@agent-compose/sdk 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +24 -1
  5. package/dist/client.d.ts +338 -534
  6. package/dist/directives.d.ts +112 -0
  7. package/dist/display.d.ts +242 -0
  8. package/dist/errors.d.ts +24 -1
  9. package/dist/index.d.ts +34 -13
  10. package/dist/index.js +2984 -861
  11. package/dist/pause/wrappers.d.ts +31 -9
  12. package/dist/processors/ask-human.d.ts +30 -0
  13. package/dist/processors/ask-human.test.d.ts +1 -0
  14. package/dist/processors/index.d.ts +1 -0
  15. package/dist/runtimes/_acp-client.d.ts +46 -1
  16. package/dist/runtimes/_cli-agent.d.ts +58 -4
  17. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  18. package/dist/runtimes/amp.d.ts +2 -2
  19. package/dist/runtimes/claude-code.d.ts +59 -0
  20. package/dist/runtimes/claude-code.test.d.ts +14 -0
  21. package/dist/runtimes/claude.d.ts +16 -0
  22. package/dist/runtimes/claude.test.d.ts +8 -0
  23. package/dist/runtimes/codex.d.ts +9 -3
  24. package/dist/runtimes/cursor.d.ts +9 -0
  25. package/dist/runtimes/droid.d.ts +9 -0
  26. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  27. package/dist/runtimes/openai-desktop.js +2922 -861
  28. package/dist/runtimes/opencode.d.ts +25 -0
  29. package/dist/runtimes/vercel.js +22 -1
  30. package/dist/sandbox/devbox.d.ts +42 -0
  31. package/dist/sandbox/exec-stream.d.ts +14 -0
  32. package/dist/sandbox/network-policy.d.ts +100 -0
  33. package/dist/sandbox/provider-def.d.ts +79 -0
  34. package/dist/sandbox/providers/desktop.d.ts +10 -0
  35. package/dist/sandbox/providers/e2b.d.ts +17 -0
  36. package/dist/sandbox/providers/local.d.ts +11 -0
  37. package/dist/sandbox/providers/vercel.d.ts +18 -0
  38. package/dist/sandbox/registry.d.ts +45 -0
  39. package/dist/sandbox/sizes.d.ts +68 -0
  40. package/dist/sandbox.d.ts +24 -299
  41. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  42. package/dist/step-invocation/invoker.d.ts +24 -1
  43. package/dist/step-invocation/protocol.d.ts +13 -0
  44. package/dist/types/api-compliance.d.ts +71 -0
  45. package/dist/types/api-conversations.d.ts +492 -0
  46. package/dist/types/api-factory.d.ts +309 -0
  47. package/dist/types/api-projects.d.ts +131 -0
  48. package/dist/types/api-runs.d.ts +377 -0
  49. package/dist/types/api-scopes.d.ts +102 -0
  50. package/dist/types/conversation-stream.d.ts +191 -0
  51. package/dist/types/execution-context.d.ts +12 -2
  52. package/dist/types/protocol.d.ts +30 -1
  53. package/dist/types/sandbox-environment.d.ts +8 -5
  54. package/dist/types/sandbox.d.ts +79 -0
  55. package/dist/types/workflow-metadata.d.ts +33 -8
  56. package/dist/types/workflow-plan.d.ts +10 -0
  57. package/dist/types/workflow.d.ts +18 -193
  58. package/dist/utils/bundler.d.ts +12 -1
  59. package/dist/utils/errors.d.ts +9 -1
  60. package/dist/workflow-steps/index.d.ts +1 -1
  61. package/dist/workflow-steps/observability.d.ts +8 -1
  62. package/dist/workflow-steps/runner.d.ts +3 -3
  63. package/dist/workflow-steps/step.d.ts +15 -1
  64. package/dist/workflow-steps/types.d.ts +19 -5
  65. package/dist/workflow-steps/workflow.d.ts +22 -1
  66. package/dist/workflows/engine.d.ts +3 -2
  67. package/dist/workflows/invoke-child.d.ts +2 -2
  68. package/package.json +1 -1
  69. package/src/agent/agent-context.ts +206 -16
  70. package/src/agent/agent-loop.ts +40 -4
  71. package/src/agent/run-agent.ts +9 -1
  72. package/src/client.ts +909 -621
  73. package/src/directives.ts +184 -0
  74. package/src/display.ts +788 -0
  75. package/src/errors.ts +39 -0
  76. package/src/index.ts +117 -10
  77. package/src/pause/wrappers.ts +44 -9
  78. package/src/processors/ask-human.ts +136 -0
  79. package/src/processors/index.ts +5 -0
  80. package/src/runtimes/_acp-client.ts +72 -3
  81. package/src/runtimes/_cli-agent.ts +171 -38
  82. package/src/runtimes/_jsonl-guard.ts +219 -0
  83. package/src/runtimes/claude-code.ts +246 -0
  84. package/src/runtimes/claude.ts +32 -2
  85. package/src/runtimes/codex.ts +55 -3
  86. package/src/runtimes/cursor.ts +59 -0
  87. package/src/runtimes/droid.ts +63 -0
  88. package/src/runtimes/openai-desktop.ts +59 -14
  89. package/src/runtimes/opencode.ts +61 -0
  90. package/src/sandbox/devbox.ts +48 -0
  91. package/src/sandbox/exec-stream.ts +48 -0
  92. package/src/sandbox/network-policy.ts +181 -0
  93. package/src/sandbox/provider-def.ts +94 -0
  94. package/src/sandbox/providers/desktop.ts +57 -0
  95. package/src/sandbox/providers/e2b.ts +354 -0
  96. package/src/sandbox/providers/local.ts +106 -0
  97. package/src/sandbox/providers/vercel.ts +331 -0
  98. package/src/sandbox/registry.ts +198 -0
  99. package/src/sandbox/sizes.ts +95 -0
  100. package/src/sandbox.ts +59 -1263
  101. package/src/step-invocation/invoker.ts +319 -34
  102. package/src/step-invocation/protocol.ts +19 -0
  103. package/src/types/api-compliance.ts +79 -0
  104. package/src/types/api-conversations.ts +522 -0
  105. package/src/types/api-factory.ts +336 -0
  106. package/src/types/api-projects.ts +140 -0
  107. package/src/types/api-runs.ts +412 -0
  108. package/src/types/api-scopes.ts +102 -0
  109. package/src/types/conversation-stream.ts +231 -0
  110. package/src/types/execution-context.ts +10 -2
  111. package/src/types/protocol.ts +33 -0
  112. package/src/types/sandbox-environment.ts +28 -9
  113. package/src/types/sandbox.ts +78 -0
  114. package/src/types/workflow-metadata.ts +35 -8
  115. package/src/types/workflow-plan.ts +11 -0
  116. package/src/types/workflow.ts +25 -280
  117. package/src/utils/bundler.ts +32 -5
  118. package/src/utils/errors.ts +34 -2
  119. package/src/workflow-steps/index.ts +1 -0
  120. package/src/workflow-steps/observability.ts +19 -8
  121. package/src/workflow-steps/runner.ts +4 -4
  122. package/src/workflow-steps/step.ts +49 -1
  123. package/src/workflow-steps/types.ts +20 -5
  124. package/src/workflow-steps/workflow.ts +22 -1
  125. package/src/workflows/engine.ts +3 -2
  126. package/src/workflows/invoke-child.ts +2 -2
@@ -7,19 +7,33 @@
7
7
  * - `output` Zod schema; validated against `run`'s return value
8
8
  * - `run` step body
9
9
  *
10
+ * Optional narrative fields (rendered on the dashboard workflow graph):
11
+ * - `summary` one plain sentence of what the step actually does
12
+ * - `deliverables` files the step promises to produce
13
+ *
10
14
  * Validation is required (not optional) because the durability story rides
11
15
  * on every step boundary being a recordable, replayable JSON value. A step
12
16
  * without a schema is invisible to the engine's persistence layer.
13
17
  *
18
+ * Narrative caps are enforced HERE, at construction — the bundler evaluates
19
+ * the module at registration, so this is the "length-capped at registration"
20
+ * point and the error names the step in the author's own environment. The
21
+ * server's manifest zod re-checks the same caps (untrusted input).
22
+ *
14
23
  * Type inference flows: `defineStep` infers TInput/TOutput from the schemas
15
24
  * so `run(ctx)` is fully typed via `ctx.input` at the call site.
16
25
  */
17
26
  import type { z } from "zod";
18
- import type { Step, StepContext } from "./types.js";
27
+ import type { Step, StepContext, StepDeliverable } from "./types.js";
19
28
  export interface DefineStepOpts<TInput, TOutput> {
20
29
  name: string;
21
30
  input: z.ZodType<TInput>;
22
31
  output: z.ZodType<TOutput>;
23
32
  run(ctx: StepContext<TInput>): TOutput | Promise<TOutput>;
33
+ /** One plain sentence of what the step actually does. 1-200 chars, no
34
+ * control characters (so no newlines). */
35
+ summary?: string;
36
+ /** Files the step promises to produce. At most 8. */
37
+ deliverables?: StepDeliverable[];
24
38
  }
25
39
  export declare function defineStep<TInput, TOutput>(opts: DefineStepOpts<TInput, TOutput>): Step<TInput, TOutput>;
@@ -14,6 +14,7 @@ import type { z } from "zod";
14
14
  import type { BaseExecutionContext } from "../types/execution-context.js";
15
15
  import type { AgentEventSink } from "../types/workflow.js";
16
16
  import type { WorkflowMetadata } from "../types/workflow-metadata.js";
17
+ import type { StepObservability } from "./observability.js";
17
18
  /**
18
19
  * Per-step execution context. Threaded into every step's `execute(...)` so
19
20
  * the step can read tenant identity and run identity, log progress, and
@@ -59,6 +60,14 @@ export interface StepContext<TInput = unknown> extends BaseExecutionContext {
59
60
  */
60
61
  agentEvents: AgentEventSink;
61
62
  }
63
+ /** One artifact a step promises to produce. `path` is workspace/drive-relative
64
+ * (e.g. "out/report.html"); the dashboard derives the format tag from the
65
+ * extension client-side — no `format` field here. */
66
+ export interface StepDeliverable {
67
+ path: string;
68
+ /** Optional one-line description of the artifact. */
69
+ description?: string;
70
+ }
62
71
  /**
63
72
  * Step definition — a single typed unit of work in a workflow chain.
64
73
  *
@@ -77,6 +86,11 @@ export interface Step<TInput, TOutput> {
77
86
  readonly output: z.ZodType<TOutput>;
78
87
  /** Step body. Receives a `StepContext<TInput>` and returns the typed output. */
79
88
  run(ctx: StepContext<TInput>): TOutput | Promise<TOutput>;
89
+ /** One plain sentence of what the step actually does — rendered on the
90
+ * dashboard workflow graph. ≤200 chars, no newlines. */
91
+ readonly summary?: string;
92
+ /** Files the step promises to produce. ≤8 entries. */
93
+ readonly deliverables?: readonly StepDeliverable[];
80
94
  }
81
95
  /**
82
96
  * The result of running one step. Engine adapters persist these into the
@@ -89,18 +103,18 @@ export type StepRunResult<TOutput = unknown> = {
89
103
  status: "completed";
90
104
  output: TOutput;
91
105
  durationMs: number;
92
- observability?: import("./observability.js").StepObservability;
106
+ observability?: StepObservability;
93
107
  } | {
94
108
  status: "failed";
95
109
  error: string;
96
110
  durationMs: number;
97
- observability?: import("./observability.js").StepObservability;
111
+ observability?: StepObservability;
98
112
  };
99
113
  /**
100
114
  * Workflow — a list of typed steps plus the workflow's input/output
101
- * schemas plus its server-side metadata bag. Returned by `defineWorkflow(...)`
102
- * (run form) and `defineWorkflow(...).step(...)...build()` (step form).
103
- * Engine adapters consume this shape.
115
+ * schemas plus its server-side metadata bag. Returned by
116
+ * `defineWorkflow(...).step(...)...build()`. Engine adapters consume
117
+ * this shape.
104
118
  *
105
119
  * `input` validates the workflow input before the first step runs.
106
120
  * `output` validates the final step's output before the workflow
@@ -20,7 +20,7 @@
20
20
  */
21
21
  import type { z } from "zod";
22
22
  import type { Step, Workflow } from "./types.js";
23
- import type { SnapshotConfig, SandboxResources } from "../types/workflow-metadata.js";
23
+ import type { SnapshotConfig, SandboxResources, ConnectorRequirements, ConnectorOperationTag, InvokePolicy, DriveMergePolicy } from "../types/workflow-metadata.js";
24
24
  import type { SandboxNetworkPolicy } from "../sandbox.js";
25
25
  import type { Processor } from "../processors/processor.js";
26
26
  export interface WorkflowBuilder<TInput, TCurrent> {
@@ -50,7 +50,28 @@ export interface StepWorkflowDefinition<TInput, TOutput> {
50
50
  * and `provider` (`vercel` | `e2b`). Vercel maps `size` to vCPUs; E2B
51
51
  * sizing is template-defined. Omit → smallest SKU on the default provider. */
52
52
  resources?: SandboxResources;
53
+ /** What happens to this workflow's drive branch when a run ends:
54
+ * `"auto"` (the default when omitted — today's behaviour) folds it into
55
+ * `main`; `"manual"` proposes a merge approval at the same terminus
56
+ * instead, leaving the branch durable until a human clicks approve. A
57
+ * cancelled run neither merges nor proposes under either policy. */
58
+ mergePolicy?: DriveMergePolicy;
53
59
  processors?: readonly Processor[];
60
+ /** Connector requirements (ADR-0007) — providers whose APIs this workflow
61
+ * calls. Dispatch resolves an authorized grant per provider and injects a
62
+ * fresh access token at the network layer. */
63
+ connectors?: ConnectorRequirements;
64
+ /** Marks this workflow as a catalogue OPERATION of a connector — e.g. the
65
+ * `create-issue` operation of the `github` connector. */
66
+ connectorOperation?: ConnectorOperationTag;
67
+ /** Tier-1 invoke ACL — who may dispatch this connector-brokering workflow.
68
+ * See `InvokePolicy`. */
69
+ invokePolicy?: InvokePolicy;
70
+ /** Internal — set by `defineSandboxEnvironment`, not by workflow authors.
71
+ * Marks the workflow as an environment build (base-env / agent-env) so the
72
+ * server skips mounting the shared factory drive for its runs (#13). See
73
+ * `WorkflowMetadata.environmentBuild`. */
74
+ environmentBuild?: boolean;
54
75
  }
55
76
  export declare function createStepWorkflow<TInput, TOutput>(opts: StepWorkflowDefinition<TInput, TOutput>): WorkflowBuilder<TInput, TInput>;
56
77
  /** Type guard — true when `value` is a `Workflow`. */
@@ -8,7 +8,8 @@
8
8
  * Errors classified into `WorkflowError` (user code threw) vs `EngineError`
9
9
  * (platform problem) for the runner harness to surface upstream.
10
10
  */
11
- import type { WorkflowHooks, WorkflowCtx } from "../types/workflow.js";
11
+ import type { WorkflowHooks } from "../types/workflow.js";
12
+ import type { InvokeChild } from "../types/execution-context.js";
12
13
  import { RequestContext } from "../request-context/request-context.js";
13
14
  import type { Workflow } from "../workflow-steps/types.js";
14
15
  import { type RunWorkflowStepsOpts } from "../workflow-steps/runner.js";
@@ -54,7 +55,7 @@ export interface RunWorkflowOptions {
54
55
  /** Provider-specific child workflow invocation. Temporal/Inngest providers
55
56
  * inject their native child-workflow primitive; the LocalProvider injects
56
57
  * the public Agent Compose API client. */
57
- invokeChild?: WorkflowCtx["invokeChild"];
58
+ invokeChild?: InvokeChild;
58
59
  }
59
60
  export declare function runWorkflow<TInput, TOutput>(wf: Workflow<TInput, TOutput>, ctx: {
60
61
  run: {
@@ -1,4 +1,4 @@
1
- import type { WorkflowCtx } from "../types/workflow.js";
1
+ import type { InvokeChild } from "../types/execution-context.js";
2
2
  /**
3
3
  * Build the public-API child workflow invoker used by legacy and sandboxed
4
4
  * workflow execution. Provider-backed engines may inject a different
@@ -7,4 +7,4 @@ import type { WorkflowCtx } from "../types/workflow.js";
7
7
  export declare function buildInvokeChild(runId: string, opts?: {
8
8
  fallbackBaseUrl?: string;
9
9
  defaultFactorySlug?: string;
10
- }): WorkflowCtx["invokeChild"];
10
+ }): InvokeChild;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.6.0",
3
+ "version": "0.8.0",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -71,29 +71,36 @@ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
71
71
 
72
72
  ## Writing workflow / agent code — the SDK
73
73
 
74
- \`@agent-compose/sdk\` is installed in \`/workspace\` — import it from any script
75
- you write there:
74
+ \`@agent-compose/sdk\` is installed in \`/workspace\`. **To author a workflow,
75
+ ALWAYS run \`/ac:generate-workflow\`** (and \`/ac:generate-agent\` for an agent
76
+ step) instead of writing source from memory — the skill scaffolds the correct,
77
+ current shape. Then \`agentc register <file.ts>\` (or \`/ac:register\`).
76
78
 
77
- import { defineWorkflow, agent, AgentComposeClient } from "@agent-compose/sdk";
79
+ The skill writes **step-form** (a builder of discrete, durable \`.step()\`s).
80
+ The legacy run-form (\`defineWorkflow({ run(ctx, sandbox) { … } })\`) has been
81
+ REMOVED from the SDK — registering one fails with an error. Step-form is the
82
+ only shape: durable per-step replay, and pause only works there.
78
83
 
79
- Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
80
- \`agentc register <file.ts>\` (or \`/ac:register\`).
84
+ ## Pausing to ask the human
81
85
 
82
- ## Pausing to ask the human — \`agentc pause\`
83
-
84
- When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
85
- **END YOUR TURN**:
86
+ To ask a human and get an answer back, use the **\`AskUserQuestion\`** tool if
87
+ you have it; otherwise run **\`agentc pause\`**:
86
88
 
87
89
  agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
90
  --option retry --option skip
89
91
 
90
- \`agentc pause\` does NOT block and does NOT print the answer. It records your
91
- question and returns immediately. The moment you end your turn, the run pauses
92
- (your sandbox is snapshotted and compute stops while the human decides) and the
93
- human's answer is delivered to you as your **next message** — you pick up
94
- exactly where you left off, with the answer in hand. So: ask, end your turn,
95
- and wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task
96
- complete after pausing.
92
+ **Both BLOCK and hand you the answer inline.** While you wait, the run is
93
+ suspended — your sandbox is frozen and compute stops, so a pause is free while
94
+ the human decides. When they answer, the call RETURNS with their decision: the
95
+ \`AskUserQuestion\` tool result, or \`agentc pause\`'s output
96
+ (\`▶ Resumed. The human answered: …\`), carries it.
97
+
98
+ **Then USE that answer to finish your work — do NOT end your turn.** This is NOT
99
+ fire-and-forget, and the answer does NOT arrive in a later message: it comes
100
+ back right where you called it, on the SAME turn. The shape is: ask → the call
101
+ blocks → it returns the human's answer → you act on it and produce your result.
102
+ Never end your turn before the call returns, never guess an answer, and never
103
+ proceed without one.
97
104
 
98
105
  Reach for it the moment you hit — or foresee — any of these:
99
106
  - **A wall only a human can clear:** a 401/403, a missing credential, an
@@ -118,14 +125,197 @@ request **without** an Authorization header and the platform adds it. Don't try
118
125
  to read or exfiltrate tokens; they aren't here. The "Connectors & access"
119
126
  section below (when present) lists exactly which providers this run can reach.
120
127
 
128
+ ## Computer Use — you have a real desktop, and it is already running
129
+
130
+ **This machine has a graphical desktop.** Every session machine does — terminal
131
+ sessions included — and the platform brings it UP AT BOOT, before your first
132
+ turn: an X server on \`DISPLAY=:0\`, the openbox window manager, wallpaper and a
133
+ panel. You do not start it, you do not wait for a human to open it, and you do
134
+ not need a viewer. Go straight to driving it.
135
+
136
+ (The one exception, and it is rare: an image built without the GUI stack has no
137
+ display at all, and \`DISPLAY=:0 xdotool getdisplaygeometry\` errors outright.
138
+ That single case is the only one where this section does not apply — a
139
+ screenshot showing only wallpaper is NOT it, and neither is an app that failed
140
+ to start.)
141
+
142
+ **This is how you SEE anything.** Any question of the form "does it render?",
143
+ "is the page actually working?", "did the markers show up?", "what does it look
144
+ like?" is answered by opening it on this desktop and screenshotting it — not by
145
+ reasoning about the code, and not by a headless render (which proves the process
146
+ starts, not that the thing draws). Verify visually before you report visually.
147
+
148
+ - **Input** — \`xdotool\` against \`DISPLAY=:0\`: \`DISPLAY=:0 xdotool mousemove <x> <y>\`,
149
+ \`DISPLAY=:0 xdotool click 1\` (1=left, 3=right), \`DISPLAY=:0 xdotool type 'text'\`,
150
+ \`DISPLAY=:0 xdotool key Return\` (also \`ctrl+c\`, \`Tab\`, \`super\`, …).
151
+ - **Screenshots** — \`scrot\` (or ImageMagick's \`import\`):
152
+ \`DISPLAY=:0 scrot /tmp/screen.png\`, then READ the PNG to see the screen,
153
+ before and after you act. A screenshot is your only eyes here.
154
+ - **Apps + windows** — a plain X session. Launch in the background:
155
+ \`DISPLAY=:0 <app> &\`. Two things that trip agents up, both normal:
156
+ - a GUI app needs a **beat to map its window** — screenshot, and if you see
157
+ only wallpaper, wait a couple of seconds and screenshot again before
158
+ concluding anything;
159
+ - **Chromium needs \`--no-sandbox\`** in this environment (nested sandbox).
160
+ The whole recipe for looking at a local page:
161
+ \`DISPLAY=:0 chromium --no-sandbox --disable-gpu --start-maximized <url> &\`
162
+ then \`sleep 5\`, then \`DISPLAY=:0 scrot /tmp/screen.png\` and read it.
163
+ If a window still never appears, read the app's own log (\`/tmp/*.log\`) — the
164
+ desktop is not the thing that failed. Do NOT abandon it for a headless
165
+ screenshot: headless cannot tell you what the human will see.
166
+ - **A human can watch** — the session header carries a **Desktop** button in the
167
+ dashboard, and what a teammate sees there is exactly this display. The desktop
168
+ runs whether or not anyone is looking; never wait for a viewer.
169
+
170
+ Nothing here changes the credentials rule above: tokens are injected at the
171
+ network layer, never present on the desktop or in any file you can read — so
172
+ there is nothing to type, paste, or screenshot a credential from.
173
+
174
+ ## Recording a demo — the desktop, captured to a video the human can play
175
+
176
+ "Record a demo of you using X" is a normal ask, and this machine does it:
177
+ start a screen recording, drive the app with \`xdotool\` exactly as in Computer
178
+ Use, stop the recording, and report the file. (For a LIVE view no recording is
179
+ needed — the session header's **Desktop** button already streams this display
180
+ to any teammate watching; a recording is the durable, replayable artifact.
181
+ Both modes exist; say so when it matters.)
182
+
183
+ **ffmpeg is NOT pre-installed** — install it first, once per machine:
184
+
185
+ sudo apt-get update -q && sudo apt-get install -y -q ffmpeg
186
+
187
+ (drop \`sudo\` if you are already root). Then the whole recipe:
188
+
189
+ DISPLAY=:0 ffmpeg -f x11grab \\
190
+ -video_size "$(DISPLAY=:0 xdotool getdisplaygeometry | tr ' ' x)" \\
191
+ -framerate 10 -i :0 -c:v libvpx -b:v 1M -deadline realtime -cpu-used 8 \\
192
+ demo.webm &
193
+ FFMPEG_PID=$!
194
+ # ... drive the app with xdotool, screenshotting as you go ...
195
+ kill -INT "$FFMPEG_PID" && wait "$FFMPEG_PID"
196
+
197
+ The gotchas, each one earned:
198
+ - **Stop with SIGINT (\`kill -INT\`), never SIGKILL** — ffmpeg finalizes the
199
+ file on SIGINT; a hard kill truncates the encode mid-write.
200
+ - **Record WebM (matroska-family), not MP4** — mp4 writes its moov atom at the
201
+ END, so a killed or crashed encode leaves an UNPLAYABLE file; webm stays
202
+ playable up to the last written frame and plays natively in the browser.
203
+ MP4's only edge is compatibility with some external players — transcode
204
+ afterwards if you truly need it, never record straight to it.
205
+ - **\`-video_size\` must match the real screen** — x11grab does not default to
206
+ it; read the geometry from \`xdotool getdisplaygeometry\` as above.
207
+ - **10–12 fps is right for a screen demo** — small files, legible UI motion;
208
+ this is not video production.
209
+ - **Write to the drive, not /tmp** — the recording must land in your working
210
+ directory to persist and show up in Files; a file in /tmp dies with the
211
+ sandbox.
212
+ - When you stop, **TELL the human the exact drive path** of the video — a
213
+ recording they cannot find might as well not exist.
214
+
215
+ ## Previews — register every server you serve (cloud sessions)
216
+
217
+ In a cloud session, a dev server listening on a port becomes a hosted,
218
+ member-gated URL the human can open — but ONLY if you register it:
219
+
220
+ agentc preview open <port> [--name <label>] [--path </landing>]
221
+ # hosted URL + an "Open preview" card
222
+ agentc preview list # the registry — what is live right now
223
+ agentc preview close <port> # take one down
224
+
225
+ (\`agentc preview announce\` is the same verb as \`open\` — announce what you
226
+ serve.) \`--name\` is the human-readable label; \`--path\` is where the app
227
+ should open (e.g. \`/dashboard\`) — the card and every chip land the human
228
+ there instead of a bare \`/\`.
229
+
230
+ Register EVERY server you start for a human, the moment it is listening, and
231
+ tell them the URL the command printed. The registry is the only discoverable
232
+ record of what this machine serves: an unregistered server keeps running, but
233
+ nobody — not the human, not the assistant — can find its URL, and when the
234
+ sandbox recycles it is gone without a trace. Never guess or hand out a raw
235
+ port; the hosted URL from \`agentc preview open\` is the only address that
236
+ works outside this machine. (Outside a cloud session the command errors
237
+ honestly — there is no session sandbox to expose.)
238
+
239
+ What registration buys you: the human sees each registered preview as a card
240
+ in the conversation and a row in the session's Previews menu — MANY at once,
241
+ one per port — and the assistant resolves "open the preview" from this same
242
+ registry (its \`list_previews\` read), so what you register is exactly what
243
+ gets opened. On deployments with subdomain previews the hosted URL is a real
244
+ origin of its own — absolute asset paths and client-side routing work, the
245
+ whole app is navigable — so serve normally and let the platform address it;
246
+ never rewrite your app to a path prefix.
247
+
121
248
  ## Tools in this environment
122
249
 
123
250
  - \`agentc\` — Agent Compose CLI (your primary interface; authed from env)
124
251
  - \`@agent-compose/sdk\` — installed in /workspace for writing workflows
125
252
  - \`/ac:*\` Claude Code skills — slash commands for the above
126
253
  - \`archil\` (factory drive), \`rtk\`, \`bun\`
254
+ - \`xdotool\` / \`scrot\` — drive + screenshot the desktop (if this machine has one; see Computer Use)
127
255
  - A world-writable \`/workspace\` working directory`;
128
256
 
257
+ /** Parameters for the `agentc session add` education brief (ADR-0055 §8). */
258
+ export interface AddedSessionBriefParams {
259
+ conversationId: string;
260
+ serverUrl: string;
261
+ dashboardUrl: string | null;
262
+ }
263
+
264
+ /**
265
+ * The education brief `agentc session add` writes into a connected LOCAL
266
+ * session's CLAUDE.md/AGENTS.md — the same content family as
267
+ * `AGENT_COMPOSE_MANUAL`, rendered for the local context (one source,
268
+ * rendered per context). The manual above must stay BYTE-IDENTICAL (the
269
+ * server and base-env bake static copies of it); this function renders a
270
+ * sibling document and never touches it. The local context differs from the
271
+ * sandbox in exactly the ways stated here: auth rides the bridge credential
272
+ * fallback instead of injected run env; no factory drive is mounted — the
273
+ * files on this machine belong to the human; and the bound conversation is
274
+ * MIRROR-ONLY (ADR-0055 §8.6) — teammates read along but can never message
275
+ * the session through it, so the brief must not promise an inbound channel.
276
+ */
277
+ export function buildAddedSessionBrief(p: AddedSessionBriefParams): string {
278
+ const dashboardSection = p.dashboardUrl === null ? "" : `
279
+
280
+ ## Dashboard
281
+
282
+ Teammates follow this conversation (and the rest of the factory) in the
283
+ Agent Compose dashboard: ${p.dashboardUrl}`;
284
+
285
+ return `# Connected to Agent Compose
286
+
287
+ Agent Compose is your team's agent platform: durable conversations,
288
+ workflow runs, and shared factory drives where humans and agents work
289
+ together. THIS terminal's claude-code session is connected to Agent
290
+ Compose conversation \`${p.conversationId}\` on ${p.serverUrl}.
291
+
292
+ That conversation is a LIVE, READ-ONLY MIRROR of this terminal session:
293
+ teammates read along in the dashboard as the work happens, but they cannot
294
+ message you through it — anything posted there is answered by the server
295
+ with a notice and never reaches this terminal. Everything you do here is
296
+ mirrored automatically; you have an audience, not a channel.
297
+
298
+ ## The \`agentc\` toolbelt
299
+
300
+ The \`agentc\` CLI works from this shell. It is already authenticated on
301
+ this machine via the bridge credential fallback — no keys to manage,
302
+ commands just work:
303
+
304
+ agentc list # registered workflows
305
+ agentc logs <run-id> # a run's logs
306
+ agentc invoke <workflow> -i '<json>' # dispatch a workflow
307
+ agentc events list # read the factory timeline
308
+
309
+ ## Scope — this is your LOCAL machine
310
+
311
+ The files here are YOURS: no factory drive is mounted in this session,
312
+ and nothing you write locally lands on a shared drive by itself. Cloud
313
+ drive/branch semantics (per-run directories on the factory drive, drive
314
+ branches, persist-by-default outputs) apply only to cloud sessions —
315
+ not here.${dashboardSection}
316
+ `;
317
+ }
318
+
129
319
  /**
130
320
  * One connector this run can reach, as the agent should see it. Strictly
131
321
  * NON-SECRET — hosts, methods, paths, identity only. The access token is
@@ -39,6 +39,24 @@ export function parseAgentStatus(text: string): AgentStatus | null {
39
39
 
40
40
  const DEFAULT_ALLOWED_TOOLS = ["Read", "Write", "Edit", "Bash", "Glob", "Grep", "WebFetch"];
41
41
 
42
+ /**
43
+ * A `responseSchema` rendered for the runtime's structured-output surface.
44
+ *
45
+ * zod v4's `toJSONSchema` stamps `$schema: "…/draft/2020-12/schema"` on the
46
+ * result. Claude Code validates `--json-schema` with a validator that has no
47
+ * 2020-12 meta-schema registered, so any schema CARRYING that header is
48
+ * rejected at CLI startup — the process exits 1 before its first API call
49
+ * and every agent with a responseSchema dies on spawn (observed live on
50
+ * claude 2.1.212: `--json-schema is not a valid JSON Schema: no schema with
51
+ * key or ref "https://json-schema.org/draft/2020-12/schema"`). The header is
52
+ * pure metadata — drop it; the schema body is draft-07-compatible for every
53
+ * shape zod emits from our workflow schemas.
54
+ */
55
+ export function runtimeJsonSchema(schema: z.ZodType<unknown>): Record<string, unknown> {
56
+ const { $schema: _$schema, ...rest } = z.toJSONSchema(schema) as Record<string, unknown>;
57
+ return rest;
58
+ }
59
+
42
60
  export interface AgentLoopResult<TResponse = unknown> {
43
61
  agentId: string;
44
62
  label: string;
@@ -71,7 +89,12 @@ function preview(value: unknown): string {
71
89
  }
72
90
  }
73
91
 
74
- export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
92
+ /** Everything but the live-only streaming chunk: `text_delta` never becomes
93
+ * an agent.message event (the terminating `text` carries the whole block) —
94
+ * the loop filters it before summarizing. */
95
+ type DurableAgentMessage = Exclude<AgentMessage, { type: "text_delta" } | { type: "usage_delta" }>;
96
+
97
+ export function summarizeAgentMessage(msg: DurableAgentMessage): AgentMessageSummary {
75
98
  switch (msg.type) {
76
99
  case "init": return { type: "init", sessionId: msg.sessionId };
77
100
  case "text": return { type: "text", text: msg.text };
@@ -164,9 +187,16 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
164
187
  const logLabel = opts.label ?? "[Agent Loop]";
165
188
  const startedAt = Date.now();
166
189
  // No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
167
- // done. A numeric budget is an explicit caller choice, not a default we impose.
190
+ // done in one unbounded session. A numeric budget is an explicit caller choice.
191
+ // Default to a SMALL allowance (not 1) so the loop can RE-PROMPT for the closing
192
+ // <status>/<response> when a session ends EARLY — e.g. a human pause (agentc
193
+ // pause / AskUserQuestion) freezes the VM mid-tool and the resumed session can
194
+ // drop its final turn after doing the work. The re-prompt (PROTOCOL_SUFFIX)
195
+ // only fires when an iteration produced no exit_signal, so a clean run still
196
+ // settles in ONE iteration — these extra iterations are a recovery path, not
197
+ // the norm.
168
198
  const turnsPerIteration = opts.turnsPerIteration;
169
- const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 1 : 8);
199
+ const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 3 : 8);
170
200
  // A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
171
201
  // validation, the loop re-prompts with the exact errors until it conforms —
172
202
  // without consuming the caller's iteration budget. The backstop below only
@@ -201,7 +231,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
201
231
  // `session/request_permission` path through CliAgentRunner.gateToolCall) can
202
232
  // raise a human-approval `ctx.pause`, not just on the loop's own hooks.
203
233
  ...(opts.pause ? { pause: opts.pause } : {}),
204
- ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: z.toJSONSchema(opts.responseSchema) as Record<string, unknown> } } : {}),
234
+ ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: runtimeJsonSchema(opts.responseSchema) } } : {}),
205
235
  });
206
236
 
207
237
  // Heads-up when a caller registers tool-call gating on a runtime that
@@ -421,6 +451,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
421
451
  signal: loopAbort.signal,
422
452
  ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
423
453
  })) {
454
+ // Live-only streaming chunk: the terminating `text` message carries
455
+ // the complete block, so the loop (accumulation, events, processors)
456
+ // ignores deltas — they exist for progressive-rendering consumers
457
+ // (the conversation cloud executor), not the workflow event stream.
458
+ if (rawMsg.type === "text_delta" || rawMsg.type === "usage_delta") continue;
424
459
  // processOutput chain — deny drops the message from accumulation;
425
460
  // abort ends the loop. Continue carries the (possibly mutated)
426
461
  // message forward.
@@ -434,6 +469,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
434
469
  continue;
435
470
  }
436
471
  const msg = outputVerdict.value;
472
+ if (msg.type === "text_delta" || msg.type === "usage_delta") continue; // a processor cannot re-introduce a live-only chunk
437
473
  opts.onAgentEvent?.(iteration, msg);
438
474
  // Usage summaries carry the resolved model so the server can price
439
475
  // token rows per model without correlating back to agent.spawned.
@@ -14,6 +14,7 @@ import { z } from "zod";
14
14
  import { randomUUID } from "node:crypto";
15
15
  import { getActiveStep, nextAgentCallInActiveStep } from "../active-step.js";
16
16
  import { corePause, type PauseRequest } from "../pause/pause-core.js";
17
+ import { createAskHumanProcessor } from "../processors/ask-human.js";
17
18
  import { agentLoop } from "./agent-loop.js";
18
19
  import { consumeSteerPending, runControlPoller } from "./steer-control.js";
19
20
  import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
@@ -330,6 +331,13 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
330
331
  // (the boundary self-pause trigger + the prompt instruction above). Human
331
332
  // steering works regardless — see the ungated control poller.
332
333
  const mode = opts.mode ?? "auto";
334
+ // Ask-a-human is TOOLS-driven, not a separate mode: the ask-human processor is
335
+ // a chain default that maps the `AskUserQuestion` tool to a server pause (the
336
+ // run freezes, the human answers, the answer returns as the tool result). It's
337
+ // a no-op for any agent that doesn't have / call `AskUserQuestion`, so granting
338
+ // that tool to an agent IS its "can ask a human" switch — withhold it and the
339
+ // agent simply can't (it reports blockers up instead).
340
+ const processors = [createAskHumanProcessor(), ...(opts.processors ?? [])];
333
341
 
334
342
  try {
335
343
  return await agentLoop({
@@ -361,7 +369,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
361
369
  ...(opts.events ? { onAgentLifecycleEvent: (event: AgentLifecycleEvent) => { void opts.events?.emit(event); } } : {}),
362
370
  ...(opts.onAgentEvent ? { onAgentEvent: opts.onAgentEvent } : {}),
363
371
  ...(opts.onIteration ? { onIteration: opts.onIteration } : {}),
364
- ...(opts.processors?.length ? { processors: opts.processors } : {}),
372
+ ...(processors.length ? { processors } : {}),
365
373
  requestContext: opts.requestContext ?? RequestContext.fromReserved({
366
374
  teamId: "", runId: "", workflowId: "",
367
375
  factoryId: null, apiKeyScopes: [], parentRunId: null,