@agent-compose/sdk 0.7.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +32 -1
  5. package/dist/agent/run-agent.d.ts +4 -0
  6. package/dist/client.d.ts +382 -534
  7. package/dist/directives.d.ts +112 -0
  8. package/dist/display.d.ts +258 -0
  9. package/dist/errors.d.ts +24 -1
  10. package/dist/index.d.ts +26 -14
  11. package/dist/index.js +3774 -1679
  12. package/dist/pause/wrappers.d.ts +31 -9
  13. package/dist/runtimes/_acp-client.d.ts +46 -1
  14. package/dist/runtimes/_cli-agent.d.ts +51 -4
  15. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  16. package/dist/runtimes/amp.d.ts +2 -2
  17. package/dist/runtimes/claude-code.d.ts +61 -0
  18. package/dist/runtimes/claude-code.test.d.ts +14 -0
  19. package/dist/runtimes/claude.d.ts +16 -0
  20. package/dist/runtimes/claude.test.d.ts +8 -0
  21. package/dist/runtimes/codex.d.ts +12 -3
  22. package/dist/runtimes/cursor.d.ts +2 -2
  23. package/dist/runtimes/droid.d.ts +2 -2
  24. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  25. package/dist/runtimes/openai-desktop.js +3718 -1680
  26. package/dist/runtimes/opencode.d.ts +2 -2
  27. package/dist/runtimes/vercel.js +12 -1
  28. package/dist/sandbox/devbox.d.ts +42 -0
  29. package/dist/sandbox/exec-stream.d.ts +14 -0
  30. package/dist/sandbox/network-policy.d.ts +100 -0
  31. package/dist/sandbox/provider-def.d.ts +79 -0
  32. package/dist/sandbox/providers/desktop.d.ts +10 -0
  33. package/dist/sandbox/providers/e2b.d.ts +17 -0
  34. package/dist/sandbox/providers/local.d.ts +11 -0
  35. package/dist/sandbox/providers/vercel.d.ts +18 -0
  36. package/dist/sandbox/registry.d.ts +45 -0
  37. package/dist/sandbox/sizes.d.ts +68 -0
  38. package/dist/sandbox.d.ts +24 -299
  39. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  40. package/dist/step-invocation/invoker.d.ts +10 -0
  41. package/dist/step-invocation/protocol.d.ts +5 -0
  42. package/dist/types/api-compliance.d.ts +71 -0
  43. package/dist/types/api-conversations.d.ts +523 -0
  44. package/dist/types/api-factory.d.ts +334 -0
  45. package/dist/types/api-projects.d.ts +131 -0
  46. package/dist/types/api-runs.d.ts +422 -0
  47. package/dist/types/api-scopes.d.ts +102 -0
  48. package/dist/types/conversation-stream.d.ts +191 -0
  49. package/dist/types/execution-context.d.ts +12 -2
  50. package/dist/types/protocol.d.ts +38 -1
  51. package/dist/types/sandbox-environment.d.ts +8 -5
  52. package/dist/types/sandbox.d.ts +74 -4
  53. package/dist/types/workflow-metadata.d.ts +41 -8
  54. package/dist/types/workflow-plan.d.ts +10 -0
  55. package/dist/types/workflow.d.ts +18 -205
  56. package/dist/utils/bundler.d.ts +68 -1
  57. package/dist/workflow-steps/index.d.ts +1 -1
  58. package/dist/workflow-steps/observability.d.ts +8 -1
  59. package/dist/workflow-steps/runner.d.ts +3 -3
  60. package/dist/workflow-steps/step.d.ts +15 -1
  61. package/dist/workflow-steps/types.d.ts +19 -5
  62. package/dist/workflow-steps/workflow.d.ts +29 -1
  63. package/dist/workflows/engine.d.ts +3 -2
  64. package/dist/workflows/invoke-child.d.ts +20 -2
  65. package/dist/workflows/invoke-child.test.d.ts +9 -0
  66. package/package.json +2 -2
  67. package/src/agent/agent-context.ts +186 -3
  68. package/src/agent/agent-loop.ts +40 -2
  69. package/src/agent/run-agent.ts +5 -0
  70. package/src/client.ts +1048 -625
  71. package/src/directives.ts +184 -0
  72. package/src/display.ts +834 -0
  73. package/src/errors.ts +39 -0
  74. package/src/index.ts +114 -12
  75. package/src/pause/wrappers.ts +44 -9
  76. package/src/runtimes/_acp-client.ts +72 -3
  77. package/src/runtimes/_cli-agent.ts +161 -36
  78. package/src/runtimes/_jsonl-guard.ts +219 -0
  79. package/src/runtimes/claude-code.ts +256 -0
  80. package/src/runtimes/claude.ts +32 -2
  81. package/src/runtimes/codex.ts +63 -3
  82. package/src/runtimes/openai-desktop.ts +59 -14
  83. package/src/sandbox/devbox.ts +48 -0
  84. package/src/sandbox/exec-stream.ts +48 -0
  85. package/src/sandbox/network-policy.ts +181 -0
  86. package/src/sandbox/provider-def.ts +94 -0
  87. package/src/sandbox/providers/desktop.ts +57 -0
  88. package/src/sandbox/providers/e2b.ts +354 -0
  89. package/src/sandbox/providers/local.ts +106 -0
  90. package/src/sandbox/providers/vercel.ts +331 -0
  91. package/src/sandbox/registry.ts +198 -0
  92. package/src/sandbox/sizes.ts +95 -0
  93. package/src/sandbox.ts +59 -1275
  94. package/src/step-invocation/invoker.ts +151 -28
  95. package/src/step-invocation/protocol.ts +8 -0
  96. package/src/types/api-compliance.ts +79 -0
  97. package/src/types/api-conversations.ts +547 -0
  98. package/src/types/api-factory.ts +368 -0
  99. package/src/types/api-projects.ts +140 -0
  100. package/src/types/api-runs.ts +459 -0
  101. package/src/types/api-scopes.ts +102 -0
  102. package/src/types/conversation-stream.ts +231 -0
  103. package/src/types/execution-context.ts +10 -2
  104. package/src/types/protocol.ts +41 -0
  105. package/src/types/sandbox-environment.ts +28 -9
  106. package/src/types/sandbox.ts +73 -4
  107. package/src/types/workflow-metadata.ts +44 -8
  108. package/src/types/workflow-plan.ts +11 -0
  109. package/src/types/workflow.ts +25 -292
  110. package/src/utils/bundler.ts +245 -8
  111. package/src/utils/errors.ts +16 -1
  112. package/src/workflow-steps/index.ts +1 -0
  113. package/src/workflow-steps/observability.ts +19 -8
  114. package/src/workflow-steps/runner.ts +4 -4
  115. package/src/workflow-steps/step.ts +49 -1
  116. package/src/workflow-steps/types.ts +20 -5
  117. package/src/workflow-steps/workflow.ts +29 -1
  118. package/src/workflows/engine.ts +3 -2
  119. package/src/workflows/invoke-child.ts +49 -13
@@ -77,9 +77,9 @@ step) instead of writing source from memory — the skill scaffolds the correct,
77
77
  current shape. Then \`agentc register <file.ts>\` (or \`/ac:register\`).
78
78
 
79
79
  The skill writes **step-form** (a builder of discrete, durable \`.step()\`s).
80
- Never write the legacy run-form (\`defineWorkflow({ run(ctx, sandbox) { … } })\`):
81
- it is one opaque step, so any failure or resume re-runs the whole body — and
82
- **pause does not work in run-form**.
80
+ The legacy run-form (\`defineWorkflow({ run(ctx, sandbox) { … } })\`) has been
81
+ REMOVED from the SDK — registering one fails with an error. Step-form is the
82
+ only shape: durable per-step replay, and pause only works there.
83
83
 
84
84
  ## Pausing to ask the human
85
85
 
@@ -125,14 +125,197 @@ request **without** an Authorization header and the platform adds it. Don't try
125
125
  to read or exfiltrate tokens; they aren't here. The "Connectors & access"
126
126
  section below (when present) lists exactly which providers this run can reach.
127
127
 
128
+ ## Computer Use — you have a real desktop, and it is already running
129
+
130
+ **This machine has a graphical desktop.** Every session machine does — terminal
131
+ sessions included — and the platform brings it UP AT BOOT, before your first
132
+ turn: an X server on \`DISPLAY=:0\`, the openbox window manager, wallpaper and a
133
+ panel. You do not start it, you do not wait for a human to open it, and you do
134
+ not need a viewer. Go straight to driving it.
135
+
136
+ (The one exception, and it is rare: an image built without the GUI stack has no
137
+ display at all, and \`DISPLAY=:0 xdotool getdisplaygeometry\` errors outright.
138
+ That single case is the only one where this section does not apply — a
139
+ screenshot showing only wallpaper is NOT it, and neither is an app that failed
140
+ to start.)
141
+
142
+ **This is how you SEE anything.** Any question of the form "does it render?",
143
+ "is the page actually working?", "did the markers show up?", "what does it look
144
+ like?" is answered by opening it on this desktop and screenshotting it — not by
145
+ reasoning about the code, and not by a headless render (which proves the process
146
+ starts, not that the thing draws). Verify visually before you report visually.
147
+
148
+ - **Input** — \`xdotool\` against \`DISPLAY=:0\`: \`DISPLAY=:0 xdotool mousemove <x> <y>\`,
149
+ \`DISPLAY=:0 xdotool click 1\` (1=left, 3=right), \`DISPLAY=:0 xdotool type 'text'\`,
150
+ \`DISPLAY=:0 xdotool key Return\` (also \`ctrl+c\`, \`Tab\`, \`super\`, …).
151
+ - **Screenshots** — \`scrot\` (or ImageMagick's \`import\`):
152
+ \`DISPLAY=:0 scrot /tmp/screen.png\`, then READ the PNG to see the screen,
153
+ before and after you act. A screenshot is your only eyes here.
154
+ - **Apps + windows** — a plain X session. Launch in the background:
155
+ \`DISPLAY=:0 <app> &\`. Two things that trip agents up, both normal:
156
+ - a GUI app needs a **beat to map its window** — screenshot, and if you see
157
+ only wallpaper, wait a couple of seconds and screenshot again before
158
+ concluding anything;
159
+ - **Chromium needs \`--no-sandbox\`** in this environment (nested sandbox).
160
+ The whole recipe for looking at a local page:
161
+ \`DISPLAY=:0 chromium --no-sandbox --disable-gpu --start-maximized <url> &\`
162
+ then \`sleep 5\`, then \`DISPLAY=:0 scrot /tmp/screen.png\` and read it.
163
+ If a window still never appears, read the app's own log (\`/tmp/*.log\`) — the
164
+ desktop is not the thing that failed. Do NOT abandon it for a headless
165
+ screenshot: headless cannot tell you what the human will see.
166
+ - **A human can watch** — the session header carries a **Desktop** button in the
167
+ dashboard, and what a teammate sees there is exactly this display. The desktop
168
+ runs whether or not anyone is looking; never wait for a viewer.
169
+
170
+ Nothing here changes the credentials rule above: tokens are injected at the
171
+ network layer, never present on the desktop or in any file you can read — so
172
+ there is nothing to type, paste, or screenshot a credential from.
173
+
174
+ ## Recording a demo — the desktop, captured to a video the human can play
175
+
176
+ "Record a demo of you using X" is a normal ask, and this machine does it:
177
+ start a screen recording, drive the app with \`xdotool\` exactly as in Computer
178
+ Use, stop the recording, and report the file. (For a LIVE view no recording is
179
+ needed — the session header's **Desktop** button already streams this display
180
+ to any teammate watching; a recording is the durable, replayable artifact.
181
+ Both modes exist; say so when it matters.)
182
+
183
+ **ffmpeg is NOT pre-installed** — install it first, once per machine:
184
+
185
+ sudo apt-get update -q && sudo apt-get install -y -q ffmpeg
186
+
187
+ (drop \`sudo\` if you are already root). Then the whole recipe:
188
+
189
+ DISPLAY=:0 ffmpeg -f x11grab \\
190
+ -video_size "$(DISPLAY=:0 xdotool getdisplaygeometry | tr ' ' x)" \\
191
+ -framerate 10 -i :0 -c:v libvpx -b:v 1M -deadline realtime -cpu-used 8 \\
192
+ demo.webm &
193
+ FFMPEG_PID=$!
194
+ # ... drive the app with xdotool, screenshotting as you go ...
195
+ kill -INT "$FFMPEG_PID" && wait "$FFMPEG_PID"
196
+
197
+ The gotchas, each one earned:
198
+ - **Stop with SIGINT (\`kill -INT\`), never SIGKILL** — ffmpeg finalizes the
199
+ file on SIGINT; a hard kill truncates the encode mid-write.
200
+ - **Record WebM (matroska-family), not MP4** — mp4 writes its moov atom at the
201
+ END, so a killed or crashed encode leaves an UNPLAYABLE file; webm stays
202
+ playable up to the last written frame and plays natively in the browser.
203
+ MP4's only edge is compatibility with some external players — transcode
204
+ afterwards if you truly need it, never record straight to it.
205
+ - **\`-video_size\` must match the real screen** — x11grab does not default to
206
+ it; read the geometry from \`xdotool getdisplaygeometry\` as above.
207
+ - **10–12 fps is right for a screen demo** — small files, legible UI motion;
208
+ this is not video production.
209
+ - **Write to the drive, not /tmp** — the recording must land in your working
210
+ directory to persist and show up in Files; a file in /tmp dies with the
211
+ sandbox.
212
+ - When you stop, **TELL the human the exact drive path** of the video — a
213
+ recording they cannot find might as well not exist.
214
+
215
+ ## Previews — register every server you serve (cloud sessions)
216
+
217
+ In a cloud session, a dev server listening on a port becomes a hosted,
218
+ member-gated URL the human can open — but ONLY if you register it:
219
+
220
+ agentc preview open <port> [--name <label>] [--path </landing>]
221
+ # hosted URL + an "Open preview" card
222
+ agentc preview list # the registry — what is live right now
223
+ agentc preview close <port> # take one down
224
+
225
+ (\`agentc preview announce\` is the same verb as \`open\` — announce what you
226
+ serve.) \`--name\` is the human-readable label; \`--path\` is where the app
227
+ should open (e.g. \`/dashboard\`) — the card and every chip land the human
228
+ there instead of a bare \`/\`.
229
+
230
+ Register EVERY server you start for a human, the moment it is listening, and
231
+ tell them the URL the command printed. The registry is the only discoverable
232
+ record of what this machine serves: an unregistered server keeps running, but
233
+ nobody — not the human, not the assistant — can find its URL, and when the
234
+ sandbox recycles it is gone without a trace. Never guess or hand out a raw
235
+ port; the hosted URL from \`agentc preview open\` is the only address that
236
+ works outside this machine. (Outside a cloud session the command errors
237
+ honestly — there is no session sandbox to expose.)
238
+
239
+ What registration buys you: the human sees each registered preview as a card
240
+ in the conversation and a row in the session's Previews menu — MANY at once,
241
+ one per port — and the assistant resolves "open the preview" from this same
242
+ registry (its \`list_previews\` read), so what you register is exactly what
243
+ gets opened. On deployments with subdomain previews the hosted URL is a real
244
+ origin of its own — absolute asset paths and client-side routing work, the
245
+ whole app is navigable — so serve normally and let the platform address it;
246
+ never rewrite your app to a path prefix.
247
+
128
248
  ## Tools in this environment
129
249
 
130
250
  - \`agentc\` — Agent Compose CLI (your primary interface; authed from env)
131
251
  - \`@agent-compose/sdk\` — installed in /workspace for writing workflows
132
252
  - \`/ac:*\` Claude Code skills — slash commands for the above
133
253
  - \`archil\` (factory drive), \`rtk\`, \`bun\`
254
+ - \`xdotool\` / \`scrot\` — drive + screenshot the desktop (if this machine has one; see Computer Use)
134
255
  - A world-writable \`/workspace\` working directory`;
135
256
 
257
+ /** Parameters for the `agentc session add` education brief (ADR-0055 §8). */
258
+ export interface AddedSessionBriefParams {
259
+ conversationId: string;
260
+ serverUrl: string;
261
+ dashboardUrl: string | null;
262
+ }
263
+
264
+ /**
265
+ * The education brief `agentc session add` writes into a connected LOCAL
266
+ * session's CLAUDE.md/AGENTS.md — the same content family as
267
+ * `AGENT_COMPOSE_MANUAL`, rendered for the local context (one source,
268
+ * rendered per context). The manual above must stay BYTE-IDENTICAL (the
269
+ * server and base-env bake static copies of it); this function renders a
270
+ * sibling document and never touches it. The local context differs from the
271
+ * sandbox in exactly the ways stated here: auth rides the bridge credential
272
+ * fallback instead of injected run env; no factory drive is mounted — the
273
+ * files on this machine belong to the human; and the bound conversation is
274
+ * MIRROR-ONLY (ADR-0055 §8.6) — teammates read along but can never message
275
+ * the session through it, so the brief must not promise an inbound channel.
276
+ */
277
+ export function buildAddedSessionBrief(p: AddedSessionBriefParams): string {
278
+ const dashboardSection = p.dashboardUrl === null ? "" : `
279
+
280
+ ## Dashboard
281
+
282
+ Teammates follow this conversation (and the rest of the factory) in the
283
+ Agent Compose dashboard: ${p.dashboardUrl}`;
284
+
285
+ return `# Connected to Agent Compose
286
+
287
+ Agent Compose is your team's agent platform: durable conversations,
288
+ workflow runs, and shared factory drives where humans and agents work
289
+ together. THIS terminal's claude-code session is connected to Agent
290
+ Compose conversation \`${p.conversationId}\` on ${p.serverUrl}.
291
+
292
+ That conversation is a LIVE, READ-ONLY MIRROR of this terminal session:
293
+ teammates read along in the dashboard as the work happens, but they cannot
294
+ message you through it — anything posted there is answered by the server
295
+ with a notice and never reaches this terminal. Everything you do here is
296
+ mirrored automatically; you have an audience, not a channel.
297
+
298
+ ## The \`agentc\` toolbelt
299
+
300
+ The \`agentc\` CLI works from this shell. It is already authenticated on
301
+ this machine via the bridge credential fallback — no keys to manage,
302
+ commands just work:
303
+
304
+ agentc list # registered workflows
305
+ agentc logs <run-id> # a run's logs
306
+ agentc invoke <workflow> -i '<json>' # dispatch a workflow
307
+ agentc events list # read the factory timeline
308
+
309
+ ## Scope — this is your LOCAL machine
310
+
311
+ The files here are YOURS: no factory drive is mounted in this session,
312
+ and nothing you write locally lands on a shared drive by itself. Cloud
313
+ drive/branch semantics (per-run directories on the factory drive, drive
314
+ branches, persist-by-default outputs) apply only to cloud sessions —
315
+ not here.${dashboardSection}
316
+ `;
317
+ }
318
+
136
319
  /**
137
320
  * One connector this run can reach, as the agent should see it. Strictly
138
321
  * NON-SECRET — hosts, methods, paths, identity only. The access token is
@@ -39,6 +39,24 @@ export function parseAgentStatus(text: string): AgentStatus | null {
39
39
 
40
40
  const DEFAULT_ALLOWED_TOOLS = ["Read", "Write", "Edit", "Bash", "Glob", "Grep", "WebFetch"];
41
41
 
42
+ /**
43
+ * A `responseSchema` rendered for the runtime's structured-output surface.
44
+ *
45
+ * zod v4's `toJSONSchema` stamps `$schema: "…/draft/2020-12/schema"` on the
46
+ * result. Claude Code validates `--json-schema` with a validator that has no
47
+ * 2020-12 meta-schema registered, so any schema CARRYING that header is
48
+ * rejected at CLI startup — the process exits 1 before its first API call
49
+ * and every agent with a responseSchema dies on spawn (observed live on
50
+ * claude 2.1.212: `--json-schema is not a valid JSON Schema: no schema with
51
+ * key or ref "https://json-schema.org/draft/2020-12/schema"`). The header is
52
+ * pure metadata — drop it; the schema body is draft-07-compatible for every
53
+ * shape zod emits from our workflow schemas.
54
+ */
55
+ export function runtimeJsonSchema(schema: z.ZodType<unknown>): Record<string, unknown> {
56
+ const { $schema: _$schema, ...rest } = z.toJSONSchema(schema) as Record<string, unknown>;
57
+ return rest;
58
+ }
59
+
42
60
  export interface AgentLoopResult<TResponse = unknown> {
43
61
  agentId: string;
44
62
  label: string;
@@ -71,7 +89,12 @@ function preview(value: unknown): string {
71
89
  }
72
90
  }
73
91
 
74
- export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
92
+ /** Everything but the live-only streaming chunk: `text_delta` never becomes
93
+ * an agent.message event (the terminating `text` carries the whole block) —
94
+ * the loop filters it before summarizing. */
95
+ type DurableAgentMessage = Exclude<AgentMessage, { type: "text_delta" } | { type: "usage_delta" }>;
96
+
97
+ export function summarizeAgentMessage(msg: DurableAgentMessage): AgentMessageSummary {
75
98
  switch (msg.type) {
76
99
  case "init": return { type: "init", sessionId: msg.sessionId };
77
100
  case "text": return { type: "text", text: msg.text };
@@ -103,6 +126,11 @@ export type AgentLifecycleEvent =
103
126
  /** Short runtime self-identifier (`claude`, `openai-desktop`, …).
104
127
  * Drives the per-agent runtime icon on the dashboard. */
105
128
  runtimeKind?: string;
129
+ /** Authored plan-phase this agent belongs to (e.g. dynamic-task's
130
+ * `phase.name`). Purely observability: the dashboard groups agents
131
+ * under named phases without parsing the label prefix. Optional and
132
+ * additive — absent for agents outside a phased plan. */
133
+ phase?: string;
106
134
  }
107
135
  | { event: "agent.message"; at: number; agentId: string; label: string; iteration: number; message: AgentMessageSummary }
108
136
  | { event: "agent.iteration"; at: number; agentId: string; label: string; iteration: number; status: AgentStatus | null }
@@ -111,6 +139,9 @@ export type AgentLifecycleEvent =
111
139
  export interface AgentLoopOpts<TResponse = unknown> {
112
140
  agentId?: string;
113
141
  label?: string;
142
+ /** Authored plan-phase name carried onto `agent.spawned` (see
143
+ * AgentLifecycleEvent.phase). Optional, observability-only. */
144
+ phase?: string;
114
145
  onAgentLifecycleEvent?: (event: AgentLifecycleEvent) => void;
115
146
  onIteration?: (iteration: number, status: AgentStatus | null) => void;
116
147
  turnsPerIteration?: number;
@@ -208,7 +239,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
208
239
  // `session/request_permission` path through CliAgentRunner.gateToolCall) can
209
240
  // raise a human-approval `ctx.pause`, not just on the loop's own hooks.
210
241
  ...(opts.pause ? { pause: opts.pause } : {}),
211
- ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: z.toJSONSchema(opts.responseSchema) as Record<string, unknown> } } : {}),
242
+ ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: runtimeJsonSchema(opts.responseSchema) } } : {}),
212
243
  });
213
244
 
214
245
  // Heads-up when a caller registers tool-call gating on a runtime that
@@ -284,6 +315,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
284
315
  allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
285
316
  ...(client.model != null ? { model: client.model } : {}),
286
317
  ...(client.kind != null ? { runtimeKind: client.kind } : {}),
318
+ ...(opts.phase != null ? { phase: opts.phase } : {}),
287
319
  });
288
320
  }
289
321
 
@@ -428,6 +460,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
428
460
  signal: loopAbort.signal,
429
461
  ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
430
462
  })) {
463
+ // Live-only streaming chunk: the terminating `text` message carries
464
+ // the complete block, so the loop (accumulation, events, processors)
465
+ // ignores deltas — they exist for progressive-rendering consumers
466
+ // (the conversation cloud executor), not the workflow event stream.
467
+ if (rawMsg.type === "text_delta" || rawMsg.type === "usage_delta") continue;
431
468
  // processOutput chain — deny drops the message from accumulation;
432
469
  // abort ends the loop. Continue carries the (possibly mutated)
433
470
  // message forward.
@@ -441,6 +478,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
441
478
  continue;
442
479
  }
443
480
  const msg = outputVerdict.value;
481
+ if (msg.type === "text_delta" || msg.type === "usage_delta") continue; // a processor cannot re-introduce a live-only chunk
444
482
  opts.onAgentEvent?.(iteration, msg);
445
483
  // Usage summaries carry the resolved model so the server can price
446
484
  // token rows per model without correlating back to agent.spawned.
@@ -183,6 +183,10 @@ export interface AgentOpts<T = unknown> {
183
183
  responseSchema?: z.ZodType<T>;
184
184
  /** Label prefix for runtime stderr ("[sbid][agent]" by default). */
185
185
  label?: string;
186
+ /** Authored plan-phase this agent belongs to. Rides `agent.spawned` so the
187
+ * dashboard groups agents under named phases. Optional, additive,
188
+ * observability-only — it never affects execution. */
189
+ phase?: string;
186
190
  /** Lifecycle event sink from workflow ctx. Emits agent.spawned / iteration / settled. */
187
191
  events?: { emit: (event: AgentLifecycleEvent) => void | Promise<void> };
188
192
  /** Per-message event callback — wire this to your workflow's event
@@ -349,6 +353,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
349
353
  runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, { ...runtimeOpts, agentManual: buildAgentContextDoc(process.env) }),
350
354
  agentId,
351
355
  ...(opts.label !== undefined ? { label: opts.label } : {}),
356
+ ...(opts.phase !== undefined ? { phase: opts.phase } : {}),
352
357
  ...(opts.budget?.turnsPerIteration !== undefined ? { turnsPerIteration: opts.budget.turnsPerIteration } : {}),
353
358
  ...(opts.budget?.maxIterations !== undefined ? { maxIterations: opts.budget.maxIterations } : {}),
354
359
  ...(opts.tools !== undefined ? { allowedTools: opts.tools } : {}),