@agent-compose/sdk 0.8.4 → 0.8.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +9 -1
  3. package/dist/agent/agent-loop.d.ts +14 -6
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +250 -59
  7. package/dist/directives.d.ts +14 -0
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +13 -11
  12. package/dist/index.js +1692 -194
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +278 -58
  15. package/dist/runtimes/claude-code.d.ts +90 -1
  16. package/dist/runtimes/claude.d.ts +1 -1
  17. package/dist/runtimes/codex.d.ts +94 -6
  18. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  19. package/dist/runtimes/openai-desktop.d.ts +50 -0
  20. package/dist/runtimes/openai-desktop.js +1689 -211
  21. package/dist/runtimes/openai-desktop.test.d.ts +20 -0
  22. package/dist/runtimes/opencode.d.ts +48 -11
  23. package/dist/runtimes/opencode.test.d.ts +14 -0
  24. package/dist/runtimes/tool-pulse.test.d.ts +17 -0
  25. package/dist/sandbox/baked-clis.d.ts +75 -0
  26. package/dist/sandbox/devbox.d.ts +5 -5
  27. package/dist/sandbox/exec-stream.d.ts +1 -2
  28. package/dist/sandbox/network-policy.d.ts +23 -5
  29. package/dist/sandbox/registry.d.ts +12 -0
  30. package/dist/sandbox/sizes.d.ts +11 -5
  31. package/dist/sandbox.d.ts +5 -3
  32. package/dist/step-invocation/protocol.d.ts +3 -4
  33. package/dist/step-invocation/server.d.ts +2 -2
  34. package/dist/step-invocation/types.d.ts +2 -2
  35. package/dist/types/api-conversations.d.ts +513 -27
  36. package/dist/types/api-factory.d.ts +183 -3
  37. package/dist/types/api-projects.d.ts +480 -0
  38. package/dist/types/api-runs.d.ts +8 -0
  39. package/dist/types/api-scopes.d.ts +32 -3
  40. package/dist/types/conversation-stream.d.ts +27 -1
  41. package/dist/types/execution-context.d.ts +1 -1
  42. package/dist/types/protocol.d.ts +182 -2
  43. package/dist/types/runtime.d.ts +80 -2
  44. package/dist/types/workflow-metadata.d.ts +2 -4
  45. package/dist/types/workflow-plan.d.ts +1 -3
  46. package/dist/utils/bundler.d.ts +23 -0
  47. package/dist/workflow-steps/observability.d.ts +2 -3
  48. package/dist/workflow-steps/runner.d.ts +5 -8
  49. package/dist/workflow-steps/types.d.ts +8 -10
  50. package/dist/workflow-steps/workflow.d.ts +2 -1
  51. package/dist/workflows/engine.d.ts +3 -5
  52. package/dist/workflows/invoke-child.d.ts +2 -2
  53. package/package.json +2 -2
  54. package/src/agent/agent-context.ts +193 -116
  55. package/src/agent/agent-loop.ts +16 -9
  56. package/src/agent/desktop-open.ts +13 -1
  57. package/src/agent/perf-sampler.ts +54 -3
  58. package/src/agent/run-agent.ts +1 -1
  59. package/src/client.ts +418 -80
  60. package/src/directives.ts +21 -1
  61. package/src/display.ts +12 -0
  62. package/src/errors.ts +1 -0
  63. package/src/generated/agentc-commands.ts +571 -0
  64. package/src/index.ts +65 -18
  65. package/src/pause/pause-core.ts +2 -1
  66. package/src/request-context/request-context.ts +1 -1
  67. package/src/runtimes/_cli-agent.ts +607 -132
  68. package/src/runtimes/claude-code.ts +427 -20
  69. package/src/runtimes/claude.ts +1 -1
  70. package/src/runtimes/codex.ts +188 -19
  71. package/src/runtimes/openai-desktop.ts +82 -19
  72. package/src/runtimes/opencode.ts +195 -26
  73. package/src/sandbox/baked-clis.ts +86 -0
  74. package/src/sandbox/devbox.ts +5 -5
  75. package/src/sandbox/exec-stream.ts +1 -2
  76. package/src/sandbox/network-policy.ts +51 -7
  77. package/src/sandbox/providers/e2b.ts +63 -19
  78. package/src/sandbox/providers/vercel.ts +6 -6
  79. package/src/sandbox/registry.ts +19 -1
  80. package/src/sandbox/sizes.ts +11 -5
  81. package/src/sandbox.ts +9 -2
  82. package/src/step-invocation/invoker.ts +2 -6
  83. package/src/step-invocation/protocol.ts +3 -4
  84. package/src/step-invocation/server.ts +2 -2
  85. package/src/types/api-conversations.ts +424 -29
  86. package/src/types/api-factory.ts +189 -3
  87. package/src/types/api-projects.ts +443 -0
  88. package/src/types/api-runs.ts +5 -0
  89. package/src/types/api-scopes.ts +32 -3
  90. package/src/types/conversation-stream.ts +29 -1
  91. package/src/types/execution-context.ts +1 -1
  92. package/src/types/protocol.ts +180 -2
  93. package/src/types/runtime.ts +71 -2
  94. package/src/types/sandbox-environment.ts +1 -2
  95. package/src/types/workflow-metadata.ts +2 -4
  96. package/src/types/workflow-plan.ts +1 -3
  97. package/src/utils/bundler.ts +88 -19
  98. package/src/workflow-steps/observability.ts +2 -3
  99. package/src/workflow-steps/runner.ts +5 -8
  100. package/src/workflow-steps/types.ts +8 -10
  101. package/src/workflow-steps/workflow.ts +2 -1
  102. package/src/workflows/engine.ts +3 -5
  103. package/src/workflows/invoke-child.ts +2 -2
  104. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  105. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  106. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
@@ -71,6 +71,21 @@ export interface AgentMessageError extends AgentMessageBase {
71
71
  type: "error";
72
72
  text: string;
73
73
  }
74
+ /** One model's share of a harness's turn-end report (claude-code
75
+ * `result.modelUsage[<model>]`): the same four token classes as the turn
76
+ * totals, plus what the harness reports beside them. `costUsd` is the
77
+ * harness's OWN estimate at its price table — never a bill. Fields the
78
+ * harness did not report are absent, never zeroed. */
79
+ export interface AgentMessageModelUsage {
80
+ inputTokens: number;
81
+ outputTokens: number;
82
+ cacheReadTokens: number;
83
+ cacheCreationTokens: number;
84
+ /** Thinking tokens, already counted inside `outputTokens`. */
85
+ thinkingTokens?: number;
86
+ webSearchRequests?: number;
87
+ costUsd?: number;
88
+ }
74
89
  export interface AgentMessageUsage extends AgentMessageBase {
75
90
  type: "usage";
76
91
  inputTokens: number;
@@ -79,6 +94,53 @@ export interface AgentMessageUsage extends AgentMessageBase {
79
94
  cacheCreationTokens: number;
80
95
  durationMs: number;
81
96
  numTurns: number;
97
+ /** Reasoning tokens, already counted inside `outputTokens` (codex
98
+ * `turn.completed.usage.reasoning_output_tokens`). Absent when the
99
+ * harness reports no such class. */
100
+ reasoningOutputTokens?: number;
101
+ /** Per-model totals the harness reported beside the turn totals
102
+ * (claude-code `result.modelUsage`): every model the query pipeline
103
+ * called — main loop, subagents, compaction. As the CLI reports them:
104
+ * CUMULATIVE for the guest session (a streaming-input or resumed
105
+ * session carries its earlier turns), so a per-turn share is the
106
+ * difference from the previous report of the same session. Absent when
107
+ * the harness reports none. */
108
+ byModel?: Record<string, AgentMessageModelUsage>;
109
+ /** The harness's own cost estimate for the same scope as `byModel`
110
+ * (claude-code `result.total_cost_usd`): list-price arithmetic, an
111
+ * estimate and never a billing statement. */
112
+ costUsd?: number;
113
+ }
114
+ /** The harness's reading of the account's PLAN LIMITS (claude-code
115
+ * `rate_limit_event`, emitted whenever its rate-limit information changes
116
+ * — subscription-funded sessions only; the headers it reads exist for
117
+ * claude.ai plans). `status` is the verdict for the request just made:
118
+ * `rejected` means the plan's wall, with `resetsAt` the authoritative
119
+ * reset. `window` names which window the verdict speaks for (the CLI's
120
+ * `rateLimitType`: five_hour, seven_day, seven_day_opus, …). `utilization`
121
+ * is carried only once a window crosses a warning threshold (the CLI omits
122
+ * it while plainly allowed), as the CLI reports it — a 0..1 fraction of
123
+ * the window. Everything the event did not carry is absent; nothing is
124
+ * invented. Additive kind: existing producers never emit it. */
125
+ export interface AgentMessagePlanLimits extends AgentMessageBase {
126
+ type: "plan_limits";
127
+ status: "allowed" | "allowed_warning" | "rejected";
128
+ window?: string;
129
+ /** ISO time the named window resets. */
130
+ resetsAt?: string;
131
+ utilization?: number;
132
+ /** The warning threshold the window crossed (as the CLI reports it). */
133
+ surpassedThreshold?: number;
134
+ /** The plan's extra-usage (overage) lane, when the event spoke of it. */
135
+ overage?: {
136
+ status?: "allowed" | "allowed_warning" | "rejected";
137
+ resetsAt?: string;
138
+ disabledReason?: string;
139
+ inUse?: boolean;
140
+ };
141
+ /** Which spend limit blocked the request when not the member's own cap. */
142
+ limitScope?: string;
143
+ errorCode?: string;
82
144
  }
83
145
  /** LIVE-ONLY incremental usage off the harness's raw provider stream — the
84
146
  * `text_delta` of token counts. claude-code's `--include-partial-messages`
@@ -109,7 +171,8 @@ export interface AgentMessagePlan extends AgentMessageBase {
109
171
  type: "plan";
110
172
  entries: {
111
173
  content: string;
112
- priority: "high" | "medium" | "low";
174
+ /** ACP names one; codex's `--json` plan names none. */
175
+ priority?: "high" | "medium" | "low";
113
176
  status: "pending" | "in_progress" | "completed";
114
177
  }[];
115
178
  }
@@ -144,6 +207,71 @@ export interface AgentMessageTaskNotification extends AgentMessageBase {
144
207
  durationMs?: number;
145
208
  };
146
209
  }
210
+ /** One entry of a harness workflow's live progress feed — the
211
+ * `workflow_progress` array claude-code's `system`/`task_progress` events
212
+ * carry for its in-harness Workflow tool (the dynamic-workflow
213
+ * orchestrator). `phase` entries are the script's declared phases (seeded
214
+ * up front, 1-based `index`); `agent` entries are the spawned workflow
215
+ * agents, updated in place as they queue → run → settle. Preview text the
216
+ * harness includes (prompt/result previews) is internal plumbing and is
217
+ * deliberately NOT forwarded — same rule as task notifications. */
218
+ export type WorkflowProgressEntry = {
219
+ kind: "phase";
220
+ index: number;
221
+ title: string;
222
+ } | {
223
+ kind: "agent";
224
+ index: number;
225
+ label: string;
226
+ /** Lifecycle: `start` (spawned) / `progress` (heartbeat) are live;
227
+ * `done` / `error` are settled. Verbatim from the harness. */
228
+ state: "start" | "progress" | "done" | "error";
229
+ phaseIndex?: number;
230
+ phaseTitle?: string;
231
+ model?: string;
232
+ agentId?: string;
233
+ /** Self-reported usage so far (cumulative for this agent). */
234
+ tokens?: number;
235
+ toolCalls?: number;
236
+ durationMs?: number;
237
+ /** Epoch ms the agent actually started (for live elapsed). */
238
+ startedAt?: number;
239
+ /** The failure message when `state: "error"`, clamped. */
240
+ error?: string;
241
+ /** Replayed from a resume cache — settled instantly, no fresh spend. */
242
+ cached?: true;
243
+ /** Skipped by the user (workflow dialog) — an error state that is
244
+ * not a failure. */
245
+ skipped?: true;
246
+ };
247
+ /** LIVE progress of a harness BACKGROUND task (claude-code
248
+ * `system`/`task_progress`). For the in-harness Workflow tool the message
249
+ * carries the CUMULATIVE `workflow_progress` entry array (the harness
250
+ * re-sends the whole picture: state changes immediately, heartbeats
251
+ * throttled ~10s), so a consumer treats the latest message as
252
+ * authoritative per entry `(kind, index)`. A plain background AGENT task
253
+ * (Task tool, run_in_background) heartbeats on the same event with NO
254
+ * workflow entries — forwarded with `workflow: []` as liveness evidence
255
+ * so the platform can declare the running task as session background
256
+ * work; its completion evidence rides `task_notification`. `toolUseId`
257
+ * names the spawning call — the correlation key to its card. Additive
258
+ * kind: existing producers never emit it. */
259
+ export interface AgentMessageTaskProgress extends AgentMessageBase {
260
+ type: "task_progress";
261
+ /** The harness's background task id. */
262
+ taskId: string;
263
+ /** The SPAWNING Workflow/Task call's tool_use id, when carried. */
264
+ toolUseId?: string;
265
+ /** The task's cumulative usage totals so far. */
266
+ usage?: {
267
+ tokens?: number;
268
+ toolUses?: number;
269
+ durationMs?: number;
270
+ };
271
+ /** The cumulative workflow progress entries — EMPTY for a plain
272
+ * background Agent-task heartbeat (only Workflow tasks carry entries). */
273
+ workflow: WorkflowProgressEntry[];
274
+ }
147
275
  /** HARNESS-authored notice text — content the CLI composed itself rather
148
276
  * than the model speaking: slash-command stdout, model/skills advisories,
149
277
  * queued-input notes. claude-code marks these structurally (assistant
@@ -156,7 +284,59 @@ export interface AgentMessageHarnessNotice extends AgentMessageBase {
156
284
  type: "harness_notice";
157
285
  text: string;
158
286
  }
159
- export type AgentMessage = AgentMessageInit | AgentMessageText | AgentMessageTextDelta | AgentMessageThinking | AgentMessageToolUse | AgentMessageToolResult | AgentMessageDone | AgentMessageError | AgentMessageUsage | AgentMessageUsageDelta | AgentMessagePlan | AgentMessageTaskNotification | AgentMessageHarnessNotice;
287
+ /** Context-compaction lifecycle — the harness summarizing its own
288
+ * conversation to reclaim context. Mapped 1:1 from claude-code's wire
289
+ * (verified live on 2.1.212, the baked sandbox pin, and 2.1.241 — both
290
+ * emit the identical shapes, `/compact` and auto alike):
291
+ *
292
+ * `system`/`status` `{status:"compacting"}` → phase "start"
293
+ * `system`/`status` `{status:null, compact_result, → phase "settled"
294
+ * compact_error?}`
295
+ * `system`/`compact_boundary` `{compact_metadata: → phase "boundary"
296
+ * {trigger, pre_tokens, post_tokens,
297
+ * cumulative_dropped_tokens, duration_ms, …}}`
298
+ *
299
+ * Order on the wire: start → settled → (fresh init) → boundary → the
300
+ * continuation summary as a SYNTHETIC user message (never forwarded — it
301
+ * quotes conversation content verbatim). "boundary" only follows a
302
+ * successful settle and only on the turn the compaction ran (verified: it
303
+ * does NOT replay on later resumes). A compaction can span MINUTES of
304
+ * otherwise-silent stream — the whole point of forwarding it is that
305
+ * downstream can show the silence as work (the 2026-08-23 dead-air
306
+ * incident: 94% auto-compact read as a dead session). Additive kind:
307
+ * existing producers never emit it. */
308
+ export interface AgentMessageCompaction extends AgentMessageBase {
309
+ type: "compaction";
310
+ phase: "start" | "settled" | "boundary";
311
+ /** settled: how it ended. Absent on start/boundary (a boundary IS a
312
+ * success by construction — the harness only emits it after one). */
313
+ result?: "success" | "failed";
314
+ /** settled+failed: the harness's own reason, clamped. */
315
+ error?: string;
316
+ /** boundary: what initiated the compaction. */
317
+ trigger?: "auto" | "manual";
318
+ /** boundary: context tokens before / after, dropped total, wall time. */
319
+ preTokens?: number;
320
+ postTokens?: number;
321
+ droppedTokens?: number;
322
+ durationMs?: number;
323
+ }
324
+ /** A user-role message landing INSIDE a subagent's thread — the delivered
325
+ * form of a steer (the parent's `SendMessage` to a RUNNING child, queued
326
+ * "for delivery at its next tool round") or any other message the harness
327
+ * folds into a child's conversation mid-flight. Emitted ONLY with sidechain
328
+ * attribution: `parentToolUseId` (the spawning Agent/Task call's tool_use
329
+ * id) is REQUIRED — an unattributed user event is the parent's own prompt
330
+ * echo, which stays unmapped as before. Lets renderers show the steer as a
331
+ * user-role message inside the child's mini-session instead of leaving it
332
+ * an opaque SendMessage tool call on the parent only (task #97, owner
333
+ * directive 2026-08-27). Additive kind: existing producers never emit it. */
334
+ export interface AgentMessageSubagentUserMessage extends AgentMessageBase {
335
+ type: "subagent_user_message";
336
+ text: string;
337
+ parentToolUseId: string;
338
+ }
339
+ export type AgentMessage = AgentMessageInit | AgentMessageText | AgentMessageTextDelta | AgentMessageThinking | AgentMessageToolUse | AgentMessageToolResult | AgentMessageDone | AgentMessageError | AgentMessageUsage | AgentMessageUsageDelta | AgentMessagePlanLimits | AgentMessagePlan | AgentMessageTaskNotification | AgentMessageTaskProgress | AgentMessageHarnessNotice | AgentMessageCompaction | AgentMessageSubagentUserMessage;
160
340
  /** Status block the agent emits to signal iteration completion or blockers. */
161
341
  export interface AgentStatus {
162
342
  summary: string;
@@ -93,6 +93,20 @@ export interface RuntimeOptions {
93
93
  * Absent ⇒ nothing extra is sourced. Must be a plain absolute path — no
94
94
  * quotes, no `..`; the runtime validates and drops anything else. */
95
95
  credEnvFile?: string;
96
+ /** Durable launch report (boot-time turn adoption): called by the durable
97
+ * detached transport the moment its in-guest runner exists — with the
98
+ * guest prompt path (every durable file derives from it), the exit
99
+ * sentinel string, and the detached wrapper's pid. The caller stamps
100
+ * these on the turn row so a SUCCESSOR process (a deploy roll's new
101
+ * server) can re-attach to the runner's durable `.out` without this
102
+ * process's memory. Fired once per detached launch (a retry with a fresh
103
+ * runner fires again with the new paths); never on transports without
104
+ * durable files. Must not throw — the transport calls it inline. */
105
+ onDetachedLaunch?: (info: {
106
+ promptPath: string;
107
+ sentinel: string;
108
+ pid: number;
109
+ }) => void;
96
110
  }
97
111
  /** Three-valued liveness verdict for a runtime's CURRENT turn, read from
98
112
  * DURABLE guest state (heartbeat file, stdout file, exit sentinel, pid) over
@@ -184,6 +198,25 @@ export interface ModelExecutionContract {
184
198
  * fallback probe it already had; null is never a verdict.
185
199
  */
186
200
  probeTurnLiveness?(): Promise<RunnerLivenessVerdict | null>;
201
+ /**
202
+ * TELEMETRY, NEVER A VERDICT (tool-run pulse, 2026-08-29): the newest
203
+ * tool-run pulse the durable liveness probe carried — the guest
204
+ * heartbeat's sample of the runner's own session (aggregate CPU jiffies,
205
+ * written bytes, live process count) plus the guest clock it was read
206
+ * against. The executor's evidence ticker peeks it AFTER its liveness
207
+ * check and compares successive samples: counters ADVANCING is proof a
208
+ * long silent foreground tool is working, fanned to clients as a live
209
+ * `tool_pulse` frame. Never probes on its own; null before any probe or
210
+ * on a transport without the pulse file. No liveness decision may ever
211
+ * read it.
212
+ */
213
+ peekTurnPulse?(): {
214
+ atMs: number;
215
+ cpuJiffies: number;
216
+ ioBytes: number;
217
+ procs: number;
218
+ guestNowMs: number;
219
+ } | null;
187
220
  /**
188
221
  * DOORBELL, NEVER A VERDICT (exit-event push, v0.10.43): wake the current
189
222
  * turn's durable watchdog NOW so it runs its normal verification pass —
@@ -233,7 +266,52 @@ export interface ModelExecutionContract {
233
266
  * phase, a spec/transport without stream input, ACP path).
234
267
  * Calls are serialized per turn; never throws.
235
268
  */
236
- injectUserMessage?(text: string): Promise<"delivered" | "pending" | "closed" | "unsupported">;
269
+ injectUserMessage?(text: string,
270
+ /** Who the message speaks for when it is not the session user
271
+ * (MidTurnEnvelopeOptions): a relayed person, named, or the thread
272
+ * agent that owns this worker. */
273
+ opts?: {
274
+ relayedFrom?: string | null;
275
+ fromOwnerAgent?: boolean;
276
+ }): Promise<"delivered" | "pending" | "closed" | "unsupported">;
277
+ /**
278
+ * Request an in-band STEP INTERRUPT of the currently running turn — the
279
+ * ESC equivalent. Where `injectUserMessage` queues content for the turn
280
+ * loop's next boundary, this rides the same durable inbox but carries a
281
+ * control line the CLI handles immediately, mid-step included: the
282
+ * running tool call aborts, the run ends within ~100ms, and the guest
283
+ * session stays resumable with the whole turn context (verified live
284
+ * against claude 2.1.236). Verdicts mirror `injectUserMessage`; only
285
+ * "delivered" means the CLI got the control line — callers escalate
286
+ * anything else (and "unsupported": no stream-input turn, or a runtime
287
+ * with no in-band interrupt, e.g. codex) to kill semantics, which stay
288
+ * honest because thread stores are durable and the successor turn
289
+ * resumes them. Never throws.
290
+ */
291
+ interruptTurn?(): Promise<"delivered" | "pending" | "closed" | "unsupported">;
292
+ /**
293
+ * Guest pid of the CURRENT turn's detached runner wrapper (the setsid
294
+ * process-group leader recorded at launch), or null when no detached
295
+ * durable-transport runner is live (boot phase, ACP path, single-exec
296
+ * transports). Advisory identity, NEVER a liveness verdict: the platform
297
+ * reads it to DECLARE harness-reported background work (an in-harness
298
+ * Workflow task) against the process tree that hosts it, so the park
299
+ * machinery can verify the tree from `/proc/<pid>` later. Runtimes
300
+ * without a detached guest simply omit the method.
301
+ */
302
+ currentRunnerPid?(): number | null;
303
+ /**
304
+ * Durable byte offset of the CURRENT turn's `.out` file just past the
305
+ * last line whose messages have ALL been yielded to the consumer — the
306
+ * safe harvest watermark for boot-time turn adoption. Null when no
307
+ * durable-transport turn is live, or before the first line completes.
308
+ * The contract is deliberately one line BEHIND the parse cursor: a
309
+ * caller that persists parts after each yielded message may stamp this
310
+ * offset at any time and a successor re-parses AT MOST the line whose
311
+ * parts were mid-persist (the same crash window the workflow tailer's
312
+ * flush-before-advance ordering accepts). Advisory, never a verdict.
313
+ */
314
+ currentTurnDurableOffset?(): number | null;
237
315
  sendMessage(opts: {
238
316
  prompt: string;
239
317
  sessionId?: string;
@@ -257,7 +335,7 @@ export interface ModelExecutionContract {
257
335
  * The sandbox provider (e.g. "vercel", "e2b") is an infrastructure concern
258
336
  * configured via SANDBOX_PROVIDER — not part of the runtime definition.
259
337
  * For non-sandbox agents (API calls, etc.) make the call directly in the workflow;
260
- * spawnAgent is a sandbox concept.
338
+ * `agent()` is a sandbox concept.
261
339
  */
262
340
  export interface AgentRuntime<S extends SandboxProvider = SandboxProvider> {
263
341
  create(sandbox: S, opts: RuntimeOptions): ModelExecutionContract;
@@ -224,10 +224,8 @@ export interface WorkflowMetadata {
224
224
  * BUILD — a setup-only workflow whose job is to leave its VM configured and
225
225
  * snapshot it (base-env / agent-env). Environment builds build a platform
226
226
  * IMAGE and never use the shared factory drive, so the server SKIPS mounting
227
- * /factory for them: a live Archil mount baked into the captured snapshot
228
- * fails the NEXT boot's re-mount ("an older Archil process is still running
229
- * for this mountpoint"), degrading /factory for every workflow booting from
230
- * that snapshot. Absent on ordinary workflows — which mount /factory exactly
227
+ * /factory for them and no live drive mount bakes into the captured
228
+ * snapshot. Absent on ordinary workflows — which mount /factory exactly
231
229
  * as before. Optional + additive: an ABSENT flag contributes nothing to the
232
230
  * canonical metadata hash (frozen-metadata rule), so existing workflows are
233
231
  * not forced to re-register. */
@@ -5,9 +5,7 @@
5
5
  * dispatch time. The CLI/bundler inspects the bundled module in the user's
6
6
  * environment and sends this compact plan as metadata.
7
7
  *
8
- * Every workflow has a step plan. Legacy `defineWorkflow({ run })`
9
- * workflows are wrapped at the SDK boundary as a single-step compiled
10
- * workflow (step name = "run"); the bundler sees the same shape regardless.
8
+ * Every workflow has a step plan.
11
9
  */
12
10
  /** One artifact a step promises to produce — mirrors `StepDeliverable`,
13
11
  * restated here so the plan stays a self-contained wire shape. */
@@ -147,6 +147,29 @@ export declare const SDK_SPECIFIER_ALIASES: readonly string[];
147
147
  * null. The single authority: the plugin's regex filter is only a fast
148
148
  * pre-filter, and this decides. */
149
149
  export declare function resolveSdkAlias(specifier: string): string | null;
150
+ /**
151
+ * The packages the PLATFORM provides to every workflow source — the SDK and
152
+ * its zod peer — which therefore resolve from the platform's own install
153
+ * roots when the source file's directory has no node_modules of its own.
154
+ * The live incident (2026-09-20, a claude-code session at the drive root):
155
+ * `agentc invoke --source /factory/files/wf.ts` died with `Could not
156
+ * resolve "@agent-compose/sdk", "zod"` because Bun walked up from
157
+ * /factory/files and found nothing, while the baked SDK sat in
158
+ * /workspace/node_modules the whole time (infra/e2b-template/template.ts —
159
+ * "SDK into /workspace/node_modules so any script written under /workspace
160
+ * can import it"). A workflow's OWN third-party deps still have to be
161
+ * installed beside the source; only these two are platform-resolved.
162
+ */
163
+ export declare const PLATFORM_RESOLVED_PACKAGES: readonly string[];
164
+ /** Env override for the fallback roots (colon-separated, tried first) — the
165
+ * test seam, and an ops knob for a sandbox image that plants the SDK
166
+ * elsewhere. */
167
+ export declare const SDK_FALLBACK_ROOTS_ENV = "AGENT_COMPOSE_SDK_FALLBACK_ROOTS";
168
+ /** Where the platform SDK lives when the source's own walk-up finds nothing:
169
+ * the env override's roots, the sandbox's baked /workspace install, then
170
+ * the bundling process's cwd (the CLI's own resolution context). Exported
171
+ * for the unit test. */
172
+ export declare function sdkFallbackRoots(env?: Record<string, string | undefined>): string[];
150
173
  /**
151
174
  * The SELF-CORRECTING layer. Turns a Bun bundle failure into a message that
152
175
  * names the real fix instead of leaking Bun's internals.
@@ -68,9 +68,8 @@ export interface StepObservability {
68
68
  * this instance. After the step finishes, the engine calls `snapshot()`
69
69
  * to extract the bundle for transport.
70
70
  *
71
- * Metadata writes merge (later keys win) — matches the legacy
72
- * `mergeRunMetadata` semantics so authors don't see a behaviour change
73
- * across migrations.
71
+ * Metadata writes merge (later keys win), as the server's run-metadata
72
+ * merge does.
74
73
  */
75
74
  export declare class StepObservabilityCollector {
76
75
  private metadata;
@@ -6,7 +6,6 @@
6
6
  * 2. For each step:
7
7
  * a. Optionally check `getCachedOutput(stepIndex, step.name)` — if a
8
8
  * previous run completed this step, skip and reuse its output.
9
- * (Phase 1b uses this for crash recovery.)
10
9
  * b. Validate current input against `step.input`.
11
10
  * c. Call `step.run({ input, ... })`.
12
11
  * d. Validate return value against `step.output`.
@@ -17,9 +16,10 @@
17
16
  * Errors during any step bubble through `onStepFailed` and re-throw so the
18
17
  * caller can decide whether to mark the run failed.
19
18
  *
20
- * The cache + completion hooks are injection points — a sandbox engine
21
- * adapter (Phase 1c) wires them to `workflow_step_runs` rows. The default
22
- * is an in-process map for tests.
19
+ * The cache + completion hooks are injection points; without a cache every
20
+ * step runs. The platform's sandbox path does not walk the chain here: the
21
+ * Temporal `executeStep` activity runs one step per runner subprocess via
22
+ * `runWorkflowSingleStep` below.
23
23
  */
24
24
  import { z } from "zod";
25
25
  import type { Workflow, StepRunResult } from "./types.js";
@@ -54,13 +54,10 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
54
54
  /**
55
55
  * Crash-recovery hook. Called before a step executes. Return the cached
56
56
  * output to skip execution; return undefined to run the step.
57
- *
58
- * Phase 1b implementations will look up `workflow_step_runs` rows for
59
- * (runId, stepIndex, stepName) and return completed step outputs here.
60
57
  * Default: always undefined (no caching).
61
58
  */
62
59
  getCachedOutput?(stepIndex: number, stepName: string): unknown | undefined | Promise<unknown | undefined>;
63
- /** Fire after a step's `execute` and output validation succeed. */
60
+ /** Fire after a step's `run` and output validation succeed. */
64
61
  onStepCompleted?(stepIndex: number, stepName: string, output: unknown, durationMs: number): void | Promise<void>;
65
62
  /** Fire when a step throws or fails validation. The error is re-thrown after this returns. */
66
63
  onStepFailed?(stepIndex: number, stepName: string, error: Error, durationMs: number): void | Promise<void>;
@@ -6,9 +6,8 @@
6
6
  *
7
7
  * Why: durable suspend/resume requires step boundaries to be addressable as
8
8
  * data, not opaque async-function bodies. Each step's input + output is
9
- * serialisable JSON so engine adapters (in-process, sandbox, future
10
- * Inngest/Temporal) can record completion and replay from the last
11
- * completed step.
9
+ * serialisable JSON so the durable engine (Temporal, one activity per step)
10
+ * can record completion and replay from the last completed step.
12
11
  */
13
12
  import type { z } from "zod";
14
13
  import type { BaseExecutionContext } from "../types/execution-context.js";
@@ -16,7 +15,7 @@ import type { AgentEventSink } from "../types/workflow.js";
16
15
  import type { WorkflowMetadata } from "../types/workflow-metadata.js";
17
16
  import type { StepObservability } from "./observability.js";
18
17
  /**
19
- * Per-step execution context. Threaded into every step's `execute(...)` so
18
+ * Per-step execution context. Threaded into every step's `run(...)` so
20
19
  * the step can read tenant identity and run identity, log progress, and
21
20
  * invoke sandbox commands.
22
21
  *
@@ -31,9 +30,9 @@ export interface StepContext<TInput = unknown> extends BaseExecutionContext {
31
30
  abortSignal: AbortSignal;
32
31
  /**
33
32
  * Merge key-value metadata onto the run record. Buffered during the
34
- * step and flushed when the step completes; the durable engine writes
35
- * it via the same `mergeRunMetadata` path the legacy runner used, so
36
- * the dashboard sees the same shape. Later keys win.
33
+ * step and flushed when the step completes; the durable engine merges
34
+ * it into the run's metadata (`persistStepObservability`). Later keys
35
+ * win.
37
36
  */
38
37
  setMetadata(data: Record<string, unknown>): Promise<void>;
39
38
  /**
@@ -93,9 +92,8 @@ export interface Step<TInput, TOutput> {
93
92
  readonly deliverables?: readonly StepDeliverable[];
94
93
  }
95
94
  /**
96
- * The result of running one step. Engine adapters persist these into the
97
- * `workflow_step_runs` table (Phase 1b) so subsequent runs can skip
98
- * completed steps. `observability` carries the snapshot of
95
+ * The result of running one step, as `runWorkflowSteps` reports it per
96
+ * step. `observability` carries the snapshot of
99
97
  * `ctx.setMetadata` / `ctx.step` / `ctx.agentEvents` recorded during
100
98
  * the step; undefined when no hooks were used.
101
99
  */
@@ -16,7 +16,8 @@
16
16
  * matches the final step's `output` schema.
17
17
  *
18
18
  * Workflows-as-data — the result is consumable by any engine adapter
19
- * (in-process today; sandbox + Inngest/Temporal in future PRs).
19
+ * (in-process `runWorkflow`; the sandbox runner one step at a time under
20
+ * Temporal).
20
21
  */
21
22
  import type { z } from "zod";
22
23
  import type { Step, Workflow } from "./types.js";
@@ -6,7 +6,7 @@
6
6
  * runner.
7
7
  *
8
8
  * Errors classified into `WorkflowError` (user code threw) vs `EngineError`
9
- * (platform problem) for the runner harness to surface upstream.
9
+ * (platform problem).
10
10
  */
11
11
  import type { WorkflowHooks } from "../types/workflow.js";
12
12
  import type { InvokeChild } from "../types/execution-context.js";
@@ -35,7 +35,7 @@ export declare class EngineError extends Error {
35
35
  readonly subsystem: EngineSubsystem;
36
36
  constructor(message: string, subsystem?: EngineSubsystem, options?: ErrorOptions);
37
37
  }
38
- /** Classify any thrown value into the wire-level `kind` expected by `/fail`. */
38
+ /** Classify any thrown value as a `workflow` or an `engine` failure. */
39
39
  export declare function classifyError(err: unknown): "workflow" | "engine";
40
40
  export declare function parseNameVersion(ref: string): {
41
41
  name: string;
@@ -52,9 +52,7 @@ export interface RunWorkflowOptions {
52
52
  onStepStarted?: RunWorkflowStepsOpts<unknown, unknown>["onStepStarted"];
53
53
  onStepCompleted?: RunWorkflowStepsOpts<unknown, unknown>["onStepCompleted"];
54
54
  onStepFailed?: RunWorkflowStepsOpts<unknown, unknown>["onStepFailed"];
55
- /** Provider-specific child workflow invocation. Temporal/Inngest providers
56
- * inject their native child-workflow primitive; the LocalProvider injects
57
- * the public Agent Compose API client. */
55
+ /** Child workflow invocation. Unset, a step's `invokeChild` throws. */
58
56
  invokeChild?: InvokeChild;
59
57
  }
60
58
  export declare function runWorkflow<TInput, TOutput>(wf: Workflow<TInput, TOutput>, ctx: {
@@ -18,8 +18,8 @@ import type { InvokeChild } from "../types/execution-context.js";
18
18
  */
19
19
  export declare function deriveInvokeChildIdempotencyKey(parentRunId: string, childName: string): string | null;
20
20
  /**
21
- * Build the public-API child workflow invoker used by legacy and sandboxed
22
- * workflow execution. Provider-backed engines may inject a different
21
+ * Build the public-API child workflow invoker used by sandboxed workflow
22
+ * execution (the step runner). Provider-backed engines may inject a different
23
23
  * implementation (Temporal child workflow, Inngest invoke, etc.).
24
24
  */
25
25
  export declare function buildInvokeChild(runId: string, opts?: {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.8.4",
3
+ "version": "0.8.6",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -71,7 +71,7 @@
71
71
  "ofetch": "^1.5.1",
72
72
  "openai": "^6.33.0",
73
73
  "p-retry": "^6.2.0",
74
- "sharp": "^0.34.5"
74
+ "sharp": "0.35.4"
75
75
  },
76
76
  "publishConfig": {
77
77
  "access": "public"