@agent-compose/sdk 0.8.4 → 0.8.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +9 -1
  3. package/dist/agent/agent-loop.d.ts +14 -6
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +250 -59
  7. package/dist/directives.d.ts +14 -0
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +13 -11
  12. package/dist/index.js +1692 -194
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +278 -58
  15. package/dist/runtimes/claude-code.d.ts +90 -1
  16. package/dist/runtimes/claude.d.ts +1 -1
  17. package/dist/runtimes/codex.d.ts +94 -6
  18. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  19. package/dist/runtimes/openai-desktop.d.ts +50 -0
  20. package/dist/runtimes/openai-desktop.js +1689 -211
  21. package/dist/runtimes/openai-desktop.test.d.ts +20 -0
  22. package/dist/runtimes/opencode.d.ts +48 -11
  23. package/dist/runtimes/opencode.test.d.ts +14 -0
  24. package/dist/runtimes/tool-pulse.test.d.ts +17 -0
  25. package/dist/sandbox/baked-clis.d.ts +75 -0
  26. package/dist/sandbox/devbox.d.ts +5 -5
  27. package/dist/sandbox/exec-stream.d.ts +1 -2
  28. package/dist/sandbox/network-policy.d.ts +23 -5
  29. package/dist/sandbox/registry.d.ts +12 -0
  30. package/dist/sandbox/sizes.d.ts +11 -5
  31. package/dist/sandbox.d.ts +5 -3
  32. package/dist/step-invocation/protocol.d.ts +3 -4
  33. package/dist/step-invocation/server.d.ts +2 -2
  34. package/dist/step-invocation/types.d.ts +2 -2
  35. package/dist/types/api-conversations.d.ts +513 -27
  36. package/dist/types/api-factory.d.ts +183 -3
  37. package/dist/types/api-projects.d.ts +480 -0
  38. package/dist/types/api-runs.d.ts +8 -0
  39. package/dist/types/api-scopes.d.ts +32 -3
  40. package/dist/types/conversation-stream.d.ts +27 -1
  41. package/dist/types/execution-context.d.ts +1 -1
  42. package/dist/types/protocol.d.ts +182 -2
  43. package/dist/types/runtime.d.ts +80 -2
  44. package/dist/types/workflow-metadata.d.ts +2 -4
  45. package/dist/types/workflow-plan.d.ts +1 -3
  46. package/dist/utils/bundler.d.ts +23 -0
  47. package/dist/workflow-steps/observability.d.ts +2 -3
  48. package/dist/workflow-steps/runner.d.ts +5 -8
  49. package/dist/workflow-steps/types.d.ts +8 -10
  50. package/dist/workflow-steps/workflow.d.ts +2 -1
  51. package/dist/workflows/engine.d.ts +3 -5
  52. package/dist/workflows/invoke-child.d.ts +2 -2
  53. package/package.json +2 -2
  54. package/src/agent/agent-context.ts +193 -116
  55. package/src/agent/agent-loop.ts +16 -9
  56. package/src/agent/desktop-open.ts +13 -1
  57. package/src/agent/perf-sampler.ts +54 -3
  58. package/src/agent/run-agent.ts +1 -1
  59. package/src/client.ts +418 -80
  60. package/src/directives.ts +21 -1
  61. package/src/display.ts +12 -0
  62. package/src/errors.ts +1 -0
  63. package/src/generated/agentc-commands.ts +571 -0
  64. package/src/index.ts +65 -18
  65. package/src/pause/pause-core.ts +2 -1
  66. package/src/request-context/request-context.ts +1 -1
  67. package/src/runtimes/_cli-agent.ts +607 -132
  68. package/src/runtimes/claude-code.ts +427 -20
  69. package/src/runtimes/claude.ts +1 -1
  70. package/src/runtimes/codex.ts +188 -19
  71. package/src/runtimes/openai-desktop.ts +82 -19
  72. package/src/runtimes/opencode.ts +195 -26
  73. package/src/sandbox/baked-clis.ts +86 -0
  74. package/src/sandbox/devbox.ts +5 -5
  75. package/src/sandbox/exec-stream.ts +1 -2
  76. package/src/sandbox/network-policy.ts +51 -7
  77. package/src/sandbox/providers/e2b.ts +63 -19
  78. package/src/sandbox/providers/vercel.ts +6 -6
  79. package/src/sandbox/registry.ts +19 -1
  80. package/src/sandbox/sizes.ts +11 -5
  81. package/src/sandbox.ts +9 -2
  82. package/src/step-invocation/invoker.ts +2 -6
  83. package/src/step-invocation/protocol.ts +3 -4
  84. package/src/step-invocation/server.ts +2 -2
  85. package/src/types/api-conversations.ts +424 -29
  86. package/src/types/api-factory.ts +189 -3
  87. package/src/types/api-projects.ts +443 -0
  88. package/src/types/api-runs.ts +5 -0
  89. package/src/types/api-scopes.ts +32 -3
  90. package/src/types/conversation-stream.ts +29 -1
  91. package/src/types/execution-context.ts +1 -1
  92. package/src/types/protocol.ts +180 -2
  93. package/src/types/runtime.ts +71 -2
  94. package/src/types/sandbox-environment.ts +1 -2
  95. package/src/types/workflow-metadata.ts +2 -4
  96. package/src/types/workflow-plan.ts +1 -3
  97. package/src/utils/bundler.ts +88 -19
  98. package/src/workflow-steps/observability.ts +2 -3
  99. package/src/workflow-steps/runner.ts +5 -8
  100. package/src/workflow-steps/types.ts +8 -10
  101. package/src/workflow-steps/workflow.ts +2 -1
  102. package/src/workflows/engine.ts +3 -5
  103. package/src/workflows/invoke-child.ts +2 -2
  104. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  105. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  106. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
@@ -74,6 +74,22 @@ export interface AgentMessageError extends AgentMessageBase {
74
74
  text: string;
75
75
  }
76
76
 
77
+ /** One model's share of a harness's turn-end report (claude-code
78
+ * `result.modelUsage[<model>]`): the same four token classes as the turn
79
+ * totals, plus what the harness reports beside them. `costUsd` is the
80
+ * harness's OWN estimate at its price table — never a bill. Fields the
81
+ * harness did not report are absent, never zeroed. */
82
+ export interface AgentMessageModelUsage {
83
+ inputTokens: number;
84
+ outputTokens: number;
85
+ cacheReadTokens: number;
86
+ cacheCreationTokens: number;
87
+ /** Thinking tokens, already counted inside `outputTokens`. */
88
+ thinkingTokens?: number;
89
+ webSearchRequests?: number;
90
+ costUsd?: number;
91
+ }
92
+
77
93
  export interface AgentMessageUsage extends AgentMessageBase {
78
94
  type: "usage";
79
95
  inputTokens: number;
@@ -82,6 +98,54 @@ export interface AgentMessageUsage extends AgentMessageBase {
82
98
  cacheCreationTokens: number;
83
99
  durationMs: number;
84
100
  numTurns: number;
101
+ /** Reasoning tokens, already counted inside `outputTokens` (codex
102
+ * `turn.completed.usage.reasoning_output_tokens`). Absent when the
103
+ * harness reports no such class. */
104
+ reasoningOutputTokens?: number;
105
+ /** Per-model totals the harness reported beside the turn totals
106
+ * (claude-code `result.modelUsage`): every model the query pipeline
107
+ * called — main loop, subagents, compaction. As the CLI reports them:
108
+ * CUMULATIVE for the guest session (a streaming-input or resumed
109
+ * session carries its earlier turns), so a per-turn share is the
110
+ * difference from the previous report of the same session. Absent when
111
+ * the harness reports none. */
112
+ byModel?: Record<string, AgentMessageModelUsage>;
113
+ /** The harness's own cost estimate for the same scope as `byModel`
114
+ * (claude-code `result.total_cost_usd`): list-price arithmetic, an
115
+ * estimate and never a billing statement. */
116
+ costUsd?: number;
117
+ }
118
+
119
+ /** The harness's reading of the account's PLAN LIMITS (claude-code
120
+ * `rate_limit_event`, emitted whenever its rate-limit information changes
121
+ * — subscription-funded sessions only; the headers it reads exist for
122
+ * claude.ai plans). `status` is the verdict for the request just made:
123
+ * `rejected` means the plan's wall, with `resetsAt` the authoritative
124
+ * reset. `window` names which window the verdict speaks for (the CLI's
125
+ * `rateLimitType`: five_hour, seven_day, seven_day_opus, …). `utilization`
126
+ * is carried only once a window crosses a warning threshold (the CLI omits
127
+ * it while plainly allowed), as the CLI reports it — a 0..1 fraction of
128
+ * the window. Everything the event did not carry is absent; nothing is
129
+ * invented. Additive kind: existing producers never emit it. */
130
+ export interface AgentMessagePlanLimits extends AgentMessageBase {
131
+ type: "plan_limits";
132
+ status: "allowed" | "allowed_warning" | "rejected";
133
+ window?: string;
134
+ /** ISO time the named window resets. */
135
+ resetsAt?: string;
136
+ utilization?: number;
137
+ /** The warning threshold the window crossed (as the CLI reports it). */
138
+ surpassedThreshold?: number;
139
+ /** The plan's extra-usage (overage) lane, when the event spoke of it. */
140
+ overage?: {
141
+ status?: "allowed" | "allowed_warning" | "rejected";
142
+ resetsAt?: string;
143
+ disabledReason?: string;
144
+ inUse?: boolean;
145
+ };
146
+ /** Which spend limit blocked the request when not the member's own cap. */
147
+ limitScope?: string;
148
+ errorCode?: string;
85
149
  }
86
150
 
87
151
  /** LIVE-ONLY incremental usage off the harness's raw provider stream — the
@@ -114,7 +178,8 @@ export interface AgentMessagePlan extends AgentMessageBase {
114
178
  type: "plan";
115
179
  entries: {
116
180
  content: string;
117
- priority: "high" | "medium" | "low";
181
+ /** ACP names one; codex's `--json` plan names none. */
182
+ priority?: "high" | "medium" | "low";
118
183
  status: "pending" | "in_progress" | "completed";
119
184
  }[];
120
185
  }
@@ -147,6 +212,61 @@ export interface AgentMessageTaskNotification extends AgentMessageBase {
147
212
  usage?: { tokens?: number; toolUses?: number; durationMs?: number };
148
213
  }
149
214
 
215
+ /** One entry of a harness workflow's live progress feed — the
216
+ * `workflow_progress` array claude-code's `system`/`task_progress` events
217
+ * carry for its in-harness Workflow tool (the dynamic-workflow
218
+ * orchestrator). `phase` entries are the script's declared phases (seeded
219
+ * up front, 1-based `index`); `agent` entries are the spawned workflow
220
+ * agents, updated in place as they queue → run → settle. Preview text the
221
+ * harness includes (prompt/result previews) is internal plumbing and is
222
+ * deliberately NOT forwarded — same rule as task notifications. */
223
+ export type WorkflowProgressEntry =
224
+ | { kind: "phase"; index: number; title: string }
225
+ | {
226
+ kind: "agent"; index: number; label: string;
227
+ /** Lifecycle: `start` (spawned) / `progress` (heartbeat) are live;
228
+ * `done` / `error` are settled. Verbatim from the harness. */
229
+ state: "start" | "progress" | "done" | "error";
230
+ phaseIndex?: number; phaseTitle?: string;
231
+ model?: string; agentId?: string;
232
+ /** Self-reported usage so far (cumulative for this agent). */
233
+ tokens?: number; toolCalls?: number; durationMs?: number;
234
+ /** Epoch ms the agent actually started (for live elapsed). */
235
+ startedAt?: number;
236
+ /** The failure message when `state: "error"`, clamped. */
237
+ error?: string;
238
+ /** Replayed from a resume cache — settled instantly, no fresh spend. */
239
+ cached?: true;
240
+ /** Skipped by the user (workflow dialog) — an error state that is
241
+ * not a failure. */
242
+ skipped?: true;
243
+ };
244
+
245
+ /** LIVE progress of a harness BACKGROUND task (claude-code
246
+ * `system`/`task_progress`). For the in-harness Workflow tool the message
247
+ * carries the CUMULATIVE `workflow_progress` entry array (the harness
248
+ * re-sends the whole picture: state changes immediately, heartbeats
249
+ * throttled ~10s), so a consumer treats the latest message as
250
+ * authoritative per entry `(kind, index)`. A plain background AGENT task
251
+ * (Task tool, run_in_background) heartbeats on the same event with NO
252
+ * workflow entries — forwarded with `workflow: []` as liveness evidence
253
+ * so the platform can declare the running task as session background
254
+ * work; its completion evidence rides `task_notification`. `toolUseId`
255
+ * names the spawning call — the correlation key to its card. Additive
256
+ * kind: existing producers never emit it. */
257
+ export interface AgentMessageTaskProgress extends AgentMessageBase {
258
+ type: "task_progress";
259
+ /** The harness's background task id. */
260
+ taskId: string;
261
+ /** The SPAWNING Workflow/Task call's tool_use id, when carried. */
262
+ toolUseId?: string;
263
+ /** The task's cumulative usage totals so far. */
264
+ usage?: { tokens?: number; toolUses?: number; durationMs?: number };
265
+ /** The cumulative workflow progress entries — EMPTY for a plain
266
+ * background Agent-task heartbeat (only Workflow tasks carry entries). */
267
+ workflow: WorkflowProgressEntry[];
268
+ }
269
+
150
270
  /** HARNESS-authored notice text — content the CLI composed itself rather
151
271
  * than the model speaking: slash-command stdout, model/skills advisories,
152
272
  * queued-input notes. claude-code marks these structurally (assistant
@@ -160,6 +280,60 @@ export interface AgentMessageHarnessNotice extends AgentMessageBase {
160
280
  text: string;
161
281
  }
162
282
 
283
+ /** Context-compaction lifecycle — the harness summarizing its own
284
+ * conversation to reclaim context. Mapped 1:1 from claude-code's wire
285
+ * (verified live on 2.1.212, the baked sandbox pin, and 2.1.241 — both
286
+ * emit the identical shapes, `/compact` and auto alike):
287
+ *
288
+ * `system`/`status` `{status:"compacting"}` → phase "start"
289
+ * `system`/`status` `{status:null, compact_result, → phase "settled"
290
+ * compact_error?}`
291
+ * `system`/`compact_boundary` `{compact_metadata: → phase "boundary"
292
+ * {trigger, pre_tokens, post_tokens,
293
+ * cumulative_dropped_tokens, duration_ms, …}}`
294
+ *
295
+ * Order on the wire: start → settled → (fresh init) → boundary → the
296
+ * continuation summary as a SYNTHETIC user message (never forwarded — it
297
+ * quotes conversation content verbatim). "boundary" only follows a
298
+ * successful settle and only on the turn the compaction ran (verified: it
299
+ * does NOT replay on later resumes). A compaction can span MINUTES of
300
+ * otherwise-silent stream — the whole point of forwarding it is that
301
+ * downstream can show the silence as work (the 2026-08-23 dead-air
302
+ * incident: 94% auto-compact read as a dead session). Additive kind:
303
+ * existing producers never emit it. */
304
+ export interface AgentMessageCompaction extends AgentMessageBase {
305
+ type: "compaction";
306
+ phase: "start" | "settled" | "boundary";
307
+ /** settled: how it ended. Absent on start/boundary (a boundary IS a
308
+ * success by construction — the harness only emits it after one). */
309
+ result?: "success" | "failed";
310
+ /** settled+failed: the harness's own reason, clamped. */
311
+ error?: string;
312
+ /** boundary: what initiated the compaction. */
313
+ trigger?: "auto" | "manual";
314
+ /** boundary: context tokens before / after, dropped total, wall time. */
315
+ preTokens?: number;
316
+ postTokens?: number;
317
+ droppedTokens?: number;
318
+ durationMs?: number;
319
+ }
320
+
321
+ /** A user-role message landing INSIDE a subagent's thread — the delivered
322
+ * form of a steer (the parent's `SendMessage` to a RUNNING child, queued
323
+ * "for delivery at its next tool round") or any other message the harness
324
+ * folds into a child's conversation mid-flight. Emitted ONLY with sidechain
325
+ * attribution: `parentToolUseId` (the spawning Agent/Task call's tool_use
326
+ * id) is REQUIRED — an unattributed user event is the parent's own prompt
327
+ * echo, which stays unmapped as before. Lets renderers show the steer as a
328
+ * user-role message inside the child's mini-session instead of leaving it
329
+ * an opaque SendMessage tool call on the parent only (task #97, owner
330
+ * directive 2026-08-27). Additive kind: existing producers never emit it. */
331
+ export interface AgentMessageSubagentUserMessage extends AgentMessageBase {
332
+ type: "subagent_user_message";
333
+ text: string;
334
+ parentToolUseId: string;
335
+ }
336
+
163
337
  export type AgentMessage =
164
338
  | AgentMessageInit
165
339
  | AgentMessageText
@@ -171,9 +345,13 @@ export type AgentMessage =
171
345
  | AgentMessageError
172
346
  | AgentMessageUsage
173
347
  | AgentMessageUsageDelta
348
+ | AgentMessagePlanLimits
174
349
  | AgentMessagePlan
175
350
  | AgentMessageTaskNotification
176
- | AgentMessageHarnessNotice;
351
+ | AgentMessageTaskProgress
352
+ | AgentMessageHarnessNotice
353
+ | AgentMessageCompaction
354
+ | AgentMessageSubagentUserMessage;
177
355
 
178
356
  /** Status block the agent emits to signal iteration completion or blockers. */
179
357
  export interface AgentStatus {
@@ -94,6 +94,16 @@ export interface RuntimeOptions {
94
94
  * Absent ⇒ nothing extra is sourced. Must be a plain absolute path — no
95
95
  * quotes, no `..`; the runtime validates and drops anything else. */
96
96
  credEnvFile?: string;
97
+ /** Durable launch report (boot-time turn adoption): called by the durable
98
+ * detached transport the moment its in-guest runner exists — with the
99
+ * guest prompt path (every durable file derives from it), the exit
100
+ * sentinel string, and the detached wrapper's pid. The caller stamps
101
+ * these on the turn row so a SUCCESSOR process (a deploy roll's new
102
+ * server) can re-attach to the runner's durable `.out` without this
103
+ * process's memory. Fired once per detached launch (a retry with a fresh
104
+ * runner fires again with the new paths); never on transports without
105
+ * durable files. Must not throw — the transport calls it inline. */
106
+ onDetachedLaunch?: (info: { promptPath: string; sentinel: string; pid: number }) => void;
97
107
  }
98
108
 
99
109
  /** Three-valued liveness verdict for a runtime's CURRENT turn, read from
@@ -182,6 +192,21 @@ export interface ModelExecutionContract {
182
192
  * fallback probe it already had; null is never a verdict.
183
193
  */
184
194
  probeTurnLiveness?(): Promise<RunnerLivenessVerdict | null>;
195
+ /**
196
+ * TELEMETRY, NEVER A VERDICT (tool-run pulse, 2026-08-29): the newest
197
+ * tool-run pulse the durable liveness probe carried — the guest
198
+ * heartbeat's sample of the runner's own session (aggregate CPU jiffies,
199
+ * written bytes, live process count) plus the guest clock it was read
200
+ * against. The executor's evidence ticker peeks it AFTER its liveness
201
+ * check and compares successive samples: counters ADVANCING is proof a
202
+ * long silent foreground tool is working, fanned to clients as a live
203
+ * `tool_pulse` frame. Never probes on its own; null before any probe or
204
+ * on a transport without the pulse file. No liveness decision may ever
205
+ * read it.
206
+ */
207
+ peekTurnPulse?(): {
208
+ atMs: number; cpuJiffies: number; ioBytes: number; procs: number; guestNowMs: number;
209
+ } | null;
185
210
  /**
186
211
  * DOORBELL, NEVER A VERDICT (exit-event push, v0.10.43): wake the current
187
212
  * turn's durable watchdog NOW so it runs its normal verification pass —
@@ -231,7 +256,51 @@ export interface ModelExecutionContract {
231
256
  * phase, a spec/transport without stream input, ACP path).
232
257
  * Calls are serialized per turn; never throws.
233
258
  */
234
- injectUserMessage?(text: string): Promise<"delivered" | "pending" | "closed" | "unsupported">;
259
+ injectUserMessage?(
260
+ text: string,
261
+ /** Who the message speaks for when it is not the session user
262
+ * (MidTurnEnvelopeOptions): a relayed person, named, or the thread
263
+ * agent that owns this worker. */
264
+ opts?: { relayedFrom?: string | null; fromOwnerAgent?: boolean },
265
+ ): Promise<"delivered" | "pending" | "closed" | "unsupported">;
266
+ /**
267
+ * Request an in-band STEP INTERRUPT of the currently running turn — the
268
+ * ESC equivalent. Where `injectUserMessage` queues content for the turn
269
+ * loop's next boundary, this rides the same durable inbox but carries a
270
+ * control line the CLI handles immediately, mid-step included: the
271
+ * running tool call aborts, the run ends within ~100ms, and the guest
272
+ * session stays resumable with the whole turn context (verified live
273
+ * against claude 2.1.236). Verdicts mirror `injectUserMessage`; only
274
+ * "delivered" means the CLI got the control line — callers escalate
275
+ * anything else (and "unsupported": no stream-input turn, or a runtime
276
+ * with no in-band interrupt, e.g. codex) to kill semantics, which stay
277
+ * honest because thread stores are durable and the successor turn
278
+ * resumes them. Never throws.
279
+ */
280
+ interruptTurn?(): Promise<"delivered" | "pending" | "closed" | "unsupported">;
281
+ /**
282
+ * Guest pid of the CURRENT turn's detached runner wrapper (the setsid
283
+ * process-group leader recorded at launch), or null when no detached
284
+ * durable-transport runner is live (boot phase, ACP path, single-exec
285
+ * transports). Advisory identity, NEVER a liveness verdict: the platform
286
+ * reads it to DECLARE harness-reported background work (an in-harness
287
+ * Workflow task) against the process tree that hosts it, so the park
288
+ * machinery can verify the tree from `/proc/<pid>` later. Runtimes
289
+ * without a detached guest simply omit the method.
290
+ */
291
+ currentRunnerPid?(): number | null;
292
+ /**
293
+ * Durable byte offset of the CURRENT turn's `.out` file just past the
294
+ * last line whose messages have ALL been yielded to the consumer — the
295
+ * safe harvest watermark for boot-time turn adoption. Null when no
296
+ * durable-transport turn is live, or before the first line completes.
297
+ * The contract is deliberately one line BEHIND the parse cursor: a
298
+ * caller that persists parts after each yielded message may stamp this
299
+ * offset at any time and a successor re-parses AT MOST the line whose
300
+ * parts were mid-persist (the same crash window the workflow tailer's
301
+ * flush-before-advance ordering accepts). Advisory, never a verdict.
302
+ */
303
+ currentTurnDurableOffset?(): number | null;
235
304
  sendMessage(opts: {
236
305
  prompt: string;
237
306
  sessionId?: string;
@@ -253,7 +322,7 @@ export interface ModelExecutionContract {
253
322
  * The sandbox provider (e.g. "vercel", "e2b") is an infrastructure concern
254
323
  * configured via SANDBOX_PROVIDER — not part of the runtime definition.
255
324
  * For non-sandbox agents (API calls, etc.) make the call directly in the workflow;
256
- * spawnAgent is a sandbox concept.
325
+ * `agent()` is a sandbox concept.
257
326
  */
258
327
  export interface AgentRuntime<S extends SandboxProvider = SandboxProvider> {
259
328
  create(sandbox: S, opts: RuntimeOptions): ModelExecutionContract;
@@ -79,8 +79,7 @@ export function defineSandboxEnvironment(
79
79
  snapshots: env.snapshots ?? { saveLatest: true },
80
80
  // Mark this as an environment build so the server skips the /factory mount
81
81
  // for its runs — an env build builds a platform image and never touches the
82
- // shared drive; baking a live Archil mount into its snapshot breaks the
83
- // re-mount of every workflow that later boots from it (#13).
82
+ // shared drive, so no live drive mount bakes into its snapshot (#13).
84
83
  environmentBuild: true,
85
84
  })
86
85
  .step({
@@ -234,10 +234,8 @@ export interface WorkflowMetadata {
234
234
  * BUILD — a setup-only workflow whose job is to leave its VM configured and
235
235
  * snapshot it (base-env / agent-env). Environment builds build a platform
236
236
  * IMAGE and never use the shared factory drive, so the server SKIPS mounting
237
- * /factory for them: a live Archil mount baked into the captured snapshot
238
- * fails the NEXT boot's re-mount ("an older Archil process is still running
239
- * for this mountpoint"), degrading /factory for every workflow booting from
240
- * that snapshot. Absent on ordinary workflows — which mount /factory exactly
237
+ * /factory for them and no live drive mount bakes into the captured
238
+ * snapshot. Absent on ordinary workflows — which mount /factory exactly
241
239
  * as before. Optional + additive: an ABSENT flag contributes nothing to the
242
240
  * canonical metadata hash (frozen-metadata rule), so existing workflows are
243
241
  * not forced to re-register. */
@@ -5,9 +5,7 @@
5
5
  * dispatch time. The CLI/bundler inspects the bundled module in the user's
6
6
  * environment and sends this compact plan as metadata.
7
7
  *
8
- * Every workflow has a step plan. Legacy `defineWorkflow({ run })`
9
- * workflows are wrapped at the SDK boundary as a single-step compiled
10
- * workflow (step name = "run"); the bundler sees the same shape regardless.
8
+ * Every workflow has a step plan.
11
9
  */
12
10
 
13
11
  /** One artifact a step promises to produce — mirrors `StepDeliverable`,
@@ -34,7 +34,8 @@
34
34
  */
35
35
 
36
36
  import { createHash } from "node:crypto";
37
- import { dirname } from "node:path";
37
+ import { realpathSync } from "node:fs";
38
+ import { dirname, join, resolve as resolvePath, sep } from "node:path";
38
39
  import { parse as babelParse } from "@babel/parser";
39
40
  import type { File, ExportDefaultDeclaration, CallExpression, Expression, Statement } from "@babel/types";
40
41
  import { importSourceModule } from "./source-loader.js";
@@ -194,38 +195,107 @@ interface BunPluginBuild {
194
195
  ): void;
195
196
  }
196
197
 
197
- /** Anchored regex over the alias table — the plugin filter. Specifiers are
198
- * literal package names, but escape anyway so a future entry with a `.`
199
- * or `+` can't widen the filter. */
198
+ /**
199
+ * The packages the PLATFORM provides to every workflow source — the SDK and
200
+ * its zod peer — which therefore resolve from the platform's own install
201
+ * roots when the source file's directory has no node_modules of its own.
202
+ * The live incident (2026-09-20, a claude-code session at the drive root):
203
+ * `agentc invoke --source /factory/files/wf.ts` died with `Could not
204
+ * resolve "@agent-compose/sdk", "zod"` because Bun walked up from
205
+ * /factory/files and found nothing, while the baked SDK sat in
206
+ * /workspace/node_modules the whole time (infra/e2b-template/template.ts —
207
+ * "SDK into /workspace/node_modules so any script written under /workspace
208
+ * can import it"). A workflow's OWN third-party deps still have to be
209
+ * installed beside the source; only these two are platform-resolved.
210
+ */
211
+ export const PLATFORM_RESOLVED_PACKAGES: readonly string[] = [SDK_PACKAGE, "zod"];
212
+
213
+ /** Env override for the fallback roots (colon-separated, tried first) — the
214
+ * test seam, and an ops knob for a sandbox image that plants the SDK
215
+ * elsewhere. */
216
+ export const SDK_FALLBACK_ROOTS_ENV = "AGENT_COMPOSE_SDK_FALLBACK_ROOTS";
217
+
218
+ /** Where the platform SDK lives when the source's own walk-up finds nothing:
219
+ * the env override's roots, the sandbox's baked /workspace install, then
220
+ * the bundling process's cwd (the CLI's own resolution context). Exported
221
+ * for the unit test. */
222
+ export function sdkFallbackRoots(env: Record<string, string | undefined> = process.env): string[] {
223
+ const fromEnv = (env[SDK_FALLBACK_ROOTS_ENV] ?? "")
224
+ .split(":")
225
+ .map((s) => s.trim())
226
+ .filter((s) => s.length > 0);
227
+ const roots = [...fromEnv, "/workspace", process.cwd()];
228
+ return roots.filter((r, i) => roots.indexOf(r) === i);
229
+ }
230
+
231
+ /** Anchored regex over the alias table PLUS the platform-resolved packages
232
+ * — the plugin filter. Specifiers are literal package names, but escape
233
+ * anyway so a future entry with a `.` or `+` can't widen the filter. */
200
234
  function aliasFilter(): RegExp {
201
- const alternation = SDK_SPECIFIER_ALIASES
235
+ const alternation = [...SDK_SPECIFIER_ALIASES, ...PLATFORM_RESOLVED_PACKAGES]
202
236
  .map((s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"))
203
237
  .join("|");
204
238
  return new RegExp(`^(?:${alternation})$`);
205
239
  }
206
240
 
207
241
  /**
208
- * The TOLERATE layer: a Bun resolve plugin that maps every known near-miss
209
- * specifier onto the real SDK. Resolution goes through Bun's own resolver
210
- * from the importing file's directory, so the alias lands on exactly the
242
+ * The TOLERATE layer: a Bun resolve plugin that (a) maps every known
243
+ * near-miss specifier onto the real SDK and (b) resolves the platform's own
244
+ * packages from the platform's install roots when the importing file's
245
+ * directory can't. Resolution goes through Bun's own resolver from the
246
+ * importing file's directory FIRST, so the alias lands on exactly the
211
247
  * `@agent-compose/sdk` the build would have used had the source spelled it
212
- * correctly. When the real SDK can't be resolved either, the plugin declines
213
- * and Bun's failure flows into `explainBundleFailure` below.
248
+ * correctly and a source that ships its own node_modules is untouched; only
249
+ * when that fails do the fallback roots (`sdkFallbackRoots`) get a turn, and
250
+ * `zod` additionally resolves from beside the SDK that was found (its
251
+ * peer, hoisted or nested). When nothing resolves, the plugin declines and
252
+ * Bun's failure flows into `explainBundleFailure` below.
214
253
  */
215
254
  function sdkAliasPlugin(bun: BunSurface): { name: string; setup(build: BunPluginBuild): void } {
255
+ // A resolution counts ONLY when it lands in a node_modules on `from`'s own
256
+ // walk-up (through symlinks — a workspace link's realpath is compared
257
+ // against the link's realpath). Bun's resolver otherwise "succeeds" via
258
+ // its global install cache (~/.bun/install/cache/<pkg>@<ver>/…), a bare
259
+ // package whose own dependencies are NOT installed — the build then dies
260
+ // on the SDK's deps ("Could not resolve @babel/parser, ai, e2b, …"), a
261
+ // far more confusing failure than the honest fallback below.
262
+ const tryResolve = (specifier: string, from: string): string | null => {
263
+ let resolved: string;
264
+ try { resolved = realpathSync(bun.resolveSync(specifier, from)); } catch { return null; }
265
+ let dir = resolvePath(from);
266
+ for (;;) {
267
+ try {
268
+ const real = realpathSync(join(dir, "node_modules", specifier));
269
+ if (resolved === real || resolved.startsWith(real + sep)) return resolved;
270
+ } catch { /* no node_modules/<specifier> at this level */ }
271
+ const parent = dirname(dir);
272
+ if (parent === dir) return null;
273
+ dir = parent;
274
+ }
275
+ };
216
276
  return {
217
277
  name: "agent-compose-sdk-alias",
218
278
  setup(build) {
219
279
  build.onResolve({ filter: aliasFilter() }, (args) => {
220
- const target = resolveSdkAlias(args.path);
280
+ const target = resolveSdkAlias(args.path)
281
+ ?? (PLATFORM_RESOLVED_PACKAGES.includes(args.path) ? args.path : null);
221
282
  if (!target) return undefined;
222
283
  // `importer` is the importing FILE, `resolveDir` already a directory.
223
284
  const from = args.importer ? dirname(args.importer) : args.resolveDir;
224
- try {
225
- return { path: bun.resolveSync(target, from && from.length > 0 ? from : process.cwd()) };
226
- } catch {
227
- return undefined;
285
+ const own = tryResolve(target, from && from.length > 0 ? from : process.cwd());
286
+ if (own) return { path: own };
287
+ for (const root of sdkFallbackRoots()) {
288
+ const hit = tryResolve(target, root);
289
+ if (hit) return { path: hit };
290
+ // zod is the SDK's peer: a root that holds the SDK resolves zod
291
+ // from the SDK's own directory even when it isn't hoisted.
292
+ if (target !== SDK_PACKAGE) {
293
+ const sdk = tryResolve(SDK_PACKAGE, root);
294
+ const beside = sdk ? tryResolve(target, dirname(sdk)) : null;
295
+ if (beside) return { path: beside };
296
+ }
228
297
  }
298
+ return undefined;
229
299
  });
230
300
  },
231
301
  };
@@ -557,10 +627,9 @@ export async function bundleWorkflow(
557
627
  // schemas → compact JSON-Schema-shaped descriptions carried in
558
628
  // template metadata. Used by the dashboard's "Registered" panel to
559
629
  // render schema tables and (eventually) client-side validation in
560
- // the playground. Run-form workflows stamp `z.unknown()` for both
561
- // — `extractIOSchema` filters those down to undefined so the
562
- // dashboard renders the "no declared schema" empty state instead
563
- // of a meaningless empty table.
630
+ // the playground. `extractIOSchema` filters a `z.unknown()` schema
631
+ // down to undefined so the dashboard renders the "no declared
632
+ // schema" empty state instead of a meaningless empty table.
564
633
  const inputSchema = extractIOSchema(workflow.input);
565
634
  const outputSchema = extractIOSchema(workflow.output);
566
635
 
@@ -78,9 +78,8 @@ const LIVE_EMIT_DRAIN_TIMEOUT_MS = 1_500;
78
78
  * this instance. After the step finishes, the engine calls `snapshot()`
79
79
  * to extract the bundle for transport.
80
80
  *
81
- * Metadata writes merge (later keys win) — matches the legacy
82
- * `mergeRunMetadata` semantics so authors don't see a behaviour change
83
- * across migrations.
81
+ * Metadata writes merge (later keys win), as the server's run-metadata
82
+ * merge does.
84
83
  */
85
84
  export class StepObservabilityCollector {
86
85
  private metadata: Record<string, unknown> = {};
@@ -6,7 +6,6 @@
6
6
  * 2. For each step:
7
7
  * a. Optionally check `getCachedOutput(stepIndex, step.name)` — if a
8
8
  * previous run completed this step, skip and reuse its output.
9
- * (Phase 1b uses this for crash recovery.)
10
9
  * b. Validate current input against `step.input`.
11
10
  * c. Call `step.run({ input, ... })`.
12
11
  * d. Validate return value against `step.output`.
@@ -17,9 +16,10 @@
17
16
  * Errors during any step bubble through `onStepFailed` and re-throw so the
18
17
  * caller can decide whether to mark the run failed.
19
18
  *
20
- * The cache + completion hooks are injection points — a sandbox engine
21
- * adapter (Phase 1c) wires them to `workflow_step_runs` rows. The default
22
- * is an in-process map for tests.
19
+ * The cache + completion hooks are injection points; without a cache every
20
+ * step runs. The platform's sandbox path does not walk the chain here: the
21
+ * Temporal `executeStep` activity runs one step per runner subprocess via
22
+ * `runWorkflowSingleStep` below.
23
23
  */
24
24
 
25
25
  import { z } from "zod";
@@ -76,13 +76,10 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
76
76
  /**
77
77
  * Crash-recovery hook. Called before a step executes. Return the cached
78
78
  * output to skip execution; return undefined to run the step.
79
- *
80
- * Phase 1b implementations will look up `workflow_step_runs` rows for
81
- * (runId, stepIndex, stepName) and return completed step outputs here.
82
79
  * Default: always undefined (no caching).
83
80
  */
84
81
  getCachedOutput?(stepIndex: number, stepName: string): unknown | undefined | Promise<unknown | undefined>;
85
- /** Fire after a step's `execute` and output validation succeed. */
82
+ /** Fire after a step's `run` and output validation succeed. */
86
83
  onStepCompleted?(stepIndex: number, stepName: string, output: unknown, durationMs: number): void | Promise<void>;
87
84
  /** Fire when a step throws or fails validation. The error is re-thrown after this returns. */
88
85
  onStepFailed?(stepIndex: number, stepName: string, error: Error, durationMs: number): void | Promise<void>;
@@ -6,9 +6,8 @@
6
6
  *
7
7
  * Why: durable suspend/resume requires step boundaries to be addressable as
8
8
  * data, not opaque async-function bodies. Each step's input + output is
9
- * serialisable JSON so engine adapters (in-process, sandbox, future
10
- * Inngest/Temporal) can record completion and replay from the last
11
- * completed step.
9
+ * serialisable JSON so the durable engine (Temporal, one activity per step)
10
+ * can record completion and replay from the last completed step.
12
11
  */
13
12
 
14
13
  import type { z } from "zod";
@@ -18,7 +17,7 @@ import type { WorkflowMetadata } from "../types/workflow-metadata.js";
18
17
  import type { StepObservability } from "./observability.js";
19
18
 
20
19
  /**
21
- * Per-step execution context. Threaded into every step's `execute(...)` so
20
+ * Per-step execution context. Threaded into every step's `run(...)` so
22
21
  * the step can read tenant identity and run identity, log progress, and
23
22
  * invoke sandbox commands.
24
23
  *
@@ -33,9 +32,9 @@ export interface StepContext<TInput = unknown> extends BaseExecutionContext {
33
32
  abortSignal: AbortSignal;
34
33
  /**
35
34
  * Merge key-value metadata onto the run record. Buffered during the
36
- * step and flushed when the step completes; the durable engine writes
37
- * it via the same `mergeRunMetadata` path the legacy runner used, so
38
- * the dashboard sees the same shape. Later keys win.
35
+ * step and flushed when the step completes; the durable engine merges
36
+ * it into the run's metadata (`persistStepObservability`). Later keys
37
+ * win.
39
38
  */
40
39
  setMetadata(data: Record<string, unknown>): Promise<void>;
41
40
  /**
@@ -98,9 +97,8 @@ export interface Step<TInput, TOutput> {
98
97
  }
99
98
 
100
99
  /**
101
- * The result of running one step. Engine adapters persist these into the
102
- * `workflow_step_runs` table (Phase 1b) so subsequent runs can skip
103
- * completed steps. `observability` carries the snapshot of
100
+ * The result of running one step, as `runWorkflowSteps` reports it per
101
+ * step. `observability` carries the snapshot of
104
102
  * `ctx.setMetadata` / `ctx.step` / `ctx.agentEvents` recorded during
105
103
  * the step; undefined when no hooks were used.
106
104
  */
@@ -16,7 +16,8 @@
16
16
  * matches the final step's `output` schema.
17
17
  *
18
18
  * Workflows-as-data — the result is consumable by any engine adapter
19
- * (in-process today; sandbox + Inngest/Temporal in future PRs).
19
+ * (in-process `runWorkflow`; the sandbox runner one step at a time under
20
+ * Temporal).
20
21
  */
21
22
 
22
23
  import type { z } from "zod";
@@ -6,7 +6,7 @@
6
6
  * runner.
7
7
  *
8
8
  * Errors classified into `WorkflowError` (user code threw) vs `EngineError`
9
- * (platform problem) for the runner harness to surface upstream.
9
+ * (platform problem).
10
10
  */
11
11
 
12
12
  import type { WorkflowHooks } from "../types/workflow.js";
@@ -53,7 +53,7 @@ export class EngineError extends Error {
53
53
  }
54
54
  }
55
55
 
56
- /** Classify any thrown value into the wire-level `kind` expected by `/fail`. */
56
+ /** Classify any thrown value as a `workflow` or an `engine` failure. */
57
57
  export function classifyError(err: unknown): "workflow" | "engine" {
58
58
  if (err instanceof WorkflowError) return "workflow";
59
59
  if (err instanceof EngineError) return "engine";
@@ -92,9 +92,7 @@ export interface RunWorkflowOptions {
92
92
  onStepStarted?: RunWorkflowStepsOpts<unknown, unknown>["onStepStarted"];
93
93
  onStepCompleted?: RunWorkflowStepsOpts<unknown, unknown>["onStepCompleted"];
94
94
  onStepFailed?: RunWorkflowStepsOpts<unknown, unknown>["onStepFailed"];
95
- /** Provider-specific child workflow invocation. Temporal/Inngest providers
96
- * inject their native child-workflow primitive; the LocalProvider injects
97
- * the public Agent Compose API client. */
95
+ /** Child workflow invocation. Unset, a step's `invokeChild` throws. */
98
96
  invokeChild?: InvokeChild;
99
97
  }
100
98