@agent-compose/sdk 0.8.5 → 0.8.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +3 -3
  3. package/dist/agent/agent-loop.d.ts +6 -5
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +119 -54
  7. package/dist/directives.d.ts +3 -3
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +12 -12
  12. package/dist/index.js +771 -204
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +185 -68
  15. package/dist/runtimes/_reported-model.d.ts +16 -0
  16. package/dist/runtimes/claude-code.d.ts +60 -1
  17. package/dist/runtimes/claude.d.ts +1 -1
  18. package/dist/runtimes/codex.d.ts +94 -6
  19. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  20. package/dist/runtimes/model-report.test.d.ts +14 -0
  21. package/dist/runtimes/openai-desktop.js +741 -200
  22. package/dist/runtimes/opencode.d.ts +48 -11
  23. package/dist/runtimes/opencode.test.d.ts +14 -0
  24. package/dist/sandbox/baked-clis.d.ts +75 -0
  25. package/dist/sandbox/exec-stream.d.ts +1 -2
  26. package/dist/sandbox/network-policy.d.ts +23 -5
  27. package/dist/sandbox.d.ts +4 -2
  28. package/dist/step-invocation/protocol.d.ts +3 -4
  29. package/dist/step-invocation/server.d.ts +2 -2
  30. package/dist/step-invocation/types.d.ts +1 -1
  31. package/dist/types/api-conversations.d.ts +442 -29
  32. package/dist/types/api-factory.d.ts +99 -10
  33. package/dist/types/api-projects.d.ts +521 -0
  34. package/dist/types/api-runs.d.ts +83 -0
  35. package/dist/types/api-scopes.d.ts +32 -3
  36. package/dist/types/conversation-stream.d.ts +5 -0
  37. package/dist/types/execution-context.d.ts +1 -1
  38. package/dist/types/protocol.d.ts +86 -2
  39. package/dist/types/runtime.d.ts +9 -2
  40. package/dist/types/workflow-metadata.d.ts +2 -4
  41. package/dist/types/workflow-plan.d.ts +1 -3
  42. package/dist/utils/bundler.d.ts +23 -0
  43. package/dist/workflow-steps/observability.d.ts +2 -3
  44. package/dist/workflow-steps/runner.d.ts +5 -8
  45. package/dist/workflow-steps/types.d.ts +8 -10
  46. package/dist/workflow-steps/workflow.d.ts +2 -1
  47. package/dist/workflows/engine.d.ts +3 -5
  48. package/dist/workflows/invoke-child.d.ts +2 -2
  49. package/package.json +2 -2
  50. package/src/agent/agent-context.ts +168 -125
  51. package/src/agent/agent-loop.ts +7 -6
  52. package/src/agent/perf-sampler.ts +54 -3
  53. package/src/agent/run-agent.ts +1 -1
  54. package/src/client.ts +226 -71
  55. package/src/directives.ts +3 -3
  56. package/src/display.ts +12 -0
  57. package/src/errors.ts +1 -0
  58. package/src/generated/agentc-commands.ts +571 -0
  59. package/src/index.ts +57 -21
  60. package/src/pause/pause-core.ts +2 -1
  61. package/src/request-context/request-context.ts +1 -1
  62. package/src/runtimes/_cli-agent.ts +318 -122
  63. package/src/runtimes/_reported-model.ts +24 -0
  64. package/src/runtimes/claude-code.ts +195 -12
  65. package/src/runtimes/claude.ts +9 -2
  66. package/src/runtimes/codex.ts +188 -19
  67. package/src/runtimes/opencode.ts +195 -26
  68. package/src/sandbox/baked-clis.ts +86 -0
  69. package/src/sandbox/exec-stream.ts +1 -2
  70. package/src/sandbox/network-policy.ts +51 -7
  71. package/src/sandbox/providers/e2b.ts +3 -3
  72. package/src/sandbox/providers/vercel.ts +6 -6
  73. package/src/sandbox.ts +8 -2
  74. package/src/step-invocation/invoker.ts +2 -6
  75. package/src/step-invocation/protocol.ts +3 -4
  76. package/src/step-invocation/server.ts +2 -2
  77. package/src/types/api-conversations.ts +366 -23
  78. package/src/types/api-factory.ts +95 -10
  79. package/src/types/api-projects.ts +477 -0
  80. package/src/types/api-runs.ts +73 -0
  81. package/src/types/api-scopes.ts +32 -3
  82. package/src/types/conversation-stream.ts +5 -0
  83. package/src/types/execution-context.ts +1 -1
  84. package/src/types/protocol.ts +91 -2
  85. package/src/types/runtime.ts +8 -2
  86. package/src/types/sandbox-environment.ts +1 -2
  87. package/src/types/workflow-metadata.ts +2 -4
  88. package/src/types/workflow-plan.ts +1 -3
  89. package/src/utils/bundler.ts +88 -19
  90. package/src/workflow-steps/observability.ts +2 -3
  91. package/src/workflow-steps/runner.ts +5 -8
  92. package/src/workflow-steps/types.ts +8 -10
  93. package/src/workflow-steps/workflow.ts +2 -1
  94. package/src/workflows/engine.ts +3 -5
  95. package/src/workflows/invoke-child.ts +2 -2
  96. package/dist/generated/verb-synopsis.d.ts +0 -34
  97. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  98. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  99. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
  100. package/src/generated/verb-synopsis.ts +0 -544
@@ -97,6 +97,11 @@ export interface InvokeInlineAndWaitOptions extends InvokeInlineOptions {
97
97
 
98
98
  export interface InvokeResult {
99
99
  id: string;
100
+ /** The run is PARKED for its funding plan's reset: the person whose plan
101
+ * pays for it has every plan out, so nothing runs until `until`, when it
102
+ * starts by itself; `text` is the one line to pass on. Absent for a run
103
+ * that boots now. */
104
+ hold?: { until: string; text: string };
100
105
  }
101
106
 
102
107
  export interface StreamRunLogsOptions {
@@ -334,6 +339,74 @@ export interface ListRunsOptions {
334
339
  limit?: number;
335
340
  }
336
341
 
342
+ /** What started a run (`GET /factories/:slug/workflow-activity*`), read off
343
+ * the dispatch stamps: a schedule's tick, a thread's scripted check, the
344
+ * chat an agent's turn dispatched it from, the person who invoked it from
345
+ * the dashboard, or the API key (the CLI, a script). */
346
+ export type RunStartedBy =
347
+ | { kind: "schedule"; name: string | null }
348
+ | { kind: "check" }
349
+ | { kind: "chat"; conversationId: string; title: string | null }
350
+ | { kind: "person"; userId: string; name: string; image: string | null; avatarUrl: string | null }
351
+ | { kind: "key"; name: string }
352
+ | { kind: "unknown" };
353
+
354
+ /** One run in a space's workflow activity: the runs list's summary plus
355
+ * what started it. */
356
+ export interface WorkflowActivityRun extends RunListEntry {
357
+ finalSummary: string | null;
358
+ /** Structured return value from the workflow, when it completed. */
359
+ output: unknown;
360
+ startedBy: RunStartedBy;
361
+ }
362
+
363
+ /** One stack in a space's workflow activity: a workflow that has run there
364
+ * and its latest run. */
365
+ export interface WorkflowActivityHead {
366
+ workflow: string;
367
+ latestRun: WorkflowActivityRun;
368
+ }
369
+
370
+ export interface ListWorkflowActivityOptions {
371
+ /** Factory the runs live in. Defaults to "default". */
372
+ factorySlug?: string;
373
+ /** Stacks per page (1–50, server default 20). */
374
+ limit?: number;
375
+ /** The previous page's `nextCursor`. */
376
+ cursor?: string;
377
+ /** A search (at most 100 characters): every run whose workflow name,
378
+ * title, failure or summary contains it, without case — reaching past
379
+ * the last 90 days the listing shows without one. */
380
+ q?: string;
381
+ }
382
+
383
+ export interface ListWorkflowRunsOptions {
384
+ /** Factory the runs live in. Defaults to "default". */
385
+ factorySlug?: string;
386
+ /** The registered workflow name, exactly. */
387
+ workflow: string;
388
+ /** Runs per page (1–50, server default 10). */
389
+ limit?: number;
390
+ /** The previous page's `nextCursor`. */
391
+ cursor?: string;
392
+ /** A search (at most 100 characters): the workflow's runs whose title,
393
+ * failure or summary contains it, without case — reaching past the last
394
+ * 90 days the listing shows without one. */
395
+ q?: string;
396
+ }
397
+
398
+ export interface WorkflowActivityPage {
399
+ workflows: WorkflowActivityHead[];
400
+ hasMore: boolean;
401
+ nextCursor: string | null;
402
+ }
403
+
404
+ export interface WorkflowRunsPage {
405
+ runs: WorkflowActivityRun[];
406
+ hasMore: boolean;
407
+ nextCursor: string | null;
408
+ }
409
+
337
410
  export interface TimelineEvent {
338
411
  kind: string;
339
412
  at: string;
@@ -36,9 +36,10 @@ export type DocumentCapability = "read" | "write";
36
36
  export type TemplateCapability = "read" | "write" | "invoke" | "see_runs";
37
37
 
38
38
  /** One grant on an artifact scope. `team`/`user` are the editable tiers
39
- * carried on a PUT (full-replace); `session`/`project` are DERIVED,
40
- * read-only arms that appear only on READ payloads (the routes that manage
41
- * them own their mutation — a scope PUT rejects them, 400 `invalid_principal`).
39
+ * carried on a PUT (full-replace); `session`/`project`/`conversation`/
40
+ * `worker` are DERIVED, read-only arms that appear only on READ payloads
41
+ * (the routes that manage them own their mutation — a scope PUT rejects
42
+ * them, 400 `invalid_principal`).
42
43
  *
43
44
  * Session and project grants carry FIXED capabilities `['read','write']`:
44
45
  * the fs-gateway is concealment-only (no read/write dimension at the mount),
@@ -58,6 +59,30 @@ export type ScopeGrant =
58
59
  * being a project member (share implies unshare). Null only for
59
60
  * transitional/legacy rows. */
60
61
  projectObjectId: string | null;
62
+ /** The people in the project: the row's reach. */
63
+ memberCount: number;
64
+ capabilities: string[];
65
+ }
66
+ /** A chat's grant (scope-at-birth, documents): reach derives LIVE from the
67
+ * chat's tier and member rows. `memberCount` is null for a public channel,
68
+ * whose audience is the whole team rather than its member rows. */
69
+ | {
70
+ principal: "conversation";
71
+ conversationId: string;
72
+ conversationTitle: string | null;
73
+ memberCount: number | null;
74
+ capabilities: string[];
75
+ }
76
+ /** The grant a chat's WORKER session holds on what it wrote: not a group
77
+ * of people. The readers of the chat it works for (`chatId`) reach the
78
+ * file through it while the worker's project boundary holds. `title` is
79
+ * the work's (the session's) title. */
80
+ | {
81
+ principal: "worker";
82
+ conversationId: string;
83
+ title: string | null;
84
+ chatId: string;
85
+ chatTitle: string | null;
61
86
  capabilities: string[];
62
87
  };
63
88
 
@@ -71,6 +96,10 @@ export interface ArtifactScope {
71
96
  /** Templates only — the owning conversation binding (a grant source:
72
97
  * members of that conversation reach the template via their role). */
73
98
  conversationId?: string | null;
99
+ /** Documents only — born of a shared chat's work (Ivy's publish, a
100
+ * worker's output for a chat). Such a document is shared by the people
101
+ * who can edit it, whoever its owner is; so is one with no owner. */
102
+ bornOfChatWork?: boolean;
74
103
  grants: ScopeGrant[];
75
104
  }
76
105
 
@@ -130,6 +130,11 @@ export interface ConversationTurnStateEvent {
130
130
  pendingCount: number;
131
131
  at: number;
132
132
  partial: true;
133
+ /** The running turn is WRITING words that will be sent — the typing
134
+ * indicator's one signal (a running turn that thinks, calls tools,
135
+ * reacts or notes is not writing). Absent when the emitter cannot know;
136
+ * consumers inherit their previous frame's value, and `idle` resets it. */
137
+ writing?: boolean;
133
138
  /** When the open turn started (ms) — carried by the connect-time
134
139
  * snapshot frame only; absent on live transition frames. */
135
140
  startedAt?: number | null;
@@ -1,4 +1,4 @@
1
- /** Shared execution context capabilities for workflow functions and steps. */
1
+ /** Shared execution context capabilities for workflow steps. */
2
2
 
3
3
  import type { InvokeAndWaitOptions, RunStatus } from "./api-runs.js";
4
4
  import type { RequestContext } from "../request-context/request-context.js";
@@ -74,6 +74,22 @@ export interface AgentMessageError extends AgentMessageBase {
74
74
  text: string;
75
75
  }
76
76
 
77
+ /** One model's share of a harness's turn-end report (claude-code
78
+ * `result.modelUsage[<model>]`): the same four token classes as the turn
79
+ * totals, plus what the harness reports beside them. `costUsd` is the
80
+ * harness's OWN estimate at its price table — never a bill. Fields the
81
+ * harness did not report are absent, never zeroed. */
82
+ export interface AgentMessageModelUsage {
83
+ inputTokens: number;
84
+ outputTokens: number;
85
+ cacheReadTokens: number;
86
+ cacheCreationTokens: number;
87
+ /** Thinking tokens, already counted inside `outputTokens`. */
88
+ thinkingTokens?: number;
89
+ webSearchRequests?: number;
90
+ costUsd?: number;
91
+ }
92
+
77
93
  export interface AgentMessageUsage extends AgentMessageBase {
78
94
  type: "usage";
79
95
  inputTokens: number;
@@ -82,6 +98,54 @@ export interface AgentMessageUsage extends AgentMessageBase {
82
98
  cacheCreationTokens: number;
83
99
  durationMs: number;
84
100
  numTurns: number;
101
+ /** Reasoning tokens, already counted inside `outputTokens` (codex
102
+ * `turn.completed.usage.reasoning_output_tokens`). Absent when the
103
+ * harness reports no such class. */
104
+ reasoningOutputTokens?: number;
105
+ /** Per-model totals the harness reported beside the turn totals
106
+ * (claude-code `result.modelUsage`): every model the query pipeline
107
+ * called — main loop, subagents, compaction. As the CLI reports them:
108
+ * CUMULATIVE for the guest session (a streaming-input or resumed
109
+ * session carries its earlier turns), so a per-turn share is the
110
+ * difference from the previous report of the same session. Absent when
111
+ * the harness reports none. */
112
+ byModel?: Record<string, AgentMessageModelUsage>;
113
+ /** The harness's own cost estimate for the same scope as `byModel`
114
+ * (claude-code `result.total_cost_usd`): list-price arithmetic, an
115
+ * estimate and never a billing statement. */
116
+ costUsd?: number;
117
+ }
118
+
119
+ /** The harness's reading of the account's PLAN LIMITS (claude-code
120
+ * `rate_limit_event`, emitted whenever its rate-limit information changes
121
+ * — subscription-funded sessions only; the headers it reads exist for
122
+ * claude.ai plans). `status` is the verdict for the request just made:
123
+ * `rejected` means the plan's wall, with `resetsAt` the authoritative
124
+ * reset. `window` names which window the verdict speaks for (the CLI's
125
+ * `rateLimitType`: five_hour, seven_day, seven_day_opus, …). `utilization`
126
+ * is carried only once a window crosses a warning threshold (the CLI omits
127
+ * it while plainly allowed), as the CLI reports it — a 0..1 fraction of
128
+ * the window. Everything the event did not carry is absent; nothing is
129
+ * invented. Additive kind: existing producers never emit it. */
130
+ export interface AgentMessagePlanLimits extends AgentMessageBase {
131
+ type: "plan_limits";
132
+ status: "allowed" | "allowed_warning" | "rejected";
133
+ window?: string;
134
+ /** ISO time the named window resets. */
135
+ resetsAt?: string;
136
+ utilization?: number;
137
+ /** The warning threshold the window crossed (as the CLI reports it). */
138
+ surpassedThreshold?: number;
139
+ /** The plan's extra-usage (overage) lane, when the event spoke of it. */
140
+ overage?: {
141
+ status?: "allowed" | "allowed_warning" | "rejected";
142
+ resetsAt?: string;
143
+ disabledReason?: string;
144
+ inUse?: boolean;
145
+ };
146
+ /** Which spend limit blocked the request when not the member's own cap. */
147
+ limitScope?: string;
148
+ errorCode?: string;
85
149
  }
86
150
 
87
151
  /** LIVE-ONLY incremental usage off the harness's raw provider stream — the
@@ -114,7 +178,8 @@ export interface AgentMessagePlan extends AgentMessageBase {
114
178
  type: "plan";
115
179
  entries: {
116
180
  content: string;
117
- priority: "high" | "medium" | "low";
181
+ /** ACP names one; codex's `--json` plan names none. */
182
+ priority?: "high" | "medium" | "low";
118
183
  status: "pending" | "in_progress" | "completed";
119
184
  }[];
120
185
  }
@@ -269,6 +334,28 @@ export interface AgentMessageSubagentUserMessage extends AgentMessageBase {
269
334
  parentToolUseId: string;
270
335
  }
271
336
 
337
+ /** The harness's OWN report of the model it is running — never the
338
+ * platform's configuration (owner 2026-10-06: "It's very difficult to
339
+ * tell which model did this work"). claude-code names it twice: once on
340
+ * `system`/`init` (`model`, the session's resolved model) and on every
341
+ * top-level `assistant` event (`message.model`, the id the API answered
342
+ * with — the truth after a `/model` switch or a fallback; `<synthetic>`
343
+ * is the harness's own voice and names none; a subagent's events name the
344
+ * subagent's model, not the worker's, and are left out). The normaliser
345
+ * emits one report per naming, so a consumer that stamps the turn's
346
+ * models dedupes in order of first appearance and keeps every distinct
347
+ * id (a mid-turn switch records both). codex's `exec --json` events and
348
+ * opencode's `run --format json` lines carry no model at all (verified
349
+ * against both sources, 2026-10-06): those harnesses emit none, and the
350
+ * record stays honestly empty. Additive kind: existing producers never
351
+ * emit it. */
352
+ export interface AgentMessageModelReport extends AgentMessageBase {
353
+ type: "model_report";
354
+ /** The model id as the harness reports it, minus claude-code's
355
+ * `[1m]`-style context-window marker (runtimes/_reported-model.ts). */
356
+ model: string;
357
+ }
358
+
272
359
  export type AgentMessage =
273
360
  | AgentMessageInit
274
361
  | AgentMessageText
@@ -280,12 +367,14 @@ export type AgentMessage =
280
367
  | AgentMessageError
281
368
  | AgentMessageUsage
282
369
  | AgentMessageUsageDelta
370
+ | AgentMessagePlanLimits
283
371
  | AgentMessagePlan
284
372
  | AgentMessageTaskNotification
285
373
  | AgentMessageTaskProgress
286
374
  | AgentMessageHarnessNotice
287
375
  | AgentMessageCompaction
288
- | AgentMessageSubagentUserMessage;
376
+ | AgentMessageSubagentUserMessage
377
+ | AgentMessageModelReport;
289
378
 
290
379
  /** Status block the agent emits to signal iteration completion or blockers. */
291
380
  export interface AgentStatus {
@@ -256,7 +256,13 @@ export interface ModelExecutionContract {
256
256
  * phase, a spec/transport without stream input, ACP path).
257
257
  * Calls are serialized per turn; never throws.
258
258
  */
259
- injectUserMessage?(text: string): Promise<"delivered" | "pending" | "closed" | "unsupported">;
259
+ injectUserMessage?(
260
+ text: string,
261
+ /** Who the message speaks for when it is not the session user
262
+ * (MidTurnEnvelopeOptions): a relayed person, named, or the thread
263
+ * agent that owns this worker. */
264
+ opts?: { relayedFrom?: string | null; fromOwnerAgent?: boolean },
265
+ ): Promise<"delivered" | "pending" | "closed" | "unsupported">;
260
266
  /**
261
267
  * Request an in-band STEP INTERRUPT of the currently running turn — the
262
268
  * ESC equivalent. Where `injectUserMessage` queues content for the turn
@@ -316,7 +322,7 @@ export interface ModelExecutionContract {
316
322
  * The sandbox provider (e.g. "vercel", "e2b") is an infrastructure concern
317
323
  * configured via SANDBOX_PROVIDER — not part of the runtime definition.
318
324
  * For non-sandbox agents (API calls, etc.) make the call directly in the workflow;
319
- * spawnAgent is a sandbox concept.
325
+ * `agent()` is a sandbox concept.
320
326
  */
321
327
  export interface AgentRuntime<S extends SandboxProvider = SandboxProvider> {
322
328
  create(sandbox: S, opts: RuntimeOptions): ModelExecutionContract;
@@ -79,8 +79,7 @@ export function defineSandboxEnvironment(
79
79
  snapshots: env.snapshots ?? { saveLatest: true },
80
80
  // Mark this as an environment build so the server skips the /factory mount
81
81
  // for its runs — an env build builds a platform image and never touches the
82
- // shared drive; baking a live Archil mount into its snapshot breaks the
83
- // re-mount of every workflow that later boots from it (#13).
82
+ // shared drive, so no live drive mount bakes into its snapshot (#13).
84
83
  environmentBuild: true,
85
84
  })
86
85
  .step({
@@ -234,10 +234,8 @@ export interface WorkflowMetadata {
234
234
  * BUILD — a setup-only workflow whose job is to leave its VM configured and
235
235
  * snapshot it (base-env / agent-env). Environment builds build a platform
236
236
  * IMAGE and never use the shared factory drive, so the server SKIPS mounting
237
- * /factory for them: a live Archil mount baked into the captured snapshot
238
- * fails the NEXT boot's re-mount ("an older Archil process is still running
239
- * for this mountpoint"), degrading /factory for every workflow booting from
240
- * that snapshot. Absent on ordinary workflows — which mount /factory exactly
237
+ * /factory for them and no live drive mount bakes into the captured
238
+ * snapshot. Absent on ordinary workflows — which mount /factory exactly
241
239
  * as before. Optional + additive: an ABSENT flag contributes nothing to the
242
240
  * canonical metadata hash (frozen-metadata rule), so existing workflows are
243
241
  * not forced to re-register. */
@@ -5,9 +5,7 @@
5
5
  * dispatch time. The CLI/bundler inspects the bundled module in the user's
6
6
  * environment and sends this compact plan as metadata.
7
7
  *
8
- * Every workflow has a step plan. Legacy `defineWorkflow({ run })`
9
- * workflows are wrapped at the SDK boundary as a single-step compiled
10
- * workflow (step name = "run"); the bundler sees the same shape regardless.
8
+ * Every workflow has a step plan.
11
9
  */
12
10
 
13
11
  /** One artifact a step promises to produce — mirrors `StepDeliverable`,
@@ -34,7 +34,8 @@
34
34
  */
35
35
 
36
36
  import { createHash } from "node:crypto";
37
- import { dirname } from "node:path";
37
+ import { realpathSync } from "node:fs";
38
+ import { dirname, join, resolve as resolvePath, sep } from "node:path";
38
39
  import { parse as babelParse } from "@babel/parser";
39
40
  import type { File, ExportDefaultDeclaration, CallExpression, Expression, Statement } from "@babel/types";
40
41
  import { importSourceModule } from "./source-loader.js";
@@ -194,38 +195,107 @@ interface BunPluginBuild {
194
195
  ): void;
195
196
  }
196
197
 
197
- /** Anchored regex over the alias table — the plugin filter. Specifiers are
198
- * literal package names, but escape anyway so a future entry with a `.`
199
- * or `+` can't widen the filter. */
198
+ /**
199
+ * The packages the PLATFORM provides to every workflow source — the SDK and
200
+ * its zod peer — which therefore resolve from the platform's own install
201
+ * roots when the source file's directory has no node_modules of its own.
202
+ * The live incident (2026-09-20, a claude-code session at the drive root):
203
+ * `agentc invoke --source /factory/files/wf.ts` died with `Could not
204
+ * resolve "@agent-compose/sdk", "zod"` because Bun walked up from
205
+ * /factory/files and found nothing, while the baked SDK sat in
206
+ * /workspace/node_modules the whole time (infra/e2b-template/template.ts —
207
+ * "SDK into /workspace/node_modules so any script written under /workspace
208
+ * can import it"). A workflow's OWN third-party deps still have to be
209
+ * installed beside the source; only these two are platform-resolved.
210
+ */
211
+ export const PLATFORM_RESOLVED_PACKAGES: readonly string[] = [SDK_PACKAGE, "zod"];
212
+
213
+ /** Env override for the fallback roots (colon-separated, tried first) — the
214
+ * test seam, and an ops knob for a sandbox image that plants the SDK
215
+ * elsewhere. */
216
+ export const SDK_FALLBACK_ROOTS_ENV = "AGENT_COMPOSE_SDK_FALLBACK_ROOTS";
217
+
218
+ /** Where the platform SDK lives when the source's own walk-up finds nothing:
219
+ * the env override's roots, the sandbox's baked /workspace install, then
220
+ * the bundling process's cwd (the CLI's own resolution context). Exported
221
+ * for the unit test. */
222
+ export function sdkFallbackRoots(env: Record<string, string | undefined> = process.env): string[] {
223
+ const fromEnv = (env[SDK_FALLBACK_ROOTS_ENV] ?? "")
224
+ .split(":")
225
+ .map((s) => s.trim())
226
+ .filter((s) => s.length > 0);
227
+ const roots = [...fromEnv, "/workspace", process.cwd()];
228
+ return roots.filter((r, i) => roots.indexOf(r) === i);
229
+ }
230
+
231
+ /** Anchored regex over the alias table PLUS the platform-resolved packages
232
+ * — the plugin filter. Specifiers are literal package names, but escape
233
+ * anyway so a future entry with a `.` or `+` can't widen the filter. */
200
234
  function aliasFilter(): RegExp {
201
- const alternation = SDK_SPECIFIER_ALIASES
235
+ const alternation = [...SDK_SPECIFIER_ALIASES, ...PLATFORM_RESOLVED_PACKAGES]
202
236
  .map((s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"))
203
237
  .join("|");
204
238
  return new RegExp(`^(?:${alternation})$`);
205
239
  }
206
240
 
207
241
  /**
208
- * The TOLERATE layer: a Bun resolve plugin that maps every known near-miss
209
- * specifier onto the real SDK. Resolution goes through Bun's own resolver
210
- * from the importing file's directory, so the alias lands on exactly the
242
+ * The TOLERATE layer: a Bun resolve plugin that (a) maps every known
243
+ * near-miss specifier onto the real SDK and (b) resolves the platform's own
244
+ * packages from the platform's install roots when the importing file's
245
+ * directory can't. Resolution goes through Bun's own resolver from the
246
+ * importing file's directory FIRST, so the alias lands on exactly the
211
247
  * `@agent-compose/sdk` the build would have used had the source spelled it
212
- * correctly. When the real SDK can't be resolved either, the plugin declines
213
- * and Bun's failure flows into `explainBundleFailure` below.
248
+ * correctly and a source that ships its own node_modules is untouched; only
249
+ * when that fails do the fallback roots (`sdkFallbackRoots`) get a turn, and
250
+ * `zod` additionally resolves from beside the SDK that was found (its
251
+ * peer, hoisted or nested). When nothing resolves, the plugin declines and
252
+ * Bun's failure flows into `explainBundleFailure` below.
214
253
  */
215
254
  function sdkAliasPlugin(bun: BunSurface): { name: string; setup(build: BunPluginBuild): void } {
255
+ // A resolution counts ONLY when it lands in a node_modules on `from`'s own
256
+ // walk-up (through symlinks — a workspace link's realpath is compared
257
+ // against the link's realpath). Bun's resolver otherwise "succeeds" via
258
+ // its global install cache (~/.bun/install/cache/<pkg>@<ver>/…), a bare
259
+ // package whose own dependencies are NOT installed — the build then dies
260
+ // on the SDK's deps ("Could not resolve @babel/parser, ai, e2b, …"), a
261
+ // far more confusing failure than the honest fallback below.
262
+ const tryResolve = (specifier: string, from: string): string | null => {
263
+ let resolved: string;
264
+ try { resolved = realpathSync(bun.resolveSync(specifier, from)); } catch { return null; }
265
+ let dir = resolvePath(from);
266
+ for (;;) {
267
+ try {
268
+ const real = realpathSync(join(dir, "node_modules", specifier));
269
+ if (resolved === real || resolved.startsWith(real + sep)) return resolved;
270
+ } catch { /* no node_modules/<specifier> at this level */ }
271
+ const parent = dirname(dir);
272
+ if (parent === dir) return null;
273
+ dir = parent;
274
+ }
275
+ };
216
276
  return {
217
277
  name: "agent-compose-sdk-alias",
218
278
  setup(build) {
219
279
  build.onResolve({ filter: aliasFilter() }, (args) => {
220
- const target = resolveSdkAlias(args.path);
280
+ const target = resolveSdkAlias(args.path)
281
+ ?? (PLATFORM_RESOLVED_PACKAGES.includes(args.path) ? args.path : null);
221
282
  if (!target) return undefined;
222
283
  // `importer` is the importing FILE, `resolveDir` already a directory.
223
284
  const from = args.importer ? dirname(args.importer) : args.resolveDir;
224
- try {
225
- return { path: bun.resolveSync(target, from && from.length > 0 ? from : process.cwd()) };
226
- } catch {
227
- return undefined;
285
+ const own = tryResolve(target, from && from.length > 0 ? from : process.cwd());
286
+ if (own) return { path: own };
287
+ for (const root of sdkFallbackRoots()) {
288
+ const hit = tryResolve(target, root);
289
+ if (hit) return { path: hit };
290
+ // zod is the SDK's peer: a root that holds the SDK resolves zod
291
+ // from the SDK's own directory even when it isn't hoisted.
292
+ if (target !== SDK_PACKAGE) {
293
+ const sdk = tryResolve(SDK_PACKAGE, root);
294
+ const beside = sdk ? tryResolve(target, dirname(sdk)) : null;
295
+ if (beside) return { path: beside };
296
+ }
228
297
  }
298
+ return undefined;
229
299
  });
230
300
  },
231
301
  };
@@ -557,10 +627,9 @@ export async function bundleWorkflow(
557
627
  // schemas → compact JSON-Schema-shaped descriptions carried in
558
628
  // template metadata. Used by the dashboard's "Registered" panel to
559
629
  // render schema tables and (eventually) client-side validation in
560
- // the playground. Run-form workflows stamp `z.unknown()` for both
561
- // — `extractIOSchema` filters those down to undefined so the
562
- // dashboard renders the "no declared schema" empty state instead
563
- // of a meaningless empty table.
630
+ // the playground. `extractIOSchema` filters a `z.unknown()` schema
631
+ // down to undefined so the dashboard renders the "no declared
632
+ // schema" empty state instead of a meaningless empty table.
564
633
  const inputSchema = extractIOSchema(workflow.input);
565
634
  const outputSchema = extractIOSchema(workflow.output);
566
635
 
@@ -78,9 +78,8 @@ const LIVE_EMIT_DRAIN_TIMEOUT_MS = 1_500;
78
78
  * this instance. After the step finishes, the engine calls `snapshot()`
79
79
  * to extract the bundle for transport.
80
80
  *
81
- * Metadata writes merge (later keys win) — matches the legacy
82
- * `mergeRunMetadata` semantics so authors don't see a behaviour change
83
- * across migrations.
81
+ * Metadata writes merge (later keys win), as the server's run-metadata
82
+ * merge does.
84
83
  */
85
84
  export class StepObservabilityCollector {
86
85
  private metadata: Record<string, unknown> = {};
@@ -6,7 +6,6 @@
6
6
  * 2. For each step:
7
7
  * a. Optionally check `getCachedOutput(stepIndex, step.name)` — if a
8
8
  * previous run completed this step, skip and reuse its output.
9
- * (Phase 1b uses this for crash recovery.)
10
9
  * b. Validate current input against `step.input`.
11
10
  * c. Call `step.run({ input, ... })`.
12
11
  * d. Validate return value against `step.output`.
@@ -17,9 +16,10 @@
17
16
  * Errors during any step bubble through `onStepFailed` and re-throw so the
18
17
  * caller can decide whether to mark the run failed.
19
18
  *
20
- * The cache + completion hooks are injection points — a sandbox engine
21
- * adapter (Phase 1c) wires them to `workflow_step_runs` rows. The default
22
- * is an in-process map for tests.
19
+ * The cache + completion hooks are injection points; without a cache every
20
+ * step runs. The platform's sandbox path does not walk the chain here: the
21
+ * Temporal `executeStep` activity runs one step per runner subprocess via
22
+ * `runWorkflowSingleStep` below.
23
23
  */
24
24
 
25
25
  import { z } from "zod";
@@ -76,13 +76,10 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
76
76
  /**
77
77
  * Crash-recovery hook. Called before a step executes. Return the cached
78
78
  * output to skip execution; return undefined to run the step.
79
- *
80
- * Phase 1b implementations will look up `workflow_step_runs` rows for
81
- * (runId, stepIndex, stepName) and return completed step outputs here.
82
79
  * Default: always undefined (no caching).
83
80
  */
84
81
  getCachedOutput?(stepIndex: number, stepName: string): unknown | undefined | Promise<unknown | undefined>;
85
- /** Fire after a step's `execute` and output validation succeed. */
82
+ /** Fire after a step's `run` and output validation succeed. */
86
83
  onStepCompleted?(stepIndex: number, stepName: string, output: unknown, durationMs: number): void | Promise<void>;
87
84
  /** Fire when a step throws or fails validation. The error is re-thrown after this returns. */
88
85
  onStepFailed?(stepIndex: number, stepName: string, error: Error, durationMs: number): void | Promise<void>;
@@ -6,9 +6,8 @@
6
6
  *
7
7
  * Why: durable suspend/resume requires step boundaries to be addressable as
8
8
  * data, not opaque async-function bodies. Each step's input + output is
9
- * serialisable JSON so engine adapters (in-process, sandbox, future
10
- * Inngest/Temporal) can record completion and replay from the last
11
- * completed step.
9
+ * serialisable JSON so the durable engine (Temporal, one activity per step)
10
+ * can record completion and replay from the last completed step.
12
11
  */
13
12
 
14
13
  import type { z } from "zod";
@@ -18,7 +17,7 @@ import type { WorkflowMetadata } from "../types/workflow-metadata.js";
18
17
  import type { StepObservability } from "./observability.js";
19
18
 
20
19
  /**
21
- * Per-step execution context. Threaded into every step's `execute(...)` so
20
+ * Per-step execution context. Threaded into every step's `run(...)` so
22
21
  * the step can read tenant identity and run identity, log progress, and
23
22
  * invoke sandbox commands.
24
23
  *
@@ -33,9 +32,9 @@ export interface StepContext<TInput = unknown> extends BaseExecutionContext {
33
32
  abortSignal: AbortSignal;
34
33
  /**
35
34
  * Merge key-value metadata onto the run record. Buffered during the
36
- * step and flushed when the step completes; the durable engine writes
37
- * it via the same `mergeRunMetadata` path the legacy runner used, so
38
- * the dashboard sees the same shape. Later keys win.
35
+ * step and flushed when the step completes; the durable engine merges
36
+ * it into the run's metadata (`persistStepObservability`). Later keys
37
+ * win.
39
38
  */
40
39
  setMetadata(data: Record<string, unknown>): Promise<void>;
41
40
  /**
@@ -98,9 +97,8 @@ export interface Step<TInput, TOutput> {
98
97
  }
99
98
 
100
99
  /**
101
- * The result of running one step. Engine adapters persist these into the
102
- * `workflow_step_runs` table (Phase 1b) so subsequent runs can skip
103
- * completed steps. `observability` carries the snapshot of
100
+ * The result of running one step, as `runWorkflowSteps` reports it per
101
+ * step. `observability` carries the snapshot of
104
102
  * `ctx.setMetadata` / `ctx.step` / `ctx.agentEvents` recorded during
105
103
  * the step; undefined when no hooks were used.
106
104
  */
@@ -16,7 +16,8 @@
16
16
  * matches the final step's `output` schema.
17
17
  *
18
18
  * Workflows-as-data — the result is consumable by any engine adapter
19
- * (in-process today; sandbox + Inngest/Temporal in future PRs).
19
+ * (in-process `runWorkflow`; the sandbox runner one step at a time under
20
+ * Temporal).
20
21
  */
21
22
 
22
23
  import type { z } from "zod";
@@ -6,7 +6,7 @@
6
6
  * runner.
7
7
  *
8
8
  * Errors classified into `WorkflowError` (user code threw) vs `EngineError`
9
- * (platform problem) for the runner harness to surface upstream.
9
+ * (platform problem).
10
10
  */
11
11
 
12
12
  import type { WorkflowHooks } from "../types/workflow.js";
@@ -53,7 +53,7 @@ export class EngineError extends Error {
53
53
  }
54
54
  }
55
55
 
56
- /** Classify any thrown value into the wire-level `kind` expected by `/fail`. */
56
+ /** Classify any thrown value as a `workflow` or an `engine` failure. */
57
57
  export function classifyError(err: unknown): "workflow" | "engine" {
58
58
  if (err instanceof WorkflowError) return "workflow";
59
59
  if (err instanceof EngineError) return "engine";
@@ -92,9 +92,7 @@ export interface RunWorkflowOptions {
92
92
  onStepStarted?: RunWorkflowStepsOpts<unknown, unknown>["onStepStarted"];
93
93
  onStepCompleted?: RunWorkflowStepsOpts<unknown, unknown>["onStepCompleted"];
94
94
  onStepFailed?: RunWorkflowStepsOpts<unknown, unknown>["onStepFailed"];
95
- /** Provider-specific child workflow invocation. Temporal/Inngest providers
96
- * inject their native child-workflow primitive; the LocalProvider injects
97
- * the public Agent Compose API client. */
95
+ /** Child workflow invocation. Unset, a step's `invokeChild` throws. */
98
96
  invokeChild?: InvokeChild;
99
97
  }
100
98
 
@@ -28,8 +28,8 @@ export function deriveInvokeChildIdempotencyKey(parentRunId: string, childName:
28
28
  }
29
29
 
30
30
  /**
31
- * Build the public-API child workflow invoker used by legacy and sandboxed
32
- * workflow execution. Provider-backed engines may inject a different
31
+ * Build the public-API child workflow invoker used by sandboxed workflow
32
+ * execution (the step runner). Provider-backed engines may inject a different
33
33
  * implementation (Temporal child workflow, Inngest invoke, etc.).
34
34
  */
35
35
  export function buildInvokeChild(