@agent-compose/sdk 0.5.1 → 0.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/dist/active-step.d.ts +60 -0
  2. package/dist/agent/agent-loop-steer.test.d.ts +1 -0
  3. package/dist/agent/agent-loop.d.ts +46 -0
  4. package/dist/agent/async-queue.d.ts +29 -0
  5. package/dist/agent/protocol.d.ts +9 -1
  6. package/dist/agent/resolve-agent-id.test.d.ts +1 -0
  7. package/dist/agent/run-agent.d.ts +16 -3
  8. package/dist/agent/steer-control.d.ts +57 -0
  9. package/dist/agent/steer-control.test.d.ts +1 -0
  10. package/dist/client.d.ts +161 -0
  11. package/dist/index.d.ts +16 -6
  12. package/dist/index.js +1586 -158
  13. package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
  14. package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
  15. package/dist/pause/__tests__/errors.test.d.ts +1 -0
  16. package/dist/pause/__tests__/manager.test.d.ts +1 -0
  17. package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
  18. package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
  19. package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
  20. package/dist/pause/checkpoint.d.ts +28 -0
  21. package/dist/pause/errors.d.ts +52 -0
  22. package/dist/pause/manager.d.ts +63 -0
  23. package/dist/pause/pause-core.d.ts +101 -0
  24. package/dist/pause/state-dir.d.ts +80 -0
  25. package/dist/pause/wrappers.d.ts +41 -0
  26. package/dist/request-context/request-context.d.ts +12 -0
  27. package/dist/runtimes/_cli-agent.d.ts +72 -0
  28. package/dist/runtimes/amp.d.ts +22 -0
  29. package/dist/runtimes/claude.d.ts +6 -0
  30. package/dist/runtimes/codex.d.ts +20 -0
  31. package/dist/runtimes/openai-desktop.d.ts +2 -0
  32. package/dist/runtimes/openai-desktop.js +1578 -157
  33. package/dist/runtimes/vercel.d.ts +53 -1
  34. package/dist/runtimes/vercel.js +60 -8
  35. package/dist/runtimes/vercel.test.d.ts +1 -0
  36. package/dist/sse.d.ts +2 -3
  37. package/dist/step-invocation/index.d.ts +2 -2
  38. package/dist/step-invocation/invoker.d.ts +3 -0
  39. package/dist/step-invocation/protocol.d.ts +12 -0
  40. package/dist/step-invocation/server.d.ts +1 -0
  41. package/dist/step-invocation/types.d.ts +40 -5
  42. package/dist/types/events.d.ts +9 -0
  43. package/dist/types/execution-context.d.ts +25 -0
  44. package/dist/types/protocol.d.ts +8 -0
  45. package/dist/types/runtime.d.ts +55 -0
  46. package/dist/types/sandbox.d.ts +4 -4
  47. package/dist/utils/schemas.d.ts +2 -0
  48. package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
  49. package/dist/workflow-steps/index.d.ts +2 -0
  50. package/dist/workflow-steps/observability.d.ts +43 -11
  51. package/dist/workflow-steps/run-callback.d.ts +39 -0
  52. package/dist/workflow-steps/runner.d.ts +8 -0
  53. package/package.json +1 -1
  54. package/src/active-step.ts +124 -0
  55. package/src/agent/agent-loop.ts +253 -19
  56. package/src/agent/async-queue.ts +61 -0
  57. package/src/agent/protocol.ts +12 -2
  58. package/src/agent/run-agent.ts +184 -8
  59. package/src/agent/steer-control.ts +125 -0
  60. package/src/client.ts +277 -0
  61. package/src/index.ts +38 -4
  62. package/src/pause/checkpoint.ts +44 -0
  63. package/src/pause/errors.ts +70 -0
  64. package/src/pause/manager.ts +177 -0
  65. package/src/pause/pause-core.ts +267 -0
  66. package/src/pause/state-dir.ts +262 -0
  67. package/src/pause/wrappers.ts +79 -0
  68. package/src/request-context/request-context.ts +17 -2
  69. package/src/runtimes/_cli-agent.ts +161 -0
  70. package/src/runtimes/amp.ts +94 -0
  71. package/src/runtimes/claude.ts +101 -6
  72. package/src/runtimes/codex.ts +109 -0
  73. package/src/runtimes/openai-desktop.ts +11 -0
  74. package/src/runtimes/vercel.ts +78 -2
  75. package/src/sandbox.ts +39 -20
  76. package/src/sse.ts +8 -6
  77. package/src/step-invocation/index.ts +2 -1
  78. package/src/step-invocation/invoker.ts +107 -29
  79. package/src/step-invocation/protocol.ts +16 -0
  80. package/src/step-invocation/server.ts +45 -12
  81. package/src/step-invocation/types.ts +43 -7
  82. package/src/tools/coding.ts +16 -5
  83. package/src/types/events.ts +9 -0
  84. package/src/types/execution-context.ts +25 -0
  85. package/src/types/protocol.ts +8 -0
  86. package/src/types/runtime.ts +52 -0
  87. package/src/types/sandbox.ts +8 -4
  88. package/src/types/workflow.ts +6 -1
  89. package/src/utils/bundler.ts +8 -3
  90. package/src/utils/schemas.ts +2 -0
  91. package/src/workflow-steps/index.ts +3 -0
  92. package/src/workflow-steps/observability.ts +84 -13
  93. package/src/workflow-steps/run-callback.ts +72 -0
  94. package/src/workflow-steps/runner.ts +70 -8
  95. package/dist/utils/discovery.d.ts +0 -2
  96. package/src/utils/discovery.ts +0 -4
@@ -22,12 +22,50 @@
22
22
  */
23
23
 
24
24
  import { randomBytes } from "node:crypto";
25
+ import { z } from "zod";
25
26
  import type { SandboxProvider } from "../types/sandbox.js";
26
- import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, requestContextPath, stepInputPath } from "./protocol.js";
27
+ import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath } from "./protocol.js";
28
+ import { StepPauseRequestSchema } from "./types.js";
27
29
  import type { StepRequest, StepResult } from "./types.js";
30
+ import type { StepObservability } from "../workflow-steps/observability.js";
28
31
 
29
32
  export { RUNNER_COMMAND } from "./protocol.js";
30
33
 
34
+ const StepObservabilitySchema = z.object({
35
+ metadata: z.record(z.string(), z.unknown()).optional(),
36
+ events: z.array(z.unknown()).optional(),
37
+ subSteps: z.array(z.object({
38
+ name: z.string(),
39
+ startedAt: z.number(),
40
+ durationMs: z.number(),
41
+ status: z.enum(["completed", "failed"]),
42
+ error: z.string().optional(),
43
+ })).optional(),
44
+ }).passthrough();
45
+
46
+ const StepResultEnvelopeSchema = z.discriminatedUnion("ok", [
47
+ z.object({
48
+ ok: z.literal(true),
49
+ output: z.unknown().optional(),
50
+ observability: StepObservabilitySchema.optional(),
51
+ }),
52
+ z.object({
53
+ ok: z.literal(false),
54
+ kind: z.enum(["protocol", "user-step"]),
55
+ error: z.string().optional(),
56
+ }),
57
+ ]);
58
+
59
+ /** Wire schema for the pause sentinel — what the runner emits on stdout
60
+ * when user code calls `ctx.pause(...)`. The inner `pauseRequest` schema
61
+ * is the canonical `StepPauseRequestSchema` from types.ts so the runtime
62
+ * type (`StepPauseRequest`) and the parsed shape can't drift. */
63
+ const StepPauseEnvelopeSchema = z.object({
64
+ pauseId: z.string().min(1),
65
+ pauseRequest: StepPauseRequestSchema,
66
+ observability: StepObservabilitySchema.optional(),
67
+ });
68
+
31
69
  /** Build the env map for one step invocation. Pure function; tests use it
32
70
  * to assert the env shape (and to assert that no server-held credentials
33
71
  * leak in) without spawning a sandbox. */
@@ -35,6 +73,7 @@ export function buildStepEnvs(args: {
35
73
  runId: string;
36
74
  stepIndex: number;
37
75
  resultToken: string;
76
+ isResume?: boolean;
38
77
  }): Record<string, string> {
39
78
  return {
40
79
  [STEP_ENV.RUN_ID]: args.runId,
@@ -43,6 +82,7 @@ export function buildStepEnvs(args: {
43
82
  [STEP_ENV.STEP_INPUT_PATH]: stepInputPath(args.stepIndex),
44
83
  [STEP_ENV.REQUEST_CONTEXT_PATH]: requestContextPath(args.stepIndex),
45
84
  [STEP_ENV.STEP_RESULT_TOKEN]: args.resultToken,
85
+ [STEP_ENV.STEP_RESUME]: args.isResume ? "1" : "0",
46
86
  };
47
87
  }
48
88
 
@@ -59,35 +99,61 @@ export function parseStepResult<TOutput = unknown>(
59
99
  stdout: string,
60
100
  resultToken: string,
61
101
  ): StepResult<TOutput> | null {
62
- const prefix = stepResultLinePrefix(resultToken);
63
- const line = stdout.split(/\r?\n/).find((l) => l.startsWith(prefix));
64
- if (!line) return null;
65
- let parsed: {
66
- ok: boolean;
67
- output?: unknown;
68
- kind?: string;
69
- error?: string;
70
- observability?: import("../workflow-steps/observability.js").StepObservability;
71
- };
102
+ // Pause sentinel takes precedence: a step that called `ctx.pause(...)`
103
+ // exits without ever emitting a result, but if the runner did emit a
104
+ // partial / stale result in a prior buffer chunk, the pause sentinel
105
+ // is the load-bearing signal. Scan one pass, dispatch by prefix.
106
+ const resultPrefix = stepResultLinePrefix(resultToken);
107
+ const pausePrefix = stepPauseLinePrefix(resultToken);
108
+ let resultLine: string | undefined;
109
+ let pauseLine: string | undefined;
110
+ for (const line of stdout.split(/\r?\n/)) {
111
+ if (!pauseLine && line.startsWith(pausePrefix)) pauseLine = line;
112
+ if (!resultLine && line.startsWith(resultPrefix)) resultLine = line;
113
+ if (pauseLine && resultLine) break;
114
+ }
115
+
116
+ if (pauseLine) {
117
+ try {
118
+ const raw = JSON.parse(pauseLine.slice(pausePrefix.length));
119
+ const parsed = StepPauseEnvelopeSchema.safeParse(raw);
120
+ if (!parsed.success) {
121
+ return { ok: false, error: { kind: "protocol", message: `step pause schema validation failed: ${parsed.error.message}` } };
122
+ }
123
+ return {
124
+ ok: "paused",
125
+ pauseId: parsed.data.pauseId,
126
+ pauseRequest: parsed.data.pauseRequest,
127
+ ...(parsed.data.observability ? { observability: parsed.data.observability as StepObservability } : {}),
128
+ };
129
+ } catch {
130
+ return { ok: false, error: { kind: "protocol", message: "step pause JSON parse failed" } };
131
+ }
132
+ }
133
+
134
+ if (!resultLine) return null;
135
+ let parsed: z.infer<typeof StepResultEnvelopeSchema>;
72
136
  try {
73
- parsed = JSON.parse(line.slice(prefix.length));
137
+ const raw = JSON.parse(resultLine.slice(resultPrefix.length));
138
+ const result = StepResultEnvelopeSchema.safeParse(raw);
139
+ if (!result.success) {
140
+ const kind = raw && typeof raw === "object" && typeof (raw as { kind?: unknown }).kind === "string"
141
+ ? ` (kind=${(raw as { kind: string }).kind})`
142
+ : "";
143
+ return { ok: false, error: { kind: "protocol", message: `step result schema validation failed${kind}: ${result.error.message}` } };
144
+ }
145
+ parsed = result.data;
74
146
  } catch {
75
147
  return { ok: false, error: { kind: "protocol", message: "step result JSON parse failed" } };
76
148
  }
77
149
  if (parsed.ok) {
78
150
  return parsed.observability
79
- ? { ok: true, output: parsed.output as TOutput, observability: parsed.observability }
151
+ ? { ok: true, output: parsed.output as TOutput, observability: parsed.observability as StepObservability }
80
152
  : { ok: true, output: parsed.output as TOutput };
81
153
  }
82
- const kind: "protocol" | "user-step" = parsed.kind === "user-step" ? "user-step" : "protocol";
154
+ const kind = parsed.kind;
83
155
  const baseMessage = parsed.error ?? "step body failed without a message";
84
- // An unknown kind value means the runner emitted something this invoker
85
- // doesn't recognise — most likely a version-skew deploy. Preserve the
86
- // original kind string in the message so operators see the actual value
87
- // instead of silently re-classifying it as "protocol".
88
- const message = parsed.kind && parsed.kind !== "user-step" && parsed.kind !== "protocol"
89
- ? `(unknown kind "${parsed.kind}") ${baseMessage}`
90
- : baseMessage;
156
+ const message = baseMessage;
91
157
  return { ok: false, error: { kind, message } };
92
158
  }
93
159
 
@@ -101,6 +167,8 @@ export function parseStepResult<TOutput = unknown>(
101
167
  */
102
168
  export interface InvokeStepOptions {
103
169
  envs?: Record<string, string>;
170
+ /** True when re-entering a step after resolving or expiring a pause. */
171
+ isResume?: boolean;
104
172
  /** Live stdout / stderr from the runner subprocess, line-by-line. Called
105
173
  * from inside `sandbox.commands.run` as chunks arrive. The sentinel line
106
174
  * carrying the protocol result token is filtered out before delivery so
@@ -131,17 +199,21 @@ export async function invokeStep<TOutput = unknown>(
131
199
  runId: request.runId,
132
200
  stepIndex: request.stepIndex,
133
201
  resultToken,
202
+ isResume: opts?.isResume,
134
203
  }),
135
204
  };
136
205
 
137
206
  // The sandbox emits stdout as raw chunks, not lines. Buffer between
138
207
  // emissions so a `console.log` split across two chunks (or a partial
139
208
  // trailing line) is delivered to onStdout/onStderr as one logical line.
140
- // The sentinel line (carrying `resultToken`) is filtered out so callers
141
- // never see protocol bytes in user-log capture. The prefix is built
142
- // from `stepResultLinePrefix` — the same helper `parseStepResult` uses,
143
- // so the filter and the parser can't drift.
144
- const sentinelPrefix = stepResultLinePrefix(resultToken);
209
+ // Sentinel lines (carrying `resultToken`) are filtered out so callers
210
+ // never see protocol bytes in user-log capture. Both the result and
211
+ // pause prefixes are built from the same shared helpers `parseStepResult`
212
+ // uses, so the filter and the parser can't drift apart.
213
+ const resultSentinel = stepResultLinePrefix(resultToken);
214
+ const pauseSentinel = stepPauseLinePrefix(resultToken);
215
+ const isSentinel = (line: string) =>
216
+ line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
145
217
  const makeLineSplitter = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
146
218
  if (!sink) return { onChunk: undefined, flush: () => {} };
147
219
  let buf = "";
@@ -152,7 +224,7 @@ export async function invokeStep<TOutput = unknown>(
152
224
  while ((nl = buf.indexOf("\n")) !== -1) {
153
225
  const line = buf.slice(0, nl);
154
226
  buf = buf.slice(nl + 1);
155
- if (filterSentinel && line.startsWith(sentinelPrefix)) continue;
227
+ if (filterSentinel && isSentinel(line)) continue;
156
228
  sink(line);
157
229
  }
158
230
  },
@@ -164,7 +236,7 @@ export async function invokeStep<TOutput = unknown>(
164
236
  if (buf.length === 0) return;
165
237
  const line = buf;
166
238
  buf = "";
167
- if (filterSentinel && line.startsWith(sentinelPrefix)) return;
239
+ if (filterSentinel && isSentinel(line)) return;
168
240
  sink(line);
169
241
  },
170
242
  };
@@ -188,11 +260,17 @@ export async function invokeStep<TOutput = unknown>(
188
260
  // before emitting" (runner-exit) from "runner exited cleanly but didn't
189
261
  // speak the protocol" (protocol) — the exit code is the evidence.
190
262
  if (result.exitCode !== 0) {
263
+ const stderrTail = result.stderr.trim().slice(-2000);
264
+ const stdoutTail = result.stdout.trim().slice(-2000);
265
+ const details = [
266
+ stderrTail ? `stderr:\n${stderrTail}` : "",
267
+ stdoutTail ? `stdout:\n${stdoutTail}` : "",
268
+ ].filter(Boolean).join("\n");
191
269
  return {
192
270
  ok: false,
193
271
  error: {
194
272
  kind: "runner-exit",
195
- message: `runner subprocess exited ${result.exitCode} before emitting a step result`,
273
+ message: `runner subprocess exited ${result.exitCode} before emitting a step result${details ? `\n${details}` : ""}`,
196
274
  exitCode: result.exitCode,
197
275
  },
198
276
  };
@@ -14,6 +14,13 @@
14
14
  * with the same prefix but a wrong token is rejected. */
15
15
  export const STEP_RESULT_PREFIX = "__AC_STEP_RESULT__";
16
16
 
17
+ /** Parallel sentinel for pause requests. The runner emits this — instead
18
+ * of (not in addition to) a step result — when user code calls
19
+ * `ctx.pause(...)`. Exits cleanly afterwards so the activity can snapshot
20
+ * the now-frozen sandbox and durably wait via Temporal `condition()`.
21
+ * See ADR-0006 §"How it actually pauses — Temporal-native, end-to-end". */
22
+ export const STEP_PAUSE_PREFIX = "__AC_STEP_PAUSE__";
23
+
17
24
  /** Build the line prefix the runner emits and the invoker scans for. The
18
25
  * runner-side serveStep writes `<prefix><token>:<json>\n`; the invoker's
19
26
  * result parser AND the stdout line splitter both match on this exact
@@ -23,6 +30,14 @@ export function stepResultLinePrefix(token: string): string {
23
30
  return `${STEP_RESULT_PREFIX}${token}:`;
24
31
  }
25
32
 
33
+ /** Build the pause sentinel line prefix. Same per-invocation token as the
34
+ * result sentinel so the two channels share one secret — a user
35
+ * `console.log` can't forge either without knowing the token, and the
36
+ * invoker can filter both prefixes from captured stdout with one token. */
37
+ export function stepPauseLinePrefix(token: string): string {
38
+ return `${STEP_PAUSE_PREFIX}${token}:`;
39
+ }
40
+
26
41
  /** Sandbox-side path where dispatch writes the compiled runner bundle.
27
42
  * Both modes (full-mode `dispatch.ts` and step-mode `invokeStep`) spawn
28
43
  * the runner from this path; single source of truth. */
@@ -41,6 +56,7 @@ export const STEP_ENV = {
41
56
  STEP_INPUT_PATH: "AC_STEP_INPUT_PATH",
42
57
  REQUEST_CONTEXT_PATH: "AC_REQUEST_CONTEXT_PATH",
43
58
  STEP_RESULT_TOKEN: "AC_STEP_RESULT_TOKEN",
59
+ STEP_RESUME: "AC_STEP_RESUME",
44
60
  } as const;
45
61
 
46
62
  /** Sandbox-side path where the invoker writes the JSON-encoded step input.
@@ -23,14 +23,17 @@
23
23
 
24
24
  import { readFileSync } from "node:fs";
25
25
  import type { RequestContextWire } from "../request-context/request-context.js";
26
+ import { RequestContextWireSchema } from "../request-context/request-context.js";
26
27
  import type { StepObservability } from "../workflow-steps/observability.js";
27
- import { STEP_ENV, STEP_RESULT_PREFIX } from "./protocol.js";
28
+ import { STEP_ENV, STEP_PAUSE_PREFIX, STEP_RESULT_PREFIX } from "./protocol.js";
29
+ import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
28
30
 
29
31
  /** What the handler receives. The serveStep already parsed the input
30
32
  * and request context from their JSON files. */
31
33
  export interface ServeStepRequest<TInput = unknown> {
32
34
  runId: string;
33
35
  stepIndex: number;
36
+ isResume: boolean;
34
37
  input: TInput;
35
38
  requestContext: RequestContextWire;
36
39
  }
@@ -64,9 +67,8 @@ type SentinelPayload =
64
67
  * immediately after a non-awaited write can drop the data on Linux
65
68
  * pipes, leaving the invoker with empty stdout and the runner-exit
66
69
  * classification pointing at the wrong thing. */
67
- function emitResult(token: string, payload: SentinelPayload): Promise<void> {
70
+ function writeSentinelLine(line: string): Promise<void> {
68
71
  return new Promise<void>((resolve, reject) => {
69
- const line = `${STEP_RESULT_PREFIX}${token}:${JSON.stringify(payload)}\n`;
70
72
  process.stdout.write(line, (err) => {
71
73
  if (err) reject(err);
72
74
  else resolve();
@@ -74,6 +76,24 @@ function emitResult(token: string, payload: SentinelPayload): Promise<void> {
74
76
  });
75
77
  }
76
78
 
79
+ function emitResult(token: string, payload: SentinelPayload): Promise<void> {
80
+ return writeSentinelLine(`${STEP_RESULT_PREFIX}${token}:${JSON.stringify(payload)}\n`);
81
+ }
82
+
83
+ /** Emit the tokenised pause sentinel — `__AC_STEP_PAUSE__<token>:{pauseId,
84
+ * pauseRequest}` — when the handler threw a `PauseSignal` instead of
85
+ * returning. Same per-invocation token + awaited-flush discipline as
86
+ * `emitResult`; the invoker's `parseStepResult` matches this exact framing
87
+ * (and pause takes precedence over any stray result line). */
88
+ function emitPause(token: string, signal: PauseSignal): Promise<void> {
89
+ const body = JSON.stringify({
90
+ pauseId: signal.pauseId,
91
+ pauseRequest: signal.pauseRequest,
92
+ ...(signal.observability ? { observability: signal.observability } : {}),
93
+ });
94
+ return writeSentinelLine(`${STEP_PAUSE_PREFIX}${token}:${body}\n`);
95
+ }
96
+
77
97
  /** Delete step-invocation transport envs so the user's workflow code
78
98
  * (loaded inside the handler) can't read them. Matters most for
79
99
  * AC_STEP_RESULT_TOKEN: if it stayed in env, a `console.log` from a dep
@@ -89,6 +109,7 @@ function scrubProtocolEnvs(): void {
89
109
  STEP_ENV.STEP_INPUT_PATH,
90
110
  STEP_ENV.REQUEST_CONTEXT_PATH,
91
111
  STEP_ENV.STEP_RESULT_TOKEN,
112
+ STEP_ENV.STEP_RESUME,
92
113
  ]) {
93
114
  delete process.env[key];
94
115
  }
@@ -129,13 +150,19 @@ export async function serveStep<TInput, TOutput>(
129
150
  // and re-emitted as a step failure.
130
151
  let exitCode = 0;
131
152
  let payload!: SentinelPayload;
153
+ // A pause is not a result. When the handler throws `PauseSignal` (user code
154
+ // called `ctx.pause`), emit the pause sentinel — not a result — and exit 0,
155
+ // so the activity snapshots the now-frozen sandbox and waits durably.
156
+ // ADR-0006 §"How it actually pauses".
157
+ let pauseSignal: PauseSignal | undefined;
132
158
 
133
- let setupResult: { runId: string; stepIndex: number; input: TInput; requestContext: RequestContextWire } | undefined;
159
+ let setupResult: { runId: string; stepIndex: number; isResume: boolean; input: TInput; requestContext: RequestContextWire } | undefined;
134
160
  try {
135
161
  const runId = process.env[STEP_ENV.RUN_ID];
136
162
  const stepIndexEnv = process.env[STEP_ENV.STEP_INDEX];
137
163
  const inputPath = process.env[STEP_ENV.STEP_INPUT_PATH];
138
164
  const ctxPath = process.env[STEP_ENV.REQUEST_CONTEXT_PATH];
165
+ const isResume = process.env[STEP_ENV.STEP_RESUME] === "1";
139
166
  const stepIndex = stepIndexEnv === undefined ? Number.NaN : Number(stepIndexEnv);
140
167
 
141
168
  if (!runId || !inputPath || !ctxPath || !Number.isFinite(stepIndex)) {
@@ -150,13 +177,9 @@ export async function serveStep<TInput, TOutput>(
150
177
  }
151
178
 
152
179
  const input = JSON.parse(readFileSync(inputPath, "utf8")) as TInput;
153
- // RequestContextWire parsing is loose here on purpose — the wire is
154
- // server-controlled, malformed shape is a server bug, and we want a
155
- // useful failure reason in `failRun`. Strict validation belongs at
156
- // the workflow-level RequestContext.parse boundary, not here.
157
- const requestContext = JSON.parse(readFileSync(ctxPath, "utf8")) as RequestContextWire;
180
+ const requestContext = RequestContextWireSchema.parse(JSON.parse(readFileSync(ctxPath, "utf8")));
158
181
 
159
- setupResult = { runId, stepIndex, input, requestContext };
182
+ setupResult = { runId, stepIndex, isResume, input, requestContext };
160
183
  } catch (err) {
161
184
  exitCode = 1;
162
185
  payload = { ok: false, kind: "protocol", error: err instanceof Error ? err.message : String(err) };
@@ -174,11 +197,21 @@ export async function serveStep<TInput, TOutput>(
174
197
  ? { ok: true, output, observability }
175
198
  : { ok: true, output };
176
199
  } catch (err) {
177
- exitCode = 1;
178
- payload = { ok: false, kind: "user-step", error: err instanceof Error ? err.message : String(err) };
200
+ // PauseSignal is control flow, not a failure — emit a pause, not a
201
+ // user-step error. Checked first so it can't be misclassified.
202
+ if (isPauseSignal(err)) {
203
+ pauseSignal = err;
204
+ } else {
205
+ exitCode = 1;
206
+ payload = { ok: false, kind: "user-step", error: err instanceof Error ? err.message : String(err) };
207
+ }
179
208
  }
180
209
  }
181
210
 
211
+ if (pauseSignal) {
212
+ await emitPause(token, pauseSignal);
213
+ process.exit(0);
214
+ }
182
215
  await emitResult(token, payload);
183
216
  process.exit(exitCode);
184
217
  }
@@ -3,6 +3,7 @@
3
3
  * the discriminated error union that classifies failure modes.
4
4
  */
5
5
 
6
+ import { z } from "zod";
6
7
  import type { RequestContextWire } from "../request-context/request-context.js";
7
8
  import type { StepObservability } from "../workflow-steps/observability.js";
8
9
 
@@ -19,15 +20,50 @@ export interface StepRequest<TInput = unknown> {
19
20
  }
20
21
 
21
22
  /**
22
- * Outcome of one step invocation. Successful runs carry the step's output;
23
- * failed runs carry a kinded error so the caller can distinguish "user
24
- * code threw" from "runner crashed before emitting" from "wire protocol
25
- * violation". The dashboard surfaces the kind to operators; the activity
26
- * uses the kind to pick a useful failRun reason.
23
+ * Outcome of one step invocation. Three terminal states:
24
+ *
25
+ * - `ok: true` — step body resolved with `output`.
26
+ * - `ok: false` — step body or runner errored; kind classifies why.
27
+ * - `ok: "paused"` — step body called `ctx.pause(...)` and exited cleanly.
28
+ * The activity captures a sandbox snapshot and the workflow waits on
29
+ * Temporal `condition()` until something resumes the pause (HTTP
30
+ * resume route, TTL expiry, cancellation). `pauseRequest` is what the
31
+ * caller passed to `ctx.pause` — the route surfaces it on the
32
+ * pending-pauses dashboard so an operator sees the question being
33
+ * asked. `observability` carries any ctx metadata/sub-step/agent events
34
+ * recorded before the pause unwound. See ADR-0006.
27
35
  */
28
36
  export type StepResult<TOutput = unknown> =
29
- | { ok: true; output: TOutput; observability?: StepObservability }
30
- | { ok: false; error: StepInvocationError };
37
+ | { ok: true; output: TOutput; observability?: StepObservability }
38
+ | { ok: "paused"; pauseId: string; pauseRequest: StepPauseRequest; observability?: StepObservability }
39
+ | { ok: false; error: StepInvocationError };
40
+
41
+ /** Wire schema for one pause request. Canonical here — `parseStepResult`
42
+ * imports it for validation, and `StepPauseRequest` is `z.infer`'d from
43
+ * it so the runtime type and the parsed shape can't drift.
44
+ *
45
+ * Loose by design (the inner zod shape stops at orchestration fields
46
+ * the engine needs): the SDK wrapper layer (`requestDecision` /
47
+ * `sleep` / `waitForEvent`) owns its own payload contract, and the
48
+ * engine treats `payload` as opaque. */
49
+ export const StepPauseRequestSchema = z.object({
50
+ /** Short human-readable label; surfaces on the pending-pauses feed. */
51
+ reason: z.string().min(1),
52
+ /** Wrapper discriminator: which SDK helper produced this pause. */
53
+ kind: z.enum(["decision", "sleep", "event", "custom"]),
54
+ /** Caller-supplied payload — opaque to the engine. */
55
+ payload: z.record(z.string(), z.unknown()).optional(),
56
+ /** Pause TTL in milliseconds. Bounded by Temporal's sleep durability. */
57
+ ttlMs: z.number().int().positive().optional(),
58
+ /** Optional second-key resume route — pending pause is unique per
59
+ * (run, correlationKey) so event-driven senders can resume without
60
+ * knowing the pause id. */
61
+ correlationKey: z.string().min(1).optional(),
62
+ /** Skip the per-pause sandbox snapshot. Default behaviour is to
63
+ * snapshot; only the lightweight wrappers (`ctx.sleep`) opt out. */
64
+ snapshot: z.boolean().optional(),
65
+ });
66
+ export type StepPauseRequest = z.infer<typeof StepPauseRequestSchema>;
31
67
 
32
68
  /**
33
69
  * Discriminated error union.
@@ -10,6 +10,13 @@ function resolvePath(path: string, cwd?: string): string {
10
10
  return cwd && !isAbsolute(path) ? join(cwd, path) : path;
11
11
  }
12
12
 
13
+ function formatCommandFailure(command: string, result: { exitCode: number; stdout: string; stderr: string }): string {
14
+ const parts = [`Command failed with exit code ${result.exitCode}: ${command}`];
15
+ if (result.stderr.trim()) parts.push(`stderr:\n${result.stderr}`);
16
+ if (result.stdout.trim()) parts.push(`stdout:\n${result.stdout}`);
17
+ return parts.join("\n");
18
+ }
19
+
13
20
  async function readFile(sandbox: SandboxProvider, path: string, opts?: { offset?: number; limit?: number; cwd?: string }): Promise<string> {
14
21
  const script = `
15
22
  const fs = require("fs");
@@ -26,14 +33,18 @@ for (let i = start; i <= end; i++) console.log(String(i) + ": " + lines[i - 1]);
26
33
  .filter(Boolean)
27
34
  .map(q)
28
35
  .join(" ");
29
- const { stdout } = await sandbox.commands.run(`node -e ${q(script)} ${args}`, { cwd: opts?.cwd, timeoutMs: 30_000 });
30
- return stdout;
36
+ const command = `node -e ${q(script)} ${args}`;
37
+ const result = await sandbox.commands.run(command, { cwd: opts?.cwd, timeoutMs: 30_000 });
38
+ if (result.exitCode !== 0) throw new Error(formatCommandFailure(command, result));
39
+ return result.stdout;
31
40
  }
32
41
 
33
42
  async function readRawFile(sandbox: SandboxProvider, path: string, cwd?: string): Promise<string> {
34
43
  const script = `const fs = require("fs"); process.stdout.write(fs.readFileSync(process.argv[1], "utf8"));`;
35
- const { stdout } = await sandbox.commands.run(`node -e ${q(script)} ${q(path)}`, { cwd, timeoutMs: 30_000 });
36
- return stdout;
44
+ const command = `node -e ${q(script)} ${q(path)}`;
45
+ const result = await sandbox.commands.run(command, { cwd, timeoutMs: 30_000 });
46
+ if (result.exitCode !== 0) throw new Error(formatCommandFailure(command, result));
47
+ return result.stdout;
37
48
  }
38
49
 
39
50
  export interface CodingTool<TInput extends Record<string, unknown> = Record<string, unknown>> {
@@ -118,7 +129,7 @@ export const bashTool: CodingTool<{
118
129
  cwd: cwd ?? ctx.cwd,
119
130
  timeoutMs: timeoutMs ?? 120_000,
120
131
  });
121
- if (result.exitCode !== 0) throw new Error(`Command failed with exit code ${result.exitCode}${result.stdout ? `\n${result.stdout}` : ""}`);
132
+ if (result.exitCode !== 0) throw new Error(formatCommandFailure(command, result));
122
133
  return result.stdout;
123
134
  },
124
135
  };
@@ -15,6 +15,15 @@ export type RunEvent =
15
15
  seq?: number;
16
16
  agentId: string;
17
17
  label: string;
18
+ /** Tool whitelist passed to `agent({ tools: [...] })`. Drives the
19
+ * per-agent "tools" badge on the dashboard. */
20
+ allowedTools?: string[];
21
+ /** Resolved model id (e.g. `claude-sonnet-4-6`). */
22
+ model?: string;
23
+ /** Short runtime self-identifier (`claude`, `openai-desktop`, …)
24
+ * read off `ModelExecutionContract.kind`. The dashboard maps
25
+ * this to a small runtime icon on the agent card header. */
26
+ runtimeKind?: string;
18
27
  }
19
28
  | {
20
29
  event: "agent.message";
@@ -2,6 +2,8 @@
2
2
 
3
3
  import type { InvokeAndWaitOptions, RunStatus } from "../client.js";
4
4
  import type { RequestContext } from "../request-context/request-context.js";
5
+ import type { PauseRequest } from "../pause/pause-core.js";
6
+ import type { RequestDecisionRequest, WaitForEventRequest } from "../pause/wrappers.js";
5
7
  import type { SandboxProvider } from "./sandbox.js";
6
8
 
7
9
  /** The identity of this workflow run. */
@@ -27,4 +29,27 @@ export interface BaseExecutionContext {
27
29
  setMetadata?: (data: Record<string, unknown>) => Promise<void>;
28
30
  /** Invoke another registered workflow and wait for it to settle. */
29
31
  invokeChild: InvokeChild;
32
+ /**
33
+ * Disk-backed memoise across pause-resume. First call runs `fn` and
34
+ * atomically writes the result to the sandbox; on resume the recorded
35
+ * value is returned and `fn` is NOT re-executed. Use for expensive
36
+ * deterministic transforms; for side effects, use `invokeChild`.
37
+ * See ADR-0006 §"`ctx.checkpoint(name, fn)` — disk-backed memoisation".
38
+ */
39
+ checkpoint<T>(name: string, fn: () => Promise<T> | T): Promise<T>;
40
+ /**
41
+ * Pause for feedback. The step exits and the workflow waits durably until
42
+ * something resolves the pause (a resume call, a TTL expiry); on resume the
43
+ * step body re-runs from the top and this call returns the resume payload.
44
+ * Provide a `schema` to validate the payload, `ttlMs` + `onExpiry` to bound
45
+ * the wait, `correlationKey` for by-key resume. Throws PauseRequestError /
46
+ * PauseExpiredError / PauseSchemaError. See ADR-0006 / ADR-0011.
47
+ */
48
+ pause<T = unknown>(req: PauseRequest<T>): Promise<T>;
49
+ /** Pause for a typed decision (a `schema` is required). Wrapper over `pause`. */
50
+ requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
51
+ /** Lightweight timed pause — resolves after `durationMs`, no snapshot. */
52
+ sleep(durationMs: number): Promise<void>;
53
+ /** Pause until an event resumes by `correlationKey`. Wrapper over `pause`. */
54
+ waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
30
55
  }
@@ -71,4 +71,12 @@ export interface AgentStatus {
71
71
  completed: string[];
72
72
  blockers: string[];
73
73
  exit_signal: boolean;
74
+ /** PR 7 self-pause (honoured only for `mode: "hitl"` agents). The agent
75
+ * cannot proceed without a human decision: it sets this true, puts the
76
+ * question in `question`, and ends its turn. The loop pauses at the next
77
+ * boundary and injects the human's answer as the next turn. `auto` agents
78
+ * ignore it and keep going. Should be paired with `exit_signal: false`. */
79
+ needs_input?: boolean;
80
+ /** The question to put to the human when `needs_input` is true. */
81
+ question?: string;
74
82
  }
@@ -48,14 +48,66 @@ export type ToolCallGateResult =
48
48
  export interface ModelExecutionContract {
49
49
  /** True when this runtime can run `processToolCall` before tool execution. */
50
50
  supportsToolCallProcessor?: boolean;
51
+ /** Short self-identifier ("claude", "openai-desktop", "vercel", …). Read
52
+ * by the agent loop and surfaced on `agent.spawned` so the dashboard
53
+ * can show a per-agent runtime icon without re-fetching template
54
+ * metadata. Optional — runtimes that omit it stay anonymous. */
55
+ kind?: string;
56
+ /** Resolved model id used by this runtime instance (already merged with
57
+ * config + runtime defaults). Surfaced on `agent.spawned` so each
58
+ * agent card on the Agent tab can label which model it ran against. */
59
+ model?: string;
51
60
  /** Runtime-owned pre-tool gate. Adapters call the shared processor chain
52
61
  * through this seam; the agent loop stays SDK-agnostic. */
53
62
  gateToolCall?(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
63
+ /**
64
+ * Capture runtime-private in-memory state that will not survive the
65
+ * runner subprocess exit. Called by the agent loop at pause time,
66
+ * AFTER the loop has flushed its own state to disk.
67
+ *
68
+ * Return value is opaque to the loop — whatever the runtime needs to
69
+ * round-trip its conversation across pause-resume. Must be JSON-
70
+ * serialisable; the loop atomically writes it to
71
+ * `/tmp/wf/state/runtime-<agentInstanceId>.json` and reads it back
72
+ * on resume to hand to `restoreCheckpoint`.
73
+ *
74
+ * Default (method omitted): runtime holds no instance state that
75
+ * needs to round-trip across pause. The shipped example is the
76
+ * Claude runtime — the conversation lives server-side at
77
+ * Anthropic, addressed by `session_id`, and the loop already holds
78
+ * `lastSessionId` as part of its own state. On resume the loop
79
+ * restores the id, the next `sendMessage` passes it through, and
80
+ * Anthropic resumes the server-side conversation.
81
+ *
82
+ * Runtimes that hold the conversation in-process — Vercel's
83
+ * `VercelRunner.messages` is the canonical case — MUST implement
84
+ * both hooks: the messages array is reconstructed from
85
+ * `response.messages` on each `streamText` and would be lost the
86
+ * moment the subprocess exits. See ADR-0006 §"Concrete examples
87
+ * for shipped runtimes" for the audit + worked examples.
88
+ */
89
+ captureCheckpoint?(): unknown;
90
+ /**
91
+ * Restore runtime-private state previously returned by
92
+ * `captureCheckpoint`. Called by the agent loop on resume, AFTER
93
+ * the loop has restored its own state but BEFORE iterations resume.
94
+ *
95
+ * `blob` is whatever this same runtime returned at pause time. If
96
+ * `captureCheckpoint` is omitted, this is never called.
97
+ */
98
+ restoreCheckpoint?(blob: unknown): void;
54
99
  sendMessage(opts: {
55
100
  prompt: string;
56
101
  sessionId?: string;
57
102
  iteration?: number;
58
103
  signal?: AbortSignal;
104
+ /** Push-iterable of mid-turn user messages from outside the agent
105
+ * loop — e.g. dashboard chat injections. Runtimes that support
106
+ * streaming-input mode (Claude Agent SDK) read from this in
107
+ * parallel with the initial `prompt`; the SDK handles delivery
108
+ * at the next safe boundary. Runtimes without streaming-input
109
+ * support ignore this and fall back to per-iteration injection. */
110
+ inboxStream?: AsyncIterable<{ text: string; senderName?: string | null }>;
59
111
  }): AsyncGenerator<AgentMessage>;
60
112
  }
61
113
 
@@ -10,17 +10,17 @@ export interface SandboxCommandRunOptions {
10
10
  envs?: Record<string, string>;
11
11
  onStdout?: (data: string) => void;
12
12
  onStderr?: (data: string) => void;
13
- background?: boolean;
14
13
  /** Run the command with root privileges. Vercel maps this to its native
15
- * `sudo` flag; E2B runs the command as `user: "root"`; the local provider
16
- * prepends `sudo`. Defaults to false. Requires the sandbox image to grant
17
- * the command root (Vercel's runtimes do — passwordless). */
14
+ * `sudo` flag; the local provider prepends `sudo`. Defaults to false.
15
+ * Requires the sandbox image to grant the command root (Vercel's runtimes
16
+ * do — passwordless). */
18
17
  sudo?: boolean;
19
18
  }
20
19
 
21
20
  export interface SandboxCommandResult {
22
21
  exitCode: number;
23
22
  stdout: string;
23
+ stderr: string;
24
24
  }
25
25
 
26
26
  export interface SandboxProvider {
@@ -28,6 +28,10 @@ export interface SandboxProvider {
28
28
  /** Working directory for the agent process. Set by onStart after environment setup. */
29
29
  cwd?: string;
30
30
  commands: {
31
+ // NOTE: does NOT throw on non-zero exit. Callers must check
32
+ // `result.exitCode` themselves. The `agent-env` setup workflow's
33
+ // `run(sb, cmd)` helper is the canonical pattern — copy it into any
34
+ // setup workflow that needs to fail loudly on command errors.
31
35
  run(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxCommandResult>;
32
36
  };
33
37
  files: {