@agent-compose/sdk 0.5.8 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +14 -12
  3. package/dist/agent/pause-client.d.ts +50 -0
  4. package/dist/agent/pause-client.test.d.ts +1 -0
  5. package/dist/agent/steer-control.d.ts +22 -6
  6. package/dist/client.d.ts +12 -1
  7. package/dist/index.d.ts +7 -5
  8. package/dist/index.js +2379 -1463
  9. package/dist/pause/checkpoint.d.ts +27 -10
  10. package/dist/pause/manager.d.ts +1 -0
  11. package/dist/pause/pause-core.d.ts +23 -0
  12. package/dist/pause/state-dir.d.ts +1 -1
  13. package/dist/processors/builtins.d.ts +20 -1
  14. package/dist/processors/gate-pause.d.ts +46 -0
  15. package/dist/processors/gate-pause.test.d.ts +1 -0
  16. package/dist/processors/index.d.ts +3 -1
  17. package/dist/processors/processor.d.ts +13 -0
  18. package/dist/runtimes/_acp-client.d.ts +140 -0
  19. package/dist/runtimes/_cli-agent.d.ts +155 -3
  20. package/dist/runtimes/amp.d.ts +2 -2
  21. package/dist/runtimes/cli-agent-acp-live.test.d.ts +30 -0
  22. package/dist/runtimes/cli-agent.test.d.ts +22 -6
  23. package/dist/runtimes/codex.d.ts +7 -2
  24. package/dist/runtimes/openai-desktop.js +2365 -1463
  25. package/dist/runtimes/vercel.js +389 -2
  26. package/dist/sandbox.d.ts +113 -19
  27. package/dist/step-invocation/__tests__/background-invoker.test.d.ts +1 -0
  28. package/dist/step-invocation/index.d.ts +2 -1
  29. package/dist/step-invocation/invoker.d.ts +36 -0
  30. package/dist/types/__tests__/environment-build-flag.test.d.ts +1 -0
  31. package/dist/types/__tests__/workflow-metadata-provider.test.d.ts +1 -0
  32. package/dist/types/execution-context.d.ts +0 -8
  33. package/dist/types/protocol.d.ts +32 -1
  34. package/dist/types/runtime.d.ts +14 -0
  35. package/dist/types/sandbox-environment.d.ts +6 -1
  36. package/dist/types/sandbox.d.ts +86 -6
  37. package/dist/types/workflow-metadata.d.ts +40 -10
  38. package/dist/types/workflow.d.ts +22 -6
  39. package/dist/utils/bundler.d.ts +5 -1
  40. package/dist/workflow-steps/observability.d.ts +28 -2
  41. package/dist/workflow-steps/types.d.ts +11 -7
  42. package/dist/workflow-steps/workflow.d.ts +3 -2
  43. package/package.json +3 -2
  44. package/src/agent/agent-context.ts +14 -6
  45. package/src/agent/agent-loop.ts +32 -10
  46. package/src/agent/pause-client.ts +108 -0
  47. package/src/agent/run-agent.ts +9 -4
  48. package/src/agent/steer-control.ts +21 -7
  49. package/src/client.ts +35 -1
  50. package/src/index.ts +20 -2
  51. package/src/pause/checkpoint.ts +33 -14
  52. package/src/pause/manager.ts +2 -2
  53. package/src/pause/pause-core.ts +35 -0
  54. package/src/pause/state-dir.ts +2 -2
  55. package/src/processors/builtins.ts +44 -1
  56. package/src/processors/gate-pause.ts +94 -0
  57. package/src/processors/index.ts +7 -0
  58. package/src/processors/processor.ts +13 -0
  59. package/src/runtimes/_acp-client.ts +516 -0
  60. package/src/runtimes/_cli-agent.ts +416 -3
  61. package/src/runtimes/claude.ts +31 -3
  62. package/src/runtimes/codex.ts +21 -1
  63. package/src/runtimes/vercel.ts +4 -1
  64. package/src/sandbox.ts +426 -56
  65. package/src/step-invocation/index.ts +2 -1
  66. package/src/step-invocation/invoker.ts +195 -84
  67. package/src/types/execution-context.ts +0 -8
  68. package/src/types/protocol.ts +27 -1
  69. package/src/types/runtime.ts +14 -0
  70. package/src/types/sandbox-environment.ts +12 -1
  71. package/src/types/sandbox.ts +84 -6
  72. package/src/types/workflow-metadata.ts +42 -10
  73. package/src/types/workflow.ts +22 -7
  74. package/src/utils/bundler.ts +6 -1
  75. package/src/workflow-steps/observability.ts +51 -5
  76. package/src/workflow-steps/runner.ts +9 -5
  77. package/src/workflow-steps/types.ts +11 -7
  78. package/src/workflow-steps/workflow.ts +3 -2
@@ -24,7 +24,7 @@
24
24
  import { randomBytes } from "node:crypto";
25
25
  import { z } from "zod";
26
26
  import { SandboxUnavailableError } from "../sandbox-errors.js";
27
- import type { SandboxProvider } from "../types/sandbox.js";
27
+ import type { SandboxProvider, SandboxBackgroundProcess } from "../types/sandbox.js";
28
28
  import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath } from "./protocol.js";
29
29
  import { StepPauseRequestSchema } from "./types.js";
30
30
  import type { StepRequest, StepResult } from "./types.js";
@@ -196,40 +196,34 @@ export interface InvokeStepOptions {
196
196
  onStderr?: (line: string) => void;
197
197
  }
198
198
 
199
- export async function invokeStep<TOutput = unknown>(
200
- sandbox: SandboxProvider,
201
- request: StepRequest,
202
- opts?: InvokeStepOptions,
203
- ): Promise<StepResult<TOutput>> {
204
- const resultToken = randomBytes(16).toString("hex");
205
-
206
- await Promise.all([
207
- sandbox.files.write(stepInputPath(request.stepIndex), JSON.stringify(request.input)),
208
- sandbox.files.write(requestContextPath(request.stepIndex), JSON.stringify(request.requestContext)),
209
- ]);
210
-
211
- const envs = {
212
- ...(opts?.envs ?? {}),
213
- ...buildStepEnvs({
214
- runId: request.runId,
215
- stepIndex: request.stepIndex,
216
- resultToken,
217
- isResume: opts?.isResume,
218
- }),
219
- };
199
+ /** A step running as a background command (ADR-0028). The activity races its
200
+ * `wait()` against a server pause request; on a pause it freezes the VM
201
+ * (`pauseProcess`) and persists `runnerPid` + `resultToken` so the resume
202
+ * activity can `reconnectStep(...)` to the SAME process — no re-run. */
203
+ export interface RunningStep<TOutput = unknown> {
204
+ /** OS pid of the background runner inside the VM — the reconnect handle. */
205
+ runnerPid: number;
206
+ /** Per-invocation token keying the durable result file the runner writes. */
207
+ resultToken: string;
208
+ /** Await the runner's exit and classify its output into a `StepResult`. */
209
+ wait(): Promise<StepResult<TOutput>>;
210
+ }
220
211
 
221
- // The sandbox emits stdout as raw chunks, not lines. Buffer between
222
- // emissions so a `console.log` split across two chunks (or a partial
223
- // trailing line) is delivered to onStdout/onStderr as one logical line.
224
- // Sentinel lines (carrying `resultToken`) are filtered out so callers
225
- // never see protocol bytes in user-log capture. Both the result and
226
- // pause prefixes are built from the same shared helpers `parseStepResult`
227
- // uses, so the filter and the parser can't drift apart.
212
+ /** Buffer the sandbox's raw stdout/stderr chunks into whole lines, filtering
213
+ * the protocol sentinel out of `onStdout` so user-log capture never sees
214
+ * protocol bytes. Shared by the foreground (`invokeStep`), background
215
+ * (`launchStep`), and resume (`reconnectStep`) paths so the filter and the
216
+ * result parser can't drift. Returns the per-stream chunk handlers (undefined
217
+ * when the caller passed no sink) plus a `flush` for trailing newline-less
218
+ * output. */
219
+ function makeStreamSplitters(
220
+ opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
221
+ resultToken: string,
222
+ ): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void } {
228
223
  const resultSentinel = stepResultLinePrefix(resultToken);
229
224
  const pauseSentinel = stepPauseLinePrefix(resultToken);
230
- const isSentinel = (line: string) =>
231
- line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
232
- const makeLineSplitter = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
225
+ const isSentinel = (line: string) => line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
226
+ const make = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
233
227
  if (!sink) return { onChunk: undefined, flush: () => {} };
234
228
  let buf = "";
235
229
  return {
@@ -243,10 +237,6 @@ export async function invokeStep<TOutput = unknown>(
243
237
  sink(line);
244
238
  }
245
239
  },
246
- // Drain any remaining buffered output that ended without a newline.
247
- // Called after `sandbox.commands.run` resolves so a runner that
248
- // exits with `process.stdout.write("final")` (no trailing \n)
249
- // doesn't silently drop its last line.
250
240
  flush: () => {
251
241
  if (buf.length === 0) return;
252
242
  const line = buf;
@@ -256,16 +246,102 @@ export async function invokeStep<TOutput = unknown>(
256
246
  },
257
247
  };
258
248
  };
259
- const stdoutSplitter = makeLineSplitter(opts?.onStdout, true);
260
- const stderrSplitter = makeLineSplitter(opts?.onStderr, false);
249
+ const stdout = make(opts?.onStdout, true);
250
+ const stderr = make(opts?.onStderr, false);
251
+ return {
252
+ ...(stdout.onChunk ? { onStdout: stdout.onChunk } : {}),
253
+ ...(stderr.onChunk ? { onStderr: stderr.onChunk } : {}),
254
+ flush: () => { stdout.flush(); stderr.flush(); },
255
+ };
256
+ }
257
+
258
+ /** Turn the runner's terminal output into a kinded `StepResult`. stdout is the
259
+ * fast path; the runner also persists the sentinel to a durable token-keyed
260
+ * file, so when stdout carries no sentinel (providers drop the tail of heavy
261
+ * streams; a reconnect captures only post-resume output) we read the file
262
+ * back. Shared by all three invocation paths. */
263
+ async function classifyRunnerOutcome<TOutput>(
264
+ sandbox: SandboxProvider,
265
+ output: { stdout: string; stderr: string; exitCode: number },
266
+ resultToken: string,
267
+ ): Promise<StepResult<TOutput>> {
268
+ const parsed = parseStepResult<TOutput>(output.stdout, resultToken);
269
+ if (parsed) return parsed;
270
+
271
+ // No sentinel on stdout — read the durable token-keyed file the runner wrote
272
+ // BEFORE its stdout emit. A runner that exited 1 after a user-step throw
273
+ // still wrote `{ok:false, kind:"user-step"}`, which beats a generic exit.
274
+ const fileRead = await sandbox.commands.run(
275
+ `cat ${stepResultFilePath(resultToken)} 2>/dev/null || true`,
276
+ ).catch(() => null);
277
+ if (fileRead?.stdout) {
278
+ const fromFile = parseStepResult<TOutput>(fileRead.stdout, resultToken);
279
+ if (fromFile) return fromFile;
280
+ }
281
+
282
+ // Distinguish "runner exited badly before emitting" (runner-exit) from
283
+ // "runner exited cleanly but didn't speak the protocol" (protocol).
284
+ if (output.exitCode !== 0) {
285
+ const stderrTail = output.stderr.trim().slice(-2000);
286
+ const stdoutTail = output.stdout.trim().slice(-2000);
287
+ const details = [
288
+ stderrTail ? `stderr:\n${stderrTail}` : "",
289
+ stdoutTail ? `stdout:\n${stdoutTail}` : "",
290
+ ].filter(Boolean).join("\n");
291
+ return {
292
+ ok: false,
293
+ error: {
294
+ kind: "runner-exit",
295
+ message: `runner subprocess exited ${output.exitCode} before emitting a step result${details ? `\n${details}` : ""}`,
296
+ exitCode: output.exitCode,
297
+ },
298
+ };
299
+ }
300
+ const tail = output.stdout.trim().slice(-500);
301
+ return {
302
+ ok: false,
303
+ error: {
304
+ kind: "protocol",
305
+ message: `no tokenised step result on stdout or in the result file (runner exited 0 without emitting)${tail ? `\nstdout tail:\n${tail}` : ""}`,
306
+ },
307
+ };
308
+ }
309
+
310
+ /** Write the step's input + request-context files and build the runner env.
311
+ * Shared by the foreground and background launch paths. Returns the
312
+ * per-invocation `resultToken` + the env map. */
313
+ async function prepareStepLaunch(
314
+ sandbox: SandboxProvider,
315
+ request: StepRequest,
316
+ opts?: InvokeStepOptions,
317
+ ): Promise<{ resultToken: string; envs: Record<string, string> }> {
318
+ const resultToken = randomBytes(16).toString("hex");
319
+ await Promise.all([
320
+ sandbox.files.write(stepInputPath(request.stepIndex), JSON.stringify(request.input)),
321
+ sandbox.files.write(requestContextPath(request.stepIndex), JSON.stringify(request.requestContext)),
322
+ ]);
323
+ const envs = {
324
+ ...(opts?.envs ?? {}),
325
+ ...buildStepEnvs({ runId: request.runId, stepIndex: request.stepIndex, resultToken, isResume: opts?.isResume }),
326
+ };
327
+ return { resultToken, envs };
328
+ }
329
+
330
+ export async function invokeStep<TOutput = unknown>(
331
+ sandbox: SandboxProvider,
332
+ request: StepRequest,
333
+ opts?: InvokeStepOptions,
334
+ ): Promise<StepResult<TOutput>> {
335
+ const { resultToken, envs } = await prepareStepLaunch(sandbox, request, opts);
336
+ const splitters = makeStreamSplitters(opts, resultToken);
261
337
 
262
338
  let result: { stdout: string; stderr: string; exitCode: number };
263
339
  try {
264
340
  result = await sandbox.commands.run(RUNNER_COMMAND, {
265
341
  envs,
266
342
  timeoutMs: 0,
267
- ...(stdoutSplitter.onChunk ? { onStdout: stdoutSplitter.onChunk } : {}),
268
- ...(stderrSplitter.onChunk ? { onStderr: stderrSplitter.onChunk } : {}),
343
+ ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
344
+ ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
269
345
  });
270
346
  } catch (e) {
271
347
  // Typed sandbox-infrastructure failure: the engine's recovery contract
@@ -286,55 +362,90 @@ export async function invokeStep<TOutput = unknown>(
286
362
  exitCode: ce.exitCode ?? ce.result?.exitCode ?? 1,
287
363
  };
288
364
  }
289
- stdoutSplitter.flush();
290
- stderrSplitter.flush();
291
-
292
- const parsed = parseStepResult<TOutput>(result.stdout, resultToken);
293
- if (parsed) return parsed;
365
+ splitters.flush();
366
+ return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
367
+ }
294
368
 
295
- // No sentinel on stdout. The runner also persists the sentinel line to a
296
- // token-keyed file before emitting — providers drop the tail of heavy
297
- // stdout streams, so read the durable copy back. Tried before exit-code
298
- // classification: a runner that exited 1 after a user-step throw still
299
- // wrote the file, and its `{ok:false, kind:"user-step"}` beats a generic
300
- // runner-exit error. A runner killed before emitting wrote no file and
301
- // falls through.
302
- const fileRead = await sandbox.commands.run(
303
- `cat ${stepResultFilePath(resultToken)} 2>/dev/null || true`,
304
- ).catch(() => null);
305
- if (fileRead?.stdout) {
306
- const fromFile = parseStepResult<TOutput>(fileRead.stdout, resultToken);
307
- if (fromFile) return fromFile;
369
+ /**
370
+ * Launch the step runner as a BACKGROUND command (ADR-0028) and return a handle
371
+ * the activity drives: it races `wait()` against a server pause request and, on
372
+ * a pause, freezes the VM (`pauseProcess`) and persists `runnerPid` +
373
+ * `resultToken` so `reconnectStep` can continue the SAME process — no re-run.
374
+ * Requires a provider with `commands.runBackground` (E2B); the foreground
375
+ * `invokeStep` is the path for providers without it (Vercel).
376
+ */
377
+ export async function launchStep<TOutput = unknown>(
378
+ sandbox: SandboxProvider,
379
+ request: StepRequest,
380
+ opts?: InvokeStepOptions,
381
+ ): Promise<RunningStep<TOutput>> {
382
+ if (!sandbox.commands.runBackground) {
383
+ throw new Error("launchStep requires a provider with background-command support (commands.runBackground)");
308
384
  }
385
+ const { resultToken, envs } = await prepareStepLaunch(sandbox, request, opts);
386
+ const splitters = makeStreamSplitters(opts, resultToken);
387
+ const proc = await sandbox.commands.runBackground(RUNNER_COMMAND, {
388
+ envs,
389
+ timeoutMs: 0,
390
+ ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
391
+ ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
392
+ });
393
+ return {
394
+ runnerPid: proc.pid,
395
+ resultToken,
396
+ async wait() {
397
+ const result = await proc.wait();
398
+ splitters.flush();
399
+ return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
400
+ },
401
+ };
402
+ }
309
403
 
310
- // No tokenised sentinel on stdout. Distinguish "runner exited badly
311
- // before emitting" (runner-exit) from "runner exited cleanly but didn't
312
- // speak the protocol" (protocol) — the exit code is the evidence.
313
- if (result.exitCode !== 0) {
314
- const stderrTail = result.stderr.trim().slice(-2000);
315
- const stdoutTail = result.stdout.trim().slice(-2000);
316
- const details = [
317
- stderrTail ? `stderr:\n${stderrTail}` : "",
318
- stdoutTail ? `stdout:\n${stdoutTail}` : "",
319
- ].filter(Boolean).join("\n");
320
- return {
321
- ok: false,
322
- error: {
323
- kind: "runner-exit",
324
- message: `runner subprocess exited ${result.exitCode} before emitting a step result${details ? `\n${details}` : ""}`,
325
- exitCode: result.exitCode,
326
- },
327
- };
404
+ /**
405
+ * Resume a previously-paused background runner (ADR-0028) and return a
406
+ * `RunningStep` handle — uniform with `launchStep` so the activity can race
407
+ * `wait()` against a fresh pause request (a resumed step can pause again). After
408
+ * the workflow reconnects the suspended VM (`Sandbox.connect` auto-resumes it),
409
+ * this re-attaches to the still-running runner by `runnerPid`; `wait()` awaits
410
+ * its exit (event-driven — no polling) and classifies the output (the durable
411
+ * result file is authoritative on this path). If the runner already exited
412
+ * during resume, the re-attach fails and `wait()` classifies from the file.
413
+ */
414
+ export async function reconnectStep<TOutput = unknown>(
415
+ sandbox: SandboxProvider,
416
+ resume: { runnerPid: number; resultToken: string; stepIndex: number },
417
+ opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr">,
418
+ ): Promise<RunningStep<TOutput>> {
419
+ if (!sandbox.commands.connectProcess) {
420
+ throw new Error("reconnectStep requires a provider with background-command support (commands.connectProcess)");
421
+ }
422
+ const { runnerPid, resultToken } = resume;
423
+ const splitters = makeStreamSplitters(opts, resultToken);
424
+ let proc: SandboxBackgroundProcess | null = null;
425
+ try {
426
+ proc = await sandbox.commands.connectProcess(runnerPid, {
427
+ ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
428
+ ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
429
+ });
430
+ } catch (e) {
431
+ // A sandbox-infrastructure failure must propagate intact (recovery
432
+ // contract). Anything else means the runner already exited during resume
433
+ // (we re-attached too late) — its durable result file is written, so the
434
+ // handle's wait() classifies from the file below.
435
+ if (e instanceof SandboxUnavailableError) throw e;
328
436
  }
329
- // Carry the stdout tail: "no tokenised step result" alone is useless to
330
- // an operator — the tail usually shows whether the runner finished its
331
- // work (output truncated by the provider) or never got there.
332
- const tail = result.stdout.trim().slice(-500);
437
+ const attached = proc;
333
438
  return {
334
- ok: false,
335
- error: {
336
- kind: "protocol",
337
- message: `no tokenised step result on stdout or in the result file (runner exited 0 without emitting)${tail ? `\nstdout tail:\n${tail}` : ""}`,
439
+ runnerPid,
440
+ resultToken,
441
+ async wait() {
442
+ let output: { stdout: string; stderr: string; exitCode: number } = { stdout: "", stderr: "", exitCode: 0 };
443
+ if (attached) {
444
+ try { output = await attached.wait(); }
445
+ catch (e) { if (e instanceof SandboxUnavailableError) throw e; }
446
+ }
447
+ splitters.flush();
448
+ return classifyRunnerOutcome<TOutput>(sandbox, output, resultToken);
338
449
  },
339
450
  };
340
451
  }
@@ -29,14 +29,6 @@ export interface BaseExecutionContext {
29
29
  setMetadata?: (data: Record<string, unknown>) => Promise<void>;
30
30
  /** Invoke another registered workflow and wait for it to settle. */
31
31
  invokeChild: InvokeChild;
32
- /**
33
- * Disk-backed memoise across pause-resume. First call runs `fn` and
34
- * atomically writes the result to the sandbox; on resume the recorded
35
- * value is returned and `fn` is NOT re-executed. Use for expensive
36
- * deterministic transforms; for side effects, use `invokeChild`.
37
- * See ADR-0006 §"`ctx.checkpoint(name, fn)` — disk-backed memoisation".
38
- */
39
- checkpoint<T>(name: string, fn: () => Promise<T> | T): Promise<T>;
40
32
  /**
41
33
  * Pause for feedback. The step exits and the workflow waits durably until
42
34
  * something resolves the pause (a resume call, a TTL expiry); on resume the
@@ -33,6 +33,16 @@ export interface AgentMessageToolResult extends AgentMessageBase {
33
33
  toolUseId: string;
34
34
  output: string;
35
35
  isError: boolean;
36
+ /** Structured file diffs produced by the tool call, carried alongside the
37
+ * rendered `output` so a richer renderer can use the structure without
38
+ * re-plumbing the normaliser. Sourced from ACP `tool_call`/`tool_call_update`
39
+ * `diff` content blocks (WS-C / ADR-0020 Q3). Optional and additive:
40
+ * existing producers (claude / vercel / legacy JSONL) omit it. `oldText` is
41
+ * null for a newly created file. */
42
+ diffs?: { path: string; oldText: string | null; newText: string }[];
43
+ /** File locations touched by the tool call (ACP `locations` field), enabling
44
+ * "follow-along" UI. Optional and additive (WS-C / ADR-0020 Q3). */
45
+ locations?: { path: string; line?: number }[];
36
46
  }
37
47
 
38
48
  export interface AgentMessageDone extends AgentMessageBase {
@@ -55,6 +65,21 @@ export interface AgentMessageUsage extends AgentMessageBase {
55
65
  numTurns: number;
56
66
  }
57
67
 
68
+ /** Structured execution plan emitted by an agent (ACP `plan` session update,
69
+ * WS-C / ADR-0020 Q2). Each `plan` notification REPLACES the whole plan — the
70
+ * normaliser emits one `AgentMessagePlan` per notification carrying the entire
71
+ * `entries` array, and downstream treats the latest as authoritative. This is
72
+ * NOT an `AgentStatus` and does not feed self-pause; it is observability only.
73
+ * Additive 9th kind: existing producers never emit it. */
74
+ export interface AgentMessagePlan extends AgentMessageBase {
75
+ type: "plan";
76
+ entries: {
77
+ content: string;
78
+ priority: "high" | "medium" | "low";
79
+ status: "pending" | "in_progress" | "completed";
80
+ }[];
81
+ }
82
+
58
83
  export type AgentMessage =
59
84
  | AgentMessageInit
60
85
  | AgentMessageText
@@ -63,7 +88,8 @@ export type AgentMessage =
63
88
  | AgentMessageToolResult
64
89
  | AgentMessageDone
65
90
  | AgentMessageError
66
- | AgentMessageUsage;
91
+ | AgentMessageUsage
92
+ | AgentMessagePlan;
67
93
 
68
94
  /** Status block the agent emits to signal iteration completion or blockers. */
69
95
  export interface AgentStatus {
@@ -5,6 +5,7 @@
5
5
  import type { SandboxProvider } from "./sandbox.js";
6
6
  import type { AgentMessage } from "./protocol.js";
7
7
  import type { Processor, ProcessorContext, ToolCall } from "../processors/processor.js";
8
+ import type { BoundaryPauseFn } from "../pause/pause-core.js";
8
9
  import type { RequestContext } from "../request-context/request-context.js";
9
10
 
10
11
  /** Configuration for a single MCP server. */
@@ -31,6 +32,19 @@ export interface RuntimeOptions {
31
32
  /** Agent id and label for processor context / adapter logs. */
32
33
  agentId?: string;
33
34
  iteration?: number;
35
+ /** The platform manual (file conventions, connectors & access, how to pause).
36
+ * `agent()` builds it per-run (`buildAgentContextDoc`) and threads it here so
37
+ * a runtime that supports a system-prompt append (the claude runtime) injects
38
+ * it directly — instead of relying on the agent to `cat` the on-disk
39
+ * AGENTS.md/CLAUDE.md, which the Agent SDK doesn't auto-load and which can
40
+ * fail to write on a read-only/degraded working dir. */
41
+ agentManual?: string;
42
+ /** The run's pause boundary, threaded from the agent loop so a runtime-driven
43
+ * pre-tool gate (e.g. the ACP `session/request_permission` path through
44
+ * `gateToolCall`) can raise a human-approval `ctx.pause`. The runtime binds it
45
+ * to the current `{ agentId, iteration }` when it builds a ProcessorContext.
46
+ * Absent ⇒ a processor pause throws (no boundary; see `boundProcessorPause`). */
47
+ pause?: BoundaryPauseFn;
34
48
  /** Optional JSON schema for runtimes with native structured-output support. */
35
49
  outputFormat?: { type: "json_schema"; schema: Record<string, unknown> };
36
50
  }
@@ -37,7 +37,7 @@
37
37
  import type { SandboxProvider } from "./sandbox.js";
38
38
  import { defineWorkflow } from "./workflow.js";
39
39
  import type { Workflow } from "../workflow-steps/types.js";
40
- import type { SnapshotConfig } from "./workflow-metadata.js";
40
+ import type { SandboxResources, SnapshotConfig } from "./workflow-metadata.js";
41
41
 
42
42
  export interface SandboxEnvironmentDefinition {
43
43
  name: string;
@@ -48,6 +48,11 @@ export interface SandboxEnvironmentDefinition {
48
48
  * useful — an env with no snapshot can't be referenced as a
49
49
  * `bootFrom` on another workflow). */
50
50
  snapshots?: SnapshotConfig;
51
+ /** Which substrate to build this environment on (and any sizing).
52
+ * An env image is provider-specific — a snapshot captured on E2B
53
+ * can't boot on Vercel and vice-versa — so building the E2B base/
54
+ * agent-env requires `resources: { provider: "e2b" }`. */
55
+ resources?: SandboxResources;
51
56
  }
52
57
 
53
58
  /** Sugar over `defineWorkflow` for setup-only workflows that exist to
@@ -61,7 +66,13 @@ export function defineSandboxEnvironment(
61
66
  }
62
67
  return defineWorkflow<void, Record<string, unknown>>({
63
68
  ...(env.description !== undefined ? { description: env.description } : {}),
69
+ ...(env.resources !== undefined ? { resources: env.resources } : {}),
64
70
  snapshots: env.snapshots ?? { saveLatest: true },
71
+ // Mark this as an environment build so the server skips the /factory mount
72
+ // for its runs — an env build builds a platform image and never touches the
73
+ // shared drive; baking a live Archil mount into its snapshot breaks the
74
+ // re-mount of every workflow that later boots from it (#13).
75
+ environmentBuild: true,
65
76
  run: async (_ctx, sandbox) => env.setup(sandbox),
66
77
  });
67
78
  }
@@ -25,6 +25,59 @@ export interface SandboxCommandResult {
25
25
  stderr: string;
26
26
  }
27
27
 
28
+ /** A long-running command launched in the background (ADR-0028). Unlike
29
+ * `commands.run` (which awaits completion on one connection), a background
30
+ * command keeps running inside the VM independent of the launching
31
+ * connection: it survives `pauseProcess()`/resume and is re-attachable by
32
+ * `pid` after a fresh `Sandbox.connect`. This is what lets the platform
33
+ * freeze an agent mid-turn for a human-in-the-loop pause and continue the
34
+ * SAME process on resume — no re-run.
35
+ *
36
+ * Implemented ONLY by process-resume-capable providers (E2B); the presence
37
+ * of `commands.runBackground` IS the capability flag, paired with
38
+ * `pauseProcess`. Providers without it leave both undefined and pause via
39
+ * `snapshot()` + re-run instead. */
40
+ export interface SandboxBackgroundProcess {
41
+ /** OS pid inside the VM — the durable handle used to reconnect after a
42
+ * pause/resume cycle via `commands.connectProcess(pid)`. */
43
+ pid: number;
44
+ /** Resolve when the process exits, with its buffered result. Live output
45
+ * streams to the `onStdout`/`onStderr` passed at launch / connect time.
46
+ * Does NOT throw on a non-zero exit — the result carries `exitCode`. */
47
+ wait(): Promise<SandboxCommandResult>;
48
+ /** Force-terminate the process. */
49
+ kill(): Promise<void>;
50
+ }
51
+
52
+ /** A spawned long-lived command with a writable stdin and readable stdout,
53
+ * exposed as byte web-streams. Unlike `commands.run` (which buffers to
54
+ * completion and exposes stdout only via an `onStdout` callback), a duplex
55
+ * handle keeps the process alive and lets the caller WRITE to its stdin —
56
+ * the half `commands.run` cannot provide. It is the transport the ACP client
57
+ * (`AcpClientPeer` over `ndJsonStream`) needs: the agent CLI reads JSON-RPC
58
+ * request frames on stdin and answers on stdout.
59
+ *
60
+ * Only `makeLocalSandboxProvider` implements it. The runner runs IN the
61
+ * sandbox VM and spawns CLIs via the local provider (a plain `child_process`
62
+ * pipe), so a duplex stdin works identically on Vercel/E2B/local. The
63
+ * vercel/e2b providers are the SERVER→sandbox view and never spawn the in-VM
64
+ * CLI, so they leave `spawnDuplex` undefined and callers fall back cleanly. */
65
+ export interface SandboxDuplexProcess {
66
+ /** Subprocess stdin. JSON-RPC request frames are written here. */
67
+ stdin: WritableStream<Uint8Array>;
68
+ /** Subprocess stdout. JSON-RPC response/notification frames arrive here. */
69
+ stdout: ReadableStream<Uint8Array>;
70
+ /** Resolves when the subprocess exits, carrying the captured stderr tail. */
71
+ exited: Promise<{ exitCode: number; stderr: string }>;
72
+ /** Force-terminate the subprocess. */
73
+ kill(): void;
74
+ }
75
+
76
+ export interface SandboxSpawnDuplexOptions {
77
+ cwd?: string;
78
+ envs?: Record<string, string>;
79
+ }
80
+
28
81
  /**
29
82
  * A sandbox provider implements the RAW provider operations only. It does NOT
30
83
  * implement transient-failure retry/backoff: reconnecting and snapshotting both
@@ -51,6 +104,21 @@ export interface SandboxProvider {
51
104
  // `run(sb, cmd)` helper is the canonical pattern — copy it into any
52
105
  // setup workflow that needs to fail loudly on command errors.
53
106
  run(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxCommandResult>;
107
+ /** Spawn a long-lived command with a real duplex stdin/stdout. OPTIONAL —
108
+ * implemented only by `makeLocalSandboxProvider` (the in-VM `child_process`
109
+ * view). The vercel/e2b providers (server→sandbox) leave it undefined; an
110
+ * ACP caller that finds it absent falls back to the JSONL transport. */
111
+ spawnDuplex?(cmd: string, opts?: SandboxSpawnDuplexOptions): SandboxDuplexProcess;
112
+ /** Launch a command in the background and return immediately with a
113
+ * reconnectable handle (ADR-0028). The process survives the launching
114
+ * connection dropping AND a `pauseProcess()`/resume cycle. OPTIONAL —
115
+ * only process-resume providers (E2B) implement it; its presence (paired
116
+ * with `pauseProcess`) is the native-pause capability flag. */
117
+ runBackground?(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxBackgroundProcess>;
118
+ /** Re-attach to a background command by `pid` after a fresh
119
+ * `Sandbox.connect` (the resume half of `runBackground`). OPTIONAL,
120
+ * E2B-only. Throws if no process with that pid is running. */
121
+ connectProcess?(pid: number, opts?: Pick<SandboxCommandRunOptions, "onStdout" | "onStderr" | "timeoutMs">): Promise<SandboxBackgroundProcess>;
54
122
  };
55
123
  files: {
56
124
  write(path: string, content: string): Promise<void>;
@@ -65,12 +133,22 @@ export interface SandboxProvider {
65
133
  * be omitted when the provider doesn't expose it; the server stores
66
134
  * `null` for missing values rather than estimating. */
67
135
  snapshot?(): Promise<{ snapshotId: string; sizeBytes?: number }>;
68
- /** Replace the live sandbox's egress policy in place. Vercel implements
69
- * it via `sandbox.update({ networkPolicy })` (2.x) so the server can
70
- * push a freshly resolved policy — with re-minted connector access
71
- * tokens — before each step instead of relying on the policy baked at
72
- * create. Providers whose enforcement lives inside the VM (E2B
73
- * iron-proxy) leave it undefined. */
136
+ /** Suspend the live VM in place and return a handle to resume it (ADR-0027).
137
+ * Present ONLY on process-resume-capable providers (E2B via `sandbox.pause()`,
138
+ * returning the sandbox id; resume is `Sandbox.connect(handle)`, which
139
+ * auto-resumes the paused VM). Unlike `snapshot()` — which captures an FS
140
+ * image, kills the origin, and re-runs the step from a fresh sandbox — a
141
+ * process-resume pause FREEZES the live process (zero compute) and continues
142
+ * it exactly where it blocked. The presence of this method IS the capability
143
+ * flag: providers without native VM-suspend leave it undefined and fall back
144
+ * to `snapshot()` + re-run. */
145
+ pauseProcess?(): Promise<{ resumeHandle: string }>;
146
+ /** Replace the live sandbox's egress policy in place — so the server can
147
+ * push a freshly resolved policy (with re-minted connector access tokens)
148
+ * before each step instead of relying on the policy baked at create.
149
+ * Vercel implements it via `sandbox.update({ networkPolicy })` (2.x); E2B
150
+ * via its native `sandbox.updateNetwork(...)`. Providers without a live
151
+ * network-update primitive leave it undefined. */
74
152
  updateNetworkPolicy?(policy: SandboxNetworkPolicy): Promise<void>;
75
153
  }
76
154