@agent-compose/sdk 0.5.9 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +0 -8
  3. package/dist/agent/pause-client.d.ts +50 -0
  4. package/dist/index.d.ts +15 -6
  5. package/dist/index.js +537 -124
  6. package/dist/processors/ask-human.d.ts +30 -0
  7. package/dist/processors/ask-human.test.d.ts +1 -0
  8. package/dist/processors/gate-pause.d.ts +46 -0
  9. package/dist/processors/gate-pause.test.d.ts +1 -0
  10. package/dist/processors/index.d.ts +3 -0
  11. package/dist/runtimes/_cli-agent.d.ts +9 -0
  12. package/dist/runtimes/cursor.d.ts +9 -0
  13. package/dist/runtimes/droid.d.ts +9 -0
  14. package/dist/runtimes/openai-desktop.js +522 -122
  15. package/dist/runtimes/opencode.d.ts +25 -0
  16. package/dist/runtimes/vercel.js +11 -1
  17. package/dist/step-invocation/__tests__/background-invoker.test.d.ts +1 -0
  18. package/dist/step-invocation/index.d.ts +2 -1
  19. package/dist/step-invocation/invoker.d.ts +49 -0
  20. package/dist/step-invocation/protocol.d.ts +8 -0
  21. package/dist/types/runtime.d.ts +7 -0
  22. package/dist/types/sandbox.d.ts +54 -0
  23. package/dist/types/workflow.d.ts +12 -0
  24. package/dist/utils/errors.d.ts +9 -1
  25. package/package.json +1 -1
  26. package/src/agent/agent-context.ts +23 -16
  27. package/src/agent/agent-loop.ts +13 -33
  28. package/src/agent/pause-client.ts +108 -0
  29. package/src/agent/run-agent.ts +16 -7
  30. package/src/index.ts +27 -4
  31. package/src/processors/ask-human.ts +136 -0
  32. package/src/processors/gate-pause.ts +94 -0
  33. package/src/processors/index.ts +11 -0
  34. package/src/runtimes/_cli-agent.ts +13 -5
  35. package/src/runtimes/claude.ts +10 -6
  36. package/src/runtimes/cursor.ts +59 -0
  37. package/src/runtimes/droid.ts +63 -0
  38. package/src/runtimes/opencode.ts +61 -0
  39. package/src/sandbox.ts +78 -3
  40. package/src/step-invocation/index.ts +2 -1
  41. package/src/step-invocation/invoker.ts +359 -86
  42. package/src/step-invocation/protocol.ts +11 -0
  43. package/src/types/runtime.ts +7 -0
  44. package/src/types/sandbox.ts +53 -0
  45. package/src/types/workflow.ts +12 -0
  46. package/src/utils/errors.ts +19 -2
  47. package/dist/agent/local-pause-request.d.ts +0 -49
  48. package/src/agent/local-pause-request.ts +0 -90
  49. /package/dist/agent/{local-pause-request.test.d.ts → pause-client.test.d.ts} +0 -0
@@ -24,8 +24,8 @@
24
24
  import { randomBytes } from "node:crypto";
25
25
  import { z } from "zod";
26
26
  import { SandboxUnavailableError } from "../sandbox-errors.js";
27
- import type { SandboxProvider } from "../types/sandbox.js";
28
- import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath } from "./protocol.js";
27
+ import type { SandboxProvider, SandboxBackgroundProcess } from "../types/sandbox.js";
28
+ import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath, stepLogFilePath } from "./protocol.js";
29
29
  import { StepPauseRequestSchema } from "./types.js";
30
30
  import type { StepRequest, StepResult } from "./types.js";
31
31
  import type { StepObservability } from "../workflow-steps/observability.js";
@@ -194,44 +194,50 @@ export interface InvokeStepOptions {
194
194
  * the caller's job; the activity batches and inserts at step completion. */
195
195
  onStdout?: (line: string) => void;
196
196
  onStderr?: (line: string) => void;
197
+ /** Called when the LIVE output stream fails mid-run (e.g. E2B's connect-web
198
+ * transport throws `received unsupported compressed output` on a large
199
+ * compressed frame) and the invoker falls back to recovering the result +
200
+ * full logs from the durable token-keyed files. The run is NOT failed — this
201
+ * is the hook to emit a structured alert so the degradation is visible/paged.
202
+ * Best-effort: keep it cheap and non-throwing. */
203
+ onStreamDegraded?: (info: { error: unknown; runnerPid: number; resultToken: string }) => void;
197
204
  }
198
205
 
199
- export async function invokeStep<TOutput = unknown>(
200
- sandbox: SandboxProvider,
201
- request: StepRequest,
202
- opts?: InvokeStepOptions,
203
- ): Promise<StepResult<TOutput>> {
204
- const resultToken = randomBytes(16).toString("hex");
205
-
206
- await Promise.all([
207
- sandbox.files.write(stepInputPath(request.stepIndex), JSON.stringify(request.input)),
208
- sandbox.files.write(requestContextPath(request.stepIndex), JSON.stringify(request.requestContext)),
209
- ]);
210
-
211
- const envs = {
212
- ...(opts?.envs ?? {}),
213
- ...buildStepEnvs({
214
- runId: request.runId,
215
- stepIndex: request.stepIndex,
216
- resultToken,
217
- isResume: opts?.isResume,
218
- }),
219
- };
206
+ /** A step running as a background command (ADR-0028). The activity races its
207
+ * `wait()` against a server pause request; on a pause it freezes the VM
208
+ * (`pauseProcess`) and persists `runnerPid` + `resultToken` so the resume
209
+ * activity can `reconnectStep(...)` to the SAME process — no re-run. */
210
+ export interface RunningStep<TOutput = unknown> {
211
+ /** OS pid of the background runner inside the VM — the reconnect handle. */
212
+ runnerPid: number;
213
+ /** Per-invocation token keying the durable result file the runner writes. */
214
+ resultToken: string;
215
+ /** Await the runner's exit and classify its output into a `StepResult`. */
216
+ wait(): Promise<StepResult<TOutput>>;
217
+ }
220
218
 
221
- // The sandbox emits stdout as raw chunks, not lines. Buffer between
222
- // emissions so a `console.log` split across two chunks (or a partial
223
- // trailing line) is delivered to onStdout/onStderr as one logical line.
224
- // Sentinel lines (carrying `resultToken`) are filtered out so callers
225
- // never see protocol bytes in user-log capture. Both the result and
226
- // pause prefixes are built from the same shared helpers `parseStepResult`
227
- // uses, so the filter and the parser can't drift apart.
219
+ /** Buffer the sandbox's raw stdout/stderr chunks into whole lines, filtering
220
+ * the protocol sentinel out of `onStdout` so user-log capture never sees
221
+ * protocol bytes. Shared by the foreground (`invokeStep`), background
222
+ * (`launchStep`), and resume (`reconnectStep`) paths so the filter and the
223
+ * result parser can't drift. Returns the per-stream chunk handlers (undefined
224
+ * when the caller passed no sink) plus a `flush` for trailing newline-less
225
+ * output. */
226
+ function makeStreamSplitters(
227
+ opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
228
+ resultToken: string,
229
+ ): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void; stdoutConsumedChars: () => number } {
228
230
  const resultSentinel = stepResultLinePrefix(resultToken);
229
231
  const pauseSentinel = stepPauseLinePrefix(resultToken);
230
- const isSentinel = (line: string) =>
231
- line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
232
- const makeLineSplitter = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
233
- if (!sink) return { onChunk: undefined, flush: () => {} };
232
+ const isSentinel = (line: string) => line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
233
+ const make = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
234
+ if (!sink) return { onChunk: undefined, flush: () => {}, consumed: () => 0 };
234
235
  let buf = "";
236
+ // Chars consumed as WHOLE lines (incl. each trailing "\n"), counting sentinel
237
+ // lines too — this is the offset into the tee'd log file up to which the live
238
+ // stream has already delivered. The recovery backfill starts exactly here, so
239
+ // an in-flight partial line is re-read whole from the file (no split, no dup).
240
+ let consumed = 0;
235
241
  return {
236
242
  onChunk: (chunk: string) => {
237
243
  buf += chunk;
@@ -239,33 +245,220 @@ export async function invokeStep<TOutput = unknown>(
239
245
  while ((nl = buf.indexOf("\n")) !== -1) {
240
246
  const line = buf.slice(0, nl);
241
247
  buf = buf.slice(nl + 1);
248
+ consumed += line.length + 1;
242
249
  if (filterSentinel && isSentinel(line)) continue;
243
250
  sink(line);
244
251
  }
245
252
  },
246
- // Drain any remaining buffered output that ended without a newline.
247
- // Called after `sandbox.commands.run` resolves so a runner that
248
- // exits with `process.stdout.write("final")` (no trailing \n)
249
- // doesn't silently drop its last line.
250
253
  flush: () => {
251
254
  if (buf.length === 0) return;
252
255
  const line = buf;
253
256
  buf = "";
257
+ consumed += line.length;
254
258
  if (filterSentinel && isSentinel(line)) return;
255
259
  sink(line);
256
260
  },
261
+ consumed: () => consumed,
262
+ };
263
+ };
264
+ const stdout = make(opts?.onStdout, true);
265
+ const stderr = make(opts?.onStderr, false);
266
+ return {
267
+ ...(stdout.onChunk ? { onStdout: stdout.onChunk } : {}),
268
+ ...(stderr.onChunk ? { onStderr: stderr.onChunk } : {}),
269
+ flush: () => { stdout.flush(); stderr.flush(); },
270
+ stdoutConsumedChars: () => stdout.consumed(),
271
+ };
272
+ }
273
+
274
+ /** Turn the runner's terminal output into a kinded `StepResult`. stdout is the
275
+ * fast path; the runner also persists the sentinel to a durable token-keyed
276
+ * file, so when stdout carries no sentinel (providers drop the tail of heavy
277
+ * streams; a reconnect captures only post-resume output) we read the file
278
+ * back. Shared by all three invocation paths. */
279
+ async function classifyRunnerOutcome<TOutput>(
280
+ sandbox: SandboxProvider,
281
+ output: { stdout: string; stderr: string; exitCode: number },
282
+ resultToken: string,
283
+ ): Promise<StepResult<TOutput>> {
284
+ const parsed = parseStepResult<TOutput>(output.stdout, resultToken);
285
+ if (parsed) return parsed;
286
+
287
+ // No sentinel on stdout — read the durable token-keyed file the runner wrote
288
+ // BEFORE its stdout emit. A runner that exited 1 after a user-step throw
289
+ // still wrote `{ok:false, kind:"user-step"}`, which beats a generic exit.
290
+ const fileRead = await sandbox.commands.run(
291
+ `cat ${stepResultFilePath(resultToken)} 2>/dev/null || true`,
292
+ ).catch(() => null);
293
+ if (fileRead?.stdout) {
294
+ const fromFile = parseStepResult<TOutput>(fileRead.stdout, resultToken);
295
+ if (fromFile) return fromFile;
296
+ }
297
+
298
+ // Distinguish "runner exited badly before emitting" (runner-exit) from
299
+ // "runner exited cleanly but didn't speak the protocol" (protocol).
300
+ if (output.exitCode !== 0) {
301
+ const stderrTail = output.stderr.trim().slice(-2000);
302
+ const stdoutTail = output.stdout.trim().slice(-2000);
303
+ const details = [
304
+ stderrTail ? `stderr:\n${stderrTail}` : "",
305
+ stdoutTail ? `stdout:\n${stdoutTail}` : "",
306
+ ].filter(Boolean).join("\n");
307
+ return {
308
+ ok: false,
309
+ error: {
310
+ kind: "runner-exit",
311
+ message: `runner subprocess exited ${output.exitCode} before emitting a step result${details ? `\n${details}` : ""}`,
312
+ exitCode: output.exitCode,
313
+ },
257
314
  };
315
+ }
316
+ const tail = output.stdout.trim().slice(-500);
317
+ return {
318
+ ok: false,
319
+ error: {
320
+ kind: "protocol",
321
+ message: `no tokenised step result on stdout or in the result file (runner exited 0 without emitting)${tail ? `\nstdout tail:\n${tail}` : ""}`,
322
+ },
258
323
  };
259
- const stdoutSplitter = makeLineSplitter(opts?.onStdout, true);
260
- const stderrSplitter = makeLineSplitter(opts?.onStderr, false);
324
+ }
325
+
326
+ /** Poll the runner's durable result file (token-keyed, written immediately before
327
+ * exit on EVERY path — success, failure, pause) as the AUTHORITATIVE completion
328
+ * signal, using `/proc/<pid>` liveness to tell work-in-progress from a crash.
329
+ * Used wherever the live stream cannot be trusted for completion: the native
330
+ * resume path (a reconnected handle's `wait()` can resolve early after a VM
331
+ * pause/resume) AND the launch path after a live-stream transport fault.
332
+ * `cat`-ing the result file is itself immune to the large-frame compression bug
333
+ * — it is a single small JSON line, far below any compression threshold. The
334
+ * activity's `startToCloseTimeout` is the real upper bound; `DEADLINE` is a
335
+ * backstop so a wedged runner can't leak this loop in the worker forever. */
336
+ async function pollDurableResult<TOutput>(
337
+ sandbox: SandboxProvider,
338
+ runnerPid: number,
339
+ resultToken: string,
340
+ signal?: AbortSignal,
341
+ ): Promise<StepResult<TOutput>> {
342
+ const resultFile = stepResultFilePath(resultToken);
343
+ const POLL_MS = 1000;
344
+ const DEADLINE = Date.now() + 45 * 60_000;
345
+ const readResult = async (): Promise<StepResult<TOutput> | null> => {
346
+ const r = await sandbox.commands.run(`cat ${resultFile} 2>/dev/null || true`).catch(() => null);
347
+ return r?.stdout ? (parseStepResult<TOutput>(r.stdout, resultToken) ?? null) : null;
348
+ };
349
+ for (let attempt = 1; ; attempt++) {
350
+ // Aborted = the activity already resolved via the pause watcher (a re-pause);
351
+ // this poll's result is now unobserved. Return (never throw — the promise is
352
+ // no longer awaited) so it settles cleanly with no unhandled rejection.
353
+ if (signal?.aborted) {
354
+ return { ok: false, error: { kind: "runner-exit", message: "reconnectStep aborted (run re-paused)", exitCode: 1 } };
355
+ }
356
+ const fromFile = await readResult();
357
+ if (fromFile) return fromFile;
358
+
359
+ const probe = await sandbox.commands
360
+ .run(`test -d /proc/${runnerPid} && echo alive || echo dead`)
361
+ .catch(() => null);
362
+ const dead = (probe?.stdout ?? "").includes("dead");
363
+ process.stderr.write(
364
+ `[recover] poll ${attempt}: result-file=absent runner=${dead ? "dead" : "alive"} pid=${runnerPid}\n`,
365
+ );
366
+ if (dead) {
367
+ // Close the write-then-exit race with one final read, else classify a crash.
368
+ const finalRead = await readResult();
369
+ if (finalRead) return finalRead;
370
+ return {
371
+ ok: false,
372
+ error: { kind: "runner-exit", message: `runner pid ${runnerPid} exited without writing a result file`, exitCode: 1 },
373
+ };
374
+ }
375
+ if (Date.now() > DEADLINE) {
376
+ return {
377
+ ok: false,
378
+ error: { kind: "runner-exit", message: `timed out after 45m waiting for runner pid ${runnerPid} to emit a result`, exitCode: 1 },
379
+ };
380
+ }
381
+ await new Promise((r) => setTimeout(r, POLL_MS));
382
+ }
383
+ }
384
+
385
+ /** Recover a step's result AND its full logs after the live output stream died
386
+ * mid-run (the dominant cause being E2B's connect-web transport throwing on a
387
+ * compressed large frame — a LOG-TRANSPORT fault, not a runner failure). Waits
388
+ * for the runner to actually finish (`pollDurableResult`), then reads the tee'd
389
+ * log file back over the HTTP file transport (compression-immune) and backfills
390
+ * only the lines the dead stream never delivered — everything past the last
391
+ * whole line already emitted live (`stdoutConsumedChars`), so no duplication and
392
+ * no split lines. Logs BEFORE the fault are already persisted by the activity's
393
+ * live flush; this restores the tail so "what happened" survives intact. */
394
+ async function recoverLogsAndResult<TOutput>(
395
+ sandbox: SandboxProvider,
396
+ runnerPid: number,
397
+ resultToken: string,
398
+ liveSplitters: { stdoutConsumedChars: () => number },
399
+ opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
400
+ ): Promise<StepResult<TOutput>> {
401
+ const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken);
402
+ // Runner has exited → the tee'd log file is complete. Best-effort: a failed
403
+ // readback just means the recovered run keeps the live logs it already had.
404
+ if (sandbox.files.read && opts?.onStdout) {
405
+ const fullLog = await sandbox.files.read(stepLogFilePath(resultToken)).catch(() => null);
406
+ if (fullLog != null) {
407
+ const already = liveSplitters.stdoutConsumedChars();
408
+ const tail = fullLog.length > already ? fullLog.slice(already) : "";
409
+ if (tail.length > 0) {
410
+ const backfill = makeStreamSplitters(opts, resultToken);
411
+ backfill.onStdout?.(tail);
412
+ backfill.flush();
413
+ }
414
+ }
415
+ }
416
+ return result;
417
+ }
418
+
419
+ /** Write the step's input + request-context files and build the runner env.
420
+ * Shared by the foreground and background launch paths. Returns the
421
+ * per-invocation `resultToken` + the env map. */
422
+ async function prepareStepLaunch(
423
+ sandbox: SandboxProvider,
424
+ request: StepRequest,
425
+ opts?: InvokeStepOptions,
426
+ ): Promise<{ resultToken: string; envs: Record<string, string> }> {
427
+ const resultToken = randomBytes(16).toString("hex");
428
+ await Promise.all([
429
+ sandbox.files.write(stepInputPath(request.stepIndex), JSON.stringify(request.input)),
430
+ sandbox.files.write(requestContextPath(request.stepIndex), JSON.stringify(request.requestContext)),
431
+ ]);
432
+ const envs = {
433
+ ...(opts?.envs ?? {}),
434
+ ...buildStepEnvs({ runId: request.runId, stepIndex: request.stepIndex, resultToken, isResume: opts?.isResume }),
435
+ };
436
+ return { resultToken, envs };
437
+ }
438
+
439
+ export async function invokeStep<TOutput = unknown>(
440
+ sandbox: SandboxProvider,
441
+ request: StepRequest,
442
+ opts?: InvokeStepOptions,
443
+ ): Promise<StepResult<TOutput>> {
444
+ const { resultToken, envs } = await prepareStepLaunch(sandbox, request, opts);
445
+ const splitters = makeStreamSplitters(opts, resultToken);
261
446
 
262
447
  let result: { stdout: string; stderr: string; exitCode: number };
263
448
  try {
449
+ // Run the agent as ROOT (ADR-0019 follow-up). The factory drive (Archil)
450
+ // presents its S3-synced files root-owned and exposes no uid-mapped mount,
451
+ // so a non-root agent EACCESes on every shared file — the reason for the
452
+ // expensive boot-time `chmod -R` walk. As root the agent writes them
453
+ // directly: the walk disappears entirely. Egress is edge-enforced with NO
454
+ // root exemption (see sandbox.ts), so root is confined exactly like the
455
+ // non-root user. `HOME=/root` so the spawned `claude` finds root's skills.
264
456
  result = await sandbox.commands.run(RUNNER_COMMAND, {
265
- envs,
457
+ envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
458
+ sudo: true,
266
459
  timeoutMs: 0,
267
- ...(stdoutSplitter.onChunk ? { onStdout: stdoutSplitter.onChunk } : {}),
268
- ...(stderrSplitter.onChunk ? { onStderr: stderrSplitter.onChunk } : {}),
460
+ ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
461
+ ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
269
462
  });
270
463
  } catch (e) {
271
464
  // Typed sandbox-infrastructure failure: the engine's recovery contract
@@ -286,55 +479,135 @@ export async function invokeStep<TOutput = unknown>(
286
479
  exitCode: ce.exitCode ?? ce.result?.exitCode ?? 1,
287
480
  };
288
481
  }
289
- stdoutSplitter.flush();
290
- stderrSplitter.flush();
482
+ splitters.flush();
483
+ return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
484
+ }
291
485
 
292
- const parsed = parseStepResult<TOutput>(result.stdout, resultToken);
293
- if (parsed) return parsed;
486
+ /**
487
+ * Launch the step runner as a BACKGROUND command (ADR-0028) and return a handle
488
+ * the activity drives: it races `wait()` against a server pause request and, on
489
+ * a pause, freezes the VM (`pauseProcess`) and persists `runnerPid` +
490
+ * `resultToken` so `reconnectStep` can continue the SAME process — no re-run.
491
+ * Requires a provider with `commands.runBackground` (E2B); the foreground
492
+ * `invokeStep` is the path for providers without it (Vercel).
493
+ */
494
+ export async function launchStep<TOutput = unknown>(
495
+ sandbox: SandboxProvider,
496
+ request: StepRequest,
497
+ opts?: InvokeStepOptions,
498
+ ): Promise<RunningStep<TOutput>> {
499
+ if (!sandbox.commands.runBackground) {
500
+ throw new Error("launchStep requires a provider with background-command support (commands.runBackground)");
501
+ }
502
+ const { resultToken, envs } = await prepareStepLaunch(sandbox, request, opts);
503
+ const splitters = makeStreamSplitters(opts, resultToken);
504
+ // Tee the runner's stdout to a durable, token-keyed LOG file (stderr stays a
505
+ // separate live stream). The live output rides E2B's connect-web command
506
+ // stream, which THROWS on a compressed large frame (gRPC-web cannot decode
507
+ // message compression) and kills the feed mid-run. The tee'd file, read back
508
+ // over the envd HTTP transport (compression-immune), lets `wait()` recover the
509
+ // full logs + result instead of failing a run that actually completed. `tee`
510
+ // runs in the same `bash -c` pipeline E2B already wraps the command in, so the
511
+ // background pid (the pause / reconnect-by-pid handle) is unchanged; the result
512
+ // sentinel still rides stdout (tee passes it through) and the result FILE is
513
+ // written by the runner directly, so the happy path is untouched. `set -o
514
+ // pipefail` is REQUIRED: without it the pipeline's exit code is tee's (0),
515
+ // masking a non-zero runner exit and breaking the runner-exit classification;
516
+ // pipefail propagates the runner's code (E2B execs via `bash -c`).
517
+ //
518
+ // Run the agent as ROOT — see invokeStep above. `sudo:true` → user:"root";
519
+ // `HOME=/root` so the spawned `claude` finds root's skills.
520
+ const proc = await sandbox.commands.runBackground(`set -o pipefail; ${RUNNER_COMMAND} | tee ${stepLogFilePath(resultToken)}`, {
521
+ envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
522
+ sudo: true,
523
+ timeoutMs: 0,
524
+ ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
525
+ ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
526
+ });
527
+ return {
528
+ runnerPid: proc.pid,
529
+ resultToken,
530
+ async wait() {
531
+ try {
532
+ const result = await proc.wait();
533
+ splitters.flush();
534
+ return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
535
+ } catch (e) {
536
+ // A genuine infra death must propagate so the engine re-provisions.
537
+ if (e instanceof SandboxUnavailableError) throw e;
538
+ // Otherwise the runner's LIVE output stream died while the runner itself
539
+ // is alive and writing its durable result + tee'd log files. The dominant
540
+ // cause is E2B's connect-web transport throwing on a compressed large
541
+ // frame ("received unsupported compressed output") — a LOG-TRANSPORT
542
+ // fault, NEVER a reason to fail a run that completed. Degrade the live
543
+ // feed, alert, and recover the result + full logs over the HTTP file
544
+ // transport. We do NOT flush the live splitter here — the recovery
545
+ // backfills from the last whole line, re-reading any in-flight partial
546
+ // line whole from the durable log (avoids a split/duplicated line).
547
+ opts?.onStreamDegraded?.({ error: e, runnerPid: proc.pid, resultToken });
548
+ return recoverLogsAndResult<TOutput>(sandbox, proc.pid, resultToken, splitters, opts);
549
+ }
550
+ },
551
+ };
552
+ }
294
553
 
295
- // No sentinel on stdout. The runner also persists the sentinel line to a
296
- // token-keyed file before emitting — providers drop the tail of heavy
297
- // stdout streams, so read the durable copy back. Tried before exit-code
298
- // classification: a runner that exited 1 after a user-step throw still
299
- // wrote the file, and its `{ok:false, kind:"user-step"}` beats a generic
300
- // runner-exit error. A runner killed before emitting wrote no file and
301
- // falls through.
302
- const fileRead = await sandbox.commands.run(
303
- `cat ${stepResultFilePath(resultToken)} 2>/dev/null || true`,
304
- ).catch(() => null);
305
- if (fileRead?.stdout) {
306
- const fromFile = parseStepResult<TOutput>(fileRead.stdout, resultToken);
307
- if (fromFile) return fromFile;
554
+ /**
555
+ * Resume a previously-paused background runner (ADR-0028) and return a
556
+ * `RunningStep` handle — uniform with `launchStep` so the activity can race
557
+ * `wait()` against a fresh pause request (a resumed step can pause again). After
558
+ * the workflow reconnects the suspended VM (`Sandbox.connect` auto-resumes it),
559
+ * this re-attaches to the still-running runner by `runnerPid`; `wait()` awaits
560
+ * its exit (event-driven — no polling) and classifies the output (the durable
561
+ * result file is authoritative on this path). If the runner already exited
562
+ * during resume, the re-attach fails and `wait()` classifies from the file.
563
+ */
564
+ export async function reconnectStep<TOutput = unknown>(
565
+ sandbox: SandboxProvider,
566
+ resume: { runnerPid: number; resultToken: string; stepIndex: number },
567
+ opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr"> & { signal?: AbortSignal },
568
+ ): Promise<RunningStep<TOutput>> {
569
+ if (!sandbox.commands.connectProcess) {
570
+ throw new Error("reconnectStep requires a provider with background-command support (commands.connectProcess)");
308
571
  }
572
+ const { runnerPid, resultToken } = resume;
573
+ const splitters = makeStreamSplitters(opts, resultToken);
574
+ const signal = opts?.signal;
309
575
 
310
- // No tokenised sentinel on stdout. Distinguish "runner exited badly
311
- // before emitting" (runner-exit) from "runner exited cleanly but didn't
312
- // speak the protocol" (protocol) — the exit code is the evidence.
313
- if (result.exitCode !== 0) {
314
- const stderrTail = result.stderr.trim().slice(-2000);
315
- const stdoutTail = result.stdout.trim().slice(-2000);
316
- const details = [
317
- stderrTail ? `stderr:\n${stderrTail}` : "",
318
- stdoutTail ? `stdout:\n${stdoutTail}` : "",
319
- ].filter(Boolean).join("\n");
320
- return {
321
- ok: false,
322
- error: {
323
- kind: "runner-exit",
324
- message: `runner subprocess exited ${result.exitCode} before emitting a step result${details ? `\n${details}` : ""}`,
325
- exitCode: result.exitCode,
326
- },
327
- };
576
+ // Re-attach to the suspended runner for LIVE stdout streaming only. This is
577
+ // best-effort: a reconnect that throws after the native resume just means no
578
+ // live feed for the dashboard — completion is read from the durable result
579
+ // file below, NOT from this handle.
580
+ let attached: SandboxBackgroundProcess | null = null;
581
+ try {
582
+ attached = await sandbox.commands.connectProcess(runnerPid, {
583
+ ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
584
+ ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
585
+ });
586
+ } catch (e) {
587
+ if (e instanceof SandboxUnavailableError) throw e;
328
588
  }
329
- // Carry the stdout tail: "no tokenised step result" alone is useless to
330
- // an operator — the tail usually shows whether the runner finished its
331
- // work (output truncated by the provider) or never got there.
332
- const tail = result.stdout.trim().slice(-500);
589
+
333
590
  return {
334
- ok: false,
335
- error: {
336
- kind: "protocol",
337
- message: `no tokenised step result on stdout or in the result file (runner exited 0 without emitting)${tail ? `\nstdout tail:\n${tail}` : ""}`,
591
+ runnerPid,
592
+ resultToken,
593
+ async wait() {
594
+ // CRITICAL: do NOT trust `connect(pid).wait()` on the resume path. After a
595
+ // native VM pause/resume, E2B's re-attached handle can resolve its wait()
596
+ // EARLY — the reconnected stdout stream closes while the runner process is
597
+ // still alive and working — returning a default `{exitCode:0}`. That made
598
+ // the worker report "exited 0 without emitting" and force-kill a runner
599
+ // that was mid-iteration. So ignore the handle for completion and poll the
600
+ // runner's durable result file + `/proc` liveness instead — the same
601
+ // authoritative signal the launch-path recovery uses (`pollDurableResult`).
602
+ try {
603
+ const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken, signal);
604
+ splitters.flush();
605
+ return result;
606
+ } finally {
607
+ // Release the streaming handle (keeps it un-GC'd for the duration of the
608
+ // poll above; a dangling reconnect after a re-pause is cleaned up here).
609
+ await attached?.kill().catch(() => {});
610
+ }
338
611
  },
339
612
  };
340
613
  }
@@ -80,3 +80,14 @@ export function requestContextPath(stepIndex: number): string {
80
80
  export function stepResultFilePath(token: string): string {
81
81
  return `/tmp/wf/step-result-${token}.json`;
82
82
  }
83
+
84
+ /** Sandbox-side path where the runner's stdout is tee'd as a durable LOG file,
85
+ * keyed by the per-invocation token. The live output rides E2B's connect-web
86
+ * command stream, which THROWS on a compressed large frame (gRPC-web cannot
87
+ * decode message compression) and kills the feed mid-run. This file, read back
88
+ * over the envd HTTP file transport (compression-immune), lets the invoker
89
+ * recover the FULL logs after such a fault instead of losing the tail — the
90
+ * log-side analogue of `stepResultFilePath` for the result. */
91
+ export function stepLogFilePath(token: string): string {
92
+ return `/tmp/wf/step-log-${token}.log`;
93
+ }
@@ -32,6 +32,13 @@ export interface RuntimeOptions {
32
32
  /** Agent id and label for processor context / adapter logs. */
33
33
  agentId?: string;
34
34
  iteration?: number;
35
+ /** The platform manual (file conventions, connectors & access, how to pause).
36
+ * `agent()` builds it per-run (`buildAgentContextDoc`) and threads it here so
37
+ * a runtime that supports a system-prompt append (the claude runtime) injects
38
+ * it directly — instead of relying on the agent to `cat` the on-disk
39
+ * AGENTS.md/CLAUDE.md, which the Agent SDK doesn't auto-load and which can
40
+ * fail to write on a read-only/degraded working dir. */
41
+ agentManual?: string;
35
42
  /** The run's pause boundary, threaded from the agent loop so a runtime-driven
36
43
  * pre-tool gate (e.g. the ACP `session/request_permission` path through
37
44
  * `gateToolCall`) can raise a human-approval `ctx.pause`. The runtime binds it
@@ -25,6 +25,30 @@ export interface SandboxCommandResult {
25
25
  stderr: string;
26
26
  }
27
27
 
28
+ /** A long-running command launched in the background (ADR-0028). Unlike
29
+ * `commands.run` (which awaits completion on one connection), a background
30
+ * command keeps running inside the VM independent of the launching
31
+ * connection: it survives `pauseProcess()`/resume and is re-attachable by
32
+ * `pid` after a fresh `Sandbox.connect`. This is what lets the platform
33
+ * freeze an agent mid-turn for a human-in-the-loop pause and continue the
34
+ * SAME process on resume — no re-run.
35
+ *
36
+ * Implemented ONLY by process-resume-capable providers (E2B); the presence
37
+ * of `commands.runBackground` IS the capability flag, paired with
38
+ * `pauseProcess`. Providers without it leave both undefined and pause via
39
+ * `snapshot()` + re-run instead. */
40
+ export interface SandboxBackgroundProcess {
41
+ /** OS pid inside the VM — the durable handle used to reconnect after a
42
+ * pause/resume cycle via `commands.connectProcess(pid)`. */
43
+ pid: number;
44
+ /** Resolve when the process exits, with its buffered result. Live output
45
+ * streams to the `onStdout`/`onStderr` passed at launch / connect time.
46
+ * Does NOT throw on a non-zero exit — the result carries `exitCode`. */
47
+ wait(): Promise<SandboxCommandResult>;
48
+ /** Force-terminate the process. */
49
+ kill(): Promise<void>;
50
+ }
51
+
28
52
  /** A spawned long-lived command with a writable stdin and readable stdout,
29
53
  * exposed as byte web-streams. Unlike `commands.run` (which buffers to
30
54
  * completion and exposes stdout only via an `onStdout` callback), a duplex
@@ -85,9 +109,28 @@ export interface SandboxProvider {
85
109
  * view). The vercel/e2b providers (server→sandbox) leave it undefined; an
86
110
  * ACP caller that finds it absent falls back to the JSONL transport. */
87
111
  spawnDuplex?(cmd: string, opts?: SandboxSpawnDuplexOptions): SandboxDuplexProcess;
112
+ /** Launch a command in the background and return immediately with a
113
+ * reconnectable handle (ADR-0028). The process survives the launching
114
+ * connection dropping AND a `pauseProcess()`/resume cycle. OPTIONAL —
115
+ * only process-resume providers (E2B) implement it; its presence (paired
116
+ * with `pauseProcess`) is the native-pause capability flag. */
117
+ runBackground?(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxBackgroundProcess>;
118
+ /** Re-attach to a background command by `pid` after a fresh
119
+ * `Sandbox.connect` (the resume half of `runBackground`). OPTIONAL,
120
+ * E2B-only. Throws if no process with that pid is running. */
121
+ connectProcess?(pid: number, opts?: Pick<SandboxCommandRunOptions, "onStdout" | "onStderr" | "timeoutMs">): Promise<SandboxBackgroundProcess>;
88
122
  };
89
123
  files: {
90
124
  write(path: string, content: string): Promise<void>;
125
+ /** Read a file's text content over the provider's FILE transport. On E2B this
126
+ * is the envd HTTP API (`Sandbox.files.read`), a DIFFERENT transport from
127
+ * `commands` — so a large readback is immune to the connect-web gRPC
128
+ * message-compression that can abort `commands.run` output on a big frame
129
+ * ("received unsupported compressed output"). This is what lets `launchStep`
130
+ * recover the full logs + result after a live-stream fault. OPTIONAL —
131
+ * implemented where durable file-readback is needed (E2B, local); providers
132
+ * that never drive the recovery path (Vercel — foreground only) may omit it. */
133
+ read?(path: string): Promise<string>;
91
134
  };
92
135
  kill(): Promise<void>;
93
136
  /** Capture the running sandbox's state as a reusable snapshot. Vercel and E2B
@@ -99,6 +142,16 @@ export interface SandboxProvider {
99
142
  * be omitted when the provider doesn't expose it; the server stores
100
143
  * `null` for missing values rather than estimating. */
101
144
  snapshot?(): Promise<{ snapshotId: string; sizeBytes?: number }>;
145
+ /** Suspend the live VM in place and return a handle to resume it (ADR-0027).
146
+ * Present ONLY on process-resume-capable providers (E2B via `sandbox.pause()`,
147
+ * returning the sandbox id; resume is `Sandbox.connect(handle)`, which
148
+ * auto-resumes the paused VM). Unlike `snapshot()` — which captures an FS
149
+ * image, kills the origin, and re-runs the step from a fresh sandbox — a
150
+ * process-resume pause FREEZES the live process (zero compute) and continues
151
+ * it exactly where it blocked. The presence of this method IS the capability
152
+ * flag: providers without native VM-suspend leave it undefined and fall back
153
+ * to `snapshot()` + re-run. */
154
+ pauseProcess?(): Promise<{ resumeHandle: string }>;
102
155
  /** Replace the live sandbox's egress policy in place — so the server can
103
156
  * push a freshly resolved policy (with re-minted connector access tokens)
104
157
  * before each step instead of relying on the policy baked at create.
@@ -110,6 +110,12 @@ export interface WorkflowDefinition<
110
110
  /** Same as `input`, for the workflow's return value. Captured into
111
111
  * `outputSchema` metadata and rendered in the IO panel. */
112
112
  output?: z.ZodType<TOutput>;
113
+ /**
114
+ * @deprecated Legacy run-form. Prefer step-form — the
115
+ * `.step(defineStep(...))` builder — for per-step durability/replay and
116
+ * working pause. A run-form body compiles to one opaque step
117
+ * (`compileRunForm`), so any failure/resume re-runs the whole body.
118
+ */
113
119
  run: WorkflowFn<TOutput, TInput>;
114
120
  /**
115
121
  * All snapshot config — boot source plus capture mode.
@@ -298,6 +304,12 @@ function compileRunForm<TOutput, TInput extends Record<string, unknown>>(
298
304
  * stores the step plan; runner subprocesses execute one step at a time
299
305
  * via the StepInvocation seam.
300
306
  */
307
+ /**
308
+ * @deprecated Run-form is legacy. Use the step-form overload —
309
+ * `defineWorkflow({ id, input, output }).step(defineStep(...)).build()` — for
310
+ * durable, replayable steps and working pause. Run-form compiles to a single
311
+ * opaque step (`compileRunForm`); there is no per-step replay.
312
+ */
301
313
  export function defineWorkflow<
302
314
  TOutput = unknown,
303
315
  TInput extends Record<string, unknown> = Record<string, unknown>,