@agent-compose/sdk 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,7 +25,7 @@ import { randomBytes } from "node:crypto";
25
25
  import { z } from "zod";
26
26
  import { SandboxUnavailableError } from "../sandbox-errors.js";
27
27
  import type { SandboxProvider, SandboxBackgroundProcess } from "../types/sandbox.js";
28
- import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath } from "./protocol.js";
28
+ import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath, stepLogFilePath } from "./protocol.js";
29
29
  import { StepPauseRequestSchema } from "./types.js";
30
30
  import type { StepRequest, StepResult } from "./types.js";
31
31
  import type { StepObservability } from "../workflow-steps/observability.js";
@@ -194,6 +194,13 @@ export interface InvokeStepOptions {
194
194
  * the caller's job; the activity batches and inserts at step completion. */
195
195
  onStdout?: (line: string) => void;
196
196
  onStderr?: (line: string) => void;
197
+ /** Called when the LIVE output stream fails mid-run (e.g. E2B's connect-web
198
+ * transport throws `received unsupported compressed output` on a large
199
+ * compressed frame) and the invoker falls back to recovering the result +
200
+ * full logs from the durable token-keyed files. The run is NOT failed — this
201
+ * is the hook to emit a structured alert so the degradation is visible/paged.
202
+ * Best-effort: keep it cheap and non-throwing. */
203
+ onStreamDegraded?: (info: { error: unknown; runnerPid: number; resultToken: string }) => void;
197
204
  }
198
205
 
199
206
  /** A step running as a background command (ADR-0028). The activity races its
@@ -219,13 +226,18 @@ export interface RunningStep<TOutput = unknown> {
219
226
  function makeStreamSplitters(
220
227
  opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
221
228
  resultToken: string,
222
- ): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void } {
229
+ ): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void; stdoutConsumedChars: () => number } {
223
230
  const resultSentinel = stepResultLinePrefix(resultToken);
224
231
  const pauseSentinel = stepPauseLinePrefix(resultToken);
225
232
  const isSentinel = (line: string) => line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
226
233
  const make = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
227
- if (!sink) return { onChunk: undefined, flush: () => {} };
234
+ if (!sink) return { onChunk: undefined, flush: () => {}, consumed: () => 0 };
228
235
  let buf = "";
236
+ // Chars consumed as WHOLE lines (incl. each trailing "\n"), counting sentinel
237
+ // lines too — this is the offset into the tee'd log file up to which the live
238
+ // stream has already delivered. The recovery backfill starts exactly here, so
239
+ // an in-flight partial line is re-read whole from the file (no split, no dup).
240
+ let consumed = 0;
229
241
  return {
230
242
  onChunk: (chunk: string) => {
231
243
  buf += chunk;
@@ -233,6 +245,7 @@ function makeStreamSplitters(
233
245
  while ((nl = buf.indexOf("\n")) !== -1) {
234
246
  const line = buf.slice(0, nl);
235
247
  buf = buf.slice(nl + 1);
248
+ consumed += line.length + 1;
236
249
  if (filterSentinel && isSentinel(line)) continue;
237
250
  sink(line);
238
251
  }
@@ -241,9 +254,11 @@ function makeStreamSplitters(
241
254
  if (buf.length === 0) return;
242
255
  const line = buf;
243
256
  buf = "";
257
+ consumed += line.length;
244
258
  if (filterSentinel && isSentinel(line)) return;
245
259
  sink(line);
246
260
  },
261
+ consumed: () => consumed,
247
262
  };
248
263
  };
249
264
  const stdout = make(opts?.onStdout, true);
@@ -252,6 +267,7 @@ function makeStreamSplitters(
252
267
  ...(stdout.onChunk ? { onStdout: stdout.onChunk } : {}),
253
268
  ...(stderr.onChunk ? { onStderr: stderr.onChunk } : {}),
254
269
  flush: () => { stdout.flush(); stderr.flush(); },
270
+ stdoutConsumedChars: () => stdout.consumed(),
255
271
  };
256
272
  }
257
273
 
@@ -307,6 +323,99 @@ async function classifyRunnerOutcome<TOutput>(
307
323
  };
308
324
  }
309
325
 
326
+ /** Poll the runner's durable result file (token-keyed, written immediately before
327
+ * exit on EVERY path — success, failure, pause) as the AUTHORITATIVE completion
328
+ * signal, using `/proc/<pid>` liveness to tell work-in-progress from a crash.
329
+ * Used wherever the live stream cannot be trusted for completion: the native
330
+ * resume path (a reconnected handle's `wait()` can resolve early after a VM
331
+ * pause/resume) AND the launch path after a live-stream transport fault.
332
+ * `cat`-ing the result file is itself immune to the large-frame compression bug
333
+ * — it is a single small JSON line, far below any compression threshold. The
334
+ * activity's `startToCloseTimeout` is the real upper bound; `DEADLINE` is a
335
+ * backstop so a wedged runner can't leak this loop in the worker forever. */
336
+ async function pollDurableResult<TOutput>(
337
+ sandbox: SandboxProvider,
338
+ runnerPid: number,
339
+ resultToken: string,
340
+ signal?: AbortSignal,
341
+ ): Promise<StepResult<TOutput>> {
342
+ const resultFile = stepResultFilePath(resultToken);
343
+ const POLL_MS = 1000;
344
+ const DEADLINE = Date.now() + 45 * 60_000;
345
+ const readResult = async (): Promise<StepResult<TOutput> | null> => {
346
+ const r = await sandbox.commands.run(`cat ${resultFile} 2>/dev/null || true`).catch(() => null);
347
+ return r?.stdout ? (parseStepResult<TOutput>(r.stdout, resultToken) ?? null) : null;
348
+ };
349
+ for (let attempt = 1; ; attempt++) {
350
+ // Aborted = the activity already resolved via the pause watcher (a re-pause);
351
+ // this poll's result is now unobserved. Return (never throw — the promise is
352
+ // no longer awaited) so it settles cleanly with no unhandled rejection.
353
+ if (signal?.aborted) {
354
+ return { ok: false, error: { kind: "runner-exit", message: "reconnectStep aborted (run re-paused)", exitCode: 1 } };
355
+ }
356
+ const fromFile = await readResult();
357
+ if (fromFile) return fromFile;
358
+
359
+ const probe = await sandbox.commands
360
+ .run(`test -d /proc/${runnerPid} && echo alive || echo dead`)
361
+ .catch(() => null);
362
+ const dead = (probe?.stdout ?? "").includes("dead");
363
+ process.stderr.write(
364
+ `[recover] poll ${attempt}: result-file=absent runner=${dead ? "dead" : "alive"} pid=${runnerPid}\n`,
365
+ );
366
+ if (dead) {
367
+ // Close the write-then-exit race with one final read, else classify a crash.
368
+ const finalRead = await readResult();
369
+ if (finalRead) return finalRead;
370
+ return {
371
+ ok: false,
372
+ error: { kind: "runner-exit", message: `runner pid ${runnerPid} exited without writing a result file`, exitCode: 1 },
373
+ };
374
+ }
375
+ if (Date.now() > DEADLINE) {
376
+ return {
377
+ ok: false,
378
+ error: { kind: "runner-exit", message: `timed out after 45m waiting for runner pid ${runnerPid} to emit a result`, exitCode: 1 },
379
+ };
380
+ }
381
+ await new Promise((r) => setTimeout(r, POLL_MS));
382
+ }
383
+ }
384
+
385
+ /** Recover a step's result AND its full logs after the live output stream died
386
+ * mid-run (the dominant cause being E2B's connect-web transport throwing on a
387
+ * compressed large frame — a LOG-TRANSPORT fault, not a runner failure). Waits
388
+ * for the runner to actually finish (`pollDurableResult`), then reads the tee'd
389
+ * log file back over the HTTP file transport (compression-immune) and backfills
390
+ * only the lines the dead stream never delivered — everything past the last
391
+ * whole line already emitted live (`stdoutConsumedChars`), so no duplication and
392
+ * no split lines. Logs BEFORE the fault are already persisted by the activity's
393
+ * live flush; this restores the tail so "what happened" survives intact. */
394
+ async function recoverLogsAndResult<TOutput>(
395
+ sandbox: SandboxProvider,
396
+ runnerPid: number,
397
+ resultToken: string,
398
+ liveSplitters: { stdoutConsumedChars: () => number },
399
+ opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
400
+ ): Promise<StepResult<TOutput>> {
401
+ const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken);
402
+ // Runner has exited → the tee'd log file is complete. Best-effort: a failed
403
+ // readback just means the recovered run keeps the live logs it already had.
404
+ if (sandbox.files.read && opts?.onStdout) {
405
+ const fullLog = await sandbox.files.read(stepLogFilePath(resultToken)).catch(() => null);
406
+ if (fullLog != null) {
407
+ const already = liveSplitters.stdoutConsumedChars();
408
+ const tail = fullLog.length > already ? fullLog.slice(already) : "";
409
+ if (tail.length > 0) {
410
+ const backfill = makeStreamSplitters(opts, resultToken);
411
+ backfill.onStdout?.(tail);
412
+ backfill.flush();
413
+ }
414
+ }
415
+ }
416
+ return result;
417
+ }
418
+
310
419
  /** Write the step's input + request-context files and build the runner env.
311
420
  * Shared by the foreground and background launch paths. Returns the
312
421
  * per-invocation `resultToken` + the env map. */
@@ -337,8 +446,16 @@ export async function invokeStep<TOutput = unknown>(
337
446
 
338
447
  let result: { stdout: string; stderr: string; exitCode: number };
339
448
  try {
449
+ // Run the agent as ROOT (ADR-0019 follow-up). The factory drive (Archil)
450
+ // presents its S3-synced files root-owned and exposes no uid-mapped mount,
451
+ // so a non-root agent EACCESes on every shared file — the reason for the
452
+ // expensive boot-time `chmod -R` walk. As root the agent writes them
453
+ // directly: the walk disappears entirely. Egress is edge-enforced with NO
454
+ // root exemption (see sandbox.ts), so root is confined exactly like the
455
+ // non-root user. `HOME=/root` so the spawned `claude` finds root's skills.
340
456
  result = await sandbox.commands.run(RUNNER_COMMAND, {
341
- envs,
457
+ envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
458
+ sudo: true,
342
459
  timeoutMs: 0,
343
460
  ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
344
461
  ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
@@ -384,8 +501,25 @@ export async function launchStep<TOutput = unknown>(
384
501
  }
385
502
  const { resultToken, envs } = await prepareStepLaunch(sandbox, request, opts);
386
503
  const splitters = makeStreamSplitters(opts, resultToken);
387
- const proc = await sandbox.commands.runBackground(RUNNER_COMMAND, {
388
- envs,
504
+ // Tee the runner's stdout to a durable, token-keyed LOG file (stderr stays a
505
+ // separate live stream). The live output rides E2B's connect-web command
506
+ // stream, which THROWS on a compressed large frame (gRPC-web cannot decode
507
+ // message compression) and kills the feed mid-run. The tee'd file, read back
508
+ // over the envd HTTP transport (compression-immune), lets `wait()` recover the
509
+ // full logs + result instead of failing a run that actually completed. `tee`
510
+ // runs in the same `bash -c` pipeline E2B already wraps the command in, so the
511
+ // background pid (the pause / reconnect-by-pid handle) is unchanged; the result
512
+ // sentinel still rides stdout (tee passes it through) and the result FILE is
513
+ // written by the runner directly, so the happy path is untouched. `set -o
514
+ // pipefail` is REQUIRED: without it the pipeline's exit code is tee's (0),
515
+ // masking a non-zero runner exit and breaking the runner-exit classification;
516
+ // pipefail propagates the runner's code (E2B execs via `bash -c`).
517
+ //
518
+ // Run the agent as ROOT — see invokeStep above. `sudo:true` → user:"root";
519
+ // `HOME=/root` so the spawned `claude` finds root's skills.
520
+ const proc = await sandbox.commands.runBackground(`set -o pipefail; ${RUNNER_COMMAND} | tee ${stepLogFilePath(resultToken)}`, {
521
+ envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
522
+ sudo: true,
389
523
  timeoutMs: 0,
390
524
  ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
391
525
  ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
@@ -394,9 +528,25 @@ export async function launchStep<TOutput = unknown>(
394
528
  runnerPid: proc.pid,
395
529
  resultToken,
396
530
  async wait() {
397
- const result = await proc.wait();
398
- splitters.flush();
399
- return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
531
+ try {
532
+ const result = await proc.wait();
533
+ splitters.flush();
534
+ return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
535
+ } catch (e) {
536
+ // A genuine infra death must propagate so the engine re-provisions.
537
+ if (e instanceof SandboxUnavailableError) throw e;
538
+ // Otherwise the runner's LIVE output stream died while the runner itself
539
+ // is alive and writing its durable result + tee'd log files. The dominant
540
+ // cause is E2B's connect-web transport throwing on a compressed large
541
+ // frame ("received unsupported compressed output") — a LOG-TRANSPORT
542
+ // fault, NEVER a reason to fail a run that completed. Degrade the live
543
+ // feed, alert, and recover the result + full logs over the HTTP file
544
+ // transport. We do NOT flush the live splitter here — the recovery
545
+ // backfills from the last whole line, re-reading any in-flight partial
546
+ // line whole from the durable log (avoids a split/duplicated line).
547
+ opts?.onStreamDegraded?.({ error: e, runnerPid: proc.pid, resultToken });
548
+ return recoverLogsAndResult<TOutput>(sandbox, proc.pid, resultToken, splitters, opts);
549
+ }
400
550
  },
401
551
  };
402
552
  }
@@ -414,38 +564,50 @@ export async function launchStep<TOutput = unknown>(
414
564
  export async function reconnectStep<TOutput = unknown>(
415
565
  sandbox: SandboxProvider,
416
566
  resume: { runnerPid: number; resultToken: string; stepIndex: number },
417
- opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr">,
567
+ opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr"> & { signal?: AbortSignal },
418
568
  ): Promise<RunningStep<TOutput>> {
419
569
  if (!sandbox.commands.connectProcess) {
420
570
  throw new Error("reconnectStep requires a provider with background-command support (commands.connectProcess)");
421
571
  }
422
572
  const { runnerPid, resultToken } = resume;
423
573
  const splitters = makeStreamSplitters(opts, resultToken);
424
- let proc: SandboxBackgroundProcess | null = null;
574
+ const signal = opts?.signal;
575
+
576
+ // Re-attach to the suspended runner for LIVE stdout streaming only. This is
577
+ // best-effort: a reconnect that throws after the native resume just means no
578
+ // live feed for the dashboard — completion is read from the durable result
579
+ // file below, NOT from this handle.
580
+ let attached: SandboxBackgroundProcess | null = null;
425
581
  try {
426
- proc = await sandbox.commands.connectProcess(runnerPid, {
582
+ attached = await sandbox.commands.connectProcess(runnerPid, {
427
583
  ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
428
584
  ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
429
585
  });
430
586
  } catch (e) {
431
- // A sandbox-infrastructure failure must propagate intact (recovery
432
- // contract). Anything else means the runner already exited during resume
433
- // (we re-attached too late) — its durable result file is written, so the
434
- // handle's wait() classifies from the file below.
435
587
  if (e instanceof SandboxUnavailableError) throw e;
436
588
  }
437
- const attached = proc;
589
+
438
590
  return {
439
591
  runnerPid,
440
592
  resultToken,
441
593
  async wait() {
442
- let output: { stdout: string; stderr: string; exitCode: number } = { stdout: "", stderr: "", exitCode: 0 };
443
- if (attached) {
444
- try { output = await attached.wait(); }
445
- catch (e) { if (e instanceof SandboxUnavailableError) throw e; }
594
+ // CRITICAL: do NOT trust `connect(pid).wait()` on the resume path. After a
595
+ // native VM pause/resume, E2B's re-attached handle can resolve its wait()
596
+ // EARLY — the reconnected stdout stream closes while the runner process is
597
+ // still alive and working — returning a default `{exitCode:0}`. That made
598
+ // the worker report "exited 0 without emitting" and force-kill a runner
599
+ // that was mid-iteration. So ignore the handle for completion and poll the
600
+ // runner's durable result file + `/proc` liveness instead — the same
601
+ // authoritative signal the launch-path recovery uses (`pollDurableResult`).
602
+ try {
603
+ const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken, signal);
604
+ splitters.flush();
605
+ return result;
606
+ } finally {
607
+ // Release the streaming handle (keeps it un-GC'd for the duration of the
608
+ // poll above; a dangling reconnect after a re-pause is cleaned up here).
609
+ await attached?.kill().catch(() => {});
446
610
  }
447
- splitters.flush();
448
- return classifyRunnerOutcome<TOutput>(sandbox, output, resultToken);
449
611
  },
450
612
  };
451
613
  }
@@ -80,3 +80,14 @@ export function requestContextPath(stepIndex: number): string {
80
80
  export function stepResultFilePath(token: string): string {
81
81
  return `/tmp/wf/step-result-${token}.json`;
82
82
  }
83
+
84
+ /** Sandbox-side path where the runner's stdout is tee'd as a durable LOG file,
85
+ * keyed by the per-invocation token. The live output rides E2B's connect-web
86
+ * command stream, which THROWS on a compressed large frame (gRPC-web cannot
87
+ * decode message compression) and kills the feed mid-run. This file, read back
88
+ * over the envd HTTP file transport (compression-immune), lets the invoker
89
+ * recover the FULL logs after such a fault instead of losing the tail — the
90
+ * log-side analogue of `stepResultFilePath` for the result. */
91
+ export function stepLogFilePath(token: string): string {
92
+ return `/tmp/wf/step-log-${token}.log`;
93
+ }
@@ -122,6 +122,15 @@ export interface SandboxProvider {
122
122
  };
123
123
  files: {
124
124
  write(path: string, content: string): Promise<void>;
125
+ /** Read a file's text content over the provider's FILE transport. On E2B this
126
+ * is the envd HTTP API (`Sandbox.files.read`), a DIFFERENT transport from
127
+ * `commands` — so a large readback is immune to the connect-web gRPC
128
+ * message-compression that can abort `commands.run` output on a big frame
129
+ * ("received unsupported compressed output"). This is what lets `launchStep`
130
+ * recover the full logs + result after a live-stream fault. OPTIONAL —
131
+ * implemented where durable file-readback is needed (E2B, local); providers
132
+ * that never drive the recovery path (Vercel — foreground only) may omit it. */
133
+ read?(path: string): Promise<string>;
125
134
  };
126
135
  kill(): Promise<void>;
127
136
  /** Capture the running sandbox's state as a reusable snapshot. Vercel and E2B
@@ -110,6 +110,12 @@ export interface WorkflowDefinition<
110
110
  /** Same as `input`, for the workflow's return value. Captured into
111
111
  * `outputSchema` metadata and rendered in the IO panel. */
112
112
  output?: z.ZodType<TOutput>;
113
+ /**
114
+ * @deprecated Legacy run-form. Prefer step-form — the
115
+ * `.step(defineStep(...))` builder — for per-step durability/replay and
116
+ * working pause. A run-form body compiles to one opaque step
117
+ * (`compileRunForm`), so any failure/resume re-runs the whole body.
118
+ */
113
119
  run: WorkflowFn<TOutput, TInput>;
114
120
  /**
115
121
  * All snapshot config — boot source plus capture mode.
@@ -298,6 +304,12 @@ function compileRunForm<TOutput, TInput extends Record<string, unknown>>(
298
304
  * stores the step plan; runner subprocesses execute one step at a time
299
305
  * via the StepInvocation seam.
300
306
  */
307
+ /**
308
+ * @deprecated Run-form is legacy. Use the step-form overload —
309
+ * `defineWorkflow({ id, input, output }).step(defineStep(...)).build()` — for
310
+ * durable, replayable steps and working pause. Run-form compiles to a single
311
+ * opaque step (`compileRunForm`); there is no per-step replay.
312
+ */
301
313
  export function defineWorkflow<
302
314
  TOutput = unknown,
303
315
  TInput extends Record<string, unknown> = Record<string, unknown>,
@@ -1,4 +1,21 @@
1
- /** Convert any thrown value to a string message. */
1
+ /** Convert any thrown value to a string message, INCLUDING its `.cause` chain.
2
+ *
3
+ * Many wrapped errors carry the real reason on `.cause` and only a generic
4
+ * summary on `.message` — the Temporal SDK's "Failed to start Workflow" is the
5
+ * canonical example (its `.cause` is the actual gRPC rejection, e.g. "search
6
+ * attribute X is not defined"). Returning `.message` alone swallowed that, so
7
+ * failures surfaced as opaque one-liners. Walk the chain and join the messages
8
+ * so the root cause is always visible. Cycle-guarded against self-referential
9
+ * `cause` links. */
2
10
  export function formatError(err: unknown): string {
3
- return err instanceof Error ? err.message : String(err);
11
+ if (!(err instanceof Error)) return String(err);
12
+ const parts: string[] = [err.message];
13
+ const seen = new Set<unknown>([err]);
14
+ let cause: unknown = (err as { cause?: unknown }).cause;
15
+ while (cause != null && !seen.has(cause)) {
16
+ seen.add(cause);
17
+ parts.push(cause instanceof Error ? cause.message : String(cause));
18
+ cause = cause instanceof Error ? (cause as { cause?: unknown }).cause : undefined;
19
+ }
20
+ return parts.join(": ");
4
21
  }