@agent-compose/sdk 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +24 -1
  5. package/dist/client.d.ts +338 -534
  6. package/dist/directives.d.ts +112 -0
  7. package/dist/display.d.ts +242 -0
  8. package/dist/errors.d.ts +24 -1
  9. package/dist/index.d.ts +34 -13
  10. package/dist/index.js +2984 -861
  11. package/dist/pause/wrappers.d.ts +31 -9
  12. package/dist/processors/ask-human.d.ts +30 -0
  13. package/dist/processors/ask-human.test.d.ts +1 -0
  14. package/dist/processors/index.d.ts +1 -0
  15. package/dist/runtimes/_acp-client.d.ts +46 -1
  16. package/dist/runtimes/_cli-agent.d.ts +58 -4
  17. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  18. package/dist/runtimes/amp.d.ts +2 -2
  19. package/dist/runtimes/claude-code.d.ts +59 -0
  20. package/dist/runtimes/claude-code.test.d.ts +14 -0
  21. package/dist/runtimes/claude.d.ts +16 -0
  22. package/dist/runtimes/claude.test.d.ts +8 -0
  23. package/dist/runtimes/codex.d.ts +9 -3
  24. package/dist/runtimes/cursor.d.ts +9 -0
  25. package/dist/runtimes/droid.d.ts +9 -0
  26. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  27. package/dist/runtimes/openai-desktop.js +2922 -861
  28. package/dist/runtimes/opencode.d.ts +25 -0
  29. package/dist/runtimes/vercel.js +22 -1
  30. package/dist/sandbox/devbox.d.ts +42 -0
  31. package/dist/sandbox/exec-stream.d.ts +14 -0
  32. package/dist/sandbox/network-policy.d.ts +100 -0
  33. package/dist/sandbox/provider-def.d.ts +79 -0
  34. package/dist/sandbox/providers/desktop.d.ts +10 -0
  35. package/dist/sandbox/providers/e2b.d.ts +17 -0
  36. package/dist/sandbox/providers/local.d.ts +11 -0
  37. package/dist/sandbox/providers/vercel.d.ts +18 -0
  38. package/dist/sandbox/registry.d.ts +45 -0
  39. package/dist/sandbox/sizes.d.ts +68 -0
  40. package/dist/sandbox.d.ts +24 -299
  41. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  42. package/dist/step-invocation/invoker.d.ts +24 -1
  43. package/dist/step-invocation/protocol.d.ts +13 -0
  44. package/dist/types/api-compliance.d.ts +71 -0
  45. package/dist/types/api-conversations.d.ts +492 -0
  46. package/dist/types/api-factory.d.ts +309 -0
  47. package/dist/types/api-projects.d.ts +131 -0
  48. package/dist/types/api-runs.d.ts +377 -0
  49. package/dist/types/api-scopes.d.ts +102 -0
  50. package/dist/types/conversation-stream.d.ts +191 -0
  51. package/dist/types/execution-context.d.ts +12 -2
  52. package/dist/types/protocol.d.ts +30 -1
  53. package/dist/types/sandbox-environment.d.ts +8 -5
  54. package/dist/types/sandbox.d.ts +79 -0
  55. package/dist/types/workflow-metadata.d.ts +33 -8
  56. package/dist/types/workflow-plan.d.ts +10 -0
  57. package/dist/types/workflow.d.ts +18 -193
  58. package/dist/utils/bundler.d.ts +12 -1
  59. package/dist/utils/errors.d.ts +9 -1
  60. package/dist/workflow-steps/index.d.ts +1 -1
  61. package/dist/workflow-steps/observability.d.ts +8 -1
  62. package/dist/workflow-steps/runner.d.ts +3 -3
  63. package/dist/workflow-steps/step.d.ts +15 -1
  64. package/dist/workflow-steps/types.d.ts +19 -5
  65. package/dist/workflow-steps/workflow.d.ts +22 -1
  66. package/dist/workflows/engine.d.ts +3 -2
  67. package/dist/workflows/invoke-child.d.ts +2 -2
  68. package/package.json +1 -1
  69. package/src/agent/agent-context.ts +206 -16
  70. package/src/agent/agent-loop.ts +40 -4
  71. package/src/agent/run-agent.ts +9 -1
  72. package/src/client.ts +909 -621
  73. package/src/directives.ts +184 -0
  74. package/src/display.ts +788 -0
  75. package/src/errors.ts +39 -0
  76. package/src/index.ts +117 -10
  77. package/src/pause/wrappers.ts +44 -9
  78. package/src/processors/ask-human.ts +136 -0
  79. package/src/processors/index.ts +5 -0
  80. package/src/runtimes/_acp-client.ts +72 -3
  81. package/src/runtimes/_cli-agent.ts +171 -38
  82. package/src/runtimes/_jsonl-guard.ts +219 -0
  83. package/src/runtimes/claude-code.ts +246 -0
  84. package/src/runtimes/claude.ts +32 -2
  85. package/src/runtimes/codex.ts +55 -3
  86. package/src/runtimes/cursor.ts +59 -0
  87. package/src/runtimes/droid.ts +63 -0
  88. package/src/runtimes/openai-desktop.ts +59 -14
  89. package/src/runtimes/opencode.ts +61 -0
  90. package/src/sandbox/devbox.ts +48 -0
  91. package/src/sandbox/exec-stream.ts +48 -0
  92. package/src/sandbox/network-policy.ts +181 -0
  93. package/src/sandbox/provider-def.ts +94 -0
  94. package/src/sandbox/providers/desktop.ts +57 -0
  95. package/src/sandbox/providers/e2b.ts +354 -0
  96. package/src/sandbox/providers/local.ts +106 -0
  97. package/src/sandbox/providers/vercel.ts +331 -0
  98. package/src/sandbox/registry.ts +198 -0
  99. package/src/sandbox/sizes.ts +95 -0
  100. package/src/sandbox.ts +59 -1263
  101. package/src/step-invocation/invoker.ts +319 -34
  102. package/src/step-invocation/protocol.ts +19 -0
  103. package/src/types/api-compliance.ts +79 -0
  104. package/src/types/api-conversations.ts +522 -0
  105. package/src/types/api-factory.ts +336 -0
  106. package/src/types/api-projects.ts +140 -0
  107. package/src/types/api-runs.ts +412 -0
  108. package/src/types/api-scopes.ts +102 -0
  109. package/src/types/conversation-stream.ts +231 -0
  110. package/src/types/execution-context.ts +10 -2
  111. package/src/types/protocol.ts +33 -0
  112. package/src/types/sandbox-environment.ts +28 -9
  113. package/src/types/sandbox.ts +78 -0
  114. package/src/types/workflow-metadata.ts +35 -8
  115. package/src/types/workflow-plan.ts +11 -0
  116. package/src/types/workflow.ts +25 -280
  117. package/src/utils/bundler.ts +32 -5
  118. package/src/utils/errors.ts +34 -2
  119. package/src/workflow-steps/index.ts +1 -0
  120. package/src/workflow-steps/observability.ts +19 -8
  121. package/src/workflow-steps/runner.ts +4 -4
  122. package/src/workflow-steps/step.ts +49 -1
  123. package/src/workflow-steps/types.ts +20 -5
  124. package/src/workflow-steps/workflow.ts +22 -1
  125. package/src/workflows/engine.ts +3 -2
  126. package/src/workflows/invoke-child.ts +2 -2
@@ -25,7 +25,7 @@ import { randomBytes } from "node:crypto";
25
25
  import { z } from "zod";
26
26
  import { SandboxUnavailableError } from "../sandbox-errors.js";
27
27
  import type { SandboxProvider, SandboxBackgroundProcess } from "../types/sandbox.js";
28
- import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath } from "./protocol.js";
28
+ import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath, stepLogFilePath, stepPidFilePath } from "./protocol.js";
29
29
  import { StepPauseRequestSchema } from "./types.js";
30
30
  import type { StepRequest, StepResult } from "./types.js";
31
31
  import type { StepObservability } from "../workflow-steps/observability.js";
@@ -194,6 +194,18 @@ export interface InvokeStepOptions {
194
194
  * the caller's job; the activity batches and inserts at step completion. */
195
195
  onStdout?: (line: string) => void;
196
196
  onStderr?: (line: string) => void;
197
+ /** Called when the LIVE output stream fails mid-run (e.g. E2B's connect-web
198
+ * transport throws `received unsupported compressed output` on a large
199
+ * compressed frame) and the invoker falls back to recovering the result +
200
+ * full logs from the durable token-keyed files. The run is NOT failed — this
201
+ * is the hook to emit a structured alert so the degradation is visible/paged.
202
+ * Best-effort: keep it cheap and non-throwing. */
203
+ onStreamDegraded?: (info: { error: unknown; runnerPid: number; resultToken: string }) => void;
204
+ /** Ceiling on the durable-result recovery poll after a live-stream fault
205
+ * (default 45m — the wedged-runner backstop). The activity passes its step
206
+ * ceiling so a stream fault on a step with hours of legitimate work left
207
+ * recovers instead of timing out; startToClose is the real wall. */
208
+ recoveryDeadlineMs?: number;
197
209
  }
198
210
 
199
211
  /** A step running as a background command (ADR-0028). The activity races its
@@ -219,13 +231,18 @@ export interface RunningStep<TOutput = unknown> {
219
231
  function makeStreamSplitters(
220
232
  opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
221
233
  resultToken: string,
222
- ): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void } {
234
+ ): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void; stdoutConsumedChars: () => number } {
223
235
  const resultSentinel = stepResultLinePrefix(resultToken);
224
236
  const pauseSentinel = stepPauseLinePrefix(resultToken);
225
237
  const isSentinel = (line: string) => line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
226
238
  const make = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
227
- if (!sink) return { onChunk: undefined, flush: () => {} };
239
+ if (!sink) return { onChunk: undefined, flush: () => {}, consumed: () => 0 };
228
240
  let buf = "";
241
+ // Chars consumed as WHOLE lines (incl. each trailing "\n"), counting sentinel
242
+ // lines too — this is the offset into the tee'd log file up to which the live
243
+ // stream has already delivered. The recovery backfill starts exactly here, so
244
+ // an in-flight partial line is re-read whole from the file (no split, no dup).
245
+ let consumed = 0;
229
246
  return {
230
247
  onChunk: (chunk: string) => {
231
248
  buf += chunk;
@@ -233,6 +250,7 @@ function makeStreamSplitters(
233
250
  while ((nl = buf.indexOf("\n")) !== -1) {
234
251
  const line = buf.slice(0, nl);
235
252
  buf = buf.slice(nl + 1);
253
+ consumed += line.length + 1;
236
254
  if (filterSentinel && isSentinel(line)) continue;
237
255
  sink(line);
238
256
  }
@@ -241,9 +259,11 @@ function makeStreamSplitters(
241
259
  if (buf.length === 0) return;
242
260
  const line = buf;
243
261
  buf = "";
262
+ consumed += line.length;
244
263
  if (filterSentinel && isSentinel(line)) return;
245
264
  sink(line);
246
265
  },
266
+ consumed: () => consumed,
247
267
  };
248
268
  };
249
269
  const stdout = make(opts?.onStdout, true);
@@ -252,6 +272,7 @@ function makeStreamSplitters(
252
272
  ...(stdout.onChunk ? { onStdout: stdout.onChunk } : {}),
253
273
  ...(stderr.onChunk ? { onStderr: stderr.onChunk } : {}),
254
274
  flush: () => { stdout.flush(); stderr.flush(); },
275
+ stdoutConsumedChars: () => stdout.consumed(),
255
276
  };
256
277
  }
257
278
 
@@ -307,6 +328,174 @@ async function classifyRunnerOutcome<TOutput>(
307
328
  };
308
329
  }
309
330
 
331
+ /** Poll the runner's durable result file (token-keyed, written immediately before
332
+ * exit on EVERY path — success, failure, pause) as the AUTHORITATIVE completion
333
+ * signal, using `/proc/<pid>` liveness to tell work-in-progress from a crash.
334
+ * Used wherever the live stream cannot be trusted for completion: the native
335
+ * resume path (a reconnected handle's `wait()` can resolve early after a VM
336
+ * pause/resume) AND the launch path after a live-stream transport fault.
337
+ * `cat`-ing the result file is itself immune to the large-frame compression bug
338
+ * — it is a single small JSON line, far below any compression threshold. The
339
+ * activity's `startToCloseTimeout` is the real upper bound; the deadline here
340
+ * is a backstop so a wedged runner can't leak this loop in the worker forever
341
+ * (default 45m; a worker-death recovery reconnecting mid-step passes the full
342
+ * step ceiling instead — hours of legitimate work may remain). */
343
+ async function pollDurableResult<TOutput>(
344
+ sandbox: SandboxProvider,
345
+ runnerPid: number,
346
+ resultToken: string,
347
+ opts?: { signal?: AbortSignal; deadlineMs?: number },
348
+ ): Promise<StepResult<TOutput>> {
349
+ const signal = opts?.signal;
350
+ const resultFile = stepResultFilePath(resultToken);
351
+ const POLL_MS = 1000;
352
+ const deadlineMs = opts?.deadlineMs ?? 45 * 60_000;
353
+ const DEADLINE = Date.now() + deadlineMs;
354
+ // Both internal commands carry an explicit budget. Required for Vercel:
355
+ // without `timeoutMs` there is no hard client-side deadline in the provider's
356
+ // `commands.run`, so a wedged VM hangs one `cat` forever and the poll
357
+ // deadline never advances. A timeout throws → the `.catch(() => null)`
358
+ // treats it as a failed read/probe and the loop continues (a hung probe
359
+ // never misclassifies as dead). Harmless on E2B — a 1-line cat is ms-fast.
360
+ const CMD_TIMEOUT_MS = 10_000;
361
+ // A destroyed sandbox (e.g. run cancelled → sandbox killed mid-recovery)
362
+ // rejects EVERY command, which a lone `.catch(() => null)` renders
363
+ // indistinguishable from a live runner — the loop would spin until the
364
+ // deadline (hours) against a VM that no longer exists. Track consecutive
365
+ // iterations where BOTH the result read and the liveness probe rejected and
366
+ // exit with an honest runner-exit once the streak is unambiguous. 20
367
+ // iterations ≈ 30s–7min of continuous rejections (each command may burn its
368
+ // 10s budget) — long enough to ride out a transient provider-API blip,
369
+ // short enough that a dead sandbox doesn't hold the worker slot for hours.
370
+ const UNREACHABLE_LIMIT = 20;
371
+ let unreachableStreak = 0;
372
+ const readResult = async (): Promise<StepResult<TOutput> | "unreachable" | null> => {
373
+ const r = await sandbox.commands.run(`cat ${resultFile} 2>/dev/null || true`, { timeoutMs: CMD_TIMEOUT_MS }).catch(() => "unreachable" as const);
374
+ if (r === "unreachable") return "unreachable";
375
+ return r.stdout ? (parseStepResult<TOutput>(r.stdout, resultToken) ?? null) : null;
376
+ };
377
+ for (let attempt = 1; ; attempt++) {
378
+ // Aborted = the activity already resolved via the pause watcher (a re-pause);
379
+ // this poll's result is now unobserved. Return (never throw — the promise is
380
+ // no longer awaited) so it settles cleanly with no unhandled rejection.
381
+ if (signal?.aborted) {
382
+ return { ok: false, error: { kind: "runner-exit", message: "reconnectStep aborted (run re-paused)", exitCode: 1 } };
383
+ }
384
+ const fromFile = await readResult();
385
+ if (fromFile !== null && fromFile !== "unreachable") return fromFile;
386
+
387
+ const probe = await sandbox.commands
388
+ .run(`test -d /proc/${runnerPid} && echo alive || echo dead`, { timeoutMs: CMD_TIMEOUT_MS })
389
+ .catch(() => null);
390
+ if (fromFile === "unreachable" && probe === null) {
391
+ unreachableStreak++;
392
+ if (unreachableStreak >= UNREACHABLE_LIMIT) {
393
+ return {
394
+ ok: false,
395
+ error: {
396
+ kind: "runner-exit",
397
+ message: `sandbox stopped responding while waiting for runner pid ${runnerPid} to emit a result (${UNREACHABLE_LIMIT} consecutive failed polls)`,
398
+ exitCode: 1,
399
+ },
400
+ };
401
+ }
402
+ } else {
403
+ unreachableStreak = 0;
404
+ }
405
+ const dead = (probe?.stdout ?? "").includes("dead");
406
+ process.stderr.write(
407
+ `[recover] poll ${attempt}: result-file=absent runner=${dead ? "dead" : "alive"} pid=${runnerPid}\n`,
408
+ );
409
+ if (dead) {
410
+ // Close the write-then-exit race with one final read, else classify a crash.
411
+ const finalRead = await readResult();
412
+ if (finalRead !== null && finalRead !== "unreachable") return finalRead;
413
+ return {
414
+ ok: false,
415
+ error: { kind: "runner-exit", message: `runner pid ${runnerPid} exited without writing a result file`, exitCode: 1 },
416
+ };
417
+ }
418
+ if (Date.now() > DEADLINE) {
419
+ return {
420
+ ok: false,
421
+ error: {
422
+ kind: "runner-exit",
423
+ message: `timed out after ${Math.round(deadlineMs / 60_000)}m waiting for runner pid ${runnerPid} to emit a result`,
424
+ exitCode: 1,
425
+ },
426
+ };
427
+ }
428
+ await new Promise((r) => setTimeout(r, POLL_MS));
429
+ }
430
+ }
431
+
432
+ /** Recover a step's result AND its full logs after the live output stream died
433
+ * mid-run (the dominant cause being E2B's connect-web transport throwing on a
434
+ * compressed large frame — a LOG-TRANSPORT fault, not a runner failure). Waits
435
+ * for the runner to actually finish (`pollDurableResult`), then reads the tee'd
436
+ * log file back over the HTTP file transport (compression-immune) and backfills
437
+ * only the lines the dead stream never delivered — everything past the last
438
+ * whole line already emitted live (`stdoutConsumedChars`), so no duplication and
439
+ * no split lines. Logs BEFORE the fault are already persisted by the activity's
440
+ * live flush; this restores the tail so "what happened" survives intact. */
441
+ async function recoverLogsAndResult<TOutput>(
442
+ sandbox: SandboxProvider,
443
+ runnerPid: number,
444
+ resultToken: string,
445
+ liveSplitters: { stdoutConsumedChars: () => number },
446
+ opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
447
+ pollOpts?: { deadlineMs?: number },
448
+ ): Promise<StepResult<TOutput>> {
449
+ const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken, pollOpts);
450
+ // Runner has exited → the tee'd log file is complete. Best-effort: a failed
451
+ // readback just means the recovered run keeps the live logs it already had.
452
+ if (sandbox.files.read && opts?.onStdout) {
453
+ const fullLog = await sandbox.files.read(stepLogFilePath(resultToken)).catch(() => null);
454
+ if (fullLog != null) {
455
+ const already = liveSplitters.stdoutConsumedChars();
456
+ const tail = fullLog.length > already ? fullLog.slice(already) : "";
457
+ if (tail.length > 0) {
458
+ const backfill = makeStreamSplitters(opts, resultToken);
459
+ backfill.onStdout?.(tail);
460
+ backfill.flush();
461
+ }
462
+ }
463
+ }
464
+ return result;
465
+ }
466
+
467
+ /** Read the foreground launch's pidfile back — the /proc liveness handle for
468
+ * the recovery poll. Returns `null` when recovery is impossible, for either
469
+ * reason: the sandbox cannot run ANY command (genuinely dead VM), or the
470
+ * pidfile never appeared. The pidfile write is the FIRST statement of the
471
+ * runner pipeline, so a still-absent pid after the retries means the runner
472
+ * never started and no result file will ever exist — recovery would poll
473
+ * blind for the full step ceiling and then misclassify; the caller must
474
+ * rethrow the original terminal error instead (fast, honest). The retries
475
+ * cover both a transiently-unreachable sandbox and the launch race where the
476
+ * stream fault fired before the VM executed the pidfile write. */
477
+ async function readRunnerPid(
478
+ sandbox: SandboxProvider,
479
+ resultToken: string,
480
+ ): Promise<number | null> {
481
+ const ATTEMPTS = 3;
482
+ const RETRY_DELAY_MS = 2_000;
483
+ for (let attempt = 1; attempt <= ATTEMPTS; attempt++) {
484
+ try {
485
+ const r = await sandbox.commands.run(
486
+ `cat ${stepPidFilePath(resultToken)} 2>/dev/null || true`,
487
+ { timeoutMs: 10_000 },
488
+ );
489
+ const pid = Number.parseInt(r.stdout.trim(), 10);
490
+ if (Number.isInteger(pid) && pid > 0) return pid;
491
+ } catch {
492
+ // Sandbox unreachable this attempt — fall through to the retry delay.
493
+ }
494
+ if (attempt < ATTEMPTS) await new Promise((r) => setTimeout(r, RETRY_DELAY_MS));
495
+ }
496
+ return null;
497
+ }
498
+
310
499
  /** Write the step's input + request-context files and build the runner env.
311
500
  * Shared by the foreground and background launch paths. Returns the
312
501
  * per-invocation `resultToken` + the env map. */
@@ -337,20 +526,60 @@ export async function invokeStep<TOutput = unknown>(
337
526
 
338
527
  let result: { stdout: string; stderr: string; exitCode: number };
339
528
  try {
340
- result = await sandbox.commands.run(RUNNER_COMMAND, {
341
- envs,
342
- timeoutMs: 0,
343
- ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
344
- ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
345
- });
529
+ // Run the agent as ROOT (ADR-0019 follow-up). The factory drive (Archil)
530
+ // presents its S3-synced files root-owned and exposes no uid-mapped mount,
531
+ // so a non-root agent EACCESes on every shared file — the reason for the
532
+ // expensive boot-time `chmod -R` walk. As root the agent writes them
533
+ // directly: the walk disappears entirely. Egress is edge-enforced with NO
534
+ // root exemption (see sandbox.ts), so root is confined exactly like the
535
+ // non-root user. `HOME=/root` so the spawned `claude` finds root's skills.
536
+ //
537
+ // Same pipeline pattern as `launchStep`: tee the runner's stdout to the
538
+ // durable token-keyed log file so a live-stream fault can backfill the
539
+ // undelivered tail, and write the wrapping shell's pid (`$$`) to a pidfile
540
+ // first — the shell waits on the pipeline, so `/proc/<pid>` liveness is a
541
+ // correct "runner may still write a result" proxy for the recovery poll.
542
+ // `set -o pipefail` is REQUIRED: without it the pipeline's exit code is
543
+ // tee's (0), masking a non-zero runner exit and breaking the runner-exit
544
+ // classification (the provider execs via `sh -c`, where pipefail works).
545
+ result = await sandbox.commands.run(
546
+ `set -o pipefail; echo $$ > ${stepPidFilePath(resultToken)}; ${RUNNER_COMMAND} | tee ${stepLogFilePath(resultToken)}`,
547
+ {
548
+ envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
549
+ sudo: true,
550
+ timeoutMs: 0,
551
+ ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
552
+ ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
553
+ },
554
+ );
346
555
  } catch (e) {
347
556
  // Typed sandbox-infrastructure failure: the engine's recovery contract
348
- // (re-provision when retryable, honest terminal classification otherwise)
349
- // keys on this error propagating INTACT — its `[sandbox-unavailable:*]`
350
- // message prefix must reach the workflow as the failure leaf, not ride an
351
- // embedded stderr tail that a later slice(-2000) can drop. Rethrow before
352
- // the CommandExitError downgrade below.
353
- if (e instanceof SandboxUnavailableError) throw e;
557
+ // keys on retryability. Its `[sandbox-unavailable:*]` message prefix must
558
+ // reach the workflow as the failure leaf, not ride an embedded stderr tail
559
+ // that a later slice(-2000) can drop. Handle before the CommandExitError
560
+ // downgrade below.
561
+ if (e instanceof SandboxUnavailableError) {
562
+ // Pre-launch (retryable): nothing user-side ran — propagate INTACT so the
563
+ // workflow layer re-provisions (identity + prefix contract; pinned test).
564
+ if (e.retryable) throw e;
565
+ // Post-launch terminal: the LIVE stream/wait call died (e.g. Vercel's raw
566
+ // "The operation timed out.") while the runner is very likely still alive
567
+ // and will write its durable result file. Recover instead of failing —
568
+ // NEVER re-launch the runner. A null pid means recovery is impossible
569
+ // (sandbox can't run ANY command, or the pidfile never appeared because
570
+ // the runner never actually started): rethrow the original terminal
571
+ // classification fast rather than polling blind for the step ceiling.
572
+ // We do NOT flush the live splitter here — the recovery backfills from
573
+ // the last whole line, re-reading any in-flight partial line whole from
574
+ // the durable log (avoids a split/duplicated line).
575
+ const runnerPid = await readRunnerPid(sandbox, resultToken);
576
+ if (runnerPid === null) throw e;
577
+ opts?.onStreamDegraded?.({ error: e, runnerPid, resultToken });
578
+ return recoverLogsAndResult<TOutput>(
579
+ sandbox, runnerPid, resultToken, splitters, opts,
580
+ opts?.recoveryDeadlineMs !== undefined ? { deadlineMs: opts.recoveryDeadlineMs } : undefined,
581
+ );
582
+ }
354
583
  // Some providers (E2B) throw a CommandExitError on a non-zero exit instead of
355
584
  // returning it. Recover stdout/stderr/exitCode from the error so we can STILL
356
585
  // parse the runner's structured `__AC_STEP_RESULT__` payload — otherwise the
@@ -384,8 +613,25 @@ export async function launchStep<TOutput = unknown>(
384
613
  }
385
614
  const { resultToken, envs } = await prepareStepLaunch(sandbox, request, opts);
386
615
  const splitters = makeStreamSplitters(opts, resultToken);
387
- const proc = await sandbox.commands.runBackground(RUNNER_COMMAND, {
388
- envs,
616
+ // Tee the runner's stdout to a durable, token-keyed LOG file (stderr stays a
617
+ // separate live stream). The live output rides E2B's connect-web command
618
+ // stream, which THROWS on a compressed large frame (gRPC-web cannot decode
619
+ // message compression) and kills the feed mid-run. The tee'd file, read back
620
+ // over the envd HTTP transport (compression-immune), lets `wait()` recover the
621
+ // full logs + result instead of failing a run that actually completed. `tee`
622
+ // runs in the same `bash -c` pipeline E2B already wraps the command in, so the
623
+ // background pid (the pause / reconnect-by-pid handle) is unchanged; the result
624
+ // sentinel still rides stdout (tee passes it through) and the result FILE is
625
+ // written by the runner directly, so the happy path is untouched. `set -o
626
+ // pipefail` is REQUIRED: without it the pipeline's exit code is tee's (0),
627
+ // masking a non-zero runner exit and breaking the runner-exit classification;
628
+ // pipefail propagates the runner's code (E2B execs via `bash -c`).
629
+ //
630
+ // Run the agent as ROOT — see invokeStep above. `sudo:true` → user:"root";
631
+ // `HOME=/root` so the spawned `claude` finds root's skills.
632
+ const proc = await sandbox.commands.runBackground(`set -o pipefail; ${RUNNER_COMMAND} | tee ${stepLogFilePath(resultToken)}`, {
633
+ envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
634
+ sudo: true,
389
635
  timeoutMs: 0,
390
636
  ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
391
637
  ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
@@ -394,9 +640,25 @@ export async function launchStep<TOutput = unknown>(
394
640
  runnerPid: proc.pid,
395
641
  resultToken,
396
642
  async wait() {
397
- const result = await proc.wait();
398
- splitters.flush();
399
- return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
643
+ try {
644
+ const result = await proc.wait();
645
+ splitters.flush();
646
+ return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
647
+ } catch (e) {
648
+ // A genuine infra death must propagate so the engine re-provisions.
649
+ if (e instanceof SandboxUnavailableError) throw e;
650
+ // Otherwise the runner's LIVE output stream died while the runner itself
651
+ // is alive and writing its durable result + tee'd log files. The dominant
652
+ // cause is E2B's connect-web transport throwing on a compressed large
653
+ // frame ("received unsupported compressed output") — a LOG-TRANSPORT
654
+ // fault, NEVER a reason to fail a run that completed. Degrade the live
655
+ // feed, alert, and recover the result + full logs over the HTTP file
656
+ // transport. We do NOT flush the live splitter here — the recovery
657
+ // backfills from the last whole line, re-reading any in-flight partial
658
+ // line whole from the durable log (avoids a split/duplicated line).
659
+ opts?.onStreamDegraded?.({ error: e, runnerPid: proc.pid, resultToken });
660
+ return recoverLogsAndResult<TOutput>(sandbox, proc.pid, resultToken, splitters, opts);
661
+ }
400
662
  },
401
663
  };
402
664
  }
@@ -414,38 +676,61 @@ export async function launchStep<TOutput = unknown>(
414
676
  export async function reconnectStep<TOutput = unknown>(
415
677
  sandbox: SandboxProvider,
416
678
  resume: { runnerPid: number; resultToken: string; stepIndex: number },
417
- opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr">,
679
+ opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr"> & {
680
+ signal?: AbortSignal;
681
+ /** Ceiling on the durable-result poll (default 45m — the wedged-runner
682
+ * backstop). A worker-death recovery reconnecting MID-step passes the
683
+ * step ceiling instead: hours of legitimate work may remain, and the
684
+ * activity's per-attempt startToClose already bounds it server-side. */
685
+ pollDeadlineMs?: number;
686
+ },
418
687
  ): Promise<RunningStep<TOutput>> {
419
688
  if (!sandbox.commands.connectProcess) {
420
689
  throw new Error("reconnectStep requires a provider with background-command support (commands.connectProcess)");
421
690
  }
422
691
  const { runnerPid, resultToken } = resume;
423
692
  const splitters = makeStreamSplitters(opts, resultToken);
424
- let proc: SandboxBackgroundProcess | null = null;
693
+ const signal = opts?.signal;
694
+ const pollDeadlineMs = opts?.pollDeadlineMs;
695
+
696
+ // Re-attach to the suspended runner for LIVE stdout streaming only. This is
697
+ // best-effort: a reconnect that throws after the native resume just means no
698
+ // live feed for the dashboard — completion is read from the durable result
699
+ // file below, NOT from this handle.
700
+ let attached: SandboxBackgroundProcess | null = null;
425
701
  try {
426
- proc = await sandbox.commands.connectProcess(runnerPid, {
702
+ attached = await sandbox.commands.connectProcess(runnerPid, {
427
703
  ...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
428
704
  ...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
429
705
  });
430
706
  } catch (e) {
431
- // A sandbox-infrastructure failure must propagate intact (recovery
432
- // contract). Anything else means the runner already exited during resume
433
- // (we re-attached too late) — its durable result file is written, so the
434
- // handle's wait() classifies from the file below.
435
707
  if (e instanceof SandboxUnavailableError) throw e;
436
708
  }
437
- const attached = proc;
709
+
438
710
  return {
439
711
  runnerPid,
440
712
  resultToken,
441
713
  async wait() {
442
- let output: { stdout: string; stderr: string; exitCode: number } = { stdout: "", stderr: "", exitCode: 0 };
443
- if (attached) {
444
- try { output = await attached.wait(); }
445
- catch (e) { if (e instanceof SandboxUnavailableError) throw e; }
714
+ // CRITICAL: do NOT trust `connect(pid).wait()` on the resume path. After a
715
+ // native VM pause/resume, E2B's re-attached handle can resolve its wait()
716
+ // EARLY — the reconnected stdout stream closes while the runner process is
717
+ // still alive and working — returning a default `{exitCode:0}`. That made
718
+ // the worker report "exited 0 without emitting" and force-kill a runner
719
+ // that was mid-iteration. So ignore the handle for completion and poll the
720
+ // runner's durable result file + `/proc` liveness instead — the same
721
+ // authoritative signal the launch-path recovery uses (`pollDurableResult`).
722
+ try {
723
+ const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken, {
724
+ ...(signal ? { signal } : {}),
725
+ ...(pollDeadlineMs !== undefined ? { deadlineMs: pollDeadlineMs } : {}),
726
+ });
727
+ splitters.flush();
728
+ return result;
729
+ } finally {
730
+ // Release the streaming handle (keeps it un-GC'd for the duration of the
731
+ // poll above; a dangling reconnect after a re-pause is cleaned up here).
732
+ await attached?.kill().catch(() => {});
446
733
  }
447
- splitters.flush();
448
- return classifyRunnerOutcome<TOutput>(sandbox, output, resultToken);
449
734
  },
450
735
  };
451
736
  }
@@ -80,3 +80,22 @@ export function requestContextPath(stepIndex: number): string {
80
80
  export function stepResultFilePath(token: string): string {
81
81
  return `/tmp/wf/step-result-${token}.json`;
82
82
  }
83
+
84
+ /** Sandbox-side pidfile the FOREGROUND launch writes (`echo $$ > …`) before
85
+ * spawning the runner pipeline — the wrapping shell's pid. The recovery path
86
+ * reads it back to probe `/proc/<pid>` liveness when the live stream/wait
87
+ * call dies mid-step (providers with no background-process pid — Vercel). */
88
+ export function stepPidFilePath(token: string): string {
89
+ return `/tmp/wf/step-pid-${token}.pid`;
90
+ }
91
+
92
+ /** Sandbox-side path where the runner's stdout is tee'd as a durable LOG file,
93
+ * keyed by the per-invocation token. The live output rides E2B's connect-web
94
+ * command stream, which THROWS on a compressed large frame (gRPC-web cannot
95
+ * decode message compression) and kills the feed mid-run. This file, read back
96
+ * over the envd HTTP file transport (compression-immune), lets the invoker
97
+ * recover the FULL logs after such a fault instead of losing the tail — the
98
+ * log-side analogue of `stepResultFilePath` for the result. */
99
+ export function stepLogFilePath(token: string): string {
100
+ return `/tmp/wf/step-log-${token}.log`;
101
+ }
@@ -0,0 +1,79 @@
1
+ /**
2
+ * Break-glass compliance session wire types (ADR-0051).
3
+ *
4
+ * A team admin has no default access to a scoped (Private/Shared) document.
5
+ * To view one they don't hold a grant on, they open a READ-ONLY,
6
+ * owner-approved, time-boxed compliance session over a scope (one factory, or
7
+ * the whole team). These are the shapes the server's `/api/v1/compliance/*`
8
+ * routes emit — `client.ts` is a thin typed wrapper over them.
9
+ */
10
+
11
+ export type ComplianceScopeKind = "factory" | "team";
12
+ export type ComplianceStatus = "pending" | "active" | "expired" | "revoked";
13
+
14
+ /** One break-glass compliance session. `expiresAt` is null while pending —
15
+ * the TTL clock starts at approval, not request. */
16
+ export interface ComplianceSession {
17
+ id: string;
18
+ teamId: string;
19
+ requestedByUserId: string | null;
20
+ requestedByLabel: string | null;
21
+ approvedByUserId: string | null;
22
+ approvedByLabel: string | null;
23
+ reason: string;
24
+ scopeKind: ComplianceScopeKind;
25
+ /** The factory this session covers (factory scope); null for team scope. */
26
+ scopeRef: string | null;
27
+ status: ComplianceStatus;
28
+ ttlSeconds: number;
29
+ expiresAt: string | null;
30
+ revokedByUserId: string | null;
31
+ revokedByLabel: string | null;
32
+ createdAt: string;
33
+ approvedAt: string | null;
34
+ revokedAt: string | null;
35
+ }
36
+
37
+ /** One per-document access recorded under an active session — the individual
38
+ * audit trail the session's reasoned entry authorizes. */
39
+ export interface ComplianceAccess {
40
+ id: string;
41
+ sessionId: string;
42
+ /** Nulled when the accessed file is later deleted (ON DELETE SET NULL);
43
+ * `filePath` is the durable snapshot that survives. */
44
+ fileId: string | null;
45
+ filePath: string;
46
+ accessedByUserId: string | null;
47
+ accessedByLabel: string | null;
48
+ accessedAt: string;
49
+ }
50
+
51
+ export interface RequestComplianceSessionInput {
52
+ scopeKind: ComplianceScopeKind;
53
+ /** Required for factory scope; must be omitted for team scope. */
54
+ scopeRef?: string | null;
55
+ reason: string;
56
+ /** Session lifetime once approved — 15 min floor, 7 day ceiling. */
57
+ ttlSeconds: number;
58
+ }
59
+
60
+ export interface ListComplianceSessionsOptions {
61
+ status?: ComplianceStatus;
62
+ limit?: number;
63
+ cursor?: string;
64
+ }
65
+
66
+ export interface ComplianceSessionsPage {
67
+ sessions: ComplianceSession[];
68
+ nextCursor: string | null;
69
+ }
70
+
71
+ export interface ListComplianceAccessesOptions {
72
+ limit?: number;
73
+ cursor?: string;
74
+ }
75
+
76
+ export interface ComplianceAccessesPage {
77
+ accesses: ComplianceAccess[];
78
+ nextCursor: string | null;
79
+ }