@agent-compose/sdk 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +8 -0
  3. package/dist/agent/run-agent.d.ts +4 -0
  4. package/dist/client.d.ts +77 -15
  5. package/dist/display.d.ts +16 -0
  6. package/dist/index.d.ts +6 -6
  7. package/dist/index.js +522 -123
  8. package/dist/runtimes/_cli-agent.d.ts +34 -7
  9. package/dist/runtimes/claude-code.d.ts +10 -8
  10. package/dist/runtimes/codex.buildcommand.test.d.ts +9 -0
  11. package/dist/runtimes/codex.d.ts +4 -1
  12. package/dist/runtimes/openai-desktop.js +507 -122
  13. package/dist/sandbox/sizes.d.ts +120 -30
  14. package/dist/sandbox.d.ts +1 -1
  15. package/dist/types/api-conversations.d.ts +198 -0
  16. package/dist/types/api-factory.d.ts +84 -7
  17. package/dist/types/api-runs.d.ts +48 -2
  18. package/dist/types/protocol.d.ts +8 -0
  19. package/dist/types/workflow-metadata.d.ts +14 -5
  20. package/dist/utils/bundler.d.ts +56 -0
  21. package/dist/workflow-steps/workflow.d.ts +7 -0
  22. package/dist/workflows/invoke-child.d.ts +18 -0
  23. package/dist/workflows/invoke-child.test.d.ts +9 -0
  24. package/package.json +2 -2
  25. package/src/agent/agent-context.ts +28 -17
  26. package/src/agent/agent-loop.ts +9 -0
  27. package/src/agent/run-agent.ts +5 -0
  28. package/src/client.ts +201 -30
  29. package/src/display.ts +61 -15
  30. package/src/index.ts +22 -9
  31. package/src/runtimes/_cli-agent.ts +302 -63
  32. package/src/runtimes/claude-code.ts +25 -15
  33. package/src/runtimes/codex.ts +19 -5
  34. package/src/sandbox/providers/e2b.ts +8 -4
  35. package/src/sandbox/sizes.ts +127 -44
  36. package/src/sandbox.ts +8 -0
  37. package/src/types/api-conversations.ts +180 -0
  38. package/src/types/api-factory.ts +89 -7
  39. package/src/types/api-runs.ts +50 -2
  40. package/src/types/protocol.ts +8 -0
  41. package/src/types/workflow-metadata.ts +15 -5
  42. package/src/utils/bundler.ts +213 -3
  43. package/src/workflow-steps/workflow.ts +7 -0
  44. package/src/workflows/invoke-child.ts +47 -11
@@ -24,7 +24,7 @@
24
24
  * values set via `agentc secrets set` are visible to the CLI.
25
25
  */
26
26
 
27
- import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxCommandResult, SandboxProvider, ToolCallGateResult } from "../index.js";
27
+ import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
28
28
  import { defineRuntime } from "../types/runtime.js";
29
29
  import { AsyncQueue } from "../agent/async-queue.js";
30
30
  import { formatError } from "../utils/errors.js";
@@ -35,12 +35,65 @@ import { RequestContext } from "../request-context/request-context.js";
35
35
  import { AcpClientPeer, ACP_PROTOCOL_VERSION } from "./_acp-client.js";
36
36
  import { isPauseSignal, boundProcessorPause } from "../pause/pause-core.js";
37
37
  import {
38
- JSONL_GUARD_LINE_CAP_BYTES, JSONL_GUARD_KEEP_BYTES,
38
+ JSONL_GUARD_LINE_CAP_BYTES, JSONL_GUARD_KEEP_BYTES, JSONL_GUARD_NO_SENTINEL_EXIT,
39
39
  jsonlGuardScript, wrapJsonlCommand,
40
40
  } from "./_jsonl-guard.js";
41
+ import type { SandboxBackgroundProcess } from "../types/sandbox.js";
41
42
 
42
43
  function now(): string { return new Date().toISOString(); }
43
44
 
45
+ /** Backoff between tail re-attach attempts after a mid-turn stream fault
46
+ * (durable detached transport). Short: the runner is alive and producing,
47
+ * and each retry costs one exec; the executor's evidence machinery — not
48
+ * this loop — bounds a truly dead sandbox. Exported for tests. */
49
+ export const TAIL_REATTACH_DELAY_MS = 2_000;
50
+
51
+ /** Provider deadline on the detached LAUNCH exec. The launch shell only
52
+ * truncates the durable files, forks the detached runner (stdio fully
53
+ * redirected — see the launch command), writes the pidfile, and echoes the
54
+ * pid — sub-second work, so 30s is generous headroom for a slow envd, not a
55
+ * budget the runner ever consumes. Exported for tests. */
56
+ export const LAUNCH_EXEC_TIMEOUT_MS = 30_000;
57
+
58
+ /** How many times (and how spaced) a THROWN launch exec falls back to reading
59
+ * the durable pidfile before the fault is surfaced. A deadlined/canceled
60
+ * launch STREAM does not mean the launch failed — the detached tree may be
61
+ * up and working — so the pid is recovered over the envd HTTP file
62
+ * transport (immune to the command-stream fault) and the turn proceeds.
63
+ * Only "no pidfile after these attempts" is a real launch failure.
64
+ * Exported for tests. */
65
+ export const LAUNCH_PID_RECOVERY_ATTEMPTS = 3;
66
+ export const LAUNCH_PID_RECOVERY_DELAY_MS = 1_000;
67
+
68
+ /** Explicit deadline on the `command -v` install probe. E2B's default command
69
+ * deadline (60s) throws the deadline_exceeded TimeoutError — every provider
70
+ * call in the turn path carries an explicit timeout so no SDK default can
71
+ * decide a turn's fate. Exported for tests. */
72
+ export const INSTALL_PROBE_TIMEOUT_MS = 30_000;
73
+
74
+ /** Parse a pid out of captured stdout (last non-empty line). NaN when absent. */
75
+ function parsePid(raw: string): number {
76
+ const pid = Number.parseInt(raw.trim().split("\n").pop() ?? "", 10);
77
+ return Number.isInteger(pid) && pid > 0 ? pid : Number.NaN;
78
+ }
79
+
80
+ /** Recover the detached runner's pid from its durable pidfile after the
81
+ * launch exec's STREAM died (deadline_exceeded, canceled context, any wire
82
+ * fault). Reads over the file transport with a short retry so the race
83
+ * where the launch shell is still writing the pidfile is absorbed. */
84
+ async function recoverLaunchPid(read: (path: string) => Promise<string>, pidPath: string): Promise<number> {
85
+ for (let attempt = 0; attempt < LAUNCH_PID_RECOVERY_ATTEMPTS; attempt++) {
86
+ try {
87
+ const pid = parsePid(await read(pidPath));
88
+ if (!Number.isNaN(pid)) return pid;
89
+ } catch { /* file plane briefly unreachable — retried */ }
90
+ if (attempt < LAUNCH_PID_RECOVERY_ATTEMPTS - 1) {
91
+ await new Promise((r) => setTimeout(r, LAUNCH_PID_RECOVERY_DELAY_MS));
92
+ }
93
+ }
94
+ return Number.NaN;
95
+ }
96
+
44
97
  /** Single-quote a value for safe interpolation into a `sh -c` command line. */
45
98
  export function shellQuote(value: string): string {
46
99
  return `'${value.replace(/'/g, `'\\''`)}'`;
@@ -119,13 +172,15 @@ async function withHandshakeTimeout<T>(p: Promise<T>, ms: number): Promise<T> {
119
172
  }
120
173
  }
121
174
 
122
- /** Reasoning-effort level a CLI turn may carry (T2 session effort). The
123
- * per-CLI mapping lives in each spec's `buildCommand` — Claude Code takes a
124
- * thinking-token budget via `MAX_THINKING_TOKENS`, codex takes
125
- * `-c model_reasoning_effort=<level>`. Specs without a real knob
126
- * (opencode/droid/cursor) never receive one: the server hides + rejects
127
- * effort for those runtimes. */
128
- export type CliReasoningEffort = "low" | "medium" | "high";
175
+ /** Reasoning-effort level a CLI turn may carry (T2 session effort) — the
176
+ * UNION of what the effort-capable CLIs accept. The per-CLI mapping lives in
177
+ * each spec's `buildCommand` — Claude Code takes its own `--effort` flag
178
+ * (all five levels), codex takes `-c model_reasoning_effort=<level>`
179
+ * (low|medium|high|xhigh — no "max"; the spec clamps it). Specs without a
180
+ * real knob (opencode/droid/cursor) never receive one: the server hides +
181
+ * rejects effort for those runtimes, and rejects levels a runtime lacks
182
+ * (sessionEffortLockError). */
183
+ export type CliReasoningEffort = "low" | "medium" | "high" | "xhigh" | "max";
129
184
 
130
185
  /** Per-CLI behaviour. The base owns the lifecycle (init/done/error) and the
131
186
  * transport (spawn + JSONL parse); a spec owns the CLI-specific bits. */
@@ -300,7 +355,8 @@ export class CliAgentRunner implements ModelExecutionContract {
300
355
  * install command runs exactly once: on the first run of a content hash. */
301
356
  private async ensureInstalled(): Promise<void> {
302
357
  if (this.installed) return;
303
- const probe = await this.sandbox.commands.run(`command -v ${this.spec.bin}`);
358
+ const probe = await this.sandbox.commands.run(
359
+ `command -v ${this.spec.bin}`, { timeoutMs: INSTALL_PROBE_TIMEOUT_MS });
304
360
  if (probe.exitCode === 0) { this.installed = true; return; }
305
361
  const res = await this.sandbox.commands.run(this.spec.install, { timeoutMs: 300_000 });
306
362
  if (res.exitCode !== 0) {
@@ -638,53 +694,224 @@ export class CliAgentRunner implements ModelExecutionContract {
638
694
  keepBytes: JSONL_GUARD_KEEP_BYTES,
639
695
  });
640
696
 
641
- // Bridge the streaming stdout callback into an async-iterable of complete
642
- // JSONL lines. `onStdout` chunks aren't line-aligned, so buffer + split.
697
+ // Bridge the transport's output into an async-iterable of complete
698
+ // JSONL lines. Two transports feed it (chosen below):
699
+ //
700
+ // - DURABLE DETACHED (providers with `runBackground` + `files.read`,
701
+ // i.e. E2B — every cloud session): the CLI runs as a DETACHED
702
+ // in-guest process whose stdout is a durable file; the server only
703
+ // ever TAILS that file, and a dead tail is re-attached (or the file
704
+ // re-read over the envd HTTP file transport) without the runner
705
+ // noticing. The exit code travels IN-BAND as a final sentinel line.
706
+ // This exists because of the 2026-08-14 prod incident (conversation
707
+ // 8bc082f0, turn bfcb318f): the turn's single streaming exec was the
708
+ // CLI's lifeline — when the connect-web stream's context was
709
+ // canceled at the wire 50s in (envd logged `context canceled`), envd
710
+ // tore down the exec's stdout, the guard died on EPIPE, the CLI took
711
+ // SIGPIPE mid-work, the finished output was lost, and `wait()` never
712
+ // settled — the turn wedged silently until the next dispatch
713
+ // superseded it as a phantom "session restarted".
714
+ //
715
+ // - SINGLE-EXEC STREAMING (everything else: local child_process,
716
+ // Vercel): the pre-incident shape — one exec, stdout streamed,
717
+ // exit code from the exec result.
643
718
  const lines = new AsyncQueue<string>();
644
- let buf = "";
645
- const onStdout = (data: string) => {
646
- buf += data;
647
- let nl: number;
648
- while ((nl = buf.indexOf("\n")) >= 0) {
649
- const line = buf.slice(0, nl).trim();
650
- buf = buf.slice(nl + 1);
651
- if (line) lines.push(line);
652
- }
653
- };
654
- const runOpts = {
655
- ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
656
- onStdout,
657
- timeoutMs: 0, // no provider deadline — see the doc comment above
658
- };
659
-
660
- // The command resolves when the process exits. Kick it off (don't await
661
- // yet); flush the trailing buffer + close the queue on completion so the
662
- // for-await below drains and we can read the exit code. Prefer the
663
- // kill-capable background handle so abort/early-exit can terminate the
664
- // process itself, not just stop reading its stream.
665
- let exited = false;
666
- const settle = (res: SandboxCommandResult): SandboxCommandResult => {
667
- exited = true;
668
- const tail = buf.trim();
669
- if (tail) lines.push(tail);
670
- lines.close();
671
- return res;
672
- };
673
- const fail = (err: unknown): never => { exited = true; lines.close(); throw err; };
674
- let kill: (() => Promise<void>) | undefined;
675
- let runPromise: Promise<SandboxCommandResult>;
719
+ // How the turn ended: the CLI's exit code once known (null while
720
+ // running), plus a best-effort stderr tail for the error message.
721
+ let exitCode: number | null = null;
722
+ let exitStderr = "";
723
+ let transportError: unknown = null;
724
+ let consumerStopped = false;
725
+ let reap: () => void = () => { /* set per transport */ };
726
+ let transport: Promise<void>;
727
+
728
+ const durableRead = this.sandbox.files.read?.bind(this.sandbox.files);
676
729
  if (this.sandbox.commands.runBackground) {
677
- const handle = await this.sandbox.commands.runBackground(guardedCmd, runOpts);
678
- kill = () => handle.kill();
679
- runPromise = handle.wait().then(settle, fail);
730
+ // ── Durable detached transport ─────────────────────────────────────
731
+ const outPath = `${promptPath}.out`;
732
+ const errPath = `${promptPath}.err`;
733
+ // The exit sentinel is appended by the DETACHED script itself after
734
+ // the guarded pipeline exits, so it lands in the durable file even if
735
+ // every server↔sandbox stream is dead by then. In node-guard mode the
736
+ // pipeline's exit code is already the CLI's (the guard re-raises the
737
+ // inner sentinel, including NO_SENTINEL_EXIT for a mid-pipe death).
738
+ const detachedScript =
739
+ `{ ${guardedCmd}; } >> ${shellQuote(outPath)} 2>> ${shellQuote(errPath)} </dev/null; `
740
+ + `printf '\\n%s %s\\n' ${shellQuote(sentinel)} "$?" >> ${shellQuote(outPath)}`;
741
+ // `setsid` detaches the runner into its own session/process group so
742
+ // it survives the launch exec ending AND gives `kill -- -pid` a whole
743
+ // tree to terminate. The fallback keeps working where setsid is
744
+ // absent; a runner that dies anyway surfaces as NO_SENTINEL below.
745
+ //
746
+ // The `>/dev/null 2>&1 </dev/null` on the forked child is LOAD-BEARING,
747
+ // not hygiene (2026-08-14 prod regression, v0.10.39): without it the
748
+ // detached `sh` inherits the LAUNCH exec's stdout/stderr pipes, and
749
+ // envd only ends a command's event stream on pipe EOF — so the launch
750
+ // exec's wait() couldn't settle until the entire detached CLI tree
751
+ // exited, the 30s launch deadline fired first, and EVERY turn died
752
+ // with E2B's `[deadline_exceeded] the operation timed out: … exceeding
753
+ // 'timeoutMs' …` while the orphaned runner kept working unobserved.
754
+ // (The runner's real stdio is redirected INSIDE detachedScript — into
755
+ // the durable out/err files — so the child needs nothing from the
756
+ // launch exec's pipes.)
757
+ //
758
+ // The pid ALSO lands in a durable pidfile so a launch whose STREAM
759
+ // dies (the same wire faults the tail survives) is recovered over the
760
+ // file transport instead of surfacing as turn death — see the catch
761
+ // below.
762
+ const pidPath = `${promptPath}.pid`;
763
+ const launchCmd =
764
+ `: > ${shellQuote(outPath)}; : > ${shellQuote(errPath)}; `
765
+ + `if command -v setsid >/dev/null 2>&1; then setsid sh -c ${shellQuote(detachedScript)} >/dev/null 2>&1 </dev/null & `
766
+ + `else sh -c ${shellQuote(detachedScript)} >/dev/null 2>&1 </dev/null & fi; `
767
+ + `echo "$!" > ${shellQuote(pidPath)}; echo "$!"`;
768
+ let pid = Number.NaN;
769
+ let launched: { exitCode: number; stdout?: string; stderr?: string } | null = null;
770
+ let launchErr: unknown = null;
771
+ try {
772
+ launched = await this.sandbox.commands.run(launchCmd, {
773
+ ...(this.options.cwd ? { cwd: this.options.cwd } : {}), timeoutMs: LAUNCH_EXEC_TIMEOUT_MS });
774
+ } catch (err) { launchErr = err; }
775
+ if (launched) {
776
+ // The exec ROUND-TRIPPED: its exit code and stdout are authoritative.
777
+ pid = parsePid(launched.stdout ?? "");
778
+ if (launched.exitCode !== 0 || Number.isNaN(pid)) {
779
+ throw new Error(`failed to launch ${this.spec.kind}: `
780
+ + `exit ${launched.exitCode}${launched.stderr ? `: ${launched.stderr.slice(-500)}` : ""}`);
781
+ }
782
+ } else {
783
+ // Structural rule of this transport: a dead STREAM is never treated
784
+ // as a dead RUNNER. The launch may well have run to completion with
785
+ // only its exec stream lost (deadline_exceeded, canceled context) —
786
+ // recover the pid from the durable pidfile and continue; only an
787
+ // absent pidfile makes the original fault real.
788
+ pid = durableRead ? await recoverLaunchPid(durableRead, pidPath) : Number.NaN;
789
+ if (Number.isNaN(pid)) throw launchErr;
790
+ console.warn(`[cli-agent] ${this.spec.kind} launch exec stream died but the detached runner is up `
791
+ + `(pid ${pid} via ${pidPath}) — continuing on the durable transport: ${formatError(launchErr)}`);
792
+ }
793
+
794
+ // Bytes of COMPLETE lines already consumed from outPath — a re-attach
795
+ // tails from here, so a dead stream never duplicates or drops lines.
796
+ let offset = 0;
797
+ let currentTail: SandboxBackgroundProcess | null = null;
798
+ const consumeLine = (raw: string): void => {
799
+ offset += Buffer.byteLength(raw, "utf8") + 1;
800
+ const line = raw.trim();
801
+ if (!line) return;
802
+ if (line.startsWith(sentinel)) {
803
+ const code = Number(line.slice(sentinel.length).trim());
804
+ exitCode = Number.isFinite(code) ? code : 0;
805
+ return;
806
+ }
807
+ lines.push(line);
808
+ };
809
+
810
+ reap = () => {
811
+ // Kill the detached tree (group first — setsid made pid the group
812
+ // leader), then whatever tail is currently attached. Fresh execs,
813
+ // deliberately independent of any possibly-dead stream.
814
+ void this.sandbox.commands.run(
815
+ `kill -TERM -- -${pid} 2>/dev/null; kill -TERM ${pid} 2>/dev/null; true`,
816
+ { timeoutMs: 10_000 }).catch(() => { /* best-effort */ });
817
+ void currentTail?.kill().catch(() => { /* already gone */ });
818
+ };
819
+
820
+ transport = (async () => {
821
+ while (exitCode === null && !consumerStopped && !opts.signal?.aborted) {
822
+ let buf = "";
823
+ let handle: SandboxBackgroundProcess | null = null;
824
+ const onStdout = (data: string): void => {
825
+ buf += data;
826
+ let nl: number;
827
+ while ((nl = buf.indexOf("\n")) >= 0) {
828
+ consumeLine(buf.slice(0, nl));
829
+ buf = buf.slice(nl + 1);
830
+ if (exitCode !== null) { void handle?.kill().catch(() => { /* racing exit */ }); return; }
831
+ }
832
+ };
833
+ try {
834
+ handle = await this.sandbox.commands.runBackground!(
835
+ `tail -c +${offset + 1} -f ${shellQuote(outPath)}`,
836
+ { timeoutMs: 0, onStdout });
837
+ } catch { /* attach failed at the wire — durable catch-up below */ }
838
+ if (handle) {
839
+ if (exitCode !== null) void handle.kill().catch(() => { /* done */ });
840
+ currentTail = handle;
841
+ await handle.wait().catch(() => undefined);
842
+ currentTail = null;
843
+ }
844
+ if (exitCode !== null || consumerStopped || opts.signal?.aborted) break;
845
+ // The tail stream died without delivering the exit line. Catch up
846
+ // over the FILE transport (envd HTTP — immune to the command-
847
+ // stream fault): this is also what recovers an already-finished
848
+ // turn's result when every stream is dead.
849
+ if (durableRead) {
850
+ try {
851
+ const bytes = Buffer.from(await durableRead(outPath), "utf8");
852
+ let rest = bytes.subarray(offset).toString("utf8");
853
+ let nl: number;
854
+ while (exitCode === null && (nl = rest.indexOf("\n")) >= 0) {
855
+ consumeLine(rest.slice(0, nl));
856
+ rest = rest.slice(nl + 1);
857
+ }
858
+ } catch { /* file plane briefly unreachable — retried below */ }
859
+ }
860
+ if (exitCode !== null) break;
861
+ // Runner still alive? A FRESH exec answers; unknown (wire fault)
862
+ // keeps retrying — the executor's evidence machinery bounds a
863
+ // truly dead sandbox, not this loop.
864
+ let alive: boolean | null = null;
865
+ try {
866
+ const probe = await this.sandbox.commands.run(
867
+ `kill -0 ${pid} 2>/dev/null`, { timeoutMs: 10_000 });
868
+ alive = probe.exitCode === 0;
869
+ } catch { alive = null; }
870
+ if (alive === false) {
871
+ // Dead with no exit line even in the durable file: abnormal
872
+ // death — surfaces as an error, never a clean done.
873
+ exitCode = JSONL_GUARD_NO_SENTINEL_EXIT;
874
+ break;
875
+ }
876
+ console.warn(`[cli-agent] ${this.spec.kind} turn stream lost with the runner alive — re-tailing ${outPath} from byte ${offset}`);
877
+ await new Promise((r) => setTimeout(r, TAIL_REATTACH_DELAY_MS));
878
+ }
879
+ if (exitCode !== null && exitCode !== 0 && durableRead) {
880
+ try { exitStderr = (await durableRead(errPath)).slice(-2000); }
881
+ catch { /* best-effort */ }
882
+ }
883
+ })().finally(() => lines.close());
680
884
  } else {
681
- runPromise = this.sandbox.commands.run(guardedCmd, runOpts).then(settle, fail);
885
+ // ── Single-exec streaming transport (no runBackground/files.read) ──
886
+ let buf = "";
887
+ const onStdout = (data: string) => {
888
+ buf += data;
889
+ let nl: number;
890
+ while ((nl = buf.indexOf("\n")) >= 0) {
891
+ const line = buf.slice(0, nl).trim();
892
+ buf = buf.slice(nl + 1);
893
+ if (line) lines.push(line);
894
+ }
895
+ };
896
+ const runOpts = {
897
+ ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
898
+ onStdout,
899
+ timeoutMs: 0, // no provider deadline — see the doc comment above
900
+ };
901
+ transport = this.sandbox.commands.run(guardedCmd, runOpts).then(
902
+ (res) => {
903
+ const tail = buf.trim();
904
+ if (tail) lines.push(tail);
905
+ exitCode = res.exitCode;
906
+ exitStderr = (res.stderr ?? "").slice(-2000);
907
+ },
908
+ (err) => { transportError = err; },
909
+ ).finally(() => lines.close());
682
910
  }
683
- // An early generator exit stops awaiting runPromise — keep its rejection
911
+ // An early generator exit stops awaiting the transport — keep it
684
912
  // observed so an aborted turn can never surface an unhandled rejection.
685
- runPromise.catch(() => { /* observed via `await runPromise` when consumed */ });
913
+ void transport.catch(() => { /* observed via `await transport` when consumed */ });
686
914
 
687
- const reap = () => { if (!exited) void kill?.().catch(() => { /* already gone */ }); };
688
915
  if (opts.signal) {
689
916
  if (opts.signal.aborted) reap();
690
917
  else opts.signal.addEventListener("abort", reap, { once: true });
@@ -705,26 +932,38 @@ export class CliAgentRunner implements ModelExecutionContract {
705
932
  }
706
933
  }
707
934
 
708
- const res = await runPromise;
709
- if (res.exitCode !== 0 && !sawError) {
710
- const tail = (res.stderr ?? "").slice(-2000);
711
- yield { type: "error", text: `${this.spec.kind} exited with code ${res.exitCode}${tail ? `: ${tail}` : ""}`, timestamp: now() };
935
+ await transport;
936
+ if (transportError) throw transportError;
937
+ if (exitCode === null) {
938
+ // Aborted/reaped before an exit line existed — honest terminal,
939
+ // never a hang and never a clean done.
940
+ if (!sawError) {
941
+ yield { type: "error", text: `${this.spec.kind} aborted — the in-sandbox runner was terminated`, timestamp: now() };
942
+ }
943
+ return;
944
+ }
945
+ if (exitCode !== 0 && !sawError) {
946
+ const detail = exitCode === JSONL_GUARD_NO_SENTINEL_EXIT
947
+ ? " (the runner died without reporting an exit code)" : "";
948
+ yield { type: "error", text: `${this.spec.kind} exited with code ${exitCode}${detail}${exitStderr ? `: ${exitStderr}` : ""}`, timestamp: now() };
712
949
  return;
713
950
  }
714
951
  if (!sawError) yield { type: "done", sessionId: sessionId ?? "", timestamp: now() };
715
952
  } finally {
716
953
  // Runs on completion AND on early generator exit (the caller broke
717
954
  // out of its for-await: abort, turn closed under the executor,
718
- // supersede). `reap` is a no-op once the process has exited.
955
+ // supersede). Reap only while the runner hasn't reported an exit.
719
956
  opts.signal?.removeEventListener("abort", reap);
720
- reap();
721
- // Opportunistic prompt-file + guard-script cleanup: paths are never
722
- // reused (see uniquePromptPath), so this is hygiene, not correctness —
723
- // fire and forget, and a dead sandbox / failed rm is fine. The shell
724
- // holds an open fd on the redirect, so unlinking under a still-exiting
725
- // CLI is harmless.
957
+ consumerStopped = true;
958
+ if (exitCode === null) reap();
959
+ // Opportunistic prompt-file + guard-script + transcript cleanup:
960
+ // paths are never reused (see uniquePromptPath), so this is hygiene,
961
+ // not correctness — fire and forget, and a dead sandbox / failed rm
962
+ // is fine. The shell holds an open fd on the redirect, so unlinking
963
+ // under a still-exiting CLI is harmless.
726
964
  void this.sandbox.commands.run(
727
- `rm -f ${shellQuote(promptPath)} ${shellQuote(guardPath)}`, { timeoutMs: 10_000 })
965
+ `rm -f ${shellQuote(promptPath)} ${shellQuote(guardPath)} ${shellQuote(`${promptPath}.out`)} ${shellQuote(`${promptPath}.err`)} ${shellQuote(`${promptPath}.pid`)}`,
966
+ { timeoutMs: 10_000 })
728
967
  .catch(() => { /* best-effort */ });
729
968
  }
730
969
  } catch (err) {
@@ -72,13 +72,15 @@ function toolResultText(content: unknown): string {
72
72
  return content == null ? "" : JSON.stringify(content) ?? "";
73
73
  }
74
74
 
75
- /** Extended-thinking token budget per effort level — Claude Code's real knob
76
- * is the `MAX_THINKING_TOKENS` env var (its documented settings env), set
77
- * per invocation below. Values mirror the platform turn loop's
78
- * `EFFORT_BUDGET_TOKENS` (server model-client) so "high" means the same
79
- * thing in a channel and in a claude-code session. */
80
- export const CLAUDE_CODE_THINKING_TOKENS: Record<CliReasoningEffort, number> =
81
- { low: 2_048, medium: 8_192, high: 16_384 };
75
+ /** Claude Code's real reasoning knob is its own `--effort <level>` flag
76
+ * (low|medium|high|xhigh|max — verified against `claude -p --help`). The
77
+ * CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
78
+ * flag verbatim; the CLI itself downgrades a level the selected model lacks
79
+ * (its documented behaviour, e.g. xhigh → high off Opus). The old
80
+ * `MAX_THINKING_TOKENS` env mapping is gone: the CLI deprecated it (treated
81
+ * as on/off on current models) and it could never express xhigh/max. */
82
+ export const CLAUDE_CODE_EFFORT_LEVELS: readonly CliReasoningEffort[] =
83
+ ["low", "medium", "high", "xhigh", "max"];
82
84
 
83
85
  export const claudeCodeSpec: CliAgentSpec = {
84
86
  kind: "claude-code",
@@ -115,17 +117,25 @@ export const claudeCodeSpec: CliAgentSpec = {
115
117
  // root-user refusal (the E2B agent-env user is root).
116
118
  "--dangerously-skip-permissions",
117
119
  ...(model ? [`--model ${shellQuote(model)}`] : []),
120
+ // Reasoning effort is the CLI's own flag; the value comes from the
121
+ // closed CliReasoningEffort set, so it is shell-safe unquoted.
122
+ ...(effort ? [`--effort ${effort}`] : []),
118
123
  ...(sessionId ? [`--resume ${shellQuote(sessionId)}`] : []),
119
124
  ].join(" ");
120
- // Reasoning effort rides as the CLI's own thinking-budget env, same
121
- // per-invocation idiom as IS_SANDBOX (never persisted into settings).
122
- const thinking = effort ? `MAX_THINKING_TOKENS=${CLAUDE_CODE_THINKING_TOKENS[effort]} ` : "";
123
- return `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}${thinking}IS_SANDBOX=1 claude ${flags} < ${shellQuote(promptPath)}`;
125
+ return `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}IS_SANDBOX=1 claude ${flags} < ${shellQuote(promptPath)}`;
124
126
  },
125
127
  // Every stream-json event carries the session id; the init event is first.
126
128
  extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
127
129
  mapEvent: (p): AgentMessage[] => {
128
130
  const ts = now();
131
+ // Sidechain attribution: stream-json stamps `parent_tool_use_id` on
132
+ // every event emitted INSIDE a subagent (the spawning Agent/Task call's
133
+ // tool_use id; null at top level). Forwarded on tool_use/tool_result so
134
+ // renderers can nest child activity under the spawning call instead of
135
+ // flattening it into the parent transcript unattributed.
136
+ const parent = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
137
+ ? { parentToolUseId: p.parent_tool_use_id }
138
+ : {};
129
139
  switch (p.type) {
130
140
  // Assistant API message: content blocks → text / thinking / tool_use.
131
141
  case "assistant": {
@@ -141,7 +151,7 @@ export const claudeCodeSpec: CliAgentSpec = {
141
151
  if (b.type === "tool_use") {
142
152
  return [{
143
153
  type: "tool_use", toolName: String(b.name ?? "tool"),
144
- toolInput: b.input ?? {}, toolUseId: String(b.id ?? ""), timestamp: ts,
154
+ toolInput: b.input ?? {}, toolUseId: String(b.id ?? ""), ...parent, timestamp: ts,
145
155
  }];
146
156
  }
147
157
  return [];
@@ -155,7 +165,7 @@ export const claudeCodeSpec: CliAgentSpec = {
155
165
  b.type === "tool_result"
156
166
  ? [{
157
167
  type: "tool_result", toolUseId: String(b.tool_use_id ?? ""),
158
- output: toolResultText(b.content), isError: b.is_error === true, timestamp: ts,
168
+ output: toolResultText(b.content), isError: b.is_error === true, ...parent, timestamp: ts,
159
169
  }]
160
170
  : []);
161
171
  }
@@ -234,8 +244,8 @@ export const claudeCodeSpec: CliAgentSpec = {
234
244
  export interface ClaudeCodeRuntimeConfig {
235
245
  /** Claude model id (`--model`). Omit to use the CLI's configured default. */
236
246
  model?: string;
237
- /** Extended-thinking effort (`MAX_THINKING_TOKENS`). Omit for the CLI's
238
- * default behaviour (no forced budget). */
247
+ /** Reasoning effort (`--effort <level>`; the CLI accepts all five levels).
248
+ * Omit for the CLI's default behaviour. */
239
249
  effort?: CliReasoningEffort;
240
250
  }
241
251
 
@@ -88,17 +88,27 @@ export const codexSpec: CliAgentSpec = {
88
88
  "--dangerously-bypass-approvals-and-sandbox",
89
89
  ...(model ? ["-m", shellQuote(model)] : []),
90
90
  // codex's own reasoning knob — a config override, valid values
91
- // low|medium|high (plus "minimal", unused here). The value comes from
91
+ // none|minimal|low|medium|high|xhigh (codex docs; xhigh is the
92
+ // codex-max-tier deep-reasoning level). There is NO "max" in codex's
93
+ // vocabulary: the server rejects it for codex sessions
94
+ // (sessionEffortLockError), and this clamp to xhigh is the type-level
95
+ // backstop for a caller that bypasses that gate. The value comes from
92
96
  // the closed CliReasoningEffort set, so it is shell-safe unquoted.
93
- ...(effort ? ["-c", `model_reasoning_effort=${effort}`] : []),
94
- ...(cwd ? ["-C", shellQuote(cwd)] : []),
97
+ ...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
95
98
  ].join(" ");
99
+ // Working directory via a shell `cd` — the idiom EVERY other runtime uses
100
+ // (claude-code/cursor/droid/opencode) — NOT codex's `-C` flag. `-C` is
101
+ // accepted by `codex exec` but REJECTED by `codex exec resume`
102
+ // ("unexpected argument '-C'"), so a fresh turn worked and every follow-up
103
+ // died. `cd` sets the cwd identically for both paths; promptPath is
104
+ // absolute (/tmp/…), so the `< prompt` redirect survives the cd.
105
+ const cd = cwd ? `cd ${shellQuote(cwd)} && ` : "";
96
106
  // Fresh turn: `codex exec <flags> - < prompt`. Continue a thread:
97
107
  // `codex exec resume <id> <flags> - < prompt`. (`-` = read prompt from stdin.)
98
108
  const exec = sessionId
99
109
  ? `codex exec resume ${shellQuote(sessionId)} ${flags}`
100
110
  : `codex exec ${flags}`;
101
- return `${exec} - < ${shellQuote(promptPath)}`;
111
+ return `${cd}${exec} - < ${shellQuote(promptPath)}`;
102
112
  },
103
113
  extractSessionId: (p) =>
104
114
  p.type === "thread.started" && typeof p.thread_id === "string" ? p.thread_id : undefined,
@@ -171,12 +181,16 @@ export const codexSpec: CliAgentSpec = {
171
181
  },
172
182
  };
173
183
 
184
+ /** The effort levels codex actually has (`model_reasoning_effort`):
185
+ * low|medium|high|xhigh — no "max" (that level is Claude Code's alone). */
186
+ export type CodexReasoningEffort = Exclude<CliReasoningEffort, "max">;
187
+
174
188
  export interface CodexRuntimeConfig {
175
189
  /** Codex model id (`-m`). Omit to use the codex CLI's configured default. */
176
190
  model?: string;
177
191
  /** Reasoning effort (`-c model_reasoning_effort=<level>`). Omit to use the
178
192
  * codex CLI's configured default. */
179
- effort?: CliReasoningEffort;
193
+ effort?: CodexReasoningEffort;
180
194
  }
181
195
 
182
196
  export function createCodexRuntime(config: CodexRuntimeConfig = {}) {
@@ -237,11 +237,15 @@ function makeE2bSandboxProvider(sb: Sandbox): SandboxProvider {
237
237
  return { resumeHandle: sb.sandboxId };
238
238
  },
239
239
  // The provider kill-clock seam: e2b's `setTimeout` REPLACES the deadline
240
- // (extend or reduce) relative to now. The server's session lifecycle
241
- // keeps this behind its own suspend clock so the deferred pause always
242
- // wins the race against the create-time timeout.
240
+ // (extend or reduce) relative to now, in MILLISECONDS. The server's
241
+ // session lifecycle uses it two ways: a FAR-HORIZON push while the
242
+ // session is active (the kill-clock is an orphan backstop, never
243
+ // load-bearing during a live turn) and a short re-arm when the deferred
244
+ // pause is about to park the VM. Clamped to the plan cap exactly like
245
+ // the create timeout — an over-cap push would 400 and leave the OLD
246
+ // (possibly short) deadline standing.
243
247
  async extendLifetime(ms) {
244
- await sb.setTimeout(ms);
248
+ await sb.setTimeout(Math.min(ms, e2bMaxSandboxMs()));
245
249
  },
246
250
  // Push a freshly-resolved egress policy onto the live sandbox via E2B's
247
251
  // native `updateNetwork` — the E2B analogue of Vercel's `update({