@agent-compose/sdk 0.7.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +32 -1
  5. package/dist/agent/run-agent.d.ts +4 -0
  6. package/dist/client.d.ts +382 -534
  7. package/dist/directives.d.ts +112 -0
  8. package/dist/display.d.ts +258 -0
  9. package/dist/errors.d.ts +24 -1
  10. package/dist/index.d.ts +26 -14
  11. package/dist/index.js +3774 -1679
  12. package/dist/pause/wrappers.d.ts +31 -9
  13. package/dist/runtimes/_acp-client.d.ts +46 -1
  14. package/dist/runtimes/_cli-agent.d.ts +51 -4
  15. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  16. package/dist/runtimes/amp.d.ts +2 -2
  17. package/dist/runtimes/claude-code.d.ts +61 -0
  18. package/dist/runtimes/claude-code.test.d.ts +14 -0
  19. package/dist/runtimes/claude.d.ts +16 -0
  20. package/dist/runtimes/claude.test.d.ts +8 -0
  21. package/dist/runtimes/codex.d.ts +12 -3
  22. package/dist/runtimes/cursor.d.ts +2 -2
  23. package/dist/runtimes/droid.d.ts +2 -2
  24. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  25. package/dist/runtimes/openai-desktop.js +3718 -1680
  26. package/dist/runtimes/opencode.d.ts +2 -2
  27. package/dist/runtimes/vercel.js +12 -1
  28. package/dist/sandbox/devbox.d.ts +42 -0
  29. package/dist/sandbox/exec-stream.d.ts +14 -0
  30. package/dist/sandbox/network-policy.d.ts +100 -0
  31. package/dist/sandbox/provider-def.d.ts +79 -0
  32. package/dist/sandbox/providers/desktop.d.ts +10 -0
  33. package/dist/sandbox/providers/e2b.d.ts +17 -0
  34. package/dist/sandbox/providers/local.d.ts +11 -0
  35. package/dist/sandbox/providers/vercel.d.ts +18 -0
  36. package/dist/sandbox/registry.d.ts +45 -0
  37. package/dist/sandbox/sizes.d.ts +68 -0
  38. package/dist/sandbox.d.ts +24 -299
  39. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  40. package/dist/step-invocation/invoker.d.ts +10 -0
  41. package/dist/step-invocation/protocol.d.ts +5 -0
  42. package/dist/types/api-compliance.d.ts +71 -0
  43. package/dist/types/api-conversations.d.ts +523 -0
  44. package/dist/types/api-factory.d.ts +334 -0
  45. package/dist/types/api-projects.d.ts +131 -0
  46. package/dist/types/api-runs.d.ts +422 -0
  47. package/dist/types/api-scopes.d.ts +102 -0
  48. package/dist/types/conversation-stream.d.ts +191 -0
  49. package/dist/types/execution-context.d.ts +12 -2
  50. package/dist/types/protocol.d.ts +38 -1
  51. package/dist/types/sandbox-environment.d.ts +8 -5
  52. package/dist/types/sandbox.d.ts +74 -4
  53. package/dist/types/workflow-metadata.d.ts +41 -8
  54. package/dist/types/workflow-plan.d.ts +10 -0
  55. package/dist/types/workflow.d.ts +18 -205
  56. package/dist/utils/bundler.d.ts +68 -1
  57. package/dist/workflow-steps/index.d.ts +1 -1
  58. package/dist/workflow-steps/observability.d.ts +8 -1
  59. package/dist/workflow-steps/runner.d.ts +3 -3
  60. package/dist/workflow-steps/step.d.ts +15 -1
  61. package/dist/workflow-steps/types.d.ts +19 -5
  62. package/dist/workflow-steps/workflow.d.ts +29 -1
  63. package/dist/workflows/engine.d.ts +3 -2
  64. package/dist/workflows/invoke-child.d.ts +20 -2
  65. package/dist/workflows/invoke-child.test.d.ts +9 -0
  66. package/package.json +2 -2
  67. package/src/agent/agent-context.ts +186 -3
  68. package/src/agent/agent-loop.ts +40 -2
  69. package/src/agent/run-agent.ts +5 -0
  70. package/src/client.ts +1048 -625
  71. package/src/directives.ts +184 -0
  72. package/src/display.ts +834 -0
  73. package/src/errors.ts +39 -0
  74. package/src/index.ts +114 -12
  75. package/src/pause/wrappers.ts +44 -9
  76. package/src/runtimes/_acp-client.ts +72 -3
  77. package/src/runtimes/_cli-agent.ts +161 -36
  78. package/src/runtimes/_jsonl-guard.ts +219 -0
  79. package/src/runtimes/claude-code.ts +256 -0
  80. package/src/runtimes/claude.ts +32 -2
  81. package/src/runtimes/codex.ts +63 -3
  82. package/src/runtimes/openai-desktop.ts +59 -14
  83. package/src/sandbox/devbox.ts +48 -0
  84. package/src/sandbox/exec-stream.ts +48 -0
  85. package/src/sandbox/network-policy.ts +181 -0
  86. package/src/sandbox/provider-def.ts +94 -0
  87. package/src/sandbox/providers/desktop.ts +57 -0
  88. package/src/sandbox/providers/e2b.ts +354 -0
  89. package/src/sandbox/providers/local.ts +106 -0
  90. package/src/sandbox/providers/vercel.ts +331 -0
  91. package/src/sandbox/registry.ts +198 -0
  92. package/src/sandbox/sizes.ts +95 -0
  93. package/src/sandbox.ts +59 -1275
  94. package/src/step-invocation/invoker.ts +151 -28
  95. package/src/step-invocation/protocol.ts +8 -0
  96. package/src/types/api-compliance.ts +79 -0
  97. package/src/types/api-conversations.ts +547 -0
  98. package/src/types/api-factory.ts +368 -0
  99. package/src/types/api-projects.ts +140 -0
  100. package/src/types/api-runs.ts +459 -0
  101. package/src/types/api-scopes.ts +102 -0
  102. package/src/types/conversation-stream.ts +231 -0
  103. package/src/types/execution-context.ts +10 -2
  104. package/src/types/protocol.ts +41 -0
  105. package/src/types/sandbox-environment.ts +28 -9
  106. package/src/types/sandbox.ts +73 -4
  107. package/src/types/workflow-metadata.ts +44 -8
  108. package/src/types/workflow-plan.ts +11 -0
  109. package/src/types/workflow.ts +25 -292
  110. package/src/utils/bundler.ts +245 -8
  111. package/src/utils/errors.ts +16 -1
  112. package/src/workflow-steps/index.ts +1 -0
  113. package/src/workflow-steps/observability.ts +19 -8
  114. package/src/workflow-steps/runner.ts +4 -4
  115. package/src/workflow-steps/step.ts +49 -1
  116. package/src/workflow-steps/types.ts +20 -5
  117. package/src/workflow-steps/workflow.ts +29 -1
  118. package/src/workflows/engine.ts +3 -2
  119. package/src/workflows/invoke-child.ts +49 -13
@@ -24,7 +24,7 @@
24
24
  * values set via `agentc secrets set` are visible to the CLI.
25
25
  */
26
26
 
27
- import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
27
+ import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxCommandResult, SandboxProvider, ToolCallGateResult } from "../index.js";
28
28
  import { defineRuntime } from "../types/runtime.js";
29
29
  import { AsyncQueue } from "../agent/async-queue.js";
30
30
  import { formatError } from "../utils/errors.js";
@@ -34,6 +34,10 @@ import type { ProcessorContext, ToolCall } from "../processors/processor.js";
34
34
  import { RequestContext } from "../request-context/request-context.js";
35
35
  import { AcpClientPeer, ACP_PROTOCOL_VERSION } from "./_acp-client.js";
36
36
  import { isPauseSignal, boundProcessorPause } from "../pause/pause-core.js";
37
+ import {
38
+ JSONL_GUARD_LINE_CAP_BYTES, JSONL_GUARD_KEEP_BYTES,
39
+ jsonlGuardScript, wrapJsonlCommand,
40
+ } from "./_jsonl-guard.js";
37
41
 
38
42
  function now(): string { return new Date().toISOString(); }
39
43
 
@@ -42,6 +46,28 @@ export function shellQuote(value: string): string {
42
46
  return `'${value.replace(/'/g, `'\\''`)}'`;
43
47
  }
44
48
 
49
+ /** Monotonic discriminator for prompt-file names within this process. */
50
+ let promptFileCounter = 0;
51
+
52
+ /**
53
+ * A UNIQUE per-turn prompt path. Uniqueness is load-bearing, not cosmetic:
54
+ * E2B's envd cannot overwrite an existing non-root file in /tmp — it opens
55
+ * with O_CREAT as root (then chowns to the user), and the sandbox kernel's
56
+ * `fs.protected_regular` policy denies O_CREAT opens by a non-owner of an
57
+ * existing file in a sticky world-writable directory — so ANY reused path
58
+ * fails with "error opening file: … permission denied" on every turn after
59
+ * the first (same boot or across suspend/resume; reproduced against e2b
60
+ * 2.30.5). Cloud sessions construct a fresh runner per turn, so an
61
+ * iteration counter alone reuses `…-0.in` per conversation — the exact
62
+ * production failure. The timestamp+counter+random suffix is collision-proof
63
+ * across runner instances AND concurrent turns in one sandbox; the readable
64
+ * prefix (kind/agent/iteration) stays for debugging.
65
+ */
66
+ export function uniquePromptPath(kind: string, agentId: string, iteration: number): string {
67
+ const unique = `${Date.now().toString(36)}-${(promptFileCounter++).toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
68
+ return `/tmp/ac-${kind}-${agentId}-${iteration}-${unique}.in`;
69
+ }
70
+
45
71
  /** Internal sentinel yielded by `sendMessageAcp` to signal the version-check
46
72
  * fallback: the ACP handshake didn't negotiate version 1, so `sendMessage`
47
73
  * must drive the legacy JSONL path for this (and every later) turn. Not an
@@ -93,6 +119,16 @@ async function withHandshakeTimeout<T>(p: Promise<T>, ms: number): Promise<T> {
93
119
  }
94
120
  }
95
121
 
122
+ /** Reasoning-effort level a CLI turn may carry (T2 session effort) — the
123
+ * UNION of what the effort-capable CLIs accept. The per-CLI mapping lives in
124
+ * each spec's `buildCommand` — Claude Code takes its own `--effort` flag
125
+ * (all five levels), codex takes `-c model_reasoning_effort=<level>`
126
+ * (low|medium|high|xhigh — no "max"; the spec clamps it). Specs without a
127
+ * real knob (opencode/droid/cursor) never receive one: the server hides +
128
+ * rejects effort for those runtimes, and rejects levels a runtime lacks
129
+ * (sessionEffortLockError). */
130
+ export type CliReasoningEffort = "low" | "medium" | "high" | "xhigh" | "max";
131
+
96
132
  /** Per-CLI behaviour. The base owns the lifecycle (init/done/error) and the
97
133
  * transport (spawn + JSONL parse); a spec owns the CLI-specific bits. */
98
134
  export interface CliAgentSpec {
@@ -117,8 +153,10 @@ export interface CliAgentSpec {
117
153
  * JSONL user message for a `--stream-json-input` CLI (amp). */
118
154
  promptPayload(prompt: string): string;
119
155
  /** Build the one-shot shell command for a turn. `promptPath` is a file in the
120
- * sandbox holding `promptPayload(prompt)`; `sessionId` continues a thread. */
121
- buildCommand(args: { promptPath: string; sessionId?: string; model?: string; cwd?: string }): string;
156
+ * sandbox holding `promptPayload(prompt)`; `sessionId` continues a thread.
157
+ * `effort` is present only when the caller configured a reasoning effort
158
+ * AND the spec has a real knob for it (see `CliReasoningEffort`). */
159
+ buildCommand(args: { promptPath: string; sessionId?: string; model?: string; cwd?: string; effort?: CliReasoningEffort }): string;
122
160
  /** Map one parsed JSONL stdout event to `AgentMessage`s. The base emits
123
161
  * `init`/`done`/`error` lifecycle itself, so a spec maps only content +
124
162
  * usage (text / thinking / tool_use / tool_result / usage).
@@ -208,6 +246,7 @@ export class CliAgentRunner implements ModelExecutionContract {
208
246
  private readonly options: RuntimeOptions,
209
247
  private readonly spec: CliAgentSpec,
210
248
  private readonly configModel?: string,
249
+ private readonly configEffort?: CliReasoningEffort,
211
250
  ) {
212
251
  this.kind = spec.kind;
213
252
  }
@@ -529,10 +568,29 @@ export class CliAgentRunner implements ModelExecutionContract {
529
568
 
530
569
  /**
531
570
  * Legacy JSONL transport (the version-mismatch fallback, ADR-0020 increment
532
- * 5 retires it). Unchanged behaviour: write the prompt to a file, run the
571
+ * 5 retires it) — and the ONLY transport on server→sandbox providers (no
572
+ * `spawnDuplex`, so ACP never engages). Write the prompt to a file, run the
533
573
  * one-shot command, line-buffer stdout, `mapEvent` each parsed line, and cap
534
574
  * with the runner-synthesised `done`. Does NOT re-yield `init` (the caller
535
575
  * already did).
576
+ *
577
+ * Deadline + termination semantics (both load-bearing on the server→E2B
578
+ * cloud-session path):
579
+ * - `timeoutMs: 0` disables the PROVIDER's own command deadline. E2B's
580
+ * `CommandStartOpts` defaults to 60s — a fraction of a real coding
581
+ * turn — and connect-web normalises `<= 0` to "no deadline" (the same
582
+ * idiom `launchStep` uses for the hours-long step runner). The local
583
+ * and Vercel providers already treat 0/absent as "no deadline". The
584
+ * turn's deadline belongs to the CALLER: the abort signal here, the
585
+ * bridge waiter's idle/absolute caps, or the in-VM step deadline.
586
+ * - Abort and early generator exit KILL the in-sandbox CLI process via
587
+ * the provider's kill-capable background handle (`runBackground` —
588
+ * present on E2B, the provider every server-driven turn runs on), so a
589
+ * cancelled/timed-out/superseded turn can never leave an orphan agent
590
+ * mutating the working tree. Providers without `runBackground` (local
591
+ * child_process, Vercel) fall back to `commands.run`: no kill handle,
592
+ * abort only stops consumption — the pre-ACP behaviour of the in-VM
593
+ * JSONL fallback, bounded by the sandbox VM's own lifetime.
536
594
  */
537
595
  private async *sendMessageJsonl(opts: {
538
596
  prompt: string;
@@ -543,18 +601,43 @@ export class CliAgentRunner implements ModelExecutionContract {
543
601
  let sessionId = opts.sessionId;
544
602
  let sawError = false;
545
603
  try {
604
+ if (opts.signal?.aborted) return;
546
605
  // Provision the CLI before the first turn (no-op when it's already
547
606
  // present, e.g. booting from a "reuse" snapshot). A failure here surfaces
548
607
  // as an `error` AgentMessage via the catch below.
549
608
  await this.ensureInstalled();
550
609
 
551
- const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
610
+ const promptPath = uniquePromptPath(this.spec.kind, this.options.agentId ?? "agent", opts.iteration ?? 0);
552
611
  await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
553
612
  const cmd = this.spec.buildCommand({
554
613
  promptPath,
555
614
  sessionId: opts.sessionId,
556
615
  model: this.model,
557
616
  cwd: this.options.cwd,
617
+ effort: this.configEffort,
618
+ });
619
+
620
+ // Frame guard (see _jsonl-guard.ts): the CLI's stdout is piped through
621
+ // an in-sandbox node filter that caps any oversized JSONL line BEFORE
622
+ // it reaches the server→sandbox command stream — a single multi-hundred-
623
+ // KB tool_result line on E2B's connect-web stream dies as `13: [internal]
624
+ // protocol error: received unsupported compressed output`, killing the
625
+ // turn and recycling the VM. Truncated payloads carry an explicit
626
+ // elision marker (the full content stays durable in the sandbox); the
627
+ // CLI's real exit code rides a per-turn random sentinel line so the
628
+ // pipe never masks it. The script travels over `files.write` — envd's
629
+ // HTTP API, immune to the connect-web fault — and the wrapper itself
630
+ // falls back to the bare command when the sandbox has no `node`.
631
+ // Derived from promptPath so it inherits the collision-proof uniqueness
632
+ // (E2B's envd cannot overwrite an existing /tmp file — see
633
+ // uniquePromptPath) and the same opportunistic cleanup below.
634
+ const guardPath = `${promptPath}.guard.js`;
635
+ const sentinel = `__AC_JSONL_EXIT_${Math.random().toString(36).slice(2, 10)}__`;
636
+ await this.sandbox.files.write(guardPath, jsonlGuardScript());
637
+ const guardedCmd = wrapJsonlCommand({
638
+ cmd, guardPath, sentinel,
639
+ maxBytes: JSONL_GUARD_LINE_CAP_BYTES,
640
+ keepBytes: JSONL_GUARD_KEEP_BYTES,
558
641
  });
559
642
 
560
643
  // Bridge the streaming stdout callback into an async-iterable of complete
@@ -570,48 +653,90 @@ export class CliAgentRunner implements ModelExecutionContract {
570
653
  if (line) lines.push(line);
571
654
  }
572
655
  };
573
-
574
- // `commands.run` resolves when the process exits. Kick it off (don't await
575
- // yet); flush the trailing buffer + close the queue on completion so the
576
- // for-await below drains and we can read the exit code.
577
- const runPromise = this.sandbox.commands.run(cmd, {
656
+ const runOpts = {
578
657
  ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
579
658
  onStdout,
580
- }).then(
581
- (res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
582
- (err) => { lines.close(); throw err; },
583
- );
584
-
585
- for await (const line of lines) {
586
- let parsed: Record<string, unknown>;
587
- try {
588
- parsed = JSON.parse(line) as Record<string, unknown>;
589
- } catch {
590
- continue; // skip any non-JSON noise that lands on stdout
591
- }
592
- const sid = this.spec.extractSessionId(parsed);
593
- if (sid) sessionId = sid;
594
- for (const msg of this.spec.mapEvent(parsed)) {
595
- if (msg.type === "error") sawError = true;
596
- yield msg;
597
- }
659
+ timeoutMs: 0, // no provider deadline — see the doc comment above
660
+ };
661
+
662
+ // The command resolves when the process exits. Kick it off (don't await
663
+ // yet); flush the trailing buffer + close the queue on completion so the
664
+ // for-await below drains and we can read the exit code. Prefer the
665
+ // kill-capable background handle so abort/early-exit can terminate the
666
+ // process itself, not just stop reading its stream.
667
+ let exited = false;
668
+ const settle = (res: SandboxCommandResult): SandboxCommandResult => {
669
+ exited = true;
670
+ const tail = buf.trim();
671
+ if (tail) lines.push(tail);
672
+ lines.close();
673
+ return res;
674
+ };
675
+ const fail = (err: unknown): never => { exited = true; lines.close(); throw err; };
676
+ let kill: (() => Promise<void>) | undefined;
677
+ let runPromise: Promise<SandboxCommandResult>;
678
+ if (this.sandbox.commands.runBackground) {
679
+ const handle = await this.sandbox.commands.runBackground(guardedCmd, runOpts);
680
+ kill = () => handle.kill();
681
+ runPromise = handle.wait().then(settle, fail);
682
+ } else {
683
+ runPromise = this.sandbox.commands.run(guardedCmd, runOpts).then(settle, fail);
598
684
  }
685
+ // An early generator exit stops awaiting runPromise — keep its rejection
686
+ // observed so an aborted turn can never surface an unhandled rejection.
687
+ runPromise.catch(() => { /* observed via `await runPromise` when consumed */ });
688
+
689
+ const reap = () => { if (!exited) void kill?.().catch(() => { /* already gone */ }); };
690
+ if (opts.signal) {
691
+ if (opts.signal.aborted) reap();
692
+ else opts.signal.addEventListener("abort", reap, { once: true });
693
+ }
694
+ try {
695
+ for await (const line of lines) {
696
+ let parsed: Record<string, unknown>;
697
+ try {
698
+ parsed = JSON.parse(line) as Record<string, unknown>;
699
+ } catch {
700
+ continue; // skip any non-JSON noise that lands on stdout
701
+ }
702
+ const sid = this.spec.extractSessionId(parsed);
703
+ if (sid) sessionId = sid;
704
+ for (const msg of this.spec.mapEvent(parsed)) {
705
+ if (msg.type === "error") sawError = true;
706
+ yield msg;
707
+ }
708
+ }
599
709
 
600
- const res = await runPromise;
601
- if (res.exitCode !== 0 && !sawError) {
602
- const tail = (res.stderr ?? "").slice(-2000);
603
- yield { type: "error", text: `${this.spec.kind} exited with code ${res.exitCode}${tail ? `: ${tail}` : ""}`, timestamp: now() };
604
- return;
710
+ const res = await runPromise;
711
+ if (res.exitCode !== 0 && !sawError) {
712
+ const tail = (res.stderr ?? "").slice(-2000);
713
+ yield { type: "error", text: `${this.spec.kind} exited with code ${res.exitCode}${tail ? `: ${tail}` : ""}`, timestamp: now() };
714
+ return;
715
+ }
716
+ if (!sawError) yield { type: "done", sessionId: sessionId ?? "", timestamp: now() };
717
+ } finally {
718
+ // Runs on completion AND on early generator exit (the caller broke
719
+ // out of its for-await: abort, turn closed under the executor,
720
+ // supersede). `reap` is a no-op once the process has exited.
721
+ opts.signal?.removeEventListener("abort", reap);
722
+ reap();
723
+ // Opportunistic prompt-file + guard-script cleanup: paths are never
724
+ // reused (see uniquePromptPath), so this is hygiene, not correctness —
725
+ // fire and forget, and a dead sandbox / failed rm is fine. The shell
726
+ // holds an open fd on the redirect, so unlinking under a still-exiting
727
+ // CLI is harmless.
728
+ void this.sandbox.commands.run(
729
+ `rm -f ${shellQuote(promptPath)} ${shellQuote(guardPath)}`, { timeoutMs: 10_000 })
730
+ .catch(() => { /* best-effort */ });
605
731
  }
606
- if (!sawError) yield { type: "done", sessionId: sessionId ?? "", timestamp: now() };
607
732
  } catch (err) {
608
733
  yield { type: "error", text: formatError(err), timestamp: now() };
609
734
  }
610
735
  }
611
736
  }
612
737
 
613
- export function createCliAgentRuntime(spec: CliAgentSpec, configModel?: string) {
738
+ export function createCliAgentRuntime(spec: CliAgentSpec, configModel?: string, configEffort?: CliReasoningEffort) {
614
739
  return defineRuntime({
615
- create: (sandbox, opts) => new CliAgentRunner(sandbox, opts, spec, configModel),
740
+ create: (sandbox, opts) => new CliAgentRunner(sandbox, opts, spec, configModel, configEffort),
616
741
  });
617
742
  }
@@ -0,0 +1,219 @@
1
+ /**
2
+ * JSONL frame guard — keeps oversized lines OFF the server→sandbox command
3
+ * stream.
4
+ *
5
+ * The live failure this exists for: a cloud session's turn stream rides
6
+ * `sandbox.commands.run` over E2B's connect-web gRPC command stream, and that
7
+ * transport dies at the wire on large payloads — connect-web throws
8
+ * `13: [internal] protocol error: received unsupported compressed output`
9
+ * the moment an envelope arrives with the compressed flag set
10
+ * (@connectrpc/connect-web connect-transport.js), which is what E2B's edge
11
+ * does to it on big frames. A single `claude` stream-json event echoing a
12
+ * whole Read file (hundreds of KB on ONE JSONL line) is exactly such a
13
+ * payload: the stream faults mid-turn, the platform classifies it as a
14
+ * protocol transport fault (server cloud-executor.ts), recycles the VM, and
15
+ * the turn's work is lost.
16
+ *
17
+ * The structural fix: the full content is already durable INSIDE the sandbox
18
+ * (the file on disk, the CLI's own session transcript), so the wire never
19
+ * needs it. A tiny dependency-free node filter runs in-sandbox, piped after
20
+ * the CLI, and rewrites any oversized JSONL line before it can reach the
21
+ * vulnerable stream: long strings inside the event are truncated with an
22
+ * EXPLICIT elision marker (a silent cut is how models and humans end up
23
+ * trusting hallucinated file contents), and a line that cannot be capped
24
+ * structurally is hard-truncated so the stream survives at the cost of that
25
+ * one event. The prompt travels the other direction over `files.write` —
26
+ * E2B's envd HTTP API, a different transport documented immune to this
27
+ * failure (see the `files.read` comment in sandbox/providers/e2b.ts) — and
28
+ * the guard script itself is delivered the same immune way.
29
+ *
30
+ * Exit codes: piping would normally replace the CLI's exit code with the
31
+ * filter's, so the wrapper prints a per-turn random sentinel line carrying
32
+ * `$?` after the CLI exits and the filter re-raises it as its own exit code.
33
+ * EOF without a sentinel means the producer died abnormally (killed
34
+ * mid-pipe) — the filter exits JSONL_GUARD_NO_SENTINEL_EXIT so a crashed CLI
35
+ * can never masquerade as a clean turn.
36
+ *
37
+ * `capJsonlLine` is deliberately SELF-CONTAINED (no imports, no captured
38
+ * helpers): `jsonlGuardScript()` embeds it via `Function.prototype.toString`
39
+ * so the exact logic the unit tests pin is the logic that runs in-sandbox —
40
+ * one implementation, two call sites.
41
+ */
42
+
43
+ /** Trigger threshold: a JSONL line at or under this many bytes passes
44
+ * through byte-identical; over it, the guard rewrites. Chosen well under
45
+ * whatever E2B's edge chokes on (the observed kills were multi-hundred-KB
46
+ * single lines) while comfortably above any sane tool result. */
47
+ export const JSONL_GUARD_LINE_CAP_BYTES =
48
+ Number(process.env.AC_JSONL_LINE_CAP_BYTES) || 131_072;
49
+
50
+ /** How many bytes of an oversized string survive elision (the head — tool
51
+ * output fronts-loads the signal: file starts, command output starts). */
52
+ export const JSONL_GUARD_KEEP_BYTES =
53
+ Number(process.env.AC_JSONL_KEEP_BYTES) || 32_768;
54
+
55
+ /** The guard's exit code when its stdin hit EOF WITHOUT the exit sentinel:
56
+ * the producer pipeline died abnormally (killed / crashed sh), so the turn
57
+ * must surface as an error, never a clean done. */
58
+ export const JSONL_GUARD_NO_SENTINEL_EXIT = 121;
59
+
60
+ /** Stable prefix of the elision marker appended to every truncated string.
61
+ * Exported so consumers/tests can detect an elided payload; a unit test
62
+ * pins that `capJsonlLine`'s embedded copy matches this constant. */
63
+ export const JSONL_ELISION_MARKER_PREFIX = "[agent-compose: PAYLOAD ELIDED IN TRANSIT";
64
+
65
+ /**
66
+ * Cap one JSONL line to at most `maxBytes` bytes.
67
+ *
68
+ * - At or under `maxBytes`: returned byte-identical (never re-serialized).
69
+ * - Over it and parseable as a JSON object/array: every string longer than
70
+ * `keepBytes` is truncated to its first `keepBytes` bytes plus an honest
71
+ * elision marker telling the reader the full content still exists in the
72
+ * sandbox and how to get it. If the result still exceeds `maxBytes`
73
+ * (many big strings), the per-string budget shrinks 8× and then 64× before
74
+ * giving up.
75
+ * - Unparseable, or uncappable even at the smallest budget: hard byte
76
+ * truncation to `maxBytes`. The line stops being valid JSON, so the
77
+ * consumer's parse skips it — one event lost, stream alive (strictly
78
+ * better than the whole turn dying).
79
+ *
80
+ * SELF-CONTAINED by contract: no imports, no outer-scope references except
81
+ * globals (`Buffer`, `JSON`) — `jsonlGuardScript()` ships `this.toString()`
82
+ * into the sandbox. Keep it that way.
83
+ */
84
+ export function capJsonlLine(line: string, maxBytes: number, keepBytes: number): string {
85
+ if (Buffer.byteLength(line, "utf8") <= maxBytes) return line;
86
+ let parsed: unknown;
87
+ try {
88
+ parsed = JSON.parse(line);
89
+ } catch {
90
+ parsed = undefined;
91
+ }
92
+ if (parsed !== null && typeof parsed === "object") {
93
+ const truncate = (value: string, budget: number): string => {
94
+ const bytes = Buffer.byteLength(value, "utf8");
95
+ if (bytes <= budget) return value;
96
+ let head = Buffer.from(value, "utf8").subarray(0, budget).toString("utf8");
97
+ // A mid-codepoint cut decodes a trailing U+FFFD — drop it.
98
+ while (head.endsWith("�")) head = head.slice(0, -1);
99
+ // At tiny budgets (an event stuffed with MANY big strings) the full
100
+ // guidance would dwarf the kept content and defeat the shrink loop —
101
+ // keep the marker prefix (consumers key on it) but drop the prose.
102
+ if (budget < 2048) {
103
+ return head + "\n[agent-compose: PAYLOAD ELIDED IN TRANSIT - kept first "
104
+ + budget + " of " + bytes + " bytes; full content remains in the sandbox]";
105
+ }
106
+ return head
107
+ + "\n\n[agent-compose: PAYLOAD ELIDED IN TRANSIT - this content was " + bytes
108
+ + " bytes; only the first " + budget + " bytes are shown. It was cut to protect the"
109
+ + " session's transport, NOT because the source is short: the full content still exists"
110
+ + " inside this sandbox (the file on disk / the command's output). Do not guess the"
111
+ + " elided part - re-read the source file, or re-run the command with narrower output"
112
+ + " (head, tail, grep, offset+limit), to see it.]";
113
+ };
114
+ const walk = (node: unknown, budget: number): unknown => {
115
+ if (typeof node === "string") return truncate(node, budget);
116
+ if (Array.isArray(node)) return node.map((item) => walk(item, budget));
117
+ if (node !== null && typeof node === "object") {
118
+ const out: Record<string, unknown> = {};
119
+ for (const key of Object.keys(node as Record<string, unknown>)) {
120
+ out[key] = walk((node as Record<string, unknown>)[key], budget);
121
+ }
122
+ return out;
123
+ }
124
+ return node;
125
+ };
126
+ for (const shrink of [1, 8, 64]) {
127
+ const budget = Math.max(64, Math.floor(keepBytes / shrink));
128
+ try {
129
+ const capped = JSON.stringify(walk(parsed, budget));
130
+ if (typeof capped === "string" && Buffer.byteLength(capped, "utf8") <= maxBytes) {
131
+ return capped;
132
+ }
133
+ } catch {
134
+ break; // structural failure — fall through to the raw truncate
135
+ }
136
+ }
137
+ }
138
+ let raw = Buffer.from(line, "utf8").subarray(0, maxBytes).toString("utf8");
139
+ while (raw.endsWith("�")) raw = raw.slice(0, -1);
140
+ return raw;
141
+ }
142
+
143
+ /**
144
+ * The in-sandbox filter's source: a dependency-free node script that
145
+ * line-buffers stdin, re-emits every line through `capJsonlLine`, and turns
146
+ * the exit sentinel into its own exit code.
147
+ *
148
+ * node <script> <sentinel> <maxBytes> <keepBytes>
149
+ */
150
+ export function jsonlGuardScript(): string {
151
+ return [
152
+ "'use strict';",
153
+ `const capJsonlLine = ${capJsonlLine.toString()};`,
154
+ `const NO_SENTINEL_EXIT = ${JSONL_GUARD_NO_SENTINEL_EXIT};`,
155
+ "const sentinel = process.argv[2];",
156
+ "const maxBytes = Number(process.argv[3]);",
157
+ "const keepBytes = Number(process.argv[4]);",
158
+ "let sawSentinel = false;",
159
+ "let buf = '';",
160
+ "// Downstream (the command-stream consumer) going away must never crash",
161
+ "// the pipe with EPIPE noise — the producer is being killed anyway.",
162
+ "process.stdout.on('error', () => process.exit(0));",
163
+ "function handle(line) {",
164
+ " const t = line.trim();",
165
+ " if (sentinel && t.startsWith(sentinel)) {",
166
+ " const code = Number(t.slice(sentinel.length).trim());",
167
+ " process.exitCode = Number.isFinite(code) ? code : 0;",
168
+ " sawSentinel = true;",
169
+ " return;",
170
+ " }",
171
+ " process.stdout.write(capJsonlLine(line, maxBytes, keepBytes) + '\\n');",
172
+ "}",
173
+ "process.stdin.setEncoding('utf8');",
174
+ "process.stdin.on('data', (d) => {",
175
+ " buf += d;",
176
+ " let nl;",
177
+ " while ((nl = buf.indexOf('\\n')) >= 0) {",
178
+ " const line = buf.slice(0, nl);",
179
+ " buf = buf.slice(nl + 1);",
180
+ " if (line.length > 0) handle(line);",
181
+ " }",
182
+ "});",
183
+ "process.stdin.on('end', () => {",
184
+ " if (buf.trim().length > 0) handle(buf);",
185
+ " if (!sawSentinel) process.exitCode = NO_SENTINEL_EXIT;",
186
+ "});",
187
+ ].join("\n");
188
+ }
189
+
190
+ /** Single-quote a value for a `sh -c` command line (local copy — this module
191
+ * must not import from _cli-agent.ts, which imports it). */
192
+ function quote(value: string): string {
193
+ return `'${value.replace(/'/g, `'\\''`)}'`;
194
+ }
195
+
196
+ /**
197
+ * Wrap a turn command so its stdout crosses the command stream through the
198
+ * guard. Self-deciding in-sandbox: when `node` is absent (a bare sandbox a
199
+ * self-provisioned CLI runs on), the command runs unguarded — the pre-guard
200
+ * status quo — instead of failing every turn.
201
+ *
202
+ * The command runs in a SUBSHELL so a `$?`-carrying sentinel line always
203
+ * prints after it exits, however it exits (a brace group would let a shell
204
+ * `exit` skip the printf). stderr stays un-piped — the provider's stderr
205
+ * accumulation is unchanged.
206
+ */
207
+ export function wrapJsonlCommand(args: {
208
+ cmd: string;
209
+ guardPath: string;
210
+ sentinel: string;
211
+ maxBytes: number;
212
+ keepBytes: number;
213
+ }): string {
214
+ const { cmd, guardPath, sentinel, maxBytes, keepBytes } = args;
215
+ return `if command -v node >/dev/null 2>&1; then `
216
+ + `{ ( ${cmd} ); printf '\\n%s %s\\n' ${quote(sentinel)} "$?"; } `
217
+ + `| node ${quote(guardPath)} ${quote(sentinel)} ${maxBytes} ${keepBytes}; `
218
+ + `else ${cmd}; fi`;
219
+ }