@agent-compose/sdk 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +24 -1
  5. package/dist/client.d.ts +338 -534
  6. package/dist/directives.d.ts +112 -0
  7. package/dist/display.d.ts +242 -0
  8. package/dist/errors.d.ts +24 -1
  9. package/dist/index.d.ts +24 -12
  10. package/dist/index.js +3545 -1667
  11. package/dist/pause/wrappers.d.ts +31 -9
  12. package/dist/runtimes/_acp-client.d.ts +46 -1
  13. package/dist/runtimes/_cli-agent.d.ts +49 -4
  14. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  15. package/dist/runtimes/amp.d.ts +2 -2
  16. package/dist/runtimes/claude-code.d.ts +59 -0
  17. package/dist/runtimes/claude-code.test.d.ts +14 -0
  18. package/dist/runtimes/claude.d.ts +16 -0
  19. package/dist/runtimes/claude.test.d.ts +8 -0
  20. package/dist/runtimes/codex.d.ts +9 -3
  21. package/dist/runtimes/cursor.d.ts +2 -2
  22. package/dist/runtimes/droid.d.ts +2 -2
  23. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  24. package/dist/runtimes/openai-desktop.js +2691 -864
  25. package/dist/runtimes/opencode.d.ts +2 -2
  26. package/dist/runtimes/vercel.js +12 -1
  27. package/dist/sandbox/devbox.d.ts +42 -0
  28. package/dist/sandbox/exec-stream.d.ts +14 -0
  29. package/dist/sandbox/network-policy.d.ts +100 -0
  30. package/dist/sandbox/provider-def.d.ts +79 -0
  31. package/dist/sandbox/providers/desktop.d.ts +10 -0
  32. package/dist/sandbox/providers/e2b.d.ts +17 -0
  33. package/dist/sandbox/providers/local.d.ts +11 -0
  34. package/dist/sandbox/providers/vercel.d.ts +18 -0
  35. package/dist/sandbox/registry.d.ts +45 -0
  36. package/dist/sandbox/sizes.d.ts +68 -0
  37. package/dist/sandbox.d.ts +24 -299
  38. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  39. package/dist/step-invocation/invoker.d.ts +10 -0
  40. package/dist/step-invocation/protocol.d.ts +5 -0
  41. package/dist/types/api-compliance.d.ts +71 -0
  42. package/dist/types/api-conversations.d.ts +492 -0
  43. package/dist/types/api-factory.d.ts +309 -0
  44. package/dist/types/api-projects.d.ts +131 -0
  45. package/dist/types/api-runs.d.ts +377 -0
  46. package/dist/types/api-scopes.d.ts +102 -0
  47. package/dist/types/conversation-stream.d.ts +191 -0
  48. package/dist/types/execution-context.d.ts +12 -2
  49. package/dist/types/protocol.d.ts +30 -1
  50. package/dist/types/sandbox-environment.d.ts +8 -5
  51. package/dist/types/sandbox.d.ts +74 -4
  52. package/dist/types/workflow-metadata.d.ts +33 -8
  53. package/dist/types/workflow-plan.d.ts +10 -0
  54. package/dist/types/workflow.d.ts +18 -205
  55. package/dist/utils/bundler.d.ts +12 -1
  56. package/dist/workflow-steps/index.d.ts +1 -1
  57. package/dist/workflow-steps/observability.d.ts +8 -1
  58. package/dist/workflow-steps/runner.d.ts +3 -3
  59. package/dist/workflow-steps/step.d.ts +15 -1
  60. package/dist/workflow-steps/types.d.ts +19 -5
  61. package/dist/workflow-steps/workflow.d.ts +22 -1
  62. package/dist/workflows/engine.d.ts +3 -2
  63. package/dist/workflows/invoke-child.d.ts +2 -2
  64. package/package.json +1 -1
  65. package/src/agent/agent-context.ts +186 -3
  66. package/src/agent/agent-loop.ts +31 -2
  67. package/src/client.ts +909 -621
  68. package/src/directives.ts +184 -0
  69. package/src/display.ts +788 -0
  70. package/src/errors.ts +39 -0
  71. package/src/index.ts +104 -10
  72. package/src/pause/wrappers.ts +44 -9
  73. package/src/runtimes/_acp-client.ts +72 -3
  74. package/src/runtimes/_cli-agent.ts +159 -36
  75. package/src/runtimes/_jsonl-guard.ts +219 -0
  76. package/src/runtimes/claude-code.ts +246 -0
  77. package/src/runtimes/claude.ts +32 -2
  78. package/src/runtimes/codex.ts +55 -3
  79. package/src/runtimes/openai-desktop.ts +59 -14
  80. package/src/sandbox/devbox.ts +48 -0
  81. package/src/sandbox/exec-stream.ts +48 -0
  82. package/src/sandbox/network-policy.ts +181 -0
  83. package/src/sandbox/provider-def.ts +94 -0
  84. package/src/sandbox/providers/desktop.ts +57 -0
  85. package/src/sandbox/providers/e2b.ts +354 -0
  86. package/src/sandbox/providers/local.ts +106 -0
  87. package/src/sandbox/providers/vercel.ts +331 -0
  88. package/src/sandbox/registry.ts +198 -0
  89. package/src/sandbox/sizes.ts +95 -0
  90. package/src/sandbox.ts +59 -1275
  91. package/src/step-invocation/invoker.ts +151 -28
  92. package/src/step-invocation/protocol.ts +8 -0
  93. package/src/types/api-compliance.ts +79 -0
  94. package/src/types/api-conversations.ts +522 -0
  95. package/src/types/api-factory.ts +336 -0
  96. package/src/types/api-projects.ts +140 -0
  97. package/src/types/api-runs.ts +412 -0
  98. package/src/types/api-scopes.ts +102 -0
  99. package/src/types/conversation-stream.ts +231 -0
  100. package/src/types/execution-context.ts +10 -2
  101. package/src/types/protocol.ts +33 -0
  102. package/src/types/sandbox-environment.ts +28 -9
  103. package/src/types/sandbox.ts +73 -4
  104. package/src/types/workflow-metadata.ts +35 -8
  105. package/src/types/workflow-plan.ts +11 -0
  106. package/src/types/workflow.ts +25 -292
  107. package/src/utils/bundler.ts +32 -5
  108. package/src/utils/errors.ts +16 -1
  109. package/src/workflow-steps/index.ts +1 -0
  110. package/src/workflow-steps/observability.ts +19 -8
  111. package/src/workflow-steps/runner.ts +4 -4
  112. package/src/workflow-steps/step.ts +49 -1
  113. package/src/workflow-steps/types.ts +20 -5
  114. package/src/workflow-steps/workflow.ts +22 -1
  115. package/src/workflows/engine.ts +3 -2
  116. package/src/workflows/invoke-child.ts +2 -2
@@ -24,7 +24,7 @@
24
24
  * values set via `agentc secrets set` are visible to the CLI.
25
25
  */
26
26
 
27
- import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
27
+ import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxCommandResult, SandboxProvider, ToolCallGateResult } from "../index.js";
28
28
  import { defineRuntime } from "../types/runtime.js";
29
29
  import { AsyncQueue } from "../agent/async-queue.js";
30
30
  import { formatError } from "../utils/errors.js";
@@ -34,6 +34,10 @@ import type { ProcessorContext, ToolCall } from "../processors/processor.js";
34
34
  import { RequestContext } from "../request-context/request-context.js";
35
35
  import { AcpClientPeer, ACP_PROTOCOL_VERSION } from "./_acp-client.js";
36
36
  import { isPauseSignal, boundProcessorPause } from "../pause/pause-core.js";
37
+ import {
38
+ JSONL_GUARD_LINE_CAP_BYTES, JSONL_GUARD_KEEP_BYTES,
39
+ jsonlGuardScript, wrapJsonlCommand,
40
+ } from "./_jsonl-guard.js";
37
41
 
38
42
  function now(): string { return new Date().toISOString(); }
39
43
 
@@ -42,6 +46,28 @@ export function shellQuote(value: string): string {
42
46
  return `'${value.replace(/'/g, `'\\''`)}'`;
43
47
  }
44
48
 
49
+ /** Monotonic discriminator for prompt-file names within this process. */
50
+ let promptFileCounter = 0;
51
+
52
+ /**
53
+ * A UNIQUE per-turn prompt path. Uniqueness is load-bearing, not cosmetic:
54
+ * E2B's envd cannot overwrite an existing non-root file in /tmp — it opens
55
+ * with O_CREAT as root (then chowns to the user), and the sandbox kernel's
56
+ * `fs.protected_regular` policy denies O_CREAT opens by a non-owner of an
57
+ * existing file in a sticky world-writable directory — so ANY reused path
58
+ * fails with "error opening file: … permission denied" on every turn after
59
+ * the first (same boot or across suspend/resume; reproduced against e2b
60
+ * 2.30.5). Cloud sessions construct a fresh runner per turn, so an
61
+ * iteration counter alone reuses `…-0.in` per conversation — the exact
62
+ * production failure. The timestamp+counter+random suffix is collision-proof
63
+ * across runner instances AND concurrent turns in one sandbox; the readable
64
+ * prefix (kind/agent/iteration) stays for debugging.
65
+ */
66
+ export function uniquePromptPath(kind: string, agentId: string, iteration: number): string {
67
+ const unique = `${Date.now().toString(36)}-${(promptFileCounter++).toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
68
+ return `/tmp/ac-${kind}-${agentId}-${iteration}-${unique}.in`;
69
+ }
70
+
45
71
  /** Internal sentinel yielded by `sendMessageAcp` to signal the version-check
46
72
  * fallback: the ACP handshake didn't negotiate version 1, so `sendMessage`
47
73
  * must drive the legacy JSONL path for this (and every later) turn. Not an
@@ -93,6 +119,14 @@ async function withHandshakeTimeout<T>(p: Promise<T>, ms: number): Promise<T> {
93
119
  }
94
120
  }
95
121
 
122
+ /** Reasoning-effort level a CLI turn may carry (T2 session effort). The
123
+ * per-CLI mapping lives in each spec's `buildCommand` — Claude Code takes a
124
+ * thinking-token budget via `MAX_THINKING_TOKENS`, codex takes
125
+ * `-c model_reasoning_effort=<level>`. Specs without a real knob
126
+ * (opencode/droid/cursor) never receive one: the server hides + rejects
127
+ * effort for those runtimes. */
128
+ export type CliReasoningEffort = "low" | "medium" | "high";
129
+
96
130
  /** Per-CLI behaviour. The base owns the lifecycle (init/done/error) and the
97
131
  * transport (spawn + JSONL parse); a spec owns the CLI-specific bits. */
98
132
  export interface CliAgentSpec {
@@ -117,8 +151,10 @@ export interface CliAgentSpec {
117
151
  * JSONL user message for a `--stream-json-input` CLI (amp). */
118
152
  promptPayload(prompt: string): string;
119
153
  /** Build the one-shot shell command for a turn. `promptPath` is a file in the
120
- * sandbox holding `promptPayload(prompt)`; `sessionId` continues a thread. */
121
- buildCommand(args: { promptPath: string; sessionId?: string; model?: string; cwd?: string }): string;
154
+ * sandbox holding `promptPayload(prompt)`; `sessionId` continues a thread.
155
+ * `effort` is present only when the caller configured a reasoning effort
156
+ * AND the spec has a real knob for it (see `CliReasoningEffort`). */
157
+ buildCommand(args: { promptPath: string; sessionId?: string; model?: string; cwd?: string; effort?: CliReasoningEffort }): string;
122
158
  /** Map one parsed JSONL stdout event to `AgentMessage`s. The base emits
123
159
  * `init`/`done`/`error` lifecycle itself, so a spec maps only content +
124
160
  * usage (text / thinking / tool_use / tool_result / usage).
@@ -208,6 +244,7 @@ export class CliAgentRunner implements ModelExecutionContract {
208
244
  private readonly options: RuntimeOptions,
209
245
  private readonly spec: CliAgentSpec,
210
246
  private readonly configModel?: string,
247
+ private readonly configEffort?: CliReasoningEffort,
211
248
  ) {
212
249
  this.kind = spec.kind;
213
250
  }
@@ -529,10 +566,29 @@ export class CliAgentRunner implements ModelExecutionContract {
529
566
 
530
567
  /**
531
568
  * Legacy JSONL transport (the version-mismatch fallback, ADR-0020 increment
532
- * 5 retires it). Unchanged behaviour: write the prompt to a file, run the
569
+ * 5 retires it) — and the ONLY transport on server→sandbox providers (no
570
+ * `spawnDuplex`, so ACP never engages). Write the prompt to a file, run the
533
571
  * one-shot command, line-buffer stdout, `mapEvent` each parsed line, and cap
534
572
  * with the runner-synthesised `done`. Does NOT re-yield `init` (the caller
535
573
  * already did).
574
+ *
575
+ * Deadline + termination semantics (both load-bearing on the server→E2B
576
+ * cloud-session path):
577
+ * - `timeoutMs: 0` disables the PROVIDER's own command deadline. E2B's
578
+ * `CommandStartOpts` defaults to 60s — a fraction of a real coding
579
+ * turn — and connect-web normalises `<= 0` to "no deadline" (the same
580
+ * idiom `launchStep` uses for the hours-long step runner). The local
581
+ * and Vercel providers already treat 0/absent as "no deadline". The
582
+ * turn's deadline belongs to the CALLER: the abort signal here, the
583
+ * bridge waiter's idle/absolute caps, or the in-VM step deadline.
584
+ * - Abort and early generator exit KILL the in-sandbox CLI process via
585
+ * the provider's kill-capable background handle (`runBackground` —
586
+ * present on E2B, the provider every server-driven turn runs on), so a
587
+ * cancelled/timed-out/superseded turn can never leave an orphan agent
588
+ * mutating the working tree. Providers without `runBackground` (local
589
+ * child_process, Vercel) fall back to `commands.run`: no kill handle,
590
+ * abort only stops consumption — the pre-ACP behaviour of the in-VM
591
+ * JSONL fallback, bounded by the sandbox VM's own lifetime.
536
592
  */
537
593
  private async *sendMessageJsonl(opts: {
538
594
  prompt: string;
@@ -543,18 +599,43 @@ export class CliAgentRunner implements ModelExecutionContract {
543
599
  let sessionId = opts.sessionId;
544
600
  let sawError = false;
545
601
  try {
602
+ if (opts.signal?.aborted) return;
546
603
  // Provision the CLI before the first turn (no-op when it's already
547
604
  // present, e.g. booting from a "reuse" snapshot). A failure here surfaces
548
605
  // as an `error` AgentMessage via the catch below.
549
606
  await this.ensureInstalled();
550
607
 
551
- const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
608
+ const promptPath = uniquePromptPath(this.spec.kind, this.options.agentId ?? "agent", opts.iteration ?? 0);
552
609
  await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
553
610
  const cmd = this.spec.buildCommand({
554
611
  promptPath,
555
612
  sessionId: opts.sessionId,
556
613
  model: this.model,
557
614
  cwd: this.options.cwd,
615
+ effort: this.configEffort,
616
+ });
617
+
618
+ // Frame guard (see _jsonl-guard.ts): the CLI's stdout is piped through
619
+ // an in-sandbox node filter that caps any oversized JSONL line BEFORE
620
+ // it reaches the server→sandbox command stream — a single multi-hundred-
621
+ // KB tool_result line on E2B's connect-web stream dies as `13: [internal]
622
+ // protocol error: received unsupported compressed output`, killing the
623
+ // turn and recycling the VM. Truncated payloads carry an explicit
624
+ // elision marker (the full content stays durable in the sandbox); the
625
+ // CLI's real exit code rides a per-turn random sentinel line so the
626
+ // pipe never masks it. The script travels over `files.write` — envd's
627
+ // HTTP API, immune to the connect-web fault — and the wrapper itself
628
+ // falls back to the bare command when the sandbox has no `node`.
629
+ // Derived from promptPath so it inherits the collision-proof uniqueness
630
+ // (E2B's envd cannot overwrite an existing /tmp file — see
631
+ // uniquePromptPath) and the same opportunistic cleanup below.
632
+ const guardPath = `${promptPath}.guard.js`;
633
+ const sentinel = `__AC_JSONL_EXIT_${Math.random().toString(36).slice(2, 10)}__`;
634
+ await this.sandbox.files.write(guardPath, jsonlGuardScript());
635
+ const guardedCmd = wrapJsonlCommand({
636
+ cmd, guardPath, sentinel,
637
+ maxBytes: JSONL_GUARD_LINE_CAP_BYTES,
638
+ keepBytes: JSONL_GUARD_KEEP_BYTES,
558
639
  });
559
640
 
560
641
  // Bridge the streaming stdout callback into an async-iterable of complete
@@ -570,48 +651,90 @@ export class CliAgentRunner implements ModelExecutionContract {
570
651
  if (line) lines.push(line);
571
652
  }
572
653
  };
573
-
574
- // `commands.run` resolves when the process exits. Kick it off (don't await
575
- // yet); flush the trailing buffer + close the queue on completion so the
576
- // for-await below drains and we can read the exit code.
577
- const runPromise = this.sandbox.commands.run(cmd, {
654
+ const runOpts = {
578
655
  ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
579
656
  onStdout,
580
- }).then(
581
- (res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
582
- (err) => { lines.close(); throw err; },
583
- );
584
-
585
- for await (const line of lines) {
586
- let parsed: Record<string, unknown>;
587
- try {
588
- parsed = JSON.parse(line) as Record<string, unknown>;
589
- } catch {
590
- continue; // skip any non-JSON noise that lands on stdout
591
- }
592
- const sid = this.spec.extractSessionId(parsed);
593
- if (sid) sessionId = sid;
594
- for (const msg of this.spec.mapEvent(parsed)) {
595
- if (msg.type === "error") sawError = true;
596
- yield msg;
597
- }
657
+ timeoutMs: 0, // no provider deadline — see the doc comment above
658
+ };
659
+
660
+ // The command resolves when the process exits. Kick it off (don't await
661
+ // yet); flush the trailing buffer + close the queue on completion so the
662
+ // for-await below drains and we can read the exit code. Prefer the
663
+ // kill-capable background handle so abort/early-exit can terminate the
664
+ // process itself, not just stop reading its stream.
665
+ let exited = false;
666
+ const settle = (res: SandboxCommandResult): SandboxCommandResult => {
667
+ exited = true;
668
+ const tail = buf.trim();
669
+ if (tail) lines.push(tail);
670
+ lines.close();
671
+ return res;
672
+ };
673
+ const fail = (err: unknown): never => { exited = true; lines.close(); throw err; };
674
+ let kill: (() => Promise<void>) | undefined;
675
+ let runPromise: Promise<SandboxCommandResult>;
676
+ if (this.sandbox.commands.runBackground) {
677
+ const handle = await this.sandbox.commands.runBackground(guardedCmd, runOpts);
678
+ kill = () => handle.kill();
679
+ runPromise = handle.wait().then(settle, fail);
680
+ } else {
681
+ runPromise = this.sandbox.commands.run(guardedCmd, runOpts).then(settle, fail);
598
682
  }
683
+ // An early generator exit stops awaiting runPromise — keep its rejection
684
+ // observed so an aborted turn can never surface an unhandled rejection.
685
+ runPromise.catch(() => { /* observed via `await runPromise` when consumed */ });
686
+
687
+ const reap = () => { if (!exited) void kill?.().catch(() => { /* already gone */ }); };
688
+ if (opts.signal) {
689
+ if (opts.signal.aborted) reap();
690
+ else opts.signal.addEventListener("abort", reap, { once: true });
691
+ }
692
+ try {
693
+ for await (const line of lines) {
694
+ let parsed: Record<string, unknown>;
695
+ try {
696
+ parsed = JSON.parse(line) as Record<string, unknown>;
697
+ } catch {
698
+ continue; // skip any non-JSON noise that lands on stdout
699
+ }
700
+ const sid = this.spec.extractSessionId(parsed);
701
+ if (sid) sessionId = sid;
702
+ for (const msg of this.spec.mapEvent(parsed)) {
703
+ if (msg.type === "error") sawError = true;
704
+ yield msg;
705
+ }
706
+ }
599
707
 
600
- const res = await runPromise;
601
- if (res.exitCode !== 0 && !sawError) {
602
- const tail = (res.stderr ?? "").slice(-2000);
603
- yield { type: "error", text: `${this.spec.kind} exited with code ${res.exitCode}${tail ? `: ${tail}` : ""}`, timestamp: now() };
604
- return;
708
+ const res = await runPromise;
709
+ if (res.exitCode !== 0 && !sawError) {
710
+ const tail = (res.stderr ?? "").slice(-2000);
711
+ yield { type: "error", text: `${this.spec.kind} exited with code ${res.exitCode}${tail ? `: ${tail}` : ""}`, timestamp: now() };
712
+ return;
713
+ }
714
+ if (!sawError) yield { type: "done", sessionId: sessionId ?? "", timestamp: now() };
715
+ } finally {
716
+ // Runs on completion AND on early generator exit (the caller broke
717
+ // out of its for-await: abort, turn closed under the executor,
718
+ // supersede). `reap` is a no-op once the process has exited.
719
+ opts.signal?.removeEventListener("abort", reap);
720
+ reap();
721
+ // Opportunistic prompt-file + guard-script cleanup: paths are never
722
+ // reused (see uniquePromptPath), so this is hygiene, not correctness —
723
+ // fire and forget, and a dead sandbox / failed rm is fine. The shell
724
+ // holds an open fd on the redirect, so unlinking under a still-exiting
725
+ // CLI is harmless.
726
+ void this.sandbox.commands.run(
727
+ `rm -f ${shellQuote(promptPath)} ${shellQuote(guardPath)}`, { timeoutMs: 10_000 })
728
+ .catch(() => { /* best-effort */ });
605
729
  }
606
- if (!sawError) yield { type: "done", sessionId: sessionId ?? "", timestamp: now() };
607
730
  } catch (err) {
608
731
  yield { type: "error", text: formatError(err), timestamp: now() };
609
732
  }
610
733
  }
611
734
  }
612
735
 
613
- export function createCliAgentRuntime(spec: CliAgentSpec, configModel?: string) {
736
+ export function createCliAgentRuntime(spec: CliAgentSpec, configModel?: string, configEffort?: CliReasoningEffort) {
614
737
  return defineRuntime({
615
- create: (sandbox, opts) => new CliAgentRunner(sandbox, opts, spec, configModel),
738
+ create: (sandbox, opts) => new CliAgentRunner(sandbox, opts, spec, configModel, configEffort),
616
739
  });
617
740
  }
@@ -0,0 +1,219 @@
1
+ /**
2
+ * JSONL frame guard — keeps oversized lines OFF the server→sandbox command
3
+ * stream.
4
+ *
5
+ * The live failure this exists for: a cloud session's turn stream rides
6
+ * `sandbox.commands.run` over E2B's connect-web gRPC command stream, and that
7
+ * transport dies at the wire on large payloads — connect-web throws
8
+ * `13: [internal] protocol error: received unsupported compressed output`
9
+ * the moment an envelope arrives with the compressed flag set
10
+ * (@connectrpc/connect-web connect-transport.js), which is what E2B's edge
11
+ * does to it on big frames. A single `claude` stream-json event echoing a
12
+ * whole Read file (hundreds of KB on ONE JSONL line) is exactly such a
13
+ * payload: the stream faults mid-turn, the platform classifies it as a
14
+ * protocol transport fault (server cloud-executor.ts), recycles the VM, and
15
+ * the turn's work is lost.
16
+ *
17
+ * The structural fix: the full content is already durable INSIDE the sandbox
18
+ * (the file on disk, the CLI's own session transcript), so the wire never
19
+ * needs it. A tiny dependency-free node filter runs in-sandbox, piped after
20
+ * the CLI, and rewrites any oversized JSONL line before it can reach the
21
+ * vulnerable stream: long strings inside the event are truncated with an
22
+ * EXPLICIT elision marker (a silent cut is how models and humans end up
23
+ * trusting hallucinated file contents), and a line that cannot be capped
24
+ * structurally is hard-truncated so the stream survives at the cost of that
25
+ * one event. The prompt travels the other direction over `files.write` —
26
+ * E2B's envd HTTP API, a different transport documented immune to this
27
+ * failure (see the `files.read` comment in sandbox/providers/e2b.ts) — and
28
+ * the guard script itself is delivered the same immune way.
29
+ *
30
+ * Exit codes: piping would normally replace the CLI's exit code with the
31
+ * filter's, so the wrapper prints a per-turn random sentinel line carrying
32
+ * `$?` after the CLI exits and the filter re-raises it as its own exit code.
33
+ * EOF without a sentinel means the producer died abnormally (killed
34
+ * mid-pipe) — the filter exits JSONL_GUARD_NO_SENTINEL_EXIT so a crashed CLI
35
+ * can never masquerade as a clean turn.
36
+ *
37
+ * `capJsonlLine` is deliberately SELF-CONTAINED (no imports, no captured
38
+ * helpers): `jsonlGuardScript()` embeds it via `Function.prototype.toString`
39
+ * so the exact logic the unit tests pin is the logic that runs in-sandbox —
40
+ * one implementation, two call sites.
41
+ */
42
+
43
+ /** Trigger threshold: a JSONL line at or under this many bytes passes
44
+ * through byte-identical; over it, the guard rewrites. Chosen well under
45
+ * whatever E2B's edge chokes on (the observed kills were multi-hundred-KB
46
+ * single lines) while comfortably above any sane tool result. */
47
+ export const JSONL_GUARD_LINE_CAP_BYTES =
48
+ Number(process.env.AC_JSONL_LINE_CAP_BYTES) || 131_072;
49
+
50
+ /** How many bytes of an oversized string survive elision (the head — tool
51
+ * output fronts-loads the signal: file starts, command output starts). */
52
+ export const JSONL_GUARD_KEEP_BYTES =
53
+ Number(process.env.AC_JSONL_KEEP_BYTES) || 32_768;
54
+
55
+ /** The guard's exit code when its stdin hit EOF WITHOUT the exit sentinel:
56
+ * the producer pipeline died abnormally (killed / crashed sh), so the turn
57
+ * must surface as an error, never a clean done. */
58
+ export const JSONL_GUARD_NO_SENTINEL_EXIT = 121;
59
+
60
+ /** Stable prefix of the elision marker appended to every truncated string.
61
+ * Exported so consumers/tests can detect an elided payload; a unit test
62
+ * pins that `capJsonlLine`'s embedded copy matches this constant. */
63
+ export const JSONL_ELISION_MARKER_PREFIX = "[agent-compose: PAYLOAD ELIDED IN TRANSIT";
64
+
65
+ /**
66
+ * Cap one JSONL line to at most `maxBytes` bytes.
67
+ *
68
+ * - At or under `maxBytes`: returned byte-identical (never re-serialized).
69
+ * - Over it and parseable as a JSON object/array: every string longer than
70
+ * `keepBytes` is truncated to its first `keepBytes` bytes plus an honest
71
+ * elision marker telling the reader the full content still exists in the
72
+ * sandbox and how to get it. If the result still exceeds `maxBytes`
73
+ * (many big strings), the per-string budget shrinks 8× and then 64× before
74
+ * giving up.
75
+ * - Unparseable, or uncappable even at the smallest budget: hard byte
76
+ * truncation to `maxBytes`. The line stops being valid JSON, so the
77
+ * consumer's parse skips it — one event lost, stream alive (strictly
78
+ * better than the whole turn dying).
79
+ *
80
+ * SELF-CONTAINED by contract: no imports, no outer-scope references except
81
+ * globals (`Buffer`, `JSON`) — `jsonlGuardScript()` ships `this.toString()`
82
+ * into the sandbox. Keep it that way.
83
+ */
84
+ export function capJsonlLine(line: string, maxBytes: number, keepBytes: number): string {
85
+ if (Buffer.byteLength(line, "utf8") <= maxBytes) return line;
86
+ let parsed: unknown;
87
+ try {
88
+ parsed = JSON.parse(line);
89
+ } catch {
90
+ parsed = undefined;
91
+ }
92
+ if (parsed !== null && typeof parsed === "object") {
93
+ const truncate = (value: string, budget: number): string => {
94
+ const bytes = Buffer.byteLength(value, "utf8");
95
+ if (bytes <= budget) return value;
96
+ let head = Buffer.from(value, "utf8").subarray(0, budget).toString("utf8");
97
+ // A mid-codepoint cut decodes a trailing U+FFFD — drop it.
98
+ while (head.endsWith("�")) head = head.slice(0, -1);
99
+ // At tiny budgets (an event stuffed with MANY big strings) the full
100
+ // guidance would dwarf the kept content and defeat the shrink loop —
101
+ // keep the marker prefix (consumers key on it) but drop the prose.
102
+ if (budget < 2048) {
103
+ return head + "\n[agent-compose: PAYLOAD ELIDED IN TRANSIT - kept first "
104
+ + budget + " of " + bytes + " bytes; full content remains in the sandbox]";
105
+ }
106
+ return head
107
+ + "\n\n[agent-compose: PAYLOAD ELIDED IN TRANSIT - this content was " + bytes
108
+ + " bytes; only the first " + budget + " bytes are shown. It was cut to protect the"
109
+ + " session's transport, NOT because the source is short: the full content still exists"
110
+ + " inside this sandbox (the file on disk / the command's output). Do not guess the"
111
+ + " elided part - re-read the source file, or re-run the command with narrower output"
112
+ + " (head, tail, grep, offset+limit), to see it.]";
113
+ };
114
+ const walk = (node: unknown, budget: number): unknown => {
115
+ if (typeof node === "string") return truncate(node, budget);
116
+ if (Array.isArray(node)) return node.map((item) => walk(item, budget));
117
+ if (node !== null && typeof node === "object") {
118
+ const out: Record<string, unknown> = {};
119
+ for (const key of Object.keys(node as Record<string, unknown>)) {
120
+ out[key] = walk((node as Record<string, unknown>)[key], budget);
121
+ }
122
+ return out;
123
+ }
124
+ return node;
125
+ };
126
+ for (const shrink of [1, 8, 64]) {
127
+ const budget = Math.max(64, Math.floor(keepBytes / shrink));
128
+ try {
129
+ const capped = JSON.stringify(walk(parsed, budget));
130
+ if (typeof capped === "string" && Buffer.byteLength(capped, "utf8") <= maxBytes) {
131
+ return capped;
132
+ }
133
+ } catch {
134
+ break; // structural failure — fall through to the raw truncate
135
+ }
136
+ }
137
+ }
138
+ let raw = Buffer.from(line, "utf8").subarray(0, maxBytes).toString("utf8");
139
+ while (raw.endsWith("�")) raw = raw.slice(0, -1);
140
+ return raw;
141
+ }
142
+
143
+ /**
144
+ * The in-sandbox filter's source: a dependency-free node script that
145
+ * line-buffers stdin, re-emits every line through `capJsonlLine`, and turns
146
+ * the exit sentinel into its own exit code.
147
+ *
148
+ * node <script> <sentinel> <maxBytes> <keepBytes>
149
+ */
150
+ export function jsonlGuardScript(): string {
151
+ return [
152
+ "'use strict';",
153
+ `const capJsonlLine = ${capJsonlLine.toString()};`,
154
+ `const NO_SENTINEL_EXIT = ${JSONL_GUARD_NO_SENTINEL_EXIT};`,
155
+ "const sentinel = process.argv[2];",
156
+ "const maxBytes = Number(process.argv[3]);",
157
+ "const keepBytes = Number(process.argv[4]);",
158
+ "let sawSentinel = false;",
159
+ "let buf = '';",
160
+ "// Downstream (the command-stream consumer) going away must never crash",
161
+ "// the pipe with EPIPE noise — the producer is being killed anyway.",
162
+ "process.stdout.on('error', () => process.exit(0));",
163
+ "function handle(line) {",
164
+ " const t = line.trim();",
165
+ " if (sentinel && t.startsWith(sentinel)) {",
166
+ " const code = Number(t.slice(sentinel.length).trim());",
167
+ " process.exitCode = Number.isFinite(code) ? code : 0;",
168
+ " sawSentinel = true;",
169
+ " return;",
170
+ " }",
171
+ " process.stdout.write(capJsonlLine(line, maxBytes, keepBytes) + '\\n');",
172
+ "}",
173
+ "process.stdin.setEncoding('utf8');",
174
+ "process.stdin.on('data', (d) => {",
175
+ " buf += d;",
176
+ " let nl;",
177
+ " while ((nl = buf.indexOf('\\n')) >= 0) {",
178
+ " const line = buf.slice(0, nl);",
179
+ " buf = buf.slice(nl + 1);",
180
+ " if (line.length > 0) handle(line);",
181
+ " }",
182
+ "});",
183
+ "process.stdin.on('end', () => {",
184
+ " if (buf.trim().length > 0) handle(buf);",
185
+ " if (!sawSentinel) process.exitCode = NO_SENTINEL_EXIT;",
186
+ "});",
187
+ ].join("\n");
188
+ }
189
+
190
+ /** Single-quote a value for a `sh -c` command line (local copy — this module
191
+ * must not import from _cli-agent.ts, which imports it). */
192
+ function quote(value: string): string {
193
+ return `'${value.replace(/'/g, `'\\''`)}'`;
194
+ }
195
+
196
+ /**
197
+ * Wrap a turn command so its stdout crosses the command stream through the
198
+ * guard. Self-deciding in-sandbox: when `node` is absent (a bare sandbox a
199
+ * self-provisioned CLI runs on), the command runs unguarded — the pre-guard
200
+ * status quo — instead of failing every turn.
201
+ *
202
+ * The command runs in a SUBSHELL so a `$?`-carrying sentinel line always
203
+ * prints after it exits, however it exits (a brace group would let a shell
204
+ * `exit` skip the printf). stderr stays un-piped — the provider's stderr
205
+ * accumulation is unchanged.
206
+ */
207
+ export function wrapJsonlCommand(args: {
208
+ cmd: string;
209
+ guardPath: string;
210
+ sentinel: string;
211
+ maxBytes: number;
212
+ keepBytes: number;
213
+ }): string {
214
+ const { cmd, guardPath, sentinel, maxBytes, keepBytes } = args;
215
+ return `if command -v node >/dev/null 2>&1; then `
216
+ + `{ ( ${cmd} ); printf '\\n%s %s\\n' ${quote(sentinel)} "$?"; } `
217
+ + `| node ${quote(guardPath)} ${quote(sentinel)} ${maxBytes} ${keepBytes}; `
218
+ + `else ${cmd}; fi`;
219
+ }