@agent-compose/sdk 0.8.5 → 0.8.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +3 -3
  3. package/dist/agent/agent-loop.d.ts +4 -5
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +105 -52
  7. package/dist/directives.d.ts +3 -3
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +12 -12
  12. package/dist/index.js +716 -203
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +182 -68
  15. package/dist/runtimes/claude-code.d.ts +60 -1
  16. package/dist/runtimes/claude.d.ts +1 -1
  17. package/dist/runtimes/codex.d.ts +94 -6
  18. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  19. package/dist/runtimes/openai-desktop.js +686 -199
  20. package/dist/runtimes/opencode.d.ts +48 -11
  21. package/dist/runtimes/opencode.test.d.ts +14 -0
  22. package/dist/sandbox/baked-clis.d.ts +75 -0
  23. package/dist/sandbox/exec-stream.d.ts +1 -2
  24. package/dist/sandbox/network-policy.d.ts +23 -5
  25. package/dist/sandbox.d.ts +4 -2
  26. package/dist/step-invocation/protocol.d.ts +3 -4
  27. package/dist/step-invocation/server.d.ts +2 -2
  28. package/dist/step-invocation/types.d.ts +1 -1
  29. package/dist/types/api-conversations.d.ts +442 -29
  30. package/dist/types/api-factory.d.ts +78 -8
  31. package/dist/types/api-projects.d.ts +480 -0
  32. package/dist/types/api-runs.d.ts +8 -0
  33. package/dist/types/api-scopes.d.ts +32 -3
  34. package/dist/types/conversation-stream.d.ts +5 -0
  35. package/dist/types/execution-context.d.ts +1 -1
  36. package/dist/types/protocol.d.ts +65 -2
  37. package/dist/types/runtime.d.ts +9 -2
  38. package/dist/types/workflow-metadata.d.ts +2 -4
  39. package/dist/types/workflow-plan.d.ts +1 -3
  40. package/dist/utils/bundler.d.ts +23 -0
  41. package/dist/workflow-steps/observability.d.ts +2 -3
  42. package/dist/workflow-steps/runner.d.ts +5 -8
  43. package/dist/workflow-steps/types.d.ts +8 -10
  44. package/dist/workflow-steps/workflow.d.ts +2 -1
  45. package/dist/workflows/engine.d.ts +3 -5
  46. package/dist/workflows/invoke-child.d.ts +2 -2
  47. package/package.json +2 -2
  48. package/src/agent/agent-context.ts +168 -125
  49. package/src/agent/agent-loop.ts +5 -4
  50. package/src/agent/perf-sampler.ts +54 -3
  51. package/src/agent/run-agent.ts +1 -1
  52. package/src/client.ts +191 -71
  53. package/src/directives.ts +3 -3
  54. package/src/display.ts +12 -0
  55. package/src/errors.ts +1 -0
  56. package/src/generated/agentc-commands.ts +571 -0
  57. package/src/index.ts +54 -21
  58. package/src/pause/pause-core.ts +2 -1
  59. package/src/request-context/request-context.ts +1 -1
  60. package/src/runtimes/_cli-agent.ts +306 -122
  61. package/src/runtimes/claude-code.ts +179 -9
  62. package/src/runtimes/claude.ts +1 -1
  63. package/src/runtimes/codex.ts +188 -19
  64. package/src/runtimes/opencode.ts +195 -26
  65. package/src/sandbox/baked-clis.ts +86 -0
  66. package/src/sandbox/exec-stream.ts +1 -2
  67. package/src/sandbox/network-policy.ts +51 -7
  68. package/src/sandbox/providers/e2b.ts +3 -3
  69. package/src/sandbox/providers/vercel.ts +6 -6
  70. package/src/sandbox.ts +8 -2
  71. package/src/step-invocation/invoker.ts +2 -6
  72. package/src/step-invocation/protocol.ts +3 -4
  73. package/src/step-invocation/server.ts +2 -2
  74. package/src/types/api-conversations.ts +366 -23
  75. package/src/types/api-factory.ts +74 -8
  76. package/src/types/api-projects.ts +443 -0
  77. package/src/types/api-runs.ts +5 -0
  78. package/src/types/api-scopes.ts +32 -3
  79. package/src/types/conversation-stream.ts +5 -0
  80. package/src/types/execution-context.ts +1 -1
  81. package/src/types/protocol.ts +67 -1
  82. package/src/types/runtime.ts +8 -2
  83. package/src/types/sandbox-environment.ts +1 -2
  84. package/src/types/workflow-metadata.ts +2 -4
  85. package/src/types/workflow-plan.ts +1 -3
  86. package/src/utils/bundler.ts +88 -19
  87. package/src/workflow-steps/observability.ts +2 -3
  88. package/src/workflow-steps/runner.ts +5 -8
  89. package/src/workflow-steps/types.ts +8 -10
  90. package/src/workflow-steps/workflow.ts +2 -1
  91. package/src/workflows/engine.ts +3 -5
  92. package/src/workflows/invoke-child.ts +2 -2
  93. package/dist/generated/verb-synopsis.d.ts +0 -34
  94. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  95. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  96. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
  97. package/src/generated/verb-synopsis.ts +0 -544
@@ -35,11 +35,13 @@
35
35
  */
36
36
 
37
37
  import type {
38
- AgentMessage, AgentMessageCompaction, AgentMessageTaskNotification, AgentMessageTaskProgress,
38
+ AgentMessage, AgentMessageCompaction, AgentMessageModelUsage, AgentMessagePlanLimits,
39
+ AgentMessageTaskNotification, AgentMessageTaskProgress, AgentMessageUsage,
39
40
  WorkflowProgressEntry,
40
41
  } from "../index.js";
41
42
  import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
42
43
  import { formatError } from "../utils/errors.js";
44
+ import { CLAUDE_CODE_VERSION } from "../sandbox/baked-clis.js";
43
45
 
44
46
  function now(): string { return new Date().toISOString(); }
45
47
 
@@ -409,6 +411,106 @@ export function parseSystemCompaction(
409
411
  };
410
412
  }
411
413
 
414
+ // ── Plan limits (the account meter, off the stream) ─────────────────────────
415
+ //
416
+ // A subscription-funded `claude -p` reads the `anthropic-ratelimit-unified-*`
417
+ // headers off every response and re-emits them as `rate_limit_event` lines
418
+ // whenever its reading changes (the public Agent SDK's SDKRateLimitEvent;
419
+ // anthropics/claude-code#50518 shows the plainly-allowed shape:
420
+ // `{status: "allowed", resetsAt: 1729281600, rateLimitType: "five_hour"}`,
421
+ // with `utilization` carried only once a window crosses a warning
422
+ // threshold). `resetsAt` is epoch SECONDS. A `rejected` event is the plan's
423
+ // wall for the request just made, with the authoritative reset. The platform
424
+ // folds these into the funding account's meter (one reading per window) and
425
+ // learns a cooldown from a rejected one — never from the notice text.
426
+
427
+ const PLAN_LIMIT_STATUSES: ReadonlySet<string> = new Set(["allowed", "allowed_warning", "rejected"]);
428
+
429
+ /** Epoch seconds (the CLI's `resetsAt`) as ISO; undefined for anything else. */
430
+ function epochSecondsIso(v: unknown): string | undefined {
431
+ if (typeof v !== "number" || !Number.isFinite(v) || v <= 0) return undefined;
432
+ return new Date(v * 1000).toISOString();
433
+ }
434
+
435
+ /** One `rate_limit_event` mapped onto the structured plan-limits message,
436
+ * or null when the event names no recognisable status. Pure and tolerant
437
+ * over untrusted harness JSON: every field is forwarded only when present
438
+ * and well-typed; nothing is defaulted or invented. Exported for tests. */
439
+ export function parsePlanLimits(
440
+ p: Record<string, unknown>, timestamp: string,
441
+ ): AgentMessagePlanLimits | null {
442
+ const info = (typeof p.rate_limit_info === "object" && p.rate_limit_info !== null
443
+ ? p.rate_limit_info : null) as Record<string, unknown> | null;
444
+ if (!info || typeof info.status !== "string" || !PLAN_LIMIT_STATUSES.has(info.status)) return null;
445
+ const num = (v: unknown): number | undefined =>
446
+ typeof v === "number" && Number.isFinite(v) ? v : undefined;
447
+ const str = (v: unknown): string | undefined =>
448
+ typeof v === "string" && v.length > 0 ? clip(v, 100) : undefined;
449
+ const window = str(info.rateLimitType);
450
+ const resetsAt = epochSecondsIso(info.resetsAt);
451
+ const utilization = num(info.utilization);
452
+ const surpassedThreshold = num(info.surpassedThreshold);
453
+ const overageStatus = typeof info.overageStatus === "string" && PLAN_LIMIT_STATUSES.has(info.overageStatus)
454
+ ? info.overageStatus as AgentMessagePlanLimits["status"] : undefined;
455
+ const overageResetsAt = epochSecondsIso(info.overageResetsAt);
456
+ const overageDisabledReason = str(info.overageDisabledReason);
457
+ const overageInUse = typeof info.isUsingOverage === "boolean" ? info.isUsingOverage
458
+ : typeof info.overageInUse === "boolean" ? info.overageInUse : undefined;
459
+ const overage = overageStatus !== undefined || overageResetsAt !== undefined
460
+ || overageDisabledReason !== undefined || overageInUse !== undefined
461
+ ? {
462
+ ...(overageStatus !== undefined ? { status: overageStatus } : {}),
463
+ ...(overageResetsAt !== undefined ? { resetsAt: overageResetsAt } : {}),
464
+ ...(overageDisabledReason !== undefined ? { disabledReason: overageDisabledReason } : {}),
465
+ ...(overageInUse !== undefined ? { inUse: overageInUse } : {}),
466
+ }
467
+ : undefined;
468
+ const limitScope = str(info.limitScope);
469
+ const errorCode = str(info.errorCode);
470
+ return {
471
+ type: "plan_limits",
472
+ status: info.status as AgentMessagePlanLimits["status"],
473
+ ...(window !== undefined ? { window } : {}),
474
+ ...(resetsAt !== undefined ? { resetsAt } : {}),
475
+ ...(utilization !== undefined ? { utilization } : {}),
476
+ ...(surpassedThreshold !== undefined ? { surpassedThreshold } : {}),
477
+ ...(overage !== undefined ? { overage } : {}),
478
+ ...(limitScope !== undefined ? { limitScope } : {}),
479
+ ...(errorCode !== undefined ? { errorCode } : {}),
480
+ timestamp,
481
+ };
482
+ }
483
+
484
+ /** The terminal result's per-model report (`result.modelUsage`, keyed by
485
+ * the raw model string) mapped onto the protocol's per-model shape, or
486
+ * undefined when the result carries none. Counts the harness did not
487
+ * report stay absent (the four classes default to 0 only when the entry
488
+ * exists at all — an entry IS a report of that model). Pure; exported for
489
+ * tests. */
490
+ export function parseModelUsage(raw: unknown): Record<string, AgentMessageModelUsage> | undefined {
491
+ if (typeof raw !== "object" || raw === null) return undefined;
492
+ const num = (v: unknown): number => (typeof v === "number" && Number.isFinite(v) ? v : 0);
493
+ const opt = (v: unknown): number | undefined => (typeof v === "number" && Number.isFinite(v) ? v : undefined);
494
+ const out: Record<string, AgentMessageModelUsage> = {};
495
+ for (const [model, entry] of Object.entries(raw as Record<string, unknown>)) {
496
+ if (typeof entry !== "object" || entry === null || model.length === 0) continue;
497
+ const e = entry as Record<string, unknown>;
498
+ const thinkingTokens = opt(e.thinkingTokens);
499
+ const webSearchRequests = opt(e.webSearchRequests);
500
+ const costUsd = opt(e.costUSD);
501
+ out[clip(model, 200)] = {
502
+ inputTokens: num(e.inputTokens),
503
+ outputTokens: num(e.outputTokens),
504
+ cacheReadTokens: num(e.cacheReadInputTokens),
505
+ cacheCreationTokens: num(e.cacheCreationInputTokens),
506
+ ...(thinkingTokens !== undefined ? { thinkingTokens } : {}),
507
+ ...(webSearchRequests !== undefined ? { webSearchRequests } : {}),
508
+ ...(costUsd !== undefined ? { costUsd } : {}),
509
+ };
510
+ }
511
+ return Object.keys(out).length > 0 ? out : undefined;
512
+ }
513
+
412
514
  /** Claude Code's real reasoning knob is its own `--effort <level>` flag
413
515
  * (low|medium|high|xhigh|max — verified against `claude -p --help`). The
414
516
  * CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
@@ -419,6 +521,50 @@ export function parseSystemCompaction(
419
521
  export const CLAUDE_CODE_EFFORT_LEVELS: readonly CliReasoningEffort[] =
420
522
  ["low", "medium", "high", "xhigh", "max"];
421
523
 
524
+ /** The Bash PreToolUse hook rtk installs for Claude Code (`rtk init -g
525
+ * --hook-only` writes exactly this entry — matcher `Bash`, command
526
+ * `rtk hook claude` — into ~/.claude/settings.json; the image bakes
527
+ * RTK_VERSION, sandbox/baked-clis.ts, which this payload was run against).
528
+ * `rtk hook claude` reads the PreToolUse JSON on stdin and, when it has a
529
+ * filter for the command, answers with `updatedInput` rewriting `git
530
+ * status` to `rtk git status` (ls, find, grep, test runners, bun, curl,
531
+ * docker, …), so the model reads rtk's compact output. A command it
532
+ * cannot compress — an unknown tool, a pipe into one, command or process
533
+ * substitution, a heredoc, a file redirect, anything already prefixed
534
+ * `rtk`, or a `RTK_DISABLED=1` prefix — gets no output and exit 0, which
535
+ * Claude Code treats as "no decision": the original command runs
536
+ * unchanged (all verified against the binary).
537
+ *
538
+ * The wrapper fails OPEN both ways. `command -v` covers images without
539
+ * rtk (the devbox, a bare Vercel VM): there a bare `rtk hook claude` would
540
+ * fail every Bash call's hook with exit 127 — non-blocking, but a stderr
541
+ * warning per call; rtk's own legacy shell hook degraded the same way
542
+ * ("binary not found: exit 0"). The trailing `exit 0` covers an rtk that
543
+ * cannot answer: Claude Code treats a PreToolUse hook's exit 2 as a BLOCK
544
+ * of the tool call with stderr fed to the model, and clap exits 2 with
545
+ * its usage for a subcommand it does not know — which is how the codex
546
+ * hook (RTK_CODEX_HOOK_COMMAND, codex.ts) blocked every codex shell
547
+ * command on 2026-10-03, when the image's rtk was a cache-served 0.45.0.
548
+ * rtk's hooks never exit non-zero on purpose, so a non-zero exit is a
549
+ * broken rtk, and the compressor must never cost the worker its shell:
550
+ * the exit code is dropped, and the smoke gate's rtk-hook-rewrite check
551
+ * is what proves the rewrite itself. */
552
+ export const RTK_BASH_HOOK_COMMAND =
553
+ "command -v rtk >/dev/null 2>&1 && rtk hook claude; exit 0";
554
+
555
+ /** Settings every platform `claude` launch passes as `--settings` (inline
556
+ * JSON). Claude Code MERGES hook entries across its settings sources
557
+ * instead of replacing them (user → project → local → flag → managed), so
558
+ * this rides beside whatever the machine's own ~/.claude/settings.json
559
+ * carries, and nothing is written to that file — the right outcome, since
560
+ * the server never owns it: home carry tar-restores it whole across
561
+ * machines and the image capture strips it as personal state. */
562
+ export const CLAUDE_CODE_PLATFORM_SETTINGS = {
563
+ hooks: {
564
+ PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_BASH_HOOK_COMMAND }] }],
565
+ },
566
+ } as const;
567
+
422
568
  export const claudeCodeSpec: CliAgentSpec = {
423
569
  kind: "claude-code",
424
570
  authEnv: "ANTHROPIC_API_KEY",
@@ -428,11 +574,14 @@ export const claudeCodeSpec: CliAgentSpec = {
428
574
  acp: { command: "npx", args: ["--yes", CLAUDE_CODE_ACP_ADAPTER] },
429
575
  // Bare-sandbox fallback only: the agent-env image bakes /usr/local/bin/claude
430
576
  // (same recipe — infra/e2b-template/build.ts), so the probe short-circuits
431
- // on the platform path. Anthropic's installer drops a versioned binary under
432
- // $HOME with a ~/.local/bin/claude launcher; resolve the symlink and copy
433
- // the self-contained binary to a world-executable system path.
577
+ // on the platform path. The SAME pinned version as the image
578
+ // (CLAUDE_CODE_VERSION), never latest, so a machine the image predates runs
579
+ // what the image's machines run. Anthropic's installer drops a versioned
580
+ // binary under $HOME with a ~/.local/bin/claude launcher; resolve the
581
+ // symlink and copy the self-contained binary to a world-executable system
582
+ // path.
434
583
  install:
435
- 'curl -fsSL https://claude.ai/install.sh | bash && ' +
584
+ `curl -fsSL https://claude.ai/install.sh | bash -s ${CLAUDE_CODE_VERSION} && ` +
436
585
  'REAL=$(readlink -f "$HOME/.local/bin/claude") && test -f "$REAL" && ' +
437
586
  'sudo cp "$REAL" /usr/local/bin/claude && sudo chmod 0755 /usr/local/bin/claude',
438
587
  // Claude reads the prompt from stdin in -p mode.
@@ -447,7 +596,8 @@ export const claudeCodeSpec: CliAgentSpec = {
447
596
  // the same build). A message that lands after the result would start a
448
597
  // NEW turn in-process, which is why the transport's feeder stops at the
449
598
  // result line instead of forwarding past it.
450
- streamInput: {
599
+ midTurnInput: {
600
+ transport: "stdin-stream",
451
601
  promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
452
602
  messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
453
603
  // The ESC equivalent (verified live against claude 2.1.236): the CLI's
@@ -479,6 +629,10 @@ export const claudeCodeSpec: CliAgentSpec = {
479
629
  // claude's own attestation env for exactly this: it lifts the flag's
480
630
  // root-user refusal (the E2B agent-env user is root).
481
631
  "--dangerously-skip-permissions",
632
+ // rtk's Bash PreToolUse hook (RTK_BASH_HOOK_COMMAND): shell output is
633
+ // compressed before it reaches the model. Flag settings merge with the
634
+ // machine's own settings; they never replace them.
635
+ `--settings ${shellQuote(JSON.stringify(CLAUDE_CODE_PLATFORM_SETTINGS))}`,
482
636
  ...(model ? [`--model ${shellQuote(model)}`] : []),
483
637
  // Reasoning effort is the CLI's own flag; the value comes from the
484
638
  // closed CliReasoningEffort set, so it is shell-safe unquoted.
@@ -614,14 +768,21 @@ export const claudeCodeSpec: CliAgentSpec = {
614
768
  return [];
615
769
  }
616
770
  // Terminal result: usage on success (the text already streamed via the
617
- // assistant events); an honest string error on failure.
771
+ // assistant events); an honest string error on failure. The turn
772
+ // totals are the main loop's `usage`; `modelUsage` (every model the
773
+ // query pipeline called, with the CLI's own cost estimate) and
774
+ // `total_cost_usd` ride beside them as the harness reported them —
775
+ // cumulative for the guest session (AgentMessageUsage.byModel).
618
776
  case "result": {
619
777
  if (p.is_error === true) {
620
778
  return [{ type: "error", text: formatError(p.result ?? p.subtype ?? p), timestamp: ts }];
621
779
  }
622
780
  const u = p.usage as Record<string, number> | undefined;
623
781
  if (!u) return [];
624
- return [{
782
+ const byModel = parseModelUsage(p.modelUsage);
783
+ const costUsd = typeof p.total_cost_usd === "number" && Number.isFinite(p.total_cost_usd)
784
+ ? p.total_cost_usd : undefined;
785
+ const usage: AgentMessageUsage = {
625
786
  type: "usage",
626
787
  inputTokens: u.input_tokens ?? 0,
627
788
  outputTokens: u.output_tokens ?? 0,
@@ -629,8 +790,17 @@ export const claudeCodeSpec: CliAgentSpec = {
629
790
  cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
630
791
  durationMs: typeof p.duration_ms === "number" ? p.duration_ms : 0,
631
792
  numTurns: typeof p.num_turns === "number" ? p.num_turns : 1,
793
+ ...(byModel !== undefined ? { byModel } : {}),
794
+ ...(costUsd !== undefined ? { costUsd } : {}),
632
795
  timestamp: ts,
633
- }];
796
+ };
797
+ return [usage];
798
+ }
799
+ // The account's plan limits as the CLI read them off the response
800
+ // headers (subscription-funded sessions; parsePlanLimits above).
801
+ case "rate_limit_event": {
802
+ const limits = parsePlanLimits(p, ts);
803
+ return limits ? [limits] : [];
634
804
  }
635
805
  // System events: init carries the session id (extractSessionId), and
636
806
  // the background-task lane rides here too — `task_notification` is
@@ -1,4 +1,4 @@
1
- /** Claude Agent SDK runtime — replaces the old Claude CLI subprocess runtime. */
1
+ /** Claude Agent SDK runtime. */
2
2
 
3
3
  import { query, type HookCallback, type PreToolUseHookInput, type SDKUserMessage, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
4
4
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
@@ -5,17 +5,28 @@
5
5
  * only stream-parse what it prints.
6
6
  *
7
7
  * Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
8
- * workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
9
- * demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
10
- * to install once and boot from the captured snapshot on every run after.
8
+ * workflow secret. The E2B session image bakes the pinned `codex` CLI
9
+ * (`@openai/codex@CODEX_CLI_VERSION`); on any other machine the runtime
10
+ * installs that same version on demand (pair with
11
+ * `snapshots: { bootFrom: "reuse" }` to install once and boot from the
12
+ * captured snapshot on every run after).
11
13
  *
12
14
  * Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
13
- * with the command_execution / reasoning / agent_message item shapes below.
15
+ * with the command_execution / reasoning / agent_message item shapes below;
16
+ * the plan's `todo_list` item against codex-rs rust-v0.153.4 and
17
+ * rust-v0.159.2 (see codexPlanMessages); the command_execution /
18
+ * agent_message / turn.completed shapes re-verified live against 0.160.0
19
+ * (the pin), whose plan-tool sources are byte-identical to rust-v0.159.2.
14
20
  */
15
21
 
16
22
  import type { AgentMessage } from "../index.js";
17
- import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
23
+ import type { AgentMessagePlan } from "../types/protocol.js";
24
+ import {
25
+ MID_TURN_DELIVERED_ENV, MID_TURN_INBOX_ENV, createCliAgentRuntime, shellQuote,
26
+ type CliAgentSpec, type CliReasoningEffort,
27
+ } from "./_cli-agent.js";
18
28
  import { formatError } from "../utils/errors.js";
29
+ import { CODEX_CLI_VERSION } from "../sandbox/baked-clis.js";
19
30
 
20
31
  function now(): string { return new Date().toISOString(); }
21
32
 
@@ -24,6 +35,97 @@ function now(): string { return new Date().toISOString(); }
24
35
  * overrides the bundled `@openai/codex` the adapter drives underneath. */
25
36
  const CODEX_ACP_ADAPTER = "@agentclientprotocol/codex-acp@0.1.0";
26
37
 
38
+ /** The Bash PreToolUse hook rtk installs for Codex (`rtk init -g --codex`
39
+ * writes exactly this entry — matcher `Bash`, command `rtk hook codex` —
40
+ * into $CODEX_HOME/hooks.json). `rtk hook codex` exists since rtk 0.50.0;
41
+ * the image bakes RTK_VERSION (sandbox/baked-clis.ts), which this payload
42
+ * was run against. Codex's shell tool matches the `Bash` matcher (codex-rs
43
+ * rust-v0.160.0, the pinned CODEX_CLI_VERSION; identical at
44
+ * rust-v0.159.2), and the hook runs under `$SHELL -lc` with the PreToolUse
45
+ * JSON on stdin. `rtk hook codex` answers `permissionDecision: "allow"` +
46
+ * `updatedInput` rewriting `git status` to `rtk git status` when it has a
47
+ * filter for the command; Codex applies the replacement before its own
48
+ * approval and sandbox checks. For a command it cannot compress (an
49
+ * unknown tool, a pipe into one, substitutions, heredocs, redirects, a
50
+ * `RTK_DISABLED=1` prefix, an unknown permission mode) it prints nothing
51
+ * and exits 0, and Codex runs the original unchanged.
52
+ *
53
+ * The wrapper fails OPEN both ways, like RTK_BASH_HOOK_COMMAND
54
+ * (claude-code.ts). `command -v` covers an image without rtk (the devbox,
55
+ * a bare Vercel VM): silent, exit 0, so Codex runs the command instead of
56
+ * failing every shell call's hook with 127. The trailing `exit 0` covers
57
+ * an rtk that cannot answer: Codex treats a PreToolUse hook's exit 2 with
58
+ * stderr as a BLOCK of the tool call (codex-rs hooks/src/events/
59
+ * pre_tool_use.rs at rust-v0.160.0 — the model reads "Command blocked by
60
+ * PreToolUse hook: <stderr>"), and clap exits 2 with its usage on stderr
61
+ * for a subcommand it does not know. That was 2026-10-03: the image's rtk
62
+ * was a cache-served 0.45.0 with no `hook codex`, and this hook — `exec`
63
+ * handing rtk's exit code to Codex — blocked every shell command of every
64
+ * codex session. rtk's hooks never exit non-zero on purpose (they fail
65
+ * open with no stdout), so a non-zero exit is always a broken rtk, and the
66
+ * compressor must never cost the worker its shell: the exit code is
67
+ * dropped, and the smoke gate's codex-rtk-hook-rewrite check is what
68
+ * proves the rewrite itself. */
69
+ export const RTK_CODEX_HOOK_COMMAND =
70
+ "command -v rtk >/dev/null 2>&1 && rtk hook codex; exit 0";
71
+
72
+ /** The awk program of the mid-turn hook below: the inbox lines not yet fed
73
+ * (JSON string literals, one per line — `codexSpec.midTurnInput.messageLine`)
74
+ * become ONE PostToolUse hook output. Each literal's quotes are stripped and
75
+ * the bodies are joined with an escaped blank line; the bodies are already
76
+ * JSON-escaped, so the result is one valid JSON string. Nothing is printed
77
+ * when no line is due, and codex ignores an empty stdout. */
78
+ const CODEX_MID_TURN_HOOK_AWK = String.raw`NR > fed && NR <= upto { s = substr($0, 2, length($0) - 2); out = (out == "" ? s : out "\\n\\n" s) } END { if (out != "") printf "{\"hookSpecificOutput\":{\"hookEventName\":\"PostToolUse\",\"additionalContext\":\"%s\"}}\n", out }`;
79
+
80
+ /** The PostToolUse command hook that carries a mid-turn message into a
81
+ * RUNNING codex turn (the 2026-10-02 relayed-steer incident: three of the
82
+ * owner's instructions waited 48-51 minutes for a codex build to end).
83
+ * Codex runs it, under `$SHELL -lc` with the event JSON on stdin and the
84
+ * codex process's own environment, after every tool call it completes
85
+ * (codex-rs core/src/tools/registry.rs → hook_runtime.rs at rust-v0.160.0,
86
+ * the pin). The launch wrapper exported the turn's inbox and delivered-
87
+ * counter paths into that environment (sdk _cli-agent.ts
88
+ * toolHookInboxFragment); the hook forwards the inbox lines past the
89
+ * counter as the hook's `additionalContext`, which codex records as
90
+ * developer context in the live turn's history before the model's next
91
+ * request (hook_runtime.rs record_additional_contexts), then advances the
92
+ * counter — the same ack the server's inject lane trusts for the stdin
93
+ * lane. Not a platform turn (no inbox in the environment): silent, exit 0.
94
+ * Codex's default spill threshold for a hook's additional context is 2,500
95
+ * tokens (hooks/src/output_spill.rs); a longer message reaches the model
96
+ * as a preview plus a file pointer, which a steer never is. */
97
+ export const CODEX_MID_TURN_HOOK_COMMAND =
98
+ // Drain the event JSON first, so codex's stdin write never meets a closed pipe.
99
+ "cat >/dev/null; "
100
+ + `[ -n "\${${MID_TURN_INBOX_ENV}:-}" ] && [ -f "$${MID_TURN_INBOX_ENV}" ] || exit 0; `
101
+ + `ac_fed=$(cat "$${MID_TURN_DELIVERED_ENV}" 2>/dev/null); ac_fed=\${ac_fed:-0}; `
102
+ + `ac_lines=$(wc -l < "$${MID_TURN_INBOX_ENV}" 2>/dev/null); ac_lines=\${ac_lines:-0}; `
103
+ + `[ "$ac_lines" -gt "$ac_fed" ] || exit 0; `
104
+ + `awk -v fed="$ac_fed" -v upto="$ac_lines" '${CODEX_MID_TURN_HOOK_AWK}' "$${MID_TURN_INBOX_ENV}" `
105
+ + `&& echo "$ac_lines" > "$${MID_TURN_DELIVERED_ENV}"; exit 0`;
106
+
107
+ /** $CODEX_HOME/hooks.json for every platform Codex session (the server
108
+ * writes it at boot, session-runtime-config.ts). Codex only RUNS a
109
+ * user-level hook it has persisted trust for — the TUI's review prompt
110
+ * has no headless counterpart, and `--dangerously-bypass-hook-trust` would
111
+ * run a cloned repository's `.codex/hooks.json` unreviewed too — so the
112
+ * server also writes the trust record for each of these exact hooks into
113
+ * the config.toml it generates (sandbox/codex-hooks.ts). Verified live
114
+ * against codex 0.159.2 and 0.160.0 for the rtk hook: with the record,
115
+ * `codex exec` rewrote `git status` through rtk with no bypass flag;
116
+ * without it, the hook was skipped.
117
+ *
118
+ * The PostToolUse group carries NO matcher: codex runs a matcher-less hook
119
+ * after every tool it completes (hooks/src/events/common.rs
120
+ * matches_matcher: an absent matcher is a match), so a mid-turn message
121
+ * lands at the next tool step whatever the tool was. */
122
+ export const CODEX_PLATFORM_HOOKS = {
123
+ hooks: {
124
+ PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_CODEX_HOOK_COMMAND }] }],
125
+ PostToolUse: [{ hooks: [{ type: "command", command: CODEX_MID_TURN_HOOK_COMMAND }] }],
126
+ },
127
+ } as const;
128
+
27
129
  /** Known-noise codex ADVISORY lines. codex emits these as `item.completed`
28
130
  * error items on the `--json` stream (exec maps every `Warning` notification
29
131
  * to an error item — verified against codex-cli 0.147.0), so without a filter
@@ -101,6 +203,41 @@ export function codexWriterLockPreflight(
101
203
  + `flock -n "$ac_lk" rm -f -- "$ac_lk" 2>/dev/null || true; fi; `;
102
204
  }
103
205
 
206
+ /**
207
+ * codex's PLAN (its update_plan tool) on the `--json` stream: one
208
+ * `todo_list` item per turn, `item.started` on the plan's first update,
209
+ * `item.updated` (same id, the whole list) on every later one, and
210
+ * `item.completed` at the turn's end. That is codex-rs exec's
211
+ * event_processor_with_jsonl_output.rs, identical in rust-v0.153.4 and
212
+ * rust-v0.159.2, where it is the only `item.updated` codex emits; the item
213
+ * `{ id, type: "todo_list", items: [{ text, completed }] }` is the shape every
214
+ * todo_list item our machines have persisted carries. Each event maps to ONE
215
+ * whole-plan `plan` message, so the transcript's checklist, and the worker's
216
+ * status line, follow the plan while the turn runs (before this, the list
217
+ * surfaced once, at the turn's end, as a raw `todo_list` tool card).
218
+ *
219
+ * A step carries only `completed`: codex's `pending` and `in_progress` both
220
+ * serialize as false. Its plan tool allows at most one step in progress and
221
+ * a plan runs in order, so the first step not completed is the one under
222
+ * way; the rest are pending. The stream names no priority, so none is set.
223
+ */
224
+ function codexPlanMessages(item: Record<string, unknown>, timestamp: string): AgentMessage[] {
225
+ const steps = Array.isArray(item.items) ? item.items : [];
226
+ let underWay = false;
227
+ const entries: AgentMessagePlan["entries"] = [];
228
+ for (const step of steps) {
229
+ if (typeof step !== "object" || step === null) continue;
230
+ const text = (step as { text?: unknown }).text;
231
+ if (typeof text !== "string" || text.trim().length === 0) continue;
232
+ let status: AgentMessagePlan["entries"][number]["status"];
233
+ if ((step as { completed?: unknown }).completed === true) status = "completed";
234
+ else if (!underWay) { status = "in_progress"; underWay = true; }
235
+ else status = "pending";
236
+ entries.push({ content: text.trim(), status });
237
+ }
238
+ return entries.length > 0 ? [{ type: "plan", entries, timestamp }] : [];
239
+ }
240
+
104
241
  /** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
105
242
  * `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
106
243
  * of the public runtime surface — `createCodexRuntime` stays the entry point. */
@@ -113,6 +250,22 @@ export const codexSpec: CliAgentSpec = {
113
250
  // every resume of the same thread, so supersede teardown must KILL it —
114
251
  // never detach it alive (server runner-kill.ts consumes this flag).
115
252
  exclusiveSessionWriter: true,
253
+ // Mid-turn input (the 2026-10-02 relayed-steer incident). codex reads
254
+ // stdin ONCE, as the prompt: `codex exec -` submits one UserTurn and never
255
+ // reads stdin again (codex-rs exec/src/lib.rs read_prompt_from_stdin and
256
+ // its single InitialOperation::UserTurn at rust-v0.160.0; the app-server's
257
+ // `turn/steer` is a different transport, not `codex exec`). So a running
258
+ // turn takes a message through the PostToolUse hook above: codex runs it
259
+ // after every tool call it completes, with the codex process's own
260
+ // environment; the hook prints the unfed inbox lines as `additionalContext`
261
+ // and codex records them as developer context in the live turn's history
262
+ // before the model's next request. One JSON string literal per inbox line,
263
+ // so the hook splices bodies without decoding. Codex builds the PostToolUse
264
+ // payload only for a tool call it counts as successful (tools/registry.rs),
265
+ // so a message lands at the next successful tool step. Source-verified at
266
+ // rust-v0.160.0 (the pinned CODEX_CLI_VERSION); the end-to-end run against
267
+ // a live codex is the acceptance check still owed.
268
+ midTurnInput: { transport: "tool-hook", messageLine: (text) => JSON.stringify(text) },
116
269
  // ACP-mode invocation (ADR-0020 increment 1). `codex` has no native `--acp`
117
270
  // flag; the adapter (a Rust binary shipped via npm) speaks ACP and drives a
118
271
  // compatible bundled `@openai/codex`. The runner attempts this first and
@@ -125,9 +278,11 @@ export const codexSpec: CliAgentSpec = {
125
278
  // (e.g. a sandbox-baked one) instead of the adapter's bundled copy.
126
279
  ...(process.env.CODEX_PATH ? { env: { CODEX_PATH: process.env.CODEX_PATH } } : {}),
127
280
  },
128
- // Global npm install; symlink onto PATH only if the global bin dir isn't
129
- // already there (so a non-login `sh -c` can find it).
130
- install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
281
+ // The E2B session image bakes this exact version (sandbox/baked-clis.ts),
282
+ // so this runs only on a machine whose image predates the bake: a global
283
+ // npm install of the SAME pin, symlinked onto PATH only if the global bin
284
+ // dir isn't already there (so a non-login `sh -c` can find it).
285
+ install: `sudo npm install -g @openai/codex@${CODEX_CLI_VERSION} && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)`,
131
286
  // Codex reads the prompt from stdin when invoked as `codex exec ... -`.
132
287
  promptPayload: (prompt) => prompt,
133
288
  buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
@@ -140,13 +295,13 @@ export const codexSpec: CliAgentSpec = {
140
295
  // for running in environments that are externally sandboxed".
141
296
  "--dangerously-bypass-approvals-and-sandbox",
142
297
  ...(model ? ["-m", shellQuote(model)] : []),
143
- // codex's own reasoning knob — a config override, valid values
144
- // none|minimal|low|medium|high|xhigh (codex docs; xhigh is the
145
- // codex-max-tier deep-reasoning level). There is NO "max" in codex's
146
- // vocabulary: the server rejects it for codex sessions
147
- // (sessionEffortLockError), and this clamp to xhigh is the type-level
148
- // backstop for a caller that bypasses that gate. The value comes from
149
- // the closed CliReasoningEffort set, so it is shell-safe unquoted.
298
+ // codex's own reasoning knob — a config override. Every codex model
299
+ // takes low|medium|high|xhigh; only the GPT-6 and GPT-5.6 presets add
300
+ // max (codex 0.153+), so the platform offers codex up to xhigh: the
301
+ // server rejects "max" for codex sessions (sessionEffortLockError), and
302
+ // this clamp to xhigh is the type-level backstop for a caller that
303
+ // bypasses that gate. The value comes from the closed
304
+ // CliReasoningEffort set, so it is shell-safe unquoted.
150
305
  ...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
151
306
  ].join(" ");
152
307
  // Working directory via a shell `cd` — the idiom EVERY other runtime uses
@@ -172,11 +327,19 @@ export const codexSpec: CliAgentSpec = {
172
327
  mapEvent: (p): AgentMessage[] => {
173
328
  const ts = now();
174
329
  switch (p.type) {
330
+ case "item.updated": {
331
+ // Only the plan updates in place (codexPlanMessages); any other
332
+ // item surfaces on its started / completed events below.
333
+ const item = p.item as Record<string, unknown> | undefined;
334
+ return item?.type === "todo_list" ? codexPlanMessages(item, ts) : [];
335
+ }
175
336
  case "item.started":
176
337
  case "item.completed": {
177
338
  const item = p.item as Record<string, unknown> | undefined;
178
339
  if (!item) return [];
179
340
  const itype = String(item.type ?? "");
341
+ // The plan, whole, on its first update and at the turn's end.
342
+ if (itype === "todo_list") return codexPlanMessages(item, ts);
180
343
  // Text + reasoning land on completion (started carries no final text).
181
344
  if (itype === "agent_message") {
182
345
  return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
@@ -209,12 +372,17 @@ export const codexSpec: CliAgentSpec = {
209
372
  }
210
373
  return [{ type: "error", text, timestamp: ts }];
211
374
  }
212
- // file_change / mcp_tool_call / web_search / todo: surface once, on completion.
375
+ // file_change / mcp_tool_call / web_search: surface once, on completion.
213
376
  if (p.type === "item.completed") {
214
377
  return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
215
378
  }
216
379
  return [];
217
380
  }
381
+ // The turn's token classes as codex reports them (codex-rs
382
+ // exec_events.rs `Usage`): input, cached input, cache-write input,
383
+ // output and reasoning output. Every class is kept — cache writes land
384
+ // in the creation class, reasoning rides beside output (it is counted
385
+ // inside output_tokens, so the four totals stay additive).
218
386
  case "turn.completed": {
219
387
  const u = p.usage as Record<string, number> | undefined;
220
388
  if (!u) return [];
@@ -223,9 +391,10 @@ export const codexSpec: CliAgentSpec = {
223
391
  inputTokens: u.input_tokens ?? 0,
224
392
  outputTokens: u.output_tokens ?? 0,
225
393
  cacheReadTokens: u.cached_input_tokens ?? 0,
226
- cacheCreationTokens: 0,
394
+ cacheCreationTokens: u.cache_write_input_tokens ?? 0,
227
395
  durationMs: 0,
228
396
  numTurns: 1,
397
+ ...(typeof u.reasoning_output_tokens === "number" ? { reasoningOutputTokens: u.reasoning_output_tokens } : {}),
229
398
  timestamp: ts,
230
399
  }];
231
400
  }
@@ -238,8 +407,8 @@ export const codexSpec: CliAgentSpec = {
238
407
  },
239
408
  };
240
409
 
241
- /** The effort levels codex actually has (`model_reasoning_effort`):
242
- * low|medium|high|xhigh — no "max" (that level is Claude Code's alone). */
410
+ /** The effort levels this runtime passes to codex (`model_reasoning_effort`):
411
+ * low|medium|high|xhigh, the levels every codex model takes. */
243
412
  export type CodexReasoningEffort = Exclude<CliReasoningEffort, "max">;
244
413
 
245
414
  export interface CodexRuntimeConfig {