@agent-compose/sdk 0.8.5 → 0.8.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +3 -3
  3. package/dist/agent/agent-loop.d.ts +6 -5
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +119 -54
  7. package/dist/directives.d.ts +3 -3
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +12 -12
  12. package/dist/index.js +771 -204
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +185 -68
  15. package/dist/runtimes/_reported-model.d.ts +16 -0
  16. package/dist/runtimes/claude-code.d.ts +60 -1
  17. package/dist/runtimes/claude.d.ts +1 -1
  18. package/dist/runtimes/codex.d.ts +94 -6
  19. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  20. package/dist/runtimes/model-report.test.d.ts +14 -0
  21. package/dist/runtimes/openai-desktop.js +741 -200
  22. package/dist/runtimes/opencode.d.ts +48 -11
  23. package/dist/runtimes/opencode.test.d.ts +14 -0
  24. package/dist/sandbox/baked-clis.d.ts +75 -0
  25. package/dist/sandbox/exec-stream.d.ts +1 -2
  26. package/dist/sandbox/network-policy.d.ts +23 -5
  27. package/dist/sandbox.d.ts +4 -2
  28. package/dist/step-invocation/protocol.d.ts +3 -4
  29. package/dist/step-invocation/server.d.ts +2 -2
  30. package/dist/step-invocation/types.d.ts +1 -1
  31. package/dist/types/api-conversations.d.ts +442 -29
  32. package/dist/types/api-factory.d.ts +99 -10
  33. package/dist/types/api-projects.d.ts +521 -0
  34. package/dist/types/api-runs.d.ts +83 -0
  35. package/dist/types/api-scopes.d.ts +32 -3
  36. package/dist/types/conversation-stream.d.ts +5 -0
  37. package/dist/types/execution-context.d.ts +1 -1
  38. package/dist/types/protocol.d.ts +86 -2
  39. package/dist/types/runtime.d.ts +9 -2
  40. package/dist/types/workflow-metadata.d.ts +2 -4
  41. package/dist/types/workflow-plan.d.ts +1 -3
  42. package/dist/utils/bundler.d.ts +23 -0
  43. package/dist/workflow-steps/observability.d.ts +2 -3
  44. package/dist/workflow-steps/runner.d.ts +5 -8
  45. package/dist/workflow-steps/types.d.ts +8 -10
  46. package/dist/workflow-steps/workflow.d.ts +2 -1
  47. package/dist/workflows/engine.d.ts +3 -5
  48. package/dist/workflows/invoke-child.d.ts +2 -2
  49. package/package.json +2 -2
  50. package/src/agent/agent-context.ts +168 -125
  51. package/src/agent/agent-loop.ts +7 -6
  52. package/src/agent/perf-sampler.ts +54 -3
  53. package/src/agent/run-agent.ts +1 -1
  54. package/src/client.ts +226 -71
  55. package/src/directives.ts +3 -3
  56. package/src/display.ts +12 -0
  57. package/src/errors.ts +1 -0
  58. package/src/generated/agentc-commands.ts +571 -0
  59. package/src/index.ts +57 -21
  60. package/src/pause/pause-core.ts +2 -1
  61. package/src/request-context/request-context.ts +1 -1
  62. package/src/runtimes/_cli-agent.ts +318 -122
  63. package/src/runtimes/_reported-model.ts +24 -0
  64. package/src/runtimes/claude-code.ts +195 -12
  65. package/src/runtimes/claude.ts +9 -2
  66. package/src/runtimes/codex.ts +188 -19
  67. package/src/runtimes/opencode.ts +195 -26
  68. package/src/sandbox/baked-clis.ts +86 -0
  69. package/src/sandbox/exec-stream.ts +1 -2
  70. package/src/sandbox/network-policy.ts +51 -7
  71. package/src/sandbox/providers/e2b.ts +3 -3
  72. package/src/sandbox/providers/vercel.ts +6 -6
  73. package/src/sandbox.ts +8 -2
  74. package/src/step-invocation/invoker.ts +2 -6
  75. package/src/step-invocation/protocol.ts +3 -4
  76. package/src/step-invocation/server.ts +2 -2
  77. package/src/types/api-conversations.ts +366 -23
  78. package/src/types/api-factory.ts +95 -10
  79. package/src/types/api-projects.ts +477 -0
  80. package/src/types/api-runs.ts +73 -0
  81. package/src/types/api-scopes.ts +32 -3
  82. package/src/types/conversation-stream.ts +5 -0
  83. package/src/types/execution-context.ts +1 -1
  84. package/src/types/protocol.ts +91 -2
  85. package/src/types/runtime.ts +8 -2
  86. package/src/types/sandbox-environment.ts +1 -2
  87. package/src/types/workflow-metadata.ts +2 -4
  88. package/src/types/workflow-plan.ts +1 -3
  89. package/src/utils/bundler.ts +88 -19
  90. package/src/workflow-steps/observability.ts +2 -3
  91. package/src/workflow-steps/runner.ts +5 -8
  92. package/src/workflow-steps/types.ts +8 -10
  93. package/src/workflow-steps/workflow.ts +2 -1
  94. package/src/workflows/engine.ts +3 -5
  95. package/src/workflows/invoke-child.ts +2 -2
  96. package/dist/generated/verb-synopsis.d.ts +0 -34
  97. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  98. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  99. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
  100. package/src/generated/verb-synopsis.ts +0 -544
@@ -0,0 +1,24 @@
1
+ /**
2
+ * A model id as a harness names it — the one derivation behind every
3
+ * `model_report` the Claude runtimes emit (AgentMessageModelReport): the
4
+ * `model` on claude-code's `system`/`init` event and the `message.model` on
5
+ * each `assistant` event, which the Agent SDK's messages mirror.
6
+ *
7
+ * - claude-code's `[1m]`-style suffix (`claude-opus-4-8[1m]`) is the CLI's
8
+ * context-window marker: a variant flag on the same model, not another
9
+ * model, so the report drops it;
10
+ * - `<synthetic>` is the harness's own voice (slash-command stdout,
11
+ * advisories) and names no model;
12
+ * - anything that is not a non-empty string names none.
13
+ *
14
+ * Machine-format rules only; nothing here reads words.
15
+ */
16
+
17
+ const MODEL_ID_MAX = 200;
18
+
19
+ export function reportedModelId(raw: unknown): string | undefined {
20
+ if (typeof raw !== "string") return undefined;
21
+ const id = raw.replace(/\[[^\]]*\]$/, "").trim();
22
+ if (id.length === 0 || id === "<synthetic>") return undefined;
23
+ return id.length > MODEL_ID_MAX ? id.slice(0, MODEL_ID_MAX) : id;
24
+ }
@@ -35,11 +35,14 @@
35
35
  */
36
36
 
37
37
  import type {
38
- AgentMessage, AgentMessageCompaction, AgentMessageTaskNotification, AgentMessageTaskProgress,
38
+ AgentMessage, AgentMessageCompaction, AgentMessageModelUsage, AgentMessagePlanLimits,
39
+ AgentMessageTaskNotification, AgentMessageTaskProgress, AgentMessageUsage,
39
40
  WorkflowProgressEntry,
40
41
  } from "../index.js";
41
42
  import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
43
+ import { reportedModelId } from "./_reported-model.js";
42
44
  import { formatError } from "../utils/errors.js";
45
+ import { CLAUDE_CODE_VERSION } from "../sandbox/baked-clis.js";
43
46
 
44
47
  function now(): string { return new Date().toISOString(); }
45
48
 
@@ -409,6 +412,106 @@ export function parseSystemCompaction(
409
412
  };
410
413
  }
411
414
 
415
+ // ── Plan limits (the account meter, off the stream) ─────────────────────────
416
+ //
417
+ // A subscription-funded `claude -p` reads the `anthropic-ratelimit-unified-*`
418
+ // headers off every response and re-emits them as `rate_limit_event` lines
419
+ // whenever its reading changes (the public Agent SDK's SDKRateLimitEvent;
420
+ // anthropics/claude-code#50518 shows the plainly-allowed shape:
421
+ // `{status: "allowed", resetsAt: 1729281600, rateLimitType: "five_hour"}`,
422
+ // with `utilization` carried only once a window crosses a warning
423
+ // threshold). `resetsAt` is epoch SECONDS. A `rejected` event is the plan's
424
+ // wall for the request just made, with the authoritative reset. The platform
425
+ // folds these into the funding account's meter (one reading per window) and
426
+ // learns a cooldown from a rejected one — never from the notice text.
427
+
428
+ const PLAN_LIMIT_STATUSES: ReadonlySet<string> = new Set(["allowed", "allowed_warning", "rejected"]);
429
+
430
+ /** Epoch seconds (the CLI's `resetsAt`) as ISO; undefined for anything else. */
431
+ function epochSecondsIso(v: unknown): string | undefined {
432
+ if (typeof v !== "number" || !Number.isFinite(v) || v <= 0) return undefined;
433
+ return new Date(v * 1000).toISOString();
434
+ }
435
+
436
+ /** One `rate_limit_event` mapped onto the structured plan-limits message,
437
+ * or null when the event names no recognisable status. Pure and tolerant
438
+ * over untrusted harness JSON: every field is forwarded only when present
439
+ * and well-typed; nothing is defaulted or invented. Exported for tests. */
440
+ export function parsePlanLimits(
441
+ p: Record<string, unknown>, timestamp: string,
442
+ ): AgentMessagePlanLimits | null {
443
+ const info = (typeof p.rate_limit_info === "object" && p.rate_limit_info !== null
444
+ ? p.rate_limit_info : null) as Record<string, unknown> | null;
445
+ if (!info || typeof info.status !== "string" || !PLAN_LIMIT_STATUSES.has(info.status)) return null;
446
+ const num = (v: unknown): number | undefined =>
447
+ typeof v === "number" && Number.isFinite(v) ? v : undefined;
448
+ const str = (v: unknown): string | undefined =>
449
+ typeof v === "string" && v.length > 0 ? clip(v, 100) : undefined;
450
+ const window = str(info.rateLimitType);
451
+ const resetsAt = epochSecondsIso(info.resetsAt);
452
+ const utilization = num(info.utilization);
453
+ const surpassedThreshold = num(info.surpassedThreshold);
454
+ const overageStatus = typeof info.overageStatus === "string" && PLAN_LIMIT_STATUSES.has(info.overageStatus)
455
+ ? info.overageStatus as AgentMessagePlanLimits["status"] : undefined;
456
+ const overageResetsAt = epochSecondsIso(info.overageResetsAt);
457
+ const overageDisabledReason = str(info.overageDisabledReason);
458
+ const overageInUse = typeof info.isUsingOverage === "boolean" ? info.isUsingOverage
459
+ : typeof info.overageInUse === "boolean" ? info.overageInUse : undefined;
460
+ const overage = overageStatus !== undefined || overageResetsAt !== undefined
461
+ || overageDisabledReason !== undefined || overageInUse !== undefined
462
+ ? {
463
+ ...(overageStatus !== undefined ? { status: overageStatus } : {}),
464
+ ...(overageResetsAt !== undefined ? { resetsAt: overageResetsAt } : {}),
465
+ ...(overageDisabledReason !== undefined ? { disabledReason: overageDisabledReason } : {}),
466
+ ...(overageInUse !== undefined ? { inUse: overageInUse } : {}),
467
+ }
468
+ : undefined;
469
+ const limitScope = str(info.limitScope);
470
+ const errorCode = str(info.errorCode);
471
+ return {
472
+ type: "plan_limits",
473
+ status: info.status as AgentMessagePlanLimits["status"],
474
+ ...(window !== undefined ? { window } : {}),
475
+ ...(resetsAt !== undefined ? { resetsAt } : {}),
476
+ ...(utilization !== undefined ? { utilization } : {}),
477
+ ...(surpassedThreshold !== undefined ? { surpassedThreshold } : {}),
478
+ ...(overage !== undefined ? { overage } : {}),
479
+ ...(limitScope !== undefined ? { limitScope } : {}),
480
+ ...(errorCode !== undefined ? { errorCode } : {}),
481
+ timestamp,
482
+ };
483
+ }
484
+
485
+ /** The terminal result's per-model report (`result.modelUsage`, keyed by
486
+ * the raw model string) mapped onto the protocol's per-model shape, or
487
+ * undefined when the result carries none. Counts the harness did not
488
+ * report stay absent (the four classes default to 0 only when the entry
489
+ * exists at all — an entry IS a report of that model). Pure; exported for
490
+ * tests. */
491
+ export function parseModelUsage(raw: unknown): Record<string, AgentMessageModelUsage> | undefined {
492
+ if (typeof raw !== "object" || raw === null) return undefined;
493
+ const num = (v: unknown): number => (typeof v === "number" && Number.isFinite(v) ? v : 0);
494
+ const opt = (v: unknown): number | undefined => (typeof v === "number" && Number.isFinite(v) ? v : undefined);
495
+ const out: Record<string, AgentMessageModelUsage> = {};
496
+ for (const [model, entry] of Object.entries(raw as Record<string, unknown>)) {
497
+ if (typeof entry !== "object" || entry === null || model.length === 0) continue;
498
+ const e = entry as Record<string, unknown>;
499
+ const thinkingTokens = opt(e.thinkingTokens);
500
+ const webSearchRequests = opt(e.webSearchRequests);
501
+ const costUsd = opt(e.costUSD);
502
+ out[clip(model, 200)] = {
503
+ inputTokens: num(e.inputTokens),
504
+ outputTokens: num(e.outputTokens),
505
+ cacheReadTokens: num(e.cacheReadInputTokens),
506
+ cacheCreationTokens: num(e.cacheCreationInputTokens),
507
+ ...(thinkingTokens !== undefined ? { thinkingTokens } : {}),
508
+ ...(webSearchRequests !== undefined ? { webSearchRequests } : {}),
509
+ ...(costUsd !== undefined ? { costUsd } : {}),
510
+ };
511
+ }
512
+ return Object.keys(out).length > 0 ? out : undefined;
513
+ }
514
+
412
515
  /** Claude Code's real reasoning knob is its own `--effort <level>` flag
413
516
  * (low|medium|high|xhigh|max — verified against `claude -p --help`). The
414
517
  * CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
@@ -419,6 +522,50 @@ export function parseSystemCompaction(
419
522
  export const CLAUDE_CODE_EFFORT_LEVELS: readonly CliReasoningEffort[] =
420
523
  ["low", "medium", "high", "xhigh", "max"];
421
524
 
525
+ /** The Bash PreToolUse hook rtk installs for Claude Code (`rtk init -g
526
+ * --hook-only` writes exactly this entry — matcher `Bash`, command
527
+ * `rtk hook claude` — into ~/.claude/settings.json; the image bakes
528
+ * RTK_VERSION, sandbox/baked-clis.ts, which this payload was run against).
529
+ * `rtk hook claude` reads the PreToolUse JSON on stdin and, when it has a
530
+ * filter for the command, answers with `updatedInput` rewriting `git
531
+ * status` to `rtk git status` (ls, find, grep, test runners, bun, curl,
532
+ * docker, …), so the model reads rtk's compact output. A command it
533
+ * cannot compress — an unknown tool, a pipe into one, command or process
534
+ * substitution, a heredoc, a file redirect, anything already prefixed
535
+ * `rtk`, or a `RTK_DISABLED=1` prefix — gets no output and exit 0, which
536
+ * Claude Code treats as "no decision": the original command runs
537
+ * unchanged (all verified against the binary).
538
+ *
539
+ * The wrapper fails OPEN both ways. `command -v` covers images without
540
+ * rtk (the devbox, a bare Vercel VM): there a bare `rtk hook claude` would
541
+ * fail every Bash call's hook with exit 127 — non-blocking, but a stderr
542
+ * warning per call; rtk's own legacy shell hook degraded the same way
543
+ * ("binary not found: exit 0"). The trailing `exit 0` covers an rtk that
544
+ * cannot answer: Claude Code treats a PreToolUse hook's exit 2 as a BLOCK
545
+ * of the tool call with stderr fed to the model, and clap exits 2 with
546
+ * its usage for a subcommand it does not know — which is how the codex
547
+ * hook (RTK_CODEX_HOOK_COMMAND, codex.ts) blocked every codex shell
548
+ * command on 2026-10-03, when the image's rtk was a cache-served 0.45.0.
549
+ * rtk's hooks never exit non-zero on purpose, so a non-zero exit is a
550
+ * broken rtk, and the compressor must never cost the worker its shell:
551
+ * the exit code is dropped, and the smoke gate's rtk-hook-rewrite check
552
+ * is what proves the rewrite itself. */
553
+ export const RTK_BASH_HOOK_COMMAND =
554
+ "command -v rtk >/dev/null 2>&1 && rtk hook claude; exit 0";
555
+
556
+ /** Settings every platform `claude` launch passes as `--settings` (inline
557
+ * JSON). Claude Code MERGES hook entries across its settings sources
558
+ * instead of replacing them (user → project → local → flag → managed), so
559
+ * this rides beside whatever the machine's own ~/.claude/settings.json
560
+ * carries, and nothing is written to that file — the right outcome, since
561
+ * the server never owns it: home carry tar-restores it whole across
562
+ * machines and the image capture strips it as personal state. */
563
+ export const CLAUDE_CODE_PLATFORM_SETTINGS = {
564
+ hooks: {
565
+ PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_BASH_HOOK_COMMAND }] }],
566
+ },
567
+ } as const;
568
+
422
569
  export const claudeCodeSpec: CliAgentSpec = {
423
570
  kind: "claude-code",
424
571
  authEnv: "ANTHROPIC_API_KEY",
@@ -428,11 +575,14 @@ export const claudeCodeSpec: CliAgentSpec = {
428
575
  acp: { command: "npx", args: ["--yes", CLAUDE_CODE_ACP_ADAPTER] },
429
576
  // Bare-sandbox fallback only: the agent-env image bakes /usr/local/bin/claude
430
577
  // (same recipe — infra/e2b-template/build.ts), so the probe short-circuits
431
- // on the platform path. Anthropic's installer drops a versioned binary under
432
- // $HOME with a ~/.local/bin/claude launcher; resolve the symlink and copy
433
- // the self-contained binary to a world-executable system path.
578
+ // on the platform path. The SAME pinned version as the image
579
+ // (CLAUDE_CODE_VERSION), never latest, so a machine the image predates runs
580
+ // what the image's machines run. Anthropic's installer drops a versioned
581
+ // binary under $HOME with a ~/.local/bin/claude launcher; resolve the
582
+ // symlink and copy the self-contained binary to a world-executable system
583
+ // path.
434
584
  install:
435
- 'curl -fsSL https://claude.ai/install.sh | bash && ' +
585
+ `curl -fsSL https://claude.ai/install.sh | bash -s ${CLAUDE_CODE_VERSION} && ` +
436
586
  'REAL=$(readlink -f "$HOME/.local/bin/claude") && test -f "$REAL" && ' +
437
587
  'sudo cp "$REAL" /usr/local/bin/claude && sudo chmod 0755 /usr/local/bin/claude',
438
588
  // Claude reads the prompt from stdin in -p mode.
@@ -447,7 +597,8 @@ export const claudeCodeSpec: CliAgentSpec = {
447
597
  // the same build). A message that lands after the result would start a
448
598
  // NEW turn in-process, which is why the transport's feeder stops at the
449
599
  // result line instead of forwarding past it.
450
- streamInput: {
600
+ midTurnInput: {
601
+ transport: "stdin-stream",
451
602
  promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
452
603
  messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
453
604
  // The ESC equivalent (verified live against claude 2.1.236): the CLI's
@@ -479,6 +630,10 @@ export const claudeCodeSpec: CliAgentSpec = {
479
630
  // claude's own attestation env for exactly this: it lifts the flag's
480
631
  // root-user refusal (the E2B agent-env user is root).
481
632
  "--dangerously-skip-permissions",
633
+ // rtk's Bash PreToolUse hook (RTK_BASH_HOOK_COMMAND): shell output is
634
+ // compressed before it reaches the model. Flag settings merge with the
635
+ // machine's own settings; they never replace them.
636
+ `--settings ${shellQuote(JSON.stringify(CLAUDE_CODE_PLATFORM_SETTINGS))}`,
482
637
  ...(model ? [`--model ${shellQuote(model)}`] : []),
483
638
  // Reasoning effort is the CLI's own flag; the value comes from the
484
639
  // closed CliReasoningEffort set, so it is shell-safe unquoted.
@@ -514,7 +669,13 @@ export const claudeCodeSpec: CliAgentSpec = {
514
669
  // Structural marker only — no content sniffing.
515
670
  const synthetic = message?.model === "<synthetic>";
516
671
  const blocks = Array.isArray(message?.content) ? message.content : [];
517
- return blocks.flatMap((b): AgentMessage[] => {
672
+ // The model the API answered with, named on every assistant event
673
+ // (AgentMessageModelReport): the truth after a /model switch or a
674
+ // fallback. Top level only — a subagent's events name the
675
+ // subagent's model, not the worker's; `<synthetic>` names none.
676
+ const model = parentId ? undefined : reportedModelId(message?.model);
677
+ const report: AgentMessage[] = model ? [{ type: "model_report", model, timestamp: ts }] : [];
678
+ return report.concat(blocks.flatMap((b): AgentMessage[] => {
518
679
  if (b.type === "text" && typeof b.text === "string" && b.text.length > 0) {
519
680
  return [{ type: synthetic ? "harness_notice" : "text", text: b.text, timestamp: ts }];
520
681
  }
@@ -528,7 +689,7 @@ export const claudeCodeSpec: CliAgentSpec = {
528
689
  }];
529
690
  }
530
691
  return [];
531
- });
692
+ }));
532
693
  }
533
694
  // User API message: the CLI echoes tool results back as user content,
534
695
  // and injects `<task-notification>` blocks (background-task completion
@@ -614,14 +775,21 @@ export const claudeCodeSpec: CliAgentSpec = {
614
775
  return [];
615
776
  }
616
777
  // Terminal result: usage on success (the text already streamed via the
617
- // assistant events); an honest string error on failure.
778
+ // assistant events); an honest string error on failure. The turn
779
+ // totals are the main loop's `usage`; `modelUsage` (every model the
780
+ // query pipeline called, with the CLI's own cost estimate) and
781
+ // `total_cost_usd` ride beside them as the harness reported them —
782
+ // cumulative for the guest session (AgentMessageUsage.byModel).
618
783
  case "result": {
619
784
  if (p.is_error === true) {
620
785
  return [{ type: "error", text: formatError(p.result ?? p.subtype ?? p), timestamp: ts }];
621
786
  }
622
787
  const u = p.usage as Record<string, number> | undefined;
623
788
  if (!u) return [];
624
- return [{
789
+ const byModel = parseModelUsage(p.modelUsage);
790
+ const costUsd = typeof p.total_cost_usd === "number" && Number.isFinite(p.total_cost_usd)
791
+ ? p.total_cost_usd : undefined;
792
+ const usage: AgentMessageUsage = {
625
793
  type: "usage",
626
794
  inputTokens: u.input_tokens ?? 0,
627
795
  outputTokens: u.output_tokens ?? 0,
@@ -629,10 +797,21 @@ export const claudeCodeSpec: CliAgentSpec = {
629
797
  cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
630
798
  durationMs: typeof p.duration_ms === "number" ? p.duration_ms : 0,
631
799
  numTurns: typeof p.num_turns === "number" ? p.num_turns : 1,
800
+ ...(byModel !== undefined ? { byModel } : {}),
801
+ ...(costUsd !== undefined ? { costUsd } : {}),
632
802
  timestamp: ts,
633
- }];
803
+ };
804
+ return [usage];
634
805
  }
635
- // System events: init carries the session id (extractSessionId), and
806
+ // The account's plan limits as the CLI read them off the response
807
+ // headers (subscription-funded sessions; parsePlanLimits above).
808
+ case "rate_limit_event": {
809
+ const limits = parsePlanLimits(p, ts);
810
+ return limits ? [limits] : [];
811
+ }
812
+ // System events: init carries the session id (extractSessionId) and
813
+ // the session's resolved model (the first model report of the turn;
814
+ // the assistant events confirm or change it), and
636
815
  // the background-task lane rides here too — `task_notification` is
637
816
  // the completion evidence stream-json actually emits (see the section
638
817
  // header above), and `task_progress` the LIVE background-task feed
@@ -644,6 +823,10 @@ export const claudeCodeSpec: CliAgentSpec = {
644
823
  // under system (task_started, task_updated, background_tasks_changed,
645
824
  // thinking_tokens) is lifecycle noise here.
646
825
  case "system": {
826
+ if (p.subtype === "init") {
827
+ const model = reportedModelId(p.model);
828
+ return model ? [{ type: "model_report", model, timestamp: ts }] : [];
829
+ }
647
830
  if (p.subtype === "task_progress") {
648
831
  const progress = parseSystemTaskProgress(p, ts);
649
832
  return progress ? [progress] : [];
@@ -1,4 +1,4 @@
1
- /** Claude Agent SDK runtime — replaces the old Claude CLI subprocess runtime. */
1
+ /** Claude Agent SDK runtime. */
2
2
 
3
3
  import { query, type HookCallback, type PreToolUseHookInput, type SDKUserMessage, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
4
4
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
@@ -9,6 +9,7 @@ import { runProcessorChain } from "../processors/runner.js";
9
9
  import type { ProcessorContext, ToolCall } from "../processors/processor.js";
10
10
  import { RequestContext } from "../request-context/request-context.js";
11
11
  import { boundProcessorPause } from "../pause/pause-core.js";
12
+ import { reportedModelId } from "./_reported-model.js";
12
13
  import { formatError } from "../utils/errors.js";
13
14
 
14
15
  function now(): string { return new Date().toISOString(); }
@@ -19,11 +20,17 @@ function translateMessage(message: Record<string, unknown>): AgentMessage[] {
19
20
 
20
21
  if (message.type === "system" && message.subtype === "init") {
21
22
  msgs.push({ type: "init", sessionId: String(message.session_id ?? ""), timestamp: ts });
23
+ // The session's resolved model, as the harness names it
24
+ // (AgentMessageModelReport); the assistant messages confirm or change it.
25
+ const model = reportedModelId(message.model);
26
+ if (model) msgs.push({ type: "model_report", model, timestamp: ts });
22
27
  return msgs;
23
28
  }
24
29
 
25
30
  if (message.type === "assistant") {
26
- const raw = message.message as { content?: unknown[] } | undefined;
31
+ const raw = message.message as { content?: unknown[]; model?: unknown } | undefined;
32
+ const model = typeof message.parent_tool_use_id === "string" ? undefined : reportedModelId(raw?.model);
33
+ if (model) msgs.push({ type: "model_report", model, timestamp: ts });
27
34
  for (const block of raw?.content ?? []) {
28
35
  const b = block as Record<string, unknown>;
29
36
  if (b.type === "text") msgs.push({ type: "text", text: String(b.text ?? ""), timestamp: ts });
@@ -5,17 +5,28 @@
5
5
  * only stream-parse what it prints.
6
6
  *
7
7
  * Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
8
- * workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
9
- * demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
10
- * to install once and boot from the captured snapshot on every run after.
8
+ * workflow secret. The E2B session image bakes the pinned `codex` CLI
9
+ * (`@openai/codex@CODEX_CLI_VERSION`); on any other machine the runtime
10
+ * installs that same version on demand (pair with
11
+ * `snapshots: { bootFrom: "reuse" }` to install once and boot from the
12
+ * captured snapshot on every run after).
11
13
  *
12
14
  * Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
13
- * with the command_execution / reasoning / agent_message item shapes below.
15
+ * with the command_execution / reasoning / agent_message item shapes below;
16
+ * the plan's `todo_list` item against codex-rs rust-v0.153.4 and
17
+ * rust-v0.159.2 (see codexPlanMessages); the command_execution /
18
+ * agent_message / turn.completed shapes re-verified live against 0.160.0
19
+ * (the pin), whose plan-tool sources are byte-identical to rust-v0.159.2.
14
20
  */
15
21
 
16
22
  import type { AgentMessage } from "../index.js";
17
- import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
23
+ import type { AgentMessagePlan } from "../types/protocol.js";
24
+ import {
25
+ MID_TURN_DELIVERED_ENV, MID_TURN_INBOX_ENV, createCliAgentRuntime, shellQuote,
26
+ type CliAgentSpec, type CliReasoningEffort,
27
+ } from "./_cli-agent.js";
18
28
  import { formatError } from "../utils/errors.js";
29
+ import { CODEX_CLI_VERSION } from "../sandbox/baked-clis.js";
19
30
 
20
31
  function now(): string { return new Date().toISOString(); }
21
32
 
@@ -24,6 +35,97 @@ function now(): string { return new Date().toISOString(); }
24
35
  * overrides the bundled `@openai/codex` the adapter drives underneath. */
25
36
  const CODEX_ACP_ADAPTER = "@agentclientprotocol/codex-acp@0.1.0";
26
37
 
38
+ /** The Bash PreToolUse hook rtk installs for Codex (`rtk init -g --codex`
39
+ * writes exactly this entry — matcher `Bash`, command `rtk hook codex` —
40
+ * into $CODEX_HOME/hooks.json). `rtk hook codex` exists since rtk 0.50.0;
41
+ * the image bakes RTK_VERSION (sandbox/baked-clis.ts), which this payload
42
+ * was run against. Codex's shell tool matches the `Bash` matcher (codex-rs
43
+ * rust-v0.160.0, the pinned CODEX_CLI_VERSION; identical at
44
+ * rust-v0.159.2), and the hook runs under `$SHELL -lc` with the PreToolUse
45
+ * JSON on stdin. `rtk hook codex` answers `permissionDecision: "allow"` +
46
+ * `updatedInput` rewriting `git status` to `rtk git status` when it has a
47
+ * filter for the command; Codex applies the replacement before its own
48
+ * approval and sandbox checks. For a command it cannot compress (an
49
+ * unknown tool, a pipe into one, substitutions, heredocs, redirects, a
50
+ * `RTK_DISABLED=1` prefix, an unknown permission mode) it prints nothing
51
+ * and exits 0, and Codex runs the original unchanged.
52
+ *
53
+ * The wrapper fails OPEN both ways, like RTK_BASH_HOOK_COMMAND
54
+ * (claude-code.ts). `command -v` covers an image without rtk (the devbox,
55
+ * a bare Vercel VM): silent, exit 0, so Codex runs the command instead of
56
+ * failing every shell call's hook with 127. The trailing `exit 0` covers
57
+ * an rtk that cannot answer: Codex treats a PreToolUse hook's exit 2 with
58
+ * stderr as a BLOCK of the tool call (codex-rs hooks/src/events/
59
+ * pre_tool_use.rs at rust-v0.160.0 — the model reads "Command blocked by
60
+ * PreToolUse hook: <stderr>"), and clap exits 2 with its usage on stderr
61
+ * for a subcommand it does not know. That was 2026-10-03: the image's rtk
62
+ * was a cache-served 0.45.0 with no `hook codex`, and this hook — `exec`
63
+ * handing rtk's exit code to Codex — blocked every shell command of every
64
+ * codex session. rtk's hooks never exit non-zero on purpose (they fail
65
+ * open with no stdout), so a non-zero exit is always a broken rtk, and the
66
+ * compressor must never cost the worker its shell: the exit code is
67
+ * dropped, and the smoke gate's codex-rtk-hook-rewrite check is what
68
+ * proves the rewrite itself. */
69
+ export const RTK_CODEX_HOOK_COMMAND =
70
+ "command -v rtk >/dev/null 2>&1 && rtk hook codex; exit 0";
71
+
72
+ /** The awk program of the mid-turn hook below: the inbox lines not yet fed
73
+ * (JSON string literals, one per line — `codexSpec.midTurnInput.messageLine`)
74
+ * become ONE PostToolUse hook output. Each literal's quotes are stripped and
75
+ * the bodies are joined with an escaped blank line; the bodies are already
76
+ * JSON-escaped, so the result is one valid JSON string. Nothing is printed
77
+ * when no line is due, and codex ignores an empty stdout. */
78
+ const CODEX_MID_TURN_HOOK_AWK = String.raw`NR > fed && NR <= upto { s = substr($0, 2, length($0) - 2); out = (out == "" ? s : out "\\n\\n" s) } END { if (out != "") printf "{\"hookSpecificOutput\":{\"hookEventName\":\"PostToolUse\",\"additionalContext\":\"%s\"}}\n", out }`;
79
+
80
+ /** The PostToolUse command hook that carries a mid-turn message into a
81
+ * RUNNING codex turn (the 2026-10-02 relayed-steer incident: three of the
82
+ * owner's instructions waited 48-51 minutes for a codex build to end).
83
+ * Codex runs it, under `$SHELL -lc` with the event JSON on stdin and the
84
+ * codex process's own environment, after every tool call it completes
85
+ * (codex-rs core/src/tools/registry.rs → hook_runtime.rs at rust-v0.160.0,
86
+ * the pin). The launch wrapper exported the turn's inbox and delivered-
87
+ * counter paths into that environment (sdk _cli-agent.ts
88
+ * toolHookInboxFragment); the hook forwards the inbox lines past the
89
+ * counter as the hook's `additionalContext`, which codex records as
90
+ * developer context in the live turn's history before the model's next
91
+ * request (hook_runtime.rs record_additional_contexts), then advances the
92
+ * counter — the same ack the server's inject lane trusts for the stdin
93
+ * lane. Not a platform turn (no inbox in the environment): silent, exit 0.
94
+ * Codex's default spill threshold for a hook's additional context is 2,500
95
+ * tokens (hooks/src/output_spill.rs); a longer message reaches the model
96
+ * as a preview plus a file pointer, which a steer never is. */
97
+ export const CODEX_MID_TURN_HOOK_COMMAND =
98
+ // Drain the event JSON first, so codex's stdin write never meets a closed pipe.
99
+ "cat >/dev/null; "
100
+ + `[ -n "\${${MID_TURN_INBOX_ENV}:-}" ] && [ -f "$${MID_TURN_INBOX_ENV}" ] || exit 0; `
101
+ + `ac_fed=$(cat "$${MID_TURN_DELIVERED_ENV}" 2>/dev/null); ac_fed=\${ac_fed:-0}; `
102
+ + `ac_lines=$(wc -l < "$${MID_TURN_INBOX_ENV}" 2>/dev/null); ac_lines=\${ac_lines:-0}; `
103
+ + `[ "$ac_lines" -gt "$ac_fed" ] || exit 0; `
104
+ + `awk -v fed="$ac_fed" -v upto="$ac_lines" '${CODEX_MID_TURN_HOOK_AWK}' "$${MID_TURN_INBOX_ENV}" `
105
+ + `&& echo "$ac_lines" > "$${MID_TURN_DELIVERED_ENV}"; exit 0`;
106
+
107
+ /** $CODEX_HOME/hooks.json for every platform Codex session (the server
108
+ * writes it at boot, session-runtime-config.ts). Codex only RUNS a
109
+ * user-level hook it has persisted trust for — the TUI's review prompt
110
+ * has no headless counterpart, and `--dangerously-bypass-hook-trust` would
111
+ * run a cloned repository's `.codex/hooks.json` unreviewed too — so the
112
+ * server also writes the trust record for each of these exact hooks into
113
+ * the config.toml it generates (sandbox/codex-hooks.ts). Verified live
114
+ * against codex 0.159.2 and 0.160.0 for the rtk hook: with the record,
115
+ * `codex exec` rewrote `git status` through rtk with no bypass flag;
116
+ * without it, the hook was skipped.
117
+ *
118
+ * The PostToolUse group carries NO matcher: codex runs a matcher-less hook
119
+ * after every tool it completes (hooks/src/events/common.rs
120
+ * matches_matcher: an absent matcher is a match), so a mid-turn message
121
+ * lands at the next tool step whatever the tool was. */
122
+ export const CODEX_PLATFORM_HOOKS = {
123
+ hooks: {
124
+ PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_CODEX_HOOK_COMMAND }] }],
125
+ PostToolUse: [{ hooks: [{ type: "command", command: CODEX_MID_TURN_HOOK_COMMAND }] }],
126
+ },
127
+ } as const;
128
+
27
129
  /** Known-noise codex ADVISORY lines. codex emits these as `item.completed`
28
130
  * error items on the `--json` stream (exec maps every `Warning` notification
29
131
  * to an error item — verified against codex-cli 0.147.0), so without a filter
@@ -101,6 +203,41 @@ export function codexWriterLockPreflight(
101
203
  + `flock -n "$ac_lk" rm -f -- "$ac_lk" 2>/dev/null || true; fi; `;
102
204
  }
103
205
 
206
+ /**
207
+ * codex's PLAN (its update_plan tool) on the `--json` stream: one
208
+ * `todo_list` item per turn, `item.started` on the plan's first update,
209
+ * `item.updated` (same id, the whole list) on every later one, and
210
+ * `item.completed` at the turn's end. That is codex-rs exec's
211
+ * event_processor_with_jsonl_output.rs, identical in rust-v0.153.4 and
212
+ * rust-v0.159.2, where it is the only `item.updated` codex emits; the item
213
+ * `{ id, type: "todo_list", items: [{ text, completed }] }` is the shape every
214
+ * todo_list item our machines have persisted carries. Each event maps to ONE
215
+ * whole-plan `plan` message, so the transcript's checklist, and the worker's
216
+ * status line, follow the plan while the turn runs (before this, the list
217
+ * surfaced once, at the turn's end, as a raw `todo_list` tool card).
218
+ *
219
+ * A step carries only `completed`: codex's `pending` and `in_progress` both
220
+ * serialize as false. Its plan tool allows at most one step in progress and
221
+ * a plan runs in order, so the first step not completed is the one under
222
+ * way; the rest are pending. The stream names no priority, so none is set.
223
+ */
224
+ function codexPlanMessages(item: Record<string, unknown>, timestamp: string): AgentMessage[] {
225
+ const steps = Array.isArray(item.items) ? item.items : [];
226
+ let underWay = false;
227
+ const entries: AgentMessagePlan["entries"] = [];
228
+ for (const step of steps) {
229
+ if (typeof step !== "object" || step === null) continue;
230
+ const text = (step as { text?: unknown }).text;
231
+ if (typeof text !== "string" || text.trim().length === 0) continue;
232
+ let status: AgentMessagePlan["entries"][number]["status"];
233
+ if ((step as { completed?: unknown }).completed === true) status = "completed";
234
+ else if (!underWay) { status = "in_progress"; underWay = true; }
235
+ else status = "pending";
236
+ entries.push({ content: text.trim(), status });
237
+ }
238
+ return entries.length > 0 ? [{ type: "plan", entries, timestamp }] : [];
239
+ }
240
+
104
241
  /** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
105
242
  * `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
106
243
  * of the public runtime surface — `createCodexRuntime` stays the entry point. */
@@ -113,6 +250,22 @@ export const codexSpec: CliAgentSpec = {
113
250
  // every resume of the same thread, so supersede teardown must KILL it —
114
251
  // never detach it alive (server runner-kill.ts consumes this flag).
115
252
  exclusiveSessionWriter: true,
253
+ // Mid-turn input (the 2026-10-02 relayed-steer incident). codex reads
254
+ // stdin ONCE, as the prompt: `codex exec -` submits one UserTurn and never
255
+ // reads stdin again (codex-rs exec/src/lib.rs read_prompt_from_stdin and
256
+ // its single InitialOperation::UserTurn at rust-v0.160.0; the app-server's
257
+ // `turn/steer` is a different transport, not `codex exec`). So a running
258
+ // turn takes a message through the PostToolUse hook above: codex runs it
259
+ // after every tool call it completes, with the codex process's own
260
+ // environment; the hook prints the unfed inbox lines as `additionalContext`
261
+ // and codex records them as developer context in the live turn's history
262
+ // before the model's next request. One JSON string literal per inbox line,
263
+ // so the hook splices bodies without decoding. Codex builds the PostToolUse
264
+ // payload only for a tool call it counts as successful (tools/registry.rs),
265
+ // so a message lands at the next successful tool step. Source-verified at
266
+ // rust-v0.160.0 (the pinned CODEX_CLI_VERSION); the end-to-end run against
267
+ // a live codex is the acceptance check still owed.
268
+ midTurnInput: { transport: "tool-hook", messageLine: (text) => JSON.stringify(text) },
116
269
  // ACP-mode invocation (ADR-0020 increment 1). `codex` has no native `--acp`
117
270
  // flag; the adapter (a Rust binary shipped via npm) speaks ACP and drives a
118
271
  // compatible bundled `@openai/codex`. The runner attempts this first and
@@ -125,9 +278,11 @@ export const codexSpec: CliAgentSpec = {
125
278
  // (e.g. a sandbox-baked one) instead of the adapter's bundled copy.
126
279
  ...(process.env.CODEX_PATH ? { env: { CODEX_PATH: process.env.CODEX_PATH } } : {}),
127
280
  },
128
- // Global npm install; symlink onto PATH only if the global bin dir isn't
129
- // already there (so a non-login `sh -c` can find it).
130
- install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
281
+ // The E2B session image bakes this exact version (sandbox/baked-clis.ts),
282
+ // so this runs only on a machine whose image predates the bake: a global
283
+ // npm install of the SAME pin, symlinked onto PATH only if the global bin
284
+ // dir isn't already there (so a non-login `sh -c` can find it).
285
+ install: `sudo npm install -g @openai/codex@${CODEX_CLI_VERSION} && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)`,
131
286
  // Codex reads the prompt from stdin when invoked as `codex exec ... -`.
132
287
  promptPayload: (prompt) => prompt,
133
288
  buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
@@ -140,13 +295,13 @@ export const codexSpec: CliAgentSpec = {
140
295
  // for running in environments that are externally sandboxed".
141
296
  "--dangerously-bypass-approvals-and-sandbox",
142
297
  ...(model ? ["-m", shellQuote(model)] : []),
143
- // codex's own reasoning knob — a config override, valid values
144
- // none|minimal|low|medium|high|xhigh (codex docs; xhigh is the
145
- // codex-max-tier deep-reasoning level). There is NO "max" in codex's
146
- // vocabulary: the server rejects it for codex sessions
147
- // (sessionEffortLockError), and this clamp to xhigh is the type-level
148
- // backstop for a caller that bypasses that gate. The value comes from
149
- // the closed CliReasoningEffort set, so it is shell-safe unquoted.
298
+ // codex's own reasoning knob — a config override. Every codex model
299
+ // takes low|medium|high|xhigh; only the GPT-6 and GPT-5.6 presets add
300
+ // max (codex 0.153+), so the platform offers codex up to xhigh: the
301
+ // server rejects "max" for codex sessions (sessionEffortLockError), and
302
+ // this clamp to xhigh is the type-level backstop for a caller that
303
+ // bypasses that gate. The value comes from the closed
304
+ // CliReasoningEffort set, so it is shell-safe unquoted.
150
305
  ...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
151
306
  ].join(" ");
152
307
  // Working directory via a shell `cd` — the idiom EVERY other runtime uses
@@ -172,11 +327,19 @@ export const codexSpec: CliAgentSpec = {
172
327
  mapEvent: (p): AgentMessage[] => {
173
328
  const ts = now();
174
329
  switch (p.type) {
330
+ case "item.updated": {
331
+ // Only the plan updates in place (codexPlanMessages); any other
332
+ // item surfaces on its started / completed events below.
333
+ const item = p.item as Record<string, unknown> | undefined;
334
+ return item?.type === "todo_list" ? codexPlanMessages(item, ts) : [];
335
+ }
175
336
  case "item.started":
176
337
  case "item.completed": {
177
338
  const item = p.item as Record<string, unknown> | undefined;
178
339
  if (!item) return [];
179
340
  const itype = String(item.type ?? "");
341
+ // The plan, whole, on its first update and at the turn's end.
342
+ if (itype === "todo_list") return codexPlanMessages(item, ts);
180
343
  // Text + reasoning land on completion (started carries no final text).
181
344
  if (itype === "agent_message") {
182
345
  return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
@@ -209,12 +372,17 @@ export const codexSpec: CliAgentSpec = {
209
372
  }
210
373
  return [{ type: "error", text, timestamp: ts }];
211
374
  }
212
- // file_change / mcp_tool_call / web_search / todo: surface once, on completion.
375
+ // file_change / mcp_tool_call / web_search: surface once, on completion.
213
376
  if (p.type === "item.completed") {
214
377
  return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
215
378
  }
216
379
  return [];
217
380
  }
381
+ // The turn's token classes as codex reports them (codex-rs
382
+ // exec_events.rs `Usage`): input, cached input, cache-write input,
383
+ // output and reasoning output. Every class is kept — cache writes land
384
+ // in the creation class, reasoning rides beside output (it is counted
385
+ // inside output_tokens, so the four totals stay additive).
218
386
  case "turn.completed": {
219
387
  const u = p.usage as Record<string, number> | undefined;
220
388
  if (!u) return [];
@@ -223,9 +391,10 @@ export const codexSpec: CliAgentSpec = {
223
391
  inputTokens: u.input_tokens ?? 0,
224
392
  outputTokens: u.output_tokens ?? 0,
225
393
  cacheReadTokens: u.cached_input_tokens ?? 0,
226
- cacheCreationTokens: 0,
394
+ cacheCreationTokens: u.cache_write_input_tokens ?? 0,
227
395
  durationMs: 0,
228
396
  numTurns: 1,
397
+ ...(typeof u.reasoning_output_tokens === "number" ? { reasoningOutputTokens: u.reasoning_output_tokens } : {}),
229
398
  timestamp: ts,
230
399
  }];
231
400
  }
@@ -238,8 +407,8 @@ export const codexSpec: CliAgentSpec = {
238
407
  },
239
408
  };
240
409
 
241
- /** The effort levels codex actually has (`model_reasoning_effort`):
242
- * low|medium|high|xhigh — no "max" (that level is Claude Code's alone). */
410
+ /** The effort levels this runtime passes to codex (`model_reasoning_effort`):
411
+ * low|medium|high|xhigh, the levels every codex model takes. */
243
412
  export type CodexReasoningEffort = Exclude<CliReasoningEffort, "max">;
244
413
 
245
414
  export interface CodexRuntimeConfig {