@agent-compose/sdk 0.8.5 → 0.8.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +213 -189
- package/dist/agent/agent-context.d.ts +3 -3
- package/dist/agent/agent-loop.d.ts +6 -5
- package/dist/agent/perf-sampler.d.ts +27 -2
- package/dist/agent/run-agent.d.ts +1 -1
- package/dist/client.d.ts +119 -54
- package/dist/directives.d.ts +3 -3
- package/dist/display.d.ts +7 -0
- package/dist/errors.d.ts +1 -1
- package/dist/generated/agentc-commands.d.ts +34 -0
- package/dist/index.d.ts +12 -12
- package/dist/index.js +771 -204
- package/dist/request-context/request-context.d.ts +1 -1
- package/dist/runtimes/_cli-agent.d.ts +185 -68
- package/dist/runtimes/_reported-model.d.ts +16 -0
- package/dist/runtimes/claude-code.d.ts +60 -1
- package/dist/runtimes/claude.d.ts +1 -1
- package/dist/runtimes/codex.d.ts +94 -6
- package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
- package/dist/runtimes/model-report.test.d.ts +14 -0
- package/dist/runtimes/openai-desktop.js +741 -200
- package/dist/runtimes/opencode.d.ts +48 -11
- package/dist/runtimes/opencode.test.d.ts +14 -0
- package/dist/sandbox/baked-clis.d.ts +75 -0
- package/dist/sandbox/exec-stream.d.ts +1 -2
- package/dist/sandbox/network-policy.d.ts +23 -5
- package/dist/sandbox.d.ts +4 -2
- package/dist/step-invocation/protocol.d.ts +3 -4
- package/dist/step-invocation/server.d.ts +2 -2
- package/dist/step-invocation/types.d.ts +1 -1
- package/dist/types/api-conversations.d.ts +442 -29
- package/dist/types/api-factory.d.ts +99 -10
- package/dist/types/api-projects.d.ts +521 -0
- package/dist/types/api-runs.d.ts +83 -0
- package/dist/types/api-scopes.d.ts +32 -3
- package/dist/types/conversation-stream.d.ts +5 -0
- package/dist/types/execution-context.d.ts +1 -1
- package/dist/types/protocol.d.ts +86 -2
- package/dist/types/runtime.d.ts +9 -2
- package/dist/types/workflow-metadata.d.ts +2 -4
- package/dist/types/workflow-plan.d.ts +1 -3
- package/dist/utils/bundler.d.ts +23 -0
- package/dist/workflow-steps/observability.d.ts +2 -3
- package/dist/workflow-steps/runner.d.ts +5 -8
- package/dist/workflow-steps/types.d.ts +8 -10
- package/dist/workflow-steps/workflow.d.ts +2 -1
- package/dist/workflows/engine.d.ts +3 -5
- package/dist/workflows/invoke-child.d.ts +2 -2
- package/package.json +2 -2
- package/src/agent/agent-context.ts +168 -125
- package/src/agent/agent-loop.ts +7 -6
- package/src/agent/perf-sampler.ts +54 -3
- package/src/agent/run-agent.ts +1 -1
- package/src/client.ts +226 -71
- package/src/directives.ts +3 -3
- package/src/display.ts +12 -0
- package/src/errors.ts +1 -0
- package/src/generated/agentc-commands.ts +571 -0
- package/src/index.ts +57 -21
- package/src/pause/pause-core.ts +2 -1
- package/src/request-context/request-context.ts +1 -1
- package/src/runtimes/_cli-agent.ts +318 -122
- package/src/runtimes/_reported-model.ts +24 -0
- package/src/runtimes/claude-code.ts +195 -12
- package/src/runtimes/claude.ts +9 -2
- package/src/runtimes/codex.ts +188 -19
- package/src/runtimes/opencode.ts +195 -26
- package/src/sandbox/baked-clis.ts +86 -0
- package/src/sandbox/exec-stream.ts +1 -2
- package/src/sandbox/network-policy.ts +51 -7
- package/src/sandbox/providers/e2b.ts +3 -3
- package/src/sandbox/providers/vercel.ts +6 -6
- package/src/sandbox.ts +8 -2
- package/src/step-invocation/invoker.ts +2 -6
- package/src/step-invocation/protocol.ts +3 -4
- package/src/step-invocation/server.ts +2 -2
- package/src/types/api-conversations.ts +366 -23
- package/src/types/api-factory.ts +95 -10
- package/src/types/api-projects.ts +477 -0
- package/src/types/api-runs.ts +73 -0
- package/src/types/api-scopes.ts +32 -3
- package/src/types/conversation-stream.ts +5 -0
- package/src/types/execution-context.ts +1 -1
- package/src/types/protocol.ts +91 -2
- package/src/types/runtime.ts +8 -2
- package/src/types/sandbox-environment.ts +1 -2
- package/src/types/workflow-metadata.ts +2 -4
- package/src/types/workflow-plan.ts +1 -3
- package/src/utils/bundler.ts +88 -19
- package/src/workflow-steps/observability.ts +2 -3
- package/src/workflow-steps/runner.ts +5 -8
- package/src/workflow-steps/types.ts +8 -10
- package/src/workflow-steps/workflow.ts +2 -1
- package/src/workflows/engine.ts +3 -5
- package/src/workflows/invoke-child.ts +2 -2
- package/dist/generated/verb-synopsis.d.ts +0 -34
- package/dist/pause/__tests__/errors.test.d.ts +0 -1
- package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
- package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
- package/src/generated/verb-synopsis.ts +0 -544
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A model id as a harness names it — the one derivation behind every
|
|
3
|
+
* `model_report` the Claude runtimes emit (AgentMessageModelReport): the
|
|
4
|
+
* `model` on claude-code's `system`/`init` event and the `message.model` on
|
|
5
|
+
* each `assistant` event, which the Agent SDK's messages mirror.
|
|
6
|
+
*
|
|
7
|
+
* - claude-code's `[1m]`-style suffix (`claude-opus-4-8[1m]`) is the CLI's
|
|
8
|
+
* context-window marker: a variant flag on the same model, not another
|
|
9
|
+
* model, so the report drops it;
|
|
10
|
+
* - `<synthetic>` is the harness's own voice (slash-command stdout,
|
|
11
|
+
* advisories) and names no model;
|
|
12
|
+
* - anything that is not a non-empty string names none.
|
|
13
|
+
*
|
|
14
|
+
* Machine-format rules only; nothing here reads words.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
const MODEL_ID_MAX = 200;
|
|
18
|
+
|
|
19
|
+
export function reportedModelId(raw: unknown): string | undefined {
|
|
20
|
+
if (typeof raw !== "string") return undefined;
|
|
21
|
+
const id = raw.replace(/\[[^\]]*\]$/, "").trim();
|
|
22
|
+
if (id.length === 0 || id === "<synthetic>") return undefined;
|
|
23
|
+
return id.length > MODEL_ID_MAX ? id.slice(0, MODEL_ID_MAX) : id;
|
|
24
|
+
}
|
|
@@ -35,11 +35,14 @@
|
|
|
35
35
|
*/
|
|
36
36
|
|
|
37
37
|
import type {
|
|
38
|
-
AgentMessage, AgentMessageCompaction,
|
|
38
|
+
AgentMessage, AgentMessageCompaction, AgentMessageModelUsage, AgentMessagePlanLimits,
|
|
39
|
+
AgentMessageTaskNotification, AgentMessageTaskProgress, AgentMessageUsage,
|
|
39
40
|
WorkflowProgressEntry,
|
|
40
41
|
} from "../index.js";
|
|
41
42
|
import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
|
|
43
|
+
import { reportedModelId } from "./_reported-model.js";
|
|
42
44
|
import { formatError } from "../utils/errors.js";
|
|
45
|
+
import { CLAUDE_CODE_VERSION } from "../sandbox/baked-clis.js";
|
|
43
46
|
|
|
44
47
|
function now(): string { return new Date().toISOString(); }
|
|
45
48
|
|
|
@@ -409,6 +412,106 @@ export function parseSystemCompaction(
|
|
|
409
412
|
};
|
|
410
413
|
}
|
|
411
414
|
|
|
415
|
+
// ── Plan limits (the account meter, off the stream) ─────────────────────────
|
|
416
|
+
//
|
|
417
|
+
// A subscription-funded `claude -p` reads the `anthropic-ratelimit-unified-*`
|
|
418
|
+
// headers off every response and re-emits them as `rate_limit_event` lines
|
|
419
|
+
// whenever its reading changes (the public Agent SDK's SDKRateLimitEvent;
|
|
420
|
+
// anthropics/claude-code#50518 shows the plainly-allowed shape:
|
|
421
|
+
// `{status: "allowed", resetsAt: 1729281600, rateLimitType: "five_hour"}`,
|
|
422
|
+
// with `utilization` carried only once a window crosses a warning
|
|
423
|
+
// threshold). `resetsAt` is epoch SECONDS. A `rejected` event is the plan's
|
|
424
|
+
// wall for the request just made, with the authoritative reset. The platform
|
|
425
|
+
// folds these into the funding account's meter (one reading per window) and
|
|
426
|
+
// learns a cooldown from a rejected one — never from the notice text.
|
|
427
|
+
|
|
428
|
+
const PLAN_LIMIT_STATUSES: ReadonlySet<string> = new Set(["allowed", "allowed_warning", "rejected"]);
|
|
429
|
+
|
|
430
|
+
/** Epoch seconds (the CLI's `resetsAt`) as ISO; undefined for anything else. */
|
|
431
|
+
function epochSecondsIso(v: unknown): string | undefined {
|
|
432
|
+
if (typeof v !== "number" || !Number.isFinite(v) || v <= 0) return undefined;
|
|
433
|
+
return new Date(v * 1000).toISOString();
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
/** One `rate_limit_event` mapped onto the structured plan-limits message,
|
|
437
|
+
* or null when the event names no recognisable status. Pure and tolerant
|
|
438
|
+
* over untrusted harness JSON: every field is forwarded only when present
|
|
439
|
+
* and well-typed; nothing is defaulted or invented. Exported for tests. */
|
|
440
|
+
export function parsePlanLimits(
|
|
441
|
+
p: Record<string, unknown>, timestamp: string,
|
|
442
|
+
): AgentMessagePlanLimits | null {
|
|
443
|
+
const info = (typeof p.rate_limit_info === "object" && p.rate_limit_info !== null
|
|
444
|
+
? p.rate_limit_info : null) as Record<string, unknown> | null;
|
|
445
|
+
if (!info || typeof info.status !== "string" || !PLAN_LIMIT_STATUSES.has(info.status)) return null;
|
|
446
|
+
const num = (v: unknown): number | undefined =>
|
|
447
|
+
typeof v === "number" && Number.isFinite(v) ? v : undefined;
|
|
448
|
+
const str = (v: unknown): string | undefined =>
|
|
449
|
+
typeof v === "string" && v.length > 0 ? clip(v, 100) : undefined;
|
|
450
|
+
const window = str(info.rateLimitType);
|
|
451
|
+
const resetsAt = epochSecondsIso(info.resetsAt);
|
|
452
|
+
const utilization = num(info.utilization);
|
|
453
|
+
const surpassedThreshold = num(info.surpassedThreshold);
|
|
454
|
+
const overageStatus = typeof info.overageStatus === "string" && PLAN_LIMIT_STATUSES.has(info.overageStatus)
|
|
455
|
+
? info.overageStatus as AgentMessagePlanLimits["status"] : undefined;
|
|
456
|
+
const overageResetsAt = epochSecondsIso(info.overageResetsAt);
|
|
457
|
+
const overageDisabledReason = str(info.overageDisabledReason);
|
|
458
|
+
const overageInUse = typeof info.isUsingOverage === "boolean" ? info.isUsingOverage
|
|
459
|
+
: typeof info.overageInUse === "boolean" ? info.overageInUse : undefined;
|
|
460
|
+
const overage = overageStatus !== undefined || overageResetsAt !== undefined
|
|
461
|
+
|| overageDisabledReason !== undefined || overageInUse !== undefined
|
|
462
|
+
? {
|
|
463
|
+
...(overageStatus !== undefined ? { status: overageStatus } : {}),
|
|
464
|
+
...(overageResetsAt !== undefined ? { resetsAt: overageResetsAt } : {}),
|
|
465
|
+
...(overageDisabledReason !== undefined ? { disabledReason: overageDisabledReason } : {}),
|
|
466
|
+
...(overageInUse !== undefined ? { inUse: overageInUse } : {}),
|
|
467
|
+
}
|
|
468
|
+
: undefined;
|
|
469
|
+
const limitScope = str(info.limitScope);
|
|
470
|
+
const errorCode = str(info.errorCode);
|
|
471
|
+
return {
|
|
472
|
+
type: "plan_limits",
|
|
473
|
+
status: info.status as AgentMessagePlanLimits["status"],
|
|
474
|
+
...(window !== undefined ? { window } : {}),
|
|
475
|
+
...(resetsAt !== undefined ? { resetsAt } : {}),
|
|
476
|
+
...(utilization !== undefined ? { utilization } : {}),
|
|
477
|
+
...(surpassedThreshold !== undefined ? { surpassedThreshold } : {}),
|
|
478
|
+
...(overage !== undefined ? { overage } : {}),
|
|
479
|
+
...(limitScope !== undefined ? { limitScope } : {}),
|
|
480
|
+
...(errorCode !== undefined ? { errorCode } : {}),
|
|
481
|
+
timestamp,
|
|
482
|
+
};
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
/** The terminal result's per-model report (`result.modelUsage`, keyed by
|
|
486
|
+
* the raw model string) mapped onto the protocol's per-model shape, or
|
|
487
|
+
* undefined when the result carries none. Counts the harness did not
|
|
488
|
+
* report stay absent (the four classes default to 0 only when the entry
|
|
489
|
+
* exists at all — an entry IS a report of that model). Pure; exported for
|
|
490
|
+
* tests. */
|
|
491
|
+
export function parseModelUsage(raw: unknown): Record<string, AgentMessageModelUsage> | undefined {
|
|
492
|
+
if (typeof raw !== "object" || raw === null) return undefined;
|
|
493
|
+
const num = (v: unknown): number => (typeof v === "number" && Number.isFinite(v) ? v : 0);
|
|
494
|
+
const opt = (v: unknown): number | undefined => (typeof v === "number" && Number.isFinite(v) ? v : undefined);
|
|
495
|
+
const out: Record<string, AgentMessageModelUsage> = {};
|
|
496
|
+
for (const [model, entry] of Object.entries(raw as Record<string, unknown>)) {
|
|
497
|
+
if (typeof entry !== "object" || entry === null || model.length === 0) continue;
|
|
498
|
+
const e = entry as Record<string, unknown>;
|
|
499
|
+
const thinkingTokens = opt(e.thinkingTokens);
|
|
500
|
+
const webSearchRequests = opt(e.webSearchRequests);
|
|
501
|
+
const costUsd = opt(e.costUSD);
|
|
502
|
+
out[clip(model, 200)] = {
|
|
503
|
+
inputTokens: num(e.inputTokens),
|
|
504
|
+
outputTokens: num(e.outputTokens),
|
|
505
|
+
cacheReadTokens: num(e.cacheReadInputTokens),
|
|
506
|
+
cacheCreationTokens: num(e.cacheCreationInputTokens),
|
|
507
|
+
...(thinkingTokens !== undefined ? { thinkingTokens } : {}),
|
|
508
|
+
...(webSearchRequests !== undefined ? { webSearchRequests } : {}),
|
|
509
|
+
...(costUsd !== undefined ? { costUsd } : {}),
|
|
510
|
+
};
|
|
511
|
+
}
|
|
512
|
+
return Object.keys(out).length > 0 ? out : undefined;
|
|
513
|
+
}
|
|
514
|
+
|
|
412
515
|
/** Claude Code's real reasoning knob is its own `--effort <level>` flag
|
|
413
516
|
* (low|medium|high|xhigh|max — verified against `claude -p --help`). The
|
|
414
517
|
* CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
|
|
@@ -419,6 +522,50 @@ export function parseSystemCompaction(
|
|
|
419
522
|
export const CLAUDE_CODE_EFFORT_LEVELS: readonly CliReasoningEffort[] =
|
|
420
523
|
["low", "medium", "high", "xhigh", "max"];
|
|
421
524
|
|
|
525
|
+
/** The Bash PreToolUse hook rtk installs for Claude Code (`rtk init -g
|
|
526
|
+
* --hook-only` writes exactly this entry — matcher `Bash`, command
|
|
527
|
+
* `rtk hook claude` — into ~/.claude/settings.json; the image bakes
|
|
528
|
+
* RTK_VERSION, sandbox/baked-clis.ts, which this payload was run against).
|
|
529
|
+
* `rtk hook claude` reads the PreToolUse JSON on stdin and, when it has a
|
|
530
|
+
* filter for the command, answers with `updatedInput` rewriting `git
|
|
531
|
+
* status` to `rtk git status` (ls, find, grep, test runners, bun, curl,
|
|
532
|
+
* docker, …), so the model reads rtk's compact output. A command it
|
|
533
|
+
* cannot compress — an unknown tool, a pipe into one, command or process
|
|
534
|
+
* substitution, a heredoc, a file redirect, anything already prefixed
|
|
535
|
+
* `rtk`, or a `RTK_DISABLED=1` prefix — gets no output and exit 0, which
|
|
536
|
+
* Claude Code treats as "no decision": the original command runs
|
|
537
|
+
* unchanged (all verified against the binary).
|
|
538
|
+
*
|
|
539
|
+
* The wrapper fails OPEN both ways. `command -v` covers images without
|
|
540
|
+
* rtk (the devbox, a bare Vercel VM): there a bare `rtk hook claude` would
|
|
541
|
+
* fail every Bash call's hook with exit 127 — non-blocking, but a stderr
|
|
542
|
+
* warning per call; rtk's own legacy shell hook degraded the same way
|
|
543
|
+
* ("binary not found: exit 0"). The trailing `exit 0` covers an rtk that
|
|
544
|
+
* cannot answer: Claude Code treats a PreToolUse hook's exit 2 as a BLOCK
|
|
545
|
+
* of the tool call with stderr fed to the model, and clap exits 2 with
|
|
546
|
+
* its usage for a subcommand it does not know — which is how the codex
|
|
547
|
+
* hook (RTK_CODEX_HOOK_COMMAND, codex.ts) blocked every codex shell
|
|
548
|
+
* command on 2026-10-03, when the image's rtk was a cache-served 0.45.0.
|
|
549
|
+
* rtk's hooks never exit non-zero on purpose, so a non-zero exit is a
|
|
550
|
+
* broken rtk, and the compressor must never cost the worker its shell:
|
|
551
|
+
* the exit code is dropped, and the smoke gate's rtk-hook-rewrite check
|
|
552
|
+
* is what proves the rewrite itself. */
|
|
553
|
+
export const RTK_BASH_HOOK_COMMAND =
|
|
554
|
+
"command -v rtk >/dev/null 2>&1 && rtk hook claude; exit 0";
|
|
555
|
+
|
|
556
|
+
/** Settings every platform `claude` launch passes as `--settings` (inline
|
|
557
|
+
* JSON). Claude Code MERGES hook entries across its settings sources
|
|
558
|
+
* instead of replacing them (user → project → local → flag → managed), so
|
|
559
|
+
* this rides beside whatever the machine's own ~/.claude/settings.json
|
|
560
|
+
* carries, and nothing is written to that file — the right outcome, since
|
|
561
|
+
* the server never owns it: home carry tar-restores it whole across
|
|
562
|
+
* machines and the image capture strips it as personal state. */
|
|
563
|
+
export const CLAUDE_CODE_PLATFORM_SETTINGS = {
|
|
564
|
+
hooks: {
|
|
565
|
+
PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_BASH_HOOK_COMMAND }] }],
|
|
566
|
+
},
|
|
567
|
+
} as const;
|
|
568
|
+
|
|
422
569
|
export const claudeCodeSpec: CliAgentSpec = {
|
|
423
570
|
kind: "claude-code",
|
|
424
571
|
authEnv: "ANTHROPIC_API_KEY",
|
|
@@ -428,11 +575,14 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
428
575
|
acp: { command: "npx", args: ["--yes", CLAUDE_CODE_ACP_ADAPTER] },
|
|
429
576
|
// Bare-sandbox fallback only: the agent-env image bakes /usr/local/bin/claude
|
|
430
577
|
// (same recipe — infra/e2b-template/build.ts), so the probe short-circuits
|
|
431
|
-
// on the platform path.
|
|
432
|
-
//
|
|
433
|
-
// the
|
|
578
|
+
// on the platform path. The SAME pinned version as the image
|
|
579
|
+
// (CLAUDE_CODE_VERSION), never latest, so a machine the image predates runs
|
|
580
|
+
// what the image's machines run. Anthropic's installer drops a versioned
|
|
581
|
+
// binary under $HOME with a ~/.local/bin/claude launcher; resolve the
|
|
582
|
+
// symlink and copy the self-contained binary to a world-executable system
|
|
583
|
+
// path.
|
|
434
584
|
install:
|
|
435
|
-
|
|
585
|
+
`curl -fsSL https://claude.ai/install.sh | bash -s ${CLAUDE_CODE_VERSION} && ` +
|
|
436
586
|
'REAL=$(readlink -f "$HOME/.local/bin/claude") && test -f "$REAL" && ' +
|
|
437
587
|
'sudo cp "$REAL" /usr/local/bin/claude && sudo chmod 0755 /usr/local/bin/claude',
|
|
438
588
|
// Claude reads the prompt from stdin in -p mode.
|
|
@@ -447,7 +597,8 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
447
597
|
// the same build). A message that lands after the result would start a
|
|
448
598
|
// NEW turn in-process, which is why the transport's feeder stops at the
|
|
449
599
|
// result line instead of forwarding past it.
|
|
450
|
-
|
|
600
|
+
midTurnInput: {
|
|
601
|
+
transport: "stdin-stream",
|
|
451
602
|
promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
|
|
452
603
|
messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
|
|
453
604
|
// The ESC equivalent (verified live against claude 2.1.236): the CLI's
|
|
@@ -479,6 +630,10 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
479
630
|
// claude's own attestation env for exactly this: it lifts the flag's
|
|
480
631
|
// root-user refusal (the E2B agent-env user is root).
|
|
481
632
|
"--dangerously-skip-permissions",
|
|
633
|
+
// rtk's Bash PreToolUse hook (RTK_BASH_HOOK_COMMAND): shell output is
|
|
634
|
+
// compressed before it reaches the model. Flag settings merge with the
|
|
635
|
+
// machine's own settings; they never replace them.
|
|
636
|
+
`--settings ${shellQuote(JSON.stringify(CLAUDE_CODE_PLATFORM_SETTINGS))}`,
|
|
482
637
|
...(model ? [`--model ${shellQuote(model)}`] : []),
|
|
483
638
|
// Reasoning effort is the CLI's own flag; the value comes from the
|
|
484
639
|
// closed CliReasoningEffort set, so it is shell-safe unquoted.
|
|
@@ -514,7 +669,13 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
514
669
|
// Structural marker only — no content sniffing.
|
|
515
670
|
const synthetic = message?.model === "<synthetic>";
|
|
516
671
|
const blocks = Array.isArray(message?.content) ? message.content : [];
|
|
517
|
-
|
|
672
|
+
// The model the API answered with, named on every assistant event
|
|
673
|
+
// (AgentMessageModelReport): the truth after a /model switch or a
|
|
674
|
+
// fallback. Top level only — a subagent's events name the
|
|
675
|
+
// subagent's model, not the worker's; `<synthetic>` names none.
|
|
676
|
+
const model = parentId ? undefined : reportedModelId(message?.model);
|
|
677
|
+
const report: AgentMessage[] = model ? [{ type: "model_report", model, timestamp: ts }] : [];
|
|
678
|
+
return report.concat(blocks.flatMap((b): AgentMessage[] => {
|
|
518
679
|
if (b.type === "text" && typeof b.text === "string" && b.text.length > 0) {
|
|
519
680
|
return [{ type: synthetic ? "harness_notice" : "text", text: b.text, timestamp: ts }];
|
|
520
681
|
}
|
|
@@ -528,7 +689,7 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
528
689
|
}];
|
|
529
690
|
}
|
|
530
691
|
return [];
|
|
531
|
-
});
|
|
692
|
+
}));
|
|
532
693
|
}
|
|
533
694
|
// User API message: the CLI echoes tool results back as user content,
|
|
534
695
|
// and injects `<task-notification>` blocks (background-task completion
|
|
@@ -614,14 +775,21 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
614
775
|
return [];
|
|
615
776
|
}
|
|
616
777
|
// Terminal result: usage on success (the text already streamed via the
|
|
617
|
-
// assistant events); an honest string error on failure.
|
|
778
|
+
// assistant events); an honest string error on failure. The turn
|
|
779
|
+
// totals are the main loop's `usage`; `modelUsage` (every model the
|
|
780
|
+
// query pipeline called, with the CLI's own cost estimate) and
|
|
781
|
+
// `total_cost_usd` ride beside them as the harness reported them —
|
|
782
|
+
// cumulative for the guest session (AgentMessageUsage.byModel).
|
|
618
783
|
case "result": {
|
|
619
784
|
if (p.is_error === true) {
|
|
620
785
|
return [{ type: "error", text: formatError(p.result ?? p.subtype ?? p), timestamp: ts }];
|
|
621
786
|
}
|
|
622
787
|
const u = p.usage as Record<string, number> | undefined;
|
|
623
788
|
if (!u) return [];
|
|
624
|
-
|
|
789
|
+
const byModel = parseModelUsage(p.modelUsage);
|
|
790
|
+
const costUsd = typeof p.total_cost_usd === "number" && Number.isFinite(p.total_cost_usd)
|
|
791
|
+
? p.total_cost_usd : undefined;
|
|
792
|
+
const usage: AgentMessageUsage = {
|
|
625
793
|
type: "usage",
|
|
626
794
|
inputTokens: u.input_tokens ?? 0,
|
|
627
795
|
outputTokens: u.output_tokens ?? 0,
|
|
@@ -629,10 +797,21 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
629
797
|
cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
|
|
630
798
|
durationMs: typeof p.duration_ms === "number" ? p.duration_ms : 0,
|
|
631
799
|
numTurns: typeof p.num_turns === "number" ? p.num_turns : 1,
|
|
800
|
+
...(byModel !== undefined ? { byModel } : {}),
|
|
801
|
+
...(costUsd !== undefined ? { costUsd } : {}),
|
|
632
802
|
timestamp: ts,
|
|
633
|
-
}
|
|
803
|
+
};
|
|
804
|
+
return [usage];
|
|
634
805
|
}
|
|
635
|
-
//
|
|
806
|
+
// The account's plan limits as the CLI read them off the response
|
|
807
|
+
// headers (subscription-funded sessions; parsePlanLimits above).
|
|
808
|
+
case "rate_limit_event": {
|
|
809
|
+
const limits = parsePlanLimits(p, ts);
|
|
810
|
+
return limits ? [limits] : [];
|
|
811
|
+
}
|
|
812
|
+
// System events: init carries the session id (extractSessionId) and
|
|
813
|
+
// the session's resolved model (the first model report of the turn;
|
|
814
|
+
// the assistant events confirm or change it), and
|
|
636
815
|
// the background-task lane rides here too — `task_notification` is
|
|
637
816
|
// the completion evidence stream-json actually emits (see the section
|
|
638
817
|
// header above), and `task_progress` the LIVE background-task feed
|
|
@@ -644,6 +823,10 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
644
823
|
// under system (task_started, task_updated, background_tasks_changed,
|
|
645
824
|
// thinking_tokens) is lifecycle noise here.
|
|
646
825
|
case "system": {
|
|
826
|
+
if (p.subtype === "init") {
|
|
827
|
+
const model = reportedModelId(p.model);
|
|
828
|
+
return model ? [{ type: "model_report", model, timestamp: ts }] : [];
|
|
829
|
+
}
|
|
647
830
|
if (p.subtype === "task_progress") {
|
|
648
831
|
const progress = parseSystemTaskProgress(p, ts);
|
|
649
832
|
return progress ? [progress] : [];
|
package/src/runtimes/claude.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
/** Claude Agent SDK runtime
|
|
1
|
+
/** Claude Agent SDK runtime. */
|
|
2
2
|
|
|
3
3
|
import { query, type HookCallback, type PreToolUseHookInput, type SDKUserMessage, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
|
|
4
4
|
import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
|
|
@@ -9,6 +9,7 @@ import { runProcessorChain } from "../processors/runner.js";
|
|
|
9
9
|
import type { ProcessorContext, ToolCall } from "../processors/processor.js";
|
|
10
10
|
import { RequestContext } from "../request-context/request-context.js";
|
|
11
11
|
import { boundProcessorPause } from "../pause/pause-core.js";
|
|
12
|
+
import { reportedModelId } from "./_reported-model.js";
|
|
12
13
|
import { formatError } from "../utils/errors.js";
|
|
13
14
|
|
|
14
15
|
function now(): string { return new Date().toISOString(); }
|
|
@@ -19,11 +20,17 @@ function translateMessage(message: Record<string, unknown>): AgentMessage[] {
|
|
|
19
20
|
|
|
20
21
|
if (message.type === "system" && message.subtype === "init") {
|
|
21
22
|
msgs.push({ type: "init", sessionId: String(message.session_id ?? ""), timestamp: ts });
|
|
23
|
+
// The session's resolved model, as the harness names it
|
|
24
|
+
// (AgentMessageModelReport); the assistant messages confirm or change it.
|
|
25
|
+
const model = reportedModelId(message.model);
|
|
26
|
+
if (model) msgs.push({ type: "model_report", model, timestamp: ts });
|
|
22
27
|
return msgs;
|
|
23
28
|
}
|
|
24
29
|
|
|
25
30
|
if (message.type === "assistant") {
|
|
26
|
-
const raw = message.message as { content?: unknown[] } | undefined;
|
|
31
|
+
const raw = message.message as { content?: unknown[]; model?: unknown } | undefined;
|
|
32
|
+
const model = typeof message.parent_tool_use_id === "string" ? undefined : reportedModelId(raw?.model);
|
|
33
|
+
if (model) msgs.push({ type: "model_report", model, timestamp: ts });
|
|
27
34
|
for (const block of raw?.content ?? []) {
|
|
28
35
|
const b = block as Record<string, unknown>;
|
|
29
36
|
if (b.type === "text") msgs.push({ type: "text", text: String(b.text ?? ""), timestamp: ts });
|
package/src/runtimes/codex.ts
CHANGED
|
@@ -5,17 +5,28 @@
|
|
|
5
5
|
* only stream-parse what it prints.
|
|
6
6
|
*
|
|
7
7
|
* Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
|
|
8
|
-
* workflow secret. The
|
|
9
|
-
*
|
|
10
|
-
*
|
|
8
|
+
* workflow secret. The E2B session image bakes the pinned `codex` CLI
|
|
9
|
+
* (`@openai/codex@CODEX_CLI_VERSION`); on any other machine the runtime
|
|
10
|
+
* installs that same version on demand (pair with
|
|
11
|
+
* `snapshots: { bootFrom: "reuse" }` to install once and boot from the
|
|
12
|
+
* captured snapshot on every run after).
|
|
11
13
|
*
|
|
12
14
|
* Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
|
|
13
|
-
* with the command_execution / reasoning / agent_message item shapes below
|
|
15
|
+
* with the command_execution / reasoning / agent_message item shapes below;
|
|
16
|
+
* the plan's `todo_list` item against codex-rs rust-v0.153.4 and
|
|
17
|
+
* rust-v0.159.2 (see codexPlanMessages); the command_execution /
|
|
18
|
+
* agent_message / turn.completed shapes re-verified live against 0.160.0
|
|
19
|
+
* (the pin), whose plan-tool sources are byte-identical to rust-v0.159.2.
|
|
14
20
|
*/
|
|
15
21
|
|
|
16
22
|
import type { AgentMessage } from "../index.js";
|
|
17
|
-
import
|
|
23
|
+
import type { AgentMessagePlan } from "../types/protocol.js";
|
|
24
|
+
import {
|
|
25
|
+
MID_TURN_DELIVERED_ENV, MID_TURN_INBOX_ENV, createCliAgentRuntime, shellQuote,
|
|
26
|
+
type CliAgentSpec, type CliReasoningEffort,
|
|
27
|
+
} from "./_cli-agent.js";
|
|
18
28
|
import { formatError } from "../utils/errors.js";
|
|
29
|
+
import { CODEX_CLI_VERSION } from "../sandbox/baked-clis.js";
|
|
19
30
|
|
|
20
31
|
function now(): string { return new Date().toISOString(); }
|
|
21
32
|
|
|
@@ -24,6 +35,97 @@ function now(): string { return new Date().toISOString(); }
|
|
|
24
35
|
* overrides the bundled `@openai/codex` the adapter drives underneath. */
|
|
25
36
|
const CODEX_ACP_ADAPTER = "@agentclientprotocol/codex-acp@0.1.0";
|
|
26
37
|
|
|
38
|
+
/** The Bash PreToolUse hook rtk installs for Codex (`rtk init -g --codex`
|
|
39
|
+
* writes exactly this entry — matcher `Bash`, command `rtk hook codex` —
|
|
40
|
+
* into $CODEX_HOME/hooks.json). `rtk hook codex` exists since rtk 0.50.0;
|
|
41
|
+
* the image bakes RTK_VERSION (sandbox/baked-clis.ts), which this payload
|
|
42
|
+
* was run against. Codex's shell tool matches the `Bash` matcher (codex-rs
|
|
43
|
+
* rust-v0.160.0, the pinned CODEX_CLI_VERSION; identical at
|
|
44
|
+
* rust-v0.159.2), and the hook runs under `$SHELL -lc` with the PreToolUse
|
|
45
|
+
* JSON on stdin. `rtk hook codex` answers `permissionDecision: "allow"` +
|
|
46
|
+
* `updatedInput` rewriting `git status` to `rtk git status` when it has a
|
|
47
|
+
* filter for the command; Codex applies the replacement before its own
|
|
48
|
+
* approval and sandbox checks. For a command it cannot compress (an
|
|
49
|
+
* unknown tool, a pipe into one, substitutions, heredocs, redirects, a
|
|
50
|
+
* `RTK_DISABLED=1` prefix, an unknown permission mode) it prints nothing
|
|
51
|
+
* and exits 0, and Codex runs the original unchanged.
|
|
52
|
+
*
|
|
53
|
+
* The wrapper fails OPEN both ways, like RTK_BASH_HOOK_COMMAND
|
|
54
|
+
* (claude-code.ts). `command -v` covers an image without rtk (the devbox,
|
|
55
|
+
* a bare Vercel VM): silent, exit 0, so Codex runs the command instead of
|
|
56
|
+
* failing every shell call's hook with 127. The trailing `exit 0` covers
|
|
57
|
+
* an rtk that cannot answer: Codex treats a PreToolUse hook's exit 2 with
|
|
58
|
+
* stderr as a BLOCK of the tool call (codex-rs hooks/src/events/
|
|
59
|
+
* pre_tool_use.rs at rust-v0.160.0 — the model reads "Command blocked by
|
|
60
|
+
* PreToolUse hook: <stderr>"), and clap exits 2 with its usage on stderr
|
|
61
|
+
* for a subcommand it does not know. That was 2026-10-03: the image's rtk
|
|
62
|
+
* was a cache-served 0.45.0 with no `hook codex`, and this hook — `exec`
|
|
63
|
+
* handing rtk's exit code to Codex — blocked every shell command of every
|
|
64
|
+
* codex session. rtk's hooks never exit non-zero on purpose (they fail
|
|
65
|
+
* open with no stdout), so a non-zero exit is always a broken rtk, and the
|
|
66
|
+
* compressor must never cost the worker its shell: the exit code is
|
|
67
|
+
* dropped, and the smoke gate's codex-rtk-hook-rewrite check is what
|
|
68
|
+
* proves the rewrite itself. */
|
|
69
|
+
export const RTK_CODEX_HOOK_COMMAND =
|
|
70
|
+
"command -v rtk >/dev/null 2>&1 && rtk hook codex; exit 0";
|
|
71
|
+
|
|
72
|
+
/** The awk program of the mid-turn hook below: the inbox lines not yet fed
|
|
73
|
+
* (JSON string literals, one per line — `codexSpec.midTurnInput.messageLine`)
|
|
74
|
+
* become ONE PostToolUse hook output. Each literal's quotes are stripped and
|
|
75
|
+
* the bodies are joined with an escaped blank line; the bodies are already
|
|
76
|
+
* JSON-escaped, so the result is one valid JSON string. Nothing is printed
|
|
77
|
+
* when no line is due, and codex ignores an empty stdout. */
|
|
78
|
+
const CODEX_MID_TURN_HOOK_AWK = String.raw`NR > fed && NR <= upto { s = substr($0, 2, length($0) - 2); out = (out == "" ? s : out "\\n\\n" s) } END { if (out != "") printf "{\"hookSpecificOutput\":{\"hookEventName\":\"PostToolUse\",\"additionalContext\":\"%s\"}}\n", out }`;
|
|
79
|
+
|
|
80
|
+
/** The PostToolUse command hook that carries a mid-turn message into a
|
|
81
|
+
* RUNNING codex turn (the 2026-10-02 relayed-steer incident: three of the
|
|
82
|
+
* owner's instructions waited 48-51 minutes for a codex build to end).
|
|
83
|
+
* Codex runs it, under `$SHELL -lc` with the event JSON on stdin and the
|
|
84
|
+
* codex process's own environment, after every tool call it completes
|
|
85
|
+
* (codex-rs core/src/tools/registry.rs → hook_runtime.rs at rust-v0.160.0,
|
|
86
|
+
* the pin). The launch wrapper exported the turn's inbox and delivered-
|
|
87
|
+
* counter paths into that environment (sdk _cli-agent.ts
|
|
88
|
+
* toolHookInboxFragment); the hook forwards the inbox lines past the
|
|
89
|
+
* counter as the hook's `additionalContext`, which codex records as
|
|
90
|
+
* developer context in the live turn's history before the model's next
|
|
91
|
+
* request (hook_runtime.rs record_additional_contexts), then advances the
|
|
92
|
+
* counter — the same ack the server's inject lane trusts for the stdin
|
|
93
|
+
* lane. Not a platform turn (no inbox in the environment): silent, exit 0.
|
|
94
|
+
* Codex's default spill threshold for a hook's additional context is 2,500
|
|
95
|
+
* tokens (hooks/src/output_spill.rs); a longer message reaches the model
|
|
96
|
+
* as a preview plus a file pointer, which a steer never is. */
|
|
97
|
+
export const CODEX_MID_TURN_HOOK_COMMAND =
|
|
98
|
+
// Drain the event JSON first, so codex's stdin write never meets a closed pipe.
|
|
99
|
+
"cat >/dev/null; "
|
|
100
|
+
+ `[ -n "\${${MID_TURN_INBOX_ENV}:-}" ] && [ -f "$${MID_TURN_INBOX_ENV}" ] || exit 0; `
|
|
101
|
+
+ `ac_fed=$(cat "$${MID_TURN_DELIVERED_ENV}" 2>/dev/null); ac_fed=\${ac_fed:-0}; `
|
|
102
|
+
+ `ac_lines=$(wc -l < "$${MID_TURN_INBOX_ENV}" 2>/dev/null); ac_lines=\${ac_lines:-0}; `
|
|
103
|
+
+ `[ "$ac_lines" -gt "$ac_fed" ] || exit 0; `
|
|
104
|
+
+ `awk -v fed="$ac_fed" -v upto="$ac_lines" '${CODEX_MID_TURN_HOOK_AWK}' "$${MID_TURN_INBOX_ENV}" `
|
|
105
|
+
+ `&& echo "$ac_lines" > "$${MID_TURN_DELIVERED_ENV}"; exit 0`;
|
|
106
|
+
|
|
107
|
+
/** $CODEX_HOME/hooks.json for every platform Codex session (the server
|
|
108
|
+
* writes it at boot, session-runtime-config.ts). Codex only RUNS a
|
|
109
|
+
* user-level hook it has persisted trust for — the TUI's review prompt
|
|
110
|
+
* has no headless counterpart, and `--dangerously-bypass-hook-trust` would
|
|
111
|
+
* run a cloned repository's `.codex/hooks.json` unreviewed too — so the
|
|
112
|
+
* server also writes the trust record for each of these exact hooks into
|
|
113
|
+
* the config.toml it generates (sandbox/codex-hooks.ts). Verified live
|
|
114
|
+
* against codex 0.159.2 and 0.160.0 for the rtk hook: with the record,
|
|
115
|
+
* `codex exec` rewrote `git status` through rtk with no bypass flag;
|
|
116
|
+
* without it, the hook was skipped.
|
|
117
|
+
*
|
|
118
|
+
* The PostToolUse group carries NO matcher: codex runs a matcher-less hook
|
|
119
|
+
* after every tool it completes (hooks/src/events/common.rs
|
|
120
|
+
* matches_matcher: an absent matcher is a match), so a mid-turn message
|
|
121
|
+
* lands at the next tool step whatever the tool was. */
|
|
122
|
+
export const CODEX_PLATFORM_HOOKS = {
|
|
123
|
+
hooks: {
|
|
124
|
+
PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_CODEX_HOOK_COMMAND }] }],
|
|
125
|
+
PostToolUse: [{ hooks: [{ type: "command", command: CODEX_MID_TURN_HOOK_COMMAND }] }],
|
|
126
|
+
},
|
|
127
|
+
} as const;
|
|
128
|
+
|
|
27
129
|
/** Known-noise codex ADVISORY lines. codex emits these as `item.completed`
|
|
28
130
|
* error items on the `--json` stream (exec maps every `Warning` notification
|
|
29
131
|
* to an error item — verified against codex-cli 0.147.0), so without a filter
|
|
@@ -101,6 +203,41 @@ export function codexWriterLockPreflight(
|
|
|
101
203
|
+ `flock -n "$ac_lk" rm -f -- "$ac_lk" 2>/dev/null || true; fi; `;
|
|
102
204
|
}
|
|
103
205
|
|
|
206
|
+
/**
|
|
207
|
+
* codex's PLAN (its update_plan tool) on the `--json` stream: one
|
|
208
|
+
* `todo_list` item per turn, `item.started` on the plan's first update,
|
|
209
|
+
* `item.updated` (same id, the whole list) on every later one, and
|
|
210
|
+
* `item.completed` at the turn's end. That is codex-rs exec's
|
|
211
|
+
* event_processor_with_jsonl_output.rs, identical in rust-v0.153.4 and
|
|
212
|
+
* rust-v0.159.2, where it is the only `item.updated` codex emits; the item
|
|
213
|
+
* `{ id, type: "todo_list", items: [{ text, completed }] }` is the shape every
|
|
214
|
+
* todo_list item our machines have persisted carries. Each event maps to ONE
|
|
215
|
+
* whole-plan `plan` message, so the transcript's checklist, and the worker's
|
|
216
|
+
* status line, follow the plan while the turn runs (before this, the list
|
|
217
|
+
* surfaced once, at the turn's end, as a raw `todo_list` tool card).
|
|
218
|
+
*
|
|
219
|
+
* A step carries only `completed`: codex's `pending` and `in_progress` both
|
|
220
|
+
* serialize as false. Its plan tool allows at most one step in progress and
|
|
221
|
+
* a plan runs in order, so the first step not completed is the one under
|
|
222
|
+
* way; the rest are pending. The stream names no priority, so none is set.
|
|
223
|
+
*/
|
|
224
|
+
function codexPlanMessages(item: Record<string, unknown>, timestamp: string): AgentMessage[] {
|
|
225
|
+
const steps = Array.isArray(item.items) ? item.items : [];
|
|
226
|
+
let underWay = false;
|
|
227
|
+
const entries: AgentMessagePlan["entries"] = [];
|
|
228
|
+
for (const step of steps) {
|
|
229
|
+
if (typeof step !== "object" || step === null) continue;
|
|
230
|
+
const text = (step as { text?: unknown }).text;
|
|
231
|
+
if (typeof text !== "string" || text.trim().length === 0) continue;
|
|
232
|
+
let status: AgentMessagePlan["entries"][number]["status"];
|
|
233
|
+
if ((step as { completed?: unknown }).completed === true) status = "completed";
|
|
234
|
+
else if (!underWay) { status = "in_progress"; underWay = true; }
|
|
235
|
+
else status = "pending";
|
|
236
|
+
entries.push({ content: text.trim(), status });
|
|
237
|
+
}
|
|
238
|
+
return entries.length > 0 ? [{ type: "plan", entries, timestamp }] : [];
|
|
239
|
+
}
|
|
240
|
+
|
|
104
241
|
/** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
|
|
105
242
|
* `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
|
|
106
243
|
* of the public runtime surface — `createCodexRuntime` stays the entry point. */
|
|
@@ -113,6 +250,22 @@ export const codexSpec: CliAgentSpec = {
|
|
|
113
250
|
// every resume of the same thread, so supersede teardown must KILL it —
|
|
114
251
|
// never detach it alive (server runner-kill.ts consumes this flag).
|
|
115
252
|
exclusiveSessionWriter: true,
|
|
253
|
+
// Mid-turn input (the 2026-10-02 relayed-steer incident). codex reads
|
|
254
|
+
// stdin ONCE, as the prompt: `codex exec -` submits one UserTurn and never
|
|
255
|
+
// reads stdin again (codex-rs exec/src/lib.rs read_prompt_from_stdin and
|
|
256
|
+
// its single InitialOperation::UserTurn at rust-v0.160.0; the app-server's
|
|
257
|
+
// `turn/steer` is a different transport, not `codex exec`). So a running
|
|
258
|
+
// turn takes a message through the PostToolUse hook above: codex runs it
|
|
259
|
+
// after every tool call it completes, with the codex process's own
|
|
260
|
+
// environment; the hook prints the unfed inbox lines as `additionalContext`
|
|
261
|
+
// and codex records them as developer context in the live turn's history
|
|
262
|
+
// before the model's next request. One JSON string literal per inbox line,
|
|
263
|
+
// so the hook splices bodies without decoding. Codex builds the PostToolUse
|
|
264
|
+
// payload only for a tool call it counts as successful (tools/registry.rs),
|
|
265
|
+
// so a message lands at the next successful tool step. Source-verified at
|
|
266
|
+
// rust-v0.160.0 (the pinned CODEX_CLI_VERSION); the end-to-end run against
|
|
267
|
+
// a live codex is the acceptance check still owed.
|
|
268
|
+
midTurnInput: { transport: "tool-hook", messageLine: (text) => JSON.stringify(text) },
|
|
116
269
|
// ACP-mode invocation (ADR-0020 increment 1). `codex` has no native `--acp`
|
|
117
270
|
// flag; the adapter (a Rust binary shipped via npm) speaks ACP and drives a
|
|
118
271
|
// compatible bundled `@openai/codex`. The runner attempts this first and
|
|
@@ -125,9 +278,11 @@ export const codexSpec: CliAgentSpec = {
|
|
|
125
278
|
// (e.g. a sandbox-baked one) instead of the adapter's bundled copy.
|
|
126
279
|
...(process.env.CODEX_PATH ? { env: { CODEX_PATH: process.env.CODEX_PATH } } : {}),
|
|
127
280
|
},
|
|
128
|
-
//
|
|
129
|
-
//
|
|
130
|
-
|
|
281
|
+
// The E2B session image bakes this exact version (sandbox/baked-clis.ts),
|
|
282
|
+
// so this runs only on a machine whose image predates the bake: a global
|
|
283
|
+
// npm install of the SAME pin, symlinked onto PATH only if the global bin
|
|
284
|
+
// dir isn't already there (so a non-login `sh -c` can find it).
|
|
285
|
+
install: `sudo npm install -g @openai/codex@${CODEX_CLI_VERSION} && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)`,
|
|
131
286
|
// Codex reads the prompt from stdin when invoked as `codex exec ... -`.
|
|
132
287
|
promptPayload: (prompt) => prompt,
|
|
133
288
|
buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
|
|
@@ -140,13 +295,13 @@ export const codexSpec: CliAgentSpec = {
|
|
|
140
295
|
// for running in environments that are externally sandboxed".
|
|
141
296
|
"--dangerously-bypass-approvals-and-sandbox",
|
|
142
297
|
...(model ? ["-m", shellQuote(model)] : []),
|
|
143
|
-
// codex's own reasoning knob — a config override
|
|
144
|
-
//
|
|
145
|
-
//
|
|
146
|
-
//
|
|
147
|
-
//
|
|
148
|
-
//
|
|
149
|
-
//
|
|
298
|
+
// codex's own reasoning knob — a config override. Every codex model
|
|
299
|
+
// takes low|medium|high|xhigh; only the GPT-6 and GPT-5.6 presets add
|
|
300
|
+
// max (codex 0.153+), so the platform offers codex up to xhigh: the
|
|
301
|
+
// server rejects "max" for codex sessions (sessionEffortLockError), and
|
|
302
|
+
// this clamp to xhigh is the type-level backstop for a caller that
|
|
303
|
+
// bypasses that gate. The value comes from the closed
|
|
304
|
+
// CliReasoningEffort set, so it is shell-safe unquoted.
|
|
150
305
|
...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
|
|
151
306
|
].join(" ");
|
|
152
307
|
// Working directory via a shell `cd` — the idiom EVERY other runtime uses
|
|
@@ -172,11 +327,19 @@ export const codexSpec: CliAgentSpec = {
|
|
|
172
327
|
mapEvent: (p): AgentMessage[] => {
|
|
173
328
|
const ts = now();
|
|
174
329
|
switch (p.type) {
|
|
330
|
+
case "item.updated": {
|
|
331
|
+
// Only the plan updates in place (codexPlanMessages); any other
|
|
332
|
+
// item surfaces on its started / completed events below.
|
|
333
|
+
const item = p.item as Record<string, unknown> | undefined;
|
|
334
|
+
return item?.type === "todo_list" ? codexPlanMessages(item, ts) : [];
|
|
335
|
+
}
|
|
175
336
|
case "item.started":
|
|
176
337
|
case "item.completed": {
|
|
177
338
|
const item = p.item as Record<string, unknown> | undefined;
|
|
178
339
|
if (!item) return [];
|
|
179
340
|
const itype = String(item.type ?? "");
|
|
341
|
+
// The plan, whole, on its first update and at the turn's end.
|
|
342
|
+
if (itype === "todo_list") return codexPlanMessages(item, ts);
|
|
180
343
|
// Text + reasoning land on completion (started carries no final text).
|
|
181
344
|
if (itype === "agent_message") {
|
|
182
345
|
return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
|
|
@@ -209,12 +372,17 @@ export const codexSpec: CliAgentSpec = {
|
|
|
209
372
|
}
|
|
210
373
|
return [{ type: "error", text, timestamp: ts }];
|
|
211
374
|
}
|
|
212
|
-
// file_change / mcp_tool_call / web_search
|
|
375
|
+
// file_change / mcp_tool_call / web_search: surface once, on completion.
|
|
213
376
|
if (p.type === "item.completed") {
|
|
214
377
|
return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
|
|
215
378
|
}
|
|
216
379
|
return [];
|
|
217
380
|
}
|
|
381
|
+
// The turn's token classes as codex reports them (codex-rs
|
|
382
|
+
// exec_events.rs `Usage`): input, cached input, cache-write input,
|
|
383
|
+
// output and reasoning output. Every class is kept — cache writes land
|
|
384
|
+
// in the creation class, reasoning rides beside output (it is counted
|
|
385
|
+
// inside output_tokens, so the four totals stay additive).
|
|
218
386
|
case "turn.completed": {
|
|
219
387
|
const u = p.usage as Record<string, number> | undefined;
|
|
220
388
|
if (!u) return [];
|
|
@@ -223,9 +391,10 @@ export const codexSpec: CliAgentSpec = {
|
|
|
223
391
|
inputTokens: u.input_tokens ?? 0,
|
|
224
392
|
outputTokens: u.output_tokens ?? 0,
|
|
225
393
|
cacheReadTokens: u.cached_input_tokens ?? 0,
|
|
226
|
-
cacheCreationTokens: 0,
|
|
394
|
+
cacheCreationTokens: u.cache_write_input_tokens ?? 0,
|
|
227
395
|
durationMs: 0,
|
|
228
396
|
numTurns: 1,
|
|
397
|
+
...(typeof u.reasoning_output_tokens === "number" ? { reasoningOutputTokens: u.reasoning_output_tokens } : {}),
|
|
229
398
|
timestamp: ts,
|
|
230
399
|
}];
|
|
231
400
|
}
|
|
@@ -238,8 +407,8 @@ export const codexSpec: CliAgentSpec = {
|
|
|
238
407
|
},
|
|
239
408
|
};
|
|
240
409
|
|
|
241
|
-
/** The effort levels
|
|
242
|
-
* low|medium|high|xhigh
|
|
410
|
+
/** The effort levels this runtime passes to codex (`model_reasoning_effort`):
|
|
411
|
+
* low|medium|high|xhigh, the levels every codex model takes. */
|
|
243
412
|
export type CodexReasoningEffort = Exclude<CliReasoningEffort, "max">;
|
|
244
413
|
|
|
245
414
|
export interface CodexRuntimeConfig {
|