@agentex/agent 0.0.33 → 0.0.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/CHANGELOG.md +204 -0
  2. package/README.md +2 -1
  3. package/dist/providers/claude/discovery.d.ts +64 -0
  4. package/dist/providers/claude/discovery.d.ts.map +1 -0
  5. package/dist/providers/claude/discovery.js +147 -0
  6. package/dist/providers/claude/discovery.js.map +1 -0
  7. package/dist/providers/claude/effort.d.ts +20 -0
  8. package/dist/providers/claude/effort.d.ts.map +1 -0
  9. package/dist/providers/claude/effort.js +32 -0
  10. package/dist/providers/claude/effort.js.map +1 -0
  11. package/dist/providers/claude/execute.d.ts.map +1 -1
  12. package/dist/providers/claude/execute.js +2 -1
  13. package/dist/providers/claude/execute.js.map +1 -1
  14. package/dist/providers/claude/index.d.ts.map +1 -1
  15. package/dist/providers/claude/index.js +2 -1
  16. package/dist/providers/claude/index.js.map +1 -1
  17. package/dist/providers/claude/parse.d.ts.map +1 -1
  18. package/dist/providers/claude/parse.js +26 -4
  19. package/dist/providers/claude/parse.js.map +1 -1
  20. package/dist/providers/claude/session.d.ts.map +1 -1
  21. package/dist/providers/claude/session.js +2 -1
  22. package/dist/providers/claude/session.js.map +1 -1
  23. package/dist/providers/codex/discovery.d.ts +20 -0
  24. package/dist/providers/codex/discovery.d.ts.map +1 -0
  25. package/dist/providers/codex/discovery.js +76 -0
  26. package/dist/providers/codex/discovery.js.map +1 -0
  27. package/dist/providers/codex/execute.d.ts.map +1 -1
  28. package/dist/providers/codex/execute.js +45 -3
  29. package/dist/providers/codex/execute.js.map +1 -1
  30. package/dist/providers/codex/index.d.ts.map +1 -1
  31. package/dist/providers/codex/index.js +2 -1
  32. package/dist/providers/codex/index.js.map +1 -1
  33. package/dist/providers/codex/parse.d.ts +14 -6
  34. package/dist/providers/codex/parse.d.ts.map +1 -1
  35. package/dist/providers/codex/parse.js +83 -1
  36. package/dist/providers/codex/parse.js.map +1 -1
  37. package/dist/providers/codex/session.d.ts +50 -1
  38. package/dist/providers/codex/session.d.ts.map +1 -1
  39. package/dist/providers/codex/session.js +611 -46
  40. package/dist/providers/codex/session.js.map +1 -1
  41. package/dist/providers/cursor/discovery.d.ts.map +1 -1
  42. package/dist/providers/cursor/discovery.js +4 -0
  43. package/dist/providers/cursor/discovery.js.map +1 -1
  44. package/dist/providers/opencode/discovery.d.ts.map +1 -1
  45. package/dist/providers/opencode/discovery.js +11 -8
  46. package/dist/providers/opencode/discovery.js.map +1 -1
  47. package/dist/providers/opencode/event-parse.d.ts +23 -0
  48. package/dist/providers/opencode/event-parse.d.ts.map +1 -1
  49. package/dist/providers/opencode/event-parse.js +44 -0
  50. package/dist/providers/opencode/event-parse.js.map +1 -1
  51. package/dist/providers/opencode/history.d.ts.map +1 -1
  52. package/dist/providers/opencode/history.js +25 -4
  53. package/dist/providers/opencode/history.js.map +1 -1
  54. package/dist/providers/opencode/http-session.d.ts.map +1 -1
  55. package/dist/providers/opencode/http-session.js +59 -8
  56. package/dist/providers/opencode/http-session.js.map +1 -1
  57. package/dist/types.d.ts +19 -5
  58. package/dist/types.d.ts.map +1 -1
  59. package/dist/utils/model-cache.d.ts +33 -0
  60. package/dist/utils/model-cache.d.ts.map +1 -0
  61. package/dist/utils/model-cache.js +57 -0
  62. package/dist/utils/model-cache.js.map +1 -0
  63. package/package.json +1 -1
  64. package/src/providers/claude/discovery.ts +215 -0
  65. package/src/providers/claude/effort.ts +34 -0
  66. package/src/providers/claude/execute.ts +2 -1
  67. package/src/providers/claude/index.ts +2 -1
  68. package/src/providers/claude/parse.ts +27 -5
  69. package/src/providers/claude/session.ts +2 -1
  70. package/src/providers/codex/discovery.ts +91 -0
  71. package/src/providers/codex/execute.ts +42 -3
  72. package/src/providers/codex/index.ts +2 -1
  73. package/src/providers/codex/parse.ts +95 -3
  74. package/src/providers/codex/session.ts +691 -48
  75. package/src/providers/cursor/discovery.ts +5 -0
  76. package/src/providers/opencode/discovery.ts +10 -7
  77. package/src/providers/opencode/event-parse.ts +63 -0
  78. package/src/providers/opencode/history.ts +32 -4
  79. package/src/providers/opencode/http-session.ts +59 -8
  80. package/src/types.ts +19 -5
  81. package/src/utils/model-cache.ts +67 -0
@@ -0,0 +1,215 @@
1
+ /**
2
+ * Claude Code model and effort discovery.
3
+ *
4
+ * Claude Code ships no catalog subcommand and its SDK init frame reports only
5
+ * the resolved model, so there is nothing to read the way `codex debug models`
6
+ * can be read. What the CLI *will* do is validate `--model` and `--effort`
7
+ * eagerly, before any API call, and say so on stderr when it does not
8
+ * recognize a value. That turns discovery into two different jobs:
9
+ *
10
+ * Efforts are enumerable. `--help` prints the accepted values inline, so the
11
+ * installed binary names them itself. Values it accepts but omits from help
12
+ * (`ultracode` today) are recovered by probing the canonical list.
13
+ *
14
+ * Models are not enumerable. Probing can only answer "do you know this one?"
15
+ * for names we already ship, so `CLAUDE_MODEL_CANDIDATES` stays a curated
16
+ * list. What probing buys is that a name the installed CLI does not know is
17
+ * dropped instead of offered and silently ignored at run time.
18
+ *
19
+ * Probes run with `--bare` when the binary supports it. The reason is side
20
+ * effects, not speed: `--bare` skips hooks, LSP, plugin sync, auto-memory, and
21
+ * keychain reads, so asking "do you know this flag value" cannot fire a user's
22
+ * SessionStart hook four times. Argument validation still runs under it.
23
+ * Probes also pass `-p ""`, so the process exits on empty input long before it
24
+ * would reach the network. No probe costs a token.
25
+ *
26
+ * Budget: two serial spawns, then one concurrent batch. `--help` and the
27
+ * control probe below are each on the critical path; only the per-candidate
28
+ * probes run in parallel. Measured cold against 2.1.232 that is ~1.8-3s
29
+ * depending on machine load, against ~0.1s for Codex's single-command
30
+ * catalog.
31
+ * Worth caching for minutes rather than seconds: callers should pass
32
+ * `cacheTtlMs` and keep a static fallback for first paint rather than blocking
33
+ * a picker on this.
34
+ */
35
+ import type { ListModelsOptions, ProviderModel } from "../../types.js";
36
+ import { findBinary, type ResolvedBinary } from "../../utils/binary.js";
37
+ import { buildEnv, ensurePathInEnv } from "../../utils/env.js";
38
+ import { runChildProcess } from "../../utils/process.js";
39
+ import { withModelCache } from "../../utils/model-cache.js";
40
+ import { CLAUDE_EFFORTS, claudeEffortFlagValue, claudeEffortFromFlagValue } from "./effort.js";
41
+
42
+ /**
43
+ * Tier aliases rather than pinned versions, so each one keeps meaning "the
44
+ * newest model in this tier" and the list does not rot between releases. A
45
+ * genuinely new tier is one line here, gated by the probe so older CLIs that
46
+ * have never heard of it simply do not show it.
47
+ */
48
+ export const CLAUDE_MODEL_CANDIDATES: ReadonlyArray<{ id: string; name: string; description: string }> = [
49
+ { id: "opus", name: "Opus", description: "Always resolves to the newest Opus release" },
50
+ { id: "sonnet", name: "Sonnet", description: "Always resolves to the newest Sonnet release" },
51
+ { id: "haiku", name: "Haiku", description: "Always resolves to the newest Haiku release" },
52
+ { id: "fable", name: "Fable", description: "Always resolves to the newest Fable release" },
53
+ ];
54
+
55
+ /**
56
+ * The error a probe earns when the flag value was accepted: the CLI got far
57
+ * enough to care that the prompt was empty. Treated as the positive signal so
58
+ * an unexpected failure (missing binary, unknown flag, changed wording) reads
59
+ * as "unsupported" rather than "supported". Discovery that fails open would
60
+ * offer values the CLI then ignores, which is the exact failure this replaces.
61
+ */
62
+ const ACCEPTED = /input must be provided/i;
63
+ const UNKNOWN_EFFORTS = /unknown --effort value/i;
64
+ const UNKNOWN_MODEL = /is not a model this version of claude code recognizes/i;
65
+
66
+ /**
67
+ * Read the effort values `--help` advertises.
68
+ *
69
+ * The option's accepted list is rendered inline and commander wraps it onto a
70
+ * continuation line, so the scan joins whitespace and stops at the next flag
71
+ * to avoid swallowing a later option's parenthetical.
72
+ */
73
+ export function claudeEffortsFromHelp(output: string): string[] {
74
+ const index = output.indexOf("--effort");
75
+ if (index === -1) return [];
76
+ const tail = output.slice(index + "--effort".length);
77
+ const stop = tail.search(/\n\s*(?:-[a-z]|--[a-z])/i);
78
+ const block = (stop === -1 ? tail : tail.slice(0, stop)).replace(/\s+/g, " ");
79
+ const list = block.match(/\(([^)]+)\)/);
80
+ if (!list?.[1]) return [];
81
+ const values = list[1]
82
+ .split(",")
83
+ .map((value) => value.trim().toLowerCase())
84
+ .filter((value) => /^[a-z][a-z0-9-]*$/.test(value));
85
+ return [...new Set(values.map(claudeEffortFromFlagValue))];
86
+ }
87
+
88
+ /** Order a discovered set by the canonical ladder, unknown values last. */
89
+ function ordered(values: Iterable<string>): string[] {
90
+ const seen = [...new Set(values)];
91
+ const rank = (value: string) => {
92
+ const index = (CLAUDE_EFFORTS as readonly string[]).indexOf(value);
93
+ return index === -1 ? CLAUDE_EFFORTS.length : index;
94
+ };
95
+ return seen.sort((left, right) => rank(left) - rank(right));
96
+ }
97
+
98
+ interface ClaudeBinary {
99
+ resolved: ResolvedBinary;
100
+ env: Record<string, string>;
101
+ cwd: string;
102
+ }
103
+
104
+ interface ProbeRuntime extends ClaudeBinary {
105
+ /** Whether the binary is new enough for `--bare` (see the file header). */
106
+ bare: boolean;
107
+ /** `--help` output, already paid for while detecting `--bare`. */
108
+ help: string;
109
+ }
110
+
111
+ async function runClaude(runtime: ClaudeBinary, args: string[], timeoutSec = 30): Promise<string> {
112
+ const result = await runChildProcess({
113
+ runId: "claude-discovery",
114
+ command: runtime.resolved.bin,
115
+ args: [...runtime.resolved.prefixArgs, ...args],
116
+ cwd: runtime.cwd,
117
+ env: runtime.env,
118
+ timeoutSec,
119
+ });
120
+ return `${result.stdout}\n${result.stderr}`;
121
+ }
122
+
123
+ async function probeFlag(
124
+ runtime: ProbeRuntime,
125
+ flag: "--model" | "--effort",
126
+ value: string,
127
+ ): Promise<boolean> {
128
+ const output = await runClaude(
129
+ runtime,
130
+ [...(runtime.bare ? ["--bare"] : []), flag, value, "-p", ""],
131
+ );
132
+ if (!ACCEPTED.test(output)) return false;
133
+ return !(flag === "--effort" ? UNKNOWN_EFFORTS : UNKNOWN_MODEL).test(output);
134
+ }
135
+
136
+ /**
137
+ * Prove the probe mechanism itself still works, with no flag under test.
138
+ *
139
+ * Without this, an empty result is ambiguous: it means both "this CLI
140
+ * recognizes none of our candidates" and "the probe broke" — a renamed flag, a
141
+ * reworded error, a binary that will not start. Failing closed only helps if
142
+ * the caller can tell those apart, and it cannot: falling back to a static list
143
+ * on a broken probe offers exactly the unvalidated names that failing open
144
+ * would have. So a broken mechanism throws, and an empty list becomes a real
145
+ * answer rather than a shrug.
146
+ */
147
+ async function probeMechanismWorks(runtime: ProbeRuntime): Promise<boolean> {
148
+ const output = await runClaude(runtime, [...(runtime.bare ? ["--bare"] : []), "-p", ""]);
149
+ return ACCEPTED.test(output);
150
+ }
151
+
152
+ /**
153
+ * Effort values the installed binary accepts, in canonical vocabulary.
154
+ *
155
+ * Exported separately because efforts are a property of the CLI rather than of
156
+ * any one model: Claude takes a single `--effort` flag and applies it to
157
+ * whatever model the turn runs on.
158
+ */
159
+ export async function listClaudeEfforts(options: ListModelsOptions = {}): Promise<string[]> {
160
+ const runtime = await probeRuntime(options);
161
+ if (!await probeMechanismWorks(runtime)) {
162
+ throw new Error("Claude Code effort discovery failed: the CLI did not respond as expected to a probe");
163
+ }
164
+ return discoverEfforts(runtime);
165
+ }
166
+
167
+ async function probeRuntime(options: ListModelsOptions): Promise<ProbeRuntime> {
168
+ const resolved = await findBinary("claude", options.config?.command);
169
+ const env = buildEnv(options.env);
170
+ ensurePathInEnv(env);
171
+ const base: ClaudeBinary = { resolved, env, cwd: options.cwd ?? process.cwd() };
172
+ // One spawn answers two questions: which efforts the binary advertises, and
173
+ // whether it understands `--bare` well enough to probe without side effects.
174
+ const help = await runClaude(base, ["--help"], 15);
175
+ return { ...base, bare: /--bare\b/.test(help), help };
176
+ }
177
+
178
+ async function discoverEfforts(runtime: ProbeRuntime): Promise<string[]> {
179
+ const advertised = claudeEffortsFromHelp(runtime.help);
180
+ // Anything help omits still gets a probe, so a value the CLI accepts without
181
+ // documenting is not lost. When help parsing yields nothing this degrades to
182
+ // probing the whole ladder rather than to returning nothing.
183
+ const unlisted: string[] = CLAUDE_EFFORTS.filter((effort) => !advertised.includes(effort));
184
+ const confirmed = await Promise.all(
185
+ unlisted.map(async (effort) => (
186
+ await probeFlag(runtime, "--effort", claudeEffortFlagValue(effort)) ? effort : null
187
+ )),
188
+ );
189
+ return ordered([...advertised, ...confirmed.filter((effort): effort is string => effort !== null)]);
190
+ }
191
+
192
+ export async function listClaudeModels(options: ListModelsOptions = {}): Promise<ProviderModel[]> {
193
+ return withModelCache("claude", options, options.cacheTtlMs, async () => {
194
+ const runtime = await probeRuntime(options);
195
+ // Throwing rather than returning [] keeps "no models" meaningful. Callers
196
+ // already treat a discovery failure as "use your fallback catalog".
197
+ if (!await probeMechanismWorks(runtime)) {
198
+ throw new Error("Claude Code model discovery failed: the CLI did not respond as expected to a probe");
199
+ }
200
+ const [supportedEfforts, recognized] = await Promise.all([
201
+ discoverEfforts(runtime),
202
+ Promise.all(CLAUDE_MODEL_CANDIDATES.map(async (candidate) => (
203
+ await probeFlag(runtime, "--model", candidate.id) ? candidate : null
204
+ ))),
205
+ ]);
206
+ return recognized
207
+ .filter((candidate): candidate is (typeof CLAUDE_MODEL_CANDIDATES)[number] => candidate !== null)
208
+ .map((candidate) => ({
209
+ id: candidate.id,
210
+ name: candidate.name,
211
+ description: candidate.description,
212
+ ...(supportedEfforts.length > 0 ? { supportedEfforts } : {}),
213
+ }));
214
+ });
215
+ }
@@ -0,0 +1,34 @@
1
+ /**
2
+ * Reasoning-effort vocabulary for Claude Code.
3
+ *
4
+ * `ProviderConfig.effort` is one shared scale across providers, but the CLIs
5
+ * do not agree on what to call its top rung: Codex accepts `ultra`, Claude
6
+ * accepts `ultracode`. Translating at the flag boundary keeps that difference
7
+ * out of every caller.
8
+ *
9
+ * This is not cosmetic. `claude --effort ultra` does not fail — it prints a
10
+ * warning to stderr and runs the turn at the session default. An untranslated
11
+ * value is therefore a silent downgrade, indistinguishable at the API surface
12
+ * from having worked.
13
+ */
14
+
15
+ /** Canonical effort ids, weakest to strongest, in the shared vocabulary. */
16
+ export const CLAUDE_EFFORTS = ["low", "medium", "high", "xhigh", "max", "ultra"] as const;
17
+
18
+ /** Canonical id -> the token `claude --effort` accepts. Identity when absent. */
19
+ const WIRE_VALUES: Record<string, string> = {
20
+ ultra: "ultracode",
21
+ };
22
+
23
+ /** The token to pass to `--effort` for a canonical effort id. */
24
+ export function claudeEffortFlagValue(effort: string): string {
25
+ return WIRE_VALUES[effort] ?? effort;
26
+ }
27
+
28
+ /** Inverse of `claudeEffortFlagValue`, for reading a CLI-advertised list. */
29
+ export function claudeEffortFromFlagValue(value: string): string {
30
+ for (const [canonical, wire] of Object.entries(WIRE_VALUES)) {
31
+ if (wire === value) return canonical;
32
+ }
33
+ return value;
34
+ }
@@ -7,6 +7,7 @@ import { runChildProcess, deriveErrorCode } from "../../utils/process.js";
7
7
  import { detectAuth } from "../../utils/auth.js";
8
8
  import { buildSkillsDir, cleanupSkillsDir } from "../../utils/skills.js";
9
9
  import { claudeFeatureArgs, cleanupMcpConfig, stageMcpConfig } from "./mcp.js";
10
+ import { claudeEffortFlagValue } from "./effort.js";
10
11
  import { createToolNameTracker } from "../../utils/tool-names.js";
11
12
  import { uuidv7 } from "../../utils/uuid.js";
12
13
  import { prepareWorkspace } from "../../utils/workspace.js";
@@ -109,7 +110,7 @@ export async function executeClaudeProvider(ctx: ExecutionContext): Promise<Exec
109
110
  args.push("--dangerously-skip-permissions");
110
111
  }
111
112
  if (model) args.push("--model", model);
112
- if (config.effort) args.push("--effort", config.effort);
113
+ if (config.effort) args.push("--effort", claudeEffortFlagValue(config.effort));
113
114
  if (config.maxTurns && config.maxTurns > 0) args.push("--max-turns", String(config.maxTurns));
114
115
  if (config.instructionsFile) args.push("--append-system-prompt-file", config.instructionsFile);
115
116
  if (skillsDir) args.push("--add-dir", skillsDir);
@@ -20,7 +20,7 @@ export const claudeProvider: ProviderModule = {
20
20
  type: "claude",
21
21
  capabilities: {
22
22
  sessions: true,
23
- modelDiscovery: false,
23
+ modelDiscovery: true,
24
24
  quotaProbing: true,
25
25
  mcp: true,
26
26
  skills: true,
@@ -56,6 +56,7 @@ export const claudeProvider: ProviderModule = {
56
56
  createSession: async (ctx: SessionContext): Promise<AgentSession> =>
57
57
  (await import("./session.js")).createClaudeSession(ctx),
58
58
  resolveAuth: (ctx) => resolveAuthForProvider("claude", ctx),
59
+ listModels: (options) => import("./discovery.js").then((m) => m.listClaudeModels(options)),
59
60
  sessionCodec: claudeSessionCodec,
60
61
  checkQuota,
61
62
  transcript: claudeTranscriptOps,
@@ -802,9 +802,19 @@ function normalizedClaudeTaskType(task: ClaudeTaskDetails): "subagent" | "proces
802
802
  return "unknown";
803
803
  }
804
804
 
805
+ /**
806
+ * Map a reported Claude status onto the neutral vocabulary.
807
+ *
808
+ * Returns null when the provider reported nothing, or reported a value this
809
+ * version does not model. Both mean "we do not know", and saying so is the
810
+ * point: a `task_updated` patch that only renames a task used to assert
811
+ * `running` over a real `paused`, and an unmodeled future status used to be
812
+ * asserted as running rather than left uncertain — which quietly undid the
813
+ * forward-compat that `claudeTaskDetailsFromRaw` establishes upstream.
814
+ */
805
815
  function normalizedClaudeTaskStatus(
806
816
  status: ClaudeTaskStatus | null,
807
- ): "pending" | "running" | "paused" | "completed" | "failed" | "stopped" {
817
+ ): "pending" | "running" | "paused" | "completed" | "failed" | "stopped" | null {
808
818
  switch (status) {
809
819
  case "pending": return "pending";
810
820
  case "paused": return "paused";
@@ -812,8 +822,8 @@ function normalizedClaudeTaskStatus(
812
822
  case "failed": return "failed";
813
823
  case "killed":
814
824
  case "stopped": return "stopped";
815
- case "running":
816
- default: return "running";
825
+ case "running": return "running";
826
+ default: return null;
817
827
  }
818
828
  }
819
829
 
@@ -824,15 +834,27 @@ function backgroundTaskEventFromClaude(
824
834
  const task = claudeTaskDetailsFromRaw(raw);
825
835
  if (!task || !task.taskId) return null;
826
836
 
837
+ // A bare notification is Claude's "this task is done" signal. One that
838
+ // carries a status is reporting that status, and must be believed.
839
+ // Inference is allowed only where the event type itself carries the fact: a
840
+ // bare notification is Claude's "done" signal, and started/progress events
841
+ // are emitted BY a running task. `task_updated` is a patch — it can carry
842
+ // nothing but a rename, so it says nothing about liveness and must not
843
+ // assert `running` over a real `paused`.
844
+ const implied = task.phase === "started" || task.phase === "progress" ? "running" : null;
827
845
  const status = task.phase === "notification" && task.status === null
828
846
  ? "completed"
829
- : normalizedClaudeTaskStatus(task.status);
847
+ : normalizedClaudeTaskStatus(task.status) ?? implied;
830
848
  const terminal = status === "completed" || status === "failed" || status === "stopped";
831
849
  return {
832
850
  type: "background_task",
833
851
  taskId: task.taskId,
834
852
  taskType: normalizedClaudeTaskType(task),
835
- phase: terminal || task.phase === "notification"
853
+ // Terminality is decided by the status alone. Treating every notification
854
+ // as terminal emitted `phase: "completed"` alongside `status: "running"`,
855
+ // and hosts are told to drop a task on `phase === "completed"` — so a
856
+ // mid-flight notification evicted a task the same event said was running.
857
+ phase: terminal
836
858
  ? "completed"
837
859
  : task.phase === "started" ? "started" : "progress",
838
860
  status,
@@ -26,6 +26,7 @@ import { claudeTranscriptOps } from "./transcript.js";
26
26
  import { findBinary } from "../../utils/binary.js";
27
27
  import { buildEnv, ensurePathInEnv } from "../../utils/env.js";
28
28
  import { translateEndpoint } from "../../utils/endpoint.js";
29
+ import { claudeEffortFlagValue } from "./effort.js";
29
30
  import { buildSkillsDir, cleanupSkillsDir } from "../../utils/skills.js";
30
31
  import { claudeFeatureArgs, cleanupMcpConfig, stageMcpConfig } from "./mcp.js";
31
32
  import { createToolNameTracker } from "../../utils/tool-names.js";
@@ -193,7 +194,7 @@ export async function createClaudeSession(ctx: SessionContext): Promise<AgentSes
193
194
  args.push("--permission-prompt-tool", "stdio");
194
195
  }
195
196
  if (config.model) args.push("--model", config.model);
196
- if (config.effort) args.push("--effort", config.effort);
197
+ if (config.effort) args.push("--effort", claudeEffortFlagValue(config.effort));
197
198
  if (config.maxTurns && config.maxTurns > 0) args.push("--max-turns", String(config.maxTurns));
198
199
  if (config.instructionsFile) args.push("--append-system-prompt-file", config.instructionsFile);
199
200
  if (skillsDir) args.push("--add-dir", skillsDir);
@@ -0,0 +1,91 @@
1
+ /**
2
+ * Codex model discovery.
3
+ *
4
+ * `codex debug models` prints the CLI's own catalog as JSON, including the
5
+ * reasoning levels each model accepts and which one it defaults to. That makes
6
+ * Codex the one provider here whose efforts are genuinely per-model: 5.6 Sol
7
+ * advertises `ultra`, 5.5 stops at `xhigh`, and hardcoding either answer would
8
+ * be wrong for the other.
9
+ */
10
+ import type { ListModelsOptions, ProviderModel } from "../../types.js";
11
+ import { findBinary } from "../../utils/binary.js";
12
+ import { buildEnv, ensurePathInEnv } from "../../utils/env.js";
13
+ import { runChildProcess } from "../../utils/process.js";
14
+ import { withModelCache } from "../../utils/model-cache.js";
15
+
16
+ interface CatalogEntry {
17
+ slug?: unknown;
18
+ display_name?: unknown;
19
+ description?: unknown;
20
+ visibility?: unknown;
21
+ supported_reasoning_levels?: unknown;
22
+ default_reasoning_level?: unknown;
23
+ }
24
+
25
+ function effortList(value: unknown): string[] {
26
+ if (!Array.isArray(value)) return [];
27
+ const seen = new Set<string>();
28
+ for (const item of value) {
29
+ // Entries are objects ({ effort, description }), not bare strings.
30
+ if (!item || typeof item !== "object") continue;
31
+ const effort = (item as { effort?: unknown }).effort;
32
+ if (typeof effort === "string" && effort) seen.add(effort);
33
+ }
34
+ return [...seen];
35
+ }
36
+
37
+ /**
38
+ * Parse `codex debug models` output.
39
+ *
40
+ * Only `visibility: "list"` entries are returned. The CLI also ships hidden
41
+ * and deprecated slugs that it will accept but does not offer, and surfacing
42
+ * those as pickable models would be a worse catalog than no catalog.
43
+ */
44
+ export function parseCodexModelCatalog(raw: string): ProviderModel[] {
45
+ const parsed = JSON.parse(raw) as { models?: unknown };
46
+ if (!Array.isArray(parsed.models)) throw new Error("Codex model catalog has no models array");
47
+
48
+ const seen = new Set<string>();
49
+ const models: ProviderModel[] = [];
50
+ for (const entry of parsed.models as CatalogEntry[]) {
51
+ if (entry.visibility !== "list") continue;
52
+ if (typeof entry.slug !== "string" || typeof entry.display_name !== "string") continue;
53
+ if (seen.has(entry.slug)) continue;
54
+ seen.add(entry.slug);
55
+ const supportedEfforts = effortList(entry.supported_reasoning_levels);
56
+ models.push({
57
+ id: entry.slug,
58
+ name: entry.display_name,
59
+ ...(typeof entry.description === "string" && entry.description
60
+ ? { description: entry.description.replace(/\.$/, "") }
61
+ : {}),
62
+ ...(supportedEfforts.length > 0 ? { supportedEfforts } : {}),
63
+ ...(typeof entry.default_reasoning_level === "string" && entry.default_reasoning_level
64
+ ? { defaultEffort: entry.default_reasoning_level }
65
+ : {}),
66
+ });
67
+ }
68
+
69
+ if (models.length === 0) throw new Error("Codex model catalog has no visible models");
70
+ return models;
71
+ }
72
+
73
+ export async function listCodexModels(options: ListModelsOptions = {}): Promise<ProviderModel[]> {
74
+ return withModelCache("codex", options, options.cacheTtlMs, async () => {
75
+ const resolved = await findBinary("codex", options.config?.command);
76
+ const env = buildEnv(options.env);
77
+ ensurePathInEnv(env);
78
+ const result = await runChildProcess({
79
+ runId: "codex-model-discovery",
80
+ command: resolved.bin,
81
+ args: [...resolved.prefixArgs, "debug", "models"],
82
+ cwd: options.cwd ?? process.cwd(),
83
+ env,
84
+ timeoutSec: 15,
85
+ });
86
+ if (result.exitCode !== 0) {
87
+ throw new Error("The installed Codex CLI does not expose `codex debug models`");
88
+ }
89
+ return parseCodexModelCatalog(result.stdout || result.stderr);
90
+ });
91
+ }
@@ -15,7 +15,7 @@ import type { PreparedWorkspace } from "../../utils/workspace.js";
15
15
  import { uuidv7 } from "../../utils/uuid.js";
16
16
  import {
17
17
  parseCodexJsonl,
18
- parseCodexStreamLine,
18
+ parseCodexStreamLines,
19
19
  stripCodexRolloutNoise,
20
20
  isCodexAuthRequired,
21
21
  isCodexUnknownSessionError,
@@ -165,6 +165,11 @@ export async function executeCodexProvider(ctx: ExecutionContext): Promise<Execu
165
165
  // Correlates tool_call → tool_result so emitted tool_result events carry
166
166
  // toolName. One tracker per attempt (a retry restarts the stream).
167
167
  const trackToolName = createToolNameTracker();
168
+ // Children still running when the one-shot process ends. `execute()` has no
169
+ // session to reconcile against, so without a terminal edge here a host that
170
+ // trusts `capabilities.backgroundTaskEvents` shows those children as
171
+ // running forever. The session path does the same on shutdown.
172
+ const openTasks = new Map<string, { description: string | null; parentTaskId: string | null }>();
168
173
 
169
174
  const handleLine = async (trimmed: string) => {
170
175
  if (!trimmed) return;
@@ -177,8 +182,14 @@ export async function executeCodexProvider(ctx: ExecutionContext): Promise<Execu
177
182
  } catch { /* ignore */ }
178
183
  }
179
184
  if (!ctx.onEvent) return;
180
- const event = parseCodexStreamLine(trimmed, streamThreadId);
181
- if (event) {
185
+ for (const event of parseCodexStreamLines(trimmed, streamThreadId)) {
186
+ if (event.type === "background_task") {
187
+ if (event.phase === "completed") openTasks.delete(event.taskId);
188
+ else openTasks.set(event.taskId, {
189
+ description: event.description,
190
+ parentTaskId: event.parentTaskId,
191
+ });
192
+ }
182
193
  try { await ctx.onEvent(trackToolName(event)); } catch { /* swallow */ }
183
194
  }
184
195
  };
@@ -225,6 +236,34 @@ export async function executeCodexProvider(ctx: ExecutionContext): Promise<Execu
225
236
  await handleLine(lineBuffer.trim());
226
237
  }
227
238
 
239
+ for (const [taskId, task] of openTasks) {
240
+ if (!ctx.onEvent) break;
241
+ try {
242
+ await ctx.onEvent({
243
+ type: "background_task",
244
+ taskId,
245
+ taskType: "subagent",
246
+ phase: "completed",
247
+ // "stopped", not "completed": the run ended while this child was
248
+ // still going. Claiming success for work we never saw finish would
249
+ // be worse than reporting that it was cut short.
250
+ status: "stopped",
251
+ description: task.description,
252
+ summary: null,
253
+ parentTaskId: task.parentTaskId,
254
+ timestamp: new Date().toISOString(),
255
+ providerType: "codex",
256
+ sessionId: streamThreadId,
257
+ messageId: null,
258
+ eventId: streamThreadId ? `codex:${streamThreadId}:background-task:${taskId}:exit` : null,
259
+ turnId: null,
260
+ parentToolCallId: null,
261
+ raw: { reason: "process_exit" },
262
+ });
263
+ } catch { /* swallow */ }
264
+ }
265
+ openTasks.clear();
266
+
228
267
  return proc;
229
268
  };
230
269
 
@@ -8,7 +8,7 @@ export const codexProvider: ProviderModule = {
8
8
  type: "codex",
9
9
  capabilities: {
10
10
  sessions: true,
11
- modelDiscovery: false,
11
+ modelDiscovery: true,
12
12
  quotaProbing: false,
13
13
  mcp: false,
14
14
  skills: true,
@@ -45,6 +45,7 @@ export const codexProvider: ProviderModule = {
45
45
  resolveAuth: (ctx) => resolveAuthForProvider("codex", ctx),
46
46
  sessionCodec: codexSessionCodec,
47
47
  transcript: codexTranscriptOps,
48
+ listModels: (options) => import("./discovery.js").then((m) => m.listCodexModels(options)),
48
49
  listModes: async (opts) => (await import("./modes.js")).listCodexModes(opts),
49
50
  attachSession: async (record, opts) =>
50
51
  (await import("./attach.js")).attachCodexSession(record, opts),