@bitkyc08/opencodex 2.52.0 → 2.53.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/gui/dist/assets/index-BBOZWGB6.css +1 -0
  2. package/gui/dist/assets/index-BlO4Yl6q.js +128 -0
  3. package/gui/dist/index.html +2 -2
  4. package/native/remote-workspace-helper/Cargo.lock +130 -0
  5. package/native/remote-workspace-helper/Cargo.toml +24 -0
  6. package/native/remote-workspace-helper/src/main.rs +49 -0
  7. package/native/remote-workspace-helper/src/protocol.rs +246 -0
  8. package/native/remote-workspace-helper/src/sandbox/macos.rs +19 -0
  9. package/native/remote-workspace-helper/src/sandbox/mod.rs +77 -0
  10. package/native/remote-workspace-helper/src/sandbox/windows.rs +15 -0
  11. package/package.json +6 -1
  12. package/src/adapters/anthropic-image-normalize.ts +30 -2
  13. package/src/adapters/anthropic.ts +1 -1
  14. package/src/adapters/base.ts +8 -2
  15. package/src/adapters/cursor/cursor-errors.ts +12 -0
  16. package/src/adapters/cursor/thread-continuity.ts +93 -0
  17. package/src/adapters/cursor.ts +104 -73
  18. package/src/adapters/devin/cloud-direct/chat.ts +312 -23
  19. package/src/adapters/devin/cloud-direct/metadata.ts +31 -2
  20. package/src/adapters/devin/live-models.ts +70 -3
  21. package/src/adapters/devin.ts +281 -21
  22. package/src/adapters/google-wire-compiler.ts +14 -6
  23. package/src/adapters/google.ts +22 -8
  24. package/src/adapters/kiro/adapter.ts +316 -0
  25. package/src/adapters/kiro/conversation.ts +136 -0
  26. package/src/adapters/kiro/payload.ts +432 -0
  27. package/src/adapters/kiro/reasoning.ts +56 -0
  28. package/src/adapters/kiro/stream.ts +1153 -0
  29. package/src/adapters/kiro/usage.ts +223 -0
  30. package/src/adapters/kiro/wire.ts +76 -0
  31. package/src/adapters/kiro.ts +8 -2319
  32. package/src/adapters/mimo-free.ts +1 -1
  33. package/src/adapters/openai-chat-images.ts +101 -0
  34. package/src/adapters/openai-chat.ts +201 -181
  35. package/src/adapters/openai-responses.ts +92 -224
  36. package/src/adapters/registry.ts +0 -7
  37. package/src/adapters/run-turn-queue.ts +13 -6
  38. package/src/bridge.ts +14 -15
  39. package/src/chat/inbound.ts +29 -4
  40. package/src/chat/outbound.ts +145 -107
  41. package/src/claude/desktop-profile.ts +4 -6
  42. package/src/cli/account-api.ts +14 -0
  43. package/src/cli/account-extended.ts +1 -1
  44. package/src/cli/account-history.ts +60 -0
  45. package/src/cli/account-main.ts +80 -0
  46. package/src/cli/account.ts +11 -3
  47. package/src/cli/capabilities.ts +113 -0
  48. package/src/cli/catalog.ts +109 -0
  49. package/src/cli/dispatch.ts +9 -0
  50. package/src/cli/help.ts +2 -0
  51. package/src/cli/index.ts +2 -2
  52. package/src/cli/observe.ts +28 -1
  53. package/src/cli/opencode.ts +42 -8
  54. package/src/cli/provider-runtime.ts +11 -1
  55. package/src/cli/provider.ts +22 -2
  56. package/src/cli/registry.ts +21 -0
  57. package/src/cli/remote-workspace.ts +154 -0
  58. package/src/cli/status.ts +39 -7
  59. package/src/cli/usage-report.ts +14 -2
  60. package/src/client/hub-client.ts +34 -0
  61. package/src/client/hub-state.ts +9 -1
  62. package/src/codex/account-store.ts +78 -0
  63. package/src/codex/auth-api.ts +81 -54
  64. package/src/codex/auth-context.ts +45 -16
  65. package/src/codex/catalog/effort.ts +1 -1
  66. package/src/codex/catalog/metadata.ts +3 -6
  67. package/src/codex/catalog/native-models.ts +4 -4
  68. package/src/codex/catalog/parsing.ts +2 -20
  69. package/src/codex/catalog/provider-fetch.ts +10 -1
  70. package/src/codex/catalog/remote.ts +233 -0
  71. package/src/codex/catalog/sync.ts +403 -35
  72. package/src/codex/convergence.ts +1 -1
  73. package/src/codex/history-manifest.ts +36 -0
  74. package/src/codex/history-provider.ts +32 -5
  75. package/src/codex/inject.ts +9 -0
  76. package/src/codex/main-account.ts +113 -0
  77. package/src/codex/main-device-reauth-api.ts +89 -0
  78. package/src/codex/main-device-reauth.ts +217 -0
  79. package/src/codex/native-residue.ts +9 -2
  80. package/src/codex/quota-auto-refresh.ts +3 -2
  81. package/src/codex/quota-capacity.ts +98 -0
  82. package/src/codex/quota-history.ts +160 -0
  83. package/src/codex/quota-types.ts +8 -0
  84. package/src/codex/quota.ts +118 -91
  85. package/src/codex/refresh.ts +2 -1
  86. package/src/codex/routing.ts +90 -17
  87. package/src/codex/sync.ts +33 -4
  88. package/src/combos/request.ts +19 -1
  89. package/src/config/multi-agent-surface.ts +61 -0
  90. package/src/config/provider-validation.ts +176 -0
  91. package/src/config.ts +213 -11
  92. package/src/generated/compatibility-version.json +436 -168
  93. package/src/images/loop.ts +119 -36
  94. package/src/lib/admission.ts +12 -6
  95. package/src/lib/redact.ts +7 -0
  96. package/src/lib/translator-budget.ts +4 -3
  97. package/src/lib/windows-atomic-replace.ts +1 -0
  98. package/src/lib/windows-elevation.ts +1 -1
  99. package/src/oauth/chatgpt-device.ts +62 -5
  100. package/src/oauth/devin/cli-import.ts +130 -0
  101. package/src/oauth/devin.ts +63 -8
  102. package/src/oauth/index.ts +29 -14
  103. package/src/oauth/kiro.ts +18 -6
  104. package/src/oauth/login-cli.ts +9 -1
  105. package/src/oauth/meta-muse-device.ts +464 -0
  106. package/src/oauth/meta-muse.ts +123 -32
  107. package/src/oauth/pool-kernel.ts +9 -0
  108. package/src/oauth/pool-settings-capability.ts +2 -2
  109. package/src/oauth/store.ts +57 -0
  110. package/src/oauth/types.ts +31 -0
  111. package/src/providers/derive.ts +13 -3
  112. package/src/providers/devin-cli-authmode-migration.ts +57 -35
  113. package/src/providers/devin-provider-merge-migration.ts +240 -0
  114. package/src/providers/muse-key-quota.ts +117 -0
  115. package/src/providers/muse-subscription-usage.ts +14 -2
  116. package/src/providers/openai-sidecar.ts +25 -3
  117. package/src/providers/opencode-zen-rate-limit.ts +58 -0
  118. package/src/providers/provider-id-rewrite.ts +20 -5
  119. package/src/providers/quota-types.ts +12 -0
  120. package/src/providers/quota.ts +143 -102
  121. package/src/providers/reasoning-metadata.ts +543 -0
  122. package/src/providers/registry.ts +80 -49
  123. package/src/reasoning-effort.ts +26 -2
  124. package/src/remote/hub-usage.ts +32 -0
  125. package/src/remote-control/index.ts +192 -41
  126. package/src/remote-control/workspace-activation.ts +9 -0
  127. package/src/remote-control/workspace-agent-connection.ts +366 -0
  128. package/src/remote-control/workspace-claude-runtime.ts +243 -0
  129. package/src/remote-control/workspace-codex-runtime.ts +531 -0
  130. package/src/remote-control/workspace-codex-sandbox.ts +115 -0
  131. package/src/remote-control/workspace-command-runner.ts +748 -0
  132. package/src/remote-control/workspace-coordinator.ts +231 -0
  133. package/src/remote-control/workspace-device.ts +585 -0
  134. package/src/remote-control/workspace-executable.ts +43 -0
  135. package/src/remote-control/workspace-executor.ts +397 -0
  136. package/src/remote-control/workspace-hub.ts +519 -0
  137. package/src/remote-control/workspace-pi-runtime.ts +382 -0
  138. package/src/remote-control/workspace-process.ts +129 -0
  139. package/src/remote-control/workspace-rpc.ts +304 -0
  140. package/src/remote-control/workspace-runtime.ts +60 -0
  141. package/src/remote-control/workspace-secret-store.ts +39 -0
  142. package/src/remote-control/workspace-sessions.ts +799 -0
  143. package/src/remote-control/workspace-tool-bridge.ts +192 -0
  144. package/src/responses/code-mode-helper-compat.ts +22 -3
  145. package/src/responses/hosted-tool-policy.ts +0 -1
  146. package/src/responses/muse-tool-name-alias.ts +379 -0
  147. package/src/responses/plaintext-v2-agent-messages.ts +902 -0
  148. package/src/router.ts +7 -0
  149. package/src/routing/compatibility/behavior.ts +0 -1
  150. package/src/server/audio-client.ts +64 -0
  151. package/src/server/audio-dictation.ts +91 -0
  152. package/src/server/audio-live.ts +185 -0
  153. package/src/server/audio-transcriptions.ts +183 -0
  154. package/src/server/audio-upstream.ts +153 -0
  155. package/src/server/auth-cors.ts +61 -2
  156. package/src/server/chat-completions.ts +1 -1
  157. package/src/server/chat-native-sse.ts +92 -48
  158. package/src/server/chat-native.ts +37 -15
  159. package/src/server/hub-usage.ts +57 -0
  160. package/src/server/images.ts +4 -0
  161. package/src/server/index.ts +722 -57
  162. package/src/server/lifecycle.ts +5 -6
  163. package/src/server/live-call-bindings.ts +60 -0
  164. package/src/server/live.ts +12 -1
  165. package/src/server/management/agent-settings-routes.ts +25 -4
  166. package/src/server/management/api-access.ts +37 -0
  167. package/src/server/management/api-key-usage.ts +7 -2
  168. package/src/server/management/config-routes.ts +1 -18
  169. package/src/server/management/context.ts +15 -0
  170. package/src/server/management/logs-usage-routes.ts +2 -0
  171. package/src/server/management/oauth-account-routes.ts +39 -12
  172. package/src/server/management/provider-routes.ts +125 -2
  173. package/src/server/management/remote-workspace-routes.ts +140 -0
  174. package/src/server/management/route-registry.ts +15 -0
  175. package/src/server/management/usage-aggregate-cache.ts +14 -15
  176. package/src/server/management/usage-summary-cache.ts +2 -0
  177. package/src/server/management-api.ts +23 -0
  178. package/src/server/ports.ts +17 -0
  179. package/src/server/relay-eager.ts +4 -1
  180. package/src/server/relay.ts +70 -10
  181. package/src/server/request-decompress.ts +6 -3
  182. package/src/server/responses/agent-task-recovery.ts +25 -32
  183. package/src/server/responses/codex-auth-error.ts +11 -0
  184. package/src/server/responses/codex-ws-exchange.ts +52 -3
  185. package/src/server/responses/codex-ws-wire.ts +55 -0
  186. package/src/server/responses/compact.ts +9 -1
  187. package/src/server/responses/core.ts +337 -73
  188. package/src/server/responses/encrypted-payload.ts +45 -2
  189. package/src/server/responses/ws-upstream.ts +4 -1
  190. package/src/server/responses-self-named-namespace-scrub.ts +1 -3
  191. package/src/server/responses-undeclared-tool-guard.ts +1 -1
  192. package/src/server/search.ts +3 -0
  193. package/src/server/sse-payload-rewrite.ts +136 -51
  194. package/src/server/ws-bridge.ts +35 -1
  195. package/src/service/cli.ts +372 -0
  196. package/src/service/diagnostics.ts +340 -0
  197. package/src/service/guards.ts +303 -0
  198. package/src/service/health.ts +222 -0
  199. package/src/service/launchd.ts +853 -0
  200. package/src/service/orchestration.ts +617 -0
  201. package/src/service/repair.ts +334 -0
  202. package/src/service/state.ts +363 -0
  203. package/src/service/systemd.ts +229 -0
  204. package/src/service/windows-ops.ts +690 -0
  205. package/src/service/windows-scheduler.ts +769 -0
  206. package/src/service/windows-taskxml.ts +613 -0
  207. package/src/service.ts +22 -5550
  208. package/src/storage/cleanup/db.ts +258 -0
  209. package/src/storage/cleanup/execute.ts +358 -0
  210. package/src/storage/cleanup/paths.ts +189 -0
  211. package/src/storage/cleanup/pending.ts +140 -0
  212. package/src/storage/cleanup/preview.ts +292 -0
  213. package/src/storage/cleanup/reconcile.ts +347 -0
  214. package/src/storage/cleanup/restore.ts +932 -0
  215. package/src/storage/cleanup/satellite.ts +474 -0
  216. package/src/storage/cleanup/staging.ts +129 -0
  217. package/src/storage/cleanup/types.ts +98 -0
  218. package/src/storage/cleanup.ts +49 -3127
  219. package/src/types/accounts.ts +2 -0
  220. package/src/types/config.ts +13 -12
  221. package/src/types/provider.ts +37 -0
  222. package/src/types/request.ts +2 -0
  223. package/src/types/tools.ts +17 -5
  224. package/src/types.ts +1 -0
  225. package/src/usage/expected-prices.ts +127 -0
  226. package/src/usage/log.ts +58 -1
  227. package/src/vision/eligibility.ts +13 -2
  228. package/src/web-search/loop.ts +56 -3
  229. package/gui/dist/assets/index-CWXut3rG.js +0 -115
  230. package/gui/dist/assets/index-EdoPnm9_.css +0 -1
  231. package/src/adapters/devin-cli/acp.ts +0 -204
  232. package/src/adapters/devin-cli/adapter.ts +0 -345
  233. package/src/adapters/devin-cli/binary.ts +0 -69
  234. package/src/adapters/devin-cli/models.ts +0 -57
  235. package/src/oauth/devin-cli.ts +0 -149
  236. package/src/server/responses-reasoning-summary-rewrite.ts +0 -178
@@ -46,16 +46,47 @@ import { resolveDevinApiBaseUrl } from '../../../oauth/devin/api-base.js';
46
46
  * we only trigger when the server has genuinely stopped responding.
47
47
  */
48
48
  const CLOUD_STREAM_IDLE_MS = 120_000;
49
- /** Time-to-first-byte timeout. */
50
- const CLOUD_STREAM_TTFB_MS = 60_000;
49
+ /**
50
+ * Budget for the response HEADERS, which is not the same thing as a connect
51
+ * timeout. Cognition holds the headers until the model produces its first
52
+ * token, so on a high-effort reasoning model this bounds generation. A 60s
53
+ * value killed live swe-2 high turns at exactly 60000ms with no output while
54
+ * a sibling call on the same account was still alive at 76s, which is the
55
+ * defect this constant exists to record.
56
+ *
57
+ * It has to be at least as generous as the body idle budget above. The cost of
58
+ * the larger value is bounded and understood: a peer that goes silent at the
59
+ * TCP level without sending RST/FIN now hangs for this long instead of 60s. A
60
+ * peer that actually dies still rejects immediately. This timer is the only
61
+ * bound on that case once `timeout: 0` is set on the fetch, so it must not be
62
+ * removed. Override with OPENCODEX_DEVIN_TTFB_MS.
63
+ */
64
+ const CLOUD_STREAM_HEADERS_DEFAULT_MS = 300_000;
65
+ /** Upper bound for the override, so a stray value cannot wedge a turn forever. */
66
+ const CLOUD_STREAM_HEADERS_MAX_MS = 1_800_000;
67
+ function cloudStreamHeadersMs(): number {
68
+ const raw = process.env.OPENCODEX_DEVIN_TTFB_MS?.trim();
69
+ if (!raw) return CLOUD_STREAM_HEADERS_DEFAULT_MS;
70
+ const parsed = Number(raw);
71
+ if (!Number.isFinite(parsed) || parsed <= 0) return CLOUD_STREAM_HEADERS_DEFAULT_MS;
72
+ return Math.min(parsed, CLOUD_STREAM_HEADERS_MAX_MS);
73
+ }
74
+ /** Test seam for the headers budget; the resolver itself stays private. */
75
+ export const cloudStreamHeadersMsForTests = cloudStreamHeadersMs;
51
76
  /** Maximum acceptable Connect-RPC frame length (16 MB). */
52
77
  const MAX_FRAME_LEN = 16 * 1024 * 1024;
53
78
 
54
79
  /**
55
- * Per-(apiKey, host) session/cascade ID cache. Cloud uses these for
56
- * server-side context caching across turns of the same conversation; if we
57
- * mint a fresh sessionId on every call (which we used to), every turn looks
58
- * like a brand-new session and the prompt-cache hit ratio is zero.
80
+ * PromptCacheOptions.type = EPHEMERAL. Marks the system prefix as a cache entry
81
+ * the server may reuse on the next turn of the same session.
82
+ */
83
+ const PROMPT_CACHE_EPHEMERAL = 1;
84
+
85
+ /**
86
+ * Per-identity session/cascade ID cache. Cloud uses these for server-side
87
+ * context caching across turns of the same conversation; if we mint a fresh
88
+ * sessionId on every call (which we used to), every turn looks like a
89
+ * brand-new session and the prompt-cache hit ratio is zero.
59
90
  * Single-process scope is enough: opencode lives in one runtime for a TUI
60
91
  * session, and CLI one-shots don't benefit from caching anyway.
61
92
  */
@@ -63,6 +94,18 @@ interface SessionIds {
63
94
  sessionId: string;
64
95
  cascadeId: string;
65
96
  }
97
+
98
+ /**
99
+ * Cache key for one Devin credential on one host.
100
+ *
101
+ * The credential itself used to be the Map key. Hashing it keeps the raw token
102
+ * out of any structure a heap dump or debugger would walk, and gives the other
103
+ * per-account caches a name they can share. 16 hex is 64 bits, which against a
104
+ * bounded single-process map is not a collision risk worth widening the key for.
105
+ */
106
+ export function devinCacheIdentity(apiKey: string, host: string): string {
107
+ return crypto.createHash('sha256').update(`${host}\x1f${apiKey}`).digest('hex').slice(0, 16);
108
+ }
66
109
  /**
67
110
  * Bounded the same way the adapter bounds its cascade-id map: a long-running
68
111
  * proxy sees one entry per (host, api_key) pair, and nothing ever evicted them.
@@ -70,7 +113,7 @@ interface SessionIds {
70
113
  const SESSION_CACHE_MAX = 256;
71
114
  const sessionCache = new Map<string, SessionIds>();
72
115
  function getOrAllocateSessionIds(apiKey: string, host: string, cascadeIdOverride?: string): SessionIds {
73
- const key = `${host}\x1f${apiKey}`;
116
+ const key = devinCacheIdentity(apiKey, host);
74
117
  let ids = sessionCache.get(key);
75
118
  if (!ids) {
76
119
  ids = {
@@ -90,9 +133,21 @@ function getOrAllocateSessionIds(apiKey: string, host: string, cascadeIdOverride
90
133
  return ids;
91
134
  }
92
135
 
93
- /** Drop the cached session IDs — call after logout so a new sign-in starts fresh. */
94
- export function clearSessionIds(): void {
95
- sessionCache.clear();
136
+ /**
137
+ * Drop the cached session for ONE identity, after that account signs out or is
138
+ * switched away from.
139
+ *
140
+ * This replaces a global clear(). The proxy serves several accounts from one
141
+ * process, so clearing every entry on a per-provider logout would strip the
142
+ * session and cascade of accounts that were mid-turn. That is why the global
143
+ * version was never safe to call, and why nothing ever called it.
144
+ *
145
+ * A turn already in flight is unaffected: it received its SessionIds object at
146
+ * request start and never re-reads the map, so it finishes on the session it
147
+ * began with and the next turn allocates fresh.
148
+ */
149
+ export function invalidateSessionIdentity(identity: string): void {
150
+ sessionCache.delete(identity);
96
151
  }
97
152
 
98
153
  // ----------------------------------------------------------------------------
@@ -120,6 +175,9 @@ export function allocateCascadeId(): string {
120
175
  * #4 num_tokens: int (rough estimate)
121
176
  * #5 safe_for_code_telemetry: bool (1 = ok to log)
122
177
  * #10 images: repeated ImageData (multimodal)
178
+ * #11 thinking: string (assistant reasoning, replayed)
179
+ * #12 signature: string (opaque attestation for #11)
180
+ * #18 signature_type: string
123
181
  * }
124
182
  *
125
183
  * ImageData (exa.codeium_common_pb.ImageData) {
@@ -153,7 +211,13 @@ function encodeChatToolCall(tc: { id: string; name: string; arguments: string })
153
211
  function encodeChatMessagePrompt(
154
212
  content: ContentPart[],
155
213
  source: number,
156
- opts?: { toolCallId?: string; toolCalls?: Array<{ id: string; name: string; arguments: string }> },
214
+ opts?: {
215
+ toolCallId?: string;
216
+ toolCalls?: Array<{ id: string; name: string; arguments: string }>;
217
+ thinking?: string;
218
+ signature?: string;
219
+ signatureType?: string;
220
+ },
157
221
  ): Buffer {
158
222
  const textParts = content.filter((p): p is { type: 'text'; text: string } => p.type === 'text');
159
223
  const imageParts = content.filter((p): p is { type: 'image'; mimeType: string; base64Data: string; caption?: string } => p.type === 'image');
@@ -178,6 +242,14 @@ function encodeChatMessagePrompt(
178
242
  for (const img of imageParts) {
179
243
  parts.push(encodeMessage(10, encodeImageData(img)));
180
244
  }
245
+ // Reasoning replay. This adapter used to assert that Cognition has no
246
+ // reasoning-replay field and drop the assistant's own thinking, so a
247
+ // reasoning model restarted its chain on every turn of a tool loop. Two
248
+ // independent clients of the same service write it here: #11 thinking,
249
+ // #12 signature, #18 signature_type on the assistant prompt.
250
+ if (opts?.thinking) parts.push(encodeString(11, opts.thinking));
251
+ if (opts?.signature) parts.push(encodeString(12, opts.signature));
252
+ if (opts?.signatureType) parts.push(encodeString(18, opts.signatureType));
181
253
  return Buffer.concat(parts);
182
254
  }
183
255
 
@@ -342,6 +414,15 @@ export interface ChatHistoryItem {
342
414
  * each ChatToolCall has #1 id, #2 name, #3 arguments_json).
343
415
  */
344
416
  tool_calls?: Array<{ id: string; name: string; arguments: string }>;
417
+ /**
418
+ * For `role: 'assistant'` only — the model's own reasoning from that turn,
419
+ * replayed so a reasoning model does not restart its chain on the next one.
420
+ * Encoded as ChatMessagePrompt #11 with its #12 signature and #18
421
+ * signature_type.
422
+ */
423
+ thinking?: string;
424
+ signature?: string;
425
+ signature_type?: string;
345
426
  }
346
427
 
347
428
  /**
@@ -399,6 +480,12 @@ export interface ToolDef {
399
480
  export type CloudChatEvent =
400
481
  | { kind: 'text'; text: string }
401
482
  | { kind: 'reasoning'; text: string }
483
+ /**
484
+ * `delta_signature` (#10) — the opaque attestation for the reasoning this
485
+ * turn produced. Without decoding it there is nothing to put in the prompt's
486
+ * #12 on the next turn, so the replay would always be unsigned.
487
+ */
488
+ | { kind: 'reasoning_signature'; signature: string }
402
489
  | { kind: 'tool_call_start'; id: string; name: string }
403
490
  | {
404
491
  kind: 'tool_call_args';
@@ -592,6 +679,9 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
592
679
  {
593
680
  toolCallId: m.role === 'tool' ? m.tool_call_id : undefined,
594
681
  toolCalls: m.role === 'assistant' ? m.tool_calls : undefined,
682
+ thinking: m.role === 'assistant' ? m.thinking : undefined,
683
+ signature: m.role === 'assistant' ? m.signature : undefined,
684
+ signatureType: m.role === 'assistant' ? m.signature_type : undefined,
595
685
  },
596
686
  ),
597
687
  ),
@@ -609,6 +699,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
609
699
  // #7 request_type (varint enum)
610
700
  // #8 completion_configuration
611
701
  // #10 tools (repeated ChatToolDefinition)
702
+ // #13 prompt_cache_options
612
703
  // #16 cascade_id (string)
613
704
  // #21 chat_model_uid (string)
614
705
  // #22 prompt_id (string)
@@ -622,6 +713,13 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
622
713
  encodeVarintField(7, args.requestType ?? 5),
623
714
  encodeMessage(8, completion),
624
715
  ...toolParts,
716
+ // #13 prompt_cache_options: { type: EPHEMERAL }. Reusing a session id is only
717
+ // half of prompt caching — without this the server creates no cache entry and
718
+ // every turn re-reads the whole prefix, which is why the sessionId reuse above
719
+ // was not producing the hit ratio its comment claims. The native client sends
720
+ // it and records real savings; sending it unconditionally matches both the
721
+ // native client and CLIProxyAPIPlus, which places it outside its tools gate.
722
+ encodeMessage(13, encodeVarintField(1, PROMPT_CACHE_EPHEMERAL)),
625
723
  // #15 session model config: { id, turn, 4 }. Present on every verified
626
724
  // request.
627
725
  encodeMessage(15, Buffer.concat([
@@ -669,7 +767,30 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
669
767
  * any non-zero to 'tool_calls' for now (and let the caller fall back to
670
768
  * 'stop' if no tool_call deltas were emitted).
671
769
  */
672
- function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
770
+ export function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
771
+ // Field 7 is `ModelUsageStats`, the authoritative per-turn accounting, and
772
+ // field 28 is `response_dimension_groups` — the rows the IDE renders. The
773
+ // decoder below reads 28 because a capture happened to expose metric-looking
774
+ // strings there (`ResponseDimension.uid` is its field 5, which is what the
775
+ // entry walker treats as `metric_id`), and that works only when the service
776
+ // chose to render cache rows. Field 7 carries cache read and cache write
777
+ // unconditionally, which is why a cached Devin turn used to report a bare
778
+ // total with no cached subset.
779
+ //
780
+ // Both fields arrive in the same message and the adapter keeps the last usage
781
+ // event it sees, so this cannot be a plain "decode both": field 7 has to
782
+ // suppress field 28 within the message. It is yielded before the rest of the
783
+ // frame rather than after it, so a frame that also carries finish (field 5)
784
+ // still reports usage ahead of the turn's end, and the order does not depend
785
+ // on where the service happens to place the field.
786
+ let authoritativeUsage: CloudChatEvent | null = null;
787
+ for (const f of iterFields(proto)) {
788
+ if (f.num === 7 && f.wire === 2 && Buffer.isBuffer(f.value)) {
789
+ authoritativeUsage = decodeModelUsageStats(f.value as Buffer);
790
+ if (authoritativeUsage) break;
791
+ }
792
+ }
793
+ if (authoritativeUsage) yield authoritativeUsage;
673
794
  for (const f of iterFields(proto)) {
674
795
  if (f.num === 3 && f.wire === 2 && Buffer.isBuffer(f.value)) {
675
796
  // Visible delta_text — what the user should SEE in the chat.
@@ -693,6 +814,9 @@ function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
693
814
  // block instead of inline with the answer.
694
815
  const s = (f.value as Buffer).toString('utf8');
695
816
  if (s) yield { kind: 'reasoning', text: s };
817
+ } else if (f.num === 10 && f.wire === 2 && Buffer.isBuffer(f.value)) {
818
+ const s = (f.value as Buffer).toString('utf8');
819
+ if (s) yield { kind: 'reasoning_signature', signature: s };
696
820
  } else if (f.num === 6 && f.wire === 2 && Buffer.isBuffer(f.value)) {
697
821
  let id: string | undefined;
698
822
  let name: string | undefined;
@@ -740,6 +864,7 @@ function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
740
864
  // else stays 'stop' for 0/2/4-9/12/13
741
865
  yield { kind: 'finish', reason };
742
866
  } else if (f.num === 28 && f.wire === 2 && Buffer.isBuffer(f.value)) {
867
+ if (authoritativeUsage) continue;
743
868
  const usage = decodeUsageBlock(f.value as Buffer);
744
869
  if (usage) yield usage;
745
870
  }
@@ -834,6 +959,68 @@ function decodeUsageBlock(buf: Buffer): CloudChatEvent | null {
834
959
  };
835
960
  }
836
961
 
962
+ /**
963
+ * `exa.codeium_common_pb.ModelUsageStats` at GetChatMessageResponse field 7.
964
+ *
965
+ * ModelUsageStats {
966
+ * #2 input_tokens uint64
967
+ * #3 output_tokens uint64
968
+ * #4 cache_write_tokens uint64
969
+ * #5 cache_read_tokens uint64
970
+ * }
971
+ *
972
+ * Plain varints, so the field-28 entry walker — which descends a
973
+ * length-delimited sub-message and reads a fixed32 float — cannot read this at
974
+ * all. It needs its own decoder.
975
+ *
976
+ * Whether Cognition's `input_tokens` already includes the cached tokens is not
977
+ * settled. oh-my-pi sums all four into its total, which suggests exclusive, but
978
+ * that is their convention rather than a measurement of this field. Guessing
979
+ * wrong in the inclusive direction is the expensive mistake: `normalizeCostTokens`
980
+ * only rejects `read + write > input`, so an inflated input passes validation and
981
+ * bills cached tokens at the uncached rate.
982
+ *
983
+ * So the shape is derived from the frame instead of assumed. An input that
984
+ * already covers the cache is left alone; one that cannot possibly cover it is
985
+ * folded. Both branches agree on the case that motivated this — a 58k prompt
986
+ * that is 57k cache read and 1k fresh reads as 58k with a 57k cached subset —
987
+ * and neither can emit `read + write > input`. Replace the derivation with a
988
+ * fixed mapping once a live frame settles the question.
989
+ */
990
+ export function decodeModelUsageStats(buf: Buffer): CloudChatEvent | null {
991
+ let wireInput: number | undefined;
992
+ let output: number | undefined;
993
+ let cacheWrite: number | undefined;
994
+ let cacheRead: number | undefined;
995
+ for (const f of iterFields(buf)) {
996
+ if (f.wire !== 0) continue;
997
+ const n = Number(f.value);
998
+ if (!Number.isFinite(n) || n < 0) continue;
999
+ if (f.num === 2) wireInput = n;
1000
+ else if (f.num === 3) output = n;
1001
+ else if (f.num === 4) cacheWrite = n;
1002
+ else if (f.num === 5) cacheRead = n;
1003
+ }
1004
+ if (wireInput === undefined && output === undefined && cacheRead === undefined && cacheWrite === undefined) {
1005
+ return null;
1006
+ }
1007
+ const read = cacheRead ?? 0;
1008
+ const write = cacheWrite ?? 0;
1009
+ const rawInput = wireInput ?? 0;
1010
+ const promptTokens = rawInput >= read + write ? rawInput : rawInput + read + write;
1011
+ const completionTokens = output ?? 0;
1012
+ const total = promptTokens + completionTokens;
1013
+ return {
1014
+ kind: 'usage',
1015
+ promptTokens,
1016
+ completionTokens,
1017
+ totalTokens: total > 0 ? total : undefined,
1018
+ cachedInputTokens: cacheRead,
1019
+ cacheCreationInputTokens: cacheWrite,
1020
+ reasoningTokens: undefined,
1021
+ };
1022
+ }
1023
+
837
1024
  // ----------------------------------------------------------------------------
838
1025
  // Public API: streamChat
839
1026
  // ----------------------------------------------------------------------------
@@ -864,7 +1051,18 @@ export interface CloudChatRequest {
864
1051
  }
865
1052
 
866
1053
  export class CloudChatError extends Error {
867
- constructor(message: string, public readonly code?: string, public readonly traceId?: string) {
1054
+ constructor(
1055
+ message: string,
1056
+ public readonly code?: string,
1057
+ public readonly traceId?: string,
1058
+ /**
1059
+ * Upstream HTTP status, when the failure was a status line rather than a
1060
+ * Connect trailer. Without it the adapter's message reaches
1061
+ * `inferHttpStatusFromAdapterMessage`, which does not parse `HTTP 429`, so
1062
+ * a live rate limit was classified 502 and core's failover never rotated.
1063
+ */
1064
+ public readonly status?: number,
1065
+ ) {
868
1066
  super(message);
869
1067
  this.name = 'CloudChatError';
870
1068
  }
@@ -872,6 +1070,45 @@ export class CloudChatError extends Error {
872
1070
 
873
1071
  const TRACE_ID_RE = /\(trace ID: ([0-9a-f]+)\)/i;
874
1072
 
1073
+ /**
1074
+ * A quota refusal Cognition delivers as `permission_denied`.
1075
+ *
1076
+ * "Your limit will reset in 13 minutes" and "Reached overall message rate
1077
+ * limit" are caps, not authorization failures. Classified as 403 they invite
1078
+ * the client to retry straight into a live cap; as 429 the proxy backs off and
1079
+ * can rotate.
1080
+ */
1081
+ const TRAILER_QUOTA_RE = /\b(?:limit will reset|rate limit|quota exceeded|out of credits)\b/i;
1082
+
1083
+ /**
1084
+ * Connect error code to HTTP status.
1085
+ *
1086
+ * Without this only the HTTP status line reached the adapter, so a cap or an
1087
+ * expired credential delivered as an EOS trailer fell through to
1088
+ * `inferHttpStatusFromAdapterMessage` and became a generic 502 — which is not
1089
+ * retryable-with-backoff, not an auth prompt, and not something core's failover
1090
+ * acts on.
1091
+ */
1092
+ export function connectTrailerHttpStatus(code: string | undefined, message: string): number | undefined {
1093
+ if (code === 'permission_denied' && TRAILER_QUOTA_RE.test(message)) return 429;
1094
+ switch (code) {
1095
+ case 'unauthenticated': return 401;
1096
+ case 'permission_denied': return 403;
1097
+ case 'resource_exhausted': return 429;
1098
+ case 'not_found': return 404;
1099
+ case 'unavailable': return 503;
1100
+ case 'deadline_exceeded': return 504;
1101
+ case 'unimplemented': return 501;
1102
+ case 'invalid_argument':
1103
+ case 'failed_precondition':
1104
+ case 'out_of_range': return 400;
1105
+ case 'internal':
1106
+ case 'unknown':
1107
+ case 'data_loss': return 502;
1108
+ default: return undefined;
1109
+ }
1110
+ }
1111
+
875
1112
  /**
876
1113
  * Stream chat events from the cloud. Yields CloudChatEvent (text deltas, tool
877
1114
  * call deltas, finish reason). Use `streamChatText` for legacy text-only iteration.
@@ -941,12 +1178,24 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
941
1178
  const framed = frameConnectStream(proto, false);
942
1179
  const body = new Blob([new Uint8Array(framed)], { type: "application/connect+proto" });
943
1180
 
944
- // Compose caller signal with a TTFB timeout. If the cloud takes longer
945
- // than CLOUD_STREAM_TTFB_MS to start the response, abort. Once any byte
946
- // arrives we cancel the TTFB timer and start the per-chunk idle timer
947
- // inside the read loop instead.
1181
+ // Compose the caller signal with a deadline on the response HEADERS. The
1182
+ // timer is cleared in the finally below, which runs when `await fetch`
1183
+ // resolves — and fetch resolves on headers, not on the first body byte. An
1184
+ // earlier comment here claimed "once any byte arrives", which was wrong and
1185
+ // hid the defect: Cognition withholds headers until the first token, so this
1186
+ // budget is a generation deadline. Body silence after headers is a separate
1187
+ // budget, the per-chunk idle timer in the read loop below.
948
1188
  const ttfbController = new AbortController();
949
- const ttfbTimer = setTimeout(() => ttfbController.abort(new Error(`cloud-direct: time-to-first-byte timeout (${CLOUD_STREAM_TTFB_MS}ms)`)), CLOUD_STREAM_TTFB_MS);
1189
+ const headersMs = cloudStreamHeadersMs();
1190
+ // Abort with no reason and remember that we are the one who fired. Bun rejects
1191
+ // the fetch with its own AbortError rather than handing back `signal.reason`,
1192
+ // so attaching a typed error to abort() would be discarded; the catch below is
1193
+ // what actually produces a classifiable failure.
1194
+ let headersDeadlineFired = false;
1195
+ const ttfbTimer = setTimeout(() => {
1196
+ headersDeadlineFired = true;
1197
+ ttfbController.abort();
1198
+ }, headersMs);
950
1199
  const ttfbSignal = ttfbController.signal;
951
1200
  // Compose req.signal + ttfbSignal. AbortSignal.any was added in Node
952
1201
  // 20.3 / Bun 1.0; our `engines` allows Node ≥18, so on Node 18-20.2 the
@@ -974,7 +1223,28 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
974
1223
  body,
975
1224
  redirect: 'error',
976
1225
  signal: initialSignal,
977
- });
1226
+ // Bun applies its own fetch idle timeout (~5 minutes) on top of ours.
1227
+ // Two independent deadlines on the same hop means the shorter one wins
1228
+ // silently and this function can no longer explain its own failure, so
1229
+ // the deadline above is made the single authority. Same reason as
1230
+ // src/server/responses/fetch-helpers.ts.
1231
+ timeout: 0,
1232
+ } as RequestInit);
1233
+ } catch (err) {
1234
+ if (headersDeadlineFired) {
1235
+ // Ours, not the upstream failing. Raised as a typed error with an explicit
1236
+ // status because devinErrorClassification reads CloudChatError.status and
1237
+ // would otherwise return {} for a bare Error, leaving src/lib/errors.ts to
1238
+ // guess from the message text. The message deliberately no longer says
1239
+ // "timeout", so the status is the only thing carrying the classification.
1240
+ throw new CloudChatError(
1241
+ `cloud-direct: no response headers within ${headersMs}ms`,
1242
+ undefined,
1243
+ undefined,
1244
+ 504,
1245
+ );
1246
+ }
1247
+ throw err;
978
1248
  } finally {
979
1249
  clearTimeout(ttfbTimer);
980
1250
  // The composed signal only guards the headers hop; the body is cancelled
@@ -987,7 +1257,11 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
987
1257
  // The body is not echoed into the message. This error reaches the adapter's
988
1258
  // error event and /api/logs, and a Connect error can quote the request that
989
1259
  // produced it - which is the request holding the api_key.
990
- throw new CloudChatError(`GetChatMessage failed (HTTP ${resp.status})`, undefined);
1260
+ //
1261
+ // The status line is carried on the error. A cap or an expired credential
1262
+ // delivered instead as a Connect EOS trailer is mapped by
1263
+ // connectTrailerHttpStatus at the trailer sites below.
1264
+ throw new CloudChatError(`GetChatMessage failed (HTTP ${resp.status})`, undefined, undefined, resp.status);
991
1265
  }
992
1266
  if (!resp.body) {
993
1267
  throw new CloudChatError('GetChatMessage response had no body stream');
@@ -1214,7 +1488,12 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
1214
1488
  `service accepts. If the request is unchanged and this is new, the ` +
1215
1489
  `account's model access is the next thing to check. ` +
1216
1490
  `(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
1217
- throw new CloudChatError(enriched, trailerError.code, trailerError.traceId);
1491
+ throw new CloudChatError(
1492
+ enriched,
1493
+ trailerError.code,
1494
+ trailerError.traceId,
1495
+ connectTrailerHttpStatus(trailerError.code, trailerError.message),
1496
+ );
1218
1497
  }
1219
1498
  // Cognition also returns `permission_denied` when a tool description
1220
1499
  // contains a blocklisted phrase that the sanitizer above did not catch
@@ -1234,9 +1513,19 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
1234
1513
  // was a phrase match at all.
1235
1514
  `(cloud message: ${trailerError.message}) ` +
1236
1515
  `(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
1237
- throw new CloudChatError(enriched, trailerError.code, trailerError.traceId);
1516
+ throw new CloudChatError(
1517
+ enriched,
1518
+ trailerError.code,
1519
+ trailerError.traceId,
1520
+ connectTrailerHttpStatus(trailerError.code, trailerError.message),
1521
+ );
1238
1522
  }
1239
- throw new CloudChatError(trailerError.message, trailerError.code, trailerError.traceId);
1523
+ throw new CloudChatError(
1524
+ trailerError.message,
1525
+ trailerError.code,
1526
+ trailerError.traceId,
1527
+ connectTrailerHttpStatus(trailerError.code, trailerError.message),
1528
+ );
1240
1529
  }
1241
1530
  // Truncation detection: the cloud always terminates a successful stream
1242
1531
  // with an EOS trailer. If we hit `done` from the body reader without one,
@@ -55,6 +55,33 @@ const CLOUD_CHAT_OS = 'windows';
55
55
  */
56
56
  const DEVICE_FINGERPRINT_BYTES = 366;
57
57
 
58
+ /** Prefix every Cognition session key carries in `Metadata.api_key`. */
59
+ const DEVIN_SESSION_TOKEN_PREFIX = 'devin-session-token$';
60
+
61
+ /** A bare JWT: three base64url segments. Nothing else is reshaped. */
62
+ const BARE_JWT_PATTERN = /^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]*$/;
63
+
64
+ /**
65
+ * Restore the `devin-session-token$` prefix on a bare JWT.
66
+ *
67
+ * Cognition reads `Metadata.api_key` as a prefixed session token. A key that
68
+ * arrives without the prefix — a JWT pasted into `apiKey` by hand, or one
69
+ * copied out of the CLI's file without its prefix — is sent verbatim and comes
70
+ * back as an opaque `permission_denied`, which reads as a revoked account
71
+ * rather than as a malformed credential.
72
+ *
73
+ * Only a bare JWT is reshaped. The other key formats this field has carried are
74
+ * not JWTs and must pass through untouched: a Codeium-classic bare UUID, an
75
+ * `sk-ws-01-…` Windsurf key, and a `cog_…` session key would all break if they
76
+ * were prefixed. Anything already containing `$` is left alone for the same
77
+ * reason.
78
+ */
79
+ export function normalizeDevinSessionToken(apiKey: string): string {
80
+ const trimmed = apiKey.trim();
81
+ if (!trimmed || trimmed.includes('$')) return apiKey;
82
+ return BARE_JWT_PATTERN.test(trimmed) ? `${DEVIN_SESSION_TOKEN_PREFIX}${trimmed}` : apiKey;
83
+ }
84
+
58
85
  export interface MetadataInput {
59
86
  /** Persistent api_key from OAuth (`devin-session-token$<JWT>`). */
60
87
  apiKey: string;
@@ -100,12 +127,14 @@ function osString(): string {
100
127
  export function buildMetadata(input: MetadataInput): Buffer {
101
128
  const version = input.windsurfVersion ?? WINDSURF_VERSION_STRING;
102
129
  const os = input.osName ?? osString();
130
+ // One boundary, so no caller has to remember the prefix rule.
131
+ const apiKey = normalizeDevinSessionToken(input.apiKey);
103
132
  if (input.cloudChatShape) {
104
133
  const clientVersion = input.windsurfVersion ?? CLOUD_CHAT_CLIENT_VERSION;
105
134
  return Buffer.concat([
106
135
  encodeString(1, CLOUD_CHAT_CLIENT_NAME),
107
136
  encodeString(2, clientVersion),
108
- encodeString(3, input.apiKey),
137
+ encodeString(3, apiKey),
109
138
  encodeString(4, 'en'),
110
139
  encodeString(5, input.osName ?? CLOUD_CHAT_OS),
111
140
  encodeString(7, clientVersion),
@@ -117,7 +146,7 @@ export function buildMetadata(input: MetadataInput): Buffer {
117
146
  const parts: Buffer[] = [
118
147
  encodeString(1, 'windsurf'), // ide_name
119
148
  encodeString(2, version), // extension_version
120
- encodeString(3, input.apiKey), // api_key
149
+ encodeString(3, apiKey), // api_key
121
150
  encodeString(4, 'en'), // locale
122
151
  encodeString(5, os), // os
123
152
  encodeString(7, version), // ide_version
@@ -72,7 +72,7 @@ export const DEVIN_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
72
72
  * Trailing tokens that the Cognition catalog appends as effort/variant
73
73
  * suffixes. Stripped to collapse suffixed UIDs to their base id.
74
74
  */
75
- const EFFORT_TOKENS = new Set([
75
+ export const EFFORT_TOKENS = new Set([
76
76
  "low", "medium", "high", "xhigh", "max", "none", "fast", "priority", "1m",
77
77
  ]);
78
78
 
@@ -85,8 +85,61 @@ export function collapseDevinModelUid(uid: string): string {
85
85
  return parts.join("-");
86
86
  }
87
87
 
88
+ /**
89
+ * The subset of catalog suffix tokens that are reasoning rungs.
90
+ *
91
+ * `fast`, `priority` and `1m` are service tiers and context variants, not effort.
92
+ * Offering them on a reasoning control would name a setting that does something
93
+ * else, so the collapse keeps stripping them while the ladder ignores them.
94
+ */
95
+ const REASONING_RUNG_TOKENS = new Set(["none", "low", "medium", "high", "xhigh", "max"]);
96
+
97
+ /**
98
+ * The reasoning rungs a catalog UID carries, in ladder order.
99
+ *
100
+ * Cognition spells effort as a suffix on the model id, so the variants an account
101
+ * actually has ARE its ladder — and collapseDevinModelUid() was throwing exactly
102
+ * that evidence away. Reading it back is what lets every model advertise the rungs
103
+ * it can really run instead of inheriting the generic six-rung default.
104
+ */
105
+ export function devinReasoningRungsOf(uid: string): string[] {
106
+ const parts = uid.split("-");
107
+ const rungs: string[] = [];
108
+ while (parts.length > 1 && EFFORT_TOKENS.has(parts[parts.length - 1]!)) {
109
+ const token = parts.pop()!;
110
+ if (REASONING_RUNG_TOKENS.has(token)) rungs.push(token);
111
+ }
112
+ return rungs;
113
+ }
114
+
115
+ /** Ladder order for display, matching the Codex rung order. */
116
+ const RUNG_ORDER = ["none", "low", "medium", "high", "xhigh", "max"];
117
+ export function sortDevinRungs(rungs: Iterable<string>): string[] {
118
+ return [...new Set(rungs)].sort((a, b) => RUNG_ORDER.indexOf(a) - RUNG_ORDER.indexOf(b));
119
+ }
120
+
121
+ /**
122
+ * Degraded-mode ladders, used only before the account catalog is readable.
123
+ *
124
+ * Only measured entries belong here. SWE-2 ships exactly three native lanes
125
+ * (see SWE2_EFFORT in src/adapters/devin.ts); inventing ladders for the rest
126
+ * would advertise rungs nobody verified, and the live catalog replaces this
127
+ * table as soon as a credential is present.
128
+ */
129
+ export const DEVIN_MODEL_EFFORTS: Record<string, string[]> = {
130
+ "swe-2": ["medium", "high", "max"],
131
+ };
132
+
133
+ /**
134
+ * Provider-level fallback ladder. No `ultra`: Cognition has no such lane, and
135
+ * the Codex catalog re-adds its own top rungs anyway (src/codex/catalog/effort.ts).
136
+ * Clients that key an effort control off this list — the Pi-shaped exports — get
137
+ * a control instead of none.
138
+ */
139
+ export const DEVIN_DEFAULT_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
140
+
88
141
  export type DevinUsableModelsResult =
89
- | { ok: true; models: string[]; contextWindows: Record<string, number> }
142
+ | { ok: true; models: string[]; contextWindows: Record<string, number>; efforts: Record<string, string[]> }
90
143
  | { ok: false; error: "auth" | "http" | "empty" | "unknown"; detail?: string };
91
144
 
92
145
  /**
@@ -105,6 +158,8 @@ export async function fetchDevinUsableModels(opts: {
105
158
  if (!catalog) return { ok: false, error: "empty" };
106
159
  const bases = new Set<string>();
107
160
  const contextWindows: Record<string, number> = {};
161
+ // Effort rungs per base, recovered from the suffixes the collapse strips.
162
+ const rungs = new Map<string, Set<string>>();
108
163
  for (const entry of catalog.byUid.values()) {
109
164
  if (entry.disabled) continue;
110
165
  // Skip internal enum constants (e.g. MODEL_GPT_5_2_LOW, MODEL_PRIVATE_*).
@@ -112,6 +167,12 @@ export async function fetchDevinUsableModels(opts: {
112
167
  if (entry.modelUid.startsWith("MODEL_")) continue;
113
168
  const base = collapseDevinModelUid(entry.modelUid);
114
169
  bases.add(base);
170
+ const found = devinReasoningRungsOf(entry.modelUid);
171
+ if (found.length > 0) {
172
+ let set = rungs.get(base);
173
+ if (!set) { set = new Set(); rungs.set(base, set); }
174
+ for (const rung of found) set.add(rung);
175
+ }
115
176
  if (entry.contextWindow && entry.contextWindow > 0) {
116
177
  // Variants of one base can disagree: the opt-in `-1m` rows report a
117
178
  // larger window than the plain row of the same base, and both collapse
@@ -124,7 +185,13 @@ export async function fetchDevinUsableModels(opts: {
124
185
  }
125
186
  }
126
187
  if (bases.size === 0) return { ok: false, error: "empty" };
127
- return { ok: true, models: [...bases].sort(), contextWindows };
188
+ const efforts: Record<string, string[]> = {};
189
+ for (const [base, set] of rungs) {
190
+ // A single rung is not a choice, so it is not a control. Advertising one
191
+ // would draw a picker whose only option is the value already in effect.
192
+ if (set.size > 1) efforts[base] = sortDevinRungs(set);
193
+ }
194
+ return { ok: true, models: [...bases].sort(), contextWindows, efforts };
128
195
  } catch (error) {
129
196
  const message = error instanceof Error ? error.message : String(error);
130
197
  if (/unauth|401|invalid token|login/i.test(message)) return { ok: false, error: "auth", detail: message };