@bitkyc08/opencodex 2.58.0 → 2.59.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. package/README.md +28 -10
  2. package/gui/dist/assets/index-C5IebErG.js +136 -0
  3. package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/gui/dist/provider-icons/crusoe.svg +1 -0
  6. package/gui/dist/provider-icons/opper.svg +3 -0
  7. package/package.json +1 -1
  8. package/src/adapters/base.ts +11 -1
  9. package/src/adapters/cursor/catalog.ts +11 -0
  10. package/src/adapters/cursor/effort-map.ts +16 -2
  11. package/src/adapters/cursor/envelope-echo.ts +55 -2
  12. package/src/adapters/cursor/message-mapper.ts +3 -2
  13. package/src/adapters/cursor/protobuf-request.ts +8 -5
  14. package/src/adapters/cursor/request-builder.ts +14 -3
  15. package/src/adapters/cursor/thread-continuity.ts +105 -31
  16. package/src/adapters/cursor/tool-guidance.ts +5 -4
  17. package/src/adapters/cursor.ts +42 -1
  18. package/src/adapters/devin/cloud-direct/chat.ts +11 -2
  19. package/src/adapters/devin/cloud-direct/index.ts +7 -0
  20. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
  21. package/src/adapters/devin.ts +75 -13
  22. package/src/adapters/google-antigravity-wire.ts +29 -2
  23. package/src/adapters/google-http.ts +8 -1
  24. package/src/adapters/google.ts +23 -4
  25. package/src/adapters/openai-chat/response-events.ts +61 -0
  26. package/src/adapters/openai-chat.ts +5 -10
  27. package/src/adapters/openai-responses/passthrough.ts +10 -1
  28. package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
  29. package/src/adapters/openai-responses/tool-schema.ts +19 -7
  30. package/src/adapters/responses-tool-schema.ts +76 -46
  31. package/src/adapters/run-turn-queue.ts +17 -4
  32. package/src/bridge/response-json.ts +1 -1
  33. package/src/bridge/sse.ts +165 -24
  34. package/src/claude/context-windows.ts +22 -0
  35. package/src/claude/outbound.ts +35 -4
  36. package/src/cli/account-api.ts +4 -3
  37. package/src/cli/account-extended.ts +22 -2
  38. package/src/cli/account-orca-import.ts +63 -0
  39. package/src/cli/account.ts +32 -4
  40. package/src/cli/capabilities.ts +40 -0
  41. package/src/cli/claude.ts +29 -1
  42. package/src/cli/codex-cli-update.ts +97 -2
  43. package/src/cli/dispatch.ts +54 -0
  44. package/src/cli/doctor.ts +197 -2
  45. package/src/cli/help.ts +4 -1
  46. package/src/cli/index.ts +88 -20
  47. package/src/cli/models-runtime.ts +33 -4
  48. package/src/cli/registry.ts +11 -1
  49. package/src/cli/runtime-api.ts +44 -0
  50. package/src/cli/start-args.ts +94 -0
  51. package/src/cli/system-command.ts +2 -0
  52. package/src/client/machine-api.ts +4 -3
  53. package/src/client/machine-listener.ts +14 -1
  54. package/src/clients/config-export/constants.ts +2 -3
  55. package/src/clients/config-export.ts +5 -5
  56. package/src/codex/account-store.ts +81 -5
  57. package/src/codex/auth-api/pool-quota-probe.ts +14 -3
  58. package/src/codex/auth-api/routes.ts +17 -2
  59. package/src/codex/auth-context.ts +16 -12
  60. package/src/codex/catalog/build-entries.ts +25 -4
  61. package/src/codex/catalog/derive-entry.ts +8 -1
  62. package/src/codex/catalog/effort.ts +10 -6
  63. package/src/codex/catalog/gather-capture.ts +1 -0
  64. package/src/codex/catalog/model-hints.ts +37 -5
  65. package/src/codex/catalog/parsing.ts +83 -5
  66. package/src/codex/catalog/reserve-warn.ts +96 -0
  67. package/src/codex/catalog/retained-sync.ts +19 -0
  68. package/src/codex/catalog/routed-gather.ts +42 -3
  69. package/src/codex/cli-installation-identity.ts +210 -0
  70. package/src/codex/cli-installation-targets.ts +158 -0
  71. package/src/codex/convergence.ts +5 -0
  72. package/src/codex/history-provider.ts +4 -1
  73. package/src/codex/history-state-open.ts +105 -0
  74. package/src/codex/inject/config-toml.ts +44 -2
  75. package/src/codex/inject.ts +3 -2
  76. package/src/codex/lineage.ts +83 -32
  77. package/src/codex/loopback-target.ts +31 -0
  78. package/src/codex/main-account-hard-lock.ts +2 -1
  79. package/src/codex/main-account.ts +10 -3
  80. package/src/codex/main-device-reauth.ts +17 -9
  81. package/src/codex/model-entitlements.ts +60 -1
  82. package/src/codex/observed-model-denials.ts +137 -0
  83. package/src/codex/orca-auth-source.ts +94 -0
  84. package/src/codex/orca-import.ts +219 -0
  85. package/src/codex/prompt-text-probe.ts +282 -12
  86. package/src/codex/quota-401-recovery.ts +12 -0
  87. package/src/codex/quota-types.ts +65 -0
  88. package/src/codex/quota.ts +24 -19
  89. package/src/codex/routing/cooldown-math.ts +8 -47
  90. package/src/codex/routing/pin-drain.ts +57 -0
  91. package/src/codex/routing.ts +13 -15
  92. package/src/codex/subagent-model-fallback.ts +94 -0
  93. package/src/codex/windows-installation-files.ts +224 -0
  94. package/src/combos/failover.ts +122 -5
  95. package/src/config/diagnostics.ts +21 -0
  96. package/src/config/load-degrade.ts +15 -0
  97. package/src/config/pending-teardown.ts +8 -0
  98. package/src/config/process-state.ts +36 -3
  99. package/src/config/provider-relative-send-path.ts +16 -0
  100. package/src/config/proxy-env.ts +23 -5
  101. package/src/config/schema/config-schema.ts +21 -0
  102. package/src/config/schema/leaf-validators.ts +64 -17
  103. package/src/generated/compatibility-version.json +235 -163
  104. package/src/generated/model-metadata.ts +1 -1
  105. package/src/lib/bounded-body.ts +4 -2
  106. package/src/lib/destination-policy.ts +48 -6
  107. package/src/lib/errors.ts +3 -15
  108. package/src/lib/local-destinations.ts +32 -5
  109. package/src/lib/provider-outbound.ts +3 -3
  110. package/src/lib/proxy-env.ts +70 -3
  111. package/src/lib/request-execution-budget.ts +11 -3
  112. package/src/lib/response-body-inactivity.ts +193 -0
  113. package/src/lib/retry-delay.ts +69 -0
  114. package/src/lib/socks5-fetch.ts +631 -0
  115. package/src/lib/spend-reservation-ledger.ts +115 -9
  116. package/src/lib/workflow-budget.ts +145 -8
  117. package/src/oauth/account-quota-rank.ts +72 -15
  118. package/src/oauth/generic-account-failover.ts +40 -27
  119. package/src/oauth/orcarouter.ts +15 -2
  120. package/src/oauth/store.ts +8 -0
  121. package/src/providers/codex-capacity.ts +9 -0
  122. package/src/providers/devin-provider-merge-migration.ts +33 -12
  123. package/src/providers/free-directory.ts +20 -2
  124. package/src/providers/key-failover.ts +261 -7
  125. package/src/providers/model-rename-migration.ts +1 -0
  126. package/src/providers/openai-sidecar.ts +4 -0
  127. package/src/providers/opencode-go-transport.ts +14 -5
  128. package/src/providers/quota/report-cache.ts +3 -0
  129. package/src/providers/registry/entries-extended.ts +96 -0
  130. package/src/providers/registry/model-seeds.ts +78 -21
  131. package/src/responses/apply-patch-envelope.ts +44 -11
  132. package/src/responses/bridge-search-replay-cache.ts +152 -0
  133. package/src/responses/code-mode-helper-compat.ts +26 -16
  134. package/src/responses/custom-tool-compat.ts +1 -1
  135. package/src/responses/hosted-tool-policy.ts +85 -2
  136. package/src/responses/schema.ts +9 -2
  137. package/src/server/auth-cors.ts +26 -0
  138. package/src/server/chat-completions.ts +9 -4
  139. package/src/server/chat-native-sse.ts +26 -9
  140. package/src/server/chat-native.ts +10 -4
  141. package/src/server/claude-messages.ts +24 -2
  142. package/src/server/gui-static.ts +36 -2
  143. package/src/server/inbound-body-admission.ts +187 -0
  144. package/src/server/index.ts +15 -19
  145. package/src/server/management/api-access.ts +3 -4
  146. package/src/server/management/config-routes.ts +31 -6
  147. package/src/server/management/provider-capability-config.ts +35 -7
  148. package/src/server/management/provider-routes.ts +70 -18
  149. package/src/server/proxy-liveness.ts +97 -2
  150. package/src/server/relay.ts +17 -24
  151. package/src/server/request-log.ts +25 -1
  152. package/src/server/responses/adapter-continuation.ts +71 -27
  153. package/src/server/responses/adapter-delivery.ts +39 -8
  154. package/src/server/responses/adapter-dispatch.ts +52 -24
  155. package/src/server/responses/compact.ts +60 -11
  156. package/src/server/responses/core-codex-account.ts +83 -22
  157. package/src/server/responses/core-normalize.ts +12 -5
  158. package/src/server/responses/fetch-helpers.ts +68 -2
  159. package/src/server/responses/passthrough-delivery.ts +10 -1
  160. package/src/server/responses/passthrough-dispatch.ts +113 -48
  161. package/src/server/responses/passthrough-execution.ts +11 -1
  162. package/src/server/responses/request-prepare.ts +29 -0
  163. package/src/server/responses/request-send-budget.ts +84 -7
  164. package/src/server/responses/request-sidecar-auth.ts +16 -8
  165. package/src/server/responses/request-spend.ts +38 -9
  166. package/src/server/responses/request-transport.ts +13 -10
  167. package/src/server/responses/run-turn-execution.ts +20 -5
  168. package/src/server/responses/sidecar-execution.ts +2 -0
  169. package/src/server/responses/ws-upstream.ts +2 -1
  170. package/src/server/responses-custom-tool-repair.ts +2 -2
  171. package/src/server/sse-frame-buffer.ts +12 -10
  172. package/src/server/sse-payload-rewrite.ts +36 -9
  173. package/src/server/system-env-shell.ts +5 -1
  174. package/src/server/system-env.ts +7 -1
  175. package/src/server/workflow-refusal.ts +56 -2
  176. package/src/service/cli.ts +16 -6
  177. package/src/service/guards.ts +10 -0
  178. package/src/service/health.ts +43 -0
  179. package/src/service/state.ts +7 -2
  180. package/src/types/accounts.ts +4 -0
  181. package/src/types/config.ts +100 -3
  182. package/src/types/provider.ts +19 -0
  183. package/src/types/request.ts +7 -1
  184. package/src/types/wire.ts +9 -1
  185. package/src/usage/expected-prices.ts +28 -0
  186. package/src/usage/log.ts +87 -4
  187. package/src/web-search/passthrough-bridge.ts +39 -5
  188. package/gui/dist/assets/index-BbrHOIY0.js +0 -128
@@ -29,6 +29,243 @@ interface KeyCooldown {
29
29
  const DEFAULT_COOLDOWN_MS = 60_000;
30
30
  const MAX_COOLDOWN_MS = 10 * 60_000; // cap at 10 min for api-key rotation
31
31
 
32
+ /**
33
+ * Cap for a cooldown the upstream itself dated, as opposed to one we inferred.
34
+ *
35
+ * `MAX_COOLDOWN_MS` is deliberately short because an undated 429 is a guess: ten
36
+ * minutes bounds how long a transient limit can park a working key. A free-tier
37
+ * quota is not a guess — OpenRouter replies `Weekly/Monthly Limit Exhausted ...
38
+ * will reset at <date>`, and until that date the key cannot serve anything. Held
39
+ * for ten minutes instead, it comes back, takes a 429, and rotates again, every
40
+ * ten minutes for the rest of the week (#4024).
41
+ *
42
+ * 32 days rather than unbounded. The wording this parses is
43
+ * `Weekly/Monthly Limit Exhausted`, so the cap has to clear a monthly window —
44
+ * 31 days plus a day of slack for timezone and month length. An earlier 8-day
45
+ * cap looked generous against the weekly case in the issue and silently clamped
46
+ * every monthly reset to ~23 days early, which puts the key back into exactly
47
+ * the 429 loop this exists to stop. Caught by the cap's own test.
48
+ *
49
+ * Bounded at all because the date is upstream-controlled input: a malformed or
50
+ * hostile `reset at 2999-01-01` must not park a working key past any horizon an
51
+ * operator would think to look at.
52
+ */
53
+ const MAX_QUOTA_COOLDOWN_MS = 32 * 24 * 60 * 60_000;
54
+ const QUOTA_RESET_PEEK_TIMEOUT_MS = 250;
55
+
56
+ interface QuotaResetReadOptions {
57
+ now?: number;
58
+ signal?: AbortSignal;
59
+ timeoutMs?: number;
60
+ }
61
+
62
+ function rebuiltResponse(response: Response, body: ReadableStream<Uint8Array>): Response {
63
+ return new Response(body, {
64
+ status: response.status,
65
+ statusText: response.statusText,
66
+ headers: response.headers,
67
+ });
68
+ }
69
+
70
+ function replayOnlyResponse(response: Response, chunks: readonly Uint8Array[]): Response {
71
+ return rebuiltResponse(response, new ReadableStream<Uint8Array>({
72
+ start(controller) {
73
+ for (const chunk of chunks) controller.enqueue(chunk);
74
+ controller.close();
75
+ },
76
+ }));
77
+ }
78
+
79
+ /**
80
+ * Read a bounded prefix of a 429 body and pull the upstream's declared reset instant.
81
+ *
82
+ * The returned response replays the bounded prefix and any boundary-chunk overflow
83
+ * before streaming the unread remainder. A rotation storm must not be gated on
84
+ * reading N full error payloads. Any failure — no body, already consumed, slow,
85
+ * malformed — returns undefined and leaves the `Retry-After` path in charge.
86
+ */
87
+ export async function readQuotaResetAt(
88
+ response: Response,
89
+ nowOrOptions: number | QuotaResetReadOptions = {},
90
+ ): Promise<{ at: number | undefined; response: Response }> {
91
+ if (!response.body) return { at: undefined, response };
92
+ const options = typeof nowOrOptions === "number" ? { now: nowOrOptions } : nowOrOptions;
93
+ const now = options.now ?? Date.now();
94
+ let reader: ReadableStreamDefaultReader<Uint8Array>;
95
+ try {
96
+ reader = response.body.getReader();
97
+ } catch {
98
+ return { at: undefined, response };
99
+ }
100
+ const chunks: Uint8Array[] = [];
101
+ let transferred = false;
102
+ let timer: ReturnType<typeof setTimeout> | undefined;
103
+ const deadline = new AbortController();
104
+ const timeoutReason = new DOMException("Quota reset body peek timed out", "TimeoutError");
105
+ try {
106
+ const decoder = new TextDecoder();
107
+ let seen = 0;
108
+ let text = "";
109
+ timer = setTimeout(() => deadline.abort(timeoutReason), options.timeoutMs ?? QUOTA_RESET_PEEK_TIMEOUT_MS);
110
+ const signal = options.signal
111
+ ? AbortSignal.any([options.signal, deadline.signal])
112
+ : deadline.signal;
113
+ while (seen < QUOTA_RESET_SCAN_BYTES) {
114
+ const read = reader.read();
115
+ let rejectAbort: ((reason: unknown) => void) | undefined;
116
+ const onAbort = () => rejectAbort?.(signal.reason);
117
+ const aborted = new Promise<never>((_resolve, reject) => {
118
+ rejectAbort = reject;
119
+ if (signal.aborted) reject(signal.reason);
120
+ else signal.addEventListener("abort", onAbort, { once: true });
121
+ });
122
+ // Derived from the reader rather than named directly: Bun's lib types
123
+ // `ReadableStreamDefaultReader.read()` as returning
124
+ // `ReadableStreamDefaultReadResult`, which is not assignable to the
125
+ // `ReadableStreamReadResult` alias.
126
+ let result: Awaited<ReturnType<typeof reader.read>>;
127
+ try {
128
+ result = await Promise.race([read, aborted]);
129
+ } finally {
130
+ signal.removeEventListener("abort", onAbort);
131
+ }
132
+ const { done, value } = result;
133
+ if (done) break;
134
+ const remaining = QUOTA_RESET_SCAN_BYTES - seen;
135
+ const prefix = value.byteLength > remaining ? value.subarray(0, remaining) : value;
136
+ const overflow = value.byteLength > remaining ? value.subarray(remaining) : undefined;
137
+ chunks.push(prefix);
138
+ if (overflow?.byteLength) chunks.push(overflow);
139
+ seen += prefix.byteLength;
140
+ text += decoder.decode(prefix, { stream: true });
141
+ }
142
+ // Hand back a Response carrying the bytes already pulled followed by whatever
143
+ // is left, so the caller can still read or cancel it. `response.clone()` is
144
+ // NOT usable here: it tees, and with the original branch undrained the tee
145
+ // stalls once its buffer fills — a 5MB error body hangs the rotation path,
146
+ // which is worse than the unbounded read this replaced.
147
+ const rest = new ReadableStream<Uint8Array>({
148
+ start(controller) {
149
+ for (const c of chunks) controller.enqueue(c);
150
+ },
151
+ async pull(controller) {
152
+ try {
153
+ const { done, value } = await reader.read();
154
+ if (done) {
155
+ controller.close();
156
+ reader.releaseLock();
157
+ return;
158
+ }
159
+ controller.enqueue(value);
160
+ } catch (error) {
161
+ controller.error(error);
162
+ reader.releaseLock();
163
+ }
164
+ },
165
+ async cancel(reason) {
166
+ try {
167
+ await reader.cancel(reason);
168
+ } finally {
169
+ reader.releaseLock();
170
+ }
171
+ },
172
+ });
173
+ transferred = true;
174
+ return { at: parseQuotaResetAt(text, now), response: rebuiltResponse(response, rest) };
175
+ } catch (error) {
176
+ const clientAborted = options.signal?.aborted === true;
177
+ void reader.cancel(error).catch(() => {}).finally(() => {
178
+ try { reader.releaseLock(); } catch { /* already released */ }
179
+ });
180
+ if (clientAborted) throw options.signal!.reason ?? error;
181
+ return { at: undefined, response: replayOnlyResponse(response, chunks) };
182
+ } finally {
183
+ if (timer !== undefined) clearTimeout(timer);
184
+ if (!transferred) {
185
+ try { reader.releaseLock(); } catch { /* already released */ }
186
+ }
187
+ }
188
+ }
189
+
190
+ /**
191
+ * How much of a 429 body is read and scanned for the reset instant.
192
+ *
193
+ * Bounds the READ, not just the parse: this runs on the rotation path, once per
194
+ * rotated key under a rate-limit storm, and the body is upstream-controlled.
195
+ * OpenRouter's rate_limit_error JSON is a few hundred bytes.
196
+ */
197
+ const QUOTA_RESET_SCAN_BYTES = 4_096;
198
+
199
+ /**
200
+ * Reset instant an upstream declared in a 429 *body*, in epoch ms.
201
+ *
202
+ * Only the body carries this: OpenRouter sends no `Retry-After` for a quota
203
+ * exhaustion, so the header path (`parseRetryAfterMs`) sees nothing and falls
204
+ * back to `DEFAULT_COOLDOWN_MS`. Returns undefined for anything it cannot read
205
+ * as a date, so an unparsable body keeps today's behaviour exactly.
206
+ */
207
+ /**
208
+ * Whether `YYYY-MM-DD…` names a day that exists.
209
+ *
210
+ * `Date.parse` does NOT reject an out-of-range day: measured on Bun,
211
+ * `2026-02-30T00:00:00Z` yields March 2 and `2026-04-31T00:00:00Z` yields
212
+ * May 1, so a malformed upstream body would park a key past the instant it
213
+ * actually named. Only the month is rejected outright (`2026-13-01` is NaN).
214
+ *
215
+ * Checked on the date text alone rather than by round-tripping the parsed
216
+ * instant, because a value carrying an explicit offset (`…T23:00+05:30`)
217
+ * legitimately lands on a different UTC day than the one written.
218
+ */
219
+ function isRealCalendarDate(value: string): boolean {
220
+ const [year, month, day] = value.slice(0, 10).split("-").map(Number);
221
+ if (month < 1 || month > 12 || day < 1) return false;
222
+ const leap = (year % 4 === 0 && year % 100 !== 0) || year % 400 === 0;
223
+ const lengths = [31, leap ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31];
224
+ return day <= lengths[month - 1]!;
225
+ }
226
+
227
+ export function parseQuotaResetAt(body: string | null | undefined, now = Date.now()): number | undefined {
228
+ const text = body?.slice(0, QUOTA_RESET_SCAN_BYTES);
229
+ if (!text) return undefined;
230
+ let parsed: unknown;
231
+ try {
232
+ parsed = JSON.parse(text);
233
+ } catch {
234
+ return undefined;
235
+ }
236
+ if (!parsed || typeof parsed !== "object") return undefined;
237
+ const error = (parsed as { error?: unknown }).error;
238
+ if (!error || typeof error !== "object") return undefined;
239
+ const code = (error as { code?: unknown }).code;
240
+ const message = (error as { message?: unknown }).message;
241
+ if ((code !== "rate_limit_error" && code !== 429) || typeof message !== "string") return undefined;
242
+ if (!/^(?:Weekly|Monthly) Limit Exhausted\b/i.test(message.trim())) return undefined;
243
+ // `will reset at 2026-09-09 03:30:06` / `... at 2026-09-09T03:30:06Z` / `resets at <date>`
244
+ const match = /reset[s]?\s+at\s+([0-9]{4}-[0-9]{2}-[0-9]{2}(?:[T ][0-9]{2}:[0-9]{2}(?::[0-9]{2})?(?:\.[0-9]+)?(?:Z|[+-][0-9]{2}:?[0-9]{2})?)?)/i.exec(message);
245
+ if (!match) return undefined;
246
+ // Pin a bare `YYYY-MM-DD hh:mm:ss` to UTC explicitly.
247
+ //
248
+ // ECMA-262 says a date-TIME form with no offset is LOCAL time, and Node follows
249
+ // that: `Date.parse("2026-09-09 03:30:06")` differs from the UTC reading by the
250
+ // host offset (7h on a PDT box, measured). Bun currently returns the UTC value
251
+ // for the same string, so on this runtime the normalisation is a no-op today —
252
+ // which is exactly why it is written out rather than relied upon. If Bun ever
253
+ // conforms, an un-normalised parse would silently shift every park-until by the
254
+ // operator's offset, and the early direction resumes the 429 loop.
255
+ //
256
+ // A consequence worth knowing: no Bun test can observe this branch being
257
+ // removed. The explicit-zone case below is the part the suite can pin.
258
+ const raw = match[1].includes("T") || /(?:Z|[+-][0-9]{2}:?[0-9]{2})$/.test(match[1])
259
+ ? match[1]
260
+ : `${match[1].replace(" ", "T")}Z`;
261
+ if (!isRealCalendarDate(match[1])) return undefined;
262
+ const at = Date.parse(raw);
263
+ if (!Number.isFinite(at)) return undefined;
264
+ // Already past, or beyond the cap: not usable as a park-until instant.
265
+ if (at <= now) return undefined;
266
+ return Math.min(at, now + MAX_QUOTA_COOLDOWN_MS);
267
+ }
268
+
32
269
  /**
33
270
  * Default same-target 429 retry policy used when a provider opts in via a bare
34
271
  * `retryOn429: {}` (presence = opt-in with these defaults).
@@ -303,11 +540,11 @@ export function rateLimitRetryPolicyFor(
303
540
 
304
541
  /**
305
542
  * Normalize a provider's `transientRetryOn5xx` policy, or return null when it is absent,
306
- * explicitly disabled, not key-auth, or not the `openai-chat` adapter.
543
+ * explicitly disabled, not key-auth, or not an adapter this policy governs.
307
544
  *
308
- * The adapter gate is part of the accepted scope, not incidental: this first version covers
309
- * key-auth `openai-chat` only, and without an explicit check any generic key-auth adapter
310
- * could opt in. Auth mode follows the same fail-closed rule as `rateLimitRetryPolicyFor` —
545
+ * The adapter gate is part of the accepted scope, not incidental: it names the adapters whose
546
+ * lanes actually read this policy, so no generic key-auth adapter can opt in by accident.
547
+ * Auth mode follows the same fail-closed rule as `rateLimitRetryPolicyFor` —
311
548
  * explicit `key` or the documented omitted default, never OAuth, forward, local, or an
312
549
  * unknown value.
313
550
  */
@@ -316,7 +553,14 @@ export function transientRetryPolicyFor(
316
553
  ): Required<TransientRetryPolicy> | null {
317
554
  const policy = provider.transientRetryOn5xx;
318
555
  if (!policy || policy.enabled === false) return null;
319
- if (provider.adapter !== "openai-chat") return null;
556
+ // Both adapters this policy governs. The first version covered chat only, which left a
557
+ // key-auth Responses provider unable to tune its ladder in either direction, because the
558
+ // Responses passthrough lane hard-coded TRANSIENT_RETRY_MAX_ATTEMPTS (#4893). Widening this
559
+ // gate is necessary and not sufficient: the lane also has to call this function, which it
560
+ // now does. Still an explicit list, so no generic key-auth adapter opts in by accident, and
561
+ // the auth check below keeps the ChatGPT forward pool out -- those providers are
562
+ // `authMode: "forward"` and keep the default ladder they have always had.
563
+ if (provider.adapter !== "openai-chat" && provider.adapter !== "openai-responses") return null;
320
564
  if (provider.authMode !== undefined && provider.authMode !== "key") return null;
321
565
  return {
322
566
  enabled: policy.enabled ?? DEFAULT_TRANSIENT_RETRY.enabled,
@@ -363,6 +607,7 @@ function rotateKeyAfterFailure(
363
607
  now = Date.now(),
364
608
  attemptedKey?: string,
365
609
  attemptedSelection?: ProviderApiKeySelection,
610
+ quotaResetAt?: number,
366
611
  ): OcxProviderConfig | null {
367
612
  const provider = config.providers[providerName];
368
613
  if (!provider) return null;
@@ -421,7 +666,12 @@ function rotateKeyAfterFailure(
421
666
  // full cap instead of the 429 default so a dead key is not re-tried once a minute.
422
667
  const cooldownMs = failureStatus === 401
423
668
  ? MAX_COOLDOWN_MS
424
- : parseRetryAfterMs(retryAfterHeader, now) ?? DEFAULT_COOLDOWN_MS;
669
+ // A reset instant the upstream dated outranks both the header and the
670
+ // default: it is the only one of the three that knows when the quota
671
+ // actually returns (#4024).
672
+ : quotaResetAt !== undefined
673
+ ? Math.max(quotaResetAt - now, 1)
674
+ : parseRetryAfterMs(retryAfterHeader, now) ?? DEFAULT_COOLDOWN_MS;
425
675
  keyCooldowns.set(cooldownKey(providerName, outcome.value.failedId), { cooldownUntil: now + cooldownMs });
426
676
  sweepExpiredOnWrite(now);
427
677
  }
@@ -448,8 +698,9 @@ export function rotateKeyOn429(
448
698
  now = Date.now(),
449
699
  attemptedKey?: string,
450
700
  attemptedSelection?: ProviderApiKeySelection,
701
+ quotaResetAt?: number,
451
702
  ): OcxProviderConfig | null {
452
- return rotateKeyAfterFailure(config, providerName, 429, retryAfterHeader, now, attemptedKey, attemptedSelection);
703
+ return rotateKeyAfterFailure(config, providerName, 429, retryAfterHeader, now, attemptedKey, attemptedSelection, quotaResetAt);
453
704
  }
454
705
 
455
706
  /**
@@ -482,6 +733,8 @@ export function sweepExpiredApiKeyCooldowns(now = Date.now()): number {
482
733
 
483
734
  interface RotateProviderTransportOptions {
484
735
  retryAfter?: string | null;
736
+ /** Epoch ms from `parseQuotaResetAt`, when the upstream dated the reset in its body. */
737
+ quotaResetAt?: number;
485
738
  now?: number;
486
739
  attemptedKey?: string;
487
740
  attemptedSelection?: ProviderApiKeySelection;
@@ -507,6 +760,7 @@ export function rotateProviderTransportOn429(
507
760
  options.now,
508
761
  options.attemptedKey,
509
762
  options.attemptedSelection ?? routedProvider._apiKeyAttempt,
763
+ options.quotaResetAt,
510
764
  );
511
765
  if (!rotated) return null;
512
766
  return applyRotatedTransport(providerName, routedProvider, rotated, options.promptCacheKey);
@@ -86,6 +86,7 @@ const MODEL_KEYED_RECORDS = [
86
86
  "modelMaxOutputTokens",
87
87
  "modelInputModalities",
88
88
  "modelReasoningEfforts",
89
+ "modelSuppressSyntheticMax",
89
90
  "modelDefaultReasoningEfforts",
90
91
  "modelReasoningEffortMap",
91
92
  ] as const;
@@ -32,6 +32,8 @@ export interface ResolvedOpenAiForwardSidecar extends OpenAiForwardSidecarCandid
32
32
  authContext: CodexAuthContext;
33
33
  headers: Headers;
34
34
  recordOutcome?: (outcome: CodexUpstreamOutcome) => void;
35
+ /** Hand back an acquired recovery probe when no sidecar request reached upstream. */
36
+ releaseProbeLease?: () => void;
35
37
  }
36
38
 
37
39
  /**
@@ -189,6 +191,7 @@ export async function resolveFirstUsableOpenAiSidecar(
189
191
  ...(authContext.kind === "pool" ? { credentialGeneration: authContext.generation } : {}),
190
192
  },
191
193
  ),
194
+ releaseProbeLease: () => releaseCodexAuthContextProbeLease(authContext),
192
195
  };
193
196
  }
194
197
  if (candidate.accountMode === "direct") {
@@ -237,6 +240,7 @@ export async function resolveFirstUsableOpenAiSidecar(
237
240
  ...(authContext.kind === "pool" ? { credentialGeneration: authContext.generation } : {}),
238
241
  },
239
242
  ),
243
+ releaseProbeLease: () => releaseCodexAuthContextProbeLease(authContext),
240
244
  }
241
245
  : {}),
242
246
  };
@@ -12,10 +12,12 @@ function hasHeaderCaseInsensitive(
12
12
  return Object.keys(headers ?? {}).some(key => key.toLowerCase() === target);
13
13
  }
14
14
 
15
- /** Derive a provider-scoped opaque value without exposing Codex task or subagent ids. */
16
- export function deriveOpenCodeGoSessionId(sessionLane: string): string {
15
+ /** Derive a provider- and wire-scoped opaque value without exposing Codex task or subagent ids. */
16
+ export function deriveOpenCodeGoSessionId(sessionLane: string, wireProtocol: string): string {
17
17
  const digest = createHash("sha256")
18
- .update("opencodex/opencode-go/session/v1\0")
18
+ .update("opencodex/opencode-go/session/v2\0")
19
+ .update(wireProtocol)
20
+ .update("\0")
19
21
  .update(sessionLane)
20
22
  .digest("hex")
21
23
  .slice(0, 32);
@@ -30,12 +32,19 @@ export function deriveOpenCodeGoSessionId(sessionLane: string): string {
30
32
  * request reaching this helper from the proxy always carries a lane. The `!sessionLane` guard stays
31
33
  * for direct callers that have no request context; it is not a per-request identity of its own, and
32
34
  * minting one here would hand each retry a different value.
35
+ *
36
+ * `provider` is already settled onto the final wire, while `destinationProvider` is the original
37
+ * routed row. Keeping both explicit prevents an Anthropic hard pin from defeating the registry's
38
+ * adapter-sensitive destination recognition.
33
39
  */
34
40
  export function resolveOpenCodeGoTransport<T extends OcxProviderConfig>(
35
41
  provider: T,
36
42
  sessionLane: string | undefined,
43
+ destinationProvider:
44
+ Pick<OcxProviderConfig, "baseUrl" | "adapter">
45
+ & Partial<Pick<OcxProviderConfig, "authMode">>,
37
46
  ): T {
38
- if (registryEntryForProviderDestination(provider)?.id !== "opencode-go") return provider;
47
+ if (registryEntryForProviderDestination(destinationProvider)?.id !== "opencode-go") return provider;
39
48
  if (!sessionLane) return provider;
40
49
  if (hasHeaderCaseInsensitive(provider.headers, OPENCODE_GO_SESSION_HEADER)) return provider;
41
50
 
@@ -43,7 +52,7 @@ export function resolveOpenCodeGoTransport<T extends OcxProviderConfig>(
43
52
  ...provider,
44
53
  headers: {
45
54
  ...(provider.headers ?? {}),
46
- [OPENCODE_GO_SESSION_HEADER]: deriveOpenCodeGoSessionId(sessionLane),
55
+ [OPENCODE_GO_SESSION_HEADER]: deriveOpenCodeGoSessionId(sessionLane, provider.adapter),
47
56
  },
48
57
  };
49
58
  }
@@ -151,6 +151,9 @@ export function providerQuotaFromCodexQuota(
151
151
  const projected: CodexCapacityQuota = {
152
152
  ...(quota.shortPercent !== undefined ? { fiveHourPercent: quota.shortPercent } : {}),
153
153
  ...(quota.shortResetAt !== undefined ? { fiveHourResetAt: quota.shortResetAt } : {}),
154
+ // Freshness for the reset-less terminal rule. Without it the dashboard evaluates that rule
155
+ // with no evidence and returns null while routing refuses the same account (#5045).
156
+ ...(quota.shortObservedAt !== undefined ? { shortObservedAt: quota.shortObservedAt } : {}),
154
157
  ...(quota.weeklyPercent !== undefined ? { weeklyPercent: quota.weeklyPercent } : {}),
155
158
  ...(quota.weeklyResetAt !== undefined ? { weeklyResetAt: quota.weeklyResetAt } : {}),
156
159
  ...(quota.monthlyPercent !== undefined ? { monthlyPercent: quota.monthlyPercent } : {}),
@@ -98,6 +98,10 @@ import {
98
98
  DIGITALOCEAN_CHAT_COMPLETION_MODELS,
99
99
  SCALEWAY_SERVERLESS_CHAT_MODELS,
100
100
  SCALEWAY_MODEL_INPUT_MODALITIES,
101
+ OPPER_MODELS,
102
+ OPPER_MODEL_CONTEXT_WINDOWS,
103
+ OPPER_MODEL_MAX_OUTPUT_TOKENS,
104
+ OPPER_MODEL_INPUT_MODALITIES,
101
105
  } from "./model-seeds";
102
106
 
103
107
  export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
@@ -219,6 +223,72 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
219
223
  },
220
224
  note: "Shared Token Factory text-output inference only; live discovery excludes embedding and image-generation rows.",
221
225
  },
226
+ {
227
+ // Primary sources checked 2026-09-11:
228
+ // - https://docs.crusoecloud.com/quickstart/getting-started-with-serverless-inference documents
229
+ // the fixed OpenAI-compatible host https://api.inference.crusoecloud.com/v1, Bearer API keys
230
+ // created in the Cloud console (Intelligence Foundry > Inference > Create API Key), and an
231
+ // OpenAI SDK chat.completions example against meta-llama/Llama-3.3-70B-Instruct.
232
+ // - https://docs.crusoecloud.com/serverless-inference/available-models lists the served models
233
+ // with slash-delimited ids; https://docs.crusoecloud.com/serverless-inference/rate-limits
234
+ // documents per-project, per-model TPM/RPM limits (429 when exceeded, 503 under shared load).
235
+ // - GET /v1/models rejects unauthenticated requests with 401 {"errors":["Authentication failed"]},
236
+ // so a successful authenticated list response is evidence that the supplied key is valid.
237
+ // An authenticated capture on 2026-09-12 returned 18 rows shaped like OpenRouter's catalog
238
+ // (`is_public`, `type`, `context_length`, `architecture.modality` of "text" or "multimodal",
239
+ // `tags`, `pricing`, `supported_parameters`); 17 were public serverless models and one was an
240
+ // account-private dedicated deployment with empty `type`/`modality`. `type` is blank on one
241
+ // public model, so the filter keys on `is_public` plus `architecture.modality` instead.
242
+ // - https://legal.crusoe.ai/ hosts the Crusoe Cloud Platform Terms of Service v1.10 (effective
243
+ // 2026-08-10), which name Crusoe Technologies LLC as the contracting entity, and the Service
244
+ // Specific Terms v5.0 (effective 2026-07-14), whose Crusoe Intelligence Foundry Terms cover the
245
+ // Managed Inference Service reached through the Crusoe API.
246
+ // - https://models.dev/api.json (provider "crusoe") records openai/gpt-oss-120b as the one served
247
+ // model with a low/medium/high reasoning_effort ladder; the other reasoning models expose an
248
+ // on/off toggle only.
249
+ // Maintainer: @acheamponge, who works at Crusoe (affiliation disclosed) and also maintains the
250
+ // models.dev crusoe entry.
251
+ id: "crusoe",
252
+ label: "Crusoe",
253
+ baseUrl: "https://api.inference.crusoecloud.com/v1",
254
+ adapter: "openai-chat",
255
+ authKind: "key",
256
+ dashboardUrl: "https://console.crusoecloud.com",
257
+ liveModels: true,
258
+ preserveCustomDestination: true,
259
+ // The getting-started guide documents tools through the OpenAI SDK but no provider-wide
260
+ // parallel tool-call contract.
261
+ parallelToolCalls: false,
262
+ // Only gpt-oss-120b has a real effort ladder; toggle-style reasoning models must not be promoted
263
+ // to Codex's full fallback ladder.
264
+ reasoningEfforts: [],
265
+ modelReasoningEfforts: { "openai/gpt-oss-120b": ["low", "medium", "high"] },
266
+ directReasoningEffortModels: ["openai/gpt-oss-120b"],
267
+ // The catalog reports `architecture.modality: "multimodal"` without an input list. Four rows
268
+ // also carry the explicit "image text to text" tag; yutori/n2 instead reports multimodal
269
+ // type/modality plus browser/computer-use tags. Those five captured rows are classified here.
270
+ modelInputModalities: {
271
+ "google/gemma-4-31b-it": ["text", "image"],
272
+ "moonshotai/Kimi-K2.6": ["text", "image"],
273
+ "nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B": ["text", "image"],
274
+ "yutori/n2": ["text", "image"],
275
+ "zai-org/GLM-5.3-Flash": ["text", "image"],
276
+ },
277
+ modelDiscovery: {
278
+ path: "models",
279
+ maxResponseBytes: 256 * 1024,
280
+ maxModels: 256,
281
+ filter: {
282
+ // Keep public serverless rows whose architecture produces text; account-private
283
+ // deployments (blank modality) and any embedding or media rows fail closed.
284
+ allOf: [
285
+ { path: ["is_public"], equalsAny: [true] },
286
+ { path: ["architecture", "modality"], equalsAny: ["text", "multimodal"] },
287
+ ],
288
+ },
289
+ },
290
+ note: "Public Serverless Inference chat models on the shared OpenAI-compatible host; account-private and self-serve dedicated deployments are excluded from discovery and out of scope.",
291
+ },
222
292
  {
223
293
  id: "digitalocean",
224
294
  label: "DigitalOcean Serverless Inference",
@@ -736,6 +806,7 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
736
806
  ...Object.fromEntries(ALIBABA_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
737
807
  ...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
738
808
  "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
809
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
739
810
  "deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
740
811
  "deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
741
812
  "deepseek-v4-flash-0731": deepseekThinkingEffortsFor("deepseek-v4-flash-0731"),
@@ -781,6 +852,7 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
781
852
  ...Object.fromEntries(ALIBABA_INTL_TOKEN_PLAN_QWEN_MODELS.map(id => [id, THINKING_BUDGET_EFFORTS])),
782
853
  ...Object.fromEntries(QWEN38_FAMILY.map(id => [id, QWEN38_REASONING_EFFORTS])),
783
854
  "glm-5.2": ZAI_GLM_52_REASONING_EFFORTS,
855
+ "glm-5.3": ZAI_GLM_53_REASONING_EFFORTS,
784
856
  "deepseek-v4-pro": deepseekThinkingEffortsFor("deepseek-v4-pro"),
785
857
  "deepseek-v4-pro-0813": deepseekThinkingEffortsFor("deepseek-v4-pro-0813"),
786
858
  "deepseek-v4-flash": deepseekThinkingEffortsFor("deepseek-v4-flash"),
@@ -957,6 +1029,30 @@ export const PROVIDER_REGISTRY_EXTENDED: readonly ProviderRegistryEntry[] = [
957
1029
  noJsonSchemaModels: [...DEEPSEEK_GATEWAY_THINKING_MODELS, ...OPENCODE_FREE_DEEPSEEK_MODELS],
958
1030
  },
959
1031
  { id: "vercel-ai-gateway", label: "Vercel AI Gateway", baseUrl: "https://ai-gateway.vercel.sh/v1", adapter: "openai-chat", authKind: "key", dashboardUrl: "https://vercel.com/dashboard" },
1032
+ {
1033
+ // Opper: EU-hosted AI gateway (Opper AI AB, Stockholm). One OpenAI-compatible endpoint and one
1034
+ // key in front of 30+ upstream providers. Seeded ids are Opper *pools* (bare names such as
1035
+ // `claude-sonnet-4-6`): the gateway chooses the provider/region per request, and a
1036
+ // `vendor/model` id (`anthropic/claude-sonnet-4-6`, `aws/claude-sonnet-4-6-eu`) pins one route.
1037
+ // The original provider author reported on 2026-09-08 that GET /v3/compat/models answers 401
1038
+ // without a key, so the default discovery URL doubles as key validation. Windows, output caps
1039
+ // and modalities live in model-seeds.ts (smallest value / shared modality across each pool's
1040
+ // members); live discovery owns which models exist.
1041
+ id: "opper",
1042
+ label: "Opper",
1043
+ adapter: "openai-chat",
1044
+ baseUrl: "https://api.opper.ai/v3/compat",
1045
+ authKind: "key",
1046
+ dashboardUrl: "https://platform.opper.ai",
1047
+ liveModels: true,
1048
+ preserveCustomDestination: true,
1049
+ defaultModel: "claude-sonnet-4-6",
1050
+ models: OPPER_MODELS,
1051
+ modelContextWindows: OPPER_MODEL_CONTEXT_WINDOWS,
1052
+ modelMaxOutputTokens: OPPER_MODEL_MAX_OUTPUT_TOKENS,
1053
+ modelInputModalities: OPPER_MODEL_INPUT_MODALITIES,
1054
+ note: "EU-hosted AI gateway: one OpenAI-compatible endpoint and one key in front of 30+ providers. Bare model ids are pools (claude-sonnet-4-6, gpt-5.5) and Opper picks the route per request; vendor/model ids (anthropic/claude-sonnet-4-6) pin one provider. The catalogue is discovered live from /v3/compat/models with your key; the public list is at opper.ai/models. Token rates are the model providers' rates with no markup; Opper charges a 3% fee when you buy credits.",
1055
+ },
960
1056
  {
961
1057
  id: "opencode-free",
962
1058
  label: "OpenCode Free",