@bitkyc08/opencodex 2.57.0 → 2.59.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (241) hide show
  1. package/README.md +28 -10
  2. package/gui/dist/assets/index-C5IebErG.js +136 -0
  3. package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/gui/dist/provider-icons/crusoe.svg +1 -0
  6. package/gui/dist/provider-icons/opper.svg +3 -0
  7. package/package.json +2 -2
  8. package/src/adapters/base.ts +11 -1
  9. package/src/adapters/codebuddy/scaffold-guard.ts +5 -4
  10. package/src/adapters/command-code.ts +13 -4
  11. package/src/adapters/cursor/catalog.ts +11 -0
  12. package/src/adapters/cursor/cursor-errors.ts +15 -0
  13. package/src/adapters/cursor/discovery.ts +65 -1
  14. package/src/adapters/cursor/effort-map.ts +16 -2
  15. package/src/adapters/cursor/envelope-echo.ts +55 -2
  16. package/src/adapters/cursor/live-transport.ts +5 -1
  17. package/src/adapters/cursor/message-mapper.ts +3 -2
  18. package/src/adapters/cursor/protobuf-events.ts +110 -11
  19. package/src/adapters/cursor/protobuf-request.ts +27 -6
  20. package/src/adapters/cursor/request-builder.ts +14 -3
  21. package/src/adapters/cursor/text-toolcall.ts +230 -0
  22. package/src/adapters/cursor/thread-continuity.ts +141 -0
  23. package/src/adapters/cursor/tool-guidance.ts +5 -4
  24. package/src/adapters/cursor/types.ts +5 -0
  25. package/src/adapters/cursor.ts +97 -6
  26. package/src/adapters/devin/cloud-direct/chat.ts +11 -2
  27. package/src/adapters/devin/cloud-direct/index.ts +7 -0
  28. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
  29. package/src/adapters/devin.ts +75 -13
  30. package/src/adapters/google-antigravity-wire.ts +29 -2
  31. package/src/adapters/google-http.ts +45 -13
  32. package/src/adapters/google.ts +23 -4
  33. package/src/adapters/mimo-free.ts +32 -17
  34. package/src/adapters/ollama-native.ts +42 -8
  35. package/src/adapters/openai-chat/response-events.ts +61 -0
  36. package/src/adapters/openai-chat.ts +5 -10
  37. package/src/adapters/openai-responses/passthrough.ts +40 -5
  38. package/src/adapters/openai-responses/request-strips.ts +43 -0
  39. package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
  40. package/src/adapters/openai-responses/tool-schema.ts +19 -7
  41. package/src/adapters/physical-send.ts +50 -0
  42. package/src/adapters/responses-tool-schema.ts +76 -46
  43. package/src/adapters/run-turn-queue.ts +17 -4
  44. package/src/bridge/response-json.ts +2 -2
  45. package/src/bridge/sse.ts +166 -25
  46. package/src/claude/context-windows.ts +22 -0
  47. package/src/claude/outbound.ts +46 -5
  48. package/src/cli/account-api.ts +4 -3
  49. package/src/cli/account-extended.ts +22 -2
  50. package/src/cli/account-orca-import.ts +63 -0
  51. package/src/cli/account.ts +32 -4
  52. package/src/cli/capabilities.ts +40 -0
  53. package/src/cli/claude.ts +29 -1
  54. package/src/cli/codex-cli-update.ts +97 -2
  55. package/src/cli/config-command.ts +35 -18
  56. package/src/cli/dispatch.ts +71 -4
  57. package/src/cli/doctor.ts +197 -2
  58. package/src/cli/help.ts +4 -1
  59. package/src/cli/index.ts +132 -22
  60. package/src/cli/models-runtime.ts +33 -4
  61. package/src/cli/registry.ts +11 -1
  62. package/src/cli/runtime-api.ts +44 -0
  63. package/src/cli/start-args.ts +94 -0
  64. package/src/cli/system-command.ts +72 -1
  65. package/src/cli/uninstall-client-state.ts +12 -0
  66. package/src/client/machine-api.ts +4 -3
  67. package/src/client/machine-listener.ts +14 -1
  68. package/src/clients/config-export/constants.ts +2 -3
  69. package/src/clients/config-export.ts +5 -5
  70. package/src/codex/account-store.ts +81 -5
  71. package/src/codex/auth-api/pool-quota-probe.ts +14 -3
  72. package/src/codex/auth-api/routes.ts +17 -2
  73. package/src/codex/auth-context.ts +58 -20
  74. package/src/codex/catalog/build-entries.ts +25 -4
  75. package/src/codex/catalog/derive-entry.ts +8 -1
  76. package/src/codex/catalog/effort.ts +10 -6
  77. package/src/codex/catalog/gather-capture.ts +1 -0
  78. package/src/codex/catalog/model-hints.ts +37 -5
  79. package/src/codex/catalog/parsing.ts +83 -5
  80. package/src/codex/catalog/reserve-warn.ts +96 -0
  81. package/src/codex/catalog/retained-sync.ts +19 -0
  82. package/src/codex/catalog/routed-gather.ts +42 -3
  83. package/src/codex/cli-installation-identity.ts +210 -0
  84. package/src/codex/cli-installation-targets.ts +158 -0
  85. package/src/codex/convergence.ts +5 -0
  86. package/src/codex/desktop-switches.ts +145 -0
  87. package/src/codex/history-job.ts +5 -1
  88. package/src/codex/history-provider.ts +37 -5
  89. package/src/codex/history-state-open.ts +105 -0
  90. package/src/codex/history-worker.ts +14 -1
  91. package/src/codex/inject/config-toml.ts +44 -2
  92. package/src/codex/inject/remove.ts +145 -7
  93. package/src/codex/inject/restore.ts +204 -32
  94. package/src/codex/inject.ts +6 -9
  95. package/src/codex/lineage.ts +83 -32
  96. package/src/codex/loopback-target.ts +40 -0
  97. package/src/codex/main-account-hard-lock.ts +2 -1
  98. package/src/codex/main-account.ts +10 -3
  99. package/src/codex/main-device-reauth.ts +17 -9
  100. package/src/codex/model-entitlements.ts +60 -1
  101. package/src/codex/native-profile-startup.ts +64 -20
  102. package/src/codex/observed-model-denials.ts +137 -0
  103. package/src/codex/orca-auth-source.ts +94 -0
  104. package/src/codex/orca-import.ts +219 -0
  105. package/src/codex/prompt-text-probe.ts +282 -12
  106. package/src/codex/quota-401-recovery.ts +12 -0
  107. package/src/codex/quota-types.ts +65 -0
  108. package/src/codex/quota.ts +24 -19
  109. package/src/codex/routing/cooldown-math.ts +8 -47
  110. package/src/codex/routing/pin-drain.ts +57 -0
  111. package/src/codex/routing.ts +13 -15
  112. package/src/codex/subagent-model-fallback.ts +94 -0
  113. package/src/codex/windows-installation-files.ts +224 -0
  114. package/src/combos/failover.ts +122 -5
  115. package/src/config/atomic-write.ts +83 -8
  116. package/src/config/diagnostics.ts +21 -0
  117. package/src/config/load-degrade.ts +15 -0
  118. package/src/config/pending-teardown.ts +8 -0
  119. package/src/config/process-state.ts +36 -3
  120. package/src/config/provider-relative-send-path.ts +16 -0
  121. package/src/config/proxy-env.ts +23 -5
  122. package/src/config/schema/config-schema.ts +23 -0
  123. package/src/config/schema/leaf-validators.ts +65 -17
  124. package/src/generated/compatibility-version.json +337 -201
  125. package/src/generated/model-metadata.ts +1 -1
  126. package/src/lib/bounded-body.ts +4 -2
  127. package/src/lib/bounded-subprocess.ts +62 -10
  128. package/src/lib/destination-policy.ts +48 -6
  129. package/src/lib/errors.ts +3 -15
  130. package/src/lib/local-destinations.ts +32 -5
  131. package/src/lib/provider-outbound.ts +3 -3
  132. package/src/lib/proxy-env.ts +70 -3
  133. package/src/lib/request-execution-budget.ts +11 -3
  134. package/src/lib/response-body-inactivity.ts +193 -0
  135. package/src/lib/retry-delay.ts +69 -0
  136. package/src/lib/socks5-fetch.ts +631 -0
  137. package/src/lib/spend-reservation-ledger.ts +115 -9
  138. package/src/lib/windows-secret-acl.ts +151 -15
  139. package/src/lib/windows-user-principal.ts +5 -1
  140. package/src/lib/workflow-budget.ts +145 -8
  141. package/src/oauth/account-quota-rank.ts +72 -15
  142. package/src/oauth/generic-account-failover.ts +40 -27
  143. package/src/oauth/orcarouter.ts +15 -2
  144. package/src/oauth/store.ts +8 -0
  145. package/src/providers/codex-capacity.ts +9 -0
  146. package/src/providers/derive.ts +6 -0
  147. package/src/providers/devin-provider-merge-migration.ts +33 -12
  148. package/src/providers/free-directory.ts +20 -2
  149. package/src/providers/key-failover.ts +261 -7
  150. package/src/providers/model-discovery.ts +19 -7
  151. package/src/providers/model-rename-migration.ts +1 -0
  152. package/src/providers/openai-sidecar.ts +4 -0
  153. package/src/providers/opencode-go-transport.ts +14 -5
  154. package/src/providers/quota/report-cache.ts +3 -0
  155. package/src/providers/registry/entries-core.ts +11 -0
  156. package/src/providers/registry/entries-extended.ts +146 -28
  157. package/src/providers/registry/model-seeds.ts +136 -29
  158. package/src/providers/registry/types.ts +9 -0
  159. package/src/responses/apply-patch-envelope.ts +44 -11
  160. package/src/responses/bridge-search-replay-cache.ts +152 -0
  161. package/src/responses/code-mode-helper-compat.ts +26 -16
  162. package/src/responses/custom-tool-compat.ts +1 -1
  163. package/src/responses/hosted-tool-policy.ts +85 -2
  164. package/src/responses/schema.ts +9 -2
  165. package/src/responses/spill-store.ts +17 -0
  166. package/src/responses/state/body-policy.ts +25 -0
  167. package/src/responses/state/spill-queue.ts +8 -6
  168. package/src/responses/state.ts +3 -22
  169. package/src/router.ts +4 -0
  170. package/src/server/auth-cors.ts +27 -0
  171. package/src/server/chat-completions.ts +9 -4
  172. package/src/server/chat-native-sse.ts +26 -9
  173. package/src/server/chat-native.ts +10 -4
  174. package/src/server/claude-messages.ts +24 -2
  175. package/src/server/gui-static.ts +36 -2
  176. package/src/server/inbound-body-admission.ts +187 -0
  177. package/src/server/index/websocket-handler.ts +48 -1
  178. package/src/server/index.ts +15 -19
  179. package/src/server/management/api-access.ts +3 -4
  180. package/src/server/management/config-routes.ts +57 -10
  181. package/src/server/management/provider-capability-config.ts +35 -7
  182. package/src/server/management/provider-routes.ts +70 -18
  183. package/src/server/models-capabilities.ts +24 -3
  184. package/src/server/proxy-liveness.ts +97 -2
  185. package/src/server/relay.ts +17 -24
  186. package/src/server/request-log.ts +25 -1
  187. package/src/server/responses/adapter-continuation.ts +71 -27
  188. package/src/server/responses/adapter-delivery.ts +39 -8
  189. package/src/server/responses/adapter-dispatch.ts +52 -24
  190. package/src/server/responses/codex-ws-exchange.ts +65 -4
  191. package/src/server/responses/combo-stream-preflight.ts +68 -5
  192. package/src/server/responses/compact.ts +60 -11
  193. package/src/server/responses/core-codex-account.ts +83 -22
  194. package/src/server/responses/core-combo.ts +26 -0
  195. package/src/server/responses/core-normalize.ts +12 -5
  196. package/src/server/responses/core-options.ts +3 -0
  197. package/src/server/responses/fetch-helpers.ts +72 -3
  198. package/src/server/responses/native-injection-protocol.ts +42 -0
  199. package/src/server/responses/native-injection-replay.ts +105 -0
  200. package/src/server/responses/native-injection.ts +242 -0
  201. package/src/server/responses/native-response-control.ts +56 -0
  202. package/src/server/responses/native-response-json.ts +14 -0
  203. package/src/server/responses/native-response-output.ts +37 -0
  204. package/src/server/responses/native-steering-log.ts +44 -0
  205. package/src/server/responses/native-steering-policy.ts +49 -0
  206. package/src/server/responses/native-steering-replay.ts +126 -0
  207. package/src/server/responses/native-steering-settings.ts +76 -0
  208. package/src/server/responses/native-steering.ts +400 -0
  209. package/src/server/responses/native-tool-results.ts +130 -0
  210. package/src/server/responses/passthrough-delivery.ts +21 -1
  211. package/src/server/responses/passthrough-dispatch.ts +146 -49
  212. package/src/server/responses/passthrough-execution.ts +11 -1
  213. package/src/server/responses/request-prepare.ts +70 -0
  214. package/src/server/responses/request-send-budget.ts +84 -7
  215. package/src/server/responses/request-sidecar-auth.ts +16 -8
  216. package/src/server/responses/request-spend.ts +38 -9
  217. package/src/server/responses/request-transport.ts +13 -10
  218. package/src/server/responses/run-turn-execution.ts +20 -5
  219. package/src/server/responses/sidecar-execution.ts +2 -0
  220. package/src/server/responses/ws-upstream.ts +23 -2
  221. package/src/server/responses-custom-tool-repair.ts +2 -2
  222. package/src/server/sse-frame-buffer.ts +12 -10
  223. package/src/server/sse-payload-rewrite.ts +36 -9
  224. package/src/server/stop-teardown.ts +8 -1
  225. package/src/server/system-env-shell.ts +5 -1
  226. package/src/server/system-env.ts +7 -1
  227. package/src/server/workflow-refusal.ts +56 -2
  228. package/src/server/ws-bridge.ts +16 -1
  229. package/src/service/cli.ts +29 -7
  230. package/src/service/guards.ts +10 -0
  231. package/src/service/health.ts +43 -0
  232. package/src/service/state.ts +7 -2
  233. package/src/types/accounts.ts +4 -0
  234. package/src/types/config.ts +104 -3
  235. package/src/types/provider.ts +32 -0
  236. package/src/types/request.ts +7 -1
  237. package/src/types/wire.ts +9 -1
  238. package/src/usage/expected-prices.ts +28 -0
  239. package/src/usage/log.ts +87 -4
  240. package/src/web-search/passthrough-bridge.ts +39 -5
  241. package/gui/dist/assets/index-Cz7CLdif.js +0 -128
@@ -3,7 +3,7 @@ import type { AdapterEvent, OcxProviderConfig } from "../types";
3
3
  import type { ProviderAdapter } from "./base";
4
4
  import { isTranslatorBudgetExceededError } from "../lib/translator-budget";
5
5
  import { cursorExecDeniedMessage, cursorRequestDeclaresFullAccess } from "./cursor/exec-policy";
6
- import { isCursorBenignCancelError, isCursorInvalidArgumentError, isCursorOverflowRemintCandidate, isCursorRootEnvelopeError, safeCursorErrorMessage, type CursorSizeContext } from "./cursor/cursor-errors";
6
+ import { isCursorBenignCancelError, isCursorIncompleteToolCallMessage, isCursorInvalidArgumentError, isCursorOverflowRemintCandidate, isCursorRootEnvelopeError, safeCursorErrorMessage, type CursorSizeContext } from "./cursor/cursor-errors";
7
7
  import { cursorCheckpointModelAffinityId, inferCursorContextWindow, isCursorExternalWireModel } from "./cursor/discovery";
8
8
  import { createCursorKvStore, type CursorKvStore } from "./cursor/kv-store";
9
9
  import { mapCursorServerMessage } from "./cursor/message-mapper";
@@ -32,8 +32,14 @@ import { isDebugEnabled } from "../lib/debug-settings";
32
32
  import { createAdapterTierMetadata } from "../providers/fastwire";
33
33
  import { estimateTokens } from "../lib/token-estimate";
34
34
  import {
35
+ clearCursorIncompleteToolRemint,
36
+ cursorIncompleteToolRemintScopeKey,
37
+ clearCursorEnvelopeEchoRemint,
38
+ cursorEnvelopeEchoRemintScopeKey,
35
39
  cursorOverflowRemintScopeKey,
36
40
  markCursorOverflowSurfaced,
41
+ recordCursorIncompleteToolRemint,
42
+ recordCursorEnvelopeEchoRemint,
37
43
  recordCursorOverflowRemint,
38
44
  rememberCursorThreadConversation,
39
45
  shouldSkipCursorOverflowRemint,
@@ -99,11 +105,20 @@ function safeCursorTransportError(err: unknown, sizeContext?: CursorSizeContext)
99
105
  * estimate over the outgoing text vs the model's context window. Only used to keep
100
106
  * SMALL requests on the 429 class — unknown/large stays on the overflow mapping.
101
107
  */
102
- function cursorRequestSizeContext(request: { modelId: string; system: string[]; messages: { content: string }[] }): CursorSizeContext {
108
+ function cursorRequestSizeContext(request: {
109
+ modelId: string;
110
+ _cursorIdentityScope?: string;
111
+ system: string[];
112
+ messages: { content: string }[];
113
+ }): CursorSizeContext {
103
114
  const text = [...request.system, ...request.messages.map(message => message.content)].join("\n");
104
115
  return {
105
116
  estimatedInputTokens: estimateTokens(text, request.modelId),
106
- contextWindow: inferCursorContextWindow(request.modelId),
117
+ // Prefers this identity scope's checkpoint `maxTokens` over the id heuristic
118
+ // so a plan-gated ceiling participates in the 0.5-window overflow vs 429 prior.
119
+ contextWindow: inferCursorContextWindow(request.modelId, {
120
+ identityScope: request._cursorIdentityScope,
121
+ }),
107
122
  };
108
123
  }
109
124
 
@@ -171,7 +186,10 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
171
186
  }
172
187
  const inheritedCheckpointRef = _parsed._providerContinuation?.cursor?.checkpointRef;
173
188
  const previousConversationId = _parsed._cursorConversationId;
174
- let request = createCursorRequest(_parsed);
189
+ let request = {
190
+ ...createCursorRequest(_parsed),
191
+ _cursorIdentityScope: _parsed._cursorIdentityScope?.trim() || "local",
192
+ };
175
193
  requestSizeContext = cursorRequestSizeContext(request);
176
194
  // The builder may derive a stable provider id from the client thread when Responses state
177
195
  // is unavailable. Rekey only existing state; there is nothing to migrate on a fresh turn,
@@ -190,6 +208,8 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
190
208
  let completedNormally = false;
191
209
  let lastTransport: { captured?: Uint8Array } | undefined;
192
210
  let emittedClientTool = false;
211
+ let sawIncompleteToolCall = false;
212
+ let sawMidstreamEnvelopeEcho = false;
193
213
  // Ordering proof for tool-suspended checkpoints: true only when the newest captured
194
214
  // checkpoint bytes arrived AFTER the turn emitted a client tool call, i.e. upstream
195
215
  // serialized its suspended-on-tool-call state. Only that snapshot can safely resume
@@ -332,6 +352,9 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
332
352
  },
333
353
  });
334
354
  for (const event of events) {
355
+ if (event.type === "error" && isCursorIncompleteToolCallMessage(event.message)) {
356
+ sawIncompleteToolCall = true;
357
+ }
335
358
  if (!guardsSettled()) {
336
359
  if (event.type === "text_delta") {
337
360
  guardHeld.push(event);
@@ -371,7 +394,9 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
371
394
  }
372
395
  if (event.type !== "heartbeat") emittedOutput = true;
373
396
  if (event.type === "done") {
374
- for (const finding of midstreamObserver?.findings() ?? []) {
397
+ const midstreamFindings = midstreamObserver?.findings() ?? [];
398
+ if (midstreamFindings.length > 0) sawMidstreamEnvelopeEcho = true;
399
+ for (const finding of midstreamFindings) {
375
400
  debugProviderDiagnostic("cursor", "midstream-envelope-echo", {
376
401
  wireModel: activeRequest.modelId,
377
402
  conversationHash: activeRequest.conversationId.slice(0, 16),
@@ -413,7 +438,10 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
413
438
  const remintConversationId = (failedConversationId: string) => {
414
439
  lastTransport = undefined;
415
440
  _parsed._cursorConversationId = undefined;
416
- const next = createCursorRequest(_parsed, { forceFreshConversation: true });
441
+ const next = {
442
+ ...createCursorRequest(_parsed, { forceFreshConversation: true }),
443
+ _cursorIdentityScope: _parsed._cursorIdentityScope?.trim() || "local",
444
+ };
417
445
  rekeyContextUsage(failedConversationId, next.conversationId);
418
446
  _parsed._cursorConversationId = next.conversationId;
419
447
  // Persist recovery for store:false clients that send any stable Cursor thread owner, so
@@ -514,6 +542,69 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
514
542
  }
515
543
  }
516
544
  }
545
+ const incompleteToolRemintScopeKey =
546
+ _parsed._cursorIsolateConversation !== true
547
+ && request.contextUsageStoreCheckpoints !== false
548
+ ? cursorIncompleteToolRemintScopeKey(
549
+ cursorClientThreadOwner(_parsed),
550
+ _parsed._cursorIdentityScope,
551
+ )
552
+ : null;
553
+ // Incomplete-tool errors are streamed, not thrown. Do not retry this turn; rotate only
554
+ // the next turn's id. request-prepare currently isolates compaction, but adapter callers
555
+ // can bypass that upstream invariant, so checkpoint storage is the local isolation boundary.
556
+ if (sawIncompleteToolCall && incompleteToolRemintScopeKey) {
557
+ if (recordCursorIncompleteToolRemint(incompleteToolRemintScopeKey)) {
558
+ if (inheritedCheckpointRef) invalidateCursorCheckpoint(inheritedCheckpointRef);
559
+ debugProviderDiagnostic("cursor", "incomplete-tool-remint", {
560
+ wireModel: request.modelId,
561
+ conversationHash: request.conversationId.slice(0, 16),
562
+ });
563
+ remintConversationId(request.conversationId);
564
+ } else {
565
+ debugProviderDiagnostic("cursor", "incomplete-tool-remint-exhausted", {
566
+ wireModel: request.modelId,
567
+ conversationHash: request.conversationId.slice(0, 16),
568
+ });
569
+ }
570
+ } else if (!sawIncompleteToolCall && completedNormally && incompleteToolRemintScopeKey) {
571
+ clearCursorIncompleteToolRemint(incompleteToolRemintScopeKey);
572
+ }
573
+ // A mid-stream envelope echo has ALREADY reached the client — the prefix sniffer only
574
+ // watches the first bytes of a turn, and grok-4.6 writes a real sentence before pasting
575
+ // the envelope. It cannot be quarantined, so the recovery is the same as the
576
+ // incomplete-tool case: leave this turn alone and rotate the next turn's id, otherwise
577
+ // the stored echo is replayed and primes the model to echo again.
578
+ //
579
+ // Its own budget, not the incomplete-tool one: echoing is cheap and repeatable while an
580
+ // incomplete client-tool stream is rare and structural, so a shared counter would let a
581
+ // persistently echoing model spend the allowance the other recovery needs. Skipped when
582
+ // the incomplete-tool arm already reminted this turn — one rotation is enough.
583
+ const envelopeEchoRemintScopeKey =
584
+ _parsed._cursorIsolateConversation !== true
585
+ && request.contextUsageStoreCheckpoints !== false
586
+ ? cursorEnvelopeEchoRemintScopeKey(
587
+ cursorClientThreadOwner(_parsed),
588
+ _parsed._cursorIdentityScope,
589
+ )
590
+ : null;
591
+ if (sawMidstreamEnvelopeEcho && !sawIncompleteToolCall && envelopeEchoRemintScopeKey) {
592
+ if (recordCursorEnvelopeEchoRemint(envelopeEchoRemintScopeKey)) {
593
+ if (inheritedCheckpointRef) invalidateCursorCheckpoint(inheritedCheckpointRef);
594
+ debugProviderDiagnostic("cursor", "midstream-envelope-echo-remint", {
595
+ wireModel: request.modelId,
596
+ conversationHash: request.conversationId.slice(0, 16),
597
+ });
598
+ remintConversationId(request.conversationId);
599
+ } else {
600
+ debugProviderDiagnostic("cursor", "midstream-envelope-echo-remint-exhausted", {
601
+ wireModel: request.modelId,
602
+ conversationHash: request.conversationId.slice(0, 16),
603
+ });
604
+ }
605
+ } else if (!sawMidstreamEnvelopeEcho && completedNormally && envelopeEchoRemintScopeKey) {
606
+ clearCursorEnvelopeEchoRemint(envelopeEchoRemintScopeKey);
607
+ }
517
608
  if (
518
609
  request.checkpointInvalidationReason
519
610
  && request.checkpointInvalidationReason !== "missing_ref"
@@ -35,7 +35,7 @@ import {
35
35
  } from './wire.js';
36
36
  import { buildMetadata } from './metadata.js';
37
37
  import { getCachedUserJwt } from './auth.js';
38
- import { getCachedCatalog, ModelNotAvailableError } from './catalog.js';
38
+ import { getCachedCatalog, ModelNotAvailableError, type CacheEntry } from './catalog.js';
39
39
  import { anySignal, cancelBodyOnAbort } from '../../../lib/abort.js';
40
40
  import { resolveDevinApiBaseUrl } from '../../../oauth/devin/api-base.js';
41
41
 
@@ -1046,6 +1046,13 @@ export interface CloudChatRequest {
1046
1046
  completionOpts?: BuildArgs['completionOpts'];
1047
1047
  /** Override request_type (default = 5, CASCADE). */
1048
1048
  requestType?: number;
1049
+ /**
1050
+ * Catalog the caller already resolved this turn. An explicit `null`
1051
+ * records a failed lookup: the pre-flight below then skips its own fetch
1052
+ * instead of paying a second catalog timeout on the same turn. Omit the
1053
+ * field to let the pre-flight perform its own cached lookup.
1054
+ */
1055
+ catalog?: CacheEntry | null;
1049
1056
  /** Abort signal — closes the fetch stream. */
1050
1057
  signal?: AbortSignal;
1051
1058
  }
@@ -1140,7 +1147,9 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
1140
1147
  // error and the trailer-error path below enriches the message in-place.
1141
1148
  // Treat an empty catalog (schema drift / unexpected response) as "no catalog"
1142
1149
  // so chat passes through instead of failing every request.
1143
- const catalog = await getCachedCatalog(req.apiKey, host, req.signal).catch(() => null);
1150
+ const catalog = req.catalog !== undefined
1151
+ ? req.catalog
1152
+ : await getCachedCatalog(req.apiKey, host, req.signal).catch(() => null);
1144
1153
  if (catalog && catalog.byUid.size > 0) {
1145
1154
  const entry = catalog.byUid.get(req.modelUid);
1146
1155
  if (!entry) {
@@ -49,6 +49,13 @@ export {
49
49
  type ToolDef,
50
50
  } from './chat.js';
51
51
 
52
+ export {
53
+ streamChatEventsWithResetRetry,
54
+ STATED_RESET_MAX_REPLAYS,
55
+ STATED_RESET_MAX_WAIT_MS,
56
+ type StatedResetRetryOptions,
57
+ } from './stated-reset-retry.js';
58
+
52
59
  export {
53
60
  mintUserJwt,
54
61
  getCachedUserJwt,
@@ -0,0 +1,103 @@
1
+ /**
2
+ * Same-target retry of an explicit, pre-output 429 refusal with a stated
3
+ * recovery delay. No event-producing attempt is ever automatically replayed.
4
+ * This is a refusal-specific policy, not a claim that every eventless POST
5
+ * is idempotent; ambiguous transport failures still propagate unchanged.
6
+ */
7
+ import { parseRetryAfterFromMessage } from '../../../lib/retry-delay.js';
8
+ import { abortError, sleepWithAbort } from '../../../lib/upstream-retry.js';
9
+ import { CloudChatError, streamChatEvents, type CloudChatEvent, type CloudChatRequest } from './chat.js';
10
+
11
+ /** 1 initial attempt plus at most 2 replays. */
12
+ export const STATED_RESET_MAX_REPLAYS = 2;
13
+ /** Default cumulative wait allowance for one invocation (30 minutes). */
14
+ export const STATED_RESET_MAX_WAIT_MS = 1_800_000;
15
+ /** Absolute maximum cumulative allowance, including explicit overrides. */
16
+ export const STATED_RESET_WAIT_CEILING_MS = 3_600_000;
17
+
18
+ function statedResetMaxWaitMs(): number {
19
+ const raw = process.env.OPENCODEX_DEVIN_STATED_RESET_WAIT_MS?.trim();
20
+ if (!raw) return STATED_RESET_MAX_WAIT_MS;
21
+ const parsed = Number(raw);
22
+ if (!Number.isFinite(parsed) || parsed < 0) return STATED_RESET_MAX_WAIT_MS;
23
+ // Zero explicitly disables local waiting. Values above one hour are capped.
24
+ return Math.min(Math.floor(parsed), STATED_RESET_WAIT_CEILING_MS);
25
+ }
26
+ export const statedResetMaxWaitMsForTests = statedResetMaxWaitMs;
27
+
28
+ export interface StatedResetRetryOptions {
29
+ /** Test seam: defaults to the real cloud stream. */
30
+ stream?: (req: CloudChatRequest) => AsyncGenerator<CloudChatEvent>;
31
+ /** Test seam: must either honour the whole delay or reject on cancellation. */
32
+ sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
33
+ maxReplays?: number;
34
+ /** CUMULATIVE wait allowance, not a fresh allowance on every failure. */
35
+ maxWaitMs?: number;
36
+ }
37
+
38
+ function replayLimit(value: number | undefined): number {
39
+ if (value === undefined) return STATED_RESET_MAX_REPLAYS;
40
+ if (!Number.isInteger(value) || value < 0 || value > STATED_RESET_MAX_REPLAYS) {
41
+ throw new RangeError('maxReplays must be an integer from 0 to 2');
42
+ }
43
+ return value;
44
+ }
45
+
46
+ function waitLimit(value: number | undefined): number {
47
+ if (value === undefined) return statedResetMaxWaitMs();
48
+ if (!Number.isFinite(value) || value < 0) {
49
+ throw new RangeError('maxWaitMs must be a finite non-negative number');
50
+ }
51
+ return Math.min(Math.floor(value), STATED_RESET_WAIT_CEILING_MS);
52
+ }
53
+
54
+ export async function* streamChatEventsWithResetRetry(
55
+ req: CloudChatRequest,
56
+ options?: StatedResetRetryOptions,
57
+ ): AsyncGenerator<CloudChatEvent> {
58
+ const stream = options?.stream ?? streamChatEvents;
59
+ const sleep = options?.sleep ?? sleepWithAbort;
60
+ const maxReplays = replayLimit(options?.maxReplays);
61
+ const maxWaitMs = waitLimit(options?.maxWaitMs);
62
+ let replays = 0;
63
+ let waitedMs = 0;
64
+ while (true) {
65
+ // Check again after sleeping: cancellation can race with timer completion.
66
+ // A pre-aborted request must not even enter a custom transport.
67
+ if (req.signal?.aborted) throw abortError(req.signal);
68
+ let yielded = false;
69
+ try {
70
+ for await (const event of stream(req)) {
71
+ // Latch before yielding, so a consumer-injected error is post-output.
72
+ yielded = true;
73
+ yield event;
74
+ }
75
+ return;
76
+ } catch (error) {
77
+ if (req.signal?.aborted) throw abortError(req.signal);
78
+ const waitSec = !yielded
79
+ && error instanceof CloudChatError
80
+ && error.status === 429
81
+ ? parseRetryAfterFromMessage(error.message)
82
+ : undefined;
83
+ const waitMs = waitSec === undefined ? undefined : waitSec * 1000;
84
+ if (
85
+ waitMs === undefined
86
+ || replays >= maxReplays
87
+ || waitMs > maxWaitMs - waitedMs
88
+ ) {
89
+ // Never shorten a provider's minimum delay to fit the local budget.
90
+ // Keep the original refusal so outer policy can preserve its metadata.
91
+ throw error;
92
+ }
93
+ replays += 1;
94
+ // Charge the complete scheduled wait once, before sleeping. This is a
95
+ // sleep allowance, not a wall-clock deadline on generation or timer
96
+ // scheduling: waking a few milliseconds late must not reject an already
97
+ // approved one-hour retry. No later wait can spend this allowance again.
98
+ waitedMs += waitMs;
99
+ await sleep(waitMs, req.signal);
100
+ if (req.signal?.aborted) throw abortError(req.signal);
101
+ }
102
+ }
103
+ }
@@ -9,9 +9,9 @@
9
9
  import type { AdapterEvent, OcxAssistantMessage, OcxContentPart, OcxMessage, OcxParsedRequest, OcxProviderConfig, OcxTool, OcxToolCall, OcxToolResultMessage, OcxUsage } from "../types";
10
10
  import { namespacedToolName } from "../types";
11
11
  import type { IncomingMeta, ProviderAdapter } from "./base";
12
- import { streamChatEvents, allocateCascadeId, CloudChatError, type ChatHistoryItem, type ToolDef } from "./devin/cloud-direct";
12
+ import { streamChatEventsWithResetRetry, allocateCascadeId, CloudChatError, type ChatHistoryItem, type ToolDef } from "./devin/cloud-direct";
13
13
  import type { ContentPart } from "./devin/cloud-direct/chat";
14
- import { getCachedCatalog } from "./devin/cloud-direct/catalog";
14
+ import { getCachedCatalog, type CacheEntry } from "./devin/cloud-direct/catalog";
15
15
  import { collapseDevinModelUid } from "./devin/live-models";
16
16
  import { buildNonOpenAIToolCatalogNudgeForTools } from "./tool-catalog-nudge";
17
17
  import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiServer } from "../oauth/devin";
@@ -155,6 +155,7 @@ async function resolveWireModelUid(
155
155
  apiKey: string,
156
156
  host: string,
157
157
  reasoningEffort?: string,
158
+ catalog?: CacheEntry | null,
158
159
  ): Promise<string> {
159
160
  const modelId = normalizeDevinModelId(rawModelId);
160
161
  // Explicit effort wins over a suffix the picker already baked into the id, so
@@ -163,15 +164,18 @@ async function resolveWireModelUid(
163
164
  const swe2 = resolveSwe2Variant(modelId, reasoningEffort);
164
165
  if (swe2) return swe2;
165
166
  if (hasEffortSuffix(modelId)) return modelId;
166
- const catalog = await getCachedCatalog(apiKey, host);
167
- if (catalog) {
168
- if (catalog.byUid.has(modelId)) return modelId;
167
+ // Callers that already read the catalog this turn pass it in; an explicit
168
+ // null records a failed lookup and must not trigger a same-turn retry —
169
+ // failures are not cached, so re-reading would only pay another timeout.
170
+ const entry = catalog !== undefined ? catalog : await getCachedCatalog(apiKey, host);
171
+ if (entry) {
172
+ if (entry.byUid.has(modelId)) return modelId;
169
173
  const effort = reasoningEffort && CALLER_EFFORT_VALUES.has(reasoningEffort) ? reasoningEffort : "medium";
170
174
  const suffixed = `${modelId}-${effort}`;
171
- if (catalog.byUid.has(suffixed)) return suffixed;
175
+ if (entry.byUid.has(suffixed)) return suffixed;
172
176
  // Fall back to any enabled variant of this base model.
173
- for (const uid of catalog.byUid.keys()) {
174
- if (uid.startsWith(modelId + "-") && !catalog.byUid.get(uid)?.disabled) return uid;
177
+ for (const uid of entry.byUid.keys()) {
178
+ if (uid.startsWith(modelId + "-") && !entry.byUid.get(uid)?.disabled) return uid;
175
179
  }
176
180
  }
177
181
  // Degraded mode: append the default effort suffix.
@@ -186,6 +190,46 @@ async function resolveWireModelUid(
186
190
  */
187
191
  export const resolveWireModelUidForTests = resolveWireModelUid;
188
192
 
193
+ /**
194
+ * Resolve the INPUT ceiling for the exact UID selected for this turn. Catalog
195
+ * ClientModelConfig #18 and CompletionConfiguration #3 both carry input tokens;
196
+ * the independent output cap is not subtracted here. Smaller operator hints
197
+ * cap live evidence, never enlarge it. No evidence leaves the encoder's 128k
198
+ * fallback intact; an unrelated or opt-in long-context variant is not evidence.
199
+ */
200
+ function resolveDevinMaxInputTokens(
201
+ provider: OcxProviderConfig,
202
+ modelUid: string,
203
+ liveWindow?: number,
204
+ ): number | undefined {
205
+ const positive = (value: unknown): number | undefined =>
206
+ typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : undefined;
207
+ const baseId = collapseDevinModelUid(modelUid);
208
+ const configured = (record: Record<string, number> | undefined): number | undefined => {
209
+ if (!record) return undefined;
210
+ for (const id of [modelUid, baseId]) {
211
+ // Prefer the canonical spelling; retain dotted/case-folded saved hints,
212
+ // matching the model-id normalization used for the inference request.
213
+ const exact = Object.hasOwn(record, id) ? positive(record[id]) : undefined;
214
+ if (exact !== undefined) return exact;
215
+ const matches = Object.entries(record)
216
+ .filter(([key]) => normalizeDevinModelId(key).toLowerCase() === id.toLowerCase())
217
+ .map(([, value]) => positive(value))
218
+ .filter((value): value is number => value !== undefined);
219
+ if (matches.length > 0) return Math.min(...matches);
220
+ }
221
+ return undefined;
222
+ };
223
+ const contextHint = configured(provider.modelContextWindows) ?? positive(provider.contextWindow);
224
+ const inputHint = configured(provider.modelMaxInputTokens);
225
+ const ceilings = [positive(liveWindow), contextHint, inputHint]
226
+ .filter((value): value is number => value !== undefined);
227
+ return ceilings.length > 0 ? Math.min(...ceilings) : undefined;
228
+ }
229
+
230
+ /** Pure test seam; runtime uses the same resolver immediately before dispatch. */
231
+ export const resolveDevinMaxInputTokensForTests = resolveDevinMaxInputTokens;
232
+
189
233
  export class DevinMissingCredentialError extends Error {
190
234
  constructor() {
191
235
  super("Devin live transport requires a Devin API key. Run ocx login devin to sign in with your Cognition/Devin account.");
@@ -493,7 +537,16 @@ export function createDevinAdapter(
493
537
  // entry: an EU or FedStart account that used provider.baseUrl would send
494
538
  // every RPC to the US server it is not provisioned on.
495
539
  const host = resolveDevinApiServer(provider.baseUrl, credentialProviderId);
496
- const modelUid = await resolveWireModelUid(rawModelId, apiKey, host, parsed.options.reasoning);
540
+ // One catalog read per turn serves model-UID resolution, the input
541
+ // ceiling, and the chat pre-flight inside streamChatEvents. Failures are
542
+ // not cached, so a second read would only pay another fetch timeout on
543
+ // an otherwise valid turn.
544
+ const catalog = await getCachedCatalog(apiKey, host, incoming.abortSignal);
545
+ if (incoming.abortSignal?.aborted) {
546
+ emit({ type: "error", message: DEVIN_CLIENT_CLOSED_MESSAGE, status: 499, retryable: false });
547
+ return;
548
+ }
549
+ const modelUid = await resolveWireModelUid(rawModelId, apiKey, host, parsed.options.reasoning, catalog);
497
550
  const returnedToolNames = buildDevinReturnedToolNameMap(parsed.context.tools);
498
551
  let openToolId: string | undefined;
499
552
  let usage: OcxUsage | undefined;
@@ -506,17 +559,26 @@ export function createDevinAdapter(
506
559
  };
507
560
 
508
561
  try {
509
- for await (const event of streamChatEvents({
562
+ // Read the selected UID's catalog row, not the picker's collapsed base.
563
+ const maxInputTokens = resolveDevinMaxInputTokens(
564
+ provider, modelUid, catalog?.byUid.get(modelUid)?.contextWindow,
565
+ );
566
+ // The reset-retry wrapper waits out a 429 that states its own recovery
567
+ // delay ("limit will reset in 35 seconds") and replays the identical
568
+ // request — but only while zero events have been yielded, so a
569
+ // post-output failure still takes the terminal path untouched.
570
+ for await (const event of streamChatEventsWithResetRetry({
510
571
  apiKey,
511
572
  apiServerUrl: host,
512
573
  modelUid,
574
+ catalog,
513
575
  messages: mapOcxMessagesToDevin(parsed),
514
576
  tools: mapOcxToolsToDevin(parsed.context.tools),
515
577
  cascadeId,
516
- // Without these the request falls back to the encoder's defaults
517
- // (8192 output, a 128k context window, temperature 0.7), so a client
518
- // that asked for a 4k cap never got one.
578
+ // Input and output ceilings are separate wire fields. Omitting the
579
+ // input hint used to force every model through the 128k default.
519
580
  completionOpts: {
581
+ ...(maxInputTokens !== undefined ? { maxInputTokens } : {}),
520
582
  ...(typeof parsed.options.maxOutputTokens === "number" ? { maxOutputTokens: parsed.options.maxOutputTokens } : {}),
521
583
  ...(typeof parsed.options.temperature === "number" ? { temperature: parsed.options.temperature } : {}),
522
584
  ...(typeof parsed.options.topP === "number" ? { topP: parsed.options.topP } : {}),
@@ -97,8 +97,35 @@ export function antigravitySessionId(parsed: OcxParsedRequest): string {
97
97
  * collision, is the failure mode this function exists to prevent.
98
98
  */
99
99
  function clientThreadAnchor(parsed: OcxParsedRequest): string | undefined {
100
- const threadId = parsed._clientThreadId?.trim();
101
- return threadId ? `codex-thread:${threadId}` : undefined;
100
+ // `_clientThreadId` carries `x-codex-parent-thread-id`, which every parallel child of one
101
+ // parent presents identically, so anchoring on it alone collapsed concurrent children onto a
102
+ // single upstream Cloud Code Assist session (#5033).
103
+ //
104
+ // A thread id is only unique WITHIN its parent, which is why `codexConversationIdentity` keys
105
+ // on both. #5054 anchored on the child alone and therefore only moved the collision: two
106
+ // parents can each have a child of the same id (#5058). The pair is the identity, joined by a
107
+ // NUL so the encoding is injective: no Codex id contains one, so a pair cannot be re-read as a
108
+ // different pair, nor as the parent-only anchor below.
109
+ //
110
+ // Deliberately NOT the general lane key: `codexConversationKeyFor` is an HMAC under a
111
+ // process-random secret, so it changes across a proxy restart — and instability, not sharing,
112
+ // is the failure mode this derivation has to avoid. These ids are Codex's own values and
113
+ // survive both compaction and restart.
114
+ const own = parsed._codexOwnThreadId?.trim();
115
+ const parent = parsed._clientThreadId?.trim();
116
+ if (own && parent) return `codex-thread:${parent}\u0000${own}`;
117
+ // Parent-only clients keep the anchor they already had.
118
+ if (parent) return `codex-thread:${parent}`;
119
+ // A parentless ROOT deliberately omits the parent header. `src/server/context-history.ts` says
120
+ // so in as many words: root model requests use (session-id=root, thread-id=root) and do not
121
+ // fabricate a parent key. It has no pair to key on, and #5054's claim that a root presents
122
+ // `thread-id` equal to its parent was simply wrong.
123
+ //
124
+ // So it keeps the pre-#5054 anchor rather than gaining an own-thread one. That is not a
125
+ // preference: durable Antigravity replay state is keyed by model plus session id, and moving a
126
+ // root's anchor on upgrade strands every signature stored under the old session — the exact
127
+ // instability this derivation exists to avoid, introduced while fixing sharing.
128
+ return undefined;
102
129
  }
103
130
 
104
131
  /** A Gemini content part as it appears in an Antigravity request body. */
@@ -1,4 +1,7 @@
1
1
  import type { AdapterFetchContext, AdapterRequest } from "./base";
2
+ import { createAdapterPhysicalSend } from "./physical-send";
3
+ import type { SendClass } from "../lib/request-execution-budget";
4
+ import type { AttemptRecoveryKind } from "../usage/log";
2
5
  import { isQuotaExhaustedBody, retryableGoogleStatus, safeGoogleHttpErrorMessage } from "./google-errors";
3
6
  import { repairGoogleInvalidRequestBody } from "./google-wire-compiler";
4
7
  import { normalizeUpstreamHttpErrorResponse, readDisplaySafeErrorPayloadText } from "./upstream-http-error";
@@ -8,6 +11,8 @@ import {
8
11
  fetchWithAttemptDeadline,
9
12
  retryBackoffDelayMs,
10
13
  sleepWithAbort,
14
+ SendBudgetExhaustedError,
15
+ isConnectionResetError,
11
16
  } from "../lib/upstream-retry";
12
17
 
13
18
  const GOOGLE_RETRY_ATTEMPTS = 3;
@@ -41,18 +46,30 @@ export async function fetchGoogleWithRetry(
41
46
  ): Promise<Response> {
42
47
  const repairInvalid400 = opts.repairInvalid400 ?? true;
43
48
  const timeoutMs = ctx.timeoutMs ?? 200_000;
44
- const executor = ctx.executor ?? globalThis.fetch;
49
+ const send = createAdapterPhysicalSend(ctx);
45
50
  let lastError: unknown;
46
51
  let activeRequest = request;
47
52
  let compatibilityReplayUsed = false;
53
+ let pendingResponse: Response | undefined;
54
+ let retryDelayMs = 0;
55
+ let sendClass: SendClass = "transient";
56
+ let recovery: AttemptRecoveryKind | undefined;
48
57
  for (let attempt = 0; attempt < GOOGLE_RETRY_ATTEMPTS; attempt++) {
49
58
  if (ctx.abortSignal?.aborted) throw abortError(ctx.abortSignal);
50
59
  try {
51
- const res = await fetchWithAttemptDeadline(activeRequest.url, {
52
- method: activeRequest.method,
53
- headers: activeRequest.headers,
54
- body: activeRequest.body,
55
- }, timeoutMs, ctx.abortSignal, ctx.stream, executor);
60
+ const res = await send({ url: activeRequest.url, sendClass, recovery,
61
+ beforeDispatch: async () => {
62
+ if (retryDelayMs > 0) await sleepWithAbort(retryDelayMs, ctx.abortSignal);
63
+ if (pendingResponse) cancelResponseBodyBestEffort(pendingResponse);
64
+ pendingResponse = undefined;
65
+ },
66
+ dispatch: executor => fetchWithAttemptDeadline(activeRequest.url, {
67
+ method: activeRequest.method, headers: activeRequest.headers, body: activeRequest.body,
68
+ }, timeoutMs, ctx.abortSignal, ctx.stream, executor),
69
+ });
70
+ retryDelayMs = 0;
71
+ sendClass = "transient";
72
+ recovery = undefined;
56
73
  if (res.status === 400 && repairInvalid400 && !compatibilityReplayUsed) {
57
74
  let payloadText = "";
58
75
  try {
@@ -64,7 +81,8 @@ export async function fetchGoogleWithRetry(
64
81
  if (repairedBody !== undefined) {
65
82
  compatibilityReplayUsed = true;
66
83
  activeRequest = { ...activeRequest, body: repairedBody };
67
- cancelResponseBodyBestEffort(res);
84
+ pendingResponse = res;
85
+ sendClass = "repair";
68
86
  attempt--; // The changed-request replay is separate from transient retry accounting.
69
87
  continue;
70
88
  }
@@ -75,7 +93,7 @@ export async function fetchGoogleWithRetry(
75
93
  // A 429 may be a transient rate limit (retry) or hard quota exhaustion (do NOT retry —
76
94
  // it won't recover for hours and burns retries). Peek the body to tell them apart.
77
95
  if (res.status === 429) {
78
- const peekTarget = ctx.returnRawErrors ? res.clone() : res;
96
+ const peekTarget = res.clone();
79
97
  const peek = await readDisplaySafeErrorPayloadText(peekTarget, ctx.abortSignal);
80
98
  if (isQuotaExhaustedBody(peek)) {
81
99
  return ctx.returnRawErrors ? res : normalizeUpstreamHttpErrorResponse(res, {
@@ -84,20 +102,34 @@ export async function fetchGoogleWithRetry(
84
102
  });
85
103
  }
86
104
  }
87
- cancelResponseBodyBestEffort(res);
88
- await sleepWithAbort(retryBackoffDelayMs(attempt, {
105
+ pendingResponse = res;
106
+ recovery = res.status === 429 ? "rate-limit-429" : "transient-5xx";
107
+ retryDelayMs = retryBackoffDelayMs(attempt, {
89
108
  baseDelayMs: GOOGLE_RETRY_BASE_MS,
90
109
  maxDelayMs: GOOGLE_RETRY_MAX_MS,
91
110
  headers: res.headers,
92
- }), ctx.abortSignal);
111
+ });
93
112
  } catch (err) {
94
113
  if (ctx.abortSignal?.aborted) throw err;
114
+ if (err instanceof SendBudgetExhaustedError) {
115
+ if (pendingResponse) {
116
+ // The ladder had already classified this response as retryable and was about to send
117
+ // again; the budget refused. Returning the original response is right — it is a real
118
+ // upstream answer — but it used to leave the log indistinguishable from a request
119
+ // where no retry was ever eligible (#5044).
120
+ ctx.onRecoveryWithheld?.({ reason: "retry-send-budget" });
121
+ return ctx.returnRawErrors ? pendingResponse : normalizeFinalGoogleError(label, pendingResponse, ctx.abortSignal);
122
+ }
123
+ throw err;
124
+ }
95
125
  lastError = err;
96
126
  if (attempt === GOOGLE_RETRY_ATTEMPTS - 1) throw err;
97
- await sleepWithAbort(retryBackoffDelayMs(attempt, {
127
+ sendClass = "transient";
128
+ recovery = isConnectionResetError(err) ? "connection-reset" : undefined;
129
+ retryDelayMs = retryBackoffDelayMs(attempt, {
98
130
  baseDelayMs: GOOGLE_RETRY_BASE_MS,
99
131
  maxDelayMs: GOOGLE_RETRY_MAX_MS,
100
- }), ctx.abortSignal);
132
+ });
101
133
  }
102
134
  }
103
135
  throw lastError ?? new Error(`${label} fetch failed`);