@bitkyc08/opencodex 2.56.0 → 2.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (206) hide show
  1. package/bin/ocx.mjs +10 -0
  2. package/gui/dist/assets/{index-D4zuyIxQ.js → index-BbrHOIY0.js} +21 -21
  3. package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/package.json +4 -4
  6. package/src/adapters/codebuddy/adapter.ts +2 -1
  7. package/src/adapters/codebuddy/scaffold-guard.ts +249 -0
  8. package/src/adapters/command-code.ts +12 -3
  9. package/src/adapters/cursor/cursor-errors.ts +15 -0
  10. package/src/adapters/cursor/discovery.ts +65 -1
  11. package/src/adapters/cursor/envelope-echo.ts +8 -2
  12. package/src/adapters/cursor/live-transport.ts +5 -1
  13. package/src/adapters/cursor/protobuf-events.ts +110 -11
  14. package/src/adapters/cursor/protobuf-request.ts +19 -1
  15. package/src/adapters/cursor/text-toolcall.ts +230 -0
  16. package/src/adapters/cursor/thread-continuity.ts +67 -0
  17. package/src/adapters/cursor/types.ts +5 -0
  18. package/src/adapters/cursor.ts +55 -5
  19. package/src/adapters/google-http.ts +38 -13
  20. package/src/adapters/google.ts +7 -7
  21. package/src/adapters/kiro/payload.ts +17 -3
  22. package/src/adapters/kiro/reasoning.ts +70 -7
  23. package/src/adapters/kiro/stream.ts +8 -2
  24. package/src/adapters/kiro/wire.ts +2 -1
  25. package/src/adapters/kiro-events.ts +21 -13
  26. package/src/adapters/mimo-free.ts +32 -17
  27. package/src/adapters/ollama-native.ts +42 -8
  28. package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
  29. package/src/adapters/openai-chat/tool-schema.ts +25 -7
  30. package/src/adapters/openai-chat.ts +8 -8
  31. package/src/adapters/openai-responses/passthrough.ts +62 -5
  32. package/src/adapters/openai-responses/request-strips.ts +43 -0
  33. package/src/adapters/physical-send.ts +50 -0
  34. package/src/bridge/errors.ts +26 -2
  35. package/src/bridge/response-json.ts +8 -2
  36. package/src/bridge/sse.ts +20 -2
  37. package/src/claude/desktop-profile.ts +66 -9
  38. package/src/claude/outbound.ts +32 -4
  39. package/src/cli/account-main.ts +1 -1
  40. package/src/cli/capabilities.ts +2 -2
  41. package/src/cli/combo.ts +10 -1
  42. package/src/cli/config-command.ts +35 -18
  43. package/src/cli/dispatch.ts +17 -4
  44. package/src/cli/index.ts +92 -7
  45. package/src/cli/registry.ts +2 -1
  46. package/src/cli/system-command.ts +74 -5
  47. package/src/cli/uninstall-client-state.ts +12 -0
  48. package/src/clients/config-export.ts +7 -3
  49. package/src/codex/account-label.ts +14 -3
  50. package/src/codex/account-store.ts +113 -26
  51. package/src/codex/account-usability.ts +21 -0
  52. package/src/codex/auth-api/login-flow.ts +14 -2
  53. package/src/codex/auth-api/reset-credit-service.ts +11 -2
  54. package/src/codex/auth-context.ts +199 -15
  55. package/src/codex/catalog/aggregation.ts +80 -1
  56. package/src/codex/catalog/model-visibility.ts +1 -0
  57. package/src/codex/catalog/remote.ts +30 -0
  58. package/src/codex/catalog/retained-sync.ts +9 -1
  59. package/src/codex/catalog/routed-gather.ts +38 -1
  60. package/src/codex/cli-install-provenance.ts +7 -1
  61. package/src/codex/convergence.ts +7 -2
  62. package/src/codex/desktop-app/types.ts +11 -2
  63. package/src/codex/desktop-app/windows.ts +5 -5
  64. package/src/codex/desktop-switches.ts +145 -0
  65. package/src/codex/history-job.ts +5 -1
  66. package/src/codex/history-provider.ts +33 -4
  67. package/src/codex/history-worker.ts +14 -1
  68. package/src/codex/inject/remove.ts +145 -7
  69. package/src/codex/inject/restore.ts +231 -32
  70. package/src/codex/inject.ts +12 -16
  71. package/src/codex/loopback-target.ts +9 -0
  72. package/src/codex/model-entitlements.ts +152 -15
  73. package/src/codex/native-profile-startup.ts +64 -20
  74. package/src/codex/pool-refresh-backoff.ts +12 -3
  75. package/src/codex/quota-rejection.ts +104 -15
  76. package/src/codex/routing/cache-affinity.ts +70 -0
  77. package/src/codex/routing/cooldown-math.ts +10 -0
  78. package/src/codex/routing/selection.ts +79 -2
  79. package/src/codex/routing/thread-affinity.ts +50 -2
  80. package/src/codex/routing/transient-hold-dispatch.ts +141 -0
  81. package/src/codex/routing.ts +29 -49
  82. package/src/codex/warmup.ts +1 -1
  83. package/src/combos/failover.ts +85 -0
  84. package/src/combos/request.ts +17 -10
  85. package/src/combos/types.ts +23 -2
  86. package/src/config/atomic-write.ts +83 -8
  87. package/src/config/pending-teardown.ts +31 -0
  88. package/src/config/schema/config-schema.ts +2 -0
  89. package/src/config/schema/leaf-validators.ts +1 -0
  90. package/src/generated/compatibility-version.json +272 -180
  91. package/src/images/loop.ts +1 -1
  92. package/src/lib/bounded-subprocess.ts +62 -10
  93. package/src/lib/errors.ts +17 -0
  94. package/src/lib/request-execution-budget.ts +147 -21
  95. package/src/lib/spend-reservation-ledger.ts +18 -0
  96. package/src/lib/state-store-registrations.ts +6 -2
  97. package/src/lib/test-home-guard.ts +85 -1
  98. package/src/lib/upstream-retry.ts +77 -10
  99. package/src/lib/windows-elevation.ts +76 -14
  100. package/src/lib/windows-secret-acl.ts +151 -15
  101. package/src/lib/windows-user-principal.ts +5 -1
  102. package/src/oauth/index.ts +2 -2
  103. package/src/oauth/key-providers.ts +2 -2
  104. package/src/providers/derive.ts +6 -0
  105. package/src/providers/kiro-models.ts +4 -3
  106. package/src/providers/label.ts +19 -1
  107. package/src/providers/model-discovery.ts +35 -7
  108. package/src/providers/registry/entries-core.ts +18 -0
  109. package/src/providers/registry/entries-extended.ts +59 -28
  110. package/src/providers/registry/model-seeds.ts +71 -17
  111. package/src/providers/registry/types.ts +9 -0
  112. package/src/responses/reasoning-envelope.ts +6 -3
  113. package/src/responses/spill-store.ts +17 -0
  114. package/src/responses/state/body-policy.ts +25 -0
  115. package/src/responses/state/spill-queue.ts +8 -6
  116. package/src/responses/state.ts +3 -22
  117. package/src/router.ts +4 -0
  118. package/src/routing/identity-domains.ts +21 -14
  119. package/src/routing/probe-lease.ts +103 -1
  120. package/src/server/auth-cors.ts +1 -0
  121. package/src/server/chat-completions.ts +3 -1
  122. package/src/server/chat-native.ts +37 -9
  123. package/src/server/index/live-sideband.ts +37 -1
  124. package/src/server/index/websocket-handler.ts +54 -3
  125. package/src/server/index.ts +5 -5
  126. package/src/server/inspection-tee.ts +107 -0
  127. package/src/server/live.ts +46 -1
  128. package/src/server/management/combo-routes.ts +10 -1
  129. package/src/server/management/config-routes.ts +27 -5
  130. package/src/server/models-capabilities.ts +24 -3
  131. package/src/server/relay-eager.ts +2 -0
  132. package/src/server/relay.ts +14 -19
  133. package/src/server/request-log.ts +127 -3
  134. package/src/server/response-log-body.ts +153 -0
  135. package/src/server/responses/account-change-state.ts +74 -0
  136. package/src/server/responses/adapter-continuation.ts +33 -7
  137. package/src/server/responses/adapter-delivery.ts +5 -11
  138. package/src/server/responses/adapter-dispatch.ts +84 -13
  139. package/src/server/responses/codex-ws-exchange.ts +65 -4
  140. package/src/server/responses/codex-ws-wire.ts +5 -0
  141. package/src/server/responses/collaboration.ts +74 -4
  142. package/src/server/responses/combo-session-recall.ts +68 -8
  143. package/src/server/responses/combo-stream-preflight.ts +68 -5
  144. package/src/server/responses/compact.ts +54 -13
  145. package/src/server/responses/core-auth.ts +2 -0
  146. package/src/server/responses/core-codex-account.ts +51 -3
  147. package/src/server/responses/core-combo.ts +129 -23
  148. package/src/server/responses/core-errors.ts +18 -0
  149. package/src/server/responses/core-options.ts +3 -0
  150. package/src/server/responses/core-replay.ts +105 -32
  151. package/src/server/responses/core.ts +3 -3
  152. package/src/server/responses/encrypted-payload.ts +0 -1
  153. package/src/server/responses/fetch-helpers.ts +4 -1
  154. package/src/server/responses/input-admission.ts +126 -6
  155. package/src/server/responses/native-injection-protocol.ts +42 -0
  156. package/src/server/responses/native-injection-replay.ts +105 -0
  157. package/src/server/responses/native-injection.ts +242 -0
  158. package/src/server/responses/native-response-control.ts +56 -0
  159. package/src/server/responses/native-response-json.ts +14 -0
  160. package/src/server/responses/native-response-output.ts +37 -0
  161. package/src/server/responses/native-steering-log.ts +44 -0
  162. package/src/server/responses/native-steering-policy.ts +49 -0
  163. package/src/server/responses/native-steering-replay.ts +126 -0
  164. package/src/server/responses/native-steering-settings.ts +76 -0
  165. package/src/server/responses/native-steering.ts +400 -0
  166. package/src/server/responses/native-tool-results.ts +130 -0
  167. package/src/server/responses/passthrough-delivery.ts +30 -6
  168. package/src/server/responses/passthrough-dispatch.ts +61 -11
  169. package/src/server/responses/passthrough-error.ts +38 -2
  170. package/src/server/responses/request-prepare.ts +173 -22
  171. package/src/server/responses/request-send-budget.ts +97 -2
  172. package/src/server/responses/request-spend.ts +147 -0
  173. package/src/server/responses/request-transport.ts +62 -3
  174. package/src/server/responses/run-turn-execution.ts +59 -31
  175. package/src/server/responses/sidecar-execution.ts +7 -13
  176. package/src/server/responses/terminal-guard.ts +65 -4
  177. package/src/server/responses/ws-upstream.ts +21 -1
  178. package/src/server/responses-undeclared-tool-guard.ts +9 -5
  179. package/src/server/stop-teardown.ts +8 -1
  180. package/src/server/ws-bridge.ts +16 -1
  181. package/src/service/cli.ts +13 -1
  182. package/src/service/windows-ops.ts +210 -16
  183. package/src/service/windows-scheduler.ts +28 -21
  184. package/src/service.ts +1 -1
  185. package/src/types/config.ts +8 -1
  186. package/src/types/provider.ts +13 -0
  187. package/src/types/request.ts +8 -5
  188. package/src/types/tools.ts +24 -0
  189. package/src/types.ts +2 -0
  190. package/src/update/index.ts +10 -0
  191. package/src/update/stop-contract.d.mts +1 -0
  192. package/src/update/stop-contract.mjs +19 -0
  193. package/src/update/stop-decision.d.mts +1 -1
  194. package/src/update/stop-decision.mjs +12 -3
  195. package/src/usage/log.ts +1 -1
  196. package/src/vision/anthropic-describe.ts +1 -1
  197. package/src/vision/describe.ts +5 -5
  198. package/src/web-search/anthropic-executor.ts +1 -1
  199. package/src/web-search/exa-executor.ts +1 -1
  200. package/src/web-search/executor.ts +1 -1
  201. package/src/web-search/gemini-executor.ts +1 -1
  202. package/src/web-search/loop.ts +1 -1
  203. package/src/web-search/ollama-executor.ts +1 -1
  204. package/src/web-search/parse.ts +67 -14
  205. package/src/web-search/passthrough-bridge.ts +64 -31
  206. package/src/web-search/xai-executor.ts +1 -1
@@ -3,7 +3,7 @@ import type { AdapterEvent, OcxProviderConfig } from "../types";
3
3
  import type { ProviderAdapter } from "./base";
4
4
  import { isTranslatorBudgetExceededError } from "../lib/translator-budget";
5
5
  import { cursorExecDeniedMessage, cursorRequestDeclaresFullAccess } from "./cursor/exec-policy";
6
- import { isCursorBenignCancelError, isCursorInvalidArgumentError, isCursorOverflowRemintCandidate, isCursorRootEnvelopeError, safeCursorErrorMessage, type CursorSizeContext } from "./cursor/cursor-errors";
6
+ import { isCursorBenignCancelError, isCursorIncompleteToolCallMessage, isCursorInvalidArgumentError, isCursorOverflowRemintCandidate, isCursorRootEnvelopeError, safeCursorErrorMessage, type CursorSizeContext } from "./cursor/cursor-errors";
7
7
  import { cursorCheckpointModelAffinityId, inferCursorContextWindow, isCursorExternalWireModel } from "./cursor/discovery";
8
8
  import { createCursorKvStore, type CursorKvStore } from "./cursor/kv-store";
9
9
  import { mapCursorServerMessage } from "./cursor/message-mapper";
@@ -32,8 +32,11 @@ import { isDebugEnabled } from "../lib/debug-settings";
32
32
  import { createAdapterTierMetadata } from "../providers/fastwire";
33
33
  import { estimateTokens } from "../lib/token-estimate";
34
34
  import {
35
+ clearCursorIncompleteToolRemint,
36
+ cursorIncompleteToolRemintScopeKey,
35
37
  cursorOverflowRemintScopeKey,
36
38
  markCursorOverflowSurfaced,
39
+ recordCursorIncompleteToolRemint,
37
40
  recordCursorOverflowRemint,
38
41
  rememberCursorThreadConversation,
39
42
  shouldSkipCursorOverflowRemint,
@@ -99,11 +102,20 @@ function safeCursorTransportError(err: unknown, sizeContext?: CursorSizeContext)
99
102
  * estimate over the outgoing text vs the model's context window. Only used to keep
100
103
  * SMALL requests on the 429 class — unknown/large stays on the overflow mapping.
101
104
  */
102
- function cursorRequestSizeContext(request: { modelId: string; system: string[]; messages: { content: string }[] }): CursorSizeContext {
105
+ function cursorRequestSizeContext(request: {
106
+ modelId: string;
107
+ _cursorIdentityScope?: string;
108
+ system: string[];
109
+ messages: { content: string }[];
110
+ }): CursorSizeContext {
103
111
  const text = [...request.system, ...request.messages.map(message => message.content)].join("\n");
104
112
  return {
105
113
  estimatedInputTokens: estimateTokens(text, request.modelId),
106
- contextWindow: inferCursorContextWindow(request.modelId),
114
+ // Prefers this identity scope's checkpoint `maxTokens` over the id heuristic
115
+ // so a plan-gated ceiling participates in the 0.5-window overflow vs 429 prior.
116
+ contextWindow: inferCursorContextWindow(request.modelId, {
117
+ identityScope: request._cursorIdentityScope,
118
+ }),
107
119
  };
108
120
  }
109
121
 
@@ -171,7 +183,10 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
171
183
  }
172
184
  const inheritedCheckpointRef = _parsed._providerContinuation?.cursor?.checkpointRef;
173
185
  const previousConversationId = _parsed._cursorConversationId;
174
- let request = createCursorRequest(_parsed);
186
+ let request = {
187
+ ...createCursorRequest(_parsed),
188
+ _cursorIdentityScope: _parsed._cursorIdentityScope?.trim() || "local",
189
+ };
175
190
  requestSizeContext = cursorRequestSizeContext(request);
176
191
  // The builder may derive a stable provider id from the client thread when Responses state
177
192
  // is unavailable. Rekey only existing state; there is nothing to migrate on a fresh turn,
@@ -190,6 +205,7 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
190
205
  let completedNormally = false;
191
206
  let lastTransport: { captured?: Uint8Array } | undefined;
192
207
  let emittedClientTool = false;
208
+ let sawIncompleteToolCall = false;
193
209
  // Ordering proof for tool-suspended checkpoints: true only when the newest captured
194
210
  // checkpoint bytes arrived AFTER the turn emitted a client tool call, i.e. upstream
195
211
  // serialized its suspended-on-tool-call state. Only that snapshot can safely resume
@@ -332,6 +348,9 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
332
348
  },
333
349
  });
334
350
  for (const event of events) {
351
+ if (event.type === "error" && isCursorIncompleteToolCallMessage(event.message)) {
352
+ sawIncompleteToolCall = true;
353
+ }
335
354
  if (!guardsSettled()) {
336
355
  if (event.type === "text_delta") {
337
356
  guardHeld.push(event);
@@ -413,7 +432,10 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
413
432
  const remintConversationId = (failedConversationId: string) => {
414
433
  lastTransport = undefined;
415
434
  _parsed._cursorConversationId = undefined;
416
- const next = createCursorRequest(_parsed, { forceFreshConversation: true });
435
+ const next = {
436
+ ...createCursorRequest(_parsed, { forceFreshConversation: true }),
437
+ _cursorIdentityScope: _parsed._cursorIdentityScope?.trim() || "local",
438
+ };
417
439
  rekeyContextUsage(failedConversationId, next.conversationId);
418
440
  _parsed._cursorConversationId = next.conversationId;
419
441
  // Persist recovery for store:false clients that send any stable Cursor thread owner, so
@@ -514,6 +536,34 @@ export function createCursorAdapter(provider: OcxProviderConfig, deps: CursorAda
514
536
  }
515
537
  }
516
538
  }
539
+ const incompleteToolRemintScopeKey =
540
+ _parsed._cursorIsolateConversation !== true
541
+ && request.contextUsageStoreCheckpoints !== false
542
+ ? cursorIncompleteToolRemintScopeKey(
543
+ cursorClientThreadOwner(_parsed),
544
+ _parsed._cursorIdentityScope,
545
+ )
546
+ : null;
547
+ // Incomplete-tool errors are streamed, not thrown. Do not retry this turn; rotate only
548
+ // the next turn's id. request-prepare currently isolates compaction, but adapter callers
549
+ // can bypass that upstream invariant, so checkpoint storage is the local isolation boundary.
550
+ if (sawIncompleteToolCall && incompleteToolRemintScopeKey) {
551
+ if (recordCursorIncompleteToolRemint(incompleteToolRemintScopeKey)) {
552
+ if (inheritedCheckpointRef) invalidateCursorCheckpoint(inheritedCheckpointRef);
553
+ debugProviderDiagnostic("cursor", "incomplete-tool-remint", {
554
+ wireModel: request.modelId,
555
+ conversationHash: request.conversationId.slice(0, 16),
556
+ });
557
+ remintConversationId(request.conversationId);
558
+ } else {
559
+ debugProviderDiagnostic("cursor", "incomplete-tool-remint-exhausted", {
560
+ wireModel: request.modelId,
561
+ conversationHash: request.conversationId.slice(0, 16),
562
+ });
563
+ }
564
+ } else if (!sawIncompleteToolCall && completedNormally && incompleteToolRemintScopeKey) {
565
+ clearCursorIncompleteToolRemint(incompleteToolRemintScopeKey);
566
+ }
517
567
  if (
518
568
  request.checkpointInvalidationReason
519
569
  && request.checkpointInvalidationReason !== "missing_ref"
@@ -1,4 +1,7 @@
1
1
  import type { AdapterFetchContext, AdapterRequest } from "./base";
2
+ import { createAdapterPhysicalSend } from "./physical-send";
3
+ import type { SendClass } from "../lib/request-execution-budget";
4
+ import type { AttemptRecoveryKind } from "../usage/log";
2
5
  import { isQuotaExhaustedBody, retryableGoogleStatus, safeGoogleHttpErrorMessage } from "./google-errors";
3
6
  import { repairGoogleInvalidRequestBody } from "./google-wire-compiler";
4
7
  import { normalizeUpstreamHttpErrorResponse, readDisplaySafeErrorPayloadText } from "./upstream-http-error";
@@ -8,6 +11,8 @@ import {
8
11
  fetchWithAttemptDeadline,
9
12
  retryBackoffDelayMs,
10
13
  sleepWithAbort,
14
+ SendBudgetExhaustedError,
15
+ isConnectionResetError,
11
16
  } from "../lib/upstream-retry";
12
17
 
13
18
  const GOOGLE_RETRY_ATTEMPTS = 3;
@@ -41,18 +46,30 @@ export async function fetchGoogleWithRetry(
41
46
  ): Promise<Response> {
42
47
  const repairInvalid400 = opts.repairInvalid400 ?? true;
43
48
  const timeoutMs = ctx.timeoutMs ?? 200_000;
44
- const executor = ctx.executor ?? globalThis.fetch;
49
+ const send = createAdapterPhysicalSend(ctx);
45
50
  let lastError: unknown;
46
51
  let activeRequest = request;
47
52
  let compatibilityReplayUsed = false;
53
+ let pendingResponse: Response | undefined;
54
+ let retryDelayMs = 0;
55
+ let sendClass: SendClass = "transient";
56
+ let recovery: AttemptRecoveryKind | undefined;
48
57
  for (let attempt = 0; attempt < GOOGLE_RETRY_ATTEMPTS; attempt++) {
49
58
  if (ctx.abortSignal?.aborted) throw abortError(ctx.abortSignal);
50
59
  try {
51
- const res = await fetchWithAttemptDeadline(activeRequest.url, {
52
- method: activeRequest.method,
53
- headers: activeRequest.headers,
54
- body: activeRequest.body,
55
- }, timeoutMs, ctx.abortSignal, ctx.stream, executor);
60
+ const res = await send({ url: activeRequest.url, sendClass, recovery,
61
+ beforeDispatch: async () => {
62
+ if (retryDelayMs > 0) await sleepWithAbort(retryDelayMs, ctx.abortSignal);
63
+ if (pendingResponse) cancelResponseBodyBestEffort(pendingResponse);
64
+ pendingResponse = undefined;
65
+ },
66
+ dispatch: executor => fetchWithAttemptDeadline(activeRequest.url, {
67
+ method: activeRequest.method, headers: activeRequest.headers, body: activeRequest.body,
68
+ }, timeoutMs, ctx.abortSignal, ctx.stream, executor),
69
+ });
70
+ retryDelayMs = 0;
71
+ sendClass = "transient";
72
+ recovery = undefined;
56
73
  if (res.status === 400 && repairInvalid400 && !compatibilityReplayUsed) {
57
74
  let payloadText = "";
58
75
  try {
@@ -64,7 +81,8 @@ export async function fetchGoogleWithRetry(
64
81
  if (repairedBody !== undefined) {
65
82
  compatibilityReplayUsed = true;
66
83
  activeRequest = { ...activeRequest, body: repairedBody };
67
- cancelResponseBodyBestEffort(res);
84
+ pendingResponse = res;
85
+ sendClass = "repair";
68
86
  attempt--; // The changed-request replay is separate from transient retry accounting.
69
87
  continue;
70
88
  }
@@ -75,7 +93,7 @@ export async function fetchGoogleWithRetry(
75
93
  // A 429 may be a transient rate limit (retry) or hard quota exhaustion (do NOT retry —
76
94
  // it won't recover for hours and burns retries). Peek the body to tell them apart.
77
95
  if (res.status === 429) {
78
- const peekTarget = ctx.returnRawErrors ? res.clone() : res;
96
+ const peekTarget = res.clone();
79
97
  const peek = await readDisplaySafeErrorPayloadText(peekTarget, ctx.abortSignal);
80
98
  if (isQuotaExhaustedBody(peek)) {
81
99
  return ctx.returnRawErrors ? res : normalizeUpstreamHttpErrorResponse(res, {
@@ -84,20 +102,27 @@ export async function fetchGoogleWithRetry(
84
102
  });
85
103
  }
86
104
  }
87
- cancelResponseBodyBestEffort(res);
88
- await sleepWithAbort(retryBackoffDelayMs(attempt, {
105
+ pendingResponse = res;
106
+ recovery = res.status === 429 ? "rate-limit-429" : "transient-5xx";
107
+ retryDelayMs = retryBackoffDelayMs(attempt, {
89
108
  baseDelayMs: GOOGLE_RETRY_BASE_MS,
90
109
  maxDelayMs: GOOGLE_RETRY_MAX_MS,
91
110
  headers: res.headers,
92
- }), ctx.abortSignal);
111
+ });
93
112
  } catch (err) {
94
113
  if (ctx.abortSignal?.aborted) throw err;
114
+ if (err instanceof SendBudgetExhaustedError) {
115
+ if (pendingResponse) return ctx.returnRawErrors ? pendingResponse : normalizeFinalGoogleError(label, pendingResponse, ctx.abortSignal);
116
+ throw err;
117
+ }
95
118
  lastError = err;
96
119
  if (attempt === GOOGLE_RETRY_ATTEMPTS - 1) throw err;
97
- await sleepWithAbort(retryBackoffDelayMs(attempt, {
120
+ sendClass = "transient";
121
+ recovery = isConnectionResetError(err) ? "connection-reset" : undefined;
122
+ retryDelayMs = retryBackoffDelayMs(attempt, {
98
123
  baseDelayMs: GOOGLE_RETRY_BASE_MS,
99
124
  maxDelayMs: GOOGLE_RETRY_MAX_MS,
100
- }), ctx.abortSignal);
125
+ });
101
126
  }
102
127
  }
103
128
  throw lastError ?? new Error(`${label} fetch failed`);
@@ -796,14 +796,14 @@ export function createGoogleAdapter(provider: OcxProviderConfig): ProviderAdapte
796
796
  // body, URL or credential.
797
797
  const requestedTextFormat = parsed.options.textFormat;
798
798
  if (requestedTextFormat) {
799
- if (provider.googleMode === "cloud-code-assist") {
800
- // Not implemented or verified by opencodex for the Cloud Code Assist envelope,
801
- // including Claude models served through it. This is not a claim that the
802
- // upstream cannot do it — silence would return unconstrained prose as success,
803
- // which is the failure this fix exists to remove.
799
+ if (provider.googleMode === "cloud-code-assist" && !parsed.modelId.startsWith("gemini-")) {
800
+ // Not implemented by opencodex for non-Gemini models (including Claude)
801
+ // served through the Cloud Code Assist envelope. This is not a claim that
802
+ // the upstream cannot do it — silence would return unconstrained prose as success,
803
+ // which is the failure this refusal exists to prevent.
804
804
  throw new Error(
805
- "google cloud-code-assist structured output is not implemented by opencodex — "
806
- + "remove response_format or route this model through AI Studio or Vertex",
805
+ "google cloud-code-assist structured output is not implemented by opencodex for non-Gemini models — "
806
+ + "remove response_format or route this model through a direct provider",
807
807
  );
808
808
  }
809
809
  if (isImageCapableModel(parsed.modelId)) {
@@ -38,7 +38,12 @@ import {
38
38
  validateKiroConversationState,
39
39
  type KiroTurn,
40
40
  } from "./conversation";
41
- import { injectKiroThinkingTags, kiroNativeEffortField, KIRO_NATIVE_EFFORTS } from "./reasoning";
41
+ import {
42
+ injectKiroThinkingTags,
43
+ kiroNativeEffortField,
44
+ kiroReasoningContent,
45
+ KIRO_NATIVE_EFFORTS,
46
+ } from "./reasoning";
42
47
  import { kiroPayloadMessages, userContentText } from "./usage";
43
48
  import {
44
49
  kiroToolWireNames,
@@ -388,7 +393,11 @@ export function buildKiroPayload(
388
393
  assistantResponseMessage: {
389
394
  content: turn.content,
390
395
  ...(turn.toolUses.length > 0 ? { toolUses: turn.toolUses } : {}),
391
- ...(turn.redactedReasoning ? { reasoningContent: { redactedContent: turn.redactedReasoning } } : {}),
396
+ // Replayed on the field it was received on: the GPT-5.6 signature is not base64 and is
397
+ // rejected when sent as `redactedContent`.
398
+ ...(turn.redactedReasoning
399
+ ? { reasoningContent: kiroReasoningContent(turn.redactedReasoning) }
400
+ : {}),
392
401
  },
393
402
  }
394
403
  : {
@@ -447,7 +456,12 @@ export function buildKiroPayload(
447
456
  if (!KIRO_NATIVE_EFFORTS.includes(effort)) {
448
457
  throw new Error(`Kiro ${normalizeKiroModelId(parsed.modelId)} does not support reasoning effort ${JSON.stringify(effort)}`);
449
458
  }
450
- payload.additionalModelRequestFields = { [effortField]: { effort } };
459
+ // Model eligibility still owns unsupported-effort validation above; wire eligibility
460
+ // is narrower for luna/terra, whose unverified rungs retain the thinking-tag path.
461
+ const verifiedEffortField = kiroNativeEffortField(parsed.modelId, effort);
462
+ if (verifiedEffortField) {
463
+ payload.additionalModelRequestFields = { [verifiedEffortField]: { effort } };
464
+ }
451
465
  }
452
466
  if (profileArn) payload.profileArn = profileArn;
453
467
  return { payload, nameMap, conversationId, completionMode };
@@ -4,21 +4,46 @@ import type { OcxParsedRequest } from "../../types";
4
4
  export type KiroReasoningMode = "native" | "emulated";
5
5
 
6
6
  // Kiro takes a verified native effort field for these models, and each model family names it
7
- // differently: the Sol-only `reasoning.effort` versus the Claude-specific `output_config.effort`.
8
- // Models absent from this table fall back to emulated thinking instructions.
7
+ // differently: the GPT-5.6 family's `reasoning.effort` versus the Claude-specific
8
+ // `output_config.effort`. Models absent from this table fall back to emulated thinking
9
+ // instructions.
10
+ //
11
+ // The GPT-5.6 entries are measured against the live runtime rather than inferred from the vendor
12
+ // schema: the field is accepted (HTTP 200) and the encrypted reasoning blob that comes back grows
13
+ // with the effort. On one fixed hard prompt — a primality search plus a 20-bit recurrence count —
14
+ // luna's blob measured 5,130 chars at `low`, 16,686 at `medium`, 30,670 at `high` and 48,594 at
15
+ // `max`, against 13,118 with no effort signal at all; terra's measured 34,590 and 38,106 at native
16
+ // `max` against 11,758 and 17,598 bare, two repetitions each. The channel this replaces — the
17
+ // emulated `<thinking_mode>` tag block, which was all those models used to receive — measured
18
+ // 21,202 (`low`) and 28,302 (`max`) for luna, i.e. between that model's native `medium` and
19
+ // `high`, never reaching native `max`. `gpt-5.6-sol`'s native `max` cross-checked at 30,498 on the
20
+ // same prompt. Terra's absence from this table was therefore an omission rather than a capability
21
+ // difference: what the earlier Sol-only scope recorded was not reproducible here.
9
22
  export const KIRO_NATIVE_EFFORT_FIELDS: Record<string, "reasoning" | "output_config"> = {
10
23
  "gpt-5.6-sol": "reasoning",
24
+ "gpt-5.6-terra": "reasoning",
25
+ "gpt-5.6-luna": "reasoning",
11
26
  "claude-opus-5": "output_config",
12
27
  };
13
28
 
14
29
  export const KIRO_NATIVE_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
15
30
 
16
- export function kiroNativeEffortField(modelId: string): "reasoning" | "output_config" | undefined {
17
- return KIRO_NATIVE_EFFORT_FIELDS[normalizeKiroModelId(modelId)];
31
+ // The newly enabled models have evidence for these rungs only. Keep the previous
32
+ // emulation for xhigh, and never widen their native wire when the shared ladder grows.
33
+ const KIRO_LUNA_TERRA_NATIVE_EFFORTS = new Set(["low", "medium", "high", "max"]);
34
+
35
+ export function kiroNativeEffortField(
36
+ modelId: string,
37
+ effort?: string,
38
+ ): "reasoning" | "output_config" | undefined {
39
+ const model = normalizeKiroModelId(modelId);
40
+ if ((model === "gpt-5.6-luna" || model === "gpt-5.6-terra")
41
+ && effort !== undefined && !KIRO_LUNA_TERRA_NATIVE_EFFORTS.has(effort)) return undefined;
42
+ return KIRO_NATIVE_EFFORT_FIELDS[model];
18
43
  }
19
44
 
20
- export function kiroReasoningMode(modelId: string): KiroReasoningMode {
21
- return kiroNativeEffortField(modelId) ? "native" : "emulated";
45
+ export function kiroReasoningMode(modelId: string, effort?: string): KiroReasoningMode {
46
+ return kiroNativeEffortField(modelId, effort) ? "native" : "emulated";
22
47
  }
23
48
 
24
49
  export function kiroThinkingBudget(parsed: OcxParsedRequest): number | undefined {
@@ -38,7 +63,7 @@ export function kiroThinkingBudget(parsed: OcxParsedRequest): number | undefined
38
63
  }
39
64
 
40
65
  export function injectKiroThinkingTags(content: string, parsed: OcxParsedRequest): string {
41
- if (kiroReasoningMode(parsed.modelId) !== "emulated") return content;
66
+ if (kiroReasoningMode(parsed.modelId, parsed.options.reasoning) !== "emulated") return content;
42
67
  const budget = kiroThinkingBudget(parsed);
43
68
  if (!budget) return content;
44
69
  const instruction = [
@@ -54,3 +79,41 @@ export function injectKiroThinkingTags(content: string, parsed: OcxParsedRequest
54
79
  content,
55
80
  ].join("\n");
56
81
  }
82
+
83
+ /**
84
+ * The blob from a Kiro `reasoningContentEvent` has two possible homes on a replayed assistant
85
+ * turn, and the wire validates the SHAPE of each rather than its content: `signature` takes the
86
+ * emitted string verbatim, while `redactedContent` is a base64 member. The `.KTR~~…` value every
87
+ * GPT-5.6 capture returns is NOT valid base64, which is exactly why replaying it as
88
+ * `redactedContent` — what this proxy did before the field was measured — came back as
89
+ * REQUEST_BODY_INVALID ("Improperly formed request").
90
+ *
91
+ * The blob travels as ONE opaque string: adapter event, `ocxr1:` reasoning envelope, then
92
+ * `OcxAssistantMessage.kiroRedactedReasoning`. The field it arrived on therefore rides that same
93
+ * string, instead of a second parallel value that could drift from it. Provider data cannot forge
94
+ * the tag: the other channel is base64, whose alphabet has no colon.
95
+ */
96
+ export const KIRO_REASONING_SIGNATURE_TAG = "signature:";
97
+
98
+ export function tagKiroReasoningBlob(field: "signature" | "redactedContent", data: string): string {
99
+ return field === "signature" ? KIRO_REASONING_SIGNATURE_TAG + data : data;
100
+ }
101
+
102
+ /** The wire field a stored blob arrived on, and its untagged value. */
103
+ export function splitKiroReasoningBlob(value: string): { field: "signature" | "redactedContent"; data: string } {
104
+ return value.startsWith(KIRO_REASONING_SIGNATURE_TAG)
105
+ ? { field: "signature", data: value.slice(KIRO_REASONING_SIGNATURE_TAG.length) }
106
+ : { field: "redactedContent", data: value };
107
+ }
108
+
109
+ /**
110
+ * The `reasoningContent` object on an `assistantResponseMessage`. Exactly one member is set: the
111
+ * wire validates the shape, so the two cannot be substituted for each other.
112
+ */
113
+ export type KiroReasoningContent = { signature: string } | { redactedContent: string };
114
+
115
+ /** `reasoningContent` for a replayed `assistantResponseMessage`, carrying the blob verbatim. */
116
+ export function kiroReasoningContent(value: string): KiroReasoningContent {
117
+ const { field, data } = splitKiroReasoningBlob(value);
118
+ return field === "signature" ? { signature: data } : { redactedContent: data };
119
+ }
@@ -19,6 +19,7 @@ import { noteKiroTransientThrottle } from "../kiro-retry";
19
19
  import { KiroThinkingParser } from "../kiro-thinking";
20
20
  import { isCompleteKiroToolInput, kiroTruncationErrorMessage } from "../kiro-truncation";
21
21
  import { isValidKiroConversationId } from "../kiro-wire";
22
+ import { tagKiroReasoningBlob } from "./reasoning";
22
23
  import { estimateKiroTokens, kiroUpstreamContextWindow } from "./usage";
23
24
 
24
25
  // Stream parsing (shared by parseStream + parseResponse)
@@ -633,8 +634,13 @@ async function* parseKiroAttemptEvents(
633
634
  if (ev.data) {
634
635
  yield* emitRetained(stage({ type: "reasoning_raw_delta", text: ev.data }));
635
636
  }
636
- if (ev.redactedContent) {
637
- yield* emitRetained(stage({ type: "kiro_redacted_reasoning", data: ev.redactedContent }));
637
+ // The blob is replayed on the field it arrived on, so remember that field here — this is
638
+ // the only place that still knows it. See kiro/reasoning.ts for why the distinction is
639
+ // load-bearing rather than cosmetic.
640
+ if (ev.signature) {
641
+ yield* emitRetained(stage({ type: "kiro_redacted_reasoning", data: tagKiroReasoningBlob("signature", ev.signature) }));
642
+ } else if (ev.redactedContent) {
643
+ yield* emitRetained(stage({ type: "kiro_redacted_reasoning", data: tagKiroReasoningBlob("redactedContent", ev.redactedContent) }));
638
644
  }
639
645
  break;
640
646
  case "context_usage":
@@ -1,5 +1,6 @@
1
1
  import type { OcxProviderConfig } from "../../types";
2
2
  import type { KiroImage } from "../kiro-images";
3
+ import type { KiroReasoningContent } from "./reasoning";
3
4
 
4
5
  export const AMZ_TARGET = "AmazonCodeWhispererStreamingService.GenerateAssistantResponse";
5
6
  export const SDK_VERSION = "1.0.27";
@@ -51,7 +52,7 @@ export interface KiroHistoryEntry {
51
52
  assistantResponseMessage?: {
52
53
  content: string;
53
54
  toolUses?: KiroToolUse[];
54
- reasoningContent?: { redactedContent: string };
55
+ reasoningContent?: KiroReasoningContent;
55
56
  };
56
57
  }
57
58
 
@@ -3,7 +3,7 @@ import { kiroTruncationReason } from "./kiro-truncation";
3
3
 
4
4
  export type ParsedKiroEvent =
5
5
  | { type: "content"; data?: string; modelId?: string }
6
- | { type: "reasoning"; data?: string; redactedContent?: string }
6
+ | { type: "reasoning"; data?: string; signature?: string; redactedContent?: string }
7
7
  | { type: "context_usage"; contextUsagePercentage: number }
8
8
  | { type: "tool"; name?: string; toolUseId?: string; input?: string; stop?: boolean }
9
9
  | { type: "truncation"; data: string }
@@ -138,18 +138,26 @@ export function parseKiroEvent(eventType: string, payload: Uint8Array): ParsedKi
138
138
  : {}),
139
139
  };
140
140
  case "reasoningContentEvent":
141
- // `text` is plaintext reasoning; `redactedContent` is the encrypted blob the GPT-5.6 family
142
- // (sol/terra/luna) actually returns — they never send `text`. Keyed off the wire field, not
143
- // the model id. Both may be absent on a bare event.
144
- return {
145
- type: "reasoning",
146
- ...(optionalString(eventType, parsed, "text") !== undefined
147
- ? { data: optionalString(eventType, parsed, "text") }
148
- : {}),
149
- ...(optionalString(eventType, parsed, "redactedContent") !== undefined
150
- ? { redactedContent: optionalString(eventType, parsed, "redactedContent") }
151
- : {}),
152
- };
141
+ // `text` is plaintext reasoning; the GPT-5.6 family (sol/terra/luna) instead returns an
142
+ // encrypted blob, and the field it arrives on has to be replayed unchanged (see
143
+ // kiro/reasoning.ts): `signature` carries the `.KTR~~…` value verbatim and is what every
144
+ // capture of those models sent, while `redactedContent` — the base64 shape a capture has
145
+ // never shown — stays accepted for any model that sends it. Keyed off the wire field, not the
146
+ // model id. Any of the three may be absent on a bare event.
147
+ {
148
+ const text = optionalString(eventType, parsed, "text");
149
+ const signature = optionalString(eventType, parsed, "signature");
150
+ const redacted = optionalString(eventType, parsed, "redactedContent");
151
+ return {
152
+ type: "reasoning",
153
+ ...(text !== undefined ? { data: text } : {}),
154
+ ...(signature !== undefined
155
+ ? { signature }
156
+ : redacted !== undefined
157
+ ? { redactedContent: redacted }
158
+ : {}),
159
+ };
160
+ }
153
161
  case "toolUseEvent":
154
162
  return {
155
163
  type: "tool",
@@ -6,6 +6,8 @@ import { recordOwnedConfigPath } from "../lib/config-ownership";
6
6
  import type { OcxProviderConfig, OcxParsedRequest } from "../types";
7
7
  import { createOpenAIChatAdapter } from "./openai-chat";
8
8
  import type { ProviderAdapter, AdapterRequest, IncomingMeta } from "./base";
9
+ import { createAdapterPhysicalSend } from "./physical-send";
10
+ import { SendBudgetExhaustedError } from "../lib/upstream-retry";
9
11
 
10
12
  const BOOTSTRAP_URL = "https://api.xiaomimimo.com/api/free-ai/bootstrap";
11
13
  export const MIMO_CHAT_URL = "https://api.xiaomimimo.com/api/free-ai/openai/chat";
@@ -248,33 +250,46 @@ export function createMimoFreeAdapter(provider: OcxProviderConfig): ProviderAdap
248
250
  },
249
251
 
250
252
  async fetchResponse(request: AdapterRequest, ctx): Promise<Response> {
251
- const response = await fetch(request.url, {
253
+ const send = createAdapterPhysicalSend(ctx);
254
+ const response = await send({ url: request.url, dispatch: executor => executor(request.url, {
252
255
  method: request.method,
253
256
  redirect: "manual",
254
257
  headers: request.headers as Record<string, string>,
255
258
  body: request.body,
256
259
  signal: ctx?.abortSignal,
257
- });
260
+ }) });
258
261
 
259
262
  // Retry predicate: 401 (expired/invalid JWT) retries ONCE with a fresh token.
260
263
  // 403 is NOT retried — Xiaomi uses it for anti-abuse "Illegal access" and there is
261
264
  // no documented token-expiry signature that would mark a 403 as retryable.
262
265
  if (response.status === 401) {
263
- // Drain the first response body before issuing the retry.
264
- try { await response.body?.cancel(); } catch { /* already consumed */ }
265
- resetMimoJwtCache();
266
- const freshJwt = await getMimoJwt(ctx?.abortSignal);
267
- const retryHeaders = {
268
- ...(request.headers as Record<string, string>),
269
- "Authorization": `Bearer ${freshJwt}`,
270
- };
271
- return fetch(request.url, {
272
- method: request.method,
273
- redirect: "manual",
274
- headers: retryHeaders,
275
- body: request.body,
276
- signal: ctx?.abortSignal,
277
- });
266
+ let retryHeaders = request.headers;
267
+ try {
268
+ return await send({ url: request.url, sendClass: "auth-recovery", recovery: "oauth-401",
269
+ beforeDispatch: async () => {
270
+ // Drain the first response body and refresh the JWT only after admission: a
271
+ // refused replay still returns THIS response to the caller, body intact.
272
+ // Draining comes first within the block because getMimoJwt issues its own
273
+ // network call and may throw, and the 401 body would then never be released.
274
+ try { void response.body?.cancel().catch(() => {}); } catch { /* already consumed */ }
275
+ resetMimoJwtCache();
276
+ const freshJwt = await getMimoJwt(ctx?.abortSignal);
277
+ retryHeaders = {
278
+ ...(request.headers as Record<string, string>),
279
+ "Authorization": `Bearer ${freshJwt}`,
280
+ };
281
+ },
282
+ dispatch: executor => executor(request.url, {
283
+ method: request.method,
284
+ redirect: "manual",
285
+ headers: retryHeaders,
286
+ body: request.body,
287
+ signal: ctx?.abortSignal,
288
+ }) });
289
+ } catch (error) {
290
+ if (error instanceof SendBudgetExhaustedError) return response;
291
+ throw error;
292
+ }
278
293
  }
279
294
 
280
295
  return response;
@@ -306,17 +306,37 @@ function buildNativeMessages(
306
306
  // owned by this adapter/request lifecycle rather than process-global state.
307
307
  reservedToolCallIds.clear();
308
308
  let pending: PendingToolBatch | undefined;
309
+ // Codex records mid-turn injections (a PostToolUse hook verdict, a context notice) between an
310
+ // assistant tool call and that call's own tool result. Native Ollama needs the call and its
311
+ // results adjacent, so those conversational messages wait here instead of closing the batch
312
+ // early. The openai-chat adapter defers them the same way; refusing the replay killed the turn.
313
+ let deferred: OllamaNativeMessage[] = [];
314
+
315
+ const releaseDeferred = (): void => {
316
+ if (deferred.length === 0) return;
317
+ messages.push(...deferred);
318
+ deferred = [];
319
+ };
309
320
 
310
321
  const flushPending = (): void => {
311
322
  if (!pending) return;
312
323
  for (const call of pending.calls) {
313
324
  if (!call.result) {
314
- throw new Error(`ollama-native tool call ${call.id} is missing its tool result; refusing interrupted replay`);
325
+ // No result exists anywhere in the replayed history: the turn was interrupted, or the
326
+ // result never reached it. State exactly that instead of inventing an outcome, and keep
327
+ // the conversation replayable.
328
+ messages.push({
329
+ role: "tool",
330
+ tool_call_id: call.id,
331
+ tool_name: call.wireName,
332
+ // Same marker text as the chat adapter (openai-chat/messages.ts), so both adapters read
333
+ // the same in an operator's log. The name is this wire's flattened tool name, which is
334
+ // what the assistant turn above it carries.
335
+ content: `[ocx] no tool result was recorded for "${call.wireName}"; execution status unknown — do not treat this as success, failure, or user-provided input.`,
336
+ });
337
+ continue;
315
338
  }
316
- }
317
- for (const call of pending.calls) {
318
- const result = call.result!;
319
- const translated = contentToNative(result.content, "tool result");
339
+ const translated = contentToNative(call.result.content, "tool result");
320
340
  messages.push({
321
341
  role: "tool",
322
342
  tool_call_id: call.id,
@@ -326,6 +346,7 @@ function buildNativeMessages(
326
346
  });
327
347
  }
328
348
  pending = undefined;
349
+ releaseDeferred();
329
350
  };
330
351
 
331
352
  for (const message of parsed.context.messages) {
@@ -347,9 +368,22 @@ function buildNativeMessages(
347
368
  continue;
348
369
  }
349
370
 
350
- // Native Ollama requires the whole assistant tool-call turn followed by its tool results. A
351
- // new conversational message is a hard boundary; unresolved calls are never fabricated.
352
- if (pending) flushPending();
371
+ // Native Ollama requires the whole assistant tool-call turn followed by its tool results. A
372
+ // conversational message that arrives while the batch is still open is held aside instead of
373
+ // closing it, so the call keeps its results adjacent; it is released right after the batch
374
+ // flushes. Anything else (a new assistant turn) settles the batch first.
375
+ if (pending) {
376
+ if (message.role === "user" || message.role === "developer") {
377
+ const translated = message.role === "user"
378
+ ? contentToNative(message.content, "user")
379
+ : contentToNative(message.content, "developer", false);
380
+ deferred.push(message.role === "user"
381
+ ? { role: "user", content: translated.content, ...(translated.images ? { images: translated.images } : {}) }
382
+ : { role: "system", content: translated.content });
383
+ continue;
384
+ }
385
+ flushPending();
386
+ }
353
387
 
354
388
  switch (message.role) {
355
389
  case "user": {