@bitkyc08/opencodex 2.58.0 → 2.59.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. package/README.md +28 -10
  2. package/gui/dist/assets/index-C5IebErG.js +136 -0
  3. package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
  4. package/gui/dist/index.html +2 -2
  5. package/gui/dist/provider-icons/crusoe.svg +1 -0
  6. package/gui/dist/provider-icons/opper.svg +3 -0
  7. package/package.json +1 -1
  8. package/src/adapters/base.ts +11 -1
  9. package/src/adapters/cursor/catalog.ts +11 -0
  10. package/src/adapters/cursor/effort-map.ts +16 -2
  11. package/src/adapters/cursor/envelope-echo.ts +55 -2
  12. package/src/adapters/cursor/message-mapper.ts +3 -2
  13. package/src/adapters/cursor/protobuf-request.ts +8 -5
  14. package/src/adapters/cursor/request-builder.ts +14 -3
  15. package/src/adapters/cursor/thread-continuity.ts +105 -31
  16. package/src/adapters/cursor/tool-guidance.ts +5 -4
  17. package/src/adapters/cursor.ts +42 -1
  18. package/src/adapters/devin/cloud-direct/chat.ts +11 -2
  19. package/src/adapters/devin/cloud-direct/index.ts +7 -0
  20. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
  21. package/src/adapters/devin.ts +75 -13
  22. package/src/adapters/google-antigravity-wire.ts +29 -2
  23. package/src/adapters/google-http.ts +8 -1
  24. package/src/adapters/google.ts +23 -4
  25. package/src/adapters/openai-chat/response-events.ts +61 -0
  26. package/src/adapters/openai-chat.ts +5 -10
  27. package/src/adapters/openai-responses/passthrough.ts +10 -1
  28. package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
  29. package/src/adapters/openai-responses/tool-schema.ts +19 -7
  30. package/src/adapters/responses-tool-schema.ts +76 -46
  31. package/src/adapters/run-turn-queue.ts +17 -4
  32. package/src/bridge/response-json.ts +1 -1
  33. package/src/bridge/sse.ts +165 -24
  34. package/src/claude/context-windows.ts +22 -0
  35. package/src/claude/outbound.ts +35 -4
  36. package/src/cli/account-api.ts +4 -3
  37. package/src/cli/account-extended.ts +22 -2
  38. package/src/cli/account-orca-import.ts +63 -0
  39. package/src/cli/account.ts +32 -4
  40. package/src/cli/capabilities.ts +40 -0
  41. package/src/cli/claude.ts +29 -1
  42. package/src/cli/codex-cli-update.ts +97 -2
  43. package/src/cli/dispatch.ts +54 -0
  44. package/src/cli/doctor.ts +197 -2
  45. package/src/cli/help.ts +4 -1
  46. package/src/cli/index.ts +88 -20
  47. package/src/cli/models-runtime.ts +33 -4
  48. package/src/cli/registry.ts +11 -1
  49. package/src/cli/runtime-api.ts +44 -0
  50. package/src/cli/start-args.ts +94 -0
  51. package/src/cli/system-command.ts +2 -0
  52. package/src/client/machine-api.ts +4 -3
  53. package/src/client/machine-listener.ts +14 -1
  54. package/src/clients/config-export/constants.ts +2 -3
  55. package/src/clients/config-export.ts +5 -5
  56. package/src/codex/account-store.ts +81 -5
  57. package/src/codex/auth-api/pool-quota-probe.ts +14 -3
  58. package/src/codex/auth-api/routes.ts +17 -2
  59. package/src/codex/auth-context.ts +16 -12
  60. package/src/codex/catalog/build-entries.ts +25 -4
  61. package/src/codex/catalog/derive-entry.ts +8 -1
  62. package/src/codex/catalog/effort.ts +10 -6
  63. package/src/codex/catalog/gather-capture.ts +1 -0
  64. package/src/codex/catalog/model-hints.ts +37 -5
  65. package/src/codex/catalog/parsing.ts +83 -5
  66. package/src/codex/catalog/reserve-warn.ts +96 -0
  67. package/src/codex/catalog/retained-sync.ts +19 -0
  68. package/src/codex/catalog/routed-gather.ts +42 -3
  69. package/src/codex/cli-installation-identity.ts +210 -0
  70. package/src/codex/cli-installation-targets.ts +158 -0
  71. package/src/codex/convergence.ts +5 -0
  72. package/src/codex/history-provider.ts +4 -1
  73. package/src/codex/history-state-open.ts +105 -0
  74. package/src/codex/inject/config-toml.ts +44 -2
  75. package/src/codex/inject.ts +3 -2
  76. package/src/codex/lineage.ts +83 -32
  77. package/src/codex/loopback-target.ts +31 -0
  78. package/src/codex/main-account-hard-lock.ts +2 -1
  79. package/src/codex/main-account.ts +10 -3
  80. package/src/codex/main-device-reauth.ts +17 -9
  81. package/src/codex/model-entitlements.ts +60 -1
  82. package/src/codex/observed-model-denials.ts +137 -0
  83. package/src/codex/orca-auth-source.ts +94 -0
  84. package/src/codex/orca-import.ts +219 -0
  85. package/src/codex/prompt-text-probe.ts +282 -12
  86. package/src/codex/quota-401-recovery.ts +12 -0
  87. package/src/codex/quota-types.ts +65 -0
  88. package/src/codex/quota.ts +24 -19
  89. package/src/codex/routing/cooldown-math.ts +8 -47
  90. package/src/codex/routing/pin-drain.ts +57 -0
  91. package/src/codex/routing.ts +13 -15
  92. package/src/codex/subagent-model-fallback.ts +94 -0
  93. package/src/codex/windows-installation-files.ts +224 -0
  94. package/src/combos/failover.ts +122 -5
  95. package/src/config/diagnostics.ts +21 -0
  96. package/src/config/load-degrade.ts +15 -0
  97. package/src/config/pending-teardown.ts +8 -0
  98. package/src/config/process-state.ts +36 -3
  99. package/src/config/provider-relative-send-path.ts +16 -0
  100. package/src/config/proxy-env.ts +23 -5
  101. package/src/config/schema/config-schema.ts +21 -0
  102. package/src/config/schema/leaf-validators.ts +64 -17
  103. package/src/generated/compatibility-version.json +235 -163
  104. package/src/generated/model-metadata.ts +1 -1
  105. package/src/lib/bounded-body.ts +4 -2
  106. package/src/lib/destination-policy.ts +48 -6
  107. package/src/lib/errors.ts +3 -15
  108. package/src/lib/local-destinations.ts +32 -5
  109. package/src/lib/provider-outbound.ts +3 -3
  110. package/src/lib/proxy-env.ts +70 -3
  111. package/src/lib/request-execution-budget.ts +11 -3
  112. package/src/lib/response-body-inactivity.ts +193 -0
  113. package/src/lib/retry-delay.ts +69 -0
  114. package/src/lib/socks5-fetch.ts +631 -0
  115. package/src/lib/spend-reservation-ledger.ts +115 -9
  116. package/src/lib/workflow-budget.ts +145 -8
  117. package/src/oauth/account-quota-rank.ts +72 -15
  118. package/src/oauth/generic-account-failover.ts +40 -27
  119. package/src/oauth/orcarouter.ts +15 -2
  120. package/src/oauth/store.ts +8 -0
  121. package/src/providers/codex-capacity.ts +9 -0
  122. package/src/providers/devin-provider-merge-migration.ts +33 -12
  123. package/src/providers/free-directory.ts +20 -2
  124. package/src/providers/key-failover.ts +261 -7
  125. package/src/providers/model-rename-migration.ts +1 -0
  126. package/src/providers/openai-sidecar.ts +4 -0
  127. package/src/providers/opencode-go-transport.ts +14 -5
  128. package/src/providers/quota/report-cache.ts +3 -0
  129. package/src/providers/registry/entries-extended.ts +96 -0
  130. package/src/providers/registry/model-seeds.ts +78 -21
  131. package/src/responses/apply-patch-envelope.ts +44 -11
  132. package/src/responses/bridge-search-replay-cache.ts +152 -0
  133. package/src/responses/code-mode-helper-compat.ts +26 -16
  134. package/src/responses/custom-tool-compat.ts +1 -1
  135. package/src/responses/hosted-tool-policy.ts +85 -2
  136. package/src/responses/schema.ts +9 -2
  137. package/src/server/auth-cors.ts +26 -0
  138. package/src/server/chat-completions.ts +9 -4
  139. package/src/server/chat-native-sse.ts +26 -9
  140. package/src/server/chat-native.ts +10 -4
  141. package/src/server/claude-messages.ts +24 -2
  142. package/src/server/gui-static.ts +36 -2
  143. package/src/server/inbound-body-admission.ts +187 -0
  144. package/src/server/index.ts +15 -19
  145. package/src/server/management/api-access.ts +3 -4
  146. package/src/server/management/config-routes.ts +31 -6
  147. package/src/server/management/provider-capability-config.ts +35 -7
  148. package/src/server/management/provider-routes.ts +70 -18
  149. package/src/server/proxy-liveness.ts +97 -2
  150. package/src/server/relay.ts +17 -24
  151. package/src/server/request-log.ts +25 -1
  152. package/src/server/responses/adapter-continuation.ts +71 -27
  153. package/src/server/responses/adapter-delivery.ts +39 -8
  154. package/src/server/responses/adapter-dispatch.ts +52 -24
  155. package/src/server/responses/compact.ts +60 -11
  156. package/src/server/responses/core-codex-account.ts +83 -22
  157. package/src/server/responses/core-normalize.ts +12 -5
  158. package/src/server/responses/fetch-helpers.ts +68 -2
  159. package/src/server/responses/passthrough-delivery.ts +10 -1
  160. package/src/server/responses/passthrough-dispatch.ts +113 -48
  161. package/src/server/responses/passthrough-execution.ts +11 -1
  162. package/src/server/responses/request-prepare.ts +29 -0
  163. package/src/server/responses/request-send-budget.ts +84 -7
  164. package/src/server/responses/request-sidecar-auth.ts +16 -8
  165. package/src/server/responses/request-spend.ts +38 -9
  166. package/src/server/responses/request-transport.ts +13 -10
  167. package/src/server/responses/run-turn-execution.ts +20 -5
  168. package/src/server/responses/sidecar-execution.ts +2 -0
  169. package/src/server/responses/ws-upstream.ts +2 -1
  170. package/src/server/responses-custom-tool-repair.ts +2 -2
  171. package/src/server/sse-frame-buffer.ts +12 -10
  172. package/src/server/sse-payload-rewrite.ts +36 -9
  173. package/src/server/system-env-shell.ts +5 -1
  174. package/src/server/system-env.ts +7 -1
  175. package/src/server/workflow-refusal.ts +56 -2
  176. package/src/service/cli.ts +16 -6
  177. package/src/service/guards.ts +10 -0
  178. package/src/service/health.ts +43 -0
  179. package/src/service/state.ts +7 -2
  180. package/src/types/accounts.ts +4 -0
  181. package/src/types/config.ts +100 -3
  182. package/src/types/provider.ts +19 -0
  183. package/src/types/request.ts +7 -1
  184. package/src/types/wire.ts +9 -1
  185. package/src/usage/expected-prices.ts +28 -0
  186. package/src/usage/log.ts +87 -4
  187. package/src/web-search/passthrough-bridge.ts +39 -5
  188. package/gui/dist/assets/index-BbrHOIY0.js +0 -128
@@ -0,0 +1,103 @@
1
+ /**
2
+ * Same-target retry of an explicit, pre-output 429 refusal with a stated
3
+ * recovery delay. No event-producing attempt is ever automatically replayed.
4
+ * This is a refusal-specific policy, not a claim that every eventless POST
5
+ * is idempotent; ambiguous transport failures still propagate unchanged.
6
+ */
7
+ import { parseRetryAfterFromMessage } from '../../../lib/retry-delay.js';
8
+ import { abortError, sleepWithAbort } from '../../../lib/upstream-retry.js';
9
+ import { CloudChatError, streamChatEvents, type CloudChatEvent, type CloudChatRequest } from './chat.js';
10
+
11
+ /** 1 initial attempt plus at most 2 replays. */
12
+ export const STATED_RESET_MAX_REPLAYS = 2;
13
+ /** Default cumulative wait allowance for one invocation (30 minutes). */
14
+ export const STATED_RESET_MAX_WAIT_MS = 1_800_000;
15
+ /** Absolute maximum cumulative allowance, including explicit overrides. */
16
+ export const STATED_RESET_WAIT_CEILING_MS = 3_600_000;
17
+
18
+ function statedResetMaxWaitMs(): number {
19
+ const raw = process.env.OPENCODEX_DEVIN_STATED_RESET_WAIT_MS?.trim();
20
+ if (!raw) return STATED_RESET_MAX_WAIT_MS;
21
+ const parsed = Number(raw);
22
+ if (!Number.isFinite(parsed) || parsed < 0) return STATED_RESET_MAX_WAIT_MS;
23
+ // Zero explicitly disables local waiting. Values above one hour are capped.
24
+ return Math.min(Math.floor(parsed), STATED_RESET_WAIT_CEILING_MS);
25
+ }
26
+ export const statedResetMaxWaitMsForTests = statedResetMaxWaitMs;
27
+
28
+ export interface StatedResetRetryOptions {
29
+ /** Test seam: defaults to the real cloud stream. */
30
+ stream?: (req: CloudChatRequest) => AsyncGenerator<CloudChatEvent>;
31
+ /** Test seam: must either honour the whole delay or reject on cancellation. */
32
+ sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
33
+ maxReplays?: number;
34
+ /** CUMULATIVE wait allowance, not a fresh allowance on every failure. */
35
+ maxWaitMs?: number;
36
+ }
37
+
38
+ function replayLimit(value: number | undefined): number {
39
+ if (value === undefined) return STATED_RESET_MAX_REPLAYS;
40
+ if (!Number.isInteger(value) || value < 0 || value > STATED_RESET_MAX_REPLAYS) {
41
+ throw new RangeError('maxReplays must be an integer from 0 to 2');
42
+ }
43
+ return value;
44
+ }
45
+
46
+ function waitLimit(value: number | undefined): number {
47
+ if (value === undefined) return statedResetMaxWaitMs();
48
+ if (!Number.isFinite(value) || value < 0) {
49
+ throw new RangeError('maxWaitMs must be a finite non-negative number');
50
+ }
51
+ return Math.min(Math.floor(value), STATED_RESET_WAIT_CEILING_MS);
52
+ }
53
+
54
+ export async function* streamChatEventsWithResetRetry(
55
+ req: CloudChatRequest,
56
+ options?: StatedResetRetryOptions,
57
+ ): AsyncGenerator<CloudChatEvent> {
58
+ const stream = options?.stream ?? streamChatEvents;
59
+ const sleep = options?.sleep ?? sleepWithAbort;
60
+ const maxReplays = replayLimit(options?.maxReplays);
61
+ const maxWaitMs = waitLimit(options?.maxWaitMs);
62
+ let replays = 0;
63
+ let waitedMs = 0;
64
+ while (true) {
65
+ // Check again after sleeping: cancellation can race with timer completion.
66
+ // A pre-aborted request must not even enter a custom transport.
67
+ if (req.signal?.aborted) throw abortError(req.signal);
68
+ let yielded = false;
69
+ try {
70
+ for await (const event of stream(req)) {
71
+ // Latch before yielding, so a consumer-injected error is post-output.
72
+ yielded = true;
73
+ yield event;
74
+ }
75
+ return;
76
+ } catch (error) {
77
+ if (req.signal?.aborted) throw abortError(req.signal);
78
+ const waitSec = !yielded
79
+ && error instanceof CloudChatError
80
+ && error.status === 429
81
+ ? parseRetryAfterFromMessage(error.message)
82
+ : undefined;
83
+ const waitMs = waitSec === undefined ? undefined : waitSec * 1000;
84
+ if (
85
+ waitMs === undefined
86
+ || replays >= maxReplays
87
+ || waitMs > maxWaitMs - waitedMs
88
+ ) {
89
+ // Never shorten a provider's minimum delay to fit the local budget.
90
+ // Keep the original refusal so outer policy can preserve its metadata.
91
+ throw error;
92
+ }
93
+ replays += 1;
94
+ // Charge the complete scheduled wait once, before sleeping. This is a
95
+ // sleep allowance, not a wall-clock deadline on generation or timer
96
+ // scheduling: waking a few milliseconds late must not reject an already
97
+ // approved one-hour retry. No later wait can spend this allowance again.
98
+ waitedMs += waitMs;
99
+ await sleep(waitMs, req.signal);
100
+ if (req.signal?.aborted) throw abortError(req.signal);
101
+ }
102
+ }
103
+ }
@@ -9,9 +9,9 @@
9
9
  import type { AdapterEvent, OcxAssistantMessage, OcxContentPart, OcxMessage, OcxParsedRequest, OcxProviderConfig, OcxTool, OcxToolCall, OcxToolResultMessage, OcxUsage } from "../types";
10
10
  import { namespacedToolName } from "../types";
11
11
  import type { IncomingMeta, ProviderAdapter } from "./base";
12
- import { streamChatEvents, allocateCascadeId, CloudChatError, type ChatHistoryItem, type ToolDef } from "./devin/cloud-direct";
12
+ import { streamChatEventsWithResetRetry, allocateCascadeId, CloudChatError, type ChatHistoryItem, type ToolDef } from "./devin/cloud-direct";
13
13
  import type { ContentPart } from "./devin/cloud-direct/chat";
14
- import { getCachedCatalog } from "./devin/cloud-direct/catalog";
14
+ import { getCachedCatalog, type CacheEntry } from "./devin/cloud-direct/catalog";
15
15
  import { collapseDevinModelUid } from "./devin/live-models";
16
16
  import { buildNonOpenAIToolCatalogNudgeForTools } from "./tool-catalog-nudge";
17
17
  import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiServer } from "../oauth/devin";
@@ -155,6 +155,7 @@ async function resolveWireModelUid(
155
155
  apiKey: string,
156
156
  host: string,
157
157
  reasoningEffort?: string,
158
+ catalog?: CacheEntry | null,
158
159
  ): Promise<string> {
159
160
  const modelId = normalizeDevinModelId(rawModelId);
160
161
  // Explicit effort wins over a suffix the picker already baked into the id, so
@@ -163,15 +164,18 @@ async function resolveWireModelUid(
163
164
  const swe2 = resolveSwe2Variant(modelId, reasoningEffort);
164
165
  if (swe2) return swe2;
165
166
  if (hasEffortSuffix(modelId)) return modelId;
166
- const catalog = await getCachedCatalog(apiKey, host);
167
- if (catalog) {
168
- if (catalog.byUid.has(modelId)) return modelId;
167
+ // Callers that already read the catalog this turn pass it in; an explicit
168
+ // null records a failed lookup and must not trigger a same-turn retry —
169
+ // failures are not cached, so re-reading would only pay another timeout.
170
+ const entry = catalog !== undefined ? catalog : await getCachedCatalog(apiKey, host);
171
+ if (entry) {
172
+ if (entry.byUid.has(modelId)) return modelId;
169
173
  const effort = reasoningEffort && CALLER_EFFORT_VALUES.has(reasoningEffort) ? reasoningEffort : "medium";
170
174
  const suffixed = `${modelId}-${effort}`;
171
- if (catalog.byUid.has(suffixed)) return suffixed;
175
+ if (entry.byUid.has(suffixed)) return suffixed;
172
176
  // Fall back to any enabled variant of this base model.
173
- for (const uid of catalog.byUid.keys()) {
174
- if (uid.startsWith(modelId + "-") && !catalog.byUid.get(uid)?.disabled) return uid;
177
+ for (const uid of entry.byUid.keys()) {
178
+ if (uid.startsWith(modelId + "-") && !entry.byUid.get(uid)?.disabled) return uid;
175
179
  }
176
180
  }
177
181
  // Degraded mode: append the default effort suffix.
@@ -186,6 +190,46 @@ async function resolveWireModelUid(
186
190
  */
187
191
  export const resolveWireModelUidForTests = resolveWireModelUid;
188
192
 
193
+ /**
194
+ * Resolve the INPUT ceiling for the exact UID selected for this turn. Catalog
195
+ * ClientModelConfig #18 and CompletionConfiguration #3 both carry input tokens;
196
+ * the independent output cap is not subtracted here. Smaller operator hints
197
+ * cap live evidence, never enlarge it. No evidence leaves the encoder's 128k
198
+ * fallback intact; an unrelated or opt-in long-context variant is not evidence.
199
+ */
200
+ function resolveDevinMaxInputTokens(
201
+ provider: OcxProviderConfig,
202
+ modelUid: string,
203
+ liveWindow?: number,
204
+ ): number | undefined {
205
+ const positive = (value: unknown): number | undefined =>
206
+ typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : undefined;
207
+ const baseId = collapseDevinModelUid(modelUid);
208
+ const configured = (record: Record<string, number> | undefined): number | undefined => {
209
+ if (!record) return undefined;
210
+ for (const id of [modelUid, baseId]) {
211
+ // Prefer the canonical spelling; retain dotted/case-folded saved hints,
212
+ // matching the model-id normalization used for the inference request.
213
+ const exact = Object.hasOwn(record, id) ? positive(record[id]) : undefined;
214
+ if (exact !== undefined) return exact;
215
+ const matches = Object.entries(record)
216
+ .filter(([key]) => normalizeDevinModelId(key).toLowerCase() === id.toLowerCase())
217
+ .map(([, value]) => positive(value))
218
+ .filter((value): value is number => value !== undefined);
219
+ if (matches.length > 0) return Math.min(...matches);
220
+ }
221
+ return undefined;
222
+ };
223
+ const contextHint = configured(provider.modelContextWindows) ?? positive(provider.contextWindow);
224
+ const inputHint = configured(provider.modelMaxInputTokens);
225
+ const ceilings = [positive(liveWindow), contextHint, inputHint]
226
+ .filter((value): value is number => value !== undefined);
227
+ return ceilings.length > 0 ? Math.min(...ceilings) : undefined;
228
+ }
229
+
230
+ /** Pure test seam; runtime uses the same resolver immediately before dispatch. */
231
+ export const resolveDevinMaxInputTokensForTests = resolveDevinMaxInputTokens;
232
+
189
233
  export class DevinMissingCredentialError extends Error {
190
234
  constructor() {
191
235
  super("Devin live transport requires a Devin API key. Run ocx login devin to sign in with your Cognition/Devin account.");
@@ -493,7 +537,16 @@ export function createDevinAdapter(
493
537
  // entry: an EU or FedStart account that used provider.baseUrl would send
494
538
  // every RPC to the US server it is not provisioned on.
495
539
  const host = resolveDevinApiServer(provider.baseUrl, credentialProviderId);
496
- const modelUid = await resolveWireModelUid(rawModelId, apiKey, host, parsed.options.reasoning);
540
+ // One catalog read per turn serves model-UID resolution, the input
541
+ // ceiling, and the chat pre-flight inside streamChatEvents. Failures are
542
+ // not cached, so a second read would only pay another fetch timeout on
543
+ // an otherwise valid turn.
544
+ const catalog = await getCachedCatalog(apiKey, host, incoming.abortSignal);
545
+ if (incoming.abortSignal?.aborted) {
546
+ emit({ type: "error", message: DEVIN_CLIENT_CLOSED_MESSAGE, status: 499, retryable: false });
547
+ return;
548
+ }
549
+ const modelUid = await resolveWireModelUid(rawModelId, apiKey, host, parsed.options.reasoning, catalog);
497
550
  const returnedToolNames = buildDevinReturnedToolNameMap(parsed.context.tools);
498
551
  let openToolId: string | undefined;
499
552
  let usage: OcxUsage | undefined;
@@ -506,17 +559,26 @@ export function createDevinAdapter(
506
559
  };
507
560
 
508
561
  try {
509
- for await (const event of streamChatEvents({
562
+ // Read the selected UID's catalog row, not the picker's collapsed base.
563
+ const maxInputTokens = resolveDevinMaxInputTokens(
564
+ provider, modelUid, catalog?.byUid.get(modelUid)?.contextWindow,
565
+ );
566
+ // The reset-retry wrapper waits out a 429 that states its own recovery
567
+ // delay ("limit will reset in 35 seconds") and replays the identical
568
+ // request — but only while zero events have been yielded, so a
569
+ // post-output failure still takes the terminal path untouched.
570
+ for await (const event of streamChatEventsWithResetRetry({
510
571
  apiKey,
511
572
  apiServerUrl: host,
512
573
  modelUid,
574
+ catalog,
513
575
  messages: mapOcxMessagesToDevin(parsed),
514
576
  tools: mapOcxToolsToDevin(parsed.context.tools),
515
577
  cascadeId,
516
- // Without these the request falls back to the encoder's defaults
517
- // (8192 output, a 128k context window, temperature 0.7), so a client
518
- // that asked for a 4k cap never got one.
578
+ // Input and output ceilings are separate wire fields. Omitting the
579
+ // input hint used to force every model through the 128k default.
519
580
  completionOpts: {
581
+ ...(maxInputTokens !== undefined ? { maxInputTokens } : {}),
520
582
  ...(typeof parsed.options.maxOutputTokens === "number" ? { maxOutputTokens: parsed.options.maxOutputTokens } : {}),
521
583
  ...(typeof parsed.options.temperature === "number" ? { temperature: parsed.options.temperature } : {}),
522
584
  ...(typeof parsed.options.topP === "number" ? { topP: parsed.options.topP } : {}),
@@ -97,8 +97,35 @@ export function antigravitySessionId(parsed: OcxParsedRequest): string {
97
97
  * collision, is the failure mode this function exists to prevent.
98
98
  */
99
99
  function clientThreadAnchor(parsed: OcxParsedRequest): string | undefined {
100
- const threadId = parsed._clientThreadId?.trim();
101
- return threadId ? `codex-thread:${threadId}` : undefined;
100
+ // `_clientThreadId` carries `x-codex-parent-thread-id`, which every parallel child of one
101
+ // parent presents identically, so anchoring on it alone collapsed concurrent children onto a
102
+ // single upstream Cloud Code Assist session (#5033).
103
+ //
104
+ // A thread id is only unique WITHIN its parent, which is why `codexConversationIdentity` keys
105
+ // on both. #5054 anchored on the child alone and therefore only moved the collision: two
106
+ // parents can each have a child of the same id (#5058). The pair is the identity, joined by a
107
+ // NUL so the encoding is injective: no Codex id contains one, so a pair cannot be re-read as a
108
+ // different pair, nor as the parent-only anchor below.
109
+ //
110
+ // Deliberately NOT the general lane key: `codexConversationKeyFor` is an HMAC under a
111
+ // process-random secret, so it changes across a proxy restart — and instability, not sharing,
112
+ // is the failure mode this derivation has to avoid. These ids are Codex's own values and
113
+ // survive both compaction and restart.
114
+ const own = parsed._codexOwnThreadId?.trim();
115
+ const parent = parsed._clientThreadId?.trim();
116
+ if (own && parent) return `codex-thread:${parent}\u0000${own}`;
117
+ // Parent-only clients keep the anchor they already had.
118
+ if (parent) return `codex-thread:${parent}`;
119
+ // A parentless ROOT deliberately omits the parent header. `src/server/context-history.ts` says
120
+ // so in as many words: root model requests use (session-id=root, thread-id=root) and do not
121
+ // fabricate a parent key. It has no pair to key on, and #5054's claim that a root presents
122
+ // `thread-id` equal to its parent was simply wrong.
123
+ //
124
+ // So it keeps the pre-#5054 anchor rather than gaining an own-thread one. That is not a
125
+ // preference: durable Antigravity replay state is keyed by model plus session id, and moving a
126
+ // root's anchor on upgrade strands every signature stored under the old session — the exact
127
+ // instability this derivation exists to avoid, introduced while fixing sharing.
128
+ return undefined;
102
129
  }
103
130
 
104
131
  /** A Gemini content part as it appears in an Antigravity request body. */
@@ -112,7 +112,14 @@ export async function fetchGoogleWithRetry(
112
112
  } catch (err) {
113
113
  if (ctx.abortSignal?.aborted) throw err;
114
114
  if (err instanceof SendBudgetExhaustedError) {
115
- if (pendingResponse) return ctx.returnRawErrors ? pendingResponse : normalizeFinalGoogleError(label, pendingResponse, ctx.abortSignal);
115
+ if (pendingResponse) {
116
+ // The ladder had already classified this response as retryable and was about to send
117
+ // again; the budget refused. Returning the original response is right — it is a real
118
+ // upstream answer — but it used to leave the log indistinguishable from a request
119
+ // where no retry was ever eligible (#5044).
120
+ ctx.onRecoveryWithheld?.({ reason: "retry-send-budget" });
121
+ return ctx.returnRawErrors ? pendingResponse : normalizeFinalGoogleError(label, pendingResponse, ctx.abortSignal);
122
+ }
116
123
  throw err;
117
124
  }
118
125
  lastError = err;
@@ -57,7 +57,6 @@ const GOOGLE_BREVITY_INSTRUCTION = [
57
57
 
58
58
  const ANTIGRAVITY_REJECTED_CLAUDE_SDK_PARAGRAPH =
59
59
  "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
60
-
61
60
  /**
62
61
  * CCA Flash generations that reject the Claude-Agent identity paragraph.
63
62
  *
@@ -102,6 +101,20 @@ function stripAntigravityRejectedClaudeSdkParagraph(systemText: string): string
102
101
  .join("\n\n");
103
102
  }
104
103
 
104
+ /**
105
+ * Strips Claude Code CLI's internal billing header (`x-anthropic-billing-header: ...`)
106
+ * at the start of the system prompt, because Cloud Code Assist / Google Antigravity inspects
107
+ * `systemInstruction` and rejects requests containing Anthropic billing metadata with
108
+ * HTTP 429 RESOURCE_EXHAUSTED.
109
+ *
110
+ * Matching is restricted to the prompt start (`^` without the `/m` multiline flag) so that
111
+ * user prompts discussing billing headers in intermediate lines are never modified, and
112
+ * prompts without a billing header preserve their leading whitespace untouched.
113
+ */
114
+ function stripAntigravityBillingHeader(systemText: string): string {
115
+ return systemText.replace(/^x-anthropic-billing-header:[^\n]*\n*/, "");
116
+ }
117
+
105
118
  /**
106
119
  * Documented output ceiling for a Google-surface model, or `undefined` when the id is not
107
120
  * recognized.
@@ -276,6 +289,7 @@ function messagesToGeminiFormat(
276
289
  parsed: OcxParsedRequest,
277
290
  identityModelId: string,
278
291
  stripRejectedClaudeSdkParagraph = false,
292
+ isCloudCodeAssist = false,
279
293
  ): { systemInstruction?: unknown; contents: unknown[]; replayedCallIds: string[] } {
280
294
  // Neutralize Codex's GPT-5 identity line (Gemini/Antigravity share this path) so a routed model
281
295
  // never misreports as GPT-5/OpenAI, and never leaks the proxy identity upstream.
@@ -285,9 +299,12 @@ function messagesToGeminiFormat(
285
299
  ...(toolCatalogNudge ? [toolCatalogNudge] : []),
286
300
  GOOGLE_BREVITY_INSTRUCTION,
287
301
  ].join("\n\n"), identityModelId);
288
- const systemText = stripRejectedClaudeSdkParagraph
289
- ? stripAntigravityRejectedClaudeSdkParagraph(identifiedSystemText)
302
+ let systemText = isCloudCodeAssist
303
+ ? stripAntigravityBillingHeader(identifiedSystemText)
290
304
  : identifiedSystemText;
305
+ if (stripRejectedClaudeSdkParagraph) {
306
+ systemText = stripAntigravityRejectedClaudeSdkParagraph(systemText);
307
+ }
291
308
  const systemInstruction = { parts: [{ text: systemText }] };
292
309
 
293
310
  const contents: unknown[] = [];
@@ -833,12 +850,14 @@ export function createGoogleAdapter(provider: OcxProviderConfig): ProviderAdapte
833
850
  && /^gemini-/.test(routedModelId) && !isImageCapableModel(parsed.modelId);
834
851
  // AI Studio's `-tiered` spelling is wire-only; CCA aliases may migrate to another generation.
835
852
  const identityModelId = provider.googleMode === "cloud-code-assist" ? routedModelId : parsed.modelId;
836
- const stripRejectedClaudeSdkParagraph = provider.googleMode === "cloud-code-assist"
853
+ const isCloudCodeAssist = provider.googleMode === "cloud-code-assist";
854
+ const stripRejectedClaudeSdkParagraph = isCloudCodeAssist
837
855
  && rejectsClaudeSdkParagraph(parsed.modelId, routedModelId);
838
856
  const { systemInstruction, contents, replayedCallIds } = messagesToGeminiFormat(
839
857
  parsed,
840
858
  identityModelId,
841
859
  stripRejectedClaudeSdkParagraph,
860
+ isCloudCodeAssist,
842
861
  );
843
862
  lastInjectedCallIds = [...replayedCallIds];
844
863
  lastReasoningReplayScope = parsed._reasoningReplayScope;
@@ -1,4 +1,5 @@
1
1
  import { diagnoseInvalidToolCalls, isRecord, type InvalidToolCallDiagnostic } from "./tool-call-validation";
2
+ import { TranslatorBudgetExceededError, type TranslatorBudget } from "../../lib/translator-budget";
2
3
  import type { AdapterEvent, OcxUsage } from "../../types";
3
4
 
4
5
  export function stopReasonFor(finishReason: unknown): "max_tokens" | "content_filter" | undefined {
@@ -22,6 +23,9 @@ export interface ReasoningDetailSegment {
22
23
  text: string;
23
24
  }
24
25
 
26
+ const MAX_REASONING_DETAIL_KEY_BYTES = 1024;
27
+ const MAX_REASONING_DETAIL_SEGMENTS = 1024;
28
+
25
29
  /**
26
30
  * Structured `reasoning_details` array (MiniMax M-series with `reasoning_split`).
27
31
  * Each segment's key scopes cumulative-snapshot tracking: upstream repeats the
@@ -35,6 +39,11 @@ export function reasoningDetailSegmentsFrom(record: Record<string, unknown>): Re
35
39
  const item: unknown = raw[i];
36
40
  if (!isRecord(item)) continue;
37
41
  if (typeof item.text !== "string" || item.text.length === 0) continue;
42
+ // Parsing retains nothing, so it rejects nothing. The non-streaming
43
+ // parseResponse path shares this function and reads only `text`; failing a
44
+ // whole valid response there because an opaque upstream id is long would be
45
+ // a new rejection unrelated to the retention bound. The key cap lives with
46
+ // the map that holds the key; see the tracker below.
38
47
  const key = typeof item.id === "string" && item.id.length > 0
39
48
  ? `id:${item.id}`
40
49
  : typeof item.index === "number"
@@ -45,6 +54,58 @@ export function reasoningDetailSegmentsFrom(record: Record<string, unknown>): Re
45
54
  return segments;
46
55
  }
47
56
 
57
+ /**
58
+ * Per-stream cumulative-snapshot store for structured `reasoning_details`. Each stream chunk
59
+ * repeats a detail's full text-so-far, so deltas are derived by prefix-diffing per segment key;
60
+ * a piece that does not extend the previous snapshot is appended whole, which keeps incremental
61
+ * senders parseable on the same path. Retained key+text bytes are charged to the translator
62
+ * budget under the `reasoning` kind, and both the key length and the segment count are capped
63
+ * here, where the map actually retains them, so a hostile upstream cannot grow it without bound.
64
+ */
65
+ export function createReasoningDetailSnapshotTracker(budget: TranslatorBudget): {
66
+ ingest(segment: ReasoningDetailSegment): string | null;
67
+ release(): void;
68
+ } {
69
+ const snapshots = new Map<string, string>();
70
+ const encoder = new TextEncoder();
71
+ let retainedBytes = 0;
72
+ return {
73
+ ingest(segment) {
74
+ const existing = snapshots.get(segment.key);
75
+ if (existing === undefined && encoder.encode(segment.key).byteLength > MAX_REASONING_DETAIL_KEY_BYTES) {
76
+ throw new TranslatorBudgetExceededError("reasoning", MAX_REASONING_DETAIL_KEY_BYTES);
77
+ }
78
+ if (existing === undefined && snapshots.size >= MAX_REASONING_DETAIL_SEGMENTS) {
79
+ throw new TranslatorBudgetExceededError("reasoning", MAX_REASONING_DETAIL_SEGMENTS);
80
+ }
81
+ const prev = existing ?? "";
82
+ if (segment.text === prev) return null;
83
+ const extendsPrev = segment.text.startsWith(prev);
84
+ const next = extendsPrev ? segment.text : prev + segment.text;
85
+ const previousBytes = existing === undefined
86
+ ? 0
87
+ : encoder.encode(segment.key).byteLength + encoder.encode(prev).byteLength;
88
+ const nextBytes = encoder.encode(segment.key).byteLength + encoder.encode(next).byteLength;
89
+ const reservation = budget.reserveTransient(nextBytes, { kind: "reasoning" });
90
+ try {
91
+ snapshots.set(segment.key, next);
92
+ reservation.commitRetained();
93
+ budget.releaseRetained(previousBytes, { kind: "reasoning" });
94
+ retainedBytes += nextBytes - previousBytes;
95
+ } catch (error) {
96
+ reservation.release();
97
+ throw error;
98
+ }
99
+ return extendsPrev ? segment.text.slice(prev.length) : segment.text;
100
+ },
101
+ release() {
102
+ budget.releaseRetained(retainedBytes, { kind: "reasoning" });
103
+ retainedBytes = 0;
104
+ snapshots.clear();
105
+ },
106
+ };
107
+ }
108
+
48
109
  /** Single-segment `reasoning_details` entry for replaying preserved reasoning (MiniMax wire shape). */
49
110
  export function reasoningDetailSegmentForWire(text: string): Record<string, unknown> {
50
111
  return { type: "reasoning.text", id: "reasoning-text-1", format: "MiniMax-response-v1", index: 0, text };
@@ -23,6 +23,7 @@ import {
23
23
  type InvalidToolCallDiagnostic,
24
24
  } from "./openai-chat/tool-call-validation";
25
25
  import {
26
+ createReasoningDetailSnapshotTracker,
26
27
  invalidChoicesEvent,
27
28
  invalidToolCallsEvent,
28
29
  reasoningDetailSegmentsFrom,
@@ -385,7 +386,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd
385
386
  // full text-so-far, so deltas are derived by prefix-diffing per segment key.
386
387
  // A piece that does not extend the previous snapshot is appended whole, which
387
388
  // keeps incremental senders parseable on the same path.
388
- const reasoningDetailSnapshots = new Map<string, string>();
389
+ const reasoningDetailTracker = createReasoningDetailSnapshotTracker(budget);
389
390
  // Gate on the routed model, not list length: a mixed openai-chat provider
390
391
  // can list MiniMax ids without putting every sibling on MiniMax semantics.
391
392
  const reasoningDetailsOptIn = modelInList(provider.reasoningDetailsModels, lastRequestedModelId ?? "");
@@ -448,15 +449,8 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd
448
449
  const detailSegments = reasoningDetailsOptIn ? reasoningDetailSegmentsFrom(delta) : [];
449
450
  if (detailSegments.length > 0) {
450
451
  for (const segment of detailSegments) {
451
- const prev = reasoningDetailSnapshots.get(segment.key) ?? "";
452
- if (segment.text === prev) continue;
453
- if (segment.text.startsWith(prev)) {
454
- reasoningDetailSnapshots.set(segment.key, segment.text);
455
- yield { type: "reasoning_raw_delta", text: segment.text.slice(prev.length) };
456
- } else {
457
- reasoningDetailSnapshots.set(segment.key, prev + segment.text);
458
- yield { type: "reasoning_raw_delta", text: segment.text };
459
- }
452
+ const reasoningDelta = reasoningDetailTracker.ingest(segment);
453
+ if (reasoningDelta !== null) yield { type: "reasoning_raw_delta", text: reasoningDelta };
460
454
  }
461
455
  } else {
462
456
  const reasoningText = reasoningTextFrom(delta);
@@ -692,6 +686,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd
692
686
  throw error;
693
687
  } finally {
694
688
  budget.releaseRetained(bufferBytes, { kind: "live_transient" });
689
+ reasoningDetailTracker.release();
695
690
  closeToolCalls();
696
691
  reader.releaseLock();
697
692
  }
@@ -36,7 +36,8 @@ import { scrubOcxCompactionItems, stripCanonicalOnlyToolFields, stripCanonicalOn
36
36
  import { stripCanonicalForwardPromptCacheOptions, stripDeprecatedPromptCacheRetention } from "./prompt-cache";
37
37
  import { isPlainObject } from "./internal";
38
38
  import { normalizeToolSchemas, promoteClientLoadedTools, stripUnsupportedHostedTools } from "./tool-schema";
39
- import { annotateEmptyResponsesToolOutputs, backfillWebSearchQueries, normalizeResponsesToolResultAdjacency, repairOrphanedInputItems, repairOversizedReplayCallIds, repairUnidentifiedToolOutputItems } from "./tool-output-recovery";
39
+ import { annotateEmptyResponsesToolOutputs, backfillWebSearchQueries, normalizeResponsesToolResultAdjacency, repairOrphanedInputItems, repairOversizedReplayCallIds, repairUnidentifiedToolOutputItems, restoreBridgedWebSearchCalls } from "./tool-output-recovery";
40
+ import { bridgeSearchReplayScope } from "../../responses/bridge-search-replay-cache";
40
41
  import { applyTierDecisionToResponsesBody, normalizeCanonicalForwardContinuationEnvelope, normalizeCanonicalForwardPromptEnvelope, stripCanonicalForwardSamplingParams, stripPreviousResponseId, stripStatefulResponsesParams, stripUnsupportedForwardParams } from "./canonical-forward";
41
42
  import { normalizeImageGenClientTools, preferConfiguredHostedTools } from "./image-gen";
42
43
  import { stripMuseSparkUnsupportedWebSearchFields, stripOpenAiOnlyWebSearchFields } from "./web-search";
@@ -321,6 +322,14 @@ export function createResponsesPassthroughAdapter(provider: OcxProviderConfig):
321
322
  outBody = repairOversizedReplayCallIds(outBody);
322
323
  }
323
324
  outBody = stripUnsupportedReasoningSummaryDelivery(outBody, parsed.modelId);
325
+ // #4587: on a bridged provider, hand the destination back the search call and result the
326
+ // proxy executed on its behalf, in place of the hosted cell the caller replays. Scoped to
327
+ // this destination and recorded by the bridge itself, so a provider without the opt-in
328
+ // computes no identity and keeps the body reference it already had. This runs before the
329
+ // query backfill below because a restored cell is no longer a web_search_call to repair.
330
+ if (provider.webSearchBridge?.enabled === true) {
331
+ outBody = restoreBridgedWebSearchCalls(outBody, bridgeSearchReplayScope(provider.baseUrl));
332
+ }
324
333
  // Repair stored history from before the bridge emitted both keys, in either
325
334
  // direction: a conversation that already recorded a web_search_call replays it
326
335
  // every turn, and a strict parser rejects the whole request over the missing key —
@@ -1,6 +1,7 @@
1
1
  import { createHash } from "node:crypto";
2
2
  import { EMPTY_TOOL_OUTPUT_ANNOTATION, isWhitespaceOnlyTextPartArray } from "../empty-tool-output-annotation";
3
3
  import { isPlainObject } from "./internal";
4
+ import { peekBridgeSearchReplay } from "../../responses/bridge-search-replay-cache";
4
5
 
5
6
  const MAX_RESPONSES_CALL_ID_LENGTH = 64;
6
7
 
@@ -265,6 +266,80 @@ export function backfillWebSearchQueries(body: unknown): unknown {
265
266
  return changed ? { ...body, input } : body;
266
267
  }
267
268
 
269
+ /**
270
+ * Give a bridged destination back its own search call and result (issue #4587).
271
+ *
272
+ * When `providers.<name>.webSearchBridge` is armed, the proxy intercepts the destination's
273
+ * `function_call` named `web_search`, runs the search, and shows the CALLER a hosted
274
+ * `web_search_call` cell. The caller stores that cell and replays it on every later turn, so the
275
+ * destination receives an item type it never produced, carrying a query and sources but no result.
276
+ * It typically responds by searching again.
277
+ *
278
+ * This restores the exchange the destination actually had: the cell becomes the destination's own
279
+ * `function_call`, immediately followed by the `function_call_output` the bridge produced for
280
+ * it, in the cell's original position. It runs before the first leg of the next turn is
281
+ * dispatched, which is the only place it can run — by the time the bridge wraps a turn, that
282
+ * turn's first leg is already on the wire.
283
+ *
284
+ * Three things it deliberately does not do:
285
+ * - It never re-runs a search. A missing memo entry means the result is gone, and paying for a
286
+ * second search would answer the model with a different search than its history claims.
287
+ * - It never invents result text. A miss leaves the item exactly as the caller sent it, which is
288
+ * the behaviour every unbridged conversation already has.
289
+ * - It never restores a call id the body already carries. If the history somehow holds that
290
+ * `function_call` too, emitting a second one would be a duplicate the upstream must reject.
291
+ *
292
+ * Entries are scoped to the upstream destination, so a history replayed against a different
293
+ * provider cannot resurrect a call that provider never made. Callers pass `undefined` for any
294
+ * provider without the bridge armed, and the common path then returns the original reference.
295
+ */
296
+ export function restoreBridgedWebSearchCalls(body: unknown, destinationScope: string | undefined): unknown {
297
+ if (destinationScope === undefined) return body;
298
+ if (!isPlainObject(body) || !Array.isArray(body.input)) return body;
299
+ const input = body.input;
300
+
301
+ // Cheap pre-check: nothing to do for a conversation that carries no hosted search cell at all,
302
+ // which is every turn before the model's first bridged search.
303
+ let hasCell = false;
304
+ for (const item of input) {
305
+ if (isPlainObject(item) && item.type === "web_search_call" && typeof item.id === "string") {
306
+ hasCell = true;
307
+ break;
308
+ }
309
+ }
310
+ if (!hasCell) return body;
311
+
312
+ const occupiedCallIds = new Set<string>();
313
+ for (const item of input) {
314
+ if (isPlainObject(item) && typeof item.call_id === "string") occupiedCallIds.add(item.call_id);
315
+ }
316
+
317
+ let changed = false;
318
+ const restored: unknown[] = [];
319
+ for (const item of input) {
320
+ if (isPlainObject(item) && item.type === "web_search_call" && typeof item.id === "string") {
321
+ const memo = peekBridgeSearchReplay(destinationScope, item.id);
322
+ if (memo && !occupiedCallIds.has(memo.callId)) {
323
+ changed = true;
324
+ occupiedCallIds.add(memo.callId);
325
+ restored.push({
326
+ type: "function_call",
327
+ ...(memo.sourceItemId ? { id: memo.sourceItemId } : {}),
328
+ call_id: memo.callId,
329
+ name: memo.name,
330
+ // The bridge records the complete arguments text from the call's own done frame; the
331
+ // empty-object fallback matches what a continuation leg would have sent.
332
+ arguments: memo.argumentsText.length > 0 ? memo.argumentsText : "{}",
333
+ });
334
+ restored.push({ type: "function_call_output", call_id: memo.callId, output: memo.output });
335
+ continue;
336
+ }
337
+ }
338
+ restored.push(item);
339
+ }
340
+ return changed ? { ...body, input: restored } : body;
341
+ }
342
+
268
343
  export function repairOrphanedInputItems(body: unknown, dropReasoning: boolean, synthesizeMissingCallOutputs = false): unknown {
269
344
  if (!isPlainObject(body) || !Array.isArray(body.input)) return body;
270
345
  const input = body.input;