@bitkyc08/opencodex 2.64.0 → 2.65.0-preview.20260925

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. package/README.md +28 -25
  2. package/bin/ocx.mjs +16 -2
  3. package/gui/dist/assets/App-8NMiZxT0.css +1 -0
  4. package/gui/dist/assets/App-CsYvvpr3.js +50 -0
  5. package/gui/dist/assets/Tray-DUvc_Wul.js +1 -0
  6. package/gui/dist/assets/{index-DiBRuK-d.css → index--EWgGQvZ.css} +1 -1
  7. package/gui/dist/assets/{index-SggB6t3z.js → index-BB0iHG8-.js} +12 -12
  8. package/gui/dist/assets/tray-data-f0lTZ4sF.js +1 -0
  9. package/gui/dist/index.html +2 -2
  10. package/package.json +1 -4
  11. package/src/adapters/anthropic.ts +30 -1
  12. package/src/adapters/claude-cli/adapter.ts +218 -0
  13. package/src/adapters/claude-cli/profiles.ts +38 -0
  14. package/src/adapters/codebuddy/profiles.ts +2 -0
  15. package/src/adapters/coding-agent/profile.ts +12 -4
  16. package/src/adapters/coding-agent/turn.ts +16 -6
  17. package/src/adapters/command-code-tool-text.ts +133 -23
  18. package/src/adapters/cursor/catalog.ts +4 -8
  19. package/src/adapters/cursor/discovery.ts +10 -5
  20. package/src/adapters/cursor/effort-map.ts +2 -6
  21. package/src/adapters/cursor/envelope-echo.ts +10 -5
  22. package/src/adapters/cursor/request-builder.ts +9 -4
  23. package/src/adapters/cursor/thread-continuity.ts +53 -0
  24. package/src/adapters/cursor.ts +29 -13
  25. package/src/adapters/devin/cloud-direct/chat.ts +3 -1
  26. package/src/adapters/google-tool-schema.ts +26 -3
  27. package/src/adapters/identity.ts +214 -2
  28. package/src/adapters/openai-chat/passthrough.ts +1 -0
  29. package/src/adapters/openai-chat/serialized-tool-call-content.ts +569 -0
  30. package/src/adapters/openai-chat.ts +43 -15
  31. package/src/adapters/openai-responses/canonical-forward.ts +17 -10
  32. package/src/adapters/openai-responses/passthrough.ts +19 -4
  33. package/src/adapters/openai-responses/request-strips.ts +25 -0
  34. package/src/adapters/openai-responses/web-search.ts +22 -4
  35. package/src/adapters/qoder/profiles.ts +2 -0
  36. package/src/adapters/registry.ts +8 -0
  37. package/src/adapters/run-turn-queue.ts +7 -0
  38. package/src/bridge/response-json.ts +6 -4
  39. package/src/bridge/sse.ts +7 -5
  40. package/src/claude/alias.ts +73 -26
  41. package/src/claude/context-windows.ts +15 -6
  42. package/src/claude/desktop-3p-library.ts +3 -2
  43. package/src/claude/desktop-3p.ts +1 -1
  44. package/src/claude/desktop-first-party.ts +63 -22
  45. package/src/claude/desktop-picker-profile.ts +302 -0
  46. package/src/claude/desktop-picker.ts +365 -0
  47. package/src/claude/desktop-risk.ts +20 -0
  48. package/src/claude/gateway-cache.ts +13 -4
  49. package/src/claude/inbound.ts +5 -2
  50. package/src/claude/intercept/connect-proxy.ts +106 -29
  51. package/src/claude/intercept/local-ca.ts +76 -18
  52. package/src/claude/intercept/model-bindings.ts +145 -0
  53. package/src/claude/intercept/picker-bootstrap.ts +141 -0
  54. package/src/claude/intercept/picker-ca.ts +126 -0
  55. package/src/claude/intercept/picker-listener.ts +213 -0
  56. package/src/claude/intercept/picker-models.ts +101 -0
  57. package/src/claude/intercept/picker-runtime.ts +330 -0
  58. package/src/claude/intercept/picker-trust.ts +125 -0
  59. package/src/claude/intercept/runtime.ts +149 -2
  60. package/src/claude/model-info.ts +18 -4
  61. package/src/cli/agent.ts +16 -2
  62. package/src/cli/capabilities.ts +71 -2
  63. package/src/cli/claude-desktop.ts +319 -33
  64. package/src/cli/claude.ts +15 -7
  65. package/src/cli/codex-shim-autorestore.ts +1 -0
  66. package/src/cli/dispatch.ts +29 -23
  67. package/src/cli/effort.ts +16 -6
  68. package/src/cli/ensure-desired-integrations.ts +29 -4
  69. package/src/cli/ready.ts +1 -1
  70. package/src/cli/registry.ts +8 -0
  71. package/src/cli/restart-scope.ts +9 -1
  72. package/src/cli/runtime-api.ts +17 -0
  73. package/src/cli/system-command.ts +11 -12
  74. package/src/client/connect.ts +12 -2
  75. package/src/client/state.ts +39 -1
  76. package/src/clients/config-export.ts +57 -9
  77. package/src/codex/auth-context.ts +51 -11
  78. package/src/codex/catalog/derive-entry.ts +5 -5
  79. package/src/codex/catalog/gather-capture.ts +25 -6
  80. package/src/codex/catalog/metadata.ts +3 -3
  81. package/src/codex/catalog/parsing.ts +39 -1
  82. package/src/codex/catalog/provider-models.ts +8 -6
  83. package/src/codex/catalog/retained-sync.ts +17 -7
  84. package/src/codex/catalog/sync.ts +2 -2
  85. package/src/codex/desktop-app-restart.ts +8 -0
  86. package/src/codex/desktop-switches.ts +9 -1
  87. package/src/codex/home.ts +43 -4
  88. package/src/codex/inject/config-toml.ts +137 -0
  89. package/src/codex/inject/plan.ts +23 -0
  90. package/src/codex/inject/remove.ts +21 -1
  91. package/src/codex/inject.ts +63 -8
  92. package/src/codex/journal.ts +36 -0
  93. package/src/codex/main-account-hard-lock.ts +14 -1
  94. package/src/codex/main-account-policy-wait.ts +109 -0
  95. package/src/codex/native-profile-startup.ts +24 -6
  96. package/src/codex/native-residue.ts +28 -10
  97. package/src/codex/quota-types.ts +8 -1
  98. package/src/codex/quota.ts +49 -7
  99. package/src/codex/runtime.ts +31 -10
  100. package/src/combos/failover.ts +7 -6
  101. package/src/combos/resolve.ts +91 -14
  102. package/src/combos/types.ts +18 -1
  103. package/src/config/load-degrade.ts +2 -0
  104. package/src/config/schema/config-schema.ts +21 -2
  105. package/src/config/schema/leaf-validators.ts +6 -3
  106. package/src/config/subagent-models.ts +3 -1
  107. package/src/generated/compatibility-version.json +275 -163
  108. package/src/images/artifacts.ts +14 -0
  109. package/src/integrations/owned-refresh.ts +1 -1
  110. package/src/integrations/ownership-policy.ts +32 -1
  111. package/src/integrations/state.ts +3 -1
  112. package/src/integrations/writer.ts +9 -0
  113. package/src/lib/config-ownership.ts +3 -0
  114. package/src/lib/errors.ts +87 -0
  115. package/src/lib/package-tree-integrity.ts +1 -1
  116. package/src/lib/request-execution-budget.ts +42 -14
  117. package/src/lib/request-resend-gate.ts +4 -3
  118. package/src/lib/retry-delay.ts +7 -1
  119. package/src/lib/tool-envelope-echo-filter.ts +236 -0
  120. package/src/lib/upstream-retry.ts +20 -12
  121. package/src/oauth/index.ts +1 -0
  122. package/src/oauth/kiro-credentials.ts +14 -1
  123. package/src/oauth/login-cli.ts +1 -0
  124. package/src/providers/derive.ts +4 -0
  125. package/src/providers/fast-opt-in.ts +31 -0
  126. package/src/providers/model-rename-fields.ts +2 -0
  127. package/src/providers/provider-id-rewrite.ts +2 -0
  128. package/src/providers/quota/vendor-probes-key.ts +8 -2
  129. package/src/providers/registry/entries-core.ts +45 -2
  130. package/src/providers/registry/entries-extended.ts +99 -12
  131. package/src/providers/registry/model-ids.ts +2 -0
  132. package/src/providers/registry/types.ts +7 -1
  133. package/src/providers/resolved-model-policy.ts +11 -4
  134. package/src/providers/service-tier.ts +10 -3
  135. package/src/responses/code-mode-helper-compat.ts +3 -0
  136. package/src/responses/parser.ts +45 -3
  137. package/src/router.ts +6 -0
  138. package/src/server/auth-cors.ts +8 -0
  139. package/src/server/chat-completions.ts +3 -1
  140. package/src/server/claude-messages.ts +69 -14
  141. package/src/server/grok-upstream-envelope-echo.ts +120 -0
  142. package/src/server/index/claude-intercept-lifecycle.ts +24 -0
  143. package/src/server/index/serve-options.ts +2 -2
  144. package/src/server/index.ts +7 -0
  145. package/src/server/management/agent-settings-routes.ts +204 -112
  146. package/src/server/management/claude-desktop-picker-routes.ts +128 -0
  147. package/src/server/management/combo-routes.ts +59 -2
  148. package/src/server/management/config-routes.ts +81 -39
  149. package/src/server/management/context.ts +3 -0
  150. package/src/server/management/native-integration-routes.ts +169 -93
  151. package/src/server/management/provider-routes.ts +46 -4
  152. package/src/server/management/route-registry.ts +10 -0
  153. package/src/server/management/routing-profile-routes.ts +9 -0
  154. package/src/server/management/sidebar-routes.ts +52 -0
  155. package/src/server/management-api.ts +2 -0
  156. package/src/server/proxy-liveness.ts +38 -3
  157. package/src/server/request-log-account-rotation.ts +53 -0
  158. package/src/server/request-log.ts +3 -16
  159. package/src/server/responses/adapter-dispatch.ts +17 -1
  160. package/src/server/responses/agent-task-recovery.ts +49 -26
  161. package/src/server/responses/codex-ws-exchange.ts +18 -7
  162. package/src/server/responses/codex-ws-wire.ts +27 -4
  163. package/src/server/responses/core-combo.ts +21 -3
  164. package/src/server/responses/core-normalize.ts +7 -0
  165. package/src/server/responses/core-opaque-recovery.ts +3 -2
  166. package/src/server/responses/encrypted-payload.ts +15 -14
  167. package/src/server/responses/fetch-helpers.ts +1 -1
  168. package/src/server/responses/input-admission.ts +10 -6
  169. package/src/server/responses/native-response-control.ts +2 -2
  170. package/src/server/responses/passthrough-delivery.ts +182 -12
  171. package/src/server/responses/passthrough-dispatch.ts +102 -44
  172. package/src/server/responses/policy-fallback.ts +3 -0
  173. package/src/server/responses/policy-refusal.ts +67 -0
  174. package/src/server/responses/request-send-budget.ts +4 -0
  175. package/src/server/responses/run-turn-execution.ts +42 -11
  176. package/src/server/responses/ws-upstream.ts +2 -1
  177. package/src/server/system-env-shell.ts +0 -1
  178. package/src/server/system-env.ts +25 -4
  179. package/src/service/cli.ts +34 -3
  180. package/src/tray/assets/opencodex-tray-offline-update.ico +0 -0
  181. package/src/tray/assets/opencodex-tray-online-update.ico +0 -0
  182. package/src/tray/assets/opencodex-tray-warning-update.ico +0 -0
  183. package/src/tray/windows-tray.ps1 +188 -7
  184. package/src/tray/windows.ts +23 -2
  185. package/src/types/config.ts +52 -12
  186. package/src/types/provider.ts +22 -5
  187. package/src/types/tools.ts +11 -5
  188. package/src/types.ts +1 -0
  189. package/src/update/async-check.ts +109 -0
  190. package/src/update/badge.ts +11 -17
  191. package/src/update/check-types.ts +9 -0
  192. package/src/update/desktop-badge.ts +102 -0
  193. package/src/update/index.ts +48 -8
  194. package/src/update/install-detection.d.mts +29 -2
  195. package/src/update/install-detection.mjs +213 -7
  196. package/src/update/job.ts +17 -12
  197. package/src/update/notify.ts +16 -11
  198. package/src/update/pnpm-owner-worker.ts +12 -0
  199. package/src/update/refresh-scheduler.ts +141 -0
  200. package/src/usage/cost.ts +31 -4
  201. package/src/usage/expected-prices.ts +57 -0
  202. package/assets/download-linux.svg +0 -10
  203. package/assets/download-macos.svg +0 -10
  204. package/assets/download-windows.svg +0 -10
  205. package/gui/dist/assets/App-CpuDF3ci.js +0 -50
  206. package/gui/dist/assets/App-I5AnaSLh.css +0 -1
  207. package/gui/dist/assets/Tray-B4uEIa1O.js +0 -1
  208. package/gui/dist/assets/usage-companion-chart-DLbKOJml.js +0 -1
@@ -20,7 +20,7 @@ import { getModelMetadata } from "../../generated/model-metadata";
20
20
  import { estimateTokens } from "../../lib/token-estimate";
21
21
  import { isCanonicalOpenAiForwardProvider, OPENAI_CODEX_PROVIDER_ID } from "../../providers/openai-tiers";
22
22
  import { modelRecordValue } from "../../reasoning-effort";
23
- import type { OcxContentPart, OcxParsedRequest, OcxProviderConfig } from "../../types";
23
+ import { modelInList, type OcxContentPart, type OcxParsedRequest, type OcxProviderConfig } from "../../types";
24
24
 
25
25
  /**
26
26
  * Multiplier applied to the ceiling before refusing.
@@ -112,10 +112,12 @@ function contentTokens(content: string | readonly OcxContentPart[], modelId: str
112
112
  * their content as `OcxAssistantContentPart[]` — text, thinking blocks, and tool calls whose
113
113
  * JSON arguments are frequently the largest single item in an agent conversation. A walk
114
114
  * that counted only `{type:"text"}` would undercount exactly the turns that trigger this
115
- * gate.
115
+ * gate. The routed provider determines whether replayed thinking reaches the wire.
116
116
  */
117
- export function estimateInputTokens(parsed: OcxParsedRequest, modelId: string): number {
117
+ export function estimateInputTokens(parsed: OcxParsedRequest, modelId: string, provider?: OcxProviderConfig): number {
118
118
  const { context } = parsed;
119
+ const countThinking = provider?.adapter !== "openai-chat"
120
+ || modelInList(provider.preserveReasoningContentModels, parsed.modelId);
119
121
  let total = 0;
120
122
 
121
123
  for (const prompt of context.systemPrompt ?? []) total += estimateTokens(prompt, modelId);
@@ -124,7 +126,9 @@ export function estimateInputTokens(parsed: OcxParsedRequest, modelId: string):
124
126
  if (message.role === "assistant") {
125
127
  for (const part of message.content) {
126
128
  if (part.type === "text") total += estimateTokens(part.text, modelId);
127
- else if (part.type === "thinking") total += estimateTokens(part.thinking, modelId);
129
+ else if (part.type === "thinking") {
130
+ if (countThinking) total += estimateTokens(part.thinking, modelId);
131
+ }
128
132
  else total += estimateTokens(part.name, modelId) + estimateTokens(JSON.stringify(part.arguments), modelId);
129
133
  }
130
134
  // Opaque provider blob replayed verbatim upstream, so it costs real input tokens.
@@ -296,7 +300,7 @@ export function checkComboTargetInputAdmission(
296
300
  const requiredOutputHeadroom = targetOutput === null
297
301
  ? requestedOutput
298
302
  : Math.min(requestedOutput, targetOutput);
299
- const estimatedTokens = estimateInputTokens(parsed, modelId);
303
+ const estimatedTokens = estimateInputTokens(parsed, modelId, provider);
300
304
  return {
301
305
  admitted: estimatedTokens <= ceiling && estimatedTokens + requiredOutputHeadroom <= window,
302
306
  estimatedTokens,
@@ -319,6 +323,6 @@ export function checkInputAdmission(
319
323
  ): InputAdmissionResult {
320
324
  const ceiling = resolveInputCeiling(provider, providerName, modelId, nativeContextCap);
321
325
  if (ceiling === null) return { admitted: true, estimatedTokens: 0, ceiling: null };
322
- const estimatedTokens = estimateInputTokens(parsed, modelId);
326
+ const estimatedTokens = estimateInputTokens(parsed, modelId, provider);
323
327
  return { admitted: estimatedTokens <= ceiling * ADMISSION_TOLERANCE, estimatedTokens, ceiling };
324
328
  }
@@ -31,9 +31,9 @@ export function markNativeControlResponse(response: Response): Response { native
31
31
  /** Recognize a marked native response by identity, not by caller-controlled content. */
32
32
  export function isNativeControlResponse(response: Response): boolean { return nativeControlResponses.has(response); }
33
33
 
34
- /** Preserve canonical ChatGPT eligibility; only injection may use the separately billed public API. */
34
+ /** Canonical ChatGPT needs its upstream WS enabled; only injection may use the separately billed public API. */
35
35
  export function nativeResponseControlEligible(provider: OcxProviderConfig, control?: NativeResponseControl): boolean {
36
- if (isCanonicalOpenAiForwardProvider(provider)) return true;
36
+ if (isCanonicalOpenAiForwardProvider(provider)) return provider.upstreamWebsocket !== false;
37
37
  return control?.kind === "injection" && provider.adapter === "openai-responses"
38
38
  && provider.upstreamWebsocket === true && provider.authMode !== "forward"
39
39
  && provider.baseUrl?.replace(/\/+$/, "") === "https://api.openai.com/v1";
@@ -35,6 +35,7 @@ import { consumeComboFailure } from "./core-combo-failure";
35
35
  import { readDisplaySafeErrorText } from "./core-errors";
36
36
  import { streamingContextOverflowResponse, jsonContextOverflowResponse } from "./context-overflow";
37
37
  import { formatPassthroughUpstreamError } from "./passthrough-error";
38
+ import { rewriteUpstreamPolicyRefusal } from "./policy-refusal";
38
39
  import {
39
40
  resolvePassthroughWebSearchBridgeAuth,
40
41
  planPassthroughWebSearchBridge,
@@ -84,6 +85,12 @@ import {
84
85
  createGrokResponsesTimestampBlockRewrite,
85
86
  } from "../grok-responses-control-frame";
86
87
  import { createGrokResponsesSparseTerminalBlockRewrite } from "../grok-responses-snapshot-repair";
88
+ import { isXaiResponsesDestination } from "../../providers/xai-transport";
89
+ import {
90
+ createGrokUpstreamEnvelopeEchoBlockRewrite,
91
+ responsesRequestMayReplayToolOutput,
92
+ stripGrokUpstreamEnvelopeEchoFromResponsesJson,
93
+ } from "../grok-upstream-envelope-echo";
87
94
  import {
88
95
  createPlaintextV2AgentMessageCallRestoreRewrite,
89
96
  restorePlaintextV2AgentMessageCallsInJsonResult,
@@ -121,12 +128,145 @@ import { linkAbortSignal, UPSTREAM_JSON_BODY_READ_OPTIONS } from "./core-lifetim
121
128
  import { registerTurn, unregisterTurn, trackStreamLifetime } from "../lifecycle";
122
129
  import { relaySseEagerBounded } from "../relay-eager";
123
130
  import { readBoundedResponseBody } from "../../lib/bounded-body";
131
+ import { idleDeadline } from "../../lib/abort";
132
+ import { resolveStallTimeoutSec } from "../../stall-timeout";
124
133
  import { formatErrorResponse } from "../../bridge";
125
134
  import { inspectResponseLogJson } from "../request-log";
126
135
  import { restoreRoutedCustomCallsInJson } from "../../responses/custom-tool-compat";
127
136
  import { restoreRoutedToolSearchCallsInJson } from "../../responses/tool-search-compat";
128
137
  import { responsesJsonToSseStream } from "../responses-json-events";
129
138
 
139
+ const PLAINTEXT_V2_SSE_PREFIX_LIMIT = 4096;
140
+
141
+ /** Prefix-probe budget: bounds one silent gap and the whole probe alike. */
142
+ interface PlaintextV2SseProbeOptions {
143
+ timeoutMs: number;
144
+ signal?: AbortSignal;
145
+ }
146
+
147
+ function classifyPlaintextV2SsePrefix(prefix: string): "sse" | "unknown" | "more" {
148
+ const lastLineEnd = prefix.lastIndexOf("\n");
149
+ if (lastLineEnd < 0) return "more";
150
+ for (const rawLine of prefix.slice(0, lastLineEnd + 1).split("\n")) {
151
+ const line = rawLine.replace(/\r$/, "").trim();
152
+ if (!line || line.startsWith(":")) continue;
153
+ if (/^(id|retry):/.test(line)) continue;
154
+ if (line.startsWith("event:")) {
155
+ return /^event:\s*(?:response\.[\w.-]+|error)$/.test(line) ? "sse" : "unknown";
156
+ }
157
+ if (line.startsWith("data:")) {
158
+ try {
159
+ const value = JSON.parse(line.slice(5).trim()) as { type?: unknown };
160
+ return typeof value.type === "string" && /^(?:response\.[\w.-]+|error)$/.test(value.type)
161
+ ? "sse" : "unknown";
162
+ } catch {
163
+ return "unknown";
164
+ }
165
+ }
166
+ return "unknown";
167
+ }
168
+ return "more";
169
+ }
170
+
171
+ /**
172
+ * Confirm an unlabeled successful body is Responses SSE before alias restoration.
173
+ *
174
+ * The probe waits at most `timeoutMs` for a first recognized event, then hands the body to the
175
+ * client with no deadline of its own: a stall in the prefix is a probe failure, and a stall after
176
+ * it belongs to the delivered stream.
177
+ */
178
+ async function classifyPlaintextV2SseResponse(
179
+ response: Response,
180
+ probe: PlaintextV2SseProbeOptions,
181
+ ): Promise<Response> {
182
+ if (!response.body) return response;
183
+ const reader = response.body.getReader();
184
+ // A non-conforming stream can throw synchronously from cancel(); neither that nor a
185
+ // rejected cancel may escape past the probe's own deadline.
186
+ const cancelReader = (reason?: unknown): void => {
187
+ try {
188
+ void reader.cancel(reason).catch(() => undefined);
189
+ } catch {
190
+ // Some stream implementations throw synchronously from cancel().
191
+ }
192
+ };
193
+ const unrecognized = (): Response => new Response(null, {
194
+ status: response.status,
195
+ statusText: response.statusText,
196
+ headers: response.headers,
197
+ });
198
+ const { timeoutMs, signal } = probe;
199
+ const stalled = new DOMException("Plaintext V2 SSE prefix probe stalled", "TimeoutError");
200
+ let rejectProbe: ((reason: unknown) => void) | undefined;
201
+ const failed = new Promise<never>((_resolve, reject) => { rejectProbe = reject; });
202
+ // The race below always observes this rejection; this covers a deadline that fires after the
203
+ // race already settled with a chunk, which would otherwise be an unhandled rejection.
204
+ void failed.catch(() => undefined);
205
+ const inactivity = idleDeadline(timeoutMs, () => rejectProbe?.(stalled));
206
+ // A drip-fed body can restart the inactivity window forever, so the probe also carries one
207
+ // total budget that starts when the probe begins.
208
+ const totalTimer = timeoutMs > 0 ? setTimeout(() => rejectProbe?.(stalled), timeoutMs) : undefined;
209
+ const onAbort = (): void => rejectProbe?.(signal?.reason);
210
+ signal?.addEventListener("abort", onAbort, { once: true });
211
+ if (signal?.aborted) onAbort();
212
+ const decoder = new TextDecoder();
213
+ const buffered: Uint8Array[] = [];
214
+ let prefix = "";
215
+ let inspectedBytes = 0;
216
+ try {
217
+ while (inspectedBytes < PLAINTEXT_V2_SSE_PREFIX_LIMIT) {
218
+ if (signal?.aborted) throw signal.reason;
219
+ // Armed for every read, so a chunk that arrives restarts the window at the next iteration.
220
+ inactivity.reset();
221
+ const read = reader.read();
222
+ // Observe a late read rejection when the deadline or the client wins the race.
223
+ void read.catch(() => undefined);
224
+ const next = await Promise.race([read, failed]);
225
+ if (signal?.aborted) throw signal.reason;
226
+ if (next.done) break;
227
+ buffered.push(next.value);
228
+ const inspected = next.value.subarray(0, PLAINTEXT_V2_SSE_PREFIX_LIMIT - inspectedBytes);
229
+ inspectedBytes += inspected.byteLength;
230
+ prefix += decoder.decode(inspected, { stream: true });
231
+ const kind = classifyPlaintextV2SsePrefix(prefix);
232
+ if (kind === "sse") {
233
+ let bufferedIndex = 0;
234
+ const body = new ReadableStream<Uint8Array>({
235
+ async pull(controller) {
236
+ if (bufferedIndex < buffered.length) {
237
+ controller.enqueue(buffered[bufferedIndex++]!);
238
+ return;
239
+ }
240
+ try {
241
+ const result = await reader.read();
242
+ if (result.done) controller.close();
243
+ else controller.enqueue(result.value);
244
+ } catch (error) {
245
+ controller.error(error);
246
+ }
247
+ },
248
+ cancel(reason) { return reader.cancel(reason); },
249
+ });
250
+ const headers = new Headers(response.headers);
251
+ headers.set("content-type", "text/event-stream");
252
+ return new Response(body, { status: response.status, statusText: response.statusText, headers });
253
+ }
254
+ if (kind === "unknown") break;
255
+ }
256
+ } catch (error) {
257
+ // Timeout, client abort, or a failed read fails closed through the unrecognized-body exit,
258
+ // which the caller already answers as the unsupported-content-type 502.
259
+ cancelReader(error);
260
+ return unrecognized();
261
+ } finally {
262
+ inactivity.cancel();
263
+ if (totalTimer !== undefined) clearTimeout(totalTimer);
264
+ signal?.removeEventListener("abort", onAbort);
265
+ }
266
+ cancelReader();
267
+ return unrecognized();
268
+ }
269
+
130
270
  /** One responsibility of the Responses request pipeline; state owners are explicit. */
131
271
  export async function deliverPassthroughResponse(
132
272
  requestContext: Pick<ResponsesRequestContext, "logCtx" | "config" | "options" | "req">,
@@ -180,7 +320,6 @@ export async function deliverPassthroughResponse(
180
320
  ): Promise<Response> {
181
321
  const { logCtx, config, options, req } = requestContext;
182
322
  const {
183
- upstreamResponse,
184
323
  codexSafetyBufferingOptions,
185
324
  upstream,
186
325
  connectMs,
@@ -205,18 +344,29 @@ export async function deliverPassthroughResponse(
205
344
  const { openAiSidecar } = sidecarState;
206
345
  const { requestBindings } = transportState;
207
346
 
347
+ let upstreamResponse = nativeExchange.upstreamResponse;
348
+ const originalContentType = upstreamResponse.headers.get("content-type");
349
+ if (isUsageDebugEnabled() && originalContentType) logCtx.usageDebugContentType = originalContentType;
350
+ if (responseEffects.plaintextV2AgentMessageToolNames.size > 0
351
+ && upstreamResponse.ok && upstreamResponse.body && parsed.stream
352
+ && !originalContentType?.toLowerCase().includes("text/event-stream")
353
+ && !originalContentType?.toLowerCase().includes("application/json")
354
+ && !isCodexWsUpstreamResponse(upstreamResponse)
355
+ && !(options.nativeControl && isNativeControlResponse(upstreamResponse))) {
356
+ upstreamResponse = await classifyPlaintextV2SseResponse(upstreamResponse, {
357
+ timeoutMs: resolveStallTimeoutSec(config.stallTimeoutSec) * 1000,
358
+ signal: options.abortSignal ?? req.signal,
359
+ });
360
+ }
361
+
208
362
  const headers = sanitizePassthroughHeaders(upstreamResponse.headers, codexSafetyBufferingOptions);
209
363
  const resolvedModel = headers.get("openai-model")?.trim();
210
364
  if (resolvedModel) {
211
365
  logCtx.servedModel = resolvedModel;
212
366
  if (!logCtx.preserveResolvedModelFromRoute) logCtx.resolvedModel = resolvedModel;
213
367
  }
214
- if (isUsageDebugEnabled()) {
215
- const upstreamContentType = upstreamResponse.headers.get("content-type");
216
- if (upstreamContentType) logCtx.usageDebugContentType = upstreamContentType;
217
- }
218
- // The chatgpt backend may omit Content-Type on SSE responses. Fall back to
219
- // treating a successful body as SSE when the caller requested streaming.
368
+ // ChatGPT may omit Content-Type on SSE responses. Plaintext V2 responses
369
+ // reach this fallback only after their first Responses event is confirmed.
220
370
  const passthroughCt = headers.get("content-type")?.toLowerCase();
221
371
  const isEventStream = passthroughCt?.includes("text/event-stream")
222
372
  || (responseEffects.plaintextV2AgentMessageToolNames.size === 0 && upstreamResponse.ok && !!upstreamResponse.body && !passthroughCt && parsed.stream);
@@ -326,6 +476,16 @@ export async function deliverPassthroughResponse(
326
476
  ? streamingContextOverflowResponse(parsed._responseModelId ?? parsed.modelId, translatorBudget)
327
477
  : jsonContextOverflowResponse();
328
478
  }
479
+ const policyRefusal = rewriteUpstreamPolicyRefusal({
480
+ status: upstreamResponse.status,
481
+ errorText,
482
+ stream: clientRequestedStream,
483
+ modelId: parsed._responseModelId ?? parsed.modelId,
484
+ destinationIsXai: isXaiResponsesDestination(route.provider),
485
+ translatorBudget,
486
+ turnAdmissionLease: options.turnAdmissionLease,
487
+ });
488
+ if (policyRefusal) return policyRefusal;
329
489
  return formatPassthroughUpstreamError(upstreamResponse.status, errorText, {
330
490
  statusText: upstreamResponse.statusText,
331
491
  headers,
@@ -356,6 +516,8 @@ export async function deliverPassthroughResponse(
356
516
  // (src/server/relay-eager.ts; policy:
357
517
  // devlog/_fin/260731_macos_rss_retention/100_darwin_eager_optin.md).
358
518
  // The bundled known-bad runtime remains on tee by default on both platforms.
519
+ const grokUpstreamEchoEnabled = isXaiResponsesDestination(route.provider)
520
+ && responsesRequestMayReplayToolOutput(parsed._rawBody);
359
521
  if (isEventStream && upstreamResponse.body) {
360
522
  // For streamed passthrough, a successful terminal response means non-error upstream status
361
523
  // before relay starts. Waiting for SSE completion would retain request state across the whole
@@ -500,7 +662,7 @@ export async function deliverPassthroughResponse(
500
662
  // are not the Responses wire shapes the snapshot must mirror.
501
663
  // Only validated client blocks may publish plaintext continuation state.
502
664
  // Raw inspection precedes rewriting on eager relays, so it cannot own this write.
503
- const plaintextInspector = responseEffects.plaintextV2AgentMessageToolNames.size > 0
665
+ const plaintextInspector = !grokUpstreamEchoEnabled && responseEffects.plaintextV2AgentMessageToolNames.size > 0
504
666
  ? createSseInspector({ onCompletedResponse: rememberPassthroughResponseChecked })
505
667
  : undefined;
506
668
  const plaintextEncoder = plaintextInspector ? new TextEncoder() : undefined;
@@ -571,6 +733,11 @@ export async function deliverPassthroughResponse(
571
733
  declaredBareWireToolNames,
572
734
  )
573
735
  : undefined,
736
+ grokUpstreamEchoEnabled
737
+ ? createGrokUpstreamEnvelopeEchoBlockRewrite(
738
+ rememberPassthroughResponse ? rememberPassthroughResponseChecked : undefined,
739
+ )
740
+ : undefined,
574
741
  rememberPlaintextBlock,
575
742
  ].filter((rewrite): rewrite is NonNullable<typeof rewrite> => rewrite !== undefined);
576
743
  const clientBlockRewrite = blockRewrites.length > 0
@@ -620,7 +787,7 @@ export async function deliverPassthroughResponse(
620
787
  const inspector = createSseInspector({
621
788
  onTerminal: reportNativeTerminal,
622
789
  logCtx,
623
- onCompletedResponse: rememberPassthroughResponse && responseEffects.plaintextV2AgentMessageToolNames.size === 0 ? rememberPassthroughResponseChecked : undefined,
790
+ onCompletedResponse: rememberPassthroughResponse && !grokUpstreamEchoEnabled && responseEffects.plaintextV2AgentMessageToolNames.size === 0 ? rememberPassthroughResponseChecked : undefined,
624
791
  onParsedPayload: noteInspectedPayload,
625
792
  onFirstOutput: options.onFirstOutput,
626
793
  pinCompletedResponseIdToFirstSeen: githubCopilotRepairEnabled,
@@ -722,7 +889,7 @@ export async function deliverPassthroughResponse(
722
889
  responseEffects.responseCompletionCancelled = true;
723
890
  options.onNativePassthroughCancel?.();
724
891
  },
725
- rememberPassthroughResponse && responseEffects.plaintextV2AgentMessageToolNames.size === 0 ? rememberPassthroughResponseChecked : undefined,
892
+ rememberPassthroughResponse && !grokUpstreamEchoEnabled && responseEffects.plaintextV2AgentMessageToolNames.size === 0 ? rememberPassthroughResponseChecked : undefined,
726
893
  options.onFirstOutput,
727
894
  inspectionConsumerOptions,
728
895
  );
@@ -732,7 +899,7 @@ export async function deliverPassthroughResponse(
732
899
  logCtx,
733
900
  turnAc.signal,
734
901
  () => unregisterTurn(turnAc),
735
- rememberPassthroughResponse && responseEffects.plaintextV2AgentMessageToolNames.size === 0 ? rememberPassthroughResponseChecked : undefined,
902
+ rememberPassthroughResponse && !grokUpstreamEchoEnabled && responseEffects.plaintextV2AgentMessageToolNames.size === 0 ? rememberPassthroughResponseChecked : undefined,
736
903
  options.onFirstOutput,
737
904
  inspectionConsumerOptions,
738
905
  );
@@ -814,6 +981,9 @@ export async function deliverPassthroughResponse(
814
981
  if (plaintextV2RestoreFailed) {
815
982
  return formatErrorResponse(502, "upstream_error", PLAINTEXT_V2_AGENT_MESSAGE_RESTORE_OVERFLOW_MESSAGE);
816
983
  }
984
+ if (grokUpstreamEchoEnabled) {
985
+ clientJson = stripGrokUpstreamEnvelopeEchoFromResponsesJson(clientJson);
986
+ }
817
987
  // #1700: same fail-closed policy as the SSE relay above. Both the plain JSON answer and
818
988
  // the reframed-SSE branch below are built from this body, so one check covers them. This
819
989
  // runs BEFORE the continuation cache write below: a refused turn must not become state a
@@ -844,7 +1014,7 @@ export async function deliverPassthroughResponse(
844
1014
  commitReasoningReplayServingRoute(nativeExchange.request.headers);
845
1015
  try {
846
1016
  rememberPassthroughResponseChecked(
847
- JSON.parse(text) as { id?: unknown; output?: unknown; status?: unknown; model?: unknown },
1017
+ JSON.parse(grokUpstreamEchoEnabled ? clientJson : text) as { id?: unknown; output?: unknown; status?: unknown; model?: unknown },
848
1018
  );
849
1019
  } catch { /* non-JSON despite content-type; recording is best-effort */ }
850
1020
  // #875: the transport-neutral reliability policy forced a bounded JSON
@@ -102,7 +102,7 @@ import {
102
102
  clearCodexModelDenialEvidence,
103
103
  recordCodexModelDenialEvidence,
104
104
  } from "../../codex/model-entitlements";
105
- import { isCodexWsUpstreamResponse, readCodexWsStage } from "./codex-ws-wire";
105
+ import { codexWsSocketDeathStage, isCodexWsUpstreamResponse, readCodexWsStage } from "./codex-ws-wire";
106
106
  import { linkAbortSignal } from "./core-lifetime";
107
107
  import type { CodexAuthContext } from "../../codex/auth-context";
108
108
  import { checkOutboundBodySize, describeOutboundBodyRefusal } from "./outbound-body-guard";
@@ -112,7 +112,10 @@ import {
112
112
  fetchWithTransientRetry,
113
113
  applyUpstreamRecoveryInit,
114
114
  isNonReplayableResponse,
115
+ isConnectionResetError,
116
+ settleOperatorReplacement,
115
117
  refetchAfterProtocolSafeReset,
118
+ replayRefusalResponse,
116
119
  prepareSameTarget429Wait,
117
120
  sleepWithAbort,
118
121
  } from "../../lib/upstream-retry";
@@ -212,6 +215,7 @@ export async function preparePassthroughExchange(
212
215
  | "recoveryClassFor"
213
216
  | "sendBudgetExhausted"
214
217
  | "claimAmbiguousResend"
218
+ | "ambiguousResendSpent"
215
219
  | "reserveCredentialHop"
216
220
  | "pendingHopPermit"
217
221
  | "workflowRootId"
@@ -751,6 +755,57 @@ export async function preparePassthroughExchange(
751
755
  */
752
756
  const claimPreHeaderResend = (): boolean =>
753
757
  authorizeResendForRecovery("pre-header", "connection-reset", ambiguousResend()).allowed;
758
+ /**
759
+ * The one replacement send the ambiguous rows at the end of the recovery loop may buy: an SSE
760
+ * body that died before any output, and a Codex WebSocket that died under its create frame
761
+ * (#4191).
762
+ *
763
+ * HTTP-only for both. A replacement HTTP body must not open a fresh WebSocket exchange: the SSE
764
+ * row replaces an HTTP stream, which a WS create frame is not, and the WebSocket row replaces
765
+ * the transport that just failed.
766
+ */
767
+ const sendAmbiguousReplacement = (
768
+ signal: AbortSignal = upstream.signal,
769
+ ): Promise<Response> => fetchWithHeaderTimeout(
770
+ request.url,
771
+ applyUpstreamRecoveryInit({
772
+ method: request.method,
773
+ headers: request.headers,
774
+ body: request.body,
775
+ }, "connection-reset"),
776
+ signal,
777
+ connectMs,
778
+ true,
779
+ providerFetch(route.provider, options.codexWsRuntimeIdentity, {
780
+ httpOnly: true,
781
+ providerName: route.providerName,
782
+ modelId: route.modelId,
783
+ dispatchOverride: oauthDispatch(request),
784
+ beforeDispatch: headers => {
785
+ if (signal.aborted) throw signal.reason;
786
+ if (!transportState.selectionIsCurrent(transportState.requestBindings.get(request))) {
787
+ throw new Error("Credential selection changed before pre-output stream recovery");
788
+ }
789
+ if (isCanonicalOpenAiForwardProvider(route.provider)) {
790
+ createCodexReserveDispatchGuard(
791
+ admissionState.authCtx,
792
+ options.codexAuthPolicy ?? config,
793
+ route.modelId,
794
+ options.admission,
795
+ options.visionDescribeTerminal === true,
796
+ )?.(headers);
797
+ }
798
+ // Recorded with the kind the gate derived its cause from, at the moment the
799
+ // send actually leaves. One authorisation, one recorded reason, one send.
800
+ transportState.noteRoutedAttemptSend(passthroughEstimate, "connection-reset");
801
+ // Charged to the SAME request counter every other send goes through. The
802
+ // replacement is bought here rather than by a nested retry helper, so there is
803
+ // one charge for one send and no per-layer counter to reconcile.
804
+ noteTransientSends(1);
805
+ },
806
+ }),
807
+ route.provider.authMode === "forward",
808
+ );
754
809
  /**
755
810
  * Refuse a built body that exceeds the operator's configured ceiling, before it is sent.
756
811
  *
@@ -1554,7 +1609,8 @@ export async function preparePassthroughExchange(
1554
1609
  if (options.abortSignal?.aborted) return transportFailureResponse(options.abortSignal.reason);
1555
1610
  upstreamResponse = preflight.response;
1556
1611
  if (preflight.kind === "failed") {
1557
- if (!configuredTransientSendBudgetExhausted()) {
1612
+ // A zero-output failure does not undo an ambiguous replacement already sent.
1613
+ if (!configuredTransientSendBudgetExhausted() && !sendBudgetState.ambiguousResendSpent) {
1558
1614
  const streamedOpaqueRecovery = await attemptOpaqueBlobRecovery({
1559
1615
  response: upstreamResponse,
1560
1616
  outboundBody: request.body,
@@ -1574,6 +1630,8 @@ export async function preparePassthroughExchange(
1574
1630
  logCtx.terminalHttpStatus = preflightLog.terminalHttpStatus;
1575
1631
  logCtx.terminalErrorCode = preflightLog.terminalErrorCode;
1576
1632
  logCtx.terminalIncompleteReason = preflightLog.terminalIncompleteReason;
1633
+ // The projected failure must not invite another send after the replacement was spent.
1634
+ if (sendBudgetState.ambiguousResendSpent) upstreamResponse = settleOperatorReplacement(upstreamResponse);
1577
1635
  }
1578
1636
  }
1579
1637
  // Console Go (opencode-zen / opencode-go) intermittently rejects a body it accepts seconds
@@ -1637,6 +1695,47 @@ export async function preparePassthroughExchange(
1637
1695
  }
1638
1696
  }
1639
1697
 
1698
+ // The WebSocket row of the same table (#4191). A Codex socket that closed or failed under its
1699
+ // create frame, before any Responses event, settled as a non-replayable 502 and every leg above
1700
+ // let it through. Whether the turn ran upstream is as unknown as after a reset before the head,
1701
+ // so the same grant decides, asked with the stage the exchange reached. The replacement's answer
1702
+ // then goes round the loop like any other, and once the grant is spent nothing may send the
1703
+ // turn a third time.
1704
+ const socketDeathStage = codexWsSocketDeathStage(upstreamResponse);
1705
+ if (
1706
+ socketDeathStage
1707
+ && !upstream.signal.aborted
1708
+ // Asked before the gate, which claims last: a replacement the budget cannot fund must not
1709
+ // spend the request's one grant.
1710
+ && remainingTransientSendBudget(transientSendAttempts()) > 0
1711
+ && authorizeResendForRecovery(socketDeathStage, "connection-reset", ambiguousResend()).allowed
1712
+ ) {
1713
+ try { void upstreamResponse.body?.cancel().catch(() => {}); } catch { /* already consumed/closed */ }
1714
+ console.warn(`[upstream-retry] codex websocket died before any Responses event (${safeHostLabel(request.url)}); `
1715
+ + "using one replacement over HTTP");
1716
+ let replacement: Response | undefined;
1717
+ while (!replacement) {
1718
+ try {
1719
+ replacement = await sendAmbiguousReplacement().then(adoptObservedResponse);
1720
+ } catch (err) {
1721
+ if (upstream.signal.aborted) return transportFailureResponse(err);
1722
+ // A replacement that reset before its head is the pre-header row again: a configured
1723
+ // second grant (and a send the budget can still fund) may buy one more send. The gate
1724
+ // claims from the request's finite allowance, so this loop is bounded by it.
1725
+ if (
1726
+ isConnectionResetError(err)
1727
+ && remainingTransientSendBudget(transientSendAttempts()) > 0
1728
+ && claimPreHeaderResend()
1729
+ ) continue;
1730
+ break;
1731
+ }
1732
+ }
1733
+ // The first send may already have run the turn, so a replacement that failed settles as
1734
+ // the refusal rather than as a transport error the client would retry.
1735
+ upstreamResponse = replacement ? settleOperatorReplacement(replacement) : replayRefusalResponse();
1736
+ continue passthroughRecovery;
1737
+ }
1738
+
1640
1739
  // The post-header row of the same table. A native SSE body can die after the head with the
1641
1740
  // caller having observed nothing, which is the identical question the pre-header helper
1642
1741
  // answers -- and the identical grant, because both claim from this request's one allowance.
@@ -1660,48 +1759,7 @@ export async function preparePassthroughExchange(
1660
1759
  upstreamResponse,
1661
1760
  { model: logCtx.model, provider: logCtx.provider },
1662
1761
  (error, stage) => refetchAfterProtocolSafeReset(
1663
- (signal = upstream.signal) => fetchWithHeaderTimeout(
1664
- request.url,
1665
- applyUpstreamRecoveryInit({
1666
- method: request.method,
1667
- headers: request.headers,
1668
- body: request.body,
1669
- }, "connection-reset"),
1670
- signal,
1671
- connectMs,
1672
- true,
1673
- providerFetch(route.provider, options.codexWsRuntimeIdentity, {
1674
- // A replacement HTTP body must not open a fresh WebSocket exchange: the turn it
1675
- // replaces was an HTTP stream, and a WS create frame is a different send.
1676
- httpOnly: true,
1677
- providerName: route.providerName,
1678
- modelId: route.modelId,
1679
- dispatchOverride: oauthDispatch(request),
1680
- beforeDispatch: headers => {
1681
- if (signal.aborted) throw signal.reason;
1682
- if (!transportState.selectionIsCurrent(transportState.requestBindings.get(request))) {
1683
- throw new Error("Credential selection changed before pre-output stream recovery");
1684
- }
1685
- if (isCanonicalOpenAiForwardProvider(route.provider)) {
1686
- createCodexReserveDispatchGuard(
1687
- admissionState.authCtx,
1688
- options.codexAuthPolicy ?? config,
1689
- route.modelId,
1690
- options.admission,
1691
- options.visionDescribeTerminal === true,
1692
- )?.(headers);
1693
- }
1694
- // Recorded with the kind the gate derived its cause from, at the moment the
1695
- // send actually leaves. One authorisation, one recorded reason, one send.
1696
- transportState.noteRoutedAttemptSend(passthroughEstimate, "connection-reset");
1697
- // Charged to the SAME request counter every other send goes through. The
1698
- // replacement is bought here rather than by a nested retry helper, so there is
1699
- // one charge for one send and no per-layer counter to reconcile.
1700
- noteTransientSends(1);
1701
- },
1702
- }),
1703
- route.provider.authMode === "forward",
1704
- ).then(adoptObservedResponse),
1762
+ (signal = upstream.signal) => sendAmbiguousReplacement(signal).then(adoptObservedResponse),
1705
1763
  error,
1706
1764
  {
1707
1765
  abortSignal: upstream.signal,
@@ -1,5 +1,6 @@
1
1
  import { comboFailureDecision } from "../../combos/failover";
2
2
  import { readBoundedResponseBody } from "../../lib/bounded-body";
3
+ import { isNonReplayableResponse } from "../../lib/upstream-retry";
3
4
  import { finishRequestAttempt, type RequestLogContext } from "../request-log";
4
5
  import { linkRequestSessionLane } from "../request-log-conversation";
5
6
  import type { OcxConfig } from "../../types";
@@ -84,6 +85,8 @@ function errorCodeFromText(text: string): string | undefined {
84
85
 
85
86
  async function shouldHopPolicyCandidate(response: Response, signal?: AbortSignal): Promise<boolean> {
86
87
  if (response.status < 400 || signal?.aborted) return false;
88
+ // A response that must not be sent again cannot open a policy-candidate retry either.
89
+ if (isNonReplayableResponse(response)) return false;
87
90
  try {
88
91
  const inspected = await readBoundedResponseBody(response.clone(), { signal });
89
92
  const text = inspected.displaySafe ? inspected.text : "";
@@ -0,0 +1,67 @@
1
+ import { bridgeToResponsesSSE, buildResponseJSON } from "../../bridge";
2
+ import type { AdmissionLease } from "../../lib/admission";
3
+ import type { TranslatorBudget } from "../../lib/translator-budget";
4
+ import type { AdapterEvent } from "../../types";
5
+ import { extractPolicyRefusalText, isUpstreamPolicyRefusal } from "../../lib/errors";
6
+ import { trackStreamLifetime } from "../lifecycle";
7
+
8
+ async function* policyRefusalEvents(message: string): AsyncGenerator<AdapterEvent> {
9
+ yield { type: "text_delta", text: message };
10
+ yield { type: "done", stopReason: "content_filter" };
11
+ }
12
+
13
+ /**
14
+ * Turn an xAI-style HTTP 403 model refusal into a Codex-facing Responses
15
+ * incomplete/content_filter payload. Returned from `prepareAdapterExchange`
16
+ * and native openai-responses passthrough (the grok-4.6 OAuth wire), like
17
+ * the 413 overflow helpers. Combo hops still see the original 403.
18
+ *
19
+ * The streamed form is a delivered turn, so it carries the turn admission lease
20
+ * the way every other streaming return does: the lease is released when the body
21
+ * finishes or the client disconnects, not when the handler returns.
22
+ *
23
+ * Only an xAI destination is rewritten. Another provider's 403 may carry the same
24
+ * sentence for an unrelated reason, and turning it into a successful turn would hide it.
25
+ */
26
+ export function rewriteUpstreamPolicyRefusal(args: {
27
+ status: number;
28
+ errorText: string;
29
+ stream: boolean;
30
+ modelId: string;
31
+ /** `isXaiResponsesDestination(route.provider)`: api.x.ai or the Grok CLI proxy. */
32
+ destinationIsXai: boolean;
33
+ translatorBudget: TranslatorBudget;
34
+ turnAdmissionLease?: AdmissionLease;
35
+ }): Response | null {
36
+ if (!args.destinationIsXai || !isUpstreamPolicyRefusal(args.status, args.errorText)) return null;
37
+ const message = extractPolicyRefusalText(args.errorText);
38
+ if (!args.stream) {
39
+ const json = buildResponseJSON(
40
+ [
41
+ { type: "text_delta", text: message },
42
+ { type: "done", stopReason: "content_filter" },
43
+ ],
44
+ args.modelId,
45
+ { translatorBudget: args.translatorBudget },
46
+ );
47
+ return Response.json(json, { status: 200, headers: { "Cache-Control": "no-store" } });
48
+ }
49
+ const sse = bridgeToResponsesSSE(
50
+ policyRefusalEvents(message),
51
+ args.modelId,
52
+ undefined,
53
+ undefined,
54
+ undefined,
55
+ undefined,
56
+ 2_000,
57
+ { translatorBudget: args.translatorBudget },
58
+ );
59
+ return new Response(trackStreamLifetime(sse, new AbortController(), undefined, args.turnAdmissionLease), {
60
+ headers: {
61
+ "Content-Type": "text/event-stream",
62
+ "Cache-Control": "no-cache",
63
+ Connection: "keep-alive",
64
+ "X-Accel-Buffering": "no",
65
+ },
66
+ });
67
+ }