@bitkyc08/opencodex 2.54.0 → 2.55.0-preview.20260914

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/gui/dist/assets/{index-CkvITofZ.js → index-DH2PUHqr.js} +10 -10
  2. package/gui/dist/index.html +1 -1
  3. package/package.json +1 -1
  4. package/src/adapters/anthropic-image-codec.ts +57 -0
  5. package/src/adapters/anthropic-image-normalize.ts +28 -1
  6. package/src/adapters/anthropic.ts +68 -6
  7. package/src/adapters/base.ts +8 -0
  8. package/src/adapters/coding-agent/protocol.ts +41 -16
  9. package/src/adapters/cursor/cursor-errors.ts +1 -1
  10. package/src/adapters/cursor/live-transport.ts +5 -1
  11. package/src/adapters/cursor/native-exec-fs.ts +10 -10
  12. package/src/adapters/cursor/native-exec-network.ts +2 -2
  13. package/src/adapters/cursor/native-exec-shell.ts +13 -12
  14. package/src/adapters/cursor/native-exec.ts +51 -10
  15. package/src/adapters/cursor/policy-error.ts +75 -0
  16. package/src/adapters/cursor/protobuf-request.ts +105 -1
  17. package/src/adapters/devin/cloud-direct/catalog.ts +34 -2
  18. package/src/adapters/devin/live-models.ts +33 -2
  19. package/src/adapters/google-wire-compiler.ts +8 -0
  20. package/src/adapters/google.ts +46 -0
  21. package/src/adapters/input-media-guard.ts +45 -0
  22. package/src/adapters/kiro/adapter.ts +8 -0
  23. package/src/adapters/kiro/payload.ts +28 -6
  24. package/src/adapters/kiro-events.ts +25 -6
  25. package/src/adapters/kiro-images.ts +30 -0
  26. package/src/adapters/kiro-retry.ts +8 -0
  27. package/src/adapters/openai-chat.ts +33 -4
  28. package/src/adapters/openai-responses.ts +26 -0
  29. package/src/adapters/registry.ts +4 -0
  30. package/src/bridge.ts +163 -116
  31. package/src/chat/image-parts.ts +151 -0
  32. package/src/chat/inbound.ts +70 -33
  33. package/src/cli/connect.ts +30 -9
  34. package/src/cli/dispatch.ts +7 -3
  35. package/src/cli/index.ts +3 -0
  36. package/src/cli/runtime-api.ts +25 -0
  37. package/src/cli/status.ts +21 -19
  38. package/src/cli/system-restart-client.ts +25 -0
  39. package/src/clients/config-export.ts +14 -4
  40. package/src/codex/app-server-processes.ts +25 -0
  41. package/src/codex/auth-context.ts +8 -0
  42. package/src/codex/autostart-health.ts +36 -2
  43. package/src/codex/catalog/provider-fetch.ts +41 -0
  44. package/src/codex/catalog-auto-refresh.ts +182 -0
  45. package/src/codex/catalog-refresh-status.ts +93 -0
  46. package/src/codex/history-provider.ts +55 -0
  47. package/src/codex/model-entitlements.ts +78 -0
  48. package/src/codex/native-profile-processes.ts +114 -15
  49. package/src/codex/prompt-text-probe.ts +274 -41
  50. package/src/codex/routing-adoption.ts +189 -0
  51. package/src/codex/routing.ts +520 -48
  52. package/src/codex/runtime.ts +249 -7
  53. package/src/combos/failover.ts +45 -0
  54. package/src/config.ts +124 -4
  55. package/src/generated/compatibility-version.json +110 -74
  56. package/src/generated/model-metadata.ts +1 -0
  57. package/src/lib/request-execution-budget.ts +202 -0
  58. package/src/lib/upstream-retry.ts +95 -8
  59. package/src/lib/workflow-budget.ts +172 -0
  60. package/src/oauth/devin.ts +57 -12
  61. package/src/providers/quota.ts +37 -6
  62. package/src/providers/registry.ts +53 -6
  63. package/src/responses/input-media.ts +65 -0
  64. package/src/responses/parser-content.ts +42 -0
  65. package/src/responses/schema.ts +12 -2
  66. package/src/server/audio-live.ts +1 -2
  67. package/src/server/audio-transcriptions.ts +1 -2
  68. package/src/server/auth-cors.ts +1 -1
  69. package/src/server/background-lifecycle.ts +18 -0
  70. package/src/server/chat-completions.ts +23 -8
  71. package/src/server/chat-native.ts +17 -17
  72. package/src/server/index.ts +24 -0
  73. package/src/server/management/request-history-routes.ts +5 -0
  74. package/src/server/request-log.ts +8 -2
  75. package/src/server/responses/compact.ts +51 -3
  76. package/src/server/responses/core.ts +238 -28
  77. package/src/server/search.ts +7 -9
  78. package/src/types/config.ts +51 -6
  79. package/src/usage/log.ts +37 -0
  80. package/src/vision/eligibility.ts +37 -4
  81. package/src/vision/index.ts +1 -0
  82. package/src/vision/plan.ts +45 -10
  83. package/src/web-search/alpha-search.ts +324 -0
  84. package/src/web-search/index.ts +13 -22
  85. package/src/web-search/passthrough-bridge.ts +195 -22
  86. package/src/web-search/sidecar-providers.ts +22 -0
@@ -20,9 +20,28 @@ import { registerUser } from "./devin/register-user";
20
20
  import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiBaseUrl, validateDevinApiBaseUrl } from "./devin/api-base";
21
21
  import { readDevinCliCredentialOutcome } from "./devin/cli-import";
22
22
  import { getCredential } from "./store";
23
+ import { DEPRECATED_OAUTH_PROVIDER_ALIASES } from "./index";
23
24
 
24
25
  export { DEVIN_DEFAULT_API_SERVER } from "./devin/api-base";
25
26
 
27
+ /**
28
+ * Credential slots the deprecated-alias map ties to `providerId`, in both
29
+ * directions: a deprecated id also reads its destination's slot, and a merge
30
+ * destination also reads every deprecated source slot pointing at it. Derived
31
+ * from DEPRECATED_OAUTH_PROVIDER_ALIASES rather than a second "devin-cli"
32
+ * literal so the map stays the single source of truth — a hard-coded pair here
33
+ * would drift the day another alias is added.
34
+ */
35
+ function devinAliasCredentialSlots(providerId: string): string[] {
36
+ const slots: string[] = [];
37
+ const destination = DEPRECATED_OAUTH_PROVIDER_ALIASES[providerId];
38
+ if (destination !== undefined) slots.push(destination);
39
+ for (const [alias, target] of Object.entries(DEPRECATED_OAUTH_PROVIDER_ALIASES)) {
40
+ if (target === providerId && alias !== providerId) slots.push(alias);
41
+ }
42
+ return slots;
43
+ }
44
+
26
45
  /**
27
46
  * The api-server host this account must talk to.
28
47
  *
@@ -33,18 +52,44 @@ export { DEVIN_DEFAULT_API_SERVER } from "./devin/api-base";
33
52
  * network value.
34
53
  */
35
54
  export function resolveDevinApiServer(configuredBaseUrl?: string, providerId = "devin"): string {
36
- return (
37
- // Provider-scoped, keyed by the configured provider id verbatim. `devin-cli`
38
- // is a deprecated alias for `devin`, but an unmigrated config row still owns
39
- // its old credential slot until the startup migration rekeys the row and the
40
- // slot together — normalizing the id here would read the wrong slot for that
41
- // window. An EU or FedStart tenant is recorded on the credential rather than
42
- // in the registry, so a fixed "devin" slot would send the key to the wrong
43
- // host either way.
44
- validateDevinApiBaseUrl(getCredential(providerId)?.apiBaseUrl) ??
45
- validateDevinApiBaseUrl(configuredBaseUrl) ??
46
- DEVIN_DEFAULT_API_SERVER
47
- );
55
+ // Provider-scoped, keyed by the configured provider id verbatim and consulted
56
+ // FIRST. `devin-cli` is a deprecated alias for `devin`, but an unmigrated
57
+ // config row still owns its old credential slot until the startup migration
58
+ // rekeys the row and the slot together — normalizing the id here would read
59
+ // the wrong slot for that window. An EU or FedStart tenant is recorded on the
60
+ // credential rather than in the registry, so a fixed "devin" slot would send
61
+ // the key to the wrong host either way.
62
+ const literalCredential = getCredential(providerId);
63
+ const literal = validateDevinApiBaseUrl(literalCredential?.apiBaseUrl);
64
+ if (literal !== undefined) return literal;
65
+
66
+ // The startup merge saves providers["devin"] synchronously but fires the
67
+ // credential rekey detached — runDevinProviderMergeStartupMigration cannot
68
+ // await inside the synchronous startServer window — so the row can already
69
+ // say "devin" while the credential still sits in the "devin-cli" slot, and it
70
+ // stays that way for the whole process when the rekey fails or refuses on an
71
+ // occupied destination slot. Reading the alias-linked slots in both
72
+ // directions closes that window: "devin" finds the not-yet-rekeyed
73
+ // "devin-cli" credential, and a lingering "devin-cli" row finds a credential
74
+ // already rekeyed to "devin". Every candidate passes the same allowlist — an
75
+ // alias slot is not trusted more than the literal one.
76
+ // Only when this id owns no credential at all. A present credential whose
77
+ // apiBaseUrl is missing or off-allowlist is a different situation: the rekey
78
+ // refuses an occupied destination slot, so both ids can hold credentials that
79
+ // belong to two different accounts. Borrowing a tenant across that pair would
80
+ // send this account's key to the other account's EU or FedStart host, which
81
+ // is the exact misdirection the provider-scoped lookup exists to prevent. An
82
+ // unusable host on a credential that does exist falls through to the
83
+ // configured base URL and then the default, as it did before this window was
84
+ // closed.
85
+ if (literalCredential === null || literalCredential === undefined) {
86
+ for (const slot of devinAliasCredentialSlots(providerId)) {
87
+ const host = validateDevinApiBaseUrl(getCredential(slot)?.apiBaseUrl);
88
+ if (host !== undefined) return host;
89
+ }
90
+ }
91
+
92
+ return validateDevinApiBaseUrl(configuredBaseUrl) ?? DEVIN_DEFAULT_API_SERVER;
48
93
  }
49
94
 
50
95
  function decodeJwtPayload(token: string): Record<string, unknown> | undefined {
@@ -2951,7 +2951,26 @@ function unavailableAntigravityQuota(failure: QuotaFailureCode): AntigravityQuot
2951
2951
  return { kind: "unavailable", failure, legacy: { kind: "null" } };
2952
2952
  }
2953
2953
 
2954
- /** Final attempt determines the safe diagnosis; a successful fallback clears the first failure. */
2954
+ /**
2955
+ * Prefer a summary network-policy diagnosis over a vaguer fallback. A blocked
2956
+ * destination is an actionable local-network fact, while "upstream_error" tells
2957
+ * the operator to go look at Google. A successful models probe still clears
2958
+ * the first failure completely.
2959
+ */
2960
+ function antigravityUnavailableFailure(
2961
+ summaryFailure: QuotaFailureCode | undefined,
2962
+ fallbackFailure: QuotaFailureCode,
2963
+ ): QuotaFailureCode {
2964
+ if (
2965
+ (summaryFailure === "destination_blocked" || summaryFailure === "dns_failed")
2966
+ && fallbackFailure !== "destination_blocked"
2967
+ && fallbackFailure !== "dns_failed"
2968
+ ) {
2969
+ return summaryFailure;
2970
+ }
2971
+ return fallbackFailure;
2972
+ }
2973
+
2955
2974
  async function probeAntigravityUsageQuota(accessToken: string, projectId: string): Promise<AntigravityQuotaProbeResult> {
2956
2975
  const fetchQuota = (url: string) => providerOutboundPost("google-antigravity", { baseUrl: ANTIGRAVITY_ACCOUNT_QUOTA_BASE }, url, {
2957
2976
  headers: {
@@ -2960,6 +2979,7 @@ async function probeAntigravityUsageQuota(accessToken: string, projectId: string
2960
2979
  },
2961
2980
  body: JSON.stringify({ project: projectId }), signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
2962
2981
  }, antigravityOutboundDependencies);
2982
+ let summaryFailure: QuotaFailureCode | undefined;
2963
2983
  try {
2964
2984
  const response = await fetchQuota(ANTIGRAVITY_QUOTA_SUMMARY_URL);
2965
2985
  if (await providerRedirectError(response, ANTIGRAVITY_QUOTA_SUMMARY_URL)) return unavailableAntigravityQuota("redirect_blocked");
@@ -2968,19 +2988,30 @@ async function probeAntigravityUsageQuota(accessToken: string, projectId: string
2968
2988
  const quota = parseAntigravityQuotaSummary(asRecord(await readQuotaJson(response)));
2969
2989
  if (quota) return { kind: "available", quota, source: "google-antigravity:retrieveUserQuotaSummary" };
2970
2990
  }
2971
- } catch {
2991
+ } catch (error) {
2972
2992
  // Existing behavior: summary transport/parse failure may recover through the models probe.
2993
+ summaryFailure = quotaTransportFailure(error);
2973
2994
  }
2974
2995
  try {
2975
2996
  const response = await fetchQuota(ANTIGRAVITY_QUOTA_MODELS_URL);
2976
- if (await providerRedirectError(response, ANTIGRAVITY_QUOTA_MODELS_URL)) return unavailableAntigravityQuota("redirect_blocked");
2977
- if (!response.ok) return unavailableAntigravityQuota(quotaHttpFailure(response.status));
2997
+ if (await providerRedirectError(response, ANTIGRAVITY_QUOTA_MODELS_URL)) {
2998
+ return unavailableAntigravityQuota(antigravityUnavailableFailure(summaryFailure, "redirect_blocked"));
2999
+ }
3000
+ if (!response.ok) {
3001
+ return unavailableAntigravityQuota(antigravityUnavailableFailure(summaryFailure, quotaHttpFailure(response.status)));
3002
+ }
2978
3003
  const customWindows = antigravityWindowsFromModels(asRecord(await readQuotaJson(response)));
2979
- if (!customWindows.length) return unavailableAntigravityQuota("response_unusable");
3004
+ if (!customWindows.length) {
3005
+ return unavailableAntigravityQuota(antigravityUnavailableFailure(summaryFailure, "response_unusable"));
3006
+ }
2980
3007
  return { kind: "available", quota: { customWindows, updatedAt: Date.now() }, source: "google-antigravity:fetchAvailableModels" };
2981
3008
  } catch (error) {
2982
3009
  // The public compatibility wrapper still rejects this exact fallback error; it never enters a DTO.
2983
- return { kind: "unavailable", failure: quotaTransportFailure(error), legacy: { kind: "throw", error } };
3010
+ return {
3011
+ kind: "unavailable",
3012
+ failure: antigravityUnavailableFailure(summaryFailure, quotaTransportFailure(error)),
3013
+ legacy: { kind: "throw", error },
3014
+ };
2984
3015
  }
2985
3016
  }
2986
3017
 
@@ -707,9 +707,35 @@ const COMMAND_CODE_IMAGE_MODELS = [
707
707
  "meta/muse-spark-1.3-contributor",
708
708
  "meta/muse-spark-1.2",
709
709
  "meta/muse-spark-1.2-contributor",
710
+ // Native Z.AI VLM (docs.z.ai/guides/vlm/glm-5.3-flash). This exact id is already
711
+ // classified as natively vision-capable in NVIDIA_NIM_VISION_MODELS in this file;
712
+ // it is not one of the verified-negative ids the header names (those are
713
+ // deepseek/deepseek-v4-flash, zai-org/GLM-5.2, zai-org/GLM-5.3, xai/grok-4.6 —
714
+ // different ids). Adding it on the shared GLM-5.3 prefix would be the family-
715
+ // resemblance mistake the header forbids; the VLM docs are the evidence (#4505).
716
+ "z-ai/glm-5.3-flash",
717
+ ] as const;
718
+ /**
719
+ * Native image stays sourced from COMMAND_CODE_IMAGE_MODELS. Text-only routes
720
+ * sit beside that list so the catalog can still advertise sidecar coverage
721
+ * without claiming the gateway itself accepts a picture.
722
+ *
723
+ * The gateway-prefixed DeepSeek V4.1 Flash route has no verified native image
724
+ * support, so declaring it image-capable would hand it a picture it drops. A
725
+ * positive text-only declaration makes it a vision-sidecar consumer
726
+ * (src/vision/eligibility.ts), so the catalog advertises image input on its
727
+ * behalf and the four-target combo in #4505 intersects to ["text","image"]
728
+ * instead of ["text"] — without claiming native vision. modelInputModalities
729
+ * is per-key filled, so this reaches an existing install even when
730
+ * noVisionModels was persisted before the id joined that list.
731
+ */
732
+ const COMMAND_CODE_TEXT_ONLY_MODELS = [
733
+ "deepseek/deepseek-v4.1-flash",
710
734
  ] as const;
711
- const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text", "image"]> =
712
- Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"]]));
735
+ const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text"] | ["text", "image"]> = {
736
+ ...Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
737
+ ...Object.fromEntries(COMMAND_CODE_TEXT_ONLY_MODELS.map(id => [id, ["text"] as ["text"]])),
738
+ };
713
739
  const OPENCODE_FREE_DEEPSEEK_MODELS = ["deepseek-v4-flash-free"];
714
740
  /*
715
741
  * Zen free models that reject `image_url` upstream (#1043, and the reproducible
@@ -722,10 +748,9 @@ const OPENCODE_FREE_DEEPSEEK_MODELS = ["deepseek-v4-flash-free"];
722
748
  * `[404] No endpoints found that support image input` and `big-pickle` with the
723
749
  * exact deserialize error quoted in #1043.
724
750
  *
725
- * `mimo-v2.5-free` and `longcat-2.0-free` ACCEPT images and are deliberately
726
- * absent. Adding them would silently replace a working image with a caption,
727
- * which is worse than the loud 400 this list exists to prevent — see the negative
728
- * assertion in tests/providers/provider-registry-parity.test.ts.
751
+ * `mimo-v2.5-free` and `longcat-2.0-free` ACCEPT images. They remain absent
752
+ * from the blind list and are recorded separately as positive input-modality evidence,
753
+ * so capability-positive dispatch can forward images without relying on blacklist absence.
729
754
  *
730
755
  * Zen's roster is discovered live while this list is static, so it is a dated
731
756
  * exception list, not a capability model. Re-probe before extending it.
@@ -739,6 +764,7 @@ const OPENCODE_ZEN_TEXT_ONLY_MODELS = [
739
764
  "laguna-s-2.1-free",
740
765
  "deepseek-v4-flash-free",
741
766
  ];
767
+ const OPENCODE_ZEN_IMAGE_MODELS = ["mimo-v2.5-free", "longcat-2.0-free"] as const;
742
768
  /*
743
769
  * DeepSeek's Codex ladder is low/high/max. With the V4 Pro GA release
744
770
  * (DeepSeek-V4-Pro-0813) the official thinking-mode table is IDENTICAL for both
@@ -1856,8 +1882,27 @@ export const PROVIDER_REGISTRY: readonly ProviderRegistryEntry[] = [
1856
1882
  },
1857
1883
  modelInputModalities: {
1858
1884
  "kimi-k3": ["text", "image"],
1885
+ // glm-5.3-flash is a native VLM (docs.z.ai/guides/vlm/glm-5.3-flash). It is
1886
+ // deliberately absent from this preset's noVisionModels, which is the
1887
+ // correct NEGATIVE half, but with no positive modelInputModalities entry
1888
+ // configuredInputModalities returns undefined and the catalog falls through
1889
+ // to the ["text"] floor. The same model is already declared ["text","image"]
1890
+ // on the zai and zhipu-bigmodel-coding presets, so the registry described
1891
+ // one model two ways (#4505).
1892
+ "glm-5.3-flash": ["text", "image"],
1859
1893
  // Experimental DeepSeek vision preview — expected to merge into deepseek-v4-flash later.
1860
1894
  [DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
1895
+ // This route is text-only upstream — it is already listed in this preset's
1896
+ // noVisionModels, which routes images through the proxy's vision sidecar and
1897
+ // makes the catalog advertise image input on its behalf. The positive
1898
+ // text-only declaration is what reaches an EXISTING install: derive.ts fills
1899
+ // noVisionModels all-or-nothing, so a config persisted before this id joined
1900
+ // the list keeps a stale list, the sidecar predicate never matches, the row
1901
+ // carries no modality at all, and any combo containing it collapses to
1902
+ // ["text"] (#4505). modelInputModalities IS per-key filled, so this
1903
+ // declaration lands on old configs. It states the route's real upstream
1904
+ // capability and keeps the sidecar explicitly distinct from native vision.
1905
+ "deepseek-v4.1-flash": ["text"],
1861
1906
  // Muse Spark Contributor is natively multimodal on Zen Go: it accepts input_image
1862
1907
  // parts over /responses (probed 2026-08-26). Without this declaration the catalog
1863
1908
  // advertises it text-only and the Codex app blocks image attachments client-side with
@@ -3209,6 +3254,7 @@ export const PROVIDER_REGISTRY: readonly ProviderRegistryEntry[] = [
3209
3254
  },
3210
3255
  modelInputModalities: {
3211
3256
  [DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
3257
+ ...Object.fromEntries(OPENCODE_ZEN_IMAGE_MODELS.map(id => [id, ["text", "image"] as string[]])),
3212
3258
  },
3213
3259
  noVisionModels: [...OPENCODE_ZEN_TEXT_ONLY_MODELS, ...DEEPSEEK_GATEWAY_THINKING_MODELS],
3214
3260
  // Same DeepSeek routes as the Go preset above, behind the same vendor, so they carry
@@ -3250,6 +3296,7 @@ export const PROVIDER_REGISTRY: readonly ProviderRegistryEntry[] = [
3250
3296
  },
3251
3297
  modelInputModalities: {
3252
3298
  [DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
3299
+ ...Object.fromEntries(OPENCODE_ZEN_IMAGE_MODELS.map(id => [id, ["text", "image"] as string[]])),
3253
3300
  },
3254
3301
  // Same Zen roster behind the same base URL, so it carries the same measured
3255
3302
  // text-only list rather than only its DeepSeek member (#1043).
@@ -0,0 +1,65 @@
1
+ /** Input kinds for which the normalized request has no lossless content carrier. */
2
+ export type UntranslatedInputMedia = "audio" | "file";
3
+
4
+ type RecordValue = Record<string, unknown>;
5
+
6
+ function isRecord(value: unknown): value is RecordValue {
7
+ return value !== null && typeof value === "object" && !Array.isArray(value);
8
+ }
9
+
10
+ function mediaKind(value: unknown): UntranslatedInputMedia | undefined {
11
+ if (!isRecord(value)) return undefined;
12
+ if (value.type === "input_audio" || value.type === "audio") return "audio";
13
+ if (value.type === "input_file" || value.type === "file" || value.type === "document") return "file";
14
+ // A file-id-only image is not pixels: translated adapters cannot dereference it.
15
+ if (value.type === "input_image" && typeof value.file_id === "string" && value.file_id.length > 0
16
+ && !(typeof value.image_url === "string" && value.image_url.length > 0)) return "file";
17
+ return undefined;
18
+ }
19
+
20
+ function contentMedia(content: unknown): UntranslatedInputMedia | undefined {
21
+ if (!Array.isArray(content)) return undefined;
22
+ for (const part of content) {
23
+ const kind = mediaKind(part);
24
+ if (kind) return kind;
25
+ }
26
+ return undefined;
27
+ }
28
+
29
+ /**
30
+ * Inspect only typed input items and their content arrays, never strings, tool
31
+ * arguments, schema properties, or arbitrary nested objects. No payload is copied,
32
+ * decoded, fetched or included in the returned value.
33
+ */
34
+ export function untranslatedResponsesInputMedia(body: unknown): UntranslatedInputMedia | undefined {
35
+ if (!isRecord(body) || !Array.isArray(body.input)) return undefined;
36
+ for (const item of body.input) {
37
+ if (!isRecord(item)) continue;
38
+ const direct = mediaKind(item);
39
+ if (direct) return direct;
40
+ if (item.type === "function_call_output" || item.type === "custom_tool_call_output") {
41
+ const kind = contentMedia(item.output);
42
+ if (kind) return kind;
43
+ } else if (item.type === "message" || item.type === undefined) {
44
+ const kind = contentMedia(item.content);
45
+ if (kind) return kind;
46
+ }
47
+ }
48
+ return undefined;
49
+ }
50
+
51
+ /** Used only when Chat is actually projected, not on the native Chat fast path. */
52
+ export function untranslatedChatInputMedia(body: unknown): UntranslatedInputMedia | undefined {
53
+ if (!isRecord(body) || !Array.isArray(body.messages)) return undefined;
54
+ for (const message of body.messages) {
55
+ if (!isRecord(message)) continue;
56
+ const kind = contentMedia(message.content);
57
+ if (kind) return kind;
58
+ }
59
+ return undefined;
60
+ }
61
+
62
+ /** Fixed vocabulary only: never interpolate filenames, URLs or client metadata. */
63
+ export function untranslatedInputMediaMessage(kind: UntranslatedInputMedia): string {
64
+ return `OpenCodex cannot translate ${kind} input on this route. Use a native input wire that supports the attachment, or convert it to text first.`;
65
+ }
@@ -9,6 +9,8 @@ type InputBlock =
9
9
  | { type: "text"; text: string }
10
10
  | { type: "input_image"; image_url?: string; file_id?: string; detail?: string }
11
11
  | { type: "input_video"; video_url?: string }
12
+ // codex-rs protocol/src/models.rs sends audio as input_audio with an audio_url.
13
+ | { type: "input_audio"; audio_url?: string; format?: string }
12
14
  | { type: "input_file"; file_id?: string; filename?: string; file_data?: string };
13
15
 
14
16
  /** A usable reference string, or undefined. Empty strings and non-strings are not references. */
@@ -16,6 +18,19 @@ function nonEmptyString(value: unknown): string | undefined {
16
18
  return typeof value === "string" && value.length > 0 ? value : undefined;
17
19
  }
18
20
 
21
+ /**
22
+ * An audio format label safe to render into model-visible prose.
23
+ *
24
+ * `format` is caller-controlled and unbounded in the schema, so interpolating it
25
+ * verbatim would let a request park newlines, injected instructions, or a signed URL
26
+ * inside text the model reads as trusted proxy output. Only a short alphanumeric
27
+ * token is echoed; anything else degrades to the bare marker.
28
+ */
29
+ function safeAudioFormat(value: unknown): string | undefined {
30
+ const raw = nonEmptyString(value);
31
+ return raw !== undefined && /^[a-z0-9]{1,12}$/i.test(raw) ? raw : undefined;
32
+ }
33
+
19
34
  export function inputContentParts(blocks: unknown): string | OcxContentPart[] {
20
35
  if (typeof blocks === "string") return blocks;
21
36
  // The catch-all can also hand back a non-array `content` (an object, a number), which would
@@ -47,6 +62,26 @@ export function inputContentParts(blocks: unknown): string | OcxContentPart[] {
47
62
  } else if (block.type === "input_video") {
48
63
  const videoUrl = nonEmptyString(block.video_url);
49
64
  if (videoUrl) parts.push({ type: "video", videoUrl });
65
+ } else if (block.type === "input_audio") {
66
+ // Upstream Codex sends input_audio with an audio_url (codex-rs
67
+ // protocol/src/models.rs). The IR has no audio carrier and no adapter consumes
68
+ // one, so this part used to vanish with no trace at all.
69
+ //
70
+ // This records PRESENCE only and is NOT audio support: never the payload, which
71
+ // is large base64 and would explode the token count, and never the URL, which
72
+ // can carry a signed token. This parser must stay non-throwing — the native
73
+ // Responses passthrough also runs through parseRequest before the adapter
74
+ // forwards _rawBody, so refusing here would regress legitimate raw passthrough.
75
+ //
76
+ // No adapter refuses audio at its wire today: by this point the part is already a
77
+ // text marker, so downstream adapters see text and continue. A typed unsupported-
78
+ // modality carrier that survives to final adapter dispatch is a separate, recorded
79
+ // residual — do not describe this branch as a refusal.
80
+ const b = block as { audio_url?: string; format?: string };
81
+ const format = safeAudioFormat(b.format);
82
+ if (nonEmptyString(b.audio_url)) {
83
+ parts.push({ type: "text", text: format ? `[audio: ${format}]` : "[audio]" });
84
+ }
50
85
  } else if (block.type === "input_file") {
51
86
  const b = block as { file_id?: string; filename?: string; file_data?: string };
52
87
  const fileId = nonEmptyString(b.file_id);
@@ -111,6 +146,13 @@ export function outputToToolResultContent(output: string | unknown[] | undefined
111
146
  } else if (fileId) {
112
147
  parts.push({ type: "text", text: `[image: ${fileId}]` });
113
148
  }
149
+ } else if (raw.type === "input_audio") {
150
+ // Same presence-only contract as the user-content branch above: Codex returns
151
+ // audio in tool output too, and it previously disappeared without trace.
152
+ const format = safeAudioFormat(raw.format);
153
+ if (nonEmptyString(raw.audio_url)) {
154
+ parts.push({ type: "text", text: format ? `[audio: ${format}]` : "[audio]" });
155
+ }
114
156
  } else if (raw.type === "encrypted_content") {
115
157
  // codex-rs FunctionCallOutputContentItem::EncryptedContent — opaque to routed models.
116
158
  parts.push({ type: "text", text: "[encrypted content omitted]" });
@@ -21,6 +21,16 @@ const inputFileBlockSchema = z.object({
21
21
  filename: z.string().optional(),
22
22
  file_data: z.string().optional(),
23
23
  });
24
+ // codex-rs protocol/src/models.rs sends audio as input_audio with an audio_url, in
25
+ // both user content and tool output. Accepting the block keeps a legitimate audio turn
26
+ // out of the malformed-item catch-all. The translated IR records only its PRESENCE —
27
+ // there is no audio carrier and no adapter-level refusal; a typed unsupported-modality
28
+ // signal reaching final adapter dispatch remains a recorded residual.
29
+ const inputAudioBlockSchema = z.object({
30
+ type: z.literal("input_audio"),
31
+ audio_url: z.string().min(1),
32
+ format: z.string().optional(),
33
+ });
24
34
  const outputTextSchema = z.object({ type: z.literal("output_text"), text: z.string() });
25
35
  const outputRefusalSchema = z.object({ type: z.literal("refusal"), refusal: z.string() });
26
36
  const summaryTextSchema = z.object({ type: z.literal("summary_text"), text: z.string() });
@@ -28,12 +38,12 @@ const reasoningTextSchema = z.object({ type: z.literal("reasoning_text"), text:
28
38
  // codex-rs FunctionCallOutputContentItem (protocol/src/models.rs): input_text | input_image | encrypted_content.
29
39
  const encryptedContentBlockSchema = z.object({ type: z.literal("encrypted_content"), encrypted_content: z.string() });
30
40
 
31
- const inputContentBlockSchema = z.union([inputTextSchema, plainTextSchema, inputImageBlockSchema, inputVideoBlockSchema, inputFileBlockSchema]);
41
+ const inputContentBlockSchema = z.union([inputTextSchema, plainTextSchema, inputImageBlockSchema, inputVideoBlockSchema, inputAudioBlockSchema, inputFileBlockSchema]);
32
42
  const outputContentBlockSchema = z.union([outputTextSchema, plainTextSchema, outputRefusalSchema]);
33
43
  // Tool outputs on the wire mix codex-rs FunctionCallOutputContentItem with legacy output blocks.
34
44
  const toolOutputContentBlockSchema = z.union([
35
45
  outputTextSchema, plainTextSchema, outputRefusalSchema,
36
- inputTextSchema, inputImageBlockSchema, encryptedContentBlockSchema,
46
+ inputTextSchema, inputImageBlockSchema, inputAudioBlockSchema, encryptedContentBlockSchema,
37
47
  ]);
38
48
  const toolOutputSchema = z.union([z.string(), z.array(toolOutputContentBlockSchema)]);
39
49
 
@@ -100,7 +100,7 @@ export async function handleExternalLive(
100
100
  ? frameless ? forwardLiveUrl(relay.providerBaseUrl, false) : keyedLiveUrl(relay.providerBaseUrl)
101
101
  : forwardLiveUrl(relay.providerBaseUrl, true);
102
102
  const upstream = await fetch(url, { method: "POST", headers, body, signal: deadline.signal, redirect: "manual" });
103
- outcome = upstream.ok ? 502 : upstream.status;
103
+ outcome = upstream.status;
104
104
  const detach = cancelBodyOnAbort(upstream.body, deadline.signal);
105
105
  let responseBody: ArrayBuffer | Response;
106
106
  try { responseBody = await readBodyCapped(upstream.body, LIVE_RESPONSE_MAX_BYTES, () => "Live answer too large", deadline.signal); }
@@ -122,7 +122,6 @@ export async function handleExternalLive(
122
122
  sidebandBaseUrl: config.experimentalRealtimeWsBaseUrl,
123
123
  });
124
124
  if (!alias) return formatErrorResponse(503, "server_busy", "Live call could not be registered");
125
- outcome = upstream.status;
126
125
  return new Response(responseBody, { status: upstream.status, headers: {
127
126
  "content-type": upstream.headers.get("content-type") ?? "application/sdp",
128
127
  location: `/v1/${frameless ? "live" : "realtime/calls"}/${alias}`,
@@ -117,7 +117,7 @@ async function transcribeAdmitted(
117
117
  form.append("response_format", "json");
118
118
  }
119
119
  const upstream = await fetch(url, { method: "POST", headers, body: form, signal: signal.signal, redirect: "manual" });
120
- outcome = upstream.ok ? 502 : upstream.status;
120
+ outcome = upstream.status;
121
121
  const detach = cancelBodyOnAbort(upstream.body, signal.signal);
122
122
  let body: ArrayBuffer | Response;
123
123
  try {
@@ -137,7 +137,6 @@ async function transcribeAdmitted(
137
137
  if (!payload || typeof payload !== "object" || !("text" in payload) || typeof payload.text !== "string") {
138
138
  return formatErrorResponse(502, "upstream_error", "Audio upstream response is missing text");
139
139
  }
140
- outcome = upstream.status;
141
140
  return input.format === "text"
142
141
  ? new Response(payload.text, { headers: { "content-type": "text/plain; charset=utf-8" } })
143
142
  : Response.json({ text: payload.text });
@@ -766,7 +766,7 @@ export function providerManagementConfigError(
766
766
  if (requestPacingError) {
767
767
  return `provider ${JSON.stringify(redactSecretString(name))} ${requestPacingError}`;
768
768
  }
769
- const webSearchBridgeError = providerWebSearchBridgeConfigError(raw.webSearchBridge);
769
+ const webSearchBridgeError = providerWebSearchBridgeConfigError(raw.webSearchBridge, name, typed);
770
770
  if (webSearchBridgeError) {
771
771
  return `provider ${JSON.stringify(redactSecretString(name))} ${webSearchBridgeError}`;
772
772
  }
@@ -11,6 +11,11 @@ import {
11
11
  stopStorageCleanupScheduler,
12
12
  } from "../storage/policy-scheduler";
13
13
  import { startQuotaResetPoller, stopQuotaResetPoller } from "../quota/reset-poller";
14
+ import {
15
+ startCatalogAutoRefresh,
16
+ stopCatalogAutoRefresh,
17
+ syncCatalogAutoRefreshCadence,
18
+ } from "../codex/catalog-auto-refresh";
14
19
  import {
15
20
  cancelQueuedStorageWorkerSpawns,
16
21
  drainStorageWorkers,
@@ -70,6 +75,17 @@ function startProcessLoops(applyPolicy: PolicyApply): ProcessLoops {
70
75
  .catch(() => {
71
76
  // The next tick adopts it.
72
77
  });
78
+ // Opt-in: the tick is a no-op unless catalogAutoRefresh.enabled is true, and the
79
+ // interval is unref'd, so a default install pays one dormant timer. The scheduler
80
+ // module keeps every heavy import inside its tick, so naming it statically here
81
+ // costs a module record and nothing else.
82
+ startCatalogAutoRefresh();
83
+ // The scheduler starts at its default cadence because resolving the operator's value
84
+ // reads the config barrel. Fire-and-forget: startup must not await an optional
85
+ // subsystem, and the next tick adopts the cadence anyway.
86
+ void syncCatalogAutoRefreshCadence().catch(() => {
87
+ // The next tick adopts it.
88
+ });
73
89
  // Install the delivery sink now rather than waiting out the first poll interval, which is 15
74
90
  // minutes by default. Without this, an enabled install would observe nothing for its first
75
91
  // quarter hour — including the live request path, which is gated on the sink existing.
@@ -85,6 +101,7 @@ function startProcessLoops(applyPolicy: PolicyApply): ProcessLoops {
85
101
  stateStoreSweeper?.stop();
86
102
  stopStorageCleanupScheduler();
87
103
  stopQuotaResetPoller();
104
+ stopCatalogAutoRefresh();
88
105
  setLivePolicyOwner(null);
89
106
  throw error;
90
107
  }
@@ -97,6 +114,7 @@ function stopProcessLoops(): void {
97
114
  loops?.stateStoreSweeper.stop();
98
115
  stopStorageCleanupScheduler();
99
116
  stopQuotaResetPoller();
117
+ stopCatalogAutoRefresh();
100
118
  setLivePolicyOwner(null);
101
119
  }
102
120
 
@@ -11,6 +11,7 @@ import {
11
11
  ChatCompletionsRequestError,
12
12
  chatCompletionsToResponsesBody,
13
13
  } from "../chat/inbound";
14
+ import { normalizeChatImageParts } from "../chat/image-parts";
14
15
  import {
15
16
  chatCompletionsErrorResponse,
16
17
  collectChatCompletion,
@@ -111,7 +112,11 @@ async function handleChatCompletionsWithBudget(
111
112
  try {
112
113
  const rawBody = await readChatBody(req, translatorBudget, resolveInboundBodyLimitBytes(config.maxInboundBodyBytes));
113
114
  assertChatCompletionsRoutingBody(rawBody);
114
- chatBody = rawBody;
115
+ // Normalize foreign image shapes BEFORE routing. isNativeChatRouteEligible below
116
+ // decides the pipeline from the image parts it can see, and the native path then
117
+ // forwards this body as-is, so both must observe the same parts. A body with no
118
+ // foreign image part is returned by reference and stays byte-identical.
119
+ chatBody = normalizeChatImageParts(rawBody);
115
120
  } catch (err) {
116
121
  const overflow = isTranslatorBudgetExceededError(err);
117
122
  const status = overflow ? 413 : err instanceof ChatCompletionsRequestError ? 400 : 500;
@@ -165,7 +170,7 @@ async function handleChatCompletionsWithBudget(
165
170
  if (chatBody.tools !== undefined) parts.push(JSON.stringify(chatBody.tools));
166
171
  logCtx.usageLogInputTokens = Math.max(1, estimateTokens(parts.join("\n"), requestedModel));
167
172
  }
168
- if (!effortRow && isNativeChatRouteEligible(route, chatBody)) chatNativeRoute = route;
173
+ if (!effortRow && isNativeChatRouteEligible(route, chatBody, config)) chatNativeRoute = route;
169
174
  } catch (err) {
170
175
  if (err instanceof UnknownRoutingPolicyError) {
171
176
  logCtx.requestedModel = requestedModel;
@@ -221,13 +226,23 @@ async function handleChatCompletionsWithBudget(
221
226
  // for non-streaming clients. Native Chat uses the caller's original stream bit.
222
227
  internalBody.stream = true;
223
228
  if (settledRoute?.provider.adapter === "openai-responses") {
224
- // ChatGPT backend rejects store:true and unsupported sampling knobs.
229
+ // The proxy never wants upstream-side retention for a translated Chat turn, so
230
+ // store stays pinned for every Responses route.
231
+ //
232
+ // The sampling and output-cap restrictions used to be applied here too, keyed on
233
+ // the adapter string. That was wrong twice over. Seven providers share this
234
+ // adapter (openai, openai-apikey, meta-model, meta-muse, zai,
235
+ // zhipu-bigmodel-responses, volcengine-agent-plan), so a generic key gateway lost
236
+ // controls it accepts. And settledRoute is the route settled at INGRESS: a combo
237
+ // or policy route resolves its concrete child later in the Responses pipeline, so
238
+ // deciding here mutates shared intent before the real target is known — a
239
+ // canonical-first combo that falls back to a key gateway had already lost the
240
+ // caller's controls, while a non-canonical-first combo that falls back to
241
+ // canonical still shipped them.
242
+ //
243
+ // Canonical-backend sanitization now happens at the final outgoing body in
244
+ // src/adapters/openai-responses.ts, where the concrete provider is known.
225
245
  internalBody.store = false;
226
- delete internalBody.max_output_tokens;
227
- delete internalBody.temperature;
228
- delete internalBody.top_p;
229
- delete internalBody.stop;
230
- delete internalBody.user;
231
246
  } else if (internalBody.store === undefined) {
232
247
  internalBody.store = false;
233
248
  }