@bitkyc08/opencodex 2.54.0 → 2.55.0-preview.20260914
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/{index-CkvITofZ.js → index-DH2PUHqr.js} +10 -10
- package/gui/dist/index.html +1 -1
- package/package.json +1 -1
- package/src/adapters/anthropic-image-codec.ts +57 -0
- package/src/adapters/anthropic-image-normalize.ts +28 -1
- package/src/adapters/anthropic.ts +68 -6
- package/src/adapters/base.ts +8 -0
- package/src/adapters/coding-agent/protocol.ts +41 -16
- package/src/adapters/cursor/cursor-errors.ts +1 -1
- package/src/adapters/cursor/live-transport.ts +5 -1
- package/src/adapters/cursor/native-exec-fs.ts +10 -10
- package/src/adapters/cursor/native-exec-network.ts +2 -2
- package/src/adapters/cursor/native-exec-shell.ts +13 -12
- package/src/adapters/cursor/native-exec.ts +51 -10
- package/src/adapters/cursor/policy-error.ts +75 -0
- package/src/adapters/cursor/protobuf-request.ts +105 -1
- package/src/adapters/devin/cloud-direct/catalog.ts +34 -2
- package/src/adapters/devin/live-models.ts +33 -2
- package/src/adapters/google-wire-compiler.ts +8 -0
- package/src/adapters/google.ts +46 -0
- package/src/adapters/input-media-guard.ts +45 -0
- package/src/adapters/kiro/adapter.ts +8 -0
- package/src/adapters/kiro/payload.ts +28 -6
- package/src/adapters/kiro-events.ts +25 -6
- package/src/adapters/kiro-images.ts +30 -0
- package/src/adapters/kiro-retry.ts +8 -0
- package/src/adapters/openai-chat.ts +33 -4
- package/src/adapters/openai-responses.ts +26 -0
- package/src/adapters/registry.ts +4 -0
- package/src/bridge.ts +163 -116
- package/src/chat/image-parts.ts +151 -0
- package/src/chat/inbound.ts +70 -33
- package/src/cli/connect.ts +30 -9
- package/src/cli/dispatch.ts +7 -3
- package/src/cli/index.ts +3 -0
- package/src/cli/runtime-api.ts +25 -0
- package/src/cli/status.ts +21 -19
- package/src/cli/system-restart-client.ts +25 -0
- package/src/clients/config-export.ts +14 -4
- package/src/codex/app-server-processes.ts +25 -0
- package/src/codex/auth-context.ts +8 -0
- package/src/codex/autostart-health.ts +36 -2
- package/src/codex/catalog/provider-fetch.ts +41 -0
- package/src/codex/catalog-auto-refresh.ts +182 -0
- package/src/codex/catalog-refresh-status.ts +93 -0
- package/src/codex/history-provider.ts +55 -0
- package/src/codex/model-entitlements.ts +78 -0
- package/src/codex/native-profile-processes.ts +114 -15
- package/src/codex/prompt-text-probe.ts +274 -41
- package/src/codex/routing-adoption.ts +189 -0
- package/src/codex/routing.ts +520 -48
- package/src/codex/runtime.ts +249 -7
- package/src/combos/failover.ts +45 -0
- package/src/config.ts +124 -4
- package/src/generated/compatibility-version.json +110 -74
- package/src/generated/model-metadata.ts +1 -0
- package/src/lib/request-execution-budget.ts +202 -0
- package/src/lib/upstream-retry.ts +95 -8
- package/src/lib/workflow-budget.ts +172 -0
- package/src/oauth/devin.ts +57 -12
- package/src/providers/quota.ts +37 -6
- package/src/providers/registry.ts +53 -6
- package/src/responses/input-media.ts +65 -0
- package/src/responses/parser-content.ts +42 -0
- package/src/responses/schema.ts +12 -2
- package/src/server/audio-live.ts +1 -2
- package/src/server/audio-transcriptions.ts +1 -2
- package/src/server/auth-cors.ts +1 -1
- package/src/server/background-lifecycle.ts +18 -0
- package/src/server/chat-completions.ts +23 -8
- package/src/server/chat-native.ts +17 -17
- package/src/server/index.ts +24 -0
- package/src/server/management/request-history-routes.ts +5 -0
- package/src/server/request-log.ts +8 -2
- package/src/server/responses/compact.ts +51 -3
- package/src/server/responses/core.ts +238 -28
- package/src/server/search.ts +7 -9
- package/src/types/config.ts +51 -6
- package/src/usage/log.ts +37 -0
- package/src/vision/eligibility.ts +37 -4
- package/src/vision/index.ts +1 -0
- package/src/vision/plan.ts +45 -10
- package/src/web-search/alpha-search.ts +324 -0
- package/src/web-search/index.ts +13 -22
- package/src/web-search/passthrough-bridge.ts +195 -22
- package/src/web-search/sidecar-providers.ts +22 -0
package/src/oauth/devin.ts
CHANGED
|
@@ -20,9 +20,28 @@ import { registerUser } from "./devin/register-user";
|
|
|
20
20
|
import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiBaseUrl, validateDevinApiBaseUrl } from "./devin/api-base";
|
|
21
21
|
import { readDevinCliCredentialOutcome } from "./devin/cli-import";
|
|
22
22
|
import { getCredential } from "./store";
|
|
23
|
+
import { DEPRECATED_OAUTH_PROVIDER_ALIASES } from "./index";
|
|
23
24
|
|
|
24
25
|
export { DEVIN_DEFAULT_API_SERVER } from "./devin/api-base";
|
|
25
26
|
|
|
27
|
+
/**
|
|
28
|
+
* Credential slots the deprecated-alias map ties to `providerId`, in both
|
|
29
|
+
* directions: a deprecated id also reads its destination's slot, and a merge
|
|
30
|
+
* destination also reads every deprecated source slot pointing at it. Derived
|
|
31
|
+
* from DEPRECATED_OAUTH_PROVIDER_ALIASES rather than a second "devin-cli"
|
|
32
|
+
* literal so the map stays the single source of truth — a hard-coded pair here
|
|
33
|
+
* would drift the day another alias is added.
|
|
34
|
+
*/
|
|
35
|
+
function devinAliasCredentialSlots(providerId: string): string[] {
|
|
36
|
+
const slots: string[] = [];
|
|
37
|
+
const destination = DEPRECATED_OAUTH_PROVIDER_ALIASES[providerId];
|
|
38
|
+
if (destination !== undefined) slots.push(destination);
|
|
39
|
+
for (const [alias, target] of Object.entries(DEPRECATED_OAUTH_PROVIDER_ALIASES)) {
|
|
40
|
+
if (target === providerId && alias !== providerId) slots.push(alias);
|
|
41
|
+
}
|
|
42
|
+
return slots;
|
|
43
|
+
}
|
|
44
|
+
|
|
26
45
|
/**
|
|
27
46
|
* The api-server host this account must talk to.
|
|
28
47
|
*
|
|
@@ -33,18 +52,44 @@ export { DEVIN_DEFAULT_API_SERVER } from "./devin/api-base";
|
|
|
33
52
|
* network value.
|
|
34
53
|
*/
|
|
35
54
|
export function resolveDevinApiServer(configuredBaseUrl?: string, providerId = "devin"): string {
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
55
|
+
// Provider-scoped, keyed by the configured provider id verbatim and consulted
|
|
56
|
+
// FIRST. `devin-cli` is a deprecated alias for `devin`, but an unmigrated
|
|
57
|
+
// config row still owns its old credential slot until the startup migration
|
|
58
|
+
// rekeys the row and the slot together — normalizing the id here would read
|
|
59
|
+
// the wrong slot for that window. An EU or FedStart tenant is recorded on the
|
|
60
|
+
// credential rather than in the registry, so a fixed "devin" slot would send
|
|
61
|
+
// the key to the wrong host either way.
|
|
62
|
+
const literalCredential = getCredential(providerId);
|
|
63
|
+
const literal = validateDevinApiBaseUrl(literalCredential?.apiBaseUrl);
|
|
64
|
+
if (literal !== undefined) return literal;
|
|
65
|
+
|
|
66
|
+
// The startup merge saves providers["devin"] synchronously but fires the
|
|
67
|
+
// credential rekey detached — runDevinProviderMergeStartupMigration cannot
|
|
68
|
+
// await inside the synchronous startServer window — so the row can already
|
|
69
|
+
// say "devin" while the credential still sits in the "devin-cli" slot, and it
|
|
70
|
+
// stays that way for the whole process when the rekey fails or refuses on an
|
|
71
|
+
// occupied destination slot. Reading the alias-linked slots in both
|
|
72
|
+
// directions closes that window: "devin" finds the not-yet-rekeyed
|
|
73
|
+
// "devin-cli" credential, and a lingering "devin-cli" row finds a credential
|
|
74
|
+
// already rekeyed to "devin". Every candidate passes the same allowlist — an
|
|
75
|
+
// alias slot is not trusted more than the literal one.
|
|
76
|
+
// Only when this id owns no credential at all. A present credential whose
|
|
77
|
+
// apiBaseUrl is missing or off-allowlist is a different situation: the rekey
|
|
78
|
+
// refuses an occupied destination slot, so both ids can hold credentials that
|
|
79
|
+
// belong to two different accounts. Borrowing a tenant across that pair would
|
|
80
|
+
// send this account's key to the other account's EU or FedStart host, which
|
|
81
|
+
// is the exact misdirection the provider-scoped lookup exists to prevent. An
|
|
82
|
+
// unusable host on a credential that does exist falls through to the
|
|
83
|
+
// configured base URL and then the default, as it did before this window was
|
|
84
|
+
// closed.
|
|
85
|
+
if (literalCredential === null || literalCredential === undefined) {
|
|
86
|
+
for (const slot of devinAliasCredentialSlots(providerId)) {
|
|
87
|
+
const host = validateDevinApiBaseUrl(getCredential(slot)?.apiBaseUrl);
|
|
88
|
+
if (host !== undefined) return host;
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
return validateDevinApiBaseUrl(configuredBaseUrl) ?? DEVIN_DEFAULT_API_SERVER;
|
|
48
93
|
}
|
|
49
94
|
|
|
50
95
|
function decodeJwtPayload(token: string): Record<string, unknown> | undefined {
|
package/src/providers/quota.ts
CHANGED
|
@@ -2951,7 +2951,26 @@ function unavailableAntigravityQuota(failure: QuotaFailureCode): AntigravityQuot
|
|
|
2951
2951
|
return { kind: "unavailable", failure, legacy: { kind: "null" } };
|
|
2952
2952
|
}
|
|
2953
2953
|
|
|
2954
|
-
/**
|
|
2954
|
+
/**
|
|
2955
|
+
* Prefer a summary network-policy diagnosis over a vaguer fallback. A blocked
|
|
2956
|
+
* destination is an actionable local-network fact, while "upstream_error" tells
|
|
2957
|
+
* the operator to go look at Google. A successful models probe still clears
|
|
2958
|
+
* the first failure completely.
|
|
2959
|
+
*/
|
|
2960
|
+
function antigravityUnavailableFailure(
|
|
2961
|
+
summaryFailure: QuotaFailureCode | undefined,
|
|
2962
|
+
fallbackFailure: QuotaFailureCode,
|
|
2963
|
+
): QuotaFailureCode {
|
|
2964
|
+
if (
|
|
2965
|
+
(summaryFailure === "destination_blocked" || summaryFailure === "dns_failed")
|
|
2966
|
+
&& fallbackFailure !== "destination_blocked"
|
|
2967
|
+
&& fallbackFailure !== "dns_failed"
|
|
2968
|
+
) {
|
|
2969
|
+
return summaryFailure;
|
|
2970
|
+
}
|
|
2971
|
+
return fallbackFailure;
|
|
2972
|
+
}
|
|
2973
|
+
|
|
2955
2974
|
async function probeAntigravityUsageQuota(accessToken: string, projectId: string): Promise<AntigravityQuotaProbeResult> {
|
|
2956
2975
|
const fetchQuota = (url: string) => providerOutboundPost("google-antigravity", { baseUrl: ANTIGRAVITY_ACCOUNT_QUOTA_BASE }, url, {
|
|
2957
2976
|
headers: {
|
|
@@ -2960,6 +2979,7 @@ async function probeAntigravityUsageQuota(accessToken: string, projectId: string
|
|
|
2960
2979
|
},
|
|
2961
2980
|
body: JSON.stringify({ project: projectId }), signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
|
|
2962
2981
|
}, antigravityOutboundDependencies);
|
|
2982
|
+
let summaryFailure: QuotaFailureCode | undefined;
|
|
2963
2983
|
try {
|
|
2964
2984
|
const response = await fetchQuota(ANTIGRAVITY_QUOTA_SUMMARY_URL);
|
|
2965
2985
|
if (await providerRedirectError(response, ANTIGRAVITY_QUOTA_SUMMARY_URL)) return unavailableAntigravityQuota("redirect_blocked");
|
|
@@ -2968,19 +2988,30 @@ async function probeAntigravityUsageQuota(accessToken: string, projectId: string
|
|
|
2968
2988
|
const quota = parseAntigravityQuotaSummary(asRecord(await readQuotaJson(response)));
|
|
2969
2989
|
if (quota) return { kind: "available", quota, source: "google-antigravity:retrieveUserQuotaSummary" };
|
|
2970
2990
|
}
|
|
2971
|
-
} catch {
|
|
2991
|
+
} catch (error) {
|
|
2972
2992
|
// Existing behavior: summary transport/parse failure may recover through the models probe.
|
|
2993
|
+
summaryFailure = quotaTransportFailure(error);
|
|
2973
2994
|
}
|
|
2974
2995
|
try {
|
|
2975
2996
|
const response = await fetchQuota(ANTIGRAVITY_QUOTA_MODELS_URL);
|
|
2976
|
-
if (await providerRedirectError(response, ANTIGRAVITY_QUOTA_MODELS_URL))
|
|
2977
|
-
|
|
2997
|
+
if (await providerRedirectError(response, ANTIGRAVITY_QUOTA_MODELS_URL)) {
|
|
2998
|
+
return unavailableAntigravityQuota(antigravityUnavailableFailure(summaryFailure, "redirect_blocked"));
|
|
2999
|
+
}
|
|
3000
|
+
if (!response.ok) {
|
|
3001
|
+
return unavailableAntigravityQuota(antigravityUnavailableFailure(summaryFailure, quotaHttpFailure(response.status)));
|
|
3002
|
+
}
|
|
2978
3003
|
const customWindows = antigravityWindowsFromModels(asRecord(await readQuotaJson(response)));
|
|
2979
|
-
if (!customWindows.length)
|
|
3004
|
+
if (!customWindows.length) {
|
|
3005
|
+
return unavailableAntigravityQuota(antigravityUnavailableFailure(summaryFailure, "response_unusable"));
|
|
3006
|
+
}
|
|
2980
3007
|
return { kind: "available", quota: { customWindows, updatedAt: Date.now() }, source: "google-antigravity:fetchAvailableModels" };
|
|
2981
3008
|
} catch (error) {
|
|
2982
3009
|
// The public compatibility wrapper still rejects this exact fallback error; it never enters a DTO.
|
|
2983
|
-
return {
|
|
3010
|
+
return {
|
|
3011
|
+
kind: "unavailable",
|
|
3012
|
+
failure: antigravityUnavailableFailure(summaryFailure, quotaTransportFailure(error)),
|
|
3013
|
+
legacy: { kind: "throw", error },
|
|
3014
|
+
};
|
|
2984
3015
|
}
|
|
2985
3016
|
}
|
|
2986
3017
|
|
|
@@ -707,9 +707,35 @@ const COMMAND_CODE_IMAGE_MODELS = [
|
|
|
707
707
|
"meta/muse-spark-1.3-contributor",
|
|
708
708
|
"meta/muse-spark-1.2",
|
|
709
709
|
"meta/muse-spark-1.2-contributor",
|
|
710
|
+
// Native Z.AI VLM (docs.z.ai/guides/vlm/glm-5.3-flash). This exact id is already
|
|
711
|
+
// classified as natively vision-capable in NVIDIA_NIM_VISION_MODELS in this file;
|
|
712
|
+
// it is not one of the verified-negative ids the header names (those are
|
|
713
|
+
// deepseek/deepseek-v4-flash, zai-org/GLM-5.2, zai-org/GLM-5.3, xai/grok-4.6 —
|
|
714
|
+
// different ids). Adding it on the shared GLM-5.3 prefix would be the family-
|
|
715
|
+
// resemblance mistake the header forbids; the VLM docs are the evidence (#4505).
|
|
716
|
+
"z-ai/glm-5.3-flash",
|
|
717
|
+
] as const;
|
|
718
|
+
/**
|
|
719
|
+
* Native image stays sourced from COMMAND_CODE_IMAGE_MODELS. Text-only routes
|
|
720
|
+
* sit beside that list so the catalog can still advertise sidecar coverage
|
|
721
|
+
* without claiming the gateway itself accepts a picture.
|
|
722
|
+
*
|
|
723
|
+
* The gateway-prefixed DeepSeek V4.1 Flash route has no verified native image
|
|
724
|
+
* support, so declaring it image-capable would hand it a picture it drops. A
|
|
725
|
+
* positive text-only declaration makes it a vision-sidecar consumer
|
|
726
|
+
* (src/vision/eligibility.ts), so the catalog advertises image input on its
|
|
727
|
+
* behalf and the four-target combo in #4505 intersects to ["text","image"]
|
|
728
|
+
* instead of ["text"] — without claiming native vision. modelInputModalities
|
|
729
|
+
* is per-key filled, so this reaches an existing install even when
|
|
730
|
+
* noVisionModels was persisted before the id joined that list.
|
|
731
|
+
*/
|
|
732
|
+
const COMMAND_CODE_TEXT_ONLY_MODELS = [
|
|
733
|
+
"deepseek/deepseek-v4.1-flash",
|
|
710
734
|
] as const;
|
|
711
|
-
const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text", "image"]> =
|
|
712
|
-
Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"]]))
|
|
735
|
+
const COMMAND_CODE_MODEL_INPUT_MODALITIES: Record<string, ["text"] | ["text", "image"]> = {
|
|
736
|
+
...Object.fromEntries(COMMAND_CODE_IMAGE_MODELS.map(id => [id, ["text", "image"] as ["text", "image"]])),
|
|
737
|
+
...Object.fromEntries(COMMAND_CODE_TEXT_ONLY_MODELS.map(id => [id, ["text"] as ["text"]])),
|
|
738
|
+
};
|
|
713
739
|
const OPENCODE_FREE_DEEPSEEK_MODELS = ["deepseek-v4-flash-free"];
|
|
714
740
|
/*
|
|
715
741
|
* Zen free models that reject `image_url` upstream (#1043, and the reproducible
|
|
@@ -722,10 +748,9 @@ const OPENCODE_FREE_DEEPSEEK_MODELS = ["deepseek-v4-flash-free"];
|
|
|
722
748
|
* `[404] No endpoints found that support image input` and `big-pickle` with the
|
|
723
749
|
* exact deserialize error quoted in #1043.
|
|
724
750
|
*
|
|
725
|
-
* `mimo-v2.5-free` and `longcat-2.0-free` ACCEPT images
|
|
726
|
-
*
|
|
727
|
-
*
|
|
728
|
-
* assertion in tests/providers/provider-registry-parity.test.ts.
|
|
751
|
+
* `mimo-v2.5-free` and `longcat-2.0-free` ACCEPT images. They remain absent
|
|
752
|
+
* from the blind list and are recorded separately as positive input-modality evidence,
|
|
753
|
+
* so capability-positive dispatch can forward images without relying on blacklist absence.
|
|
729
754
|
*
|
|
730
755
|
* Zen's roster is discovered live while this list is static, so it is a dated
|
|
731
756
|
* exception list, not a capability model. Re-probe before extending it.
|
|
@@ -739,6 +764,7 @@ const OPENCODE_ZEN_TEXT_ONLY_MODELS = [
|
|
|
739
764
|
"laguna-s-2.1-free",
|
|
740
765
|
"deepseek-v4-flash-free",
|
|
741
766
|
];
|
|
767
|
+
const OPENCODE_ZEN_IMAGE_MODELS = ["mimo-v2.5-free", "longcat-2.0-free"] as const;
|
|
742
768
|
/*
|
|
743
769
|
* DeepSeek's Codex ladder is low/high/max. With the V4 Pro GA release
|
|
744
770
|
* (DeepSeek-V4-Pro-0813) the official thinking-mode table is IDENTICAL for both
|
|
@@ -1856,8 +1882,27 @@ export const PROVIDER_REGISTRY: readonly ProviderRegistryEntry[] = [
|
|
|
1856
1882
|
},
|
|
1857
1883
|
modelInputModalities: {
|
|
1858
1884
|
"kimi-k3": ["text", "image"],
|
|
1885
|
+
// glm-5.3-flash is a native VLM (docs.z.ai/guides/vlm/glm-5.3-flash). It is
|
|
1886
|
+
// deliberately absent from this preset's noVisionModels, which is the
|
|
1887
|
+
// correct NEGATIVE half, but with no positive modelInputModalities entry
|
|
1888
|
+
// configuredInputModalities returns undefined and the catalog falls through
|
|
1889
|
+
// to the ["text"] floor. The same model is already declared ["text","image"]
|
|
1890
|
+
// on the zai and zhipu-bigmodel-coding presets, so the registry described
|
|
1891
|
+
// one model two ways (#4505).
|
|
1892
|
+
"glm-5.3-flash": ["text", "image"],
|
|
1859
1893
|
// Experimental DeepSeek vision preview — expected to merge into deepseek-v4-flash later.
|
|
1860
1894
|
[DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
|
|
1895
|
+
// This route is text-only upstream — it is already listed in this preset's
|
|
1896
|
+
// noVisionModels, which routes images through the proxy's vision sidecar and
|
|
1897
|
+
// makes the catalog advertise image input on its behalf. The positive
|
|
1898
|
+
// text-only declaration is what reaches an EXISTING install: derive.ts fills
|
|
1899
|
+
// noVisionModels all-or-nothing, so a config persisted before this id joined
|
|
1900
|
+
// the list keeps a stale list, the sidecar predicate never matches, the row
|
|
1901
|
+
// carries no modality at all, and any combo containing it collapses to
|
|
1902
|
+
// ["text"] (#4505). modelInputModalities IS per-key filled, so this
|
|
1903
|
+
// declaration lands on old configs. It states the route's real upstream
|
|
1904
|
+
// capability and keeps the sidecar explicitly distinct from native vision.
|
|
1905
|
+
"deepseek-v4.1-flash": ["text"],
|
|
1861
1906
|
// Muse Spark Contributor is natively multimodal on Zen Go: it accepts input_image
|
|
1862
1907
|
// parts over /responses (probed 2026-08-26). Without this declaration the catalog
|
|
1863
1908
|
// advertises it text-only and the Codex app blocks image attachments client-side with
|
|
@@ -3209,6 +3254,7 @@ export const PROVIDER_REGISTRY: readonly ProviderRegistryEntry[] = [
|
|
|
3209
3254
|
},
|
|
3210
3255
|
modelInputModalities: {
|
|
3211
3256
|
[DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
|
|
3257
|
+
...Object.fromEntries(OPENCODE_ZEN_IMAGE_MODELS.map(id => [id, ["text", "image"] as string[]])),
|
|
3212
3258
|
},
|
|
3213
3259
|
noVisionModels: [...OPENCODE_ZEN_TEXT_ONLY_MODELS, ...DEEPSEEK_GATEWAY_THINKING_MODELS],
|
|
3214
3260
|
// Same DeepSeek routes as the Go preset above, behind the same vendor, so they carry
|
|
@@ -3250,6 +3296,7 @@ export const PROVIDER_REGISTRY: readonly ProviderRegistryEntry[] = [
|
|
|
3250
3296
|
},
|
|
3251
3297
|
modelInputModalities: {
|
|
3252
3298
|
[DEEPSEEK_VISION_PREVIEW_MODEL]: ["text", "image"],
|
|
3299
|
+
...Object.fromEntries(OPENCODE_ZEN_IMAGE_MODELS.map(id => [id, ["text", "image"] as string[]])),
|
|
3253
3300
|
},
|
|
3254
3301
|
// Same Zen roster behind the same base URL, so it carries the same measured
|
|
3255
3302
|
// text-only list rather than only its DeepSeek member (#1043).
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
/** Input kinds for which the normalized request has no lossless content carrier. */
|
|
2
|
+
export type UntranslatedInputMedia = "audio" | "file";
|
|
3
|
+
|
|
4
|
+
type RecordValue = Record<string, unknown>;
|
|
5
|
+
|
|
6
|
+
function isRecord(value: unknown): value is RecordValue {
|
|
7
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
function mediaKind(value: unknown): UntranslatedInputMedia | undefined {
|
|
11
|
+
if (!isRecord(value)) return undefined;
|
|
12
|
+
if (value.type === "input_audio" || value.type === "audio") return "audio";
|
|
13
|
+
if (value.type === "input_file" || value.type === "file" || value.type === "document") return "file";
|
|
14
|
+
// A file-id-only image is not pixels: translated adapters cannot dereference it.
|
|
15
|
+
if (value.type === "input_image" && typeof value.file_id === "string" && value.file_id.length > 0
|
|
16
|
+
&& !(typeof value.image_url === "string" && value.image_url.length > 0)) return "file";
|
|
17
|
+
return undefined;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
function contentMedia(content: unknown): UntranslatedInputMedia | undefined {
|
|
21
|
+
if (!Array.isArray(content)) return undefined;
|
|
22
|
+
for (const part of content) {
|
|
23
|
+
const kind = mediaKind(part);
|
|
24
|
+
if (kind) return kind;
|
|
25
|
+
}
|
|
26
|
+
return undefined;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Inspect only typed input items and their content arrays, never strings, tool
|
|
31
|
+
* arguments, schema properties, or arbitrary nested objects. No payload is copied,
|
|
32
|
+
* decoded, fetched or included in the returned value.
|
|
33
|
+
*/
|
|
34
|
+
export function untranslatedResponsesInputMedia(body: unknown): UntranslatedInputMedia | undefined {
|
|
35
|
+
if (!isRecord(body) || !Array.isArray(body.input)) return undefined;
|
|
36
|
+
for (const item of body.input) {
|
|
37
|
+
if (!isRecord(item)) continue;
|
|
38
|
+
const direct = mediaKind(item);
|
|
39
|
+
if (direct) return direct;
|
|
40
|
+
if (item.type === "function_call_output" || item.type === "custom_tool_call_output") {
|
|
41
|
+
const kind = contentMedia(item.output);
|
|
42
|
+
if (kind) return kind;
|
|
43
|
+
} else if (item.type === "message" || item.type === undefined) {
|
|
44
|
+
const kind = contentMedia(item.content);
|
|
45
|
+
if (kind) return kind;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
return undefined;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Used only when Chat is actually projected, not on the native Chat fast path. */
|
|
52
|
+
export function untranslatedChatInputMedia(body: unknown): UntranslatedInputMedia | undefined {
|
|
53
|
+
if (!isRecord(body) || !Array.isArray(body.messages)) return undefined;
|
|
54
|
+
for (const message of body.messages) {
|
|
55
|
+
if (!isRecord(message)) continue;
|
|
56
|
+
const kind = contentMedia(message.content);
|
|
57
|
+
if (kind) return kind;
|
|
58
|
+
}
|
|
59
|
+
return undefined;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Fixed vocabulary only: never interpolate filenames, URLs or client metadata. */
|
|
63
|
+
export function untranslatedInputMediaMessage(kind: UntranslatedInputMedia): string {
|
|
64
|
+
return `OpenCodex cannot translate ${kind} input on this route. Use a native input wire that supports the attachment, or convert it to text first.`;
|
|
65
|
+
}
|
|
@@ -9,6 +9,8 @@ type InputBlock =
|
|
|
9
9
|
| { type: "text"; text: string }
|
|
10
10
|
| { type: "input_image"; image_url?: string; file_id?: string; detail?: string }
|
|
11
11
|
| { type: "input_video"; video_url?: string }
|
|
12
|
+
// codex-rs protocol/src/models.rs sends audio as input_audio with an audio_url.
|
|
13
|
+
| { type: "input_audio"; audio_url?: string; format?: string }
|
|
12
14
|
| { type: "input_file"; file_id?: string; filename?: string; file_data?: string };
|
|
13
15
|
|
|
14
16
|
/** A usable reference string, or undefined. Empty strings and non-strings are not references. */
|
|
@@ -16,6 +18,19 @@ function nonEmptyString(value: unknown): string | undefined {
|
|
|
16
18
|
return typeof value === "string" && value.length > 0 ? value : undefined;
|
|
17
19
|
}
|
|
18
20
|
|
|
21
|
+
/**
|
|
22
|
+
* An audio format label safe to render into model-visible prose.
|
|
23
|
+
*
|
|
24
|
+
* `format` is caller-controlled and unbounded in the schema, so interpolating it
|
|
25
|
+
* verbatim would let a request park newlines, injected instructions, or a signed URL
|
|
26
|
+
* inside text the model reads as trusted proxy output. Only a short alphanumeric
|
|
27
|
+
* token is echoed; anything else degrades to the bare marker.
|
|
28
|
+
*/
|
|
29
|
+
function safeAudioFormat(value: unknown): string | undefined {
|
|
30
|
+
const raw = nonEmptyString(value);
|
|
31
|
+
return raw !== undefined && /^[a-z0-9]{1,12}$/i.test(raw) ? raw : undefined;
|
|
32
|
+
}
|
|
33
|
+
|
|
19
34
|
export function inputContentParts(blocks: unknown): string | OcxContentPart[] {
|
|
20
35
|
if (typeof blocks === "string") return blocks;
|
|
21
36
|
// The catch-all can also hand back a non-array `content` (an object, a number), which would
|
|
@@ -47,6 +62,26 @@ export function inputContentParts(blocks: unknown): string | OcxContentPart[] {
|
|
|
47
62
|
} else if (block.type === "input_video") {
|
|
48
63
|
const videoUrl = nonEmptyString(block.video_url);
|
|
49
64
|
if (videoUrl) parts.push({ type: "video", videoUrl });
|
|
65
|
+
} else if (block.type === "input_audio") {
|
|
66
|
+
// Upstream Codex sends input_audio with an audio_url (codex-rs
|
|
67
|
+
// protocol/src/models.rs). The IR has no audio carrier and no adapter consumes
|
|
68
|
+
// one, so this part used to vanish with no trace at all.
|
|
69
|
+
//
|
|
70
|
+
// This records PRESENCE only and is NOT audio support: never the payload, which
|
|
71
|
+
// is large base64 and would explode the token count, and never the URL, which
|
|
72
|
+
// can carry a signed token. This parser must stay non-throwing — the native
|
|
73
|
+
// Responses passthrough also runs through parseRequest before the adapter
|
|
74
|
+
// forwards _rawBody, so refusing here would regress legitimate raw passthrough.
|
|
75
|
+
//
|
|
76
|
+
// No adapter refuses audio at its wire today: by this point the part is already a
|
|
77
|
+
// text marker, so downstream adapters see text and continue. A typed unsupported-
|
|
78
|
+
// modality carrier that survives to final adapter dispatch is a separate, recorded
|
|
79
|
+
// residual — do not describe this branch as a refusal.
|
|
80
|
+
const b = block as { audio_url?: string; format?: string };
|
|
81
|
+
const format = safeAudioFormat(b.format);
|
|
82
|
+
if (nonEmptyString(b.audio_url)) {
|
|
83
|
+
parts.push({ type: "text", text: format ? `[audio: ${format}]` : "[audio]" });
|
|
84
|
+
}
|
|
50
85
|
} else if (block.type === "input_file") {
|
|
51
86
|
const b = block as { file_id?: string; filename?: string; file_data?: string };
|
|
52
87
|
const fileId = nonEmptyString(b.file_id);
|
|
@@ -111,6 +146,13 @@ export function outputToToolResultContent(output: string | unknown[] | undefined
|
|
|
111
146
|
} else if (fileId) {
|
|
112
147
|
parts.push({ type: "text", text: `[image: ${fileId}]` });
|
|
113
148
|
}
|
|
149
|
+
} else if (raw.type === "input_audio") {
|
|
150
|
+
// Same presence-only contract as the user-content branch above: Codex returns
|
|
151
|
+
// audio in tool output too, and it previously disappeared without trace.
|
|
152
|
+
const format = safeAudioFormat(raw.format);
|
|
153
|
+
if (nonEmptyString(raw.audio_url)) {
|
|
154
|
+
parts.push({ type: "text", text: format ? `[audio: ${format}]` : "[audio]" });
|
|
155
|
+
}
|
|
114
156
|
} else if (raw.type === "encrypted_content") {
|
|
115
157
|
// codex-rs FunctionCallOutputContentItem::EncryptedContent — opaque to routed models.
|
|
116
158
|
parts.push({ type: "text", text: "[encrypted content omitted]" });
|
package/src/responses/schema.ts
CHANGED
|
@@ -21,6 +21,16 @@ const inputFileBlockSchema = z.object({
|
|
|
21
21
|
filename: z.string().optional(),
|
|
22
22
|
file_data: z.string().optional(),
|
|
23
23
|
});
|
|
24
|
+
// codex-rs protocol/src/models.rs sends audio as input_audio with an audio_url, in
|
|
25
|
+
// both user content and tool output. Accepting the block keeps a legitimate audio turn
|
|
26
|
+
// out of the malformed-item catch-all. The translated IR records only its PRESENCE —
|
|
27
|
+
// there is no audio carrier and no adapter-level refusal; a typed unsupported-modality
|
|
28
|
+
// signal reaching final adapter dispatch remains a recorded residual.
|
|
29
|
+
const inputAudioBlockSchema = z.object({
|
|
30
|
+
type: z.literal("input_audio"),
|
|
31
|
+
audio_url: z.string().min(1),
|
|
32
|
+
format: z.string().optional(),
|
|
33
|
+
});
|
|
24
34
|
const outputTextSchema = z.object({ type: z.literal("output_text"), text: z.string() });
|
|
25
35
|
const outputRefusalSchema = z.object({ type: z.literal("refusal"), refusal: z.string() });
|
|
26
36
|
const summaryTextSchema = z.object({ type: z.literal("summary_text"), text: z.string() });
|
|
@@ -28,12 +38,12 @@ const reasoningTextSchema = z.object({ type: z.literal("reasoning_text"), text:
|
|
|
28
38
|
// codex-rs FunctionCallOutputContentItem (protocol/src/models.rs): input_text | input_image | encrypted_content.
|
|
29
39
|
const encryptedContentBlockSchema = z.object({ type: z.literal("encrypted_content"), encrypted_content: z.string() });
|
|
30
40
|
|
|
31
|
-
const inputContentBlockSchema = z.union([inputTextSchema, plainTextSchema, inputImageBlockSchema, inputVideoBlockSchema, inputFileBlockSchema]);
|
|
41
|
+
const inputContentBlockSchema = z.union([inputTextSchema, plainTextSchema, inputImageBlockSchema, inputVideoBlockSchema, inputAudioBlockSchema, inputFileBlockSchema]);
|
|
32
42
|
const outputContentBlockSchema = z.union([outputTextSchema, plainTextSchema, outputRefusalSchema]);
|
|
33
43
|
// Tool outputs on the wire mix codex-rs FunctionCallOutputContentItem with legacy output blocks.
|
|
34
44
|
const toolOutputContentBlockSchema = z.union([
|
|
35
45
|
outputTextSchema, plainTextSchema, outputRefusalSchema,
|
|
36
|
-
inputTextSchema, inputImageBlockSchema, encryptedContentBlockSchema,
|
|
46
|
+
inputTextSchema, inputImageBlockSchema, inputAudioBlockSchema, encryptedContentBlockSchema,
|
|
37
47
|
]);
|
|
38
48
|
const toolOutputSchema = z.union([z.string(), z.array(toolOutputContentBlockSchema)]);
|
|
39
49
|
|
package/src/server/audio-live.ts
CHANGED
|
@@ -100,7 +100,7 @@ export async function handleExternalLive(
|
|
|
100
100
|
? frameless ? forwardLiveUrl(relay.providerBaseUrl, false) : keyedLiveUrl(relay.providerBaseUrl)
|
|
101
101
|
: forwardLiveUrl(relay.providerBaseUrl, true);
|
|
102
102
|
const upstream = await fetch(url, { method: "POST", headers, body, signal: deadline.signal, redirect: "manual" });
|
|
103
|
-
outcome = upstream.
|
|
103
|
+
outcome = upstream.status;
|
|
104
104
|
const detach = cancelBodyOnAbort(upstream.body, deadline.signal);
|
|
105
105
|
let responseBody: ArrayBuffer | Response;
|
|
106
106
|
try { responseBody = await readBodyCapped(upstream.body, LIVE_RESPONSE_MAX_BYTES, () => "Live answer too large", deadline.signal); }
|
|
@@ -122,7 +122,6 @@ export async function handleExternalLive(
|
|
|
122
122
|
sidebandBaseUrl: config.experimentalRealtimeWsBaseUrl,
|
|
123
123
|
});
|
|
124
124
|
if (!alias) return formatErrorResponse(503, "server_busy", "Live call could not be registered");
|
|
125
|
-
outcome = upstream.status;
|
|
126
125
|
return new Response(responseBody, { status: upstream.status, headers: {
|
|
127
126
|
"content-type": upstream.headers.get("content-type") ?? "application/sdp",
|
|
128
127
|
location: `/v1/${frameless ? "live" : "realtime/calls"}/${alias}`,
|
|
@@ -117,7 +117,7 @@ async function transcribeAdmitted(
|
|
|
117
117
|
form.append("response_format", "json");
|
|
118
118
|
}
|
|
119
119
|
const upstream = await fetch(url, { method: "POST", headers, body: form, signal: signal.signal, redirect: "manual" });
|
|
120
|
-
outcome = upstream.
|
|
120
|
+
outcome = upstream.status;
|
|
121
121
|
const detach = cancelBodyOnAbort(upstream.body, signal.signal);
|
|
122
122
|
let body: ArrayBuffer | Response;
|
|
123
123
|
try {
|
|
@@ -137,7 +137,6 @@ async function transcribeAdmitted(
|
|
|
137
137
|
if (!payload || typeof payload !== "object" || !("text" in payload) || typeof payload.text !== "string") {
|
|
138
138
|
return formatErrorResponse(502, "upstream_error", "Audio upstream response is missing text");
|
|
139
139
|
}
|
|
140
|
-
outcome = upstream.status;
|
|
141
140
|
return input.format === "text"
|
|
142
141
|
? new Response(payload.text, { headers: { "content-type": "text/plain; charset=utf-8" } })
|
|
143
142
|
: Response.json({ text: payload.text });
|
package/src/server/auth-cors.ts
CHANGED
|
@@ -766,7 +766,7 @@ export function providerManagementConfigError(
|
|
|
766
766
|
if (requestPacingError) {
|
|
767
767
|
return `provider ${JSON.stringify(redactSecretString(name))} ${requestPacingError}`;
|
|
768
768
|
}
|
|
769
|
-
const webSearchBridgeError = providerWebSearchBridgeConfigError(raw.webSearchBridge);
|
|
769
|
+
const webSearchBridgeError = providerWebSearchBridgeConfigError(raw.webSearchBridge, name, typed);
|
|
770
770
|
if (webSearchBridgeError) {
|
|
771
771
|
return `provider ${JSON.stringify(redactSecretString(name))} ${webSearchBridgeError}`;
|
|
772
772
|
}
|
|
@@ -11,6 +11,11 @@ import {
|
|
|
11
11
|
stopStorageCleanupScheduler,
|
|
12
12
|
} from "../storage/policy-scheduler";
|
|
13
13
|
import { startQuotaResetPoller, stopQuotaResetPoller } from "../quota/reset-poller";
|
|
14
|
+
import {
|
|
15
|
+
startCatalogAutoRefresh,
|
|
16
|
+
stopCatalogAutoRefresh,
|
|
17
|
+
syncCatalogAutoRefreshCadence,
|
|
18
|
+
} from "../codex/catalog-auto-refresh";
|
|
14
19
|
import {
|
|
15
20
|
cancelQueuedStorageWorkerSpawns,
|
|
16
21
|
drainStorageWorkers,
|
|
@@ -70,6 +75,17 @@ function startProcessLoops(applyPolicy: PolicyApply): ProcessLoops {
|
|
|
70
75
|
.catch(() => {
|
|
71
76
|
// The next tick adopts it.
|
|
72
77
|
});
|
|
78
|
+
// Opt-in: the tick is a no-op unless catalogAutoRefresh.enabled is true, and the
|
|
79
|
+
// interval is unref'd, so a default install pays one dormant timer. The scheduler
|
|
80
|
+
// module keeps every heavy import inside its tick, so naming it statically here
|
|
81
|
+
// costs a module record and nothing else.
|
|
82
|
+
startCatalogAutoRefresh();
|
|
83
|
+
// The scheduler starts at its default cadence because resolving the operator's value
|
|
84
|
+
// reads the config barrel. Fire-and-forget: startup must not await an optional
|
|
85
|
+
// subsystem, and the next tick adopts the cadence anyway.
|
|
86
|
+
void syncCatalogAutoRefreshCadence().catch(() => {
|
|
87
|
+
// The next tick adopts it.
|
|
88
|
+
});
|
|
73
89
|
// Install the delivery sink now rather than waiting out the first poll interval, which is 15
|
|
74
90
|
// minutes by default. Without this, an enabled install would observe nothing for its first
|
|
75
91
|
// quarter hour — including the live request path, which is gated on the sink existing.
|
|
@@ -85,6 +101,7 @@ function startProcessLoops(applyPolicy: PolicyApply): ProcessLoops {
|
|
|
85
101
|
stateStoreSweeper?.stop();
|
|
86
102
|
stopStorageCleanupScheduler();
|
|
87
103
|
stopQuotaResetPoller();
|
|
104
|
+
stopCatalogAutoRefresh();
|
|
88
105
|
setLivePolicyOwner(null);
|
|
89
106
|
throw error;
|
|
90
107
|
}
|
|
@@ -97,6 +114,7 @@ function stopProcessLoops(): void {
|
|
|
97
114
|
loops?.stateStoreSweeper.stop();
|
|
98
115
|
stopStorageCleanupScheduler();
|
|
99
116
|
stopQuotaResetPoller();
|
|
117
|
+
stopCatalogAutoRefresh();
|
|
100
118
|
setLivePolicyOwner(null);
|
|
101
119
|
}
|
|
102
120
|
|
|
@@ -11,6 +11,7 @@ import {
|
|
|
11
11
|
ChatCompletionsRequestError,
|
|
12
12
|
chatCompletionsToResponsesBody,
|
|
13
13
|
} from "../chat/inbound";
|
|
14
|
+
import { normalizeChatImageParts } from "../chat/image-parts";
|
|
14
15
|
import {
|
|
15
16
|
chatCompletionsErrorResponse,
|
|
16
17
|
collectChatCompletion,
|
|
@@ -111,7 +112,11 @@ async function handleChatCompletionsWithBudget(
|
|
|
111
112
|
try {
|
|
112
113
|
const rawBody = await readChatBody(req, translatorBudget, resolveInboundBodyLimitBytes(config.maxInboundBodyBytes));
|
|
113
114
|
assertChatCompletionsRoutingBody(rawBody);
|
|
114
|
-
|
|
115
|
+
// Normalize foreign image shapes BEFORE routing. isNativeChatRouteEligible below
|
|
116
|
+
// decides the pipeline from the image parts it can see, and the native path then
|
|
117
|
+
// forwards this body as-is, so both must observe the same parts. A body with no
|
|
118
|
+
// foreign image part is returned by reference and stays byte-identical.
|
|
119
|
+
chatBody = normalizeChatImageParts(rawBody);
|
|
115
120
|
} catch (err) {
|
|
116
121
|
const overflow = isTranslatorBudgetExceededError(err);
|
|
117
122
|
const status = overflow ? 413 : err instanceof ChatCompletionsRequestError ? 400 : 500;
|
|
@@ -165,7 +170,7 @@ async function handleChatCompletionsWithBudget(
|
|
|
165
170
|
if (chatBody.tools !== undefined) parts.push(JSON.stringify(chatBody.tools));
|
|
166
171
|
logCtx.usageLogInputTokens = Math.max(1, estimateTokens(parts.join("\n"), requestedModel));
|
|
167
172
|
}
|
|
168
|
-
if (!effortRow && isNativeChatRouteEligible(route, chatBody)) chatNativeRoute = route;
|
|
173
|
+
if (!effortRow && isNativeChatRouteEligible(route, chatBody, config)) chatNativeRoute = route;
|
|
169
174
|
} catch (err) {
|
|
170
175
|
if (err instanceof UnknownRoutingPolicyError) {
|
|
171
176
|
logCtx.requestedModel = requestedModel;
|
|
@@ -221,13 +226,23 @@ async function handleChatCompletionsWithBudget(
|
|
|
221
226
|
// for non-streaming clients. Native Chat uses the caller's original stream bit.
|
|
222
227
|
internalBody.stream = true;
|
|
223
228
|
if (settledRoute?.provider.adapter === "openai-responses") {
|
|
224
|
-
//
|
|
229
|
+
// The proxy never wants upstream-side retention for a translated Chat turn, so
|
|
230
|
+
// store stays pinned for every Responses route.
|
|
231
|
+
//
|
|
232
|
+
// The sampling and output-cap restrictions used to be applied here too, keyed on
|
|
233
|
+
// the adapter string. That was wrong twice over. Seven providers share this
|
|
234
|
+
// adapter (openai, openai-apikey, meta-model, meta-muse, zai,
|
|
235
|
+
// zhipu-bigmodel-responses, volcengine-agent-plan), so a generic key gateway lost
|
|
236
|
+
// controls it accepts. And settledRoute is the route settled at INGRESS: a combo
|
|
237
|
+
// or policy route resolves its concrete child later in the Responses pipeline, so
|
|
238
|
+
// deciding here mutates shared intent before the real target is known — a
|
|
239
|
+
// canonical-first combo that falls back to a key gateway had already lost the
|
|
240
|
+
// caller's controls, while a non-canonical-first combo that falls back to
|
|
241
|
+
// canonical still shipped them.
|
|
242
|
+
//
|
|
243
|
+
// Canonical-backend sanitization now happens at the final outgoing body in
|
|
244
|
+
// src/adapters/openai-responses.ts, where the concrete provider is known.
|
|
225
245
|
internalBody.store = false;
|
|
226
|
-
delete internalBody.max_output_tokens;
|
|
227
|
-
delete internalBody.temperature;
|
|
228
|
-
delete internalBody.top_p;
|
|
229
|
-
delete internalBody.stop;
|
|
230
|
-
delete internalBody.user;
|
|
231
246
|
} else if (internalBody.store === undefined) {
|
|
232
247
|
internalBody.store = false;
|
|
233
248
|
}
|