@sayknow-cli/ai 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2788 -0
- package/README.md +1183 -0
- package/dist/types/api-registry.d.ts +30 -0
- package/dist/types/auth-broker/client.d.ts +66 -0
- package/dist/types/auth-broker/index.d.ts +5 -0
- package/dist/types/auth-broker/refresher.d.ts +25 -0
- package/dist/types/auth-broker/remote-store.d.ts +96 -0
- package/dist/types/auth-broker/server.d.ts +32 -0
- package/dist/types/auth-broker/types.d.ts +105 -0
- package/dist/types/auth-broker/wire-schemas.d.ts +412 -0
- package/dist/types/auth-gateway/http.d.ts +39 -0
- package/dist/types/auth-gateway/index.d.ts +3 -0
- package/dist/types/auth-gateway/server.d.ts +17 -0
- package/dist/types/auth-gateway/types.d.ts +115 -0
- package/dist/types/auth-storage.d.ts +660 -0
- package/dist/types/cli.d.ts +2 -0
- package/dist/types/index.d.ts +51 -0
- package/dist/types/model-cache.d.ts +17 -0
- package/dist/types/model-manager.d.ts +62 -0
- package/dist/types/model-thinking.d.ts +74 -0
- package/dist/types/models.d.ts +12 -0
- package/dist/types/provider-details.d.ts +24 -0
- package/dist/types/provider-models/bundled-references.d.ts +4 -0
- package/dist/types/provider-models/descriptors.d.ts +48 -0
- package/dist/types/provider-models/google.d.ts +20 -0
- package/dist/types/provider-models/index.d.ts +5 -0
- package/dist/types/provider-models/ollama.d.ts +7 -0
- package/dist/types/provider-models/openai-compat.d.ts +244 -0
- package/dist/types/provider-models/special.d.ts +16 -0
- package/dist/types/providers/amazon-bedrock.d.ts +60 -0
- package/dist/types/providers/anthropic-messages-server-schema.d.ts +450 -0
- package/dist/types/providers/anthropic-messages-server.d.ts +17 -0
- package/dist/types/providers/anthropic.d.ts +198 -0
- package/dist/types/providers/aws-credentials.d.ts +43 -0
- package/dist/types/providers/aws-eventstream.d.ts +38 -0
- package/dist/types/providers/aws-sigv4.d.ts +55 -0
- package/dist/types/providers/azure-openai-responses.d.ts +15 -0
- package/dist/types/providers/composer-discipline.d.ts +26 -0
- package/dist/types/providers/cursor/gen/agent_pb.d.ts +13022 -0
- package/dist/types/providers/cursor.d.ts +44 -0
- package/dist/types/providers/error-message.d.ts +27 -0
- package/dist/types/providers/github-copilot-headers.d.ts +40 -0
- package/dist/types/providers/gitlab-duo.d.ts +27 -0
- package/dist/types/providers/google-auth.d.ts +24 -0
- package/dist/types/providers/google-gemini-cli.d.ts +72 -0
- package/dist/types/providers/google-gemini-headers.d.ts +18 -0
- package/dist/types/providers/google-shared.d.ts +173 -0
- package/dist/types/providers/google-types.d.ts +138 -0
- package/dist/types/providers/google-vertex.d.ts +7 -0
- package/dist/types/providers/google.d.ts +4 -0
- package/dist/types/providers/grammar.d.ts +1 -0
- package/dist/types/providers/kimi.d.ts +27 -0
- package/dist/types/providers/mock.d.ts +175 -0
- package/dist/types/providers/ollama.d.ts +41 -0
- package/dist/types/providers/openai-anthropic-shim.d.ts +31 -0
- package/dist/types/providers/openai-chat-server-schema.d.ts +815 -0
- package/dist/types/providers/openai-chat-server.d.ts +16 -0
- package/dist/types/providers/openai-codex/constants.d.ts +26 -0
- package/dist/types/providers/openai-codex/request-transformer.d.ts +49 -0
- package/dist/types/providers/openai-codex/response-handler.d.ts +17 -0
- package/dist/types/providers/openai-codex-responses.d.ts +67 -0
- package/dist/types/providers/openai-completions-compat.d.ts +27 -0
- package/dist/types/providers/openai-completions.d.ts +33 -0
- package/dist/types/providers/openai-request-transform.d.ts +4 -0
- package/dist/types/providers/openai-responses-server-schema.d.ts +392 -0
- package/dist/types/providers/openai-responses-server.d.ts +17 -0
- package/dist/types/providers/openai-responses-shared.d.ts +89 -0
- package/dist/types/providers/openai-responses.d.ts +32 -0
- package/dist/types/providers/pi-native-client.d.ts +13 -0
- package/dist/types/providers/pi-native-server.d.ts +68 -0
- package/dist/types/providers/register-builtins.d.ts +31 -0
- package/dist/types/providers/synthetic.d.ts +26 -0
- package/dist/types/providers/transform-messages.d.ts +14 -0
- package/dist/types/providers/vision-guard.d.ts +8 -0
- package/dist/types/rate-limit-utils.d.ts +19 -0
- package/dist/types/stream.d.ts +43 -0
- package/dist/types/types.d.ts +811 -0
- package/dist/types/usage/claude.d.ts +3 -0
- package/dist/types/usage/gemini.d.ts +2 -0
- package/dist/types/usage/github-copilot.d.ts +7 -0
- package/dist/types/usage/google-antigravity.d.ts +2 -0
- package/dist/types/usage/grok-cli.d.ts +10 -0
- package/dist/types/usage/kimi.d.ts +2 -0
- package/dist/types/usage/minimax-code.d.ts +2 -0
- package/dist/types/usage/openai-codex.d.ts +3 -0
- package/dist/types/usage/shared.d.ts +1 -0
- package/dist/types/usage/zai.d.ts +2 -0
- package/dist/types/usage.d.ts +258 -0
- package/dist/types/utils/abort.d.ts +19 -0
- package/dist/types/utils/anthropic-auth.d.ts +31 -0
- package/dist/types/utils/discovery/antigravity.d.ts +61 -0
- package/dist/types/utils/discovery/codex.d.ts +38 -0
- package/dist/types/utils/discovery/cursor.d.ts +23 -0
- package/dist/types/utils/discovery/gemini.d.ts +25 -0
- package/dist/types/utils/discovery/index.d.ts +4 -0
- package/dist/types/utils/discovery/openai-compatible.d.ts +74 -0
- package/dist/types/utils/event-stream.d.ts +33 -0
- package/dist/types/utils/fireworks-model-id.d.ts +10 -0
- package/dist/types/utils/foundry.d.ts +1 -0
- package/dist/types/utils/h2-fetch.d.ts +22 -0
- package/dist/types/utils/http-inspector.d.ts +35 -0
- package/dist/types/utils/idle-iterator.d.ts +67 -0
- package/dist/types/utils/json-parse.d.ts +10 -0
- package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +18 -0
- package/dist/types/utils/oauth/anthropic.d.ts +22 -0
- package/dist/types/utils/oauth/api-key-login.d.ts +35 -0
- package/dist/types/utils/oauth/api-key-validation.d.ts +27 -0
- package/dist/types/utils/oauth/callback-server.d.ts +60 -0
- package/dist/types/utils/oauth/cerebras.d.ts +1 -0
- package/dist/types/utils/oauth/cloudflare-ai-gateway.d.ts +18 -0
- package/dist/types/utils/oauth/cursor.d.ts +15 -0
- package/dist/types/utils/oauth/deepseek.d.ts +10 -0
- package/dist/types/utils/oauth/firepass.d.ts +1 -0
- package/dist/types/utils/oauth/fireworks.d.ts +1 -0
- package/dist/types/utils/oauth/github-copilot.d.ts +38 -0
- package/dist/types/utils/oauth/gitlab-duo.d.ts +3 -0
- package/dist/types/utils/oauth/google-antigravity.d.ts +11 -0
- package/dist/types/utils/oauth/google-gemini-cli.d.ts +10 -0
- package/dist/types/utils/oauth/google-oauth-shared.d.ts +28 -0
- package/dist/types/utils/oauth/huggingface.d.ts +19 -0
- package/dist/types/utils/oauth/index.d.ts +38 -0
- package/dist/types/utils/oauth/kagi.d.ts +17 -0
- package/dist/types/utils/oauth/kilo.d.ts +5 -0
- package/dist/types/utils/oauth/kimi.d.ts +21 -0
- package/dist/types/utils/oauth/litellm.d.ts +18 -0
- package/dist/types/utils/oauth/lm-studio.d.ts +17 -0
- package/dist/types/utils/oauth/minimax-code.d.ts +28 -0
- package/dist/types/utils/oauth/moonshot.d.ts +1 -0
- package/dist/types/utils/oauth/nanogpt.d.ts +1 -0
- package/dist/types/utils/oauth/nvidia.d.ts +18 -0
- package/dist/types/utils/oauth/ollama-cloud.d.ts +2 -0
- package/dist/types/utils/oauth/ollama.d.ts +18 -0
- package/dist/types/utils/oauth/openai-codex.d.ts +21 -0
- package/dist/types/utils/oauth/opencode.d.ts +18 -0
- package/dist/types/utils/oauth/parallel.d.ts +17 -0
- package/dist/types/utils/oauth/perplexity.d.ts +9 -0
- package/dist/types/utils/oauth/pkce.d.ts +8 -0
- package/dist/types/utils/oauth/qianfan.d.ts +17 -0
- package/dist/types/utils/oauth/qwen-portal.d.ts +19 -0
- package/dist/types/utils/oauth/synthetic.d.ts +1 -0
- package/dist/types/utils/oauth/tavily.d.ts +17 -0
- package/dist/types/utils/oauth/together.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +45 -0
- package/dist/types/utils/oauth/venice.d.ts +18 -0
- package/dist/types/utils/oauth/vercel-ai-gateway.d.ts +18 -0
- package/dist/types/utils/oauth/vllm.d.ts +16 -0
- package/dist/types/utils/oauth/xai.d.ts +30 -0
- package/dist/types/utils/oauth/xiaomi.d.ts +25 -0
- package/dist/types/utils/oauth/zai.d.ts +18 -0
- package/dist/types/utils/oauth/zenmux.d.ts +1 -0
- package/dist/types/utils/overflow.d.ts +54 -0
- package/dist/types/utils/parse-bind.d.ts +23 -0
- package/dist/types/utils/provider-response.d.ts +3 -0
- package/dist/types/utils/retry-after.d.ts +3 -0
- package/dist/types/utils/retry-budget.d.ts +1 -0
- package/dist/types/utils/retry.d.ts +26 -0
- package/dist/types/utils/schema/adapt.d.ts +24 -0
- package/dist/types/utils/schema/compatibility.d.ts +30 -0
- package/dist/types/utils/schema/dereference.d.ts +11 -0
- package/dist/types/utils/schema/draft.d.ts +10 -0
- package/dist/types/utils/schema/equality.d.ts +4 -0
- package/dist/types/utils/schema/fields.d.ts +49 -0
- package/dist/types/utils/schema/index.d.ts +13 -0
- package/dist/types/utils/schema/json-schema-validator.d.ts +12 -0
- package/dist/types/utils/schema/meta-validator.d.ts +2 -0
- package/dist/types/utils/schema/normalize.d.ts +93 -0
- package/dist/types/utils/schema/spill.d.ts +8 -0
- package/dist/types/utils/schema/stamps.d.ts +25 -0
- package/dist/types/utils/schema/types.d.ts +4 -0
- package/dist/types/utils/schema/wire.d.ts +54 -0
- package/dist/types/utils/schema/zod-decontaminate.d.ts +31 -0
- package/dist/types/utils/sse-debug.d.ts +10 -0
- package/dist/types/utils/tool-call-healing.d.ts +71 -0
- package/dist/types/utils/tool-choice-capability.d.ts +41 -0
- package/dist/types/utils/tool-choice.d.ts +50 -0
- package/dist/types/utils/validation.d.ts +17 -0
- package/dist/types/utils.d.ts +34 -0
- package/package.json +146 -0
- package/src/api-registry.ts +96 -0
- package/src/auth-broker/client.ts +358 -0
- package/src/auth-broker/index.ts +5 -0
- package/src/auth-broker/refresher.ts +127 -0
- package/src/auth-broker/remote-store.ts +623 -0
- package/src/auth-broker/server.ts +644 -0
- package/src/auth-broker/types.ts +127 -0
- package/src/auth-broker/wire-schemas.ts +200 -0
- package/src/auth-gateway/http.ts +194 -0
- package/src/auth-gateway/index.ts +3 -0
- package/src/auth-gateway/server.ts +717 -0
- package/src/auth-gateway/types.ts +134 -0
- package/src/auth-storage.ts +4179 -0
- package/src/cli.ts +263 -0
- package/src/index.ts +56 -0
- package/src/model-cache.ts +129 -0
- package/src/model-manager.ts +486 -0
- package/src/model-thinking.ts +772 -0
- package/src/models.json +75437 -0
- package/src/models.json.d.ts +9 -0
- package/src/models.ts +82 -0
- package/src/prompts/turn-aborted-guidance.md +4 -0
- package/src/provider-details.ts +90 -0
- package/src/provider-models/bundled-references.ts +38 -0
- package/src/provider-models/descriptors.ts +327 -0
- package/src/provider-models/google.ts +91 -0
- package/src/provider-models/index.ts +5 -0
- package/src/provider-models/ollama.ts +153 -0
- package/src/provider-models/openai-compat.ts +2352 -0
- package/src/provider-models/special.ts +67 -0
- package/src/providers/amazon-bedrock.ts +937 -0
- package/src/providers/anthropic-messages-server-schema.ts +229 -0
- package/src/providers/anthropic-messages-server.ts +677 -0
- package/src/providers/anthropic.ts +2940 -0
- package/src/providers/aws-credentials.ts +501 -0
- package/src/providers/aws-eventstream.ts +185 -0
- package/src/providers/aws-sigv4.ts +218 -0
- package/src/providers/azure-openai-responses.ts +379 -0
- package/src/providers/composer-discipline.ts +41 -0
- package/src/providers/cursor/gen/agent_pb.ts +15274 -0
- package/src/providers/cursor/proto/agent.proto +3526 -0
- package/src/providers/cursor/proto/buf.gen.yaml +6 -0
- package/src/providers/cursor/proto/buf.yaml +17 -0
- package/src/providers/cursor.ts +2671 -0
- package/src/providers/error-message.ts +21 -0
- package/src/providers/github-copilot-headers.ts +140 -0
- package/src/providers/gitlab-duo.ts +372 -0
- package/src/providers/google-auth.ts +252 -0
- package/src/providers/google-gemini-cli.ts +856 -0
- package/src/providers/google-gemini-headers.ts +41 -0
- package/src/providers/google-shared.ts +951 -0
- package/src/providers/google-types.ts +167 -0
- package/src/providers/google-vertex.ts +88 -0
- package/src/providers/google.ts +41 -0
- package/src/providers/grammar.ts +70 -0
- package/src/providers/kimi.ts +52 -0
- package/src/providers/mock.ts +500 -0
- package/src/providers/ollama.ts +603 -0
- package/src/providers/openai-anthropic-shim.ts +138 -0
- package/src/providers/openai-chat-server-schema.ts +243 -0
- package/src/providers/openai-chat-server.ts +635 -0
- package/src/providers/openai-codex/constants.ts +43 -0
- package/src/providers/openai-codex/request-transformer.ts +161 -0
- package/src/providers/openai-codex/response-handler.ts +81 -0
- package/src/providers/openai-codex-responses.ts +2774 -0
- package/src/providers/openai-completions-compat.ts +289 -0
- package/src/providers/openai-completions.ts +1935 -0
- package/src/providers/openai-request-transform.ts +136 -0
- package/src/providers/openai-responses-server-schema.ts +290 -0
- package/src/providers/openai-responses-server.ts +1190 -0
- package/src/providers/openai-responses-shared.ts +800 -0
- package/src/providers/openai-responses.ts +738 -0
- package/src/providers/pi-native-client.ts +227 -0
- package/src/providers/pi-native-server.ts +210 -0
- package/src/providers/register-builtins.ts +411 -0
- package/src/providers/synthetic.ts +50 -0
- package/src/providers/transform-messages.ts +319 -0
- package/src/providers/vision-guard.ts +31 -0
- package/src/rate-limit-utils.ts +93 -0
- package/src/stream.ts +960 -0
- package/src/types.ts +967 -0
- package/src/usage/claude.ts +431 -0
- package/src/usage/gemini.ts +250 -0
- package/src/usage/github-copilot.ts +421 -0
- package/src/usage/google-antigravity.ts +201 -0
- package/src/usage/grok-cli.ts +163 -0
- package/src/usage/kimi.ts +271 -0
- package/src/usage/minimax-code.ts +31 -0
- package/src/usage/openai-codex.ts +503 -0
- package/src/usage/shared.ts +10 -0
- package/src/usage/zai.ts +247 -0
- package/src/usage.ts +183 -0
- package/src/utils/abort.ts +51 -0
- package/src/utils/anthropic-auth.ts +87 -0
- package/src/utils/discovery/antigravity.ts +261 -0
- package/src/utils/discovery/codex.ts +371 -0
- package/src/utils/discovery/cursor.ts +306 -0
- package/src/utils/discovery/gemini.ts +248 -0
- package/src/utils/discovery/index.ts +4 -0
- package/src/utils/discovery/openai-compatible.ts +230 -0
- package/src/utils/event-stream.ts +172 -0
- package/src/utils/fireworks-model-id.ts +30 -0
- package/src/utils/foundry.ts +8 -0
- package/src/utils/h2-fetch.ts +60 -0
- package/src/utils/http-inspector.ts +255 -0
- package/src/utils/idle-iterator.ts +257 -0
- package/src/utils/json-parse.ts +148 -0
- package/src/utils/oauth/alibaba-coding-plan.ts +59 -0
- package/src/utils/oauth/anthropic.ts +200 -0
- package/src/utils/oauth/api-key-login.ts +87 -0
- package/src/utils/oauth/api-key-validation.ts +92 -0
- package/src/utils/oauth/callback-server.ts +281 -0
- package/src/utils/oauth/cerebras.ts +16 -0
- package/src/utils/oauth/cloudflare-ai-gateway.ts +48 -0
- package/src/utils/oauth/cursor.ts +157 -0
- package/src/utils/oauth/deepseek.ts +53 -0
- package/src/utils/oauth/firepass.ts +24 -0
- package/src/utils/oauth/fireworks.ts +15 -0
- package/src/utils/oauth/github-copilot.ts +362 -0
- package/src/utils/oauth/gitlab-duo.ts +123 -0
- package/src/utils/oauth/google-antigravity.ts +200 -0
- package/src/utils/oauth/google-gemini-cli.ts +256 -0
- package/src/utils/oauth/google-oauth-shared.ts +110 -0
- package/src/utils/oauth/huggingface.ts +62 -0
- package/src/utils/oauth/index.ts +469 -0
- package/src/utils/oauth/kagi.ts +47 -0
- package/src/utils/oauth/kilo.ts +87 -0
- package/src/utils/oauth/kimi.ts +254 -0
- package/src/utils/oauth/litellm.ts +47 -0
- package/src/utils/oauth/lm-studio.ts +38 -0
- package/src/utils/oauth/minimax-code.ts +78 -0
- package/src/utils/oauth/moonshot.ts +16 -0
- package/src/utils/oauth/nanogpt.ts +15 -0
- package/src/utils/oauth/nvidia.ts +70 -0
- package/src/utils/oauth/oauth.html +199 -0
- package/src/utils/oauth/ollama-cloud.ts +28 -0
- package/src/utils/oauth/ollama.ts +47 -0
- package/src/utils/oauth/openai-codex.ts +299 -0
- package/src/utils/oauth/opencode.ts +49 -0
- package/src/utils/oauth/parallel.ts +46 -0
- package/src/utils/oauth/perplexity.ts +206 -0
- package/src/utils/oauth/pkce.ts +18 -0
- package/src/utils/oauth/qianfan.ts +58 -0
- package/src/utils/oauth/qwen-portal.ts +60 -0
- package/src/utils/oauth/synthetic.ts +16 -0
- package/src/utils/oauth/tavily.ts +46 -0
- package/src/utils/oauth/together.ts +16 -0
- package/src/utils/oauth/types.ts +99 -0
- package/src/utils/oauth/venice.ts +59 -0
- package/src/utils/oauth/vercel-ai-gateway.ts +47 -0
- package/src/utils/oauth/vllm.ts +40 -0
- package/src/utils/oauth/xai.ts +246 -0
- package/src/utils/oauth/xiaomi.ts +199 -0
- package/src/utils/oauth/zai.ts +60 -0
- package/src/utils/oauth/zenmux.ts +15 -0
- package/src/utils/overflow.ts +137 -0
- package/src/utils/parse-bind.ts +54 -0
- package/src/utils/provider-response.ts +30 -0
- package/src/utils/retry-after.ts +110 -0
- package/src/utils/retry-budget.ts +4 -0
- package/src/utils/retry.ts +54 -0
- package/src/utils/schema/CONSTRAINTS.md +164 -0
- package/src/utils/schema/adapt.ts +36 -0
- package/src/utils/schema/compatibility.ts +435 -0
- package/src/utils/schema/dereference.ts +98 -0
- package/src/utils/schema/draft.ts +341 -0
- package/src/utils/schema/equality.ts +97 -0
- package/src/utils/schema/fields.ts +190 -0
- package/src/utils/schema/index.ts +13 -0
- package/src/utils/schema/json-schema-validator.ts +577 -0
- package/src/utils/schema/meta-validator.ts +167 -0
- package/src/utils/schema/normalize.ts +1588 -0
- package/src/utils/schema/spill.ts +43 -0
- package/src/utils/schema/stamps.ts +97 -0
- package/src/utils/schema/types.ts +11 -0
- package/src/utils/schema/wire.ts +213 -0
- package/src/utils/schema/zod-decontaminate.ts +331 -0
- package/src/utils/sse-debug.ts +289 -0
- package/src/utils/tool-call-healing.ts +271 -0
- package/src/utils/tool-choice-capability.ts +220 -0
- package/src/utils/tool-choice.ts +99 -0
- package/src/utils/validation.ts +1019 -0
- package/src/utils.ts +178 -0
|
@@ -0,0 +1,772 @@
|
|
|
1
|
+
import { resolveOpenAICompat } from "./providers/openai-completions-compat";
|
|
2
|
+
import type { Api, Model as ApiModel, ThinkingConfig } from "./types";
|
|
3
|
+
import { isClaudeForcedToolChoiceIncapableModelId } from "./utils/tool-choice-capability";
|
|
4
|
+
|
|
5
|
+
/** User-facing thinking levels, ordered least to most intensive. */
|
|
6
|
+
export const enum Effort {
|
|
7
|
+
Minimal = "minimal",
|
|
8
|
+
Low = "low",
|
|
9
|
+
Medium = "medium",
|
|
10
|
+
High = "high",
|
|
11
|
+
XHigh = "xhigh",
|
|
12
|
+
Max = "max",
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export const THINKING_EFFORTS: readonly Effort[] = [
|
|
16
|
+
Effort.Minimal,
|
|
17
|
+
Effort.Low,
|
|
18
|
+
Effort.Medium,
|
|
19
|
+
Effort.High,
|
|
20
|
+
Effort.XHigh,
|
|
21
|
+
Effort.Max,
|
|
22
|
+
];
|
|
23
|
+
|
|
24
|
+
const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
|
|
25
|
+
const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [
|
|
26
|
+
Effort.Minimal,
|
|
27
|
+
Effort.Low,
|
|
28
|
+
Effort.Medium,
|
|
29
|
+
Effort.High,
|
|
30
|
+
Effort.XHigh,
|
|
31
|
+
];
|
|
32
|
+
const DEFAULT_REASONING_EFFORTS_WITH_MAX: readonly Effort[] = [
|
|
33
|
+
Effort.Minimal,
|
|
34
|
+
Effort.Low,
|
|
35
|
+
Effort.Medium,
|
|
36
|
+
Effort.High,
|
|
37
|
+
Effort.Max,
|
|
38
|
+
];
|
|
39
|
+
const DEFAULT_REASONING_EFFORTS_WITH_XHIGH_AND_MAX: readonly Effort[] = [
|
|
40
|
+
Effort.Minimal,
|
|
41
|
+
Effort.Low,
|
|
42
|
+
Effort.Medium,
|
|
43
|
+
Effort.High,
|
|
44
|
+
Effort.XHigh,
|
|
45
|
+
Effort.Max,
|
|
46
|
+
];
|
|
47
|
+
const GEMINI_3_PRO_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High];
|
|
48
|
+
const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
|
|
49
|
+
const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
|
|
50
|
+
const GPT_5_5_DEFAULT_EFFORT = Effort.XHigh;
|
|
51
|
+
|
|
52
|
+
const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
|
|
53
|
+
const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
|
|
54
|
+
|
|
55
|
+
type SemVer = {
|
|
56
|
+
major: number;
|
|
57
|
+
minor: number;
|
|
58
|
+
patch: number;
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
type GeminiKind = "pro" | "flash";
|
|
62
|
+
type AnthropicKind = "opus" | "sonnet";
|
|
63
|
+
type OpenAIVariant = "base" | "codex" | "codex-max" | "codex-mini" | "codex-spark" | "mini" | "max" | "nano";
|
|
64
|
+
|
|
65
|
+
const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>> = {
|
|
66
|
+
base: 0,
|
|
67
|
+
mini: 1,
|
|
68
|
+
nano: 2,
|
|
69
|
+
};
|
|
70
|
+
|
|
71
|
+
const COPILOT_GENERATED_LIMITS: Record<string, { contextWindow: number; maxTokens: number }> = {
|
|
72
|
+
"claude-opus-4.6": { contextWindow: 168000, maxTokens: 32000 },
|
|
73
|
+
"gpt-5.2": { contextWindow: 272000, maxTokens: 128000 },
|
|
74
|
+
"gpt-5.4": { contextWindow: 272000, maxTokens: 128000 },
|
|
75
|
+
"gpt-5.4-mini": { contextWindow: 272000, maxTokens: 128000 },
|
|
76
|
+
"grok-code-fast-1": { contextWindow: 192000, maxTokens: 64000 },
|
|
77
|
+
};
|
|
78
|
+
|
|
79
|
+
interface GeminiModel {
|
|
80
|
+
family: "gemini";
|
|
81
|
+
kind: GeminiKind;
|
|
82
|
+
version: SemVer;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
interface AnthropicModel {
|
|
86
|
+
family: "anthropic";
|
|
87
|
+
kind: AnthropicKind;
|
|
88
|
+
version: SemVer;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
interface OpenAIModel {
|
|
92
|
+
family: "openai";
|
|
93
|
+
variant: OpenAIVariant;
|
|
94
|
+
version: SemVer;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
interface UnknownModel {
|
|
98
|
+
family: "unknown";
|
|
99
|
+
id: string;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
type ParsedModel = GeminiModel | AnthropicModel | OpenAIModel | UnknownModel;
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Static fallback model injected when Cloudflare AI Gateway discovery
|
|
106
|
+
* returns no results. Ensures the provider always has at least one usable
|
|
107
|
+
* model entry in the catalog.
|
|
108
|
+
*/
|
|
109
|
+
export const CLOUDFLARE_FALLBACK_MODEL: ApiModel<"anthropic-messages"> = {
|
|
110
|
+
id: "claude-sonnet-4-5",
|
|
111
|
+
name: "Anthropic Sonnet 4.5",
|
|
112
|
+
api: "anthropic-messages",
|
|
113
|
+
provider: "cloudflare-ai-gateway",
|
|
114
|
+
baseUrl: CLOUDFLARE_AI_GATEWAY_BASE_URL,
|
|
115
|
+
reasoning: true,
|
|
116
|
+
input: ["text", "image"],
|
|
117
|
+
cost: {
|
|
118
|
+
input: 3,
|
|
119
|
+
output: 15,
|
|
120
|
+
cacheRead: 0.3,
|
|
121
|
+
cacheWrite: 3.75,
|
|
122
|
+
},
|
|
123
|
+
contextWindow: 200000,
|
|
124
|
+
maxTokens: 64000,
|
|
125
|
+
};
|
|
126
|
+
|
|
127
|
+
const kEnrichedModel = Symbol("model-thinking.enrichedModel");
|
|
128
|
+
type ModelWithEnriched = ApiModel<Api> & { [kEnrichedModel]?: ApiModel<Api> };
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Returns a copy of the model with canonical thinking metadata attached.
|
|
132
|
+
*
|
|
133
|
+
* This helper belongs to catalog enrichment only. Runtime consumers should
|
|
134
|
+
* trust `model.thinking` and avoid inferring capabilities on demand.
|
|
135
|
+
*/
|
|
136
|
+
export function enrichModelThinking<TApi extends Api>(model: ApiModel<TApi>): ApiModel<TApi> {
|
|
137
|
+
const tagged = model as ModelWithEnriched;
|
|
138
|
+
const cached = tagged[kEnrichedModel];
|
|
139
|
+
if (cached !== undefined) {
|
|
140
|
+
return cached as ApiModel<TApi>;
|
|
141
|
+
}
|
|
142
|
+
const normalizedThinking = normalizeThinkingConfig(model.thinking);
|
|
143
|
+
let result: ApiModel<TApi>;
|
|
144
|
+
if (!model.reasoning) {
|
|
145
|
+
result =
|
|
146
|
+
normalizedThinking === undefined && model.thinking === undefined ? model : { ...model, thinking: undefined };
|
|
147
|
+
} else {
|
|
148
|
+
const thinking = normalizedThinking ?? inferModelThinking(model);
|
|
149
|
+
result = thinkingsEqual(normalizedThinking, thinking) ? model : { ...model, thinking };
|
|
150
|
+
}
|
|
151
|
+
// Stash the enriched copy on a non-enumerable slot so callers that hand us
|
|
152
|
+
// the same reference twice skip the work. `enumerable: false` is critical:
|
|
153
|
+
// many call sites build derived models via `{ ...model, ...overrides }`,
|
|
154
|
+
// which would otherwise copy this cache slot and trick us into returning
|
|
155
|
+
// the *original* enriched model — silently discarding the overrides.
|
|
156
|
+
Object.defineProperty(tagged, kEnrichedModel, {
|
|
157
|
+
value: result,
|
|
158
|
+
enumerable: false,
|
|
159
|
+
configurable: true,
|
|
160
|
+
writable: true,
|
|
161
|
+
});
|
|
162
|
+
return result;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Returns a copy of the model with thinking metadata recomputed from the
|
|
167
|
+
* canonical rules, replacing any existing `thinking`.
|
|
168
|
+
*/
|
|
169
|
+
export function refreshModelThinking<TApi extends Api>(model: ApiModel<TApi>): ApiModel<TApi> {
|
|
170
|
+
if (!model.reasoning) {
|
|
171
|
+
const normalizedThinking = normalizeThinkingConfig(model.thinking);
|
|
172
|
+
return normalizedThinking === undefined && model.thinking === undefined
|
|
173
|
+
? model
|
|
174
|
+
: { ...model, thinking: undefined };
|
|
175
|
+
}
|
|
176
|
+
return { ...model, thinking: inferModelThinking(model) };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* Apply upstream metadata corrections to a mutable array of models.
|
|
181
|
+
*
|
|
182
|
+
* Each model is first normalized through `refreshModelThinking()` so generated
|
|
183
|
+
* catalogs keep canonical thinking metadata and policy fixes in one pass.
|
|
184
|
+
*/
|
|
185
|
+
export function applyGeneratedModelPolicies(models: ApiModel<Api>[]): void {
|
|
186
|
+
for (let index = 0; index < models.length; index++) {
|
|
187
|
+
const model = refreshModelThinking(models[index]!);
|
|
188
|
+
applyGeneratedModelPolicy(model);
|
|
189
|
+
models[index] = model;
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Link OpenAI model variants to their context promotion targets.
|
|
195
|
+
*
|
|
196
|
+
* When a model's context is exhausted, the agent can promote to a sibling
|
|
197
|
+
* model with a larger context window on the same provider:
|
|
198
|
+
* - `OpenAI code backend-spark` variants promote to `gpt-5.5`.
|
|
199
|
+
*
|
|
200
|
+
* `gpt-5.5` itself is a 400K-context model and is not demoted to `gpt-5.4`
|
|
201
|
+
* (which has a smaller window), so it has no promotion target.
|
|
202
|
+
*/
|
|
203
|
+
export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
|
|
204
|
+
for (const candidate of models) {
|
|
205
|
+
const parsedCandidate = parseKnownModel(candidate.id);
|
|
206
|
+
if (parsedCandidate.family !== "openai") continue;
|
|
207
|
+
let targetId: string | undefined;
|
|
208
|
+
if (parsedCandidate.variant === "codex-spark") {
|
|
209
|
+
targetId = "gpt-5.5";
|
|
210
|
+
} else {
|
|
211
|
+
continue;
|
|
212
|
+
}
|
|
213
|
+
const fallback = models.find(
|
|
214
|
+
model => model.provider === candidate.provider && model.api === candidate.api && model.id === targetId,
|
|
215
|
+
);
|
|
216
|
+
if (!fallback) continue;
|
|
217
|
+
candidate.contextPromotionTarget = `${fallback.provider}/${fallback.id}`;
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Returns the supported thinking efforts declared on the model metadata.
|
|
223
|
+
*
|
|
224
|
+
* Catalog enrichment is responsible for normalizing bundled model metadata up front.
|
|
225
|
+
* Runtime callers must treat explicit `model.thinking` on custom models as authoritative
|
|
226
|
+
* so proxy-specific overrides from `models.yml` survive request construction.
|
|
227
|
+
*
|
|
228
|
+
* @throws Error when a reasoning-capable model is missing thinking metadata
|
|
229
|
+
*/
|
|
230
|
+
export function getSupportedEfforts<TApi extends Api>(model: ApiModel<TApi>): readonly Effort[] {
|
|
231
|
+
if (!model.reasoning) {
|
|
232
|
+
return [];
|
|
233
|
+
}
|
|
234
|
+
if (!model.thinking) {
|
|
235
|
+
throw new Error(`Model ${model.provider}/${model.id} is missing thinking metadata`);
|
|
236
|
+
}
|
|
237
|
+
return expandEffortRange(model.thinking);
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Clamps a requested thinking level against explicit model metadata.
|
|
242
|
+
*
|
|
243
|
+
* Non-reasoning models always resolve to `undefined`.
|
|
244
|
+
*/
|
|
245
|
+
export function clampThinkingLevelForModel<TApi extends Api>(
|
|
246
|
+
model: ApiModel<TApi> | undefined,
|
|
247
|
+
requested: Effort | undefined,
|
|
248
|
+
): Effort | undefined {
|
|
249
|
+
if (!model) {
|
|
250
|
+
return requested;
|
|
251
|
+
}
|
|
252
|
+
if (!model.reasoning || requested === undefined) {
|
|
253
|
+
return undefined;
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
const levels = getSupportedEfforts(model);
|
|
257
|
+
if (levels.includes(requested)) {
|
|
258
|
+
return requested;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
const requestedIndex = THINKING_EFFORTS.indexOf(requested);
|
|
262
|
+
if (requestedIndex === -1) {
|
|
263
|
+
return undefined;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
let clamped: Effort | undefined;
|
|
267
|
+
for (const effort of levels) {
|
|
268
|
+
if (THINKING_EFFORTS.indexOf(effort) > requestedIndex) {
|
|
269
|
+
break;
|
|
270
|
+
}
|
|
271
|
+
clamped = effort;
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
return clamped ?? levels[0];
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
export function requireSupportedEffort<TApi extends Api>(model: ApiModel<TApi>, effort: Effort): Effort {
|
|
278
|
+
if (!model.reasoning) {
|
|
279
|
+
throw new Error(`Model ${model.provider}/${model.id} does not support thinking`);
|
|
280
|
+
}
|
|
281
|
+
const levels = getSupportedEfforts(model);
|
|
282
|
+
if (!levels.includes(effort)) {
|
|
283
|
+
throw new Error(
|
|
284
|
+
`Thinking effort ${effort} is not supported by ${model.provider}/${model.id}. Supported efforts: ${levels.join(", ")}`,
|
|
285
|
+
);
|
|
286
|
+
}
|
|
287
|
+
return effort;
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. */
|
|
291
|
+
export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
|
|
292
|
+
model: ApiModel<TApi>,
|
|
293
|
+
effort: Effort,
|
|
294
|
+
): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
|
|
295
|
+
switch (requireSupportedEffort(model, effort)) {
|
|
296
|
+
case Effort.Minimal:
|
|
297
|
+
return "MINIMAL";
|
|
298
|
+
case Effort.Low:
|
|
299
|
+
return "LOW";
|
|
300
|
+
case Effort.Medium:
|
|
301
|
+
return "MEDIUM";
|
|
302
|
+
case Effort.High:
|
|
303
|
+
case Effort.XHigh:
|
|
304
|
+
case Effort.Max:
|
|
305
|
+
return "HIGH";
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
/** Maps a normalized thinking effort to Anthropic adaptive effort values. */
|
|
310
|
+
export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
|
|
311
|
+
model: ApiModel<TApi>,
|
|
312
|
+
effort: Effort,
|
|
313
|
+
): "low" | "medium" | "high" | "xhigh" | "max" {
|
|
314
|
+
switch (requireSupportedEffort(model, effort)) {
|
|
315
|
+
case Effort.Minimal:
|
|
316
|
+
case Effort.Low:
|
|
317
|
+
return "low";
|
|
318
|
+
case Effort.Medium:
|
|
319
|
+
return "medium";
|
|
320
|
+
case Effort.High:
|
|
321
|
+
return "high";
|
|
322
|
+
case Effort.XHigh:
|
|
323
|
+
case Effort.Max:
|
|
324
|
+
return effort === Effort.XHigh ? "xhigh" : "max";
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
/**
|
|
329
|
+
* Returns true for Anthropic models with Opus 4.7 API restrictions:
|
|
330
|
+
* - Sampling parameters (temperature/top_p/top_k) return 400 error
|
|
331
|
+
* - Thinking content is omitted by default (needs display: "summarized")
|
|
332
|
+
*/
|
|
333
|
+
export function hasOpus47ApiRestrictions(modelId: string): boolean {
|
|
334
|
+
const parsed = parseAnthropicModel(getCanonicalModelId(modelId));
|
|
335
|
+
if (!parsed) return false;
|
|
336
|
+
return semverGte(parsed.version, "4.7") && parsed.kind === "opus";
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
|
340
|
+
if (model.api !== "anthropic-messages") return false;
|
|
341
|
+
const parsedModel = parseKnownModel(model.id);
|
|
342
|
+
if (parsedModel.family !== "anthropic" || parsedModel.kind !== "opus") return false;
|
|
343
|
+
return semverGte(parsedModel.version, "4.7");
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
|
|
347
|
+
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
|
|
348
|
+
if (copilotLimits) {
|
|
349
|
+
model.contextWindow = copilotLimits.contextWindow;
|
|
350
|
+
model.maxTokens = copilotLimits.maxTokens;
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
if (
|
|
354
|
+
model.api === "openai-completions" &&
|
|
355
|
+
(model.provider === "minimax-code" || model.provider === "minimax-code-cn")
|
|
356
|
+
) {
|
|
357
|
+
model.compat = {
|
|
358
|
+
...(model.compat ?? {}),
|
|
359
|
+
supportsStore: false,
|
|
360
|
+
supportsDeveloperRole: false,
|
|
361
|
+
supportsReasoningEffort: false,
|
|
362
|
+
reasoningContentField: "reasoning_content",
|
|
363
|
+
};
|
|
364
|
+
delete model.compat.thinkingFormat;
|
|
365
|
+
}
|
|
366
|
+
model.name = scrubGeneratedModelName(model.name);
|
|
367
|
+
if (
|
|
368
|
+
model.api === "openai-completions" &&
|
|
369
|
+
model.provider === "opencode-go" &&
|
|
370
|
+
(model.id === "deepseek-v4-flash" || model.id === "deepseek-v4-pro")
|
|
371
|
+
) {
|
|
372
|
+
model.compat = {
|
|
373
|
+
...(model.compat ?? {}),
|
|
374
|
+
supportsToolChoice: false,
|
|
375
|
+
reasoningContentField: "reasoning_content",
|
|
376
|
+
requiresReasoningContentForToolCalls: true,
|
|
377
|
+
};
|
|
378
|
+
}
|
|
379
|
+
const parsedModel = parseKnownModel(model.id);
|
|
380
|
+
const applyPatchToolType = inferGeneratedApplyPatchToolType(model, parsedModel);
|
|
381
|
+
if (applyPatchToolType) {
|
|
382
|
+
model.applyPatchToolType = applyPatchToolType;
|
|
383
|
+
} else {
|
|
384
|
+
delete model.applyPatchToolType;
|
|
385
|
+
}
|
|
386
|
+
if (
|
|
387
|
+
(model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") &&
|
|
388
|
+
isClaudeForcedToolChoiceIncapableModelId(model.id)
|
|
389
|
+
) {
|
|
390
|
+
// Claude Mythos accepts tools but rejects forced tool use (Anthropic
|
|
391
|
+
// 400: "tool_choice forces tool use is not compatible with this model").
|
|
392
|
+
model.compat = { ...(model.compat ?? {}), toolChoiceSupport: "auto" } as typeof model.compat;
|
|
393
|
+
}
|
|
394
|
+
if (parsedModel.family === "anthropic") {
|
|
395
|
+
applyAnthropicCatalogPolicy(model, parsedModel);
|
|
396
|
+
}
|
|
397
|
+
if (parsedModel.family === "openai") {
|
|
398
|
+
applyOpenAICatalogPolicy(model, parsedModel);
|
|
399
|
+
}
|
|
400
|
+
// GLM-5.2 (Zhipu/ZAI): ships a 1M lossless context window, but the bundled
|
|
401
|
+
// catalog copied GLM-5.1's 200K and that stale value survives generate-models
|
|
402
|
+
// (provider-scoped models bypass the models.dev refresh in applyGlobalModelsDevFallback).
|
|
403
|
+
// Pin to the true 1M so context-cap / auto-compaction thresholds aren't tripped ~5x early.
|
|
404
|
+
if (model.provider === "zai" && model.id === "glm-5.2") {
|
|
405
|
+
model.contextWindow = 1_000_000;
|
|
406
|
+
}
|
|
407
|
+
// MiniMax-M3: official MiniMax docs (platform.minimax.io/docs/guides/models-intro)
|
|
408
|
+
// document a 1M context window, but models.dev and the bundled catalog both report
|
|
409
|
+
// 512K. The stale 512K survives generate-models (provider-scoped models bypass the
|
|
410
|
+
// models.dev refresh in applyGlobalModelsDevFallback), tripping auto-compaction /
|
|
411
|
+
// context-cap thresholds 2x early on MiniMax sessions. Pin to the true 1M.
|
|
412
|
+
if (model.id === "minimax-m3") {
|
|
413
|
+
model.contextWindow = 1_000_000;
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
function scrubGeneratedModelName(name: string): string {
|
|
418
|
+
return name
|
|
419
|
+
.replaceAll("Claude", "Anthropic")
|
|
420
|
+
.replaceAll("claude", "anthropic")
|
|
421
|
+
.replaceAll("Codex", "OpenAI code")
|
|
422
|
+
.replaceAll("codex", "openai-code");
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
function applyAnthropicCatalogPolicy(model: ApiModel<Api>, parsedModel: AnthropicModel): void {
|
|
426
|
+
// Anthropic model Opus 4.5: models.dev reports 3x the correct cache pricing.
|
|
427
|
+
if (model.provider === "anthropic" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.5")) {
|
|
428
|
+
model.cost.cacheRead = 0.5;
|
|
429
|
+
model.cost.cacheWrite = 6.25;
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
// Bedrock Opus 4.6: upstream metadata is stale for cache pricing and context.
|
|
433
|
+
if (model.provider === "amazon-bedrock" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.6")) {
|
|
434
|
+
model.cost.cacheRead = 0.5;
|
|
435
|
+
model.cost.cacheWrite = 6.25;
|
|
436
|
+
model.contextWindow = 1000000;
|
|
437
|
+
model.maxTokens = 128000;
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
function inferGeneratedApplyPatchToolType(
|
|
442
|
+
model: ApiModel<Api>,
|
|
443
|
+
parsedModel: ParsedModel,
|
|
444
|
+
): ApiModel<Api>["applyPatchToolType"] {
|
|
445
|
+
if (parsedModel.family !== "openai" || parsedModel.version.major !== 5) {
|
|
446
|
+
return undefined;
|
|
447
|
+
}
|
|
448
|
+
if (model.provider === "openai" && model.api === "openai-responses") {
|
|
449
|
+
return "freeform";
|
|
450
|
+
}
|
|
451
|
+
if (model.provider === "openai-codex" && model.api === "openai-codex-responses") {
|
|
452
|
+
return "freeform";
|
|
453
|
+
}
|
|
454
|
+
return undefined;
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
function applyGpt55ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel): boolean {
|
|
458
|
+
// gpt-5.5 is a 400K-context model. OpenAI code backend discovery can omit the
|
|
459
|
+
// context window, falling back to the 272K default, which incorrectly trips
|
|
460
|
+
// context-cap / auto-promote thresholds (a ~272K session would look over-cap
|
|
461
|
+
// and demote to gpt-5.4). Pin gpt-5.5 to its true 400K window.
|
|
462
|
+
if (parsedModel.variant === "base" && semverEqual(parsedModel.version, "5.5")) {
|
|
463
|
+
model.contextWindow = 400000;
|
|
464
|
+
return true;
|
|
465
|
+
}
|
|
466
|
+
return false;
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel): void {
|
|
470
|
+
if (applyGpt55ContextWindow(model, parsedModel)) {
|
|
471
|
+
return;
|
|
472
|
+
}
|
|
473
|
+
// OpenAI code backend models: 400K figure includes output budget; input window is 272K.
|
|
474
|
+
if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
|
|
475
|
+
model.contextWindow = 272000;
|
|
476
|
+
return;
|
|
477
|
+
}
|
|
478
|
+
// GPT-5.4 mini/nano use plain OpenAI IDs on the OpenAI code backend transport, but OpenAI code backend still
|
|
479
|
+
// enforces the lower prompt budget for these variants. OpenAI code backend discovery can also
|
|
480
|
+
// report inconsistent priorities for the GPT-5.4 family, so normalize by parsed
|
|
481
|
+
// variant instead of special-casing raw model ids.
|
|
482
|
+
if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.4")) {
|
|
483
|
+
const normalizedPriority = CODEX_GPT_5_4_PRIORITY_BY_VARIANT[parsedModel.variant];
|
|
484
|
+
if (normalizedPriority !== undefined) {
|
|
485
|
+
model.priority = normalizedPriority;
|
|
486
|
+
}
|
|
487
|
+
if (parsedModel.variant === "mini" || parsedModel.variant === "nano") {
|
|
488
|
+
model.contextWindow = 272000;
|
|
489
|
+
}
|
|
490
|
+
}
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
function inferDefaultEffort<TApi extends Api>(model: ApiModel<TApi>, parsedModel: ParsedModel): Effort | undefined {
|
|
494
|
+
if (
|
|
495
|
+
parsedModel.family === "openai" &&
|
|
496
|
+
model.provider === "openai-codex" &&
|
|
497
|
+
semverEqual(parsedModel.version, "5.5")
|
|
498
|
+
) {
|
|
499
|
+
return GPT_5_5_DEFAULT_EFFORT;
|
|
500
|
+
}
|
|
501
|
+
return undefined;
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
function inferModelThinking<TApi extends Api>(model: ApiModel<TApi>): ThinkingConfig {
|
|
505
|
+
const parsedModel = parseKnownModel(model.id);
|
|
506
|
+
const efforts = inferSupportedEfforts(parsedModel, model);
|
|
507
|
+
const minLevel = efforts[0];
|
|
508
|
+
const maxLevel = efforts.at(-1);
|
|
509
|
+
if (!minLevel || !maxLevel) {
|
|
510
|
+
throw new Error(`Model ${model.provider}/${model.id} resolved to an empty thinking range`);
|
|
511
|
+
}
|
|
512
|
+
const config: ThinkingConfig = {
|
|
513
|
+
mode: inferThinkingControlMode(model, parsedModel),
|
|
514
|
+
minLevel,
|
|
515
|
+
maxLevel,
|
|
516
|
+
};
|
|
517
|
+
const defaultLevel = inferDefaultEffort(model, parsedModel);
|
|
518
|
+
if (defaultLevel && efforts.includes(defaultLevel)) {
|
|
519
|
+
config.defaultLevel = defaultLevel;
|
|
520
|
+
}
|
|
521
|
+
// Encode explicit levels only when the inferred set has gaps the min..max range cannot represent.
|
|
522
|
+
const minIndex = THINKING_EFFORTS.indexOf(minLevel);
|
|
523
|
+
const maxIndex = THINKING_EFFORTS.indexOf(maxLevel);
|
|
524
|
+
const expandedRange = THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
|
|
525
|
+
if (expandedRange.length !== efforts.length) {
|
|
526
|
+
config.levels = efforts;
|
|
527
|
+
}
|
|
528
|
+
return config;
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
function normalizeThinkingConfig(thinking: ThinkingConfig | undefined): ThinkingConfig | undefined {
|
|
532
|
+
if (!thinking || expandEffortRange(thinking).length === 0) {
|
|
533
|
+
return undefined;
|
|
534
|
+
}
|
|
535
|
+
return thinking;
|
|
536
|
+
}
|
|
537
|
+
|
|
538
|
+
function thinkingsEqual(left: ThinkingConfig | undefined, right: ThinkingConfig | undefined): boolean {
|
|
539
|
+
if (left === right) return true;
|
|
540
|
+
if (!left || !right) return false;
|
|
541
|
+
if (
|
|
542
|
+
left.mode !== right.mode ||
|
|
543
|
+
left.minLevel !== right.minLevel ||
|
|
544
|
+
left.maxLevel !== right.maxLevel ||
|
|
545
|
+
left.defaultLevel !== right.defaultLevel
|
|
546
|
+
)
|
|
547
|
+
return false;
|
|
548
|
+
const leftLevels = left.levels;
|
|
549
|
+
const rightLevels = right.levels;
|
|
550
|
+
if (leftLevels === rightLevels) return true;
|
|
551
|
+
if (!leftLevels || !rightLevels) return false;
|
|
552
|
+
if (leftLevels.length !== rightLevels.length) return false;
|
|
553
|
+
return leftLevels.every((level, index) => level === rightLevels[index]);
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
function expandEffortRange(thinking: ThinkingConfig): readonly Effort[] {
|
|
557
|
+
if (thinking.levels && thinking.levels.length > 0) {
|
|
558
|
+
return thinking.levels;
|
|
559
|
+
}
|
|
560
|
+
const minIndex = THINKING_EFFORTS.indexOf(thinking.minLevel);
|
|
561
|
+
const maxIndex = THINKING_EFFORTS.indexOf(thinking.maxLevel);
|
|
562
|
+
if (minIndex === -1 || maxIndex === -1 || minIndex > maxIndex) {
|
|
563
|
+
return [];
|
|
564
|
+
}
|
|
565
|
+
return THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
function inferSupportedEfforts<TApi extends Api>(parsedModel: ParsedModel, model: ApiModel<TApi>): readonly Effort[] {
|
|
569
|
+
switch (parsedModel.family) {
|
|
570
|
+
case "openai":
|
|
571
|
+
return inferOpenAISupportedEfforts(parsedModel);
|
|
572
|
+
case "gemini":
|
|
573
|
+
return inferGeminiSupportedEfforts(parsedModel);
|
|
574
|
+
case "anthropic":
|
|
575
|
+
return inferAnthropicSupportedEfforts(parsedModel, model);
|
|
576
|
+
case "unknown":
|
|
577
|
+
return inferFallbackEfforts(model);
|
|
578
|
+
}
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] {
|
|
582
|
+
if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) {
|
|
583
|
+
return GPT_5_1_CODEX_MINI_EFFORTS;
|
|
584
|
+
}
|
|
585
|
+
if (semverGte(model.version, "5.2")) {
|
|
586
|
+
return GPT_5_2_PLUS_EFFORTS;
|
|
587
|
+
}
|
|
588
|
+
return DEFAULT_REASONING_EFFORTS;
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
function inferGeminiSupportedEfforts(model: GeminiModel): readonly Effort[] {
|
|
592
|
+
if (!semverGte(model.version, "3.0")) {
|
|
593
|
+
return DEFAULT_REASONING_EFFORTS;
|
|
594
|
+
}
|
|
595
|
+
return model.kind === "pro" ? GEMINI_3_PRO_EFFORTS : GEMINI_3_FLASH_EFFORTS;
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
function inferAnthropicSupportedEfforts<TApi extends Api>(
|
|
599
|
+
parsedModel: AnthropicModel,
|
|
600
|
+
model: ApiModel<TApi>,
|
|
601
|
+
): readonly Effort[] {
|
|
602
|
+
if (
|
|
603
|
+
(model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") &&
|
|
604
|
+
semverGte(parsedModel.version, "4.6")
|
|
605
|
+
) {
|
|
606
|
+
if (parsedModel.kind !== "opus") return DEFAULT_REASONING_EFFORTS;
|
|
607
|
+
return anthropicModelHasRealXHighEffort(model)
|
|
608
|
+
? DEFAULT_REASONING_EFFORTS_WITH_XHIGH_AND_MAX
|
|
609
|
+
: DEFAULT_REASONING_EFFORTS_WITH_MAX;
|
|
610
|
+
}
|
|
611
|
+
return inferFallbackEfforts(model);
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
function inferFallbackEfforts<TApi extends Api>(model: ApiModel<TApi>): readonly Effort[] {
|
|
615
|
+
if (model.api === "anthropic-messages") {
|
|
616
|
+
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
|
617
|
+
}
|
|
618
|
+
if (model.name.includes("deepseek-v4")) {
|
|
619
|
+
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
|
620
|
+
}
|
|
621
|
+
if (model.api === "bedrock-converse-stream") {
|
|
622
|
+
return DEFAULT_REASONING_EFFORTS;
|
|
623
|
+
}
|
|
624
|
+
if (model.api === "openai-completions") {
|
|
625
|
+
const compat = resolveOpenAICompat(model as ApiModel<"openai-completions">);
|
|
626
|
+
if (compat.thinkingFormat === "openai" && compat.supportsReasoningEffort) {
|
|
627
|
+
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
|
628
|
+
}
|
|
629
|
+
return DEFAULT_REASONING_EFFORTS;
|
|
630
|
+
}
|
|
631
|
+
// OpenAI Responses APIs encode discrete effort levels, including xhigh.
|
|
632
|
+
if (model.api === "openai-responses" || model.api === "openai-codex-responses") {
|
|
633
|
+
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
|
634
|
+
}
|
|
635
|
+
return DEFAULT_REASONING_EFFORTS;
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
function inferThinkingControlMode<TApi extends Api>(
|
|
639
|
+
model: ApiModel<TApi>,
|
|
640
|
+
parsedModel: ParsedModel,
|
|
641
|
+
): ThinkingConfig["mode"] {
|
|
642
|
+
switch (model.api) {
|
|
643
|
+
case "google-generative-ai":
|
|
644
|
+
case "google-gemini-cli":
|
|
645
|
+
case "google-vertex":
|
|
646
|
+
return parsedModel.family === "gemini" &&
|
|
647
|
+
semverGte(parsedModel.version, "3.0") &&
|
|
648
|
+
parsedModel.version.major === 3
|
|
649
|
+
? "google-level"
|
|
650
|
+
: "budget";
|
|
651
|
+
|
|
652
|
+
case "anthropic-messages":
|
|
653
|
+
if (parsedModel.family === "anthropic") {
|
|
654
|
+
if (semverGte(parsedModel.version, "4.6")) {
|
|
655
|
+
return "anthropic-adaptive";
|
|
656
|
+
}
|
|
657
|
+
if (semverGte(parsedModel.version, "4.5")) {
|
|
658
|
+
return "anthropic-budget-effort";
|
|
659
|
+
}
|
|
660
|
+
}
|
|
661
|
+
return "budget";
|
|
662
|
+
|
|
663
|
+
case "bedrock-converse-stream":
|
|
664
|
+
if (parsedModel.family === "anthropic") {
|
|
665
|
+
if (semverGte(parsedModel.version, "4.6") && parsedModel.kind === "opus") {
|
|
666
|
+
return "anthropic-adaptive";
|
|
667
|
+
}
|
|
668
|
+
if (semverGte(parsedModel.version, "4.5")) {
|
|
669
|
+
return "anthropic-budget-effort";
|
|
670
|
+
}
|
|
671
|
+
}
|
|
672
|
+
return "budget";
|
|
673
|
+
|
|
674
|
+
default:
|
|
675
|
+
return "effort";
|
|
676
|
+
}
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
function parseKnownModel(modelId: string): ParsedModel {
|
|
680
|
+
const canonicalId = getCanonicalModelId(modelId);
|
|
681
|
+
return (
|
|
682
|
+
parseGeminiModel(canonicalId) ??
|
|
683
|
+
parseAnthropicModel(canonicalId) ??
|
|
684
|
+
parseOpenAIModel(canonicalId) ?? { family: "unknown", id: canonicalId }
|
|
685
|
+
);
|
|
686
|
+
}
|
|
687
|
+
|
|
688
|
+
const GEMINI_SUFFIX = "-preview";
|
|
689
|
+
function parseGeminiModel(modelId: string): GeminiModel | null {
|
|
690
|
+
if (modelId.endsWith(GEMINI_SUFFIX)) {
|
|
691
|
+
modelId = modelId.slice(0, -GEMINI_SUFFIX.length);
|
|
692
|
+
}
|
|
693
|
+
const match = /gemini-(\d+(?:\.\d+){0,2})-(pro|flash)\b/.exec(modelId);
|
|
694
|
+
if (!match) {
|
|
695
|
+
return null;
|
|
696
|
+
}
|
|
697
|
+
const version = parseSemVer(match[1]);
|
|
698
|
+
if (!version) {
|
|
699
|
+
return null;
|
|
700
|
+
}
|
|
701
|
+
return { family: "gemini", kind: match[2] as GeminiKind, version };
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
function parseAnthropicModel(modelId: string): AnthropicModel | null {
|
|
705
|
+
const match = /claude-(opus|sonnet)-(\d{1,2}(?:[.-]\d{1,2}){0,2})\b/.exec(modelId);
|
|
706
|
+
if (!match) {
|
|
707
|
+
return null;
|
|
708
|
+
}
|
|
709
|
+
const version = parseSemVer(match[2]);
|
|
710
|
+
if (!version) {
|
|
711
|
+
return null;
|
|
712
|
+
}
|
|
713
|
+
return { family: "anthropic", kind: match[1] as AnthropicKind, version };
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
function parseOpenAIModel(modelId: string): OpenAIModel | null {
|
|
717
|
+
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))?$/.exec(modelId);
|
|
718
|
+
if (!match) {
|
|
719
|
+
return null;
|
|
720
|
+
}
|
|
721
|
+
const version = parseSemVer(match[1]);
|
|
722
|
+
if (!version) {
|
|
723
|
+
return null;
|
|
724
|
+
}
|
|
725
|
+
return { family: "openai", variant: (match[2] as OpenAIVariant | undefined) ?? "base", version };
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
function createSemVer(major: number, minor: number, patch = 0): SemVer {
|
|
729
|
+
return { major, minor, patch };
|
|
730
|
+
}
|
|
731
|
+
|
|
732
|
+
// extend this table if we need anything more than 9.10
|
|
733
|
+
const precomputeTable: Record<string, SemVer> = {};
|
|
734
|
+
for (let major = 0; major <= 9; major++) {
|
|
735
|
+
for (let minor = 0; minor <= 10; minor++) {
|
|
736
|
+
const version = createSemVer(major, minor, 0);
|
|
737
|
+
precomputeTable[`${major}.${minor}`] = version;
|
|
738
|
+
precomputeTable[`${major}-${minor}`] = version;
|
|
739
|
+
}
|
|
740
|
+
precomputeTable[`${major}`] = createSemVer(major, 0, 0);
|
|
741
|
+
}
|
|
742
|
+
|
|
743
|
+
function parseSemVer(version: string): SemVer | null {
|
|
744
|
+
return precomputeTable[version] ?? null;
|
|
745
|
+
}
|
|
746
|
+
|
|
747
|
+
function semverGte(left: SemVer | string, right: SemVer | string): boolean {
|
|
748
|
+
return compareSemVer(left, right) >= 0;
|
|
749
|
+
}
|
|
750
|
+
|
|
751
|
+
function semverEqual(left: SemVer | string, right: SemVer | string): boolean {
|
|
752
|
+
return compareSemVer(left, right) === 0;
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
function compareSemVer(left: SemVer | string | null, right: SemVer | string | null): number {
|
|
756
|
+
left = typeof left === "string" ? parseSemVer(left) : left;
|
|
757
|
+
right = typeof right === "string" ? parseSemVer(right) : right;
|
|
758
|
+
if (!left || !right) return (left ? 1 : 0) - (right ? 1 : 0);
|
|
759
|
+
|
|
760
|
+
if (left.major !== right.major) {
|
|
761
|
+
return left.major - right.major;
|
|
762
|
+
}
|
|
763
|
+
if (left.minor !== right.minor) {
|
|
764
|
+
return left.minor - right.minor;
|
|
765
|
+
}
|
|
766
|
+
return left.patch - right.patch;
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
function getCanonicalModelId(modelId: string): string {
|
|
770
|
+
const p = modelId.lastIndexOf("/");
|
|
771
|
+
return p !== -1 ? modelId.slice(p + 1) : modelId;
|
|
772
|
+
}
|