@ai-sdk/gateway 4.0.90 → 4.0.92
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/README.md +1 -1
- package/dist/index.d.ts +14 -8
- package/dist/index.js +1 -1
- package/docs/00-ai-gateway.mdx +29 -27
- package/package.json +2 -2
- package/src/gateway-image-model-settings.ts +1 -0
- package/src/gateway-language-model-settings.ts +3 -1
- package/src/gateway-provider-options.ts +17 -4
- package/src/gateway-speech-model-settings.ts +2 -0
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,25 @@
|
|
|
1
1
|
# @ai-sdk/gateway
|
|
2
2
|
|
|
3
|
+
## 4.0.92
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- 124b6ef: chore(provider/gateway): update gateway model settings files
|
|
8
|
+
|
|
9
|
+
## 4.0.91
|
|
10
|
+
|
|
11
|
+
### Patch Changes
|
|
12
|
+
|
|
13
|
+
- 2693319: Add Gemini 3.8 TTS support with structured speech metadata and per-turn speaker and style controls for prebuilt voices. Preserve native WAV responses without adding a second header, support explicit raw PCM, mu-law, and A-law output, and identify headerless audio formats correctly. Add the Gemini 3.8 speech model IDs to Google and Gateway types.
|
|
14
|
+
|
|
15
|
+
Share transcript and custom-voice inspection through the Google provider internal export, and reject empty speech transcripts before sending a request. Default newer and custom model IDs to structured speech while preserving the legacy format for Gemini 2.5 and 3.1.
|
|
16
|
+
|
|
17
|
+
- b73f2f9: feat(provider/gateway): add quantization conditions to the has provider option
|
|
18
|
+
- Updated dependencies [fe07867]
|
|
19
|
+
- Updated dependencies [a4b0940]
|
|
20
|
+
- Updated dependencies [771e74b]
|
|
21
|
+
- @ai-sdk/provider-utils@5.0.47
|
|
22
|
+
|
|
3
23
|
## 4.0.90
|
|
4
24
|
|
|
5
25
|
### Patch Changes
|
package/README.md
CHANGED
|
@@ -33,7 +33,7 @@ import { gateway } from '@ai-sdk/gateway';
|
|
|
33
33
|
import { generateText } from 'ai';
|
|
34
34
|
|
|
35
35
|
const { text } = await generateText({
|
|
36
|
-
model: gateway('
|
|
36
|
+
model: gateway('spacexai/grok-4.7'),
|
|
37
37
|
prompt:
|
|
38
38
|
'Tell me about the history of the San Francisco Mission-style burrito.',
|
|
39
39
|
});
|
package/dist/index.d.ts
CHANGED
|
@@ -6,9 +6,9 @@ type GatewayEmbeddingModelId = 'alibaba/qwen3-embedding-0.6b' | 'alibaba/qwen3-e
|
|
|
6
6
|
|
|
7
7
|
type GatewayEvaluationModelId = 'typesafe-ai/jev' | (string & {});
|
|
8
8
|
|
|
9
|
-
type GatewayImageModelId = 'bfl/flux-2-flex' | 'bfl/flux-2-klein-4b' | 'bfl/flux-2-klein-9b' | 'bfl/flux-2-max' | 'bfl/flux-2-pro' | 'bfl/flux-kontext-max' | 'bfl/flux-kontext-pro' | 'bfl/flux-pro-1.0-fill' | 'bfl/flux-pro-1.1' | 'bfl/flux-pro-1.1-ultra' | 'bytedance/seedream-4.0' | 'bytedance/seedream-4.5' | 'bytedance/seedream-5.0-lite' | 'bytedance/seedream-5.0-pro' | 'meta/muse-image-1.0' | 'openai/gpt-image-1' | 'openai/gpt-image-1-mini' | 'openai/gpt-image-1.5' | 'openai/gpt-image-2' | 'openai/gpt-image-2.5-flare' | 'openai/gpt-image-2.5-sunburst' | 'prodia/flux-fast-schnell' | 'quiverai/arrow-1.1' | 'recraft/recraft-v2' | 'recraft/recraft-v3' | 'recraft/recraft-v4' | 'recraft/recraft-v4-pro' | 'recraft/recraft-v4.1' | 'recraft/recraft-v4.1-pro' | 'recraft/recraft-v4.1-utility' | 'recraft/recraft-v4.1-utility-pro' | 'spacexai/grok-imagine-image' | 'spacexai/grok-imagine-image-2.0' | (string & {});
|
|
9
|
+
type GatewayImageModelId = 'bfl/flux-2-flex' | 'bfl/flux-2-klein-4b' | 'bfl/flux-2-klein-9b' | 'bfl/flux-2-max' | 'bfl/flux-2-pro' | 'bfl/flux-kontext-max' | 'bfl/flux-kontext-pro' | 'bfl/flux-pro-1.0-fill' | 'bfl/flux-pro-1.1' | 'bfl/flux-pro-1.1-ultra' | 'bytedance/seedream-4.0' | 'bytedance/seedream-4.5' | 'bytedance/seedream-5.0-lite' | 'bytedance/seedream-5.0-pro' | 'meta/muse-image-1.0' | 'openai/gpt-image-1' | 'openai/gpt-image-1-mini' | 'openai/gpt-image-1.5' | 'openai/gpt-image-2' | 'openai/gpt-image-2.5-flare' | 'openai/gpt-image-2.5-sunburst' | 'prodia/flux-fast-schnell' | 'quiverai/arrow-1.1' | 'recraft/recraft-v2' | 'recraft/recraft-v3' | 'recraft/recraft-v4' | 'recraft/recraft-v4-pro' | 'recraft/recraft-v4.1' | 'recraft/recraft-v4.1-flash' | 'recraft/recraft-v4.1-pro' | 'recraft/recraft-v4.1-utility' | 'recraft/recraft-v4.1-utility-pro' | 'spacexai/grok-imagine-image' | 'spacexai/grok-imagine-image-2.0' | (string & {});
|
|
10
10
|
|
|
11
|
-
type GatewayModelId = 'alibaba/qwen-3-14b' | 'alibaba/qwen-3-235b' | 'alibaba/qwen-3-30b' | 'alibaba/qwen-3-32b' | 'alibaba/qwen-3.6-max-preview' | 'alibaba/qwen3-235b-a22b-thinking' | 'alibaba/qwen3-coder' | 'alibaba/qwen3-coder-30b-a3b' | 'alibaba/qwen3-coder-next' | 'alibaba/qwen3-coder-plus' | 'alibaba/qwen3-max' | 'alibaba/qwen3-max-preview' | 'alibaba/qwen3-max-thinking' | 'alibaba/qwen3-next-80b-a3b-instruct' | 'alibaba/qwen3-next-80b-a3b-thinking' | 'alibaba/qwen3-vl-235b-a22b-instruct' | 'alibaba/qwen3-vl-instruct' | 'alibaba/qwen3-vl-thinking' | 'alibaba/qwen3.5-flash' | 'alibaba/qwen3.5-plus' | 'alibaba/qwen3.6-27b' | 'alibaba/qwen3.6-plus' | 'alibaba/qwen3.7-flash' | 'alibaba/qwen3.7-max' | 'alibaba/qwen3.7-plus' | 'alibaba/qwen3.8-2.4t-a95b' | 'alibaba/qwen3.8-27b' | 'alibaba/qwen3.8-flash' | 'alibaba/qwen3.8-max' | 'alibaba/qwen3.8-max-0902' | 'alibaba/qwen3.8-omni-flash' | 'amazon/nova-2-lite' | 'amazon/nova-lite' | 'amazon/nova-micro' | 'amazon/nova-pro' | 'anthropic/claude-3-haiku' | 'anthropic/claude-fable-5' | 'anthropic/claude-fable-5.1' | 'anthropic/claude-haiku-4.5' | 'anthropic/claude-opus-4' | 'anthropic/claude-opus-4.5' | 'anthropic/claude-opus-4.6' | 'anthropic/claude-opus-4.7' | 'anthropic/claude-opus-4.8' | 'anthropic/claude-opus-4.8-fast' | 'anthropic/claude-opus-5' | 'anthropic/claude-opus-5-fast' | 'anthropic/claude-opus-5.5' | 'anthropic/claude-sonnet-4' | 'anthropic/claude-sonnet-4.5' | 'anthropic/claude-sonnet-4.6' | 'anthropic/claude-sonnet-5' | 'arcee-ai/trinity-large-thinking' | 'bytedance/seed-1.6' | 'bytedance/seed-1.8' | 'bytedance/seed-2.1-turbo' | 'cohere/command-a' | 'deepseek/deepseek-r1' | 'deepseek/deepseek-v3.1' | 'deepseek/deepseek-v3.1-terminus' | 'deepseek/deepseek-v3.2' | 'deepseek/deepseek-v3.2-thinking' | 'deepseek/deepseek-v4-flash' | 'deepseek/deepseek-v4-flash-0731' | 'deepseek/deepseek-v4-flash-vision-exp' | 'deepseek/deepseek-v4-pro' | 'deepseek/deepseek-v4-pro-0813' | 'deepseek/deepseek-v4.1-flash' | 'google/gemini-2.5-flash' | 'google/gemini-2.5-flash-image' | 'google/gemini-2.5-flash-lite' | 'google/gemini-2.5-pro' | 'google/gemini-3-flash' | 'google/gemini-3-pro-image' | 'google/gemini-3.1-flash-image' | 'google/gemini-3.1-flash-image-preview' | 'google/gemini-3.1-flash-lite' | 'google/gemini-3.1-flash-lite-image' | 'google/gemini-3.1-pro-preview' | 'google/gemini-3.5-flash' | 'google/gemini-3.5-flash-lite' | 'google/gemini-3.6-flash' | 'google/gemini-3.7-flash' | 'google/gemini-3.8-flash' | 'google/gemini-omni-flash-preview' | 'google/gemma-4-26b-a4b-it' | 'google/gemma-4-31b-it' | 'inception/mercury-2' | 'inception/mercury-2.5' | 'inception/mercury-coder-small' | 'inclusionai/ling-3.0-flash' | 'inclusionai/ling-3.0-flash-fin' | 'inclusionai/ling-3.0-flash-fin-free' | 'inclusionai/ling-3.0-flash-sante' | 'inclusionai/ling-3.0-flash-sante-free' | 'inclusionai/ling-3.0-flash-vl' | '
|
|
11
|
+
type GatewayModelId = 'alibaba/qwen-3-14b' | 'alibaba/qwen-3-235b' | 'alibaba/qwen-3-30b' | 'alibaba/qwen-3-32b' | 'alibaba/qwen-3.6-max-preview' | 'alibaba/qwen3-235b-a22b-thinking' | 'alibaba/qwen3-coder' | 'alibaba/qwen3-coder-30b-a3b' | 'alibaba/qwen3-coder-next' | 'alibaba/qwen3-coder-plus' | 'alibaba/qwen3-max' | 'alibaba/qwen3-max-preview' | 'alibaba/qwen3-max-thinking' | 'alibaba/qwen3-next-80b-a3b-instruct' | 'alibaba/qwen3-next-80b-a3b-thinking' | 'alibaba/qwen3-vl-235b-a22b-instruct' | 'alibaba/qwen3-vl-instruct' | 'alibaba/qwen3-vl-thinking' | 'alibaba/qwen3.5-flash' | 'alibaba/qwen3.5-plus' | 'alibaba/qwen3.6-27b' | 'alibaba/qwen3.6-plus' | 'alibaba/qwen3.7-flash' | 'alibaba/qwen3.7-max' | 'alibaba/qwen3.7-plus' | 'alibaba/qwen3.8-2.4t-a95b' | 'alibaba/qwen3.8-27b' | 'alibaba/qwen3.8-flash' | 'alibaba/qwen3.8-max' | 'alibaba/qwen3.8-max-0902' | 'alibaba/qwen3.8-max-prime' | 'alibaba/qwen3.8-omni-flash' | 'amazon/nova-2-lite' | 'amazon/nova-lite' | 'amazon/nova-micro' | 'amazon/nova-pro' | 'anthropic/claude-3-haiku' | 'anthropic/claude-fable-5' | 'anthropic/claude-fable-5.1' | 'anthropic/claude-haiku-4.5' | 'anthropic/claude-opus-4' | 'anthropic/claude-opus-4.5' | 'anthropic/claude-opus-4.6' | 'anthropic/claude-opus-4.7' | 'anthropic/claude-opus-4.8' | 'anthropic/claude-opus-4.8-fast' | 'anthropic/claude-opus-5' | 'anthropic/claude-opus-5-fast' | 'anthropic/claude-opus-5.5' | 'anthropic/claude-opus-5.5-fast' | 'anthropic/claude-sonnet-4' | 'anthropic/claude-sonnet-4.5' | 'anthropic/claude-sonnet-4.6' | 'anthropic/claude-sonnet-5' | 'arcee-ai/trinity-large-thinking' | 'bytedance/seed-1.6' | 'bytedance/seed-1.8' | 'bytedance/seed-2.1-turbo' | 'cohere/command-a' | 'deepseek/deepseek-r1' | 'deepseek/deepseek-v3.1' | 'deepseek/deepseek-v3.1-terminus' | 'deepseek/deepseek-v3.2' | 'deepseek/deepseek-v3.2-thinking' | 'deepseek/deepseek-v4-flash' | 'deepseek/deepseek-v4-flash-0731' | 'deepseek/deepseek-v4-flash-vision-exp' | 'deepseek/deepseek-v4-pro' | 'deepseek/deepseek-v4-pro-0813' | 'deepseek/deepseek-v4.1-flash' | 'fireworks/ember-1' | 'google/gemini-2.5-flash' | 'google/gemini-2.5-flash-image' | 'google/gemini-2.5-flash-lite' | 'google/gemini-2.5-pro' | 'google/gemini-3-flash' | 'google/gemini-3-pro-image' | 'google/gemini-3.1-flash-image' | 'google/gemini-3.1-flash-image-preview' | 'google/gemini-3.1-flash-lite' | 'google/gemini-3.1-flash-lite-image' | 'google/gemini-3.1-pro-preview' | 'google/gemini-3.5-flash' | 'google/gemini-3.5-flash-lite' | 'google/gemini-3.6-flash' | 'google/gemini-3.7-flash' | 'google/gemini-3.8-flash' | 'google/gemini-omni-flash-preview' | 'google/gemma-4-26b-a4b-it' | 'google/gemma-4-31b-it' | 'inception/mercury-2' | 'inception/mercury-2.5' | 'inception/mercury-coder-small' | 'inclusionai/ling-3.0-flash' | 'inclusionai/ling-3.0-flash-fin' | 'inclusionai/ling-3.0-flash-fin-free' | 'inclusionai/ling-3.0-flash-sante' | 'inclusionai/ling-3.0-flash-sante-free' | 'inclusionai/ling-3.0-flash-vl' | 'inference-net/schematron-v2-small' | 'inference-net/schematron-v2-turbo' | 'interfaze/interfaze-beta' | 'meta/llama-3.1-70b' | 'meta/llama-3.1-8b' | 'meta/llama-3.3-70b' | 'meta/llama-4-maverick' | 'meta/llama-4-scout' | 'meta/muse-glimmer-30b' | 'meta/muse-spark-1.1' | 'meta/muse-spark-1.2' | 'meta/muse-spark-1.2-contributor' | 'meta/muse-spark-1.3' | 'meta/muse-spark-1.3-contributor' | 'minimax/minimax-m2' | 'minimax/minimax-m2.1' | 'minimax/minimax-m2.1-lightning' | 'minimax/minimax-m2.5' | 'minimax/minimax-m2.5-highspeed' | 'minimax/minimax-m2.7' | 'minimax/minimax-m2.7-highspeed' | 'minimax/minimax-m3' | 'mistral/codestral' | 'mistral/ministral-14b' | 'mistral/ministral-3b' | 'mistral/ministral-8b' | 'mistral/mistral-large-3' | 'mistral/mistral-medium-3.5' | 'mistral/mistral-nemo' | 'mistral/mistral-small' | 'mixedbread/toast-1' | 'moonshotai/kimi-k2' | 'moonshotai/kimi-k2-thinking' | 'moonshotai/kimi-k2.5' | 'moonshotai/kimi-k2.6' | 'moonshotai/kimi-k2.7-code' | 'moonshotai/kimi-k2.7-code-highspeed' | 'moonshotai/kimi-k3' | 'moonshotai/kimi-k3-fast' | 'morph/morph-v3-fast' | 'morph/morph-v3-large' | 'nvidia/nemotron-3-nano-30b-a3b' | 'nvidia/nemotron-3-super-120b-a12b' | 'nvidia/nemotron-3-ultra-550b-a55b' | 'nvidia/nemotron-3.5-lightning' | 'nvidia/nemotron-nano-12b-v2-vl' | 'nvidia/nemotron-nano-9b-v2' | 'openai/gpt-3.5-turbo' | 'openai/gpt-4-turbo' | 'openai/gpt-4.1' | 'openai/gpt-4.1-fast' | 'openai/gpt-4.1-mini' | 'openai/gpt-4.1-mini-fast' | 'openai/gpt-4.1-nano' | 'openai/gpt-4.1-nano-fast' | 'openai/gpt-4o' | 'openai/gpt-4o-fast' | 'openai/gpt-4o-mini' | 'openai/gpt-4o-mini-fast' | 'openai/gpt-5' | 'openai/gpt-5-codex' | 'openai/gpt-5-fast' | 'openai/gpt-5-mini' | 'openai/gpt-5-mini-fast' | 'openai/gpt-5-nano' | 'openai/gpt-5-pro' | 'openai/gpt-5.1-codex' | 'openai/gpt-5.1-codex-max' | 'openai/gpt-5.1-codex-mini' | 'openai/gpt-5.1-thinking' | 'openai/gpt-5.1-thinking-fast' | 'openai/gpt-5.2' | 'openai/gpt-5.2-codex' | 'openai/gpt-5.2-fast' | 'openai/gpt-5.2-pro' | 'openai/gpt-5.3-codex' | 'openai/gpt-5.3-codex-fast' | 'openai/gpt-5.4' | 'openai/gpt-5.4-fast' | 'openai/gpt-5.4-mini' | 'openai/gpt-5.4-mini-fast' | 'openai/gpt-5.4-nano' | 'openai/gpt-5.4-pro' | 'openai/gpt-5.5' | 'openai/gpt-5.5-fast' | 'openai/gpt-5.5-pro' | 'openai/gpt-5.6-luna' | 'openai/gpt-5.6-luna-fast' | 'openai/gpt-5.6-sol' | 'openai/gpt-5.6-sol-fast' | 'openai/gpt-5.6-terra' | 'openai/gpt-5.6-terra-fast' | 'openai/gpt-6-astra' | 'openai/gpt-6-astra-fast' | 'openai/gpt-6-luna' | 'openai/gpt-6-luna-fast' | 'openai/gpt-6-sol' | 'openai/gpt-6-sol-fast' | 'openai/gpt-oss-120b' | 'openai/gpt-oss-20b' | 'openai/gpt-oss-safeguard-120b' | 'openai/gpt-oss-safeguard-20b' | 'openai/o1' | 'openai/o3' | 'openai/o3-fast' | 'openai/o3-mini' | 'openai/o3-pro' | 'openai/o4-mini' | 'openai/o4-mini-fast' | 'perplexity/sonar' | 'perplexity/sonar-pro' | 'perplexity/sonar-reasoning-pro' | 'poolside/laguna-s-2.1' | 'poolside/laguna-s-2.1-free' | 'quiverai/arrow-2' | 'quiverai/arrow-2-telos' | 'sakana/fugu-max' | 'sakana/fugu-ultra' | 'sakana/fugu-ultra-v2' | 'sakana/namazu' | 'spacexai/grok-4.1-fast-non-reasoning' | 'spacexai/grok-4.1-fast-reasoning' | 'spacexai/grok-4.20-multi-agent' | 'spacexai/grok-4.20-multi-agent-beta' | 'spacexai/grok-4.20-non-reasoning' | 'spacexai/grok-4.20-non-reasoning-beta' | 'spacexai/grok-4.20-reasoning' | 'spacexai/grok-4.20-reasoning-beta' | 'spacexai/grok-4.3' | 'spacexai/grok-4.5' | 'spacexai/grok-4.6' | 'spacexai/grok-4.7' | 'spacexai/grok-build-0.1' | 'stepfun/step-3.5-flash' | 'stepfun/step-3.7-flash' | 'stepfun/step-5-preview' | 'tencent/hy-mt2-lite' | 'tencent/hy-mt2-plus' | 'tencent/hy-mt2-pro' | 'tencent/hy3' | 'tencent/hy4-preview' | 'thinkingmachines/inkling' | 'thinkingmachines/inkling-small' | 'xiaomi/mimo-v2.5' | 'xiaomi/mimo-v2.5-pro' | 'xiaomi/mimo-v2.6-flash' | 'xiaomi/mimo-v2.6-pro' | 'xiaomi/mimo-v2.6-pro-ultraspeed' | 'zai/glm-4.5' | 'zai/glm-4.5-air' | 'zai/glm-4.5v' | 'zai/glm-4.6' | 'zai/glm-4.7' | 'zai/glm-4.7-flash' | 'zai/glm-4.7-flashx' | 'zai/glm-5' | 'zai/glm-5-turbo' | 'zai/glm-5.1' | 'zai/glm-5.2' | 'zai/glm-5.2-fast' | 'zai/glm-5.3' | 'zai/glm-5.3-fast' | 'zai/glm-5.3-flash' | 'zai/glm-5.3-flashx' | 'zai/glm-5v-turbo' | (string & {});
|
|
12
12
|
|
|
13
13
|
/**
|
|
14
14
|
* Shared WebSocket subprotocol contract for AI Gateway realtime and streaming
|
|
@@ -81,7 +81,7 @@ type GatewayRealtimeModelId = 'google/gemini-3.8-live' | 'google/gemini-3.8-live
|
|
|
81
81
|
|
|
82
82
|
type GatewayRerankingModelId = 'cohere/rerank-v3.5' | 'cohere/rerank-v4-fast' | 'cohere/rerank-v4-pro' | 'voyage/rerank-2.5' | 'voyage/rerank-2.5-lite' | (string & {});
|
|
83
83
|
|
|
84
|
-
type GatewaySpeechModelId = 'fish-audio/s1' | 'fish-audio/s2-pro' | 'fish-audio/s2.1-pro' | 'openai/tts-1' | 'openai/tts-1-hd' | 'spacexai/grok-tts' | (string & {});
|
|
84
|
+
type GatewaySpeechModelId = 'fish-audio/s1' | 'fish-audio/s2-pro' | 'fish-audio/s2.1-pro' | 'google/gemini-3.8-flash-lite-tts' | 'google/gemini-3.8-flash-tts' | 'openai/tts-1' | 'openai/tts-1-hd' | 'spacexai/grok-tts' | (string & {});
|
|
85
85
|
|
|
86
86
|
type GatewayTranscriptionModelId = 'fish-audio/transcribe-1' | 'google/gemini-3.5-transcribe' | 'google/gemini-3.5-transcribe-live' | 'openai/gpt-4o-mini-transcribe' | 'openai/gpt-4o-transcribe' | 'openai/gpt-realtime-whisper' | 'openai/whisper-1' | 'spacexai/grok-stt' | (string & {});
|
|
87
87
|
|
|
@@ -1120,11 +1120,17 @@ type GatewayProviderOptions = {
|
|
|
1120
1120
|
/** Filter to providers that do not train on prompt data. */
|
|
1121
1121
|
disallowPromptTraining?: boolean;
|
|
1122
1122
|
/**
|
|
1123
|
-
* Restrict routing to models that
|
|
1124
|
-
*
|
|
1125
|
-
* `'
|
|
1126
|
-
|
|
1127
|
-
|
|
1123
|
+
* Restrict routing to provider models that satisfy every given entry.
|
|
1124
|
+
*
|
|
1125
|
+
* Entries are capability tags (`'implicit-caching'`, `'reasoning'`,
|
|
1126
|
+
* `'tool-use'`, `'vision'`) or weight-format conditions: `'quantization:fp8'`
|
|
1127
|
+
* requires the serving provider to report that weight format,
|
|
1128
|
+
* `'!quantization:fp8'` excludes it (providers with no recorded format still
|
|
1129
|
+
* pass an exclusion). Format values are an open space but must match
|
|
1130
|
+
* `[a-zA-Z0-9._-]{1,32}` and compare case-insensitively; unknown capability
|
|
1131
|
+
* names are rejected by the Gateway with a 400.
|
|
1132
|
+
*/
|
|
1133
|
+
has?: Array<'implicit-caching' | 'reasoning' | 'tool-use' | 'vision' | `quantization:${string}` | `!quantization:${string}`>;
|
|
1128
1134
|
/**
|
|
1129
1135
|
* Idempotency key for `experimental_startBatch`: retries with the same
|
|
1130
1136
|
* key replay the original batch instead of creating a duplicate.
|
package/dist/index.js
CHANGED
|
@@ -3662,7 +3662,7 @@ async function getVercelRequestId() {
|
|
|
3662
3662
|
}
|
|
3663
3663
|
|
|
3664
3664
|
// src/version.ts
|
|
3665
|
-
var VERSION = true ? "4.0.
|
|
3665
|
+
var VERSION = true ? "4.0.92" : "0.0.0-test";
|
|
3666
3666
|
|
|
3667
3667
|
// src/gateway-provider.ts
|
|
3668
3668
|
var AI_GATEWAY_PROTOCOL_VERSION = "0.0.1";
|
package/docs/00-ai-gateway.mdx
CHANGED
|
@@ -29,7 +29,7 @@ For most use cases, you can use the AI Gateway directly with a model string:
|
|
|
29
29
|
import { generateText } from 'ai';
|
|
30
30
|
|
|
31
31
|
const { text } = await generateText({
|
|
32
|
-
model: 'openai/gpt-
|
|
32
|
+
model: 'openai/gpt-6-astra',
|
|
33
33
|
prompt: 'Hello world',
|
|
34
34
|
});
|
|
35
35
|
```
|
|
@@ -39,7 +39,7 @@ const { text } = await generateText({
|
|
|
39
39
|
import { generateText, gateway } from 'ai';
|
|
40
40
|
|
|
41
41
|
const { text } = await generateText({
|
|
42
|
-
model: gateway('openai/gpt-
|
|
42
|
+
model: gateway('openai/gpt-6-astra'),
|
|
43
43
|
prompt: 'Hello world',
|
|
44
44
|
});
|
|
45
45
|
```
|
|
@@ -200,7 +200,7 @@ You can create language models using a provider instance. The first argument is
|
|
|
200
200
|
import { generateText } from 'ai';
|
|
201
201
|
|
|
202
202
|
const { text } = await generateText({
|
|
203
|
-
model: 'openai/gpt-
|
|
203
|
+
model: 'openai/gpt-6-astra',
|
|
204
204
|
prompt: 'Explain quantum computing in simple terms',
|
|
205
205
|
});
|
|
206
206
|
```
|
|
@@ -586,7 +586,7 @@ availableModels.models.forEach(model => {
|
|
|
586
586
|
|
|
587
587
|
// Use any discovered model with plain string
|
|
588
588
|
const { text } = await generateText({
|
|
589
|
-
model: availableModels.models[0].id, // e.g., 'openai/gpt-
|
|
589
|
+
model: availableModels.models[0].id, // e.g., 'openai/gpt-6-astra'
|
|
590
590
|
prompt: 'Hello world',
|
|
591
591
|
});
|
|
592
592
|
```
|
|
@@ -620,7 +620,7 @@ import { gateway, generateText } from 'ai';
|
|
|
620
620
|
|
|
621
621
|
// Make a request
|
|
622
622
|
const result = await generateText({
|
|
623
|
-
model: gateway('anthropic/claude-sonnet-
|
|
623
|
+
model: gateway('anthropic/claude-sonnet-5'),
|
|
624
624
|
prompt: 'Explain quantum entanglement briefly',
|
|
625
625
|
});
|
|
626
626
|
|
|
@@ -643,7 +643,7 @@ With `streamText`, you can capture the generation ID from the first chunk via `s
|
|
|
643
643
|
import { gateway, streamText } from 'ai';
|
|
644
644
|
|
|
645
645
|
const result = streamText({
|
|
646
|
-
model: gateway('anthropic/claude-sonnet-
|
|
646
|
+
model: gateway('anthropic/claude-sonnet-5'),
|
|
647
647
|
prompt: 'Explain quantum entanglement briefly',
|
|
648
648
|
});
|
|
649
649
|
|
|
@@ -673,7 +673,7 @@ a tool is running:
|
|
|
673
673
|
import { gateway, generateText } from 'ai';
|
|
674
674
|
|
|
675
675
|
await generateText({
|
|
676
|
-
model: gateway('anthropic/claude-sonnet-
|
|
676
|
+
model: gateway('anthropic/claude-sonnet-5'),
|
|
677
677
|
prompt: 'Explain quantum entanglement briefly',
|
|
678
678
|
onLanguageModelCallEnd({ providerMetadata }) {
|
|
679
679
|
const generationId = providerMetadata?.gateway?.generationId as
|
|
@@ -720,7 +720,7 @@ It returns a `GatewayGenerationInfo` object with the following fields:
|
|
|
720
720
|
import { generateText } from 'ai';
|
|
721
721
|
|
|
722
722
|
const { text } = await generateText({
|
|
723
|
-
model: 'anthropic/claude-sonnet-
|
|
723
|
+
model: 'anthropic/claude-sonnet-5',
|
|
724
724
|
prompt: 'Write a haiku about programming',
|
|
725
725
|
});
|
|
726
726
|
|
|
@@ -733,7 +733,7 @@ console.log(text);
|
|
|
733
733
|
import { streamText } from 'ai';
|
|
734
734
|
|
|
735
735
|
const { textStream } = await streamText({
|
|
736
|
-
model: 'openai/gpt-
|
|
736
|
+
model: 'openai/gpt-6-astra',
|
|
737
737
|
prompt: 'Explain the benefits of serverless architecture',
|
|
738
738
|
});
|
|
739
739
|
|
|
@@ -749,7 +749,7 @@ import { generateText, tool } from 'ai';
|
|
|
749
749
|
import { z } from 'zod';
|
|
750
750
|
|
|
751
751
|
const { text } = await generateText({
|
|
752
|
-
model: '
|
|
752
|
+
model: 'spacexai/grok-4.7',
|
|
753
753
|
prompt: 'What is the weather like in San Francisco?',
|
|
754
754
|
tools: {
|
|
755
755
|
getWeather: tool({
|
|
@@ -775,7 +775,7 @@ import { generateText, isStepCount } from 'ai';
|
|
|
775
775
|
import { openai } from '@ai-sdk/openai';
|
|
776
776
|
|
|
777
777
|
const result = await generateText({
|
|
778
|
-
model: 'openai/gpt-
|
|
778
|
+
model: 'openai/gpt-6-luna',
|
|
779
779
|
prompt: 'What is the Vercel AI Gateway?',
|
|
780
780
|
stopWhen: isStepCount(10),
|
|
781
781
|
tools: {
|
|
@@ -1207,7 +1207,7 @@ import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
|
1207
1207
|
import { generateText } from 'ai';
|
|
1208
1208
|
|
|
1209
1209
|
const { text } = await generateText({
|
|
1210
|
-
model: 'openai/gpt-
|
|
1210
|
+
model: 'openai/gpt-6-astra',
|
|
1211
1211
|
prompt: 'Summarize this document...',
|
|
1212
1212
|
providerOptions: {
|
|
1213
1213
|
gateway: {
|
|
@@ -1249,7 +1249,7 @@ The `getSpendReport()` method accepts the following parameters:
|
|
|
1249
1249
|
- **groupBy** _string_ - Aggregation dimension: `'day'` (default), `'user'`, `'model'`, `'tag'`, `'provider'`, or `'credential_type'`
|
|
1250
1250
|
- **datePart** _string_ - Time granularity when `groupBy` is `'day'`: `'day'` or `'hour'`
|
|
1251
1251
|
- **userId** _string_ - Filter to a specific user
|
|
1252
|
-
- **model** _string_ - Filter to a specific model (e.g. `'anthropic/claude-sonnet-
|
|
1252
|
+
- **model** _string_ - Filter to a specific model (e.g. `'anthropic/claude-sonnet-5'`)
|
|
1253
1253
|
- **provider** _string_ - Filter to a specific provider (e.g. `'anthropic'`)
|
|
1254
1254
|
- **credentialType** _string_ - Filter by `'byok'` or `'system'` credentials
|
|
1255
1255
|
- **tags** _string[]_ - Filter to requests matching these tags
|
|
@@ -1308,7 +1308,7 @@ import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
|
1308
1308
|
import { generateText } from 'ai';
|
|
1309
1309
|
|
|
1310
1310
|
const { text } = await generateText({
|
|
1311
|
-
model: 'anthropic/claude-sonnet-
|
|
1311
|
+
model: 'anthropic/claude-sonnet-5',
|
|
1312
1312
|
prompt: 'Explain quantum computing',
|
|
1313
1313
|
providerOptions: {
|
|
1314
1314
|
gateway: {
|
|
@@ -1350,7 +1350,7 @@ The following gateway provider options are available:
|
|
|
1350
1350
|
|
|
1351
1351
|
Specifies fallback models to use when the primary model fails or is unavailable. The gateway will try the primary model first (specified in the `model` parameter), then try each model in this array in order until one succeeds.
|
|
1352
1352
|
|
|
1353
|
-
Example: `models: ['openai/gpt-5.4-nano', 'gemini-3-flash
|
|
1353
|
+
Example: `models: ['openai/gpt-5.4-nano', 'google/gemini-3.8-flash']` will try the fallback models in order if the primary model fails.
|
|
1354
1354
|
|
|
1355
1355
|
- **user** _string_
|
|
1356
1356
|
|
|
@@ -1390,7 +1390,7 @@ The following gateway provider options are available:
|
|
|
1390
1390
|
|
|
1391
1391
|
The unique identifier for the entity against which quota is tracked. Used for quota management and enforcement purposes.
|
|
1392
1392
|
|
|
1393
|
-
- **has** _Array<'implicit-caching' | 'reasoning' | 'tool-use' | 'vision'
|
|
1393
|
+
- **has** _Array<'implicit-caching' | 'reasoning' | 'tool-use' | 'vision' | `quantization:${string}` | `!quantization:${string}`>_
|
|
1394
1394
|
|
|
1395
1395
|
Restricts routing to provider models that have all of the specified capabilities. Applies to both BYOK and system credentials, since the capability is a property of the model rather than the credential. If no provider model for the requested model satisfies the capabilities, the request fails. Unsupported values are rejected.
|
|
1396
1396
|
|
|
@@ -1400,7 +1400,9 @@ The following gateway provider options are available:
|
|
|
1400
1400
|
- `'tool-use'` — models that support tool calling.
|
|
1401
1401
|
- `'vision'` — models that accept image input.
|
|
1402
1402
|
|
|
1403
|
-
|
|
1403
|
+
Weight-format conditions route on the serving provider's recorded weight format: `'quantization:fp8'` requires a provider serving fp8 weights, while `'!quantization:fp8'` excludes fp8 providers. A negated condition also matches providers whose weight format is not recorded, so exclusions remain usable while the catalog is populated.
|
|
1404
|
+
|
|
1405
|
+
Example: `has: ['implicit-caching']` will only route to models that support implicit caching. Example: `has: ['vision']` will only route to models that accept image input. Example: `has: ['tool-use']` will only route to models that support tool calling. Example: `has: ['!quantization:fp8']` will only route to providers that do not serve fp8 weights.
|
|
1404
1406
|
|
|
1405
1407
|
- **providerTimeouts** _object_
|
|
1406
1408
|
|
|
@@ -1426,7 +1428,7 @@ The following gateway provider options are available:
|
|
|
1426
1428
|
import { generateText } from 'ai';
|
|
1427
1429
|
|
|
1428
1430
|
const { text } = await generateText({
|
|
1429
|
-
model: 'openai/gpt-
|
|
1431
|
+
model: 'openai/gpt-6-luna',
|
|
1430
1432
|
prompt: 'Hello',
|
|
1431
1433
|
providerOptions: {
|
|
1432
1434
|
gateway: {
|
|
@@ -1443,7 +1445,7 @@ import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
|
1443
1445
|
import { generateText } from 'ai';
|
|
1444
1446
|
|
|
1445
1447
|
const { text } = await generateText({
|
|
1446
|
-
model: 'anthropic/claude-sonnet-
|
|
1448
|
+
model: 'anthropic/claude-sonnet-5',
|
|
1447
1449
|
prompt: 'Write a haiku about programming',
|
|
1448
1450
|
providerOptions: {
|
|
1449
1451
|
gateway: {
|
|
@@ -1463,11 +1465,11 @@ import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
|
1463
1465
|
import { generateText } from 'ai';
|
|
1464
1466
|
|
|
1465
1467
|
const { text } = await generateText({
|
|
1466
|
-
model: 'openai/gpt-
|
|
1468
|
+
model: 'openai/gpt-6-astra', // Primary model
|
|
1467
1469
|
prompt: 'Write a TypeScript haiku',
|
|
1468
1470
|
providerOptions: {
|
|
1469
1471
|
gateway: {
|
|
1470
|
-
models: ['openai/gpt-5.4-nano', 'gemini-3-flash
|
|
1472
|
+
models: ['openai/gpt-5.4-nano', 'gemini-3.8-flash'], // Fallback models
|
|
1471
1473
|
} satisfies GatewayProviderOptions,
|
|
1472
1474
|
},
|
|
1473
1475
|
});
|
|
@@ -1488,7 +1490,7 @@ import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
|
1488
1490
|
import { generateText } from 'ai';
|
|
1489
1491
|
|
|
1490
1492
|
const { text } = await generateText({
|
|
1491
|
-
model: 'anthropic/claude-sonnet-
|
|
1493
|
+
model: 'anthropic/claude-sonnet-5',
|
|
1492
1494
|
prompt: 'Analyze this sensitive document...',
|
|
1493
1495
|
providerOptions: {
|
|
1494
1496
|
gateway: {
|
|
@@ -1507,7 +1509,7 @@ import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
|
1507
1509
|
import { generateText } from 'ai';
|
|
1508
1510
|
|
|
1509
1511
|
const { text } = await generateText({
|
|
1510
|
-
model: 'anthropic/claude-sonnet-
|
|
1512
|
+
model: 'anthropic/claude-sonnet-5',
|
|
1511
1513
|
prompt: 'Analyze this proprietary business data...',
|
|
1512
1514
|
providerOptions: {
|
|
1513
1515
|
gateway: {
|
|
@@ -1526,7 +1528,7 @@ import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
|
1526
1528
|
import { generateText } from 'ai';
|
|
1527
1529
|
|
|
1528
1530
|
const { text } = await generateText({
|
|
1529
|
-
model: 'anthropic/claude-sonnet-
|
|
1531
|
+
model: 'anthropic/claude-sonnet-5',
|
|
1530
1532
|
prompt: 'Summarize this report...',
|
|
1531
1533
|
providerOptions: {
|
|
1532
1534
|
gateway: {
|
|
@@ -1538,14 +1540,14 @@ const { text } = await generateText({
|
|
|
1538
1540
|
|
|
1539
1541
|
#### Filtering by Model Capability
|
|
1540
1542
|
|
|
1541
|
-
Set `has` to restrict routing to provider models that have the specified capabilities. This applies to both BYOK and system credentials, since the capability is a property of the model rather than the credential. `'implicit-caching'` limits routing to models that perform automatic prompt caching, `'vision'` limits routing to models that accept image input, `'reasoning'` limits routing to models that support reasoning, and `'tool-use'` limits routing to models that support tool calling. If no provider model for the requested model satisfies the capabilities, the request fails.
|
|
1543
|
+
Set `has` to restrict routing to provider models that have the specified capabilities. This applies to both BYOK and system credentials, since the capability is a property of the model rather than the credential. `'implicit-caching'` limits routing to models that perform automatic prompt caching, `'vision'` limits routing to models that accept image input, `'reasoning'` limits routing to models that support reasoning, and `'tool-use'` limits routing to models that support tool calling. Weight-format conditions (`'quantization:fp8'` to require, `'!quantization:fp8'` to exclude) limit routing by the serving provider's recorded weight format. If no provider model for the requested model satisfies the capabilities, the request fails.
|
|
1542
1544
|
|
|
1543
1545
|
```ts
|
|
1544
1546
|
import type { GatewayProviderOptions } from '@ai-sdk/gateway';
|
|
1545
1547
|
import { generateText } from 'ai';
|
|
1546
1548
|
|
|
1547
1549
|
const { text } = await generateText({
|
|
1548
|
-
model: 'openai/gpt-
|
|
1550
|
+
model: 'openai/gpt-6-astra',
|
|
1549
1551
|
prompt: 'Summarize this report...',
|
|
1550
1552
|
providerOptions: {
|
|
1551
1553
|
gateway: {
|
|
@@ -1564,7 +1566,7 @@ import { generateText } from 'ai';
|
|
|
1564
1566
|
import fs from 'node:fs';
|
|
1565
1567
|
|
|
1566
1568
|
const { text } = await generateText({
|
|
1567
|
-
model: 'anthropic/claude-sonnet-
|
|
1569
|
+
model: 'anthropic/claude-sonnet-5',
|
|
1568
1570
|
messages: [
|
|
1569
1571
|
{
|
|
1570
1572
|
role: 'user',
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ai-sdk/gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "4.0.
|
|
4
|
+
"version": "4.0.92",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
7
7
|
"sideEffects": false,
|
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
},
|
|
32
32
|
"dependencies": {
|
|
33
33
|
"@ai-sdk/provider": "4.0.18",
|
|
34
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
34
|
+
"@ai-sdk/provider-utils": "5.0.47",
|
|
35
35
|
"@vercel/oidc": "3.2.0"
|
|
36
36
|
},
|
|
37
37
|
"devDependencies": {
|
|
@@ -29,6 +29,7 @@ export type GatewayModelId =
|
|
|
29
29
|
| 'alibaba/qwen3.8-flash'
|
|
30
30
|
| 'alibaba/qwen3.8-max'
|
|
31
31
|
| 'alibaba/qwen3.8-max-0902'
|
|
32
|
+
| 'alibaba/qwen3.8-max-prime'
|
|
32
33
|
| 'alibaba/qwen3.8-omni-flash'
|
|
33
34
|
| 'amazon/nova-2-lite'
|
|
34
35
|
| 'amazon/nova-lite'
|
|
@@ -47,6 +48,7 @@ export type GatewayModelId =
|
|
|
47
48
|
| 'anthropic/claude-opus-5'
|
|
48
49
|
| 'anthropic/claude-opus-5-fast'
|
|
49
50
|
| 'anthropic/claude-opus-5.5'
|
|
51
|
+
| 'anthropic/claude-opus-5.5-fast'
|
|
50
52
|
| 'anthropic/claude-sonnet-4'
|
|
51
53
|
| 'anthropic/claude-sonnet-4.5'
|
|
52
54
|
| 'anthropic/claude-sonnet-4.6'
|
|
@@ -67,6 +69,7 @@ export type GatewayModelId =
|
|
|
67
69
|
| 'deepseek/deepseek-v4-pro'
|
|
68
70
|
| 'deepseek/deepseek-v4-pro-0813'
|
|
69
71
|
| 'deepseek/deepseek-v4.1-flash'
|
|
72
|
+
| 'fireworks/ember-1'
|
|
70
73
|
| 'google/gemini-2.5-flash'
|
|
71
74
|
| 'google/gemini-2.5-flash-image'
|
|
72
75
|
| 'google/gemini-2.5-flash-lite'
|
|
@@ -95,7 +98,6 @@ export type GatewayModelId =
|
|
|
95
98
|
| 'inclusionai/ling-3.0-flash-sante'
|
|
96
99
|
| 'inclusionai/ling-3.0-flash-sante-free'
|
|
97
100
|
| 'inclusionai/ling-3.0-flash-vl'
|
|
98
|
-
| 'inclusionai/ling-3.0-flash-vl-free'
|
|
99
101
|
| 'inference-net/schematron-v2-small'
|
|
100
102
|
| 'inference-net/schematron-v2-turbo'
|
|
101
103
|
| 'interfaze/interfaze-beta'
|
|
@@ -16,11 +16,24 @@ export type GatewayProviderOptions = {
|
|
|
16
16
|
disallowPromptTraining?: boolean;
|
|
17
17
|
|
|
18
18
|
/**
|
|
19
|
-
* Restrict routing to models that
|
|
20
|
-
*
|
|
21
|
-
* `'
|
|
19
|
+
* Restrict routing to provider models that satisfy every given entry.
|
|
20
|
+
*
|
|
21
|
+
* Entries are capability tags (`'implicit-caching'`, `'reasoning'`,
|
|
22
|
+
* `'tool-use'`, `'vision'`) or weight-format conditions: `'quantization:fp8'`
|
|
23
|
+
* requires the serving provider to report that weight format,
|
|
24
|
+
* `'!quantization:fp8'` excludes it (providers with no recorded format still
|
|
25
|
+
* pass an exclusion). Format values are an open space but must match
|
|
26
|
+
* `[a-zA-Z0-9._-]{1,32}` and compare case-insensitively; unknown capability
|
|
27
|
+
* names are rejected by the Gateway with a 400.
|
|
22
28
|
*/
|
|
23
|
-
has?: Array<
|
|
29
|
+
has?: Array<
|
|
30
|
+
| 'implicit-caching'
|
|
31
|
+
| 'reasoning'
|
|
32
|
+
| 'tool-use'
|
|
33
|
+
| 'vision'
|
|
34
|
+
| `quantization:${string}`
|
|
35
|
+
| `!quantization:${string}`
|
|
36
|
+
>;
|
|
24
37
|
|
|
25
38
|
/**
|
|
26
39
|
* Idempotency key for `experimental_startBatch`: retries with the same
|