@mastra/mcp-docs-server 1.3.0-alpha.2 → 1.3.0-alpha.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.docs/models/gateways/netlify.md +2 -1
- package/.docs/models/gateways/openrouter.md +4 -3
- package/.docs/models/gateways/vercel.md +2 -1
- package/.docs/models/index.md +1 -1
- package/.docs/models/providers/above.md +12 -11
- package/.docs/models/providers/cortecs.md +5 -2
- package/.docs/models/providers/deepinfra.md +4 -2
- package/.docs/models/providers/edenai.md +6 -4
- package/.docs/models/providers/empiriolabs.md +3 -3
- package/.docs/models/providers/fireworks-ai.md +2 -1
- package/.docs/models/providers/google.md +3 -3
- package/.docs/models/providers/kilo.md +13 -12
- package/.docs/models/providers/llmgateway-providers.md +3 -1
- package/.docs/models/providers/nano-gpt.md +5 -3
- package/.docs/models/providers/opencode-go.md +2 -1
- package/.docs/models/providers/snowflake-cortex.md +4 -4
- package/.docs/models/providers/tempr.md +12 -2
- package/.docs/models/providers/togetherai.md +2 -2
- package/.docs/models/providers/vivgrid.md +4 -1
- package/.docs/reference/agents/durable-agent.md +2 -0
- package/.docs/reference/agents/inngest-agent.md +2 -0
- package/.docs/reference/observability/tracing/trace-query.md +17 -1
- package/.docs/reference/processors/memory-input-filter.md +1 -1
- package/.docs/reference/processors/token-limiter-processor.md +15 -10
- package/package.json +3 -3
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Netlify
|
|
6
6
|
|
|
7
|
-
Netlify AI Gateway provides unified access to multiple providers with built-in caching and observability. Access
|
|
7
|
+
Netlify AI Gateway provides unified access to multiple providers with built-in caching and observability. Access 265 models through Mastra's model router.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Netlify documentation](https://docs.netlify.com/build/ai-gateway/overview/).
|
|
10
10
|
|
|
@@ -153,6 +153,7 @@ ANTHROPIC_API_KEY=ant-...
|
|
|
153
153
|
| `openrouter/deepseek/deepseek-v4-pro` |
|
|
154
154
|
| `openrouter/deepseek/deepseek-v4-pro-0813` |
|
|
155
155
|
| `openrouter/deepseek/deepseek-v4.1-flash` |
|
|
156
|
+
| `openrouter/fireworks/ember-1` |
|
|
156
157
|
| `openrouter/google/gemma-2-27b-it` |
|
|
157
158
|
| `openrouter/google/gemma-3-12b-it` |
|
|
158
159
|
| `openrouter/google/gemma-3-27b-it` |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# OpenRouter
|
|
6
6
|
|
|
7
|
-
OpenRouter aggregates models from multiple providers with enhanced features like rate limiting and failover. Access
|
|
7
|
+
OpenRouter aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 384 models through Mastra's model router.
|
|
8
8
|
|
|
9
9
|
Learn more in the [OpenRouter documentation](https://openrouter.ai/models).
|
|
10
10
|
|
|
@@ -211,9 +211,7 @@ ANTHROPIC_API_KEY=ant-...
|
|
|
211
211
|
| `moonshotai/kimi-k3` |
|
|
212
212
|
| `morph/morph-v3-fast` |
|
|
213
213
|
| `morph/morph-v3-large` |
|
|
214
|
-
| `nex-agi/nex-n2.5-mini` |
|
|
215
214
|
| `nex-agi/nex-n2.5-mini:free` |
|
|
216
|
-
| `nex-agi/nex-n2.5-pro` |
|
|
217
215
|
| `nex-agi/nex-n2.5-pro:free` |
|
|
218
216
|
| `nousresearch/hermes-3-llama-3.1-405b` |
|
|
219
217
|
| `nousresearch/hermes-3-llama-3.1-70b` |
|
|
@@ -359,6 +357,7 @@ ANTHROPIC_API_KEY=ant-...
|
|
|
359
357
|
| `qwen/qwen3.8-27b:free` |
|
|
360
358
|
| `qwen/qwen3.8-flash` |
|
|
361
359
|
| `qwen/qwen3.8-max-0902` |
|
|
360
|
+
| `qwen/qwen3.8-max-prime` |
|
|
362
361
|
| `qwen/qwen3.8-omni-flash` |
|
|
363
362
|
| `rekaai/reka-edge` |
|
|
364
363
|
| `rekaai/reka-flash-3` |
|
|
@@ -371,6 +370,7 @@ ANTHROPIC_API_KEY=ant-...
|
|
|
371
370
|
| `sao10k/l3-lunaris-8b` |
|
|
372
371
|
| `sao10k/l3.1-euryale-70b` |
|
|
373
372
|
| `sao10k/l3.3-euryale-70b` |
|
|
373
|
+
| `stealth/space-bunny-alpha` |
|
|
374
374
|
| `stepfun/step-3.5-flash` |
|
|
375
375
|
| `stepfun/step-3.7-flash` |
|
|
376
376
|
| `tencent/hunyuan-a13b-instruct` |
|
|
@@ -420,4 +420,5 @@ ANTHROPIC_API_KEY=ant-...
|
|
|
420
420
|
| `z-ai/glm-5.3` |
|
|
421
421
|
| `z-ai/glm-5.3-flash` |
|
|
422
422
|
| `z-ai/glm-5.3-flashx` |
|
|
423
|
+
| `z-ai/glm-5.3-prime` |
|
|
423
424
|
| `z-ai/glm-5v-turbo` |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Vercel
|
|
6
6
|
|
|
7
|
-
Vercel aggregates models from multiple providers with enhanced features like rate limiting and failover. Access
|
|
7
|
+
Vercel aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 388 models through Mastra's model router.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Vercel documentation](https://ai-sdk.dev/providers/ai-sdk-providers).
|
|
10
10
|
|
|
@@ -71,6 +71,7 @@ ANTHROPIC_API_KEY=ant-...
|
|
|
71
71
|
| `alibaba/qwen3.8-flash` |
|
|
72
72
|
| `alibaba/qwen3.8-max` |
|
|
73
73
|
| `alibaba/qwen3.8-max-0902` |
|
|
74
|
+
| `alibaba/qwen3.8-max-prime` |
|
|
74
75
|
| `alibaba/qwen3.8-omni-flash` |
|
|
75
76
|
| `alibaba/wan-v2.5-t2v-preview` |
|
|
76
77
|
| `alibaba/wan-v2.6-i2v` |
|
package/.docs/models/index.md
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Model Providers
|
|
6
6
|
|
|
7
|
-
Mastra provides a unified interface for working with LLMs across multiple providers, giving you access to
|
|
7
|
+
Mastra provides a unified interface for working with LLMs across multiple providers, giving you access to 7614 models from 210 providers through a single API.
|
|
8
8
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# above.dev
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 9 above.dev models through Mastra's model router. Authentication is handled automatically using the `ABOVE_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [above.dev documentation](https://above.dev/docs).
|
|
10
10
|
|
|
@@ -36,16 +36,17 @@ for await (const chunk of stream) {
|
|
|
36
36
|
|
|
37
37
|
## Models
|
|
38
38
|
|
|
39
|
-
| Model
|
|
40
|
-
|
|
|
41
|
-
| `above/deepseek-v4-flash`
|
|
42
|
-
| `above/deepseek-v4-
|
|
43
|
-
| `above/
|
|
44
|
-
| `above/glm-5.2`
|
|
45
|
-
| `above/glm-5.
|
|
46
|
-
| `above/
|
|
47
|
-
| `above/mimo-v2.
|
|
48
|
-
| `above/
|
|
39
|
+
| Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
|
|
40
|
+
| -------------------------------- | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
|
|
41
|
+
| `above/deepseek-v4-flash` | 1.0M | | | | | | $0.17 | $0.66 |
|
|
42
|
+
| `above/deepseek-v4-pro` | 1.0M | | | | | | $0.73 | $2 |
|
|
43
|
+
| `above/glm-5.2` | 1.0M | | | | | | $2 | $5 |
|
|
44
|
+
| `above/glm-5.2-fast` | 1.0M | | | | | | $2 | $7 |
|
|
45
|
+
| `above/glm-5.3-flash` | 1.0M | | | | | | $0.17 | $0.55 |
|
|
46
|
+
| `above/mimo-v2.6-flash` | 1.0M | | | | | | $0.17 | $0.34 |
|
|
47
|
+
| `above/mimo-v2.6-pro` | 1.0M | | | | | | $0.51 | $1 |
|
|
48
|
+
| `above/mimo-v2.6-pro-ultraspeed` | 1.0M | | | | | | $5 | $10 |
|
|
49
|
+
| `above/qwen3.8-max` | 1.0M | | | | | | $2 | $7 |
|
|
49
50
|
|
|
50
51
|
Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
|
|
51
52
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Cortecs
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 109 Cortecs models through Mastra's model router. Authentication is handled automatically using the `CORTECS_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Cortecs documentation](https://cortecs.ai).
|
|
10
10
|
|
|
@@ -43,6 +43,7 @@ for await (const chunk of stream) {
|
|
|
43
43
|
| `cortecs/claude-4-6-sonnet` | 1.0M | | | | | | $3 | $16 |
|
|
44
44
|
| `cortecs/claude-haiku-4-5` | 200K | | | | | | $1.00 | $5 |
|
|
45
45
|
| `cortecs/claude-opus-5` | 1.0M | | | | | | $6 | $27 |
|
|
46
|
+
| `cortecs/claude-opus-5.5` | 1.0M | | | | | | $4 | $22 |
|
|
46
47
|
| `cortecs/claude-opus4-5` | 200K | | | | | | $5 | $27 |
|
|
47
48
|
| `cortecs/claude-opus4-6` | 1.0M | | | | | | $5 | $27 |
|
|
48
49
|
| `cortecs/claude-opus4-7` | 1.0M | | | | | | $5 | $27 |
|
|
@@ -87,8 +88,10 @@ for await (const chunk of stream) {
|
|
|
87
88
|
| `cortecs/gpt-5.1` | 400K | | | | | | $1 | $11 |
|
|
88
89
|
| `cortecs/gpt-5.4` | 1.1M | | | | | | $3 | $15 |
|
|
89
90
|
| `cortecs/gpt-5.6-luna` | 1.1M | | | | | | $0.22 | $1 |
|
|
90
|
-
| `cortecs/gpt-5.6-sol` | 1.1M | | | | | | $
|
|
91
|
+
| `cortecs/gpt-5.6-sol` | 1.1M | | | | | | $4 | $22 |
|
|
91
92
|
| `cortecs/gpt-5.6-terra` | 1.1M | | | | | | $2 | $13 |
|
|
93
|
+
| `cortecs/gpt-6-luna` | 1.1M | | | | | | $0.12 | $0.60 |
|
|
94
|
+
| `cortecs/gpt-6-sol` | 1.1M | | | | | | $2 | $12 |
|
|
92
95
|
| `cortecs/gpt-oss-120b` | 131K | | | | | | $0.09 | $0.45 |
|
|
93
96
|
| `cortecs/gpt-oss-20b` | 131K | | | | | | $0.04 | $0.17 |
|
|
94
97
|
| `cortecs/gpt-oss-safeguard-120b` | 128K | | | | | | $0.18 | $0.70 |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Deep Infra
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 70 Deep Infra models through Mastra's model router. Authentication is handled automatically using the `DEEPINFRA_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Deep Infra documentation](https://deepinfra.com/models).
|
|
10
10
|
|
|
@@ -91,11 +91,13 @@ for await (const chunk of stream) {
|
|
|
91
91
|
| `deepinfra/thinkingmachines/Inkling-Small` | 524K | | | | | | $0.45 | $1 |
|
|
92
92
|
| `deepinfra/XiaomiMiMo/MiMo-V2.5` | 262K | | | | | | $0.14 | $0.28 |
|
|
93
93
|
| `deepinfra/XiaomiMiMo/MiMo-V2.5-Pro` | 1.0M | | | | | | $1 | $3 |
|
|
94
|
+
| `deepinfra/XiaomiMiMo/MiMo-V2.6-Flash` | 1.0M | | | | | | $0.14 | $0.28 |
|
|
95
|
+
| `deepinfra/XiaomiMiMo/MiMo-V2.6-Pro` | 1.0M | | | | | | $0.43 | $0.87 |
|
|
94
96
|
| `deepinfra/zai-org/GLM-4.6` | 203K | | | | | | $0.50 | $2 |
|
|
95
97
|
| `deepinfra/zai-org/GLM-4.7` | 203K | | | | | | $0.40 | $2 |
|
|
96
98
|
| `deepinfra/zai-org/GLM-5.1` | 203K | | | | | | $1 | $4 |
|
|
97
99
|
| `deepinfra/zai-org/GLM-5.2` | 1.0M | | | | | | $0.75 | $2 |
|
|
98
|
-
| `deepinfra/zai-org/GLM-5.3` | 1.0M | | | | | | $
|
|
100
|
+
| `deepinfra/zai-org/GLM-5.3` | 1.0M | | | | | | $0.90 | $4 |
|
|
99
101
|
| `deepinfra/zai-org/GLM-5.3-Flash` | 1.0M | | | | | | $0.15 | $0.50 |
|
|
100
102
|
|
|
101
103
|
Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Eden AI
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 285 Eden AI models through Mastra's model router. Authentication is handled automatically using the `EDENAI_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Eden AI documentation](https://docs.edenai.co).
|
|
10
10
|
|
|
@@ -136,11 +136,13 @@ for await (const chunk of stream) {
|
|
|
136
136
|
| `edenai/fireworks_ai/accounts/fireworks/models/inkling` | 1.0M | | | | | | $1 | $4 |
|
|
137
137
|
| `edenai/fireworks_ai/accounts/fireworks/models/muse-glimmer-30b` | 131K | | | | | | $0.35 | $2 |
|
|
138
138
|
| `edenai/fireworks_ai/gpt-oss-120b` | 131K | | | | | | $0.15 | $0.60 |
|
|
139
|
-
| `edenai/flexai/DeepSeek-V4-Flash-0731` | 1.0M | | | | | | $0.
|
|
140
|
-
| `edenai/flexai/gpt-oss-120b` | 131K | | | | | | $0.
|
|
141
|
-
| `edenai/flexai/gpt-oss-20b` | 131K | | | | | | $0.02 | $0.
|
|
139
|
+
| `edenai/flexai/DeepSeek-V4-Flash-0731` | 1.0M | | | | | | $0.06 | $0.18 |
|
|
140
|
+
| `edenai/flexai/gpt-oss-120b` | 131K | | | | | | $0.03 | $0.17 |
|
|
141
|
+
| `edenai/flexai/gpt-oss-20b` | 131K | | | | | | $0.02 | $0.10 |
|
|
142
142
|
| `edenai/flexai/Muse-Glimmer-30B` | 131K | | | | | | $0.30 | $1 |
|
|
143
143
|
| `edenai/flexai/Step-3.7-Flash` | 262K | | | | | | $0.20 | $1 |
|
|
144
|
+
| `edenai/google/deep-research-max-preview-04-2026` | 131K | | | | | | $2 | $12 |
|
|
145
|
+
| `edenai/google/deep-research-preview-04-2026` | 131K | | | | | | $2 | $12 |
|
|
144
146
|
| `edenai/google/gemini-2.5-flash-image` | 33K | | | | | | $0.30 | $3 |
|
|
145
147
|
| `edenai/google/gemini-3-flash-preview` | 1.0M | | | | | | $0.50 | $3 |
|
|
146
148
|
| `edenai/google/gemini-3-pro-image` | 66K | | | | | | $2 | $12 |
|
|
@@ -62,9 +62,9 @@ for await (const chunk of stream) {
|
|
|
62
62
|
| `empiriolabs/kimi-k3` | 1.0M | | | | | | $3 | $15 |
|
|
63
63
|
| `empiriolabs/mimo-v2-5` | 1.0M | | | | | | $0.70 | $1 |
|
|
64
64
|
| `empiriolabs/mimo-v2-5-pro` | 1.0M | | | | | | $2 | $4 |
|
|
65
|
-
| `empiriolabs/mimo-v2-6-flash` | 1.0M | | | | | | $0.
|
|
66
|
-
| `empiriolabs/mimo-v2-6-pro` | 1.0M | | | | | | $
|
|
67
|
-
| `empiriolabs/mimo-v2-6-pro-ultraspeed` | 1.0M | | | | | | $
|
|
65
|
+
| `empiriolabs/mimo-v2-6-flash` | 1.0M | | | | | | $0.14 | $0.28 |
|
|
66
|
+
| `empiriolabs/mimo-v2-6-pro` | 1.0M | | | | | | $0.43 | $0.87 |
|
|
67
|
+
| `empiriolabs/mimo-v2-6-pro-ultraspeed` | 1.0M | | | | | | $4 | $9 |
|
|
68
68
|
| `empiriolabs/minimax-m2-7` | 200K | | | | | | $0.15 | $0.60 |
|
|
69
69
|
| `empiriolabs/minimax-m2-7-highspeed` | 200K | | | | | | $0.30 | $1 |
|
|
70
70
|
| `empiriolabs/minimax-m3` | 1.0M | | | | | | $0.23 | $0.90 |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Fireworks AI
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 34 Fireworks AI models through Mastra's model router. Authentication is handled automatically using the `FIREWORKS_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Fireworks AI documentation](https://fireworks.ai/docs/).
|
|
10
10
|
|
|
@@ -39,6 +39,7 @@ for await (const chunk of stream) {
|
|
|
39
39
|
| Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
|
|
40
40
|
| ----------------------------------------------------------------------- | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
|
|
41
41
|
| `fireworks-ai/accounts/fireworks/models/deepseek-v4p1-flash` | 1.0M | | | | | | $0.22 | $0.66 |
|
|
42
|
+
| `fireworks-ai/accounts/fireworks/models/ember-1` | 1.0M | | | | | | $3 | $15 |
|
|
42
43
|
| `fireworks-ai/accounts/fireworks/models/glm-5p3` | 1.0M | | | | | | $1 | $4 |
|
|
43
44
|
| `fireworks-ai/accounts/fireworks/models/glm-5p3-flash` | 1.0M | | | | | | $0.15 | $0.50 |
|
|
44
45
|
| `fireworks-ai/accounts/fireworks/models/gpt-oss-120b` | 131K | | | | | | $0.15 | $0.60 |
|
|
@@ -39,7 +39,7 @@ for await (const chunk of stream) {
|
|
|
39
39
|
| ------------------------------------------------ | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
|
|
40
40
|
| `google/deep-research-max-preview-04-2026` | 131K | | | | | | $2 | $12 |
|
|
41
41
|
| `google/deep-research-preview-04-2026` | 131K | | | | | | $2 | $12 |
|
|
42
|
-
| `google/gemini-2.5-computer-use-preview-10-2025` |
|
|
42
|
+
| `google/gemini-2.5-computer-use-preview-10-2025` | 128K | | | | | | $1 | $10 |
|
|
43
43
|
| `google/gemini-2.5-flash` | 1.0M | | | | | | $0.30 | $3 |
|
|
44
44
|
| `google/gemini-2.5-flash-image` | 33K | | | | | | $0.30 | $30 |
|
|
45
45
|
| `google/gemini-2.5-flash-lite` | 1.0M | | | | | | $0.10 | $0.40 |
|
|
@@ -47,9 +47,9 @@ for await (const chunk of stream) {
|
|
|
47
47
|
| `google/gemini-2.5-pro` | 1.0M | | | | | | $1 | $10 |
|
|
48
48
|
| `google/gemini-2.5-pro-preview-tts` | 8K | | | | | | $1 | $20 |
|
|
49
49
|
| `google/gemini-3-flash-preview` | 1.0M | | | | | | $0.50 | $3 |
|
|
50
|
-
| `google/gemini-3-pro-image` |
|
|
50
|
+
| `google/gemini-3-pro-image` | 66K | | | | | | $2 | $120 |
|
|
51
51
|
| `google/gemini-3-pro-image-preview` | 131K | | | | | | $2 | $120 |
|
|
52
|
-
| `google/gemini-3.1-flash-image` |
|
|
52
|
+
| `google/gemini-3.1-flash-image` | 131K | | | | | | $0.50 | $60 |
|
|
53
53
|
| `google/gemini-3.1-flash-image-preview` | 66K | | | | | | $0.50 | $60 |
|
|
54
54
|
| `google/gemini-3.1-flash-lite` | 1.0M | | | | | | $0.25 | $2 |
|
|
55
55
|
| `google/gemini-3.1-flash-lite-image` | 66K | | | | | | $0.25 | $30 |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Kilo Gateway
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 391 Kilo Gateway models through Mastra's model router. Authentication is handled automatically using the `KILO_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Kilo Gateway documentation](https://kilo.ai).
|
|
10
10
|
|
|
@@ -42,8 +42,8 @@ for await (const chunk of stream) {
|
|
|
42
42
|
| `kilo/~anthropic/claude-haiku-latest` | 200K | | | | | | $1 | $5 |
|
|
43
43
|
| `kilo/~anthropic/claude-opus-latest` | 1.0M | | | | | | $4 | $20 |
|
|
44
44
|
| `kilo/~anthropic/claude-sonnet-latest` | 1.0M | | | | | | $2 | $10 |
|
|
45
|
-
| `kilo/~deepseek/deepseek-flash-latest` | 1.0M | | | | | | $0.
|
|
46
|
-
| `kilo/~deepseek/deepseek-pro-latest` | 1.0M | | | | | | $0.
|
|
45
|
+
| `kilo/~deepseek/deepseek-flash-latest` | 1.0M | | | | | | $0.05 | $0.25 |
|
|
46
|
+
| `kilo/~deepseek/deepseek-pro-latest` | 1.0M | | | | | | $0.39 | $3 |
|
|
47
47
|
| `kilo/~deepseek/deepseek-v4-flash-latest` | 1.0M | | | | | | $0.04 | $0.55 |
|
|
48
48
|
| `kilo/~google/gemini-flash-latest` | 1.0M | | | | | | $0.75 | $4 |
|
|
49
49
|
| `kilo/~google/gemini-pro-latest` | 1.0M | | | | | | $2 | $12 |
|
|
@@ -54,8 +54,8 @@ for await (const chunk of stream) {
|
|
|
54
54
|
| `kilo/~openai/gpt-sol-latest` | 1.1M | | | | | | $2 | $10 |
|
|
55
55
|
| `kilo/~openai/gpt-terra-latest` | 1.1M | | | | | | $2 | $12 |
|
|
56
56
|
| `kilo/~x-ai/grok-latest` | 500K | | | | | | $2 | $5 |
|
|
57
|
-
| `kilo/~z-ai/glm-flash-latest` | 1.0M | | | | | | $0.
|
|
58
|
-
| `kilo/~z-ai/glm-latest` | 1.0M | | | | | | $0.56 | $
|
|
57
|
+
| `kilo/~z-ai/glm-flash-latest` | 1.0M | | | | | | $0.04 | $0.14 |
|
|
58
|
+
| `kilo/~z-ai/glm-latest` | 1.0M | | | | | | $0.56 | $3 |
|
|
59
59
|
| `kilo/aion-labs/aion-2.0` | 131K | | | | | | $0.80 | $2 |
|
|
60
60
|
| `kilo/aion-labs/aion-3.0` | 131K | | | | | | $3 | $6 |
|
|
61
61
|
| `kilo/aion-labs/aion-3.0-mini` | 131K | | | | | | $0.70 | $1 |
|
|
@@ -116,7 +116,7 @@ for await (const chunk of stream) {
|
|
|
116
116
|
| `kilo/deepseek/deepseek-v4.1-flash` | 1.0M | | | | | | $0.30 | $1 |
|
|
117
117
|
| `kilo/dots-studio/dots-3-note-preview:free` | 512K | | | | | | — | — |
|
|
118
118
|
| `kilo/google/gemini-2.5-flash` | 1.0M | | | | | | $0.30 | $3 |
|
|
119
|
-
| `kilo/google/gemini-2.5-flash-image` | 33K | | | | | | $0.
|
|
119
|
+
| `kilo/google/gemini-2.5-flash-image` | 33K | | | | | | $0.15 | $1 |
|
|
120
120
|
| `kilo/google/gemini-2.5-flash-lite` | 1.0M | | | | | | $0.10 | $0.40 |
|
|
121
121
|
| `kilo/google/gemini-2.5-pro` | 1.0M | | | | | | $1 | $10 |
|
|
122
122
|
| `kilo/google/gemini-2.5-pro-preview` | 1.0M | | | | | | $1 | $10 |
|
|
@@ -127,7 +127,7 @@ for await (const chunk of stream) {
|
|
|
127
127
|
| `kilo/google/gemini-3.1-flash-image-preview` | 66K | | | | | | $0.50 | $3 |
|
|
128
128
|
| `kilo/google/gemini-3.1-flash-lite` | 1.0M | | | | | | $0.13 | $0.75 |
|
|
129
129
|
| `kilo/google/gemini-3.1-flash-lite-image` | 66K | | | | | | $0.25 | $2 |
|
|
130
|
-
| `kilo/google/gemini-3.1-flash-lite-preview` | 1.0M | | | | | | $0.
|
|
130
|
+
| `kilo/google/gemini-3.1-flash-lite-preview` | 1.0M | | | | | | $0.13 | $0.75 |
|
|
131
131
|
| `kilo/google/gemini-3.1-pro-preview` | 1.0M | | | | | | $1 | $6 |
|
|
132
132
|
| `kilo/google/gemini-3.1-pro-preview-customtools` | 1.0M | | | | | | $2 | $12 |
|
|
133
133
|
| `kilo/google/gemini-3.5-flash` | 1.0M | | | | | | $0.75 | $5 |
|
|
@@ -182,7 +182,7 @@ for await (const chunk of stream) {
|
|
|
182
182
|
| `kilo/microsoft/wizardlm-2-8x22b` | 66K | | | | | | $0.62 | $0.62 |
|
|
183
183
|
| `kilo/minimax/minimax-01` | 1.0M | | | | | | $0.20 | $1 |
|
|
184
184
|
| `kilo/minimax/minimax-m1` | 1.0M | | | | | | $0.40 | $2 |
|
|
185
|
-
| `kilo/minimax/minimax-m2` |
|
|
185
|
+
| `kilo/minimax/minimax-m2` | 197K | | | | | | $0.30 | $1 |
|
|
186
186
|
| `kilo/minimax/minimax-m2-her` | 66K | | | | | | $0.30 | $1 |
|
|
187
187
|
| `kilo/minimax/minimax-m2.1` | 205K | | | | | | $0.30 | $1 |
|
|
188
188
|
| `kilo/minimax/minimax-m2.5` | 200K | | | | | | $0.30 | $1 |
|
|
@@ -214,9 +214,7 @@ for await (const chunk of stream) {
|
|
|
214
214
|
| `kilo/moonshotai/kimi-k3` | 1.0M | | | | | | $3 | $15 |
|
|
215
215
|
| `kilo/morph/morph-v3-fast` | 82K | | | | | | $0.80 | $1 |
|
|
216
216
|
| `kilo/morph/morph-v3-large` | 262K | | | | | | $0.90 | $2 |
|
|
217
|
-
| `kilo/nex-agi/nex-n2.5-mini` | 262K | | | | | | $0.03 | $0.10 |
|
|
218
217
|
| `kilo/nex-agi/nex-n2.5-mini:free` | 262K | | | | | | — | — |
|
|
219
|
-
| `kilo/nex-agi/nex-n2.5-pro` | 262K | | | | | | $0.07 | $0.25 |
|
|
220
218
|
| `kilo/nex-agi/nex-n2.5-pro:free` | 262K | | | | | | — | — |
|
|
221
219
|
| `kilo/nousresearch/hermes-3-llama-3.1-405b` | 131K | | | | | | $1 | $1 |
|
|
222
220
|
| `kilo/nousresearch/hermes-3-llama-3.1-70b` | 131K | | | | | | $0.70 | $0.70 |
|
|
@@ -361,6 +359,7 @@ for await (const chunk of stream) {
|
|
|
361
359
|
| `kilo/qwen/qwen3.8-27b:free` | 262K | | | | | | — | — |
|
|
362
360
|
| `kilo/qwen/qwen3.8-flash` | 1.0M | | | | | | $0.15 | $0.47 |
|
|
363
361
|
| `kilo/qwen/qwen3.8-max-0902` | 1.0M | | | | | | $2 | $6 |
|
|
362
|
+
| `kilo/qwen/qwen3.8-max-prime` | 1.0M | | | | | | $4 | $12 |
|
|
364
363
|
| `kilo/qwen/qwen3.8-omni-flash` | 1.0M | | | | | | $0.15 | $0.47 |
|
|
365
364
|
| `kilo/rekaai/reka-edge` | 16K | | | | | | $0.10 | $0.10 |
|
|
366
365
|
| `kilo/rekaai/reka-flash-3` | 66K | | | | | | $0.10 | $0.20 |
|
|
@@ -378,14 +377,15 @@ for await (const chunk of stream) {
|
|
|
378
377
|
| `kilo/stealth/claude-opus-4.8` | 1.0M | | | | | | $4 | $20 |
|
|
379
378
|
| `kilo/stealth/claude-sonnet-4.6` | 1.0M | | | | | | $2 | $12 |
|
|
380
379
|
| `kilo/stealth/qwen3.6-plus` | 1.0M | | | | | | $0.25 | $2 |
|
|
380
|
+
| `kilo/stealth/space-bunny-alpha` | 1.0M | | | | | | — | — |
|
|
381
381
|
| `kilo/stepfun/step-3.5-flash` | 262K | | | | | | $0.10 | $0.30 |
|
|
382
382
|
| `kilo/stepfun/step-3.7-flash` | 256K | | | | | | $0.20 | $1 |
|
|
383
383
|
| `kilo/stepfun/step-3.7-flash:free` | 262K | | | | | | — | — |
|
|
384
384
|
| `kilo/tencent/hunyuan-a13b-instruct` | 131K | | | | | | $0.14 | $0.57 |
|
|
385
385
|
| `kilo/tencent/hy-mt2-1.8b` | 8K | | | | | | $0.04 | $0.18 |
|
|
386
|
-
| `kilo/tencent/hy-mt2-30b-a3b` | 8K | | | | | | $0.07 | $0.
|
|
386
|
+
| `kilo/tencent/hy-mt2-30b-a3b` | 8K | | | | | | $0.07 | $0.28 |
|
|
387
387
|
| `kilo/tencent/hy-mt2-7b` | 8K | | | | | | $0.07 | $0.29 |
|
|
388
|
-
| `kilo/tencent/hy3` | 262K | | | | | | $0.
|
|
388
|
+
| `kilo/tencent/hy3` | 262K | | | | | | $0.13 | $0.53 |
|
|
389
389
|
| `kilo/tencent/hy3-preview` | 262K | | | | | | $0.18 | $0.60 |
|
|
390
390
|
| `kilo/tencent/hy4-preview` | 1.0M | | | | | | $0.83 | $3 |
|
|
391
391
|
| `kilo/thedrummer/cydonia-24b-v4.1` | 131K | | | | | | $0.30 | $0.50 |
|
|
@@ -427,6 +427,7 @@ for await (const chunk of stream) {
|
|
|
427
427
|
| `kilo/z-ai/glm-5.3` | 1.0M | | | | | | $1 | $4 |
|
|
428
428
|
| `kilo/z-ai/glm-5.3-flash` | 1.0M | | | | | | $0.15 | $0.50 |
|
|
429
429
|
| `kilo/z-ai/glm-5.3-flashx` | 1.0M | | | | | | $0.37 | $1 |
|
|
430
|
+
| `kilo/z-ai/glm-5.3-prime` | 1.0M | | | | | | $3 | $9 |
|
|
430
431
|
| `kilo/z-ai/glm-5v-turbo` | 203K | | | | | | $1 | $4 |
|
|
431
432
|
|
|
432
433
|
Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# LLM Gateway
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 428 LLM Gateway models through Mastra's model router. Authentication is handled automatically using the `LLMGATEWAY_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [LLM Gateway documentation](https://llmgateway.io/docs).
|
|
10
10
|
|
|
@@ -140,6 +140,8 @@ for await (const chunk of stream) {
|
|
|
140
140
|
| `llmgateway-providers/azure/gpt-5.6-sol` | 1.1M | | | | | | $4 | $20 |
|
|
141
141
|
| `llmgateway-providers/azure/gpt-5.6-terra` | 1.1M | | | | | | $2 | $12 |
|
|
142
142
|
| `llmgateway-providers/azure/gpt-6-astra` | 1.1M | | | | | | $10 | $50 |
|
|
143
|
+
| `llmgateway-providers/azure/gpt-6-luna` | 1.1M | | | | | | $0.10 | $0.50 |
|
|
144
|
+
| `llmgateway-providers/azure/gpt-6-sol` | 1.1M | | | | | | $2 | $10 |
|
|
143
145
|
| `llmgateway-providers/azure/gpt-oss-120b` | 131K | | | | | | $0.15 | $0.60 |
|
|
144
146
|
| `llmgateway-providers/azure/o1` | 200K | | | | | | $15 | $60 |
|
|
145
147
|
| `llmgateway-providers/azure/o3` | 200K | | | | | | $2 | $8 |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# NanoGPT
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 592 NanoGPT models through Mastra's model router. Authentication is handled automatically using the `NANO_GPT_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [NanoGPT documentation](https://docs.nano-gpt.com).
|
|
10
10
|
|
|
@@ -393,8 +393,8 @@ for await (const chunk of stream) {
|
|
|
393
393
|
| `nano-gpt/openai/gpt-5.6-terra-pro` | 1.1M | | | | | | $2 | $12 |
|
|
394
394
|
| `nano-gpt/openai/gpt-6-astra` | 1.1M | | | | | | $10 | $50 |
|
|
395
395
|
| `nano-gpt/openai/gpt-6-astra-pro` | 1.1M | | | | | | $10 | $50 |
|
|
396
|
-
| `nano-gpt/openai/gpt-6-luna` | 1.1M | | | | | | $0.
|
|
397
|
-
| `nano-gpt/openai/gpt-6-luna-pro` | 1.1M | | | | | | $0.
|
|
396
|
+
| `nano-gpt/openai/gpt-6-luna` | 1.1M | | | | | | $0.10 | $0.50 |
|
|
397
|
+
| `nano-gpt/openai/gpt-6-luna-pro` | 1.1M | | | | | | $0.10 | $0.50 |
|
|
398
398
|
| `nano-gpt/openai/gpt-6-sol` | 1.1M | | | | | | $2 | $10 |
|
|
399
399
|
| `nano-gpt/openai/gpt-6-sol-pro` | 1.1M | | | | | | $2 | $10 |
|
|
400
400
|
| `nano-gpt/openai/gpt-astra-latest` | 1.1M | | | | | | $10 | $50 |
|
|
@@ -491,6 +491,7 @@ for await (const chunk of stream) {
|
|
|
491
491
|
| `nano-gpt/qwen/qwen3.8-flash` | 992K | | | | | | $0.14 | $0.42 |
|
|
492
492
|
| `nano-gpt/qwen/qwen3.8-max` | 991K | | | | | | $2 | $6 |
|
|
493
493
|
| `nano-gpt/qwen/qwen3.8-max-0902` | 992K | | | | | | $2 | $6 |
|
|
494
|
+
| `nano-gpt/qwen/qwen3.8-max-prime` | 1.0M | | | | | | $4 | $12 |
|
|
494
495
|
| `nano-gpt/qwen/qwen3.8-max:thinking` | 991K | | | | | | $2 | $6 |
|
|
495
496
|
| `nano-gpt/qwen/qwen3.8-omni-flash` | 992K | | | | | | $0.15 | $0.47 |
|
|
496
497
|
| `nano-gpt/qwen3-vl-235b-a22b-instruct-original` | 33K | | | | | | $0.30 | $2 |
|
|
@@ -511,6 +512,7 @@ for await (const chunk of stream) {
|
|
|
511
512
|
| `nano-gpt/soob3123/amoral-gemma3-27B-v2` | 33K | | | | | | $0.30 | $0.30 |
|
|
512
513
|
| `nano-gpt/soob3123/GrayLine-Qwen3-8B` | 33K | | | | | | $0.30 | $0.30 |
|
|
513
514
|
| `nano-gpt/soob3123/Veiled-Calla-12B` | 33K | | | | | | $0.30 | $0.30 |
|
|
515
|
+
| `nano-gpt/stealth/space-bunny-alpha` | 1.0M | | | | | | $0.05 | $0.15 |
|
|
514
516
|
| `nano-gpt/Steelskull/L3.3-Cu-Mai-R1-70b` | 33K | | | | | | $0.49 | $0.49 |
|
|
515
517
|
| `nano-gpt/Steelskull/L3.3-Electra-R1-70b` | 33K | | | | | | $0.70 | $0.70 |
|
|
516
518
|
| `nano-gpt/Steelskull/L3.3-MS-Nevoria-70b` | 33K | | | | | | $0.49 | $0.49 |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# OpenCode Go
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 41 OpenCode Go models through Mastra's model router. Authentication is handled automatically using the `OPENCODE_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [OpenCode Go documentation](https://opencode.ai/docs/go).
|
|
10
10
|
|
|
@@ -47,6 +47,7 @@ for await (const chunk of stream) {
|
|
|
47
47
|
| `opencode-go/glm-5.3` | 1.0M | | | | | | $1 | $4 |
|
|
48
48
|
| `opencode-go/glm-5.3-flash` | 1.0M | | | | | | $0.15 | $0.50 |
|
|
49
49
|
| `opencode-go/gpt-5.6-luna` | 1.1M | | | | | | $0.20 | $1 |
|
|
50
|
+
| `opencode-go/gpt-6-luna` | 1.1M | | | | | | $0.10 | $0.50 |
|
|
50
51
|
| `opencode-go/grok-4.6` | 500K | | | | | | $2 | $6 |
|
|
51
52
|
| `opencode-go/grok-4.7` | 500K | | | | | | $2 | $6 |
|
|
52
53
|
| `opencode-go/hy3` | 256K | | | | | | $0.14 | $0.58 |
|
|
@@ -52,11 +52,11 @@ for await (const chunk of stream) {
|
|
|
52
52
|
| `snowflake-cortex/deepseek-r1` | 128K | | | | | | — | — |
|
|
53
53
|
| `snowflake-cortex/gemini-3.1-pro` | 1.0M | | | | | | — | — |
|
|
54
54
|
| `snowflake-cortex/mistral-large2` | 262K | | | | | | — | — |
|
|
55
|
-
| `snowflake-cortex/openai-gpt-4.1` |
|
|
56
|
-
| `snowflake-cortex/openai-gpt-5` |
|
|
55
|
+
| `snowflake-cortex/openai-gpt-4.1` | 128K | | | | | | — | — |
|
|
56
|
+
| `snowflake-cortex/openai-gpt-5` | 272K | | | | | | — | — |
|
|
57
57
|
| `snowflake-cortex/openai-gpt-5-mini` | 272K | | | | | | — | — |
|
|
58
|
-
| `snowflake-cortex/openai-gpt-5-nano` |
|
|
59
|
-
| `snowflake-cortex/openai-gpt-5.1` |
|
|
58
|
+
| `snowflake-cortex/openai-gpt-5-nano` | 272K | | | | | | — | — |
|
|
59
|
+
| `snowflake-cortex/openai-gpt-5.1` | 272K | | | | | | — | — |
|
|
60
60
|
| `snowflake-cortex/openai-gpt-5.2` | 400K | | | | | | — | — |
|
|
61
61
|
| `snowflake-cortex/openai-gpt-5.4` | 1.1M | | | | | | — | — |
|
|
62
62
|
| `snowflake-cortex/openai-gpt-5.5` | 1.1M | | | | | | — | — |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Tempr
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 39 Tempr models through Mastra's model router. Authentication is handled automatically using the `TEMPR_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Tempr documentation](https://temprhq.io/docs/gateway-reference.html).
|
|
10
10
|
|
|
@@ -67,6 +67,16 @@ for await (const chunk of stream) {
|
|
|
67
67
|
| `tempr/google/gemini-flash-lite-latest` | 1.0M | | | | | | $0.30 | $3 |
|
|
68
68
|
| `tempr/google/gemma-4-26b-a4b-it` | 262K | | | | | | — | — |
|
|
69
69
|
| `tempr/google/gemma-4-31b-it` | 262K | | | | | | — | — |
|
|
70
|
+
| `tempr/mistral/mistral-embed` | 8K | | | | | | $0.10 | — |
|
|
71
|
+
| `tempr/mistral/mistral-large-2512` | 262K | | | | | | $0.50 | $2 |
|
|
72
|
+
| `tempr/mistral/mistral-large-latest` | 262K | | | | | | $0.50 | $2 |
|
|
73
|
+
| `tempr/mistral/mistral-medium-2604` | 262K | | | | | | $2 | $8 |
|
|
74
|
+
| `tempr/mistral/mistral-medium-latest` | 262K | | | | | | $2 | $8 |
|
|
75
|
+
| `tempr/mistral/mistral-small-2603` | 256K | | | | | | $0.15 | $0.60 |
|
|
76
|
+
| `tempr/mistral/mistral-small-latest` | 256K | | | | | | $0.15 | $0.60 |
|
|
77
|
+
| `tempr/mistral/voxtral-small-latest` | 32K | | | | | | $0.10 | $0.30 |
|
|
78
|
+
| `tempr/mistral/zai-glm-5-2` | 1.0M | | | | | | $1 | $4 |
|
|
79
|
+
| `tempr/mistral/zai-glm-5-3` | 1.0M | | | | | | $1 | $4 |
|
|
70
80
|
|
|
71
81
|
Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
|
|
72
82
|
|
|
@@ -98,7 +108,7 @@ const agent = new Agent({
|
|
|
98
108
|
model: ({ requestContext }) => {
|
|
99
109
|
const useAdvanced = requestContext.task === "complex";
|
|
100
110
|
return useAdvanced
|
|
101
|
-
? "tempr/
|
|
111
|
+
? "tempr/mistral/zai-glm-5-3"
|
|
102
112
|
: "tempr/anthropic/claude-fable-5";
|
|
103
113
|
}
|
|
104
114
|
});
|
|
@@ -46,7 +46,7 @@ for await (const chunk of stream) {
|
|
|
46
46
|
| `togetherai/LiquidAI/LFM2-24B-A2B` | 33K | | | | | | $0.03 | $0.12 |
|
|
47
47
|
| `togetherai/meta-llama/Llama-3.3-70B-Instruct-Turbo` | 131K | | | | | | $1 | $1 |
|
|
48
48
|
| `togetherai/meta-llama/Meta-Llama-3-8B-Instruct-Lite` | 8K | | | | | | $0.14 | $0.14 |
|
|
49
|
-
| `togetherai/MiniMaxAI/MiniMax-M2.7` |
|
|
49
|
+
| `togetherai/MiniMaxAI/MiniMax-M2.7` | 197K | | | | | | $0.30 | $1 |
|
|
50
50
|
| `togetherai/MiniMaxAI/MiniMax-M3` | 524K | | | | | | $0.30 | $1 |
|
|
51
51
|
| `togetherai/moonshotai/Kimi-K2.6` | 262K | | | | | | $1 | $5 |
|
|
52
52
|
| `togetherai/moonshotai/Kimi-K2.7-Code` | 262K | | | | | | $0.95 | $4 |
|
|
@@ -60,7 +60,7 @@ for await (const chunk of stream) {
|
|
|
60
60
|
| `togetherai/Qwen/Qwen3.6-Plus` | 1.0M | | | | | | $0.50 | $3 |
|
|
61
61
|
| `togetherai/Qwen/Qwen3.7-Max` | 1.0M | | | | | | $1 | $4 |
|
|
62
62
|
| `togetherai/thinkingmachines/Inkling` | 524K | | | | | | $1 | $4 |
|
|
63
|
-
| `togetherai/zai-org/GLM-5.2` |
|
|
63
|
+
| `togetherai/zai-org/GLM-5.2` | 1.0M | | | | | | $1 | $4 |
|
|
64
64
|
| `togetherai/zai-org/GLM-5.3` | 1.0M | | | | | | $1 | $4 |
|
|
65
65
|
| `togetherai/zai-org/GLM-5.3-Flash` | 1.0M | | | | | | $0.15 | $0.50 |
|
|
66
66
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Vivgrid
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 34 Vivgrid models through Mastra's model router. Authentication is handled automatically using the `VIVGRID_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Vivgrid documentation](https://docs.vivgrid.com/models).
|
|
10
10
|
|
|
@@ -41,6 +41,7 @@ for await (const chunk of stream) {
|
|
|
41
41
|
| `vivgrid/claude-fable-5` | 1.0M | | | | | | $10 | $50 |
|
|
42
42
|
| `vivgrid/claude-fable-5-1` | 1.0M | | | | | | $10 | $50 |
|
|
43
43
|
| `vivgrid/claude-opus-5` | 1.0M | | | | | | $5 | $25 |
|
|
44
|
+
| `vivgrid/claude-opus-5-5` | 1.0M | | | | | | $4 | $20 |
|
|
44
45
|
| `vivgrid/claude-sonnet-5` | 1.0M | | | | | | $2 | $10 |
|
|
45
46
|
| `vivgrid/deepseek-v3.2` | 128K | | | | | | $0.28 | $0.42 |
|
|
46
47
|
| `vivgrid/deepseek-v4-flash` | 1.0M | | | | | | $0.15 | $0.30 |
|
|
@@ -67,6 +68,8 @@ for await (const chunk of stream) {
|
|
|
67
68
|
| `vivgrid/gpt-5.6-sol` | 1.1M | | | | | | $5 | $30 |
|
|
68
69
|
| `vivgrid/gpt-5.6-terra` | 1.1M | | | | | | $3 | $15 |
|
|
69
70
|
| `vivgrid/gpt-6-astra` | 1.1M | | | | | | $10 | $50 |
|
|
71
|
+
| `vivgrid/gpt-6-luna` | 1.1M | | | | | | $0.10 | $0.50 |
|
|
72
|
+
| `vivgrid/gpt-6-sol` | 1.1M | | | | | | $2 | $10 |
|
|
70
73
|
| `vivgrid/kimi-k3` | 1.0M | | | | | | $3 | $15 |
|
|
71
74
|
| `vivgrid/viv-fast` | 1.0M | | | | | | $0.13 | $0.40 |
|
|
72
75
|
|
|
@@ -390,6 +390,8 @@ Returns: [`Promise<DurableAgentStreamResult>`](#durableagentstreamresult)
|
|
|
390
390
|
|
|
391
391
|
**autoResumeSuspendedTools** (`boolean`): Automatically resume tools that suspended, instead of waiting for an external resume() call.
|
|
392
392
|
|
|
393
|
+
**closeOnSuspend** (`boolean`): When true, close the stream when the run suspends (for example for tool approval) so fullStream, text, and getFullOutput() resolve at the suspension boundary instead of staying open, matching non-durable Agent.stream(). Useful for callers such as AG-UI or A2A that need the stream to end. Defaults to false, which keeps the stream open across suspension so a later resume can continue streaming on the same reader. resume()/resumeStream() always return a fresh stream. (Default: `false`)
|
|
394
|
+
|
|
393
395
|
**toolCallConcurrency** (`number`): Maximum number of tool calls to execute concurrently.
|
|
394
396
|
|
|
395
397
|
**includeRawChunks** (`boolean`): Include raw provider chunks in the stream output.
|
|
@@ -275,6 +275,8 @@ Returns: `boolean`
|
|
|
275
275
|
|
|
276
276
|
**autoResumeSuspendedTools** (`boolean`): Automatically resume tools that suspended, instead of waiting for an external resume() call.
|
|
277
277
|
|
|
278
|
+
**closeOnSuspend** (`boolean`): When true, close the stream when the run suspends (for example for tool approval) so fullStream, text, and getFullOutput() resolve at the suspension boundary instead of staying open, matching non-durable Agent.stream(). Useful for callers such as AG-UI or A2A that need the stream to end. Defaults to false, which keeps the stream open across suspension so a later resume can continue streaming on the same reader. resume()/resumeStream() always return a fresh stream. (Default: `false`)
|
|
279
|
+
|
|
278
280
|
**toolCallConcurrency** (`number`): Maximum number of tool calls to execute concurrently.
|
|
279
281
|
|
|
280
282
|
**includeRawChunks** (`boolean`): Include raw provider chunks in the stream output.
|
|
@@ -287,7 +287,7 @@ Hono and Fastify enforce the request-body limit before JSON parsing. Express and
|
|
|
287
287
|
| ----------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------- |
|
|
288
288
|
| Trace | `traceId`, `threadId`, `resourceId`, `entityName`, `entityType`, `environment`, `status` | `eq`, `ne`, `in`, `notIn`, `exists`, `notExists` |
|
|
289
289
|
| Trace metadata | `metadata.<key>` | `eq`, `ne`, `in`, `notIn`, `exists`, `notExists` |
|
|
290
|
-
| Trace | `startedAt`, `endedAt`
|
|
290
|
+
| Trace | `startedAt`, `endedAt`, `durationMs` | `eq`, `ne`, `in`, `notIn`, `lt`, `lte`, `gt`, `gte`, `exists`, `notExists` |
|
|
291
291
|
| Trace | `tags` | `includes`, `notIncludes`, `exists`, `notExists` |
|
|
292
292
|
| Span | `name`, `spanType`, `model`, `provider`, `status`, `entityType`, `entityId`, `entityName`, `entityVersionId`, `parentEntityVersionId`, `rootEntityVersionId` | `eq`, `ne`, `in`, `notIn`, `exists`, `notExists` |
|
|
293
293
|
| Span | `startedAt`, `endedAt`, `durationMs` | `eq`, `ne`, `in`, `notIn`, `lt`, `lte`, `gt`, `gte`, `exists`, `notExists` |
|
|
@@ -303,6 +303,22 @@ Compose predicates with `{ op: 'and', args: [...] }`, `{ op: 'or', args: [...] }
|
|
|
303
303
|
|
|
304
304
|
String comparisons are case-sensitive, and literals are never coerced. Canonical string fields compare exact stored values. Metadata string values are trimmed before comparison, as described below. Numeric predicates, including `score` and `durationMs`, require numbers, while timestamp predicates require ISO timestamp strings. A missing value satisfies neither positive nor ordered predicates, although it does satisfy the negative operators `ne` and `notIn`. Combine a negative predicate with `exists` when the field must also be present.
|
|
305
305
|
|
|
306
|
+
### Filter by root duration
|
|
307
|
+
|
|
308
|
+
At trace scope, `durationMs` is the current completed root span's elapsed time in milliseconds. The value is derived from `endedAt - startedAt` when the query runs:
|
|
309
|
+
|
|
310
|
+
```typescript
|
|
311
|
+
const slowRootTraces = {
|
|
312
|
+
op: 'gt',
|
|
313
|
+
left: { path: 'durationMs' },
|
|
314
|
+
right: { literal: 5000 },
|
|
315
|
+
}
|
|
316
|
+
```
|
|
317
|
+
|
|
318
|
+
This top-level predicate doesn't inspect child spans. In contrast, `spans.some` with a `durationMs` predicate examines the current root span and current child spans, so a long child can satisfy that clause even when its root is shorter.
|
|
319
|
+
|
|
320
|
+
Trace queries exclude incomplete roots before evaluating predicates. As a result, `{ op: 'notExists', path: 'durationMs' }` doesn't find running traces or malformed roots without a usable `endedAt`.
|
|
321
|
+
|
|
306
322
|
### Filter by span properties
|
|
307
323
|
|
|
308
324
|
Every condition inside one `spans.some` or `spans.none` clause applies to the same current span. For example, this predicate finds a failed tool span whose name is `medication_lookup`; a matching name on one span and an error on another don't satisfy it:
|
|
@@ -51,7 +51,7 @@ Trailing user messages are treated as new messages and kept as sent. When they c
|
|
|
51
51
|
|
|
52
52
|
### Stored messages as the base layer
|
|
53
53
|
|
|
54
|
-
Loaders add stored messages with a `memory` source. When a stored message and an input message share an ID, the stored copy
|
|
54
|
+
Loaders add stored messages with a `memory` source. When a stored message and an input message share an ID, the stored copy keeps its text, reasoning, provider metadata, and `createdAt`. The only thing taken from the input is a tool outcome for a call that's still pending in the stored copy, such as a client tool result or an approval answer. Text and metadata from the input are ignored, so a client can't change a stored message by sending a different copy of it. To change a stored message, update it in storage. This also applies with `retainFullInput`.
|
|
55
55
|
|
|
56
56
|
## Related
|
|
57
57
|
|
|
@@ -4,10 +4,10 @@
|
|
|
4
4
|
|
|
5
5
|
# TokenLimiterProcessor
|
|
6
6
|
|
|
7
|
-
The `TokenLimiterProcessor` limits the number of tokens in messages.
|
|
7
|
+
The `TokenLimiterProcessor` limits the number of tokens in messages. Depending on `trimMode`, it acts as a prompt processor, an input processor, and an output processor:
|
|
8
8
|
|
|
9
|
-
- **
|
|
10
|
-
- **
|
|
9
|
+
- **Prompt processor** (`processLLMRequest`): In the default `best-fit` and `contiguous` trim modes, enforces the input budget on the provider prompt right before each model call, at every step of the agentic loop. The prompt is measured after earlier prompt processors (such as `ToolCallFilter`) have transformed it, so only tokens that actually reach the model are counted. Tool call and tool result messages are grouped so they're kept or removed together, and trimming is transient: stored messages are never modified.
|
|
10
|
+
- **Input processor** (`processInput`): In `memory-only` trim mode, filters historical messages to fit within the context window before the agentic loop starts, prioritizing recent messages
|
|
11
11
|
- **Output processor**: Limits generated response tokens via streaming or non-streaming with configurable strategies for handling exceeded limits
|
|
12
12
|
|
|
13
13
|
## Usage example
|
|
@@ -34,7 +34,7 @@ const processor = new TokenLimiterProcessor({
|
|
|
34
34
|
|
|
35
35
|
**options.countMode** (`'cumulative' | 'part'`): Whether to count tokens from the beginning of the stream or just the current part: 'cumulative' counts all tokens from start, 'part' only counts tokens in current part
|
|
36
36
|
|
|
37
|
-
**options.trimMode** (`'best-fit' | 'contiguous'`): Controls how
|
|
37
|
+
**options.trimMode** (`'best-fit' | 'contiguous' | 'memory-only'`): Controls how the token limit is enforced: 'best-fit' trims the provider prompt while keeping as many messages as possible (may create gaps), 'contiguous' trims the provider prompt but stops at the first message that does not fit (keeping a continuous suffix of conversation history), and 'memory-only' trims stored history in processInput instead of the prompt
|
|
38
38
|
|
|
39
39
|
## Returns
|
|
40
40
|
|
|
@@ -42,9 +42,11 @@ const processor = new TokenLimiterProcessor({
|
|
|
42
42
|
|
|
43
43
|
**name** (`string`): Optional processor display name
|
|
44
44
|
|
|
45
|
-
**processInput** (`(args: { messages: MastraDBMessage[]; abort: (reason?: string) => never }) => Promise<MastraDBMessage[]>`):
|
|
45
|
+
**processInput** (`(args: { messages: MastraDBMessage[]; abort: (reason?: string) => never }) => Promise<MastraDBMessage[]>`): Trims stored history to fit within the token limit before the agentic loop starts in 'memory-only' trim mode, prioritizing recent messages while preserving system messages and the current turn
|
|
46
46
|
|
|
47
|
-
**processInputStep** (`(args: ProcessInputStepArgs) => Promise<void>`):
|
|
47
|
+
**processInputStep** (`(args: ProcessInputStepArgs) => Promise<void>`): In 'memory-only' trim mode, applies stored-history trimming at each step. In the 'best-fit' and 'contiguous' trim modes, it leaves trimming to processLLMRequest when the agent runs that method for this processor. Otherwise it trims stored history, for example in generateLegacy() and streamLegacy() or when the limiter is inside a processor workflow.
|
|
48
|
+
|
|
49
|
+
**processLLMRequest** (`(args: ProcessLLMRequestArgs) => Promise<ProcessLLMRequestResult>`): Enforces the input budget on the provider prompt in the 'best-fit' and 'contiguous' trim modes. Runs after earlier prompt processors have transformed the prompt, counts that exact prompt, and returns a trimmed copy for the model call only. System messages are always preserved, and tool call and tool result messages are grouped so they are kept or removed together.
|
|
48
50
|
|
|
49
51
|
**processOutputStream** (`(args: ProcessOutputStreamArgs) => Promise<ChunkType | null>`): Processes streaming output parts to limit token count during streaming. Only text and object parts count against the limit and can be withheld; lifecycle, reasoning and tool parts always pass through.
|
|
50
52
|
|
|
@@ -72,10 +74,11 @@ Images and file attachments are estimated instead of tokenized, including `file`
|
|
|
72
74
|
|
|
73
75
|
## Error behavior
|
|
74
76
|
|
|
75
|
-
When
|
|
77
|
+
When trimming input, `TokenLimiterProcessor` throws a `TripWire` error in the following cases:
|
|
76
78
|
|
|
77
|
-
- **Empty messages**: If there are no messages to process, a TripWire is thrown because you can't send an LLM request with no messages.
|
|
79
|
+
- **Empty messages**: If there are no non-system messages to process, a TripWire is thrown because you can't send an LLM request with no messages.
|
|
78
80
|
- **System messages exceed limit**: If system messages alone exceed the token limit, a TripWire is thrown because you can't send an LLM request with only system messages and no user/assistant messages.
|
|
81
|
+
- **No messages fit**: If no message fits within the remaining token budget, a TripWire is thrown because you can't send an LLM request with no messages.
|
|
79
82
|
|
|
80
83
|
```typescript
|
|
81
84
|
import { TripWire } from '@mastra/core/agent'
|
|
@@ -112,9 +115,9 @@ export const agent = new Agent({
|
|
|
112
115
|
})
|
|
113
116
|
```
|
|
114
117
|
|
|
115
|
-
### As a per-step
|
|
118
|
+
### As a per-step processor (limit multi-step token growth)
|
|
116
119
|
|
|
117
|
-
When an agent uses tools across multiple steps (e.g. `maxSteps > 1`), each step accumulates conversation history from all previous steps.
|
|
120
|
+
When an agent uses tools across multiple steps (e.g. `maxSteps > 1`), each step accumulates conversation history from all previous steps. `TokenLimiterProcessor` applies its limit at every step, measuring the prompt that's about to be sent after any earlier prompt processors have run. Register prompt-shrinking processors such as `ToolCallFilter` before it, so the limiter counts the prompt the model actually receives:
|
|
118
121
|
|
|
119
122
|
```typescript
|
|
120
123
|
import { Agent } from '@mastra/core/agent'
|
|
@@ -136,6 +139,8 @@ const result = await agent.generate('Research this topic using your tools', {
|
|
|
136
139
|
})
|
|
137
140
|
```
|
|
138
141
|
|
|
142
|
+
Processor workflows don't run `processLLMRequest`, so a `TokenLimiterProcessor` inside a processor workflow trims stored messages in `processInputStep`, before prompt processors such as `ToolCallFilter` remove anything from the request. To count the prompt the model receives, add the limiter directly to `inputProcessors`.
|
|
143
|
+
|
|
139
144
|
### As an output processor (limit response length)
|
|
140
145
|
|
|
141
146
|
Use `outputProcessors` to limit the length of generated responses:
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mastra/mcp-docs-server",
|
|
3
|
-
"version": "1.3.0-alpha.
|
|
3
|
+
"version": "1.3.0-alpha.5",
|
|
4
4
|
"description": "MCP server for accessing Mastra.ai documentation, changelogs, and news.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -26,7 +26,7 @@
|
|
|
26
26
|
"@mastra/mcp-legacy": "npm:@mastra/mcp@^1.18.0",
|
|
27
27
|
"local-pkg": "^1.1.2",
|
|
28
28
|
"zod": "^4.6.4",
|
|
29
|
-
"@mastra/core": "1.70.0-alpha.
|
|
29
|
+
"@mastra/core": "1.70.0-alpha.3",
|
|
30
30
|
"@mastra/mcp": "^2.1.0-alpha.0"
|
|
31
31
|
},
|
|
32
32
|
"devDependencies": {
|
|
@@ -45,7 +45,7 @@
|
|
|
45
45
|
"vitest": "4.1.11",
|
|
46
46
|
"@internal/lint": "0.0.135",
|
|
47
47
|
"@internal/types-builder": "0.0.110",
|
|
48
|
-
"@mastra/core": "1.70.0-alpha.
|
|
48
|
+
"@mastra/core": "1.70.0-alpha.3"
|
|
49
49
|
},
|
|
50
50
|
"homepage": "https://mastra.ai",
|
|
51
51
|
"repository": {
|