@mastra/mcp-docs-server 1.2.25 → 1.2.26-alpha.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/.docs/docs/agents/overview.md +1 -1
  2. package/.docs/docs/agents/processors.md +21 -0
  3. package/.docs/docs/connections/connect-mcp-client.md +211 -0
  4. package/.docs/docs/deployment/mastra-server.md +8 -2
  5. package/.docs/docs/deployment/monorepo.md +12 -0
  6. package/.docs/docs/evals/datasets.md +5 -1
  7. package/.docs/docs/guides/context-engineering.md +1 -1
  8. package/.docs/docs/harness/background-tasks.md +30 -24
  9. package/.docs/docs/harness/signals.md +39 -0
  10. package/.docs/docs/index.md +1 -1
  11. package/.docs/docs/memory/message-history.md +6 -2
  12. package/.docs/docs/subagents.md +25 -0
  13. package/.docs/integrations/agentic-ui/ai-sdk-ui.md +7 -0
  14. package/.docs/integrations/file-storage/amazon-s3.md +7 -1
  15. package/.docs/integrations/file-storage/archil.md +3 -3
  16. package/.docs/integrations/frameworks/electron.md +1 -1
  17. package/.docs/integrations/observability/langfuse.md +11 -1
  18. package/.docs/integrations/sandboxes/cloudflare-sandbox.md +2 -0
  19. package/.docs/integrations/sandboxes/daytona.md +33 -0
  20. package/.docs/integrations/sandboxes/docker.md +13 -0
  21. package/.docs/integrations/voice/livekit.md +26 -2
  22. package/.docs/models/gateways/merge-gateway.md +7 -1
  23. package/.docs/models/gateways/netlify.md +5 -1
  24. package/.docs/models/gateways/openrouter.md +11 -2
  25. package/.docs/models/gateways/vercel.md +1 -1
  26. package/.docs/models/index.md +1 -1
  27. package/.docs/models/providers/above.md +2 -3
  28. package/.docs/models/providers/agentrouter.md +8 -6
  29. package/.docs/models/providers/alibaba-token-plan-cn.md +0 -1
  30. package/.docs/models/providers/alibaba-token-plan.md +0 -1
  31. package/.docs/models/providers/baseten.md +2 -1
  32. package/.docs/models/providers/bothub.md +10 -4
  33. package/.docs/models/providers/cline-pass.md +18 -17
  34. package/.docs/models/providers/cortecs.md +3 -3
  35. package/.docs/models/providers/deepinfra.md +7 -5
  36. package/.docs/models/providers/deepseek.md +7 -6
  37. package/.docs/models/providers/digitalocean.md +1 -1
  38. package/.docs/models/providers/edenai.md +27 -4
  39. package/.docs/models/providers/empiriolabs.md +3 -1
  40. package/.docs/models/providers/fireworks-ai.md +3 -1
  41. package/.docs/models/providers/greenpt.md +2 -1
  42. package/.docs/models/providers/huggingface.md +5 -1
  43. package/.docs/models/providers/hyper.md +6 -5
  44. package/.docs/models/providers/inception.md +2 -1
  45. package/.docs/models/providers/kilo.md +24 -16
  46. package/.docs/models/providers/llmgateway-providers.md +34 -2
  47. package/.docs/models/providers/llmgateway.md +9 -3
  48. package/.docs/models/providers/nan.md +1 -1
  49. package/.docs/models/providers/nano-gpt.md +20 -13
  50. package/.docs/models/providers/nvidia.md +3 -2
  51. package/.docs/models/providers/ofox.md +31 -3
  52. package/.docs/models/providers/ollama-cloud.md +3 -1
  53. package/.docs/models/providers/opencode-go.md +4 -4
  54. package/.docs/models/providers/pioneer.md +11 -2
  55. package/.docs/models/providers/requesty.md +4 -4
  56. package/.docs/models/providers/scnet-token-plan.md +4 -6
  57. package/.docs/models/providers/togetherai.md +2 -1
  58. package/.docs/models/providers/volcengine-coding-plan.md +3 -1
  59. package/.docs/reference/agents/agent.md +47 -1
  60. package/.docs/reference/agents/generate.md +2 -0
  61. package/.docs/reference/ai-sdk/to-ai-sdk-messages.md +16 -0
  62. package/.docs/reference/cli/mastra.md +28 -0
  63. package/.docs/reference/client-js/datasets.md +1 -1
  64. package/.docs/reference/configuration.md +2 -2
  65. package/.docs/reference/datasets/purgeItem.md +3 -3
  66. package/.docs/reference/index.md +3 -0
  67. package/.docs/reference/memory/cloneThread.md +2 -0
  68. package/.docs/reference/memory/copyThread.md +65 -0
  69. package/.docs/reference/memory/memory-class.md +2 -1
  70. package/.docs/reference/memory/recall.md +51 -0
  71. package/.docs/reference/memory/updateThreadResourceId.md +46 -0
  72. package/.docs/reference/processors/agents-md-injector.md +55 -0
  73. package/.docs/reference/processors/processor-interface.md +2 -0
  74. package/.docs/reference/pubsub/redis-streams.md +6 -0
  75. package/.docs/reference/pubsub/valkey-streams.md +6 -0
  76. package/.docs/reference/streaming/agents/stream.md +30 -0
  77. package/.docs/reference/tools/mcp-server.md +28 -0
  78. package/.docs/reference/workspace/filesystem.md +72 -0
  79. package/package.json +5 -5
@@ -430,7 +430,7 @@ export default function App(): React.JSX.Element {
430
430
  }
431
431
  ```
432
432
 
433
- This connects [`useChat()`](https://ai-sdk.dev/docs/reference/ai-sdk-ui/use-chat) to the `/chat/weather-agent` endpoint, sending propmts there and streaming the response back in chunks.
433
+ This connects [`useChat()`](https://ai-sdk.dev/docs/reference/ai-sdk-ui/use-chat) to the `/chat/weather-agent` endpoint, sending prompts there and streaming the response back in chunks.
434
434
 
435
435
  ## Test your agent
436
436
 
@@ -216,7 +216,7 @@ const tracingOptions = {
216
216
 
217
217
  This example produces `langfuse.trace.metadata.customerId` and `langfuse.trace.metadata.tier`.
218
218
 
219
- Metadata on the root span is also forwarded. Mastra sets `runId` and `resourceId` on every agent and workflow root span, and you can add your own keys through `tracingOptions.metadata`. The exporter forwards each of these root span keys to `langfuse.trace.metadata.<key>`. Keys that map to a dedicated Langfuse field (`userId`, `sessionId`, `threadId`, `traceName`, and `version`) are not duplicated as trace metadata. Metadata on child spans stays on the observation and never changes the trace.
219
+ Metadata on the root span is also forwarded. Mastra sets `runId` and `resourceId` on every agent and workflow root span, and you can add your own keys through `tracingOptions.metadata`. The exporter forwards each of these root span keys to `langfuse.trace.metadata.<key>`. Keys that map to a dedicated Langfuse field (`userId`, `sessionId`, `threadId`, `traceName`, and `version`) are not duplicated as trace metadata. Other metadata on child spans stays on the observation. Dedicated fields such as session IDs follow their own mapping rules.
220
220
 
221
221
  Notes:
222
222
 
@@ -225,6 +225,16 @@ Notes:
225
225
  - Keys under `langfuse` take precedence over root span metadata with the same name.
226
226
  - Values are sent as strings, because Langfuse maps trace metadata attributes as strings. Numbers, booleans, and objects are serialized with JSON. Langfuse Cloud restores them to their original types on ingestion.
227
227
 
228
+ ## Conversation sessions
229
+
230
+ The exporter maps span metadata to Langfuse sessions using `sessionId`, or `threadId` when `sessionId` is absent or `null`. An explicit empty `sessionId` suppresses this fallback and sends no session ID.
231
+
232
+ For [Observational Memory](https://mastra.ai/docs/memory/observational-memory), the Langfuse exporter keeps observer and reflector spans in the caller's session, even when child spans arrive before their parents. An explicit `sessionId`, including an empty string, takes precedence. Otherwise, the exporter uses the original caller thread carried by Observational Memory before falling back to the span's own `threadId`. Observational Memory captures that caller identity from the caller span's non-empty `threadId`, or the original request's thread ID. Nested observation preserves the outer caller identity.
233
+
234
+ This thread-to-session fallback is specific to Langfuse. Observational Memory doesn't synthesize generic `sessionId` metadata from a thread ID for other exporters.
235
+
236
+ Internal observer and reflector execution threads remain separate. Multi-thread observation uses the invoking caller's session, not the first thread in the batch. This session behavior applies regardless of the `bufferOnIdle` setting and doesn't change buffering or span parentage.
237
+
228
238
  ## Prompt linking
229
239
 
230
240
  You can link LLM generations to prompts stored in [Langfuse Prompt Management](https://langfuse.com/docs/prompt-management). It enables version tracking and metrics for your prompts.
@@ -82,6 +82,8 @@ const result = await workspace.sandbox?.executeCommand?.('npm', ['test'], {
82
82
 
83
83
  Relative paths are resolved under `/workspace`. Absolute paths must also resolve within `/workspace`. Each file is sent as its own bridge request, and the bridge caps a single file at 32 MiB.
84
84
 
85
+ The Cloudflare sandbox cannot set per-file permissions, so a `writeFiles` call that includes a `mode` is rejected rather than silently ignored.
86
+
85
87
  ```typescript
86
88
  await workspace.sandbox?.writeFiles?.([
87
89
  { path: 'src/index.ts', content: "console.log('hello')\n" },
@@ -200,6 +200,39 @@ const workspace = new Workspace({
200
200
 
201
201
  When the workspace starts, the filesystems are automatically mounted at the specified paths. Code running in the sandbox can then access files at `/s3-data` and `/gcs-data` as if they were local directories.
202
202
 
203
+ #### Temporary S3 credentials
204
+
205
+ Pass all three credential values to `S3Filesystem` when using temporary credentials. Daytona forwards the session token to s3fs:
206
+
207
+ ```typescript
208
+ import { Workspace } from '@mastra/core/workspace'
209
+ import { DaytonaSandbox } from '@mastra/daytona'
210
+ import { S3Filesystem } from '@mastra/s3'
211
+
212
+ const workspace = new Workspace({
213
+ mounts: {
214
+ '/s3-data': new S3Filesystem({
215
+ bucket: process.env.S3_BUCKET!,
216
+ region: process.env.S3_REGION ?? 'us-east-1',
217
+ endpoint: process.env.S3_ENDPOINT,
218
+ prefix: 'resource-123/thread-456/',
219
+ accessKeyId: process.env.SCOPED_S3_ACCESS_KEY_ID!,
220
+ secretAccessKey: process.env.SCOPED_S3_SECRET_ACCESS_KEY!,
221
+ sessionToken: process.env.SCOPED_S3_SESSION_TOKEN!,
222
+ }),
223
+ },
224
+ sandbox: new DaytonaSandbox({ language: 'python', ephemeral: true }),
225
+ })
226
+ ```
227
+
228
+ Before starting the workspace, obtain credentials from your storage provider that authorize only the intended prefix. Credential issuance is provider-specific; Mastra doesn't mint or restrict credentials. A `prefix` selects a directory but doesn't enforce authorization. Authenticate each request, authorize its resource and thread, and use separate sandboxes for separate security scopes.
229
+
230
+ Temporary credentials are loaded when the mount starts and aren't automatically refreshed. Host-side credential provider functions don't refresh the mount. Keep runs within the credential lifetime and create a new sandbox with fresh credentials for later runs. Reconnecting to a sandbox doesn't renew its credentials.
231
+
232
+ Credentials are uploaded into owner-only files inside private directories. After launching s3fs, Daytona removes the temporary-credential staging file and directory. The daemon retains the credentials in its environment, which sandbox code running as the same user or root can still read. Never supply broader credentials than the sandbox needs. Mounts using long-lived credentials retain their password files while s3fs needs them. Unmount and reconnect cleanup remove those files after verifying that the daemon has exited. If a daemon is still active, including after a mount is moved aside, or its status can't be checked, the files are retained and a warning is logged. Cleanup on a later unmount of the same path retries removal; deleting the sandbox removes any remaining files.
233
+
234
+ Prefixed filesystems require permission to list their prefix during initialization. With s3fs, a prefixed mount may also require a zero-byte object at the exact `<prefix>/` key. Provision that directory marker before mounting. After launching s3fs, Daytona checks the mounted directory's metadata without listing its contents, with a 15-second timeout and forced termination after another 5 seconds. A failed check reports a mount failure. Cleanup attempts to unmount the failed mount, moving a stuck mount aside if necessary to free the original path for retry. Moved stale mounts may remain until sandbox deletion. This startup check doesn't guarantee read or write access to individual files or ongoing daemon health; verify a read and write through the mounted path.
235
+
203
236
  #### Via `sandbox.mount()`
204
237
 
205
238
  Mount manually at any point after the sandbox has started:
@@ -162,6 +162,19 @@ const sandbox = new DockerSandbox({
162
162
  })
163
163
  ```
164
164
 
165
+ ## Write files
166
+
167
+ Upload multiple files in one call with `writeFiles`. Relative paths resolve under the working directory. Set an optional per-file `mode` to control POSIX permissions; it must be an integer between `0o001` and `0o777`. When `mode` is omitted, new files are created with `0644`.
168
+
169
+ ```typescript
170
+ await sandbox.writeFiles([
171
+ { path: 'src/index.js', content: "console.log('hello')\n" },
172
+ { path: 'run.sh', content: '#!/bin/sh\n', mode: 0o755 },
173
+ ])
174
+ ```
175
+
176
+ Docker is the only built-in sandbox that applies an explicit `mode`. Providers that cannot honor per-file permissions (Vercel, E2B, Daytona, Cloudflare) reject a `writeFiles` call that includes a `mode` instead of silently dropping it.
177
+
165
178
  ## Bind mounts
166
179
 
167
180
  Mount host directories into the container using the `volumes` option:
@@ -205,6 +205,28 @@ export default createLiveKitWorker({
205
205
 
206
206
  `configuration.stt` works the same way for per-call transcription, for example a different transcription model or language per tenant. The greeting has a matching per-call form: `configuration.greeting.text` accepts a resolver with the same call context, so one worker can open with each tenant's own phrasing.
207
207
 
208
+ ### Per-call turn detection
209
+
210
+ LiveKit's `TurnDetector` classes read the job's inference executor when constructed, so they can only be created inside a LiveKit job, not at module scope where the worker options live. To use one, set the `configuration.turnDetection` resolver. It runs once per call with the same call context as `configuration.stt` and returns anything the top-level `turnDetection` option accepts. Return `undefined` to fall back to the top-level option.
211
+
212
+ ```typescript
213
+ import { turnDetector } from '@livekit/agents-plugin-livekit'
214
+
215
+ export default createLiveKitWorker({
216
+ mastra,
217
+ agent: 'support',
218
+ stt: 'deepgram/nova-3',
219
+ tts: 'cartesia/sonic-3',
220
+ turnDetection: 'multilingual',
221
+ configuration: {
222
+ // Constructed inside the job, where the inference executor is available.
223
+ turnDetection: () => new turnDetector.MultilingualModel(0.2),
224
+ },
225
+ })
226
+ ```
227
+
228
+ The semantic model's inference runners must be registered before the agent server boots, so when this resolver is set the worker imports `@livekit/agents-plugin-livekit` up front. Keep the top-level `turnDetection` set to `'multilingual'` or `'english'` to pre-register only that model; otherwise both stay available.
229
+
208
230
  ### Memory and threads
209
231
 
210
232
  When the resolved Mastra agent has memory configured, each call becomes one memory thread:
@@ -539,7 +561,7 @@ if (process.argv[1] === fileURLToPath(import.meta.url)) {
539
561
 
540
562
  **vad** (`VAD | 'silero' | false`): Voice activity detection. 'silero' loads the Silero VAD from @livekit/agents-plugin-silero during prewarm. Pass an instance to bring your own, or false to disable. (Default: `'silero'`)
541
563
 
542
- **turnDetection** (`'multilingual' | 'english' | TurnDetectionMode`): End-of-turn detection. 'multilingual' and 'english' load LiveKit's semantic turn detector from @livekit/agents-plugin-livekit. Other values such as 'vad', 'stt', or 'manual' pass through.
564
+ **turnDetection** (`'multilingual' | 'english' | TurnDetectionMode`): End-of-turn detection. 'multilingual' and 'english' load LiveKit's semantic turn detector from @livekit/agents-plugin-livekit. Other values such as 'vad', 'stt', or 'manual' pass through. To construct a TurnDetector instance per call, set the configuration.turnDetection resolver — it takes precedence, with this option as the fallback.
543
565
 
544
566
  **turnHandling** (`Partial<TurnHandlingOptions>`): Turn handling tuning: endpointing delays, interruption sensitivity, preemptive generation. The worker disables preemptiveGeneration unless set here — each preemptive attempt re-runs the Mastra agent and persists partial user and assistant messages unless memory.options.readOnly is set.
545
567
 
@@ -551,7 +573,7 @@ if (process.argv[1] === fileURLToPath(import.meta.url)) {
551
573
 
552
574
  **onTurnComplete** (`(ctx: VoiceTurnCompleteContext) => void | Promise<void>`): Called once per turn after the reply finished streaming to text-to-speech. Runs off the audio path and is not awaited. The context carries the produced reply (text, toolCalls, interrupted, usage) and the resolved memory mapping.
553
575
 
554
- **configuration** (`LiveKitWorkerConfiguration`): Grouped conversation and compliance configuration: the opening greeting and AI disclosure, consent requirements, agent-initiated hang-up, and per-call STT/TTS selection.
576
+ **configuration** (`LiveKitWorkerConfiguration`): Grouped conversation and compliance configuration: the opening greeting and AI disclosure, consent requirements, agent-initiated hang-up, and per-call STT/TTS/turn detection selection.
555
577
 
556
578
  **configuration.greeting** (`GreetingConfiguration`): The opening greeting and AI disclosure: text (a fixed string or a per-call resolver for per-tenant greetings), allowInterruptions, awaitPlayout, persist, and periodic re-disclosure via repeatEvery and repeatText.
557
579
 
@@ -563,6 +585,8 @@ if (process.argv[1] === fileURLToPath(import.meta.url)) {
563
585
 
564
586
  **configuration.tts** (`(context: VoiceCallContext) => TTS | string | undefined`): Per-call text-to-speech: a resolver invoked once per call (post-connect) with { metadata, requestContext, roomName, ctx }, returning anything the top-level tts option accepts — one voice or language per tenant. Return undefined to fall back to the top-level tts. Cache plugin instances across calls.
565
587
 
588
+ **configuration.turnDetection** (`(context: VoiceCallContext) => TurnDetectionMode | undefined`): Per-call end-of-turn detection: a resolver invoked once per call (post-connect, inside the LiveKit job) with { metadata, requestContext, roomName, ctx }, returning anything the top-level turnDetection option accepts. Use it to construct LiveKit TurnDetector instances, which need the job's inference executor. Return undefined to fall back to the top-level turnDetection.
589
+
566
590
  **greeting** (`string`): Static greeting spoken when the session starts. Deprecated: prefer configuration.greeting.text.
567
591
 
568
592
  **persistGreeting** (`boolean`): Save the spoken greeting to the memory thread as an assistant message, making the saved thread a faithful call transcript. Only applies when a greeting is set and memory is enabled. Deprecated: prefer configuration.greeting.persist. (Default: `true`)
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![Merge Gateway logo](https://models.dev/logos/merge-gateway.svg)Merge Gateway
6
6
 
7
- Merge Gateway aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 181 models through Mastra's model router.
7
+ Merge Gateway aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 187 models through Mastra's model router.
8
8
 
9
9
  Learn more in the [Merge Gateway documentation](https://docs.merge.dev/merge-gateway).
10
10
 
@@ -67,11 +67,13 @@ ANTHROPIC_API_KEY=ant-...
67
67
  | `deepseek/deepseek-v3.1` |
68
68
  | `deepseek/deepseek-v3.2` |
69
69
  | `deepseek/deepseek-v4-flash` |
70
+ | `deepseek/deepseek-v4-flash-0423` |
70
71
  | `deepseek/deepseek-v4-flash-0731` |
71
72
  | `deepseek/deepseek-v4-flash-0731-fast` |
72
73
  | `deepseek/deepseek-v4-pro` |
73
74
  | `deepseek/deepseek-v4-pro-0423` |
74
75
  | `deepseek/deepseek-v4-pro-0813` |
76
+ | `deepseek/deepseek-v4.1-flash` |
75
77
  | `google/gemini-2.5-computer-use-preview-10-2025` |
76
78
  | `google/gemini-2.5-flash` |
77
79
  | `google/gemini-2.5-flash-image` |
@@ -93,6 +95,9 @@ ANTHROPIC_API_KEY=ant-...
93
95
  | `google/gemini-embedding-001` |
94
96
  | `google/gemini-flash-latest` |
95
97
  | `google/gemini-flash-lite-latest` |
98
+ | `google/gemma-3-12b-it` |
99
+ | `google/gemma-3-27b-it` |
100
+ | `google/gemma-3-4b-it` |
96
101
  | `google/gemma-4-26b-a4b-it` |
97
102
  | `google/gemma-4-31b-it` |
98
103
  | `meta/llama-3.1-70b-instruct` |
@@ -159,6 +164,7 @@ ANTHROPIC_API_KEY=ant-...
159
164
  | `openai/gpt-oss-120b` |
160
165
  | `openai/gpt-oss-20b` |
161
166
  | `openai/gpt-oss-safeguard-120b` |
167
+ | `openai/gpt-oss-safeguard-20b` |
162
168
  | `openai/o1` |
163
169
  | `openai/o3` |
164
170
  | `openai/o3-mini` |
@@ -4,7 +4,7 @@
4
4
 
5
5
  # Netlify
6
6
 
7
- Netlify AI Gateway provides unified access to multiple providers with built-in caching and observability. Access 251 models through Mastra's model router.
7
+ Netlify AI Gateway provides unified access to multiple providers with built-in caching and observability. Access 255 models through Mastra's model router.
8
8
 
9
9
  Learn more in the [Netlify documentation](https://docs.netlify.com/build/ai-gateway/overview/).
10
10
 
@@ -161,6 +161,10 @@ ANTHROPIC_API_KEY=ant-...
161
161
  | `openrouter/inclusionai/ling-3.0-flash-fin` |
162
162
  | `openrouter/inclusionai/ling-3.0-flash-fin:free` |
163
163
  | `openrouter/inclusionai/ling-3.0-flash-sante:free` |
164
+ | `openrouter/inclusionai/ling-3.0-flash-vl` |
165
+ | `openrouter/inclusionai/ling-3.0-flash-vl:free` |
166
+ | `openrouter/inference-net/schematron-v2-small` |
167
+ | `openrouter/inference-net/schematron-v2-turbo` |
164
168
  | `openrouter/mancer/weaver` |
165
169
  | `openrouter/meta-llama/llama-3.1-70b-instruct` |
166
170
  | `openrouter/meta-llama/llama-3.1-8b-instruct` |
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![OpenRouter logo](https://models.dev/logos/openrouter.svg)OpenRouter
6
6
 
7
- OpenRouter aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 358 models through Mastra's model router.
7
+ OpenRouter aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 367 models through Mastra's model router.
8
8
 
9
9
  Learn more in the [OpenRouter documentation](https://openrouter.ai/models).
10
10
 
@@ -46,8 +46,11 @@ ANTHROPIC_API_KEY=ant-...
46
46
  | `~google/gemini-flash-latest` |
47
47
  | `~google/gemini-pro-latest` |
48
48
  | `~moonshotai/kimi-latest` |
49
- | `~openai/gpt-latest` |
49
+ | `~openai/gpt-astra-latest` |
50
+ | `~openai/gpt-luna-latest` |
50
51
  | `~openai/gpt-mini-latest` |
52
+ | `~openai/gpt-sol-latest` |
53
+ | `~openai/gpt-terra-latest` |
51
54
  | `~x-ai/grok-latest` |
52
55
  | `~z-ai/glm-flash-latest` |
53
56
  | `~z-ai/glm-latest` |
@@ -147,6 +150,10 @@ ANTHROPIC_API_KEY=ant-...
147
150
  | `inclusionai/ling-3.0-flash-fin` |
148
151
  | `inclusionai/ling-3.0-flash-fin:free` |
149
152
  | `inclusionai/ling-3.0-flash-sante:free` |
153
+ | `inclusionai/ling-3.0-flash-vl` |
154
+ | `inclusionai/ling-3.0-flash-vl:free` |
155
+ | `inference-net/schematron-v2-small` |
156
+ | `inference-net/schematron-v2-turbo` |
150
157
  | `kwaipilot/kat-coder-pro-v2` |
151
158
  | `kwaipilot/kat-coder-pro-v2.5` |
152
159
  | `liquid/lfm-2.5-2.6b:free` |
@@ -349,7 +356,9 @@ ANTHROPIC_API_KEY=ant-...
349
356
  | `rekaai/reka-flash-3` |
350
357
  | `relace/relace-apply-3` |
351
358
  | `relace/relace-search` |
359
+ | `sakana/fugu-max` |
352
360
  | `sakana/fugu-ultra` |
361
+ | `sakana/fugu-ultra-v2` |
353
362
  | `sakana/sakana-namazu` |
354
363
  | `sao10k/l3-lunaris-8b` |
355
364
  | `sao10k/l3.1-euryale-70b` |
@@ -143,7 +143,7 @@ ANTHROPIC_API_KEY=ant-...
143
143
  | `deepseek/deepseek-v4-flash-vision-exp` |
144
144
  | `deepseek/deepseek-v4-pro` |
145
145
  | `deepseek/deepseek-v4-pro-0813` |
146
- | `deepseek/deepseek-v4.1-flash-beta` |
146
+ | `deepseek/deepseek-v4.1-flash` |
147
147
  | `fish-audio/s1` |
148
148
  | `fish-audio/s1-free` |
149
149
  | `fish-audio/s2-pro` |
@@ -4,7 +4,7 @@
4
4
 
5
5
  # Model Providers
6
6
 
7
- Mastra provides a unified interface for working with LLMs across multiple providers, giving you access to 7132 models from 200 providers through a single API.
7
+ Mastra provides a unified interface for working with LLMs across multiple providers, giving you access to 7294 models from 200 providers through a single API.
8
8
 
9
9
  ## Features
10
10
 
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![above.dev logo](https://models.dev/logos/above.svg)above.dev
6
6
 
7
- Access 9 above.dev models through Mastra's model router. Authentication is handled automatically using the `ABOVE_API_KEY` environment variable.
7
+ Access 8 above.dev models through Mastra's model router. Authentication is handled automatically using the `ABOVE_API_KEY` environment variable.
8
8
 
9
9
  Learn more in the [above.dev documentation](https://above.dev/docs).
10
10
 
@@ -38,14 +38,13 @@ for await (const chunk of stream) {
38
38
 
39
39
  | Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
40
40
  | ------------------------------------ | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
41
- | `above/deepseek-v4-flash` | 1.0M | | | | | | $0.24 | $0.73 |
41
+ | `above/deepseek-v4-flash` | 1.0M | | | | | | $0.17 | $0.66 |
42
42
  | `above/deepseek-v4-flash-vision-exp` | 1.0M | | | | | | $0.24 | $0.73 |
43
43
  | `above/deepseek-v4-pro` | 1.0M | | | | | | $0.73 | $2 |
44
44
  | `above/glm-5.2` | 1.0M | | | | | | $2 | $5 |
45
45
  | `above/glm-5.2-fast` | 1.0M | | | | | | $2 | $7 |
46
46
  | `above/glm-5.3-flash` | 1.0M | | | | | | $0.17 | $0.55 |
47
47
  | `above/mimo-v2.5-pro` | 1.0M | | | | | | $0.51 | $1 |
48
- | `above/mimo-v2.5-pro-ultraspeed` | 1.0M | | | | | | $2 | $3 |
49
48
  | `above/qwen3.8-max` | 1.0M | | | | | | $2 | $7 |
50
49
 
51
50
  Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![AgentRouter logo](https://models.dev/logos/agentrouter.svg)AgentRouter
6
6
 
7
- Access 3 AgentRouter models through Mastra's model router. Authentication is handled automatically using the `AGENTROUTER_API_KEY` environment variable.
7
+ Access 5 AgentRouter models through Mastra's model router. Authentication is handled automatically using the `AGENTROUTER_API_KEY` environment variable.
8
8
 
9
9
  Learn more in the [AgentRouter documentation](https://agentrouter.org/docs/opencode.html).
10
10
 
@@ -36,11 +36,13 @@ for await (const chunk of stream) {
36
36
 
37
37
  ## Models
38
38
 
39
- | Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
40
- | ----------------------------- | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
41
- | `agentrouter/claude-opus-4-8` | 1.0M | | | | | | — | — |
42
- | `agentrouter/claude-opus-5` | 1.0M | | | | | | — | — |
43
- | `agentrouter/gpt-5.6-sol` | 1.1M | | | | | | — | — |
39
+ | Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
40
+ | ------------------------------- | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
41
+ | `agentrouter/claude-opus-4-8` | 1.0M | | | | | | — | — |
42
+ | `agentrouter/claude-opus-5` | 1.0M | | | | | | — | — |
43
+ | `agentrouter/deepseek-v4-flash` | 1.0M | | | | | | — | — |
44
+ | `agentrouter/glm-5.3` | 1.0M | | | | | | — | — |
45
+ | `agentrouter/gpt-5.6-sol` | 1.1M | | | | | | — | — |
44
46
 
45
47
  Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
46
48
 
@@ -61,7 +61,6 @@ for await (const chunk of stream) {
61
61
  | `alibaba-token-plan-cn/qwen3.7-plus` | 1.0M | | | | | | — | — |
62
62
  | `alibaba-token-plan-cn/qwen3.8-flash` | 1.0M | | | | | | — | — |
63
63
  | `alibaba-token-plan-cn/qwen3.8-max` | 1.0M | | | | | | — | — |
64
- | `alibaba-token-plan-cn/qwen3.8-max-preview` | 1.0M | | | | | | — | — |
65
64
  | `alibaba-token-plan-cn/wan2.7-image` | 8K | | | | | | — | — |
66
65
  | `alibaba-token-plan-cn/wan2.7-image-pro` | 8K | | | | | | — | — |
67
66
 
@@ -61,7 +61,6 @@ for await (const chunk of stream) {
61
61
  | `alibaba-token-plan/qwen3.7-plus` | 1.0M | | | | | | — | — |
62
62
  | `alibaba-token-plan/qwen3.8-flash` | 1.0M | | | | | | — | — |
63
63
  | `alibaba-token-plan/qwen3.8-max` | 1.0M | | | | | | — | — |
64
- | `alibaba-token-plan/qwen3.8-max-preview` | 1.0M | | | | | | — | — |
65
64
  | `alibaba-token-plan/wan2.7-image` | 8K | | | | | | — | — |
66
65
  | `alibaba-token-plan/wan2.7-image-pro` | 8K | | | | | | — | — |
67
66
 
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![Baseten logo](https://models.dev/logos/baseten.svg)Baseten
6
6
 
7
- Access 22 Baseten models through Mastra's model router. Authentication is handled automatically using the `BASETEN_API_KEY` environment variable.
7
+ Access 23 Baseten models through Mastra's model router. Authentication is handled automatically using the `BASETEN_API_KEY` environment variable.
8
8
 
9
9
  Learn more in the [Baseten documentation](https://docs.baseten.co).
10
10
 
@@ -41,6 +41,7 @@ for await (const chunk of stream) {
41
41
  | `baseten/deepseek-ai/DeepSeek-V4-Flash-0731` | 1.0M | | | | | | $0.13 | $0.26 |
42
42
  | `baseten/deepseek-ai/DeepSeek-V4-Pro` | 1.0M | | | | | | $2 | $3 |
43
43
  | `baseten/deepseek-ai/DeepSeek-V4-Pro-0813` | 1.0M | | | | | | $1 | $4 |
44
+ | `baseten/deepseek-ai/DeepSeek-V4.1-Flash` | 1.0M | | | | | | $0.30 | $1 |
44
45
  | `baseten/moonshotai/Kimi-K2.5` | 262K | | | | | | $0.60 | $3 |
45
46
  | `baseten/moonshotai/Kimi-K2.6` | 262K | | | | | | $0.95 | $4 |
46
47
  | `baseten/moonshotai/Kimi-K2.7-Code` | 262K | | | | | | $0.95 | $4 |
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![Bothub logo](https://models.dev/logos/bothub.svg)Bothub
6
6
 
7
- Access 2 Bothub models through Mastra's model router. Authentication is handled automatically using the `BOTHUB_API_KEY` environment variable.
7
+ Access 8 Bothub models through Mastra's model router. Authentication is handled automatically using the `BOTHUB_API_KEY` environment variable.
8
8
 
9
9
  Learn more in the [Bothub documentation](https://bothub.ru/models).
10
10
 
@@ -19,7 +19,7 @@ const agent = new Agent({
19
19
  id: "my-agent",
20
20
  name: "My Agent",
21
21
  instructions: "You are a helpful assistant",
22
- model: "bothub/gemma-4-31b-it:free"
22
+ model: "bothub/deepseek-v4-flash-0731"
23
23
  });
24
24
 
25
25
  // Generate a response
@@ -38,7 +38,13 @@ for await (const chunk of stream) {
38
38
 
39
39
  | Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
40
40
  | ---------------------------------------- | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
41
+ | `bothub/deepseek-v4-flash-0731` | 1.0M | | | | | | $0.10 | $0.28 |
42
+ | `bothub/deepseek-v4-pro-0813` | 1.0M | | | | | | $2 | $5 |
41
43
  | `bothub/gemma-4-31b-it:free` | 262K | | | | | | — | — |
44
+ | `bothub/glm-5.3` | 1.0M | | | | | | $2 | $5 |
45
+ | `bothub/glm-5.3-flash` | 1.0M | | | | | | $0.12 | $0.44 |
46
+ | `bothub/gpt-5.6-luna` | 1.1M | | | | | | $0.06 | $0.37 |
47
+ | `bothub/muse-spark-1.3-contributor` | 1.0M | | | | | | $0.10 | $0.20 |
42
48
  | `bothub/nemotron-3-ultra-550b-a55b:free` | 1.0M | | | | | | — | — |
43
49
 
44
50
  Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
@@ -53,7 +59,7 @@ const agent = new Agent({
53
59
  name: "custom-agent",
54
60
  model: {
55
61
  url: "https://openai.bothub.ru/v1",
56
- id: "bothub/gemma-4-31b-it:free",
62
+ id: "bothub/deepseek-v4-flash-0731",
57
63
  apiKey: process.env.BOTHUB_API_KEY,
58
64
  headers: {
59
65
  "X-Custom-Header": "value"
@@ -72,7 +78,7 @@ const agent = new Agent({
72
78
  const useAdvanced = requestContext.task === "complex";
73
79
  return useAdvanced
74
80
  ? "bothub/nemotron-3-ultra-550b-a55b:free"
75
- : "bothub/gemma-4-31b-it:free";
81
+ : "bothub/deepseek-v4-flash-0731";
76
82
  }
77
83
  });
78
84
  ```
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![ClinePass logo](https://models.dev/logos/cline-pass.svg)ClinePass
6
6
 
7
- Access 14 ClinePass models through Mastra's model router. Authentication is handled automatically using the `CLINE_API_KEY` environment variable.
7
+ Access 15 ClinePass models through Mastra's model router. Authentication is handled automatically using the `CLINE_API_KEY` environment variable.
8
8
 
9
9
  Learn more in the [ClinePass documentation](https://docs.cline.bot/getting-started/clinepass).
10
10
 
@@ -36,22 +36,23 @@ for await (const chunk of stream) {
36
36
 
37
37
  ## Models
38
38
 
39
- | Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
40
- | ----------------------------------------- | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
41
- | `cline-pass/cline-pass/deepseek-v4-flash` | 1.0M | | | | | | $0.14 | $0.28 |
42
- | `cline-pass/cline-pass/deepseek-v4-pro` | 1.0M | | | | | | $2 | $3 |
43
- | `cline-pass/cline-pass/glm-5.2` | 1.0M | | | | | | $1 | $4 |
44
- | `cline-pass/cline-pass/glm-5.3` | 1.0M | | | | | | $1 | $4 |
45
- | `cline-pass/cline-pass/glm-5.3-flash` | 1.0M | | | | | | $0.15 | $0.50 |
46
- | `cline-pass/cline-pass/kimi-k2.6` | 262K | | | | | | $0.95 | $4 |
47
- | `cline-pass/cline-pass/kimi-k2.7-code` | 262K | | | | | | $0.95 | $4 |
48
- | `cline-pass/cline-pass/kimi-k3` | 1.0M | | | | | | $3 | $15 |
49
- | `cline-pass/cline-pass/mimo-v2.5` | 1.0M | | | | | | $0.14 | $0.28 |
50
- | `cline-pass/cline-pass/mimo-v2.5-pro` | 1.0M | | | | | | $2 | $3 |
51
- | `cline-pass/cline-pass/minimax-m3` | 1.0M | | | | | | $0.30 | $1 |
52
- | `cline-pass/cline-pass/qwen3.7-max` | 1.0M | | | | | | $3 | $8 |
53
- | `cline-pass/cline-pass/qwen3.7-plus` | 1.0M | | | | | | $0.40 | $2 |
54
- | `cline-pass/cline-pass/qwen3.8-max` | 1.0M | | | | | | $2 | $6 |
39
+ | Model | Context | Tools | Reasoning | Image | Audio | Video | Input $/1M | Output $/1M |
40
+ | ------------------------------------------- | ------- | ----- | --------- | ----- | ----- | ----- | ---------- | ----------- |
41
+ | `cline-pass/cline-pass/deepseek-v4-flash` | 1.0M | | | | | | $0.14 | $0.28 |
42
+ | `cline-pass/cline-pass/deepseek-v4-pro` | 1.0M | | | | | | $2 | $3 |
43
+ | `cline-pass/cline-pass/deepseek-v4.1-flash` | 1.0M | | | | | | $0.15 | $0.60 |
44
+ | `cline-pass/cline-pass/glm-5.2` | 1.0M | | | | | | $1 | $4 |
45
+ | `cline-pass/cline-pass/glm-5.3` | 1.0M | | | | | | $1 | $4 |
46
+ | `cline-pass/cline-pass/glm-5.3-flash` | 1.0M | | | | | | $0.15 | $0.50 |
47
+ | `cline-pass/cline-pass/kimi-k2.6` | 262K | | | | | | $0.95 | $4 |
48
+ | `cline-pass/cline-pass/kimi-k2.7-code` | 262K | | | | | | $0.95 | $4 |
49
+ | `cline-pass/cline-pass/kimi-k3` | 1.0M | | | | | | $3 | $15 |
50
+ | `cline-pass/cline-pass/mimo-v2.5` | 1.0M | | | | | | $0.14 | $0.28 |
51
+ | `cline-pass/cline-pass/mimo-v2.5-pro` | 1.0M | | | | | | $2 | $3 |
52
+ | `cline-pass/cline-pass/minimax-m3` | 1.0M | | | | | | $0.30 | $1 |
53
+ | `cline-pass/cline-pass/qwen3.7-max` | 1.0M | | | | | | $3 | $8 |
54
+ | `cline-pass/cline-pass/qwen3.7-plus` | 1.0M | | | | | | $0.40 | $2 |
55
+ | `cline-pass/cline-pass/qwen3.8-max` | 1.0M | | | | | | $2 | $6 |
55
56
 
56
57
  Model availability, capabilities, context windows, and pricing are sourced from [models.dev](https://models.dev) and may change.
57
58
 
@@ -52,7 +52,7 @@ for await (const chunk of stream) {
52
52
  | `cortecs/codestral-2508` | 256K | | | | | | $0.33 | $1 |
53
53
  | `cortecs/deepseek-r1-0528` | 164K | | | | | | $0.65 | $3 |
54
54
  | `cortecs/deepseek-v3.2` | 164K | | | | | | $0.30 | $0.49 |
55
- | `cortecs/deepseek-v4-flash-0731` | 1.0M | | | | | | $0.13 | $0.28 |
55
+ | `cortecs/deepseek-v4-flash-0731` | 1.0M | | | | | | $0.09 | $0.17 |
56
56
  | `cortecs/deepseek-v4-pro` | 1.0M | | | | | | $2 | $3 |
57
57
  | `cortecs/deepseek-v4-pro-0813` | 1.0M | | | | | | $2 | $4 |
58
58
  | `cortecs/devstral-2512` | 256K | | | | | | $0.48 | $2 |
@@ -73,7 +73,7 @@ for await (const chunk of stream) {
73
73
  | `cortecs/glm-5.1` | 203K | | | | | | $1 | $4 |
74
74
  | `cortecs/glm-5.2` | 1.0M | | | | | | $1 | $4 |
75
75
  | `cortecs/glm-5.3` | 1.0M | | | | | | $1 | $4 |
76
- | `cortecs/glm-5.3-flash` | 1.0M | | | | | | $0.20 | $0.50 |
76
+ | `cortecs/glm-5.3-flash` | 1.0M | | | | | | $0.10 | $0.35 |
77
77
  | `cortecs/glm-5v-turbo` | 203K | | | | | | $1 | $4 |
78
78
  | `cortecs/gpt-4.1` | 1.0M | | | | | | $2 | $9 |
79
79
  | `cortecs/gpt-4.1-mini` | 1.0M | | | | | | $0.43 | $2 |
@@ -139,7 +139,7 @@ for await (const chunk of stream) {
139
139
  | `cortecs/qwen3.6-27b` | 262K | | | | | | $0.45 | $3 |
140
140
  | `cortecs/qwen3.6-35b-a3b` | 262K | | | | | | $0.17 | $0.56 |
141
141
  | `cortecs/qwen3.8-2.4t-a95b` | 262K | | | | | | $3 | $6 |
142
- | `cortecs/qwen3.8-27b` | 262K | | | | | | $0.33 | $2 |
142
+ | `cortecs/qwen3.8-27b` | 262K | | | | | | $0.10 | $0.40 |
143
143
  | `cortecs/qwen3.8-flash-next` | 262K | | | | | | $0.20 | $0.50 |
144
144
  | `cortecs/qwen3guard-gen-0.6b` | 32K | | | | | | — | — |
145
145
  | `cortecs/qwen3guard-gen-8b` | 32K | | | | | | — | — |
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![Deep Infra logo](https://models.dev/logos/deepinfra.svg)Deep Infra
6
6
 
7
- Access 63 Deep Infra models through Mastra's model router. Authentication is handled automatically using the `DEEPINFRA_API_KEY` environment variable.
7
+ Access 68 Deep Infra models through Mastra's model router. Authentication is handled automatically using the `DEEPINFRA_API_KEY` environment variable.
8
8
 
9
9
  Learn more in the [Deep Infra documentation](https://deepinfra.com/models).
10
10
 
@@ -49,13 +49,16 @@ for await (const chunk of stream) {
49
49
  | `deepinfra/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp` | 1.0M | | | | | | $0.44 | $1 |
50
50
  | `deepinfra/deepseek-ai/DeepSeek-V4-Pro` | 1.0M | | | | | | $1 | $3 |
51
51
  | `deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813` | 1.0M | | | | | | $1 | $3 |
52
+ | `deepinfra/deepseek-ai/DeepSeek-V4.1-Flash` | 1.0M | | | | | | $0.20 | $0.60 |
53
+ | `deepinfra/google/gemma-3-12b-it` | 131K | | | | | | $0.05 | $0.15 |
54
+ | `deepinfra/google/gemma-3-27b-it` | 131K | | | | | | $0.08 | $0.16 |
55
+ | `deepinfra/google/gemma-3-4b-it` | 131K | | | | | | $0.05 | $0.10 |
52
56
  | `deepinfra/google/gemma-4-26B-A4B-it` | 262K | | | | | | $0.07 | $0.34 |
53
57
  | `deepinfra/google/gemma-4-31B-it` | 262K | | | | | | $0.13 | $0.38 |
54
58
  | `deepinfra/google/gemma-4-E4B-it` | 131K | | | | | | $0.02 | $0.10 |
55
59
  | `deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo` | 131K | | | | | | $0.10 | $0.32 |
56
60
  | `deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8` | 1.0M | | | | | | $0.20 | $0.80 |
57
61
  | `deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct` | 328K | | | | | | $0.10 | $0.30 |
58
- | `deepinfra/MiniMaxAI/MiniMax-M2.7` | 197K | | | | | | $0.25 | $1 |
59
62
  | `deepinfra/MiniMaxAI/MiniMax-M3` | 524K | | | | | | $0.28 | $1 |
60
63
  | `deepinfra/moonshotai/Kimi-K2.6` | 262K | | | | | | $0.75 | $4 |
61
64
  | `deepinfra/moonshotai/Kimi-K2.7-Code` | 262K | | | | | | $0.68 | $3 |
@@ -80,17 +83,16 @@ for await (const chunk of stream) {
80
83
  | `deepinfra/Qwen/Qwen3.7-Max` | 256K | | | | | | $3 | $8 |
81
84
  | `deepinfra/Qwen/Qwen3.8-2.4T-A95B` | 262K | | | | | | $2 | $6 |
82
85
  | `deepinfra/Qwen/Qwen3.8-27B` | 262K | | | | | | $0.40 | $3 |
86
+ | `deepinfra/Qwen/Qwen3.8-Flash` | 1.0M | | | | | | $0.11 | $0.38 |
83
87
  | `deepinfra/Qwen/Qwen3.8-Max` | 256K | | | | | | $2 | $5 |
84
88
  | `deepinfra/stepfun-ai/Step-3.7-Flash` | 262K | | | | | | $0.20 | $1 |
85
89
  | `deepinfra/tencent/Hy3` | 262K | | | | | | $0.14 | $0.58 |
86
90
  | `deepinfra/thinkingmachines/Inkling` | 524K | | | | | | $0.95 | $4 |
87
91
  | `deepinfra/thinkingmachines/Inkling-Small` | 524K | | | | | | $0.45 | $1 |
88
- | `deepinfra/XiaomiMiMo/MiMo-V2.5` | 262K | | | | | | $0.40 | $2 |
92
+ | `deepinfra/XiaomiMiMo/MiMo-V2.5` | 262K | | | | | | $0.14 | $0.28 |
89
93
  | `deepinfra/XiaomiMiMo/MiMo-V2.5-Pro` | 1.0M | | | | | | $1 | $3 |
90
94
  | `deepinfra/zai-org/GLM-4.6` | 203K | | | | | | $0.50 | $2 |
91
95
  | `deepinfra/zai-org/GLM-4.7` | 203K | | | | | | $0.40 | $2 |
92
- | `deepinfra/zai-org/GLM-4.7-Flash` | 203K | | | | | | $0.06 | $0.40 |
93
- | `deepinfra/zai-org/GLM-5` | 203K | | | | | | $0.60 | $2 |
94
96
  | `deepinfra/zai-org/GLM-5.1` | 203K | | | | | | $1 | $4 |
95
97
  | `deepinfra/zai-org/GLM-5.2` | 1.0M | | | | | | $0.75 | $2 |
96
98
  | `deepinfra/zai-org/GLM-5.3` | 1.0M | | | | | | $1 | $4 |