ai 7.0.103 → 7.0.104

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/CHANGELOG.md +17 -0
  2. package/dist/index.d.ts +37 -7
  3. package/dist/index.js +522 -298
  4. package/dist/index.js.map +1 -1
  5. package/dist/internal/index.d.ts +1 -0
  6. package/dist/internal/index.js +4 -1
  7. package/dist/internal/index.js.map +1 -1
  8. package/docs/03-ai-sdk-core/18-code-mode.mdx +11 -0
  9. package/docs/03-ai-sdk-core/19-tool-search.mdx +80 -0
  10. package/docs/03-ai-sdk-core/32-evaluation.mdx +152 -7
  11. package/docs/03-ai-sdk-core/45-provider-management.mdx +10 -0
  12. package/docs/03-ai-sdk-core/index.mdx +6 -0
  13. package/docs/07-reference/01-ai-sdk-core/14-evaluate.mdx +22 -9
  14. package/docs/07-reference/01-ai-sdk-core/20-tool.mdx +7 -0
  15. package/docs/07-reference/01-ai-sdk-core/22-dynamic-tool.mdx +7 -0
  16. package/docs/07-reference/01-ai-sdk-core/23-tool-search.mdx +75 -0
  17. package/docs/07-reference/01-ai-sdk-core/40-provider-registry.mdx +17 -0
  18. package/docs/07-reference/01-ai-sdk-core/42-custom-provider.mdx +16 -0
  19. package/docs/07-reference/01-ai-sdk-core/index.mdx +6 -0
  20. package/docs/07-reference/05-ai-sdk-errors/ai-no-such-model-error.mdx +4 -0
  21. package/docs/07-reference/05-ai-sdk-errors/ai-no-such-provider-error.mdx +4 -0
  22. package/package.json +12 -12
  23. package/src/evaluate/evaluate.ts +4 -10
  24. package/src/evaluate/evaluation-provider.ts +6 -0
  25. package/src/evaluate/evaluation-result.ts +1 -1
  26. package/src/generate-text/generate-text.ts +9 -1
  27. package/src/generate-text/stream-text.ts +9 -1
  28. package/src/generate-text/tool-caller-configuration.ts +1 -1
  29. package/src/global.ts +1 -0
  30. package/src/index.ts +1 -0
  31. package/src/model/resolve-model.ts +55 -10
  32. package/src/prompt/prepare-tools.ts +1 -1
  33. package/src/realtime/browser-realtime-transport.ts +12 -1
  34. package/src/realtime/realtime-event-channel.ts +4 -0
  35. package/src/realtime/realtime-session.ts +9 -4
  36. package/src/registry/custom-provider.ts +37 -0
  37. package/src/registry/index.ts +2 -0
  38. package/src/registry/no-such-provider-error.ts +2 -1
  39. package/src/registry/provider-registry.ts +65 -6
  40. package/src/tool-search/prepare-tool-search.ts +143 -0
  41. package/src/tool-search/tool-search.ts +59 -0
@@ -173,6 +173,17 @@ const user = await tools['lookup-user']({ userId: 'user_123' });
173
173
  return { id: user.id, plan: user.plan };
174
174
  ```
175
175
 
176
+ ### Searching Deferred Tools
177
+
178
+ Use `toolSearch()` with `deferLoading: true` to expose tools only when the model
179
+ needs them. With `toolDiscovery: 'conversation'`, discovered definitions arrive
180
+ in user messages, preserving the tool-definition cache by keeping the
181
+ provider-visible code mode tool unchanged. Actual prompt-cache reuse depends on
182
+ the provider.
183
+
184
+ See [Tool Search](/docs/ai-sdk-core/tool-search) for direct-calling and code mode
185
+ examples.
186
+
176
187
  ## Writing Code Mode Programs
177
188
 
178
189
  Generated programs support:
@@ -0,0 +1,80 @@
1
+ ---
2
+ title: Tool Search
3
+ description: Let models discover tools on demand, with direct calling or cache-preserving code mode.
4
+ ---
5
+
6
+ # Tool Search
7
+
8
+ `toolSearch()` lets a model find the tools it needs without loading every tool's
9
+ definition into its initial context. Register tools with `deferLoading: true`;
10
+ search matches their names and descriptions and makes them available on the
11
+ **next model step**.
12
+
13
+ Use it with `generateText`, `streamText`, or `ToolLoopAgent`. The factory takes no
14
+ arguments; the model supplies a search query.
15
+
16
+ ## Direct Tool Calling
17
+
18
+ The model initially sees only `search`. After searching, matching definitions are
19
+ added to the provider's tool list and the model calls those tools directly. This
20
+ changes the tool definitions and can invalidate the cached prompt prefix.
21
+
22
+ ```ts
23
+ import { generateText, isStepCount, tool, toolSearch } from 'ai';
24
+ import { z } from 'zod/v4';
25
+
26
+ const weather = tool({
27
+ deferLoading: true,
28
+ description: 'Get the weather forecast for a city.',
29
+ inputSchema: z.object({ city: z.string() }),
30
+ execute: async ({ city }) => ({ city, forecast: 'Rain tomorrow.' }),
31
+ });
32
+
33
+ const result = await generateText({
34
+ model: __MODEL__,
35
+ tools: { search: toolSearch(), weather },
36
+ stopWhen: isStepCount(5),
37
+ prompt: 'Will it rain in Bangalore tomorrow?',
38
+ });
39
+ ```
40
+
41
+ ## Code Mode
42
+
43
+ With [code mode](/docs/ai-sdk-core/code-mode), the model searches and calls tools
44
+ through generated code. Set `toolDiscovery: 'conversation'` to announce discovered
45
+ definitions in user messages while keeping the provider-visible code tool
46
+ unchanged. **This preserves the tool-definition cache as tools are discovered.**
47
+ Actual prompt-cache reuse depends on the provider.
48
+
49
+ Using the `weather` tool defined above:
50
+
51
+ ```ts
52
+ import { experimental_codeModeTool as codeModeTool } from '@ai-sdk/code-mode';
53
+
54
+ const result = await generateText({
55
+ model: __MODEL__,
56
+ tools: {
57
+ code: codeModeTool({ toolDiscovery: 'conversation' }),
58
+ search: toolSearch(),
59
+ weather,
60
+ },
61
+ experimental_toolCallers: {
62
+ search: ['code'],
63
+ weather: ['code'],
64
+ },
65
+ stopWhen: isStepCount(5),
66
+ prompt: 'Will it rain in Bangalore tomorrow?',
67
+ });
68
+ ```
69
+
70
+ The model first runs `tools.search({ query: 'weather forecast' })`. On the next
71
+ step, it receives the updated capability catalog and can call
72
+ `tools.weather({ city: 'Bangalore' })`.
73
+
74
+ In both modes, search loads up to five matches, respects `activeTools` and caller
75
+ routing, and keeps discovered tools available for the rest of the generation.
76
+ MCP client tools work too: add `deferLoading: true` to the tools returned by
77
+ `client.tools()`.
78
+
79
+ See the [`toolSearch()` reference](/docs/reference/ai-sdk-core/tool-search) for
80
+ input, output, and matching details.
@@ -10,9 +10,9 @@ an evaluation model. State can be a string, JSON object, or JSON array. An array
10
10
  is one state, not a batch of unrelated inputs.
11
11
 
12
12
  This API and the evaluation model specification are experimental and may change
13
- in patch releases. Pass a model instance implementing
14
- `Experimental_EvaluationModelV4`; string model IDs and registry integration are
15
- not supported yet.
13
+ in patch releases. Pass an evaluation model instance, resolve a model through a
14
+ provider registry, or use a string ID with an evaluation-capable default provider.
15
+ Evaluation does not fall back to Vercel AI Gateway.
16
16
 
17
17
  ```ts
18
18
  import { experimental_evaluate, type Experimental_EvaluationModel } from 'ai';
@@ -44,6 +44,123 @@ async function triage(model: Experimental_EvaluationModel, message: string) {
44
44
  }
45
45
  ```
46
46
 
47
+ ## Provider models
48
+
49
+ Use a provider's `evaluationModel` factory:
50
+
51
+ | Provider | Example model |
52
+ | ----------- | -------------------------------------------------------- |
53
+ | TypeSafe AI | `typeSafeAi.evaluationModel('jev-latest')` |
54
+ | OpenAI | `openai.evaluationModel('gpt-5.6-luna')` |
55
+ | Anthropic | `anthropic.evaluationModel('claude-haiku-4-5-20251001')` |
56
+ | Google | `google.evaluationModel('gemini-3.5-flash-lite')` |
57
+
58
+ TypeSafe AI supplies native Choice, Score, and Boolean evaluations. OpenAI,
59
+ Anthropic, and Google adapt structured language-model output for all three types.
60
+ Boolean answers contain prompted estimates of P(true), validated to be finite
61
+ and in `[0, 1]`. These estimates are not guaranteed to be calibrated. Choice and
62
+ Score answers do not include probability distributions. Select a model that
63
+ supports the provider's structured-output API; judge quality against your own
64
+ labeled examples before choosing a model for a task.
65
+
66
+ The language-model adapters evaluate all questions in one prompt. They do not
67
+ provide TypeSafe's native independent-question execution semantics. The example
68
+ model IDs demonstrate API compatibility; they are not benchmark-selected defaults.
69
+
70
+ Language-model adapters request `reasoning: 'none'` by default. Each provider
71
+ maps this setting to its model's reasoning controls; it does not guarantee that
72
+ every model runs without thinking. To enable reasoning for more demanding
73
+ evaluations, pass the model's supported reasoning settings through
74
+ `providerOptions`, which take precedence over this default. For example, use
75
+ `providerOptions: { openai: { reasoningEffort: 'high' } }` with an OpenAI model
76
+ that supports that effort.
77
+
78
+ ## Model aliases and registries
79
+
80
+ Use `customProvider` to give models application-specific names, then register
81
+ providers with `createProviderRegistry`:
82
+
83
+ ```ts
84
+ import { typeSafeAi } from '@ai-sdk/typesafe-ai';
85
+ import { openai } from '@ai-sdk/openai';
86
+ import {
87
+ customProvider,
88
+ createProviderRegistry,
89
+ experimental_evaluate,
90
+ } from 'ai';
91
+
92
+ const registry = createProviderRegistry({
93
+ triage: customProvider({
94
+ evaluationModels: {
95
+ native: typeSafeAi.evaluationModel('jev-latest'),
96
+ compact: openai.evaluationModel('gpt-5.6-luna'),
97
+ },
98
+ fallbackProvider: typeSafeAi,
99
+ }),
100
+ openai,
101
+ });
102
+
103
+ const result = await experimental_evaluate({
104
+ model: registry.evaluationModel('triage:native'),
105
+ state: 'I was charged twice.',
106
+ questions: {
107
+ department: {
108
+ type: 'choice',
109
+ instructions: 'Which team should handle this?',
110
+ criteria: { billing: 'Charges and refunds', support: 'Other requests' },
111
+ },
112
+ },
113
+ });
114
+
115
+ result.answers.department.choice; // 'billing' | 'support'
116
+ ```
117
+
118
+ Registry IDs use `providerId:modelId`; the `separator` option changes the
119
+ separator. Only the first separator is used, so model IDs can contain it.
120
+ Custom aliases take precedence over a fallback provider. A fallback only resolves
121
+ unknown model IDs; it does not retry failed evaluations or substitute a model
122
+ when a question type is unsupported. Registered providers keep their own
123
+ credentials and settings. Registry language/image middleware does not wrap
124
+ evaluation models.
125
+
126
+ ### Default-provider strings
127
+
128
+ To use strings directly, configure the default provider once at application
129
+ startup:
130
+
131
+ ```ts
132
+ // Using the registry above, expose an alias through a custom provider:
133
+ globalThis.AI_SDK_DEFAULT_PROVIDER = customProvider({
134
+ evaluationModels: { native: registry.evaluationModel('triage:native') },
135
+ });
136
+
137
+ const result = await experimental_evaluate({
138
+ model: 'native',
139
+ state: 'I was charged twice.',
140
+ questions: {
141
+ refund: {
142
+ type: 'boolean',
143
+ instructions: 'Is the customer asking for a refund?',
144
+ },
145
+ },
146
+ });
147
+ ```
148
+
149
+ A direct provider such as `typeSafeAi` can also be the default; then use its
150
+ unprefixed model ID, such as `'jev-latest'`. String values in `evaluationModels`
151
+ also resolve through this global default. Prefer model instances in aliases to
152
+ avoid resolution cycles. Global configuration affects other AI SDK functions
153
+ too; avoid changing it per request in a shared process.
154
+
155
+ Strings require an explicitly configured provider with an `evaluationModel`
156
+ method. There is no assumed Gateway evaluation support. Missing providers throw
157
+ `NoSuchProviderError`; missing models or evaluation capabilities throw
158
+ `NoSuchModelError` with `modelType: 'evaluationModel'`. Unsupported model versions
159
+ throw `UnsupportedModelVersionError`, including models returned by registries.
160
+ The stable `ProviderV4` and `ProviderRegistryProvider` interfaces are unchanged;
161
+ keep the inferred registry type, or use `Experimental_EvaluationProviderRegistry`,
162
+ to retain its experimental `evaluationModel` method.
163
+
47
164
  ## Question types
48
165
 
49
166
  | Type | Criteria | Answer |
@@ -77,11 +194,27 @@ are preserved, never silently normalized.
77
194
 
78
195
  Choice and Score distributions are optional. Boolean probability is required:
79
196
  `0.98` means a strong yes and `0.02` means a strong no. It is not confidence in
80
- either outcome. The SDK does not promise calibration across providers or
81
- synthesize missing probabilities. Provider-specific confidence statistics belong
82
- in `providerMetadata`.
197
+ either outcome. The SDK does not promise calibration across providers. Structured
198
+ language-model adapters prompt the model to estimate P(true); native evaluation providers return
199
+ their API probabilities. Provider-specific confidence statistics belong
200
+ in `providerMetadata`. TypeSafe exposes its separate Choice/Score confidence
201
+ statistic at `result.providerMetadata?.typesafe?.confidence`, keyed by question
202
+ ID. It is not the selected option's probability or a portable confidence measure.
83
203
 
84
- Choose thresholds in application code:
204
+ Check for optional distributions before using them. For example, an application
205
+ can route only when a provider supplies a sufficiently high selected-option
206
+ probability:
207
+
208
+ ```ts
209
+ const answer = result.answers.department;
210
+ const selectedProbability = answer.probabilities?.[answer.choice];
211
+ if (selectedProbability != null && selectedProbability >= 0.9) {
212
+ // Route automatically; otherwise use the application's review path.
213
+ }
214
+ ```
215
+
216
+ Choose Boolean thresholds in application code, using labeled data from the task
217
+ rather than assuming that the same threshold behaves identically across providers:
85
218
 
86
219
  ```ts
87
220
  if (result.answers.requestsRefund.probability >= 0.8) {
@@ -108,3 +241,15 @@ Unknown token counts stay `undefined`; `totalTokens` is available only when both
108
241
  input and output counts are known.
109
242
 
110
243
  For tests, use `Experimental_EvaluationMockModelV4` from `ai/test`.
244
+
245
+ ## Scope and examples
246
+
247
+ Evaluation currently returns one complete result for one shared state. It does
248
+ not stream answers, perform multilabel classification, or batch unrelated
249
+ states. Run separate calls for separate states. Provider support and judgment
250
+ quality depend on the chosen model; the SDK does not choose a model automatically.
251
+
252
+ Runnable examples are in
253
+ [`examples/ai-functions/src/evaluate`](https://github.com/vercel/ai/tree/main/examples/ai-functions/src/evaluate),
254
+ including basic examples for TypeSafe, OpenAI, Anthropic, and Google, model
255
+ registries, custom aliases, default-provider strings, and probability-based routing.
@@ -468,3 +468,13 @@ const result = await streamText({
468
468
  ```
469
469
 
470
470
  This simplifies provider usage and makes it easier to switch between providers without changing your model references throughout your codebase.
471
+
472
+ ## Experimental evaluation models
473
+
474
+ Custom providers accept `evaluationModels` aliases, and registries expose
475
+ `evaluationModel('provider:model')`. These methods return model instances for
476
+ `experimental_evaluate`. An explicitly configured default provider with an
477
+ `evaluationModel` method also enables direct string IDs; evaluation does not
478
+ implicitly use Gateway. Registry middleware for language and image models does
479
+ not apply to evaluation. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
480
+ for aliases, default-provider configuration, and capability differences.
@@ -28,6 +28,12 @@ description: Learn about AI SDK Core.
28
28
  description: 'Learn how to do tool calling with AI SDK Core.',
29
29
  href: '/docs/ai-sdk-core/tools-and-tool-calling',
30
30
  },
31
+ {
32
+ title: 'Tool Search',
33
+ description:
34
+ 'Discover tools on demand with direct calling or cache-preserving code mode.',
35
+ href: '/docs/ai-sdk-core/tool-search',
36
+ },
31
37
  {
32
38
  title: 'Code Mode',
33
39
  description:
@@ -14,15 +14,15 @@ state. See [Evaluation](/docs/ai-sdk-core/evaluation) for examples and semantics
14
14
 
15
15
  ## Parameters
16
16
 
17
- | Parameter | Type | Description |
18
- | ----------------- | ------------------------------------------------- | ------------------------------------------------------------------------- |
19
- | `model` | `Experimental_EvaluationModel` | Required model instance implementing the experimental v4 evaluation spec. |
20
- | `state` | `string \| object \| array` | Required JSON-compatible shared state. |
21
- | `questions` | `Record<string, Experimental_EvaluationQuestion>` | Required nonempty question map. |
22
- | `maxRetries` | `number` | Nonnegative integer; defaults to 2. |
23
- | `abortSignal` | `AbortSignal` | Cancels evaluation. |
24
- | `headers` | `Record<string, string>` | Additional HTTP headers. |
25
- | `providerOptions` | `ProviderOptions` | Provider-specific options. |
17
+ | Parameter | Type | Description |
18
+ | ----------------- | ------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------- |
19
+ | `model` | `Experimental_EvaluationModel` | Required experimental v4 model instance or a string ID resolved by an explicitly configured evaluation-capable default provider. |
20
+ | `state` | `string \| object \| array` | Required JSON-compatible shared state. |
21
+ | `questions` | `Record<string, Experimental_EvaluationQuestion>` | Required nonempty question map. |
22
+ | `maxRetries` | `number` | Nonnegative integer; defaults to 2. |
23
+ | `abortSignal` | `AbortSignal` | Cancels evaluation. |
24
+ | `headers` | `Record<string, string>` | Additional HTTP headers. |
25
+ | `providerOptions` | `ProviderOptions` | Provider-specific options. |
26
26
 
27
27
  ## Result
28
28
 
@@ -52,3 +52,16 @@ Unsupported types throw `Experimental_EvaluationUnsupportedQuestionTypeError`
52
52
  before provider I/O. Invalid inputs throw `InvalidArgumentError`; malformed
53
53
  answers throw `InvalidResponseDataError`. Invalid answers are not retried.
54
54
  Neither partial results nor missing probability synthesis are supported.
55
+
56
+ ## Model resolution
57
+
58
+ Use `registry.evaluationModel('provider:model')` or
59
+ `customProvider({ evaluationModels: { alias: model } }).evaluationModel('alias')`
60
+ to resolve models. Strings passed directly to `experimental_evaluate` use
61
+ `globalThis.AI_SDK_DEFAULT_PROVIDER.evaluationModel(id)`, when available. Evaluation
62
+ never implicitly falls back to Gateway.
63
+
64
+ Resolution errors use the existing `NoSuchModelError` and `NoSuchProviderError`
65
+ classes with `modelType: 'evaluationModel'`. Model instances and resolved models
66
+ must implement v4; other versions throw `UnsupportedModelVersionError`.
67
+ See [model resolution examples](/docs/ai-sdk-core/evaluation#model-aliases-and-registries).
@@ -65,6 +65,13 @@ export const weatherTool = tool({
65
65
  description:
66
66
  'Information about the purpose of the tool including details on how and when it can be used by the model. Provide a string for a fixed description, or a function to derive the description from the tool-specific context and optional experimental sandbox before each model call.',
67
67
  },
68
+ {
69
+ name: 'deferLoading',
70
+ isOptional: true,
71
+ type: 'boolean',
72
+ description:
73
+ "Keep this tool out of the model context until toolSearch discovers it. Supports direct calling or code mode with toolDiscovery: 'conversation'. Discovered tools become available on the next model step. Defaults to false.",
74
+ },
68
75
  {
69
76
  name: 'title',
70
77
  isOptional: true,
@@ -58,6 +58,13 @@ export const customTool = dynamicTool({
58
58
  description:
59
59
  'Information about the purpose of the tool including details on how and when it can be used by the model. Provide a string for a fixed description, or a function to derive the description from the tool-specific context and optional experimental sandbox before each model call.'
60
60
  },
61
+ {
62
+ name: 'deferLoading',
63
+ isOptional: true,
64
+ type: 'boolean',
65
+ description:
66
+ "Keep this tool out of the model context until toolSearch discovers it. Supports direct calling or code mode with toolDiscovery: 'conversation'. Discovered tools become available on the next model step. Defaults to false.",
67
+ },
61
68
  {
62
69
  name: 'title',
63
70
  isOptional: true,
@@ -0,0 +1,75 @@
1
+ ---
2
+ title: toolSearch
3
+ description: Search deferred tools and load their definitions on demand for direct calling or code mode.
4
+ ---
5
+
6
+ # `toolSearch()`
7
+
8
+ Creates a tool that searches the surrounding generation's deferred tools by name
9
+ and description. The factory takes no arguments. Use it with `generateText`,
10
+ `streamText`, or `ToolLoopAgent`, either with direct tool calling or with code mode
11
+ configured with `toolDiscovery: 'conversation'`.
12
+
13
+ ```ts
14
+ import { toolSearch } from 'ai';
15
+
16
+ const search = toolSearch();
17
+ ```
18
+
19
+ See the [Tool Search guide](/docs/ai-sdk-core/tool-search) for direct-calling and
20
+ code mode examples, including how code mode preserves the tool-definition cache.
21
+
22
+ ## Model Input
23
+
24
+ The model supplies the following input to the search tool:
25
+
26
+ ```json
27
+ { "query": "weather forecast" }
28
+ ```
29
+
30
+ `query` is a required, nonempty string of search keywords. Search is local and
31
+ case-insensitive, matching words in tool names and descriptions. Camel-case names
32
+ are split into words. Name matches rank above description matches; equal scores
33
+ preserve registration order. Function descriptions are resolved with the current
34
+ tool context and sandbox. No embedding service or additional model call is used.
35
+
36
+ ## Output
37
+
38
+ ```ts
39
+ {
40
+ tools: [
41
+ { name: 'getForecast', description: 'Get the weather forecast for a city.' },
42
+ ],
43
+ }
44
+ ```
45
+
46
+ Results contain at most five matching tools, with their names and optional
47
+ descriptions. They do not include schemas. No matches returns `{ tools: [] }`.
48
+ Every returned match is queued for discovery. Newly discovered tools can only be
49
+ called on the next model step, after their definitions have been provided. A
50
+ parallel call in the same response as the search cannot use a newly discovered
51
+ tool. For code mode, finish the current execution and wait for the capability
52
+ update.
53
+
54
+ ## Discovery Lifecycle
55
+
56
+ Before the next model step, the SDK makes discovered tools available through
57
+ their configured callers:
58
+
59
+ - **Direct calling:** the provider receives the updated tool definitions.
60
+ - **Code mode:** the SDK appends a user message containing the updated capability
61
+ catalog. The provider-visible code mode definition stays unchanged. Existing
62
+ catalogs remain in the conversation; the latest catalog describes the complete
63
+ currently available tool set.
64
+
65
+ Actual prompt-cache reuse depends on the provider. Code mode search still requires
66
+ `toolDiscovery: 'conversation'`; description discovery and provider callers are
67
+ not supported.
68
+
69
+ Discovered tools remain loaded for the rest of the generation, subject to
70
+ `activeTools`. Search cannot discover tools excluded by `activeTools`. Discovery
71
+ state is isolated between generation calls, including calls that reuse the same
72
+ agent or tool instances.
73
+
74
+ Set a multi-step `stopWhen` condition with `generateText` and `streamText` so the
75
+ model can search, use the discovered tools, and answer.
@@ -298,3 +298,20 @@ The `createProviderRegistry` function returns a `Provider` instance. It has the
298
298
  },
299
299
  ]}
300
300
  />
301
+
302
+ ## Experimental evaluation models
303
+
304
+ The inferred return type also exposes `evaluationModel('providerId:modelId')`,
305
+ returning `Experimental_EvaluationModelV4`. The provider must expose an
306
+ `evaluationModel` factory. Custom separators and model ID
307
+ inference work as they do for video models. Language and image middleware do not
308
+ wrap evaluation models. `ProviderRegistryProvider` remains a stable interface;
309
+ use the inferred return type or `Experimental_EvaluationProviderRegistry` to
310
+ retain experimental evaluation access.
311
+
312
+ Unavailable evaluation capabilities or models throw `NoSuchModelError` with
313
+ `modelType: 'evaluationModel'`; unknown registry providers throw
314
+ `NoSuchProviderError`. These capabilities are structural extensions and are not
315
+ added to the stable `ProviderV4` contract. Evaluation does not assume Gateway
316
+ support. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
317
+ for runnable usage patterns.
@@ -201,3 +201,19 @@ The `customProvider` function returns a `Provider` instance. It has the followin
201
201
  },
202
202
  ]}
203
203
  />
204
+
205
+ ## Experimental evaluation models
206
+
207
+ Pass `evaluationModels: Record<string, Experimental_EvaluationModel>` to define
208
+ aliases for evaluation model instances or string IDs. The returned provider adds
209
+ `evaluationModel(alias): Experimental_EvaluationModelV4`. String aliases resolve
210
+ through an explicitly configured default provider with an `evaluationModel`
211
+ method. Unknown aliases use the fallback provider's evaluation factory when
212
+ available; model failures and unsupported questions do not trigger substitution.
213
+
214
+ Unavailable evaluation capabilities or models throw `NoSuchModelError` with
215
+ `modelType: 'evaluationModel'`; unknown registry providers throw
216
+ `NoSuchProviderError`. These capabilities are structural extensions and are not
217
+ added to the stable `ProviderV4` contract. Evaluation does not assume Gateway
218
+ support. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
219
+ for runnable usage patterns.
@@ -115,6 +115,12 @@ It also contains the following helper functions:
115
115
 
116
116
  <IndexCards
117
117
  cards={[
118
+ {
119
+ title: 'toolSearch()',
120
+ description:
121
+ 'Search deferred tools and load their definitions on demand.',
122
+ href: '/docs/reference/ai-sdk-core/tool-search',
123
+ },
118
124
  {
119
125
  title: 'tool()',
120
126
  description: 'Type inference helper function for tools.',
@@ -24,3 +24,7 @@ if (NoSuchModelError.isInstance(error)) {
24
24
  // Handle the error
25
25
  }
26
26
  ```
27
+
28
+ Experimental evaluation resolution uses `modelType: 'evaluationModel'`. This
29
+ includes unavailable evaluation capabilities and unknown evaluation model or
30
+ provider IDs. See [Evaluation](/docs/ai-sdk-core/evaluation#default-provider-strings).
@@ -26,3 +26,7 @@ if (NoSuchProviderError.isInstance(error)) {
26
26
  // Handle the error
27
27
  }
28
28
  ```
29
+
30
+ Experimental evaluation resolution uses `modelType: 'evaluationModel'`. This
31
+ includes unavailable evaluation capabilities and unknown evaluation model or
32
+ provider IDs. See [Evaluation](/docs/ai-sdk-core/evaluation#default-provider-strings).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ai",
3
- "version": "7.0.103",
3
+ "version": "7.0.104",
4
4
  "type": "module",
5
5
  "description": "AI SDK by Vercel - build apps like ChatGPT, Claude, Gemini, and more with a single interface for any model using the Vercel AI Gateway or go direct to OpenAI, Anthropic, Google, or any other model provider.",
6
6
  "license": "Apache-2.0",
@@ -42,20 +42,20 @@
42
42
  }
43
43
  },
44
44
  "dependencies": {
45
- "@ai-sdk/gateway": "4.0.83",
46
- "@ai-sdk/provider": "4.0.16",
47
- "@ai-sdk/provider-utils": "5.0.42"
45
+ "@ai-sdk/gateway": "4.0.84",
46
+ "@ai-sdk/provider": "4.0.17",
47
+ "@ai-sdk/provider-utils": "5.0.43"
48
48
  },
49
49
  "devDependencies": {
50
- "@ai-sdk/amazon-bedrock": "5.0.85",
51
- "@ai-sdk/deepseek": "3.0.46",
52
- "@ai-sdk/google": "4.0.73",
53
- "@ai-sdk/groq": "4.0.43",
54
- "@ai-sdk/huggingface": "2.0.50",
55
- "@ai-sdk/moonshotai": "3.0.51",
56
- "@ai-sdk/openai": "4.0.68",
50
+ "@ai-sdk/amazon-bedrock": "5.0.86",
51
+ "@ai-sdk/deepseek": "3.0.47",
52
+ "@ai-sdk/google": "4.0.74",
53
+ "@ai-sdk/groq": "4.0.44",
54
+ "@ai-sdk/huggingface": "2.0.51",
55
+ "@ai-sdk/moonshotai": "3.0.52",
56
+ "@ai-sdk/openai": "4.0.69",
57
57
  "@ai-sdk/test-server": "2.0.1",
58
- "@ai-sdk/xai": "5.0.1",
58
+ "@ai-sdk/xai": "5.0.2",
59
59
  "@edge-runtime/vm": "^5.0.0",
60
60
  "@smithy/eventstream-codec": "^4.3.3",
61
61
  "@smithy/util-utf8": "^4.3.3",
@@ -6,7 +6,7 @@ import {
6
6
  withUserAgentSuffix,
7
7
  type ProviderOptions,
8
8
  } from '@ai-sdk/provider-utils';
9
- import { UnsupportedModelVersionError } from '../error/unsupported-model-version-error';
9
+ import { resolveEvaluationModel } from '../model/resolve-model';
10
10
  import { logWarnings } from '../logger/log-warnings';
11
11
  import { prepareRetries } from '../util/prepare-retries';
12
12
  import { VERSION } from '../version';
@@ -24,7 +24,7 @@ import {
24
24
  export async function evaluate<
25
25
  const QUESTIONS extends Record<string, EvaluationQuestion>,
26
26
  >({
27
- model,
27
+ model: modelArg,
28
28
  state,
29
29
  questions,
30
30
  maxRetries,
@@ -32,7 +32,7 @@ export async function evaluate<
32
32
  headers,
33
33
  providerOptions = {},
34
34
  }: {
35
- /** An evaluation model instance. String model resolution is not yet supported. */
35
+ /** An evaluation model instance or an ID resolved by the configured default provider. */
36
36
  model: EvaluationModel;
37
37
  state: EvaluationModelV4CallOptions['state'];
38
38
  questions: QUESTIONS;
@@ -42,13 +42,7 @@ export async function evaluate<
42
42
  headers?: Record<string, string>;
43
43
  providerOptions?: ProviderOptions;
44
44
  }): Promise<EvaluationResult<QUESTIONS>> {
45
- if (model.specificationVersion !== 'v4') {
46
- throw new UnsupportedModelVersionError({
47
- version: model.specificationVersion,
48
- provider: model.provider,
49
- modelId: model.modelId,
50
- });
51
- }
45
+ const model = resolveEvaluationModel(modelArg);
52
46
 
53
47
  validateEvaluationInput({ state, questions });
54
48
 
@@ -0,0 +1,6 @@
1
+ import type { Experimental_EvaluationModelV4 as EvaluationModelV4 } from '@ai-sdk/provider';
2
+
3
+ /** Structural extension; evaluation is not part of the stable provider contract. */
4
+ export type EvaluationProvider = {
5
+ evaluationModel?: (modelId: string) => EvaluationModelV4;
6
+ };
@@ -4,7 +4,7 @@ import type {
4
4
  Experimental_EvaluationModelV4Result as EvaluationModelV4Result,
5
5
  } from '@ai-sdk/provider';
6
6
 
7
- export type EvaluationModel = EvaluationModelV4;
7
+ export type EvaluationModel = string | EvaluationModelV4;
8
8
  export type EvaluationQuestion = EvaluationModelV4Question;
9
9
 
10
10
  export type EvaluationAnswer<QUESTION extends EvaluationQuestion> =