ai 7.0.103 → 7.0.104
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/dist/index.d.ts +37 -7
- package/dist/index.js +522 -298
- package/dist/index.js.map +1 -1
- package/dist/internal/index.d.ts +1 -0
- package/dist/internal/index.js +4 -1
- package/dist/internal/index.js.map +1 -1
- package/docs/03-ai-sdk-core/18-code-mode.mdx +11 -0
- package/docs/03-ai-sdk-core/19-tool-search.mdx +80 -0
- package/docs/03-ai-sdk-core/32-evaluation.mdx +152 -7
- package/docs/03-ai-sdk-core/45-provider-management.mdx +10 -0
- package/docs/03-ai-sdk-core/index.mdx +6 -0
- package/docs/07-reference/01-ai-sdk-core/14-evaluate.mdx +22 -9
- package/docs/07-reference/01-ai-sdk-core/20-tool.mdx +7 -0
- package/docs/07-reference/01-ai-sdk-core/22-dynamic-tool.mdx +7 -0
- package/docs/07-reference/01-ai-sdk-core/23-tool-search.mdx +75 -0
- package/docs/07-reference/01-ai-sdk-core/40-provider-registry.mdx +17 -0
- package/docs/07-reference/01-ai-sdk-core/42-custom-provider.mdx +16 -0
- package/docs/07-reference/01-ai-sdk-core/index.mdx +6 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-no-such-model-error.mdx +4 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-no-such-provider-error.mdx +4 -0
- package/package.json +12 -12
- package/src/evaluate/evaluate.ts +4 -10
- package/src/evaluate/evaluation-provider.ts +6 -0
- package/src/evaluate/evaluation-result.ts +1 -1
- package/src/generate-text/generate-text.ts +9 -1
- package/src/generate-text/stream-text.ts +9 -1
- package/src/generate-text/tool-caller-configuration.ts +1 -1
- package/src/global.ts +1 -0
- package/src/index.ts +1 -0
- package/src/model/resolve-model.ts +55 -10
- package/src/prompt/prepare-tools.ts +1 -1
- package/src/realtime/browser-realtime-transport.ts +12 -1
- package/src/realtime/realtime-event-channel.ts +4 -0
- package/src/realtime/realtime-session.ts +9 -4
- package/src/registry/custom-provider.ts +37 -0
- package/src/registry/index.ts +2 -0
- package/src/registry/no-such-provider-error.ts +2 -1
- package/src/registry/provider-registry.ts +65 -6
- package/src/tool-search/prepare-tool-search.ts +143 -0
- package/src/tool-search/tool-search.ts +59 -0
|
@@ -173,6 +173,17 @@ const user = await tools['lookup-user']({ userId: 'user_123' });
|
|
|
173
173
|
return { id: user.id, plan: user.plan };
|
|
174
174
|
```
|
|
175
175
|
|
|
176
|
+
### Searching Deferred Tools
|
|
177
|
+
|
|
178
|
+
Use `toolSearch()` with `deferLoading: true` to expose tools only when the model
|
|
179
|
+
needs them. With `toolDiscovery: 'conversation'`, discovered definitions arrive
|
|
180
|
+
in user messages, preserving the tool-definition cache by keeping the
|
|
181
|
+
provider-visible code mode tool unchanged. Actual prompt-cache reuse depends on
|
|
182
|
+
the provider.
|
|
183
|
+
|
|
184
|
+
See [Tool Search](/docs/ai-sdk-core/tool-search) for direct-calling and code mode
|
|
185
|
+
examples.
|
|
186
|
+
|
|
176
187
|
## Writing Code Mode Programs
|
|
177
188
|
|
|
178
189
|
Generated programs support:
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: Tool Search
|
|
3
|
+
description: Let models discover tools on demand, with direct calling or cache-preserving code mode.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Tool Search
|
|
7
|
+
|
|
8
|
+
`toolSearch()` lets a model find the tools it needs without loading every tool's
|
|
9
|
+
definition into its initial context. Register tools with `deferLoading: true`;
|
|
10
|
+
search matches their names and descriptions and makes them available on the
|
|
11
|
+
**next model step**.
|
|
12
|
+
|
|
13
|
+
Use it with `generateText`, `streamText`, or `ToolLoopAgent`. The factory takes no
|
|
14
|
+
arguments; the model supplies a search query.
|
|
15
|
+
|
|
16
|
+
## Direct Tool Calling
|
|
17
|
+
|
|
18
|
+
The model initially sees only `search`. After searching, matching definitions are
|
|
19
|
+
added to the provider's tool list and the model calls those tools directly. This
|
|
20
|
+
changes the tool definitions and can invalidate the cached prompt prefix.
|
|
21
|
+
|
|
22
|
+
```ts
|
|
23
|
+
import { generateText, isStepCount, tool, toolSearch } from 'ai';
|
|
24
|
+
import { z } from 'zod/v4';
|
|
25
|
+
|
|
26
|
+
const weather = tool({
|
|
27
|
+
deferLoading: true,
|
|
28
|
+
description: 'Get the weather forecast for a city.',
|
|
29
|
+
inputSchema: z.object({ city: z.string() }),
|
|
30
|
+
execute: async ({ city }) => ({ city, forecast: 'Rain tomorrow.' }),
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
const result = await generateText({
|
|
34
|
+
model: __MODEL__,
|
|
35
|
+
tools: { search: toolSearch(), weather },
|
|
36
|
+
stopWhen: isStepCount(5),
|
|
37
|
+
prompt: 'Will it rain in Bangalore tomorrow?',
|
|
38
|
+
});
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Code Mode
|
|
42
|
+
|
|
43
|
+
With [code mode](/docs/ai-sdk-core/code-mode), the model searches and calls tools
|
|
44
|
+
through generated code. Set `toolDiscovery: 'conversation'` to announce discovered
|
|
45
|
+
definitions in user messages while keeping the provider-visible code tool
|
|
46
|
+
unchanged. **This preserves the tool-definition cache as tools are discovered.**
|
|
47
|
+
Actual prompt-cache reuse depends on the provider.
|
|
48
|
+
|
|
49
|
+
Using the `weather` tool defined above:
|
|
50
|
+
|
|
51
|
+
```ts
|
|
52
|
+
import { experimental_codeModeTool as codeModeTool } from '@ai-sdk/code-mode';
|
|
53
|
+
|
|
54
|
+
const result = await generateText({
|
|
55
|
+
model: __MODEL__,
|
|
56
|
+
tools: {
|
|
57
|
+
code: codeModeTool({ toolDiscovery: 'conversation' }),
|
|
58
|
+
search: toolSearch(),
|
|
59
|
+
weather,
|
|
60
|
+
},
|
|
61
|
+
experimental_toolCallers: {
|
|
62
|
+
search: ['code'],
|
|
63
|
+
weather: ['code'],
|
|
64
|
+
},
|
|
65
|
+
stopWhen: isStepCount(5),
|
|
66
|
+
prompt: 'Will it rain in Bangalore tomorrow?',
|
|
67
|
+
});
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
The model first runs `tools.search({ query: 'weather forecast' })`. On the next
|
|
71
|
+
step, it receives the updated capability catalog and can call
|
|
72
|
+
`tools.weather({ city: 'Bangalore' })`.
|
|
73
|
+
|
|
74
|
+
In both modes, search loads up to five matches, respects `activeTools` and caller
|
|
75
|
+
routing, and keeps discovered tools available for the rest of the generation.
|
|
76
|
+
MCP client tools work too: add `deferLoading: true` to the tools returned by
|
|
77
|
+
`client.tools()`.
|
|
78
|
+
|
|
79
|
+
See the [`toolSearch()` reference](/docs/reference/ai-sdk-core/tool-search) for
|
|
80
|
+
input, output, and matching details.
|
|
@@ -10,9 +10,9 @@ an evaluation model. State can be a string, JSON object, or JSON array. An array
|
|
|
10
10
|
is one state, not a batch of unrelated inputs.
|
|
11
11
|
|
|
12
12
|
This API and the evaluation model specification are experimental and may change
|
|
13
|
-
in patch releases. Pass
|
|
14
|
-
|
|
15
|
-
not
|
|
13
|
+
in patch releases. Pass an evaluation model instance, resolve a model through a
|
|
14
|
+
provider registry, or use a string ID with an evaluation-capable default provider.
|
|
15
|
+
Evaluation does not fall back to Vercel AI Gateway.
|
|
16
16
|
|
|
17
17
|
```ts
|
|
18
18
|
import { experimental_evaluate, type Experimental_EvaluationModel } from 'ai';
|
|
@@ -44,6 +44,123 @@ async function triage(model: Experimental_EvaluationModel, message: string) {
|
|
|
44
44
|
}
|
|
45
45
|
```
|
|
46
46
|
|
|
47
|
+
## Provider models
|
|
48
|
+
|
|
49
|
+
Use a provider's `evaluationModel` factory:
|
|
50
|
+
|
|
51
|
+
| Provider | Example model |
|
|
52
|
+
| ----------- | -------------------------------------------------------- |
|
|
53
|
+
| TypeSafe AI | `typeSafeAi.evaluationModel('jev-latest')` |
|
|
54
|
+
| OpenAI | `openai.evaluationModel('gpt-5.6-luna')` |
|
|
55
|
+
| Anthropic | `anthropic.evaluationModel('claude-haiku-4-5-20251001')` |
|
|
56
|
+
| Google | `google.evaluationModel('gemini-3.5-flash-lite')` |
|
|
57
|
+
|
|
58
|
+
TypeSafe AI supplies native Choice, Score, and Boolean evaluations. OpenAI,
|
|
59
|
+
Anthropic, and Google adapt structured language-model output for all three types.
|
|
60
|
+
Boolean answers contain prompted estimates of P(true), validated to be finite
|
|
61
|
+
and in `[0, 1]`. These estimates are not guaranteed to be calibrated. Choice and
|
|
62
|
+
Score answers do not include probability distributions. Select a model that
|
|
63
|
+
supports the provider's structured-output API; judge quality against your own
|
|
64
|
+
labeled examples before choosing a model for a task.
|
|
65
|
+
|
|
66
|
+
The language-model adapters evaluate all questions in one prompt. They do not
|
|
67
|
+
provide TypeSafe's native independent-question execution semantics. The example
|
|
68
|
+
model IDs demonstrate API compatibility; they are not benchmark-selected defaults.
|
|
69
|
+
|
|
70
|
+
Language-model adapters request `reasoning: 'none'` by default. Each provider
|
|
71
|
+
maps this setting to its model's reasoning controls; it does not guarantee that
|
|
72
|
+
every model runs without thinking. To enable reasoning for more demanding
|
|
73
|
+
evaluations, pass the model's supported reasoning settings through
|
|
74
|
+
`providerOptions`, which take precedence over this default. For example, use
|
|
75
|
+
`providerOptions: { openai: { reasoningEffort: 'high' } }` with an OpenAI model
|
|
76
|
+
that supports that effort.
|
|
77
|
+
|
|
78
|
+
## Model aliases and registries
|
|
79
|
+
|
|
80
|
+
Use `customProvider` to give models application-specific names, then register
|
|
81
|
+
providers with `createProviderRegistry`:
|
|
82
|
+
|
|
83
|
+
```ts
|
|
84
|
+
import { typeSafeAi } from '@ai-sdk/typesafe-ai';
|
|
85
|
+
import { openai } from '@ai-sdk/openai';
|
|
86
|
+
import {
|
|
87
|
+
customProvider,
|
|
88
|
+
createProviderRegistry,
|
|
89
|
+
experimental_evaluate,
|
|
90
|
+
} from 'ai';
|
|
91
|
+
|
|
92
|
+
const registry = createProviderRegistry({
|
|
93
|
+
triage: customProvider({
|
|
94
|
+
evaluationModels: {
|
|
95
|
+
native: typeSafeAi.evaluationModel('jev-latest'),
|
|
96
|
+
compact: openai.evaluationModel('gpt-5.6-luna'),
|
|
97
|
+
},
|
|
98
|
+
fallbackProvider: typeSafeAi,
|
|
99
|
+
}),
|
|
100
|
+
openai,
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
const result = await experimental_evaluate({
|
|
104
|
+
model: registry.evaluationModel('triage:native'),
|
|
105
|
+
state: 'I was charged twice.',
|
|
106
|
+
questions: {
|
|
107
|
+
department: {
|
|
108
|
+
type: 'choice',
|
|
109
|
+
instructions: 'Which team should handle this?',
|
|
110
|
+
criteria: { billing: 'Charges and refunds', support: 'Other requests' },
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
result.answers.department.choice; // 'billing' | 'support'
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Registry IDs use `providerId:modelId`; the `separator` option changes the
|
|
119
|
+
separator. Only the first separator is used, so model IDs can contain it.
|
|
120
|
+
Custom aliases take precedence over a fallback provider. A fallback only resolves
|
|
121
|
+
unknown model IDs; it does not retry failed evaluations or substitute a model
|
|
122
|
+
when a question type is unsupported. Registered providers keep their own
|
|
123
|
+
credentials and settings. Registry language/image middleware does not wrap
|
|
124
|
+
evaluation models.
|
|
125
|
+
|
|
126
|
+
### Default-provider strings
|
|
127
|
+
|
|
128
|
+
To use strings directly, configure the default provider once at application
|
|
129
|
+
startup:
|
|
130
|
+
|
|
131
|
+
```ts
|
|
132
|
+
// Using the registry above, expose an alias through a custom provider:
|
|
133
|
+
globalThis.AI_SDK_DEFAULT_PROVIDER = customProvider({
|
|
134
|
+
evaluationModels: { native: registry.evaluationModel('triage:native') },
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
const result = await experimental_evaluate({
|
|
138
|
+
model: 'native',
|
|
139
|
+
state: 'I was charged twice.',
|
|
140
|
+
questions: {
|
|
141
|
+
refund: {
|
|
142
|
+
type: 'boolean',
|
|
143
|
+
instructions: 'Is the customer asking for a refund?',
|
|
144
|
+
},
|
|
145
|
+
},
|
|
146
|
+
});
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
A direct provider such as `typeSafeAi` can also be the default; then use its
|
|
150
|
+
unprefixed model ID, such as `'jev-latest'`. String values in `evaluationModels`
|
|
151
|
+
also resolve through this global default. Prefer model instances in aliases to
|
|
152
|
+
avoid resolution cycles. Global configuration affects other AI SDK functions
|
|
153
|
+
too; avoid changing it per request in a shared process.
|
|
154
|
+
|
|
155
|
+
Strings require an explicitly configured provider with an `evaluationModel`
|
|
156
|
+
method. There is no assumed Gateway evaluation support. Missing providers throw
|
|
157
|
+
`NoSuchProviderError`; missing models or evaluation capabilities throw
|
|
158
|
+
`NoSuchModelError` with `modelType: 'evaluationModel'`. Unsupported model versions
|
|
159
|
+
throw `UnsupportedModelVersionError`, including models returned by registries.
|
|
160
|
+
The stable `ProviderV4` and `ProviderRegistryProvider` interfaces are unchanged;
|
|
161
|
+
keep the inferred registry type, or use `Experimental_EvaluationProviderRegistry`,
|
|
162
|
+
to retain its experimental `evaluationModel` method.
|
|
163
|
+
|
|
47
164
|
## Question types
|
|
48
165
|
|
|
49
166
|
| Type | Criteria | Answer |
|
|
@@ -77,11 +194,27 @@ are preserved, never silently normalized.
|
|
|
77
194
|
|
|
78
195
|
Choice and Score distributions are optional. Boolean probability is required:
|
|
79
196
|
`0.98` means a strong yes and `0.02` means a strong no. It is not confidence in
|
|
80
|
-
either outcome. The SDK does not promise calibration across providers
|
|
81
|
-
|
|
82
|
-
|
|
197
|
+
either outcome. The SDK does not promise calibration across providers. Structured
|
|
198
|
+
language-model adapters prompt the model to estimate P(true); native evaluation providers return
|
|
199
|
+
their API probabilities. Provider-specific confidence statistics belong
|
|
200
|
+
in `providerMetadata`. TypeSafe exposes its separate Choice/Score confidence
|
|
201
|
+
statistic at `result.providerMetadata?.typesafe?.confidence`, keyed by question
|
|
202
|
+
ID. It is not the selected option's probability or a portable confidence measure.
|
|
83
203
|
|
|
84
|
-
|
|
204
|
+
Check for optional distributions before using them. For example, an application
|
|
205
|
+
can route only when a provider supplies a sufficiently high selected-option
|
|
206
|
+
probability:
|
|
207
|
+
|
|
208
|
+
```ts
|
|
209
|
+
const answer = result.answers.department;
|
|
210
|
+
const selectedProbability = answer.probabilities?.[answer.choice];
|
|
211
|
+
if (selectedProbability != null && selectedProbability >= 0.9) {
|
|
212
|
+
// Route automatically; otherwise use the application's review path.
|
|
213
|
+
}
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Choose Boolean thresholds in application code, using labeled data from the task
|
|
217
|
+
rather than assuming that the same threshold behaves identically across providers:
|
|
85
218
|
|
|
86
219
|
```ts
|
|
87
220
|
if (result.answers.requestsRefund.probability >= 0.8) {
|
|
@@ -108,3 +241,15 @@ Unknown token counts stay `undefined`; `totalTokens` is available only when both
|
|
|
108
241
|
input and output counts are known.
|
|
109
242
|
|
|
110
243
|
For tests, use `Experimental_EvaluationMockModelV4` from `ai/test`.
|
|
244
|
+
|
|
245
|
+
## Scope and examples
|
|
246
|
+
|
|
247
|
+
Evaluation currently returns one complete result for one shared state. It does
|
|
248
|
+
not stream answers, perform multilabel classification, or batch unrelated
|
|
249
|
+
states. Run separate calls for separate states. Provider support and judgment
|
|
250
|
+
quality depend on the chosen model; the SDK does not choose a model automatically.
|
|
251
|
+
|
|
252
|
+
Runnable examples are in
|
|
253
|
+
[`examples/ai-functions/src/evaluate`](https://github.com/vercel/ai/tree/main/examples/ai-functions/src/evaluate),
|
|
254
|
+
including basic examples for TypeSafe, OpenAI, Anthropic, and Google, model
|
|
255
|
+
registries, custom aliases, default-provider strings, and probability-based routing.
|
|
@@ -468,3 +468,13 @@ const result = await streamText({
|
|
|
468
468
|
```
|
|
469
469
|
|
|
470
470
|
This simplifies provider usage and makes it easier to switch between providers without changing your model references throughout your codebase.
|
|
471
|
+
|
|
472
|
+
## Experimental evaluation models
|
|
473
|
+
|
|
474
|
+
Custom providers accept `evaluationModels` aliases, and registries expose
|
|
475
|
+
`evaluationModel('provider:model')`. These methods return model instances for
|
|
476
|
+
`experimental_evaluate`. An explicitly configured default provider with an
|
|
477
|
+
`evaluationModel` method also enables direct string IDs; evaluation does not
|
|
478
|
+
implicitly use Gateway. Registry middleware for language and image models does
|
|
479
|
+
not apply to evaluation. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
|
|
480
|
+
for aliases, default-provider configuration, and capability differences.
|
|
@@ -28,6 +28,12 @@ description: Learn about AI SDK Core.
|
|
|
28
28
|
description: 'Learn how to do tool calling with AI SDK Core.',
|
|
29
29
|
href: '/docs/ai-sdk-core/tools-and-tool-calling',
|
|
30
30
|
},
|
|
31
|
+
{
|
|
32
|
+
title: 'Tool Search',
|
|
33
|
+
description:
|
|
34
|
+
'Discover tools on demand with direct calling or cache-preserving code mode.',
|
|
35
|
+
href: '/docs/ai-sdk-core/tool-search',
|
|
36
|
+
},
|
|
31
37
|
{
|
|
32
38
|
title: 'Code Mode',
|
|
33
39
|
description:
|
|
@@ -14,15 +14,15 @@ state. See [Evaluation](/docs/ai-sdk-core/evaluation) for examples and semantics
|
|
|
14
14
|
|
|
15
15
|
## Parameters
|
|
16
16
|
|
|
17
|
-
| Parameter | Type | Description
|
|
18
|
-
| ----------------- | ------------------------------------------------- |
|
|
19
|
-
| `model` | `Experimental_EvaluationModel` | Required model instance
|
|
20
|
-
| `state` | `string \| object \| array` | Required JSON-compatible shared state.
|
|
21
|
-
| `questions` | `Record<string, Experimental_EvaluationQuestion>` | Required nonempty question map.
|
|
22
|
-
| `maxRetries` | `number` | Nonnegative integer; defaults to 2.
|
|
23
|
-
| `abortSignal` | `AbortSignal` | Cancels evaluation.
|
|
24
|
-
| `headers` | `Record<string, string>` | Additional HTTP headers.
|
|
25
|
-
| `providerOptions` | `ProviderOptions` | Provider-specific options.
|
|
17
|
+
| Parameter | Type | Description |
|
|
18
|
+
| ----------------- | ------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------- |
|
|
19
|
+
| `model` | `Experimental_EvaluationModel` | Required experimental v4 model instance or a string ID resolved by an explicitly configured evaluation-capable default provider. |
|
|
20
|
+
| `state` | `string \| object \| array` | Required JSON-compatible shared state. |
|
|
21
|
+
| `questions` | `Record<string, Experimental_EvaluationQuestion>` | Required nonempty question map. |
|
|
22
|
+
| `maxRetries` | `number` | Nonnegative integer; defaults to 2. |
|
|
23
|
+
| `abortSignal` | `AbortSignal` | Cancels evaluation. |
|
|
24
|
+
| `headers` | `Record<string, string>` | Additional HTTP headers. |
|
|
25
|
+
| `providerOptions` | `ProviderOptions` | Provider-specific options. |
|
|
26
26
|
|
|
27
27
|
## Result
|
|
28
28
|
|
|
@@ -52,3 +52,16 @@ Unsupported types throw `Experimental_EvaluationUnsupportedQuestionTypeError`
|
|
|
52
52
|
before provider I/O. Invalid inputs throw `InvalidArgumentError`; malformed
|
|
53
53
|
answers throw `InvalidResponseDataError`. Invalid answers are not retried.
|
|
54
54
|
Neither partial results nor missing probability synthesis are supported.
|
|
55
|
+
|
|
56
|
+
## Model resolution
|
|
57
|
+
|
|
58
|
+
Use `registry.evaluationModel('provider:model')` or
|
|
59
|
+
`customProvider({ evaluationModels: { alias: model } }).evaluationModel('alias')`
|
|
60
|
+
to resolve models. Strings passed directly to `experimental_evaluate` use
|
|
61
|
+
`globalThis.AI_SDK_DEFAULT_PROVIDER.evaluationModel(id)`, when available. Evaluation
|
|
62
|
+
never implicitly falls back to Gateway.
|
|
63
|
+
|
|
64
|
+
Resolution errors use the existing `NoSuchModelError` and `NoSuchProviderError`
|
|
65
|
+
classes with `modelType: 'evaluationModel'`. Model instances and resolved models
|
|
66
|
+
must implement v4; other versions throw `UnsupportedModelVersionError`.
|
|
67
|
+
See [model resolution examples](/docs/ai-sdk-core/evaluation#model-aliases-and-registries).
|
|
@@ -65,6 +65,13 @@ export const weatherTool = tool({
|
|
|
65
65
|
description:
|
|
66
66
|
'Information about the purpose of the tool including details on how and when it can be used by the model. Provide a string for a fixed description, or a function to derive the description from the tool-specific context and optional experimental sandbox before each model call.',
|
|
67
67
|
},
|
|
68
|
+
{
|
|
69
|
+
name: 'deferLoading',
|
|
70
|
+
isOptional: true,
|
|
71
|
+
type: 'boolean',
|
|
72
|
+
description:
|
|
73
|
+
"Keep this tool out of the model context until toolSearch discovers it. Supports direct calling or code mode with toolDiscovery: 'conversation'. Discovered tools become available on the next model step. Defaults to false.",
|
|
74
|
+
},
|
|
68
75
|
{
|
|
69
76
|
name: 'title',
|
|
70
77
|
isOptional: true,
|
|
@@ -58,6 +58,13 @@ export const customTool = dynamicTool({
|
|
|
58
58
|
description:
|
|
59
59
|
'Information about the purpose of the tool including details on how and when it can be used by the model. Provide a string for a fixed description, or a function to derive the description from the tool-specific context and optional experimental sandbox before each model call.'
|
|
60
60
|
},
|
|
61
|
+
{
|
|
62
|
+
name: 'deferLoading',
|
|
63
|
+
isOptional: true,
|
|
64
|
+
type: 'boolean',
|
|
65
|
+
description:
|
|
66
|
+
"Keep this tool out of the model context until toolSearch discovers it. Supports direct calling or code mode with toolDiscovery: 'conversation'. Discovered tools become available on the next model step. Defaults to false.",
|
|
67
|
+
},
|
|
61
68
|
{
|
|
62
69
|
name: 'title',
|
|
63
70
|
isOptional: true,
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: toolSearch
|
|
3
|
+
description: Search deferred tools and load their definitions on demand for direct calling or code mode.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# `toolSearch()`
|
|
7
|
+
|
|
8
|
+
Creates a tool that searches the surrounding generation's deferred tools by name
|
|
9
|
+
and description. The factory takes no arguments. Use it with `generateText`,
|
|
10
|
+
`streamText`, or `ToolLoopAgent`, either with direct tool calling or with code mode
|
|
11
|
+
configured with `toolDiscovery: 'conversation'`.
|
|
12
|
+
|
|
13
|
+
```ts
|
|
14
|
+
import { toolSearch } from 'ai';
|
|
15
|
+
|
|
16
|
+
const search = toolSearch();
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
See the [Tool Search guide](/docs/ai-sdk-core/tool-search) for direct-calling and
|
|
20
|
+
code mode examples, including how code mode preserves the tool-definition cache.
|
|
21
|
+
|
|
22
|
+
## Model Input
|
|
23
|
+
|
|
24
|
+
The model supplies the following input to the search tool:
|
|
25
|
+
|
|
26
|
+
```json
|
|
27
|
+
{ "query": "weather forecast" }
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
`query` is a required, nonempty string of search keywords. Search is local and
|
|
31
|
+
case-insensitive, matching words in tool names and descriptions. Camel-case names
|
|
32
|
+
are split into words. Name matches rank above description matches; equal scores
|
|
33
|
+
preserve registration order. Function descriptions are resolved with the current
|
|
34
|
+
tool context and sandbox. No embedding service or additional model call is used.
|
|
35
|
+
|
|
36
|
+
## Output
|
|
37
|
+
|
|
38
|
+
```ts
|
|
39
|
+
{
|
|
40
|
+
tools: [
|
|
41
|
+
{ name: 'getForecast', description: 'Get the weather forecast for a city.' },
|
|
42
|
+
],
|
|
43
|
+
}
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Results contain at most five matching tools, with their names and optional
|
|
47
|
+
descriptions. They do not include schemas. No matches returns `{ tools: [] }`.
|
|
48
|
+
Every returned match is queued for discovery. Newly discovered tools can only be
|
|
49
|
+
called on the next model step, after their definitions have been provided. A
|
|
50
|
+
parallel call in the same response as the search cannot use a newly discovered
|
|
51
|
+
tool. For code mode, finish the current execution and wait for the capability
|
|
52
|
+
update.
|
|
53
|
+
|
|
54
|
+
## Discovery Lifecycle
|
|
55
|
+
|
|
56
|
+
Before the next model step, the SDK makes discovered tools available through
|
|
57
|
+
their configured callers:
|
|
58
|
+
|
|
59
|
+
- **Direct calling:** the provider receives the updated tool definitions.
|
|
60
|
+
- **Code mode:** the SDK appends a user message containing the updated capability
|
|
61
|
+
catalog. The provider-visible code mode definition stays unchanged. Existing
|
|
62
|
+
catalogs remain in the conversation; the latest catalog describes the complete
|
|
63
|
+
currently available tool set.
|
|
64
|
+
|
|
65
|
+
Actual prompt-cache reuse depends on the provider. Code mode search still requires
|
|
66
|
+
`toolDiscovery: 'conversation'`; description discovery and provider callers are
|
|
67
|
+
not supported.
|
|
68
|
+
|
|
69
|
+
Discovered tools remain loaded for the rest of the generation, subject to
|
|
70
|
+
`activeTools`. Search cannot discover tools excluded by `activeTools`. Discovery
|
|
71
|
+
state is isolated between generation calls, including calls that reuse the same
|
|
72
|
+
agent or tool instances.
|
|
73
|
+
|
|
74
|
+
Set a multi-step `stopWhen` condition with `generateText` and `streamText` so the
|
|
75
|
+
model can search, use the discovered tools, and answer.
|
|
@@ -298,3 +298,20 @@ The `createProviderRegistry` function returns a `Provider` instance. It has the
|
|
|
298
298
|
},
|
|
299
299
|
]}
|
|
300
300
|
/>
|
|
301
|
+
|
|
302
|
+
## Experimental evaluation models
|
|
303
|
+
|
|
304
|
+
The inferred return type also exposes `evaluationModel('providerId:modelId')`,
|
|
305
|
+
returning `Experimental_EvaluationModelV4`. The provider must expose an
|
|
306
|
+
`evaluationModel` factory. Custom separators and model ID
|
|
307
|
+
inference work as they do for video models. Language and image middleware do not
|
|
308
|
+
wrap evaluation models. `ProviderRegistryProvider` remains a stable interface;
|
|
309
|
+
use the inferred return type or `Experimental_EvaluationProviderRegistry` to
|
|
310
|
+
retain experimental evaluation access.
|
|
311
|
+
|
|
312
|
+
Unavailable evaluation capabilities or models throw `NoSuchModelError` with
|
|
313
|
+
`modelType: 'evaluationModel'`; unknown registry providers throw
|
|
314
|
+
`NoSuchProviderError`. These capabilities are structural extensions and are not
|
|
315
|
+
added to the stable `ProviderV4` contract. Evaluation does not assume Gateway
|
|
316
|
+
support. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
|
|
317
|
+
for runnable usage patterns.
|
|
@@ -201,3 +201,19 @@ The `customProvider` function returns a `Provider` instance. It has the followin
|
|
|
201
201
|
},
|
|
202
202
|
]}
|
|
203
203
|
/>
|
|
204
|
+
|
|
205
|
+
## Experimental evaluation models
|
|
206
|
+
|
|
207
|
+
Pass `evaluationModels: Record<string, Experimental_EvaluationModel>` to define
|
|
208
|
+
aliases for evaluation model instances or string IDs. The returned provider adds
|
|
209
|
+
`evaluationModel(alias): Experimental_EvaluationModelV4`. String aliases resolve
|
|
210
|
+
through an explicitly configured default provider with an `evaluationModel`
|
|
211
|
+
method. Unknown aliases use the fallback provider's evaluation factory when
|
|
212
|
+
available; model failures and unsupported questions do not trigger substitution.
|
|
213
|
+
|
|
214
|
+
Unavailable evaluation capabilities or models throw `NoSuchModelError` with
|
|
215
|
+
`modelType: 'evaluationModel'`; unknown registry providers throw
|
|
216
|
+
`NoSuchProviderError`. These capabilities are structural extensions and are not
|
|
217
|
+
added to the stable `ProviderV4` contract. Evaluation does not assume Gateway
|
|
218
|
+
support. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
|
|
219
|
+
for runnable usage patterns.
|
|
@@ -115,6 +115,12 @@ It also contains the following helper functions:
|
|
|
115
115
|
|
|
116
116
|
<IndexCards
|
|
117
117
|
cards={[
|
|
118
|
+
{
|
|
119
|
+
title: 'toolSearch()',
|
|
120
|
+
description:
|
|
121
|
+
'Search deferred tools and load their definitions on demand.',
|
|
122
|
+
href: '/docs/reference/ai-sdk-core/tool-search',
|
|
123
|
+
},
|
|
118
124
|
{
|
|
119
125
|
title: 'tool()',
|
|
120
126
|
description: 'Type inference helper function for tools.',
|
|
@@ -24,3 +24,7 @@ if (NoSuchModelError.isInstance(error)) {
|
|
|
24
24
|
// Handle the error
|
|
25
25
|
}
|
|
26
26
|
```
|
|
27
|
+
|
|
28
|
+
Experimental evaluation resolution uses `modelType: 'evaluationModel'`. This
|
|
29
|
+
includes unavailable evaluation capabilities and unknown evaluation model or
|
|
30
|
+
provider IDs. See [Evaluation](/docs/ai-sdk-core/evaluation#default-provider-strings).
|
|
@@ -26,3 +26,7 @@ if (NoSuchProviderError.isInstance(error)) {
|
|
|
26
26
|
// Handle the error
|
|
27
27
|
}
|
|
28
28
|
```
|
|
29
|
+
|
|
30
|
+
Experimental evaluation resolution uses `modelType: 'evaluationModel'`. This
|
|
31
|
+
includes unavailable evaluation capabilities and unknown evaluation model or
|
|
32
|
+
provider IDs. See [Evaluation](/docs/ai-sdk-core/evaluation#default-provider-strings).
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ai",
|
|
3
|
-
"version": "7.0.
|
|
3
|
+
"version": "7.0.104",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "AI SDK by Vercel - build apps like ChatGPT, Claude, Gemini, and more with a single interface for any model using the Vercel AI Gateway or go direct to OpenAI, Anthropic, Google, or any other model provider.",
|
|
6
6
|
"license": "Apache-2.0",
|
|
@@ -42,20 +42,20 @@
|
|
|
42
42
|
}
|
|
43
43
|
},
|
|
44
44
|
"dependencies": {
|
|
45
|
-
"@ai-sdk/gateway": "4.0.
|
|
46
|
-
"@ai-sdk/provider": "4.0.
|
|
47
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
45
|
+
"@ai-sdk/gateway": "4.0.84",
|
|
46
|
+
"@ai-sdk/provider": "4.0.17",
|
|
47
|
+
"@ai-sdk/provider-utils": "5.0.43"
|
|
48
48
|
},
|
|
49
49
|
"devDependencies": {
|
|
50
|
-
"@ai-sdk/amazon-bedrock": "5.0.
|
|
51
|
-
"@ai-sdk/deepseek": "3.0.
|
|
52
|
-
"@ai-sdk/google": "4.0.
|
|
53
|
-
"@ai-sdk/groq": "4.0.
|
|
54
|
-
"@ai-sdk/huggingface": "2.0.
|
|
55
|
-
"@ai-sdk/moonshotai": "3.0.
|
|
56
|
-
"@ai-sdk/openai": "4.0.
|
|
50
|
+
"@ai-sdk/amazon-bedrock": "5.0.86",
|
|
51
|
+
"@ai-sdk/deepseek": "3.0.47",
|
|
52
|
+
"@ai-sdk/google": "4.0.74",
|
|
53
|
+
"@ai-sdk/groq": "4.0.44",
|
|
54
|
+
"@ai-sdk/huggingface": "2.0.51",
|
|
55
|
+
"@ai-sdk/moonshotai": "3.0.52",
|
|
56
|
+
"@ai-sdk/openai": "4.0.69",
|
|
57
57
|
"@ai-sdk/test-server": "2.0.1",
|
|
58
|
-
"@ai-sdk/xai": "5.0.
|
|
58
|
+
"@ai-sdk/xai": "5.0.2",
|
|
59
59
|
"@edge-runtime/vm": "^5.0.0",
|
|
60
60
|
"@smithy/eventstream-codec": "^4.3.3",
|
|
61
61
|
"@smithy/util-utf8": "^4.3.3",
|
package/src/evaluate/evaluate.ts
CHANGED
|
@@ -6,7 +6,7 @@ import {
|
|
|
6
6
|
withUserAgentSuffix,
|
|
7
7
|
type ProviderOptions,
|
|
8
8
|
} from '@ai-sdk/provider-utils';
|
|
9
|
-
import {
|
|
9
|
+
import { resolveEvaluationModel } from '../model/resolve-model';
|
|
10
10
|
import { logWarnings } from '../logger/log-warnings';
|
|
11
11
|
import { prepareRetries } from '../util/prepare-retries';
|
|
12
12
|
import { VERSION } from '../version';
|
|
@@ -24,7 +24,7 @@ import {
|
|
|
24
24
|
export async function evaluate<
|
|
25
25
|
const QUESTIONS extends Record<string, EvaluationQuestion>,
|
|
26
26
|
>({
|
|
27
|
-
model,
|
|
27
|
+
model: modelArg,
|
|
28
28
|
state,
|
|
29
29
|
questions,
|
|
30
30
|
maxRetries,
|
|
@@ -32,7 +32,7 @@ export async function evaluate<
|
|
|
32
32
|
headers,
|
|
33
33
|
providerOptions = {},
|
|
34
34
|
}: {
|
|
35
|
-
/** An evaluation model instance
|
|
35
|
+
/** An evaluation model instance or an ID resolved by the configured default provider. */
|
|
36
36
|
model: EvaluationModel;
|
|
37
37
|
state: EvaluationModelV4CallOptions['state'];
|
|
38
38
|
questions: QUESTIONS;
|
|
@@ -42,13 +42,7 @@ export async function evaluate<
|
|
|
42
42
|
headers?: Record<string, string>;
|
|
43
43
|
providerOptions?: ProviderOptions;
|
|
44
44
|
}): Promise<EvaluationResult<QUESTIONS>> {
|
|
45
|
-
|
|
46
|
-
throw new UnsupportedModelVersionError({
|
|
47
|
-
version: model.specificationVersion,
|
|
48
|
-
provider: model.provider,
|
|
49
|
-
modelId: model.modelId,
|
|
50
|
-
});
|
|
51
|
-
}
|
|
45
|
+
const model = resolveEvaluationModel(modelArg);
|
|
52
46
|
|
|
53
47
|
validateEvaluationInput({ state, questions });
|
|
54
48
|
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import type { Experimental_EvaluationModelV4 as EvaluationModelV4 } from '@ai-sdk/provider';
|
|
2
|
+
|
|
3
|
+
/** Structural extension; evaluation is not part of the stable provider contract. */
|
|
4
|
+
export type EvaluationProvider = {
|
|
5
|
+
evaluationModel?: (modelId: string) => EvaluationModelV4;
|
|
6
|
+
};
|
|
@@ -4,7 +4,7 @@ import type {
|
|
|
4
4
|
Experimental_EvaluationModelV4Result as EvaluationModelV4Result,
|
|
5
5
|
} from '@ai-sdk/provider';
|
|
6
6
|
|
|
7
|
-
export type EvaluationModel = EvaluationModelV4;
|
|
7
|
+
export type EvaluationModel = string | EvaluationModelV4;
|
|
8
8
|
export type EvaluationQuestion = EvaluationModelV4Question;
|
|
9
9
|
|
|
10
10
|
export type EvaluationAnswer<QUESTION extends EvaluationQuestion> =
|