ai 7.0.102 → 7.0.104
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -0
- package/dist/index.d.ts +90 -6
- package/dist/index.js +900 -299
- package/dist/index.js.map +1 -1
- package/dist/internal/index.d.ts +1 -0
- package/dist/internal/index.js +5 -1
- package/dist/internal/index.js.map +1 -1
- package/dist/test/index.d.ts +16 -2
- package/dist/test/index.js +17 -0
- package/dist/test/index.js.map +1 -1
- package/docs/03-ai-sdk-core/18-code-mode.mdx +50 -0
- package/docs/03-ai-sdk-core/19-tool-search.mdx +80 -0
- package/docs/03-ai-sdk-core/32-evaluation.mdx +255 -0
- package/docs/03-ai-sdk-core/42-batch.mdx +1 -2
- package/docs/03-ai-sdk-core/45-provider-management.mdx +10 -0
- package/docs/03-ai-sdk-core/index.mdx +6 -0
- package/docs/04-ai-sdk-ui/03-chatbot-message-persistence.mdx +35 -0
- package/docs/06-advanced/11-secure-url-fetching.mdx +8 -2
- package/docs/07-reference/01-ai-sdk-core/14-evaluate.mdx +67 -0
- package/docs/07-reference/01-ai-sdk-core/20-tool.mdx +7 -0
- package/docs/07-reference/01-ai-sdk-core/22-dynamic-tool.mdx +7 -0
- package/docs/07-reference/01-ai-sdk-core/23-tool-search.mdx +75 -0
- package/docs/07-reference/01-ai-sdk-core/32-validate-ui-messages.mdx +12 -0
- package/docs/07-reference/01-ai-sdk-core/33-safe-validate-ui-messages.mdx +12 -0
- package/docs/07-reference/01-ai-sdk-core/40-provider-registry.mdx +17 -0
- package/docs/07-reference/01-ai-sdk-core/42-custom-provider.mdx +16 -0
- package/docs/07-reference/01-ai-sdk-core/index.mdx +6 -0
- package/docs/07-reference/02-ai-sdk-ui/31-convert-to-model-messages.mdx +12 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-evaluation-unsupported-question-type-error.mdx +31 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-no-such-model-error.mdx +4 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-no-such-provider-error.mdx +4 -0
- package/package.json +12 -12
- package/src/error/index.ts +1 -0
- package/src/evaluate/evaluate.ts +106 -0
- package/src/evaluate/evaluation-provider.ts +6 -0
- package/src/evaluate/evaluation-result.ts +39 -0
- package/src/evaluate/index.ts +7 -0
- package/src/evaluate/validate-evaluation.ts +298 -0
- package/src/generate-text/generate-text.ts +25 -13
- package/src/generate-text/stream-text.ts +15 -2
- package/src/generate-text/tool-caller-configuration.ts +59 -5
- package/src/global.ts +1 -0
- package/src/index.ts +2 -0
- package/src/model/resolve-model.ts +55 -10
- package/src/prompt/prepare-tools.ts +1 -1
- package/src/realtime/browser-realtime-transport.ts +12 -1
- package/src/realtime/realtime-event-channel.ts +4 -0
- package/src/realtime/realtime-session.ts +9 -4
- package/src/registry/custom-provider.ts +37 -0
- package/src/registry/index.ts +2 -0
- package/src/registry/no-such-provider-error.ts +2 -1
- package/src/registry/provider-registry.ts +65 -6
- package/src/test/evaluation-mock-model-v4.ts +27 -0
- package/src/tool-search/prepare-tool-search.ts +143 -0
- package/src/tool-search/tool-search.ts +59 -0
- package/src/ui/convert-to-model-messages.ts +3 -0
- package/src/ui/process-ui-message-stream.ts +4 -0
- package/src/ui/ui-messages.ts +5 -1
- package/src/ui/validate-ui-messages.ts +3 -0
- package/src/ui/warn-if-ui-message-has-deprecated-raw-input.ts +36 -0
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: Evaluation
|
|
3
|
+
description: Evaluate Choice, Score, and Boolean questions against shared state.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Evaluation
|
|
7
|
+
|
|
8
|
+
`experimental_evaluate` evaluates named questions against one shared state using
|
|
9
|
+
an evaluation model. State can be a string, JSON object, or JSON array. An array
|
|
10
|
+
is one state, not a batch of unrelated inputs.
|
|
11
|
+
|
|
12
|
+
This API and the evaluation model specification are experimental and may change
|
|
13
|
+
in patch releases. Pass an evaluation model instance, resolve a model through a
|
|
14
|
+
provider registry, or use a string ID with an evaluation-capable default provider.
|
|
15
|
+
Evaluation does not fall back to Vercel AI Gateway.
|
|
16
|
+
|
|
17
|
+
```ts
|
|
18
|
+
import { experimental_evaluate, type Experimental_EvaluationModel } from 'ai';
|
|
19
|
+
|
|
20
|
+
async function triage(model: Experimental_EvaluationModel, message: string) {
|
|
21
|
+
return experimental_evaluate({
|
|
22
|
+
model,
|
|
23
|
+
state: { message },
|
|
24
|
+
questions: {
|
|
25
|
+
department: {
|
|
26
|
+
type: 'choice',
|
|
27
|
+
instructions: 'Which team should handle this?',
|
|
28
|
+
criteria: {
|
|
29
|
+
billing: 'Payments and refunds',
|
|
30
|
+
support: 'Other requests',
|
|
31
|
+
},
|
|
32
|
+
},
|
|
33
|
+
severity: {
|
|
34
|
+
type: 'score',
|
|
35
|
+
instructions: 'How severe is the issue?',
|
|
36
|
+
criteria: ['Cosmetic', 'Workaround exists', 'Blocking; no workaround'],
|
|
37
|
+
},
|
|
38
|
+
requestsRefund: {
|
|
39
|
+
type: 'boolean',
|
|
40
|
+
instructions: 'Is the customer requesting money back?',
|
|
41
|
+
},
|
|
42
|
+
},
|
|
43
|
+
});
|
|
44
|
+
}
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Provider models
|
|
48
|
+
|
|
49
|
+
Use a provider's `evaluationModel` factory:
|
|
50
|
+
|
|
51
|
+
| Provider | Example model |
|
|
52
|
+
| ----------- | -------------------------------------------------------- |
|
|
53
|
+
| TypeSafe AI | `typeSafeAi.evaluationModel('jev-latest')` |
|
|
54
|
+
| OpenAI | `openai.evaluationModel('gpt-5.6-luna')` |
|
|
55
|
+
| Anthropic | `anthropic.evaluationModel('claude-haiku-4-5-20251001')` |
|
|
56
|
+
| Google | `google.evaluationModel('gemini-3.5-flash-lite')` |
|
|
57
|
+
|
|
58
|
+
TypeSafe AI supplies native Choice, Score, and Boolean evaluations. OpenAI,
|
|
59
|
+
Anthropic, and Google adapt structured language-model output for all three types.
|
|
60
|
+
Boolean answers contain prompted estimates of P(true), validated to be finite
|
|
61
|
+
and in `[0, 1]`. These estimates are not guaranteed to be calibrated. Choice and
|
|
62
|
+
Score answers do not include probability distributions. Select a model that
|
|
63
|
+
supports the provider's structured-output API; judge quality against your own
|
|
64
|
+
labeled examples before choosing a model for a task.
|
|
65
|
+
|
|
66
|
+
The language-model adapters evaluate all questions in one prompt. They do not
|
|
67
|
+
provide TypeSafe's native independent-question execution semantics. The example
|
|
68
|
+
model IDs demonstrate API compatibility; they are not benchmark-selected defaults.
|
|
69
|
+
|
|
70
|
+
Language-model adapters request `reasoning: 'none'` by default. Each provider
|
|
71
|
+
maps this setting to its model's reasoning controls; it does not guarantee that
|
|
72
|
+
every model runs without thinking. To enable reasoning for more demanding
|
|
73
|
+
evaluations, pass the model's supported reasoning settings through
|
|
74
|
+
`providerOptions`, which take precedence over this default. For example, use
|
|
75
|
+
`providerOptions: { openai: { reasoningEffort: 'high' } }` with an OpenAI model
|
|
76
|
+
that supports that effort.
|
|
77
|
+
|
|
78
|
+
## Model aliases and registries
|
|
79
|
+
|
|
80
|
+
Use `customProvider` to give models application-specific names, then register
|
|
81
|
+
providers with `createProviderRegistry`:
|
|
82
|
+
|
|
83
|
+
```ts
|
|
84
|
+
import { typeSafeAi } from '@ai-sdk/typesafe-ai';
|
|
85
|
+
import { openai } from '@ai-sdk/openai';
|
|
86
|
+
import {
|
|
87
|
+
customProvider,
|
|
88
|
+
createProviderRegistry,
|
|
89
|
+
experimental_evaluate,
|
|
90
|
+
} from 'ai';
|
|
91
|
+
|
|
92
|
+
const registry = createProviderRegistry({
|
|
93
|
+
triage: customProvider({
|
|
94
|
+
evaluationModels: {
|
|
95
|
+
native: typeSafeAi.evaluationModel('jev-latest'),
|
|
96
|
+
compact: openai.evaluationModel('gpt-5.6-luna'),
|
|
97
|
+
},
|
|
98
|
+
fallbackProvider: typeSafeAi,
|
|
99
|
+
}),
|
|
100
|
+
openai,
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
const result = await experimental_evaluate({
|
|
104
|
+
model: registry.evaluationModel('triage:native'),
|
|
105
|
+
state: 'I was charged twice.',
|
|
106
|
+
questions: {
|
|
107
|
+
department: {
|
|
108
|
+
type: 'choice',
|
|
109
|
+
instructions: 'Which team should handle this?',
|
|
110
|
+
criteria: { billing: 'Charges and refunds', support: 'Other requests' },
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
result.answers.department.choice; // 'billing' | 'support'
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Registry IDs use `providerId:modelId`; the `separator` option changes the
|
|
119
|
+
separator. Only the first separator is used, so model IDs can contain it.
|
|
120
|
+
Custom aliases take precedence over a fallback provider. A fallback only resolves
|
|
121
|
+
unknown model IDs; it does not retry failed evaluations or substitute a model
|
|
122
|
+
when a question type is unsupported. Registered providers keep their own
|
|
123
|
+
credentials and settings. Registry language/image middleware does not wrap
|
|
124
|
+
evaluation models.
|
|
125
|
+
|
|
126
|
+
### Default-provider strings
|
|
127
|
+
|
|
128
|
+
To use strings directly, configure the default provider once at application
|
|
129
|
+
startup:
|
|
130
|
+
|
|
131
|
+
```ts
|
|
132
|
+
// Using the registry above, expose an alias through a custom provider:
|
|
133
|
+
globalThis.AI_SDK_DEFAULT_PROVIDER = customProvider({
|
|
134
|
+
evaluationModels: { native: registry.evaluationModel('triage:native') },
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
const result = await experimental_evaluate({
|
|
138
|
+
model: 'native',
|
|
139
|
+
state: 'I was charged twice.',
|
|
140
|
+
questions: {
|
|
141
|
+
refund: {
|
|
142
|
+
type: 'boolean',
|
|
143
|
+
instructions: 'Is the customer asking for a refund?',
|
|
144
|
+
},
|
|
145
|
+
},
|
|
146
|
+
});
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
A direct provider such as `typeSafeAi` can also be the default; then use its
|
|
150
|
+
unprefixed model ID, such as `'jev-latest'`. String values in `evaluationModels`
|
|
151
|
+
also resolve through this global default. Prefer model instances in aliases to
|
|
152
|
+
avoid resolution cycles. Global configuration affects other AI SDK functions
|
|
153
|
+
too; avoid changing it per request in a shared process.
|
|
154
|
+
|
|
155
|
+
Strings require an explicitly configured provider with an `evaluationModel`
|
|
156
|
+
method. There is no assumed Gateway evaluation support. Missing providers throw
|
|
157
|
+
`NoSuchProviderError`; missing models or evaluation capabilities throw
|
|
158
|
+
`NoSuchModelError` with `modelType: 'evaluationModel'`. Unsupported model versions
|
|
159
|
+
throw `UnsupportedModelVersionError`, including models returned by registries.
|
|
160
|
+
The stable `ProviderV4` and `ProviderRegistryProvider` interfaces are unchanged;
|
|
161
|
+
keep the inferred registry type, or use `Experimental_EvaluationProviderRegistry`,
|
|
162
|
+
to retain its experimental `evaluationModel` method.
|
|
163
|
+
|
|
164
|
+
## Question types
|
|
165
|
+
|
|
166
|
+
| Type | Criteria | Answer |
|
|
167
|
+
| --------- | ----------------------------------------- | ------------------------------------------------------------------------ |
|
|
168
|
+
| `choice` | A nonempty map of options to descriptions | `choice`, inferred as a union of option keys; optional `probabilities` |
|
|
169
|
+
| `score` | At least two ordered level descriptions | Fractional `score` in `[0, levels.length - 1]`; optional `probabilities` |
|
|
170
|
+
| `boolean` | Optional `true` and `false` descriptions | Required `probability`, the model-estimated probability of true |
|
|
171
|
+
|
|
172
|
+
Instructions and descriptions can be strings, JSON objects, or JSON arrays.
|
|
173
|
+
Descriptions can also be `null`. Core treats structured descriptions as content;
|
|
174
|
+
it does not interpret their keys. Functions, class instances, cycles, undefined
|
|
175
|
+
values, and nonfinite numbers are not JSON-compatible.
|
|
176
|
+
|
|
177
|
+
Answers retain question IDs and have the same `type` as their question. When a
|
|
178
|
+
Choice distribution is supplied, it includes every option and the selected
|
|
179
|
+
choice has maximal probability. Score distributions use string keys for
|
|
180
|
+
zero-based level indices, and the score equals the probability-weighted mean.
|
|
181
|
+
Without a distribution, a score is the model's estimated position on the rubric.
|
|
182
|
+
|
|
183
|
+
Distributions must sum to one, and weighted scores must agree with their
|
|
184
|
+
distributions. The default absolute tolerance is `0.000001`. Providers that round
|
|
185
|
+
their output can declare `rounding.probabilityDecimals` and
|
|
186
|
+
`rounding.scoreDecimals` (integers from 0 to 15). Validation then also allows half
|
|
187
|
+
a unit in the last decimal place per rounded probability or score, accumulated
|
|
188
|
+
over the sum or weighted mean. For example, probabilities rounded to two decimal
|
|
189
|
+
places can sum to `0.99` even when their unrounded values sum to one. The result
|
|
190
|
+
includes this `rounding` information. Invalid output is rejected; native values
|
|
191
|
+
are preserved, never silently normalized.
|
|
192
|
+
|
|
193
|
+
## Probabilities and confidence
|
|
194
|
+
|
|
195
|
+
Choice and Score distributions are optional. Boolean probability is required:
|
|
196
|
+
`0.98` means a strong yes and `0.02` means a strong no. It is not confidence in
|
|
197
|
+
either outcome. The SDK does not promise calibration across providers. Structured
|
|
198
|
+
language-model adapters prompt the model to estimate P(true); native evaluation providers return
|
|
199
|
+
their API probabilities. Provider-specific confidence statistics belong
|
|
200
|
+
in `providerMetadata`. TypeSafe exposes its separate Choice/Score confidence
|
|
201
|
+
statistic at `result.providerMetadata?.typesafe?.confidence`, keyed by question
|
|
202
|
+
ID. It is not the selected option's probability or a portable confidence measure.
|
|
203
|
+
|
|
204
|
+
Check for optional distributions before using them. For example, an application
|
|
205
|
+
can route only when a provider supplies a sufficiently high selected-option
|
|
206
|
+
probability:
|
|
207
|
+
|
|
208
|
+
```ts
|
|
209
|
+
const answer = result.answers.department;
|
|
210
|
+
const selectedProbability = answer.probabilities?.[answer.choice];
|
|
211
|
+
if (selectedProbability != null && selectedProbability >= 0.9) {
|
|
212
|
+
// Route automatically; otherwise use the application's review path.
|
|
213
|
+
}
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Choose Boolean thresholds in application code, using labeled data from the task
|
|
217
|
+
rather than assuming that the same threshold behaves identically across providers:
|
|
218
|
+
|
|
219
|
+
```ts
|
|
220
|
+
if (result.answers.requestsRefund.probability >= 0.8) {
|
|
221
|
+
// Route to the refunds queue.
|
|
222
|
+
}
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
## Errors and cancellation
|
|
226
|
+
|
|
227
|
+
The model's `supportedQuestionTypes` are checked before calling the provider.
|
|
228
|
+
Any unsupported question fails the entire call with
|
|
229
|
+
`Experimental_EvaluationUnsupportedQuestionTypeError`. Successful calls return
|
|
230
|
+
an answer for every question; there is no partial success or automatic model
|
|
231
|
+
substitution.
|
|
232
|
+
|
|
233
|
+
Invalid inputs throw `InvalidArgumentError`. Missing answers, mismatched answer
|
|
234
|
+
types, invalid options, scores, or probabilities throw `InvalidResponseDataError`.
|
|
235
|
+
Transient provider failures use the normal retry policy (`maxRetries: 2` by
|
|
236
|
+
default). Use `abortSignal` to cancel evaluation, `headers` for request headers,
|
|
237
|
+
and `providerOptions` for provider-specific settings.
|
|
238
|
+
|
|
239
|
+
The result includes `usage`, `warnings`, `providerMetadata`, and `response`.
|
|
240
|
+
Unknown token counts stay `undefined`; `totalTokens` is available only when both
|
|
241
|
+
input and output counts are known.
|
|
242
|
+
|
|
243
|
+
For tests, use `Experimental_EvaluationMockModelV4` from `ai/test`.
|
|
244
|
+
|
|
245
|
+
## Scope and examples
|
|
246
|
+
|
|
247
|
+
Evaluation currently returns one complete result for one shared state. It does
|
|
248
|
+
not stream answers, perform multilabel classification, or batch unrelated
|
|
249
|
+
states. Run separate calls for separate states. Provider support and judgment
|
|
250
|
+
quality depend on the chosen model; the SDK does not choose a model automatically.
|
|
251
|
+
|
|
252
|
+
Runnable examples are in
|
|
253
|
+
[`examples/ai-functions/src/evaluate`](https://github.com/vercel/ai/tree/main/examples/ai-functions/src/evaluate),
|
|
254
|
+
including basic examples for TypeSafe, OpenAI, Anthropic, and Google, model
|
|
255
|
+
registries, custom aliases, default-provider strings, and probability-based routing.
|
|
@@ -66,8 +66,7 @@ support are:
|
|
|
66
66
|
|
|
67
67
|
See the provider documentation for the supported models, limits, and native
|
|
68
68
|
batch behavior. For example, OpenAI batch support is available through the
|
|
69
|
-
Responses API, not `openai.chat()
|
|
70
|
-
the Responses API, not `xai.chat()`.
|
|
69
|
+
Responses API, not `openai.chat()`.
|
|
71
70
|
|
|
72
71
|
## Starting a batch
|
|
73
72
|
|
|
@@ -468,3 +468,13 @@ const result = await streamText({
|
|
|
468
468
|
```
|
|
469
469
|
|
|
470
470
|
This simplifies provider usage and makes it easier to switch between providers without changing your model references throughout your codebase.
|
|
471
|
+
|
|
472
|
+
## Experimental evaluation models
|
|
473
|
+
|
|
474
|
+
Custom providers accept `evaluationModels` aliases, and registries expose
|
|
475
|
+
`evaluationModel('provider:model')`. These methods return model instances for
|
|
476
|
+
`experimental_evaluate`. An explicitly configured default provider with an
|
|
477
|
+
`evaluationModel` method also enables direct string IDs; evaluation does not
|
|
478
|
+
implicitly use Gateway. Registry middleware for language and image models does
|
|
479
|
+
not apply to evaluation. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
|
|
480
|
+
for aliases, default-provider configuration, and capability differences.
|
|
@@ -28,6 +28,12 @@ description: Learn about AI SDK Core.
|
|
|
28
28
|
description: 'Learn how to do tool calling with AI SDK Core.',
|
|
29
29
|
href: '/docs/ai-sdk-core/tools-and-tool-calling',
|
|
30
30
|
},
|
|
31
|
+
{
|
|
32
|
+
title: 'Tool Search',
|
|
33
|
+
description:
|
|
34
|
+
'Discover tools on demand with direct calling or cache-preserving code mode.',
|
|
35
|
+
href: '/docs/ai-sdk-core/tool-search',
|
|
36
|
+
},
|
|
31
37
|
{
|
|
32
38
|
title: 'Code Mode',
|
|
33
39
|
description:
|
|
@@ -92,6 +92,41 @@ export async function loadChat(id: string): Promise<UIMessage[]> {
|
|
|
92
92
|
|
|
93
93
|
When processing messages on the server that contain tool calls, custom metadata, or data parts, you should validate them using `validateUIMessages` before sending them to the model.
|
|
94
94
|
|
|
95
|
+
### Migrating deprecated tool `rawInput`
|
|
96
|
+
|
|
97
|
+
Older persisted tool parts in the `output-error` state may contain `rawInput`.
|
|
98
|
+
This field is deprecated and will be removed in the next major version. Store
|
|
99
|
+
tool arguments in `input` instead. AI SDK emits a deprecation warning when
|
|
100
|
+
`validateUIMessages`, `safeValidateUIMessages`, or `convertToModelMessages`
|
|
101
|
+
encounters a defined `rawInput` value. UI stream processing also warns when it
|
|
102
|
+
reconstructs a static `output-error` part with `rawInput`.
|
|
103
|
+
|
|
104
|
+
Migrate each legacy part by copying `rawInput` to `input` when `input` is
|
|
105
|
+
`null` or `undefined`, then remove `rawInput`. This preserves the current
|
|
106
|
+
backward-compatible conversion behavior:
|
|
107
|
+
|
|
108
|
+
```ts
|
|
109
|
+
import { isToolUIPart } from 'ai';
|
|
110
|
+
|
|
111
|
+
const migratedParts = message.parts.map(part => {
|
|
112
|
+
if (
|
|
113
|
+
!isToolUIPart(part) ||
|
|
114
|
+
part.state !== 'output-error' ||
|
|
115
|
+
!('rawInput' in part) ||
|
|
116
|
+
part.rawInput === undefined
|
|
117
|
+
) {
|
|
118
|
+
return part;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const { rawInput, ...partWithoutRawInput } = part;
|
|
122
|
+
|
|
123
|
+
return {
|
|
124
|
+
...partWithoutRawInput,
|
|
125
|
+
input: part.input ?? rawInput,
|
|
126
|
+
};
|
|
127
|
+
});
|
|
128
|
+
```
|
|
129
|
+
|
|
95
130
|
### Validation with tools
|
|
96
131
|
|
|
97
132
|
When your messages include tool calls, validate them against your tool definitions:
|
|
@@ -60,7 +60,13 @@ On Node.js, the default validated download fetch uses `node:dns` and an
|
|
|
60
60
|
The connector uses those exact results, closing both hostname-to-private-IP and
|
|
61
61
|
DNS-rebinding bypasses.
|
|
62
62
|
|
|
63
|
-
|
|
63
|
+
Wrapping or replacing global `fetch` does not disable this protection: the
|
|
64
|
+
default Node.js download transport is independent of global `fetch`.
|
|
65
|
+
|
|
66
|
+
Bun, Deno, Cloudflare Workers, and framework edge runtimes use their platform
|
|
67
|
+
fetch, even when they expose a Node-compatible `process` object.
|
|
68
|
+
|
|
69
|
+
If you explicitly inject a custom `fetch`, it is responsible for equivalent DNS
|
|
64
70
|
validation and connection pinning. Other runtimes do not expose Node's
|
|
65
71
|
DNS/socket hooks, so server deployments on those runtimes should restrict
|
|
66
72
|
network egress to private, loopback, link-local, and cloud-metadata ranges.
|
|
@@ -78,7 +84,7 @@ code.
|
|
|
78
84
|
|
|
79
85
|
### 2. Harden an injected `fetch`
|
|
80
86
|
|
|
81
|
-
The Node.js default is already pinned. If you inject
|
|
87
|
+
The Node.js default is already pinned. If you explicitly inject a custom
|
|
82
88
|
`fetch`, back it with an `undici`
|
|
83
89
|
`Agent` whose `connect.lookup` validates the resolved IP and lets the socket
|
|
84
90
|
connect only to a safe address — closing both the hostname-to-private and the
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: experimental_evaluate
|
|
3
|
+
description: Evaluate typed questions against shared state with an evaluation model.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# `experimental_evaluate()`
|
|
7
|
+
|
|
8
|
+
```ts
|
|
9
|
+
import { experimental_evaluate } from 'ai';
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Evaluates a nonempty map of `choice`, `score`, and `boolean` questions against one
|
|
13
|
+
state. See [Evaluation](/docs/ai-sdk-core/evaluation) for examples and semantics.
|
|
14
|
+
|
|
15
|
+
## Parameters
|
|
16
|
+
|
|
17
|
+
| Parameter | Type | Description |
|
|
18
|
+
| ----------------- | ------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------- |
|
|
19
|
+
| `model` | `Experimental_EvaluationModel` | Required experimental v4 model instance or a string ID resolved by an explicitly configured evaluation-capable default provider. |
|
|
20
|
+
| `state` | `string \| object \| array` | Required JSON-compatible shared state. |
|
|
21
|
+
| `questions` | `Record<string, Experimental_EvaluationQuestion>` | Required nonempty question map. |
|
|
22
|
+
| `maxRetries` | `number` | Nonnegative integer; defaults to 2. |
|
|
23
|
+
| `abortSignal` | `AbortSignal` | Cancels evaluation. |
|
|
24
|
+
| `headers` | `Record<string, string>` | Additional HTTP headers. |
|
|
25
|
+
| `providerOptions` | `ProviderOptions` | Provider-specific options. |
|
|
26
|
+
|
|
27
|
+
## Result
|
|
28
|
+
|
|
29
|
+
Returns `Promise<Experimental_EvaluationResult<QUESTIONS>>`:
|
|
30
|
+
|
|
31
|
+
- `answers`: One typed answer per question ID, with literal Choice option inference.
|
|
32
|
+
- `usage`: `inputTokens`, `outputTokens`, and `totalTokens`, each possibly undefined.
|
|
33
|
+
- `warnings`: Provider warnings, also passed to the SDK warning logger.
|
|
34
|
+
- `rounding`: Optional provider-declared decimal precision for probabilities and scores.
|
|
35
|
+
- `providerMetadata`: Optional provider-specific metadata.
|
|
36
|
+
- `response`: Timestamp, model ID, and optional response ID, headers, and body.
|
|
37
|
+
|
|
38
|
+
## Provider specification
|
|
39
|
+
|
|
40
|
+
`Experimental_EvaluationModelV4` is exported from `@ai-sdk/provider` and declares
|
|
41
|
+
`specificationVersion: 'v4'`, `provider`, `modelId`, `supportedQuestionTypes`, and
|
|
42
|
+
`doEvaluate(options)`. Evaluation is isolated from stable `ProviderV4`.
|
|
43
|
+
|
|
44
|
+
The public core types are `Experimental_EvaluationModel`,
|
|
45
|
+
`Experimental_EvaluationQuestion`, `Experimental_EvaluationAnswer`, and
|
|
46
|
+
`Experimental_EvaluationResult`. All evaluation-specific classes use the
|
|
47
|
+
`Evaluation` prefix, with `Experimental_` aliases at package boundaries.
|
|
48
|
+
|
|
49
|
+
## Errors
|
|
50
|
+
|
|
51
|
+
Unsupported types throw `Experimental_EvaluationUnsupportedQuestionTypeError`
|
|
52
|
+
before provider I/O. Invalid inputs throw `InvalidArgumentError`; malformed
|
|
53
|
+
answers throw `InvalidResponseDataError`. Invalid answers are not retried.
|
|
54
|
+
Neither partial results nor missing probability synthesis are supported.
|
|
55
|
+
|
|
56
|
+
## Model resolution
|
|
57
|
+
|
|
58
|
+
Use `registry.evaluationModel('provider:model')` or
|
|
59
|
+
`customProvider({ evaluationModels: { alias: model } }).evaluationModel('alias')`
|
|
60
|
+
to resolve models. Strings passed directly to `experimental_evaluate` use
|
|
61
|
+
`globalThis.AI_SDK_DEFAULT_PROVIDER.evaluationModel(id)`, when available. Evaluation
|
|
62
|
+
never implicitly falls back to Gateway.
|
|
63
|
+
|
|
64
|
+
Resolution errors use the existing `NoSuchModelError` and `NoSuchProviderError`
|
|
65
|
+
classes with `modelType: 'evaluationModel'`. Model instances and resolved models
|
|
66
|
+
must implement v4; other versions throw `UnsupportedModelVersionError`.
|
|
67
|
+
See [model resolution examples](/docs/ai-sdk-core/evaluation#model-aliases-and-registries).
|
|
@@ -65,6 +65,13 @@ export const weatherTool = tool({
|
|
|
65
65
|
description:
|
|
66
66
|
'Information about the purpose of the tool including details on how and when it can be used by the model. Provide a string for a fixed description, or a function to derive the description from the tool-specific context and optional experimental sandbox before each model call.',
|
|
67
67
|
},
|
|
68
|
+
{
|
|
69
|
+
name: 'deferLoading',
|
|
70
|
+
isOptional: true,
|
|
71
|
+
type: 'boolean',
|
|
72
|
+
description:
|
|
73
|
+
"Keep this tool out of the model context until toolSearch discovers it. Supports direct calling or code mode with toolDiscovery: 'conversation'. Discovered tools become available on the next model step. Defaults to false.",
|
|
74
|
+
},
|
|
68
75
|
{
|
|
69
76
|
name: 'title',
|
|
70
77
|
isOptional: true,
|
|
@@ -58,6 +58,13 @@ export const customTool = dynamicTool({
|
|
|
58
58
|
description:
|
|
59
59
|
'Information about the purpose of the tool including details on how and when it can be used by the model. Provide a string for a fixed description, or a function to derive the description from the tool-specific context and optional experimental sandbox before each model call.'
|
|
60
60
|
},
|
|
61
|
+
{
|
|
62
|
+
name: 'deferLoading',
|
|
63
|
+
isOptional: true,
|
|
64
|
+
type: 'boolean',
|
|
65
|
+
description:
|
|
66
|
+
"Keep this tool out of the model context until toolSearch discovers it. Supports direct calling or code mode with toolDiscovery: 'conversation'. Discovered tools become available on the next model step. Defaults to false.",
|
|
67
|
+
},
|
|
61
68
|
{
|
|
62
69
|
name: 'title',
|
|
63
70
|
isOptional: true,
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: toolSearch
|
|
3
|
+
description: Search deferred tools and load their definitions on demand for direct calling or code mode.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# `toolSearch()`
|
|
7
|
+
|
|
8
|
+
Creates a tool that searches the surrounding generation's deferred tools by name
|
|
9
|
+
and description. The factory takes no arguments. Use it with `generateText`,
|
|
10
|
+
`streamText`, or `ToolLoopAgent`, either with direct tool calling or with code mode
|
|
11
|
+
configured with `toolDiscovery: 'conversation'`.
|
|
12
|
+
|
|
13
|
+
```ts
|
|
14
|
+
import { toolSearch } from 'ai';
|
|
15
|
+
|
|
16
|
+
const search = toolSearch();
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
See the [Tool Search guide](/docs/ai-sdk-core/tool-search) for direct-calling and
|
|
20
|
+
code mode examples, including how code mode preserves the tool-definition cache.
|
|
21
|
+
|
|
22
|
+
## Model Input
|
|
23
|
+
|
|
24
|
+
The model supplies the following input to the search tool:
|
|
25
|
+
|
|
26
|
+
```json
|
|
27
|
+
{ "query": "weather forecast" }
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
`query` is a required, nonempty string of search keywords. Search is local and
|
|
31
|
+
case-insensitive, matching words in tool names and descriptions. Camel-case names
|
|
32
|
+
are split into words. Name matches rank above description matches; equal scores
|
|
33
|
+
preserve registration order. Function descriptions are resolved with the current
|
|
34
|
+
tool context and sandbox. No embedding service or additional model call is used.
|
|
35
|
+
|
|
36
|
+
## Output
|
|
37
|
+
|
|
38
|
+
```ts
|
|
39
|
+
{
|
|
40
|
+
tools: [
|
|
41
|
+
{ name: 'getForecast', description: 'Get the weather forecast for a city.' },
|
|
42
|
+
],
|
|
43
|
+
}
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Results contain at most five matching tools, with their names and optional
|
|
47
|
+
descriptions. They do not include schemas. No matches returns `{ tools: [] }`.
|
|
48
|
+
Every returned match is queued for discovery. Newly discovered tools can only be
|
|
49
|
+
called on the next model step, after their definitions have been provided. A
|
|
50
|
+
parallel call in the same response as the search cannot use a newly discovered
|
|
51
|
+
tool. For code mode, finish the current execution and wait for the capability
|
|
52
|
+
update.
|
|
53
|
+
|
|
54
|
+
## Discovery Lifecycle
|
|
55
|
+
|
|
56
|
+
Before the next model step, the SDK makes discovered tools available through
|
|
57
|
+
their configured callers:
|
|
58
|
+
|
|
59
|
+
- **Direct calling:** the provider receives the updated tool definitions.
|
|
60
|
+
- **Code mode:** the SDK appends a user message containing the updated capability
|
|
61
|
+
catalog. The provider-visible code mode definition stays unchanged. Existing
|
|
62
|
+
catalogs remain in the conversation; the latest catalog describes the complete
|
|
63
|
+
currently available tool set.
|
|
64
|
+
|
|
65
|
+
Actual prompt-cache reuse depends on the provider. Code mode search still requires
|
|
66
|
+
`toolDiscovery: 'conversation'`; description discovery and provider callers are
|
|
67
|
+
not supported.
|
|
68
|
+
|
|
69
|
+
Discovered tools remain loaded for the rest of the generation, subject to
|
|
70
|
+
`activeTools`. Search cannot discover tools excluded by `activeTools`. Discovery
|
|
71
|
+
state is isolated between generation calls, including calls that reuse the same
|
|
72
|
+
agent or tool instances.
|
|
73
|
+
|
|
74
|
+
Set a multi-step `stopWhen` condition with `generateText` and `streamText` so the
|
|
75
|
+
model can search, use the discovered tools, and answer.
|
|
@@ -99,3 +99,15 @@ const validatedMessages = await validateUIMessages({
|
|
|
99
99
|
tools,
|
|
100
100
|
});
|
|
101
101
|
```
|
|
102
|
+
|
|
103
|
+
## Deprecated `rawInput` field
|
|
104
|
+
|
|
105
|
+
For backward compatibility, validation still accepts `rawInput` on tool parts
|
|
106
|
+
in the `output-error` state. When a defined `rawInput` value is found,
|
|
107
|
+
`validateUIMessages` emits an AI SDK deprecation warning through
|
|
108
|
+
`AI_SDK_LOG_WARNINGS`.
|
|
109
|
+
|
|
110
|
+
Migrate persisted messages to store tool arguments in `input` and remove
|
|
111
|
+
`rawInput`. For backward compatibility, conversion uses `rawInput` as a
|
|
112
|
+
fallback when `input` is `null` or `undefined`. `rawInput` will be removed in
|
|
113
|
+
the next major version.
|
|
@@ -33,6 +33,18 @@ if (!result.success) {
|
|
|
33
33
|
}
|
|
34
34
|
```
|
|
35
35
|
|
|
36
|
+
## Deprecated `rawInput` field
|
|
37
|
+
|
|
38
|
+
For backward compatibility, validation still accepts `rawInput` on tool parts
|
|
39
|
+
in the `output-error` state. When a defined `rawInput` value is found,
|
|
40
|
+
`safeValidateUIMessages` emits an AI SDK deprecation warning through
|
|
41
|
+
`AI_SDK_LOG_WARNINGS`.
|
|
42
|
+
|
|
43
|
+
Migrate persisted messages to store tool arguments in `input` and remove
|
|
44
|
+
`rawInput`. For backward compatibility, conversion uses `rawInput` as a
|
|
45
|
+
fallback when `input` is `null` or `undefined`. `rawInput` will be removed in
|
|
46
|
+
the next major version.
|
|
47
|
+
|
|
36
48
|
## Advanced Usage
|
|
37
49
|
|
|
38
50
|
Comprehensive validation with custom metadata, data parts, and tools:
|
|
@@ -298,3 +298,20 @@ The `createProviderRegistry` function returns a `Provider` instance. It has the
|
|
|
298
298
|
},
|
|
299
299
|
]}
|
|
300
300
|
/>
|
|
301
|
+
|
|
302
|
+
## Experimental evaluation models
|
|
303
|
+
|
|
304
|
+
The inferred return type also exposes `evaluationModel('providerId:modelId')`,
|
|
305
|
+
returning `Experimental_EvaluationModelV4`. The provider must expose an
|
|
306
|
+
`evaluationModel` factory. Custom separators and model ID
|
|
307
|
+
inference work as they do for video models. Language and image middleware do not
|
|
308
|
+
wrap evaluation models. `ProviderRegistryProvider` remains a stable interface;
|
|
309
|
+
use the inferred return type or `Experimental_EvaluationProviderRegistry` to
|
|
310
|
+
retain experimental evaluation access.
|
|
311
|
+
|
|
312
|
+
Unavailable evaluation capabilities or models throw `NoSuchModelError` with
|
|
313
|
+
`modelType: 'evaluationModel'`; unknown registry providers throw
|
|
314
|
+
`NoSuchProviderError`. These capabilities are structural extensions and are not
|
|
315
|
+
added to the stable `ProviderV4` contract. Evaluation does not assume Gateway
|
|
316
|
+
support. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
|
|
317
|
+
for runnable usage patterns.
|
|
@@ -201,3 +201,19 @@ The `customProvider` function returns a `Provider` instance. It has the followin
|
|
|
201
201
|
},
|
|
202
202
|
]}
|
|
203
203
|
/>
|
|
204
|
+
|
|
205
|
+
## Experimental evaluation models
|
|
206
|
+
|
|
207
|
+
Pass `evaluationModels: Record<string, Experimental_EvaluationModel>` to define
|
|
208
|
+
aliases for evaluation model instances or string IDs. The returned provider adds
|
|
209
|
+
`evaluationModel(alias): Experimental_EvaluationModelV4`. String aliases resolve
|
|
210
|
+
through an explicitly configured default provider with an `evaluationModel`
|
|
211
|
+
method. Unknown aliases use the fallback provider's evaluation factory when
|
|
212
|
+
available; model failures and unsupported questions do not trigger substitution.
|
|
213
|
+
|
|
214
|
+
Unavailable evaluation capabilities or models throw `NoSuchModelError` with
|
|
215
|
+
`modelType: 'evaluationModel'`; unknown registry providers throw
|
|
216
|
+
`NoSuchProviderError`. These capabilities are structural extensions and are not
|
|
217
|
+
added to the stable `ProviderV4` contract. Evaluation does not assume Gateway
|
|
218
|
+
support. See [Evaluation](/docs/ai-sdk-core/evaluation#model-aliases-and-registries)
|
|
219
|
+
for runnable usage patterns.
|
|
@@ -115,6 +115,12 @@ It also contains the following helper functions:
|
|
|
115
115
|
|
|
116
116
|
<IndexCards
|
|
117
117
|
cards={[
|
|
118
|
+
{
|
|
119
|
+
title: 'toolSearch()',
|
|
120
|
+
description:
|
|
121
|
+
'Search deferred tools and load their definitions on demand.',
|
|
122
|
+
href: '/docs/reference/ai-sdk-core/tool-search',
|
|
123
|
+
},
|
|
118
124
|
{
|
|
119
125
|
title: 'tool()',
|
|
120
126
|
description: 'Type inference helper function for tools.',
|
|
@@ -70,6 +70,18 @@ A Promise that resolves to an array of [`ModelMessage`](/docs/reference/ai-sdk-c
|
|
|
70
70
|
]}
|
|
71
71
|
/>
|
|
72
72
|
|
|
73
|
+
## Deprecated `rawInput` field
|
|
74
|
+
|
|
75
|
+
Tool parts in the `output-error` state should store their tool arguments in
|
|
76
|
+
`input`. The legacy `rawInput` field remains supported for persisted messages,
|
|
77
|
+
but `convertToModelMessages` emits an AI SDK deprecation warning when it
|
|
78
|
+
encounters a defined value.
|
|
79
|
+
|
|
80
|
+
When both fields are present, `input` takes precedence when it is non-nullish.
|
|
81
|
+
For backward compatibility, `rawInput` remains the fallback when `input` is
|
|
82
|
+
`null` or `undefined`. Migrate stored messages to `input` before the next major
|
|
83
|
+
version, when `rawInput` will be removed.
|
|
84
|
+
|
|
73
85
|
## Tool Approval States
|
|
74
86
|
|
|
75
87
|
`convertToModelMessages` preserves tool approval state from UI messages when converting them back into `ModelMessage`s for a follow-up `generateText` or `streamText` call.
|