@ai-sdk/anthropic 4.0.55 → 4.0.57

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1893,3 +1893,52 @@ and the `mediaType` should be set to `'application/pdf'`.
1893
1893
  of available models. The table above lists popular models. You can also pass
1894
1894
  any available provider model ID as a string if needed.
1895
1895
  </Note>
1896
+
1897
+ ## Evaluation Models
1898
+
1899
+ Create an experimental evaluation model with `anthropic.evaluationModel(modelId)`.
1900
+ It uses the Messages API's structured-output support for Choice, Score, and Boolean
1901
+ questions. `claude-haiku-4-5-20251001` is one supported model.
1902
+
1903
+ ```ts
1904
+ import { anthropic } from '@ai-sdk/anthropic';
1905
+ import { experimental_evaluate } from 'ai';
1906
+
1907
+ const { answers } = await experimental_evaluate({
1908
+ model: anthropic.evaluationModel('claude-haiku-4-5-20251001'),
1909
+ state: 'I was charged twice.',
1910
+ questions: {
1911
+ requestsRefund: {
1912
+ type: 'boolean',
1913
+ instructions: 'Is the customer requesting money back?',
1914
+ },
1915
+ department: {
1916
+ type: 'choice',
1917
+ instructions: 'Which team should handle this?',
1918
+ criteria: { billing: 'Charges and refunds', support: 'Other requests' },
1919
+ },
1920
+ },
1921
+ });
1922
+ ```
1923
+
1924
+ Choice labels are preserved exactly, and Scores are finite fractional positions
1925
+ on the ordered rubric. Anthropic does not support numeric bounds in its native
1926
+ output schema, so the adapter describes them in the prompt and validates them
1927
+ after parsing. Invalid answers, refusals, and truncation fail the entire call.
1928
+
1929
+ The adapter returns no probability distributions for Choice or Score. Boolean
1930
+ answers contain prompted estimates of P(true), validated to be finite and in
1931
+ `[0, 1]`. These estimates are not guaranteed to be calibrated. Apply thresholds
1932
+ in application code, for example `answers.requestsRefund.probability >= 0.5`.
1933
+ It respects
1934
+ `createAnthropic` configuration and forwards `providerOptions.anthropic`, including
1935
+ `structuredOutputMode`. Supported models use native structured output by default;
1936
+ `jsonTool` uses the existing Messages JSON-tool fallback. Usage, warnings,
1937
+ response information, and provider metadata are preserved.
1938
+ See [Evaluation](/docs/ai-sdk-core/evaluation).
1939
+
1940
+ Evaluation models can also be accessed through `customProvider` aliases or
1941
+ `createProviderRegistry().evaluationModel('provider:model')`. Direct string IDs
1942
+ use Gateway by default, or an explicitly configured default provider with an
1943
+ `evaluationModel` method. See
1944
+ [model aliases and registries](/docs/ai-sdk-core/evaluation#model-aliases-and-registries).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/anthropic",
3
- "version": "4.0.55",
3
+ "version": "4.0.57",
4
4
  "type": "module",
5
5
  "license": "Apache-2.0",
6
6
  "sideEffects": false,
@@ -35,8 +35,8 @@
35
35
  }
36
36
  },
37
37
  "dependencies": {
38
- "@ai-sdk/provider": "4.0.16",
39
- "@ai-sdk/provider-utils": "5.0.42"
38
+ "@ai-sdk/provider": "4.0.17",
39
+ "@ai-sdk/provider-utils": "5.0.44"
40
40
  },
41
41
  "devDependencies": {
42
42
  "@ai-sdk/test-server": "2.0.1",
@@ -2,6 +2,7 @@ import {
2
2
  InvalidArgumentError,
3
3
  NoSuchModelError,
4
4
  type Experimental_BatchV4 as BatchV4,
5
+ type Experimental_EvaluationModelV4 as EvaluationModelV4,
5
6
  type FilesV4,
6
7
  type LanguageModelV4,
7
8
  type ProviderV4,
@@ -16,6 +17,7 @@ import {
16
17
  withUserAgentSuffix,
17
18
  type FetchFunction,
18
19
  } from '@ai-sdk/provider-utils';
20
+ import { Experimental_EvaluationLanguageModel as EvaluationLanguageModel } from '@ai-sdk/provider-utils/experimental-evaluation';
19
21
  import { AnthropicFiles } from './anthropic-files';
20
22
  import { AnthropicLanguageModel } from './anthropic-language-model';
21
23
  import { AnthropicBatch } from './anthropic-batch';
@@ -52,6 +54,9 @@ export interface AnthropicProvider extends ProviderV4 {
52
54
 
53
55
  messages(modelId: AnthropicModelId): LanguageModelV4;
54
56
 
57
+ /** Creates an experimental Choice/Score/Boolean evaluation model using Messages. */
58
+ evaluationModel(modelId: AnthropicModelId): EvaluationModelV4;
59
+
55
60
  experimental_batch(): BatchV4<{ text: AnthropicModelId }>;
56
61
 
57
62
  /**
@@ -204,6 +209,11 @@ export function createAnthropic(
204
209
  provider.languageModel = createChatModel;
205
210
  provider.chat = createChatModel;
206
211
  provider.messages = createChatModel;
212
+ provider.evaluationModel = (modelId: AnthropicModelId) =>
213
+ new EvaluationLanguageModel({
214
+ model: createChatModel(modelId),
215
+ provider: `${providerName.replace(/\.messages$/, '')}.evaluation`,
216
+ });
207
217
  provider.experimental_batch = createBatch;
208
218
 
209
219
  provider.embeddingModel = (modelId: string) => {