@ai-sdk/openai 4.0.67 → 4.0.69

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3397,3 +3397,48 @@ const result = await generateSpeech({
3397
3397
  | `tts-1` | <Check /> |
3398
3398
  | `tts-1-hd` | <Check /> |
3399
3399
  | `gpt-4o-mini-tts` | <Check /> |
3400
+
3401
+ ## Evaluation Models
3402
+
3403
+ Create an experimental evaluation model with `openai.evaluationModel(modelId)`.
3404
+ It uses the Responses API with structured output to answer Choice, Score, and
3405
+ Boolean questions. Choose a model that supports structured output, such as `gpt-5.6-luna`.
3406
+
3407
+ ```ts
3408
+ import { openai } from '@ai-sdk/openai';
3409
+ import { experimental_evaluate } from 'ai';
3410
+
3411
+ const { answers } = await experimental_evaluate({
3412
+ model: openai.evaluationModel('gpt-5.6-luna'),
3413
+ state: 'I was charged twice.',
3414
+ questions: {
3415
+ requestsRefund: {
3416
+ type: 'boolean',
3417
+ instructions: 'Is the customer requesting money back?',
3418
+ },
3419
+ department: {
3420
+ type: 'choice',
3421
+ instructions: 'Which team should handle this?',
3422
+ criteria: { billing: 'Charges and refunds', support: 'Other requests' },
3423
+ },
3424
+ },
3425
+ });
3426
+ ```
3427
+
3428
+ Scores are finite fractional positions on the ordered rubric. Choice labels are
3429
+ returned exactly as supplied. These are model judgments: no probability
3430
+ distributions are returned for Choice or Score. Boolean answers contain a prompted
3431
+ estimate of P(true), validated to be finite and in `[0, 1]`. These estimates are
3432
+ not guaranteed to be calibrated. Apply thresholds in application code, for example
3433
+ `answers.requestsRefund.probability >= 0.5`.
3434
+
3435
+ The factory respects `createOpenAI` settings, including custom fetch and base URL.
3436
+ Pass Responses options through `providerOptions.openai`. Usage, warnings,
3437
+ response information, and provider metadata are preserved. Refusals, truncated
3438
+ output, or invalid answers fail the entire call. See [Evaluation](/docs/ai-sdk-core/evaluation).
3439
+
3440
+ Evaluation models can also be accessed through `customProvider` aliases or
3441
+ `createProviderRegistry().evaluationModel('provider:model')`. Direct string IDs
3442
+ require an explicitly configured default provider with an `evaluationModel`
3443
+ method; they do not automatically use Gateway. See
3444
+ [model aliases and registries](/docs/ai-sdk-core/evaluation#model-aliases-and-registries).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/openai",
3
- "version": "4.0.67",
3
+ "version": "4.0.69",
4
4
  "type": "module",
5
5
  "license": "Apache-2.0",
6
6
  "sideEffects": false,
@@ -35,8 +35,8 @@
35
35
  }
36
36
  },
37
37
  "dependencies": {
38
- "@ai-sdk/provider": "4.0.15",
39
- "@ai-sdk/provider-utils": "5.0.41"
38
+ "@ai-sdk/provider": "4.0.17",
39
+ "@ai-sdk/provider-utils": "5.0.43"
40
40
  },
41
41
  "devDependencies": {
42
42
  "@ai-sdk/test-server": "2.0.1",
@@ -1,5 +1,6 @@
1
1
  import type {
2
2
  Experimental_BatchV4 as BatchV4,
3
+ Experimental_EvaluationModelV4 as EvaluationModelV4,
3
4
  EmbeddingModelV4,
4
5
  FilesV4,
5
6
  ImageModelV4,
@@ -19,6 +20,7 @@ import {
19
20
  type FetchFunction,
20
21
  type WebSocketConstructor,
21
22
  } from '@ai-sdk/provider-utils';
23
+ import { Experimental_EvaluationLanguageModel as EvaluationLanguageModel } from '@ai-sdk/provider-utils/experimental-evaluation';
22
24
  import { OpenAIChatLanguageModel } from './chat/openai-chat-language-model';
23
25
  import type { OpenAIChatModelId } from './chat/openai-chat-language-model-options';
24
26
  import { OpenAICompletionLanguageModel } from './completion/openai-completion-language-model';
@@ -48,6 +50,9 @@ import { VERSION } from './version';
48
50
  export interface OpenAIProvider extends ProviderV4 {
49
51
  (modelId: OpenAIResponsesModelId): LanguageModelV4;
50
52
 
53
+ /** Creates an experimental Choice/Score/Boolean evaluation model using the Responses API. */
54
+ evaluationModel(modelId: OpenAIResponsesModelId): EvaluationModelV4;
55
+
51
56
  /**
52
57
  * Creates an OpenAI model for text generation.
53
58
  */
@@ -347,6 +352,11 @@ export function createOpenAI(
347
352
  provider.chat = createChatModel;
348
353
  provider.completion = createCompletionModel;
349
354
  provider.responses = createResponsesModel;
355
+ provider.evaluationModel = (modelId: OpenAIResponsesModelId) =>
356
+ new EvaluationLanguageModel({
357
+ model: createResponsesModel(modelId),
358
+ provider: `${providerName}.evaluation`,
359
+ });
350
360
  provider.embedding = createEmbeddingModel;
351
361
  provider.embeddingModel = createEmbeddingModel;
352
362
  provider.textEmbedding = createEmbeddingModel;