@ai-sdk/openai 4.0.68 → 4.0.70

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3397,3 +3397,48 @@ const result = await generateSpeech({
3397
3397
  | `tts-1` | <Check /> |
3398
3398
  | `tts-1-hd` | <Check /> |
3399
3399
  | `gpt-4o-mini-tts` | <Check /> |
3400
+
3401
+ ## Evaluation Models
3402
+
3403
+ Create an experimental evaluation model with `openai.evaluationModel(modelId)`.
3404
+ It uses the Responses API with structured output to answer Choice, Score, and
3405
+ Boolean questions. Choose a model that supports structured output, such as `gpt-5.6-luna`.
3406
+
3407
+ ```ts
3408
+ import { openai } from '@ai-sdk/openai';
3409
+ import { experimental_evaluate } from 'ai';
3410
+
3411
+ const { answers } = await experimental_evaluate({
3412
+ model: openai.evaluationModel('gpt-5.6-luna'),
3413
+ state: 'I was charged twice.',
3414
+ questions: {
3415
+ requestsRefund: {
3416
+ type: 'boolean',
3417
+ instructions: 'Is the customer requesting money back?',
3418
+ },
3419
+ department: {
3420
+ type: 'choice',
3421
+ instructions: 'Which team should handle this?',
3422
+ criteria: { billing: 'Charges and refunds', support: 'Other requests' },
3423
+ },
3424
+ },
3425
+ });
3426
+ ```
3427
+
3428
+ Scores are finite fractional positions on the ordered rubric. Choice labels are
3429
+ returned exactly as supplied. These are model judgments: no probability
3430
+ distributions are returned for Choice or Score. Boolean answers contain a prompted
3431
+ estimate of P(true), validated to be finite and in `[0, 1]`. These estimates are
3432
+ not guaranteed to be calibrated. Apply thresholds in application code, for example
3433
+ `answers.requestsRefund.probability >= 0.5`.
3434
+
3435
+ The factory respects `createOpenAI` settings, including custom fetch and base URL.
3436
+ Pass Responses options through `providerOptions.openai`. Usage, warnings,
3437
+ response information, and provider metadata are preserved. Refusals, truncated
3438
+ output, or invalid answers fail the entire call. See [Evaluation](/docs/ai-sdk-core/evaluation).
3439
+
3440
+ Evaluation models can also be accessed through `customProvider` aliases or
3441
+ `createProviderRegistry().evaluationModel('provider:model')`. Direct string IDs
3442
+ use Gateway by default, or an explicitly configured default provider with an
3443
+ `evaluationModel` method. See
3444
+ [model aliases and registries](/docs/ai-sdk-core/evaluation#model-aliases-and-registries).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/openai",
3
- "version": "4.0.68",
3
+ "version": "4.0.70",
4
4
  "type": "module",
5
5
  "license": "Apache-2.0",
6
6
  "sideEffects": false,
@@ -35,8 +35,8 @@
35
35
  }
36
36
  },
37
37
  "dependencies": {
38
- "@ai-sdk/provider": "4.0.16",
39
- "@ai-sdk/provider-utils": "5.0.42"
38
+ "@ai-sdk/provider": "4.0.17",
39
+ "@ai-sdk/provider-utils": "5.0.44"
40
40
  },
41
41
  "devDependencies": {
42
42
  "@ai-sdk/test-server": "2.0.1",
@@ -1,5 +1,6 @@
1
1
  import type {
2
2
  Experimental_BatchV4 as BatchV4,
3
+ Experimental_EvaluationModelV4 as EvaluationModelV4,
3
4
  EmbeddingModelV4,
4
5
  FilesV4,
5
6
  ImageModelV4,
@@ -19,6 +20,7 @@ import {
19
20
  type FetchFunction,
20
21
  type WebSocketConstructor,
21
22
  } from '@ai-sdk/provider-utils';
23
+ import { Experimental_EvaluationLanguageModel as EvaluationLanguageModel } from '@ai-sdk/provider-utils/experimental-evaluation';
22
24
  import { OpenAIChatLanguageModel } from './chat/openai-chat-language-model';
23
25
  import type { OpenAIChatModelId } from './chat/openai-chat-language-model-options';
24
26
  import { OpenAICompletionLanguageModel } from './completion/openai-completion-language-model';
@@ -48,6 +50,9 @@ import { VERSION } from './version';
48
50
  export interface OpenAIProvider extends ProviderV4 {
49
51
  (modelId: OpenAIResponsesModelId): LanguageModelV4;
50
52
 
53
+ /** Creates an experimental Choice/Score/Boolean evaluation model using the Responses API. */
54
+ evaluationModel(modelId: OpenAIResponsesModelId): EvaluationModelV4;
55
+
51
56
  /**
52
57
  * Creates an OpenAI model for text generation.
53
58
  */
@@ -347,6 +352,11 @@ export function createOpenAI(
347
352
  provider.chat = createChatModel;
348
353
  provider.completion = createCompletionModel;
349
354
  provider.responses = createResponsesModel;
355
+ provider.evaluationModel = (modelId: OpenAIResponsesModelId) =>
356
+ new EvaluationLanguageModel({
357
+ model: createResponsesModel(modelId),
358
+ provider: `${providerName}.evaluation`,
359
+ });
350
360
  provider.embedding = createEmbeddingModel;
351
361
  provider.embeddingModel = createEmbeddingModel;
352
362
  provider.textEmbedding = createEmbeddingModel;
@@ -137,6 +137,32 @@ async function convertFunctionToolResultOutput({
137
137
  const imageDetail =
138
138
  item.providerOptions?.[providerOptionsName]?.imageDetail;
139
139
 
140
+ if (item.data.type === 'reference') {
141
+ const fileId = resolveProviderReference({
142
+ reference: item.data.reference,
143
+ provider: providerOptionsName,
144
+ });
145
+
146
+ if (topLevel === 'image') {
147
+ return {
148
+ type: 'input_image' as const,
149
+ file_id: fileId,
150
+ detail: imageDetail,
151
+ ...(promptCacheBreakpoint != null && {
152
+ prompt_cache_breakpoint: promptCacheBreakpoint,
153
+ }),
154
+ };
155
+ }
156
+
157
+ return {
158
+ type: 'input_file' as const,
159
+ file_id: fileId,
160
+ ...(promptCacheBreakpoint != null && {
161
+ prompt_cache_breakpoint: promptCacheBreakpoint,
162
+ }),
163
+ };
164
+ }
165
+
140
166
  if (item.data.type === 'data') {
141
167
  const fullMediaType = resolveFullMediaType({ part: item });
142
168
  if (topLevel === 'image') {
@@ -611,8 +637,7 @@ export async function convertToOpenAIResponsesInput({
611
637
  input.push({
612
638
  ...(explicitMessageItemType && { type: 'message' as const }),
613
639
  role: 'assistant',
614
- content: [{ type: 'output_text', text: part.text }],
615
- id,
640
+ content: part.text,
616
641
  ...(phase != null && { phase }),
617
642
  });
618
643
 
@@ -266,8 +266,7 @@ export type OpenAIResponsesUserMessage = {
266
266
  export type OpenAIResponsesAssistantMessage = {
267
267
  type?: 'message';
268
268
  role: 'assistant';
269
- content: Array<{ type: 'output_text'; text: string }>;
270
- id?: string;
269
+ content: string;
271
270
  phase?: 'commentary' | 'final_answer' | null;
272
271
  };
273
272
 
@@ -298,6 +297,11 @@ export type OpenAIResponsesFunctionCallOutput = {
298
297
  image_url: string;
299
298
  prompt_cache_breakpoint?: { mode: 'explicit' };
300
299
  }
300
+ | {
301
+ type: 'input_image';
302
+ file_id: string;
303
+ prompt_cache_breakpoint?: { mode: 'explicit' };
304
+ }
301
305
  | {
302
306
  type: 'input_file';
303
307
  filename: string;
@@ -309,6 +313,11 @@ export type OpenAIResponsesFunctionCallOutput = {
309
313
  file_url: string;
310
314
  prompt_cache_breakpoint?: { mode: 'explicit' };
311
315
  }
316
+ | {
317
+ type: 'input_file';
318
+ file_id: string;
319
+ prompt_cache_breakpoint?: { mode: 'explicit' };
320
+ }
312
321
  >;
313
322
  caller?: OpenAIResponsesToolCaller;
314
323
  };