@ai-sdk/openai 3.0.108 → 3.0.109

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -206,9 +206,23 @@ The following provider options are available:
206
206
 
207
207
  <Note>
208
208
  Supported reasoning efforts vary by model. GPT-5.6 supports `'none'`, `'low'`,
209
- `'medium'`, `'high'`, `'xhigh'`, and `'max'`.
209
+ `'medium'`, `'high'`, `'xhigh'`, and `'max'`. GPT-6 and later models support
210
+ `'low'`, `'medium'`, `'high'`, `'xhigh'`, and `'max'`.
210
211
  </Note>
211
212
 
213
+ <Note>
214
+ GPT-6 and later models do not support `temperature`, `topP`, `logprobs`, or
215
+ the legacy `promptCacheRetention` option. The provider removes these settings
216
+ and returns a warning. Use the Responses API for GPT-6 tool calling.
217
+ </Note>
218
+
219
+ - **reasoningEffortUpdate** _'low' | 'medium' | 'high' | 'xhigh' | 'max'_
220
+ Updates the reasoning effort for GPT-6 and later models starting with the
221
+ current response without changing the request-level effort. Use this with
222
+ `previousResponseId` to preserve the original prompt prefix for caching.
223
+ Configuration updates require standard, single-agent mode and cannot be
224
+ combined with automatic truncation.
225
+
212
226
  - **reasoningMode** _'standard' | 'pro'_
213
227
  Controls how much model work GPT-5.6 performs before returning a final answer. `'standard'` is the default. Use `'pro'` for difficult tasks where quality matters more than latency and token usage.
214
228
 
@@ -308,6 +322,58 @@ The following OpenAI-specific metadata may be returned:
308
322
  - **reasoningContext** _(optional)_
309
323
  Effective persisted-reasoning context returned by GPT-5.6 (`'current_turn'` or `'all_turns'`).
310
324
 
325
+ #### Changing Reasoning Effort Mid-Conversation
326
+
327
+ GPT-6 and later models can change reasoning effort between responses without
328
+ changing the request-level `reasoningEffort` setting. The provider sends
329
+ `reasoningEffortUpdate` as an OpenAI `configuration_update` input item before
330
+ the next user message. Keeping the request-level effort unchanged preserves the
331
+ original prompt prefix for prompt caching.
332
+
333
+ ```ts highlight="13,32,34"
334
+ import {
335
+ openai,
336
+ type OpenAILanguageModelResponsesOptions,
337
+ type OpenaiResponsesProviderMetadata,
338
+ } from '@ai-sdk/openai';
339
+ import { generateText } from 'ai';
340
+
341
+ const first = await generateText({
342
+ model: openai.responses('gpt-6-astra'),
343
+ prompt: 'Draft a database migration plan.',
344
+ providerOptions: {
345
+ openai: {
346
+ reasoningEffort: 'low',
347
+ } satisfies OpenAILanguageModelResponsesOptions,
348
+ },
349
+ });
350
+
351
+ const metadata = first.providerMetadata as
352
+ | OpenaiResponsesProviderMetadata
353
+ | undefined;
354
+ const previousResponseId = metadata?.openai.responseId;
355
+
356
+ if (!previousResponseId) {
357
+ throw new Error('OpenAI did not return a response ID.');
358
+ }
359
+
360
+ const second = await generateText({
361
+ model: openai.responses('gpt-6-astra'),
362
+ prompt: 'Analyze the failure modes and propose rollback steps.',
363
+ providerOptions: {
364
+ openai: {
365
+ previousResponseId,
366
+ reasoningEffort: 'low',
367
+ reasoningEffortUpdate: 'high',
368
+ } satisfies OpenAILanguageModelResponsesOptions,
369
+ },
370
+ });
371
+ ```
372
+
373
+ The response metadata continues to report the request-level reasoning effort,
374
+ not the effective effort selected by `reasoningEffortUpdate`. Configuration
375
+ updates are not supported with `reasoningMode: 'pro'` or `truncation: 'auto'`.
376
+
311
377
  #### Reasoning Output
312
378
 
313
379
  For reasoning models like `gpt-5`, you can enable reasoning summaries to see the model's thought process. Different models support different summarizers—for example, `o4-mini` supports detailed summaries. Set `reasoningSummary: "auto"` to automatically receive the richest level available.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/openai",
3
- "version": "3.0.108",
3
+ "version": "3.0.109",
4
4
  "license": "Apache-2.0",
5
5
  "sideEffects": false,
6
6
  "main": "./dist/index.js",
@@ -97,6 +97,23 @@ export class OpenAIChatLanguageModel implements LanguageModelV3 {
97
97
  const isReasoningModel =
98
98
  openaiOptions.forceReasoning ?? modelCapabilities.isReasoningModel;
99
99
 
100
+ let resolvedReasoningEffort = openaiOptions.reasoningEffort;
101
+
102
+ if (
103
+ resolvedReasoningEffort != null &&
104
+ modelCapabilities.supportedReasoningEfforts != null &&
105
+ !modelCapabilities.supportedReasoningEfforts.includes(
106
+ resolvedReasoningEffort,
107
+ )
108
+ ) {
109
+ warnings.push({
110
+ type: 'unsupported',
111
+ feature: 'reasoningEffort',
112
+ details: `${this.modelId} only supports the following reasoning efforts: ${modelCapabilities.supportedReasoningEfforts.join(', ')}`,
113
+ });
114
+ resolvedReasoningEffort = undefined;
115
+ }
116
+
100
117
  if (topK != null) {
101
118
  warnings.push({ type: 'unsupported', feature: 'topK' });
102
119
  }
@@ -168,7 +185,7 @@ export class OpenAIChatLanguageModel implements LanguageModelV3 {
168
185
  store: openaiOptions.store,
169
186
  metadata: openaiOptions.metadata,
170
187
  prediction: openaiOptions.prediction,
171
- reasoning_effort: openaiOptions.reasoningEffort,
188
+ reasoning_effort: resolvedReasoningEffort,
172
189
  service_tier: openaiOptions.serviceTier,
173
190
  prompt_cache_key: openaiOptions.promptCacheKey,
174
191
  prompt_cache_options: openaiOptions.promptCacheOptions,
@@ -179,6 +196,19 @@ export class OpenAIChatLanguageModel implements LanguageModelV3 {
179
196
  messages,
180
197
  };
181
198
 
199
+ if (
200
+ modelCapabilities.supportedReasoningEfforts != null &&
201
+ baseArgs.prompt_cache_retention != null
202
+ ) {
203
+ baseArgs.prompt_cache_retention = undefined;
204
+ warnings.push({
205
+ type: 'unsupported',
206
+ feature: 'promptCacheRetention',
207
+ details:
208
+ 'promptCacheRetention is not supported by GPT-6 and later models; use promptCacheOptions instead',
209
+ });
210
+ }
211
+
182
212
  // remove unsupported settings for reasoning models
183
213
  // see https://platform.openai.com/docs/guides/reasoning#limitations
184
214
  if (isReasoningModel) {
@@ -108,6 +108,7 @@ export const openaiLanguageModelChatOptions = lazySchema(() =>
108
108
 
109
109
  /**
110
110
  * Reasoning effort for reasoning models. Defaults to `medium`.
111
+ * GPT-6 and later models support 'low' | 'medium' | 'high' | 'xhigh' | 'max'.
111
112
  */
112
113
  reasoningEffort: z
113
114
  .enum(['none', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max'])
@@ -3,6 +3,8 @@ export type OpenAILanguageModelCapabilities = {
3
3
  systemMessageMode: 'remove' | 'system' | 'developer';
4
4
  supportsFlexProcessing: boolean;
5
5
  supportsPriorityProcessing: boolean;
6
+ supportsConfigurationUpdate: boolean;
7
+ supportedReasoningEfforts: readonly string[] | undefined;
6
8
 
7
9
  /**
8
10
  * Allow temperature, topP, logProbs when reasoningEffort is none.
@@ -19,6 +21,7 @@ export function getOpenAILanguageModelCapabilities(
19
21
  gptVersion?.minor == null &&
20
22
  (gptVersion?.variant?.startsWith('chat') ?? false);
21
23
  const isGptNanoModel = gptVersion?.variant?.startsWith('nano') ?? false;
24
+ const isGpt6OrLaterModel = gptVersion != null && gptVersion.major >= 6;
22
25
 
23
26
  const supportsFlexProcessing =
24
27
  (oSeriesVersion != null && oSeriesVersion >= 3) ||
@@ -41,6 +44,7 @@ export function getOpenAILanguageModelCapabilities(
41
44
  // https://platform.openai.com/docs/guides/latest-model#gpt-5-1-parameter-compatibility
42
45
  // GPT-5.1 and later model families support temperature, topP, logProbs when reasoningEffort is none.
43
46
  const supportsNonReasoningParameters =
47
+ !isGpt6OrLaterModel &&
44
48
  gptVersion != null &&
45
49
  (gptVersion.major > 5 ||
46
50
  (gptVersion.major === 5 && (gptVersion.minor ?? 0) >= 1));
@@ -50,6 +54,10 @@ export function getOpenAILanguageModelCapabilities(
50
54
  return {
51
55
  supportsFlexProcessing,
52
56
  supportsPriorityProcessing,
57
+ supportsConfigurationUpdate: isGpt6OrLaterModel,
58
+ supportedReasoningEfforts: isGpt6OrLaterModel
59
+ ? ['low', 'medium', 'high', 'xhigh', 'max']
60
+ : undefined,
53
61
  isReasoningModel,
54
62
  systemMessageMode,
55
63
  supportsNonReasoningParameters,
@@ -87,7 +87,8 @@ export type OpenAIResponsesInputItem =
87
87
  | OpenAIResponsesToolSearchCall
88
88
  | OpenAIResponsesToolSearchOutput
89
89
  | OpenAIResponsesReasoning
90
- | OpenAIResponsesItemReference;
90
+ | OpenAIResponsesItemReference
91
+ | OpenAIResponsesConfigurationUpdate;
91
92
 
92
93
  export type OpenAIResponsesIncludeValue =
93
94
  | 'web_search_call.action.sources'
@@ -579,6 +580,13 @@ export type OpenAIResponsesReasoning = {
579
580
  }>;
580
581
  };
581
582
 
583
+ export type OpenAIResponsesConfigurationUpdate = {
584
+ type: 'configuration_update';
585
+ reasoning: {
586
+ effort: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
587
+ };
588
+ };
589
+
582
590
  // Captured from the Responses API when OpenAI returned an early
583
591
  // insufficient_quota stream error after HTTP 200. This shape differs from the
584
592
  // currently documented ResponseErrorEvent below.
@@ -186,6 +186,23 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV3 {
186
186
  const isReasoningModel =
187
187
  openaiOptions?.forceReasoning ?? modelCapabilities.isReasoningModel;
188
188
 
189
+ let resolvedReasoningEffort = openaiOptions?.reasoningEffort;
190
+
191
+ if (
192
+ resolvedReasoningEffort != null &&
193
+ modelCapabilities.supportedReasoningEfforts != null &&
194
+ !modelCapabilities.supportedReasoningEfforts.includes(
195
+ resolvedReasoningEffort,
196
+ )
197
+ ) {
198
+ warnings.push({
199
+ type: 'unsupported',
200
+ feature: 'reasoningEffort',
201
+ details: `${this.modelId} only supports the following reasoning efforts: ${modelCapabilities.supportedReasoningEfforts.join(', ')}`,
202
+ });
203
+ resolvedReasoningEffort = undefined;
204
+ }
205
+
189
206
  if (openaiOptions?.conversation && openaiOptions?.previousResponseId) {
190
207
  warnings.push({
191
208
  type: 'unsupported',
@@ -255,6 +272,28 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV3 {
255
272
 
256
273
  warnings.push(...inputWarnings);
257
274
 
275
+ const reasoningEffortUpdate = openaiOptions?.reasoningEffortUpdate;
276
+ const configurationUpdateIsSupported =
277
+ reasoningEffortUpdate == null ||
278
+ (modelCapabilities.supportsConfigurationUpdate &&
279
+ openaiOptions?.reasoningMode !== 'pro' &&
280
+ openaiOptions?.truncation !== 'auto');
281
+
282
+ if (reasoningEffortUpdate != null && !configurationUpdateIsSupported) {
283
+ warnings.push({
284
+ type: 'unsupported',
285
+ feature: 'reasoningEffortUpdate',
286
+ details: !modelCapabilities.supportsConfigurationUpdate
287
+ ? 'reasoningEffortUpdate is only supported by GPT-6 and later models'
288
+ : 'reasoningEffortUpdate requires standard reasoning mode without automatic truncation',
289
+ });
290
+ } else if (reasoningEffortUpdate != null) {
291
+ input.unshift({
292
+ type: 'configuration_update',
293
+ reasoning: { effort: reasoningEffortUpdate },
294
+ });
295
+ }
296
+
258
297
  const strictJsonSchema = openaiOptions?.strictJsonSchema ?? true;
259
298
 
260
299
  let include: OpenAIResponsesIncludeOptions = openaiOptions?.include;
@@ -361,13 +400,13 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV3 {
361
400
 
362
401
  // model-specific settings:
363
402
  ...(isReasoningModel &&
364
- (openaiOptions?.reasoningEffort != null ||
403
+ (resolvedReasoningEffort != null ||
365
404
  openaiOptions?.reasoningSummary != null ||
366
405
  openaiOptions?.reasoningMode != null ||
367
406
  openaiOptions?.reasoningContext != null) && {
368
407
  reasoning: {
369
- ...(openaiOptions?.reasoningEffort != null && {
370
- effort: openaiOptions.reasoningEffort,
408
+ ...(resolvedReasoningEffort != null && {
409
+ effort: resolvedReasoningEffort,
371
410
  }),
372
411
  ...(openaiOptions?.reasoningSummary != null && {
373
412
  summary: openaiOptions.reasoningSummary,
@@ -382,6 +421,19 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV3 {
382
421
  }),
383
422
  };
384
423
 
424
+ if (
425
+ modelCapabilities.supportsConfigurationUpdate &&
426
+ baseArgs.prompt_cache_retention != null
427
+ ) {
428
+ baseArgs.prompt_cache_retention = undefined;
429
+ warnings.push({
430
+ type: 'unsupported',
431
+ feature: 'promptCacheRetention',
432
+ details:
433
+ 'promptCacheRetention is not supported by GPT-6 and later models; use promptCacheOptions instead',
434
+ });
435
+ }
436
+
385
437
  // remove unsupported settings for reasoning models
386
438
  // see https://platform.openai.com/docs/guides/reasoning#limitations
387
439
  if (isReasoningModel) {
@@ -411,6 +463,26 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV3 {
411
463
  });
412
464
  }
413
465
  }
466
+
467
+ if (
468
+ modelCapabilities.supportedReasoningEfforts != null &&
469
+ (baseArgs.top_logprobs != null ||
470
+ baseArgs.include?.includes('message.output_text.logprobs'))
471
+ ) {
472
+ baseArgs.top_logprobs = undefined;
473
+ const filteredInclude = baseArgs.include?.filter(
474
+ value => value !== 'message.output_text.logprobs',
475
+ );
476
+ baseArgs.include =
477
+ filteredInclude != null && filteredInclude.length > 0
478
+ ? filteredInclude
479
+ : undefined;
480
+ warnings.push({
481
+ type: 'unsupported',
482
+ feature: 'logprobs',
483
+ details: 'logprobs is not supported for reasoning models',
484
+ });
485
+ }
414
486
  } else {
415
487
  if (openaiOptions?.reasoningEffort != null) {
416
488
  warnings.push({
@@ -260,10 +260,23 @@ export const openaiLanguageModelResponsesOptionsSchema = lazySchema(() =>
260
260
  * Reasoning effort for reasoning models. Defaults to `medium`. If you use
261
261
  * `providerOptions` to set the `reasoningEffort` option, this model setting will be ignored.
262
262
  * GPT-5.6 supports 'none' | 'low' | 'medium' | 'high' | 'xhigh' | 'max'.
263
+ * GPT-6 and later models support 'low' | 'medium' | 'high' | 'xhigh' | 'max'.
263
264
  * Supported values vary by model.
264
265
  */
265
266
  reasoningEffort: z.string().nullish(),
266
267
 
268
+ /**
269
+ * Updates the reasoning effort for GPT-6 and later models starting with this response
270
+ * without changing the request-level reasoning effort. This preserves the
271
+ * request prefix for prompt caching.
272
+ *
273
+ * Only supported by GPT-6 and later models in standard, single-agent mode. Cannot be
274
+ * combined with automatic truncation.
275
+ */
276
+ reasoningEffortUpdate: z
277
+ .enum(['low', 'medium', 'high', 'xhigh', 'max'])
278
+ .optional(),
279
+
267
280
  /**
268
281
  * Controls how much model work GPT-5.6 performs before returning a final answer.
269
282
  * `standard` is the default. `pro` increases quality, latency, and token usage.