@ai-sdk/openai 4.0.58 → 4.0.60
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/dist/index.d.ts +16 -8
- package/dist/index.js +166 -26
- package/dist/index.js.map +1 -1
- package/dist/internal/index.d.ts +24 -11
- package/dist/internal/index.js +165 -25
- package/dist/internal/index.js.map +1 -1
- package/docs/03-openai.mdx +95 -7
- package/package.json +1 -1
- package/src/chat/openai-chat-language-model-options.ts +1 -0
- package/src/chat/openai-chat-language-model.ts +29 -1
- package/src/openai-language-model-capabilities.ts +8 -0
- package/src/responses/openai-responses-api.ts +8 -0
- package/src/responses/openai-responses-language-model-options.ts +14 -0
- package/src/responses/openai-responses-language-model.ts +73 -1
- package/src/transcription/openai-transcription-api.ts +22 -12
- package/src/transcription/openai-transcription-model-options.ts +22 -0
- package/src/transcription/openai-transcription-model.ts +55 -1
package/docs/03-openai.mdx
CHANGED
|
@@ -192,9 +192,23 @@ The following provider options are available:
|
|
|
192
192
|
|
|
193
193
|
<Note>
|
|
194
194
|
Supported reasoning efforts vary by model. GPT-5.6 supports `'none'`, `'low'`,
|
|
195
|
-
`'medium'`, `'high'`, `'xhigh'`, and `'max'`.
|
|
195
|
+
`'medium'`, `'high'`, `'xhigh'`, and `'max'`. GPT-6 and later models support
|
|
196
|
+
`'low'`, `'medium'`, `'high'`, `'xhigh'`, and `'max'`.
|
|
196
197
|
</Note>
|
|
197
198
|
|
|
199
|
+
<Note type="warning">
|
|
200
|
+
GPT-6 and later models do not support `temperature`, `topP`, `logprobs`, or
|
|
201
|
+
the legacy `promptCacheRetention` option. The provider removes these settings
|
|
202
|
+
and returns a warning. Use the Responses API for GPT-6 tool calling.
|
|
203
|
+
</Note>
|
|
204
|
+
|
|
205
|
+
- **reasoningEffortUpdate** _'low' | 'medium' | 'high' | 'xhigh' | 'max'_
|
|
206
|
+
Updates the reasoning effort for GPT-6 and later models starting with the
|
|
207
|
+
current response without changing the request-level effort. Use this with
|
|
208
|
+
`previousResponseId` to preserve the original prompt prefix for caching.
|
|
209
|
+
Configuration updates require standard, single-agent mode and cannot be
|
|
210
|
+
combined with automatic compaction or automatic truncation.
|
|
211
|
+
|
|
198
212
|
- **reasoningMode** _'standard' | 'pro'_
|
|
199
213
|
Controls how much model work GPT-5.6 performs before returning a final answer. `'standard'` is the default. Use `'pro'` for difficult tasks where quality matters more than latency and token usage.
|
|
200
214
|
|
|
@@ -302,6 +316,57 @@ The following OpenAI-specific metadata may be returned:
|
|
|
302
316
|
- **reasoningContext** _(optional)_
|
|
303
317
|
Effective persisted-reasoning context returned by GPT-5.6 (`'current_turn'` or `'all_turns'`).
|
|
304
318
|
|
|
319
|
+
#### Changing Reasoning Effort Mid-Conversation
|
|
320
|
+
|
|
321
|
+
GPT-6 and later models can change reasoning effort between responses without
|
|
322
|
+
changing the request-level `reasoning` setting. The provider sends
|
|
323
|
+
`reasoningEffortUpdate` as an OpenAI `configuration_update` input item before
|
|
324
|
+
the next user message. Keeping the request-level effort unchanged preserves the
|
|
325
|
+
original prompt prefix for prompt caching.
|
|
326
|
+
|
|
327
|
+
```ts highlight="8,26,32"
|
|
328
|
+
import {
|
|
329
|
+
openai,
|
|
330
|
+
type OpenAILanguageModelResponsesOptions,
|
|
331
|
+
type OpenaiResponsesProviderMetadata,
|
|
332
|
+
} from '@ai-sdk/openai';
|
|
333
|
+
import { generateText } from 'ai';
|
|
334
|
+
|
|
335
|
+
const first = await generateText({
|
|
336
|
+
model: openai.responses('gpt-6-astra'),
|
|
337
|
+
reasoning: 'low',
|
|
338
|
+
prompt: 'Draft a database migration plan.',
|
|
339
|
+
});
|
|
340
|
+
|
|
341
|
+
const metadata = first.finalStep.providerMetadata as
|
|
342
|
+
| OpenaiResponsesProviderMetadata
|
|
343
|
+
| undefined;
|
|
344
|
+
const previousResponseId = metadata?.openai.responseId;
|
|
345
|
+
|
|
346
|
+
if (!previousResponseId) {
|
|
347
|
+
throw new Error('OpenAI did not return a response ID.');
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
const second = await generateText({
|
|
351
|
+
model: openai.responses('gpt-6-astra'),
|
|
352
|
+
reasoning: 'low',
|
|
353
|
+
prompt: 'Analyze the failure modes and propose rollback steps.',
|
|
354
|
+
providerOptions: {
|
|
355
|
+
openai: {
|
|
356
|
+
previousResponseId,
|
|
357
|
+
reasoningEffortUpdate: 'high',
|
|
358
|
+
} satisfies OpenAILanguageModelResponsesOptions,
|
|
359
|
+
},
|
|
360
|
+
});
|
|
361
|
+
```
|
|
362
|
+
|
|
363
|
+
The response metadata continues to report the request-level reasoning effort,
|
|
364
|
+
not the effective effort selected by `reasoningEffortUpdate`. Configuration
|
|
365
|
+
updates are not supported with `reasoningMode: 'pro'`,
|
|
366
|
+
`contextManagement`, or `truncation: 'auto'`. Explicit compaction with
|
|
367
|
+
`compactionTrigger` remains supported; send a new `reasoningEffortUpdate` after
|
|
368
|
+
compaction when the effort should change again.
|
|
369
|
+
|
|
305
370
|
#### Reasoning Output
|
|
306
371
|
|
|
307
372
|
For reasoning models like `gpt-5`, you can enable reasoning summaries to see the model's thought process. Different models support different summarizers—for example, `o4-mini` supports detailed summaries. Set `reasoningSummary: "auto"` to automatically receive the richest level available. When `reasoningEffort` is set to a value other than `'none'`, the OpenAI Responses provider defaults `reasoningSummary` to `'detailed'`; set `reasoningSummary: null` to omit reasoning summaries.
|
|
@@ -2720,6 +2785,7 @@ The following optional provider options are available for OpenAI completion mode
|
|
|
2720
2785
|
|
|
2721
2786
|
| Model | Image Input | Audio Input | Object Generation | Tool Usage |
|
|
2722
2787
|
| --------------------- | ----------- | ----------- | ----------------- | ---------- |
|
|
2788
|
+
| `gpt-6-astra` | <Check /> | <Cross /> | <Check /> | <Check /> |
|
|
2723
2789
|
| `gpt-5.6` | <Check /> | <Cross /> | <Check /> | <Check /> |
|
|
2724
2790
|
| `gpt-5.6-luna` | <Check /> | <Cross /> | <Check /> | <Check /> |
|
|
2725
2791
|
| `gpt-5.6-sol` | <Check /> | <Cross /> | <Check /> | <Check /> |
|
|
@@ -3022,6 +3088,21 @@ const result = await transcribe({
|
|
|
3022
3088
|
console.log(result.segments); // Array of segments with startSecond/endSecond
|
|
3023
3089
|
```
|
|
3024
3090
|
|
|
3091
|
+
`gpt-4o-transcribe-diarize` identifies speakers in a recording. It defaults to
|
|
3092
|
+
the `diarized_json` response format and automatic chunking:
|
|
3093
|
+
|
|
3094
|
+
```ts
|
|
3095
|
+
import { transcribe } from 'ai';
|
|
3096
|
+
import { openai } from '@ai-sdk/openai';
|
|
3097
|
+
|
|
3098
|
+
const result = await transcribe({
|
|
3099
|
+
model: openai.transcription('gpt-4o-transcribe-diarize'),
|
|
3100
|
+
audio: new Uint8Array([1, 2, 3, 4]),
|
|
3101
|
+
});
|
|
3102
|
+
|
|
3103
|
+
console.log(result.providerMetadata.openai.segments);
|
|
3104
|
+
```
|
|
3105
|
+
|
|
3025
3106
|
The following provider options are available:
|
|
3026
3107
|
|
|
3027
3108
|
- **timestampGranularities** _string[]_
|
|
@@ -3046,6 +3127,12 @@ The following provider options are available:
|
|
|
3046
3127
|
- **include** _string[]_
|
|
3047
3128
|
Additional information to include in the transcription response.
|
|
3048
3129
|
|
|
3130
|
+
- **responseFormat** _'json' | 'verbose_json' | 'diarized_json'_
|
|
3131
|
+
The format of the transcription response. `gpt-4o-transcribe-diarize` defaults to `diarized_json`.
|
|
3132
|
+
|
|
3133
|
+
- **chunkingStrategy** _'auto' | object_
|
|
3134
|
+
Controls how the audio is split into chunks. `gpt-4o-transcribe-diarize` defaults to `'auto'`; provide this option to override the default. The object form configures OpenAI server-side VAD with `type: 'server_vad'` and optional `threshold`, `prefixPaddingMs`, and `silenceDurationMs` fields.
|
|
3135
|
+
|
|
3049
3136
|
- **streaming** _object_
|
|
3050
3137
|
Options for streaming transcription models such as `gpt-realtime-whisper`.
|
|
3051
3138
|
Use with `experimental_streamTranscribe`.
|
|
@@ -3057,12 +3144,13 @@ The following provider options are available:
|
|
|
3057
3144
|
|
|
3058
3145
|
### Model Capabilities
|
|
3059
3146
|
|
|
3060
|
-
| Model
|
|
3061
|
-
|
|
|
3062
|
-
| `whisper-1`
|
|
3063
|
-
| `gpt-4o-mini-transcribe`
|
|
3064
|
-
| `gpt-4o-transcribe`
|
|
3065
|
-
| `gpt-
|
|
3147
|
+
| Model | Transcription | Streaming | Duration | Segments | Language |
|
|
3148
|
+
| --------------------------- | ------------- | --------- | --------- | --------- | --------- |
|
|
3149
|
+
| `whisper-1` | <Check /> | <Cross /> | <Check /> | <Check /> | <Check /> |
|
|
3150
|
+
| `gpt-4o-mini-transcribe` | <Check /> | <Cross /> | <Cross /> | <Cross /> | <Cross /> |
|
|
3151
|
+
| `gpt-4o-transcribe` | <Check /> | <Cross /> | <Cross /> | <Cross /> | <Cross /> |
|
|
3152
|
+
| `gpt-4o-transcribe-diarize` | <Check /> | <Cross /> | <Check /> | <Check /> | <Cross /> |
|
|
3153
|
+
| `gpt-realtime-whisper` | <Cross /> | <Check /> | <Cross /> | <Cross /> | <Cross /> |
|
|
3066
3154
|
|
|
3067
3155
|
## Translation Models
|
|
3068
3156
|
|
package/package.json
CHANGED
|
@@ -118,10 +118,25 @@ export class OpenAIChatLanguageModel implements LanguageModelV4 {
|
|
|
118
118
|
const modelCapabilities = getOpenAILanguageModelCapabilities(this.modelId);
|
|
119
119
|
|
|
120
120
|
// AI SDK reasoning values map directly to the OpenAI reasoning values.
|
|
121
|
-
|
|
121
|
+
let resolvedReasoningEffort =
|
|
122
122
|
openaiOptions.reasoningEffort ??
|
|
123
123
|
(isCustomReasoning(reasoning) ? reasoning : undefined);
|
|
124
124
|
|
|
125
|
+
if (
|
|
126
|
+
resolvedReasoningEffort != null &&
|
|
127
|
+
modelCapabilities.supportedReasoningEfforts != null &&
|
|
128
|
+
!modelCapabilities.supportedReasoningEfforts.includes(
|
|
129
|
+
resolvedReasoningEffort,
|
|
130
|
+
)
|
|
131
|
+
) {
|
|
132
|
+
warnings.push({
|
|
133
|
+
type: 'unsupported',
|
|
134
|
+
feature: 'reasoningEffort',
|
|
135
|
+
details: `${this.modelId} only supports the following reasoning efforts: ${modelCapabilities.supportedReasoningEfforts.join(', ')}`,
|
|
136
|
+
});
|
|
137
|
+
resolvedReasoningEffort = undefined;
|
|
138
|
+
}
|
|
139
|
+
|
|
125
140
|
const isReasoningModel =
|
|
126
141
|
openaiOptions.forceReasoning ?? modelCapabilities.isReasoningModel;
|
|
127
142
|
|
|
@@ -207,6 +222,19 @@ export class OpenAIChatLanguageModel implements LanguageModelV4 {
|
|
|
207
222
|
messages,
|
|
208
223
|
};
|
|
209
224
|
|
|
225
|
+
if (
|
|
226
|
+
modelCapabilities.supportedReasoningEfforts != null &&
|
|
227
|
+
baseArgs.prompt_cache_retention != null
|
|
228
|
+
) {
|
|
229
|
+
baseArgs.prompt_cache_retention = undefined;
|
|
230
|
+
warnings.push({
|
|
231
|
+
type: 'unsupported',
|
|
232
|
+
feature: 'promptCacheRetention',
|
|
233
|
+
details:
|
|
234
|
+
'promptCacheRetention is not supported by GPT-6 and later models; use promptCacheOptions instead',
|
|
235
|
+
});
|
|
236
|
+
}
|
|
237
|
+
|
|
210
238
|
// remove unsupported settings for reasoning models
|
|
211
239
|
// see https://platform.openai.com/docs/guides/reasoning#limitations
|
|
212
240
|
if (isReasoningModel) {
|
|
@@ -3,6 +3,8 @@ export type OpenAILanguageModelCapabilities = {
|
|
|
3
3
|
systemMessageMode: 'remove' | 'system' | 'developer';
|
|
4
4
|
supportsFlexProcessing: boolean;
|
|
5
5
|
supportsPriorityProcessing: boolean;
|
|
6
|
+
supportsConfigurationUpdate: boolean;
|
|
7
|
+
supportedReasoningEfforts: readonly string[] | undefined;
|
|
6
8
|
|
|
7
9
|
/**
|
|
8
10
|
* Allow temperature, topP, logProbs when reasoningEffort is none.
|
|
@@ -19,6 +21,7 @@ export function getOpenAILanguageModelCapabilities(
|
|
|
19
21
|
gptVersion?.minor == null &&
|
|
20
22
|
(gptVersion?.variant?.startsWith('chat') ?? false);
|
|
21
23
|
const isGptNanoModel = gptVersion?.variant?.startsWith('nano') ?? false;
|
|
24
|
+
const isGpt6OrLaterModel = gptVersion != null && gptVersion.major >= 6;
|
|
22
25
|
|
|
23
26
|
const supportsFlexProcessing =
|
|
24
27
|
(oSeriesVersion != null && oSeriesVersion >= 3) ||
|
|
@@ -41,6 +44,7 @@ export function getOpenAILanguageModelCapabilities(
|
|
|
41
44
|
// https://platform.openai.com/docs/guides/latest-model#gpt-5-1-parameter-compatibility
|
|
42
45
|
// GPT-5.1 and later model families support temperature, topP, logProbs when reasoningEffort is none.
|
|
43
46
|
const supportsNonReasoningParameters =
|
|
47
|
+
!isGpt6OrLaterModel &&
|
|
44
48
|
gptVersion != null &&
|
|
45
49
|
(gptVersion.major > 5 ||
|
|
46
50
|
(gptVersion.major === 5 && (gptVersion.minor ?? 0) >= 1));
|
|
@@ -50,6 +54,10 @@ export function getOpenAILanguageModelCapabilities(
|
|
|
50
54
|
return {
|
|
51
55
|
supportsFlexProcessing,
|
|
52
56
|
supportsPriorityProcessing,
|
|
57
|
+
supportsConfigurationUpdate: isGpt6OrLaterModel,
|
|
58
|
+
supportedReasoningEfforts: isGpt6OrLaterModel
|
|
59
|
+
? ['low', 'medium', 'high', 'xhigh', 'max']
|
|
60
|
+
: undefined,
|
|
53
61
|
isReasoningModel,
|
|
54
62
|
systemMessageMode,
|
|
55
63
|
supportsNonReasoningParameters,
|
|
@@ -180,6 +180,7 @@ export type OpenAIResponsesInputItem =
|
|
|
180
180
|
| OpenAIResponsesReasoning
|
|
181
181
|
| OpenAIResponsesItemReference
|
|
182
182
|
| OpenAIResponsesCompactionItem
|
|
183
|
+
| OpenAIResponsesConfigurationUpdate
|
|
183
184
|
| OpenAIResponsesCompactionTrigger;
|
|
184
185
|
|
|
185
186
|
export type OpenAIResponsesIncludeValue =
|
|
@@ -468,6 +469,13 @@ export type OpenAIResponsesCompactionItem = {
|
|
|
468
469
|
encrypted_content: string;
|
|
469
470
|
};
|
|
470
471
|
|
|
472
|
+
export type OpenAIResponsesConfigurationUpdate = {
|
|
473
|
+
type: 'configuration_update';
|
|
474
|
+
reasoning: {
|
|
475
|
+
effort: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
476
|
+
};
|
|
477
|
+
};
|
|
478
|
+
|
|
471
479
|
export type OpenAIResponsesCompactionTrigger = {
|
|
472
480
|
type: 'compaction_trigger';
|
|
473
481
|
};
|
|
@@ -57,6 +57,7 @@ export const openaiResponsesReasoningModelIds = [
|
|
|
57
57
|
'gpt-5.6-luna',
|
|
58
58
|
'gpt-5.6-sol',
|
|
59
59
|
'gpt-5.6-terra',
|
|
60
|
+
'gpt-6-astra',
|
|
60
61
|
] as const;
|
|
61
62
|
|
|
62
63
|
export const openaiResponsesModelIds = [
|
|
@@ -129,6 +130,7 @@ export type OpenAIResponsesModelId =
|
|
|
129
130
|
| 'gpt-5.6-luna'
|
|
130
131
|
| 'gpt-5.6-sol'
|
|
131
132
|
| 'gpt-5.6-terra'
|
|
133
|
+
| 'gpt-6-astra'
|
|
132
134
|
| 'gpt-5-2025-08-07'
|
|
133
135
|
| 'gpt-5-chat-latest'
|
|
134
136
|
| 'gpt-5-codex'
|
|
@@ -262,6 +264,18 @@ export const openaiLanguageModelResponsesOptionsSchema = lazySchema(() =>
|
|
|
262
264
|
*/
|
|
263
265
|
reasoningEffort: z.string().nullish(),
|
|
264
266
|
|
|
267
|
+
/**
|
|
268
|
+
* Updates the reasoning effort for GPT-6 and later models starting with this response
|
|
269
|
+
* without changing the request-level reasoning effort. This preserves the
|
|
270
|
+
* request prefix for prompt caching.
|
|
271
|
+
*
|
|
272
|
+
* Only supported by GPT-6 and later models in standard, single-agent mode. Cannot be
|
|
273
|
+
* combined with automatic compaction or automatic truncation.
|
|
274
|
+
*/
|
|
275
|
+
reasoningEffortUpdate: z
|
|
276
|
+
.enum(['low', 'medium', 'high', 'xhigh', 'max'])
|
|
277
|
+
.optional(),
|
|
278
|
+
|
|
265
279
|
/**
|
|
266
280
|
* Controls how much model work GPT-5.6 performs before returning a final answer.
|
|
267
281
|
* `standard` is the default. `pro` increases quality, latency, and token usage.
|
|
@@ -296,9 +296,25 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV4 {
|
|
|
296
296
|
});
|
|
297
297
|
}
|
|
298
298
|
|
|
299
|
-
|
|
299
|
+
let resolvedReasoningEffort =
|
|
300
300
|
openaiOptions?.reasoningEffort ??
|
|
301
301
|
(isCustomReasoning(reasoning) ? reasoning : undefined);
|
|
302
|
+
|
|
303
|
+
if (
|
|
304
|
+
resolvedReasoningEffort != null &&
|
|
305
|
+
modelCapabilities.supportedReasoningEfforts != null &&
|
|
306
|
+
!modelCapabilities.supportedReasoningEfforts.includes(
|
|
307
|
+
resolvedReasoningEffort,
|
|
308
|
+
)
|
|
309
|
+
) {
|
|
310
|
+
warnings.push({
|
|
311
|
+
type: 'unsupported',
|
|
312
|
+
feature: 'reasoningEffort',
|
|
313
|
+
details: `${this.modelId} only supports the following reasoning efforts: ${modelCapabilities.supportedReasoningEfforts.join(', ')}`,
|
|
314
|
+
});
|
|
315
|
+
resolvedReasoningEffort = undefined;
|
|
316
|
+
}
|
|
317
|
+
|
|
302
318
|
const resolvedReasoningSummary =
|
|
303
319
|
openaiOptions?.reasoningSummary !== undefined
|
|
304
320
|
? openaiOptions.reasoningSummary
|
|
@@ -384,6 +400,29 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV4 {
|
|
|
384
400
|
|
|
385
401
|
warnings.push(...inputWarnings);
|
|
386
402
|
|
|
403
|
+
const reasoningEffortUpdate = openaiOptions?.reasoningEffortUpdate;
|
|
404
|
+
const configurationUpdateIsSupported =
|
|
405
|
+
reasoningEffortUpdate == null ||
|
|
406
|
+
(modelCapabilities.supportsConfigurationUpdate &&
|
|
407
|
+
openaiOptions?.reasoningMode !== 'pro' &&
|
|
408
|
+
openaiOptions?.contextManagement == null &&
|
|
409
|
+
openaiOptions?.truncation !== 'auto');
|
|
410
|
+
|
|
411
|
+
if (reasoningEffortUpdate != null && !configurationUpdateIsSupported) {
|
|
412
|
+
warnings.push({
|
|
413
|
+
type: 'unsupported',
|
|
414
|
+
feature: 'reasoningEffortUpdate',
|
|
415
|
+
details: !modelCapabilities.supportsConfigurationUpdate
|
|
416
|
+
? 'reasoningEffortUpdate is only supported by GPT-6 and later models'
|
|
417
|
+
: 'reasoningEffortUpdate requires standard reasoning mode without automatic compaction or automatic truncation',
|
|
418
|
+
});
|
|
419
|
+
} else if (reasoningEffortUpdate != null) {
|
|
420
|
+
input.unshift({
|
|
421
|
+
type: 'configuration_update',
|
|
422
|
+
reasoning: { effort: reasoningEffortUpdate },
|
|
423
|
+
});
|
|
424
|
+
}
|
|
425
|
+
|
|
387
426
|
// A compaction trigger is a request control, not conversation history.
|
|
388
427
|
// OpenAI requires it to be the final input item, so append it only after
|
|
389
428
|
// the complete prompt has been converted.
|
|
@@ -526,6 +565,19 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV4 {
|
|
|
526
565
|
}),
|
|
527
566
|
};
|
|
528
567
|
|
|
568
|
+
if (
|
|
569
|
+
modelCapabilities.supportsConfigurationUpdate &&
|
|
570
|
+
baseArgs.prompt_cache_retention != null
|
|
571
|
+
) {
|
|
572
|
+
baseArgs.prompt_cache_retention = undefined;
|
|
573
|
+
warnings.push({
|
|
574
|
+
type: 'unsupported',
|
|
575
|
+
feature: 'promptCacheRetention',
|
|
576
|
+
details:
|
|
577
|
+
'promptCacheRetention is not supported by GPT-6 and later models; use promptCacheOptions instead',
|
|
578
|
+
});
|
|
579
|
+
}
|
|
580
|
+
|
|
529
581
|
// remove unsupported settings for reasoning models
|
|
530
582
|
// see https://platform.openai.com/docs/guides/reasoning#limitations
|
|
531
583
|
if (isReasoningModel) {
|
|
@@ -554,6 +606,26 @@ export class OpenAIResponsesLanguageModel implements LanguageModelV4 {
|
|
|
554
606
|
details: 'topP is not supported for reasoning models',
|
|
555
607
|
});
|
|
556
608
|
}
|
|
609
|
+
|
|
610
|
+
if (
|
|
611
|
+
modelCapabilities.supportedReasoningEfforts != null &&
|
|
612
|
+
(baseArgs.top_logprobs != null ||
|
|
613
|
+
baseArgs.include?.includes('message.output_text.logprobs'))
|
|
614
|
+
) {
|
|
615
|
+
baseArgs.top_logprobs = undefined;
|
|
616
|
+
const filteredInclude = baseArgs.include?.filter(
|
|
617
|
+
value => value !== 'message.output_text.logprobs',
|
|
618
|
+
);
|
|
619
|
+
baseArgs.include =
|
|
620
|
+
filteredInclude != null && filteredInclude.length > 0
|
|
621
|
+
? filteredInclude
|
|
622
|
+
: undefined;
|
|
623
|
+
warnings.push({
|
|
624
|
+
type: 'unsupported',
|
|
625
|
+
feature: 'logprobs',
|
|
626
|
+
details: 'logprobs is not supported for reasoning models',
|
|
627
|
+
});
|
|
628
|
+
}
|
|
557
629
|
}
|
|
558
630
|
} else {
|
|
559
631
|
if (openaiOptions?.reasoningEffort != null) {
|
|
@@ -18,18 +18,28 @@ export const openaiTranscriptionResponseSchema = lazySchema(() =>
|
|
|
18
18
|
.nullish(),
|
|
19
19
|
segments: z
|
|
20
20
|
.array(
|
|
21
|
-
z.
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
21
|
+
z.union([
|
|
22
|
+
z.object({
|
|
23
|
+
id: z.number(),
|
|
24
|
+
seek: z.number(),
|
|
25
|
+
start: z.number(),
|
|
26
|
+
end: z.number(),
|
|
27
|
+
text: z.string(),
|
|
28
|
+
tokens: z.array(z.number()),
|
|
29
|
+
temperature: z.number(),
|
|
30
|
+
avg_logprob: z.number(),
|
|
31
|
+
compression_ratio: z.number(),
|
|
32
|
+
no_speech_prob: z.number(),
|
|
33
|
+
}),
|
|
34
|
+
z.object({
|
|
35
|
+
type: z.literal('transcript.text.segment'),
|
|
36
|
+
id: z.string(),
|
|
37
|
+
start: z.number(),
|
|
38
|
+
end: z.number(),
|
|
39
|
+
text: z.string(),
|
|
40
|
+
speaker: z.string(),
|
|
41
|
+
}),
|
|
42
|
+
]),
|
|
33
43
|
)
|
|
34
44
|
.nullish(),
|
|
35
45
|
}),
|
|
@@ -50,6 +50,28 @@ export const openAITranscriptionModelOptions = lazySchema(() =>
|
|
|
50
50
|
.default(['segment'])
|
|
51
51
|
.optional(),
|
|
52
52
|
|
|
53
|
+
/**
|
|
54
|
+
* The format of the transcription response.
|
|
55
|
+
*/
|
|
56
|
+
responseFormat: z
|
|
57
|
+
.enum(['json', 'verbose_json', 'diarized_json'])
|
|
58
|
+
.optional(),
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Controls how the audio is split into chunks before transcription.
|
|
62
|
+
*/
|
|
63
|
+
chunkingStrategy: z
|
|
64
|
+
.union([
|
|
65
|
+
z.literal('auto'),
|
|
66
|
+
z.object({
|
|
67
|
+
type: z.literal('server_vad'),
|
|
68
|
+
threshold: z.number().min(0).max(1).optional(),
|
|
69
|
+
prefixPaddingMs: z.number().int().min(0).optional(),
|
|
70
|
+
silenceDurationMs: z.number().int().min(0).optional(),
|
|
71
|
+
}),
|
|
72
|
+
])
|
|
73
|
+
.optional(),
|
|
74
|
+
|
|
53
75
|
/**
|
|
54
76
|
* Options for streaming transcription models such as `gpt-realtime-whisper`.
|
|
55
77
|
*/
|
|
@@ -195,6 +195,11 @@ export class OpenAITranscriptionModel implements TranscriptionModelV4 {
|
|
|
195
195
|
formData.append('response_format', 'verbose_json');
|
|
196
196
|
}
|
|
197
197
|
|
|
198
|
+
const isDiarizationModel = this.modelId === 'gpt-4o-transcribe-diarize';
|
|
199
|
+
const chunkingStrategy =
|
|
200
|
+
openAIOptions?.chunkingStrategy ??
|
|
201
|
+
(isDiarizationModel ? 'auto' : undefined);
|
|
202
|
+
|
|
198
203
|
// Add provider-specific options
|
|
199
204
|
if (openAIOptions) {
|
|
200
205
|
const isGpt4oTranscribeModel = [
|
|
@@ -209,7 +214,13 @@ export class OpenAITranscriptionModel implements TranscriptionModelV4 {
|
|
|
209
214
|
// https://platform.openai.com/docs/api-reference/audio/createTranscription#audio_createtranscription-response_format
|
|
210
215
|
// prefer verbose_json to get segments for models that support it
|
|
211
216
|
...(this.modelId !== 'whisper-1' && {
|
|
212
|
-
response_format:
|
|
217
|
+
response_format:
|
|
218
|
+
openAIOptions.responseFormat ??
|
|
219
|
+
(isDiarizationModel
|
|
220
|
+
? 'diarized_json'
|
|
221
|
+
: isGpt4oTranscribeModel
|
|
222
|
+
? 'json'
|
|
223
|
+
: 'verbose_json'),
|
|
213
224
|
}),
|
|
214
225
|
temperature: openAIOptions.temperature,
|
|
215
226
|
timestamp_granularities: openAIOptions.timestampGranularities,
|
|
@@ -226,6 +237,28 @@ export class OpenAITranscriptionModel implements TranscriptionModelV4 {
|
|
|
226
237
|
}
|
|
227
238
|
}
|
|
228
239
|
}
|
|
240
|
+
} else if (isDiarizationModel) {
|
|
241
|
+
formData.append('response_format', 'diarized_json');
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
if (chunkingStrategy != null) {
|
|
245
|
+
formData.append(
|
|
246
|
+
'chunking_strategy',
|
|
247
|
+
typeof chunkingStrategy === 'string'
|
|
248
|
+
? chunkingStrategy
|
|
249
|
+
: JSON.stringify({
|
|
250
|
+
type: chunkingStrategy.type,
|
|
251
|
+
...(chunkingStrategy.threshold != null && {
|
|
252
|
+
threshold: chunkingStrategy.threshold,
|
|
253
|
+
}),
|
|
254
|
+
...(chunkingStrategy.prefixPaddingMs != null && {
|
|
255
|
+
prefix_padding_ms: chunkingStrategy.prefixPaddingMs,
|
|
256
|
+
}),
|
|
257
|
+
...(chunkingStrategy.silenceDurationMs != null && {
|
|
258
|
+
silence_duration_ms: chunkingStrategy.silenceDurationMs,
|
|
259
|
+
}),
|
|
260
|
+
}),
|
|
261
|
+
);
|
|
229
262
|
}
|
|
230
263
|
|
|
231
264
|
return {
|
|
@@ -270,6 +303,19 @@ export class OpenAITranscriptionModel implements TranscriptionModelV4 {
|
|
|
270
303
|
? languageMap[response.language as keyof typeof languageMap]
|
|
271
304
|
: undefined;
|
|
272
305
|
|
|
306
|
+
const diarizedSegments = response.segments?.flatMap(segment =>
|
|
307
|
+
'speaker' in segment
|
|
308
|
+
? [
|
|
309
|
+
{
|
|
310
|
+
text: segment.text,
|
|
311
|
+
startSecond: segment.start,
|
|
312
|
+
endSecond: segment.end,
|
|
313
|
+
speaker: segment.speaker,
|
|
314
|
+
},
|
|
315
|
+
]
|
|
316
|
+
: [],
|
|
317
|
+
);
|
|
318
|
+
|
|
273
319
|
return {
|
|
274
320
|
text: response.text,
|
|
275
321
|
segments:
|
|
@@ -293,6 +339,14 @@ export class OpenAITranscriptionModel implements TranscriptionModelV4 {
|
|
|
293
339
|
headers: responseHeaders,
|
|
294
340
|
body: rawResponse,
|
|
295
341
|
},
|
|
342
|
+
...(diarizedSegments != null &&
|
|
343
|
+
diarizedSegments.length > 0 && {
|
|
344
|
+
providerMetadata: {
|
|
345
|
+
openai: {
|
|
346
|
+
segments: diarizedSegments,
|
|
347
|
+
},
|
|
348
|
+
},
|
|
349
|
+
}),
|
|
296
350
|
};
|
|
297
351
|
}
|
|
298
352
|
|