@ai-sdk/azure 3.0.127 → 3.0.129

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -21,11 +21,17 @@ import {
21
21
  loadApiKey,
22
22
  loadSetting,
23
23
  normalizeHeaders,
24
+ parseProviderOptions,
24
25
  withoutTrailingSlash,
25
26
  withUserAgentSuffix,
26
27
  type FetchFunction,
27
28
  } from '@ai-sdk/provider-utils';
28
29
  import { azureOpenaiTools } from './azure-openai-tools';
30
+ import { AzureSpeechTranscriptionModel } from './azure-speech-transcription-model';
31
+ import {
32
+ azureTranscriptionModelOptions,
33
+ isMAITranscribe2,
34
+ } from './azure-transcription-model-options';
29
35
  import { VERSION } from './version';
30
36
 
31
37
  export interface AzureOpenAIProvider extends ProviderV3 {
@@ -87,10 +93,16 @@ export interface AzureOpenAIProvider extends ProviderV3 {
87
93
  imageModel(deploymentId: string): ImageModelV3;
88
94
 
89
95
  /**
90
- * Creates an Azure OpenAI model for audio transcription.
96
+ * Creates an Azure transcription model. MAI-Transcribe-2 uses the Speech API
97
+ * by default; other IDs use OpenAI. Override with providerOptions.azure.api.
91
98
  */
92
99
  transcription(deploymentId: string): TranscriptionModelV3;
93
100
 
101
+ /**
102
+ * Creates an Azure transcription model. Alias of `transcription`.
103
+ */
104
+ transcriptionModel(deploymentId: string): TranscriptionModelV3;
105
+
94
106
  /**
95
107
  * Creates an Azure OpenAI model for speech generation.
96
108
  */
@@ -155,6 +167,14 @@ export interface AzureOpenAIProviderSettings {
155
167
  * `{baseURL}/v1{path}?api-version={apiVersion}`.
156
168
  */
157
169
  useDeploymentBasedUrls?: boolean;
170
+
171
+ /**
172
+ * URL prefix for Azure Speech transcription (MAI-Transcribe-2), e.g. a
173
+ * regional endpoint like `https://eastus.api.cognitive.microsoft.com`.
174
+ * Defaults to `https://{resourceName}.cognitiveservices.azure.com`.
175
+ * Speech requests do not use `baseURL` or `apiVersion`.
176
+ */
177
+ speechBaseURL?: string;
158
178
  }
159
179
 
160
180
  function getAzureOpenAIBaseURLInfo(baseURL: string | undefined) {
@@ -199,15 +219,16 @@ export function createAzure(
199
219
  });
200
220
  }
201
221
 
202
- const getHeaders = () => {
222
+ const getHeaders = (api: 'openai' | 'speech' = 'openai') => {
203
223
  const authHeaders = tokenProvider
204
224
  ? {}
205
225
  : {
206
- 'api-key': loadApiKey({
207
- apiKey: options.apiKey,
208
- environmentVariableName: 'AZURE_API_KEY',
209
- description: 'Azure OpenAI',
210
- }),
226
+ [api === 'speech' ? 'Ocp-Apim-Subscription-Key' : 'api-key']:
227
+ loadApiKey({
228
+ apiKey: options.apiKey,
229
+ environmentVariableName: 'AZURE_API_KEY',
230
+ description: api === 'speech' ? 'Azure Speech' : 'Azure OpenAI',
231
+ }),
211
232
  };
212
233
 
213
234
  return withUserAgentSuffix(
@@ -332,12 +353,25 @@ export function createAzure(
332
353
  });
333
354
 
334
355
  const createTranscriptionModel = (modelId: string) =>
335
- new OpenAITranscriptionModel(modelId, {
336
- provider: 'azure.transcription',
337
- url,
338
- headers: getHeaders,
339
- fetch,
340
- });
356
+ new AzureTranscriptionModel(
357
+ modelId,
358
+ options,
359
+ new OpenAITranscriptionModel(modelId, {
360
+ provider: 'azure.transcription',
361
+ url,
362
+ headers: getHeaders,
363
+ fetch,
364
+ }),
365
+ new AzureSpeechTranscriptionModel(modelId, {
366
+ url: () =>
367
+ `${
368
+ withoutTrailingSlash(options.speechBaseURL) ??
369
+ `https://${getResourceName()}.cognitiveservices.azure.com`
370
+ }/speechtotext/transcriptions:transcribe?api-version=2025-10-15`,
371
+ headers: () => getHeaders('speech'),
372
+ fetch,
373
+ }),
374
+ );
341
375
 
342
376
  const createSpeechModel = (modelId: string) =>
343
377
  new OpenAISpeechModel(modelId, {
@@ -370,6 +404,7 @@ export function createAzure(
370
404
  provider.imageModel = createImageModel;
371
405
  provider.responses = createResponsesModel;
372
406
  provider.transcription = createTranscriptionModel;
407
+ provider.transcriptionModel = createTranscriptionModel;
373
408
  provider.speech = createSpeechModel;
374
409
  provider.tools = azureOpenaiTools;
375
410
  return provider;
@@ -379,3 +414,55 @@ export function createAzure(
379
414
  * Default Azure OpenAI provider instance.
380
415
  */
381
416
  export const azure = createAzure();
417
+
418
+ // Resolves the API per request: providerOptions also reach this model via Gateway.
419
+ class AzureTranscriptionModel implements TranscriptionModelV3 {
420
+ readonly specificationVersion = 'v3';
421
+ readonly provider = 'azure.transcription';
422
+
423
+ constructor(
424
+ readonly modelId: string,
425
+ private readonly config: AzureOpenAIProviderSettings,
426
+ private readonly openai: OpenAITranscriptionModel,
427
+ private readonly speech: AzureSpeechTranscriptionModel,
428
+ ) {}
429
+
430
+ async doGenerate(options: Parameters<TranscriptionModelV3['doGenerate']>[0]) {
431
+ const { api, ...speechOptions } = await this.getOptions(
432
+ options.providerOptions,
433
+ );
434
+ if (api === 'speech') {
435
+ return this.speech.doGenerate(options, speechOptions);
436
+ }
437
+
438
+ const result = await this.openai.doGenerate(options);
439
+ return {
440
+ ...result,
441
+ warnings: [
442
+ ...result.warnings,
443
+ ...Object.keys(speechOptions).map(key => ({
444
+ type: 'unsupported' as const,
445
+ feature: `providerOptions.azure.${key}`,
446
+ details: 'This option requires the Azure Speech API.',
447
+ })),
448
+ ],
449
+ };
450
+ }
451
+
452
+ private async getOptions(
453
+ providerOptions: Parameters<
454
+ TranscriptionModelV3['doGenerate']
455
+ >[0]['providerOptions'],
456
+ ) {
457
+ const options = await parseProviderOptions({
458
+ provider: 'azure',
459
+ providerOptions,
460
+ schema: azureTranscriptionModelOptions,
461
+ });
462
+ return {
463
+ ...options,
464
+ api:
465
+ options?.api ?? (isMAITranscribe2(this.modelId) ? 'speech' : 'openai'),
466
+ };
467
+ }
468
+ }
@@ -0,0 +1,46 @@
1
+ import {
2
+ lazySchema,
3
+ zodSchema,
4
+ type InferSchema,
5
+ } from '@ai-sdk/provider-utils';
6
+ import { z } from 'zod/v4';
7
+
8
+ // Strict objects reject unknown keys (e.g. `diarization.maxSpeakers`, which
9
+ // MAI-Transcribe-2 does not support) instead of silently dropping them.
10
+ export const azureSpeechTranscriptionModelOptionsShape = () => ({
11
+ /**
12
+ * Timing granularity. Defaults to `segment` so that transcription results
13
+ * include timed segments (Azure's own default is `none`).
14
+ */
15
+ timestamps: z.enum(['word', 'segment', 'none']).optional(),
16
+
17
+ /**
18
+ * `verbatim` keeps fillers and false starts, `clean` removes them.
19
+ * Azure defaults to `verbatim`.
20
+ */
21
+ transcribeStyle: z.enum(['verbatim', 'clean']).optional(),
22
+
23
+ /**
24
+ * Forces a single language, e.g. `['en']`. Omit for automatic language
25
+ * detection and code switching.
26
+ */
27
+ locales: z.array(z.string()).length(1).optional(),
28
+
29
+ /**
30
+ * Speaker diarization. Speaker IDs are available in provider metadata.
31
+ */
32
+ diarization: z.strictObject({ enabled: z.boolean() }).optional(),
33
+
34
+ /**
35
+ * Keyword biasing for names and domain terminology.
36
+ */
37
+ phraseList: z.strictObject({ phrases: z.array(z.string()) }).optional(),
38
+ });
39
+
40
+ export const azureSpeechTranscriptionModelOptions = lazySchema(() =>
41
+ zodSchema(z.strictObject(azureSpeechTranscriptionModelOptionsShape())),
42
+ );
43
+
44
+ export type AzureTranscriptionModelSpeechOptions = InferSchema<
45
+ typeof azureSpeechTranscriptionModelOptions
46
+ >;
@@ -0,0 +1,155 @@
1
+ import type { TranscriptionModelV3 } from '@ai-sdk/provider';
2
+ import {
3
+ combineHeaders,
4
+ convertBase64ToUint8Array,
5
+ createJsonErrorResponseHandler,
6
+ createJsonResponseHandler,
7
+ mediaTypeToExtension,
8
+ postFormDataToApi,
9
+ type FetchFunction,
10
+ } from '@ai-sdk/provider-utils';
11
+ import { z } from 'zod/v4';
12
+ import type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';
13
+ import type { AzureTranscriptionModelSpeechOptions } from './azure-speech-transcription-model-options';
14
+ import { isMAITranscribe2 } from './azure-transcription-model-options';
15
+
16
+ export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
17
+ readonly specificationVersion = 'v3';
18
+ readonly provider = 'azure.transcription';
19
+
20
+ constructor(
21
+ readonly modelId: string,
22
+ private readonly config: {
23
+ url: () => string;
24
+ headers: () => Record<string, string | undefined>;
25
+ fetch?: FetchFunction;
26
+ },
27
+ ) {}
28
+
29
+ async doGenerate(
30
+ options: Parameters<TranscriptionModelV3['doGenerate']>[0],
31
+ azureOptions: AzureTranscriptionModelSpeechOptions = {},
32
+ ): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>> {
33
+ const timestamp = new Date();
34
+ const formData = new FormData();
35
+ formData.append(
36
+ 'audio',
37
+ new Blob(
38
+ [
39
+ typeof options.audio === 'string'
40
+ ? convertBase64ToUint8Array(options.audio)
41
+ : options.audio,
42
+ ],
43
+ { type: options.mediaType },
44
+ ),
45
+ `audio.${mediaTypeToExtension(options.mediaType)}`,
46
+ );
47
+ formData.append(
48
+ 'definition',
49
+ JSON.stringify({
50
+ enhancedMode: {
51
+ enabled: true,
52
+ model: isMAITranscribe2(this.modelId)
53
+ ? 'MAI-Transcribe-2'
54
+ : this.modelId,
55
+ modelOptions: {
56
+ timestamps: azureOptions.timestamps ?? 'segment',
57
+ transcribeStyle: azureOptions.transcribeStyle,
58
+ },
59
+ },
60
+ locales: azureOptions.locales,
61
+ diarization: azureOptions.diarization,
62
+ phraseList: azureOptions.phraseList,
63
+ }),
64
+ );
65
+
66
+ const { value, rawValue, responseHeaders } = await postFormDataToApi({
67
+ url: this.config.url(),
68
+ headers: combineHeaders(this.config.headers(), options.headers),
69
+ formData,
70
+ abortSignal: options.abortSignal,
71
+ fetch: this.config.fetch,
72
+ failedResponseHandler: createJsonErrorResponseHandler({
73
+ errorSchema: z.union([
74
+ z.object({ error: z.object({ message: z.string() }) }),
75
+ z.object({ message: z.string() }),
76
+ ]),
77
+ errorToMessage: data =>
78
+ 'error' in data ? data.error.message : data.message,
79
+ }),
80
+ successfulResponseHandler: createJsonResponseHandler(responseSchema),
81
+ });
82
+
83
+ const phrases = value.phrases ?? [];
84
+ // Consider every reported locale so mixed results (e.g. en + fil) are not
85
+ // collapsed to one language; only ISO-639-1 codes are reported.
86
+ const languages = new Set(
87
+ phrases.flatMap(phrase =>
88
+ phrase.locale ? [phrase.locale.split('-')[0].toLowerCase()] : [],
89
+ ),
90
+ );
91
+ const [language] = languages;
92
+ const reportedLanguage =
93
+ languages.size === 1 && /^[a-z]{2}$/.test(language)
94
+ ? language
95
+ : undefined;
96
+
97
+ return {
98
+ text: value.combinedPhrases.map(phrase => phrase.text).join(' '),
99
+ segments: phrases.flatMap(phrase =>
100
+ phrase.offsetMilliseconds != null && phrase.durationMilliseconds != null
101
+ ? [
102
+ {
103
+ text: phrase.text,
104
+ startSecond: phrase.offsetMilliseconds / 1000,
105
+ endSecond:
106
+ (phrase.offsetMilliseconds + phrase.durationMilliseconds) /
107
+ 1000,
108
+ },
109
+ ]
110
+ : [],
111
+ ),
112
+ language: reportedLanguage,
113
+ durationInSeconds:
114
+ value.durationMilliseconds != null
115
+ ? value.durationMilliseconds / 1000
116
+ : undefined,
117
+ warnings: [],
118
+ providerMetadata: {
119
+ azure: { phrases },
120
+ } satisfies AzureTranscriptionProviderMetadata,
121
+ response: {
122
+ timestamp,
123
+ modelId: this.modelId,
124
+ headers: responseHeaders,
125
+ body: rawValue,
126
+ },
127
+ };
128
+ }
129
+ }
130
+
131
+ const responseSchema = z.object({
132
+ combinedPhrases: z.array(z.object({ text: z.string() })),
133
+ durationMilliseconds: z.number().nullish(),
134
+ phrases: z
135
+ .array(
136
+ z.object({
137
+ text: z.string(),
138
+ offsetMilliseconds: z.number().nullish(),
139
+ durationMilliseconds: z.number().nullish(),
140
+ locale: z.string().nullish(),
141
+ speaker: z.number().nullish(),
142
+ confidence: z.number().nullish(),
143
+ words: z
144
+ .array(
145
+ z.object({
146
+ text: z.string(),
147
+ offsetMilliseconds: z.number().nullish(),
148
+ durationMilliseconds: z.number().nullish(),
149
+ }),
150
+ )
151
+ .nullish(),
152
+ }),
153
+ )
154
+ .nullish(),
155
+ });
@@ -0,0 +1,27 @@
1
+ import {
2
+ lazySchema,
3
+ zodSchema,
4
+ type InferSchema,
5
+ } from '@ai-sdk/provider-utils';
6
+ import { z } from 'zod/v4';
7
+ import { azureSpeechTranscriptionModelOptionsShape } from './azure-speech-transcription-model-options';
8
+
9
+ export const azureTranscriptionModelOptions = lazySchema(() =>
10
+ zodSchema(
11
+ z.strictObject({
12
+ /**
13
+ * API to use. Defaults to Speech for MAI-Transcribe-2, OpenAI otherwise.
14
+ */
15
+ api: z.enum(['openai', 'speech']).optional(),
16
+ ...azureSpeechTranscriptionModelOptionsShape(),
17
+ }),
18
+ ),
19
+ );
20
+
21
+ export type AzureTranscriptionModelOptions = InferSchema<
22
+ typeof azureTranscriptionModelOptions
23
+ >;
24
+
25
+ export function isMAITranscribe2(modelId: string): boolean {
26
+ return modelId.toLowerCase() === 'mai-transcribe-2';
27
+ }
@@ -0,0 +1,17 @@
1
+ export type AzureTranscriptionProviderMetadata = {
2
+ azure: {
3
+ phrases: Array<{
4
+ text: string;
5
+ offsetMilliseconds?: number | null;
6
+ durationMilliseconds?: number | null;
7
+ locale?: string | null;
8
+ speaker?: number | null;
9
+ confidence?: number | null;
10
+ words?: Array<{
11
+ text: string;
12
+ offsetMilliseconds?: number | null;
13
+ durationMilliseconds?: number | null;
14
+ }> | null;
15
+ }>;
16
+ };
17
+ };
package/src/index.ts CHANGED
@@ -29,3 +29,5 @@ export type {
29
29
  AzureResponsesSourceDocumentProviderMetadata,
30
30
  } from './azure-openai-provider-metadata';
31
31
  export { VERSION } from './version';
32
+ export type { AzureTranscriptionModelOptions } from './azure-transcription-model-options';
33
+ export type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';