@ai-sdk/azure 3.0.127 → 3.0.129
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/dist/index.d.mts +48 -3
- package/dist/index.d.ts +48 -3
- package/dist/index.js +235 -15
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +244 -10
- package/dist/index.mjs.map +1 -1
- package/docs/04-azure.mdx +108 -1
- package/package.json +4 -4
- package/src/azure-openai-provider.ts +100 -13
- package/src/azure-speech-transcription-model-options.ts +46 -0
- package/src/azure-speech-transcription-model.ts +155 -0
- package/src/azure-transcription-model-options.ts +27 -0
- package/src/azure-transcription-provider-metadata.ts +17 -0
- package/src/index.ts +2 -0
|
@@ -21,11 +21,17 @@ import {
|
|
|
21
21
|
loadApiKey,
|
|
22
22
|
loadSetting,
|
|
23
23
|
normalizeHeaders,
|
|
24
|
+
parseProviderOptions,
|
|
24
25
|
withoutTrailingSlash,
|
|
25
26
|
withUserAgentSuffix,
|
|
26
27
|
type FetchFunction,
|
|
27
28
|
} from '@ai-sdk/provider-utils';
|
|
28
29
|
import { azureOpenaiTools } from './azure-openai-tools';
|
|
30
|
+
import { AzureSpeechTranscriptionModel } from './azure-speech-transcription-model';
|
|
31
|
+
import {
|
|
32
|
+
azureTranscriptionModelOptions,
|
|
33
|
+
isMAITranscribe2,
|
|
34
|
+
} from './azure-transcription-model-options';
|
|
29
35
|
import { VERSION } from './version';
|
|
30
36
|
|
|
31
37
|
export interface AzureOpenAIProvider extends ProviderV3 {
|
|
@@ -87,10 +93,16 @@ export interface AzureOpenAIProvider extends ProviderV3 {
|
|
|
87
93
|
imageModel(deploymentId: string): ImageModelV3;
|
|
88
94
|
|
|
89
95
|
/**
|
|
90
|
-
* Creates an Azure
|
|
96
|
+
* Creates an Azure transcription model. MAI-Transcribe-2 uses the Speech API
|
|
97
|
+
* by default; other IDs use OpenAI. Override with providerOptions.azure.api.
|
|
91
98
|
*/
|
|
92
99
|
transcription(deploymentId: string): TranscriptionModelV3;
|
|
93
100
|
|
|
101
|
+
/**
|
|
102
|
+
* Creates an Azure transcription model. Alias of `transcription`.
|
|
103
|
+
*/
|
|
104
|
+
transcriptionModel(deploymentId: string): TranscriptionModelV3;
|
|
105
|
+
|
|
94
106
|
/**
|
|
95
107
|
* Creates an Azure OpenAI model for speech generation.
|
|
96
108
|
*/
|
|
@@ -155,6 +167,14 @@ export interface AzureOpenAIProviderSettings {
|
|
|
155
167
|
* `{baseURL}/v1{path}?api-version={apiVersion}`.
|
|
156
168
|
*/
|
|
157
169
|
useDeploymentBasedUrls?: boolean;
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* URL prefix for Azure Speech transcription (MAI-Transcribe-2), e.g. a
|
|
173
|
+
* regional endpoint like `https://eastus.api.cognitive.microsoft.com`.
|
|
174
|
+
* Defaults to `https://{resourceName}.cognitiveservices.azure.com`.
|
|
175
|
+
* Speech requests do not use `baseURL` or `apiVersion`.
|
|
176
|
+
*/
|
|
177
|
+
speechBaseURL?: string;
|
|
158
178
|
}
|
|
159
179
|
|
|
160
180
|
function getAzureOpenAIBaseURLInfo(baseURL: string | undefined) {
|
|
@@ -199,15 +219,16 @@ export function createAzure(
|
|
|
199
219
|
});
|
|
200
220
|
}
|
|
201
221
|
|
|
202
|
-
const getHeaders = () => {
|
|
222
|
+
const getHeaders = (api: 'openai' | 'speech' = 'openai') => {
|
|
203
223
|
const authHeaders = tokenProvider
|
|
204
224
|
? {}
|
|
205
225
|
: {
|
|
206
|
-
'api-key':
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
226
|
+
[api === 'speech' ? 'Ocp-Apim-Subscription-Key' : 'api-key']:
|
|
227
|
+
loadApiKey({
|
|
228
|
+
apiKey: options.apiKey,
|
|
229
|
+
environmentVariableName: 'AZURE_API_KEY',
|
|
230
|
+
description: api === 'speech' ? 'Azure Speech' : 'Azure OpenAI',
|
|
231
|
+
}),
|
|
211
232
|
};
|
|
212
233
|
|
|
213
234
|
return withUserAgentSuffix(
|
|
@@ -332,12 +353,25 @@ export function createAzure(
|
|
|
332
353
|
});
|
|
333
354
|
|
|
334
355
|
const createTranscriptionModel = (modelId: string) =>
|
|
335
|
-
new
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
356
|
+
new AzureTranscriptionModel(
|
|
357
|
+
modelId,
|
|
358
|
+
options,
|
|
359
|
+
new OpenAITranscriptionModel(modelId, {
|
|
360
|
+
provider: 'azure.transcription',
|
|
361
|
+
url,
|
|
362
|
+
headers: getHeaders,
|
|
363
|
+
fetch,
|
|
364
|
+
}),
|
|
365
|
+
new AzureSpeechTranscriptionModel(modelId, {
|
|
366
|
+
url: () =>
|
|
367
|
+
`${
|
|
368
|
+
withoutTrailingSlash(options.speechBaseURL) ??
|
|
369
|
+
`https://${getResourceName()}.cognitiveservices.azure.com`
|
|
370
|
+
}/speechtotext/transcriptions:transcribe?api-version=2025-10-15`,
|
|
371
|
+
headers: () => getHeaders('speech'),
|
|
372
|
+
fetch,
|
|
373
|
+
}),
|
|
374
|
+
);
|
|
341
375
|
|
|
342
376
|
const createSpeechModel = (modelId: string) =>
|
|
343
377
|
new OpenAISpeechModel(modelId, {
|
|
@@ -370,6 +404,7 @@ export function createAzure(
|
|
|
370
404
|
provider.imageModel = createImageModel;
|
|
371
405
|
provider.responses = createResponsesModel;
|
|
372
406
|
provider.transcription = createTranscriptionModel;
|
|
407
|
+
provider.transcriptionModel = createTranscriptionModel;
|
|
373
408
|
provider.speech = createSpeechModel;
|
|
374
409
|
provider.tools = azureOpenaiTools;
|
|
375
410
|
return provider;
|
|
@@ -379,3 +414,55 @@ export function createAzure(
|
|
|
379
414
|
* Default Azure OpenAI provider instance.
|
|
380
415
|
*/
|
|
381
416
|
export const azure = createAzure();
|
|
417
|
+
|
|
418
|
+
// Resolves the API per request: providerOptions also reach this model via Gateway.
|
|
419
|
+
class AzureTranscriptionModel implements TranscriptionModelV3 {
|
|
420
|
+
readonly specificationVersion = 'v3';
|
|
421
|
+
readonly provider = 'azure.transcription';
|
|
422
|
+
|
|
423
|
+
constructor(
|
|
424
|
+
readonly modelId: string,
|
|
425
|
+
private readonly config: AzureOpenAIProviderSettings,
|
|
426
|
+
private readonly openai: OpenAITranscriptionModel,
|
|
427
|
+
private readonly speech: AzureSpeechTranscriptionModel,
|
|
428
|
+
) {}
|
|
429
|
+
|
|
430
|
+
async doGenerate(options: Parameters<TranscriptionModelV3['doGenerate']>[0]) {
|
|
431
|
+
const { api, ...speechOptions } = await this.getOptions(
|
|
432
|
+
options.providerOptions,
|
|
433
|
+
);
|
|
434
|
+
if (api === 'speech') {
|
|
435
|
+
return this.speech.doGenerate(options, speechOptions);
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
const result = await this.openai.doGenerate(options);
|
|
439
|
+
return {
|
|
440
|
+
...result,
|
|
441
|
+
warnings: [
|
|
442
|
+
...result.warnings,
|
|
443
|
+
...Object.keys(speechOptions).map(key => ({
|
|
444
|
+
type: 'unsupported' as const,
|
|
445
|
+
feature: `providerOptions.azure.${key}`,
|
|
446
|
+
details: 'This option requires the Azure Speech API.',
|
|
447
|
+
})),
|
|
448
|
+
],
|
|
449
|
+
};
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
private async getOptions(
|
|
453
|
+
providerOptions: Parameters<
|
|
454
|
+
TranscriptionModelV3['doGenerate']
|
|
455
|
+
>[0]['providerOptions'],
|
|
456
|
+
) {
|
|
457
|
+
const options = await parseProviderOptions({
|
|
458
|
+
provider: 'azure',
|
|
459
|
+
providerOptions,
|
|
460
|
+
schema: azureTranscriptionModelOptions,
|
|
461
|
+
});
|
|
462
|
+
return {
|
|
463
|
+
...options,
|
|
464
|
+
api:
|
|
465
|
+
options?.api ?? (isMAITranscribe2(this.modelId) ? 'speech' : 'openai'),
|
|
466
|
+
};
|
|
467
|
+
}
|
|
468
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import {
|
|
2
|
+
lazySchema,
|
|
3
|
+
zodSchema,
|
|
4
|
+
type InferSchema,
|
|
5
|
+
} from '@ai-sdk/provider-utils';
|
|
6
|
+
import { z } from 'zod/v4';
|
|
7
|
+
|
|
8
|
+
// Strict objects reject unknown keys (e.g. `diarization.maxSpeakers`, which
|
|
9
|
+
// MAI-Transcribe-2 does not support) instead of silently dropping them.
|
|
10
|
+
export const azureSpeechTranscriptionModelOptionsShape = () => ({
|
|
11
|
+
/**
|
|
12
|
+
* Timing granularity. Defaults to `segment` so that transcription results
|
|
13
|
+
* include timed segments (Azure's own default is `none`).
|
|
14
|
+
*/
|
|
15
|
+
timestamps: z.enum(['word', 'segment', 'none']).optional(),
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* `verbatim` keeps fillers and false starts, `clean` removes them.
|
|
19
|
+
* Azure defaults to `verbatim`.
|
|
20
|
+
*/
|
|
21
|
+
transcribeStyle: z.enum(['verbatim', 'clean']).optional(),
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Forces a single language, e.g. `['en']`. Omit for automatic language
|
|
25
|
+
* detection and code switching.
|
|
26
|
+
*/
|
|
27
|
+
locales: z.array(z.string()).length(1).optional(),
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Speaker diarization. Speaker IDs are available in provider metadata.
|
|
31
|
+
*/
|
|
32
|
+
diarization: z.strictObject({ enabled: z.boolean() }).optional(),
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Keyword biasing for names and domain terminology.
|
|
36
|
+
*/
|
|
37
|
+
phraseList: z.strictObject({ phrases: z.array(z.string()) }).optional(),
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
export const azureSpeechTranscriptionModelOptions = lazySchema(() =>
|
|
41
|
+
zodSchema(z.strictObject(azureSpeechTranscriptionModelOptionsShape())),
|
|
42
|
+
);
|
|
43
|
+
|
|
44
|
+
export type AzureTranscriptionModelSpeechOptions = InferSchema<
|
|
45
|
+
typeof azureSpeechTranscriptionModelOptions
|
|
46
|
+
>;
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
import type { TranscriptionModelV3 } from '@ai-sdk/provider';
|
|
2
|
+
import {
|
|
3
|
+
combineHeaders,
|
|
4
|
+
convertBase64ToUint8Array,
|
|
5
|
+
createJsonErrorResponseHandler,
|
|
6
|
+
createJsonResponseHandler,
|
|
7
|
+
mediaTypeToExtension,
|
|
8
|
+
postFormDataToApi,
|
|
9
|
+
type FetchFunction,
|
|
10
|
+
} from '@ai-sdk/provider-utils';
|
|
11
|
+
import { z } from 'zod/v4';
|
|
12
|
+
import type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';
|
|
13
|
+
import type { AzureTranscriptionModelSpeechOptions } from './azure-speech-transcription-model-options';
|
|
14
|
+
import { isMAITranscribe2 } from './azure-transcription-model-options';
|
|
15
|
+
|
|
16
|
+
export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
|
|
17
|
+
readonly specificationVersion = 'v3';
|
|
18
|
+
readonly provider = 'azure.transcription';
|
|
19
|
+
|
|
20
|
+
constructor(
|
|
21
|
+
readonly modelId: string,
|
|
22
|
+
private readonly config: {
|
|
23
|
+
url: () => string;
|
|
24
|
+
headers: () => Record<string, string | undefined>;
|
|
25
|
+
fetch?: FetchFunction;
|
|
26
|
+
},
|
|
27
|
+
) {}
|
|
28
|
+
|
|
29
|
+
async doGenerate(
|
|
30
|
+
options: Parameters<TranscriptionModelV3['doGenerate']>[0],
|
|
31
|
+
azureOptions: AzureTranscriptionModelSpeechOptions = {},
|
|
32
|
+
): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>> {
|
|
33
|
+
const timestamp = new Date();
|
|
34
|
+
const formData = new FormData();
|
|
35
|
+
formData.append(
|
|
36
|
+
'audio',
|
|
37
|
+
new Blob(
|
|
38
|
+
[
|
|
39
|
+
typeof options.audio === 'string'
|
|
40
|
+
? convertBase64ToUint8Array(options.audio)
|
|
41
|
+
: options.audio,
|
|
42
|
+
],
|
|
43
|
+
{ type: options.mediaType },
|
|
44
|
+
),
|
|
45
|
+
`audio.${mediaTypeToExtension(options.mediaType)}`,
|
|
46
|
+
);
|
|
47
|
+
formData.append(
|
|
48
|
+
'definition',
|
|
49
|
+
JSON.stringify({
|
|
50
|
+
enhancedMode: {
|
|
51
|
+
enabled: true,
|
|
52
|
+
model: isMAITranscribe2(this.modelId)
|
|
53
|
+
? 'MAI-Transcribe-2'
|
|
54
|
+
: this.modelId,
|
|
55
|
+
modelOptions: {
|
|
56
|
+
timestamps: azureOptions.timestamps ?? 'segment',
|
|
57
|
+
transcribeStyle: azureOptions.transcribeStyle,
|
|
58
|
+
},
|
|
59
|
+
},
|
|
60
|
+
locales: azureOptions.locales,
|
|
61
|
+
diarization: azureOptions.diarization,
|
|
62
|
+
phraseList: azureOptions.phraseList,
|
|
63
|
+
}),
|
|
64
|
+
);
|
|
65
|
+
|
|
66
|
+
const { value, rawValue, responseHeaders } = await postFormDataToApi({
|
|
67
|
+
url: this.config.url(),
|
|
68
|
+
headers: combineHeaders(this.config.headers(), options.headers),
|
|
69
|
+
formData,
|
|
70
|
+
abortSignal: options.abortSignal,
|
|
71
|
+
fetch: this.config.fetch,
|
|
72
|
+
failedResponseHandler: createJsonErrorResponseHandler({
|
|
73
|
+
errorSchema: z.union([
|
|
74
|
+
z.object({ error: z.object({ message: z.string() }) }),
|
|
75
|
+
z.object({ message: z.string() }),
|
|
76
|
+
]),
|
|
77
|
+
errorToMessage: data =>
|
|
78
|
+
'error' in data ? data.error.message : data.message,
|
|
79
|
+
}),
|
|
80
|
+
successfulResponseHandler: createJsonResponseHandler(responseSchema),
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
const phrases = value.phrases ?? [];
|
|
84
|
+
// Consider every reported locale so mixed results (e.g. en + fil) are not
|
|
85
|
+
// collapsed to one language; only ISO-639-1 codes are reported.
|
|
86
|
+
const languages = new Set(
|
|
87
|
+
phrases.flatMap(phrase =>
|
|
88
|
+
phrase.locale ? [phrase.locale.split('-')[0].toLowerCase()] : [],
|
|
89
|
+
),
|
|
90
|
+
);
|
|
91
|
+
const [language] = languages;
|
|
92
|
+
const reportedLanguage =
|
|
93
|
+
languages.size === 1 && /^[a-z]{2}$/.test(language)
|
|
94
|
+
? language
|
|
95
|
+
: undefined;
|
|
96
|
+
|
|
97
|
+
return {
|
|
98
|
+
text: value.combinedPhrases.map(phrase => phrase.text).join(' '),
|
|
99
|
+
segments: phrases.flatMap(phrase =>
|
|
100
|
+
phrase.offsetMilliseconds != null && phrase.durationMilliseconds != null
|
|
101
|
+
? [
|
|
102
|
+
{
|
|
103
|
+
text: phrase.text,
|
|
104
|
+
startSecond: phrase.offsetMilliseconds / 1000,
|
|
105
|
+
endSecond:
|
|
106
|
+
(phrase.offsetMilliseconds + phrase.durationMilliseconds) /
|
|
107
|
+
1000,
|
|
108
|
+
},
|
|
109
|
+
]
|
|
110
|
+
: [],
|
|
111
|
+
),
|
|
112
|
+
language: reportedLanguage,
|
|
113
|
+
durationInSeconds:
|
|
114
|
+
value.durationMilliseconds != null
|
|
115
|
+
? value.durationMilliseconds / 1000
|
|
116
|
+
: undefined,
|
|
117
|
+
warnings: [],
|
|
118
|
+
providerMetadata: {
|
|
119
|
+
azure: { phrases },
|
|
120
|
+
} satisfies AzureTranscriptionProviderMetadata,
|
|
121
|
+
response: {
|
|
122
|
+
timestamp,
|
|
123
|
+
modelId: this.modelId,
|
|
124
|
+
headers: responseHeaders,
|
|
125
|
+
body: rawValue,
|
|
126
|
+
},
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
const responseSchema = z.object({
|
|
132
|
+
combinedPhrases: z.array(z.object({ text: z.string() })),
|
|
133
|
+
durationMilliseconds: z.number().nullish(),
|
|
134
|
+
phrases: z
|
|
135
|
+
.array(
|
|
136
|
+
z.object({
|
|
137
|
+
text: z.string(),
|
|
138
|
+
offsetMilliseconds: z.number().nullish(),
|
|
139
|
+
durationMilliseconds: z.number().nullish(),
|
|
140
|
+
locale: z.string().nullish(),
|
|
141
|
+
speaker: z.number().nullish(),
|
|
142
|
+
confidence: z.number().nullish(),
|
|
143
|
+
words: z
|
|
144
|
+
.array(
|
|
145
|
+
z.object({
|
|
146
|
+
text: z.string(),
|
|
147
|
+
offsetMilliseconds: z.number().nullish(),
|
|
148
|
+
durationMilliseconds: z.number().nullish(),
|
|
149
|
+
}),
|
|
150
|
+
)
|
|
151
|
+
.nullish(),
|
|
152
|
+
}),
|
|
153
|
+
)
|
|
154
|
+
.nullish(),
|
|
155
|
+
});
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import {
|
|
2
|
+
lazySchema,
|
|
3
|
+
zodSchema,
|
|
4
|
+
type InferSchema,
|
|
5
|
+
} from '@ai-sdk/provider-utils';
|
|
6
|
+
import { z } from 'zod/v4';
|
|
7
|
+
import { azureSpeechTranscriptionModelOptionsShape } from './azure-speech-transcription-model-options';
|
|
8
|
+
|
|
9
|
+
export const azureTranscriptionModelOptions = lazySchema(() =>
|
|
10
|
+
zodSchema(
|
|
11
|
+
z.strictObject({
|
|
12
|
+
/**
|
|
13
|
+
* API to use. Defaults to Speech for MAI-Transcribe-2, OpenAI otherwise.
|
|
14
|
+
*/
|
|
15
|
+
api: z.enum(['openai', 'speech']).optional(),
|
|
16
|
+
...azureSpeechTranscriptionModelOptionsShape(),
|
|
17
|
+
}),
|
|
18
|
+
),
|
|
19
|
+
);
|
|
20
|
+
|
|
21
|
+
export type AzureTranscriptionModelOptions = InferSchema<
|
|
22
|
+
typeof azureTranscriptionModelOptions
|
|
23
|
+
>;
|
|
24
|
+
|
|
25
|
+
export function isMAITranscribe2(modelId: string): boolean {
|
|
26
|
+
return modelId.toLowerCase() === 'mai-transcribe-2';
|
|
27
|
+
}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
export type AzureTranscriptionProviderMetadata = {
|
|
2
|
+
azure: {
|
|
3
|
+
phrases: Array<{
|
|
4
|
+
text: string;
|
|
5
|
+
offsetMilliseconds?: number | null;
|
|
6
|
+
durationMilliseconds?: number | null;
|
|
7
|
+
locale?: string | null;
|
|
8
|
+
speaker?: number | null;
|
|
9
|
+
confidence?: number | null;
|
|
10
|
+
words?: Array<{
|
|
11
|
+
text: string;
|
|
12
|
+
offsetMilliseconds?: number | null;
|
|
13
|
+
durationMilliseconds?: number | null;
|
|
14
|
+
}> | null;
|
|
15
|
+
}>;
|
|
16
|
+
};
|
|
17
|
+
};
|
package/src/index.ts
CHANGED
|
@@ -29,3 +29,5 @@ export type {
|
|
|
29
29
|
AzureResponsesSourceDocumentProviderMetadata,
|
|
30
30
|
} from './azure-openai-provider-metadata';
|
|
31
31
|
export { VERSION } from './version';
|
|
32
|
+
export type { AzureTranscriptionModelOptions } from './azure-transcription-model-options';
|
|
33
|
+
export type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';
|