@ai-sdk/azure 3.0.130 → 3.0.131
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +8 -0
- package/dist/index.d.mts +20 -6
- package/dist/index.d.ts +20 -6
- package/dist/index.js +380 -74
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +384 -64
- package/dist/index.mjs.map +1 -1
- package/docs/04-azure.mdx +104 -18
- package/package.json +1 -1
- package/src/azure-openai-provider.ts +101 -19
- package/src/azure-speech-model-options.ts +33 -0
- package/src/azure-speech-speech-model-options.ts +27 -0
- package/src/azure-speech-speech-model.ts +282 -0
- package/src/azure-speech-transcription-model.ts +6 -5
- package/src/azure-transcription-model-options.ts +13 -3
- package/src/index.ts +1 -0
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import {
|
|
2
|
+
lazySchema,
|
|
3
|
+
zodSchema,
|
|
4
|
+
type InferSchema,
|
|
5
|
+
} from '@ai-sdk/provider-utils';
|
|
6
|
+
import { z } from 'zod/v4';
|
|
7
|
+
import { azureSpeechSpeechModelOptionsShape } from './azure-speech-speech-model-options';
|
|
8
|
+
|
|
9
|
+
export const azureSpeechModelOptions = lazySchema(() =>
|
|
10
|
+
zodSchema(
|
|
11
|
+
z.strictObject({
|
|
12
|
+
/**
|
|
13
|
+
* API to use. Defaults to Speech for MAI-Voice models, OpenAI otherwise.
|
|
14
|
+
*/
|
|
15
|
+
api: z.enum(['openai', 'speech']).optional(),
|
|
16
|
+
...azureSpeechSpeechModelOptionsShape(),
|
|
17
|
+
}),
|
|
18
|
+
),
|
|
19
|
+
);
|
|
20
|
+
|
|
21
|
+
export type AzureSpeechModelOptions = InferSchema<
|
|
22
|
+
typeof azureSpeechModelOptions
|
|
23
|
+
>;
|
|
24
|
+
|
|
25
|
+
// Azure voice-name suffixes by lowercase model ID.
|
|
26
|
+
const maiVoiceModels = new Map([
|
|
27
|
+
['mai-voice-2-flash', 'MAI-Voice-2-Flash'],
|
|
28
|
+
['mai-voice-2', 'MAI-Voice-2'],
|
|
29
|
+
]);
|
|
30
|
+
|
|
31
|
+
export function getMAIVoiceModel(modelId: string) {
|
|
32
|
+
return maiVoiceModels.get(modelId.toLowerCase());
|
|
33
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import {
|
|
2
|
+
lazySchema,
|
|
3
|
+
zodSchema,
|
|
4
|
+
type InferSchema,
|
|
5
|
+
} from '@ai-sdk/provider-utils';
|
|
6
|
+
import { z } from 'zod/v4';
|
|
7
|
+
|
|
8
|
+
export const azureSpeechSpeechModelOptionsShape = () => ({
|
|
9
|
+
/**
|
|
10
|
+
* Speaking style applied with `mstts:express-as`, e.g. `excited` or
|
|
11
|
+
* `whispering`. Supported styles vary by voice.
|
|
12
|
+
*/
|
|
13
|
+
style: z.string().min(1).optional(),
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Intensity of `style`, from 0.01 to 2. Azure defaults to 1.
|
|
17
|
+
*/
|
|
18
|
+
styleDegree: z.number().min(0.01).max(2).optional(),
|
|
19
|
+
});
|
|
20
|
+
|
|
21
|
+
export const azureSpeechSpeechModelOptions = lazySchema(() =>
|
|
22
|
+
zodSchema(z.strictObject(azureSpeechSpeechModelOptionsShape())),
|
|
23
|
+
);
|
|
24
|
+
|
|
25
|
+
export type AzureSpeechModelSpeechOptions = InferSchema<
|
|
26
|
+
typeof azureSpeechSpeechModelOptions
|
|
27
|
+
>;
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
import {
|
|
2
|
+
APICallError,
|
|
3
|
+
type SharedV3Warning,
|
|
4
|
+
type SpeechModelV3,
|
|
5
|
+
} from '@ai-sdk/provider';
|
|
6
|
+
import {
|
|
7
|
+
combineHeaders,
|
|
8
|
+
createBinaryResponseHandler,
|
|
9
|
+
extractResponseHeaders,
|
|
10
|
+
postToApi,
|
|
11
|
+
safeParseJSON,
|
|
12
|
+
type FetchFunction,
|
|
13
|
+
type ResponseHandler,
|
|
14
|
+
} from '@ai-sdk/provider-utils';
|
|
15
|
+
import { z } from 'zod/v4';
|
|
16
|
+
import { getMAIVoiceModel } from './azure-speech-model-options';
|
|
17
|
+
import type { AzureSpeechModelSpeechOptions } from './azure-speech-speech-model-options';
|
|
18
|
+
|
|
19
|
+
const DEFAULT_VOICE = 'en-US-Harper';
|
|
20
|
+
|
|
21
|
+
// Default voice per ISO 639-1 language; available on MAI-Voice-2 and
|
|
22
|
+
// MAI-Voice-2-Flash.
|
|
23
|
+
const DEFAULT_VOICES = new Map([
|
|
24
|
+
['de', 'de-DE-Mia'],
|
|
25
|
+
['en', DEFAULT_VOICE],
|
|
26
|
+
['es', 'es-MX-Valeria'],
|
|
27
|
+
['fr', 'fr-FR-Soleil'],
|
|
28
|
+
['hi', 'hi-IN-Kavya'],
|
|
29
|
+
['hu', 'hu-HU-Lilla'],
|
|
30
|
+
['it', 'it-IT-Rosa'],
|
|
31
|
+
['ko', 'ko-KR-Haena'],
|
|
32
|
+
['nl', 'nl-NL-Fleur'],
|
|
33
|
+
['pt', 'pt-BR-Luana'],
|
|
34
|
+
['ro', 'ro-RO-Elena'],
|
|
35
|
+
['ru', 'ru-RU-Masha'],
|
|
36
|
+
['th', 'th-TH-Krit'],
|
|
37
|
+
['tr', 'tr-TR-Elif'],
|
|
38
|
+
['zh', 'zh-CN-Mei'],
|
|
39
|
+
]);
|
|
40
|
+
const DEFAULT_OUTPUT_FORMAT = 'audio-24khz-160kbitrate-mono-mp3';
|
|
41
|
+
|
|
42
|
+
// Shorthand formats for `outputFormat`; other X-Microsoft-OutputFormat values
|
|
43
|
+
// (e.g. `audio-48khz-192kbitrate-mono-mp3`) are passed through unchanged.
|
|
44
|
+
const OUTPUT_FORMATS = new Map([
|
|
45
|
+
['mp3', DEFAULT_OUTPUT_FORMAT],
|
|
46
|
+
['opus', 'ogg-24khz-16bit-mono-opus'],
|
|
47
|
+
['pcm', 'raw-24khz-16bit-mono-pcm'],
|
|
48
|
+
['wav', 'riff-24khz-16bit-mono-pcm'],
|
|
49
|
+
]);
|
|
50
|
+
const NATIVE_OUTPUT_FORMAT =
|
|
51
|
+
/^(?:amr|audio|g722|ogg|raw|riff|webm)-[a-z0-9-]+$/;
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Azure Speech text to speech with SSML. The model is appended to the voice
|
|
55
|
+
* name, e.g. `en-US-Harper:MAI-Voice-2`.
|
|
56
|
+
*/
|
|
57
|
+
export class AzureSpeechSpeechModel implements SpeechModelV3 {
|
|
58
|
+
readonly specificationVersion = 'v3';
|
|
59
|
+
readonly provider = 'azure.speech';
|
|
60
|
+
|
|
61
|
+
constructor(
|
|
62
|
+
readonly modelId: string,
|
|
63
|
+
private readonly config: {
|
|
64
|
+
url: () => string;
|
|
65
|
+
headers: () => Record<string, string | undefined>;
|
|
66
|
+
fetch?: FetchFunction;
|
|
67
|
+
_internal?: { currentDate?: () => Date };
|
|
68
|
+
},
|
|
69
|
+
) {}
|
|
70
|
+
|
|
71
|
+
async doGenerate(
|
|
72
|
+
options: Parameters<SpeechModelV3['doGenerate']>[0],
|
|
73
|
+
azureOptions: AzureSpeechModelSpeechOptions = {},
|
|
74
|
+
): Promise<Awaited<ReturnType<SpeechModelV3['doGenerate']>>> {
|
|
75
|
+
const currentDate = this.config._internal?.currentDate?.() ?? new Date();
|
|
76
|
+
const warnings: SharedV3Warning[] = [];
|
|
77
|
+
const { style, styleDegree } = azureOptions;
|
|
78
|
+
|
|
79
|
+
if (options.instructions != null) {
|
|
80
|
+
warnings.push({
|
|
81
|
+
type: 'unsupported',
|
|
82
|
+
feature: 'instructions',
|
|
83
|
+
details: 'Use providerOptions.azure.style to control speaking style.',
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
const { voice, languageWarning } = resolveVoice(
|
|
87
|
+
options.voice,
|
|
88
|
+
options.language,
|
|
89
|
+
);
|
|
90
|
+
if (languageWarning != null) {
|
|
91
|
+
warnings.push({
|
|
92
|
+
type: 'unsupported',
|
|
93
|
+
feature: 'language',
|
|
94
|
+
details: languageWarning,
|
|
95
|
+
});
|
|
96
|
+
}
|
|
97
|
+
if (styleDegree != null && style == null) {
|
|
98
|
+
warnings.push({
|
|
99
|
+
type: 'unsupported',
|
|
100
|
+
feature: 'providerOptions.azure.styleDegree',
|
|
101
|
+
details: 'styleDegree requires style.',
|
|
102
|
+
});
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
let outputFormat = DEFAULT_OUTPUT_FORMAT;
|
|
106
|
+
if (options.outputFormat != null) {
|
|
107
|
+
const format = options.outputFormat.toLowerCase();
|
|
108
|
+
const shorthand = OUTPUT_FORMATS.get(format);
|
|
109
|
+
if (shorthand != null) {
|
|
110
|
+
outputFormat = shorthand;
|
|
111
|
+
} else if (NATIVE_OUTPUT_FORMAT.test(format)) {
|
|
112
|
+
outputFormat = format;
|
|
113
|
+
} else {
|
|
114
|
+
warnings.push({
|
|
115
|
+
type: 'unsupported',
|
|
116
|
+
feature: 'outputFormat',
|
|
117
|
+
details: `Unsupported output format: ${options.outputFormat}. Using mp3 instead.`,
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
const ssml = buildSsml({
|
|
123
|
+
text: options.text,
|
|
124
|
+
voiceName: voice.includes(':')
|
|
125
|
+
? voice
|
|
126
|
+
: `${voice}:${getMAIVoiceModel(this.modelId) ?? this.modelId}`,
|
|
127
|
+
locale: /^([a-z]{2,3}-[a-z]{2,4})-/i.exec(voice)?.[1],
|
|
128
|
+
speed: options.speed,
|
|
129
|
+
style,
|
|
130
|
+
styleDegree: style != null ? styleDegree : undefined,
|
|
131
|
+
});
|
|
132
|
+
|
|
133
|
+
const {
|
|
134
|
+
value: audio,
|
|
135
|
+
responseHeaders,
|
|
136
|
+
rawValue,
|
|
137
|
+
} = await postToApi({
|
|
138
|
+
url: this.config.url(),
|
|
139
|
+
headers: combineHeaders(
|
|
140
|
+
this.config.headers(),
|
|
141
|
+
{
|
|
142
|
+
'Content-Type': 'application/ssml+xml',
|
|
143
|
+
'X-Microsoft-OutputFormat': outputFormat,
|
|
144
|
+
},
|
|
145
|
+
options.headers,
|
|
146
|
+
),
|
|
147
|
+
body: { content: ssml, values: ssml },
|
|
148
|
+
failedResponseHandler,
|
|
149
|
+
successfulResponseHandler: createBinaryResponseHandler(),
|
|
150
|
+
abortSignal: options.abortSignal,
|
|
151
|
+
fetch: this.config.fetch,
|
|
152
|
+
});
|
|
153
|
+
|
|
154
|
+
return {
|
|
155
|
+
audio,
|
|
156
|
+
warnings,
|
|
157
|
+
request: { body: ssml },
|
|
158
|
+
response: {
|
|
159
|
+
timestamp: currentDate,
|
|
160
|
+
modelId: this.modelId,
|
|
161
|
+
headers: responseHeaders,
|
|
162
|
+
body: rawValue,
|
|
163
|
+
},
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
function buildSsml({
|
|
169
|
+
text,
|
|
170
|
+
voiceName,
|
|
171
|
+
locale = 'en-US',
|
|
172
|
+
speed,
|
|
173
|
+
style,
|
|
174
|
+
styleDegree,
|
|
175
|
+
}: {
|
|
176
|
+
text: string;
|
|
177
|
+
voiceName: string;
|
|
178
|
+
locale?: string;
|
|
179
|
+
speed?: number;
|
|
180
|
+
style?: string;
|
|
181
|
+
styleDegree?: number;
|
|
182
|
+
}) {
|
|
183
|
+
let content = escapeXml(text);
|
|
184
|
+
if (speed != null) {
|
|
185
|
+
content = `<prosody rate="${speed}">${content}</prosody>`;
|
|
186
|
+
}
|
|
187
|
+
if (style != null) {
|
|
188
|
+
const degree = styleDegree != null ? ` styledegree="${styleDegree}"` : '';
|
|
189
|
+
content = `<mstts:express-as style="${escapeXml(style)}"${degree}>${content}</mstts:express-as>`;
|
|
190
|
+
}
|
|
191
|
+
return (
|
|
192
|
+
`<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" ` +
|
|
193
|
+
`xmlns:mstts="http://www.w3.org/2001/mstts" xml:lang="${escapeXml(locale)}">` +
|
|
194
|
+
`<voice name="${escapeXml(voiceName)}">${content}</voice></speak>`
|
|
195
|
+
);
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
const XML_ESCAPES: Record<string, string> = {
|
|
199
|
+
'&': '&',
|
|
200
|
+
'<': '<',
|
|
201
|
+
'>': '>',
|
|
202
|
+
'"': '"',
|
|
203
|
+
"'": ''',
|
|
204
|
+
};
|
|
205
|
+
|
|
206
|
+
function escapeXml(value: string) {
|
|
207
|
+
return value.replace(/[&<>"']/g, char => XML_ESCAPES[char]);
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
// Uses the language's default voice when no voice is set. An explicit voice
|
|
211
|
+
// wins, and its locale selects the language.
|
|
212
|
+
function resolveVoice(
|
|
213
|
+
voice: string | undefined,
|
|
214
|
+
language: string | undefined,
|
|
215
|
+
): { voice: string; languageWarning?: string } {
|
|
216
|
+
const code = language ? language.split('-')[0].toLowerCase() : undefined;
|
|
217
|
+
if (voice == null) {
|
|
218
|
+
if (code == null) return { voice: DEFAULT_VOICE };
|
|
219
|
+
if (code === 'auto') {
|
|
220
|
+
return {
|
|
221
|
+
voice: DEFAULT_VOICE,
|
|
222
|
+
languageWarning: `Automatic language detection is not supported. ${DEFAULT_VOICE} was used.`,
|
|
223
|
+
};
|
|
224
|
+
}
|
|
225
|
+
const defaultVoice = DEFAULT_VOICES.get(code);
|
|
226
|
+
return defaultVoice != null
|
|
227
|
+
? { voice: defaultVoice }
|
|
228
|
+
: {
|
|
229
|
+
voice: DEFAULT_VOICE,
|
|
230
|
+
languageWarning: `No default MAI voice for language "${language}". ${DEFAULT_VOICE} was used.`,
|
|
231
|
+
};
|
|
232
|
+
}
|
|
233
|
+
const voiceLanguage = /^([a-z]{2,3})-[a-z]{2,4}-/i
|
|
234
|
+
.exec(voice)?.[1]
|
|
235
|
+
?.toLowerCase();
|
|
236
|
+
return code != null &&
|
|
237
|
+
code !== 'auto' &&
|
|
238
|
+
voiceLanguage != null &&
|
|
239
|
+
code !== voiceLanguage
|
|
240
|
+
? {
|
|
241
|
+
voice,
|
|
242
|
+
languageWarning: `The voice ${voice} selects the language. Language "${language}" was ignored.`,
|
|
243
|
+
}
|
|
244
|
+
: { voice };
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
const errorSchema = z.object({
|
|
248
|
+
error: z.object({ message: z.string() }),
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
// Azure Speech answers invalid requests (unknown voice, style, or output
|
|
252
|
+
// format) with an empty 400 body.
|
|
253
|
+
const failedResponseHandler: ResponseHandler<APICallError> = async ({
|
|
254
|
+
response,
|
|
255
|
+
url,
|
|
256
|
+
requestBodyValues,
|
|
257
|
+
}) => {
|
|
258
|
+
const responseHeaders = extractResponseHeaders(response);
|
|
259
|
+
const responseBody = await response.text();
|
|
260
|
+
const parsed = await safeParseJSON({
|
|
261
|
+
text: responseBody,
|
|
262
|
+
schema: errorSchema,
|
|
263
|
+
});
|
|
264
|
+
|
|
265
|
+
const message = parsed.success
|
|
266
|
+
? parsed.value.error.message
|
|
267
|
+
: response.status === 400
|
|
268
|
+
? 'Azure Speech request failed with status 400. Check the voice name, style, and output format.'
|
|
269
|
+
: `Azure Speech request failed with status ${response.status}.`;
|
|
270
|
+
|
|
271
|
+
return {
|
|
272
|
+
responseHeaders,
|
|
273
|
+
value: new APICallError({
|
|
274
|
+
message,
|
|
275
|
+
url,
|
|
276
|
+
requestBodyValues,
|
|
277
|
+
statusCode: response.status,
|
|
278
|
+
responseHeaders,
|
|
279
|
+
responseBody,
|
|
280
|
+
}),
|
|
281
|
+
};
|
|
282
|
+
};
|
|
@@ -11,7 +11,7 @@ import {
|
|
|
11
11
|
import { z } from 'zod/v4';
|
|
12
12
|
import type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';
|
|
13
13
|
import type { AzureTranscriptionModelSpeechOptions } from './azure-speech-transcription-model-options';
|
|
14
|
-
import {
|
|
14
|
+
import { getMAITranscribeModel } from './azure-transcription-model-options';
|
|
15
15
|
|
|
16
16
|
export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
|
|
17
17
|
readonly specificationVersion = 'v3';
|
|
@@ -31,6 +31,7 @@ export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
|
|
|
31
31
|
azureOptions: AzureTranscriptionModelSpeechOptions = {},
|
|
32
32
|
): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>> {
|
|
33
33
|
const timestamp = new Date();
|
|
34
|
+
const maiModel = getMAITranscribeModel(this.modelId);
|
|
34
35
|
const formData = new FormData();
|
|
35
36
|
formData.append(
|
|
36
37
|
'audio',
|
|
@@ -49,11 +50,11 @@ export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
|
|
|
49
50
|
JSON.stringify({
|
|
50
51
|
enhancedMode: {
|
|
51
52
|
enabled: true,
|
|
52
|
-
model:
|
|
53
|
-
? 'MAI-Transcribe-2'
|
|
54
|
-
: this.modelId,
|
|
53
|
+
model: maiModel?.name ?? this.modelId,
|
|
55
54
|
modelOptions: {
|
|
56
|
-
timestamps:
|
|
55
|
+
timestamps:
|
|
56
|
+
azureOptions.timestamps ??
|
|
57
|
+
(maiModel?.supportsTimestamps === false ? undefined : 'segment'),
|
|
57
58
|
transcribeStyle: azureOptions.transcribeStyle,
|
|
58
59
|
},
|
|
59
60
|
},
|
|
@@ -10,7 +10,7 @@ export const azureTranscriptionModelOptions = lazySchema(() =>
|
|
|
10
10
|
zodSchema(
|
|
11
11
|
z.strictObject({
|
|
12
12
|
/**
|
|
13
|
-
* API to use. Defaults to Speech for MAI-Transcribe
|
|
13
|
+
* API to use. Defaults to Speech for MAI-Transcribe models, OpenAI otherwise.
|
|
14
14
|
*/
|
|
15
15
|
api: z.enum(['openai', 'speech']).optional(),
|
|
16
16
|
...azureSpeechTranscriptionModelOptionsShape(),
|
|
@@ -22,6 +22,16 @@ export type AzureTranscriptionModelOptions = InferSchema<
|
|
|
22
22
|
typeof azureTranscriptionModelOptions
|
|
23
23
|
>;
|
|
24
24
|
|
|
25
|
-
|
|
26
|
-
|
|
25
|
+
// Azure Speech model names by lowercase model ID. MAI-Transcribe-1.5 only
|
|
26
|
+
// accepts Azure's default timestamps (`none`).
|
|
27
|
+
const maiTranscribeModels = new Map([
|
|
28
|
+
['mai-transcribe-2', { name: 'MAI-Transcribe-2', supportsTimestamps: true }],
|
|
29
|
+
[
|
|
30
|
+
'mai-transcribe-1.5',
|
|
31
|
+
{ name: 'MAI-Transcribe-1.5', supportsTimestamps: false },
|
|
32
|
+
],
|
|
33
|
+
]);
|
|
34
|
+
|
|
35
|
+
export function getMAITranscribeModel(modelId: string) {
|
|
36
|
+
return maiTranscribeModels.get(modelId.toLowerCase());
|
|
27
37
|
}
|
package/src/index.ts
CHANGED
|
@@ -29,5 +29,6 @@ export type {
|
|
|
29
29
|
AzureResponsesSourceDocumentProviderMetadata,
|
|
30
30
|
} from './azure-openai-provider-metadata';
|
|
31
31
|
export { VERSION } from './version';
|
|
32
|
+
export type { AzureSpeechModelOptions } from './azure-speech-model-options';
|
|
32
33
|
export type { AzureTranscriptionModelOptions } from './azure-transcription-model-options';
|
|
33
34
|
export type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';
|