@ai-sdk/azure 3.0.130 → 3.0.132

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,33 @@
1
+ import {
2
+ lazySchema,
3
+ zodSchema,
4
+ type InferSchema,
5
+ } from '@ai-sdk/provider-utils';
6
+ import { z } from 'zod/v4';
7
+ import { azureSpeechSpeechModelOptionsShape } from './azure-speech-speech-model-options';
8
+
9
+ export const azureSpeechModelOptions = lazySchema(() =>
10
+ zodSchema(
11
+ z.strictObject({
12
+ /**
13
+ * API to use. Defaults to Speech for MAI-Voice models, OpenAI otherwise.
14
+ */
15
+ api: z.enum(['openai', 'speech']).optional(),
16
+ ...azureSpeechSpeechModelOptionsShape(),
17
+ }),
18
+ ),
19
+ );
20
+
21
+ export type AzureSpeechModelOptions = InferSchema<
22
+ typeof azureSpeechModelOptions
23
+ >;
24
+
25
+ // Azure voice-name suffixes by lowercase model ID.
26
+ const maiVoiceModels = new Map([
27
+ ['mai-voice-2-flash', 'MAI-Voice-2-Flash'],
28
+ ['mai-voice-2', 'MAI-Voice-2'],
29
+ ]);
30
+
31
+ export function getMAIVoiceModel(modelId: string) {
32
+ return maiVoiceModels.get(modelId.toLowerCase());
33
+ }
@@ -0,0 +1,27 @@
1
+ import {
2
+ lazySchema,
3
+ zodSchema,
4
+ type InferSchema,
5
+ } from '@ai-sdk/provider-utils';
6
+ import { z } from 'zod/v4';
7
+
8
+ export const azureSpeechSpeechModelOptionsShape = () => ({
9
+ /**
10
+ * Speaking style applied with `mstts:express-as`, e.g. `excited` or
11
+ * `whispering`. Supported styles vary by voice.
12
+ */
13
+ style: z.string().min(1).optional(),
14
+
15
+ /**
16
+ * Intensity of `style`, from 0.01 to 2. Azure defaults to 1.
17
+ */
18
+ styleDegree: z.number().min(0.01).max(2).optional(),
19
+ });
20
+
21
+ export const azureSpeechSpeechModelOptions = lazySchema(() =>
22
+ zodSchema(z.strictObject(azureSpeechSpeechModelOptionsShape())),
23
+ );
24
+
25
+ export type AzureSpeechModelSpeechOptions = InferSchema<
26
+ typeof azureSpeechSpeechModelOptions
27
+ >;
@@ -0,0 +1,282 @@
1
+ import {
2
+ APICallError,
3
+ type SharedV3Warning,
4
+ type SpeechModelV3,
5
+ } from '@ai-sdk/provider';
6
+ import {
7
+ combineHeaders,
8
+ createBinaryResponseHandler,
9
+ extractResponseHeaders,
10
+ postToApi,
11
+ safeParseJSON,
12
+ type FetchFunction,
13
+ type ResponseHandler,
14
+ } from '@ai-sdk/provider-utils';
15
+ import { z } from 'zod/v4';
16
+ import { getMAIVoiceModel } from './azure-speech-model-options';
17
+ import type { AzureSpeechModelSpeechOptions } from './azure-speech-speech-model-options';
18
+
19
+ const DEFAULT_VOICE = 'en-US-Harper';
20
+
21
+ // Default voice per ISO 639-1 language; available on MAI-Voice-2 and
22
+ // MAI-Voice-2-Flash.
23
+ const DEFAULT_VOICES = new Map([
24
+ ['de', 'de-DE-Mia'],
25
+ ['en', DEFAULT_VOICE],
26
+ ['es', 'es-MX-Valeria'],
27
+ ['fr', 'fr-FR-Soleil'],
28
+ ['hi', 'hi-IN-Kavya'],
29
+ ['hu', 'hu-HU-Lilla'],
30
+ ['it', 'it-IT-Rosa'],
31
+ ['ko', 'ko-KR-Haena'],
32
+ ['nl', 'nl-NL-Fleur'],
33
+ ['pt', 'pt-BR-Luana'],
34
+ ['ro', 'ro-RO-Elena'],
35
+ ['ru', 'ru-RU-Masha'],
36
+ ['th', 'th-TH-Krit'],
37
+ ['tr', 'tr-TR-Elif'],
38
+ ['zh', 'zh-CN-Mei'],
39
+ ]);
40
+ const DEFAULT_OUTPUT_FORMAT = 'audio-24khz-160kbitrate-mono-mp3';
41
+
42
+ // Shorthand formats for `outputFormat`; other X-Microsoft-OutputFormat values
43
+ // (e.g. `audio-48khz-192kbitrate-mono-mp3`) are passed through unchanged.
44
+ const OUTPUT_FORMATS = new Map([
45
+ ['mp3', DEFAULT_OUTPUT_FORMAT],
46
+ ['opus', 'ogg-24khz-16bit-mono-opus'],
47
+ ['pcm', 'raw-24khz-16bit-mono-pcm'],
48
+ ['wav', 'riff-24khz-16bit-mono-pcm'],
49
+ ]);
50
+ const NATIVE_OUTPUT_FORMAT =
51
+ /^(?:amr|audio|g722|ogg|raw|riff|webm)-[a-z0-9-]+$/;
52
+
53
+ /**
54
+ * Azure Speech text to speech with SSML. The model is appended to the voice
55
+ * name, e.g. `en-US-Harper:MAI-Voice-2`.
56
+ */
57
+ export class AzureSpeechSpeechModel implements SpeechModelV3 {
58
+ readonly specificationVersion = 'v3';
59
+ readonly provider = 'azure.speech';
60
+
61
+ constructor(
62
+ readonly modelId: string,
63
+ private readonly config: {
64
+ url: () => string;
65
+ headers: () => Record<string, string | undefined>;
66
+ fetch?: FetchFunction;
67
+ _internal?: { currentDate?: () => Date };
68
+ },
69
+ ) {}
70
+
71
+ async doGenerate(
72
+ options: Parameters<SpeechModelV3['doGenerate']>[0],
73
+ azureOptions: AzureSpeechModelSpeechOptions = {},
74
+ ): Promise<Awaited<ReturnType<SpeechModelV3['doGenerate']>>> {
75
+ const currentDate = this.config._internal?.currentDate?.() ?? new Date();
76
+ const warnings: SharedV3Warning[] = [];
77
+ const { style, styleDegree } = azureOptions;
78
+
79
+ if (options.instructions != null) {
80
+ warnings.push({
81
+ type: 'unsupported',
82
+ feature: 'instructions',
83
+ details: 'Use providerOptions.azure.style to control speaking style.',
84
+ });
85
+ }
86
+ const { voice, languageWarning } = resolveVoice(
87
+ options.voice,
88
+ options.language,
89
+ );
90
+ if (languageWarning != null) {
91
+ warnings.push({
92
+ type: 'unsupported',
93
+ feature: 'language',
94
+ details: languageWarning,
95
+ });
96
+ }
97
+ if (styleDegree != null && style == null) {
98
+ warnings.push({
99
+ type: 'unsupported',
100
+ feature: 'providerOptions.azure.styleDegree',
101
+ details: 'styleDegree requires style.',
102
+ });
103
+ }
104
+
105
+ let outputFormat = DEFAULT_OUTPUT_FORMAT;
106
+ if (options.outputFormat != null) {
107
+ const format = options.outputFormat.toLowerCase();
108
+ const shorthand = OUTPUT_FORMATS.get(format);
109
+ if (shorthand != null) {
110
+ outputFormat = shorthand;
111
+ } else if (NATIVE_OUTPUT_FORMAT.test(format)) {
112
+ outputFormat = format;
113
+ } else {
114
+ warnings.push({
115
+ type: 'unsupported',
116
+ feature: 'outputFormat',
117
+ details: `Unsupported output format: ${options.outputFormat}. Using mp3 instead.`,
118
+ });
119
+ }
120
+ }
121
+
122
+ const ssml = buildSsml({
123
+ text: options.text,
124
+ voiceName: voice.includes(':')
125
+ ? voice
126
+ : `${voice}:${getMAIVoiceModel(this.modelId) ?? this.modelId}`,
127
+ locale: /^([a-z]{2,3}-[a-z]{2,4})-/i.exec(voice)?.[1],
128
+ speed: options.speed,
129
+ style,
130
+ styleDegree: style != null ? styleDegree : undefined,
131
+ });
132
+
133
+ const {
134
+ value: audio,
135
+ responseHeaders,
136
+ rawValue,
137
+ } = await postToApi({
138
+ url: this.config.url(),
139
+ headers: combineHeaders(
140
+ this.config.headers(),
141
+ {
142
+ 'Content-Type': 'application/ssml+xml',
143
+ 'X-Microsoft-OutputFormat': outputFormat,
144
+ },
145
+ options.headers,
146
+ ),
147
+ body: { content: ssml, values: ssml },
148
+ failedResponseHandler,
149
+ successfulResponseHandler: createBinaryResponseHandler(),
150
+ abortSignal: options.abortSignal,
151
+ fetch: this.config.fetch,
152
+ });
153
+
154
+ return {
155
+ audio,
156
+ warnings,
157
+ request: { body: ssml },
158
+ response: {
159
+ timestamp: currentDate,
160
+ modelId: this.modelId,
161
+ headers: responseHeaders,
162
+ body: rawValue,
163
+ },
164
+ };
165
+ }
166
+ }
167
+
168
+ function buildSsml({
169
+ text,
170
+ voiceName,
171
+ locale = 'en-US',
172
+ speed,
173
+ style,
174
+ styleDegree,
175
+ }: {
176
+ text: string;
177
+ voiceName: string;
178
+ locale?: string;
179
+ speed?: number;
180
+ style?: string;
181
+ styleDegree?: number;
182
+ }) {
183
+ let content = escapeXml(text);
184
+ if (speed != null) {
185
+ content = `<prosody rate="${speed}">${content}</prosody>`;
186
+ }
187
+ if (style != null) {
188
+ const degree = styleDegree != null ? ` styledegree="${styleDegree}"` : '';
189
+ content = `<mstts:express-as style="${escapeXml(style)}"${degree}>${content}</mstts:express-as>`;
190
+ }
191
+ return (
192
+ `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" ` +
193
+ `xmlns:mstts="http://www.w3.org/2001/mstts" xml:lang="${escapeXml(locale)}">` +
194
+ `<voice name="${escapeXml(voiceName)}">${content}</voice></speak>`
195
+ );
196
+ }
197
+
198
+ const XML_ESCAPES: Record<string, string> = {
199
+ '&': '&amp;',
200
+ '<': '&lt;',
201
+ '>': '&gt;',
202
+ '"': '&quot;',
203
+ "'": '&apos;',
204
+ };
205
+
206
+ function escapeXml(value: string) {
207
+ return value.replace(/[&<>"']/g, char => XML_ESCAPES[char]);
208
+ }
209
+
210
+ // Uses the language's default voice when no voice is set. An explicit voice
211
+ // wins, and its locale selects the language.
212
+ function resolveVoice(
213
+ voice: string | undefined,
214
+ language: string | undefined,
215
+ ): { voice: string; languageWarning?: string } {
216
+ const code = language ? language.split('-')[0].toLowerCase() : undefined;
217
+ if (voice == null) {
218
+ if (code == null) return { voice: DEFAULT_VOICE };
219
+ if (code === 'auto') {
220
+ return {
221
+ voice: DEFAULT_VOICE,
222
+ languageWarning: `Automatic language detection is not supported. ${DEFAULT_VOICE} was used.`,
223
+ };
224
+ }
225
+ const defaultVoice = DEFAULT_VOICES.get(code);
226
+ return defaultVoice != null
227
+ ? { voice: defaultVoice }
228
+ : {
229
+ voice: DEFAULT_VOICE,
230
+ languageWarning: `No default MAI voice for language "${language}". ${DEFAULT_VOICE} was used.`,
231
+ };
232
+ }
233
+ const voiceLanguage = /^([a-z]{2,3})-[a-z]{2,4}-/i
234
+ .exec(voice)?.[1]
235
+ ?.toLowerCase();
236
+ return code != null &&
237
+ code !== 'auto' &&
238
+ voiceLanguage != null &&
239
+ code !== voiceLanguage
240
+ ? {
241
+ voice,
242
+ languageWarning: `The voice ${voice} selects the language. Language "${language}" was ignored.`,
243
+ }
244
+ : { voice };
245
+ }
246
+
247
+ const errorSchema = z.object({
248
+ error: z.object({ message: z.string() }),
249
+ });
250
+
251
+ // Azure Speech answers invalid requests (unknown voice, style, or output
252
+ // format) with an empty 400 body.
253
+ const failedResponseHandler: ResponseHandler<APICallError> = async ({
254
+ response,
255
+ url,
256
+ requestBodyValues,
257
+ }) => {
258
+ const responseHeaders = extractResponseHeaders(response);
259
+ const responseBody = await response.text();
260
+ const parsed = await safeParseJSON({
261
+ text: responseBody,
262
+ schema: errorSchema,
263
+ });
264
+
265
+ const message = parsed.success
266
+ ? parsed.value.error.message
267
+ : response.status === 400
268
+ ? 'Azure Speech request failed with status 400. Check the voice name, style, and output format.'
269
+ : `Azure Speech request failed with status ${response.status}.`;
270
+
271
+ return {
272
+ responseHeaders,
273
+ value: new APICallError({
274
+ message,
275
+ url,
276
+ requestBodyValues,
277
+ statusCode: response.status,
278
+ responseHeaders,
279
+ responseBody,
280
+ }),
281
+ };
282
+ };
@@ -11,7 +11,7 @@ import {
11
11
  import { z } from 'zod/v4';
12
12
  import type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';
13
13
  import type { AzureTranscriptionModelSpeechOptions } from './azure-speech-transcription-model-options';
14
- import { isMAITranscribe2 } from './azure-transcription-model-options';
14
+ import { getMAITranscribeModel } from './azure-transcription-model-options';
15
15
 
16
16
  export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
17
17
  readonly specificationVersion = 'v3';
@@ -31,6 +31,7 @@ export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
31
31
  azureOptions: AzureTranscriptionModelSpeechOptions = {},
32
32
  ): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>> {
33
33
  const timestamp = new Date();
34
+ const maiModel = getMAITranscribeModel(this.modelId);
34
35
  const formData = new FormData();
35
36
  formData.append(
36
37
  'audio',
@@ -49,11 +50,11 @@ export class AzureSpeechTranscriptionModel implements TranscriptionModelV3 {
49
50
  JSON.stringify({
50
51
  enhancedMode: {
51
52
  enabled: true,
52
- model: isMAITranscribe2(this.modelId)
53
- ? 'MAI-Transcribe-2'
54
- : this.modelId,
53
+ model: maiModel?.name ?? this.modelId,
55
54
  modelOptions: {
56
- timestamps: azureOptions.timestamps ?? 'segment',
55
+ timestamps:
56
+ azureOptions.timestamps ??
57
+ (maiModel?.supportsTimestamps === false ? undefined : 'segment'),
57
58
  transcribeStyle: azureOptions.transcribeStyle,
58
59
  },
59
60
  },
@@ -10,7 +10,7 @@ export const azureTranscriptionModelOptions = lazySchema(() =>
10
10
  zodSchema(
11
11
  z.strictObject({
12
12
  /**
13
- * API to use. Defaults to Speech for MAI-Transcribe-2, OpenAI otherwise.
13
+ * API to use. Defaults to Speech for MAI-Transcribe models, OpenAI otherwise.
14
14
  */
15
15
  api: z.enum(['openai', 'speech']).optional(),
16
16
  ...azureSpeechTranscriptionModelOptionsShape(),
@@ -22,6 +22,16 @@ export type AzureTranscriptionModelOptions = InferSchema<
22
22
  typeof azureTranscriptionModelOptions
23
23
  >;
24
24
 
25
- export function isMAITranscribe2(modelId: string): boolean {
26
- return modelId.toLowerCase() === 'mai-transcribe-2';
25
+ // Azure Speech model names by lowercase model ID. MAI-Transcribe-1.5 only
26
+ // accepts Azure's default timestamps (`none`).
27
+ const maiTranscribeModels = new Map([
28
+ ['mai-transcribe-2', { name: 'MAI-Transcribe-2', supportsTimestamps: true }],
29
+ [
30
+ 'mai-transcribe-1.5',
31
+ { name: 'MAI-Transcribe-1.5', supportsTimestamps: false },
32
+ ],
33
+ ]);
34
+
35
+ export function getMAITranscribeModel(modelId: string) {
36
+ return maiTranscribeModels.get(modelId.toLowerCase());
27
37
  }
package/src/index.ts CHANGED
@@ -29,5 +29,6 @@ export type {
29
29
  AzureResponsesSourceDocumentProviderMetadata,
30
30
  } from './azure-openai-provider-metadata';
31
31
  export { VERSION } from './version';
32
+ export type { AzureSpeechModelOptions } from './azure-speech-model-options';
32
33
  export type { AzureTranscriptionModelOptions } from './azure-transcription-model-options';
33
34
  export type { AzureTranscriptionProviderMetadata } from './azure-transcription-provider-metadata';