@ai-sdk/azure 3.0.130 → 3.0.132

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -29,8 +29,8 @@ module.exports = __toCommonJS(index_exports);
29
29
  // src/azure-openai-provider.ts
30
30
  var import_internal2 = require("@ai-sdk/openai/internal");
31
31
  var import_internal3 = require("@ai-sdk/deepseek/internal");
32
- var import_provider = require("@ai-sdk/provider");
33
- var import_provider_utils4 = require("@ai-sdk/provider-utils");
32
+ var import_provider2 = require("@ai-sdk/provider");
33
+ var import_provider_utils7 = require("@ai-sdk/provider-utils");
34
34
 
35
35
  // src/azure-openai-tools.ts
36
36
  var import_internal = require("@ai-sdk/openai/internal");
@@ -42,60 +42,312 @@ var azureOpenaiTools = {
42
42
  webSearchPreview: import_internal.webSearchPreview
43
43
  };
44
44
 
45
- // src/azure-speech-transcription-model.ts
45
+ // src/azure-speech-model-options.ts
46
+ var import_provider_utils2 = require("@ai-sdk/provider-utils");
47
+ var import_v42 = require("zod/v4");
48
+
49
+ // src/azure-speech-speech-model-options.ts
50
+ var import_provider_utils = require("@ai-sdk/provider-utils");
51
+ var import_v4 = require("zod/v4");
52
+ var azureSpeechSpeechModelOptionsShape = () => ({
53
+ /**
54
+ * Speaking style applied with `mstts:express-as`, e.g. `excited` or
55
+ * `whispering`. Supported styles vary by voice.
56
+ */
57
+ style: import_v4.z.string().min(1).optional(),
58
+ /**
59
+ * Intensity of `style`, from 0.01 to 2. Azure defaults to 1.
60
+ */
61
+ styleDegree: import_v4.z.number().min(0.01).max(2).optional()
62
+ });
63
+ var azureSpeechSpeechModelOptions = (0, import_provider_utils.lazySchema)(
64
+ () => (0, import_provider_utils.zodSchema)(import_v4.z.strictObject(azureSpeechSpeechModelOptionsShape()))
65
+ );
66
+
67
+ // src/azure-speech-model-options.ts
68
+ var azureSpeechModelOptions = (0, import_provider_utils2.lazySchema)(
69
+ () => (0, import_provider_utils2.zodSchema)(
70
+ import_v42.z.strictObject({
71
+ /**
72
+ * API to use. Defaults to Speech for MAI-Voice models, OpenAI otherwise.
73
+ */
74
+ api: import_v42.z.enum(["openai", "speech"]).optional(),
75
+ ...azureSpeechSpeechModelOptionsShape()
76
+ })
77
+ )
78
+ );
79
+ var maiVoiceModels = /* @__PURE__ */ new Map([
80
+ ["mai-voice-2-flash", "MAI-Voice-2-Flash"],
81
+ ["mai-voice-2", "MAI-Voice-2"]
82
+ ]);
83
+ function getMAIVoiceModel(modelId) {
84
+ return maiVoiceModels.get(modelId.toLowerCase());
85
+ }
86
+
87
+ // src/azure-speech-speech-model.ts
88
+ var import_provider = require("@ai-sdk/provider");
46
89
  var import_provider_utils3 = require("@ai-sdk/provider-utils");
47
90
  var import_v43 = require("zod/v4");
91
+ var DEFAULT_VOICE = "en-US-Harper";
92
+ var DEFAULT_VOICES = /* @__PURE__ */ new Map([
93
+ ["de", "de-DE-Mia"],
94
+ ["en", DEFAULT_VOICE],
95
+ ["es", "es-MX-Valeria"],
96
+ ["fr", "fr-FR-Soleil"],
97
+ ["hi", "hi-IN-Kavya"],
98
+ ["hu", "hu-HU-Lilla"],
99
+ ["it", "it-IT-Rosa"],
100
+ ["ko", "ko-KR-Haena"],
101
+ ["nl", "nl-NL-Fleur"],
102
+ ["pt", "pt-BR-Luana"],
103
+ ["ro", "ro-RO-Elena"],
104
+ ["ru", "ru-RU-Masha"],
105
+ ["th", "th-TH-Krit"],
106
+ ["tr", "tr-TR-Elif"],
107
+ ["zh", "zh-CN-Mei"]
108
+ ]);
109
+ var DEFAULT_OUTPUT_FORMAT = "audio-24khz-160kbitrate-mono-mp3";
110
+ var OUTPUT_FORMATS = /* @__PURE__ */ new Map([
111
+ ["mp3", DEFAULT_OUTPUT_FORMAT],
112
+ ["opus", "ogg-24khz-16bit-mono-opus"],
113
+ ["pcm", "raw-24khz-16bit-mono-pcm"],
114
+ ["wav", "riff-24khz-16bit-mono-pcm"]
115
+ ]);
116
+ var NATIVE_OUTPUT_FORMAT = /^(?:amr|audio|g722|ogg|raw|riff|webm)-[a-z0-9-]+$/;
117
+ var AzureSpeechSpeechModel = class {
118
+ constructor(modelId, config) {
119
+ this.modelId = modelId;
120
+ this.config = config;
121
+ this.specificationVersion = "v3";
122
+ this.provider = "azure.speech";
123
+ }
124
+ async doGenerate(options, azureOptions = {}) {
125
+ var _a, _b, _c, _d, _e;
126
+ const currentDate = (_c = (_b = (_a = this.config._internal) == null ? void 0 : _a.currentDate) == null ? void 0 : _b.call(_a)) != null ? _c : /* @__PURE__ */ new Date();
127
+ const warnings = [];
128
+ const { style, styleDegree } = azureOptions;
129
+ if (options.instructions != null) {
130
+ warnings.push({
131
+ type: "unsupported",
132
+ feature: "instructions",
133
+ details: "Use providerOptions.azure.style to control speaking style."
134
+ });
135
+ }
136
+ const { voice, languageWarning } = resolveVoice(
137
+ options.voice,
138
+ options.language
139
+ );
140
+ if (languageWarning != null) {
141
+ warnings.push({
142
+ type: "unsupported",
143
+ feature: "language",
144
+ details: languageWarning
145
+ });
146
+ }
147
+ if (styleDegree != null && style == null) {
148
+ warnings.push({
149
+ type: "unsupported",
150
+ feature: "providerOptions.azure.styleDegree",
151
+ details: "styleDegree requires style."
152
+ });
153
+ }
154
+ let outputFormat = DEFAULT_OUTPUT_FORMAT;
155
+ if (options.outputFormat != null) {
156
+ const format = options.outputFormat.toLowerCase();
157
+ const shorthand = OUTPUT_FORMATS.get(format);
158
+ if (shorthand != null) {
159
+ outputFormat = shorthand;
160
+ } else if (NATIVE_OUTPUT_FORMAT.test(format)) {
161
+ outputFormat = format;
162
+ } else {
163
+ warnings.push({
164
+ type: "unsupported",
165
+ feature: "outputFormat",
166
+ details: `Unsupported output format: ${options.outputFormat}. Using mp3 instead.`
167
+ });
168
+ }
169
+ }
170
+ const ssml = buildSsml({
171
+ text: options.text,
172
+ voiceName: voice.includes(":") ? voice : `${voice}:${(_d = getMAIVoiceModel(this.modelId)) != null ? _d : this.modelId}`,
173
+ locale: (_e = /^([a-z]{2,3}-[a-z]{2,4})-/i.exec(voice)) == null ? void 0 : _e[1],
174
+ speed: options.speed,
175
+ style,
176
+ styleDegree: style != null ? styleDegree : void 0
177
+ });
178
+ const {
179
+ value: audio,
180
+ responseHeaders,
181
+ rawValue
182
+ } = await (0, import_provider_utils3.postToApi)({
183
+ url: this.config.url(),
184
+ headers: (0, import_provider_utils3.combineHeaders)(
185
+ this.config.headers(),
186
+ {
187
+ "Content-Type": "application/ssml+xml",
188
+ "X-Microsoft-OutputFormat": outputFormat
189
+ },
190
+ options.headers
191
+ ),
192
+ body: { content: ssml, values: ssml },
193
+ failedResponseHandler,
194
+ successfulResponseHandler: (0, import_provider_utils3.createBinaryResponseHandler)(),
195
+ abortSignal: options.abortSignal,
196
+ fetch: this.config.fetch
197
+ });
198
+ return {
199
+ audio,
200
+ warnings,
201
+ request: { body: ssml },
202
+ response: {
203
+ timestamp: currentDate,
204
+ modelId: this.modelId,
205
+ headers: responseHeaders,
206
+ body: rawValue
207
+ }
208
+ };
209
+ }
210
+ };
211
+ function buildSsml({
212
+ text,
213
+ voiceName,
214
+ locale = "en-US",
215
+ speed,
216
+ style,
217
+ styleDegree
218
+ }) {
219
+ let content = escapeXml(text);
220
+ if (speed != null) {
221
+ content = `<prosody rate="${speed}">${content}</prosody>`;
222
+ }
223
+ if (style != null) {
224
+ const degree = styleDegree != null ? ` styledegree="${styleDegree}"` : "";
225
+ content = `<mstts:express-as style="${escapeXml(style)}"${degree}>${content}</mstts:express-as>`;
226
+ }
227
+ return `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="http://www.w3.org/2001/mstts" xml:lang="${escapeXml(locale)}"><voice name="${escapeXml(voiceName)}">${content}</voice></speak>`;
228
+ }
229
+ var XML_ESCAPES = {
230
+ "&": "&amp;",
231
+ "<": "&lt;",
232
+ ">": "&gt;",
233
+ '"': "&quot;",
234
+ "'": "&apos;"
235
+ };
236
+ function escapeXml(value) {
237
+ return value.replace(/[&<>"']/g, (char) => XML_ESCAPES[char]);
238
+ }
239
+ function resolveVoice(voice, language) {
240
+ var _a, _b;
241
+ const code = language ? language.split("-")[0].toLowerCase() : void 0;
242
+ if (voice == null) {
243
+ if (code == null) return { voice: DEFAULT_VOICE };
244
+ if (code === "auto") {
245
+ return {
246
+ voice: DEFAULT_VOICE,
247
+ languageWarning: `Automatic language detection is not supported. ${DEFAULT_VOICE} was used.`
248
+ };
249
+ }
250
+ const defaultVoice = DEFAULT_VOICES.get(code);
251
+ return defaultVoice != null ? { voice: defaultVoice } : {
252
+ voice: DEFAULT_VOICE,
253
+ languageWarning: `No default MAI voice for language "${language}". ${DEFAULT_VOICE} was used.`
254
+ };
255
+ }
256
+ const voiceLanguage = (_b = (_a = /^([a-z]{2,3})-[a-z]{2,4}-/i.exec(voice)) == null ? void 0 : _a[1]) == null ? void 0 : _b.toLowerCase();
257
+ return code != null && code !== "auto" && voiceLanguage != null && code !== voiceLanguage ? {
258
+ voice,
259
+ languageWarning: `The voice ${voice} selects the language. Language "${language}" was ignored.`
260
+ } : { voice };
261
+ }
262
+ var errorSchema = import_v43.z.object({
263
+ error: import_v43.z.object({ message: import_v43.z.string() })
264
+ });
265
+ var failedResponseHandler = async ({
266
+ response,
267
+ url,
268
+ requestBodyValues
269
+ }) => {
270
+ const responseHeaders = (0, import_provider_utils3.extractResponseHeaders)(response);
271
+ const responseBody = await response.text();
272
+ const parsed = await (0, import_provider_utils3.safeParseJSON)({
273
+ text: responseBody,
274
+ schema: errorSchema
275
+ });
276
+ const message = parsed.success ? parsed.value.error.message : response.status === 400 ? "Azure Speech request failed with status 400. Check the voice name, style, and output format." : `Azure Speech request failed with status ${response.status}.`;
277
+ return {
278
+ responseHeaders,
279
+ value: new import_provider.APICallError({
280
+ message,
281
+ url,
282
+ requestBodyValues,
283
+ statusCode: response.status,
284
+ responseHeaders,
285
+ responseBody
286
+ })
287
+ };
288
+ };
289
+
290
+ // src/azure-speech-transcription-model.ts
291
+ var import_provider_utils6 = require("@ai-sdk/provider-utils");
292
+ var import_v46 = require("zod/v4");
48
293
 
49
294
  // src/azure-transcription-model-options.ts
50
- var import_provider_utils2 = require("@ai-sdk/provider-utils");
51
- var import_v42 = require("zod/v4");
295
+ var import_provider_utils5 = require("@ai-sdk/provider-utils");
296
+ var import_v45 = require("zod/v4");
52
297
 
53
298
  // src/azure-speech-transcription-model-options.ts
54
- var import_provider_utils = require("@ai-sdk/provider-utils");
55
- var import_v4 = require("zod/v4");
299
+ var import_provider_utils4 = require("@ai-sdk/provider-utils");
300
+ var import_v44 = require("zod/v4");
56
301
  var azureSpeechTranscriptionModelOptionsShape = () => ({
57
302
  /**
58
303
  * Timing granularity. Defaults to `segment` so that transcription results
59
304
  * include timed segments (Azure's own default is `none`).
60
305
  */
61
- timestamps: import_v4.z.enum(["word", "segment", "none"]).optional(),
306
+ timestamps: import_v44.z.enum(["word", "segment", "none"]).optional(),
62
307
  /**
63
308
  * `verbatim` keeps fillers and false starts, `clean` removes them.
64
309
  * Azure defaults to `verbatim`.
65
310
  */
66
- transcribeStyle: import_v4.z.enum(["verbatim", "clean"]).optional(),
311
+ transcribeStyle: import_v44.z.enum(["verbatim", "clean"]).optional(),
67
312
  /**
68
313
  * Forces a single language, e.g. `['en']`. Omit for automatic language
69
314
  * detection and code switching.
70
315
  */
71
- locales: import_v4.z.array(import_v4.z.string()).length(1).optional(),
316
+ locales: import_v44.z.array(import_v44.z.string()).length(1).optional(),
72
317
  /**
73
318
  * Speaker diarization. Speaker IDs are available in provider metadata.
74
319
  */
75
- diarization: import_v4.z.strictObject({ enabled: import_v4.z.boolean() }).optional(),
320
+ diarization: import_v44.z.strictObject({ enabled: import_v44.z.boolean() }).optional(),
76
321
  /**
77
322
  * Keyword biasing for names and domain terminology.
78
323
  */
79
- phraseList: import_v4.z.strictObject({ phrases: import_v4.z.array(import_v4.z.string()) }).optional()
324
+ phraseList: import_v44.z.strictObject({ phrases: import_v44.z.array(import_v44.z.string()) }).optional()
80
325
  });
81
- var azureSpeechTranscriptionModelOptions = (0, import_provider_utils.lazySchema)(
82
- () => (0, import_provider_utils.zodSchema)(import_v4.z.strictObject(azureSpeechTranscriptionModelOptionsShape()))
326
+ var azureSpeechTranscriptionModelOptions = (0, import_provider_utils4.lazySchema)(
327
+ () => (0, import_provider_utils4.zodSchema)(import_v44.z.strictObject(azureSpeechTranscriptionModelOptionsShape()))
83
328
  );
84
329
 
85
330
  // src/azure-transcription-model-options.ts
86
- var azureTranscriptionModelOptions = (0, import_provider_utils2.lazySchema)(
87
- () => (0, import_provider_utils2.zodSchema)(
88
- import_v42.z.strictObject({
331
+ var azureTranscriptionModelOptions = (0, import_provider_utils5.lazySchema)(
332
+ () => (0, import_provider_utils5.zodSchema)(
333
+ import_v45.z.strictObject({
89
334
  /**
90
- * API to use. Defaults to Speech for MAI-Transcribe-2, OpenAI otherwise.
335
+ * API to use. Defaults to Speech for MAI-Transcribe models, OpenAI otherwise.
91
336
  */
92
- api: import_v42.z.enum(["openai", "speech"]).optional(),
337
+ api: import_v45.z.enum(["openai", "speech"]).optional(),
93
338
  ...azureSpeechTranscriptionModelOptionsShape()
94
339
  })
95
340
  )
96
341
  );
97
- function isMAITranscribe2(modelId) {
98
- return modelId.toLowerCase() === "mai-transcribe-2";
342
+ var maiTranscribeModels = /* @__PURE__ */ new Map([
343
+ ["mai-transcribe-2", { name: "MAI-Transcribe-2", supportsTimestamps: true }],
344
+ [
345
+ "mai-transcribe-1.5",
346
+ { name: "MAI-Transcribe-1.5", supportsTimestamps: false }
347
+ ]
348
+ ]);
349
+ function getMAITranscribeModel(modelId) {
350
+ return maiTranscribeModels.get(modelId.toLowerCase());
99
351
  }
100
352
 
101
353
  // src/azure-speech-transcription-model.ts
@@ -107,27 +359,28 @@ var AzureSpeechTranscriptionModel = class {
107
359
  this.provider = "azure.transcription";
108
360
  }
109
361
  async doGenerate(options, azureOptions = {}) {
110
- var _a, _b;
362
+ var _a, _b, _c;
111
363
  const timestamp = /* @__PURE__ */ new Date();
364
+ const maiModel = getMAITranscribeModel(this.modelId);
112
365
  const formData = new FormData();
113
366
  formData.append(
114
367
  "audio",
115
368
  new Blob(
116
369
  [
117
- typeof options.audio === "string" ? (0, import_provider_utils3.convertBase64ToUint8Array)(options.audio) : options.audio
370
+ typeof options.audio === "string" ? (0, import_provider_utils6.convertBase64ToUint8Array)(options.audio) : options.audio
118
371
  ],
119
372
  { type: options.mediaType }
120
373
  ),
121
- `audio.${(0, import_provider_utils3.mediaTypeToExtension)(options.mediaType)}`
374
+ `audio.${(0, import_provider_utils6.mediaTypeToExtension)(options.mediaType)}`
122
375
  );
123
376
  formData.append(
124
377
  "definition",
125
378
  JSON.stringify({
126
379
  enhancedMode: {
127
380
  enabled: true,
128
- model: isMAITranscribe2(this.modelId) ? "MAI-Transcribe-2" : this.modelId,
381
+ model: (_a = maiModel == null ? void 0 : maiModel.name) != null ? _a : this.modelId,
129
382
  modelOptions: {
130
- timestamps: (_a = azureOptions.timestamps) != null ? _a : "segment",
383
+ timestamps: (_b = azureOptions.timestamps) != null ? _b : (maiModel == null ? void 0 : maiModel.supportsTimestamps) === false ? void 0 : "segment",
131
384
  transcribeStyle: azureOptions.transcribeStyle
132
385
  }
133
386
  },
@@ -136,22 +389,22 @@ var AzureSpeechTranscriptionModel = class {
136
389
  phraseList: azureOptions.phraseList
137
390
  })
138
391
  );
139
- const { value, rawValue, responseHeaders } = await (0, import_provider_utils3.postFormDataToApi)({
392
+ const { value, rawValue, responseHeaders } = await (0, import_provider_utils6.postFormDataToApi)({
140
393
  url: this.config.url(),
141
- headers: (0, import_provider_utils3.combineHeaders)(this.config.headers(), options.headers),
394
+ headers: (0, import_provider_utils6.combineHeaders)(this.config.headers(), options.headers),
142
395
  formData,
143
396
  abortSignal: options.abortSignal,
144
397
  fetch: this.config.fetch,
145
- failedResponseHandler: (0, import_provider_utils3.createJsonErrorResponseHandler)({
146
- errorSchema: import_v43.z.union([
147
- import_v43.z.object({ error: import_v43.z.object({ message: import_v43.z.string() }) }),
148
- import_v43.z.object({ message: import_v43.z.string() })
398
+ failedResponseHandler: (0, import_provider_utils6.createJsonErrorResponseHandler)({
399
+ errorSchema: import_v46.z.union([
400
+ import_v46.z.object({ error: import_v46.z.object({ message: import_v46.z.string() }) }),
401
+ import_v46.z.object({ message: import_v46.z.string() })
149
402
  ]),
150
403
  errorToMessage: (data) => "error" in data ? data.error.message : data.message
151
404
  }),
152
- successfulResponseHandler: (0, import_provider_utils3.createJsonResponseHandler)(responseSchema)
405
+ successfulResponseHandler: (0, import_provider_utils6.createJsonResponseHandler)(responseSchema)
153
406
  });
154
- const phrases = (_b = value.phrases) != null ? _b : [];
407
+ const phrases = (_c = value.phrases) != null ? _c : [];
155
408
  const languages = new Set(
156
409
  phrases.flatMap(
157
410
  (phrase) => phrase.locale ? [phrase.locale.split("-")[0].toLowerCase()] : []
@@ -185,22 +438,22 @@ var AzureSpeechTranscriptionModel = class {
185
438
  };
186
439
  }
187
440
  };
188
- var responseSchema = import_v43.z.object({
189
- combinedPhrases: import_v43.z.array(import_v43.z.object({ text: import_v43.z.string() })),
190
- durationMilliseconds: import_v43.z.number().nullish(),
191
- phrases: import_v43.z.array(
192
- import_v43.z.object({
193
- text: import_v43.z.string(),
194
- offsetMilliseconds: import_v43.z.number().nullish(),
195
- durationMilliseconds: import_v43.z.number().nullish(),
196
- locale: import_v43.z.string().nullish(),
197
- speaker: import_v43.z.number().nullish(),
198
- confidence: import_v43.z.number().nullish(),
199
- words: import_v43.z.array(
200
- import_v43.z.object({
201
- text: import_v43.z.string(),
202
- offsetMilliseconds: import_v43.z.number().nullish(),
203
- durationMilliseconds: import_v43.z.number().nullish()
441
+ var responseSchema = import_v46.z.object({
442
+ combinedPhrases: import_v46.z.array(import_v46.z.object({ text: import_v46.z.string() })),
443
+ durationMilliseconds: import_v46.z.number().nullish(),
444
+ phrases: import_v46.z.array(
445
+ import_v46.z.object({
446
+ text: import_v46.z.string(),
447
+ offsetMilliseconds: import_v46.z.number().nullish(),
448
+ durationMilliseconds: import_v46.z.number().nullish(),
449
+ locale: import_v46.z.string().nullish(),
450
+ speaker: import_v46.z.number().nullish(),
451
+ confidence: import_v46.z.number().nullish(),
452
+ words: import_v46.z.array(
453
+ import_v46.z.object({
454
+ text: import_v46.z.string(),
455
+ offsetMilliseconds: import_v46.z.number().nullish(),
456
+ durationMilliseconds: import_v46.z.number().nullish()
204
457
  })
205
458
  ).nullish()
206
459
  })
@@ -208,7 +461,7 @@ var responseSchema = import_v43.z.object({
208
461
  });
209
462
 
210
463
  // src/version.ts
211
- var VERSION = true ? "3.0.130" : "0.0.0-test";
464
+ var VERSION = true ? "3.0.132" : "0.0.0-test";
212
465
 
213
466
  // src/azure-openai-provider.ts
214
467
  function getAzureOpenAIBaseURLInfo(baseURL) {
@@ -233,20 +486,20 @@ function createAzure(options = {}) {
233
486
  var _a;
234
487
  const tokenProvider = options.tokenProvider;
235
488
  if (options.apiKey && tokenProvider) {
236
- throw new import_provider.InvalidArgumentError({
489
+ throw new import_provider2.InvalidArgumentError({
237
490
  argument: "apiKey/tokenProvider",
238
491
  message: "Both apiKey and tokenProvider were provided. Please use only one authentication method."
239
492
  });
240
493
  }
241
494
  const getHeaders = (api = "openai") => {
242
495
  const authHeaders = tokenProvider ? {} : {
243
- [api === "speech" ? "Ocp-Apim-Subscription-Key" : "api-key"]: (0, import_provider_utils4.loadApiKey)({
496
+ [api === "speech" ? "Ocp-Apim-Subscription-Key" : "api-key"]: (0, import_provider_utils7.loadApiKey)({
244
497
  apiKey: options.apiKey,
245
498
  environmentVariableName: "AZURE_API_KEY",
246
499
  description: api === "speech" ? "Azure Speech" : "Azure OpenAI"
247
500
  })
248
501
  };
249
- return (0, import_provider_utils4.withUserAgentSuffix)(
502
+ return (0, import_provider_utils7.withUserAgentSuffix)(
250
503
  {
251
504
  ...authHeaders,
252
505
  ...options.headers
@@ -256,7 +509,7 @@ function createAzure(options = {}) {
256
509
  };
257
510
  const fetch = tokenProvider ? async (input, init) => {
258
511
  var _a2;
259
- const headers = (0, import_provider_utils4.normalizeHeaders)(init == null ? void 0 : init.headers);
512
+ const headers = (0, import_provider_utils7.normalizeHeaders)(init == null ? void 0 : init.headers);
260
513
  if (headers.authorization == null) {
261
514
  headers.authorization = `Bearer ${await tokenProvider()}`;
262
515
  }
@@ -265,12 +518,21 @@ function createAzure(options = {}) {
265
518
  headers
266
519
  });
267
520
  } : options.fetch;
268
- const getResourceName = () => (0, import_provider_utils4.loadSetting)({
269
- settingValue: options.resourceName,
270
- settingName: "resourceName",
271
- environmentVariableName: "AZURE_RESOURCE_NAME",
272
- description: "Azure OpenAI resource name"
273
- });
521
+ const getResourceName = () => {
522
+ const resourceName = (0, import_provider_utils7.loadSetting)({
523
+ settingValue: options.resourceName,
524
+ settingName: "resourceName",
525
+ environmentVariableName: "AZURE_RESOURCE_NAME",
526
+ description: "Azure OpenAI resource name"
527
+ });
528
+ if (!/^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/i.test(resourceName)) {
529
+ throw new import_provider2.InvalidArgumentError({
530
+ argument: "resourceName",
531
+ message: "Invalid Azure resource name. Expected a single DNS label (letters, digits, and hyphens). Use `baseURL` for custom endpoints."
532
+ });
533
+ }
534
+ return resourceName;
535
+ };
274
536
  const apiVersion = (_a = options.apiVersion) != null ? _a : "v1";
275
537
  const {
276
538
  isAzureOpenAI,
@@ -279,7 +541,7 @@ function createAzure(options = {}) {
279
541
  } = getAzureOpenAIBaseURLInfo(options.baseURL);
280
542
  const url = ({ path, modelId }) => {
281
543
  var _a2;
282
- const baseUrlPrefix = (0, import_provider_utils4.withoutTrailingSlash)(
544
+ const baseUrlPrefix = (0, import_provider_utils7.withoutTrailingSlash)(
283
545
  (_a2 = options.baseURL) != null ? _a2 : `https://${getResourceName()}.openai.azure.com/openai`
284
546
  );
285
547
  let fullUrl;
@@ -337,6 +599,10 @@ function createAzure(options = {}) {
337
599
  headers: getHeaders,
338
600
  fetch
339
601
  });
602
+ const speechBaseURL = () => {
603
+ var _a2;
604
+ return (_a2 = (0, import_provider_utils7.withoutTrailingSlash)(options.speechBaseURL)) != null ? _a2 : `https://${getResourceName()}.cognitiveservices.azure.com`;
605
+ };
340
606
  const createTranscriptionModel = (modelId) => new AzureTranscriptionModel(
341
607
  modelId,
342
608
  options,
@@ -347,20 +613,25 @@ function createAzure(options = {}) {
347
613
  fetch
348
614
  }),
349
615
  new AzureSpeechTranscriptionModel(modelId, {
350
- url: () => {
351
- var _a2;
352
- return `${(_a2 = (0, import_provider_utils4.withoutTrailingSlash)(options.speechBaseURL)) != null ? _a2 : `https://${getResourceName()}.cognitiveservices.azure.com`}/speechtotext/transcriptions:transcribe?api-version=2025-10-15`;
353
- },
616
+ url: () => `${speechBaseURL()}/speechtotext/transcriptions:transcribe?api-version=2025-10-15`,
617
+ headers: () => getHeaders("speech"),
618
+ fetch
619
+ })
620
+ );
621
+ const createSpeechModel = (modelId) => new AzureSpeechModel(
622
+ modelId,
623
+ new import_internal2.OpenAISpeechModel(modelId, {
624
+ provider: "azure.speech",
625
+ url,
626
+ headers: getHeaders,
627
+ fetch
628
+ }),
629
+ new AzureSpeechSpeechModel(modelId, {
630
+ url: () => `${speechBaseURL()}/tts/cognitiveservices/v1`,
354
631
  headers: () => getHeaders("speech"),
355
632
  fetch
356
633
  })
357
634
  );
358
- const createSpeechModel = (modelId) => new import_internal2.OpenAISpeechModel(modelId, {
359
- provider: "azure.speech",
360
- url,
361
- headers: getHeaders,
362
- fetch
363
- });
364
635
  const provider = function(deploymentId) {
365
636
  if (new.target) {
366
637
  throw new Error(
@@ -384,6 +655,7 @@ function createAzure(options = {}) {
384
655
  provider.transcription = createTranscriptionModel;
385
656
  provider.transcriptionModel = createTranscriptionModel;
386
657
  provider.speech = createSpeechModel;
658
+ provider.speechModel = createSpeechModel;
387
659
  provider.tools = azureOpenaiTools;
388
660
  return provider;
389
661
  }
@@ -419,14 +691,48 @@ var AzureTranscriptionModel = class {
419
691
  }
420
692
  async getOptions(providerOptions) {
421
693
  var _a;
422
- const options = await (0, import_provider_utils4.parseProviderOptions)({
694
+ const options = await (0, import_provider_utils7.parseProviderOptions)({
423
695
  provider: "azure",
424
696
  providerOptions,
425
697
  schema: azureTranscriptionModelOptions
426
698
  });
427
699
  return {
428
700
  ...options,
429
- api: (_a = options == null ? void 0 : options.api) != null ? _a : isMAITranscribe2(this.modelId) ? "speech" : "openai"
701
+ api: (_a = options == null ? void 0 : options.api) != null ? _a : getMAITranscribeModel(this.modelId) ? "speech" : "openai"
702
+ };
703
+ }
704
+ };
705
+ var AzureSpeechModel = class {
706
+ constructor(modelId, openai, speech) {
707
+ this.modelId = modelId;
708
+ this.openai = openai;
709
+ this.speech = speech;
710
+ this.specificationVersion = "v3";
711
+ this.provider = "azure.speech";
712
+ }
713
+ async doGenerate(options) {
714
+ var _a;
715
+ const { api, ...speechOptions } = (_a = await (0, import_provider_utils7.parseProviderOptions)({
716
+ provider: "azure",
717
+ providerOptions: options.providerOptions,
718
+ schema: azureSpeechModelOptions
719
+ })) != null ? _a : {};
720
+ if ((api != null ? api : getMAIVoiceModel(this.modelId) ? "speech" : "openai") === "speech") {
721
+ return this.speech.doGenerate(options, speechOptions);
722
+ }
723
+ const result = await this.openai.doGenerate(options);
724
+ return {
725
+ ...result,
726
+ warnings: [
727
+ ...result.warnings,
728
+ ...Object.entries(speechOptions).filter(([, value]) => value !== void 0).map(
729
+ ([key]) => ({
730
+ type: "unsupported",
731
+ feature: `providerOptions.azure.${key}`,
732
+ details: "This option requires the Azure Speech API."
733
+ })
734
+ )
735
+ ]
430
736
  };
431
737
  }
432
738
  };