@ai-sdk/azure 3.0.130 → 3.0.131
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +8 -0
- package/dist/index.d.mts +20 -6
- package/dist/index.d.ts +20 -6
- package/dist/index.js +380 -74
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +384 -64
- package/dist/index.mjs.map +1 -1
- package/docs/04-azure.mdx +104 -18
- package/package.json +1 -1
- package/src/azure-openai-provider.ts +101 -19
- package/src/azure-speech-model-options.ts +33 -0
- package/src/azure-speech-speech-model-options.ts +27 -0
- package/src/azure-speech-speech-model.ts +282 -0
- package/src/azure-speech-transcription-model.ts +6 -5
- package/src/azure-transcription-model-options.ts +13 -3
- package/src/index.ts +1 -0
package/dist/index.mjs
CHANGED
|
@@ -37,73 +37,339 @@ var azureOpenaiTools = {
|
|
|
37
37
|
webSearchPreview
|
|
38
38
|
};
|
|
39
39
|
|
|
40
|
-
// src/azure-speech-
|
|
40
|
+
// src/azure-speech-model-options.ts
|
|
41
|
+
import {
|
|
42
|
+
lazySchema as lazySchema2,
|
|
43
|
+
zodSchema as zodSchema2
|
|
44
|
+
} from "@ai-sdk/provider-utils";
|
|
45
|
+
import { z as z2 } from "zod/v4";
|
|
46
|
+
|
|
47
|
+
// src/azure-speech-speech-model-options.ts
|
|
48
|
+
import {
|
|
49
|
+
lazySchema,
|
|
50
|
+
zodSchema
|
|
51
|
+
} from "@ai-sdk/provider-utils";
|
|
52
|
+
import { z } from "zod/v4";
|
|
53
|
+
var azureSpeechSpeechModelOptionsShape = () => ({
|
|
54
|
+
/**
|
|
55
|
+
* Speaking style applied with `mstts:express-as`, e.g. `excited` or
|
|
56
|
+
* `whispering`. Supported styles vary by voice.
|
|
57
|
+
*/
|
|
58
|
+
style: z.string().min(1).optional(),
|
|
59
|
+
/**
|
|
60
|
+
* Intensity of `style`, from 0.01 to 2. Azure defaults to 1.
|
|
61
|
+
*/
|
|
62
|
+
styleDegree: z.number().min(0.01).max(2).optional()
|
|
63
|
+
});
|
|
64
|
+
var azureSpeechSpeechModelOptions = lazySchema(
|
|
65
|
+
() => zodSchema(z.strictObject(azureSpeechSpeechModelOptionsShape()))
|
|
66
|
+
);
|
|
67
|
+
|
|
68
|
+
// src/azure-speech-model-options.ts
|
|
69
|
+
var azureSpeechModelOptions = lazySchema2(
|
|
70
|
+
() => zodSchema2(
|
|
71
|
+
z2.strictObject({
|
|
72
|
+
/**
|
|
73
|
+
* API to use. Defaults to Speech for MAI-Voice models, OpenAI otherwise.
|
|
74
|
+
*/
|
|
75
|
+
api: z2.enum(["openai", "speech"]).optional(),
|
|
76
|
+
...azureSpeechSpeechModelOptionsShape()
|
|
77
|
+
})
|
|
78
|
+
)
|
|
79
|
+
);
|
|
80
|
+
var maiVoiceModels = /* @__PURE__ */ new Map([
|
|
81
|
+
["mai-voice-2-flash", "MAI-Voice-2-Flash"],
|
|
82
|
+
["mai-voice-2", "MAI-Voice-2"]
|
|
83
|
+
]);
|
|
84
|
+
function getMAIVoiceModel(modelId) {
|
|
85
|
+
return maiVoiceModels.get(modelId.toLowerCase());
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// src/azure-speech-speech-model.ts
|
|
89
|
+
import {
|
|
90
|
+
APICallError
|
|
91
|
+
} from "@ai-sdk/provider";
|
|
41
92
|
import {
|
|
42
93
|
combineHeaders,
|
|
94
|
+
createBinaryResponseHandler,
|
|
95
|
+
extractResponseHeaders,
|
|
96
|
+
postToApi,
|
|
97
|
+
safeParseJSON
|
|
98
|
+
} from "@ai-sdk/provider-utils";
|
|
99
|
+
import { z as z3 } from "zod/v4";
|
|
100
|
+
var DEFAULT_VOICE = "en-US-Harper";
|
|
101
|
+
var DEFAULT_VOICES = /* @__PURE__ */ new Map([
|
|
102
|
+
["de", "de-DE-Mia"],
|
|
103
|
+
["en", DEFAULT_VOICE],
|
|
104
|
+
["es", "es-MX-Valeria"],
|
|
105
|
+
["fr", "fr-FR-Soleil"],
|
|
106
|
+
["hi", "hi-IN-Kavya"],
|
|
107
|
+
["hu", "hu-HU-Lilla"],
|
|
108
|
+
["it", "it-IT-Rosa"],
|
|
109
|
+
["ko", "ko-KR-Haena"],
|
|
110
|
+
["nl", "nl-NL-Fleur"],
|
|
111
|
+
["pt", "pt-BR-Luana"],
|
|
112
|
+
["ro", "ro-RO-Elena"],
|
|
113
|
+
["ru", "ru-RU-Masha"],
|
|
114
|
+
["th", "th-TH-Krit"],
|
|
115
|
+
["tr", "tr-TR-Elif"],
|
|
116
|
+
["zh", "zh-CN-Mei"]
|
|
117
|
+
]);
|
|
118
|
+
var DEFAULT_OUTPUT_FORMAT = "audio-24khz-160kbitrate-mono-mp3";
|
|
119
|
+
var OUTPUT_FORMATS = /* @__PURE__ */ new Map([
|
|
120
|
+
["mp3", DEFAULT_OUTPUT_FORMAT],
|
|
121
|
+
["opus", "ogg-24khz-16bit-mono-opus"],
|
|
122
|
+
["pcm", "raw-24khz-16bit-mono-pcm"],
|
|
123
|
+
["wav", "riff-24khz-16bit-mono-pcm"]
|
|
124
|
+
]);
|
|
125
|
+
var NATIVE_OUTPUT_FORMAT = /^(?:amr|audio|g722|ogg|raw|riff|webm)-[a-z0-9-]+$/;
|
|
126
|
+
var AzureSpeechSpeechModel = class {
|
|
127
|
+
constructor(modelId, config) {
|
|
128
|
+
this.modelId = modelId;
|
|
129
|
+
this.config = config;
|
|
130
|
+
this.specificationVersion = "v3";
|
|
131
|
+
this.provider = "azure.speech";
|
|
132
|
+
}
|
|
133
|
+
async doGenerate(options, azureOptions = {}) {
|
|
134
|
+
var _a, _b, _c, _d, _e;
|
|
135
|
+
const currentDate = (_c = (_b = (_a = this.config._internal) == null ? void 0 : _a.currentDate) == null ? void 0 : _b.call(_a)) != null ? _c : /* @__PURE__ */ new Date();
|
|
136
|
+
const warnings = [];
|
|
137
|
+
const { style, styleDegree } = azureOptions;
|
|
138
|
+
if (options.instructions != null) {
|
|
139
|
+
warnings.push({
|
|
140
|
+
type: "unsupported",
|
|
141
|
+
feature: "instructions",
|
|
142
|
+
details: "Use providerOptions.azure.style to control speaking style."
|
|
143
|
+
});
|
|
144
|
+
}
|
|
145
|
+
const { voice, languageWarning } = resolveVoice(
|
|
146
|
+
options.voice,
|
|
147
|
+
options.language
|
|
148
|
+
);
|
|
149
|
+
if (languageWarning != null) {
|
|
150
|
+
warnings.push({
|
|
151
|
+
type: "unsupported",
|
|
152
|
+
feature: "language",
|
|
153
|
+
details: languageWarning
|
|
154
|
+
});
|
|
155
|
+
}
|
|
156
|
+
if (styleDegree != null && style == null) {
|
|
157
|
+
warnings.push({
|
|
158
|
+
type: "unsupported",
|
|
159
|
+
feature: "providerOptions.azure.styleDegree",
|
|
160
|
+
details: "styleDegree requires style."
|
|
161
|
+
});
|
|
162
|
+
}
|
|
163
|
+
let outputFormat = DEFAULT_OUTPUT_FORMAT;
|
|
164
|
+
if (options.outputFormat != null) {
|
|
165
|
+
const format = options.outputFormat.toLowerCase();
|
|
166
|
+
const shorthand = OUTPUT_FORMATS.get(format);
|
|
167
|
+
if (shorthand != null) {
|
|
168
|
+
outputFormat = shorthand;
|
|
169
|
+
} else if (NATIVE_OUTPUT_FORMAT.test(format)) {
|
|
170
|
+
outputFormat = format;
|
|
171
|
+
} else {
|
|
172
|
+
warnings.push({
|
|
173
|
+
type: "unsupported",
|
|
174
|
+
feature: "outputFormat",
|
|
175
|
+
details: `Unsupported output format: ${options.outputFormat}. Using mp3 instead.`
|
|
176
|
+
});
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
const ssml = buildSsml({
|
|
180
|
+
text: options.text,
|
|
181
|
+
voiceName: voice.includes(":") ? voice : `${voice}:${(_d = getMAIVoiceModel(this.modelId)) != null ? _d : this.modelId}`,
|
|
182
|
+
locale: (_e = /^([a-z]{2,3}-[a-z]{2,4})-/i.exec(voice)) == null ? void 0 : _e[1],
|
|
183
|
+
speed: options.speed,
|
|
184
|
+
style,
|
|
185
|
+
styleDegree: style != null ? styleDegree : void 0
|
|
186
|
+
});
|
|
187
|
+
const {
|
|
188
|
+
value: audio,
|
|
189
|
+
responseHeaders,
|
|
190
|
+
rawValue
|
|
191
|
+
} = await postToApi({
|
|
192
|
+
url: this.config.url(),
|
|
193
|
+
headers: combineHeaders(
|
|
194
|
+
this.config.headers(),
|
|
195
|
+
{
|
|
196
|
+
"Content-Type": "application/ssml+xml",
|
|
197
|
+
"X-Microsoft-OutputFormat": outputFormat
|
|
198
|
+
},
|
|
199
|
+
options.headers
|
|
200
|
+
),
|
|
201
|
+
body: { content: ssml, values: ssml },
|
|
202
|
+
failedResponseHandler,
|
|
203
|
+
successfulResponseHandler: createBinaryResponseHandler(),
|
|
204
|
+
abortSignal: options.abortSignal,
|
|
205
|
+
fetch: this.config.fetch
|
|
206
|
+
});
|
|
207
|
+
return {
|
|
208
|
+
audio,
|
|
209
|
+
warnings,
|
|
210
|
+
request: { body: ssml },
|
|
211
|
+
response: {
|
|
212
|
+
timestamp: currentDate,
|
|
213
|
+
modelId: this.modelId,
|
|
214
|
+
headers: responseHeaders,
|
|
215
|
+
body: rawValue
|
|
216
|
+
}
|
|
217
|
+
};
|
|
218
|
+
}
|
|
219
|
+
};
|
|
220
|
+
function buildSsml({
|
|
221
|
+
text,
|
|
222
|
+
voiceName,
|
|
223
|
+
locale = "en-US",
|
|
224
|
+
speed,
|
|
225
|
+
style,
|
|
226
|
+
styleDegree
|
|
227
|
+
}) {
|
|
228
|
+
let content = escapeXml(text);
|
|
229
|
+
if (speed != null) {
|
|
230
|
+
content = `<prosody rate="${speed}">${content}</prosody>`;
|
|
231
|
+
}
|
|
232
|
+
if (style != null) {
|
|
233
|
+
const degree = styleDegree != null ? ` styledegree="${styleDegree}"` : "";
|
|
234
|
+
content = `<mstts:express-as style="${escapeXml(style)}"${degree}>${content}</mstts:express-as>`;
|
|
235
|
+
}
|
|
236
|
+
return `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="http://www.w3.org/2001/mstts" xml:lang="${escapeXml(locale)}"><voice name="${escapeXml(voiceName)}">${content}</voice></speak>`;
|
|
237
|
+
}
|
|
238
|
+
var XML_ESCAPES = {
|
|
239
|
+
"&": "&",
|
|
240
|
+
"<": "<",
|
|
241
|
+
">": ">",
|
|
242
|
+
'"': """,
|
|
243
|
+
"'": "'"
|
|
244
|
+
};
|
|
245
|
+
function escapeXml(value) {
|
|
246
|
+
return value.replace(/[&<>"']/g, (char) => XML_ESCAPES[char]);
|
|
247
|
+
}
|
|
248
|
+
function resolveVoice(voice, language) {
|
|
249
|
+
var _a, _b;
|
|
250
|
+
const code = language ? language.split("-")[0].toLowerCase() : void 0;
|
|
251
|
+
if (voice == null) {
|
|
252
|
+
if (code == null) return { voice: DEFAULT_VOICE };
|
|
253
|
+
if (code === "auto") {
|
|
254
|
+
return {
|
|
255
|
+
voice: DEFAULT_VOICE,
|
|
256
|
+
languageWarning: `Automatic language detection is not supported. ${DEFAULT_VOICE} was used.`
|
|
257
|
+
};
|
|
258
|
+
}
|
|
259
|
+
const defaultVoice = DEFAULT_VOICES.get(code);
|
|
260
|
+
return defaultVoice != null ? { voice: defaultVoice } : {
|
|
261
|
+
voice: DEFAULT_VOICE,
|
|
262
|
+
languageWarning: `No default MAI voice for language "${language}". ${DEFAULT_VOICE} was used.`
|
|
263
|
+
};
|
|
264
|
+
}
|
|
265
|
+
const voiceLanguage = (_b = (_a = /^([a-z]{2,3})-[a-z]{2,4}-/i.exec(voice)) == null ? void 0 : _a[1]) == null ? void 0 : _b.toLowerCase();
|
|
266
|
+
return code != null && code !== "auto" && voiceLanguage != null && code !== voiceLanguage ? {
|
|
267
|
+
voice,
|
|
268
|
+
languageWarning: `The voice ${voice} selects the language. Language "${language}" was ignored.`
|
|
269
|
+
} : { voice };
|
|
270
|
+
}
|
|
271
|
+
var errorSchema = z3.object({
|
|
272
|
+
error: z3.object({ message: z3.string() })
|
|
273
|
+
});
|
|
274
|
+
var failedResponseHandler = async ({
|
|
275
|
+
response,
|
|
276
|
+
url,
|
|
277
|
+
requestBodyValues
|
|
278
|
+
}) => {
|
|
279
|
+
const responseHeaders = extractResponseHeaders(response);
|
|
280
|
+
const responseBody = await response.text();
|
|
281
|
+
const parsed = await safeParseJSON({
|
|
282
|
+
text: responseBody,
|
|
283
|
+
schema: errorSchema
|
|
284
|
+
});
|
|
285
|
+
const message = parsed.success ? parsed.value.error.message : response.status === 400 ? "Azure Speech request failed with status 400. Check the voice name, style, and output format." : `Azure Speech request failed with status ${response.status}.`;
|
|
286
|
+
return {
|
|
287
|
+
responseHeaders,
|
|
288
|
+
value: new APICallError({
|
|
289
|
+
message,
|
|
290
|
+
url,
|
|
291
|
+
requestBodyValues,
|
|
292
|
+
statusCode: response.status,
|
|
293
|
+
responseHeaders,
|
|
294
|
+
responseBody
|
|
295
|
+
})
|
|
296
|
+
};
|
|
297
|
+
};
|
|
298
|
+
|
|
299
|
+
// src/azure-speech-transcription-model.ts
|
|
300
|
+
import {
|
|
301
|
+
combineHeaders as combineHeaders2,
|
|
43
302
|
convertBase64ToUint8Array,
|
|
44
303
|
createJsonErrorResponseHandler,
|
|
45
304
|
createJsonResponseHandler,
|
|
46
305
|
mediaTypeToExtension,
|
|
47
306
|
postFormDataToApi
|
|
48
307
|
} from "@ai-sdk/provider-utils";
|
|
49
|
-
import { z as
|
|
308
|
+
import { z as z6 } from "zod/v4";
|
|
50
309
|
|
|
51
310
|
// src/azure-transcription-model-options.ts
|
|
52
311
|
import {
|
|
53
|
-
lazySchema as
|
|
54
|
-
zodSchema as
|
|
312
|
+
lazySchema as lazySchema4,
|
|
313
|
+
zodSchema as zodSchema4
|
|
55
314
|
} from "@ai-sdk/provider-utils";
|
|
56
|
-
import { z as
|
|
315
|
+
import { z as z5 } from "zod/v4";
|
|
57
316
|
|
|
58
317
|
// src/azure-speech-transcription-model-options.ts
|
|
59
318
|
import {
|
|
60
|
-
lazySchema,
|
|
61
|
-
zodSchema
|
|
319
|
+
lazySchema as lazySchema3,
|
|
320
|
+
zodSchema as zodSchema3
|
|
62
321
|
} from "@ai-sdk/provider-utils";
|
|
63
|
-
import { z } from "zod/v4";
|
|
322
|
+
import { z as z4 } from "zod/v4";
|
|
64
323
|
var azureSpeechTranscriptionModelOptionsShape = () => ({
|
|
65
324
|
/**
|
|
66
325
|
* Timing granularity. Defaults to `segment` so that transcription results
|
|
67
326
|
* include timed segments (Azure's own default is `none`).
|
|
68
327
|
*/
|
|
69
|
-
timestamps:
|
|
328
|
+
timestamps: z4.enum(["word", "segment", "none"]).optional(),
|
|
70
329
|
/**
|
|
71
330
|
* `verbatim` keeps fillers and false starts, `clean` removes them.
|
|
72
331
|
* Azure defaults to `verbatim`.
|
|
73
332
|
*/
|
|
74
|
-
transcribeStyle:
|
|
333
|
+
transcribeStyle: z4.enum(["verbatim", "clean"]).optional(),
|
|
75
334
|
/**
|
|
76
335
|
* Forces a single language, e.g. `['en']`. Omit for automatic language
|
|
77
336
|
* detection and code switching.
|
|
78
337
|
*/
|
|
79
|
-
locales:
|
|
338
|
+
locales: z4.array(z4.string()).length(1).optional(),
|
|
80
339
|
/**
|
|
81
340
|
* Speaker diarization. Speaker IDs are available in provider metadata.
|
|
82
341
|
*/
|
|
83
|
-
diarization:
|
|
342
|
+
diarization: z4.strictObject({ enabled: z4.boolean() }).optional(),
|
|
84
343
|
/**
|
|
85
344
|
* Keyword biasing for names and domain terminology.
|
|
86
345
|
*/
|
|
87
|
-
phraseList:
|
|
346
|
+
phraseList: z4.strictObject({ phrases: z4.array(z4.string()) }).optional()
|
|
88
347
|
});
|
|
89
|
-
var azureSpeechTranscriptionModelOptions =
|
|
90
|
-
() =>
|
|
348
|
+
var azureSpeechTranscriptionModelOptions = lazySchema3(
|
|
349
|
+
() => zodSchema3(z4.strictObject(azureSpeechTranscriptionModelOptionsShape()))
|
|
91
350
|
);
|
|
92
351
|
|
|
93
352
|
// src/azure-transcription-model-options.ts
|
|
94
|
-
var azureTranscriptionModelOptions =
|
|
95
|
-
() =>
|
|
96
|
-
|
|
353
|
+
var azureTranscriptionModelOptions = lazySchema4(
|
|
354
|
+
() => zodSchema4(
|
|
355
|
+
z5.strictObject({
|
|
97
356
|
/**
|
|
98
|
-
* API to use. Defaults to Speech for MAI-Transcribe
|
|
357
|
+
* API to use. Defaults to Speech for MAI-Transcribe models, OpenAI otherwise.
|
|
99
358
|
*/
|
|
100
|
-
api:
|
|
359
|
+
api: z5.enum(["openai", "speech"]).optional(),
|
|
101
360
|
...azureSpeechTranscriptionModelOptionsShape()
|
|
102
361
|
})
|
|
103
362
|
)
|
|
104
363
|
);
|
|
105
|
-
|
|
106
|
-
|
|
364
|
+
var maiTranscribeModels = /* @__PURE__ */ new Map([
|
|
365
|
+
["mai-transcribe-2", { name: "MAI-Transcribe-2", supportsTimestamps: true }],
|
|
366
|
+
[
|
|
367
|
+
"mai-transcribe-1.5",
|
|
368
|
+
{ name: "MAI-Transcribe-1.5", supportsTimestamps: false }
|
|
369
|
+
]
|
|
370
|
+
]);
|
|
371
|
+
function getMAITranscribeModel(modelId) {
|
|
372
|
+
return maiTranscribeModels.get(modelId.toLowerCase());
|
|
107
373
|
}
|
|
108
374
|
|
|
109
375
|
// src/azure-speech-transcription-model.ts
|
|
@@ -115,8 +381,9 @@ var AzureSpeechTranscriptionModel = class {
|
|
|
115
381
|
this.provider = "azure.transcription";
|
|
116
382
|
}
|
|
117
383
|
async doGenerate(options, azureOptions = {}) {
|
|
118
|
-
var _a, _b;
|
|
384
|
+
var _a, _b, _c;
|
|
119
385
|
const timestamp = /* @__PURE__ */ new Date();
|
|
386
|
+
const maiModel = getMAITranscribeModel(this.modelId);
|
|
120
387
|
const formData = new FormData();
|
|
121
388
|
formData.append(
|
|
122
389
|
"audio",
|
|
@@ -133,9 +400,9 @@ var AzureSpeechTranscriptionModel = class {
|
|
|
133
400
|
JSON.stringify({
|
|
134
401
|
enhancedMode: {
|
|
135
402
|
enabled: true,
|
|
136
|
-
model:
|
|
403
|
+
model: (_a = maiModel == null ? void 0 : maiModel.name) != null ? _a : this.modelId,
|
|
137
404
|
modelOptions: {
|
|
138
|
-
timestamps: (
|
|
405
|
+
timestamps: (_b = azureOptions.timestamps) != null ? _b : (maiModel == null ? void 0 : maiModel.supportsTimestamps) === false ? void 0 : "segment",
|
|
139
406
|
transcribeStyle: azureOptions.transcribeStyle
|
|
140
407
|
}
|
|
141
408
|
},
|
|
@@ -146,20 +413,20 @@ var AzureSpeechTranscriptionModel = class {
|
|
|
146
413
|
);
|
|
147
414
|
const { value, rawValue, responseHeaders } = await postFormDataToApi({
|
|
148
415
|
url: this.config.url(),
|
|
149
|
-
headers:
|
|
416
|
+
headers: combineHeaders2(this.config.headers(), options.headers),
|
|
150
417
|
formData,
|
|
151
418
|
abortSignal: options.abortSignal,
|
|
152
419
|
fetch: this.config.fetch,
|
|
153
420
|
failedResponseHandler: createJsonErrorResponseHandler({
|
|
154
|
-
errorSchema:
|
|
155
|
-
|
|
156
|
-
|
|
421
|
+
errorSchema: z6.union([
|
|
422
|
+
z6.object({ error: z6.object({ message: z6.string() }) }),
|
|
423
|
+
z6.object({ message: z6.string() })
|
|
157
424
|
]),
|
|
158
425
|
errorToMessage: (data) => "error" in data ? data.error.message : data.message
|
|
159
426
|
}),
|
|
160
427
|
successfulResponseHandler: createJsonResponseHandler(responseSchema)
|
|
161
428
|
});
|
|
162
|
-
const phrases = (
|
|
429
|
+
const phrases = (_c = value.phrases) != null ? _c : [];
|
|
163
430
|
const languages = new Set(
|
|
164
431
|
phrases.flatMap(
|
|
165
432
|
(phrase) => phrase.locale ? [phrase.locale.split("-")[0].toLowerCase()] : []
|
|
@@ -193,22 +460,22 @@ var AzureSpeechTranscriptionModel = class {
|
|
|
193
460
|
};
|
|
194
461
|
}
|
|
195
462
|
};
|
|
196
|
-
var responseSchema =
|
|
197
|
-
combinedPhrases:
|
|
198
|
-
durationMilliseconds:
|
|
199
|
-
phrases:
|
|
200
|
-
|
|
201
|
-
text:
|
|
202
|
-
offsetMilliseconds:
|
|
203
|
-
durationMilliseconds:
|
|
204
|
-
locale:
|
|
205
|
-
speaker:
|
|
206
|
-
confidence:
|
|
207
|
-
words:
|
|
208
|
-
|
|
209
|
-
text:
|
|
210
|
-
offsetMilliseconds:
|
|
211
|
-
durationMilliseconds:
|
|
463
|
+
var responseSchema = z6.object({
|
|
464
|
+
combinedPhrases: z6.array(z6.object({ text: z6.string() })),
|
|
465
|
+
durationMilliseconds: z6.number().nullish(),
|
|
466
|
+
phrases: z6.array(
|
|
467
|
+
z6.object({
|
|
468
|
+
text: z6.string(),
|
|
469
|
+
offsetMilliseconds: z6.number().nullish(),
|
|
470
|
+
durationMilliseconds: z6.number().nullish(),
|
|
471
|
+
locale: z6.string().nullish(),
|
|
472
|
+
speaker: z6.number().nullish(),
|
|
473
|
+
confidence: z6.number().nullish(),
|
|
474
|
+
words: z6.array(
|
|
475
|
+
z6.object({
|
|
476
|
+
text: z6.string(),
|
|
477
|
+
offsetMilliseconds: z6.number().nullish(),
|
|
478
|
+
durationMilliseconds: z6.number().nullish()
|
|
212
479
|
})
|
|
213
480
|
).nullish()
|
|
214
481
|
})
|
|
@@ -216,7 +483,7 @@ var responseSchema = z3.object({
|
|
|
216
483
|
});
|
|
217
484
|
|
|
218
485
|
// src/version.ts
|
|
219
|
-
var VERSION = true ? "3.0.
|
|
486
|
+
var VERSION = true ? "3.0.131" : "0.0.0-test";
|
|
220
487
|
|
|
221
488
|
// src/azure-openai-provider.ts
|
|
222
489
|
function getAzureOpenAIBaseURLInfo(baseURL) {
|
|
@@ -273,12 +540,21 @@ function createAzure(options = {}) {
|
|
|
273
540
|
headers
|
|
274
541
|
});
|
|
275
542
|
} : options.fetch;
|
|
276
|
-
const getResourceName = () =>
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
543
|
+
const getResourceName = () => {
|
|
544
|
+
const resourceName = loadSetting({
|
|
545
|
+
settingValue: options.resourceName,
|
|
546
|
+
settingName: "resourceName",
|
|
547
|
+
environmentVariableName: "AZURE_RESOURCE_NAME",
|
|
548
|
+
description: "Azure OpenAI resource name"
|
|
549
|
+
});
|
|
550
|
+
if (!/^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/i.test(resourceName)) {
|
|
551
|
+
throw new InvalidArgumentError({
|
|
552
|
+
argument: "resourceName",
|
|
553
|
+
message: "Invalid Azure resource name. Expected a single DNS label (letters, digits, and hyphens). Use `baseURL` for custom endpoints."
|
|
554
|
+
});
|
|
555
|
+
}
|
|
556
|
+
return resourceName;
|
|
557
|
+
};
|
|
282
558
|
const apiVersion = (_a = options.apiVersion) != null ? _a : "v1";
|
|
283
559
|
const {
|
|
284
560
|
isAzureOpenAI,
|
|
@@ -345,6 +621,10 @@ function createAzure(options = {}) {
|
|
|
345
621
|
headers: getHeaders,
|
|
346
622
|
fetch
|
|
347
623
|
});
|
|
624
|
+
const speechBaseURL = () => {
|
|
625
|
+
var _a2;
|
|
626
|
+
return (_a2 = withoutTrailingSlash(options.speechBaseURL)) != null ? _a2 : `https://${getResourceName()}.cognitiveservices.azure.com`;
|
|
627
|
+
};
|
|
348
628
|
const createTranscriptionModel = (modelId) => new AzureTranscriptionModel(
|
|
349
629
|
modelId,
|
|
350
630
|
options,
|
|
@@ -355,20 +635,25 @@ function createAzure(options = {}) {
|
|
|
355
635
|
fetch
|
|
356
636
|
}),
|
|
357
637
|
new AzureSpeechTranscriptionModel(modelId, {
|
|
358
|
-
url: () => {
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
638
|
+
url: () => `${speechBaseURL()}/speechtotext/transcriptions:transcribe?api-version=2025-10-15`,
|
|
639
|
+
headers: () => getHeaders("speech"),
|
|
640
|
+
fetch
|
|
641
|
+
})
|
|
642
|
+
);
|
|
643
|
+
const createSpeechModel = (modelId) => new AzureSpeechModel(
|
|
644
|
+
modelId,
|
|
645
|
+
new OpenAISpeechModel(modelId, {
|
|
646
|
+
provider: "azure.speech",
|
|
647
|
+
url,
|
|
648
|
+
headers: getHeaders,
|
|
649
|
+
fetch
|
|
650
|
+
}),
|
|
651
|
+
new AzureSpeechSpeechModel(modelId, {
|
|
652
|
+
url: () => `${speechBaseURL()}/tts/cognitiveservices/v1`,
|
|
362
653
|
headers: () => getHeaders("speech"),
|
|
363
654
|
fetch
|
|
364
655
|
})
|
|
365
656
|
);
|
|
366
|
-
const createSpeechModel = (modelId) => new OpenAISpeechModel(modelId, {
|
|
367
|
-
provider: "azure.speech",
|
|
368
|
-
url,
|
|
369
|
-
headers: getHeaders,
|
|
370
|
-
fetch
|
|
371
|
-
});
|
|
372
657
|
const provider = function(deploymentId) {
|
|
373
658
|
if (new.target) {
|
|
374
659
|
throw new Error(
|
|
@@ -392,6 +677,7 @@ function createAzure(options = {}) {
|
|
|
392
677
|
provider.transcription = createTranscriptionModel;
|
|
393
678
|
provider.transcriptionModel = createTranscriptionModel;
|
|
394
679
|
provider.speech = createSpeechModel;
|
|
680
|
+
provider.speechModel = createSpeechModel;
|
|
395
681
|
provider.tools = azureOpenaiTools;
|
|
396
682
|
return provider;
|
|
397
683
|
}
|
|
@@ -434,7 +720,41 @@ var AzureTranscriptionModel = class {
|
|
|
434
720
|
});
|
|
435
721
|
return {
|
|
436
722
|
...options,
|
|
437
|
-
api: (_a = options == null ? void 0 : options.api) != null ? _a :
|
|
723
|
+
api: (_a = options == null ? void 0 : options.api) != null ? _a : getMAITranscribeModel(this.modelId) ? "speech" : "openai"
|
|
724
|
+
};
|
|
725
|
+
}
|
|
726
|
+
};
|
|
727
|
+
var AzureSpeechModel = class {
|
|
728
|
+
constructor(modelId, openai, speech) {
|
|
729
|
+
this.modelId = modelId;
|
|
730
|
+
this.openai = openai;
|
|
731
|
+
this.speech = speech;
|
|
732
|
+
this.specificationVersion = "v3";
|
|
733
|
+
this.provider = "azure.speech";
|
|
734
|
+
}
|
|
735
|
+
async doGenerate(options) {
|
|
736
|
+
var _a;
|
|
737
|
+
const { api, ...speechOptions } = (_a = await parseProviderOptions({
|
|
738
|
+
provider: "azure",
|
|
739
|
+
providerOptions: options.providerOptions,
|
|
740
|
+
schema: azureSpeechModelOptions
|
|
741
|
+
})) != null ? _a : {};
|
|
742
|
+
if ((api != null ? api : getMAIVoiceModel(this.modelId) ? "speech" : "openai") === "speech") {
|
|
743
|
+
return this.speech.doGenerate(options, speechOptions);
|
|
744
|
+
}
|
|
745
|
+
const result = await this.openai.doGenerate(options);
|
|
746
|
+
return {
|
|
747
|
+
...result,
|
|
748
|
+
warnings: [
|
|
749
|
+
...result.warnings,
|
|
750
|
+
...Object.entries(speechOptions).filter(([, value]) => value !== void 0).map(
|
|
751
|
+
([key]) => ({
|
|
752
|
+
type: "unsupported",
|
|
753
|
+
feature: `providerOptions.azure.${key}`,
|
|
754
|
+
details: "This option requires the Azure Speech API."
|
|
755
|
+
})
|
|
756
|
+
)
|
|
757
|
+
]
|
|
438
758
|
};
|
|
439
759
|
}
|
|
440
760
|
};
|