ssml-builder-js 2.13.0 → 2.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -2
- package/dist/{chunk-LCTE26LS.mjs → chunk-BDY2Q2JL.mjs} +2 -2
- package/dist/{chunk-25LOR4AJ.mjs → chunk-FXUM45ZY.mjs} +402 -169
- package/dist/chunk-FXUM45ZY.mjs.map +1 -0
- package/dist/{chunk-CZ2F3TET.mjs → chunk-WXFLUCLR.mjs} +227 -9
- package/dist/chunk-WXFLUCLR.mjs.map +1 -0
- package/dist/core.d.mts +65 -17
- package/dist/core.d.ts +65 -17
- package/dist/core.js +401 -166
- package/dist/core.js.map +1 -1
- package/dist/core.mjs +5 -1
- package/dist/elements.js +101 -8
- package/dist/elements.js.map +1 -1
- package/dist/elements.mjs +2 -2
- package/dist/{index.d-Xbr6ZWnK.d.mts → index.d-8BvkB9gz.d.mts} +40 -2
- package/dist/{index.d-Xbr6ZWnK.d.ts → index.d-8BvkB9gz.d.ts} +40 -2
- package/dist/index.d.mts +170 -18
- package/dist/index.d.ts +170 -18
- package/dist/index.js +1512 -489
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +612 -47
- package/dist/index.mjs.map +1 -1
- package/dist/react.d.mts +15 -2
- package/dist/react.d.ts +15 -2
- package/dist/react.js +204 -23
- package/dist/react.js.map +1 -1
- package/dist/react.mjs +105 -17
- package/dist/react.mjs.map +1 -1
- package/package.json +1 -1
- package/dist/chunk-25LOR4AJ.mjs.map +0 -1
- package/dist/chunk-CZ2F3TET.mjs.map +0 -1
- /package/dist/{chunk-LCTE26LS.mjs.map → chunk-BDY2Q2JL.mjs.map} +0 -0
package/dist/index.mjs
CHANGED
|
@@ -2,11 +2,13 @@ import {
|
|
|
2
2
|
areAzureLanguagesEquivalent,
|
|
3
3
|
buildPartialSsml,
|
|
4
4
|
buildSsml,
|
|
5
|
+
createAzureUrlValidatorRunner,
|
|
5
6
|
extractSsmlText,
|
|
6
7
|
extractSsmlTranslatableText,
|
|
7
8
|
fromPlainTextToSsml,
|
|
8
9
|
getAzureVoiceCatalogMetadata,
|
|
9
10
|
getBuiltInVoiceCatalogMetadata,
|
|
11
|
+
getSsmlSourceMap,
|
|
10
12
|
isValidAzureAudioDuration,
|
|
11
13
|
mapSsmlTextNodes,
|
|
12
14
|
normalizeAzureLanguage,
|
|
@@ -15,10 +17,11 @@ import {
|
|
|
15
17
|
validateAzureSsml,
|
|
16
18
|
validateSsml,
|
|
17
19
|
validateSsmlStructureIntegrity
|
|
18
|
-
} from "./chunk-
|
|
20
|
+
} from "./chunk-FXUM45ZY.mjs";
|
|
19
21
|
import {
|
|
22
|
+
getSsmlSourceMap as getSsmlSourceMap2,
|
|
20
23
|
validateAzureSsml as validateAzureSsml2
|
|
21
|
-
} from "./chunk-
|
|
24
|
+
} from "./chunk-WXFLUCLR.mjs";
|
|
22
25
|
import {
|
|
23
26
|
__privateAdd,
|
|
24
27
|
__privateGet,
|
|
@@ -29,6 +32,7 @@ import {
|
|
|
29
32
|
var AzureTtsError = class extends Error {
|
|
30
33
|
constructor(status, statusText, responseBody, requestId) {
|
|
31
34
|
super(`Azure TTS request failed: ${status} ${statusText}`);
|
|
35
|
+
this.kind = "azure-api-error";
|
|
32
36
|
this.name = "AzureTtsError";
|
|
33
37
|
this.status = status;
|
|
34
38
|
this.statusText = statusText;
|
|
@@ -44,6 +48,44 @@ var AzureTtsSdkError = class extends AzureTtsError {
|
|
|
44
48
|
this.errorDetails = errorDetails;
|
|
45
49
|
}
|
|
46
50
|
};
|
|
51
|
+
var SynthesisCancelledError = class extends Error {
|
|
52
|
+
constructor(message = "Speech synthesis was cancelled.") {
|
|
53
|
+
super(message);
|
|
54
|
+
this.kind = "cancelled";
|
|
55
|
+
this.name = "SynthesisCancelledError";
|
|
56
|
+
}
|
|
57
|
+
};
|
|
58
|
+
var SynthesisTimeoutError = class extends Error {
|
|
59
|
+
constructor(message) {
|
|
60
|
+
super(message);
|
|
61
|
+
this.kind = "timeout";
|
|
62
|
+
this.name = "SynthesisTimeoutError";
|
|
63
|
+
}
|
|
64
|
+
};
|
|
65
|
+
var MergeError = class extends Error {
|
|
66
|
+
constructor(message, cause) {
|
|
67
|
+
super(message);
|
|
68
|
+
this.kind = "merge-error";
|
|
69
|
+
this.name = "MergeError";
|
|
70
|
+
this.cause = cause;
|
|
71
|
+
}
|
|
72
|
+
};
|
|
73
|
+
var UnsupportedMergeFormatError = class extends Error {
|
|
74
|
+
constructor(format) {
|
|
75
|
+
super(`Audio format "${format}" cannot be safely concatenated; container re-multiplexing is required.`);
|
|
76
|
+
this.kind = "unsupported-format-error";
|
|
77
|
+
this.name = "UnsupportedMergeFormatError";
|
|
78
|
+
this.format = format;
|
|
79
|
+
}
|
|
80
|
+
};
|
|
81
|
+
function toSynthesisError(error) {
|
|
82
|
+
if (error instanceof AzureTtsError || error instanceof MergeError || error instanceof UnsupportedMergeFormatError || error instanceof SynthesisCancelledError || error instanceof SynthesisTimeoutError)
|
|
83
|
+
return error;
|
|
84
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
85
|
+
if (/cancel|abort/i.test(message)) return new SynthesisCancelledError(message);
|
|
86
|
+
if (/tim(?:e|ed) ?out/i.test(message)) return new SynthesisTimeoutError(message);
|
|
87
|
+
return createSpeechSdkError(error);
|
|
88
|
+
}
|
|
47
89
|
function createSpeechSdkError(error) {
|
|
48
90
|
const message = error instanceof Error ? error.message : String(error);
|
|
49
91
|
return new AzureTtsSdkError(message);
|
|
@@ -52,9 +94,6 @@ function createSpeechSdkError(error) {
|
|
|
52
94
|
// packages/azure-tts-client/src/synthesis.ts
|
|
53
95
|
import * as SpeechSDK2 from "microsoft-cognitiveservices-speech-sdk";
|
|
54
96
|
|
|
55
|
-
// packages/azure-tts-client/src/speechConfig.ts
|
|
56
|
-
import { SpeechConfig } from "microsoft-cognitiveservices-speech-sdk";
|
|
57
|
-
|
|
58
97
|
// packages/azure-tts-client/src/outputFormats.ts
|
|
59
98
|
import * as SpeechSDK from "microsoft-cognitiveservices-speech-sdk";
|
|
60
99
|
var DEFAULT_OUTPUT_FORMAT = "audio-16khz-128kbitrate-mono-mp3";
|
|
@@ -99,6 +138,14 @@ var OUTPUT_FORMATS = {
|
|
|
99
138
|
"amr-wb-16000hz": SpeechSDK.SpeechSynthesisOutputFormat.AmrWb16000Hz,
|
|
100
139
|
"g722-16khz-64kbps": SpeechSDK.SpeechSynthesisOutputFormat.G72216Khz64Kbps
|
|
101
140
|
};
|
|
141
|
+
function resolveMimeType(outputFormat) {
|
|
142
|
+
if (/(?:wav|wave|riff)/i.test(outputFormat)) return "audio/wav";
|
|
143
|
+
if (/(?:mp3|mpeg)/i.test(outputFormat)) return "audio/mpeg";
|
|
144
|
+
if (/ogg/i.test(outputFormat)) return "audio/ogg";
|
|
145
|
+
if (/webm/i.test(outputFormat)) return "audio/webm";
|
|
146
|
+
if (/raw/i.test(outputFormat)) return "audio/L16";
|
|
147
|
+
return "application/octet-stream";
|
|
148
|
+
}
|
|
102
149
|
function resolveOutputFormat(outputFormat) {
|
|
103
150
|
const resolvedFormat = OUTPUT_FORMATS[outputFormat];
|
|
104
151
|
if (resolvedFormat === void 0) {
|
|
@@ -108,6 +155,7 @@ function resolveOutputFormat(outputFormat) {
|
|
|
108
155
|
}
|
|
109
156
|
|
|
110
157
|
// packages/azure-tts-client/src/speechConfig.ts
|
|
158
|
+
import { SpeechConfig } from "microsoft-cognitiveservices-speech-sdk";
|
|
111
159
|
function resolveEndpoint(config) {
|
|
112
160
|
const endpoint = config.endpoint?.trim() || "https://{region}.tts.speech.microsoft.com/cognitiveservices/v1";
|
|
113
161
|
return endpoint.replace(/\{region\}/g, encodeURIComponent(config.region));
|
|
@@ -121,6 +169,156 @@ function createSpeechConfig(config) {
|
|
|
121
169
|
}
|
|
122
170
|
|
|
123
171
|
// packages/azure-tts-client/src/synthesis.ts
|
|
172
|
+
function ascii(bytes, offset, value) {
|
|
173
|
+
return [...value].every((character, index) => bytes[offset + index] === character.charCodeAt(0));
|
|
174
|
+
}
|
|
175
|
+
function readUint32(bytes, offset) {
|
|
176
|
+
return new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength).getUint32(offset, true);
|
|
177
|
+
}
|
|
178
|
+
function parseWav(buffer) {
|
|
179
|
+
const bytes = new Uint8Array(buffer);
|
|
180
|
+
if (bytes.byteLength < 12 || !ascii(bytes, 0, "RIFF") || !ascii(bytes, 8, "WAVE")) {
|
|
181
|
+
throw new Error("Invalid WAV/RIFF audio buffer.");
|
|
182
|
+
}
|
|
183
|
+
const chunks = [];
|
|
184
|
+
const dataParts = [];
|
|
185
|
+
let format;
|
|
186
|
+
let offset = 12;
|
|
187
|
+
while (offset < bytes.byteLength) {
|
|
188
|
+
if (offset + 8 > bytes.byteLength) throw new Error("Invalid WAV chunk header.");
|
|
189
|
+
const id = String.fromCharCode(...bytes.slice(offset, offset + 4));
|
|
190
|
+
const size = readUint32(bytes, offset + 4);
|
|
191
|
+
const dataStart = offset + 8;
|
|
192
|
+
const dataEnd = dataStart + size;
|
|
193
|
+
if (dataEnd > bytes.byteLength) throw new Error(`WAV chunk "${id}" exceeds the audio buffer.`);
|
|
194
|
+
const data2 = bytes.slice(dataStart, dataEnd);
|
|
195
|
+
chunks.push({ id, data: data2 });
|
|
196
|
+
if (id === "fmt ") format ?? (format = data2);
|
|
197
|
+
if (id === "data") dataParts.push(data2);
|
|
198
|
+
offset = dataEnd + (size & 1);
|
|
199
|
+
if (offset > bytes.byteLength) throw new Error("Invalid WAV chunk padding.");
|
|
200
|
+
}
|
|
201
|
+
if (!format || dataParts.length === 0) throw new Error("WAV audio must contain fmt and data chunks.");
|
|
202
|
+
const dataLength = dataParts.reduce((total, part) => total + part.byteLength, 0);
|
|
203
|
+
const data = new Uint8Array(dataLength);
|
|
204
|
+
let dataOffset = 0;
|
|
205
|
+
for (const part of dataParts) {
|
|
206
|
+
data.set(part, dataOffset);
|
|
207
|
+
dataOffset += part.byteLength;
|
|
208
|
+
}
|
|
209
|
+
return { chunks, data, format };
|
|
210
|
+
}
|
|
211
|
+
function writeUint32(target, offset, value) {
|
|
212
|
+
new DataView(target.buffer).setUint32(offset, value, true);
|
|
213
|
+
}
|
|
214
|
+
function writeChunk(target, offset, id, data) {
|
|
215
|
+
for (let index = 0; index < 4; index += 1) target[offset + index] = id.charCodeAt(index) ?? 0;
|
|
216
|
+
writeUint32(target, offset + 4, data.byteLength);
|
|
217
|
+
target.set(data, offset + 8);
|
|
218
|
+
const end = offset + 8 + data.byteLength;
|
|
219
|
+
if (data.byteLength & 1) target[end] = 0;
|
|
220
|
+
return end + (data.byteLength & 1);
|
|
221
|
+
}
|
|
222
|
+
function mergeWavBuffers(buffers) {
|
|
223
|
+
if (buffers.length === 0) return new ArrayBuffer(0);
|
|
224
|
+
const parsed = buffers.map(parseWav);
|
|
225
|
+
const first = parsed[0];
|
|
226
|
+
if (!first) throw new Error("At least one WAV buffer is required.");
|
|
227
|
+
if (parsed.some(
|
|
228
|
+
(item) => item.format.length !== first.format.length || item.format.some((value, i) => value !== first.format[i])
|
|
229
|
+
))
|
|
230
|
+
throw new Error("WAV buffers have incompatible fmt chunks.");
|
|
231
|
+
const dataLength = parsed.reduce((total, item) => total + item.data.byteLength, 0);
|
|
232
|
+
const nonDataLength = first.chunks.reduce(
|
|
233
|
+
(total, chunk) => chunk.id === "data" ? total : total + 8 + chunk.data.byteLength + (chunk.data.byteLength & 1),
|
|
234
|
+
0
|
|
235
|
+
);
|
|
236
|
+
const outputLength = 12 + nonDataLength + 8 + dataLength + (dataLength & 1);
|
|
237
|
+
if (outputLength - 8 > 4294967295) throw new RangeError("Merged WAV exceeds the RIFF format size limit.");
|
|
238
|
+
const output = new Uint8Array(outputLength);
|
|
239
|
+
output.set(Uint8Array.from([82, 73, 70, 70]), 0);
|
|
240
|
+
writeUint32(output, 4, outputLength - 8);
|
|
241
|
+
output.set(Uint8Array.from([87, 65, 86, 69]), 8);
|
|
242
|
+
let outputOffset = 12;
|
|
243
|
+
let dataWritten = false;
|
|
244
|
+
for (const chunk of first.chunks) {
|
|
245
|
+
if (chunk.id === "data") {
|
|
246
|
+
if (dataWritten) continue;
|
|
247
|
+
const data = new Uint8Array(dataLength);
|
|
248
|
+
let dataOffset = 0;
|
|
249
|
+
for (const item of parsed) {
|
|
250
|
+
data.set(item.data, dataOffset);
|
|
251
|
+
dataOffset += item.data.byteLength;
|
|
252
|
+
}
|
|
253
|
+
outputOffset = writeChunk(output, outputOffset, "data", data);
|
|
254
|
+
dataWritten = true;
|
|
255
|
+
} else {
|
|
256
|
+
outputOffset = writeChunk(output, outputOffset, chunk.id, chunk.data);
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
if (!dataWritten) throw new Error("WAV audio must contain a data chunk.");
|
|
260
|
+
return output.buffer;
|
|
261
|
+
}
|
|
262
|
+
function skipId3v2(bytes) {
|
|
263
|
+
if (!ascii(bytes, 0, "ID3") || bytes.byteLength < 10) return 0;
|
|
264
|
+
const size = [bytes[6], bytes[7], bytes[8], bytes[9]].reduce((total, value) => total << 7 | value & 127, 0);
|
|
265
|
+
const hasFooter = (bytes[5] & 16) !== 0;
|
|
266
|
+
return Math.min(bytes.byteLength, 10 + size + (hasFooter ? 10 : 0));
|
|
267
|
+
}
|
|
268
|
+
function stripMp3Tags(buffer) {
|
|
269
|
+
const bytes = new Uint8Array(buffer);
|
|
270
|
+
const start = skipId3v2(bytes);
|
|
271
|
+
const end = bytes.byteLength >= 128 && ascii(bytes, bytes.byteLength - 128, "TAG") ? bytes.byteLength - 128 : bytes.byteLength;
|
|
272
|
+
return bytes.slice(Math.min(start, end), end);
|
|
273
|
+
}
|
|
274
|
+
function isMp3Format(format) {
|
|
275
|
+
return /(?:mp3|mpeg)/i.test(format);
|
|
276
|
+
}
|
|
277
|
+
function isWavFormat(format) {
|
|
278
|
+
return /(?:wav|wave|riff)/i.test(format);
|
|
279
|
+
}
|
|
280
|
+
function isRawFormat(format) {
|
|
281
|
+
return /^raw(?:-|$)/i.test(format);
|
|
282
|
+
}
|
|
283
|
+
function resolveMergeAudioFormat(format) {
|
|
284
|
+
if (isWavFormat(format)) return "wav";
|
|
285
|
+
if (isMp3Format(format)) return "mp3";
|
|
286
|
+
if (isRawFormat(format)) return "raw";
|
|
287
|
+
return void 0;
|
|
288
|
+
}
|
|
289
|
+
function canMergeAudioFormat(format) {
|
|
290
|
+
return resolveMergeAudioFormat(format) !== void 0;
|
|
291
|
+
}
|
|
292
|
+
function mergeAudioBuffers(buffers, options) {
|
|
293
|
+
const format = typeof options === "string" ? options : options?.format;
|
|
294
|
+
if (!format) throw new UnsupportedMergeFormatError("");
|
|
295
|
+
try {
|
|
296
|
+
if (isWavFormat(format)) return mergeWavBuffers(buffers);
|
|
297
|
+
if (isMp3Format(format)) {
|
|
298
|
+
const parts = buffers.map(stripMp3Tags);
|
|
299
|
+
const output = new Uint8Array(parts.reduce((total, part) => total + part.byteLength, 0));
|
|
300
|
+
let offset = 0;
|
|
301
|
+
for (const part of parts) {
|
|
302
|
+
output.set(part, offset);
|
|
303
|
+
offset += part.byteLength;
|
|
304
|
+
}
|
|
305
|
+
return output.buffer;
|
|
306
|
+
}
|
|
307
|
+
if (isRawFormat(format)) {
|
|
308
|
+
const output = new Uint8Array(buffers.reduce((total, buffer) => total + buffer.byteLength, 0));
|
|
309
|
+
let offset = 0;
|
|
310
|
+
for (const buffer of buffers) {
|
|
311
|
+
output.set(new Uint8Array(buffer), offset);
|
|
312
|
+
offset += buffer.byteLength;
|
|
313
|
+
}
|
|
314
|
+
return output.buffer;
|
|
315
|
+
}
|
|
316
|
+
throw new UnsupportedMergeFormatError(format);
|
|
317
|
+
} catch (error) {
|
|
318
|
+
if (error instanceof UnsupportedMergeFormatError || error instanceof MergeError) throw error;
|
|
319
|
+
throw new MergeError(`Audio buffers could not be merged for format "${format}".`, error);
|
|
320
|
+
}
|
|
321
|
+
}
|
|
124
322
|
function closeSpeechResources(speechConfig, synthesizer) {
|
|
125
323
|
try {
|
|
126
324
|
synthesizer.close();
|
|
@@ -134,7 +332,7 @@ function closeSpeechResources(speechConfig, synthesizer) {
|
|
|
134
332
|
var ticksToMilliseconds = (ticks) => Math.max(0, ticks) / 1e4;
|
|
135
333
|
async function synthesizeSsml(ssml, config) {
|
|
136
334
|
if (config.signal?.aborted) {
|
|
137
|
-
throw
|
|
335
|
+
throw new SynthesisCancelledError();
|
|
138
336
|
}
|
|
139
337
|
const speechConfig = createSpeechConfig(config);
|
|
140
338
|
const synthesizer = new SpeechSDK2.SpeechSynthesizer(speechConfig, null);
|
|
@@ -157,23 +355,94 @@ async function synthesizeSsml(ssml, config) {
|
|
|
157
355
|
settled = true;
|
|
158
356
|
cleanup();
|
|
159
357
|
closeResources();
|
|
160
|
-
reject(
|
|
358
|
+
reject(toSynthesisError(error));
|
|
161
359
|
};
|
|
162
360
|
const boundaries = [];
|
|
163
361
|
const visemes = [];
|
|
164
362
|
const bookmarks = [];
|
|
363
|
+
let sourceEventCursor = 0;
|
|
364
|
+
let generatedSourceMap;
|
|
365
|
+
if (!config.sourceTextSegments && !config.sourceMarkers) {
|
|
366
|
+
try {
|
|
367
|
+
generatedSourceMap = getSsmlSourceMap2(ssml);
|
|
368
|
+
} catch {
|
|
369
|
+
generatedSourceMap = void 0;
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
const sourceBaseOffset = config.sourceTextRange?.start ?? 0;
|
|
373
|
+
const sourceSegments = config.sourceTextSegments ?? generatedSourceMap?.segments.map((segment) => ({
|
|
374
|
+
...segment,
|
|
375
|
+
range: {
|
|
376
|
+
start: segment.range.start + sourceBaseOffset,
|
|
377
|
+
end: segment.range.end + sourceBaseOffset
|
|
378
|
+
},
|
|
379
|
+
sourceNodePath: [...segment.sourceNodePath]
|
|
380
|
+
})) ?? [];
|
|
381
|
+
const sourceMarkers = config.sourceMarkers ?? generatedSourceMap?.markers.map((marker) => ({
|
|
382
|
+
...marker,
|
|
383
|
+
originalTextRange: {
|
|
384
|
+
start: marker.originalTextRange.start + sourceBaseOffset,
|
|
385
|
+
end: marker.originalTextRange.end + sourceBaseOffset
|
|
386
|
+
},
|
|
387
|
+
sourceNodePath: [...marker.sourceNodePath]
|
|
388
|
+
})) ?? [];
|
|
389
|
+
const sourceText = sourceSegments.map((segment) => segment.text).join("");
|
|
390
|
+
const mapSourceEvent = (text, offsetHint, markerName) => {
|
|
391
|
+
const marker = markerName ? sourceMarkers.find((candidate) => candidate.name === markerName) : void 0;
|
|
392
|
+
if (marker) {
|
|
393
|
+
return {
|
|
394
|
+
originalTextRange: { ...marker.originalTextRange },
|
|
395
|
+
sourceNodePath: [...marker.sourceNodePath],
|
|
396
|
+
textRange: { ...marker.originalTextRange }
|
|
397
|
+
};
|
|
398
|
+
}
|
|
399
|
+
if (sourceSegments.length === 0 && !config.sourceTextRange && !config.sourceNodePath) return {};
|
|
400
|
+
const value = text ?? "";
|
|
401
|
+
let localStart = Number.isFinite(offsetHint) && (offsetHint ?? 0) >= 0 ? offsetHint : -1;
|
|
402
|
+
if (value && localStart >= 0 && sourceText.slice(localStart, localStart + value.length) !== value)
|
|
403
|
+
localStart = -1;
|
|
404
|
+
if (localStart < 0 || localStart > sourceText.length) {
|
|
405
|
+
localStart = value ? sourceText.indexOf(value, sourceEventCursor) : sourceEventCursor;
|
|
406
|
+
if (localStart < 0) localStart = value ? sourceText.indexOf(value) : sourceEventCursor;
|
|
407
|
+
}
|
|
408
|
+
localStart = Math.max(0, localStart);
|
|
409
|
+
const localEnd = Math.min(sourceText.length, localStart + value.length);
|
|
410
|
+
sourceEventCursor = Math.max(sourceEventCursor, localEnd);
|
|
411
|
+
const baseStart = config.sourceTextRange?.start ?? sourceSegments[0]?.range.start ?? 0;
|
|
412
|
+
const fallbackRange = { start: baseStart + localStart, end: baseStart + localEnd };
|
|
413
|
+
const segment = sourceSegments.find(({ range }) => range.start <= fallbackRange.start && range.end > fallbackRange.start) ?? sourceSegments.find(({ range }) => range.end > fallbackRange.start) ?? (value.length === 0 ? sourceSegments.find(({ range }) => range.start <= fallbackRange.start && range.end >= fallbackRange.start) : void 0);
|
|
414
|
+
return {
|
|
415
|
+
originalTextRange: { ...fallbackRange },
|
|
416
|
+
textRange: { ...fallbackRange },
|
|
417
|
+
...segment ? { sourceNodePath: [...segment.sourceNodePath] } : config.sourceNodePath ? { sourceNodePath: [...config.sourceNodePath] } : {}
|
|
418
|
+
};
|
|
419
|
+
};
|
|
165
420
|
synthesizer.wordBoundary = (_sender, event) => {
|
|
166
421
|
boundaries.push({
|
|
167
422
|
text: event.text,
|
|
168
423
|
audioOffsetMs: ticksToMilliseconds(event.audioOffset),
|
|
169
|
-
durationMs: ticksToMilliseconds(event.duration)
|
|
424
|
+
durationMs: ticksToMilliseconds(event.duration),
|
|
425
|
+
...mapSourceEvent(
|
|
426
|
+
event.text,
|
|
427
|
+
event.textOffset
|
|
428
|
+
)
|
|
170
429
|
});
|
|
171
430
|
};
|
|
172
431
|
synthesizer.visemeReceived = (_sender, event) => {
|
|
173
|
-
|
|
432
|
+
const eventWithOffset = event;
|
|
433
|
+
visemes.push({
|
|
434
|
+
visemeId: event.visemeId,
|
|
435
|
+
audioOffsetMs: ticksToMilliseconds(event.audioOffset),
|
|
436
|
+
...mapSourceEvent(void 0, eventWithOffset.textOffset)
|
|
437
|
+
});
|
|
174
438
|
};
|
|
175
439
|
synthesizer.bookmarkReached = (_sender, event) => {
|
|
176
|
-
|
|
440
|
+
const eventWithOffset = event;
|
|
441
|
+
bookmarks.push({
|
|
442
|
+
name: event.text,
|
|
443
|
+
audioOffsetMs: ticksToMilliseconds(event.audioOffset),
|
|
444
|
+
...mapSourceEvent(void 0, eventWithOffset.textOffset, event.text)
|
|
445
|
+
});
|
|
177
446
|
};
|
|
178
447
|
const cb = (result) => {
|
|
179
448
|
if (settled) return;
|
|
@@ -196,7 +465,10 @@ async function synthesizeSsml(ssml, config) {
|
|
|
196
465
|
const requestId = result.resultId;
|
|
197
466
|
const addSourceMetadata = (event) => ({
|
|
198
467
|
...event,
|
|
199
|
-
...config.sourceTextRange ? { textRange: { ...config.sourceTextRange } } : {},
|
|
468
|
+
...config.sourceTextRange && !("textRange" in event) ? { textRange: { ...config.sourceTextRange } } : {},
|
|
469
|
+
...config.sourceTextRange && !("originalTextRange" in event) ? { originalTextRange: { ...config.sourceTextRange } } : {},
|
|
470
|
+
...config.chunkIndex !== void 0 ? { chunkIndex: config.chunkIndex } : {},
|
|
471
|
+
...config.sourceNodePath ? { sourceNodePath: [...config.sourceNodePath] } : {},
|
|
200
472
|
...requestId ? { requestId } : {}
|
|
201
473
|
});
|
|
202
474
|
const sourceBoundaries = boundaries.map((boundary) => addSourceMetadata(boundary));
|
|
@@ -214,12 +486,12 @@ async function synthesizeSsml(ssml, config) {
|
|
|
214
486
|
};
|
|
215
487
|
try {
|
|
216
488
|
if (config.signal) {
|
|
217
|
-
abortHandler = () => rejectWithError(
|
|
489
|
+
abortHandler = () => rejectWithError(new SynthesisCancelledError());
|
|
218
490
|
config.signal.addEventListener("abort", abortHandler, { once: true });
|
|
219
491
|
}
|
|
220
492
|
if (config.timeoutMs !== void 0 && config.timeoutMs > 0) {
|
|
221
493
|
timeout = setTimeout(
|
|
222
|
-
() => rejectWithError(`Speech synthesis timed out after ${config.timeoutMs} ms.`),
|
|
494
|
+
() => rejectWithError(new SynthesisTimeoutError(`Speech synthesis timed out after ${config.timeoutMs} ms.`)),
|
|
223
495
|
config.timeoutMs
|
|
224
496
|
);
|
|
225
497
|
}
|
|
@@ -232,60 +504,117 @@ async function synthesizeSsml(ssml, config) {
|
|
|
232
504
|
async function synthesizeSsmlChunks(chunks, config) {
|
|
233
505
|
const results = [];
|
|
234
506
|
const totalChunks = chunks.length;
|
|
507
|
+
const report = (event) => config.onProgress?.(event);
|
|
235
508
|
for (const [index, chunk] of chunks.entries()) {
|
|
236
509
|
const input = typeof chunk === "string" ? { ssml: chunk } : chunk;
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
510
|
+
report({
|
|
511
|
+
currentChunk: index,
|
|
512
|
+
totalChunks,
|
|
513
|
+
percent: totalChunks === 0 ? 100 : Math.round(index / totalChunks * 100),
|
|
514
|
+
chunkIndex: index,
|
|
515
|
+
originalTextRange: input.originalTextRange,
|
|
516
|
+
status: "pending",
|
|
517
|
+
durationMs: 0
|
|
241
518
|
});
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
519
|
+
}
|
|
520
|
+
for (const [index, chunk] of chunks.entries()) {
|
|
521
|
+
const input = typeof chunk === "string" ? { ssml: chunk } : chunk;
|
|
522
|
+
report({
|
|
523
|
+
currentChunk: index,
|
|
245
524
|
totalChunks,
|
|
246
|
-
percent: totalChunks === 0 ? 100 : Math.round(
|
|
525
|
+
percent: totalChunks === 0 ? 100 : Math.round(index / totalChunks * 100),
|
|
526
|
+
chunkIndex: index,
|
|
527
|
+
originalTextRange: input.originalTextRange,
|
|
528
|
+
status: "synthesizing",
|
|
529
|
+
durationMs: 0
|
|
247
530
|
});
|
|
531
|
+
const startedAt = Date.now();
|
|
532
|
+
try {
|
|
533
|
+
const result = await synthesizeSsml(input.ssml, {
|
|
534
|
+
...config,
|
|
535
|
+
...input.originalTextRange ? { sourceTextRange: input.originalTextRange } : {},
|
|
536
|
+
...input.sourceNodePath ?? config.sourceNodePath ? { sourceNodePath: [...input.sourceNodePath ?? config.sourceNodePath ?? []] } : {},
|
|
537
|
+
...input.sourceTextSegments ? { sourceTextSegments: input.sourceTextSegments } : {},
|
|
538
|
+
...input.sourceMarkers ? { sourceMarkers: input.sourceMarkers } : {},
|
|
539
|
+
chunkIndex: index,
|
|
540
|
+
onProgress: void 0
|
|
541
|
+
});
|
|
542
|
+
results.push(result);
|
|
543
|
+
report({
|
|
544
|
+
currentChunk: index + 1,
|
|
545
|
+
totalChunks,
|
|
546
|
+
percent: totalChunks === 0 ? 100 : Math.round((index + 1) / totalChunks * 100),
|
|
547
|
+
chunkIndex: index,
|
|
548
|
+
originalTextRange: input.originalTextRange,
|
|
549
|
+
status: "success",
|
|
550
|
+
durationMs: Date.now() - startedAt
|
|
551
|
+
});
|
|
552
|
+
} catch (error) {
|
|
553
|
+
report({
|
|
554
|
+
currentChunk: index,
|
|
555
|
+
totalChunks,
|
|
556
|
+
percent: totalChunks === 0 ? 100 : Math.round(index / totalChunks * 100),
|
|
557
|
+
chunkIndex: index,
|
|
558
|
+
originalTextRange: input.originalTextRange,
|
|
559
|
+
status: "failed",
|
|
560
|
+
durationMs: Date.now() - startedAt,
|
|
561
|
+
error
|
|
562
|
+
});
|
|
563
|
+
throw error;
|
|
564
|
+
}
|
|
248
565
|
}
|
|
249
|
-
return mergeSynthesisResults(results
|
|
566
|
+
return mergeSynthesisResults(results, {
|
|
567
|
+
format: config.outputFormat ?? "audio-16khz-128kbitrate-mono-mp3"
|
|
568
|
+
});
|
|
250
569
|
}
|
|
251
|
-
function
|
|
252
|
-
const audioLength = results.reduce((total, result) => total + result.audioData.byteLength, 0);
|
|
253
|
-
const audioData = new Uint8Array(audioLength);
|
|
570
|
+
function createMergedResult(results, audioData, format) {
|
|
254
571
|
const boundaries = [];
|
|
255
572
|
const visemes = [];
|
|
256
573
|
const bookmarks = [];
|
|
257
|
-
let byteOffset = 0;
|
|
258
574
|
let durationOffset = 0;
|
|
259
|
-
for (const result of results) {
|
|
260
|
-
audioData.set(new Uint8Array(result.audioData), byteOffset);
|
|
261
|
-
byteOffset += result.audioData.byteLength;
|
|
575
|
+
for (const [resultIndex, result] of results.entries()) {
|
|
262
576
|
const chunkBoundaries = result.boundaries && result.boundaries.length > 0 ? result.boundaries : result.wordBoundary ?? result.wordBoundaries ?? [];
|
|
263
577
|
for (const boundary of chunkBoundaries) {
|
|
264
578
|
const textRange = boundary.textRange ?? result.textRange;
|
|
579
|
+
const originalTextRange = boundary.originalTextRange ?? textRange;
|
|
265
580
|
const requestId = boundary.requestId ?? result.requestId;
|
|
266
581
|
boundaries.push({
|
|
267
582
|
...boundary,
|
|
268
583
|
audioOffsetMs: boundary.audioOffsetMs + durationOffset,
|
|
584
|
+
chunkAudioOffsetMs: boundary.chunkAudioOffsetMs ?? boundary.audioOffsetMs,
|
|
585
|
+
...boundary.chunkIndex === void 0 ? { chunkIndex: resultIndex } : {},
|
|
586
|
+
...boundary.sourceNodePath ? { sourceNodePath: [...boundary.sourceNodePath] } : {},
|
|
587
|
+
...originalTextRange ? { originalTextRange: { ...originalTextRange } } : {},
|
|
269
588
|
...textRange ? { textRange: { ...textRange } } : {},
|
|
270
589
|
...requestId ? { requestId } : {}
|
|
271
590
|
});
|
|
272
591
|
}
|
|
273
592
|
for (const viseme of result.visemes ?? []) {
|
|
274
593
|
const textRange = viseme.textRange ?? result.textRange;
|
|
594
|
+
const originalTextRange = viseme.originalTextRange ?? textRange;
|
|
275
595
|
const requestId = viseme.requestId ?? result.requestId;
|
|
276
596
|
visemes.push({
|
|
277
597
|
...viseme,
|
|
278
598
|
audioOffsetMs: viseme.audioOffsetMs + durationOffset,
|
|
599
|
+
chunkAudioOffsetMs: viseme.chunkAudioOffsetMs ?? viseme.audioOffsetMs,
|
|
600
|
+
...viseme.chunkIndex === void 0 ? { chunkIndex: resultIndex } : {},
|
|
601
|
+
...viseme.sourceNodePath ? { sourceNodePath: [...viseme.sourceNodePath] } : {},
|
|
602
|
+
...originalTextRange ? { originalTextRange: { ...originalTextRange } } : {},
|
|
279
603
|
...textRange ? { textRange: { ...textRange } } : {},
|
|
280
604
|
...requestId ? { requestId } : {}
|
|
281
605
|
});
|
|
282
606
|
}
|
|
283
607
|
for (const bookmark of result.bookmarks ?? []) {
|
|
284
608
|
const textRange = bookmark.textRange ?? result.textRange;
|
|
609
|
+
const originalTextRange = bookmark.originalTextRange ?? textRange;
|
|
285
610
|
const requestId = bookmark.requestId ?? result.requestId;
|
|
286
611
|
bookmarks.push({
|
|
287
612
|
...bookmark,
|
|
288
613
|
audioOffsetMs: bookmark.audioOffsetMs + durationOffset,
|
|
614
|
+
chunkAudioOffsetMs: bookmark.chunkAudioOffsetMs ?? bookmark.audioOffsetMs,
|
|
615
|
+
...bookmark.chunkIndex === void 0 ? { chunkIndex: resultIndex } : {},
|
|
616
|
+
...bookmark.sourceNodePath ? { sourceNodePath: [...bookmark.sourceNodePath] } : {},
|
|
617
|
+
...originalTextRange ? { originalTextRange: { ...originalTextRange } } : {},
|
|
289
618
|
...textRange ? { textRange: { ...textRange } } : {},
|
|
290
619
|
...requestId ? { requestId } : {}
|
|
291
620
|
});
|
|
@@ -293,8 +622,9 @@ function mergeSynthesisResults(results) {
|
|
|
293
622
|
durationOffset += Math.max(0, result.durationMs);
|
|
294
623
|
}
|
|
295
624
|
return {
|
|
296
|
-
audioData
|
|
625
|
+
audioData,
|
|
297
626
|
durationMs: durationOffset,
|
|
627
|
+
mimeType: resolveMimeType(format),
|
|
298
628
|
...boundaries.length > 0 ? { boundaries, wordBoundary: boundaries, wordBoundaries: boundaries } : {},
|
|
299
629
|
...visemes.length > 0 ? { visemes } : {},
|
|
300
630
|
...bookmarks.length > 0 ? { bookmarks } : {},
|
|
@@ -302,34 +632,224 @@ function mergeSynthesisResults(results) {
|
|
|
302
632
|
...results.length === 1 && results[0]?.textRange ? { textRange: { ...results[0].textRange } } : {}
|
|
303
633
|
};
|
|
304
634
|
}
|
|
635
|
+
function mergeSynthesisResults(results, options) {
|
|
636
|
+
const resolvedOptions = typeof options === "string" ? { format: options } : options;
|
|
637
|
+
const format = resolvedOptions?.format;
|
|
638
|
+
if (!format) throw new UnsupportedMergeFormatError("");
|
|
639
|
+
const buffers = results.map((result) => result.audioData);
|
|
640
|
+
if (resolvedOptions.customMerger) {
|
|
641
|
+
return Promise.resolve().then(() => resolvedOptions.customMerger?.(buffers, format)).then((merged) => {
|
|
642
|
+
if (!merged) throw new MergeError("The custom audio merger returned no audio buffer.");
|
|
643
|
+
return createMergedResult(results, merged, format);
|
|
644
|
+
}).catch((error) => {
|
|
645
|
+
if (error instanceof UnsupportedMergeFormatError || error instanceof MergeError) throw error;
|
|
646
|
+
throw new MergeError(`Custom audio merger failed for format "${format}".`, error);
|
|
647
|
+
});
|
|
648
|
+
}
|
|
649
|
+
try {
|
|
650
|
+
return createMergedResult(results, mergeAudioBuffers(buffers, { format }), format);
|
|
651
|
+
} catch (error) {
|
|
652
|
+
if (error instanceof UnsupportedMergeFormatError || error instanceof MergeError) throw error;
|
|
653
|
+
throw new MergeError(`Audio buffers could not be merged for format "${format}".`, error);
|
|
654
|
+
}
|
|
655
|
+
}
|
|
305
656
|
async function synthesizeSpeech(ssml, config) {
|
|
306
657
|
return (await synthesizeSsml(ssml, config)).audioData;
|
|
307
658
|
}
|
|
308
659
|
|
|
309
660
|
// packages/azure-tts-client/src/safe.ts
|
|
661
|
+
var ChunkValidationError = class extends Error {
|
|
662
|
+
constructor(chunkIndex, diagnostics) {
|
|
663
|
+
super(`SSML validation failed for chunk ${chunkIndex}; the Azure Speech API was not called.`);
|
|
664
|
+
this.kind = "validation-error";
|
|
665
|
+
this.name = "ChunkValidationError";
|
|
666
|
+
this.chunkIndex = chunkIndex;
|
|
667
|
+
this.diagnostics = diagnostics;
|
|
668
|
+
}
|
|
669
|
+
};
|
|
670
|
+
function failure(error) {
|
|
671
|
+
return { ok: false, success: false, status: error.kind, error };
|
|
672
|
+
}
|
|
310
673
|
async function synthesizeSsmlSafe(client, ssml, options = {}) {
|
|
311
|
-
const validationOptions = options.validation ?? options;
|
|
674
|
+
const validationOptions = withValidationSignal(options.validation ?? options, options.signal);
|
|
312
675
|
const diagnostics = await Promise.resolve(validateAzureSsml2(ssml, validationOptions));
|
|
676
|
+
if (options.signal?.aborted) {
|
|
677
|
+
const error = toSynthesisError(new Error("Speech synthesis was cancelled."));
|
|
678
|
+
return failure(error);
|
|
679
|
+
}
|
|
313
680
|
const errors = diagnostics.filter((diagnostic) => diagnostic.severity === "error");
|
|
314
681
|
if (errors.length > 0) {
|
|
682
|
+
return failure({
|
|
683
|
+
kind: "validation-error",
|
|
684
|
+
message: "SSML validation failed; the Azure Speech API was not called.",
|
|
685
|
+
diagnostics: errors
|
|
686
|
+
});
|
|
687
|
+
}
|
|
688
|
+
try {
|
|
315
689
|
return {
|
|
316
|
-
ok:
|
|
317
|
-
success:
|
|
318
|
-
status: "
|
|
319
|
-
|
|
320
|
-
kind: "validation",
|
|
321
|
-
message: "SSML validation failed; the Azure Speech API was not called.",
|
|
322
|
-
diagnostics: errors
|
|
323
|
-
}
|
|
690
|
+
ok: true,
|
|
691
|
+
success: true,
|
|
692
|
+
status: "success",
|
|
693
|
+
value: await client.synthesizeSsml(ssml, { signal: options.signal })
|
|
324
694
|
};
|
|
695
|
+
} catch (error) {
|
|
696
|
+
const synthesisError = toSynthesisError(error);
|
|
697
|
+
return failure(synthesisError);
|
|
698
|
+
}
|
|
699
|
+
}
|
|
700
|
+
async function synthesizeSsmlChunksSafe(client, chunks, options = {}) {
|
|
701
|
+
const validationOptions = withValidationSignal(options.validation ?? options, options.signal);
|
|
702
|
+
if (options.signal?.aborted) {
|
|
703
|
+
const error = toSynthesisError(new Error("Speech synthesis was cancelled."));
|
|
704
|
+
return failure(error);
|
|
705
|
+
}
|
|
706
|
+
const pending = (index, status, error) => {
|
|
707
|
+
options.onProgress?.({
|
|
708
|
+
currentChunk: status === "success" ? index + 1 : index,
|
|
709
|
+
totalChunks: chunks.length,
|
|
710
|
+
percent: chunks.length === 0 ? 100 : Math.round((status === "success" ? index + 1 : index) / chunks.length * 100),
|
|
711
|
+
chunkIndex: index,
|
|
712
|
+
originalTextRange: typeof chunks[index] === "string" ? void 0 : chunks[index]?.originalTextRange,
|
|
713
|
+
status,
|
|
714
|
+
durationMs: 0,
|
|
715
|
+
...error ? { error } : {}
|
|
716
|
+
});
|
|
717
|
+
};
|
|
718
|
+
chunks.forEach((_chunk, index) => {
|
|
719
|
+
pending(index, "pending");
|
|
720
|
+
});
|
|
721
|
+
const validations = await Promise.all(
|
|
722
|
+
chunks.map(async (chunk) => {
|
|
723
|
+
const ssml = typeof chunk === "string" ? chunk : chunk.ssml;
|
|
724
|
+
const sourceNodePath = typeof chunk === "string" ? options.sourceNodePath : chunk.sourceNodePath ?? options.sourceNodePath;
|
|
725
|
+
const diagnostics = await Promise.resolve(
|
|
726
|
+
validateAzureSsml2(ssml, { ...validationOptions, ...sourceNodePath ? { sourceNodePath } : {} })
|
|
727
|
+
);
|
|
728
|
+
return diagnostics.filter((diagnostic) => diagnostic.severity === "error");
|
|
729
|
+
})
|
|
730
|
+
);
|
|
731
|
+
const firstInvalidIndex = validations.findIndex((diagnostics) => diagnostics.length > 0);
|
|
732
|
+
if (firstInvalidIndex >= 0) {
|
|
733
|
+
const error = new ChunkValidationError(firstInvalidIndex, validations[firstInvalidIndex] ?? []);
|
|
734
|
+
pending(firstInvalidIndex, "failed", error);
|
|
735
|
+
return failure(error);
|
|
325
736
|
}
|
|
326
737
|
try {
|
|
327
|
-
|
|
738
|
+
if (client.synthesizeChunks) {
|
|
739
|
+
const normalizedChunks = chunks.map((chunk) => {
|
|
740
|
+
if (typeof chunk === "string" || chunk.sourceNodePath || !options.sourceNodePath) return chunk;
|
|
741
|
+
return { ...chunk, sourceNodePath: [...options.sourceNodePath] };
|
|
742
|
+
});
|
|
743
|
+
const value = await client.synthesizeChunks(normalizedChunks, {
|
|
744
|
+
onProgress: options.onProgress,
|
|
745
|
+
outputFormat: options.outputFormat,
|
|
746
|
+
signal: options.signal,
|
|
747
|
+
timeoutMs: options.timeoutMs,
|
|
748
|
+
sourceNodePath: options.sourceNodePath
|
|
749
|
+
});
|
|
750
|
+
return { ok: true, success: true, status: "success", value };
|
|
751
|
+
}
|
|
752
|
+
const results = [];
|
|
753
|
+
for (const [index, chunk] of chunks.entries()) {
|
|
754
|
+
const input = typeof chunk === "string" ? { ssml: chunk } : chunk;
|
|
755
|
+
const sourceNodePath = input.sourceNodePath;
|
|
756
|
+
const originalTextRange = input.originalTextRange;
|
|
757
|
+
pending(index, "synthesizing");
|
|
758
|
+
const startedAt = Date.now();
|
|
759
|
+
try {
|
|
760
|
+
const result = await client.synthesizeSsml(input.ssml, {
|
|
761
|
+
outputFormat: options.outputFormat,
|
|
762
|
+
signal: options.signal,
|
|
763
|
+
timeoutMs: options.timeoutMs,
|
|
764
|
+
sourceNodePath: input.sourceNodePath ?? options.sourceNodePath
|
|
765
|
+
});
|
|
766
|
+
results.push({
|
|
767
|
+
...result,
|
|
768
|
+
...input.originalTextRange ? { textRange: { ...input.originalTextRange } } : {},
|
|
769
|
+
...sourceNodePath ? {
|
|
770
|
+
boundaries: result.boundaries?.map((event) => ({
|
|
771
|
+
...event,
|
|
772
|
+
sourceNodePath: [...sourceNodePath],
|
|
773
|
+
...event.originalTextRange ? { originalTextRange: { ...event.originalTextRange } } : input.originalTextRange ? { originalTextRange: { ...input.originalTextRange } } : {}
|
|
774
|
+
})),
|
|
775
|
+
visemes: result.visemes?.map((event) => ({
|
|
776
|
+
...event,
|
|
777
|
+
sourceNodePath: [...sourceNodePath],
|
|
778
|
+
...event.originalTextRange ? { originalTextRange: { ...event.originalTextRange } } : input.originalTextRange ? { originalTextRange: { ...input.originalTextRange } } : {}
|
|
779
|
+
})),
|
|
780
|
+
bookmarks: result.bookmarks?.map((event) => ({
|
|
781
|
+
...event,
|
|
782
|
+
sourceNodePath: [...sourceNodePath],
|
|
783
|
+
...event.originalTextRange ? { originalTextRange: { ...event.originalTextRange } } : input.originalTextRange ? { originalTextRange: { ...input.originalTextRange } } : {}
|
|
784
|
+
}))
|
|
785
|
+
} : {},
|
|
786
|
+
...originalTextRange ? {
|
|
787
|
+
boundaries: result.boundaries?.map((event) => ({
|
|
788
|
+
...event,
|
|
789
|
+
originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
|
|
790
|
+
})),
|
|
791
|
+
wordBoundary: result.wordBoundary?.map((event) => ({
|
|
792
|
+
...event,
|
|
793
|
+
originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
|
|
794
|
+
})),
|
|
795
|
+
wordBoundaries: result.wordBoundaries?.map((event) => ({
|
|
796
|
+
...event,
|
|
797
|
+
originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
|
|
798
|
+
})),
|
|
799
|
+
visemes: result.visemes?.map((event) => ({
|
|
800
|
+
...event,
|
|
801
|
+
originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
|
|
802
|
+
})),
|
|
803
|
+
bookmarks: result.bookmarks?.map((event) => ({
|
|
804
|
+
...event,
|
|
805
|
+
originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
|
|
806
|
+
}))
|
|
807
|
+
} : {}
|
|
808
|
+
});
|
|
809
|
+
options.onProgress?.({
|
|
810
|
+
currentChunk: index + 1,
|
|
811
|
+
totalChunks: chunks.length,
|
|
812
|
+
percent: chunks.length === 0 ? 100 : Math.round((index + 1) / chunks.length * 100),
|
|
813
|
+
chunkIndex: index,
|
|
814
|
+
originalTextRange: input.originalTextRange,
|
|
815
|
+
status: "success",
|
|
816
|
+
durationMs: Date.now() - startedAt
|
|
817
|
+
});
|
|
818
|
+
} catch (error) {
|
|
819
|
+
options.onProgress?.({
|
|
820
|
+
currentChunk: index,
|
|
821
|
+
totalChunks: chunks.length,
|
|
822
|
+
percent: chunks.length === 0 ? 100 : Math.round(index / chunks.length * 100),
|
|
823
|
+
chunkIndex: index,
|
|
824
|
+
originalTextRange: input.originalTextRange,
|
|
825
|
+
status: "failed",
|
|
826
|
+
durationMs: Date.now() - startedAt,
|
|
827
|
+
error
|
|
828
|
+
});
|
|
829
|
+
throw error;
|
|
830
|
+
}
|
|
831
|
+
}
|
|
832
|
+
return {
|
|
833
|
+
ok: true,
|
|
834
|
+
success: true,
|
|
835
|
+
status: "success",
|
|
836
|
+
value: mergeSynthesisResults(results, {
|
|
837
|
+
format: options.outputFormat ?? "audio-16khz-128kbitrate-mono-mp3"
|
|
838
|
+
})
|
|
839
|
+
};
|
|
328
840
|
} catch (error) {
|
|
329
|
-
const
|
|
330
|
-
return
|
|
841
|
+
const synthesisError = toSynthesisError(error);
|
|
842
|
+
return failure(synthesisError);
|
|
331
843
|
}
|
|
332
844
|
}
|
|
845
|
+
function withValidationSignal(options, signal) {
|
|
846
|
+
if (!signal) return options;
|
|
847
|
+
return {
|
|
848
|
+
...options,
|
|
849
|
+
urlValidatorSignal: signal,
|
|
850
|
+
urlValidation: { ...options.urlValidation ?? {}, signal }
|
|
851
|
+
};
|
|
852
|
+
}
|
|
333
853
|
|
|
334
854
|
// packages/azure-tts-client/src/client.ts
|
|
335
855
|
var ENDPOINT_TEMPLATE = "https://{region}.tts.speech.microsoft.com/cognitiveservices/v1";
|
|
@@ -346,11 +866,21 @@ var AzureTtsClient = class {
|
|
|
346
866
|
const config = { endpoint, region, subscriptionKey, outputFormat, signal, timeoutMs };
|
|
347
867
|
return synthesizeSpeech(ssml, config);
|
|
348
868
|
}
|
|
349
|
-
async synthesizeSsml(ssml) {
|
|
869
|
+
async synthesizeSsml(ssml, options = {}) {
|
|
350
870
|
const { region, subscriptionKey, outputFormat, signal, timeoutMs } = __privateGet(this, _options);
|
|
351
871
|
const endpoint = __privateGet(this, _options).endpoint?.trim() || ENDPOINT_TEMPLATE.replace("{region}", region);
|
|
352
872
|
__privateGet(this, _options).logger?.debug?.("Using Azure TTS endpoint:", endpoint);
|
|
353
|
-
return synthesizeSsml(ssml, {
|
|
873
|
+
return synthesizeSsml(ssml, {
|
|
874
|
+
endpoint,
|
|
875
|
+
region,
|
|
876
|
+
subscriptionKey,
|
|
877
|
+
outputFormat: options.outputFormat ?? outputFormat,
|
|
878
|
+
signal: options.signal ?? signal,
|
|
879
|
+
timeoutMs: options.timeoutMs ?? timeoutMs,
|
|
880
|
+
sourceNodePath: options.sourceNodePath,
|
|
881
|
+
sourceTextSegments: options.sourceTextSegments,
|
|
882
|
+
sourceMarkers: options.sourceMarkers
|
|
883
|
+
});
|
|
354
884
|
}
|
|
355
885
|
async synthesizeChunks(chunks, options = {}) {
|
|
356
886
|
const { region, subscriptionKey, outputFormat, signal, timeoutMs } = __privateGet(this, _options);
|
|
@@ -359,15 +889,28 @@ var AzureTtsClient = class {
|
|
|
359
889
|
endpoint,
|
|
360
890
|
region,
|
|
361
891
|
subscriptionKey,
|
|
362
|
-
outputFormat,
|
|
363
|
-
signal,
|
|
364
|
-
timeoutMs,
|
|
892
|
+
outputFormat: options.outputFormat ?? outputFormat,
|
|
893
|
+
signal: options.signal ?? signal,
|
|
894
|
+
timeoutMs: options.timeoutMs ?? timeoutMs,
|
|
895
|
+
sourceNodePath: options.sourceNodePath,
|
|
365
896
|
onProgress: options.onProgress ?? __privateGet(this, _options).onProgress
|
|
366
897
|
});
|
|
367
898
|
}
|
|
368
899
|
async synthesizeSsmlSafe(ssml, options = {}) {
|
|
369
900
|
return synthesizeSsmlSafe(this, ssml, options);
|
|
370
901
|
}
|
|
902
|
+
async synthesizeChunksSafe(chunks, options = {}) {
|
|
903
|
+
return synthesizeSsmlChunksSafe(this, chunks, {
|
|
904
|
+
...options,
|
|
905
|
+
outputFormat: options.outputFormat ?? __privateGet(this, _options).outputFormat,
|
|
906
|
+
signal: options.signal ?? __privateGet(this, _options).signal,
|
|
907
|
+
timeoutMs: options.timeoutMs ?? __privateGet(this, _options).timeoutMs,
|
|
908
|
+
onProgress: options.onProgress ?? __privateGet(this, _options).onProgress
|
|
909
|
+
});
|
|
910
|
+
}
|
|
911
|
+
async synthesizeSsmlChunksSafe(chunks, options = {}) {
|
|
912
|
+
return this.synthesizeChunksSafe(chunks, options);
|
|
913
|
+
}
|
|
371
914
|
};
|
|
372
915
|
_options = new WeakMap();
|
|
373
916
|
|
|
@@ -423,6 +966,9 @@ async function fetchAzureVoiceCatalog(options) {
|
|
|
423
966
|
const secondaryLocales = stringList(record.SecondaryLocaleList);
|
|
424
967
|
const styles = stringList(record.StyleList);
|
|
425
968
|
const status = normalizeStatus(record.Status);
|
|
969
|
+
const supportedTags = stringList(record.SupportedTags);
|
|
970
|
+
const unsupportedTags = stringList(record.UnsupportedTags);
|
|
971
|
+
const models = stringList(record.Models);
|
|
426
972
|
const merged = {
|
|
427
973
|
name: existing?.name ?? name,
|
|
428
974
|
locale: existing?.locale ?? locale,
|
|
@@ -432,6 +978,12 @@ async function fetchAzureVoiceCatalog(options) {
|
|
|
432
978
|
if (mergedSecondaryLocales.length > 0) merged.secondaryLocales = mergedSecondaryLocales;
|
|
433
979
|
const mergedStyles = [.../* @__PURE__ */ new Set([...existing?.styles ?? [], ...styles])];
|
|
434
980
|
if (mergedStyles.length > 0) merged.styles = mergedStyles;
|
|
981
|
+
const mergedSupportedTags = [.../* @__PURE__ */ new Set([...existing?.supportedTags ?? [], ...supportedTags])];
|
|
982
|
+
if (mergedSupportedTags.length > 0) merged.supportedTags = mergedSupportedTags;
|
|
983
|
+
const mergedUnsupportedTags = [.../* @__PURE__ */ new Set([...existing?.unsupportedTags ?? [], ...unsupportedTags])];
|
|
984
|
+
if (mergedUnsupportedTags.length > 0) merged.unsupportedTags = mergedUnsupportedTags;
|
|
985
|
+
const mergedModels = [.../* @__PURE__ */ new Set([...existing?.models ?? [], ...models])];
|
|
986
|
+
if (mergedModels.length > 0) merged.models = mergedModels;
|
|
435
987
|
if (status) merged.status = status;
|
|
436
988
|
else if (existing?.status) merged.status = existing.status;
|
|
437
989
|
voices.set(key, merged);
|
|
@@ -452,24 +1004,37 @@ export {
|
|
|
452
1004
|
AzureTtsClient,
|
|
453
1005
|
AzureTtsError,
|
|
454
1006
|
AzureTtsSdkError,
|
|
1007
|
+
ChunkValidationError,
|
|
1008
|
+
DEFAULT_OUTPUT_FORMAT,
|
|
1009
|
+
MergeError,
|
|
1010
|
+
SynthesisCancelledError,
|
|
1011
|
+
SynthesisTimeoutError,
|
|
1012
|
+
UnsupportedMergeFormatError,
|
|
455
1013
|
areAzureLanguagesEquivalent,
|
|
456
1014
|
buildPartialSsml,
|
|
457
1015
|
buildSsml,
|
|
1016
|
+
canMergeAudioFormat,
|
|
1017
|
+
createAzureUrlValidatorRunner,
|
|
458
1018
|
extractSsmlText,
|
|
459
1019
|
extractSsmlTranslatableText,
|
|
460
1020
|
fetchAzureVoiceCatalog,
|
|
461
1021
|
fromPlainTextToSsml,
|
|
462
1022
|
getAzureVoiceCatalogMetadata,
|
|
463
1023
|
getBuiltInVoiceCatalogMetadata,
|
|
1024
|
+
getSsmlSourceMap,
|
|
464
1025
|
isValidAzureAudioDuration,
|
|
465
1026
|
mapSsmlTextNodes,
|
|
1027
|
+
mergeAudioBuffers,
|
|
466
1028
|
mergeSynthesisResults,
|
|
467
1029
|
normalizeAzureLanguage,
|
|
468
1030
|
parseSsml,
|
|
1031
|
+
resolveMergeAudioFormat,
|
|
1032
|
+
resolveMimeType,
|
|
469
1033
|
splitSsmlDocument,
|
|
470
1034
|
synthesizeSpeech,
|
|
471
1035
|
synthesizeSsml,
|
|
472
1036
|
synthesizeSsmlChunks,
|
|
1037
|
+
synthesizeSsmlChunksSafe,
|
|
473
1038
|
synthesizeSsmlSafe,
|
|
474
1039
|
validateAzureSsml,
|
|
475
1040
|
validateSsml,
|