ssml-builder-js 2.13.0 → 2.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -2,11 +2,13 @@ import {
2
2
  areAzureLanguagesEquivalent,
3
3
  buildPartialSsml,
4
4
  buildSsml,
5
+ createAzureUrlValidatorRunner,
5
6
  extractSsmlText,
6
7
  extractSsmlTranslatableText,
7
8
  fromPlainTextToSsml,
8
9
  getAzureVoiceCatalogMetadata,
9
10
  getBuiltInVoiceCatalogMetadata,
11
+ getSsmlSourceMap,
10
12
  isValidAzureAudioDuration,
11
13
  mapSsmlTextNodes,
12
14
  normalizeAzureLanguage,
@@ -15,10 +17,11 @@ import {
15
17
  validateAzureSsml,
16
18
  validateSsml,
17
19
  validateSsmlStructureIntegrity
18
- } from "./chunk-25LOR4AJ.mjs";
20
+ } from "./chunk-FXUM45ZY.mjs";
19
21
  import {
22
+ getSsmlSourceMap as getSsmlSourceMap2,
20
23
  validateAzureSsml as validateAzureSsml2
21
- } from "./chunk-CZ2F3TET.mjs";
24
+ } from "./chunk-WXFLUCLR.mjs";
22
25
  import {
23
26
  __privateAdd,
24
27
  __privateGet,
@@ -29,6 +32,7 @@ import {
29
32
  var AzureTtsError = class extends Error {
30
33
  constructor(status, statusText, responseBody, requestId) {
31
34
  super(`Azure TTS request failed: ${status} ${statusText}`);
35
+ this.kind = "azure-api-error";
32
36
  this.name = "AzureTtsError";
33
37
  this.status = status;
34
38
  this.statusText = statusText;
@@ -44,6 +48,44 @@ var AzureTtsSdkError = class extends AzureTtsError {
44
48
  this.errorDetails = errorDetails;
45
49
  }
46
50
  };
51
+ var SynthesisCancelledError = class extends Error {
52
+ constructor(message = "Speech synthesis was cancelled.") {
53
+ super(message);
54
+ this.kind = "cancelled";
55
+ this.name = "SynthesisCancelledError";
56
+ }
57
+ };
58
+ var SynthesisTimeoutError = class extends Error {
59
+ constructor(message) {
60
+ super(message);
61
+ this.kind = "timeout";
62
+ this.name = "SynthesisTimeoutError";
63
+ }
64
+ };
65
+ var MergeError = class extends Error {
66
+ constructor(message, cause) {
67
+ super(message);
68
+ this.kind = "merge-error";
69
+ this.name = "MergeError";
70
+ this.cause = cause;
71
+ }
72
+ };
73
+ var UnsupportedMergeFormatError = class extends Error {
74
+ constructor(format) {
75
+ super(`Audio format "${format}" cannot be safely concatenated; container re-multiplexing is required.`);
76
+ this.kind = "unsupported-format-error";
77
+ this.name = "UnsupportedMergeFormatError";
78
+ this.format = format;
79
+ }
80
+ };
81
+ function toSynthesisError(error) {
82
+ if (error instanceof AzureTtsError || error instanceof MergeError || error instanceof UnsupportedMergeFormatError || error instanceof SynthesisCancelledError || error instanceof SynthesisTimeoutError)
83
+ return error;
84
+ const message = error instanceof Error ? error.message : String(error);
85
+ if (/cancel|abort/i.test(message)) return new SynthesisCancelledError(message);
86
+ if (/tim(?:e|ed) ?out/i.test(message)) return new SynthesisTimeoutError(message);
87
+ return createSpeechSdkError(error);
88
+ }
47
89
  function createSpeechSdkError(error) {
48
90
  const message = error instanceof Error ? error.message : String(error);
49
91
  return new AzureTtsSdkError(message);
@@ -52,9 +94,6 @@ function createSpeechSdkError(error) {
52
94
  // packages/azure-tts-client/src/synthesis.ts
53
95
  import * as SpeechSDK2 from "microsoft-cognitiveservices-speech-sdk";
54
96
 
55
- // packages/azure-tts-client/src/speechConfig.ts
56
- import { SpeechConfig } from "microsoft-cognitiveservices-speech-sdk";
57
-
58
97
  // packages/azure-tts-client/src/outputFormats.ts
59
98
  import * as SpeechSDK from "microsoft-cognitiveservices-speech-sdk";
60
99
  var DEFAULT_OUTPUT_FORMAT = "audio-16khz-128kbitrate-mono-mp3";
@@ -99,6 +138,14 @@ var OUTPUT_FORMATS = {
99
138
  "amr-wb-16000hz": SpeechSDK.SpeechSynthesisOutputFormat.AmrWb16000Hz,
100
139
  "g722-16khz-64kbps": SpeechSDK.SpeechSynthesisOutputFormat.G72216Khz64Kbps
101
140
  };
141
+ function resolveMimeType(outputFormat) {
142
+ if (/(?:wav|wave|riff)/i.test(outputFormat)) return "audio/wav";
143
+ if (/(?:mp3|mpeg)/i.test(outputFormat)) return "audio/mpeg";
144
+ if (/ogg/i.test(outputFormat)) return "audio/ogg";
145
+ if (/webm/i.test(outputFormat)) return "audio/webm";
146
+ if (/raw/i.test(outputFormat)) return "audio/L16";
147
+ return "application/octet-stream";
148
+ }
102
149
  function resolveOutputFormat(outputFormat) {
103
150
  const resolvedFormat = OUTPUT_FORMATS[outputFormat];
104
151
  if (resolvedFormat === void 0) {
@@ -108,6 +155,7 @@ function resolveOutputFormat(outputFormat) {
108
155
  }
109
156
 
110
157
  // packages/azure-tts-client/src/speechConfig.ts
158
+ import { SpeechConfig } from "microsoft-cognitiveservices-speech-sdk";
111
159
  function resolveEndpoint(config) {
112
160
  const endpoint = config.endpoint?.trim() || "https://{region}.tts.speech.microsoft.com/cognitiveservices/v1";
113
161
  return endpoint.replace(/\{region\}/g, encodeURIComponent(config.region));
@@ -121,6 +169,156 @@ function createSpeechConfig(config) {
121
169
  }
122
170
 
123
171
  // packages/azure-tts-client/src/synthesis.ts
172
+ function ascii(bytes, offset, value) {
173
+ return [...value].every((character, index) => bytes[offset + index] === character.charCodeAt(0));
174
+ }
175
+ function readUint32(bytes, offset) {
176
+ return new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength).getUint32(offset, true);
177
+ }
178
+ function parseWav(buffer) {
179
+ const bytes = new Uint8Array(buffer);
180
+ if (bytes.byteLength < 12 || !ascii(bytes, 0, "RIFF") || !ascii(bytes, 8, "WAVE")) {
181
+ throw new Error("Invalid WAV/RIFF audio buffer.");
182
+ }
183
+ const chunks = [];
184
+ const dataParts = [];
185
+ let format;
186
+ let offset = 12;
187
+ while (offset < bytes.byteLength) {
188
+ if (offset + 8 > bytes.byteLength) throw new Error("Invalid WAV chunk header.");
189
+ const id = String.fromCharCode(...bytes.slice(offset, offset + 4));
190
+ const size = readUint32(bytes, offset + 4);
191
+ const dataStart = offset + 8;
192
+ const dataEnd = dataStart + size;
193
+ if (dataEnd > bytes.byteLength) throw new Error(`WAV chunk "${id}" exceeds the audio buffer.`);
194
+ const data2 = bytes.slice(dataStart, dataEnd);
195
+ chunks.push({ id, data: data2 });
196
+ if (id === "fmt ") format ?? (format = data2);
197
+ if (id === "data") dataParts.push(data2);
198
+ offset = dataEnd + (size & 1);
199
+ if (offset > bytes.byteLength) throw new Error("Invalid WAV chunk padding.");
200
+ }
201
+ if (!format || dataParts.length === 0) throw new Error("WAV audio must contain fmt and data chunks.");
202
+ const dataLength = dataParts.reduce((total, part) => total + part.byteLength, 0);
203
+ const data = new Uint8Array(dataLength);
204
+ let dataOffset = 0;
205
+ for (const part of dataParts) {
206
+ data.set(part, dataOffset);
207
+ dataOffset += part.byteLength;
208
+ }
209
+ return { chunks, data, format };
210
+ }
211
+ function writeUint32(target, offset, value) {
212
+ new DataView(target.buffer).setUint32(offset, value, true);
213
+ }
214
+ function writeChunk(target, offset, id, data) {
215
+ for (let index = 0; index < 4; index += 1) target[offset + index] = id.charCodeAt(index) ?? 0;
216
+ writeUint32(target, offset + 4, data.byteLength);
217
+ target.set(data, offset + 8);
218
+ const end = offset + 8 + data.byteLength;
219
+ if (data.byteLength & 1) target[end] = 0;
220
+ return end + (data.byteLength & 1);
221
+ }
222
+ function mergeWavBuffers(buffers) {
223
+ if (buffers.length === 0) return new ArrayBuffer(0);
224
+ const parsed = buffers.map(parseWav);
225
+ const first = parsed[0];
226
+ if (!first) throw new Error("At least one WAV buffer is required.");
227
+ if (parsed.some(
228
+ (item) => item.format.length !== first.format.length || item.format.some((value, i) => value !== first.format[i])
229
+ ))
230
+ throw new Error("WAV buffers have incompatible fmt chunks.");
231
+ const dataLength = parsed.reduce((total, item) => total + item.data.byteLength, 0);
232
+ const nonDataLength = first.chunks.reduce(
233
+ (total, chunk) => chunk.id === "data" ? total : total + 8 + chunk.data.byteLength + (chunk.data.byteLength & 1),
234
+ 0
235
+ );
236
+ const outputLength = 12 + nonDataLength + 8 + dataLength + (dataLength & 1);
237
+ if (outputLength - 8 > 4294967295) throw new RangeError("Merged WAV exceeds the RIFF format size limit.");
238
+ const output = new Uint8Array(outputLength);
239
+ output.set(Uint8Array.from([82, 73, 70, 70]), 0);
240
+ writeUint32(output, 4, outputLength - 8);
241
+ output.set(Uint8Array.from([87, 65, 86, 69]), 8);
242
+ let outputOffset = 12;
243
+ let dataWritten = false;
244
+ for (const chunk of first.chunks) {
245
+ if (chunk.id === "data") {
246
+ if (dataWritten) continue;
247
+ const data = new Uint8Array(dataLength);
248
+ let dataOffset = 0;
249
+ for (const item of parsed) {
250
+ data.set(item.data, dataOffset);
251
+ dataOffset += item.data.byteLength;
252
+ }
253
+ outputOffset = writeChunk(output, outputOffset, "data", data);
254
+ dataWritten = true;
255
+ } else {
256
+ outputOffset = writeChunk(output, outputOffset, chunk.id, chunk.data);
257
+ }
258
+ }
259
+ if (!dataWritten) throw new Error("WAV audio must contain a data chunk.");
260
+ return output.buffer;
261
+ }
262
+ function skipId3v2(bytes) {
263
+ if (!ascii(bytes, 0, "ID3") || bytes.byteLength < 10) return 0;
264
+ const size = [bytes[6], bytes[7], bytes[8], bytes[9]].reduce((total, value) => total << 7 | value & 127, 0);
265
+ const hasFooter = (bytes[5] & 16) !== 0;
266
+ return Math.min(bytes.byteLength, 10 + size + (hasFooter ? 10 : 0));
267
+ }
268
+ function stripMp3Tags(buffer) {
269
+ const bytes = new Uint8Array(buffer);
270
+ const start = skipId3v2(bytes);
271
+ const end = bytes.byteLength >= 128 && ascii(bytes, bytes.byteLength - 128, "TAG") ? bytes.byteLength - 128 : bytes.byteLength;
272
+ return bytes.slice(Math.min(start, end), end);
273
+ }
274
+ function isMp3Format(format) {
275
+ return /(?:mp3|mpeg)/i.test(format);
276
+ }
277
+ function isWavFormat(format) {
278
+ return /(?:wav|wave|riff)/i.test(format);
279
+ }
280
+ function isRawFormat(format) {
281
+ return /^raw(?:-|$)/i.test(format);
282
+ }
283
+ function resolveMergeAudioFormat(format) {
284
+ if (isWavFormat(format)) return "wav";
285
+ if (isMp3Format(format)) return "mp3";
286
+ if (isRawFormat(format)) return "raw";
287
+ return void 0;
288
+ }
289
+ function canMergeAudioFormat(format) {
290
+ return resolveMergeAudioFormat(format) !== void 0;
291
+ }
292
+ function mergeAudioBuffers(buffers, options) {
293
+ const format = typeof options === "string" ? options : options?.format;
294
+ if (!format) throw new UnsupportedMergeFormatError("");
295
+ try {
296
+ if (isWavFormat(format)) return mergeWavBuffers(buffers);
297
+ if (isMp3Format(format)) {
298
+ const parts = buffers.map(stripMp3Tags);
299
+ const output = new Uint8Array(parts.reduce((total, part) => total + part.byteLength, 0));
300
+ let offset = 0;
301
+ for (const part of parts) {
302
+ output.set(part, offset);
303
+ offset += part.byteLength;
304
+ }
305
+ return output.buffer;
306
+ }
307
+ if (isRawFormat(format)) {
308
+ const output = new Uint8Array(buffers.reduce((total, buffer) => total + buffer.byteLength, 0));
309
+ let offset = 0;
310
+ for (const buffer of buffers) {
311
+ output.set(new Uint8Array(buffer), offset);
312
+ offset += buffer.byteLength;
313
+ }
314
+ return output.buffer;
315
+ }
316
+ throw new UnsupportedMergeFormatError(format);
317
+ } catch (error) {
318
+ if (error instanceof UnsupportedMergeFormatError || error instanceof MergeError) throw error;
319
+ throw new MergeError(`Audio buffers could not be merged for format "${format}".`, error);
320
+ }
321
+ }
124
322
  function closeSpeechResources(speechConfig, synthesizer) {
125
323
  try {
126
324
  synthesizer.close();
@@ -134,7 +332,7 @@ function closeSpeechResources(speechConfig, synthesizer) {
134
332
  var ticksToMilliseconds = (ticks) => Math.max(0, ticks) / 1e4;
135
333
  async function synthesizeSsml(ssml, config) {
136
334
  if (config.signal?.aborted) {
137
- throw createSpeechSdkError("Speech synthesis was cancelled.");
335
+ throw new SynthesisCancelledError();
138
336
  }
139
337
  const speechConfig = createSpeechConfig(config);
140
338
  const synthesizer = new SpeechSDK2.SpeechSynthesizer(speechConfig, null);
@@ -157,23 +355,94 @@ async function synthesizeSsml(ssml, config) {
157
355
  settled = true;
158
356
  cleanup();
159
357
  closeResources();
160
- reject(createSpeechSdkError(error));
358
+ reject(toSynthesisError(error));
161
359
  };
162
360
  const boundaries = [];
163
361
  const visemes = [];
164
362
  const bookmarks = [];
363
+ let sourceEventCursor = 0;
364
+ let generatedSourceMap;
365
+ if (!config.sourceTextSegments && !config.sourceMarkers) {
366
+ try {
367
+ generatedSourceMap = getSsmlSourceMap2(ssml);
368
+ } catch {
369
+ generatedSourceMap = void 0;
370
+ }
371
+ }
372
+ const sourceBaseOffset = config.sourceTextRange?.start ?? 0;
373
+ const sourceSegments = config.sourceTextSegments ?? generatedSourceMap?.segments.map((segment) => ({
374
+ ...segment,
375
+ range: {
376
+ start: segment.range.start + sourceBaseOffset,
377
+ end: segment.range.end + sourceBaseOffset
378
+ },
379
+ sourceNodePath: [...segment.sourceNodePath]
380
+ })) ?? [];
381
+ const sourceMarkers = config.sourceMarkers ?? generatedSourceMap?.markers.map((marker) => ({
382
+ ...marker,
383
+ originalTextRange: {
384
+ start: marker.originalTextRange.start + sourceBaseOffset,
385
+ end: marker.originalTextRange.end + sourceBaseOffset
386
+ },
387
+ sourceNodePath: [...marker.sourceNodePath]
388
+ })) ?? [];
389
+ const sourceText = sourceSegments.map((segment) => segment.text).join("");
390
+ const mapSourceEvent = (text, offsetHint, markerName) => {
391
+ const marker = markerName ? sourceMarkers.find((candidate) => candidate.name === markerName) : void 0;
392
+ if (marker) {
393
+ return {
394
+ originalTextRange: { ...marker.originalTextRange },
395
+ sourceNodePath: [...marker.sourceNodePath],
396
+ textRange: { ...marker.originalTextRange }
397
+ };
398
+ }
399
+ if (sourceSegments.length === 0 && !config.sourceTextRange && !config.sourceNodePath) return {};
400
+ const value = text ?? "";
401
+ let localStart = Number.isFinite(offsetHint) && (offsetHint ?? 0) >= 0 ? offsetHint : -1;
402
+ if (value && localStart >= 0 && sourceText.slice(localStart, localStart + value.length) !== value)
403
+ localStart = -1;
404
+ if (localStart < 0 || localStart > sourceText.length) {
405
+ localStart = value ? sourceText.indexOf(value, sourceEventCursor) : sourceEventCursor;
406
+ if (localStart < 0) localStart = value ? sourceText.indexOf(value) : sourceEventCursor;
407
+ }
408
+ localStart = Math.max(0, localStart);
409
+ const localEnd = Math.min(sourceText.length, localStart + value.length);
410
+ sourceEventCursor = Math.max(sourceEventCursor, localEnd);
411
+ const baseStart = config.sourceTextRange?.start ?? sourceSegments[0]?.range.start ?? 0;
412
+ const fallbackRange = { start: baseStart + localStart, end: baseStart + localEnd };
413
+ const segment = sourceSegments.find(({ range }) => range.start <= fallbackRange.start && range.end > fallbackRange.start) ?? sourceSegments.find(({ range }) => range.end > fallbackRange.start) ?? (value.length === 0 ? sourceSegments.find(({ range }) => range.start <= fallbackRange.start && range.end >= fallbackRange.start) : void 0);
414
+ return {
415
+ originalTextRange: { ...fallbackRange },
416
+ textRange: { ...fallbackRange },
417
+ ...segment ? { sourceNodePath: [...segment.sourceNodePath] } : config.sourceNodePath ? { sourceNodePath: [...config.sourceNodePath] } : {}
418
+ };
419
+ };
165
420
  synthesizer.wordBoundary = (_sender, event) => {
166
421
  boundaries.push({
167
422
  text: event.text,
168
423
  audioOffsetMs: ticksToMilliseconds(event.audioOffset),
169
- durationMs: ticksToMilliseconds(event.duration)
424
+ durationMs: ticksToMilliseconds(event.duration),
425
+ ...mapSourceEvent(
426
+ event.text,
427
+ event.textOffset
428
+ )
170
429
  });
171
430
  };
172
431
  synthesizer.visemeReceived = (_sender, event) => {
173
- visemes.push({ visemeId: event.visemeId, audioOffsetMs: ticksToMilliseconds(event.audioOffset) });
432
+ const eventWithOffset = event;
433
+ visemes.push({
434
+ visemeId: event.visemeId,
435
+ audioOffsetMs: ticksToMilliseconds(event.audioOffset),
436
+ ...mapSourceEvent(void 0, eventWithOffset.textOffset)
437
+ });
174
438
  };
175
439
  synthesizer.bookmarkReached = (_sender, event) => {
176
- bookmarks.push({ name: event.text, audioOffsetMs: ticksToMilliseconds(event.audioOffset) });
440
+ const eventWithOffset = event;
441
+ bookmarks.push({
442
+ name: event.text,
443
+ audioOffsetMs: ticksToMilliseconds(event.audioOffset),
444
+ ...mapSourceEvent(void 0, eventWithOffset.textOffset, event.text)
445
+ });
177
446
  };
178
447
  const cb = (result) => {
179
448
  if (settled) return;
@@ -196,7 +465,10 @@ async function synthesizeSsml(ssml, config) {
196
465
  const requestId = result.resultId;
197
466
  const addSourceMetadata = (event) => ({
198
467
  ...event,
199
- ...config.sourceTextRange ? { textRange: { ...config.sourceTextRange } } : {},
468
+ ...config.sourceTextRange && !("textRange" in event) ? { textRange: { ...config.sourceTextRange } } : {},
469
+ ...config.sourceTextRange && !("originalTextRange" in event) ? { originalTextRange: { ...config.sourceTextRange } } : {},
470
+ ...config.chunkIndex !== void 0 ? { chunkIndex: config.chunkIndex } : {},
471
+ ...config.sourceNodePath ? { sourceNodePath: [...config.sourceNodePath] } : {},
200
472
  ...requestId ? { requestId } : {}
201
473
  });
202
474
  const sourceBoundaries = boundaries.map((boundary) => addSourceMetadata(boundary));
@@ -214,12 +486,12 @@ async function synthesizeSsml(ssml, config) {
214
486
  };
215
487
  try {
216
488
  if (config.signal) {
217
- abortHandler = () => rejectWithError("Speech synthesis was cancelled.");
489
+ abortHandler = () => rejectWithError(new SynthesisCancelledError());
218
490
  config.signal.addEventListener("abort", abortHandler, { once: true });
219
491
  }
220
492
  if (config.timeoutMs !== void 0 && config.timeoutMs > 0) {
221
493
  timeout = setTimeout(
222
- () => rejectWithError(`Speech synthesis timed out after ${config.timeoutMs} ms.`),
494
+ () => rejectWithError(new SynthesisTimeoutError(`Speech synthesis timed out after ${config.timeoutMs} ms.`)),
223
495
  config.timeoutMs
224
496
  );
225
497
  }
@@ -232,60 +504,117 @@ async function synthesizeSsml(ssml, config) {
232
504
  async function synthesizeSsmlChunks(chunks, config) {
233
505
  const results = [];
234
506
  const totalChunks = chunks.length;
507
+ const report = (event) => config.onProgress?.(event);
235
508
  for (const [index, chunk] of chunks.entries()) {
236
509
  const input = typeof chunk === "string" ? { ssml: chunk } : chunk;
237
- const result = await synthesizeSsml(input.ssml, {
238
- ...config,
239
- ...input.originalTextRange ? { sourceTextRange: input.originalTextRange } : {},
240
- onProgress: void 0
510
+ report({
511
+ currentChunk: index,
512
+ totalChunks,
513
+ percent: totalChunks === 0 ? 100 : Math.round(index / totalChunks * 100),
514
+ chunkIndex: index,
515
+ originalTextRange: input.originalTextRange,
516
+ status: "pending",
517
+ durationMs: 0
241
518
  });
242
- results.push(result);
243
- config.onProgress?.({
244
- currentChunk: index + 1,
519
+ }
520
+ for (const [index, chunk] of chunks.entries()) {
521
+ const input = typeof chunk === "string" ? { ssml: chunk } : chunk;
522
+ report({
523
+ currentChunk: index,
245
524
  totalChunks,
246
- percent: totalChunks === 0 ? 100 : Math.round((index + 1) / totalChunks * 100)
525
+ percent: totalChunks === 0 ? 100 : Math.round(index / totalChunks * 100),
526
+ chunkIndex: index,
527
+ originalTextRange: input.originalTextRange,
528
+ status: "synthesizing",
529
+ durationMs: 0
247
530
  });
531
+ const startedAt = Date.now();
532
+ try {
533
+ const result = await synthesizeSsml(input.ssml, {
534
+ ...config,
535
+ ...input.originalTextRange ? { sourceTextRange: input.originalTextRange } : {},
536
+ ...input.sourceNodePath ?? config.sourceNodePath ? { sourceNodePath: [...input.sourceNodePath ?? config.sourceNodePath ?? []] } : {},
537
+ ...input.sourceTextSegments ? { sourceTextSegments: input.sourceTextSegments } : {},
538
+ ...input.sourceMarkers ? { sourceMarkers: input.sourceMarkers } : {},
539
+ chunkIndex: index,
540
+ onProgress: void 0
541
+ });
542
+ results.push(result);
543
+ report({
544
+ currentChunk: index + 1,
545
+ totalChunks,
546
+ percent: totalChunks === 0 ? 100 : Math.round((index + 1) / totalChunks * 100),
547
+ chunkIndex: index,
548
+ originalTextRange: input.originalTextRange,
549
+ status: "success",
550
+ durationMs: Date.now() - startedAt
551
+ });
552
+ } catch (error) {
553
+ report({
554
+ currentChunk: index,
555
+ totalChunks,
556
+ percent: totalChunks === 0 ? 100 : Math.round(index / totalChunks * 100),
557
+ chunkIndex: index,
558
+ originalTextRange: input.originalTextRange,
559
+ status: "failed",
560
+ durationMs: Date.now() - startedAt,
561
+ error
562
+ });
563
+ throw error;
564
+ }
248
565
  }
249
- return mergeSynthesisResults(results);
566
+ return mergeSynthesisResults(results, {
567
+ format: config.outputFormat ?? "audio-16khz-128kbitrate-mono-mp3"
568
+ });
250
569
  }
251
- function mergeSynthesisResults(results) {
252
- const audioLength = results.reduce((total, result) => total + result.audioData.byteLength, 0);
253
- const audioData = new Uint8Array(audioLength);
570
+ function createMergedResult(results, audioData, format) {
254
571
  const boundaries = [];
255
572
  const visemes = [];
256
573
  const bookmarks = [];
257
- let byteOffset = 0;
258
574
  let durationOffset = 0;
259
- for (const result of results) {
260
- audioData.set(new Uint8Array(result.audioData), byteOffset);
261
- byteOffset += result.audioData.byteLength;
575
+ for (const [resultIndex, result] of results.entries()) {
262
576
  const chunkBoundaries = result.boundaries && result.boundaries.length > 0 ? result.boundaries : result.wordBoundary ?? result.wordBoundaries ?? [];
263
577
  for (const boundary of chunkBoundaries) {
264
578
  const textRange = boundary.textRange ?? result.textRange;
579
+ const originalTextRange = boundary.originalTextRange ?? textRange;
265
580
  const requestId = boundary.requestId ?? result.requestId;
266
581
  boundaries.push({
267
582
  ...boundary,
268
583
  audioOffsetMs: boundary.audioOffsetMs + durationOffset,
584
+ chunkAudioOffsetMs: boundary.chunkAudioOffsetMs ?? boundary.audioOffsetMs,
585
+ ...boundary.chunkIndex === void 0 ? { chunkIndex: resultIndex } : {},
586
+ ...boundary.sourceNodePath ? { sourceNodePath: [...boundary.sourceNodePath] } : {},
587
+ ...originalTextRange ? { originalTextRange: { ...originalTextRange } } : {},
269
588
  ...textRange ? { textRange: { ...textRange } } : {},
270
589
  ...requestId ? { requestId } : {}
271
590
  });
272
591
  }
273
592
  for (const viseme of result.visemes ?? []) {
274
593
  const textRange = viseme.textRange ?? result.textRange;
594
+ const originalTextRange = viseme.originalTextRange ?? textRange;
275
595
  const requestId = viseme.requestId ?? result.requestId;
276
596
  visemes.push({
277
597
  ...viseme,
278
598
  audioOffsetMs: viseme.audioOffsetMs + durationOffset,
599
+ chunkAudioOffsetMs: viseme.chunkAudioOffsetMs ?? viseme.audioOffsetMs,
600
+ ...viseme.chunkIndex === void 0 ? { chunkIndex: resultIndex } : {},
601
+ ...viseme.sourceNodePath ? { sourceNodePath: [...viseme.sourceNodePath] } : {},
602
+ ...originalTextRange ? { originalTextRange: { ...originalTextRange } } : {},
279
603
  ...textRange ? { textRange: { ...textRange } } : {},
280
604
  ...requestId ? { requestId } : {}
281
605
  });
282
606
  }
283
607
  for (const bookmark of result.bookmarks ?? []) {
284
608
  const textRange = bookmark.textRange ?? result.textRange;
609
+ const originalTextRange = bookmark.originalTextRange ?? textRange;
285
610
  const requestId = bookmark.requestId ?? result.requestId;
286
611
  bookmarks.push({
287
612
  ...bookmark,
288
613
  audioOffsetMs: bookmark.audioOffsetMs + durationOffset,
614
+ chunkAudioOffsetMs: bookmark.chunkAudioOffsetMs ?? bookmark.audioOffsetMs,
615
+ ...bookmark.chunkIndex === void 0 ? { chunkIndex: resultIndex } : {},
616
+ ...bookmark.sourceNodePath ? { sourceNodePath: [...bookmark.sourceNodePath] } : {},
617
+ ...originalTextRange ? { originalTextRange: { ...originalTextRange } } : {},
289
618
  ...textRange ? { textRange: { ...textRange } } : {},
290
619
  ...requestId ? { requestId } : {}
291
620
  });
@@ -293,8 +622,9 @@ function mergeSynthesisResults(results) {
293
622
  durationOffset += Math.max(0, result.durationMs);
294
623
  }
295
624
  return {
296
- audioData: audioData.buffer,
625
+ audioData,
297
626
  durationMs: durationOffset,
627
+ mimeType: resolveMimeType(format),
298
628
  ...boundaries.length > 0 ? { boundaries, wordBoundary: boundaries, wordBoundaries: boundaries } : {},
299
629
  ...visemes.length > 0 ? { visemes } : {},
300
630
  ...bookmarks.length > 0 ? { bookmarks } : {},
@@ -302,34 +632,224 @@ function mergeSynthesisResults(results) {
302
632
  ...results.length === 1 && results[0]?.textRange ? { textRange: { ...results[0].textRange } } : {}
303
633
  };
304
634
  }
635
+ function mergeSynthesisResults(results, options) {
636
+ const resolvedOptions = typeof options === "string" ? { format: options } : options;
637
+ const format = resolvedOptions?.format;
638
+ if (!format) throw new UnsupportedMergeFormatError("");
639
+ const buffers = results.map((result) => result.audioData);
640
+ if (resolvedOptions.customMerger) {
641
+ return Promise.resolve().then(() => resolvedOptions.customMerger?.(buffers, format)).then((merged) => {
642
+ if (!merged) throw new MergeError("The custom audio merger returned no audio buffer.");
643
+ return createMergedResult(results, merged, format);
644
+ }).catch((error) => {
645
+ if (error instanceof UnsupportedMergeFormatError || error instanceof MergeError) throw error;
646
+ throw new MergeError(`Custom audio merger failed for format "${format}".`, error);
647
+ });
648
+ }
649
+ try {
650
+ return createMergedResult(results, mergeAudioBuffers(buffers, { format }), format);
651
+ } catch (error) {
652
+ if (error instanceof UnsupportedMergeFormatError || error instanceof MergeError) throw error;
653
+ throw new MergeError(`Audio buffers could not be merged for format "${format}".`, error);
654
+ }
655
+ }
305
656
  async function synthesizeSpeech(ssml, config) {
306
657
  return (await synthesizeSsml(ssml, config)).audioData;
307
658
  }
308
659
 
309
660
  // packages/azure-tts-client/src/safe.ts
661
+ var ChunkValidationError = class extends Error {
662
+ constructor(chunkIndex, diagnostics) {
663
+ super(`SSML validation failed for chunk ${chunkIndex}; the Azure Speech API was not called.`);
664
+ this.kind = "validation-error";
665
+ this.name = "ChunkValidationError";
666
+ this.chunkIndex = chunkIndex;
667
+ this.diagnostics = diagnostics;
668
+ }
669
+ };
670
+ function failure(error) {
671
+ return { ok: false, success: false, status: error.kind, error };
672
+ }
310
673
  async function synthesizeSsmlSafe(client, ssml, options = {}) {
311
- const validationOptions = options.validation ?? options;
674
+ const validationOptions = withValidationSignal(options.validation ?? options, options.signal);
312
675
  const diagnostics = await Promise.resolve(validateAzureSsml2(ssml, validationOptions));
676
+ if (options.signal?.aborted) {
677
+ const error = toSynthesisError(new Error("Speech synthesis was cancelled."));
678
+ return failure(error);
679
+ }
313
680
  const errors = diagnostics.filter((diagnostic) => diagnostic.severity === "error");
314
681
  if (errors.length > 0) {
682
+ return failure({
683
+ kind: "validation-error",
684
+ message: "SSML validation failed; the Azure Speech API was not called.",
685
+ diagnostics: errors
686
+ });
687
+ }
688
+ try {
315
689
  return {
316
- ok: false,
317
- success: false,
318
- status: "validation-error",
319
- error: {
320
- kind: "validation",
321
- message: "SSML validation failed; the Azure Speech API was not called.",
322
- diagnostics: errors
323
- }
690
+ ok: true,
691
+ success: true,
692
+ status: "success",
693
+ value: await client.synthesizeSsml(ssml, { signal: options.signal })
324
694
  };
695
+ } catch (error) {
696
+ const synthesisError = toSynthesisError(error);
697
+ return failure(synthesisError);
698
+ }
699
+ }
700
+ async function synthesizeSsmlChunksSafe(client, chunks, options = {}) {
701
+ const validationOptions = withValidationSignal(options.validation ?? options, options.signal);
702
+ if (options.signal?.aborted) {
703
+ const error = toSynthesisError(new Error("Speech synthesis was cancelled."));
704
+ return failure(error);
705
+ }
706
+ const pending = (index, status, error) => {
707
+ options.onProgress?.({
708
+ currentChunk: status === "success" ? index + 1 : index,
709
+ totalChunks: chunks.length,
710
+ percent: chunks.length === 0 ? 100 : Math.round((status === "success" ? index + 1 : index) / chunks.length * 100),
711
+ chunkIndex: index,
712
+ originalTextRange: typeof chunks[index] === "string" ? void 0 : chunks[index]?.originalTextRange,
713
+ status,
714
+ durationMs: 0,
715
+ ...error ? { error } : {}
716
+ });
717
+ };
718
+ chunks.forEach((_chunk, index) => {
719
+ pending(index, "pending");
720
+ });
721
+ const validations = await Promise.all(
722
+ chunks.map(async (chunk) => {
723
+ const ssml = typeof chunk === "string" ? chunk : chunk.ssml;
724
+ const sourceNodePath = typeof chunk === "string" ? options.sourceNodePath : chunk.sourceNodePath ?? options.sourceNodePath;
725
+ const diagnostics = await Promise.resolve(
726
+ validateAzureSsml2(ssml, { ...validationOptions, ...sourceNodePath ? { sourceNodePath } : {} })
727
+ );
728
+ return diagnostics.filter((diagnostic) => diagnostic.severity === "error");
729
+ })
730
+ );
731
+ const firstInvalidIndex = validations.findIndex((diagnostics) => diagnostics.length > 0);
732
+ if (firstInvalidIndex >= 0) {
733
+ const error = new ChunkValidationError(firstInvalidIndex, validations[firstInvalidIndex] ?? []);
734
+ pending(firstInvalidIndex, "failed", error);
735
+ return failure(error);
325
736
  }
326
737
  try {
327
- return { ok: true, success: true, status: "success", value: await client.synthesizeSsml(ssml) };
738
+ if (client.synthesizeChunks) {
739
+ const normalizedChunks = chunks.map((chunk) => {
740
+ if (typeof chunk === "string" || chunk.sourceNodePath || !options.sourceNodePath) return chunk;
741
+ return { ...chunk, sourceNodePath: [...options.sourceNodePath] };
742
+ });
743
+ const value = await client.synthesizeChunks(normalizedChunks, {
744
+ onProgress: options.onProgress,
745
+ outputFormat: options.outputFormat,
746
+ signal: options.signal,
747
+ timeoutMs: options.timeoutMs,
748
+ sourceNodePath: options.sourceNodePath
749
+ });
750
+ return { ok: true, success: true, status: "success", value };
751
+ }
752
+ const results = [];
753
+ for (const [index, chunk] of chunks.entries()) {
754
+ const input = typeof chunk === "string" ? { ssml: chunk } : chunk;
755
+ const sourceNodePath = input.sourceNodePath;
756
+ const originalTextRange = input.originalTextRange;
757
+ pending(index, "synthesizing");
758
+ const startedAt = Date.now();
759
+ try {
760
+ const result = await client.synthesizeSsml(input.ssml, {
761
+ outputFormat: options.outputFormat,
762
+ signal: options.signal,
763
+ timeoutMs: options.timeoutMs,
764
+ sourceNodePath: input.sourceNodePath ?? options.sourceNodePath
765
+ });
766
+ results.push({
767
+ ...result,
768
+ ...input.originalTextRange ? { textRange: { ...input.originalTextRange } } : {},
769
+ ...sourceNodePath ? {
770
+ boundaries: result.boundaries?.map((event) => ({
771
+ ...event,
772
+ sourceNodePath: [...sourceNodePath],
773
+ ...event.originalTextRange ? { originalTextRange: { ...event.originalTextRange } } : input.originalTextRange ? { originalTextRange: { ...input.originalTextRange } } : {}
774
+ })),
775
+ visemes: result.visemes?.map((event) => ({
776
+ ...event,
777
+ sourceNodePath: [...sourceNodePath],
778
+ ...event.originalTextRange ? { originalTextRange: { ...event.originalTextRange } } : input.originalTextRange ? { originalTextRange: { ...input.originalTextRange } } : {}
779
+ })),
780
+ bookmarks: result.bookmarks?.map((event) => ({
781
+ ...event,
782
+ sourceNodePath: [...sourceNodePath],
783
+ ...event.originalTextRange ? { originalTextRange: { ...event.originalTextRange } } : input.originalTextRange ? { originalTextRange: { ...input.originalTextRange } } : {}
784
+ }))
785
+ } : {},
786
+ ...originalTextRange ? {
787
+ boundaries: result.boundaries?.map((event) => ({
788
+ ...event,
789
+ originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
790
+ })),
791
+ wordBoundary: result.wordBoundary?.map((event) => ({
792
+ ...event,
793
+ originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
794
+ })),
795
+ wordBoundaries: result.wordBoundaries?.map((event) => ({
796
+ ...event,
797
+ originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
798
+ })),
799
+ visemes: result.visemes?.map((event) => ({
800
+ ...event,
801
+ originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
802
+ })),
803
+ bookmarks: result.bookmarks?.map((event) => ({
804
+ ...event,
805
+ originalTextRange: event.originalTextRange ? { ...event.originalTextRange } : { ...originalTextRange }
806
+ }))
807
+ } : {}
808
+ });
809
+ options.onProgress?.({
810
+ currentChunk: index + 1,
811
+ totalChunks: chunks.length,
812
+ percent: chunks.length === 0 ? 100 : Math.round((index + 1) / chunks.length * 100),
813
+ chunkIndex: index,
814
+ originalTextRange: input.originalTextRange,
815
+ status: "success",
816
+ durationMs: Date.now() - startedAt
817
+ });
818
+ } catch (error) {
819
+ options.onProgress?.({
820
+ currentChunk: index,
821
+ totalChunks: chunks.length,
822
+ percent: chunks.length === 0 ? 100 : Math.round(index / chunks.length * 100),
823
+ chunkIndex: index,
824
+ originalTextRange: input.originalTextRange,
825
+ status: "failed",
826
+ durationMs: Date.now() - startedAt,
827
+ error
828
+ });
829
+ throw error;
830
+ }
831
+ }
832
+ return {
833
+ ok: true,
834
+ success: true,
835
+ status: "success",
836
+ value: mergeSynthesisResults(results, {
837
+ format: options.outputFormat ?? "audio-16khz-128kbitrate-mono-mp3"
838
+ })
839
+ };
328
840
  } catch (error) {
329
- const azureError = error instanceof AzureTtsError ? error : createSpeechSdkError(error);
330
- return { ok: false, success: false, status: "azure-api-error", error: azureError };
841
+ const synthesisError = toSynthesisError(error);
842
+ return failure(synthesisError);
331
843
  }
332
844
  }
845
+ function withValidationSignal(options, signal) {
846
+ if (!signal) return options;
847
+ return {
848
+ ...options,
849
+ urlValidatorSignal: signal,
850
+ urlValidation: { ...options.urlValidation ?? {}, signal }
851
+ };
852
+ }
333
853
 
334
854
  // packages/azure-tts-client/src/client.ts
335
855
  var ENDPOINT_TEMPLATE = "https://{region}.tts.speech.microsoft.com/cognitiveservices/v1";
@@ -346,11 +866,21 @@ var AzureTtsClient = class {
346
866
  const config = { endpoint, region, subscriptionKey, outputFormat, signal, timeoutMs };
347
867
  return synthesizeSpeech(ssml, config);
348
868
  }
349
- async synthesizeSsml(ssml) {
869
+ async synthesizeSsml(ssml, options = {}) {
350
870
  const { region, subscriptionKey, outputFormat, signal, timeoutMs } = __privateGet(this, _options);
351
871
  const endpoint = __privateGet(this, _options).endpoint?.trim() || ENDPOINT_TEMPLATE.replace("{region}", region);
352
872
  __privateGet(this, _options).logger?.debug?.("Using Azure TTS endpoint:", endpoint);
353
- return synthesizeSsml(ssml, { endpoint, region, subscriptionKey, outputFormat, signal, timeoutMs });
873
+ return synthesizeSsml(ssml, {
874
+ endpoint,
875
+ region,
876
+ subscriptionKey,
877
+ outputFormat: options.outputFormat ?? outputFormat,
878
+ signal: options.signal ?? signal,
879
+ timeoutMs: options.timeoutMs ?? timeoutMs,
880
+ sourceNodePath: options.sourceNodePath,
881
+ sourceTextSegments: options.sourceTextSegments,
882
+ sourceMarkers: options.sourceMarkers
883
+ });
354
884
  }
355
885
  async synthesizeChunks(chunks, options = {}) {
356
886
  const { region, subscriptionKey, outputFormat, signal, timeoutMs } = __privateGet(this, _options);
@@ -359,15 +889,28 @@ var AzureTtsClient = class {
359
889
  endpoint,
360
890
  region,
361
891
  subscriptionKey,
362
- outputFormat,
363
- signal,
364
- timeoutMs,
892
+ outputFormat: options.outputFormat ?? outputFormat,
893
+ signal: options.signal ?? signal,
894
+ timeoutMs: options.timeoutMs ?? timeoutMs,
895
+ sourceNodePath: options.sourceNodePath,
365
896
  onProgress: options.onProgress ?? __privateGet(this, _options).onProgress
366
897
  });
367
898
  }
368
899
  async synthesizeSsmlSafe(ssml, options = {}) {
369
900
  return synthesizeSsmlSafe(this, ssml, options);
370
901
  }
902
+ async synthesizeChunksSafe(chunks, options = {}) {
903
+ return synthesizeSsmlChunksSafe(this, chunks, {
904
+ ...options,
905
+ outputFormat: options.outputFormat ?? __privateGet(this, _options).outputFormat,
906
+ signal: options.signal ?? __privateGet(this, _options).signal,
907
+ timeoutMs: options.timeoutMs ?? __privateGet(this, _options).timeoutMs,
908
+ onProgress: options.onProgress ?? __privateGet(this, _options).onProgress
909
+ });
910
+ }
911
+ async synthesizeSsmlChunksSafe(chunks, options = {}) {
912
+ return this.synthesizeChunksSafe(chunks, options);
913
+ }
371
914
  };
372
915
  _options = new WeakMap();
373
916
 
@@ -423,6 +966,9 @@ async function fetchAzureVoiceCatalog(options) {
423
966
  const secondaryLocales = stringList(record.SecondaryLocaleList);
424
967
  const styles = stringList(record.StyleList);
425
968
  const status = normalizeStatus(record.Status);
969
+ const supportedTags = stringList(record.SupportedTags);
970
+ const unsupportedTags = stringList(record.UnsupportedTags);
971
+ const models = stringList(record.Models);
426
972
  const merged = {
427
973
  name: existing?.name ?? name,
428
974
  locale: existing?.locale ?? locale,
@@ -432,6 +978,12 @@ async function fetchAzureVoiceCatalog(options) {
432
978
  if (mergedSecondaryLocales.length > 0) merged.secondaryLocales = mergedSecondaryLocales;
433
979
  const mergedStyles = [.../* @__PURE__ */ new Set([...existing?.styles ?? [], ...styles])];
434
980
  if (mergedStyles.length > 0) merged.styles = mergedStyles;
981
+ const mergedSupportedTags = [.../* @__PURE__ */ new Set([...existing?.supportedTags ?? [], ...supportedTags])];
982
+ if (mergedSupportedTags.length > 0) merged.supportedTags = mergedSupportedTags;
983
+ const mergedUnsupportedTags = [.../* @__PURE__ */ new Set([...existing?.unsupportedTags ?? [], ...unsupportedTags])];
984
+ if (mergedUnsupportedTags.length > 0) merged.unsupportedTags = mergedUnsupportedTags;
985
+ const mergedModels = [.../* @__PURE__ */ new Set([...existing?.models ?? [], ...models])];
986
+ if (mergedModels.length > 0) merged.models = mergedModels;
435
987
  if (status) merged.status = status;
436
988
  else if (existing?.status) merged.status = existing.status;
437
989
  voices.set(key, merged);
@@ -452,24 +1004,37 @@ export {
452
1004
  AzureTtsClient,
453
1005
  AzureTtsError,
454
1006
  AzureTtsSdkError,
1007
+ ChunkValidationError,
1008
+ DEFAULT_OUTPUT_FORMAT,
1009
+ MergeError,
1010
+ SynthesisCancelledError,
1011
+ SynthesisTimeoutError,
1012
+ UnsupportedMergeFormatError,
455
1013
  areAzureLanguagesEquivalent,
456
1014
  buildPartialSsml,
457
1015
  buildSsml,
1016
+ canMergeAudioFormat,
1017
+ createAzureUrlValidatorRunner,
458
1018
  extractSsmlText,
459
1019
  extractSsmlTranslatableText,
460
1020
  fetchAzureVoiceCatalog,
461
1021
  fromPlainTextToSsml,
462
1022
  getAzureVoiceCatalogMetadata,
463
1023
  getBuiltInVoiceCatalogMetadata,
1024
+ getSsmlSourceMap,
464
1025
  isValidAzureAudioDuration,
465
1026
  mapSsmlTextNodes,
1027
+ mergeAudioBuffers,
466
1028
  mergeSynthesisResults,
467
1029
  normalizeAzureLanguage,
468
1030
  parseSsml,
1031
+ resolveMergeAudioFormat,
1032
+ resolveMimeType,
469
1033
  splitSsmlDocument,
470
1034
  synthesizeSpeech,
471
1035
  synthesizeSsml,
472
1036
  synthesizeSsmlChunks,
1037
+ synthesizeSsmlChunksSafe,
473
1038
  synthesizeSsmlSafe,
474
1039
  validateAzureSsml,
475
1040
  validateSsml,