echogarden 0.11.12 → 0.11.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/data/schemas/options.json +16 -0
  2. package/dist/api/Alignment.js +2 -2
  3. package/dist/api/Alignment.js.map +1 -1
  4. package/dist/api/Recognition.js +2 -2
  5. package/dist/api/Recognition.js.map +1 -1
  6. package/dist/api/Synthesis.js +5 -4
  7. package/dist/api/Synthesis.js.map +1 -1
  8. package/dist/api/Translation.js +2 -2
  9. package/dist/api/Translation.js.map +1 -1
  10. package/dist/audio/AudioUtilities.d.ts +1 -0
  11. package/dist/audio/AudioUtilities.js +25 -7
  12. package/dist/audio/AudioUtilities.js.map +1 -1
  13. package/dist/cli/CLI.js +2 -2
  14. package/dist/cli/CLI.js.map +1 -1
  15. package/dist/recognition/WhisperSTT.js +2 -2
  16. package/dist/recognition/WhisperSTT.js.map +1 -1
  17. package/dist/subtitles/Subtitles.d.ts +10 -7
  18. package/dist/subtitles/Subtitles.js +268 -207
  19. package/dist/subtitles/Subtitles.js.map +1 -1
  20. package/docs/Options.md +4 -2
  21. package/package.json +7 -6
  22. package/src/alignment/DTWMfccSequenceAlignment.ts +43 -0
  23. package/src/alignment/DTWSequenceAlignment.ts +121 -0
  24. package/src/alignment/DTWSequenceAlignmentWindowed.ts +210 -0
  25. package/src/alignment/LevenshteinSequenceAlignment.ts +126 -0
  26. package/src/alignment/SpeechAlignment.ts +488 -0
  27. package/src/api/API.ts +12 -0
  28. package/src/api/APIOptions.ts +15 -0
  29. package/src/api/Alignment.ts +329 -0
  30. package/src/api/Common.ts +16 -0
  31. package/src/api/Denoising.ts +120 -0
  32. package/src/api/LanguageDetection.ts +286 -0
  33. package/src/api/Recognition.ts +344 -0
  34. package/src/api/Synthesis.ts +1735 -0
  35. package/src/api/Translation.ts +143 -0
  36. package/src/api/Vad.ts +172 -0
  37. package/src/audio/AudioBufferConversion.ts +248 -0
  38. package/src/audio/AudioPlayer.ts +358 -0
  39. package/src/audio/AudioRecorder.ts +91 -0
  40. package/src/audio/AudioUtilities.ts +392 -0
  41. package/src/audio/SoxPath.ts +24 -0
  42. package/src/cli/CLI.ts +1360 -0
  43. package/src/cli/CLIConfigFile.ts +91 -0
  44. package/src/cli/CLILauncher.ts +26 -0
  45. package/src/cli/CLIOptionsSchema.ts +54 -0
  46. package/src/cli/CLIParser.ts +41 -0
  47. package/src/cli/CLIStarter.ts +40 -0
  48. package/src/codecs/FFMpegTranscoder.ts +214 -0
  49. package/src/codecs/TIMITCodec.ts +17 -0
  50. package/src/codecs/WaveCodec.ts +260 -0
  51. package/src/denoising/RNNoise.ts +95 -0
  52. package/src/dsp/BiquadFilter.ts +488 -0
  53. package/src/dsp/FFT.ts +187 -0
  54. package/src/dsp/MFCC.ts +227 -0
  55. package/src/dsp/MelSpectogram.ts +145 -0
  56. package/src/dsp/Rubberband.ts +249 -0
  57. package/src/dsp/Sonic.ts +59 -0
  58. package/src/dsp/SpeexResampler.ts +79 -0
  59. package/src/math/VectorMath.ts +812 -0
  60. package/src/nlp/ChineseSegmentation.ts +68 -0
  61. package/src/nlp/CompromiseNLP.ts +113 -0
  62. package/src/nlp/EspeakPhonemizer.ts +168 -0
  63. package/src/nlp/IPA.ts +139 -0
  64. package/src/nlp/JapaneseSegmentation.ts +53 -0
  65. package/src/nlp/Lexicon.ts +119 -0
  66. package/src/nlp/PhoneConversion.ts +508 -0
  67. package/src/nlp/Segmentation.ts +237 -0
  68. package/src/nlp/TextNormalizer.ts +160 -0
  69. package/src/recognition/AmazonTranscribeSTT.ts +112 -0
  70. package/src/recognition/AzureCognitiveServicesSTT.ts +76 -0
  71. package/src/recognition/GoogleCloudSTT.ts +92 -0
  72. package/src/recognition/SileroSTT.ts +173 -0
  73. package/src/recognition/VoskSTT.ts +112 -0
  74. package/src/recognition/WhisperSTT.ts +1518 -0
  75. package/src/server/Client.ts +297 -0
  76. package/src/server/Server.ts +178 -0
  77. package/src/server/ServerStarter.ts +12 -0
  78. package/src/server/Worker.ts +400 -0
  79. package/src/server/WorkerStarter.ts +38 -0
  80. package/src/speech-language-detection/SileroLanguageDetection.ts +105 -0
  81. package/src/subtitles/Subtitles.ts +478 -0
  82. package/src/synthesis/AwsPollyTTS.ts +78 -0
  83. package/src/synthesis/AzureCognitiveServicesTTS.ts +146 -0
  84. package/src/synthesis/CoquiServerTTS.ts +29 -0
  85. package/src/synthesis/ElevenLabsTTS.ts +104 -0
  86. package/src/synthesis/EspeakTTS.ts +552 -0
  87. package/src/synthesis/FliteTTS.ts +387 -0
  88. package/src/synthesis/GoogleCloudTTS.ts +112 -0
  89. package/src/synthesis/GoogleTranslateTTS.ts +210 -0
  90. package/src/synthesis/MicrosoftEdgeTTS.ts +298 -0
  91. package/src/synthesis/SamTTS.ts +30 -0
  92. package/src/synthesis/SapiTTS.ts +222 -0
  93. package/src/synthesis/StreamlabsPollyTTS.ts +114 -0
  94. package/src/synthesis/SvoxPicoTTS.ts +318 -0
  95. package/src/synthesis/VitsTTS.ts +734 -0
  96. package/src/tests/Test.ts +24 -0
  97. package/src/text-language-detection/FastTextLanguageDetection.ts +53 -0
  98. package/src/text-language-detection/TinyLDLanguageDetection.ts +16 -0
  99. package/src/typings/Fillers.d.ts +41 -0
  100. package/src/utilities/BinaryArrayConversion.ts +159 -0
  101. package/src/utilities/Compression.ts +91 -0
  102. package/src/utilities/FileDownloader.ts +201 -0
  103. package/src/utilities/FileSystem.ts +265 -0
  104. package/src/utilities/Hashing.ts +230 -0
  105. package/src/utilities/Locale.ts +119 -0
  106. package/src/utilities/Logger.ts +72 -0
  107. package/src/utilities/NdArrayUtilities.ts +31 -0
  108. package/src/utilities/ObjectUtilities.ts +169 -0
  109. package/src/utilities/OpenPromise.ts +13 -0
  110. package/src/utilities/PackageManager.ts +97 -0
  111. package/src/utilities/Queue.ts +17 -0
  112. package/src/utilities/RandomGenerator.ts +237 -0
  113. package/src/utilities/SignalChannel.ts +22 -0
  114. package/src/utilities/TarballMaker.ts +68 -0
  115. package/src/utilities/Timeline.ts +231 -0
  116. package/src/utilities/Timer.ts +93 -0
  117. package/src/utilities/Utilities.ts +574 -0
  118. package/src/utilities/WasmMemoryManager.ts +516 -0
  119. package/src/utilities/WebReader.ts +55 -0
  120. package/src/utilities/WikipediaReader.ts +41 -0
  121. package/src/voice-activity-detection/SileroVAD.ts +86 -0
  122. package/src/voice-activity-detection/WebRtcVAD.ts +76 -0
@@ -0,0 +1,478 @@
1
+ import { convert as convertHtmlToText } from 'html-to-text'
2
+
3
+ import { formatHMS, formatMS, secondsToHMS, secondsToMS, startsWithAnyOf } from '../utilities/Utilities.js'
4
+ import { isWord, isWordOrSymbolWord } from '../nlp/Segmentation.js'
5
+ import { charactersToWriteAhead } from '../audio/AudioPlayer.js'
6
+ import { Timeline, TimelineEntry } from '../utilities/Timeline.js'
7
+ import { readFile } from '../utilities/FileSystem.js'
8
+ import { deepClone } from '../utilities/ObjectUtilities.js'
9
+
10
+ export async function subtitlesFileToText(filename: string) {
11
+ return subtitlesToText(await readFile(filename, 'utf8'))
12
+ }
13
+
14
+ export function subtitlesToText(subtitles: string) {
15
+ return subtitlesToTimeline(subtitles, true).map(entry => entry.text).join(' ')
16
+ }
17
+
18
+ export function subtitlesToTimeline(subtitles: string, removeMarkup = true) {
19
+ const lines = subtitles.split(/\r?\n/)
20
+
21
+ const timeline: Timeline = []
22
+
23
+ let isWithinCue = false
24
+
25
+ // Parse lines of subtitles text
26
+ for (let line of lines) {
27
+ line = line.trim()
28
+
29
+ if (line.length == 0) {
30
+ isWithinCue = false
31
+
32
+ continue
33
+ }
34
+
35
+ let result = tryParseTimeRangePatternWithHours(line)
36
+
37
+ if (!result.succeeded) {
38
+ result = tryParseTimeRangePatternWithoutHours(line)
39
+ }
40
+
41
+ if (result.succeeded) {
42
+ timeline.push({
43
+ type: 'segment',
44
+ startTime: result.startTime,
45
+ endTime: result.endTime,
46
+ text: ''
47
+ })
48
+
49
+ isWithinCue = true
50
+ } else if (isWithinCue && timeline.length > 0) {
51
+ const lastEntry = timeline[timeline.length - 1]
52
+
53
+ if (lastEntry.text == '') {
54
+ lastEntry.text = line
55
+ } else {
56
+ lastEntry.text += ' ' + line
57
+ }
58
+ }
59
+ }
60
+
61
+ if (!removeMarkup) {
62
+ return timeline
63
+ }
64
+
65
+ // Remove markup in each entry text
66
+ const timelineWithoutMarkup = timeline.map((entry) => {
67
+ let plainText: string = entry.text
68
+
69
+ plainText = plainText.replaceAll(/<[^>]*>/g, '')
70
+
71
+ plainText = convertHtmlToText(plainText, { wordwrap: false })
72
+
73
+ plainText = plainText.replaceAll(/\s+/g, ' ').trim()
74
+
75
+ return { ...entry, text: plainText }
76
+ })
77
+
78
+ return timelineWithoutMarkup
79
+ }
80
+
81
+ export function timelineToSubtitles(timeline: Timeline, subtitlesConfig?: SubtitlesConfig) {
82
+ // Prepare subtitle configuration
83
+ timeline = deepClone(timeline)
84
+
85
+ let config = subtitlesConfig || {}
86
+
87
+ if (config.format && config.format == 'webvtt') {
88
+ config = { ...defaultSubtitlesBaseConfig, ...webVttConfigExtension, ...config }
89
+ } else {
90
+ config = { ...defaultSubtitlesBaseConfig, ...srtConfigExtension, ...config }
91
+ }
92
+
93
+ // Initialize subtitle file content
94
+ const lineBreakString = config.lineBreakString
95
+
96
+ let outText = ''
97
+
98
+ if (config.format == 'webvtt') {
99
+ outText += `WEBVTT${lineBreakString}Kind: captions${lineBreakString}`
100
+
101
+ if (config.language) {
102
+ outText += `Language: ${config.language}${lineBreakString}`
103
+ }
104
+
105
+ outText += lineBreakString
106
+ }
107
+
108
+ // Generate the cues from the given timeline
109
+ let cues: Cue[]
110
+
111
+ if (config.kind == 'segment' || config.kind == 'sentence') {
112
+ cues = getCuesFromTimeline_IsolateSegmentSentence(timeline, config)
113
+
114
+ // Extend cue end times to maximum duration set, if possible
115
+ if (config.maxAddedDuration! > 0) {
116
+ for (let i = 1; i < cues.length; i++) {
117
+ const currentCue = cues[i]
118
+ const previousCue = cues[i - 1]
119
+
120
+ previousCue.endTime = Math.min(previousCue.endTime + config.maxAddedDuration!, currentCue.startTime)
121
+ }
122
+ }
123
+ } else if (config.kind == 'word' || config.kind == 'phone' || config.kind == 'word-phone') {
124
+ cues = getCuesFromTimeline_IsolateWordPhone(timeline, config)
125
+ } else {
126
+ throw new Error('Invalid subtitles mode.')
127
+ }
128
+
129
+ // Write cues to output text
130
+ for (let cueIndex = 0; cueIndex < cues.length; cueIndex++) {
131
+ outText += cueObjectToText(cues[cueIndex], cueIndex + 1, config)
132
+ }
133
+
134
+ return outText
135
+ }
136
+
137
+ // Generates subtitle cues from timeline. Ensures each segment or sentence starts in a new cue.
138
+ function getCuesFromTimeline_IsolateSegmentSentence(timeline: Timeline, config: SubtitlesConfig) {
139
+ if (timeline.length == 0) {
140
+ return []
141
+ }
142
+
143
+ // If the given timeline is a word timeline, wrap it with a segment and call again
144
+ if (timeline[0].type == 'word') {
145
+ const wordTimeline = timeline.filter(entry => isWordOrSymbolWord(entry.text))
146
+
147
+ const text = wordTimeline.map(entry => entry.text).join(' ')
148
+
149
+ const segmentEntry: TimelineEntry = {
150
+ type: 'segment',
151
+ text: text,
152
+ startTime: wordTimeline[0].startTime,
153
+ endTime: wordTimeline[wordTimeline.length - 1].endTime,
154
+ timeline: wordTimeline
155
+ }
156
+
157
+ return getCuesFromTimeline_IsolateSegmentSentence([segmentEntry], config)
158
+ }
159
+
160
+ const cues: Cue[] = []
161
+
162
+ // Generate one or more cues from each segment or sentence in the timeline.
163
+ for (let entry of timeline) {
164
+ if (entry.type == 'segment' && entry.timeline?.[0].type == 'sentence') {
165
+ if (config.kind == 'segment') {
166
+ // If the mode is 'segment', flatten all sentences to a single word timeline
167
+ entry.timeline = entry.timeline!.flatMap(t => t.timeline!)
168
+ } else {
169
+ cues.push(...getCuesFromTimeline_IsolateSegmentSentence(entry.timeline!, config))
170
+
171
+ continue
172
+ }
173
+ }
174
+
175
+ const entryText = entry.text
176
+ const maxLineWidth = config.maxLineWidth!
177
+
178
+ if (entryText.length <= maxLineWidth) {
179
+ cues.push({
180
+ lines: [entryText],
181
+ startTime: entry.startTime,
182
+ endTime: entry.endTime
183
+ })
184
+
185
+ continue
186
+ }
187
+
188
+ if (!entry.timeline || entry.timeline?.[0]?.type != 'word') {
189
+ continue
190
+ }
191
+
192
+ const wordTimeline = entry.timeline!.filter(entry => isWord(entry.text))
193
+
194
+ // First, add word start and end offsets for all word entries
195
+ let lastWordEndOffset = 0
196
+ for (const wordEntry of wordTimeline) {
197
+ const wordStartOffset = entryText.indexOf(wordEntry.text, lastWordEndOffset)
198
+
199
+ if (wordStartOffset == -1) {
200
+ throw new Error(`Couldn't find word '${wordEntry.text}' in its parent entry text`)
201
+ }
202
+
203
+ let wordEndOffset = wordStartOffset + wordEntry.text.length
204
+ lastWordEndOffset = wordEndOffset
205
+
206
+ wordEntry.startOffsetUtf16 = wordStartOffset
207
+ wordEntry.endOffsetUtf16 = wordEndOffset
208
+ }
209
+
210
+ // Add cues
211
+ let currentCue: Cue = {
212
+ lines: [],
213
+ startTime: -1,
214
+ endTime: -1
215
+ }
216
+
217
+ let lineStartWordOffset = 0
218
+ let lineStartOffset = 0
219
+
220
+ for (let wordIndex = 0; wordIndex < wordTimeline.length; wordIndex++) {
221
+ const isLastWord = wordIndex == wordTimeline.length - 1
222
+
223
+ const wordEntry = wordTimeline[wordIndex]
224
+ const wordEndOffset = wordEntry.endOffsetUtf16!
225
+
226
+ function getExtendedEndOffset(offset: number | undefined) {
227
+ if (offset == undefined) {
228
+ return entryText.length
229
+ }
230
+
231
+ while (charactersToWriteAhead.includes(entryText[offset])) {
232
+ offset += 1
233
+ }
234
+
235
+ return offset
236
+ }
237
+
238
+ const wordExtendedEndOffset = getExtendedEndOffset(wordEndOffset)
239
+
240
+ const nextWordEntry = wordTimeline[wordIndex + 1]
241
+ const nextWordExtendedEndOffset = getExtendedEndOffset(nextWordEntry?.endOffsetUtf16)
242
+
243
+ // Decide if to add to a new line
244
+ const lineLength = wordExtendedEndOffset - lineStartOffset
245
+ const lineLengthWithNextWord = nextWordExtendedEndOffset - lineStartOffset
246
+ const wordsRemaining = wordTimeline.length - wordIndex - 1
247
+
248
+ const phraseSeparators = [',', ',', '、', ';', ':', '),', '",', '”,', '.', '".', '”.', '."', '.”', '。']
249
+
250
+ const lineLengthWithNextWordExceedsMaxLineWidth = lineLengthWithNextWord >= maxLineWidth
251
+ const lineLengthExceedsHalfMaxLineWidth = lineLength >= maxLineWidth / 2
252
+
253
+ const wordsRemainingAreEqualOrLessToMinimumWordsInLine = wordsRemaining <= config.minWordsInLine!
254
+ const remainingTextExceedsMaxLineWidth = entryText.length - lineStartOffset > maxLineWidth
255
+ const followingSubstringIsPhraseSeparator = startsWithAnyOf(entryText.substring(wordEndOffset), phraseSeparators)
256
+
257
+ const shouldAddNewLine =
258
+ isLastWord ||
259
+ lineLengthWithNextWordExceedsMaxLineWidth ||
260
+ (remainingTextExceedsMaxLineWidth &&
261
+ lineLengthExceedsHalfMaxLineWidth &&
262
+ (wordsRemainingAreEqualOrLessToMinimumWordsInLine || (config.separatePhrases && followingSubstringIsPhraseSeparator)))
263
+
264
+ // If it was decided to add a new line
265
+ if (shouldAddNewLine) {
266
+ // Extend line end offset to end of sentence entry if last word encountered
267
+ let lineEndOffset: number
268
+
269
+ if (isLastWord) {
270
+ lineEndOffset = entryText.length
271
+ } else {
272
+ lineEndOffset = wordExtendedEndOffset
273
+ }
274
+
275
+ // Get line text
276
+ const lineText = entryText.substring(lineStartOffset, lineEndOffset)
277
+
278
+ // Find start and end times of line
279
+ const nextWordStartTime = isLastWord ? entry.endTime : wordTimeline[wordIndex + 1].startTime
280
+
281
+ const lineStartTime = wordTimeline[lineStartWordOffset].startTime
282
+ const lineEndTime = nextWordStartTime
283
+
284
+ // Add new line to cue
285
+ currentCue.lines.push(lineText)
286
+
287
+ // Update cue start and end times
288
+ if (currentCue.startTime == -1) {
289
+ currentCue.startTime = lineStartTime
290
+ }
291
+
292
+ currentCue.endTime = lineEndTime
293
+
294
+ // Finalize cue if needed
295
+ if (isLastWord || currentCue.lines.length == config.maxLineCount) {
296
+ cues.push(currentCue)
297
+
298
+ currentCue = {
299
+ lines: [],
300
+ startTime: -1,
301
+ endTime: -1
302
+ }
303
+ }
304
+
305
+ // Update offsets
306
+ lineStartOffset = lineEndOffset
307
+ lineStartWordOffset = wordIndex + 1
308
+ }
309
+ }
310
+ }
311
+
312
+ return cues
313
+ }
314
+
315
+ // Generates cues from timeline. Isolate words or phones in individual cues.
316
+ function getCuesFromTimeline_IsolateWordPhone(timeline: Timeline, config: SubtitlesConfig) {
317
+ if (timeline.length == 0) {
318
+ return []
319
+ }
320
+
321
+ const kind = config.kind!
322
+
323
+ const cues: Cue[] = []
324
+
325
+ for (const entry of timeline) {
326
+ const entryIsWord = entry.type == 'word'
327
+ const entryIsPhone = entry.type == 'phone'
328
+
329
+ const shouldIncludeEntry =
330
+ (entryIsWord && (kind == 'word' || kind == 'word-phone')) ||
331
+ (entryIsPhone && (kind == 'phone' || kind == 'word-phone'))
332
+
333
+ if (shouldIncludeEntry) {
334
+ cues.push({
335
+ lines: [entry.text],
336
+ startTime: entry.startTime,
337
+ endTime: entry.endTime,
338
+ })
339
+ }
340
+
341
+ if (entry.timeline) {
342
+ cues.push(...getCuesFromTimeline_IsolateWordPhone(entry.timeline, config))
343
+ }
344
+ }
345
+
346
+ return cues
347
+ }
348
+
349
+ function tryParseTimeRangePatternWithHours(line: string) {
350
+ const timeRangePatternWithHours = /^(\d+)\:(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)\:(\d+)[\.,](\d+)/
351
+ const match = timeRangePatternWithHours.exec(line)
352
+
353
+ if (!match) {
354
+ return { startTime: -1, endTime: -1, succeeded: false }
355
+ }
356
+
357
+ const startHours = parseInt(match[1])
358
+ const startMinutes = parseInt(match[2])
359
+ const startSeconds = parseInt(match[3])
360
+ const startMilliseconds = parseInt(match[4])
361
+
362
+ const endHours = parseInt(match[5])
363
+ const endMinutes = parseInt(match[6])
364
+ const endSeconds = parseInt(match[7])
365
+ const endMilliseconds = parseInt(match[8])
366
+
367
+ const startTime = (startMilliseconds / 1000) + (startSeconds) + (startMinutes * 60) + (startHours * 60 * 60)
368
+ const endTime = (endMilliseconds / 1000) + (endSeconds) + (endMinutes * 60) + (endHours * 60 * 60)
369
+
370
+ return { startTime, endTime, succeeded: true }
371
+ }
372
+
373
+ function tryParseTimeRangePatternWithoutHours(line: string) {
374
+ const timeRangePatternWithHours = /^(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)[\.,](\d+)/
375
+ const match = timeRangePatternWithHours.exec(line)
376
+
377
+ if (!match) {
378
+ return { startTime: -1, endTime: -1, succeeded: false }
379
+ }
380
+
381
+ const startMinutes = parseInt(match[1])
382
+ const startSeconds = parseInt(match[2])
383
+ const startMilliseconds = parseInt(match[3])
384
+
385
+ const endMinutes = parseInt(match[4])
386
+ const endSeconds = parseInt(match[5])
387
+ const endMilliseconds = parseInt(match[6])
388
+
389
+ const startTime = (startMilliseconds / 1000) + (startSeconds) + (startMinutes * 60)
390
+ const endTime = (endMilliseconds / 1000) + (endSeconds) + (endMinutes * 60)
391
+
392
+ return { startTime, endTime, succeeded: true }
393
+ }
394
+
395
+ function cueObjectToText(cue: Cue, cueIndex: number, config: SubtitlesConfig) {
396
+ if (!cue || !cue.lines || cue.lines.length == 0) {
397
+ throw new Error(`Cue is empty`)
398
+ }
399
+
400
+ const lineBreakString = config.lineBreakString
401
+
402
+ let outText = ''
403
+
404
+ if (config.includeCueIndexes) {
405
+ outText += `${cueIndex}${lineBreakString}`
406
+ }
407
+
408
+ let formattedStartTime: string
409
+ let formattedEndTime: string
410
+
411
+ if (config.includeHours == true) {
412
+ formattedStartTime = formatHMS(secondsToHMS(cue.startTime), config.decimalSeparator)
413
+ formattedEndTime = formatHMS(secondsToHMS(cue.endTime), config.decimalSeparator)
414
+ } else {
415
+ formattedStartTime = formatMS(secondsToMS(cue.startTime), config.decimalSeparator)
416
+ formattedEndTime = formatMS(secondsToMS(cue.endTime), config.decimalSeparator)
417
+ }
418
+
419
+ outText += `${formattedStartTime} --> ${formattedEndTime}`
420
+ outText += `${lineBreakString}`
421
+
422
+ outText += cue.lines.map(line => line.trim()).join(lineBreakString)
423
+
424
+ outText += `${lineBreakString}`
425
+ outText += `${lineBreakString}`
426
+
427
+ return outText
428
+ }
429
+
430
+ export type Cue = {
431
+ lines: string[]
432
+ startTime: number
433
+ endTime: number
434
+ }
435
+
436
+ export type SubtitlesKind = 'segment' | 'sentence' | 'word' | 'phone' | 'word-phone'
437
+
438
+ export interface SubtitlesConfig {
439
+ format?: 'srt' | 'webvtt'
440
+ language?: string
441
+ kind?: SubtitlesKind
442
+
443
+ maxLineCount?: number
444
+ maxLineWidth?: number
445
+ minWordsInLine?: number
446
+ separatePhrases?: boolean
447
+ maxAddedDuration?: number
448
+
449
+ decimalSeparator?: ',' | '.'
450
+ includeCueIndexes?: boolean
451
+ includeHours?: boolean
452
+ lineBreakString?: '\n' | '\r\n'
453
+ }
454
+
455
+ export const defaultSubtitlesBaseConfig: SubtitlesConfig = {
456
+ format: 'srt',
457
+ kind: 'sentence',
458
+
459
+ maxLineCount: 2,
460
+ maxLineWidth: 42,
461
+ minWordsInLine: 4,
462
+ separatePhrases: true,
463
+ maxAddedDuration: 3.0,
464
+ }
465
+
466
+ export const srtConfigExtension: SubtitlesConfig = {
467
+ decimalSeparator: ',',
468
+ includeCueIndexes: true,
469
+ includeHours: true,
470
+ lineBreakString: '\n',
471
+ }
472
+
473
+ export const webVttConfigExtension: SubtitlesConfig = {
474
+ decimalSeparator: '.',
475
+ includeCueIndexes: false,
476
+ includeHours: true,
477
+ lineBreakString: '\n',
478
+ }
@@ -0,0 +1,78 @@
1
+ import type { SynthesizeSpeechCommandInput } from "@aws-sdk/client-polly"
2
+ import { IncomingMessage } from "http"
3
+ import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
4
+ import { Logger } from "../utilities/Logger.js"
5
+
6
+ import { readBinaryIncomingMessage } from "../utilities/Utilities.js"
7
+
8
+ export async function synthesize(text: string, language: string | undefined, voice: string, region: string, accessKeyId: string, secretAccessKey: string, engine: "standard" | "neural" = "standard", ssmlEnabled = false, lexiconNames?: string[]) {
9
+ const logger = new Logger()
10
+ logger.start("Load AWS SDK client module")
11
+
12
+ const polly = await import("@aws-sdk/client-polly")
13
+
14
+ const pollyClient = new polly.PollyClient({
15
+ region,
16
+ credentials: {
17
+ accessKeyId,
18
+ secretAccessKey
19
+ }
20
+ })
21
+
22
+ const params: SynthesizeSpeechCommandInput = {
23
+ VoiceId: voice,
24
+
25
+ LanguageCode: language,
26
+
27
+ Engine: engine,
28
+ Text: text,
29
+ LexiconNames: lexiconNames,
30
+
31
+ TextType: ssmlEnabled ? "ssml" : "text",
32
+
33
+ OutputFormat: "mp3",
34
+ }
35
+
36
+ logger.start("Request synthesis from AWS Polly")
37
+
38
+ const command = new polly.SynthesizeSpeechCommand(params)
39
+
40
+ const result = await pollyClient.send(command)
41
+
42
+ const audioStream: IncomingMessage = result.AudioStream as any
43
+
44
+ const audioData = await readBinaryIncomingMessage(audioStream)
45
+
46
+ logger.end()
47
+
48
+ const rawAudio = await FFMpegTranscoder.decodeToChannels(audioData as any)
49
+
50
+ return { rawAudio }
51
+ }
52
+
53
+ export async function getVoiceList(region: string, accessKeyId: string, secretAccessKey: string) {
54
+ const logger = new Logger()
55
+ logger.start("Load AWS SDK client module")
56
+
57
+ const polly = await import("@aws-sdk/client-polly")
58
+
59
+ logger.start("Request voice list from AWS Polly")
60
+
61
+ const pollyClient = new polly.PollyClient({
62
+ region,
63
+ credentials: {
64
+ accessKeyId,
65
+ secretAccessKey
66
+ }
67
+ })
68
+
69
+ const command = new polly.DescribeVoicesCommand({})
70
+
71
+ const result = await pollyClient.send(command)
72
+
73
+ const voices = result.Voices!
74
+
75
+ logger.end()
76
+
77
+ return voices
78
+ }
@@ -0,0 +1,146 @@
1
+ import SpeechSDK from 'microsoft-cognitiveservices-speech-sdk'
2
+
3
+ import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
4
+
5
+ import { escape } from 'html-escaper'
6
+
7
+ import { Logger } from '../utilities/Logger.js'
8
+ import { Timeline } from '../utilities/Timeline.js'
9
+ import { RawAudio, getRawAudioDuration } from '../audio/AudioUtilities.js'
10
+
11
+ export async function synthesize(
12
+ text: string,
13
+ subscriptionKey: string,
14
+ serviceRegion: string,
15
+ languageCode = "en-US",
16
+ voice = "Microsoft Server Speech Text to Speech Voice (en-US, AriaNeural)",
17
+ ssmlEnabled = false,
18
+ ssmlPitchString = "+0Hz",
19
+ ssmlRateString = "+0%") {
20
+
21
+ return new Promise<{ rawAudio: RawAudio, timeline: Timeline }>((resolve, reject) => {
22
+ const logger = new Logger()
23
+ logger.start("Request synthesis from Azure Cognitive Services")
24
+
25
+ const speechConfig = SpeechSDK.SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
26
+
27
+ speechConfig.speechSynthesisLanguage = languageCode
28
+ speechConfig.speechSynthesisVoiceName = voice
29
+ speechConfig.speechSynthesisOutputFormat = SpeechSDK.SpeechSynthesisOutputFormat.Ogg24Khz16BitMonoOpus
30
+
31
+ const audioOutputStream = SpeechSDK.AudioOutputStream.createPullStream()
32
+
33
+ const audioConfig = SpeechSDK.AudioConfig.fromStreamOutput(audioOutputStream)
34
+
35
+ const synthesis = new SpeechSDK.SpeechSynthesizer(speechConfig, audioConfig)
36
+
37
+ const events: SpeechSDK.SpeechSynthesisWordBoundaryEventArgs[] = []
38
+
39
+ synthesis.wordBoundary = (sender, event) => {
40
+ events.push(event)
41
+ }
42
+
43
+ const onResult = async (result: SpeechSDK.SpeechSynthesisResult) => {
44
+ if (result.errorDetails != null) {
45
+ reject(result.errorDetails)
46
+ return
47
+ }
48
+
49
+ /*
50
+ const bufferSize = 2 ** 16
51
+ const buffers: Buffer[] = []
52
+
53
+ while (true) {
54
+
55
+ const buffer = Buffer.alloc(bufferSize)
56
+ const amountRead = await audioOutputStream.read(buffer)
57
+
58
+ if (amountRead == 0) {
59
+ audioOutputStream.close()
60
+ break
61
+ }
62
+
63
+ buffers.push(buffer.subarray(0, amountRead))
64
+ }
65
+
66
+ const encodedAudio = Buffer.concat(buffers)
67
+ */
68
+
69
+ const encodedAudio = Buffer.from(result.audioData)
70
+
71
+ logger.end()
72
+
73
+ const rawAudio = await FFMpegTranscoder.decodeToChannels(encodedAudio, 24000, 1)
74
+
75
+ logger.start("Convert boundary events to a timeline")
76
+
77
+ const timeline = boundaryEventsToTimeline(events, getRawAudioDuration(rawAudio))
78
+
79
+ logger.end()
80
+
81
+ resolve({ rawAudio, timeline: timeline })
82
+ }
83
+
84
+ const onError = (error: string) => {
85
+ reject(error)
86
+ }
87
+
88
+ if (!ssmlEnabled && ssmlPitchString != "+0%" || ssmlRateString != "+0Hz") {
89
+ ssmlEnabled = true
90
+ text = escape(text)
91
+ }
92
+
93
+ if (ssmlEnabled) {
94
+ text =
95
+ `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="en-US">` +
96
+ `<voice name="${voice}">` +
97
+ `<prosody pitch="${ssmlPitchString}" rate="${ssmlRateString}">` +
98
+ text +
99
+ `</prosody>` +
100
+ `</voice>` +
101
+ `</speak>`
102
+
103
+ synthesis.speakSsmlAsync(text, onResult, onError)
104
+ } else {
105
+ synthesis.speakTextAsync(text, onResult, onError)
106
+ }
107
+ })
108
+ }
109
+
110
+ export async function getVoiceList(subscriptionKey: string, serviceRegion: string) {
111
+ const speechConfig = SpeechSDK.SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
112
+
113
+ const synthesis = new SpeechSDK.SpeechSynthesizer(speechConfig, undefined)
114
+
115
+ const result = await synthesis.getVoicesAsync()
116
+
117
+ return result.voices
118
+ }
119
+
120
+ export function boundaryEventsToTimeline(events: any[], totalDuration: number) {
121
+ const timeline: Timeline = []
122
+
123
+ for (const event of events) {
124
+ const boundaryType = event.boundaryType != null ? event.boundaryType : event.Type
125
+
126
+ if (boundaryType != "WordBoundary") {
127
+ continue
128
+ }
129
+
130
+ const text: string = event.text != null ? event.text : event.Data.text.Text
131
+ const offset: number = event.audioOffset != null ? event.audioOffset : event.Data.Offset
132
+ const duration: number = event.duration != null ? event.duration : event.Data.Duration
133
+
134
+ const startTime = offset / 10000000
135
+ const endTime = (offset + duration) / 10000000
136
+
137
+ timeline.push({
138
+ type: "word",
139
+ text,
140
+ startTime,
141
+ endTime
142
+ })
143
+ }
144
+
145
+ return timeline
146
+ }