echogarden 0.11.12 → 0.11.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +16 -0
- package/dist/api/Alignment.js +2 -2
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/Recognition.js +2 -2
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/Synthesis.js +5 -4
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.js +2 -2
- package/dist/api/Translation.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +1 -0
- package/dist/audio/AudioUtilities.js +25 -7
- package/dist/audio/AudioUtilities.js.map +1 -1
- package/dist/cli/CLI.js +2 -2
- package/dist/cli/CLI.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +2 -2
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/subtitles/Subtitles.d.ts +10 -7
- package/dist/subtitles/Subtitles.js +268 -207
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/docs/Options.md +4 -2
- package/package.json +7 -6
- package/src/alignment/DTWMfccSequenceAlignment.ts +43 -0
- package/src/alignment/DTWSequenceAlignment.ts +121 -0
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +210 -0
- package/src/alignment/LevenshteinSequenceAlignment.ts +126 -0
- package/src/alignment/SpeechAlignment.ts +488 -0
- package/src/api/API.ts +12 -0
- package/src/api/APIOptions.ts +15 -0
- package/src/api/Alignment.ts +329 -0
- package/src/api/Common.ts +16 -0
- package/src/api/Denoising.ts +120 -0
- package/src/api/LanguageDetection.ts +286 -0
- package/src/api/Recognition.ts +344 -0
- package/src/api/Synthesis.ts +1735 -0
- package/src/api/Translation.ts +143 -0
- package/src/api/Vad.ts +172 -0
- package/src/audio/AudioBufferConversion.ts +248 -0
- package/src/audio/AudioPlayer.ts +358 -0
- package/src/audio/AudioRecorder.ts +91 -0
- package/src/audio/AudioUtilities.ts +392 -0
- package/src/audio/SoxPath.ts +24 -0
- package/src/cli/CLI.ts +1360 -0
- package/src/cli/CLIConfigFile.ts +91 -0
- package/src/cli/CLILauncher.ts +26 -0
- package/src/cli/CLIOptionsSchema.ts +54 -0
- package/src/cli/CLIParser.ts +41 -0
- package/src/cli/CLIStarter.ts +40 -0
- package/src/codecs/FFMpegTranscoder.ts +214 -0
- package/src/codecs/TIMITCodec.ts +17 -0
- package/src/codecs/WaveCodec.ts +260 -0
- package/src/denoising/RNNoise.ts +95 -0
- package/src/dsp/BiquadFilter.ts +488 -0
- package/src/dsp/FFT.ts +187 -0
- package/src/dsp/MFCC.ts +227 -0
- package/src/dsp/MelSpectogram.ts +145 -0
- package/src/dsp/Rubberband.ts +249 -0
- package/src/dsp/Sonic.ts +59 -0
- package/src/dsp/SpeexResampler.ts +79 -0
- package/src/math/VectorMath.ts +812 -0
- package/src/nlp/ChineseSegmentation.ts +68 -0
- package/src/nlp/CompromiseNLP.ts +113 -0
- package/src/nlp/EspeakPhonemizer.ts +168 -0
- package/src/nlp/IPA.ts +139 -0
- package/src/nlp/JapaneseSegmentation.ts +53 -0
- package/src/nlp/Lexicon.ts +119 -0
- package/src/nlp/PhoneConversion.ts +508 -0
- package/src/nlp/Segmentation.ts +237 -0
- package/src/nlp/TextNormalizer.ts +160 -0
- package/src/recognition/AmazonTranscribeSTT.ts +112 -0
- package/src/recognition/AzureCognitiveServicesSTT.ts +76 -0
- package/src/recognition/GoogleCloudSTT.ts +92 -0
- package/src/recognition/SileroSTT.ts +173 -0
- package/src/recognition/VoskSTT.ts +112 -0
- package/src/recognition/WhisperSTT.ts +1518 -0
- package/src/server/Client.ts +297 -0
- package/src/server/Server.ts +178 -0
- package/src/server/ServerStarter.ts +12 -0
- package/src/server/Worker.ts +400 -0
- package/src/server/WorkerStarter.ts +38 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +105 -0
- package/src/subtitles/Subtitles.ts +478 -0
- package/src/synthesis/AwsPollyTTS.ts +78 -0
- package/src/synthesis/AzureCognitiveServicesTTS.ts +146 -0
- package/src/synthesis/CoquiServerTTS.ts +29 -0
- package/src/synthesis/ElevenLabsTTS.ts +104 -0
- package/src/synthesis/EspeakTTS.ts +552 -0
- package/src/synthesis/FliteTTS.ts +387 -0
- package/src/synthesis/GoogleCloudTTS.ts +112 -0
- package/src/synthesis/GoogleTranslateTTS.ts +210 -0
- package/src/synthesis/MicrosoftEdgeTTS.ts +298 -0
- package/src/synthesis/SamTTS.ts +30 -0
- package/src/synthesis/SapiTTS.ts +222 -0
- package/src/synthesis/StreamlabsPollyTTS.ts +114 -0
- package/src/synthesis/SvoxPicoTTS.ts +318 -0
- package/src/synthesis/VitsTTS.ts +734 -0
- package/src/tests/Test.ts +24 -0
- package/src/text-language-detection/FastTextLanguageDetection.ts +53 -0
- package/src/text-language-detection/TinyLDLanguageDetection.ts +16 -0
- package/src/typings/Fillers.d.ts +41 -0
- package/src/utilities/BinaryArrayConversion.ts +159 -0
- package/src/utilities/Compression.ts +91 -0
- package/src/utilities/FileDownloader.ts +201 -0
- package/src/utilities/FileSystem.ts +265 -0
- package/src/utilities/Hashing.ts +230 -0
- package/src/utilities/Locale.ts +119 -0
- package/src/utilities/Logger.ts +72 -0
- package/src/utilities/NdArrayUtilities.ts +31 -0
- package/src/utilities/ObjectUtilities.ts +169 -0
- package/src/utilities/OpenPromise.ts +13 -0
- package/src/utilities/PackageManager.ts +97 -0
- package/src/utilities/Queue.ts +17 -0
- package/src/utilities/RandomGenerator.ts +237 -0
- package/src/utilities/SignalChannel.ts +22 -0
- package/src/utilities/TarballMaker.ts +68 -0
- package/src/utilities/Timeline.ts +231 -0
- package/src/utilities/Timer.ts +93 -0
- package/src/utilities/Utilities.ts +574 -0
- package/src/utilities/WasmMemoryManager.ts +516 -0
- package/src/utilities/WebReader.ts +55 -0
- package/src/utilities/WikipediaReader.ts +41 -0
- package/src/voice-activity-detection/SileroVAD.ts +86 -0
- package/src/voice-activity-detection/WebRtcVAD.ts +76 -0
|
@@ -0,0 +1,478 @@
|
|
|
1
|
+
import { convert as convertHtmlToText } from 'html-to-text'
|
|
2
|
+
|
|
3
|
+
import { formatHMS, formatMS, secondsToHMS, secondsToMS, startsWithAnyOf } from '../utilities/Utilities.js'
|
|
4
|
+
import { isWord, isWordOrSymbolWord } from '../nlp/Segmentation.js'
|
|
5
|
+
import { charactersToWriteAhead } from '../audio/AudioPlayer.js'
|
|
6
|
+
import { Timeline, TimelineEntry } from '../utilities/Timeline.js'
|
|
7
|
+
import { readFile } from '../utilities/FileSystem.js'
|
|
8
|
+
import { deepClone } from '../utilities/ObjectUtilities.js'
|
|
9
|
+
|
|
10
|
+
export async function subtitlesFileToText(filename: string) {
|
|
11
|
+
return subtitlesToText(await readFile(filename, 'utf8'))
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export function subtitlesToText(subtitles: string) {
|
|
15
|
+
return subtitlesToTimeline(subtitles, true).map(entry => entry.text).join(' ')
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export function subtitlesToTimeline(subtitles: string, removeMarkup = true) {
|
|
19
|
+
const lines = subtitles.split(/\r?\n/)
|
|
20
|
+
|
|
21
|
+
const timeline: Timeline = []
|
|
22
|
+
|
|
23
|
+
let isWithinCue = false
|
|
24
|
+
|
|
25
|
+
// Parse lines of subtitles text
|
|
26
|
+
for (let line of lines) {
|
|
27
|
+
line = line.trim()
|
|
28
|
+
|
|
29
|
+
if (line.length == 0) {
|
|
30
|
+
isWithinCue = false
|
|
31
|
+
|
|
32
|
+
continue
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
let result = tryParseTimeRangePatternWithHours(line)
|
|
36
|
+
|
|
37
|
+
if (!result.succeeded) {
|
|
38
|
+
result = tryParseTimeRangePatternWithoutHours(line)
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
if (result.succeeded) {
|
|
42
|
+
timeline.push({
|
|
43
|
+
type: 'segment',
|
|
44
|
+
startTime: result.startTime,
|
|
45
|
+
endTime: result.endTime,
|
|
46
|
+
text: ''
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
isWithinCue = true
|
|
50
|
+
} else if (isWithinCue && timeline.length > 0) {
|
|
51
|
+
const lastEntry = timeline[timeline.length - 1]
|
|
52
|
+
|
|
53
|
+
if (lastEntry.text == '') {
|
|
54
|
+
lastEntry.text = line
|
|
55
|
+
} else {
|
|
56
|
+
lastEntry.text += ' ' + line
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
if (!removeMarkup) {
|
|
62
|
+
return timeline
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// Remove markup in each entry text
|
|
66
|
+
const timelineWithoutMarkup = timeline.map((entry) => {
|
|
67
|
+
let plainText: string = entry.text
|
|
68
|
+
|
|
69
|
+
plainText = plainText.replaceAll(/<[^>]*>/g, '')
|
|
70
|
+
|
|
71
|
+
plainText = convertHtmlToText(plainText, { wordwrap: false })
|
|
72
|
+
|
|
73
|
+
plainText = plainText.replaceAll(/\s+/g, ' ').trim()
|
|
74
|
+
|
|
75
|
+
return { ...entry, text: plainText }
|
|
76
|
+
})
|
|
77
|
+
|
|
78
|
+
return timelineWithoutMarkup
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export function timelineToSubtitles(timeline: Timeline, subtitlesConfig?: SubtitlesConfig) {
|
|
82
|
+
// Prepare subtitle configuration
|
|
83
|
+
timeline = deepClone(timeline)
|
|
84
|
+
|
|
85
|
+
let config = subtitlesConfig || {}
|
|
86
|
+
|
|
87
|
+
if (config.format && config.format == 'webvtt') {
|
|
88
|
+
config = { ...defaultSubtitlesBaseConfig, ...webVttConfigExtension, ...config }
|
|
89
|
+
} else {
|
|
90
|
+
config = { ...defaultSubtitlesBaseConfig, ...srtConfigExtension, ...config }
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Initialize subtitle file content
|
|
94
|
+
const lineBreakString = config.lineBreakString
|
|
95
|
+
|
|
96
|
+
let outText = ''
|
|
97
|
+
|
|
98
|
+
if (config.format == 'webvtt') {
|
|
99
|
+
outText += `WEBVTT${lineBreakString}Kind: captions${lineBreakString}`
|
|
100
|
+
|
|
101
|
+
if (config.language) {
|
|
102
|
+
outText += `Language: ${config.language}${lineBreakString}`
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
outText += lineBreakString
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// Generate the cues from the given timeline
|
|
109
|
+
let cues: Cue[]
|
|
110
|
+
|
|
111
|
+
if (config.kind == 'segment' || config.kind == 'sentence') {
|
|
112
|
+
cues = getCuesFromTimeline_IsolateSegmentSentence(timeline, config)
|
|
113
|
+
|
|
114
|
+
// Extend cue end times to maximum duration set, if possible
|
|
115
|
+
if (config.maxAddedDuration! > 0) {
|
|
116
|
+
for (let i = 1; i < cues.length; i++) {
|
|
117
|
+
const currentCue = cues[i]
|
|
118
|
+
const previousCue = cues[i - 1]
|
|
119
|
+
|
|
120
|
+
previousCue.endTime = Math.min(previousCue.endTime + config.maxAddedDuration!, currentCue.startTime)
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
} else if (config.kind == 'word' || config.kind == 'phone' || config.kind == 'word-phone') {
|
|
124
|
+
cues = getCuesFromTimeline_IsolateWordPhone(timeline, config)
|
|
125
|
+
} else {
|
|
126
|
+
throw new Error('Invalid subtitles mode.')
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// Write cues to output text
|
|
130
|
+
for (let cueIndex = 0; cueIndex < cues.length; cueIndex++) {
|
|
131
|
+
outText += cueObjectToText(cues[cueIndex], cueIndex + 1, config)
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
return outText
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// Generates subtitle cues from timeline. Ensures each segment or sentence starts in a new cue.
|
|
138
|
+
function getCuesFromTimeline_IsolateSegmentSentence(timeline: Timeline, config: SubtitlesConfig) {
|
|
139
|
+
if (timeline.length == 0) {
|
|
140
|
+
return []
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// If the given timeline is a word timeline, wrap it with a segment and call again
|
|
144
|
+
if (timeline[0].type == 'word') {
|
|
145
|
+
const wordTimeline = timeline.filter(entry => isWordOrSymbolWord(entry.text))
|
|
146
|
+
|
|
147
|
+
const text = wordTimeline.map(entry => entry.text).join(' ')
|
|
148
|
+
|
|
149
|
+
const segmentEntry: TimelineEntry = {
|
|
150
|
+
type: 'segment',
|
|
151
|
+
text: text,
|
|
152
|
+
startTime: wordTimeline[0].startTime,
|
|
153
|
+
endTime: wordTimeline[wordTimeline.length - 1].endTime,
|
|
154
|
+
timeline: wordTimeline
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
return getCuesFromTimeline_IsolateSegmentSentence([segmentEntry], config)
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
const cues: Cue[] = []
|
|
161
|
+
|
|
162
|
+
// Generate one or more cues from each segment or sentence in the timeline.
|
|
163
|
+
for (let entry of timeline) {
|
|
164
|
+
if (entry.type == 'segment' && entry.timeline?.[0].type == 'sentence') {
|
|
165
|
+
if (config.kind == 'segment') {
|
|
166
|
+
// If the mode is 'segment', flatten all sentences to a single word timeline
|
|
167
|
+
entry.timeline = entry.timeline!.flatMap(t => t.timeline!)
|
|
168
|
+
} else {
|
|
169
|
+
cues.push(...getCuesFromTimeline_IsolateSegmentSentence(entry.timeline!, config))
|
|
170
|
+
|
|
171
|
+
continue
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
const entryText = entry.text
|
|
176
|
+
const maxLineWidth = config.maxLineWidth!
|
|
177
|
+
|
|
178
|
+
if (entryText.length <= maxLineWidth) {
|
|
179
|
+
cues.push({
|
|
180
|
+
lines: [entryText],
|
|
181
|
+
startTime: entry.startTime,
|
|
182
|
+
endTime: entry.endTime
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
continue
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
if (!entry.timeline || entry.timeline?.[0]?.type != 'word') {
|
|
189
|
+
continue
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
const wordTimeline = entry.timeline!.filter(entry => isWord(entry.text))
|
|
193
|
+
|
|
194
|
+
// First, add word start and end offsets for all word entries
|
|
195
|
+
let lastWordEndOffset = 0
|
|
196
|
+
for (const wordEntry of wordTimeline) {
|
|
197
|
+
const wordStartOffset = entryText.indexOf(wordEntry.text, lastWordEndOffset)
|
|
198
|
+
|
|
199
|
+
if (wordStartOffset == -1) {
|
|
200
|
+
throw new Error(`Couldn't find word '${wordEntry.text}' in its parent entry text`)
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
let wordEndOffset = wordStartOffset + wordEntry.text.length
|
|
204
|
+
lastWordEndOffset = wordEndOffset
|
|
205
|
+
|
|
206
|
+
wordEntry.startOffsetUtf16 = wordStartOffset
|
|
207
|
+
wordEntry.endOffsetUtf16 = wordEndOffset
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
// Add cues
|
|
211
|
+
let currentCue: Cue = {
|
|
212
|
+
lines: [],
|
|
213
|
+
startTime: -1,
|
|
214
|
+
endTime: -1
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
let lineStartWordOffset = 0
|
|
218
|
+
let lineStartOffset = 0
|
|
219
|
+
|
|
220
|
+
for (let wordIndex = 0; wordIndex < wordTimeline.length; wordIndex++) {
|
|
221
|
+
const isLastWord = wordIndex == wordTimeline.length - 1
|
|
222
|
+
|
|
223
|
+
const wordEntry = wordTimeline[wordIndex]
|
|
224
|
+
const wordEndOffset = wordEntry.endOffsetUtf16!
|
|
225
|
+
|
|
226
|
+
function getExtendedEndOffset(offset: number | undefined) {
|
|
227
|
+
if (offset == undefined) {
|
|
228
|
+
return entryText.length
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
while (charactersToWriteAhead.includes(entryText[offset])) {
|
|
232
|
+
offset += 1
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
return offset
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
const wordExtendedEndOffset = getExtendedEndOffset(wordEndOffset)
|
|
239
|
+
|
|
240
|
+
const nextWordEntry = wordTimeline[wordIndex + 1]
|
|
241
|
+
const nextWordExtendedEndOffset = getExtendedEndOffset(nextWordEntry?.endOffsetUtf16)
|
|
242
|
+
|
|
243
|
+
// Decide if to add to a new line
|
|
244
|
+
const lineLength = wordExtendedEndOffset - lineStartOffset
|
|
245
|
+
const lineLengthWithNextWord = nextWordExtendedEndOffset - lineStartOffset
|
|
246
|
+
const wordsRemaining = wordTimeline.length - wordIndex - 1
|
|
247
|
+
|
|
248
|
+
const phraseSeparators = [',', ',', '、', ';', ':', '),', '",', '”,', '.', '".', '”.', '."', '.”', '。']
|
|
249
|
+
|
|
250
|
+
const lineLengthWithNextWordExceedsMaxLineWidth = lineLengthWithNextWord >= maxLineWidth
|
|
251
|
+
const lineLengthExceedsHalfMaxLineWidth = lineLength >= maxLineWidth / 2
|
|
252
|
+
|
|
253
|
+
const wordsRemainingAreEqualOrLessToMinimumWordsInLine = wordsRemaining <= config.minWordsInLine!
|
|
254
|
+
const remainingTextExceedsMaxLineWidth = entryText.length - lineStartOffset > maxLineWidth
|
|
255
|
+
const followingSubstringIsPhraseSeparator = startsWithAnyOf(entryText.substring(wordEndOffset), phraseSeparators)
|
|
256
|
+
|
|
257
|
+
const shouldAddNewLine =
|
|
258
|
+
isLastWord ||
|
|
259
|
+
lineLengthWithNextWordExceedsMaxLineWidth ||
|
|
260
|
+
(remainingTextExceedsMaxLineWidth &&
|
|
261
|
+
lineLengthExceedsHalfMaxLineWidth &&
|
|
262
|
+
(wordsRemainingAreEqualOrLessToMinimumWordsInLine || (config.separatePhrases && followingSubstringIsPhraseSeparator)))
|
|
263
|
+
|
|
264
|
+
// If it was decided to add a new line
|
|
265
|
+
if (shouldAddNewLine) {
|
|
266
|
+
// Extend line end offset to end of sentence entry if last word encountered
|
|
267
|
+
let lineEndOffset: number
|
|
268
|
+
|
|
269
|
+
if (isLastWord) {
|
|
270
|
+
lineEndOffset = entryText.length
|
|
271
|
+
} else {
|
|
272
|
+
lineEndOffset = wordExtendedEndOffset
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
// Get line text
|
|
276
|
+
const lineText = entryText.substring(lineStartOffset, lineEndOffset)
|
|
277
|
+
|
|
278
|
+
// Find start and end times of line
|
|
279
|
+
const nextWordStartTime = isLastWord ? entry.endTime : wordTimeline[wordIndex + 1].startTime
|
|
280
|
+
|
|
281
|
+
const lineStartTime = wordTimeline[lineStartWordOffset].startTime
|
|
282
|
+
const lineEndTime = nextWordStartTime
|
|
283
|
+
|
|
284
|
+
// Add new line to cue
|
|
285
|
+
currentCue.lines.push(lineText)
|
|
286
|
+
|
|
287
|
+
// Update cue start and end times
|
|
288
|
+
if (currentCue.startTime == -1) {
|
|
289
|
+
currentCue.startTime = lineStartTime
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
currentCue.endTime = lineEndTime
|
|
293
|
+
|
|
294
|
+
// Finalize cue if needed
|
|
295
|
+
if (isLastWord || currentCue.lines.length == config.maxLineCount) {
|
|
296
|
+
cues.push(currentCue)
|
|
297
|
+
|
|
298
|
+
currentCue = {
|
|
299
|
+
lines: [],
|
|
300
|
+
startTime: -1,
|
|
301
|
+
endTime: -1
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// Update offsets
|
|
306
|
+
lineStartOffset = lineEndOffset
|
|
307
|
+
lineStartWordOffset = wordIndex + 1
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
return cues
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
// Generates cues from timeline. Isolate words or phones in individual cues.
|
|
316
|
+
function getCuesFromTimeline_IsolateWordPhone(timeline: Timeline, config: SubtitlesConfig) {
|
|
317
|
+
if (timeline.length == 0) {
|
|
318
|
+
return []
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
const kind = config.kind!
|
|
322
|
+
|
|
323
|
+
const cues: Cue[] = []
|
|
324
|
+
|
|
325
|
+
for (const entry of timeline) {
|
|
326
|
+
const entryIsWord = entry.type == 'word'
|
|
327
|
+
const entryIsPhone = entry.type == 'phone'
|
|
328
|
+
|
|
329
|
+
const shouldIncludeEntry =
|
|
330
|
+
(entryIsWord && (kind == 'word' || kind == 'word-phone')) ||
|
|
331
|
+
(entryIsPhone && (kind == 'phone' || kind == 'word-phone'))
|
|
332
|
+
|
|
333
|
+
if (shouldIncludeEntry) {
|
|
334
|
+
cues.push({
|
|
335
|
+
lines: [entry.text],
|
|
336
|
+
startTime: entry.startTime,
|
|
337
|
+
endTime: entry.endTime,
|
|
338
|
+
})
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
if (entry.timeline) {
|
|
342
|
+
cues.push(...getCuesFromTimeline_IsolateWordPhone(entry.timeline, config))
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
return cues
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
function tryParseTimeRangePatternWithHours(line: string) {
|
|
350
|
+
const timeRangePatternWithHours = /^(\d+)\:(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)\:(\d+)[\.,](\d+)/
|
|
351
|
+
const match = timeRangePatternWithHours.exec(line)
|
|
352
|
+
|
|
353
|
+
if (!match) {
|
|
354
|
+
return { startTime: -1, endTime: -1, succeeded: false }
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
const startHours = parseInt(match[1])
|
|
358
|
+
const startMinutes = parseInt(match[2])
|
|
359
|
+
const startSeconds = parseInt(match[3])
|
|
360
|
+
const startMilliseconds = parseInt(match[4])
|
|
361
|
+
|
|
362
|
+
const endHours = parseInt(match[5])
|
|
363
|
+
const endMinutes = parseInt(match[6])
|
|
364
|
+
const endSeconds = parseInt(match[7])
|
|
365
|
+
const endMilliseconds = parseInt(match[8])
|
|
366
|
+
|
|
367
|
+
const startTime = (startMilliseconds / 1000) + (startSeconds) + (startMinutes * 60) + (startHours * 60 * 60)
|
|
368
|
+
const endTime = (endMilliseconds / 1000) + (endSeconds) + (endMinutes * 60) + (endHours * 60 * 60)
|
|
369
|
+
|
|
370
|
+
return { startTime, endTime, succeeded: true }
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
function tryParseTimeRangePatternWithoutHours(line: string) {
|
|
374
|
+
const timeRangePatternWithHours = /^(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)[\.,](\d+)/
|
|
375
|
+
const match = timeRangePatternWithHours.exec(line)
|
|
376
|
+
|
|
377
|
+
if (!match) {
|
|
378
|
+
return { startTime: -1, endTime: -1, succeeded: false }
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
const startMinutes = parseInt(match[1])
|
|
382
|
+
const startSeconds = parseInt(match[2])
|
|
383
|
+
const startMilliseconds = parseInt(match[3])
|
|
384
|
+
|
|
385
|
+
const endMinutes = parseInt(match[4])
|
|
386
|
+
const endSeconds = parseInt(match[5])
|
|
387
|
+
const endMilliseconds = parseInt(match[6])
|
|
388
|
+
|
|
389
|
+
const startTime = (startMilliseconds / 1000) + (startSeconds) + (startMinutes * 60)
|
|
390
|
+
const endTime = (endMilliseconds / 1000) + (endSeconds) + (endMinutes * 60)
|
|
391
|
+
|
|
392
|
+
return { startTime, endTime, succeeded: true }
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
function cueObjectToText(cue: Cue, cueIndex: number, config: SubtitlesConfig) {
|
|
396
|
+
if (!cue || !cue.lines || cue.lines.length == 0) {
|
|
397
|
+
throw new Error(`Cue is empty`)
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
const lineBreakString = config.lineBreakString
|
|
401
|
+
|
|
402
|
+
let outText = ''
|
|
403
|
+
|
|
404
|
+
if (config.includeCueIndexes) {
|
|
405
|
+
outText += `${cueIndex}${lineBreakString}`
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
let formattedStartTime: string
|
|
409
|
+
let formattedEndTime: string
|
|
410
|
+
|
|
411
|
+
if (config.includeHours == true) {
|
|
412
|
+
formattedStartTime = formatHMS(secondsToHMS(cue.startTime), config.decimalSeparator)
|
|
413
|
+
formattedEndTime = formatHMS(secondsToHMS(cue.endTime), config.decimalSeparator)
|
|
414
|
+
} else {
|
|
415
|
+
formattedStartTime = formatMS(secondsToMS(cue.startTime), config.decimalSeparator)
|
|
416
|
+
formattedEndTime = formatMS(secondsToMS(cue.endTime), config.decimalSeparator)
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
outText += `${formattedStartTime} --> ${formattedEndTime}`
|
|
420
|
+
outText += `${lineBreakString}`
|
|
421
|
+
|
|
422
|
+
outText += cue.lines.map(line => line.trim()).join(lineBreakString)
|
|
423
|
+
|
|
424
|
+
outText += `${lineBreakString}`
|
|
425
|
+
outText += `${lineBreakString}`
|
|
426
|
+
|
|
427
|
+
return outText
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
export type Cue = {
|
|
431
|
+
lines: string[]
|
|
432
|
+
startTime: number
|
|
433
|
+
endTime: number
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
export type SubtitlesKind = 'segment' | 'sentence' | 'word' | 'phone' | 'word-phone'
|
|
437
|
+
|
|
438
|
+
export interface SubtitlesConfig {
|
|
439
|
+
format?: 'srt' | 'webvtt'
|
|
440
|
+
language?: string
|
|
441
|
+
kind?: SubtitlesKind
|
|
442
|
+
|
|
443
|
+
maxLineCount?: number
|
|
444
|
+
maxLineWidth?: number
|
|
445
|
+
minWordsInLine?: number
|
|
446
|
+
separatePhrases?: boolean
|
|
447
|
+
maxAddedDuration?: number
|
|
448
|
+
|
|
449
|
+
decimalSeparator?: ',' | '.'
|
|
450
|
+
includeCueIndexes?: boolean
|
|
451
|
+
includeHours?: boolean
|
|
452
|
+
lineBreakString?: '\n' | '\r\n'
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
export const defaultSubtitlesBaseConfig: SubtitlesConfig = {
|
|
456
|
+
format: 'srt',
|
|
457
|
+
kind: 'sentence',
|
|
458
|
+
|
|
459
|
+
maxLineCount: 2,
|
|
460
|
+
maxLineWidth: 42,
|
|
461
|
+
minWordsInLine: 4,
|
|
462
|
+
separatePhrases: true,
|
|
463
|
+
maxAddedDuration: 3.0,
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
export const srtConfigExtension: SubtitlesConfig = {
|
|
467
|
+
decimalSeparator: ',',
|
|
468
|
+
includeCueIndexes: true,
|
|
469
|
+
includeHours: true,
|
|
470
|
+
lineBreakString: '\n',
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
export const webVttConfigExtension: SubtitlesConfig = {
|
|
474
|
+
decimalSeparator: '.',
|
|
475
|
+
includeCueIndexes: false,
|
|
476
|
+
includeHours: true,
|
|
477
|
+
lineBreakString: '\n',
|
|
478
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
import type { SynthesizeSpeechCommandInput } from "@aws-sdk/client-polly"
|
|
2
|
+
import { IncomingMessage } from "http"
|
|
3
|
+
import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
|
|
4
|
+
import { Logger } from "../utilities/Logger.js"
|
|
5
|
+
|
|
6
|
+
import { readBinaryIncomingMessage } from "../utilities/Utilities.js"
|
|
7
|
+
|
|
8
|
+
export async function synthesize(text: string, language: string | undefined, voice: string, region: string, accessKeyId: string, secretAccessKey: string, engine: "standard" | "neural" = "standard", ssmlEnabled = false, lexiconNames?: string[]) {
|
|
9
|
+
const logger = new Logger()
|
|
10
|
+
logger.start("Load AWS SDK client module")
|
|
11
|
+
|
|
12
|
+
const polly = await import("@aws-sdk/client-polly")
|
|
13
|
+
|
|
14
|
+
const pollyClient = new polly.PollyClient({
|
|
15
|
+
region,
|
|
16
|
+
credentials: {
|
|
17
|
+
accessKeyId,
|
|
18
|
+
secretAccessKey
|
|
19
|
+
}
|
|
20
|
+
})
|
|
21
|
+
|
|
22
|
+
const params: SynthesizeSpeechCommandInput = {
|
|
23
|
+
VoiceId: voice,
|
|
24
|
+
|
|
25
|
+
LanguageCode: language,
|
|
26
|
+
|
|
27
|
+
Engine: engine,
|
|
28
|
+
Text: text,
|
|
29
|
+
LexiconNames: lexiconNames,
|
|
30
|
+
|
|
31
|
+
TextType: ssmlEnabled ? "ssml" : "text",
|
|
32
|
+
|
|
33
|
+
OutputFormat: "mp3",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
logger.start("Request synthesis from AWS Polly")
|
|
37
|
+
|
|
38
|
+
const command = new polly.SynthesizeSpeechCommand(params)
|
|
39
|
+
|
|
40
|
+
const result = await pollyClient.send(command)
|
|
41
|
+
|
|
42
|
+
const audioStream: IncomingMessage = result.AudioStream as any
|
|
43
|
+
|
|
44
|
+
const audioData = await readBinaryIncomingMessage(audioStream)
|
|
45
|
+
|
|
46
|
+
logger.end()
|
|
47
|
+
|
|
48
|
+
const rawAudio = await FFMpegTranscoder.decodeToChannels(audioData as any)
|
|
49
|
+
|
|
50
|
+
return { rawAudio }
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export async function getVoiceList(region: string, accessKeyId: string, secretAccessKey: string) {
|
|
54
|
+
const logger = new Logger()
|
|
55
|
+
logger.start("Load AWS SDK client module")
|
|
56
|
+
|
|
57
|
+
const polly = await import("@aws-sdk/client-polly")
|
|
58
|
+
|
|
59
|
+
logger.start("Request voice list from AWS Polly")
|
|
60
|
+
|
|
61
|
+
const pollyClient = new polly.PollyClient({
|
|
62
|
+
region,
|
|
63
|
+
credentials: {
|
|
64
|
+
accessKeyId,
|
|
65
|
+
secretAccessKey
|
|
66
|
+
}
|
|
67
|
+
})
|
|
68
|
+
|
|
69
|
+
const command = new polly.DescribeVoicesCommand({})
|
|
70
|
+
|
|
71
|
+
const result = await pollyClient.send(command)
|
|
72
|
+
|
|
73
|
+
const voices = result.Voices!
|
|
74
|
+
|
|
75
|
+
logger.end()
|
|
76
|
+
|
|
77
|
+
return voices
|
|
78
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
import SpeechSDK from 'microsoft-cognitiveservices-speech-sdk'
|
|
2
|
+
|
|
3
|
+
import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
|
|
4
|
+
|
|
5
|
+
import { escape } from 'html-escaper'
|
|
6
|
+
|
|
7
|
+
import { Logger } from '../utilities/Logger.js'
|
|
8
|
+
import { Timeline } from '../utilities/Timeline.js'
|
|
9
|
+
import { RawAudio, getRawAudioDuration } from '../audio/AudioUtilities.js'
|
|
10
|
+
|
|
11
|
+
export async function synthesize(
|
|
12
|
+
text: string,
|
|
13
|
+
subscriptionKey: string,
|
|
14
|
+
serviceRegion: string,
|
|
15
|
+
languageCode = "en-US",
|
|
16
|
+
voice = "Microsoft Server Speech Text to Speech Voice (en-US, AriaNeural)",
|
|
17
|
+
ssmlEnabled = false,
|
|
18
|
+
ssmlPitchString = "+0Hz",
|
|
19
|
+
ssmlRateString = "+0%") {
|
|
20
|
+
|
|
21
|
+
return new Promise<{ rawAudio: RawAudio, timeline: Timeline }>((resolve, reject) => {
|
|
22
|
+
const logger = new Logger()
|
|
23
|
+
logger.start("Request synthesis from Azure Cognitive Services")
|
|
24
|
+
|
|
25
|
+
const speechConfig = SpeechSDK.SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
|
|
26
|
+
|
|
27
|
+
speechConfig.speechSynthesisLanguage = languageCode
|
|
28
|
+
speechConfig.speechSynthesisVoiceName = voice
|
|
29
|
+
speechConfig.speechSynthesisOutputFormat = SpeechSDK.SpeechSynthesisOutputFormat.Ogg24Khz16BitMonoOpus
|
|
30
|
+
|
|
31
|
+
const audioOutputStream = SpeechSDK.AudioOutputStream.createPullStream()
|
|
32
|
+
|
|
33
|
+
const audioConfig = SpeechSDK.AudioConfig.fromStreamOutput(audioOutputStream)
|
|
34
|
+
|
|
35
|
+
const synthesis = new SpeechSDK.SpeechSynthesizer(speechConfig, audioConfig)
|
|
36
|
+
|
|
37
|
+
const events: SpeechSDK.SpeechSynthesisWordBoundaryEventArgs[] = []
|
|
38
|
+
|
|
39
|
+
synthesis.wordBoundary = (sender, event) => {
|
|
40
|
+
events.push(event)
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const onResult = async (result: SpeechSDK.SpeechSynthesisResult) => {
|
|
44
|
+
if (result.errorDetails != null) {
|
|
45
|
+
reject(result.errorDetails)
|
|
46
|
+
return
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/*
|
|
50
|
+
const bufferSize = 2 ** 16
|
|
51
|
+
const buffers: Buffer[] = []
|
|
52
|
+
|
|
53
|
+
while (true) {
|
|
54
|
+
|
|
55
|
+
const buffer = Buffer.alloc(bufferSize)
|
|
56
|
+
const amountRead = await audioOutputStream.read(buffer)
|
|
57
|
+
|
|
58
|
+
if (amountRead == 0) {
|
|
59
|
+
audioOutputStream.close()
|
|
60
|
+
break
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
buffers.push(buffer.subarray(0, amountRead))
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const encodedAudio = Buffer.concat(buffers)
|
|
67
|
+
*/
|
|
68
|
+
|
|
69
|
+
const encodedAudio = Buffer.from(result.audioData)
|
|
70
|
+
|
|
71
|
+
logger.end()
|
|
72
|
+
|
|
73
|
+
const rawAudio = await FFMpegTranscoder.decodeToChannels(encodedAudio, 24000, 1)
|
|
74
|
+
|
|
75
|
+
logger.start("Convert boundary events to a timeline")
|
|
76
|
+
|
|
77
|
+
const timeline = boundaryEventsToTimeline(events, getRawAudioDuration(rawAudio))
|
|
78
|
+
|
|
79
|
+
logger.end()
|
|
80
|
+
|
|
81
|
+
resolve({ rawAudio, timeline: timeline })
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
const onError = (error: string) => {
|
|
85
|
+
reject(error)
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
if (!ssmlEnabled && ssmlPitchString != "+0%" || ssmlRateString != "+0Hz") {
|
|
89
|
+
ssmlEnabled = true
|
|
90
|
+
text = escape(text)
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
if (ssmlEnabled) {
|
|
94
|
+
text =
|
|
95
|
+
`<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="en-US">` +
|
|
96
|
+
`<voice name="${voice}">` +
|
|
97
|
+
`<prosody pitch="${ssmlPitchString}" rate="${ssmlRateString}">` +
|
|
98
|
+
text +
|
|
99
|
+
`</prosody>` +
|
|
100
|
+
`</voice>` +
|
|
101
|
+
`</speak>`
|
|
102
|
+
|
|
103
|
+
synthesis.speakSsmlAsync(text, onResult, onError)
|
|
104
|
+
} else {
|
|
105
|
+
synthesis.speakTextAsync(text, onResult, onError)
|
|
106
|
+
}
|
|
107
|
+
})
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export async function getVoiceList(subscriptionKey: string, serviceRegion: string) {
|
|
111
|
+
const speechConfig = SpeechSDK.SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
|
|
112
|
+
|
|
113
|
+
const synthesis = new SpeechSDK.SpeechSynthesizer(speechConfig, undefined)
|
|
114
|
+
|
|
115
|
+
const result = await synthesis.getVoicesAsync()
|
|
116
|
+
|
|
117
|
+
return result.voices
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
export function boundaryEventsToTimeline(events: any[], totalDuration: number) {
|
|
121
|
+
const timeline: Timeline = []
|
|
122
|
+
|
|
123
|
+
for (const event of events) {
|
|
124
|
+
const boundaryType = event.boundaryType != null ? event.boundaryType : event.Type
|
|
125
|
+
|
|
126
|
+
if (boundaryType != "WordBoundary") {
|
|
127
|
+
continue
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const text: string = event.text != null ? event.text : event.Data.text.Text
|
|
131
|
+
const offset: number = event.audioOffset != null ? event.audioOffset : event.Data.Offset
|
|
132
|
+
const duration: number = event.duration != null ? event.duration : event.Data.Duration
|
|
133
|
+
|
|
134
|
+
const startTime = offset / 10000000
|
|
135
|
+
const endTime = (offset + duration) / 10000000
|
|
136
|
+
|
|
137
|
+
timeline.push({
|
|
138
|
+
type: "word",
|
|
139
|
+
text,
|
|
140
|
+
startTime,
|
|
141
|
+
endTime
|
|
142
|
+
})
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
return timeline
|
|
146
|
+
}
|