echogarden 0.12.2 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (334) hide show
  1. package/README.md +15 -14
  2. package/data/schemas/options.json +398 -111
  3. package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
  4. package/dist/alignment/DTWMfccSequenceAlignment.js +8 -8
  5. package/dist/alignment/DTWSequenceAlignment.d.ts +1 -1
  6. package/dist/alignment/DTWSequenceAlignment.js +1 -1
  7. package/dist/alignment/DTWSequenceAlignmentWindowed.d.ts +1 -1
  8. package/dist/alignment/DTWSequenceAlignmentWindowed.js +2 -2
  9. package/dist/alignment/LevenshteinSequenceAlignment.d.ts +1 -1
  10. package/dist/alignment/LevenshteinSequenceAlignment.js +1 -1
  11. package/dist/alignment/SpeechAlignment.d.ts +9 -10
  12. package/dist/alignment/SpeechAlignment.js +136 -105
  13. package/dist/alignment/SpeechAlignment.js.map +1 -1
  14. package/dist/api/API.d.ts +13 -12
  15. package/dist/api/API.js +14 -13
  16. package/dist/api/API.js.map +1 -1
  17. package/dist/api/APIOptions.d.ts +5 -4
  18. package/dist/api/Alignment.d.ts +15 -9
  19. package/dist/api/Alignment.js +88 -74
  20. package/dist/api/Alignment.js.map +1 -1
  21. package/dist/api/Common.js +1 -1
  22. package/dist/api/Denoising.d.ts +6 -6
  23. package/dist/api/Denoising.js +23 -23
  24. package/dist/api/Denoising.js.map +1 -1
  25. package/dist/api/LanguageDetection.d.ts +19 -12
  26. package/dist/api/LanguageDetection.js +88 -38
  27. package/dist/api/LanguageDetection.js.map +1 -1
  28. package/dist/api/Recognition.d.ts +16 -6
  29. package/dist/api/Recognition.js +129 -55
  30. package/dist/api/Recognition.js.map +1 -1
  31. package/dist/api/SourceSeparation.d.ts +17 -0
  32. package/dist/api/SourceSeparation.js +61 -0
  33. package/dist/api/SourceSeparation.js.map +1 -0
  34. package/dist/api/Synthesis.d.ts +18 -18
  35. package/dist/api/Synthesis.js +191 -164
  36. package/dist/api/Synthesis.js.map +1 -1
  37. package/dist/api/Translation.d.ts +19 -8
  38. package/dist/api/Translation.js +132 -35
  39. package/dist/api/Translation.js.map +1 -1
  40. package/dist/api/Vad.d.ts +10 -5
  41. package/dist/api/Vad.js +76 -38
  42. package/dist/api/Vad.js.map +1 -1
  43. package/dist/audio/AudioBufferConversion.d.ts +1 -1
  44. package/dist/audio/AudioBufferConversion.js +4 -4
  45. package/dist/audio/AudioPlayer.d.ts +1 -1
  46. package/dist/audio/AudioPlayer.js +26 -26
  47. package/dist/audio/AudioPlayer.js.map +1 -1
  48. package/dist/audio/AudioRecorder.d.ts +1 -1
  49. package/dist/audio/AudioRecorder.js +5 -5
  50. package/dist/audio/AudioUtilities.d.ts +13 -9
  51. package/dist/audio/AudioUtilities.js +86 -24
  52. package/dist/audio/AudioUtilities.js.map +1 -1
  53. package/dist/cli/CLI.d.ts +3 -3
  54. package/dist/cli/CLI.js +271 -162
  55. package/dist/cli/CLI.js.map +1 -1
  56. package/dist/cli/CLIConfigFile.js +8 -8
  57. package/dist/cli/CLILauncher.js +6 -6
  58. package/dist/cli/CLIOptionsSchema.js +2 -2
  59. package/dist/cli/CLIParser.js +5 -5
  60. package/dist/cli/CLIStarter.js +4 -4
  61. package/dist/codecs/FFMpegTranscoder.d.ts +2 -2
  62. package/dist/codecs/FFMpegTranscoder.js +37 -37
  63. package/dist/codecs/FFMpegTranscoder.js.map +1 -1
  64. package/dist/codecs/TIMITCodec.js +5 -5
  65. package/dist/codecs/WaveCodec.d.ts +1 -1
  66. package/dist/codecs/WaveCodec.js +22 -22
  67. package/dist/denoising/RNNoise.d.ts +1 -1
  68. package/dist/denoising/RNNoise.js +9 -9
  69. package/dist/dsp/BiquadFilter.d.ts +3 -2
  70. package/dist/dsp/BiquadFilter.js +18 -11
  71. package/dist/dsp/BiquadFilter.js.map +1 -1
  72. package/dist/dsp/DecayingPeakEstimator.d.ts +16 -0
  73. package/dist/dsp/DecayingPeakEstimator.js +23 -0
  74. package/dist/dsp/DecayingPeakEstimator.js.map +1 -0
  75. package/dist/dsp/FFT.d.ts +8 -4
  76. package/dist/dsp/FFT.js +76 -30
  77. package/dist/dsp/FFT.js.map +1 -1
  78. package/dist/dsp/KWeightingFilter.d.ts +9 -0
  79. package/dist/dsp/KWeightingFilter.js +40 -0
  80. package/dist/dsp/KWeightingFilter.js.map +1 -0
  81. package/dist/dsp/LoudnessEstimator.d.ts +21 -0
  82. package/dist/dsp/LoudnessEstimator.js +47 -0
  83. package/dist/dsp/LoudnessEstimator.js.map +1 -0
  84. package/dist/dsp/MFCC.d.ts +2 -2
  85. package/dist/dsp/MFCC.js +15 -15
  86. package/dist/dsp/MelSpectogram.d.ts +1 -1
  87. package/dist/dsp/MelSpectogram.js +6 -6
  88. package/dist/dsp/Rubberband.d.ts +11 -11
  89. package/dist/dsp/Rubberband.js +27 -27
  90. package/dist/dsp/Sonic.d.ts +1 -1
  91. package/dist/dsp/Sonic.js +3 -3
  92. package/dist/dsp/SpeexResampler.d.ts +1 -1
  93. package/dist/dsp/SpeexResampler.js +2 -2
  94. package/dist/math/VectorMath.d.ts +12 -8
  95. package/dist/math/VectorMath.js +35 -32
  96. package/dist/math/VectorMath.js.map +1 -1
  97. package/dist/nlp/ChineseSegmentation.js +2 -2
  98. package/dist/nlp/CompromiseNLP.js +3 -3
  99. package/dist/nlp/EspeakPhonemizer.js +30 -30
  100. package/dist/nlp/IPA.js +20 -20
  101. package/dist/nlp/JapaneseSegmentation.js +6 -6
  102. package/dist/nlp/Lexicon.d.ts +1 -1
  103. package/dist/nlp/Lexicon.js +7 -7
  104. package/dist/nlp/Segmentation.d.ts +3 -0
  105. package/dist/nlp/Segmentation.js +21 -14
  106. package/dist/nlp/Segmentation.js.map +1 -1
  107. package/dist/nlp/TextNormalizer.js +16 -16
  108. package/dist/recognition/AmazonTranscribeSTT.d.ts +2 -2
  109. package/dist/recognition/AmazonTranscribeSTT.js +13 -14
  110. package/dist/recognition/AmazonTranscribeSTT.js.map +1 -1
  111. package/dist/recognition/AzureCognitiveServicesSTT.js +5 -6
  112. package/dist/recognition/AzureCognitiveServicesSTT.js.map +1 -1
  113. package/dist/recognition/GoogleCloudSTT.d.ts +3 -3
  114. package/dist/recognition/GoogleCloudSTT.js +18 -18
  115. package/dist/recognition/OpenAICloudSTT.d.ts +19 -0
  116. package/dist/recognition/OpenAICloudSTT.js +81 -0
  117. package/dist/recognition/OpenAICloudSTT.js.map +1 -0
  118. package/dist/recognition/SileroSTT.d.ts +2 -2
  119. package/dist/recognition/SileroSTT.js +25 -25
  120. package/dist/recognition/VoskSTT.d.ts +2 -2
  121. package/dist/recognition/VoskSTT.js +8 -8
  122. package/dist/recognition/WhisperCppSTT.d.ts +88 -0
  123. package/dist/recognition/WhisperCppSTT.js +332 -0
  124. package/dist/recognition/WhisperCppSTT.js.map +1 -0
  125. package/dist/recognition/WhisperSTT.d.ts +49 -25
  126. package/dist/recognition/WhisperSTT.js +626 -481
  127. package/dist/recognition/WhisperSTT.js.map +1 -1
  128. package/dist/server/Client.d.ts +1 -1
  129. package/dist/server/Client.js +22 -22
  130. package/dist/server/Server.js +9 -9
  131. package/dist/server/Server.js.map +1 -1
  132. package/dist/server/Worker.d.ts +22 -22
  133. package/dist/server/Worker.js +36 -36
  134. package/dist/server/Worker.js.map +1 -1
  135. package/dist/server/WorkerStarter.js +2 -2
  136. package/dist/source-separation/MDXNetSourceSeparation.d.ts +11 -0
  137. package/dist/source-separation/MDXNetSourceSeparation.js +161 -0
  138. package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -0
  139. package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -1
  140. package/dist/speech-language-detection/SileroLanguageDetection.js +7 -7
  141. package/dist/subtitles/Subtitles.d.ts +10 -0
  142. package/dist/subtitles/Subtitles.js +2 -2
  143. package/dist/subtitles/Subtitles.js.map +1 -1
  144. package/dist/synthesis/AwsPollyTTS.d.ts +1 -1
  145. package/dist/synthesis/AwsPollyTTS.js +12 -12
  146. package/dist/synthesis/AzureCognitiveServicesTTS.js +7 -7
  147. package/dist/synthesis/CoquiServerTTS.js +10 -10
  148. package/dist/synthesis/CoquiServerTTS.js.map +1 -1
  149. package/dist/synthesis/ElevenlabsTTS.d.ts +23 -0
  150. package/dist/synthesis/ElevenlabsTTS.js +103 -0
  151. package/dist/synthesis/ElevenlabsTTS.js.map +1 -0
  152. package/dist/synthesis/EspeakTTS.d.ts +6 -5
  153. package/dist/synthesis/EspeakTTS.js +81 -69
  154. package/dist/synthesis/EspeakTTS.js.map +1 -1
  155. package/dist/synthesis/FliteTTS.d.ts +3 -3
  156. package/dist/synthesis/FliteTTS.js +154 -154
  157. package/dist/synthesis/FliteTTS.js.map +1 -1
  158. package/dist/synthesis/GoogleCloudTTS.d.ts +3 -3
  159. package/dist/synthesis/GoogleCloudTTS.js +17 -17
  160. package/dist/synthesis/GoogleCloudTTS.js.map +1 -1
  161. package/dist/synthesis/GoogleTranslateTTS.d.ts +1 -1
  162. package/dist/synthesis/GoogleTranslateTTS.js +103 -103
  163. package/dist/synthesis/MicrosoftEdgeTTS.d.ts +2 -2
  164. package/dist/synthesis/MicrosoftEdgeTTS.js +74 -74
  165. package/dist/synthesis/OpenAICloudTTS.d.ts +13 -0
  166. package/dist/synthesis/OpenAICloudTTS.js +169 -0
  167. package/dist/synthesis/OpenAICloudTTS.js.map +1 -0
  168. package/dist/synthesis/SamTTS.js +3 -3
  169. package/dist/synthesis/SapiTTS.d.ts +3 -3
  170. package/dist/synthesis/SapiTTS.js +26 -26
  171. package/dist/synthesis/StreamlabsPollyTTS.d.ts +2 -2
  172. package/dist/synthesis/StreamlabsPollyTTS.js +27 -27
  173. package/dist/synthesis/SvoxPicoTTS.d.ts +2 -2
  174. package/dist/synthesis/SvoxPicoTTS.js +65 -65
  175. package/dist/synthesis/SvoxPicoTTS.js.map +1 -1
  176. package/dist/synthesis/VitsTTS.d.ts +3 -3
  177. package/dist/synthesis/VitsTTS.js +378 -378
  178. package/dist/synthesis/VitsTTS.js.map +1 -1
  179. package/dist/tests/Test.js +2 -2
  180. package/dist/utilities/Compression.d.ts +5 -0
  181. package/dist/utilities/Compression.js +29 -13
  182. package/dist/utilities/Compression.js.map +1 -1
  183. package/dist/utilities/FileDownloader.d.ts +1 -1
  184. package/dist/utilities/FileDownloader.js +16 -16
  185. package/dist/utilities/FileSystem.js +7 -7
  186. package/dist/utilities/Locale.d.ts +7 -7
  187. package/dist/utilities/Locale.js +15 -15
  188. package/dist/utilities/Logger.js +3 -3
  189. package/dist/utilities/ObjectUtilities.js +19 -19
  190. package/dist/utilities/OpenPromise.js +2 -2
  191. package/dist/utilities/OpenPromise.js.map +1 -1
  192. package/dist/utilities/PackageManager.js +31 -0
  193. package/dist/utilities/PackageManager.js.map +1 -1
  194. package/dist/utilities/PathUtilities.js +8 -8
  195. package/dist/utilities/RandomGenerator.js +2 -2
  196. package/dist/utilities/SmoothEstimator.d.ts +8 -0
  197. package/dist/utilities/SmoothEstimator.js +25 -0
  198. package/dist/utilities/SmoothEstimator.js.map +1 -0
  199. package/dist/utilities/TarballMaker.js +8 -8
  200. package/dist/utilities/Timeline.d.ts +3 -2
  201. package/dist/utilities/Timeline.js +11 -11
  202. package/dist/utilities/Timeline.js.map +1 -1
  203. package/dist/utilities/Timer.js +4 -4
  204. package/dist/utilities/Utilities.d.ts +4 -0
  205. package/dist/utilities/Utilities.js +38 -15
  206. package/dist/utilities/Utilities.js.map +1 -1
  207. package/dist/utilities/WasmMemoryManager.js +7 -7
  208. package/dist/utilities/WebReader.js +23 -23
  209. package/dist/utilities/WikipediaReader.js +2 -2
  210. package/dist/voice-activity-detection/AdaptiveGateVAD.d.ts +28 -0
  211. package/dist/voice-activity-detection/AdaptiveGateVAD.js +138 -0
  212. package/dist/voice-activity-detection/AdaptiveGateVAD.js.map +1 -0
  213. package/dist/voice-activity-detection/SileroVAD.d.ts +1 -1
  214. package/dist/voice-activity-detection/SileroVAD.js +5 -5
  215. package/dist/voice-activity-detection/SileroVAD.js.map +1 -1
  216. package/dist/voice-activity-detection/WebRtcVAD.d.ts +1 -1
  217. package/dist/voice-activity-detection/WebRtcVAD.js +4 -4
  218. package/docs/API.md +29 -11
  219. package/docs/CLI.md +31 -7
  220. package/docs/Contributing.md +38 -0
  221. package/docs/Development.md +93 -19
  222. package/docs/Engines.md +28 -16
  223. package/docs/Licenses.md +4 -1
  224. package/docs/Options.md +158 -78
  225. package/docs/Releases.md +262 -0
  226. package/docs/Server.md +7 -7
  227. package/docs/Tasklist.md +95 -76
  228. package/docs/Technical.md +4 -4
  229. package/package.json +13 -14
  230. package/src/alignment/DTWMfccSequenceAlignment.ts +9 -9
  231. package/src/alignment/DTWSequenceAlignment.ts +2 -2
  232. package/src/alignment/DTWSequenceAlignmentWindowed.ts +3 -3
  233. package/src/alignment/LevenshteinSequenceAlignment.ts +2 -2
  234. package/src/alignment/SpeechAlignment.ts +204 -119
  235. package/src/api/API.ts +14 -13
  236. package/src/api/APIOptions.ts +12 -11
  237. package/src/api/Alignment.ts +147 -90
  238. package/src/api/Common.ts +1 -1
  239. package/src/api/Denoising.ts +28 -28
  240. package/src/api/LanguageDetection.ts +135 -48
  241. package/src/api/Recognition.ts +198 -59
  242. package/src/api/SourceSeparation.ts +99 -0
  243. package/src/api/Synthesis.ts +217 -181
  244. package/src/api/Translation.ts +193 -40
  245. package/src/api/Vad.ts +110 -41
  246. package/src/audio/AudioBufferConversion.ts +4 -4
  247. package/src/audio/AudioPlayer.ts +27 -27
  248. package/src/audio/AudioRecorder.ts +5 -5
  249. package/src/audio/AudioUtilities.ts +107 -24
  250. package/src/cli/CLI.ts +313 -164
  251. package/src/cli/CLIConfigFile.ts +8 -8
  252. package/src/cli/CLILauncher.ts +6 -6
  253. package/src/cli/CLIOptionsSchema.ts +2 -2
  254. package/src/cli/CLIParser.ts +5 -5
  255. package/src/cli/CLIStarter.ts +4 -4
  256. package/src/codecs/FFMpegTranscoder.ts +38 -38
  257. package/src/codecs/TIMITCodec.ts +5 -5
  258. package/src/codecs/WaveCodec.ts +22 -22
  259. package/src/denoising/RNNoise.ts +9 -9
  260. package/src/dsp/BiquadFilter.ts +19 -11
  261. package/src/dsp/DecayingPeakEstimator.ts +35 -0
  262. package/src/dsp/FFT.ts +103 -35
  263. package/src/dsp/KWeightingFilter.ts +43 -0
  264. package/src/dsp/LoudnessEstimator.ts +74 -0
  265. package/src/dsp/MFCC.ts +15 -15
  266. package/src/dsp/MelSpectogram.ts +7 -7
  267. package/src/dsp/Rubberband.ts +38 -38
  268. package/src/dsp/Sonic.ts +4 -4
  269. package/src/dsp/SpeexResampler.ts +2 -2
  270. package/src/math/VectorMath.ts +42 -33
  271. package/src/nlp/ChineseSegmentation.ts +3 -3
  272. package/src/nlp/CompromiseNLP.ts +3 -3
  273. package/src/nlp/EspeakPhonemizer.ts +30 -30
  274. package/src/nlp/IPA.ts +20 -20
  275. package/src/nlp/JapaneseSegmentation.ts +6 -6
  276. package/src/nlp/Lexicon.ts +8 -8
  277. package/src/nlp/Segmentation.ts +23 -14
  278. package/src/nlp/TextNormalizer.ts +16 -16
  279. package/src/recognition/AmazonTranscribeSTT.ts +16 -17
  280. package/src/recognition/AzureCognitiveServicesSTT.ts +8 -6
  281. package/src/recognition/GoogleCloudSTT.ts +21 -21
  282. package/src/recognition/OpenAICloudSTT.ts +142 -0
  283. package/src/recognition/SileroSTT.ts +26 -26
  284. package/src/recognition/VoskSTT.ts +10 -10
  285. package/src/recognition/WhisperCppSTT.ts +555 -0
  286. package/src/recognition/WhisperSTT.ts +760 -507
  287. package/src/server/Client.ts +23 -23
  288. package/src/server/Server.ts +9 -9
  289. package/src/server/Worker.ts +53 -53
  290. package/src/server/WorkerStarter.ts +2 -2
  291. package/src/source-separation/MDXNetSourceSeparation.ts +228 -0
  292. package/src/speech-language-detection/SileroLanguageDetection.ts +8 -8
  293. package/src/subtitles/Subtitles.ts +3 -3
  294. package/src/synthesis/AwsPollyTTS.ts +14 -14
  295. package/src/synthesis/AzureCognitiveServicesTTS.ts +10 -10
  296. package/src/synthesis/CoquiServerTTS.ts +10 -10
  297. package/src/synthesis/ElevenlabsTTS.ts +137 -0
  298. package/src/synthesis/EspeakTTS.ts +90 -71
  299. package/src/synthesis/FliteTTS.ts +157 -157
  300. package/src/synthesis/GoogleCloudTTS.ts +19 -19
  301. package/src/synthesis/GoogleTranslateTTS.ts +104 -104
  302. package/src/synthesis/MicrosoftEdgeTTS.ts +80 -80
  303. package/src/synthesis/OpenAICloudTTS.ts +196 -0
  304. package/src/synthesis/SamTTS.ts +3 -3
  305. package/src/synthesis/SapiTTS.ts +29 -29
  306. package/src/synthesis/StreamlabsPollyTTS.ts +29 -29
  307. package/src/synthesis/SvoxPicoTTS.ts +67 -67
  308. package/src/synthesis/VitsTTS.ts +380 -380
  309. package/src/tests/Test.ts +4 -4
  310. package/src/utilities/Compression.ts +34 -13
  311. package/src/utilities/FileDownloader.ts +19 -19
  312. package/src/utilities/FileSystem.ts +7 -7
  313. package/src/utilities/Locale.ts +22 -22
  314. package/src/utilities/Logger.ts +4 -4
  315. package/src/utilities/ObjectUtilities.ts +19 -19
  316. package/src/utilities/OpenPromise.ts +2 -2
  317. package/src/utilities/PackageManager.ts +40 -0
  318. package/src/utilities/PathUtilities.ts +8 -8
  319. package/src/utilities/RandomGenerator.ts +3 -3
  320. package/src/utilities/SmoothEstimator.ts +35 -0
  321. package/src/utilities/TarballMaker.ts +9 -9
  322. package/src/utilities/Timeline.ts +15 -13
  323. package/src/utilities/Timer.ts +4 -4
  324. package/src/utilities/Utilities.ts +49 -15
  325. package/src/utilities/WasmMemoryManager.ts +7 -7
  326. package/src/utilities/WebReader.ts +23 -23
  327. package/src/utilities/WikipediaReader.ts +2 -2
  328. package/src/voice-activity-detection/AdaptiveGateVAD.ts +202 -0
  329. package/src/voice-activity-detection/SileroVAD.ts +5 -5
  330. package/src/voice-activity-detection/WebRtcVAD.ts +5 -5
  331. package/dist/synthesis/ElevenLabsTTS.d.ts +0 -8
  332. package/dist/synthesis/ElevenLabsTTS.js +0 -82
  333. package/dist/synthesis/ElevenLabsTTS.js.map +0 -1
  334. package/src/synthesis/ElevenLabsTTS.ts +0 -104
@@ -0,0 +1,228 @@
1
+ import Onnx from 'onnxruntime-node'
2
+ import { RawAudio } from '../audio/AudioUtilities.js';
3
+ import { binBufferToComplex, complexToBinBuffer, getWindowWeights, stftr, stiftr } from '../dsp/FFT.js';
4
+ import { ComplexNumber } from '../math/VectorMath.js';
5
+ import { logToStderr } from '../utilities/Utilities.js';
6
+ import { Logger } from '../utilities/Logger.js';
7
+
8
+ const log = logToStderr
9
+
10
+ export async function isolate(rawAudio: RawAudio, modelFilePath: string) {
11
+ const model = new MDXNet(modelFilePath)
12
+
13
+ return model.processAudio(rawAudio)
14
+ }
15
+
16
+ export class MDXNet {
17
+ session?: Onnx.InferenceSession
18
+
19
+ constructor(public readonly modelFilePath: string) {
20
+ }
21
+
22
+ async processAudio(rawAudio: RawAudio) {
23
+ if (rawAudio.audioChannels.length != 2) {
24
+ throw new Error(`Input audio must be stereo`)
25
+ }
26
+
27
+ if (rawAudio.sampleRate != 44100) {
28
+ throw new Error(`Input audio must have a 44100 Hz sampling rate`)
29
+ }
30
+
31
+ if (!this.session) {
32
+ await this.initializeSession(this.modelFilePath)
33
+ }
34
+
35
+ const logger = new Logger()
36
+
37
+ const session = this.session!
38
+
39
+ const sampleRate = rawAudio.sampleRate
40
+ const fftSize = 6144
41
+ const fftCount = 2048
42
+ const fftWindowSize = fftSize
43
+ const fftHopSize = 1024
44
+
45
+ const segmentSize = 256
46
+ const segmentHopSize = 240
47
+
48
+ const sampleCount = rawAudio.audioChannels[0].length
49
+
50
+ logger.start('Compute STFT of full waveform')
51
+
52
+ const fftFramesLeft = await stftr(rawAudio.audioChannels[0], fftSize, fftWindowSize, fftHopSize, 'hann')
53
+ const fftFramesRight = await stftr(rawAudio.audioChannels[1], fftSize, fftWindowSize, fftHopSize, 'hann')
54
+
55
+ const fftFramesLeftComplex = fftFramesLeft.map(frame => binBufferToComplex(frame).slice(0, fftCount))
56
+ const fftFramesRightComplex = fftFramesRight.map(frame => binBufferToComplex(frame).slice(0, fftCount))
57
+
58
+ const audioForSegments: Float32Array[][] = []
59
+
60
+ for (let segmentOffset = 0; segmentOffset < fftFramesLeft.length; segmentOffset += segmentHopSize) {
61
+ const timePosition = segmentOffset * (fftHopSize / sampleRate)
62
+
63
+ logger.start(`Process segment at time position ${timePosition.toFixed(2)}`)
64
+
65
+ const fftFramesLeftComplexForSegment = fftFramesLeftComplex.slice(segmentOffset, segmentOffset + segmentSize)
66
+ const fftFramesRightComplexForSegment = fftFramesRightComplex.slice(segmentOffset, segmentOffset + segmentSize)
67
+
68
+ const segmentLength = fftFramesLeftComplexForSegment.length
69
+
70
+ const flattenedInputTensor = new Float32Array(1 * 4 * fftCount * segmentSize)
71
+
72
+ {
73
+ let writePosition = 0
74
+
75
+ for (let tensorChannelIndex = 0; tensorChannelIndex < 4; tensorChannelIndex++) {
76
+ for (let binIndex = 0; binIndex < fftCount; binIndex++) {
77
+ for (let frameIndex = 0; frameIndex < segmentSize; frameIndex++) {
78
+ let value = 0
79
+
80
+ if (frameIndex < segmentLength && binIndex >= 0) {
81
+ let frame: ComplexNumber[]
82
+
83
+ if (tensorChannelIndex < 2) {
84
+ frame = fftFramesLeftComplexForSegment[frameIndex]
85
+ } else {
86
+ frame = fftFramesRightComplexForSegment[frameIndex]
87
+ }
88
+
89
+ const bin = frame[binIndex]
90
+
91
+ if (tensorChannelIndex % 2 === 0) {
92
+ value = bin.real
93
+ } else {
94
+ value = bin.imaginary
95
+ }
96
+ }
97
+
98
+ flattenedInputTensor[writePosition++] = value
99
+ }
100
+ }
101
+ }
102
+ }
103
+
104
+ const inputTensor = new Onnx.Tensor('float32', flattenedInputTensor, [1, 4, 2048, 256])
105
+
106
+ const { output: outputTensor } = await session.run({ input: inputTensor })
107
+
108
+ const flattenedOutputTensor = outputTensor.data as Float32Array
109
+
110
+ const outputChannelComplexFrames: ComplexNumber[][][] = []
111
+
112
+ {
113
+ for (let outChannelIndex = 0; outChannelIndex < 2; outChannelIndex++) {
114
+ const framesForChannel: ComplexNumber[][] = []
115
+
116
+ for (let frameIndex = 0; frameIndex < 256; frameIndex++) {
117
+ const frame: ComplexNumber[] = []
118
+
119
+ for (let binIndex = 0; binIndex < fftSize; binIndex++) {
120
+ frame.push({ real: 0, imaginary: 0 })
121
+ }
122
+
123
+ framesForChannel.push(frame)
124
+ }
125
+
126
+ outputChannelComplexFrames.push(framesForChannel)
127
+ }
128
+
129
+ let readPosition = 0
130
+
131
+ for (let tensorChannelIndex = 0; tensorChannelIndex < 4; tensorChannelIndex++) {
132
+ const outChannelIndex = tensorChannelIndex < 2 ? 0 : 1
133
+
134
+ for (let binIndex = 0; binIndex < 2048; binIndex++) {
135
+ for (let frameIndex = 0; frameIndex < 256; frameIndex++) {
136
+ const bin = outputChannelComplexFrames[outChannelIndex][frameIndex][binIndex]
137
+
138
+ if (tensorChannelIndex % 2 === 0) {
139
+ bin.real = flattenedOutputTensor[readPosition++]
140
+ } else {
141
+ bin.imaginary = flattenedOutputTensor[readPosition++]
142
+ }
143
+ }
144
+ }
145
+ }
146
+ }
147
+
148
+ const outputAudioChannels: Float32Array[] = []
149
+
150
+ //logger.start(`Compute inverse STFT for segment`)
151
+ for (let channelIndex = 0; channelIndex < 2; channelIndex++) {
152
+ let outputChannelFlattenedFrames = outputChannelComplexFrames[channelIndex]
153
+ .map(frame => complexToBinBuffer(frame).map(value => value / fftSize))
154
+
155
+ const samples = await stiftr(
156
+ outputChannelFlattenedFrames,
157
+ fftSize,
158
+ fftWindowSize,
159
+ fftHopSize,
160
+ 'hann')
161
+
162
+ outputAudioChannels.push(samples)
163
+ }
164
+
165
+ //logger.log(`Reconstructed waveform peak: ${getAudioPeakDecibels(outputAudioChannels).toFixed(3)}dB`)
166
+ //await playAudioSamples({ audioChannels: outputAudioChannels, sampleRate })
167
+
168
+ audioForSegments.push(outputAudioChannels)
169
+ }
170
+
171
+ // Join segments using overlapping Hann windows
172
+ logger.start(`Join segments`)
173
+ const concatenatedAudioChannels = [new Float32Array(sampleCount), new Float32Array(sampleCount)]
174
+
175
+ {
176
+ const segmentCount = audioForSegments.length
177
+
178
+ const segmentSampleCount = audioForSegments[0][0].length
179
+
180
+ const windowWeights = getWindowWeights('hann', segmentSampleCount)
181
+
182
+ const sumOfWeightsForSample = new Float32Array(sampleCount)
183
+
184
+ for (let segmentIndex = 0; segmentIndex < segmentCount; segmentIndex++) {
185
+ const segmentStartFrameIndex = segmentIndex * segmentHopSize
186
+ const segmentStartSampleIndex = segmentStartFrameIndex * fftHopSize
187
+
188
+ const segmentSamples = audioForSegments[segmentIndex]
189
+
190
+ for (let segmentSampleOffset = 0; segmentSampleOffset < segmentSampleCount; segmentSampleOffset++) {
191
+ const sampleIndex = segmentStartSampleIndex + segmentSampleOffset
192
+
193
+ if (sampleIndex >= sampleCount) {
194
+ break
195
+ }
196
+
197
+ const weight = windowWeights[segmentSampleOffset]
198
+
199
+ for (let channelIndex = 0; channelIndex < 2; channelIndex++) {
200
+ concatenatedAudioChannels[channelIndex][sampleIndex] += segmentSamples[channelIndex][segmentSampleOffset] * weight
201
+ }
202
+
203
+ sumOfWeightsForSample[sampleIndex] += weight
204
+ }
205
+ }
206
+
207
+ for (let sampleIndex = 0; sampleIndex < sampleCount; sampleIndex++) {
208
+ for (let channelIndex = 0; channelIndex < 2; channelIndex++) {
209
+ concatenatedAudioChannels[channelIndex][sampleIndex] /= sumOfWeightsForSample[sampleIndex] + 1e-8
210
+ }
211
+ }
212
+ }
213
+
214
+ const isolatedRawAudio: RawAudio = { audioChannels: concatenatedAudioChannels, sampleRate }
215
+
216
+ logger.end()
217
+
218
+ return isolatedRawAudio
219
+ }
220
+
221
+ private async initializeSession(modelPath: string) {
222
+ const onnxOptions: Onnx.InferenceSession.SessionOptions = {
223
+ logSeverityLevel: 3
224
+ }
225
+
226
+ this.session = await Onnx.InferenceSession.create(modelPath, onnxOptions)
227
+ }
228
+ }
@@ -1,7 +1,7 @@
1
1
  import Onnx from 'onnxruntime-node'
2
2
  import { softmax } from '../math/VectorMath.js'
3
- import { Logger } from "../utilities/Logger.js"
4
- import { RawAudio } from "../audio/AudioUtilities.js"
3
+ import { Logger } from '../utilities/Logger.js'
4
+ import { RawAudio } from '../audio/AudioUtilities.js'
5
5
  import { readAndParseJsonFile } from '../utilities/FileSystem.js'
6
6
  import { detectSpeechLanguageByParts, type LanguageDetectionResults } from '../api/LanguageDetection.js'
7
7
  import { languageCodeToName } from '../utilities/Locale.js'
@@ -41,7 +41,7 @@ export class SileroLanguageDetection {
41
41
 
42
42
  async initialize() {
43
43
  const logger = new Logger()
44
- logger.start("Initialize ONNX inference session")
44
+ logger.start('Initialize ONNX inference session')
45
45
 
46
46
  this.languageDictionary = await readAndParseJsonFile(this.languageDictionaryPath)
47
47
  this.languageGroupDictionary = await readAndParseJsonFile(this.languageGroupDictionaryPath)
@@ -58,7 +58,7 @@ export class SileroLanguageDetection {
58
58
  async detectLanguage(rawAudio: RawAudio) {
59
59
  const logger = new Logger()
60
60
 
61
- logger.start("Detect language with Silero")
61
+ logger.start('Detect language with Silero')
62
62
 
63
63
  const audioSamples = rawAudio.audioChannels[0]
64
64
 
@@ -68,10 +68,10 @@ export class SileroLanguageDetection {
68
68
 
69
69
  const results = await this.session!.run(inputs)
70
70
 
71
- logger.start("Parse model results")
71
+ logger.start('Parse model results')
72
72
 
73
- const languageLogits = results["output"].data
74
- const languageGroupLogits = results["2038"].data
73
+ const languageLogits = results['output'].data
74
+ const languageGroupLogits = results['2038'].data
75
75
 
76
76
  const languageProbabilities = softmax(languageLogits as any)
77
77
  const languageGroupProbabilities = softmax(languageGroupLogits as any)
@@ -80,7 +80,7 @@ export class SileroLanguageDetection {
80
80
 
81
81
  for (let i = 0; i < languageProbabilities.length; i++) {
82
82
  const languageString = this.languageDictionary[i]
83
- const languageCode = languageString.replace(/,.*$/, "")
83
+ const languageCode = languageString.replace(/,.*$/, '')
84
84
 
85
85
  languageResults.push({
86
86
  language: languageCode,
@@ -428,11 +428,11 @@ function getCuesFromTimeline_IsolateLines(timeline: Timeline, config: SubtitlesC
428
428
 
429
429
  addCuesFrom(timeline)
430
430
  addCueFromCurrentWords() // Add any remaining words
431
-
431
+
432
432
  return cues
433
433
  }
434
434
 
435
- function tryParseTimeRangePatternWithHours(line: string) {
435
+ export function tryParseTimeRangePatternWithHours(line: string) {
436
436
  const timeRangePatternWithHours = /^(\d+)\:(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)\:(\d+)[\.,](\d+)/
437
437
  const match = timeRangePatternWithHours.exec(line)
438
438
 
@@ -456,7 +456,7 @@ function tryParseTimeRangePatternWithHours(line: string) {
456
456
  return { startTime, endTime, succeeded: true }
457
457
  }
458
458
 
459
- function tryParseTimeRangePatternWithoutHours(line: string) {
459
+ export function tryParseTimeRangePatternWithoutHours(line: string) {
460
460
  const timeRangePatternWithHours = /^(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)[\.,](\d+)/
461
461
  const match = timeRangePatternWithHours.exec(line)
462
462
 
@@ -1,15 +1,15 @@
1
- import type { LanguageCode, SynthesizeSpeechCommandInput, VoiceId } from "@aws-sdk/client-polly"
2
- import { IncomingMessage } from "http"
3
- import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
4
- import { Logger } from "../utilities/Logger.js"
1
+ import type { LanguageCode, SynthesizeSpeechCommandInput, VoiceId } from '@aws-sdk/client-polly'
2
+ import { IncomingMessage } from 'http'
3
+ import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
4
+ import { Logger } from '../utilities/Logger.js'
5
5
 
6
- import { readBinaryIncomingMessage } from "../utilities/Utilities.js"
6
+ import { readBinaryIncomingMessage } from '../utilities/Utilities.js'
7
7
 
8
- export async function synthesize(text: string, language: string | undefined, voice: string, region: string, accessKeyId: string, secretAccessKey: string, engine: "standard" | "neural" = "standard", ssmlEnabled = false, lexiconNames?: string[]) {
8
+ export async function synthesize(text: string, language: string | undefined, voice: string, region: string, accessKeyId: string, secretAccessKey: string, engine: 'standard' | 'neural' = 'standard', ssmlEnabled = false, lexiconNames?: string[]) {
9
9
  const logger = new Logger()
10
- logger.start("Load AWS SDK client module")
10
+ logger.start('Load AWS SDK client module')
11
11
 
12
- const polly = await import("@aws-sdk/client-polly")
12
+ const polly = await import('@aws-sdk/client-polly')
13
13
 
14
14
  const pollyClient = new polly.PollyClient({
15
15
  region,
@@ -28,12 +28,12 @@ export async function synthesize(text: string, language: string | undefined, voi
28
28
  Text: text,
29
29
  LexiconNames: lexiconNames,
30
30
 
31
- TextType: ssmlEnabled ? "ssml" : "text",
31
+ TextType: ssmlEnabled ? 'ssml' : 'text',
32
32
 
33
- OutputFormat: "mp3",
33
+ OutputFormat: 'mp3',
34
34
  }
35
35
 
36
- logger.start("Request synthesis from AWS Polly")
36
+ logger.start('Request synthesis from AWS Polly')
37
37
 
38
38
  const command = new polly.SynthesizeSpeechCommand(params)
39
39
 
@@ -52,11 +52,11 @@ export async function synthesize(text: string, language: string | undefined, voi
52
52
 
53
53
  export async function getVoiceList(region: string, accessKeyId: string, secretAccessKey: string) {
54
54
  const logger = new Logger()
55
- logger.start("Load AWS SDK client module")
55
+ logger.start('Load AWS SDK client module')
56
56
 
57
- const polly = await import("@aws-sdk/client-polly")
57
+ const polly = await import('@aws-sdk/client-polly')
58
58
 
59
- logger.start("Request voice list from AWS Polly")
59
+ logger.start('Request voice list from AWS Polly')
60
60
 
61
61
  const pollyClient = new polly.PollyClient({
62
62
  region,
@@ -1,6 +1,6 @@
1
1
  import * as SpeechSDK from 'microsoft-cognitiveservices-speech-sdk'
2
2
 
3
- import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
3
+ import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
4
4
 
5
5
  import { escape } from 'html-escaper'
6
6
 
@@ -12,15 +12,15 @@ export async function synthesize(
12
12
  text: string,
13
13
  subscriptionKey: string,
14
14
  serviceRegion: string,
15
- languageCode = "en-US",
16
- voice = "Microsoft Server Speech Text to Speech Voice (en-US, AvaNeural)",
15
+ languageCode = 'en-US',
16
+ voice = 'Microsoft Server Speech Text to Speech Voice (en-US, AvaNeural)',
17
17
  ssmlEnabled = false,
18
- ssmlPitchString = "+0Hz",
19
- ssmlRateString = "+0%") {
18
+ ssmlPitchString = '+0Hz',
19
+ ssmlRateString = '+0%') {
20
20
 
21
21
  return new Promise<{ rawAudio: RawAudio, timeline: Timeline }>((resolve, reject) => {
22
22
  const logger = new Logger()
23
- logger.start("Request synthesis from Azure Cognitive Services")
23
+ logger.start('Request synthesis from Azure Cognitive Services')
24
24
 
25
25
  const speechConfig = SpeechSDK.SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
26
26
 
@@ -72,7 +72,7 @@ export async function synthesize(
72
72
 
73
73
  const rawAudio = await FFMpegTranscoder.decodeToChannels(encodedAudio, 24000, 1)
74
74
 
75
- logger.start("Convert boundary events to a timeline")
75
+ logger.start('Convert boundary events to a timeline')
76
76
 
77
77
  const timeline = boundaryEventsToTimeline(events, getRawAudioDuration(rawAudio))
78
78
 
@@ -85,7 +85,7 @@ export async function synthesize(
85
85
  reject(error)
86
86
  }
87
87
 
88
- if (!ssmlEnabled && ssmlPitchString != "+0%" || ssmlRateString != "+0Hz") {
88
+ if (!ssmlEnabled && ssmlPitchString != '+0%' || ssmlRateString != '+0Hz') {
89
89
  ssmlEnabled = true
90
90
  text = escape(text)
91
91
  }
@@ -123,7 +123,7 @@ export function boundaryEventsToTimeline(events: any[], totalDuration: number) {
123
123
  for (const event of events) {
124
124
  const boundaryType = event.boundaryType != null ? event.boundaryType : event.Type
125
125
 
126
- if (boundaryType != "WordBoundary") {
126
+ if (boundaryType != 'WordBoundary') {
127
127
  continue
128
128
  }
129
129
 
@@ -135,7 +135,7 @@ export function boundaryEventsToTimeline(events: any[], totalDuration: number) {
135
135
  const endTime = (offset + duration) / 10000000
136
136
 
137
137
  timeline.push({
138
- type: "word",
138
+ type: 'word',
139
139
  text,
140
140
  startTime,
141
141
  endTime
@@ -1,27 +1,27 @@
1
- import { request } from "gaxios"
2
- import { decodeWaveBuffer } from "../audio/AudioUtilities.js"
3
- import { Logger } from "../utilities/Logger.js"
4
- import { logToStderr } from "../utilities/Utilities.js"
1
+ import { request } from 'gaxios'
2
+ import { decodeWaveToRawAudio } from '../audio/AudioUtilities.js'
3
+ import { Logger } from '../utilities/Logger.js'
4
+ import { logToStderr } from '../utilities/Utilities.js'
5
5
  const log = logToStderr
6
6
 
7
- export async function synthesize(text: string, speakerId: string | null, serverURL = "http://[::1]:5002") {
7
+ export async function synthesize(text: string, speakerId: string | null, serverURL = 'http://[::1]:5002') {
8
8
  const logger = new Logger()
9
- logger.start("Request synthesis from Coqui Server")
9
+ logger.start('Request synthesis from Coqui Server')
10
10
 
11
11
  const response = await request<Buffer>({
12
12
  url: `${serverURL}/api/tts`,
13
13
 
14
14
  params: {
15
- "text": text,
16
- "speaker_id": speakerId
15
+ 'text': text,
16
+ 'speaker_id': speakerId
17
17
  },
18
18
 
19
- responseType: "arraybuffer"
19
+ responseType: 'arraybuffer'
20
20
  })
21
21
 
22
22
  const waveData = Buffer.from(response.data)
23
23
 
24
- const rawAudio = decodeWaveBuffer(waveData).rawAudio
24
+ const rawAudio = decodeWaveToRawAudio(waveData).rawAudio
25
25
 
26
26
  logger.end()
27
27
 
@@ -0,0 +1,137 @@
1
+ import { GaxiosResponse, request } from 'gaxios'
2
+ import { SynthesisVoice, VoiceGender } from '../api/API.js'
3
+ import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
4
+ import { Logger } from '../utilities/Logger.js'
5
+ import { logToStderr } from '../utilities/Utilities.js'
6
+ import { extendDeep } from '../utilities/ObjectUtilities.js'
7
+
8
+ const log = logToStderr
9
+
10
+ export async function synthesize(text: string, voiceId: string, modelId: string, options: ElevenlabsTTSOptions) {
11
+ const logger = new Logger()
12
+ logger.start('Request synthesis from ElevenLabs')
13
+
14
+ options = extendDeep(defaultElevenlabsTTSOptions, options)
15
+
16
+ let response: GaxiosResponse<any>
17
+
18
+ try {
19
+ response = await request<any>({
20
+ url: `https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`,
21
+
22
+ method: 'POST',
23
+
24
+ headers: {
25
+ 'accept': 'audio/mpeg',
26
+ 'xi-api-key': options.apiKey,
27
+ },
28
+
29
+ data: {
30
+ text,
31
+
32
+ model_id: modelId,
33
+
34
+ voice_setting: {
35
+ stability: options.stability,
36
+ similarity_boost: options.similarityBoost,
37
+ style: options.style,
38
+ use_speaker_boost: options.useSpeakerBoost
39
+ }
40
+ },
41
+
42
+ responseType: 'arraybuffer'
43
+ })
44
+ } catch (e: any) {
45
+ const response = e.response
46
+
47
+ if (response) {
48
+ logger.log(`Request failed with status code ${response.status}`)
49
+
50
+ if (response.data) {
51
+ logger.log(`Server responded with:`)
52
+ logger.log(response.data)
53
+ }
54
+ }
55
+
56
+ throw e
57
+ }
58
+
59
+ logger.start('Decode synthesized audio')
60
+ const rawAudio = await FFMpegTranscoder.decodeToChannels(Buffer.from(response.data))
61
+
62
+ logger.end()
63
+
64
+ return { rawAudio }
65
+ }
66
+
67
+ export async function getVoiceList(apiKey: string) {
68
+ const response = await request<any>({
69
+ method: 'GET',
70
+
71
+ url: 'https://api.elevenlabs.io/v1/voices',
72
+
73
+ headers: {
74
+ 'accept': 'accept: application/json',
75
+ 'xi-api-key': apiKey
76
+ },
77
+
78
+ responseType: 'json'
79
+ })
80
+
81
+ const elevenlabsVoices: any[] = response.data.voices
82
+
83
+ const voices: SynthesisVoice[] = elevenlabsVoices.map(elevenlabsVoice => {
84
+ const modelId: string = elevenlabsVoice?.high_quality_base_model_ids?.[0] ?? 'eleven_monolingual_v1'
85
+ const accent: string | undefined = elevenlabsVoice?.labels?.accent
86
+ const gender: VoiceGender = elevenlabsVoice?.labels?.gender ?? 'unknown'
87
+
88
+ const supportedLanguages: string[] = []
89
+
90
+ if (accent) {
91
+ if (accent.startsWith('american')) {
92
+ supportedLanguages.push('en-US')
93
+ } else if (accent.startsWith('british')) {
94
+ supportedLanguages.push('en-GB')
95
+ } else if (accent === 'irish') {
96
+ supportedLanguages.push('en-IE')
97
+ } else if (accent == 'australian') {
98
+ supportedLanguages.push('en-AU')
99
+ }
100
+ }
101
+
102
+ if (modelId.includes('multilingual')) {
103
+ supportedLanguages.push('en', ...supporteMultilingualLanguages)
104
+ } else {
105
+ supportedLanguages.push('en')
106
+ }
107
+
108
+ return {
109
+ name: elevenlabsVoice.name,
110
+ languages: supportedLanguages,
111
+ gender,
112
+
113
+ elevenLabsVoiceId: elevenlabsVoice.voice_id,
114
+ elevenLabsModelId: modelId
115
+ }
116
+ })
117
+
118
+ return voices
119
+ }
120
+
121
+ export interface ElevenlabsTTSOptions {
122
+ apiKey?: string
123
+ stability?: number
124
+ similarityBoost?: number
125
+ style?: number
126
+ useSpeakerBoost?: boolean
127
+ }
128
+
129
+ export const defaultElevenlabsTTSOptions = {
130
+ apiKey: undefined,
131
+ stability: 0.5,
132
+ similarityBoost: 0.5,
133
+ style: 0,
134
+ useSpeakerBoost: true
135
+ }
136
+
137
+ export const supporteMultilingualLanguages = ['zh', 'ko', 'nl', 'tr', 'sv', 'id', 'tl', 'ja', 'uk', 'el', 'cs', 'fi', 'ro', 'ru', 'da', 'bg', 'ms', 'sk', 'hr', 'ar', 'ta', 'pl', 'de', 'es', 'fr', 'it', 'hi', 'pt']