echogarden 0.12.1 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (334) hide show
  1. package/README.md +15 -14
  2. package/data/schemas/options.json +398 -111
  3. package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
  4. package/dist/alignment/DTWMfccSequenceAlignment.js +8 -8
  5. package/dist/alignment/DTWSequenceAlignment.d.ts +1 -1
  6. package/dist/alignment/DTWSequenceAlignment.js +1 -1
  7. package/dist/alignment/DTWSequenceAlignmentWindowed.d.ts +1 -1
  8. package/dist/alignment/DTWSequenceAlignmentWindowed.js +2 -2
  9. package/dist/alignment/LevenshteinSequenceAlignment.d.ts +1 -1
  10. package/dist/alignment/LevenshteinSequenceAlignment.js +1 -1
  11. package/dist/alignment/SpeechAlignment.d.ts +9 -10
  12. package/dist/alignment/SpeechAlignment.js +136 -105
  13. package/dist/alignment/SpeechAlignment.js.map +1 -1
  14. package/dist/api/API.d.ts +13 -12
  15. package/dist/api/API.js +14 -13
  16. package/dist/api/API.js.map +1 -1
  17. package/dist/api/APIOptions.d.ts +5 -4
  18. package/dist/api/Alignment.d.ts +15 -9
  19. package/dist/api/Alignment.js +88 -74
  20. package/dist/api/Alignment.js.map +1 -1
  21. package/dist/api/Common.js +1 -1
  22. package/dist/api/Denoising.d.ts +6 -6
  23. package/dist/api/Denoising.js +23 -23
  24. package/dist/api/Denoising.js.map +1 -1
  25. package/dist/api/LanguageDetection.d.ts +19 -12
  26. package/dist/api/LanguageDetection.js +88 -38
  27. package/dist/api/LanguageDetection.js.map +1 -1
  28. package/dist/api/Recognition.d.ts +16 -6
  29. package/dist/api/Recognition.js +129 -55
  30. package/dist/api/Recognition.js.map +1 -1
  31. package/dist/api/SourceSeparation.d.ts +17 -0
  32. package/dist/api/SourceSeparation.js +61 -0
  33. package/dist/api/SourceSeparation.js.map +1 -0
  34. package/dist/api/Synthesis.d.ts +18 -18
  35. package/dist/api/Synthesis.js +191 -164
  36. package/dist/api/Synthesis.js.map +1 -1
  37. package/dist/api/Translation.d.ts +19 -8
  38. package/dist/api/Translation.js +132 -35
  39. package/dist/api/Translation.js.map +1 -1
  40. package/dist/api/Vad.d.ts +10 -5
  41. package/dist/api/Vad.js +76 -38
  42. package/dist/api/Vad.js.map +1 -1
  43. package/dist/audio/AudioBufferConversion.d.ts +1 -1
  44. package/dist/audio/AudioBufferConversion.js +4 -4
  45. package/dist/audio/AudioPlayer.d.ts +1 -1
  46. package/dist/audio/AudioPlayer.js +26 -26
  47. package/dist/audio/AudioPlayer.js.map +1 -1
  48. package/dist/audio/AudioRecorder.d.ts +1 -1
  49. package/dist/audio/AudioRecorder.js +5 -5
  50. package/dist/audio/AudioUtilities.d.ts +13 -9
  51. package/dist/audio/AudioUtilities.js +86 -24
  52. package/dist/audio/AudioUtilities.js.map +1 -1
  53. package/dist/cli/CLI.d.ts +3 -3
  54. package/dist/cli/CLI.js +271 -162
  55. package/dist/cli/CLI.js.map +1 -1
  56. package/dist/cli/CLIConfigFile.js +8 -8
  57. package/dist/cli/CLILauncher.js +6 -6
  58. package/dist/cli/CLIOptionsSchema.js +2 -2
  59. package/dist/cli/CLIParser.js +5 -5
  60. package/dist/cli/CLIStarter.js +4 -4
  61. package/dist/codecs/FFMpegTranscoder.d.ts +2 -2
  62. package/dist/codecs/FFMpegTranscoder.js +37 -37
  63. package/dist/codecs/FFMpegTranscoder.js.map +1 -1
  64. package/dist/codecs/TIMITCodec.js +5 -5
  65. package/dist/codecs/WaveCodec.d.ts +1 -1
  66. package/dist/codecs/WaveCodec.js +22 -22
  67. package/dist/denoising/RNNoise.d.ts +1 -1
  68. package/dist/denoising/RNNoise.js +9 -9
  69. package/dist/dsp/BiquadFilter.d.ts +3 -2
  70. package/dist/dsp/BiquadFilter.js +18 -11
  71. package/dist/dsp/BiquadFilter.js.map +1 -1
  72. package/dist/dsp/DecayingPeakEstimator.d.ts +16 -0
  73. package/dist/dsp/DecayingPeakEstimator.js +23 -0
  74. package/dist/dsp/DecayingPeakEstimator.js.map +1 -0
  75. package/dist/dsp/FFT.d.ts +8 -4
  76. package/dist/dsp/FFT.js +76 -30
  77. package/dist/dsp/FFT.js.map +1 -1
  78. package/dist/dsp/KWeightingFilter.d.ts +9 -0
  79. package/dist/dsp/KWeightingFilter.js +40 -0
  80. package/dist/dsp/KWeightingFilter.js.map +1 -0
  81. package/dist/dsp/LoudnessEstimator.d.ts +21 -0
  82. package/dist/dsp/LoudnessEstimator.js +47 -0
  83. package/dist/dsp/LoudnessEstimator.js.map +1 -0
  84. package/dist/dsp/MFCC.d.ts +2 -2
  85. package/dist/dsp/MFCC.js +15 -15
  86. package/dist/dsp/MelSpectogram.d.ts +1 -1
  87. package/dist/dsp/MelSpectogram.js +6 -6
  88. package/dist/dsp/Rubberband.d.ts +11 -11
  89. package/dist/dsp/Rubberband.js +27 -27
  90. package/dist/dsp/Sonic.d.ts +1 -1
  91. package/dist/dsp/Sonic.js +3 -3
  92. package/dist/dsp/SpeexResampler.d.ts +1 -1
  93. package/dist/dsp/SpeexResampler.js +2 -2
  94. package/dist/math/VectorMath.d.ts +12 -8
  95. package/dist/math/VectorMath.js +35 -32
  96. package/dist/math/VectorMath.js.map +1 -1
  97. package/dist/nlp/ChineseSegmentation.js +2 -2
  98. package/dist/nlp/CompromiseNLP.js +3 -3
  99. package/dist/nlp/EspeakPhonemizer.js +30 -30
  100. package/dist/nlp/IPA.js +20 -20
  101. package/dist/nlp/JapaneseSegmentation.js +6 -6
  102. package/dist/nlp/Lexicon.d.ts +1 -1
  103. package/dist/nlp/Lexicon.js +7 -7
  104. package/dist/nlp/Segmentation.d.ts +3 -0
  105. package/dist/nlp/Segmentation.js +21 -14
  106. package/dist/nlp/Segmentation.js.map +1 -1
  107. package/dist/nlp/TextNormalizer.js +16 -16
  108. package/dist/recognition/AmazonTranscribeSTT.d.ts +2 -2
  109. package/dist/recognition/AmazonTranscribeSTT.js +13 -14
  110. package/dist/recognition/AmazonTranscribeSTT.js.map +1 -1
  111. package/dist/recognition/AzureCognitiveServicesSTT.js +5 -6
  112. package/dist/recognition/AzureCognitiveServicesSTT.js.map +1 -1
  113. package/dist/recognition/GoogleCloudSTT.d.ts +3 -3
  114. package/dist/recognition/GoogleCloudSTT.js +18 -18
  115. package/dist/recognition/OpenAICloudSTT.d.ts +19 -0
  116. package/dist/recognition/OpenAICloudSTT.js +81 -0
  117. package/dist/recognition/OpenAICloudSTT.js.map +1 -0
  118. package/dist/recognition/SileroSTT.d.ts +2 -2
  119. package/dist/recognition/SileroSTT.js +25 -25
  120. package/dist/recognition/VoskSTT.d.ts +2 -2
  121. package/dist/recognition/VoskSTT.js +8 -8
  122. package/dist/recognition/WhisperCppSTT.d.ts +88 -0
  123. package/dist/recognition/WhisperCppSTT.js +332 -0
  124. package/dist/recognition/WhisperCppSTT.js.map +1 -0
  125. package/dist/recognition/WhisperSTT.d.ts +49 -25
  126. package/dist/recognition/WhisperSTT.js +626 -481
  127. package/dist/recognition/WhisperSTT.js.map +1 -1
  128. package/dist/server/Client.d.ts +1 -1
  129. package/dist/server/Client.js +22 -22
  130. package/dist/server/Server.js +9 -9
  131. package/dist/server/Server.js.map +1 -1
  132. package/dist/server/Worker.d.ts +22 -22
  133. package/dist/server/Worker.js +36 -36
  134. package/dist/server/Worker.js.map +1 -1
  135. package/dist/server/WorkerStarter.js +2 -2
  136. package/dist/source-separation/MDXNetSourceSeparation.d.ts +11 -0
  137. package/dist/source-separation/MDXNetSourceSeparation.js +161 -0
  138. package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -0
  139. package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -1
  140. package/dist/speech-language-detection/SileroLanguageDetection.js +7 -7
  141. package/dist/subtitles/Subtitles.d.ts +10 -0
  142. package/dist/subtitles/Subtitles.js +2 -2
  143. package/dist/subtitles/Subtitles.js.map +1 -1
  144. package/dist/synthesis/AwsPollyTTS.d.ts +1 -1
  145. package/dist/synthesis/AwsPollyTTS.js +12 -12
  146. package/dist/synthesis/AzureCognitiveServicesTTS.js +7 -7
  147. package/dist/synthesis/CoquiServerTTS.js +10 -10
  148. package/dist/synthesis/CoquiServerTTS.js.map +1 -1
  149. package/dist/synthesis/ElevenlabsTTS.d.ts +23 -0
  150. package/dist/synthesis/ElevenlabsTTS.js +103 -0
  151. package/dist/synthesis/ElevenlabsTTS.js.map +1 -0
  152. package/dist/synthesis/EspeakTTS.d.ts +6 -5
  153. package/dist/synthesis/EspeakTTS.js +82 -68
  154. package/dist/synthesis/EspeakTTS.js.map +1 -1
  155. package/dist/synthesis/FliteTTS.d.ts +3 -3
  156. package/dist/synthesis/FliteTTS.js +154 -154
  157. package/dist/synthesis/FliteTTS.js.map +1 -1
  158. package/dist/synthesis/GoogleCloudTTS.d.ts +3 -3
  159. package/dist/synthesis/GoogleCloudTTS.js +17 -17
  160. package/dist/synthesis/GoogleCloudTTS.js.map +1 -1
  161. package/dist/synthesis/GoogleTranslateTTS.d.ts +1 -1
  162. package/dist/synthesis/GoogleTranslateTTS.js +103 -103
  163. package/dist/synthesis/MicrosoftEdgeTTS.d.ts +2 -2
  164. package/dist/synthesis/MicrosoftEdgeTTS.js +74 -74
  165. package/dist/synthesis/OpenAICloudTTS.d.ts +13 -0
  166. package/dist/synthesis/OpenAICloudTTS.js +169 -0
  167. package/dist/synthesis/OpenAICloudTTS.js.map +1 -0
  168. package/dist/synthesis/SamTTS.js +3 -3
  169. package/dist/synthesis/SapiTTS.d.ts +3 -3
  170. package/dist/synthesis/SapiTTS.js +26 -26
  171. package/dist/synthesis/StreamlabsPollyTTS.d.ts +2 -2
  172. package/dist/synthesis/StreamlabsPollyTTS.js +27 -27
  173. package/dist/synthesis/SvoxPicoTTS.d.ts +2 -2
  174. package/dist/synthesis/SvoxPicoTTS.js +65 -65
  175. package/dist/synthesis/SvoxPicoTTS.js.map +1 -1
  176. package/dist/synthesis/VitsTTS.d.ts +3 -3
  177. package/dist/synthesis/VitsTTS.js +378 -378
  178. package/dist/synthesis/VitsTTS.js.map +1 -1
  179. package/dist/tests/Test.js +2 -2
  180. package/dist/utilities/Compression.d.ts +5 -0
  181. package/dist/utilities/Compression.js +29 -13
  182. package/dist/utilities/Compression.js.map +1 -1
  183. package/dist/utilities/FileDownloader.d.ts +1 -1
  184. package/dist/utilities/FileDownloader.js +16 -16
  185. package/dist/utilities/FileSystem.js +7 -7
  186. package/dist/utilities/Locale.d.ts +7 -7
  187. package/dist/utilities/Locale.js +15 -15
  188. package/dist/utilities/Logger.js +3 -3
  189. package/dist/utilities/ObjectUtilities.js +19 -19
  190. package/dist/utilities/OpenPromise.js +2 -2
  191. package/dist/utilities/OpenPromise.js.map +1 -1
  192. package/dist/utilities/PackageManager.js +31 -0
  193. package/dist/utilities/PackageManager.js.map +1 -1
  194. package/dist/utilities/PathUtilities.js +8 -8
  195. package/dist/utilities/RandomGenerator.js +2 -2
  196. package/dist/utilities/SmoothEstimator.d.ts +8 -0
  197. package/dist/utilities/SmoothEstimator.js +25 -0
  198. package/dist/utilities/SmoothEstimator.js.map +1 -0
  199. package/dist/utilities/TarballMaker.js +8 -8
  200. package/dist/utilities/Timeline.d.ts +3 -2
  201. package/dist/utilities/Timeline.js +11 -11
  202. package/dist/utilities/Timeline.js.map +1 -1
  203. package/dist/utilities/Timer.js +4 -4
  204. package/dist/utilities/Utilities.d.ts +4 -0
  205. package/dist/utilities/Utilities.js +38 -15
  206. package/dist/utilities/Utilities.js.map +1 -1
  207. package/dist/utilities/WasmMemoryManager.js +7 -7
  208. package/dist/utilities/WebReader.js +23 -23
  209. package/dist/utilities/WikipediaReader.js +2 -2
  210. package/dist/voice-activity-detection/AdaptiveGateVAD.d.ts +28 -0
  211. package/dist/voice-activity-detection/AdaptiveGateVAD.js +138 -0
  212. package/dist/voice-activity-detection/AdaptiveGateVAD.js.map +1 -0
  213. package/dist/voice-activity-detection/SileroVAD.d.ts +1 -1
  214. package/dist/voice-activity-detection/SileroVAD.js +5 -5
  215. package/dist/voice-activity-detection/SileroVAD.js.map +1 -1
  216. package/dist/voice-activity-detection/WebRtcVAD.d.ts +1 -1
  217. package/dist/voice-activity-detection/WebRtcVAD.js +4 -4
  218. package/docs/API.md +29 -11
  219. package/docs/CLI.md +31 -7
  220. package/docs/Contributing.md +38 -0
  221. package/docs/Development.md +93 -19
  222. package/docs/Engines.md +28 -16
  223. package/docs/Licenses.md +4 -1
  224. package/docs/Options.md +158 -78
  225. package/docs/Releases.md +262 -0
  226. package/docs/Server.md +7 -7
  227. package/docs/Tasklist.md +95 -76
  228. package/docs/Technical.md +4 -4
  229. package/package.json +13 -14
  230. package/src/alignment/DTWMfccSequenceAlignment.ts +9 -9
  231. package/src/alignment/DTWSequenceAlignment.ts +2 -2
  232. package/src/alignment/DTWSequenceAlignmentWindowed.ts +3 -3
  233. package/src/alignment/LevenshteinSequenceAlignment.ts +2 -2
  234. package/src/alignment/SpeechAlignment.ts +204 -119
  235. package/src/api/API.ts +14 -13
  236. package/src/api/APIOptions.ts +12 -11
  237. package/src/api/Alignment.ts +147 -90
  238. package/src/api/Common.ts +1 -1
  239. package/src/api/Denoising.ts +28 -28
  240. package/src/api/LanguageDetection.ts +135 -48
  241. package/src/api/Recognition.ts +198 -59
  242. package/src/api/SourceSeparation.ts +99 -0
  243. package/src/api/Synthesis.ts +217 -181
  244. package/src/api/Translation.ts +193 -40
  245. package/src/api/Vad.ts +110 -41
  246. package/src/audio/AudioBufferConversion.ts +4 -4
  247. package/src/audio/AudioPlayer.ts +27 -27
  248. package/src/audio/AudioRecorder.ts +5 -5
  249. package/src/audio/AudioUtilities.ts +107 -24
  250. package/src/cli/CLI.ts +313 -164
  251. package/src/cli/CLIConfigFile.ts +8 -8
  252. package/src/cli/CLILauncher.ts +6 -6
  253. package/src/cli/CLIOptionsSchema.ts +2 -2
  254. package/src/cli/CLIParser.ts +5 -5
  255. package/src/cli/CLIStarter.ts +4 -4
  256. package/src/codecs/FFMpegTranscoder.ts +38 -38
  257. package/src/codecs/TIMITCodec.ts +5 -5
  258. package/src/codecs/WaveCodec.ts +22 -22
  259. package/src/denoising/RNNoise.ts +9 -9
  260. package/src/dsp/BiquadFilter.ts +19 -11
  261. package/src/dsp/DecayingPeakEstimator.ts +35 -0
  262. package/src/dsp/FFT.ts +103 -35
  263. package/src/dsp/KWeightingFilter.ts +43 -0
  264. package/src/dsp/LoudnessEstimator.ts +74 -0
  265. package/src/dsp/MFCC.ts +15 -15
  266. package/src/dsp/MelSpectogram.ts +7 -7
  267. package/src/dsp/Rubberband.ts +38 -38
  268. package/src/dsp/Sonic.ts +4 -4
  269. package/src/dsp/SpeexResampler.ts +2 -2
  270. package/src/math/VectorMath.ts +42 -33
  271. package/src/nlp/ChineseSegmentation.ts +3 -3
  272. package/src/nlp/CompromiseNLP.ts +3 -3
  273. package/src/nlp/EspeakPhonemizer.ts +30 -30
  274. package/src/nlp/IPA.ts +20 -20
  275. package/src/nlp/JapaneseSegmentation.ts +6 -6
  276. package/src/nlp/Lexicon.ts +8 -8
  277. package/src/nlp/Segmentation.ts +23 -14
  278. package/src/nlp/TextNormalizer.ts +16 -16
  279. package/src/recognition/AmazonTranscribeSTT.ts +16 -17
  280. package/src/recognition/AzureCognitiveServicesSTT.ts +8 -6
  281. package/src/recognition/GoogleCloudSTT.ts +21 -21
  282. package/src/recognition/OpenAICloudSTT.ts +142 -0
  283. package/src/recognition/SileroSTT.ts +26 -26
  284. package/src/recognition/VoskSTT.ts +10 -10
  285. package/src/recognition/WhisperCppSTT.ts +555 -0
  286. package/src/recognition/WhisperSTT.ts +760 -507
  287. package/src/server/Client.ts +23 -23
  288. package/src/server/Server.ts +9 -9
  289. package/src/server/Worker.ts +53 -53
  290. package/src/server/WorkerStarter.ts +2 -2
  291. package/src/source-separation/MDXNetSourceSeparation.ts +228 -0
  292. package/src/speech-language-detection/SileroLanguageDetection.ts +8 -8
  293. package/src/subtitles/Subtitles.ts +3 -3
  294. package/src/synthesis/AwsPollyTTS.ts +14 -14
  295. package/src/synthesis/AzureCognitiveServicesTTS.ts +10 -10
  296. package/src/synthesis/CoquiServerTTS.ts +10 -10
  297. package/src/synthesis/ElevenlabsTTS.ts +137 -0
  298. package/src/synthesis/EspeakTTS.ts +92 -70
  299. package/src/synthesis/FliteTTS.ts +157 -157
  300. package/src/synthesis/GoogleCloudTTS.ts +19 -19
  301. package/src/synthesis/GoogleTranslateTTS.ts +104 -104
  302. package/src/synthesis/MicrosoftEdgeTTS.ts +80 -80
  303. package/src/synthesis/OpenAICloudTTS.ts +196 -0
  304. package/src/synthesis/SamTTS.ts +3 -3
  305. package/src/synthesis/SapiTTS.ts +29 -29
  306. package/src/synthesis/StreamlabsPollyTTS.ts +29 -29
  307. package/src/synthesis/SvoxPicoTTS.ts +67 -67
  308. package/src/synthesis/VitsTTS.ts +380 -380
  309. package/src/tests/Test.ts +4 -4
  310. package/src/utilities/Compression.ts +34 -13
  311. package/src/utilities/FileDownloader.ts +19 -19
  312. package/src/utilities/FileSystem.ts +7 -7
  313. package/src/utilities/Locale.ts +22 -22
  314. package/src/utilities/Logger.ts +4 -4
  315. package/src/utilities/ObjectUtilities.ts +19 -19
  316. package/src/utilities/OpenPromise.ts +2 -2
  317. package/src/utilities/PackageManager.ts +40 -0
  318. package/src/utilities/PathUtilities.ts +8 -8
  319. package/src/utilities/RandomGenerator.ts +3 -3
  320. package/src/utilities/SmoothEstimator.ts +35 -0
  321. package/src/utilities/TarballMaker.ts +9 -9
  322. package/src/utilities/Timeline.ts +15 -13
  323. package/src/utilities/Timer.ts +4 -4
  324. package/src/utilities/Utilities.ts +49 -15
  325. package/src/utilities/WasmMemoryManager.ts +7 -7
  326. package/src/utilities/WebReader.ts +23 -23
  327. package/src/utilities/WikipediaReader.ts +2 -2
  328. package/src/voice-activity-detection/AdaptiveGateVAD.ts +202 -0
  329. package/src/voice-activity-detection/SileroVAD.ts +5 -5
  330. package/src/voice-activity-detection/WebRtcVAD.ts +5 -5
  331. package/dist/synthesis/ElevenLabsTTS.d.ts +0 -8
  332. package/dist/synthesis/ElevenLabsTTS.js +0 -82
  333. package/dist/synthesis/ElevenLabsTTS.js.map +0 -1
  334. package/src/synthesis/ElevenLabsTTS.ts +0 -104
@@ -1,15 +1,15 @@
1
- import { concatFloat32Arrays, logToStderr, objToString, simplifyPunctuationCharacters } from "../utilities/Utilities.js"
2
- import { int16PcmToFloat32 } from "../audio/AudioBufferConversion.js"
1
+ import { concatFloat32Arrays, logToStderr, objToString, simplifyPunctuationCharacters } from '../utilities/Utilities.js'
2
+ import { int16PcmToFloat32 } from '../audio/AudioBufferConversion.js'
3
3
  import { Logger } from '../utilities/Logger.js'
4
- import { WasmMemoryManager } from "../utilities/WasmMemoryManager.js"
5
- import { RawAudio, getEmptyRawAudio } from "../audio/AudioUtilities.js"
6
- import { playAudioWithTimelinePhones } from "../audio/AudioPlayer.js"
7
- import { getNormalizedFragmentsForSpeech } from "../nlp/TextNormalizer.js"
8
- import { ipaPhoneToKirshenbaum } from "../nlp/PhoneConversion.js"
9
- import { splitToWords, wordCharacterPattern } from "../nlp/Segmentation.js"
10
- import { Lexicon, tryGetFirstLexiconSubstitution } from "../nlp/Lexicon.js"
11
- import { phonemizeSentence } from "../nlp/EspeakPhonemizer.js"
12
- import { Timeline, TimelineEntry } from "../utilities/Timeline.js"
4
+ import { WasmMemoryManager } from '../utilities/WasmMemoryManager.js'
5
+ import { RawAudio, getEmptyRawAudio } from '../audio/AudioUtilities.js'
6
+ import { playAudioWithTimelinePhones } from '../audio/AudioPlayer.js'
7
+ import { getNormalizedFragmentsForSpeech } from '../nlp/TextNormalizer.js'
8
+ import { ipaPhoneToKirshenbaum } from '../nlp/PhoneConversion.js'
9
+ import { splitToWords, wordCharacterPattern } from '../nlp/Segmentation.js'
10
+ import { Lexicon, tryGetFirstLexiconSubstitution } from '../nlp/Lexicon.js'
11
+ import { phonemizeSentence } from '../nlp/EspeakPhonemizer.js'
12
+ import { Timeline, TimelineEntry } from '../utilities/Timeline.js'
13
13
 
14
14
  const log = logToStderr
15
15
 
@@ -19,12 +19,12 @@ let espeakModule: any
19
19
  export async function preprocessAndSynthesize(text: string, language: string, espeakOptions: EspeakOptions, lexicons: Lexicon[] = []) {
20
20
  const logger = new Logger()
21
21
 
22
- await logger.startAsync("Tokenize and analyze text")
22
+ await logger.startAsync('Tokenize and analyze text')
23
23
 
24
24
  let lowerCaseLanguageCode = language.toLowerCase()
25
25
 
26
- if (lowerCaseLanguageCode == "en-gb") {
27
- lowerCaseLanguageCode = "en-gb-x-rp"
26
+ if (lowerCaseLanguageCode == 'en-gb') {
27
+ lowerCaseLanguageCode = 'en-gb-x-rp'
28
28
  }
29
29
 
30
30
  let fragments: string[]
@@ -53,7 +53,7 @@ export async function preprocessAndSynthesize(text: string, language: string, es
53
53
  words = wordsWithMerges
54
54
 
55
55
  // Remove words containing only whitespace
56
- words = words.filter(word => word.trim() != "")
56
+ words = words.filter(word => word.trim() != '')
57
57
 
58
58
  const { normalizedFragments, referenceFragments } = getNormalizedFragmentsForSpeech(words, language)
59
59
 
@@ -69,12 +69,12 @@ export async function preprocessAndSynthesize(text: string, language: string, es
69
69
  }
70
70
 
71
71
  phonemizedFragmentsSubstitutions.set(fragmentIndex, substitutionPhonemes)
72
- const referenceIPA = (await textToPhonemes(fragment, espeakOptions.voice, true)).replaceAll("_", " ")
73
- const referenceKirshenbaum = (await textToPhonemes(fragment, espeakOptions.voice, false)).replaceAll("_", "")
72
+ const referenceIPA = (await textToPhonemes(fragment, espeakOptions.voice, true)).replaceAll('_', ' ')
73
+ const referenceKirshenbaum = (await textToPhonemes(fragment, espeakOptions.voice, false)).replaceAll('_', '')
74
74
 
75
- const kirshenbaumPhonemes = substitutionPhonemes.map(phone => ipaPhoneToKirshenbaum(phone)).join("")
75
+ const kirshenbaumPhonemes = substitutionPhonemes.map(phone => ipaPhoneToKirshenbaum(phone)).join('')
76
76
 
77
- logger.logTitledMessage(`\nLexicon substitution for '${fragment}'`, `IPA: ${substitutionPhonemes.join(" ")} (original: ${referenceIPA}), Kirshenbaum: ${kirshenbaumPhonemes} (reference: ${referenceKirshenbaum})`)
77
+ logger.logTitledMessage(`\nLexicon substitution for '${fragment}'`, `IPA: ${substitutionPhonemes.join(' ')} (original: ${referenceIPA}), Kirshenbaum: ${kirshenbaumPhonemes} (reference: ${referenceKirshenbaum})`)
78
78
 
79
79
  const substitutionPhonemesFragment = ` [[${kirshenbaumPhonemes}]] `
80
80
 
@@ -84,11 +84,11 @@ export async function preprocessAndSynthesize(text: string, language: string, es
84
84
  fragments = referenceFragments
85
85
  preprocessedFragments = normalizedFragments
86
86
 
87
- logger.start("Synthesize preprocessed fragments with eSpeak")
87
+ logger.start('Synthesize preprocessed fragments with eSpeak')
88
88
 
89
89
  const { rawAudio: referenceSynthesizedAudio, timeline: referenceTimeline } = await synthesizeFragments(preprocessedFragments, espeakOptions)
90
90
 
91
- await logger.startAsync("Build phonemized tokens")
91
+ await logger.startAsync('Build phonemized tokens')
92
92
 
93
93
  const phonemizedSentence: string[][][] = []
94
94
 
@@ -125,19 +125,19 @@ export async function preprocessAndSynthesize(text: string, language: string, es
125
125
  }
126
126
  }
127
127
 
128
- logger.log(phonemizedSentence.map(phrase => phrase.map(word => word.join(" ")).join(" | ")).join(" || "))
128
+ logger.log(phonemizedSentence.map(phrase => phrase.map(word => word.join(' ')).join(' | ')).join(' || '))
129
129
 
130
130
  logger.end()
131
131
 
132
132
  return { referenceSynthesizedAudio, referenceTimeline, fragments, preprocessedFragments, phonemizedFragmentsSubstitutions, phonemizedSentence }
133
133
  }
134
134
 
135
- export async function synthesizeFragments(fragments: string[], espeakOptions: EspeakOptions, insertSeparators = false) {
135
+ export async function synthesizeFragments(fragments: string[], espeakOptions: EspeakOptions) {
136
136
  const logger = new Logger()
137
137
 
138
138
  const sampleRate = await getSampleRate()
139
139
 
140
- //fragments = fragments.filter(fragment => fragment.trim() != "")
140
+ //fragments = fragments.filter(fragment => fragment.trim() != '')
141
141
 
142
142
  if (fragments.length == 0) {
143
143
  return {
@@ -147,7 +147,7 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
147
147
  }
148
148
  }
149
149
 
150
- let textWithMarkers = '() | '
150
+ let textWithMarkers = '() '
151
151
 
152
152
  for (let i = 0; i < fragments.length; i++) {
153
153
  let fragment = fragments[i]
@@ -155,14 +155,16 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
155
155
  fragment = simplifyPunctuationCharacters(fragment)
156
156
 
157
157
  fragment = fragment
158
- .replaceAll("<", "&lt;")
159
- .replaceAll(">", "&gt;")
158
+ .replaceAll('<', '&lt;')
159
+ .replaceAll('>', '&gt;')
160
160
 
161
- if (insertSeparators) {
162
- textWithMarkers += `<mark name="s-${i}"/> | ${fragment} | <mark name="e-${i}"/>`
161
+ if (espeakOptions.insertSeparators) {
162
+ const separator = ` | `
163
+
164
+ textWithMarkers += `<mark name="s-${i}"/>${separator}${fragment}${separator}<mark name="e-${i}"/>`
163
165
  } else {
164
- if (fragment.endsWith(".")) {
165
- fragment += " ()"
166
+ if (fragment.endsWith('.')) {
167
+ fragment += ' ()'
166
168
  }
167
169
 
168
170
  textWithMarkers += `<mark name="s-${i}"/>${fragment}<mark name="e-${i}"/> `
@@ -175,13 +177,13 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
175
177
 
176
178
  // Build word timeline from events
177
179
  const wordTimeline: Timeline = fragments.map(word => ({
178
- type: "word",
180
+ type: 'word',
179
181
  text: word,
180
182
  startTime: -1,
181
183
  endTime: -1,
182
184
  timeline: [{
183
- type: "token",
184
- text: "",
185
+ type: 'token',
186
+ text: '',
185
187
  startTime: -1,
186
188
  endTime: -1,
187
189
  timeline: []
@@ -207,7 +209,7 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
207
209
  lastPhoneEntry.endTime = eventTime
208
210
  }
209
211
 
210
- if (event.type == "word") {
212
+ if (event.type == 'word') {
211
213
  if (!event.id || currentPhoneTimeline.length == 0) {
212
214
  continue
213
215
  }
@@ -217,21 +219,21 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
217
219
  }
218
220
 
219
221
  currentTokenTimeline.push({
220
- type: "token",
221
- text: "",
222
+ type: 'token',
223
+ text: '',
222
224
  startTime: eventTime,
223
225
  endTime: -1,
224
226
  timeline: []
225
227
  })
226
- } else if (event.type == "phoneme") {
228
+ } else if (event.type == 'phoneme') {
227
229
  const phoneText = event.id as string
228
230
 
229
- if (!phoneText || phoneText.startsWith("(")) {
231
+ if (!phoneText || phoneText.startsWith('(')) {
230
232
  continue
231
233
  }
232
234
 
233
235
  currentPhoneTimeline.push({
234
- type: "phone",
236
+ type: 'phone',
235
237
  text: phoneText,
236
238
  startTime: eventTime,
237
239
  endTime: -1
@@ -239,10 +241,10 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
239
241
 
240
242
  currentTokenEntry.text += phoneText
241
243
  currentTokenEntry.startTime = currentPhoneTimeline[0].startTime
242
- } else if (event.type == "mark") {
244
+ } else if (event.type == 'mark') {
243
245
  const markerName = event.id! as string
244
246
 
245
- if (markerName.startsWith("s-")) {
247
+ if (markerName.startsWith('s-')) {
246
248
  const markerIndex = parseInt(markerName.substring(2))
247
249
 
248
250
  if (markerIndex != wordIndex) {
@@ -255,7 +257,7 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
255
257
 
256
258
  currentWordEntry.startTime = eventTime
257
259
  currentTokenEntry.startTime = eventTime
258
- } else if (markerName.startsWith("e-")) {
260
+ } else if (markerName.startsWith('e-')) {
259
261
  const markerIndex = parseInt(markerName.substring(2))
260
262
 
261
263
  if (markerIndex != wordIndex) {
@@ -275,7 +277,7 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
275
277
  } else {
276
278
  continue
277
279
  }
278
- } else if (event.type == "end") {
280
+ } else if (event.type == 'end') {
279
281
  clauseEndIndexes.push(wordIndex)
280
282
  }
281
283
  }
@@ -291,22 +293,22 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
291
293
  }
292
294
 
293
295
  if (!tokenTimeline || tokenTimeline.length == 0) {
294
- throw new Error("Unexpected: token timeline should exist and have at least one token")
296
+ throw new Error('Unexpected: token timeline should exist and have at least one token')
295
297
  }
296
298
 
297
- if (tokenTimeline[0].text != '') {
299
+ if (tokenTimeline.length !== 1 && tokenTimeline[0].text != '') {
298
300
  continue
299
301
  }
300
302
 
301
- const wordReferencePhonemes = (await textToPhonemes(wordEntry.text, espeakOptions.voice, true)).split("_")
303
+ const wordReferencePhonemes = (await textToPhonemes(wordEntry.text, espeakOptions.voice, true)).split('_')
302
304
 
303
- const wordReferenceIPA = wordReferencePhonemes.join(" ")
305
+ const wordReferenceIPA = wordReferencePhonemes.join(' ')
304
306
 
305
307
  if (wordReferenceIPA.trim().length == 0) {
306
308
  continue
307
309
  }
308
310
 
309
- const wordReferenceIPAWithoutStress = wordReferenceIPA.replaceAll("ˈ", "").replaceAll("ˌ", "")
311
+ const wordReferenceIPAWithoutStress = wordReferenceIPA.replaceAll('ˈ', '').replaceAll('ˌ', '')
310
312
 
311
313
  const previousWordEntry = wordTimeline[index - 1]
312
314
 
@@ -314,13 +316,31 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
314
316
  continue
315
317
  }
316
318
 
317
- const previousWordTokenEntry = previousWordEntry.timeline[0]
319
+ const previousWordTokenEntry = previousWordEntry.timeline[previousWordEntry.timeline.length - 1]
318
320
 
319
- if (!previousWordTokenEntry.timeline || previousWordTokenEntry.timeline.length <= wordReferencePhonemes.length) {
321
+ if (!previousWordTokenEntry.timeline) {
320
322
  continue
321
323
  }
322
324
 
323
- const previousWordTokenIPAWithoutStress = previousWordTokenEntry.timeline.map(phoneEntry => phoneEntry.text.replaceAll("ˈ", "").replaceAll("ˌ", "")).join(" ")
325
+ const previousWordTokenIPAWithoutStress = previousWordTokenEntry.timeline.map(phoneEntry => phoneEntry.text.replaceAll('ˈ', '').replaceAll('ˌ', '')).join(' ')
326
+
327
+ if (previousWordEntry.timeline.length > 1 && previousWordTokenIPAWithoutStress === wordReferenceIPAWithoutStress) {
328
+ tokenTimeline.pop()
329
+
330
+ const tokenEntryToInsert = previousWordEntry.timeline.pop()!
331
+ tokenTimeline.push(tokenEntryToInsert)
332
+
333
+ previousWordEntry.endTime = previousWordEntry.timeline[previousWordEntry.timeline.length - 1].endTime
334
+
335
+ wordEntry.startTime = tokenEntryToInsert.startTime
336
+ wordEntry.endTime = tokenEntryToInsert.endTime
337
+
338
+ continue
339
+ }
340
+
341
+ if (previousWordTokenEntry.timeline.length <= wordReferencePhonemes.length) {
342
+ continue
343
+ }
324
344
 
325
345
  if (!previousWordTokenIPAWithoutStress.endsWith(wordReferenceIPAWithoutStress)) {
326
346
  continue
@@ -329,14 +349,14 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
329
349
  const tokenEntry = tokenTimeline[0]
330
350
 
331
351
  tokenEntry.timeline = previousWordTokenEntry.timeline.splice(previousWordTokenEntry.timeline.length - wordReferencePhonemes.length)
332
- tokenEntry.text = tokenEntry.timeline.map(phoneEntry => phoneEntry.text).join("")
352
+ tokenEntry.text = tokenEntry.timeline.map(phoneEntry => phoneEntry.text).join('')
333
353
 
334
354
  tokenEntry.startTime = tokenEntry.timeline[0].startTime
335
355
  tokenEntry.endTime = tokenEntry.timeline[tokenEntry.timeline.length - 1].endTime
336
356
  wordEntry.startTime = tokenEntry.startTime
337
357
  wordEntry.endTime = tokenEntry.endTime
338
358
 
339
- previousWordTokenEntry.text = previousWordTokenEntry.timeline.map(phoneEntry => phoneEntry.text).join("")
359
+ previousWordTokenEntry.text = previousWordTokenEntry.timeline.map(phoneEntry => phoneEntry.text).join('')
340
360
  previousWordTokenEntry.endTime = previousWordTokenEntry.timeline[previousWordTokenEntry.timeline.length - 1].endTime
341
361
  previousWordEntry.endTime = previousWordTokenEntry.endTime
342
362
  }
@@ -348,8 +368,8 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
348
368
 
349
369
  for (const clauseEndIndex of clauseEndIndexes) {
350
370
  const newClause: TimelineEntry = {
351
- type: "clause",
352
- text: "",
371
+ type: 'clause',
372
+ text: '',
353
373
  startTime: -1,
354
374
  endTime: -1,
355
375
  timeline: []
@@ -379,7 +399,7 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
379
399
 
380
400
  export async function synthesize(text: string, espeakOptions: EspeakOptions) {
381
401
  const logger = new Logger()
382
- logger.start("Get eSpeak Emscripten instance")
402
+ logger.start('Get eSpeak Emscripten instance')
383
403
 
384
404
  if (!espeakOptions.ssml) {
385
405
  const { escape } = await import('html-escaper')
@@ -392,7 +412,7 @@ export async function synthesize(text: string, espeakOptions: EspeakOptions) {
392
412
  const sampleChunks: Float32Array[] = []
393
413
  const allEvents: EspeakEvent[] = []
394
414
 
395
- logger.start("Synthesize with eSpeak")
415
+ logger.start('Synthesize with eSpeak')
396
416
 
397
417
  if (espeakOptions.useKlatt) {
398
418
  await setVoice(`${espeakOptions.voice}+klatt6`)
@@ -410,9 +430,9 @@ export async function synthesize(text: string, espeakOptions: EspeakOptions) {
410
430
  }
411
431
 
412
432
  for (const event of events) {
413
- if (event.type == "word") {
433
+ if (event.type == 'word') {
414
434
  const textPosition = event.text_position - 1;
415
- (event as any)["text"] = text.substring(textPosition, textPosition + event.word_length)
435
+ (event as any)['text'] = text.substring(textPosition, textPosition + event.word_length)
416
436
  }
417
437
  }
418
438
 
@@ -512,7 +532,7 @@ async function getEspeakInstance() {
512
532
  return { instance: espeakInstance, module: espeakModule }
513
533
  }
514
534
 
515
- export type EspeakEventType = "sentence" | "word" | "phoneme" | "end" | "mark" | "play" | "msg_terminated" | "list_terminated" | "samplerate"
535
+ export type EspeakEventType = 'sentence' | 'word' | 'phoneme' | 'end' | 'mark' | 'play' | 'msg_terminated' | 'list_terminated' | 'samplerate'
516
536
 
517
537
  export interface EspeakEvent {
518
538
  audio_position: number
@@ -528,6 +548,7 @@ export interface EspeakOptions {
528
548
  pitch: number
529
549
  pitchRange: number
530
550
  useKlatt: boolean
551
+ insertSeparators: boolean
531
552
  }
532
553
 
533
554
  export const defaultEspeakOptions: EspeakOptions = {
@@ -536,17 +557,18 @@ export const defaultEspeakOptions: EspeakOptions = {
536
557
  rate: 1.0,
537
558
  pitch: 1.0,
538
559
  pitchRange: 1.0,
539
- useKlatt: false
560
+ useKlatt: false,
561
+ insertSeparators: false
540
562
  }
541
563
 
542
564
  export async function testEspeakSynthesisWithPrePhonemizedInputs(text: string) {
543
- const ipaPhonemizedSentence = (await phonemizeSentence(text, "en-us")).flatMap(clause => clause)
544
- const kirshenbaumPhonemizedSentence = (await phonemizeSentence(text, "en-us", undefined, false)).flatMap(clause => clause)
565
+ const ipaPhonemizedSentence = (await phonemizeSentence(text, 'en-us')).flatMap(clause => clause)
566
+ const kirshenbaumPhonemizedSentence = (await phonemizeSentence(text, 'en-us', undefined, false)).flatMap(clause => clause)
545
567
  log(kirshenbaumPhonemizedSentence)
546
568
 
547
569
  const fragments = ipaPhonemizedSentence.map(word =>
548
570
  word.map(phoneme =>
549
- ipaPhoneToKirshenbaum(phoneme)).join("")).map(word => ` [[${word}]] `)
571
+ ipaPhoneToKirshenbaum(phoneme)).join('')).map(word => ` [[${word}]] `)
550
572
 
551
573
  const { rawAudio, timeline } = await synthesizeFragments(fragments, defaultEspeakOptions)
552
574
 
@@ -554,16 +576,16 @@ export async function testEspeakSynthesisWithPrePhonemizedInputs(text: string) {
554
576
  }
555
577
 
556
578
  export async function testKirshenbaumPhonemization(text: string) {
557
- const ipaPhonemizedSentence = (await phonemizeSentence(text, "en-us")).flatMap(clause => clause)
558
- const kirshenbaumPhonemizedSentence = (await phonemizeSentence(text, "en-us", undefined, false)).flatMap(clause => clause)
579
+ const ipaPhonemizedSentence = (await phonemizeSentence(text, 'en-us')).flatMap(clause => clause)
580
+ const kirshenbaumPhonemizedSentence = (await phonemizeSentence(text, 'en-us', undefined, false)).flatMap(clause => clause)
559
581
 
560
- const ipaFragments = ipaPhonemizedSentence.map(word => word.join(""))
582
+ const ipaFragments = ipaPhonemizedSentence.map(word => word.join(''))
561
583
 
562
- const kirshenbaumFragments = kirshenbaumPhonemizedSentence.map(word => word.join(""))
584
+ const kirshenbaumFragments = kirshenbaumPhonemizedSentence.map(word => word.join(''))
563
585
 
564
586
  const fragments = ipaPhonemizedSentence.map(word =>
565
587
  word.map(phoneme =>
566
- ipaPhoneToKirshenbaum(phoneme)).join(""))
588
+ ipaPhoneToKirshenbaum(phoneme)).join(''))
567
589
 
568
590
  for (let i = 0; i < fragments.length; i++) {
569
591
  log(`IPA: ${ipaFragments[i]} | converted: ${fragments[i]} | ground truth: ${kirshenbaumFragments[i]}`)