echogarden 1.4.4 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/data/schemas/options.json +310 -25
  2. package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
  3. package/dist/alignment/DTWMfccSequenceAlignment.js +5 -5
  4. package/dist/alignment/DTWSequenceAlignmentWindowed.js +1 -3
  5. package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
  6. package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +4 -2
  7. package/dist/alignment/SemanticTextAlignment.js +336 -0
  8. package/dist/alignment/SemanticTextAlignment.js.map +1 -0
  9. package/dist/alignment/SpeechAlignment.d.ts +4 -3
  10. package/dist/alignment/SpeechAlignment.js +130 -39
  11. package/dist/alignment/SpeechAlignment.js.map +1 -1
  12. package/dist/api/API.d.ts +7 -3
  13. package/dist/api/API.js +7 -2
  14. package/dist/api/API.js.map +1 -1
  15. package/dist/api/APIOptions.d.ts +4 -1
  16. package/dist/api/Alignment.d.ts +1 -1
  17. package/dist/api/Alignment.js +13 -5
  18. package/dist/api/Alignment.js.map +1 -1
  19. package/dist/api/LanguageDetectionCommon.d.ts +6 -0
  20. package/dist/api/LanguageDetectionCommon.js +2 -0
  21. package/dist/api/LanguageDetectionCommon.js.map +1 -0
  22. package/dist/api/Recognition.js.map +1 -1
  23. package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
  24. package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
  25. package/dist/api/SpeechLanguageDetection.js.map +1 -0
  26. package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
  27. package/dist/api/SpeechTranslation.js.map +1 -0
  28. package/dist/api/Synthesis.d.ts +0 -1
  29. package/dist/api/Synthesis.js +4 -4
  30. package/dist/api/TextLanguageDetection.d.ts +21 -0
  31. package/dist/api/TextLanguageDetection.js +67 -0
  32. package/dist/api/TextLanguageDetection.js.map +1 -0
  33. package/dist/api/TextTranslation.d.ts +25 -0
  34. package/dist/api/TextTranslation.js +101 -0
  35. package/dist/api/TextTranslation.js.map +1 -0
  36. package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
  37. package/dist/api/TimelineTranslationAlignment.js +92 -0
  38. package/dist/api/TimelineTranslationAlignment.js.map +1 -0
  39. package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
  40. package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
  41. package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
  42. package/dist/api/TranslationAlignment.d.ts +4 -3
  43. package/dist/api/TranslationAlignment.js +9 -8
  44. package/dist/api/TranslationAlignment.js.map +1 -1
  45. package/dist/api/VoiceActivityDetection.js +16 -1
  46. package/dist/api/VoiceActivityDetection.js.map +1 -1
  47. package/dist/audio/AudioBufferConversion.d.ts +0 -1
  48. package/dist/audio/AudioPlayer.d.ts +0 -1
  49. package/dist/audio/AudioPlayer.js +62 -41
  50. package/dist/audio/AudioPlayer.js.map +1 -1
  51. package/dist/audio/AudioUtilities.d.ts +0 -1
  52. package/dist/cli/CLI.d.ts +28 -7
  53. package/dist/cli/CLI.js +265 -37
  54. package/dist/cli/CLI.js.map +1 -1
  55. package/dist/codecs/FFMpegTranscoder.d.ts +0 -1
  56. package/dist/codecs/FFMpegTranscoder.js +7 -0
  57. package/dist/codecs/FFMpegTranscoder.js.map +1 -1
  58. package/dist/codecs/TIMITCodec.d.ts +0 -1
  59. package/dist/codecs/WaveCodec.d.ts +0 -1
  60. package/dist/dsp/FFT.d.ts +1 -1
  61. package/dist/dsp/FFT.js +6 -0
  62. package/dist/dsp/FFT.js.map +1 -1
  63. package/dist/dsp/KWeightingFilter.js +1 -1
  64. package/dist/dsp/KWeightingFilter.js.map +1 -1
  65. package/dist/dsp/MelSpectogram.d.ts +3 -2
  66. package/dist/dsp/MelSpectogram.js +14 -8
  67. package/dist/dsp/MelSpectogram.js.map +1 -1
  68. package/dist/math/VectorMath.d.ts +9 -9
  69. package/dist/math/VectorMath.js +10 -10
  70. package/dist/math/VectorMath.js.map +1 -1
  71. package/dist/nlp/ChineseSegmentation.js +4 -4
  72. package/dist/nlp/ChineseSegmentation.js.map +1 -1
  73. package/dist/nlp/Segmentation.d.ts +2 -2
  74. package/dist/nlp/Segmentation.js +20 -13
  75. package/dist/nlp/Segmentation.js.map +1 -1
  76. package/dist/recognition/OpenAICloudSTT.d.ts +2 -1
  77. package/dist/recognition/OpenAICloudSTT.js +30 -19
  78. package/dist/recognition/OpenAICloudSTT.js.map +1 -1
  79. package/dist/recognition/SileroSTT.d.ts +0 -1
  80. package/dist/recognition/WhisperCppSTT.d.ts +3 -3
  81. package/dist/recognition/WhisperCppSTT.js +21 -9
  82. package/dist/recognition/WhisperCppSTT.js.map +1 -1
  83. package/dist/recognition/WhisperSTT.d.ts +9 -6
  84. package/dist/recognition/WhisperSTT.js +227 -46
  85. package/dist/recognition/WhisperSTT.js.map +1 -1
  86. package/dist/server/Client.d.ts +3 -4
  87. package/dist/server/Client.js.map +1 -1
  88. package/dist/server/Worker.d.ts +3 -3
  89. package/dist/server/Worker.js +3 -2
  90. package/dist/server/Worker.js.map +1 -1
  91. package/dist/source-separation/MDXNetSourceSeparation.d.ts +0 -1
  92. package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
  93. package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
  94. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +12 -0
  95. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
  96. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
  97. package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -2
  98. package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
  99. package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
  100. package/dist/subtitles/Subtitles.js +2 -2
  101. package/dist/subtitles/Subtitles.js.map +1 -1
  102. package/dist/synthesis/GoogleCloudTTS.d.ts +0 -1
  103. package/dist/synthesis/GoogleTranslateTTS.d.ts +0 -1
  104. package/dist/synthesis/GoogleTranslateTTS.js +6 -21
  105. package/dist/synthesis/GoogleTranslateTTS.js.map +1 -1
  106. package/dist/synthesis/StreamlabsPollyTTS.d.ts +0 -1
  107. package/dist/synthesis/VitsTTS.d.ts +0 -1
  108. package/dist/synthesis/VitsTTS.js +30 -0
  109. package/dist/synthesis/VitsTTS.js.map +1 -1
  110. package/dist/tests/Test.js +0 -31
  111. package/dist/tests/Test.js.map +1 -1
  112. package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
  113. package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
  114. package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
  115. package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
  116. package/dist/text-translation/DeepLTextTranslation.d.ts +2 -0
  117. package/dist/text-translation/DeepLTextTranslation.js +67 -0
  118. package/dist/text-translation/DeepLTextTranslation.js.map +1 -0
  119. package/dist/text-translation/GoogleTranslateTextTranslation.d.ts +10 -0
  120. package/dist/text-translation/GoogleTranslateTextTranslation.js +554 -0
  121. package/dist/text-translation/GoogleTranslateTextTranslation.js.map +1 -0
  122. package/dist/text-translation/NLLBTextTranslation.d.ts +2 -1
  123. package/dist/text-translation/NLLBTextTranslation.js +249 -19
  124. package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
  125. package/dist/utilities/BinaryArrayConversion.d.ts +0 -1
  126. package/dist/utilities/BrowserRequestHeaders.d.ts +6 -0
  127. package/dist/utilities/BrowserRequestHeaders.js +52 -0
  128. package/dist/utilities/BrowserRequestHeaders.js.map +1 -0
  129. package/dist/utilities/BufferFileReadStream.d.ts +20 -0
  130. package/dist/utilities/BufferFileReadStream.js +81 -0
  131. package/dist/utilities/BufferFileReadStream.js.map +1 -0
  132. package/dist/utilities/DynamicUint8Array.d.ts +9 -0
  133. package/dist/utilities/DynamicUint8Array.js +31 -0
  134. package/dist/utilities/DynamicUint8Array.js.map +1 -0
  135. package/dist/utilities/FileSystem.d.ts +0 -2
  136. package/dist/utilities/Hashing.d.ts +3 -10
  137. package/dist/utilities/Hashing.js +10 -127
  138. package/dist/utilities/Hashing.js.map +1 -1
  139. package/dist/utilities/LEB128.d.ts +15 -5
  140. package/dist/utilities/LEB128.js +199 -119
  141. package/dist/utilities/LEB128.js.map +1 -1
  142. package/dist/utilities/LPVarInt.d.ts +11 -0
  143. package/dist/utilities/LPVarInt.js +187 -0
  144. package/dist/utilities/LPVarInt.js.map +1 -0
  145. package/dist/utilities/Locale.d.ts +1 -1
  146. package/dist/utilities/Locale.js +1 -1
  147. package/dist/utilities/OnnxUtilities.d.ts +1 -2
  148. package/dist/utilities/PVarInt.d.ts +4 -0
  149. package/dist/utilities/PVarInt.js +166 -0
  150. package/dist/utilities/PVarInt.js.map +1 -0
  151. package/dist/utilities/PackageManager.js +48 -25
  152. package/dist/utilities/PackageManager.js.map +1 -1
  153. package/dist/utilities/RandomGenerator.d.ts +3 -17
  154. package/dist/utilities/RandomGenerator.js +12 -81
  155. package/dist/utilities/RandomGenerator.js.map +1 -1
  156. package/dist/utilities/Timeline.d.ts +2 -0
  157. package/dist/utilities/Timeline.js +129 -20
  158. package/dist/utilities/Timeline.js.map +1 -1
  159. package/dist/utilities/Utilities.d.ts +1 -3
  160. package/dist/utilities/Utilities.js +30 -3
  161. package/dist/utilities/Utilities.js.map +1 -1
  162. package/dist/utilities/VarInt.d.ts +4 -0
  163. package/dist/utilities/VarInt.js +166 -0
  164. package/dist/utilities/VarInt.js.map +1 -0
  165. package/dist/utilities/VirtualFileReadStream.d.ts +20 -0
  166. package/dist/utilities/VirtualFileReadStream.js +79 -0
  167. package/dist/utilities/VirtualFileReadStream.js.map +1 -0
  168. package/dist/utilities/WebReader.js +7 -23
  169. package/dist/utilities/WebReader.js.map +1 -1
  170. package/dist/voice-activity-detection/SileroVAD.d.ts +0 -1
  171. package/docs/API.md +105 -3
  172. package/docs/CLI.md +51 -1
  173. package/docs/Engines.md +32 -3
  174. package/docs/Options.md +53 -12
  175. package/docs/Tasklist.md +1 -13
  176. package/package.json +20 -24
  177. package/src/alignment/DTWMfccSequenceAlignment.ts +5 -5
  178. package/src/alignment/DTWSequenceAlignmentWindowed.ts +1 -3
  179. package/src/alignment/SemanticTextAlignment.ts +467 -0
  180. package/src/alignment/SpeechAlignment.ts +214 -56
  181. package/src/api/API.ts +18 -2
  182. package/src/api/APIOptions.ts +14 -1
  183. package/src/api/Alignment.ts +31 -9
  184. package/src/api/LanguageDetectionCommon.ts +7 -0
  185. package/src/api/Recognition.ts +2 -0
  186. package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
  187. package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
  188. package/src/api/Synthesis.ts +4 -4
  189. package/src/api/TextLanguageDetection.ts +116 -0
  190. package/src/api/TextTranslation.ts +177 -0
  191. package/src/api/TimelineTranslationAlignment.ts +162 -0
  192. package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
  193. package/src/api/TranslationAlignment.ts +12 -10
  194. package/src/api/VoiceActivityDetection.ts +24 -3
  195. package/src/audio/AudioPlayer.ts +2 -0
  196. package/src/cli/CLI.ts +376 -40
  197. package/src/codecs/FFMpegTranscoder.ts +6 -0
  198. package/src/dsp/FFT.ts +8 -2
  199. package/src/dsp/KWeightingFilter.ts +1 -1
  200. package/src/dsp/MelSpectogram.ts +17 -8
  201. package/src/math/VectorMath.ts +15 -15
  202. package/src/nlp/ChineseSegmentation.ts +6 -4
  203. package/src/nlp/Segmentation.ts +18 -13
  204. package/src/recognition/OpenAICloudSTT.ts +47 -29
  205. package/src/recognition/WhisperCppSTT.ts +26 -11
  206. package/src/recognition/WhisperSTT.ts +364 -49
  207. package/src/server/Client.ts +3 -2
  208. package/src/server/Worker.ts +3 -2
  209. package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
  210. package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
  211. package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
  212. package/src/subtitles/Subtitles.ts +2 -2
  213. package/src/synthesis/GoogleTranslateTTS.ts +7 -21
  214. package/src/synthesis/VitsTTS.ts +31 -3
  215. package/src/tests/Test.ts +1 -38
  216. package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
  217. package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
  218. package/src/text-translation/DeepLTextTranslation.ts +88 -0
  219. package/src/text-translation/GoogleTranslateTextTranslation.ts +667 -0
  220. package/src/text-translation/NLLBTextTranslation.ts +261 -21
  221. package/src/typings/Fillers.d.ts +25 -2
  222. package/src/utilities/BrowserRequestHeaders.ts +59 -0
  223. package/src/utilities/DynamicUint8Array.ts +39 -0
  224. package/src/utilities/Hashing.ts +14 -167
  225. package/src/utilities/LEB128.ts +273 -148
  226. package/src/utilities/LPVarInt.ts +292 -0
  227. package/src/utilities/Locale.ts +1 -1
  228. package/src/utilities/OnnxUtilities.ts +1 -1
  229. package/src/utilities/PackageManager.ts +51 -30
  230. package/src/utilities/RandomGenerator.ts +12 -113
  231. package/src/utilities/Timeline.ts +162 -23
  232. package/src/utilities/Utilities.ts +40 -3
  233. package/src/utilities/VirtualFileReadStream.ts +109 -0
  234. package/src/utilities/WebReader.ts +9 -23
  235. package/dist/alignment/TextAlignment.js +0 -156
  236. package/dist/alignment/TextAlignment.js.map +0 -1
  237. package/dist/api/LanguageDetection.js.map +0 -1
  238. package/dist/api/Translation.js.map +0 -1
  239. package/src/alignment/TextAlignment.ts +0 -234
  240. /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
package/docs/Options.md CHANGED
@@ -141,24 +141,24 @@ Applies to CLI operation: `transcribe`, API method: `recognize`
141
141
  * `sourceSeparation`: prefix to provide options for source separation when `isolate` is set to `true`. Options detailed in section for source separation
142
142
 
143
143
  **Whisper**:
144
- * `whisper.model`: selects which Whisper model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en`, `large` (same as `large-v2`), `large-v1`, `large-v2`, `large-v3` (**Note**: large models aren't yet supported by `onnxruntime-node` due to their size). Defaults to `tiny` or `tiny.en`
144
+ * `whisper.model`: selects which Whisper model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en` or `large-v3-turbo`. Defaults to `tiny` or `tiny.en`
145
145
  * `whisper.temperature`: temperature setting for the text decoder. Impacts the amount of randomization for token selection. It is recommended to leave at `0.1` (close to no randomization - almost always chooses the top ranked token) or choose a relatively low value (`0.25` or lower) for best results. Defaults to `0.1`
146
146
  * `whisper.prompt`: initial text to give the Whisper model. Can be a vocabulary, or example text of some sort. Note that if the prompt is very similar to the transcript, the model may intentionally avoid producing the transcript tokens as it may assume that they have already been transcribed. Optional
147
147
  * `whisper.topCandidateCount`: the number of top candidate tokens to consider. Defaults to `5`
148
- * `whisper.punctuationThreshold`: the minimal probability for a punctuation token, included in the top candidates, to be chosen unconditionally. A lower threshold encourages the model to output more punctuation symbols. Defaults to `0.2`
149
- * `whisper.autoPromptParts`: use previous part's recognized text as the prompt for the next part. Disabling this may help to prevent repetition carrying over between parts, in some cases. Defaults to `true`
148
+ * `whisper.punctuationThreshold`: the minimal probability for a punctuation token, included in the top candidates, to be chosen unconditionally. A lower threshold encourages the model to output more punctuation characters. Defaults to `0.2`
149
+ * `whisper.autoPromptParts`: use previous part's recognized text as the prompt for the next part. Disabling this may help to prevent repetition carrying over between parts, in some cases. Defaults to `true` (**Note**: currently always disabled for `large-v3-turbo` model due to an apparent issue with corrupt output when prompted)
150
150
  * `whisper.maxTokensPerPart`: maximum number of tokens to decode for each audio part. Defaults to `250`
151
151
  * `whisper.suppressRepetition`: attempt to suppress decoding of repeating token patterns. Defaults to `true`
152
152
  * `whisper.repetitionThreshold`: minimal repetition / compressibility score to cause a part not to be auto-prompted to the next part. Defaults to `2.4`
153
153
  * `whisper.decodeTimestampTokens`: enable/disable decoding of timestamp tokens. Setting to `false` can reduce the occurrence of hallucinations and token repetition loops, possibly due to the overall reduction in the number of tokens decoded. This has no impact on the accuracy of timestamps, since they are derived independently using cross-attention weights. However, there are cases where this can cause the model to end a part prematurely, especially in singing and less speech-like voice segments, or when there are multiple speakers. Defaults to `true`
154
- * `whisper.encoderProvider`: identifier for the ONNX execution provider to use with the encoder model. Can be `cpu` or `dml` ([DirectML](https://microsoft.github.io/DirectML/)-based GPU acceleration - Windows only). In general, GPU-based encoding should be significantly faster. Defaults to `cpu`, or `dml` if available
155
- * `whisper.decoderProvider`: identifier for the ONNX execution provider to use with the decoder model. Can be `cpu` or `dml` (Windows only). Using GPU acceleration for the decoder may be faster than CPU, especially for larger models, but that depends on your particular combination of CPU and GPU. Defaults to `cpu`
154
+ * `whisper.encoderProvider`: identifier for the ONNX execution provider to use with the encoder model. Can be `cpu`, `dml` ([DirectML](https://microsoft.github.io/DirectML/)-based GPU acceleration - Windows only) or `cuda` (Linux only). In general, GPU-based encoding should be significantly faster. Defaults to `cpu`, or `dml` if available
155
+ * `whisper.decoderProvider`: identifier for the ONNX execution provider to use with the decoder model. Can be `cpu`, `dml` (Windows only) or `cuda` (Linux only). Using GPU acceleration for the decoder may be faster than CPU, especially for larger models, but that depends on your particular combination of CPU and GPU. Defaults to `cpu`, and on Windows, `dml` if available for larger models (`small`, `medium`, `large`)
156
156
  * `whisper.seed`: provide a custom random seed for token selection when temperature is greater than 0. Uses a constant seed by default to ensure reproducibility
157
157
 
158
158
  **Whisper.cpp**:
159
- * `whisperCpp.model`: selects which `whisper.cpp` model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en`, `large` (same as `large-v2`), `large-v1`, `large-v2`, `large-v3`. These quantized models are also supported: `tiny-q5_1`, `tiny.en-q5_1`, `tiny.en-q8_0`,`base-q5_1`, `base.en-q5_1`, `small-q5_1`, `small.en-q5_1`, `medium-q5_0`, `medium.en-q5_0`, `large-v2-q5_0`, `large-v3-q5_0`. Defaults to `base` or `base.en`
160
- * `whisperCpp.executablePath`: custom `whisper.cpp` executable path (currently required for macOS)
161
- * `whisperCpp.build`: type of `whisper.cpp` build to use. Can be set `cpu`, `cublas-11.8.0`, `cublas-12.4.0`. By default, builds are auto-selected and downloaded for Windows x64 (`cpu`, `cublas-11.8.0`, `cublas-12.4.0`) and Linux x64 (`cpu`). Using other builds requires providing a custom `executablePath`
159
+ * `whisperCpp.model`: selects which `whisper.cpp` model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en`, `large` (same as `large-v2`), `large-v1`, `large-v2`, `large-v3`. These quantized models are also supported: `tiny-q5_1`, `tiny.en-q5_1`, `tiny.en-q8_0`,`base-q5_1`, `base.en-q5_1`, `small-q5_1`, `small.en-q5_1`, `medium-q5_0`, `medium.en-q5_0`, `large-v2-q5_0`, `large-v3-q5_0`, `large-v3-turbo`, `large-v3-turbo-q5_0`. Defaults to `base` or `base.en`
160
+ * `whisperCpp.executablePath`: a path to a custom `whisper.cpp` `main` executable (currently required for macOS)
161
+ * `whisperCpp.build`: type of `whisper.cpp` build to use. Can be set to `cpu` or `cublas-12.4.0`. By default, builds are auto-selected and downloaded for Windows x64 (`cpu`, `cublas-12.4.0`) and Linux x64 (`cpu`). Using other builds requires providing a custom `executablePath`
162
162
  * `whisperCpp.threadCount`: number of threads to use, defaults to `4`
163
163
  * `whisperCpp.splitCount`: number of splits of the audio data to process in parallel (called `--processors` in the `whisper.cpp` CLI). A value greater than `1` can increase memory use significantly, reduce timing accuracy, and slow down execution in some cases. Defaults to `1` (highly recommended)
164
164
  * `whisperCpp.enableGPU`: enable GPU processing. Setting to `true` will try to use a CUDA build, if available for your system. Defaults to `true` when a CUDA-enabled build is selected via `whisperCpp.build`, otherwise `false`
@@ -166,7 +166,7 @@ Applies to CLI operation: `transcribe`, API method: `recognize`
166
166
  * `whisperCpp.beamCount`: the number of decoding paths to use during beam search. Defaults to `5`
167
167
  * `whisperCpp.repetitionThreshold`: minimal repetition / compressibility score to cause a decoded segment to be discarded. Defaults to `2.4`
168
168
  * `whisperCpp.prompt`: initial text to give the Whisper model. Can be a vocabulary, or example text of some sort. Note that if the prompt is very similar to the transcript, the model may intentionally avoid producing the transcript tokens as it may assume that they have already been transcribed. Optional
169
- * `whisperCpp.enableDTW`: enable experimental `whisper.cpp` internal DTW-based token alignment to be used to derive timestamps. Defaults to `false` (recommended for now)
169
+ * `whisperCpp.enableDTW`: enable `whisper.cpp`'s own experimental DTW-based token alignment to be used to derive timestamps. Defaults to `false` (highly recommended)
170
170
  * `whisperCpp.verbose`: show all CLI messages during execution. Defaults to `false`
171
171
 
172
172
  **Vosk**:
@@ -194,9 +194,9 @@ Applies to CLI operation: `transcribe`, API method: `recognize`
194
194
 
195
195
  **OpenAI Cloud**:
196
196
  * `openAICloud.apiKey`: API key (required)
197
- * `openAICloud.model`: model to use. Can only be `whisper-1`
197
+ * `openAICloud.model`: model to use. When using the default provider (OpenAI), can only be `whisper-1`. For a custom provider, like Groq, see its documentation
198
198
  * `openAICloud.organization`: organization identifier. Optional
199
- * `openAICloud.baseURL`: override the default base URL used by the API. Optional
199
+ * `openAICloud.baseURL`: override the default base URL used by the API. For example, set `https://api.groq.com/openai/v1` to use Groq's OpenAI compatible Whisper API instead. Optional
200
200
  * `openAICloud.temperature`: temperature. Choosing `0` uses a dynamic temperature approach. Defaults to `0`
201
201
  * `openAICloud.prompt`: initial prompt for the model. Optional
202
202
  * `openAICloud.timeout`: request timeout. Optional
@@ -265,13 +265,23 @@ Applies to CLI operation: `translate-speech`, API method: `translateSpeech`
265
265
 
266
266
  * `openAICloud`: prefix to provide options for OpenAI cloud. Same options as detailed in the recognition section above
267
267
 
268
+ ## Text-to-text translation
269
+
270
+ Applies to CLI operation: `translate-text`, API method: `translateText`
271
+
272
+ * `engine`: only `google-translate` supported
273
+ * `sourceLanguage`: the source language code for the input text. Auto-detected if not set
274
+ * `targetLanguage`: the target language code for the output text. Required
275
+ * `languageDetection`: language detection options. Optional
276
+
268
277
  ## Speech-to-translated-transcript alignment
269
278
 
270
279
  Applies to CLI operation: `align-translation`, API method: `alignTranslation`
271
280
 
272
281
  **General**:
273
282
  * `engine`: alignment algorithm to use, can only be `whisper`. Defaults to `whisper`
274
- * `language`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
283
+ * `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
284
+ * `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
275
285
  * `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
276
286
  * `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
277
287
  * `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
@@ -284,6 +294,37 @@ Applies to CLI operation: `align-translation`, API method: `alignTranslation`
284
294
  * `whisper.encoderProvider`: encoder ONNX execution provider. See details in recognition section above
285
295
  * `whisper.decoderProvider`: decoder ONNX execution provider. See details in recognition section above
286
296
 
297
+ ## Speech-to-transcript-and-translation alignment
298
+
299
+ Applies to CLI operation: `align-transcript-and-translation`, API method: `alignTranscriptAndTranslation`
300
+
301
+ **General**:
302
+ * `engine`: can only be `two-stage`. Defaults to `two-stage`
303
+ * `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
304
+ * `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
305
+ * `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
306
+ * `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
307
+ * `alignment`: prefix to provide options for alignment. Options detailed in section for alignment
308
+ * `timelineAlignment`: prefix to provide options for timeline alignment. Options detailed in section for timeline alignment
309
+ * `vad`: prefix to provide options for voice activity detection when `crop` is set to `true`. Options detailed in section for voice activity detection
310
+ * `sourceSeparation`: prefix to provide options for source separation when `isolate` is set to `true`. Options detailed in section for source separation
311
+ * `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
312
+
313
+ ## Timeline-to-translated-text alignment
314
+
315
+ Applies to CLI operation: `align-timeline-translation`, API method: `alignTimelineTranslation`
316
+
317
+ **General**:
318
+ * `engine`: alignment engine to use. Can only be `e5`. Defaults to `e5`
319
+ * `sourceLanguage`: language code for the source timeline. Auto-detected from timeline if not set
320
+ * `targetLanguage`: language code for the translated transcript. Auto-detected if not set
321
+ * `audio`: spoken audio to play when previewing the result in the CLI (not required or used by the alignment itself). Optional
322
+ * `languageDetection`: prefix to provide options for language detection. Options detailed in section for text language detection
323
+ * `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
324
+
325
+ **E5**:
326
+ * `e5.model`: E5 model to use. Defaults to `e5-small-fp16` (support for additional models will be added in the future)
327
+
287
328
  ## Language detection
288
329
 
289
330
  ### Speech language detection
package/docs/Tasklist.md CHANGED
@@ -2,11 +2,6 @@
2
2
 
3
3
  ## Bugs
4
4
 
5
- ### Alignment / DTW-RA
6
-
7
- * In DTW-RA, a recognized transcript including something like "Question 2.What does Juan", where "2.What" has a point in the middle, is breaking playback of the timeline
8
-
9
- ### Synthesis
10
5
 
11
6
  ### eSpeak
12
7
 
@@ -24,6 +19,7 @@
24
19
  * `espeak-ng`: 'Oh dear!”' is read as "oh dear exclamation mark", because of the special quote character following the exclamation mark
25
20
  * `espeak-ng`: [Marker right after sentence end is not reported as an event](https://github.com/espeak-ng/espeak-ng/issues/920)
26
21
  * `espeak-ng`: On Japanese text, it says "Chinese character" or "Japanese character" for characters it doesn't know
22
+ * `espeak-ng`: Broken markers on the Korean voice
27
23
  * `wtf_wikipedia` Sometimes fails on `getResult.js` without throwing a humanly readable error
28
24
  * `wtf_wikipedia` Sometimes captures markup like `.svg` etc.
29
25
  * `msspeech`: Initialization fails on Chinese and Japanese voices (but not Korean)
@@ -255,15 +251,7 @@
255
251
  * Allow `dtw` mode work with more speech synthesizers to produce its reference
256
252
  * Predict timing for individual letters (graphemes) based on phoneme timestamps (especially useful for Chinese and Japanese)
257
253
 
258
- ### Translation
259
- * Add text-to-text translation with cloud translation APIs like Google Translate or DeepL, or offline models like OpenNMT or NLLB-200
260
-
261
- ### Translation alignment
262
- * Perform word-level alignment of text-to-text translations from and to any language (not just English) using methods like multilingual embeddings, or specialized models, and then use the text-based alignment to align speech in any recognized language, to its translated transcript in any language supported by the text-to-text alignment approach
263
-
264
- ### Speech-to-text translation
265
254
 
266
- * Hybrid approach: recognize speech in its native language using any recognition model, then translate the resulting transcript using a text-to-text translation engine, and then align the translated transcript to the original one using text-to-text alignment, and map back to the original speech using the recognition timestamps, to get word-level alignment for the translated transcript
267
255
 
268
256
  ## Possible new engines or platforms
269
257
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "echogarden",
3
- "version": "1.4.4",
3
+ "version": "1.6.0",
4
4
  "description": "An easy-to-use speech toolset. Includes tools for synthesis, recognition, alignment, speech translation, language detection, source separation and more.",
5
5
  "author": "Rotem Dan",
6
6
  "license": "GPL-3.0",
@@ -55,8 +55,8 @@
55
55
  "echogarden": "./dist/cli/CLILauncher.js"
56
56
  },
57
57
  "dependencies": {
58
- "@aws-sdk/client-polly": "^3.576.0",
59
- "@aws-sdk/client-transcribe-streaming": "^3.576.0",
58
+ "@aws-sdk/client-polly": "^3.664.0",
59
+ "@aws-sdk/client-transcribe-streaming": "^3.664.0",
60
60
  "@echogarden/espeak-ng-emscripten": "^0.1.2",
61
61
  "@echogarden/fasttext-wasm": "^0.1.0",
62
62
  "@echogarden/flite-wasi": "^0.1.1",
@@ -74,42 +74,38 @@
74
74
  "chalk": "^5.3.0",
75
75
  "cldr-segmentation": "^2.2.1",
76
76
  "command-exists": "^1.2.9",
77
- "compromise": "^14.13.0",
77
+ "compromise": "^14.14.0",
78
78
  "fs-extra": "^11.2.0",
79
- "gaxios": "^6.5.0",
79
+ "gaxios": "^6.7.1",
80
80
  "graceful-fs": "^4.2.11",
81
81
  "html-escaper": "^3.0.3",
82
82
  "html-to-text": "^9.0.5",
83
83
  "import-meta-resolve": "^4.1.0",
84
- "jieba-wasm": "^0.0.2",
85
- "jsdom": "^24.0.0",
84
+ "jieba-wasm": "^2.1.1",
85
+ "jsdom": "^25.0.1",
86
86
  "json5": "^2.2.3",
87
87
  "kuromoji": "^0.1.2",
88
- "microsoft-cognitiveservices-speech-sdk": "^1.36.0",
88
+ "microsoft-cognitiveservices-speech-sdk": "^1.40.0",
89
89
  "moving-median": "^1.0.0",
90
90
  "msgpack-lite": "^0.1.26",
91
- "onnxruntime-node": "^1.17.3",
92
- "openai": "^4.47.1",
93
- "sam-js": "^0.2.1",
91
+ "onnxruntime-node": "^1.19.2",
92
+ "openai": "^4.67.1",
93
+ "sam-js": "^0.3.1",
94
94
  "strip-ansi": "^7.1.0",
95
- "tar": "^7.1.0",
96
- "tiktoken": "^1.0.15",
95
+ "tar": "^7.4.3",
96
+ "tiktoken": "^1.0.16",
97
97
  "tinyld": "^1.3.4",
98
- "ws": "^8.17.0",
99
- "wtf_wikipedia": "^10.3.1"
98
+ "ws": "^8.18.0",
99
+ "wtf_wikipedia": "^10.3.2"
100
100
  },
101
101
  "peerDependencies": {
102
102
  "@echogarden/vosk": "^0.3.39-patched.1",
103
- "speaker": "^0.5.5",
104
103
  "winax": "^3.4.2"
105
104
  },
106
105
  "peerDependenciesMeta": {
107
106
  "@echogarden/vosk": {
108
107
  "optional": true
109
108
  },
110
- "speaker": {
111
- "optional": true
112
- },
113
109
  "winax": {
114
110
  "optional": true
115
111
  }
@@ -118,13 +114,13 @@
118
114
  "@types/buffer-split": "^1.0.2",
119
115
  "@types/fs-extra": "^11.0.4",
120
116
  "@types/graceful-fs": "^4.1.9",
121
- "@types/jsdom": "^21.1.6",
117
+ "@types/jsdom": "^21.1.7",
122
118
  "@types/msgpack-lite": "^0.1.11",
123
- "@types/node": "^20.12.12",
119
+ "@types/node": "^22.7.4",
124
120
  "@types/recursive-readdir": "^2.2.4",
125
121
  "@types/tar": "^6.1.13",
126
- "@types/ws": "^8.5.10",
127
- "ts-json-schema-generator": "^2.1.2-next.1",
128
- "typescript": "^5.4.5"
122
+ "@types/ws": "^8.5.12",
123
+ "ts-json-schema-generator": "^2.3.0",
124
+ "typescript": "^5.6.2"
129
125
  }
130
126
  }
@@ -1,15 +1,15 @@
1
- import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclidianDistance, euclidianDistance13Dim, magnitude } from '../math/VectorMath.js'
1
+ import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclideanDistance, euclideanDistance13Dim, magnitude } from '../math/VectorMath.js'
2
2
  import { logToStderr } from '../utilities/Utilities.js'
3
3
  import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
4
4
 
5
5
  const log = logToStderr
6
6
 
7
- export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: 'euclidian' | 'cosine' = 'euclidian', centerIndexes?: number[]) {
8
- if (distanceFunctionKind == 'euclidian') {
9
- let distanceFunction = euclidianDistance
7
+ export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: 'euclidean' | 'cosine' = 'euclidean', centerIndexes?: number[]) {
8
+ if (distanceFunctionKind == 'euclidean') {
9
+ let distanceFunction = euclideanDistance
10
10
 
11
11
  if (mfccFrames1.length > 0 && mfccFrames1[0].length === 13) {
12
- distanceFunction = euclidianDistance13Dim
12
+ distanceFunction = euclideanDistance13Dim
13
13
  }
14
14
 
15
15
  const { path } = alignDTWWindowed(
@@ -4,9 +4,7 @@ import { AlignmentPath } from './SpeechAlignment.js'
4
4
  const log = logToStderr
5
5
 
6
6
  export function alignDTWWindowed<T, U>(sequence1: T[], sequence2: U[], costFunction: (a: T, b: U) => number, windowMaxLength: number, centerIndexes?: number[]) {
7
- if (windowMaxLength < 2) {
8
- throw new Error('Window length must be greater or equal to 2')
9
- }
7
+ windowMaxLength = Math.max(windowMaxLength, 2)
10
8
 
11
9
  if (sequence1.length == 0 || sequence2.length == 0) {
12
10
  return {