echogarden 1.4.4 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +310 -25
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWMfccSequenceAlignment.js +5 -5
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +1 -3
- package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
- package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +4 -2
- package/dist/alignment/SemanticTextAlignment.js +336 -0
- package/dist/alignment/SemanticTextAlignment.js.map +1 -0
- package/dist/alignment/SpeechAlignment.d.ts +4 -3
- package/dist/alignment/SpeechAlignment.js +130 -39
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +7 -3
- package/dist/api/API.js +7 -2
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +4 -1
- package/dist/api/Alignment.d.ts +1 -1
- package/dist/api/Alignment.js +13 -5
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetectionCommon.d.ts +6 -0
- package/dist/api/LanguageDetectionCommon.js +2 -0
- package/dist/api/LanguageDetectionCommon.js.map +1 -0
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
- package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
- package/dist/api/SpeechLanguageDetection.js.map +1 -0
- package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
- package/dist/api/SpeechTranslation.js.map +1 -0
- package/dist/api/Synthesis.d.ts +0 -1
- package/dist/api/Synthesis.js +4 -4
- package/dist/api/TextLanguageDetection.d.ts +21 -0
- package/dist/api/TextLanguageDetection.js +67 -0
- package/dist/api/TextLanguageDetection.js.map +1 -0
- package/dist/api/TextTranslation.d.ts +25 -0
- package/dist/api/TextTranslation.js +101 -0
- package/dist/api/TextTranslation.js.map +1 -0
- package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
- package/dist/api/TimelineTranslationAlignment.js +92 -0
- package/dist/api/TimelineTranslationAlignment.js.map +1 -0
- package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
- package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
- package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
- package/dist/api/TranslationAlignment.d.ts +4 -3
- package/dist/api/TranslationAlignment.js +9 -8
- package/dist/api/TranslationAlignment.js.map +1 -1
- package/dist/api/VoiceActivityDetection.js +16 -1
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/audio/AudioBufferConversion.d.ts +0 -1
- package/dist/audio/AudioPlayer.d.ts +0 -1
- package/dist/audio/AudioPlayer.js +62 -41
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +0 -1
- package/dist/cli/CLI.d.ts +28 -7
- package/dist/cli/CLI.js +265 -37
- package/dist/cli/CLI.js.map +1 -1
- package/dist/codecs/FFMpegTranscoder.d.ts +0 -1
- package/dist/codecs/FFMpegTranscoder.js +7 -0
- package/dist/codecs/FFMpegTranscoder.js.map +1 -1
- package/dist/codecs/TIMITCodec.d.ts +0 -1
- package/dist/codecs/WaveCodec.d.ts +0 -1
- package/dist/dsp/FFT.d.ts +1 -1
- package/dist/dsp/FFT.js +6 -0
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/dsp/KWeightingFilter.js +1 -1
- package/dist/dsp/KWeightingFilter.js.map +1 -1
- package/dist/dsp/MelSpectogram.d.ts +3 -2
- package/dist/dsp/MelSpectogram.js +14 -8
- package/dist/dsp/MelSpectogram.js.map +1 -1
- package/dist/math/VectorMath.d.ts +9 -9
- package/dist/math/VectorMath.js +10 -10
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/nlp/ChineseSegmentation.js +4 -4
- package/dist/nlp/ChineseSegmentation.js.map +1 -1
- package/dist/nlp/Segmentation.d.ts +2 -2
- package/dist/nlp/Segmentation.js +20 -13
- package/dist/nlp/Segmentation.js.map +1 -1
- package/dist/recognition/OpenAICloudSTT.d.ts +2 -1
- package/dist/recognition/OpenAICloudSTT.js +30 -19
- package/dist/recognition/OpenAICloudSTT.js.map +1 -1
- package/dist/recognition/SileroSTT.d.ts +0 -1
- package/dist/recognition/WhisperCppSTT.d.ts +3 -3
- package/dist/recognition/WhisperCppSTT.js +21 -9
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.d.ts +9 -6
- package/dist/recognition/WhisperSTT.js +227 -46
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Client.d.ts +3 -4
- package/dist/server/Client.js.map +1 -1
- package/dist/server/Worker.d.ts +3 -3
- package/dist/server/Worker.js +3 -2
- package/dist/server/Worker.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +0 -1
- package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +12 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -2
- package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/subtitles/Subtitles.js +2 -2
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.js +6 -21
- package/dist/synthesis/GoogleTranslateTTS.js.map +1 -1
- package/dist/synthesis/StreamlabsPollyTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.js +30 -0
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/tests/Test.js +0 -31
- package/dist/tests/Test.js.map +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
- package/dist/text-translation/DeepLTextTranslation.d.ts +2 -0
- package/dist/text-translation/DeepLTextTranslation.js +67 -0
- package/dist/text-translation/DeepLTextTranslation.js.map +1 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.d.ts +10 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js +554 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js.map +1 -0
- package/dist/text-translation/NLLBTextTranslation.d.ts +2 -1
- package/dist/text-translation/NLLBTextTranslation.js +249 -19
- package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
- package/dist/utilities/BinaryArrayConversion.d.ts +0 -1
- package/dist/utilities/BrowserRequestHeaders.d.ts +6 -0
- package/dist/utilities/BrowserRequestHeaders.js +52 -0
- package/dist/utilities/BrowserRequestHeaders.js.map +1 -0
- package/dist/utilities/BufferFileReadStream.d.ts +20 -0
- package/dist/utilities/BufferFileReadStream.js +81 -0
- package/dist/utilities/BufferFileReadStream.js.map +1 -0
- package/dist/utilities/DynamicUint8Array.d.ts +9 -0
- package/dist/utilities/DynamicUint8Array.js +31 -0
- package/dist/utilities/DynamicUint8Array.js.map +1 -0
- package/dist/utilities/FileSystem.d.ts +0 -2
- package/dist/utilities/Hashing.d.ts +3 -10
- package/dist/utilities/Hashing.js +10 -127
- package/dist/utilities/Hashing.js.map +1 -1
- package/dist/utilities/LEB128.d.ts +15 -5
- package/dist/utilities/LEB128.js +199 -119
- package/dist/utilities/LEB128.js.map +1 -1
- package/dist/utilities/LPVarInt.d.ts +11 -0
- package/dist/utilities/LPVarInt.js +187 -0
- package/dist/utilities/LPVarInt.js.map +1 -0
- package/dist/utilities/Locale.d.ts +1 -1
- package/dist/utilities/Locale.js +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +1 -2
- package/dist/utilities/PVarInt.d.ts +4 -0
- package/dist/utilities/PVarInt.js +166 -0
- package/dist/utilities/PVarInt.js.map +1 -0
- package/dist/utilities/PackageManager.js +48 -25
- package/dist/utilities/PackageManager.js.map +1 -1
- package/dist/utilities/RandomGenerator.d.ts +3 -17
- package/dist/utilities/RandomGenerator.js +12 -81
- package/dist/utilities/RandomGenerator.js.map +1 -1
- package/dist/utilities/Timeline.d.ts +2 -0
- package/dist/utilities/Timeline.js +129 -20
- package/dist/utilities/Timeline.js.map +1 -1
- package/dist/utilities/Utilities.d.ts +1 -3
- package/dist/utilities/Utilities.js +30 -3
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/utilities/VarInt.d.ts +4 -0
- package/dist/utilities/VarInt.js +166 -0
- package/dist/utilities/VarInt.js.map +1 -0
- package/dist/utilities/VirtualFileReadStream.d.ts +20 -0
- package/dist/utilities/VirtualFileReadStream.js +79 -0
- package/dist/utilities/VirtualFileReadStream.js.map +1 -0
- package/dist/utilities/WebReader.js +7 -23
- package/dist/utilities/WebReader.js.map +1 -1
- package/dist/voice-activity-detection/SileroVAD.d.ts +0 -1
- package/docs/API.md +105 -3
- package/docs/CLI.md +51 -1
- package/docs/Engines.md +32 -3
- package/docs/Options.md +53 -12
- package/docs/Tasklist.md +1 -13
- package/package.json +20 -24
- package/src/alignment/DTWMfccSequenceAlignment.ts +5 -5
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +1 -3
- package/src/alignment/SemanticTextAlignment.ts +467 -0
- package/src/alignment/SpeechAlignment.ts +214 -56
- package/src/api/API.ts +18 -2
- package/src/api/APIOptions.ts +14 -1
- package/src/api/Alignment.ts +31 -9
- package/src/api/LanguageDetectionCommon.ts +7 -0
- package/src/api/Recognition.ts +2 -0
- package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
- package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
- package/src/api/Synthesis.ts +4 -4
- package/src/api/TextLanguageDetection.ts +116 -0
- package/src/api/TextTranslation.ts +177 -0
- package/src/api/TimelineTranslationAlignment.ts +162 -0
- package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
- package/src/api/TranslationAlignment.ts +12 -10
- package/src/api/VoiceActivityDetection.ts +24 -3
- package/src/audio/AudioPlayer.ts +2 -0
- package/src/cli/CLI.ts +376 -40
- package/src/codecs/FFMpegTranscoder.ts +6 -0
- package/src/dsp/FFT.ts +8 -2
- package/src/dsp/KWeightingFilter.ts +1 -1
- package/src/dsp/MelSpectogram.ts +17 -8
- package/src/math/VectorMath.ts +15 -15
- package/src/nlp/ChineseSegmentation.ts +6 -4
- package/src/nlp/Segmentation.ts +18 -13
- package/src/recognition/OpenAICloudSTT.ts +47 -29
- package/src/recognition/WhisperCppSTT.ts +26 -11
- package/src/recognition/WhisperSTT.ts +364 -49
- package/src/server/Client.ts +3 -2
- package/src/server/Worker.ts +3 -2
- package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
- package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
- package/src/subtitles/Subtitles.ts +2 -2
- package/src/synthesis/GoogleTranslateTTS.ts +7 -21
- package/src/synthesis/VitsTTS.ts +31 -3
- package/src/tests/Test.ts +1 -38
- package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
- package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
- package/src/text-translation/DeepLTextTranslation.ts +88 -0
- package/src/text-translation/GoogleTranslateTextTranslation.ts +667 -0
- package/src/text-translation/NLLBTextTranslation.ts +261 -21
- package/src/typings/Fillers.d.ts +25 -2
- package/src/utilities/BrowserRequestHeaders.ts +59 -0
- package/src/utilities/DynamicUint8Array.ts +39 -0
- package/src/utilities/Hashing.ts +14 -167
- package/src/utilities/LEB128.ts +273 -148
- package/src/utilities/LPVarInt.ts +292 -0
- package/src/utilities/Locale.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +1 -1
- package/src/utilities/PackageManager.ts +51 -30
- package/src/utilities/RandomGenerator.ts +12 -113
- package/src/utilities/Timeline.ts +162 -23
- package/src/utilities/Utilities.ts +40 -3
- package/src/utilities/VirtualFileReadStream.ts +109 -0
- package/src/utilities/WebReader.ts +9 -23
- package/dist/alignment/TextAlignment.js +0 -156
- package/dist/alignment/TextAlignment.js.map +0 -1
- package/dist/api/LanguageDetection.js.map +0 -1
- package/dist/api/Translation.js.map +0 -1
- package/src/alignment/TextAlignment.ts +0 -234
- /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
package/docs/Options.md
CHANGED
|
@@ -141,24 +141,24 @@ Applies to CLI operation: `transcribe`, API method: `recognize`
|
|
|
141
141
|
* `sourceSeparation`: prefix to provide options for source separation when `isolate` is set to `true`. Options detailed in section for source separation
|
|
142
142
|
|
|
143
143
|
**Whisper**:
|
|
144
|
-
* `whisper.model`: selects which Whisper model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en
|
|
144
|
+
* `whisper.model`: selects which Whisper model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en` or `large-v3-turbo`. Defaults to `tiny` or `tiny.en`
|
|
145
145
|
* `whisper.temperature`: temperature setting for the text decoder. Impacts the amount of randomization for token selection. It is recommended to leave at `0.1` (close to no randomization - almost always chooses the top ranked token) or choose a relatively low value (`0.25` or lower) for best results. Defaults to `0.1`
|
|
146
146
|
* `whisper.prompt`: initial text to give the Whisper model. Can be a vocabulary, or example text of some sort. Note that if the prompt is very similar to the transcript, the model may intentionally avoid producing the transcript tokens as it may assume that they have already been transcribed. Optional
|
|
147
147
|
* `whisper.topCandidateCount`: the number of top candidate tokens to consider. Defaults to `5`
|
|
148
|
-
* `whisper.punctuationThreshold`: the minimal probability for a punctuation token, included in the top candidates, to be chosen unconditionally. A lower threshold encourages the model to output more punctuation
|
|
149
|
-
* `whisper.autoPromptParts`: use previous part's recognized text as the prompt for the next part. Disabling this may help to prevent repetition carrying over between parts, in some cases. Defaults to `true`
|
|
148
|
+
* `whisper.punctuationThreshold`: the minimal probability for a punctuation token, included in the top candidates, to be chosen unconditionally. A lower threshold encourages the model to output more punctuation characters. Defaults to `0.2`
|
|
149
|
+
* `whisper.autoPromptParts`: use previous part's recognized text as the prompt for the next part. Disabling this may help to prevent repetition carrying over between parts, in some cases. Defaults to `true` (**Note**: currently always disabled for `large-v3-turbo` model due to an apparent issue with corrupt output when prompted)
|
|
150
150
|
* `whisper.maxTokensPerPart`: maximum number of tokens to decode for each audio part. Defaults to `250`
|
|
151
151
|
* `whisper.suppressRepetition`: attempt to suppress decoding of repeating token patterns. Defaults to `true`
|
|
152
152
|
* `whisper.repetitionThreshold`: minimal repetition / compressibility score to cause a part not to be auto-prompted to the next part. Defaults to `2.4`
|
|
153
153
|
* `whisper.decodeTimestampTokens`: enable/disable decoding of timestamp tokens. Setting to `false` can reduce the occurrence of hallucinations and token repetition loops, possibly due to the overall reduction in the number of tokens decoded. This has no impact on the accuracy of timestamps, since they are derived independently using cross-attention weights. However, there are cases where this can cause the model to end a part prematurely, especially in singing and less speech-like voice segments, or when there are multiple speakers. Defaults to `true`
|
|
154
|
-
* `whisper.encoderProvider`: identifier for the ONNX execution provider to use with the encoder model. Can be `cpu
|
|
155
|
-
* `whisper.decoderProvider`: identifier for the ONNX execution provider to use with the decoder model. Can be `cpu` or `
|
|
154
|
+
* `whisper.encoderProvider`: identifier for the ONNX execution provider to use with the encoder model. Can be `cpu`, `dml` ([DirectML](https://microsoft.github.io/DirectML/)-based GPU acceleration - Windows only) or `cuda` (Linux only). In general, GPU-based encoding should be significantly faster. Defaults to `cpu`, or `dml` if available
|
|
155
|
+
* `whisper.decoderProvider`: identifier for the ONNX execution provider to use with the decoder model. Can be `cpu`, `dml` (Windows only) or `cuda` (Linux only). Using GPU acceleration for the decoder may be faster than CPU, especially for larger models, but that depends on your particular combination of CPU and GPU. Defaults to `cpu`, and on Windows, `dml` if available for larger models (`small`, `medium`, `large`)
|
|
156
156
|
* `whisper.seed`: provide a custom random seed for token selection when temperature is greater than 0. Uses a constant seed by default to ensure reproducibility
|
|
157
157
|
|
|
158
158
|
**Whisper.cpp**:
|
|
159
|
-
* `whisperCpp.model`: selects which `whisper.cpp` model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en`, `large` (same as `large-v2`), `large-v1`, `large-v2`, `large-v3`. These quantized models are also supported: `tiny-q5_1`, `tiny.en-q5_1`, `tiny.en-q8_0`,`base-q5_1`, `base.en-q5_1`, `small-q5_1`, `small.en-q5_1`, `medium-q5_0`, `medium.en-q5_0`, `large-v2-q5_0`, `large-v3-q5_0`. Defaults to `base` or `base.en`
|
|
160
|
-
* `whisperCpp.executablePath`: custom `whisper.cpp` executable
|
|
161
|
-
* `whisperCpp.build`: type of `whisper.cpp` build to use. Can be set `cpu
|
|
159
|
+
* `whisperCpp.model`: selects which `whisper.cpp` model to use. Can be `tiny`, `tiny.en`, `base`, `base.en`, `small`, `small.en`, `medium`, `medium.en`, `large` (same as `large-v2`), `large-v1`, `large-v2`, `large-v3`. These quantized models are also supported: `tiny-q5_1`, `tiny.en-q5_1`, `tiny.en-q8_0`,`base-q5_1`, `base.en-q5_1`, `small-q5_1`, `small.en-q5_1`, `medium-q5_0`, `medium.en-q5_0`, `large-v2-q5_0`, `large-v3-q5_0`, `large-v3-turbo`, `large-v3-turbo-q5_0`. Defaults to `base` or `base.en`
|
|
160
|
+
* `whisperCpp.executablePath`: a path to a custom `whisper.cpp` `main` executable (currently required for macOS)
|
|
161
|
+
* `whisperCpp.build`: type of `whisper.cpp` build to use. Can be set to `cpu` or `cublas-12.4.0`. By default, builds are auto-selected and downloaded for Windows x64 (`cpu`, `cublas-12.4.0`) and Linux x64 (`cpu`). Using other builds requires providing a custom `executablePath`
|
|
162
162
|
* `whisperCpp.threadCount`: number of threads to use, defaults to `4`
|
|
163
163
|
* `whisperCpp.splitCount`: number of splits of the audio data to process in parallel (called `--processors` in the `whisper.cpp` CLI). A value greater than `1` can increase memory use significantly, reduce timing accuracy, and slow down execution in some cases. Defaults to `1` (highly recommended)
|
|
164
164
|
* `whisperCpp.enableGPU`: enable GPU processing. Setting to `true` will try to use a CUDA build, if available for your system. Defaults to `true` when a CUDA-enabled build is selected via `whisperCpp.build`, otherwise `false`
|
|
@@ -166,7 +166,7 @@ Applies to CLI operation: `transcribe`, API method: `recognize`
|
|
|
166
166
|
* `whisperCpp.beamCount`: the number of decoding paths to use during beam search. Defaults to `5`
|
|
167
167
|
* `whisperCpp.repetitionThreshold`: minimal repetition / compressibility score to cause a decoded segment to be discarded. Defaults to `2.4`
|
|
168
168
|
* `whisperCpp.prompt`: initial text to give the Whisper model. Can be a vocabulary, or example text of some sort. Note that if the prompt is very similar to the transcript, the model may intentionally avoid producing the transcript tokens as it may assume that they have already been transcribed. Optional
|
|
169
|
-
* `whisperCpp.enableDTW`: enable
|
|
169
|
+
* `whisperCpp.enableDTW`: enable `whisper.cpp`'s own experimental DTW-based token alignment to be used to derive timestamps. Defaults to `false` (highly recommended)
|
|
170
170
|
* `whisperCpp.verbose`: show all CLI messages during execution. Defaults to `false`
|
|
171
171
|
|
|
172
172
|
**Vosk**:
|
|
@@ -194,9 +194,9 @@ Applies to CLI operation: `transcribe`, API method: `recognize`
|
|
|
194
194
|
|
|
195
195
|
**OpenAI Cloud**:
|
|
196
196
|
* `openAICloud.apiKey`: API key (required)
|
|
197
|
-
* `openAICloud.model`: model to use.
|
|
197
|
+
* `openAICloud.model`: model to use. When using the default provider (OpenAI), can only be `whisper-1`. For a custom provider, like Groq, see its documentation
|
|
198
198
|
* `openAICloud.organization`: organization identifier. Optional
|
|
199
|
-
* `openAICloud.baseURL`: override the default base URL used by the API. Optional
|
|
199
|
+
* `openAICloud.baseURL`: override the default base URL used by the API. For example, set `https://api.groq.com/openai/v1` to use Groq's OpenAI compatible Whisper API instead. Optional
|
|
200
200
|
* `openAICloud.temperature`: temperature. Choosing `0` uses a dynamic temperature approach. Defaults to `0`
|
|
201
201
|
* `openAICloud.prompt`: initial prompt for the model. Optional
|
|
202
202
|
* `openAICloud.timeout`: request timeout. Optional
|
|
@@ -265,13 +265,23 @@ Applies to CLI operation: `translate-speech`, API method: `translateSpeech`
|
|
|
265
265
|
|
|
266
266
|
* `openAICloud`: prefix to provide options for OpenAI cloud. Same options as detailed in the recognition section above
|
|
267
267
|
|
|
268
|
+
## Text-to-text translation
|
|
269
|
+
|
|
270
|
+
Applies to CLI operation: `translate-text`, API method: `translateText`
|
|
271
|
+
|
|
272
|
+
* `engine`: only `google-translate` supported
|
|
273
|
+
* `sourceLanguage`: the source language code for the input text. Auto-detected if not set
|
|
274
|
+
* `targetLanguage`: the target language code for the output text. Required
|
|
275
|
+
* `languageDetection`: language detection options. Optional
|
|
276
|
+
|
|
268
277
|
## Speech-to-translated-transcript alignment
|
|
269
278
|
|
|
270
279
|
Applies to CLI operation: `align-translation`, API method: `alignTranslation`
|
|
271
280
|
|
|
272
281
|
**General**:
|
|
273
282
|
* `engine`: alignment algorithm to use, can only be `whisper`. Defaults to `whisper`
|
|
274
|
-
* `
|
|
283
|
+
* `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
|
|
284
|
+
* `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
|
|
275
285
|
* `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
|
|
276
286
|
* `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
|
|
277
287
|
* `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
|
|
@@ -284,6 +294,37 @@ Applies to CLI operation: `align-translation`, API method: `alignTranslation`
|
|
|
284
294
|
* `whisper.encoderProvider`: encoder ONNX execution provider. See details in recognition section above
|
|
285
295
|
* `whisper.decoderProvider`: decoder ONNX execution provider. See details in recognition section above
|
|
286
296
|
|
|
297
|
+
## Speech-to-transcript-and-translation alignment
|
|
298
|
+
|
|
299
|
+
Applies to CLI operation: `align-transcript-and-translation`, API method: `alignTranscriptAndTranslation`
|
|
300
|
+
|
|
301
|
+
**General**:
|
|
302
|
+
* `engine`: can only be `two-stage`. Defaults to `two-stage`
|
|
303
|
+
* `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
|
|
304
|
+
* `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
|
|
305
|
+
* `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
|
|
306
|
+
* `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
|
|
307
|
+
* `alignment`: prefix to provide options for alignment. Options detailed in section for alignment
|
|
308
|
+
* `timelineAlignment`: prefix to provide options for timeline alignment. Options detailed in section for timeline alignment
|
|
309
|
+
* `vad`: prefix to provide options for voice activity detection when `crop` is set to `true`. Options detailed in section for voice activity detection
|
|
310
|
+
* `sourceSeparation`: prefix to provide options for source separation when `isolate` is set to `true`. Options detailed in section for source separation
|
|
311
|
+
* `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
|
|
312
|
+
|
|
313
|
+
## Timeline-to-translated-text alignment
|
|
314
|
+
|
|
315
|
+
Applies to CLI operation: `align-timeline-translation`, API method: `alignTimelineTranslation`
|
|
316
|
+
|
|
317
|
+
**General**:
|
|
318
|
+
* `engine`: alignment engine to use. Can only be `e5`. Defaults to `e5`
|
|
319
|
+
* `sourceLanguage`: language code for the source timeline. Auto-detected from timeline if not set
|
|
320
|
+
* `targetLanguage`: language code for the translated transcript. Auto-detected if not set
|
|
321
|
+
* `audio`: spoken audio to play when previewing the result in the CLI (not required or used by the alignment itself). Optional
|
|
322
|
+
* `languageDetection`: prefix to provide options for language detection. Options detailed in section for text language detection
|
|
323
|
+
* `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
|
|
324
|
+
|
|
325
|
+
**E5**:
|
|
326
|
+
* `e5.model`: E5 model to use. Defaults to `e5-small-fp16` (support for additional models will be added in the future)
|
|
327
|
+
|
|
287
328
|
## Language detection
|
|
288
329
|
|
|
289
330
|
### Speech language detection
|
package/docs/Tasklist.md
CHANGED
|
@@ -2,11 +2,6 @@
|
|
|
2
2
|
|
|
3
3
|
## Bugs
|
|
4
4
|
|
|
5
|
-
### Alignment / DTW-RA
|
|
6
|
-
|
|
7
|
-
* In DTW-RA, a recognized transcript including something like "Question 2.What does Juan", where "2.What" has a point in the middle, is breaking playback of the timeline
|
|
8
|
-
|
|
9
|
-
### Synthesis
|
|
10
5
|
|
|
11
6
|
### eSpeak
|
|
12
7
|
|
|
@@ -24,6 +19,7 @@
|
|
|
24
19
|
* `espeak-ng`: 'Oh dear!”' is read as "oh dear exclamation mark", because of the special quote character following the exclamation mark
|
|
25
20
|
* `espeak-ng`: [Marker right after sentence end is not reported as an event](https://github.com/espeak-ng/espeak-ng/issues/920)
|
|
26
21
|
* `espeak-ng`: On Japanese text, it says "Chinese character" or "Japanese character" for characters it doesn't know
|
|
22
|
+
* `espeak-ng`: Broken markers on the Korean voice
|
|
27
23
|
* `wtf_wikipedia` Sometimes fails on `getResult.js` without throwing a humanly readable error
|
|
28
24
|
* `wtf_wikipedia` Sometimes captures markup like `.svg` etc.
|
|
29
25
|
* `msspeech`: Initialization fails on Chinese and Japanese voices (but not Korean)
|
|
@@ -255,15 +251,7 @@
|
|
|
255
251
|
* Allow `dtw` mode work with more speech synthesizers to produce its reference
|
|
256
252
|
* Predict timing for individual letters (graphemes) based on phoneme timestamps (especially useful for Chinese and Japanese)
|
|
257
253
|
|
|
258
|
-
### Translation
|
|
259
|
-
* Add text-to-text translation with cloud translation APIs like Google Translate or DeepL, or offline models like OpenNMT or NLLB-200
|
|
260
|
-
|
|
261
|
-
### Translation alignment
|
|
262
|
-
* Perform word-level alignment of text-to-text translations from and to any language (not just English) using methods like multilingual embeddings, or specialized models, and then use the text-based alignment to align speech in any recognized language, to its translated transcript in any language supported by the text-to-text alignment approach
|
|
263
|
-
|
|
264
|
-
### Speech-to-text translation
|
|
265
254
|
|
|
266
|
-
* Hybrid approach: recognize speech in its native language using any recognition model, then translate the resulting transcript using a text-to-text translation engine, and then align the translated transcript to the original one using text-to-text alignment, and map back to the original speech using the recognition timestamps, to get word-level alignment for the translated transcript
|
|
267
255
|
|
|
268
256
|
## Possible new engines or platforms
|
|
269
257
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "echogarden",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"description": "An easy-to-use speech toolset. Includes tools for synthesis, recognition, alignment, speech translation, language detection, source separation and more.",
|
|
5
5
|
"author": "Rotem Dan",
|
|
6
6
|
"license": "GPL-3.0",
|
|
@@ -55,8 +55,8 @@
|
|
|
55
55
|
"echogarden": "./dist/cli/CLILauncher.js"
|
|
56
56
|
},
|
|
57
57
|
"dependencies": {
|
|
58
|
-
"@aws-sdk/client-polly": "^3.
|
|
59
|
-
"@aws-sdk/client-transcribe-streaming": "^3.
|
|
58
|
+
"@aws-sdk/client-polly": "^3.664.0",
|
|
59
|
+
"@aws-sdk/client-transcribe-streaming": "^3.664.0",
|
|
60
60
|
"@echogarden/espeak-ng-emscripten": "^0.1.2",
|
|
61
61
|
"@echogarden/fasttext-wasm": "^0.1.0",
|
|
62
62
|
"@echogarden/flite-wasi": "^0.1.1",
|
|
@@ -74,42 +74,38 @@
|
|
|
74
74
|
"chalk": "^5.3.0",
|
|
75
75
|
"cldr-segmentation": "^2.2.1",
|
|
76
76
|
"command-exists": "^1.2.9",
|
|
77
|
-
"compromise": "^14.
|
|
77
|
+
"compromise": "^14.14.0",
|
|
78
78
|
"fs-extra": "^11.2.0",
|
|
79
|
-
"gaxios": "^6.
|
|
79
|
+
"gaxios": "^6.7.1",
|
|
80
80
|
"graceful-fs": "^4.2.11",
|
|
81
81
|
"html-escaper": "^3.0.3",
|
|
82
82
|
"html-to-text": "^9.0.5",
|
|
83
83
|
"import-meta-resolve": "^4.1.0",
|
|
84
|
-
"jieba-wasm": "^
|
|
85
|
-
"jsdom": "^
|
|
84
|
+
"jieba-wasm": "^2.1.1",
|
|
85
|
+
"jsdom": "^25.0.1",
|
|
86
86
|
"json5": "^2.2.3",
|
|
87
87
|
"kuromoji": "^0.1.2",
|
|
88
|
-
"microsoft-cognitiveservices-speech-sdk": "^1.
|
|
88
|
+
"microsoft-cognitiveservices-speech-sdk": "^1.40.0",
|
|
89
89
|
"moving-median": "^1.0.0",
|
|
90
90
|
"msgpack-lite": "^0.1.26",
|
|
91
|
-
"onnxruntime-node": "^1.
|
|
92
|
-
"openai": "^4.
|
|
93
|
-
"sam-js": "^0.
|
|
91
|
+
"onnxruntime-node": "^1.19.2",
|
|
92
|
+
"openai": "^4.67.1",
|
|
93
|
+
"sam-js": "^0.3.1",
|
|
94
94
|
"strip-ansi": "^7.1.0",
|
|
95
|
-
"tar": "^7.
|
|
96
|
-
"tiktoken": "^1.0.
|
|
95
|
+
"tar": "^7.4.3",
|
|
96
|
+
"tiktoken": "^1.0.16",
|
|
97
97
|
"tinyld": "^1.3.4",
|
|
98
|
-
"ws": "^8.
|
|
99
|
-
"wtf_wikipedia": "^10.3.
|
|
98
|
+
"ws": "^8.18.0",
|
|
99
|
+
"wtf_wikipedia": "^10.3.2"
|
|
100
100
|
},
|
|
101
101
|
"peerDependencies": {
|
|
102
102
|
"@echogarden/vosk": "^0.3.39-patched.1",
|
|
103
|
-
"speaker": "^0.5.5",
|
|
104
103
|
"winax": "^3.4.2"
|
|
105
104
|
},
|
|
106
105
|
"peerDependenciesMeta": {
|
|
107
106
|
"@echogarden/vosk": {
|
|
108
107
|
"optional": true
|
|
109
108
|
},
|
|
110
|
-
"speaker": {
|
|
111
|
-
"optional": true
|
|
112
|
-
},
|
|
113
109
|
"winax": {
|
|
114
110
|
"optional": true
|
|
115
111
|
}
|
|
@@ -118,13 +114,13 @@
|
|
|
118
114
|
"@types/buffer-split": "^1.0.2",
|
|
119
115
|
"@types/fs-extra": "^11.0.4",
|
|
120
116
|
"@types/graceful-fs": "^4.1.9",
|
|
121
|
-
"@types/jsdom": "^21.1.
|
|
117
|
+
"@types/jsdom": "^21.1.7",
|
|
122
118
|
"@types/msgpack-lite": "^0.1.11",
|
|
123
|
-
"@types/node": "^
|
|
119
|
+
"@types/node": "^22.7.4",
|
|
124
120
|
"@types/recursive-readdir": "^2.2.4",
|
|
125
121
|
"@types/tar": "^6.1.13",
|
|
126
|
-
"@types/ws": "^8.5.
|
|
127
|
-
"ts-json-schema-generator": "^2.
|
|
128
|
-
"typescript": "^5.
|
|
122
|
+
"@types/ws": "^8.5.12",
|
|
123
|
+
"ts-json-schema-generator": "^2.3.0",
|
|
124
|
+
"typescript": "^5.6.2"
|
|
129
125
|
}
|
|
130
126
|
}
|
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange,
|
|
1
|
+
import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclideanDistance, euclideanDistance13Dim, magnitude } from '../math/VectorMath.js'
|
|
2
2
|
import { logToStderr } from '../utilities/Utilities.js'
|
|
3
3
|
import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
|
|
4
4
|
|
|
5
5
|
const log = logToStderr
|
|
6
6
|
|
|
7
|
-
export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: '
|
|
8
|
-
if (distanceFunctionKind == '
|
|
9
|
-
let distanceFunction =
|
|
7
|
+
export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: 'euclidean' | 'cosine' = 'euclidean', centerIndexes?: number[]) {
|
|
8
|
+
if (distanceFunctionKind == 'euclidean') {
|
|
9
|
+
let distanceFunction = euclideanDistance
|
|
10
10
|
|
|
11
11
|
if (mfccFrames1.length > 0 && mfccFrames1[0].length === 13) {
|
|
12
|
-
distanceFunction =
|
|
12
|
+
distanceFunction = euclideanDistance13Dim
|
|
13
13
|
}
|
|
14
14
|
|
|
15
15
|
const { path } = alignDTWWindowed(
|
|
@@ -4,9 +4,7 @@ import { AlignmentPath } from './SpeechAlignment.js'
|
|
|
4
4
|
const log = logToStderr
|
|
5
5
|
|
|
6
6
|
export function alignDTWWindowed<T, U>(sequence1: T[], sequence2: U[], costFunction: (a: T, b: U) => number, windowMaxLength: number, centerIndexes?: number[]) {
|
|
7
|
-
|
|
8
|
-
throw new Error('Window length must be greater or equal to 2')
|
|
9
|
-
}
|
|
7
|
+
windowMaxLength = Math.max(windowMaxLength, 2)
|
|
10
8
|
|
|
11
9
|
if (sequence1.length == 0 || sequence2.length == 0) {
|
|
12
10
|
return {
|