echogarden 0.12.2 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -14
- package/data/schemas/options.json +398 -111
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWMfccSequenceAlignment.js +8 -8
- package/dist/alignment/DTWSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWSequenceAlignment.js +1 -1
- package/dist/alignment/DTWSequenceAlignmentWindowed.d.ts +1 -1
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +2 -2
- package/dist/alignment/LevenshteinSequenceAlignment.d.ts +1 -1
- package/dist/alignment/LevenshteinSequenceAlignment.js +1 -1
- package/dist/alignment/SpeechAlignment.d.ts +9 -10
- package/dist/alignment/SpeechAlignment.js +136 -105
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +13 -12
- package/dist/api/API.js +14 -13
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +5 -4
- package/dist/api/Alignment.d.ts +15 -9
- package/dist/api/Alignment.js +88 -74
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/Common.js +1 -1
- package/dist/api/Denoising.d.ts +6 -6
- package/dist/api/Denoising.js +23 -23
- package/dist/api/Denoising.js.map +1 -1
- package/dist/api/LanguageDetection.d.ts +19 -12
- package/dist/api/LanguageDetection.js +88 -38
- package/dist/api/LanguageDetection.js.map +1 -1
- package/dist/api/Recognition.d.ts +16 -6
- package/dist/api/Recognition.js +129 -55
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/SourceSeparation.d.ts +17 -0
- package/dist/api/SourceSeparation.js +61 -0
- package/dist/api/SourceSeparation.js.map +1 -0
- package/dist/api/Synthesis.d.ts +18 -18
- package/dist/api/Synthesis.js +191 -164
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.d.ts +19 -8
- package/dist/api/Translation.js +132 -35
- package/dist/api/Translation.js.map +1 -1
- package/dist/api/Vad.d.ts +10 -5
- package/dist/api/Vad.js +76 -38
- package/dist/api/Vad.js.map +1 -1
- package/dist/audio/AudioBufferConversion.d.ts +1 -1
- package/dist/audio/AudioBufferConversion.js +4 -4
- package/dist/audio/AudioPlayer.d.ts +1 -1
- package/dist/audio/AudioPlayer.js +26 -26
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/audio/AudioRecorder.d.ts +1 -1
- package/dist/audio/AudioRecorder.js +5 -5
- package/dist/audio/AudioUtilities.d.ts +13 -9
- package/dist/audio/AudioUtilities.js +86 -24
- package/dist/audio/AudioUtilities.js.map +1 -1
- package/dist/cli/CLI.d.ts +3 -3
- package/dist/cli/CLI.js +271 -162
- package/dist/cli/CLI.js.map +1 -1
- package/dist/cli/CLIConfigFile.js +8 -8
- package/dist/cli/CLILauncher.js +6 -6
- package/dist/cli/CLIOptionsSchema.js +2 -2
- package/dist/cli/CLIParser.js +5 -5
- package/dist/cli/CLIStarter.js +4 -4
- package/dist/codecs/FFMpegTranscoder.d.ts +2 -2
- package/dist/codecs/FFMpegTranscoder.js +37 -37
- package/dist/codecs/FFMpegTranscoder.js.map +1 -1
- package/dist/codecs/TIMITCodec.js +5 -5
- package/dist/codecs/WaveCodec.d.ts +1 -1
- package/dist/codecs/WaveCodec.js +22 -22
- package/dist/denoising/RNNoise.d.ts +1 -1
- package/dist/denoising/RNNoise.js +9 -9
- package/dist/dsp/BiquadFilter.d.ts +3 -2
- package/dist/dsp/BiquadFilter.js +18 -11
- package/dist/dsp/BiquadFilter.js.map +1 -1
- package/dist/dsp/DecayingPeakEstimator.d.ts +16 -0
- package/dist/dsp/DecayingPeakEstimator.js +23 -0
- package/dist/dsp/DecayingPeakEstimator.js.map +1 -0
- package/dist/dsp/FFT.d.ts +8 -4
- package/dist/dsp/FFT.js +76 -30
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/dsp/KWeightingFilter.d.ts +9 -0
- package/dist/dsp/KWeightingFilter.js +40 -0
- package/dist/dsp/KWeightingFilter.js.map +1 -0
- package/dist/dsp/LoudnessEstimator.d.ts +21 -0
- package/dist/dsp/LoudnessEstimator.js +47 -0
- package/dist/dsp/LoudnessEstimator.js.map +1 -0
- package/dist/dsp/MFCC.d.ts +2 -2
- package/dist/dsp/MFCC.js +15 -15
- package/dist/dsp/MelSpectogram.d.ts +1 -1
- package/dist/dsp/MelSpectogram.js +6 -6
- package/dist/dsp/Rubberband.d.ts +11 -11
- package/dist/dsp/Rubberband.js +27 -27
- package/dist/dsp/Sonic.d.ts +1 -1
- package/dist/dsp/Sonic.js +3 -3
- package/dist/dsp/SpeexResampler.d.ts +1 -1
- package/dist/dsp/SpeexResampler.js +2 -2
- package/dist/math/VectorMath.d.ts +12 -8
- package/dist/math/VectorMath.js +35 -32
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/nlp/ChineseSegmentation.js +2 -2
- package/dist/nlp/CompromiseNLP.js +3 -3
- package/dist/nlp/EspeakPhonemizer.js +30 -30
- package/dist/nlp/IPA.js +20 -20
- package/dist/nlp/JapaneseSegmentation.js +6 -6
- package/dist/nlp/Lexicon.d.ts +1 -1
- package/dist/nlp/Lexicon.js +7 -7
- package/dist/nlp/Segmentation.d.ts +3 -0
- package/dist/nlp/Segmentation.js +21 -14
- package/dist/nlp/Segmentation.js.map +1 -1
- package/dist/nlp/TextNormalizer.js +16 -16
- package/dist/recognition/AmazonTranscribeSTT.d.ts +2 -2
- package/dist/recognition/AmazonTranscribeSTT.js +13 -14
- package/dist/recognition/AmazonTranscribeSTT.js.map +1 -1
- package/dist/recognition/AzureCognitiveServicesSTT.js +5 -6
- package/dist/recognition/AzureCognitiveServicesSTT.js.map +1 -1
- package/dist/recognition/GoogleCloudSTT.d.ts +3 -3
- package/dist/recognition/GoogleCloudSTT.js +18 -18
- package/dist/recognition/OpenAICloudSTT.d.ts +19 -0
- package/dist/recognition/OpenAICloudSTT.js +81 -0
- package/dist/recognition/OpenAICloudSTT.js.map +1 -0
- package/dist/recognition/SileroSTT.d.ts +2 -2
- package/dist/recognition/SileroSTT.js +25 -25
- package/dist/recognition/VoskSTT.d.ts +2 -2
- package/dist/recognition/VoskSTT.js +8 -8
- package/dist/recognition/WhisperCppSTT.d.ts +88 -0
- package/dist/recognition/WhisperCppSTT.js +332 -0
- package/dist/recognition/WhisperCppSTT.js.map +1 -0
- package/dist/recognition/WhisperSTT.d.ts +49 -25
- package/dist/recognition/WhisperSTT.js +626 -481
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Client.d.ts +1 -1
- package/dist/server/Client.js +22 -22
- package/dist/server/Server.js +9 -9
- package/dist/server/Server.js.map +1 -1
- package/dist/server/Worker.d.ts +22 -22
- package/dist/server/Worker.js +36 -36
- package/dist/server/Worker.js.map +1 -1
- package/dist/server/WorkerStarter.js +2 -2
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +11 -0
- package/dist/source-separation/MDXNetSourceSeparation.js +161 -0
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -0
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js +7 -7
- package/dist/subtitles/Subtitles.d.ts +10 -0
- package/dist/subtitles/Subtitles.js +2 -2
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/dist/synthesis/AwsPollyTTS.d.ts +1 -1
- package/dist/synthesis/AwsPollyTTS.js +12 -12
- package/dist/synthesis/AzureCognitiveServicesTTS.js +7 -7
- package/dist/synthesis/CoquiServerTTS.js +10 -10
- package/dist/synthesis/CoquiServerTTS.js.map +1 -1
- package/dist/synthesis/ElevenlabsTTS.d.ts +23 -0
- package/dist/synthesis/ElevenlabsTTS.js +103 -0
- package/dist/synthesis/ElevenlabsTTS.js.map +1 -0
- package/dist/synthesis/EspeakTTS.d.ts +6 -5
- package/dist/synthesis/EspeakTTS.js +81 -69
- package/dist/synthesis/EspeakTTS.js.map +1 -1
- package/dist/synthesis/FliteTTS.d.ts +3 -3
- package/dist/synthesis/FliteTTS.js +154 -154
- package/dist/synthesis/FliteTTS.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.d.ts +3 -3
- package/dist/synthesis/GoogleCloudTTS.js +17 -17
- package/dist/synthesis/GoogleCloudTTS.js.map +1 -1
- package/dist/synthesis/GoogleTranslateTTS.d.ts +1 -1
- package/dist/synthesis/GoogleTranslateTTS.js +103 -103
- package/dist/synthesis/MicrosoftEdgeTTS.d.ts +2 -2
- package/dist/synthesis/MicrosoftEdgeTTS.js +74 -74
- package/dist/synthesis/OpenAICloudTTS.d.ts +13 -0
- package/dist/synthesis/OpenAICloudTTS.js +169 -0
- package/dist/synthesis/OpenAICloudTTS.js.map +1 -0
- package/dist/synthesis/SamTTS.js +3 -3
- package/dist/synthesis/SapiTTS.d.ts +3 -3
- package/dist/synthesis/SapiTTS.js +26 -26
- package/dist/synthesis/StreamlabsPollyTTS.d.ts +2 -2
- package/dist/synthesis/StreamlabsPollyTTS.js +27 -27
- package/dist/synthesis/SvoxPicoTTS.d.ts +2 -2
- package/dist/synthesis/SvoxPicoTTS.js +65 -65
- package/dist/synthesis/SvoxPicoTTS.js.map +1 -1
- package/dist/synthesis/VitsTTS.d.ts +3 -3
- package/dist/synthesis/VitsTTS.js +378 -378
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/tests/Test.js +2 -2
- package/dist/utilities/Compression.d.ts +5 -0
- package/dist/utilities/Compression.js +29 -13
- package/dist/utilities/Compression.js.map +1 -1
- package/dist/utilities/FileDownloader.d.ts +1 -1
- package/dist/utilities/FileDownloader.js +16 -16
- package/dist/utilities/FileSystem.js +7 -7
- package/dist/utilities/Locale.d.ts +7 -7
- package/dist/utilities/Locale.js +15 -15
- package/dist/utilities/Logger.js +3 -3
- package/dist/utilities/ObjectUtilities.js +19 -19
- package/dist/utilities/OpenPromise.js +2 -2
- package/dist/utilities/OpenPromise.js.map +1 -1
- package/dist/utilities/PackageManager.js +31 -0
- package/dist/utilities/PackageManager.js.map +1 -1
- package/dist/utilities/PathUtilities.js +8 -8
- package/dist/utilities/RandomGenerator.js +2 -2
- package/dist/utilities/SmoothEstimator.d.ts +8 -0
- package/dist/utilities/SmoothEstimator.js +25 -0
- package/dist/utilities/SmoothEstimator.js.map +1 -0
- package/dist/utilities/TarballMaker.js +8 -8
- package/dist/utilities/Timeline.d.ts +3 -2
- package/dist/utilities/Timeline.js +11 -11
- package/dist/utilities/Timeline.js.map +1 -1
- package/dist/utilities/Timer.js +4 -4
- package/dist/utilities/Utilities.d.ts +4 -0
- package/dist/utilities/Utilities.js +38 -15
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/utilities/WasmMemoryManager.js +7 -7
- package/dist/utilities/WebReader.js +23 -23
- package/dist/utilities/WikipediaReader.js +2 -2
- package/dist/voice-activity-detection/AdaptiveGateVAD.d.ts +28 -0
- package/dist/voice-activity-detection/AdaptiveGateVAD.js +138 -0
- package/dist/voice-activity-detection/AdaptiveGateVAD.js.map +1 -0
- package/dist/voice-activity-detection/SileroVAD.d.ts +1 -1
- package/dist/voice-activity-detection/SileroVAD.js +5 -5
- package/dist/voice-activity-detection/SileroVAD.js.map +1 -1
- package/dist/voice-activity-detection/WebRtcVAD.d.ts +1 -1
- package/dist/voice-activity-detection/WebRtcVAD.js +4 -4
- package/docs/API.md +29 -11
- package/docs/CLI.md +31 -7
- package/docs/Contributing.md +38 -0
- package/docs/Development.md +93 -19
- package/docs/Engines.md +28 -16
- package/docs/Licenses.md +4 -1
- package/docs/Options.md +158 -78
- package/docs/Releases.md +262 -0
- package/docs/Server.md +7 -7
- package/docs/Tasklist.md +95 -76
- package/docs/Technical.md +4 -4
- package/package.json +13 -14
- package/src/alignment/DTWMfccSequenceAlignment.ts +9 -9
- package/src/alignment/DTWSequenceAlignment.ts +2 -2
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +3 -3
- package/src/alignment/LevenshteinSequenceAlignment.ts +2 -2
- package/src/alignment/SpeechAlignment.ts +204 -119
- package/src/api/API.ts +14 -13
- package/src/api/APIOptions.ts +12 -11
- package/src/api/Alignment.ts +147 -90
- package/src/api/Common.ts +1 -1
- package/src/api/Denoising.ts +28 -28
- package/src/api/LanguageDetection.ts +135 -48
- package/src/api/Recognition.ts +198 -59
- package/src/api/SourceSeparation.ts +99 -0
- package/src/api/Synthesis.ts +217 -181
- package/src/api/Translation.ts +193 -40
- package/src/api/Vad.ts +110 -41
- package/src/audio/AudioBufferConversion.ts +4 -4
- package/src/audio/AudioPlayer.ts +27 -27
- package/src/audio/AudioRecorder.ts +5 -5
- package/src/audio/AudioUtilities.ts +107 -24
- package/src/cli/CLI.ts +313 -164
- package/src/cli/CLIConfigFile.ts +8 -8
- package/src/cli/CLILauncher.ts +6 -6
- package/src/cli/CLIOptionsSchema.ts +2 -2
- package/src/cli/CLIParser.ts +5 -5
- package/src/cli/CLIStarter.ts +4 -4
- package/src/codecs/FFMpegTranscoder.ts +38 -38
- package/src/codecs/TIMITCodec.ts +5 -5
- package/src/codecs/WaveCodec.ts +22 -22
- package/src/denoising/RNNoise.ts +9 -9
- package/src/dsp/BiquadFilter.ts +19 -11
- package/src/dsp/DecayingPeakEstimator.ts +35 -0
- package/src/dsp/FFT.ts +103 -35
- package/src/dsp/KWeightingFilter.ts +43 -0
- package/src/dsp/LoudnessEstimator.ts +74 -0
- package/src/dsp/MFCC.ts +15 -15
- package/src/dsp/MelSpectogram.ts +7 -7
- package/src/dsp/Rubberband.ts +38 -38
- package/src/dsp/Sonic.ts +4 -4
- package/src/dsp/SpeexResampler.ts +2 -2
- package/src/math/VectorMath.ts +42 -33
- package/src/nlp/ChineseSegmentation.ts +3 -3
- package/src/nlp/CompromiseNLP.ts +3 -3
- package/src/nlp/EspeakPhonemizer.ts +30 -30
- package/src/nlp/IPA.ts +20 -20
- package/src/nlp/JapaneseSegmentation.ts +6 -6
- package/src/nlp/Lexicon.ts +8 -8
- package/src/nlp/Segmentation.ts +23 -14
- package/src/nlp/TextNormalizer.ts +16 -16
- package/src/recognition/AmazonTranscribeSTT.ts +16 -17
- package/src/recognition/AzureCognitiveServicesSTT.ts +8 -6
- package/src/recognition/GoogleCloudSTT.ts +21 -21
- package/src/recognition/OpenAICloudSTT.ts +142 -0
- package/src/recognition/SileroSTT.ts +26 -26
- package/src/recognition/VoskSTT.ts +10 -10
- package/src/recognition/WhisperCppSTT.ts +555 -0
- package/src/recognition/WhisperSTT.ts +760 -507
- package/src/server/Client.ts +23 -23
- package/src/server/Server.ts +9 -9
- package/src/server/Worker.ts +53 -53
- package/src/server/WorkerStarter.ts +2 -2
- package/src/source-separation/MDXNetSourceSeparation.ts +228 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +8 -8
- package/src/subtitles/Subtitles.ts +3 -3
- package/src/synthesis/AwsPollyTTS.ts +14 -14
- package/src/synthesis/AzureCognitiveServicesTTS.ts +10 -10
- package/src/synthesis/CoquiServerTTS.ts +10 -10
- package/src/synthesis/ElevenlabsTTS.ts +137 -0
- package/src/synthesis/EspeakTTS.ts +90 -71
- package/src/synthesis/FliteTTS.ts +157 -157
- package/src/synthesis/GoogleCloudTTS.ts +19 -19
- package/src/synthesis/GoogleTranslateTTS.ts +104 -104
- package/src/synthesis/MicrosoftEdgeTTS.ts +80 -80
- package/src/synthesis/OpenAICloudTTS.ts +196 -0
- package/src/synthesis/SamTTS.ts +3 -3
- package/src/synthesis/SapiTTS.ts +29 -29
- package/src/synthesis/StreamlabsPollyTTS.ts +29 -29
- package/src/synthesis/SvoxPicoTTS.ts +67 -67
- package/src/synthesis/VitsTTS.ts +380 -380
- package/src/tests/Test.ts +4 -4
- package/src/utilities/Compression.ts +34 -13
- package/src/utilities/FileDownloader.ts +19 -19
- package/src/utilities/FileSystem.ts +7 -7
- package/src/utilities/Locale.ts +22 -22
- package/src/utilities/Logger.ts +4 -4
- package/src/utilities/ObjectUtilities.ts +19 -19
- package/src/utilities/OpenPromise.ts +2 -2
- package/src/utilities/PackageManager.ts +40 -0
- package/src/utilities/PathUtilities.ts +8 -8
- package/src/utilities/RandomGenerator.ts +3 -3
- package/src/utilities/SmoothEstimator.ts +35 -0
- package/src/utilities/TarballMaker.ts +9 -9
- package/src/utilities/Timeline.ts +15 -13
- package/src/utilities/Timer.ts +4 -4
- package/src/utilities/Utilities.ts +49 -15
- package/src/utilities/WasmMemoryManager.ts +7 -7
- package/src/utilities/WebReader.ts +23 -23
- package/src/utilities/WikipediaReader.ts +2 -2
- package/src/voice-activity-detection/AdaptiveGateVAD.ts +202 -0
- package/src/voice-activity-detection/SileroVAD.ts +5 -5
- package/src/voice-activity-detection/WebRtcVAD.ts +5 -5
- package/dist/synthesis/ElevenLabsTTS.d.ts +0 -8
- package/dist/synthesis/ElevenLabsTTS.js +0 -82
- package/dist/synthesis/ElevenLabsTTS.js.map +0 -1
- package/src/synthesis/ElevenLabsTTS.ts +0 -104
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
import Onnx from 'onnxruntime-node'
|
|
2
|
+
import { RawAudio } from '../audio/AudioUtilities.js';
|
|
3
|
+
import { binBufferToComplex, complexToBinBuffer, getWindowWeights, stftr, stiftr } from '../dsp/FFT.js';
|
|
4
|
+
import { ComplexNumber } from '../math/VectorMath.js';
|
|
5
|
+
import { logToStderr } from '../utilities/Utilities.js';
|
|
6
|
+
import { Logger } from '../utilities/Logger.js';
|
|
7
|
+
|
|
8
|
+
const log = logToStderr
|
|
9
|
+
|
|
10
|
+
export async function isolate(rawAudio: RawAudio, modelFilePath: string) {
|
|
11
|
+
const model = new MDXNet(modelFilePath)
|
|
12
|
+
|
|
13
|
+
return model.processAudio(rawAudio)
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export class MDXNet {
|
|
17
|
+
session?: Onnx.InferenceSession
|
|
18
|
+
|
|
19
|
+
constructor(public readonly modelFilePath: string) {
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
async processAudio(rawAudio: RawAudio) {
|
|
23
|
+
if (rawAudio.audioChannels.length != 2) {
|
|
24
|
+
throw new Error(`Input audio must be stereo`)
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
if (rawAudio.sampleRate != 44100) {
|
|
28
|
+
throw new Error(`Input audio must have a 44100 Hz sampling rate`)
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
if (!this.session) {
|
|
32
|
+
await this.initializeSession(this.modelFilePath)
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const logger = new Logger()
|
|
36
|
+
|
|
37
|
+
const session = this.session!
|
|
38
|
+
|
|
39
|
+
const sampleRate = rawAudio.sampleRate
|
|
40
|
+
const fftSize = 6144
|
|
41
|
+
const fftCount = 2048
|
|
42
|
+
const fftWindowSize = fftSize
|
|
43
|
+
const fftHopSize = 1024
|
|
44
|
+
|
|
45
|
+
const segmentSize = 256
|
|
46
|
+
const segmentHopSize = 240
|
|
47
|
+
|
|
48
|
+
const sampleCount = rawAudio.audioChannels[0].length
|
|
49
|
+
|
|
50
|
+
logger.start('Compute STFT of full waveform')
|
|
51
|
+
|
|
52
|
+
const fftFramesLeft = await stftr(rawAudio.audioChannels[0], fftSize, fftWindowSize, fftHopSize, 'hann')
|
|
53
|
+
const fftFramesRight = await stftr(rawAudio.audioChannels[1], fftSize, fftWindowSize, fftHopSize, 'hann')
|
|
54
|
+
|
|
55
|
+
const fftFramesLeftComplex = fftFramesLeft.map(frame => binBufferToComplex(frame).slice(0, fftCount))
|
|
56
|
+
const fftFramesRightComplex = fftFramesRight.map(frame => binBufferToComplex(frame).slice(0, fftCount))
|
|
57
|
+
|
|
58
|
+
const audioForSegments: Float32Array[][] = []
|
|
59
|
+
|
|
60
|
+
for (let segmentOffset = 0; segmentOffset < fftFramesLeft.length; segmentOffset += segmentHopSize) {
|
|
61
|
+
const timePosition = segmentOffset * (fftHopSize / sampleRate)
|
|
62
|
+
|
|
63
|
+
logger.start(`Process segment at time position ${timePosition.toFixed(2)}`)
|
|
64
|
+
|
|
65
|
+
const fftFramesLeftComplexForSegment = fftFramesLeftComplex.slice(segmentOffset, segmentOffset + segmentSize)
|
|
66
|
+
const fftFramesRightComplexForSegment = fftFramesRightComplex.slice(segmentOffset, segmentOffset + segmentSize)
|
|
67
|
+
|
|
68
|
+
const segmentLength = fftFramesLeftComplexForSegment.length
|
|
69
|
+
|
|
70
|
+
const flattenedInputTensor = new Float32Array(1 * 4 * fftCount * segmentSize)
|
|
71
|
+
|
|
72
|
+
{
|
|
73
|
+
let writePosition = 0
|
|
74
|
+
|
|
75
|
+
for (let tensorChannelIndex = 0; tensorChannelIndex < 4; tensorChannelIndex++) {
|
|
76
|
+
for (let binIndex = 0; binIndex < fftCount; binIndex++) {
|
|
77
|
+
for (let frameIndex = 0; frameIndex < segmentSize; frameIndex++) {
|
|
78
|
+
let value = 0
|
|
79
|
+
|
|
80
|
+
if (frameIndex < segmentLength && binIndex >= 0) {
|
|
81
|
+
let frame: ComplexNumber[]
|
|
82
|
+
|
|
83
|
+
if (tensorChannelIndex < 2) {
|
|
84
|
+
frame = fftFramesLeftComplexForSegment[frameIndex]
|
|
85
|
+
} else {
|
|
86
|
+
frame = fftFramesRightComplexForSegment[frameIndex]
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
const bin = frame[binIndex]
|
|
90
|
+
|
|
91
|
+
if (tensorChannelIndex % 2 === 0) {
|
|
92
|
+
value = bin.real
|
|
93
|
+
} else {
|
|
94
|
+
value = bin.imaginary
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
flattenedInputTensor[writePosition++] = value
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const inputTensor = new Onnx.Tensor('float32', flattenedInputTensor, [1, 4, 2048, 256])
|
|
105
|
+
|
|
106
|
+
const { output: outputTensor } = await session.run({ input: inputTensor })
|
|
107
|
+
|
|
108
|
+
const flattenedOutputTensor = outputTensor.data as Float32Array
|
|
109
|
+
|
|
110
|
+
const outputChannelComplexFrames: ComplexNumber[][][] = []
|
|
111
|
+
|
|
112
|
+
{
|
|
113
|
+
for (let outChannelIndex = 0; outChannelIndex < 2; outChannelIndex++) {
|
|
114
|
+
const framesForChannel: ComplexNumber[][] = []
|
|
115
|
+
|
|
116
|
+
for (let frameIndex = 0; frameIndex < 256; frameIndex++) {
|
|
117
|
+
const frame: ComplexNumber[] = []
|
|
118
|
+
|
|
119
|
+
for (let binIndex = 0; binIndex < fftSize; binIndex++) {
|
|
120
|
+
frame.push({ real: 0, imaginary: 0 })
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
framesForChannel.push(frame)
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
outputChannelComplexFrames.push(framesForChannel)
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
let readPosition = 0
|
|
130
|
+
|
|
131
|
+
for (let tensorChannelIndex = 0; tensorChannelIndex < 4; tensorChannelIndex++) {
|
|
132
|
+
const outChannelIndex = tensorChannelIndex < 2 ? 0 : 1
|
|
133
|
+
|
|
134
|
+
for (let binIndex = 0; binIndex < 2048; binIndex++) {
|
|
135
|
+
for (let frameIndex = 0; frameIndex < 256; frameIndex++) {
|
|
136
|
+
const bin = outputChannelComplexFrames[outChannelIndex][frameIndex][binIndex]
|
|
137
|
+
|
|
138
|
+
if (tensorChannelIndex % 2 === 0) {
|
|
139
|
+
bin.real = flattenedOutputTensor[readPosition++]
|
|
140
|
+
} else {
|
|
141
|
+
bin.imaginary = flattenedOutputTensor[readPosition++]
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const outputAudioChannels: Float32Array[] = []
|
|
149
|
+
|
|
150
|
+
//logger.start(`Compute inverse STFT for segment`)
|
|
151
|
+
for (let channelIndex = 0; channelIndex < 2; channelIndex++) {
|
|
152
|
+
let outputChannelFlattenedFrames = outputChannelComplexFrames[channelIndex]
|
|
153
|
+
.map(frame => complexToBinBuffer(frame).map(value => value / fftSize))
|
|
154
|
+
|
|
155
|
+
const samples = await stiftr(
|
|
156
|
+
outputChannelFlattenedFrames,
|
|
157
|
+
fftSize,
|
|
158
|
+
fftWindowSize,
|
|
159
|
+
fftHopSize,
|
|
160
|
+
'hann')
|
|
161
|
+
|
|
162
|
+
outputAudioChannels.push(samples)
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
//logger.log(`Reconstructed waveform peak: ${getAudioPeakDecibels(outputAudioChannels).toFixed(3)}dB`)
|
|
166
|
+
//await playAudioSamples({ audioChannels: outputAudioChannels, sampleRate })
|
|
167
|
+
|
|
168
|
+
audioForSegments.push(outputAudioChannels)
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
// Join segments using overlapping Hann windows
|
|
172
|
+
logger.start(`Join segments`)
|
|
173
|
+
const concatenatedAudioChannels = [new Float32Array(sampleCount), new Float32Array(sampleCount)]
|
|
174
|
+
|
|
175
|
+
{
|
|
176
|
+
const segmentCount = audioForSegments.length
|
|
177
|
+
|
|
178
|
+
const segmentSampleCount = audioForSegments[0][0].length
|
|
179
|
+
|
|
180
|
+
const windowWeights = getWindowWeights('hann', segmentSampleCount)
|
|
181
|
+
|
|
182
|
+
const sumOfWeightsForSample = new Float32Array(sampleCount)
|
|
183
|
+
|
|
184
|
+
for (let segmentIndex = 0; segmentIndex < segmentCount; segmentIndex++) {
|
|
185
|
+
const segmentStartFrameIndex = segmentIndex * segmentHopSize
|
|
186
|
+
const segmentStartSampleIndex = segmentStartFrameIndex * fftHopSize
|
|
187
|
+
|
|
188
|
+
const segmentSamples = audioForSegments[segmentIndex]
|
|
189
|
+
|
|
190
|
+
for (let segmentSampleOffset = 0; segmentSampleOffset < segmentSampleCount; segmentSampleOffset++) {
|
|
191
|
+
const sampleIndex = segmentStartSampleIndex + segmentSampleOffset
|
|
192
|
+
|
|
193
|
+
if (sampleIndex >= sampleCount) {
|
|
194
|
+
break
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
const weight = windowWeights[segmentSampleOffset]
|
|
198
|
+
|
|
199
|
+
for (let channelIndex = 0; channelIndex < 2; channelIndex++) {
|
|
200
|
+
concatenatedAudioChannels[channelIndex][sampleIndex] += segmentSamples[channelIndex][segmentSampleOffset] * weight
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
sumOfWeightsForSample[sampleIndex] += weight
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
for (let sampleIndex = 0; sampleIndex < sampleCount; sampleIndex++) {
|
|
208
|
+
for (let channelIndex = 0; channelIndex < 2; channelIndex++) {
|
|
209
|
+
concatenatedAudioChannels[channelIndex][sampleIndex] /= sumOfWeightsForSample[sampleIndex] + 1e-8
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
const isolatedRawAudio: RawAudio = { audioChannels: concatenatedAudioChannels, sampleRate }
|
|
215
|
+
|
|
216
|
+
logger.end()
|
|
217
|
+
|
|
218
|
+
return isolatedRawAudio
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
private async initializeSession(modelPath: string) {
|
|
222
|
+
const onnxOptions: Onnx.InferenceSession.SessionOptions = {
|
|
223
|
+
logSeverityLevel: 3
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
this.session = await Onnx.InferenceSession.create(modelPath, onnxOptions)
|
|
227
|
+
}
|
|
228
|
+
}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import Onnx from 'onnxruntime-node'
|
|
2
2
|
import { softmax } from '../math/VectorMath.js'
|
|
3
|
-
import { Logger } from
|
|
4
|
-
import { RawAudio } from
|
|
3
|
+
import { Logger } from '../utilities/Logger.js'
|
|
4
|
+
import { RawAudio } from '../audio/AudioUtilities.js'
|
|
5
5
|
import { readAndParseJsonFile } from '../utilities/FileSystem.js'
|
|
6
6
|
import { detectSpeechLanguageByParts, type LanguageDetectionResults } from '../api/LanguageDetection.js'
|
|
7
7
|
import { languageCodeToName } from '../utilities/Locale.js'
|
|
@@ -41,7 +41,7 @@ export class SileroLanguageDetection {
|
|
|
41
41
|
|
|
42
42
|
async initialize() {
|
|
43
43
|
const logger = new Logger()
|
|
44
|
-
logger.start(
|
|
44
|
+
logger.start('Initialize ONNX inference session')
|
|
45
45
|
|
|
46
46
|
this.languageDictionary = await readAndParseJsonFile(this.languageDictionaryPath)
|
|
47
47
|
this.languageGroupDictionary = await readAndParseJsonFile(this.languageGroupDictionaryPath)
|
|
@@ -58,7 +58,7 @@ export class SileroLanguageDetection {
|
|
|
58
58
|
async detectLanguage(rawAudio: RawAudio) {
|
|
59
59
|
const logger = new Logger()
|
|
60
60
|
|
|
61
|
-
logger.start(
|
|
61
|
+
logger.start('Detect language with Silero')
|
|
62
62
|
|
|
63
63
|
const audioSamples = rawAudio.audioChannels[0]
|
|
64
64
|
|
|
@@ -68,10 +68,10 @@ export class SileroLanguageDetection {
|
|
|
68
68
|
|
|
69
69
|
const results = await this.session!.run(inputs)
|
|
70
70
|
|
|
71
|
-
logger.start(
|
|
71
|
+
logger.start('Parse model results')
|
|
72
72
|
|
|
73
|
-
const languageLogits = results[
|
|
74
|
-
const languageGroupLogits = results[
|
|
73
|
+
const languageLogits = results['output'].data
|
|
74
|
+
const languageGroupLogits = results['2038'].data
|
|
75
75
|
|
|
76
76
|
const languageProbabilities = softmax(languageLogits as any)
|
|
77
77
|
const languageGroupProbabilities = softmax(languageGroupLogits as any)
|
|
@@ -80,7 +80,7 @@ export class SileroLanguageDetection {
|
|
|
80
80
|
|
|
81
81
|
for (let i = 0; i < languageProbabilities.length; i++) {
|
|
82
82
|
const languageString = this.languageDictionary[i]
|
|
83
|
-
const languageCode = languageString.replace(/,.*$/,
|
|
83
|
+
const languageCode = languageString.replace(/,.*$/, '')
|
|
84
84
|
|
|
85
85
|
languageResults.push({
|
|
86
86
|
language: languageCode,
|
|
@@ -428,11 +428,11 @@ function getCuesFromTimeline_IsolateLines(timeline: Timeline, config: SubtitlesC
|
|
|
428
428
|
|
|
429
429
|
addCuesFrom(timeline)
|
|
430
430
|
addCueFromCurrentWords() // Add any remaining words
|
|
431
|
-
|
|
431
|
+
|
|
432
432
|
return cues
|
|
433
433
|
}
|
|
434
434
|
|
|
435
|
-
function tryParseTimeRangePatternWithHours(line: string) {
|
|
435
|
+
export function tryParseTimeRangePatternWithHours(line: string) {
|
|
436
436
|
const timeRangePatternWithHours = /^(\d+)\:(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)\:(\d+)[\.,](\d+)/
|
|
437
437
|
const match = timeRangePatternWithHours.exec(line)
|
|
438
438
|
|
|
@@ -456,7 +456,7 @@ function tryParseTimeRangePatternWithHours(line: string) {
|
|
|
456
456
|
return { startTime, endTime, succeeded: true }
|
|
457
457
|
}
|
|
458
458
|
|
|
459
|
-
function tryParseTimeRangePatternWithoutHours(line: string) {
|
|
459
|
+
export function tryParseTimeRangePatternWithoutHours(line: string) {
|
|
460
460
|
const timeRangePatternWithHours = /^(\d+)\:(\d+)[\.,](\d+)[ ]*-->[ ]*(\d+)\:(\d+)[\.,](\d+)/
|
|
461
461
|
const match = timeRangePatternWithHours.exec(line)
|
|
462
462
|
|
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
import type { LanguageCode, SynthesizeSpeechCommandInput, VoiceId } from
|
|
2
|
-
import { IncomingMessage } from
|
|
3
|
-
import * as FFMpegTranscoder from
|
|
4
|
-
import { Logger } from
|
|
1
|
+
import type { LanguageCode, SynthesizeSpeechCommandInput, VoiceId } from '@aws-sdk/client-polly'
|
|
2
|
+
import { IncomingMessage } from 'http'
|
|
3
|
+
import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
|
|
4
|
+
import { Logger } from '../utilities/Logger.js'
|
|
5
5
|
|
|
6
|
-
import { readBinaryIncomingMessage } from
|
|
6
|
+
import { readBinaryIncomingMessage } from '../utilities/Utilities.js'
|
|
7
7
|
|
|
8
|
-
export async function synthesize(text: string, language: string | undefined, voice: string, region: string, accessKeyId: string, secretAccessKey: string, engine:
|
|
8
|
+
export async function synthesize(text: string, language: string | undefined, voice: string, region: string, accessKeyId: string, secretAccessKey: string, engine: 'standard' | 'neural' = 'standard', ssmlEnabled = false, lexiconNames?: string[]) {
|
|
9
9
|
const logger = new Logger()
|
|
10
|
-
logger.start(
|
|
10
|
+
logger.start('Load AWS SDK client module')
|
|
11
11
|
|
|
12
|
-
const polly = await import(
|
|
12
|
+
const polly = await import('@aws-sdk/client-polly')
|
|
13
13
|
|
|
14
14
|
const pollyClient = new polly.PollyClient({
|
|
15
15
|
region,
|
|
@@ -28,12 +28,12 @@ export async function synthesize(text: string, language: string | undefined, voi
|
|
|
28
28
|
Text: text,
|
|
29
29
|
LexiconNames: lexiconNames,
|
|
30
30
|
|
|
31
|
-
TextType: ssmlEnabled ?
|
|
31
|
+
TextType: ssmlEnabled ? 'ssml' : 'text',
|
|
32
32
|
|
|
33
|
-
OutputFormat:
|
|
33
|
+
OutputFormat: 'mp3',
|
|
34
34
|
}
|
|
35
35
|
|
|
36
|
-
logger.start(
|
|
36
|
+
logger.start('Request synthesis from AWS Polly')
|
|
37
37
|
|
|
38
38
|
const command = new polly.SynthesizeSpeechCommand(params)
|
|
39
39
|
|
|
@@ -52,11 +52,11 @@ export async function synthesize(text: string, language: string | undefined, voi
|
|
|
52
52
|
|
|
53
53
|
export async function getVoiceList(region: string, accessKeyId: string, secretAccessKey: string) {
|
|
54
54
|
const logger = new Logger()
|
|
55
|
-
logger.start(
|
|
55
|
+
logger.start('Load AWS SDK client module')
|
|
56
56
|
|
|
57
|
-
const polly = await import(
|
|
57
|
+
const polly = await import('@aws-sdk/client-polly')
|
|
58
58
|
|
|
59
|
-
logger.start(
|
|
59
|
+
logger.start('Request voice list from AWS Polly')
|
|
60
60
|
|
|
61
61
|
const pollyClient = new polly.PollyClient({
|
|
62
62
|
region,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import * as SpeechSDK from 'microsoft-cognitiveservices-speech-sdk'
|
|
2
2
|
|
|
3
|
-
import * as FFMpegTranscoder from
|
|
3
|
+
import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
|
|
4
4
|
|
|
5
5
|
import { escape } from 'html-escaper'
|
|
6
6
|
|
|
@@ -12,15 +12,15 @@ export async function synthesize(
|
|
|
12
12
|
text: string,
|
|
13
13
|
subscriptionKey: string,
|
|
14
14
|
serviceRegion: string,
|
|
15
|
-
languageCode =
|
|
16
|
-
voice =
|
|
15
|
+
languageCode = 'en-US',
|
|
16
|
+
voice = 'Microsoft Server Speech Text to Speech Voice (en-US, AvaNeural)',
|
|
17
17
|
ssmlEnabled = false,
|
|
18
|
-
ssmlPitchString =
|
|
19
|
-
ssmlRateString =
|
|
18
|
+
ssmlPitchString = '+0Hz',
|
|
19
|
+
ssmlRateString = '+0%') {
|
|
20
20
|
|
|
21
21
|
return new Promise<{ rawAudio: RawAudio, timeline: Timeline }>((resolve, reject) => {
|
|
22
22
|
const logger = new Logger()
|
|
23
|
-
logger.start(
|
|
23
|
+
logger.start('Request synthesis from Azure Cognitive Services')
|
|
24
24
|
|
|
25
25
|
const speechConfig = SpeechSDK.SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
|
|
26
26
|
|
|
@@ -72,7 +72,7 @@ export async function synthesize(
|
|
|
72
72
|
|
|
73
73
|
const rawAudio = await FFMpegTranscoder.decodeToChannels(encodedAudio, 24000, 1)
|
|
74
74
|
|
|
75
|
-
logger.start(
|
|
75
|
+
logger.start('Convert boundary events to a timeline')
|
|
76
76
|
|
|
77
77
|
const timeline = boundaryEventsToTimeline(events, getRawAudioDuration(rawAudio))
|
|
78
78
|
|
|
@@ -85,7 +85,7 @@ export async function synthesize(
|
|
|
85
85
|
reject(error)
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
-
if (!ssmlEnabled && ssmlPitchString !=
|
|
88
|
+
if (!ssmlEnabled && ssmlPitchString != '+0%' || ssmlRateString != '+0Hz') {
|
|
89
89
|
ssmlEnabled = true
|
|
90
90
|
text = escape(text)
|
|
91
91
|
}
|
|
@@ -123,7 +123,7 @@ export function boundaryEventsToTimeline(events: any[], totalDuration: number) {
|
|
|
123
123
|
for (const event of events) {
|
|
124
124
|
const boundaryType = event.boundaryType != null ? event.boundaryType : event.Type
|
|
125
125
|
|
|
126
|
-
if (boundaryType !=
|
|
126
|
+
if (boundaryType != 'WordBoundary') {
|
|
127
127
|
continue
|
|
128
128
|
}
|
|
129
129
|
|
|
@@ -135,7 +135,7 @@ export function boundaryEventsToTimeline(events: any[], totalDuration: number) {
|
|
|
135
135
|
const endTime = (offset + duration) / 10000000
|
|
136
136
|
|
|
137
137
|
timeline.push({
|
|
138
|
-
type:
|
|
138
|
+
type: 'word',
|
|
139
139
|
text,
|
|
140
140
|
startTime,
|
|
141
141
|
endTime
|
|
@@ -1,27 +1,27 @@
|
|
|
1
|
-
import { request } from
|
|
2
|
-
import {
|
|
3
|
-
import { Logger } from
|
|
4
|
-
import { logToStderr } from
|
|
1
|
+
import { request } from 'gaxios'
|
|
2
|
+
import { decodeWaveToRawAudio } from '../audio/AudioUtilities.js'
|
|
3
|
+
import { Logger } from '../utilities/Logger.js'
|
|
4
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
5
5
|
const log = logToStderr
|
|
6
6
|
|
|
7
|
-
export async function synthesize(text: string, speakerId: string | null, serverURL =
|
|
7
|
+
export async function synthesize(text: string, speakerId: string | null, serverURL = 'http://[::1]:5002') {
|
|
8
8
|
const logger = new Logger()
|
|
9
|
-
logger.start(
|
|
9
|
+
logger.start('Request synthesis from Coqui Server')
|
|
10
10
|
|
|
11
11
|
const response = await request<Buffer>({
|
|
12
12
|
url: `${serverURL}/api/tts`,
|
|
13
13
|
|
|
14
14
|
params: {
|
|
15
|
-
|
|
16
|
-
|
|
15
|
+
'text': text,
|
|
16
|
+
'speaker_id': speakerId
|
|
17
17
|
},
|
|
18
18
|
|
|
19
|
-
responseType:
|
|
19
|
+
responseType: 'arraybuffer'
|
|
20
20
|
})
|
|
21
21
|
|
|
22
22
|
const waveData = Buffer.from(response.data)
|
|
23
23
|
|
|
24
|
-
const rawAudio =
|
|
24
|
+
const rawAudio = decodeWaveToRawAudio(waveData).rawAudio
|
|
25
25
|
|
|
26
26
|
logger.end()
|
|
27
27
|
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
import { GaxiosResponse, request } from 'gaxios'
|
|
2
|
+
import { SynthesisVoice, VoiceGender } from '../api/API.js'
|
|
3
|
+
import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
|
|
4
|
+
import { Logger } from '../utilities/Logger.js'
|
|
5
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
6
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
7
|
+
|
|
8
|
+
const log = logToStderr
|
|
9
|
+
|
|
10
|
+
export async function synthesize(text: string, voiceId: string, modelId: string, options: ElevenlabsTTSOptions) {
|
|
11
|
+
const logger = new Logger()
|
|
12
|
+
logger.start('Request synthesis from ElevenLabs')
|
|
13
|
+
|
|
14
|
+
options = extendDeep(defaultElevenlabsTTSOptions, options)
|
|
15
|
+
|
|
16
|
+
let response: GaxiosResponse<any>
|
|
17
|
+
|
|
18
|
+
try {
|
|
19
|
+
response = await request<any>({
|
|
20
|
+
url: `https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`,
|
|
21
|
+
|
|
22
|
+
method: 'POST',
|
|
23
|
+
|
|
24
|
+
headers: {
|
|
25
|
+
'accept': 'audio/mpeg',
|
|
26
|
+
'xi-api-key': options.apiKey,
|
|
27
|
+
},
|
|
28
|
+
|
|
29
|
+
data: {
|
|
30
|
+
text,
|
|
31
|
+
|
|
32
|
+
model_id: modelId,
|
|
33
|
+
|
|
34
|
+
voice_setting: {
|
|
35
|
+
stability: options.stability,
|
|
36
|
+
similarity_boost: options.similarityBoost,
|
|
37
|
+
style: options.style,
|
|
38
|
+
use_speaker_boost: options.useSpeakerBoost
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
|
|
42
|
+
responseType: 'arraybuffer'
|
|
43
|
+
})
|
|
44
|
+
} catch (e: any) {
|
|
45
|
+
const response = e.response
|
|
46
|
+
|
|
47
|
+
if (response) {
|
|
48
|
+
logger.log(`Request failed with status code ${response.status}`)
|
|
49
|
+
|
|
50
|
+
if (response.data) {
|
|
51
|
+
logger.log(`Server responded with:`)
|
|
52
|
+
logger.log(response.data)
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
throw e
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
logger.start('Decode synthesized audio')
|
|
60
|
+
const rawAudio = await FFMpegTranscoder.decodeToChannels(Buffer.from(response.data))
|
|
61
|
+
|
|
62
|
+
logger.end()
|
|
63
|
+
|
|
64
|
+
return { rawAudio }
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export async function getVoiceList(apiKey: string) {
|
|
68
|
+
const response = await request<any>({
|
|
69
|
+
method: 'GET',
|
|
70
|
+
|
|
71
|
+
url: 'https://api.elevenlabs.io/v1/voices',
|
|
72
|
+
|
|
73
|
+
headers: {
|
|
74
|
+
'accept': 'accept: application/json',
|
|
75
|
+
'xi-api-key': apiKey
|
|
76
|
+
},
|
|
77
|
+
|
|
78
|
+
responseType: 'json'
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
const elevenlabsVoices: any[] = response.data.voices
|
|
82
|
+
|
|
83
|
+
const voices: SynthesisVoice[] = elevenlabsVoices.map(elevenlabsVoice => {
|
|
84
|
+
const modelId: string = elevenlabsVoice?.high_quality_base_model_ids?.[0] ?? 'eleven_monolingual_v1'
|
|
85
|
+
const accent: string | undefined = elevenlabsVoice?.labels?.accent
|
|
86
|
+
const gender: VoiceGender = elevenlabsVoice?.labels?.gender ?? 'unknown'
|
|
87
|
+
|
|
88
|
+
const supportedLanguages: string[] = []
|
|
89
|
+
|
|
90
|
+
if (accent) {
|
|
91
|
+
if (accent.startsWith('american')) {
|
|
92
|
+
supportedLanguages.push('en-US')
|
|
93
|
+
} else if (accent.startsWith('british')) {
|
|
94
|
+
supportedLanguages.push('en-GB')
|
|
95
|
+
} else if (accent === 'irish') {
|
|
96
|
+
supportedLanguages.push('en-IE')
|
|
97
|
+
} else if (accent == 'australian') {
|
|
98
|
+
supportedLanguages.push('en-AU')
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
if (modelId.includes('multilingual')) {
|
|
103
|
+
supportedLanguages.push('en', ...supporteMultilingualLanguages)
|
|
104
|
+
} else {
|
|
105
|
+
supportedLanguages.push('en')
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
return {
|
|
109
|
+
name: elevenlabsVoice.name,
|
|
110
|
+
languages: supportedLanguages,
|
|
111
|
+
gender,
|
|
112
|
+
|
|
113
|
+
elevenLabsVoiceId: elevenlabsVoice.voice_id,
|
|
114
|
+
elevenLabsModelId: modelId
|
|
115
|
+
}
|
|
116
|
+
})
|
|
117
|
+
|
|
118
|
+
return voices
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export interface ElevenlabsTTSOptions {
|
|
122
|
+
apiKey?: string
|
|
123
|
+
stability?: number
|
|
124
|
+
similarityBoost?: number
|
|
125
|
+
style?: number
|
|
126
|
+
useSpeakerBoost?: boolean
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
export const defaultElevenlabsTTSOptions = {
|
|
130
|
+
apiKey: undefined,
|
|
131
|
+
stability: 0.5,
|
|
132
|
+
similarityBoost: 0.5,
|
|
133
|
+
style: 0,
|
|
134
|
+
useSpeakerBoost: true
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export const supporteMultilingualLanguages = ['zh', 'ko', 'nl', 'tr', 'sv', 'id', 'tl', 'ja', 'uk', 'el', 'cs', 'fi', 'ro', 'ru', 'da', 'bg', 'ms', 'sk', 'hr', 'ar', 'ta', 'pl', 'de', 'es', 'fr', 'it', 'hi', 'pt']
|