echogarden 1.4.4 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +310 -25
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWMfccSequenceAlignment.js +5 -5
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +1 -3
- package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
- package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +4 -2
- package/dist/alignment/SemanticTextAlignment.js +336 -0
- package/dist/alignment/SemanticTextAlignment.js.map +1 -0
- package/dist/alignment/SpeechAlignment.d.ts +4 -3
- package/dist/alignment/SpeechAlignment.js +130 -39
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +7 -3
- package/dist/api/API.js +7 -2
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +4 -1
- package/dist/api/Alignment.d.ts +1 -1
- package/dist/api/Alignment.js +13 -5
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetectionCommon.d.ts +6 -0
- package/dist/api/LanguageDetectionCommon.js +2 -0
- package/dist/api/LanguageDetectionCommon.js.map +1 -0
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
- package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
- package/dist/api/SpeechLanguageDetection.js.map +1 -0
- package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
- package/dist/api/SpeechTranslation.js.map +1 -0
- package/dist/api/Synthesis.d.ts +0 -1
- package/dist/api/Synthesis.js +4 -4
- package/dist/api/TextLanguageDetection.d.ts +21 -0
- package/dist/api/TextLanguageDetection.js +67 -0
- package/dist/api/TextLanguageDetection.js.map +1 -0
- package/dist/api/TextTranslation.d.ts +25 -0
- package/dist/api/TextTranslation.js +101 -0
- package/dist/api/TextTranslation.js.map +1 -0
- package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
- package/dist/api/TimelineTranslationAlignment.js +92 -0
- package/dist/api/TimelineTranslationAlignment.js.map +1 -0
- package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
- package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
- package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
- package/dist/api/TranslationAlignment.d.ts +4 -3
- package/dist/api/TranslationAlignment.js +9 -8
- package/dist/api/TranslationAlignment.js.map +1 -1
- package/dist/api/VoiceActivityDetection.js +16 -1
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/audio/AudioBufferConversion.d.ts +0 -1
- package/dist/audio/AudioPlayer.d.ts +0 -1
- package/dist/audio/AudioPlayer.js +62 -41
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +0 -1
- package/dist/cli/CLI.d.ts +28 -7
- package/dist/cli/CLI.js +265 -37
- package/dist/cli/CLI.js.map +1 -1
- package/dist/codecs/FFMpegTranscoder.d.ts +0 -1
- package/dist/codecs/FFMpegTranscoder.js +7 -0
- package/dist/codecs/FFMpegTranscoder.js.map +1 -1
- package/dist/codecs/TIMITCodec.d.ts +0 -1
- package/dist/codecs/WaveCodec.d.ts +0 -1
- package/dist/dsp/FFT.d.ts +1 -1
- package/dist/dsp/FFT.js +6 -0
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/dsp/KWeightingFilter.js +1 -1
- package/dist/dsp/KWeightingFilter.js.map +1 -1
- package/dist/dsp/MelSpectogram.d.ts +3 -2
- package/dist/dsp/MelSpectogram.js +14 -8
- package/dist/dsp/MelSpectogram.js.map +1 -1
- package/dist/math/VectorMath.d.ts +9 -9
- package/dist/math/VectorMath.js +10 -10
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/nlp/ChineseSegmentation.js +4 -4
- package/dist/nlp/ChineseSegmentation.js.map +1 -1
- package/dist/nlp/Segmentation.d.ts +2 -2
- package/dist/nlp/Segmentation.js +20 -13
- package/dist/nlp/Segmentation.js.map +1 -1
- package/dist/recognition/OpenAICloudSTT.d.ts +2 -1
- package/dist/recognition/OpenAICloudSTT.js +30 -19
- package/dist/recognition/OpenAICloudSTT.js.map +1 -1
- package/dist/recognition/SileroSTT.d.ts +0 -1
- package/dist/recognition/WhisperCppSTT.d.ts +3 -3
- package/dist/recognition/WhisperCppSTT.js +21 -9
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.d.ts +9 -6
- package/dist/recognition/WhisperSTT.js +227 -46
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Client.d.ts +3 -4
- package/dist/server/Client.js.map +1 -1
- package/dist/server/Worker.d.ts +3 -3
- package/dist/server/Worker.js +3 -2
- package/dist/server/Worker.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +0 -1
- package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +12 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -2
- package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/subtitles/Subtitles.js +2 -2
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.js +6 -21
- package/dist/synthesis/GoogleTranslateTTS.js.map +1 -1
- package/dist/synthesis/StreamlabsPollyTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.js +30 -0
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/tests/Test.js +0 -31
- package/dist/tests/Test.js.map +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
- package/dist/text-translation/DeepLTextTranslation.d.ts +2 -0
- package/dist/text-translation/DeepLTextTranslation.js +67 -0
- package/dist/text-translation/DeepLTextTranslation.js.map +1 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.d.ts +10 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js +554 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js.map +1 -0
- package/dist/text-translation/NLLBTextTranslation.d.ts +2 -1
- package/dist/text-translation/NLLBTextTranslation.js +249 -19
- package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
- package/dist/utilities/BinaryArrayConversion.d.ts +0 -1
- package/dist/utilities/BrowserRequestHeaders.d.ts +6 -0
- package/dist/utilities/BrowserRequestHeaders.js +52 -0
- package/dist/utilities/BrowserRequestHeaders.js.map +1 -0
- package/dist/utilities/BufferFileReadStream.d.ts +20 -0
- package/dist/utilities/BufferFileReadStream.js +81 -0
- package/dist/utilities/BufferFileReadStream.js.map +1 -0
- package/dist/utilities/DynamicUint8Array.d.ts +9 -0
- package/dist/utilities/DynamicUint8Array.js +31 -0
- package/dist/utilities/DynamicUint8Array.js.map +1 -0
- package/dist/utilities/FileSystem.d.ts +0 -2
- package/dist/utilities/Hashing.d.ts +3 -10
- package/dist/utilities/Hashing.js +10 -127
- package/dist/utilities/Hashing.js.map +1 -1
- package/dist/utilities/LEB128.d.ts +15 -5
- package/dist/utilities/LEB128.js +199 -119
- package/dist/utilities/LEB128.js.map +1 -1
- package/dist/utilities/LPVarInt.d.ts +11 -0
- package/dist/utilities/LPVarInt.js +187 -0
- package/dist/utilities/LPVarInt.js.map +1 -0
- package/dist/utilities/Locale.d.ts +1 -1
- package/dist/utilities/Locale.js +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +1 -2
- package/dist/utilities/PVarInt.d.ts +4 -0
- package/dist/utilities/PVarInt.js +166 -0
- package/dist/utilities/PVarInt.js.map +1 -0
- package/dist/utilities/PackageManager.js +48 -25
- package/dist/utilities/PackageManager.js.map +1 -1
- package/dist/utilities/RandomGenerator.d.ts +3 -17
- package/dist/utilities/RandomGenerator.js +12 -81
- package/dist/utilities/RandomGenerator.js.map +1 -1
- package/dist/utilities/Timeline.d.ts +2 -0
- package/dist/utilities/Timeline.js +129 -20
- package/dist/utilities/Timeline.js.map +1 -1
- package/dist/utilities/Utilities.d.ts +1 -3
- package/dist/utilities/Utilities.js +30 -3
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/utilities/VarInt.d.ts +4 -0
- package/dist/utilities/VarInt.js +166 -0
- package/dist/utilities/VarInt.js.map +1 -0
- package/dist/utilities/VirtualFileReadStream.d.ts +20 -0
- package/dist/utilities/VirtualFileReadStream.js +79 -0
- package/dist/utilities/VirtualFileReadStream.js.map +1 -0
- package/dist/utilities/WebReader.js +7 -23
- package/dist/utilities/WebReader.js.map +1 -1
- package/dist/voice-activity-detection/SileroVAD.d.ts +0 -1
- package/docs/API.md +105 -3
- package/docs/CLI.md +51 -1
- package/docs/Engines.md +32 -3
- package/docs/Options.md +53 -12
- package/docs/Tasklist.md +1 -13
- package/package.json +20 -24
- package/src/alignment/DTWMfccSequenceAlignment.ts +5 -5
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +1 -3
- package/src/alignment/SemanticTextAlignment.ts +467 -0
- package/src/alignment/SpeechAlignment.ts +214 -56
- package/src/api/API.ts +18 -2
- package/src/api/APIOptions.ts +14 -1
- package/src/api/Alignment.ts +31 -9
- package/src/api/LanguageDetectionCommon.ts +7 -0
- package/src/api/Recognition.ts +2 -0
- package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
- package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
- package/src/api/Synthesis.ts +4 -4
- package/src/api/TextLanguageDetection.ts +116 -0
- package/src/api/TextTranslation.ts +177 -0
- package/src/api/TimelineTranslationAlignment.ts +162 -0
- package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
- package/src/api/TranslationAlignment.ts +12 -10
- package/src/api/VoiceActivityDetection.ts +24 -3
- package/src/audio/AudioPlayer.ts +2 -0
- package/src/cli/CLI.ts +376 -40
- package/src/codecs/FFMpegTranscoder.ts +6 -0
- package/src/dsp/FFT.ts +8 -2
- package/src/dsp/KWeightingFilter.ts +1 -1
- package/src/dsp/MelSpectogram.ts +17 -8
- package/src/math/VectorMath.ts +15 -15
- package/src/nlp/ChineseSegmentation.ts +6 -4
- package/src/nlp/Segmentation.ts +18 -13
- package/src/recognition/OpenAICloudSTT.ts +47 -29
- package/src/recognition/WhisperCppSTT.ts +26 -11
- package/src/recognition/WhisperSTT.ts +364 -49
- package/src/server/Client.ts +3 -2
- package/src/server/Worker.ts +3 -2
- package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
- package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
- package/src/subtitles/Subtitles.ts +2 -2
- package/src/synthesis/GoogleTranslateTTS.ts +7 -21
- package/src/synthesis/VitsTTS.ts +31 -3
- package/src/tests/Test.ts +1 -38
- package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
- package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
- package/src/text-translation/DeepLTextTranslation.ts +88 -0
- package/src/text-translation/GoogleTranslateTextTranslation.ts +667 -0
- package/src/text-translation/NLLBTextTranslation.ts +261 -21
- package/src/typings/Fillers.d.ts +25 -2
- package/src/utilities/BrowserRequestHeaders.ts +59 -0
- package/src/utilities/DynamicUint8Array.ts +39 -0
- package/src/utilities/Hashing.ts +14 -167
- package/src/utilities/LEB128.ts +273 -148
- package/src/utilities/LPVarInt.ts +292 -0
- package/src/utilities/Locale.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +1 -1
- package/src/utilities/PackageManager.ts +51 -30
- package/src/utilities/RandomGenerator.ts +12 -113
- package/src/utilities/Timeline.ts +162 -23
- package/src/utilities/Utilities.ts +40 -3
- package/src/utilities/VirtualFileReadStream.ts +109 -0
- package/src/utilities/WebReader.ts +9 -23
- package/dist/alignment/TextAlignment.js +0 -156
- package/dist/alignment/TextAlignment.js.map +0 -1
- package/dist/api/LanguageDetection.js.map +0 -1
- package/dist/api/Translation.js.map +0 -1
- package/src/alignment/TextAlignment.ts +0 -234
- /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
|
@@ -1,18 +1,21 @@
|
|
|
1
|
-
import { clip } from '../utilities/Utilities.js'
|
|
1
|
+
import { clip, splitFloat32Array } from '../utilities/Utilities.js'
|
|
2
2
|
|
|
3
3
|
import * as API from '../api/API.js'
|
|
4
4
|
|
|
5
5
|
import { computeMFCCs, extendDefaultMfccOptions, MfccOptions } from '../dsp/MFCC.js'
|
|
6
6
|
import { alignMFCC_DTW, getCostMatrixMemorySizeMB } from './DTWMfccSequenceAlignment.js'
|
|
7
7
|
import { Logger } from '../utilities/Logger.js'
|
|
8
|
-
import { Timeline, TimelineEntry } from '../utilities/Timeline.js'
|
|
9
|
-
import { downmixToMonoAndNormalize, getEndingSilentSampleCount, getRawAudioDuration, getStartingSilentSampleCount, RawAudio } from '../audio/AudioUtilities.js'
|
|
8
|
+
import { addTimeOffsetToTimeline, Timeline, TimelineEntry } from '../utilities/Timeline.js'
|
|
9
|
+
import { concatAudioSegments, downmixToMonoAndNormalize, getEmptyRawAudio, getEndingSilentSampleCount, getRawAudioDuration, getStartingSilentSampleCount, RawAudio } from '../audio/AudioUtilities.js'
|
|
10
10
|
import chalk from 'chalk'
|
|
11
11
|
import { synthesize } from '../api/API.js'
|
|
12
12
|
import { resampleAudioSpeex } from '../dsp/SpeexResampler.js'
|
|
13
13
|
import { deepClone } from '../utilities/ObjectUtilities.js'
|
|
14
|
-
import { zeroIfNaN } from '../math/VectorMath.js'
|
|
15
|
-
import { EspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
14
|
+
import { cosineDistance, euclideanDistance, zeroIfNaN } from '../math/VectorMath.js'
|
|
15
|
+
import { EspeakEvent, EspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
16
|
+
import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
|
|
17
|
+
import { loadPackage } from '../utilities/PackageManager.js'
|
|
18
|
+
import path from 'path'
|
|
16
19
|
|
|
17
20
|
export async function alignUsingDtw(
|
|
18
21
|
sourceRawAudio: RawAudio,
|
|
@@ -97,54 +100,7 @@ export async function alignUsingDtw(
|
|
|
97
100
|
|
|
98
101
|
logger.start('\nConvert path to timeline')
|
|
99
102
|
|
|
100
|
-
|
|
101
|
-
const referenceStartFrameIndex = Math.floor(timelineEntry.startTime * framesPerSecond)
|
|
102
|
-
const referenceEndFrameIndex = Math.floor(timelineEntry.endTime * framesPerSecond)
|
|
103
|
-
|
|
104
|
-
if (referenceStartFrameIndex < 0 || referenceEndFrameIndex < 0) {
|
|
105
|
-
throw new Error('Unexpected: encountered a negative timestamp in timeline')
|
|
106
|
-
}
|
|
107
|
-
|
|
108
|
-
const mappedStartFrameIndex = getMappedFrameIndexForPath(referenceStartFrameIndex, compactedPath, 'first')
|
|
109
|
-
const mappedEndFrameIndex = getMappedFrameIndexForPath(referenceEndFrameIndex, compactedPath, 'first')
|
|
110
|
-
|
|
111
|
-
let innerTimeline: Timeline | undefined
|
|
112
|
-
|
|
113
|
-
if (recurse && timelineEntry.timeline != null) {
|
|
114
|
-
innerTimeline = timelineEntry.timeline.map((entry) => getMappedTimelineEntry(entry))
|
|
115
|
-
}
|
|
116
|
-
|
|
117
|
-
// Trim silent samples from start and end of mapped entry range
|
|
118
|
-
const sourceSamplesPerFrame = Math.floor(sourceRawAudio.sampleRate / framesPerSecond)
|
|
119
|
-
|
|
120
|
-
let startSampleIndex = mappedStartFrameIndex * sourceSamplesPerFrame
|
|
121
|
-
let endSampleIndex = mappedEndFrameIndex * sourceSamplesPerFrame
|
|
122
|
-
|
|
123
|
-
const frameSamples = sourceRawAudio.audioChannels[0].subarray(startSampleIndex, endSampleIndex)
|
|
124
|
-
|
|
125
|
-
const silenceThresholdDecibels = -40
|
|
126
|
-
|
|
127
|
-
startSampleIndex += getStartingSilentSampleCount(frameSamples, silenceThresholdDecibels)
|
|
128
|
-
endSampleIndex -= getEndingSilentSampleCount(frameSamples, silenceThresholdDecibels)
|
|
129
|
-
|
|
130
|
-
endSampleIndex = Math.max(endSampleIndex, startSampleIndex)
|
|
131
|
-
|
|
132
|
-
// Build mapped timeline entry
|
|
133
|
-
const startTime = startSampleIndex / sourceRawAudio.sampleRate
|
|
134
|
-
const endTime = endSampleIndex / sourceRawAudio.sampleRate
|
|
135
|
-
|
|
136
|
-
return {
|
|
137
|
-
type: timelineEntry.type,
|
|
138
|
-
text: timelineEntry.text,
|
|
139
|
-
|
|
140
|
-
startTime,
|
|
141
|
-
endTime,
|
|
142
|
-
|
|
143
|
-
timeline: innerTimeline
|
|
144
|
-
}
|
|
145
|
-
}
|
|
146
|
-
|
|
147
|
-
const mappedTimeline = referenceTimeline.map((timelineEntry) => getMappedTimelineEntry(timelineEntry))
|
|
103
|
+
const mappedTimeline = referenceTimeline.map(entry => getMappedTimelineEntry(entry, sourceRawAudio, framesPerSecond, compactedPath))
|
|
148
104
|
|
|
149
105
|
logger.end()
|
|
150
106
|
|
|
@@ -322,6 +278,170 @@ export async function alignUsingDtwWithRecognition(
|
|
|
322
278
|
return result
|
|
323
279
|
}
|
|
324
280
|
|
|
281
|
+
// This is experimental code. It doesn't work well enough to be usable for anything.
|
|
282
|
+
// Just testing some alternative approaches.
|
|
283
|
+
export async function alignUsingDtwWithEmbeddings(
|
|
284
|
+
sourceRawAudio: RawAudio,
|
|
285
|
+
referenceRawAudio: RawAudio,
|
|
286
|
+
referenceTimeline: Timeline,
|
|
287
|
+
language: string,
|
|
288
|
+
granularities: DtwGranularity[],
|
|
289
|
+
windowDurations: number[]) {
|
|
290
|
+
|
|
291
|
+
const logger = new Logger()
|
|
292
|
+
|
|
293
|
+
if (sourceRawAudio.sampleRate != 16000) {
|
|
294
|
+
throw new Error('Source audio must have a sample rate of 16000 Hz')
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
if (referenceRawAudio.sampleRate != 16000) {
|
|
298
|
+
throw new Error('Reference audio must have a sample rate of 16000 Hz')
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
const embeddingType: 'w2v-bert-2.0' | 'whisper' = 'w2v-bert-2.0'
|
|
302
|
+
|
|
303
|
+
let sourceEmbeddings: Float32Array[]
|
|
304
|
+
let referenceEmbeddings: Float32Array[]
|
|
305
|
+
let framesPerSecond: number
|
|
306
|
+
|
|
307
|
+
if (embeddingType === 'w2v-bert-2.0') {
|
|
308
|
+
const packageName = 'w2v-bert-2.0-uint8'
|
|
309
|
+
const modelDir = await loadPackage(packageName)
|
|
310
|
+
const modelFilePath = path.join(modelDir, `${packageName}.onnx`)
|
|
311
|
+
|
|
312
|
+
const { Wav2Vec2BertFeatureEmbeddings } = await import('../speech-embeddings/WavToVec2BertFeatureEmbeddings.js')
|
|
313
|
+
|
|
314
|
+
const wav2vecBert = new Wav2Vec2BertFeatureEmbeddings(
|
|
315
|
+
modelFilePath,
|
|
316
|
+
['cpu'],
|
|
317
|
+
)
|
|
318
|
+
|
|
319
|
+
logger.start(`Extract source audio embeddings using the W2V-BERT-2.0 model`)
|
|
320
|
+
sourceEmbeddings = await wav2vecBert.computeEmbeddings(sourceRawAudio)
|
|
321
|
+
|
|
322
|
+
logger.start(`Extract reference audio embeddings using the W2V-BERT-2.0 model`)
|
|
323
|
+
referenceEmbeddings = await wav2vecBert.computeEmbeddings(referenceRawAudio)
|
|
324
|
+
|
|
325
|
+
framesPerSecond = 1000 / 10 / 2
|
|
326
|
+
} else if (embeddingType === 'whisper') {
|
|
327
|
+
const sourceSamples = sourceRawAudio.audioChannels[0]
|
|
328
|
+
const referenceSamples = referenceRawAudio.audioChannels[0]
|
|
329
|
+
|
|
330
|
+
const WhisperSTT = await import(`../recognition/WhisperSTT.js`)
|
|
331
|
+
|
|
332
|
+
const { modelName, modelDir } = await WhisperSTT.loadPackagesAndGetPaths('base.en', language)
|
|
333
|
+
|
|
334
|
+
const whisper = new WhisperSTT.Whisper(modelName, modelDir, ['dml', 'cpu'], ['cpu'])
|
|
335
|
+
|
|
336
|
+
async function encodeToAudioFeatures(samples: Float32Array) {
|
|
337
|
+
const featureVectors: Float32Array[] = []
|
|
338
|
+
|
|
339
|
+
for (let i = 0; i < samples.length; i += 16000 * 30) {
|
|
340
|
+
const startSampleIndex = i
|
|
341
|
+
const endSampleIndex = Math.min(samples.length, i + 16000 * 30)
|
|
342
|
+
const partSampleCount = endSampleIndex - startSampleIndex
|
|
343
|
+
|
|
344
|
+
const audioPart = samples.subarray(startSampleIndex, endSampleIndex)
|
|
345
|
+
const rawAudioForPart = { audioChannels: [audioPart], sampleRate: 16000 } as RawAudio
|
|
346
|
+
|
|
347
|
+
const resultTensor = await whisper.encodeAudio(rawAudioForPart)
|
|
348
|
+
|
|
349
|
+
const vectorLength = resultTensor.dims[2]
|
|
350
|
+
|
|
351
|
+
let featureVectorsForPart = splitFloat32Array(resultTensor.data as Float32Array, vectorLength)
|
|
352
|
+
|
|
353
|
+
featureVectorsForPart = featureVectorsForPart.slice(0, Math.floor((partSampleCount / (16000 * 30)) * 1500))
|
|
354
|
+
|
|
355
|
+
featureVectors.push(...featureVectorsForPart)
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
return featureVectors
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
logger.start(`Extract source audio embeddings using the Whisper encoder model`)
|
|
362
|
+
sourceEmbeddings = await encodeToAudioFeatures(sourceSamples)
|
|
363
|
+
|
|
364
|
+
logger.start(`Extract reference audio embeddings using the Whisper encoder model`)
|
|
365
|
+
referenceEmbeddings = await encodeToAudioFeatures(referenceSamples)
|
|
366
|
+
|
|
367
|
+
framesPerSecond = 1500 / 30
|
|
368
|
+
} else {
|
|
369
|
+
throw new Error(`Unknown embedding type: ${embeddingType}`)
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
logger.start(`Align source and reference audio embeddings using DTW`)
|
|
373
|
+
|
|
374
|
+
const { path: alignmentPath } = alignDTWWindowed(
|
|
375
|
+
referenceEmbeddings,
|
|
376
|
+
sourceEmbeddings,
|
|
377
|
+
cosineDistance,
|
|
378
|
+
1000 * 1000
|
|
379
|
+
)
|
|
380
|
+
|
|
381
|
+
const compactedPath = compactPath(alignmentPath)
|
|
382
|
+
|
|
383
|
+
logger.start('\nConvert path to timeline')
|
|
384
|
+
|
|
385
|
+
const mappedTimeline = referenceTimeline.map(entry => getMappedTimelineEntry(entry, sourceRawAudio, framesPerSecond, compactedPath))
|
|
386
|
+
|
|
387
|
+
logger.end()
|
|
388
|
+
|
|
389
|
+
return mappedTimeline
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
function getMappedTimelineEntry(
|
|
393
|
+
timelineEntry: TimelineEntry,
|
|
394
|
+
sourceRawAudio: RawAudio,
|
|
395
|
+
framesPerSecond: number,
|
|
396
|
+
compactedPath: CompactedPath,
|
|
397
|
+
recurse = true): TimelineEntry {
|
|
398
|
+
|
|
399
|
+
const referenceStartFrameIndex = Math.floor(timelineEntry.startTime * framesPerSecond)
|
|
400
|
+
const referenceEndFrameIndex = Math.floor(timelineEntry.endTime * framesPerSecond)
|
|
401
|
+
|
|
402
|
+
if (referenceStartFrameIndex < 0 || referenceEndFrameIndex < 0) {
|
|
403
|
+
throw new Error('Unexpected: encountered a negative timestamp in timeline')
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
const mappedStartFrameIndex = getMappedFrameIndexForPath(referenceStartFrameIndex, compactedPath, 'first')
|
|
407
|
+
const mappedEndFrameIndex = getMappedFrameIndexForPath(referenceEndFrameIndex, compactedPath, 'first')
|
|
408
|
+
|
|
409
|
+
let innerTimeline: Timeline | undefined
|
|
410
|
+
|
|
411
|
+
if (recurse && timelineEntry.timeline != null) {
|
|
412
|
+
innerTimeline = timelineEntry.timeline.map((entry) => getMappedTimelineEntry(entry, sourceRawAudio, framesPerSecond, compactedPath, recurse))
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Trim silent samples from start and end of mapped entry range
|
|
416
|
+
const sourceSamplesPerFrame = Math.floor(sourceRawAudio.sampleRate / framesPerSecond)
|
|
417
|
+
|
|
418
|
+
let startSampleIndex = mappedStartFrameIndex * sourceSamplesPerFrame
|
|
419
|
+
let endSampleIndex = mappedEndFrameIndex * sourceSamplesPerFrame
|
|
420
|
+
|
|
421
|
+
const frameSamples = sourceRawAudio.audioChannels[0].subarray(startSampleIndex, endSampleIndex)
|
|
422
|
+
|
|
423
|
+
const silenceThresholdDecibels = -40
|
|
424
|
+
|
|
425
|
+
startSampleIndex += getStartingSilentSampleCount(frameSamples, silenceThresholdDecibels)
|
|
426
|
+
endSampleIndex -= getEndingSilentSampleCount(frameSamples, silenceThresholdDecibels)
|
|
427
|
+
|
|
428
|
+
endSampleIndex = Math.max(endSampleIndex, startSampleIndex)
|
|
429
|
+
|
|
430
|
+
// Build mapped timeline entry
|
|
431
|
+
const startTime = startSampleIndex / sourceRawAudio.sampleRate
|
|
432
|
+
const endTime = endSampleIndex / sourceRawAudio.sampleRate
|
|
433
|
+
|
|
434
|
+
return {
|
|
435
|
+
type: timelineEntry.type,
|
|
436
|
+
text: timelineEntry.text,
|
|
437
|
+
|
|
438
|
+
startTime,
|
|
439
|
+
endTime,
|
|
440
|
+
|
|
441
|
+
timeline: innerTimeline
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
|
|
325
445
|
export async function interpolatePhoneTimelines(sourceTimeline: Timeline, referenceTimeline: Timeline) {
|
|
326
446
|
const interpolatedTimeline: Timeline = []
|
|
327
447
|
|
|
@@ -460,7 +580,38 @@ export async function createAlignmentReferenceUsingEspeakForFragments(fragments:
|
|
|
460
580
|
|
|
461
581
|
progressLogger.start("Synthesize alignment reference with eSpeak")
|
|
462
582
|
|
|
463
|
-
const result =
|
|
583
|
+
const result = {
|
|
584
|
+
rawAudio: getEmptyRawAudio(1, await Espeak.getSampleRate()) as RawAudio,
|
|
585
|
+
timeline: [] as Timeline,
|
|
586
|
+
events: [] as EspeakEvent[],
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
// Split fragments to chunks and process each chunk individually,
|
|
590
|
+
// and incrementally merge the chunks to the final result.
|
|
591
|
+
{
|
|
592
|
+
const maxFragmentsInChunk = 1000
|
|
593
|
+
|
|
594
|
+
let timeOffset = 0
|
|
595
|
+
|
|
596
|
+
for (let startOffset = 0; startOffset < fragments.length; startOffset += maxFragmentsInChunk) {
|
|
597
|
+
const chunk = fragments.slice(startOffset, startOffset + maxFragmentsInChunk)
|
|
598
|
+
|
|
599
|
+
const chunkResult = await Espeak.synthesizeFragments(chunk, espeakOptions)
|
|
600
|
+
|
|
601
|
+
result.rawAudio = {
|
|
602
|
+
sampleRate: result.rawAudio.sampleRate,
|
|
603
|
+
audioChannels: concatAudioSegments([result.rawAudio.audioChannels, chunkResult.rawAudio.audioChannels])
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
const chunkTimeline = addTimeOffsetToTimeline(chunkResult.timeline, timeOffset)
|
|
607
|
+
|
|
608
|
+
result.timeline = [...result.timeline, ...chunkTimeline]
|
|
609
|
+
|
|
610
|
+
result.events = [...result.events, ...chunkResult.events]
|
|
611
|
+
|
|
612
|
+
timeOffset += getRawAudioDuration(chunkResult.rawAudio)
|
|
613
|
+
}
|
|
614
|
+
}
|
|
464
615
|
|
|
465
616
|
result.timeline = result.timeline.flatMap(clause => clause.timeline!)
|
|
466
617
|
|
|
@@ -473,7 +624,14 @@ export async function createAlignmentReferenceUsingEspeakForFragments(fragments:
|
|
|
473
624
|
return result
|
|
474
625
|
}
|
|
475
626
|
|
|
476
|
-
export async function createAlignmentReferenceUsingEspeak(
|
|
627
|
+
export async function createAlignmentReferenceUsingEspeak(
|
|
628
|
+
transcript: string,
|
|
629
|
+
language: string,
|
|
630
|
+
plaintextOptions?: API.PlainTextOptions,
|
|
631
|
+
customLexiconPaths?: string[],
|
|
632
|
+
insertSeparators?: boolean,
|
|
633
|
+
useKlatt?: boolean) {
|
|
634
|
+
|
|
477
635
|
const logger = new Logger()
|
|
478
636
|
|
|
479
637
|
logger.start('Synthesize alignment reference with eSpeak')
|
|
@@ -486,7 +644,7 @@ export async function createAlignmentReferenceUsingEspeak(transcript: string, la
|
|
|
486
644
|
customLexiconPaths: customLexiconPaths,
|
|
487
645
|
|
|
488
646
|
espeak: {
|
|
489
|
-
useKlatt
|
|
647
|
+
useKlatt,
|
|
490
648
|
insertSeparators,
|
|
491
649
|
}
|
|
492
650
|
}
|
package/src/api/API.ts
CHANGED
|
@@ -2,15 +2,31 @@
|
|
|
2
2
|
|
|
3
3
|
export * from './Common.js'
|
|
4
4
|
export * from './GlobalOptions.js'
|
|
5
|
+
|
|
5
6
|
export * from './Synthesis.js'
|
|
7
|
+
|
|
6
8
|
export * from './Recognition.js'
|
|
9
|
+
|
|
7
10
|
export * from './Alignment.js'
|
|
8
|
-
|
|
11
|
+
|
|
12
|
+
export * from './SpeechTranslation.js'
|
|
13
|
+
export * from './TextTranslation.js'
|
|
14
|
+
|
|
9
15
|
export * from './TranslationAlignment.js'
|
|
10
|
-
export * from './
|
|
16
|
+
export * from './TranscriptAndTranslationAlignment.js'
|
|
17
|
+
export * from './TimelineTranslationAlignment.js'
|
|
18
|
+
|
|
19
|
+
export * from './LanguageDetectionCommon.js'
|
|
20
|
+
export * from './SpeechLanguageDetection.js'
|
|
21
|
+
export * from './TextLanguageDetection.js'
|
|
22
|
+
|
|
11
23
|
export * from './VoiceActivityDetection.js'
|
|
24
|
+
|
|
12
25
|
export * from './Denoising.js'
|
|
26
|
+
|
|
13
27
|
export * from './SourceSeparation.js'
|
|
28
|
+
|
|
14
29
|
export * from '../server/Server.js'
|
|
15
30
|
export * from '../server/Client.js'
|
|
31
|
+
|
|
16
32
|
export { timelineToSubtitles, subtitlesToTimeline } from '../subtitles/Subtitles.js'
|
package/src/api/APIOptions.ts
CHANGED
|
@@ -3,18 +3,31 @@ import type { ServerOptions } from '../server/Server.js'
|
|
|
3
3
|
import { CLIOptions } from '../cli/CLIOptions.js'
|
|
4
4
|
|
|
5
5
|
export interface APIOptions {
|
|
6
|
-
VoiceListRequestOptions: API.VoiceListRequestOptions
|
|
7
6
|
SynthesisOptions: API.SynthesisOptions
|
|
7
|
+
VoiceListRequestOptions: API.VoiceListRequestOptions
|
|
8
|
+
|
|
8
9
|
RecognitionOptions: API.RecognitionOptions
|
|
10
|
+
|
|
9
11
|
AlignmentOptions: API.AlignmentOptions
|
|
12
|
+
|
|
10
13
|
TranslationAlignmentOptions: API.TranslationAlignmentOptions
|
|
14
|
+
TranscriptAndTranslationAlignmentOptions: API.TranscriptAndTranslationAlignmentOptions
|
|
15
|
+
TimelineTranslationAlignmentOptions: API.TimelineTranslationAlignmentOptions
|
|
16
|
+
|
|
11
17
|
SpeechTranslationOptions: API.SpeechTranslationOptions
|
|
18
|
+
TextTranslationOptions: API.TextTranslationOptions
|
|
19
|
+
|
|
12
20
|
SpeechLanguageDetectionOptions: API.SpeechLanguageDetectionOptions
|
|
13
21
|
TextLanguageDetectionOptions: API.TextLanguageDetectionOptions
|
|
22
|
+
|
|
14
23
|
VADOptions: API.VADOptions
|
|
24
|
+
|
|
15
25
|
DenoisingOptions: API.DenoisingOptions
|
|
26
|
+
|
|
16
27
|
SourceSeparationOptions: API.SourceSeparationOptions
|
|
28
|
+
|
|
17
29
|
ServerOptions: ServerOptions
|
|
30
|
+
|
|
18
31
|
GlobalOptions: API.GlobalOptions
|
|
19
32
|
CLIOptions: CLIOptions
|
|
20
33
|
}
|
package/src/api/Alignment.ts
CHANGED
|
@@ -9,19 +9,14 @@ import { Timeline, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, wordTi
|
|
|
9
9
|
import { formatLanguageCodeWithName, getDefaultDialectForLanguageCodeIfPossible, getShortLanguageCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
10
10
|
import { type WhisperAlignmentOptions } from '../recognition/WhisperSTT.js'
|
|
11
11
|
import chalk from 'chalk'
|
|
12
|
-
import { DtwGranularity, createAlignmentReferenceUsingEspeak } from '../alignment/SpeechAlignment.js'
|
|
12
|
+
import { DtwGranularity, alignUsingDtwWithEmbeddings, createAlignmentReferenceUsingEspeak } from '../alignment/SpeechAlignment.js'
|
|
13
13
|
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
14
14
|
import { type EspeakOptions, defaultEspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
15
15
|
import { isWord } from '../nlp/Segmentation.js'
|
|
16
|
-
import { alignText } from '../alignment/TextAlignment.js'
|
|
17
|
-
import { translateText } from '../text-translation/NLLBTextTranslation.js'
|
|
18
16
|
|
|
19
17
|
const log = logToStderr
|
|
20
18
|
|
|
21
19
|
export async function align(input: AudioSourceParam, transcript: string, options: AlignmentOptions): Promise<AlignmentResult> {
|
|
22
|
-
//await alignText(transcript, transcript)
|
|
23
|
-
//await translateText(transcript, 'en', 'de')
|
|
24
|
-
|
|
25
20
|
const logger = new Logger()
|
|
26
21
|
|
|
27
22
|
const startTimestamp = logger.getTimestamp()
|
|
@@ -164,7 +159,7 @@ export async function align(input: AudioSourceParam, transcript: string, options
|
|
|
164
159
|
const {
|
|
165
160
|
referenceRawAudio,
|
|
166
161
|
referenceTimeline
|
|
167
|
-
} = await createAlignmentReferenceUsingEspeak(transcript, language, options.plainText, options.customLexiconPaths, false)
|
|
162
|
+
} = await createAlignmentReferenceUsingEspeak(transcript, language, options.plainText, options.customLexiconPaths, false, false)
|
|
168
163
|
|
|
169
164
|
logger.end()
|
|
170
165
|
|
|
@@ -196,7 +191,7 @@ export async function align(input: AudioSourceParam, transcript: string, options
|
|
|
196
191
|
referenceRawAudio,
|
|
197
192
|
referenceTimeline,
|
|
198
193
|
espeakVoice,
|
|
199
|
-
} = await createAlignmentReferenceUsingEspeak(transcript, language, options.plainText, options.customLexiconPaths, false)
|
|
194
|
+
} = await createAlignmentReferenceUsingEspeak(transcript, language, options.plainText, options.customLexiconPaths, false, false)
|
|
200
195
|
|
|
201
196
|
logger.end()
|
|
202
197
|
|
|
@@ -225,6 +220,33 @@ export async function align(input: AudioSourceParam, transcript: string, options
|
|
|
225
220
|
break
|
|
226
221
|
}
|
|
227
222
|
|
|
223
|
+
case 'dtw-ea': {
|
|
224
|
+
const { windowDurations, granularities } = getDtwWindowGranularitiesAndDurations()
|
|
225
|
+
|
|
226
|
+
logger.end()
|
|
227
|
+
|
|
228
|
+
logger.logTitledMessage(`Warning`, `The dtw-ea alignment engine is just an early experiment and doesn't currently perform as well as, or as efficiently as other alignment engines.`, chalk.yellow, 'warning')
|
|
229
|
+
|
|
230
|
+
const {
|
|
231
|
+
referenceRawAudio,
|
|
232
|
+
referenceTimeline
|
|
233
|
+
} = await createAlignmentReferenceUsingEspeak(transcript, language, options.plainText, options.customLexiconPaths, false, true)
|
|
234
|
+
|
|
235
|
+
logger.end()
|
|
236
|
+
|
|
237
|
+
const shortLanguageCode = getShortLanguageCode(language)
|
|
238
|
+
|
|
239
|
+
mappedTimeline = await alignUsingDtwWithEmbeddings(
|
|
240
|
+
sourceRawAudio,
|
|
241
|
+
referenceRawAudio,
|
|
242
|
+
referenceTimeline,
|
|
243
|
+
shortLanguageCode,
|
|
244
|
+
granularities,
|
|
245
|
+
windowDurations)
|
|
246
|
+
|
|
247
|
+
break
|
|
248
|
+
}
|
|
249
|
+
|
|
228
250
|
case 'whisper': {
|
|
229
251
|
const WhisperSTT = await import('../recognition/WhisperSTT.js')
|
|
230
252
|
|
|
@@ -315,7 +337,7 @@ export interface AlignmentResult {
|
|
|
315
337
|
backgroundRawAudio?: RawAudio
|
|
316
338
|
}
|
|
317
339
|
|
|
318
|
-
export type AlignmentEngine = 'dtw' | 'dtw-ra' | 'whisper'
|
|
340
|
+
export type AlignmentEngine = 'dtw' | 'dtw-ra' | 'dtw-ea' | 'whisper'
|
|
319
341
|
export type PhoneAlignmentMethod = 'interpolation' | 'dtw'
|
|
320
342
|
|
|
321
343
|
export interface AlignmentOptions {
|
package/src/api/Recognition.ts
CHANGED
|
@@ -319,8 +319,10 @@ export async function recognize(input: AudioSourceParam, options: RecognitionOpt
|
|
|
319
319
|
|
|
320
320
|
export interface RecognitionResult {
|
|
321
321
|
transcript: string
|
|
322
|
+
|
|
322
323
|
timeline: Timeline
|
|
323
324
|
wordTimeline: Timeline
|
|
325
|
+
|
|
324
326
|
language: string
|
|
325
327
|
|
|
326
328
|
inputRawAudio: RawAudio
|
|
@@ -13,12 +13,10 @@ import chalk from 'chalk'
|
|
|
13
13
|
import { type WhisperCppOptions } from '../recognition/WhisperCppSTT.js'
|
|
14
14
|
import { type SileroLanguageDetectionOptions } from '../speech-language-detection/SileroLanguageDetection.js'
|
|
15
15
|
import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js'
|
|
16
|
+
import { LanguageDetectionResults } from './LanguageDetectionCommon.js'
|
|
16
17
|
|
|
17
18
|
const log = logToStderr
|
|
18
19
|
|
|
19
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
20
|
-
// Speech language detection
|
|
21
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
22
20
|
export async function detectSpeechLanguage(input: AudioSourceParam, options: SpeechLanguageDetectionOptions): Promise<SpeechLanguageDetectionResult> {
|
|
23
21
|
const logger = new Logger()
|
|
24
22
|
|
|
@@ -238,107 +236,6 @@ export const defaultSpeechLanguageDetectionOptions: SpeechLanguageDetectionOptio
|
|
|
238
236
|
}
|
|
239
237
|
}
|
|
240
238
|
|
|
241
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
242
|
-
// Text language detection
|
|
243
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
244
|
-
export async function detectTextLanguage(input: string, options: TextLanguageDetectionOptions): Promise<TextLanguageDetectionResult> {
|
|
245
|
-
const logger = new Logger()
|
|
246
|
-
|
|
247
|
-
options = extendDeep(defaultTextLanguageDetectionOptions, options)
|
|
248
|
-
|
|
249
|
-
const defaultLanguage = options.defaultLanguage!
|
|
250
|
-
const fallbackThresholdProbability = options.fallbackThresholdProbability!
|
|
251
|
-
|
|
252
|
-
let detectedLanguageProbabilities: LanguageDetectionResults
|
|
253
|
-
|
|
254
|
-
logger.start(`Initialize ${options.engine} module`)
|
|
255
|
-
|
|
256
|
-
switch (options.engine) {
|
|
257
|
-
case 'tinyld': {
|
|
258
|
-
const { detectLanguage } = await import('../text-language-detection/TinyLDLanguageDetection.js')
|
|
259
|
-
|
|
260
|
-
logger.start('Detect text language using tinyld')
|
|
261
|
-
|
|
262
|
-
detectedLanguageProbabilities = await detectLanguage(input)
|
|
263
|
-
|
|
264
|
-
break
|
|
265
|
-
}
|
|
266
|
-
|
|
267
|
-
case 'fasttext': {
|
|
268
|
-
const { detectLanguage } = await import('../text-language-detection/FastTextLanguageDetection.js')
|
|
269
|
-
|
|
270
|
-
logger.start('Detect text language using FastText')
|
|
271
|
-
|
|
272
|
-
detectedLanguageProbabilities = await detectLanguage(input)
|
|
273
|
-
|
|
274
|
-
break
|
|
275
|
-
}
|
|
276
|
-
|
|
277
|
-
default: {
|
|
278
|
-
throw new Error(`Engine '${options.engine}' is not supported`)
|
|
279
|
-
}
|
|
280
|
-
}
|
|
281
|
-
|
|
282
|
-
let detectedLanguage: string
|
|
283
|
-
|
|
284
|
-
if (detectedLanguageProbabilities.length == 0 ||
|
|
285
|
-
detectedLanguageProbabilities[0].probability < fallbackThresholdProbability) {
|
|
286
|
-
|
|
287
|
-
detectedLanguage = defaultLanguage
|
|
288
|
-
} else {
|
|
289
|
-
detectedLanguage = detectedLanguageProbabilities[0].language
|
|
290
|
-
}
|
|
291
|
-
|
|
292
|
-
logger.end()
|
|
293
|
-
|
|
294
|
-
return {
|
|
295
|
-
detectedLanguage,
|
|
296
|
-
detectedLanguageName: languageCodeToName(detectedLanguage),
|
|
297
|
-
detectedLanguageProbabilities
|
|
298
|
-
}
|
|
299
|
-
}
|
|
300
|
-
|
|
301
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
302
|
-
// Types
|
|
303
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
304
|
-
|
|
305
|
-
export interface TextLanguageDetectionResult {
|
|
306
|
-
detectedLanguage: string
|
|
307
|
-
detectedLanguageName: string
|
|
308
|
-
detectedLanguageProbabilities: LanguageDetectionResults
|
|
309
|
-
}
|
|
310
|
-
|
|
311
|
-
export type LanguageDetectionResults = LanguageDetectionResultsEntry[]
|
|
312
|
-
export interface LanguageDetectionResultsEntry {
|
|
313
|
-
language: string
|
|
314
|
-
languageName: string
|
|
315
|
-
probability: number
|
|
316
|
-
}
|
|
317
|
-
|
|
318
|
-
export type LanguageDetectionGroupResults = LanguageDetectionGroupResultsEntry[]
|
|
319
|
-
export interface LanguageDetectionGroupResultsEntry {
|
|
320
|
-
languageGroup: string
|
|
321
|
-
probability: number
|
|
322
|
-
}
|
|
323
|
-
|
|
324
|
-
export type TextLanguageDetectionEngine = 'tinyld' | 'fasttext'
|
|
325
|
-
|
|
326
|
-
export interface TextLanguageDetectionOptions {
|
|
327
|
-
engine?: TextLanguageDetectionEngine
|
|
328
|
-
defaultLanguage?: string
|
|
329
|
-
fallbackThresholdProbability?: number
|
|
330
|
-
}
|
|
331
|
-
|
|
332
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
333
|
-
// Constants
|
|
334
|
-
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
335
|
-
|
|
336
|
-
export const defaultTextLanguageDetectionOptions: TextLanguageDetectionOptions = {
|
|
337
|
-
engine: 'tinyld',
|
|
338
|
-
defaultLanguage: 'en',
|
|
339
|
-
fallbackThresholdProbability: 0.05,
|
|
340
|
-
}
|
|
341
|
-
|
|
342
239
|
export const speechLanguageDetectionEngines: API.EngineMetadata[] = [
|
|
343
240
|
{
|
|
344
241
|
id: 'silero',
|
|
@@ -359,18 +256,3 @@ export const speechLanguageDetectionEngines: API.EngineMetadata[] = [
|
|
|
359
256
|
type: 'local'
|
|
360
257
|
},
|
|
361
258
|
]
|
|
362
|
-
|
|
363
|
-
export const textLanguageDetectionEngines: API.EngineMetadata[] = [
|
|
364
|
-
{
|
|
365
|
-
id: 'tinyld',
|
|
366
|
-
name: 'TinyLD',
|
|
367
|
-
description: 'A simple language detection library.',
|
|
368
|
-
type: 'local'
|
|
369
|
-
},
|
|
370
|
-
{
|
|
371
|
-
id: 'fasttext',
|
|
372
|
-
name: 'FastText',
|
|
373
|
-
description: 'A library for word representations and sentence classification by Facebook research.',
|
|
374
|
-
type: 'local'
|
|
375
|
-
},
|
|
376
|
-
]
|
|
@@ -6,7 +6,7 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
|
|
7
7
|
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
8
8
|
import { type WhisperOptions } from '../recognition/WhisperSTT.js'
|
|
9
|
-
import { formatLanguageCodeWithName, getShortLanguageCode,
|
|
9
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeIdentifierToLanguageCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
10
10
|
import { EngineMetadata } from './Common.js'
|
|
11
11
|
import { type SpeechLanguageDetectionOptions, detectSpeechLanguage } from './API.js'
|
|
12
12
|
import chalk from 'chalk'
|
|
@@ -81,7 +81,7 @@ export async function translateSpeech(input: AudioSourceParam, options: SpeechTr
|
|
|
81
81
|
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
-
options.targetLanguage = await
|
|
84
|
+
options.targetLanguage = await normalizeIdentifierToLanguageCode(options.targetLanguage!)
|
|
85
85
|
|
|
86
86
|
logger.logTitledMessage('Target language', formatLanguageCodeWithName(options.targetLanguage))
|
|
87
87
|
|