echogarden 1.0.4 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -23
- package/data/schemas/options.json +177 -36
- package/dist/alignment/SpeechAlignment.d.ts +1 -1
- package/dist/alignment/SpeechAlignment.js +1 -1
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +1 -0
- package/dist/api/API.js +1 -0
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +1 -0
- package/dist/api/Alignment.d.ts +3 -3
- package/dist/api/Alignment.js +5 -10
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetection.d.ts +5 -7
- package/dist/api/LanguageDetection.js +3 -2
- package/dist/api/LanguageDetection.js.map +1 -1
- package/dist/api/Recognition.d.ts +4 -5
- package/dist/api/Recognition.js +5 -8
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/SourceSeparation.d.ts +2 -0
- package/dist/api/SourceSeparation.js +4 -2
- package/dist/api/SourceSeparation.js.map +1 -1
- package/dist/api/Synthesis.d.ts +3 -1
- package/dist/api/Synthesis.js +9 -10
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.d.ts +1 -1
- package/dist/api/Translation.js +4 -8
- package/dist/api/Translation.js.map +1 -1
- package/dist/api/TranslationAlignment.d.ts +31 -0
- package/dist/api/TranslationAlignment.js +121 -0
- package/dist/api/TranslationAlignment.js.map +1 -0
- package/dist/api/VoiceActivityDetection.d.ts +5 -1
- package/dist/api/VoiceActivityDetection.js +38 -2
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/audio/AudioPlayer.js +6 -1
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/cli/CLI.js +85 -0
- package/dist/cli/CLI.js.map +1 -1
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/math/MedianFilter.d.ts +5 -0
- package/dist/math/MedianFilter.js +102 -0
- package/dist/math/MedianFilter.js.map +1 -0
- package/dist/math/VectorMath.d.ts +0 -2
- package/dist/math/VectorMath.js +1 -25
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/recognition/OpenAICloudSTT.d.ts +1 -1
- package/dist/recognition/OpenAICloudSTT.js.map +1 -1
- package/dist/recognition/SileroSTT.d.ts +22 -1
- package/dist/recognition/SileroSTT.js +122 -95
- package/dist/recognition/SileroSTT.js.map +1 -1
- package/dist/recognition/WhisperCppSTT.js +1 -1
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.d.ts +52 -19
- package/dist/recognition/WhisperSTT.js +645 -494
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Server.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +5 -3
- package/dist/source-separation/MDXNetSourceSeparation.js +26 -19
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +15 -9
- package/dist/speech-language-detection/SileroLanguageDetection.js +23 -16
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/synthesis/EspeakTTS.js +4 -0
- package/dist/synthesis/EspeakTTS.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.js.map +1 -1
- package/dist/synthesis/VitsTTS.d.ts +8 -6
- package/dist/synthesis/VitsTTS.js +36 -31
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/tests/Test.js.map +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +14 -0
- package/dist/utilities/OnnxUtilities.js +43 -0
- package/dist/utilities/OnnxUtilities.js.map +1 -0
- package/dist/utilities/Utilities.d.ts +4 -8
- package/dist/utilities/Utilities.js +35 -58
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/voice-activity-detection/SileroVAD.d.ts +5 -3
- package/dist/voice-activity-detection/SileroVAD.js +9 -11
- package/dist/voice-activity-detection/SileroVAD.js.map +1 -1
- package/docs/API.md +54 -34
- package/docs/CLI.md +25 -13
- package/docs/Contributing.md +4 -2
- package/docs/Engines.md +43 -32
- package/docs/Licenses.md +3 -4
- package/docs/Options.md +47 -11
- package/docs/Releases.md +4 -0
- package/docs/Server.md +8 -6
- package/docs/Tasklist.md +39 -52
- package/docs/Technical.md +1 -1
- package/package.json +8 -12
- package/src/alignment/SpeechAlignment.ts +1 -1
- package/src/api/API.ts +1 -0
- package/src/api/APIOptions.ts +1 -0
- package/src/api/Alignment.ts +10 -14
- package/src/api/LanguageDetection.ts +14 -10
- package/src/api/Recognition.ts +17 -10
- package/src/api/SourceSeparation.ts +7 -2
- package/src/api/Synthesis.ts +26 -11
- package/src/api/Translation.ts +14 -8
- package/src/api/TranslationAlignment.ts +213 -0
- package/src/api/VoiceActivityDetection.ts +66 -3
- package/src/audio/AudioPlayer.ts +6 -2
- package/src/cli/CLI.ts +121 -2
- package/src/dsp/FFT.ts +3 -0
- package/src/math/MedianFilter.ts +124 -0
- package/src/math/VectorMath.ts +1 -36
- package/src/recognition/OpenAICloudSTT.ts +27 -27
- package/src/recognition/SileroSTT.ts +149 -102
- package/src/recognition/WhisperCppSTT.ts +1 -1
- package/src/recognition/WhisperSTT.ts +961 -684
- package/src/server/Server.ts +1 -1
- package/src/source-separation/MDXNetSourceSeparation.ts +35 -19
- package/src/speech-language-detection/SileroLanguageDetection.ts +53 -33
- package/src/synthesis/EspeakTTS.ts +8 -0
- package/src/synthesis/GoogleCloudTTS.ts +12 -1
- package/src/synthesis/VitsTTS.ts +57 -46
- package/src/tests/Test.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +68 -0
- package/src/utilities/Utilities.ts +38 -66
- package/src/voice-activity-detection/SileroVAD.ts +15 -15
- package/dist/utilities/NdArrayUtilities.d.ts +0 -3
- package/dist/utilities/NdArrayUtilities.js +0 -23
- package/dist/utilities/NdArrayUtilities.js.map +0 -1
- package/src/utilities/NdArrayUtilities.ts +0 -31
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
2
|
+
|
|
3
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
4
|
+
import { AudioSourceParam, RawAudio, ensureRawAudio, normalizeAudioLevel, trimAudioEnd } from '../audio/AudioUtilities.js'
|
|
5
|
+
import { Logger } from '../utilities/Logger.js'
|
|
6
|
+
|
|
7
|
+
import * as API from './API.js'
|
|
8
|
+
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
9
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeLanguageCode } from '../utilities/Locale.js'
|
|
10
|
+
import { type WhisperAlignmentOptions } from '../recognition/WhisperSTT.js'
|
|
11
|
+
import chalk from 'chalk'
|
|
12
|
+
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
13
|
+
|
|
14
|
+
const log = logToStderr
|
|
15
|
+
|
|
16
|
+
export async function alignTranslation(input: AudioSourceParam, transcript: string, options: TranslationAlignmentOptions): Promise<TranslationAlignmentResult> {
|
|
17
|
+
const logger = new Logger()
|
|
18
|
+
|
|
19
|
+
const startTimestamp = logger.getTimestamp()
|
|
20
|
+
|
|
21
|
+
options = extendDeep(defaultTranslationAlignmentOptions, options)
|
|
22
|
+
|
|
23
|
+
const inputRawAudio = await ensureRawAudio(input)
|
|
24
|
+
|
|
25
|
+
let sourceRawAudio: RawAudio
|
|
26
|
+
let isolatedRawAudio: RawAudio | undefined
|
|
27
|
+
let backgroundRawAudio: RawAudio | undefined
|
|
28
|
+
|
|
29
|
+
if (options.isolate) {
|
|
30
|
+
logger.log(``)
|
|
31
|
+
logger.end();
|
|
32
|
+
|
|
33
|
+
({ isolatedRawAudio, backgroundRawAudio } = await API.isolate(inputRawAudio, options.sourceSeparation!))
|
|
34
|
+
|
|
35
|
+
logger.end()
|
|
36
|
+
logger.log(``)
|
|
37
|
+
|
|
38
|
+
sourceRawAudio = await ensureRawAudio(isolatedRawAudio, 16000, 1)
|
|
39
|
+
} else {
|
|
40
|
+
sourceRawAudio = await ensureRawAudio(inputRawAudio, 16000, 1)
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
let sourceUncropTimeline: Timeline | undefined
|
|
44
|
+
|
|
45
|
+
if (options.crop) {
|
|
46
|
+
logger.start('Crop using voice activity detection');
|
|
47
|
+
({ timeline: sourceUncropTimeline, croppedRawAudio: sourceRawAudio } = await API.detectVoiceActivity(sourceRawAudio, options.vad!))
|
|
48
|
+
|
|
49
|
+
logger.end()
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
logger.start('Prepare for alignment')
|
|
53
|
+
|
|
54
|
+
sourceRawAudio = normalizeAudioLevel(sourceRawAudio)
|
|
55
|
+
sourceRawAudio.audioChannels[0] = trimAudioEnd(sourceRawAudio.audioChannels[0])
|
|
56
|
+
|
|
57
|
+
let sourceLanguage: string
|
|
58
|
+
|
|
59
|
+
if (options.sourceLanguage) {
|
|
60
|
+
sourceLanguage = normalizeLanguageCode(options.sourceLanguage!)
|
|
61
|
+
} else {
|
|
62
|
+
logger.start('No source language specified. Detecting speech language')
|
|
63
|
+
const { detectedLanguage } = await API.detectSpeechLanguage(sourceRawAudio, options.languageDetection || {})
|
|
64
|
+
|
|
65
|
+
logger.end()
|
|
66
|
+
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
67
|
+
|
|
68
|
+
sourceLanguage = detectedLanguage
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const targetLanguage = normalizeLanguageCode(options.targetLanguage!)
|
|
72
|
+
|
|
73
|
+
let mappedTimeline: Timeline
|
|
74
|
+
|
|
75
|
+
switch (options.engine) {
|
|
76
|
+
case 'whisper': {
|
|
77
|
+
const WhisperSTT = await import('../recognition/WhisperSTT.js')
|
|
78
|
+
|
|
79
|
+
const shortSourceLanguageCode = getShortLanguageCode(sourceLanguage)
|
|
80
|
+
const shortTargetLanguageCode = getShortLanguageCode(targetLanguage)
|
|
81
|
+
|
|
82
|
+
if (shortTargetLanguageCode != 'en') {
|
|
83
|
+
throw new Error('Whisper translation only supports English as target language')
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
if (shortSourceLanguageCode == 'en' && shortTargetLanguageCode == 'en') {
|
|
87
|
+
throw new Error('Both translation source and target languages are English')
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const whisperAlignmnentOptions = options.whisper!
|
|
91
|
+
|
|
92
|
+
const { modelName, modelDir } = await WhisperSTT.loadPackagesAndGetPaths(whisperAlignmnentOptions.model, shortSourceLanguageCode)
|
|
93
|
+
|
|
94
|
+
logger.end()
|
|
95
|
+
|
|
96
|
+
if (modelName.endsWith('.en')) {
|
|
97
|
+
throw new Error('Whisper translation tasks are only possible with a multilingual model')
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
mappedTimeline = await WhisperSTT.alignEnglishTranslation(sourceRawAudio, transcript, modelName, modelDir, shortSourceLanguageCode, whisperAlignmnentOptions)
|
|
101
|
+
|
|
102
|
+
break
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
default: {
|
|
106
|
+
throw new Error(`Engine '${options.engine}' is not supported`)
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
// If the audio was cropped before recognition, map the timestamps back to the original audio
|
|
111
|
+
if (sourceUncropTimeline && sourceUncropTimeline.length > 0) {
|
|
112
|
+
API.convertCroppedToUncroppedTimeline(mappedTimeline, sourceUncropTimeline)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// Add text offsets
|
|
116
|
+
addWordTextOffsetsToTimeline(mappedTimeline, transcript)
|
|
117
|
+
|
|
118
|
+
// Make segment timeline
|
|
119
|
+
const { segmentTimeline } = await wordTimelineToSegmentSentenceTimeline(mappedTimeline, transcript, sourceLanguage, options.plainText?.paragraphBreaks, options.plainText?.whitespace)
|
|
120
|
+
|
|
121
|
+
logger.end()
|
|
122
|
+
logger.logDuration(`Total translation alignment time`, startTimestamp, chalk.magentaBright)
|
|
123
|
+
|
|
124
|
+
return {
|
|
125
|
+
timeline: segmentTimeline,
|
|
126
|
+
wordTimeline: mappedTimeline,
|
|
127
|
+
|
|
128
|
+
transcript,
|
|
129
|
+
language: sourceLanguage,
|
|
130
|
+
|
|
131
|
+
inputRawAudio,
|
|
132
|
+
isolatedRawAudio,
|
|
133
|
+
backgroundRawAudio,
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export interface TranslationAlignmentResult {
|
|
138
|
+
timeline: Timeline
|
|
139
|
+
wordTimeline: Timeline
|
|
140
|
+
|
|
141
|
+
transcript: string
|
|
142
|
+
language: string
|
|
143
|
+
|
|
144
|
+
inputRawAudio: RawAudio
|
|
145
|
+
isolatedRawAudio?: RawAudio
|
|
146
|
+
backgroundRawAudio?: RawAudio
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
export type TranslationAlignmentEngine = 'whisper'
|
|
150
|
+
|
|
151
|
+
export interface TranslationAlignmentOptions {
|
|
152
|
+
engine?: TranslationAlignmentEngine
|
|
153
|
+
|
|
154
|
+
sourceLanguage?: string
|
|
155
|
+
targetLanguage?: string
|
|
156
|
+
|
|
157
|
+
isolate?: boolean
|
|
158
|
+
|
|
159
|
+
crop?: boolean
|
|
160
|
+
|
|
161
|
+
languageDetection?: API.SpeechLanguageDetectionOptions
|
|
162
|
+
|
|
163
|
+
vad?: API.VADOptions
|
|
164
|
+
|
|
165
|
+
plainText?: API.PlainTextOptions
|
|
166
|
+
|
|
167
|
+
subtitles?: SubtitlesConfig
|
|
168
|
+
|
|
169
|
+
sourceSeparation?: API.SourceSeparationOptions
|
|
170
|
+
|
|
171
|
+
whisper?: WhisperAlignmentOptions
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
export const defaultTranslationAlignmentOptions: TranslationAlignmentOptions = {
|
|
175
|
+
engine: 'whisper',
|
|
176
|
+
|
|
177
|
+
sourceLanguage: undefined,
|
|
178
|
+
targetLanguage: 'en',
|
|
179
|
+
|
|
180
|
+
isolate: false,
|
|
181
|
+
|
|
182
|
+
crop: true,
|
|
183
|
+
|
|
184
|
+
languageDetection: {
|
|
185
|
+
},
|
|
186
|
+
|
|
187
|
+
plainText: {
|
|
188
|
+
paragraphBreaks: 'double',
|
|
189
|
+
whitespace: 'collapse'
|
|
190
|
+
},
|
|
191
|
+
|
|
192
|
+
subtitles: {
|
|
193
|
+
},
|
|
194
|
+
|
|
195
|
+
vad: {
|
|
196
|
+
engine: 'adaptive-gate'
|
|
197
|
+
},
|
|
198
|
+
|
|
199
|
+
sourceSeparation: {
|
|
200
|
+
},
|
|
201
|
+
|
|
202
|
+
whisper: {
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
export const translationAlignmentEngines: API.EngineMetadata[] = [
|
|
207
|
+
{
|
|
208
|
+
id: 'whisper',
|
|
209
|
+
name: 'OpenAI Whisper',
|
|
210
|
+
description: 'Extracts timestamps by guiding the Whisper recognition model to recognize the translated transcript tokens.',
|
|
211
|
+
type: 'local'
|
|
212
|
+
}
|
|
213
|
+
]
|
|
@@ -10,6 +10,8 @@ import { loadPackage } from '../utilities/PackageManager.js'
|
|
|
10
10
|
import { EngineMetadata } from './Common.js'
|
|
11
11
|
import chalk from 'chalk'
|
|
12
12
|
import { type AdaptiveGateVADOptions } from '../voice-activity-detection/AdaptiveGateVAD.js'
|
|
13
|
+
import { type WhisperVADOptions } from '../recognition/WhisperSTT.js'
|
|
14
|
+
import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js'
|
|
13
15
|
|
|
14
16
|
const log = logToStderr
|
|
15
17
|
|
|
@@ -56,7 +58,14 @@ export async function detectVoiceActivity(input: AudioSourceParam, options: VADO
|
|
|
56
58
|
const modelPath = path.join(modelDir, 'silero-vad.onnx')
|
|
57
59
|
const frameDuration = sileroOptions.frameDuration!
|
|
58
60
|
|
|
59
|
-
const
|
|
61
|
+
const onnxExecutionProviders: OnnxExecutionProvider[] = sileroOptions.provider ? [sileroOptions.provider] : []
|
|
62
|
+
|
|
63
|
+
const frameProbabilities = await SileroVAD.detectVoiceActivity(
|
|
64
|
+
sourceRawAudio,
|
|
65
|
+
modelPath,
|
|
66
|
+
frameDuration,
|
|
67
|
+
onnxExecutionProviders)
|
|
68
|
+
|
|
60
69
|
const frameDurationSeconds = sileroOptions.frameDuration! / 1000
|
|
61
70
|
|
|
62
71
|
verboseTimeline = frameProbabilitiesToTimeline(frameProbabilities, frameDurationSeconds, activityThreshold)
|
|
@@ -81,6 +90,46 @@ export async function detectVoiceActivity(input: AudioSourceParam, options: VADO
|
|
|
81
90
|
break
|
|
82
91
|
}
|
|
83
92
|
|
|
93
|
+
case 'whisper': {
|
|
94
|
+
const WhisperSTT = await import('../recognition/WhisperSTT.js')
|
|
95
|
+
|
|
96
|
+
const whisperVADOptions = options.whisper!
|
|
97
|
+
|
|
98
|
+
logger.end()
|
|
99
|
+
|
|
100
|
+
const { modelName, modelDir } = await WhisperSTT.loadPackagesAndGetPaths(whisperVADOptions.model, 'de')
|
|
101
|
+
|
|
102
|
+
logger.end();
|
|
103
|
+
|
|
104
|
+
const { partProbabilities } = await WhisperSTT.detectVoiceActivity(
|
|
105
|
+
sourceRawAudio,
|
|
106
|
+
modelName,
|
|
107
|
+
modelDir,
|
|
108
|
+
whisperVADOptions,
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
verboseTimeline = []
|
|
112
|
+
|
|
113
|
+
for (const entry of partProbabilities) {
|
|
114
|
+
const hasSpeech = entry.confidence! >= activityThreshold
|
|
115
|
+
|
|
116
|
+
const text = hasSpeech ? 'active' : 'inactive'
|
|
117
|
+
|
|
118
|
+
if (verboseTimeline.length === 0 || verboseTimeline[verboseTimeline.length - 1].text != text) {
|
|
119
|
+
verboseTimeline.push({
|
|
120
|
+
type: 'segment',
|
|
121
|
+
text,
|
|
122
|
+
startTime: entry.startTime,
|
|
123
|
+
endTime: entry.endTime
|
|
124
|
+
})
|
|
125
|
+
} else {
|
|
126
|
+
verboseTimeline[verboseTimeline.length - 1].endTime = entry.endTime
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
break
|
|
131
|
+
}
|
|
132
|
+
|
|
84
133
|
case 'adaptive-gate': {
|
|
85
134
|
const AdaptiveGateVAD = await import('../voice-activity-detection/AdaptiveGateVAD.js')
|
|
86
135
|
|
|
@@ -104,7 +153,12 @@ export async function detectVoiceActivity(input: AudioSourceParam, options: VADO
|
|
|
104
153
|
logger.log('')
|
|
105
154
|
logger.logDuration(`Total voice activity detection time`, startTimestamp, chalk.magentaBright)
|
|
106
155
|
|
|
107
|
-
return {
|
|
156
|
+
return {
|
|
157
|
+
timeline,
|
|
158
|
+
verboseTimeline,
|
|
159
|
+
inputRawAudio,
|
|
160
|
+
croppedRawAudio
|
|
161
|
+
}
|
|
108
162
|
}
|
|
109
163
|
|
|
110
164
|
function frameProbabilitiesToTimeline(frameProbabilities: number[], frameDurationSeconds: number, activityThreshold: number) {
|
|
@@ -176,7 +230,7 @@ export interface VADResult {
|
|
|
176
230
|
croppedRawAudio: RawAudio
|
|
177
231
|
}
|
|
178
232
|
|
|
179
|
-
export type VADEngine = 'webrtc' | 'silero' | 'rnnoise' | 'adaptive-gate'
|
|
233
|
+
export type VADEngine = 'webrtc' | 'silero' | 'rnnoise' | 'whisper' | 'adaptive-gate'
|
|
180
234
|
|
|
181
235
|
export interface VADOptions {
|
|
182
236
|
engine?: VADEngine
|
|
@@ -190,11 +244,14 @@ export interface VADOptions {
|
|
|
190
244
|
|
|
191
245
|
silero?: {
|
|
192
246
|
frameDuration?: 30 | 60 | 90
|
|
247
|
+
provider?: OnnxExecutionProvider
|
|
193
248
|
}
|
|
194
249
|
|
|
195
250
|
rnnoise?: {
|
|
196
251
|
}
|
|
197
252
|
|
|
253
|
+
whisper?: WhisperVADOptions
|
|
254
|
+
|
|
198
255
|
adaptiveGate?: AdaptiveGateVADOptions
|
|
199
256
|
}
|
|
200
257
|
|
|
@@ -210,11 +267,17 @@ export const defaultVADOptions: VADOptions = {
|
|
|
210
267
|
|
|
211
268
|
silero: {
|
|
212
269
|
frameDuration: 90,
|
|
270
|
+
provider: undefined,
|
|
213
271
|
},
|
|
214
272
|
|
|
215
273
|
rnnoise: {
|
|
216
274
|
},
|
|
217
275
|
|
|
276
|
+
whisper: {
|
|
277
|
+
model: 'tiny',
|
|
278
|
+
temperature: 1.0,
|
|
279
|
+
},
|
|
280
|
+
|
|
218
281
|
adaptiveGate: {
|
|
219
282
|
}
|
|
220
283
|
}
|
package/src/audio/AudioPlayer.ts
CHANGED
|
@@ -355,5 +355,9 @@ export function playAudioSamples_Speaker(rawAudio: RawAudio, onTimePosition?: (t
|
|
|
355
355
|
})
|
|
356
356
|
}
|
|
357
357
|
|
|
358
|
-
export const charactersToWriteAhead =
|
|
359
|
-
|
|
358
|
+
export const charactersToWriteAhead = [
|
|
359
|
+
',', '.', ',', '、', ':', ';',
|
|
360
|
+
'。', ':', ';', '?', '?', '!', '!',
|
|
361
|
+
')', ']', '}', `"`, `'`, '”', '’',
|
|
362
|
+
'-', '—', '»', '،', '؟'
|
|
363
|
+
]
|
package/src/cli/CLI.ts
CHANGED
|
@@ -13,8 +13,8 @@ import { encodeFromChannels, getDefaultFFMpegOptionsForSpeech } from '../codecs/
|
|
|
13
13
|
import path, { parse as parsePath } from 'node:path'
|
|
14
14
|
import { splitToParagraphs, splitToWords, wordCharacterPattern } from '../nlp/Segmentation.js'
|
|
15
15
|
import { playAudioSamples, playAudioWithWordTimeline } from '../audio/AudioPlayer.js'
|
|
16
|
-
import {
|
|
17
|
-
import { Timeline, TimelineEntry, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, roundTimelineProperties
|
|
16
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
17
|
+
import { Timeline, TimelineEntry, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, roundTimelineProperties } from '../utilities/Timeline.js'
|
|
18
18
|
import { ensureDir, existsSync, readAndParseJsonFile, readFile, readdir, writeFileSafe } from '../utilities/FileSystem.js'
|
|
19
19
|
import { formatLanguageCodeWithName, getShortLanguageCode } from '../utilities/Locale.js'
|
|
20
20
|
import { APIOptions } from '../api/APIOptions.js'
|
|
@@ -167,6 +167,8 @@ const commandHelp = [
|
|
|
167
167
|
` Align audio file to the reference transcript file\n`,
|
|
168
168
|
`${executableName} ${chalk.magentaBright('translate-speech')} inputFile [output files...] [options...]`,
|
|
169
169
|
` Transcribe audio file directly to a different language\n`,
|
|
170
|
+
`${executableName} ${chalk.magentaBright('align-translation')} audioFile referenceFile [output files...] [options...]`,
|
|
171
|
+
` Align audio file to the reference translated transcript file\n`,
|
|
170
172
|
`${executableName} ${chalk.magentaBright('detect-speech-language')} audioFile [output files...] [options...]`,
|
|
171
173
|
` Detect language of audio file\n`,
|
|
172
174
|
`${executableName} ${chalk.magentaBright('detect-text-language')} inputFile [output files...] [options...]`,
|
|
@@ -220,6 +222,11 @@ async function startWithArgs(parsedArgs: CLIArguments) {
|
|
|
220
222
|
break
|
|
221
223
|
}
|
|
222
224
|
|
|
225
|
+
case 'align-translation': {
|
|
226
|
+
await alignTranslation(parsedArgs.commandArgs, parsedArgs.options)
|
|
227
|
+
break
|
|
228
|
+
}
|
|
229
|
+
|
|
223
230
|
case 'detect-language': {
|
|
224
231
|
await detectLanguage(parsedArgs.commandArgs, parsedArgs.options, 'auto')
|
|
225
232
|
break
|
|
@@ -607,6 +614,112 @@ async function align(commandArgs: string[], cliOptions: Map<string, string>) {
|
|
|
607
614
|
}
|
|
608
615
|
}
|
|
609
616
|
|
|
617
|
+
async function alignTranslation(commandArgs: string[], cliOptions: Map<string, string>) {
|
|
618
|
+
const logger = new Logger()
|
|
619
|
+
|
|
620
|
+
const audioFilename = commandArgs[0]
|
|
621
|
+
const outputFilenames = commandArgs.slice(2)
|
|
622
|
+
|
|
623
|
+
if (audioFilename == undefined) {
|
|
624
|
+
throw new Error(`align-translation requires an argument containing the audio file path.`)
|
|
625
|
+
}
|
|
626
|
+
|
|
627
|
+
if (!existsSync(audioFilename)) {
|
|
628
|
+
throw new Error(`The given source file '${audioFilename}' was not found.`)
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
const alignmentReferenceFile = commandArgs[1]
|
|
632
|
+
|
|
633
|
+
if (alignmentReferenceFile == undefined) {
|
|
634
|
+
throw new Error(`align-translation requires a second argument containing the translated reference file path.`)
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
if (!existsSync(alignmentReferenceFile)) {
|
|
638
|
+
throw new Error(`The given reference file '${alignmentReferenceFile}' was not found.`)
|
|
639
|
+
}
|
|
640
|
+
|
|
641
|
+
const referenceFileExtension = getLowercaseFileExtension(alignmentReferenceFile)
|
|
642
|
+
const fileContent = await readFile(alignmentReferenceFile, { encoding: 'utf-8' })
|
|
643
|
+
|
|
644
|
+
let text: string
|
|
645
|
+
|
|
646
|
+
if (referenceFileExtension == 'txt') {
|
|
647
|
+
text = fileContent
|
|
648
|
+
} else if (referenceFileExtension == 'html' || referenceFileExtension == 'htm') {
|
|
649
|
+
text = await convertHtmlToText(fileContent)
|
|
650
|
+
} else if (referenceFileExtension == 'srt' || referenceFileExtension == 'vtt') {
|
|
651
|
+
text = subtitlesToText(fileContent)
|
|
652
|
+
} else {
|
|
653
|
+
throw new Error(`align only supports reference files with extensions 'txt', 'html', 'htm', 'srt' or 'vtt'`)
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
|
|
657
|
+
additionalOptionsSchema.set('play', { type: 'boolean' })
|
|
658
|
+
additionalOptionsSchema.set('overwrite', { type: 'boolean' })
|
|
659
|
+
|
|
660
|
+
if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
|
|
661
|
+
cliOptions.set('play', `${outputFilenames.length == 0}`)
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
const options: API.TranslationAlignmentOptions = await cliOptionsMapToOptionsObject(cliOptions, 'TranslationAlignmentOptions', additionalOptionsSchema)
|
|
665
|
+
|
|
666
|
+
const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
|
|
667
|
+
const { includesPlaceholderPattern } = await checkOutputFilenames(outputFilenames, true, true, true)
|
|
668
|
+
|
|
669
|
+
const {
|
|
670
|
+
timeline,
|
|
671
|
+
wordTimeline,
|
|
672
|
+
transcript,
|
|
673
|
+
language,
|
|
674
|
+
inputRawAudio,
|
|
675
|
+
isolatedRawAudio,
|
|
676
|
+
backgroundRawAudio } = await API.alignTranslation(audioFilename, text, options)
|
|
677
|
+
|
|
678
|
+
if (outputFilenames.length > 0) {
|
|
679
|
+
logger.start('\nWrite output files')
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
if (includesPlaceholderPattern) {
|
|
683
|
+
for (let segmentIndex = 0; segmentIndex < timeline.length; segmentIndex++) {
|
|
684
|
+
const segmentEntry = timeline[segmentIndex]
|
|
685
|
+
const segmentAudio = sliceRawAudioByTime(inputRawAudio, segmentEntry.startTime, segmentEntry.endTime)
|
|
686
|
+
const sentenceTimeline = addTimeOffsetToTimeline(segmentEntry.timeline!, -segmentEntry.startTime)
|
|
687
|
+
|
|
688
|
+
await writeOutputFilesForSegment(outputFilenames, segmentIndex, timeline.length, segmentAudio, sentenceTimeline, segmentEntry.text, language, allowOverwrite)
|
|
689
|
+
}
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
for (const outputFilename of outputFilenames) {
|
|
693
|
+
const partPatternMatch = outputFilename.match(filenamePlaceholderPattern)
|
|
694
|
+
|
|
695
|
+
if (partPatternMatch) {
|
|
696
|
+
continue
|
|
697
|
+
}
|
|
698
|
+
|
|
699
|
+
const fileSaver = getFileSaver(outputFilename, allowOverwrite)
|
|
700
|
+
|
|
701
|
+
await fileSaver(inputRawAudio, timeline, transcript, options.subtitles)
|
|
702
|
+
|
|
703
|
+
await writeSourceSeparationOutputIfNeeded(outputFilename, isolatedRawAudio, backgroundRawAudio, allowOverwrite, true)
|
|
704
|
+
}
|
|
705
|
+
|
|
706
|
+
logger.end()
|
|
707
|
+
|
|
708
|
+
if ((options as any).play) {
|
|
709
|
+
let audioToPlay: RawAudio
|
|
710
|
+
|
|
711
|
+
if (isolatedRawAudio) {
|
|
712
|
+
audioToPlay = isolatedRawAudio
|
|
713
|
+
} else {
|
|
714
|
+
audioToPlay = inputRawAudio
|
|
715
|
+
}
|
|
716
|
+
|
|
717
|
+
const normalizedAudioToPlay = normalizeAudioLevel(audioToPlay)
|
|
718
|
+
|
|
719
|
+
await playAudioWithWordTimeline(normalizedAudioToPlay, wordTimeline, transcript)
|
|
720
|
+
}
|
|
721
|
+
}
|
|
722
|
+
|
|
610
723
|
async function translateSpeech(commandArgs: string[], cliOptions: Map<string, string>) {
|
|
611
724
|
const logger = new Logger()
|
|
612
725
|
|
|
@@ -976,6 +1089,12 @@ async function listEngines(commandArgs: string[], cliOptions: Map<string, string
|
|
|
976
1089
|
break
|
|
977
1090
|
}
|
|
978
1091
|
|
|
1092
|
+
case 'align-translation': {
|
|
1093
|
+
engines = API.translationAlignmentEngines
|
|
1094
|
+
|
|
1095
|
+
break
|
|
1096
|
+
}
|
|
1097
|
+
|
|
979
1098
|
case 'translate-speech': {
|
|
980
1099
|
engines = API.speechTranslationEngines
|
|
981
1100
|
|
package/src/dsp/FFT.ts
CHANGED
|
@@ -48,6 +48,7 @@ export async function stftr(samples: Float32Array, fftOrder: number, windowSize:
|
|
|
48
48
|
}
|
|
49
49
|
|
|
50
50
|
binsBufferRef.clear()
|
|
51
|
+
|
|
51
52
|
m._kiss_fftr(statePtr, frameBufferRef.address, binsBufferRef.address)
|
|
52
53
|
|
|
53
54
|
const bins = binsBufferRef.view.slice(0, fftOrder + 2)
|
|
@@ -103,7 +104,9 @@ export async function stiftr(binsForFrames: Float32Array[], fftOrder: number, wi
|
|
|
103
104
|
binsRef.view.set(binsForFrame)
|
|
104
105
|
|
|
105
106
|
frameBufferRef.clear()
|
|
107
|
+
|
|
106
108
|
m._kiss_fftri(statePtr, binsRef.address, frameBufferRef.address)
|
|
109
|
+
|
|
107
110
|
const frameSamples = frameBufferRef.view
|
|
108
111
|
|
|
109
112
|
const frameStartOffset = frameIndex * hopSize
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import { createVector } from "./VectorMath.js"
|
|
2
|
+
|
|
3
|
+
export function medianOf5Filter(points: number[]) {
|
|
4
|
+
// This function computes the moving median with a window of 5 elements.
|
|
5
|
+
|
|
6
|
+
// I initialized the window such that at the edges of the range no median would be computed.
|
|
7
|
+
// This is a form of optimization assuming that computing a median edge points
|
|
8
|
+
// is less important.
|
|
9
|
+
|
|
10
|
+
const pointCount = points.length
|
|
11
|
+
|
|
12
|
+
if (pointCount < 5) {
|
|
13
|
+
return points
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
const medians = createVector(pointCount)
|
|
17
|
+
|
|
18
|
+
medians[0] = points[0]
|
|
19
|
+
medians[1] = points[1]
|
|
20
|
+
medians[pointCount - 2] = points[pointCount - 2]
|
|
21
|
+
medians[pointCount - 1] = points[pointCount - 1]
|
|
22
|
+
|
|
23
|
+
for (let i = 2; i < pointCount - 2; i++) {
|
|
24
|
+
medians[i] = medianOf5(points[i - 2], points[i - 1], points[i], points[i + 1], points[i + 2])
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
return medians
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function medianOf3Filter(points: ArrayLike<number>) {
|
|
31
|
+
// This function computes the moving median with a window of 3 elements.
|
|
32
|
+
|
|
33
|
+
// I initialized the window such that at the edges of the range no median would be computed.
|
|
34
|
+
|
|
35
|
+
const pointCount = points.length
|
|
36
|
+
|
|
37
|
+
if (pointCount < 3) {
|
|
38
|
+
return points
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
const medians: number[] = createVector(pointCount)
|
|
42
|
+
|
|
43
|
+
medians[0] = points[0]
|
|
44
|
+
medians[pointCount - 1] = points[pointCount - 1]
|
|
45
|
+
|
|
46
|
+
for (let i = 1; i < pointCount - 1; i++) {
|
|
47
|
+
medians[i] = medianOf3(points[i - 1], points[i], points[i + 1])
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
return medians
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export function medianOf5(a: number, b: number, c: number, d: number, e: number) {
|
|
54
|
+
// These swapping computation should be faster than separately using the minimum and maximum
|
|
55
|
+
// functions but maybe less readable.
|
|
56
|
+
|
|
57
|
+
// Ensure b is greater or equal to a (swap if needed)
|
|
58
|
+
if (b < a) {
|
|
59
|
+
[a, b] = [b, a]
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Ensure d is greater or equal to c (swap if needed)
|
|
63
|
+
if (d < c) {
|
|
64
|
+
[c, d] = [d, c]
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// What this part does is compute the two middle medians of the first 4 elements
|
|
68
|
+
// given to the function (a, b, c, d), but it doesn't actually determine their relative order:
|
|
69
|
+
const firstMedianOfABCD = Math.max(a, c) // First median of a, b, c, d
|
|
70
|
+
const secondMedianOfABCD = Math.min(b, d) // Second median of a, b, c, d
|
|
71
|
+
|
|
72
|
+
// Now in relation to all five numbers, the median can only be either
|
|
73
|
+
// the first median of ABCD, the second median of ABCD, or E:
|
|
74
|
+
return medianOf3(firstMedianOfABCD, secondMedianOfABCD, e)
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
export function medianOf3(a: number, b: number, c: number) {
|
|
78
|
+
// This function uses a decision tree to find the median of three numbers.
|
|
79
|
+
//
|
|
80
|
+
// I tried to ensure that the comparison preserved the natural altering of
|
|
81
|
+
// a, b and c such that in case that they are given already in order,
|
|
82
|
+
// then all the initial branches would be directly taken.
|
|
83
|
+
|
|
84
|
+
// Possible orderings:
|
|
85
|
+
//
|
|
86
|
+
// a, b, c
|
|
87
|
+
// a, c, b
|
|
88
|
+
// b, a, c
|
|
89
|
+
// b, c, a
|
|
90
|
+
// c, a, b
|
|
91
|
+
// c, b, a
|
|
92
|
+
|
|
93
|
+
if (a <= b) {
|
|
94
|
+
if (b <= c) {
|
|
95
|
+
return b // a, b, c
|
|
96
|
+
} else if (a <= c) {
|
|
97
|
+
return c // a, c, b
|
|
98
|
+
} else {
|
|
99
|
+
return a // c, a, b
|
|
100
|
+
}
|
|
101
|
+
} else {
|
|
102
|
+
if (a <= c) {
|
|
103
|
+
return a // b, a, c
|
|
104
|
+
} else if (b <= c) {
|
|
105
|
+
return c // b, c, a
|
|
106
|
+
} else {
|
|
107
|
+
return b // c, b, a
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// Slower, variable-width median filter using the `moving-median` package
|
|
113
|
+
export async function medianFilter(points: number[], width: number) {
|
|
114
|
+
const { default: createMedianFilter } = await import('moving-median')
|
|
115
|
+
|
|
116
|
+
const filter = createMedianFilter(width)
|
|
117
|
+
const result = []
|
|
118
|
+
|
|
119
|
+
for (let i = 0; i < points.length; i++) {
|
|
120
|
+
result.push(filter(points[i]))
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
return result
|
|
124
|
+
}
|