echogarden 1.0.4 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/README.md +26 -23
  2. package/data/schemas/options.json +177 -36
  3. package/dist/alignment/SpeechAlignment.d.ts +1 -1
  4. package/dist/alignment/SpeechAlignment.js +1 -1
  5. package/dist/alignment/SpeechAlignment.js.map +1 -1
  6. package/dist/api/API.d.ts +1 -0
  7. package/dist/api/API.js +1 -0
  8. package/dist/api/API.js.map +1 -1
  9. package/dist/api/APIOptions.d.ts +1 -0
  10. package/dist/api/Alignment.d.ts +3 -3
  11. package/dist/api/Alignment.js +5 -10
  12. package/dist/api/Alignment.js.map +1 -1
  13. package/dist/api/LanguageDetection.d.ts +5 -7
  14. package/dist/api/LanguageDetection.js +3 -2
  15. package/dist/api/LanguageDetection.js.map +1 -1
  16. package/dist/api/Recognition.d.ts +4 -5
  17. package/dist/api/Recognition.js +5 -8
  18. package/dist/api/Recognition.js.map +1 -1
  19. package/dist/api/SourceSeparation.d.ts +2 -0
  20. package/dist/api/SourceSeparation.js +4 -2
  21. package/dist/api/SourceSeparation.js.map +1 -1
  22. package/dist/api/Synthesis.d.ts +3 -1
  23. package/dist/api/Synthesis.js +9 -10
  24. package/dist/api/Synthesis.js.map +1 -1
  25. package/dist/api/Translation.d.ts +1 -1
  26. package/dist/api/Translation.js +4 -8
  27. package/dist/api/Translation.js.map +1 -1
  28. package/dist/api/TranslationAlignment.d.ts +31 -0
  29. package/dist/api/TranslationAlignment.js +121 -0
  30. package/dist/api/TranslationAlignment.js.map +1 -0
  31. package/dist/api/VoiceActivityDetection.d.ts +5 -1
  32. package/dist/api/VoiceActivityDetection.js +38 -2
  33. package/dist/api/VoiceActivityDetection.js.map +1 -1
  34. package/dist/audio/AudioPlayer.js +6 -1
  35. package/dist/audio/AudioPlayer.js.map +1 -1
  36. package/dist/cli/CLI.js +85 -0
  37. package/dist/cli/CLI.js.map +1 -1
  38. package/dist/dsp/FFT.js.map +1 -1
  39. package/dist/math/MedianFilter.d.ts +5 -0
  40. package/dist/math/MedianFilter.js +102 -0
  41. package/dist/math/MedianFilter.js.map +1 -0
  42. package/dist/math/VectorMath.d.ts +0 -2
  43. package/dist/math/VectorMath.js +1 -25
  44. package/dist/math/VectorMath.js.map +1 -1
  45. package/dist/recognition/OpenAICloudSTT.d.ts +1 -1
  46. package/dist/recognition/OpenAICloudSTT.js.map +1 -1
  47. package/dist/recognition/SileroSTT.d.ts +22 -1
  48. package/dist/recognition/SileroSTT.js +122 -95
  49. package/dist/recognition/SileroSTT.js.map +1 -1
  50. package/dist/recognition/WhisperCppSTT.js +1 -1
  51. package/dist/recognition/WhisperCppSTT.js.map +1 -1
  52. package/dist/recognition/WhisperSTT.d.ts +52 -19
  53. package/dist/recognition/WhisperSTT.js +645 -494
  54. package/dist/recognition/WhisperSTT.js.map +1 -1
  55. package/dist/server/Server.js.map +1 -1
  56. package/dist/source-separation/MDXNetSourceSeparation.d.ts +5 -3
  57. package/dist/source-separation/MDXNetSourceSeparation.js +26 -19
  58. package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
  59. package/dist/speech-language-detection/SileroLanguageDetection.d.ts +15 -9
  60. package/dist/speech-language-detection/SileroLanguageDetection.js +23 -16
  61. package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
  62. package/dist/synthesis/EspeakTTS.js +4 -0
  63. package/dist/synthesis/EspeakTTS.js.map +1 -1
  64. package/dist/synthesis/GoogleCloudTTS.js.map +1 -1
  65. package/dist/synthesis/VitsTTS.d.ts +8 -6
  66. package/dist/synthesis/VitsTTS.js +36 -31
  67. package/dist/synthesis/VitsTTS.js.map +1 -1
  68. package/dist/tests/Test.js.map +1 -1
  69. package/dist/utilities/OnnxUtilities.d.ts +14 -0
  70. package/dist/utilities/OnnxUtilities.js +43 -0
  71. package/dist/utilities/OnnxUtilities.js.map +1 -0
  72. package/dist/utilities/Utilities.d.ts +4 -8
  73. package/dist/utilities/Utilities.js +35 -58
  74. package/dist/utilities/Utilities.js.map +1 -1
  75. package/dist/voice-activity-detection/SileroVAD.d.ts +5 -3
  76. package/dist/voice-activity-detection/SileroVAD.js +9 -11
  77. package/dist/voice-activity-detection/SileroVAD.js.map +1 -1
  78. package/docs/API.md +54 -34
  79. package/docs/CLI.md +25 -13
  80. package/docs/Contributing.md +4 -2
  81. package/docs/Engines.md +43 -32
  82. package/docs/Licenses.md +3 -4
  83. package/docs/Options.md +47 -11
  84. package/docs/Releases.md +4 -0
  85. package/docs/Server.md +8 -6
  86. package/docs/Tasklist.md +39 -52
  87. package/docs/Technical.md +1 -1
  88. package/package.json +8 -12
  89. package/src/alignment/SpeechAlignment.ts +1 -1
  90. package/src/api/API.ts +1 -0
  91. package/src/api/APIOptions.ts +1 -0
  92. package/src/api/Alignment.ts +10 -14
  93. package/src/api/LanguageDetection.ts +14 -10
  94. package/src/api/Recognition.ts +17 -10
  95. package/src/api/SourceSeparation.ts +7 -2
  96. package/src/api/Synthesis.ts +26 -11
  97. package/src/api/Translation.ts +14 -8
  98. package/src/api/TranslationAlignment.ts +213 -0
  99. package/src/api/VoiceActivityDetection.ts +66 -3
  100. package/src/audio/AudioPlayer.ts +6 -2
  101. package/src/cli/CLI.ts +121 -2
  102. package/src/dsp/FFT.ts +3 -0
  103. package/src/math/MedianFilter.ts +124 -0
  104. package/src/math/VectorMath.ts +1 -36
  105. package/src/recognition/OpenAICloudSTT.ts +27 -27
  106. package/src/recognition/SileroSTT.ts +149 -102
  107. package/src/recognition/WhisperCppSTT.ts +1 -1
  108. package/src/recognition/WhisperSTT.ts +961 -684
  109. package/src/server/Server.ts +1 -1
  110. package/src/source-separation/MDXNetSourceSeparation.ts +35 -19
  111. package/src/speech-language-detection/SileroLanguageDetection.ts +53 -33
  112. package/src/synthesis/EspeakTTS.ts +8 -0
  113. package/src/synthesis/GoogleCloudTTS.ts +12 -1
  114. package/src/synthesis/VitsTTS.ts +57 -46
  115. package/src/tests/Test.ts +1 -1
  116. package/src/utilities/OnnxUtilities.ts +68 -0
  117. package/src/utilities/Utilities.ts +38 -66
  118. package/src/voice-activity-detection/SileroVAD.ts +15 -15
  119. package/dist/utilities/NdArrayUtilities.d.ts +0 -3
  120. package/dist/utilities/NdArrayUtilities.js +0 -23
  121. package/dist/utilities/NdArrayUtilities.js.map +0 -1
  122. package/src/utilities/NdArrayUtilities.ts +0 -31
@@ -0,0 +1,213 @@
1
+ import { extendDeep } from '../utilities/ObjectUtilities.js'
2
+
3
+ import { logToStderr } from '../utilities/Utilities.js'
4
+ import { AudioSourceParam, RawAudio, ensureRawAudio, normalizeAudioLevel, trimAudioEnd } from '../audio/AudioUtilities.js'
5
+ import { Logger } from '../utilities/Logger.js'
6
+
7
+ import * as API from './API.js'
8
+ import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
9
+ import { formatLanguageCodeWithName, getShortLanguageCode, normalizeLanguageCode } from '../utilities/Locale.js'
10
+ import { type WhisperAlignmentOptions } from '../recognition/WhisperSTT.js'
11
+ import chalk from 'chalk'
12
+ import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
13
+
14
+ const log = logToStderr
15
+
16
+ export async function alignTranslation(input: AudioSourceParam, transcript: string, options: TranslationAlignmentOptions): Promise<TranslationAlignmentResult> {
17
+ const logger = new Logger()
18
+
19
+ const startTimestamp = logger.getTimestamp()
20
+
21
+ options = extendDeep(defaultTranslationAlignmentOptions, options)
22
+
23
+ const inputRawAudio = await ensureRawAudio(input)
24
+
25
+ let sourceRawAudio: RawAudio
26
+ let isolatedRawAudio: RawAudio | undefined
27
+ let backgroundRawAudio: RawAudio | undefined
28
+
29
+ if (options.isolate) {
30
+ logger.log(``)
31
+ logger.end();
32
+
33
+ ({ isolatedRawAudio, backgroundRawAudio } = await API.isolate(inputRawAudio, options.sourceSeparation!))
34
+
35
+ logger.end()
36
+ logger.log(``)
37
+
38
+ sourceRawAudio = await ensureRawAudio(isolatedRawAudio, 16000, 1)
39
+ } else {
40
+ sourceRawAudio = await ensureRawAudio(inputRawAudio, 16000, 1)
41
+ }
42
+
43
+ let sourceUncropTimeline: Timeline | undefined
44
+
45
+ if (options.crop) {
46
+ logger.start('Crop using voice activity detection');
47
+ ({ timeline: sourceUncropTimeline, croppedRawAudio: sourceRawAudio } = await API.detectVoiceActivity(sourceRawAudio, options.vad!))
48
+
49
+ logger.end()
50
+ }
51
+
52
+ logger.start('Prepare for alignment')
53
+
54
+ sourceRawAudio = normalizeAudioLevel(sourceRawAudio)
55
+ sourceRawAudio.audioChannels[0] = trimAudioEnd(sourceRawAudio.audioChannels[0])
56
+
57
+ let sourceLanguage: string
58
+
59
+ if (options.sourceLanguage) {
60
+ sourceLanguage = normalizeLanguageCode(options.sourceLanguage!)
61
+ } else {
62
+ logger.start('No source language specified. Detecting speech language')
63
+ const { detectedLanguage } = await API.detectSpeechLanguage(sourceRawAudio, options.languageDetection || {})
64
+
65
+ logger.end()
66
+ logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
67
+
68
+ sourceLanguage = detectedLanguage
69
+ }
70
+
71
+ const targetLanguage = normalizeLanguageCode(options.targetLanguage!)
72
+
73
+ let mappedTimeline: Timeline
74
+
75
+ switch (options.engine) {
76
+ case 'whisper': {
77
+ const WhisperSTT = await import('../recognition/WhisperSTT.js')
78
+
79
+ const shortSourceLanguageCode = getShortLanguageCode(sourceLanguage)
80
+ const shortTargetLanguageCode = getShortLanguageCode(targetLanguage)
81
+
82
+ if (shortTargetLanguageCode != 'en') {
83
+ throw new Error('Whisper translation only supports English as target language')
84
+ }
85
+
86
+ if (shortSourceLanguageCode == 'en' && shortTargetLanguageCode == 'en') {
87
+ throw new Error('Both translation source and target languages are English')
88
+ }
89
+
90
+ const whisperAlignmnentOptions = options.whisper!
91
+
92
+ const { modelName, modelDir } = await WhisperSTT.loadPackagesAndGetPaths(whisperAlignmnentOptions.model, shortSourceLanguageCode)
93
+
94
+ logger.end()
95
+
96
+ if (modelName.endsWith('.en')) {
97
+ throw new Error('Whisper translation tasks are only possible with a multilingual model')
98
+ }
99
+
100
+ mappedTimeline = await WhisperSTT.alignEnglishTranslation(sourceRawAudio, transcript, modelName, modelDir, shortSourceLanguageCode, whisperAlignmnentOptions)
101
+
102
+ break
103
+ }
104
+
105
+ default: {
106
+ throw new Error(`Engine '${options.engine}' is not supported`)
107
+ }
108
+ }
109
+
110
+ // If the audio was cropped before recognition, map the timestamps back to the original audio
111
+ if (sourceUncropTimeline && sourceUncropTimeline.length > 0) {
112
+ API.convertCroppedToUncroppedTimeline(mappedTimeline, sourceUncropTimeline)
113
+ }
114
+
115
+ // Add text offsets
116
+ addWordTextOffsetsToTimeline(mappedTimeline, transcript)
117
+
118
+ // Make segment timeline
119
+ const { segmentTimeline } = await wordTimelineToSegmentSentenceTimeline(mappedTimeline, transcript, sourceLanguage, options.plainText?.paragraphBreaks, options.plainText?.whitespace)
120
+
121
+ logger.end()
122
+ logger.logDuration(`Total translation alignment time`, startTimestamp, chalk.magentaBright)
123
+
124
+ return {
125
+ timeline: segmentTimeline,
126
+ wordTimeline: mappedTimeline,
127
+
128
+ transcript,
129
+ language: sourceLanguage,
130
+
131
+ inputRawAudio,
132
+ isolatedRawAudio,
133
+ backgroundRawAudio,
134
+ }
135
+ }
136
+
137
+ export interface TranslationAlignmentResult {
138
+ timeline: Timeline
139
+ wordTimeline: Timeline
140
+
141
+ transcript: string
142
+ language: string
143
+
144
+ inputRawAudio: RawAudio
145
+ isolatedRawAudio?: RawAudio
146
+ backgroundRawAudio?: RawAudio
147
+ }
148
+
149
+ export type TranslationAlignmentEngine = 'whisper'
150
+
151
+ export interface TranslationAlignmentOptions {
152
+ engine?: TranslationAlignmentEngine
153
+
154
+ sourceLanguage?: string
155
+ targetLanguage?: string
156
+
157
+ isolate?: boolean
158
+
159
+ crop?: boolean
160
+
161
+ languageDetection?: API.SpeechLanguageDetectionOptions
162
+
163
+ vad?: API.VADOptions
164
+
165
+ plainText?: API.PlainTextOptions
166
+
167
+ subtitles?: SubtitlesConfig
168
+
169
+ sourceSeparation?: API.SourceSeparationOptions
170
+
171
+ whisper?: WhisperAlignmentOptions
172
+ }
173
+
174
+ export const defaultTranslationAlignmentOptions: TranslationAlignmentOptions = {
175
+ engine: 'whisper',
176
+
177
+ sourceLanguage: undefined,
178
+ targetLanguage: 'en',
179
+
180
+ isolate: false,
181
+
182
+ crop: true,
183
+
184
+ languageDetection: {
185
+ },
186
+
187
+ plainText: {
188
+ paragraphBreaks: 'double',
189
+ whitespace: 'collapse'
190
+ },
191
+
192
+ subtitles: {
193
+ },
194
+
195
+ vad: {
196
+ engine: 'adaptive-gate'
197
+ },
198
+
199
+ sourceSeparation: {
200
+ },
201
+
202
+ whisper: {
203
+ }
204
+ }
205
+
206
+ export const translationAlignmentEngines: API.EngineMetadata[] = [
207
+ {
208
+ id: 'whisper',
209
+ name: 'OpenAI Whisper',
210
+ description: 'Extracts timestamps by guiding the Whisper recognition model to recognize the translated transcript tokens.',
211
+ type: 'local'
212
+ }
213
+ ]
@@ -10,6 +10,8 @@ import { loadPackage } from '../utilities/PackageManager.js'
10
10
  import { EngineMetadata } from './Common.js'
11
11
  import chalk from 'chalk'
12
12
  import { type AdaptiveGateVADOptions } from '../voice-activity-detection/AdaptiveGateVAD.js'
13
+ import { type WhisperVADOptions } from '../recognition/WhisperSTT.js'
14
+ import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js'
13
15
 
14
16
  const log = logToStderr
15
17
 
@@ -56,7 +58,14 @@ export async function detectVoiceActivity(input: AudioSourceParam, options: VADO
56
58
  const modelPath = path.join(modelDir, 'silero-vad.onnx')
57
59
  const frameDuration = sileroOptions.frameDuration!
58
60
 
59
- const frameProbabilities = await SileroVAD.detectVoiceActivity(sourceRawAudio, modelPath, frameDuration)
61
+ const onnxExecutionProviders: OnnxExecutionProvider[] = sileroOptions.provider ? [sileroOptions.provider] : []
62
+
63
+ const frameProbabilities = await SileroVAD.detectVoiceActivity(
64
+ sourceRawAudio,
65
+ modelPath,
66
+ frameDuration,
67
+ onnxExecutionProviders)
68
+
60
69
  const frameDurationSeconds = sileroOptions.frameDuration! / 1000
61
70
 
62
71
  verboseTimeline = frameProbabilitiesToTimeline(frameProbabilities, frameDurationSeconds, activityThreshold)
@@ -81,6 +90,46 @@ export async function detectVoiceActivity(input: AudioSourceParam, options: VADO
81
90
  break
82
91
  }
83
92
 
93
+ case 'whisper': {
94
+ const WhisperSTT = await import('../recognition/WhisperSTT.js')
95
+
96
+ const whisperVADOptions = options.whisper!
97
+
98
+ logger.end()
99
+
100
+ const { modelName, modelDir } = await WhisperSTT.loadPackagesAndGetPaths(whisperVADOptions.model, 'de')
101
+
102
+ logger.end();
103
+
104
+ const { partProbabilities } = await WhisperSTT.detectVoiceActivity(
105
+ sourceRawAudio,
106
+ modelName,
107
+ modelDir,
108
+ whisperVADOptions,
109
+ )
110
+
111
+ verboseTimeline = []
112
+
113
+ for (const entry of partProbabilities) {
114
+ const hasSpeech = entry.confidence! >= activityThreshold
115
+
116
+ const text = hasSpeech ? 'active' : 'inactive'
117
+
118
+ if (verboseTimeline.length === 0 || verboseTimeline[verboseTimeline.length - 1].text != text) {
119
+ verboseTimeline.push({
120
+ type: 'segment',
121
+ text,
122
+ startTime: entry.startTime,
123
+ endTime: entry.endTime
124
+ })
125
+ } else {
126
+ verboseTimeline[verboseTimeline.length - 1].endTime = entry.endTime
127
+ }
128
+ }
129
+
130
+ break
131
+ }
132
+
84
133
  case 'adaptive-gate': {
85
134
  const AdaptiveGateVAD = await import('../voice-activity-detection/AdaptiveGateVAD.js')
86
135
 
@@ -104,7 +153,12 @@ export async function detectVoiceActivity(input: AudioSourceParam, options: VADO
104
153
  logger.log('')
105
154
  logger.logDuration(`Total voice activity detection time`, startTimestamp, chalk.magentaBright)
106
155
 
107
- return { timeline, verboseTimeline, inputRawAudio, croppedRawAudio }
156
+ return {
157
+ timeline,
158
+ verboseTimeline,
159
+ inputRawAudio,
160
+ croppedRawAudio
161
+ }
108
162
  }
109
163
 
110
164
  function frameProbabilitiesToTimeline(frameProbabilities: number[], frameDurationSeconds: number, activityThreshold: number) {
@@ -176,7 +230,7 @@ export interface VADResult {
176
230
  croppedRawAudio: RawAudio
177
231
  }
178
232
 
179
- export type VADEngine = 'webrtc' | 'silero' | 'rnnoise' | 'adaptive-gate'
233
+ export type VADEngine = 'webrtc' | 'silero' | 'rnnoise' | 'whisper' | 'adaptive-gate'
180
234
 
181
235
  export interface VADOptions {
182
236
  engine?: VADEngine
@@ -190,11 +244,14 @@ export interface VADOptions {
190
244
 
191
245
  silero?: {
192
246
  frameDuration?: 30 | 60 | 90
247
+ provider?: OnnxExecutionProvider
193
248
  }
194
249
 
195
250
  rnnoise?: {
196
251
  }
197
252
 
253
+ whisper?: WhisperVADOptions
254
+
198
255
  adaptiveGate?: AdaptiveGateVADOptions
199
256
  }
200
257
 
@@ -210,11 +267,17 @@ export const defaultVADOptions: VADOptions = {
210
267
 
211
268
  silero: {
212
269
  frameDuration: 90,
270
+ provider: undefined,
213
271
  },
214
272
 
215
273
  rnnoise: {
216
274
  },
217
275
 
276
+ whisper: {
277
+ model: 'tiny',
278
+ temperature: 1.0,
279
+ },
280
+
218
281
  adaptiveGate: {
219
282
  }
220
283
  }
@@ -355,5 +355,9 @@ export function playAudioSamples_Speaker(rawAudio: RawAudio, onTimePosition?: (t
355
355
  })
356
356
  }
357
357
 
358
- export const charactersToWriteAhead =
359
- [',', '.', ',', '、', ':', ';', '。', ':', ';', '?', '!', ')', ']', '}', `"`, `'`, '”', '’', '-', '—', '»']
358
+ export const charactersToWriteAhead = [
359
+ ',', '.', ',', '、', ':', ';',
360
+ '。', ':', ';', '?', '?', '!', '!',
361
+ ')', ']', '}', `"`, `'`, '”', '’',
362
+ '-', '—', '»', '،', '؟'
363
+ ]
package/src/cli/CLI.ts CHANGED
@@ -13,8 +13,8 @@ import { encodeFromChannels, getDefaultFFMpegOptionsForSpeech } from '../codecs/
13
13
  import path, { parse as parsePath } from 'node:path'
14
14
  import { splitToParagraphs, splitToWords, wordCharacterPattern } from '../nlp/Segmentation.js'
15
15
  import { playAudioSamples, playAudioWithWordTimeline } from '../audio/AudioPlayer.js'
16
- import { deepClone, extendDeep } from '../utilities/ObjectUtilities.js'
17
- import { Timeline, TimelineEntry, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, roundTimelineProperties, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
16
+ import { extendDeep } from '../utilities/ObjectUtilities.js'
17
+ import { Timeline, TimelineEntry, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, roundTimelineProperties } from '../utilities/Timeline.js'
18
18
  import { ensureDir, existsSync, readAndParseJsonFile, readFile, readdir, writeFileSafe } from '../utilities/FileSystem.js'
19
19
  import { formatLanguageCodeWithName, getShortLanguageCode } from '../utilities/Locale.js'
20
20
  import { APIOptions } from '../api/APIOptions.js'
@@ -167,6 +167,8 @@ const commandHelp = [
167
167
  ` Align audio file to the reference transcript file\n`,
168
168
  `${executableName} ${chalk.magentaBright('translate-speech')} inputFile [output files...] [options...]`,
169
169
  ` Transcribe audio file directly to a different language\n`,
170
+ `${executableName} ${chalk.magentaBright('align-translation')} audioFile referenceFile [output files...] [options...]`,
171
+ ` Align audio file to the reference translated transcript file\n`,
170
172
  `${executableName} ${chalk.magentaBright('detect-speech-language')} audioFile [output files...] [options...]`,
171
173
  ` Detect language of audio file\n`,
172
174
  `${executableName} ${chalk.magentaBright('detect-text-language')} inputFile [output files...] [options...]`,
@@ -220,6 +222,11 @@ async function startWithArgs(parsedArgs: CLIArguments) {
220
222
  break
221
223
  }
222
224
 
225
+ case 'align-translation': {
226
+ await alignTranslation(parsedArgs.commandArgs, parsedArgs.options)
227
+ break
228
+ }
229
+
223
230
  case 'detect-language': {
224
231
  await detectLanguage(parsedArgs.commandArgs, parsedArgs.options, 'auto')
225
232
  break
@@ -607,6 +614,112 @@ async function align(commandArgs: string[], cliOptions: Map<string, string>) {
607
614
  }
608
615
  }
609
616
 
617
+ async function alignTranslation(commandArgs: string[], cliOptions: Map<string, string>) {
618
+ const logger = new Logger()
619
+
620
+ const audioFilename = commandArgs[0]
621
+ const outputFilenames = commandArgs.slice(2)
622
+
623
+ if (audioFilename == undefined) {
624
+ throw new Error(`align-translation requires an argument containing the audio file path.`)
625
+ }
626
+
627
+ if (!existsSync(audioFilename)) {
628
+ throw new Error(`The given source file '${audioFilename}' was not found.`)
629
+ }
630
+
631
+ const alignmentReferenceFile = commandArgs[1]
632
+
633
+ if (alignmentReferenceFile == undefined) {
634
+ throw new Error(`align-translation requires a second argument containing the translated reference file path.`)
635
+ }
636
+
637
+ if (!existsSync(alignmentReferenceFile)) {
638
+ throw new Error(`The given reference file '${alignmentReferenceFile}' was not found.`)
639
+ }
640
+
641
+ const referenceFileExtension = getLowercaseFileExtension(alignmentReferenceFile)
642
+ const fileContent = await readFile(alignmentReferenceFile, { encoding: 'utf-8' })
643
+
644
+ let text: string
645
+
646
+ if (referenceFileExtension == 'txt') {
647
+ text = fileContent
648
+ } else if (referenceFileExtension == 'html' || referenceFileExtension == 'htm') {
649
+ text = await convertHtmlToText(fileContent)
650
+ } else if (referenceFileExtension == 'srt' || referenceFileExtension == 'vtt') {
651
+ text = subtitlesToText(fileContent)
652
+ } else {
653
+ throw new Error(`align only supports reference files with extensions 'txt', 'html', 'htm', 'srt' or 'vtt'`)
654
+ }
655
+
656
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
657
+ additionalOptionsSchema.set('play', { type: 'boolean' })
658
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
659
+
660
+ if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
661
+ cliOptions.set('play', `${outputFilenames.length == 0}`)
662
+ }
663
+
664
+ const options: API.TranslationAlignmentOptions = await cliOptionsMapToOptionsObject(cliOptions, 'TranslationAlignmentOptions', additionalOptionsSchema)
665
+
666
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
667
+ const { includesPlaceholderPattern } = await checkOutputFilenames(outputFilenames, true, true, true)
668
+
669
+ const {
670
+ timeline,
671
+ wordTimeline,
672
+ transcript,
673
+ language,
674
+ inputRawAudio,
675
+ isolatedRawAudio,
676
+ backgroundRawAudio } = await API.alignTranslation(audioFilename, text, options)
677
+
678
+ if (outputFilenames.length > 0) {
679
+ logger.start('\nWrite output files')
680
+ }
681
+
682
+ if (includesPlaceholderPattern) {
683
+ for (let segmentIndex = 0; segmentIndex < timeline.length; segmentIndex++) {
684
+ const segmentEntry = timeline[segmentIndex]
685
+ const segmentAudio = sliceRawAudioByTime(inputRawAudio, segmentEntry.startTime, segmentEntry.endTime)
686
+ const sentenceTimeline = addTimeOffsetToTimeline(segmentEntry.timeline!, -segmentEntry.startTime)
687
+
688
+ await writeOutputFilesForSegment(outputFilenames, segmentIndex, timeline.length, segmentAudio, sentenceTimeline, segmentEntry.text, language, allowOverwrite)
689
+ }
690
+ }
691
+
692
+ for (const outputFilename of outputFilenames) {
693
+ const partPatternMatch = outputFilename.match(filenamePlaceholderPattern)
694
+
695
+ if (partPatternMatch) {
696
+ continue
697
+ }
698
+
699
+ const fileSaver = getFileSaver(outputFilename, allowOverwrite)
700
+
701
+ await fileSaver(inputRawAudio, timeline, transcript, options.subtitles)
702
+
703
+ await writeSourceSeparationOutputIfNeeded(outputFilename, isolatedRawAudio, backgroundRawAudio, allowOverwrite, true)
704
+ }
705
+
706
+ logger.end()
707
+
708
+ if ((options as any).play) {
709
+ let audioToPlay: RawAudio
710
+
711
+ if (isolatedRawAudio) {
712
+ audioToPlay = isolatedRawAudio
713
+ } else {
714
+ audioToPlay = inputRawAudio
715
+ }
716
+
717
+ const normalizedAudioToPlay = normalizeAudioLevel(audioToPlay)
718
+
719
+ await playAudioWithWordTimeline(normalizedAudioToPlay, wordTimeline, transcript)
720
+ }
721
+ }
722
+
610
723
  async function translateSpeech(commandArgs: string[], cliOptions: Map<string, string>) {
611
724
  const logger = new Logger()
612
725
 
@@ -976,6 +1089,12 @@ async function listEngines(commandArgs: string[], cliOptions: Map<string, string
976
1089
  break
977
1090
  }
978
1091
 
1092
+ case 'align-translation': {
1093
+ engines = API.translationAlignmentEngines
1094
+
1095
+ break
1096
+ }
1097
+
979
1098
  case 'translate-speech': {
980
1099
  engines = API.speechTranslationEngines
981
1100
 
package/src/dsp/FFT.ts CHANGED
@@ -48,6 +48,7 @@ export async function stftr(samples: Float32Array, fftOrder: number, windowSize:
48
48
  }
49
49
 
50
50
  binsBufferRef.clear()
51
+
51
52
  m._kiss_fftr(statePtr, frameBufferRef.address, binsBufferRef.address)
52
53
 
53
54
  const bins = binsBufferRef.view.slice(0, fftOrder + 2)
@@ -103,7 +104,9 @@ export async function stiftr(binsForFrames: Float32Array[], fftOrder: number, wi
103
104
  binsRef.view.set(binsForFrame)
104
105
 
105
106
  frameBufferRef.clear()
107
+
106
108
  m._kiss_fftri(statePtr, binsRef.address, frameBufferRef.address)
109
+
107
110
  const frameSamples = frameBufferRef.view
108
111
 
109
112
  const frameStartOffset = frameIndex * hopSize
@@ -0,0 +1,124 @@
1
+ import { createVector } from "./VectorMath.js"
2
+
3
+ export function medianOf5Filter(points: number[]) {
4
+ // This function computes the moving median with a window of 5 elements.
5
+
6
+ // I initialized the window such that at the edges of the range no median would be computed.
7
+ // This is a form of optimization assuming that computing a median edge points
8
+ // is less important.
9
+
10
+ const pointCount = points.length
11
+
12
+ if (pointCount < 5) {
13
+ return points
14
+ }
15
+
16
+ const medians = createVector(pointCount)
17
+
18
+ medians[0] = points[0]
19
+ medians[1] = points[1]
20
+ medians[pointCount - 2] = points[pointCount - 2]
21
+ medians[pointCount - 1] = points[pointCount - 1]
22
+
23
+ for (let i = 2; i < pointCount - 2; i++) {
24
+ medians[i] = medianOf5(points[i - 2], points[i - 1], points[i], points[i + 1], points[i + 2])
25
+ }
26
+
27
+ return medians
28
+ }
29
+
30
+ export function medianOf3Filter(points: ArrayLike<number>) {
31
+ // This function computes the moving median with a window of 3 elements.
32
+
33
+ // I initialized the window such that at the edges of the range no median would be computed.
34
+
35
+ const pointCount = points.length
36
+
37
+ if (pointCount < 3) {
38
+ return points
39
+ }
40
+
41
+ const medians: number[] = createVector(pointCount)
42
+
43
+ medians[0] = points[0]
44
+ medians[pointCount - 1] = points[pointCount - 1]
45
+
46
+ for (let i = 1; i < pointCount - 1; i++) {
47
+ medians[i] = medianOf3(points[i - 1], points[i], points[i + 1])
48
+ }
49
+
50
+ return medians
51
+ }
52
+
53
+ export function medianOf5(a: number, b: number, c: number, d: number, e: number) {
54
+ // These swapping computation should be faster than separately using the minimum and maximum
55
+ // functions but maybe less readable.
56
+
57
+ // Ensure b is greater or equal to a (swap if needed)
58
+ if (b < a) {
59
+ [a, b] = [b, a]
60
+ }
61
+
62
+ // Ensure d is greater or equal to c (swap if needed)
63
+ if (d < c) {
64
+ [c, d] = [d, c]
65
+ }
66
+
67
+ // What this part does is compute the two middle medians of the first 4 elements
68
+ // given to the function (a, b, c, d), but it doesn't actually determine their relative order:
69
+ const firstMedianOfABCD = Math.max(a, c) // First median of a, b, c, d
70
+ const secondMedianOfABCD = Math.min(b, d) // Second median of a, b, c, d
71
+
72
+ // Now in relation to all five numbers, the median can only be either
73
+ // the first median of ABCD, the second median of ABCD, or E:
74
+ return medianOf3(firstMedianOfABCD, secondMedianOfABCD, e)
75
+ }
76
+
77
+ export function medianOf3(a: number, b: number, c: number) {
78
+ // This function uses a decision tree to find the median of three numbers.
79
+ //
80
+ // I tried to ensure that the comparison preserved the natural altering of
81
+ // a, b and c such that in case that they are given already in order,
82
+ // then all the initial branches would be directly taken.
83
+
84
+ // Possible orderings:
85
+ //
86
+ // a, b, c
87
+ // a, c, b
88
+ // b, a, c
89
+ // b, c, a
90
+ // c, a, b
91
+ // c, b, a
92
+
93
+ if (a <= b) {
94
+ if (b <= c) {
95
+ return b // a, b, c
96
+ } else if (a <= c) {
97
+ return c // a, c, b
98
+ } else {
99
+ return a // c, a, b
100
+ }
101
+ } else {
102
+ if (a <= c) {
103
+ return a // b, a, c
104
+ } else if (b <= c) {
105
+ return c // b, c, a
106
+ } else {
107
+ return b // c, b, a
108
+ }
109
+ }
110
+ }
111
+
112
+ // Slower, variable-width median filter using the `moving-median` package
113
+ export async function medianFilter(points: number[], width: number) {
114
+ const { default: createMedianFilter } = await import('moving-median')
115
+
116
+ const filter = createMedianFilter(width)
117
+ const result = []
118
+
119
+ for (let i = 0; i < points.length; i++) {
120
+ result.push(filter(points[i]))
121
+ }
122
+
123
+ return result
124
+ }