echogarden 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +43 -6
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWMfccSequenceAlignment.js +5 -5
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +1 -3
- package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
- package/dist/alignment/SemanticTextAlignment.js +1 -1
- package/dist/alignment/SemanticTextAlignment.js.map +1 -1
- package/dist/alignment/SpeechAlignment.d.ts +2 -2
- package/dist/alignment/SpeechAlignment.js +26 -3
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +0 -1
- package/dist/api/Synthesis.d.ts +0 -1
- package/dist/api/TextTranslation.d.ts +21 -1
- package/dist/api/TextTranslation.js +99 -2
- package/dist/api/TextTranslation.js.map +1 -1
- package/dist/api/TimelineTranslationAlignment.js +1 -1
- package/dist/api/TimelineTranslationAlignment.js.map +1 -1
- package/dist/audio/AudioBufferConversion.d.ts +0 -1
- package/dist/audio/AudioPlayer.d.ts +0 -1
- package/dist/audio/AudioPlayer.js +62 -41
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +0 -1
- package/dist/cli/CLI.d.ts +1 -0
- package/dist/cli/CLI.js +61 -4
- package/dist/cli/CLI.js.map +1 -1
- package/dist/codecs/FFMpegTranscoder.d.ts +0 -1
- package/dist/codecs/TIMITCodec.d.ts +0 -1
- package/dist/codecs/WaveCodec.d.ts +0 -1
- package/dist/math/VectorMath.d.ts +4 -4
- package/dist/math/VectorMath.js +6 -6
- package/dist/nlp/ChineseSegmentation.js +4 -4
- package/dist/nlp/ChineseSegmentation.js.map +1 -1
- package/dist/nlp/Segmentation.d.ts +2 -2
- package/dist/nlp/Segmentation.js +20 -13
- package/dist/nlp/Segmentation.js.map +1 -1
- package/dist/recognition/OpenAICloudSTT.d.ts +2 -1
- package/dist/recognition/OpenAICloudSTT.js +30 -19
- package/dist/recognition/OpenAICloudSTT.js.map +1 -1
- package/dist/recognition/SileroSTT.d.ts +0 -1
- package/dist/recognition/WhisperCppSTT.d.ts +2 -2
- package/dist/recognition/WhisperCppSTT.js +19 -7
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.d.ts +9 -6
- package/dist/recognition/WhisperSTT.js +218 -40
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Client.d.ts +0 -2
- package/dist/server/Worker.d.ts +0 -1
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +0 -1
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +0 -1
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +0 -1
- package/dist/subtitles/Subtitles.js +2 -2
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.js +6 -21
- package/dist/synthesis/GoogleTranslateTTS.js.map +1 -1
- package/dist/synthesis/StreamlabsPollyTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.js +30 -0
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/text-translation/DeepLTextTranslation.d.ts +2 -0
- package/dist/text-translation/DeepLTextTranslation.js +67 -0
- package/dist/text-translation/DeepLTextTranslation.js.map +1 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.d.ts +10 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js +554 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js.map +1 -0
- package/dist/text-translation/NLLBTextTranslation.d.ts +2 -1
- package/dist/text-translation/NLLBTextTranslation.js +248 -18
- package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
- package/dist/utilities/BinaryArrayConversion.d.ts +0 -1
- package/dist/utilities/BrowserRequestHeaders.d.ts +6 -0
- package/dist/utilities/BrowserRequestHeaders.js +52 -0
- package/dist/utilities/BrowserRequestHeaders.js.map +1 -0
- package/dist/utilities/BufferFileReadStream.d.ts +20 -0
- package/dist/utilities/BufferFileReadStream.js +81 -0
- package/dist/utilities/BufferFileReadStream.js.map +1 -0
- package/dist/utilities/DynamicUint8Array.d.ts +9 -0
- package/dist/utilities/DynamicUint8Array.js +31 -0
- package/dist/utilities/DynamicUint8Array.js.map +1 -0
- package/dist/utilities/FileSystem.d.ts +0 -2
- package/dist/utilities/Hashing.d.ts +3 -10
- package/dist/utilities/Hashing.js +10 -127
- package/dist/utilities/Hashing.js.map +1 -1
- package/dist/utilities/LEB128.d.ts +15 -5
- package/dist/utilities/LEB128.js +199 -119
- package/dist/utilities/LEB128.js.map +1 -1
- package/dist/utilities/LPVarInt.d.ts +11 -0
- package/dist/utilities/LPVarInt.js +187 -0
- package/dist/utilities/LPVarInt.js.map +1 -0
- package/dist/utilities/OnnxUtilities.d.ts +0 -1
- package/dist/utilities/PVarInt.d.ts +4 -0
- package/dist/utilities/PVarInt.js +166 -0
- package/dist/utilities/PVarInt.js.map +1 -0
- package/dist/utilities/PackageManager.js +43 -22
- package/dist/utilities/PackageManager.js.map +1 -1
- package/dist/utilities/RandomGenerator.d.ts +3 -17
- package/dist/utilities/RandomGenerator.js +12 -81
- package/dist/utilities/RandomGenerator.js.map +1 -1
- package/dist/utilities/Timeline.d.ts +1 -0
- package/dist/utilities/Timeline.js +117 -20
- package/dist/utilities/Timeline.js.map +1 -1
- package/dist/utilities/Utilities.d.ts +1 -3
- package/dist/utilities/Utilities.js +30 -3
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/utilities/VarInt.d.ts +4 -0
- package/dist/utilities/VarInt.js +166 -0
- package/dist/utilities/VarInt.js.map +1 -0
- package/dist/utilities/VirtualFileReadStream.d.ts +20 -0
- package/dist/utilities/VirtualFileReadStream.js +79 -0
- package/dist/utilities/VirtualFileReadStream.js.map +1 -0
- package/dist/utilities/WebReader.js +7 -23
- package/dist/utilities/WebReader.js.map +1 -1
- package/dist/voice-activity-detection/SileroVAD.d.ts +0 -1
- package/docs/API.md +24 -0
- package/docs/Engines.md +8 -3
- package/docs/Options.md +20 -11
- package/docs/Tasklist.md +1 -13
- package/package.json +20 -24
- package/src/alignment/DTWMfccSequenceAlignment.ts +5 -5
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +1 -3
- package/src/alignment/SemanticTextAlignment.ts +1 -1
- package/src/alignment/SpeechAlignment.ts +37 -6
- package/src/api/TextTranslation.ts +170 -2
- package/src/api/TimelineTranslationAlignment.ts +1 -1
- package/src/audio/AudioPlayer.ts +2 -0
- package/src/cli/CLI.ts +101 -7
- package/src/math/VectorMath.ts +6 -6
- package/src/nlp/ChineseSegmentation.ts +6 -4
- package/src/nlp/Segmentation.ts +18 -13
- package/src/recognition/OpenAICloudSTT.ts +47 -29
- package/src/recognition/WhisperCppSTT.ts +24 -9
- package/src/recognition/WhisperSTT.ts +354 -43
- package/src/subtitles/Subtitles.ts +2 -2
- package/src/synthesis/GoogleTranslateTTS.ts +7 -21
- package/src/synthesis/VitsTTS.ts +31 -3
- package/src/text-translation/DeepLTextTranslation.ts +88 -0
- package/src/text-translation/GoogleTranslateTextTranslation.ts +667 -0
- package/src/text-translation/NLLBTextTranslation.ts +260 -20
- package/src/typings/Fillers.d.ts +25 -2
- package/src/utilities/BrowserRequestHeaders.ts +59 -0
- package/src/utilities/DynamicUint8Array.ts +39 -0
- package/src/utilities/Hashing.ts +14 -167
- package/src/utilities/LEB128.ts +273 -148
- package/src/utilities/LPVarInt.ts +292 -0
- package/src/utilities/PackageManager.ts +45 -27
- package/src/utilities/RandomGenerator.ts +12 -113
- package/src/utilities/Timeline.ts +148 -23
- package/src/utilities/Utilities.ts +40 -3
- package/src/utilities/VirtualFileReadStream.ts +109 -0
- package/src/utilities/WebReader.ts +9 -23
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange,
|
|
1
|
+
import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclideanDistance, euclideanDistance13Dim, magnitude } from '../math/VectorMath.js'
|
|
2
2
|
import { logToStderr } from '../utilities/Utilities.js'
|
|
3
3
|
import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
|
|
4
4
|
|
|
5
5
|
const log = logToStderr
|
|
6
6
|
|
|
7
|
-
export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: '
|
|
8
|
-
if (distanceFunctionKind == '
|
|
9
|
-
let distanceFunction =
|
|
7
|
+
export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: 'euclidean' | 'cosine' = 'euclidean', centerIndexes?: number[]) {
|
|
8
|
+
if (distanceFunctionKind == 'euclidean') {
|
|
9
|
+
let distanceFunction = euclideanDistance
|
|
10
10
|
|
|
11
11
|
if (mfccFrames1.length > 0 && mfccFrames1[0].length === 13) {
|
|
12
|
-
distanceFunction =
|
|
12
|
+
distanceFunction = euclideanDistance13Dim
|
|
13
13
|
}
|
|
14
14
|
|
|
15
15
|
const { path } = alignDTWWindowed(
|
|
@@ -4,9 +4,7 @@ import { AlignmentPath } from './SpeechAlignment.js'
|
|
|
4
4
|
const log = logToStderr
|
|
5
5
|
|
|
6
6
|
export function alignDTWWindowed<T, U>(sequence1: T[], sequence2: U[], costFunction: (a: T, b: U) => number, windowMaxLength: number, centerIndexes?: number[]) {
|
|
7
|
-
|
|
8
|
-
throw new Error('Window length must be greater or equal to 2')
|
|
9
|
-
}
|
|
7
|
+
windowMaxLength = Math.max(windowMaxLength, 2)
|
|
10
8
|
|
|
11
9
|
if (sequence1.length == 0 || sequence2.length == 0) {
|
|
12
10
|
return {
|
|
@@ -118,7 +118,7 @@ export async function alignTimelineToTextSemantically(timeline: Timeline, text:
|
|
|
118
118
|
return resultTimeline
|
|
119
119
|
}
|
|
120
120
|
|
|
121
|
-
export async function alignWordsToWordsSemantically(wordsGroups1: string[][], wordsGroups2: string[][], windowTokenCount =
|
|
121
|
+
export async function alignWordsToWordsSemantically(wordsGroups1: string[][], wordsGroups2: string[][], windowTokenCount = 20000) {
|
|
122
122
|
const logger = new Logger()
|
|
123
123
|
|
|
124
124
|
// Load embedding model
|
|
@@ -5,14 +5,14 @@ import * as API from '../api/API.js'
|
|
|
5
5
|
import { computeMFCCs, extendDefaultMfccOptions, MfccOptions } from '../dsp/MFCC.js'
|
|
6
6
|
import { alignMFCC_DTW, getCostMatrixMemorySizeMB } from './DTWMfccSequenceAlignment.js'
|
|
7
7
|
import { Logger } from '../utilities/Logger.js'
|
|
8
|
-
import { Timeline, TimelineEntry } from '../utilities/Timeline.js'
|
|
9
|
-
import { downmixToMonoAndNormalize, getEndingSilentSampleCount, getRawAudioDuration, getStartingSilentSampleCount, RawAudio } from '../audio/AudioUtilities.js'
|
|
8
|
+
import { addTimeOffsetToTimeline, Timeline, TimelineEntry } from '../utilities/Timeline.js'
|
|
9
|
+
import { concatAudioSegments, downmixToMonoAndNormalize, getEmptyRawAudio, getEndingSilentSampleCount, getRawAudioDuration, getStartingSilentSampleCount, RawAudio } from '../audio/AudioUtilities.js'
|
|
10
10
|
import chalk from 'chalk'
|
|
11
11
|
import { synthesize } from '../api/API.js'
|
|
12
12
|
import { resampleAudioSpeex } from '../dsp/SpeexResampler.js'
|
|
13
13
|
import { deepClone } from '../utilities/ObjectUtilities.js'
|
|
14
|
-
import { cosineDistance,
|
|
15
|
-
import { EspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
14
|
+
import { cosineDistance, euclideanDistance, zeroIfNaN } from '../math/VectorMath.js'
|
|
15
|
+
import { EspeakEvent, EspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
16
16
|
import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
|
|
17
17
|
import { loadPackage } from '../utilities/PackageManager.js'
|
|
18
18
|
import path from 'path'
|
|
@@ -278,7 +278,7 @@ export async function alignUsingDtwWithRecognition(
|
|
|
278
278
|
return result
|
|
279
279
|
}
|
|
280
280
|
|
|
281
|
-
// This is experimental code. It
|
|
281
|
+
// This is experimental code. It doesn't work well enough to be usable for anything.
|
|
282
282
|
// Just testing some alternative approaches.
|
|
283
283
|
export async function alignUsingDtwWithEmbeddings(
|
|
284
284
|
sourceRawAudio: RawAudio,
|
|
@@ -580,7 +580,38 @@ export async function createAlignmentReferenceUsingEspeakForFragments(fragments:
|
|
|
580
580
|
|
|
581
581
|
progressLogger.start("Synthesize alignment reference with eSpeak")
|
|
582
582
|
|
|
583
|
-
const result =
|
|
583
|
+
const result = {
|
|
584
|
+
rawAudio: getEmptyRawAudio(1, await Espeak.getSampleRate()) as RawAudio,
|
|
585
|
+
timeline: [] as Timeline,
|
|
586
|
+
events: [] as EspeakEvent[],
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
// Split fragments to chunks and process each chunk individually,
|
|
590
|
+
// and incrementally merge the chunks to the final result.
|
|
591
|
+
{
|
|
592
|
+
const maxFragmentsInChunk = 1000
|
|
593
|
+
|
|
594
|
+
let timeOffset = 0
|
|
595
|
+
|
|
596
|
+
for (let startOffset = 0; startOffset < fragments.length; startOffset += maxFragmentsInChunk) {
|
|
597
|
+
const chunk = fragments.slice(startOffset, startOffset + maxFragmentsInChunk)
|
|
598
|
+
|
|
599
|
+
const chunkResult = await Espeak.synthesizeFragments(chunk, espeakOptions)
|
|
600
|
+
|
|
601
|
+
result.rawAudio = {
|
|
602
|
+
sampleRate: result.rawAudio.sampleRate,
|
|
603
|
+
audioChannels: concatAudioSegments([result.rawAudio.audioChannels, chunkResult.rawAudio.audioChannels])
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
const chunkTimeline = addTimeOffsetToTimeline(chunkResult.timeline, timeOffset)
|
|
607
|
+
|
|
608
|
+
result.timeline = [...result.timeline, ...chunkTimeline]
|
|
609
|
+
|
|
610
|
+
result.events = [...result.events, ...chunkResult.events]
|
|
611
|
+
|
|
612
|
+
timeOffset += getRawAudioDuration(chunkResult.rawAudio)
|
|
613
|
+
}
|
|
614
|
+
}
|
|
584
615
|
|
|
585
616
|
result.timeline = result.timeline.flatMap(clause => clause.timeline!)
|
|
586
617
|
|
|
@@ -1,9 +1,177 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
1
|
+
import chalk from 'chalk'
|
|
2
|
+
import { formatLanguageCodeWithName, normalizeIdentifierToLanguageCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
3
|
+
import { Logger } from '../utilities/Logger.js'
|
|
4
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
5
|
+
import * as API from './API.js'
|
|
6
|
+
|
|
7
|
+
export async function translateText(inputText: string, options: TextTranslationOptions): Promise<TextTranslationResult> {
|
|
8
|
+
const logger = new Logger()
|
|
9
|
+
|
|
10
|
+
const startTimestamp = logger.getTimestamp()
|
|
11
|
+
|
|
12
|
+
options = extendDeep(defaultTextTranslationOptions, options)
|
|
13
|
+
|
|
14
|
+
if (options.sourceLanguage) {
|
|
15
|
+
const languageData = await parseLangIdentifier(options.sourceLanguage)
|
|
16
|
+
|
|
17
|
+
options.sourceLanguage = languageData.Name
|
|
18
|
+
|
|
19
|
+
logger.end()
|
|
20
|
+
logger.logTitledMessage('Source language specified', formatLanguageCodeWithName(options.sourceLanguage))
|
|
21
|
+
} else {
|
|
22
|
+
logger.start('No source language specified. Detect text language')
|
|
23
|
+
const { detectedLanguage } = await API.detectTextLanguage(inputText, options.languageDetection || {})
|
|
24
|
+
|
|
25
|
+
options.sourceLanguage = detectedLanguage
|
|
26
|
+
|
|
27
|
+
logger.end()
|
|
28
|
+
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
options.targetLanguage = await normalizeIdentifierToLanguageCode(options.targetLanguage!)
|
|
32
|
+
|
|
33
|
+
logger.logTitledMessage('Target language', formatLanguageCodeWithName(options.targetLanguage))
|
|
34
|
+
|
|
35
|
+
logger.start(`Load ${options.engine} module`)
|
|
36
|
+
|
|
37
|
+
let translationPairs: TranslationPair[]
|
|
38
|
+
|
|
39
|
+
switch (options.engine) {
|
|
40
|
+
case 'nllb': {
|
|
41
|
+
const NLLBTextTranslation = await import('../text-translation/NLLBTextTranslation.js')
|
|
42
|
+
|
|
43
|
+
logger.end()
|
|
44
|
+
|
|
45
|
+
logger.logTitledMessage(`Warning`, `The nllb engine is currently an early prototype implementation and doesn't work correctly.`, chalk.yellow, 'warning')
|
|
46
|
+
|
|
47
|
+
translationPairs = await NLLBTextTranslation.translateText(inputText, options.sourceLanguage, options.targetLanguage)
|
|
48
|
+
|
|
49
|
+
break
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
case 'google-translate': {
|
|
53
|
+
const GoogleTranslateTextTranslation = await import('../text-translation/GoogleTranslateTextTranslation.js')
|
|
54
|
+
|
|
55
|
+
logger.end()
|
|
56
|
+
|
|
57
|
+
translationPairs = await GoogleTranslateTextTranslation.translateText(inputText, options.sourceLanguage, options.targetLanguage)
|
|
58
|
+
|
|
59
|
+
break
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
case 'deepl': {
|
|
63
|
+
const DeepLTextTranslation = await import('../text-translation/DeepLTextTranslation.js')
|
|
64
|
+
|
|
65
|
+
logger.end()
|
|
66
|
+
|
|
67
|
+
logger.logTitledMessage(`Warning`, `The deepl engine is currently an early prototype implementation and doesn't work correctly.`, chalk.yellow, 'warning')
|
|
68
|
+
|
|
69
|
+
translationPairs = await DeepLTextTranslation.translateText(inputText, options.sourceLanguage, options.targetLanguage)
|
|
70
|
+
|
|
71
|
+
break
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
default: {
|
|
75
|
+
throw new Error(`'${options.engine}' is not a supported text translation engine.`)
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const translatedText = translationPairs.map(pair => {
|
|
80
|
+
const translated = pair.translatedText
|
|
81
|
+
|
|
82
|
+
if (translated.endsWith(' ') || translated.endsWith('\n')) {
|
|
83
|
+
return pair.translatedText
|
|
84
|
+
} else {
|
|
85
|
+
return pair.translatedText + ' '
|
|
86
|
+
}
|
|
87
|
+
}).join('').trim()
|
|
88
|
+
|
|
89
|
+
logger.end()
|
|
90
|
+
|
|
91
|
+
logger.log('')
|
|
92
|
+
logger.logDuration(`Total text translation time`, startTimestamp, chalk.magentaBright)
|
|
93
|
+
|
|
94
|
+
return {
|
|
95
|
+
text: inputText,
|
|
96
|
+
translatedText,
|
|
97
|
+
|
|
98
|
+
translationPairs,
|
|
99
|
+
|
|
100
|
+
sourceLanguage: options.sourceLanguage!,
|
|
101
|
+
targetLanguage: options.targetLanguage!,
|
|
102
|
+
}
|
|
3
103
|
}
|
|
4
104
|
|
|
5
105
|
export interface TextTranslationOptions {
|
|
106
|
+
engine?: TextTranslationEngine
|
|
107
|
+
|
|
108
|
+
sourceLanguage?: string
|
|
109
|
+
targetLanguage?: string
|
|
110
|
+
|
|
111
|
+
languageDetection?: API.TextLanguageDetectionOptions
|
|
112
|
+
|
|
113
|
+
nllb?: {
|
|
114
|
+
},
|
|
115
|
+
|
|
116
|
+
googleTranslate?: {
|
|
117
|
+
},
|
|
118
|
+
|
|
119
|
+
deepl?: {
|
|
120
|
+
},
|
|
6
121
|
}
|
|
7
122
|
|
|
8
123
|
export interface TextTranslationResult {
|
|
124
|
+
text: string
|
|
125
|
+
translatedText: string
|
|
126
|
+
|
|
127
|
+
translationPairs: TranslationPair[]
|
|
128
|
+
|
|
129
|
+
sourceLanguage: string
|
|
130
|
+
targetLanguage: string
|
|
9
131
|
}
|
|
132
|
+
|
|
133
|
+
export type TextTranslationEngine = 'nllb' | 'google-translate' | 'deepl'
|
|
134
|
+
|
|
135
|
+
export interface TranslationPair {
|
|
136
|
+
sourceText: string
|
|
137
|
+
translatedText: string
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export const defaultTextTranslationOptions: TextTranslationOptions = {
|
|
141
|
+
engine: 'google-translate',
|
|
142
|
+
|
|
143
|
+
sourceLanguage: undefined,
|
|
144
|
+
targetLanguage: 'en',
|
|
145
|
+
|
|
146
|
+
languageDetection: undefined,
|
|
147
|
+
|
|
148
|
+
nllb: {
|
|
149
|
+
},
|
|
150
|
+
|
|
151
|
+
googleTranslate: {
|
|
152
|
+
},
|
|
153
|
+
|
|
154
|
+
deepl: {
|
|
155
|
+
},
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
export const textTranslationEngines: API.EngineMetadata[] = [
|
|
159
|
+
{
|
|
160
|
+
id: 'nllb',
|
|
161
|
+
name: 'NLLB',
|
|
162
|
+
description: 'No Language Left Behind (NLLB) is a deep learning machine translation model by Facebook Research (early prototype implementation).',
|
|
163
|
+
type: 'local'
|
|
164
|
+
},
|
|
165
|
+
{
|
|
166
|
+
id: 'google-translate',
|
|
167
|
+
name: 'Google Translate',
|
|
168
|
+
description: 'Unoffical text translation API used by the Google Translate web interface.',
|
|
169
|
+
type: 'cloud'
|
|
170
|
+
},
|
|
171
|
+
{
|
|
172
|
+
id: 'deepl',
|
|
173
|
+
name: 'DeepL',
|
|
174
|
+
description: 'Unoffical text translation API used by the DeepL web interface (early prototype implementation).',
|
|
175
|
+
type: 'cloud'
|
|
176
|
+
},
|
|
177
|
+
]
|
|
@@ -60,7 +60,7 @@ export async function alignTimelineTranslation(inputTimeline: Timeline, translat
|
|
|
60
60
|
logger.logTitledMessage('Target language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
61
61
|
}
|
|
62
62
|
|
|
63
|
-
logger.
|
|
63
|
+
logger.start(`Load ${options.engine} module`)
|
|
64
64
|
|
|
65
65
|
let mappedWordTimeline: Timeline
|
|
66
66
|
|
package/src/audio/AudioPlayer.ts
CHANGED
|
@@ -274,6 +274,7 @@ export function playAudioSamples(rawAudio: RawAudio, onTimePosition?: (timePosit
|
|
|
274
274
|
})
|
|
275
275
|
}
|
|
276
276
|
|
|
277
|
+
/*
|
|
277
278
|
export function playAudioSamples_Speaker(rawAudio: RawAudio, onTimePosition?: (timePosition: number) => void, microFadeInOut = true) {
|
|
278
279
|
return new Promise<void>(async (resolve, reject) => {
|
|
279
280
|
if (microFadeInOut) {
|
|
@@ -354,6 +355,7 @@ export function playAudioSamples_Speaker(rawAudio: RawAudio, onTimePosition?: (t
|
|
|
354
355
|
}
|
|
355
356
|
})
|
|
356
357
|
}
|
|
358
|
+
*/
|
|
357
359
|
|
|
358
360
|
export const charactersToWriteAhead = [
|
|
359
361
|
',', '.', ',', '、', ':', ';',
|
package/src/cli/CLI.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import * as API from '../api/API.js'
|
|
2
2
|
import { parseCLIArguments } from './CLIParser.js'
|
|
3
|
-
import {
|
|
3
|
+
import { convertHtmlToText, formatIntegerWithLeadingZeros, formatListWithQuotedElements, getWithDefault, logToStderr, setupUnhandledExceptionListeners, splitFilenameOnExtendedExtension, stringifyAndFormatJson } from '../utilities/Utilities.js'
|
|
4
4
|
import { getOptionTypeFromSchema, SchemaTypeDefinition } from './CLIOptionsSchema.js'
|
|
5
5
|
import { ParsedConfigFile, parseConfigFile, parseJSONConfigFile } from './CLIConfigFile.js'
|
|
6
6
|
|
|
7
7
|
import chalk from 'chalk'
|
|
8
8
|
import { RawAudio, applyGainDecibels, encodeRawAudioToWave, getEmptyRawAudio, getRawAudioDuration, normalizeAudioLevel, sliceRawAudioByTime } from '../audio/AudioUtilities.js'
|
|
9
|
-
import { SubtitlesConfig, subtitlesToText,
|
|
9
|
+
import { SubtitlesConfig, subtitlesToText, timelineToSubtitles } from '../subtitles/Subtitles.js'
|
|
10
10
|
import { Logger, resetActiveLogger } from '../utilities/Logger.js'
|
|
11
11
|
import { isMainThread, parentPort } from 'node:worker_threads'
|
|
12
12
|
import { encodeFromChannels, getDefaultFFMpegOptionsForSpeech } from '../codecs/FFMpegTranscoder.js'
|
|
@@ -224,18 +224,20 @@ const help = [
|
|
|
224
224
|
` Transcribe a spoken audio file\n`,
|
|
225
225
|
`${executableName} ${chalk.magentaBright('align')} audioFile transcriptFile [output files...] [options...]`,
|
|
226
226
|
` Align spoken audio file to its transcript\n`,
|
|
227
|
+
`${executableName} ${chalk.magentaBright('translate-text')} inputFile [output files...] [options...]`,
|
|
228
|
+
` Translate text to a different language\n`,
|
|
227
229
|
`${executableName} ${chalk.magentaBright('translate-speech')} audioFile [output files...] [options...]`,
|
|
228
|
-
` Transcribe audio file directly to a different language\n`,
|
|
230
|
+
` Transcribe spoken audio file directly to a different language\n`,
|
|
229
231
|
`${executableName} ${chalk.magentaBright('align-translation')} audioFile translatedTranscriptFile [output files...] [options...]`,
|
|
230
232
|
` Align spoken audio file to its translated transcript\n`,
|
|
231
233
|
`${executableName} ${chalk.magentaBright('align-transcript-and-translation')} audioFile transcriptFile translatedTranscriptFile [output files...] [options...]`,
|
|
232
234
|
` Align spoken audio file to both its transcript and its translated transcript using a two-stage approach.\n`,
|
|
233
235
|
`${executableName} ${chalk.magentaBright('align-timeline-translation')} timelineFile translatedFile [output files...] [options...]`,
|
|
234
236
|
` Align a given timeline file to its translated text\n`,
|
|
235
|
-
`${executableName} ${chalk.magentaBright('detect-speech-language')} audioFile [output files...] [options...]`,
|
|
236
|
-
` Detect language of spoken audio file\n`,
|
|
237
237
|
`${executableName} ${chalk.magentaBright('detect-text-language')} inputFile [output files...] [options...]`,
|
|
238
238
|
` Detect language of textual file\n`,
|
|
239
|
+
`${executableName} ${chalk.magentaBright('detect-speech-language')} audioFile [output files...] [options...]`,
|
|
240
|
+
` Detect language of spoken audio file\n`,
|
|
239
241
|
`${executableName} ${chalk.magentaBright('detect-voice-activity')} audioFile [output files...] [options...]`,
|
|
240
242
|
` Detect voice activity in audio file\n`,
|
|
241
243
|
`${executableName} ${chalk.magentaBright('denoise')} audioFile [output files...] [options...]`,
|
|
@@ -279,6 +281,11 @@ async function startWithArgs(operationData: CLIOperationData) {
|
|
|
279
281
|
break
|
|
280
282
|
}
|
|
281
283
|
|
|
284
|
+
case 'translate-text': {
|
|
285
|
+
await translateText(operationData)
|
|
286
|
+
break
|
|
287
|
+
}
|
|
288
|
+
|
|
282
289
|
case 'translate-speech': {
|
|
283
290
|
await translateSpeech(operationData)
|
|
284
291
|
break
|
|
@@ -1014,6 +1021,75 @@ export async function alignTimelineTranslation(operationData: CLIOperationData)
|
|
|
1014
1021
|
}
|
|
1015
1022
|
}
|
|
1016
1023
|
|
|
1024
|
+
export async function translateText(operationData: CLIOperationData) {
|
|
1025
|
+
const logger = new Logger()
|
|
1026
|
+
|
|
1027
|
+
const { operationArgs, operationOptionsLookup, cliOptions } = operationData
|
|
1028
|
+
|
|
1029
|
+
const inputFilename = operationArgs[0]
|
|
1030
|
+
const outputFilenames = operationArgs.slice(1)
|
|
1031
|
+
|
|
1032
|
+
if (inputFilename == undefined) {
|
|
1033
|
+
throw new Error(`translate-text requires an argument containing the input file path.`)
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
if (!existsSync(inputFilename)) {
|
|
1037
|
+
throw new Error(`The given input file '${inputFilename}' was not found.`)
|
|
1038
|
+
}
|
|
1039
|
+
|
|
1040
|
+
const inputFileExtension = getLowercaseFileExtension(inputFilename)
|
|
1041
|
+
const inputFileContent = await readFile(inputFilename, { encoding: 'utf-8' })
|
|
1042
|
+
|
|
1043
|
+
let inputText: string
|
|
1044
|
+
|
|
1045
|
+
if (inputFileExtension === 'txt') {
|
|
1046
|
+
inputText = inputFileContent
|
|
1047
|
+
} else if (inputFileExtension === 'html' || inputFileExtension === 'htm') {
|
|
1048
|
+
inputText = await convertHtmlToText(inputFileContent)
|
|
1049
|
+
} else if (inputFileExtension == 'srt' || inputFileExtension == 'vtt') {
|
|
1050
|
+
inputText = subtitlesToText(inputFileContent)
|
|
1051
|
+
} else {
|
|
1052
|
+
throw new Error(`align only supports reference files with extensions 'txt', 'html', 'htm', 'srt' or 'vtt'`)
|
|
1053
|
+
}
|
|
1054
|
+
|
|
1055
|
+
const options = await optionsLookupToTypedObject(operationOptionsLookup, 'TextTranslationOptions')
|
|
1056
|
+
|
|
1057
|
+
const allowOverwrite = getWithDefault(cliOptions.overwrite, overwriteByDefault)
|
|
1058
|
+
|
|
1059
|
+
await checkOutputFilenames(outputFilenames, false, true, true)
|
|
1060
|
+
|
|
1061
|
+
const {
|
|
1062
|
+
text,
|
|
1063
|
+
translatedText,
|
|
1064
|
+
|
|
1065
|
+
translationPairs,
|
|
1066
|
+
|
|
1067
|
+
sourceLanguage,
|
|
1068
|
+
targetLanguage,
|
|
1069
|
+
} = await API.translateText(inputFileContent, options)
|
|
1070
|
+
|
|
1071
|
+
if (outputFilenames.length > 0) {
|
|
1072
|
+
logger.start('\nWrite output files')
|
|
1073
|
+
|
|
1074
|
+
for (const outputFilename of outputFilenames) {
|
|
1075
|
+
const partPatternMatch = outputFilename.match(filenamePlaceholderPattern)
|
|
1076
|
+
|
|
1077
|
+
if (partPatternMatch) {
|
|
1078
|
+
continue
|
|
1079
|
+
}
|
|
1080
|
+
|
|
1081
|
+
const fileSaver = getFileSaver(outputFilename, allowOverwrite)
|
|
1082
|
+
|
|
1083
|
+
await fileSaver(getEmptyRawAudio(1, 16000), translationPairs as any as Timeline, translatedText, undefined)
|
|
1084
|
+
}
|
|
1085
|
+
|
|
1086
|
+
logger.end()
|
|
1087
|
+
} else {
|
|
1088
|
+
logger.log(``)
|
|
1089
|
+
logger.log(translatedText)
|
|
1090
|
+
}
|
|
1091
|
+
}
|
|
1092
|
+
|
|
1017
1093
|
export async function translateSpeech(operationData: CLIOperationData) {
|
|
1018
1094
|
const logger = new Logger()
|
|
1019
1095
|
|
|
@@ -1040,7 +1116,19 @@ export async function translateSpeech(operationData: CLIOperationData) {
|
|
|
1040
1116
|
|
|
1041
1117
|
await checkOutputFilenames(outputFilenames, true, true, true)
|
|
1042
1118
|
|
|
1043
|
-
const {
|
|
1119
|
+
const {
|
|
1120
|
+
transcript,
|
|
1121
|
+
|
|
1122
|
+
timeline,
|
|
1123
|
+
wordTimeline,
|
|
1124
|
+
|
|
1125
|
+
sourceLanguage,
|
|
1126
|
+
targetLanguage,
|
|
1127
|
+
|
|
1128
|
+
inputRawAudio,
|
|
1129
|
+
isolatedRawAudio,
|
|
1130
|
+
backgroundRawAudio
|
|
1131
|
+
} = await API.translateSpeech(inputFilename, options)
|
|
1044
1132
|
|
|
1045
1133
|
if (outputFilenames.length > 0) {
|
|
1046
1134
|
logger.start('\nWrite output files')
|
|
@@ -1382,6 +1470,12 @@ export async function listEngines(operationData: CLIOperationData) {
|
|
|
1382
1470
|
break
|
|
1383
1471
|
}
|
|
1384
1472
|
|
|
1473
|
+
case 'translate-text': {
|
|
1474
|
+
engines = API.textTranslationEngines
|
|
1475
|
+
|
|
1476
|
+
break
|
|
1477
|
+
}
|
|
1478
|
+
|
|
1385
1479
|
case 'translate-speech': {
|
|
1386
1480
|
engines = API.speechTranslationEngines
|
|
1387
1481
|
|
|
@@ -1432,7 +1526,7 @@ export async function listEngines(operationData: CLIOperationData) {
|
|
|
1432
1526
|
}
|
|
1433
1527
|
|
|
1434
1528
|
default: {
|
|
1435
|
-
throw new Error(`Unrecognized operation: '${targetOperation}'`)
|
|
1529
|
+
throw new Error(`Unrecognized operation name: '${targetOperation}'`)
|
|
1436
1530
|
}
|
|
1437
1531
|
}
|
|
1438
1532
|
|
package/src/math/VectorMath.ts
CHANGED
|
@@ -456,11 +456,11 @@ export function meanSquaredError(actual: ArrayLike<number>, expected: ArrayLike<
|
|
|
456
456
|
return sum / featureCount
|
|
457
457
|
}
|
|
458
458
|
|
|
459
|
-
export function
|
|
460
|
-
return Math.sqrt(
|
|
459
|
+
export function euclideanDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
460
|
+
return Math.sqrt(squaredEuclideanDistance(vector1, vector2))
|
|
461
461
|
}
|
|
462
462
|
|
|
463
|
-
export function
|
|
463
|
+
export function squaredEuclideanDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
464
464
|
if (vector1.length !== vector2.length) {
|
|
465
465
|
throw new Error('Vectors are not the same length')
|
|
466
466
|
}
|
|
@@ -480,11 +480,11 @@ export function squaredEuclidianDistance(vector1: ArrayLike<number>, vector2: Ar
|
|
|
480
480
|
return sum
|
|
481
481
|
}
|
|
482
482
|
|
|
483
|
-
export function
|
|
484
|
-
return Math.sqrt(
|
|
483
|
+
export function euclideanDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
484
|
+
return Math.sqrt(squaredEuclideanDistance13Dim(vector1, vector2))
|
|
485
485
|
}
|
|
486
486
|
|
|
487
|
-
export function
|
|
487
|
+
export function squaredEuclideanDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
488
488
|
// Assumes the input has 13 dimensions (optimized for 13-dimensional MFCC vectors)
|
|
489
489
|
|
|
490
490
|
const result =
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
export async function splitChineseTextToWords_Jieba(text: string, fineGrained = false, useHMM = true) {
|
|
2
|
-
const jieba = await
|
|
2
|
+
const jieba = await getJiebaWasmInstance()
|
|
3
3
|
|
|
4
4
|
if (!fineGrained) {
|
|
5
5
|
return jieba.cut(text, useHMM)
|
|
@@ -58,10 +58,12 @@ export async function splitChineseTextToWords_Jieba(text: string, fineGrained =
|
|
|
58
58
|
}
|
|
59
59
|
|
|
60
60
|
let JiebaWasmInstance: typeof import('jieba-wasm')
|
|
61
|
-
|
|
61
|
+
|
|
62
|
+
async function getJiebaWasmInstance() {
|
|
62
63
|
if (!JiebaWasmInstance) {
|
|
63
|
-
const { default:
|
|
64
|
-
|
|
64
|
+
const { default: JiebaWasm } = await import('jieba-wasm')
|
|
65
|
+
|
|
66
|
+
JiebaWasmInstance = JiebaWasm
|
|
65
67
|
}
|
|
66
68
|
|
|
67
69
|
return JiebaWasmInstance
|
package/src/nlp/Segmentation.ts
CHANGED
|
@@ -11,8 +11,7 @@ const log = logToStderr
|
|
|
11
11
|
export const wordCharacterPattern = /[\p{Letter}\p{Number}]/u
|
|
12
12
|
export const punctuationPattern = /[\p{Punctuation}]/u
|
|
13
13
|
|
|
14
|
-
export const phraseSeparators = [',', ';', ':']
|
|
15
|
-
export const sentenceSeparators = ['.', '?', '!']
|
|
14
|
+
export const phraseSeparators = [',', ';', ':', ',', '、']
|
|
16
15
|
export const symbolWords = ['$', '€', '¢', '£', '¥', '©', '®', '™', '%', '&', '#', '~', '@', '+', '±', '÷', '/', '*', '=', '¼', '½', '¾']
|
|
17
16
|
|
|
18
17
|
export function isWordOrSymbolWord(str: string) {
|
|
@@ -226,29 +225,35 @@ export async function splitToWords(text: string, langCode: string): Promise<stri
|
|
|
226
225
|
}
|
|
227
226
|
}
|
|
228
227
|
|
|
229
|
-
export function splitToParagraphs(text: string, paragraphBreaks: ParagraphBreakType,
|
|
228
|
+
export function splitToParagraphs(text: string, paragraphBreaks: ParagraphBreakType, whitespaceProcessingMethod: WhitespaceProcessing) {
|
|
230
229
|
let paragraphs: string[] = []
|
|
231
230
|
|
|
232
|
-
if (paragraphBreaks
|
|
231
|
+
if (paragraphBreaks === 'single') {
|
|
233
232
|
paragraphs = text.split(/(\r?\n)+/g)
|
|
234
|
-
} else if (paragraphBreaks
|
|
233
|
+
} else if (paragraphBreaks === 'double') {
|
|
235
234
|
paragraphs = text.split(/(\r?\n)(\r?\n)+/g)
|
|
236
235
|
} else {
|
|
237
|
-
throw new Error(`Invalid paragraph break type: ${paragraphBreaks}`)
|
|
236
|
+
throw new Error(`Invalid paragraph break type: '${paragraphBreaks}'`)
|
|
238
237
|
}
|
|
239
238
|
|
|
240
|
-
|
|
241
|
-
paragraphs = paragraphs.map(p => p.replaceAll(/(\r?\n)+/g, ' '))
|
|
242
|
-
} else if (whitespace == 'collapse') {
|
|
243
|
-
paragraphs = paragraphs.map(p => p.replaceAll(/\s+/g, ' '))
|
|
244
|
-
}
|
|
245
|
-
|
|
246
|
-
paragraphs = paragraphs.map(p => p.trim())
|
|
239
|
+
paragraphs = paragraphs.map(p => applyWhitespaceProcessing(p.trim(), whitespaceProcessingMethod))
|
|
247
240
|
paragraphs = paragraphs.filter(p => p.length > 0)
|
|
248
241
|
|
|
249
242
|
return paragraphs
|
|
250
243
|
}
|
|
251
244
|
|
|
245
|
+
export function applyWhitespaceProcessing(text: string, whitespaceProcessingMethod: WhitespaceProcessing) {
|
|
246
|
+
if (whitespaceProcessingMethod === 'removeLineBreaks') {
|
|
247
|
+
return text.replaceAll(/(\r?\n)+/g, ' ')
|
|
248
|
+
} else if (whitespaceProcessingMethod === 'collapse') {
|
|
249
|
+
return text.replaceAll(/\s+/g, ' ')
|
|
250
|
+
} else if (whitespaceProcessingMethod === 'preserve') {
|
|
251
|
+
return text
|
|
252
|
+
} else {
|
|
253
|
+
throw new Error(`Invalid whitespace processing method: '${whitespaceProcessingMethod}'`)
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
|
|
252
257
|
export function splitToLines(text: string) {
|
|
253
258
|
return text.split(/\r?\n/g)
|
|
254
259
|
}
|