echogarden 1.4.3 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/data/schemas/options.json +267 -19
- package/data/tables/lcid-table.json +9 -0
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWMfccSequenceAlignment.js +9 -5
- package/dist/alignment/DTWMfccSequenceAlignment.js.map +1 -1
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +4 -4
- package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
- package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +10 -1
- package/dist/alignment/SemanticTextAlignment.js +336 -0
- package/dist/alignment/SemanticTextAlignment.js.map +1 -0
- package/dist/alignment/SpeechAlignment.d.ts +2 -1
- package/dist/alignment/SpeechAlignment.js +106 -38
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +7 -2
- package/dist/api/API.js +7 -2
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +4 -1
- package/dist/api/Alignment.d.ts +1 -1
- package/dist/api/Alignment.js +14 -6
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetectionCommon.d.ts +6 -0
- package/dist/api/LanguageDetectionCommon.js +2 -0
- package/dist/api/LanguageDetectionCommon.js.map +1 -0
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
- package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
- package/dist/api/SpeechLanguageDetection.js.map +1 -0
- package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
- package/dist/api/SpeechTranslation.js.map +1 -0
- package/dist/api/Synthesis.js +4 -4
- package/dist/api/TextLanguageDetection.d.ts +21 -0
- package/dist/api/TextLanguageDetection.js +67 -0
- package/dist/api/TextLanguageDetection.js.map +1 -0
- package/dist/api/TextTranslation.d.ts +5 -0
- package/dist/api/TextTranslation.js +4 -0
- package/dist/api/TextTranslation.js.map +1 -0
- package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
- package/dist/api/TimelineTranslationAlignment.js +92 -0
- package/dist/api/TimelineTranslationAlignment.js.map +1 -0
- package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
- package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
- package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
- package/dist/api/TranslationAlignment.d.ts +4 -3
- package/dist/api/TranslationAlignment.js +9 -8
- package/dist/api/TranslationAlignment.js.map +1 -1
- package/dist/api/VoiceActivityDetection.js +16 -1
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/cli/CLI.d.ts +27 -7
- package/dist/cli/CLI.js +205 -34
- package/dist/cli/CLI.js.map +1 -1
- package/dist/codecs/FFMpegTranscoder.js +7 -0
- package/dist/codecs/FFMpegTranscoder.js.map +1 -1
- package/dist/dsp/FFT.d.ts +1 -1
- package/dist/dsp/FFT.js +6 -0
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/dsp/KWeightingFilter.js +1 -1
- package/dist/dsp/KWeightingFilter.js.map +1 -1
- package/dist/dsp/MelSpectogram.d.ts +3 -2
- package/dist/dsp/MelSpectogram.js +14 -8
- package/dist/dsp/MelSpectogram.js.map +1 -1
- package/dist/math/VectorMath.d.ts +22 -20
- package/dist/math/VectorMath.js +57 -30
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/recognition/WhisperCppSTT.d.ts +1 -1
- package/dist/recognition/WhisperCppSTT.js +2 -2
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +9 -6
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Client.d.ts +3 -2
- package/dist/server/Client.js.map +1 -1
- package/dist/server/Worker.d.ts +3 -2
- package/dist/server/Worker.js +3 -2
- package/dist/server/Worker.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +13 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/synthesis/EspeakTTS.js +4 -5
- package/dist/synthesis/EspeakTTS.js.map +1 -1
- package/dist/tests/Test.js +0 -8
- package/dist/tests/Test.js.map +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
- package/dist/text-translation/NLLBTextTranslation.js +1 -1
- package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
- package/dist/utilities/Locale.d.ts +1 -1
- package/dist/utilities/Locale.js +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +1 -1
- package/dist/utilities/PackageManager.js +8 -2
- package/dist/utilities/PackageManager.js.map +1 -1
- package/dist/utilities/Timeline.d.ts +1 -0
- package/dist/utilities/Timeline.js +12 -0
- package/dist/utilities/Timeline.js.map +1 -1
- package/docs/API.md +81 -3
- package/docs/CLI.md +51 -1
- package/docs/Engines.md +24 -0
- package/docs/Options.md +33 -1
- package/docs/Tasklist.md +4 -1
- package/package.json +11 -11
- package/src/alignment/DTWMfccSequenceAlignment.ts +11 -5
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +4 -4
- package/src/alignment/SemanticTextAlignment.ts +467 -0
- package/src/alignment/SpeechAlignment.ts +180 -53
- package/src/api/API.ts +18 -2
- package/src/api/APIOptions.ts +14 -1
- package/src/api/Alignment.ts +32 -10
- package/src/api/LanguageDetectionCommon.ts +7 -0
- package/src/api/Recognition.ts +2 -0
- package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
- package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
- package/src/api/Synthesis.ts +4 -4
- package/src/api/TextLanguageDetection.ts +116 -0
- package/src/api/TextTranslation.ts +9 -0
- package/src/api/TimelineTranslationAlignment.ts +162 -0
- package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
- package/src/api/TranslationAlignment.ts +12 -10
- package/src/api/VoiceActivityDetection.ts +24 -3
- package/src/cli/CLI.ts +276 -34
- package/src/codecs/FFMpegTranscoder.ts +6 -0
- package/src/dsp/FFT.ts +8 -2
- package/src/dsp/KWeightingFilter.ts +1 -1
- package/src/dsp/MelSpectogram.ts +17 -8
- package/src/math/VectorMath.ts +87 -52
- package/src/recognition/WhisperCppSTT.ts +2 -2
- package/src/recognition/WhisperSTT.ts +10 -6
- package/src/server/Client.ts +3 -2
- package/src/server/Worker.ts +3 -2
- package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
- package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
- package/src/synthesis/EspeakTTS.ts +6 -8
- package/src/tests/Test.ts +1 -12
- package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
- package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
- package/src/text-translation/NLLBTextTranslation.ts +1 -1
- package/src/utilities/Locale.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +1 -1
- package/src/utilities/PackageManager.ts +9 -2
- package/src/utilities/Timeline.ts +14 -0
- package/dist/alignment/TextAlignment.js +0 -63
- package/dist/alignment/TextAlignment.js.map +0 -1
- package/dist/api/LanguageDetection.js.map +0 -1
- package/dist/api/Translation.js.map +0 -1
- package/src/alignment/TextAlignment.ts +0 -96
- /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
2
|
+
|
|
3
|
+
import { Logger } from '../utilities/Logger.js'
|
|
4
|
+
|
|
5
|
+
import * as API from './API.js'
|
|
6
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
7
|
+
import { languageCodeToName } from '../utilities/Locale.js'
|
|
8
|
+
import { LanguageDetectionResults } from './LanguageDetectionCommon.js'
|
|
9
|
+
|
|
10
|
+
const log = logToStderr
|
|
11
|
+
|
|
12
|
+
export async function detectTextLanguage(input: string, options: TextLanguageDetectionOptions): Promise<TextLanguageDetectionResult> {
|
|
13
|
+
const logger = new Logger()
|
|
14
|
+
|
|
15
|
+
options = extendDeep(defaultTextLanguageDetectionOptions, options)
|
|
16
|
+
|
|
17
|
+
const defaultLanguage = options.defaultLanguage!
|
|
18
|
+
const fallbackThresholdProbability = options.fallbackThresholdProbability!
|
|
19
|
+
|
|
20
|
+
let detectedLanguageProbabilities: LanguageDetectionResults
|
|
21
|
+
|
|
22
|
+
logger.start(`Initialize ${options.engine} module`)
|
|
23
|
+
|
|
24
|
+
switch (options.engine) {
|
|
25
|
+
case 'tinyld': {
|
|
26
|
+
const { detectLanguage } = await import('../text-language-detection/TinyLDLanguageDetection.js')
|
|
27
|
+
|
|
28
|
+
logger.start('Detect text language using tinyld')
|
|
29
|
+
|
|
30
|
+
detectedLanguageProbabilities = await detectLanguage(input)
|
|
31
|
+
|
|
32
|
+
break
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
case 'fasttext': {
|
|
36
|
+
const { detectLanguage } = await import('../text-language-detection/FastTextLanguageDetection.js')
|
|
37
|
+
|
|
38
|
+
logger.start('Detect text language using FastText')
|
|
39
|
+
|
|
40
|
+
detectedLanguageProbabilities = await detectLanguage(input)
|
|
41
|
+
|
|
42
|
+
break
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
default: {
|
|
46
|
+
throw new Error(`Engine '${options.engine}' is not supported`)
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
let detectedLanguage: string
|
|
51
|
+
|
|
52
|
+
if (detectedLanguageProbabilities.length == 0 ||
|
|
53
|
+
detectedLanguageProbabilities[0].probability < fallbackThresholdProbability) {
|
|
54
|
+
|
|
55
|
+
detectedLanguage = defaultLanguage
|
|
56
|
+
} else {
|
|
57
|
+
detectedLanguage = detectedLanguageProbabilities[0].language
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
logger.end()
|
|
61
|
+
|
|
62
|
+
return {
|
|
63
|
+
detectedLanguage,
|
|
64
|
+
detectedLanguageName: languageCodeToName(detectedLanguage),
|
|
65
|
+
detectedLanguageProbabilities
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
70
|
+
// Types
|
|
71
|
+
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
72
|
+
|
|
73
|
+
export interface TextLanguageDetectionResult {
|
|
74
|
+
detectedLanguage: string
|
|
75
|
+
detectedLanguageName: string
|
|
76
|
+
detectedLanguageProbabilities: LanguageDetectionResults
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export type LanguageDetectionGroupResults = LanguageDetectionGroupResultsEntry[]
|
|
80
|
+
export interface LanguageDetectionGroupResultsEntry {
|
|
81
|
+
languageGroup: string
|
|
82
|
+
probability: number
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export type TextLanguageDetectionEngine = 'tinyld' | 'fasttext'
|
|
86
|
+
|
|
87
|
+
export interface TextLanguageDetectionOptions {
|
|
88
|
+
engine?: TextLanguageDetectionEngine
|
|
89
|
+
defaultLanguage?: string
|
|
90
|
+
fallbackThresholdProbability?: number
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
94
|
+
// Constants
|
|
95
|
+
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
96
|
+
|
|
97
|
+
export const defaultTextLanguageDetectionOptions: TextLanguageDetectionOptions = {
|
|
98
|
+
engine: 'tinyld',
|
|
99
|
+
defaultLanguage: 'en',
|
|
100
|
+
fallbackThresholdProbability: 0.05,
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export const textLanguageDetectionEngines: API.EngineMetadata[] = [
|
|
104
|
+
{
|
|
105
|
+
id: 'tinyld',
|
|
106
|
+
name: 'TinyLD',
|
|
107
|
+
description: 'A simple language detection library.',
|
|
108
|
+
type: 'local'
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
id: 'fasttext',
|
|
112
|
+
name: 'FastText',
|
|
113
|
+
description: 'A library for word representations and sentence classification by Facebook research.',
|
|
114
|
+
type: 'local'
|
|
115
|
+
},
|
|
116
|
+
]
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
import chalk from 'chalk'
|
|
2
|
+
import { AudioSourceParam, RawAudio, ensureRawAudio } from '../audio/AudioUtilities.js'
|
|
3
|
+
import { SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
4
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
5
|
+
import { Logger } from '../utilities/Logger.js'
|
|
6
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
7
|
+
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
8
|
+
import * as API from './API.js'
|
|
9
|
+
|
|
10
|
+
export async function alignTimelineTranslation(inputTimeline: Timeline, translatedTranscript: string, options: TimelineTranslationAlignmentOptions): Promise<TimelineTranslationAlignmentResult> {
|
|
11
|
+
const logger = new Logger()
|
|
12
|
+
|
|
13
|
+
const startTimestamp = logger.getTimestamp()
|
|
14
|
+
|
|
15
|
+
options = extendDeep(defaultTimelineTranslationAlignmentOptions, options)
|
|
16
|
+
|
|
17
|
+
let rawAudio: RawAudio | undefined
|
|
18
|
+
|
|
19
|
+
if (options.audio) {
|
|
20
|
+
rawAudio = await ensureRawAudio(options.audio)
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
let sourceLanguage = options.sourceLanguage
|
|
24
|
+
|
|
25
|
+
if (options.sourceLanguage) {
|
|
26
|
+
const languageData = await parseLangIdentifier(options.sourceLanguage)
|
|
27
|
+
|
|
28
|
+
sourceLanguage = languageData.Name
|
|
29
|
+
|
|
30
|
+
logger.end()
|
|
31
|
+
logger.logTitledMessage('Source language specified', formatLanguageCodeWithName(sourceLanguage))
|
|
32
|
+
} else {
|
|
33
|
+
logger.start('No source language specified. Detect source language')
|
|
34
|
+
|
|
35
|
+
const timelineText = inputTimeline.map(entry => entry.text).join(' ')
|
|
36
|
+
const { detectedLanguage } = await API.detectTextLanguage(timelineText, options.languageDetection || {})
|
|
37
|
+
|
|
38
|
+
sourceLanguage = detectedLanguage
|
|
39
|
+
|
|
40
|
+
logger.end()
|
|
41
|
+
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
let targetLanguage: string
|
|
45
|
+
|
|
46
|
+
if (options.targetLanguage) {
|
|
47
|
+
const languageData = await parseLangIdentifier(options.targetLanguage)
|
|
48
|
+
|
|
49
|
+
targetLanguage = languageData.Name
|
|
50
|
+
|
|
51
|
+
logger.end()
|
|
52
|
+
logger.logTitledMessage('Target language specified', formatLanguageCodeWithName(targetLanguage))
|
|
53
|
+
} else {
|
|
54
|
+
logger.start('No target language specified. Detect target language')
|
|
55
|
+
const { detectedLanguage } = await API.detectTextLanguage(translatedTranscript, options.languageDetection || {})
|
|
56
|
+
|
|
57
|
+
targetLanguage = detectedLanguage
|
|
58
|
+
|
|
59
|
+
logger.end()
|
|
60
|
+
logger.logTitledMessage('Target language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
logger.log(`Load ${options.engine} module`)
|
|
64
|
+
|
|
65
|
+
let mappedWordTimeline: Timeline
|
|
66
|
+
|
|
67
|
+
switch (options.engine) {
|
|
68
|
+
case 'e5': {
|
|
69
|
+
const { alignTimelineToTextSemantically, e5SupportedLanguages } = await import('../alignment/SemanticTextAlignment.js')
|
|
70
|
+
|
|
71
|
+
const shortSourceLanguageCode = getShortLanguageCode(sourceLanguage)
|
|
72
|
+
if (!e5SupportedLanguages.includes(shortSourceLanguageCode)) {
|
|
73
|
+
throw new Error(`Source language ${formatLanguageCodeWithName(sourceLanguage)} is not supported by the E5 embedding model.`)
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const shortTargetLanguageCode = getShortLanguageCode(targetLanguage)
|
|
77
|
+
if (!e5SupportedLanguages.includes(shortTargetLanguageCode)) {
|
|
78
|
+
throw new Error(`Target language ${formatLanguageCodeWithName(targetLanguage)} is not supported by the E5 embedding model.`)
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
logger.end()
|
|
82
|
+
|
|
83
|
+
mappedWordTimeline = await alignTimelineToTextSemantically(
|
|
84
|
+
inputTimeline,
|
|
85
|
+
translatedTranscript,
|
|
86
|
+
targetLanguage)
|
|
87
|
+
|
|
88
|
+
break
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
default: {
|
|
92
|
+
throw new Error(`Unsupported engine: ${options.engine}`)
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
logger.start(`Postprocess timeline`)
|
|
97
|
+
|
|
98
|
+
addWordTextOffsetsToTimeline(mappedWordTimeline, translatedTranscript)
|
|
99
|
+
|
|
100
|
+
const { segmentTimeline: mappedTimeline } = await wordTimelineToSegmentSentenceTimeline(mappedWordTimeline, translatedTranscript, targetLanguage)
|
|
101
|
+
|
|
102
|
+
logger.end()
|
|
103
|
+
logger.logDuration(`Total timeline translation alignment time`, startTimestamp, chalk.magentaBright)
|
|
104
|
+
|
|
105
|
+
logger.end()
|
|
106
|
+
|
|
107
|
+
return {
|
|
108
|
+
timeline: mappedTimeline,
|
|
109
|
+
wordTimeline: mappedWordTimeline,
|
|
110
|
+
|
|
111
|
+
sourceLanguage,
|
|
112
|
+
targetLanguage,
|
|
113
|
+
|
|
114
|
+
rawAudio,
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
// Types
|
|
119
|
+
export interface TimelineTranslationAlignmentResult {
|
|
120
|
+
timeline: Timeline
|
|
121
|
+
wordTimeline: Timeline
|
|
122
|
+
|
|
123
|
+
sourceLanguage?: string
|
|
124
|
+
targetLanguage: string
|
|
125
|
+
|
|
126
|
+
rawAudio?: RawAudio
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
export interface TimelineTranslationAlignmentOptions {
|
|
130
|
+
engine?: 'e5'
|
|
131
|
+
|
|
132
|
+
sourceLanguage?: string
|
|
133
|
+
targetLanguage?: string
|
|
134
|
+
|
|
135
|
+
audio?: AudioSourceParam
|
|
136
|
+
|
|
137
|
+
languageDetection?: API.TextLanguageDetectionOptions
|
|
138
|
+
|
|
139
|
+
subtitles?: SubtitlesConfig
|
|
140
|
+
|
|
141
|
+
e5?: {
|
|
142
|
+
model: 'small-fp16'
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
// Constants
|
|
147
|
+
const defaultTimelineTranslationAlignmentOptions: TimelineTranslationAlignmentOptions = {
|
|
148
|
+
engine: 'e5',
|
|
149
|
+
|
|
150
|
+
sourceLanguage: undefined,
|
|
151
|
+
targetLanguage: undefined,
|
|
152
|
+
|
|
153
|
+
audio: undefined,
|
|
154
|
+
|
|
155
|
+
languageDetection: undefined,
|
|
156
|
+
|
|
157
|
+
subtitles: undefined,
|
|
158
|
+
|
|
159
|
+
e5: {
|
|
160
|
+
model: 'small-fp16',
|
|
161
|
+
}
|
|
162
|
+
}
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
2
|
+
|
|
3
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
4
|
+
import { AudioSourceParam, RawAudio } from '../audio/AudioUtilities.js'
|
|
5
|
+
import { Logger } from '../utilities/Logger.js'
|
|
6
|
+
|
|
7
|
+
import * as API from './API.js'
|
|
8
|
+
import { Timeline } from '../utilities/Timeline.js'
|
|
9
|
+
import chalk from 'chalk'
|
|
10
|
+
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
11
|
+
|
|
12
|
+
const log = logToStderr
|
|
13
|
+
|
|
14
|
+
export async function alignTranscriptAndTranslation(input: AudioSourceParam, transcript: string, translatedTranscript: string, options: TranscriptAndTranslationAlignmentOptions): Promise<TranscriptAndTranslationAlignmentResult> {
|
|
15
|
+
const logger = new Logger()
|
|
16
|
+
|
|
17
|
+
const startTimestamp = logger.getTimestamp()
|
|
18
|
+
|
|
19
|
+
options = extendDeep(defaultTranscriptAndTranslationAlignmentOptions, options)
|
|
20
|
+
|
|
21
|
+
if (options.sourceLanguage && !options.alignment?.language) {
|
|
22
|
+
options.alignment = extendDeep(options.alignment || {}, { language: options.sourceLanguage })
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
if (options.targetLanguage && !options.timelineAlignment?.targetLanguage) {
|
|
26
|
+
options.timelineAlignment = extendDeep(options.timelineAlignment || {}, { targetLanguage: options.targetLanguage })
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
let alignmentResult: API.AlignmentResult
|
|
30
|
+
let timelineAlignmentResult: API.TimelineTranslationAlignmentResult
|
|
31
|
+
|
|
32
|
+
switch (options.engine) {
|
|
33
|
+
case 'two-stage': {
|
|
34
|
+
logger.logTitledMessage(`Start stage 1`, `Align speech to transcript`, chalk.magentaBright)
|
|
35
|
+
logger.end()
|
|
36
|
+
|
|
37
|
+
alignmentResult = await API.align(input, transcript, options.alignment || {})
|
|
38
|
+
|
|
39
|
+
logger.log(``)
|
|
40
|
+
logger.logTitledMessage(`Start stage 2`, `Align timeline to translated transcript`, chalk.magentaBright)
|
|
41
|
+
logger.end()
|
|
42
|
+
|
|
43
|
+
timelineAlignmentResult = await API.alignTimelineTranslation(alignmentResult.timeline, translatedTranscript, options.timelineAlignment || {})
|
|
44
|
+
|
|
45
|
+
break
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
default: {
|
|
49
|
+
throw new Error(`Engine '${options.engine}' is not supported`)
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
logger.end()
|
|
54
|
+
|
|
55
|
+
logger.log(``)
|
|
56
|
+
logger.logDuration(`Total transcript and translation alignment time`, startTimestamp, chalk.magentaBright)
|
|
57
|
+
|
|
58
|
+
return {
|
|
59
|
+
timeline: alignmentResult.timeline,
|
|
60
|
+
wordTimeline: alignmentResult.wordTimeline,
|
|
61
|
+
|
|
62
|
+
translatedTimeline: timelineAlignmentResult.timeline,
|
|
63
|
+
translatedWordTimeline: timelineAlignmentResult.wordTimeline,
|
|
64
|
+
|
|
65
|
+
transcript,
|
|
66
|
+
translatedTranscript,
|
|
67
|
+
|
|
68
|
+
sourceLanguage: alignmentResult.language,
|
|
69
|
+
targetLanguage: timelineAlignmentResult.targetLanguage,
|
|
70
|
+
|
|
71
|
+
inputRawAudio: alignmentResult.inputRawAudio,
|
|
72
|
+
isolatedRawAudio: alignmentResult.isolatedRawAudio,
|
|
73
|
+
backgroundRawAudio: alignmentResult.backgroundRawAudio,
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
export interface TranscriptAndTranslationAlignmentResult {
|
|
78
|
+
timeline: Timeline
|
|
79
|
+
wordTimeline: Timeline
|
|
80
|
+
|
|
81
|
+
translatedTimeline: Timeline
|
|
82
|
+
translatedWordTimeline: Timeline
|
|
83
|
+
|
|
84
|
+
transcript: string
|
|
85
|
+
translatedTranscript: string
|
|
86
|
+
|
|
87
|
+
sourceLanguage: string
|
|
88
|
+
targetLanguage: string
|
|
89
|
+
|
|
90
|
+
inputRawAudio: RawAudio
|
|
91
|
+
isolatedRawAudio?: RawAudio
|
|
92
|
+
backgroundRawAudio?: RawAudio
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export type TranscriptAndTranslationAlignmentEngine = 'two-stage'
|
|
96
|
+
|
|
97
|
+
export interface TranscriptAndTranslationAlignmentOptions {
|
|
98
|
+
engine?: TranscriptAndTranslationAlignmentEngine
|
|
99
|
+
|
|
100
|
+
sourceLanguage?: string
|
|
101
|
+
targetLanguage?: string
|
|
102
|
+
|
|
103
|
+
isolate?: boolean
|
|
104
|
+
|
|
105
|
+
crop?: boolean
|
|
106
|
+
|
|
107
|
+
alignment?: API.AlignmentOptions
|
|
108
|
+
|
|
109
|
+
timelineAlignment?: API.TimelineTranslationAlignmentOptions
|
|
110
|
+
|
|
111
|
+
languageDetection?: API.TextLanguageDetectionOptions
|
|
112
|
+
|
|
113
|
+
vad?: API.VADOptions
|
|
114
|
+
|
|
115
|
+
plainText?: API.PlainTextOptions
|
|
116
|
+
|
|
117
|
+
subtitles?: SubtitlesConfig
|
|
118
|
+
|
|
119
|
+
sourceSeparation?: API.SourceSeparationOptions
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
export const defaultTranscriptAndTranslationAlignmentOptions: TranscriptAndTranslationAlignmentOptions = {
|
|
123
|
+
engine: 'two-stage',
|
|
124
|
+
|
|
125
|
+
sourceLanguage: undefined,
|
|
126
|
+
targetLanguage: undefined,
|
|
127
|
+
|
|
128
|
+
isolate: false,
|
|
129
|
+
|
|
130
|
+
crop: true,
|
|
131
|
+
|
|
132
|
+
alignment: {
|
|
133
|
+
},
|
|
134
|
+
|
|
135
|
+
timelineAlignment: {
|
|
136
|
+
},
|
|
137
|
+
|
|
138
|
+
languageDetection: {
|
|
139
|
+
},
|
|
140
|
+
|
|
141
|
+
plainText: {
|
|
142
|
+
paragraphBreaks: 'double',
|
|
143
|
+
whitespace: 'collapse'
|
|
144
|
+
},
|
|
145
|
+
|
|
146
|
+
subtitles: {
|
|
147
|
+
},
|
|
148
|
+
|
|
149
|
+
vad: {
|
|
150
|
+
engine: 'adaptive-gate'
|
|
151
|
+
},
|
|
152
|
+
|
|
153
|
+
sourceSeparation: {
|
|
154
|
+
},
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
export const TranscriptAndTranslationAlignmentEngines: API.EngineMetadata[] = [
|
|
158
|
+
{
|
|
159
|
+
id: 'two-stage',
|
|
160
|
+
name: 'Two-stage translation alignment',
|
|
161
|
+
description: 'Applies two-stage translation alignment to the spoken audio. First stage aligns the speech to the native language transcript. Second stage aligns the resulting timeline with the translated transcript.',
|
|
162
|
+
type: 'local'
|
|
163
|
+
}
|
|
164
|
+
]
|
|
@@ -6,14 +6,14 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
|
|
7
7
|
import * as API from './API.js'
|
|
8
8
|
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
9
|
-
import { formatLanguageCodeWithName, getShortLanguageCode,
|
|
9
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeIdentifierToLanguageCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
10
10
|
import { type WhisperAlignmentOptions } from '../recognition/WhisperSTT.js'
|
|
11
11
|
import chalk from 'chalk'
|
|
12
12
|
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
13
13
|
|
|
14
14
|
const log = logToStderr
|
|
15
15
|
|
|
16
|
-
export async function alignTranslation(input: AudioSourceParam,
|
|
16
|
+
export async function alignTranslation(input: AudioSourceParam, translatedTranscript: string, options: TranslationAlignmentOptions): Promise<TranslationAlignmentResult> {
|
|
17
17
|
const logger = new Logger()
|
|
18
18
|
|
|
19
19
|
const startTimestamp = logger.getTimestamp()
|
|
@@ -77,7 +77,7 @@ export async function alignTranslation(input: AudioSourceParam, transcript: stri
|
|
|
77
77
|
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
78
78
|
}
|
|
79
79
|
|
|
80
|
-
const targetLanguage = await
|
|
80
|
+
const targetLanguage = await normalizeIdentifierToLanguageCode(options.targetLanguage!)
|
|
81
81
|
|
|
82
82
|
logger.logTitledMessage('Target language', formatLanguageCodeWithName(targetLanguage))
|
|
83
83
|
|
|
@@ -108,7 +108,7 @@ export async function alignTranslation(input: AudioSourceParam, transcript: stri
|
|
|
108
108
|
throw new Error('Whisper translation tasks are only possible with a multilingual model')
|
|
109
109
|
}
|
|
110
110
|
|
|
111
|
-
mappedTimeline = await WhisperSTT.alignEnglishTranslation(sourceRawAudio,
|
|
111
|
+
mappedTimeline = await WhisperSTT.alignEnglishTranslation(sourceRawAudio, translatedTranscript, modelName, modelDir, shortSourceLanguageCode, whisperAlignmnentOptions)
|
|
112
112
|
|
|
113
113
|
break
|
|
114
114
|
}
|
|
@@ -124,10 +124,10 @@ export async function alignTranslation(input: AudioSourceParam, transcript: stri
|
|
|
124
124
|
}
|
|
125
125
|
|
|
126
126
|
// Add text offsets
|
|
127
|
-
addWordTextOffsetsToTimeline(mappedTimeline,
|
|
127
|
+
addWordTextOffsetsToTimeline(mappedTimeline, translatedTranscript)
|
|
128
128
|
|
|
129
129
|
// Make segment timeline
|
|
130
|
-
const { segmentTimeline } = await wordTimelineToSegmentSentenceTimeline(mappedTimeline,
|
|
130
|
+
const { segmentTimeline } = await wordTimelineToSegmentSentenceTimeline(mappedTimeline, translatedTranscript, sourceLanguage, options.plainText?.paragraphBreaks, options.plainText?.whitespace)
|
|
131
131
|
|
|
132
132
|
logger.end()
|
|
133
133
|
logger.logDuration(`Total translation alignment time`, startTimestamp, chalk.magentaBright)
|
|
@@ -136,8 +136,9 @@ export async function alignTranslation(input: AudioSourceParam, transcript: stri
|
|
|
136
136
|
timeline: segmentTimeline,
|
|
137
137
|
wordTimeline: mappedTimeline,
|
|
138
138
|
|
|
139
|
-
|
|
140
|
-
|
|
139
|
+
translatedTranscript,
|
|
140
|
+
sourceLanguage,
|
|
141
|
+
targetLanguage,
|
|
141
142
|
|
|
142
143
|
inputRawAudio,
|
|
143
144
|
isolatedRawAudio,
|
|
@@ -149,8 +150,9 @@ export interface TranslationAlignmentResult {
|
|
|
149
150
|
timeline: Timeline
|
|
150
151
|
wordTimeline: Timeline
|
|
151
152
|
|
|
152
|
-
|
|
153
|
-
|
|
153
|
+
translatedTranscript: string
|
|
154
|
+
sourceLanguage: string
|
|
155
|
+
targetLanguage: string
|
|
154
156
|
|
|
155
157
|
inputRawAudio: RawAudio
|
|
156
158
|
isolatedRawAudio?: RawAudio
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
2
2
|
|
|
3
|
-
import { logToStderr
|
|
3
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
4
4
|
import { AudioSourceParam, RawAudio, cropToTimeline, ensureRawAudio, } from '../audio/AudioUtilities.js'
|
|
5
5
|
import { Logger } from '../utilities/Logger.js'
|
|
6
6
|
|
|
@@ -220,7 +220,14 @@ export function convertCroppedToUncroppedTimeline(timeline: Timeline, uncropTime
|
|
|
220
220
|
}
|
|
221
221
|
}
|
|
222
222
|
|
|
223
|
-
function mapUsingUncropTimeline(startTimeInCroppedAudio: number, endTimeInCroppedAudio: number, uncropTimeline: Timeline) {
|
|
223
|
+
function mapUsingUncropTimeline(startTimeInCroppedAudio: number, endTimeInCroppedAudio: number, uncropTimeline: Timeline): UncropTimelineMapResult {
|
|
224
|
+
if (uncropTimeline.length === 0) {
|
|
225
|
+
return {
|
|
226
|
+
mappedStartTime: 0,
|
|
227
|
+
mappedEndTime: 0,
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
|
|
224
231
|
let offsetInCroppedAudio = 0
|
|
225
232
|
|
|
226
233
|
if (endTimeInCroppedAudio < startTimeInCroppedAudio) {
|
|
@@ -250,7 +257,16 @@ function mapUsingUncropTimeline(startTimeInCroppedAudio: number, endTimeInCroppe
|
|
|
250
257
|
}
|
|
251
258
|
|
|
252
259
|
if (bestOverlapDuration === -1) {
|
|
253
|
-
|
|
260
|
+
if (startTimeInCroppedAudio >= offsetInCroppedAudio) {
|
|
261
|
+
const maxTimestamp = uncropTimeline[uncropTimeline.length - 1].endTime
|
|
262
|
+
|
|
263
|
+
return {
|
|
264
|
+
mappedStartTime: maxTimestamp,
|
|
265
|
+
mappedEndTime: maxTimestamp
|
|
266
|
+
}
|
|
267
|
+
} else {
|
|
268
|
+
throw new Error(`Given start time ${startTimeInCroppedAudio} was smaller than audio duration but no match was found in uncrop timeline (should not occur)`)
|
|
269
|
+
}
|
|
254
270
|
}
|
|
255
271
|
|
|
256
272
|
return {
|
|
@@ -259,6 +275,11 @@ function mapUsingUncropTimeline(startTimeInCroppedAudio: number, endTimeInCroppe
|
|
|
259
275
|
}
|
|
260
276
|
}
|
|
261
277
|
|
|
278
|
+
interface UncropTimelineMapResult {
|
|
279
|
+
mappedStartTime: number
|
|
280
|
+
mappedEndTime: number
|
|
281
|
+
}
|
|
282
|
+
|
|
262
283
|
export interface VADResult {
|
|
263
284
|
timeline: Timeline
|
|
264
285
|
verboseTimeline: Timeline
|