echogarden 1.2.1 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/tables/lcid-table.json +7427 -7427
- package/dist/api/Alignment.js +5 -3
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/Recognition.js +12 -10
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/Synthesis.js +10 -14
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.js +21 -14
- package/dist/api/Translation.js.map +1 -1
- package/dist/api/TranslationAlignment.js +8 -4
- package/dist/api/TranslationAlignment.js.map +1 -1
- package/dist/api/VoiceActivityDetection.d.ts +0 -4
- package/dist/api/VoiceActivityDetection.js +38 -18
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +4 -4
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/utilities/Locale.d.ts +14 -8
- package/dist/utilities/Locale.js +52 -11
- package/dist/utilities/Locale.js.map +1 -1
- package/docs/Tasklist.md +11 -6
- package/package.json +6 -6
- package/src/api/Alignment.ts +9 -3
- package/src/api/Recognition.ts +15 -12
- package/src/api/Synthesis.ts +12 -20
- package/src/api/Translation.ts +26 -21
- package/src/api/TranslationAlignment.ts +12 -5
- package/src/api/VoiceActivityDetection.ts +49 -18
- package/src/recognition/WhisperSTT.ts +4 -4
- package/src/utilities/Locale.ts +84 -22
package/docs/Tasklist.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## Bugs
|
|
4
4
|
|
|
5
|
-
### Alignment
|
|
5
|
+
### Alignment / DTW-RA
|
|
6
6
|
|
|
7
7
|
* In DTW-RA, a recognized transcript including something like "Question 2.What does Juan", where "2.What" has a point in the middle, is breaking playback of the timeline
|
|
8
8
|
* DTW-RA will not work correctly with Polish language texts, due to issues with the eSpeak engine pronouncing `|` characters, which are intended to be used as separators and ignored by all other eSpeak languages
|
|
@@ -83,7 +83,6 @@
|
|
|
83
83
|
* `speak-youtube-subtitles`: To speak the subtitles of a YouTube video
|
|
84
84
|
|
|
85
85
|
### API
|
|
86
|
-
* Accept full language names as language identifiers
|
|
87
86
|
* Validate timelines to ensure timestamps are always increasing: no negative timestamps or timestamps over the duration of the audio. No sentences without words, etc. and correct if needed
|
|
88
87
|
* See whether it's possible to detect and include / remove Emoji characters in timelines
|
|
89
88
|
* Add support for phrases in timelines
|
|
@@ -120,7 +119,7 @@
|
|
|
120
119
|
* Decide whether asterisk `*` should be spoken when using `speak-url` or `speak-wikipedia`
|
|
121
120
|
* Add partial SSML support for all engines. In particular, allow changing language or voice using the `<voice>` and `<lang>` tags, `<say-as>` and `<phoneme>` where possible
|
|
122
121
|
* Try to remove reliance on `()` after `.` character hack in `EspeakTTS.synthesizeFragments`
|
|
123
|
-
* eSpeak IPA output
|
|
122
|
+
* eSpeak IPA output applies stress marks on vowels, not syllables - which is usually the standard for IPA. Consider how to make a conversion to and from these two approaches (possibly detect it automatically), to provide users with more useful phonemizations
|
|
124
123
|
* Decide if `msspeech` engine should be selected if available. This would require attempting to load a matching voice, and falling back if it is not installed
|
|
125
124
|
* Speaker-specific voice option
|
|
126
125
|
* Use VAD on the synthesized audio file to get more accurate sentence or word segmentation
|
|
@@ -149,7 +148,6 @@
|
|
|
149
148
|
* Currently, when input is set to be SSML, it is wrapped in a `<speak>` tag. Handle the case where the user made their own SSML document wrapped with a `<speak>` tag as well. Currently, it may send invalid input to Azure
|
|
150
149
|
|
|
151
150
|
### Recognition
|
|
152
|
-
* Recognized word entries that span VAD segment boundaries can be split
|
|
153
151
|
* Show alternatives when playing in the CLI. Clear current line and rewrite already printed text for alternatives during the speech recognition process
|
|
154
152
|
|
|
155
153
|
### Recognition / Whisper
|
|
@@ -161,8 +159,6 @@
|
|
|
161
159
|
|
|
162
160
|
### Alignment
|
|
163
161
|
|
|
164
|
-
* Aligned words entries that span VAD boundaries may be split
|
|
165
|
-
|
|
166
162
|
### Alignment / DTW-RA
|
|
167
163
|
|
|
168
164
|
### Alignment / Whisper
|
|
@@ -256,6 +252,15 @@
|
|
|
256
252
|
* Allow `dtw` mode work with more speech synthesizers to produce its reference
|
|
257
253
|
* Predict timing for individual letters (graphemes) based on phoneme timestamps (especially useful for Chinese and Japanese)
|
|
258
254
|
|
|
255
|
+
### Translation
|
|
256
|
+
* Add text-to-text translation with cloud translation APIs like Google Translate or DeepL, or offline models like OpenNMT or NLLB-200
|
|
257
|
+
|
|
258
|
+
### Translation alignment
|
|
259
|
+
* Perform word-level alignment of text-to-text translations from and to any language (not just English) using methods like multilingual embeddings, or specialized models, and then use the text-based alignment to align speech in any recognized language, to its translated transcript in any language supported by the text-to-text alignment approach
|
|
260
|
+
|
|
261
|
+
### Speech-to-text translation
|
|
262
|
+
|
|
263
|
+
* Hybrid approach: recognize speech in its language using any recognition model, then translate the resulting transcript using a text-to-text translation engine, and then align the translated transcript to the original one using text-to-text alignment, and map back to the original speech using the recognition timestamps, to get word-level alignment for the translated transcript
|
|
259
264
|
|
|
260
265
|
## Possible new engines or platforms
|
|
261
266
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "echogarden",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.3.0",
|
|
4
4
|
"description": "An easy-to-use speech toolset. Includes tools for synthesis, recognition, alignment, speech translation, language detection, source separation and more.",
|
|
5
5
|
"author": "Rotem Dan",
|
|
6
6
|
"license": "GPL-3.0",
|
|
@@ -55,8 +55,8 @@
|
|
|
55
55
|
"echogarden": "./dist/cli/CLILauncher.js"
|
|
56
56
|
},
|
|
57
57
|
"dependencies": {
|
|
58
|
-
"@aws-sdk/client-polly": "^3.
|
|
59
|
-
"@aws-sdk/client-transcribe-streaming": "^3.
|
|
58
|
+
"@aws-sdk/client-polly": "^3.567.0",
|
|
59
|
+
"@aws-sdk/client-transcribe-streaming": "^3.567.0",
|
|
60
60
|
"@echogarden/espeak-ng-emscripten": "^0.1.2",
|
|
61
61
|
"@echogarden/fasttext-wasm": "^0.1.0",
|
|
62
62
|
"@echogarden/flite-wasi": "^0.1.1",
|
|
@@ -79,7 +79,7 @@
|
|
|
79
79
|
"graceful-fs": "^4.2.11",
|
|
80
80
|
"html-escaper": "^3.0.3",
|
|
81
81
|
"html-to-text": "^9.0.5",
|
|
82
|
-
"import-meta-resolve": "^4.
|
|
82
|
+
"import-meta-resolve": "^4.1.0",
|
|
83
83
|
"jieba-wasm": "^0.0.2",
|
|
84
84
|
"jsdom": "^24.0.0",
|
|
85
85
|
"json5": "^2.2.3",
|
|
@@ -88,7 +88,7 @@
|
|
|
88
88
|
"moving-median": "^1.0.0",
|
|
89
89
|
"msgpack-lite": "^0.1.26",
|
|
90
90
|
"onnxruntime-node": "^1.17.3",
|
|
91
|
-
"openai": "^4.
|
|
91
|
+
"openai": "^4.40.0",
|
|
92
92
|
"sam-js": "^0.2.1",
|
|
93
93
|
"strip-ansi": "^7.1.0",
|
|
94
94
|
"tar": "^7.0.1",
|
|
@@ -119,7 +119,7 @@
|
|
|
119
119
|
"@types/graceful-fs": "^4.1.9",
|
|
120
120
|
"@types/jsdom": "^21.1.6",
|
|
121
121
|
"@types/msgpack-lite": "^0.1.11",
|
|
122
|
-
"@types/node": "^20.12.
|
|
122
|
+
"@types/node": "^20.12.8",
|
|
123
123
|
"@types/recursive-readdir": "^2.2.4",
|
|
124
124
|
"@types/tar": "^6.1.13",
|
|
125
125
|
"@types/ws": "^8.5.10",
|
package/src/api/Alignment.ts
CHANGED
|
@@ -6,7 +6,7 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
|
|
7
7
|
import * as API from './API.js'
|
|
8
8
|
import { Timeline, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
9
|
-
import { formatLanguageCodeWithName, getDefaultDialectForLanguageCodeIfPossible, getShortLanguageCode,
|
|
9
|
+
import { formatLanguageCodeWithName, getDefaultDialectForLanguageCodeIfPossible, getShortLanguageCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
10
10
|
import { type WhisperAlignmentOptions } from '../recognition/WhisperSTT.js'
|
|
11
11
|
import chalk from 'chalk'
|
|
12
12
|
import { DtwGranularity, createAlignmentReferenceUsingEspeak } from '../alignment/SpeechAlignment.js'
|
|
@@ -72,14 +72,20 @@ export async function align(input: AudioSourceParam, transcript: string, options
|
|
|
72
72
|
let language: string
|
|
73
73
|
|
|
74
74
|
if (options.language) {
|
|
75
|
-
|
|
75
|
+
const languageData = await parseLangIdentifier(options.language)
|
|
76
|
+
|
|
77
|
+
language = languageData.Name
|
|
78
|
+
|
|
79
|
+
logger.logTitledMessage('Language specified', formatLanguageCodeWithName(language))
|
|
76
80
|
} else {
|
|
77
81
|
logger.start('No language specified. Detecting language')
|
|
78
82
|
const { detectedLanguage } = await API.detectTextLanguage(transcript, options.languageDetection || {})
|
|
83
|
+
|
|
79
84
|
language = detectedLanguage
|
|
80
85
|
|
|
81
86
|
logger.end()
|
|
82
|
-
|
|
87
|
+
|
|
88
|
+
logger.logTitledMessage('Language detected', formatLanguageCodeWithName(language))
|
|
83
89
|
}
|
|
84
90
|
|
|
85
91
|
language = getDefaultDialectForLanguageCodeIfPossible(language)
|
package/src/api/Recognition.ts
CHANGED
|
@@ -6,7 +6,7 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
|
|
7
7
|
import * as API from './API.js'
|
|
8
8
|
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
9
|
-
import { formatLanguageCodeWithName,
|
|
9
|
+
import { formatLanguageCodeWithName, parseLangIdentifier } from '../utilities/Locale.js'
|
|
10
10
|
import { loadPackage } from '../utilities/PackageManager.js'
|
|
11
11
|
import chalk from 'chalk'
|
|
12
12
|
|
|
@@ -62,24 +62,27 @@ export async function recognize(input: AudioSourceParam, options: RecognitionOpt
|
|
|
62
62
|
|
|
63
63
|
const engine = options.engine!
|
|
64
64
|
|
|
65
|
-
if (
|
|
66
|
-
|
|
67
|
-
const { detectedLanguage } = await API.detectSpeechLanguage(sourceRawAudio, options.languageDetection!)
|
|
65
|
+
if (options.language) {
|
|
66
|
+
const languageData = await parseLangIdentifier(options.language)
|
|
68
67
|
|
|
69
|
-
|
|
70
|
-
logger.logTitledMessage('Language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
68
|
+
options.language = languageData.Name
|
|
71
69
|
|
|
72
|
-
options.language = detectedLanguage
|
|
73
|
-
} else {
|
|
74
70
|
logger.end()
|
|
71
|
+
logger.logTitledMessage('Language specified', formatLanguageCodeWithName(options.language))
|
|
72
|
+
} else {
|
|
73
|
+
logger.start('No language specified. Detecting speech language')
|
|
74
|
+
const { detectedLanguage } = await API.detectSpeechLanguage(sourceRawAudio, options.languageDetection!)
|
|
75
75
|
|
|
76
|
-
|
|
76
|
+
options.language = detectedLanguage
|
|
77
77
|
|
|
78
|
-
logger.
|
|
78
|
+
logger.end()
|
|
79
|
+
logger.logTitledMessage('Language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
79
80
|
}
|
|
80
81
|
|
|
81
|
-
const
|
|
82
|
-
|
|
82
|
+
const languageData = await parseLangIdentifier(options.language)
|
|
83
|
+
|
|
84
|
+
const languageCode = languageData.Name
|
|
85
|
+
const shortLanguageCode = languageData.TwoLetterISOLanguageName
|
|
83
86
|
|
|
84
87
|
let transcript: string
|
|
85
88
|
let timeline: Timeline | undefined
|
package/src/api/Synthesis.ts
CHANGED
|
@@ -15,7 +15,7 @@ import { loadLexiconsForLanguage } from '../nlp/Lexicon.js'
|
|
|
15
15
|
import * as API from './API.js'
|
|
16
16
|
import { Timeline, TimelineEntry, addTimeOffsetToTimeline, multiplyTimelineByFactor } from '../utilities/Timeline.js'
|
|
17
17
|
import { getAppDataDir, ensureDir, existsSync, isFileIsUpToDate, readAndParseJsonFile, writeFileSafe } from '../utilities/FileSystem.js'
|
|
18
|
-
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeLanguageCode, defaultDialectForLanguageCode } from '../utilities/Locale.js'
|
|
18
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeLanguageCode, defaultDialectForLanguageCode, parseLangIdentifier, normalizeIdentifierToLangaugeCode } from '../utilities/Locale.js'
|
|
19
19
|
import { loadPackage } from '../utilities/PackageManager.js'
|
|
20
20
|
import { EngineMetadata, appName } from './Common.js'
|
|
21
21
|
import { shouldCancelCurrentTask } from '../server/Worker.js'
|
|
@@ -328,7 +328,15 @@ async function synthesizeSegment(text: string, options: SynthesisOptions) {
|
|
|
328
328
|
logger.start(`Initialize ${engine} module`)
|
|
329
329
|
|
|
330
330
|
const voice = selectedVoice.name
|
|
331
|
-
|
|
331
|
+
|
|
332
|
+
let language: string
|
|
333
|
+
|
|
334
|
+
if (options.language) {
|
|
335
|
+
language = await normalizeIdentifierToLangaugeCode(options.language)
|
|
336
|
+
} else {
|
|
337
|
+
language = selectedVoice.languages[0]
|
|
338
|
+
}
|
|
339
|
+
|
|
332
340
|
const voiceGender = selectedVoice.gender
|
|
333
341
|
|
|
334
342
|
const speed = clip(options.speed!, 0.1, 10.0)
|
|
@@ -1548,7 +1556,7 @@ export async function requestVoiceList(options: VoiceListRequestOptions): Promis
|
|
|
1548
1556
|
voiceList = await loadVoiceList()
|
|
1549
1557
|
}
|
|
1550
1558
|
|
|
1551
|
-
const languageCode =
|
|
1559
|
+
const languageCode = await normalizeIdentifierToLangaugeCode(options.language || '')
|
|
1552
1560
|
|
|
1553
1561
|
if (languageCode) {
|
|
1554
1562
|
let filteredVoiceList = voiceList.filter(voice => voice.languages.includes(languageCode))
|
|
@@ -1611,7 +1619,7 @@ export interface RequestVoiceListResult {
|
|
|
1611
1619
|
}
|
|
1612
1620
|
|
|
1613
1621
|
export async function selectBestOfflineEngineForLanguage(language: string): Promise<SynthesisEngine> {
|
|
1614
|
-
language =
|
|
1622
|
+
language = await normalizeIdentifierToLangaugeCode(language)
|
|
1615
1623
|
|
|
1616
1624
|
const VitsTTS = await import('../synthesis/VitsTTS.js')
|
|
1617
1625
|
|
|
@@ -1621,22 +1629,6 @@ export async function selectBestOfflineEngineForLanguage(language: string): Prom
|
|
|
1621
1629
|
return 'vits'
|
|
1622
1630
|
}
|
|
1623
1631
|
|
|
1624
|
-
const FliteTTS = await import('../synthesis/FliteTTS.js')
|
|
1625
|
-
|
|
1626
|
-
const fliteLanguages = getAllLangCodesFromVoiceList(FliteTTS.voiceList)
|
|
1627
|
-
|
|
1628
|
-
if (fliteLanguages.includes(language)) {
|
|
1629
|
-
return 'flite'
|
|
1630
|
-
}
|
|
1631
|
-
|
|
1632
|
-
const SvoxPicoTTS = await import('../synthesis/SvoxPicoTTS.js')
|
|
1633
|
-
|
|
1634
|
-
const picoLanguages = getAllLangCodesFromVoiceList(SvoxPicoTTS.voiceList)
|
|
1635
|
-
|
|
1636
|
-
if (picoLanguages.includes(language)) {
|
|
1637
|
-
return 'pico'
|
|
1638
|
-
}
|
|
1639
|
-
|
|
1640
1632
|
return 'espeak'
|
|
1641
1633
|
}
|
|
1642
1634
|
|
package/src/api/Translation.ts
CHANGED
|
@@ -6,7 +6,7 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
|
|
7
7
|
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
8
8
|
import { type WhisperOptions } from '../recognition/WhisperSTT.js'
|
|
9
|
-
import { formatLanguageCodeWithName, getShortLanguageCode,
|
|
9
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeIdentifierToLangaugeCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
10
10
|
import { EngineMetadata } from './Common.js'
|
|
11
11
|
import { type SpeechLanguageDetectionOptions, detectSpeechLanguage } from './API.js'
|
|
12
12
|
import chalk from 'chalk'
|
|
@@ -62,28 +62,31 @@ export async function translateSpeech(input: AudioSourceParam, options: SpeechTr
|
|
|
62
62
|
sourceRawAudio = normalizeAudioLevel(sourceRawAudio)
|
|
63
63
|
sourceRawAudio.audioChannels[0] = trimAudioEnd(sourceRawAudio.audioChannels[0])
|
|
64
64
|
|
|
65
|
-
if (
|
|
66
|
-
|
|
67
|
-
const { detectedLanguage } = await detectSpeechLanguage(sourceRawAudio, options.languageDetection || {})
|
|
65
|
+
if (options.sourceLanguage) {
|
|
66
|
+
const languageData = await parseLangIdentifier(options.sourceLanguage)
|
|
68
67
|
|
|
69
|
-
|
|
70
|
-
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
68
|
+
options.sourceLanguage = languageData.Name
|
|
71
69
|
|
|
72
|
-
options.sourceLanguage = detectedLanguage
|
|
73
|
-
} else {
|
|
74
70
|
logger.end()
|
|
71
|
+
logger.logTitledMessage('Source language specified', formatLanguageCodeWithName(options.sourceLanguage))
|
|
72
|
+
} else {
|
|
73
|
+
logger.start('No source language specified. Detecting speech language')
|
|
74
|
+
const { detectedLanguage } = await detectSpeechLanguage(sourceRawAudio, options.languageDetection || {})
|
|
75
75
|
|
|
76
|
-
|
|
76
|
+
options.sourceLanguage = detectedLanguage
|
|
77
77
|
|
|
78
|
-
logger.
|
|
78
|
+
logger.end()
|
|
79
|
+
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
79
80
|
}
|
|
80
81
|
|
|
81
|
-
|
|
82
|
+
options.targetLanguage = await normalizeIdentifierToLangaugeCode(options.targetLanguage!)
|
|
83
|
+
|
|
84
|
+
logger.logTitledMessage('Target language', formatLanguageCodeWithName(options.targetLanguage))
|
|
82
85
|
|
|
83
86
|
logger.start('Preprocess audio for translation')
|
|
84
87
|
|
|
85
88
|
const engine = options.engine!
|
|
86
|
-
const sourceLanguage =
|
|
89
|
+
const sourceLanguage = options.sourceLanguage!
|
|
87
90
|
const targetLanguage = options.targetLanguage!
|
|
88
91
|
|
|
89
92
|
let transcript: string
|
|
@@ -119,10 +122,6 @@ export async function translateSpeech(input: AudioSourceParam, options: SpeechTr
|
|
|
119
122
|
|
|
120
123
|
({ transcript, timeline: wordTimeline } = await WhisperSTT.recognize(sourceRawAudio, modelName, modelDir, 'translate', sourceLanguage, whisperOptions))
|
|
121
124
|
|
|
122
|
-
addWordTextOffsetsToTimeline(wordTimeline, transcript);
|
|
123
|
-
|
|
124
|
-
({ segmentTimeline } = await wordTimelineToSegmentSentenceTimeline(wordTimeline, transcript, targetLanguage, 'single', 'preserve'))
|
|
125
|
-
|
|
126
125
|
break
|
|
127
126
|
}
|
|
128
127
|
|
|
@@ -155,9 +154,7 @@ export async function translateSpeech(input: AudioSourceParam, options: SpeechTr
|
|
|
155
154
|
modelName,
|
|
156
155
|
modelPath,
|
|
157
156
|
whisperCppOptions,
|
|
158
|
-
))
|
|
159
|
-
|
|
160
|
-
({ segmentTimeline } = await wordTimelineToSegmentSentenceTimeline(wordTimeline, transcript, targetLanguage, 'single', 'preserve'))
|
|
157
|
+
))
|
|
161
158
|
|
|
162
159
|
break
|
|
163
160
|
}
|
|
@@ -194,13 +191,21 @@ export async function translateSpeech(input: AudioSourceParam, options: SpeechTr
|
|
|
194
191
|
|
|
195
192
|
// If the audio was cropped before recognition, map the timestamps back to the original audio
|
|
196
193
|
if (sourceUncropTimeline && sourceUncropTimeline.length > 0) {
|
|
197
|
-
API.convertCroppedToUncroppedTimeline(segmentTimeline, sourceUncropTimeline)
|
|
198
|
-
|
|
199
194
|
if (wordTimeline) {
|
|
200
195
|
API.convertCroppedToUncroppedTimeline(wordTimeline, sourceUncropTimeline)
|
|
196
|
+
} else if (segmentTimeline) {
|
|
197
|
+
API.convertCroppedToUncroppedTimeline(segmentTimeline, sourceUncropTimeline)
|
|
201
198
|
}
|
|
202
199
|
}
|
|
203
200
|
|
|
201
|
+
if (wordTimeline) {
|
|
202
|
+
addWordTextOffsetsToTimeline(wordTimeline, transcript)
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
if (!segmentTimeline) {
|
|
206
|
+
({ segmentTimeline } = await wordTimelineToSegmentSentenceTimeline(wordTimeline!, transcript, targetLanguage, 'single', 'preserve'))
|
|
207
|
+
}
|
|
208
|
+
|
|
204
209
|
logger.log('')
|
|
205
210
|
logger.logDuration(`Total speech translation time`, startTimestamp, chalk.magentaBright)
|
|
206
211
|
|
|
@@ -6,7 +6,7 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
|
|
7
7
|
import * as API from './API.js'
|
|
8
8
|
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
9
|
-
import { formatLanguageCodeWithName, getShortLanguageCode,
|
|
9
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeIdentifierToLangaugeCode, parseLangIdentifier } from '../utilities/Locale.js'
|
|
10
10
|
import { type WhisperAlignmentOptions } from '../recognition/WhisperSTT.js'
|
|
11
11
|
import chalk from 'chalk'
|
|
12
12
|
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
@@ -57,18 +57,25 @@ export async function alignTranslation(input: AudioSourceParam, transcript: stri
|
|
|
57
57
|
let sourceLanguage: string
|
|
58
58
|
|
|
59
59
|
if (options.sourceLanguage) {
|
|
60
|
-
|
|
60
|
+
const languageData = await parseLangIdentifier(options.sourceLanguage)
|
|
61
|
+
|
|
62
|
+
sourceLanguage = languageData.Name
|
|
63
|
+
|
|
64
|
+
logger.end()
|
|
65
|
+
logger.logTitledMessage('Source language specified', formatLanguageCodeWithName(sourceLanguage))
|
|
61
66
|
} else {
|
|
62
67
|
logger.start('No source language specified. Detecting speech language')
|
|
63
68
|
const { detectedLanguage } = await API.detectSpeechLanguage(sourceRawAudio, options.languageDetection || {})
|
|
64
69
|
|
|
70
|
+
sourceLanguage = detectedLanguage
|
|
71
|
+
|
|
65
72
|
logger.end()
|
|
66
73
|
logger.logTitledMessage('Source language detected', formatLanguageCodeWithName(detectedLanguage))
|
|
67
|
-
|
|
68
|
-
sourceLanguage = detectedLanguage
|
|
69
74
|
}
|
|
70
75
|
|
|
71
|
-
const targetLanguage =
|
|
76
|
+
const targetLanguage = await normalizeIdentifierToLangaugeCode(options.targetLanguage!)
|
|
77
|
+
|
|
78
|
+
logger.logTitledMessage('Target language', formatLanguageCodeWithName(targetLanguage))
|
|
72
79
|
|
|
73
80
|
let mappedTimeline: Timeline
|
|
74
81
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
2
2
|
|
|
3
|
-
import { logToStderr } from '../utilities/Utilities.js'
|
|
3
|
+
import { logToStderr, roundToDigits } from '../utilities/Utilities.js'
|
|
4
4
|
import { AudioSourceParam, RawAudio, cropToTimeline, ensureRawAudio, } from '../audio/AudioUtilities.js'
|
|
5
5
|
import { Logger } from '../utilities/Logger.js'
|
|
6
6
|
|
|
@@ -191,38 +191,69 @@ function frameProbabilitiesToTimeline(frameProbabilities: number[], frameDuratio
|
|
|
191
191
|
}
|
|
192
192
|
|
|
193
193
|
export function convertCroppedToUncroppedTimeline(timeline: Timeline, uncropTimeline: Timeline) {
|
|
194
|
+
if (timeline.length === 0) {
|
|
195
|
+
return
|
|
196
|
+
}
|
|
197
|
+
|
|
194
198
|
for (const entry of timeline) {
|
|
195
|
-
|
|
196
|
-
|
|
199
|
+
const {
|
|
200
|
+
mappedStartTime,
|
|
201
|
+
mappedEndTime
|
|
202
|
+
} = mapUsingUncropTimeline(entry.startTime, entry.endTime, uncropTimeline)
|
|
203
|
+
|
|
204
|
+
const mapSubTimeline = (subTimeline: Timeline | undefined) => {
|
|
205
|
+
if (!subTimeline) {
|
|
206
|
+
return
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
for (const subEntry of subTimeline) {
|
|
210
|
+
subEntry.startTime = Math.min(mappedStartTime + (subEntry.startTime - entry.startTime), mappedEndTime)
|
|
211
|
+
subEntry.endTime = Math.min(mappedStartTime + (subEntry.endTime - entry.startTime), mappedEndTime)
|
|
197
212
|
|
|
198
|
-
|
|
199
|
-
|
|
213
|
+
mapSubTimeline(subEntry.timeline)
|
|
214
|
+
}
|
|
200
215
|
}
|
|
216
|
+
|
|
217
|
+
mapSubTimeline(entry.timeline)
|
|
218
|
+
|
|
219
|
+
entry.startTime = mappedStartTime
|
|
220
|
+
entry.endTime = mappedEndTime
|
|
201
221
|
}
|
|
202
222
|
}
|
|
203
223
|
|
|
204
|
-
|
|
224
|
+
function mapUsingUncropTimeline(startTimeInCroppedAudio: number, endTimeInCroppedAudio: number, uncropTimeline: Timeline) {
|
|
205
225
|
let offsetInCroppedAudio = 0
|
|
206
226
|
|
|
207
|
-
|
|
208
|
-
|
|
227
|
+
let bestOverlapDuration = -1
|
|
228
|
+
let mappedStartTime = -1
|
|
229
|
+
let mappedEndTime = -1
|
|
209
230
|
|
|
210
|
-
|
|
231
|
+
for (const uncropEntry of uncropTimeline) {
|
|
232
|
+
const uncropEntryDuration = uncropEntry.endTime - uncropEntry.startTime
|
|
211
233
|
|
|
212
|
-
const
|
|
234
|
+
const overlapStartTime = Math.max(startTimeInCroppedAudio, offsetInCroppedAudio)
|
|
235
|
+
const overlapEndTime = Math.min(endTimeInCroppedAudio, offsetInCroppedAudio + uncropEntryDuration)
|
|
213
236
|
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
237
|
+
const overlapDuration = overlapEndTime - overlapStartTime
|
|
238
|
+
|
|
239
|
+
if (overlapDuration >= 0 && overlapDuration > bestOverlapDuration) {
|
|
240
|
+
bestOverlapDuration = overlapDuration
|
|
241
|
+
|
|
242
|
+
mappedStartTime = uncropEntry.startTime + (overlapStartTime - offsetInCroppedAudio)
|
|
243
|
+
mappedEndTime = uncropEntry.startTime + (overlapEndTime - offsetInCroppedAudio)
|
|
220
244
|
}
|
|
221
245
|
|
|
222
|
-
offsetInCroppedAudio +=
|
|
246
|
+
offsetInCroppedAudio += uncropEntryDuration
|
|
223
247
|
}
|
|
224
248
|
|
|
225
|
-
|
|
249
|
+
if (bestOverlapDuration === -1) {
|
|
250
|
+
throw new Error(`No match found in uncrop timeline (should not occur)`)
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
return {
|
|
254
|
+
mappedStartTime,
|
|
255
|
+
mappedEndTime
|
|
256
|
+
}
|
|
226
257
|
}
|
|
227
258
|
|
|
228
259
|
export interface VADResult {
|
|
@@ -13,7 +13,7 @@ import { getRawAudioDuration, RawAudio, sliceRawAudio } from '../audio/AudioUtil
|
|
|
13
13
|
import { readFile } from '../utilities/FileSystem.js'
|
|
14
14
|
import path from 'path'
|
|
15
15
|
import type { LanguageDetectionResults } from '../api/API.js'
|
|
16
|
-
import { getShortLanguageCode, languageCodeToName } from '../utilities/Locale.js'
|
|
16
|
+
import { formatLanguageCodeWithName, getShortLanguageCode, languageCodeToName } from '../utilities/Locale.js'
|
|
17
17
|
import { loadPackage } from '../utilities/PackageManager.js'
|
|
18
18
|
import chalk from 'chalk'
|
|
19
19
|
import { XorShift32RNG } from '../utilities/RandomGenerator.js'
|
|
@@ -41,7 +41,7 @@ export async function recognize(
|
|
|
41
41
|
sourceLanguage = getShortLanguageCode(sourceLanguage)
|
|
42
42
|
|
|
43
43
|
if (!(sourceLanguage in languageIdLookup)) {
|
|
44
|
-
throw new Error(`The language ${
|
|
44
|
+
throw new Error(`The language ${formatLanguageCodeWithName(sourceLanguage)} is not supported by the Whisper engine.`)
|
|
45
45
|
}
|
|
46
46
|
|
|
47
47
|
if (isEnglishOnlyModel(modelName) && sourceLanguage != 'en') {
|
|
@@ -93,7 +93,7 @@ export async function align(
|
|
|
93
93
|
sourceLanguage = getShortLanguageCode(sourceLanguage)
|
|
94
94
|
|
|
95
95
|
if (!(sourceLanguage in languageIdLookup)) {
|
|
96
|
-
throw new Error(`The language ${
|
|
96
|
+
throw new Error(`The language ${formatLanguageCodeWithName(sourceLanguage)} is not supported by the Whisper engine.`)
|
|
97
97
|
}
|
|
98
98
|
|
|
99
99
|
if (isEnglishOnlyModel(modelName) && sourceLanguage != 'en') {
|
|
@@ -134,7 +134,7 @@ export async function alignEnglishTranslation(
|
|
|
134
134
|
sourceLanguage = getShortLanguageCode(sourceLanguage)
|
|
135
135
|
|
|
136
136
|
if (!(sourceLanguage in languageIdLookup)) {
|
|
137
|
-
throw new Error(`The source language ${
|
|
137
|
+
throw new Error(`The source language ${formatLanguageCodeWithName(sourceLanguage)} is not supported by the Whisper engine.`)
|
|
138
138
|
}
|
|
139
139
|
|
|
140
140
|
if (isEnglishOnlyModel(modelName)) {
|
package/src/utilities/Locale.ts
CHANGED
|
@@ -22,6 +22,39 @@ export function formatLanguageCodeWithName(languageCode: string, styleId: 1 | 2
|
|
|
22
22
|
}
|
|
23
23
|
}
|
|
24
24
|
|
|
25
|
+
export async function normalizeIdentifierToLangaugeCode(langIdentifier: string) {
|
|
26
|
+
const result = await parseLangIdentifier(langIdentifier)
|
|
27
|
+
|
|
28
|
+
return result.Name
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export async function normalizeIdentifierToShortLanguageCode(langIdentifier: string) {
|
|
32
|
+
const result = await parseLangIdentifier(langIdentifier)
|
|
33
|
+
|
|
34
|
+
return result.TwoLetterISOLanguageName
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export async function parseLangIdentifier(langIdentifier: string) {
|
|
38
|
+
if (!langIdentifier) {
|
|
39
|
+
return emptyLangInfoEntry
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
await loadLangInfoEntriesIfNeeded()
|
|
43
|
+
|
|
44
|
+
langIdentifier = langIdentifier.trim().toLowerCase()
|
|
45
|
+
|
|
46
|
+
for (const entry of langInfoEntries) {
|
|
47
|
+
if (langIdentifier === entry.NameLowerCase ||
|
|
48
|
+
langIdentifier === entry.ThreeLetterISOLanguageName ||
|
|
49
|
+
langIdentifier === entry.EnglishNameLowerCase) {
|
|
50
|
+
|
|
51
|
+
return entry
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
throw new Error(`Couldn't parse language identifier '${langIdentifier}'.`)
|
|
56
|
+
}
|
|
57
|
+
|
|
25
58
|
export function getShortLanguageCode(langCode: string) {
|
|
26
59
|
const dashIndex = langCode.indexOf('-')
|
|
27
60
|
|
|
@@ -48,7 +81,7 @@ export function normalizeLanguageCode(langCode: string) {
|
|
|
48
81
|
|
|
49
82
|
const isoToLcidLookup = new Map<string, number>()
|
|
50
83
|
const lcidToIsoLookup = new Map<number, string[]>()
|
|
51
|
-
|
|
84
|
+
let langInfoEntries: LangInfoEntry[] = []
|
|
52
85
|
|
|
53
86
|
export async function isoToLcidLanguageCode(iso: string) {
|
|
54
87
|
await loadLcidLookupIfNeeded()
|
|
@@ -63,19 +96,13 @@ export async function lcidToIsoLanguageCode(lcid: number) {
|
|
|
63
96
|
}
|
|
64
97
|
|
|
65
98
|
async function loadLcidLookupIfNeeded() {
|
|
66
|
-
|
|
67
|
-
return lcidEntries
|
|
68
|
-
}
|
|
69
|
-
|
|
70
|
-
const lcidLookup: LCIDLookup = await readAndParseJsonFile(resolveToModuleRootDir('data/tables/lcid-table.json'))
|
|
71
|
-
|
|
72
|
-
for (const isoName in lcidLookup) {
|
|
73
|
-
const lcidEntry = lcidLookup[isoName]
|
|
74
|
-
lcidEntries.push(lcidEntry)
|
|
99
|
+
await loadLangInfoEntriesIfNeeded()
|
|
75
100
|
|
|
101
|
+
for (const lcidEntry of langInfoEntries) {
|
|
102
|
+
const name = lcidEntry.Name
|
|
76
103
|
const lcidValue = lcidEntry.LCID
|
|
77
104
|
|
|
78
|
-
isoToLcidLookup.set(
|
|
105
|
+
isoToLcidLookup.set(name, lcidValue)
|
|
79
106
|
|
|
80
107
|
let entry = lcidToIsoLookup.get(lcidValue)
|
|
81
108
|
|
|
@@ -84,10 +111,25 @@ async function loadLcidLookupIfNeeded() {
|
|
|
84
111
|
lcidToIsoLookup.set(lcidValue, entry)
|
|
85
112
|
}
|
|
86
113
|
|
|
87
|
-
entry.push(
|
|
114
|
+
entry.push(name)
|
|
88
115
|
}
|
|
89
116
|
|
|
90
|
-
return
|
|
117
|
+
return langInfoEntries
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
async function loadLangInfoEntriesIfNeeded() {
|
|
121
|
+
if (langInfoEntries.length > 0) {
|
|
122
|
+
return
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const entries = await readAndParseJsonFile(resolveToModuleRootDir('data/tables/lcid-table.json')) as LangInfoEntry[]
|
|
126
|
+
|
|
127
|
+
for (const entry of entries) {
|
|
128
|
+
entry.NameLowerCase = entry.Name.toLowerCase()
|
|
129
|
+
entry.EnglishNameLowerCase = entry.EnglishName.toLowerCase()
|
|
130
|
+
|
|
131
|
+
langInfoEntries.push(entry)
|
|
132
|
+
}
|
|
91
133
|
}
|
|
92
134
|
|
|
93
135
|
export function getDefaultDialectForLanguageCodeIfPossible(langCode: string) {
|
|
@@ -107,14 +149,34 @@ export const defaultDialectForLanguageCode: { [lang: string]: string } = {
|
|
|
107
149
|
'nl': 'nl-NL'
|
|
108
150
|
}
|
|
109
151
|
|
|
110
|
-
|
|
152
|
+
export interface LangInfoEntry {
|
|
153
|
+
LCID: number
|
|
154
|
+
|
|
155
|
+
Name: string
|
|
156
|
+
NameLowerCase: string
|
|
157
|
+
|
|
158
|
+
TwoLetterISOLanguageName: string
|
|
159
|
+
ThreeLetterISOLanguageName: string
|
|
160
|
+
ThreeLetterWindowsLanguageName: string
|
|
161
|
+
|
|
162
|
+
EnglishName: string
|
|
163
|
+
EnglishNameLowerCase: string
|
|
164
|
+
|
|
165
|
+
ANSICodePage: string
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
export const emptyLangInfoEntry: LangInfoEntry = {
|
|
169
|
+
LCID: -1,
|
|
170
|
+
|
|
171
|
+
Name: '',
|
|
172
|
+
NameLowerCase: '',
|
|
173
|
+
|
|
174
|
+
TwoLetterISOLanguageName: '',
|
|
175
|
+
ThreeLetterISOLanguageName: '',
|
|
176
|
+
ThreeLetterWindowsLanguageName: '',
|
|
177
|
+
|
|
178
|
+
EnglishName: 'Empty',
|
|
179
|
+
EnglishNameLowerCase: 'empty',
|
|
111
180
|
|
|
112
|
-
|
|
113
|
-
'LCID': number
|
|
114
|
-
'Name': string
|
|
115
|
-
'TwoLetterISOLanguageName': string,
|
|
116
|
-
'ThreeLetterISOLanguageName': string,
|
|
117
|
-
'ThreeLetterWindowsLanguageName': string,
|
|
118
|
-
'EnglishName': string
|
|
119
|
-
'ANSICodePage': string
|
|
181
|
+
ANSICodePage: ''
|
|
120
182
|
}
|