echogarden 1.0.4 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -23
- package/data/schemas/options.json +177 -36
- package/dist/alignment/SpeechAlignment.d.ts +1 -1
- package/dist/alignment/SpeechAlignment.js +1 -1
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +1 -0
- package/dist/api/API.js +1 -0
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +1 -0
- package/dist/api/Alignment.d.ts +3 -3
- package/dist/api/Alignment.js +5 -10
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetection.d.ts +5 -7
- package/dist/api/LanguageDetection.js +3 -2
- package/dist/api/LanguageDetection.js.map +1 -1
- package/dist/api/Recognition.d.ts +4 -5
- package/dist/api/Recognition.js +5 -8
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/SourceSeparation.d.ts +2 -0
- package/dist/api/SourceSeparation.js +4 -2
- package/dist/api/SourceSeparation.js.map +1 -1
- package/dist/api/Synthesis.d.ts +3 -1
- package/dist/api/Synthesis.js +9 -10
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.d.ts +1 -1
- package/dist/api/Translation.js +4 -8
- package/dist/api/Translation.js.map +1 -1
- package/dist/api/TranslationAlignment.d.ts +31 -0
- package/dist/api/TranslationAlignment.js +121 -0
- package/dist/api/TranslationAlignment.js.map +1 -0
- package/dist/api/VoiceActivityDetection.d.ts +5 -1
- package/dist/api/VoiceActivityDetection.js +38 -2
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/audio/AudioPlayer.js +6 -1
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/cli/CLI.js +85 -0
- package/dist/cli/CLI.js.map +1 -1
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/math/MedianFilter.d.ts +5 -0
- package/dist/math/MedianFilter.js +102 -0
- package/dist/math/MedianFilter.js.map +1 -0
- package/dist/math/VectorMath.d.ts +0 -2
- package/dist/math/VectorMath.js +1 -25
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/recognition/OpenAICloudSTT.d.ts +1 -1
- package/dist/recognition/OpenAICloudSTT.js.map +1 -1
- package/dist/recognition/SileroSTT.d.ts +22 -1
- package/dist/recognition/SileroSTT.js +122 -95
- package/dist/recognition/SileroSTT.js.map +1 -1
- package/dist/recognition/WhisperCppSTT.js +1 -1
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.d.ts +52 -19
- package/dist/recognition/WhisperSTT.js +645 -494
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Server.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +5 -3
- package/dist/source-separation/MDXNetSourceSeparation.js +26 -19
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +15 -9
- package/dist/speech-language-detection/SileroLanguageDetection.js +23 -16
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/synthesis/EspeakTTS.js +4 -0
- package/dist/synthesis/EspeakTTS.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.js.map +1 -1
- package/dist/synthesis/VitsTTS.d.ts +8 -6
- package/dist/synthesis/VitsTTS.js +36 -31
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/tests/Test.js.map +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +14 -0
- package/dist/utilities/OnnxUtilities.js +43 -0
- package/dist/utilities/OnnxUtilities.js.map +1 -0
- package/dist/utilities/Utilities.d.ts +4 -8
- package/dist/utilities/Utilities.js +35 -58
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/voice-activity-detection/SileroVAD.d.ts +5 -3
- package/dist/voice-activity-detection/SileroVAD.js +9 -11
- package/dist/voice-activity-detection/SileroVAD.js.map +1 -1
- package/docs/API.md +54 -34
- package/docs/CLI.md +25 -13
- package/docs/Contributing.md +4 -2
- package/docs/Engines.md +43 -32
- package/docs/Licenses.md +3 -4
- package/docs/Options.md +47 -11
- package/docs/Releases.md +4 -0
- package/docs/Server.md +8 -6
- package/docs/Tasklist.md +39 -52
- package/docs/Technical.md +1 -1
- package/package.json +8 -12
- package/src/alignment/SpeechAlignment.ts +1 -1
- package/src/api/API.ts +1 -0
- package/src/api/APIOptions.ts +1 -0
- package/src/api/Alignment.ts +10 -14
- package/src/api/LanguageDetection.ts +14 -10
- package/src/api/Recognition.ts +17 -10
- package/src/api/SourceSeparation.ts +7 -2
- package/src/api/Synthesis.ts +26 -11
- package/src/api/Translation.ts +14 -8
- package/src/api/TranslationAlignment.ts +213 -0
- package/src/api/VoiceActivityDetection.ts +66 -3
- package/src/audio/AudioPlayer.ts +6 -2
- package/src/cli/CLI.ts +121 -2
- package/src/dsp/FFT.ts +3 -0
- package/src/math/MedianFilter.ts +124 -0
- package/src/math/VectorMath.ts +1 -36
- package/src/recognition/OpenAICloudSTT.ts +27 -27
- package/src/recognition/SileroSTT.ts +149 -102
- package/src/recognition/WhisperCppSTT.ts +1 -1
- package/src/recognition/WhisperSTT.ts +961 -684
- package/src/server/Server.ts +1 -1
- package/src/source-separation/MDXNetSourceSeparation.ts +35 -19
- package/src/speech-language-detection/SileroLanguageDetection.ts +53 -33
- package/src/synthesis/EspeakTTS.ts +8 -0
- package/src/synthesis/GoogleCloudTTS.ts +12 -1
- package/src/synthesis/VitsTTS.ts +57 -46
- package/src/tests/Test.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +68 -0
- package/src/utilities/Utilities.ts +38 -66
- package/src/voice-activity-detection/SileroVAD.ts +15 -15
- package/dist/utilities/NdArrayUtilities.d.ts +0 -3
- package/dist/utilities/NdArrayUtilities.js +0 -23
- package/dist/utilities/NdArrayUtilities.js.map +0 -1
- package/src/utilities/NdArrayUtilities.ts +0 -31
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "echogarden",
|
|
3
|
-
"version": "1.0
|
|
4
|
-
"description": "An
|
|
3
|
+
"version": "1.1.0",
|
|
4
|
+
"description": "An easy-to-use speech toolset. Includes tools for synthesis, recognition, alignment, speech translation, language detection, source separation and more.",
|
|
5
5
|
"author": "Rotem Dan",
|
|
6
6
|
"license": "GPL-3.0",
|
|
7
7
|
"keywords": [
|
|
@@ -14,7 +14,8 @@
|
|
|
14
14
|
"forced alignment",
|
|
15
15
|
"speech translation",
|
|
16
16
|
"language identification",
|
|
17
|
-
"language detection"
|
|
17
|
+
"language detection",
|
|
18
|
+
"source separation"
|
|
18
19
|
],
|
|
19
20
|
"repository": {
|
|
20
21
|
"type": "git",
|
|
@@ -73,7 +74,6 @@
|
|
|
73
74
|
"cldr-segmentation": "^2.2.0",
|
|
74
75
|
"command-exists": "^1.2.9",
|
|
75
76
|
"compromise": "^14.13.0",
|
|
76
|
-
"compromise-dates": "^3.5.0",
|
|
77
77
|
"fs-extra": "^11.2.0",
|
|
78
78
|
"gaxios": "^6.5.0",
|
|
79
79
|
"graceful-fs": "^4.2.11",
|
|
@@ -87,10 +87,8 @@
|
|
|
87
87
|
"microsoft-cognitiveservices-speech-sdk": "^1.36.0",
|
|
88
88
|
"moving-median": "^1.0.0",
|
|
89
89
|
"msgpack-lite": "^0.1.26",
|
|
90
|
-
"
|
|
91
|
-
"
|
|
92
|
-
"onnxruntime-node": "^1.17.0",
|
|
93
|
-
"openai": "^4.37.0",
|
|
90
|
+
"onnxruntime-node": "^1.17.3",
|
|
91
|
+
"openai": "^4.38.2",
|
|
94
92
|
"sam-js": "^0.2.1",
|
|
95
93
|
"strip-ansi": "^7.1.0",
|
|
96
94
|
"tar": "^7.0.1",
|
|
@@ -121,13 +119,11 @@
|
|
|
121
119
|
"@types/graceful-fs": "^4.1.9",
|
|
122
120
|
"@types/jsdom": "^21.1.6",
|
|
123
121
|
"@types/msgpack-lite": "^0.1.11",
|
|
124
|
-
"@types/ndarray": "^1.0.14",
|
|
125
|
-
"@types/ndarray-ops": "^1.2.7",
|
|
126
122
|
"@types/node": "^20.12.7",
|
|
127
123
|
"@types/recursive-readdir": "^2.2.4",
|
|
128
|
-
"@types/tar": "^6.1.
|
|
124
|
+
"@types/tar": "^6.1.13",
|
|
129
125
|
"@types/ws": "^8.5.10",
|
|
130
|
-
"ts-json-schema-generator": "^2.0
|
|
126
|
+
"ts-json-schema-generator": "^2.1.0",
|
|
131
127
|
"typescript": "^5.4.5"
|
|
132
128
|
}
|
|
133
129
|
}
|
|
@@ -452,7 +452,7 @@ export async function alignPhoneTimelines(
|
|
|
452
452
|
return alignedWordTimeline
|
|
453
453
|
}
|
|
454
454
|
|
|
455
|
-
export async function createAlignmentReferenceUsingEspeakForFragments(fragments: string[], espeakOptions: EspeakOptions
|
|
455
|
+
export async function createAlignmentReferenceUsingEspeakForFragments(fragments: string[], espeakOptions: EspeakOptions) {
|
|
456
456
|
const progressLogger = new Logger()
|
|
457
457
|
|
|
458
458
|
progressLogger.start("Load espeak module")
|
package/src/api/API.ts
CHANGED
|
@@ -6,6 +6,7 @@ export * from './Synthesis.js'
|
|
|
6
6
|
export * from './Recognition.js'
|
|
7
7
|
export * from './Alignment.js'
|
|
8
8
|
export * from './Translation.js'
|
|
9
|
+
export * from './TranslationAlignment.js'
|
|
9
10
|
export * from './LanguageDetection.js'
|
|
10
11
|
export * from './VoiceActivityDetection.js'
|
|
11
12
|
export * from './Denoising.js'
|
package/src/api/APIOptions.ts
CHANGED
|
@@ -6,6 +6,7 @@ export interface APIOptions {
|
|
|
6
6
|
SynthesisOptions: API.SynthesisOptions
|
|
7
7
|
RecognitionOptions: API.RecognitionOptions
|
|
8
8
|
AlignmentOptions: API.AlignmentOptions
|
|
9
|
+
TranslationAlignmentOptions: API.TranslationAlignmentOptions
|
|
9
10
|
SpeechTranslationOptions: API.SpeechTranslationOptions
|
|
10
11
|
SpeechLanguageDetectionOptions: API.SpeechLanguageDetectionOptions
|
|
11
12
|
TextLanguageDetectionOptions: API.TextLanguageDetectionOptions
|
package/src/api/Alignment.ts
CHANGED
|
@@ -7,11 +7,11 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
7
7
|
import * as API from './API.js'
|
|
8
8
|
import { Timeline, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
9
9
|
import { formatLanguageCodeWithName, getDefaultDialectForLanguageCodeIfPossible, getShortLanguageCode, normalizeLanguageCode } from '../utilities/Locale.js'
|
|
10
|
-
import { type
|
|
10
|
+
import { type WhisperAlignmentOptions } from '../recognition/WhisperSTT.js'
|
|
11
11
|
import chalk from 'chalk'
|
|
12
12
|
import { DtwGranularity, createAlignmentReferenceUsingEspeak } from '../alignment/SpeechAlignment.js'
|
|
13
|
-
import { SubtitlesConfig
|
|
14
|
-
import { EspeakOptions, defaultEspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
13
|
+
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
14
|
+
import { type EspeakOptions, defaultEspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
15
15
|
import { isWord } from '../nlp/Segmentation.js'
|
|
16
16
|
|
|
17
17
|
const log = logToStderr
|
|
@@ -194,19 +194,15 @@ export async function align(input: AudioSourceParam, transcript: string, options
|
|
|
194
194
|
case 'whisper': {
|
|
195
195
|
const WhisperSTT = await import('../recognition/WhisperSTT.js')
|
|
196
196
|
|
|
197
|
-
const
|
|
197
|
+
const whisperAlignmnentOptions = options.whisper!
|
|
198
198
|
|
|
199
199
|
const shortLanguageCode = getShortLanguageCode(language)
|
|
200
200
|
|
|
201
|
-
const { modelName, modelDir } = await WhisperSTT.loadPackagesAndGetPaths(
|
|
202
|
-
|
|
203
|
-
if (getRawAudioDuration(sourceRawAudio) > 30) {
|
|
204
|
-
throw new Error('Whisper based alignment currently only supports audio inputs that are 30s or less')
|
|
205
|
-
}
|
|
201
|
+
const { modelName, modelDir } = await WhisperSTT.loadPackagesAndGetPaths(whisperAlignmnentOptions.model, shortLanguageCode)
|
|
206
202
|
|
|
207
203
|
logger.end()
|
|
208
204
|
|
|
209
|
-
mappedTimeline = await WhisperSTT.align(sourceRawAudio, transcript, modelName, modelDir, shortLanguageCode)
|
|
205
|
+
mappedTimeline = await WhisperSTT.align(sourceRawAudio, transcript, modelName, modelDir, shortLanguageCode, whisperAlignmnentOptions)
|
|
210
206
|
|
|
211
207
|
break
|
|
212
208
|
}
|
|
@@ -315,7 +311,7 @@ export interface AlignmentOptions {
|
|
|
315
311
|
|
|
316
312
|
sourceSeparation?: API.SourceSeparationOptions
|
|
317
313
|
|
|
318
|
-
whisper?:
|
|
314
|
+
whisper?: WhisperAlignmentOptions
|
|
319
315
|
}
|
|
320
316
|
|
|
321
317
|
export const defaultAlignmentOptions: AlignmentOptions = {
|
|
@@ -337,7 +333,8 @@ export const defaultAlignmentOptions: AlignmentOptions = {
|
|
|
337
333
|
whitespace: 'collapse'
|
|
338
334
|
},
|
|
339
335
|
|
|
340
|
-
subtitles:
|
|
336
|
+
subtitles: {
|
|
337
|
+
},
|
|
341
338
|
|
|
342
339
|
dtw: {
|
|
343
340
|
granularity: 'auto',
|
|
@@ -354,7 +351,6 @@ export const defaultAlignmentOptions: AlignmentOptions = {
|
|
|
354
351
|
autoPromptParts: false,
|
|
355
352
|
suppressRepetition: true,
|
|
356
353
|
decodeTimestampTokens: true,
|
|
357
|
-
seed: undefined,
|
|
358
354
|
}
|
|
359
355
|
},
|
|
360
356
|
|
|
@@ -385,7 +381,7 @@ export const alignmentEngines: API.EngineMetadata[] = [
|
|
|
385
381
|
{
|
|
386
382
|
id: 'whisper',
|
|
387
383
|
name: 'OpenAI Whisper',
|
|
388
|
-
description: 'Extracts timestamps
|
|
384
|
+
description: 'Extracts timestamps by guiding the Whisper recognition model to recognize the transcript tokens.',
|
|
389
385
|
type: 'local'
|
|
390
386
|
}
|
|
391
387
|
]
|
|
@@ -6,11 +6,13 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
import * as API from './API.js'
|
|
7
7
|
import { logToStderr } from '../utilities/Utilities.js'
|
|
8
8
|
import path from 'path'
|
|
9
|
-
import { type
|
|
9
|
+
import { type WhisperLanguageDetectionOptions } from '../recognition/WhisperSTT.js'
|
|
10
10
|
import { formatLanguageCodeWithName, languageCodeToName } from '../utilities/Locale.js'
|
|
11
11
|
import { loadPackage } from '../utilities/PackageManager.js'
|
|
12
12
|
import chalk from 'chalk'
|
|
13
|
-
import { WhisperCppOptions } from '../recognition/WhisperCppSTT.js'
|
|
13
|
+
import { type WhisperCppOptions } from '../recognition/WhisperCppSTT.js'
|
|
14
|
+
import { type SileroLanguageDetectionOptions } from '../speech-language-detection/SileroLanguageDetection.js'
|
|
15
|
+
import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js'
|
|
14
16
|
|
|
15
17
|
const log = logToStderr
|
|
16
18
|
|
|
@@ -59,12 +61,14 @@ export async function detectSpeechLanguage(input: AudioSourceParam, options: Spe
|
|
|
59
61
|
const modelPath = path.join(modelDir, 'lang_classifier_95.onnx')
|
|
60
62
|
const languageDictionaryPath = path.join(modelDir, 'lang_dict_95.json')
|
|
61
63
|
const languageGroupDictionaryPath = path.join(modelDir, 'lang_group_dict_95.json')
|
|
64
|
+
const onnxExecutionProviders: OnnxExecutionProvider[] = sileroOptions.provider ? [sileroOptions.provider] : []
|
|
62
65
|
|
|
63
66
|
const languageResults = await SileroLanguageDetection.detectLanguage(
|
|
64
67
|
sourceRawAudio,
|
|
65
68
|
modelPath,
|
|
66
69
|
languageDictionaryPath,
|
|
67
|
-
languageGroupDictionaryPath
|
|
70
|
+
languageGroupDictionaryPath,
|
|
71
|
+
onnxExecutionProviders)
|
|
68
72
|
|
|
69
73
|
detectedLanguageProbabilities = languageResults
|
|
70
74
|
|
|
@@ -80,7 +84,11 @@ export async function detectSpeechLanguage(input: AudioSourceParam, options: Spe
|
|
|
80
84
|
|
|
81
85
|
logger.end()
|
|
82
86
|
|
|
83
|
-
detectedLanguageProbabilities = await WhisperSTT.detectLanguage(
|
|
87
|
+
detectedLanguageProbabilities = await WhisperSTT.detectLanguage(
|
|
88
|
+
sourceRawAudio,
|
|
89
|
+
modelName,
|
|
90
|
+
modelDir,
|
|
91
|
+
whisperOptions)
|
|
84
92
|
|
|
85
93
|
break
|
|
86
94
|
}
|
|
@@ -198,13 +206,9 @@ export interface SpeechLanguageDetectionOptions {
|
|
|
198
206
|
|
|
199
207
|
crop?: boolean
|
|
200
208
|
|
|
201
|
-
silero?:
|
|
202
|
-
}
|
|
209
|
+
silero?: SileroLanguageDetectionOptions
|
|
203
210
|
|
|
204
|
-
whisper?:
|
|
205
|
-
model?: WhisperModelName
|
|
206
|
-
temperature?: number
|
|
207
|
-
}
|
|
211
|
+
whisper?: WhisperLanguageDetectionOptions
|
|
208
212
|
|
|
209
213
|
whisperCpp?: WhisperCppOptions
|
|
210
214
|
|
package/src/api/Recognition.ts
CHANGED
|
@@ -6,13 +6,16 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
6
6
|
|
|
7
7
|
import * as API from './API.js'
|
|
8
8
|
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
9
|
-
import { type WhisperOptions } from '../recognition/WhisperSTT.js'
|
|
10
9
|
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeLanguageCode } from '../utilities/Locale.js'
|
|
11
10
|
import { loadPackage } from '../utilities/PackageManager.js'
|
|
12
11
|
import chalk from 'chalk'
|
|
13
|
-
|
|
14
|
-
import {
|
|
12
|
+
|
|
13
|
+
import { type WhisperOptions } from '../recognition/WhisperSTT.js'
|
|
14
|
+
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
15
|
+
import { type OpenAICloudSTTOptions } from '../recognition/OpenAICloudSTT.js'
|
|
15
16
|
import { type WhisperCppOptions } from '../recognition/WhisperCppSTT.js'
|
|
17
|
+
import { type SileroRecognitionOptions } from '../recognition/SileroSTT.js'
|
|
18
|
+
import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js'
|
|
16
19
|
|
|
17
20
|
const log = logToStderr
|
|
18
21
|
|
|
@@ -172,9 +175,14 @@ export async function recognize(input: AudioSourceParam, options: RecognitionOpt
|
|
|
172
175
|
modelPath = await loadPackage(packageName)
|
|
173
176
|
}
|
|
174
177
|
|
|
178
|
+
const onnxExecutionProviders: OnnxExecutionProvider[] = sileroOptions.provider ? [sileroOptions.provider] : []
|
|
179
|
+
|
|
175
180
|
logger.end();
|
|
176
181
|
|
|
177
|
-
({ transcript, timeline } = await SileroSTT.recognize(
|
|
182
|
+
({ transcript, timeline } = await SileroSTT.recognize(
|
|
183
|
+
sourceRawAudio,
|
|
184
|
+
modelPath,
|
|
185
|
+
onnxExecutionProviders))
|
|
178
186
|
|
|
179
187
|
break
|
|
180
188
|
}
|
|
@@ -346,9 +354,7 @@ export interface RecognitionOptions {
|
|
|
346
354
|
modelPath?: string
|
|
347
355
|
}
|
|
348
356
|
|
|
349
|
-
silero?:
|
|
350
|
-
modelPath?: string
|
|
351
|
-
}
|
|
357
|
+
silero?: SileroRecognitionOptions
|
|
352
358
|
|
|
353
359
|
googleCloud?: {
|
|
354
360
|
apiKey?: string
|
|
@@ -389,7 +395,8 @@ export const defaultRecognitionOptions: RecognitionOptions = {
|
|
|
389
395
|
languageDetection: {
|
|
390
396
|
},
|
|
391
397
|
|
|
392
|
-
subtitles:
|
|
398
|
+
subtitles: {
|
|
399
|
+
},
|
|
393
400
|
|
|
394
401
|
vad: {
|
|
395
402
|
engine: 'adaptive-gate'
|
|
@@ -406,7 +413,6 @@ export const defaultRecognitionOptions: RecognitionOptions = {
|
|
|
406
413
|
},
|
|
407
414
|
|
|
408
415
|
silero: {
|
|
409
|
-
modelPath: undefined
|
|
410
416
|
},
|
|
411
417
|
|
|
412
418
|
googleCloud: {
|
|
@@ -428,7 +434,8 @@ export const defaultRecognitionOptions: RecognitionOptions = {
|
|
|
428
434
|
secretAccessKey: undefined,
|
|
429
435
|
},
|
|
430
436
|
|
|
431
|
-
openAICloud:
|
|
437
|
+
openAICloud: {
|
|
438
|
+
},
|
|
432
439
|
}
|
|
433
440
|
|
|
434
441
|
export const recognitionEngines: API.EngineMetadata[] = [
|
|
@@ -6,6 +6,7 @@ import { EngineMetadata } from './Common.js';
|
|
|
6
6
|
import chalk from 'chalk';
|
|
7
7
|
import { readdir } from '../utilities/FileSystem.js';
|
|
8
8
|
import path from 'node:path';
|
|
9
|
+
import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js';
|
|
9
10
|
|
|
10
11
|
export async function isolate(input: AudioSourceParam, options: SourceSeparationOptions): Promise<SourceSeparationResult> {
|
|
11
12
|
const logger = new Logger()
|
|
@@ -26,6 +27,8 @@ export async function isolate(input: AudioSourceParam, options: SourceSeparation
|
|
|
26
27
|
|
|
27
28
|
const mdxNetOptions = options.mdxNet!
|
|
28
29
|
|
|
30
|
+
const executionProviders: OnnxExecutionProvider[] = mdxNetOptions.executionProvider ? [mdxNetOptions.executionProvider] : []
|
|
31
|
+
|
|
29
32
|
const packageDir = await loadPackage(`mdxnet-${mdxNetOptions.model!}`)
|
|
30
33
|
const modelFilename = (await readdir(packageDir)).filter(name => name.endsWith('onnx'))[0]
|
|
31
34
|
|
|
@@ -39,7 +42,7 @@ export async function isolate(input: AudioSourceParam, options: SourceSeparation
|
|
|
39
42
|
|
|
40
43
|
const audioStereo44100 = await ensureRawAudio(inputRawAudio, 44100, 2)
|
|
41
44
|
|
|
42
|
-
isolatedRawAudio = await MDXNetSourceSeparation.isolate(audioStereo44100, modelPath)
|
|
45
|
+
isolatedRawAudio = await MDXNetSourceSeparation.isolate(audioStereo44100, modelPath, executionProviders)
|
|
43
46
|
|
|
44
47
|
logger.end()
|
|
45
48
|
|
|
@@ -72,6 +75,7 @@ export interface SourceSeparationOptions {
|
|
|
72
75
|
|
|
73
76
|
mdxNet?: {
|
|
74
77
|
model?: string
|
|
78
|
+
executionProvider?: OnnxExecutionProvider
|
|
75
79
|
}
|
|
76
80
|
}
|
|
77
81
|
|
|
@@ -79,7 +83,8 @@ export const defaultSourceSeparationOptions: SourceSeparationOptions = {
|
|
|
79
83
|
engine: 'mdx-net',
|
|
80
84
|
|
|
81
85
|
mdxNet: {
|
|
82
|
-
model: 'UVR_MDXNET_1_9703'
|
|
86
|
+
model: 'UVR_MDXNET_1_9703',
|
|
87
|
+
executionProvider: undefined,
|
|
83
88
|
}
|
|
84
89
|
}
|
|
85
90
|
|
package/src/api/Synthesis.ts
CHANGED
|
@@ -4,7 +4,7 @@ import { deepClone, extendDeep } from '../utilities/ObjectUtilities.js'
|
|
|
4
4
|
|
|
5
5
|
import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
|
|
6
6
|
|
|
7
|
-
import { clip, convertHtmlToText, sha256AsHex, simplifyPunctuationCharacters, stringifyAndFormatJson, logToStderr, yieldToEventLoop,
|
|
7
|
+
import { clip, convertHtmlToText, sha256AsHex, simplifyPunctuationCharacters, stringifyAndFormatJson, logToStderr, yieldToEventLoop, runOperationWithRetries } from '../utilities/Utilities.js'
|
|
8
8
|
import { RawAudio, attenuateIfClipping, concatAudioSegments, downmixToMono, encodeRawAudioToWave, getSamplePeakDecibels, getEmptyRawAudio, getRawAudioDuration, normalizeAudioLevel, trimAudioEnd, trimAudioStart } from '../audio/AudioUtilities.js'
|
|
9
9
|
import { Logger } from '../utilities/Logger.js'
|
|
10
10
|
|
|
@@ -20,10 +20,11 @@ import { loadPackage } from '../utilities/PackageManager.js'
|
|
|
20
20
|
import { EngineMetadata, appName } from './Common.js'
|
|
21
21
|
import { shouldCancelCurrentTask } from '../server/Worker.js'
|
|
22
22
|
import chalk from 'chalk'
|
|
23
|
-
import { SubtitlesConfig
|
|
23
|
+
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
24
24
|
import { type EspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
25
|
-
import { type OpenAICloudTTSOptions
|
|
26
|
-
import { type ElevenlabsTTSOptions
|
|
25
|
+
import { type OpenAICloudTTSOptions } from '../synthesis/OpenAICloudTTS.js'
|
|
26
|
+
import { type ElevenlabsTTSOptions } from '../synthesis/ElevenlabsTTS.js'
|
|
27
|
+
import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js'
|
|
27
28
|
|
|
28
29
|
const log = logToStderr
|
|
29
30
|
|
|
@@ -358,9 +359,9 @@ async function synthesizeSegment(text: string, options: SynthesisOptions) {
|
|
|
358
359
|
|
|
359
360
|
const lengthScale = 1 / speed
|
|
360
361
|
|
|
361
|
-
const
|
|
362
|
+
const vitsOptions = options.vits!
|
|
362
363
|
|
|
363
|
-
const speakerId =
|
|
364
|
+
const speakerId = vitsOptions.speakerId
|
|
364
365
|
|
|
365
366
|
if (speakerId != undefined) {
|
|
366
367
|
if (selectedVoice.speakerCount == undefined) {
|
|
@@ -368,7 +369,7 @@ async function synthesizeSegment(text: string, options: SynthesisOptions) {
|
|
|
368
369
|
throw new Error('Selected VITS model has only one speaker. Speaker ID must be 0 if specified.')
|
|
369
370
|
}
|
|
370
371
|
} else if (speakerId < 0 || speakerId >= selectedVoice.speakerCount) {
|
|
371
|
-
throw new Error(`Selected VITS model has ${selectedVoice.speakerCount}
|
|
372
|
+
throw new Error(`Selected VITS model has ${selectedVoice.speakerCount} speaker IDs. Speaker ID should be in the range ${0} to ${selectedVoice.speakerCount - 1}`)
|
|
372
373
|
}
|
|
373
374
|
}
|
|
374
375
|
|
|
@@ -376,9 +377,18 @@ async function synthesizeSegment(text: string, options: SynthesisOptions) {
|
|
|
376
377
|
|
|
377
378
|
const modelPath = voicePackagePath!
|
|
378
379
|
|
|
380
|
+
const onnxExecutionProviders: OnnxExecutionProvider[] = vitsOptions.provider ? [vitsOptions.provider] : []
|
|
381
|
+
|
|
379
382
|
logger.end()
|
|
380
383
|
|
|
381
|
-
const { rawAudio, timeline: outTimeline } = await vitsTTS.synthesizeSentence(
|
|
384
|
+
const { rawAudio, timeline: outTimeline } = await vitsTTS.synthesizeSentence(
|
|
385
|
+
text,
|
|
386
|
+
voice,
|
|
387
|
+
modelPath,
|
|
388
|
+
lengthScale,
|
|
389
|
+
speakerId ?? 0,
|
|
390
|
+
lexicons,
|
|
391
|
+
onnxExecutionProviders)
|
|
382
392
|
|
|
383
393
|
synthesizedAudio = rawAudio
|
|
384
394
|
timeline = outTimeline
|
|
@@ -1024,6 +1034,7 @@ export interface SynthesisOptions {
|
|
|
1024
1034
|
|
|
1025
1035
|
vits?: {
|
|
1026
1036
|
speakerId?: number
|
|
1037
|
+
provider?: OnnxExecutionProvider
|
|
1027
1038
|
}
|
|
1028
1039
|
|
|
1029
1040
|
pico?: {
|
|
@@ -1153,10 +1164,12 @@ export const defaultSynthesisOptions: SynthesisOptions = {
|
|
|
1153
1164
|
|
|
1154
1165
|
languageDetection: undefined,
|
|
1155
1166
|
|
|
1156
|
-
subtitles:
|
|
1167
|
+
subtitles: {
|
|
1168
|
+
},
|
|
1157
1169
|
|
|
1158
1170
|
vits: {
|
|
1159
1171
|
speakerId: undefined,
|
|
1172
|
+
provider: undefined,
|
|
1160
1173
|
},
|
|
1161
1174
|
|
|
1162
1175
|
pico: {
|
|
@@ -1216,9 +1229,11 @@ export const defaultSynthesisOptions: SynthesisOptions = {
|
|
|
1216
1229
|
lexiconNames: undefined,
|
|
1217
1230
|
},
|
|
1218
1231
|
|
|
1219
|
-
openAICloud:
|
|
1232
|
+
openAICloud: {
|
|
1233
|
+
},
|
|
1220
1234
|
|
|
1221
|
-
elevenlabs:
|
|
1235
|
+
elevenlabs: {
|
|
1236
|
+
},
|
|
1222
1237
|
|
|
1223
1238
|
googleTranslate: {
|
|
1224
1239
|
tld: 'us'
|
package/src/api/Translation.ts
CHANGED
|
@@ -5,16 +5,16 @@ import { AudioSourceParam, RawAudio, ensureRawAudio, normalizeAudioLevel, trimAu
|
|
|
5
5
|
import { Logger } from '../utilities/Logger.js'
|
|
6
6
|
|
|
7
7
|
import { Timeline, addWordTextOffsetsToTimeline, wordTimelineToSegmentSentenceTimeline } from '../utilities/Timeline.js'
|
|
8
|
-
import {
|
|
8
|
+
import { type WhisperOptions } from '../recognition/WhisperSTT.js'
|
|
9
9
|
import { formatLanguageCodeWithName, getShortLanguageCode, normalizeLanguageCode } from '../utilities/Locale.js'
|
|
10
10
|
import { EngineMetadata } from './Common.js'
|
|
11
11
|
import { type SpeechLanguageDetectionOptions, detectSpeechLanguage } from './API.js'
|
|
12
12
|
import chalk from 'chalk'
|
|
13
|
-
import { SubtitlesConfig
|
|
13
|
+
import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
14
14
|
|
|
15
15
|
import * as API from './API.js'
|
|
16
|
-
import { type OpenAICloudSTTOptions
|
|
17
|
-
import { type WhisperCppOptions
|
|
16
|
+
import { type OpenAICloudSTTOptions } from '../recognition/OpenAICloudSTT.js'
|
|
17
|
+
import { type WhisperCppOptions } from '../recognition/WhisperCppSTT.js'
|
|
18
18
|
|
|
19
19
|
const log = logToStderr
|
|
20
20
|
|
|
@@ -263,15 +263,21 @@ export const defaultSpeechTranslationOptions: SpeechTranslationOptions = {
|
|
|
263
263
|
|
|
264
264
|
languageDetection: undefined,
|
|
265
265
|
|
|
266
|
-
subtitles:
|
|
266
|
+
subtitles: {
|
|
267
|
+
},
|
|
267
268
|
|
|
268
269
|
vad: {
|
|
269
270
|
engine: 'adaptive-gate'
|
|
270
271
|
},
|
|
271
272
|
|
|
272
|
-
whisper:
|
|
273
|
-
|
|
274
|
-
|
|
273
|
+
whisper: {
|
|
274
|
+
},
|
|
275
|
+
|
|
276
|
+
whisperCpp: {
|
|
277
|
+
},
|
|
278
|
+
|
|
279
|
+
openAICloud: {
|
|
280
|
+
},
|
|
275
281
|
}
|
|
276
282
|
|
|
277
283
|
export const speechTranslationEngines: EngineMetadata[] = [
|