echogarden 2.2.0 → 2.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/data/lexicons/heteronyms.en.json +1 -1
- package/data/schemas/options.json +27 -2
- package/dist/api/Synthesis.d.ts +5 -2
- package/dist/api/Synthesis.js +36 -7
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/recognition/DeepgramSTT.d.ts +3 -2
- package/dist/recognition/DeepgramSTT.js +54 -27
- package/dist/recognition/DeepgramSTT.js.map +1 -1
- package/dist/synthesis/DeepgramTTS.d.ts +15 -0
- package/dist/synthesis/DeepgramTTS.js +127 -0
- package/dist/synthesis/DeepgramTTS.js.map +1 -0
- package/dist/synthesis/ElevenLabsTTS.d.ts +8 -2
- package/dist/synthesis/ElevenLabsTTS.js +91 -33
- package/dist/synthesis/ElevenLabsTTS.js.map +1 -1
- package/dist/synthesis/EspeakTTS.js +4 -1
- package/dist/synthesis/EspeakTTS.js.map +1 -1
- package/dist/utilities/ObjectUtilities.js +6 -1
- package/dist/utilities/ObjectUtilities.js.map +1 -1
- package/dist/utilities/WasmMemoryManager.d.ts +1 -1
- package/docs/Engines.md +2 -1
- package/docs/Options.md +13 -2
- package/docs/Tasklist.md +0 -1
- package/package.json +2 -2
- package/src/api/Synthesis.ts +58 -10
- package/src/recognition/DeepgramSTT.ts +66 -29
- package/src/synthesis/DeepgramTTS.ts +149 -0
- package/src/synthesis/ElevenLabsTTS.ts +109 -32
- package/src/synthesis/EspeakTTS.ts +5 -1
- package/src/utilities/ObjectUtilities.ts +30 -13
package/src/api/Synthesis.ts
CHANGED
|
@@ -22,6 +22,7 @@ import { type SubtitlesConfig } from '../subtitles/Subtitles.js'
|
|
|
22
22
|
import { type EspeakOptions } from '../synthesis/EspeakTTS.js'
|
|
23
23
|
import { type OpenAICloudTTSOptions } from '../synthesis/OpenAICloudTTS.js'
|
|
24
24
|
import { type ElevenLabsTTSOptions } from '../synthesis/ElevenLabsTTS.js'
|
|
25
|
+
import { type DeepgramTTSOptions } from '../synthesis/DeepgramTTS.js'
|
|
25
26
|
import { OnnxExecutionProvider } from '../utilities/OnnxUtilities.js'
|
|
26
27
|
import { simplifyPunctuationCharacters } from '../nlp/TextNormalizer.js'
|
|
27
28
|
import { convertHtmlToText } from '../utilities/StringUtilities.js'
|
|
@@ -808,23 +809,50 @@ async function synthesizeSegment(text: string, options: SynthesisOptions) {
|
|
|
808
809
|
|
|
809
810
|
case 'elevenlabs': {
|
|
810
811
|
if (inputIsSSML) {
|
|
811
|
-
throw new Error(`The
|
|
812
|
+
throw new Error(`The ElevenLabs engine doesn't support SSML inputs`)
|
|
812
813
|
}
|
|
813
814
|
|
|
814
815
|
const ElevenLabsTTS = await import('../synthesis/ElevenLabsTTS.js')
|
|
815
816
|
|
|
816
|
-
const engineOptions = options.
|
|
817
|
+
const engineOptions = options.elevenLabs!
|
|
817
818
|
|
|
818
819
|
if (!engineOptions.apiKey) {
|
|
819
820
|
throw new Error(`No ElevenLabs API key provided`)
|
|
820
821
|
}
|
|
821
822
|
|
|
822
823
|
const voiceId = (selectedVoice as any)['elevenLabsVoiceId']
|
|
823
|
-
const modelId = (selectedVoice as any)['elevenLabsModelId']
|
|
824
824
|
|
|
825
825
|
logger.end()
|
|
826
826
|
|
|
827
|
-
const { rawAudio } = await ElevenLabsTTS.synthesize(text, voiceId,
|
|
827
|
+
const { rawAudio, timeline: outTimeline } = await ElevenLabsTTS.synthesize(text, voiceId, language, engineOptions)
|
|
828
|
+
|
|
829
|
+
synthesizedAudio = rawAudio
|
|
830
|
+
timeline = outTimeline
|
|
831
|
+
|
|
832
|
+
shouldPostprocessSpeed = true
|
|
833
|
+
shouldPostprocessPitch = true
|
|
834
|
+
|
|
835
|
+
break
|
|
836
|
+
}
|
|
837
|
+
|
|
838
|
+
case 'deepgram': {
|
|
839
|
+
if (inputIsSSML) {
|
|
840
|
+
throw new Error(`The Deepgram engine doesn't support SSML inputs`)
|
|
841
|
+
}
|
|
842
|
+
|
|
843
|
+
const DeepgramTTS = await import('../synthesis/DeepgramTTS.js')
|
|
844
|
+
|
|
845
|
+
const engineOptions = options.deepgram!
|
|
846
|
+
|
|
847
|
+
if (!engineOptions.apiKey) {
|
|
848
|
+
throw new Error(`No Deepgram API key provided`)
|
|
849
|
+
}
|
|
850
|
+
|
|
851
|
+
const modelId = selectedVoice.deepgramModelId
|
|
852
|
+
|
|
853
|
+
logger.end()
|
|
854
|
+
|
|
855
|
+
const { rawAudio } = await DeepgramTTS.synthesize(text, modelId, engineOptions)
|
|
828
856
|
|
|
829
857
|
synthesizedAudio = rawAudio
|
|
830
858
|
|
|
@@ -1071,8 +1099,8 @@ export type SynthesisEngine =
|
|
|
1071
1099
|
'vits' | 'kokoro' | 'pico' | 'flite' | 'gnuspeech' |
|
|
1072
1100
|
'espeak' | 'sam' | 'sapi' | 'msspeech' | 'coqui-server' |
|
|
1073
1101
|
'google-cloud' | 'microsoft-azure' | 'amazon-polly' |
|
|
1074
|
-
'openai-cloud' | 'elevenlabs' | '
|
|
1075
|
-
'microsoft-edge' | 'streamlabs-polly'
|
|
1102
|
+
'openai-cloud' | 'elevenlabs' | 'deepgram' |
|
|
1103
|
+
'google-translate' | 'microsoft-edge' | 'streamlabs-polly'
|
|
1076
1104
|
|
|
1077
1105
|
export type TimePitchShiftingMethod = 'sonic' | 'rubberband'
|
|
1078
1106
|
|
|
@@ -1198,7 +1226,9 @@ export interface SynthesisOptions {
|
|
|
1198
1226
|
|
|
1199
1227
|
openAICloud?: OpenAICloudTTSOptions
|
|
1200
1228
|
|
|
1201
|
-
|
|
1229
|
+
elevenLabs?: ElevenLabsTTSOptions,
|
|
1230
|
+
|
|
1231
|
+
deepgram?: DeepgramTTSOptions
|
|
1202
1232
|
|
|
1203
1233
|
googleTranslate?: {
|
|
1204
1234
|
tld?: string
|
|
@@ -1340,7 +1370,10 @@ export const defaultSynthesisOptions: SynthesisOptions = {
|
|
|
1340
1370
|
openAICloud: {
|
|
1341
1371
|
},
|
|
1342
1372
|
|
|
1343
|
-
|
|
1373
|
+
elevenLabs: {
|
|
1374
|
+
},
|
|
1375
|
+
|
|
1376
|
+
deepgram: {
|
|
1344
1377
|
},
|
|
1345
1378
|
|
|
1346
1379
|
googleTranslate: {
|
|
@@ -1595,7 +1628,7 @@ export async function requestVoiceList(options: VoiceListRequestOptions): Promis
|
|
|
1595
1628
|
case 'elevenlabs': {
|
|
1596
1629
|
const ElevenLabsTTS = await import('../synthesis/ElevenLabsTTS.js')
|
|
1597
1630
|
|
|
1598
|
-
const engineOptions = options.
|
|
1631
|
+
const engineOptions = options.elevenLabs!
|
|
1599
1632
|
|
|
1600
1633
|
const apiKey = engineOptions.apiKey
|
|
1601
1634
|
|
|
@@ -1608,6 +1641,14 @@ export async function requestVoiceList(options: VoiceListRequestOptions): Promis
|
|
|
1608
1641
|
break
|
|
1609
1642
|
}
|
|
1610
1643
|
|
|
1644
|
+
case 'deepgram': {
|
|
1645
|
+
const DeepgramTTS = await import('../synthesis/DeepgramTTS.js')
|
|
1646
|
+
|
|
1647
|
+
voiceList = DeepgramTTS.voiceList
|
|
1648
|
+
|
|
1649
|
+
break
|
|
1650
|
+
}
|
|
1651
|
+
|
|
1611
1652
|
case 'google-translate': {
|
|
1612
1653
|
const GoogleTranslateTTS = await import('../synthesis/GoogleTranslateTTS.js')
|
|
1613
1654
|
|
|
@@ -1800,6 +1841,7 @@ export interface SynthesisVoice {
|
|
|
1800
1841
|
gender: VoiceGender
|
|
1801
1842
|
speakerCount?: number
|
|
1802
1843
|
packageName?: string
|
|
1844
|
+
[key: string]: any
|
|
1803
1845
|
}
|
|
1804
1846
|
|
|
1805
1847
|
export type VoiceGender = 'male' | 'female' | 'unknown'
|
|
@@ -1891,7 +1933,13 @@ export const synthesisEngines: EngineMetadata[] = [
|
|
|
1891
1933
|
},
|
|
1892
1934
|
{
|
|
1893
1935
|
id: 'elevenlabs',
|
|
1894
|
-
name: '
|
|
1936
|
+
name: 'ElevenLabs',
|
|
1937
|
+
description: 'A generative AI text-to-speech cloud service.',
|
|
1938
|
+
type: 'cloud'
|
|
1939
|
+
},
|
|
1940
|
+
{
|
|
1941
|
+
id: 'deepgram',
|
|
1942
|
+
name: 'Deepgram',
|
|
1895
1943
|
description: 'A generative AI text-to-speech cloud service.',
|
|
1896
1944
|
type: 'cloud'
|
|
1897
1945
|
},
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { request } from 'gaxios'
|
|
1
|
+
import { GaxiosResponse, request } from 'gaxios'
|
|
2
2
|
import { RawAudio } from '../audio/AudioUtilities.js'
|
|
3
3
|
import { Logger } from '../utilities/Logger.js'
|
|
4
4
|
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
@@ -18,9 +18,9 @@ export async function recognize(rawAudio: RawAudio, languageCode: string | undef
|
|
|
18
18
|
|
|
19
19
|
// Prepare API request
|
|
20
20
|
const params: Record<string, string> = {
|
|
21
|
-
model: options.model
|
|
22
|
-
encoding: '
|
|
23
|
-
|
|
21
|
+
model: options.model!,
|
|
22
|
+
encoding: 'opus',
|
|
23
|
+
punctuate: options.punctuate ? 'true' : 'false',
|
|
24
24
|
}
|
|
25
25
|
|
|
26
26
|
// Set language or enable auto-detection
|
|
@@ -31,32 +31,49 @@ export async function recognize(rawAudio: RawAudio, languageCode: string | undef
|
|
|
31
31
|
}
|
|
32
32
|
|
|
33
33
|
// Set audio encoding parameters
|
|
34
|
-
logger.start('Convert audio to
|
|
34
|
+
logger.start('Convert audio to Opus format')
|
|
35
35
|
|
|
36
36
|
const audioData = await FFMpegTranscoder.encodeFromChannels(
|
|
37
37
|
rawAudio,
|
|
38
|
-
FFMpegTranscoder.getDefaultFFMpegOptionsForSpeech('
|
|
38
|
+
FFMpegTranscoder.getDefaultFFMpegOptionsForSpeech('opus')
|
|
39
39
|
)
|
|
40
40
|
|
|
41
41
|
logger.start('Send request to Deepgram API')
|
|
42
42
|
|
|
43
|
-
|
|
44
|
-
method: 'POST',
|
|
43
|
+
let response: GaxiosResponse<any>
|
|
45
44
|
|
|
46
|
-
|
|
45
|
+
try {
|
|
46
|
+
response = await request<any>({
|
|
47
|
+
method: 'POST',
|
|
47
48
|
|
|
48
|
-
|
|
49
|
+
url: 'https://api.deepgram.com/v1/listen',
|
|
49
50
|
|
|
50
|
-
|
|
51
|
-
'Authorization': `Token ${options.apiKey}`,
|
|
52
|
-
'Content-Type': 'audio/flac',
|
|
53
|
-
'Accept': 'application/json'
|
|
54
|
-
},
|
|
51
|
+
params,
|
|
55
52
|
|
|
56
|
-
|
|
53
|
+
headers: {
|
|
54
|
+
'Authorization': `Token ${options.apiKey}`,
|
|
55
|
+
'Content-Type': 'audio/ogg',
|
|
56
|
+
'Accept': 'application/json'
|
|
57
|
+
},
|
|
57
58
|
|
|
58
|
-
|
|
59
|
-
|
|
59
|
+
body: audioData,
|
|
60
|
+
|
|
61
|
+
responseType: 'json',
|
|
62
|
+
})
|
|
63
|
+
} catch (e: any) {
|
|
64
|
+
const response = e.response
|
|
65
|
+
|
|
66
|
+
if (response) {
|
|
67
|
+
logger.log(`Request failed with status code ${response.status}`)
|
|
68
|
+
|
|
69
|
+
if (response.data) {
|
|
70
|
+
logger.log(`Server responded with:`)
|
|
71
|
+
logger.log(response.data)
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
throw e
|
|
76
|
+
}
|
|
60
77
|
|
|
61
78
|
const deepgramResponse: DeepgramResponse = response.data
|
|
62
79
|
|
|
@@ -65,19 +82,37 @@ export async function recognize(rawAudio: RawAudio, languageCode: string | undef
|
|
|
65
82
|
// Extract transcript and create timeline
|
|
66
83
|
const transcript = firstAlternative?.transcript || ''
|
|
67
84
|
|
|
68
|
-
let timeline: Timeline = []
|
|
69
|
-
|
|
70
85
|
// Extract word-level timing information if available
|
|
71
86
|
const words = firstAlternative?.words || []
|
|
72
87
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
88
|
+
const timeline = words.map((wordEntry: DeepgramWordEntry) => ({
|
|
89
|
+
type: 'word',
|
|
90
|
+
text: wordEntry.word,
|
|
91
|
+
startTime: wordEntry.start,
|
|
92
|
+
endTime: wordEntry.end,
|
|
93
|
+
confidence: wordEntry.confidence,
|
|
94
|
+
} as TimelineEntry))
|
|
95
|
+
|
|
96
|
+
// If `punctuate` is set to `true`, modify the text of all words to match their exact case in the transcript.
|
|
97
|
+
// This is required, otherwise it would later fail deriving word offsets.
|
|
98
|
+
if (options.punctuate) {
|
|
99
|
+
const lowerCaseTranscript = transcript.toLocaleLowerCase()
|
|
100
|
+
|
|
101
|
+
let readOffset = 0
|
|
102
|
+
|
|
103
|
+
for (const wordEntry of timeline) {
|
|
104
|
+
const wordEntryTextLowercase = wordEntry.text.toLocaleLowerCase()
|
|
105
|
+
|
|
106
|
+
const matchPosition = lowerCaseTranscript.indexOf(wordEntryTextLowercase, readOffset)
|
|
107
|
+
|
|
108
|
+
if (matchPosition === -1) {
|
|
109
|
+
throw new Error(`Couldn't match the word '${wordEntry.text}' in the lowercase transcript`)
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
wordEntry.text = transcript.substring(matchPosition, matchPosition + wordEntryTextLowercase.length)
|
|
113
|
+
|
|
114
|
+
readOffset = matchPosition + wordEntry.text.length
|
|
115
|
+
}
|
|
81
116
|
}
|
|
82
117
|
|
|
83
118
|
logger.end()
|
|
@@ -88,11 +123,13 @@ export async function recognize(rawAudio: RawAudio, languageCode: string | undef
|
|
|
88
123
|
export interface DeepgramSTTOptions {
|
|
89
124
|
apiKey?: string
|
|
90
125
|
model?: string
|
|
126
|
+
punctuate?: boolean
|
|
91
127
|
}
|
|
92
128
|
|
|
93
129
|
export const defaultDeepgramSTTOptions: DeepgramSTTOptions = {
|
|
94
130
|
apiKey: undefined,
|
|
95
|
-
model: 'nova-2'
|
|
131
|
+
model: 'nova-2',
|
|
132
|
+
punctuate: true,
|
|
96
133
|
}
|
|
97
134
|
|
|
98
135
|
interface DeepgramWordEntry {
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
import { GaxiosResponse, request } from 'gaxios'
|
|
2
|
+
import { SynthesisVoice } from '../api/API.js'
|
|
3
|
+
import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
|
|
4
|
+
import { Logger } from '../utilities/Logger.js'
|
|
5
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
6
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
7
|
+
|
|
8
|
+
const log = logToStderr
|
|
9
|
+
|
|
10
|
+
export async function synthesize(text: string, modelId: string, options: DeepgramTTSOptions) {
|
|
11
|
+
const logger = new Logger()
|
|
12
|
+
logger.start('Request synthesis from Deepgram')
|
|
13
|
+
|
|
14
|
+
options = extendDeep(defaultDeepgramTTSOptions, options)
|
|
15
|
+
|
|
16
|
+
let response: GaxiosResponse<any>
|
|
17
|
+
|
|
18
|
+
try {
|
|
19
|
+
response = await request<any>({
|
|
20
|
+
url: `https://api.deepgram.com/v1/speak`,
|
|
21
|
+
|
|
22
|
+
params: {
|
|
23
|
+
model: modelId,
|
|
24
|
+
encoding: 'mp3',
|
|
25
|
+
bit_rate: 48000,
|
|
26
|
+
},
|
|
27
|
+
|
|
28
|
+
method: 'POST',
|
|
29
|
+
|
|
30
|
+
headers: {
|
|
31
|
+
'Content-Type': 'application/json',
|
|
32
|
+
'Authorization': `Token ${options.apiKey}`,
|
|
33
|
+
},
|
|
34
|
+
|
|
35
|
+
data: {
|
|
36
|
+
text,
|
|
37
|
+
},
|
|
38
|
+
|
|
39
|
+
responseType: 'arraybuffer'
|
|
40
|
+
})
|
|
41
|
+
} catch (e: any) {
|
|
42
|
+
const response = e.response
|
|
43
|
+
|
|
44
|
+
if (response) {
|
|
45
|
+
logger.log(`Request failed with status code ${response.status}`)
|
|
46
|
+
|
|
47
|
+
if (response.data) {
|
|
48
|
+
logger.log(`Server responded with:`)
|
|
49
|
+
logger.log(response.data)
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
throw e
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
logger.start('Decode synthesized audio')
|
|
57
|
+
const rawAudio = await FFMpegTranscoder.decodeToChannels(new Uint8Array(response.data))
|
|
58
|
+
|
|
59
|
+
logger.end()
|
|
60
|
+
|
|
61
|
+
return { rawAudio }
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export async function getVoiceList() {
|
|
65
|
+
return voiceList
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export interface DeepgramTTSOptions {
|
|
69
|
+
apiKey?: string
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export const defaultDeepgramTTSOptions = {
|
|
73
|
+
apiKey: undefined,
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export const voiceList: SynthesisVoice[] = [
|
|
77
|
+
{
|
|
78
|
+
name: 'Asteria',
|
|
79
|
+
deepgramModelId: 'aura-asteria-en',
|
|
80
|
+
languages: ['en-US', 'en'],
|
|
81
|
+
gender: 'female',
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
name: 'Luna',
|
|
85
|
+
deepgramModelId: 'aura-luna-en',
|
|
86
|
+
languages: ['en-US', 'en'],
|
|
87
|
+
gender: 'female',
|
|
88
|
+
},
|
|
89
|
+
{
|
|
90
|
+
name: 'Stella',
|
|
91
|
+
deepgramModelId: 'aura-stella-en',
|
|
92
|
+
languages: ['en-US', 'en'],
|
|
93
|
+
gender: 'female',
|
|
94
|
+
},
|
|
95
|
+
{
|
|
96
|
+
name: 'Athena',
|
|
97
|
+
deepgramModelId: 'aura-athena-en',
|
|
98
|
+
languages: ['en-GB', 'en'],
|
|
99
|
+
gender: 'female',
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
name: 'Hera',
|
|
103
|
+
deepgramModelId: 'aura-hera-en',
|
|
104
|
+
languages: ['en-US', 'en'],
|
|
105
|
+
gender: 'female',
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
name: 'Orion',
|
|
109
|
+
deepgramModelId: 'aura-orion-en',
|
|
110
|
+
languages: ['en-US', 'en'],
|
|
111
|
+
gender: 'male',
|
|
112
|
+
},
|
|
113
|
+
{
|
|
114
|
+
name: 'Arcas',
|
|
115
|
+
deepgramModelId: 'aura-arcas-en',
|
|
116
|
+
languages: ['en-US', 'en'],
|
|
117
|
+
gender: 'male',
|
|
118
|
+
},
|
|
119
|
+
{
|
|
120
|
+
name: 'Perseus',
|
|
121
|
+
deepgramModelId: 'aura-perseus-en',
|
|
122
|
+
languages: ['en-US', 'en'],
|
|
123
|
+
gender: 'male',
|
|
124
|
+
},
|
|
125
|
+
{
|
|
126
|
+
name: 'Angus',
|
|
127
|
+
deepgramModelId: 'aura-angus-en',
|
|
128
|
+
languages: ['en-US', 'en'],
|
|
129
|
+
gender: 'male',
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
name: 'Orpheus',
|
|
133
|
+
deepgramModelId: 'aura-orpheus-en',
|
|
134
|
+
languages: ['en-US', 'en'],
|
|
135
|
+
gender: 'male',
|
|
136
|
+
},
|
|
137
|
+
{
|
|
138
|
+
name: 'Helios',
|
|
139
|
+
deepgramModelId: 'aura-helios-en',
|
|
140
|
+
languages: ['en-US', 'en'],
|
|
141
|
+
gender: 'male',
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
name: 'Zeus',
|
|
145
|
+
deepgramModelId: 'aura-zeus-en',
|
|
146
|
+
languages: ['en-US', 'en'],
|
|
147
|
+
gender: 'male',
|
|
148
|
+
},
|
|
149
|
+
]
|
|
@@ -4,10 +4,13 @@ import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
|
|
|
4
4
|
import { Logger } from '../utilities/Logger.js'
|
|
5
5
|
import { logToStderr } from '../utilities/Utilities.js'
|
|
6
6
|
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
7
|
+
import { decodeBase64 } from '../encodings/Base64.js'
|
|
8
|
+
import { isWordOrSymbolWord, splitToWords } from '../nlp/Segmentation.js'
|
|
9
|
+
import { Timeline } from '../utilities/Timeline.js'
|
|
7
10
|
|
|
8
11
|
const log = logToStderr
|
|
9
12
|
|
|
10
|
-
export async function synthesize(text: string, voiceId: string,
|
|
13
|
+
export async function synthesize(text: string, voiceId: string, language: string, options: ElevenLabsTTSOptions) {
|
|
11
14
|
const logger = new Logger()
|
|
12
15
|
logger.start('Request synthesis from ElevenLabs')
|
|
13
16
|
|
|
@@ -17,7 +20,7 @@ export async function synthesize(text: string, voiceId: string, modelId: string,
|
|
|
17
20
|
|
|
18
21
|
try {
|
|
19
22
|
response = await request<any>({
|
|
20
|
-
url: `https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`,
|
|
23
|
+
url: `https://api.elevenlabs.io/v1/text-to-speech/${voiceId}/with-timestamps`,
|
|
21
24
|
|
|
22
25
|
method: 'POST',
|
|
23
26
|
|
|
@@ -26,20 +29,26 @@ export async function synthesize(text: string, voiceId: string, modelId: string,
|
|
|
26
29
|
'xi-api-key': options.apiKey,
|
|
27
30
|
},
|
|
28
31
|
|
|
32
|
+
params: {
|
|
33
|
+
output_format: 'mp3_44100_64',
|
|
34
|
+
},
|
|
35
|
+
|
|
29
36
|
data: {
|
|
30
37
|
text,
|
|
31
38
|
|
|
32
|
-
model_id: modelId,
|
|
39
|
+
model_id: options.modelId,
|
|
33
40
|
|
|
34
41
|
voice_setting: {
|
|
35
42
|
stability: options.stability,
|
|
36
43
|
similarity_boost: options.similarityBoost,
|
|
37
44
|
style: options.style,
|
|
38
|
-
use_speaker_boost: options.useSpeakerBoost
|
|
39
|
-
}
|
|
45
|
+
use_speaker_boost: options.useSpeakerBoost,
|
|
46
|
+
},
|
|
47
|
+
|
|
48
|
+
seed: options.seed
|
|
40
49
|
},
|
|
41
50
|
|
|
42
|
-
responseType: '
|
|
51
|
+
responseType: 'json'
|
|
43
52
|
})
|
|
44
53
|
} catch (e: any) {
|
|
45
54
|
const response = e.response
|
|
@@ -57,11 +66,43 @@ export async function synthesize(text: string, voiceId: string, modelId: string,
|
|
|
57
66
|
}
|
|
58
67
|
|
|
59
68
|
logger.start('Decode synthesized audio')
|
|
60
|
-
const
|
|
69
|
+
const audioData = decodeBase64(response.data.audio_base64)
|
|
70
|
+
const rawAudio = await FFMpegTranscoder.decodeToChannels(audioData)
|
|
71
|
+
|
|
72
|
+
let timeline: Timeline | undefined
|
|
73
|
+
|
|
74
|
+
const characters: string[] = response.data.alignment?.characters
|
|
75
|
+
const characterStartTimes: number[] = response.data.alignment?.character_start_times_seconds
|
|
76
|
+
const characterEndTimes: number[] = response.data.alignment?.character_end_times_seconds
|
|
77
|
+
|
|
78
|
+
if (characters && characterStartTimes && characterEndTimes) {
|
|
79
|
+
logger.start('Create timeline from returned character timings')
|
|
80
|
+
|
|
81
|
+
const referenceText = characters.join('')
|
|
82
|
+
const words = (await splitToWords(referenceText, language)).filter(w => isWordOrSymbolWord(w))
|
|
83
|
+
|
|
84
|
+
timeline = []
|
|
85
|
+
|
|
86
|
+
let offset = 0
|
|
87
|
+
|
|
88
|
+
for (const word of words) {
|
|
89
|
+
const wordStartIndex = referenceText.indexOf(word, offset)
|
|
90
|
+
const wordEndIndex = wordStartIndex + word.length
|
|
91
|
+
|
|
92
|
+
timeline.push({
|
|
93
|
+
type: 'word',
|
|
94
|
+
text: word,
|
|
95
|
+
startTime: characterStartTimes[wordStartIndex],
|
|
96
|
+
endTime: characterEndTimes[wordEndIndex] ?? characterEndTimes[wordEndIndex - 1]
|
|
97
|
+
})
|
|
98
|
+
|
|
99
|
+
offset = wordEndIndex
|
|
100
|
+
}
|
|
101
|
+
}
|
|
61
102
|
|
|
62
103
|
logger.end()
|
|
63
104
|
|
|
64
|
-
return { rawAudio }
|
|
105
|
+
return { rawAudio, timeline }
|
|
65
106
|
}
|
|
66
107
|
|
|
67
108
|
export async function getVoiceList(apiKey: string) {
|
|
@@ -78,40 +119,36 @@ export async function getVoiceList(apiKey: string) {
|
|
|
78
119
|
responseType: 'json'
|
|
79
120
|
})
|
|
80
121
|
|
|
81
|
-
const
|
|
122
|
+
const elevenLabsVoices: any[] = response.data.voices
|
|
82
123
|
|
|
83
|
-
const voices: SynthesisVoice[] =
|
|
84
|
-
const
|
|
85
|
-
const accent: string | undefined = elevenlabsVoice?.labels?.accent
|
|
86
|
-
const gender: VoiceGender = elevenlabsVoice?.labels?.gender ?? 'unknown'
|
|
124
|
+
const voices: SynthesisVoice[] = elevenLabsVoices.map(elevenLabsVoice => {
|
|
125
|
+
const gender: VoiceGender = elevenLabsVoice?.labels?.gender ?? 'unknown'
|
|
87
126
|
|
|
88
127
|
const supportedLanguages: string[] = []
|
|
89
128
|
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
if (modelId.includes('multilingual')) {
|
|
103
|
-
supportedLanguages.push('en', ...supporteMultilingualLanguages)
|
|
129
|
+
let accent: string | undefined = elevenLabsVoice?.labels?.accent
|
|
130
|
+
accent = accent?.toLowerCase() ?? ''
|
|
131
|
+
|
|
132
|
+
if (accent.startsWith('american')) {
|
|
133
|
+
supportedLanguages.push('en-US')
|
|
134
|
+
} else if (accent.startsWith('british')) {
|
|
135
|
+
supportedLanguages.push('en-GB')
|
|
136
|
+
} else if (accent === 'irish') {
|
|
137
|
+
supportedLanguages.push('en-IE')
|
|
138
|
+
} else if (accent == 'australian') {
|
|
139
|
+
supportedLanguages.push('en-AU')
|
|
104
140
|
} else {
|
|
105
141
|
supportedLanguages.push('en')
|
|
106
142
|
}
|
|
107
143
|
|
|
144
|
+
supportedLanguages.push(...supportedLanguagesInMultilingualModels)
|
|
145
|
+
|
|
108
146
|
return {
|
|
109
|
-
name:
|
|
147
|
+
name: elevenLabsVoice.name,
|
|
110
148
|
languages: supportedLanguages,
|
|
111
149
|
gender,
|
|
112
150
|
|
|
113
|
-
elevenLabsVoiceId:
|
|
114
|
-
elevenLabsModelId: modelId
|
|
151
|
+
elevenLabsVoiceId: elevenLabsVoice.voice_id,
|
|
115
152
|
}
|
|
116
153
|
})
|
|
117
154
|
|
|
@@ -120,18 +157,58 @@ export async function getVoiceList(apiKey: string) {
|
|
|
120
157
|
|
|
121
158
|
export interface ElevenLabsTTSOptions {
|
|
122
159
|
apiKey?: string
|
|
160
|
+
modelId?: string
|
|
161
|
+
|
|
123
162
|
stability?: number
|
|
124
163
|
similarityBoost?: number
|
|
125
164
|
style?: number
|
|
126
165
|
useSpeakerBoost?: boolean
|
|
166
|
+
|
|
167
|
+
seed?: number
|
|
127
168
|
}
|
|
128
169
|
|
|
129
170
|
export const defaultElevenLabsTTSOptions = {
|
|
130
171
|
apiKey: undefined,
|
|
172
|
+
modelId: 'eleven_multilingual_v2',
|
|
173
|
+
|
|
131
174
|
stability: 0.5,
|
|
132
175
|
similarityBoost: 0.5,
|
|
133
176
|
style: 0,
|
|
134
|
-
useSpeakerBoost: true
|
|
177
|
+
useSpeakerBoost: true,
|
|
178
|
+
|
|
179
|
+
seed: undefined,
|
|
135
180
|
}
|
|
136
181
|
|
|
137
|
-
export const
|
|
182
|
+
export const supportedLanguagesInMultilingualModels = [
|
|
183
|
+
'ja',
|
|
184
|
+
'zh',
|
|
185
|
+
'de',
|
|
186
|
+
'hi',
|
|
187
|
+
'fr',
|
|
188
|
+
'ko',
|
|
189
|
+
'pt',
|
|
190
|
+
'it',
|
|
191
|
+
'es',
|
|
192
|
+
'id',
|
|
193
|
+
'nl',
|
|
194
|
+
'tr',
|
|
195
|
+
'fil',
|
|
196
|
+
'pl',
|
|
197
|
+
'sv',
|
|
198
|
+
'bg',
|
|
199
|
+
'ro',
|
|
200
|
+
'ar',
|
|
201
|
+
'cs',
|
|
202
|
+
'el',
|
|
203
|
+
'fi',
|
|
204
|
+
'hr',
|
|
205
|
+
'ms',
|
|
206
|
+
'sk',
|
|
207
|
+
'da',
|
|
208
|
+
'ta',
|
|
209
|
+
'uk',
|
|
210
|
+
'ru',
|
|
211
|
+
'hu',
|
|
212
|
+
'no',
|
|
213
|
+
'vi',
|
|
214
|
+
]
|
|
@@ -170,7 +170,11 @@ export async function synthesizeFragments(fragments: string[], espeakOptions: Es
|
|
|
170
170
|
fragment = simplifyPunctuationCharacters(fragment)
|
|
171
171
|
|
|
172
172
|
//fragment = encodeHTMLAngleBrackets(fragment)
|
|
173
|
-
fragment = fragment.replaceAll('<', '_').replaceAll('>', '_')
|
|
173
|
+
fragment = fragment.replaceAll('<', '_').replaceAll('>', '_')
|
|
174
|
+
|
|
175
|
+
if (fragment.split('').every(c => c === ':')) {
|
|
176
|
+
fragment = ','
|
|
177
|
+
}
|
|
174
178
|
|
|
175
179
|
if (espeakOptions.insertSeparators && canInsertSeparators) {
|
|
176
180
|
const separator = ` | `
|