voxy 0.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- voxy-0.0.2/LICENSE +21 -0
- voxy-0.0.2/PKG-INFO +212 -0
- voxy-0.0.2/README.md +201 -0
- voxy-0.0.2/setup.cfg +25 -0
- voxy-0.0.2/setup.py +3 -0
- voxy-0.0.2/voxy/__init__.py +5 -0
- voxy-0.0.2/voxy/base.py +576 -0
- voxy-0.0.2/voxy.egg-info/PKG-INFO +212 -0
- voxy-0.0.2/voxy.egg-info/SOURCES.txt +11 -0
- voxy-0.0.2/voxy.egg-info/dependency_links.txt +1 -0
- voxy-0.0.2/voxy.egg-info/not-zip-safe +1 -0
- voxy-0.0.2/voxy.egg-info/top_level.txt +1 -0
voxy-0.0.2/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) [year] [fullname]
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
voxy-0.0.2/PKG-INFO
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
Metadata-Version: 2.2
|
|
2
|
+
Name: voxy
|
|
3
|
+
Version: 0.0.2
|
|
4
|
+
Summary: Facade for voice cloning and speech synthesis
|
|
5
|
+
Home-page: https://github.com/thorwhalen/voxy
|
|
6
|
+
Author: Thor Whalen
|
|
7
|
+
License: mit
|
|
8
|
+
Platform: any
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
|
|
12
|
+
# voxy
|
|
13
|
+
|
|
14
|
+
Facade for voice cloning and speech synthesis
|
|
15
|
+
|
|
16
|
+
To install: ```pip install voxy```
|
|
17
|
+
|
|
18
|
+
Voxy is a flexible Python module for speech synthesis and voice cloning, with initial support for the Sesame CSM-1B model. It provides a plugin architecture that can be extended to support other models in the future.
|
|
19
|
+
|
|
20
|
+
## Features
|
|
21
|
+
|
|
22
|
+
- Voice cloning from audio samples
|
|
23
|
+
- High-quality speech synthesis
|
|
24
|
+
- Flexible input formats (file paths, bytes, streams, tensors)
|
|
25
|
+
- Audio cleanup utilities
|
|
26
|
+
- Automatic audio transcription (using Whisper)
|
|
27
|
+
- Plugin architecture for different speech models
|
|
28
|
+
|
|
29
|
+
## Installation
|
|
30
|
+
|
|
31
|
+
### Prerequisites
|
|
32
|
+
|
|
33
|
+
- Python 3.10+
|
|
34
|
+
- PyTorch and TorchAudio
|
|
35
|
+
- CUDA-compatible GPU (recommended)
|
|
36
|
+
- FFmpeg for audio processing
|
|
37
|
+
|
|
38
|
+
### Install the CSM Model
|
|
39
|
+
|
|
40
|
+
The intention is to make `voxy` into a plugin-enabled facade, where you can chose your
|
|
41
|
+
own engine (for voice cloning, voice synthesis, etc.).
|
|
42
|
+
But for now, we just support, what seems to be the best open-source model out there
|
|
43
|
+
(at the time of writing this):
|
|
44
|
+
[Sesame AI Lab's](https://www.sesame.com/research/crossing_the_uncanny_valley_of_voice)
|
|
45
|
+
CSM model. It's just that, well, they did an amazing job at the model, but a terrible one
|
|
46
|
+
(so far) for the python interface -- which is what inspired me to develop `voxy`
|
|
47
|
+
in the first place.
|
|
48
|
+
|
|
49
|
+
Follow the instructions in the [CSM repository](https://github.com/SesameAILabs/csm)
|
|
50
|
+
to install the CSM model and its dependencies.
|
|
51
|
+
|
|
52
|
+
## Quick Start
|
|
53
|
+
|
|
54
|
+
### Basic Usage
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from voxy import create_speech_model
|
|
58
|
+
|
|
59
|
+
# Create a speech model
|
|
60
|
+
model = create_speech_model(model_type="csm")
|
|
61
|
+
|
|
62
|
+
# Generate speech with default voice
|
|
63
|
+
audio = model.generate_speech(
|
|
64
|
+
text="Hello, this is a test of the CSM speech model.",
|
|
65
|
+
output_path="output.wav"
|
|
66
|
+
)
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### Voice Cloning
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from voxy import create_speech_model
|
|
73
|
+
|
|
74
|
+
# Create a speech model
|
|
75
|
+
model = create_speech_model(model_type="csm")
|
|
76
|
+
|
|
77
|
+
# Clone a voice from an audio file
|
|
78
|
+
voice_profile = model.clone_voice(
|
|
79
|
+
audio_input="sample_voice.wav",
|
|
80
|
+
transcript="This is a sample of my voice for cloning purposes."
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
# Generate speech with the cloned voice
|
|
84
|
+
audio = model.generate_speech(
|
|
85
|
+
text="This is my cloned voice speaking. Isn't it amazing?",
|
|
86
|
+
voice_profile=voice_profile,
|
|
87
|
+
output_path="cloned_voice.wav"
|
|
88
|
+
)
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
### Automatic Transcription
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
from voxy import create_speech_model
|
|
95
|
+
|
|
96
|
+
# Create a speech model
|
|
97
|
+
model = create_speech_model(model_type="csm")
|
|
98
|
+
|
|
99
|
+
# Clone a voice with automatic transcription
|
|
100
|
+
voice_profile = model.clone_voice(
|
|
101
|
+
audio_input="sample_voice.wav",
|
|
102
|
+
# No transcript provided, will use automatic transcription
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
# Generate speech with the cloned voice
|
|
106
|
+
audio = model.generate_speech(
|
|
107
|
+
text="This voice was cloned using automatic transcription.",
|
|
108
|
+
voice_profile=voice_profile,
|
|
109
|
+
output_path="auto_transcribed_voice.wav"
|
|
110
|
+
)
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Flexible Input Formats
|
|
114
|
+
|
|
115
|
+
The module supports various input formats:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
# From file path
|
|
119
|
+
voice_profile1 = model.clone_voice(
|
|
120
|
+
audio_input="sample_voice.wav",
|
|
121
|
+
transcript="Text transcript."
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
# From bytes
|
|
125
|
+
with open("sample_voice.wav", "rb") as f:
|
|
126
|
+
audio_bytes = f.read()
|
|
127
|
+
voice_profile2 = model.clone_voice(
|
|
128
|
+
audio_input=audio_bytes,
|
|
129
|
+
transcript="Text transcript."
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
# From file object
|
|
133
|
+
with open("sample_voice.wav", "rb") as f:
|
|
134
|
+
voice_profile3 = model.clone_voice(
|
|
135
|
+
audio_input=f,
|
|
136
|
+
transcript="Text transcript."
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
# From tensor
|
|
140
|
+
import torch
|
|
141
|
+
import torchaudio
|
|
142
|
+
audio_tensor, sample_rate = torchaudio.load("sample_voice.wav")
|
|
143
|
+
voice_profile4 = model.clone_voice(
|
|
144
|
+
audio_input=audio_tensor,
|
|
145
|
+
transcript="Text transcript."
|
|
146
|
+
)
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## Configuration
|
|
150
|
+
|
|
151
|
+
You can configure the default device by setting the `DFLT_VOXY_DEVICE` environment variable:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
# Use CUDA
|
|
155
|
+
export DFLT_VOXY_DEVICE=cuda
|
|
156
|
+
|
|
157
|
+
# Use CPU
|
|
158
|
+
export DFLT_VOXY_DEVICE=cpu
|
|
159
|
+
|
|
160
|
+
# Use MPS (Apple Silicon)
|
|
161
|
+
export DFLT_VOXY_DEVICE=mps
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## Advanced Usage
|
|
165
|
+
|
|
166
|
+
### Audio Cleanup
|
|
167
|
+
|
|
168
|
+
The module includes an audio cleanup function that normalizes volume and removes silence:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from voxy import cleanup_audio
|
|
172
|
+
import torchaudio
|
|
173
|
+
|
|
174
|
+
# Load audio
|
|
175
|
+
audio, sample_rate = torchaudio.load("noisy_audio.wav")
|
|
176
|
+
|
|
177
|
+
# Clean up audio
|
|
178
|
+
cleaned_audio = cleanup_audio(
|
|
179
|
+
audio=audio,
|
|
180
|
+
sample_rate=sample_rate,
|
|
181
|
+
normalize=True,
|
|
182
|
+
remove_silence=True,
|
|
183
|
+
silence_threshold=0.02,
|
|
184
|
+
min_silence_duration=0.2
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
# Save cleaned audio
|
|
188
|
+
torchaudio.save("cleaned_audio.wav", cleaned_audio, sample_rate)
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
### Disabling Audio Cleanup
|
|
192
|
+
|
|
193
|
+
You can disable audio cleanup when cloning a voice:
|
|
194
|
+
|
|
195
|
+
```python
|
|
196
|
+
voice_profile = model.clone_voice(
|
|
197
|
+
audio_input="sample_voice.wav",
|
|
198
|
+
transcript="This is a sample of my voice.",
|
|
199
|
+
cleanup_audio_fn=None # Disable audio cleanup
|
|
200
|
+
)
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
### Custom Audio Cleanup
|
|
204
|
+
|
|
205
|
+
You can also provide your own audio cleanup function:
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
def my_custom_cleanup(audio, sample_rate, **kwargs):
|
|
209
|
+
# Custom cleanup logic
|
|
210
|
+
return processed_audio
|
|
211
|
+
|
|
212
|
+
voice_profile = model.
|
voxy-0.0.2/README.md
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# voxy
|
|
2
|
+
|
|
3
|
+
Facade for voice cloning and speech synthesis
|
|
4
|
+
|
|
5
|
+
To install: ```pip install voxy```
|
|
6
|
+
|
|
7
|
+
Voxy is a flexible Python module for speech synthesis and voice cloning, with initial support for the Sesame CSM-1B model. It provides a plugin architecture that can be extended to support other models in the future.
|
|
8
|
+
|
|
9
|
+
## Features
|
|
10
|
+
|
|
11
|
+
- Voice cloning from audio samples
|
|
12
|
+
- High-quality speech synthesis
|
|
13
|
+
- Flexible input formats (file paths, bytes, streams, tensors)
|
|
14
|
+
- Audio cleanup utilities
|
|
15
|
+
- Automatic audio transcription (using Whisper)
|
|
16
|
+
- Plugin architecture for different speech models
|
|
17
|
+
|
|
18
|
+
## Installation
|
|
19
|
+
|
|
20
|
+
### Prerequisites
|
|
21
|
+
|
|
22
|
+
- Python 3.10+
|
|
23
|
+
- PyTorch and TorchAudio
|
|
24
|
+
- CUDA-compatible GPU (recommended)
|
|
25
|
+
- FFmpeg for audio processing
|
|
26
|
+
|
|
27
|
+
### Install the CSM Model
|
|
28
|
+
|
|
29
|
+
The intention is to make `voxy` into a plugin-enabled facade, where you can chose your
|
|
30
|
+
own engine (for voice cloning, voice synthesis, etc.).
|
|
31
|
+
But for now, we just support, what seems to be the best open-source model out there
|
|
32
|
+
(at the time of writing this):
|
|
33
|
+
[Sesame AI Lab's](https://www.sesame.com/research/crossing_the_uncanny_valley_of_voice)
|
|
34
|
+
CSM model. It's just that, well, they did an amazing job at the model, but a terrible one
|
|
35
|
+
(so far) for the python interface -- which is what inspired me to develop `voxy`
|
|
36
|
+
in the first place.
|
|
37
|
+
|
|
38
|
+
Follow the instructions in the [CSM repository](https://github.com/SesameAILabs/csm)
|
|
39
|
+
to install the CSM model and its dependencies.
|
|
40
|
+
|
|
41
|
+
## Quick Start
|
|
42
|
+
|
|
43
|
+
### Basic Usage
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from voxy import create_speech_model
|
|
47
|
+
|
|
48
|
+
# Create a speech model
|
|
49
|
+
model = create_speech_model(model_type="csm")
|
|
50
|
+
|
|
51
|
+
# Generate speech with default voice
|
|
52
|
+
audio = model.generate_speech(
|
|
53
|
+
text="Hello, this is a test of the CSM speech model.",
|
|
54
|
+
output_path="output.wav"
|
|
55
|
+
)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
### Voice Cloning
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from voxy import create_speech_model
|
|
62
|
+
|
|
63
|
+
# Create a speech model
|
|
64
|
+
model = create_speech_model(model_type="csm")
|
|
65
|
+
|
|
66
|
+
# Clone a voice from an audio file
|
|
67
|
+
voice_profile = model.clone_voice(
|
|
68
|
+
audio_input="sample_voice.wav",
|
|
69
|
+
transcript="This is a sample of my voice for cloning purposes."
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
# Generate speech with the cloned voice
|
|
73
|
+
audio = model.generate_speech(
|
|
74
|
+
text="This is my cloned voice speaking. Isn't it amazing?",
|
|
75
|
+
voice_profile=voice_profile,
|
|
76
|
+
output_path="cloned_voice.wav"
|
|
77
|
+
)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### Automatic Transcription
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
from voxy import create_speech_model
|
|
84
|
+
|
|
85
|
+
# Create a speech model
|
|
86
|
+
model = create_speech_model(model_type="csm")
|
|
87
|
+
|
|
88
|
+
# Clone a voice with automatic transcription
|
|
89
|
+
voice_profile = model.clone_voice(
|
|
90
|
+
audio_input="sample_voice.wav",
|
|
91
|
+
# No transcript provided, will use automatic transcription
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
# Generate speech with the cloned voice
|
|
95
|
+
audio = model.generate_speech(
|
|
96
|
+
text="This voice was cloned using automatic transcription.",
|
|
97
|
+
voice_profile=voice_profile,
|
|
98
|
+
output_path="auto_transcribed_voice.wav"
|
|
99
|
+
)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
### Flexible Input Formats
|
|
103
|
+
|
|
104
|
+
The module supports various input formats:
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
# From file path
|
|
108
|
+
voice_profile1 = model.clone_voice(
|
|
109
|
+
audio_input="sample_voice.wav",
|
|
110
|
+
transcript="Text transcript."
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
# From bytes
|
|
114
|
+
with open("sample_voice.wav", "rb") as f:
|
|
115
|
+
audio_bytes = f.read()
|
|
116
|
+
voice_profile2 = model.clone_voice(
|
|
117
|
+
audio_input=audio_bytes,
|
|
118
|
+
transcript="Text transcript."
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
# From file object
|
|
122
|
+
with open("sample_voice.wav", "rb") as f:
|
|
123
|
+
voice_profile3 = model.clone_voice(
|
|
124
|
+
audio_input=f,
|
|
125
|
+
transcript="Text transcript."
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
# From tensor
|
|
129
|
+
import torch
|
|
130
|
+
import torchaudio
|
|
131
|
+
audio_tensor, sample_rate = torchaudio.load("sample_voice.wav")
|
|
132
|
+
voice_profile4 = model.clone_voice(
|
|
133
|
+
audio_input=audio_tensor,
|
|
134
|
+
transcript="Text transcript."
|
|
135
|
+
)
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
## Configuration
|
|
139
|
+
|
|
140
|
+
You can configure the default device by setting the `DFLT_VOXY_DEVICE` environment variable:
|
|
141
|
+
|
|
142
|
+
```bash
|
|
143
|
+
# Use CUDA
|
|
144
|
+
export DFLT_VOXY_DEVICE=cuda
|
|
145
|
+
|
|
146
|
+
# Use CPU
|
|
147
|
+
export DFLT_VOXY_DEVICE=cpu
|
|
148
|
+
|
|
149
|
+
# Use MPS (Apple Silicon)
|
|
150
|
+
export DFLT_VOXY_DEVICE=mps
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
## Advanced Usage
|
|
154
|
+
|
|
155
|
+
### Audio Cleanup
|
|
156
|
+
|
|
157
|
+
The module includes an audio cleanup function that normalizes volume and removes silence:
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
from voxy import cleanup_audio
|
|
161
|
+
import torchaudio
|
|
162
|
+
|
|
163
|
+
# Load audio
|
|
164
|
+
audio, sample_rate = torchaudio.load("noisy_audio.wav")
|
|
165
|
+
|
|
166
|
+
# Clean up audio
|
|
167
|
+
cleaned_audio = cleanup_audio(
|
|
168
|
+
audio=audio,
|
|
169
|
+
sample_rate=sample_rate,
|
|
170
|
+
normalize=True,
|
|
171
|
+
remove_silence=True,
|
|
172
|
+
silence_threshold=0.02,
|
|
173
|
+
min_silence_duration=0.2
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
# Save cleaned audio
|
|
177
|
+
torchaudio.save("cleaned_audio.wav", cleaned_audio, sample_rate)
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
### Disabling Audio Cleanup
|
|
181
|
+
|
|
182
|
+
You can disable audio cleanup when cloning a voice:
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
voice_profile = model.clone_voice(
|
|
186
|
+
audio_input="sample_voice.wav",
|
|
187
|
+
transcript="This is a sample of my voice.",
|
|
188
|
+
cleanup_audio_fn=None # Disable audio cleanup
|
|
189
|
+
)
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
### Custom Audio Cleanup
|
|
193
|
+
|
|
194
|
+
You can also provide your own audio cleanup function:
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
def my_custom_cleanup(audio, sample_rate, **kwargs):
|
|
198
|
+
# Custom cleanup logic
|
|
199
|
+
return processed_audio
|
|
200
|
+
|
|
201
|
+
voice_profile = model.
|
voxy-0.0.2/setup.cfg
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
[metadata]
|
|
2
|
+
name = voxy
|
|
3
|
+
version = 0.0.2
|
|
4
|
+
url = https://github.com/thorwhalen/voxy
|
|
5
|
+
platforms = any
|
|
6
|
+
description_file = README.md
|
|
7
|
+
root_url = https://github.com/thorwhalen/
|
|
8
|
+
license = mit
|
|
9
|
+
author = Thor Whalen
|
|
10
|
+
description = Facade for voice cloning and speech synthesis
|
|
11
|
+
long_description = file:README.md
|
|
12
|
+
long_description_content_type = text/markdown
|
|
13
|
+
keywords =
|
|
14
|
+
display_name = voxy
|
|
15
|
+
|
|
16
|
+
[options]
|
|
17
|
+
packages = find:
|
|
18
|
+
include_package_data = True
|
|
19
|
+
zip_safe = False
|
|
20
|
+
install_requires =
|
|
21
|
+
|
|
22
|
+
[egg_info]
|
|
23
|
+
tag_build =
|
|
24
|
+
tag_date = 0
|
|
25
|
+
|
voxy-0.0.2/setup.py
ADDED
voxy-0.0.2/voxy/base.py
ADDED
|
@@ -0,0 +1,576 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Voxy: A flexible speech synthesis and voice cloning module.
|
|
3
|
+
|
|
4
|
+
This module provides a plugin architecture for working with different speech synthesis
|
|
5
|
+
models, with initial support for the CSM-1B model.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
import io
|
|
10
|
+
import pathlib
|
|
11
|
+
from typing import Union, Optional, List, Dict, Any, Callable, BinaryIO, Tuple
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
import tempfile
|
|
14
|
+
|
|
15
|
+
import torch
|
|
16
|
+
import torchaudio
|
|
17
|
+
import numpy as np
|
|
18
|
+
from huggingface_hub import hf_hub_download
|
|
19
|
+
|
|
20
|
+
# Try to import whisper for transcription, but don't fail if it's not available
|
|
21
|
+
try:
|
|
22
|
+
import whisper
|
|
23
|
+
|
|
24
|
+
_HAS_WHISPER = True
|
|
25
|
+
except ImportError:
|
|
26
|
+
_HAS_WHISPER = False
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Determine the default device for model inference
|
|
30
|
+
DFLT_VOXY_DEVICE = os.environ.get("DFLT_VOXY_DEVICE", None)
|
|
31
|
+
|
|
32
|
+
if DFLT_VOXY_DEVICE is None:
|
|
33
|
+
if torch.backends.mps.is_available():
|
|
34
|
+
DFLT_VOXY_DEVICE = "mps"
|
|
35
|
+
elif torch.cuda.is_available():
|
|
36
|
+
DFLT_VOXY_DEVICE = "cuda"
|
|
37
|
+
else:
|
|
38
|
+
DFLT_VOXY_DEVICE = "cpu"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# -----------------------------------------------------------------------------
|
|
42
|
+
# Helper functions for input normalization
|
|
43
|
+
# -----------------------------------------------------------------------------
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _resolve_audio_input(
|
|
47
|
+
audio_input: Union[str, bytes, BinaryIO, torch.Tensor, np.ndarray],
|
|
48
|
+
) -> Tuple[torch.Tensor, int]:
|
|
49
|
+
"""
|
|
50
|
+
Resolves various audio input formats to a torch.Tensor and sample rate.
|
|
51
|
+
|
|
52
|
+
Args:
|
|
53
|
+
audio_input: Audio in various formats:
|
|
54
|
+
- str: Path to an audio file
|
|
55
|
+
- bytes: Raw audio data
|
|
56
|
+
- BinaryIO: File-like object containing audio data
|
|
57
|
+
- torch.Tensor: Direct audio tensor
|
|
58
|
+
- np.ndarray: Numpy array of audio samples
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
Tuple of (audio_tensor, sample_rate)
|
|
62
|
+
"""
|
|
63
|
+
if isinstance(audio_input, str):
|
|
64
|
+
# Check if it's a file path
|
|
65
|
+
if os.path.isfile(audio_input):
|
|
66
|
+
return torchaudio.load(audio_input)
|
|
67
|
+
else:
|
|
68
|
+
raise ValueError(f"Audio path does not exist: {audio_input}")
|
|
69
|
+
|
|
70
|
+
elif isinstance(audio_input, bytes):
|
|
71
|
+
# Convert bytes to file-like object
|
|
72
|
+
byte_stream = io.BytesIO(audio_input)
|
|
73
|
+
return torchaudio.load(byte_stream)
|
|
74
|
+
|
|
75
|
+
elif isinstance(audio_input, (io.IOBase, BinaryIO)):
|
|
76
|
+
# File-like object
|
|
77
|
+
return torchaudio.load(audio_input)
|
|
78
|
+
|
|
79
|
+
elif isinstance(audio_input, torch.Tensor):
|
|
80
|
+
# Assume default sample rate of 16000 if directly passed tensor
|
|
81
|
+
# and the tensor shape is [channels, samples] or [samples]
|
|
82
|
+
if len(audio_input.shape) > 2:
|
|
83
|
+
raise ValueError(f"Invalid audio tensor shape: {audio_input.shape}")
|
|
84
|
+
return audio_input, 16000
|
|
85
|
+
|
|
86
|
+
elif isinstance(audio_input, np.ndarray):
|
|
87
|
+
# Convert numpy array to tensor
|
|
88
|
+
# Assume default sample rate of 16000
|
|
89
|
+
audio_tensor = torch.from_numpy(audio_input)
|
|
90
|
+
if len(audio_tensor.shape) == 1:
|
|
91
|
+
# Add channel dimension if not present
|
|
92
|
+
audio_tensor = audio_tensor.unsqueeze(0)
|
|
93
|
+
return audio_tensor, 16000
|
|
94
|
+
|
|
95
|
+
else:
|
|
96
|
+
raise TypeError(f"Unsupported audio input type: {type(audio_input)}")
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _resolve_text_input(text_input: Union[str, bytes, io.TextIOBase]) -> str:
|
|
100
|
+
"""
|
|
101
|
+
Resolves various text input formats to a string.
|
|
102
|
+
|
|
103
|
+
Args:
|
|
104
|
+
text_input: Text in various formats:
|
|
105
|
+
- str: Direct text or path to a text file
|
|
106
|
+
- bytes: UTF-8 encoded text
|
|
107
|
+
- TextIOBase: File-like object containing text
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
String containing the text
|
|
111
|
+
"""
|
|
112
|
+
if isinstance(text_input, str):
|
|
113
|
+
# If it starts with / and is a file, read the content
|
|
114
|
+
if text_input.startswith('/') and os.path.isfile(text_input):
|
|
115
|
+
with open(text_input, 'r') as f:
|
|
116
|
+
return f.read()
|
|
117
|
+
# Otherwise, use the string directly
|
|
118
|
+
return text_input
|
|
119
|
+
|
|
120
|
+
elif isinstance(text_input, bytes):
|
|
121
|
+
# Decode bytes to string
|
|
122
|
+
return text_input.decode('utf-8')
|
|
123
|
+
|
|
124
|
+
elif isinstance(text_input, io.TextIOBase):
|
|
125
|
+
# Read from file-like object
|
|
126
|
+
return text_input.read()
|
|
127
|
+
|
|
128
|
+
else:
|
|
129
|
+
raise TypeError(f"Unsupported text input type: {type(text_input)}")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# -----------------------------------------------------------------------------
|
|
133
|
+
# Audio processing functions
|
|
134
|
+
# -----------------------------------------------------------------------------
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def cleanup_audio(
|
|
138
|
+
audio: torch.Tensor,
|
|
139
|
+
sample_rate: int,
|
|
140
|
+
normalize: bool = True,
|
|
141
|
+
remove_silence: bool = True,
|
|
142
|
+
silence_threshold: float = 0.02,
|
|
143
|
+
min_silence_duration: float = 0.2,
|
|
144
|
+
) -> torch.Tensor:
|
|
145
|
+
"""
|
|
146
|
+
Clean up audio by normalizing volume and removing silence.
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
audio: Audio tensor [channels, samples] or [samples]
|
|
150
|
+
sample_rate: Sample rate of the audio
|
|
151
|
+
normalize: Whether to normalize the audio volume
|
|
152
|
+
remove_silence: Whether to remove silence
|
|
153
|
+
silence_threshold: Threshold for silence detection (0.0-1.0)
|
|
154
|
+
min_silence_duration: Minimum silence duration in seconds
|
|
155
|
+
|
|
156
|
+
Returns:
|
|
157
|
+
Processed audio tensor
|
|
158
|
+
"""
|
|
159
|
+
# Ensure input is 2D with shape [channels, samples]
|
|
160
|
+
if len(audio.shape) == 1:
|
|
161
|
+
audio = audio.unsqueeze(0)
|
|
162
|
+
|
|
163
|
+
# Convert to mono if not already
|
|
164
|
+
if audio.shape[0] > 1:
|
|
165
|
+
audio = torch.mean(audio, dim=0, keepdim=True)
|
|
166
|
+
|
|
167
|
+
# Move to CPU for processing
|
|
168
|
+
device = audio.device
|
|
169
|
+
audio = audio.cpu()
|
|
170
|
+
|
|
171
|
+
# Normalize volume
|
|
172
|
+
if normalize:
|
|
173
|
+
max_val = torch.max(torch.abs(audio))
|
|
174
|
+
if max_val > 0:
|
|
175
|
+
audio = audio / (max_val + 1e-8)
|
|
176
|
+
|
|
177
|
+
# Remove silence
|
|
178
|
+
if remove_silence:
|
|
179
|
+
# Convert to numpy for easier processing
|
|
180
|
+
audio_np = audio.squeeze(0).numpy()
|
|
181
|
+
|
|
182
|
+
# Calculate energy
|
|
183
|
+
energy = np.abs(audio_np)
|
|
184
|
+
|
|
185
|
+
# Find regions above threshold (speech)
|
|
186
|
+
is_speech = energy > silence_threshold
|
|
187
|
+
|
|
188
|
+
# Convert min_silence_duration to samples
|
|
189
|
+
min_silence_samples = int(min_silence_duration * sample_rate)
|
|
190
|
+
|
|
191
|
+
# Find speech segments
|
|
192
|
+
speech_segments = []
|
|
193
|
+
in_speech = False
|
|
194
|
+
speech_start = 0
|
|
195
|
+
|
|
196
|
+
for i in range(len(is_speech)):
|
|
197
|
+
if is_speech[i] and not in_speech:
|
|
198
|
+
# Start of speech segment
|
|
199
|
+
in_speech = True
|
|
200
|
+
speech_start = i
|
|
201
|
+
elif not is_speech[i] and in_speech:
|
|
202
|
+
# Potential end of speech segment
|
|
203
|
+
# Only end if silence is long enough
|
|
204
|
+
silence_count = 0
|
|
205
|
+
for j in range(i, min(len(is_speech), i + min_silence_samples)):
|
|
206
|
+
if not is_speech[j]:
|
|
207
|
+
silence_count += 1
|
|
208
|
+
else:
|
|
209
|
+
break
|
|
210
|
+
|
|
211
|
+
if silence_count >= min_silence_samples:
|
|
212
|
+
# End of speech segment
|
|
213
|
+
in_speech = False
|
|
214
|
+
speech_segments.append((speech_start, i))
|
|
215
|
+
|
|
216
|
+
# Handle case where audio ends during speech
|
|
217
|
+
if in_speech:
|
|
218
|
+
speech_segments.append((speech_start, len(is_speech)))
|
|
219
|
+
|
|
220
|
+
# Concatenate speech segments
|
|
221
|
+
if not speech_segments:
|
|
222
|
+
# If no speech found, return original audio
|
|
223
|
+
processed_audio = audio
|
|
224
|
+
else:
|
|
225
|
+
# Add small buffer around segments
|
|
226
|
+
buffer_samples = int(0.05 * sample_rate) # 50ms buffer
|
|
227
|
+
processed_segments = []
|
|
228
|
+
|
|
229
|
+
for start, end in speech_segments:
|
|
230
|
+
buffered_start = max(0, start - buffer_samples)
|
|
231
|
+
buffered_end = min(len(audio_np), end + buffer_samples)
|
|
232
|
+
processed_segments.append(audio_np[buffered_start:buffered_end])
|
|
233
|
+
|
|
234
|
+
# Concatenate all segments
|
|
235
|
+
processed_audio_np = np.concatenate(processed_segments)
|
|
236
|
+
processed_audio = torch.tensor(processed_audio_np, device='cpu').unsqueeze(
|
|
237
|
+
0
|
|
238
|
+
)
|
|
239
|
+
else:
|
|
240
|
+
processed_audio = audio
|
|
241
|
+
|
|
242
|
+
# Return to original device
|
|
243
|
+
return processed_audio.to(device)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def audio_to_text(
|
|
247
|
+
audio_input: Union[str, bytes, BinaryIO, torch.Tensor, np.ndarray],
|
|
248
|
+
model_size: str = "base",
|
|
249
|
+
) -> str:
|
|
250
|
+
"""
|
|
251
|
+
Transcribe audio to text using Whisper.
|
|
252
|
+
|
|
253
|
+
Args:
|
|
254
|
+
audio_input: Audio in various formats
|
|
255
|
+
model_size: Whisper model size ('tiny', 'base', 'small', 'medium', 'large')
|
|
256
|
+
|
|
257
|
+
Returns:
|
|
258
|
+
Transcribed text
|
|
259
|
+
|
|
260
|
+
Raises:
|
|
261
|
+
ImportError: If whisper is not installed
|
|
262
|
+
"""
|
|
263
|
+
if not _HAS_WHISPER:
|
|
264
|
+
raise ImportError(
|
|
265
|
+
"whisper is required for transcription. Install with 'pip install whisper'"
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
# Resolve audio input
|
|
269
|
+
audio, sample_rate = _resolve_audio_input(audio_input)
|
|
270
|
+
|
|
271
|
+
# Load whisper model
|
|
272
|
+
model = whisper.load_model(model_size)
|
|
273
|
+
|
|
274
|
+
# If audio is a torch tensor, convert to numpy array
|
|
275
|
+
if isinstance(audio, torch.Tensor):
|
|
276
|
+
# Ensure mono
|
|
277
|
+
if len(audio.shape) > 1 and audio.shape[0] > 1:
|
|
278
|
+
audio = torch.mean(audio, dim=0)
|
|
279
|
+
else:
|
|
280
|
+
audio = audio.squeeze(0)
|
|
281
|
+
|
|
282
|
+
# Convert to numpy
|
|
283
|
+
audio_np = audio.cpu().numpy()
|
|
284
|
+
else:
|
|
285
|
+
audio_np = audio
|
|
286
|
+
|
|
287
|
+
# Resample if needed
|
|
288
|
+
if sample_rate != 16000:
|
|
289
|
+
# Whisper expects 16kHz
|
|
290
|
+
# Use torchaudio for resampling
|
|
291
|
+
audio_tensor = torch.tensor(audio_np).unsqueeze(0)
|
|
292
|
+
audio_tensor = torchaudio.functional.resample(
|
|
293
|
+
audio_tensor, orig_freq=sample_rate, new_freq=16000
|
|
294
|
+
)
|
|
295
|
+
audio_np = audio_tensor.squeeze(0).numpy()
|
|
296
|
+
|
|
297
|
+
# Transcribe
|
|
298
|
+
result = model.transcribe(audio_np)
|
|
299
|
+
|
|
300
|
+
return result["text"].strip()
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
# -----------------------------------------------------------------------------
|
|
304
|
+
# Main SpeechModel classes
|
|
305
|
+
# -----------------------------------------------------------------------------
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
@dataclass
|
|
309
|
+
class VoiceProfile:
|
|
310
|
+
"""Data class to store voice cloning information."""
|
|
311
|
+
|
|
312
|
+
segment: Any # Model-specific voice segment
|
|
313
|
+
speaker_id: int
|
|
314
|
+
model_type: str
|
|
315
|
+
sample_rate: int
|
|
316
|
+
metadata: Dict[str, Any] = None
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
class SpeechModel:
|
|
320
|
+
"""Base class for speech synthesis models."""
|
|
321
|
+
|
|
322
|
+
def __init__(self, device: str = DFLT_VOXY_DEVICE):
|
|
323
|
+
"""
|
|
324
|
+
Initialize the speech model.
|
|
325
|
+
|
|
326
|
+
Args:
|
|
327
|
+
device: Device for model inference ('cuda', 'cpu', 'mps')
|
|
328
|
+
"""
|
|
329
|
+
self.device = device
|
|
330
|
+
|
|
331
|
+
def clone_voice(
|
|
332
|
+
self,
|
|
333
|
+
audio_input: Union[str, bytes, BinaryIO, torch.Tensor, np.ndarray],
|
|
334
|
+
transcript: Optional[str] = None,
|
|
335
|
+
speaker_id: int = 999,
|
|
336
|
+
*,
|
|
337
|
+
cleanup_audio_fn: Optional[Callable] = cleanup_audio,
|
|
338
|
+
) -> VoiceProfile:
|
|
339
|
+
"""
|
|
340
|
+
Create a voice profile from an audio sample and its transcript.
|
|
341
|
+
|
|
342
|
+
Args:
|
|
343
|
+
audio_input: Audio in various formats
|
|
344
|
+
transcript: Text transcription of the audio (if None, auto-transcribed)
|
|
345
|
+
speaker_id: Unique ID for this voice
|
|
346
|
+
cleanup_audio_fn: Function to clean up audio (None to skip)
|
|
347
|
+
|
|
348
|
+
Returns:
|
|
349
|
+
VoiceProfile: A packaged voice profile
|
|
350
|
+
"""
|
|
351
|
+
raise NotImplementedError("Subclasses must implement this method")
|
|
352
|
+
|
|
353
|
+
def generate_speech(
|
|
354
|
+
self,
|
|
355
|
+
text: Union[str, bytes, io.TextIOBase],
|
|
356
|
+
voice_profile: Optional[VoiceProfile] = None,
|
|
357
|
+
output_path: Optional[str] = None,
|
|
358
|
+
max_length_ms: int = 10000,
|
|
359
|
+
**kwargs,
|
|
360
|
+
) -> torch.Tensor:
|
|
361
|
+
"""
|
|
362
|
+
Generate speech using a voice profile.
|
|
363
|
+
|
|
364
|
+
Args:
|
|
365
|
+
text: Text to synthesize
|
|
366
|
+
voice_profile: Voice profile from clone_voice()
|
|
367
|
+
output_path: Path to save the audio (optional)
|
|
368
|
+
max_length_ms: Maximum audio length in milliseconds
|
|
369
|
+
**kwargs: Additional model-specific parameters
|
|
370
|
+
|
|
371
|
+
Returns:
|
|
372
|
+
Generated audio tensor
|
|
373
|
+
"""
|
|
374
|
+
raise NotImplementedError("Subclasses must implement this method")
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
class CSMSpeechModel(SpeechModel):
|
|
378
|
+
"""Speech model implementation using Sesame's CSM-1B model."""
|
|
379
|
+
|
|
380
|
+
def __init__(
|
|
381
|
+
self, model_path: Optional[str] = None, device: str = DFLT_VOXY_DEVICE
|
|
382
|
+
):
|
|
383
|
+
"""
|
|
384
|
+
Initialize the CSM speech model.
|
|
385
|
+
|
|
386
|
+
Args:
|
|
387
|
+
model_path: Path to the model checkpoint (None to download from HF)
|
|
388
|
+
device: Device for model inference ('cuda', 'cpu', 'mps')
|
|
389
|
+
"""
|
|
390
|
+
super().__init__(device)
|
|
391
|
+
self.model_path = model_path
|
|
392
|
+
self._generator = None # Lazy initialization
|
|
393
|
+
|
|
394
|
+
def _ensure_generator_loaded(self):
|
|
395
|
+
"""Ensure the generator is loaded."""
|
|
396
|
+
if self._generator is None:
|
|
397
|
+
# Import here to avoid dependencies if not using CSM
|
|
398
|
+
from generator import load_csm_1b, Segment
|
|
399
|
+
|
|
400
|
+
if self.model_path is None:
|
|
401
|
+
# Download the model if not provided
|
|
402
|
+
try:
|
|
403
|
+
self.model_path = hf_hub_download(
|
|
404
|
+
repo_id="sesame/csm-1b", filename="ckpt.pt"
|
|
405
|
+
)
|
|
406
|
+
except Exception as e:
|
|
407
|
+
raise RuntimeError(
|
|
408
|
+
"Failed to download CSM-1B model. Ensure you have huggingface-cli "
|
|
409
|
+
f"installed and are logged in with appropriate permissions: {e}"
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
# Load the generator
|
|
413
|
+
self._generator = load_csm_1b(self.model_path, self.device)
|
|
414
|
+
|
|
415
|
+
# Save a reference to the Segment class
|
|
416
|
+
self.Segment = Segment
|
|
417
|
+
|
|
418
|
+
def clone_voice(
|
|
419
|
+
self,
|
|
420
|
+
audio_input: Union[str, bytes, BinaryIO, torch.Tensor, np.ndarray],
|
|
421
|
+
transcript: Optional[str] = None,
|
|
422
|
+
speaker_id: int = 999,
|
|
423
|
+
*,
|
|
424
|
+
cleanup_audio_fn: Optional[Callable] = cleanup_audio,
|
|
425
|
+
) -> VoiceProfile:
|
|
426
|
+
"""
|
|
427
|
+
Create a voice profile from an audio sample and its transcript.
|
|
428
|
+
|
|
429
|
+
Args:
|
|
430
|
+
audio_input: Audio in various formats
|
|
431
|
+
transcript: Text transcription of the audio (if None, auto-transcribed)
|
|
432
|
+
speaker_id: Unique ID for this voice
|
|
433
|
+
cleanup_audio_fn: Function to clean up audio (None to skip)
|
|
434
|
+
|
|
435
|
+
Returns:
|
|
436
|
+
VoiceProfile: A packaged voice profile
|
|
437
|
+
"""
|
|
438
|
+
# Load model if not already loaded
|
|
439
|
+
self._ensure_generator_loaded()
|
|
440
|
+
|
|
441
|
+
# Resolve audio input
|
|
442
|
+
audio_tensor, sample_rate = _resolve_audio_input(audio_input)
|
|
443
|
+
|
|
444
|
+
# Clean up audio if requested
|
|
445
|
+
if cleanup_audio_fn is not None:
|
|
446
|
+
audio_tensor = cleanup_audio_fn(audio_tensor, sample_rate)
|
|
447
|
+
|
|
448
|
+
# Convert to mono if stereo
|
|
449
|
+
if audio_tensor.shape[0] > 1:
|
|
450
|
+
audio_tensor = torch.mean(audio_tensor, dim=0, keepdim=True)
|
|
451
|
+
|
|
452
|
+
# Squeeze out channel dimension if present
|
|
453
|
+
audio_tensor = audio_tensor.squeeze(0)
|
|
454
|
+
|
|
455
|
+
# Resample if needed
|
|
456
|
+
if sample_rate != self._generator.sample_rate:
|
|
457
|
+
audio_tensor = torchaudio.functional.resample(
|
|
458
|
+
audio_tensor,
|
|
459
|
+
orig_freq=sample_rate,
|
|
460
|
+
new_freq=self._generator.sample_rate,
|
|
461
|
+
)
|
|
462
|
+
|
|
463
|
+
# Auto-transcribe if no transcript provided
|
|
464
|
+
if transcript is None:
|
|
465
|
+
transcript = audio_to_text(audio_tensor)
|
|
466
|
+
else:
|
|
467
|
+
# Resolve transcript if not a string
|
|
468
|
+
transcript = _resolve_text_input(transcript)
|
|
469
|
+
|
|
470
|
+
# Create segment for voice profile
|
|
471
|
+
segment = self.Segment(
|
|
472
|
+
text=transcript, speaker=speaker_id, audio=audio_tensor.to(self.device)
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
# Create and return voice profile
|
|
476
|
+
return VoiceProfile(
|
|
477
|
+
segment=segment,
|
|
478
|
+
speaker_id=speaker_id,
|
|
479
|
+
model_type="csm",
|
|
480
|
+
sample_rate=self._generator.sample_rate,
|
|
481
|
+
metadata={
|
|
482
|
+
"transcript_length": len(transcript),
|
|
483
|
+
"audio_length_seconds": len(audio_tensor) / self._generator.sample_rate,
|
|
484
|
+
},
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
def generate_speech(
|
|
488
|
+
self,
|
|
489
|
+
text: Union[str, bytes, io.TextIOBase],
|
|
490
|
+
voice_profile: Optional[VoiceProfile] = None,
|
|
491
|
+
output_path: Optional[str] = None,
|
|
492
|
+
max_length_ms: int = 10000,
|
|
493
|
+
temperature: float = 0.7,
|
|
494
|
+
topk: int = 30,
|
|
495
|
+
) -> torch.Tensor:
|
|
496
|
+
"""
|
|
497
|
+
Generate speech using a voice profile.
|
|
498
|
+
|
|
499
|
+
Args:
|
|
500
|
+
text: Text to synthesize
|
|
501
|
+
voice_profile: Voice profile from clone_voice()
|
|
502
|
+
output_path: Path to save the audio (optional)
|
|
503
|
+
max_length_ms: Maximum audio length in milliseconds
|
|
504
|
+
temperature: Sampling temperature (lower = more deterministic)
|
|
505
|
+
topk: Top-k sampling parameter
|
|
506
|
+
|
|
507
|
+
Returns:
|
|
508
|
+
Generated audio tensor
|
|
509
|
+
"""
|
|
510
|
+
# Load model if not already loaded
|
|
511
|
+
self._ensure_generator_loaded()
|
|
512
|
+
|
|
513
|
+
# Resolve text input
|
|
514
|
+
text = _resolve_text_input(text)
|
|
515
|
+
|
|
516
|
+
# Set up context and speaker ID
|
|
517
|
+
if voice_profile is not None:
|
|
518
|
+
if voice_profile.model_type != "csm":
|
|
519
|
+
raise ValueError(
|
|
520
|
+
f"Incompatible voice profile type: {voice_profile.model_type}"
|
|
521
|
+
)
|
|
522
|
+
|
|
523
|
+
context = [voice_profile.segment]
|
|
524
|
+
speaker_id = voice_profile.speaker_id
|
|
525
|
+
else:
|
|
526
|
+
# No voice profile, use default speaker
|
|
527
|
+
context = []
|
|
528
|
+
speaker_id = 0
|
|
529
|
+
|
|
530
|
+
# Add punctuation if missing to help with phrasing
|
|
531
|
+
if not any(p in text for p in ['.', ',', '!', '?']):
|
|
532
|
+
text = text + '.'
|
|
533
|
+
|
|
534
|
+
# Generate audio
|
|
535
|
+
audio = self._generator.generate(
|
|
536
|
+
text=text,
|
|
537
|
+
speaker=speaker_id,
|
|
538
|
+
context=context,
|
|
539
|
+
max_audio_length_ms=max_length_ms,
|
|
540
|
+
temperature=temperature,
|
|
541
|
+
topk=topk,
|
|
542
|
+
)
|
|
543
|
+
|
|
544
|
+
# Save if path provided
|
|
545
|
+
if output_path:
|
|
546
|
+
os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True)
|
|
547
|
+
torchaudio.save(
|
|
548
|
+
output_path, audio.unsqueeze(0).cpu(), self._generator.sample_rate
|
|
549
|
+
)
|
|
550
|
+
|
|
551
|
+
return audio
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
# -----------------------------------------------------------------------------
|
|
555
|
+
# Factory function for creating speech models
|
|
556
|
+
# -----------------------------------------------------------------------------
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def create_speech_model(model_type: str = "csm", **kwargs) -> SpeechModel:
|
|
560
|
+
"""
|
|
561
|
+
Create a speech model of the specified type.
|
|
562
|
+
|
|
563
|
+
Args:
|
|
564
|
+
model_type: Type of speech model ('csm' currently supported)
|
|
565
|
+
**kwargs: Additional model-specific parameters
|
|
566
|
+
|
|
567
|
+
Returns:
|
|
568
|
+
SpeechModel instance
|
|
569
|
+
|
|
570
|
+
Raises:
|
|
571
|
+
ValueError: If the model type is not supported
|
|
572
|
+
"""
|
|
573
|
+
if model_type.lower() == "csm":
|
|
574
|
+
return CSMSpeechModel(**kwargs)
|
|
575
|
+
else:
|
|
576
|
+
raise ValueError(f"Unsupported model type: {model_type}")
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
Metadata-Version: 2.2
|
|
2
|
+
Name: voxy
|
|
3
|
+
Version: 0.0.2
|
|
4
|
+
Summary: Facade for voice cloning and speech synthesis
|
|
5
|
+
Home-page: https://github.com/thorwhalen/voxy
|
|
6
|
+
Author: Thor Whalen
|
|
7
|
+
License: mit
|
|
8
|
+
Platform: any
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
|
|
12
|
+
# voxy
|
|
13
|
+
|
|
14
|
+
Facade for voice cloning and speech synthesis
|
|
15
|
+
|
|
16
|
+
To install: ```pip install voxy```
|
|
17
|
+
|
|
18
|
+
Voxy is a flexible Python module for speech synthesis and voice cloning, with initial support for the Sesame CSM-1B model. It provides a plugin architecture that can be extended to support other models in the future.
|
|
19
|
+
|
|
20
|
+
## Features
|
|
21
|
+
|
|
22
|
+
- Voice cloning from audio samples
|
|
23
|
+
- High-quality speech synthesis
|
|
24
|
+
- Flexible input formats (file paths, bytes, streams, tensors)
|
|
25
|
+
- Audio cleanup utilities
|
|
26
|
+
- Automatic audio transcription (using Whisper)
|
|
27
|
+
- Plugin architecture for different speech models
|
|
28
|
+
|
|
29
|
+
## Installation
|
|
30
|
+
|
|
31
|
+
### Prerequisites
|
|
32
|
+
|
|
33
|
+
- Python 3.10+
|
|
34
|
+
- PyTorch and TorchAudio
|
|
35
|
+
- CUDA-compatible GPU (recommended)
|
|
36
|
+
- FFmpeg for audio processing
|
|
37
|
+
|
|
38
|
+
### Install the CSM Model
|
|
39
|
+
|
|
40
|
+
The intention is to make `voxy` into a plugin-enabled facade, where you can chose your
|
|
41
|
+
own engine (for voice cloning, voice synthesis, etc.).
|
|
42
|
+
But for now, we just support, what seems to be the best open-source model out there
|
|
43
|
+
(at the time of writing this):
|
|
44
|
+
[Sesame AI Lab's](https://www.sesame.com/research/crossing_the_uncanny_valley_of_voice)
|
|
45
|
+
CSM model. It's just that, well, they did an amazing job at the model, but a terrible one
|
|
46
|
+
(so far) for the python interface -- which is what inspired me to develop `voxy`
|
|
47
|
+
in the first place.
|
|
48
|
+
|
|
49
|
+
Follow the instructions in the [CSM repository](https://github.com/SesameAILabs/csm)
|
|
50
|
+
to install the CSM model and its dependencies.
|
|
51
|
+
|
|
52
|
+
## Quick Start
|
|
53
|
+
|
|
54
|
+
### Basic Usage
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from voxy import create_speech_model
|
|
58
|
+
|
|
59
|
+
# Create a speech model
|
|
60
|
+
model = create_speech_model(model_type="csm")
|
|
61
|
+
|
|
62
|
+
# Generate speech with default voice
|
|
63
|
+
audio = model.generate_speech(
|
|
64
|
+
text="Hello, this is a test of the CSM speech model.",
|
|
65
|
+
output_path="output.wav"
|
|
66
|
+
)
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### Voice Cloning
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from voxy import create_speech_model
|
|
73
|
+
|
|
74
|
+
# Create a speech model
|
|
75
|
+
model = create_speech_model(model_type="csm")
|
|
76
|
+
|
|
77
|
+
# Clone a voice from an audio file
|
|
78
|
+
voice_profile = model.clone_voice(
|
|
79
|
+
audio_input="sample_voice.wav",
|
|
80
|
+
transcript="This is a sample of my voice for cloning purposes."
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
# Generate speech with the cloned voice
|
|
84
|
+
audio = model.generate_speech(
|
|
85
|
+
text="This is my cloned voice speaking. Isn't it amazing?",
|
|
86
|
+
voice_profile=voice_profile,
|
|
87
|
+
output_path="cloned_voice.wav"
|
|
88
|
+
)
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
### Automatic Transcription
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
from voxy import create_speech_model
|
|
95
|
+
|
|
96
|
+
# Create a speech model
|
|
97
|
+
model = create_speech_model(model_type="csm")
|
|
98
|
+
|
|
99
|
+
# Clone a voice with automatic transcription
|
|
100
|
+
voice_profile = model.clone_voice(
|
|
101
|
+
audio_input="sample_voice.wav",
|
|
102
|
+
# No transcript provided, will use automatic transcription
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
# Generate speech with the cloned voice
|
|
106
|
+
audio = model.generate_speech(
|
|
107
|
+
text="This voice was cloned using automatic transcription.",
|
|
108
|
+
voice_profile=voice_profile,
|
|
109
|
+
output_path="auto_transcribed_voice.wav"
|
|
110
|
+
)
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Flexible Input Formats
|
|
114
|
+
|
|
115
|
+
The module supports various input formats:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
# From file path
|
|
119
|
+
voice_profile1 = model.clone_voice(
|
|
120
|
+
audio_input="sample_voice.wav",
|
|
121
|
+
transcript="Text transcript."
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
# From bytes
|
|
125
|
+
with open("sample_voice.wav", "rb") as f:
|
|
126
|
+
audio_bytes = f.read()
|
|
127
|
+
voice_profile2 = model.clone_voice(
|
|
128
|
+
audio_input=audio_bytes,
|
|
129
|
+
transcript="Text transcript."
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
# From file object
|
|
133
|
+
with open("sample_voice.wav", "rb") as f:
|
|
134
|
+
voice_profile3 = model.clone_voice(
|
|
135
|
+
audio_input=f,
|
|
136
|
+
transcript="Text transcript."
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
# From tensor
|
|
140
|
+
import torch
|
|
141
|
+
import torchaudio
|
|
142
|
+
audio_tensor, sample_rate = torchaudio.load("sample_voice.wav")
|
|
143
|
+
voice_profile4 = model.clone_voice(
|
|
144
|
+
audio_input=audio_tensor,
|
|
145
|
+
transcript="Text transcript."
|
|
146
|
+
)
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## Configuration
|
|
150
|
+
|
|
151
|
+
You can configure the default device by setting the `DFLT_VOXY_DEVICE` environment variable:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
# Use CUDA
|
|
155
|
+
export DFLT_VOXY_DEVICE=cuda
|
|
156
|
+
|
|
157
|
+
# Use CPU
|
|
158
|
+
export DFLT_VOXY_DEVICE=cpu
|
|
159
|
+
|
|
160
|
+
# Use MPS (Apple Silicon)
|
|
161
|
+
export DFLT_VOXY_DEVICE=mps
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## Advanced Usage
|
|
165
|
+
|
|
166
|
+
### Audio Cleanup
|
|
167
|
+
|
|
168
|
+
The module includes an audio cleanup function that normalizes volume and removes silence:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from voxy import cleanup_audio
|
|
172
|
+
import torchaudio
|
|
173
|
+
|
|
174
|
+
# Load audio
|
|
175
|
+
audio, sample_rate = torchaudio.load("noisy_audio.wav")
|
|
176
|
+
|
|
177
|
+
# Clean up audio
|
|
178
|
+
cleaned_audio = cleanup_audio(
|
|
179
|
+
audio=audio,
|
|
180
|
+
sample_rate=sample_rate,
|
|
181
|
+
normalize=True,
|
|
182
|
+
remove_silence=True,
|
|
183
|
+
silence_threshold=0.02,
|
|
184
|
+
min_silence_duration=0.2
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
# Save cleaned audio
|
|
188
|
+
torchaudio.save("cleaned_audio.wav", cleaned_audio, sample_rate)
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
### Disabling Audio Cleanup
|
|
192
|
+
|
|
193
|
+
You can disable audio cleanup when cloning a voice:
|
|
194
|
+
|
|
195
|
+
```python
|
|
196
|
+
voice_profile = model.clone_voice(
|
|
197
|
+
audio_input="sample_voice.wav",
|
|
198
|
+
transcript="This is a sample of my voice.",
|
|
199
|
+
cleanup_audio_fn=None # Disable audio cleanup
|
|
200
|
+
)
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
### Custom Audio Cleanup
|
|
204
|
+
|
|
205
|
+
You can also provide your own audio cleanup function:
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
def my_custom_cleanup(audio, sample_rate, **kwargs):
|
|
209
|
+
# Custom cleanup logic
|
|
210
|
+
return processed_audio
|
|
211
|
+
|
|
212
|
+
voice_profile = model.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
voxy
|