MyanmarTTS 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- myanmartts-1.0.0/LICENSE +21 -0
- myanmartts-1.0.0/PKG-INFO +173 -0
- myanmartts-1.0.0/README.md +148 -0
- myanmartts-1.0.0/pyproject.toml +36 -0
- myanmartts-1.0.0/src/myanmar_tts/__init__.py +42 -0
- myanmartts-1.0.0/src/myanmar_tts/api.py +82 -0
- myanmartts-1.0.0/src/myanmar_tts/burmese.py +42 -0
- myanmartts-1.0.0/src/myanmar_tts/config.py +50 -0
- myanmartts-1.0.0/src/myanmar_tts/datas/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/datas/all_sources_dataset.py +160 -0
- myanmartts-1.0.0/src/myanmar_tts/datas/collate_wav.py +39 -0
- myanmartts-1.0.0/src/myanmar_tts/datas/dataset.py +69 -0
- myanmartts-1.0.0/src/myanmar_tts/datas/local_shards_dataset.py +73 -0
- myanmartts-1.0.0/src/myanmar_tts/models/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/models/diffusion_transformer.py +205 -0
- myanmartts-1.0.0/src/myanmar_tts/models/duration_predictor.py +40 -0
- myanmartts-1.0.0/src/myanmar_tts/models/estimator.py +138 -0
- myanmartts-1.0.0/src/myanmar_tts/models/flow_matching.py +100 -0
- myanmartts-1.0.0/src/myanmar_tts/models/model.py +178 -0
- myanmartts-1.0.0/src/myanmar_tts/models/reference_encoder.py +168 -0
- myanmartts-1.0.0/src/myanmar_tts/models/text_encoder.py +44 -0
- myanmartts-1.0.0/src/myanmar_tts/monotonic_align/__init__.py +16 -0
- myanmartts-1.0.0/src/myanmar_tts/monotonic_align/core.py +46 -0
- myanmartts-1.0.0/src/myanmar_tts/symbols.py +57 -0
- myanmartts-1.0.0/src/myanmar_tts/text/LICENSE +19 -0
- myanmartts-1.0.0/src/myanmar_tts/text/__init__.py +16 -0
- myanmartts-1.0.0/src/myanmar_tts/text/burmese.py +42 -0
- myanmartts-1.0.0/src/myanmar_tts/text/cleaners.py +10 -0
- myanmartts-1.0.0/src/myanmar_tts/text/symbols.py +57 -0
- myanmartts-1.0.0/src/myanmar_tts/utils/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/utils/audio.py +74 -0
- myanmartts-1.0.0/src/myanmar_tts/utils/load.py +43 -0
- myanmartts-1.0.0/src/myanmar_tts/utils/mask.py +8 -0
- myanmartts-1.0.0/src/myanmar_tts/utils/scheduler.py +428 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/backbone.py +214 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/head.py +257 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/model.py +57 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/unify.py +60 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/README.md +41 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/config.py +41 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/dataset.py +57 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/inference.ipynb +79 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/backbone.py +57 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/discriminator.py +171 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/head.py +118 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/loss.py +66 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/model.py +20 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/module.py +47 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/preprocess.py +45 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/requirements.txt +2 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/train.py +165 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/__init__.py +0 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/audio.py +74 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/load.py +53 -0
- myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/scheduler.py +298 -0
myanmartts-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 KdaiP
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: MyanmarTTS
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: A from-scratch Burmese (Myanmar) text-to-speech model (CC0).
|
|
5
|
+
Project-URL: Homepage, https://huggingface.co/freococo/MyanmarTTS
|
|
6
|
+
Project-URL: Repository, https://huggingface.co/freococo/MyanmarTTS
|
|
7
|
+
Author: freococo
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: burmese,myanmar,speech,text-to-speech,tts
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Requires-Python: >=3.9
|
|
15
|
+
Requires-Dist: huggingface-hub>=0.20
|
|
16
|
+
Requires-Dist: numba
|
|
17
|
+
Requires-Dist: numpy<2.3
|
|
18
|
+
Requires-Dist: safetensors
|
|
19
|
+
Requires-Dist: scipy
|
|
20
|
+
Requires-Dist: soundfile
|
|
21
|
+
Requires-Dist: torch>=2.0
|
|
22
|
+
Requires-Dist: torchaudio>=2.0
|
|
23
|
+
Requires-Dist: torchdiffeq
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
<div align="center">
|
|
27
|
+
|
|
28
|
+
# StableTTS
|
|
29
|
+
|
|
30
|
+
Next-generation TTS model using flow-matching and DiT, inspired by [Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3).
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
</div>
|
|
34
|
+
|
|
35
|
+
## Introduction
|
|
36
|
+
|
|
37
|
+
As the first open-source TTS model that tried to combine flow-matching and DiT, **StableTTS** is a fast and lightweight TTS model for chinese, english and japanese speech generation. It has 31M parameters.
|
|
38
|
+
|
|
39
|
+
✨ **Huggingface demo:** [🤗](https://huggingface.co/spaces/KdaiP/StableTTS1.1)
|
|
40
|
+
|
|
41
|
+
## News
|
|
42
|
+
|
|
43
|
+
2024/10: A new autoregressive TTS model is coming soon...
|
|
44
|
+
|
|
45
|
+
2024/9: 🚀 **StableTTS V1.1 Released** ⭐ Audio quality is largely improved ⭐
|
|
46
|
+
|
|
47
|
+
⭐ **V1.1 Release Highlights:**
|
|
48
|
+
|
|
49
|
+
- Fixed critical issues that cause the audio quality being much lower than expected. (Mainly in Mel spectrogram and Attention mask)
|
|
50
|
+
- Introduced U-Net-like long skip connections to the DiT in the Flow-matching Decoder.
|
|
51
|
+
- Use cosine timestep scheduler from [Cosyvoice](https://github.com/FunAudioLLM/CosyVoice)
|
|
52
|
+
- Add support for CFG (Classifier-Free Guidance).
|
|
53
|
+
- Add support for [FireflyGAN vocoder](https://github.com/fishaudio/vocoder/releases/tag/1.0.0).
|
|
54
|
+
- Switched to [torchdiffeq](https://github.com/rtqichen/torchdiffeq) for ODE solvers.
|
|
55
|
+
- Improved Chinese text frontend (partially based on [gpt-sovits2](https://github.com/RVC-Boss/GPT-SoVITS)).
|
|
56
|
+
- Multilingual support (Chinese, English, Japanese) in a single checkpoint.
|
|
57
|
+
- Increased parameters: 10M -> 31M.
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
## Pretrained models
|
|
61
|
+
|
|
62
|
+
### Text-To-Mel model
|
|
63
|
+
|
|
64
|
+
Download and place the model in the `./checkpoints` directory, it is ready for inference, finetuning and webui.
|
|
65
|
+
|
|
66
|
+
| Model Name | Task Details | Dataset | Download Link |
|
|
67
|
+
|:----------:|:------------:|:-------------:|:-------------:|
|
|
68
|
+
| StableTTS | text to mel | 600 hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/StableTTS/checkpoint_0.pt)|
|
|
69
|
+
|
|
70
|
+
### Mel-To-Wav model
|
|
71
|
+
|
|
72
|
+
Choose a vocoder (`vocos` or `firefly-gan` ) and place it in the `./vocoders/pretrained` directory.
|
|
73
|
+
|
|
74
|
+
| Model Name | Task Details | Dataset | Download Link |
|
|
75
|
+
|:----------:|:------------:|:-------------:|:-------------:|
|
|
76
|
+
| Vocos | mel to wav | 2k hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/vocoders/vocos.pt)|
|
|
77
|
+
| firefly-gan-base | mel to wav | HiFi-16kh | [download from fishaudio](https://github.com/fishaudio/vocoder/releases/download/1.0.0/firefly-gan-base-generator.ckpt)|
|
|
78
|
+
|
|
79
|
+
## Installation
|
|
80
|
+
|
|
81
|
+
1. **Install pytorch**: Follow the [official PyTorch guide](https://pytorch.org/get-started/locally/) to install pytorch and torchaudio. We recommend the latest version (tested with PyTorch 2.4 and Python 3.12).
|
|
82
|
+
|
|
83
|
+
2. **Install Dependencies**: Run the following command to install the required Python packages:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
pip install -r requirements.txt
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## Inference
|
|
90
|
+
|
|
91
|
+
For detailed inference instructions, please refer to `inference.ipynb`
|
|
92
|
+
|
|
93
|
+
We also provide a webui based on gradio, please refer to `webui.py`
|
|
94
|
+
|
|
95
|
+
## Training
|
|
96
|
+
|
|
97
|
+
StableTTS is designed to be trained easily. We only need text and audio pairs, without any speaker id or extra feature extraction. Here’s how to get started:
|
|
98
|
+
|
|
99
|
+
### Preparing Your Data
|
|
100
|
+
|
|
101
|
+
1. **Generate Text and Audio pairs**: Generate the text and audio pair filelist as `./filelists/example.txt`. Some recipes of open-source datasets could be found in `./recipes`.
|
|
102
|
+
|
|
103
|
+
2. **Run Preprocessing**: Adjust the `DataConfig` in `preprocess.py` to set your input and output paths, then run the script. This will process the audio and text according to your list, outputting a JSON file with paths to mel features and phonemes.
|
|
104
|
+
|
|
105
|
+
**Note: Process multilingual data separately by changing the `language` setting in `DataConfig`**
|
|
106
|
+
|
|
107
|
+
### Start training
|
|
108
|
+
|
|
109
|
+
1. **Adjust Training Configuration**: In `config.py`, modify `TrainConfig` to set your file list path and adjust training parameters (such as batch_size) as needed.
|
|
110
|
+
|
|
111
|
+
2. **Start the Training Process**: Launch `train.py` to start training your model.
|
|
112
|
+
|
|
113
|
+
Note: For finetuning, download the pretrained model and place it in the `model_save_path` directory specified in `TrainConfig`. Training script will automatically detect and load the pretrained checkpoint.
|
|
114
|
+
|
|
115
|
+
### (Optional) Vocoder training
|
|
116
|
+
|
|
117
|
+
The `./vocoder/vocos` folder contains the training and finetuning codes for vocos vocoder.
|
|
118
|
+
|
|
119
|
+
For other types of vocoders, we recommend to train by using [fishaudio vocoder](https://github.com/fishaudio/vocoder): an uniform interface for developing various vocoders. We use the same spectrogram transform so the vocoders trained is compatible with StableTTS.
|
|
120
|
+
|
|
121
|
+
## Model structure
|
|
122
|
+
|
|
123
|
+
<div align="center">
|
|
124
|
+
|
|
125
|
+
<p style="text-align: center;">
|
|
126
|
+
<img src="./figures/structure.jpg" height="512"/>
|
|
127
|
+
</p>
|
|
128
|
+
|
|
129
|
+
</div>
|
|
130
|
+
|
|
131
|
+
- We use the Diffusion Convolution Transformer block from [Hierspeech++](https://github.com/sh-lee-prml/HierSpeechpp), which is a combination of original [DiT](https://github.com/sh-lee-prml/HierSpeechpp) and [FFT](https://arxiv.org/pdf/1905.09263.pdf)(Feed forward Transformer from fastspeech) for better prosody.
|
|
132
|
+
|
|
133
|
+
- In flow-matching decoder, we add a [FiLM layer](https://arxiv.org/abs/1709.07871) before DiT block to condition timestep embedding into model.
|
|
134
|
+
|
|
135
|
+
## References
|
|
136
|
+
|
|
137
|
+
The development of our models heavily relies on insights and code from various projects. We express our heartfelt thanks to the creators of the following:
|
|
138
|
+
|
|
139
|
+
### Direct Inspirations
|
|
140
|
+
|
|
141
|
+
[Matcha TTS](https://github.com/shivammehta25/Matcha-TTS): Essential flow-matching code.
|
|
142
|
+
|
|
143
|
+
[Grad TTS](https://github.com/huawei-noah/Speech-Backbones/tree/main/Grad-TTS): Diffusion model structure.
|
|
144
|
+
|
|
145
|
+
[Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3): Idea of combining flow-matching and DiT.
|
|
146
|
+
|
|
147
|
+
[Vits](https://github.com/jaywalnut310/vits): Code style and MAS insights, DistributedBucketSampler.
|
|
148
|
+
|
|
149
|
+
### Additional References:
|
|
150
|
+
|
|
151
|
+
[plowtts-pytorch](https://github.com/p0p4k/pflowtts_pytorch): codes of MAS in training
|
|
152
|
+
|
|
153
|
+
[Bert-VITS2](https://github.com/Plachtaa/VITS-fast-fine-tuning) : numba version of MAS and modern pytorch codes of Vits
|
|
154
|
+
|
|
155
|
+
[fish-speech](https://github.com/fishaudio/fish-speech): dataclass usage and mel-spectrogram transforms using torchaudio, gradio webui
|
|
156
|
+
|
|
157
|
+
[gpt-sovits](https://github.com/RVC-Boss/GPT-SoVITS): melstyle encoder for voice clone
|
|
158
|
+
|
|
159
|
+
[coqui xtts](https://huggingface.co/spaces/coqui/xtts): gradio webui
|
|
160
|
+
|
|
161
|
+
Chinese Dirtionary Of DiffSinger: [Multi-langs_Dictionary](https://github.com/colstone/Multi-langs_Dictionary) and [atonyxu's fork](https://github.com/atonyxu/Multi-langs_Dictionary)
|
|
162
|
+
|
|
163
|
+
## TODO
|
|
164
|
+
|
|
165
|
+
- [x] Release pretrained models.
|
|
166
|
+
- [x] Support Japanese language.
|
|
167
|
+
- [x] User friendly preprocess and inference script.
|
|
168
|
+
- [x] Enhance documentation and citations.
|
|
169
|
+
- [x] Release multilingual checkpoint.
|
|
170
|
+
|
|
171
|
+
## Disclaimer
|
|
172
|
+
|
|
173
|
+
Any organization or individual is prohibited from using any technology in this repo to generate or edit someone's speech without his/her consent, including but not limited to government leaders, political figures, and celebrities. If you do not comply with this item, you could be in violation of copyright laws.
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+
# StableTTS
|
|
4
|
+
|
|
5
|
+
Next-generation TTS model using flow-matching and DiT, inspired by [Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3).
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
</div>
|
|
9
|
+
|
|
10
|
+
## Introduction
|
|
11
|
+
|
|
12
|
+
As the first open-source TTS model that tried to combine flow-matching and DiT, **StableTTS** is a fast and lightweight TTS model for chinese, english and japanese speech generation. It has 31M parameters.
|
|
13
|
+
|
|
14
|
+
✨ **Huggingface demo:** [🤗](https://huggingface.co/spaces/KdaiP/StableTTS1.1)
|
|
15
|
+
|
|
16
|
+
## News
|
|
17
|
+
|
|
18
|
+
2024/10: A new autoregressive TTS model is coming soon...
|
|
19
|
+
|
|
20
|
+
2024/9: 🚀 **StableTTS V1.1 Released** ⭐ Audio quality is largely improved ⭐
|
|
21
|
+
|
|
22
|
+
⭐ **V1.1 Release Highlights:**
|
|
23
|
+
|
|
24
|
+
- Fixed critical issues that cause the audio quality being much lower than expected. (Mainly in Mel spectrogram and Attention mask)
|
|
25
|
+
- Introduced U-Net-like long skip connections to the DiT in the Flow-matching Decoder.
|
|
26
|
+
- Use cosine timestep scheduler from [Cosyvoice](https://github.com/FunAudioLLM/CosyVoice)
|
|
27
|
+
- Add support for CFG (Classifier-Free Guidance).
|
|
28
|
+
- Add support for [FireflyGAN vocoder](https://github.com/fishaudio/vocoder/releases/tag/1.0.0).
|
|
29
|
+
- Switched to [torchdiffeq](https://github.com/rtqichen/torchdiffeq) for ODE solvers.
|
|
30
|
+
- Improved Chinese text frontend (partially based on [gpt-sovits2](https://github.com/RVC-Boss/GPT-SoVITS)).
|
|
31
|
+
- Multilingual support (Chinese, English, Japanese) in a single checkpoint.
|
|
32
|
+
- Increased parameters: 10M -> 31M.
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
## Pretrained models
|
|
36
|
+
|
|
37
|
+
### Text-To-Mel model
|
|
38
|
+
|
|
39
|
+
Download and place the model in the `./checkpoints` directory, it is ready for inference, finetuning and webui.
|
|
40
|
+
|
|
41
|
+
| Model Name | Task Details | Dataset | Download Link |
|
|
42
|
+
|:----------:|:------------:|:-------------:|:-------------:|
|
|
43
|
+
| StableTTS | text to mel | 600 hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/StableTTS/checkpoint_0.pt)|
|
|
44
|
+
|
|
45
|
+
### Mel-To-Wav model
|
|
46
|
+
|
|
47
|
+
Choose a vocoder (`vocos` or `firefly-gan` ) and place it in the `./vocoders/pretrained` directory.
|
|
48
|
+
|
|
49
|
+
| Model Name | Task Details | Dataset | Download Link |
|
|
50
|
+
|:----------:|:------------:|:-------------:|:-------------:|
|
|
51
|
+
| Vocos | mel to wav | 2k hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/vocoders/vocos.pt)|
|
|
52
|
+
| firefly-gan-base | mel to wav | HiFi-16kh | [download from fishaudio](https://github.com/fishaudio/vocoder/releases/download/1.0.0/firefly-gan-base-generator.ckpt)|
|
|
53
|
+
|
|
54
|
+
## Installation
|
|
55
|
+
|
|
56
|
+
1. **Install pytorch**: Follow the [official PyTorch guide](https://pytorch.org/get-started/locally/) to install pytorch and torchaudio. We recommend the latest version (tested with PyTorch 2.4 and Python 3.12).
|
|
57
|
+
|
|
58
|
+
2. **Install Dependencies**: Run the following command to install the required Python packages:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install -r requirements.txt
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Inference
|
|
65
|
+
|
|
66
|
+
For detailed inference instructions, please refer to `inference.ipynb`
|
|
67
|
+
|
|
68
|
+
We also provide a webui based on gradio, please refer to `webui.py`
|
|
69
|
+
|
|
70
|
+
## Training
|
|
71
|
+
|
|
72
|
+
StableTTS is designed to be trained easily. We only need text and audio pairs, without any speaker id or extra feature extraction. Here’s how to get started:
|
|
73
|
+
|
|
74
|
+
### Preparing Your Data
|
|
75
|
+
|
|
76
|
+
1. **Generate Text and Audio pairs**: Generate the text and audio pair filelist as `./filelists/example.txt`. Some recipes of open-source datasets could be found in `./recipes`.
|
|
77
|
+
|
|
78
|
+
2. **Run Preprocessing**: Adjust the `DataConfig` in `preprocess.py` to set your input and output paths, then run the script. This will process the audio and text according to your list, outputting a JSON file with paths to mel features and phonemes.
|
|
79
|
+
|
|
80
|
+
**Note: Process multilingual data separately by changing the `language` setting in `DataConfig`**
|
|
81
|
+
|
|
82
|
+
### Start training
|
|
83
|
+
|
|
84
|
+
1. **Adjust Training Configuration**: In `config.py`, modify `TrainConfig` to set your file list path and adjust training parameters (such as batch_size) as needed.
|
|
85
|
+
|
|
86
|
+
2. **Start the Training Process**: Launch `train.py` to start training your model.
|
|
87
|
+
|
|
88
|
+
Note: For finetuning, download the pretrained model and place it in the `model_save_path` directory specified in `TrainConfig`. Training script will automatically detect and load the pretrained checkpoint.
|
|
89
|
+
|
|
90
|
+
### (Optional) Vocoder training
|
|
91
|
+
|
|
92
|
+
The `./vocoder/vocos` folder contains the training and finetuning codes for vocos vocoder.
|
|
93
|
+
|
|
94
|
+
For other types of vocoders, we recommend to train by using [fishaudio vocoder](https://github.com/fishaudio/vocoder): an uniform interface for developing various vocoders. We use the same spectrogram transform so the vocoders trained is compatible with StableTTS.
|
|
95
|
+
|
|
96
|
+
## Model structure
|
|
97
|
+
|
|
98
|
+
<div align="center">
|
|
99
|
+
|
|
100
|
+
<p style="text-align: center;">
|
|
101
|
+
<img src="./figures/structure.jpg" height="512"/>
|
|
102
|
+
</p>
|
|
103
|
+
|
|
104
|
+
</div>
|
|
105
|
+
|
|
106
|
+
- We use the Diffusion Convolution Transformer block from [Hierspeech++](https://github.com/sh-lee-prml/HierSpeechpp), which is a combination of original [DiT](https://github.com/sh-lee-prml/HierSpeechpp) and [FFT](https://arxiv.org/pdf/1905.09263.pdf)(Feed forward Transformer from fastspeech) for better prosody.
|
|
107
|
+
|
|
108
|
+
- In flow-matching decoder, we add a [FiLM layer](https://arxiv.org/abs/1709.07871) before DiT block to condition timestep embedding into model.
|
|
109
|
+
|
|
110
|
+
## References
|
|
111
|
+
|
|
112
|
+
The development of our models heavily relies on insights and code from various projects. We express our heartfelt thanks to the creators of the following:
|
|
113
|
+
|
|
114
|
+
### Direct Inspirations
|
|
115
|
+
|
|
116
|
+
[Matcha TTS](https://github.com/shivammehta25/Matcha-TTS): Essential flow-matching code.
|
|
117
|
+
|
|
118
|
+
[Grad TTS](https://github.com/huawei-noah/Speech-Backbones/tree/main/Grad-TTS): Diffusion model structure.
|
|
119
|
+
|
|
120
|
+
[Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3): Idea of combining flow-matching and DiT.
|
|
121
|
+
|
|
122
|
+
[Vits](https://github.com/jaywalnut310/vits): Code style and MAS insights, DistributedBucketSampler.
|
|
123
|
+
|
|
124
|
+
### Additional References:
|
|
125
|
+
|
|
126
|
+
[plowtts-pytorch](https://github.com/p0p4k/pflowtts_pytorch): codes of MAS in training
|
|
127
|
+
|
|
128
|
+
[Bert-VITS2](https://github.com/Plachtaa/VITS-fast-fine-tuning) : numba version of MAS and modern pytorch codes of Vits
|
|
129
|
+
|
|
130
|
+
[fish-speech](https://github.com/fishaudio/fish-speech): dataclass usage and mel-spectrogram transforms using torchaudio, gradio webui
|
|
131
|
+
|
|
132
|
+
[gpt-sovits](https://github.com/RVC-Boss/GPT-SoVITS): melstyle encoder for voice clone
|
|
133
|
+
|
|
134
|
+
[coqui xtts](https://huggingface.co/spaces/coqui/xtts): gradio webui
|
|
135
|
+
|
|
136
|
+
Chinese Dirtionary Of DiffSinger: [Multi-langs_Dictionary](https://github.com/colstone/Multi-langs_Dictionary) and [atonyxu's fork](https://github.com/atonyxu/Multi-langs_Dictionary)
|
|
137
|
+
|
|
138
|
+
## TODO
|
|
139
|
+
|
|
140
|
+
- [x] Release pretrained models.
|
|
141
|
+
- [x] Support Japanese language.
|
|
142
|
+
- [x] User friendly preprocess and inference script.
|
|
143
|
+
- [x] Enhance documentation and citations.
|
|
144
|
+
- [x] Release multilingual checkpoint.
|
|
145
|
+
|
|
146
|
+
## Disclaimer
|
|
147
|
+
|
|
148
|
+
Any organization or individual is prohibited from using any technology in this repo to generate or edit someone's speech without his/her consent, including but not limited to government leaders, political figures, and celebrities. If you do not comply with this item, you could be in violation of copyright laws.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "MyanmarTTS"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "A from-scratch Burmese (Myanmar) text-to-speech model (CC0)."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
authors = [{ name = "freococo" }]
|
|
12
|
+
keywords = ["tts", "burmese", "myanmar", "speech", "text-to-speech"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Programming Language :: Python :: 3",
|
|
15
|
+
"Operating System :: OS Independent",
|
|
16
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
17
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
18
|
+
]
|
|
19
|
+
dependencies = [
|
|
20
|
+
"torch>=2.0",
|
|
21
|
+
"torchaudio>=2.0",
|
|
22
|
+
"huggingface_hub>=0.20",
|
|
23
|
+
"numpy<2.3",
|
|
24
|
+
"numba",
|
|
25
|
+
"scipy",
|
|
26
|
+
"soundfile",
|
|
27
|
+
"safetensors",
|
|
28
|
+
"torchdiffeq",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
Homepage = "https://huggingface.co/freococo/MyanmarTTS"
|
|
33
|
+
Repository = "https://huggingface.co/freococo/MyanmarTTS"
|
|
34
|
+
|
|
35
|
+
[tool.hatch.build.targets.wheel]
|
|
36
|
+
packages = ["src/myanmar_tts"]
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""MyanmarTTS - Burmese text-to-speech, from scratch, CC0."""
|
|
2
|
+
import os
|
|
3
|
+
from huggingface_hub import hf_hub_download
|
|
4
|
+
|
|
5
|
+
HF_REPO = "freococo/MyanmarTTS"
|
|
6
|
+
MODEL_FILE = "model_fp16.safetensors"
|
|
7
|
+
VOCODER_FILE = "vocos.pt"
|
|
8
|
+
REF_AUDIO = "samples/sample_0.wav"
|
|
9
|
+
|
|
10
|
+
__version__ = "1.0.0"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class MyanmarTTS:
|
|
14
|
+
def __init__(self, device="cuda"):
|
|
15
|
+
self.device = device
|
|
16
|
+
self._model = None
|
|
17
|
+
self._ref = None
|
|
18
|
+
|
|
19
|
+
def _load(self):
|
|
20
|
+
if self._model is not None:
|
|
21
|
+
return
|
|
22
|
+
print(f"Downloading model files from {HF_REPO}...")
|
|
23
|
+
model_path = hf_hub_download(repo_id=HF_REPO, filename=MODEL_FILE)
|
|
24
|
+
vocos_path = hf_hub_download(repo_id=HF_REPO, filename=VOCODER_FILE)
|
|
25
|
+
self._ref = hf_hub_download(repo_id=HF_REPO, filename=REF_AUDIO)
|
|
26
|
+
|
|
27
|
+
from .api import StableTTSAPI
|
|
28
|
+
print("Loading model...")
|
|
29
|
+
self._model = StableTTSAPI(model_path, vocos_path, "vocos")
|
|
30
|
+
self._model.to(self.device)
|
|
31
|
+
print("Ready.")
|
|
32
|
+
|
|
33
|
+
def tts(self, text):
|
|
34
|
+
"""Return a numpy float32 waveform at 44100 Hz."""
|
|
35
|
+
self._load()
|
|
36
|
+
audio, _ = self._model.inference(
|
|
37
|
+
text, self._ref, "burmese", step=32, solver="dopri5", cfg=3.0
|
|
38
|
+
)
|
|
39
|
+
return audio.squeeze(0).cpu().numpy()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
__all__ = ["MyanmarTTS"]
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
import torch
|
|
2
|
+
import torch.nn as nn
|
|
3
|
+
from dataclasses import asdict
|
|
4
|
+
|
|
5
|
+
from .utils.audio import LogMelSpectrogram
|
|
6
|
+
from .config import ModelConfig, MelConfig
|
|
7
|
+
from .models.model import StableTTS
|
|
8
|
+
|
|
9
|
+
from .text import symbols
|
|
10
|
+
from .text import cleaned_text_to_sequence
|
|
11
|
+
from .text.burmese import burmese_to_ipa2
|
|
12
|
+
from .datas.dataset import intersperse
|
|
13
|
+
from .utils.audio import load_and_resample_audio
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def get_vocoder(model_path, model_name='vocos'):
|
|
17
|
+
if model_name == 'vocos':
|
|
18
|
+
from .vocoders.vocos.models.model import Vocos
|
|
19
|
+
from .config import VocosConfig, MelConfig
|
|
20
|
+
vocoder = Vocos(VocosConfig(), MelConfig())
|
|
21
|
+
vocoder.load_state_dict(torch.load(model_path, weights_only=True, map_location='cpu'))
|
|
22
|
+
vocoder.eval()
|
|
23
|
+
else:
|
|
24
|
+
raise NotImplementedError(f"Unsupported vocoder: {model_name}")
|
|
25
|
+
return vocoder
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class StableTTSAPI(nn.Module):
|
|
29
|
+
def __init__(self, tts_model_path, vocoder_model_path, vocoder_name='vocos'):
|
|
30
|
+
super().__init__()
|
|
31
|
+
self.mel_config = MelConfig()
|
|
32
|
+
self.tts_model_config = ModelConfig()
|
|
33
|
+
|
|
34
|
+
self.mel_extractor = LogMelSpectrogram(**asdict(self.mel_config))
|
|
35
|
+
|
|
36
|
+
self.tts_model = StableTTS(len(symbols), self.mel_config.n_mels, **asdict(self.tts_model_config))
|
|
37
|
+
|
|
38
|
+
if tts_model_path.endswith(".safetensors"):
|
|
39
|
+
from safetensors.torch import load_file
|
|
40
|
+
state = load_file(tts_model_path)
|
|
41
|
+
else:
|
|
42
|
+
state = torch.load(tts_model_path, map_location='cpu', weights_only=True)
|
|
43
|
+
self.tts_model.load_state_dict(state)
|
|
44
|
+
self.tts_model.eval()
|
|
45
|
+
|
|
46
|
+
self.vocoder_model = get_vocoder(vocoder_model_path, vocoder_name)
|
|
47
|
+
self.vocoder_model.eval()
|
|
48
|
+
|
|
49
|
+
self.g2p_mapping = {
|
|
50
|
+
'burmese': burmese_to_ipa2,
|
|
51
|
+
}
|
|
52
|
+
self.supported_languages = self.g2p_mapping.keys()
|
|
53
|
+
|
|
54
|
+
@torch.inference_mode()
|
|
55
|
+
def inference(self, text, ref_audio, language, step, temperature=1.0,
|
|
56
|
+
length_scale=1.0, solver=None, cfg=3.0):
|
|
57
|
+
device = next(self.parameters()).device
|
|
58
|
+
phonemizer = self.g2p_mapping.get(language)
|
|
59
|
+
if phonemizer is None:
|
|
60
|
+
raise ValueError(f"Unsupported language: {language}")
|
|
61
|
+
|
|
62
|
+
text = phonemizer(text)
|
|
63
|
+
text = torch.tensor(
|
|
64
|
+
intersperse(cleaned_text_to_sequence(text), item=0),
|
|
65
|
+
dtype=torch.long, device=device
|
|
66
|
+
).unsqueeze(0)
|
|
67
|
+
text_length = torch.tensor([text.size(-1)], dtype=torch.long, device=device)
|
|
68
|
+
|
|
69
|
+
ref_audio = load_and_resample_audio(ref_audio, self.mel_config.sample_rate).to(device)
|
|
70
|
+
ref_audio = self.mel_extractor(ref_audio)
|
|
71
|
+
|
|
72
|
+
mel_output = self.tts_model.synthesise(
|
|
73
|
+
text, text_length, step, temperature, ref_audio,
|
|
74
|
+
length_scale, solver, cfg
|
|
75
|
+
)['decoder_outputs']
|
|
76
|
+
audio_output = self.vocoder_model(mel_output)
|
|
77
|
+
return audio_output.cpu(), mel_output.cpu()
|
|
78
|
+
|
|
79
|
+
def get_params(self):
|
|
80
|
+
tts_param = sum(p.numel() for p in self.tts_model.parameters()) / 1e6
|
|
81
|
+
vocoder_param = sum(p.numel() for p in self.vocoder_model.parameters()) / 1e6
|
|
82
|
+
return tts_param, vocoder_param
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Character-level text frontend for Burmese.
|
|
3
|
+
No real G2P: we clean the text and pass characters through directly.
|
|
4
|
+
Matches the interface of english_to_ipa2 / japanese_to_ipa2 (returns a list).
|
|
5
|
+
"""
|
|
6
|
+
import re
|
|
7
|
+
|
|
8
|
+
# Burmese Unicode ranges
|
|
9
|
+
_BURMESE_RANGES = (
|
|
10
|
+
(0x1000, 0x109F), # Myanmar
|
|
11
|
+
(0xAA60, 0xAA7F), # Myanmar Extended-A
|
|
12
|
+
(0xA9E0, 0xA9FF), # Myanmar Extended-B
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
# Allow these through the cleaner; everything else gets dropped
|
|
16
|
+
_ALLOWED_ASCII_PUNCT = set(".,!?'\"-:;()")
|
|
17
|
+
_ALLOWED_BURMESE_PUNCT = set("\u104A\u104B\u104C\u104D\u104E\u104F")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _is_burmese(ch):
|
|
21
|
+
o = ord(ch)
|
|
22
|
+
return any(lo <= o <= hi for lo, hi in _BURMESE_RANGES)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def clean_burmese(text):
|
|
26
|
+
"""Keep only Burmese chars, spaces, and whitelisted punctuation."""
|
|
27
|
+
out = []
|
|
28
|
+
for ch in text:
|
|
29
|
+
if ch in (" ", "\t", "\n", "\r"):
|
|
30
|
+
out.append(" ")
|
|
31
|
+
elif _is_burmese(ch):
|
|
32
|
+
out.append(ch)
|
|
33
|
+
elif ch in _ALLOWED_ASCII_PUNCT or ch in _ALLOWED_BURMESE_PUNCT:
|
|
34
|
+
out.append(ch)
|
|
35
|
+
# else: drop silently
|
|
36
|
+
return re.sub(r"\s+", " ", "".join(out)).strip()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def burmese_to_ipa2(text):
|
|
40
|
+
"""Return a list of character tokens (matches the IPA-G2P interface)."""
|
|
41
|
+
cleaned = clean_burmese(text)
|
|
42
|
+
return list(cleaned)
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
|
|
3
|
+
@dataclass
|
|
4
|
+
class MelConfig:
|
|
5
|
+
sample_rate: int = 44100
|
|
6
|
+
n_fft: int = 2048
|
|
7
|
+
win_length: int = 2048
|
|
8
|
+
hop_length: int = 512
|
|
9
|
+
f_min: float = 0.0
|
|
10
|
+
f_max: float = None
|
|
11
|
+
pad: int = 0
|
|
12
|
+
n_mels: int = 128
|
|
13
|
+
center: bool = False
|
|
14
|
+
pad_mode: str = "reflect"
|
|
15
|
+
mel_scale: str = "slaney"
|
|
16
|
+
|
|
17
|
+
def __post_init__(self):
|
|
18
|
+
if self.pad == 0:
|
|
19
|
+
self.pad = (self.n_fft - self.hop_length) // 2
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class ModelConfig:
|
|
23
|
+
hidden_channels: int = 256
|
|
24
|
+
filter_channels: int = 1024
|
|
25
|
+
n_heads: int = 4
|
|
26
|
+
n_enc_layers: int = 3
|
|
27
|
+
n_dec_layers: int = 6
|
|
28
|
+
kernel_size: int = 3
|
|
29
|
+
p_dropout: int = 0.1
|
|
30
|
+
gin_channels: int = 256
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class TrainConfig:
|
|
34
|
+
train_dataset_path: str = 'filelists/filelist.json'
|
|
35
|
+
test_dataset_path: str = 'filelists/filelist.json' # not used
|
|
36
|
+
batch_size: int = 32
|
|
37
|
+
learning_rate: float = 1e-4
|
|
38
|
+
num_epochs: int = 50
|
|
39
|
+
model_save_path: str = './checkpoints'
|
|
40
|
+
log_dir: str = './runs'
|
|
41
|
+
log_interval: int = 5
|
|
42
|
+
save_interval: int = 1
|
|
43
|
+
warmup_steps: int = 200
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class VocosConfig:
|
|
47
|
+
input_channels: int = 128
|
|
48
|
+
dim: int = 512
|
|
49
|
+
intermediate_dim: int = 1536
|
|
50
|
+
num_layers: int = 8
|
|
File without changes
|