MyanmarTTS 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. myanmartts-1.0.0/LICENSE +21 -0
  2. myanmartts-1.0.0/PKG-INFO +173 -0
  3. myanmartts-1.0.0/README.md +148 -0
  4. myanmartts-1.0.0/pyproject.toml +36 -0
  5. myanmartts-1.0.0/src/myanmar_tts/__init__.py +42 -0
  6. myanmartts-1.0.0/src/myanmar_tts/api.py +82 -0
  7. myanmartts-1.0.0/src/myanmar_tts/burmese.py +42 -0
  8. myanmartts-1.0.0/src/myanmar_tts/config.py +50 -0
  9. myanmartts-1.0.0/src/myanmar_tts/datas/__init__.py +0 -0
  10. myanmartts-1.0.0/src/myanmar_tts/datas/all_sources_dataset.py +160 -0
  11. myanmartts-1.0.0/src/myanmar_tts/datas/collate_wav.py +39 -0
  12. myanmartts-1.0.0/src/myanmar_tts/datas/dataset.py +69 -0
  13. myanmartts-1.0.0/src/myanmar_tts/datas/local_shards_dataset.py +73 -0
  14. myanmartts-1.0.0/src/myanmar_tts/models/__init__.py +0 -0
  15. myanmartts-1.0.0/src/myanmar_tts/models/diffusion_transformer.py +205 -0
  16. myanmartts-1.0.0/src/myanmar_tts/models/duration_predictor.py +40 -0
  17. myanmartts-1.0.0/src/myanmar_tts/models/estimator.py +138 -0
  18. myanmartts-1.0.0/src/myanmar_tts/models/flow_matching.py +100 -0
  19. myanmartts-1.0.0/src/myanmar_tts/models/model.py +178 -0
  20. myanmartts-1.0.0/src/myanmar_tts/models/reference_encoder.py +168 -0
  21. myanmartts-1.0.0/src/myanmar_tts/models/text_encoder.py +44 -0
  22. myanmartts-1.0.0/src/myanmar_tts/monotonic_align/__init__.py +16 -0
  23. myanmartts-1.0.0/src/myanmar_tts/monotonic_align/core.py +46 -0
  24. myanmartts-1.0.0/src/myanmar_tts/symbols.py +57 -0
  25. myanmartts-1.0.0/src/myanmar_tts/text/LICENSE +19 -0
  26. myanmartts-1.0.0/src/myanmar_tts/text/__init__.py +16 -0
  27. myanmartts-1.0.0/src/myanmar_tts/text/burmese.py +42 -0
  28. myanmartts-1.0.0/src/myanmar_tts/text/cleaners.py +10 -0
  29. myanmartts-1.0.0/src/myanmar_tts/text/symbols.py +57 -0
  30. myanmartts-1.0.0/src/myanmar_tts/utils/__init__.py +0 -0
  31. myanmartts-1.0.0/src/myanmar_tts/utils/audio.py +74 -0
  32. myanmartts-1.0.0/src/myanmar_tts/utils/load.py +43 -0
  33. myanmartts-1.0.0/src/myanmar_tts/utils/mask.py +8 -0
  34. myanmartts-1.0.0/src/myanmar_tts/utils/scheduler.py +428 -0
  35. myanmartts-1.0.0/src/myanmar_tts/vocoders/__init__.py +0 -0
  36. myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/__init__.py +0 -0
  37. myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/backbone.py +214 -0
  38. myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/head.py +257 -0
  39. myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/model.py +57 -0
  40. myanmartts-1.0.0/src/myanmar_tts/vocoders/ffgan/unify.py +60 -0
  41. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/README.md +41 -0
  42. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/__init__.py +0 -0
  43. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/config.py +41 -0
  44. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/dataset.py +57 -0
  45. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/inference.ipynb +79 -0
  46. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/__init__.py +0 -0
  47. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/backbone.py +57 -0
  48. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/discriminator.py +171 -0
  49. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/head.py +118 -0
  50. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/loss.py +66 -0
  51. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/model.py +20 -0
  52. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/models/module.py +47 -0
  53. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/preprocess.py +45 -0
  54. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/requirements.txt +2 -0
  55. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/train.py +165 -0
  56. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/__init__.py +0 -0
  57. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/audio.py +74 -0
  58. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/load.py +53 -0
  59. myanmartts-1.0.0/src/myanmar_tts/vocoders/vocos/utils/scheduler.py +298 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 KdaiP
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,173 @@
1
+ Metadata-Version: 2.5
2
+ Name: MyanmarTTS
3
+ Version: 1.0.0
4
+ Summary: A from-scratch Burmese (Myanmar) text-to-speech model (CC0).
5
+ Project-URL: Homepage, https://huggingface.co/freococo/MyanmarTTS
6
+ Project-URL: Repository, https://huggingface.co/freococo/MyanmarTTS
7
+ Author: freococo
8
+ License-File: LICENSE
9
+ Keywords: burmese,myanmar,speech,text-to-speech,tts
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
13
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
14
+ Requires-Python: >=3.9
15
+ Requires-Dist: huggingface-hub>=0.20
16
+ Requires-Dist: numba
17
+ Requires-Dist: numpy<2.3
18
+ Requires-Dist: safetensors
19
+ Requires-Dist: scipy
20
+ Requires-Dist: soundfile
21
+ Requires-Dist: torch>=2.0
22
+ Requires-Dist: torchaudio>=2.0
23
+ Requires-Dist: torchdiffeq
24
+ Description-Content-Type: text/markdown
25
+
26
+ <div align="center">
27
+
28
+ # StableTTS
29
+
30
+ Next-generation TTS model using flow-matching and DiT, inspired by [Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3).
31
+
32
+
33
+ </div>
34
+
35
+ ## Introduction
36
+
37
+ As the first open-source TTS model that tried to combine flow-matching and DiT, **StableTTS** is a fast and lightweight TTS model for chinese, english and japanese speech generation. It has 31M parameters.
38
+
39
+ ✨ **Huggingface demo:** [🤗](https://huggingface.co/spaces/KdaiP/StableTTS1.1)
40
+
41
+ ## News
42
+
43
+ 2024/10: A new autoregressive TTS model is coming soon...
44
+
45
+ 2024/9: 🚀 **StableTTS V1.1 Released** ⭐ Audio quality is largely improved ⭐
46
+
47
+ ⭐ **V1.1 Release Highlights:**
48
+
49
+ - Fixed critical issues that cause the audio quality being much lower than expected. (Mainly in Mel spectrogram and Attention mask)
50
+ - Introduced U-Net-like long skip connections to the DiT in the Flow-matching Decoder.
51
+ - Use cosine timestep scheduler from [Cosyvoice](https://github.com/FunAudioLLM/CosyVoice)
52
+ - Add support for CFG (Classifier-Free Guidance).
53
+ - Add support for [FireflyGAN vocoder](https://github.com/fishaudio/vocoder/releases/tag/1.0.0).
54
+ - Switched to [torchdiffeq](https://github.com/rtqichen/torchdiffeq) for ODE solvers.
55
+ - Improved Chinese text frontend (partially based on [gpt-sovits2](https://github.com/RVC-Boss/GPT-SoVITS)).
56
+ - Multilingual support (Chinese, English, Japanese) in a single checkpoint.
57
+ - Increased parameters: 10M -> 31M.
58
+
59
+
60
+ ## Pretrained models
61
+
62
+ ### Text-To-Mel model
63
+
64
+ Download and place the model in the `./checkpoints` directory, it is ready for inference, finetuning and webui.
65
+
66
+ | Model Name | Task Details | Dataset | Download Link |
67
+ |:----------:|:------------:|:-------------:|:-------------:|
68
+ | StableTTS | text to mel | 600 hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/StableTTS/checkpoint_0.pt)|
69
+
70
+ ### Mel-To-Wav model
71
+
72
+ Choose a vocoder (`vocos` or `firefly-gan` ) and place it in the `./vocoders/pretrained` directory.
73
+
74
+ | Model Name | Task Details | Dataset | Download Link |
75
+ |:----------:|:------------:|:-------------:|:-------------:|
76
+ | Vocos | mel to wav | 2k hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/vocoders/vocos.pt)|
77
+ | firefly-gan-base | mel to wav | HiFi-16kh | [download from fishaudio](https://github.com/fishaudio/vocoder/releases/download/1.0.0/firefly-gan-base-generator.ckpt)|
78
+
79
+ ## Installation
80
+
81
+ 1. **Install pytorch**: Follow the [official PyTorch guide](https://pytorch.org/get-started/locally/) to install pytorch and torchaudio. We recommend the latest version (tested with PyTorch 2.4 and Python 3.12).
82
+
83
+ 2. **Install Dependencies**: Run the following command to install the required Python packages:
84
+
85
+ ```bash
86
+ pip install -r requirements.txt
87
+ ```
88
+
89
+ ## Inference
90
+
91
+ For detailed inference instructions, please refer to `inference.ipynb`
92
+
93
+ We also provide a webui based on gradio, please refer to `webui.py`
94
+
95
+ ## Training
96
+
97
+ StableTTS is designed to be trained easily. We only need text and audio pairs, without any speaker id or extra feature extraction. Here’s how to get started:
98
+
99
+ ### Preparing Your Data
100
+
101
+ 1. **Generate Text and Audio pairs**: Generate the text and audio pair filelist as `./filelists/example.txt`. Some recipes of open-source datasets could be found in `./recipes`.
102
+
103
+ 2. **Run Preprocessing**: Adjust the `DataConfig` in `preprocess.py` to set your input and output paths, then run the script. This will process the audio and text according to your list, outputting a JSON file with paths to mel features and phonemes.
104
+
105
+ **Note: Process multilingual data separately by changing the `language` setting in `DataConfig`**
106
+
107
+ ### Start training
108
+
109
+ 1. **Adjust Training Configuration**: In `config.py`, modify `TrainConfig` to set your file list path and adjust training parameters (such as batch_size) as needed.
110
+
111
+ 2. **Start the Training Process**: Launch `train.py` to start training your model.
112
+
113
+ Note: For finetuning, download the pretrained model and place it in the `model_save_path` directory specified in `TrainConfig`. Training script will automatically detect and load the pretrained checkpoint.
114
+
115
+ ### (Optional) Vocoder training
116
+
117
+ The `./vocoder/vocos` folder contains the training and finetuning codes for vocos vocoder.
118
+
119
+ For other types of vocoders, we recommend to train by using [fishaudio vocoder](https://github.com/fishaudio/vocoder): an uniform interface for developing various vocoders. We use the same spectrogram transform so the vocoders trained is compatible with StableTTS.
120
+
121
+ ## Model structure
122
+
123
+ <div align="center">
124
+
125
+ <p style="text-align: center;">
126
+ <img src="./figures/structure.jpg" height="512"/>
127
+ </p>
128
+
129
+ </div>
130
+
131
+ - We use the Diffusion Convolution Transformer block from [Hierspeech++](https://github.com/sh-lee-prml/HierSpeechpp), which is a combination of original [DiT](https://github.com/sh-lee-prml/HierSpeechpp) and [FFT](https://arxiv.org/pdf/1905.09263.pdf)(Feed forward Transformer from fastspeech) for better prosody.
132
+
133
+ - In flow-matching decoder, we add a [FiLM layer](https://arxiv.org/abs/1709.07871) before DiT block to condition timestep embedding into model.
134
+
135
+ ## References
136
+
137
+ The development of our models heavily relies on insights and code from various projects. We express our heartfelt thanks to the creators of the following:
138
+
139
+ ### Direct Inspirations
140
+
141
+ [Matcha TTS](https://github.com/shivammehta25/Matcha-TTS): Essential flow-matching code.
142
+
143
+ [Grad TTS](https://github.com/huawei-noah/Speech-Backbones/tree/main/Grad-TTS): Diffusion model structure.
144
+
145
+ [Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3): Idea of combining flow-matching and DiT.
146
+
147
+ [Vits](https://github.com/jaywalnut310/vits): Code style and MAS insights, DistributedBucketSampler.
148
+
149
+ ### Additional References:
150
+
151
+ [plowtts-pytorch](https://github.com/p0p4k/pflowtts_pytorch): codes of MAS in training
152
+
153
+ [Bert-VITS2](https://github.com/Plachtaa/VITS-fast-fine-tuning) : numba version of MAS and modern pytorch codes of Vits
154
+
155
+ [fish-speech](https://github.com/fishaudio/fish-speech): dataclass usage and mel-spectrogram transforms using torchaudio, gradio webui
156
+
157
+ [gpt-sovits](https://github.com/RVC-Boss/GPT-SoVITS): melstyle encoder for voice clone
158
+
159
+ [coqui xtts](https://huggingface.co/spaces/coqui/xtts): gradio webui
160
+
161
+ Chinese Dirtionary Of DiffSinger: [Multi-langs_Dictionary](https://github.com/colstone/Multi-langs_Dictionary) and [atonyxu's fork](https://github.com/atonyxu/Multi-langs_Dictionary)
162
+
163
+ ## TODO
164
+
165
+ - [x] Release pretrained models.
166
+ - [x] Support Japanese language.
167
+ - [x] User friendly preprocess and inference script.
168
+ - [x] Enhance documentation and citations.
169
+ - [x] Release multilingual checkpoint.
170
+
171
+ ## Disclaimer
172
+
173
+ Any organization or individual is prohibited from using any technology in this repo to generate or edit someone's speech without his/her consent, including but not limited to government leaders, political figures, and celebrities. If you do not comply with this item, you could be in violation of copyright laws.
@@ -0,0 +1,148 @@
1
+ <div align="center">
2
+
3
+ # StableTTS
4
+
5
+ Next-generation TTS model using flow-matching and DiT, inspired by [Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3).
6
+
7
+
8
+ </div>
9
+
10
+ ## Introduction
11
+
12
+ As the first open-source TTS model that tried to combine flow-matching and DiT, **StableTTS** is a fast and lightweight TTS model for chinese, english and japanese speech generation. It has 31M parameters.
13
+
14
+ ✨ **Huggingface demo:** [🤗](https://huggingface.co/spaces/KdaiP/StableTTS1.1)
15
+
16
+ ## News
17
+
18
+ 2024/10: A new autoregressive TTS model is coming soon...
19
+
20
+ 2024/9: 🚀 **StableTTS V1.1 Released** ⭐ Audio quality is largely improved ⭐
21
+
22
+ ⭐ **V1.1 Release Highlights:**
23
+
24
+ - Fixed critical issues that cause the audio quality being much lower than expected. (Mainly in Mel spectrogram and Attention mask)
25
+ - Introduced U-Net-like long skip connections to the DiT in the Flow-matching Decoder.
26
+ - Use cosine timestep scheduler from [Cosyvoice](https://github.com/FunAudioLLM/CosyVoice)
27
+ - Add support for CFG (Classifier-Free Guidance).
28
+ - Add support for [FireflyGAN vocoder](https://github.com/fishaudio/vocoder/releases/tag/1.0.0).
29
+ - Switched to [torchdiffeq](https://github.com/rtqichen/torchdiffeq) for ODE solvers.
30
+ - Improved Chinese text frontend (partially based on [gpt-sovits2](https://github.com/RVC-Boss/GPT-SoVITS)).
31
+ - Multilingual support (Chinese, English, Japanese) in a single checkpoint.
32
+ - Increased parameters: 10M -> 31M.
33
+
34
+
35
+ ## Pretrained models
36
+
37
+ ### Text-To-Mel model
38
+
39
+ Download and place the model in the `./checkpoints` directory, it is ready for inference, finetuning and webui.
40
+
41
+ | Model Name | Task Details | Dataset | Download Link |
42
+ |:----------:|:------------:|:-------------:|:-------------:|
43
+ | StableTTS | text to mel | 600 hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/StableTTS/checkpoint_0.pt)|
44
+
45
+ ### Mel-To-Wav model
46
+
47
+ Choose a vocoder (`vocos` or `firefly-gan` ) and place it in the `./vocoders/pretrained` directory.
48
+
49
+ | Model Name | Task Details | Dataset | Download Link |
50
+ |:----------:|:------------:|:-------------:|:-------------:|
51
+ | Vocos | mel to wav | 2k hours | [🤗](https://huggingface.co/KdaiP/StableTTS1.1/resolve/main/vocoders/vocos.pt)|
52
+ | firefly-gan-base | mel to wav | HiFi-16kh | [download from fishaudio](https://github.com/fishaudio/vocoder/releases/download/1.0.0/firefly-gan-base-generator.ckpt)|
53
+
54
+ ## Installation
55
+
56
+ 1. **Install pytorch**: Follow the [official PyTorch guide](https://pytorch.org/get-started/locally/) to install pytorch and torchaudio. We recommend the latest version (tested with PyTorch 2.4 and Python 3.12).
57
+
58
+ 2. **Install Dependencies**: Run the following command to install the required Python packages:
59
+
60
+ ```bash
61
+ pip install -r requirements.txt
62
+ ```
63
+
64
+ ## Inference
65
+
66
+ For detailed inference instructions, please refer to `inference.ipynb`
67
+
68
+ We also provide a webui based on gradio, please refer to `webui.py`
69
+
70
+ ## Training
71
+
72
+ StableTTS is designed to be trained easily. We only need text and audio pairs, without any speaker id or extra feature extraction. Here’s how to get started:
73
+
74
+ ### Preparing Your Data
75
+
76
+ 1. **Generate Text and Audio pairs**: Generate the text and audio pair filelist as `./filelists/example.txt`. Some recipes of open-source datasets could be found in `./recipes`.
77
+
78
+ 2. **Run Preprocessing**: Adjust the `DataConfig` in `preprocess.py` to set your input and output paths, then run the script. This will process the audio and text according to your list, outputting a JSON file with paths to mel features and phonemes.
79
+
80
+ **Note: Process multilingual data separately by changing the `language` setting in `DataConfig`**
81
+
82
+ ### Start training
83
+
84
+ 1. **Adjust Training Configuration**: In `config.py`, modify `TrainConfig` to set your file list path and adjust training parameters (such as batch_size) as needed.
85
+
86
+ 2. **Start the Training Process**: Launch `train.py` to start training your model.
87
+
88
+ Note: For finetuning, download the pretrained model and place it in the `model_save_path` directory specified in `TrainConfig`. Training script will automatically detect and load the pretrained checkpoint.
89
+
90
+ ### (Optional) Vocoder training
91
+
92
+ The `./vocoder/vocos` folder contains the training and finetuning codes for vocos vocoder.
93
+
94
+ For other types of vocoders, we recommend to train by using [fishaudio vocoder](https://github.com/fishaudio/vocoder): an uniform interface for developing various vocoders. We use the same spectrogram transform so the vocoders trained is compatible with StableTTS.
95
+
96
+ ## Model structure
97
+
98
+ <div align="center">
99
+
100
+ <p style="text-align: center;">
101
+ <img src="./figures/structure.jpg" height="512"/>
102
+ </p>
103
+
104
+ </div>
105
+
106
+ - We use the Diffusion Convolution Transformer block from [Hierspeech++](https://github.com/sh-lee-prml/HierSpeechpp), which is a combination of original [DiT](https://github.com/sh-lee-prml/HierSpeechpp) and [FFT](https://arxiv.org/pdf/1905.09263.pdf)(Feed forward Transformer from fastspeech) for better prosody.
107
+
108
+ - In flow-matching decoder, we add a [FiLM layer](https://arxiv.org/abs/1709.07871) before DiT block to condition timestep embedding into model.
109
+
110
+ ## References
111
+
112
+ The development of our models heavily relies on insights and code from various projects. We express our heartfelt thanks to the creators of the following:
113
+
114
+ ### Direct Inspirations
115
+
116
+ [Matcha TTS](https://github.com/shivammehta25/Matcha-TTS): Essential flow-matching code.
117
+
118
+ [Grad TTS](https://github.com/huawei-noah/Speech-Backbones/tree/main/Grad-TTS): Diffusion model structure.
119
+
120
+ [Stable Diffusion 3](https://stability.ai/news/stable-diffusion-3): Idea of combining flow-matching and DiT.
121
+
122
+ [Vits](https://github.com/jaywalnut310/vits): Code style and MAS insights, DistributedBucketSampler.
123
+
124
+ ### Additional References:
125
+
126
+ [plowtts-pytorch](https://github.com/p0p4k/pflowtts_pytorch): codes of MAS in training
127
+
128
+ [Bert-VITS2](https://github.com/Plachtaa/VITS-fast-fine-tuning) : numba version of MAS and modern pytorch codes of Vits
129
+
130
+ [fish-speech](https://github.com/fishaudio/fish-speech): dataclass usage and mel-spectrogram transforms using torchaudio, gradio webui
131
+
132
+ [gpt-sovits](https://github.com/RVC-Boss/GPT-SoVITS): melstyle encoder for voice clone
133
+
134
+ [coqui xtts](https://huggingface.co/spaces/coqui/xtts): gradio webui
135
+
136
+ Chinese Dirtionary Of DiffSinger: [Multi-langs_Dictionary](https://github.com/colstone/Multi-langs_Dictionary) and [atonyxu's fork](https://github.com/atonyxu/Multi-langs_Dictionary)
137
+
138
+ ## TODO
139
+
140
+ - [x] Release pretrained models.
141
+ - [x] Support Japanese language.
142
+ - [x] User friendly preprocess and inference script.
143
+ - [x] Enhance documentation and citations.
144
+ - [x] Release multilingual checkpoint.
145
+
146
+ ## Disclaimer
147
+
148
+ Any organization or individual is prohibited from using any technology in this repo to generate or edit someone's speech without his/her consent, including but not limited to government leaders, political figures, and celebrities. If you do not comply with this item, you could be in violation of copyright laws.
@@ -0,0 +1,36 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "MyanmarTTS"
7
+ version = "1.0.0"
8
+ description = "A from-scratch Burmese (Myanmar) text-to-speech model (CC0)."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ authors = [{ name = "freococo" }]
12
+ keywords = ["tts", "burmese", "myanmar", "speech", "text-to-speech"]
13
+ classifiers = [
14
+ "Programming Language :: Python :: 3",
15
+ "Operating System :: OS Independent",
16
+ "Topic :: Multimedia :: Sound/Audio :: Speech",
17
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
18
+ ]
19
+ dependencies = [
20
+ "torch>=2.0",
21
+ "torchaudio>=2.0",
22
+ "huggingface_hub>=0.20",
23
+ "numpy<2.3",
24
+ "numba",
25
+ "scipy",
26
+ "soundfile",
27
+ "safetensors",
28
+ "torchdiffeq",
29
+ ]
30
+
31
+ [project.urls]
32
+ Homepage = "https://huggingface.co/freococo/MyanmarTTS"
33
+ Repository = "https://huggingface.co/freococo/MyanmarTTS"
34
+
35
+ [tool.hatch.build.targets.wheel]
36
+ packages = ["src/myanmar_tts"]
@@ -0,0 +1,42 @@
1
+ """MyanmarTTS - Burmese text-to-speech, from scratch, CC0."""
2
+ import os
3
+ from huggingface_hub import hf_hub_download
4
+
5
+ HF_REPO = "freococo/MyanmarTTS"
6
+ MODEL_FILE = "model_fp16.safetensors"
7
+ VOCODER_FILE = "vocos.pt"
8
+ REF_AUDIO = "samples/sample_0.wav"
9
+
10
+ __version__ = "1.0.0"
11
+
12
+
13
+ class MyanmarTTS:
14
+ def __init__(self, device="cuda"):
15
+ self.device = device
16
+ self._model = None
17
+ self._ref = None
18
+
19
+ def _load(self):
20
+ if self._model is not None:
21
+ return
22
+ print(f"Downloading model files from {HF_REPO}...")
23
+ model_path = hf_hub_download(repo_id=HF_REPO, filename=MODEL_FILE)
24
+ vocos_path = hf_hub_download(repo_id=HF_REPO, filename=VOCODER_FILE)
25
+ self._ref = hf_hub_download(repo_id=HF_REPO, filename=REF_AUDIO)
26
+
27
+ from .api import StableTTSAPI
28
+ print("Loading model...")
29
+ self._model = StableTTSAPI(model_path, vocos_path, "vocos")
30
+ self._model.to(self.device)
31
+ print("Ready.")
32
+
33
+ def tts(self, text):
34
+ """Return a numpy float32 waveform at 44100 Hz."""
35
+ self._load()
36
+ audio, _ = self._model.inference(
37
+ text, self._ref, "burmese", step=32, solver="dopri5", cfg=3.0
38
+ )
39
+ return audio.squeeze(0).cpu().numpy()
40
+
41
+
42
+ __all__ = ["MyanmarTTS"]
@@ -0,0 +1,82 @@
1
+ import torch
2
+ import torch.nn as nn
3
+ from dataclasses import asdict
4
+
5
+ from .utils.audio import LogMelSpectrogram
6
+ from .config import ModelConfig, MelConfig
7
+ from .models.model import StableTTS
8
+
9
+ from .text import symbols
10
+ from .text import cleaned_text_to_sequence
11
+ from .text.burmese import burmese_to_ipa2
12
+ from .datas.dataset import intersperse
13
+ from .utils.audio import load_and_resample_audio
14
+
15
+
16
+ def get_vocoder(model_path, model_name='vocos'):
17
+ if model_name == 'vocos':
18
+ from .vocoders.vocos.models.model import Vocos
19
+ from .config import VocosConfig, MelConfig
20
+ vocoder = Vocos(VocosConfig(), MelConfig())
21
+ vocoder.load_state_dict(torch.load(model_path, weights_only=True, map_location='cpu'))
22
+ vocoder.eval()
23
+ else:
24
+ raise NotImplementedError(f"Unsupported vocoder: {model_name}")
25
+ return vocoder
26
+
27
+
28
+ class StableTTSAPI(nn.Module):
29
+ def __init__(self, tts_model_path, vocoder_model_path, vocoder_name='vocos'):
30
+ super().__init__()
31
+ self.mel_config = MelConfig()
32
+ self.tts_model_config = ModelConfig()
33
+
34
+ self.mel_extractor = LogMelSpectrogram(**asdict(self.mel_config))
35
+
36
+ self.tts_model = StableTTS(len(symbols), self.mel_config.n_mels, **asdict(self.tts_model_config))
37
+
38
+ if tts_model_path.endswith(".safetensors"):
39
+ from safetensors.torch import load_file
40
+ state = load_file(tts_model_path)
41
+ else:
42
+ state = torch.load(tts_model_path, map_location='cpu', weights_only=True)
43
+ self.tts_model.load_state_dict(state)
44
+ self.tts_model.eval()
45
+
46
+ self.vocoder_model = get_vocoder(vocoder_model_path, vocoder_name)
47
+ self.vocoder_model.eval()
48
+
49
+ self.g2p_mapping = {
50
+ 'burmese': burmese_to_ipa2,
51
+ }
52
+ self.supported_languages = self.g2p_mapping.keys()
53
+
54
+ @torch.inference_mode()
55
+ def inference(self, text, ref_audio, language, step, temperature=1.0,
56
+ length_scale=1.0, solver=None, cfg=3.0):
57
+ device = next(self.parameters()).device
58
+ phonemizer = self.g2p_mapping.get(language)
59
+ if phonemizer is None:
60
+ raise ValueError(f"Unsupported language: {language}")
61
+
62
+ text = phonemizer(text)
63
+ text = torch.tensor(
64
+ intersperse(cleaned_text_to_sequence(text), item=0),
65
+ dtype=torch.long, device=device
66
+ ).unsqueeze(0)
67
+ text_length = torch.tensor([text.size(-1)], dtype=torch.long, device=device)
68
+
69
+ ref_audio = load_and_resample_audio(ref_audio, self.mel_config.sample_rate).to(device)
70
+ ref_audio = self.mel_extractor(ref_audio)
71
+
72
+ mel_output = self.tts_model.synthesise(
73
+ text, text_length, step, temperature, ref_audio,
74
+ length_scale, solver, cfg
75
+ )['decoder_outputs']
76
+ audio_output = self.vocoder_model(mel_output)
77
+ return audio_output.cpu(), mel_output.cpu()
78
+
79
+ def get_params(self):
80
+ tts_param = sum(p.numel() for p in self.tts_model.parameters()) / 1e6
81
+ vocoder_param = sum(p.numel() for p in self.vocoder_model.parameters()) / 1e6
82
+ return tts_param, vocoder_param
@@ -0,0 +1,42 @@
1
+ """
2
+ Character-level text frontend for Burmese.
3
+ No real G2P: we clean the text and pass characters through directly.
4
+ Matches the interface of english_to_ipa2 / japanese_to_ipa2 (returns a list).
5
+ """
6
+ import re
7
+
8
+ # Burmese Unicode ranges
9
+ _BURMESE_RANGES = (
10
+ (0x1000, 0x109F), # Myanmar
11
+ (0xAA60, 0xAA7F), # Myanmar Extended-A
12
+ (0xA9E0, 0xA9FF), # Myanmar Extended-B
13
+ )
14
+
15
+ # Allow these through the cleaner; everything else gets dropped
16
+ _ALLOWED_ASCII_PUNCT = set(".,!?'\"-:;()")
17
+ _ALLOWED_BURMESE_PUNCT = set("\u104A\u104B\u104C\u104D\u104E\u104F")
18
+
19
+
20
+ def _is_burmese(ch):
21
+ o = ord(ch)
22
+ return any(lo <= o <= hi for lo, hi in _BURMESE_RANGES)
23
+
24
+
25
+ def clean_burmese(text):
26
+ """Keep only Burmese chars, spaces, and whitelisted punctuation."""
27
+ out = []
28
+ for ch in text:
29
+ if ch in (" ", "\t", "\n", "\r"):
30
+ out.append(" ")
31
+ elif _is_burmese(ch):
32
+ out.append(ch)
33
+ elif ch in _ALLOWED_ASCII_PUNCT or ch in _ALLOWED_BURMESE_PUNCT:
34
+ out.append(ch)
35
+ # else: drop silently
36
+ return re.sub(r"\s+", " ", "".join(out)).strip()
37
+
38
+
39
+ def burmese_to_ipa2(text):
40
+ """Return a list of character tokens (matches the IPA-G2P interface)."""
41
+ cleaned = clean_burmese(text)
42
+ return list(cleaned)
@@ -0,0 +1,50 @@
1
+ from dataclasses import dataclass
2
+
3
+ @dataclass
4
+ class MelConfig:
5
+ sample_rate: int = 44100
6
+ n_fft: int = 2048
7
+ win_length: int = 2048
8
+ hop_length: int = 512
9
+ f_min: float = 0.0
10
+ f_max: float = None
11
+ pad: int = 0
12
+ n_mels: int = 128
13
+ center: bool = False
14
+ pad_mode: str = "reflect"
15
+ mel_scale: str = "slaney"
16
+
17
+ def __post_init__(self):
18
+ if self.pad == 0:
19
+ self.pad = (self.n_fft - self.hop_length) // 2
20
+
21
+ @dataclass
22
+ class ModelConfig:
23
+ hidden_channels: int = 256
24
+ filter_channels: int = 1024
25
+ n_heads: int = 4
26
+ n_enc_layers: int = 3
27
+ n_dec_layers: int = 6
28
+ kernel_size: int = 3
29
+ p_dropout: int = 0.1
30
+ gin_channels: int = 256
31
+
32
+ @dataclass
33
+ class TrainConfig:
34
+ train_dataset_path: str = 'filelists/filelist.json'
35
+ test_dataset_path: str = 'filelists/filelist.json' # not used
36
+ batch_size: int = 32
37
+ learning_rate: float = 1e-4
38
+ num_epochs: int = 50
39
+ model_save_path: str = './checkpoints'
40
+ log_dir: str = './runs'
41
+ log_interval: int = 5
42
+ save_interval: int = 1
43
+ warmup_steps: int = 200
44
+
45
+ @dataclass
46
+ class VocosConfig:
47
+ input_channels: int = 128
48
+ dim: int = 512
49
+ intermediate_dim: int = 1536
50
+ num_layers: int = 8
File without changes