melotts-mblt 0.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- melotts_mblt-0.0.0/LICENSE +28 -0
- melotts_mblt-0.0.0/MANIFEST.in +3 -0
- melotts_mblt-0.0.0/PKG-INFO +160 -0
- melotts_mblt-0.0.0/README.md +117 -0
- melotts_mblt-0.0.0/melotts_mblt/LICENSE +19 -0
- melotts_mblt-0.0.0/melotts_mblt/__init__.py +26 -0
- melotts_mblt-0.0.0/melotts_mblt/api.py +310 -0
- melotts_mblt-0.0.0/melotts_mblt/app.py +98 -0
- melotts_mblt-0.0.0/melotts_mblt/cli/__init__.py +3 -0
- melotts_mblt-0.0.0/melotts_mblt/cli/__main__.py +4 -0
- melotts_mblt-0.0.0/melotts_mblt/cli/_click.py +37 -0
- melotts_mblt-0.0.0/melotts_mblt/cli/download.py +80 -0
- melotts_mblt-0.0.0/melotts_mblt/cli/main.py +50 -0
- melotts_mblt-0.0.0/melotts_mblt/cli/tts.py +58 -0
- melotts_mblt-0.0.0/melotts_mblt/cli/ui.py +70 -0
- melotts_mblt-0.0.0/melotts_mblt/commons.py +136 -0
- melotts_mblt-0.0.0/melotts_mblt/download_utils.py +54 -0
- melotts_mblt-0.0.0/melotts_mblt/main.py +50 -0
- melotts_mblt-0.0.0/melotts_mblt/models.py +476 -0
- melotts_mblt-0.0.0/melotts_mblt/py.typed +0 -0
- melotts_mblt-0.0.0/melotts_mblt/split_utils.py +175 -0
- melotts_mblt-0.0.0/melotts_mblt/text/__init__.py +49 -0
- melotts_mblt-0.0.0/melotts_mblt/text/cleaner.py +19 -0
- melotts_mblt-0.0.0/melotts_mblt/text/cmudict.rep +129530 -0
- melotts_mblt-0.0.0/melotts_mblt/text/cmudict_cache.pickle +0 -0
- melotts_mblt-0.0.0/melotts_mblt/text/english.py +236 -0
- melotts_mblt-0.0.0/melotts_mblt/text/english_utils/__init__.py +0 -0
- melotts_mblt-0.0.0/melotts_mblt/text/english_utils/abbreviations.py +36 -0
- melotts_mblt-0.0.0/melotts_mblt/text/english_utils/number_norm.py +99 -0
- melotts_mblt-0.0.0/melotts_mblt/text/english_utils/time_norm.py +47 -0
- melotts_mblt-0.0.0/melotts_mblt/text/japanese.py +7 -0
- melotts_mblt-0.0.0/melotts_mblt/text/ko_dictionary.py +44 -0
- melotts_mblt-0.0.0/melotts_mblt/text/korean.py +161 -0
- melotts_mblt-0.0.0/melotts_mblt/text/symbols.py +358 -0
- melotts_mblt-0.0.0/melotts_mblt/utils.py +143 -0
- melotts_mblt-0.0.0/melotts_mblt.egg-info/PKG-INFO +160 -0
- melotts_mblt-0.0.0/melotts_mblt.egg-info/SOURCES.txt +41 -0
- melotts_mblt-0.0.0/melotts_mblt.egg-info/dependency_links.txt +1 -0
- melotts_mblt-0.0.0/melotts_mblt.egg-info/entry_points.txt +2 -0
- melotts_mblt-0.0.0/melotts_mblt.egg-info/requires.txt +16 -0
- melotts_mblt-0.0.0/melotts_mblt.egg-info/top_level.txt +1 -0
- melotts_mblt-0.0.0/pyproject.toml +110 -0
- melotts_mblt-0.0.0/setup.cfg +4 -0
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Mobilint, Inc.
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
22
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
23
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
24
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
25
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
26
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
27
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
28
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: melotts-mblt
|
|
3
|
+
Version: 0.0.0
|
|
4
|
+
Summary: MeloTTS text-to-speech on Mobilint NPUs: English and Korean speech synthesis with a CLI and Gradio WebUI
|
|
5
|
+
Author-email: "Mobilint Inc." <tech-support@mobilint.com>
|
|
6
|
+
License: BSD-3-Clause AND MIT
|
|
7
|
+
Project-URL: Home, https://www.mobilint.com/
|
|
8
|
+
Project-URL: Repository, https://github.com/mobilint/MeloTTS-mblt
|
|
9
|
+
Keywords: tts,text-to-speech,melotts,NPU,inference,mobilint,mblt,aries,regulus
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: License :: OSI Approved :: BSD License
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Requires-Python: <3.13,>=3.10
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: mblt-npu-python>=0.1.0
|
|
27
|
+
Requires-Dist: mobilint-qb-runtime>=1.4.0
|
|
28
|
+
Requires-Dist: transformers-mblt>=0.0.0
|
|
29
|
+
Requires-Dist: torch>=2.4.1
|
|
30
|
+
Requires-Dist: numpy>=1.26.0
|
|
31
|
+
Requires-Dist: huggingface-hub
|
|
32
|
+
Requires-Dist: tqdm
|
|
33
|
+
Requires-Dist: g2p_en>=2.1.0
|
|
34
|
+
Requires-Dist: anyascii>=0.3.2
|
|
35
|
+
Requires-Dist: jamo>=0.4.1
|
|
36
|
+
Requires-Dist: g2pkk>=0.1.1
|
|
37
|
+
Requires-Dist: unidic>=1.1.0
|
|
38
|
+
Requires-Dist: python-mecab-ko>=1.3.7
|
|
39
|
+
Requires-Dist: soundfile
|
|
40
|
+
Requires-Dist: gradio
|
|
41
|
+
Requires-Dist: click
|
|
42
|
+
Dynamic: license-file
|
|
43
|
+
|
|
44
|
+
# melotts-mblt
|
|
45
|
+
|
|
46
|
+
<!-- markdownlint-disable MD033 -->
|
|
47
|
+
<div align="center">
|
|
48
|
+
<p>
|
|
49
|
+
<a href="https://www.mobilint.com/" target="_blank">
|
|
50
|
+
<img src="https://raw.githubusercontent.com/mobilint/.github/main/assets/Mobilint_Logo_Primary.png" alt="Mobilint Logo" width="60%">
|
|
51
|
+
</a>
|
|
52
|
+
</p>
|
|
53
|
+
</div>
|
|
54
|
+
<!-- markdownlint-enable MD033 -->
|
|
55
|
+
|
|
56
|
+
Run [MeloTTS](https://github.com/myshell-ai/MeloTTS) text-to-speech on Mobilint NPUs. `melotts-mblt` ships
|
|
57
|
+
pre-quantized MeloTTS synthesizers for English and Korean. It provides a Python API, a command line, and a Gradio
|
|
58
|
+
WebUI. The acoustic model runs on the NPU, and its Mobilint BERT prosody encoder loads through
|
|
59
|
+
[transformers-mblt](https://github.com/mobilint/transformers-mblt).
|
|
60
|
+
|
|
61
|
+
Models run on Mobilint [ARIES](https://www.mobilint.com/aries) and [REGULUS](https://www.mobilint.com/regulus)
|
|
62
|
+
boards. Supported target-device identifiers are `aries-rb`, `regulus-ra`, `regulus-rb`, `regulus-ra-usb`, and
|
|
63
|
+
`regulus-rb-usb`.
|
|
64
|
+
|
|
65
|
+
Version `0.0.0` is the first standalone release. It was extracted from `mblt-model-zoo` 2.11.0.
|
|
66
|
+
|
|
67
|
+
## Installation
|
|
68
|
+
|
|
69
|
+
[](https://pypi.org/project/melotts-mblt/)
|
|
70
|
+
[](https://clickpy.clickhouse.com/dashboard/melotts-mblt)
|
|
71
|
+
[](https://pypi.org/project/melotts-mblt/)
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install melotts-mblt
|
|
75
|
+
# One-time download of the NLTK tagger (English G2P) and the UniDic dictionary
|
|
76
|
+
melotts-mblt download
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
NPU execution requires a supported Mobilint NPU driver and device. Python packaging cannot run post-install steps, so
|
|
80
|
+
run `melotts-mblt download` once after installing.
|
|
81
|
+
|
|
82
|
+
## Quick start
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from melotts_mblt import TTS
|
|
86
|
+
|
|
87
|
+
model = TTS(language="EN_NEWEST", device="cpu", trust_remote_code=True)
|
|
88
|
+
try:
|
|
89
|
+
speaker_ids = model.hps.data.spk2id
|
|
90
|
+
model.tts_to_file("Did you ever hear a folk tale about a giant turtle?", speaker_ids["EN-Newest"], "en.wav", speed=1.0)
|
|
91
|
+
finally:
|
|
92
|
+
model.dispose() # release the NPU backends
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
`trust_remote_code=True` lets the Mobilint BERT prosody encoder load its Hub remote code; the `melotts-mblt` CLI and
|
|
96
|
+
WebUI pass it for you. Use `language="KR"` with speaker `"KR"` for Korean. Each language downloads its model from the Mobilint Hugging Face
|
|
97
|
+
organization: [`mobilint/MeloTTS-English-v3`](https://huggingface.co/mobilint/MeloTTS-English-v3) and
|
|
98
|
+
[`mobilint/MeloTTS-Korean`](https://huggingface.co/mobilint/MeloTTS-Korean).
|
|
99
|
+
|
|
100
|
+
These `TTS(...)` arguments select the NPU: `dev_no`, `target_core`, `target_device`, `encoder_mxq_path`, and
|
|
101
|
+
`decoder_mxq_path`. The [API reference](melotts_mblt/README.md) has more examples.
|
|
102
|
+
|
|
103
|
+
## Command line
|
|
104
|
+
|
|
105
|
+
The `melotts-mblt` command provides three subcommands:
|
|
106
|
+
|
|
107
|
+
- `tts` synthesizes speech with the MeloTTS CLI.
|
|
108
|
+
- `ui` launches the Gradio WebUI.
|
|
109
|
+
- `download` fetches the NLTK and UniDic resources.
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
melotts-mblt tts "Text to read" output.wav --language EN_NEWEST --speed 1.2
|
|
113
|
+
melotts-mblt tts "text-to-speech 안녕하세요" kr.wav --language KR
|
|
114
|
+
melotts-mblt tts file.txt out.wav --file
|
|
115
|
+
melotts-mblt ui --host 0.0.0.0 --port 7860
|
|
116
|
+
melotts-mblt download
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
`python -m melotts_mblt.cli` is equivalent to `melotts-mblt`. Run `melotts-mblt tts --help` for every TTS option.
|
|
120
|
+
|
|
121
|
+
## Using with mblt-model-zoo
|
|
122
|
+
|
|
123
|
+
This package replaces `mblt_model_zoo.MeloTTS`. The modules have the same contents, and only the import prefix and the
|
|
124
|
+
command names differ:
|
|
125
|
+
|
|
126
|
+
| mblt-model-zoo | melotts-mblt |
|
|
127
|
+
| --- | --- |
|
|
128
|
+
| `from mblt_model_zoo.MeloTTS.api import TTS` | `from melotts_mblt import TTS` |
|
|
129
|
+
| `mblt_model_zoo.MeloTTS.<module>` | `melotts_mblt.<module>` |
|
|
130
|
+
| `mblt-model-zoo melo ...` / `mblt-model-zoo melotts ...` | `melotts-mblt tts ...` |
|
|
131
|
+
| `mblt-model-zoo melo-ui` | `melotts-mblt ui` |
|
|
132
|
+
| `mblt-melotts-download` | `melotts-mblt download` |
|
|
133
|
+
|
|
134
|
+
## Development
|
|
135
|
+
|
|
136
|
+
Use [uv](https://docs.astral.sh/uv/) to manage the development environment:
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
git clone https://github.com/mobilint/MeloTTS-mblt.git
|
|
140
|
+
cd MeloTTS-mblt
|
|
141
|
+
uv venv --python 3.12
|
|
142
|
+
source .venv/bin/activate
|
|
143
|
+
uv pip install -e . --group dev
|
|
144
|
+
pre-commit install
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
The [test guide](tests/TEST.md) explains the hardware-free and NPU test runs.
|
|
148
|
+
|
|
149
|
+
## Support and issues
|
|
150
|
+
|
|
151
|
+
For installation, model, or runtime support, visit the [Mobilint forum](https://discuss.mobilint.com/)
|
|
152
|
+
or contact [tech-support@mobilint.com](mailto:tech-support@mobilint.com). Report reproducible
|
|
153
|
+
package issues in the [MeloTTS-mblt issue tracker](https://github.com/mobilint/MeloTTS-mblt/issues).
|
|
154
|
+
|
|
155
|
+
## License
|
|
156
|
+
|
|
157
|
+
Mobilint's code is distributed under the [BSD 3-Clause License](LICENSE). The MeloTTS-derived code in `melotts_mblt`
|
|
158
|
+
remains under [MyShell.ai's MIT License](melotts_mblt/LICENSE), and that license ships in the wheel. MeloTTS was
|
|
159
|
+
created by Wenliang Zhao, Xumin Yu, and Zengyi Qin; see the [API reference](melotts_mblt/README.md#original-authors)
|
|
160
|
+
for the original authors and the citation.
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
# melotts-mblt
|
|
2
|
+
|
|
3
|
+
<!-- markdownlint-disable MD033 -->
|
|
4
|
+
<div align="center">
|
|
5
|
+
<p>
|
|
6
|
+
<a href="https://www.mobilint.com/" target="_blank">
|
|
7
|
+
<img src="https://raw.githubusercontent.com/mobilint/.github/main/assets/Mobilint_Logo_Primary.png" alt="Mobilint Logo" width="60%">
|
|
8
|
+
</a>
|
|
9
|
+
</p>
|
|
10
|
+
</div>
|
|
11
|
+
<!-- markdownlint-enable MD033 -->
|
|
12
|
+
|
|
13
|
+
Run [MeloTTS](https://github.com/myshell-ai/MeloTTS) text-to-speech on Mobilint NPUs. `melotts-mblt` ships
|
|
14
|
+
pre-quantized MeloTTS synthesizers for English and Korean. It provides a Python API, a command line, and a Gradio
|
|
15
|
+
WebUI. The acoustic model runs on the NPU, and its Mobilint BERT prosody encoder loads through
|
|
16
|
+
[transformers-mblt](https://github.com/mobilint/transformers-mblt).
|
|
17
|
+
|
|
18
|
+
Models run on Mobilint [ARIES](https://www.mobilint.com/aries) and [REGULUS](https://www.mobilint.com/regulus)
|
|
19
|
+
boards. Supported target-device identifiers are `aries-rb`, `regulus-ra`, `regulus-rb`, `regulus-ra-usb`, and
|
|
20
|
+
`regulus-rb-usb`.
|
|
21
|
+
|
|
22
|
+
Version `0.0.0` is the first standalone release. It was extracted from `mblt-model-zoo` 2.11.0.
|
|
23
|
+
|
|
24
|
+
## Installation
|
|
25
|
+
|
|
26
|
+
[](https://pypi.org/project/melotts-mblt/)
|
|
27
|
+
[](https://clickpy.clickhouse.com/dashboard/melotts-mblt)
|
|
28
|
+
[](https://pypi.org/project/melotts-mblt/)
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install melotts-mblt
|
|
32
|
+
# One-time download of the NLTK tagger (English G2P) and the UniDic dictionary
|
|
33
|
+
melotts-mblt download
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
NPU execution requires a supported Mobilint NPU driver and device. Python packaging cannot run post-install steps, so
|
|
37
|
+
run `melotts-mblt download` once after installing.
|
|
38
|
+
|
|
39
|
+
## Quick start
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from melotts_mblt import TTS
|
|
43
|
+
|
|
44
|
+
model = TTS(language="EN_NEWEST", device="cpu", trust_remote_code=True)
|
|
45
|
+
try:
|
|
46
|
+
speaker_ids = model.hps.data.spk2id
|
|
47
|
+
model.tts_to_file("Did you ever hear a folk tale about a giant turtle?", speaker_ids["EN-Newest"], "en.wav", speed=1.0)
|
|
48
|
+
finally:
|
|
49
|
+
model.dispose() # release the NPU backends
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
`trust_remote_code=True` lets the Mobilint BERT prosody encoder load its Hub remote code; the `melotts-mblt` CLI and
|
|
53
|
+
WebUI pass it for you. Use `language="KR"` with speaker `"KR"` for Korean. Each language downloads its model from the Mobilint Hugging Face
|
|
54
|
+
organization: [`mobilint/MeloTTS-English-v3`](https://huggingface.co/mobilint/MeloTTS-English-v3) and
|
|
55
|
+
[`mobilint/MeloTTS-Korean`](https://huggingface.co/mobilint/MeloTTS-Korean).
|
|
56
|
+
|
|
57
|
+
These `TTS(...)` arguments select the NPU: `dev_no`, `target_core`, `target_device`, `encoder_mxq_path`, and
|
|
58
|
+
`decoder_mxq_path`. The [API reference](melotts_mblt/README.md) has more examples.
|
|
59
|
+
|
|
60
|
+
## Command line
|
|
61
|
+
|
|
62
|
+
The `melotts-mblt` command provides three subcommands:
|
|
63
|
+
|
|
64
|
+
- `tts` synthesizes speech with the MeloTTS CLI.
|
|
65
|
+
- `ui` launches the Gradio WebUI.
|
|
66
|
+
- `download` fetches the NLTK and UniDic resources.
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
melotts-mblt tts "Text to read" output.wav --language EN_NEWEST --speed 1.2
|
|
70
|
+
melotts-mblt tts "text-to-speech 안녕하세요" kr.wav --language KR
|
|
71
|
+
melotts-mblt tts file.txt out.wav --file
|
|
72
|
+
melotts-mblt ui --host 0.0.0.0 --port 7860
|
|
73
|
+
melotts-mblt download
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
`python -m melotts_mblt.cli` is equivalent to `melotts-mblt`. Run `melotts-mblt tts --help` for every TTS option.
|
|
77
|
+
|
|
78
|
+
## Using with mblt-model-zoo
|
|
79
|
+
|
|
80
|
+
This package replaces `mblt_model_zoo.MeloTTS`. The modules have the same contents, and only the import prefix and the
|
|
81
|
+
command names differ:
|
|
82
|
+
|
|
83
|
+
| mblt-model-zoo | melotts-mblt |
|
|
84
|
+
| --- | --- |
|
|
85
|
+
| `from mblt_model_zoo.MeloTTS.api import TTS` | `from melotts_mblt import TTS` |
|
|
86
|
+
| `mblt_model_zoo.MeloTTS.<module>` | `melotts_mblt.<module>` |
|
|
87
|
+
| `mblt-model-zoo melo ...` / `mblt-model-zoo melotts ...` | `melotts-mblt tts ...` |
|
|
88
|
+
| `mblt-model-zoo melo-ui` | `melotts-mblt ui` |
|
|
89
|
+
| `mblt-melotts-download` | `melotts-mblt download` |
|
|
90
|
+
|
|
91
|
+
## Development
|
|
92
|
+
|
|
93
|
+
Use [uv](https://docs.astral.sh/uv/) to manage the development environment:
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
git clone https://github.com/mobilint/MeloTTS-mblt.git
|
|
97
|
+
cd MeloTTS-mblt
|
|
98
|
+
uv venv --python 3.12
|
|
99
|
+
source .venv/bin/activate
|
|
100
|
+
uv pip install -e . --group dev
|
|
101
|
+
pre-commit install
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
The [test guide](tests/TEST.md) explains the hardware-free and NPU test runs.
|
|
105
|
+
|
|
106
|
+
## Support and issues
|
|
107
|
+
|
|
108
|
+
For installation, model, or runtime support, visit the [Mobilint forum](https://discuss.mobilint.com/)
|
|
109
|
+
or contact [tech-support@mobilint.com](mailto:tech-support@mobilint.com). Report reproducible
|
|
110
|
+
package issues in the [MeloTTS-mblt issue tracker](https://github.com/mobilint/MeloTTS-mblt/issues).
|
|
111
|
+
|
|
112
|
+
## License
|
|
113
|
+
|
|
114
|
+
Mobilint's code is distributed under the [BSD 3-Clause License](LICENSE). The MeloTTS-derived code in `melotts_mblt`
|
|
115
|
+
remains under [MyShell.ai's MIT License](melotts_mblt/LICENSE), and that license ships in the wheel. MeloTTS was
|
|
116
|
+
created by Wenliang Zhao, Xumin Yu, and Zengyi Qin; see the [API reference](melotts_mblt/README.md#original-authors)
|
|
117
|
+
for the original authors and the citation.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
Copyright (c) 2024 MyShell.ai
|
|
2
|
+
|
|
3
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
4
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
5
|
+
in the Software without restriction, including without limitation the rights
|
|
6
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
7
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
8
|
+
furnished to do so, subject to the following conditions:
|
|
9
|
+
|
|
10
|
+
The above copyright notice and this permission notice shall be included in all
|
|
11
|
+
copies or substantial portions of the Software.
|
|
12
|
+
|
|
13
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
14
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
15
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
16
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
17
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
18
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
19
|
+
SOFTWARE.
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""MeloTTS text-to-speech for Mobilint NPUs.
|
|
2
|
+
|
|
3
|
+
``from melotts_mblt import TTS`` loads the speech synthesizer. The import is lazy, so ``import melotts_mblt`` does not
|
|
4
|
+
import ``torch`` or ``transformers``.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from typing import TYPE_CHECKING
|
|
8
|
+
|
|
9
|
+
__version__ = "0.0.0"
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from .api import TTS
|
|
13
|
+
|
|
14
|
+
__all__ = ["TTS", "__version__"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def __getattr__(name: str):
|
|
18
|
+
if name == "TTS":
|
|
19
|
+
from .api import TTS
|
|
20
|
+
|
|
21
|
+
return TTS
|
|
22
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def __dir__() -> list[str]:
|
|
26
|
+
return sorted(set(globals()) | set(__all__))
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
import math
|
|
2
|
+
import re
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import soundfile
|
|
7
|
+
import torch
|
|
8
|
+
from torch import nn
|
|
9
|
+
from tqdm import tqdm
|
|
10
|
+
from transformers.models.auto.modeling_auto import AutoModelForMaskedLM
|
|
11
|
+
from transformers.models.auto.tokenization_auto import AutoTokenizer
|
|
12
|
+
|
|
13
|
+
from . import utils
|
|
14
|
+
from .download_utils import (
|
|
15
|
+
LANG_TO_HF_REPO_ID,
|
|
16
|
+
load_or_download_config,
|
|
17
|
+
load_or_download_model,
|
|
18
|
+
resolve_local_bert_mxq,
|
|
19
|
+
resolve_local_mxq,
|
|
20
|
+
)
|
|
21
|
+
from .models import MobilintSynthesizerTrn
|
|
22
|
+
from .split_utils import split_sentence
|
|
23
|
+
from .text.cleaner import clean_text
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _describe_value(value):
|
|
27
|
+
"""``repr(value)`` for error messages, falling back to the type when ``repr`` itself fails (e.g. huge ints)."""
|
|
28
|
+
try:
|
|
29
|
+
return repr(value)
|
|
30
|
+
except Exception:
|
|
31
|
+
return f"<{type(value).__name__}>"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _validate_speed(speed):
|
|
35
|
+
"""Return ``speed`` as a ``float`` if it is a finite real number greater than 0, else raise ``ValueError``.
|
|
36
|
+
|
|
37
|
+
Any real scalar ``float()`` accepts is allowed (``int``, ``float``, NumPy scalars, 0-d tensors, ``Fraction``,
|
|
38
|
+
``Decimal``). Booleans and strings are rejected even though ``float()`` would convert them.
|
|
39
|
+
"""
|
|
40
|
+
message = "speed must be a finite number greater than 0, got {}"
|
|
41
|
+
if isinstance(speed, (bool, np.bool_, str, bytes)):
|
|
42
|
+
raise ValueError(message.format(_describe_value(speed)))
|
|
43
|
+
try:
|
|
44
|
+
value = float(speed)
|
|
45
|
+
except (TypeError, ValueError, OverflowError):
|
|
46
|
+
# OverflowError: a finite real too large for a float, e.g. 10**10000.
|
|
47
|
+
raise ValueError(message.format(_describe_value(speed))) from None
|
|
48
|
+
if not math.isfinite(value) or value <= 0:
|
|
49
|
+
raise ValueError(message.format(_describe_value(speed)))
|
|
50
|
+
return value
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class TTS(nn.Module):
|
|
54
|
+
def __init__(self,
|
|
55
|
+
language,
|
|
56
|
+
device='auto',
|
|
57
|
+
config_path=None,
|
|
58
|
+
ckpt_path=None,
|
|
59
|
+
|
|
60
|
+
trust_remote_code: Optional[bool]=None,
|
|
61
|
+
local_files_only: Optional[bool]=None,
|
|
62
|
+
|
|
63
|
+
dev_no: Optional[int] = None,
|
|
64
|
+
target_core: Optional[str] = None,
|
|
65
|
+
encoder_mxq_path: Optional[str] = None,
|
|
66
|
+
decoder_mxq_path: Optional[str] = None,
|
|
67
|
+
target_device: Optional[str] = None,
|
|
68
|
+
):
|
|
69
|
+
nn.Module.__init__(self)
|
|
70
|
+
if device == "auto":
|
|
71
|
+
device = "cpu"
|
|
72
|
+
if torch.cuda.is_available():
|
|
73
|
+
device = "cuda"
|
|
74
|
+
if torch.backends.mps.is_available():
|
|
75
|
+
device = "mps"
|
|
76
|
+
if "cuda" in device:
|
|
77
|
+
assert torch.cuda.is_available()
|
|
78
|
+
|
|
79
|
+
# config_path =
|
|
80
|
+
hps = load_or_download_config(language, config_path=config_path, local_files_only=local_files_only)
|
|
81
|
+
|
|
82
|
+
if dev_no is not None:
|
|
83
|
+
hps.model.dev_no = dev_no
|
|
84
|
+
|
|
85
|
+
if target_core is not None:
|
|
86
|
+
hps.model.target_core = target_core
|
|
87
|
+
|
|
88
|
+
if encoder_mxq_path is not None:
|
|
89
|
+
hps.model.encoder_mxq_path = encoder_mxq_path
|
|
90
|
+
|
|
91
|
+
if decoder_mxq_path is not None:
|
|
92
|
+
hps.model.decoder_mxq_path = decoder_mxq_path
|
|
93
|
+
|
|
94
|
+
resolved_target_device = (
|
|
95
|
+
target_device
|
|
96
|
+
if target_device is not None
|
|
97
|
+
else getattr(hps.model, "target_device", "aries-rb")
|
|
98
|
+
)
|
|
99
|
+
hps.model.target_device = resolved_target_device
|
|
100
|
+
|
|
101
|
+
if local_files_only:
|
|
102
|
+
# mblt_npu downloads a missing MXQ from the Hub; resolve both synthesizer MXQs from the cache first so
|
|
103
|
+
# local_files_only never reaches the network (LocalEntryNotFoundError when they are not cached).
|
|
104
|
+
repo_id = LANG_TO_HF_REPO_ID[language]
|
|
105
|
+
hps.model.encoder_mxq_path = resolve_local_mxq(repo_id, hps.model.encoder_mxq_path)
|
|
106
|
+
hps.model.decoder_mxq_path = resolve_local_mxq(repo_id, hps.model.decoder_mxq_path)
|
|
107
|
+
|
|
108
|
+
num_languages = hps.num_languages
|
|
109
|
+
num_tones = hps.num_tones
|
|
110
|
+
symbols = hps.symbols
|
|
111
|
+
|
|
112
|
+
# A failure while building the synthesizer is cleaned up by MobilintSynthesizerTrn itself; from here on the
|
|
113
|
+
# NPU backends exist, so assign self.model before anything else can fail (including the device transfer).
|
|
114
|
+
self.model = MobilintSynthesizerTrn(
|
|
115
|
+
len(symbols),
|
|
116
|
+
hps.data.filter_length // 2 + 1,
|
|
117
|
+
hps.train.segment_size // hps.data.hop_length,
|
|
118
|
+
n_speakers=hps.data.n_speakers,
|
|
119
|
+
num_tones=num_tones,
|
|
120
|
+
num_languages=num_languages,
|
|
121
|
+
name_or_path=LANG_TO_HF_REPO_ID[language],
|
|
122
|
+
**hps.model,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
try:
|
|
126
|
+
model = self.model.to(device)
|
|
127
|
+
model.eval()
|
|
128
|
+
self.model = model
|
|
129
|
+
self.symbol_to_id = {s: i for i, s in enumerate(symbols)}
|
|
130
|
+
self.hps = hps
|
|
131
|
+
self.device = device
|
|
132
|
+
|
|
133
|
+
# load state_dict
|
|
134
|
+
checkpoint_dict = load_or_download_model(language, device, ckpt_path=ckpt_path, local_files_only=local_files_only)
|
|
135
|
+
self.model.load_state_dict(checkpoint_dict['model'], strict=True)
|
|
136
|
+
|
|
137
|
+
language = language.split('_')[0]
|
|
138
|
+
self.language = 'ZH_MIX_EN' if language == 'ZH' else language # we support a ZH_MIX_EN model
|
|
139
|
+
|
|
140
|
+
self.tokenizer = AutoTokenizer.from_pretrained(
|
|
141
|
+
hps.model.bert_model_id,
|
|
142
|
+
trust_remote_code=trust_remote_code,
|
|
143
|
+
local_files_only=local_files_only,
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
bert_kwargs = {}
|
|
147
|
+
if local_files_only:
|
|
148
|
+
# Same reason as the synthesizer MXQs: resolve the BERT MXQ from the cache before mblt_npu sees it.
|
|
149
|
+
bert_mxq_path = resolve_local_bert_mxq(hps.model.bert_model_id)
|
|
150
|
+
if bert_mxq_path is not None:
|
|
151
|
+
bert_kwargs["mxq_path"] = bert_mxq_path
|
|
152
|
+
|
|
153
|
+
# Keep a reference to the loaded BERT (it owns an NPU backend) before the transfer, so dispose() can
|
|
154
|
+
# still release it if .to(device) fails.
|
|
155
|
+
self.bert = AutoModelForMaskedLM.from_pretrained(
|
|
156
|
+
hps.model.bert_model_id,
|
|
157
|
+
trust_remote_code=trust_remote_code,
|
|
158
|
+
local_files_only=local_files_only,
|
|
159
|
+
|
|
160
|
+
dev_no=hps.model.dev_no,
|
|
161
|
+
target_cores=[hps.model.target_core],
|
|
162
|
+
target_device=hps.model.target_device,
|
|
163
|
+
**bert_kwargs,
|
|
164
|
+
)
|
|
165
|
+
self.bert = self.bert.to(device)
|
|
166
|
+
|
|
167
|
+
except BaseException:
|
|
168
|
+
# Checkpoint, tokenizer, or BERT loading failed after the synthesizer's NPU backends were created.
|
|
169
|
+
self.dispose()
|
|
170
|
+
raise
|
|
171
|
+
|
|
172
|
+
@staticmethod
|
|
173
|
+
def audio_numpy_concat(segment_data_list, sr, speed=1.0):
|
|
174
|
+
audio_segments = []
|
|
175
|
+
for segment_data in segment_data_list:
|
|
176
|
+
audio_segments += segment_data.reshape(-1).tolist()
|
|
177
|
+
audio_segments += [0] * int((sr * 0.05) / speed)
|
|
178
|
+
audio_segments = np.array(audio_segments).astype(np.float32)
|
|
179
|
+
return audio_segments
|
|
180
|
+
|
|
181
|
+
@staticmethod
|
|
182
|
+
def split_sentences_into_pieces(text, language, quiet=False):
|
|
183
|
+
texts = split_sentence(text, language_str=language)
|
|
184
|
+
if not quiet:
|
|
185
|
+
print(" > Text split to sentences.")
|
|
186
|
+
print("\n".join(texts))
|
|
187
|
+
print(" > ===========================")
|
|
188
|
+
return texts
|
|
189
|
+
|
|
190
|
+
def _bert_token_count(self, text, language):
|
|
191
|
+
"""Number of BERT tokens (special tokens included) the text produces after normalization."""
|
|
192
|
+
if language in ["EN", "ZH_MIX_EN"]:
|
|
193
|
+
text = re.sub(r"([a-z])([A-Z])", r"\1 \2", text)
|
|
194
|
+
_, _, _, word2ph = clean_text(text, language, tokenizer=self.tokenizer)
|
|
195
|
+
return len(word2ph)
|
|
196
|
+
|
|
197
|
+
def _bert_max_tokens(self):
|
|
198
|
+
config = getattr(getattr(self, "bert", None), "config", None)
|
|
199
|
+
limit = getattr(config, "max_position_embeddings", None) or getattr(self.tokenizer, "model_max_length", None)
|
|
200
|
+
# Tokenizers without a configured maximum report a huge sentinel value; fall back to BERT's usual 512.
|
|
201
|
+
return int(limit) if limit and limit < 100_000 else 512
|
|
202
|
+
|
|
203
|
+
def _fit_piece_to_bert(self, text, language, limit):
|
|
204
|
+
"""Split ``text`` until every piece fits BERT's position limit.
|
|
205
|
+
|
|
206
|
+
Sentence splitting only breaks at punctuation, so unpunctuated input can exceed BERT's position embeddings,
|
|
207
|
+
which are computed for the whole input before the MXQ runs (synthesizer chunking cannot help). Bisect at
|
|
208
|
+
whitespace, or at the middle character for an unbroken run; each piece then goes through normalization,
|
|
209
|
+
G2P, and BERT on its own, so ``word2ph`` / phoneme alignment is recomputed per piece instead of truncated.
|
|
210
|
+
"""
|
|
211
|
+
if self._bert_token_count(text, language) <= limit:
|
|
212
|
+
return [text]
|
|
213
|
+
words = text.split()
|
|
214
|
+
if len(words) > 1:
|
|
215
|
+
middle = len(words) // 2
|
|
216
|
+
left, right = " ".join(words[:middle]), " ".join(words[middle:])
|
|
217
|
+
else:
|
|
218
|
+
middle = len(text) // 2
|
|
219
|
+
left, right = text[:middle], text[middle:]
|
|
220
|
+
if not left.strip() or not right.strip():
|
|
221
|
+
return [text]
|
|
222
|
+
return self._fit_piece_to_bert(left, language, limit) + self._fit_piece_to_bert(right, language, limit)
|
|
223
|
+
|
|
224
|
+
def fit_pieces_to_bert(self, texts, language):
|
|
225
|
+
"""Return ``texts`` with every piece split as needed to fit BERT's maximum input length."""
|
|
226
|
+
if getattr(self.hps.data, "disable_bert", False):
|
|
227
|
+
return list(texts)
|
|
228
|
+
limit = self._bert_max_tokens()
|
|
229
|
+
pieces = []
|
|
230
|
+
for text in texts:
|
|
231
|
+
pieces.extend(self._fit_piece_to_bert(text, language, limit))
|
|
232
|
+
return pieces
|
|
233
|
+
|
|
234
|
+
def tts_to_file(self, text, speaker_id, output_path=None, sdp_ratio=0.2, noise_scale=0.6, noise_scale_w=0.8, speed=1.0, pbar=None, format=None, position=None, quiet=False):
|
|
235
|
+
speed = _validate_speed(speed)
|
|
236
|
+
language = self.language
|
|
237
|
+
texts = self.fit_pieces_to_bert(self.split_sentences_into_pieces(text, language, quiet), language)
|
|
238
|
+
audio_list = []
|
|
239
|
+
if pbar:
|
|
240
|
+
tx = pbar(texts)
|
|
241
|
+
else:
|
|
242
|
+
if position:
|
|
243
|
+
tx = tqdm(texts, position=position)
|
|
244
|
+
elif quiet:
|
|
245
|
+
tx = texts
|
|
246
|
+
else:
|
|
247
|
+
tx = tqdm(texts)
|
|
248
|
+
for t in tx:
|
|
249
|
+
if language in ["EN", "ZH_MIX_EN"]:
|
|
250
|
+
t = re.sub(r"([a-z])([A-Z])", r"\1 \2", t)
|
|
251
|
+
device = self.device
|
|
252
|
+
bert, ja_bert, phones, tones, lang_ids = utils.get_text_for_tts_infer(t, language, self.hps, device, self.symbol_to_id, tokenizer=self.tokenizer, bert=self.bert)
|
|
253
|
+
with torch.no_grad():
|
|
254
|
+
x_tst = phones.to(device).unsqueeze(0)
|
|
255
|
+
tones = tones.to(device).unsqueeze(0)
|
|
256
|
+
lang_ids = lang_ids.to(device).unsqueeze(0)
|
|
257
|
+
bert = bert.to(device).unsqueeze(0)
|
|
258
|
+
ja_bert = ja_bert.to(device).unsqueeze(0)
|
|
259
|
+
x_tst_lengths = torch.LongTensor([phones.size(0)]).to(device)
|
|
260
|
+
del phones
|
|
261
|
+
speakers = torch.LongTensor([speaker_id]).to(device)
|
|
262
|
+
audio = (
|
|
263
|
+
self.model.infer(
|
|
264
|
+
x_tst,
|
|
265
|
+
x_tst_lengths,
|
|
266
|
+
speakers,
|
|
267
|
+
tones,
|
|
268
|
+
lang_ids,
|
|
269
|
+
bert,
|
|
270
|
+
ja_bert,
|
|
271
|
+
sdp_ratio=sdp_ratio,
|
|
272
|
+
noise_scale=noise_scale,
|
|
273
|
+
noise_scale_w=noise_scale_w,
|
|
274
|
+
length_scale=1.0 / speed,
|
|
275
|
+
)[0][0, 0]
|
|
276
|
+
.data.cpu()
|
|
277
|
+
.float()
|
|
278
|
+
.numpy()
|
|
279
|
+
)
|
|
280
|
+
del x_tst, tones, lang_ids, bert, ja_bert, x_tst_lengths, speakers
|
|
281
|
+
#
|
|
282
|
+
audio_list.append(audio)
|
|
283
|
+
torch.cuda.empty_cache()
|
|
284
|
+
audio = self.audio_numpy_concat(audio_list, sr=self.hps.data.sampling_rate, speed=speed)
|
|
285
|
+
|
|
286
|
+
if output_path is None:
|
|
287
|
+
return audio
|
|
288
|
+
else:
|
|
289
|
+
if format:
|
|
290
|
+
soundfile.write(
|
|
291
|
+
output_path, audio, self.hps.data.sampling_rate, format=format
|
|
292
|
+
)
|
|
293
|
+
else:
|
|
294
|
+
soundfile.write(output_path, audio, self.hps.data.sampling_rate)
|
|
295
|
+
|
|
296
|
+
def launch(self):
|
|
297
|
+
self.model.launch()
|
|
298
|
+
|
|
299
|
+
def dispose(self):
|
|
300
|
+
model = getattr(self, "model", None)
|
|
301
|
+
if model is not None:
|
|
302
|
+
model.dispose()
|
|
303
|
+
# ``self.bert`` is a sibling NPU-backed module built via
|
|
304
|
+
# ``AutoModelForMaskedLM.from_pretrained``; if the loaded class is a
|
|
305
|
+
# Mobilint Bert (``MobilintBertForMaskedLM``) it owns its own NPU
|
|
306
|
+
# backend and must be released alongside the synthesizer, otherwise
|
|
307
|
+
# LPDDR stays pinned between TTS instances.
|
|
308
|
+
bert_dispose = getattr(getattr(self, "bert", None), "dispose", None)
|
|
309
|
+
if callable(bert_dispose):
|
|
310
|
+
bert_dispose()
|