melotts-mblt 0.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. melotts_mblt-0.0.0/LICENSE +28 -0
  2. melotts_mblt-0.0.0/MANIFEST.in +3 -0
  3. melotts_mblt-0.0.0/PKG-INFO +160 -0
  4. melotts_mblt-0.0.0/README.md +117 -0
  5. melotts_mblt-0.0.0/melotts_mblt/LICENSE +19 -0
  6. melotts_mblt-0.0.0/melotts_mblt/__init__.py +26 -0
  7. melotts_mblt-0.0.0/melotts_mblt/api.py +310 -0
  8. melotts_mblt-0.0.0/melotts_mblt/app.py +98 -0
  9. melotts_mblt-0.0.0/melotts_mblt/cli/__init__.py +3 -0
  10. melotts_mblt-0.0.0/melotts_mblt/cli/__main__.py +4 -0
  11. melotts_mblt-0.0.0/melotts_mblt/cli/_click.py +37 -0
  12. melotts_mblt-0.0.0/melotts_mblt/cli/download.py +80 -0
  13. melotts_mblt-0.0.0/melotts_mblt/cli/main.py +50 -0
  14. melotts_mblt-0.0.0/melotts_mblt/cli/tts.py +58 -0
  15. melotts_mblt-0.0.0/melotts_mblt/cli/ui.py +70 -0
  16. melotts_mblt-0.0.0/melotts_mblt/commons.py +136 -0
  17. melotts_mblt-0.0.0/melotts_mblt/download_utils.py +54 -0
  18. melotts_mblt-0.0.0/melotts_mblt/main.py +50 -0
  19. melotts_mblt-0.0.0/melotts_mblt/models.py +476 -0
  20. melotts_mblt-0.0.0/melotts_mblt/py.typed +0 -0
  21. melotts_mblt-0.0.0/melotts_mblt/split_utils.py +175 -0
  22. melotts_mblt-0.0.0/melotts_mblt/text/__init__.py +49 -0
  23. melotts_mblt-0.0.0/melotts_mblt/text/cleaner.py +19 -0
  24. melotts_mblt-0.0.0/melotts_mblt/text/cmudict.rep +129530 -0
  25. melotts_mblt-0.0.0/melotts_mblt/text/cmudict_cache.pickle +0 -0
  26. melotts_mblt-0.0.0/melotts_mblt/text/english.py +236 -0
  27. melotts_mblt-0.0.0/melotts_mblt/text/english_utils/__init__.py +0 -0
  28. melotts_mblt-0.0.0/melotts_mblt/text/english_utils/abbreviations.py +36 -0
  29. melotts_mblt-0.0.0/melotts_mblt/text/english_utils/number_norm.py +99 -0
  30. melotts_mblt-0.0.0/melotts_mblt/text/english_utils/time_norm.py +47 -0
  31. melotts_mblt-0.0.0/melotts_mblt/text/japanese.py +7 -0
  32. melotts_mblt-0.0.0/melotts_mblt/text/ko_dictionary.py +44 -0
  33. melotts_mblt-0.0.0/melotts_mblt/text/korean.py +161 -0
  34. melotts_mblt-0.0.0/melotts_mblt/text/symbols.py +358 -0
  35. melotts_mblt-0.0.0/melotts_mblt/utils.py +143 -0
  36. melotts_mblt-0.0.0/melotts_mblt.egg-info/PKG-INFO +160 -0
  37. melotts_mblt-0.0.0/melotts_mblt.egg-info/SOURCES.txt +41 -0
  38. melotts_mblt-0.0.0/melotts_mblt.egg-info/dependency_links.txt +1 -0
  39. melotts_mblt-0.0.0/melotts_mblt.egg-info/entry_points.txt +2 -0
  40. melotts_mblt-0.0.0/melotts_mblt.egg-info/requires.txt +16 -0
  41. melotts_mblt-0.0.0/melotts_mblt.egg-info/top_level.txt +1 -0
  42. melotts_mblt-0.0.0/pyproject.toml +110 -0
  43. melotts_mblt-0.0.0/setup.cfg +4 -0
@@ -0,0 +1,28 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2026, Mobilint, Inc.
4
+
5
+ Redistribution and use in source and binary forms, with or without
6
+ modification, are permitted provided that the following conditions are met:
7
+
8
+ 1. Redistributions of source code must retain the above copyright notice, this
9
+ list of conditions and the following disclaimer.
10
+
11
+ 2. Redistributions in binary form must reproduce the above copyright notice,
12
+ this list of conditions and the following disclaimer in the documentation
13
+ and/or other materials provided with the distribution.
14
+
15
+ 3. Neither the name of the copyright holder nor the names of its
16
+ contributors may be used to endorse or promote products derived from
17
+ this software without specific prior written permission.
18
+
19
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
20
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
21
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
22
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
23
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
24
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
25
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
26
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
27
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
28
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,3 @@
1
+ # Tests are development-only. setuptools already builds only the `melotts_mblt*` packages; prune them from the sdist
2
+ # to pin that intent.
3
+ prune tests
@@ -0,0 +1,160 @@
1
+ Metadata-Version: 2.4
2
+ Name: melotts-mblt
3
+ Version: 0.0.0
4
+ Summary: MeloTTS text-to-speech on Mobilint NPUs: English and Korean speech synthesis with a CLI and Gradio WebUI
5
+ Author-email: "Mobilint Inc." <tech-support@mobilint.com>
6
+ License: BSD-3-Clause AND MIT
7
+ Project-URL: Home, https://www.mobilint.com/
8
+ Project-URL: Repository, https://github.com/mobilint/MeloTTS-mblt
9
+ Keywords: tts,text-to-speech,melotts,NPU,inference,mobilint,mblt,aries,regulus
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Environment :: Console
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: BSD License
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Operating System :: POSIX :: Linux
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Requires-Python: <3.13,>=3.10
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: mblt-npu-python>=0.1.0
27
+ Requires-Dist: mobilint-qb-runtime>=1.4.0
28
+ Requires-Dist: transformers-mblt>=0.0.0
29
+ Requires-Dist: torch>=2.4.1
30
+ Requires-Dist: numpy>=1.26.0
31
+ Requires-Dist: huggingface-hub
32
+ Requires-Dist: tqdm
33
+ Requires-Dist: g2p_en>=2.1.0
34
+ Requires-Dist: anyascii>=0.3.2
35
+ Requires-Dist: jamo>=0.4.1
36
+ Requires-Dist: g2pkk>=0.1.1
37
+ Requires-Dist: unidic>=1.1.0
38
+ Requires-Dist: python-mecab-ko>=1.3.7
39
+ Requires-Dist: soundfile
40
+ Requires-Dist: gradio
41
+ Requires-Dist: click
42
+ Dynamic: license-file
43
+
44
+ # melotts-mblt
45
+
46
+ <!-- markdownlint-disable MD033 -->
47
+ <div align="center">
48
+ <p>
49
+ <a href="https://www.mobilint.com/" target="_blank">
50
+ <img src="https://raw.githubusercontent.com/mobilint/.github/main/assets/Mobilint_Logo_Primary.png" alt="Mobilint Logo" width="60%">
51
+ </a>
52
+ </p>
53
+ </div>
54
+ <!-- markdownlint-enable MD033 -->
55
+
56
+ Run [MeloTTS](https://github.com/myshell-ai/MeloTTS) text-to-speech on Mobilint NPUs. `melotts-mblt` ships
57
+ pre-quantized MeloTTS synthesizers for English and Korean. It provides a Python API, a command line, and a Gradio
58
+ WebUI. The acoustic model runs on the NPU, and its Mobilint BERT prosody encoder loads through
59
+ [transformers-mblt](https://github.com/mobilint/transformers-mblt).
60
+
61
+ Models run on Mobilint [ARIES](https://www.mobilint.com/aries) and [REGULUS](https://www.mobilint.com/regulus)
62
+ boards. Supported target-device identifiers are `aries-rb`, `regulus-ra`, `regulus-rb`, `regulus-ra-usb`, and
63
+ `regulus-rb-usb`.
64
+
65
+ Version `0.0.0` is the first standalone release. It was extracted from `mblt-model-zoo` 2.11.0.
66
+
67
+ ## Installation
68
+
69
+ [![PyPI - Version](https://img.shields.io/pypi/v/melotts-mblt?logo=pypi&logoColor=white)](https://pypi.org/project/melotts-mblt/)
70
+ [![PyPI Downloads](https://static.pepy.tech/badge/melotts-mblt?period=total&units=INTERNATIONAL_SYSTEM&left_color=BLACK&right_color=GREEN&left_text=downloads)](https://clickpy.clickhouse.com/dashboard/melotts-mblt)
71
+ [![PyPI - Python Version](https://img.shields.io/pypi/pyversions/melotts-mblt?logo=python&logoColor=gold)](https://pypi.org/project/melotts-mblt/)
72
+
73
+ ```bash
74
+ pip install melotts-mblt
75
+ # One-time download of the NLTK tagger (English G2P) and the UniDic dictionary
76
+ melotts-mblt download
77
+ ```
78
+
79
+ NPU execution requires a supported Mobilint NPU driver and device. Python packaging cannot run post-install steps, so
80
+ run `melotts-mblt download` once after installing.
81
+
82
+ ## Quick start
83
+
84
+ ```python
85
+ from melotts_mblt import TTS
86
+
87
+ model = TTS(language="EN_NEWEST", device="cpu", trust_remote_code=True)
88
+ try:
89
+ speaker_ids = model.hps.data.spk2id
90
+ model.tts_to_file("Did you ever hear a folk tale about a giant turtle?", speaker_ids["EN-Newest"], "en.wav", speed=1.0)
91
+ finally:
92
+ model.dispose() # release the NPU backends
93
+ ```
94
+
95
+ `trust_remote_code=True` lets the Mobilint BERT prosody encoder load its Hub remote code; the `melotts-mblt` CLI and
96
+ WebUI pass it for you. Use `language="KR"` with speaker `"KR"` for Korean. Each language downloads its model from the Mobilint Hugging Face
97
+ organization: [`mobilint/MeloTTS-English-v3`](https://huggingface.co/mobilint/MeloTTS-English-v3) and
98
+ [`mobilint/MeloTTS-Korean`](https://huggingface.co/mobilint/MeloTTS-Korean).
99
+
100
+ These `TTS(...)` arguments select the NPU: `dev_no`, `target_core`, `target_device`, `encoder_mxq_path`, and
101
+ `decoder_mxq_path`. The [API reference](melotts_mblt/README.md) has more examples.
102
+
103
+ ## Command line
104
+
105
+ The `melotts-mblt` command provides three subcommands:
106
+
107
+ - `tts` synthesizes speech with the MeloTTS CLI.
108
+ - `ui` launches the Gradio WebUI.
109
+ - `download` fetches the NLTK and UniDic resources.
110
+
111
+ ```bash
112
+ melotts-mblt tts "Text to read" output.wav --language EN_NEWEST --speed 1.2
113
+ melotts-mblt tts "text-to-speech 안녕하세요" kr.wav --language KR
114
+ melotts-mblt tts file.txt out.wav --file
115
+ melotts-mblt ui --host 0.0.0.0 --port 7860
116
+ melotts-mblt download
117
+ ```
118
+
119
+ `python -m melotts_mblt.cli` is equivalent to `melotts-mblt`. Run `melotts-mblt tts --help` for every TTS option.
120
+
121
+ ## Using with mblt-model-zoo
122
+
123
+ This package replaces `mblt_model_zoo.MeloTTS`. The modules have the same contents, and only the import prefix and the
124
+ command names differ:
125
+
126
+ | mblt-model-zoo | melotts-mblt |
127
+ | --- | --- |
128
+ | `from mblt_model_zoo.MeloTTS.api import TTS` | `from melotts_mblt import TTS` |
129
+ | `mblt_model_zoo.MeloTTS.<module>` | `melotts_mblt.<module>` |
130
+ | `mblt-model-zoo melo ...` / `mblt-model-zoo melotts ...` | `melotts-mblt tts ...` |
131
+ | `mblt-model-zoo melo-ui` | `melotts-mblt ui` |
132
+ | `mblt-melotts-download` | `melotts-mblt download` |
133
+
134
+ ## Development
135
+
136
+ Use [uv](https://docs.astral.sh/uv/) to manage the development environment:
137
+
138
+ ```bash
139
+ git clone https://github.com/mobilint/MeloTTS-mblt.git
140
+ cd MeloTTS-mblt
141
+ uv venv --python 3.12
142
+ source .venv/bin/activate
143
+ uv pip install -e . --group dev
144
+ pre-commit install
145
+ ```
146
+
147
+ The [test guide](tests/TEST.md) explains the hardware-free and NPU test runs.
148
+
149
+ ## Support and issues
150
+
151
+ For installation, model, or runtime support, visit the [Mobilint forum](https://discuss.mobilint.com/)
152
+ or contact [tech-support@mobilint.com](mailto:tech-support@mobilint.com). Report reproducible
153
+ package issues in the [MeloTTS-mblt issue tracker](https://github.com/mobilint/MeloTTS-mblt/issues).
154
+
155
+ ## License
156
+
157
+ Mobilint's code is distributed under the [BSD 3-Clause License](LICENSE). The MeloTTS-derived code in `melotts_mblt`
158
+ remains under [MyShell.ai's MIT License](melotts_mblt/LICENSE), and that license ships in the wheel. MeloTTS was
159
+ created by Wenliang Zhao, Xumin Yu, and Zengyi Qin; see the [API reference](melotts_mblt/README.md#original-authors)
160
+ for the original authors and the citation.
@@ -0,0 +1,117 @@
1
+ # melotts-mblt
2
+
3
+ <!-- markdownlint-disable MD033 -->
4
+ <div align="center">
5
+ <p>
6
+ <a href="https://www.mobilint.com/" target="_blank">
7
+ <img src="https://raw.githubusercontent.com/mobilint/.github/main/assets/Mobilint_Logo_Primary.png" alt="Mobilint Logo" width="60%">
8
+ </a>
9
+ </p>
10
+ </div>
11
+ <!-- markdownlint-enable MD033 -->
12
+
13
+ Run [MeloTTS](https://github.com/myshell-ai/MeloTTS) text-to-speech on Mobilint NPUs. `melotts-mblt` ships
14
+ pre-quantized MeloTTS synthesizers for English and Korean. It provides a Python API, a command line, and a Gradio
15
+ WebUI. The acoustic model runs on the NPU, and its Mobilint BERT prosody encoder loads through
16
+ [transformers-mblt](https://github.com/mobilint/transformers-mblt).
17
+
18
+ Models run on Mobilint [ARIES](https://www.mobilint.com/aries) and [REGULUS](https://www.mobilint.com/regulus)
19
+ boards. Supported target-device identifiers are `aries-rb`, `regulus-ra`, `regulus-rb`, `regulus-ra-usb`, and
20
+ `regulus-rb-usb`.
21
+
22
+ Version `0.0.0` is the first standalone release. It was extracted from `mblt-model-zoo` 2.11.0.
23
+
24
+ ## Installation
25
+
26
+ [![PyPI - Version](https://img.shields.io/pypi/v/melotts-mblt?logo=pypi&logoColor=white)](https://pypi.org/project/melotts-mblt/)
27
+ [![PyPI Downloads](https://static.pepy.tech/badge/melotts-mblt?period=total&units=INTERNATIONAL_SYSTEM&left_color=BLACK&right_color=GREEN&left_text=downloads)](https://clickpy.clickhouse.com/dashboard/melotts-mblt)
28
+ [![PyPI - Python Version](https://img.shields.io/pypi/pyversions/melotts-mblt?logo=python&logoColor=gold)](https://pypi.org/project/melotts-mblt/)
29
+
30
+ ```bash
31
+ pip install melotts-mblt
32
+ # One-time download of the NLTK tagger (English G2P) and the UniDic dictionary
33
+ melotts-mblt download
34
+ ```
35
+
36
+ NPU execution requires a supported Mobilint NPU driver and device. Python packaging cannot run post-install steps, so
37
+ run `melotts-mblt download` once after installing.
38
+
39
+ ## Quick start
40
+
41
+ ```python
42
+ from melotts_mblt import TTS
43
+
44
+ model = TTS(language="EN_NEWEST", device="cpu", trust_remote_code=True)
45
+ try:
46
+ speaker_ids = model.hps.data.spk2id
47
+ model.tts_to_file("Did you ever hear a folk tale about a giant turtle?", speaker_ids["EN-Newest"], "en.wav", speed=1.0)
48
+ finally:
49
+ model.dispose() # release the NPU backends
50
+ ```
51
+
52
+ `trust_remote_code=True` lets the Mobilint BERT prosody encoder load its Hub remote code; the `melotts-mblt` CLI and
53
+ WebUI pass it for you. Use `language="KR"` with speaker `"KR"` for Korean. Each language downloads its model from the Mobilint Hugging Face
54
+ organization: [`mobilint/MeloTTS-English-v3`](https://huggingface.co/mobilint/MeloTTS-English-v3) and
55
+ [`mobilint/MeloTTS-Korean`](https://huggingface.co/mobilint/MeloTTS-Korean).
56
+
57
+ These `TTS(...)` arguments select the NPU: `dev_no`, `target_core`, `target_device`, `encoder_mxq_path`, and
58
+ `decoder_mxq_path`. The [API reference](melotts_mblt/README.md) has more examples.
59
+
60
+ ## Command line
61
+
62
+ The `melotts-mblt` command provides three subcommands:
63
+
64
+ - `tts` synthesizes speech with the MeloTTS CLI.
65
+ - `ui` launches the Gradio WebUI.
66
+ - `download` fetches the NLTK and UniDic resources.
67
+
68
+ ```bash
69
+ melotts-mblt tts "Text to read" output.wav --language EN_NEWEST --speed 1.2
70
+ melotts-mblt tts "text-to-speech 안녕하세요" kr.wav --language KR
71
+ melotts-mblt tts file.txt out.wav --file
72
+ melotts-mblt ui --host 0.0.0.0 --port 7860
73
+ melotts-mblt download
74
+ ```
75
+
76
+ `python -m melotts_mblt.cli` is equivalent to `melotts-mblt`. Run `melotts-mblt tts --help` for every TTS option.
77
+
78
+ ## Using with mblt-model-zoo
79
+
80
+ This package replaces `mblt_model_zoo.MeloTTS`. The modules have the same contents, and only the import prefix and the
81
+ command names differ:
82
+
83
+ | mblt-model-zoo | melotts-mblt |
84
+ | --- | --- |
85
+ | `from mblt_model_zoo.MeloTTS.api import TTS` | `from melotts_mblt import TTS` |
86
+ | `mblt_model_zoo.MeloTTS.<module>` | `melotts_mblt.<module>` |
87
+ | `mblt-model-zoo melo ...` / `mblt-model-zoo melotts ...` | `melotts-mblt tts ...` |
88
+ | `mblt-model-zoo melo-ui` | `melotts-mblt ui` |
89
+ | `mblt-melotts-download` | `melotts-mblt download` |
90
+
91
+ ## Development
92
+
93
+ Use [uv](https://docs.astral.sh/uv/) to manage the development environment:
94
+
95
+ ```bash
96
+ git clone https://github.com/mobilint/MeloTTS-mblt.git
97
+ cd MeloTTS-mblt
98
+ uv venv --python 3.12
99
+ source .venv/bin/activate
100
+ uv pip install -e . --group dev
101
+ pre-commit install
102
+ ```
103
+
104
+ The [test guide](tests/TEST.md) explains the hardware-free and NPU test runs.
105
+
106
+ ## Support and issues
107
+
108
+ For installation, model, or runtime support, visit the [Mobilint forum](https://discuss.mobilint.com/)
109
+ or contact [tech-support@mobilint.com](mailto:tech-support@mobilint.com). Report reproducible
110
+ package issues in the [MeloTTS-mblt issue tracker](https://github.com/mobilint/MeloTTS-mblt/issues).
111
+
112
+ ## License
113
+
114
+ Mobilint's code is distributed under the [BSD 3-Clause License](LICENSE). The MeloTTS-derived code in `melotts_mblt`
115
+ remains under [MyShell.ai's MIT License](melotts_mblt/LICENSE), and that license ships in the wheel. MeloTTS was
116
+ created by Wenliang Zhao, Xumin Yu, and Zengyi Qin; see the [API reference](melotts_mblt/README.md#original-authors)
117
+ for the original authors and the citation.
@@ -0,0 +1,19 @@
1
+ Copyright (c) 2024 MyShell.ai
2
+
3
+ Permission is hereby granted, free of charge, to any person obtaining a copy
4
+ of this software and associated documentation files (the "Software"), to deal
5
+ in the Software without restriction, including without limitation the rights
6
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
7
+ copies of the Software, and to permit persons to whom the Software is
8
+ furnished to do so, subject to the following conditions:
9
+
10
+ The above copyright notice and this permission notice shall be included in all
11
+ copies or substantial portions of the Software.
12
+
13
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
14
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
15
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
16
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
17
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
18
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
19
+ SOFTWARE.
@@ -0,0 +1,26 @@
1
+ """MeloTTS text-to-speech for Mobilint NPUs.
2
+
3
+ ``from melotts_mblt import TTS`` loads the speech synthesizer. The import is lazy, so ``import melotts_mblt`` does not
4
+ import ``torch`` or ``transformers``.
5
+ """
6
+
7
+ from typing import TYPE_CHECKING
8
+
9
+ __version__ = "0.0.0"
10
+
11
+ if TYPE_CHECKING:
12
+ from .api import TTS
13
+
14
+ __all__ = ["TTS", "__version__"]
15
+
16
+
17
+ def __getattr__(name: str):
18
+ if name == "TTS":
19
+ from .api import TTS
20
+
21
+ return TTS
22
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
23
+
24
+
25
+ def __dir__() -> list[str]:
26
+ return sorted(set(globals()) | set(__all__))
@@ -0,0 +1,310 @@
1
+ import math
2
+ import re
3
+ from typing import Optional
4
+
5
+ import numpy as np
6
+ import soundfile
7
+ import torch
8
+ from torch import nn
9
+ from tqdm import tqdm
10
+ from transformers.models.auto.modeling_auto import AutoModelForMaskedLM
11
+ from transformers.models.auto.tokenization_auto import AutoTokenizer
12
+
13
+ from . import utils
14
+ from .download_utils import (
15
+ LANG_TO_HF_REPO_ID,
16
+ load_or_download_config,
17
+ load_or_download_model,
18
+ resolve_local_bert_mxq,
19
+ resolve_local_mxq,
20
+ )
21
+ from .models import MobilintSynthesizerTrn
22
+ from .split_utils import split_sentence
23
+ from .text.cleaner import clean_text
24
+
25
+
26
+ def _describe_value(value):
27
+ """``repr(value)`` for error messages, falling back to the type when ``repr`` itself fails (e.g. huge ints)."""
28
+ try:
29
+ return repr(value)
30
+ except Exception:
31
+ return f"<{type(value).__name__}>"
32
+
33
+
34
+ def _validate_speed(speed):
35
+ """Return ``speed`` as a ``float`` if it is a finite real number greater than 0, else raise ``ValueError``.
36
+
37
+ Any real scalar ``float()`` accepts is allowed (``int``, ``float``, NumPy scalars, 0-d tensors, ``Fraction``,
38
+ ``Decimal``). Booleans and strings are rejected even though ``float()`` would convert them.
39
+ """
40
+ message = "speed must be a finite number greater than 0, got {}"
41
+ if isinstance(speed, (bool, np.bool_, str, bytes)):
42
+ raise ValueError(message.format(_describe_value(speed)))
43
+ try:
44
+ value = float(speed)
45
+ except (TypeError, ValueError, OverflowError):
46
+ # OverflowError: a finite real too large for a float, e.g. 10**10000.
47
+ raise ValueError(message.format(_describe_value(speed))) from None
48
+ if not math.isfinite(value) or value <= 0:
49
+ raise ValueError(message.format(_describe_value(speed)))
50
+ return value
51
+
52
+
53
+ class TTS(nn.Module):
54
+ def __init__(self,
55
+ language,
56
+ device='auto',
57
+ config_path=None,
58
+ ckpt_path=None,
59
+
60
+ trust_remote_code: Optional[bool]=None,
61
+ local_files_only: Optional[bool]=None,
62
+
63
+ dev_no: Optional[int] = None,
64
+ target_core: Optional[str] = None,
65
+ encoder_mxq_path: Optional[str] = None,
66
+ decoder_mxq_path: Optional[str] = None,
67
+ target_device: Optional[str] = None,
68
+ ):
69
+ nn.Module.__init__(self)
70
+ if device == "auto":
71
+ device = "cpu"
72
+ if torch.cuda.is_available():
73
+ device = "cuda"
74
+ if torch.backends.mps.is_available():
75
+ device = "mps"
76
+ if "cuda" in device:
77
+ assert torch.cuda.is_available()
78
+
79
+ # config_path =
80
+ hps = load_or_download_config(language, config_path=config_path, local_files_only=local_files_only)
81
+
82
+ if dev_no is not None:
83
+ hps.model.dev_no = dev_no
84
+
85
+ if target_core is not None:
86
+ hps.model.target_core = target_core
87
+
88
+ if encoder_mxq_path is not None:
89
+ hps.model.encoder_mxq_path = encoder_mxq_path
90
+
91
+ if decoder_mxq_path is not None:
92
+ hps.model.decoder_mxq_path = decoder_mxq_path
93
+
94
+ resolved_target_device = (
95
+ target_device
96
+ if target_device is not None
97
+ else getattr(hps.model, "target_device", "aries-rb")
98
+ )
99
+ hps.model.target_device = resolved_target_device
100
+
101
+ if local_files_only:
102
+ # mblt_npu downloads a missing MXQ from the Hub; resolve both synthesizer MXQs from the cache first so
103
+ # local_files_only never reaches the network (LocalEntryNotFoundError when they are not cached).
104
+ repo_id = LANG_TO_HF_REPO_ID[language]
105
+ hps.model.encoder_mxq_path = resolve_local_mxq(repo_id, hps.model.encoder_mxq_path)
106
+ hps.model.decoder_mxq_path = resolve_local_mxq(repo_id, hps.model.decoder_mxq_path)
107
+
108
+ num_languages = hps.num_languages
109
+ num_tones = hps.num_tones
110
+ symbols = hps.symbols
111
+
112
+ # A failure while building the synthesizer is cleaned up by MobilintSynthesizerTrn itself; from here on the
113
+ # NPU backends exist, so assign self.model before anything else can fail (including the device transfer).
114
+ self.model = MobilintSynthesizerTrn(
115
+ len(symbols),
116
+ hps.data.filter_length // 2 + 1,
117
+ hps.train.segment_size // hps.data.hop_length,
118
+ n_speakers=hps.data.n_speakers,
119
+ num_tones=num_tones,
120
+ num_languages=num_languages,
121
+ name_or_path=LANG_TO_HF_REPO_ID[language],
122
+ **hps.model,
123
+ )
124
+
125
+ try:
126
+ model = self.model.to(device)
127
+ model.eval()
128
+ self.model = model
129
+ self.symbol_to_id = {s: i for i, s in enumerate(symbols)}
130
+ self.hps = hps
131
+ self.device = device
132
+
133
+ # load state_dict
134
+ checkpoint_dict = load_or_download_model(language, device, ckpt_path=ckpt_path, local_files_only=local_files_only)
135
+ self.model.load_state_dict(checkpoint_dict['model'], strict=True)
136
+
137
+ language = language.split('_')[0]
138
+ self.language = 'ZH_MIX_EN' if language == 'ZH' else language # we support a ZH_MIX_EN model
139
+
140
+ self.tokenizer = AutoTokenizer.from_pretrained(
141
+ hps.model.bert_model_id,
142
+ trust_remote_code=trust_remote_code,
143
+ local_files_only=local_files_only,
144
+ )
145
+
146
+ bert_kwargs = {}
147
+ if local_files_only:
148
+ # Same reason as the synthesizer MXQs: resolve the BERT MXQ from the cache before mblt_npu sees it.
149
+ bert_mxq_path = resolve_local_bert_mxq(hps.model.bert_model_id)
150
+ if bert_mxq_path is not None:
151
+ bert_kwargs["mxq_path"] = bert_mxq_path
152
+
153
+ # Keep a reference to the loaded BERT (it owns an NPU backend) before the transfer, so dispose() can
154
+ # still release it if .to(device) fails.
155
+ self.bert = AutoModelForMaskedLM.from_pretrained(
156
+ hps.model.bert_model_id,
157
+ trust_remote_code=trust_remote_code,
158
+ local_files_only=local_files_only,
159
+
160
+ dev_no=hps.model.dev_no,
161
+ target_cores=[hps.model.target_core],
162
+ target_device=hps.model.target_device,
163
+ **bert_kwargs,
164
+ )
165
+ self.bert = self.bert.to(device)
166
+
167
+ except BaseException:
168
+ # Checkpoint, tokenizer, or BERT loading failed after the synthesizer's NPU backends were created.
169
+ self.dispose()
170
+ raise
171
+
172
+ @staticmethod
173
+ def audio_numpy_concat(segment_data_list, sr, speed=1.0):
174
+ audio_segments = []
175
+ for segment_data in segment_data_list:
176
+ audio_segments += segment_data.reshape(-1).tolist()
177
+ audio_segments += [0] * int((sr * 0.05) / speed)
178
+ audio_segments = np.array(audio_segments).astype(np.float32)
179
+ return audio_segments
180
+
181
+ @staticmethod
182
+ def split_sentences_into_pieces(text, language, quiet=False):
183
+ texts = split_sentence(text, language_str=language)
184
+ if not quiet:
185
+ print(" > Text split to sentences.")
186
+ print("\n".join(texts))
187
+ print(" > ===========================")
188
+ return texts
189
+
190
+ def _bert_token_count(self, text, language):
191
+ """Number of BERT tokens (special tokens included) the text produces after normalization."""
192
+ if language in ["EN", "ZH_MIX_EN"]:
193
+ text = re.sub(r"([a-z])([A-Z])", r"\1 \2", text)
194
+ _, _, _, word2ph = clean_text(text, language, tokenizer=self.tokenizer)
195
+ return len(word2ph)
196
+
197
+ def _bert_max_tokens(self):
198
+ config = getattr(getattr(self, "bert", None), "config", None)
199
+ limit = getattr(config, "max_position_embeddings", None) or getattr(self.tokenizer, "model_max_length", None)
200
+ # Tokenizers without a configured maximum report a huge sentinel value; fall back to BERT's usual 512.
201
+ return int(limit) if limit and limit < 100_000 else 512
202
+
203
+ def _fit_piece_to_bert(self, text, language, limit):
204
+ """Split ``text`` until every piece fits BERT's position limit.
205
+
206
+ Sentence splitting only breaks at punctuation, so unpunctuated input can exceed BERT's position embeddings,
207
+ which are computed for the whole input before the MXQ runs (synthesizer chunking cannot help). Bisect at
208
+ whitespace, or at the middle character for an unbroken run; each piece then goes through normalization,
209
+ G2P, and BERT on its own, so ``word2ph`` / phoneme alignment is recomputed per piece instead of truncated.
210
+ """
211
+ if self._bert_token_count(text, language) <= limit:
212
+ return [text]
213
+ words = text.split()
214
+ if len(words) > 1:
215
+ middle = len(words) // 2
216
+ left, right = " ".join(words[:middle]), " ".join(words[middle:])
217
+ else:
218
+ middle = len(text) // 2
219
+ left, right = text[:middle], text[middle:]
220
+ if not left.strip() or not right.strip():
221
+ return [text]
222
+ return self._fit_piece_to_bert(left, language, limit) + self._fit_piece_to_bert(right, language, limit)
223
+
224
+ def fit_pieces_to_bert(self, texts, language):
225
+ """Return ``texts`` with every piece split as needed to fit BERT's maximum input length."""
226
+ if getattr(self.hps.data, "disable_bert", False):
227
+ return list(texts)
228
+ limit = self._bert_max_tokens()
229
+ pieces = []
230
+ for text in texts:
231
+ pieces.extend(self._fit_piece_to_bert(text, language, limit))
232
+ return pieces
233
+
234
+ def tts_to_file(self, text, speaker_id, output_path=None, sdp_ratio=0.2, noise_scale=0.6, noise_scale_w=0.8, speed=1.0, pbar=None, format=None, position=None, quiet=False):
235
+ speed = _validate_speed(speed)
236
+ language = self.language
237
+ texts = self.fit_pieces_to_bert(self.split_sentences_into_pieces(text, language, quiet), language)
238
+ audio_list = []
239
+ if pbar:
240
+ tx = pbar(texts)
241
+ else:
242
+ if position:
243
+ tx = tqdm(texts, position=position)
244
+ elif quiet:
245
+ tx = texts
246
+ else:
247
+ tx = tqdm(texts)
248
+ for t in tx:
249
+ if language in ["EN", "ZH_MIX_EN"]:
250
+ t = re.sub(r"([a-z])([A-Z])", r"\1 \2", t)
251
+ device = self.device
252
+ bert, ja_bert, phones, tones, lang_ids = utils.get_text_for_tts_infer(t, language, self.hps, device, self.symbol_to_id, tokenizer=self.tokenizer, bert=self.bert)
253
+ with torch.no_grad():
254
+ x_tst = phones.to(device).unsqueeze(0)
255
+ tones = tones.to(device).unsqueeze(0)
256
+ lang_ids = lang_ids.to(device).unsqueeze(0)
257
+ bert = bert.to(device).unsqueeze(0)
258
+ ja_bert = ja_bert.to(device).unsqueeze(0)
259
+ x_tst_lengths = torch.LongTensor([phones.size(0)]).to(device)
260
+ del phones
261
+ speakers = torch.LongTensor([speaker_id]).to(device)
262
+ audio = (
263
+ self.model.infer(
264
+ x_tst,
265
+ x_tst_lengths,
266
+ speakers,
267
+ tones,
268
+ lang_ids,
269
+ bert,
270
+ ja_bert,
271
+ sdp_ratio=sdp_ratio,
272
+ noise_scale=noise_scale,
273
+ noise_scale_w=noise_scale_w,
274
+ length_scale=1.0 / speed,
275
+ )[0][0, 0]
276
+ .data.cpu()
277
+ .float()
278
+ .numpy()
279
+ )
280
+ del x_tst, tones, lang_ids, bert, ja_bert, x_tst_lengths, speakers
281
+ #
282
+ audio_list.append(audio)
283
+ torch.cuda.empty_cache()
284
+ audio = self.audio_numpy_concat(audio_list, sr=self.hps.data.sampling_rate, speed=speed)
285
+
286
+ if output_path is None:
287
+ return audio
288
+ else:
289
+ if format:
290
+ soundfile.write(
291
+ output_path, audio, self.hps.data.sampling_rate, format=format
292
+ )
293
+ else:
294
+ soundfile.write(output_path, audio, self.hps.data.sampling_rate)
295
+
296
+ def launch(self):
297
+ self.model.launch()
298
+
299
+ def dispose(self):
300
+ model = getattr(self, "model", None)
301
+ if model is not None:
302
+ model.dispose()
303
+ # ``self.bert`` is a sibling NPU-backed module built via
304
+ # ``AutoModelForMaskedLM.from_pretrained``; if the loaded class is a
305
+ # Mobilint Bert (``MobilintBertForMaskedLM``) it owns its own NPU
306
+ # backend and must be released alongside the synthesizer, otherwise
307
+ # LPDDR stays pinned between TTS instances.
308
+ bert_dispose = getattr(getattr(self, "bert", None), "dispose", None)
309
+ if callable(bert_dispose):
310
+ bert_dispose()