termux-tts 1.4.4 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +25 -1
- package/README.md +284 -71
- package/README.pypi.md +92 -31
- package/doc.config.yaml +8 -9
- package/package.json +1 -1
- package/pyproject.toml +2 -2
- package/setup.py +1 -1
- package/termux_tts/__init__.py +1 -5
- package/termux_tts/cli.py +10 -3
- package/termux_tts/engine.py +47 -28
- package/termux_tts/engine_multilingual.py +41 -6
- package/termux_tts/engine_sherpa.py +155 -16
- package/termux_tts/engine_sherpa_capi.py +23 -6
- package/termux_tts/engine_vulkan.py +258 -103
- package/termux_tts/hardware.py +28 -16
- package/termux_tts/installer.py +22 -1
- package/termux_tts/tokenizer_melo.py +159 -0
- package/termux_tts/transliteration.py +250 -0
- package/termux_tts/engine_dsp.py +0 -340
package/doc.config.yaml
CHANGED
|
@@ -8,8 +8,8 @@ name: "termux-tts"
|
|
|
8
8
|
display_name: "Termux-TTS"
|
|
9
9
|
package_name_pypi: "termux-tts"
|
|
10
10
|
package_name_npm: "termux-tts"
|
|
11
|
-
version: "v1.
|
|
12
|
-
release_name: "
|
|
11
|
+
version: "v1.5.0"
|
|
12
|
+
release_name: "100% Native Vulkan Hardware GPU Pipeline & Zero-Silent-Fallback Architecture"
|
|
13
13
|
license: "Apache-2.0"
|
|
14
14
|
platform: "Android ARM64 / Qualcomm Adreno & ARM Mali Vulkan 1.3 / Linux"
|
|
15
15
|
github_repo_url: "https://github.com/uno-km/termux-tts"
|
|
@@ -142,14 +142,13 @@ code_example_js: |
|
|
|
142
142
|
main();
|
|
143
143
|
|
|
144
144
|
benchmarks:
|
|
145
|
-
headers: ["Hardware Target", "SoC / GPU Architecture", "
|
|
145
|
+
headers: ["Hardware Target", "SoC / Physical GPU Architecture", "Vulkan Driver ABI", "Audio Duration", "Synthesis Latency", "Real-Time Factor (RTF)", "Execution Status"]
|
|
146
146
|
rows:
|
|
147
|
-
- ["Galaxy
|
|
148
|
-
- ["Galaxy S25", "Snapdragon 8 Elite / Adreno 830", "Vulkan
|
|
149
|
-
- ["Galaxy
|
|
150
|
-
- ["Galaxy A35", "Exynos 1380 / ARM Mali-G68 MP5", "Vulkan
|
|
151
|
-
- ["Heterogeneous
|
|
152
|
-
- ["Android Physical Speaker", "AudioTrack / OpenSL ES Bridge", "Android Native Service", "N/A", "0.012 s", "0.0020x", "Validated (Hardware Out)"]
|
|
147
|
+
- ["Galaxy S21", "Exynos 2100 / ARM Mali-G78 (0x9800000)", "Vulkan 1.1 (/system/lib64)", "4.25 s", "9.37 s", "2.2055x", "Validated (100% Native GPU)"]
|
|
148
|
+
- ["Galaxy S25", "Snapdragon 8 Elite / Adreno 830 (0x80320040)", "Vulkan 1.3 (/system/lib64)", "4.25 s", "16.21 s", "3.8145x", "Validated (100% Native GPU)"]
|
|
149
|
+
- ["Galaxy S22", "Snapdragon 8 Gen 1 / Adreno 730 (0x80267062)", "Vulkan 1.1 (/system/lib64)", "4.27 s", "38.09 s", "8.9159x", "Validated (100% Native GPU)"]
|
|
150
|
+
- ["Galaxy A35", "Exynos 1380 / ARM Mali-G68 MP5 (0x9801000)", "Vulkan 1.1 (/system/lib64)", "4.27 s", "51.91 s", "12.1505x", "Validated (100% Native GPU)"]
|
|
151
|
+
- ["Heterogeneous CPU", "Cortex-A78 / A55 CPU Core", "Sherpa NEON SIMD C++", "4.25 s", "1.85 s", "0.4350x", "Validated (CPU Reference)"]
|
|
153
152
|
|
|
154
153
|
api_reference:
|
|
155
154
|
- symbol: "termux_tts.load(engine='auto'|'vulkan'|'sherpa'|'dsp'|'native', model_tier='high'|'medium', preset='balanced')"
|
package/package.json
CHANGED
package/pyproject.toml
CHANGED
|
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "termux-tts"
|
|
7
|
-
version = "1.
|
|
8
|
-
description = "On-device 4-Tier Text-to-Speech framework utilizing device resources (
|
|
7
|
+
version = "1.5.0"
|
|
8
|
+
description = "On-device 4-Tier Text-to-Speech framework utilizing device resources (C++ Sherpa-ONNX Neural, Android Native & Expressive)"
|
|
9
9
|
readme = "README.pypi.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
license = { text = "Apache-2.0" }
|
package/setup.py
CHANGED
|
@@ -3,7 +3,7 @@ from setuptools import setup, find_packages
|
|
|
3
3
|
|
|
4
4
|
setup(
|
|
5
5
|
name="termux-tts",
|
|
6
|
-
version="1.
|
|
6
|
+
version="1.5.0",
|
|
7
7
|
description="Ultra-Fast On-Device 4-Tier Text-to-Speech Framework (DSP Synth, Android Native, C++ Sherpa-ONNX Neural & Expressive)",
|
|
8
8
|
long_description=open("README.pypi.md", encoding="utf-8").read() if os.path.exists("README.pypi.md") else open("README.md", encoding="utf-8").read(),
|
|
9
9
|
long_description_content_type="text/markdown",
|
package/termux_tts/__init__.py
CHANGED
|
@@ -8,7 +8,6 @@ termux-tts: Production-Grade 4-Tier TTS Framework for Android Termux.
|
|
|
8
8
|
|
|
9
9
|
from .engine import TTSEngine, load, doctor
|
|
10
10
|
from .engine_native import NativeAndroidEngine, NativeResult
|
|
11
|
-
from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS, DSPSynthesizer
|
|
12
11
|
from .engine_sherpa import SherpaNeuralEngine, SherpaResult
|
|
13
12
|
from .engine_vulkan import VulkanNeuralEngine, VulkanResult
|
|
14
13
|
from .engine_expressive import ExpressiveEngine, ExpressiveResult
|
|
@@ -32,14 +31,11 @@ from .exceptions import (
|
|
|
32
31
|
ONNXNeuralEngine = SherpaNeuralEngine
|
|
33
32
|
ONNXResult = SherpaResult
|
|
34
33
|
|
|
35
|
-
__version__ = "1.
|
|
34
|
+
__version__ = "1.5.0"
|
|
36
35
|
__all__ = [
|
|
37
36
|
"TTSEngine",
|
|
38
37
|
"load",
|
|
39
38
|
"doctor",
|
|
40
|
-
"ParametricDSPEngine",
|
|
41
|
-
"DSPSynthesizer",
|
|
42
|
-
"DSPResult",
|
|
43
39
|
"NativeAndroidEngine",
|
|
44
40
|
"NativeResult",
|
|
45
41
|
"SherpaNeuralEngine",
|
package/termux_tts/cli.py
CHANGED
|
@@ -31,8 +31,8 @@ def main():
|
|
|
31
31
|
synth_parser.add_argument("-l", "--lang", default="auto", help="Language code (auto=Multi-language auto switch, ko/kor=Korean only, en/eng=English only, ja/jpn=Japanese only)")
|
|
32
32
|
synth_parser.add_argument(
|
|
33
33
|
"-e", "--engine", default="auto",
|
|
34
|
-
choices=["auto", "vulkan", "ncnn", "gpu", "
|
|
35
|
-
help="Synthesis engine tier (auto=Smart Routing, vulkan=GPU NCNN,
|
|
34
|
+
choices=["auto", "vulkan", "ncnn", "gpu", "native", "neural", "onnx", "expressive", "multilingual", "hybrid", "kokoro", "melo", "supertonic"],
|
|
35
|
+
help="Synthesis engine tier (auto=Smart Routing, vulkan=GPU NCNN, native=Android voice, neural=VITS C++, kokoro=StyleTTS2 82M, melo=MeloTTS Bilingual, supertonic=Supertonic On-Device)"
|
|
36
36
|
)
|
|
37
37
|
synth_parser.add_argument("-m", "--model", default=None, help="Path to model file or directory")
|
|
38
38
|
synth_parser.add_argument("-p", "--preset", default="balanced", choices=["fast", "balanced", "expressive", "ultra"])
|
|
@@ -42,6 +42,7 @@ def main():
|
|
|
42
42
|
synth_parser.add_argument("--cpu", dest="device", action="store_const", const="cpu", help="Force CPU compute mode")
|
|
43
43
|
synth_parser.add_argument("--tier", default=None, choices=["high", "medium", "balanced", "fast", "ultra"], help="Target model tier (high=Studio FP16, medium=Balanced)")
|
|
44
44
|
synth_parser.add_argument("-s", "--speed", type=float, default=1.0, help="Speech speed multiplier (0.5 to 2.0)")
|
|
45
|
+
synth_parser.add_argument("--mode", default="unified", choices=["unified", "stitch"], help="Multilingual synthesis mode (unified=Single-pass G2P transliteration [BigTech Standard], stitch=Multi-model chunk concatenation)")
|
|
45
46
|
synth_parser.add_argument("--threads", type=int, default=4, help="Compute worker threads (ARM NEON)")
|
|
46
47
|
synth_parser.add_argument("--volume", type=int, default=None, help="Set Android media volume (1 to 15)")
|
|
47
48
|
synth_parser.add_argument("--play", action="store_true", help="Play synthesized audio through physical speaker immediately")
|
|
@@ -101,7 +102,13 @@ def main():
|
|
|
101
102
|
engine=args.engine,
|
|
102
103
|
tier=getattr(args, "tier", None),
|
|
103
104
|
) as engine:
|
|
104
|
-
res = engine.synthesize(
|
|
105
|
+
res = engine.synthesize(
|
|
106
|
+
target_text,
|
|
107
|
+
output=out_path,
|
|
108
|
+
speed=args.speed,
|
|
109
|
+
language=args.lang,
|
|
110
|
+
mode=getattr(args, "mode", "unified")
|
|
111
|
+
)
|
|
105
112
|
backend_name = getattr(res, "backend", "UNKNOWN")
|
|
106
113
|
model_name = getattr(res, "model_name", "model")
|
|
107
114
|
dur = getattr(res, "duration_sec", 0.0)
|
package/termux_tts/engine.py
CHANGED
|
@@ -18,7 +18,6 @@ from typing import Optional, Union, Dict, Any
|
|
|
18
18
|
|
|
19
19
|
from .exceptions import TTSInferenceError, TTSModelLoadError, VulkanInitializationError
|
|
20
20
|
from .engine_native import NativeAndroidEngine, NativeResult
|
|
21
|
-
from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS
|
|
22
21
|
from .engine_sherpa import SherpaNeuralEngine, SherpaResult
|
|
23
22
|
from .engine_vulkan import VulkanNeuralEngine, VulkanResult
|
|
24
23
|
from .engine_expressive import ExpressiveEngine, ExpressiveResult
|
|
@@ -34,6 +33,14 @@ from .hardware import (
|
|
|
34
33
|
logger = logging.getLogger("termux_tts.engine")
|
|
35
34
|
|
|
36
35
|
|
|
36
|
+
QUALITY_PRESETS = {
|
|
37
|
+
"fast": {"sample_rate": 16000, "description": "16kHz Low Latency"},
|
|
38
|
+
"balanced": {"sample_rate": 22050, "description": "22.05kHz Standard Audio"},
|
|
39
|
+
"expressive": {"sample_rate": 24000, "description": "24kHz Expressive High Fidelity"},
|
|
40
|
+
"ultra": {"sample_rate": 44100, "description": "44.1kHz Studio Master"},
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
37
44
|
class TTSEngine:
|
|
38
45
|
"""Production 4-Tier Multi-Backend Gateway supporting Synth, Native, Neural, and Expressive engines."""
|
|
39
46
|
|
|
@@ -86,6 +93,18 @@ class TTSEngine:
|
|
|
86
93
|
if norm_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar"):
|
|
87
94
|
return self._get_multilingual_engine()
|
|
88
95
|
|
|
96
|
+
# Explicit Vulkan GPU MeloTTS Tier (Plan 1 NCNN Sliced / Plan 2 MNN Vulkan)
|
|
97
|
+
if t in ("melo", "melo_vulkan", "melo_ncnn", "melo_mnn") and (self.requested_device in ("vulkan", "gpu") or self.device == "vulkan"):
|
|
98
|
+
return VulkanNeuralEngine(
|
|
99
|
+
model_path=self.model_path,
|
|
100
|
+
language=self.language,
|
|
101
|
+
device=self.requested_device,
|
|
102
|
+
threads=self.threads,
|
|
103
|
+
sample_rate=self.sample_rate or 44100,
|
|
104
|
+
model_tier=self.model_tier,
|
|
105
|
+
model_type=t,
|
|
106
|
+
)
|
|
107
|
+
|
|
89
108
|
# Explicit Vulkan GPU Tier (Vulkan NCNN engine targets English Lessac)
|
|
90
109
|
if (t in ("vulkan", "gpu", "ncnn") or (self.requested_device in ("vulkan", "gpu") and t in ("neural", "vits", "auto"))) and norm_lang in ("en", "auto"):
|
|
91
110
|
try:
|
|
@@ -96,6 +115,7 @@ class TTSEngine:
|
|
|
96
115
|
threads=self.threads,
|
|
97
116
|
sample_rate=self.sample_rate or 22050,
|
|
98
117
|
model_tier=self.model_tier,
|
|
118
|
+
model_type="vits",
|
|
99
119
|
)
|
|
100
120
|
except (VulkanInitializationError, TTSModelLoadError) as err:
|
|
101
121
|
if t in ("vulkan", "gpu", "ncnn") or self.requested_device in ("vulkan", "gpu"):
|
|
@@ -113,24 +133,25 @@ class TTSEngine:
|
|
|
113
133
|
model_type="vits",
|
|
114
134
|
)
|
|
115
135
|
|
|
116
|
-
#
|
|
117
|
-
elif t in ("
|
|
118
|
-
return
|
|
136
|
+
# BigTech 3rd-Party Neural Speech Engines (StyleTTS2/Kokoro, MeloTTS, Supertonic)
|
|
137
|
+
elif t in ("kokoro", "melo", "supertonic"):
|
|
138
|
+
return SherpaNeuralEngine(
|
|
119
139
|
model_path=self.model_path,
|
|
120
140
|
language=self.language,
|
|
121
141
|
device=self.requested_device,
|
|
122
142
|
threads=self.threads,
|
|
123
143
|
sample_rate=self.sample_rate or 22050,
|
|
144
|
+
model_type=t,
|
|
124
145
|
)
|
|
125
146
|
|
|
126
|
-
# Explicit Tier
|
|
127
|
-
elif t in ("
|
|
128
|
-
return
|
|
147
|
+
# Explicit Tier 4: Expressive (Fail-Fast)
|
|
148
|
+
elif t in ("expressive", "chat", "conversational"):
|
|
149
|
+
return ExpressiveEngine(
|
|
129
150
|
model_path=self.model_path,
|
|
130
151
|
language=self.language,
|
|
131
|
-
preset=self.preset,
|
|
132
152
|
device=self.requested_device,
|
|
133
|
-
|
|
153
|
+
threads=self.threads,
|
|
154
|
+
sample_rate=self.sample_rate or 22050,
|
|
134
155
|
)
|
|
135
156
|
|
|
136
157
|
# Explicit Multilingual / Code-Switching Tier
|
|
@@ -141,7 +162,7 @@ class TTSEngine:
|
|
|
141
162
|
elif t == "native":
|
|
142
163
|
return self.native_engine
|
|
143
164
|
|
|
144
|
-
# Auto Mode
|
|
165
|
+
# Auto Mode (Zero-Silent-Fallback)
|
|
145
166
|
elif t == "auto":
|
|
146
167
|
# 1. Check if SherpaNeuralEngine assets exist
|
|
147
168
|
try:
|
|
@@ -159,18 +180,17 @@ class TTSEngine:
|
|
|
159
180
|
if self.native_engine.binary:
|
|
160
181
|
return self.native_engine
|
|
161
182
|
|
|
162
|
-
# 3.
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
sample_rate=self.sample_rate,
|
|
183
|
+
# 3. Fail-Fast: No silent robotic fallback
|
|
184
|
+
raise TTSModelLoadError(
|
|
185
|
+
"[FAIL-FAST] No TTS voice engine available. Neural speech model is not installed, "
|
|
186
|
+
"and native Android TTS engine is unavailable.\n"
|
|
187
|
+
" To install official neural models automatically, run:\n"
|
|
188
|
+
" termux-tts install\n"
|
|
169
189
|
)
|
|
170
190
|
else:
|
|
171
191
|
raise TTSInferenceError(
|
|
172
192
|
f"[FAIL-FAST] Unknown engine_type '{self.requested_engine_type}'. "
|
|
173
|
-
f"Available tiers: ['auto', '
|
|
193
|
+
f"Available tiers: ['auto', 'native', 'neural', 'expressive', 'multilingual', 'kokoro', 'melo', 'supertonic']"
|
|
174
194
|
)
|
|
175
195
|
|
|
176
196
|
@property
|
|
@@ -207,7 +227,8 @@ class TTSEngine:
|
|
|
207
227
|
speed: float = 1.0,
|
|
208
228
|
preset: Optional[str] = None,
|
|
209
229
|
language: Optional[str] = None,
|
|
210
|
-
|
|
230
|
+
mode: str = "unified",
|
|
231
|
+
) -> Union[SherpaResult, ExpressiveResult, NativeResult, MultilingualResult]:
|
|
211
232
|
"""Synthesize text into speech audio buffer / WAV file with Zero-Config intelligent routing."""
|
|
212
233
|
if self._is_closed:
|
|
213
234
|
raise TTSInferenceError("Cannot synthesize: Engine session is closed.")
|
|
@@ -219,24 +240,21 @@ class TTSEngine:
|
|
|
219
240
|
from .script_classifier import normalize_language_code
|
|
220
241
|
target_lang = normalize_language_code(language or self.language)
|
|
221
242
|
|
|
222
|
-
# 1. If user explicitly pinned to
|
|
223
|
-
if self.requested_engine_type in ("dsp", "synth", "formant") or self.device == "dsp":
|
|
224
|
-
return self.synth_engine.synthesize(clean_text, output=output, speed=speed, preset=preset)
|
|
225
|
-
|
|
226
|
-
# 2. If user explicitly pinned to OS Native voice, speak directly
|
|
243
|
+
# 1. If user explicitly pinned to OS Native voice, speak directly
|
|
227
244
|
if self.requested_engine_type == "native":
|
|
228
245
|
return self.native_engine.speak(clean_text)
|
|
229
246
|
|
|
230
|
-
#
|
|
247
|
+
# 2. Multilingual auto-detection:
|
|
231
248
|
# Route to MultilingualNeuralEngine if multiple languages are present in text
|
|
232
249
|
# or if caller specifically requested multilingual / hybrid engine.
|
|
233
250
|
detected_langs = MultilingualTokenizer.detect_languages(clean_text)
|
|
234
251
|
is_mixed_text = len(detected_langs) > 1
|
|
235
252
|
|
|
236
253
|
should_route_multilingual = (
|
|
237
|
-
(self.requested_engine_type in ("
|
|
238
|
-
(
|
|
239
|
-
(
|
|
254
|
+
(self.requested_engine_type in ("multilingual", "codeswitch", "hybrid")) or
|
|
255
|
+
(self.requested_engine_type == "auto" and (is_mixed_text or self._multilingual_engine is not None)) or
|
|
256
|
+
(is_mixed_text and self.requested_engine_type != "native") or
|
|
257
|
+
(target_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar") and self.requested_engine_type != "native")
|
|
240
258
|
)
|
|
241
259
|
|
|
242
260
|
if should_route_multilingual:
|
|
@@ -248,6 +266,7 @@ class TTSEngine:
|
|
|
248
266
|
clean_text,
|
|
249
267
|
output=output,
|
|
250
268
|
speed=speed,
|
|
269
|
+
mode=mode,
|
|
251
270
|
**kwargs
|
|
252
271
|
)
|
|
253
272
|
|
|
@@ -8,6 +8,7 @@ from __future__ import annotations
|
|
|
8
8
|
|
|
9
9
|
import logging
|
|
10
10
|
import os
|
|
11
|
+
import re
|
|
11
12
|
import shutil
|
|
12
13
|
import subprocess
|
|
13
14
|
import tempfile
|
|
@@ -284,10 +285,12 @@ class MultilingualNeuralEngine:
|
|
|
284
285
|
output: Optional[str] = None,
|
|
285
286
|
speed: float = 1.0,
|
|
286
287
|
language: Optional[str] = None,
|
|
288
|
+
mode: str = "unified",
|
|
287
289
|
) -> MultilingualResult:
|
|
288
290
|
"""
|
|
289
|
-
Synthesizes speech with zero cross-
|
|
290
|
-
|
|
291
|
+
Synthesizes speech with natural prosody and zero cross-speaker distortion.
|
|
292
|
+
Defaults to 'unified' single-pass mode (phonetic transliteration preserving prosody).
|
|
293
|
+
'stitch' mode retains explicit multi-model chunk concatenation when requested.
|
|
291
294
|
"""
|
|
292
295
|
if self._is_closed:
|
|
293
296
|
raise TTSInferenceError("Cannot synthesize: Multilingual session is closed.")
|
|
@@ -297,17 +300,50 @@ class MultilingualNeuralEngine:
|
|
|
297
300
|
raise TTSInferenceError("Cannot synthesize empty text.")
|
|
298
301
|
|
|
299
302
|
t0 = time.perf_counter()
|
|
303
|
+
target_sample_rate = self.sample_rate
|
|
300
304
|
|
|
301
|
-
#
|
|
305
|
+
# -------------------------------------------------------------------
|
|
306
|
+
# Mode 1: Unified Single-Pass Pipeline (BigTech Architecture Standard)
|
|
307
|
+
# -------------------------------------------------------------------
|
|
308
|
+
has_korean = bool(re.search(r"[\uac00-\ud7a3]", clean_text))
|
|
309
|
+
has_latin = bool(re.search(r"[a-zA-Z]", clean_text))
|
|
310
|
+
|
|
311
|
+
if mode == "unified" and (language == "ko" or (language is None and has_korean)):
|
|
312
|
+
from .transliteration import transliterate_mixed_text
|
|
313
|
+
unified_text = transliterate_mixed_text(clean_text)
|
|
314
|
+
logger.info("[Multilingual] Unified single-pass execution: '%s' -> '%s'", clean_text, unified_text)
|
|
315
|
+
|
|
316
|
+
chunk_buf = self._synthesize_chunk_audio(unified_text, "ko", speed=speed)
|
|
317
|
+
elapsed_ms = (time.perf_counter() - t0) * 1000.0
|
|
318
|
+
dur_sec = chunk_buf.duration_seconds
|
|
319
|
+
rtf = (elapsed_ms / 1000.0) / max(0.001, dur_sec)
|
|
320
|
+
|
|
321
|
+
if output:
|
|
322
|
+
chunk_buf.save(output)
|
|
323
|
+
|
|
324
|
+
dummy_chunk = LanguageChunk(text=clean_text, language="ko", pause_after=0.0)
|
|
325
|
+
return MultilingualResult(
|
|
326
|
+
text=clean_text,
|
|
327
|
+
audio_buffer=chunk_buf,
|
|
328
|
+
sample_rate=target_sample_rate,
|
|
329
|
+
duration_sec=dur_sec,
|
|
330
|
+
elapsed_ms=elapsed_ms,
|
|
331
|
+
rtf=rtf,
|
|
332
|
+
chunks=[dummy_chunk],
|
|
333
|
+
languages_detected=["ko", "en"] if has_latin else ["ko"],
|
|
334
|
+
backend="UNIFIED_SINGLE_PASS_NEURAL",
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
# -------------------------------------------------------------------
|
|
338
|
+
# Mode 2: Multi-Script Chunk Concatenation (Explicit Stitch Mode)
|
|
339
|
+
# -------------------------------------------------------------------
|
|
302
340
|
chunks = self.tokenizer.tokenize(clean_text, force_language=language)
|
|
303
341
|
if not chunks:
|
|
304
342
|
raise TTSInferenceError("Tokenization yielded no synthesizable segments.")
|
|
305
343
|
|
|
306
344
|
detected_langs = list(dict.fromkeys(c.language for c in chunks))
|
|
307
345
|
|
|
308
|
-
# 2. Multi-chunk synthesis & seamless concatenation
|
|
309
346
|
assembled_samples: List[np.ndarray] = []
|
|
310
|
-
target_sample_rate = self.sample_rate
|
|
311
347
|
|
|
312
348
|
for i, chunk in enumerate(chunks):
|
|
313
349
|
chunk_buf = self._synthesize_chunk_audio(chunk.text, chunk.language, speed=speed)
|
|
@@ -323,7 +359,6 @@ class MultilingualNeuralEngine:
|
|
|
323
359
|
silence_gap = np.zeros(pause_len, dtype=np.float32)
|
|
324
360
|
assembled_samples.append(silence_gap)
|
|
325
361
|
|
|
326
|
-
# 3. Concatenate and build final AudioBuffer
|
|
327
362
|
if assembled_samples:
|
|
328
363
|
final_samples = np.concatenate(assembled_samples)
|
|
329
364
|
else:
|
|
@@ -96,7 +96,7 @@ class SherpaNeuralEngine:
|
|
|
96
96
|
)
|
|
97
97
|
|
|
98
98
|
def _resolve_model_assets(self, model_path: Optional[str]) -> Dict[str, str]:
|
|
99
|
-
"""Locate model onnx
|
|
99
|
+
"""Locate model onnx and auxiliary metadata assets or fail-fast."""
|
|
100
100
|
search_dirs: List[Path] = []
|
|
101
101
|
if model_path:
|
|
102
102
|
p = Path(model_path).expanduser().resolve()
|
|
@@ -111,13 +111,98 @@ class SherpaNeuralEngine:
|
|
|
111
111
|
else:
|
|
112
112
|
from .hardware import get_unified_model_search_dirs
|
|
113
113
|
search_dirs = list(get_unified_model_search_dirs("tts"))
|
|
114
|
+
search_dirs.extend([Path.home(), Path("/data/data/com.termux/files/home")])
|
|
115
|
+
search_dirs.extend(self.STANDARD_MODEL_DIRS)
|
|
116
|
+
|
|
117
|
+
# 1. Kokoro Model Search
|
|
118
|
+
if self.model_type == "kokoro":
|
|
119
|
+
for sdir in search_dirs:
|
|
120
|
+
if not sdir.exists():
|
|
121
|
+
continue
|
|
122
|
+
candidates = [sdir] + [sdir / name for name in ["kokoro-int8-en-v0_19", "kokoro-en-v0_19", "kokoro-int8-multi-lang-v1_1", "kokoro-multi-lang-v1_0", "kokoro"]]
|
|
123
|
+
for cand in candidates:
|
|
124
|
+
if not cand.is_dir():
|
|
125
|
+
continue
|
|
126
|
+
voices = cand / "voices.bin"
|
|
127
|
+
tokens = cand / "tokens.txt"
|
|
128
|
+
data_dir = cand / "espeak-ng-data"
|
|
129
|
+
onnx_files = list(cand.glob("*.onnx"))
|
|
130
|
+
if onnx_files and voices.exists() and tokens.exists():
|
|
131
|
+
return {
|
|
132
|
+
"model": str(onnx_files[0]),
|
|
133
|
+
"voices": str(voices),
|
|
134
|
+
"tokens": str(tokens),
|
|
135
|
+
"data_dir": str(data_dir) if data_dir.exists() else "",
|
|
136
|
+
"model_name": cand.name,
|
|
137
|
+
}
|
|
138
|
+
raise TTSModelLoadError(
|
|
139
|
+
f"[FAIL-FAST] Kokoro model assets (model.onnx, voices.bin, tokens.txt) not found.\n"
|
|
140
|
+
f" Run 'termux-tts install --models kokoro' to download and provision Kokoro-82M."
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
# 2. Supertonic Model Search
|
|
144
|
+
if self.model_type == "supertonic":
|
|
145
|
+
for sdir in search_dirs:
|
|
146
|
+
if not sdir.exists():
|
|
147
|
+
continue
|
|
148
|
+
candidates = [sdir] + [sdir / name for name in ["sherpa-onnx-supertonic-3-tts-int8-2026-05-11", "sherpa-onnx-supertonic-tts-int8-2026-03-06", "supertonic"]]
|
|
149
|
+
for cand in candidates:
|
|
150
|
+
if not cand.is_dir():
|
|
151
|
+
continue
|
|
152
|
+
dp = cand / "duration_predictor.int8.onnx" if (cand / "duration_predictor.int8.onnx").exists() else cand / "duration_predictor.onnx"
|
|
153
|
+
te = cand / "text_encoder.int8.onnx" if (cand / "text_encoder.int8.onnx").exists() else cand / "text_encoder.onnx"
|
|
154
|
+
ve = cand / "vector_estimator.int8.onnx" if (cand / "vector_estimator.int8.onnx").exists() else cand / "vector_estimator.onnx"
|
|
155
|
+
voc = cand / "vocoder.int8.onnx" if (cand / "vocoder.int8.onnx").exists() else cand / "vocoder.onnx"
|
|
156
|
+
tts_json = cand / "tts.json"
|
|
157
|
+
ui = cand / "unicode_indexer.bin"
|
|
158
|
+
vs = cand / "voice_styles.bin"
|
|
159
|
+
if not vs.exists():
|
|
160
|
+
vs = cand / "voice.bin"
|
|
161
|
+
if dp.exists() and te.exists() and ve.exists() and voc.exists() and tts_json.exists() and ui.exists() and vs.exists():
|
|
162
|
+
return {
|
|
163
|
+
"duration_predictor": str(dp),
|
|
164
|
+
"text_encoder": str(te),
|
|
165
|
+
"vector_estimator": str(ve),
|
|
166
|
+
"vocoder": str(voc),
|
|
167
|
+
"tts_json": str(tts_json),
|
|
168
|
+
"unicode_indexer": str(ui),
|
|
169
|
+
"voice_style": str(vs),
|
|
170
|
+
"model_name": cand.name,
|
|
171
|
+
}
|
|
172
|
+
raise TTSModelLoadError(
|
|
173
|
+
f"[FAIL-FAST] Supertonic model assets (duration_predictor, text_encoder, vocoder, tts.json) not found.\n"
|
|
174
|
+
f" Run 'termux-tts install --models supertonic' to download and provision Supertonic."
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
# 3. MeloTTS Model Search
|
|
178
|
+
if self.model_type == "melo":
|
|
179
|
+
for sdir in search_dirs:
|
|
180
|
+
if not sdir.exists():
|
|
181
|
+
continue
|
|
182
|
+
candidates = [sdir] + [sdir / name for name in ["vits-melo-tts-zh_en", "melo-tts", "melotts"]]
|
|
183
|
+
for cand in candidates:
|
|
184
|
+
if not cand.is_dir():
|
|
185
|
+
continue
|
|
186
|
+
tokens = cand / "tokens.txt"
|
|
187
|
+
lexicon = cand / "lexicon.txt"
|
|
188
|
+
onnx_files = list(cand.glob("*.onnx"))
|
|
189
|
+
if onnx_files and tokens.exists() and lexicon.exists():
|
|
190
|
+
return {
|
|
191
|
+
"model": str(onnx_files[0]),
|
|
192
|
+
"tokens": str(tokens),
|
|
193
|
+
"lexicon": str(lexicon),
|
|
194
|
+
"model_name": cand.name,
|
|
195
|
+
}
|
|
196
|
+
raise TTSModelLoadError(
|
|
197
|
+
f"[FAIL-FAST] MeloTTS model assets (model.onnx, tokens.txt, lexicon.txt) not found.\n"
|
|
198
|
+
f" Run 'termux-tts install --models melo' to download and provision MeloTTS."
|
|
199
|
+
)
|
|
114
200
|
|
|
115
|
-
#
|
|
201
|
+
# 4. Standard VITS ONNX model search
|
|
116
202
|
for sdir in search_dirs:
|
|
117
203
|
if not sdir.exists():
|
|
118
204
|
continue
|
|
119
205
|
|
|
120
|
-
# Look for VITS ONNX model
|
|
121
206
|
onnx_files = list(sdir.glob("*.onnx"))
|
|
122
207
|
if not onnx_files and (sdir / "vits-mimic3-ko_KO-kss_low").exists():
|
|
123
208
|
sdir = sdir / "vits-mimic3-ko_KO-kss_low"
|
|
@@ -127,19 +212,26 @@ class SherpaNeuralEngine:
|
|
|
127
212
|
onnx_model = str(onnx_files[0])
|
|
128
213
|
tokens_file = sdir / "tokens.txt"
|
|
129
214
|
espeak_dir = sdir / "espeak-ng-data"
|
|
215
|
+
lexicon_file = sdir / "lexicon.txt"
|
|
130
216
|
|
|
131
217
|
if not tokens_file.exists():
|
|
132
218
|
tokens_file = sdir.parent / "tokens.txt"
|
|
133
219
|
if not espeak_dir.exists():
|
|
134
220
|
espeak_dir = sdir.parent / "espeak-ng-data"
|
|
221
|
+
if not lexicon_file.exists():
|
|
222
|
+
lexicon_file = sdir.parent / "lexicon.txt"
|
|
135
223
|
|
|
136
|
-
if tokens_file.exists() and espeak_dir.exists():
|
|
137
|
-
|
|
224
|
+
if tokens_file.exists() and (espeak_dir.exists() or lexicon_file.exists()):
|
|
225
|
+
res = {
|
|
138
226
|
"model": str(onnx_model),
|
|
139
227
|
"tokens": str(tokens_file),
|
|
140
|
-
"data_dir": str(espeak_dir),
|
|
141
228
|
"model_name": Path(onnx_model).stem,
|
|
142
229
|
}
|
|
230
|
+
if espeak_dir.exists():
|
|
231
|
+
res["data_dir"] = str(espeak_dir)
|
|
232
|
+
if lexicon_file.exists():
|
|
233
|
+
res["lexicon"] = str(lexicon_file)
|
|
234
|
+
return res
|
|
143
235
|
|
|
144
236
|
raise TTSModelLoadError(
|
|
145
237
|
f"[FAIL-FAST] Required TTS model assets (onnx model, tokens.txt, espeak-ng-data) "
|
|
@@ -173,18 +265,65 @@ class SherpaNeuralEngine:
|
|
|
173
265
|
temp_wav = tmp_file.name
|
|
174
266
|
|
|
175
267
|
try:
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
268
|
+
if self.model_type == "kokoro":
|
|
269
|
+
cmd = [
|
|
270
|
+
self.binary,
|
|
271
|
+
f"--kokoro-model={self.model_assets['model']}",
|
|
272
|
+
f"--kokoro-voices={self.model_assets['voices']}",
|
|
273
|
+
f"--kokoro-tokens={self.model_assets['tokens']}",
|
|
274
|
+
f"--output-filename={temp_wav}",
|
|
275
|
+
f"--num-threads={self.threads}",
|
|
276
|
+
f"--speed={speed:.2f}",
|
|
277
|
+
]
|
|
278
|
+
if self.model_assets.get("data_dir"):
|
|
279
|
+
cmd.append(f"--kokoro-data-dir={self.model_assets['data_dir']}")
|
|
280
|
+
cmd.append(normalized_text)
|
|
281
|
+
elif self.model_type == "supertonic":
|
|
282
|
+
cmd = [
|
|
283
|
+
self.binary,
|
|
284
|
+
f"--supertonic-duration-predictor={self.model_assets['duration_predictor']}",
|
|
285
|
+
f"--supertonic-text-encoder={self.model_assets['text_encoder']}",
|
|
286
|
+
f"--supertonic-vector-estimator={self.model_assets['vector_estimator']}",
|
|
287
|
+
f"--supertonic-vocoder={self.model_assets['vocoder']}",
|
|
288
|
+
f"--supertonic-tts-json={self.model_assets['tts_json']}",
|
|
289
|
+
f"--supertonic-unicode-indexer={self.model_assets['unicode_indexer']}",
|
|
290
|
+
f"--supertonic-voice-style={self.model_assets['voice_style']}",
|
|
291
|
+
f"--lang={self.language if self.language not in ('auto', '') else 'ko'}",
|
|
292
|
+
f"--output-filename={temp_wav}",
|
|
293
|
+
f"--num-threads={self.threads}",
|
|
294
|
+
f"--speed={speed:.2f}",
|
|
295
|
+
normalized_text,
|
|
296
|
+
]
|
|
297
|
+
elif self.model_type == "melo":
|
|
298
|
+
cmd = [
|
|
299
|
+
self.binary,
|
|
300
|
+
f"--vits-model={self.model_assets['model']}",
|
|
301
|
+
f"--vits-tokens={self.model_assets['tokens']}",
|
|
302
|
+
f"--vits-lexicon={self.model_assets['lexicon']}",
|
|
303
|
+
f"--output-filename={temp_wav}",
|
|
304
|
+
f"--num-threads={self.threads}",
|
|
305
|
+
f"--speed={speed:.2f}",
|
|
306
|
+
normalized_text,
|
|
307
|
+
]
|
|
308
|
+
else:
|
|
309
|
+
cmd = [
|
|
310
|
+
self.binary,
|
|
311
|
+
f"--vits-model={self.model_assets['model']}",
|
|
312
|
+
f"--vits-tokens={self.model_assets['tokens']}",
|
|
313
|
+
f"--output-filename={temp_wav}",
|
|
314
|
+
f"--num-threads={self.threads}",
|
|
315
|
+
f"--speed={speed:.2f}",
|
|
316
|
+
]
|
|
317
|
+
if self.model_assets.get("data_dir"):
|
|
318
|
+
cmd.append(f"--vits-data-dir={self.model_assets['data_dir']}")
|
|
319
|
+
elif self.model_assets.get("lexicon"):
|
|
320
|
+
cmd.append(f"--vits-lexicon={self.model_assets['lexicon']}")
|
|
321
|
+
cmd.append(normalized_text)
|
|
186
322
|
|
|
187
323
|
env = os.environ.copy()
|
|
324
|
+
env["PYTHONIOENCODING"] = "utf-8"
|
|
325
|
+
env["LANG"] = "en_US.UTF-8"
|
|
326
|
+
env["LC_ALL"] = "en_US.UTF-8"
|
|
188
327
|
# Ensure proper thread affinity and libraries
|
|
189
328
|
if os.path.exists("/system/lib64/libvulkan.so"):
|
|
190
329
|
current_ld = env.get("LD_LIBRARY_PATH", "")
|