termux-tts 1.4.4 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/doc.config.yaml CHANGED
@@ -8,8 +8,8 @@ name: "termux-tts"
8
8
  display_name: "Termux-TTS"
9
9
  package_name_pypi: "termux-tts"
10
10
  package_name_npm: "termux-tts"
11
- version: "v1.4.4"
12
- release_name: "Multilingual Neural Orchestrator & Zero-Config Ergonomics"
11
+ version: "v1.5.0"
12
+ release_name: "100% Native Vulkan Hardware GPU Pipeline & Zero-Silent-Fallback Architecture"
13
13
  license: "Apache-2.0"
14
14
  platform: "Android ARM64 / Qualcomm Adreno & ARM Mali Vulkan 1.3 / Linux"
15
15
  github_repo_url: "https://github.com/uno-km/termux-tts"
@@ -142,14 +142,13 @@ code_example_js: |
142
142
  main();
143
143
 
144
144
  benchmarks:
145
- headers: ["Hardware Target", "SoC / GPU Architecture", "Synthesis Engine", "Audio Duration", "Compute Latency", "Real-Time Factor (RTF)", "Execution Status"]
145
+ headers: ["Hardware Target", "SoC / Physical GPU Architecture", "Vulkan Driver ABI", "Audio Duration", "Synthesis Latency", "Real-Time Factor (RTF)", "Execution Status"]
146
146
  rows:
147
- - ["Galaxy S25", "Snapdragon 8 Elite / Adreno 830", "Vulkan GPU (high-fp16)", "6.70 s", "6.65 s", "0.993x", "Validated (Real-time)"]
148
- - ["Galaxy S25", "Snapdragon 8 Elite / Adreno 830", "Vulkan GPU (medium)", "4.59 s", "1.21 s", "0.264x", "Validated (3.79x faster)"]
149
- - ["Galaxy A35", "Exynos 1380 / ARM Mali-G68 MP5", "Vulkan GPU (medium)", "4.52 s", "5.18 s", "1.146x", "Validated (Stable)"]
150
- - ["Galaxy A35", "Exynos 1380 / ARM Mali-G68 MP5", "Vulkan GPU (high-fp16)", "6.73 s", "34.33 s", "5.098x", "Validated (High Fidelity)"]
151
- - ["Heterogeneous ARM64", "Cortex-A78 / A55 CPU Core", "Zero-Dep DSP Formant", "4.15 s", "0.054 s", "0.0130x", "Validated (Instant)"]
152
- - ["Android Physical Speaker", "AudioTrack / OpenSL ES Bridge", "Android Native Service", "N/A", "0.012 s", "0.0020x", "Validated (Hardware Out)"]
147
+ - ["Galaxy S21", "Exynos 2100 / ARM Mali-G78 (0x9800000)", "Vulkan 1.1 (/system/lib64)", "4.25 s", "9.37 s", "2.2055x", "Validated (100% Native GPU)"]
148
+ - ["Galaxy S25", "Snapdragon 8 Elite / Adreno 830 (0x80320040)", "Vulkan 1.3 (/system/lib64)", "4.25 s", "16.21 s", "3.8145x", "Validated (100% Native GPU)"]
149
+ - ["Galaxy S22", "Snapdragon 8 Gen 1 / Adreno 730 (0x80267062)", "Vulkan 1.1 (/system/lib64)", "4.27 s", "38.09 s", "8.9159x", "Validated (100% Native GPU)"]
150
+ - ["Galaxy A35", "Exynos 1380 / ARM Mali-G68 MP5 (0x9801000)", "Vulkan 1.1 (/system/lib64)", "4.27 s", "51.91 s", "12.1505x", "Validated (100% Native GPU)"]
151
+ - ["Heterogeneous CPU", "Cortex-A78 / A55 CPU Core", "Sherpa NEON SIMD C++", "4.25 s", "1.85 s", "0.4350x", "Validated (CPU Reference)"]
153
152
 
154
153
  api_reference:
155
154
  - symbol: "termux_tts.load(engine='auto'|'vulkan'|'sherpa'|'dsp'|'native', model_tier='high'|'medium', preset='balanced')"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "termux-tts",
3
- "version": "1.4.4",
3
+ "version": "1.5.0",
4
4
  "description": "On-device Text-to-Speech framework utilizing device resources (DSP Formant Vocoder, ONNX Neural Runtime & Android Native Voice)",
5
5
  "main": "index.js",
6
6
  "bin": {
package/pyproject.toml CHANGED
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "termux-tts"
7
- version = "1.4.4"
8
- description = "On-device 4-Tier Text-to-Speech framework utilizing device resources (DSP Formant Vocoder, C++ Sherpa-ONNX Neural, Android Native & Expressive)"
7
+ version = "1.5.0"
8
+ description = "On-device 4-Tier Text-to-Speech framework utilizing device resources (C++ Sherpa-ONNX Neural, Android Native & Expressive)"
9
9
  readme = "README.pypi.md"
10
10
  requires-python = ">=3.10"
11
11
  license = { text = "Apache-2.0" }
package/setup.py CHANGED
@@ -3,7 +3,7 @@ from setuptools import setup, find_packages
3
3
 
4
4
  setup(
5
5
  name="termux-tts",
6
- version="1.4.2",
6
+ version="1.5.0",
7
7
  description="Ultra-Fast On-Device 4-Tier Text-to-Speech Framework (DSP Synth, Android Native, C++ Sherpa-ONNX Neural & Expressive)",
8
8
  long_description=open("README.pypi.md", encoding="utf-8").read() if os.path.exists("README.pypi.md") else open("README.md", encoding="utf-8").read(),
9
9
  long_description_content_type="text/markdown",
@@ -8,7 +8,6 @@ termux-tts: Production-Grade 4-Tier TTS Framework for Android Termux.
8
8
 
9
9
  from .engine import TTSEngine, load, doctor
10
10
  from .engine_native import NativeAndroidEngine, NativeResult
11
- from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS, DSPSynthesizer
12
11
  from .engine_sherpa import SherpaNeuralEngine, SherpaResult
13
12
  from .engine_vulkan import VulkanNeuralEngine, VulkanResult
14
13
  from .engine_expressive import ExpressiveEngine, ExpressiveResult
@@ -32,14 +31,11 @@ from .exceptions import (
32
31
  ONNXNeuralEngine = SherpaNeuralEngine
33
32
  ONNXResult = SherpaResult
34
33
 
35
- __version__ = "1.4.4"
34
+ __version__ = "1.5.0"
36
35
  __all__ = [
37
36
  "TTSEngine",
38
37
  "load",
39
38
  "doctor",
40
- "ParametricDSPEngine",
41
- "DSPSynthesizer",
42
- "DSPResult",
43
39
  "NativeAndroidEngine",
44
40
  "NativeResult",
45
41
  "SherpaNeuralEngine",
package/termux_tts/cli.py CHANGED
@@ -31,8 +31,8 @@ def main():
31
31
  synth_parser.add_argument("-l", "--lang", default="auto", help="Language code (auto=Multi-language auto switch, ko/kor=Korean only, en/eng=English only, ja/jpn=Japanese only)")
32
32
  synth_parser.add_argument(
33
33
  "-e", "--engine", default="auto",
34
- choices=["auto", "vulkan", "ncnn", "gpu", "synth", "dsp", "native", "neural", "onnx", "expressive", "multilingual", "hybrid"],
35
- help="Synthesis engine tier (auto=Smart Routing, vulkan=GPU NCNN, synth=0MB DSP, native=Android voice, neural=VITS C++, expressive=emotional, multilingual=Cross-language)"
34
+ choices=["auto", "vulkan", "ncnn", "gpu", "native", "neural", "onnx", "expressive", "multilingual", "hybrid", "kokoro", "melo", "supertonic"],
35
+ help="Synthesis engine tier (auto=Smart Routing, vulkan=GPU NCNN, native=Android voice, neural=VITS C++, kokoro=StyleTTS2 82M, melo=MeloTTS Bilingual, supertonic=Supertonic On-Device)"
36
36
  )
37
37
  synth_parser.add_argument("-m", "--model", default=None, help="Path to model file or directory")
38
38
  synth_parser.add_argument("-p", "--preset", default="balanced", choices=["fast", "balanced", "expressive", "ultra"])
@@ -42,6 +42,7 @@ def main():
42
42
  synth_parser.add_argument("--cpu", dest="device", action="store_const", const="cpu", help="Force CPU compute mode")
43
43
  synth_parser.add_argument("--tier", default=None, choices=["high", "medium", "balanced", "fast", "ultra"], help="Target model tier (high=Studio FP16, medium=Balanced)")
44
44
  synth_parser.add_argument("-s", "--speed", type=float, default=1.0, help="Speech speed multiplier (0.5 to 2.0)")
45
+ synth_parser.add_argument("--mode", default="unified", choices=["unified", "stitch"], help="Multilingual synthesis mode (unified=Single-pass G2P transliteration [BigTech Standard], stitch=Multi-model chunk concatenation)")
45
46
  synth_parser.add_argument("--threads", type=int, default=4, help="Compute worker threads (ARM NEON)")
46
47
  synth_parser.add_argument("--volume", type=int, default=None, help="Set Android media volume (1 to 15)")
47
48
  synth_parser.add_argument("--play", action="store_true", help="Play synthesized audio through physical speaker immediately")
@@ -101,7 +102,13 @@ def main():
101
102
  engine=args.engine,
102
103
  tier=getattr(args, "tier", None),
103
104
  ) as engine:
104
- res = engine.synthesize(target_text, output=out_path, speed=args.speed, language=args.lang)
105
+ res = engine.synthesize(
106
+ target_text,
107
+ output=out_path,
108
+ speed=args.speed,
109
+ language=args.lang,
110
+ mode=getattr(args, "mode", "unified")
111
+ )
105
112
  backend_name = getattr(res, "backend", "UNKNOWN")
106
113
  model_name = getattr(res, "model_name", "model")
107
114
  dur = getattr(res, "duration_sec", 0.0)
@@ -18,7 +18,6 @@ from typing import Optional, Union, Dict, Any
18
18
 
19
19
  from .exceptions import TTSInferenceError, TTSModelLoadError, VulkanInitializationError
20
20
  from .engine_native import NativeAndroidEngine, NativeResult
21
- from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS
22
21
  from .engine_sherpa import SherpaNeuralEngine, SherpaResult
23
22
  from .engine_vulkan import VulkanNeuralEngine, VulkanResult
24
23
  from .engine_expressive import ExpressiveEngine, ExpressiveResult
@@ -34,6 +33,14 @@ from .hardware import (
34
33
  logger = logging.getLogger("termux_tts.engine")
35
34
 
36
35
 
36
+ QUALITY_PRESETS = {
37
+ "fast": {"sample_rate": 16000, "description": "16kHz Low Latency"},
38
+ "balanced": {"sample_rate": 22050, "description": "22.05kHz Standard Audio"},
39
+ "expressive": {"sample_rate": 24000, "description": "24kHz Expressive High Fidelity"},
40
+ "ultra": {"sample_rate": 44100, "description": "44.1kHz Studio Master"},
41
+ }
42
+
43
+
37
44
  class TTSEngine:
38
45
  """Production 4-Tier Multi-Backend Gateway supporting Synth, Native, Neural, and Expressive engines."""
39
46
 
@@ -86,6 +93,18 @@ class TTSEngine:
86
93
  if norm_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar"):
87
94
  return self._get_multilingual_engine()
88
95
 
96
+ # Explicit Vulkan GPU MeloTTS Tier (Plan 1 NCNN Sliced / Plan 2 MNN Vulkan)
97
+ if t in ("melo", "melo_vulkan", "melo_ncnn", "melo_mnn") and (self.requested_device in ("vulkan", "gpu") or self.device == "vulkan"):
98
+ return VulkanNeuralEngine(
99
+ model_path=self.model_path,
100
+ language=self.language,
101
+ device=self.requested_device,
102
+ threads=self.threads,
103
+ sample_rate=self.sample_rate or 44100,
104
+ model_tier=self.model_tier,
105
+ model_type=t,
106
+ )
107
+
89
108
  # Explicit Vulkan GPU Tier (Vulkan NCNN engine targets English Lessac)
90
109
  if (t in ("vulkan", "gpu", "ncnn") or (self.requested_device in ("vulkan", "gpu") and t in ("neural", "vits", "auto"))) and norm_lang in ("en", "auto"):
91
110
  try:
@@ -96,6 +115,7 @@ class TTSEngine:
96
115
  threads=self.threads,
97
116
  sample_rate=self.sample_rate or 22050,
98
117
  model_tier=self.model_tier,
118
+ model_type="vits",
99
119
  )
100
120
  except (VulkanInitializationError, TTSModelLoadError) as err:
101
121
  if t in ("vulkan", "gpu", "ncnn") or self.requested_device in ("vulkan", "gpu"):
@@ -113,24 +133,25 @@ class TTSEngine:
113
133
  model_type="vits",
114
134
  )
115
135
 
116
- # Explicit Tier 4: Expressive (Fail-Fast)
117
- elif t in ("expressive", "chat", "conversational"):
118
- return ExpressiveEngine(
136
+ # BigTech 3rd-Party Neural Speech Engines (StyleTTS2/Kokoro, MeloTTS, Supertonic)
137
+ elif t in ("kokoro", "melo", "supertonic"):
138
+ return SherpaNeuralEngine(
119
139
  model_path=self.model_path,
120
140
  language=self.language,
121
141
  device=self.requested_device,
122
142
  threads=self.threads,
123
143
  sample_rate=self.sample_rate or 22050,
144
+ model_type=t,
124
145
  )
125
146
 
126
- # Explicit Tier 1: Synth / DSP
127
- elif t in ("synth", "dsp", "formant"):
128
- return ParametricDSPEngine(
147
+ # Explicit Tier 4: Expressive (Fail-Fast)
148
+ elif t in ("expressive", "chat", "conversational"):
149
+ return ExpressiveEngine(
129
150
  model_path=self.model_path,
130
151
  language=self.language,
131
- preset=self.preset,
132
152
  device=self.requested_device,
133
- sample_rate=self.sample_rate,
153
+ threads=self.threads,
154
+ sample_rate=self.sample_rate or 22050,
134
155
  )
135
156
 
136
157
  # Explicit Multilingual / Code-Switching Tier
@@ -141,7 +162,7 @@ class TTSEngine:
141
162
  elif t == "native":
142
163
  return self.native_engine
143
164
 
144
- # Auto Mode
165
+ # Auto Mode (Zero-Silent-Fallback)
145
166
  elif t == "auto":
146
167
  # 1. Check if SherpaNeuralEngine assets exist
147
168
  try:
@@ -159,18 +180,17 @@ class TTSEngine:
159
180
  if self.native_engine.binary:
160
181
  return self.native_engine
161
182
 
162
- # 3. Fallback to zero-dependency DSP Synth
163
- return ParametricDSPEngine(
164
- model_path=self.model_path,
165
- language=self.language,
166
- preset=self.preset,
167
- device=self.requested_device,
168
- sample_rate=self.sample_rate,
183
+ # 3. Fail-Fast: No silent robotic fallback
184
+ raise TTSModelLoadError(
185
+ "[FAIL-FAST] No TTS voice engine available. Neural speech model is not installed, "
186
+ "and native Android TTS engine is unavailable.\n"
187
+ " To install official neural models automatically, run:\n"
188
+ " termux-tts install\n"
169
189
  )
170
190
  else:
171
191
  raise TTSInferenceError(
172
192
  f"[FAIL-FAST] Unknown engine_type '{self.requested_engine_type}'. "
173
- f"Available tiers: ['auto', 'synth', 'native', 'neural', 'expressive']"
193
+ f"Available tiers: ['auto', 'native', 'neural', 'expressive', 'multilingual', 'kokoro', 'melo', 'supertonic']"
174
194
  )
175
195
 
176
196
  @property
@@ -207,7 +227,8 @@ class TTSEngine:
207
227
  speed: float = 1.0,
208
228
  preset: Optional[str] = None,
209
229
  language: Optional[str] = None,
210
- ) -> Union[DSPResult, SherpaResult, ExpressiveResult, NativeResult, MultilingualResult]:
230
+ mode: str = "unified",
231
+ ) -> Union[SherpaResult, ExpressiveResult, NativeResult, MultilingualResult]:
211
232
  """Synthesize text into speech audio buffer / WAV file with Zero-Config intelligent routing."""
212
233
  if self._is_closed:
213
234
  raise TTSInferenceError("Cannot synthesize: Engine session is closed.")
@@ -219,24 +240,21 @@ class TTSEngine:
219
240
  from .script_classifier import normalize_language_code
220
241
  target_lang = normalize_language_code(language or self.language)
221
242
 
222
- # 1. If user explicitly pinned to lightweight DSP (0MB), respect choice
223
- if self.requested_engine_type in ("dsp", "synth", "formant") or self.device == "dsp":
224
- return self.synth_engine.synthesize(clean_text, output=output, speed=speed, preset=preset)
225
-
226
- # 2. If user explicitly pinned to OS Native voice, speak directly
243
+ # 1. If user explicitly pinned to OS Native voice, speak directly
227
244
  if self.requested_engine_type == "native":
228
245
  return self.native_engine.speak(clean_text)
229
246
 
230
- # 3. Multilingual auto-detection:
247
+ # 2. Multilingual auto-detection:
231
248
  # Route to MultilingualNeuralEngine if multiple languages are present in text
232
249
  # or if caller specifically requested multilingual / hybrid engine.
233
250
  detected_langs = MultilingualTokenizer.detect_languages(clean_text)
234
251
  is_mixed_text = len(detected_langs) > 1
235
252
 
236
253
  should_route_multilingual = (
237
- (self.requested_engine_type in ("auto", "multilingual", "codeswitch", "hybrid")) or
238
- (is_mixed_text and self.requested_engine_type not in ("dsp", "synth", "native") and self.device != "dsp" and (self._multilingual_engine is not None or not isinstance(self.synth_engine, ParametricDSPEngine))) or
239
- (target_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar") and self.requested_engine_type not in ("dsp", "synth", "native"))
254
+ (self.requested_engine_type in ("multilingual", "codeswitch", "hybrid")) or
255
+ (self.requested_engine_type == "auto" and (is_mixed_text or self._multilingual_engine is not None)) or
256
+ (is_mixed_text and self.requested_engine_type != "native") or
257
+ (target_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar") and self.requested_engine_type != "native")
240
258
  )
241
259
 
242
260
  if should_route_multilingual:
@@ -248,6 +266,7 @@ class TTSEngine:
248
266
  clean_text,
249
267
  output=output,
250
268
  speed=speed,
269
+ mode=mode,
251
270
  **kwargs
252
271
  )
253
272
 
@@ -8,6 +8,7 @@ from __future__ import annotations
8
8
 
9
9
  import logging
10
10
  import os
11
+ import re
11
12
  import shutil
12
13
  import subprocess
13
14
  import tempfile
@@ -284,10 +285,12 @@ class MultilingualNeuralEngine:
284
285
  output: Optional[str] = None,
285
286
  speed: float = 1.0,
286
287
  language: Optional[str] = None,
288
+ mode: str = "unified",
287
289
  ) -> MultilingualResult:
288
290
  """
289
- Synthesizes speech with zero cross-linguistic phoneme distortion.
290
- Supports automatic multi-script code-switching or forced language isolation.
291
+ Synthesizes speech with natural prosody and zero cross-speaker distortion.
292
+ Defaults to 'unified' single-pass mode (phonetic transliteration preserving prosody).
293
+ 'stitch' mode retains explicit multi-model chunk concatenation when requested.
291
294
  """
292
295
  if self._is_closed:
293
296
  raise TTSInferenceError("Cannot synthesize: Multilingual session is closed.")
@@ -297,17 +300,50 @@ class MultilingualNeuralEngine:
297
300
  raise TTSInferenceError("Cannot synthesize empty text.")
298
301
 
299
302
  t0 = time.perf_counter()
303
+ target_sample_rate = self.sample_rate
300
304
 
301
- # 1. Parse text into language-tagged chunks (respecting forced language if provided)
305
+ # -------------------------------------------------------------------
306
+ # Mode 1: Unified Single-Pass Pipeline (BigTech Architecture Standard)
307
+ # -------------------------------------------------------------------
308
+ has_korean = bool(re.search(r"[\uac00-\ud7a3]", clean_text))
309
+ has_latin = bool(re.search(r"[a-zA-Z]", clean_text))
310
+
311
+ if mode == "unified" and (language == "ko" or (language is None and has_korean)):
312
+ from .transliteration import transliterate_mixed_text
313
+ unified_text = transliterate_mixed_text(clean_text)
314
+ logger.info("[Multilingual] Unified single-pass execution: '%s' -> '%s'", clean_text, unified_text)
315
+
316
+ chunk_buf = self._synthesize_chunk_audio(unified_text, "ko", speed=speed)
317
+ elapsed_ms = (time.perf_counter() - t0) * 1000.0
318
+ dur_sec = chunk_buf.duration_seconds
319
+ rtf = (elapsed_ms / 1000.0) / max(0.001, dur_sec)
320
+
321
+ if output:
322
+ chunk_buf.save(output)
323
+
324
+ dummy_chunk = LanguageChunk(text=clean_text, language="ko", pause_after=0.0)
325
+ return MultilingualResult(
326
+ text=clean_text,
327
+ audio_buffer=chunk_buf,
328
+ sample_rate=target_sample_rate,
329
+ duration_sec=dur_sec,
330
+ elapsed_ms=elapsed_ms,
331
+ rtf=rtf,
332
+ chunks=[dummy_chunk],
333
+ languages_detected=["ko", "en"] if has_latin else ["ko"],
334
+ backend="UNIFIED_SINGLE_PASS_NEURAL",
335
+ )
336
+
337
+ # -------------------------------------------------------------------
338
+ # Mode 2: Multi-Script Chunk Concatenation (Explicit Stitch Mode)
339
+ # -------------------------------------------------------------------
302
340
  chunks = self.tokenizer.tokenize(clean_text, force_language=language)
303
341
  if not chunks:
304
342
  raise TTSInferenceError("Tokenization yielded no synthesizable segments.")
305
343
 
306
344
  detected_langs = list(dict.fromkeys(c.language for c in chunks))
307
345
 
308
- # 2. Multi-chunk synthesis & seamless concatenation
309
346
  assembled_samples: List[np.ndarray] = []
310
- target_sample_rate = self.sample_rate
311
347
 
312
348
  for i, chunk in enumerate(chunks):
313
349
  chunk_buf = self._synthesize_chunk_audio(chunk.text, chunk.language, speed=speed)
@@ -323,7 +359,6 @@ class MultilingualNeuralEngine:
323
359
  silence_gap = np.zeros(pause_len, dtype=np.float32)
324
360
  assembled_samples.append(silence_gap)
325
361
 
326
- # 3. Concatenate and build final AudioBuffer
327
362
  if assembled_samples:
328
363
  final_samples = np.concatenate(assembled_samples)
329
364
  else:
@@ -96,7 +96,7 @@ class SherpaNeuralEngine:
96
96
  )
97
97
 
98
98
  def _resolve_model_assets(self, model_path: Optional[str]) -> Dict[str, str]:
99
- """Locate model onnx, tokens.txt, and espeak-ng-data directory or fail-fast."""
99
+ """Locate model onnx and auxiliary metadata assets or fail-fast."""
100
100
  search_dirs: List[Path] = []
101
101
  if model_path:
102
102
  p = Path(model_path).expanduser().resolve()
@@ -111,13 +111,98 @@ class SherpaNeuralEngine:
111
111
  else:
112
112
  from .hardware import get_unified_model_search_dirs
113
113
  search_dirs = list(get_unified_model_search_dirs("tts"))
114
+ search_dirs.extend([Path.home(), Path("/data/data/com.termux/files/home")])
115
+ search_dirs.extend(self.STANDARD_MODEL_DIRS)
116
+
117
+ # 1. Kokoro Model Search
118
+ if self.model_type == "kokoro":
119
+ for sdir in search_dirs:
120
+ if not sdir.exists():
121
+ continue
122
+ candidates = [sdir] + [sdir / name for name in ["kokoro-int8-en-v0_19", "kokoro-en-v0_19", "kokoro-int8-multi-lang-v1_1", "kokoro-multi-lang-v1_0", "kokoro"]]
123
+ for cand in candidates:
124
+ if not cand.is_dir():
125
+ continue
126
+ voices = cand / "voices.bin"
127
+ tokens = cand / "tokens.txt"
128
+ data_dir = cand / "espeak-ng-data"
129
+ onnx_files = list(cand.glob("*.onnx"))
130
+ if onnx_files and voices.exists() and tokens.exists():
131
+ return {
132
+ "model": str(onnx_files[0]),
133
+ "voices": str(voices),
134
+ "tokens": str(tokens),
135
+ "data_dir": str(data_dir) if data_dir.exists() else "",
136
+ "model_name": cand.name,
137
+ }
138
+ raise TTSModelLoadError(
139
+ f"[FAIL-FAST] Kokoro model assets (model.onnx, voices.bin, tokens.txt) not found.\n"
140
+ f" Run 'termux-tts install --models kokoro' to download and provision Kokoro-82M."
141
+ )
142
+
143
+ # 2. Supertonic Model Search
144
+ if self.model_type == "supertonic":
145
+ for sdir in search_dirs:
146
+ if not sdir.exists():
147
+ continue
148
+ candidates = [sdir] + [sdir / name for name in ["sherpa-onnx-supertonic-3-tts-int8-2026-05-11", "sherpa-onnx-supertonic-tts-int8-2026-03-06", "supertonic"]]
149
+ for cand in candidates:
150
+ if not cand.is_dir():
151
+ continue
152
+ dp = cand / "duration_predictor.int8.onnx" if (cand / "duration_predictor.int8.onnx").exists() else cand / "duration_predictor.onnx"
153
+ te = cand / "text_encoder.int8.onnx" if (cand / "text_encoder.int8.onnx").exists() else cand / "text_encoder.onnx"
154
+ ve = cand / "vector_estimator.int8.onnx" if (cand / "vector_estimator.int8.onnx").exists() else cand / "vector_estimator.onnx"
155
+ voc = cand / "vocoder.int8.onnx" if (cand / "vocoder.int8.onnx").exists() else cand / "vocoder.onnx"
156
+ tts_json = cand / "tts.json"
157
+ ui = cand / "unicode_indexer.bin"
158
+ vs = cand / "voice_styles.bin"
159
+ if not vs.exists():
160
+ vs = cand / "voice.bin"
161
+ if dp.exists() and te.exists() and ve.exists() and voc.exists() and tts_json.exists() and ui.exists() and vs.exists():
162
+ return {
163
+ "duration_predictor": str(dp),
164
+ "text_encoder": str(te),
165
+ "vector_estimator": str(ve),
166
+ "vocoder": str(voc),
167
+ "tts_json": str(tts_json),
168
+ "unicode_indexer": str(ui),
169
+ "voice_style": str(vs),
170
+ "model_name": cand.name,
171
+ }
172
+ raise TTSModelLoadError(
173
+ f"[FAIL-FAST] Supertonic model assets (duration_predictor, text_encoder, vocoder, tts.json) not found.\n"
174
+ f" Run 'termux-tts install --models supertonic' to download and provision Supertonic."
175
+ )
176
+
177
+ # 3. MeloTTS Model Search
178
+ if self.model_type == "melo":
179
+ for sdir in search_dirs:
180
+ if not sdir.exists():
181
+ continue
182
+ candidates = [sdir] + [sdir / name for name in ["vits-melo-tts-zh_en", "melo-tts", "melotts"]]
183
+ for cand in candidates:
184
+ if not cand.is_dir():
185
+ continue
186
+ tokens = cand / "tokens.txt"
187
+ lexicon = cand / "lexicon.txt"
188
+ onnx_files = list(cand.glob("*.onnx"))
189
+ if onnx_files and tokens.exists() and lexicon.exists():
190
+ return {
191
+ "model": str(onnx_files[0]),
192
+ "tokens": str(tokens),
193
+ "lexicon": str(lexicon),
194
+ "model_name": cand.name,
195
+ }
196
+ raise TTSModelLoadError(
197
+ f"[FAIL-FAST] MeloTTS model assets (model.onnx, tokens.txt, lexicon.txt) not found.\n"
198
+ f" Run 'termux-tts install --models melo' to download and provision MeloTTS."
199
+ )
114
200
 
115
- # 1. Find directory containing .onnx model, tokens.txt, and espeak-ng-data
201
+ # 4. Standard VITS ONNX model search
116
202
  for sdir in search_dirs:
117
203
  if not sdir.exists():
118
204
  continue
119
205
 
120
- # Look for VITS ONNX model
121
206
  onnx_files = list(sdir.glob("*.onnx"))
122
207
  if not onnx_files and (sdir / "vits-mimic3-ko_KO-kss_low").exists():
123
208
  sdir = sdir / "vits-mimic3-ko_KO-kss_low"
@@ -127,19 +212,26 @@ class SherpaNeuralEngine:
127
212
  onnx_model = str(onnx_files[0])
128
213
  tokens_file = sdir / "tokens.txt"
129
214
  espeak_dir = sdir / "espeak-ng-data"
215
+ lexicon_file = sdir / "lexicon.txt"
130
216
 
131
217
  if not tokens_file.exists():
132
218
  tokens_file = sdir.parent / "tokens.txt"
133
219
  if not espeak_dir.exists():
134
220
  espeak_dir = sdir.parent / "espeak-ng-data"
221
+ if not lexicon_file.exists():
222
+ lexicon_file = sdir.parent / "lexicon.txt"
135
223
 
136
- if tokens_file.exists() and espeak_dir.exists():
137
- return {
224
+ if tokens_file.exists() and (espeak_dir.exists() or lexicon_file.exists()):
225
+ res = {
138
226
  "model": str(onnx_model),
139
227
  "tokens": str(tokens_file),
140
- "data_dir": str(espeak_dir),
141
228
  "model_name": Path(onnx_model).stem,
142
229
  }
230
+ if espeak_dir.exists():
231
+ res["data_dir"] = str(espeak_dir)
232
+ if lexicon_file.exists():
233
+ res["lexicon"] = str(lexicon_file)
234
+ return res
143
235
 
144
236
  raise TTSModelLoadError(
145
237
  f"[FAIL-FAST] Required TTS model assets (onnx model, tokens.txt, espeak-ng-data) "
@@ -173,18 +265,65 @@ class SherpaNeuralEngine:
173
265
  temp_wav = tmp_file.name
174
266
 
175
267
  try:
176
- cmd = [
177
- self.binary,
178
- f"--vits-model={self.model_assets['model']}",
179
- f"--vits-tokens={self.model_assets['tokens']}",
180
- f"--vits-data-dir={self.model_assets['data_dir']}",
181
- f"--output-filename={temp_wav}",
182
- f"--num-threads={self.threads}",
183
- f"--speed={speed:.2f}",
184
- normalized_text,
185
- ]
268
+ if self.model_type == "kokoro":
269
+ cmd = [
270
+ self.binary,
271
+ f"--kokoro-model={self.model_assets['model']}",
272
+ f"--kokoro-voices={self.model_assets['voices']}",
273
+ f"--kokoro-tokens={self.model_assets['tokens']}",
274
+ f"--output-filename={temp_wav}",
275
+ f"--num-threads={self.threads}",
276
+ f"--speed={speed:.2f}",
277
+ ]
278
+ if self.model_assets.get("data_dir"):
279
+ cmd.append(f"--kokoro-data-dir={self.model_assets['data_dir']}")
280
+ cmd.append(normalized_text)
281
+ elif self.model_type == "supertonic":
282
+ cmd = [
283
+ self.binary,
284
+ f"--supertonic-duration-predictor={self.model_assets['duration_predictor']}",
285
+ f"--supertonic-text-encoder={self.model_assets['text_encoder']}",
286
+ f"--supertonic-vector-estimator={self.model_assets['vector_estimator']}",
287
+ f"--supertonic-vocoder={self.model_assets['vocoder']}",
288
+ f"--supertonic-tts-json={self.model_assets['tts_json']}",
289
+ f"--supertonic-unicode-indexer={self.model_assets['unicode_indexer']}",
290
+ f"--supertonic-voice-style={self.model_assets['voice_style']}",
291
+ f"--lang={self.language if self.language not in ('auto', '') else 'ko'}",
292
+ f"--output-filename={temp_wav}",
293
+ f"--num-threads={self.threads}",
294
+ f"--speed={speed:.2f}",
295
+ normalized_text,
296
+ ]
297
+ elif self.model_type == "melo":
298
+ cmd = [
299
+ self.binary,
300
+ f"--vits-model={self.model_assets['model']}",
301
+ f"--vits-tokens={self.model_assets['tokens']}",
302
+ f"--vits-lexicon={self.model_assets['lexicon']}",
303
+ f"--output-filename={temp_wav}",
304
+ f"--num-threads={self.threads}",
305
+ f"--speed={speed:.2f}",
306
+ normalized_text,
307
+ ]
308
+ else:
309
+ cmd = [
310
+ self.binary,
311
+ f"--vits-model={self.model_assets['model']}",
312
+ f"--vits-tokens={self.model_assets['tokens']}",
313
+ f"--output-filename={temp_wav}",
314
+ f"--num-threads={self.threads}",
315
+ f"--speed={speed:.2f}",
316
+ ]
317
+ if self.model_assets.get("data_dir"):
318
+ cmd.append(f"--vits-data-dir={self.model_assets['data_dir']}")
319
+ elif self.model_assets.get("lexicon"):
320
+ cmd.append(f"--vits-lexicon={self.model_assets['lexicon']}")
321
+ cmd.append(normalized_text)
186
322
 
187
323
  env = os.environ.copy()
324
+ env["PYTHONIOENCODING"] = "utf-8"
325
+ env["LANG"] = "en_US.UTF-8"
326
+ env["LC_ALL"] = "en_US.UTF-8"
188
327
  # Ensure proper thread affinity and libraries
189
328
  if os.path.exists("/system/lib64/libvulkan.so"):
190
329
  current_ld = env.get("LD_LIBRARY_PATH", "")