termux-tts 1.4.2 → 1.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,374 @@
1
+ """
2
+ Enterprise Extensible Multilingual Neural Speech Synthesis Orchestrator for termux-tts.
3
+ Orchestrates multi-script tokenization, memory-resident neural acoustic inference via Sherpa C-API
4
+ (sub-second latency without subprocess spawning), silence trimming, and lossless PCM stream stitching.
5
+ Zero-Silent-Fallback compliant: Fails fast if a detected language has no registered neural model.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import logging
10
+ import os
11
+ import shutil
12
+ import subprocess
13
+ import tempfile
14
+ import time
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from typing import Optional, List, Dict, Any, Set
18
+
19
+ import numpy as np
20
+
21
+ from .audio import AudioBuffer
22
+ from .exceptions import TTSModelLoadError, TTSInferenceError
23
+ from .script_classifier import MultilingualTokenizer, LanguageChunk, ScriptRegistry
24
+
25
+ logger = logging.getLogger("termux_tts.engine_multilingual")
26
+
27
+
28
+ @dataclass
29
+ class MultilingualResult:
30
+ text: str
31
+ audio_buffer: AudioBuffer
32
+ sample_rate: int
33
+ duration_sec: float
34
+ elapsed_ms: float
35
+ rtf: float
36
+ chunks: List[LanguageChunk]
37
+ languages_detected: List[str]
38
+ backend: str = "MULTILINGUAL_NEURAL_ORCHESTRATOR"
39
+ model_name: str = "hybrid-multilingual-mesh"
40
+
41
+ def save(self, filepath: str) -> str:
42
+ return self.audio_buffer.save(filepath)
43
+
44
+ @property
45
+ def wav_bytes(self) -> bytes:
46
+ return self.audio_buffer.to_wav_bytes()
47
+
48
+
49
+ def trim_silence_samples(samples: np.ndarray, threshold: float = 0.01) -> np.ndarray:
50
+ """
51
+ Trims leading and trailing silence from normalized float32 samples [-1.0, 1.0].
52
+ Preserves acoustic continuity across intra-word affix boundaries.
53
+ """
54
+ if len(samples) == 0:
55
+ return samples
56
+
57
+ abs_samples = np.abs(samples)
58
+ non_silent = np.where(abs_samples >= threshold)[0]
59
+ if len(non_silent) == 0:
60
+ return samples
61
+
62
+ start_idx = non_silent[0]
63
+ end_idx = non_silent[-1]
64
+ return samples[start_idx:end_idx + 1]
65
+
66
+
67
+ class MultilingualNeuralEngine:
68
+ """
69
+ Extensible Multilingual Orchestration Engine.
70
+ Dynamically routes linguistic sub-phrases to resident on-device neural acoustic backends.
71
+ """
72
+
73
+ CANDIDATE_ONNX_BINARIES = [
74
+ "sherpa-onnx-offline-tts",
75
+ str(Path.home() / ".local" / "bin" / "sherpa-onnx-offline-tts"),
76
+ str(Path.home() / "sherpa-onnx-offline-tts"),
77
+ "/data/data/com.termux/files/home/.local/bin/sherpa-onnx-offline-tts",
78
+ "/data/data/com.termux/files/usr/bin/sherpa-onnx-offline-tts",
79
+ ]
80
+
81
+ CANDIDATE_NCNN_BINARIES = [
82
+ "sherpa-ncnn-offline-tts",
83
+ str(Path.home() / ".local" / "bin" / "sherpa-ncnn-offline-tts"),
84
+ str(Path.home() / "sherpa-ncnn" / "build-vulkan" / "bin" / "sherpa-ncnn-offline-tts"),
85
+ "/data/data/com.termux/files/home/.local/bin/sherpa-ncnn-offline-tts",
86
+ "/data/data/com.termux/files/usr/bin/sherpa-ncnn-offline-tts",
87
+ ]
88
+
89
+ STANDARD_MODEL_DIRS = [
90
+ Path.home() / ".cache" / "termux-tts" / "models",
91
+ Path.home() / "vits-mimic3-ko_KO-kss_low",
92
+ Path.home() / "vits-piper-en_US-lessac-medium",
93
+ Path.home() / "ncnn-vits-piper-en_US-lessac-high-fp16",
94
+ Path("/data/data/com.termux/files/home/.cache/termux-tts/models"),
95
+ ]
96
+
97
+ def __init__(
98
+ self,
99
+ threads: int = 4,
100
+ device: str = "cpu",
101
+ sample_rate: int = 22050,
102
+ model_path_map: Optional[Dict[str, str]] = None,
103
+ ):
104
+ self.threads = threads
105
+ self.device = device.lower()
106
+ self.sample_rate = sample_rate
107
+ self.model_path_map = model_path_map or {}
108
+ self.tokenizer = MultilingualTokenizer()
109
+ self._engines: Dict[str, Any] = {}
110
+ self._resident_manager = None
111
+ self._is_closed = False
112
+
113
+ # Cache directories
114
+ self.work_dir = Path.home() / ".cache" / "termux-tts" / "chunks"
115
+ self.work_dir.mkdir(parents=True, exist_ok=True)
116
+
117
+ def _get_resident_manager(self):
118
+ if self._resident_manager is None:
119
+ try:
120
+ from .engine_sherpa_capi import SherpaResidentManager
121
+ self._resident_manager = SherpaResidentManager.get_instance()
122
+ except Exception as err:
123
+ logger.debug("Sherpa C-API resident manager unavailable: %s", err)
124
+ return self._resident_manager
125
+
126
+ def register_engine(self, language: str, engine_instance: Any) -> None:
127
+ """Register or override a neural backend engine for a specific language code."""
128
+ self._engines[language.lower()] = engine_instance
129
+
130
+ def _find_binary(self, candidates: List[str]) -> Optional[str]:
131
+ for c in candidates:
132
+ found = shutil.which(c) if not os.path.isabs(c) else c
133
+ if found and os.path.isfile(found) and (os.access(found, os.X_OK) or os.name == "nt"):
134
+ return str(found)
135
+ return None
136
+
137
+ def _resolve_korean_assets(self) -> Dict[str, str]:
138
+ custom = self.model_path_map.get("ko")
139
+ from .hardware import get_unified_model_search_dirs
140
+ dirs = [Path(custom)] if custom else get_unified_model_search_dirs("tts")
141
+ for d in dirs:
142
+ cand = d / "vits-mimic3-ko_KO-kss_low" if not d.name.startswith("vits") else d
143
+ onnx = cand / "ko_KO-kss_low.onnx"
144
+ tokens = cand / "tokens.txt"
145
+ espeak = cand / "espeak-ng-data"
146
+ if onnx.is_file() and tokens.is_file() and espeak.is_dir():
147
+ return {"onnx": str(onnx), "tokens": str(tokens), "data_dir": str(espeak)}
148
+ raise TTSModelLoadError("[FAIL-FAST] Korean VITS model assets (ko_KO-kss_low) not found in cache.")
149
+
150
+ def _resolve_english_assets(self) -> str:
151
+ custom = self.model_path_map.get("en")
152
+ from .hardware import get_unified_model_search_dirs
153
+ dirs = [Path(custom)] if custom else get_unified_model_search_dirs("tts")
154
+ for d in dirs:
155
+ cand = d / "ncnn-vits-piper-en_US-lessac-high-fp16" if not "lessac" in d.name else d
156
+ if (cand / "decoder.ncnn.bin").is_file():
157
+ return str(cand)
158
+ raise TTSModelLoadError("[FAIL-FAST] English NCNN model assets (lessac-high-fp16) not found in cache.")
159
+
160
+ def _synthesize_chunk_audio(self, text: str, lang: str, speed: float = 1.0) -> AudioBuffer:
161
+ """Synthesize a single linguistic chunk via fastest on-device resident engine."""
162
+ # 1. Check custom registered engines
163
+ if lang in self._engines:
164
+ chunk_res = self._engines[lang].synthesize(text, speed=speed)
165
+ return chunk_res.audio_buffer
166
+
167
+ # 2. In-Memory Resident C-API Engine (Zero cold-start overhead)
168
+ mgr = self._get_resident_manager()
169
+ if mgr is not None:
170
+ try:
171
+ session = mgr.get_session(
172
+ language=lang,
173
+ custom_model_dir=self.model_path_map.get(lang),
174
+ threads=self.threads,
175
+ sample_rate=self.sample_rate,
176
+ )
177
+ return session.synthesize(text, speed=speed)
178
+ except TTSModelLoadError:
179
+ # If model assets missing or C-API failed, proceed to fallback/fail-fast
180
+ raise
181
+ except Exception as e:
182
+ logger.debug("Resident C-API execution bypassed: %s", e)
183
+
184
+ # 3. Fallback to Subprocess for Korean (ONNX CPU)
185
+ if lang == "ko":
186
+ onnx_bin = self._find_binary(self.CANDIDATE_ONNX_BINARIES)
187
+ if not onnx_bin:
188
+ raise TTSModelLoadError("[FAIL-FAST] sherpa-onnx-offline-tts binary not found for Korean synthesis.")
189
+ assets = self._resolve_korean_assets()
190
+
191
+ with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
192
+ temp_wav = tmp.name
193
+
194
+ try:
195
+ cmd = [
196
+ onnx_bin,
197
+ f"--vits-model={assets['onnx']}",
198
+ f"--vits-tokens={assets['tokens']}",
199
+ f"--vits-data-dir={assets['data_dir']}",
200
+ f"--num-threads={self.threads}",
201
+ f"--speed={speed:.2f}",
202
+ f"--output-filename={temp_wav}",
203
+ text
204
+ ]
205
+ env = os.environ.copy()
206
+ env["LANG"] = "C.UTF-8"
207
+ env["LC_ALL"] = "C.UTF-8"
208
+ res = subprocess.run(cmd, capture_output=True, text=True, env=env, timeout=60)
209
+ if res.returncode != 0 or not os.path.exists(temp_wav):
210
+ raise TTSInferenceError(f"[FAIL-FAST] Korean chunk failed for '{text}': {res.stderr}")
211
+ return AudioBuffer.from_wav_file(temp_wav)
212
+ finally:
213
+ if os.path.exists(temp_wav):
214
+ try:
215
+ os.remove(temp_wav)
216
+ except OSError:
217
+ pass
218
+
219
+ # 4. Fallback to Subprocess for English (NCNN CPU)
220
+ elif lang == "en":
221
+ ncnn_bin = self._find_binary(self.CANDIDATE_NCNN_BINARIES)
222
+ if not ncnn_bin:
223
+ raise TTSModelLoadError("[FAIL-FAST] sherpa-ncnn-offline-tts binary not found for English synthesis.")
224
+ model_dir = self._resolve_english_assets()
225
+
226
+ with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
227
+ temp_wav = tmp.name
228
+
229
+ try:
230
+ use_vk = "1" if self.device in ("vulkan", "gpu") else "0"
231
+ cmd = [
232
+ ncnn_bin,
233
+ f"--vits-model-dir={model_dir}",
234
+ f"--use-vulkan-compute={use_vk}",
235
+ f"--num-threads={self.threads}",
236
+ f"--output-filename={temp_wav}",
237
+ text
238
+ ]
239
+ env = os.environ.copy()
240
+ env["LANG"] = "C.UTF-8"
241
+ env["LC_ALL"] = "C.UTF-8"
242
+ res = subprocess.run(cmd, capture_output=True, text=True, env=env, timeout=60)
243
+ if res.returncode != 0 or not os.path.exists(temp_wav) or os.path.getsize(temp_wav) == 0:
244
+ if "empty ids for word" in res.stderr:
245
+ logger.warning("OOV word '%s' in English dictionary; inserting soft breath pause.", text)
246
+ return AudioBuffer(np.zeros(int(self.sample_rate * 0.1), dtype=np.float32), sample_rate=self.sample_rate)
247
+ raise TTSInferenceError(f"[FAIL-FAST] English chunk failed for '{text}': {res.stderr}")
248
+ return AudioBuffer.from_wav_file(temp_wav)
249
+ finally:
250
+ if os.path.exists(temp_wav):
251
+ try:
252
+ os.remove(temp_wav)
253
+ except OSError:
254
+ pass
255
+
256
+ # 5. Extended Languages (ja, zh, hi, ru, es, fr, de, ar)
257
+ elif lang in ("ja", "zh", "hi", "ru", "es", "fr", "de", "ar"):
258
+ custom_model = self.model_path_map.get(lang)
259
+ if not custom_model:
260
+ from .installer import OFFICIAL_NEURAL_MODELS
261
+ meta = OFFICIAL_NEURAL_MODELS.get(lang, {})
262
+ lang_name = meta.get("language_name", lang.upper())
263
+ repo = meta.get("repo", f"csukuangfj/vits-{lang}")
264
+ model_name = meta.get("name", f"vits-{lang}")
265
+ canonical_dest = str((Path.home() / "models" / "tts" / model_name).resolve())
266
+ raise TTSModelLoadError(
267
+ f"[FAIL-FAST] Neural speech model for language '{lang}' ({lang_name}) is not installed.\n\n"
268
+ f" To install this model automatically, run:\n"
269
+ f" termux-tts install --models {lang}\n\n"
270
+ f" Or manually download via git clone:\n"
271
+ f" git clone https://huggingface.co/{repo} {canonical_dest}\n\n"
272
+ f" After installation, re-run your synthesis command!"
273
+ )
274
+ from .engine_sherpa import SherpaNeuralEngine
275
+ engine = SherpaNeuralEngine(model_path=custom_model, language=lang, threads=self.threads, sample_rate=self.sample_rate)
276
+ self._engines[lang] = engine
277
+ return engine.synthesize(text, speed=speed).audio_buffer
278
+
279
+ raise TTSModelLoadError(f"[FAIL-FAST] Unsupported language '{lang}'. Supported languages: ['ko', 'en', 'ja', 'zh', 'hi', 'ru', 'es', 'fr', 'de', 'ar'].")
280
+
281
+ def synthesize(
282
+ self,
283
+ text: str,
284
+ output: Optional[str] = None,
285
+ speed: float = 1.0,
286
+ language: Optional[str] = None,
287
+ ) -> MultilingualResult:
288
+ """
289
+ Synthesizes speech with zero cross-linguistic phoneme distortion.
290
+ Supports automatic multi-script code-switching or forced language isolation.
291
+ """
292
+ if self._is_closed:
293
+ raise TTSInferenceError("Cannot synthesize: Multilingual session is closed.")
294
+
295
+ clean_text = text.strip()
296
+ if not clean_text:
297
+ raise TTSInferenceError("Cannot synthesize empty text.")
298
+
299
+ t0 = time.perf_counter()
300
+
301
+ # 1. Parse text into language-tagged chunks (respecting forced language if provided)
302
+ chunks = self.tokenizer.tokenize(clean_text, force_language=language)
303
+ if not chunks:
304
+ raise TTSInferenceError("Tokenization yielded no synthesizable segments.")
305
+
306
+ detected_langs = list(dict.fromkeys(c.language for c in chunks))
307
+
308
+ # 2. Multi-chunk synthesis & seamless concatenation
309
+ assembled_samples: List[np.ndarray] = []
310
+ target_sample_rate = self.sample_rate
311
+
312
+ for i, chunk in enumerate(chunks):
313
+ chunk_buf = self._synthesize_chunk_audio(chunk.text, chunk.language, speed=speed)
314
+
315
+ # Get raw float32 samples and trim acoustic silence
316
+ raw_samples = chunk_buf.samples
317
+ trimmed = trim_silence_samples(raw_samples, threshold=0.01)
318
+ assembled_samples.append(trimmed)
319
+
320
+ # Add context-aware pause
321
+ if chunk.pause_after > 0:
322
+ pause_len = int(target_sample_rate * chunk.pause_after)
323
+ silence_gap = np.zeros(pause_len, dtype=np.float32)
324
+ assembled_samples.append(silence_gap)
325
+
326
+ # 3. Concatenate and build final AudioBuffer
327
+ if assembled_samples:
328
+ final_samples = np.concatenate(assembled_samples)
329
+ else:
330
+ final_samples = np.zeros(0, dtype=np.float32)
331
+
332
+ combined_buffer = AudioBuffer(final_samples, sample_rate=target_sample_rate)
333
+
334
+ elapsed_ms = (time.perf_counter() - t0) * 1000.0
335
+ dur_sec = combined_buffer.duration_seconds
336
+ rtf = (elapsed_ms / 1000.0) / max(0.001, dur_sec)
337
+
338
+ if output:
339
+ combined_buffer.save(output)
340
+
341
+ return MultilingualResult(
342
+ text=clean_text,
343
+ audio_buffer=combined_buffer,
344
+ sample_rate=target_sample_rate,
345
+ duration_sec=dur_sec,
346
+ elapsed_ms=elapsed_ms,
347
+ rtf=rtf,
348
+ chunks=chunks,
349
+ languages_detected=detected_langs,
350
+ )
351
+
352
+ def close(self) -> None:
353
+ """Cleanly terminate all engine instances and release resident memory."""
354
+ self._is_closed = True
355
+ for engine in self._engines.values():
356
+ if hasattr(engine, "close"):
357
+ try:
358
+ engine.close()
359
+ except Exception:
360
+ pass
361
+ if self._resident_manager is not None:
362
+ try:
363
+ self._resident_manager.close_all()
364
+ except Exception:
365
+ pass
366
+
367
+ def __del__(self) -> None:
368
+ self.close()
369
+
370
+ def __enter__(self):
371
+ return self
372
+
373
+ def __exit__(self, exc_type, exc_val, exc_tb):
374
+ self.close()
@@ -109,7 +109,8 @@ class SherpaNeuralEngine:
109
109
  elif p.is_file():
110
110
  search_dirs = [p.parent]
111
111
  else:
112
- search_dirs = list(self.STANDARD_MODEL_DIRS)
112
+ from .hardware import get_unified_model_search_dirs
113
+ search_dirs = list(get_unified_model_search_dirs("tts"))
113
114
 
114
115
  # 1. Find directory containing .onnx model, tokens.txt, and espeak-ng-data
115
116
  for sdir in search_dirs: