termux-tts 1.4.2 → 1.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +81 -381
- package/README.pypi.md +26 -410
- package/doc.config.yaml +101 -37
- package/package.json +1 -1
- package/pyproject.toml +1 -1
- package/termux_tts/__init__.py +11 -1
- package/termux_tts/audio.py +6 -3
- package/termux_tts/cli.py +27 -10
- package/termux_tts/engine.py +73 -7
- package/termux_tts/engine_multilingual.py +374 -0
- package/termux_tts/engine_sherpa.py +2 -1
- package/termux_tts/engine_sherpa_capi.py +491 -0
- package/termux_tts/hardware.py +34 -1
- package/termux_tts/installer.py +213 -11
- package/termux_tts/script_classifier.py +253 -0
- package/termux_tts/tokenizer.py +3 -1
|
@@ -0,0 +1,374 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Enterprise Extensible Multilingual Neural Speech Synthesis Orchestrator for termux-tts.
|
|
3
|
+
Orchestrates multi-script tokenization, memory-resident neural acoustic inference via Sherpa C-API
|
|
4
|
+
(sub-second latency without subprocess spawning), silence trimming, and lossless PCM stream stitching.
|
|
5
|
+
Zero-Silent-Fallback compliant: Fails fast if a detected language has no registered neural model.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import logging
|
|
10
|
+
import os
|
|
11
|
+
import shutil
|
|
12
|
+
import subprocess
|
|
13
|
+
import tempfile
|
|
14
|
+
import time
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Optional, List, Dict, Any, Set
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
|
|
21
|
+
from .audio import AudioBuffer
|
|
22
|
+
from .exceptions import TTSModelLoadError, TTSInferenceError
|
|
23
|
+
from .script_classifier import MultilingualTokenizer, LanguageChunk, ScriptRegistry
|
|
24
|
+
|
|
25
|
+
logger = logging.getLogger("termux_tts.engine_multilingual")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class MultilingualResult:
|
|
30
|
+
text: str
|
|
31
|
+
audio_buffer: AudioBuffer
|
|
32
|
+
sample_rate: int
|
|
33
|
+
duration_sec: float
|
|
34
|
+
elapsed_ms: float
|
|
35
|
+
rtf: float
|
|
36
|
+
chunks: List[LanguageChunk]
|
|
37
|
+
languages_detected: List[str]
|
|
38
|
+
backend: str = "MULTILINGUAL_NEURAL_ORCHESTRATOR"
|
|
39
|
+
model_name: str = "hybrid-multilingual-mesh"
|
|
40
|
+
|
|
41
|
+
def save(self, filepath: str) -> str:
|
|
42
|
+
return self.audio_buffer.save(filepath)
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def wav_bytes(self) -> bytes:
|
|
46
|
+
return self.audio_buffer.to_wav_bytes()
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def trim_silence_samples(samples: np.ndarray, threshold: float = 0.01) -> np.ndarray:
|
|
50
|
+
"""
|
|
51
|
+
Trims leading and trailing silence from normalized float32 samples [-1.0, 1.0].
|
|
52
|
+
Preserves acoustic continuity across intra-word affix boundaries.
|
|
53
|
+
"""
|
|
54
|
+
if len(samples) == 0:
|
|
55
|
+
return samples
|
|
56
|
+
|
|
57
|
+
abs_samples = np.abs(samples)
|
|
58
|
+
non_silent = np.where(abs_samples >= threshold)[0]
|
|
59
|
+
if len(non_silent) == 0:
|
|
60
|
+
return samples
|
|
61
|
+
|
|
62
|
+
start_idx = non_silent[0]
|
|
63
|
+
end_idx = non_silent[-1]
|
|
64
|
+
return samples[start_idx:end_idx + 1]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class MultilingualNeuralEngine:
|
|
68
|
+
"""
|
|
69
|
+
Extensible Multilingual Orchestration Engine.
|
|
70
|
+
Dynamically routes linguistic sub-phrases to resident on-device neural acoustic backends.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
CANDIDATE_ONNX_BINARIES = [
|
|
74
|
+
"sherpa-onnx-offline-tts",
|
|
75
|
+
str(Path.home() / ".local" / "bin" / "sherpa-onnx-offline-tts"),
|
|
76
|
+
str(Path.home() / "sherpa-onnx-offline-tts"),
|
|
77
|
+
"/data/data/com.termux/files/home/.local/bin/sherpa-onnx-offline-tts",
|
|
78
|
+
"/data/data/com.termux/files/usr/bin/sherpa-onnx-offline-tts",
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
CANDIDATE_NCNN_BINARIES = [
|
|
82
|
+
"sherpa-ncnn-offline-tts",
|
|
83
|
+
str(Path.home() / ".local" / "bin" / "sherpa-ncnn-offline-tts"),
|
|
84
|
+
str(Path.home() / "sherpa-ncnn" / "build-vulkan" / "bin" / "sherpa-ncnn-offline-tts"),
|
|
85
|
+
"/data/data/com.termux/files/home/.local/bin/sherpa-ncnn-offline-tts",
|
|
86
|
+
"/data/data/com.termux/files/usr/bin/sherpa-ncnn-offline-tts",
|
|
87
|
+
]
|
|
88
|
+
|
|
89
|
+
STANDARD_MODEL_DIRS = [
|
|
90
|
+
Path.home() / ".cache" / "termux-tts" / "models",
|
|
91
|
+
Path.home() / "vits-mimic3-ko_KO-kss_low",
|
|
92
|
+
Path.home() / "vits-piper-en_US-lessac-medium",
|
|
93
|
+
Path.home() / "ncnn-vits-piper-en_US-lessac-high-fp16",
|
|
94
|
+
Path("/data/data/com.termux/files/home/.cache/termux-tts/models"),
|
|
95
|
+
]
|
|
96
|
+
|
|
97
|
+
def __init__(
|
|
98
|
+
self,
|
|
99
|
+
threads: int = 4,
|
|
100
|
+
device: str = "cpu",
|
|
101
|
+
sample_rate: int = 22050,
|
|
102
|
+
model_path_map: Optional[Dict[str, str]] = None,
|
|
103
|
+
):
|
|
104
|
+
self.threads = threads
|
|
105
|
+
self.device = device.lower()
|
|
106
|
+
self.sample_rate = sample_rate
|
|
107
|
+
self.model_path_map = model_path_map or {}
|
|
108
|
+
self.tokenizer = MultilingualTokenizer()
|
|
109
|
+
self._engines: Dict[str, Any] = {}
|
|
110
|
+
self._resident_manager = None
|
|
111
|
+
self._is_closed = False
|
|
112
|
+
|
|
113
|
+
# Cache directories
|
|
114
|
+
self.work_dir = Path.home() / ".cache" / "termux-tts" / "chunks"
|
|
115
|
+
self.work_dir.mkdir(parents=True, exist_ok=True)
|
|
116
|
+
|
|
117
|
+
def _get_resident_manager(self):
|
|
118
|
+
if self._resident_manager is None:
|
|
119
|
+
try:
|
|
120
|
+
from .engine_sherpa_capi import SherpaResidentManager
|
|
121
|
+
self._resident_manager = SherpaResidentManager.get_instance()
|
|
122
|
+
except Exception as err:
|
|
123
|
+
logger.debug("Sherpa C-API resident manager unavailable: %s", err)
|
|
124
|
+
return self._resident_manager
|
|
125
|
+
|
|
126
|
+
def register_engine(self, language: str, engine_instance: Any) -> None:
|
|
127
|
+
"""Register or override a neural backend engine for a specific language code."""
|
|
128
|
+
self._engines[language.lower()] = engine_instance
|
|
129
|
+
|
|
130
|
+
def _find_binary(self, candidates: List[str]) -> Optional[str]:
|
|
131
|
+
for c in candidates:
|
|
132
|
+
found = shutil.which(c) if not os.path.isabs(c) else c
|
|
133
|
+
if found and os.path.isfile(found) and (os.access(found, os.X_OK) or os.name == "nt"):
|
|
134
|
+
return str(found)
|
|
135
|
+
return None
|
|
136
|
+
|
|
137
|
+
def _resolve_korean_assets(self) -> Dict[str, str]:
|
|
138
|
+
custom = self.model_path_map.get("ko")
|
|
139
|
+
from .hardware import get_unified_model_search_dirs
|
|
140
|
+
dirs = [Path(custom)] if custom else get_unified_model_search_dirs("tts")
|
|
141
|
+
for d in dirs:
|
|
142
|
+
cand = d / "vits-mimic3-ko_KO-kss_low" if not d.name.startswith("vits") else d
|
|
143
|
+
onnx = cand / "ko_KO-kss_low.onnx"
|
|
144
|
+
tokens = cand / "tokens.txt"
|
|
145
|
+
espeak = cand / "espeak-ng-data"
|
|
146
|
+
if onnx.is_file() and tokens.is_file() and espeak.is_dir():
|
|
147
|
+
return {"onnx": str(onnx), "tokens": str(tokens), "data_dir": str(espeak)}
|
|
148
|
+
raise TTSModelLoadError("[FAIL-FAST] Korean VITS model assets (ko_KO-kss_low) not found in cache.")
|
|
149
|
+
|
|
150
|
+
def _resolve_english_assets(self) -> str:
|
|
151
|
+
custom = self.model_path_map.get("en")
|
|
152
|
+
from .hardware import get_unified_model_search_dirs
|
|
153
|
+
dirs = [Path(custom)] if custom else get_unified_model_search_dirs("tts")
|
|
154
|
+
for d in dirs:
|
|
155
|
+
cand = d / "ncnn-vits-piper-en_US-lessac-high-fp16" if not "lessac" in d.name else d
|
|
156
|
+
if (cand / "decoder.ncnn.bin").is_file():
|
|
157
|
+
return str(cand)
|
|
158
|
+
raise TTSModelLoadError("[FAIL-FAST] English NCNN model assets (lessac-high-fp16) not found in cache.")
|
|
159
|
+
|
|
160
|
+
def _synthesize_chunk_audio(self, text: str, lang: str, speed: float = 1.0) -> AudioBuffer:
|
|
161
|
+
"""Synthesize a single linguistic chunk via fastest on-device resident engine."""
|
|
162
|
+
# 1. Check custom registered engines
|
|
163
|
+
if lang in self._engines:
|
|
164
|
+
chunk_res = self._engines[lang].synthesize(text, speed=speed)
|
|
165
|
+
return chunk_res.audio_buffer
|
|
166
|
+
|
|
167
|
+
# 2. In-Memory Resident C-API Engine (Zero cold-start overhead)
|
|
168
|
+
mgr = self._get_resident_manager()
|
|
169
|
+
if mgr is not None:
|
|
170
|
+
try:
|
|
171
|
+
session = mgr.get_session(
|
|
172
|
+
language=lang,
|
|
173
|
+
custom_model_dir=self.model_path_map.get(lang),
|
|
174
|
+
threads=self.threads,
|
|
175
|
+
sample_rate=self.sample_rate,
|
|
176
|
+
)
|
|
177
|
+
return session.synthesize(text, speed=speed)
|
|
178
|
+
except TTSModelLoadError:
|
|
179
|
+
# If model assets missing or C-API failed, proceed to fallback/fail-fast
|
|
180
|
+
raise
|
|
181
|
+
except Exception as e:
|
|
182
|
+
logger.debug("Resident C-API execution bypassed: %s", e)
|
|
183
|
+
|
|
184
|
+
# 3. Fallback to Subprocess for Korean (ONNX CPU)
|
|
185
|
+
if lang == "ko":
|
|
186
|
+
onnx_bin = self._find_binary(self.CANDIDATE_ONNX_BINARIES)
|
|
187
|
+
if not onnx_bin:
|
|
188
|
+
raise TTSModelLoadError("[FAIL-FAST] sherpa-onnx-offline-tts binary not found for Korean synthesis.")
|
|
189
|
+
assets = self._resolve_korean_assets()
|
|
190
|
+
|
|
191
|
+
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
|
|
192
|
+
temp_wav = tmp.name
|
|
193
|
+
|
|
194
|
+
try:
|
|
195
|
+
cmd = [
|
|
196
|
+
onnx_bin,
|
|
197
|
+
f"--vits-model={assets['onnx']}",
|
|
198
|
+
f"--vits-tokens={assets['tokens']}",
|
|
199
|
+
f"--vits-data-dir={assets['data_dir']}",
|
|
200
|
+
f"--num-threads={self.threads}",
|
|
201
|
+
f"--speed={speed:.2f}",
|
|
202
|
+
f"--output-filename={temp_wav}",
|
|
203
|
+
text
|
|
204
|
+
]
|
|
205
|
+
env = os.environ.copy()
|
|
206
|
+
env["LANG"] = "C.UTF-8"
|
|
207
|
+
env["LC_ALL"] = "C.UTF-8"
|
|
208
|
+
res = subprocess.run(cmd, capture_output=True, text=True, env=env, timeout=60)
|
|
209
|
+
if res.returncode != 0 or not os.path.exists(temp_wav):
|
|
210
|
+
raise TTSInferenceError(f"[FAIL-FAST] Korean chunk failed for '{text}': {res.stderr}")
|
|
211
|
+
return AudioBuffer.from_wav_file(temp_wav)
|
|
212
|
+
finally:
|
|
213
|
+
if os.path.exists(temp_wav):
|
|
214
|
+
try:
|
|
215
|
+
os.remove(temp_wav)
|
|
216
|
+
except OSError:
|
|
217
|
+
pass
|
|
218
|
+
|
|
219
|
+
# 4. Fallback to Subprocess for English (NCNN CPU)
|
|
220
|
+
elif lang == "en":
|
|
221
|
+
ncnn_bin = self._find_binary(self.CANDIDATE_NCNN_BINARIES)
|
|
222
|
+
if not ncnn_bin:
|
|
223
|
+
raise TTSModelLoadError("[FAIL-FAST] sherpa-ncnn-offline-tts binary not found for English synthesis.")
|
|
224
|
+
model_dir = self._resolve_english_assets()
|
|
225
|
+
|
|
226
|
+
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
|
|
227
|
+
temp_wav = tmp.name
|
|
228
|
+
|
|
229
|
+
try:
|
|
230
|
+
use_vk = "1" if self.device in ("vulkan", "gpu") else "0"
|
|
231
|
+
cmd = [
|
|
232
|
+
ncnn_bin,
|
|
233
|
+
f"--vits-model-dir={model_dir}",
|
|
234
|
+
f"--use-vulkan-compute={use_vk}",
|
|
235
|
+
f"--num-threads={self.threads}",
|
|
236
|
+
f"--output-filename={temp_wav}",
|
|
237
|
+
text
|
|
238
|
+
]
|
|
239
|
+
env = os.environ.copy()
|
|
240
|
+
env["LANG"] = "C.UTF-8"
|
|
241
|
+
env["LC_ALL"] = "C.UTF-8"
|
|
242
|
+
res = subprocess.run(cmd, capture_output=True, text=True, env=env, timeout=60)
|
|
243
|
+
if res.returncode != 0 or not os.path.exists(temp_wav) or os.path.getsize(temp_wav) == 0:
|
|
244
|
+
if "empty ids for word" in res.stderr:
|
|
245
|
+
logger.warning("OOV word '%s' in English dictionary; inserting soft breath pause.", text)
|
|
246
|
+
return AudioBuffer(np.zeros(int(self.sample_rate * 0.1), dtype=np.float32), sample_rate=self.sample_rate)
|
|
247
|
+
raise TTSInferenceError(f"[FAIL-FAST] English chunk failed for '{text}': {res.stderr}")
|
|
248
|
+
return AudioBuffer.from_wav_file(temp_wav)
|
|
249
|
+
finally:
|
|
250
|
+
if os.path.exists(temp_wav):
|
|
251
|
+
try:
|
|
252
|
+
os.remove(temp_wav)
|
|
253
|
+
except OSError:
|
|
254
|
+
pass
|
|
255
|
+
|
|
256
|
+
# 5. Extended Languages (ja, zh, hi, ru, es, fr, de, ar)
|
|
257
|
+
elif lang in ("ja", "zh", "hi", "ru", "es", "fr", "de", "ar"):
|
|
258
|
+
custom_model = self.model_path_map.get(lang)
|
|
259
|
+
if not custom_model:
|
|
260
|
+
from .installer import OFFICIAL_NEURAL_MODELS
|
|
261
|
+
meta = OFFICIAL_NEURAL_MODELS.get(lang, {})
|
|
262
|
+
lang_name = meta.get("language_name", lang.upper())
|
|
263
|
+
repo = meta.get("repo", f"csukuangfj/vits-{lang}")
|
|
264
|
+
model_name = meta.get("name", f"vits-{lang}")
|
|
265
|
+
canonical_dest = str((Path.home() / "models" / "tts" / model_name).resolve())
|
|
266
|
+
raise TTSModelLoadError(
|
|
267
|
+
f"[FAIL-FAST] Neural speech model for language '{lang}' ({lang_name}) is not installed.\n\n"
|
|
268
|
+
f" To install this model automatically, run:\n"
|
|
269
|
+
f" termux-tts install --models {lang}\n\n"
|
|
270
|
+
f" Or manually download via git clone:\n"
|
|
271
|
+
f" git clone https://huggingface.co/{repo} {canonical_dest}\n\n"
|
|
272
|
+
f" After installation, re-run your synthesis command!"
|
|
273
|
+
)
|
|
274
|
+
from .engine_sherpa import SherpaNeuralEngine
|
|
275
|
+
engine = SherpaNeuralEngine(model_path=custom_model, language=lang, threads=self.threads, sample_rate=self.sample_rate)
|
|
276
|
+
self._engines[lang] = engine
|
|
277
|
+
return engine.synthesize(text, speed=speed).audio_buffer
|
|
278
|
+
|
|
279
|
+
raise TTSModelLoadError(f"[FAIL-FAST] Unsupported language '{lang}'. Supported languages: ['ko', 'en', 'ja', 'zh', 'hi', 'ru', 'es', 'fr', 'de', 'ar'].")
|
|
280
|
+
|
|
281
|
+
def synthesize(
|
|
282
|
+
self,
|
|
283
|
+
text: str,
|
|
284
|
+
output: Optional[str] = None,
|
|
285
|
+
speed: float = 1.0,
|
|
286
|
+
language: Optional[str] = None,
|
|
287
|
+
) -> MultilingualResult:
|
|
288
|
+
"""
|
|
289
|
+
Synthesizes speech with zero cross-linguistic phoneme distortion.
|
|
290
|
+
Supports automatic multi-script code-switching or forced language isolation.
|
|
291
|
+
"""
|
|
292
|
+
if self._is_closed:
|
|
293
|
+
raise TTSInferenceError("Cannot synthesize: Multilingual session is closed.")
|
|
294
|
+
|
|
295
|
+
clean_text = text.strip()
|
|
296
|
+
if not clean_text:
|
|
297
|
+
raise TTSInferenceError("Cannot synthesize empty text.")
|
|
298
|
+
|
|
299
|
+
t0 = time.perf_counter()
|
|
300
|
+
|
|
301
|
+
# 1. Parse text into language-tagged chunks (respecting forced language if provided)
|
|
302
|
+
chunks = self.tokenizer.tokenize(clean_text, force_language=language)
|
|
303
|
+
if not chunks:
|
|
304
|
+
raise TTSInferenceError("Tokenization yielded no synthesizable segments.")
|
|
305
|
+
|
|
306
|
+
detected_langs = list(dict.fromkeys(c.language for c in chunks))
|
|
307
|
+
|
|
308
|
+
# 2. Multi-chunk synthesis & seamless concatenation
|
|
309
|
+
assembled_samples: List[np.ndarray] = []
|
|
310
|
+
target_sample_rate = self.sample_rate
|
|
311
|
+
|
|
312
|
+
for i, chunk in enumerate(chunks):
|
|
313
|
+
chunk_buf = self._synthesize_chunk_audio(chunk.text, chunk.language, speed=speed)
|
|
314
|
+
|
|
315
|
+
# Get raw float32 samples and trim acoustic silence
|
|
316
|
+
raw_samples = chunk_buf.samples
|
|
317
|
+
trimmed = trim_silence_samples(raw_samples, threshold=0.01)
|
|
318
|
+
assembled_samples.append(trimmed)
|
|
319
|
+
|
|
320
|
+
# Add context-aware pause
|
|
321
|
+
if chunk.pause_after > 0:
|
|
322
|
+
pause_len = int(target_sample_rate * chunk.pause_after)
|
|
323
|
+
silence_gap = np.zeros(pause_len, dtype=np.float32)
|
|
324
|
+
assembled_samples.append(silence_gap)
|
|
325
|
+
|
|
326
|
+
# 3. Concatenate and build final AudioBuffer
|
|
327
|
+
if assembled_samples:
|
|
328
|
+
final_samples = np.concatenate(assembled_samples)
|
|
329
|
+
else:
|
|
330
|
+
final_samples = np.zeros(0, dtype=np.float32)
|
|
331
|
+
|
|
332
|
+
combined_buffer = AudioBuffer(final_samples, sample_rate=target_sample_rate)
|
|
333
|
+
|
|
334
|
+
elapsed_ms = (time.perf_counter() - t0) * 1000.0
|
|
335
|
+
dur_sec = combined_buffer.duration_seconds
|
|
336
|
+
rtf = (elapsed_ms / 1000.0) / max(0.001, dur_sec)
|
|
337
|
+
|
|
338
|
+
if output:
|
|
339
|
+
combined_buffer.save(output)
|
|
340
|
+
|
|
341
|
+
return MultilingualResult(
|
|
342
|
+
text=clean_text,
|
|
343
|
+
audio_buffer=combined_buffer,
|
|
344
|
+
sample_rate=target_sample_rate,
|
|
345
|
+
duration_sec=dur_sec,
|
|
346
|
+
elapsed_ms=elapsed_ms,
|
|
347
|
+
rtf=rtf,
|
|
348
|
+
chunks=chunks,
|
|
349
|
+
languages_detected=detected_langs,
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
def close(self) -> None:
|
|
353
|
+
"""Cleanly terminate all engine instances and release resident memory."""
|
|
354
|
+
self._is_closed = True
|
|
355
|
+
for engine in self._engines.values():
|
|
356
|
+
if hasattr(engine, "close"):
|
|
357
|
+
try:
|
|
358
|
+
engine.close()
|
|
359
|
+
except Exception:
|
|
360
|
+
pass
|
|
361
|
+
if self._resident_manager is not None:
|
|
362
|
+
try:
|
|
363
|
+
self._resident_manager.close_all()
|
|
364
|
+
except Exception:
|
|
365
|
+
pass
|
|
366
|
+
|
|
367
|
+
def __del__(self) -> None:
|
|
368
|
+
self.close()
|
|
369
|
+
|
|
370
|
+
def __enter__(self):
|
|
371
|
+
return self
|
|
372
|
+
|
|
373
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
374
|
+
self.close()
|
|
@@ -109,7 +109,8 @@ class SherpaNeuralEngine:
|
|
|
109
109
|
elif p.is_file():
|
|
110
110
|
search_dirs = [p.parent]
|
|
111
111
|
else:
|
|
112
|
-
|
|
112
|
+
from .hardware import get_unified_model_search_dirs
|
|
113
|
+
search_dirs = list(get_unified_model_search_dirs("tts"))
|
|
113
114
|
|
|
114
115
|
# 1. Find directory containing .onnx model, tokens.txt, and espeak-ng-data
|
|
115
116
|
for sdir in search_dirs:
|