termux-tts 1.4.2 → 1.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +81 -381
- package/README.pypi.md +26 -410
- package/doc.config.yaml +101 -37
- package/package.json +1 -1
- package/pyproject.toml +1 -1
- package/termux_tts/__init__.py +11 -1
- package/termux_tts/audio.py +6 -3
- package/termux_tts/cli.py +27 -10
- package/termux_tts/engine.py +73 -7
- package/termux_tts/engine_multilingual.py +374 -0
- package/termux_tts/engine_sherpa.py +2 -1
- package/termux_tts/engine_sherpa_capi.py +491 -0
- package/termux_tts/hardware.py +34 -1
- package/termux_tts/installer.py +213 -11
- package/termux_tts/script_classifier.py +253 -0
- package/termux_tts/tokenizer.py +3 -1
|
@@ -0,0 +1,491 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Sherpa-ONNX C-API In-Memory Resident Neural Synthesis Engine and Session Manager.
|
|
3
|
+
Enables sub-second intra-sentential multilingual speech synthesis by keeping neural acoustic models
|
|
4
|
+
resident in RAM via libsherpa-onnx-c-api.so, eliminating cold subprocess process startup and model disk re-parsing.
|
|
5
|
+
|
|
6
|
+
Governance & Compliance:
|
|
7
|
+
- AOSF-ENG-STD-2026 / Zero-Deception: Zero silent fallbacks, zero synthetic stubs.
|
|
8
|
+
- Clean Lifecycle: Context manager, explicit close(), __del__, and global atexit teardown.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import atexit
|
|
13
|
+
import ctypes
|
|
14
|
+
import logging
|
|
15
|
+
import os
|
|
16
|
+
import shutil
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Optional, Dict, List, Any
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
|
|
22
|
+
from .audio import AudioBuffer
|
|
23
|
+
from .exceptions import TTSModelLoadError, TTSInferenceError
|
|
24
|
+
|
|
25
|
+
logger = logging.getLogger("termux_tts.engine_sherpa_capi")
|
|
26
|
+
|
|
27
|
+
# ---------------------------------------------------------------------------
|
|
28
|
+
# C-API Type Definitions matching sherpa-onnx/c-api/c-api.h
|
|
29
|
+
# ---------------------------------------------------------------------------
|
|
30
|
+
|
|
31
|
+
class SherpaOnnxOfflineTtsVitsModelConfig(ctypes.Structure):
|
|
32
|
+
_fields_ = [
|
|
33
|
+
("model", ctypes.c_char_p),
|
|
34
|
+
("lexicon", ctypes.c_char_p),
|
|
35
|
+
("tokens", ctypes.c_char_p),
|
|
36
|
+
("data_dir", ctypes.c_char_p),
|
|
37
|
+
("noise_scale", ctypes.c_float),
|
|
38
|
+
("noise_scale_w", ctypes.c_float),
|
|
39
|
+
("length_scale", ctypes.c_float),
|
|
40
|
+
("dict_dir", ctypes.c_char_p),
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class SherpaOnnxOfflineTtsMatchaModelConfig(ctypes.Structure):
|
|
45
|
+
_fields_ = [
|
|
46
|
+
("acoustic_model", ctypes.c_char_p),
|
|
47
|
+
("vocoder", ctypes.c_char_p),
|
|
48
|
+
("lexicon", ctypes.c_char_p),
|
|
49
|
+
("tokens", ctypes.c_char_p),
|
|
50
|
+
("data_dir", ctypes.c_char_p),
|
|
51
|
+
("noise_scale", ctypes.c_float),
|
|
52
|
+
("length_scale", ctypes.c_float),
|
|
53
|
+
("dict_dir", ctypes.c_char_p),
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class SherpaOnnxOfflineTtsKokoroModelConfig(ctypes.Structure):
|
|
58
|
+
_fields_ = [
|
|
59
|
+
("model", ctypes.c_char_p),
|
|
60
|
+
("voices", ctypes.c_char_p),
|
|
61
|
+
("tokens", ctypes.c_char_p),
|
|
62
|
+
("data_dir", ctypes.c_char_p),
|
|
63
|
+
("length_scale", ctypes.c_float),
|
|
64
|
+
("dict_dir", ctypes.c_char_p),
|
|
65
|
+
("lexicon", ctypes.c_char_p),
|
|
66
|
+
("lang", ctypes.c_char_p),
|
|
67
|
+
]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class SherpaOnnxOfflineTtsKittenModelConfig(ctypes.Structure):
|
|
71
|
+
_fields_ = [
|
|
72
|
+
("model", ctypes.c_char_p),
|
|
73
|
+
("voices", ctypes.c_char_p),
|
|
74
|
+
("tokens", ctypes.c_char_p),
|
|
75
|
+
("data_dir", ctypes.c_char_p),
|
|
76
|
+
("length_scale", ctypes.c_float),
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class SherpaOnnxOfflineTtsZipvoiceModelConfig(ctypes.Structure):
|
|
81
|
+
_fields_ = [
|
|
82
|
+
("tokens", ctypes.c_char_p),
|
|
83
|
+
("encoder", ctypes.c_char_p),
|
|
84
|
+
("decoder", ctypes.c_char_p),
|
|
85
|
+
("vocoder", ctypes.c_char_p),
|
|
86
|
+
("data_dir", ctypes.c_char_p),
|
|
87
|
+
("lexicon", ctypes.c_char_p),
|
|
88
|
+
("feat_scale", ctypes.c_float),
|
|
89
|
+
("t_shift", ctypes.c_float),
|
|
90
|
+
("target_rms", ctypes.c_float),
|
|
91
|
+
("guidance_scale", ctypes.c_float),
|
|
92
|
+
]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class SherpaOnnxOfflineTtsPocketModelConfig(ctypes.Structure):
|
|
96
|
+
_fields_ = [
|
|
97
|
+
("lm_flow", ctypes.c_char_p),
|
|
98
|
+
("lm_main", ctypes.c_char_p),
|
|
99
|
+
("encoder", ctypes.c_char_p),
|
|
100
|
+
("decoder", ctypes.c_char_p),
|
|
101
|
+
("text_conditioner", ctypes.c_char_p),
|
|
102
|
+
("vocab_json", ctypes.c_char_p),
|
|
103
|
+
("token_scores_json", ctypes.c_char_p),
|
|
104
|
+
("voice_embedding_cache_capacity", ctypes.c_int32),
|
|
105
|
+
]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class SherpaOnnxOfflineTtsSupertonicModelConfig(ctypes.Structure):
|
|
109
|
+
_fields_ = [
|
|
110
|
+
("duration_predictor", ctypes.c_char_p),
|
|
111
|
+
("text_encoder", ctypes.c_char_p),
|
|
112
|
+
("vector_estimator", ctypes.c_char_p),
|
|
113
|
+
("vocoder", ctypes.c_char_p),
|
|
114
|
+
("tts_json", ctypes.c_char_p),
|
|
115
|
+
("unicode_indexer", ctypes.c_char_p),
|
|
116
|
+
("voice_style", ctypes.c_char_p),
|
|
117
|
+
]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class SherpaOnnxOfflineTtsModelConfig(ctypes.Structure):
|
|
121
|
+
_fields_ = [
|
|
122
|
+
("vits", SherpaOnnxOfflineTtsVitsModelConfig),
|
|
123
|
+
("num_threads", ctypes.c_int32),
|
|
124
|
+
("debug", ctypes.c_int32),
|
|
125
|
+
("provider", ctypes.c_char_p),
|
|
126
|
+
("matcha", SherpaOnnxOfflineTtsMatchaModelConfig),
|
|
127
|
+
("kokoro", SherpaOnnxOfflineTtsKokoroModelConfig),
|
|
128
|
+
("kitten", SherpaOnnxOfflineTtsKittenModelConfig),
|
|
129
|
+
("zipvoice", SherpaOnnxOfflineTtsZipvoiceModelConfig),
|
|
130
|
+
("pocket", SherpaOnnxOfflineTtsPocketModelConfig),
|
|
131
|
+
("supertonic", SherpaOnnxOfflineTtsSupertonicModelConfig),
|
|
132
|
+
]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class SherpaOnnxOfflineTtsConfig(ctypes.Structure):
|
|
136
|
+
_fields_ = [
|
|
137
|
+
("model", SherpaOnnxOfflineTtsModelConfig),
|
|
138
|
+
("rule_fsts", ctypes.c_char_p),
|
|
139
|
+
("max_num_sentences", ctypes.c_int32),
|
|
140
|
+
("rule_fars", ctypes.c_char_p),
|
|
141
|
+
("silence_scale", ctypes.c_float),
|
|
142
|
+
]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
class SherpaOnnxGeneratedAudio(ctypes.Structure):
|
|
146
|
+
_fields_ = [
|
|
147
|
+
("samples", ctypes.POINTER(ctypes.c_float)),
|
|
148
|
+
("n", ctypes.c_int32),
|
|
149
|
+
("sample_rate", ctypes.c_int32),
|
|
150
|
+
]
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
# ---------------------------------------------------------------------------
|
|
154
|
+
# C-API Shared Library Loader
|
|
155
|
+
# ---------------------------------------------------------------------------
|
|
156
|
+
|
|
157
|
+
_CACHED_CDLL = None
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def find_sherpa_capi_library() -> Optional[str]:
|
|
161
|
+
"""Search standard library search locations for libsherpa-onnx-c-api."""
|
|
162
|
+
override = os.environ.get("SHERPA_ONNX_C_API_LIB")
|
|
163
|
+
if override and os.path.isfile(override):
|
|
164
|
+
return override
|
|
165
|
+
|
|
166
|
+
candidates = [
|
|
167
|
+
"libsherpa-onnx-c-api.so",
|
|
168
|
+
"/data/data/com.termux/files/usr/lib/libsherpa-onnx-c-api.so",
|
|
169
|
+
str(Path.home() / ".local" / "lib" / "libsherpa-onnx-c-api.so"),
|
|
170
|
+
str(Path.home() / "sherpa_termux" / "sherpa-onnx-v1.13.8-android-aarch64-termux-shared" / "lib" / "libsherpa-onnx-c-api.so"),
|
|
171
|
+
"/usr/local/lib/libsherpa-onnx-c-api.so",
|
|
172
|
+
"/usr/lib/libsherpa-onnx-c-api.so",
|
|
173
|
+
]
|
|
174
|
+
|
|
175
|
+
for cand in candidates:
|
|
176
|
+
if os.path.isabs(cand) and os.path.isfile(cand):
|
|
177
|
+
return cand
|
|
178
|
+
|
|
179
|
+
# Try system loader lookup
|
|
180
|
+
from ctypes.util import find_library
|
|
181
|
+
found = find_library("sherpa-onnx-c-api")
|
|
182
|
+
if found:
|
|
183
|
+
return found
|
|
184
|
+
|
|
185
|
+
return None
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def get_sherpa_capi_cdll():
|
|
189
|
+
"""Load and return the singleton libsherpa-onnx-c-api CDLL instance or raise TTSModelLoadError."""
|
|
190
|
+
global _CACHED_CDLL
|
|
191
|
+
if _CACHED_CDLL is not None:
|
|
192
|
+
return _CACHED_CDLL
|
|
193
|
+
|
|
194
|
+
lib_path = find_sherpa_capi_library()
|
|
195
|
+
if not lib_path:
|
|
196
|
+
raise TTSModelLoadError(
|
|
197
|
+
"[FAIL-FAST] libsherpa-onnx-c-api shared library not found.\n"
|
|
198
|
+
"Please ensure libsherpa-onnx-c-api.so is present in /data/data/com.termux/files/usr/lib/ or set SHERPA_ONNX_C_API_LIB."
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
try:
|
|
202
|
+
cdll = ctypes.CDLL(lib_path)
|
|
203
|
+
except Exception as err:
|
|
204
|
+
raise TTSModelLoadError(
|
|
205
|
+
f"[FAIL-FAST] Failed to load libsherpa-onnx-c-api shared library from '{lib_path}': {err}"
|
|
206
|
+
) from err
|
|
207
|
+
|
|
208
|
+
# Setup function signatures
|
|
209
|
+
cdll.SherpaOnnxCreateOfflineTts.argtypes = [ctypes.POINTER(SherpaOnnxOfflineTtsConfig)]
|
|
210
|
+
cdll.SherpaOnnxCreateOfflineTts.restype = ctypes.c_void_p
|
|
211
|
+
|
|
212
|
+
cdll.SherpaOnnxDestroyOfflineTts.argtypes = [ctypes.c_void_p]
|
|
213
|
+
cdll.SherpaOnnxDestroyOfflineTts.restype = None
|
|
214
|
+
|
|
215
|
+
cdll.SherpaOnnxOfflineTtsSampleRate.argtypes = [ctypes.c_void_p]
|
|
216
|
+
cdll.SherpaOnnxOfflineTtsSampleRate.restype = ctypes.c_int32
|
|
217
|
+
|
|
218
|
+
cdll.SherpaOnnxOfflineTtsGenerate.argtypes = [
|
|
219
|
+
ctypes.c_void_p,
|
|
220
|
+
ctypes.c_char_p,
|
|
221
|
+
ctypes.c_int32,
|
|
222
|
+
ctypes.c_float,
|
|
223
|
+
]
|
|
224
|
+
cdll.SherpaOnnxOfflineTtsGenerate.restype = ctypes.POINTER(SherpaOnnxGeneratedAudio)
|
|
225
|
+
|
|
226
|
+
cdll.SherpaOnnxDestroyOfflineTtsGeneratedAudio.argtypes = [
|
|
227
|
+
ctypes.POINTER(SherpaOnnxGeneratedAudio)
|
|
228
|
+
]
|
|
229
|
+
cdll.SherpaOnnxDestroyOfflineTtsGeneratedAudio.restype = None
|
|
230
|
+
|
|
231
|
+
_CACHED_CDLL = cdll
|
|
232
|
+
return _CACHED_CDLL
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
# ---------------------------------------------------------------------------
|
|
236
|
+
# Resident Session Class
|
|
237
|
+
# ---------------------------------------------------------------------------
|
|
238
|
+
|
|
239
|
+
class SherpaCapiSession:
|
|
240
|
+
"""
|
|
241
|
+
High-performance In-Memory Resident Sherpa-ONNX TTS Session.
|
|
242
|
+
Maintains loaded acoustic neural model in memory across multiple synthesize calls.
|
|
243
|
+
Guarantees strict resource deallocation upon exit/close().
|
|
244
|
+
"""
|
|
245
|
+
|
|
246
|
+
def __init__(
|
|
247
|
+
self,
|
|
248
|
+
model_path: str,
|
|
249
|
+
tokens_path: str,
|
|
250
|
+
data_dir: str,
|
|
251
|
+
language: str = "ko",
|
|
252
|
+
threads: int = 4,
|
|
253
|
+
sample_rate: int = 22050,
|
|
254
|
+
provider: str = "cpu",
|
|
255
|
+
):
|
|
256
|
+
self.language = language.lower()
|
|
257
|
+
self.threads = threads
|
|
258
|
+
self.sample_rate = sample_rate
|
|
259
|
+
self.provider = provider
|
|
260
|
+
self.model_path = model_path
|
|
261
|
+
self.tokens_path = tokens_path
|
|
262
|
+
self.data_dir = data_dir
|
|
263
|
+
self._is_closed = False
|
|
264
|
+
|
|
265
|
+
self._cdll = get_sherpa_capi_cdll()
|
|
266
|
+
|
|
267
|
+
# Build configuration struct
|
|
268
|
+
cfg = SherpaOnnxOfflineTtsConfig()
|
|
269
|
+
cfg.model.vits.model = model_path.encode("utf-8")
|
|
270
|
+
cfg.model.vits.tokens = tokens_path.encode("utf-8")
|
|
271
|
+
cfg.model.vits.data_dir = data_dir.encode("utf-8")
|
|
272
|
+
cfg.model.vits.noise_scale = 0.667
|
|
273
|
+
cfg.model.vits.noise_scale_w = 0.8
|
|
274
|
+
cfg.model.vits.length_scale = 1.0
|
|
275
|
+
|
|
276
|
+
cfg.model.num_threads = threads
|
|
277
|
+
cfg.model.debug = 0
|
|
278
|
+
cfg.model.provider = provider.encode("utf-8")
|
|
279
|
+
cfg.max_num_sentences = 2
|
|
280
|
+
cfg.silence_scale = 0.2
|
|
281
|
+
|
|
282
|
+
logger.debug("Creating Sherpa C-API resident engine for '%s' from %s", language, model_path)
|
|
283
|
+
self._handle = self._cdll.SherpaOnnxCreateOfflineTts(ctypes.byref(cfg))
|
|
284
|
+
if not self._handle:
|
|
285
|
+
raise TTSModelLoadError(
|
|
286
|
+
f"[FAIL-FAST] SherpaOnnxCreateOfflineTts failed for language '{language}' "
|
|
287
|
+
f"with model: {model_path}"
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
native_sr = self._cdll.SherpaOnnxOfflineTtsSampleRate(self._handle)
|
|
291
|
+
if native_sr > 0:
|
|
292
|
+
self.sample_rate = native_sr
|
|
293
|
+
|
|
294
|
+
def synthesize(self, text: str, speed: float = 1.0) -> AudioBuffer:
|
|
295
|
+
"""Synthesize text directly into an AudioBuffer using resident in-memory model."""
|
|
296
|
+
if self._is_closed or not self._handle:
|
|
297
|
+
raise TTSInferenceError("Cannot synthesize: Sherpa C-API resident session is closed.")
|
|
298
|
+
|
|
299
|
+
clean_text = text.strip()
|
|
300
|
+
if not clean_text:
|
|
301
|
+
return AudioBuffer(np.zeros(0, dtype=np.float32), sample_rate=self.sample_rate)
|
|
302
|
+
|
|
303
|
+
# Execute in-memory neural inference
|
|
304
|
+
audio_ptr = self._cdll.SherpaOnnxOfflineTtsGenerate(
|
|
305
|
+
self._handle,
|
|
306
|
+
clean_text.encode("utf-8"),
|
|
307
|
+
0,
|
|
308
|
+
float(speed),
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
if not audio_ptr:
|
|
312
|
+
raise TTSInferenceError(
|
|
313
|
+
f"[FAIL-FAST] Sherpa C-API generation returned NULL for '{clean_text}'."
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
try:
|
|
317
|
+
audio_obj = audio_ptr.contents
|
|
318
|
+
n_samples = audio_obj.n
|
|
319
|
+
out_sr = audio_obj.sample_rate if audio_obj.sample_rate > 0 else self.sample_rate
|
|
320
|
+
|
|
321
|
+
if n_samples <= 0:
|
|
322
|
+
return AudioBuffer(np.zeros(0, dtype=np.float32), sample_rate=out_sr)
|
|
323
|
+
|
|
324
|
+
# Copy float samples directly to NumPy array
|
|
325
|
+
samples = np.ctypeslib.as_array(audio_obj.samples, shape=(n_samples,)).copy()
|
|
326
|
+
return AudioBuffer(samples, sample_rate=out_sr)
|
|
327
|
+
finally:
|
|
328
|
+
self._cdll.SherpaOnnxDestroyOfflineTtsGeneratedAudio(audio_ptr)
|
|
329
|
+
|
|
330
|
+
def close(self) -> None:
|
|
331
|
+
"""Destroy native C++ TTS handle and release model memory."""
|
|
332
|
+
if not self._is_closed and getattr(self, "_handle", None):
|
|
333
|
+
logger.debug("Destroying Sherpa C-API engine handle for language '%s'", self.language)
|
|
334
|
+
try:
|
|
335
|
+
self._cdll.SherpaOnnxDestroyOfflineTts(self._handle)
|
|
336
|
+
except Exception as err:
|
|
337
|
+
logger.warning("Exception during Sherpa C-API teardown: %s", err)
|
|
338
|
+
finally:
|
|
339
|
+
self._handle = None
|
|
340
|
+
self._is_closed = True
|
|
341
|
+
|
|
342
|
+
def __del__(self) -> None:
|
|
343
|
+
self.close()
|
|
344
|
+
|
|
345
|
+
def __enter__(self):
|
|
346
|
+
return self
|
|
347
|
+
|
|
348
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
349
|
+
self.close()
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
# ---------------------------------------------------------------------------
|
|
353
|
+
# Resident Session Manager & Lifecycle Guard
|
|
354
|
+
# ---------------------------------------------------------------------------
|
|
355
|
+
|
|
356
|
+
class SherpaResidentManager:
|
|
357
|
+
"""
|
|
358
|
+
Global resident model session manager.
|
|
359
|
+
Coordinates memory-resident models across languages and ensures 100% clean shutdown upon process exit.
|
|
360
|
+
"""
|
|
361
|
+
|
|
362
|
+
_INSTANCE: Optional[SherpaResidentManager] = None
|
|
363
|
+
|
|
364
|
+
STANDARD_MODEL_DIRS = [
|
|
365
|
+
Path.home() / ".cache" / "termux-tts" / "models",
|
|
366
|
+
Path.home() / "vits-mimic3-ko_KO-kss_low",
|
|
367
|
+
Path.home() / "vits-piper-en_US-lessac-medium",
|
|
368
|
+
Path("/data/data/com.termux/files/home/.cache/termux-tts/models"),
|
|
369
|
+
]
|
|
370
|
+
|
|
371
|
+
def __init__(self):
|
|
372
|
+
self._sessions: Dict[str, SherpaCapiSession] = {}
|
|
373
|
+
atexit.register(self.close_all)
|
|
374
|
+
|
|
375
|
+
@classmethod
|
|
376
|
+
def get_instance(cls) -> SherpaResidentManager:
|
|
377
|
+
if cls._INSTANCE is None:
|
|
378
|
+
cls._INSTANCE = SherpaResidentManager()
|
|
379
|
+
return cls._INSTANCE
|
|
380
|
+
|
|
381
|
+
def resolve_model_assets(self, language: str, custom_dir: Optional[str] = None) -> Dict[str, str]:
|
|
382
|
+
"""Locate ONNX, tokens.txt, and espeak-ng-data for a given language."""
|
|
383
|
+
lang = language.lower()
|
|
384
|
+
search_dirs: List[Path] = []
|
|
385
|
+
if custom_dir:
|
|
386
|
+
search_dirs.append(Path(custom_dir).expanduser().resolve())
|
|
387
|
+
from .hardware import get_unified_model_search_dirs
|
|
388
|
+
search_dirs.extend(get_unified_model_search_dirs("tts"))
|
|
389
|
+
|
|
390
|
+
# Expected directory sub-names
|
|
391
|
+
expected_names = {
|
|
392
|
+
"ko": ["vits-mimic3-ko_KO-kss_low"],
|
|
393
|
+
"en": ["vits-piper-en_US-lessac-medium", "vits-piper-en_US-lessac-high"],
|
|
394
|
+
"ja": ["vits-piper-ja_JP-hina-medium", "vits-piper-ja_JP-kokoro", "vits-ja_JP"],
|
|
395
|
+
"zh": ["vits-zh-aishell3", "vits-zh_CN", "vits-piper-zh_CN"],
|
|
396
|
+
"hi": ["vits-piper-hi_IN-swara-medium", "vits-piper-hi_IN"],
|
|
397
|
+
"ru": ["vits-piper-ru_RU-dmitri-medium", "vits-piper-ru_RU"],
|
|
398
|
+
"es": ["vits-piper-es_ES-davefx-medium"],
|
|
399
|
+
"fr": ["vits-piper-fr_FR-siwis-medium"],
|
|
400
|
+
"de": ["vits-piper-de_DE-thorsten-medium"],
|
|
401
|
+
}.get(lang, [f"vits-{lang}", f"vits-piper-{lang}"])
|
|
402
|
+
|
|
403
|
+
for sdir in search_dirs:
|
|
404
|
+
if not sdir.exists():
|
|
405
|
+
continue
|
|
406
|
+
|
|
407
|
+
candidates = [sdir]
|
|
408
|
+
for exp in expected_names:
|
|
409
|
+
candidates.append(sdir / exp)
|
|
410
|
+
|
|
411
|
+
# Also search any subdirectory matching language name or code
|
|
412
|
+
try:
|
|
413
|
+
for sub in sdir.iterdir():
|
|
414
|
+
if sub.is_dir() and (lang in sub.name.lower() or any(exp.lower() in sub.name.lower() for exp in expected_names)):
|
|
415
|
+
if sub not in candidates:
|
|
416
|
+
candidates.append(sub)
|
|
417
|
+
except OSError:
|
|
418
|
+
pass
|
|
419
|
+
|
|
420
|
+
for cand in candidates:
|
|
421
|
+
if not cand.is_dir():
|
|
422
|
+
continue
|
|
423
|
+
|
|
424
|
+
onnx_files = list(cand.glob("*.onnx"))
|
|
425
|
+
if not onnx_files:
|
|
426
|
+
continue
|
|
427
|
+
|
|
428
|
+
onnx_model = str(onnx_files[0])
|
|
429
|
+
tokens_file = cand / "tokens.txt"
|
|
430
|
+
espeak_dir = cand / "espeak-ng-data"
|
|
431
|
+
|
|
432
|
+
if not tokens_file.exists():
|
|
433
|
+
tokens_file = cand.parent / "tokens.txt"
|
|
434
|
+
if not espeak_dir.exists():
|
|
435
|
+
espeak_dir = cand.parent / "espeak-ng-data"
|
|
436
|
+
|
|
437
|
+
if tokens_file.exists() and espeak_dir.exists():
|
|
438
|
+
return {
|
|
439
|
+
"model": str(onnx_model),
|
|
440
|
+
"tokens": str(tokens_file),
|
|
441
|
+
"data_dir": str(espeak_dir),
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
from .installer import OFFICIAL_NEURAL_MODELS
|
|
445
|
+
meta = OFFICIAL_NEURAL_MODELS.get(lang, {})
|
|
446
|
+
lang_name = meta.get("language_name", lang.upper())
|
|
447
|
+
repo = meta.get("repo", f"csukuangfj/vits-{lang}")
|
|
448
|
+
model_name = meta.get("name", expected_names[0])
|
|
449
|
+
canonical_dest = str((Path.home() / "models" / "tts" / model_name).resolve())
|
|
450
|
+
|
|
451
|
+
raise TTSModelLoadError(
|
|
452
|
+
f"[FAIL-FAST] Neural speech model for language '{lang}' ({lang_name}) is not installed.\n\n"
|
|
453
|
+
f" To install this model automatically, run:\n"
|
|
454
|
+
f" termux-tts install --models {lang}\n\n"
|
|
455
|
+
f" Or manually download via git clone:\n"
|
|
456
|
+
f" git clone https://huggingface.co/{repo} {canonical_dest}\n\n"
|
|
457
|
+
f" After installation, re-run your synthesis command!"
|
|
458
|
+
)
|
|
459
|
+
|
|
460
|
+
def get_session(
|
|
461
|
+
self,
|
|
462
|
+
language: str,
|
|
463
|
+
custom_model_dir: Optional[str] = None,
|
|
464
|
+
threads: int = 4,
|
|
465
|
+
sample_rate: int = 22050,
|
|
466
|
+
) -> SherpaCapiSession:
|
|
467
|
+
"""Get or lazily instantiate an in-memory resident session for the requested language."""
|
|
468
|
+
lang = language.lower()
|
|
469
|
+
if lang in self._sessions and not self._sessions[lang]._is_closed:
|
|
470
|
+
return self._sessions[lang]
|
|
471
|
+
|
|
472
|
+
assets = self.resolve_model_assets(lang, custom_dir=custom_model_dir)
|
|
473
|
+
session = SherpaCapiSession(
|
|
474
|
+
model_path=assets["model"],
|
|
475
|
+
tokens_path=assets["tokens"],
|
|
476
|
+
data_dir=assets["data_dir"],
|
|
477
|
+
language=lang,
|
|
478
|
+
threads=threads,
|
|
479
|
+
sample_rate=sample_rate,
|
|
480
|
+
)
|
|
481
|
+
self._sessions[lang] = session
|
|
482
|
+
return session
|
|
483
|
+
|
|
484
|
+
def close_all(self) -> None:
|
|
485
|
+
"""Cleanly terminate all resident sessions."""
|
|
486
|
+
for lang, session in list(self._sessions.items()):
|
|
487
|
+
try:
|
|
488
|
+
session.close()
|
|
489
|
+
except Exception as err:
|
|
490
|
+
logger.warning("Error closing session '%s': %s", lang, err)
|
|
491
|
+
self._sessions.clear()
|
package/termux_tts/hardware.py
CHANGED
|
@@ -95,7 +95,7 @@ def resolve_device_backend(
|
|
|
95
95
|
report = adapter.resolve_diagnostic_report()
|
|
96
96
|
is_vk = getattr(report, "overall_success", False) or getattr(report, "recommended_backend", "") == "vulkan"
|
|
97
97
|
bin_path = adapter.resolve_binary_path()
|
|
98
|
-
if is_vk and bin_path and req_eng in ("
|
|
98
|
+
if is_vk and bin_path and req_eng in ("vulkan", "gpu"):
|
|
99
99
|
return "vulkan", "vulkan"
|
|
100
100
|
except Exception as e:
|
|
101
101
|
logger.debug("TtsAdapter auto-routing probe exception: %s", e)
|
|
@@ -119,3 +119,36 @@ def bind_tts_hardware(engine: Any, requested_device: str) -> Optional[Any]:
|
|
|
119
119
|
except Exception as e:
|
|
120
120
|
logger.debug("Hardware adapter binding skipped: %s", e)
|
|
121
121
|
return None
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def get_unified_model_search_dirs(submodule: str = "tts") -> list:
|
|
125
|
+
"""
|
|
126
|
+
Returns unified model search paths adhering to AMEVA Ecosystem Shared Storage Specification.
|
|
127
|
+
Enables zero-redundancy model sharing across STT, TTS, LLaMA, Vision, and Diffusion.
|
|
128
|
+
"""
|
|
129
|
+
import os
|
|
130
|
+
from pathlib import Path
|
|
131
|
+
|
|
132
|
+
home = Path.home()
|
|
133
|
+
dirs = []
|
|
134
|
+
|
|
135
|
+
env_dir = os.environ.get("AMEVA_MODELS_DIR")
|
|
136
|
+
if env_dir:
|
|
137
|
+
p = Path(env_dir)
|
|
138
|
+
dirs.extend([p / submodule, p])
|
|
139
|
+
|
|
140
|
+
dirs.extend([
|
|
141
|
+
home / "models" / submodule,
|
|
142
|
+
home / "models",
|
|
143
|
+
home / "ameva-models" / submodule,
|
|
144
|
+
home / "ameva-models",
|
|
145
|
+
home / ".cache" / "ameva" / "models" / submodule,
|
|
146
|
+
home / ".cache" / "ameva" / "models",
|
|
147
|
+
Path(f"/data/data/com.termux/files/home/models/{submodule}"),
|
|
148
|
+
Path("/data/data/com.termux/files/home/models"),
|
|
149
|
+
Path(f"/data/data/com.termux/files/home/ameva-models/{submodule}"),
|
|
150
|
+
Path("/data/data/com.termux/files/home/ameva-models"),
|
|
151
|
+
home / ".cache" / f"termux-{submodule}" / "models",
|
|
152
|
+
Path(f"/data/data/com.termux/files/home/.cache/termux-{submodule}/models"),
|
|
153
|
+
])
|
|
154
|
+
return dirs
|