termux-tts 1.4.2 → 1.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +81 -381
- package/README.pypi.md +26 -410
- package/doc.config.yaml +101 -37
- package/package.json +1 -1
- package/pyproject.toml +1 -1
- package/termux_tts/__init__.py +11 -1
- package/termux_tts/audio.py +6 -3
- package/termux_tts/cli.py +27 -10
- package/termux_tts/engine.py +73 -7
- package/termux_tts/engine_multilingual.py +374 -0
- package/termux_tts/engine_sherpa.py +2 -1
- package/termux_tts/engine_sherpa_capi.py +491 -0
- package/termux_tts/hardware.py +34 -1
- package/termux_tts/installer.py +213 -11
- package/termux_tts/script_classifier.py +253 -0
- package/termux_tts/tokenizer.py +3 -1
package/doc.config.yaml
CHANGED
|
@@ -8,8 +8,8 @@ name: "termux-tts"
|
|
|
8
8
|
display_name: "Termux-TTS"
|
|
9
9
|
package_name_pypi: "termux-tts"
|
|
10
10
|
package_name_npm: "termux-tts"
|
|
11
|
-
version: "v1.
|
|
12
|
-
release_name: "
|
|
11
|
+
version: "v1.4.4"
|
|
12
|
+
release_name: "Multilingual Neural Orchestrator & Zero-Config Ergonomics"
|
|
13
13
|
license: "Apache-2.0"
|
|
14
14
|
platform: "Android ARM64 / Qualcomm Adreno & ARM Mali Vulkan 1.3 / Linux"
|
|
15
15
|
github_repo_url: "https://github.com/uno-km/termux-tts"
|
|
@@ -20,22 +20,42 @@ custom_pages:
|
|
|
20
20
|
title_ko: "Vulkan C++ 네이티브 가속 논문"
|
|
21
21
|
file: "lib/tts/vulkan-engineering-paper.html"
|
|
22
22
|
|
|
23
|
-
tagline_en: "Production-Grade 4-Tier On-Device Speech Synthesis Framework (Zero-Dependency DSP Formant
|
|
24
|
-
tagline_ko: "모바일 및 엣지 환경을 위한 프로덕션급 4-Tier 온디바이스 음성 합성 프레임워크 (무의존성 DSP 포먼트, C++
|
|
23
|
+
tagline_en: "Production-Grade 4-Tier On-Device Speech Synthesis Framework (Multilingual Neural Orchestrator, Zero-Dependency DSP Formant & C++ Acceleration)"
|
|
24
|
+
tagline_ko: "모바일 및 엣지 환경을 위한 프로덕션급 4-Tier 온디바이스 음성 합성 프레임워크 (다국어 신경망 오케스트레이터, 무의존성 DSP 포먼트, C++ 가속 엔진)"
|
|
25
25
|
|
|
26
|
-
why_challenge_en: "Constrained mobile edge environments frequently suffer from execution instability, excessive thermal throttling, and unpredictable runtime memory spikes when running conventional deep learning text-to-speech stacks. Heavyweight dependencies like full PyTorch or unoptimized runtime graph compilers exhaust mobile DRAM, while
|
|
27
|
-
why_challenge_ko: "제약된 모바일 엣지 환경에서는 과도한 메모리 점유, 발열 스로틀링, 드라이버 파편화로 인해 생성형 음성 합성 모델 구동이 불안정합니다. 기존 무거운 프레임워크는 수 기가바이트의 런타임 의존성으로 모바일 OOM을 유발하며,
|
|
26
|
+
why_challenge_en: "Constrained mobile edge environments frequently suffer from execution instability, excessive thermal throttling, and unpredictable runtime memory spikes when running conventional deep learning text-to-speech stacks. Heavyweight dependencies like full PyTorch or unoptimized runtime graph compilers exhaust mobile DRAM, while cross-language code-switching historically required multiple disconnected runtimes or heavy cloud APIs."
|
|
27
|
+
why_challenge_ko: "제약된 모바일 엣지 환경에서는 과도한 메모리 점유, 발열 스로틀링, 드라이버 파편화로 인해 생성형 음성 합성 모델 구동이 불안정합니다. 기존 무거운 프레임워크는 수 기가바이트의 런타임 의존성으로 모바일 OOM을 유발하며, 다국어 교차 발화(한/영/일 등)는 단일 기기 로컬에서 매끄럽게 처리하기 어려웠습니다."
|
|
28
28
|
|
|
29
|
-
description_en: "Termux-TTS delivers a resilient 4-Tier on-device text-to-speech architecture designed for deterministic latency and
|
|
30
|
-
description_ko: "Termux-TTS는 결정론적 지연
|
|
29
|
+
description_en: "Termux-TTS delivers a resilient 4-Tier on-device text-to-speech architecture designed for deterministic latency, zero-config ergonomics, and multi-language acoustic fidelity. It bridges lightweight parametric DSP synthesis (<50ms compute, 0MB model download) with high-fidelity resident C-API multilingual neural models (9 languages including Korean, English, Japanese, Hindi, and Russian at 22.05kHz), alongside Android system native speech service routing. With automated 1-click provisioning, Unicode script classification, and resident in-memory model caching, Termux-TTS achieves high acoustic fidelity with sub-0.18x real-time factor."
|
|
30
|
+
description_ko: "Termux-TTS는 결정론적 지연 시간, Zero-Config 사용성, 다국어 고품질 음향을 실현하는 프로덕션급 4-Tier 온디바이스 음성 합성 프레임워크입니다. 0MB 무의존성 파라메트릭 DSP 보코더부터 9개 공식 언어(한국어, 영어, 일본어, 힌디어, 러시아어 등)를 지원하는 레지던트 C-API 다국어 신경망 오케스트레이터, 그리고 안드로이드 네이티브 음성 브릿지까지 통합 제공합니다."
|
|
31
31
|
|
|
32
32
|
quick_install_cmd: |
|
|
33
33
|
pip install termux-tts
|
|
34
34
|
# or: npm install termux-tts
|
|
35
|
-
# 1-Click Provision
|
|
36
|
-
termux-tts install
|
|
35
|
+
# 1-Click Provision Default Models (Korean KSS + English Lessac):
|
|
36
|
+
termux-tts install
|
|
37
|
+
# On-demand provision specific languages (e.g. Hindi, Japanese, Russian, or all):
|
|
38
|
+
termux-tts install --models hi
|
|
39
|
+
# Instant speech synthesis:
|
|
40
|
+
termux-tts "Hello 방가방가 나는 parrot 이라고 해." -o out.wav --play
|
|
37
41
|
|
|
38
42
|
features:
|
|
43
|
+
- title_en: "Multilingual Neural Orchestrator"
|
|
44
|
+
desc_en: "Dynamic cross-language code-switching and single-language synthesis supporting 9 official languages (Korean, English, Japanese, Chinese, Hindi, Russian, Spanish, French, German) with 50ms context-aware pause padding."
|
|
45
|
+
title_ko: "다국어 신경망 오케스트레이터"
|
|
46
|
+
desc_ko: "한국어, 영어, 일본어, 중국어, 힌디어, 러시아어, 스페인어, 프랑스어, 독일어 9대 언어의 다국어 교차 발화 및 단독 발화를 50ms 문맥 묵음 패딩과 함께 실시간 합성합니다."
|
|
47
|
+
- title_en: "Resident C-API In-Memory Acceleration"
|
|
48
|
+
desc_en: "Native C-API residency with SherpaResidentManager eliminates subprocess startup latency and achieves sub-0.18x real-time factor with ARM NEON SIMD acceleration."
|
|
49
|
+
title_ko: "레지던트 C-API 인메모리 가속"
|
|
50
|
+
desc_ko: "SherpaResidentManager를 통한 C-API 모델 인메모리 상주로 프로세스 기동 지연을 전면 배제하고 ARM NEON SIMD 가속으로 0.18x 이하의 RTF를 달성합니다."
|
|
51
|
+
- title_en: "Zero-Config CLI Ergonomics"
|
|
52
|
+
desc_en: "Top-level speech synthesis by default (termux-tts 'Hello' --play) without requiring subcommands or explicit engine flags. Explicit 'speak' command routes to Android native system voice."
|
|
53
|
+
title_ko: "Zero-Config CLI 사용성"
|
|
54
|
+
desc_ko: "서브 커맨드나 엔진 플래그 없이 문장을 직접 입력(termux-tts '문장' --play)하면 즉시 신경망 합성이 수행되며, 안드로이드 시스템 음성은 'speak'로 명시 분기합니다."
|
|
55
|
+
- title_en: "On-Demand Self-Healing Provisioner"
|
|
56
|
+
desc_en: "1-Click automated provisioning (termux-tts install --models default/hi/ja/ru/zh/all) with actionable English guidance and absolute paths for uninstalled models."
|
|
57
|
+
title_ko: "온디맨드 자가 치유 프로비저너"
|
|
58
|
+
desc_ko: "최초 설치 시 기본 모델(한/영)만 경량 배포하고, 미설치 언어 호출 시 절대 경로와 함께 즉시 복구 가능한 영문 안내문 및 온디맨드 설치 명령을 제공합니다."
|
|
39
59
|
- title_en: "4-Tier Resilient Architecture"
|
|
40
60
|
desc_en: "Tier 1: Zero-Dependency Parametric DSP Formant (0MB footprint). Tier 2: Android Native System Voice Bridge. Tier 3: Subprocess-Isolated Sherpa C++ CPU Engine. Tier 4: Pure Vulkan GPU Hardware Neural Acceleration."
|
|
41
61
|
title_ko: "4-Tier 복원형 합성 아키텍처"
|
|
@@ -78,20 +98,28 @@ matrix_table:
|
|
|
78
98
|
code_example_py: |
|
|
79
99
|
import termux_tts as tts
|
|
80
100
|
|
|
81
|
-
# 1.
|
|
101
|
+
# 1. Zero-Config Multilingual Neural Synthesis (Korean + English Code-Switching)
|
|
102
|
+
with tts.load() as engine:
|
|
103
|
+
result = engine.synthesize(
|
|
104
|
+
"Hello 방가방가 키키키키 나는 parrot 이라고 해. Natural cross-language neural voice.",
|
|
105
|
+
output="multilingual.wav"
|
|
106
|
+
)
|
|
107
|
+
print(f"Synthesized {result.duration_sec:.2f}s in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
|
|
108
|
+
|
|
109
|
+
# 2. Pure Vulkan GPU Neural Synthesis (Studio Tier)
|
|
82
110
|
with tts.load(engine="vulkan", model_tier="high") as engine:
|
|
83
111
|
result = engine.synthesize(
|
|
84
112
|
"The neural speech synthesis engine is operating with pure Vulkan hardware acceleration.",
|
|
85
113
|
output="studio_vulkan.wav"
|
|
86
114
|
)
|
|
87
|
-
print(f"Synthesized {result.
|
|
115
|
+
print(f"Synthesized {result.duration_sec:.2f}s in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
|
|
88
116
|
|
|
89
|
-
#
|
|
117
|
+
# 3. Instant Zero-Dependency DSP Formant Synthesis
|
|
90
118
|
with tts.load(engine="dsp", preset="balanced") as engine:
|
|
91
119
|
result = engine.synthesize("Instant speech generation with zero external weights.", output="dsp.wav")
|
|
92
120
|
print(f"DSP Synthesis Latency: {result.elapsed_ms:.1f}ms")
|
|
93
121
|
|
|
94
|
-
#
|
|
122
|
+
# 4. Direct Android Hardware Speaker Playback
|
|
95
123
|
with tts.load(engine="native", language="en") as engine:
|
|
96
124
|
engine.speak("Direct hardware speaker output via Android native service.")
|
|
97
125
|
|
|
@@ -304,10 +332,22 @@ advanced_parameters_body: |
|
|
|
304
332
|
<p>In-process audio playback via native libraries frequently triggered memory corruption and GIL deadlocks on Android Bionic. Termux-TTS orchestrates playback using isolated subprocess worker pools, safeguarding the primary application runtime.</p>
|
|
305
333
|
|
|
306
334
|
changelog:
|
|
335
|
+
- version: "v1.4.4"
|
|
336
|
+
date: "2026-09-14"
|
|
337
|
+
title: "Multilingual Neural Orchestrator & Zero-Config Ergonomics"
|
|
338
|
+
type: "Production Release (Latest)"
|
|
339
|
+
changes:
|
|
340
|
+
- "Implemented MultilingualNeuralEngine supporting dynamic cross-language code-switching across 9 languages (ko, en, ja, zh, hi, ru, es, fr, de)."
|
|
341
|
+
- "Built SherpaResidentManager for native in-memory C-API residency with sub-0.18x RTF and zero subprocess startup lag."
|
|
342
|
+
- "Promoted speech synthesis to default top-level subcommand, accepting positional text without '-t' or '-e multilingual'."
|
|
343
|
+
- "Integrated universal Unicode script classifier covering Hangul, Latin, Devanagari, Cyrillic, CJK, and Arabic scripts."
|
|
344
|
+
- "Enhanced 1-click installer with lightweight default models (ko+en) and on-demand provisioning (--models <lang> / all)."
|
|
345
|
+
- "Hardened defensive path handling with Path.expanduser() across audio exports and CLI arguments."
|
|
346
|
+
|
|
307
347
|
- version: "v1.3.0"
|
|
308
348
|
date: "2026-09-05"
|
|
309
349
|
title: "Production 4-Tier Synthesizer & Studio Vulkan GPU Acceleration"
|
|
310
|
-
type: "
|
|
350
|
+
type: "Stable Release"
|
|
311
351
|
changes:
|
|
312
352
|
- "Implemented Tier 4 Pure Vulkan GPU Neural Engine (sherpa-ncnn-offline-tts-vulkan) with zero CPU fallback."
|
|
313
353
|
- "Integrated studio-grade reference model (vits-piper-en_US-lessac-high-fp16, 22.05kHz 16-bit PCM)."
|
|
@@ -319,7 +359,7 @@ changelog:
|
|
|
319
359
|
- version: "v1.1.5"
|
|
320
360
|
date: "2026-09-04"
|
|
321
361
|
title: "Subprocess Isolation & Conversational Tag Expansion"
|
|
322
|
-
type: "
|
|
362
|
+
type: "Archive Release"
|
|
323
363
|
changes:
|
|
324
364
|
- "Implemented subprocess IPC audio playback isolation."
|
|
325
365
|
- "Added expressive conversational token handling."
|
|
@@ -333,7 +373,7 @@ readme_content: |
|
|
|
333
373
|
[](https://www.npmjs.com/package/termux-tts)
|
|
334
374
|
[](https://github.com/uno-km/termux-tts)
|
|
335
375
|
|
|
336
|
-
> Production-Grade 4-Tier On-Device Speech Synthesis Framework (Zero-Dependency DSP Formant
|
|
376
|
+
> Production-Grade 4-Tier On-Device Speech Synthesis Framework (Multilingual Neural Orchestrator, Zero-Dependency DSP Formant & C++ Acceleration)
|
|
337
377
|
|
|
338
378
|
---
|
|
339
379
|
|
|
@@ -341,6 +381,8 @@ readme_content: |
|
|
|
341
381
|
|
|
342
382
|
Termux-TTS is an enterprise-grade, on-device text-to-speech framework optimized for mobile edge hardware and Android Termux environments. Built to eliminate heavy dependency stacks and fragile driver behaviors, it features a resilient 4-Tier architecture:
|
|
343
383
|
|
|
384
|
+
- **Multilingual Neural Orchestrator**: Dynamic cross-language code-switching mesh covering 9 official languages (Korean, English, Japanese, Chinese, Hindi, Russian, Spanish, French, German) with Unicode script tokenization and 50ms context-aware silence padding.
|
|
385
|
+
- **Resident C-API In-Memory Engine**: Direct C-API memory residency with `SherpaResidentManager` achieving sub-0.18x real-time factor with ARM NEON SIMD acceleration and zero subprocess lag.
|
|
344
386
|
- **Tier 1: Zero-Dependency Parametric DSP Formant Vocoder**: 0MB disk footprint, Rosenberg glottal pulse formulation, and 5-band biquad formant filters providing deterministic speech synthesis in under 50 milliseconds (RTF 0.013x).
|
|
345
387
|
- **Tier 2: Android System Native Voice Bridge**: Direct IPC integration to physical Samsung and Google speech engines via the Termux-API service layer.
|
|
346
388
|
- **Tier 3: Subprocess-Isolated Sherpa C++ CPU Engine**: Subprocess-isolated VITS acoustic modeling on ARM64 NEON with memory leak protection.
|
|
@@ -353,16 +395,16 @@ readme_content: |
|
|
|
353
395
|
Measurements gathered on physical Android 16 hardware running Termux ARM64:
|
|
354
396
|
|
|
355
397
|
| Target Device | Hardware Architecture | Synthesis Engine | Model Profile | Audio Length | Synthesis Time | Real-Time Factor (RTF) | Status |
|
|
356
|
-
| :--- | :--- | :--- | :--- |
|
|
398
|
+
| :--- | :--- | :--- | :--- | :--- | :---: | :---: | :---: |
|
|
399
|
+
| **Galaxy A53** | Exynos 1280 / ARM64 NEON | Multilingual Neural C-API | `KSS + Lessac` (Bilingual) | 10.50 s | **1.82 s** | **0.173x** | Validated (5.78x RT) |
|
|
357
400
|
| **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | Vulkan GPU Neural | `lessac-high-fp16` | 6.70 s | **6.65 s** | **0.993x** | Validated |
|
|
358
|
-
| **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | Vulkan GPU Neural | `lessac-medium` | 4.59 s | **1.21 s** | **0.264x** | Validated |
|
|
401
|
+
| **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | Vulkan GPU Neural | `lessac-medium` | 4.59 s | **1.21 s** | **0.264x** | Validated (3.79x RT) |
|
|
359
402
|
| **Galaxy A35** | Exynos 1380 / Mali-G68 MP5 | Vulkan GPU Neural | `lessac-medium` | 4.52 s | **5.18 s** | **1.146x** | Validated |
|
|
360
|
-
| **
|
|
361
|
-
| **ARM64 CPU** | Cortex-A78 / A55 | Parametric DSP | 5-Band Biquad | 4.15 s | **0.054 s** | **0.0130x** | Validated |
|
|
403
|
+
| **ARM64 CPU** | Cortex-A78 / A55 | Parametric DSP | 5-Band Biquad | 4.15 s | **0.054 s** | **0.0130x** | Validated (76x RT) |
|
|
362
404
|
|
|
363
405
|
---
|
|
364
406
|
|
|
365
|
-
## Installation &
|
|
407
|
+
## Installation & Automated Provisioning
|
|
366
408
|
|
|
367
409
|
### 1. Package Installation
|
|
368
410
|
```bash
|
|
@@ -373,32 +415,46 @@ readme_content: |
|
|
|
373
415
|
npm install termux-tts
|
|
374
416
|
```
|
|
375
417
|
|
|
376
|
-
### 2. Automated
|
|
377
|
-
Automate
|
|
418
|
+
### 2. Automated Model Provisioning
|
|
419
|
+
Automate downloading and linking precompiled models to the official immutable path (`/data/data/com.termux/files/home/models/tts/`):
|
|
378
420
|
```bash
|
|
379
|
-
#
|
|
380
|
-
termux-tts install
|
|
421
|
+
# Default lightweight installation (Korean KSS + English Lessac)
|
|
422
|
+
termux-tts install
|
|
423
|
+
|
|
424
|
+
# On-demand provision specific language models:
|
|
425
|
+
termux-tts install --models hi # Hindi (Piper Swara)
|
|
426
|
+
termux-tts install --models ja # Japanese (Piper Hina)
|
|
427
|
+
termux-tts install --models ru # Russian (Piper Dmitri)
|
|
428
|
+
termux-tts install --models zh # Chinese (AISHELL3)
|
|
429
|
+
termux-tts install --models all # All 9 official languages
|
|
381
430
|
|
|
382
|
-
#
|
|
383
|
-
termux-tts install --tier
|
|
431
|
+
# Provision Studio Vulkan GPU engine:
|
|
432
|
+
termux-tts install --tier high
|
|
384
433
|
```
|
|
385
434
|
|
|
386
435
|
---
|
|
387
436
|
|
|
388
437
|
## Quickstart
|
|
389
438
|
|
|
390
|
-
### Global Command-Line Interface (
|
|
439
|
+
### Global Command-Line Interface (Zero-Config)
|
|
391
440
|
```bash
|
|
392
|
-
#
|
|
393
|
-
termux-tts
|
|
441
|
+
# 1. Zero-config synthesis with physical speaker playback
|
|
442
|
+
termux-tts "Hello 방가방가 나는 parrot 이라고 해." -o ~/out.wav --play
|
|
443
|
+
|
|
444
|
+
# 2. Flag-based syntax
|
|
445
|
+
termux-tts -t "Multilingual neural speech synthesis on device." --play
|
|
394
446
|
|
|
395
|
-
#
|
|
447
|
+
# 3. Force language pinning
|
|
448
|
+
termux-tts -l en -t "Pure English text output." --play
|
|
449
|
+
termux-tts -l ko -t "한국어 단독 신경망 음성 합성." --play
|
|
450
|
+
|
|
451
|
+
# 4. Instant DSP Formant synthesis (0MB footprint)
|
|
396
452
|
termux-tts synth -e dsp -t "Zero dependency DSP synthesis." -o dsp.wav
|
|
397
453
|
|
|
398
|
-
# Direct
|
|
399
|
-
termux-tts speak -t "Hardware speaker broadcast."
|
|
454
|
+
# 5. Direct Android system native speaker broadcast
|
|
455
|
+
termux-tts speak -t "Hardware speaker broadcast via Android service."
|
|
400
456
|
|
|
401
|
-
# Hardware diagnostics
|
|
457
|
+
# 6. Hardware diagnostics
|
|
402
458
|
termux-tts doctor
|
|
403
459
|
```
|
|
404
460
|
|
|
@@ -406,12 +462,20 @@ readme_content: |
|
|
|
406
462
|
```python
|
|
407
463
|
import termux_tts as tts
|
|
408
464
|
|
|
409
|
-
#
|
|
465
|
+
# 1. Zero-Config Multilingual Neural Synthesis (Korean + English Code-Switching)
|
|
466
|
+
with tts.load() as engine:
|
|
467
|
+
result = engine.synthesize(
|
|
468
|
+
"Hello 방가방가 키키키키 나는 parrot 이라고 해.",
|
|
469
|
+
output="multilingual.wav"
|
|
470
|
+
)
|
|
471
|
+
print(f"Generated {result.duration_sec:.2f}s in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
|
|
472
|
+
|
|
473
|
+
# 2. Pure Vulkan GPU Neural Synthesis (Studio Tier)
|
|
410
474
|
with tts.load(engine="vulkan", model_tier="high") as engine:
|
|
411
475
|
result = engine.synthesize("Pure Vulkan neural execution on mobile.", output="speech.wav")
|
|
412
476
|
print(f"Synthesized in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
|
|
413
477
|
|
|
414
|
-
# Zero-Dependency DSP Formant Synthesis
|
|
478
|
+
# 3. Zero-Dependency DSP Formant Synthesis
|
|
415
479
|
with tts.load(engine="dsp", preset="balanced") as engine:
|
|
416
480
|
result = engine.synthesize("Instant speech without model downloads.", output="dsp.wav")
|
|
417
481
|
print(f"DSP Latency: {result.elapsed_ms:.1f}ms")
|
package/package.json
CHANGED
package/pyproject.toml
CHANGED
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "termux-tts"
|
|
7
|
-
version = "1.4.
|
|
7
|
+
version = "1.4.4"
|
|
8
8
|
description = "On-device 4-Tier Text-to-Speech framework utilizing device resources (DSP Formant Vocoder, C++ Sherpa-ONNX Neural, Android Native & Expressive)"
|
|
9
9
|
readme = "README.pypi.md"
|
|
10
10
|
requires-python = ">=3.10"
|
package/termux_tts/__init__.py
CHANGED
|
@@ -12,6 +12,9 @@ from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS, DSPSynt
|
|
|
12
12
|
from .engine_sherpa import SherpaNeuralEngine, SherpaResult
|
|
13
13
|
from .engine_vulkan import VulkanNeuralEngine, VulkanResult
|
|
14
14
|
from .engine_expressive import ExpressiveEngine, ExpressiveResult
|
|
15
|
+
from .engine_multilingual import MultilingualNeuralEngine, MultilingualResult
|
|
16
|
+
from .engine_sherpa_capi import SherpaResidentManager, SherpaCapiSession
|
|
17
|
+
from .script_classifier import MultilingualTokenizer, ScriptRegistry, LanguageChunk
|
|
15
18
|
from .installer import run_installation
|
|
16
19
|
from .tokenizer import PhoneticTokenizer, EXPRESSIVE_TAGS
|
|
17
20
|
from .g2p_korean import KoreanG2PEngine, korean_text_to_phonemes
|
|
@@ -29,7 +32,7 @@ from .exceptions import (
|
|
|
29
32
|
ONNXNeuralEngine = SherpaNeuralEngine
|
|
30
33
|
ONNXResult = SherpaResult
|
|
31
34
|
|
|
32
|
-
__version__ = "1.4.
|
|
35
|
+
__version__ = "1.4.4"
|
|
33
36
|
__all__ = [
|
|
34
37
|
"TTSEngine",
|
|
35
38
|
"load",
|
|
@@ -45,6 +48,13 @@ __all__ = [
|
|
|
45
48
|
"VulkanResult",
|
|
46
49
|
"ExpressiveEngine",
|
|
47
50
|
"ExpressiveResult",
|
|
51
|
+
"MultilingualNeuralEngine",
|
|
52
|
+
"MultilingualResult",
|
|
53
|
+
"SherpaResidentManager",
|
|
54
|
+
"SherpaCapiSession",
|
|
55
|
+
"MultilingualTokenizer",
|
|
56
|
+
"ScriptRegistry",
|
|
57
|
+
"LanguageChunk",
|
|
48
58
|
"run_installation",
|
|
49
59
|
"ONNXNeuralEngine",
|
|
50
60
|
"ONNXResult",
|
package/termux_tts/audio.py
CHANGED
|
@@ -5,8 +5,9 @@ Handles 16-bit Linear PCM formatting with soft-clipping protection.
|
|
|
5
5
|
|
|
6
6
|
import io
|
|
7
7
|
import wave
|
|
8
|
-
|
|
8
|
+
from pathlib import Path
|
|
9
9
|
from typing import Union
|
|
10
|
+
import numpy as np
|
|
10
11
|
from .exceptions import TTSAudioEncodingError
|
|
11
12
|
|
|
12
13
|
class AudioBuffer:
|
|
@@ -89,10 +90,12 @@ class AudioBuffer:
|
|
|
89
90
|
|
|
90
91
|
def save(self, filepath: str) -> str:
|
|
91
92
|
"""Save the audio buffer to a WAV file on disk."""
|
|
93
|
+
target = Path(filepath).expanduser().resolve()
|
|
92
94
|
wav_data = self.to_wav_bytes()
|
|
93
95
|
try:
|
|
94
|
-
|
|
96
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
97
|
+
with open(target, "wb") as f:
|
|
95
98
|
f.write(wav_data)
|
|
96
|
-
return
|
|
99
|
+
return str(target)
|
|
97
100
|
except Exception as e:
|
|
98
101
|
raise TTSAudioEncodingError(f"Failed to save WAV to '{filepath}': {e}") from e
|
package/termux_tts/cli.py
CHANGED
|
@@ -11,6 +11,12 @@ import argparse
|
|
|
11
11
|
from .engine import load, doctor
|
|
12
12
|
|
|
13
13
|
def main():
|
|
14
|
+
# Route to default subcommand 'synth' if no recognized subcommand is given
|
|
15
|
+
KNOWN_COMMANDS = {"synth", "speak", "doctor", "install", "component", "model", "instance"}
|
|
16
|
+
raw_args = sys.argv[1:]
|
|
17
|
+
if raw_args and raw_args[0] not in KNOWN_COMMANDS and raw_args[0] not in ("-h", "--help", "-v", "--version"):
|
|
18
|
+
sys.argv.insert(1, "synth")
|
|
19
|
+
|
|
14
20
|
parser = argparse.ArgumentParser(
|
|
15
21
|
prog="termux-tts",
|
|
16
22
|
description="Termux Neural & Native Text-to-Speech Engine"
|
|
@@ -19,13 +25,14 @@ def main():
|
|
|
19
25
|
|
|
20
26
|
# 1. Synth (4-Tier Speech Synthesis)
|
|
21
27
|
synth_parser = subparsers.add_parser("synth", help="Synthesize text to audio WAV file (Synth, Neural, Expressive)")
|
|
22
|
-
synth_parser.add_argument("
|
|
28
|
+
synth_parser.add_argument("text_pos", nargs="*", default=[], help="Input text to synthesize (positional)")
|
|
29
|
+
synth_parser.add_argument("-t", "--text", default=None, help="Input text to synthesize (flag)")
|
|
23
30
|
synth_parser.add_argument("-o", "--output", default="output.wav", help="Output WAV filepath")
|
|
24
|
-
synth_parser.add_argument("-l", "--lang", default="
|
|
31
|
+
synth_parser.add_argument("-l", "--lang", default="auto", help="Language code (auto=Multi-language auto switch, ko/kor=Korean only, en/eng=English only, ja/jpn=Japanese only)")
|
|
25
32
|
synth_parser.add_argument(
|
|
26
33
|
"-e", "--engine", default="auto",
|
|
27
|
-
choices=["auto", "vulkan", "ncnn", "gpu", "synth", "dsp", "native", "neural", "onnx", "expressive"],
|
|
28
|
-
help="Synthesis engine tier (vulkan=GPU NCNN, synth=0MB DSP, native=Android voice, neural=VITS C++, expressive=emotional)"
|
|
34
|
+
choices=["auto", "vulkan", "ncnn", "gpu", "synth", "dsp", "native", "neural", "onnx", "expressive", "multilingual", "hybrid"],
|
|
35
|
+
help="Synthesis engine tier (auto=Smart Routing, vulkan=GPU NCNN, synth=0MB DSP, native=Android voice, neural=VITS C++, expressive=emotional, multilingual=Cross-language)"
|
|
29
36
|
)
|
|
30
37
|
synth_parser.add_argument("-m", "--model", default=None, help="Path to model file or directory")
|
|
31
38
|
synth_parser.add_argument("-p", "--preset", default="balanced", choices=["fast", "balanced", "expressive", "ultra"])
|
|
@@ -52,6 +59,10 @@ def main():
|
|
|
52
59
|
# 4. Install (One-Click Automated Provisioner)
|
|
53
60
|
install_parser = subparsers.add_parser("install", help="1-Click download and provision precompiled Vulkan binary & VITS studio models")
|
|
54
61
|
install_parser.add_argument("--tier", default="high", choices=["high", "medium"], help="Model resolution tier (high=57MB Studio FP16, medium=25MB Fast)")
|
|
62
|
+
install_parser.add_argument(
|
|
63
|
+
"--models", default="default",
|
|
64
|
+
help="Language model packages to provision: default (Korean & English only), or specific code: hi, ja, zh, ru, es, fr, de, or 'all'"
|
|
65
|
+
)
|
|
55
66
|
install_parser.add_argument("--force", action="store_true", help="Force overwrite existing binary and model assets")
|
|
56
67
|
install_parser.add_argument("--no-play", action="store_true", help="Skip playback verification during self-test")
|
|
57
68
|
|
|
@@ -75,6 +86,12 @@ def main():
|
|
|
75
86
|
subprocess.run(["termux-volume", "music", str(args.volume)], check=False)
|
|
76
87
|
|
|
77
88
|
if args.command == "synth":
|
|
89
|
+
pos_text = " ".join(args.text_pos).strip() if getattr(args, "text_pos", None) else None
|
|
90
|
+
target_text = args.text or (pos_text if pos_text else None)
|
|
91
|
+
if not target_text:
|
|
92
|
+
synth_parser.error("the following arguments are required: text (as positional arguments or -t/--text)")
|
|
93
|
+
|
|
94
|
+
out_path = os.path.expanduser(args.output) if args.output else None
|
|
78
95
|
with load(
|
|
79
96
|
model=args.model,
|
|
80
97
|
language=args.lang,
|
|
@@ -84,20 +101,20 @@ def main():
|
|
|
84
101
|
engine=args.engine,
|
|
85
102
|
tier=getattr(args, "tier", None),
|
|
86
103
|
) as engine:
|
|
87
|
-
res = engine.synthesize(
|
|
104
|
+
res = engine.synthesize(target_text, output=out_path, speed=args.speed, language=args.lang)
|
|
88
105
|
backend_name = getattr(res, "backend", "UNKNOWN")
|
|
89
106
|
model_name = getattr(res, "model_name", "model")
|
|
90
107
|
dur = getattr(res, "duration_sec", 0.0)
|
|
91
108
|
elapsed = getattr(res, "elapsed_ms", 0.0)
|
|
92
109
|
rtf = getattr(res, "rtf", 0.0)
|
|
93
|
-
print(f"[SUCCESS] Synthesized via {backend_name} ({model_name}) -> {args.output}")
|
|
110
|
+
print(f"[SUCCESS] Synthesized via {backend_name} ({model_name}) -> {out_path or args.output}")
|
|
94
111
|
print(f" Duration: {dur:.2f}s | Elapsed: {elapsed:.1f}ms | RTF: {rtf:.4f}x")
|
|
95
112
|
|
|
96
|
-
if args.play and
|
|
113
|
+
if args.play and out_path and os.path.exists(out_path):
|
|
97
114
|
if shutil.which("termux-media-player"):
|
|
98
|
-
subprocess.run(["termux-media-player", "play",
|
|
115
|
+
subprocess.run(["termux-media-player", "play", out_path], check=False)
|
|
99
116
|
elif shutil.which("play-audio"):
|
|
100
|
-
subprocess.run(["play-audio",
|
|
117
|
+
subprocess.run(["play-audio", out_path], check=False)
|
|
101
118
|
|
|
102
119
|
elif args.command == "speak":
|
|
103
120
|
with load(language=args.lang) as engine:
|
|
@@ -115,7 +132,7 @@ def main():
|
|
|
115
132
|
|
|
116
133
|
elif args.command == "install":
|
|
117
134
|
from .installer import run_installation
|
|
118
|
-
run_installation(tier=args.tier, force=args.force, play=not args.no_play)
|
|
135
|
+
run_installation(tier=args.tier, models=getattr(args, "models", "all"), force=args.force, play=not args.no_play)
|
|
119
136
|
|
|
120
137
|
elif args.command in ("component", "model", "instance") and _protocol_available:
|
|
121
138
|
from ameva_component.cli_support import dispatch_protocol
|
package/termux_tts/engine.py
CHANGED
|
@@ -22,6 +22,8 @@ from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS
|
|
|
22
22
|
from .engine_sherpa import SherpaNeuralEngine, SherpaResult
|
|
23
23
|
from .engine_vulkan import VulkanNeuralEngine, VulkanResult
|
|
24
24
|
from .engine_expressive import ExpressiveEngine, ExpressiveResult
|
|
25
|
+
from .engine_multilingual import MultilingualNeuralEngine, MultilingualResult
|
|
26
|
+
from .script_classifier import MultilingualTokenizer
|
|
25
27
|
from .hardware import (
|
|
26
28
|
resolve_device_backend,
|
|
27
29
|
bind_tts_hardware,
|
|
@@ -69,6 +71,7 @@ class TTSEngine:
|
|
|
69
71
|
self._binding_plan = self._bind_hardware()
|
|
70
72
|
|
|
71
73
|
self.native_engine = NativeAndroidEngine(language=language)
|
|
74
|
+
self._multilingual_engine: Optional[MultilingualNeuralEngine] = None
|
|
72
75
|
self.synth_engine = self._resolve_synth_engine()
|
|
73
76
|
|
|
74
77
|
def _bind_hardware(self):
|
|
@@ -76,9 +79,15 @@ class TTSEngine:
|
|
|
76
79
|
|
|
77
80
|
def _resolve_synth_engine(self):
|
|
78
81
|
t = self.requested_engine_type
|
|
82
|
+
from .script_classifier import normalize_language_code
|
|
83
|
+
norm_lang = normalize_language_code(self.language)
|
|
79
84
|
|
|
80
|
-
#
|
|
81
|
-
if
|
|
85
|
+
# Extended Languages (hi, ru, ja, zh, es, fr, de, ar) route to MultilingualNeuralEngine
|
|
86
|
+
if norm_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar"):
|
|
87
|
+
return self._get_multilingual_engine()
|
|
88
|
+
|
|
89
|
+
# Explicit Vulkan GPU Tier (Vulkan NCNN engine targets English Lessac)
|
|
90
|
+
if (t in ("vulkan", "gpu", "ncnn") or (self.requested_device in ("vulkan", "gpu") and t in ("neural", "vits", "auto"))) and norm_lang in ("en", "auto"):
|
|
82
91
|
try:
|
|
83
92
|
return VulkanNeuralEngine(
|
|
84
93
|
model_path=self.model_path,
|
|
@@ -124,6 +133,10 @@ class TTSEngine:
|
|
|
124
133
|
sample_rate=self.sample_rate,
|
|
125
134
|
)
|
|
126
135
|
|
|
136
|
+
# Explicit Multilingual / Code-Switching Tier
|
|
137
|
+
elif t in ("multilingual", "codeswitch", "hybrid"):
|
|
138
|
+
return self._get_multilingual_engine()
|
|
139
|
+
|
|
127
140
|
# Explicit Tier 2: Native
|
|
128
141
|
elif t == "native":
|
|
129
142
|
return self.native_engine
|
|
@@ -172,6 +185,15 @@ class TTSEngine:
|
|
|
172
185
|
def binary(self) -> Optional[str]:
|
|
173
186
|
return getattr(self.synth_engine, "binary", getattr(self.native_engine, "binary", None))
|
|
174
187
|
|
|
188
|
+
def _get_multilingual_engine(self) -> MultilingualNeuralEngine:
|
|
189
|
+
if self._multilingual_engine is None:
|
|
190
|
+
self._multilingual_engine = MultilingualNeuralEngine(
|
|
191
|
+
threads=self.threads,
|
|
192
|
+
device=self.device,
|
|
193
|
+
sample_rate=self.sample_rate or 22050,
|
|
194
|
+
)
|
|
195
|
+
return self._multilingual_engine
|
|
196
|
+
|
|
175
197
|
def speak(self, text: str, stream: Optional[str] = None) -> NativeResult:
|
|
176
198
|
"""Speak text directly through physical Android speaker (Native Engine)."""
|
|
177
199
|
if self._is_closed:
|
|
@@ -184,14 +206,56 @@ class TTSEngine:
|
|
|
184
206
|
output: Optional[str] = None,
|
|
185
207
|
speed: float = 1.0,
|
|
186
208
|
preset: Optional[str] = None,
|
|
187
|
-
|
|
188
|
-
|
|
209
|
+
language: Optional[str] = None,
|
|
210
|
+
) -> Union[DSPResult, SherpaResult, ExpressiveResult, NativeResult, MultilingualResult]:
|
|
211
|
+
"""Synthesize text into speech audio buffer / WAV file with Zero-Config intelligent routing."""
|
|
189
212
|
if self._is_closed:
|
|
190
213
|
raise TTSInferenceError("Cannot synthesize: Engine session is closed.")
|
|
214
|
+
|
|
215
|
+
clean_text = text.strip() if text else ""
|
|
216
|
+
if not clean_text:
|
|
217
|
+
raise TTSInferenceError("Cannot synthesize empty text.")
|
|
218
|
+
|
|
219
|
+
from .script_classifier import normalize_language_code
|
|
220
|
+
target_lang = normalize_language_code(language or self.language)
|
|
221
|
+
|
|
222
|
+
# 1. If user explicitly pinned to lightweight DSP (0MB), respect choice
|
|
223
|
+
if self.requested_engine_type in ("dsp", "synth", "formant") or self.device == "dsp":
|
|
224
|
+
return self.synth_engine.synthesize(clean_text, output=output, speed=speed, preset=preset)
|
|
225
|
+
|
|
226
|
+
# 2. If user explicitly pinned to OS Native voice, speak directly
|
|
227
|
+
if self.requested_engine_type == "native":
|
|
228
|
+
return self.native_engine.speak(clean_text)
|
|
229
|
+
|
|
230
|
+
# 3. Multilingual auto-detection:
|
|
231
|
+
# Route to MultilingualNeuralEngine if multiple languages are present in text
|
|
232
|
+
# or if caller specifically requested multilingual / hybrid engine.
|
|
233
|
+
detected_langs = MultilingualTokenizer.detect_languages(clean_text)
|
|
234
|
+
is_mixed_text = len(detected_langs) > 1
|
|
235
|
+
|
|
236
|
+
should_route_multilingual = (
|
|
237
|
+
(self.requested_engine_type in ("auto", "multilingual", "codeswitch", "hybrid")) or
|
|
238
|
+
(is_mixed_text and self.requested_engine_type not in ("dsp", "synth", "native") and self.device != "dsp" and (self._multilingual_engine is not None or not isinstance(self.synth_engine, ParametricDSPEngine))) or
|
|
239
|
+
(target_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar") and self.requested_engine_type not in ("dsp", "synth", "native"))
|
|
240
|
+
)
|
|
241
|
+
|
|
242
|
+
if should_route_multilingual:
|
|
243
|
+
multi_engine = self._get_multilingual_engine()
|
|
244
|
+
kwargs = {}
|
|
245
|
+
if target_lang != "auto":
|
|
246
|
+
kwargs["language"] = target_lang
|
|
247
|
+
return multi_engine.synthesize(
|
|
248
|
+
clean_text,
|
|
249
|
+
output=output,
|
|
250
|
+
speed=speed,
|
|
251
|
+
**kwargs
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
# 4. Standard single-language engine
|
|
191
255
|
if hasattr(self.synth_engine, "synthesize"):
|
|
192
|
-
return self.synth_engine.synthesize(
|
|
256
|
+
return self.synth_engine.synthesize(clean_text, output=output, speed=speed, preset=preset)
|
|
193
257
|
elif hasattr(self.synth_engine, "speak"):
|
|
194
|
-
return self.synth_engine.speak(
|
|
258
|
+
return self.synth_engine.speak(clean_text)
|
|
195
259
|
raise TTSInferenceError(f"Selected engine '{type(self.synth_engine).__name__}' does not support synthesize.")
|
|
196
260
|
|
|
197
261
|
def close(self) -> None:
|
|
@@ -199,6 +263,8 @@ class TTSEngine:
|
|
|
199
263
|
self.native_engine.close()
|
|
200
264
|
if hasattr(self.synth_engine, "close"):
|
|
201
265
|
self.synth_engine.close()
|
|
266
|
+
if self._multilingual_engine is not None:
|
|
267
|
+
self._multilingual_engine.close()
|
|
202
268
|
|
|
203
269
|
def __enter__(self):
|
|
204
270
|
return self
|
|
@@ -209,7 +275,7 @@ class TTSEngine:
|
|
|
209
275
|
|
|
210
276
|
def load(
|
|
211
277
|
model: Optional[str] = None,
|
|
212
|
-
language: str = "
|
|
278
|
+
language: str = "auto",
|
|
213
279
|
preset: str = "balanced",
|
|
214
280
|
device: str = "auto",
|
|
215
281
|
threads: int = 4,
|