termux-tts 1.4.2 → 1.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/doc.config.yaml CHANGED
@@ -8,8 +8,8 @@ name: "termux-tts"
8
8
  display_name: "Termux-TTS"
9
9
  package_name_pypi: "termux-tts"
10
10
  package_name_npm: "termux-tts"
11
- version: "v1.3.0"
12
- release_name: "4-Tier Architecture & Studio Vulkan GPU Acceleration"
11
+ version: "v1.4.4"
12
+ release_name: "Multilingual Neural Orchestrator & Zero-Config Ergonomics"
13
13
  license: "Apache-2.0"
14
14
  platform: "Android ARM64 / Qualcomm Adreno & ARM Mali Vulkan 1.3 / Linux"
15
15
  github_repo_url: "https://github.com/uno-km/termux-tts"
@@ -20,22 +20,42 @@ custom_pages:
20
20
  title_ko: "Vulkan C++ 네이티브 가속 논문"
21
21
  file: "lib/tts/vulkan-engineering-paper.html"
22
22
 
23
- tagline_en: "Production-Grade 4-Tier On-Device Speech Synthesis Framework (Zero-Dependency DSP Formant, C++ Vulkan GPU Neural Engine & Android Native Voice Bridge)"
24
- tagline_ko: "모바일 및 엣지 환경을 위한 프로덕션급 4-Tier 온디바이스 음성 합성 프레임워크 (무의존성 DSP 포먼트, C++ Vulkan GPU 신경망 엔진, 안드로이드 네이티브 브릿지)"
23
+ tagline_en: "Production-Grade 4-Tier On-Device Speech Synthesis Framework (Multilingual Neural Orchestrator, Zero-Dependency DSP Formant & C++ Acceleration)"
24
+ tagline_ko: "모바일 및 엣지 환경을 위한 프로덕션급 4-Tier 온디바이스 음성 합성 프레임워크 (다국어 신경망 오케스트레이터, 무의존성 DSP 포먼트, C++ 가속 엔진)"
25
25
 
26
- why_challenge_en: "Constrained mobile edge environments frequently suffer from execution instability, excessive thermal throttling, and unpredictable runtime memory spikes when running conventional deep learning text-to-speech stacks. Heavyweight dependencies like full PyTorch or unoptimized runtime graph compilers exhaust mobile DRAM, while vendor-specific Vulkan driver quirks (such as ARM Mali subgroup index truncation or Qualcomm Adreno JIT pipeline compilation crashes) historically prevented reliable on-device GPU speech synthesis."
27
- why_challenge_ko: "제약된 모바일 엣지 환경에서는 과도한 메모리 점유, 발열 스로틀링, 드라이버 파편화로 인해 생성형 음성 합성 모델 구동이 불안정합니다. 기존 무거운 프레임워크는 수 기가바이트의 런타임 의존성으로 모바일 OOM을 유발하며, ARM Mali 및 Qualcomm Adreno의 셰이더 컴파일러 결함으로 인해 안정적인 GPU 음성 가속이 어려웠습니다."
26
+ why_challenge_en: "Constrained mobile edge environments frequently suffer from execution instability, excessive thermal throttling, and unpredictable runtime memory spikes when running conventional deep learning text-to-speech stacks. Heavyweight dependencies like full PyTorch or unoptimized runtime graph compilers exhaust mobile DRAM, while cross-language code-switching historically required multiple disconnected runtimes or heavy cloud APIs."
27
+ why_challenge_ko: "제약된 모바일 엣지 환경에서는 과도한 메모리 점유, 발열 스로틀링, 드라이버 파편화로 인해 생성형 음성 합성 모델 구동이 불안정합니다. 기존 무거운 프레임워크는 수 기가바이트의 런타임 의존성으로 모바일 OOM을 유발하며, 다국어 교차 발화(한/영/일 등)는 단일 기기 로컬에서 매끄럽게 처리하기 어려웠습니다."
28
28
 
29
- description_en: "Termux-TTS delivers a resilient 4-Tier on-device text-to-speech architecture designed for deterministic latency and hardware acceleration. It bridges lightweight parametric DSP synthesis (<50ms compute, 0MB model download) with high-fidelity C++ Vulkan neural acceleration (lessac-high-fp16 at 22.05kHz), alongside Android system native speech service routing. With automated 1-click provisioning and subprocess IPC isolation, Termux-TTS achieves high acoustic fidelity while mitigating audio thread jitter and driver locks."
30
- description_ko: "Termux-TTS는 결정론적 지연 시간과 하드웨어 가속을 실현하는 프로덕션급 4-Tier 온디바이스 음성 합성 프레임워크입니다. 0MB 무의존성 파라메트릭 DSP 보코더부터 고해상도 C++ Vulkan GPU 가속 신경망 모델(22.05kHz FP16), 그리고 안드로이드 네이티브 음성 브릿지까지 통합 제공합니다."
29
+ description_en: "Termux-TTS delivers a resilient 4-Tier on-device text-to-speech architecture designed for deterministic latency, zero-config ergonomics, and multi-language acoustic fidelity. It bridges lightweight parametric DSP synthesis (<50ms compute, 0MB model download) with high-fidelity resident C-API multilingual neural models (9 languages including Korean, English, Japanese, Hindi, and Russian at 22.05kHz), alongside Android system native speech service routing. With automated 1-click provisioning, Unicode script classification, and resident in-memory model caching, Termux-TTS achieves high acoustic fidelity with sub-0.18x real-time factor."
30
+ description_ko: "Termux-TTS는 결정론적 지연 시간, Zero-Config 사용성, 다국어 고품질 음향을 실현하는 프로덕션급 4-Tier 온디바이스 음성 합성 프레임워크입니다. 0MB 무의존성 파라메트릭 DSP 보코더부터 9개 공식 언어(한국어, 영어, 일본어, 힌디어, 러시아어 등)를 지원하는 레지던트 C-API 다국어 신경망 오케스트레이터, 그리고 안드로이드 네이티브 음성 브릿지까지 통합 제공합니다."
31
31
 
32
32
  quick_install_cmd: |
33
33
  pip install termux-tts
34
34
  # or: npm install termux-tts
35
- # 1-Click Provision Studio Vulkan Engine:
36
- termux-tts install --tier high
35
+ # 1-Click Provision Default Models (Korean KSS + English Lessac):
36
+ termux-tts install
37
+ # On-demand provision specific languages (e.g. Hindi, Japanese, Russian, or all):
38
+ termux-tts install --models hi
39
+ # Instant speech synthesis:
40
+ termux-tts "Hello 방가방가 나는 parrot 이라고 해." -o out.wav --play
37
41
 
38
42
  features:
43
+ - title_en: "Multilingual Neural Orchestrator"
44
+ desc_en: "Dynamic cross-language code-switching and single-language synthesis supporting 9 official languages (Korean, English, Japanese, Chinese, Hindi, Russian, Spanish, French, German) with 50ms context-aware pause padding."
45
+ title_ko: "다국어 신경망 오케스트레이터"
46
+ desc_ko: "한국어, 영어, 일본어, 중국어, 힌디어, 러시아어, 스페인어, 프랑스어, 독일어 9대 언어의 다국어 교차 발화 및 단독 발화를 50ms 문맥 묵음 패딩과 함께 실시간 합성합니다."
47
+ - title_en: "Resident C-API In-Memory Acceleration"
48
+ desc_en: "Native C-API residency with SherpaResidentManager eliminates subprocess startup latency and achieves sub-0.18x real-time factor with ARM NEON SIMD acceleration."
49
+ title_ko: "레지던트 C-API 인메모리 가속"
50
+ desc_ko: "SherpaResidentManager를 통한 C-API 모델 인메모리 상주로 프로세스 기동 지연을 전면 배제하고 ARM NEON SIMD 가속으로 0.18x 이하의 RTF를 달성합니다."
51
+ - title_en: "Zero-Config CLI Ergonomics"
52
+ desc_en: "Top-level speech synthesis by default (termux-tts 'Hello' --play) without requiring subcommands or explicit engine flags. Explicit 'speak' command routes to Android native system voice."
53
+ title_ko: "Zero-Config CLI 사용성"
54
+ desc_ko: "서브 커맨드나 엔진 플래그 없이 문장을 직접 입력(termux-tts '문장' --play)하면 즉시 신경망 합성이 수행되며, 안드로이드 시스템 음성은 'speak'로 명시 분기합니다."
55
+ - title_en: "On-Demand Self-Healing Provisioner"
56
+ desc_en: "1-Click automated provisioning (termux-tts install --models default/hi/ja/ru/zh/all) with actionable English guidance and absolute paths for uninstalled models."
57
+ title_ko: "온디맨드 자가 치유 프로비저너"
58
+ desc_ko: "최초 설치 시 기본 모델(한/영)만 경량 배포하고, 미설치 언어 호출 시 절대 경로와 함께 즉시 복구 가능한 영문 안내문 및 온디맨드 설치 명령을 제공합니다."
39
59
  - title_en: "4-Tier Resilient Architecture"
40
60
  desc_en: "Tier 1: Zero-Dependency Parametric DSP Formant (0MB footprint). Tier 2: Android Native System Voice Bridge. Tier 3: Subprocess-Isolated Sherpa C++ CPU Engine. Tier 4: Pure Vulkan GPU Hardware Neural Acceleration."
41
61
  title_ko: "4-Tier 복원형 합성 아키텍처"
@@ -78,20 +98,28 @@ matrix_table:
78
98
  code_example_py: |
79
99
  import termux_tts as tts
80
100
 
81
- # 1. Pure Vulkan GPU Neural Synthesis (Studio Tier)
101
+ # 1. Zero-Config Multilingual Neural Synthesis (Korean + English Code-Switching)
102
+ with tts.load() as engine:
103
+ result = engine.synthesize(
104
+ "Hello 방가방가 키키키키 나는 parrot 이라고 해. Natural cross-language neural voice.",
105
+ output="multilingual.wav"
106
+ )
107
+ print(f"Synthesized {result.duration_sec:.2f}s in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
108
+
109
+ # 2. Pure Vulkan GPU Neural Synthesis (Studio Tier)
82
110
  with tts.load(engine="vulkan", model_tier="high") as engine:
83
111
  result = engine.synthesize(
84
112
  "The neural speech synthesis engine is operating with pure Vulkan hardware acceleration.",
85
113
  output="studio_vulkan.wav"
86
114
  )
87
- print(f"Synthesized {result.audio_duration:.2f}s in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
115
+ print(f"Synthesized {result.duration_sec:.2f}s in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
88
116
 
89
- # 2. Instant Zero-Dependency DSP Formant Synthesis
117
+ # 3. Instant Zero-Dependency DSP Formant Synthesis
90
118
  with tts.load(engine="dsp", preset="balanced") as engine:
91
119
  result = engine.synthesize("Instant speech generation with zero external weights.", output="dsp.wav")
92
120
  print(f"DSP Synthesis Latency: {result.elapsed_ms:.1f}ms")
93
121
 
94
- # 3. Direct Android Hardware Speaker Playback
122
+ # 4. Direct Android Hardware Speaker Playback
95
123
  with tts.load(engine="native", language="en") as engine:
96
124
  engine.speak("Direct hardware speaker output via Android native service.")
97
125
 
@@ -304,10 +332,22 @@ advanced_parameters_body: |
304
332
  <p>In-process audio playback via native libraries frequently triggered memory corruption and GIL deadlocks on Android Bionic. Termux-TTS orchestrates playback using isolated subprocess worker pools, safeguarding the primary application runtime.</p>
305
333
 
306
334
  changelog:
335
+ - version: "v1.4.4"
336
+ date: "2026-09-14"
337
+ title: "Multilingual Neural Orchestrator & Zero-Config Ergonomics"
338
+ type: "Production Release (Latest)"
339
+ changes:
340
+ - "Implemented MultilingualNeuralEngine supporting dynamic cross-language code-switching across 9 languages (ko, en, ja, zh, hi, ru, es, fr, de)."
341
+ - "Built SherpaResidentManager for native in-memory C-API residency with sub-0.18x RTF and zero subprocess startup lag."
342
+ - "Promoted speech synthesis to default top-level subcommand, accepting positional text without '-t' or '-e multilingual'."
343
+ - "Integrated universal Unicode script classifier covering Hangul, Latin, Devanagari, Cyrillic, CJK, and Arabic scripts."
344
+ - "Enhanced 1-click installer with lightweight default models (ko+en) and on-demand provisioning (--models <lang> / all)."
345
+ - "Hardened defensive path handling with Path.expanduser() across audio exports and CLI arguments."
346
+
307
347
  - version: "v1.3.0"
308
348
  date: "2026-09-05"
309
349
  title: "Production 4-Tier Synthesizer & Studio Vulkan GPU Acceleration"
310
- type: "Production Release (Latest)"
350
+ type: "Stable Release"
311
351
  changes:
312
352
  - "Implemented Tier 4 Pure Vulkan GPU Neural Engine (sherpa-ncnn-offline-tts-vulkan) with zero CPU fallback."
313
353
  - "Integrated studio-grade reference model (vits-piper-en_US-lessac-high-fp16, 22.05kHz 16-bit PCM)."
@@ -319,7 +359,7 @@ changelog:
319
359
  - version: "v1.1.5"
320
360
  date: "2026-09-04"
321
361
  title: "Subprocess Isolation & Conversational Tag Expansion"
322
- type: "Stable Release"
362
+ type: "Archive Release"
323
363
  changes:
324
364
  - "Implemented subprocess IPC audio playback isolation."
325
365
  - "Added expressive conversational token handling."
@@ -333,7 +373,7 @@ readme_content: |
333
373
  [![npm](https://img.shields.io/npm/v/termux-tts.svg?style=flat-square&color=b91c1c)](https://www.npmjs.com/package/termux-tts)
334
374
  [![License](https://img.shields.io/badge/License-Apache_2.0-004499.svg?style=flat-square)](https://github.com/uno-km/termux-tts)
335
375
 
336
- > Production-Grade 4-Tier On-Device Speech Synthesis Framework (Zero-Dependency DSP Formant, C++ Vulkan GPU Neural Engine & Android Native Voice Bridge)
376
+ > Production-Grade 4-Tier On-Device Speech Synthesis Framework (Multilingual Neural Orchestrator, Zero-Dependency DSP Formant & C++ Acceleration)
337
377
 
338
378
  ---
339
379
 
@@ -341,6 +381,8 @@ readme_content: |
341
381
 
342
382
  Termux-TTS is an enterprise-grade, on-device text-to-speech framework optimized for mobile edge hardware and Android Termux environments. Built to eliminate heavy dependency stacks and fragile driver behaviors, it features a resilient 4-Tier architecture:
343
383
 
384
+ - **Multilingual Neural Orchestrator**: Dynamic cross-language code-switching mesh covering 9 official languages (Korean, English, Japanese, Chinese, Hindi, Russian, Spanish, French, German) with Unicode script tokenization and 50ms context-aware silence padding.
385
+ - **Resident C-API In-Memory Engine**: Direct C-API memory residency with `SherpaResidentManager` achieving sub-0.18x real-time factor with ARM NEON SIMD acceleration and zero subprocess lag.
344
386
  - **Tier 1: Zero-Dependency Parametric DSP Formant Vocoder**: 0MB disk footprint, Rosenberg glottal pulse formulation, and 5-band biquad formant filters providing deterministic speech synthesis in under 50 milliseconds (RTF 0.013x).
345
387
  - **Tier 2: Android System Native Voice Bridge**: Direct IPC integration to physical Samsung and Google speech engines via the Termux-API service layer.
346
388
  - **Tier 3: Subprocess-Isolated Sherpa C++ CPU Engine**: Subprocess-isolated VITS acoustic modeling on ARM64 NEON with memory leak protection.
@@ -353,16 +395,16 @@ readme_content: |
353
395
  Measurements gathered on physical Android 16 hardware running Termux ARM64:
354
396
 
355
397
  | Target Device | Hardware Architecture | Synthesis Engine | Model Profile | Audio Length | Synthesis Time | Real-Time Factor (RTF) | Status |
356
- | :--- | :--- | :--- | :--- | :---: | :---: | :---: | :---: |
398
+ | :--- | :--- | :--- | :--- | :--- | :---: | :---: | :---: |
399
+ | **Galaxy A53** | Exynos 1280 / ARM64 NEON | Multilingual Neural C-API | `KSS + Lessac` (Bilingual) | 10.50 s | **1.82 s** | **0.173x** | Validated (5.78x RT) |
357
400
  | **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | Vulkan GPU Neural | `lessac-high-fp16` | 6.70 s | **6.65 s** | **0.993x** | Validated |
358
- | **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | Vulkan GPU Neural | `lessac-medium` | 4.59 s | **1.21 s** | **0.264x** | Validated |
401
+ | **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | Vulkan GPU Neural | `lessac-medium` | 4.59 s | **1.21 s** | **0.264x** | Validated (3.79x RT) |
359
402
  | **Galaxy A35** | Exynos 1380 / Mali-G68 MP5 | Vulkan GPU Neural | `lessac-medium` | 4.52 s | **5.18 s** | **1.146x** | Validated |
360
- | **Galaxy A35** | Exynos 1380 / Mali-G68 MP5 | Vulkan GPU Neural | `lessac-high-fp16` | 6.73 s | **34.33 s** | **5.098x** | Validated |
361
- | **ARM64 CPU** | Cortex-A78 / A55 | Parametric DSP | 5-Band Biquad | 4.15 s | **0.054 s** | **0.0130x** | Validated |
403
+ | **ARM64 CPU** | Cortex-A78 / A55 | Parametric DSP | 5-Band Biquad | 4.15 s | **0.054 s** | **0.0130x** | Validated (76x RT) |
362
404
 
363
405
  ---
364
406
 
365
- ## Installation & 1-Click Provisioning
407
+ ## Installation & Automated Provisioning
366
408
 
367
409
  ### 1. Package Installation
368
410
  ```bash
@@ -373,32 +415,46 @@ readme_content: |
373
415
  npm install termux-tts
374
416
  ```
375
417
 
376
- ### 2. Automated Engine & Weights Provisioning
377
- Automate the installation of precompiled ARM64 Vulkan C++ binaries and HuggingFace weights with self-test verification:
418
+ ### 2. Automated Model Provisioning
419
+ Automate downloading and linking precompiled models to the official immutable path (`/data/data/com.termux/files/home/models/tts/`):
378
420
  ```bash
379
- # Install Studio Tier (22.05kHz High-Fidelity)
380
- termux-tts install --tier high
421
+ # Default lightweight installation (Korean KSS + English Lessac)
422
+ termux-tts install
423
+
424
+ # On-demand provision specific language models:
425
+ termux-tts install --models hi # Hindi (Piper Swara)
426
+ termux-tts install --models ja # Japanese (Piper Hina)
427
+ termux-tts install --models ru # Russian (Piper Dmitri)
428
+ termux-tts install --models zh # Chinese (AISHELL3)
429
+ termux-tts install --models all # All 9 official languages
381
430
 
382
- # Or install Medium Tier (Balanced Performance)
383
- termux-tts install --tier medium
431
+ # Provision Studio Vulkan GPU engine:
432
+ termux-tts install --tier high
384
433
  ```
385
434
 
386
435
  ---
387
436
 
388
437
  ## Quickstart
389
438
 
390
- ### Global Command-Line Interface (CLI)
439
+ ### Global Command-Line Interface (Zero-Config)
391
440
  ```bash
392
- # Synthesize using Vulkan GPU with speaker playback
393
- termux-tts synth -e vulkan --tier high -t "Speech synthesis via Vulkan GPU." -o out.wav --play
441
+ # 1. Zero-config synthesis with physical speaker playback
442
+ termux-tts "Hello 방가방가 나는 parrot 이라고 해." -o ~/out.wav --play
443
+
444
+ # 2. Flag-based syntax
445
+ termux-tts -t "Multilingual neural speech synthesis on device." --play
394
446
 
395
- # Instant DSP Formant synthesis
447
+ # 3. Force language pinning
448
+ termux-tts -l en -t "Pure English text output." --play
449
+ termux-tts -l ko -t "한국어 단독 신경망 음성 합성." --play
450
+
451
+ # 4. Instant DSP Formant synthesis (0MB footprint)
396
452
  termux-tts synth -e dsp -t "Zero dependency DSP synthesis." -o dsp.wav
397
453
 
398
- # Direct hardware speaker broadcast
399
- termux-tts speak -t "Hardware speaker broadcast." -l en
454
+ # 5. Direct Android system native speaker broadcast
455
+ termux-tts speak -t "Hardware speaker broadcast via Android service."
400
456
 
401
- # Hardware diagnostics
457
+ # 6. Hardware diagnostics
402
458
  termux-tts doctor
403
459
  ```
404
460
 
@@ -406,12 +462,20 @@ readme_content: |
406
462
  ```python
407
463
  import termux_tts as tts
408
464
 
409
- # High-Resolution Vulkan GPU Neural Synthesis
465
+ # 1. Zero-Config Multilingual Neural Synthesis (Korean + English Code-Switching)
466
+ with tts.load() as engine:
467
+ result = engine.synthesize(
468
+ "Hello 방가방가 키키키키 나는 parrot 이라고 해.",
469
+ output="multilingual.wav"
470
+ )
471
+ print(f"Generated {result.duration_sec:.2f}s in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
472
+
473
+ # 2. Pure Vulkan GPU Neural Synthesis (Studio Tier)
410
474
  with tts.load(engine="vulkan", model_tier="high") as engine:
411
475
  result = engine.synthesize("Pure Vulkan neural execution on mobile.", output="speech.wav")
412
476
  print(f"Synthesized in {result.elapsed_ms:.1f}ms (RTF: {result.rtf:.4f}x)")
413
477
 
414
- # Zero-Dependency DSP Formant Synthesis
478
+ # 3. Zero-Dependency DSP Formant Synthesis
415
479
  with tts.load(engine="dsp", preset="balanced") as engine:
416
480
  result = engine.synthesize("Instant speech without model downloads.", output="dsp.wav")
417
481
  print(f"DSP Latency: {result.elapsed_ms:.1f}ms")
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "termux-tts",
3
- "version": "1.4.2",
3
+ "version": "1.4.4",
4
4
  "description": "On-device Text-to-Speech framework utilizing device resources (DSP Formant Vocoder, ONNX Neural Runtime & Android Native Voice)",
5
5
  "main": "index.js",
6
6
  "bin": {
package/pyproject.toml CHANGED
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "termux-tts"
7
- version = "1.4.2"
7
+ version = "1.4.4"
8
8
  description = "On-device 4-Tier Text-to-Speech framework utilizing device resources (DSP Formant Vocoder, C++ Sherpa-ONNX Neural, Android Native & Expressive)"
9
9
  readme = "README.pypi.md"
10
10
  requires-python = ">=3.10"
@@ -12,6 +12,9 @@ from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS, DSPSynt
12
12
  from .engine_sherpa import SherpaNeuralEngine, SherpaResult
13
13
  from .engine_vulkan import VulkanNeuralEngine, VulkanResult
14
14
  from .engine_expressive import ExpressiveEngine, ExpressiveResult
15
+ from .engine_multilingual import MultilingualNeuralEngine, MultilingualResult
16
+ from .engine_sherpa_capi import SherpaResidentManager, SherpaCapiSession
17
+ from .script_classifier import MultilingualTokenizer, ScriptRegistry, LanguageChunk
15
18
  from .installer import run_installation
16
19
  from .tokenizer import PhoneticTokenizer, EXPRESSIVE_TAGS
17
20
  from .g2p_korean import KoreanG2PEngine, korean_text_to_phonemes
@@ -29,7 +32,7 @@ from .exceptions import (
29
32
  ONNXNeuralEngine = SherpaNeuralEngine
30
33
  ONNXResult = SherpaResult
31
34
 
32
- __version__ = "1.4.2"
35
+ __version__ = "1.4.4"
33
36
  __all__ = [
34
37
  "TTSEngine",
35
38
  "load",
@@ -45,6 +48,13 @@ __all__ = [
45
48
  "VulkanResult",
46
49
  "ExpressiveEngine",
47
50
  "ExpressiveResult",
51
+ "MultilingualNeuralEngine",
52
+ "MultilingualResult",
53
+ "SherpaResidentManager",
54
+ "SherpaCapiSession",
55
+ "MultilingualTokenizer",
56
+ "ScriptRegistry",
57
+ "LanguageChunk",
48
58
  "run_installation",
49
59
  "ONNXNeuralEngine",
50
60
  "ONNXResult",
@@ -5,8 +5,9 @@ Handles 16-bit Linear PCM formatting with soft-clipping protection.
5
5
 
6
6
  import io
7
7
  import wave
8
- import numpy as np
8
+ from pathlib import Path
9
9
  from typing import Union
10
+ import numpy as np
10
11
  from .exceptions import TTSAudioEncodingError
11
12
 
12
13
  class AudioBuffer:
@@ -89,10 +90,12 @@ class AudioBuffer:
89
90
 
90
91
  def save(self, filepath: str) -> str:
91
92
  """Save the audio buffer to a WAV file on disk."""
93
+ target = Path(filepath).expanduser().resolve()
92
94
  wav_data = self.to_wav_bytes()
93
95
  try:
94
- with open(filepath, "wb") as f:
96
+ target.parent.mkdir(parents=True, exist_ok=True)
97
+ with open(target, "wb") as f:
95
98
  f.write(wav_data)
96
- return filepath
99
+ return str(target)
97
100
  except Exception as e:
98
101
  raise TTSAudioEncodingError(f"Failed to save WAV to '{filepath}': {e}") from e
package/termux_tts/cli.py CHANGED
@@ -11,6 +11,12 @@ import argparse
11
11
  from .engine import load, doctor
12
12
 
13
13
  def main():
14
+ # Route to default subcommand 'synth' if no recognized subcommand is given
15
+ KNOWN_COMMANDS = {"synth", "speak", "doctor", "install", "component", "model", "instance"}
16
+ raw_args = sys.argv[1:]
17
+ if raw_args and raw_args[0] not in KNOWN_COMMANDS and raw_args[0] not in ("-h", "--help", "-v", "--version"):
18
+ sys.argv.insert(1, "synth")
19
+
14
20
  parser = argparse.ArgumentParser(
15
21
  prog="termux-tts",
16
22
  description="Termux Neural & Native Text-to-Speech Engine"
@@ -19,13 +25,14 @@ def main():
19
25
 
20
26
  # 1. Synth (4-Tier Speech Synthesis)
21
27
  synth_parser = subparsers.add_parser("synth", help="Synthesize text to audio WAV file (Synth, Neural, Expressive)")
22
- synth_parser.add_argument("-t", "--text", required=True, help="Input text to synthesize")
28
+ synth_parser.add_argument("text_pos", nargs="*", default=[], help="Input text to synthesize (positional)")
29
+ synth_parser.add_argument("-t", "--text", default=None, help="Input text to synthesize (flag)")
23
30
  synth_parser.add_argument("-o", "--output", default="output.wav", help="Output WAV filepath")
24
- synth_parser.add_argument("-l", "--lang", default="ko", help="Language code (ko, en)")
31
+ synth_parser.add_argument("-l", "--lang", default="auto", help="Language code (auto=Multi-language auto switch, ko/kor=Korean only, en/eng=English only, ja/jpn=Japanese only)")
25
32
  synth_parser.add_argument(
26
33
  "-e", "--engine", default="auto",
27
- choices=["auto", "vulkan", "ncnn", "gpu", "synth", "dsp", "native", "neural", "onnx", "expressive"],
28
- help="Synthesis engine tier (vulkan=GPU NCNN, synth=0MB DSP, native=Android voice, neural=VITS C++, expressive=emotional)"
34
+ choices=["auto", "vulkan", "ncnn", "gpu", "synth", "dsp", "native", "neural", "onnx", "expressive", "multilingual", "hybrid"],
35
+ help="Synthesis engine tier (auto=Smart Routing, vulkan=GPU NCNN, synth=0MB DSP, native=Android voice, neural=VITS C++, expressive=emotional, multilingual=Cross-language)"
29
36
  )
30
37
  synth_parser.add_argument("-m", "--model", default=None, help="Path to model file or directory")
31
38
  synth_parser.add_argument("-p", "--preset", default="balanced", choices=["fast", "balanced", "expressive", "ultra"])
@@ -52,6 +59,10 @@ def main():
52
59
  # 4. Install (One-Click Automated Provisioner)
53
60
  install_parser = subparsers.add_parser("install", help="1-Click download and provision precompiled Vulkan binary & VITS studio models")
54
61
  install_parser.add_argument("--tier", default="high", choices=["high", "medium"], help="Model resolution tier (high=57MB Studio FP16, medium=25MB Fast)")
62
+ install_parser.add_argument(
63
+ "--models", default="default",
64
+ help="Language model packages to provision: default (Korean & English only), or specific code: hi, ja, zh, ru, es, fr, de, or 'all'"
65
+ )
55
66
  install_parser.add_argument("--force", action="store_true", help="Force overwrite existing binary and model assets")
56
67
  install_parser.add_argument("--no-play", action="store_true", help="Skip playback verification during self-test")
57
68
 
@@ -75,6 +86,12 @@ def main():
75
86
  subprocess.run(["termux-volume", "music", str(args.volume)], check=False)
76
87
 
77
88
  if args.command == "synth":
89
+ pos_text = " ".join(args.text_pos).strip() if getattr(args, "text_pos", None) else None
90
+ target_text = args.text or (pos_text if pos_text else None)
91
+ if not target_text:
92
+ synth_parser.error("the following arguments are required: text (as positional arguments or -t/--text)")
93
+
94
+ out_path = os.path.expanduser(args.output) if args.output else None
78
95
  with load(
79
96
  model=args.model,
80
97
  language=args.lang,
@@ -84,20 +101,20 @@ def main():
84
101
  engine=args.engine,
85
102
  tier=getattr(args, "tier", None),
86
103
  ) as engine:
87
- res = engine.synthesize(args.text, output=args.output, speed=args.speed)
104
+ res = engine.synthesize(target_text, output=out_path, speed=args.speed, language=args.lang)
88
105
  backend_name = getattr(res, "backend", "UNKNOWN")
89
106
  model_name = getattr(res, "model_name", "model")
90
107
  dur = getattr(res, "duration_sec", 0.0)
91
108
  elapsed = getattr(res, "elapsed_ms", 0.0)
92
109
  rtf = getattr(res, "rtf", 0.0)
93
- print(f"[SUCCESS] Synthesized via {backend_name} ({model_name}) -> {args.output}")
110
+ print(f"[SUCCESS] Synthesized via {backend_name} ({model_name}) -> {out_path or args.output}")
94
111
  print(f" Duration: {dur:.2f}s | Elapsed: {elapsed:.1f}ms | RTF: {rtf:.4f}x")
95
112
 
96
- if args.play and args.output and os.path.exists(args.output):
113
+ if args.play and out_path and os.path.exists(out_path):
97
114
  if shutil.which("termux-media-player"):
98
- subprocess.run(["termux-media-player", "play", args.output], check=False)
115
+ subprocess.run(["termux-media-player", "play", out_path], check=False)
99
116
  elif shutil.which("play-audio"):
100
- subprocess.run(["play-audio", args.output], check=False)
117
+ subprocess.run(["play-audio", out_path], check=False)
101
118
 
102
119
  elif args.command == "speak":
103
120
  with load(language=args.lang) as engine:
@@ -115,7 +132,7 @@ def main():
115
132
 
116
133
  elif args.command == "install":
117
134
  from .installer import run_installation
118
- run_installation(tier=args.tier, force=args.force, play=not args.no_play)
135
+ run_installation(tier=args.tier, models=getattr(args, "models", "all"), force=args.force, play=not args.no_play)
119
136
 
120
137
  elif args.command in ("component", "model", "instance") and _protocol_available:
121
138
  from ameva_component.cli_support import dispatch_protocol
@@ -22,6 +22,8 @@ from .engine_dsp import ParametricDSPEngine, DSPResult, QUALITY_PRESETS
22
22
  from .engine_sherpa import SherpaNeuralEngine, SherpaResult
23
23
  from .engine_vulkan import VulkanNeuralEngine, VulkanResult
24
24
  from .engine_expressive import ExpressiveEngine, ExpressiveResult
25
+ from .engine_multilingual import MultilingualNeuralEngine, MultilingualResult
26
+ from .script_classifier import MultilingualTokenizer
25
27
  from .hardware import (
26
28
  resolve_device_backend,
27
29
  bind_tts_hardware,
@@ -69,6 +71,7 @@ class TTSEngine:
69
71
  self._binding_plan = self._bind_hardware()
70
72
 
71
73
  self.native_engine = NativeAndroidEngine(language=language)
74
+ self._multilingual_engine: Optional[MultilingualNeuralEngine] = None
72
75
  self.synth_engine = self._resolve_synth_engine()
73
76
 
74
77
  def _bind_hardware(self):
@@ -76,9 +79,15 @@ class TTSEngine:
76
79
 
77
80
  def _resolve_synth_engine(self):
78
81
  t = self.requested_engine_type
82
+ from .script_classifier import normalize_language_code
83
+ norm_lang = normalize_language_code(self.language)
79
84
 
80
- # Explicit Vulkan GPU Tier (Fail-Fast)
81
- if t in ("vulkan", "gpu", "ncnn") or (self.requested_device in ("vulkan", "gpu") and t in ("neural", "vits", "auto")):
85
+ # Extended Languages (hi, ru, ja, zh, es, fr, de, ar) route to MultilingualNeuralEngine
86
+ if norm_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar"):
87
+ return self._get_multilingual_engine()
88
+
89
+ # Explicit Vulkan GPU Tier (Vulkan NCNN engine targets English Lessac)
90
+ if (t in ("vulkan", "gpu", "ncnn") or (self.requested_device in ("vulkan", "gpu") and t in ("neural", "vits", "auto"))) and norm_lang in ("en", "auto"):
82
91
  try:
83
92
  return VulkanNeuralEngine(
84
93
  model_path=self.model_path,
@@ -124,6 +133,10 @@ class TTSEngine:
124
133
  sample_rate=self.sample_rate,
125
134
  )
126
135
 
136
+ # Explicit Multilingual / Code-Switching Tier
137
+ elif t in ("multilingual", "codeswitch", "hybrid"):
138
+ return self._get_multilingual_engine()
139
+
127
140
  # Explicit Tier 2: Native
128
141
  elif t == "native":
129
142
  return self.native_engine
@@ -172,6 +185,15 @@ class TTSEngine:
172
185
  def binary(self) -> Optional[str]:
173
186
  return getattr(self.synth_engine, "binary", getattr(self.native_engine, "binary", None))
174
187
 
188
+ def _get_multilingual_engine(self) -> MultilingualNeuralEngine:
189
+ if self._multilingual_engine is None:
190
+ self._multilingual_engine = MultilingualNeuralEngine(
191
+ threads=self.threads,
192
+ device=self.device,
193
+ sample_rate=self.sample_rate or 22050,
194
+ )
195
+ return self._multilingual_engine
196
+
175
197
  def speak(self, text: str, stream: Optional[str] = None) -> NativeResult:
176
198
  """Speak text directly through physical Android speaker (Native Engine)."""
177
199
  if self._is_closed:
@@ -184,14 +206,56 @@ class TTSEngine:
184
206
  output: Optional[str] = None,
185
207
  speed: float = 1.0,
186
208
  preset: Optional[str] = None,
187
- ) -> Union[DSPResult, SherpaResult, ExpressiveResult, NativeResult]:
188
- """Synthesize text into speech audio buffer / WAV file."""
209
+ language: Optional[str] = None,
210
+ ) -> Union[DSPResult, SherpaResult, ExpressiveResult, NativeResult, MultilingualResult]:
211
+ """Synthesize text into speech audio buffer / WAV file with Zero-Config intelligent routing."""
189
212
  if self._is_closed:
190
213
  raise TTSInferenceError("Cannot synthesize: Engine session is closed.")
214
+
215
+ clean_text = text.strip() if text else ""
216
+ if not clean_text:
217
+ raise TTSInferenceError("Cannot synthesize empty text.")
218
+
219
+ from .script_classifier import normalize_language_code
220
+ target_lang = normalize_language_code(language or self.language)
221
+
222
+ # 1. If user explicitly pinned to lightweight DSP (0MB), respect choice
223
+ if self.requested_engine_type in ("dsp", "synth", "formant") or self.device == "dsp":
224
+ return self.synth_engine.synthesize(clean_text, output=output, speed=speed, preset=preset)
225
+
226
+ # 2. If user explicitly pinned to OS Native voice, speak directly
227
+ if self.requested_engine_type == "native":
228
+ return self.native_engine.speak(clean_text)
229
+
230
+ # 3. Multilingual auto-detection:
231
+ # Route to MultilingualNeuralEngine if multiple languages are present in text
232
+ # or if caller specifically requested multilingual / hybrid engine.
233
+ detected_langs = MultilingualTokenizer.detect_languages(clean_text)
234
+ is_mixed_text = len(detected_langs) > 1
235
+
236
+ should_route_multilingual = (
237
+ (self.requested_engine_type in ("auto", "multilingual", "codeswitch", "hybrid")) or
238
+ (is_mixed_text and self.requested_engine_type not in ("dsp", "synth", "native") and self.device != "dsp" and (self._multilingual_engine is not None or not isinstance(self.synth_engine, ParametricDSPEngine))) or
239
+ (target_lang in ("hi", "ru", "ja", "zh", "es", "fr", "de", "ar") and self.requested_engine_type not in ("dsp", "synth", "native"))
240
+ )
241
+
242
+ if should_route_multilingual:
243
+ multi_engine = self._get_multilingual_engine()
244
+ kwargs = {}
245
+ if target_lang != "auto":
246
+ kwargs["language"] = target_lang
247
+ return multi_engine.synthesize(
248
+ clean_text,
249
+ output=output,
250
+ speed=speed,
251
+ **kwargs
252
+ )
253
+
254
+ # 4. Standard single-language engine
191
255
  if hasattr(self.synth_engine, "synthesize"):
192
- return self.synth_engine.synthesize(text, output=output, speed=speed, preset=preset)
256
+ return self.synth_engine.synthesize(clean_text, output=output, speed=speed, preset=preset)
193
257
  elif hasattr(self.synth_engine, "speak"):
194
- return self.synth_engine.speak(text)
258
+ return self.synth_engine.speak(clean_text)
195
259
  raise TTSInferenceError(f"Selected engine '{type(self.synth_engine).__name__}' does not support synthesize.")
196
260
 
197
261
  def close(self) -> None:
@@ -199,6 +263,8 @@ class TTSEngine:
199
263
  self.native_engine.close()
200
264
  if hasattr(self.synth_engine, "close"):
201
265
  self.synth_engine.close()
266
+ if self._multilingual_engine is not None:
267
+ self._multilingual_engine.close()
202
268
 
203
269
  def __enter__(self):
204
270
  return self
@@ -209,7 +275,7 @@ class TTSEngine:
209
275
 
210
276
  def load(
211
277
  model: Optional[str] = None,
212
- language: str = "ko",
278
+ language: str = "auto",
213
279
  preset: str = "balanced",
214
280
  device: str = "auto",
215
281
  threads: int = 4,