termux-tts 1.4.0 → 1.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/release.yml +98 -0
- package/CHANGELOG.md +48 -27
- package/LICENSE +17 -17
- package/README.md +438 -114
- package/README.pypi.md +438 -54
- package/bin/cli.js +33 -33
- package/doc.config.yaml +532 -532
- package/docs/benchmarks.md +11 -11
- package/index.js +5 -5
- package/install.sh +53 -53
- package/package.json +57 -31
- package/pyproject.toml +81 -44
- package/setup.py +38 -38
- package/termux_tts/__init__.py +2 -1
- package/termux_tts/control/component.py +398 -398
- package/termux_tts/engine_native.py +104 -104
- package/termux_tts/exceptions.py +28 -28
- package/termux_tts/tokenizer.py +184 -184
|
@@ -1,104 +1,104 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Android OS System Native TTS Engine Bridge (Option B).
|
|
3
|
-
Directly communicates with Samsung / Google Voice Engine via Termux IPC.
|
|
4
|
-
Provides authentic human speech output with zero download overhead.
|
|
5
|
-
"""
|
|
6
|
-
|
|
7
|
-
import os
|
|
8
|
-
import time
|
|
9
|
-
import subprocess
|
|
10
|
-
import shutil
|
|
11
|
-
from dataclasses import dataclass
|
|
12
|
-
from typing import Optional
|
|
13
|
-
|
|
14
|
-
from .exceptions import TTSInferenceError
|
|
15
|
-
|
|
16
|
-
@dataclass
|
|
17
|
-
class NativeResult:
|
|
18
|
-
text: str
|
|
19
|
-
output_path: Optional[str]
|
|
20
|
-
language: str
|
|
21
|
-
pitch: float
|
|
22
|
-
rate: float
|
|
23
|
-
elapsed_ms: float
|
|
24
|
-
engine_name: str
|
|
25
|
-
|
|
26
|
-
class NativeAndroidEngine:
|
|
27
|
-
"""Android System Native TTS Engine Bridge (Samsung / Google Voice)."""
|
|
28
|
-
|
|
29
|
-
def __init__(
|
|
30
|
-
self,
|
|
31
|
-
language: str = "ko",
|
|
32
|
-
pitch: float = 1.0,
|
|
33
|
-
rate: float = 1.0,
|
|
34
|
-
stream: str = "MUSIC"
|
|
35
|
-
):
|
|
36
|
-
self.language = language
|
|
37
|
-
self.pitch = pitch
|
|
38
|
-
self.rate = rate
|
|
39
|
-
self.stream = stream
|
|
40
|
-
self.binary = self._find_binary()
|
|
41
|
-
|
|
42
|
-
def _find_binary(self) -> Optional[str]:
|
|
43
|
-
bin_path = shutil.which("termux-tts-speak")
|
|
44
|
-
if bin_path and os.access(bin_path, os.X_OK):
|
|
45
|
-
return bin_path
|
|
46
|
-
default_p = "/data/data/com.termux/files/usr/bin/termux-tts-speak"
|
|
47
|
-
if os.path.exists(default_p) and os.access(default_p, os.X_OK):
|
|
48
|
-
return default_p
|
|
49
|
-
return None
|
|
50
|
-
|
|
51
|
-
def speak(self, text: str, stream: Optional[str] = None) -> NativeResult:
|
|
52
|
-
"""Speak text directly through physical Android device speakers via termux-tts-speak IPC."""
|
|
53
|
-
if not text or not text.strip():
|
|
54
|
-
raise TTSInferenceError("Input text cannot be empty or whitespace only.")
|
|
55
|
-
|
|
56
|
-
if not self.binary:
|
|
57
|
-
raise TTSInferenceError(
|
|
58
|
-
"[FAIL-FAST] 'termux-tts-speak' binary not found on this system. "
|
|
59
|
-
"NativeAndroidEngine requires an Android Termux environment with 'termux-api' installed (run 'pkg install termux-api'). "
|
|
60
|
-
"If running on non-Android Linux/macOS/Windows, please use '--engine dsp' or '--engine onnx'."
|
|
61
|
-
)
|
|
62
|
-
|
|
63
|
-
t0 = time.perf_counter()
|
|
64
|
-
target_stream = stream or self.stream
|
|
65
|
-
cmd = [
|
|
66
|
-
self.binary,
|
|
67
|
-
"-l", self.language,
|
|
68
|
-
"-p", str(self.pitch),
|
|
69
|
-
"-r", str(self.rate),
|
|
70
|
-
"-s", target_stream,
|
|
71
|
-
text
|
|
72
|
-
]
|
|
73
|
-
|
|
74
|
-
try:
|
|
75
|
-
res = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
|
|
76
|
-
if res.returncode != 0:
|
|
77
|
-
err_msg = res.stderr.strip() or f"Process exited with code {res.returncode}"
|
|
78
|
-
raise TTSInferenceError(f"Native Android TTS engine execution failed: {err_msg}")
|
|
79
|
-
except subprocess.TimeoutExpired as e:
|
|
80
|
-
raise TTSInferenceError(f"Native Android TTS execution timed out (30s): {e}") from e
|
|
81
|
-
except Exception as e:
|
|
82
|
-
raise TTSInferenceError(f"Failed to execute native Android TTS: {e}") from e
|
|
83
|
-
|
|
84
|
-
elapsed_ms = (time.perf_counter() - t0) * 1000.0
|
|
85
|
-
return NativeResult(
|
|
86
|
-
text=text,
|
|
87
|
-
output_path=None,
|
|
88
|
-
language=self.language,
|
|
89
|
-
pitch=self.pitch,
|
|
90
|
-
rate=self.rate,
|
|
91
|
-
elapsed_ms=elapsed_ms,
|
|
92
|
-
engine_name="Android_Native_Voice_Engine"
|
|
93
|
-
)
|
|
94
|
-
|
|
95
|
-
def close(self) -> None:
|
|
96
|
-
pass
|
|
97
|
-
|
|
98
|
-
def __enter__(self):
|
|
99
|
-
return self
|
|
100
|
-
|
|
101
|
-
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
102
|
-
self.close()
|
|
103
|
-
|
|
104
|
-
|
|
1
|
+
"""
|
|
2
|
+
Android OS System Native TTS Engine Bridge (Option B).
|
|
3
|
+
Directly communicates with Samsung / Google Voice Engine via Termux IPC.
|
|
4
|
+
Provides authentic human speech output with zero download overhead.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import time
|
|
9
|
+
import subprocess
|
|
10
|
+
import shutil
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
from .exceptions import TTSInferenceError
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class NativeResult:
|
|
18
|
+
text: str
|
|
19
|
+
output_path: Optional[str]
|
|
20
|
+
language: str
|
|
21
|
+
pitch: float
|
|
22
|
+
rate: float
|
|
23
|
+
elapsed_ms: float
|
|
24
|
+
engine_name: str
|
|
25
|
+
|
|
26
|
+
class NativeAndroidEngine:
|
|
27
|
+
"""Android System Native TTS Engine Bridge (Samsung / Google Voice)."""
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
language: str = "ko",
|
|
32
|
+
pitch: float = 1.0,
|
|
33
|
+
rate: float = 1.0,
|
|
34
|
+
stream: str = "MUSIC"
|
|
35
|
+
):
|
|
36
|
+
self.language = language
|
|
37
|
+
self.pitch = pitch
|
|
38
|
+
self.rate = rate
|
|
39
|
+
self.stream = stream
|
|
40
|
+
self.binary = self._find_binary()
|
|
41
|
+
|
|
42
|
+
def _find_binary(self) -> Optional[str]:
|
|
43
|
+
bin_path = shutil.which("termux-tts-speak")
|
|
44
|
+
if bin_path and os.access(bin_path, os.X_OK):
|
|
45
|
+
return bin_path
|
|
46
|
+
default_p = "/data/data/com.termux/files/usr/bin/termux-tts-speak"
|
|
47
|
+
if os.path.exists(default_p) and os.access(default_p, os.X_OK):
|
|
48
|
+
return default_p
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
def speak(self, text: str, stream: Optional[str] = None) -> NativeResult:
|
|
52
|
+
"""Speak text directly through physical Android device speakers via termux-tts-speak IPC."""
|
|
53
|
+
if not text or not text.strip():
|
|
54
|
+
raise TTSInferenceError("Input text cannot be empty or whitespace only.")
|
|
55
|
+
|
|
56
|
+
if not self.binary:
|
|
57
|
+
raise TTSInferenceError(
|
|
58
|
+
"[FAIL-FAST] 'termux-tts-speak' binary not found on this system. "
|
|
59
|
+
"NativeAndroidEngine requires an Android Termux environment with 'termux-api' installed (run 'pkg install termux-api'). "
|
|
60
|
+
"If running on non-Android Linux/macOS/Windows, please use '--engine dsp' or '--engine onnx'."
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
t0 = time.perf_counter()
|
|
64
|
+
target_stream = stream or self.stream
|
|
65
|
+
cmd = [
|
|
66
|
+
self.binary,
|
|
67
|
+
"-l", self.language,
|
|
68
|
+
"-p", str(self.pitch),
|
|
69
|
+
"-r", str(self.rate),
|
|
70
|
+
"-s", target_stream,
|
|
71
|
+
text
|
|
72
|
+
]
|
|
73
|
+
|
|
74
|
+
try:
|
|
75
|
+
res = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
|
|
76
|
+
if res.returncode != 0:
|
|
77
|
+
err_msg = res.stderr.strip() or f"Process exited with code {res.returncode}"
|
|
78
|
+
raise TTSInferenceError(f"Native Android TTS engine execution failed: {err_msg}")
|
|
79
|
+
except subprocess.TimeoutExpired as e:
|
|
80
|
+
raise TTSInferenceError(f"Native Android TTS execution timed out (30s): {e}") from e
|
|
81
|
+
except Exception as e:
|
|
82
|
+
raise TTSInferenceError(f"Failed to execute native Android TTS: {e}") from e
|
|
83
|
+
|
|
84
|
+
elapsed_ms = (time.perf_counter() - t0) * 1000.0
|
|
85
|
+
return NativeResult(
|
|
86
|
+
text=text,
|
|
87
|
+
output_path=None,
|
|
88
|
+
language=self.language,
|
|
89
|
+
pitch=self.pitch,
|
|
90
|
+
rate=self.rate,
|
|
91
|
+
elapsed_ms=elapsed_ms,
|
|
92
|
+
engine_name="Android_Native_Voice_Engine"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
def close(self) -> None:
|
|
96
|
+
pass
|
|
97
|
+
|
|
98
|
+
def __enter__(self):
|
|
99
|
+
return self
|
|
100
|
+
|
|
101
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
102
|
+
self.close()
|
|
103
|
+
|
|
104
|
+
|
package/termux_tts/exceptions.py
CHANGED
|
@@ -1,28 +1,28 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Domain Specific Exceptions for termux-tts (Strict Fail-Fast Protocol).
|
|
3
|
-
Adheres to AOSF-ENG-STD-2026-V1 No-Fallback Governance.
|
|
4
|
-
"""
|
|
5
|
-
|
|
6
|
-
class TTSError(Exception):
|
|
7
|
-
"""Base exception for all termux-tts domain errors."""
|
|
8
|
-
pass
|
|
9
|
-
|
|
10
|
-
class TTSModelLoadError(TTSError):
|
|
11
|
-
"""Raised when the neural TTS model file cannot be loaded or is corrupted."""
|
|
12
|
-
pass
|
|
13
|
-
|
|
14
|
-
class TTSInferenceError(TTSError):
|
|
15
|
-
"""Raised when tensor forward pass or audio synthesis fails."""
|
|
16
|
-
pass
|
|
17
|
-
|
|
18
|
-
class VulkanInitializationError(TTSInferenceError):
|
|
19
|
-
"""Raised when Vulkan GPU is explicitly requested but unavailable (Strict Fail-Fast)."""
|
|
20
|
-
pass
|
|
21
|
-
|
|
22
|
-
class TTSAudioEncodingError(TTSError):
|
|
23
|
-
"""Raised when raw PCM cannot be encoded to standard WAV."""
|
|
24
|
-
pass
|
|
25
|
-
|
|
26
|
-
class TTSLanguageNotSupportedError(TTSError):
|
|
27
|
-
"""Raised when the requested language is not supported by the current tokenizer."""
|
|
28
|
-
pass
|
|
1
|
+
"""
|
|
2
|
+
Domain Specific Exceptions for termux-tts (Strict Fail-Fast Protocol).
|
|
3
|
+
Adheres to AOSF-ENG-STD-2026-V1 No-Fallback Governance.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
class TTSError(Exception):
|
|
7
|
+
"""Base exception for all termux-tts domain errors."""
|
|
8
|
+
pass
|
|
9
|
+
|
|
10
|
+
class TTSModelLoadError(TTSError):
|
|
11
|
+
"""Raised when the neural TTS model file cannot be loaded or is corrupted."""
|
|
12
|
+
pass
|
|
13
|
+
|
|
14
|
+
class TTSInferenceError(TTSError):
|
|
15
|
+
"""Raised when tensor forward pass or audio synthesis fails."""
|
|
16
|
+
pass
|
|
17
|
+
|
|
18
|
+
class VulkanInitializationError(TTSInferenceError):
|
|
19
|
+
"""Raised when Vulkan GPU is explicitly requested but unavailable (Strict Fail-Fast)."""
|
|
20
|
+
pass
|
|
21
|
+
|
|
22
|
+
class TTSAudioEncodingError(TTSError):
|
|
23
|
+
"""Raised when raw PCM cannot be encoded to standard WAV."""
|
|
24
|
+
pass
|
|
25
|
+
|
|
26
|
+
class TTSLanguageNotSupportedError(TTSError):
|
|
27
|
+
"""Raised when the requested language is not supported by the current tokenizer."""
|
|
28
|
+
pass
|
package/termux_tts/tokenizer.py
CHANGED
|
@@ -1,184 +1,184 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Grapheme-to-Phoneme (G2P) Phonetic Tokenizer for Korean and English.
|
|
3
|
-
Includes Number-to-Speech Normalizer and Inline Expressive Tag Parser.
|
|
4
|
-
"""
|
|
5
|
-
|
|
6
|
-
import re
|
|
7
|
-
from typing import List, Dict
|
|
8
|
-
from .exceptions import TTSLanguageNotSupportedError
|
|
9
|
-
from .g2p_korean import korean_text_to_phonemes
|
|
10
|
-
|
|
11
|
-
HANGUL_BASE = 0xAC00
|
|
12
|
-
HANGUL_END = 0xD7A3
|
|
13
|
-
|
|
14
|
-
CHO = [
|
|
15
|
-
"ㄱ", "ㄲ", "ㄴ", "ㄷ", "ㄸ", "ㄹ", "ㅁ", "ㅂ", "ㅃ", "ㅅ",
|
|
16
|
-
"ㅆ", "ㅇ", "ㅈ", "ㅉ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
|
|
17
|
-
]
|
|
18
|
-
JUNG = [
|
|
19
|
-
"ㅏ", "ㅐ", "ㅑ", "ㅒ", "ㅓ", "ㅔ", "ㅕ", "ㅖ", "ㅗ", "ㅘ",
|
|
20
|
-
"ㅙ", "ㅚ", "ㅛ", "ㅜ", "ㅝ", "ㅞ", "ㅟ", "ㅠ", "ㅡ", "ㅢ", "ㅣ"
|
|
21
|
-
]
|
|
22
|
-
JONG = [
|
|
23
|
-
"", "ㄱ", "ㄲ", "ㄳ", "ㄴ", "ㄵ", "ㄶ", "ㄷ", "ㄹ", "ㄺ",
|
|
24
|
-
"ㄻ", "ㄼ", "ㄽ", "ㄾ", "ㄿ", "ㅀ", "ㅁ", "ㅂ", "ㅄ", "ㅅ",
|
|
25
|
-
"ㅆ", "ㅇ", "ㅈ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
|
|
26
|
-
]
|
|
27
|
-
|
|
28
|
-
EXPRESSIVE_TAGS = {
|
|
29
|
-
"[laugh]": 1001,
|
|
30
|
-
"[sigh]": 1002,
|
|
31
|
-
"[breath]": 1003,
|
|
32
|
-
"[uv_break]": 1004,
|
|
33
|
-
"[clears_throat]": 1005,
|
|
34
|
-
"[pause]": 1006
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
VOCAB: List[str] = [
|
|
38
|
-
"_", " ", "!", "?", ",", ".", "~", "-",
|
|
39
|
-
*CHO, *JUNG, *[j for j in JONG if j],
|
|
40
|
-
*"abcdefghijklmnopqrstuvwxyz"
|
|
41
|
-
]
|
|
42
|
-
VOCAB_TO_ID: Dict[str, int] = {sym: idx for idx, sym in enumerate(VOCAB)}
|
|
43
|
-
PAD_ID: int = VOCAB_TO_ID["_"]
|
|
44
|
-
SPACE_ID: int = VOCAB_TO_ID[" "]
|
|
45
|
-
|
|
46
|
-
KOREAN_DIGITS = ["", "일", "이", "삼", "사", "오", "육", "칠", "팔", "구"]
|
|
47
|
-
SMALL_UNITS = ["", "십", "백", "천"]
|
|
48
|
-
BIG_UNITS = ["", "만", "억", "조", "경"]
|
|
49
|
-
|
|
50
|
-
ENGLISH_WORDS = {
|
|
51
|
-
"0": "zero", "1": "one", "2": "two", "3": "three", "4": "four",
|
|
52
|
-
"5": "five", "6": "six", "7": "seven", "8": "eight", "9": "nine",
|
|
53
|
-
"10": "ten", "11": "eleven", "12": "twelve", "13": "thirteen", "14": "fourteen",
|
|
54
|
-
"15": "fifteen", "16": "sixteen", "17": "seventeen", "18": "eighteen", "19": "nineteen",
|
|
55
|
-
"20": "twenty", "30": "thirty", "40": "forty", "50": "fifty",
|
|
56
|
-
"60": "sixty", "70": "seventy", "80": "eighty", "90": "ninety",
|
|
57
|
-
"100": "hundred", "1000": "thousand"
|
|
58
|
-
}
|
|
59
|
-
|
|
60
|
-
def decompose_hangul(char: str) -> List[str]:
|
|
61
|
-
code = ord(char)
|
|
62
|
-
if HANGUL_BASE <= code <= HANGUL_END:
|
|
63
|
-
offset = code - HANGUL_BASE
|
|
64
|
-
cho_idx = offset // (21 * 28)
|
|
65
|
-
jung_idx = (offset % (21 * 28)) // 28
|
|
66
|
-
jong_idx = offset % 28
|
|
67
|
-
res = [CHO[cho_idx], JUNG[jung_idx]]
|
|
68
|
-
if jong_idx > 0:
|
|
69
|
-
res.append(JONG[jong_idx])
|
|
70
|
-
return res
|
|
71
|
-
return [char]
|
|
72
|
-
|
|
73
|
-
def _convert_4digits_korean(chunk: str) -> str:
|
|
74
|
-
num = int(chunk)
|
|
75
|
-
if num == 0:
|
|
76
|
-
return ""
|
|
77
|
-
res = []
|
|
78
|
-
str_num = str(num).zfill(4)
|
|
79
|
-
for i, ch in enumerate(str_num):
|
|
80
|
-
d = int(ch)
|
|
81
|
-
if d > 0:
|
|
82
|
-
unit = SMALL_UNITS[3 - i]
|
|
83
|
-
if d == 1 and unit != "":
|
|
84
|
-
res.append(unit)
|
|
85
|
-
else:
|
|
86
|
-
res.append(KOREAN_DIGITS[d] + unit)
|
|
87
|
-
return "".join(res)
|
|
88
|
-
|
|
89
|
-
def number_to_korean_sino(num_str: str) -> str:
|
|
90
|
-
"""Convert integer string to authentic Sino-Korean place-value numerals (e.g. 1234 -> 천이백삼십사, 10000 -> 만)."""
|
|
91
|
-
try:
|
|
92
|
-
n = int(num_str)
|
|
93
|
-
except ValueError:
|
|
94
|
-
return num_str
|
|
95
|
-
if n == 0:
|
|
96
|
-
return "영"
|
|
97
|
-
rev_str = str(n)[::-1]
|
|
98
|
-
chunks = [rev_str[i:i+4][::-1] for i in range(0, len(rev_str), 4)]
|
|
99
|
-
parts = []
|
|
100
|
-
for i, chunk in enumerate(chunks):
|
|
101
|
-
c_korean = _convert_4digits_korean(chunk)
|
|
102
|
-
if c_korean:
|
|
103
|
-
unit = BIG_UNITS[i]
|
|
104
|
-
# 10000일 때 '일만' 대신 '만' (단, 210000 -> 이십일만)
|
|
105
|
-
if c_korean == "일" and unit != "" and len(chunks) == i + 1:
|
|
106
|
-
parts.append(unit)
|
|
107
|
-
else:
|
|
108
|
-
parts.append(c_korean + unit)
|
|
109
|
-
return "".join(reversed(parts))
|
|
110
|
-
|
|
111
|
-
def normalize_numbers_korean(text: str) -> str:
|
|
112
|
-
"""Convert digit sequences into spoken Korean place-value numerals."""
|
|
113
|
-
return re.sub(r"\d+", lambda m: number_to_korean_sino(m.group(0)), text)
|
|
114
|
-
|
|
115
|
-
def normalize_numbers_english(text: str) -> str:
|
|
116
|
-
"""Convert digit sequences into spoken English words."""
|
|
117
|
-
def _en_repl(m):
|
|
118
|
-
num_str = m.group(0)
|
|
119
|
-
if num_str in ENGLISH_WORDS:
|
|
120
|
-
return " " + ENGLISH_WORDS[num_str] + " "
|
|
121
|
-
# Digit-by-digit for phone numbers / codes
|
|
122
|
-
return " " + " ".join(ENGLISH_WORDS.get(d, d) for d in num_str) + " "
|
|
123
|
-
return re.sub(r"\d+", _en_repl, text)
|
|
124
|
-
|
|
125
|
-
class PhoneticTokenizer:
|
|
126
|
-
def __init__(self, language: str = "ko"):
|
|
127
|
-
self.language = language.lower()
|
|
128
|
-
self.vocab = VOCAB
|
|
129
|
-
if self.language not in ["ko", "korean", "en", "english"]:
|
|
130
|
-
raise TTSLanguageNotSupportedError(f"Language '{language}' is not supported. Supported: ['ko', 'en']")
|
|
131
|
-
|
|
132
|
-
def normalize_text(self, text: str) -> str:
|
|
133
|
-
if not text or not text.strip():
|
|
134
|
-
return ""
|
|
135
|
-
# 1. Normalize linebreaks and tabs
|
|
136
|
-
text = re.sub(r"[\r\n\t]+", " ", text)
|
|
137
|
-
|
|
138
|
-
# 2. Digits to spoken words
|
|
139
|
-
if self.language in ["ko", "korean"]:
|
|
140
|
-
text = normalize_numbers_korean(text)
|
|
141
|
-
text = korean_text_to_phonemes(text)
|
|
142
|
-
else:
|
|
143
|
-
text = normalize_numbers_english(text)
|
|
144
|
-
|
|
145
|
-
# 3. Collapse multiple spaces
|
|
146
|
-
text = re.sub(r"\s{2,}", " ", text)
|
|
147
|
-
return text.strip()
|
|
148
|
-
|
|
149
|
-
def tokenize(self, text: str) -> List[int]:
|
|
150
|
-
normalized = self.normalize_text(text)
|
|
151
|
-
if not normalized:
|
|
152
|
-
return []
|
|
153
|
-
|
|
154
|
-
# Parse expressive tags first
|
|
155
|
-
tag_pattern = re.compile(r"(\[[a-zA-Z_]+\])")
|
|
156
|
-
parts = tag_pattern.split(normalized)
|
|
157
|
-
|
|
158
|
-
tokens: List[int] = []
|
|
159
|
-
for part in parts:
|
|
160
|
-
if not part:
|
|
161
|
-
continue
|
|
162
|
-
if part in EXPRESSIVE_TAGS:
|
|
163
|
-
tokens.append(EXPRESSIVE_TAGS[part])
|
|
164
|
-
continue
|
|
165
|
-
|
|
166
|
-
if self.language in ["ko", "korean"]:
|
|
167
|
-
for char in part:
|
|
168
|
-
if char == " ":
|
|
169
|
-
tokens.append(SPACE_ID)
|
|
170
|
-
elif char in [".", ",", "!", "?", "~", "-"]:
|
|
171
|
-
if char in VOCAB_TO_ID:
|
|
172
|
-
tokens.append(VOCAB_TO_ID[char])
|
|
173
|
-
else:
|
|
174
|
-
jamos = decompose_hangul(char)
|
|
175
|
-
for j in jamos:
|
|
176
|
-
if j in VOCAB_TO_ID:
|
|
177
|
-
tokens.append(VOCAB_TO_ID[j])
|
|
178
|
-
else: # en
|
|
179
|
-
for char in part.lower():
|
|
180
|
-
if char in VOCAB_TO_ID:
|
|
181
|
-
tokens.append(VOCAB_TO_ID[char])
|
|
182
|
-
|
|
183
|
-
return tokens
|
|
184
|
-
|
|
1
|
+
"""
|
|
2
|
+
Grapheme-to-Phoneme (G2P) Phonetic Tokenizer for Korean and English.
|
|
3
|
+
Includes Number-to-Speech Normalizer and Inline Expressive Tag Parser.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import re
|
|
7
|
+
from typing import List, Dict
|
|
8
|
+
from .exceptions import TTSLanguageNotSupportedError
|
|
9
|
+
from .g2p_korean import korean_text_to_phonemes
|
|
10
|
+
|
|
11
|
+
HANGUL_BASE = 0xAC00
|
|
12
|
+
HANGUL_END = 0xD7A3
|
|
13
|
+
|
|
14
|
+
CHO = [
|
|
15
|
+
"ㄱ", "ㄲ", "ㄴ", "ㄷ", "ㄸ", "ㄹ", "ㅁ", "ㅂ", "ㅃ", "ㅅ",
|
|
16
|
+
"ㅆ", "ㅇ", "ㅈ", "ㅉ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
|
|
17
|
+
]
|
|
18
|
+
JUNG = [
|
|
19
|
+
"ㅏ", "ㅐ", "ㅑ", "ㅒ", "ㅓ", "ㅔ", "ㅕ", "ㅖ", "ㅗ", "ㅘ",
|
|
20
|
+
"ㅙ", "ㅚ", "ㅛ", "ㅜ", "ㅝ", "ㅞ", "ㅟ", "ㅠ", "ㅡ", "ㅢ", "ㅣ"
|
|
21
|
+
]
|
|
22
|
+
JONG = [
|
|
23
|
+
"", "ㄱ", "ㄲ", "ㄳ", "ㄴ", "ㄵ", "ㄶ", "ㄷ", "ㄹ", "ㄺ",
|
|
24
|
+
"ㄻ", "ㄼ", "ㄽ", "ㄾ", "ㄿ", "ㅀ", "ㅁ", "ㅂ", "ㅄ", "ㅅ",
|
|
25
|
+
"ㅆ", "ㅇ", "ㅈ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
EXPRESSIVE_TAGS = {
|
|
29
|
+
"[laugh]": 1001,
|
|
30
|
+
"[sigh]": 1002,
|
|
31
|
+
"[breath]": 1003,
|
|
32
|
+
"[uv_break]": 1004,
|
|
33
|
+
"[clears_throat]": 1005,
|
|
34
|
+
"[pause]": 1006
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
VOCAB: List[str] = [
|
|
38
|
+
"_", " ", "!", "?", ",", ".", "~", "-",
|
|
39
|
+
*CHO, *JUNG, *[j for j in JONG if j],
|
|
40
|
+
*"abcdefghijklmnopqrstuvwxyz"
|
|
41
|
+
]
|
|
42
|
+
VOCAB_TO_ID: Dict[str, int] = {sym: idx for idx, sym in enumerate(VOCAB)}
|
|
43
|
+
PAD_ID: int = VOCAB_TO_ID["_"]
|
|
44
|
+
SPACE_ID: int = VOCAB_TO_ID[" "]
|
|
45
|
+
|
|
46
|
+
KOREAN_DIGITS = ["", "일", "이", "삼", "사", "오", "육", "칠", "팔", "구"]
|
|
47
|
+
SMALL_UNITS = ["", "십", "백", "천"]
|
|
48
|
+
BIG_UNITS = ["", "만", "억", "조", "경"]
|
|
49
|
+
|
|
50
|
+
ENGLISH_WORDS = {
|
|
51
|
+
"0": "zero", "1": "one", "2": "two", "3": "three", "4": "four",
|
|
52
|
+
"5": "five", "6": "six", "7": "seven", "8": "eight", "9": "nine",
|
|
53
|
+
"10": "ten", "11": "eleven", "12": "twelve", "13": "thirteen", "14": "fourteen",
|
|
54
|
+
"15": "fifteen", "16": "sixteen", "17": "seventeen", "18": "eighteen", "19": "nineteen",
|
|
55
|
+
"20": "twenty", "30": "thirty", "40": "forty", "50": "fifty",
|
|
56
|
+
"60": "sixty", "70": "seventy", "80": "eighty", "90": "ninety",
|
|
57
|
+
"100": "hundred", "1000": "thousand"
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
def decompose_hangul(char: str) -> List[str]:
|
|
61
|
+
code = ord(char)
|
|
62
|
+
if HANGUL_BASE <= code <= HANGUL_END:
|
|
63
|
+
offset = code - HANGUL_BASE
|
|
64
|
+
cho_idx = offset // (21 * 28)
|
|
65
|
+
jung_idx = (offset % (21 * 28)) // 28
|
|
66
|
+
jong_idx = offset % 28
|
|
67
|
+
res = [CHO[cho_idx], JUNG[jung_idx]]
|
|
68
|
+
if jong_idx > 0:
|
|
69
|
+
res.append(JONG[jong_idx])
|
|
70
|
+
return res
|
|
71
|
+
return [char]
|
|
72
|
+
|
|
73
|
+
def _convert_4digits_korean(chunk: str) -> str:
|
|
74
|
+
num = int(chunk)
|
|
75
|
+
if num == 0:
|
|
76
|
+
return ""
|
|
77
|
+
res = []
|
|
78
|
+
str_num = str(num).zfill(4)
|
|
79
|
+
for i, ch in enumerate(str_num):
|
|
80
|
+
d = int(ch)
|
|
81
|
+
if d > 0:
|
|
82
|
+
unit = SMALL_UNITS[3 - i]
|
|
83
|
+
if d == 1 and unit != "":
|
|
84
|
+
res.append(unit)
|
|
85
|
+
else:
|
|
86
|
+
res.append(KOREAN_DIGITS[d] + unit)
|
|
87
|
+
return "".join(res)
|
|
88
|
+
|
|
89
|
+
def number_to_korean_sino(num_str: str) -> str:
|
|
90
|
+
"""Convert integer string to authentic Sino-Korean place-value numerals (e.g. 1234 -> 천이백삼십사, 10000 -> 만)."""
|
|
91
|
+
try:
|
|
92
|
+
n = int(num_str)
|
|
93
|
+
except ValueError:
|
|
94
|
+
return num_str
|
|
95
|
+
if n == 0:
|
|
96
|
+
return "영"
|
|
97
|
+
rev_str = str(n)[::-1]
|
|
98
|
+
chunks = [rev_str[i:i+4][::-1] for i in range(0, len(rev_str), 4)]
|
|
99
|
+
parts = []
|
|
100
|
+
for i, chunk in enumerate(chunks):
|
|
101
|
+
c_korean = _convert_4digits_korean(chunk)
|
|
102
|
+
if c_korean:
|
|
103
|
+
unit = BIG_UNITS[i]
|
|
104
|
+
# 10000일 때 '일만' 대신 '만' (단, 210000 -> 이십일만)
|
|
105
|
+
if c_korean == "일" and unit != "" and len(chunks) == i + 1:
|
|
106
|
+
parts.append(unit)
|
|
107
|
+
else:
|
|
108
|
+
parts.append(c_korean + unit)
|
|
109
|
+
return "".join(reversed(parts))
|
|
110
|
+
|
|
111
|
+
def normalize_numbers_korean(text: str) -> str:
|
|
112
|
+
"""Convert digit sequences into spoken Korean place-value numerals."""
|
|
113
|
+
return re.sub(r"\d+", lambda m: number_to_korean_sino(m.group(0)), text)
|
|
114
|
+
|
|
115
|
+
def normalize_numbers_english(text: str) -> str:
|
|
116
|
+
"""Convert digit sequences into spoken English words."""
|
|
117
|
+
def _en_repl(m):
|
|
118
|
+
num_str = m.group(0)
|
|
119
|
+
if num_str in ENGLISH_WORDS:
|
|
120
|
+
return " " + ENGLISH_WORDS[num_str] + " "
|
|
121
|
+
# Digit-by-digit for phone numbers / codes
|
|
122
|
+
return " " + " ".join(ENGLISH_WORDS.get(d, d) for d in num_str) + " "
|
|
123
|
+
return re.sub(r"\d+", _en_repl, text)
|
|
124
|
+
|
|
125
|
+
class PhoneticTokenizer:
|
|
126
|
+
def __init__(self, language: str = "ko"):
|
|
127
|
+
self.language = language.lower()
|
|
128
|
+
self.vocab = VOCAB
|
|
129
|
+
if self.language not in ["ko", "korean", "en", "english"]:
|
|
130
|
+
raise TTSLanguageNotSupportedError(f"Language '{language}' is not supported. Supported: ['ko', 'en']")
|
|
131
|
+
|
|
132
|
+
def normalize_text(self, text: str) -> str:
|
|
133
|
+
if not text or not text.strip():
|
|
134
|
+
return ""
|
|
135
|
+
# 1. Normalize linebreaks and tabs
|
|
136
|
+
text = re.sub(r"[\r\n\t]+", " ", text)
|
|
137
|
+
|
|
138
|
+
# 2. Digits to spoken words
|
|
139
|
+
if self.language in ["ko", "korean"]:
|
|
140
|
+
text = normalize_numbers_korean(text)
|
|
141
|
+
text = korean_text_to_phonemes(text)
|
|
142
|
+
else:
|
|
143
|
+
text = normalize_numbers_english(text)
|
|
144
|
+
|
|
145
|
+
# 3. Collapse multiple spaces
|
|
146
|
+
text = re.sub(r"\s{2,}", " ", text)
|
|
147
|
+
return text.strip()
|
|
148
|
+
|
|
149
|
+
def tokenize(self, text: str) -> List[int]:
|
|
150
|
+
normalized = self.normalize_text(text)
|
|
151
|
+
if not normalized:
|
|
152
|
+
return []
|
|
153
|
+
|
|
154
|
+
# Parse expressive tags first
|
|
155
|
+
tag_pattern = re.compile(r"(\[[a-zA-Z_]+\])")
|
|
156
|
+
parts = tag_pattern.split(normalized)
|
|
157
|
+
|
|
158
|
+
tokens: List[int] = []
|
|
159
|
+
for part in parts:
|
|
160
|
+
if not part:
|
|
161
|
+
continue
|
|
162
|
+
if part in EXPRESSIVE_TAGS:
|
|
163
|
+
tokens.append(EXPRESSIVE_TAGS[part])
|
|
164
|
+
continue
|
|
165
|
+
|
|
166
|
+
if self.language in ["ko", "korean"]:
|
|
167
|
+
for char in part:
|
|
168
|
+
if char == " ":
|
|
169
|
+
tokens.append(SPACE_ID)
|
|
170
|
+
elif char in [".", ",", "!", "?", "~", "-"]:
|
|
171
|
+
if char in VOCAB_TO_ID:
|
|
172
|
+
tokens.append(VOCAB_TO_ID[char])
|
|
173
|
+
else:
|
|
174
|
+
jamos = decompose_hangul(char)
|
|
175
|
+
for j in jamos:
|
|
176
|
+
if j in VOCAB_TO_ID:
|
|
177
|
+
tokens.append(VOCAB_TO_ID[j])
|
|
178
|
+
else: # en
|
|
179
|
+
for char in part.lower():
|
|
180
|
+
if char in VOCAB_TO_ID:
|
|
181
|
+
tokens.append(VOCAB_TO_ID[char])
|
|
182
|
+
|
|
183
|
+
return tokens
|
|
184
|
+
|