termux-tts 1.4.0 → 1.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,104 +1,104 @@
1
- """
2
- Android OS System Native TTS Engine Bridge (Option B).
3
- Directly communicates with Samsung / Google Voice Engine via Termux IPC.
4
- Provides authentic human speech output with zero download overhead.
5
- """
6
-
7
- import os
8
- import time
9
- import subprocess
10
- import shutil
11
- from dataclasses import dataclass
12
- from typing import Optional
13
-
14
- from .exceptions import TTSInferenceError
15
-
16
- @dataclass
17
- class NativeResult:
18
- text: str
19
- output_path: Optional[str]
20
- language: str
21
- pitch: float
22
- rate: float
23
- elapsed_ms: float
24
- engine_name: str
25
-
26
- class NativeAndroidEngine:
27
- """Android System Native TTS Engine Bridge (Samsung / Google Voice)."""
28
-
29
- def __init__(
30
- self,
31
- language: str = "ko",
32
- pitch: float = 1.0,
33
- rate: float = 1.0,
34
- stream: str = "MUSIC"
35
- ):
36
- self.language = language
37
- self.pitch = pitch
38
- self.rate = rate
39
- self.stream = stream
40
- self.binary = self._find_binary()
41
-
42
- def _find_binary(self) -> Optional[str]:
43
- bin_path = shutil.which("termux-tts-speak")
44
- if bin_path and os.access(bin_path, os.X_OK):
45
- return bin_path
46
- default_p = "/data/data/com.termux/files/usr/bin/termux-tts-speak"
47
- if os.path.exists(default_p) and os.access(default_p, os.X_OK):
48
- return default_p
49
- return None
50
-
51
- def speak(self, text: str, stream: Optional[str] = None) -> NativeResult:
52
- """Speak text directly through physical Android device speakers via termux-tts-speak IPC."""
53
- if not text or not text.strip():
54
- raise TTSInferenceError("Input text cannot be empty or whitespace only.")
55
-
56
- if not self.binary:
57
- raise TTSInferenceError(
58
- "[FAIL-FAST] 'termux-tts-speak' binary not found on this system. "
59
- "NativeAndroidEngine requires an Android Termux environment with 'termux-api' installed (run 'pkg install termux-api'). "
60
- "If running on non-Android Linux/macOS/Windows, please use '--engine dsp' or '--engine onnx'."
61
- )
62
-
63
- t0 = time.perf_counter()
64
- target_stream = stream or self.stream
65
- cmd = [
66
- self.binary,
67
- "-l", self.language,
68
- "-p", str(self.pitch),
69
- "-r", str(self.rate),
70
- "-s", target_stream,
71
- text
72
- ]
73
-
74
- try:
75
- res = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
76
- if res.returncode != 0:
77
- err_msg = res.stderr.strip() or f"Process exited with code {res.returncode}"
78
- raise TTSInferenceError(f"Native Android TTS engine execution failed: {err_msg}")
79
- except subprocess.TimeoutExpired as e:
80
- raise TTSInferenceError(f"Native Android TTS execution timed out (30s): {e}") from e
81
- except Exception as e:
82
- raise TTSInferenceError(f"Failed to execute native Android TTS: {e}") from e
83
-
84
- elapsed_ms = (time.perf_counter() - t0) * 1000.0
85
- return NativeResult(
86
- text=text,
87
- output_path=None,
88
- language=self.language,
89
- pitch=self.pitch,
90
- rate=self.rate,
91
- elapsed_ms=elapsed_ms,
92
- engine_name="Android_Native_Voice_Engine"
93
- )
94
-
95
- def close(self) -> None:
96
- pass
97
-
98
- def __enter__(self):
99
- return self
100
-
101
- def __exit__(self, exc_type, exc_val, exc_tb):
102
- self.close()
103
-
104
-
1
+ """
2
+ Android OS System Native TTS Engine Bridge (Option B).
3
+ Directly communicates with Samsung / Google Voice Engine via Termux IPC.
4
+ Provides authentic human speech output with zero download overhead.
5
+ """
6
+
7
+ import os
8
+ import time
9
+ import subprocess
10
+ import shutil
11
+ from dataclasses import dataclass
12
+ from typing import Optional
13
+
14
+ from .exceptions import TTSInferenceError
15
+
16
+ @dataclass
17
+ class NativeResult:
18
+ text: str
19
+ output_path: Optional[str]
20
+ language: str
21
+ pitch: float
22
+ rate: float
23
+ elapsed_ms: float
24
+ engine_name: str
25
+
26
+ class NativeAndroidEngine:
27
+ """Android System Native TTS Engine Bridge (Samsung / Google Voice)."""
28
+
29
+ def __init__(
30
+ self,
31
+ language: str = "ko",
32
+ pitch: float = 1.0,
33
+ rate: float = 1.0,
34
+ stream: str = "MUSIC"
35
+ ):
36
+ self.language = language
37
+ self.pitch = pitch
38
+ self.rate = rate
39
+ self.stream = stream
40
+ self.binary = self._find_binary()
41
+
42
+ def _find_binary(self) -> Optional[str]:
43
+ bin_path = shutil.which("termux-tts-speak")
44
+ if bin_path and os.access(bin_path, os.X_OK):
45
+ return bin_path
46
+ default_p = "/data/data/com.termux/files/usr/bin/termux-tts-speak"
47
+ if os.path.exists(default_p) and os.access(default_p, os.X_OK):
48
+ return default_p
49
+ return None
50
+
51
+ def speak(self, text: str, stream: Optional[str] = None) -> NativeResult:
52
+ """Speak text directly through physical Android device speakers via termux-tts-speak IPC."""
53
+ if not text or not text.strip():
54
+ raise TTSInferenceError("Input text cannot be empty or whitespace only.")
55
+
56
+ if not self.binary:
57
+ raise TTSInferenceError(
58
+ "[FAIL-FAST] 'termux-tts-speak' binary not found on this system. "
59
+ "NativeAndroidEngine requires an Android Termux environment with 'termux-api' installed (run 'pkg install termux-api'). "
60
+ "If running on non-Android Linux/macOS/Windows, please use '--engine dsp' or '--engine onnx'."
61
+ )
62
+
63
+ t0 = time.perf_counter()
64
+ target_stream = stream or self.stream
65
+ cmd = [
66
+ self.binary,
67
+ "-l", self.language,
68
+ "-p", str(self.pitch),
69
+ "-r", str(self.rate),
70
+ "-s", target_stream,
71
+ text
72
+ ]
73
+
74
+ try:
75
+ res = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
76
+ if res.returncode != 0:
77
+ err_msg = res.stderr.strip() or f"Process exited with code {res.returncode}"
78
+ raise TTSInferenceError(f"Native Android TTS engine execution failed: {err_msg}")
79
+ except subprocess.TimeoutExpired as e:
80
+ raise TTSInferenceError(f"Native Android TTS execution timed out (30s): {e}") from e
81
+ except Exception as e:
82
+ raise TTSInferenceError(f"Failed to execute native Android TTS: {e}") from e
83
+
84
+ elapsed_ms = (time.perf_counter() - t0) * 1000.0
85
+ return NativeResult(
86
+ text=text,
87
+ output_path=None,
88
+ language=self.language,
89
+ pitch=self.pitch,
90
+ rate=self.rate,
91
+ elapsed_ms=elapsed_ms,
92
+ engine_name="Android_Native_Voice_Engine"
93
+ )
94
+
95
+ def close(self) -> None:
96
+ pass
97
+
98
+ def __enter__(self):
99
+ return self
100
+
101
+ def __exit__(self, exc_type, exc_val, exc_tb):
102
+ self.close()
103
+
104
+
@@ -1,28 +1,28 @@
1
- """
2
- Domain Specific Exceptions for termux-tts (Strict Fail-Fast Protocol).
3
- Adheres to AOSF-ENG-STD-2026-V1 No-Fallback Governance.
4
- """
5
-
6
- class TTSError(Exception):
7
- """Base exception for all termux-tts domain errors."""
8
- pass
9
-
10
- class TTSModelLoadError(TTSError):
11
- """Raised when the neural TTS model file cannot be loaded or is corrupted."""
12
- pass
13
-
14
- class TTSInferenceError(TTSError):
15
- """Raised when tensor forward pass or audio synthesis fails."""
16
- pass
17
-
18
- class VulkanInitializationError(TTSInferenceError):
19
- """Raised when Vulkan GPU is explicitly requested but unavailable (Strict Fail-Fast)."""
20
- pass
21
-
22
- class TTSAudioEncodingError(TTSError):
23
- """Raised when raw PCM cannot be encoded to standard WAV."""
24
- pass
25
-
26
- class TTSLanguageNotSupportedError(TTSError):
27
- """Raised when the requested language is not supported by the current tokenizer."""
28
- pass
1
+ """
2
+ Domain Specific Exceptions for termux-tts (Strict Fail-Fast Protocol).
3
+ Adheres to AOSF-ENG-STD-2026-V1 No-Fallback Governance.
4
+ """
5
+
6
+ class TTSError(Exception):
7
+ """Base exception for all termux-tts domain errors."""
8
+ pass
9
+
10
+ class TTSModelLoadError(TTSError):
11
+ """Raised when the neural TTS model file cannot be loaded or is corrupted."""
12
+ pass
13
+
14
+ class TTSInferenceError(TTSError):
15
+ """Raised when tensor forward pass or audio synthesis fails."""
16
+ pass
17
+
18
+ class VulkanInitializationError(TTSInferenceError):
19
+ """Raised when Vulkan GPU is explicitly requested but unavailable (Strict Fail-Fast)."""
20
+ pass
21
+
22
+ class TTSAudioEncodingError(TTSError):
23
+ """Raised when raw PCM cannot be encoded to standard WAV."""
24
+ pass
25
+
26
+ class TTSLanguageNotSupportedError(TTSError):
27
+ """Raised when the requested language is not supported by the current tokenizer."""
28
+ pass
@@ -1,184 +1,184 @@
1
- """
2
- Grapheme-to-Phoneme (G2P) Phonetic Tokenizer for Korean and English.
3
- Includes Number-to-Speech Normalizer and Inline Expressive Tag Parser.
4
- """
5
-
6
- import re
7
- from typing import List, Dict
8
- from .exceptions import TTSLanguageNotSupportedError
9
- from .g2p_korean import korean_text_to_phonemes
10
-
11
- HANGUL_BASE = 0xAC00
12
- HANGUL_END = 0xD7A3
13
-
14
- CHO = [
15
- "ㄱ", "ㄲ", "ㄴ", "ㄷ", "ㄸ", "ㄹ", "ㅁ", "ㅂ", "ㅃ", "ㅅ",
16
- "ㅆ", "ㅇ", "ㅈ", "ㅉ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
17
- ]
18
- JUNG = [
19
- "ㅏ", "ㅐ", "ㅑ", "ㅒ", "ㅓ", "ㅔ", "ㅕ", "ㅖ", "ㅗ", "ㅘ",
20
- "ㅙ", "ㅚ", "ㅛ", "ㅜ", "ㅝ", "ㅞ", "ㅟ", "ㅠ", "ㅡ", "ㅢ", "ㅣ"
21
- ]
22
- JONG = [
23
- "", "ㄱ", "ㄲ", "ㄳ", "ㄴ", "ㄵ", "ㄶ", "ㄷ", "ㄹ", "ㄺ",
24
- "ㄻ", "ㄼ", "ㄽ", "ㄾ", "ㄿ", "ㅀ", "ㅁ", "ㅂ", "ㅄ", "ㅅ",
25
- "ㅆ", "ㅇ", "ㅈ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
26
- ]
27
-
28
- EXPRESSIVE_TAGS = {
29
- "[laugh]": 1001,
30
- "[sigh]": 1002,
31
- "[breath]": 1003,
32
- "[uv_break]": 1004,
33
- "[clears_throat]": 1005,
34
- "[pause]": 1006
35
- }
36
-
37
- VOCAB: List[str] = [
38
- "_", " ", "!", "?", ",", ".", "~", "-",
39
- *CHO, *JUNG, *[j for j in JONG if j],
40
- *"abcdefghijklmnopqrstuvwxyz"
41
- ]
42
- VOCAB_TO_ID: Dict[str, int] = {sym: idx for idx, sym in enumerate(VOCAB)}
43
- PAD_ID: int = VOCAB_TO_ID["_"]
44
- SPACE_ID: int = VOCAB_TO_ID[" "]
45
-
46
- KOREAN_DIGITS = ["", "일", "이", "삼", "사", "오", "육", "칠", "팔", "구"]
47
- SMALL_UNITS = ["", "십", "백", "천"]
48
- BIG_UNITS = ["", "만", "억", "조", "경"]
49
-
50
- ENGLISH_WORDS = {
51
- "0": "zero", "1": "one", "2": "two", "3": "three", "4": "four",
52
- "5": "five", "6": "six", "7": "seven", "8": "eight", "9": "nine",
53
- "10": "ten", "11": "eleven", "12": "twelve", "13": "thirteen", "14": "fourteen",
54
- "15": "fifteen", "16": "sixteen", "17": "seventeen", "18": "eighteen", "19": "nineteen",
55
- "20": "twenty", "30": "thirty", "40": "forty", "50": "fifty",
56
- "60": "sixty", "70": "seventy", "80": "eighty", "90": "ninety",
57
- "100": "hundred", "1000": "thousand"
58
- }
59
-
60
- def decompose_hangul(char: str) -> List[str]:
61
- code = ord(char)
62
- if HANGUL_BASE <= code <= HANGUL_END:
63
- offset = code - HANGUL_BASE
64
- cho_idx = offset // (21 * 28)
65
- jung_idx = (offset % (21 * 28)) // 28
66
- jong_idx = offset % 28
67
- res = [CHO[cho_idx], JUNG[jung_idx]]
68
- if jong_idx > 0:
69
- res.append(JONG[jong_idx])
70
- return res
71
- return [char]
72
-
73
- def _convert_4digits_korean(chunk: str) -> str:
74
- num = int(chunk)
75
- if num == 0:
76
- return ""
77
- res = []
78
- str_num = str(num).zfill(4)
79
- for i, ch in enumerate(str_num):
80
- d = int(ch)
81
- if d > 0:
82
- unit = SMALL_UNITS[3 - i]
83
- if d == 1 and unit != "":
84
- res.append(unit)
85
- else:
86
- res.append(KOREAN_DIGITS[d] + unit)
87
- return "".join(res)
88
-
89
- def number_to_korean_sino(num_str: str) -> str:
90
- """Convert integer string to authentic Sino-Korean place-value numerals (e.g. 1234 -> 천이백삼십사, 10000 -> 만)."""
91
- try:
92
- n = int(num_str)
93
- except ValueError:
94
- return num_str
95
- if n == 0:
96
- return "영"
97
- rev_str = str(n)[::-1]
98
- chunks = [rev_str[i:i+4][::-1] for i in range(0, len(rev_str), 4)]
99
- parts = []
100
- for i, chunk in enumerate(chunks):
101
- c_korean = _convert_4digits_korean(chunk)
102
- if c_korean:
103
- unit = BIG_UNITS[i]
104
- # 10000일 때 '일만' 대신 '만' (단, 210000 -> 이십일만)
105
- if c_korean == "일" and unit != "" and len(chunks) == i + 1:
106
- parts.append(unit)
107
- else:
108
- parts.append(c_korean + unit)
109
- return "".join(reversed(parts))
110
-
111
- def normalize_numbers_korean(text: str) -> str:
112
- """Convert digit sequences into spoken Korean place-value numerals."""
113
- return re.sub(r"\d+", lambda m: number_to_korean_sino(m.group(0)), text)
114
-
115
- def normalize_numbers_english(text: str) -> str:
116
- """Convert digit sequences into spoken English words."""
117
- def _en_repl(m):
118
- num_str = m.group(0)
119
- if num_str in ENGLISH_WORDS:
120
- return " " + ENGLISH_WORDS[num_str] + " "
121
- # Digit-by-digit for phone numbers / codes
122
- return " " + " ".join(ENGLISH_WORDS.get(d, d) for d in num_str) + " "
123
- return re.sub(r"\d+", _en_repl, text)
124
-
125
- class PhoneticTokenizer:
126
- def __init__(self, language: str = "ko"):
127
- self.language = language.lower()
128
- self.vocab = VOCAB
129
- if self.language not in ["ko", "korean", "en", "english"]:
130
- raise TTSLanguageNotSupportedError(f"Language '{language}' is not supported. Supported: ['ko', 'en']")
131
-
132
- def normalize_text(self, text: str) -> str:
133
- if not text or not text.strip():
134
- return ""
135
- # 1. Normalize linebreaks and tabs
136
- text = re.sub(r"[\r\n\t]+", " ", text)
137
-
138
- # 2. Digits to spoken words
139
- if self.language in ["ko", "korean"]:
140
- text = normalize_numbers_korean(text)
141
- text = korean_text_to_phonemes(text)
142
- else:
143
- text = normalize_numbers_english(text)
144
-
145
- # 3. Collapse multiple spaces
146
- text = re.sub(r"\s{2,}", " ", text)
147
- return text.strip()
148
-
149
- def tokenize(self, text: str) -> List[int]:
150
- normalized = self.normalize_text(text)
151
- if not normalized:
152
- return []
153
-
154
- # Parse expressive tags first
155
- tag_pattern = re.compile(r"(\[[a-zA-Z_]+\])")
156
- parts = tag_pattern.split(normalized)
157
-
158
- tokens: List[int] = []
159
- for part in parts:
160
- if not part:
161
- continue
162
- if part in EXPRESSIVE_TAGS:
163
- tokens.append(EXPRESSIVE_TAGS[part])
164
- continue
165
-
166
- if self.language in ["ko", "korean"]:
167
- for char in part:
168
- if char == " ":
169
- tokens.append(SPACE_ID)
170
- elif char in [".", ",", "!", "?", "~", "-"]:
171
- if char in VOCAB_TO_ID:
172
- tokens.append(VOCAB_TO_ID[char])
173
- else:
174
- jamos = decompose_hangul(char)
175
- for j in jamos:
176
- if j in VOCAB_TO_ID:
177
- tokens.append(VOCAB_TO_ID[j])
178
- else: # en
179
- for char in part.lower():
180
- if char in VOCAB_TO_ID:
181
- tokens.append(VOCAB_TO_ID[char])
182
-
183
- return tokens
184
-
1
+ """
2
+ Grapheme-to-Phoneme (G2P) Phonetic Tokenizer for Korean and English.
3
+ Includes Number-to-Speech Normalizer and Inline Expressive Tag Parser.
4
+ """
5
+
6
+ import re
7
+ from typing import List, Dict
8
+ from .exceptions import TTSLanguageNotSupportedError
9
+ from .g2p_korean import korean_text_to_phonemes
10
+
11
+ HANGUL_BASE = 0xAC00
12
+ HANGUL_END = 0xD7A3
13
+
14
+ CHO = [
15
+ "ㄱ", "ㄲ", "ㄴ", "ㄷ", "ㄸ", "ㄹ", "ㅁ", "ㅂ", "ㅃ", "ㅅ",
16
+ "ㅆ", "ㅇ", "ㅈ", "ㅉ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
17
+ ]
18
+ JUNG = [
19
+ "ㅏ", "ㅐ", "ㅑ", "ㅒ", "ㅓ", "ㅔ", "ㅕ", "ㅖ", "ㅗ", "ㅘ",
20
+ "ㅙ", "ㅚ", "ㅛ", "ㅜ", "ㅝ", "ㅞ", "ㅟ", "ㅠ", "ㅡ", "ㅢ", "ㅣ"
21
+ ]
22
+ JONG = [
23
+ "", "ㄱ", "ㄲ", "ㄳ", "ㄴ", "ㄵ", "ㄶ", "ㄷ", "ㄹ", "ㄺ",
24
+ "ㄻ", "ㄼ", "ㄽ", "ㄾ", "ㄿ", "ㅀ", "ㅁ", "ㅂ", "ㅄ", "ㅅ",
25
+ "ㅆ", "ㅇ", "ㅈ", "ㅊ", "ㅋ", "ㅌ", "ㅍ", "ㅎ"
26
+ ]
27
+
28
+ EXPRESSIVE_TAGS = {
29
+ "[laugh]": 1001,
30
+ "[sigh]": 1002,
31
+ "[breath]": 1003,
32
+ "[uv_break]": 1004,
33
+ "[clears_throat]": 1005,
34
+ "[pause]": 1006
35
+ }
36
+
37
+ VOCAB: List[str] = [
38
+ "_", " ", "!", "?", ",", ".", "~", "-",
39
+ *CHO, *JUNG, *[j for j in JONG if j],
40
+ *"abcdefghijklmnopqrstuvwxyz"
41
+ ]
42
+ VOCAB_TO_ID: Dict[str, int] = {sym: idx for idx, sym in enumerate(VOCAB)}
43
+ PAD_ID: int = VOCAB_TO_ID["_"]
44
+ SPACE_ID: int = VOCAB_TO_ID[" "]
45
+
46
+ KOREAN_DIGITS = ["", "일", "이", "삼", "사", "오", "육", "칠", "팔", "구"]
47
+ SMALL_UNITS = ["", "십", "백", "천"]
48
+ BIG_UNITS = ["", "만", "억", "조", "경"]
49
+
50
+ ENGLISH_WORDS = {
51
+ "0": "zero", "1": "one", "2": "two", "3": "three", "4": "four",
52
+ "5": "five", "6": "six", "7": "seven", "8": "eight", "9": "nine",
53
+ "10": "ten", "11": "eleven", "12": "twelve", "13": "thirteen", "14": "fourteen",
54
+ "15": "fifteen", "16": "sixteen", "17": "seventeen", "18": "eighteen", "19": "nineteen",
55
+ "20": "twenty", "30": "thirty", "40": "forty", "50": "fifty",
56
+ "60": "sixty", "70": "seventy", "80": "eighty", "90": "ninety",
57
+ "100": "hundred", "1000": "thousand"
58
+ }
59
+
60
+ def decompose_hangul(char: str) -> List[str]:
61
+ code = ord(char)
62
+ if HANGUL_BASE <= code <= HANGUL_END:
63
+ offset = code - HANGUL_BASE
64
+ cho_idx = offset // (21 * 28)
65
+ jung_idx = (offset % (21 * 28)) // 28
66
+ jong_idx = offset % 28
67
+ res = [CHO[cho_idx], JUNG[jung_idx]]
68
+ if jong_idx > 0:
69
+ res.append(JONG[jong_idx])
70
+ return res
71
+ return [char]
72
+
73
+ def _convert_4digits_korean(chunk: str) -> str:
74
+ num = int(chunk)
75
+ if num == 0:
76
+ return ""
77
+ res = []
78
+ str_num = str(num).zfill(4)
79
+ for i, ch in enumerate(str_num):
80
+ d = int(ch)
81
+ if d > 0:
82
+ unit = SMALL_UNITS[3 - i]
83
+ if d == 1 and unit != "":
84
+ res.append(unit)
85
+ else:
86
+ res.append(KOREAN_DIGITS[d] + unit)
87
+ return "".join(res)
88
+
89
+ def number_to_korean_sino(num_str: str) -> str:
90
+ """Convert integer string to authentic Sino-Korean place-value numerals (e.g. 1234 -> 천이백삼십사, 10000 -> 만)."""
91
+ try:
92
+ n = int(num_str)
93
+ except ValueError:
94
+ return num_str
95
+ if n == 0:
96
+ return "영"
97
+ rev_str = str(n)[::-1]
98
+ chunks = [rev_str[i:i+4][::-1] for i in range(0, len(rev_str), 4)]
99
+ parts = []
100
+ for i, chunk in enumerate(chunks):
101
+ c_korean = _convert_4digits_korean(chunk)
102
+ if c_korean:
103
+ unit = BIG_UNITS[i]
104
+ # 10000일 때 '일만' 대신 '만' (단, 210000 -> 이십일만)
105
+ if c_korean == "일" and unit != "" and len(chunks) == i + 1:
106
+ parts.append(unit)
107
+ else:
108
+ parts.append(c_korean + unit)
109
+ return "".join(reversed(parts))
110
+
111
+ def normalize_numbers_korean(text: str) -> str:
112
+ """Convert digit sequences into spoken Korean place-value numerals."""
113
+ return re.sub(r"\d+", lambda m: number_to_korean_sino(m.group(0)), text)
114
+
115
+ def normalize_numbers_english(text: str) -> str:
116
+ """Convert digit sequences into spoken English words."""
117
+ def _en_repl(m):
118
+ num_str = m.group(0)
119
+ if num_str in ENGLISH_WORDS:
120
+ return " " + ENGLISH_WORDS[num_str] + " "
121
+ # Digit-by-digit for phone numbers / codes
122
+ return " " + " ".join(ENGLISH_WORDS.get(d, d) for d in num_str) + " "
123
+ return re.sub(r"\d+", _en_repl, text)
124
+
125
+ class PhoneticTokenizer:
126
+ def __init__(self, language: str = "ko"):
127
+ self.language = language.lower()
128
+ self.vocab = VOCAB
129
+ if self.language not in ["ko", "korean", "en", "english"]:
130
+ raise TTSLanguageNotSupportedError(f"Language '{language}' is not supported. Supported: ['ko', 'en']")
131
+
132
+ def normalize_text(self, text: str) -> str:
133
+ if not text or not text.strip():
134
+ return ""
135
+ # 1. Normalize linebreaks and tabs
136
+ text = re.sub(r"[\r\n\t]+", " ", text)
137
+
138
+ # 2. Digits to spoken words
139
+ if self.language in ["ko", "korean"]:
140
+ text = normalize_numbers_korean(text)
141
+ text = korean_text_to_phonemes(text)
142
+ else:
143
+ text = normalize_numbers_english(text)
144
+
145
+ # 3. Collapse multiple spaces
146
+ text = re.sub(r"\s{2,}", " ", text)
147
+ return text.strip()
148
+
149
+ def tokenize(self, text: str) -> List[int]:
150
+ normalized = self.normalize_text(text)
151
+ if not normalized:
152
+ return []
153
+
154
+ # Parse expressive tags first
155
+ tag_pattern = re.compile(r"(\[[a-zA-Z_]+\])")
156
+ parts = tag_pattern.split(normalized)
157
+
158
+ tokens: List[int] = []
159
+ for part in parts:
160
+ if not part:
161
+ continue
162
+ if part in EXPRESSIVE_TAGS:
163
+ tokens.append(EXPRESSIVE_TAGS[part])
164
+ continue
165
+
166
+ if self.language in ["ko", "korean"]:
167
+ for char in part:
168
+ if char == " ":
169
+ tokens.append(SPACE_ID)
170
+ elif char in [".", ",", "!", "?", "~", "-"]:
171
+ if char in VOCAB_TO_ID:
172
+ tokens.append(VOCAB_TO_ID[char])
173
+ else:
174
+ jamos = decompose_hangul(char)
175
+ for j in jamos:
176
+ if j in VOCAB_TO_ID:
177
+ tokens.append(VOCAB_TO_ID[j])
178
+ else: # en
179
+ for char in part.lower():
180
+ if char in VOCAB_TO_ID:
181
+ tokens.append(VOCAB_TO_ID[char])
182
+
183
+ return tokens
184
+