echoact 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echoact/__init__.py +3 -0
- echoact/__main__.py +117 -0
- echoact/app.py +315 -0
- echoact/audio/__init__.py +0 -0
- echoact/audio/devices.py +192 -0
- echoact/audio/player.py +611 -0
- echoact/audio/wav.py +854 -0
- echoact/config/__init__.py +0 -0
- echoact/config/budget.py +370 -0
- echoact/config/settings.py +1244 -0
- echoact/db/__init__.py +0 -0
- echoact/db/backup.py +2429 -0
- echoact/db/migrations.py +434 -0
- echoact/db/schema.sql +214 -0
- echoact/db/store.py +2062 -0
- echoact/diagnostics.py +902 -0
- echoact/domain.py +487 -0
- echoact/engine/__init__.py +0 -0
- echoact/engine/container.py +843 -0
- echoact/engine/protocol.py +241 -0
- echoact/engine/runtime.py +324 -0
- echoact/engine/supervisor.py +961 -0
- echoact/engine/worker.py +659 -0
- echoact/errors.py +281 -0
- echoact/instance.py +172 -0
- echoact/jobs/__init__.py +0 -0
- echoact/jobs/engine.py +776 -0
- echoact/jobs/request.py +300 -0
- echoact/mcp/__init__.py +0 -0
- echoact/mcp/__main__.py +50 -0
- echoact/mcp/client.py +202 -0
- echoact/mcp/config.py +112 -0
- echoact/mcp/server.py +340 -0
- echoact/models/__init__.py +0 -0
- echoact/models/catalog.py +273 -0
- echoact/models/manifest.py +278 -0
- echoact/models/registry.py +1551 -0
- echoact/paths.py +93 -0
- echoact/policy.py +189 -0
- echoact/security/__init__.py +0 -0
- echoact/security/credentials.py +930 -0
- echoact/security/ratelimit.py +534 -0
- echoact/service/__init__.py +20 -0
- echoact/service/app.py +182 -0
- echoact/service/deps.py +563 -0
- echoact/service/errors.py +241 -0
- echoact/service/routes.py +1125 -0
- echoact/service/schemas.py +509 -0
- echoact/service/server.py +270 -0
- echoact/text/__init__.py +0 -0
- echoact/text/language.py +44 -0
- echoact/text/loader.py +577 -0
- echoact/text/normalize.py +924 -0
- echoact/text/segment.py +499 -0
- echoact/text/sniff.py +1202 -0
- echoact/ui/__init__.py +0 -0
- echoact/ui/bridge.py +50 -0
- echoact/ui/controls.py +360 -0
- echoact/ui/credential_dialog.py +131 -0
- echoact/ui/fonts.py +94 -0
- echoact/ui/i18n.py +260 -0
- echoact/ui/icons.py +440 -0
- echoact/ui/library.py +1642 -0
- echoact/ui/licence.py +162 -0
- echoact/ui/main_window.py +1202 -0
- echoact/ui/mcp_setup.py +494 -0
- echoact/ui/models_view.py +1142 -0
- echoact/ui/notifications.py +202 -0
- echoact/ui/reading.py +494 -0
- echoact/ui/settings_view.py +2258 -0
- echoact/ui/status_view.py +1193 -0
- echoact/ui/theme.py +579 -0
- echoact/util/__init__.py +0 -0
- echoact/util/ids.py +62 -0
- echoact/util/logging.py +127 -0
- echoact-0.1.0.dist-info/METADATA +162 -0
- echoact-0.1.0.dist-info/RECORD +80 -0
- echoact-0.1.0.dist-info/WHEEL +4 -0
- echoact-0.1.0.dist-info/entry_points.txt +3 -0
- echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,924 @@
|
|
|
1
|
+
"""F-27's normalisation, carrying an alignment back to what the user typed.
|
|
2
|
+
|
|
3
|
+
The engine is given an expanded reading -- "1,234" becomes "천이백삼십사" or
|
|
4
|
+
"one thousand two hundred thirty-four" -- but every segment range the rest of
|
|
5
|
+
the product uses must still point at the original characters. So the
|
|
6
|
+
expansion is recorded as an *edit script*: an ordered list of
|
|
7
|
+
``Piece(source_range, produced_text)`` that tiles the source exactly. A
|
|
8
|
+
per-character index array was the obvious alternative and is worse in the two
|
|
9
|
+
ways that matter here: it cannot represent a span that produces nothing (an
|
|
10
|
+
emoji, a run of whitespace), and an off-by-one in it is invisible, whereas a
|
|
11
|
+
gap or an overlap between pieces is an assertion failure at construction.
|
|
12
|
+
|
|
13
|
+
Two consequences the callers depend on:
|
|
14
|
+
|
|
15
|
+
* Pieces are atomic. A number, a date, or an abbreviation is one piece, so a
|
|
16
|
+
segmenter that only ever cuts at a piece boundary cannot cut inside one --
|
|
17
|
+
which is half of F-81's "never inside a word, a number, or a grapheme
|
|
18
|
+
cluster".
|
|
19
|
+
* A piece whose produced text is empty still occupies its source range. A.3
|
|
20
|
+
decided emoji and decorative symbols are not sent for synthesis, because
|
|
21
|
+
A.5 measured the engine vocalising three emoji as 1.32 s of audio; F-27
|
|
22
|
+
requires their characters to stay in the source and the highlight to travel
|
|
23
|
+
across them. An empty piece is exactly that.
|
|
24
|
+
|
|
25
|
+
Expansion is deliberately conservative. A reading we are not sure of is
|
|
26
|
+
worse than no reading at all, because a wrong expansion changes what is
|
|
27
|
+
spoken while leaving the source text -- and therefore the user's ability to
|
|
28
|
+
notice -- untouched. So "1.2.3", "St.", and "3rd" are left exactly as typed.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import re
|
|
34
|
+
import unicodedata
|
|
35
|
+
from bisect import bisect_right
|
|
36
|
+
from collections.abc import Callable, Iterable
|
|
37
|
+
from dataclasses import dataclass, field
|
|
38
|
+
from typing import Final
|
|
39
|
+
|
|
40
|
+
from ..domain import TextRange
|
|
41
|
+
from .language import EN, KO
|
|
42
|
+
|
|
43
|
+
# ======================================================================
|
|
44
|
+
# Grapheme clusters (UAX #29, the subset this product can encounter)
|
|
45
|
+
# ======================================================================
|
|
46
|
+
#
|
|
47
|
+
# The `regex` package is not a dependency, and `str` indexing counts code
|
|
48
|
+
# points, so the cluster rules are implemented here. They are needed twice:
|
|
49
|
+
# a split must never fall inside a cluster (F-81), and an emoji sequence must
|
|
50
|
+
# be suppressed whole rather than leaving a stray joiner or skin-tone
|
|
51
|
+
# modifier behind for the engine to read.
|
|
52
|
+
|
|
53
|
+
_CR: Final = 1
|
|
54
|
+
_LF: Final = 2
|
|
55
|
+
_CONTROL: Final = 3
|
|
56
|
+
_EXTEND: Final = 4
|
|
57
|
+
_ZWJ: Final = 5
|
|
58
|
+
_RI: Final = 6
|
|
59
|
+
_PREPEND: Final = 7
|
|
60
|
+
_SPACINGMARK: Final = 8
|
|
61
|
+
_L: Final = 9
|
|
62
|
+
_V: Final = 10
|
|
63
|
+
_T: Final = 11
|
|
64
|
+
_LV: Final = 12
|
|
65
|
+
_LVT: Final = 13
|
|
66
|
+
_EXTPICT: Final = 14
|
|
67
|
+
_OTHER: Final = 0
|
|
68
|
+
|
|
69
|
+
#: Extended_Pictographic, approximated by block ranges. Unicode data files
|
|
70
|
+
#: are not shipped with the app, and the exact property is only needed to
|
|
71
|
+
#: decide "emoji-ish", so the blocks are enumerated instead.
|
|
72
|
+
_EXTPICT_RANGES: Final[tuple[tuple[int, int], ...]] = (
|
|
73
|
+
(0x00A9, 0x00A9),
|
|
74
|
+
(0x00AE, 0x00AE),
|
|
75
|
+
(0x203C, 0x203C),
|
|
76
|
+
(0x2049, 0x2049),
|
|
77
|
+
(0x2122, 0x2122),
|
|
78
|
+
(0x2139, 0x2139),
|
|
79
|
+
(0x2190, 0x21FF),
|
|
80
|
+
(0x2300, 0x23FF),
|
|
81
|
+
(0x24C2, 0x24C2),
|
|
82
|
+
(0x25A0, 0x25FF),
|
|
83
|
+
(0x2600, 0x27BF),
|
|
84
|
+
(0x2900, 0x297F),
|
|
85
|
+
(0x2934, 0x2935),
|
|
86
|
+
(0x2B00, 0x2BFF),
|
|
87
|
+
(0x3030, 0x3030),
|
|
88
|
+
(0x303D, 0x303D),
|
|
89
|
+
(0x3297, 0x3297),
|
|
90
|
+
(0x3299, 0x3299),
|
|
91
|
+
(0x1F000, 0x1FAFF),
|
|
92
|
+
(0x1FC00, 0x1FFFD),
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
#: Characters that are decorative in the F-27 sense: pictographs, dingbats,
|
|
96
|
+
#: the symbol blocks, and the bullets a pasted list carries. Currency signs
|
|
97
|
+
#: and mathematical operators are deliberately absent -- they carry meaning a
|
|
98
|
+
#: listener needs, and several of them are expanded by the rules below.
|
|
99
|
+
_DECORATIVE_RANGES: Final[tuple[tuple[int, int], ...]] = tuple(
|
|
100
|
+
sorted(
|
|
101
|
+
_EXTPICT_RANGES
|
|
102
|
+
+ (
|
|
103
|
+
(0x2022, 0x2023),
|
|
104
|
+
(0x2043, 0x2043),
|
|
105
|
+
(0x20D0, 0x20FF),
|
|
106
|
+
(0xFE00, 0xFE0F),
|
|
107
|
+
(0xE0100, 0xE01EF),
|
|
108
|
+
)
|
|
109
|
+
)
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
#: Attach to a preceding pictograph without starting a new cluster.
|
|
113
|
+
_EMOJI_MODIFIERS: Final[tuple[tuple[int, int], ...]] = (
|
|
114
|
+
(0x1F3FB, 0x1F3FF),
|
|
115
|
+
(0xFE00, 0xFE0F),
|
|
116
|
+
(0x20E3, 0x20E3),
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
_PREPEND_CODEPOINTS: Final[frozenset[int]] = frozenset(
|
|
120
|
+
{0x0600, 0x0601, 0x0602, 0x0603, 0x0604, 0x0605, 0x06DD, 0x070F, 0x0890,
|
|
121
|
+
0x0891, 0x08E2, 0x0D4E, 0x110BD, 0x110CD}
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _in_ranges(cp: int, ranges: tuple[tuple[int, int], ...]) -> bool:
|
|
126
|
+
"""Membership in an ascending, non-overlapping range table."""
|
|
127
|
+
for lo, hi in ranges:
|
|
128
|
+
if lo <= cp <= hi:
|
|
129
|
+
return True
|
|
130
|
+
if cp < lo:
|
|
131
|
+
break
|
|
132
|
+
return False
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
_CLASS_CACHE: dict[int, int] = {}
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _gb_class(ch: str) -> int:
|
|
139
|
+
cp = ord(ch)
|
|
140
|
+
cached = _CLASS_CACHE.get(cp)
|
|
141
|
+
if cached is not None:
|
|
142
|
+
return cached
|
|
143
|
+
_CLASS_CACHE[cp] = value = _compute_gb_class(cp, ch)
|
|
144
|
+
return value
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _compute_gb_class(cp: int, ch: str) -> int:
|
|
148
|
+
if cp == 0x0D:
|
|
149
|
+
return _CR
|
|
150
|
+
if cp == 0x0A:
|
|
151
|
+
return _LF
|
|
152
|
+
if cp == 0x200D:
|
|
153
|
+
return _ZWJ
|
|
154
|
+
if 0x1F1E6 <= cp <= 0x1F1FF:
|
|
155
|
+
return _RI
|
|
156
|
+
if 0x1100 <= cp <= 0x115F:
|
|
157
|
+
return _L
|
|
158
|
+
if 0x1160 <= cp <= 0x11A7:
|
|
159
|
+
return _V
|
|
160
|
+
if 0x11A8 <= cp <= 0x11FF:
|
|
161
|
+
return _T
|
|
162
|
+
if 0xAC00 <= cp <= 0xD7A3:
|
|
163
|
+
return _LV if (cp - 0xAC00) % 28 == 0 else _LVT
|
|
164
|
+
if _in_ranges(cp, _EMOJI_MODIFIERS):
|
|
165
|
+
return _EXTEND
|
|
166
|
+
if cp in _PREPEND_CODEPOINTS:
|
|
167
|
+
return _PREPEND
|
|
168
|
+
category = unicodedata.category(ch)
|
|
169
|
+
if category in ("Mn", "Me"):
|
|
170
|
+
return _EXTEND
|
|
171
|
+
if category == "Mc":
|
|
172
|
+
return _SPACINGMARK
|
|
173
|
+
if category in ("Cc", "Cf", "Zl", "Zp"):
|
|
174
|
+
return _CONTROL
|
|
175
|
+
if _in_ranges(cp, _EXTPICT_RANGES):
|
|
176
|
+
return _EXTPICT
|
|
177
|
+
return _OTHER
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def grapheme_boundaries(text: str) -> tuple[int, ...]:
|
|
181
|
+
"""Every index at which ``text`` may be cut, including 0 and ``len``."""
|
|
182
|
+
if not text:
|
|
183
|
+
return (0,)
|
|
184
|
+
out = [0]
|
|
185
|
+
chain = False # an ExtPict Extend* run is open
|
|
186
|
+
zwj_after_pict = False
|
|
187
|
+
ri_run = 0
|
|
188
|
+
prev = _gb_class(text[0])
|
|
189
|
+
_, chain, zwj_after_pict, ri_run = _advance(prev, chain, zwj_after_pict, ri_run)
|
|
190
|
+
for i in range(1, len(text)):
|
|
191
|
+
cur = _gb_class(text[i])
|
|
192
|
+
if _breaks(prev, cur, zwj_after_pict, ri_run):
|
|
193
|
+
out.append(i)
|
|
194
|
+
prev, chain, zwj_after_pict, ri_run = _advance(cur, chain, zwj_after_pict, ri_run)
|
|
195
|
+
out.append(len(text))
|
|
196
|
+
return tuple(out)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _advance(
|
|
200
|
+
cur: int, chain: bool, zwj_after_pict: bool, ri_run: int
|
|
201
|
+
) -> tuple[int, bool, bool, int]:
|
|
202
|
+
if cur == _EXTPICT:
|
|
203
|
+
chain, zwj_after_pict = True, False
|
|
204
|
+
elif cur == _EXTEND:
|
|
205
|
+
zwj_after_pict = False
|
|
206
|
+
elif cur == _ZWJ:
|
|
207
|
+
zwj_after_pict = chain
|
|
208
|
+
else:
|
|
209
|
+
chain, zwj_after_pict = False, False
|
|
210
|
+
ri_run = ri_run + 1 if cur == _RI else 0
|
|
211
|
+
return cur, chain, zwj_after_pict, ri_run
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _breaks(prev: int, cur: int, zwj_after_pict: bool, ri_run: int) -> bool:
|
|
215
|
+
if prev == _CR and cur == _LF:
|
|
216
|
+
return False
|
|
217
|
+
if prev in (_CR, _LF, _CONTROL) or cur in (_CR, _LF, _CONTROL):
|
|
218
|
+
return True
|
|
219
|
+
if cur in (_EXTEND, _ZWJ, _SPACINGMARK):
|
|
220
|
+
return False
|
|
221
|
+
if prev == _PREPEND:
|
|
222
|
+
return False
|
|
223
|
+
if prev == _L and cur in (_L, _V, _LV, _LVT):
|
|
224
|
+
return False
|
|
225
|
+
if prev in (_LV, _V) and cur in (_V, _T):
|
|
226
|
+
return False
|
|
227
|
+
if prev in (_LVT, _T) and cur == _T:
|
|
228
|
+
return False
|
|
229
|
+
if prev == _ZWJ and cur == _EXTPICT and zwj_after_pict:
|
|
230
|
+
return False
|
|
231
|
+
if prev == _RI and cur == _RI and ri_run % 2 == 1:
|
|
232
|
+
return False
|
|
233
|
+
return True
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def grapheme_clusters(text: str) -> list[str]:
|
|
237
|
+
bounds = grapheme_boundaries(text)
|
|
238
|
+
return [text[a:b] for a, b in zip(bounds, bounds[1:], strict=False)]
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
#: How far back ``is_grapheme_boundary`` re-derives the machine's state. A
|
|
242
|
+
#: cluster longer than this does not occur in text a person typed; the flag
|
|
243
|
+
#: sequences and ZWJ families that motivate the rules are well under it.
|
|
244
|
+
_RESTART_WINDOW: Final = 64
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def is_grapheme_boundary(text: str, index: int) -> bool:
|
|
248
|
+
"""Whether ``index`` splits ``text`` without cutting a cluster.
|
|
249
|
+
|
|
250
|
+
Scanning the whole string for one question costs O(n) per candidate split
|
|
251
|
+
and F-81 asks that question often on a 50,000-character document, so the
|
|
252
|
+
state machine is restarted from a nearby character that cannot be inside
|
|
253
|
+
a cluster instead.
|
|
254
|
+
"""
|
|
255
|
+
if index <= 0 or index >= len(text):
|
|
256
|
+
return True
|
|
257
|
+
start = max(0, index - _RESTART_WINDOW)
|
|
258
|
+
while start > 0 and _gb_class(text[start]) not in (_OTHER, _CONTROL, _CR, _LF):
|
|
259
|
+
start -= 1
|
|
260
|
+
return (index - start) in {b for b in grapheme_boundaries(text[start:index + 1])}
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
# ======================================================================
|
|
264
|
+
# Number readings
|
|
265
|
+
# ======================================================================
|
|
266
|
+
|
|
267
|
+
_EN_ONES: Final = (
|
|
268
|
+
"zero", "one", "two", "three", "four", "five", "six", "seven", "eight", "nine",
|
|
269
|
+
"ten", "eleven", "twelve", "thirteen", "fourteen", "fifteen", "sixteen",
|
|
270
|
+
"seventeen", "eighteen", "nineteen",
|
|
271
|
+
)
|
|
272
|
+
_EN_TENS: Final = (
|
|
273
|
+
"", "", "twenty", "thirty", "forty", "fifty", "sixty", "seventy", "eighty", "ninety",
|
|
274
|
+
)
|
|
275
|
+
_EN_SCALES: Final = ((10**12, "trillion"), (10**9, "billion"), (10**6, "million"), (1000, "thousand"))
|
|
276
|
+
_EN_ORDINALS: Final = {
|
|
277
|
+
"one": "first", "two": "second", "three": "third", "five": "fifth",
|
|
278
|
+
"eight": "eighth", "nine": "ninth", "twelve": "twelfth",
|
|
279
|
+
"twenty": "twentieth", "thirty": "thirtieth", "forty": "fortieth",
|
|
280
|
+
"fifty": "fiftieth", "sixty": "sixtieth", "seventy": "seventieth",
|
|
281
|
+
"eighty": "eightieth", "ninety": "ninetieth", "hundred": "hundredth",
|
|
282
|
+
"thousand": "thousandth",
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
_KO_SINO_DIGITS: Final = "영일이삼사오육칠팔구"
|
|
286
|
+
_KO_SMALL_UNITS: Final = ("", "십", "백", "천")
|
|
287
|
+
_KO_BIG_UNITS: Final = ("", "만", "억", "조", "경")
|
|
288
|
+
_KO_NATIVE_ONES: Final = (
|
|
289
|
+
"", "하나", "둘", "셋", "넷", "다섯", "여섯", "일곱", "여덟", "아홉",
|
|
290
|
+
)
|
|
291
|
+
_KO_NATIVE_ONES_ATTR: Final = (
|
|
292
|
+
"", "한", "두", "세", "네", "다섯", "여섯", "일곱", "여덟", "아홉",
|
|
293
|
+
)
|
|
294
|
+
_KO_NATIVE_TENS: Final = (
|
|
295
|
+
"", "열", "스물", "서른", "마흔", "쉰", "예순", "일흔", "여든", "아흔",
|
|
296
|
+
)
|
|
297
|
+
_KO_NATIVE_TENS_ATTR: Final = (
|
|
298
|
+
"", "열", "스무", "서른", "마흔", "쉰", "예순", "일흔", "여든", "아흔",
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def english_cardinal(n: int) -> str:
|
|
303
|
+
if n < 0:
|
|
304
|
+
return "minus " + english_cardinal(-n)
|
|
305
|
+
if n < 20:
|
|
306
|
+
return _EN_ONES[n]
|
|
307
|
+
if n < 100:
|
|
308
|
+
tens, ones = divmod(n, 10)
|
|
309
|
+
return _EN_TENS[tens] + (f"-{_EN_ONES[ones]}" if ones else "")
|
|
310
|
+
if n < 1000:
|
|
311
|
+
hundreds, rest = divmod(n, 100)
|
|
312
|
+
head = f"{_EN_ONES[hundreds]} hundred"
|
|
313
|
+
return f"{head} {english_cardinal(rest)}" if rest else head
|
|
314
|
+
for value, name in _EN_SCALES:
|
|
315
|
+
if n >= value:
|
|
316
|
+
count, rest = divmod(n, value)
|
|
317
|
+
head = f"{english_cardinal(count)} {name}"
|
|
318
|
+
return f"{head} {english_cardinal(rest)}" if rest else head
|
|
319
|
+
return str(n)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def english_ordinal(n: int) -> str:
|
|
323
|
+
words = english_cardinal(n)
|
|
324
|
+
head, sep, last = words.rpartition("-") if "-" in words.rsplit(" ", 1)[-1] else words.rpartition(" ")
|
|
325
|
+
ordinal = _EN_ORDINALS.get(last, last + "th")
|
|
326
|
+
return head + sep + ordinal
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def english_year(year: int) -> str:
|
|
330
|
+
"""The reading a listener expects for a year, not the plain cardinal.
|
|
331
|
+
|
|
332
|
+
1984 is "nineteen eighty-four" and 2026 is "twenty twenty-six"; only the
|
|
333
|
+
2000s are read as a cardinal, because "twenty oh five" is a style choice
|
|
334
|
+
and "two thousand five" is not.
|
|
335
|
+
"""
|
|
336
|
+
if 1100 <= year <= 1999 or 2010 <= year <= 2099:
|
|
337
|
+
high, low = divmod(year, 100)
|
|
338
|
+
if low == 0:
|
|
339
|
+
return f"{english_cardinal(high)} hundred"
|
|
340
|
+
if low < 10:
|
|
341
|
+
return f"{english_cardinal(high)} oh {english_cardinal(low)}"
|
|
342
|
+
return f"{english_cardinal(high)} {english_cardinal(low)}"
|
|
343
|
+
return english_cardinal(year)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def korean_sino(n: int) -> str:
|
|
347
|
+
"""Sino-Korean reading: 1234 -> 천이백삼십사."""
|
|
348
|
+
if n < 0:
|
|
349
|
+
return "마이너스 " + korean_sino(-n)
|
|
350
|
+
if n == 0:
|
|
351
|
+
return "영"
|
|
352
|
+
groups: list[int] = []
|
|
353
|
+
rest = n
|
|
354
|
+
while rest:
|
|
355
|
+
rest, group = divmod(rest, 10_000)
|
|
356
|
+
groups.append(group)
|
|
357
|
+
if len(groups) > len(_KO_BIG_UNITS):
|
|
358
|
+
return str(n)
|
|
359
|
+
parts: list[str] = []
|
|
360
|
+
for index in range(len(groups) - 1, -1, -1):
|
|
361
|
+
group = groups[index]
|
|
362
|
+
if not group:
|
|
363
|
+
continue
|
|
364
|
+
body = _korean_sino_group(group)
|
|
365
|
+
if index == 1 and group == 1:
|
|
366
|
+
body = "" # 10,000 is 만, never 일만
|
|
367
|
+
parts.append(body + _KO_BIG_UNITS[index])
|
|
368
|
+
return " ".join(parts)
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _korean_sino_group(group: int) -> str:
|
|
372
|
+
out: list[str] = []
|
|
373
|
+
for power in range(3, -1, -1):
|
|
374
|
+
digit = (group // 10**power) % 10
|
|
375
|
+
if not digit:
|
|
376
|
+
continue
|
|
377
|
+
if digit == 1 and power > 0:
|
|
378
|
+
out.append(_KO_SMALL_UNITS[power]) # 십, not 일십
|
|
379
|
+
else:
|
|
380
|
+
out.append(_KO_SINO_DIGITS[digit] + _KO_SMALL_UNITS[power])
|
|
381
|
+
return "".join(out)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def korean_native(n: int, *, attributive: bool = True) -> str | None:
|
|
385
|
+
"""Native-Korean reading, or ``None`` where the series does not reach.
|
|
386
|
+
|
|
387
|
+
Counters like 개 and 시 take 하나/둘/셋, and before a counter those become
|
|
388
|
+
한/두/세 -- "3개" is "세 개", never "삼 개". The series is only used up to
|
|
389
|
+
99 because beyond that Korean itself switches to the Sino reading.
|
|
390
|
+
"""
|
|
391
|
+
if not 1 <= n <= 99:
|
|
392
|
+
return None
|
|
393
|
+
tens, ones = divmod(n, 10)
|
|
394
|
+
tens_table = _KO_NATIVE_TENS_ATTR if attributive and ones == 0 else _KO_NATIVE_TENS
|
|
395
|
+
ones_table = _KO_NATIVE_ONES_ATTR if attributive else _KO_NATIVE_ONES
|
|
396
|
+
return tens_table[tens] + ones_table[ones]
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def spoken_number(literal: str, lang: str) -> str:
|
|
400
|
+
"""Read a bare numeric literal such as ``1,234`` or ``3.14``."""
|
|
401
|
+
digits = literal.replace(",", "")
|
|
402
|
+
whole, _, fraction = digits.partition(".")
|
|
403
|
+
value = int(whole) if whole else 0
|
|
404
|
+
head = english_cardinal(value) if lang == EN else korean_sino(value)
|
|
405
|
+
if not fraction:
|
|
406
|
+
return head
|
|
407
|
+
if lang == EN:
|
|
408
|
+
tail = " ".join(_EN_ONES[int(d)] for d in fraction)
|
|
409
|
+
return f"{head} point {tail}"
|
|
410
|
+
tail = " ".join(_KO_SINO_DIGITS[int(d)] for d in fraction)
|
|
411
|
+
return f"{head} 점 {tail}"
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
# ======================================================================
|
|
415
|
+
# Rule tables
|
|
416
|
+
# ======================================================================
|
|
417
|
+
|
|
418
|
+
_MONTHS: Final = {
|
|
419
|
+
"jan": 1, "january": 1, "feb": 2, "february": 2, "mar": 3, "march": 3,
|
|
420
|
+
"apr": 4, "april": 4, "may": 5, "jun": 6, "june": 6, "jul": 7, "july": 7,
|
|
421
|
+
"aug": 8, "august": 8, "sep": 9, "sept": 9, "september": 9,
|
|
422
|
+
"oct": 10, "october": 10, "nov": 11, "november": 11, "dec": 12, "december": 12,
|
|
423
|
+
}
|
|
424
|
+
_MONTH_NAMES: Final = (
|
|
425
|
+
"", "January", "February", "March", "April", "May", "June",
|
|
426
|
+
"July", "August", "September", "October", "November", "December",
|
|
427
|
+
)
|
|
428
|
+
#: 6월 is 유월 and 10월 is 시월, not 육월 and 십월. The irregularity is the
|
|
429
|
+
#: whole reason months are not just "Sino number + 월".
|
|
430
|
+
_KO_MONTHS: Final = (
|
|
431
|
+
"", "일월", "이월", "삼월", "사월", "오월", "유월",
|
|
432
|
+
"칠월", "팔월", "구월", "시월", "십일월", "십이월",
|
|
433
|
+
)
|
|
434
|
+
|
|
435
|
+
# token -> (english singular, english plural, korean)
|
|
436
|
+
_UNITS: Final[dict[str, tuple[str, str, str]]] = {
|
|
437
|
+
"kHz": ("kilohertz", "kilohertz", "킬로헤르츠"),
|
|
438
|
+
"MHz": ("megahertz", "megahertz", "메가헤르츠"),
|
|
439
|
+
"GHz": ("gigahertz", "gigahertz", "기가헤르츠"),
|
|
440
|
+
"Hz": ("hertz", "hertz", "헤르츠"),
|
|
441
|
+
"km": ("kilometer", "kilometers", "킬로미터"),
|
|
442
|
+
"cm": ("centimeter", "centimeters", "센티미터"),
|
|
443
|
+
"mm": ("millimeter", "millimeters", "밀리미터"),
|
|
444
|
+
"kg": ("kilogram", "kilograms", "킬로그램"),
|
|
445
|
+
"mg": ("milligram", "milligrams", "밀리그램"),
|
|
446
|
+
"KB": ("kilobyte", "kilobytes", "킬로바이트"),
|
|
447
|
+
"MB": ("megabyte", "megabytes", "메가바이트"),
|
|
448
|
+
"GB": ("gigabyte", "gigabytes", "기가바이트"),
|
|
449
|
+
"TB": ("terabyte", "terabytes", "테라바이트"),
|
|
450
|
+
"ms": ("millisecond", "milliseconds", "밀리초"),
|
|
451
|
+
"°C": ("degree Celsius", "degrees Celsius", "도"),
|
|
452
|
+
"°F": ("degree Fahrenheit", "degrees Fahrenheit", "도"),
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
#: counter -> (native reading?, a space between number and counter?)
|
|
456
|
+
#: 번 is absent on purpose: "3번" is 삼번 for a bus and 세 번 for three times,
|
|
457
|
+
#: and nothing in the surrounding text settles which.
|
|
458
|
+
_KO_COUNTERS: Final[dict[str, tuple[bool, bool]]] = {
|
|
459
|
+
"시간": (True, True),
|
|
460
|
+
"개월": (False, False),
|
|
461
|
+
"주일": (False, False),
|
|
462
|
+
"학년": (False, False),
|
|
463
|
+
"인분": (False, True),
|
|
464
|
+
"개": (True, True),
|
|
465
|
+
"명": (True, True),
|
|
466
|
+
"사람": (True, True),
|
|
467
|
+
"살": (True, True),
|
|
468
|
+
"마리": (True, True),
|
|
469
|
+
"권": (True, True),
|
|
470
|
+
"대": (True, True),
|
|
471
|
+
"병": (True, True),
|
|
472
|
+
"잔": (True, True),
|
|
473
|
+
"그루": (True, True),
|
|
474
|
+
"가지": (True, True),
|
|
475
|
+
"켤레": (True, True),
|
|
476
|
+
"벌": (True, True),
|
|
477
|
+
"채": (True, True),
|
|
478
|
+
"시": (True, False),
|
|
479
|
+
"년": (False, False),
|
|
480
|
+
"월": (False, False),
|
|
481
|
+
"일": (False, False),
|
|
482
|
+
"원": (False, False),
|
|
483
|
+
"분": (False, False),
|
|
484
|
+
"초": (False, False),
|
|
485
|
+
"층": (False, False),
|
|
486
|
+
"주": (False, False),
|
|
487
|
+
"호": (False, False),
|
|
488
|
+
"도": (False, False),
|
|
489
|
+
"회": (False, True),
|
|
490
|
+
"세": (False, False),
|
|
491
|
+
"미터": (False, True),
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
# abbreviation -> (english reading, korean reading)
|
|
495
|
+
#
|
|
496
|
+
# "Ms.", "St.", and "No." are deliberately absent: their expansions are
|
|
497
|
+
# ambiguous (Saint or Street; Number or the Spanish negative), and F-27's
|
|
498
|
+
# risk is asymmetric -- leaving text alone is recoverable, speaking the wrong
|
|
499
|
+
# word is not.
|
|
500
|
+
_ABBREVIATIONS: Final[dict[str, tuple[str, str]]] = {
|
|
501
|
+
"Dr.": ("Doctor", "Doctor"),
|
|
502
|
+
"Mr.": ("Mister", "Mister"),
|
|
503
|
+
"Mrs.": ("Missus", "Missus"),
|
|
504
|
+
"Prof.": ("Professor", "Professor"),
|
|
505
|
+
"etc.": ("et cetera", "et cetera"),
|
|
506
|
+
"e.g.": ("for example", "for example"),
|
|
507
|
+
"i.e.": ("that is", "that is"),
|
|
508
|
+
"vs.": ("versus", "versus"),
|
|
509
|
+
# Spelled as letters rather than "in the morning": p.m. spans both
|
|
510
|
+
# afternoon and evening, so the wordy reading is wrong half the time.
|
|
511
|
+
"a.m.": ("A M", "오전"),
|
|
512
|
+
"p.m.": ("P M", "오후"),
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
#: Tokens whose trailing period never ends a sentence (F-81). Used by the
|
|
516
|
+
#: segmenter, which must agree with this table or "Dr. Kim" splits in two.
|
|
517
|
+
NON_TERMINAL_ABBREVIATIONS: Final[frozenset[str]] = frozenset(
|
|
518
|
+
{"Dr", "Mr", "Mrs", "Ms", "Prof", "St", "Jr", "Sr", "Fig", "No", "Inc",
|
|
519
|
+
"Ltd", "Co", "Corp", "Rev", "Gen", "Sgt", "Capt", "vs", "cf", "al",
|
|
520
|
+
"e.g", "i.e", "a.m", "p.m", "A.M", "P.M", "U.S", "Ph.D", "approx", "est"}
|
|
521
|
+
)
|
|
522
|
+
#: Tokens that may or may not end a sentence; the segmenter looks at what
|
|
523
|
+
#: follows.
|
|
524
|
+
AMBIGUOUS_ABBREVIATIONS: Final[frozenset[str]] = frozenset({"etc", "Ave", "Rd"})
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _char_class(ranges: tuple[tuple[int, int], ...]) -> str:
|
|
528
|
+
return "".join(
|
|
529
|
+
f"\\U{lo:08x}" if lo == hi else f"\\U{lo:08x}-\\U{hi:08x}" for lo, hi in ranges
|
|
530
|
+
)
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
_DECORATIVE_CLASS: Final = _char_class(_DECORATIVE_RANGES) + _char_class(
|
|
534
|
+
((0x1F1E6, 0x1F1FF), (0x1F3FB, 0x1F3FF), (0x20E3, 0x20E3), (0x200D, 0x200D))
|
|
535
|
+
)
|
|
536
|
+
|
|
537
|
+
_NUM: Final = r"\d{1,3}(?:,\d{3})+|\d+"
|
|
538
|
+
_DEC: Final = rf"(?:{_NUM})(?:\.\d+)?"
|
|
539
|
+
#: Nothing that would make the digits part of an identifier or a version.
|
|
540
|
+
#: Nothing may start a numeric match immediately after one of these. The
|
|
541
|
+
#: dashes are what keeps "010-1234-5678" a phone number: with them absent, a
|
|
542
|
+
#: rule could start on the second group and read half of it as a range.
|
|
543
|
+
_LEFT: Final = r"(?<![-–~\d.,A-Za-z_])"
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _alternation(tokens: Iterable[str]) -> str:
|
|
547
|
+
"""Longest token first, so 시간 wins over 시 and 개월 over 개."""
|
|
548
|
+
return "|".join(re.escape(t) for t in sorted(tokens, key=len, reverse=True))
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
#: Month names are matched capitalised only. Lower-case "may" is a common
|
|
552
|
+
#: English verb, and "I may 3 times" is not a date.
|
|
553
|
+
_MONTH_TOKENS: Final = tuple(sorted({n.capitalize() for n in _MONTHS}, key=len, reverse=True))
|
|
554
|
+
|
|
555
|
+
_RULES: Final[tuple[tuple[str, str], ...]] = (
|
|
556
|
+
("iso_date", r"(?<![\d-])\d{4}-\d{2}-\d{2}(?![\d-])"),
|
|
557
|
+
(
|
|
558
|
+
"en_date",
|
|
559
|
+
r"(?<![A-Za-z])(?:"
|
|
560
|
+
+ "|".join(_MONTH_TOKENS)
|
|
561
|
+
+ r")\.?\s+\d{1,2}(?:st|nd|rd|th)?(?:\s*,?\s*\d{4})?(?![A-Za-z\d])",
|
|
562
|
+
),
|
|
563
|
+
("currency", _LEFT + rf"[$₩]\s?(?:{_NUM})(?:\.\d{{1,2}})?(?![\d])"),
|
|
564
|
+
("percent", _LEFT + rf"(?:{_DEC})\s?%"),
|
|
565
|
+
("unit", _LEFT + rf"(?:{_DEC})\s?(?:{_alternation(_UNITS)})(?![A-Za-z])"),
|
|
566
|
+
("ko_counter", _LEFT + rf"(?:{_DEC})\s?(?:{_alternation(_KO_COUNTERS)})"),
|
|
567
|
+
(
|
|
568
|
+
"number_range",
|
|
569
|
+
# ``(?!\d)`` stops the second number backtracking to a shorter one to
|
|
570
|
+
# satisfy the "no third group" lookahead that follows it.
|
|
571
|
+
_LEFT + rf"(?:{_NUM})\s?[-–~]\s?(?:{_NUM})(?!\d)(?!\s?[-–~]\s?\d)(?![A-Za-z])",
|
|
572
|
+
),
|
|
573
|
+
("abbrev", r"(?<![A-Za-z.])(?:" + _alternation(_ABBREVIATIONS) + r")"),
|
|
574
|
+
# ``(?!,\d)`` keeps "1,2,3" whole: the group is not a thousands group, so
|
|
575
|
+
# reading only the "1" would speak a different list from the one written.
|
|
576
|
+
# ``(?![-–~]\d)`` does the same for "010-1234-5678", which the range rule
|
|
577
|
+
# above has already declined: a part of a phone number is not a number.
|
|
578
|
+
("decimal", _LEFT + rf"(?:{_NUM})\.\d+(?![\d.])(?!,\d)(?![-–~]\d)(?![A-Za-z])"),
|
|
579
|
+
("integer", _LEFT + rf"(?:{_NUM})(?![\d.]*\d)(?!,\d)(?![-–~]\d)(?![A-Za-z])"),
|
|
580
|
+
("decorative", rf"[{_DECORATIVE_CLASS}]+"),
|
|
581
|
+
("space", r"\s+"),
|
|
582
|
+
)
|
|
583
|
+
|
|
584
|
+
#: One alternation, scanned once. Rule order is priority order: at any
|
|
585
|
+
#: position the first rule that can match wins, which is why the specific
|
|
586
|
+
#: patterns (a date, a currency amount) precede the general ones (a decimal,
|
|
587
|
+
#: an integer).
|
|
588
|
+
_MASTER: Final = re.compile("|".join(f"(?P<{name}>{pattern})" for name, pattern in _RULES))
|
|
589
|
+
|
|
590
|
+
_RX_ISO: Final = re.compile(r"(\d{4})-(\d{2})-(\d{2})")
|
|
591
|
+
_RX_EN_DATE: Final = re.compile(
|
|
592
|
+
r"([A-Za-z]+)\.?\s+(\d{1,2})(?:st|nd|rd|th)?(?:\s*,?\s*(\d{4}))?", re.ASCII
|
|
593
|
+
)
|
|
594
|
+
_RX_CURRENCY: Final = re.compile(r"([$₩])\s?([\d,]+)(?:\.(\d{1,2}))?")
|
|
595
|
+
_RX_VALUE_TAIL: Final = re.compile(r"([\d,.]+)\s?(.+)", re.DOTALL)
|
|
596
|
+
_RX_RANGE: Final = re.compile(r"([\d,]+)\s?[-–~]\s?([\d,]+)")
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
# ======================================================================
|
|
600
|
+
# Rule handlers
|
|
601
|
+
# ======================================================================
|
|
602
|
+
|
|
603
|
+
|
|
604
|
+
def _h_iso_date(match: re.Match[str], lang: str) -> str:
|
|
605
|
+
parsed = _RX_ISO.fullmatch(match.group())
|
|
606
|
+
if parsed is None:
|
|
607
|
+
return match.group()
|
|
608
|
+
year, month, day = (int(g) for g in parsed.groups())
|
|
609
|
+
if not (1 <= month <= 12 and 1 <= day <= 31):
|
|
610
|
+
return match.group() # a part number or a range, not a date
|
|
611
|
+
return _spoken_date(year, month, day, lang)
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
def _h_en_date(match: re.Match[str], lang: str) -> str:
|
|
615
|
+
parsed = _RX_EN_DATE.fullmatch(match.group())
|
|
616
|
+
if parsed is None:
|
|
617
|
+
return match.group()
|
|
618
|
+
name, day_text, year_text = parsed.groups()
|
|
619
|
+
month = _MONTHS.get(name.lower())
|
|
620
|
+
day = int(day_text)
|
|
621
|
+
if month is None or not 1 <= day <= 31:
|
|
622
|
+
return match.group()
|
|
623
|
+
year = int(year_text) if year_text else None
|
|
624
|
+
if lang == KO:
|
|
625
|
+
head = "" if year is None else f"{korean_sino(year)}년 "
|
|
626
|
+
return f"{head}{_KO_MONTHS[month]} {korean_sino(day)}일"
|
|
627
|
+
tail = "" if year is None else f", {english_year(year)}"
|
|
628
|
+
return f"{_MONTH_NAMES[month]} {english_ordinal(day)}{tail}"
|
|
629
|
+
|
|
630
|
+
|
|
631
|
+
def _spoken_date(year: int, month: int, day: int, lang: str) -> str:
|
|
632
|
+
if lang == KO:
|
|
633
|
+
return f"{korean_sino(year)}년 {_KO_MONTHS[month]} {korean_sino(day)}일"
|
|
634
|
+
return f"{_MONTH_NAMES[month]} {english_ordinal(day)}, {english_year(year)}"
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
def _h_currency(match: re.Match[str], lang: str) -> str:
|
|
638
|
+
parsed = _RX_CURRENCY.fullmatch(match.group())
|
|
639
|
+
if parsed is None:
|
|
640
|
+
return match.group()
|
|
641
|
+
sign, whole_text, cents_text = parsed.groups()
|
|
642
|
+
whole = int(whole_text.replace(",", ""))
|
|
643
|
+
cents = int(cents_text.ljust(2, "0")) if cents_text else 0
|
|
644
|
+
if sign == "₩":
|
|
645
|
+
# The won has no everyday subunit, so a decimal here is a plain
|
|
646
|
+
# fractional amount rather than 100ths of a unit.
|
|
647
|
+
literal = whole_text if not cents_text else f"{whole_text}.{cents_text}"
|
|
648
|
+
reading = spoken_number(literal, lang)
|
|
649
|
+
return f"{reading} 원" if lang == KO else f"{reading} won"
|
|
650
|
+
say_whole = bool(whole) or not cents
|
|
651
|
+
if lang == KO:
|
|
652
|
+
head = f"{korean_sino(whole)} 달러" if say_whole else ""
|
|
653
|
+
tail = f"{korean_sino(cents)} 센트" if cents else ""
|
|
654
|
+
return " ".join(part for part in (head, tail) if part)
|
|
655
|
+
head = f"{english_cardinal(whole)} dollar{'' if whole == 1 else 's'}" if say_whole else ""
|
|
656
|
+
tail = f"{english_cardinal(cents)} cent{'' if cents == 1 else 's'}" if cents else ""
|
|
657
|
+
if head and tail:
|
|
658
|
+
return f"{head} and {tail}"
|
|
659
|
+
return head or tail
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
def _h_percent(match: re.Match[str], lang: str) -> str:
|
|
663
|
+
literal = match.group().rstrip("%").strip()
|
|
664
|
+
reading = spoken_number(literal, lang)
|
|
665
|
+
return f"{reading} 퍼센트" if lang == KO else f"{reading} percent"
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
def _h_unit(match: re.Match[str], lang: str) -> str:
|
|
669
|
+
parsed = _RX_VALUE_TAIL.fullmatch(match.group())
|
|
670
|
+
if parsed is None:
|
|
671
|
+
return match.group()
|
|
672
|
+
literal, token = parsed.group(1), parsed.group(2).strip()
|
|
673
|
+
singular, plural, korean = _UNITS[token]
|
|
674
|
+
reading = spoken_number(literal, lang)
|
|
675
|
+
if lang == KO:
|
|
676
|
+
return f"{reading} {korean}"
|
|
677
|
+
exact_one = literal.replace(",", "") in ("1", "1.0")
|
|
678
|
+
return f"{reading} {singular if exact_one else plural}"
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def _h_ko_counter(match: re.Match[str], lang: str) -> str:
|
|
682
|
+
"""Korean counters are read in Korean whatever the sentence's language.
|
|
683
|
+
|
|
684
|
+
The counter itself is a Korean word, so an English reading of the number
|
|
685
|
+
in front of it ("three 개") is not a reading anyone would want; F-05's
|
|
686
|
+
per-sentence language decides the voice, not the arithmetic.
|
|
687
|
+
"""
|
|
688
|
+
parsed = _RX_VALUE_TAIL.fullmatch(match.group())
|
|
689
|
+
if parsed is None:
|
|
690
|
+
return match.group()
|
|
691
|
+
literal, counter = parsed.group(1), parsed.group(2).strip()
|
|
692
|
+
native, spaced = _KO_COUNTERS[counter]
|
|
693
|
+
digits = literal.replace(",", "")
|
|
694
|
+
if counter == "월" and "." not in digits and 1 <= int(digits) <= 12:
|
|
695
|
+
return _KO_MONTHS[int(digits)]
|
|
696
|
+
reading: str | None = None
|
|
697
|
+
if native and "." not in digits:
|
|
698
|
+
reading = korean_native(int(digits))
|
|
699
|
+
if reading is None:
|
|
700
|
+
reading = spoken_number(literal, KO)
|
|
701
|
+
return f"{reading} {counter}" if spaced else f"{reading}{counter}"
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
def _h_number_range(match: re.Match[str], lang: str) -> str:
|
|
705
|
+
parsed = _RX_RANGE.fullmatch(match.group())
|
|
706
|
+
if parsed is None:
|
|
707
|
+
return match.group()
|
|
708
|
+
low, high = (spoken_number(g, lang) for g in parsed.groups())
|
|
709
|
+
return f"{low}에서 {high}" if lang == KO else f"{low} to {high}"
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
def _h_abbrev(match: re.Match[str], lang: str) -> str:
|
|
713
|
+
english, korean = _ABBREVIATIONS[match.group()]
|
|
714
|
+
return korean if lang == KO else english
|
|
715
|
+
|
|
716
|
+
|
|
717
|
+
def _h_number(match: re.Match[str], lang: str) -> str:
|
|
718
|
+
return spoken_number(match.group(), lang)
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
def _h_decorative(match: re.Match[str], lang: str) -> str:
|
|
722
|
+
"""A.3: not sent for synthesis, but the source range survives.
|
|
723
|
+
|
|
724
|
+
The run collapses to nothing, except between two non-space characters,
|
|
725
|
+
where it collapses to a single space -- otherwise "hello🎉world" would
|
|
726
|
+
reach the engine as one invented word.
|
|
727
|
+
"""
|
|
728
|
+
text = match.string
|
|
729
|
+
before = text[match.start() - 1] if match.start() else ""
|
|
730
|
+
after = text[match.end()] if match.end() < len(text) else ""
|
|
731
|
+
if before and after and not before.isspace() and not after.isspace():
|
|
732
|
+
return " "
|
|
733
|
+
return ""
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
def _h_space(match: re.Match[str], lang: str) -> str:
|
|
737
|
+
"""One space, whatever the run contained.
|
|
738
|
+
|
|
739
|
+
Line breaks must not reach the engine, and a paragraph break is expressed
|
|
740
|
+
as silence between segments (F-82, F-08) rather than as characters.
|
|
741
|
+
"""
|
|
742
|
+
return " "
|
|
743
|
+
|
|
744
|
+
|
|
745
|
+
_HANDLERS: Final[dict[str, Callable[[re.Match[str], str], str]]] = {
|
|
746
|
+
"iso_date": _h_iso_date,
|
|
747
|
+
"en_date": _h_en_date,
|
|
748
|
+
"currency": _h_currency,
|
|
749
|
+
"percent": _h_percent,
|
|
750
|
+
"unit": _h_unit,
|
|
751
|
+
"ko_counter": _h_ko_counter,
|
|
752
|
+
"number_range": _h_number_range,
|
|
753
|
+
"abbrev": _h_abbrev,
|
|
754
|
+
"decimal": _h_number,
|
|
755
|
+
"integer": _h_number,
|
|
756
|
+
"decorative": _h_decorative,
|
|
757
|
+
"space": _h_space,
|
|
758
|
+
}
|
|
759
|
+
|
|
760
|
+
|
|
761
|
+
# ======================================================================
|
|
762
|
+
# The alignment
|
|
763
|
+
# ======================================================================
|
|
764
|
+
|
|
765
|
+
|
|
766
|
+
@dataclass(frozen=True, slots=True)
|
|
767
|
+
class Piece:
|
|
768
|
+
"""One entry of the edit script: a source span and what it is read as."""
|
|
769
|
+
|
|
770
|
+
source: TextRange
|
|
771
|
+
text: str
|
|
772
|
+
|
|
773
|
+
@property
|
|
774
|
+
def is_identity(self) -> bool:
|
|
775
|
+
return len(self.text) == self.source.length
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
@dataclass(frozen=True, slots=True)
|
|
779
|
+
class Normalized:
|
|
780
|
+
"""Normalised text plus the alignment F-27 requires.
|
|
781
|
+
|
|
782
|
+
``pieces`` tiles ``source`` exactly: consecutive, no gaps, no overlaps,
|
|
783
|
+
covering ``[0, len(source))``. ``text`` is their concatenation. All
|
|
784
|
+
offsets are Unicode code points into the string that was normalised, and
|
|
785
|
+
a caller normalising a slice of a document adds that slice's own offset.
|
|
786
|
+
"""
|
|
787
|
+
|
|
788
|
+
source: str
|
|
789
|
+
text: str
|
|
790
|
+
pieces: tuple[Piece, ...]
|
|
791
|
+
# Index arrays for the two mappings, derived in __post_init__ (which is
|
|
792
|
+
# also where the tiling is verified: a gap or an overlap is a defect in a
|
|
793
|
+
# rule, and F-27 makes it one worth crashing on rather than shipping).
|
|
794
|
+
_source_starts: tuple[int, ...] = field(default=(), init=False, repr=False, compare=False)
|
|
795
|
+
_produced_starts: tuple[int, ...] = field(default=(), init=False, repr=False, compare=False)
|
|
796
|
+
|
|
797
|
+
def __post_init__(self) -> None:
|
|
798
|
+
cursor = 0
|
|
799
|
+
produced = 0
|
|
800
|
+
source_starts: list[int] = []
|
|
801
|
+
produced_starts: list[int] = []
|
|
802
|
+
for piece in self.pieces:
|
|
803
|
+
if piece.source.start != cursor:
|
|
804
|
+
raise AssertionError(
|
|
805
|
+
f"alignment is not a tiling at {piece.source.start} (expected {cursor})"
|
|
806
|
+
)
|
|
807
|
+
source_starts.append(cursor)
|
|
808
|
+
produced_starts.append(produced)
|
|
809
|
+
cursor = piece.source.end
|
|
810
|
+
produced += len(piece.text)
|
|
811
|
+
if cursor != len(self.source):
|
|
812
|
+
raise AssertionError(f"alignment covers {cursor} of {len(self.source)} code points")
|
|
813
|
+
if produced != len(self.text):
|
|
814
|
+
# Lengths, not contents: the offsets are all the mappings use,
|
|
815
|
+
# and rebuilding the joined string here would double the cost of
|
|
816
|
+
# normalising a 50,000-character document to catch nothing that
|
|
817
|
+
# this module can produce.
|
|
818
|
+
raise AssertionError("alignment pieces do not span the normalised text")
|
|
819
|
+
object.__setattr__(self, "_source_starts", tuple(source_starts))
|
|
820
|
+
object.__setattr__(self, "_produced_starts", tuple(produced_starts))
|
|
821
|
+
|
|
822
|
+
# -- mapping ------------------------------------------------------
|
|
823
|
+
|
|
824
|
+
def source_offset(self, produced_offset: int) -> int:
|
|
825
|
+
"""Where in the *source* a produced offset came from.
|
|
826
|
+
|
|
827
|
+
Inside a piece that changed length there is no character-level truth
|
|
828
|
+
to report, so the piece's own start is returned: a highlight may cover
|
|
829
|
+
one character too many, but it can never point outside the span that
|
|
830
|
+
produced the audio.
|
|
831
|
+
"""
|
|
832
|
+
if produced_offset <= 0:
|
|
833
|
+
return 0
|
|
834
|
+
if produced_offset >= len(self.text):
|
|
835
|
+
return len(self.source)
|
|
836
|
+
index = bisect_right(self._produced_starts, produced_offset) - 1
|
|
837
|
+
piece = self.pieces[index]
|
|
838
|
+
within = produced_offset - self._produced_starts[index]
|
|
839
|
+
if piece.is_identity:
|
|
840
|
+
return piece.source.start + within
|
|
841
|
+
return piece.source.start
|
|
842
|
+
|
|
843
|
+
def produced_offset(self, source_offset: int) -> int:
|
|
844
|
+
"""The inverse. Offsets inside a rewritten piece floor to its start."""
|
|
845
|
+
if source_offset <= 0:
|
|
846
|
+
return 0
|
|
847
|
+
if source_offset >= len(self.source):
|
|
848
|
+
return len(self.text)
|
|
849
|
+
index = bisect_right(self._source_starts, source_offset) - 1
|
|
850
|
+
piece = self.pieces[index]
|
|
851
|
+
within = source_offset - piece.source.start
|
|
852
|
+
if piece.is_identity:
|
|
853
|
+
return self._produced_starts[index] + within
|
|
854
|
+
return self._produced_starts[index]
|
|
855
|
+
|
|
856
|
+
def spoken_between(self, start: int, end: int) -> str:
|
|
857
|
+
"""The normalised text produced by source range ``[start, end)``."""
|
|
858
|
+
return self.text[self.produced_offset(start) : self.produced_offset(end)]
|
|
859
|
+
|
|
860
|
+
def source_range_of(self, produced: TextRange) -> TextRange:
|
|
861
|
+
return TextRange(self.source_offset(produced.start), self.source_offset(produced.end))
|
|
862
|
+
|
|
863
|
+
@property
|
|
864
|
+
def boundaries(self) -> tuple[int, ...]:
|
|
865
|
+
"""Source offsets a caller may cut at without splitting a piece."""
|
|
866
|
+
return self._source_starts + (len(self.source),)
|
|
867
|
+
|
|
868
|
+
|
|
869
|
+
def normalize(text: str, lang: str = EN) -> Normalized:
|
|
870
|
+
"""Expand numbers, dates, currency, and abbreviations for one language.
|
|
871
|
+
|
|
872
|
+
F-27. ``lang`` is the engine language F-05 resolved for this sentence;
|
|
873
|
+
it selects the reading, never which characters exist.
|
|
874
|
+
"""
|
|
875
|
+
pieces: list[Piece] = []
|
|
876
|
+
out: list[str] = []
|
|
877
|
+
cursor = 0
|
|
878
|
+
for match in _MASTER.finditer(text):
|
|
879
|
+
start, end = match.span()
|
|
880
|
+
if start > cursor:
|
|
881
|
+
_append(pieces, out, cursor, start, text[cursor:start])
|
|
882
|
+
# Exactly one rule group can participate in a match, so the last
|
|
883
|
+
# named group that matched is the rule that fired.
|
|
884
|
+
name = match.lastgroup or ""
|
|
885
|
+
_append(pieces, out, start, end, _HANDLERS[name](match, lang))
|
|
886
|
+
cursor = end
|
|
887
|
+
if cursor < len(text):
|
|
888
|
+
_append(pieces, out, cursor, len(text), text[cursor:])
|
|
889
|
+
return Normalized(source=text, text="".join(out), pieces=tuple(pieces))
|
|
890
|
+
|
|
891
|
+
|
|
892
|
+
def _append(pieces: list[Piece], out: list[str], start: int, end: int, produced: str) -> None:
|
|
893
|
+
pieces.append(Piece(TextRange(start, end), produced))
|
|
894
|
+
out.append(produced)
|
|
895
|
+
|
|
896
|
+
|
|
897
|
+
def is_decorative(ch: str) -> bool:
|
|
898
|
+
"""Whether one code point is emoji or decorative, per A.3."""
|
|
899
|
+
cp = ord(ch)
|
|
900
|
+
return (
|
|
901
|
+
_in_ranges(cp, _DECORATIVE_RANGES)
|
|
902
|
+
or 0x1F1E6 <= cp <= 0x1F1FF
|
|
903
|
+
or 0x1F3FB <= cp <= 0x1F3FF
|
|
904
|
+
or cp in (0x200D, 0x20E3)
|
|
905
|
+
)
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
__all__ = [
|
|
909
|
+
"AMBIGUOUS_ABBREVIATIONS",
|
|
910
|
+
"NON_TERMINAL_ABBREVIATIONS",
|
|
911
|
+
"Normalized",
|
|
912
|
+
"Piece",
|
|
913
|
+
"english_cardinal",
|
|
914
|
+
"english_ordinal",
|
|
915
|
+
"english_year",
|
|
916
|
+
"grapheme_boundaries",
|
|
917
|
+
"grapheme_clusters",
|
|
918
|
+
"is_decorative",
|
|
919
|
+
"is_grapheme_boundary",
|
|
920
|
+
"korean_native",
|
|
921
|
+
"korean_sino",
|
|
922
|
+
"normalize",
|
|
923
|
+
"spoken_number",
|
|
924
|
+
]
|