echoact 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. echoact/__init__.py +3 -0
  2. echoact/__main__.py +117 -0
  3. echoact/app.py +315 -0
  4. echoact/audio/__init__.py +0 -0
  5. echoact/audio/devices.py +192 -0
  6. echoact/audio/player.py +611 -0
  7. echoact/audio/wav.py +854 -0
  8. echoact/config/__init__.py +0 -0
  9. echoact/config/budget.py +370 -0
  10. echoact/config/settings.py +1244 -0
  11. echoact/db/__init__.py +0 -0
  12. echoact/db/backup.py +2429 -0
  13. echoact/db/migrations.py +434 -0
  14. echoact/db/schema.sql +214 -0
  15. echoact/db/store.py +2062 -0
  16. echoact/diagnostics.py +902 -0
  17. echoact/domain.py +487 -0
  18. echoact/engine/__init__.py +0 -0
  19. echoact/engine/container.py +843 -0
  20. echoact/engine/protocol.py +241 -0
  21. echoact/engine/runtime.py +324 -0
  22. echoact/engine/supervisor.py +961 -0
  23. echoact/engine/worker.py +659 -0
  24. echoact/errors.py +281 -0
  25. echoact/instance.py +172 -0
  26. echoact/jobs/__init__.py +0 -0
  27. echoact/jobs/engine.py +776 -0
  28. echoact/jobs/request.py +300 -0
  29. echoact/mcp/__init__.py +0 -0
  30. echoact/mcp/__main__.py +50 -0
  31. echoact/mcp/client.py +202 -0
  32. echoact/mcp/config.py +112 -0
  33. echoact/mcp/server.py +340 -0
  34. echoact/models/__init__.py +0 -0
  35. echoact/models/catalog.py +273 -0
  36. echoact/models/manifest.py +278 -0
  37. echoact/models/registry.py +1551 -0
  38. echoact/paths.py +93 -0
  39. echoact/policy.py +189 -0
  40. echoact/security/__init__.py +0 -0
  41. echoact/security/credentials.py +930 -0
  42. echoact/security/ratelimit.py +534 -0
  43. echoact/service/__init__.py +20 -0
  44. echoact/service/app.py +182 -0
  45. echoact/service/deps.py +563 -0
  46. echoact/service/errors.py +241 -0
  47. echoact/service/routes.py +1125 -0
  48. echoact/service/schemas.py +509 -0
  49. echoact/service/server.py +270 -0
  50. echoact/text/__init__.py +0 -0
  51. echoact/text/language.py +44 -0
  52. echoact/text/loader.py +577 -0
  53. echoact/text/normalize.py +924 -0
  54. echoact/text/segment.py +499 -0
  55. echoact/text/sniff.py +1202 -0
  56. echoact/ui/__init__.py +0 -0
  57. echoact/ui/bridge.py +50 -0
  58. echoact/ui/controls.py +360 -0
  59. echoact/ui/credential_dialog.py +131 -0
  60. echoact/ui/fonts.py +94 -0
  61. echoact/ui/i18n.py +260 -0
  62. echoact/ui/icons.py +440 -0
  63. echoact/ui/library.py +1642 -0
  64. echoact/ui/licence.py +162 -0
  65. echoact/ui/main_window.py +1202 -0
  66. echoact/ui/mcp_setup.py +494 -0
  67. echoact/ui/models_view.py +1142 -0
  68. echoact/ui/notifications.py +202 -0
  69. echoact/ui/reading.py +494 -0
  70. echoact/ui/settings_view.py +2258 -0
  71. echoact/ui/status_view.py +1193 -0
  72. echoact/ui/theme.py +579 -0
  73. echoact/util/__init__.py +0 -0
  74. echoact/util/ids.py +62 -0
  75. echoact/util/logging.py +127 -0
  76. echoact-0.1.0.dist-info/METADATA +162 -0
  77. echoact-0.1.0.dist-info/RECORD +80 -0
  78. echoact-0.1.0.dist-info/WHEEL +4 -0
  79. echoact-0.1.0.dist-info/entry_points.txt +3 -0
  80. echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,924 @@
1
+ """F-27's normalisation, carrying an alignment back to what the user typed.
2
+
3
+ The engine is given an expanded reading -- "1,234" becomes "천이백삼십사" or
4
+ "one thousand two hundred thirty-four" -- but every segment range the rest of
5
+ the product uses must still point at the original characters. So the
6
+ expansion is recorded as an *edit script*: an ordered list of
7
+ ``Piece(source_range, produced_text)`` that tiles the source exactly. A
8
+ per-character index array was the obvious alternative and is worse in the two
9
+ ways that matter here: it cannot represent a span that produces nothing (an
10
+ emoji, a run of whitespace), and an off-by-one in it is invisible, whereas a
11
+ gap or an overlap between pieces is an assertion failure at construction.
12
+
13
+ Two consequences the callers depend on:
14
+
15
+ * Pieces are atomic. A number, a date, or an abbreviation is one piece, so a
16
+ segmenter that only ever cuts at a piece boundary cannot cut inside one --
17
+ which is half of F-81's "never inside a word, a number, or a grapheme
18
+ cluster".
19
+ * A piece whose produced text is empty still occupies its source range. A.3
20
+ decided emoji and decorative symbols are not sent for synthesis, because
21
+ A.5 measured the engine vocalising three emoji as 1.32 s of audio; F-27
22
+ requires their characters to stay in the source and the highlight to travel
23
+ across them. An empty piece is exactly that.
24
+
25
+ Expansion is deliberately conservative. A reading we are not sure of is
26
+ worse than no reading at all, because a wrong expansion changes what is
27
+ spoken while leaving the source text -- and therefore the user's ability to
28
+ notice -- untouched. So "1.2.3", "St.", and "3rd" are left exactly as typed.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import re
34
+ import unicodedata
35
+ from bisect import bisect_right
36
+ from collections.abc import Callable, Iterable
37
+ from dataclasses import dataclass, field
38
+ from typing import Final
39
+
40
+ from ..domain import TextRange
41
+ from .language import EN, KO
42
+
43
+ # ======================================================================
44
+ # Grapheme clusters (UAX #29, the subset this product can encounter)
45
+ # ======================================================================
46
+ #
47
+ # The `regex` package is not a dependency, and `str` indexing counts code
48
+ # points, so the cluster rules are implemented here. They are needed twice:
49
+ # a split must never fall inside a cluster (F-81), and an emoji sequence must
50
+ # be suppressed whole rather than leaving a stray joiner or skin-tone
51
+ # modifier behind for the engine to read.
52
+
53
+ _CR: Final = 1
54
+ _LF: Final = 2
55
+ _CONTROL: Final = 3
56
+ _EXTEND: Final = 4
57
+ _ZWJ: Final = 5
58
+ _RI: Final = 6
59
+ _PREPEND: Final = 7
60
+ _SPACINGMARK: Final = 8
61
+ _L: Final = 9
62
+ _V: Final = 10
63
+ _T: Final = 11
64
+ _LV: Final = 12
65
+ _LVT: Final = 13
66
+ _EXTPICT: Final = 14
67
+ _OTHER: Final = 0
68
+
69
+ #: Extended_Pictographic, approximated by block ranges. Unicode data files
70
+ #: are not shipped with the app, and the exact property is only needed to
71
+ #: decide "emoji-ish", so the blocks are enumerated instead.
72
+ _EXTPICT_RANGES: Final[tuple[tuple[int, int], ...]] = (
73
+ (0x00A9, 0x00A9),
74
+ (0x00AE, 0x00AE),
75
+ (0x203C, 0x203C),
76
+ (0x2049, 0x2049),
77
+ (0x2122, 0x2122),
78
+ (0x2139, 0x2139),
79
+ (0x2190, 0x21FF),
80
+ (0x2300, 0x23FF),
81
+ (0x24C2, 0x24C2),
82
+ (0x25A0, 0x25FF),
83
+ (0x2600, 0x27BF),
84
+ (0x2900, 0x297F),
85
+ (0x2934, 0x2935),
86
+ (0x2B00, 0x2BFF),
87
+ (0x3030, 0x3030),
88
+ (0x303D, 0x303D),
89
+ (0x3297, 0x3297),
90
+ (0x3299, 0x3299),
91
+ (0x1F000, 0x1FAFF),
92
+ (0x1FC00, 0x1FFFD),
93
+ )
94
+
95
+ #: Characters that are decorative in the F-27 sense: pictographs, dingbats,
96
+ #: the symbol blocks, and the bullets a pasted list carries. Currency signs
97
+ #: and mathematical operators are deliberately absent -- they carry meaning a
98
+ #: listener needs, and several of them are expanded by the rules below.
99
+ _DECORATIVE_RANGES: Final[tuple[tuple[int, int], ...]] = tuple(
100
+ sorted(
101
+ _EXTPICT_RANGES
102
+ + (
103
+ (0x2022, 0x2023),
104
+ (0x2043, 0x2043),
105
+ (0x20D0, 0x20FF),
106
+ (0xFE00, 0xFE0F),
107
+ (0xE0100, 0xE01EF),
108
+ )
109
+ )
110
+ )
111
+
112
+ #: Attach to a preceding pictograph without starting a new cluster.
113
+ _EMOJI_MODIFIERS: Final[tuple[tuple[int, int], ...]] = (
114
+ (0x1F3FB, 0x1F3FF),
115
+ (0xFE00, 0xFE0F),
116
+ (0x20E3, 0x20E3),
117
+ )
118
+
119
+ _PREPEND_CODEPOINTS: Final[frozenset[int]] = frozenset(
120
+ {0x0600, 0x0601, 0x0602, 0x0603, 0x0604, 0x0605, 0x06DD, 0x070F, 0x0890,
121
+ 0x0891, 0x08E2, 0x0D4E, 0x110BD, 0x110CD}
122
+ )
123
+
124
+
125
+ def _in_ranges(cp: int, ranges: tuple[tuple[int, int], ...]) -> bool:
126
+ """Membership in an ascending, non-overlapping range table."""
127
+ for lo, hi in ranges:
128
+ if lo <= cp <= hi:
129
+ return True
130
+ if cp < lo:
131
+ break
132
+ return False
133
+
134
+
135
+ _CLASS_CACHE: dict[int, int] = {}
136
+
137
+
138
+ def _gb_class(ch: str) -> int:
139
+ cp = ord(ch)
140
+ cached = _CLASS_CACHE.get(cp)
141
+ if cached is not None:
142
+ return cached
143
+ _CLASS_CACHE[cp] = value = _compute_gb_class(cp, ch)
144
+ return value
145
+
146
+
147
+ def _compute_gb_class(cp: int, ch: str) -> int:
148
+ if cp == 0x0D:
149
+ return _CR
150
+ if cp == 0x0A:
151
+ return _LF
152
+ if cp == 0x200D:
153
+ return _ZWJ
154
+ if 0x1F1E6 <= cp <= 0x1F1FF:
155
+ return _RI
156
+ if 0x1100 <= cp <= 0x115F:
157
+ return _L
158
+ if 0x1160 <= cp <= 0x11A7:
159
+ return _V
160
+ if 0x11A8 <= cp <= 0x11FF:
161
+ return _T
162
+ if 0xAC00 <= cp <= 0xD7A3:
163
+ return _LV if (cp - 0xAC00) % 28 == 0 else _LVT
164
+ if _in_ranges(cp, _EMOJI_MODIFIERS):
165
+ return _EXTEND
166
+ if cp in _PREPEND_CODEPOINTS:
167
+ return _PREPEND
168
+ category = unicodedata.category(ch)
169
+ if category in ("Mn", "Me"):
170
+ return _EXTEND
171
+ if category == "Mc":
172
+ return _SPACINGMARK
173
+ if category in ("Cc", "Cf", "Zl", "Zp"):
174
+ return _CONTROL
175
+ if _in_ranges(cp, _EXTPICT_RANGES):
176
+ return _EXTPICT
177
+ return _OTHER
178
+
179
+
180
+ def grapheme_boundaries(text: str) -> tuple[int, ...]:
181
+ """Every index at which ``text`` may be cut, including 0 and ``len``."""
182
+ if not text:
183
+ return (0,)
184
+ out = [0]
185
+ chain = False # an ExtPict Extend* run is open
186
+ zwj_after_pict = False
187
+ ri_run = 0
188
+ prev = _gb_class(text[0])
189
+ _, chain, zwj_after_pict, ri_run = _advance(prev, chain, zwj_after_pict, ri_run)
190
+ for i in range(1, len(text)):
191
+ cur = _gb_class(text[i])
192
+ if _breaks(prev, cur, zwj_after_pict, ri_run):
193
+ out.append(i)
194
+ prev, chain, zwj_after_pict, ri_run = _advance(cur, chain, zwj_after_pict, ri_run)
195
+ out.append(len(text))
196
+ return tuple(out)
197
+
198
+
199
+ def _advance(
200
+ cur: int, chain: bool, zwj_after_pict: bool, ri_run: int
201
+ ) -> tuple[int, bool, bool, int]:
202
+ if cur == _EXTPICT:
203
+ chain, zwj_after_pict = True, False
204
+ elif cur == _EXTEND:
205
+ zwj_after_pict = False
206
+ elif cur == _ZWJ:
207
+ zwj_after_pict = chain
208
+ else:
209
+ chain, zwj_after_pict = False, False
210
+ ri_run = ri_run + 1 if cur == _RI else 0
211
+ return cur, chain, zwj_after_pict, ri_run
212
+
213
+
214
+ def _breaks(prev: int, cur: int, zwj_after_pict: bool, ri_run: int) -> bool:
215
+ if prev == _CR and cur == _LF:
216
+ return False
217
+ if prev in (_CR, _LF, _CONTROL) or cur in (_CR, _LF, _CONTROL):
218
+ return True
219
+ if cur in (_EXTEND, _ZWJ, _SPACINGMARK):
220
+ return False
221
+ if prev == _PREPEND:
222
+ return False
223
+ if prev == _L and cur in (_L, _V, _LV, _LVT):
224
+ return False
225
+ if prev in (_LV, _V) and cur in (_V, _T):
226
+ return False
227
+ if prev in (_LVT, _T) and cur == _T:
228
+ return False
229
+ if prev == _ZWJ and cur == _EXTPICT and zwj_after_pict:
230
+ return False
231
+ if prev == _RI and cur == _RI and ri_run % 2 == 1:
232
+ return False
233
+ return True
234
+
235
+
236
+ def grapheme_clusters(text: str) -> list[str]:
237
+ bounds = grapheme_boundaries(text)
238
+ return [text[a:b] for a, b in zip(bounds, bounds[1:], strict=False)]
239
+
240
+
241
+ #: How far back ``is_grapheme_boundary`` re-derives the machine's state. A
242
+ #: cluster longer than this does not occur in text a person typed; the flag
243
+ #: sequences and ZWJ families that motivate the rules are well under it.
244
+ _RESTART_WINDOW: Final = 64
245
+
246
+
247
+ def is_grapheme_boundary(text: str, index: int) -> bool:
248
+ """Whether ``index`` splits ``text`` without cutting a cluster.
249
+
250
+ Scanning the whole string for one question costs O(n) per candidate split
251
+ and F-81 asks that question often on a 50,000-character document, so the
252
+ state machine is restarted from a nearby character that cannot be inside
253
+ a cluster instead.
254
+ """
255
+ if index <= 0 or index >= len(text):
256
+ return True
257
+ start = max(0, index - _RESTART_WINDOW)
258
+ while start > 0 and _gb_class(text[start]) not in (_OTHER, _CONTROL, _CR, _LF):
259
+ start -= 1
260
+ return (index - start) in {b for b in grapheme_boundaries(text[start:index + 1])}
261
+
262
+
263
+ # ======================================================================
264
+ # Number readings
265
+ # ======================================================================
266
+
267
+ _EN_ONES: Final = (
268
+ "zero", "one", "two", "three", "four", "five", "six", "seven", "eight", "nine",
269
+ "ten", "eleven", "twelve", "thirteen", "fourteen", "fifteen", "sixteen",
270
+ "seventeen", "eighteen", "nineteen",
271
+ )
272
+ _EN_TENS: Final = (
273
+ "", "", "twenty", "thirty", "forty", "fifty", "sixty", "seventy", "eighty", "ninety",
274
+ )
275
+ _EN_SCALES: Final = ((10**12, "trillion"), (10**9, "billion"), (10**6, "million"), (1000, "thousand"))
276
+ _EN_ORDINALS: Final = {
277
+ "one": "first", "two": "second", "three": "third", "five": "fifth",
278
+ "eight": "eighth", "nine": "ninth", "twelve": "twelfth",
279
+ "twenty": "twentieth", "thirty": "thirtieth", "forty": "fortieth",
280
+ "fifty": "fiftieth", "sixty": "sixtieth", "seventy": "seventieth",
281
+ "eighty": "eightieth", "ninety": "ninetieth", "hundred": "hundredth",
282
+ "thousand": "thousandth",
283
+ }
284
+
285
+ _KO_SINO_DIGITS: Final = "영일이삼사오육칠팔구"
286
+ _KO_SMALL_UNITS: Final = ("", "십", "백", "천")
287
+ _KO_BIG_UNITS: Final = ("", "만", "억", "조", "경")
288
+ _KO_NATIVE_ONES: Final = (
289
+ "", "하나", "둘", "셋", "넷", "다섯", "여섯", "일곱", "여덟", "아홉",
290
+ )
291
+ _KO_NATIVE_ONES_ATTR: Final = (
292
+ "", "한", "두", "세", "네", "다섯", "여섯", "일곱", "여덟", "아홉",
293
+ )
294
+ _KO_NATIVE_TENS: Final = (
295
+ "", "열", "스물", "서른", "마흔", "쉰", "예순", "일흔", "여든", "아흔",
296
+ )
297
+ _KO_NATIVE_TENS_ATTR: Final = (
298
+ "", "열", "스무", "서른", "마흔", "쉰", "예순", "일흔", "여든", "아흔",
299
+ )
300
+
301
+
302
+ def english_cardinal(n: int) -> str:
303
+ if n < 0:
304
+ return "minus " + english_cardinal(-n)
305
+ if n < 20:
306
+ return _EN_ONES[n]
307
+ if n < 100:
308
+ tens, ones = divmod(n, 10)
309
+ return _EN_TENS[tens] + (f"-{_EN_ONES[ones]}" if ones else "")
310
+ if n < 1000:
311
+ hundreds, rest = divmod(n, 100)
312
+ head = f"{_EN_ONES[hundreds]} hundred"
313
+ return f"{head} {english_cardinal(rest)}" if rest else head
314
+ for value, name in _EN_SCALES:
315
+ if n >= value:
316
+ count, rest = divmod(n, value)
317
+ head = f"{english_cardinal(count)} {name}"
318
+ return f"{head} {english_cardinal(rest)}" if rest else head
319
+ return str(n)
320
+
321
+
322
+ def english_ordinal(n: int) -> str:
323
+ words = english_cardinal(n)
324
+ head, sep, last = words.rpartition("-") if "-" in words.rsplit(" ", 1)[-1] else words.rpartition(" ")
325
+ ordinal = _EN_ORDINALS.get(last, last + "th")
326
+ return head + sep + ordinal
327
+
328
+
329
+ def english_year(year: int) -> str:
330
+ """The reading a listener expects for a year, not the plain cardinal.
331
+
332
+ 1984 is "nineteen eighty-four" and 2026 is "twenty twenty-six"; only the
333
+ 2000s are read as a cardinal, because "twenty oh five" is a style choice
334
+ and "two thousand five" is not.
335
+ """
336
+ if 1100 <= year <= 1999 or 2010 <= year <= 2099:
337
+ high, low = divmod(year, 100)
338
+ if low == 0:
339
+ return f"{english_cardinal(high)} hundred"
340
+ if low < 10:
341
+ return f"{english_cardinal(high)} oh {english_cardinal(low)}"
342
+ return f"{english_cardinal(high)} {english_cardinal(low)}"
343
+ return english_cardinal(year)
344
+
345
+
346
+ def korean_sino(n: int) -> str:
347
+ """Sino-Korean reading: 1234 -> 천이백삼십사."""
348
+ if n < 0:
349
+ return "마이너스 " + korean_sino(-n)
350
+ if n == 0:
351
+ return "영"
352
+ groups: list[int] = []
353
+ rest = n
354
+ while rest:
355
+ rest, group = divmod(rest, 10_000)
356
+ groups.append(group)
357
+ if len(groups) > len(_KO_BIG_UNITS):
358
+ return str(n)
359
+ parts: list[str] = []
360
+ for index in range(len(groups) - 1, -1, -1):
361
+ group = groups[index]
362
+ if not group:
363
+ continue
364
+ body = _korean_sino_group(group)
365
+ if index == 1 and group == 1:
366
+ body = "" # 10,000 is 만, never 일만
367
+ parts.append(body + _KO_BIG_UNITS[index])
368
+ return " ".join(parts)
369
+
370
+
371
+ def _korean_sino_group(group: int) -> str:
372
+ out: list[str] = []
373
+ for power in range(3, -1, -1):
374
+ digit = (group // 10**power) % 10
375
+ if not digit:
376
+ continue
377
+ if digit == 1 and power > 0:
378
+ out.append(_KO_SMALL_UNITS[power]) # 십, not 일십
379
+ else:
380
+ out.append(_KO_SINO_DIGITS[digit] + _KO_SMALL_UNITS[power])
381
+ return "".join(out)
382
+
383
+
384
+ def korean_native(n: int, *, attributive: bool = True) -> str | None:
385
+ """Native-Korean reading, or ``None`` where the series does not reach.
386
+
387
+ Counters like 개 and 시 take 하나/둘/셋, and before a counter those become
388
+ 한/두/세 -- "3개" is "세 개", never "삼 개". The series is only used up to
389
+ 99 because beyond that Korean itself switches to the Sino reading.
390
+ """
391
+ if not 1 <= n <= 99:
392
+ return None
393
+ tens, ones = divmod(n, 10)
394
+ tens_table = _KO_NATIVE_TENS_ATTR if attributive and ones == 0 else _KO_NATIVE_TENS
395
+ ones_table = _KO_NATIVE_ONES_ATTR if attributive else _KO_NATIVE_ONES
396
+ return tens_table[tens] + ones_table[ones]
397
+
398
+
399
+ def spoken_number(literal: str, lang: str) -> str:
400
+ """Read a bare numeric literal such as ``1,234`` or ``3.14``."""
401
+ digits = literal.replace(",", "")
402
+ whole, _, fraction = digits.partition(".")
403
+ value = int(whole) if whole else 0
404
+ head = english_cardinal(value) if lang == EN else korean_sino(value)
405
+ if not fraction:
406
+ return head
407
+ if lang == EN:
408
+ tail = " ".join(_EN_ONES[int(d)] for d in fraction)
409
+ return f"{head} point {tail}"
410
+ tail = " ".join(_KO_SINO_DIGITS[int(d)] for d in fraction)
411
+ return f"{head} 점 {tail}"
412
+
413
+
414
+ # ======================================================================
415
+ # Rule tables
416
+ # ======================================================================
417
+
418
+ _MONTHS: Final = {
419
+ "jan": 1, "january": 1, "feb": 2, "february": 2, "mar": 3, "march": 3,
420
+ "apr": 4, "april": 4, "may": 5, "jun": 6, "june": 6, "jul": 7, "july": 7,
421
+ "aug": 8, "august": 8, "sep": 9, "sept": 9, "september": 9,
422
+ "oct": 10, "october": 10, "nov": 11, "november": 11, "dec": 12, "december": 12,
423
+ }
424
+ _MONTH_NAMES: Final = (
425
+ "", "January", "February", "March", "April", "May", "June",
426
+ "July", "August", "September", "October", "November", "December",
427
+ )
428
+ #: 6월 is 유월 and 10월 is 시월, not 육월 and 십월. The irregularity is the
429
+ #: whole reason months are not just "Sino number + 월".
430
+ _KO_MONTHS: Final = (
431
+ "", "일월", "이월", "삼월", "사월", "오월", "유월",
432
+ "칠월", "팔월", "구월", "시월", "십일월", "십이월",
433
+ )
434
+
435
+ # token -> (english singular, english plural, korean)
436
+ _UNITS: Final[dict[str, tuple[str, str, str]]] = {
437
+ "kHz": ("kilohertz", "kilohertz", "킬로헤르츠"),
438
+ "MHz": ("megahertz", "megahertz", "메가헤르츠"),
439
+ "GHz": ("gigahertz", "gigahertz", "기가헤르츠"),
440
+ "Hz": ("hertz", "hertz", "헤르츠"),
441
+ "km": ("kilometer", "kilometers", "킬로미터"),
442
+ "cm": ("centimeter", "centimeters", "센티미터"),
443
+ "mm": ("millimeter", "millimeters", "밀리미터"),
444
+ "kg": ("kilogram", "kilograms", "킬로그램"),
445
+ "mg": ("milligram", "milligrams", "밀리그램"),
446
+ "KB": ("kilobyte", "kilobytes", "킬로바이트"),
447
+ "MB": ("megabyte", "megabytes", "메가바이트"),
448
+ "GB": ("gigabyte", "gigabytes", "기가바이트"),
449
+ "TB": ("terabyte", "terabytes", "테라바이트"),
450
+ "ms": ("millisecond", "milliseconds", "밀리초"),
451
+ "°C": ("degree Celsius", "degrees Celsius", "도"),
452
+ "°F": ("degree Fahrenheit", "degrees Fahrenheit", "도"),
453
+ }
454
+
455
+ #: counter -> (native reading?, a space between number and counter?)
456
+ #: 번 is absent on purpose: "3번" is 삼번 for a bus and 세 번 for three times,
457
+ #: and nothing in the surrounding text settles which.
458
+ _KO_COUNTERS: Final[dict[str, tuple[bool, bool]]] = {
459
+ "시간": (True, True),
460
+ "개월": (False, False),
461
+ "주일": (False, False),
462
+ "학년": (False, False),
463
+ "인분": (False, True),
464
+ "개": (True, True),
465
+ "명": (True, True),
466
+ "사람": (True, True),
467
+ "살": (True, True),
468
+ "마리": (True, True),
469
+ "권": (True, True),
470
+ "대": (True, True),
471
+ "병": (True, True),
472
+ "잔": (True, True),
473
+ "그루": (True, True),
474
+ "가지": (True, True),
475
+ "켤레": (True, True),
476
+ "벌": (True, True),
477
+ "채": (True, True),
478
+ "시": (True, False),
479
+ "년": (False, False),
480
+ "월": (False, False),
481
+ "일": (False, False),
482
+ "원": (False, False),
483
+ "분": (False, False),
484
+ "초": (False, False),
485
+ "층": (False, False),
486
+ "주": (False, False),
487
+ "호": (False, False),
488
+ "도": (False, False),
489
+ "회": (False, True),
490
+ "세": (False, False),
491
+ "미터": (False, True),
492
+ }
493
+
494
+ # abbreviation -> (english reading, korean reading)
495
+ #
496
+ # "Ms.", "St.", and "No." are deliberately absent: their expansions are
497
+ # ambiguous (Saint or Street; Number or the Spanish negative), and F-27's
498
+ # risk is asymmetric -- leaving text alone is recoverable, speaking the wrong
499
+ # word is not.
500
+ _ABBREVIATIONS: Final[dict[str, tuple[str, str]]] = {
501
+ "Dr.": ("Doctor", "Doctor"),
502
+ "Mr.": ("Mister", "Mister"),
503
+ "Mrs.": ("Missus", "Missus"),
504
+ "Prof.": ("Professor", "Professor"),
505
+ "etc.": ("et cetera", "et cetera"),
506
+ "e.g.": ("for example", "for example"),
507
+ "i.e.": ("that is", "that is"),
508
+ "vs.": ("versus", "versus"),
509
+ # Spelled as letters rather than "in the morning": p.m. spans both
510
+ # afternoon and evening, so the wordy reading is wrong half the time.
511
+ "a.m.": ("A M", "오전"),
512
+ "p.m.": ("P M", "오후"),
513
+ }
514
+
515
+ #: Tokens whose trailing period never ends a sentence (F-81). Used by the
516
+ #: segmenter, which must agree with this table or "Dr. Kim" splits in two.
517
+ NON_TERMINAL_ABBREVIATIONS: Final[frozenset[str]] = frozenset(
518
+ {"Dr", "Mr", "Mrs", "Ms", "Prof", "St", "Jr", "Sr", "Fig", "No", "Inc",
519
+ "Ltd", "Co", "Corp", "Rev", "Gen", "Sgt", "Capt", "vs", "cf", "al",
520
+ "e.g", "i.e", "a.m", "p.m", "A.M", "P.M", "U.S", "Ph.D", "approx", "est"}
521
+ )
522
+ #: Tokens that may or may not end a sentence; the segmenter looks at what
523
+ #: follows.
524
+ AMBIGUOUS_ABBREVIATIONS: Final[frozenset[str]] = frozenset({"etc", "Ave", "Rd"})
525
+
526
+
527
+ def _char_class(ranges: tuple[tuple[int, int], ...]) -> str:
528
+ return "".join(
529
+ f"\\U{lo:08x}" if lo == hi else f"\\U{lo:08x}-\\U{hi:08x}" for lo, hi in ranges
530
+ )
531
+
532
+
533
+ _DECORATIVE_CLASS: Final = _char_class(_DECORATIVE_RANGES) + _char_class(
534
+ ((0x1F1E6, 0x1F1FF), (0x1F3FB, 0x1F3FF), (0x20E3, 0x20E3), (0x200D, 0x200D))
535
+ )
536
+
537
+ _NUM: Final = r"\d{1,3}(?:,\d{3})+|\d+"
538
+ _DEC: Final = rf"(?:{_NUM})(?:\.\d+)?"
539
+ #: Nothing that would make the digits part of an identifier or a version.
540
+ #: Nothing may start a numeric match immediately after one of these. The
541
+ #: dashes are what keeps "010-1234-5678" a phone number: with them absent, a
542
+ #: rule could start on the second group and read half of it as a range.
543
+ _LEFT: Final = r"(?<![-–~\d.,A-Za-z_])"
544
+
545
+
546
+ def _alternation(tokens: Iterable[str]) -> str:
547
+ """Longest token first, so 시간 wins over 시 and 개월 over 개."""
548
+ return "|".join(re.escape(t) for t in sorted(tokens, key=len, reverse=True))
549
+
550
+
551
+ #: Month names are matched capitalised only. Lower-case "may" is a common
552
+ #: English verb, and "I may 3 times" is not a date.
553
+ _MONTH_TOKENS: Final = tuple(sorted({n.capitalize() for n in _MONTHS}, key=len, reverse=True))
554
+
555
+ _RULES: Final[tuple[tuple[str, str], ...]] = (
556
+ ("iso_date", r"(?<![\d-])\d{4}-\d{2}-\d{2}(?![\d-])"),
557
+ (
558
+ "en_date",
559
+ r"(?<![A-Za-z])(?:"
560
+ + "|".join(_MONTH_TOKENS)
561
+ + r")\.?\s+\d{1,2}(?:st|nd|rd|th)?(?:\s*,?\s*\d{4})?(?![A-Za-z\d])",
562
+ ),
563
+ ("currency", _LEFT + rf"[$₩]\s?(?:{_NUM})(?:\.\d{{1,2}})?(?![\d])"),
564
+ ("percent", _LEFT + rf"(?:{_DEC})\s?%"),
565
+ ("unit", _LEFT + rf"(?:{_DEC})\s?(?:{_alternation(_UNITS)})(?![A-Za-z])"),
566
+ ("ko_counter", _LEFT + rf"(?:{_DEC})\s?(?:{_alternation(_KO_COUNTERS)})"),
567
+ (
568
+ "number_range",
569
+ # ``(?!\d)`` stops the second number backtracking to a shorter one to
570
+ # satisfy the "no third group" lookahead that follows it.
571
+ _LEFT + rf"(?:{_NUM})\s?[-–~]\s?(?:{_NUM})(?!\d)(?!\s?[-–~]\s?\d)(?![A-Za-z])",
572
+ ),
573
+ ("abbrev", r"(?<![A-Za-z.])(?:" + _alternation(_ABBREVIATIONS) + r")"),
574
+ # ``(?!,\d)`` keeps "1,2,3" whole: the group is not a thousands group, so
575
+ # reading only the "1" would speak a different list from the one written.
576
+ # ``(?![-–~]\d)`` does the same for "010-1234-5678", which the range rule
577
+ # above has already declined: a part of a phone number is not a number.
578
+ ("decimal", _LEFT + rf"(?:{_NUM})\.\d+(?![\d.])(?!,\d)(?![-–~]\d)(?![A-Za-z])"),
579
+ ("integer", _LEFT + rf"(?:{_NUM})(?![\d.]*\d)(?!,\d)(?![-–~]\d)(?![A-Za-z])"),
580
+ ("decorative", rf"[{_DECORATIVE_CLASS}]+"),
581
+ ("space", r"\s+"),
582
+ )
583
+
584
+ #: One alternation, scanned once. Rule order is priority order: at any
585
+ #: position the first rule that can match wins, which is why the specific
586
+ #: patterns (a date, a currency amount) precede the general ones (a decimal,
587
+ #: an integer).
588
+ _MASTER: Final = re.compile("|".join(f"(?P<{name}>{pattern})" for name, pattern in _RULES))
589
+
590
+ _RX_ISO: Final = re.compile(r"(\d{4})-(\d{2})-(\d{2})")
591
+ _RX_EN_DATE: Final = re.compile(
592
+ r"([A-Za-z]+)\.?\s+(\d{1,2})(?:st|nd|rd|th)?(?:\s*,?\s*(\d{4}))?", re.ASCII
593
+ )
594
+ _RX_CURRENCY: Final = re.compile(r"([$₩])\s?([\d,]+)(?:\.(\d{1,2}))?")
595
+ _RX_VALUE_TAIL: Final = re.compile(r"([\d,.]+)\s?(.+)", re.DOTALL)
596
+ _RX_RANGE: Final = re.compile(r"([\d,]+)\s?[-–~]\s?([\d,]+)")
597
+
598
+
599
+ # ======================================================================
600
+ # Rule handlers
601
+ # ======================================================================
602
+
603
+
604
+ def _h_iso_date(match: re.Match[str], lang: str) -> str:
605
+ parsed = _RX_ISO.fullmatch(match.group())
606
+ if parsed is None:
607
+ return match.group()
608
+ year, month, day = (int(g) for g in parsed.groups())
609
+ if not (1 <= month <= 12 and 1 <= day <= 31):
610
+ return match.group() # a part number or a range, not a date
611
+ return _spoken_date(year, month, day, lang)
612
+
613
+
614
+ def _h_en_date(match: re.Match[str], lang: str) -> str:
615
+ parsed = _RX_EN_DATE.fullmatch(match.group())
616
+ if parsed is None:
617
+ return match.group()
618
+ name, day_text, year_text = parsed.groups()
619
+ month = _MONTHS.get(name.lower())
620
+ day = int(day_text)
621
+ if month is None or not 1 <= day <= 31:
622
+ return match.group()
623
+ year = int(year_text) if year_text else None
624
+ if lang == KO:
625
+ head = "" if year is None else f"{korean_sino(year)}년 "
626
+ return f"{head}{_KO_MONTHS[month]} {korean_sino(day)}일"
627
+ tail = "" if year is None else f", {english_year(year)}"
628
+ return f"{_MONTH_NAMES[month]} {english_ordinal(day)}{tail}"
629
+
630
+
631
+ def _spoken_date(year: int, month: int, day: int, lang: str) -> str:
632
+ if lang == KO:
633
+ return f"{korean_sino(year)}년 {_KO_MONTHS[month]} {korean_sino(day)}일"
634
+ return f"{_MONTH_NAMES[month]} {english_ordinal(day)}, {english_year(year)}"
635
+
636
+
637
+ def _h_currency(match: re.Match[str], lang: str) -> str:
638
+ parsed = _RX_CURRENCY.fullmatch(match.group())
639
+ if parsed is None:
640
+ return match.group()
641
+ sign, whole_text, cents_text = parsed.groups()
642
+ whole = int(whole_text.replace(",", ""))
643
+ cents = int(cents_text.ljust(2, "0")) if cents_text else 0
644
+ if sign == "₩":
645
+ # The won has no everyday subunit, so a decimal here is a plain
646
+ # fractional amount rather than 100ths of a unit.
647
+ literal = whole_text if not cents_text else f"{whole_text}.{cents_text}"
648
+ reading = spoken_number(literal, lang)
649
+ return f"{reading} 원" if lang == KO else f"{reading} won"
650
+ say_whole = bool(whole) or not cents
651
+ if lang == KO:
652
+ head = f"{korean_sino(whole)} 달러" if say_whole else ""
653
+ tail = f"{korean_sino(cents)} 센트" if cents else ""
654
+ return " ".join(part for part in (head, tail) if part)
655
+ head = f"{english_cardinal(whole)} dollar{'' if whole == 1 else 's'}" if say_whole else ""
656
+ tail = f"{english_cardinal(cents)} cent{'' if cents == 1 else 's'}" if cents else ""
657
+ if head and tail:
658
+ return f"{head} and {tail}"
659
+ return head or tail
660
+
661
+
662
+ def _h_percent(match: re.Match[str], lang: str) -> str:
663
+ literal = match.group().rstrip("%").strip()
664
+ reading = spoken_number(literal, lang)
665
+ return f"{reading} 퍼센트" if lang == KO else f"{reading} percent"
666
+
667
+
668
+ def _h_unit(match: re.Match[str], lang: str) -> str:
669
+ parsed = _RX_VALUE_TAIL.fullmatch(match.group())
670
+ if parsed is None:
671
+ return match.group()
672
+ literal, token = parsed.group(1), parsed.group(2).strip()
673
+ singular, plural, korean = _UNITS[token]
674
+ reading = spoken_number(literal, lang)
675
+ if lang == KO:
676
+ return f"{reading} {korean}"
677
+ exact_one = literal.replace(",", "") in ("1", "1.0")
678
+ return f"{reading} {singular if exact_one else plural}"
679
+
680
+
681
+ def _h_ko_counter(match: re.Match[str], lang: str) -> str:
682
+ """Korean counters are read in Korean whatever the sentence's language.
683
+
684
+ The counter itself is a Korean word, so an English reading of the number
685
+ in front of it ("three 개") is not a reading anyone would want; F-05's
686
+ per-sentence language decides the voice, not the arithmetic.
687
+ """
688
+ parsed = _RX_VALUE_TAIL.fullmatch(match.group())
689
+ if parsed is None:
690
+ return match.group()
691
+ literal, counter = parsed.group(1), parsed.group(2).strip()
692
+ native, spaced = _KO_COUNTERS[counter]
693
+ digits = literal.replace(",", "")
694
+ if counter == "월" and "." not in digits and 1 <= int(digits) <= 12:
695
+ return _KO_MONTHS[int(digits)]
696
+ reading: str | None = None
697
+ if native and "." not in digits:
698
+ reading = korean_native(int(digits))
699
+ if reading is None:
700
+ reading = spoken_number(literal, KO)
701
+ return f"{reading} {counter}" if spaced else f"{reading}{counter}"
702
+
703
+
704
+ def _h_number_range(match: re.Match[str], lang: str) -> str:
705
+ parsed = _RX_RANGE.fullmatch(match.group())
706
+ if parsed is None:
707
+ return match.group()
708
+ low, high = (spoken_number(g, lang) for g in parsed.groups())
709
+ return f"{low}에서 {high}" if lang == KO else f"{low} to {high}"
710
+
711
+
712
+ def _h_abbrev(match: re.Match[str], lang: str) -> str:
713
+ english, korean = _ABBREVIATIONS[match.group()]
714
+ return korean if lang == KO else english
715
+
716
+
717
+ def _h_number(match: re.Match[str], lang: str) -> str:
718
+ return spoken_number(match.group(), lang)
719
+
720
+
721
+ def _h_decorative(match: re.Match[str], lang: str) -> str:
722
+ """A.3: not sent for synthesis, but the source range survives.
723
+
724
+ The run collapses to nothing, except between two non-space characters,
725
+ where it collapses to a single space -- otherwise "hello🎉world" would
726
+ reach the engine as one invented word.
727
+ """
728
+ text = match.string
729
+ before = text[match.start() - 1] if match.start() else ""
730
+ after = text[match.end()] if match.end() < len(text) else ""
731
+ if before and after and not before.isspace() and not after.isspace():
732
+ return " "
733
+ return ""
734
+
735
+
736
+ def _h_space(match: re.Match[str], lang: str) -> str:
737
+ """One space, whatever the run contained.
738
+
739
+ Line breaks must not reach the engine, and a paragraph break is expressed
740
+ as silence between segments (F-82, F-08) rather than as characters.
741
+ """
742
+ return " "
743
+
744
+
745
+ _HANDLERS: Final[dict[str, Callable[[re.Match[str], str], str]]] = {
746
+ "iso_date": _h_iso_date,
747
+ "en_date": _h_en_date,
748
+ "currency": _h_currency,
749
+ "percent": _h_percent,
750
+ "unit": _h_unit,
751
+ "ko_counter": _h_ko_counter,
752
+ "number_range": _h_number_range,
753
+ "abbrev": _h_abbrev,
754
+ "decimal": _h_number,
755
+ "integer": _h_number,
756
+ "decorative": _h_decorative,
757
+ "space": _h_space,
758
+ }
759
+
760
+
761
+ # ======================================================================
762
+ # The alignment
763
+ # ======================================================================
764
+
765
+
766
+ @dataclass(frozen=True, slots=True)
767
+ class Piece:
768
+ """One entry of the edit script: a source span and what it is read as."""
769
+
770
+ source: TextRange
771
+ text: str
772
+
773
+ @property
774
+ def is_identity(self) -> bool:
775
+ return len(self.text) == self.source.length
776
+
777
+
778
+ @dataclass(frozen=True, slots=True)
779
+ class Normalized:
780
+ """Normalised text plus the alignment F-27 requires.
781
+
782
+ ``pieces`` tiles ``source`` exactly: consecutive, no gaps, no overlaps,
783
+ covering ``[0, len(source))``. ``text`` is their concatenation. All
784
+ offsets are Unicode code points into the string that was normalised, and
785
+ a caller normalising a slice of a document adds that slice's own offset.
786
+ """
787
+
788
+ source: str
789
+ text: str
790
+ pieces: tuple[Piece, ...]
791
+ # Index arrays for the two mappings, derived in __post_init__ (which is
792
+ # also where the tiling is verified: a gap or an overlap is a defect in a
793
+ # rule, and F-27 makes it one worth crashing on rather than shipping).
794
+ _source_starts: tuple[int, ...] = field(default=(), init=False, repr=False, compare=False)
795
+ _produced_starts: tuple[int, ...] = field(default=(), init=False, repr=False, compare=False)
796
+
797
+ def __post_init__(self) -> None:
798
+ cursor = 0
799
+ produced = 0
800
+ source_starts: list[int] = []
801
+ produced_starts: list[int] = []
802
+ for piece in self.pieces:
803
+ if piece.source.start != cursor:
804
+ raise AssertionError(
805
+ f"alignment is not a tiling at {piece.source.start} (expected {cursor})"
806
+ )
807
+ source_starts.append(cursor)
808
+ produced_starts.append(produced)
809
+ cursor = piece.source.end
810
+ produced += len(piece.text)
811
+ if cursor != len(self.source):
812
+ raise AssertionError(f"alignment covers {cursor} of {len(self.source)} code points")
813
+ if produced != len(self.text):
814
+ # Lengths, not contents: the offsets are all the mappings use,
815
+ # and rebuilding the joined string here would double the cost of
816
+ # normalising a 50,000-character document to catch nothing that
817
+ # this module can produce.
818
+ raise AssertionError("alignment pieces do not span the normalised text")
819
+ object.__setattr__(self, "_source_starts", tuple(source_starts))
820
+ object.__setattr__(self, "_produced_starts", tuple(produced_starts))
821
+
822
+ # -- mapping ------------------------------------------------------
823
+
824
+ def source_offset(self, produced_offset: int) -> int:
825
+ """Where in the *source* a produced offset came from.
826
+
827
+ Inside a piece that changed length there is no character-level truth
828
+ to report, so the piece's own start is returned: a highlight may cover
829
+ one character too many, but it can never point outside the span that
830
+ produced the audio.
831
+ """
832
+ if produced_offset <= 0:
833
+ return 0
834
+ if produced_offset >= len(self.text):
835
+ return len(self.source)
836
+ index = bisect_right(self._produced_starts, produced_offset) - 1
837
+ piece = self.pieces[index]
838
+ within = produced_offset - self._produced_starts[index]
839
+ if piece.is_identity:
840
+ return piece.source.start + within
841
+ return piece.source.start
842
+
843
+ def produced_offset(self, source_offset: int) -> int:
844
+ """The inverse. Offsets inside a rewritten piece floor to its start."""
845
+ if source_offset <= 0:
846
+ return 0
847
+ if source_offset >= len(self.source):
848
+ return len(self.text)
849
+ index = bisect_right(self._source_starts, source_offset) - 1
850
+ piece = self.pieces[index]
851
+ within = source_offset - piece.source.start
852
+ if piece.is_identity:
853
+ return self._produced_starts[index] + within
854
+ return self._produced_starts[index]
855
+
856
+ def spoken_between(self, start: int, end: int) -> str:
857
+ """The normalised text produced by source range ``[start, end)``."""
858
+ return self.text[self.produced_offset(start) : self.produced_offset(end)]
859
+
860
+ def source_range_of(self, produced: TextRange) -> TextRange:
861
+ return TextRange(self.source_offset(produced.start), self.source_offset(produced.end))
862
+
863
+ @property
864
+ def boundaries(self) -> tuple[int, ...]:
865
+ """Source offsets a caller may cut at without splitting a piece."""
866
+ return self._source_starts + (len(self.source),)
867
+
868
+
869
+ def normalize(text: str, lang: str = EN) -> Normalized:
870
+ """Expand numbers, dates, currency, and abbreviations for one language.
871
+
872
+ F-27. ``lang`` is the engine language F-05 resolved for this sentence;
873
+ it selects the reading, never which characters exist.
874
+ """
875
+ pieces: list[Piece] = []
876
+ out: list[str] = []
877
+ cursor = 0
878
+ for match in _MASTER.finditer(text):
879
+ start, end = match.span()
880
+ if start > cursor:
881
+ _append(pieces, out, cursor, start, text[cursor:start])
882
+ # Exactly one rule group can participate in a match, so the last
883
+ # named group that matched is the rule that fired.
884
+ name = match.lastgroup or ""
885
+ _append(pieces, out, start, end, _HANDLERS[name](match, lang))
886
+ cursor = end
887
+ if cursor < len(text):
888
+ _append(pieces, out, cursor, len(text), text[cursor:])
889
+ return Normalized(source=text, text="".join(out), pieces=tuple(pieces))
890
+
891
+
892
+ def _append(pieces: list[Piece], out: list[str], start: int, end: int, produced: str) -> None:
893
+ pieces.append(Piece(TextRange(start, end), produced))
894
+ out.append(produced)
895
+
896
+
897
+ def is_decorative(ch: str) -> bool:
898
+ """Whether one code point is emoji or decorative, per A.3."""
899
+ cp = ord(ch)
900
+ return (
901
+ _in_ranges(cp, _DECORATIVE_RANGES)
902
+ or 0x1F1E6 <= cp <= 0x1F1FF
903
+ or 0x1F3FB <= cp <= 0x1F3FF
904
+ or cp in (0x200D, 0x20E3)
905
+ )
906
+
907
+
908
+ __all__ = [
909
+ "AMBIGUOUS_ABBREVIATIONS",
910
+ "NON_TERMINAL_ABBREVIATIONS",
911
+ "Normalized",
912
+ "Piece",
913
+ "english_cardinal",
914
+ "english_ordinal",
915
+ "english_year",
916
+ "grapheme_boundaries",
917
+ "grapheme_clusters",
918
+ "is_decorative",
919
+ "is_grapheme_boundary",
920
+ "korean_native",
921
+ "korean_sino",
922
+ "normalize",
923
+ "spoken_number",
924
+ ]