alifbe 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- alifbe/__init__.py +85 -0
- alifbe/__main__.py +95 -0
- alifbe/apostrophe.py +71 -0
- alifbe/case.py +111 -0
- alifbe/confusables.py +81 -0
- alifbe/detect.py +66 -0
- alifbe/exceptions_data.py +62 -0
- alifbe/oldnew.py +110 -0
- alifbe/py.typed +0 -0
- alifbe/result.py +42 -0
- alifbe/searchkey.py +61 -0
- alifbe/transliterate.py +306 -0
- alifbe-0.2.0.dist-info/METADATA +177 -0
- alifbe-0.2.0.dist-info/RECORD +18 -0
- alifbe-0.2.0.dist-info/WHEEL +5 -0
- alifbe-0.2.0.dist-info/entry_points.txt +2 -0
- alifbe-0.2.0.dist-info/licenses/LICENSE +21 -0
- alifbe-0.2.0.dist-info/top_level.txt +1 -0
alifbe/__init__.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""
|
|
2
|
+
alifbe — robust text normalization utilities for the Uzbek language.
|
|
3
|
+
|
|
4
|
+
Handles the sharp edges of real-world Uzbek text:
|
|
5
|
+
|
|
6
|
+
* Apostrophe chaos (oʻ / gʻ / tutuq belgisi typed with 8+ different glyphs)
|
|
7
|
+
* The "Isʼhoq" problem (s+h is not always the "sh" digraph)
|
|
8
|
+
* ş vs ș confusable Unicode lookalikes
|
|
9
|
+
* The "Turkish I" casing bug
|
|
10
|
+
* Ambiguous Cyrillic letters (е, ц, ё) that depend on context
|
|
11
|
+
* The pre-2026 vs. post-reform (Sept 2026 Senate-approved) Latin
|
|
12
|
+
orthographies -- sh/ch/oʻ/gʻ vs ş/ç/ö/ğ
|
|
13
|
+
* Cross-script/cross-orthography search-key folding, so "oʻzbek",
|
|
14
|
+
"özbek", and "ўзбек" are recognized as the same word
|
|
15
|
+
|
|
16
|
+
Nothing here silently guesses when a guess would corrupt data — ambiguous
|
|
17
|
+
cases are surfaced as warnings so the caller (or a human) can decide.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from .apostrophe import normalize_apostrophes, TURNED_COMMA, APOSTROPHE
|
|
21
|
+
from .confusables import normalize_confusables, find_confusables, Confusable
|
|
22
|
+
from .case import (
|
|
23
|
+
uz_lower,
|
|
24
|
+
uz_upper,
|
|
25
|
+
uz_title,
|
|
26
|
+
find_turkish_i_corruption,
|
|
27
|
+
fix_turkish_i_corruption,
|
|
28
|
+
CorruptionHit,
|
|
29
|
+
TURKISH_DOTTED_I,
|
|
30
|
+
TURKISH_DOTLESS_I,
|
|
31
|
+
)
|
|
32
|
+
from .exceptions_data import SH_BOUNDARY_EXCEPTIONS, is_sh_boundary_word
|
|
33
|
+
from .result import Result, Warning_
|
|
34
|
+
from .transliterate import (
|
|
35
|
+
TransliterationResult,
|
|
36
|
+
AmbiguousTransliterationError,
|
|
37
|
+
latin_to_cyrillic,
|
|
38
|
+
cyrillic_to_latin,
|
|
39
|
+
)
|
|
40
|
+
from .oldnew import ConversionResult, to_new_latin, to_old_latin
|
|
41
|
+
from .detect import detect_alphabet, AlphabetDetection
|
|
42
|
+
from .searchkey import fold_search_key, fold_search_key_loose
|
|
43
|
+
|
|
44
|
+
__all__ = [
|
|
45
|
+
# apostrophe
|
|
46
|
+
"normalize_apostrophes",
|
|
47
|
+
"TURNED_COMMA",
|
|
48
|
+
"APOSTROPHE",
|
|
49
|
+
# confusables
|
|
50
|
+
"normalize_confusables",
|
|
51
|
+
"find_confusables",
|
|
52
|
+
"Confusable",
|
|
53
|
+
# case
|
|
54
|
+
"uz_lower",
|
|
55
|
+
"uz_upper",
|
|
56
|
+
"uz_title",
|
|
57
|
+
"find_turkish_i_corruption",
|
|
58
|
+
"fix_turkish_i_corruption",
|
|
59
|
+
"CorruptionHit",
|
|
60
|
+
"TURKISH_DOTTED_I",
|
|
61
|
+
"TURKISH_DOTLESS_I",
|
|
62
|
+
# exceptions
|
|
63
|
+
"SH_BOUNDARY_EXCEPTIONS",
|
|
64
|
+
"is_sh_boundary_word",
|
|
65
|
+
# shared result type
|
|
66
|
+
"Result",
|
|
67
|
+
"Warning_",
|
|
68
|
+
# transliteration (Latin <-> Cyrillic)
|
|
69
|
+
"TransliterationResult",
|
|
70
|
+
"AmbiguousTransliterationError",
|
|
71
|
+
"latin_to_cyrillic",
|
|
72
|
+
"cyrillic_to_latin",
|
|
73
|
+
# 2026 reform (old Latin <-> new Latin)
|
|
74
|
+
"ConversionResult",
|
|
75
|
+
"to_new_latin",
|
|
76
|
+
"to_old_latin",
|
|
77
|
+
# detection
|
|
78
|
+
"detect_alphabet",
|
|
79
|
+
"AlphabetDetection",
|
|
80
|
+
# search keys
|
|
81
|
+
"fold_search_key",
|
|
82
|
+
"fold_search_key_loose",
|
|
83
|
+
]
|
|
84
|
+
|
|
85
|
+
__version__ = "0.2.0"
|
alifbe/__main__.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command-line interface: `python -m alifbe <command> [text]`.
|
|
3
|
+
|
|
4
|
+
Reads TEXT from the positional argument if given, otherwise from stdin
|
|
5
|
+
(so it works well in pipelines). Warnings, if any, go to stderr so stdout
|
|
6
|
+
stays clean for piping the converted text onward.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
from . import (
|
|
13
|
+
normalize_apostrophes,
|
|
14
|
+
normalize_confusables,
|
|
15
|
+
uz_upper,
|
|
16
|
+
uz_lower,
|
|
17
|
+
uz_title,
|
|
18
|
+
find_turkish_i_corruption,
|
|
19
|
+
latin_to_cyrillic,
|
|
20
|
+
cyrillic_to_latin,
|
|
21
|
+
to_new_latin,
|
|
22
|
+
to_old_latin,
|
|
23
|
+
detect_alphabet,
|
|
24
|
+
fold_search_key,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _read_text(args) -> str:
|
|
29
|
+
if args.text is not None:
|
|
30
|
+
return args.text
|
|
31
|
+
return sys.stdin.read().rstrip("\n")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _emit(result_text: str, warnings=()) -> None:
|
|
35
|
+
print(result_text)
|
|
36
|
+
for w in warnings:
|
|
37
|
+
print(f"warning: {w.message}", file=sys.stderr)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def main(argv=None) -> int:
|
|
41
|
+
parser = argparse.ArgumentParser(
|
|
42
|
+
prog="alifbe",
|
|
43
|
+
description="Normalize and convert Uzbek text (apostrophes, ş/ș confusables, "
|
|
44
|
+
"old/new Latin orthography, Cyrillic transliteration).",
|
|
45
|
+
)
|
|
46
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
47
|
+
|
|
48
|
+
for name in ("normalize-apostrophes", "normalize-confusables",
|
|
49
|
+
"upper", "lower", "title",
|
|
50
|
+
"to-cyrillic", "to-latin", "to-new-latin", "to-old-latin",
|
|
51
|
+
"detect", "fold-key", "check"):
|
|
52
|
+
p = sub.add_parser(name)
|
|
53
|
+
p.add_argument("text", nargs="?", help="Text to process (reads stdin if omitted)")
|
|
54
|
+
|
|
55
|
+
args = parser.parse_args(argv)
|
|
56
|
+
text = _read_text(args)
|
|
57
|
+
|
|
58
|
+
if args.command == "normalize-apostrophes":
|
|
59
|
+
_emit(normalize_apostrophes(text))
|
|
60
|
+
elif args.command == "normalize-confusables":
|
|
61
|
+
_emit(normalize_confusables(text))
|
|
62
|
+
elif args.command == "upper":
|
|
63
|
+
_emit(uz_upper(text))
|
|
64
|
+
elif args.command == "lower":
|
|
65
|
+
_emit(uz_lower(text))
|
|
66
|
+
elif args.command == "title":
|
|
67
|
+
_emit(uz_title(text))
|
|
68
|
+
elif args.command == "to-cyrillic":
|
|
69
|
+
result = latin_to_cyrillic(text)
|
|
70
|
+
_emit(result.text, result.warnings)
|
|
71
|
+
elif args.command == "to-latin":
|
|
72
|
+
result = cyrillic_to_latin(text)
|
|
73
|
+
_emit(result.text, result.warnings)
|
|
74
|
+
elif args.command == "to-new-latin":
|
|
75
|
+
_emit(to_new_latin(text).text)
|
|
76
|
+
elif args.command == "to-old-latin":
|
|
77
|
+
_emit(to_old_latin(text).text)
|
|
78
|
+
elif args.command == "detect":
|
|
79
|
+
detection = detect_alphabet(text)
|
|
80
|
+
print(f"{detection.alphabet} (confidence: {detection.confidence})")
|
|
81
|
+
elif args.command == "fold-key":
|
|
82
|
+
_emit(fold_search_key(text))
|
|
83
|
+
elif args.command == "check":
|
|
84
|
+
hits = find_turkish_i_corruption(text)
|
|
85
|
+
if hits:
|
|
86
|
+
for h in hits:
|
|
87
|
+
print(f"corruption: '{h.char}' ({h.codepoint}) at index {h.index}")
|
|
88
|
+
return 1
|
|
89
|
+
print("clean: no Turkish-I corruption found")
|
|
90
|
+
|
|
91
|
+
return 0
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
if __name__ == "__main__":
|
|
95
|
+
raise SystemExit(main())
|
alifbe/apostrophe.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Normalize the ~8+ different characters people type for Uzbek apostrophes
|
|
3
|
+
into the two *correct* Unicode code points:
|
|
4
|
+
|
|
5
|
+
ʻ U+02BB MODIFIER LETTER TURNED COMMA -> used only inside oʻ / gʻ
|
|
6
|
+
ʼ U+02BC MODIFIER LETTER APOSTROPHE -> the tutuq belgisi (glottal
|
|
7
|
+
stop) used everywhere else,
|
|
8
|
+
e.g. sanʼat, Isʼhoq, mashʼal
|
|
9
|
+
|
|
10
|
+
People type either of these using: ' (U+0027), ‘ ’ (curly quotes),
|
|
11
|
+
` ´ (grave/acute accent), ʹ (prime), ʻʼ themselves, and more — often
|
|
12
|
+
inconsistently within the same document.
|
|
13
|
+
|
|
14
|
+
The rule used to decide *which* canonical character a lookalike becomes:
|
|
15
|
+
if it immediately follows o/O/g/G it is part of the oʻ/gʻ letter (turned
|
|
16
|
+
comma); otherwise it is the tutuq belgisi (plain apostrophe).
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
import unicodedata
|
|
21
|
+
|
|
22
|
+
TURNED_COMMA = "\u02BB" # ʻ — part of the letters oʻ / gʻ
|
|
23
|
+
APOSTROPHE = "\u02BC" # ʼ — tutuq belgisi / glottal stop marker
|
|
24
|
+
|
|
25
|
+
# Every character that, in the wild, gets used to mean "apostrophe" in
|
|
26
|
+
# Uzbek text. This list is intentionally generous.
|
|
27
|
+
_APOSTROPHE_LOOKALIKES = {
|
|
28
|
+
"'", # U+0027 APOSTROPHE (ascii)
|
|
29
|
+
"\u2018", # ‘ LEFT SINGLE QUOTATION MARK
|
|
30
|
+
"\u2019", # ’ RIGHT SINGLE QUOTATION MARK
|
|
31
|
+
"\u201A", # ‚ SINGLE LOW-9 QUOTATION MARK
|
|
32
|
+
"\u201B", # ‛ SINGLE HIGH-REVERSED-9 QUOTATION MARK
|
|
33
|
+
"`", # U+0060 GRAVE ACCENT
|
|
34
|
+
"\u00B4", # ´ ACUTE ACCENT
|
|
35
|
+
"\u02B9", # ʹ MODIFIER LETTER PRIME
|
|
36
|
+
"\u02BA", # ʺ MODIFIER LETTER DOUBLE PRIME
|
|
37
|
+
"\u02BB", # ʻ MODIFIER LETTER TURNED COMMA (already canonical form #1)
|
|
38
|
+
"\u02BC", # ʼ MODIFIER LETTER APOSTROPHE (already canonical form #2)
|
|
39
|
+
"\u02BD", # ʽ MODIFIER LETTER REVERSED COMMA
|
|
40
|
+
"\u02BE", # ʾ MODIFIER LETTER RIGHT HALF RING
|
|
41
|
+
"\u02BF", # ʿ MODIFIER LETTER LEFT HALF RING
|
|
42
|
+
"\u02C8", # ˈ MODIFIER LETTER VERTICAL LINE (stress mark, sometimes misused)
|
|
43
|
+
"\u02CA", # ˊ MODIFIER LETTER ACUTE ACCENT
|
|
44
|
+
"\u02EE", # ˮ MODIFIER LETTER DOUBLE APOSTROPHE
|
|
45
|
+
"\u2032", # ′ PRIME
|
|
46
|
+
"\u0301", # combining acute accent (when it ends up standalone)
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
_PATTERN = re.compile("[" + "".join(re.escape(c) for c in _APOSTROPHE_LOOKALIKES) + "]")
|
|
50
|
+
|
|
51
|
+
_OG_TRIGGERS = set("oOgG")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def normalize_apostrophes(text: str) -> str:
|
|
55
|
+
"""
|
|
56
|
+
Rewrite every apostrophe-lookalike in `text` to the correct Uzbek
|
|
57
|
+
character: U+02BB (ʻ) right after o/g, otherwise U+02BC (ʼ).
|
|
58
|
+
|
|
59
|
+
NFC-normalizes first so precomposed vs. decomposed forms don't sneak
|
|
60
|
+
through as false negatives.
|
|
61
|
+
"""
|
|
62
|
+
text = unicodedata.normalize("NFC", text)
|
|
63
|
+
|
|
64
|
+
def repl(match: "re.Match[str]") -> str:
|
|
65
|
+
idx = match.start()
|
|
66
|
+
prev_char = text[idx - 1] if idx > 0 else ""
|
|
67
|
+
if prev_char in _OG_TRIGGERS:
|
|
68
|
+
return TURNED_COMMA
|
|
69
|
+
return APOSTROPHE
|
|
70
|
+
|
|
71
|
+
return _PATTERN.sub(repl, text)
|
alifbe/case.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Locale-independent, explicit-table casing for Uzbek Latin text.
|
|
3
|
+
|
|
4
|
+
Uzbek Latin uses plain ASCII I/i — it has no dotless ı or dotted İ.
|
|
5
|
+
Those two characters (U+0131 ı, U+0130 İ) belong to Turkish. But a
|
|
6
|
+
surprising number of pipelines silently produce them anyway: ICU / a
|
|
7
|
+
database collation / a JS `.toLocaleUpperCase('tr')` call / an OS with a
|
|
8
|
+
Turkish locale set somewhere upstream all quietly turn plain "i" into
|
|
9
|
+
İ (dotted capital I) or "I" into ı (dotless i). Once that happens, Uzbek
|
|
10
|
+
text is corrupted — "olib" becomes "OLİB" instead of "OLIB", and later
|
|
11
|
+
lookups against the original ASCII string fail.
|
|
12
|
+
|
|
13
|
+
This module never depends on the platform locale or ICU. It uses fixed,
|
|
14
|
+
explicit translation tables so `uz_upper("olib") == "OLIB"` everywhere,
|
|
15
|
+
always, regardless of what locale the process happens to be running under.
|
|
16
|
+
|
|
17
|
+
Covers both orthographies in current use:
|
|
18
|
+
* the pre-2026 Latin alphabet (digraphs sh/ch/ng, letters oʻ/gʻ)
|
|
19
|
+
* the Latin alphabet approved by Uzbekistan's Senate in Sept 2026
|
|
20
|
+
(single letters ş/ç/ö/ğ) — not yet in force as of this writing,
|
|
21
|
+
pending presidential signature
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
import re
|
|
25
|
+
from typing import List, NamedTuple
|
|
26
|
+
|
|
27
|
+
TURKISH_DOTTED_I = "\u0130" # İ — must never appear in Uzbek text
|
|
28
|
+
TURKISH_DOTLESS_I = "\u0131" # ı — must never appear in Uzbek text
|
|
29
|
+
|
|
30
|
+
# Explicit lower -> upper map. Covers the ASCII letters plus every
|
|
31
|
+
# Uzbek-specific character from both orthographies. c/w are included even
|
|
32
|
+
# though they only really appear in loanwords/foreign names — silently
|
|
33
|
+
# leaving them un-cased would be a worse surprise than mapping them.
|
|
34
|
+
_LOWER_TO_UPPER = {
|
|
35
|
+
"a": "A", "b": "B", "c": "C", "d": "D", "e": "E", "f": "F", "g": "G",
|
|
36
|
+
"h": "H", "i": "I", "j": "J", "k": "K", "l": "L", "m": "M", "n": "N",
|
|
37
|
+
"o": "O", "p": "P", "q": "Q", "r": "R", "s": "S", "t": "T", "u": "U",
|
|
38
|
+
"v": "V", "w": "W", "x": "X", "y": "Y", "z": "Z",
|
|
39
|
+
"\u02bb": "\u02bb", # ʻ modifier letter turned comma has no case
|
|
40
|
+
"\u02bc": "\u02bc", # ʼ modifier letter apostrophe has no case
|
|
41
|
+
# 2026 reform letters
|
|
42
|
+
"\u00f6": "\u00d6", # ö -> Ö
|
|
43
|
+
"\u011f": "\u011e", # ğ -> Ğ
|
|
44
|
+
"\u015f": "\u015e", # ş -> Ş
|
|
45
|
+
"\u00e7": "\u00c7", # ç -> Ç
|
|
46
|
+
}
|
|
47
|
+
_UPPER_TO_LOWER = {v: k for k, v in _LOWER_TO_UPPER.items()}
|
|
48
|
+
|
|
49
|
+
_LOWER_TABLE = str.maketrans(_UPPER_TO_LOWER)
|
|
50
|
+
_UPPER_TABLE = str.maketrans(_LOWER_TO_UPPER)
|
|
51
|
+
|
|
52
|
+
# Digraphs that must be title-cased as a unit ("Sh", not "SH" or "sH")
|
|
53
|
+
# when title-casing a word, e.g. "Shahzoda", "Gʻofur", "Choʻntak".
|
|
54
|
+
_DIGRAPHS = ["sh", "ch", "ng", "o\u02bb", "g\u02bb"]
|
|
55
|
+
_DIGRAPH_RE = re.compile(
|
|
56
|
+
"|".join(re.escape(d) for d in sorted(_DIGRAPHS, key=len, reverse=True)),
|
|
57
|
+
re.IGNORECASE,
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def uz_lower(text: str) -> str:
|
|
62
|
+
"""Lowercase Uzbek Latin text using a fixed table — never locale-dependent."""
|
|
63
|
+
return text.translate(_LOWER_TABLE)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def uz_upper(text: str) -> str:
|
|
67
|
+
"""Uppercase Uzbek Latin text using a fixed table — never locale-dependent."""
|
|
68
|
+
return text.translate(_UPPER_TABLE)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def uz_title(text: str) -> str:
|
|
72
|
+
"""
|
|
73
|
+
Title-case each word, keeping Uzbek digraphs (sh, ch, ng, oʻ, gʻ) as a
|
|
74
|
+
single capitalized unit when they start a word, e.g.
|
|
75
|
+
"shahzoda" -> "Shahzoda", not "SHahzoda" or "SHAHZODA".
|
|
76
|
+
"""
|
|
77
|
+
def cap_word(word: str) -> str:
|
|
78
|
+
if not word:
|
|
79
|
+
return word
|
|
80
|
+
m = _DIGRAPH_RE.match(word)
|
|
81
|
+
if m:
|
|
82
|
+
digraph = m.group(0)
|
|
83
|
+
rest = word[len(digraph):]
|
|
84
|
+
return uz_upper(digraph[0]) + uz_lower(digraph[1:]) + uz_lower(rest)
|
|
85
|
+
return uz_upper(word[0]) + uz_lower(word[1:])
|
|
86
|
+
|
|
87
|
+
return re.sub(r"[^\s\-]+", lambda m: cap_word(m.group(0)), text)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class CorruptionHit(NamedTuple):
|
|
91
|
+
index: int
|
|
92
|
+
char: str
|
|
93
|
+
codepoint: str
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def find_turkish_i_corruption(text: str) -> List[CorruptionHit]:
|
|
97
|
+
"""
|
|
98
|
+
Flag any Turkish dotted/dotless I (İ, ı) in text that claims to be
|
|
99
|
+
Uzbek — these should never legitimately appear and are a strong signal
|
|
100
|
+
that a Turkish-locale case operation ran on the text somewhere upstream.
|
|
101
|
+
"""
|
|
102
|
+
hits = []
|
|
103
|
+
for i, ch in enumerate(text):
|
|
104
|
+
if ch in (TURKISH_DOTTED_I, TURKISH_DOTLESS_I):
|
|
105
|
+
hits.append(CorruptionHit(index=i, char=ch, codepoint=f"U+{ord(ch):04X}"))
|
|
106
|
+
return hits
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def fix_turkish_i_corruption(text: str) -> str:
|
|
110
|
+
"""Best-effort repair: İ -> I, ı -> i. Only safe if the source was Uzbek."""
|
|
111
|
+
return text.replace(TURKISH_DOTTED_I, "I").replace(TURKISH_DOTLESS_I, "i")
|
alifbe/confusables.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Detect and normalize Unicode "confusable" characters that *look* identical
|
|
3
|
+
(or nearly identical) but are different code points — meaning two spellings
|
|
4
|
+
of "the same word" silently fail to match in search, sorting, or a database
|
|
5
|
+
unique constraint.
|
|
6
|
+
|
|
7
|
+
The canonical example: ş (U+015F LATIN SMALL LETTER S WITH CEDILLA, used in
|
|
8
|
+
Turkish/Romanian text people paste in) vs ș (U+0219 LATIN SMALL LETTER S
|
|
9
|
+
WITH COMMA BELOW, the "correct" Romanian character). Uzbek text should not
|
|
10
|
+
contain either — but they leak in from copy-pasted foreign text, and
|
|
11
|
+
because they render almost identically most fonts, humans never notice.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import unicodedata
|
|
15
|
+
from typing import Dict, List, NamedTuple
|
|
16
|
+
|
|
17
|
+
# Map of "confusable code point" -> "canonical code point we normalize to".
|
|
18
|
+
# Extend this table as new lookalike pairs are discovered in real data.
|
|
19
|
+
CONFUSABLE_MAP: Dict[str, str] = {
|
|
20
|
+
"\u0219": "\u015f", # ș -> ş
|
|
21
|
+
"\u0218": "\u015e", # Ș -> Ş
|
|
22
|
+
"\u021b": "\u0163", # ț -> ţ
|
|
23
|
+
"\u021a": "\u0162", # Ț -> Ţ
|
|
24
|
+
# Cyrillic/Latin homoglyphs that commonly leak into "Latin" Uzbek text
|
|
25
|
+
# when copy-pasted from a Cyrillic-typing keyboard layout.
|
|
26
|
+
"\u0430": "a", # а (Cyrillic) -> a (Latin)
|
|
27
|
+
"\u0435": "e", # е (Cyrillic) -> e (Latin)
|
|
28
|
+
"\u043e": "o", # о (Cyrillic) -> o (Latin)
|
|
29
|
+
"\u0440": "p", # р (Cyrillic) -> p (Latin)
|
|
30
|
+
"\u0441": "c", # с (Cyrillic) -> c (Latin)
|
|
31
|
+
"\u0445": "x", # х (Cyrillic) -> x (Latin)
|
|
32
|
+
"\u0443": "y", # у (Cyrillic) -> y (Latin) -- careful: not always desired
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class Confusable(NamedTuple):
|
|
37
|
+
index: int
|
|
38
|
+
char: str
|
|
39
|
+
codepoint: str
|
|
40
|
+
suggestion: str
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def find_confusables(text: str) -> List[Confusable]:
|
|
44
|
+
"""Return every confusable character found in `text`, without changing it."""
|
|
45
|
+
text = unicodedata.normalize("NFC", text)
|
|
46
|
+
found = []
|
|
47
|
+
for i, ch in enumerate(text):
|
|
48
|
+
if ch in CONFUSABLE_MAP:
|
|
49
|
+
found.append(
|
|
50
|
+
Confusable(
|
|
51
|
+
index=i,
|
|
52
|
+
char=ch,
|
|
53
|
+
codepoint=f"U+{ord(ch):04X}",
|
|
54
|
+
suggestion=CONFUSABLE_MAP[ch],
|
|
55
|
+
)
|
|
56
|
+
)
|
|
57
|
+
return found
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def normalize_confusables(text: str, mixed_script_letters: bool = False) -> str:
|
|
61
|
+
"""
|
|
62
|
+
Rewrite confusable characters to their canonical form.
|
|
63
|
+
|
|
64
|
+
mixed_script_letters: if False (default), only fixes diacritic
|
|
65
|
+
confusables (ş/ș etc.) and leaves bare Cyrillic look-alike letters
|
|
66
|
+
(а/е/о/р/с/х/у) untouched, since collapsing those can be a real script
|
|
67
|
+
decision, not just cleanup. Set True to also fold those.
|
|
68
|
+
"""
|
|
69
|
+
text = unicodedata.normalize("NFC", text)
|
|
70
|
+
out = []
|
|
71
|
+
for ch in text:
|
|
72
|
+
target = CONFUSABLE_MAP.get(ch)
|
|
73
|
+
if target and (mixed_script_letters or not _is_bare_cyrillic_letter(ch)):
|
|
74
|
+
out.append(target)
|
|
75
|
+
else:
|
|
76
|
+
out.append(ch)
|
|
77
|
+
return "".join(out)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _is_bare_cyrillic_letter(ch: str) -> bool:
|
|
81
|
+
return "\u0400" <= ch <= "\u04FF"
|
alifbe/detect.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Detect which script/orthography a piece of Uzbek text is written in.
|
|
3
|
+
|
|
4
|
+
Useful before deciding which conversion function to run, or for flagging
|
|
5
|
+
documents that mix scripts (very common in real-world Uzbek data: a
|
|
6
|
+
Cyrillic paragraph with a Latin brand name, or vice versa).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Literal
|
|
12
|
+
|
|
13
|
+
from .apostrophe import TURNED_COMMA
|
|
14
|
+
|
|
15
|
+
Alphabet = Literal["cyrillic", "old-latin", "new-latin", "mixed", "unknown"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class AlphabetDetection:
|
|
20
|
+
alphabet: Alphabet
|
|
21
|
+
confidence: float # 0.0-1.0, rough and heuristic -- not a probability
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
_NEW_LATIN_MARKS = set("öÖğĞşŞçÇ")
|
|
25
|
+
_OLD_LATIN_DIGRAPH_RE = re.compile(r"sh|ch|ng", re.IGNORECASE)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def detect_alphabet(text: str) -> AlphabetDetection:
|
|
29
|
+
"""
|
|
30
|
+
Best-effort guess at which script `text` is written in.
|
|
31
|
+
|
|
32
|
+
This is a heuristic over a short string, not a language model — treat
|
|
33
|
+
the confidence score as "how much signal was there", not as a
|
|
34
|
+
statistically calibrated probability.
|
|
35
|
+
"""
|
|
36
|
+
letters = [c for c in text if c.isalpha()]
|
|
37
|
+
if not letters:
|
|
38
|
+
return AlphabetDetection("unknown", 0.0)
|
|
39
|
+
|
|
40
|
+
cyrillic_count = sum(1 for c in letters if "\u0400" <= c <= "\u04ff")
|
|
41
|
+
cyrillic_ratio = cyrillic_count / len(letters)
|
|
42
|
+
|
|
43
|
+
new_latin_hits = sum(1 for c in text if c in _NEW_LATIN_MARKS)
|
|
44
|
+
old_latin_hits = len(_OLD_LATIN_DIGRAPH_RE.findall(text)) + text.count(TURNED_COMMA)
|
|
45
|
+
|
|
46
|
+
if cyrillic_ratio > 0.5:
|
|
47
|
+
if new_latin_hits or old_latin_hits:
|
|
48
|
+
return AlphabetDetection("mixed", round(cyrillic_ratio, 2))
|
|
49
|
+
return AlphabetDetection("cyrillic", round(cyrillic_ratio, 2))
|
|
50
|
+
|
|
51
|
+
if new_latin_hits and old_latin_hits:
|
|
52
|
+
return AlphabetDetection("mixed", 0.5)
|
|
53
|
+
if new_latin_hits:
|
|
54
|
+
return AlphabetDetection("new-latin", min(1.0, 0.6 + 0.1 * new_latin_hits))
|
|
55
|
+
if old_latin_hits:
|
|
56
|
+
return AlphabetDetection("old-latin", min(1.0, 0.6 + 0.05 * old_latin_hits))
|
|
57
|
+
if cyrillic_count:
|
|
58
|
+
# Some Cyrillic present but not a majority, and no Latin-specific
|
|
59
|
+
# markers found either -- genuinely mixed.
|
|
60
|
+
return AlphabetDetection("mixed", round(cyrillic_ratio, 2))
|
|
61
|
+
|
|
62
|
+
# Plain ASCII letters only, no digraphs/apostrophes/new letters seen:
|
|
63
|
+
# could be either orthography (most words look the same in both), or
|
|
64
|
+
# not Uzbek at all. Low-confidence new-latin is the more useful default
|
|
65
|
+
# since it's the superset that round-trips cleanly either way.
|
|
66
|
+
return AlphabetDetection("new-latin", 0.3)
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Words where "s" and "h" sit next to each other but do NOT form the "sh"
|
|
3
|
+
digraph (ш / ş) — they belong to different morphemes, usually across a
|
|
4
|
+
tutuq belgisi (ʼ) that marks an Arabic-origin glottal stop, e.g.:
|
|
5
|
+
|
|
6
|
+
Isʼhoq (Исҳоқ, the name Isaac/Ishaq) — is + hoq, not "ish" + "oq"
|
|
7
|
+
asʼhob (асҳоб, companions) — as + hob
|
|
8
|
+
vasʼhat (васҳат) — vas + hat
|
|
9
|
+
mashʼal (машъал, torch) — even though "mash" looks
|
|
10
|
+
like a digraph, it's mash + ʼal
|
|
11
|
+
|
|
12
|
+
The apostrophe usually disambiguates this already — as long as it survives.
|
|
13
|
+
The real danger is upstream code that strips punctuation *before*
|
|
14
|
+
transliterating ("Isʼhoq" -> "Ishoq"), at which point "sh" looks exactly
|
|
15
|
+
like a normal digraph and silently gets merged.
|
|
16
|
+
|
|
17
|
+
This list lets `is_sh_boundary_word` catch that even when the apostrophe
|
|
18
|
+
is already gone, by checking the word (apostrophes stripped, lowercased)
|
|
19
|
+
against known exceptions. It is necessarily incomplete — a starting point,
|
|
20
|
+
not a linguistic corpus. Pass your own additions via `extra=`.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from typing import Iterable, Set
|
|
24
|
+
|
|
25
|
+
# Stored WITHOUT apostrophes, lowercase — the same normalization
|
|
26
|
+
# is_sh_boundary_word applies to its input before comparing.
|
|
27
|
+
SH_BOUNDARY_EXCEPTIONS: Set[str] = {
|
|
28
|
+
"ishoq", # Isʼhoq — the name Isaac/Ishaq
|
|
29
|
+
"ishoqov", # surname derived from Isʼhoq
|
|
30
|
+
"ishoqova",
|
|
31
|
+
"ashob", # asʼhob — companions (Islamic term)
|
|
32
|
+
"vashat", # vasʼhat
|
|
33
|
+
"mashal", # mashʼal — torch
|
|
34
|
+
"mashur", # mashʼur -- well-known/famous (also seen without apostrophe as "mashhur",
|
|
35
|
+
# which genuinely IS a plain digraph word; kept out on purpose, see note below
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
# NOTE on "mashhur": this is a real ambiguity in the source data itself —
|
|
39
|
+
# "mashhur" (famous) is written with a genuine "sh" digraph in most modern
|
|
40
|
+
# usage and is NOT an s+h boundary word, despite superficially resembling
|
|
41
|
+
# "mashʼal". It's deliberately excluded from the exception set. This kind
|
|
42
|
+
# of collision is exactly why the exception list should be reviewed by a
|
|
43
|
+
# native speaker before being trusted for anything production-critical.
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def is_sh_boundary_word(word: str, extra: Iterable[str] = ()) -> bool:
|
|
47
|
+
"""
|
|
48
|
+
Check whether `word` (with or without an apostrophe already present)
|
|
49
|
+
is a known case where s+h is a morpheme boundary, not the "sh"/"ş"
|
|
50
|
+
digraph.
|
|
51
|
+
|
|
52
|
+
Matching is done on the lowercased word with apostrophes stripped, so
|
|
53
|
+
it catches the word whether or not the apostrophe survived upstream.
|
|
54
|
+
"""
|
|
55
|
+
normalized = (
|
|
56
|
+
word.lower()
|
|
57
|
+
.replace("\u02bc", "") # ʼ
|
|
58
|
+
.replace("\u02bb", "") # ʻ
|
|
59
|
+
.replace("'", "")
|
|
60
|
+
)
|
|
61
|
+
exception_set = SH_BOUNDARY_EXCEPTIONS | {e.lower() for e in extra}
|
|
62
|
+
return normalized in exception_set
|
alifbe/oldnew.py
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Convert between the pre-2026 Uzbek Latin alphabet (26 letters + 3 digraphs
|
|
3
|
+
sh/ch/ng + apostrophe) and the alphabet approved by Uzbekistan's Senate on
|
|
4
|
+
10 September 2026 (28 letters + 1 apostrophe):
|
|
5
|
+
|
|
6
|
+
Old New
|
|
7
|
+
--- ---
|
|
8
|
+
sh, Sh ş, Ş
|
|
9
|
+
ch, Ch ç, Ç
|
|
10
|
+
oʻ, Oʻ ö, Ö
|
|
11
|
+
gʻ, Gʻ ğ, Ğ
|
|
12
|
+
|
|
13
|
+
"ng" is dropped as a distinct letter but is NOT rewritten — it was already
|
|
14
|
+
spelled with plain n+g and continues to be; only how the sound is taught/
|
|
15
|
+
classified changes, not the spelling. The tutuq belgisi (ʼ) is unaffected;
|
|
16
|
+
it remains the alphabet's one apostrophe under the new system too.
|
|
17
|
+
|
|
18
|
+
STATUS: as of this writing the amending law has been approved by the
|
|
19
|
+
Senate and sent to the president, but is **not yet in force**. Treat
|
|
20
|
+
`to_new_latin` as a forward-looking convenience, not a claim about which
|
|
21
|
+
spelling is currently official — check the latest guidance from lex.uz or
|
|
22
|
+
the Uzbek Ministry of Education before assuming this is mandatory.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
import re
|
|
26
|
+
from typing import Iterable, List, Tuple
|
|
27
|
+
|
|
28
|
+
from .apostrophe import normalize_apostrophes, TURNED_COMMA
|
|
29
|
+
from .result import Result
|
|
30
|
+
|
|
31
|
+
ConversionResult = Result
|
|
32
|
+
|
|
33
|
+
# Matches http(s) URLs, emails, and `backtick-quoted` spans -- things that
|
|
34
|
+
# are very unlikely to be running Uzbek prose and where an automatic sh/ch
|
|
35
|
+
# rewrite is more likely to break a slug/identifier than help.
|
|
36
|
+
_AUTO_PROTECT_PATTERN = re.compile(
|
|
37
|
+
r"https?://\S+"
|
|
38
|
+
r"|\b[\w.+-]+@[\w-]+\.[\w.-]+\b"
|
|
39
|
+
r"|`[^`]*`"
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
_PLACEHOLDER_TEMPLATE = "\uE000{index}\uE001" # private-use-area markers
|
|
43
|
+
_PLACEHOLDER_RE = re.compile("\uE000(\\d+)\uE001")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _protect(text: str, protected_terms: Iterable[str], protect_spans: bool
|
|
47
|
+
) -> Tuple[str, List[str]]:
|
|
48
|
+
"""Replace protected substrings with placeholders; return (text, originals)."""
|
|
49
|
+
originals: List[str] = []
|
|
50
|
+
|
|
51
|
+
def stash(match_text: str) -> str:
|
|
52
|
+
originals.append(match_text)
|
|
53
|
+
return _PLACEHOLDER_TEMPLATE.format(index=len(originals) - 1)
|
|
54
|
+
|
|
55
|
+
# Protect explicit terms first (longest first, so overlapping/prefix
|
|
56
|
+
# terms don't partially shadow each other), then automatic spans.
|
|
57
|
+
for term in sorted({t for t in protected_terms if t}, key=len, reverse=True):
|
|
58
|
+
if term in text:
|
|
59
|
+
text = text.replace(term, stash(term))
|
|
60
|
+
if protect_spans:
|
|
61
|
+
text = _AUTO_PROTECT_PATTERN.sub(lambda m: stash(m.group(0)), text)
|
|
62
|
+
return text, originals
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _restore(text: str, originals: List[str]) -> str:
|
|
66
|
+
return _PLACEHOLDER_RE.sub(lambda m: originals[int(m.group(1))], text)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def to_new_latin(text: str, protect_spans: bool = True,
|
|
70
|
+
protected_terms: Iterable[str] = ()) -> Result:
|
|
71
|
+
"""
|
|
72
|
+
Convert pre-2026 Uzbek Latin text to the 2026-reform Latin alphabet.
|
|
73
|
+
|
|
74
|
+
protect_spans: if True (default), URLs, emails, and `backtick`
|
|
75
|
+
spans are left untouched.
|
|
76
|
+
protected_terms: exact substrings (e.g. brand names) to leave
|
|
77
|
+
untouched, e.g. protected_terms=["MyShop"] keeps
|
|
78
|
+
"MyShop" from becoming "MyŞop".
|
|
79
|
+
"""
|
|
80
|
+
# Protect spans FIRST, using the raw input -- apostrophe normalization
|
|
81
|
+
# would otherwise eat backticks (U+0060 is itself an apostrophe
|
|
82
|
+
# lookalike) before the protection regex ever gets to see them.
|
|
83
|
+
protected, originals = _protect(text, protected_terms, protect_spans)
|
|
84
|
+
protected = normalize_apostrophes(protected)
|
|
85
|
+
|
|
86
|
+
converted = re.sub(r"sh", lambda m: _case(m.group(0), "ş", "Ş"), protected, flags=re.IGNORECASE)
|
|
87
|
+
converted = re.sub(r"ch", lambda m: _case(m.group(0), "ç", "Ç"), converted, flags=re.IGNORECASE)
|
|
88
|
+
converted = re.sub(f"[oO]{TURNED_COMMA}", lambda m: "Ö" if m.group(0)[0].isupper() else "ö", converted)
|
|
89
|
+
converted = re.sub(f"[gG]{TURNED_COMMA}", lambda m: "Ğ" if m.group(0)[0].isupper() else "ğ", converted)
|
|
90
|
+
|
|
91
|
+
result_text = _restore(converted, originals)
|
|
92
|
+
return Result(text=result_text, warnings=())
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def to_old_latin(text: str, protect_spans: bool = True,
|
|
96
|
+
protected_terms: Iterable[str] = ()) -> Result:
|
|
97
|
+
"""Convert 2026-reform Uzbek Latin text back to the pre-2026 alphabet."""
|
|
98
|
+
protected, originals = _protect(text, protected_terms, protect_spans)
|
|
99
|
+
|
|
100
|
+
converted = protected.replace("ş", "sh").replace("Ş", "Sh")
|
|
101
|
+
converted = converted.replace("ç", "ch").replace("Ç", "Ch")
|
|
102
|
+
converted = converted.replace("ö", f"o{TURNED_COMMA}").replace("Ö", f"O{TURNED_COMMA}")
|
|
103
|
+
converted = converted.replace("ğ", f"g{TURNED_COMMA}").replace("Ğ", f"G{TURNED_COMMA}")
|
|
104
|
+
|
|
105
|
+
result_text = _restore(converted, originals)
|
|
106
|
+
return Result(text=result_text, warnings=())
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _case(matched: str, lower: str, upper: str) -> str:
|
|
110
|
+
return upper if any(c.isupper() for c in matched) else lower
|
alifbe/py.typed
ADDED
|
File without changes
|
alifbe/result.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Shared result types.
|
|
3
|
+
|
|
4
|
+
Every conversion function in alifbe returns the same shape: the converted
|
|
5
|
+
text, plus a tuple of warnings for anything it wasn't fully certain about.
|
|
6
|
+
Both are frozen (immutable) — a Result you get back can't be silently
|
|
7
|
+
mutated somewhere downstream and cause confusing bugs.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Tuple
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class Warning_:
|
|
16
|
+
"""
|
|
17
|
+
One thing the conversion wasn't fully sure about.
|
|
18
|
+
|
|
19
|
+
index: code-point offset into the *input* text where this applies.
|
|
20
|
+
length: how many input code points this warning covers (usually 1).
|
|
21
|
+
rule: a short machine-readable identifier, e.g. "cyrillic.e.positional",
|
|
22
|
+
so callers can filter/handle specific kinds of ambiguity.
|
|
23
|
+
message: a human-readable explanation.
|
|
24
|
+
"""
|
|
25
|
+
index: int
|
|
26
|
+
length: int
|
|
27
|
+
rule: str
|
|
28
|
+
message: str
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class Result:
|
|
33
|
+
"""The output of a conversion: the text, and zero or more warnings."""
|
|
34
|
+
text: str
|
|
35
|
+
warnings: Tuple[Warning_, ...] = field(default_factory=tuple)
|
|
36
|
+
|
|
37
|
+
def __str__(self) -> str:
|
|
38
|
+
return self.text
|
|
39
|
+
|
|
40
|
+
def __bool__(self) -> bool:
|
|
41
|
+
# A Result is "clean" (truthy) when there were no warnings.
|
|
42
|
+
return not self.warnings
|
alifbe/searchkey.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Canonical search keys for Uzbek text.
|
|
3
|
+
|
|
4
|
+
The whole point of this library is that "the same word" can arrive typed
|
|
5
|
+
eight different ways. That's a display problem for humans, but it's a
|
|
6
|
+
*data* problem for search indexes, database unique constraints, and
|
|
7
|
+
deduplication: "oʻzbek", "o'zbek", "özbek", and "ўзбек" all mean the same
|
|
8
|
+
thing, but a naive `.lower()` treats them as four different strings.
|
|
9
|
+
|
|
10
|
+
fold_search_key() produces one canonical Latin (new-orthography), NFC,
|
|
11
|
+
locale-safe-lowercased string regardless of which script or apostrophe
|
|
12
|
+
style the input used, so equality/hashing/indexing "just works".
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import unicodedata
|
|
16
|
+
|
|
17
|
+
from .apostrophe import normalize_apostrophes
|
|
18
|
+
from .case import uz_lower
|
|
19
|
+
from .confusables import normalize_confusables
|
|
20
|
+
from .detect import detect_alphabet
|
|
21
|
+
from .oldnew import to_new_latin
|
|
22
|
+
from .transliterate import cyrillic_to_latin
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def fold_search_key(text: str) -> str:
|
|
26
|
+
"""
|
|
27
|
+
Canonical search key: same output for the same word regardless of
|
|
28
|
+
script (Cyrillic/Latin) or apostrophe style. Lossless with respect to
|
|
29
|
+
the letters ö/ğ/ş/ç/oʻ/gʻ/etc. — different *sounding* words won't
|
|
30
|
+
collide, only different spellings of the same word do.
|
|
31
|
+
"""
|
|
32
|
+
text = normalize_apostrophes(text)
|
|
33
|
+
text = normalize_confusables(text)
|
|
34
|
+
|
|
35
|
+
detection = detect_alphabet(text)
|
|
36
|
+
if detection.alphabet == "cyrillic":
|
|
37
|
+
text = cyrillic_to_latin(text, on_ambiguous="best_guess", new_orthography=True).text
|
|
38
|
+
else:
|
|
39
|
+
# old-latin, new-latin, mixed, or unknown: run the old->new pass
|
|
40
|
+
# regardless. It's idempotent on text that's already new-latin,
|
|
41
|
+
# and it upgrades any old-latin digraphs it finds in mixed text.
|
|
42
|
+
text = to_new_latin(text, protect_spans=False).text
|
|
43
|
+
|
|
44
|
+
return uz_lower(text)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def fold_search_key_loose(text: str) -> str:
|
|
48
|
+
"""
|
|
49
|
+
Like fold_search_key(), but also strips diacritics (ö->o, ğ->g, ş->s,
|
|
50
|
+
ç->c, apostrophes removed entirely).
|
|
51
|
+
|
|
52
|
+
This is intentionally lossy: "sen" (you) and "şen" (cheerful) would
|
|
53
|
+
fold to the same loose key. Only use this for "did you maybe mean"
|
|
54
|
+
style fuzzy search, never as a unique key or for anything where a
|
|
55
|
+
false match would be a real problem.
|
|
56
|
+
"""
|
|
57
|
+
key = fold_search_key(text)
|
|
58
|
+
decomposed = unicodedata.normalize("NFKD", key)
|
|
59
|
+
stripped = "".join(c for c in decomposed if not unicodedata.combining(c))
|
|
60
|
+
stripped = stripped.replace("\u02bb", "").replace("\u02bc", "").replace("'", "")
|
|
61
|
+
return unicodedata.normalize("NFC", stripped)
|
alifbe/transliterate.py
ADDED
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Uzbek Latin <-> Cyrillic transliteration.
|
|
3
|
+
|
|
4
|
+
Two hard problems this module refuses to paper over:
|
|
5
|
+
|
|
6
|
+
1. The "Isʼhoq" problem: "sh" is usually the digraph for ш, but not when
|
|
7
|
+
s and h belong to different morphemes. We check the apostrophe *and*
|
|
8
|
+
a known-word exception list before ever treating "sh" as one letter.
|
|
9
|
+
|
|
10
|
+
2. Cyrillic е, ц, ё are position/word-dependent and sometimes genuinely
|
|
11
|
+
ambiguous (native vs. borrowed word spelling rules differ). Rather
|
|
12
|
+
than guess silently, cyrillic_to_latin() always returns a `warnings`
|
|
13
|
+
tuple alongside the text, and can be configured to raise instead.
|
|
14
|
+
|
|
15
|
+
Digraph matching is done with case-insensitive regexes rather than an
|
|
16
|
+
explicit table of pre-listed case combinations (`"Sh"`, `"SH"`, ...) —
|
|
17
|
+
a fixed table is exactly the kind of thing that quietly mishandles an
|
|
18
|
+
unusual input like `"sH"` or text typed with caps-lock stuck on. The
|
|
19
|
+
regex approach handles every case combination by construction.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import re
|
|
23
|
+
import unicodedata
|
|
24
|
+
from typing import Iterable, List, Tuple
|
|
25
|
+
|
|
26
|
+
from .apostrophe import normalize_apostrophes, TURNED_COMMA, APOSTROPHE
|
|
27
|
+
from .exceptions_data import is_sh_boundary_word
|
|
28
|
+
from .result import Result, Warning_
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class AmbiguousTransliterationError(ValueError):
|
|
32
|
+
"""Raised when on_ambiguous='raise' and an ambiguous letter is hit."""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# Backwards/forwards-compatible alias: transliteration results and
|
|
36
|
+
# old<->new-latin conversion results share the same shape.
|
|
37
|
+
TransliterationResult = Result
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# ---------------------------------------------------------------------------
|
|
41
|
+
# Latin -> Cyrillic
|
|
42
|
+
# ---------------------------------------------------------------------------
|
|
43
|
+
|
|
44
|
+
_L2C_SINGLES = {
|
|
45
|
+
"a": "а", "b": "б", "c": "к", "d": "д", "e": "е", "f": "ф", "g": "г",
|
|
46
|
+
"h": "ҳ", "i": "и", "j": "ж", "k": "к", "l": "л", "m": "м", "n": "н",
|
|
47
|
+
"o": "о", "p": "п", "q": "қ", "r": "р", "s": "с", "t": "т", "u": "у",
|
|
48
|
+
"v": "в", "w": "в", "x": "х", "y": "й", "z": "з",
|
|
49
|
+
APOSTROPHE: "ъ",
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _cased(latin_lower: str, cyr_lower: str, cyr_upper: str):
|
|
54
|
+
"""Build a (compiled pattern, lower, upper) triple for a digraph."""
|
|
55
|
+
return re.compile(re.escape(latin_lower), re.IGNORECASE), cyr_lower, cyr_upper
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
_L2C_DIGRAPH_SPECS = [
|
|
59
|
+
_cased("sh", "ш", "Ш"),
|
|
60
|
+
_cased("ch", "ч", "Ч"),
|
|
61
|
+
_cased("ng", "нг", "Нг"),
|
|
62
|
+
_cased("yo", "ё", "Ё"),
|
|
63
|
+
_cased("yu", "ю", "Ю"),
|
|
64
|
+
_cased("ya", "я", "Я"),
|
|
65
|
+
]
|
|
66
|
+
|
|
67
|
+
# oʻ / gʻ: the turned comma itself carries no case, so casing is decided
|
|
68
|
+
# by the o/g alone.
|
|
69
|
+
_OG_PATTERN = re.compile(f"([oOgG]){TURNED_COMMA}")
|
|
70
|
+
_OG_TARGET = {"o": "ў", "g": "ғ"}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _apply_case(match_text: str, lower: str, upper: str) -> str:
|
|
74
|
+
# Any uppercase letter in the matched span means "this is a capital
|
|
75
|
+
# letter" -- there's no meaningful notion of "SH" vs "Sh" once it
|
|
76
|
+
# becomes a single Cyrillic/new-Latin character, so any-uppercase wins.
|
|
77
|
+
return upper if any(c.isupper() for c in match_text) else lower
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
_WORD_RE = re.compile(r"[^\W\d_]+", re.UNICODE)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def latin_to_cyrillic(text: str, extra_sh_exceptions: Iterable[str] = ()) -> Result:
|
|
84
|
+
"""
|
|
85
|
+
Convert Uzbek Latin text (old or new orthography) to Cyrillic.
|
|
86
|
+
|
|
87
|
+
Correctly refuses to merge "sh" into ш when the word is a known
|
|
88
|
+
s+h morpheme-boundary exception (e.g. "Isʼhoq") or when an
|
|
89
|
+
apostrophe already separates them, even if a prior "cleanup" step
|
|
90
|
+
stripped the apostrophe from an otherwise-recognized word.
|
|
91
|
+
|
|
92
|
+
Also understands the 2026-reform single letters (ş/ç/ö/ğ) directly,
|
|
93
|
+
so text doesn't need to be downgraded to the old orthography first.
|
|
94
|
+
"""
|
|
95
|
+
text = normalize_apostrophes(text)
|
|
96
|
+
warnings: List[Warning_] = []
|
|
97
|
+
out_parts: List[str] = []
|
|
98
|
+
|
|
99
|
+
last = 0
|
|
100
|
+
for m in _WORD_RE.finditer(text):
|
|
101
|
+
out_parts.append(text[last:m.start()])
|
|
102
|
+
word = m.group(0)
|
|
103
|
+
out_parts.append(_convert_word_l2c(word, warnings, m.start(), extra_sh_exceptions))
|
|
104
|
+
last = m.end()
|
|
105
|
+
out_parts.append(text[last:])
|
|
106
|
+
|
|
107
|
+
return Result(text="".join(out_parts), warnings=tuple(warnings))
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _convert_word_l2c(word: str, warnings: List[Warning_], base_index: int,
|
|
111
|
+
extra_sh_exceptions: Iterable[str]) -> str:
|
|
112
|
+
protect_sh = is_sh_boundary_word(word, extra=extra_sh_exceptions)
|
|
113
|
+
if protect_sh and APOSTROPHE not in word and re.search(r"sh", word, re.IGNORECASE):
|
|
114
|
+
warnings.append(Warning_(
|
|
115
|
+
index=base_index,
|
|
116
|
+
length=len(word),
|
|
117
|
+
rule="latin.sh_boundary.no_apostrophe",
|
|
118
|
+
message=(
|
|
119
|
+
f"'{word}' looks like it contains the 'sh' digraph but is a known "
|
|
120
|
+
f"s+h morpheme-boundary word with no apostrophe present. Converted "
|
|
121
|
+
f"as с+ҳ, not ш. Consider restoring the ʼ (tutuq belgisi) at the source."
|
|
122
|
+
),
|
|
123
|
+
))
|
|
124
|
+
|
|
125
|
+
i = 0
|
|
126
|
+
n = len(word)
|
|
127
|
+
result: List[str] = []
|
|
128
|
+
while i < n:
|
|
129
|
+
if protect_sh and word[i:i + 2].lower() == "sh":
|
|
130
|
+
result.append(_L2C_SINGLES.get(word[i].lower(), word[i]))
|
|
131
|
+
result.append(_L2C_SINGLES.get(word[i + 1].lower(), word[i + 1]))
|
|
132
|
+
i += 2
|
|
133
|
+
continue
|
|
134
|
+
|
|
135
|
+
# New-orthography single letters (ş ç ö ğ)
|
|
136
|
+
new_letter = _NEW_LATIN_TO_CYR.get(word[i].lower())
|
|
137
|
+
if new_letter is not None:
|
|
138
|
+
result.append(new_letter.upper() if word[i].isupper() else new_letter)
|
|
139
|
+
i += 1
|
|
140
|
+
continue
|
|
141
|
+
|
|
142
|
+
og = _OG_PATTERN.match(word, i)
|
|
143
|
+
if og:
|
|
144
|
+
base = og.group(1).lower()
|
|
145
|
+
cyr = _OG_TARGET[base]
|
|
146
|
+
result.append(cyr.upper() if og.group(1).isupper() else cyr)
|
|
147
|
+
i = og.end()
|
|
148
|
+
continue
|
|
149
|
+
|
|
150
|
+
matched_digraph = False
|
|
151
|
+
for pattern, lower_cyr, upper_cyr in _L2C_DIGRAPH_SPECS:
|
|
152
|
+
m = pattern.match(word, i)
|
|
153
|
+
if m:
|
|
154
|
+
result.append(_apply_case(m.group(0), lower_cyr, upper_cyr))
|
|
155
|
+
i = m.end()
|
|
156
|
+
matched_digraph = True
|
|
157
|
+
break
|
|
158
|
+
if matched_digraph:
|
|
159
|
+
continue
|
|
160
|
+
|
|
161
|
+
ch = word[i]
|
|
162
|
+
mapped = _L2C_SINGLES.get(ch.lower())
|
|
163
|
+
if mapped is not None:
|
|
164
|
+
result.append(mapped.upper() if ch.isupper() else mapped)
|
|
165
|
+
else:
|
|
166
|
+
result.append(ch)
|
|
167
|
+
i += 1
|
|
168
|
+
|
|
169
|
+
return "".join(result)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
# New-orthography (post-reform) single letters, lowercase-keyed.
|
|
173
|
+
_NEW_LATIN_TO_CYR = {
|
|
174
|
+
"ş": "ш",
|
|
175
|
+
"ç": "ч",
|
|
176
|
+
"ö": "ў",
|
|
177
|
+
"ğ": "ғ",
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
# ---------------------------------------------------------------------------
|
|
182
|
+
# Cyrillic -> Latin (with explicit ambiguity warnings for е, ц, ё)
|
|
183
|
+
# ---------------------------------------------------------------------------
|
|
184
|
+
|
|
185
|
+
_C2L_SIMPLE = {
|
|
186
|
+
"а": "a", "б": "b", "в": "v", "г": "g", "д": "d", "ж": "j",
|
|
187
|
+
"з": "z", "и": "i", "й": "y", "к": "k", "л": "l", "м": "m",
|
|
188
|
+
"н": "n", "о": "o", "п": "p", "р": "r", "с": "s", "т": "t",
|
|
189
|
+
"у": "u", "ф": "f", "х": "x", "ш": "sh", "ч": "ch", "қ": "q",
|
|
190
|
+
"ғ": f"g{TURNED_COMMA}", "ў": f"o{TURNED_COMMA}", "ҳ": "h",
|
|
191
|
+
"ъ": APOSTROPHE, "ь": "", "ю": "yu", "я": "ya",
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
# Letters whose correct romanization genuinely depends on position or on
|
|
195
|
+
# whether the word is native vs. borrowed. We never guess these silently
|
|
196
|
+
# without recording why.
|
|
197
|
+
_AMBIGUOUS = {"е", "ц", "ё"}
|
|
198
|
+
|
|
199
|
+
_VOWELS = set("аеёиоуўэюя")
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def cyrillic_to_latin(text: str, on_ambiguous: str = "warn",
|
|
203
|
+
new_orthography: bool = False) -> Result:
|
|
204
|
+
"""
|
|
205
|
+
Convert Uzbek Cyrillic text to Latin.
|
|
206
|
+
|
|
207
|
+
on_ambiguous:
|
|
208
|
+
"warn" (default) - use a reasonable positional best-guess for
|
|
209
|
+
е (ye at word start / after a vowel or ъ/ь, else e),
|
|
210
|
+
and for ц/ё fall back to a documented default — but
|
|
211
|
+
always record a Warning_ so the caller knows a guess
|
|
212
|
+
was made.
|
|
213
|
+
"raise" - raise AmbiguousTransliterationError on the first
|
|
214
|
+
ambiguous letter instead of guessing.
|
|
215
|
+
"best_guess" - same positional guessing as "warn", but does not
|
|
216
|
+
populate the warnings tuple.
|
|
217
|
+
|
|
218
|
+
new_orthography: if True, emit the 2026-reform single letters
|
|
219
|
+
(ş/ç/ö/ğ) instead of the digraphs (sh/ch/oʻ/gʻ).
|
|
220
|
+
"""
|
|
221
|
+
if on_ambiguous not in ("warn", "raise", "best_guess"):
|
|
222
|
+
raise ValueError("on_ambiguous must be 'warn', 'raise', or 'best_guess'")
|
|
223
|
+
|
|
224
|
+
text = unicodedata.normalize("NFC", text)
|
|
225
|
+
warnings: List[Warning_] = []
|
|
226
|
+
out: List[str] = []
|
|
227
|
+
|
|
228
|
+
for i, ch in enumerate(text):
|
|
229
|
+
lower = ch.lower()
|
|
230
|
+
is_upper = ch.isupper()
|
|
231
|
+
|
|
232
|
+
if lower in _AMBIGUOUS:
|
|
233
|
+
if on_ambiguous == "raise":
|
|
234
|
+
raise AmbiguousTransliterationError(
|
|
235
|
+
f"Ambiguous Cyrillic letter '{ch}' at index {i} "
|
|
236
|
+
f"(context: ...{text[max(0, i - 5):i + 6]}...). "
|
|
237
|
+
f"Its correct Latin form depends on word origin/position; "
|
|
238
|
+
f"resolve manually or pass on_ambiguous='best_guess'."
|
|
239
|
+
)
|
|
240
|
+
guess, rule, note = _guess_ambiguous(lower, text, i)
|
|
241
|
+
out.append(guess.upper() if is_upper else guess)
|
|
242
|
+
if on_ambiguous == "warn":
|
|
243
|
+
warnings.append(Warning_(index=i, length=1, rule=rule, message=note))
|
|
244
|
+
continue
|
|
245
|
+
|
|
246
|
+
mapped = _C2L_SIMPLE.get(lower)
|
|
247
|
+
if mapped is None:
|
|
248
|
+
out.append(ch)
|
|
249
|
+
continue
|
|
250
|
+
|
|
251
|
+
if is_upper and mapped:
|
|
252
|
+
mapped = mapped[0].upper() + mapped[1:]
|
|
253
|
+
out.append(mapped)
|
|
254
|
+
|
|
255
|
+
result_text = "".join(out)
|
|
256
|
+
if new_orthography:
|
|
257
|
+
result_text = _digraphs_to_new_letters(result_text)
|
|
258
|
+
|
|
259
|
+
return Result(text=result_text, warnings=tuple(warnings))
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _digraphs_to_new_letters(text: str) -> str:
|
|
263
|
+
text = re.sub(r"sh", lambda m: _apply_case(m.group(0), "ş", "Ş"), text, flags=re.IGNORECASE)
|
|
264
|
+
text = re.sub(r"ch", lambda m: _apply_case(m.group(0), "ç", "Ç"), text, flags=re.IGNORECASE)
|
|
265
|
+
text = re.sub(f"[oO]{TURNED_COMMA}", lambda m: "Ö" if m.group(0)[0].isupper() else "ö", text)
|
|
266
|
+
text = re.sub(f"[gG]{TURNED_COMMA}", lambda m: "Ğ" if m.group(0)[0].isupper() else "ğ", text)
|
|
267
|
+
return text
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _guess_ambiguous(lower_ch: str, text: str, i: int) -> Tuple[str, str, str]:
|
|
271
|
+
prev = text[i - 1] if i > 0 else ""
|
|
272
|
+
is_word_start = i == 0 or not prev.isalpha()
|
|
273
|
+
|
|
274
|
+
if lower_ch == "е":
|
|
275
|
+
ye_context = is_word_start or prev.lower() in _VOWELS or prev.lower() in ("ъ", "ь")
|
|
276
|
+
latin = "ye" if ye_context else "e"
|
|
277
|
+
rule = "cyrillic.e.positional"
|
|
278
|
+
note = (
|
|
279
|
+
f"'е' romanized as '{latin}' based on position "
|
|
280
|
+
f"({'word-initial/after vowel or ъ-ь' if ye_context else 'after consonant'}); "
|
|
281
|
+
f"this is a positional heuristic, not a certainty for borrowed words."
|
|
282
|
+
)
|
|
283
|
+
return latin, rule, note
|
|
284
|
+
|
|
285
|
+
if lower_ch == "ц":
|
|
286
|
+
# No reliable positional rule: some words render as 's', direct
|
|
287
|
+
# borrowings (proper nouns, scientific/technical terms) as 'ts'.
|
|
288
|
+
# We default to 'ts' (the more literal, information-preserving
|
|
289
|
+
# choice) and flag it loudly.
|
|
290
|
+
rule = "cyrillic.ts.ambiguous"
|
|
291
|
+
note = (
|
|
292
|
+
"'ц' has no reliable rule (can be 's' or 'ts' depending on the "
|
|
293
|
+
"word's origin). Defaulted to 'ts' — verify against a dictionary "
|
|
294
|
+
"for native-feeling words."
|
|
295
|
+
)
|
|
296
|
+
return "ts", rule, note
|
|
297
|
+
|
|
298
|
+
if lower_ch == "ё":
|
|
299
|
+
rule = "cyrillic.yo.foreign"
|
|
300
|
+
note = (
|
|
301
|
+
"'ё' is not part of the native Uzbek Cyrillic alphabet and "
|
|
302
|
+
"usually only appears in Russian loanwords; defaulted to 'yo'."
|
|
303
|
+
)
|
|
304
|
+
return "yo", rule, note
|
|
305
|
+
|
|
306
|
+
return lower_ch, "", ""
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: alifbe
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Robust normalization for real-world Uzbek text: apostrophe chaos, sh/s+h word boundaries, ş/ș confusables, the Turkish-I casing bug, ambiguous Cyrillic letters, and conversion between the pre-2026 and Sept-2026-reform Latin alphabets.
|
|
5
|
+
Author: you
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/yourname/alifbe
|
|
8
|
+
Project-URL: Repository, https://github.com/yourname/alifbe
|
|
9
|
+
Keywords: uzbek,unicode,normalization,transliteration,i18n,nlp,cyrillic,latin-alphabet
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
15
|
+
Classifier: Topic :: Software Development :: Internationalization
|
|
16
|
+
Classifier: Typing :: Typed
|
|
17
|
+
Requires-Python: >=3.8
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Requires-Dist: pytest; extra == "dev"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# alifbe
|
|
25
|
+
|
|
26
|
+
Robust text normalization for real-world Uzbek text — the eight problems
|
|
27
|
+
that quietly corrupt Uzbek data in most pipelines.
|
|
28
|
+
|
|
29
|
+
| Problem | What alifbe does |
|
|
30
|
+
|---|---|
|
|
31
|
+
| 8+ different "apostrophe" characters (`'` `‘` `’` `` ` `` `´` ...) all meaning oʻ / gʻ / tutuq belgisi | Normalizes all of them to the two *correct* Unicode letters, based on context |
|
|
32
|
+
| `"Isʼhoq"` → `"Işoq"` | Detects s+h morpheme boundaries (via apostrophe *and* a known-word list) so `sh` isn't wrongly merged into one letter |
|
|
33
|
+
| `ş` vs `ș` (cedilla vs. comma-below — different code points, identical glyph) | Detects and normalizes confusable characters so search/dedup actually works |
|
|
34
|
+
| The "Turkish I" bug (`.upper()`/`.lower()` under some locales turns `i` into `İ`) | Locale-independent, explicit-table casing that never produces `İ`/`ı` |
|
|
35
|
+
| Cyrillic `е`, `ц`, `ё` are position/origin-dependent | Transliteration always returns warnings for ambiguous letters instead of silently guessing (and can `raise` instead, if you'd rather fail loudly) |
|
|
36
|
+
| The **Sept 2026 alphabet reform** (sh→ş, ch→ç, oʻ→ö, gʻ→ğ) | `to_new_latin()` / `to_old_latin()` convert between the two orthographies, with brand-name/URL/code protection |
|
|
37
|
+
| "Which script is this text even in?" | `detect_alphabet()` — cyrillic / old-latin / new-latin / mixed / unknown |
|
|
38
|
+
| "oʻzbek", "özbek", and "ўзбек" are the same word, but `==` doesn't think so | `fold_search_key()` gives every spelling the same canonical key |
|
|
39
|
+
|
|
40
|
+
## Install
|
|
41
|
+
|
|
42
|
+
Not yet published to PyPI — the name `alifbe` is free, checked via
|
|
43
|
+
`pip download alifbe` returning no match, but publishing itself is a
|
|
44
|
+
manual step (PyPI account + 2FA + `twine upload`). Install locally for now:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install /path/to/alifbe # normal install
|
|
48
|
+
pip install -e /path/to/alifbe # editable, for development
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
This also registers an `alifbe` command-line tool (see below).
|
|
52
|
+
|
|
53
|
+
## Quick tour
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
import alifbe as uz
|
|
57
|
+
|
|
58
|
+
# 1. Apostrophes: every variant collapses to the correct, same string
|
|
59
|
+
uz.normalize_apostrophes("o'zbek") # -> "oʻzbek" (U+02BB, turned comma)
|
|
60
|
+
uz.normalize_apostrophes("o‘zbek") # -> "oʻzbek" (same result)
|
|
61
|
+
uz.normalize_apostrophes("san'at") # -> "sanʼat" (U+02BC — different rule, not after o/g)
|
|
62
|
+
|
|
63
|
+
# 2. "Isʼhoq" stays "Isʼhoq" -- sh is not wrongly merged into ш
|
|
64
|
+
uz.latin_to_cyrillic("Isʼhoq").text # -> "Исъҳоқ" (not "Ишоқ")
|
|
65
|
+
uz.latin_to_cyrillic("Ishoq").warnings # non-empty: flags this as a known
|
|
66
|
+
# s+h boundary word even without
|
|
67
|
+
# the apostrophe
|
|
68
|
+
|
|
69
|
+
# 3. ş vs ș
|
|
70
|
+
uz.find_confusables("Kraiova munșasi") # -> [Confusable(char='ș', codepoint='U+0219', ...)]
|
|
71
|
+
uz.normalize_confusables("munşa") == uz.normalize_confusables("munșa") # -> True
|
|
72
|
+
|
|
73
|
+
# 4. Turkish I bug
|
|
74
|
+
uz.uz_upper("olib") # -> "OLIB" (never "OLİB")
|
|
75
|
+
uz.find_turkish_i_corruption("OLİB") # -> flags the İ as corruption
|
|
76
|
+
uz.fix_turkish_i_corruption("OLİB") # -> "OLIB"
|
|
77
|
+
|
|
78
|
+
# 5. Ambiguous Cyrillic letters warn instead of silently guessing
|
|
79
|
+
result = uz.cyrillic_to_latin("центр")
|
|
80
|
+
result.text # -> "tsentr" (best guess)
|
|
81
|
+
result.warnings # -> non-empty, explains ц is ambiguous
|
|
82
|
+
uz.cyrillic_to_latin("ёлғон", on_ambiguous="raise") # -> raises instead
|
|
83
|
+
|
|
84
|
+
# 6. The Sept 2026 alphabet reform
|
|
85
|
+
uz.to_new_latin("Shahzoda Oʻzbekistonda choy ichdi").text
|
|
86
|
+
# -> "Şahzoda Özbekistonda çoy içdi"
|
|
87
|
+
uz.to_old_latin("Şahzoda Özbekistonda çoy içdi").text
|
|
88
|
+
# -> "Shahzoda Oʻzbekistonda choy ichdi"
|
|
89
|
+
|
|
90
|
+
# ... with brand names / URLs / code protected from conversion
|
|
91
|
+
uz.to_new_latin("MyShop: sotib oling", protected_terms=["MyShop"]).text
|
|
92
|
+
# -> "MyShop: sotib oling" (MyShop untouched, rest still converts if applicable)
|
|
93
|
+
uz.to_new_latin("See https://x.com/shahar for info").text
|
|
94
|
+
# -> "See https://x.com/shahar for info" (URL untouched)
|
|
95
|
+
|
|
96
|
+
# 7. What script is this?
|
|
97
|
+
uz.detect_alphabet("Shahzoda") # -> AlphabetDetection(alphabet='old-latin', confidence=0.65)
|
|
98
|
+
uz.detect_alphabet("Şahzoda") # -> AlphabetDetection(alphabet='new-latin', confidence=0.7)
|
|
99
|
+
uz.detect_alphabet("Шаҳзода") # -> AlphabetDetection(alphabet='cyrillic', confidence=1.0)
|
|
100
|
+
|
|
101
|
+
# 8. One search key regardless of script or apostrophe style
|
|
102
|
+
uz.fold_search_key("o'zbek") == uz.fold_search_key("özbek") == uz.fold_search_key("ўзбек")
|
|
103
|
+
# -> True (all fold to "özbek")
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
## Command line
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
echo "o'zbek" | alifbe normalize-apostrophes # oʻzbek
|
|
110
|
+
alifbe to-new-latin "Shahzoda choy ichdi" # Şahzoda çoy içdi
|
|
111
|
+
alifbe to-cyrillic "Ishoq" # Исъҳоқ (+ warning on stderr)
|
|
112
|
+
alifbe detect "Шаҳзода" # cyrillic (confidence: 1.0)
|
|
113
|
+
alifbe fold-key "oʻzbek" # özbek
|
|
114
|
+
alifbe check "OLİB" # corruption: 'İ' (U+0130) at index 2
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Every subcommand reads from the positional argument if given, or stdin
|
|
118
|
+
otherwise — so it pipes cleanly. Conversion warnings go to stderr, so
|
|
119
|
+
stdout stays clean for piping the result onward.
|
|
120
|
+
|
|
121
|
+
## Design principle
|
|
122
|
+
|
|
123
|
+
Every function that could plausibly get something wrong either:
|
|
124
|
+
- makes the correct choice deterministically (apostrophes, casing,
|
|
125
|
+
alphabet-reform digraphs), or
|
|
126
|
+
- tells you it's not sure, instead of guessing silently (Cyrillic
|
|
127
|
+
е/ц/ё, the sh/s+h boundary when no apostrophe survives).
|
|
128
|
+
|
|
129
|
+
That second category is deliberate: a library that *always* looks
|
|
130
|
+
confident is more dangerous than one that sometimes says "I'm not sure,
|
|
131
|
+
here's my best guess and why." All conversion functions return the same
|
|
132
|
+
`Result(text, warnings)` shape — `warnings` is empty when the library is
|
|
133
|
+
confident, and populated (with a machine-readable `rule` plus a
|
|
134
|
+
human-readable `message`) when it made a judgment call.
|
|
135
|
+
|
|
136
|
+
## On the Sept 2026 alphabet reform
|
|
137
|
+
|
|
138
|
+
Uzbekistan's Senate approved a bill on 10 September 2026 replacing the
|
|
139
|
+
sh/ch/oʻ/gʻ digraphs with single letters ş/ç/ö/ğ. As of this writing the
|
|
140
|
+
bill has been sent to the president and is **not yet in force** — school
|
|
141
|
+
materials are expected to transition starting 2027. `to_new_latin()` is a
|
|
142
|
+
forward-looking convenience, not a claim about which spelling is
|
|
143
|
+
currently mandatory. Check lex.uz or the Ministry of Education for the
|
|
144
|
+
authoritative status before treating conversion as required.
|
|
145
|
+
|
|
146
|
+
## Known simplifications (read before relying on this for anything critical)
|
|
147
|
+
|
|
148
|
+
- `SH_BOUNDARY_EXCEPTIONS` (in `exceptions_data.py`) is a small starting
|
|
149
|
+
list, not a linguistic corpus — extend it via `extra=` for your data,
|
|
150
|
+
and get a native speaker to review it before production use.
|
|
151
|
+
- Cyrillic `ц` has no reliable positional rule (can be "s" or "ts"
|
|
152
|
+
depending on the word's origin); alifbe defaults to "ts" and always
|
|
153
|
+
warns. Cyrillic `е` uses a positional heuristic (ye at word start/after
|
|
154
|
+
a vowel, else e) which is usually right but isn't a certainty for
|
|
155
|
+
borrowed words.
|
|
156
|
+
- `latin_to_cyrillic("c")` maps to "к" (since old-orthography Uzbek Latin
|
|
157
|
+
has no standalone "c" outside the "ch" digraph) — for text with
|
|
158
|
+
loanwords spelled with a bare "c", double-check the output.
|
|
159
|
+
- `fold_search_key_loose()` is intentionally lossy (strips ö/ğ/ş/ç
|
|
160
|
+
diacritics and apostrophes) — never use it as a unique key, only for
|
|
161
|
+
fuzzy "did you mean" style matching.
|
|
162
|
+
|
|
163
|
+
## Extending the sh-boundary word list
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
uz.latin_to_cyrillic("Asʼhad", extra_sh_exceptions={"ashad"})
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
## Tests
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
pip install -e ".[dev]"
|
|
173
|
+
pytest tests/ -v
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
102 tests across normalization, transliteration, the alphabet-reform
|
|
177
|
+
converter, script detection, search-key folding, and the CLI.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
alifbe/__init__.py,sha256=dTQxZxFW_zZyUtMKsbqSKfVlzNTRIPD5UOJA_SiFsDc,2516
|
|
2
|
+
alifbe/__main__.py,sha256=rS5kqQ4tMWCxHnwzompzTWfTrQDMNUnmNYXREV4yHmY,2963
|
|
3
|
+
alifbe/apostrophe.py,sha256=cFJBs6gBdQ5SDaPwKZMuEJZask04tdqNw4wooFuWsBI,2861
|
|
4
|
+
alifbe/case.py,sha256=yj0yLNlfGAOGJlyJyZWWet5LzJwfCBa9unLuZwSjRfE,4468
|
|
5
|
+
alifbe/confusables.py,sha256=DePw1dgS-4EMhdVjsRZg74P-1sHgL6NUbXn9WILebFM,2984
|
|
6
|
+
alifbe/detect.py,sha256=kL7O1_hsnM_qmEih7RfsCoU2Wz0o-gwnqNIGNh9YZc0,2516
|
|
7
|
+
alifbe/exceptions_data.py,sha256=JnhaHqLSMSMTVHpEOya4p1L1y0jM2bOM3zYcGfYVgPs,2893
|
|
8
|
+
alifbe/oldnew.py,sha256=NufVW5VEgUVl0FKZdL27dFg9pDXMS413i_qvtNz9Hnw,4737
|
|
9
|
+
alifbe/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
10
|
+
alifbe/result.py,sha256=-nBOvmBZeR6mUIbJ8Itbxt3WvW3qulkFLPEDCCEwoN4,1273
|
|
11
|
+
alifbe/searchkey.py,sha256=ofICm8YgRKiIdLwry-Gc0XZicVy77YnD1fm0tf3Mg3o,2510
|
|
12
|
+
alifbe/transliterate.py,sha256=OH-fzfxtuy6gIqs9HgwWLXRVDKpeajiNz70kbLUgizo,11708
|
|
13
|
+
alifbe-0.2.0.dist-info/licenses/LICENSE,sha256=ESYyLizI0WWtxMeS7rGVcX3ivMezm-HOd5WdeOh-9oU,1056
|
|
14
|
+
alifbe-0.2.0.dist-info/METADATA,sha256=l8YZ_ZdWbgo1gpRbWqAfkW_MXQsFw0tPAb_8YcTOD4w,8999
|
|
15
|
+
alifbe-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
16
|
+
alifbe-0.2.0.dist-info/entry_points.txt,sha256=2pwRxqXw7Tr06nX21Z-7gnkUJfvt9Lv-wYUHB7qNPrE,48
|
|
17
|
+
alifbe-0.2.0.dist-info/top_level.txt,sha256=hmzgmWiQBnPNViz9FN9RGr2uZDgKt5GWaWLiBqkndkw,7
|
|
18
|
+
alifbe-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
alifbe
|