unicode-blocks-py 6.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_blocks/__init__.py +4 -0
- unicode_blocks/blocks.py +492 -0
- unicode_blocks/charNormaliser.py +17 -0
- unicode_blocks/cjk.py +149 -0
- unicode_blocks/errors.py +10 -0
- unicode_blocks/globals.py +44 -0
- unicode_blocks/unicodeBlock.py +112 -0
- unicode_blocks_py-6.1.0.dist-info/METADATA +171 -0
- unicode_blocks_py-6.1.0.dist-info/RECORD +12 -0
- unicode_blocks_py-6.1.0.dist-info/WHEEL +5 -0
- unicode_blocks_py-6.1.0.dist-info/licenses/LICENSE +21 -0
- unicode_blocks_py-6.1.0.dist-info/top_level.txt +1 -0
unicode_blocks/blocks.py
ADDED
|
@@ -0,0 +1,492 @@
|
|
|
1
|
+
# This file is auto-generated on build. Do not edit.
|
|
2
|
+
# This code is licensed under the MIT License.
|
|
3
|
+
# Modified under Unicode License v3. See https://www.unicode.org/license.txt for details.
|
|
4
|
+
|
|
5
|
+
from .unicodeBlock import UnicodeBlock
|
|
6
|
+
|
|
7
|
+
__version__ = '6.1.0'
|
|
8
|
+
|
|
9
|
+
NO_BLOCK = UnicodeBlock(name='No Block', start=-1, end=-1)
|
|
10
|
+
BASIC_LATIN = UnicodeBlock(name='Basic Latin', start=0x0000, end=0x007f, assigned_ranges=[(0x0000, 0x007f)], aliases=['ASCII'])
|
|
11
|
+
LATIN_1_SUPPLEMENT = UnicodeBlock(name='Latin-1 Supplement', start=0x0080, end=0x00ff, assigned_ranges=[(0x0080, 0x00ff)], aliases=['Latin_1_Sup', 'Latin_1'])
|
|
12
|
+
LATIN_EXTENDED_A = UnicodeBlock(name='Latin Extended-A', start=0x0100, end=0x017f, assigned_ranges=[(0x0100, 0x017f)], aliases=['Latin_Ext_A'])
|
|
13
|
+
LATIN_EXTENDED_B = UnicodeBlock(name='Latin Extended-B', start=0x0180, end=0x024f, assigned_ranges=[(0x0180, 0x024f)], aliases=['Latin_Ext_B'])
|
|
14
|
+
IPA_EXTENSIONS = UnicodeBlock(name='IPA Extensions', start=0x0250, end=0x02af, assigned_ranges=[(0x0250, 0x02af)], aliases=['IPA_Ext'])
|
|
15
|
+
SPACING_MODIFIER_LETTERS = UnicodeBlock(name='Spacing Modifier Letters', start=0x02b0, end=0x02ff, assigned_ranges=[(0x02b0, 0x02ff)], aliases=['Modifier_Letters'])
|
|
16
|
+
COMBINING_DIACRITICAL_MARKS = UnicodeBlock(name='Combining Diacritical Marks', start=0x0300, end=0x036f, assigned_ranges=[(0x0300, 0x036f)], aliases=['Diacriticals'])
|
|
17
|
+
GREEK_AND_COPTIC = UnicodeBlock(name='Greek and Coptic', start=0x0370, end=0x03ff, assigned_ranges=[(0x0370, 0x0377), (0x037a, 0x037e), (0x0384, 0x038a), (0x038c, 0x038c), (0x038e, 0x03a1), (0x03a3, 0x03ff)], aliases=['Greek'])
|
|
18
|
+
CYRILLIC = UnicodeBlock(name='Cyrillic', start=0x0400, end=0x04ff, assigned_ranges=[(0x0400, 0x04ff)])
|
|
19
|
+
CYRILLIC_SUPPLEMENT = UnicodeBlock(name='Cyrillic Supplement', start=0x0500, end=0x052f, assigned_ranges=[(0x0500, 0x0527)], aliases=['Cyrillic_Sup', 'Cyrillic_Supplementary'])
|
|
20
|
+
ARMENIAN = UnicodeBlock(name='Armenian', start=0x0530, end=0x058f, assigned_ranges=[(0x0531, 0x0556), (0x0559, 0x055f), (0x0561, 0x0587), (0x0589, 0x058a), (0x058f, 0x058f)])
|
|
21
|
+
HEBREW = UnicodeBlock(name='Hebrew', start=0x0590, end=0x05ff, assigned_ranges=[(0x0591, 0x05c7), (0x05d0, 0x05ea), (0x05f0, 0x05f4)])
|
|
22
|
+
ARABIC = UnicodeBlock(name='Arabic', start=0x0600, end=0x06ff, assigned_ranges=[(0x0600, 0x0604), (0x0606, 0x061b), (0x061e, 0x06ff)])
|
|
23
|
+
SYRIAC = UnicodeBlock(name='Syriac', start=0x0700, end=0x074f, assigned_ranges=[(0x0700, 0x070d), (0x070f, 0x074a), (0x074d, 0x074f)])
|
|
24
|
+
ARABIC_SUPPLEMENT = UnicodeBlock(name='Arabic Supplement', start=0x0750, end=0x077f, assigned_ranges=[(0x0750, 0x077f)], aliases=['Arabic_Sup'])
|
|
25
|
+
THAANA = UnicodeBlock(name='Thaana', start=0x0780, end=0x07bf, assigned_ranges=[(0x0780, 0x07b1)])
|
|
26
|
+
NKO = UnicodeBlock(name='NKo', start=0x07c0, end=0x07ff, assigned_ranges=[(0x07c0, 0x07fa)])
|
|
27
|
+
SAMARITAN = UnicodeBlock(name='Samaritan', start=0x0800, end=0x083f, assigned_ranges=[(0x0800, 0x082d), (0x0830, 0x083e)])
|
|
28
|
+
MANDAIC = UnicodeBlock(name='Mandaic', start=0x0840, end=0x085f, assigned_ranges=[(0x0840, 0x085b), (0x085e, 0x085e)])
|
|
29
|
+
ARABIC_EXTENDED_A = UnicodeBlock(name='Arabic Extended-A', start=0x08a0, end=0x08ff, assigned_ranges=[(0x08a0, 0x08a0), (0x08a2, 0x08ac), (0x08e4, 0x08fe)], aliases=['Arabic_Ext_A'])
|
|
30
|
+
DEVANAGARI = UnicodeBlock(name='Devanagari', start=0x0900, end=0x097f, assigned_ranges=[(0x0900, 0x0977), (0x0979, 0x097f)])
|
|
31
|
+
BENGALI = UnicodeBlock(name='Bengali', start=0x0980, end=0x09ff, assigned_ranges=[(0x0981, 0x0983), (0x0985, 0x098c), (0x098f, 0x0990), (0x0993, 0x09a8), (0x09aa, 0x09b0), (0x09b2, 0x09b2), (0x09b6, 0x09b9), (0x09bc, 0x09c4), (0x09c7, 0x09c8), (0x09cb, 0x09ce), (0x09d7, 0x09d7), (0x09dc, 0x09dd), (0x09df, 0x09e3), (0x09e6, 0x09fb)])
|
|
32
|
+
GURMUKHI = UnicodeBlock(name='Gurmukhi', start=0x0a00, end=0x0a7f, assigned_ranges=[(0x0a01, 0x0a03), (0x0a05, 0x0a0a), (0x0a0f, 0x0a10), (0x0a13, 0x0a28), (0x0a2a, 0x0a30), (0x0a32, 0x0a33), (0x0a35, 0x0a36), (0x0a38, 0x0a39), (0x0a3c, 0x0a3c), (0x0a3e, 0x0a42), (0x0a47, 0x0a48), (0x0a4b, 0x0a4d), (0x0a51, 0x0a51), (0x0a59, 0x0a5c), (0x0a5e, 0x0a5e), (0x0a66, 0x0a75)])
|
|
33
|
+
GUJARATI = UnicodeBlock(name='Gujarati', start=0x0a80, end=0x0aff, assigned_ranges=[(0x0a81, 0x0a83), (0x0a85, 0x0a8d), (0x0a8f, 0x0a91), (0x0a93, 0x0aa8), (0x0aaa, 0x0ab0), (0x0ab2, 0x0ab3), (0x0ab5, 0x0ab9), (0x0abc, 0x0ac5), (0x0ac7, 0x0ac9), (0x0acb, 0x0acd), (0x0ad0, 0x0ad0), (0x0ae0, 0x0ae3), (0x0ae6, 0x0af1)])
|
|
34
|
+
ORIYA = UnicodeBlock(name='Oriya', start=0x0b00, end=0x0b7f, assigned_ranges=[(0x0b01, 0x0b03), (0x0b05, 0x0b0c), (0x0b0f, 0x0b10), (0x0b13, 0x0b28), (0x0b2a, 0x0b30), (0x0b32, 0x0b33), (0x0b35, 0x0b39), (0x0b3c, 0x0b44), (0x0b47, 0x0b48), (0x0b4b, 0x0b4d), (0x0b56, 0x0b57), (0x0b5c, 0x0b5d), (0x0b5f, 0x0b63), (0x0b66, 0x0b77)])
|
|
35
|
+
TAMIL = UnicodeBlock(name='Tamil', start=0x0b80, end=0x0bff, assigned_ranges=[(0x0b82, 0x0b83), (0x0b85, 0x0b8a), (0x0b8e, 0x0b90), (0x0b92, 0x0b95), (0x0b99, 0x0b9a), (0x0b9c, 0x0b9c), (0x0b9e, 0x0b9f), (0x0ba3, 0x0ba4), (0x0ba8, 0x0baa), (0x0bae, 0x0bb9), (0x0bbe, 0x0bc2), (0x0bc6, 0x0bc8), (0x0bca, 0x0bcd), (0x0bd0, 0x0bd0), (0x0bd7, 0x0bd7), (0x0be6, 0x0bfa)])
|
|
36
|
+
TELUGU = UnicodeBlock(name='Telugu', start=0x0c00, end=0x0c7f, assigned_ranges=[(0x0c01, 0x0c03), (0x0c05, 0x0c0c), (0x0c0e, 0x0c10), (0x0c12, 0x0c28), (0x0c2a, 0x0c33), (0x0c35, 0x0c39), (0x0c3d, 0x0c44), (0x0c46, 0x0c48), (0x0c4a, 0x0c4d), (0x0c55, 0x0c56), (0x0c58, 0x0c59), (0x0c60, 0x0c63), (0x0c66, 0x0c6f), (0x0c78, 0x0c7f)])
|
|
37
|
+
KANNADA = UnicodeBlock(name='Kannada', start=0x0c80, end=0x0cff, assigned_ranges=[(0x0c82, 0x0c83), (0x0c85, 0x0c8c), (0x0c8e, 0x0c90), (0x0c92, 0x0ca8), (0x0caa, 0x0cb3), (0x0cb5, 0x0cb9), (0x0cbc, 0x0cc4), (0x0cc6, 0x0cc8), (0x0cca, 0x0ccd), (0x0cd5, 0x0cd6), (0x0cde, 0x0cde), (0x0ce0, 0x0ce3), (0x0ce6, 0x0cef), (0x0cf1, 0x0cf2)])
|
|
38
|
+
MALAYALAM = UnicodeBlock(name='Malayalam', start=0x0d00, end=0x0d7f, assigned_ranges=[(0x0d02, 0x0d03), (0x0d05, 0x0d0c), (0x0d0e, 0x0d10), (0x0d12, 0x0d3a), (0x0d3d, 0x0d44), (0x0d46, 0x0d48), (0x0d4a, 0x0d4e), (0x0d57, 0x0d57), (0x0d60, 0x0d63), (0x0d66, 0x0d75), (0x0d79, 0x0d7f)])
|
|
39
|
+
SINHALA = UnicodeBlock(name='Sinhala', start=0x0d80, end=0x0dff, assigned_ranges=[(0x0d82, 0x0d83), (0x0d85, 0x0d96), (0x0d9a, 0x0db1), (0x0db3, 0x0dbb), (0x0dbd, 0x0dbd), (0x0dc0, 0x0dc6), (0x0dca, 0x0dca), (0x0dcf, 0x0dd4), (0x0dd6, 0x0dd6), (0x0dd8, 0x0ddf), (0x0df2, 0x0df4)])
|
|
40
|
+
THAI = UnicodeBlock(name='Thai', start=0x0e00, end=0x0e7f, assigned_ranges=[(0x0e01, 0x0e3a), (0x0e3f, 0x0e5b)])
|
|
41
|
+
LAO = UnicodeBlock(name='Lao', start=0x0e80, end=0x0eff, assigned_ranges=[(0x0e81, 0x0e82), (0x0e84, 0x0e84), (0x0e87, 0x0e88), (0x0e8a, 0x0e8a), (0x0e8d, 0x0e8d), (0x0e94, 0x0e97), (0x0e99, 0x0e9f), (0x0ea1, 0x0ea3), (0x0ea5, 0x0ea5), (0x0ea7, 0x0ea7), (0x0eaa, 0x0eab), (0x0ead, 0x0eb9), (0x0ebb, 0x0ebd), (0x0ec0, 0x0ec4), (0x0ec6, 0x0ec6), (0x0ec8, 0x0ecd), (0x0ed0, 0x0ed9), (0x0edc, 0x0edf)])
|
|
42
|
+
TIBETAN = UnicodeBlock(name='Tibetan', start=0x0f00, end=0x0fff, assigned_ranges=[(0x0f00, 0x0f47), (0x0f49, 0x0f6c), (0x0f71, 0x0f97), (0x0f99, 0x0fbc), (0x0fbe, 0x0fcc), (0x0fce, 0x0fda)])
|
|
43
|
+
MYANMAR = UnicodeBlock(name='Myanmar', start=0x1000, end=0x109f, assigned_ranges=[(0x1000, 0x109f)])
|
|
44
|
+
GEORGIAN = UnicodeBlock(name='Georgian', start=0x10a0, end=0x10ff, assigned_ranges=[(0x10a0, 0x10c5), (0x10c7, 0x10c7), (0x10cd, 0x10cd), (0x10d0, 0x10ff)])
|
|
45
|
+
HANGUL_JAMO = UnicodeBlock(name='Hangul Jamo', start=0x1100, end=0x11ff, assigned_ranges=[(0x1100, 0x11ff)], aliases=['Jamo'])
|
|
46
|
+
ETHIOPIC = UnicodeBlock(name='Ethiopic', start=0x1200, end=0x137f, assigned_ranges=[(0x1200, 0x1248), (0x124a, 0x124d), (0x1250, 0x1256), (0x1258, 0x1258), (0x125a, 0x125d), (0x1260, 0x1288), (0x128a, 0x128d), (0x1290, 0x12b0), (0x12b2, 0x12b5), (0x12b8, 0x12be), (0x12c0, 0x12c0), (0x12c2, 0x12c5), (0x12c8, 0x12d6), (0x12d8, 0x1310), (0x1312, 0x1315), (0x1318, 0x135a), (0x135d, 0x137c)])
|
|
47
|
+
ETHIOPIC_SUPPLEMENT = UnicodeBlock(name='Ethiopic Supplement', start=0x1380, end=0x139f, assigned_ranges=[(0x1380, 0x1399)], aliases=['Ethiopic_Sup'])
|
|
48
|
+
CHEROKEE = UnicodeBlock(name='Cherokee', start=0x13a0, end=0x13ff, assigned_ranges=[(0x13a0, 0x13f4)])
|
|
49
|
+
UNIFIED_CANADIAN_ABORIGINAL_SYLLABICS = UnicodeBlock(name='Unified Canadian Aboriginal Syllabics', start=0x1400, end=0x167f, assigned_ranges=[(0x1400, 0x167f)], aliases=['UCAS', 'Canadian_Syllabics'])
|
|
50
|
+
OGHAM = UnicodeBlock(name='Ogham', start=0x1680, end=0x169f, assigned_ranges=[(0x1680, 0x169c)])
|
|
51
|
+
RUNIC = UnicodeBlock(name='Runic', start=0x16a0, end=0x16ff, assigned_ranges=[(0x16a0, 0x16f0)])
|
|
52
|
+
TAGALOG = UnicodeBlock(name='Tagalog', start=0x1700, end=0x171f, assigned_ranges=[(0x1700, 0x170c), (0x170e, 0x1714)])
|
|
53
|
+
HANUNOO = UnicodeBlock(name='Hanunoo', start=0x1720, end=0x173f, assigned_ranges=[(0x1720, 0x1736)])
|
|
54
|
+
BUHID = UnicodeBlock(name='Buhid', start=0x1740, end=0x175f, assigned_ranges=[(0x1740, 0x1753)])
|
|
55
|
+
TAGBANWA = UnicodeBlock(name='Tagbanwa', start=0x1760, end=0x177f, assigned_ranges=[(0x1760, 0x176c), (0x176e, 0x1770), (0x1772, 0x1773)])
|
|
56
|
+
KHMER = UnicodeBlock(name='Khmer', start=0x1780, end=0x17ff, assigned_ranges=[(0x1780, 0x17dd), (0x17e0, 0x17e9), (0x17f0, 0x17f9)])
|
|
57
|
+
MONGOLIAN = UnicodeBlock(name='Mongolian', start=0x1800, end=0x18af, assigned_ranges=[(0x1800, 0x180e), (0x1810, 0x1819), (0x1820, 0x1877), (0x1880, 0x18aa)])
|
|
58
|
+
UNIFIED_CANADIAN_ABORIGINAL_SYLLABICS_EXTENDED = UnicodeBlock(name='Unified Canadian Aboriginal Syllabics Extended', start=0x18b0, end=0x18ff, assigned_ranges=[(0x18b0, 0x18f5)], aliases=['UCAS_Ext'])
|
|
59
|
+
LIMBU = UnicodeBlock(name='Limbu', start=0x1900, end=0x194f, assigned_ranges=[(0x1900, 0x191c), (0x1920, 0x192b), (0x1930, 0x193b), (0x1940, 0x1940), (0x1944, 0x194f)])
|
|
60
|
+
TAI_LE = UnicodeBlock(name='Tai Le', start=0x1950, end=0x197f, assigned_ranges=[(0x1950, 0x196d), (0x1970, 0x1974)])
|
|
61
|
+
NEW_TAI_LUE = UnicodeBlock(name='New Tai Lue', start=0x1980, end=0x19df, assigned_ranges=[(0x1980, 0x19ab), (0x19b0, 0x19c9), (0x19d0, 0x19da), (0x19de, 0x19df)])
|
|
62
|
+
KHMER_SYMBOLS = UnicodeBlock(name='Khmer Symbols', start=0x19e0, end=0x19ff, assigned_ranges=[(0x19e0, 0x19ff)])
|
|
63
|
+
BUGINESE = UnicodeBlock(name='Buginese', start=0x1a00, end=0x1a1f, assigned_ranges=[(0x1a00, 0x1a1b), (0x1a1e, 0x1a1f)])
|
|
64
|
+
TAI_THAM = UnicodeBlock(name='Tai Tham', start=0x1a20, end=0x1aaf, assigned_ranges=[(0x1a20, 0x1a5e), (0x1a60, 0x1a7c), (0x1a7f, 0x1a89), (0x1a90, 0x1a99), (0x1aa0, 0x1aad)])
|
|
65
|
+
BALINESE = UnicodeBlock(name='Balinese', start=0x1b00, end=0x1b7f, assigned_ranges=[(0x1b00, 0x1b4b), (0x1b50, 0x1b7c)])
|
|
66
|
+
SUNDANESE = UnicodeBlock(name='Sundanese', start=0x1b80, end=0x1bbf, assigned_ranges=[(0x1b80, 0x1bbf)])
|
|
67
|
+
BATAK = UnicodeBlock(name='Batak', start=0x1bc0, end=0x1bff, assigned_ranges=[(0x1bc0, 0x1bf3), (0x1bfc, 0x1bff)])
|
|
68
|
+
LEPCHA = UnicodeBlock(name='Lepcha', start=0x1c00, end=0x1c4f, assigned_ranges=[(0x1c00, 0x1c37), (0x1c3b, 0x1c49), (0x1c4d, 0x1c4f)])
|
|
69
|
+
OL_CHIKI = UnicodeBlock(name='Ol Chiki', start=0x1c50, end=0x1c7f, assigned_ranges=[(0x1c50, 0x1c7f)])
|
|
70
|
+
SUNDANESE_SUPPLEMENT = UnicodeBlock(name='Sundanese Supplement', start=0x1cc0, end=0x1ccf, assigned_ranges=[(0x1cc0, 0x1cc7)], aliases=['Sundanese_Sup'])
|
|
71
|
+
VEDIC_EXTENSIONS = UnicodeBlock(name='Vedic Extensions', start=0x1cd0, end=0x1cff, assigned_ranges=[(0x1cd0, 0x1cf6)], aliases=['Vedic_Ext'])
|
|
72
|
+
PHONETIC_EXTENSIONS = UnicodeBlock(name='Phonetic Extensions', start=0x1d00, end=0x1d7f, assigned_ranges=[(0x1d00, 0x1d7f)], aliases=['Phonetic_Ext'])
|
|
73
|
+
PHONETIC_EXTENSIONS_SUPPLEMENT = UnicodeBlock(name='Phonetic Extensions Supplement', start=0x1d80, end=0x1dbf, assigned_ranges=[(0x1d80, 0x1dbf)], aliases=['Phonetic_Ext_Sup'])
|
|
74
|
+
COMBINING_DIACRITICAL_MARKS_SUPPLEMENT = UnicodeBlock(name='Combining Diacritical Marks Supplement', start=0x1dc0, end=0x1dff, assigned_ranges=[(0x1dc0, 0x1de6), (0x1dfc, 0x1dff)], aliases=['Diacriticals_Sup'])
|
|
75
|
+
LATIN_EXTENDED_ADDITIONAL = UnicodeBlock(name='Latin Extended Additional', start=0x1e00, end=0x1eff, assigned_ranges=[(0x1e00, 0x1eff)], aliases=['Latin_Ext_Additional'])
|
|
76
|
+
GREEK_EXTENDED = UnicodeBlock(name='Greek Extended', start=0x1f00, end=0x1fff, assigned_ranges=[(0x1f00, 0x1f15), (0x1f18, 0x1f1d), (0x1f20, 0x1f45), (0x1f48, 0x1f4d), (0x1f50, 0x1f57), (0x1f59, 0x1f59), (0x1f5b, 0x1f5b), (0x1f5d, 0x1f5d), (0x1f5f, 0x1f7d), (0x1f80, 0x1fb4), (0x1fb6, 0x1fc4), (0x1fc6, 0x1fd3), (0x1fd6, 0x1fdb), (0x1fdd, 0x1fef), (0x1ff2, 0x1ff4), (0x1ff6, 0x1ffe)], aliases=['Greek_Ext'])
|
|
77
|
+
GENERAL_PUNCTUATION = UnicodeBlock(name='General Punctuation', start=0x2000, end=0x206f, assigned_ranges=[(0x2000, 0x2064), (0x206a, 0x206f)], aliases=['Punctuation'])
|
|
78
|
+
SUPERSCRIPTS_AND_SUBSCRIPTS = UnicodeBlock(name='Superscripts and Subscripts', start=0x2070, end=0x209f, assigned_ranges=[(0x2070, 0x2071), (0x2074, 0x208e), (0x2090, 0x209c)], aliases=['Super_And_Sub'])
|
|
79
|
+
CURRENCY_SYMBOLS = UnicodeBlock(name='Currency Symbols', start=0x20a0, end=0x20cf, assigned_ranges=[(0x20a0, 0x20b9)])
|
|
80
|
+
COMBINING_DIACRITICAL_MARKS_FOR_SYMBOLS = UnicodeBlock(name='Combining Diacritical Marks for Symbols', start=0x20d0, end=0x20ff, assigned_ranges=[(0x20d0, 0x20f0)], aliases=['Diacriticals_For_Symbols', 'Combining_Marks_For_Symbols'])
|
|
81
|
+
LETTERLIKE_SYMBOLS = UnicodeBlock(name='Letterlike Symbols', start=0x2100, end=0x214f, assigned_ranges=[(0x2100, 0x214f)])
|
|
82
|
+
NUMBER_FORMS = UnicodeBlock(name='Number Forms', start=0x2150, end=0x218f, assigned_ranges=[(0x2150, 0x2189)])
|
|
83
|
+
ARROWS = UnicodeBlock(name='Arrows', start=0x2190, end=0x21ff, assigned_ranges=[(0x2190, 0x21ff)])
|
|
84
|
+
MATHEMATICAL_OPERATORS = UnicodeBlock(name='Mathematical Operators', start=0x2200, end=0x22ff, assigned_ranges=[(0x2200, 0x22ff)], aliases=['Math_Operators'])
|
|
85
|
+
MISCELLANEOUS_TECHNICAL = UnicodeBlock(name='Miscellaneous Technical', start=0x2300, end=0x23ff, assigned_ranges=[(0x2300, 0x23f3)], aliases=['Misc_Technical'])
|
|
86
|
+
CONTROL_PICTURES = UnicodeBlock(name='Control Pictures', start=0x2400, end=0x243f, assigned_ranges=[(0x2400, 0x2426)])
|
|
87
|
+
OPTICAL_CHARACTER_RECOGNITION = UnicodeBlock(name='Optical Character Recognition', start=0x2440, end=0x245f, assigned_ranges=[(0x2440, 0x244a)], aliases=['OCR'])
|
|
88
|
+
ENCLOSED_ALPHANUMERICS = UnicodeBlock(name='Enclosed Alphanumerics', start=0x2460, end=0x24ff, assigned_ranges=[(0x2460, 0x24ff)], aliases=['Enclosed_Alphanum'])
|
|
89
|
+
BOX_DRAWING = UnicodeBlock(name='Box Drawing', start=0x2500, end=0x257f, assigned_ranges=[(0x2500, 0x257f)])
|
|
90
|
+
BLOCK_ELEMENTS = UnicodeBlock(name='Block Elements', start=0x2580, end=0x259f, assigned_ranges=[(0x2580, 0x259f)])
|
|
91
|
+
GEOMETRIC_SHAPES = UnicodeBlock(name='Geometric Shapes', start=0x25a0, end=0x25ff, assigned_ranges=[(0x25a0, 0x25ff)])
|
|
92
|
+
MISCELLANEOUS_SYMBOLS = UnicodeBlock(name='Miscellaneous Symbols', start=0x2600, end=0x26ff, assigned_ranges=[(0x2600, 0x26ff)], aliases=['Misc_Symbols'])
|
|
93
|
+
DINGBATS = UnicodeBlock(name='Dingbats', start=0x2700, end=0x27bf, assigned_ranges=[(0x2701, 0x27bf)])
|
|
94
|
+
MISCELLANEOUS_MATHEMATICAL_SYMBOLS_A = UnicodeBlock(name='Miscellaneous Mathematical Symbols-A', start=0x27c0, end=0x27ef, assigned_ranges=[(0x27c0, 0x27ef)], aliases=['Misc_Math_Symbols_A'])
|
|
95
|
+
SUPPLEMENTAL_ARROWS_A = UnicodeBlock(name='Supplemental Arrows-A', start=0x27f0, end=0x27ff, assigned_ranges=[(0x27f0, 0x27ff)], aliases=['Sup_Arrows_A'])
|
|
96
|
+
BRAILLE_PATTERNS = UnicodeBlock(name='Braille Patterns', start=0x2800, end=0x28ff, assigned_ranges=[(0x2800, 0x28ff)], aliases=['Braille'])
|
|
97
|
+
SUPPLEMENTAL_ARROWS_B = UnicodeBlock(name='Supplemental Arrows-B', start=0x2900, end=0x297f, assigned_ranges=[(0x2900, 0x297f)], aliases=['Sup_Arrows_B'])
|
|
98
|
+
MISCELLANEOUS_MATHEMATICAL_SYMBOLS_B = UnicodeBlock(name='Miscellaneous Mathematical Symbols-B', start=0x2980, end=0x29ff, assigned_ranges=[(0x2980, 0x29ff)], aliases=['Misc_Math_Symbols_B'])
|
|
99
|
+
SUPPLEMENTAL_MATHEMATICAL_OPERATORS = UnicodeBlock(name='Supplemental Mathematical Operators', start=0x2a00, end=0x2aff, assigned_ranges=[(0x2a00, 0x2aff)], aliases=['Sup_Math_Operators'])
|
|
100
|
+
MISCELLANEOUS_SYMBOLS_AND_ARROWS = UnicodeBlock(name='Miscellaneous Symbols and Arrows', start=0x2b00, end=0x2bff, assigned_ranges=[(0x2b00, 0x2b4c), (0x2b50, 0x2b59)], aliases=['Misc_Arrows'])
|
|
101
|
+
GLAGOLITIC = UnicodeBlock(name='Glagolitic', start=0x2c00, end=0x2c5f, assigned_ranges=[(0x2c00, 0x2c2e), (0x2c30, 0x2c5e)])
|
|
102
|
+
LATIN_EXTENDED_C = UnicodeBlock(name='Latin Extended-C', start=0x2c60, end=0x2c7f, assigned_ranges=[(0x2c60, 0x2c7f)], aliases=['Latin_Ext_C'])
|
|
103
|
+
COPTIC = UnicodeBlock(name='Coptic', start=0x2c80, end=0x2cff, assigned_ranges=[(0x2c80, 0x2cf3), (0x2cf9, 0x2cff)])
|
|
104
|
+
GEORGIAN_SUPPLEMENT = UnicodeBlock(name='Georgian Supplement', start=0x2d00, end=0x2d2f, assigned_ranges=[(0x2d00, 0x2d25), (0x2d27, 0x2d27), (0x2d2d, 0x2d2d)], aliases=['Georgian_Sup'])
|
|
105
|
+
TIFINAGH = UnicodeBlock(name='Tifinagh', start=0x2d30, end=0x2d7f, assigned_ranges=[(0x2d30, 0x2d67), (0x2d6f, 0x2d70), (0x2d7f, 0x2d7f)])
|
|
106
|
+
ETHIOPIC_EXTENDED = UnicodeBlock(name='Ethiopic Extended', start=0x2d80, end=0x2ddf, assigned_ranges=[(0x2d80, 0x2d96), (0x2da0, 0x2da6), (0x2da8, 0x2dae), (0x2db0, 0x2db6), (0x2db8, 0x2dbe), (0x2dc0, 0x2dc6), (0x2dc8, 0x2dce), (0x2dd0, 0x2dd6), (0x2dd8, 0x2dde)], aliases=['Ethiopic_Ext'])
|
|
107
|
+
CYRILLIC_EXTENDED_A = UnicodeBlock(name='Cyrillic Extended-A', start=0x2de0, end=0x2dff, assigned_ranges=[(0x2de0, 0x2dff)], aliases=['Cyrillic_Ext_A'])
|
|
108
|
+
SUPPLEMENTAL_PUNCTUATION = UnicodeBlock(name='Supplemental Punctuation', start=0x2e00, end=0x2e7f, assigned_ranges=[(0x2e00, 0x2e3b)], aliases=['Sup_Punctuation'])
|
|
109
|
+
CJK_RADICALS_SUPPLEMENT = UnicodeBlock(name='CJK Radicals Supplement', start=0x2e80, end=0x2eff, assigned_ranges=[(0x2e80, 0x2e99), (0x2e9b, 0x2ef3)], aliases=['CJK_Radicals_Sup'])
|
|
110
|
+
KANGXI_RADICALS = UnicodeBlock(name='Kangxi Radicals', start=0x2f00, end=0x2fdf, assigned_ranges=[(0x2f00, 0x2fd5)], aliases=['Kangxi'])
|
|
111
|
+
IDEOGRAPHIC_DESCRIPTION_CHARACTERS = UnicodeBlock(name='Ideographic Description Characters', start=0x2ff0, end=0x2fff, assigned_ranges=[(0x2ff0, 0x2ffb)], aliases=['IDC'])
|
|
112
|
+
CJK_SYMBOLS_AND_PUNCTUATION = UnicodeBlock(name='CJK Symbols and Punctuation', start=0x3000, end=0x303f, assigned_ranges=[(0x3000, 0x303f)], aliases=['CJK_Symbols'])
|
|
113
|
+
HIRAGANA = UnicodeBlock(name='Hiragana', start=0x3040, end=0x309f, assigned_ranges=[(0x3041, 0x3096), (0x3099, 0x309f)])
|
|
114
|
+
KATAKANA = UnicodeBlock(name='Katakana', start=0x30a0, end=0x30ff, assigned_ranges=[(0x30a0, 0x30ff)])
|
|
115
|
+
BOPOMOFO = UnicodeBlock(name='Bopomofo', start=0x3100, end=0x312f, assigned_ranges=[(0x3105, 0x312d)])
|
|
116
|
+
HANGUL_COMPATIBILITY_JAMO = UnicodeBlock(name='Hangul Compatibility Jamo', start=0x3130, end=0x318f, assigned_ranges=[(0x3131, 0x318e)], aliases=['Compat_Jamo'])
|
|
117
|
+
KANBUN = UnicodeBlock(name='Kanbun', start=0x3190, end=0x319f, assigned_ranges=[(0x3190, 0x319f)])
|
|
118
|
+
BOPOMOFO_EXTENDED = UnicodeBlock(name='Bopomofo Extended', start=0x31a0, end=0x31bf, assigned_ranges=[(0x31a0, 0x31ba)], aliases=['Bopomofo_Ext'])
|
|
119
|
+
CJK_STROKES = UnicodeBlock(name='CJK Strokes', start=0x31c0, end=0x31ef, assigned_ranges=[(0x31c0, 0x31e3)])
|
|
120
|
+
KATAKANA_PHONETIC_EXTENSIONS = UnicodeBlock(name='Katakana Phonetic Extensions', start=0x31f0, end=0x31ff, assigned_ranges=[(0x31f0, 0x31ff)], aliases=['Katakana_Ext'])
|
|
121
|
+
ENCLOSED_CJK_LETTERS_AND_MONTHS = UnicodeBlock(name='Enclosed CJK Letters and Months', start=0x3200, end=0x32ff, assigned_ranges=[(0x3200, 0x321e), (0x3220, 0x32fe)], aliases=['Enclosed_CJK'])
|
|
122
|
+
CJK_COMPATIBILITY = UnicodeBlock(name='CJK Compatibility', start=0x3300, end=0x33ff, assigned_ranges=[(0x3300, 0x33ff)], aliases=['CJK_Compat'])
|
|
123
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A = UnicodeBlock(name='CJK Unified Ideographs Extension A', start=0x3400, end=0x4dbf, assigned_ranges=[(0x3400, 0x4db5)], aliases=['CJK_Ext_A'])
|
|
124
|
+
YIJING_HEXAGRAM_SYMBOLS = UnicodeBlock(name='Yijing Hexagram Symbols', start=0x4dc0, end=0x4dff, assigned_ranges=[(0x4dc0, 0x4dff)], aliases=['Yijing'])
|
|
125
|
+
CJK_UNIFIED_IDEOGRAPHS = UnicodeBlock(name='CJK Unified Ideographs', start=0x4e00, end=0x9fff, assigned_ranges=[(0x4e00, 0x9fcc)], aliases=['CJK'])
|
|
126
|
+
YI_SYLLABLES = UnicodeBlock(name='Yi Syllables', start=0xa000, end=0xa48f, assigned_ranges=[(0xa000, 0xa48c)])
|
|
127
|
+
YI_RADICALS = UnicodeBlock(name='Yi Radicals', start=0xa490, end=0xa4cf, assigned_ranges=[(0xa490, 0xa4c6)])
|
|
128
|
+
LISU = UnicodeBlock(name='Lisu', start=0xa4d0, end=0xa4ff, assigned_ranges=[(0xa4d0, 0xa4ff)])
|
|
129
|
+
VAI = UnicodeBlock(name='Vai', start=0xa500, end=0xa63f, assigned_ranges=[(0xa500, 0xa62b)])
|
|
130
|
+
CYRILLIC_EXTENDED_B = UnicodeBlock(name='Cyrillic Extended-B', start=0xa640, end=0xa69f, assigned_ranges=[(0xa640, 0xa697), (0xa69f, 0xa69f)], aliases=['Cyrillic_Ext_B'])
|
|
131
|
+
BAMUM = UnicodeBlock(name='Bamum', start=0xa6a0, end=0xa6ff, assigned_ranges=[(0xa6a0, 0xa6f7)])
|
|
132
|
+
MODIFIER_TONE_LETTERS = UnicodeBlock(name='Modifier Tone Letters', start=0xa700, end=0xa71f, assigned_ranges=[(0xa700, 0xa71f)])
|
|
133
|
+
LATIN_EXTENDED_D = UnicodeBlock(name='Latin Extended-D', start=0xa720, end=0xa7ff, assigned_ranges=[(0xa720, 0xa78e), (0xa790, 0xa793), (0xa7a0, 0xa7aa), (0xa7f8, 0xa7ff)], aliases=['Latin_Ext_D'])
|
|
134
|
+
SYLOTI_NAGRI = UnicodeBlock(name='Syloti Nagri', start=0xa800, end=0xa82f, assigned_ranges=[(0xa800, 0xa82b)])
|
|
135
|
+
COMMON_INDIC_NUMBER_FORMS = UnicodeBlock(name='Common Indic Number Forms', start=0xa830, end=0xa83f, assigned_ranges=[(0xa830, 0xa839)], aliases=['Indic_Number_Forms'])
|
|
136
|
+
PHAGS_PA = UnicodeBlock(name='Phags-pa', start=0xa840, end=0xa87f, assigned_ranges=[(0xa840, 0xa877)])
|
|
137
|
+
SAURASHTRA = UnicodeBlock(name='Saurashtra', start=0xa880, end=0xa8df, assigned_ranges=[(0xa880, 0xa8c4), (0xa8ce, 0xa8d9)])
|
|
138
|
+
DEVANAGARI_EXTENDED = UnicodeBlock(name='Devanagari Extended', start=0xa8e0, end=0xa8ff, assigned_ranges=[(0xa8e0, 0xa8fb)], aliases=['Devanagari_Ext'])
|
|
139
|
+
KAYAH_LI = UnicodeBlock(name='Kayah Li', start=0xa900, end=0xa92f, assigned_ranges=[(0xa900, 0xa92f)])
|
|
140
|
+
REJANG = UnicodeBlock(name='Rejang', start=0xa930, end=0xa95f, assigned_ranges=[(0xa930, 0xa953), (0xa95f, 0xa95f)])
|
|
141
|
+
HANGUL_JAMO_EXTENDED_A = UnicodeBlock(name='Hangul Jamo Extended-A', start=0xa960, end=0xa97f, assigned_ranges=[(0xa960, 0xa97c)], aliases=['Jamo_Ext_A'])
|
|
142
|
+
JAVANESE = UnicodeBlock(name='Javanese', start=0xa980, end=0xa9df, assigned_ranges=[(0xa980, 0xa9cd), (0xa9cf, 0xa9d9), (0xa9de, 0xa9df)])
|
|
143
|
+
CHAM = UnicodeBlock(name='Cham', start=0xaa00, end=0xaa5f, assigned_ranges=[(0xaa00, 0xaa36), (0xaa40, 0xaa4d), (0xaa50, 0xaa59), (0xaa5c, 0xaa5f)])
|
|
144
|
+
MYANMAR_EXTENDED_A = UnicodeBlock(name='Myanmar Extended-A', start=0xaa60, end=0xaa7f, assigned_ranges=[(0xaa60, 0xaa7b)], aliases=['Myanmar_Ext_A'])
|
|
145
|
+
TAI_VIET = UnicodeBlock(name='Tai Viet', start=0xaa80, end=0xaadf, assigned_ranges=[(0xaa80, 0xaac2), (0xaadb, 0xaadf)])
|
|
146
|
+
MEETEI_MAYEK_EXTENSIONS = UnicodeBlock(name='Meetei Mayek Extensions', start=0xaae0, end=0xaaff, assigned_ranges=[(0xaae0, 0xaaf6)], aliases=['Meetei_Mayek_Ext'])
|
|
147
|
+
ETHIOPIC_EXTENDED_A = UnicodeBlock(name='Ethiopic Extended-A', start=0xab00, end=0xab2f, assigned_ranges=[(0xab01, 0xab06), (0xab09, 0xab0e), (0xab11, 0xab16), (0xab20, 0xab26), (0xab28, 0xab2e)], aliases=['Ethiopic_Ext_A'])
|
|
148
|
+
MEETEI_MAYEK = UnicodeBlock(name='Meetei Mayek', start=0xabc0, end=0xabff, assigned_ranges=[(0xabc0, 0xabed), (0xabf0, 0xabf9)])
|
|
149
|
+
HANGUL_SYLLABLES = UnicodeBlock(name='Hangul Syllables', start=0xac00, end=0xd7af, assigned_ranges=[(0xac00, 0xd7a3)], aliases=['Hangul'])
|
|
150
|
+
HANGUL_JAMO_EXTENDED_B = UnicodeBlock(name='Hangul Jamo Extended-B', start=0xd7b0, end=0xd7ff, assigned_ranges=[(0xd7b0, 0xd7c6), (0xd7cb, 0xd7fb)], aliases=['Jamo_Ext_B'])
|
|
151
|
+
HIGH_SURROGATES = UnicodeBlock(name='High Surrogates', start=0xd800, end=0xdb7f, assigned_ranges=[(0xd800, 0xdb7f)])
|
|
152
|
+
HIGH_PRIVATE_USE_SURROGATES = UnicodeBlock(name='High Private Use Surrogates', start=0xdb80, end=0xdbff, assigned_ranges=[(0xdb80, 0xdbff)], aliases=['High_PU_Surrogates'])
|
|
153
|
+
LOW_SURROGATES = UnicodeBlock(name='Low Surrogates', start=0xdc00, end=0xdfff, assigned_ranges=[(0xdc00, 0xdfff)])
|
|
154
|
+
PRIVATE_USE_AREA = UnicodeBlock(name='Private Use Area', start=0xe000, end=0xf8ff, assigned_ranges=[(0xe000, 0xf8ff)], aliases=['PUA', 'Private_Use'])
|
|
155
|
+
CJK_COMPATIBILITY_IDEOGRAPHS = UnicodeBlock(name='CJK Compatibility Ideographs', start=0xf900, end=0xfaff, assigned_ranges=[(0xf900, 0xfa6d), (0xfa70, 0xfad9)], aliases=['CJK_Compat_Ideographs'])
|
|
156
|
+
ALPHABETIC_PRESENTATION_FORMS = UnicodeBlock(name='Alphabetic Presentation Forms', start=0xfb00, end=0xfb4f, assigned_ranges=[(0xfb00, 0xfb06), (0xfb13, 0xfb17), (0xfb1d, 0xfb36), (0xfb38, 0xfb3c), (0xfb3e, 0xfb3e), (0xfb40, 0xfb41), (0xfb43, 0xfb44), (0xfb46, 0xfb4f)], aliases=['Alphabetic_PF'])
|
|
157
|
+
ARABIC_PRESENTATION_FORMS_A = UnicodeBlock(name='Arabic Presentation Forms-A', start=0xfb50, end=0xfdff, assigned_ranges=[(0xfb50, 0xfbc1), (0xfbd3, 0xfd3f), (0xfd50, 0xfd8f), (0xfd92, 0xfdc7), (0xfdf0, 0xfdfd)], aliases=['Arabic_PF_A', 'Arabic_Presentation_Forms-A'])
|
|
158
|
+
VARIATION_SELECTORS = UnicodeBlock(name='Variation Selectors', start=0xfe00, end=0xfe0f, assigned_ranges=[(0xfe00, 0xfe0f)], aliases=['VS'])
|
|
159
|
+
VERTICAL_FORMS = UnicodeBlock(name='Vertical Forms', start=0xfe10, end=0xfe1f, assigned_ranges=[(0xfe10, 0xfe19)])
|
|
160
|
+
COMBINING_HALF_MARKS = UnicodeBlock(name='Combining Half Marks', start=0xfe20, end=0xfe2f, assigned_ranges=[(0xfe20, 0xfe26)], aliases=['Half_Marks'])
|
|
161
|
+
CJK_COMPATIBILITY_FORMS = UnicodeBlock(name='CJK Compatibility Forms', start=0xfe30, end=0xfe4f, assigned_ranges=[(0xfe30, 0xfe4f)], aliases=['CJK_Compat_Forms'])
|
|
162
|
+
SMALL_FORM_VARIANTS = UnicodeBlock(name='Small Form Variants', start=0xfe50, end=0xfe6f, assigned_ranges=[(0xfe50, 0xfe52), (0xfe54, 0xfe66), (0xfe68, 0xfe6b)], aliases=['Small_Forms'])
|
|
163
|
+
ARABIC_PRESENTATION_FORMS_B = UnicodeBlock(name='Arabic Presentation Forms-B', start=0xfe70, end=0xfeff, assigned_ranges=[(0xfe70, 0xfe74), (0xfe76, 0xfefc), (0xfeff, 0xfeff)], aliases=['Arabic_PF_B'])
|
|
164
|
+
HALFWIDTH_AND_FULLWIDTH_FORMS = UnicodeBlock(name='Halfwidth and Fullwidth Forms', start=0xff00, end=0xffef, assigned_ranges=[(0xff01, 0xffbe), (0xffc2, 0xffc7), (0xffca, 0xffcf), (0xffd2, 0xffd7), (0xffda, 0xffdc), (0xffe0, 0xffe6), (0xffe8, 0xffee)], aliases=['Half_And_Full_Forms'])
|
|
165
|
+
SPECIALS = UnicodeBlock(name='Specials', start=0xfff0, end=0xffff, assigned_ranges=[(0xfff9, 0xfffd)])
|
|
166
|
+
LINEAR_B_SYLLABARY = UnicodeBlock(name='Linear B Syllabary', start=0x10000, end=0x1007f, assigned_ranges=[(0x10000, 0x1000b), (0x1000d, 0x10026), (0x10028, 0x1003a), (0x1003c, 0x1003d), (0x1003f, 0x1004d), (0x10050, 0x1005d)])
|
|
167
|
+
LINEAR_B_IDEOGRAMS = UnicodeBlock(name='Linear B Ideograms', start=0x10080, end=0x100ff, assigned_ranges=[(0x10080, 0x100fa)])
|
|
168
|
+
AEGEAN_NUMBERS = UnicodeBlock(name='Aegean Numbers', start=0x10100, end=0x1013f, assigned_ranges=[(0x10100, 0x10102), (0x10107, 0x10133), (0x10137, 0x1013f)])
|
|
169
|
+
ANCIENT_GREEK_NUMBERS = UnicodeBlock(name='Ancient Greek Numbers', start=0x10140, end=0x1018f, assigned_ranges=[(0x10140, 0x1018a)])
|
|
170
|
+
ANCIENT_SYMBOLS = UnicodeBlock(name='Ancient Symbols', start=0x10190, end=0x101cf, assigned_ranges=[(0x10190, 0x1019b)])
|
|
171
|
+
PHAISTOS_DISC = UnicodeBlock(name='Phaistos Disc', start=0x101d0, end=0x101ff, assigned_ranges=[(0x101d0, 0x101fd)], aliases=['Phaistos'])
|
|
172
|
+
LYCIAN = UnicodeBlock(name='Lycian', start=0x10280, end=0x1029f, assigned_ranges=[(0x10280, 0x1029c)])
|
|
173
|
+
CARIAN = UnicodeBlock(name='Carian', start=0x102a0, end=0x102df, assigned_ranges=[(0x102a0, 0x102d0)])
|
|
174
|
+
OLD_ITALIC = UnicodeBlock(name='Old Italic', start=0x10300, end=0x1032f, assigned_ranges=[(0x10300, 0x1031e), (0x10320, 0x10323)])
|
|
175
|
+
GOTHIC = UnicodeBlock(name='Gothic', start=0x10330, end=0x1034f, assigned_ranges=[(0x10330, 0x1034a)])
|
|
176
|
+
UGARITIC = UnicodeBlock(name='Ugaritic', start=0x10380, end=0x1039f, assigned_ranges=[(0x10380, 0x1039d), (0x1039f, 0x1039f)])
|
|
177
|
+
OLD_PERSIAN = UnicodeBlock(name='Old Persian', start=0x103a0, end=0x103df, assigned_ranges=[(0x103a0, 0x103c3), (0x103c8, 0x103d5)])
|
|
178
|
+
DESERET = UnicodeBlock(name='Deseret', start=0x10400, end=0x1044f, assigned_ranges=[(0x10400, 0x1044f)])
|
|
179
|
+
SHAVIAN = UnicodeBlock(name='Shavian', start=0x10450, end=0x1047f, assigned_ranges=[(0x10450, 0x1047f)])
|
|
180
|
+
OSMANYA = UnicodeBlock(name='Osmanya', start=0x10480, end=0x104af, assigned_ranges=[(0x10480, 0x1049d), (0x104a0, 0x104a9)])
|
|
181
|
+
CYPRIOT_SYLLABARY = UnicodeBlock(name='Cypriot Syllabary', start=0x10800, end=0x1083f, assigned_ranges=[(0x10800, 0x10805), (0x10808, 0x10808), (0x1080a, 0x10835), (0x10837, 0x10838), (0x1083c, 0x1083c), (0x1083f, 0x1083f)])
|
|
182
|
+
IMPERIAL_ARAMAIC = UnicodeBlock(name='Imperial Aramaic', start=0x10840, end=0x1085f, assigned_ranges=[(0x10840, 0x10855), (0x10857, 0x1085f)])
|
|
183
|
+
PHOENICIAN = UnicodeBlock(name='Phoenician', start=0x10900, end=0x1091f, assigned_ranges=[(0x10900, 0x1091b), (0x1091f, 0x1091f)])
|
|
184
|
+
LYDIAN = UnicodeBlock(name='Lydian', start=0x10920, end=0x1093f, assigned_ranges=[(0x10920, 0x10939), (0x1093f, 0x1093f)])
|
|
185
|
+
MEROITIC_HIEROGLYPHS = UnicodeBlock(name='Meroitic Hieroglyphs', start=0x10980, end=0x1099f, assigned_ranges=[(0x10980, 0x1099f)])
|
|
186
|
+
MEROITIC_CURSIVE = UnicodeBlock(name='Meroitic Cursive', start=0x109a0, end=0x109ff, assigned_ranges=[(0x109a0, 0x109b7), (0x109be, 0x109bf)])
|
|
187
|
+
KHAROSHTHI = UnicodeBlock(name='Kharoshthi', start=0x10a00, end=0x10a5f, assigned_ranges=[(0x10a00, 0x10a03), (0x10a05, 0x10a06), (0x10a0c, 0x10a13), (0x10a15, 0x10a17), (0x10a19, 0x10a33), (0x10a38, 0x10a3a), (0x10a3f, 0x10a47), (0x10a50, 0x10a58)])
|
|
188
|
+
OLD_SOUTH_ARABIAN = UnicodeBlock(name='Old South Arabian', start=0x10a60, end=0x10a7f, assigned_ranges=[(0x10a60, 0x10a7f)])
|
|
189
|
+
AVESTAN = UnicodeBlock(name='Avestan', start=0x10b00, end=0x10b3f, assigned_ranges=[(0x10b00, 0x10b35), (0x10b39, 0x10b3f)])
|
|
190
|
+
INSCRIPTIONAL_PARTHIAN = UnicodeBlock(name='Inscriptional Parthian', start=0x10b40, end=0x10b5f, assigned_ranges=[(0x10b40, 0x10b55), (0x10b58, 0x10b5f)])
|
|
191
|
+
INSCRIPTIONAL_PAHLAVI = UnicodeBlock(name='Inscriptional Pahlavi', start=0x10b60, end=0x10b7f, assigned_ranges=[(0x10b60, 0x10b72), (0x10b78, 0x10b7f)])
|
|
192
|
+
OLD_TURKIC = UnicodeBlock(name='Old Turkic', start=0x10c00, end=0x10c4f, assigned_ranges=[(0x10c00, 0x10c48)])
|
|
193
|
+
RUMI_NUMERAL_SYMBOLS = UnicodeBlock(name='Rumi Numeral Symbols', start=0x10e60, end=0x10e7f, assigned_ranges=[(0x10e60, 0x10e7e)], aliases=['Rumi'])
|
|
194
|
+
BRAHMI = UnicodeBlock(name='Brahmi', start=0x11000, end=0x1107f, assigned_ranges=[(0x11000, 0x1104d), (0x11052, 0x1106f)])
|
|
195
|
+
KAITHI = UnicodeBlock(name='Kaithi', start=0x11080, end=0x110cf, assigned_ranges=[(0x11080, 0x110c1)])
|
|
196
|
+
SORA_SOMPENG = UnicodeBlock(name='Sora Sompeng', start=0x110d0, end=0x110ff, assigned_ranges=[(0x110d0, 0x110e8), (0x110f0, 0x110f9)])
|
|
197
|
+
CHAKMA = UnicodeBlock(name='Chakma', start=0x11100, end=0x1114f, assigned_ranges=[(0x11100, 0x11134), (0x11136, 0x11143)])
|
|
198
|
+
SHARADA = UnicodeBlock(name='Sharada', start=0x11180, end=0x111df, assigned_ranges=[(0x11180, 0x111c8), (0x111d0, 0x111d9)])
|
|
199
|
+
TAKRI = UnicodeBlock(name='Takri', start=0x11680, end=0x116cf, assigned_ranges=[(0x11680, 0x116b7), (0x116c0, 0x116c9)])
|
|
200
|
+
CUNEIFORM = UnicodeBlock(name='Cuneiform', start=0x12000, end=0x123ff, assigned_ranges=[(0x12000, 0x1236e)])
|
|
201
|
+
CUNEIFORM_NUMBERS_AND_PUNCTUATION = UnicodeBlock(name='Cuneiform Numbers and Punctuation', start=0x12400, end=0x1247f, assigned_ranges=[(0x12400, 0x12462), (0x12470, 0x12473)], aliases=['Cuneiform_Numbers'])
|
|
202
|
+
EGYPTIAN_HIEROGLYPHS = UnicodeBlock(name='Egyptian Hieroglyphs', start=0x13000, end=0x1342f, assigned_ranges=[(0x13000, 0x1342e)])
|
|
203
|
+
BAMUM_SUPPLEMENT = UnicodeBlock(name='Bamum Supplement', start=0x16800, end=0x16a3f, assigned_ranges=[(0x16800, 0x16a38)], aliases=['Bamum_Sup'])
|
|
204
|
+
MIAO = UnicodeBlock(name='Miao', start=0x16f00, end=0x16f9f, assigned_ranges=[(0x16f00, 0x16f44), (0x16f50, 0x16f7e), (0x16f8f, 0x16f9f)])
|
|
205
|
+
KANA_SUPPLEMENT = UnicodeBlock(name='Kana Supplement', start=0x1b000, end=0x1b0ff, assigned_ranges=[(0x1b000, 0x1b001)], aliases=['Kana_Sup'])
|
|
206
|
+
BYZANTINE_MUSICAL_SYMBOLS = UnicodeBlock(name='Byzantine Musical Symbols', start=0x1d000, end=0x1d0ff, assigned_ranges=[(0x1d000, 0x1d0f5)], aliases=['Byzantine_Music'])
|
|
207
|
+
MUSICAL_SYMBOLS = UnicodeBlock(name='Musical Symbols', start=0x1d100, end=0x1d1ff, assigned_ranges=[(0x1d100, 0x1d126), (0x1d129, 0x1d1dd)], aliases=['Music'])
|
|
208
|
+
ANCIENT_GREEK_MUSICAL_NOTATION = UnicodeBlock(name='Ancient Greek Musical Notation', start=0x1d200, end=0x1d24f, assigned_ranges=[(0x1d200, 0x1d245)], aliases=['Ancient_Greek_Music'])
|
|
209
|
+
TAI_XUAN_JING_SYMBOLS = UnicodeBlock(name='Tai Xuan Jing Symbols', start=0x1d300, end=0x1d35f, assigned_ranges=[(0x1d300, 0x1d356)], aliases=['Tai_Xuan_Jing'])
|
|
210
|
+
COUNTING_ROD_NUMERALS = UnicodeBlock(name='Counting Rod Numerals', start=0x1d360, end=0x1d37f, assigned_ranges=[(0x1d360, 0x1d371)], aliases=['Counting_Rod'])
|
|
211
|
+
MATHEMATICAL_ALPHANUMERIC_SYMBOLS = UnicodeBlock(name='Mathematical Alphanumeric Symbols', start=0x1d400, end=0x1d7ff, assigned_ranges=[(0x1d400, 0x1d454), (0x1d456, 0x1d49c), (0x1d49e, 0x1d49f), (0x1d4a2, 0x1d4a2), (0x1d4a5, 0x1d4a6), (0x1d4a9, 0x1d4ac), (0x1d4ae, 0x1d4b9), (0x1d4bb, 0x1d4bb), (0x1d4bd, 0x1d4c3), (0x1d4c5, 0x1d505), (0x1d507, 0x1d50a), (0x1d50d, 0x1d514), (0x1d516, 0x1d51c), (0x1d51e, 0x1d539), (0x1d53b, 0x1d53e), (0x1d540, 0x1d544), (0x1d546, 0x1d546), (0x1d54a, 0x1d550), (0x1d552, 0x1d6a5), (0x1d6a8, 0x1d7cb), (0x1d7ce, 0x1d7ff)], aliases=['Math_Alphanum'])
|
|
212
|
+
ARABIC_MATHEMATICAL_ALPHABETIC_SYMBOLS = UnicodeBlock(name='Arabic Mathematical Alphabetic Symbols', start=0x1ee00, end=0x1eeff, assigned_ranges=[(0x1ee00, 0x1ee03), (0x1ee05, 0x1ee1f), (0x1ee21, 0x1ee22), (0x1ee24, 0x1ee24), (0x1ee27, 0x1ee27), (0x1ee29, 0x1ee32), (0x1ee34, 0x1ee37), (0x1ee39, 0x1ee39), (0x1ee3b, 0x1ee3b), (0x1ee42, 0x1ee42), (0x1ee47, 0x1ee47), (0x1ee49, 0x1ee49), (0x1ee4b, 0x1ee4b), (0x1ee4d, 0x1ee4f), (0x1ee51, 0x1ee52), (0x1ee54, 0x1ee54), (0x1ee57, 0x1ee57), (0x1ee59, 0x1ee59), (0x1ee5b, 0x1ee5b), (0x1ee5d, 0x1ee5d), (0x1ee5f, 0x1ee5f), (0x1ee61, 0x1ee62), (0x1ee64, 0x1ee64), (0x1ee67, 0x1ee6a), (0x1ee6c, 0x1ee72), (0x1ee74, 0x1ee77), (0x1ee79, 0x1ee7c), (0x1ee7e, 0x1ee7e), (0x1ee80, 0x1ee89), (0x1ee8b, 0x1ee9b), (0x1eea1, 0x1eea3), (0x1eea5, 0x1eea9), (0x1eeab, 0x1eebb), (0x1eef0, 0x1eef1)], aliases=['Arabic_Math'])
|
|
213
|
+
MAHJONG_TILES = UnicodeBlock(name='Mahjong Tiles', start=0x1f000, end=0x1f02f, assigned_ranges=[(0x1f000, 0x1f02b)], aliases=['Mahjong'])
|
|
214
|
+
DOMINO_TILES = UnicodeBlock(name='Domino Tiles', start=0x1f030, end=0x1f09f, assigned_ranges=[(0x1f030, 0x1f093)], aliases=['Domino'])
|
|
215
|
+
PLAYING_CARDS = UnicodeBlock(name='Playing Cards', start=0x1f0a0, end=0x1f0ff, assigned_ranges=[(0x1f0a0, 0x1f0ae), (0x1f0b1, 0x1f0be), (0x1f0c1, 0x1f0cf), (0x1f0d1, 0x1f0df)])
|
|
216
|
+
ENCLOSED_ALPHANUMERIC_SUPPLEMENT = UnicodeBlock(name='Enclosed Alphanumeric Supplement', start=0x1f100, end=0x1f1ff, assigned_ranges=[(0x1f100, 0x1f10a), (0x1f110, 0x1f12e), (0x1f130, 0x1f16b), (0x1f170, 0x1f19a), (0x1f1e6, 0x1f1ff)], aliases=['Enclosed_Alphanum_Sup'])
|
|
217
|
+
ENCLOSED_IDEOGRAPHIC_SUPPLEMENT = UnicodeBlock(name='Enclosed Ideographic Supplement', start=0x1f200, end=0x1f2ff, assigned_ranges=[(0x1f200, 0x1f202), (0x1f210, 0x1f23a), (0x1f240, 0x1f248), (0x1f250, 0x1f251)], aliases=['Enclosed_Ideographic_Sup'])
|
|
218
|
+
MISCELLANEOUS_SYMBOLS_AND_PICTOGRAPHS = UnicodeBlock(name='Miscellaneous Symbols And Pictographs', start=0x1f300, end=0x1f5ff, assigned_ranges=[(0x1f300, 0x1f320), (0x1f330, 0x1f335), (0x1f337, 0x1f37c), (0x1f380, 0x1f393), (0x1f3a0, 0x1f3c4), (0x1f3c6, 0x1f3ca), (0x1f3e0, 0x1f3f0), (0x1f400, 0x1f43e), (0x1f440, 0x1f440), (0x1f442, 0x1f4f7), (0x1f4f9, 0x1f4fc), (0x1f500, 0x1f53d), (0x1f540, 0x1f543), (0x1f550, 0x1f567), (0x1f5fb, 0x1f5ff)], aliases=['Misc_Pictographs'])
|
|
219
|
+
EMOTICONS = UnicodeBlock(name='Emoticons', start=0x1f600, end=0x1f64f, assigned_ranges=[(0x1f600, 0x1f640), (0x1f645, 0x1f64f)])
|
|
220
|
+
TRANSPORT_AND_MAP_SYMBOLS = UnicodeBlock(name='Transport And Map Symbols', start=0x1f680, end=0x1f6ff, assigned_ranges=[(0x1f680, 0x1f6c5)], aliases=['Transport_And_Map'])
|
|
221
|
+
ALCHEMICAL_SYMBOLS = UnicodeBlock(name='Alchemical Symbols', start=0x1f700, end=0x1f77f, assigned_ranges=[(0x1f700, 0x1f773)], aliases=['Alchemical'])
|
|
222
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B = UnicodeBlock(name='CJK Unified Ideographs Extension B', start=0x20000, end=0x2a6df, assigned_ranges=[(0x20000, 0x2a6d6)], aliases=['CJK_Ext_B'])
|
|
223
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_C = UnicodeBlock(name='CJK Unified Ideographs Extension C', start=0x2a700, end=0x2b73f, assigned_ranges=[(0x2a700, 0x2b734)], aliases=['CJK_Ext_C'])
|
|
224
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_D = UnicodeBlock(name='CJK Unified Ideographs Extension D', start=0x2b740, end=0x2b81f, assigned_ranges=[(0x2b740, 0x2b81d)], aliases=['CJK_Ext_D'])
|
|
225
|
+
CJK_COMPATIBILITY_IDEOGRAPHS_SUPPLEMENT = UnicodeBlock(name='CJK Compatibility Ideographs Supplement', start=0x2f800, end=0x2fa1f, assigned_ranges=[(0x2f800, 0x2fa1d)], aliases=['CJK_Compat_Ideographs_Sup'])
|
|
226
|
+
TAGS = UnicodeBlock(name='Tags', start=0xe0000, end=0xe007f, assigned_ranges=[(0xe0001, 0xe0001), (0xe0020, 0xe007f)])
|
|
227
|
+
VARIATION_SELECTORS_SUPPLEMENT = UnicodeBlock(name='Variation Selectors Supplement', start=0xe0100, end=0xe01ef, assigned_ranges=[(0xe0100, 0xe01ef)], aliases=['VS_Sup'])
|
|
228
|
+
SUPPLEMENTARY_PRIVATE_USE_AREA_A = UnicodeBlock(name='Supplementary Private Use Area-A', start=0xf0000, end=0xfffff, assigned_ranges=[(0xf0000, 0xffffd)], aliases=['Sup_PUA_A'])
|
|
229
|
+
SUPPLEMENTARY_PRIVATE_USE_AREA_B = UnicodeBlock(name='Supplementary Private Use Area-B', start=0x100000, end=0x10ffff, assigned_ranges=[(0x100000, 0x10fffd)], aliases=['Sup_PUA_B'])
|
|
230
|
+
|
|
231
|
+
ALL_BLOCKS = [
|
|
232
|
+
BASIC_LATIN,
|
|
233
|
+
LATIN_1_SUPPLEMENT,
|
|
234
|
+
LATIN_EXTENDED_A,
|
|
235
|
+
LATIN_EXTENDED_B,
|
|
236
|
+
IPA_EXTENSIONS,
|
|
237
|
+
SPACING_MODIFIER_LETTERS,
|
|
238
|
+
COMBINING_DIACRITICAL_MARKS,
|
|
239
|
+
GREEK_AND_COPTIC,
|
|
240
|
+
CYRILLIC,
|
|
241
|
+
CYRILLIC_SUPPLEMENT,
|
|
242
|
+
ARMENIAN,
|
|
243
|
+
HEBREW,
|
|
244
|
+
ARABIC,
|
|
245
|
+
SYRIAC,
|
|
246
|
+
ARABIC_SUPPLEMENT,
|
|
247
|
+
THAANA,
|
|
248
|
+
NKO,
|
|
249
|
+
SAMARITAN,
|
|
250
|
+
MANDAIC,
|
|
251
|
+
ARABIC_EXTENDED_A,
|
|
252
|
+
DEVANAGARI,
|
|
253
|
+
BENGALI,
|
|
254
|
+
GURMUKHI,
|
|
255
|
+
GUJARATI,
|
|
256
|
+
ORIYA,
|
|
257
|
+
TAMIL,
|
|
258
|
+
TELUGU,
|
|
259
|
+
KANNADA,
|
|
260
|
+
MALAYALAM,
|
|
261
|
+
SINHALA,
|
|
262
|
+
THAI,
|
|
263
|
+
LAO,
|
|
264
|
+
TIBETAN,
|
|
265
|
+
MYANMAR,
|
|
266
|
+
GEORGIAN,
|
|
267
|
+
HANGUL_JAMO,
|
|
268
|
+
ETHIOPIC,
|
|
269
|
+
ETHIOPIC_SUPPLEMENT,
|
|
270
|
+
CHEROKEE,
|
|
271
|
+
UNIFIED_CANADIAN_ABORIGINAL_SYLLABICS,
|
|
272
|
+
OGHAM,
|
|
273
|
+
RUNIC,
|
|
274
|
+
TAGALOG,
|
|
275
|
+
HANUNOO,
|
|
276
|
+
BUHID,
|
|
277
|
+
TAGBANWA,
|
|
278
|
+
KHMER,
|
|
279
|
+
MONGOLIAN,
|
|
280
|
+
UNIFIED_CANADIAN_ABORIGINAL_SYLLABICS_EXTENDED,
|
|
281
|
+
LIMBU,
|
|
282
|
+
TAI_LE,
|
|
283
|
+
NEW_TAI_LUE,
|
|
284
|
+
KHMER_SYMBOLS,
|
|
285
|
+
BUGINESE,
|
|
286
|
+
TAI_THAM,
|
|
287
|
+
BALINESE,
|
|
288
|
+
SUNDANESE,
|
|
289
|
+
BATAK,
|
|
290
|
+
LEPCHA,
|
|
291
|
+
OL_CHIKI,
|
|
292
|
+
SUNDANESE_SUPPLEMENT,
|
|
293
|
+
VEDIC_EXTENSIONS,
|
|
294
|
+
PHONETIC_EXTENSIONS,
|
|
295
|
+
PHONETIC_EXTENSIONS_SUPPLEMENT,
|
|
296
|
+
COMBINING_DIACRITICAL_MARKS_SUPPLEMENT,
|
|
297
|
+
LATIN_EXTENDED_ADDITIONAL,
|
|
298
|
+
GREEK_EXTENDED,
|
|
299
|
+
GENERAL_PUNCTUATION,
|
|
300
|
+
SUPERSCRIPTS_AND_SUBSCRIPTS,
|
|
301
|
+
CURRENCY_SYMBOLS,
|
|
302
|
+
COMBINING_DIACRITICAL_MARKS_FOR_SYMBOLS,
|
|
303
|
+
LETTERLIKE_SYMBOLS,
|
|
304
|
+
NUMBER_FORMS,
|
|
305
|
+
ARROWS,
|
|
306
|
+
MATHEMATICAL_OPERATORS,
|
|
307
|
+
MISCELLANEOUS_TECHNICAL,
|
|
308
|
+
CONTROL_PICTURES,
|
|
309
|
+
OPTICAL_CHARACTER_RECOGNITION,
|
|
310
|
+
ENCLOSED_ALPHANUMERICS,
|
|
311
|
+
BOX_DRAWING,
|
|
312
|
+
BLOCK_ELEMENTS,
|
|
313
|
+
GEOMETRIC_SHAPES,
|
|
314
|
+
MISCELLANEOUS_SYMBOLS,
|
|
315
|
+
DINGBATS,
|
|
316
|
+
MISCELLANEOUS_MATHEMATICAL_SYMBOLS_A,
|
|
317
|
+
SUPPLEMENTAL_ARROWS_A,
|
|
318
|
+
BRAILLE_PATTERNS,
|
|
319
|
+
SUPPLEMENTAL_ARROWS_B,
|
|
320
|
+
MISCELLANEOUS_MATHEMATICAL_SYMBOLS_B,
|
|
321
|
+
SUPPLEMENTAL_MATHEMATICAL_OPERATORS,
|
|
322
|
+
MISCELLANEOUS_SYMBOLS_AND_ARROWS,
|
|
323
|
+
GLAGOLITIC,
|
|
324
|
+
LATIN_EXTENDED_C,
|
|
325
|
+
COPTIC,
|
|
326
|
+
GEORGIAN_SUPPLEMENT,
|
|
327
|
+
TIFINAGH,
|
|
328
|
+
ETHIOPIC_EXTENDED,
|
|
329
|
+
CYRILLIC_EXTENDED_A,
|
|
330
|
+
SUPPLEMENTAL_PUNCTUATION,
|
|
331
|
+
CJK_RADICALS_SUPPLEMENT,
|
|
332
|
+
KANGXI_RADICALS,
|
|
333
|
+
IDEOGRAPHIC_DESCRIPTION_CHARACTERS,
|
|
334
|
+
CJK_SYMBOLS_AND_PUNCTUATION,
|
|
335
|
+
HIRAGANA,
|
|
336
|
+
KATAKANA,
|
|
337
|
+
BOPOMOFO,
|
|
338
|
+
HANGUL_COMPATIBILITY_JAMO,
|
|
339
|
+
KANBUN,
|
|
340
|
+
BOPOMOFO_EXTENDED,
|
|
341
|
+
CJK_STROKES,
|
|
342
|
+
KATAKANA_PHONETIC_EXTENSIONS,
|
|
343
|
+
ENCLOSED_CJK_LETTERS_AND_MONTHS,
|
|
344
|
+
CJK_COMPATIBILITY,
|
|
345
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A,
|
|
346
|
+
YIJING_HEXAGRAM_SYMBOLS,
|
|
347
|
+
CJK_UNIFIED_IDEOGRAPHS,
|
|
348
|
+
YI_SYLLABLES,
|
|
349
|
+
YI_RADICALS,
|
|
350
|
+
LISU,
|
|
351
|
+
VAI,
|
|
352
|
+
CYRILLIC_EXTENDED_B,
|
|
353
|
+
BAMUM,
|
|
354
|
+
MODIFIER_TONE_LETTERS,
|
|
355
|
+
LATIN_EXTENDED_D,
|
|
356
|
+
SYLOTI_NAGRI,
|
|
357
|
+
COMMON_INDIC_NUMBER_FORMS,
|
|
358
|
+
PHAGS_PA,
|
|
359
|
+
SAURASHTRA,
|
|
360
|
+
DEVANAGARI_EXTENDED,
|
|
361
|
+
KAYAH_LI,
|
|
362
|
+
REJANG,
|
|
363
|
+
HANGUL_JAMO_EXTENDED_A,
|
|
364
|
+
JAVANESE,
|
|
365
|
+
CHAM,
|
|
366
|
+
MYANMAR_EXTENDED_A,
|
|
367
|
+
TAI_VIET,
|
|
368
|
+
MEETEI_MAYEK_EXTENSIONS,
|
|
369
|
+
ETHIOPIC_EXTENDED_A,
|
|
370
|
+
MEETEI_MAYEK,
|
|
371
|
+
HANGUL_SYLLABLES,
|
|
372
|
+
HANGUL_JAMO_EXTENDED_B,
|
|
373
|
+
HIGH_SURROGATES,
|
|
374
|
+
HIGH_PRIVATE_USE_SURROGATES,
|
|
375
|
+
LOW_SURROGATES,
|
|
376
|
+
PRIVATE_USE_AREA,
|
|
377
|
+
CJK_COMPATIBILITY_IDEOGRAPHS,
|
|
378
|
+
ALPHABETIC_PRESENTATION_FORMS,
|
|
379
|
+
ARABIC_PRESENTATION_FORMS_A,
|
|
380
|
+
VARIATION_SELECTORS,
|
|
381
|
+
VERTICAL_FORMS,
|
|
382
|
+
COMBINING_HALF_MARKS,
|
|
383
|
+
CJK_COMPATIBILITY_FORMS,
|
|
384
|
+
SMALL_FORM_VARIANTS,
|
|
385
|
+
ARABIC_PRESENTATION_FORMS_B,
|
|
386
|
+
HALFWIDTH_AND_FULLWIDTH_FORMS,
|
|
387
|
+
SPECIALS,
|
|
388
|
+
LINEAR_B_SYLLABARY,
|
|
389
|
+
LINEAR_B_IDEOGRAMS,
|
|
390
|
+
AEGEAN_NUMBERS,
|
|
391
|
+
ANCIENT_GREEK_NUMBERS,
|
|
392
|
+
ANCIENT_SYMBOLS,
|
|
393
|
+
PHAISTOS_DISC,
|
|
394
|
+
LYCIAN,
|
|
395
|
+
CARIAN,
|
|
396
|
+
OLD_ITALIC,
|
|
397
|
+
GOTHIC,
|
|
398
|
+
UGARITIC,
|
|
399
|
+
OLD_PERSIAN,
|
|
400
|
+
DESERET,
|
|
401
|
+
SHAVIAN,
|
|
402
|
+
OSMANYA,
|
|
403
|
+
CYPRIOT_SYLLABARY,
|
|
404
|
+
IMPERIAL_ARAMAIC,
|
|
405
|
+
PHOENICIAN,
|
|
406
|
+
LYDIAN,
|
|
407
|
+
MEROITIC_HIEROGLYPHS,
|
|
408
|
+
MEROITIC_CURSIVE,
|
|
409
|
+
KHAROSHTHI,
|
|
410
|
+
OLD_SOUTH_ARABIAN,
|
|
411
|
+
AVESTAN,
|
|
412
|
+
INSCRIPTIONAL_PARTHIAN,
|
|
413
|
+
INSCRIPTIONAL_PAHLAVI,
|
|
414
|
+
OLD_TURKIC,
|
|
415
|
+
RUMI_NUMERAL_SYMBOLS,
|
|
416
|
+
BRAHMI,
|
|
417
|
+
KAITHI,
|
|
418
|
+
SORA_SOMPENG,
|
|
419
|
+
CHAKMA,
|
|
420
|
+
SHARADA,
|
|
421
|
+
TAKRI,
|
|
422
|
+
CUNEIFORM,
|
|
423
|
+
CUNEIFORM_NUMBERS_AND_PUNCTUATION,
|
|
424
|
+
EGYPTIAN_HIEROGLYPHS,
|
|
425
|
+
BAMUM_SUPPLEMENT,
|
|
426
|
+
MIAO,
|
|
427
|
+
KANA_SUPPLEMENT,
|
|
428
|
+
BYZANTINE_MUSICAL_SYMBOLS,
|
|
429
|
+
MUSICAL_SYMBOLS,
|
|
430
|
+
ANCIENT_GREEK_MUSICAL_NOTATION,
|
|
431
|
+
TAI_XUAN_JING_SYMBOLS,
|
|
432
|
+
COUNTING_ROD_NUMERALS,
|
|
433
|
+
MATHEMATICAL_ALPHANUMERIC_SYMBOLS,
|
|
434
|
+
ARABIC_MATHEMATICAL_ALPHABETIC_SYMBOLS,
|
|
435
|
+
MAHJONG_TILES,
|
|
436
|
+
DOMINO_TILES,
|
|
437
|
+
PLAYING_CARDS,
|
|
438
|
+
ENCLOSED_ALPHANUMERIC_SUPPLEMENT,
|
|
439
|
+
ENCLOSED_IDEOGRAPHIC_SUPPLEMENT,
|
|
440
|
+
MISCELLANEOUS_SYMBOLS_AND_PICTOGRAPHS,
|
|
441
|
+
EMOTICONS,
|
|
442
|
+
TRANSPORT_AND_MAP_SYMBOLS,
|
|
443
|
+
ALCHEMICAL_SYMBOLS,
|
|
444
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B,
|
|
445
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_C,
|
|
446
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_D,
|
|
447
|
+
CJK_COMPATIBILITY_IDEOGRAPHS_SUPPLEMENT,
|
|
448
|
+
TAGS,
|
|
449
|
+
VARIATION_SELECTORS_SUPPLEMENT,
|
|
450
|
+
SUPPLEMENTARY_PRIVATE_USE_AREA_A,
|
|
451
|
+
SUPPLEMENTARY_PRIVATE_USE_AREA_B,
|
|
452
|
+
]
|
|
453
|
+
|
|
454
|
+
IDEO_BLOCKS = [
|
|
455
|
+
KANGXI_RADICALS,
|
|
456
|
+
CJK_RADICALS_SUPPLEMENT,
|
|
457
|
+
CJK_COMPATIBILITY_IDEOGRAPHS,
|
|
458
|
+
CJK_COMPATIBILITY_IDEOGRAPHS_SUPPLEMENT,
|
|
459
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A,
|
|
460
|
+
CJK_UNIFIED_IDEOGRAPHS,
|
|
461
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B,
|
|
462
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_C,
|
|
463
|
+
CJK_UNIFIED_IDEOGRAPHS_EXTENSION_D,
|
|
464
|
+
]
|
|
465
|
+
|
|
466
|
+
JPAN_BLOCKS = [
|
|
467
|
+
CJK_COMPATIBILITY,
|
|
468
|
+
HIRAGANA,
|
|
469
|
+
KATAKANA,
|
|
470
|
+
KATAKANA_PHONETIC_EXTENSIONS,
|
|
471
|
+
KANA_SUPPLEMENT,
|
|
472
|
+
]
|
|
473
|
+
|
|
474
|
+
KORE_BLOCKS = [
|
|
475
|
+
HANGUL_SYLLABLES,
|
|
476
|
+
HANGUL_JAMO,
|
|
477
|
+
HANGUL_COMPATIBILITY_JAMO,
|
|
478
|
+
HANGUL_JAMO_EXTENDED_A,
|
|
479
|
+
HANGUL_JAMO_EXTENDED_B,
|
|
480
|
+
]
|
|
481
|
+
|
|
482
|
+
PUNC_BLOCKS = [
|
|
483
|
+
CJK_COMPATIBILITY,
|
|
484
|
+
CJK_COMPATIBILITY_FORMS,
|
|
485
|
+
CJK_STROKES,
|
|
486
|
+
CJK_SYMBOLS_AND_PUNCTUATION,
|
|
487
|
+
HALFWIDTH_AND_FULLWIDTH_FORMS,
|
|
488
|
+
ENCLOSED_CJK_LETTERS_AND_MONTHS,
|
|
489
|
+
ENCLOSED_IDEOGRAPHIC_SUPPLEMENT,
|
|
490
|
+
IDEOGRAPHIC_DESCRIPTION_CHARACTERS,
|
|
491
|
+
]
|
|
492
|
+
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
class CharNormaliser:
|
|
2
|
+
@staticmethod
|
|
3
|
+
def to_codepoint(char: int | str | bytes) -> int:
|
|
4
|
+
"""Convert a character to its Unicode code point.
|
|
5
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
6
|
+
Returns the Unicode code point as an integer.
|
|
7
|
+
"""
|
|
8
|
+
if isinstance(char, int):
|
|
9
|
+
unidec = char
|
|
10
|
+
elif isinstance(char, bytes):
|
|
11
|
+
unidec = ord(char.decode("utf-8")[0])
|
|
12
|
+
elif isinstance(char, str):
|
|
13
|
+
unidec = ord(char[0])
|
|
14
|
+
else:
|
|
15
|
+
raise TypeError(f"Expected str, int, or bytes, got {type(char).__name__}")
|
|
16
|
+
assert 0 <= unidec < 0x110000, "Code point must be in the range [0, 0x10FFFF]"
|
|
17
|
+
return unidec
|
unicode_blocks/cjk.py
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
from .unicodeBlock import UnicodeBlock
|
|
2
|
+
from .charNormaliser import CharNormaliser
|
|
3
|
+
from .blocks import IDEO_BLOCKS, JPAN_BLOCKS, KORE_BLOCKS, PUNC_BLOCKS
|
|
4
|
+
|
|
5
|
+
CJK_BLOCKS = IDEO_BLOCKS + JPAN_BLOCKS + KORE_BLOCKS + PUNC_BLOCKS
|
|
6
|
+
|
|
7
|
+
### Individual character checks ###
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def is_in_blocks(char: str | int | bytes, blocks: list[UnicodeBlock]) -> bool:
|
|
11
|
+
unidec = CharNormaliser.to_codepoint(char)
|
|
12
|
+
return any(unidec in block for block in blocks)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def is_ideographic_zero(char: str | int | bytes) -> bool:
|
|
16
|
+
"""Check if a character is ideographic zero (U+3007).
|
|
17
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
18
|
+
"""
|
|
19
|
+
return CharNormaliser.to_codepoint(char) == 0x3007
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def is_ideographic(char: str | int | bytes) -> bool:
|
|
23
|
+
"""Check if a character is ideographic (hanzi/kanji/hanja/漢字/汉字).
|
|
24
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
25
|
+
"""
|
|
26
|
+
return is_ideographic_zero(char) or is_in_blocks(char, IDEO_BLOCKS)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def is_japanese_kana(char: str | int | bytes) -> bool:
|
|
30
|
+
"""Check if a character is Japanese kana (ひらがな/カタカナ).
|
|
31
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
32
|
+
"""
|
|
33
|
+
return is_in_blocks(char, JPAN_BLOCKS)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def is_japanese(char: str | int | bytes) -> bool:
|
|
37
|
+
"""Check if a character is Japanese (kanji + kana).
|
|
38
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
39
|
+
"""
|
|
40
|
+
return is_japanese_kana(char) or is_ideographic(char)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def is_korean_hangul(char: str | int | bytes) -> bool:
|
|
44
|
+
"""Check if a character is Korean hangul (한글).
|
|
45
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
46
|
+
"""
|
|
47
|
+
return is_in_blocks(char, KORE_BLOCKS)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def is_korean(char: str | int | bytes) -> bool:
|
|
51
|
+
"""Check if a character is Korean (hanja + hangul).
|
|
52
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
53
|
+
"""
|
|
54
|
+
return is_korean_hangul(char) or is_ideographic(char)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def is_cjk_punctuation(char: str | int | bytes) -> bool:
|
|
58
|
+
"""Check if a character is CJK symbol or punctuation.
|
|
59
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
60
|
+
"""
|
|
61
|
+
return is_in_blocks(char, PUNC_BLOCKS)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def is_cjk(char: str | int | bytes) -> bool:
|
|
65
|
+
"""Determine if a character is used in CJK (Chinese, Japanese, Korean) languages.
|
|
66
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
67
|
+
"""
|
|
68
|
+
return is_ideographic_zero(char) or is_in_blocks(char, CJK_BLOCKS)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
### Block checks ###
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def is_ideographic_block(block: UnicodeBlock) -> bool:
|
|
75
|
+
"""Check if a block is used in ideographic (hanzi/kanji/hanja/漢字/汉字)."""
|
|
76
|
+
return block in IDEO_BLOCKS
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def is_japanese_kana_block(block: UnicodeBlock) -> bool:
|
|
80
|
+
"""Check if a block is used in Japanese kana (ひらがな/カタカナ)."""
|
|
81
|
+
return block in JPAN_BLOCKS
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def is_japanese_block(block: UnicodeBlock) -> bool:
|
|
85
|
+
"""Check if a block is used in Japanese (kanji + kana)."""
|
|
86
|
+
return is_japanese_kana_block(block) or is_ideographic_block(block)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def is_korean_hangul_block(block: UnicodeBlock) -> bool:
|
|
90
|
+
"""Check if a block is used in Korean hangul (한글)."""
|
|
91
|
+
return block in KORE_BLOCKS
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def is_korean_block(block: UnicodeBlock) -> bool:
|
|
95
|
+
"""Check if a block is used in Korean (hanja + hangul)."""
|
|
96
|
+
return is_korean_hangul_block(block) or is_ideographic_block(block)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def is_cjk_punctuation_block(block: UnicodeBlock) -> bool:
|
|
100
|
+
"""Check if a block is used in CJK symbol or punctuation."""
|
|
101
|
+
return block in PUNC_BLOCKS
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def is_cjk_block(block: UnicodeBlock) -> bool:
|
|
105
|
+
"""Determine if a block is used in used in CJK (Chinese, Japanese, Korean) languages."""
|
|
106
|
+
return (
|
|
107
|
+
is_ideographic_block(block)
|
|
108
|
+
or is_japanese_block(block)
|
|
109
|
+
or is_korean_block(block)
|
|
110
|
+
or is_cjk_punctuation_block(block)
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
### Block accessors ###
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def get_ideographic_blocks() -> list[UnicodeBlock]:
|
|
118
|
+
"""Get all Unicode blocks used in ideographic (hanzi/kanji/hanja/漢字/汉字)."""
|
|
119
|
+
return IDEO_BLOCKS
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def get_japanese_kana_blocks() -> list[UnicodeBlock]:
|
|
123
|
+
"""Get all Unicode blocks used in Japanese kana (ひらがな/カタカナ)."""
|
|
124
|
+
return JPAN_BLOCKS
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def get_japanese_blocks() -> list[UnicodeBlock]:
|
|
128
|
+
"""Get all Unicode blocks used in Japanese (kanji + kana)."""
|
|
129
|
+
return JPAN_BLOCKS + IDEO_BLOCKS
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def get_korean_hangul_blocks() -> list[UnicodeBlock]:
|
|
133
|
+
"""Get all Unicode blocks used in Korean hangul (한글)."""
|
|
134
|
+
return KORE_BLOCKS
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def get_korean_blocks() -> list[UnicodeBlock]:
|
|
138
|
+
"""Get all Unicode blocks used in Korean (hanja + hangul)."""
|
|
139
|
+
return KORE_BLOCKS + IDEO_BLOCKS
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def get_cjk_punctuation_blocks() -> list[UnicodeBlock]:
|
|
143
|
+
"""Get all Unicode blocks used in CJK symbol or punctuation."""
|
|
144
|
+
return PUNC_BLOCKS
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def get_cjk_blocks() -> list[UnicodeBlock]:
|
|
148
|
+
"""Get all Unicode blocks used in used in CJK (Chinese, Japanese, Korean) languages."""
|
|
149
|
+
return CJK_BLOCKS
|
unicode_blocks/errors.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
from .unicodeBlock import UnicodeBlock
|
|
2
|
+
from .charNormaliser import CharNormaliser
|
|
3
|
+
from .blocks import ALL_BLOCKS, NO_BLOCK
|
|
4
|
+
from .errors import InvalidUnicodeBlockNameError
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def all() -> list[UnicodeBlock]:
|
|
8
|
+
"""
|
|
9
|
+
Get all Unicode blocks.
|
|
10
|
+
"""
|
|
11
|
+
return ALL_BLOCKS
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def for_name(name: str) -> UnicodeBlock:
|
|
15
|
+
"""
|
|
16
|
+
Get the Unicode block for a given name.
|
|
17
|
+
"""
|
|
18
|
+
blocks = all()
|
|
19
|
+
for block in blocks:
|
|
20
|
+
if block.normalised_name == UnicodeBlock.normalise_name(name):
|
|
21
|
+
return block
|
|
22
|
+
if UnicodeBlock.normalise_name(name) in block.aliases:
|
|
23
|
+
return block
|
|
24
|
+
raise InvalidUnicodeBlockNameError(name)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def of(char: str | int | bytes) -> UnicodeBlock:
|
|
28
|
+
"""
|
|
29
|
+
Get the name of the Unicode block for a given character.
|
|
30
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8.
|
|
31
|
+
"""
|
|
32
|
+
unidec = CharNormaliser.to_codepoint(char)
|
|
33
|
+
blocks = all()
|
|
34
|
+
low, high = 0, len(blocks) - 1
|
|
35
|
+
while low <= high:
|
|
36
|
+
mid = (low + high) // 2
|
|
37
|
+
block = blocks[mid]
|
|
38
|
+
if unidec in block:
|
|
39
|
+
return block
|
|
40
|
+
elif unidec < block.start:
|
|
41
|
+
high = mid - 1
|
|
42
|
+
else:
|
|
43
|
+
low = mid + 1
|
|
44
|
+
return NO_BLOCK
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from functools import total_ordering
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
from .charNormaliser import CharNormaliser
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@total_ordering
|
|
10
|
+
class UnicodeBlock:
|
|
11
|
+
def __init__(
|
|
12
|
+
self,
|
|
13
|
+
name: str,
|
|
14
|
+
start: int,
|
|
15
|
+
end: int,
|
|
16
|
+
assigned_ranges: Optional[list[tuple[int, int]]] = None,
|
|
17
|
+
aliases: Optional[list[str]] = None,
|
|
18
|
+
):
|
|
19
|
+
self.name = name
|
|
20
|
+
self.start = start
|
|
21
|
+
self.end = end
|
|
22
|
+
self.assigned_ranges = AssignedRanges(assigned_ranges) if assigned_ranges else []
|
|
23
|
+
self.aliases = [self.normalise_name(a) for a in aliases] if aliases else []
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def normalised_name(self) -> str:
|
|
27
|
+
"""Return the normalised name of this Unicode block."""
|
|
28
|
+
return self.normalise_name(self.name)
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def variable_name(self) -> str:
|
|
32
|
+
"""Return the variable name of this Unicode block."""
|
|
33
|
+
return self.to_variable_name(self.name)
|
|
34
|
+
|
|
35
|
+
def __contains__(self, char: str | int | bytes) -> bool:
|
|
36
|
+
"""Check if a character is in this Unicode block.
|
|
37
|
+
May be a str, int, or bytes. Bytes are decoded to str using utf-8."""
|
|
38
|
+
unidec = CharNormaliser.to_codepoint(char)
|
|
39
|
+
return self.start <= unidec <= self.end
|
|
40
|
+
|
|
41
|
+
def __eq__(self, other: object) -> bool:
|
|
42
|
+
if not isinstance(other, UnicodeBlock):
|
|
43
|
+
return NotImplemented
|
|
44
|
+
return self.start == other.start
|
|
45
|
+
|
|
46
|
+
def __lt__(self, other: object) -> bool:
|
|
47
|
+
"""Compare two UnicodeBlock objects based on their start values."""
|
|
48
|
+
if not isinstance(other, UnicodeBlock):
|
|
49
|
+
return NotImplemented
|
|
50
|
+
return self.start < other.start
|
|
51
|
+
|
|
52
|
+
def __hash__(self) -> int:
|
|
53
|
+
return hash((self.start, self.end))
|
|
54
|
+
|
|
55
|
+
def __repr__(self) -> str:
|
|
56
|
+
parts = [
|
|
57
|
+
f"name={self.name!r}",
|
|
58
|
+
f"start={self.start:#06x}",
|
|
59
|
+
f"end={self.end:#06x}",
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
if self.assigned_ranges:
|
|
63
|
+
ranges_str = [f"({start:#06x}, {end:#06x})" for start, end in self.assigned_ranges]
|
|
64
|
+
parts.append(f"assigned_ranges=[{', '.join(ranges_str)}]")
|
|
65
|
+
|
|
66
|
+
if self.aliases:
|
|
67
|
+
parts.append(f"aliases={self.aliases!r}")
|
|
68
|
+
|
|
69
|
+
return f"{self.__class__.__name__}({', '.join(parts)})"
|
|
70
|
+
|
|
71
|
+
def __len__(self) -> int:
|
|
72
|
+
"""Return the number of available code points in this block."""
|
|
73
|
+
return self.end - self.start + 1
|
|
74
|
+
|
|
75
|
+
@staticmethod
|
|
76
|
+
def normalise_name(name: str) -> str:
|
|
77
|
+
"""Normalise the name of a Unicode block."""
|
|
78
|
+
name = name.upper().replace(" ", "").replace("-", "").replace("_", "")
|
|
79
|
+
if name.startswith("IS"):
|
|
80
|
+
name = name[2:]
|
|
81
|
+
return name
|
|
82
|
+
|
|
83
|
+
@staticmethod
|
|
84
|
+
def to_variable_name(name: str) -> str:
|
|
85
|
+
"""Normalise the variable name of a Unicode block."""
|
|
86
|
+
return name.upper().replace(" ", "_").replace("-", "_")
|
|
87
|
+
|
|
88
|
+
class AssignedRanges:
|
|
89
|
+
"""A class to represent assigned ranges within a Unicode block."""
|
|
90
|
+
|
|
91
|
+
def __init__(self, ranges: list[tuple[int, int]]):
|
|
92
|
+
self.ranges = ranges
|
|
93
|
+
|
|
94
|
+
def __iter__(self):
|
|
95
|
+
return iter(self.ranges)
|
|
96
|
+
|
|
97
|
+
def __repr__(self) -> str:
|
|
98
|
+
return f"AssignedRanges({self.ranges})"
|
|
99
|
+
|
|
100
|
+
def __len__(self) -> int:
|
|
101
|
+
"""Return the number of assigned ranges."""
|
|
102
|
+
if self.ranges is None:
|
|
103
|
+
return 0
|
|
104
|
+
return sum(end - start + 1 for start, end in self.ranges)
|
|
105
|
+
|
|
106
|
+
def __contains__(self, char: str | int | bytes) -> bool:
|
|
107
|
+
"""Check if a character is in any of the assigned ranges."""
|
|
108
|
+
unidec = CharNormaliser.to_codepoint(char)
|
|
109
|
+
for start, end in self.ranges:
|
|
110
|
+
if start <= unidec <= end:
|
|
111
|
+
return True
|
|
112
|
+
return False
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: unicode-blocks-py
|
|
3
|
+
Version: 6.1.0
|
|
4
|
+
Summary: Unicode blocks data utility module
|
|
5
|
+
Author-email: NightFurySL2001 <nfsl-fonts@outlook.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/NightFurySL2001/unicode-blocks-py
|
|
8
|
+
Project-URL: Issues, https://github.com/NightFurySL2001/unicode-blocks-py/issues
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Natural Language :: English
|
|
14
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
15
|
+
Requires-Python: >=3.11
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# 🧱 Unicode_Blocks 🧱
|
|
21
|
+
|
|
22
|
+
`unicode_blocks` is a simple utility module for working with Unicode blocks data. [Unicode blocks](https://www.unicode.org/versions/latest/core-spec/chapter-3/#G64189) are continuous ranges of code points defined by the Unicode standard, used to group characters with generally similar purposes or origins.
|
|
23
|
+
|
|
24
|
+
## Usage
|
|
25
|
+
|
|
26
|
+
Install this package from PyPI:
|
|
27
|
+
|
|
28
|
+
```sh
|
|
29
|
+
pip install unicode-blocks-py
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
The module interface is heavily inspired by Java [`Character.UnicodeBlock`](https://docs.oracle.com/en/java/javase/21/docs/api/java.base/java/lang/Character.UnicodeBlock.html) class and Rust [`unicode_blocks`](https://docs.rs/unicode-blocks/latest/unicode_blocks/) module.
|
|
33
|
+
|
|
34
|
+
```py
|
|
35
|
+
>>> import unicode_blocks
|
|
36
|
+
>>> unicode_major_version = int(unicode_blocks.__version__.split(".")[0])
|
|
37
|
+
|
|
38
|
+
# To get Unicode block of a character, input a character string of length 1,
|
|
39
|
+
# UTF-8 encoded bytes, or a positive integer representing a Unicode code point.
|
|
40
|
+
# The following are the same: they decode the character 'a'.
|
|
41
|
+
>>> block = unicode_blocks.of('a')
|
|
42
|
+
>>> block2 = unicode_blocks.of(b'\x61')
|
|
43
|
+
>>> block3 = unicode_blocks.of(97)
|
|
44
|
+
>>> assert block == block2 == block3
|
|
45
|
+
|
|
46
|
+
# To get Unicode block using name, input the block name.
|
|
47
|
+
# Cases, whitespace, dashes, underscrolls and prefix "is" will be ignored for comparison. See UAX44-LM3.
|
|
48
|
+
# Block name aliases from PropertyValueAliases are also usable here
|
|
49
|
+
>>> ascii_block = unicode_blocks.for_name("BASIC_LATIN")
|
|
50
|
+
>>> ascii_block2 = unicode_blocks.for_name("basiclatin")
|
|
51
|
+
>>> ascii_block3 = unicode_blocks.for_name("isBasicLatin")
|
|
52
|
+
>>> from unicode_blocks import BASIC_LATIN
|
|
53
|
+
>>> assert ascii_block == ascii_block2 == ascii_block3 == BASIC_LATIN
|
|
54
|
+
>>> if unicode_major_version >= 6:
|
|
55
|
+
... ascii_block4 = unicode_blocks.for_name("ASCII")
|
|
56
|
+
... assert ascii_block4 == BASIC_LATIN
|
|
57
|
+
|
|
58
|
+
# Unicode characters currently not assigned will receive No_Block object as per
|
|
59
|
+
# rule D10b in Section 3.4, *Characters and Encoding*, of Unicode
|
|
60
|
+
>>> assert unicode_blocks.of(0xEDCBA) == unicode_blocks.NO_BLOCK
|
|
61
|
+
|
|
62
|
+
# List through all the defined Unicode blocks at the version
|
|
63
|
+
# NO_BLOCK is not in the list of all blocks
|
|
64
|
+
>>> for block in unicode_blocks.all():
|
|
65
|
+
... print(block) # doctest: +ELLIPSIS
|
|
66
|
+
UnicodeBlock(...)
|
|
67
|
+
|
|
68
|
+
# Pythonic helpers: comparisons between blocks, where earlier blocks is smaller than later blocks
|
|
69
|
+
# useful for sorting a list of UnicodeBlocks
|
|
70
|
+
>>> latin1_block = unicode_blocks.for_name("Latin-1 Supplement")
|
|
71
|
+
>>> assert ascii_block < latin1_block
|
|
72
|
+
|
|
73
|
+
# Get the total defined code points in a block. Does not represent if the block is filled in or not.
|
|
74
|
+
>>> assert len(ascii_block) == 128
|
|
75
|
+
|
|
76
|
+
# Additional helpers: check for assigned characters in the block
|
|
77
|
+
# Data is loaded from UCD and may change between Unicode versions
|
|
78
|
+
>>> assert len(ascii_block.assigned_ranges) == 128
|
|
79
|
+
>>> assert 'B' in ascii_block.assigned_ranges
|
|
80
|
+
|
|
81
|
+
# Example where defined Unicode block range is not fully utilised
|
|
82
|
+
>>> bopo_block = unicode_blocks.of('ㄅ')
|
|
83
|
+
>>> assert len(bopo_block) == 48
|
|
84
|
+
>>> bopo_assigned_count = 41 if unicode_major_version < 10 else 42 if unicode_major_version == 10 else 43
|
|
85
|
+
>>> assert len(bopo_block.assigned_ranges) == bopo_assigned_count # first 5 code points should be unassigned, at least in <=17.0
|
|
86
|
+
>>> assert len(bopo_block) != len(bopo_block.assigned_ranges)
|
|
87
|
+
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The lists of Unicode block objects are available directly in the namespace, or under the `blocks` module.
|
|
91
|
+
|
|
92
|
+
```py
|
|
93
|
+
# both are equivalent
|
|
94
|
+
>>> from unicode_blocks import BASIC_LATIN
|
|
95
|
+
>>> from unicode_blocks.blocks import BASIC_LATIN
|
|
96
|
+
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Various names are also available in the block:
|
|
100
|
+
|
|
101
|
+
```py
|
|
102
|
+
>>> from unicode_blocks import BASIC_LATIN
|
|
103
|
+
>>> assert BASIC_LATIN.name == "Basic Latin" # Official Unicode name as in Blocks.txt
|
|
104
|
+
>>> assert BASIC_LATIN.normalised_name == "BASICLATIN" # Normalised name under UAX44-LM3
|
|
105
|
+
>>> assert BASIC_LATIN.variable_name == "BASIC_LATIN" # Variable name in `unicode_blocks.blocks`
|
|
106
|
+
>>> if unicode_major_version >= 6:
|
|
107
|
+
... assert BASIC_LATIN.aliases == ["ASCII"] # Official block aliases as in PropertyValueAliases.txt
|
|
108
|
+
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Additional utilities for CJK are specially provided referencing the oxidised version of the module. Selected samples are shown below.
|
|
112
|
+
|
|
113
|
+
```py
|
|
114
|
+
>>> from unicode_blocks import cjk
|
|
115
|
+
>>> assert cjk.is_cjk('中')
|
|
116
|
+
>>> assert cjk.is_japanese_kana('あ')
|
|
117
|
+
>>> assert cjk.is_korean_hangul('글')
|
|
118
|
+
>>> assert cjk.is_cjk_punctuation('。')
|
|
119
|
+
|
|
120
|
+
>>> from unicode_blocks import blocks
|
|
121
|
+
>>> assert cjk.is_ideographic_block(blocks.CJK_UNIFIED_IDEOGRAPHS)
|
|
122
|
+
>>> assert cjk.is_cjk_block(blocks.KANGXI_RADICALS)
|
|
123
|
+
>>> assert cjk.is_japanese_block(blocks.KATAKANA_PHONETIC_EXTENSIONS)
|
|
124
|
+
>>> assert cjk.is_korean_block(blocks.HANGUL_COMPATIBILITY_JAMO)
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
> [!WARNING]
|
|
129
|
+
> Checking `char in unicode_blocks.for_name("is_CJK")` is **NOT** the same as `cjk.is_cjk(char)`!
|
|
130
|
+
> `unicode_blocks.for_name("is_CJK")` refers to the "CJK" block alias for CJK Unified Ideographs block, while `cjk.is_cjk` checks through (roughly) all Unicode blocks related to CJK including kana, hangul and punctuations.
|
|
131
|
+
|
|
132
|
+
To check which Unicode version data is used, check against the `__version__` variable in the namespace. (Bug fix release will use `+1` notation)
|
|
133
|
+
|
|
134
|
+
```sh
|
|
135
|
+
$ python3
|
|
136
|
+
>>> import unicode_blocks
|
|
137
|
+
>>> unicode_blocks.__version__ # doctest: +SKIP
|
|
138
|
+
'17.0.0'
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
The version will follow the Unicode semver of the data files, optionally followed by additional numbering from this module for bug fixes after a plus sign, i.e. `<Unicode major.minor.patch>(+<additional numbering>)`.
|
|
142
|
+
|
|
143
|
+
## Update
|
|
144
|
+
|
|
145
|
+
To update the blocks data from Unicode Character Database, update the `project.version` key in `pyproject.toml` to the Unicode version number, and then run `python3 build_blocks.py`. This will update the `src/unicode_blocks/blocks.py` file, which is automatically generated from UCD data.
|
|
146
|
+
|
|
147
|
+
Most of these steps should be directly runnable through GitHub Actions.
|
|
148
|
+
|
|
149
|
+
## Contributing
|
|
150
|
+
|
|
151
|
+
Contributions are welcome! Please follow these steps:
|
|
152
|
+
|
|
153
|
+
1. Clone the repository and install as development mode:
|
|
154
|
+
```sh
|
|
155
|
+
git clone https://github.com/NightFurySL2001/unicode-blocks.git
|
|
156
|
+
cd unicode-blocks
|
|
157
|
+
pip install -e .
|
|
158
|
+
```
|
|
159
|
+
2. Create a new branch for your feature or bug fix.
|
|
160
|
+
3. Work on the feature and run or develop relevant test cases.
|
|
161
|
+
4. Test the changes by running `pytest`.
|
|
162
|
+
5. Ensure this README.md is updated with `python -m doctest README.md`.
|
|
163
|
+
6. Submit a pull request with a clear description of your changes.
|
|
164
|
+
|
|
165
|
+
## License
|
|
166
|
+
|
|
167
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
168
|
+
|
|
169
|
+
## Acknowledgments
|
|
170
|
+
|
|
171
|
+
- [Unicode Consortium](https://unicode.org) for maintaining the Unicode standard and providing the Unicode Character Database (UCD). Data modification are done under [Unicode License v3](https://www.unicode.org/license.txt).
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
unicode_blocks/__init__.py,sha256=iUzxCYc1FxVX8PowNApt-jSp5hpYtVBtvElZMadV1uY,115
|
|
2
|
+
unicode_blocks/blocks.py,sha256=5j2rqPuG-Ju3sKKP4uAfSqmh_A-tcv9FvLQTqtRMo-U,42566
|
|
3
|
+
unicode_blocks/charNormaliser.py,sha256=lxAhk9Cf5-qK-O1FC89uUw-yZf4jU9PDDWIwWcCv2pU,721
|
|
4
|
+
unicode_blocks/cjk.py,sha256=6Xxz-QXdOGNpz1qBs4zBD2oNK7qlXZHLPp6raXtwhdo,4950
|
|
5
|
+
unicode_blocks/errors.py,sha256=gKvNQMxlpGGOeWcJMwvMMikJPmb0fbqpkYx7pChGpvs,225
|
|
6
|
+
unicode_blocks/globals.py,sha256=pud_WrINy5gXaVcAzTMot14_gaOc6t60SOtu_qo1NrY,1207
|
|
7
|
+
unicode_blocks/unicodeBlock.py,sha256=cwcU4HAJc_Fv9d9KlOJTReyyUR5uhZaR05ywXJvdLw8,3712
|
|
8
|
+
unicode_blocks_py-6.1.0.dist-info/licenses/LICENSE,sha256=hceFGJLSH-s78KSb0ZEsMHuXjDG6IS8FlDxpgOuf06c,1070
|
|
9
|
+
unicode_blocks_py-6.1.0.dist-info/METADATA,sha256=ufucodoQpcIVgfG3Bw2JleLprN8PuxiaUH4xJFTh5w4,7474
|
|
10
|
+
unicode_blocks_py-6.1.0.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
11
|
+
unicode_blocks_py-6.1.0.dist-info/top_level.txt,sha256=u-QN9HeyE8pcxDzlh3uuDP4Hb3W4uXMyQHnhDI5fAbw,15
|
|
12
|
+
unicode_blocks_py-6.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright © 2025 NightFurySL2001
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
unicode_blocks
|