unicode-blocks-py 6.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright © 2025 NightFurySL2001
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,171 @@
1
+ Metadata-Version: 2.4
2
+ Name: unicode-blocks-py
3
+ Version: 6.1.0
4
+ Summary: Unicode blocks data utility module
5
+ Author-email: NightFurySL2001 <nfsl-fonts@outlook.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/NightFurySL2001/unicode-blocks-py
8
+ Project-URL: Issues, https://github.com/NightFurySL2001/unicode-blocks-py/issues
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Natural Language :: English
14
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
15
+ Requires-Python: >=3.11
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Dynamic: license-file
19
+
20
+ # 🧱 Unicode_Blocks 🧱
21
+
22
+ `unicode_blocks` is a simple utility module for working with Unicode blocks data. [Unicode blocks](https://www.unicode.org/versions/latest/core-spec/chapter-3/#G64189) are continuous ranges of code points defined by the Unicode standard, used to group characters with generally similar purposes or origins.
23
+
24
+ ## Usage
25
+
26
+ Install this package from PyPI:
27
+
28
+ ```sh
29
+ pip install unicode-blocks-py
30
+ ```
31
+
32
+ The module interface is heavily inspired by Java [`Character.UnicodeBlock`](https://docs.oracle.com/en/java/javase/21/docs/api/java.base/java/lang/Character.UnicodeBlock.html) class and Rust [`unicode_blocks`](https://docs.rs/unicode-blocks/latest/unicode_blocks/) module.
33
+
34
+ ```py
35
+ >>> import unicode_blocks
36
+ >>> unicode_major_version = int(unicode_blocks.__version__.split(".")[0])
37
+
38
+ # To get Unicode block of a character, input a character string of length 1,
39
+ # UTF-8 encoded bytes, or a positive integer representing a Unicode code point.
40
+ # The following are the same: they decode the character 'a'.
41
+ >>> block = unicode_blocks.of('a')
42
+ >>> block2 = unicode_blocks.of(b'\x61')
43
+ >>> block3 = unicode_blocks.of(97)
44
+ >>> assert block == block2 == block3
45
+
46
+ # To get Unicode block using name, input the block name.
47
+ # Cases, whitespace, dashes, underscrolls and prefix "is" will be ignored for comparison. See UAX44-LM3.
48
+ # Block name aliases from PropertyValueAliases are also usable here
49
+ >>> ascii_block = unicode_blocks.for_name("BASIC_LATIN")
50
+ >>> ascii_block2 = unicode_blocks.for_name("basiclatin")
51
+ >>> ascii_block3 = unicode_blocks.for_name("isBasicLatin")
52
+ >>> from unicode_blocks import BASIC_LATIN
53
+ >>> assert ascii_block == ascii_block2 == ascii_block3 == BASIC_LATIN
54
+ >>> if unicode_major_version >= 6:
55
+ ... ascii_block4 = unicode_blocks.for_name("ASCII")
56
+ ... assert ascii_block4 == BASIC_LATIN
57
+
58
+ # Unicode characters currently not assigned will receive No_Block object as per
59
+ # rule D10b in Section 3.4, *Characters and Encoding*, of Unicode
60
+ >>> assert unicode_blocks.of(0xEDCBA) == unicode_blocks.NO_BLOCK
61
+
62
+ # List through all the defined Unicode blocks at the version
63
+ # NO_BLOCK is not in the list of all blocks
64
+ >>> for block in unicode_blocks.all():
65
+ ... print(block) # doctest: +ELLIPSIS
66
+ UnicodeBlock(...)
67
+
68
+ # Pythonic helpers: comparisons between blocks, where earlier blocks is smaller than later blocks
69
+ # useful for sorting a list of UnicodeBlocks
70
+ >>> latin1_block = unicode_blocks.for_name("Latin-1 Supplement")
71
+ >>> assert ascii_block < latin1_block
72
+
73
+ # Get the total defined code points in a block. Does not represent if the block is filled in or not.
74
+ >>> assert len(ascii_block) == 128
75
+
76
+ # Additional helpers: check for assigned characters in the block
77
+ # Data is loaded from UCD and may change between Unicode versions
78
+ >>> assert len(ascii_block.assigned_ranges) == 128
79
+ >>> assert 'B' in ascii_block.assigned_ranges
80
+
81
+ # Example where defined Unicode block range is not fully utilised
82
+ >>> bopo_block = unicode_blocks.of('ㄅ')
83
+ >>> assert len(bopo_block) == 48
84
+ >>> bopo_assigned_count = 41 if unicode_major_version < 10 else 42 if unicode_major_version == 10 else 43
85
+ >>> assert len(bopo_block.assigned_ranges) == bopo_assigned_count # first 5 code points should be unassigned, at least in <=17.0
86
+ >>> assert len(bopo_block) != len(bopo_block.assigned_ranges)
87
+
88
+ ```
89
+
90
+ The lists of Unicode block objects are available directly in the namespace, or under the `blocks` module.
91
+
92
+ ```py
93
+ # both are equivalent
94
+ >>> from unicode_blocks import BASIC_LATIN
95
+ >>> from unicode_blocks.blocks import BASIC_LATIN
96
+
97
+ ```
98
+
99
+ Various names are also available in the block:
100
+
101
+ ```py
102
+ >>> from unicode_blocks import BASIC_LATIN
103
+ >>> assert BASIC_LATIN.name == "Basic Latin" # Official Unicode name as in Blocks.txt
104
+ >>> assert BASIC_LATIN.normalised_name == "BASICLATIN" # Normalised name under UAX44-LM3
105
+ >>> assert BASIC_LATIN.variable_name == "BASIC_LATIN" # Variable name in `unicode_blocks.blocks`
106
+ >>> if unicode_major_version >= 6:
107
+ ... assert BASIC_LATIN.aliases == ["ASCII"] # Official block aliases as in PropertyValueAliases.txt
108
+
109
+ ```
110
+
111
+ Additional utilities for CJK are specially provided referencing the oxidised version of the module. Selected samples are shown below.
112
+
113
+ ```py
114
+ >>> from unicode_blocks import cjk
115
+ >>> assert cjk.is_cjk('中')
116
+ >>> assert cjk.is_japanese_kana('あ')
117
+ >>> assert cjk.is_korean_hangul('글')
118
+ >>> assert cjk.is_cjk_punctuation('。')
119
+
120
+ >>> from unicode_blocks import blocks
121
+ >>> assert cjk.is_ideographic_block(blocks.CJK_UNIFIED_IDEOGRAPHS)
122
+ >>> assert cjk.is_cjk_block(blocks.KANGXI_RADICALS)
123
+ >>> assert cjk.is_japanese_block(blocks.KATAKANA_PHONETIC_EXTENSIONS)
124
+ >>> assert cjk.is_korean_block(blocks.HANGUL_COMPATIBILITY_JAMO)
125
+
126
+ ```
127
+
128
+ > [!WARNING]
129
+ > Checking `char in unicode_blocks.for_name("is_CJK")` is **NOT** the same as `cjk.is_cjk(char)`!
130
+ > `unicode_blocks.for_name("is_CJK")` refers to the "CJK" block alias for CJK Unified Ideographs block, while `cjk.is_cjk` checks through (roughly) all Unicode blocks related to CJK including kana, hangul and punctuations.
131
+
132
+ To check which Unicode version data is used, check against the `__version__` variable in the namespace. (Bug fix release will use `+1` notation)
133
+
134
+ ```sh
135
+ $ python3
136
+ >>> import unicode_blocks
137
+ >>> unicode_blocks.__version__ # doctest: +SKIP
138
+ '17.0.0'
139
+ ```
140
+
141
+ The version will follow the Unicode semver of the data files, optionally followed by additional numbering from this module for bug fixes after a plus sign, i.e. `<Unicode major.minor.patch>(+<additional numbering>)`.
142
+
143
+ ## Update
144
+
145
+ To update the blocks data from Unicode Character Database, update the `project.version` key in `pyproject.toml` to the Unicode version number, and then run `python3 build_blocks.py`. This will update the `src/unicode_blocks/blocks.py` file, which is automatically generated from UCD data.
146
+
147
+ Most of these steps should be directly runnable through GitHub Actions.
148
+
149
+ ## Contributing
150
+
151
+ Contributions are welcome! Please follow these steps:
152
+
153
+ 1. Clone the repository and install as development mode:
154
+ ```sh
155
+ git clone https://github.com/NightFurySL2001/unicode-blocks.git
156
+ cd unicode-blocks
157
+ pip install -e .
158
+ ```
159
+ 2. Create a new branch for your feature or bug fix.
160
+ 3. Work on the feature and run or develop relevant test cases.
161
+ 4. Test the changes by running `pytest`.
162
+ 5. Ensure this README.md is updated with `python -m doctest README.md`.
163
+ 6. Submit a pull request with a clear description of your changes.
164
+
165
+ ## License
166
+
167
+ This project is licensed under the [MIT License](LICENSE).
168
+
169
+ ## Acknowledgments
170
+
171
+ - [Unicode Consortium](https://unicode.org) for maintaining the Unicode standard and providing the Unicode Character Database (UCD). Data modification are done under [Unicode License v3](https://www.unicode.org/license.txt).
@@ -0,0 +1,152 @@
1
+ # 🧱 Unicode_Blocks 🧱
2
+
3
+ `unicode_blocks` is a simple utility module for working with Unicode blocks data. [Unicode blocks](https://www.unicode.org/versions/latest/core-spec/chapter-3/#G64189) are continuous ranges of code points defined by the Unicode standard, used to group characters with generally similar purposes or origins.
4
+
5
+ ## Usage
6
+
7
+ Install this package from PyPI:
8
+
9
+ ```sh
10
+ pip install unicode-blocks-py
11
+ ```
12
+
13
+ The module interface is heavily inspired by Java [`Character.UnicodeBlock`](https://docs.oracle.com/en/java/javase/21/docs/api/java.base/java/lang/Character.UnicodeBlock.html) class and Rust [`unicode_blocks`](https://docs.rs/unicode-blocks/latest/unicode_blocks/) module.
14
+
15
+ ```py
16
+ >>> import unicode_blocks
17
+ >>> unicode_major_version = int(unicode_blocks.__version__.split(".")[0])
18
+
19
+ # To get Unicode block of a character, input a character string of length 1,
20
+ # UTF-8 encoded bytes, or a positive integer representing a Unicode code point.
21
+ # The following are the same: they decode the character 'a'.
22
+ >>> block = unicode_blocks.of('a')
23
+ >>> block2 = unicode_blocks.of(b'\x61')
24
+ >>> block3 = unicode_blocks.of(97)
25
+ >>> assert block == block2 == block3
26
+
27
+ # To get Unicode block using name, input the block name.
28
+ # Cases, whitespace, dashes, underscrolls and prefix "is" will be ignored for comparison. See UAX44-LM3.
29
+ # Block name aliases from PropertyValueAliases are also usable here
30
+ >>> ascii_block = unicode_blocks.for_name("BASIC_LATIN")
31
+ >>> ascii_block2 = unicode_blocks.for_name("basiclatin")
32
+ >>> ascii_block3 = unicode_blocks.for_name("isBasicLatin")
33
+ >>> from unicode_blocks import BASIC_LATIN
34
+ >>> assert ascii_block == ascii_block2 == ascii_block3 == BASIC_LATIN
35
+ >>> if unicode_major_version >= 6:
36
+ ... ascii_block4 = unicode_blocks.for_name("ASCII")
37
+ ... assert ascii_block4 == BASIC_LATIN
38
+
39
+ # Unicode characters currently not assigned will receive No_Block object as per
40
+ # rule D10b in Section 3.4, *Characters and Encoding*, of Unicode
41
+ >>> assert unicode_blocks.of(0xEDCBA) == unicode_blocks.NO_BLOCK
42
+
43
+ # List through all the defined Unicode blocks at the version
44
+ # NO_BLOCK is not in the list of all blocks
45
+ >>> for block in unicode_blocks.all():
46
+ ... print(block) # doctest: +ELLIPSIS
47
+ UnicodeBlock(...)
48
+
49
+ # Pythonic helpers: comparisons between blocks, where earlier blocks is smaller than later blocks
50
+ # useful for sorting a list of UnicodeBlocks
51
+ >>> latin1_block = unicode_blocks.for_name("Latin-1 Supplement")
52
+ >>> assert ascii_block < latin1_block
53
+
54
+ # Get the total defined code points in a block. Does not represent if the block is filled in or not.
55
+ >>> assert len(ascii_block) == 128
56
+
57
+ # Additional helpers: check for assigned characters in the block
58
+ # Data is loaded from UCD and may change between Unicode versions
59
+ >>> assert len(ascii_block.assigned_ranges) == 128
60
+ >>> assert 'B' in ascii_block.assigned_ranges
61
+
62
+ # Example where defined Unicode block range is not fully utilised
63
+ >>> bopo_block = unicode_blocks.of('ㄅ')
64
+ >>> assert len(bopo_block) == 48
65
+ >>> bopo_assigned_count = 41 if unicode_major_version < 10 else 42 if unicode_major_version == 10 else 43
66
+ >>> assert len(bopo_block.assigned_ranges) == bopo_assigned_count # first 5 code points should be unassigned, at least in <=17.0
67
+ >>> assert len(bopo_block) != len(bopo_block.assigned_ranges)
68
+
69
+ ```
70
+
71
+ The lists of Unicode block objects are available directly in the namespace, or under the `blocks` module.
72
+
73
+ ```py
74
+ # both are equivalent
75
+ >>> from unicode_blocks import BASIC_LATIN
76
+ >>> from unicode_blocks.blocks import BASIC_LATIN
77
+
78
+ ```
79
+
80
+ Various names are also available in the block:
81
+
82
+ ```py
83
+ >>> from unicode_blocks import BASIC_LATIN
84
+ >>> assert BASIC_LATIN.name == "Basic Latin" # Official Unicode name as in Blocks.txt
85
+ >>> assert BASIC_LATIN.normalised_name == "BASICLATIN" # Normalised name under UAX44-LM3
86
+ >>> assert BASIC_LATIN.variable_name == "BASIC_LATIN" # Variable name in `unicode_blocks.blocks`
87
+ >>> if unicode_major_version >= 6:
88
+ ... assert BASIC_LATIN.aliases == ["ASCII"] # Official block aliases as in PropertyValueAliases.txt
89
+
90
+ ```
91
+
92
+ Additional utilities for CJK are specially provided referencing the oxidised version of the module. Selected samples are shown below.
93
+
94
+ ```py
95
+ >>> from unicode_blocks import cjk
96
+ >>> assert cjk.is_cjk('中')
97
+ >>> assert cjk.is_japanese_kana('あ')
98
+ >>> assert cjk.is_korean_hangul('글')
99
+ >>> assert cjk.is_cjk_punctuation('。')
100
+
101
+ >>> from unicode_blocks import blocks
102
+ >>> assert cjk.is_ideographic_block(blocks.CJK_UNIFIED_IDEOGRAPHS)
103
+ >>> assert cjk.is_cjk_block(blocks.KANGXI_RADICALS)
104
+ >>> assert cjk.is_japanese_block(blocks.KATAKANA_PHONETIC_EXTENSIONS)
105
+ >>> assert cjk.is_korean_block(blocks.HANGUL_COMPATIBILITY_JAMO)
106
+
107
+ ```
108
+
109
+ > [!WARNING]
110
+ > Checking `char in unicode_blocks.for_name("is_CJK")` is **NOT** the same as `cjk.is_cjk(char)`!
111
+ > `unicode_blocks.for_name("is_CJK")` refers to the "CJK" block alias for CJK Unified Ideographs block, while `cjk.is_cjk` checks through (roughly) all Unicode blocks related to CJK including kana, hangul and punctuations.
112
+
113
+ To check which Unicode version data is used, check against the `__version__` variable in the namespace. (Bug fix release will use `+1` notation)
114
+
115
+ ```sh
116
+ $ python3
117
+ >>> import unicode_blocks
118
+ >>> unicode_blocks.__version__ # doctest: +SKIP
119
+ '17.0.0'
120
+ ```
121
+
122
+ The version will follow the Unicode semver of the data files, optionally followed by additional numbering from this module for bug fixes after a plus sign, i.e. `<Unicode major.minor.patch>(+<additional numbering>)`.
123
+
124
+ ## Update
125
+
126
+ To update the blocks data from Unicode Character Database, update the `project.version` key in `pyproject.toml` to the Unicode version number, and then run `python3 build_blocks.py`. This will update the `src/unicode_blocks/blocks.py` file, which is automatically generated from UCD data.
127
+
128
+ Most of these steps should be directly runnable through GitHub Actions.
129
+
130
+ ## Contributing
131
+
132
+ Contributions are welcome! Please follow these steps:
133
+
134
+ 1. Clone the repository and install as development mode:
135
+ ```sh
136
+ git clone https://github.com/NightFurySL2001/unicode-blocks.git
137
+ cd unicode-blocks
138
+ pip install -e .
139
+ ```
140
+ 2. Create a new branch for your feature or bug fix.
141
+ 3. Work on the feature and run or develop relevant test cases.
142
+ 4. Test the changes by running `pytest`.
143
+ 5. Ensure this README.md is updated with `python -m doctest README.md`.
144
+ 6. Submit a pull request with a clear description of your changes.
145
+
146
+ ## License
147
+
148
+ This project is licensed under the [MIT License](LICENSE).
149
+
150
+ ## Acknowledgments
151
+
152
+ - [Unicode Consortium](https://unicode.org) for maintaining the Unicode standard and providing the Unicode Character Database (UCD). Data modification are done under [Unicode License v3](https://www.unicode.org/license.txt).
@@ -0,0 +1,33 @@
1
+ [project]
2
+ name = "unicode-blocks-py"
3
+ version = "6.1.0"
4
+ authors = [
5
+ { name="NightFurySL2001", email="nfsl-fonts@outlook.com" },
6
+ ]
7
+ description = "Unicode blocks data utility module"
8
+ readme = "README.md"
9
+ requires-python = ">=3.11"
10
+ classifiers = [
11
+ "Programming Language :: Python :: 3",
12
+ "Operating System :: OS Independent",
13
+ "Development Status :: 4 - Beta",
14
+ "Intended Audience :: Developers",
15
+ "Natural Language :: English",
16
+ "Topic :: Software Development :: Libraries :: Python Modules",
17
+ ]
18
+ license = "MIT"
19
+ license-files = ["LICEN[CS]E*"]
20
+
21
+ [project.urls]
22
+ Homepage = "https://github.com/NightFurySL2001/unicode-blocks-py"
23
+ Issues = "https://github.com/NightFurySL2001/unicode-blocks-py/issues"
24
+
25
+ [build-system]
26
+ requires = ["setuptools >= 77.0.3"]
27
+ build-backend = "setuptools.build_meta"
28
+
29
+ [tool.pytest.ini_options]
30
+ minversion = "6.0"
31
+ testpaths = [
32
+ "tests",
33
+ ]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,4 @@
1
+ from .blocks import __version__
2
+ from .blocks import *
3
+ from .cjk import is_cjk, is_cjk_block
4
+ from .globals import *