uchr 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- uchr-0.3.0.dist-info/LICENSE +21 -0
- uchr-0.3.0.dist-info/METADATA +314 -0
- uchr-0.3.0.dist-info/RECORD +11 -0
- uchr-0.3.0.dist-info/WHEEL +4 -0
- uchr-0.3.0.dist-info/entry_points.txt +3 -0
- unicode_tools/__init__.py +0 -0
- unicode_tools/database.py +331 -0
- unicode_tools/db.py +141 -0
- unicode_tools/normalize.py +112 -0
- unicode_tools/search.py +83 -0
- unicode_tools/uchr.py +242 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2023 Miki Yutani
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: uchr
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Unicode character tools
|
|
5
|
+
License: MIT
|
|
6
|
+
Keywords: unicode,character,tools,cli
|
|
7
|
+
Author: Miki Yutani
|
|
8
|
+
Author-email: mkyutani@gmail.com
|
|
9
|
+
Requires-Python: >=3.8,<4.0
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
23
|
+
Classifier: Topic :: Text Processing :: General
|
|
24
|
+
Classifier: Topic :: Utilities
|
|
25
|
+
Requires-Dist: requests (>=2.25.0,<3.0.0)
|
|
26
|
+
Project-URL: Homepage, https://github.com/mkyutani/unicode-tools
|
|
27
|
+
Project-URL: Repository, https://github.com/mkyutani/unicode-tools
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# Unicode Tools
|
|
31
|
+
|
|
32
|
+
[](https://opensource.org/licenses/MIT)
|
|
33
|
+
[](https://www.python.org/downloads/)
|
|
34
|
+
[](https://unicode.org/)
|
|
35
|
+
|
|
36
|
+
A powerful command-line tool for searching and exploring Unicode characters, emoji sequences, and character properties.
|
|
37
|
+
|
|
38
|
+
## 🚀 Features
|
|
39
|
+
|
|
40
|
+
- **Search by name**: Find characters by their Unicode name
|
|
41
|
+
- **Search by code**: Look up characters by code point or range
|
|
42
|
+
- **Search by character**: Reverse lookup from character to details
|
|
43
|
+
- **Search by block**: Explore characters within Unicode blocks
|
|
44
|
+
- **Emoji support**: Full support for emoji sequences and ZWJ sequences
|
|
45
|
+
- **CJK details**: Enhanced descriptions for CJK characters using kDefinition
|
|
46
|
+
- **Flexible output**: Multiple output formats for different use cases
|
|
47
|
+
|
|
48
|
+
## 📖 Table of Contents
|
|
49
|
+
|
|
50
|
+
- [Installation](#installation)
|
|
51
|
+
- [Quick Start](#quick-start)
|
|
52
|
+
- [Usage Examples](#usage-examples)
|
|
53
|
+
- [Command Reference](#command-reference)
|
|
54
|
+
- [Database Management](#database-management)
|
|
55
|
+
- [Contributing](#contributing)
|
|
56
|
+
- [License](#license)
|
|
57
|
+
|
|
58
|
+
## 🛠 Installation
|
|
59
|
+
|
|
60
|
+
### Install from source
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
git clone https://github.com/mkyutani/unicode-tools.git
|
|
64
|
+
cd unicode-tools
|
|
65
|
+
pip install -e .
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
### Initialize database
|
|
69
|
+
|
|
70
|
+
Create the Unicode database (required for first use):
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
uchr db create
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
This downloads Unicode 15.0 data and creates a local SQLite database (~13MB) at:
|
|
77
|
+
- Linux/macOS: `~/.local/share/unicode-tools/unicode.db`
|
|
78
|
+
- Root users: Automatically chooses between system (`/var/lib/unicode-tools/`) or personal location
|
|
79
|
+
|
|
80
|
+
## ⚡ Quick Start
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
# Search for ghost-related characters
|
|
84
|
+
uchr search ghost
|
|
85
|
+
|
|
86
|
+
# Find characters in a code range
|
|
87
|
+
uchr search -c 1F47A-1F480
|
|
88
|
+
|
|
89
|
+
# Search by character
|
|
90
|
+
uchr search -x 👻
|
|
91
|
+
|
|
92
|
+
# Search within a Unicode block
|
|
93
|
+
uchr search -b "Emoticons"
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## 📋 Usage Examples
|
|
97
|
+
|
|
98
|
+
### Search by Name
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
uchr search goblin
|
|
102
|
+
```
|
|
103
|
+
```
|
|
104
|
+
👺 1F47A JAPANESE GOBLIN
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
### Search by Code Range
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
uchr search -c 1F479-1F47B
|
|
111
|
+
```
|
|
112
|
+
```
|
|
113
|
+
👹 1F479 JAPANESE OGRE
|
|
114
|
+
👺 1F47A JAPANESE GOBLIN
|
|
115
|
+
👻 1F47B GHOST
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### Search by Character
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
uchr search -x 👻
|
|
122
|
+
```
|
|
123
|
+
```
|
|
124
|
+
👻 1F47B GHOST
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
### Search by Unicode Block
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
uchr search -b "Misc_Pictographs"
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
### Search with Details (CJK Characters)
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
uchr search -d "pray for happiness"
|
|
137
|
+
```
|
|
138
|
+
```
|
|
139
|
+
祝 795D CJK UNIFIED IDEOGRAPH-#; PRAY FOR HAPPINESS OR BLESSINGS
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
### Output Formatting
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
# Simple format (characters only)
|
|
146
|
+
uchr search ghost -f simple
|
|
147
|
+
👻
|
|
148
|
+
|
|
149
|
+
# UTF-8 format
|
|
150
|
+
uchr search ghost -f utf8
|
|
151
|
+
👻 F0 9F 91 BB GHOST
|
|
152
|
+
|
|
153
|
+
# Custom delimiter
|
|
154
|
+
uchr search ghost -D "|"
|
|
155
|
+
👻|1F47B|GHOST
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
## 🔧 Command Reference
|
|
159
|
+
|
|
160
|
+
### uchr
|
|
161
|
+
|
|
162
|
+
Main command with subcommands for all Unicode operations.
|
|
163
|
+
|
|
164
|
+
#### uchr search
|
|
165
|
+
|
|
166
|
+
Search Unicode characters with various criteria.
|
|
167
|
+
|
|
168
|
+
| Option | Short | Description |
|
|
169
|
+
|--------|-------|-------------|
|
|
170
|
+
| `--name` | | Search by character name (default) |
|
|
171
|
+
| `--code` | `-c` | Search by code point or range |
|
|
172
|
+
| `--char` | `-x` | Search by character |
|
|
173
|
+
| `--block` | `-b` | Search by Unicode block |
|
|
174
|
+
| `--detail` | `-d` | Search in character details |
|
|
175
|
+
| `--strict` | `-s` | Exact match (case insensitive) |
|
|
176
|
+
| `--first` | `-1` | Show first result only |
|
|
177
|
+
| `--format` | `-f` | Output format: `utf8`, `simple` |
|
|
178
|
+
| `--delimiter` | `-D` | Custom delimiter (default: space) |
|
|
179
|
+
|
|
180
|
+
#### uchr db
|
|
181
|
+
|
|
182
|
+
Database management operations.
|
|
183
|
+
|
|
184
|
+
| Subcommand | Description |
|
|
185
|
+
|------------|-------------|
|
|
186
|
+
| `uchr db create` | Create/update Unicode database |
|
|
187
|
+
| `uchr db delete` | Remove Unicode database |
|
|
188
|
+
| `uchr db info` | Show database location |
|
|
189
|
+
|
|
190
|
+
## 💾 Database Management
|
|
191
|
+
|
|
192
|
+
### Create Database
|
|
193
|
+
```bash
|
|
194
|
+
uchr db create
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
### Check Database Location
|
|
198
|
+
```bash
|
|
199
|
+
uchr db info
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
### Remove Database
|
|
203
|
+
```bash
|
|
204
|
+
uchr db delete
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
### Environment Variables
|
|
208
|
+
|
|
209
|
+
- `UNICODE_DB_PATH`: Override default database location
|
|
210
|
+
|
|
211
|
+
```bash
|
|
212
|
+
export UNICODE_DB_PATH="/custom/path/unicode.db"
|
|
213
|
+
uchr db create
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
## 🌟 Advanced Examples
|
|
217
|
+
|
|
218
|
+
### Finding Emoji Sequences
|
|
219
|
+
|
|
220
|
+
```bash
|
|
221
|
+
# National flags
|
|
222
|
+
uchr search -b "RGI_Emoji_Flag_Sequence"
|
|
223
|
+
|
|
224
|
+
# Family emoji with ZWJ sequences
|
|
225
|
+
uchr search family
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
### Terminal Display vs. Browser/Application Support
|
|
229
|
+
|
|
230
|
+
Many terminals don't properly display complex emoji sequences, but the characters work correctly when copied to browsers or applications.
|
|
231
|
+
|
|
232
|
+
#### National Flag Example
|
|
233
|
+
|
|
234
|
+
When searching for flags, you might see separate letters in your terminal:
|
|
235
|
+
|
|
236
|
+
```bash
|
|
237
|
+
uchr search -b "RGI_Emoji_Flag_Sequence" | grep -i norway
|
|
238
|
+
```
|
|
239
|
+
```
|
|
240
|
+
🇳🇴 1F1F3 1F1F4 flag: Norway
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+

|
|
244
|
+
|
|
245
|
+
Even though you see two separate letters (🇳🇴) in the terminal, when you copy and paste them into a browser or application like Twitter, they combine to display the Norwegian flag 🇳🇴.
|
|
246
|
+
|
|
247
|
+

|
|
248
|
+
|
|
249
|
+
#### ZWJ Sequence Example
|
|
250
|
+
|
|
251
|
+
The same applies to Zero Width Joiner (ZWJ) sequences. Complex emoji like family groups or professional emoji might not render correctly in terminals:
|
|
252
|
+
|
|
253
|
+
```bash
|
|
254
|
+
uchr search "polar bear"
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
In a terminal without proper font support:
|
|
258
|
+
|
|
259
|
+

|
|
260
|
+
|
|
261
|
+
But when pasted in Twitter or other applications:
|
|
262
|
+
|
|
263
|
+

|
|
264
|
+
|
|
265
|
+
> **💡 Tip**: This is expected behavior. The Unicode data is correct, and the characters will work properly in applications that support modern emoji rendering.
|
|
266
|
+
|
|
267
|
+
### Pipe Operations
|
|
268
|
+
|
|
269
|
+
```bash
|
|
270
|
+
# Get just the character
|
|
271
|
+
uchr search ghost -f simple
|
|
272
|
+
|
|
273
|
+
# First match only
|
|
274
|
+
uchr search snow -1
|
|
275
|
+
|
|
276
|
+
# Custom format for scripting
|
|
277
|
+
uchr search ghost -D "," | cut -d',' -f1
|
|
278
|
+
```
|
|
279
|
+
|
|
280
|
+
### Complex Searches
|
|
281
|
+
|
|
282
|
+
```bash
|
|
283
|
+
# CJK characters with specific meanings
|
|
284
|
+
uchr search -d "dragon"
|
|
285
|
+
|
|
286
|
+
# Characters in multiple blocks
|
|
287
|
+
uchr search -b "Mathematical" | head -10
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
## 🏗 Data Sources
|
|
291
|
+
|
|
292
|
+
This tool uses official Unicode 15.0 data:
|
|
293
|
+
|
|
294
|
+
- [Unicode Character Database](https://www.unicode.org/Public/15.0.0/ucdxml/ucd.all.flat.zip)
|
|
295
|
+
- [Emoji Sequences](https://www.unicode.org/Public/emoji/15.0/emoji-sequences.txt)
|
|
296
|
+
- [Emoji ZWJ Sequences](https://www.unicode.org/Public/emoji/15.0/emoji-zwj-sequences.txt)
|
|
297
|
+
|
|
298
|
+
## 🤝 Contributing
|
|
299
|
+
|
|
300
|
+
1. Fork the repository
|
|
301
|
+
2. Create a feature branch (`git checkout -b feature/amazing-feature`)
|
|
302
|
+
3. Commit your changes (`git commit -m 'Add amazing feature'`)
|
|
303
|
+
4. Push to the branch (`git push origin feature/amazing-feature`)
|
|
304
|
+
5. Open a Pull Request
|
|
305
|
+
|
|
306
|
+
## 📄 License
|
|
307
|
+
|
|
308
|
+
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
|
309
|
+
|
|
310
|
+
## 🙏 Acknowledgments
|
|
311
|
+
|
|
312
|
+
- [Unicode Consortium](https://unicode.org/) for maintaining Unicode standards
|
|
313
|
+
- Contributors and users of this project
|
|
314
|
+
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
unicode_tools/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
unicode_tools/database.py,sha256=WqPBF7Rmyq74ONv3awcuUsSQkB5hGJZ3MB2v7rbQsck,11329
|
|
3
|
+
unicode_tools/db.py,sha256=G2y13GA_XdoV13Lmy-wcPEQ93-QdU03tWRrL6ALYyhs,4463
|
|
4
|
+
unicode_tools/normalize.py,sha256=ZeybMGBIWTaQjjyrg75DNsqyNwM_4p4SAqn2Wfm23QU,3560
|
|
5
|
+
unicode_tools/search.py,sha256=ovhGPTFzXuRs0YX_ljhzttTqw5IC20WAtOcwEexn4DM,2636
|
|
6
|
+
unicode_tools/uchr.py,sha256=qAutAuD8hNnMnhpulxveYkiIVTcvEzMqQg6O3OPFNQQ,6807
|
|
7
|
+
uchr-0.3.0.dist-info/LICENSE,sha256=9btWvx8-OK73XlFie43RPuK8tWhKKu_IcEjqtVQjC-c,1068
|
|
8
|
+
uchr-0.3.0.dist-info/METADATA,sha256=5ddxWmIyOvFMbNZBF8cu8vrFAimPbKMSkxz5dtVSBhc,7977
|
|
9
|
+
uchr-0.3.0.dist-info/WHEEL,sha256=fGIA9gx4Qxk2KDKeNJCbOEwSrmLtjWCwzBz351GyrPQ,88
|
|
10
|
+
uchr-0.3.0.dist-info/entry_points.txt,sha256=l5mV9P_tkP_q8Z4_rGP1CbP7vV2dyMTos6C-jwzkR3c,48
|
|
11
|
+
uchr-0.3.0.dist-info/RECORD,,
|
|
File without changes
|
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import io
|
|
4
|
+
import os
|
|
5
|
+
import re
|
|
6
|
+
import sqlite3
|
|
7
|
+
import sys
|
|
8
|
+
import tempfile
|
|
9
|
+
import xml.etree.ElementTree as et
|
|
10
|
+
import zipfile
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from urllib.parse import urlparse
|
|
13
|
+
|
|
14
|
+
import requests
|
|
15
|
+
|
|
16
|
+
from .db import AutoID, Connection, Cursor, Database
|
|
17
|
+
|
|
18
|
+
namespace = "{http://www.unicode.org/ns/2003/ucd/1.0}"
|
|
19
|
+
tag_ucd = namespace + "ucd"
|
|
20
|
+
tag_description = namespace + "description"
|
|
21
|
+
tag_repertoire = namespace + "repertoire"
|
|
22
|
+
tag_char = namespace + "char"
|
|
23
|
+
tag_noncharacter = namespace + "noncharacter"
|
|
24
|
+
tag_reserved = namespace + "reserved"
|
|
25
|
+
tag_surrogate = namespace + "surrogate"
|
|
26
|
+
tag_name_alias = namespace + "name-alias"
|
|
27
|
+
|
|
28
|
+
table_char_autoincrement_id = AutoID().init()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def download_ucd(ucd_zip_url):
|
|
32
|
+
ucd_zip_url_path = Path(urlparse(ucd_zip_url)[2])
|
|
33
|
+
ucd_zip_filename = ucd_zip_url_path.name
|
|
34
|
+
ucd_xml_filename = ucd_zip_url_path.with_suffix(".xml").name
|
|
35
|
+
|
|
36
|
+
zip_filepath = "(Not assigned)"
|
|
37
|
+
|
|
38
|
+
try:
|
|
39
|
+
with tempfile.TemporaryDirectory() as tmpdir:
|
|
40
|
+
zip_filepath = os.path.join(tmpdir, ucd_zip_filename)
|
|
41
|
+
|
|
42
|
+
print(f"Downloading {ucd_zip_url} ...", file=sys.stderr)
|
|
43
|
+
|
|
44
|
+
res = requests.get(ucd_zip_url, stream=True)
|
|
45
|
+
if res.status_code >= 400:
|
|
46
|
+
print(f"Fetch error: {res.status_code}", file=sys.stderr)
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
content_type = res.headers["Content-Type"]
|
|
50
|
+
if content_type != "application/zip":
|
|
51
|
+
print(f"Invalid content type: {content_type}")
|
|
52
|
+
return None
|
|
53
|
+
|
|
54
|
+
with open(zip_filepath, "wb") as fzip:
|
|
55
|
+
for chunk in res.iter_content(chunk_size=1024):
|
|
56
|
+
if chunk:
|
|
57
|
+
fzip.write(chunk)
|
|
58
|
+
fzip.flush()
|
|
59
|
+
|
|
60
|
+
print(f"Downloaded {zip_filepath}", file=sys.stderr)
|
|
61
|
+
|
|
62
|
+
with zipfile.ZipFile(zip_filepath, "r") as zip:
|
|
63
|
+
xml_list = zip.read(ucd_xml_filename)
|
|
64
|
+
|
|
65
|
+
print("Extracted unicode data xml", file=sys.stderr)
|
|
66
|
+
|
|
67
|
+
return xml_list
|
|
68
|
+
|
|
69
|
+
except Exception as e:
|
|
70
|
+
print(
|
|
71
|
+
f"Failed to download zip from {ucd_zip_url} to {zip_filepath}",
|
|
72
|
+
file=sys.stderr,
|
|
73
|
+
)
|
|
74
|
+
print(f"{type(e).__name__}: {str(e)}", file=sys.stderr)
|
|
75
|
+
return None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def get_ucd_cp(tag):
|
|
79
|
+
cp = tag.attrib.get("cp")
|
|
80
|
+
first_cp = tag.attrib.get("first-cp")
|
|
81
|
+
last_cp = tag.attrib.get("last-cp")
|
|
82
|
+
|
|
83
|
+
if cp:
|
|
84
|
+
first_cp = cp
|
|
85
|
+
last_cp = cp
|
|
86
|
+
|
|
87
|
+
if not (first_cp and last_cp):
|
|
88
|
+
print(
|
|
89
|
+
f"Invalid code range: cp={cp}, first_cp={first_cp}, last_cp={last_cp}",
|
|
90
|
+
file=sys.stderr,
|
|
91
|
+
)
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
min = int(first_cp, 16)
|
|
95
|
+
max = int(last_cp, 16)
|
|
96
|
+
|
|
97
|
+
return (min, max)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def get_ucd_char_cp(char):
|
|
101
|
+
value = None
|
|
102
|
+
r = get_ucd_cp(char)
|
|
103
|
+
if r:
|
|
104
|
+
min = r[0]
|
|
105
|
+
max = r[1]
|
|
106
|
+
if char.tag != tag_char:
|
|
107
|
+
if min == max:
|
|
108
|
+
code_range = f"{min:X}"
|
|
109
|
+
else:
|
|
110
|
+
code_range = f"{min:X}-{max:X}"
|
|
111
|
+
|
|
112
|
+
if char.tag == tag_reserved:
|
|
113
|
+
print(f"Found reserved code(s): {code_range}", file=sys.stderr)
|
|
114
|
+
elif char.tag == tag_noncharacter:
|
|
115
|
+
print(f"Found non character code(s): {code_range}", file=sys.stderr)
|
|
116
|
+
elif char.tag == tag_surrogate:
|
|
117
|
+
print(f"Found surrogate code(s): {code_range}", file=sys.stderr)
|
|
118
|
+
else:
|
|
119
|
+
print(f"Found unknown tag: {char.tag} {code_range}", file=sys.stderr)
|
|
120
|
+
return []
|
|
121
|
+
|
|
122
|
+
value = range(min, max + 1)
|
|
123
|
+
return value
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def get_name(char):
|
|
127
|
+
value = []
|
|
128
|
+
name = char.attrib.get("na")
|
|
129
|
+
name1 = char.attrib.get("na1")
|
|
130
|
+
if name and len(name) > 0:
|
|
131
|
+
value.append(name)
|
|
132
|
+
if name1 and len(name1) > 0 and name != name1:
|
|
133
|
+
value.append(name1)
|
|
134
|
+
for alias in char:
|
|
135
|
+
if alias.tag == tag_name_alias:
|
|
136
|
+
alias_name = alias.attrib.get("alias")
|
|
137
|
+
if alias_name and len(alias_name) > 0 and alias_name not in value:
|
|
138
|
+
value.append(alias_name)
|
|
139
|
+
return "; ".join(value)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def get_detail(char, name):
|
|
143
|
+
definition = char.attrib.get("kDefinition")
|
|
144
|
+
if definition and len(definition) > 0:
|
|
145
|
+
return "; ".join([name, definition.upper()])
|
|
146
|
+
else:
|
|
147
|
+
return name
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def store_ucd(xml_list):
|
|
151
|
+
if not xml_list:
|
|
152
|
+
return None
|
|
153
|
+
|
|
154
|
+
root = et.parse(io.BytesIO(xml_list)).getroot()
|
|
155
|
+
if root.tag != tag_ucd:
|
|
156
|
+
print(f"Unexpected XML scheme: {root.tag}", file=sys.stderr)
|
|
157
|
+
return None
|
|
158
|
+
|
|
159
|
+
repertoire = root.find(tag_repertoire)
|
|
160
|
+
|
|
161
|
+
with Connection() as conn:
|
|
162
|
+
with Cursor(conn) as cur:
|
|
163
|
+
count = 0
|
|
164
|
+
for char in repertoire:
|
|
165
|
+
code_range = get_ucd_char_cp(char)
|
|
166
|
+
for code in code_range:
|
|
167
|
+
value_code = code
|
|
168
|
+
value_code_text = f'"{value_code:X}"'
|
|
169
|
+
name = get_name(char)
|
|
170
|
+
if not name:
|
|
171
|
+
print(f"Found no character: {code:X}", file=sys.stderr)
|
|
172
|
+
continue
|
|
173
|
+
value_name = f'"{name}"'
|
|
174
|
+
detail = get_detail(char, name)
|
|
175
|
+
value_detail = f'"{detail}"'
|
|
176
|
+
|
|
177
|
+
try:
|
|
178
|
+
if code == 0:
|
|
179
|
+
value_char = "NULL"
|
|
180
|
+
else:
|
|
181
|
+
escaped_char = str(chr(code)).replace('"', '""')
|
|
182
|
+
value_char = f'"{escaped_char}"'
|
|
183
|
+
except ValueError:
|
|
184
|
+
print(f"Invalid character {code:X} ({name})", file=sys.stderr)
|
|
185
|
+
continue
|
|
186
|
+
|
|
187
|
+
block = char.attrib.get("blk")
|
|
188
|
+
if not block:
|
|
189
|
+
print(f"No block name: {code:X}", file=sys.stderr)
|
|
190
|
+
value_block = '"(None)"'
|
|
191
|
+
else:
|
|
192
|
+
value_block = f'"{block}"'
|
|
193
|
+
|
|
194
|
+
id = table_char_autoincrement_id.next()
|
|
195
|
+
dml = f"insert into char(id, name, detail, codetext, char, block) values({id}, {value_name}, {value_detail}, {value_code_text}, {value_char}, {value_block})"
|
|
196
|
+
cur.execute(dml)
|
|
197
|
+
dml_seq = f"insert into codepoint(char, seq, code) values({id}, 1, {value_code})"
|
|
198
|
+
cur.execute(dml_seq)
|
|
199
|
+
count = count + 1
|
|
200
|
+
|
|
201
|
+
conn.commit()
|
|
202
|
+
print(f"Stored {count} characters", file=sys.stderr)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def download_emoji(emoji_txt_url):
|
|
206
|
+
try:
|
|
207
|
+
res = requests.get(emoji_txt_url)
|
|
208
|
+
if res.status_code >= 400:
|
|
209
|
+
print(f"Fetch error: {res.status_code}", file=sys.stderr)
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
content_type = res.headers["Content-Type"]
|
|
213
|
+
if not ("text/plain" in content_type and "charset=utf-8" in content_type):
|
|
214
|
+
print(f"Invalid content type: {content_type}")
|
|
215
|
+
return None
|
|
216
|
+
|
|
217
|
+
return res.text.splitlines()
|
|
218
|
+
|
|
219
|
+
except Exception as e:
|
|
220
|
+
print(f"Failed to download emoji from {emoji_txt_url}", file=sys.stderr)
|
|
221
|
+
print(f"{type(e).__name__}: {str(e)}", file=sys.stderr)
|
|
222
|
+
return None
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def store_emoji(emoji_sequences):
|
|
226
|
+
if not emoji_sequences:
|
|
227
|
+
return
|
|
228
|
+
|
|
229
|
+
emoji_sequence_line_pattern = re.compile("^(.+);(.+);([^#]+)#")
|
|
230
|
+
emoji_sequence_cp_pattern = re.compile("([0-9A-Fa-f]+)")
|
|
231
|
+
emoji_sequence_multi_pattern = re.compile("([0-9A-Fa-f]+)")
|
|
232
|
+
emoji_sequence_continuous_pattern = re.compile(r"([0-9A-Fa-f]+)\.\.([0-9A-Fa-f]+)")
|
|
233
|
+
|
|
234
|
+
with Connection() as conn:
|
|
235
|
+
with Cursor(conn) as cur:
|
|
236
|
+
count = 0
|
|
237
|
+
for sequence in emoji_sequences:
|
|
238
|
+
if len(sequence) == 0 or sequence.startswith("#"):
|
|
239
|
+
continue
|
|
240
|
+
emoji = emoji_sequence_line_pattern.match(sequence)
|
|
241
|
+
emoji_codes = emoji.group(1).strip()
|
|
242
|
+
emoji_type = emoji.group(2).strip()
|
|
243
|
+
emoji_name = emoji.group(3).strip()
|
|
244
|
+
|
|
245
|
+
cp_list = []
|
|
246
|
+
cp = re.fullmatch(emoji_sequence_cp_pattern, emoji_codes)
|
|
247
|
+
if cp:
|
|
248
|
+
cp_list.append(int(emoji_codes, 16))
|
|
249
|
+
else:
|
|
250
|
+
cp = re.match(emoji_sequence_continuous_pattern, emoji_codes)
|
|
251
|
+
if cp:
|
|
252
|
+
min = int(cp.group(1), 16)
|
|
253
|
+
max = int(cp.group(2), 16)
|
|
254
|
+
cp_list.extend(list(range(min, max + 1)))
|
|
255
|
+
else:
|
|
256
|
+
seq = []
|
|
257
|
+
for cp in re.finditer(
|
|
258
|
+
emoji_sequence_multi_pattern, emoji_codes
|
|
259
|
+
):
|
|
260
|
+
seq.append(int(cp.group(1), 16))
|
|
261
|
+
cp_list.append(seq)
|
|
262
|
+
if len(cp_list) == 0:
|
|
263
|
+
print(
|
|
264
|
+
f"Failed to get code points: {emoji_codes}, {emoji_name}",
|
|
265
|
+
file=sys.stderr,
|
|
266
|
+
)
|
|
267
|
+
continue
|
|
268
|
+
|
|
269
|
+
value_name = f'"{emoji_name}"'
|
|
270
|
+
value_code_text = f'"{emoji_codes}"'
|
|
271
|
+
value_block = f'"{emoji_type}"'
|
|
272
|
+
for code in cp_list:
|
|
273
|
+
if type(code) is int:
|
|
274
|
+
value_code_text = f'"{code:X}"'
|
|
275
|
+
char = chr(code)
|
|
276
|
+
value_char = f'"{char}"'
|
|
277
|
+
else:
|
|
278
|
+
char = ""
|
|
279
|
+
for c in code:
|
|
280
|
+
char = char + chr(c)
|
|
281
|
+
value_char = f'"{char}"'
|
|
282
|
+
|
|
283
|
+
try:
|
|
284
|
+
id = table_char_autoincrement_id.next()
|
|
285
|
+
dml = f"insert into char(id, name, codetext, char, block) values({id}, {value_name}, {value_code_text}, {value_char}, {value_block})"
|
|
286
|
+
cur.execute(dml)
|
|
287
|
+
if type(code) is int:
|
|
288
|
+
dml_seq = f"insert into codepoint(char, seq, code) values({id}, 1, {code})"
|
|
289
|
+
cur.execute(dml_seq)
|
|
290
|
+
else:
|
|
291
|
+
for i, c in enumerate(code):
|
|
292
|
+
dml_seq = f"insert into codepoint(char, seq, code) values({id}, {i}, {c})"
|
|
293
|
+
cur.execute(dml_seq)
|
|
294
|
+
count = count + 1
|
|
295
|
+
except sqlite3.IntegrityError:
|
|
296
|
+
print(
|
|
297
|
+
f"Already registered: {value_code_text} {value_name}",
|
|
298
|
+
file=sys.stderr,
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
conn.commit()
|
|
302
|
+
print(f"Stored {count} emoji characters", file=sys.stderr)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def create_database():
|
|
306
|
+
"""Create Unicode database"""
|
|
307
|
+
Database().create()
|
|
308
|
+
store_ucd(
|
|
309
|
+
download_ucd("https://www.unicode.org/Public/15.0.0/ucdxml/ucd.all.flat.zip")
|
|
310
|
+
)
|
|
311
|
+
store_emoji(
|
|
312
|
+
download_emoji("https://www.unicode.org/Public/emoji/15.0/emoji-sequences.txt")
|
|
313
|
+
)
|
|
314
|
+
store_emoji(
|
|
315
|
+
download_emoji(
|
|
316
|
+
"https://www.unicode.org/Public/emoji/15.0/emoji-zwj-sequences.txt"
|
|
317
|
+
)
|
|
318
|
+
)
|
|
319
|
+
return 0
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def delete_database():
|
|
323
|
+
"""Delete Unicode database"""
|
|
324
|
+
Database().delete()
|
|
325
|
+
return 0
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def database_info():
|
|
329
|
+
"""Show database information"""
|
|
330
|
+
print(Database().get_path())
|
|
331
|
+
return 0
|
unicode_tools/db.py
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import sqlite3
|
|
3
|
+
import sys
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def get_data_dir():
|
|
8
|
+
"""データディレクトリのパスを決定する"""
|
|
9
|
+
# 環境変数でオーバーライド可能
|
|
10
|
+
if "UNICODE_DB_PATH" in os.environ:
|
|
11
|
+
return Path(os.environ["UNICODE_DB_PATH"]).parent
|
|
12
|
+
|
|
13
|
+
if os.getuid() == 0: # rootユーザー
|
|
14
|
+
# /procを使って親プロセスをチェック(Linuxのみ)
|
|
15
|
+
try:
|
|
16
|
+
with open("/proc/self/stat") as f:
|
|
17
|
+
stats = f.read().split()
|
|
18
|
+
parent_pid = int(stats[3]) # 4番目が親プロセスID
|
|
19
|
+
|
|
20
|
+
with open(f"/proc/{parent_pid}/comm") as f:
|
|
21
|
+
parent_name = f.read().strip()
|
|
22
|
+
|
|
23
|
+
if parent_name in ["systemd", "cron"]:
|
|
24
|
+
# システムサービスとして実行されている
|
|
25
|
+
return Path("/var/lib/unicode-tools")
|
|
26
|
+
except (FileNotFoundError, ValueError, IndexError):
|
|
27
|
+
# /procが読めない場合やLinux以外の場合
|
|
28
|
+
pass
|
|
29
|
+
|
|
30
|
+
# rootの個人利用
|
|
31
|
+
return Path("/root/.local/share/unicode-tools")
|
|
32
|
+
else:
|
|
33
|
+
# 通常ユーザー
|
|
34
|
+
return Path.home() / ".local/share/unicode-tools"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
# データベースパスの設定
|
|
38
|
+
data_dir = get_data_dir()
|
|
39
|
+
unicode_sqlite3_database_path = str(data_dir / "unicode.db")
|
|
40
|
+
unicode_sqlite3_database_dir = str(data_dir)
|
|
41
|
+
|
|
42
|
+
# ディレクトリが存在しない場合は作成(ただし警告を出す)
|
|
43
|
+
if not os.path.exists(unicode_sqlite3_database_dir):
|
|
44
|
+
try:
|
|
45
|
+
os.makedirs(unicode_sqlite3_database_dir)
|
|
46
|
+
print(f"Created directory: {unicode_sqlite3_database_dir}", file=sys.stderr)
|
|
47
|
+
except PermissionError:
|
|
48
|
+
print(
|
|
49
|
+
f"Permission denied: Cannot create directory {unicode_sqlite3_database_dir}",
|
|
50
|
+
file=sys.stderr,
|
|
51
|
+
)
|
|
52
|
+
print(
|
|
53
|
+
"Please create the directory manually or set UNICODE_DB_PATH environment variable",
|
|
54
|
+
file=sys.stderr,
|
|
55
|
+
)
|
|
56
|
+
sys.exit(1)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Database:
|
|
60
|
+
def create(self):
|
|
61
|
+
def execute(conn, name, dml):
|
|
62
|
+
with Cursor(conn) as cur:
|
|
63
|
+
try:
|
|
64
|
+
cur.execute(dml)
|
|
65
|
+
conn.commit()
|
|
66
|
+
print(f"Created table: {name}")
|
|
67
|
+
except Exception as e:
|
|
68
|
+
t = type(e)
|
|
69
|
+
s = str(e)
|
|
70
|
+
if t == sqlite3.OperationalError and s.endswith(" already exists"):
|
|
71
|
+
print(f"Table already exists: {name}", file=sys.stderr)
|
|
72
|
+
else:
|
|
73
|
+
print(f"Failed to create table: {name}", file=sys.stderr)
|
|
74
|
+
print(f"{type(e).__name__}: {str(e)}", file=sys.stderr)
|
|
75
|
+
|
|
76
|
+
with Connection() as conn:
|
|
77
|
+
execute(
|
|
78
|
+
conn,
|
|
79
|
+
"char",
|
|
80
|
+
"create table char(id integer primary key, name text, detail text, codetext text, char text, block text)",
|
|
81
|
+
)
|
|
82
|
+
execute(
|
|
83
|
+
conn,
|
|
84
|
+
"codepoint",
|
|
85
|
+
"create table codepoint(char integer, seq integer, code integer, primary key(char, seq))",
|
|
86
|
+
)
|
|
87
|
+
execute(
|
|
88
|
+
conn, "char_index", "create unique index char_index on char(codetext)"
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
def delete(self):
|
|
92
|
+
if not os.path.exists(unicode_sqlite3_database_path):
|
|
93
|
+
print(f"No database file: {unicode_sqlite3_database_path}", file=sys.stderr)
|
|
94
|
+
else:
|
|
95
|
+
os.remove(unicode_sqlite3_database_path)
|
|
96
|
+
print(
|
|
97
|
+
f"Deleted database file: {unicode_sqlite3_database_path}",
|
|
98
|
+
file=sys.stderr,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
def get_path(self):
|
|
102
|
+
return unicode_sqlite3_database_path
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class Connection:
|
|
106
|
+
def __init__(self):
|
|
107
|
+
self.conn = None
|
|
108
|
+
|
|
109
|
+
def __enter__(self):
|
|
110
|
+
self.conn = sqlite3.connect(unicode_sqlite3_database_path)
|
|
111
|
+
return self.conn
|
|
112
|
+
|
|
113
|
+
def __exit__(self, *args):
|
|
114
|
+
if self.conn:
|
|
115
|
+
self.conn.close()
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class Cursor:
|
|
119
|
+
def __init__(self, conn):
|
|
120
|
+
self.conn = conn
|
|
121
|
+
self.cur = None
|
|
122
|
+
|
|
123
|
+
def __enter__(self):
|
|
124
|
+
self.cur = self.conn.cursor()
|
|
125
|
+
return self.cur
|
|
126
|
+
|
|
127
|
+
def __exit__(self, *args):
|
|
128
|
+
if self.cur:
|
|
129
|
+
self.cur.close()
|
|
130
|
+
self.cur = None
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class AutoID:
|
|
134
|
+
def init(self):
|
|
135
|
+
self.value = 1
|
|
136
|
+
return self
|
|
137
|
+
|
|
138
|
+
def next(self):
|
|
139
|
+
value = self.value
|
|
140
|
+
self.value = self.value + 1
|
|
141
|
+
return value
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
import unicodedata
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def convert_to_halfwidth(text):
|
|
8
|
+
"""Convert fullwidth characters to halfwidth"""
|
|
9
|
+
return "".join(
|
|
10
|
+
chr(ord(char) - 0xFEE0)
|
|
11
|
+
if 0xFF01 <= ord(char) <= 0xFF5E
|
|
12
|
+
else " "
|
|
13
|
+
if char == " " # Fullwidth space to regular space
|
|
14
|
+
else char
|
|
15
|
+
for char in text
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def get_binary_representation(text):
|
|
20
|
+
"""Get binary representation of text as hex string with colon separators"""
|
|
21
|
+
return ":".join(f"{b:02x}" for b in text.encode("utf-8"))
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def get_unicode_representation(text):
|
|
25
|
+
"""Get Unicode code point representation with colon separators"""
|
|
26
|
+
return ":".join(f"U+{ord(c):04X}" for c in text)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def escape_control_chars(text):
|
|
30
|
+
"""Replace control characters and space with dots for display"""
|
|
31
|
+
return "".join("." if ord(char) <= 32 else char for char in text)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def format_detailed_output(text, form, delimiter=" "):
|
|
35
|
+
"""Format text with detailed information: result form binary unicode"""
|
|
36
|
+
# Escape control characters in display text
|
|
37
|
+
display_text = escape_control_chars(text)
|
|
38
|
+
binary = get_binary_representation(text)
|
|
39
|
+
unicode_repr = get_unicode_representation(text)
|
|
40
|
+
return delimiter.join([display_text, form, binary, unicode_repr])
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def normalize_text(text, form="NFC", halfwidth=False):
|
|
44
|
+
"""Normalize text using specified Unicode normalization form and halfwidth conversion"""
|
|
45
|
+
# Apply Unicode normalization
|
|
46
|
+
normalized = (
|
|
47
|
+
unicodedata.normalize(form.upper(), text)
|
|
48
|
+
if form.upper() in ["NFC", "NFD", "NFKC", "NFKD"]
|
|
49
|
+
else text
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
# Apply halfwidth conversion if specified
|
|
53
|
+
return convert_to_halfwidth(normalized) if halfwidth else normalized
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def analyze_normalization(content, delimiter=" "):
|
|
57
|
+
"""Analyze text and show all normalization forms in consistent format"""
|
|
58
|
+
forms = ["Original", "NFC", "NFD", "NFKC", "NFKD"]
|
|
59
|
+
|
|
60
|
+
for form in forms:
|
|
61
|
+
if form == "Original":
|
|
62
|
+
result_text = content
|
|
63
|
+
else:
|
|
64
|
+
result_text = unicodedata.normalize(form, content)
|
|
65
|
+
|
|
66
|
+
print(format_detailed_output(result_text, form, delimiter))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def normalize_command(
|
|
70
|
+
form="NFC",
|
|
71
|
+
halfwidth=False,
|
|
72
|
+
compare=False,
|
|
73
|
+
detailed=False,
|
|
74
|
+
delimiter=" ",
|
|
75
|
+
input_file=None,
|
|
76
|
+
):
|
|
77
|
+
"""Main normalize command function"""
|
|
78
|
+
try:
|
|
79
|
+
# Read entire content as single string
|
|
80
|
+
if input_file and input_file != "-":
|
|
81
|
+
with open(input_file, encoding="utf-8") as f:
|
|
82
|
+
content = f.read().rstrip("\n")
|
|
83
|
+
else:
|
|
84
|
+
content = sys.stdin.read().rstrip("\n")
|
|
85
|
+
|
|
86
|
+
# Process content
|
|
87
|
+
if compare:
|
|
88
|
+
analyze_normalization(content, delimiter)
|
|
89
|
+
else:
|
|
90
|
+
result = normalize_text(content, form=form, halfwidth=halfwidth)
|
|
91
|
+
if detailed:
|
|
92
|
+
# Determine actual form used
|
|
93
|
+
actual_form = form.upper()
|
|
94
|
+
if halfwidth:
|
|
95
|
+
actual_form += "+HALFWIDTH"
|
|
96
|
+
print(format_detailed_output(result, actual_form, delimiter))
|
|
97
|
+
else:
|
|
98
|
+
print(
|
|
99
|
+
result, end=""
|
|
100
|
+
) # Don't add extra newline since content may already have it
|
|
101
|
+
|
|
102
|
+
except FileNotFoundError:
|
|
103
|
+
print(f"Error: File '{input_file}' not found", file=sys.stderr)
|
|
104
|
+
return 1
|
|
105
|
+
except UnicodeDecodeError:
|
|
106
|
+
print("Error: Unable to decode file as UTF-8", file=sys.stderr)
|
|
107
|
+
return 1
|
|
108
|
+
except BrokenPipeError:
|
|
109
|
+
# Handle pipe operations gracefully
|
|
110
|
+
pass
|
|
111
|
+
|
|
112
|
+
return 0
|
unicode_tools/search.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from .db import Connection, Cursor
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def get_code_range(fragment):
|
|
9
|
+
if "-" in fragment:
|
|
10
|
+
m = re.match("([0-9A-Fa-f]+)-([0-9A-Fa-f]+)", fragment)
|
|
11
|
+
min = int(m.group(1), 16)
|
|
12
|
+
max = int(m.group(2), 16)
|
|
13
|
+
r = (min, max)
|
|
14
|
+
else:
|
|
15
|
+
m = re.match("[0-9A-Fa-f]+", fragment)
|
|
16
|
+
r = int(fragment, 16)
|
|
17
|
+
return r
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def search(fragment, by, delimiter, strict=False, first=False, format=None):
|
|
21
|
+
if format is not None:
|
|
22
|
+
format = format.upper()
|
|
23
|
+
|
|
24
|
+
with Connection() as conn:
|
|
25
|
+
char_list = []
|
|
26
|
+
head = "select char.id, char.codetext, char.name, char.char from char"
|
|
27
|
+
head_detail = "select char.id, char.codetext, char.detail, char.char from char"
|
|
28
|
+
if by == "code":
|
|
29
|
+
code_range = get_code_range(fragment)
|
|
30
|
+
if type(code_range) is tuple:
|
|
31
|
+
cond = (
|
|
32
|
+
f"where cp.code >= {code_range[0]} and cp.code <= {code_range[1]}"
|
|
33
|
+
)
|
|
34
|
+
else:
|
|
35
|
+
cond = f"where cp.code = {code_range}"
|
|
36
|
+
dml = " ".join(
|
|
37
|
+
[
|
|
38
|
+
head,
|
|
39
|
+
"inner join codepoint as cp on char.id = cp.char",
|
|
40
|
+
cond,
|
|
41
|
+
"order by char.char",
|
|
42
|
+
]
|
|
43
|
+
)
|
|
44
|
+
elif by == "char":
|
|
45
|
+
cond = f'where char.char = "{fragment}"'
|
|
46
|
+
dml = " ".join([head, cond])
|
|
47
|
+
else:
|
|
48
|
+
if strict:
|
|
49
|
+
by = f"upper({by})"
|
|
50
|
+
matched = f'= "{fragment.upper()}"'
|
|
51
|
+
else:
|
|
52
|
+
matched = f'like "%{fragment}%"'
|
|
53
|
+
if by == "detail":
|
|
54
|
+
dml = " ".join(
|
|
55
|
+
[head_detail, "where", by, matched, "order by char.char"]
|
|
56
|
+
)
|
|
57
|
+
else:
|
|
58
|
+
dml = " ".join([head, "where", by, matched, "order by char.char"])
|
|
59
|
+
|
|
60
|
+
with Cursor(conn) as cur:
|
|
61
|
+
cur.execute(dml)
|
|
62
|
+
char_list = cur.fetchall()
|
|
63
|
+
|
|
64
|
+
if first == True:
|
|
65
|
+
char_list = char_list[0:1]
|
|
66
|
+
|
|
67
|
+
for id, codetext, name, char in char_list:
|
|
68
|
+
if not char:
|
|
69
|
+
char = str(char)
|
|
70
|
+
|
|
71
|
+
if format == "SIMPLE":
|
|
72
|
+
print(char, end="")
|
|
73
|
+
else:
|
|
74
|
+
if format == "UTF8":
|
|
75
|
+
codetext = " ".join(
|
|
76
|
+
f"{u:X}"
|
|
77
|
+
for u in [
|
|
78
|
+
int.from_bytes(chr(int(c, 16)).encode(), "big")
|
|
79
|
+
for c in codetext.split(" ")
|
|
80
|
+
]
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
print(delimiter.join([char, codetext, name]))
|
unicode_tools/uchr.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import io
|
|
5
|
+
import sys
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def wrap_io():
|
|
9
|
+
"""Wrap stdin/stdout/stderr with UTF-8 encoding"""
|
|
10
|
+
sys.stdin = io.TextIOWrapper(sys.stdin.buffer, encoding="utf-8")
|
|
11
|
+
sys.stdout = io.TextIOWrapper(
|
|
12
|
+
sys.stdout.buffer, encoding="utf-8", line_buffering=True
|
|
13
|
+
)
|
|
14
|
+
sys.stderr = io.TextIOWrapper(
|
|
15
|
+
sys.stderr.buffer, encoding="utf-8", line_buffering=True
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def search_command(args):
|
|
20
|
+
"""Handle uchr search subcommand"""
|
|
21
|
+
from .search import search
|
|
22
|
+
|
|
23
|
+
# Convert args to match search.search signature
|
|
24
|
+
by = args.by if args.by else "name"
|
|
25
|
+
|
|
26
|
+
if (by == "code" or by == "char") and args.strict:
|
|
27
|
+
print(f"warning: Ignore --strict in {by} search")
|
|
28
|
+
|
|
29
|
+
search(
|
|
30
|
+
args.expression,
|
|
31
|
+
by,
|
|
32
|
+
args.delimiter,
|
|
33
|
+
strict=args.strict,
|
|
34
|
+
first=args.first,
|
|
35
|
+
format=args.format,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def normalize_command(args):
|
|
40
|
+
"""Handle uchr normalize subcommand"""
|
|
41
|
+
from .normalize import normalize_command as normalize_func
|
|
42
|
+
|
|
43
|
+
return normalize_func(
|
|
44
|
+
form=args.form,
|
|
45
|
+
halfwidth=args.halfwidth,
|
|
46
|
+
compare=args.compare,
|
|
47
|
+
detailed=args.detail,
|
|
48
|
+
delimiter=args.delimiter,
|
|
49
|
+
input_file=args.input_file,
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def db_create_command(args):
|
|
54
|
+
"""Handle uchr db create subcommand"""
|
|
55
|
+
from .database import create_database
|
|
56
|
+
|
|
57
|
+
return create_database()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def db_delete_command(args):
|
|
61
|
+
"""Handle uchr db delete subcommand"""
|
|
62
|
+
from .database import delete_database
|
|
63
|
+
|
|
64
|
+
return delete_database()
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def db_info_command(args):
|
|
68
|
+
"""Handle uchr db info subcommand"""
|
|
69
|
+
from .database import database_info
|
|
70
|
+
|
|
71
|
+
return database_info()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def create_parser():
|
|
75
|
+
"""Create the main argument parser with subcommands"""
|
|
76
|
+
parser = argparse.ArgumentParser(
|
|
77
|
+
prog="uchr",
|
|
78
|
+
description="Unicode character tools",
|
|
79
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
80
|
+
epilog="""
|
|
81
|
+
Examples:
|
|
82
|
+
uchr search ghost # Search for characters named 'ghost'
|
|
83
|
+
uchr search -c 1F47A-1F480 # Search by code range
|
|
84
|
+
uchr search -x 👻 # Search by character
|
|
85
|
+
uchr search -b "Emoticons" # Search by Unicode block
|
|
86
|
+
uchr normalize # Normalize text from stdin (NFC)
|
|
87
|
+
uchr normalize --form nfd # Normalize to NFD form
|
|
88
|
+
uchr normalize --halfwidth # Convert fullwidth to halfwidth
|
|
89
|
+
uchr normalize --detail # Show result form binary unicode
|
|
90
|
+
uchr normalize --compare # Show all normalization forms
|
|
91
|
+
uchr db create # Create Unicode database
|
|
92
|
+
uchr db info # Show database location
|
|
93
|
+
""",
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
subparsers = parser.add_subparsers(dest="command", help="Available commands")
|
|
97
|
+
|
|
98
|
+
# Search subcommand
|
|
99
|
+
search_parser = subparsers.add_parser("search", help="Search Unicode characters")
|
|
100
|
+
search_parser.add_argument(
|
|
101
|
+
"expression", metavar="EXPR", help="Expression to search"
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
search_by_group = search_parser.add_mutually_exclusive_group()
|
|
105
|
+
search_by_group.add_argument(
|
|
106
|
+
"-b",
|
|
107
|
+
"--block",
|
|
108
|
+
action="store_const",
|
|
109
|
+
dest="by",
|
|
110
|
+
const="block",
|
|
111
|
+
help="Search by block name",
|
|
112
|
+
)
|
|
113
|
+
search_by_group.add_argument(
|
|
114
|
+
"-c",
|
|
115
|
+
"--code",
|
|
116
|
+
action="store_const",
|
|
117
|
+
dest="by",
|
|
118
|
+
const="code",
|
|
119
|
+
help="Search by code point or range",
|
|
120
|
+
)
|
|
121
|
+
search_by_group.add_argument(
|
|
122
|
+
"-x",
|
|
123
|
+
"--char",
|
|
124
|
+
action="store_const",
|
|
125
|
+
dest="by",
|
|
126
|
+
const="char",
|
|
127
|
+
help="Search by character",
|
|
128
|
+
)
|
|
129
|
+
search_by_group.add_argument(
|
|
130
|
+
"-d",
|
|
131
|
+
"--detail",
|
|
132
|
+
action="store_const",
|
|
133
|
+
dest="by",
|
|
134
|
+
const="detail",
|
|
135
|
+
help="Search by character details",
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
search_parser.add_argument(
|
|
139
|
+
"-s",
|
|
140
|
+
"--strict",
|
|
141
|
+
action="store_true",
|
|
142
|
+
help="Match name strictly (case insensitive)",
|
|
143
|
+
)
|
|
144
|
+
search_parser.add_argument(
|
|
145
|
+
"-1", "--first", action="store_true", help="Show first result only"
|
|
146
|
+
)
|
|
147
|
+
search_parser.add_argument(
|
|
148
|
+
"-f", "--format", choices=["utf8", "simple"], default=None, help="Output format"
|
|
149
|
+
)
|
|
150
|
+
search_parser.add_argument(
|
|
151
|
+
"-D", "--delimiter", default=" ", help="Output delimiter (default: space)"
|
|
152
|
+
)
|
|
153
|
+
search_parser.set_defaults(func=search_command)
|
|
154
|
+
|
|
155
|
+
# Normalize subcommand
|
|
156
|
+
normalize_parser = subparsers.add_parser(
|
|
157
|
+
"normalize", help="Normalize and convert Unicode text"
|
|
158
|
+
)
|
|
159
|
+
normalize_parser.add_argument(
|
|
160
|
+
"input_file", nargs="?", default=None, help="Input file (default: stdin)"
|
|
161
|
+
)
|
|
162
|
+
normalize_parser.add_argument(
|
|
163
|
+
"--form",
|
|
164
|
+
choices=["nfc", "nfd", "nfkc", "nfkd"],
|
|
165
|
+
default="nfc",
|
|
166
|
+
help="Unicode normalization form (default: nfc)",
|
|
167
|
+
)
|
|
168
|
+
normalize_parser.add_argument(
|
|
169
|
+
"--halfwidth",
|
|
170
|
+
action="store_true",
|
|
171
|
+
help="Convert fullwidth characters to halfwidth",
|
|
172
|
+
)
|
|
173
|
+
normalize_parser.add_argument(
|
|
174
|
+
"--compare",
|
|
175
|
+
action="store_true",
|
|
176
|
+
help="Show comparison of all normalization forms",
|
|
177
|
+
)
|
|
178
|
+
normalize_parser.add_argument(
|
|
179
|
+
"--detail",
|
|
180
|
+
action="store_true",
|
|
181
|
+
help="Show detailed output: result form binary unicode",
|
|
182
|
+
)
|
|
183
|
+
normalize_parser.add_argument(
|
|
184
|
+
"--delimiter",
|
|
185
|
+
default=" ",
|
|
186
|
+
help="Delimiter for detailed output (default: space)",
|
|
187
|
+
)
|
|
188
|
+
normalize_parser.set_defaults(func=normalize_command)
|
|
189
|
+
|
|
190
|
+
# Database subcommand
|
|
191
|
+
db_parser = subparsers.add_parser("db", help="Database management")
|
|
192
|
+
db_subparsers = db_parser.add_subparsers(
|
|
193
|
+
dest="db_command", help="Database operations"
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
# db create
|
|
197
|
+
db_create_parser = db_subparsers.add_parser(
|
|
198
|
+
"create", help="Create Unicode database"
|
|
199
|
+
)
|
|
200
|
+
db_create_parser.set_defaults(func=db_create_command)
|
|
201
|
+
|
|
202
|
+
# db delete
|
|
203
|
+
db_delete_parser = db_subparsers.add_parser(
|
|
204
|
+
"delete", help="Delete Unicode database"
|
|
205
|
+
)
|
|
206
|
+
db_delete_parser.set_defaults(func=db_delete_command)
|
|
207
|
+
|
|
208
|
+
# db info
|
|
209
|
+
db_info_parser = db_subparsers.add_parser("info", help="Show database information")
|
|
210
|
+
db_info_parser.set_defaults(func=db_info_command)
|
|
211
|
+
|
|
212
|
+
return parser
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def main():
|
|
216
|
+
"""Main entry point for uchr command"""
|
|
217
|
+
wrap_io()
|
|
218
|
+
|
|
219
|
+
parser = create_parser()
|
|
220
|
+
args = parser.parse_args()
|
|
221
|
+
|
|
222
|
+
if not hasattr(args, "func"):
|
|
223
|
+
parser.print_help()
|
|
224
|
+
return 1
|
|
225
|
+
|
|
226
|
+
try:
|
|
227
|
+
return args.func(args) or 0
|
|
228
|
+
except KeyboardInterrupt:
|
|
229
|
+
print("\nInterrupted", file=sys.stderr)
|
|
230
|
+
return 1
|
|
231
|
+
except Exception as e:
|
|
232
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
233
|
+
return 1
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def uchr():
|
|
237
|
+
"""Entry point for console_scripts"""
|
|
238
|
+
return main()
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
if __name__ == "__main__":
|
|
242
|
+
sys.exit(main())
|