uchr 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of uchr might be problematic. Click here for more details.
- uchr-0.3.0/LICENSE +21 -0
- uchr-0.3.0/PKG-INFO +314 -0
- uchr-0.3.0/README.md +284 -0
- uchr-0.3.0/pyproject.toml +53 -0
- uchr-0.3.0/src/unicode_tools/__init__.py +0 -0
- uchr-0.3.0/src/unicode_tools/database.py +331 -0
- uchr-0.3.0/src/unicode_tools/db.py +141 -0
- uchr-0.3.0/src/unicode_tools/normalize.py +112 -0
- uchr-0.3.0/src/unicode_tools/search.py +83 -0
- uchr-0.3.0/src/unicode_tools/uchr.py +242 -0
uchr-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2023 Miki Yutani
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
uchr-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: uchr
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Unicode character tools
|
|
5
|
+
License: MIT
|
|
6
|
+
Keywords: unicode,character,tools,cli
|
|
7
|
+
Author: Miki Yutani
|
|
8
|
+
Author-email: mkyutani@gmail.com
|
|
9
|
+
Requires-Python: >=3.8,<4.0
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
23
|
+
Classifier: Topic :: Text Processing :: General
|
|
24
|
+
Classifier: Topic :: Utilities
|
|
25
|
+
Requires-Dist: requests (>=2.25.0,<3.0.0)
|
|
26
|
+
Project-URL: Homepage, https://github.com/mkyutani/unicode-tools
|
|
27
|
+
Project-URL: Repository, https://github.com/mkyutani/unicode-tools
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# Unicode Tools
|
|
31
|
+
|
|
32
|
+
[](https://opensource.org/licenses/MIT)
|
|
33
|
+
[](https://www.python.org/downloads/)
|
|
34
|
+
[](https://unicode.org/)
|
|
35
|
+
|
|
36
|
+
A powerful command-line tool for searching and exploring Unicode characters, emoji sequences, and character properties.
|
|
37
|
+
|
|
38
|
+
## 🚀 Features
|
|
39
|
+
|
|
40
|
+
- **Search by name**: Find characters by their Unicode name
|
|
41
|
+
- **Search by code**: Look up characters by code point or range
|
|
42
|
+
- **Search by character**: Reverse lookup from character to details
|
|
43
|
+
- **Search by block**: Explore characters within Unicode blocks
|
|
44
|
+
- **Emoji support**: Full support for emoji sequences and ZWJ sequences
|
|
45
|
+
- **CJK details**: Enhanced descriptions for CJK characters using kDefinition
|
|
46
|
+
- **Flexible output**: Multiple output formats for different use cases
|
|
47
|
+
|
|
48
|
+
## 📖 Table of Contents
|
|
49
|
+
|
|
50
|
+
- [Installation](#installation)
|
|
51
|
+
- [Quick Start](#quick-start)
|
|
52
|
+
- [Usage Examples](#usage-examples)
|
|
53
|
+
- [Command Reference](#command-reference)
|
|
54
|
+
- [Database Management](#database-management)
|
|
55
|
+
- [Contributing](#contributing)
|
|
56
|
+
- [License](#license)
|
|
57
|
+
|
|
58
|
+
## 🛠 Installation
|
|
59
|
+
|
|
60
|
+
### Install from source
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
git clone https://github.com/mkyutani/unicode-tools.git
|
|
64
|
+
cd unicode-tools
|
|
65
|
+
pip install -e .
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
### Initialize database
|
|
69
|
+
|
|
70
|
+
Create the Unicode database (required for first use):
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
uchr db create
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
This downloads Unicode 15.0 data and creates a local SQLite database (~13MB) at:
|
|
77
|
+
- Linux/macOS: `~/.local/share/unicode-tools/unicode.db`
|
|
78
|
+
- Root users: Automatically chooses between system (`/var/lib/unicode-tools/`) or personal location
|
|
79
|
+
|
|
80
|
+
## ⚡ Quick Start
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
# Search for ghost-related characters
|
|
84
|
+
uchr search ghost
|
|
85
|
+
|
|
86
|
+
# Find characters in a code range
|
|
87
|
+
uchr search -c 1F47A-1F480
|
|
88
|
+
|
|
89
|
+
# Search by character
|
|
90
|
+
uchr search -x 👻
|
|
91
|
+
|
|
92
|
+
# Search within a Unicode block
|
|
93
|
+
uchr search -b "Emoticons"
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## 📋 Usage Examples
|
|
97
|
+
|
|
98
|
+
### Search by Name
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
uchr search goblin
|
|
102
|
+
```
|
|
103
|
+
```
|
|
104
|
+
👺 1F47A JAPANESE GOBLIN
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
### Search by Code Range
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
uchr search -c 1F479-1F47B
|
|
111
|
+
```
|
|
112
|
+
```
|
|
113
|
+
👹 1F479 JAPANESE OGRE
|
|
114
|
+
👺 1F47A JAPANESE GOBLIN
|
|
115
|
+
👻 1F47B GHOST
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### Search by Character
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
uchr search -x 👻
|
|
122
|
+
```
|
|
123
|
+
```
|
|
124
|
+
👻 1F47B GHOST
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
### Search by Unicode Block
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
uchr search -b "Misc_Pictographs"
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
### Search with Details (CJK Characters)
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
uchr search -d "pray for happiness"
|
|
137
|
+
```
|
|
138
|
+
```
|
|
139
|
+
祝 795D CJK UNIFIED IDEOGRAPH-#; PRAY FOR HAPPINESS OR BLESSINGS
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
### Output Formatting
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
# Simple format (characters only)
|
|
146
|
+
uchr search ghost -f simple
|
|
147
|
+
👻
|
|
148
|
+
|
|
149
|
+
# UTF-8 format
|
|
150
|
+
uchr search ghost -f utf8
|
|
151
|
+
👻 F0 9F 91 BB GHOST
|
|
152
|
+
|
|
153
|
+
# Custom delimiter
|
|
154
|
+
uchr search ghost -D "|"
|
|
155
|
+
👻|1F47B|GHOST
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
## 🔧 Command Reference
|
|
159
|
+
|
|
160
|
+
### uchr
|
|
161
|
+
|
|
162
|
+
Main command with subcommands for all Unicode operations.
|
|
163
|
+
|
|
164
|
+
#### uchr search
|
|
165
|
+
|
|
166
|
+
Search Unicode characters with various criteria.
|
|
167
|
+
|
|
168
|
+
| Option | Short | Description |
|
|
169
|
+
|--------|-------|-------------|
|
|
170
|
+
| `--name` | | Search by character name (default) |
|
|
171
|
+
| `--code` | `-c` | Search by code point or range |
|
|
172
|
+
| `--char` | `-x` | Search by character |
|
|
173
|
+
| `--block` | `-b` | Search by Unicode block |
|
|
174
|
+
| `--detail` | `-d` | Search in character details |
|
|
175
|
+
| `--strict` | `-s` | Exact match (case insensitive) |
|
|
176
|
+
| `--first` | `-1` | Show first result only |
|
|
177
|
+
| `--format` | `-f` | Output format: `utf8`, `simple` |
|
|
178
|
+
| `--delimiter` | `-D` | Custom delimiter (default: space) |
|
|
179
|
+
|
|
180
|
+
#### uchr db
|
|
181
|
+
|
|
182
|
+
Database management operations.
|
|
183
|
+
|
|
184
|
+
| Subcommand | Description |
|
|
185
|
+
|------------|-------------|
|
|
186
|
+
| `uchr db create` | Create/update Unicode database |
|
|
187
|
+
| `uchr db delete` | Remove Unicode database |
|
|
188
|
+
| `uchr db info` | Show database location |
|
|
189
|
+
|
|
190
|
+
## 💾 Database Management
|
|
191
|
+
|
|
192
|
+
### Create Database
|
|
193
|
+
```bash
|
|
194
|
+
uchr db create
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
### Check Database Location
|
|
198
|
+
```bash
|
|
199
|
+
uchr db info
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
### Remove Database
|
|
203
|
+
```bash
|
|
204
|
+
uchr db delete
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
### Environment Variables
|
|
208
|
+
|
|
209
|
+
- `UNICODE_DB_PATH`: Override default database location
|
|
210
|
+
|
|
211
|
+
```bash
|
|
212
|
+
export UNICODE_DB_PATH="/custom/path/unicode.db"
|
|
213
|
+
uchr db create
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
## 🌟 Advanced Examples
|
|
217
|
+
|
|
218
|
+
### Finding Emoji Sequences
|
|
219
|
+
|
|
220
|
+
```bash
|
|
221
|
+
# National flags
|
|
222
|
+
uchr search -b "RGI_Emoji_Flag_Sequence"
|
|
223
|
+
|
|
224
|
+
# Family emoji with ZWJ sequences
|
|
225
|
+
uchr search family
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
### Terminal Display vs. Browser/Application Support
|
|
229
|
+
|
|
230
|
+
Many terminals don't properly display complex emoji sequences, but the characters work correctly when copied to browsers or applications.
|
|
231
|
+
|
|
232
|
+
#### National Flag Example
|
|
233
|
+
|
|
234
|
+
When searching for flags, you might see separate letters in your terminal:
|
|
235
|
+
|
|
236
|
+
```bash
|
|
237
|
+
uchr search -b "RGI_Emoji_Flag_Sequence" | grep -i norway
|
|
238
|
+
```
|
|
239
|
+
```
|
|
240
|
+
🇳🇴 1F1F3 1F1F4 flag: Norway
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+

|
|
244
|
+
|
|
245
|
+
Even though you see two separate letters (🇳🇴) in the terminal, when you copy and paste them into a browser or application like Twitter, they combine to display the Norwegian flag 🇳🇴.
|
|
246
|
+
|
|
247
|
+

|
|
248
|
+
|
|
249
|
+
#### ZWJ Sequence Example
|
|
250
|
+
|
|
251
|
+
The same applies to Zero Width Joiner (ZWJ) sequences. Complex emoji like family groups or professional emoji might not render correctly in terminals:
|
|
252
|
+
|
|
253
|
+
```bash
|
|
254
|
+
uchr search "polar bear"
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
In a terminal without proper font support:
|
|
258
|
+
|
|
259
|
+

|
|
260
|
+
|
|
261
|
+
But when pasted in Twitter or other applications:
|
|
262
|
+
|
|
263
|
+

|
|
264
|
+
|
|
265
|
+
> **💡 Tip**: This is expected behavior. The Unicode data is correct, and the characters will work properly in applications that support modern emoji rendering.
|
|
266
|
+
|
|
267
|
+
### Pipe Operations
|
|
268
|
+
|
|
269
|
+
```bash
|
|
270
|
+
# Get just the character
|
|
271
|
+
uchr search ghost -f simple
|
|
272
|
+
|
|
273
|
+
# First match only
|
|
274
|
+
uchr search snow -1
|
|
275
|
+
|
|
276
|
+
# Custom format for scripting
|
|
277
|
+
uchr search ghost -D "," | cut -d',' -f1
|
|
278
|
+
```
|
|
279
|
+
|
|
280
|
+
### Complex Searches
|
|
281
|
+
|
|
282
|
+
```bash
|
|
283
|
+
# CJK characters with specific meanings
|
|
284
|
+
uchr search -d "dragon"
|
|
285
|
+
|
|
286
|
+
# Characters in multiple blocks
|
|
287
|
+
uchr search -b "Mathematical" | head -10
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
## 🏗 Data Sources
|
|
291
|
+
|
|
292
|
+
This tool uses official Unicode 15.0 data:
|
|
293
|
+
|
|
294
|
+
- [Unicode Character Database](https://www.unicode.org/Public/15.0.0/ucdxml/ucd.all.flat.zip)
|
|
295
|
+
- [Emoji Sequences](https://www.unicode.org/Public/emoji/15.0/emoji-sequences.txt)
|
|
296
|
+
- [Emoji ZWJ Sequences](https://www.unicode.org/Public/emoji/15.0/emoji-zwj-sequences.txt)
|
|
297
|
+
|
|
298
|
+
## 🤝 Contributing
|
|
299
|
+
|
|
300
|
+
1. Fork the repository
|
|
301
|
+
2. Create a feature branch (`git checkout -b feature/amazing-feature`)
|
|
302
|
+
3. Commit your changes (`git commit -m 'Add amazing feature'`)
|
|
303
|
+
4. Push to the branch (`git push origin feature/amazing-feature`)
|
|
304
|
+
5. Open a Pull Request
|
|
305
|
+
|
|
306
|
+
## 📄 License
|
|
307
|
+
|
|
308
|
+
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
|
309
|
+
|
|
310
|
+
## 🙏 Acknowledgments
|
|
311
|
+
|
|
312
|
+
- [Unicode Consortium](https://unicode.org/) for maintaining Unicode standards
|
|
313
|
+
- Contributors and users of this project
|
|
314
|
+
|
uchr-0.3.0/README.md
ADDED
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
# Unicode Tools
|
|
2
|
+
|
|
3
|
+
[](https://opensource.org/licenses/MIT)
|
|
4
|
+
[](https://www.python.org/downloads/)
|
|
5
|
+
[](https://unicode.org/)
|
|
6
|
+
|
|
7
|
+
A powerful command-line tool for searching and exploring Unicode characters, emoji sequences, and character properties.
|
|
8
|
+
|
|
9
|
+
## 🚀 Features
|
|
10
|
+
|
|
11
|
+
- **Search by name**: Find characters by their Unicode name
|
|
12
|
+
- **Search by code**: Look up characters by code point or range
|
|
13
|
+
- **Search by character**: Reverse lookup from character to details
|
|
14
|
+
- **Search by block**: Explore characters within Unicode blocks
|
|
15
|
+
- **Emoji support**: Full support for emoji sequences and ZWJ sequences
|
|
16
|
+
- **CJK details**: Enhanced descriptions for CJK characters using kDefinition
|
|
17
|
+
- **Flexible output**: Multiple output formats for different use cases
|
|
18
|
+
|
|
19
|
+
## 📖 Table of Contents
|
|
20
|
+
|
|
21
|
+
- [Installation](#installation)
|
|
22
|
+
- [Quick Start](#quick-start)
|
|
23
|
+
- [Usage Examples](#usage-examples)
|
|
24
|
+
- [Command Reference](#command-reference)
|
|
25
|
+
- [Database Management](#database-management)
|
|
26
|
+
- [Contributing](#contributing)
|
|
27
|
+
- [License](#license)
|
|
28
|
+
|
|
29
|
+
## 🛠 Installation
|
|
30
|
+
|
|
31
|
+
### Install from source
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
git clone https://github.com/mkyutani/unicode-tools.git
|
|
35
|
+
cd unicode-tools
|
|
36
|
+
pip install -e .
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
### Initialize database
|
|
40
|
+
|
|
41
|
+
Create the Unicode database (required for first use):
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
uchr db create
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
This downloads Unicode 15.0 data and creates a local SQLite database (~13MB) at:
|
|
48
|
+
- Linux/macOS: `~/.local/share/unicode-tools/unicode.db`
|
|
49
|
+
- Root users: Automatically chooses between system (`/var/lib/unicode-tools/`) or personal location
|
|
50
|
+
|
|
51
|
+
## ⚡ Quick Start
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
# Search for ghost-related characters
|
|
55
|
+
uchr search ghost
|
|
56
|
+
|
|
57
|
+
# Find characters in a code range
|
|
58
|
+
uchr search -c 1F47A-1F480
|
|
59
|
+
|
|
60
|
+
# Search by character
|
|
61
|
+
uchr search -x 👻
|
|
62
|
+
|
|
63
|
+
# Search within a Unicode block
|
|
64
|
+
uchr search -b "Emoticons"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## 📋 Usage Examples
|
|
68
|
+
|
|
69
|
+
### Search by Name
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
uchr search goblin
|
|
73
|
+
```
|
|
74
|
+
```
|
|
75
|
+
👺 1F47A JAPANESE GOBLIN
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### Search by Code Range
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
uchr search -c 1F479-1F47B
|
|
82
|
+
```
|
|
83
|
+
```
|
|
84
|
+
👹 1F479 JAPANESE OGRE
|
|
85
|
+
👺 1F47A JAPANESE GOBLIN
|
|
86
|
+
👻 1F47B GHOST
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
### Search by Character
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
uchr search -x 👻
|
|
93
|
+
```
|
|
94
|
+
```
|
|
95
|
+
👻 1F47B GHOST
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
### Search by Unicode Block
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
uchr search -b "Misc_Pictographs"
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
### Search with Details (CJK Characters)
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
uchr search -d "pray for happiness"
|
|
108
|
+
```
|
|
109
|
+
```
|
|
110
|
+
祝 795D CJK UNIFIED IDEOGRAPH-#; PRAY FOR HAPPINESS OR BLESSINGS
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Output Formatting
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
# Simple format (characters only)
|
|
117
|
+
uchr search ghost -f simple
|
|
118
|
+
👻
|
|
119
|
+
|
|
120
|
+
# UTF-8 format
|
|
121
|
+
uchr search ghost -f utf8
|
|
122
|
+
👻 F0 9F 91 BB GHOST
|
|
123
|
+
|
|
124
|
+
# Custom delimiter
|
|
125
|
+
uchr search ghost -D "|"
|
|
126
|
+
👻|1F47B|GHOST
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## 🔧 Command Reference
|
|
130
|
+
|
|
131
|
+
### uchr
|
|
132
|
+
|
|
133
|
+
Main command with subcommands for all Unicode operations.
|
|
134
|
+
|
|
135
|
+
#### uchr search
|
|
136
|
+
|
|
137
|
+
Search Unicode characters with various criteria.
|
|
138
|
+
|
|
139
|
+
| Option | Short | Description |
|
|
140
|
+
|--------|-------|-------------|
|
|
141
|
+
| `--name` | | Search by character name (default) |
|
|
142
|
+
| `--code` | `-c` | Search by code point or range |
|
|
143
|
+
| `--char` | `-x` | Search by character |
|
|
144
|
+
| `--block` | `-b` | Search by Unicode block |
|
|
145
|
+
| `--detail` | `-d` | Search in character details |
|
|
146
|
+
| `--strict` | `-s` | Exact match (case insensitive) |
|
|
147
|
+
| `--first` | `-1` | Show first result only |
|
|
148
|
+
| `--format` | `-f` | Output format: `utf8`, `simple` |
|
|
149
|
+
| `--delimiter` | `-D` | Custom delimiter (default: space) |
|
|
150
|
+
|
|
151
|
+
#### uchr db
|
|
152
|
+
|
|
153
|
+
Database management operations.
|
|
154
|
+
|
|
155
|
+
| Subcommand | Description |
|
|
156
|
+
|------------|-------------|
|
|
157
|
+
| `uchr db create` | Create/update Unicode database |
|
|
158
|
+
| `uchr db delete` | Remove Unicode database |
|
|
159
|
+
| `uchr db info` | Show database location |
|
|
160
|
+
|
|
161
|
+
## 💾 Database Management
|
|
162
|
+
|
|
163
|
+
### Create Database
|
|
164
|
+
```bash
|
|
165
|
+
uchr db create
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
### Check Database Location
|
|
169
|
+
```bash
|
|
170
|
+
uchr db info
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
### Remove Database
|
|
174
|
+
```bash
|
|
175
|
+
uchr db delete
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
### Environment Variables
|
|
179
|
+
|
|
180
|
+
- `UNICODE_DB_PATH`: Override default database location
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
export UNICODE_DB_PATH="/custom/path/unicode.db"
|
|
184
|
+
uchr db create
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
## 🌟 Advanced Examples
|
|
188
|
+
|
|
189
|
+
### Finding Emoji Sequences
|
|
190
|
+
|
|
191
|
+
```bash
|
|
192
|
+
# National flags
|
|
193
|
+
uchr search -b "RGI_Emoji_Flag_Sequence"
|
|
194
|
+
|
|
195
|
+
# Family emoji with ZWJ sequences
|
|
196
|
+
uchr search family
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
### Terminal Display vs. Browser/Application Support
|
|
200
|
+
|
|
201
|
+
Many terminals don't properly display complex emoji sequences, but the characters work correctly when copied to browsers or applications.
|
|
202
|
+
|
|
203
|
+
#### National Flag Example
|
|
204
|
+
|
|
205
|
+
When searching for flags, you might see separate letters in your terminal:
|
|
206
|
+
|
|
207
|
+
```bash
|
|
208
|
+
uchr search -b "RGI_Emoji_Flag_Sequence" | grep -i norway
|
|
209
|
+
```
|
|
210
|
+
```
|
|
211
|
+
🇳🇴 1F1F3 1F1F4 flag: Norway
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+

|
|
215
|
+
|
|
216
|
+
Even though you see two separate letters (🇳🇴) in the terminal, when you copy and paste them into a browser or application like Twitter, they combine to display the Norwegian flag 🇳🇴.
|
|
217
|
+
|
|
218
|
+

|
|
219
|
+
|
|
220
|
+
#### ZWJ Sequence Example
|
|
221
|
+
|
|
222
|
+
The same applies to Zero Width Joiner (ZWJ) sequences. Complex emoji like family groups or professional emoji might not render correctly in terminals:
|
|
223
|
+
|
|
224
|
+
```bash
|
|
225
|
+
uchr search "polar bear"
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
In a terminal without proper font support:
|
|
229
|
+
|
|
230
|
+

|
|
231
|
+
|
|
232
|
+
But when pasted in Twitter or other applications:
|
|
233
|
+
|
|
234
|
+

|
|
235
|
+
|
|
236
|
+
> **💡 Tip**: This is expected behavior. The Unicode data is correct, and the characters will work properly in applications that support modern emoji rendering.
|
|
237
|
+
|
|
238
|
+
### Pipe Operations
|
|
239
|
+
|
|
240
|
+
```bash
|
|
241
|
+
# Get just the character
|
|
242
|
+
uchr search ghost -f simple
|
|
243
|
+
|
|
244
|
+
# First match only
|
|
245
|
+
uchr search snow -1
|
|
246
|
+
|
|
247
|
+
# Custom format for scripting
|
|
248
|
+
uchr search ghost -D "," | cut -d',' -f1
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
### Complex Searches
|
|
252
|
+
|
|
253
|
+
```bash
|
|
254
|
+
# CJK characters with specific meanings
|
|
255
|
+
uchr search -d "dragon"
|
|
256
|
+
|
|
257
|
+
# Characters in multiple blocks
|
|
258
|
+
uchr search -b "Mathematical" | head -10
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
## 🏗 Data Sources
|
|
262
|
+
|
|
263
|
+
This tool uses official Unicode 15.0 data:
|
|
264
|
+
|
|
265
|
+
- [Unicode Character Database](https://www.unicode.org/Public/15.0.0/ucdxml/ucd.all.flat.zip)
|
|
266
|
+
- [Emoji Sequences](https://www.unicode.org/Public/emoji/15.0/emoji-sequences.txt)
|
|
267
|
+
- [Emoji ZWJ Sequences](https://www.unicode.org/Public/emoji/15.0/emoji-zwj-sequences.txt)
|
|
268
|
+
|
|
269
|
+
## 🤝 Contributing
|
|
270
|
+
|
|
271
|
+
1. Fork the repository
|
|
272
|
+
2. Create a feature branch (`git checkout -b feature/amazing-feature`)
|
|
273
|
+
3. Commit your changes (`git commit -m 'Add amazing feature'`)
|
|
274
|
+
4. Push to the branch (`git push origin feature/amazing-feature`)
|
|
275
|
+
5. Open a Pull Request
|
|
276
|
+
|
|
277
|
+
## 📄 License
|
|
278
|
+
|
|
279
|
+
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
|
280
|
+
|
|
281
|
+
## 🙏 Acknowledgments
|
|
282
|
+
|
|
283
|
+
- [Unicode Consortium](https://unicode.org/) for maintaining Unicode standards
|
|
284
|
+
- Contributors and users of this project
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
[tool.poetry]
|
|
2
|
+
name = "uchr"
|
|
3
|
+
version = "0.3.0"
|
|
4
|
+
description = "Unicode character tools"
|
|
5
|
+
authors = ["Miki Yutani <mkyutani@gmail.com>"]
|
|
6
|
+
license = "MIT"
|
|
7
|
+
readme = "README.md"
|
|
8
|
+
homepage = "https://github.com/mkyutani/unicode-tools"
|
|
9
|
+
repository = "https://github.com/mkyutani/unicode-tools"
|
|
10
|
+
keywords = ["unicode", "character", "tools", "cli"]
|
|
11
|
+
classifiers = [
|
|
12
|
+
"Development Status :: 4 - Beta",
|
|
13
|
+
"Environment :: Console",
|
|
14
|
+
"Intended Audience :: Developers",
|
|
15
|
+
"License :: OSI Approved :: MIT License",
|
|
16
|
+
"Operating System :: OS Independent",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.8",
|
|
19
|
+
"Programming Language :: Python :: 3.9",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
24
|
+
"Topic :: Text Processing :: General",
|
|
25
|
+
"Topic :: Utilities",
|
|
26
|
+
]
|
|
27
|
+
packages = [{include = "unicode_tools", from = "src"}]
|
|
28
|
+
|
|
29
|
+
[tool.poetry.dependencies]
|
|
30
|
+
python = "^3.8"
|
|
31
|
+
requests = "^2.25.0"
|
|
32
|
+
|
|
33
|
+
[tool.poetry.group.dev.dependencies]
|
|
34
|
+
pytest = "^7.0.0"
|
|
35
|
+
black = "^23.0.0"
|
|
36
|
+
ruff = "^0.1.0"
|
|
37
|
+
|
|
38
|
+
[tool.poetry.scripts]
|
|
39
|
+
uchr = "unicode_tools.uchr:uchr"
|
|
40
|
+
|
|
41
|
+
[build-system]
|
|
42
|
+
requires = ["poetry-core"]
|
|
43
|
+
build-backend = "poetry.core.masonry.api"
|
|
44
|
+
|
|
45
|
+
[tool.black]
|
|
46
|
+
line-length = 88
|
|
47
|
+
target-version = ['py38']
|
|
48
|
+
|
|
49
|
+
[tool.ruff]
|
|
50
|
+
target-version = "py38"
|
|
51
|
+
line-length = 88
|
|
52
|
+
select = ["E", "F", "W", "I", "N", "UP", "B", "A", "C4", "T20"]
|
|
53
|
+
ignore = ["E501"] # Line too long (handled by black)
|
|
File without changes
|