vn-collate 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vn_collate-0.1.0/.gitignore +32 -0
- vn_collate-0.1.0/LICENSE +21 -0
- vn_collate-0.1.0/PKG-INFO +187 -0
- vn_collate-0.1.0/README.md +160 -0
- vn_collate-0.1.0/pyproject.toml +61 -0
- vn_collate-0.1.0/src/vn_collate/__init__.py +45 -0
- vn_collate-0.1.0/src/vn_collate/core.py +187 -0
- vn_collate-0.1.0/src/vn_collate/py.typed +0 -0
- vn_collate-0.1.0/src/vn_collate/weights.py +72 -0
- vn_collate-0.1.0/tests/test_vn_collate.py +202 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
node_modules/
|
|
2
|
+
dist/
|
|
3
|
+
*.tsbuildinfo
|
|
4
|
+
|
|
5
|
+
# Python
|
|
6
|
+
__pycache__/
|
|
7
|
+
*.py[cod]
|
|
8
|
+
.venv/
|
|
9
|
+
venv/
|
|
10
|
+
*.egg-info/
|
|
11
|
+
.pytest_cache/
|
|
12
|
+
.mypy_cache/
|
|
13
|
+
.ruff_cache/
|
|
14
|
+
|
|
15
|
+
# Editors / OS
|
|
16
|
+
.DS_Store
|
|
17
|
+
Thumbs.db
|
|
18
|
+
.idea/
|
|
19
|
+
.vscode/
|
|
20
|
+
|
|
21
|
+
# Build output of the conformance tooling
|
|
22
|
+
.tmp/
|
|
23
|
+
|
|
24
|
+
# Traction snapshots. Machine-generated history for scripts/track-traction.mjs,
|
|
25
|
+
# not something to review in a diff.
|
|
26
|
+
.traction/
|
|
27
|
+
|
|
28
|
+
# LICENSE copies staged into each package at pack time by
|
|
29
|
+
# scripts/stage-license.mjs. One canonical file at the repository root; these
|
|
30
|
+
# are build output, and a staged one in a diff is a bug, not a licence update.
|
|
31
|
+
packages/*/LICENSE
|
|
32
|
+
packages/*/*/LICENSE
|
vn_collate-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 vn-toolkit contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: vn-collate
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Vietnamese collation sort keys, derived from ICU. Store the key in a binary column and ORDER BY it, because Postgres and MySQL have no Vietnamese collation. Standard library only.
|
|
5
|
+
Project-URL: Homepage, https://github.com/leeloc1809/vn-toolkit
|
|
6
|
+
Project-URL: Repository, https://github.com/leeloc1809/vn-toolkit
|
|
7
|
+
Project-URL: Issues, https://github.com/leeloc1809/vn-toolkit/issues
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: collate,collation,i18n,icu,mysql,postgres,sort,sort-key,unicode,vietnam,vietnamese
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Natural Language :: Vietnamese
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# vn-collate
|
|
29
|
+
|
|
30
|
+
Vietnamese collation sort keys, derived from ICU. **No runtime dependencies** —
|
|
31
|
+
standard library only.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install vn-collate
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from vn_collate import collate_key, compare, sort
|
|
39
|
+
|
|
40
|
+
sort(['Đặng Minh', 'An Nguyễn', 'Bảo Châu', 'Lê Duẩn'])
|
|
41
|
+
# ['An Nguyễn', 'Bảo Châu', 'Đặng Minh', 'Lê Duẩn']
|
|
42
|
+
|
|
43
|
+
compare('Đặng', 'Dũng') # 1 — Đ follows D, it does not fall through to the end
|
|
44
|
+
compare('bao', 'Bảo') # -1 — tone order is huyền hỏi ngã sắc nặng
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Why
|
|
48
|
+
|
|
49
|
+
Postgres has no Vietnamese collation. MySQL's `utf8mb4_vietnamese_ci` exists,
|
|
50
|
+
but it is one of a fixed set of orders the server ships, and whichever one you
|
|
51
|
+
get depends on the server's version and configuration rather than on anything
|
|
52
|
+
you can pin. The result is that a Vietnamese `ORDER BY` is either
|
|
53
|
+
unreproducible between environments or not Vietnamese at all.
|
|
54
|
+
|
|
55
|
+
This library takes the order out of the server's hands. It builds a **sort
|
|
56
|
+
key** — a string whose byte order is the Vietnamese collation order — so you
|
|
57
|
+
store the key once and the database only has to do byte comparison, which
|
|
58
|
+
every database already does correctly.
|
|
59
|
+
|
|
60
|
+
```sql
|
|
61
|
+
-- Postgres
|
|
62
|
+
CREATE TABLE people (
|
|
63
|
+
name text,
|
|
64
|
+
sortkey bytea -- or text COLLATE "C"
|
|
65
|
+
);
|
|
66
|
+
CREATE INDEX ON people (sortkey);
|
|
67
|
+
|
|
68
|
+
INSERT INTO people VALUES ('Đặng Minh', decode(:key, 'hex'));
|
|
69
|
+
SELECT name FROM people ORDER BY sortkey; -- correct Vietnamese order
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
**The column has to be binary.** `ORDER BY key` on a `text` or `varchar` column
|
|
73
|
+
re-sorts the key using the column's own collation rules and produces nonsense.
|
|
74
|
+
This is the one way to use the library incorrectly, and it fails silently.
|
|
75
|
+
|
|
76
|
+
## The order, and where it comes from
|
|
77
|
+
|
|
78
|
+
| | |
|
|
79
|
+
|---|---|
|
|
80
|
+
| letters | `a ă â b c d đ e ê f g h i j k l m n o ô ơ p q r s t u ư v w x y z` |
|
|
81
|
+
| tones | `none > huyền > hỏi > ngã > sắc > nặng` |
|
|
82
|
+
| digits | before every letter, `0`–`9` in numeric order |
|
|
83
|
+
| case | secondary to tone: `a A á Á` |
|
|
84
|
+
| unknown | after everything, ordered deterministically |
|
|
85
|
+
|
|
86
|
+
33 letters, not 29: ICU's `vi-VN` locale still collates `w` and `z` as letters
|
|
87
|
+
in meaningful positions rather than dropping them in with the punctuation, so
|
|
88
|
+
they are kept where ICU puts them. That is what makes "sorts like ICU" true
|
|
89
|
+
rather than approximately true.
|
|
90
|
+
|
|
91
|
+
**Every value in that table was read out of `Intl.Collator('vi-VN')`, not
|
|
92
|
+
written from memory.** `scripts/generate-collation-conformance.mjs` probes the
|
|
93
|
+
runtime and emits both the table and the conformance suite, which records the
|
|
94
|
+
runtime it came from:
|
|
95
|
+
|
|
96
|
+
```json
|
|
97
|
+
"derivedFrom": "Intl.Collator('vi-VN') on win32/x64, ICU 75.1, Unicode 15.1"
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
If ICU ever changes the order, the next regeneration shows it as a diff landing
|
|
101
|
+
in review — instead of a silent behaviour change shipping to users.
|
|
102
|
+
|
|
103
|
+
## Why a sort key and not a comparison function
|
|
104
|
+
|
|
105
|
+
A comparator answers one question. A sort key answers every question the
|
|
106
|
+
database will ever be asked, including the ones you did not anticipate: range
|
|
107
|
+
queries on names, exact lookup, `ORDER BY` on one or several key columns, and
|
|
108
|
+
an index on any of them.
|
|
109
|
+
|
|
110
|
+
A key is also the only formulation where a Python service and a Node service
|
|
111
|
+
agree. Both ports emit byte-identical keys, checked on every push by
|
|
112
|
+
`scripts/verify-collate-parity.mjs` across 445 strings — so a key written by one
|
|
113
|
+
service sorts correctly against data written by the other.
|
|
114
|
+
|
|
115
|
+
## Known limitations
|
|
116
|
+
|
|
117
|
+
Stated precisely, because the boundaries are where this kind of library
|
|
118
|
+
usually lies.
|
|
119
|
+
|
|
120
|
+
**Punctuation and spaces sort after every letter, not before.** ICU sorts them
|
|
121
|
+
first. This only changes the order of records that differ *only* in
|
|
122
|
+
punctuation:
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
sort(['Nguyen Van An', 'NguyenVanAn']) # ours: NguyenVanAn, Nguyen Van An
|
|
126
|
+
# ICU: Nguyen Van An, NguyenVanAn
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
A corpus that is internally consistent — every name punctuated the same way,
|
|
130
|
+
which is the normal case — orders identically to ICU. Verified: six real
|
|
131
|
+
multi-word Vietnamese names sort the same in both. A corpus that mixes
|
|
132
|
+
`Nguyễn Văn An` with `NguyễnVănAn` does not, and ICU's answer is the more
|
|
133
|
+
natural one. Handling this properly means ICU's variable weighting, which is a
|
|
134
|
+
configuration axis rather than a fixed order; that is v1 work.
|
|
135
|
+
|
|
136
|
+
**An unrecognised combining mark is dropped.** `ä` is `a` + U+0308, U+0308 is
|
|
137
|
+
not a Vietnamese tone or letter modifier, so it contributes no weight and `ä`
|
|
138
|
+
collates identically to `a`:
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
collate_key('a') == collate_key('ä') # True
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
This never happens for Vietnamese text. It is a real key collision for text
|
|
145
|
+
that mixes in other languages, and it is deliberate — a guessed weight for an
|
|
146
|
+
unknown mark would be wrong in a way that is much harder to notice.
|
|
147
|
+
|
|
148
|
+
**Latin letters outside the Vietnamese set are in the fallback bucket.**
|
|
149
|
+
`ç`, `ñ`, `š`, `ž` and friends sort after every letter rather than beside their
|
|
150
|
+
base letter as ICU does. `ß` and `æ` are in the same bucket.
|
|
151
|
+
|
|
152
|
+
**No numeric collation.** `a2 > a10`, matching ICU's default.
|
|
153
|
+
|
|
154
|
+
## Validation
|
|
155
|
+
|
|
156
|
+
519 cases in
|
|
157
|
+
[`conformance/vn-collate-1.0.0.json`](../../conformance/vn-collate-1.0.0.json),
|
|
158
|
+
generated from ICU rather than hand-authored, every one carrying the note
|
|
159
|
+
explaining what breaks if you get it wrong. The TypeScript port reads the same
|
|
160
|
+
file.
|
|
161
|
+
|
|
162
|
+
The test file runs with or without pytest, because a library whose correctness
|
|
163
|
+
guarantee depends on a test runner is a library you cannot check on a machine
|
|
164
|
+
that has no test runner:
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
python packages/vn-collate-py/tests/test_vn_collate.py # stdlib only
|
|
168
|
+
pytest packages/vn-collate-py/tests -q # if you have it
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
## API parity
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
collate_key(text) -> str
|
|
175
|
+
compare(a, b) -> int # -1, 0 or 1
|
|
176
|
+
sort(items, key=None) -> list # stable, does not mutate the input
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
`collateKey` is exported as an alias so a call site moves between the Python and
|
|
180
|
+
TypeScript ports without renaming. The tables are exported as constants for
|
|
181
|
+
callers who want to inspect or extend them.
|
|
182
|
+
|
|
183
|
+
`compare` is defined in terms of `collate_key`, so the two can never disagree.
|
|
184
|
+
|
|
185
|
+
## Licence
|
|
186
|
+
|
|
187
|
+
MIT
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
# vn-collate
|
|
2
|
+
|
|
3
|
+
Vietnamese collation sort keys, derived from ICU. **No runtime dependencies** —
|
|
4
|
+
standard library only.
|
|
5
|
+
|
|
6
|
+
```bash
|
|
7
|
+
pip install vn-collate
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
from vn_collate import collate_key, compare, sort
|
|
12
|
+
|
|
13
|
+
sort(['Đặng Minh', 'An Nguyễn', 'Bảo Châu', 'Lê Duẩn'])
|
|
14
|
+
# ['An Nguyễn', 'Bảo Châu', 'Đặng Minh', 'Lê Duẩn']
|
|
15
|
+
|
|
16
|
+
compare('Đặng', 'Dũng') # 1 — Đ follows D, it does not fall through to the end
|
|
17
|
+
compare('bao', 'Bảo') # -1 — tone order is huyền hỏi ngã sắc nặng
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## Why
|
|
21
|
+
|
|
22
|
+
Postgres has no Vietnamese collation. MySQL's `utf8mb4_vietnamese_ci` exists,
|
|
23
|
+
but it is one of a fixed set of orders the server ships, and whichever one you
|
|
24
|
+
get depends on the server's version and configuration rather than on anything
|
|
25
|
+
you can pin. The result is that a Vietnamese `ORDER BY` is either
|
|
26
|
+
unreproducible between environments or not Vietnamese at all.
|
|
27
|
+
|
|
28
|
+
This library takes the order out of the server's hands. It builds a **sort
|
|
29
|
+
key** — a string whose byte order is the Vietnamese collation order — so you
|
|
30
|
+
store the key once and the database only has to do byte comparison, which
|
|
31
|
+
every database already does correctly.
|
|
32
|
+
|
|
33
|
+
```sql
|
|
34
|
+
-- Postgres
|
|
35
|
+
CREATE TABLE people (
|
|
36
|
+
name text,
|
|
37
|
+
sortkey bytea -- or text COLLATE "C"
|
|
38
|
+
);
|
|
39
|
+
CREATE INDEX ON people (sortkey);
|
|
40
|
+
|
|
41
|
+
INSERT INTO people VALUES ('Đặng Minh', decode(:key, 'hex'));
|
|
42
|
+
SELECT name FROM people ORDER BY sortkey; -- correct Vietnamese order
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
**The column has to be binary.** `ORDER BY key` on a `text` or `varchar` column
|
|
46
|
+
re-sorts the key using the column's own collation rules and produces nonsense.
|
|
47
|
+
This is the one way to use the library incorrectly, and it fails silently.
|
|
48
|
+
|
|
49
|
+
## The order, and where it comes from
|
|
50
|
+
|
|
51
|
+
| | |
|
|
52
|
+
|---|---|
|
|
53
|
+
| letters | `a ă â b c d đ e ê f g h i j k l m n o ô ơ p q r s t u ư v w x y z` |
|
|
54
|
+
| tones | `none > huyền > hỏi > ngã > sắc > nặng` |
|
|
55
|
+
| digits | before every letter, `0`–`9` in numeric order |
|
|
56
|
+
| case | secondary to tone: `a A á Á` |
|
|
57
|
+
| unknown | after everything, ordered deterministically |
|
|
58
|
+
|
|
59
|
+
33 letters, not 29: ICU's `vi-VN` locale still collates `w` and `z` as letters
|
|
60
|
+
in meaningful positions rather than dropping them in with the punctuation, so
|
|
61
|
+
they are kept where ICU puts them. That is what makes "sorts like ICU" true
|
|
62
|
+
rather than approximately true.
|
|
63
|
+
|
|
64
|
+
**Every value in that table was read out of `Intl.Collator('vi-VN')`, not
|
|
65
|
+
written from memory.** `scripts/generate-collation-conformance.mjs` probes the
|
|
66
|
+
runtime and emits both the table and the conformance suite, which records the
|
|
67
|
+
runtime it came from:
|
|
68
|
+
|
|
69
|
+
```json
|
|
70
|
+
"derivedFrom": "Intl.Collator('vi-VN') on win32/x64, ICU 75.1, Unicode 15.1"
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
If ICU ever changes the order, the next regeneration shows it as a diff landing
|
|
74
|
+
in review — instead of a silent behaviour change shipping to users.
|
|
75
|
+
|
|
76
|
+
## Why a sort key and not a comparison function
|
|
77
|
+
|
|
78
|
+
A comparator answers one question. A sort key answers every question the
|
|
79
|
+
database will ever be asked, including the ones you did not anticipate: range
|
|
80
|
+
queries on names, exact lookup, `ORDER BY` on one or several key columns, and
|
|
81
|
+
an index on any of them.
|
|
82
|
+
|
|
83
|
+
A key is also the only formulation where a Python service and a Node service
|
|
84
|
+
agree. Both ports emit byte-identical keys, checked on every push by
|
|
85
|
+
`scripts/verify-collate-parity.mjs` across 445 strings — so a key written by one
|
|
86
|
+
service sorts correctly against data written by the other.
|
|
87
|
+
|
|
88
|
+
## Known limitations
|
|
89
|
+
|
|
90
|
+
Stated precisely, because the boundaries are where this kind of library
|
|
91
|
+
usually lies.
|
|
92
|
+
|
|
93
|
+
**Punctuation and spaces sort after every letter, not before.** ICU sorts them
|
|
94
|
+
first. This only changes the order of records that differ *only* in
|
|
95
|
+
punctuation:
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
sort(['Nguyen Van An', 'NguyenVanAn']) # ours: NguyenVanAn, Nguyen Van An
|
|
99
|
+
# ICU: Nguyen Van An, NguyenVanAn
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
A corpus that is internally consistent — every name punctuated the same way,
|
|
103
|
+
which is the normal case — orders identically to ICU. Verified: six real
|
|
104
|
+
multi-word Vietnamese names sort the same in both. A corpus that mixes
|
|
105
|
+
`Nguyễn Văn An` with `NguyễnVănAn` does not, and ICU's answer is the more
|
|
106
|
+
natural one. Handling this properly means ICU's variable weighting, which is a
|
|
107
|
+
configuration axis rather than a fixed order; that is v1 work.
|
|
108
|
+
|
|
109
|
+
**An unrecognised combining mark is dropped.** `ä` is `a` + U+0308, U+0308 is
|
|
110
|
+
not a Vietnamese tone or letter modifier, so it contributes no weight and `ä`
|
|
111
|
+
collates identically to `a`:
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
collate_key('a') == collate_key('ä') # True
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
This never happens for Vietnamese text. It is a real key collision for text
|
|
118
|
+
that mixes in other languages, and it is deliberate — a guessed weight for an
|
|
119
|
+
unknown mark would be wrong in a way that is much harder to notice.
|
|
120
|
+
|
|
121
|
+
**Latin letters outside the Vietnamese set are in the fallback bucket.**
|
|
122
|
+
`ç`, `ñ`, `š`, `ž` and friends sort after every letter rather than beside their
|
|
123
|
+
base letter as ICU does. `ß` and `æ` are in the same bucket.
|
|
124
|
+
|
|
125
|
+
**No numeric collation.** `a2 > a10`, matching ICU's default.
|
|
126
|
+
|
|
127
|
+
## Validation
|
|
128
|
+
|
|
129
|
+
519 cases in
|
|
130
|
+
[`conformance/vn-collate-1.0.0.json`](../../conformance/vn-collate-1.0.0.json),
|
|
131
|
+
generated from ICU rather than hand-authored, every one carrying the note
|
|
132
|
+
explaining what breaks if you get it wrong. The TypeScript port reads the same
|
|
133
|
+
file.
|
|
134
|
+
|
|
135
|
+
The test file runs with or without pytest, because a library whose correctness
|
|
136
|
+
guarantee depends on a test runner is a library you cannot check on a machine
|
|
137
|
+
that has no test runner:
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
python packages/vn-collate-py/tests/test_vn_collate.py # stdlib only
|
|
141
|
+
pytest packages/vn-collate-py/tests -q # if you have it
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
## API parity
|
|
145
|
+
|
|
146
|
+
```python
|
|
147
|
+
collate_key(text) -> str
|
|
148
|
+
compare(a, b) -> int # -1, 0 or 1
|
|
149
|
+
sort(items, key=None) -> list # stable, does not mutate the input
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
`collateKey` is exported as an alias so a call site moves between the Python and
|
|
153
|
+
TypeScript ports without renaming. The tables are exported as constants for
|
|
154
|
+
callers who want to inspect or extend them.
|
|
155
|
+
|
|
156
|
+
`compare` is defined in terms of `collate_key`, so the two can never disagree.
|
|
157
|
+
|
|
158
|
+
## Licence
|
|
159
|
+
|
|
160
|
+
MIT
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.24"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "vn-collate"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Vietnamese collation sort keys, derived from ICU. Store the key in a binary column and ORDER BY it, because Postgres and MySQL have no Vietnamese collation. Standard library only."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
license-files = { paths = ["LICENSE"] }
|
|
13
|
+
keywords = [
|
|
14
|
+
"vietnamese",
|
|
15
|
+
"collation",
|
|
16
|
+
"collate",
|
|
17
|
+
"sort",
|
|
18
|
+
"sort-key",
|
|
19
|
+
"unicode",
|
|
20
|
+
"icu",
|
|
21
|
+
"postgres",
|
|
22
|
+
"mysql",
|
|
23
|
+
"i18n",
|
|
24
|
+
"vietnam",
|
|
25
|
+
]
|
|
26
|
+
classifiers = [
|
|
27
|
+
"Development Status :: 3 - Alpha",
|
|
28
|
+
"Intended Audience :: Developers",
|
|
29
|
+
"License :: OSI Approved :: MIT License",
|
|
30
|
+
"Natural Language :: Vietnamese",
|
|
31
|
+
"Programming Language :: Python :: 3",
|
|
32
|
+
"Programming Language :: Python :: 3.9",
|
|
33
|
+
"Programming Language :: Python :: 3.10",
|
|
34
|
+
"Programming Language :: Python :: 3.11",
|
|
35
|
+
"Programming Language :: Python :: 3.12",
|
|
36
|
+
"Programming Language :: Python :: 3.13",
|
|
37
|
+
"Topic :: Text Processing :: Linguistic",
|
|
38
|
+
"Typing :: Typed",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
# Intentionally empty. A collation library that drags in dependencies is a
|
|
42
|
+
# dependency every consumer has to audit, and this one has to stay auditable
|
|
43
|
+
# itself.
|
|
44
|
+
dependencies = []
|
|
45
|
+
|
|
46
|
+
[project.optional-dependencies]
|
|
47
|
+
dev = ["pytest>=8.0"]
|
|
48
|
+
|
|
49
|
+
[project.urls]
|
|
50
|
+
Homepage = "https://github.com/leeloc1809/vn-toolkit"
|
|
51
|
+
Repository = "https://github.com/leeloc1809/vn-toolkit"
|
|
52
|
+
Issues = "https://github.com/leeloc1809/vn-toolkit/issues"
|
|
53
|
+
|
|
54
|
+
[tool.hatch.build.targets.wheel]
|
|
55
|
+
packages = ["src/vn_collate"]
|
|
56
|
+
|
|
57
|
+
[tool.hatch.build.targets.sdist]
|
|
58
|
+
include = ["src/vn_collate", "tests", "README.md", "LICENSE"]
|
|
59
|
+
|
|
60
|
+
[tool.pytest.ini_options]
|
|
61
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""vn-collate — Vietnamese collation sort keys.
|
|
2
|
+
|
|
3
|
+
Standard library only. Mirrors the TypeScript port function for function, and
|
|
4
|
+
both are validated against the shared conformance suite in
|
|
5
|
+
``conformance/vn-collate-1.0.0.json``.
|
|
6
|
+
|
|
7
|
+
>>> from vn_collate import collate_key, compare, sort
|
|
8
|
+
>>> sort(['Đặng', 'Anh', 'Bảo', 'bao'])
|
|
9
|
+
['Anh', 'bao', 'Bảo', 'Đặng']
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from .core import (
|
|
15
|
+
DIGIT_PRIMARY_BASE,
|
|
16
|
+
FALLBACK_PRIMARY,
|
|
17
|
+
LETTER_MODIFIERS,
|
|
18
|
+
LETTER_PRIMARY_BASE,
|
|
19
|
+
PRIMARY_ORDER,
|
|
20
|
+
TONE_MARKS,
|
|
21
|
+
TONE_ORDER,
|
|
22
|
+
collateKey,
|
|
23
|
+
collate_key,
|
|
24
|
+
compare,
|
|
25
|
+
sort,
|
|
26
|
+
)
|
|
27
|
+
from .weights import TERMINATOR
|
|
28
|
+
|
|
29
|
+
__version__ = "0.1.0"
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"__version__",
|
|
33
|
+
"collate_key",
|
|
34
|
+
"compare",
|
|
35
|
+
"sort",
|
|
36
|
+
"collateKey",
|
|
37
|
+
"PRIMARY_ORDER",
|
|
38
|
+
"TONE_ORDER",
|
|
39
|
+
"LETTER_MODIFIERS",
|
|
40
|
+
"TONE_MARKS",
|
|
41
|
+
"TERMINATOR",
|
|
42
|
+
"FALLBACK_PRIMARY",
|
|
43
|
+
"DIGIT_PRIMARY_BASE",
|
|
44
|
+
"LETTER_PRIMARY_BASE",
|
|
45
|
+
]
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""Vietnamese collation: sort keys, comparison, sorting.
|
|
2
|
+
|
|
3
|
+
Standard library only, deliberately. This has to be trustworthy, and it has to
|
|
4
|
+
be the same answer in both ports, so it has as few moving parts as possible.
|
|
5
|
+
|
|
6
|
+
Every function is validated against ``conformance/vn-collate-1.0.0.json``,
|
|
7
|
+
which the TypeScript port also reads.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import unicodedata
|
|
13
|
+
from typing import Callable, Iterable, Sequence, TypeVar
|
|
14
|
+
|
|
15
|
+
from .weights import (
|
|
16
|
+
DIGITS,
|
|
17
|
+
DIGIT_PRIMARY_BASE,
|
|
18
|
+
FALLBACK_PRIMARY,
|
|
19
|
+
LETTER_MODIFIERS,
|
|
20
|
+
LETTER_PRIMARY_BASE,
|
|
21
|
+
PRIMARY_INDEX,
|
|
22
|
+
PRIMARY_ORDER,
|
|
23
|
+
TONE_MARKS,
|
|
24
|
+
TONE_ORDER,
|
|
25
|
+
TERMINATOR,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"collate_key",
|
|
30
|
+
"compare",
|
|
31
|
+
"sort",
|
|
32
|
+
"collateKey",
|
|
33
|
+
"PRIMARY_ORDER",
|
|
34
|
+
"TONE_ORDER",
|
|
35
|
+
"LETTER_MODIFIERS",
|
|
36
|
+
"TONE_MARKS",
|
|
37
|
+
"FALLBACK_PRIMARY",
|
|
38
|
+
"DIGIT_PRIMARY_BASE",
|
|
39
|
+
"LETTER_PRIMARY_BASE",
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
T = TypeVar("T")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _clusters(text: str) -> list[str]:
|
|
46
|
+
"""Split a string into one base character plus its combining marks.
|
|
47
|
+
|
|
48
|
+
This has to run on the NFD form. NFC does not guarantee that a combining
|
|
49
|
+
mark stays attached to its base -- ``B`` followed by U+0301 has no
|
|
50
|
+
precomposed form, so NFC leaves it as two code points, and walking that
|
|
51
|
+
string would treat the bare acute accent as a character of its own. NFD
|
|
52
|
+
always leaves marks trailing their base, so the grouping is unambiguous.
|
|
53
|
+
"""
|
|
54
|
+
out: list[str] = []
|
|
55
|
+
for ch in unicodedata.normalize("NFD", text):
|
|
56
|
+
if out and unicodedata.combining(ch):
|
|
57
|
+
out[-1] += ch
|
|
58
|
+
else:
|
|
59
|
+
out.append(ch)
|
|
60
|
+
return out
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _weights_for(cluster: str) -> tuple[int, int, int]:
|
|
64
|
+
base = ""
|
|
65
|
+
modifiers = ""
|
|
66
|
+
tone = 0
|
|
67
|
+
|
|
68
|
+
for mark in cluster:
|
|
69
|
+
if mark in TONE_MARKS:
|
|
70
|
+
weight = TONE_MARKS[mark]
|
|
71
|
+
# Deterministic when a character carries more than one tone mark:
|
|
72
|
+
# the heaviest wins. Documented rather than accidental.
|
|
73
|
+
if weight > tone:
|
|
74
|
+
tone = weight
|
|
75
|
+
elif mark in LETTER_MODIFIERS:
|
|
76
|
+
modifiers += mark
|
|
77
|
+
elif not base:
|
|
78
|
+
base = mark
|
|
79
|
+
|
|
80
|
+
# Recompose the base letter from its modifiers so ă, â, ê, ô, ơ and ư find
|
|
81
|
+
# their own primary instead of collapsing onto a, e, o or u.
|
|
82
|
+
letter = unicodedata.normalize("NFC", base + modifiers)
|
|
83
|
+
|
|
84
|
+
# The table is lowercase. Case belongs on the tertiary level, so look the
|
|
85
|
+
# letter up folded and record the case separately -- otherwise every
|
|
86
|
+
# capital lands in the fallback bucket and sorts after all the lowercase.
|
|
87
|
+
folded = letter.lower()
|
|
88
|
+
|
|
89
|
+
primary = PRIMARY_INDEX.get(folded)
|
|
90
|
+
if primary is None:
|
|
91
|
+
digit = DIGITS.find(folded)
|
|
92
|
+
primary = DIGIT_PRIMARY_BASE + digit if digit >= 0 else FALLBACK_PRIMARY
|
|
93
|
+
|
|
94
|
+
return primary, tone, 1 if letter != folded else 0
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def collate_key(text: str) -> str:
|
|
98
|
+
"""Build a sort key for ``text``, as a hex string.
|
|
99
|
+
|
|
100
|
+
Why the key is emitted level by level
|
|
101
|
+
-------------------------------------
|
|
102
|
+
Interleaving the levels per character is the obvious encoding and it is
|
|
103
|
+
wrong. Comparing ``"ab"`` against ``"á"`` that way looks at the tone of the
|
|
104
|
+
second character before noticing that ``"á"`` has no second character, and
|
|
105
|
+
orders ``"ab"`` first. ICU orders ``"á"`` first, because a string that ends
|
|
106
|
+
earlier in the primary level sorts first. So the key is laid out primaries,
|
|
107
|
+
then secondaries, then tertiaries, each terminated -- the same shape the
|
|
108
|
+
Unicode Collation Algorithm uses.
|
|
109
|
+
|
|
110
|
+
Storing the key
|
|
111
|
+
---------------
|
|
112
|
+
The key is **hex**, not raw bytes, so that both ports return the same type
|
|
113
|
+
and the value can live in a text column.
|
|
114
|
+
|
|
115
|
+
That does not mean you can sort it with the database's default collation.
|
|
116
|
+
``ORDER BY key`` on a ``text``/``varchar`` column re-sorts the key with the
|
|
117
|
+
column's own rules and produces nonsense. It has to be a binary
|
|
118
|
+
comparison:
|
|
119
|
+
|
|
120
|
+
- Postgres: ``bytea``, or ``text COLLATE "C"``
|
|
121
|
+
- MySQL: ``varbinary``, or ``varchar`` with ``COLLATE utf8mb4_bin``
|
|
122
|
+
|
|
123
|
+
>>> collate_key('a') < collate_key('A')
|
|
124
|
+
True
|
|
125
|
+
>>> collate_key('Đặng') < collate_key('Dũng')
|
|
126
|
+
False
|
|
127
|
+
"""
|
|
128
|
+
primary: list[int] = []
|
|
129
|
+
secondary: list[int] = []
|
|
130
|
+
tertiary: list[int] = []
|
|
131
|
+
|
|
132
|
+
for cluster in _clusters(text):
|
|
133
|
+
p, s, t = _weights_for(cluster)
|
|
134
|
+
primary.append(p)
|
|
135
|
+
secondary.append(s)
|
|
136
|
+
tertiary.append(t)
|
|
137
|
+
|
|
138
|
+
raw = (
|
|
139
|
+
primary + [TERMINATOR] + secondary + [TERMINATOR] + tertiary + [TERMINATOR]
|
|
140
|
+
)
|
|
141
|
+
return "".join(f"{b:02x}" for b in raw)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def compare(a: str, b: str) -> int:
|
|
145
|
+
"""Compare two strings the way ICU's Vietnamese collation does.
|
|
146
|
+
|
|
147
|
+
Defined as a comparison of their sort keys, so ``compare`` and
|
|
148
|
+
``collate_key`` can never disagree with each other. Returns -1, 0 or 1.
|
|
149
|
+
|
|
150
|
+
>>> compare('Đặng', 'Dũng')
|
|
151
|
+
1
|
|
152
|
+
>>> compare('Thảo', 'Thao')
|
|
153
|
+
1
|
|
154
|
+
>>> compare('bao', 'Bảo')
|
|
155
|
+
-1
|
|
156
|
+
"""
|
|
157
|
+
ka = collate_key(a)
|
|
158
|
+
kb = collate_key(b)
|
|
159
|
+
if ka == kb:
|
|
160
|
+
return 0
|
|
161
|
+
# Hex digits compare in value order under Python code point order, so a
|
|
162
|
+
# plain string comparison is a byte comparison.
|
|
163
|
+
return -1 if ka < kb else 1
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def sort(items: Iterable[T], key: Callable[[T], str] | None = None) -> list[T]:
|
|
167
|
+
"""Sort a list of items by Vietnamese collation order.
|
|
168
|
+
|
|
169
|
+
Stable, and does not mutate the input. ``key`` defaults to treating each
|
|
170
|
+
item as a string, mirroring the TypeScript port.
|
|
171
|
+
|
|
172
|
+
>>> sort(['Đặng', 'Anh', 'Bảo', 'bao'])
|
|
173
|
+
['Anh', 'bao', 'Bảo', 'Đặng']
|
|
174
|
+
"""
|
|
175
|
+
key_fn: Callable[[T], str] = key if key is not None else (lambda x: x) # type: ignore[assignment,arg-type]
|
|
176
|
+
indexed = [
|
|
177
|
+
(collate_key(key_fn(item)), index, item)
|
|
178
|
+
for index, item in enumerate(items)
|
|
179
|
+
]
|
|
180
|
+
# Sort on the key and the original index only. Including the item would ask
|
|
181
|
+
# Python to compare items of a type that may not be orderable.
|
|
182
|
+
indexed.sort(key=lambda entry: (entry[0], entry[1]))
|
|
183
|
+
return [item for _, _, item in indexed]
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
#: API-parity aliases, so call sites move between the ports without renaming.
|
|
187
|
+
collateKey = collate_key
|
|
File without changes
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Collation weights for Vietnamese, derived from ICU.
|
|
2
|
+
|
|
3
|
+
Do not hand-edit this table. Every value was read out of ICU by
|
|
4
|
+
``scripts/generate-collation-conformance.mjs`` and pinned by the 519 cases in
|
|
5
|
+
``conformance/vn-collate-1.0.0.json``. The suite records the exact runtime it
|
|
6
|
+
came from, so a future ICU change surfaces as a failing test rather than as a
|
|
7
|
+
silent behaviour change.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
#: Primary weights, in ICU order. 33 letters.
|
|
13
|
+
#:
|
|
14
|
+
#: The first 29 are the Vietnamese alphabet in its official order. The last four
|
|
15
|
+
#: are w, x, y, z: x and y are Vietnamese, but w and z are not -- they appear
|
|
16
|
+
#: because ICU's vi-VN locale still collates them as letters in meaningful
|
|
17
|
+
#: positions instead of dropping them into the fallback bucket alongside
|
|
18
|
+
#: punctuation. Keeping them where ICU puts them is what makes "sorts like ICU"
|
|
19
|
+
#: true rather than approximately true.
|
|
20
|
+
PRIMARY_ORDER: tuple[str, ...] = (
|
|
21
|
+
"a", "ă", "â", "b", "c", "d", "đ", "e", "ê",
|
|
22
|
+
"f", "g", "h", "i", "j", "k", "l", "m", "n",
|
|
23
|
+
"o", "ô", "ơ", "p", "q", "r", "s", "t", "u",
|
|
24
|
+
"ư", "v", "w", "x", "y", "z",
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
#: Tone weights. Uniform across every base letter, confirmed by probing all 33.
|
|
28
|
+
TONE_ORDER: tuple[str, ...] = ("none", "huyền", "hỏi", "ngã", "sắc", "nặng")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def cp(code_point: int) -> str:
|
|
32
|
+
"""Return the character for a code point.
|
|
33
|
+
|
|
34
|
+
Combining marks are written as code points, never as literal characters.
|
|
35
|
+
They are invisible in an editor, and a stray reformat can silently corrupt
|
|
36
|
+
a lookup table that still compiles.
|
|
37
|
+
"""
|
|
38
|
+
return chr(code_point)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
#: Marks that modify the base letter rather than carrying tone.
|
|
42
|
+
LETTER_MODIFIERS: frozenset[str] = frozenset(
|
|
43
|
+
{cp(0x0302), cp(0x0306), cp(0x031B)} # circumflex, breve, horn
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
#: Marks that carry tone, mapped to their weight in TONE_ORDER.
|
|
47
|
+
TONE_MARKS: dict[str, int] = {
|
|
48
|
+
cp(0x0300): 1, # huyền
|
|
49
|
+
cp(0x0309): 2, # hỏi
|
|
50
|
+
cp(0x0303): 3, # ngã
|
|
51
|
+
cp(0x0301): 4, # sắc
|
|
52
|
+
cp(0x0323): 5, # nặng
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
#: Primary byte layout.
|
|
56
|
+
#:
|
|
57
|
+
#: 0x01..0x0A digits 0-9, which ICU sorts before every letter
|
|
58
|
+
#: 0x0B..0x2B the 33 letters above
|
|
59
|
+
#: 0xFE anything else
|
|
60
|
+
#:
|
|
61
|
+
#: 0x00 is reserved as the level terminator, so no weight may be zero.
|
|
62
|
+
DIGIT_PRIMARY_BASE = 0x01
|
|
63
|
+
LETTER_PRIMARY_BASE = 0x0B
|
|
64
|
+
FALLBACK_PRIMARY = 0xFE
|
|
65
|
+
|
|
66
|
+
DIGITS = "0123456789"
|
|
67
|
+
|
|
68
|
+
PRIMARY_INDEX: dict[str, int] = {
|
|
69
|
+
letter: LETTER_PRIMARY_BASE + i for i, letter in enumerate(PRIMARY_ORDER)
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
TERMINATOR = 0x00
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""Run the shared collation conformance suite against the Python port.
|
|
2
|
+
|
|
3
|
+
Runnable two ways, on purpose:
|
|
4
|
+
|
|
5
|
+
pytest packages/vn-collate-py/tests -q
|
|
6
|
+
python packages/vn-collate-py/tests/test_vn_collate.py
|
|
7
|
+
|
|
8
|
+
The second form needs nothing installed. The conformance suite is the contract
|
|
9
|
+
between the two ports, and being able to check it with nothing but a Python
|
|
10
|
+
interpreter is what stops them drifting apart.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import sys
|
|
17
|
+
import unicodedata
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
22
|
+
|
|
23
|
+
from vn_collate import ( # noqa: E402 (path set up above)
|
|
24
|
+
PRIMARY_ORDER,
|
|
25
|
+
TONE_ORDER,
|
|
26
|
+
collate_key,
|
|
27
|
+
compare,
|
|
28
|
+
sort,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
CONFORMANCE_PATH = (
|
|
32
|
+
Path(__file__).resolve().parents[3] / "conformance" / "vn-collate-1.0.0.json"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def load_suite() -> dict[str, Any]:
|
|
37
|
+
with CONFORMANCE_PATH.open(encoding="utf-8") as fh:
|
|
38
|
+
return json.load(fh)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
SUITE = load_suite()
|
|
42
|
+
CASES = SUITE["cases"]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_schema_is_understood() -> None:
|
|
46
|
+
assert SUITE["schema"] == "vn-collate-conformance/1", SUITE["schema"]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def test_package_ships_a_py_typed_marker() -> None:
|
|
50
|
+
# The package declares "Typing :: Typed" on the PyPI page. Without this
|
|
51
|
+
# marker a type checker ignores every annotation in the module, so the claim
|
|
52
|
+
# is false and a consumer running mypy gets "untyped import" from a library
|
|
53
|
+
# that is annotated throughout.
|
|
54
|
+
import vn_collate
|
|
55
|
+
|
|
56
|
+
marker = Path(vn_collate.__file__).resolve().parent / "py.typed"
|
|
57
|
+
assert marker.exists(), (
|
|
58
|
+
f"{marker} is missing. The wheel must contain it, or drop the "
|
|
59
|
+
f"Typing :: Typed classifier instead of advertising types it hides."
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def test_suite_records_its_provenance() -> None:
|
|
64
|
+
# If a future ICU disagrees, this string is how you find out what moved.
|
|
65
|
+
assert "ICU " in SUITE["derivedFrom"], SUITE["derivedFrom"]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_case_ids_are_unique() -> None:
|
|
69
|
+
ids = [c["id"] for c in CASES]
|
|
70
|
+
assert len(set(ids)) == len(ids), "duplicate case ids in conformance file"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def test_suite_pins_both_directions_and_reflexivity() -> None:
|
|
74
|
+
expected = {c["expected"] for c in CASES}
|
|
75
|
+
assert expected == {-1, 0, 1}, f"suite cannot reject a constant comparator: {expected}"
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_table_matches_the_suite() -> None:
|
|
79
|
+
assert list(PRIMARY_ORDER) == SUITE["primaryOrder"]
|
|
80
|
+
assert list(TONE_ORDER) == SUITE["toneOrder"]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_conformance_suite() -> None:
|
|
84
|
+
failures: list[str] = []
|
|
85
|
+
for case in CASES:
|
|
86
|
+
actual = compare(case["a"], case["b"])
|
|
87
|
+
if actual != case["expected"]:
|
|
88
|
+
failures.append(
|
|
89
|
+
f"{case['id']} compare({case['a']!r}, {case['b']!r})\n"
|
|
90
|
+
f" expected: {case['expected']}\n"
|
|
91
|
+
f" actual: {actual}\n"
|
|
92
|
+
f" note: {case.get('note', '')}"
|
|
93
|
+
)
|
|
94
|
+
assert not failures, f"{len(failures)} conformance failures:\n" + "\n".join(failures[:10])
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
# --- Invariants ---------------------------------------------------------------
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def test_compare_agrees_with_comparing_the_keys() -> None:
|
|
101
|
+
for a, b in [
|
|
102
|
+
("a", "A"),
|
|
103
|
+
("a", "á"),
|
|
104
|
+
("Đặng", "Dũng"),
|
|
105
|
+
("1", "An"),
|
|
106
|
+
("ab", "á"),
|
|
107
|
+
("", "a"),
|
|
108
|
+
]:
|
|
109
|
+
ka, kb = collate_key(a), collate_key(b)
|
|
110
|
+
expected = 0 if ka == kb else (-1 if ka < kb else 1)
|
|
111
|
+
assert compare(a, b) == expected, f"{a!r} vs {b!r}"
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def test_antisymmetric() -> None:
|
|
115
|
+
words = ["a", "A", "á", "Á", "à", "b", "đ", "Đ", "Thảo", "Thao", "1", ""]
|
|
116
|
+
for a in words:
|
|
117
|
+
for b in words:
|
|
118
|
+
assert compare(a, b) + compare(b, a) == 0, f"{a!r} vs {b!r}"
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_prefix_sorts_before_its_extension() -> None:
|
|
122
|
+
# The case that breaks per-character interleaved keys.
|
|
123
|
+
assert compare("á", "ab") == -1
|
|
124
|
+
assert compare("a", "ab") == -1
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def test_nfc_and_nfd_input_collate_identically() -> None:
|
|
128
|
+
# The last two have no precomposed form at all, so NFD is their only
|
|
129
|
+
# spelling and cluster splitting cannot be avoided. The combining mark is
|
|
130
|
+
# written as a code point because a literal one is invisible in an editor,
|
|
131
|
+
# and an invisible mark in a test about cluster splitting is precisely the
|
|
132
|
+
# bug the test exists to catch.
|
|
133
|
+
cases = [
|
|
134
|
+
"Đặng",
|
|
135
|
+
"Thảo",
|
|
136
|
+
"đường",
|
|
137
|
+
"Ăn",
|
|
138
|
+
"B" + chr(0x0301),
|
|
139
|
+
"b" + chr(0x0303) + chr(0x031B),
|
|
140
|
+
]
|
|
141
|
+
for s in cases:
|
|
142
|
+
assert compare(s, unicodedata.normalize("NFD", s)) == 0, s
|
|
143
|
+
assert collate_key(s) == collate_key(unicodedata.normalize("NFD", s)), s
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def test_digits_sort_before_every_letter() -> None:
|
|
147
|
+
for digit in "0123456789":
|
|
148
|
+
for letter in PRIMARY_ORDER:
|
|
149
|
+
assert compare(digit, letter) == -1, f"{digit} vs {letter}"
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def test_unknown_characters_sort_last_deterministically() -> None:
|
|
153
|
+
assert compare("z", "☃") == -1
|
|
154
|
+
assert compare("☃", "☃") == 0
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def test_key_is_hex_only() -> None:
|
|
158
|
+
# Hex only, so the key is safe in a text column with binary collation.
|
|
159
|
+
key = collate_key("Đặng Thảo 1")
|
|
160
|
+
assert all(c in "0123456789abcdef" for c in key), key
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def test_sort_is_stable_and_does_not_mutate() -> None:
|
|
164
|
+
original = ["b", "B", "b", "B"]
|
|
165
|
+
snapshot = list(original)
|
|
166
|
+
out = sort(original)
|
|
167
|
+
assert original == snapshot, "sort mutated its input"
|
|
168
|
+
# b sorts before B, and the two of each keep their input order.
|
|
169
|
+
assert out == ["b", "b", "B", "B"], out
|
|
170
|
+
# An already-sorted list comes back unchanged.
|
|
171
|
+
already = ["a", "A", "á", "Á"]
|
|
172
|
+
assert sort(already) == already, already
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def test_sort_accepts_a_key_function() -> None:
|
|
176
|
+
people = [{"n": "Đặng"}, {"n": "Anh"}, {"n": "Bảo"}]
|
|
177
|
+
assert [p["n"] for p in sort(people, key=lambda p: p["n"])] == ["Anh", "Bảo", "Đặng"]
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _run_standalone() -> int:
|
|
181
|
+
tests = [
|
|
182
|
+
(name, obj)
|
|
183
|
+
for name, obj in sorted(globals().items())
|
|
184
|
+
if name.startswith("test_") and callable(obj)
|
|
185
|
+
]
|
|
186
|
+
failed = 0
|
|
187
|
+
for name, fn in tests:
|
|
188
|
+
try:
|
|
189
|
+
fn()
|
|
190
|
+
except AssertionError as exc:
|
|
191
|
+
failed += 1
|
|
192
|
+
print(f"FAIL {name}\n{exc}\n")
|
|
193
|
+
else:
|
|
194
|
+
print(f"ok {name}")
|
|
195
|
+
print()
|
|
196
|
+
print(f"{len(tests) - failed}/{len(tests)} test functions passed")
|
|
197
|
+
print(f"{len(CASES)} conformance cases across {len(SUITE['primaryOrder'])} letters")
|
|
198
|
+
return 1 if failed else 0
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
if __name__ == "__main__":
|
|
202
|
+
raise SystemExit(_run_standalone())
|