vn-collate 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,32 @@
1
+ node_modules/
2
+ dist/
3
+ *.tsbuildinfo
4
+
5
+ # Python
6
+ __pycache__/
7
+ *.py[cod]
8
+ .venv/
9
+ venv/
10
+ *.egg-info/
11
+ .pytest_cache/
12
+ .mypy_cache/
13
+ .ruff_cache/
14
+
15
+ # Editors / OS
16
+ .DS_Store
17
+ Thumbs.db
18
+ .idea/
19
+ .vscode/
20
+
21
+ # Build output of the conformance tooling
22
+ .tmp/
23
+
24
+ # Traction snapshots. Machine-generated history for scripts/track-traction.mjs,
25
+ # not something to review in a diff.
26
+ .traction/
27
+
28
+ # LICENSE copies staged into each package at pack time by
29
+ # scripts/stage-license.mjs. One canonical file at the repository root; these
30
+ # are build output, and a staged one in a diff is a bug, not a licence update.
31
+ packages/*/LICENSE
32
+ packages/*/*/LICENSE
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 vn-toolkit contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,187 @@
1
+ Metadata-Version: 2.5
2
+ Name: vn-collate
3
+ Version: 0.1.0
4
+ Summary: Vietnamese collation sort keys, derived from ICU. Store the key in a binary column and ORDER BY it, because Postgres and MySQL have no Vietnamese collation. Standard library only.
5
+ Project-URL: Homepage, https://github.com/leeloc1809/vn-toolkit
6
+ Project-URL: Repository, https://github.com/leeloc1809/vn-toolkit
7
+ Project-URL: Issues, https://github.com/leeloc1809/vn-toolkit/issues
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: collate,collation,i18n,icu,mysql,postgres,sort,sort-key,unicode,vietnam,vietnamese
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Natural Language :: Vietnamese
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Text Processing :: Linguistic
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.9
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest>=8.0; extra == 'dev'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # vn-collate
29
+
30
+ Vietnamese collation sort keys, derived from ICU. **No runtime dependencies** —
31
+ standard library only.
32
+
33
+ ```bash
34
+ pip install vn-collate
35
+ ```
36
+
37
+ ```python
38
+ from vn_collate import collate_key, compare, sort
39
+
40
+ sort(['Đặng Minh', 'An Nguyễn', 'Bảo Châu', 'Lê Duẩn'])
41
+ # ['An Nguyễn', 'Bảo Châu', 'Đặng Minh', 'Lê Duẩn']
42
+
43
+ compare('Đặng', 'Dũng') # 1 — Đ follows D, it does not fall through to the end
44
+ compare('bao', 'Bảo') # -1 — tone order is huyền hỏi ngã sắc nặng
45
+ ```
46
+
47
+ ## Why
48
+
49
+ Postgres has no Vietnamese collation. MySQL's `utf8mb4_vietnamese_ci` exists,
50
+ but it is one of a fixed set of orders the server ships, and whichever one you
51
+ get depends on the server's version and configuration rather than on anything
52
+ you can pin. The result is that a Vietnamese `ORDER BY` is either
53
+ unreproducible between environments or not Vietnamese at all.
54
+
55
+ This library takes the order out of the server's hands. It builds a **sort
56
+ key** — a string whose byte order is the Vietnamese collation order — so you
57
+ store the key once and the database only has to do byte comparison, which
58
+ every database already does correctly.
59
+
60
+ ```sql
61
+ -- Postgres
62
+ CREATE TABLE people (
63
+ name text,
64
+ sortkey bytea -- or text COLLATE "C"
65
+ );
66
+ CREATE INDEX ON people (sortkey);
67
+
68
+ INSERT INTO people VALUES ('Đặng Minh', decode(:key, 'hex'));
69
+ SELECT name FROM people ORDER BY sortkey; -- correct Vietnamese order
70
+ ```
71
+
72
+ **The column has to be binary.** `ORDER BY key` on a `text` or `varchar` column
73
+ re-sorts the key using the column's own collation rules and produces nonsense.
74
+ This is the one way to use the library incorrectly, and it fails silently.
75
+
76
+ ## The order, and where it comes from
77
+
78
+ | | |
79
+ |---|---|
80
+ | letters | `a ă â b c d đ e ê f g h i j k l m n o ô ơ p q r s t u ư v w x y z` |
81
+ | tones | `none > huyền > hỏi > ngã > sắc > nặng` |
82
+ | digits | before every letter, `0`–`9` in numeric order |
83
+ | case | secondary to tone: `a A á Á` |
84
+ | unknown | after everything, ordered deterministically |
85
+
86
+ 33 letters, not 29: ICU's `vi-VN` locale still collates `w` and `z` as letters
87
+ in meaningful positions rather than dropping them in with the punctuation, so
88
+ they are kept where ICU puts them. That is what makes "sorts like ICU" true
89
+ rather than approximately true.
90
+
91
+ **Every value in that table was read out of `Intl.Collator('vi-VN')`, not
92
+ written from memory.** `scripts/generate-collation-conformance.mjs` probes the
93
+ runtime and emits both the table and the conformance suite, which records the
94
+ runtime it came from:
95
+
96
+ ```json
97
+ "derivedFrom": "Intl.Collator('vi-VN') on win32/x64, ICU 75.1, Unicode 15.1"
98
+ ```
99
+
100
+ If ICU ever changes the order, the next regeneration shows it as a diff landing
101
+ in review — instead of a silent behaviour change shipping to users.
102
+
103
+ ## Why a sort key and not a comparison function
104
+
105
+ A comparator answers one question. A sort key answers every question the
106
+ database will ever be asked, including the ones you did not anticipate: range
107
+ queries on names, exact lookup, `ORDER BY` on one or several key columns, and
108
+ an index on any of them.
109
+
110
+ A key is also the only formulation where a Python service and a Node service
111
+ agree. Both ports emit byte-identical keys, checked on every push by
112
+ `scripts/verify-collate-parity.mjs` across 445 strings — so a key written by one
113
+ service sorts correctly against data written by the other.
114
+
115
+ ## Known limitations
116
+
117
+ Stated precisely, because the boundaries are where this kind of library
118
+ usually lies.
119
+
120
+ **Punctuation and spaces sort after every letter, not before.** ICU sorts them
121
+ first. This only changes the order of records that differ *only* in
122
+ punctuation:
123
+
124
+ ```python
125
+ sort(['Nguyen Van An', 'NguyenVanAn']) # ours: NguyenVanAn, Nguyen Van An
126
+ # ICU: Nguyen Van An, NguyenVanAn
127
+ ```
128
+
129
+ A corpus that is internally consistent — every name punctuated the same way,
130
+ which is the normal case — orders identically to ICU. Verified: six real
131
+ multi-word Vietnamese names sort the same in both. A corpus that mixes
132
+ `Nguyễn Văn An` with `NguyễnVănAn` does not, and ICU's answer is the more
133
+ natural one. Handling this properly means ICU's variable weighting, which is a
134
+ configuration axis rather than a fixed order; that is v1 work.
135
+
136
+ **An unrecognised combining mark is dropped.** `ä` is `a` + U+0308, U+0308 is
137
+ not a Vietnamese tone or letter modifier, so it contributes no weight and `ä`
138
+ collates identically to `a`:
139
+
140
+ ```python
141
+ collate_key('a') == collate_key('ä') # True
142
+ ```
143
+
144
+ This never happens for Vietnamese text. It is a real key collision for text
145
+ that mixes in other languages, and it is deliberate — a guessed weight for an
146
+ unknown mark would be wrong in a way that is much harder to notice.
147
+
148
+ **Latin letters outside the Vietnamese set are in the fallback bucket.**
149
+ `ç`, `ñ`, `š`, `ž` and friends sort after every letter rather than beside their
150
+ base letter as ICU does. `ß` and `æ` are in the same bucket.
151
+
152
+ **No numeric collation.** `a2 > a10`, matching ICU's default.
153
+
154
+ ## Validation
155
+
156
+ 519 cases in
157
+ [`conformance/vn-collate-1.0.0.json`](../../conformance/vn-collate-1.0.0.json),
158
+ generated from ICU rather than hand-authored, every one carrying the note
159
+ explaining what breaks if you get it wrong. The TypeScript port reads the same
160
+ file.
161
+
162
+ The test file runs with or without pytest, because a library whose correctness
163
+ guarantee depends on a test runner is a library you cannot check on a machine
164
+ that has no test runner:
165
+
166
+ ```bash
167
+ python packages/vn-collate-py/tests/test_vn_collate.py # stdlib only
168
+ pytest packages/vn-collate-py/tests -q # if you have it
169
+ ```
170
+
171
+ ## API parity
172
+
173
+ ```python
174
+ collate_key(text) -> str
175
+ compare(a, b) -> int # -1, 0 or 1
176
+ sort(items, key=None) -> list # stable, does not mutate the input
177
+ ```
178
+
179
+ `collateKey` is exported as an alias so a call site moves between the Python and
180
+ TypeScript ports without renaming. The tables are exported as constants for
181
+ callers who want to inspect or extend them.
182
+
183
+ `compare` is defined in terms of `collate_key`, so the two can never disagree.
184
+
185
+ ## Licence
186
+
187
+ MIT
@@ -0,0 +1,160 @@
1
+ # vn-collate
2
+
3
+ Vietnamese collation sort keys, derived from ICU. **No runtime dependencies** —
4
+ standard library only.
5
+
6
+ ```bash
7
+ pip install vn-collate
8
+ ```
9
+
10
+ ```python
11
+ from vn_collate import collate_key, compare, sort
12
+
13
+ sort(['Đặng Minh', 'An Nguyễn', 'Bảo Châu', 'Lê Duẩn'])
14
+ # ['An Nguyễn', 'Bảo Châu', 'Đặng Minh', 'Lê Duẩn']
15
+
16
+ compare('Đặng', 'Dũng') # 1 — Đ follows D, it does not fall through to the end
17
+ compare('bao', 'Bảo') # -1 — tone order is huyền hỏi ngã sắc nặng
18
+ ```
19
+
20
+ ## Why
21
+
22
+ Postgres has no Vietnamese collation. MySQL's `utf8mb4_vietnamese_ci` exists,
23
+ but it is one of a fixed set of orders the server ships, and whichever one you
24
+ get depends on the server's version and configuration rather than on anything
25
+ you can pin. The result is that a Vietnamese `ORDER BY` is either
26
+ unreproducible between environments or not Vietnamese at all.
27
+
28
+ This library takes the order out of the server's hands. It builds a **sort
29
+ key** — a string whose byte order is the Vietnamese collation order — so you
30
+ store the key once and the database only has to do byte comparison, which
31
+ every database already does correctly.
32
+
33
+ ```sql
34
+ -- Postgres
35
+ CREATE TABLE people (
36
+ name text,
37
+ sortkey bytea -- or text COLLATE "C"
38
+ );
39
+ CREATE INDEX ON people (sortkey);
40
+
41
+ INSERT INTO people VALUES ('Đặng Minh', decode(:key, 'hex'));
42
+ SELECT name FROM people ORDER BY sortkey; -- correct Vietnamese order
43
+ ```
44
+
45
+ **The column has to be binary.** `ORDER BY key` on a `text` or `varchar` column
46
+ re-sorts the key using the column's own collation rules and produces nonsense.
47
+ This is the one way to use the library incorrectly, and it fails silently.
48
+
49
+ ## The order, and where it comes from
50
+
51
+ | | |
52
+ |---|---|
53
+ | letters | `a ă â b c d đ e ê f g h i j k l m n o ô ơ p q r s t u ư v w x y z` |
54
+ | tones | `none > huyền > hỏi > ngã > sắc > nặng` |
55
+ | digits | before every letter, `0`–`9` in numeric order |
56
+ | case | secondary to tone: `a A á Á` |
57
+ | unknown | after everything, ordered deterministically |
58
+
59
+ 33 letters, not 29: ICU's `vi-VN` locale still collates `w` and `z` as letters
60
+ in meaningful positions rather than dropping them in with the punctuation, so
61
+ they are kept where ICU puts them. That is what makes "sorts like ICU" true
62
+ rather than approximately true.
63
+
64
+ **Every value in that table was read out of `Intl.Collator('vi-VN')`, not
65
+ written from memory.** `scripts/generate-collation-conformance.mjs` probes the
66
+ runtime and emits both the table and the conformance suite, which records the
67
+ runtime it came from:
68
+
69
+ ```json
70
+ "derivedFrom": "Intl.Collator('vi-VN') on win32/x64, ICU 75.1, Unicode 15.1"
71
+ ```
72
+
73
+ If ICU ever changes the order, the next regeneration shows it as a diff landing
74
+ in review — instead of a silent behaviour change shipping to users.
75
+
76
+ ## Why a sort key and not a comparison function
77
+
78
+ A comparator answers one question. A sort key answers every question the
79
+ database will ever be asked, including the ones you did not anticipate: range
80
+ queries on names, exact lookup, `ORDER BY` on one or several key columns, and
81
+ an index on any of them.
82
+
83
+ A key is also the only formulation where a Python service and a Node service
84
+ agree. Both ports emit byte-identical keys, checked on every push by
85
+ `scripts/verify-collate-parity.mjs` across 445 strings — so a key written by one
86
+ service sorts correctly against data written by the other.
87
+
88
+ ## Known limitations
89
+
90
+ Stated precisely, because the boundaries are where this kind of library
91
+ usually lies.
92
+
93
+ **Punctuation and spaces sort after every letter, not before.** ICU sorts them
94
+ first. This only changes the order of records that differ *only* in
95
+ punctuation:
96
+
97
+ ```python
98
+ sort(['Nguyen Van An', 'NguyenVanAn']) # ours: NguyenVanAn, Nguyen Van An
99
+ # ICU: Nguyen Van An, NguyenVanAn
100
+ ```
101
+
102
+ A corpus that is internally consistent — every name punctuated the same way,
103
+ which is the normal case — orders identically to ICU. Verified: six real
104
+ multi-word Vietnamese names sort the same in both. A corpus that mixes
105
+ `Nguyễn Văn An` with `NguyễnVănAn` does not, and ICU's answer is the more
106
+ natural one. Handling this properly means ICU's variable weighting, which is a
107
+ configuration axis rather than a fixed order; that is v1 work.
108
+
109
+ **An unrecognised combining mark is dropped.** `ä` is `a` + U+0308, U+0308 is
110
+ not a Vietnamese tone or letter modifier, so it contributes no weight and `ä`
111
+ collates identically to `a`:
112
+
113
+ ```python
114
+ collate_key('a') == collate_key('ä') # True
115
+ ```
116
+
117
+ This never happens for Vietnamese text. It is a real key collision for text
118
+ that mixes in other languages, and it is deliberate — a guessed weight for an
119
+ unknown mark would be wrong in a way that is much harder to notice.
120
+
121
+ **Latin letters outside the Vietnamese set are in the fallback bucket.**
122
+ `ç`, `ñ`, `š`, `ž` and friends sort after every letter rather than beside their
123
+ base letter as ICU does. `ß` and `æ` are in the same bucket.
124
+
125
+ **No numeric collation.** `a2 > a10`, matching ICU's default.
126
+
127
+ ## Validation
128
+
129
+ 519 cases in
130
+ [`conformance/vn-collate-1.0.0.json`](../../conformance/vn-collate-1.0.0.json),
131
+ generated from ICU rather than hand-authored, every one carrying the note
132
+ explaining what breaks if you get it wrong. The TypeScript port reads the same
133
+ file.
134
+
135
+ The test file runs with or without pytest, because a library whose correctness
136
+ guarantee depends on a test runner is a library you cannot check on a machine
137
+ that has no test runner:
138
+
139
+ ```bash
140
+ python packages/vn-collate-py/tests/test_vn_collate.py # stdlib only
141
+ pytest packages/vn-collate-py/tests -q # if you have it
142
+ ```
143
+
144
+ ## API parity
145
+
146
+ ```python
147
+ collate_key(text) -> str
148
+ compare(a, b) -> int # -1, 0 or 1
149
+ sort(items, key=None) -> list # stable, does not mutate the input
150
+ ```
151
+
152
+ `collateKey` is exported as an alias so a call site moves between the Python and
153
+ TypeScript ports without renaming. The tables are exported as constants for
154
+ callers who want to inspect or extend them.
155
+
156
+ `compare` is defined in terms of `collate_key`, so the two can never disagree.
157
+
158
+ ## Licence
159
+
160
+ MIT
@@ -0,0 +1,61 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.24"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "vn-collate"
7
+ version = "0.1.0"
8
+ description = "Vietnamese collation sort keys, derived from ICU. Store the key in a binary column and ORDER BY it, because Postgres and MySQL have no Vietnamese collation. Standard library only."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ license-files = { paths = ["LICENSE"] }
13
+ keywords = [
14
+ "vietnamese",
15
+ "collation",
16
+ "collate",
17
+ "sort",
18
+ "sort-key",
19
+ "unicode",
20
+ "icu",
21
+ "postgres",
22
+ "mysql",
23
+ "i18n",
24
+ "vietnam",
25
+ ]
26
+ classifiers = [
27
+ "Development Status :: 3 - Alpha",
28
+ "Intended Audience :: Developers",
29
+ "License :: OSI Approved :: MIT License",
30
+ "Natural Language :: Vietnamese",
31
+ "Programming Language :: Python :: 3",
32
+ "Programming Language :: Python :: 3.9",
33
+ "Programming Language :: Python :: 3.10",
34
+ "Programming Language :: Python :: 3.11",
35
+ "Programming Language :: Python :: 3.12",
36
+ "Programming Language :: Python :: 3.13",
37
+ "Topic :: Text Processing :: Linguistic",
38
+ "Typing :: Typed",
39
+ ]
40
+
41
+ # Intentionally empty. A collation library that drags in dependencies is a
42
+ # dependency every consumer has to audit, and this one has to stay auditable
43
+ # itself.
44
+ dependencies = []
45
+
46
+ [project.optional-dependencies]
47
+ dev = ["pytest>=8.0"]
48
+
49
+ [project.urls]
50
+ Homepage = "https://github.com/leeloc1809/vn-toolkit"
51
+ Repository = "https://github.com/leeloc1809/vn-toolkit"
52
+ Issues = "https://github.com/leeloc1809/vn-toolkit/issues"
53
+
54
+ [tool.hatch.build.targets.wheel]
55
+ packages = ["src/vn_collate"]
56
+
57
+ [tool.hatch.build.targets.sdist]
58
+ include = ["src/vn_collate", "tests", "README.md", "LICENSE"]
59
+
60
+ [tool.pytest.ini_options]
61
+ testpaths = ["tests"]
@@ -0,0 +1,45 @@
1
+ """vn-collate — Vietnamese collation sort keys.
2
+
3
+ Standard library only. Mirrors the TypeScript port function for function, and
4
+ both are validated against the shared conformance suite in
5
+ ``conformance/vn-collate-1.0.0.json``.
6
+
7
+ >>> from vn_collate import collate_key, compare, sort
8
+ >>> sort(['Đặng', 'Anh', 'Bảo', 'bao'])
9
+ ['Anh', 'bao', 'Bảo', 'Đặng']
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from .core import (
15
+ DIGIT_PRIMARY_BASE,
16
+ FALLBACK_PRIMARY,
17
+ LETTER_MODIFIERS,
18
+ LETTER_PRIMARY_BASE,
19
+ PRIMARY_ORDER,
20
+ TONE_MARKS,
21
+ TONE_ORDER,
22
+ collateKey,
23
+ collate_key,
24
+ compare,
25
+ sort,
26
+ )
27
+ from .weights import TERMINATOR
28
+
29
+ __version__ = "0.1.0"
30
+
31
+ __all__ = [
32
+ "__version__",
33
+ "collate_key",
34
+ "compare",
35
+ "sort",
36
+ "collateKey",
37
+ "PRIMARY_ORDER",
38
+ "TONE_ORDER",
39
+ "LETTER_MODIFIERS",
40
+ "TONE_MARKS",
41
+ "TERMINATOR",
42
+ "FALLBACK_PRIMARY",
43
+ "DIGIT_PRIMARY_BASE",
44
+ "LETTER_PRIMARY_BASE",
45
+ ]
@@ -0,0 +1,187 @@
1
+ """Vietnamese collation: sort keys, comparison, sorting.
2
+
3
+ Standard library only, deliberately. This has to be trustworthy, and it has to
4
+ be the same answer in both ports, so it has as few moving parts as possible.
5
+
6
+ Every function is validated against ``conformance/vn-collate-1.0.0.json``,
7
+ which the TypeScript port also reads.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import unicodedata
13
+ from typing import Callable, Iterable, Sequence, TypeVar
14
+
15
+ from .weights import (
16
+ DIGITS,
17
+ DIGIT_PRIMARY_BASE,
18
+ FALLBACK_PRIMARY,
19
+ LETTER_MODIFIERS,
20
+ LETTER_PRIMARY_BASE,
21
+ PRIMARY_INDEX,
22
+ PRIMARY_ORDER,
23
+ TONE_MARKS,
24
+ TONE_ORDER,
25
+ TERMINATOR,
26
+ )
27
+
28
+ __all__ = [
29
+ "collate_key",
30
+ "compare",
31
+ "sort",
32
+ "collateKey",
33
+ "PRIMARY_ORDER",
34
+ "TONE_ORDER",
35
+ "LETTER_MODIFIERS",
36
+ "TONE_MARKS",
37
+ "FALLBACK_PRIMARY",
38
+ "DIGIT_PRIMARY_BASE",
39
+ "LETTER_PRIMARY_BASE",
40
+ ]
41
+
42
+ T = TypeVar("T")
43
+
44
+
45
+ def _clusters(text: str) -> list[str]:
46
+ """Split a string into one base character plus its combining marks.
47
+
48
+ This has to run on the NFD form. NFC does not guarantee that a combining
49
+ mark stays attached to its base -- ``B`` followed by U+0301 has no
50
+ precomposed form, so NFC leaves it as two code points, and walking that
51
+ string would treat the bare acute accent as a character of its own. NFD
52
+ always leaves marks trailing their base, so the grouping is unambiguous.
53
+ """
54
+ out: list[str] = []
55
+ for ch in unicodedata.normalize("NFD", text):
56
+ if out and unicodedata.combining(ch):
57
+ out[-1] += ch
58
+ else:
59
+ out.append(ch)
60
+ return out
61
+
62
+
63
+ def _weights_for(cluster: str) -> tuple[int, int, int]:
64
+ base = ""
65
+ modifiers = ""
66
+ tone = 0
67
+
68
+ for mark in cluster:
69
+ if mark in TONE_MARKS:
70
+ weight = TONE_MARKS[mark]
71
+ # Deterministic when a character carries more than one tone mark:
72
+ # the heaviest wins. Documented rather than accidental.
73
+ if weight > tone:
74
+ tone = weight
75
+ elif mark in LETTER_MODIFIERS:
76
+ modifiers += mark
77
+ elif not base:
78
+ base = mark
79
+
80
+ # Recompose the base letter from its modifiers so ă, â, ê, ô, ơ and ư find
81
+ # their own primary instead of collapsing onto a, e, o or u.
82
+ letter = unicodedata.normalize("NFC", base + modifiers)
83
+
84
+ # The table is lowercase. Case belongs on the tertiary level, so look the
85
+ # letter up folded and record the case separately -- otherwise every
86
+ # capital lands in the fallback bucket and sorts after all the lowercase.
87
+ folded = letter.lower()
88
+
89
+ primary = PRIMARY_INDEX.get(folded)
90
+ if primary is None:
91
+ digit = DIGITS.find(folded)
92
+ primary = DIGIT_PRIMARY_BASE + digit if digit >= 0 else FALLBACK_PRIMARY
93
+
94
+ return primary, tone, 1 if letter != folded else 0
95
+
96
+
97
+ def collate_key(text: str) -> str:
98
+ """Build a sort key for ``text``, as a hex string.
99
+
100
+ Why the key is emitted level by level
101
+ -------------------------------------
102
+ Interleaving the levels per character is the obvious encoding and it is
103
+ wrong. Comparing ``"ab"`` against ``"á"`` that way looks at the tone of the
104
+ second character before noticing that ``"á"`` has no second character, and
105
+ orders ``"ab"`` first. ICU orders ``"á"`` first, because a string that ends
106
+ earlier in the primary level sorts first. So the key is laid out primaries,
107
+ then secondaries, then tertiaries, each terminated -- the same shape the
108
+ Unicode Collation Algorithm uses.
109
+
110
+ Storing the key
111
+ ---------------
112
+ The key is **hex**, not raw bytes, so that both ports return the same type
113
+ and the value can live in a text column.
114
+
115
+ That does not mean you can sort it with the database's default collation.
116
+ ``ORDER BY key`` on a ``text``/``varchar`` column re-sorts the key with the
117
+ column's own rules and produces nonsense. It has to be a binary
118
+ comparison:
119
+
120
+ - Postgres: ``bytea``, or ``text COLLATE "C"``
121
+ - MySQL: ``varbinary``, or ``varchar`` with ``COLLATE utf8mb4_bin``
122
+
123
+ >>> collate_key('a') < collate_key('A')
124
+ True
125
+ >>> collate_key('Đặng') < collate_key('Dũng')
126
+ False
127
+ """
128
+ primary: list[int] = []
129
+ secondary: list[int] = []
130
+ tertiary: list[int] = []
131
+
132
+ for cluster in _clusters(text):
133
+ p, s, t = _weights_for(cluster)
134
+ primary.append(p)
135
+ secondary.append(s)
136
+ tertiary.append(t)
137
+
138
+ raw = (
139
+ primary + [TERMINATOR] + secondary + [TERMINATOR] + tertiary + [TERMINATOR]
140
+ )
141
+ return "".join(f"{b:02x}" for b in raw)
142
+
143
+
144
+ def compare(a: str, b: str) -> int:
145
+ """Compare two strings the way ICU's Vietnamese collation does.
146
+
147
+ Defined as a comparison of their sort keys, so ``compare`` and
148
+ ``collate_key`` can never disagree with each other. Returns -1, 0 or 1.
149
+
150
+ >>> compare('Đặng', 'Dũng')
151
+ 1
152
+ >>> compare('Thảo', 'Thao')
153
+ 1
154
+ >>> compare('bao', 'Bảo')
155
+ -1
156
+ """
157
+ ka = collate_key(a)
158
+ kb = collate_key(b)
159
+ if ka == kb:
160
+ return 0
161
+ # Hex digits compare in value order under Python code point order, so a
162
+ # plain string comparison is a byte comparison.
163
+ return -1 if ka < kb else 1
164
+
165
+
166
+ def sort(items: Iterable[T], key: Callable[[T], str] | None = None) -> list[T]:
167
+ """Sort a list of items by Vietnamese collation order.
168
+
169
+ Stable, and does not mutate the input. ``key`` defaults to treating each
170
+ item as a string, mirroring the TypeScript port.
171
+
172
+ >>> sort(['Đặng', 'Anh', 'Bảo', 'bao'])
173
+ ['Anh', 'bao', 'Bảo', 'Đặng']
174
+ """
175
+ key_fn: Callable[[T], str] = key if key is not None else (lambda x: x) # type: ignore[assignment,arg-type]
176
+ indexed = [
177
+ (collate_key(key_fn(item)), index, item)
178
+ for index, item in enumerate(items)
179
+ ]
180
+ # Sort on the key and the original index only. Including the item would ask
181
+ # Python to compare items of a type that may not be orderable.
182
+ indexed.sort(key=lambda entry: (entry[0], entry[1]))
183
+ return [item for _, _, item in indexed]
184
+
185
+
186
+ #: API-parity aliases, so call sites move between the ports without renaming.
187
+ collateKey = collate_key
File without changes
@@ -0,0 +1,72 @@
1
+ """Collation weights for Vietnamese, derived from ICU.
2
+
3
+ Do not hand-edit this table. Every value was read out of ICU by
4
+ ``scripts/generate-collation-conformance.mjs`` and pinned by the 519 cases in
5
+ ``conformance/vn-collate-1.0.0.json``. The suite records the exact runtime it
6
+ came from, so a future ICU change surfaces as a failing test rather than as a
7
+ silent behaviour change.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ #: Primary weights, in ICU order. 33 letters.
13
+ #:
14
+ #: The first 29 are the Vietnamese alphabet in its official order. The last four
15
+ #: are w, x, y, z: x and y are Vietnamese, but w and z are not -- they appear
16
+ #: because ICU's vi-VN locale still collates them as letters in meaningful
17
+ #: positions instead of dropping them into the fallback bucket alongside
18
+ #: punctuation. Keeping them where ICU puts them is what makes "sorts like ICU"
19
+ #: true rather than approximately true.
20
+ PRIMARY_ORDER: tuple[str, ...] = (
21
+ "a", "ă", "â", "b", "c", "d", "đ", "e", "ê",
22
+ "f", "g", "h", "i", "j", "k", "l", "m", "n",
23
+ "o", "ô", "ơ", "p", "q", "r", "s", "t", "u",
24
+ "ư", "v", "w", "x", "y", "z",
25
+ )
26
+
27
+ #: Tone weights. Uniform across every base letter, confirmed by probing all 33.
28
+ TONE_ORDER: tuple[str, ...] = ("none", "huyền", "hỏi", "ngã", "sắc", "nặng")
29
+
30
+
31
+ def cp(code_point: int) -> str:
32
+ """Return the character for a code point.
33
+
34
+ Combining marks are written as code points, never as literal characters.
35
+ They are invisible in an editor, and a stray reformat can silently corrupt
36
+ a lookup table that still compiles.
37
+ """
38
+ return chr(code_point)
39
+
40
+
41
+ #: Marks that modify the base letter rather than carrying tone.
42
+ LETTER_MODIFIERS: frozenset[str] = frozenset(
43
+ {cp(0x0302), cp(0x0306), cp(0x031B)} # circumflex, breve, horn
44
+ )
45
+
46
+ #: Marks that carry tone, mapped to their weight in TONE_ORDER.
47
+ TONE_MARKS: dict[str, int] = {
48
+ cp(0x0300): 1, # huyền
49
+ cp(0x0309): 2, # hỏi
50
+ cp(0x0303): 3, # ngã
51
+ cp(0x0301): 4, # sắc
52
+ cp(0x0323): 5, # nặng
53
+ }
54
+
55
+ #: Primary byte layout.
56
+ #:
57
+ #: 0x01..0x0A digits 0-9, which ICU sorts before every letter
58
+ #: 0x0B..0x2B the 33 letters above
59
+ #: 0xFE anything else
60
+ #:
61
+ #: 0x00 is reserved as the level terminator, so no weight may be zero.
62
+ DIGIT_PRIMARY_BASE = 0x01
63
+ LETTER_PRIMARY_BASE = 0x0B
64
+ FALLBACK_PRIMARY = 0xFE
65
+
66
+ DIGITS = "0123456789"
67
+
68
+ PRIMARY_INDEX: dict[str, int] = {
69
+ letter: LETTER_PRIMARY_BASE + i for i, letter in enumerate(PRIMARY_ORDER)
70
+ }
71
+
72
+ TERMINATOR = 0x00
@@ -0,0 +1,202 @@
1
+ """Run the shared collation conformance suite against the Python port.
2
+
3
+ Runnable two ways, on purpose:
4
+
5
+ pytest packages/vn-collate-py/tests -q
6
+ python packages/vn-collate-py/tests/test_vn_collate.py
7
+
8
+ The second form needs nothing installed. The conformance suite is the contract
9
+ between the two ports, and being able to check it with nothing but a Python
10
+ interpreter is what stops them drifting apart.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import sys
17
+ import unicodedata
18
+ from pathlib import Path
19
+ from typing import Any
20
+
21
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
22
+
23
+ from vn_collate import ( # noqa: E402 (path set up above)
24
+ PRIMARY_ORDER,
25
+ TONE_ORDER,
26
+ collate_key,
27
+ compare,
28
+ sort,
29
+ )
30
+
31
+ CONFORMANCE_PATH = (
32
+ Path(__file__).resolve().parents[3] / "conformance" / "vn-collate-1.0.0.json"
33
+ )
34
+
35
+
36
+ def load_suite() -> dict[str, Any]:
37
+ with CONFORMANCE_PATH.open(encoding="utf-8") as fh:
38
+ return json.load(fh)
39
+
40
+
41
+ SUITE = load_suite()
42
+ CASES = SUITE["cases"]
43
+
44
+
45
+ def test_schema_is_understood() -> None:
46
+ assert SUITE["schema"] == "vn-collate-conformance/1", SUITE["schema"]
47
+
48
+
49
+ def test_package_ships_a_py_typed_marker() -> None:
50
+ # The package declares "Typing :: Typed" on the PyPI page. Without this
51
+ # marker a type checker ignores every annotation in the module, so the claim
52
+ # is false and a consumer running mypy gets "untyped import" from a library
53
+ # that is annotated throughout.
54
+ import vn_collate
55
+
56
+ marker = Path(vn_collate.__file__).resolve().parent / "py.typed"
57
+ assert marker.exists(), (
58
+ f"{marker} is missing. The wheel must contain it, or drop the "
59
+ f"Typing :: Typed classifier instead of advertising types it hides."
60
+ )
61
+
62
+
63
+ def test_suite_records_its_provenance() -> None:
64
+ # If a future ICU disagrees, this string is how you find out what moved.
65
+ assert "ICU " in SUITE["derivedFrom"], SUITE["derivedFrom"]
66
+
67
+
68
+ def test_case_ids_are_unique() -> None:
69
+ ids = [c["id"] for c in CASES]
70
+ assert len(set(ids)) == len(ids), "duplicate case ids in conformance file"
71
+
72
+
73
+ def test_suite_pins_both_directions_and_reflexivity() -> None:
74
+ expected = {c["expected"] for c in CASES}
75
+ assert expected == {-1, 0, 1}, f"suite cannot reject a constant comparator: {expected}"
76
+
77
+
78
+ def test_table_matches_the_suite() -> None:
79
+ assert list(PRIMARY_ORDER) == SUITE["primaryOrder"]
80
+ assert list(TONE_ORDER) == SUITE["toneOrder"]
81
+
82
+
83
+ def test_conformance_suite() -> None:
84
+ failures: list[str] = []
85
+ for case in CASES:
86
+ actual = compare(case["a"], case["b"])
87
+ if actual != case["expected"]:
88
+ failures.append(
89
+ f"{case['id']} compare({case['a']!r}, {case['b']!r})\n"
90
+ f" expected: {case['expected']}\n"
91
+ f" actual: {actual}\n"
92
+ f" note: {case.get('note', '')}"
93
+ )
94
+ assert not failures, f"{len(failures)} conformance failures:\n" + "\n".join(failures[:10])
95
+
96
+
97
+ # --- Invariants ---------------------------------------------------------------
98
+
99
+
100
+ def test_compare_agrees_with_comparing_the_keys() -> None:
101
+ for a, b in [
102
+ ("a", "A"),
103
+ ("a", "á"),
104
+ ("Đặng", "Dũng"),
105
+ ("1", "An"),
106
+ ("ab", "á"),
107
+ ("", "a"),
108
+ ]:
109
+ ka, kb = collate_key(a), collate_key(b)
110
+ expected = 0 if ka == kb else (-1 if ka < kb else 1)
111
+ assert compare(a, b) == expected, f"{a!r} vs {b!r}"
112
+
113
+
114
+ def test_antisymmetric() -> None:
115
+ words = ["a", "A", "á", "Á", "à", "b", "đ", "Đ", "Thảo", "Thao", "1", ""]
116
+ for a in words:
117
+ for b in words:
118
+ assert compare(a, b) + compare(b, a) == 0, f"{a!r} vs {b!r}"
119
+
120
+
121
+ def test_prefix_sorts_before_its_extension() -> None:
122
+ # The case that breaks per-character interleaved keys.
123
+ assert compare("á", "ab") == -1
124
+ assert compare("a", "ab") == -1
125
+
126
+
127
+ def test_nfc_and_nfd_input_collate_identically() -> None:
128
+ # The last two have no precomposed form at all, so NFD is their only
129
+ # spelling and cluster splitting cannot be avoided. The combining mark is
130
+ # written as a code point because a literal one is invisible in an editor,
131
+ # and an invisible mark in a test about cluster splitting is precisely the
132
+ # bug the test exists to catch.
133
+ cases = [
134
+ "Đặng",
135
+ "Thảo",
136
+ "đường",
137
+ "Ăn",
138
+ "B" + chr(0x0301),
139
+ "b" + chr(0x0303) + chr(0x031B),
140
+ ]
141
+ for s in cases:
142
+ assert compare(s, unicodedata.normalize("NFD", s)) == 0, s
143
+ assert collate_key(s) == collate_key(unicodedata.normalize("NFD", s)), s
144
+
145
+
146
+ def test_digits_sort_before_every_letter() -> None:
147
+ for digit in "0123456789":
148
+ for letter in PRIMARY_ORDER:
149
+ assert compare(digit, letter) == -1, f"{digit} vs {letter}"
150
+
151
+
152
+ def test_unknown_characters_sort_last_deterministically() -> None:
153
+ assert compare("z", "☃") == -1
154
+ assert compare("☃", "☃") == 0
155
+
156
+
157
+ def test_key_is_hex_only() -> None:
158
+ # Hex only, so the key is safe in a text column with binary collation.
159
+ key = collate_key("Đặng Thảo 1")
160
+ assert all(c in "0123456789abcdef" for c in key), key
161
+
162
+
163
+ def test_sort_is_stable_and_does_not_mutate() -> None:
164
+ original = ["b", "B", "b", "B"]
165
+ snapshot = list(original)
166
+ out = sort(original)
167
+ assert original == snapshot, "sort mutated its input"
168
+ # b sorts before B, and the two of each keep their input order.
169
+ assert out == ["b", "b", "B", "B"], out
170
+ # An already-sorted list comes back unchanged.
171
+ already = ["a", "A", "á", "Á"]
172
+ assert sort(already) == already, already
173
+
174
+
175
+ def test_sort_accepts_a_key_function() -> None:
176
+ people = [{"n": "Đặng"}, {"n": "Anh"}, {"n": "Bảo"}]
177
+ assert [p["n"] for p in sort(people, key=lambda p: p["n"])] == ["Anh", "Bảo", "Đặng"]
178
+
179
+
180
+ def _run_standalone() -> int:
181
+ tests = [
182
+ (name, obj)
183
+ for name, obj in sorted(globals().items())
184
+ if name.startswith("test_") and callable(obj)
185
+ ]
186
+ failed = 0
187
+ for name, fn in tests:
188
+ try:
189
+ fn()
190
+ except AssertionError as exc:
191
+ failed += 1
192
+ print(f"FAIL {name}\n{exc}\n")
193
+ else:
194
+ print(f"ok {name}")
195
+ print()
196
+ print(f"{len(tests) - failed}/{len(tests)} test functions passed")
197
+ print(f"{len(CASES)} conformance cases across {len(SUITE['primaryOrder'])} letters")
198
+ return 1 if failed else 0
199
+
200
+
201
+ if __name__ == "__main__":
202
+ raise SystemExit(_run_standalone())