vn-text 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
vn_text/__init__.py ADDED
@@ -0,0 +1,57 @@
1
+ """vn-text — correct Unicode primitives for Vietnamese text.
2
+
3
+ Standard library only. Mirrors the TypeScript port function-for-function, and
4
+ both are validated against the shared conformance suite in
5
+ ``conformance/vn-text-1.0.0.json``.
6
+
7
+ >>> from vn_text import fold
8
+ >>> fold("Đặng Minh Anh")
9
+ 'dang minh anh'
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from ._unicode import (
15
+ COMBINING_MARKS,
16
+ D_WITH_STROKE_LOWER,
17
+ D_WITH_STROKE_UPPER,
18
+ ETH_LOWER,
19
+ ETH_UPPER,
20
+ TONE_MARKS,
21
+ VIETNAMESE_SPECIFIC,
22
+ )
23
+ from .core import (
24
+ decompose,
25
+ deaccent,
26
+ fold,
27
+ is_vietnamese,
28
+ isVietnamese,
29
+ normalize,
30
+ repair_mojibake,
31
+ repairMojibake,
32
+ strip_stroke,
33
+ stripStroke,
34
+ )
35
+
36
+ __version__ = "0.1.0"
37
+
38
+ __all__ = [
39
+ "__version__",
40
+ "normalize",
41
+ "decompose",
42
+ "deaccent",
43
+ "strip_stroke",
44
+ "repair_mojibake",
45
+ "fold",
46
+ "is_vietnamese",
47
+ "stripStroke",
48
+ "repairMojibake",
49
+ "isVietnamese",
50
+ "D_WITH_STROKE_UPPER",
51
+ "D_WITH_STROKE_LOWER",
52
+ "ETH_UPPER",
53
+ "ETH_LOWER",
54
+ "TONE_MARKS",
55
+ "COMBINING_MARKS",
56
+ "VIETNAMESE_SPECIFIC",
57
+ ]
vn_text/_unicode.py ADDED
@@ -0,0 +1,76 @@
1
+ """Unicode constants that matter for Vietnamese text handling.
2
+
3
+ The trap
4
+ --------
5
+
6
+ Four characters look like a capital or lowercase D with a stroke::
7
+
8
+ U+0110 LATIN CAPITAL LETTER D WITH STROKE -> Vietnamese
9
+ U+0111 LATIN SMALL LETTER D WITH STROKE -> Vietnamese
10
+ U+00D0 LATIN CAPITAL LETTER ETH -> Icelandic, Danish
11
+ U+00F0 LATIN SMALL LETTER ETH -> Icelandic, Danish
12
+
13
+ None has a canonical decomposition, and none has a compatibility decomposition
14
+ either -- verified against both NFD and NFKD. Unicode treats each as a letter in
15
+ its own right rather than a decorated ``D``.
16
+
17
+ Why that breaks search
18
+ ----------------------
19
+
20
+ The obvious implementation, normalise to NFD and strip the combining marks,
21
+ silently fails to fold Vietnamese ``Đ``::
22
+
23
+ 'Đặng Minh' -> 'Đang Minh'
24
+
25
+ The surviving ``Đ`` means the index key no longer matches anything a user can
26
+ type, because nobody types ``Đang`` to find ``Đặng``. Nothing throws. Search
27
+ just quietly returns fewer results than it should.
28
+
29
+ The obvious repair is worse. A blanket "fold any stroked D" rule also turns ETH
30
+ into ``D``, corrupting Icelandic and Danish names. Transliteration tables that
31
+ treat ``Đ`` as decoration hit exactly that.
32
+
33
+ So the two must be distinguished explicitly. That is what
34
+ :func:`vn_text.strip_stroke` does: U+0110 and U+0111, nothing else.
35
+
36
+ All code points below are written as :func:`chr` calls rather than as literal
37
+ characters or ``\\uXXXX`` escapes. Combining marks are invisible in an editor
38
+ and a stray reformat or copy-paste can silently corrupt a character class that
39
+ happens to still compile. Keeping this module pure ASCII makes that class of
40
+ accident impossible, and keeps diffs readable.
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ import re
46
+
47
+ #: LATIN CAPITAL LETTER D WITH STROKE (Vietnamese).
48
+ D_WITH_STROKE_UPPER = chr(0x0110)
49
+
50
+ #: LATIN SMALL LETTER D WITH STROKE (Vietnamese).
51
+ D_WITH_STROKE_LOWER = chr(0x0111)
52
+
53
+ #: LATIN CAPITAL LETTER ETH (Icelandic, Danish). Never fold to D.
54
+ ETH_UPPER = chr(0x00D0)
55
+
56
+ #: LATIN SMALL LETTER ETH (Icelandic, Danish). Never fold to d.
57
+ ETH_LOWER = chr(0x00F0)
58
+
59
+ #: The five Vietnamese tone marks as combining marks, in NFD form.
60
+ #: sắc, huyền, hỏi, ngã, nặng.
61
+ TONE_MARKS: tuple[str, ...] = (
62
+ chr(0x0301), # sắc - COMBINING ACUTE ACCENT
63
+ chr(0x0300), # huyền - COMBINING GRAVE ACCENT
64
+ chr(0x0309), # hỏi - COMBINING HOOK ABOVE
65
+ chr(0x0303), # ngã - COMBINING TILDE
66
+ chr(0x0323), # nặng - COMBINING DOT BELOW
67
+ )
68
+
69
+ #: Full Combining Diacritical Marks block, U+0300 to U+036F.
70
+ COMBINING_MARKS = re.compile(f"[{chr(0x0300)}-{chr(0x036F)}]")
71
+
72
+ #: Vietnamese-specific letters: D WITH STROKE plus the precomposed block
73
+ #: U+1EA0 to U+1EF9 (a-circumflex-dot-below through y-tilde-dot-below).
74
+ VIETNAMESE_SPECIFIC = re.compile(
75
+ f"[{D_WITH_STROKE_UPPER}{D_WITH_STROKE_LOWER}{chr(0x1EA0)}-{chr(0x1EF9)}]"
76
+ )
vn_text/core.py ADDED
@@ -0,0 +1,206 @@
1
+ """Core Vietnamese text transforms.
2
+
3
+ Standard library only, on purpose. A text-normalisation library that drags in
4
+ a dependency is a dependency every consumer has to audit, and this one has to
5
+ stay auditable itself.
6
+
7
+ Every function here has a counterpart in the TypeScript port, and both ports
8
+ are validated against the same conformance file at
9
+ ``conformance/vn-text-1.0.0.json``.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import unicodedata
15
+
16
+ from ._unicode import (
17
+ COMBINING_MARKS,
18
+ D_WITH_STROKE_LOWER,
19
+ D_WITH_STROKE_UPPER,
20
+ ETH_LOWER,
21
+ ETH_UPPER,
22
+ VIETNAMESE_SPECIFIC,
23
+ )
24
+
25
+ __all__ = [
26
+ "normalize",
27
+ "decompose",
28
+ "deaccent",
29
+ "strip_stroke",
30
+ "repair_mojibake",
31
+ "fold",
32
+ "is_vietnamese",
33
+ # camelCase aliases for API parity with the TypeScript port.
34
+ "stripStroke",
35
+ "repairMojibake",
36
+ "isVietnamese",
37
+ ]
38
+
39
+
40
+ def normalize(text: str) -> str:
41
+ """Normalise to Unicode Normalization Form C (composed).
42
+
43
+ The form to use for storage, database keys, equality checks and
44
+ deduplication. Two strings that look identical to a user can differ at the
45
+ code-point level if one came from a Windows-1252 editor and the other from
46
+ a UTF-8 terminal; NFC collapses them.
47
+
48
+ >>> normalize(decompose("Cái gì thế này")) == "Cái gì thế này"
49
+ True
50
+ """
51
+ return unicodedata.normalize("NFC", text)
52
+
53
+
54
+ def decompose(text: str) -> str:
55
+ """Normalise to Unicode Normalization Form D (decomposed).
56
+
57
+ Vietnamese letters such as ``ế`` become a base ``e`` plus the combining
58
+ marks U+0302 (circumflex) and U+0301 (acute). Exposed because downstream
59
+ code that walks combining marks needs the decomposed form.
60
+ """
61
+ return unicodedata.normalize("NFD", text)
62
+
63
+
64
+ def deaccent(text: str) -> str:
65
+ """Remove every combining diacritical mark, leaving base letters intact.
66
+
67
+ Applies to all of Unicode, not just Vietnamese -- ``café`` becomes
68
+ ``cafe``, ``Ѐ`` becomes ``Е``.
69
+
70
+ What this function deliberately does NOT do:
71
+
72
+ It does not fold ``đ`` (U+0111) to ``d``. Vietnamese D-WITH-STROKE has no
73
+ canonical decomposition, so it survives NFD untouched and comes back out
74
+ as ``đ``. That is correct: stripping tone marks should not change which
75
+ letter you are looking at. Use :func:`strip_stroke` or :func:`fold` when
76
+ you need ``đ`` to become ``d``.
77
+
78
+ It also leaves ETH (U+00D0) alone, because ETH is a real letter in
79
+ Icelandic and Danish, not a decorated D.
80
+
81
+ >>> deaccent("Tiếng Việt")
82
+ 'Tieng Viet'
83
+ >>> deaccent("Đặng Minh")
84
+ 'Đang Minh'
85
+ >>> deaccent("Ðor")
86
+ 'Ðor'
87
+ """
88
+ return unicodedata.normalize(
89
+ "NFC", COMBINING_MARKS.sub("", unicodedata.normalize("NFD", text))
90
+ )
91
+
92
+
93
+ def strip_stroke(text: str) -> str:
94
+ """Replace Vietnamese D-WITH-STROKE with plain ``d`` / ``D``.
95
+
96
+ Nothing else is changed. In particular tone marks are left alone -- that is
97
+ :func:`deaccent`'s job -- and ETH (U+00D0, U+00F0) is preserved because it
98
+ is a different letter belonging to another script.
99
+
100
+ >>> strip_stroke("Đặng")
101
+ 'Dặng'
102
+ >>> strip_stroke("Ðor")
103
+ 'Ðor'
104
+ """
105
+ return text.replace(D_WITH_STROKE_UPPER, "D").replace(
106
+ D_WITH_STROKE_LOWER, "d"
107
+ )
108
+
109
+
110
+ def repair_mojibake(text: str) -> str:
111
+ """Repair the most common corruption in stored Vietnamese text.
112
+
113
+ Maps ``Ð`` (U+00D0) to ``Đ`` (U+0110) and ``ð`` (U+00F0) to ``đ`` (U+0111).
114
+
115
+ This is a **data repair** operation, not a search-key operation. It answers
116
+ a different question from :func:`deaccent` and :func:`fold`:
117
+
118
+ - :func:`fold` produces a search key and never reinterprets which letter you
119
+ have. It leaves ``Ð`` alone, because in Icelandic and Danish it is a real
120
+ letter called eth.
121
+ - ``repair_mojibake`` assumes the text *is* Vietnamese and that ``Ð`` is
122
+ damage. In Vietnamese text produced by legacy systems, ``Ð`` is
123
+ essentially always a mangled ``Đ``, and leaving it in place means a user
124
+ can never find the record again.
125
+
126
+ Compose them when the corpus is Vietnamese and may be damaged::
127
+
128
+ fold(repair_mojibake('Ðảm baỏ')) # 'dam bao' -> finds the record
129
+
130
+ What it deliberately does not do:
131
+
132
+ - **It is Vietnamese-biased by design.** ``repair_mojibake('Ðor')`` returns
133
+ ``'Đor'``, which is wrong for an Icelandic name. That is the trade, and it
134
+ is the right one for a Vietnamese library, but do not call it on text you
135
+ know contains Scandinavian or Icelandic content.
136
+ - **It does not recover byte-level mojibake.** Damage of the form ``Ä Ä¡``
137
+ comes from UTF-8 bytes decoded as Windows-1252, and undoing it needs the
138
+ original bytes, not a character mapping. There is a dedicated library for
139
+ that (``ftfy``); guessing at it from a Unicode string is not reliable
140
+ enough to ship.
141
+ - **It does not fix spelling or vowel composition.** ``lựơng`` stays
142
+ ``lựơng``. Those are separate problems, and ``VietnameseTextNormalizer``
143
+ and ``underthesea.text_normalize`` are the right tools for them.
144
+
145
+ >>> repair_mojibake('Ðảm baỏ')
146
+ 'Đảm bảo'
147
+ """
148
+ return text.replace(ETH_UPPER, D_WITH_STROKE_UPPER).replace(
149
+ ETH_LOWER, D_WITH_STROKE_LOWER
150
+ )
151
+
152
+
153
+ def fold(text: str) -> str:
154
+ """Produce a search key.
155
+
156
+ Accent-free, stroke-free, lower-cased, with every run of punctuation and
157
+ symbols collapsed to a single space. This is the canonical form to index
158
+ under for Vietnamese search: two strings a user would consider the same
159
+ hit produce the same key.
160
+
161
+ >>> fold("Cái gì thế này")
162
+ 'cai gi the nay'
163
+ >>> fold("Đặng Minh Anh")
164
+ 'dang minh anh'
165
+ >>> fold("Hà Nội, Việt Nam!")
166
+ 'ha noi viet nam'
167
+
168
+ Letters from other scripts are preserved rather than dropped, so
169
+ ``fold("東京 Tokyo")`` gives ``"東京 tokyo"`` -- the ideographs stay
170
+ searchable even though they carry no diacritics.
171
+ """
172
+ folded = strip_stroke(deaccent(text)).lower()
173
+ # Lower-casing can reintroduce combining marks, e.g. U+0130 -> i + U+0307.
174
+ folded = COMBINING_MARKS.sub("", unicodedata.normalize("NFD", folded))
175
+ # str.isalnum() is the closest stdlib equivalent of JavaScript's
176
+ # \p{L}\p{N}, covering the L* and N* general categories.
177
+ spaced = "".join(ch if ch.isalnum() else " " for ch in folded)
178
+ return " ".join(spaced.split())
179
+
180
+
181
+ def is_vietnamese(text: str) -> bool:
182
+ """True when the string contains at least one Vietnamese-specific character.
183
+
184
+ Useful for deciding whether a string needs Vietnamese-aware handling at
185
+ all.
186
+
187
+ This is a fast heuristic, not a language detector: the common Vietnamese
188
+ surnames ``Nguyen``, ``Tran`` and ``Le`` are pure ASCII and return False.
189
+
190
+ >>> is_vietnamese("Đặng")
191
+ True
192
+ >>> is_vietnamese("Nguyen")
193
+ False
194
+ """
195
+ return VIETNAMESE_SPECIFIC.search(text) is not None
196
+
197
+
198
+ #: API-parity alias, so code can move between the Python and TypeScript ports
199
+ #: without renaming every call site.
200
+ stripStroke = strip_stroke
201
+
202
+ #: API-parity alias, see :func:`stripStroke`.
203
+ repairMojibake = repair_mojibake
204
+
205
+ #: API-parity alias, see :func:`stripStroke`.
206
+ isVietnamese = is_vietnamese
vn_text/py.typed ADDED
File without changes
@@ -0,0 +1,98 @@
1
+ Metadata-Version: 2.5
2
+ Name: vn-text
3
+ Version: 0.1.0
4
+ Summary: Correct Unicode primitives for Vietnamese text: deaccent, fold, normalize, strip_stroke. Standard library only. Ships a cross-language conformance suite shared with the TypeScript port.
5
+ Project-URL: Homepage, https://github.com/leeloc1809/vn-toolkit
6
+ Project-URL: Repository, https://github.com/leeloc1809/vn-toolkit
7
+ Project-URL: Issues, https://github.com/leeloc1809/vn-toolkit/issues
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: diacritics,i18n,nfc,nfd,normalization,search,slugify,unicode,vietnam,vietnamese
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Natural Language :: Vietnamese
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Text Processing :: Linguistic
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.9
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest>=8.0; extra == 'dev'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # vn-text
29
+
30
+ Correct Unicode primitives for Vietnamese text. **No runtime dependencies** —
31
+ standard library only.
32
+
33
+ ```bash
34
+ pip install vn-text
35
+ ```
36
+
37
+ ```python
38
+ from vn_text import fold, deaccent, strip_stroke, is_vietnamese
39
+
40
+ fold('Đặng Minh Anh') # 'dang minh anh'
41
+ fold('Hà Nội, Việt Nam!') # 'ha noi viet nam'
42
+ deaccent('Tiếng Việt') # 'Tieng Viet'
43
+ strip_stroke('Đặng') # 'Dặng'
44
+ is_vietnamese('Nguyễn') # True
45
+ ```
46
+
47
+ ## Why
48
+
49
+ Vietnamese `Đ` (U+0110) has no canonical decomposition, so NFD cannot fold it.
50
+ The usual "NFD then strip combining marks" recipe leaves `Đặng Minh` as
51
+ `Đang Minh` — and nobody types `Đang` to find `Đặng`, so search quietly returns
52
+ too few results. The tempting blanket fix also turns the Icelandic letter `Ð`
53
+ (U+00D0) into `D` and corrupts those names.
54
+
55
+ `strip_stroke` handles U+0110 and U+0111 explicitly, and nothing else.
56
+
57
+ ## Validation
58
+
59
+ Every function is checked against the shared conformance suite at
60
+ [`conformance/vn-text-1.0.0.json`](../../conformance/vn-text-1.0.0.json) — 156
61
+ hand-authored cases that the TypeScript port also consumes, so the two
62
+ implementations cannot drift apart.
63
+
64
+ The test file runs with or without pytest, because a library whose correctness
65
+ guarantee depends on a test runner is a library you cannot check on a machine
66
+ that has no test runner:
67
+
68
+ ```bash
69
+ python packages/vn-text-py/tests/test_vn_text.py # stdlib only, nothing to install
70
+
71
+ pytest packages/vn-text-py/tests -q # if you have it
72
+ ```
73
+
74
+ ## API parity
75
+
76
+ `snake_case` is the Python convention, but `stripStroke` and `isVietnamese` are
77
+ exported as aliases so a call site can move between the Python and TypeScript
78
+ ports without renaming.
79
+
80
+ ## Known port risk
81
+
82
+ `fold()` classifies characters using `str.isalnum()`, while the TypeScript port
83
+ uses `\p{L}\p{N}`. These agree across all 127 distinct characters in the current
84
+ corpus, measured by `scripts/verify-port-parity.mjs`. They are not guaranteed to
85
+ agree for every code point; the conformance suite is how a divergence gets
86
+ caught.
87
+
88
+ ## Scope
89
+
90
+ Deliberately narrow. Tone-mark canonicalisation (`hóa` vs `hòa`) and fuzzy
91
+ matching for unaccented-keyboard input are not here. Sorting is not here
92
+ either -- it is [`vn-collate`](../vn-collate-py), which shares this
93
+ repository's conformance infrastructure. See the
94
+ [root README](../../README.md#what-this-is-not) for the reasoning.
95
+
96
+ ## Licence
97
+
98
+ MIT
@@ -0,0 +1,8 @@
1
+ vn_text/__init__.py,sha256=3Qg0FKvvCSb2ZONeigBog2dCkkltzexCip99AfTOUHI,1122
2
+ vn_text/_unicode.py,sha256=TuP33WgIWgbZBoc0MT_xChl3m8jXm-d5qb6K-uAm55I,2922
3
+ vn_text/core.py,sha256=TFvX66DTXF6T9unUbaL24YfJJsA6RdwcvIe3CzPdNHk,7225
4
+ vn_text/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ vn_text-0.1.0.dist-info/METADATA,sha256=B41YqxdVqVW4au5t0WboJ1IKwBKsw6R_f-lZIKPUwc4,3683
6
+ vn_text-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
7
+ vn_text-0.1.0.dist-info/licenses/LICENSE,sha256=jBDksFrsccC6mMMWfqpJtTCSLSe1WQXAJW0I1K99mB0,1080
8
+ vn_text-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 vn-toolkit contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.