vn-text 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,32 @@
1
+ node_modules/
2
+ dist/
3
+ *.tsbuildinfo
4
+
5
+ # Python
6
+ __pycache__/
7
+ *.py[cod]
8
+ .venv/
9
+ venv/
10
+ *.egg-info/
11
+ .pytest_cache/
12
+ .mypy_cache/
13
+ .ruff_cache/
14
+
15
+ # Editors / OS
16
+ .DS_Store
17
+ Thumbs.db
18
+ .idea/
19
+ .vscode/
20
+
21
+ # Build output of the conformance tooling
22
+ .tmp/
23
+
24
+ # Traction snapshots. Machine-generated history for scripts/track-traction.mjs,
25
+ # not something to review in a diff.
26
+ .traction/
27
+
28
+ # LICENSE copies staged into each package at pack time by
29
+ # scripts/stage-license.mjs. One canonical file at the repository root; these
30
+ # are build output, and a staged one in a diff is a bug, not a licence update.
31
+ packages/*/LICENSE
32
+ packages/*/*/LICENSE
vn_text-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 vn-toolkit contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
vn_text-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,98 @@
1
+ Metadata-Version: 2.5
2
+ Name: vn-text
3
+ Version: 0.1.0
4
+ Summary: Correct Unicode primitives for Vietnamese text: deaccent, fold, normalize, strip_stroke. Standard library only. Ships a cross-language conformance suite shared with the TypeScript port.
5
+ Project-URL: Homepage, https://github.com/leeloc1809/vn-toolkit
6
+ Project-URL: Repository, https://github.com/leeloc1809/vn-toolkit
7
+ Project-URL: Issues, https://github.com/leeloc1809/vn-toolkit/issues
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: diacritics,i18n,nfc,nfd,normalization,search,slugify,unicode,vietnam,vietnamese
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Natural Language :: Vietnamese
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Text Processing :: Linguistic
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.9
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest>=8.0; extra == 'dev'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # vn-text
29
+
30
+ Correct Unicode primitives for Vietnamese text. **No runtime dependencies** —
31
+ standard library only.
32
+
33
+ ```bash
34
+ pip install vn-text
35
+ ```
36
+
37
+ ```python
38
+ from vn_text import fold, deaccent, strip_stroke, is_vietnamese
39
+
40
+ fold('Đặng Minh Anh') # 'dang minh anh'
41
+ fold('Hà Nội, Việt Nam!') # 'ha noi viet nam'
42
+ deaccent('Tiếng Việt') # 'Tieng Viet'
43
+ strip_stroke('Đặng') # 'Dặng'
44
+ is_vietnamese('Nguyễn') # True
45
+ ```
46
+
47
+ ## Why
48
+
49
+ Vietnamese `Đ` (U+0110) has no canonical decomposition, so NFD cannot fold it.
50
+ The usual "NFD then strip combining marks" recipe leaves `Đặng Minh` as
51
+ `Đang Minh` — and nobody types `Đang` to find `Đặng`, so search quietly returns
52
+ too few results. The tempting blanket fix also turns the Icelandic letter `Ð`
53
+ (U+00D0) into `D` and corrupts those names.
54
+
55
+ `strip_stroke` handles U+0110 and U+0111 explicitly, and nothing else.
56
+
57
+ ## Validation
58
+
59
+ Every function is checked against the shared conformance suite at
60
+ [`conformance/vn-text-1.0.0.json`](../../conformance/vn-text-1.0.0.json) — 156
61
+ hand-authored cases that the TypeScript port also consumes, so the two
62
+ implementations cannot drift apart.
63
+
64
+ The test file runs with or without pytest, because a library whose correctness
65
+ guarantee depends on a test runner is a library you cannot check on a machine
66
+ that has no test runner:
67
+
68
+ ```bash
69
+ python packages/vn-text-py/tests/test_vn_text.py # stdlib only, nothing to install
70
+
71
+ pytest packages/vn-text-py/tests -q # if you have it
72
+ ```
73
+
74
+ ## API parity
75
+
76
+ `snake_case` is the Python convention, but `stripStroke` and `isVietnamese` are
77
+ exported as aliases so a call site can move between the Python and TypeScript
78
+ ports without renaming.
79
+
80
+ ## Known port risk
81
+
82
+ `fold()` classifies characters using `str.isalnum()`, while the TypeScript port
83
+ uses `\p{L}\p{N}`. These agree across all 127 distinct characters in the current
84
+ corpus, measured by `scripts/verify-port-parity.mjs`. They are not guaranteed to
85
+ agree for every code point; the conformance suite is how a divergence gets
86
+ caught.
87
+
88
+ ## Scope
89
+
90
+ Deliberately narrow. Tone-mark canonicalisation (`hóa` vs `hòa`) and fuzzy
91
+ matching for unaccented-keyboard input are not here. Sorting is not here
92
+ either -- it is [`vn-collate`](../vn-collate-py), which shares this
93
+ repository's conformance infrastructure. See the
94
+ [root README](../../README.md#what-this-is-not) for the reasoning.
95
+
96
+ ## Licence
97
+
98
+ MIT
@@ -0,0 +1,71 @@
1
+ # vn-text
2
+
3
+ Correct Unicode primitives for Vietnamese text. **No runtime dependencies** —
4
+ standard library only.
5
+
6
+ ```bash
7
+ pip install vn-text
8
+ ```
9
+
10
+ ```python
11
+ from vn_text import fold, deaccent, strip_stroke, is_vietnamese
12
+
13
+ fold('Đặng Minh Anh') # 'dang minh anh'
14
+ fold('Hà Nội, Việt Nam!') # 'ha noi viet nam'
15
+ deaccent('Tiếng Việt') # 'Tieng Viet'
16
+ strip_stroke('Đặng') # 'Dặng'
17
+ is_vietnamese('Nguyễn') # True
18
+ ```
19
+
20
+ ## Why
21
+
22
+ Vietnamese `Đ` (U+0110) has no canonical decomposition, so NFD cannot fold it.
23
+ The usual "NFD then strip combining marks" recipe leaves `Đặng Minh` as
24
+ `Đang Minh` — and nobody types `Đang` to find `Đặng`, so search quietly returns
25
+ too few results. The tempting blanket fix also turns the Icelandic letter `Ð`
26
+ (U+00D0) into `D` and corrupts those names.
27
+
28
+ `strip_stroke` handles U+0110 and U+0111 explicitly, and nothing else.
29
+
30
+ ## Validation
31
+
32
+ Every function is checked against the shared conformance suite at
33
+ [`conformance/vn-text-1.0.0.json`](../../conformance/vn-text-1.0.0.json) — 156
34
+ hand-authored cases that the TypeScript port also consumes, so the two
35
+ implementations cannot drift apart.
36
+
37
+ The test file runs with or without pytest, because a library whose correctness
38
+ guarantee depends on a test runner is a library you cannot check on a machine
39
+ that has no test runner:
40
+
41
+ ```bash
42
+ python packages/vn-text-py/tests/test_vn_text.py # stdlib only, nothing to install
43
+
44
+ pytest packages/vn-text-py/tests -q # if you have it
45
+ ```
46
+
47
+ ## API parity
48
+
49
+ `snake_case` is the Python convention, but `stripStroke` and `isVietnamese` are
50
+ exported as aliases so a call site can move between the Python and TypeScript
51
+ ports without renaming.
52
+
53
+ ## Known port risk
54
+
55
+ `fold()` classifies characters using `str.isalnum()`, while the TypeScript port
56
+ uses `\p{L}\p{N}`. These agree across all 127 distinct characters in the current
57
+ corpus, measured by `scripts/verify-port-parity.mjs`. They are not guaranteed to
58
+ agree for every code point; the conformance suite is how a divergence gets
59
+ caught.
60
+
61
+ ## Scope
62
+
63
+ Deliberately narrow. Tone-mark canonicalisation (`hóa` vs `hòa`) and fuzzy
64
+ matching for unaccented-keyboard input are not here. Sorting is not here
65
+ either -- it is [`vn-collate`](../vn-collate-py), which shares this
66
+ repository's conformance infrastructure. See the
67
+ [root README](../../README.md#what-this-is-not) for the reasoning.
68
+
69
+ ## Licence
70
+
71
+ MIT
@@ -0,0 +1,60 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.24"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "vn-text"
7
+ version = "0.1.0"
8
+ description = "Correct Unicode primitives for Vietnamese text: deaccent, fold, normalize, strip_stroke. Standard library only. Ships a cross-language conformance suite shared with the TypeScript port."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ license-files = { paths = ["LICENSE"] }
13
+ keywords = [
14
+ "vietnamese",
15
+ "unicode",
16
+ "normalization",
17
+ "search",
18
+ "nfc",
19
+ "nfd",
20
+ "diacritics",
21
+ "slugify",
22
+ "i18n",
23
+ "vietnam",
24
+ ]
25
+ classifiers = [
26
+ "Development Status :: 3 - Alpha",
27
+ "Intended Audience :: Developers",
28
+ "License :: OSI Approved :: MIT License",
29
+ "Natural Language :: Vietnamese",
30
+ "Programming Language :: Python :: 3",
31
+ "Programming Language :: Python :: 3.9",
32
+ "Programming Language :: Python :: 3.10",
33
+ "Programming Language :: Python :: 3.11",
34
+ "Programming Language :: Python :: 3.12",
35
+ "Programming Language :: Python :: 3.13",
36
+ "Topic :: Text Processing :: Linguistic",
37
+ "Typing :: Typed",
38
+ ]
39
+
40
+ # Intentionally empty. A text-normalisation library that drags in a
41
+ # dependency is a dependency every consumer has to audit, and this one has to
42
+ # stay auditable itself.
43
+ dependencies = []
44
+
45
+ [project.optional-dependencies]
46
+ dev = ["pytest>=8.0"]
47
+
48
+ [project.urls]
49
+ Homepage = "https://github.com/leeloc1809/vn-toolkit"
50
+ Repository = "https://github.com/leeloc1809/vn-toolkit"
51
+ Issues = "https://github.com/leeloc1809/vn-toolkit/issues"
52
+
53
+ [tool.hatch.build.targets.wheel]
54
+ packages = ["src/vn_text"]
55
+
56
+ [tool.hatch.build.targets.sdist]
57
+ include = ["src/vn_text", "tests", "README.md", "LICENSE"]
58
+
59
+ [tool.pytest.ini_options]
60
+ testpaths = ["tests"]
@@ -0,0 +1,57 @@
1
+ """vn-text — correct Unicode primitives for Vietnamese text.
2
+
3
+ Standard library only. Mirrors the TypeScript port function-for-function, and
4
+ both are validated against the shared conformance suite in
5
+ ``conformance/vn-text-1.0.0.json``.
6
+
7
+ >>> from vn_text import fold
8
+ >>> fold("Đặng Minh Anh")
9
+ 'dang minh anh'
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from ._unicode import (
15
+ COMBINING_MARKS,
16
+ D_WITH_STROKE_LOWER,
17
+ D_WITH_STROKE_UPPER,
18
+ ETH_LOWER,
19
+ ETH_UPPER,
20
+ TONE_MARKS,
21
+ VIETNAMESE_SPECIFIC,
22
+ )
23
+ from .core import (
24
+ decompose,
25
+ deaccent,
26
+ fold,
27
+ is_vietnamese,
28
+ isVietnamese,
29
+ normalize,
30
+ repair_mojibake,
31
+ repairMojibake,
32
+ strip_stroke,
33
+ stripStroke,
34
+ )
35
+
36
+ __version__ = "0.1.0"
37
+
38
+ __all__ = [
39
+ "__version__",
40
+ "normalize",
41
+ "decompose",
42
+ "deaccent",
43
+ "strip_stroke",
44
+ "repair_mojibake",
45
+ "fold",
46
+ "is_vietnamese",
47
+ "stripStroke",
48
+ "repairMojibake",
49
+ "isVietnamese",
50
+ "D_WITH_STROKE_UPPER",
51
+ "D_WITH_STROKE_LOWER",
52
+ "ETH_UPPER",
53
+ "ETH_LOWER",
54
+ "TONE_MARKS",
55
+ "COMBINING_MARKS",
56
+ "VIETNAMESE_SPECIFIC",
57
+ ]
@@ -0,0 +1,76 @@
1
+ """Unicode constants that matter for Vietnamese text handling.
2
+
3
+ The trap
4
+ --------
5
+
6
+ Four characters look like a capital or lowercase D with a stroke::
7
+
8
+ U+0110 LATIN CAPITAL LETTER D WITH STROKE -> Vietnamese
9
+ U+0111 LATIN SMALL LETTER D WITH STROKE -> Vietnamese
10
+ U+00D0 LATIN CAPITAL LETTER ETH -> Icelandic, Danish
11
+ U+00F0 LATIN SMALL LETTER ETH -> Icelandic, Danish
12
+
13
+ None has a canonical decomposition, and none has a compatibility decomposition
14
+ either -- verified against both NFD and NFKD. Unicode treats each as a letter in
15
+ its own right rather than a decorated ``D``.
16
+
17
+ Why that breaks search
18
+ ----------------------
19
+
20
+ The obvious implementation, normalise to NFD and strip the combining marks,
21
+ silently fails to fold Vietnamese ``Đ``::
22
+
23
+ 'Đặng Minh' -> 'Đang Minh'
24
+
25
+ The surviving ``Đ`` means the index key no longer matches anything a user can
26
+ type, because nobody types ``Đang`` to find ``Đặng``. Nothing throws. Search
27
+ just quietly returns fewer results than it should.
28
+
29
+ The obvious repair is worse. A blanket "fold any stroked D" rule also turns ETH
30
+ into ``D``, corrupting Icelandic and Danish names. Transliteration tables that
31
+ treat ``Đ`` as decoration hit exactly that.
32
+
33
+ So the two must be distinguished explicitly. That is what
34
+ :func:`vn_text.strip_stroke` does: U+0110 and U+0111, nothing else.
35
+
36
+ All code points below are written as :func:`chr` calls rather than as literal
37
+ characters or ``\\uXXXX`` escapes. Combining marks are invisible in an editor
38
+ and a stray reformat or copy-paste can silently corrupt a character class that
39
+ happens to still compile. Keeping this module pure ASCII makes that class of
40
+ accident impossible, and keeps diffs readable.
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ import re
46
+
47
+ #: LATIN CAPITAL LETTER D WITH STROKE (Vietnamese).
48
+ D_WITH_STROKE_UPPER = chr(0x0110)
49
+
50
+ #: LATIN SMALL LETTER D WITH STROKE (Vietnamese).
51
+ D_WITH_STROKE_LOWER = chr(0x0111)
52
+
53
+ #: LATIN CAPITAL LETTER ETH (Icelandic, Danish). Never fold to D.
54
+ ETH_UPPER = chr(0x00D0)
55
+
56
+ #: LATIN SMALL LETTER ETH (Icelandic, Danish). Never fold to d.
57
+ ETH_LOWER = chr(0x00F0)
58
+
59
+ #: The five Vietnamese tone marks as combining marks, in NFD form.
60
+ #: sắc, huyền, hỏi, ngã, nặng.
61
+ TONE_MARKS: tuple[str, ...] = (
62
+ chr(0x0301), # sắc - COMBINING ACUTE ACCENT
63
+ chr(0x0300), # huyền - COMBINING GRAVE ACCENT
64
+ chr(0x0309), # hỏi - COMBINING HOOK ABOVE
65
+ chr(0x0303), # ngã - COMBINING TILDE
66
+ chr(0x0323), # nặng - COMBINING DOT BELOW
67
+ )
68
+
69
+ #: Full Combining Diacritical Marks block, U+0300 to U+036F.
70
+ COMBINING_MARKS = re.compile(f"[{chr(0x0300)}-{chr(0x036F)}]")
71
+
72
+ #: Vietnamese-specific letters: D WITH STROKE plus the precomposed block
73
+ #: U+1EA0 to U+1EF9 (a-circumflex-dot-below through y-tilde-dot-below).
74
+ VIETNAMESE_SPECIFIC = re.compile(
75
+ f"[{D_WITH_STROKE_UPPER}{D_WITH_STROKE_LOWER}{chr(0x1EA0)}-{chr(0x1EF9)}]"
76
+ )
@@ -0,0 +1,206 @@
1
+ """Core Vietnamese text transforms.
2
+
3
+ Standard library only, on purpose. A text-normalisation library that drags in
4
+ a dependency is a dependency every consumer has to audit, and this one has to
5
+ stay auditable itself.
6
+
7
+ Every function here has a counterpart in the TypeScript port, and both ports
8
+ are validated against the same conformance file at
9
+ ``conformance/vn-text-1.0.0.json``.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import unicodedata
15
+
16
+ from ._unicode import (
17
+ COMBINING_MARKS,
18
+ D_WITH_STROKE_LOWER,
19
+ D_WITH_STROKE_UPPER,
20
+ ETH_LOWER,
21
+ ETH_UPPER,
22
+ VIETNAMESE_SPECIFIC,
23
+ )
24
+
25
+ __all__ = [
26
+ "normalize",
27
+ "decompose",
28
+ "deaccent",
29
+ "strip_stroke",
30
+ "repair_mojibake",
31
+ "fold",
32
+ "is_vietnamese",
33
+ # camelCase aliases for API parity with the TypeScript port.
34
+ "stripStroke",
35
+ "repairMojibake",
36
+ "isVietnamese",
37
+ ]
38
+
39
+
40
+ def normalize(text: str) -> str:
41
+ """Normalise to Unicode Normalization Form C (composed).
42
+
43
+ The form to use for storage, database keys, equality checks and
44
+ deduplication. Two strings that look identical to a user can differ at the
45
+ code-point level if one came from a Windows-1252 editor and the other from
46
+ a UTF-8 terminal; NFC collapses them.
47
+
48
+ >>> normalize(decompose("Cái gì thế này")) == "Cái gì thế này"
49
+ True
50
+ """
51
+ return unicodedata.normalize("NFC", text)
52
+
53
+
54
+ def decompose(text: str) -> str:
55
+ """Normalise to Unicode Normalization Form D (decomposed).
56
+
57
+ Vietnamese letters such as ``ế`` become a base ``e`` plus the combining
58
+ marks U+0302 (circumflex) and U+0301 (acute). Exposed because downstream
59
+ code that walks combining marks needs the decomposed form.
60
+ """
61
+ return unicodedata.normalize("NFD", text)
62
+
63
+
64
+ def deaccent(text: str) -> str:
65
+ """Remove every combining diacritical mark, leaving base letters intact.
66
+
67
+ Applies to all of Unicode, not just Vietnamese -- ``café`` becomes
68
+ ``cafe``, ``Ѐ`` becomes ``Е``.
69
+
70
+ What this function deliberately does NOT do:
71
+
72
+ It does not fold ``đ`` (U+0111) to ``d``. Vietnamese D-WITH-STROKE has no
73
+ canonical decomposition, so it survives NFD untouched and comes back out
74
+ as ``đ``. That is correct: stripping tone marks should not change which
75
+ letter you are looking at. Use :func:`strip_stroke` or :func:`fold` when
76
+ you need ``đ`` to become ``d``.
77
+
78
+ It also leaves ETH (U+00D0) alone, because ETH is a real letter in
79
+ Icelandic and Danish, not a decorated D.
80
+
81
+ >>> deaccent("Tiếng Việt")
82
+ 'Tieng Viet'
83
+ >>> deaccent("Đặng Minh")
84
+ 'Đang Minh'
85
+ >>> deaccent("Ðor")
86
+ 'Ðor'
87
+ """
88
+ return unicodedata.normalize(
89
+ "NFC", COMBINING_MARKS.sub("", unicodedata.normalize("NFD", text))
90
+ )
91
+
92
+
93
+ def strip_stroke(text: str) -> str:
94
+ """Replace Vietnamese D-WITH-STROKE with plain ``d`` / ``D``.
95
+
96
+ Nothing else is changed. In particular tone marks are left alone -- that is
97
+ :func:`deaccent`'s job -- and ETH (U+00D0, U+00F0) is preserved because it
98
+ is a different letter belonging to another script.
99
+
100
+ >>> strip_stroke("Đặng")
101
+ 'Dặng'
102
+ >>> strip_stroke("Ðor")
103
+ 'Ðor'
104
+ """
105
+ return text.replace(D_WITH_STROKE_UPPER, "D").replace(
106
+ D_WITH_STROKE_LOWER, "d"
107
+ )
108
+
109
+
110
+ def repair_mojibake(text: str) -> str:
111
+ """Repair the most common corruption in stored Vietnamese text.
112
+
113
+ Maps ``Ð`` (U+00D0) to ``Đ`` (U+0110) and ``ð`` (U+00F0) to ``đ`` (U+0111).
114
+
115
+ This is a **data repair** operation, not a search-key operation. It answers
116
+ a different question from :func:`deaccent` and :func:`fold`:
117
+
118
+ - :func:`fold` produces a search key and never reinterprets which letter you
119
+ have. It leaves ``Ð`` alone, because in Icelandic and Danish it is a real
120
+ letter called eth.
121
+ - ``repair_mojibake`` assumes the text *is* Vietnamese and that ``Ð`` is
122
+ damage. In Vietnamese text produced by legacy systems, ``Ð`` is
123
+ essentially always a mangled ``Đ``, and leaving it in place means a user
124
+ can never find the record again.
125
+
126
+ Compose them when the corpus is Vietnamese and may be damaged::
127
+
128
+ fold(repair_mojibake('Ðảm baỏ')) # 'dam bao' -> finds the record
129
+
130
+ What it deliberately does not do:
131
+
132
+ - **It is Vietnamese-biased by design.** ``repair_mojibake('Ðor')`` returns
133
+ ``'Đor'``, which is wrong for an Icelandic name. That is the trade, and it
134
+ is the right one for a Vietnamese library, but do not call it on text you
135
+ know contains Scandinavian or Icelandic content.
136
+ - **It does not recover byte-level mojibake.** Damage of the form ``Ä Ä¡``
137
+ comes from UTF-8 bytes decoded as Windows-1252, and undoing it needs the
138
+ original bytes, not a character mapping. There is a dedicated library for
139
+ that (``ftfy``); guessing at it from a Unicode string is not reliable
140
+ enough to ship.
141
+ - **It does not fix spelling or vowel composition.** ``lựơng`` stays
142
+ ``lựơng``. Those are separate problems, and ``VietnameseTextNormalizer``
143
+ and ``underthesea.text_normalize`` are the right tools for them.
144
+
145
+ >>> repair_mojibake('Ðảm baỏ')
146
+ 'Đảm bảo'
147
+ """
148
+ return text.replace(ETH_UPPER, D_WITH_STROKE_UPPER).replace(
149
+ ETH_LOWER, D_WITH_STROKE_LOWER
150
+ )
151
+
152
+
153
+ def fold(text: str) -> str:
154
+ """Produce a search key.
155
+
156
+ Accent-free, stroke-free, lower-cased, with every run of punctuation and
157
+ symbols collapsed to a single space. This is the canonical form to index
158
+ under for Vietnamese search: two strings a user would consider the same
159
+ hit produce the same key.
160
+
161
+ >>> fold("Cái gì thế này")
162
+ 'cai gi the nay'
163
+ >>> fold("Đặng Minh Anh")
164
+ 'dang minh anh'
165
+ >>> fold("Hà Nội, Việt Nam!")
166
+ 'ha noi viet nam'
167
+
168
+ Letters from other scripts are preserved rather than dropped, so
169
+ ``fold("東京 Tokyo")`` gives ``"東京 tokyo"`` -- the ideographs stay
170
+ searchable even though they carry no diacritics.
171
+ """
172
+ folded = strip_stroke(deaccent(text)).lower()
173
+ # Lower-casing can reintroduce combining marks, e.g. U+0130 -> i + U+0307.
174
+ folded = COMBINING_MARKS.sub("", unicodedata.normalize("NFD", folded))
175
+ # str.isalnum() is the closest stdlib equivalent of JavaScript's
176
+ # \p{L}\p{N}, covering the L* and N* general categories.
177
+ spaced = "".join(ch if ch.isalnum() else " " for ch in folded)
178
+ return " ".join(spaced.split())
179
+
180
+
181
+ def is_vietnamese(text: str) -> bool:
182
+ """True when the string contains at least one Vietnamese-specific character.
183
+
184
+ Useful for deciding whether a string needs Vietnamese-aware handling at
185
+ all.
186
+
187
+ This is a fast heuristic, not a language detector: the common Vietnamese
188
+ surnames ``Nguyen``, ``Tran`` and ``Le`` are pure ASCII and return False.
189
+
190
+ >>> is_vietnamese("Đặng")
191
+ True
192
+ >>> is_vietnamese("Nguyen")
193
+ False
194
+ """
195
+ return VIETNAMESE_SPECIFIC.search(text) is not None
196
+
197
+
198
+ #: API-parity alias, so code can move between the Python and TypeScript ports
199
+ #: without renaming every call site.
200
+ stripStroke = strip_stroke
201
+
202
+ #: API-parity alias, see :func:`stripStroke`.
203
+ repairMojibake = repair_mojibake
204
+
205
+ #: API-parity alias, see :func:`stripStroke`.
206
+ isVietnamese = is_vietnamese
File without changes
@@ -0,0 +1,227 @@
1
+ """Run the shared conformance suite against the Python port.
2
+
3
+ This file is deliberately runnable two ways:
4
+
5
+ pytest packages/py/tests -q # full runner, nice reporting
6
+ python packages/vn-text-py/tests/test_vn_text.py # stdlib only, no install
7
+
8
+ The second form exists because the conformance suite is the contract between
9
+ the two ports. Being able to check it with nothing but a Python interpreter
10
+ means the Python side can never silently drift from the TypeScript side.
11
+
12
+ Both ports read the exact same JSON file. There is no second copy to keep in
13
+ sync, which is the whole point.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import json
19
+ import sys
20
+ import unicodedata
21
+ from pathlib import Path
22
+ from typing import Any, Callable
23
+
24
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
25
+
26
+ from vn_text import ( # noqa: E402 (path set up above)
27
+ deaccent,
28
+ decompose,
29
+ fold,
30
+ is_vietnamese,
31
+ normalize,
32
+ repair_mojibake,
33
+ strip_stroke,
34
+ )
35
+
36
+ CONFORMANCE_PATH = (
37
+ Path(__file__).resolve().parents[3] / "conformance" / "vn-text-1.0.0.json"
38
+ )
39
+
40
+
41
+ def test_package_ships_a_py_typed_marker() -> None:
42
+ # The package declares "Typing :: Typed" on the PyPI page. Without this
43
+ # marker a type checker ignores every annotation in the module, so the claim
44
+ # is false and a consumer running mypy gets "untyped import" from a library
45
+ # that is annotated throughout.
46
+ import vn_text
47
+
48
+ marker = Path(vn_text.__file__).resolve().parent / "py.typed"
49
+ assert marker.exists(), (
50
+ f"{marker} is missing. The wheel must contain it, or drop the "
51
+ f"Typing :: Typed classifier instead of advertising types it hides."
52
+ )
53
+
54
+
55
+ def load_suite() -> dict[str, Any]:
56
+ with CONFORMANCE_PATH.open(encoding="utf-8") as fh:
57
+ return json.load(fh)
58
+
59
+
60
+ #: Dispatch table keyed by the function names used in the conformance file.
61
+ #: These are the TypeScript names, so the mapping stays one-to-one.
62
+ IMPLS: dict[str, Callable[[str], str | bool]] = {
63
+ "normalize": normalize,
64
+ "deaccent": deaccent,
65
+ "stripStroke": strip_stroke,
66
+ "repairMojibake": repair_mojibake,
67
+ "fold": fold,
68
+ "isVietnamese": is_vietnamese,
69
+ }
70
+
71
+ SUITE = load_suite()
72
+ CASES = SUITE["cases"]
73
+
74
+
75
+ # --------------------------------------------------------------------------
76
+ # Metadata checks -- fail loudly if the suite itself drifts out of shape.
77
+ # --------------------------------------------------------------------------
78
+
79
+
80
+ def test_schema_is_understood() -> None:
81
+ assert SUITE["schema"] == "vn-text-conformance/1", SUITE["schema"]
82
+
83
+
84
+ def test_no_case_references_an_unknown_function() -> None:
85
+ unknown = [f"{c['id']} -> {c['fn']}" for c in CASES if c["fn"] not in IMPLS]
86
+ assert unknown == [], f"unknown functions in conformance file: {unknown}"
87
+
88
+
89
+ def test_case_ids_are_unique() -> None:
90
+ ids = [c["id"] for c in CASES]
91
+ assert len(set(ids)) == len(ids), "duplicate case ids in conformance file"
92
+
93
+
94
+ def test_every_declared_function_has_cases() -> None:
95
+ for fn in SUITE["functions"]:
96
+ count = sum(1 for c in CASES if c["fn"] == fn)
97
+ assert count > 0, f'function "{fn}" has no conformance cases'
98
+
99
+
100
+ # --------------------------------------------------------------------------
101
+ # The suite itself.
102
+ # --------------------------------------------------------------------------
103
+
104
+
105
+ def test_conformance_suite() -> None:
106
+ failures: list[str] = []
107
+ for case in CASES:
108
+ impl = IMPLS[case["fn"]]
109
+ actual = impl(case["input"])
110
+ if actual != case["expected"]:
111
+ failures.append(
112
+ f"{case['id']} {case['fn']}({case['input']!r})\n"
113
+ f" expected: {case['expected']!r}\n"
114
+ f" actual: {actual!r}"
115
+ )
116
+ assert not failures, "conformance failures:\n" + "\n".join(failures)
117
+
118
+
119
+ # --------------------------------------------------------------------------
120
+ # Regression tests for the mistakes that make existing libraries wrong.
121
+ # Duplicated on purpose so the headline behaviour stays visible in the report.
122
+ # --------------------------------------------------------------------------
123
+
124
+
125
+ def test_d_with_stroke_folds_to_d() -> None:
126
+ assert fold("Đặng Minh") == "dang minh"
127
+
128
+
129
+ def test_eth_never_folds_to_d() -> None:
130
+ assert fold("Ðor") == "ðor"
131
+ assert deaccent("Ðor") == "Ðor"
132
+ assert strip_stroke("Ðor") == "Ðor"
133
+
134
+
135
+ def test_deaccent_preserves_letter_identity_but_fold_collapses_it() -> None:
136
+ # Stripping tone marks should not silently rewrite which letter you have.
137
+ assert deaccent("Đặng") == "Đang"
138
+ # A search key has to collapse it, or users cannot find "Đặng" by
139
+ # typing "Dang".
140
+ assert fold("Đặng") == "dang"
141
+
142
+
143
+ def test_search_equivalence() -> None:
144
+ assert fold("Cai gi the nay") == fold("Cái gì thế này")
145
+ assert fold("Dang") == fold("Đặng")
146
+ assert fold("HA NOI") == fold("Hà Nội")
147
+
148
+
149
+ def test_fold_is_idempotent() -> None:
150
+ for s in [
151
+ "Cái gì thế này",
152
+ "Đặng Minh",
153
+ "Hà Nội, Việt Nam!",
154
+ "Ðor",
155
+ "東京 Tokyo",
156
+ ]:
157
+ assert fold(s) == fold(fold(s)), s
158
+
159
+
160
+ def test_whitespace_and_punctuation_collapse_consistently() -> None:
161
+ assert fold(" Hà Nội ") == fold("Hà-Nội")
162
+ assert fold("TP. Hồ Chí Minh") == "tp ho chi minh"
163
+
164
+
165
+ def test_nfd_and_nfc_spellings_agree() -> None:
166
+ nfc = "Cái gì thế này"
167
+ nfd = unicodedata.normalize("NFD", nfc)
168
+ assert nfc != nfd
169
+ assert normalize(nfd) == nfc
170
+ assert fold(nfc) == fold(nfd)
171
+ assert decompose(nfc) == nfd
172
+
173
+
174
+ def test_uses_nfd_not_nfkd_so_compatibility_ligatures_survive() -> None:
175
+ # NFKD would rewrite this to "fi" and change English text.
176
+ ligature = chr(0xFB01)
177
+ assert deaccent(ligature) == ligature
178
+
179
+
180
+ def test_repair_mojibake_is_idempotent() -> None:
181
+ for s in ["Ðảm baỏ", "Ðor", "Đặng", "", "Không có gì"]:
182
+ assert repair_mojibake(s) == repair_mojibake(repair_mojibake(s)), s
183
+
184
+
185
+ def test_repair_then_fold_makes_damaged_text_findable() -> None:
186
+ # The whole point of the function: a record indexed after repair is
187
+ # reachable by the string a user would actually type.
188
+ assert fold(repair_mojibake("Ðặng Minh")) == fold("Đặng Minh") == "dang minh"
189
+ # Without the repair step the record is unreachable, and silently so.
190
+ assert fold("Ðặng Minh") != fold("Đặng Minh")
191
+
192
+
193
+ def test_repair_and_fold_answer_different_questions() -> None:
194
+ # fold never reinterprets a letter; repair_mojibake assumes the corpus is
195
+ # Vietnamese and that ETH is damage. Both are correct, for different inputs.
196
+ assert fold("Ðor") == "ðor"
197
+ assert repair_mojibake("Ðor") == "Đor"
198
+
199
+
200
+ # --------------------------------------------------------------------------
201
+ # Zero-dependency runner, for when pytest is not available.
202
+ # --------------------------------------------------------------------------
203
+
204
+
205
+ def _run_standalone() -> int:
206
+ tests = [
207
+ (name, obj)
208
+ for name, obj in sorted(globals().items())
209
+ if name.startswith("test_") and callable(obj)
210
+ ]
211
+ failed = 0
212
+ for name, fn in tests:
213
+ try:
214
+ fn()
215
+ except AssertionError as exc:
216
+ failed += 1
217
+ print(f"FAIL {name}\n{exc}\n")
218
+ else:
219
+ print(f"ok {name}")
220
+ print()
221
+ print(f"{len(tests) - failed}/{len(tests)} test functions passed")
222
+ print(f"{len(CASES)} conformance cases across {len(SUITE['functions'])} functions")
223
+ return 1 if failed else 0
224
+
225
+
226
+ if __name__ == "__main__":
227
+ raise SystemExit(_run_standalone())