vn-text 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vn_text/__init__.py +57 -0
- vn_text/_unicode.py +76 -0
- vn_text/core.py +206 -0
- vn_text/py.typed +0 -0
- vn_text-0.1.0.dist-info/METADATA +98 -0
- vn_text-0.1.0.dist-info/RECORD +8 -0
- vn_text-0.1.0.dist-info/WHEEL +4 -0
- vn_text-0.1.0.dist-info/licenses/LICENSE +21 -0
vn_text/__init__.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""vn-text — correct Unicode primitives for Vietnamese text.
|
|
2
|
+
|
|
3
|
+
Standard library only. Mirrors the TypeScript port function-for-function, and
|
|
4
|
+
both are validated against the shared conformance suite in
|
|
5
|
+
``conformance/vn-text-1.0.0.json``.
|
|
6
|
+
|
|
7
|
+
>>> from vn_text import fold
|
|
8
|
+
>>> fold("Đặng Minh Anh")
|
|
9
|
+
'dang minh anh'
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from ._unicode import (
|
|
15
|
+
COMBINING_MARKS,
|
|
16
|
+
D_WITH_STROKE_LOWER,
|
|
17
|
+
D_WITH_STROKE_UPPER,
|
|
18
|
+
ETH_LOWER,
|
|
19
|
+
ETH_UPPER,
|
|
20
|
+
TONE_MARKS,
|
|
21
|
+
VIETNAMESE_SPECIFIC,
|
|
22
|
+
)
|
|
23
|
+
from .core import (
|
|
24
|
+
decompose,
|
|
25
|
+
deaccent,
|
|
26
|
+
fold,
|
|
27
|
+
is_vietnamese,
|
|
28
|
+
isVietnamese,
|
|
29
|
+
normalize,
|
|
30
|
+
repair_mojibake,
|
|
31
|
+
repairMojibake,
|
|
32
|
+
strip_stroke,
|
|
33
|
+
stripStroke,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
__version__ = "0.1.0"
|
|
37
|
+
|
|
38
|
+
__all__ = [
|
|
39
|
+
"__version__",
|
|
40
|
+
"normalize",
|
|
41
|
+
"decompose",
|
|
42
|
+
"deaccent",
|
|
43
|
+
"strip_stroke",
|
|
44
|
+
"repair_mojibake",
|
|
45
|
+
"fold",
|
|
46
|
+
"is_vietnamese",
|
|
47
|
+
"stripStroke",
|
|
48
|
+
"repairMojibake",
|
|
49
|
+
"isVietnamese",
|
|
50
|
+
"D_WITH_STROKE_UPPER",
|
|
51
|
+
"D_WITH_STROKE_LOWER",
|
|
52
|
+
"ETH_UPPER",
|
|
53
|
+
"ETH_LOWER",
|
|
54
|
+
"TONE_MARKS",
|
|
55
|
+
"COMBINING_MARKS",
|
|
56
|
+
"VIETNAMESE_SPECIFIC",
|
|
57
|
+
]
|
vn_text/_unicode.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Unicode constants that matter for Vietnamese text handling.
|
|
2
|
+
|
|
3
|
+
The trap
|
|
4
|
+
--------
|
|
5
|
+
|
|
6
|
+
Four characters look like a capital or lowercase D with a stroke::
|
|
7
|
+
|
|
8
|
+
U+0110 LATIN CAPITAL LETTER D WITH STROKE -> Vietnamese
|
|
9
|
+
U+0111 LATIN SMALL LETTER D WITH STROKE -> Vietnamese
|
|
10
|
+
U+00D0 LATIN CAPITAL LETTER ETH -> Icelandic, Danish
|
|
11
|
+
U+00F0 LATIN SMALL LETTER ETH -> Icelandic, Danish
|
|
12
|
+
|
|
13
|
+
None has a canonical decomposition, and none has a compatibility decomposition
|
|
14
|
+
either -- verified against both NFD and NFKD. Unicode treats each as a letter in
|
|
15
|
+
its own right rather than a decorated ``D``.
|
|
16
|
+
|
|
17
|
+
Why that breaks search
|
|
18
|
+
----------------------
|
|
19
|
+
|
|
20
|
+
The obvious implementation, normalise to NFD and strip the combining marks,
|
|
21
|
+
silently fails to fold Vietnamese ``Đ``::
|
|
22
|
+
|
|
23
|
+
'Đặng Minh' -> 'Đang Minh'
|
|
24
|
+
|
|
25
|
+
The surviving ``Đ`` means the index key no longer matches anything a user can
|
|
26
|
+
type, because nobody types ``Đang`` to find ``Đặng``. Nothing throws. Search
|
|
27
|
+
just quietly returns fewer results than it should.
|
|
28
|
+
|
|
29
|
+
The obvious repair is worse. A blanket "fold any stroked D" rule also turns ETH
|
|
30
|
+
into ``D``, corrupting Icelandic and Danish names. Transliteration tables that
|
|
31
|
+
treat ``Đ`` as decoration hit exactly that.
|
|
32
|
+
|
|
33
|
+
So the two must be distinguished explicitly. That is what
|
|
34
|
+
:func:`vn_text.strip_stroke` does: U+0110 and U+0111, nothing else.
|
|
35
|
+
|
|
36
|
+
All code points below are written as :func:`chr` calls rather than as literal
|
|
37
|
+
characters or ``\\uXXXX`` escapes. Combining marks are invisible in an editor
|
|
38
|
+
and a stray reformat or copy-paste can silently corrupt a character class that
|
|
39
|
+
happens to still compile. Keeping this module pure ASCII makes that class of
|
|
40
|
+
accident impossible, and keeps diffs readable.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
import re
|
|
46
|
+
|
|
47
|
+
#: LATIN CAPITAL LETTER D WITH STROKE (Vietnamese).
|
|
48
|
+
D_WITH_STROKE_UPPER = chr(0x0110)
|
|
49
|
+
|
|
50
|
+
#: LATIN SMALL LETTER D WITH STROKE (Vietnamese).
|
|
51
|
+
D_WITH_STROKE_LOWER = chr(0x0111)
|
|
52
|
+
|
|
53
|
+
#: LATIN CAPITAL LETTER ETH (Icelandic, Danish). Never fold to D.
|
|
54
|
+
ETH_UPPER = chr(0x00D0)
|
|
55
|
+
|
|
56
|
+
#: LATIN SMALL LETTER ETH (Icelandic, Danish). Never fold to d.
|
|
57
|
+
ETH_LOWER = chr(0x00F0)
|
|
58
|
+
|
|
59
|
+
#: The five Vietnamese tone marks as combining marks, in NFD form.
|
|
60
|
+
#: sắc, huyền, hỏi, ngã, nặng.
|
|
61
|
+
TONE_MARKS: tuple[str, ...] = (
|
|
62
|
+
chr(0x0301), # sắc - COMBINING ACUTE ACCENT
|
|
63
|
+
chr(0x0300), # huyền - COMBINING GRAVE ACCENT
|
|
64
|
+
chr(0x0309), # hỏi - COMBINING HOOK ABOVE
|
|
65
|
+
chr(0x0303), # ngã - COMBINING TILDE
|
|
66
|
+
chr(0x0323), # nặng - COMBINING DOT BELOW
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
#: Full Combining Diacritical Marks block, U+0300 to U+036F.
|
|
70
|
+
COMBINING_MARKS = re.compile(f"[{chr(0x0300)}-{chr(0x036F)}]")
|
|
71
|
+
|
|
72
|
+
#: Vietnamese-specific letters: D WITH STROKE plus the precomposed block
|
|
73
|
+
#: U+1EA0 to U+1EF9 (a-circumflex-dot-below through y-tilde-dot-below).
|
|
74
|
+
VIETNAMESE_SPECIFIC = re.compile(
|
|
75
|
+
f"[{D_WITH_STROKE_UPPER}{D_WITH_STROKE_LOWER}{chr(0x1EA0)}-{chr(0x1EF9)}]"
|
|
76
|
+
)
|
vn_text/core.py
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Core Vietnamese text transforms.
|
|
2
|
+
|
|
3
|
+
Standard library only, on purpose. A text-normalisation library that drags in
|
|
4
|
+
a dependency is a dependency every consumer has to audit, and this one has to
|
|
5
|
+
stay auditable itself.
|
|
6
|
+
|
|
7
|
+
Every function here has a counterpart in the TypeScript port, and both ports
|
|
8
|
+
are validated against the same conformance file at
|
|
9
|
+
``conformance/vn-text-1.0.0.json``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import unicodedata
|
|
15
|
+
|
|
16
|
+
from ._unicode import (
|
|
17
|
+
COMBINING_MARKS,
|
|
18
|
+
D_WITH_STROKE_LOWER,
|
|
19
|
+
D_WITH_STROKE_UPPER,
|
|
20
|
+
ETH_LOWER,
|
|
21
|
+
ETH_UPPER,
|
|
22
|
+
VIETNAMESE_SPECIFIC,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"normalize",
|
|
27
|
+
"decompose",
|
|
28
|
+
"deaccent",
|
|
29
|
+
"strip_stroke",
|
|
30
|
+
"repair_mojibake",
|
|
31
|
+
"fold",
|
|
32
|
+
"is_vietnamese",
|
|
33
|
+
# camelCase aliases for API parity with the TypeScript port.
|
|
34
|
+
"stripStroke",
|
|
35
|
+
"repairMojibake",
|
|
36
|
+
"isVietnamese",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def normalize(text: str) -> str:
|
|
41
|
+
"""Normalise to Unicode Normalization Form C (composed).
|
|
42
|
+
|
|
43
|
+
The form to use for storage, database keys, equality checks and
|
|
44
|
+
deduplication. Two strings that look identical to a user can differ at the
|
|
45
|
+
code-point level if one came from a Windows-1252 editor and the other from
|
|
46
|
+
a UTF-8 terminal; NFC collapses them.
|
|
47
|
+
|
|
48
|
+
>>> normalize(decompose("Cái gì thế này")) == "Cái gì thế này"
|
|
49
|
+
True
|
|
50
|
+
"""
|
|
51
|
+
return unicodedata.normalize("NFC", text)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def decompose(text: str) -> str:
|
|
55
|
+
"""Normalise to Unicode Normalization Form D (decomposed).
|
|
56
|
+
|
|
57
|
+
Vietnamese letters such as ``ế`` become a base ``e`` plus the combining
|
|
58
|
+
marks U+0302 (circumflex) and U+0301 (acute). Exposed because downstream
|
|
59
|
+
code that walks combining marks needs the decomposed form.
|
|
60
|
+
"""
|
|
61
|
+
return unicodedata.normalize("NFD", text)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def deaccent(text: str) -> str:
|
|
65
|
+
"""Remove every combining diacritical mark, leaving base letters intact.
|
|
66
|
+
|
|
67
|
+
Applies to all of Unicode, not just Vietnamese -- ``café`` becomes
|
|
68
|
+
``cafe``, ``Ѐ`` becomes ``Е``.
|
|
69
|
+
|
|
70
|
+
What this function deliberately does NOT do:
|
|
71
|
+
|
|
72
|
+
It does not fold ``đ`` (U+0111) to ``d``. Vietnamese D-WITH-STROKE has no
|
|
73
|
+
canonical decomposition, so it survives NFD untouched and comes back out
|
|
74
|
+
as ``đ``. That is correct: stripping tone marks should not change which
|
|
75
|
+
letter you are looking at. Use :func:`strip_stroke` or :func:`fold` when
|
|
76
|
+
you need ``đ`` to become ``d``.
|
|
77
|
+
|
|
78
|
+
It also leaves ETH (U+00D0) alone, because ETH is a real letter in
|
|
79
|
+
Icelandic and Danish, not a decorated D.
|
|
80
|
+
|
|
81
|
+
>>> deaccent("Tiếng Việt")
|
|
82
|
+
'Tieng Viet'
|
|
83
|
+
>>> deaccent("Đặng Minh")
|
|
84
|
+
'Đang Minh'
|
|
85
|
+
>>> deaccent("Ðor")
|
|
86
|
+
'Ðor'
|
|
87
|
+
"""
|
|
88
|
+
return unicodedata.normalize(
|
|
89
|
+
"NFC", COMBINING_MARKS.sub("", unicodedata.normalize("NFD", text))
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def strip_stroke(text: str) -> str:
|
|
94
|
+
"""Replace Vietnamese D-WITH-STROKE with plain ``d`` / ``D``.
|
|
95
|
+
|
|
96
|
+
Nothing else is changed. In particular tone marks are left alone -- that is
|
|
97
|
+
:func:`deaccent`'s job -- and ETH (U+00D0, U+00F0) is preserved because it
|
|
98
|
+
is a different letter belonging to another script.
|
|
99
|
+
|
|
100
|
+
>>> strip_stroke("Đặng")
|
|
101
|
+
'Dặng'
|
|
102
|
+
>>> strip_stroke("Ðor")
|
|
103
|
+
'Ðor'
|
|
104
|
+
"""
|
|
105
|
+
return text.replace(D_WITH_STROKE_UPPER, "D").replace(
|
|
106
|
+
D_WITH_STROKE_LOWER, "d"
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def repair_mojibake(text: str) -> str:
|
|
111
|
+
"""Repair the most common corruption in stored Vietnamese text.
|
|
112
|
+
|
|
113
|
+
Maps ``Ð`` (U+00D0) to ``Đ`` (U+0110) and ``ð`` (U+00F0) to ``đ`` (U+0111).
|
|
114
|
+
|
|
115
|
+
This is a **data repair** operation, not a search-key operation. It answers
|
|
116
|
+
a different question from :func:`deaccent` and :func:`fold`:
|
|
117
|
+
|
|
118
|
+
- :func:`fold` produces a search key and never reinterprets which letter you
|
|
119
|
+
have. It leaves ``Ð`` alone, because in Icelandic and Danish it is a real
|
|
120
|
+
letter called eth.
|
|
121
|
+
- ``repair_mojibake`` assumes the text *is* Vietnamese and that ``Ð`` is
|
|
122
|
+
damage. In Vietnamese text produced by legacy systems, ``Ð`` is
|
|
123
|
+
essentially always a mangled ``Đ``, and leaving it in place means a user
|
|
124
|
+
can never find the record again.
|
|
125
|
+
|
|
126
|
+
Compose them when the corpus is Vietnamese and may be damaged::
|
|
127
|
+
|
|
128
|
+
fold(repair_mojibake('Ðảm baỏ')) # 'dam bao' -> finds the record
|
|
129
|
+
|
|
130
|
+
What it deliberately does not do:
|
|
131
|
+
|
|
132
|
+
- **It is Vietnamese-biased by design.** ``repair_mojibake('Ðor')`` returns
|
|
133
|
+
``'Đor'``, which is wrong for an Icelandic name. That is the trade, and it
|
|
134
|
+
is the right one for a Vietnamese library, but do not call it on text you
|
|
135
|
+
know contains Scandinavian or Icelandic content.
|
|
136
|
+
- **It does not recover byte-level mojibake.** Damage of the form ``Ä Ä¡``
|
|
137
|
+
comes from UTF-8 bytes decoded as Windows-1252, and undoing it needs the
|
|
138
|
+
original bytes, not a character mapping. There is a dedicated library for
|
|
139
|
+
that (``ftfy``); guessing at it from a Unicode string is not reliable
|
|
140
|
+
enough to ship.
|
|
141
|
+
- **It does not fix spelling or vowel composition.** ``lựơng`` stays
|
|
142
|
+
``lựơng``. Those are separate problems, and ``VietnameseTextNormalizer``
|
|
143
|
+
and ``underthesea.text_normalize`` are the right tools for them.
|
|
144
|
+
|
|
145
|
+
>>> repair_mojibake('Ðảm baỏ')
|
|
146
|
+
'Đảm bảo'
|
|
147
|
+
"""
|
|
148
|
+
return text.replace(ETH_UPPER, D_WITH_STROKE_UPPER).replace(
|
|
149
|
+
ETH_LOWER, D_WITH_STROKE_LOWER
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def fold(text: str) -> str:
|
|
154
|
+
"""Produce a search key.
|
|
155
|
+
|
|
156
|
+
Accent-free, stroke-free, lower-cased, with every run of punctuation and
|
|
157
|
+
symbols collapsed to a single space. This is the canonical form to index
|
|
158
|
+
under for Vietnamese search: two strings a user would consider the same
|
|
159
|
+
hit produce the same key.
|
|
160
|
+
|
|
161
|
+
>>> fold("Cái gì thế này")
|
|
162
|
+
'cai gi the nay'
|
|
163
|
+
>>> fold("Đặng Minh Anh")
|
|
164
|
+
'dang minh anh'
|
|
165
|
+
>>> fold("Hà Nội, Việt Nam!")
|
|
166
|
+
'ha noi viet nam'
|
|
167
|
+
|
|
168
|
+
Letters from other scripts are preserved rather than dropped, so
|
|
169
|
+
``fold("東京 Tokyo")`` gives ``"東京 tokyo"`` -- the ideographs stay
|
|
170
|
+
searchable even though they carry no diacritics.
|
|
171
|
+
"""
|
|
172
|
+
folded = strip_stroke(deaccent(text)).lower()
|
|
173
|
+
# Lower-casing can reintroduce combining marks, e.g. U+0130 -> i + U+0307.
|
|
174
|
+
folded = COMBINING_MARKS.sub("", unicodedata.normalize("NFD", folded))
|
|
175
|
+
# str.isalnum() is the closest stdlib equivalent of JavaScript's
|
|
176
|
+
# \p{L}\p{N}, covering the L* and N* general categories.
|
|
177
|
+
spaced = "".join(ch if ch.isalnum() else " " for ch in folded)
|
|
178
|
+
return " ".join(spaced.split())
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def is_vietnamese(text: str) -> bool:
|
|
182
|
+
"""True when the string contains at least one Vietnamese-specific character.
|
|
183
|
+
|
|
184
|
+
Useful for deciding whether a string needs Vietnamese-aware handling at
|
|
185
|
+
all.
|
|
186
|
+
|
|
187
|
+
This is a fast heuristic, not a language detector: the common Vietnamese
|
|
188
|
+
surnames ``Nguyen``, ``Tran`` and ``Le`` are pure ASCII and return False.
|
|
189
|
+
|
|
190
|
+
>>> is_vietnamese("Đặng")
|
|
191
|
+
True
|
|
192
|
+
>>> is_vietnamese("Nguyen")
|
|
193
|
+
False
|
|
194
|
+
"""
|
|
195
|
+
return VIETNAMESE_SPECIFIC.search(text) is not None
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
#: API-parity alias, so code can move between the Python and TypeScript ports
|
|
199
|
+
#: without renaming every call site.
|
|
200
|
+
stripStroke = strip_stroke
|
|
201
|
+
|
|
202
|
+
#: API-parity alias, see :func:`stripStroke`.
|
|
203
|
+
repairMojibake = repair_mojibake
|
|
204
|
+
|
|
205
|
+
#: API-parity alias, see :func:`stripStroke`.
|
|
206
|
+
isVietnamese = is_vietnamese
|
vn_text/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: vn-text
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Correct Unicode primitives for Vietnamese text: deaccent, fold, normalize, strip_stroke. Standard library only. Ships a cross-language conformance suite shared with the TypeScript port.
|
|
5
|
+
Project-URL: Homepage, https://github.com/leeloc1809/vn-toolkit
|
|
6
|
+
Project-URL: Repository, https://github.com/leeloc1809/vn-toolkit
|
|
7
|
+
Project-URL: Issues, https://github.com/leeloc1809/vn-toolkit/issues
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: diacritics,i18n,nfc,nfd,normalization,search,slugify,unicode,vietnam,vietnamese
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Natural Language :: Vietnamese
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# vn-text
|
|
29
|
+
|
|
30
|
+
Correct Unicode primitives for Vietnamese text. **No runtime dependencies** —
|
|
31
|
+
standard library only.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install vn-text
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from vn_text import fold, deaccent, strip_stroke, is_vietnamese
|
|
39
|
+
|
|
40
|
+
fold('Đặng Minh Anh') # 'dang minh anh'
|
|
41
|
+
fold('Hà Nội, Việt Nam!') # 'ha noi viet nam'
|
|
42
|
+
deaccent('Tiếng Việt') # 'Tieng Viet'
|
|
43
|
+
strip_stroke('Đặng') # 'Dặng'
|
|
44
|
+
is_vietnamese('Nguyễn') # True
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Why
|
|
48
|
+
|
|
49
|
+
Vietnamese `Đ` (U+0110) has no canonical decomposition, so NFD cannot fold it.
|
|
50
|
+
The usual "NFD then strip combining marks" recipe leaves `Đặng Minh` as
|
|
51
|
+
`Đang Minh` — and nobody types `Đang` to find `Đặng`, so search quietly returns
|
|
52
|
+
too few results. The tempting blanket fix also turns the Icelandic letter `Ð`
|
|
53
|
+
(U+00D0) into `D` and corrupts those names.
|
|
54
|
+
|
|
55
|
+
`strip_stroke` handles U+0110 and U+0111 explicitly, and nothing else.
|
|
56
|
+
|
|
57
|
+
## Validation
|
|
58
|
+
|
|
59
|
+
Every function is checked against the shared conformance suite at
|
|
60
|
+
[`conformance/vn-text-1.0.0.json`](../../conformance/vn-text-1.0.0.json) — 156
|
|
61
|
+
hand-authored cases that the TypeScript port also consumes, so the two
|
|
62
|
+
implementations cannot drift apart.
|
|
63
|
+
|
|
64
|
+
The test file runs with or without pytest, because a library whose correctness
|
|
65
|
+
guarantee depends on a test runner is a library you cannot check on a machine
|
|
66
|
+
that has no test runner:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
python packages/vn-text-py/tests/test_vn_text.py # stdlib only, nothing to install
|
|
70
|
+
|
|
71
|
+
pytest packages/vn-text-py/tests -q # if you have it
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## API parity
|
|
75
|
+
|
|
76
|
+
`snake_case` is the Python convention, but `stripStroke` and `isVietnamese` are
|
|
77
|
+
exported as aliases so a call site can move between the Python and TypeScript
|
|
78
|
+
ports without renaming.
|
|
79
|
+
|
|
80
|
+
## Known port risk
|
|
81
|
+
|
|
82
|
+
`fold()` classifies characters using `str.isalnum()`, while the TypeScript port
|
|
83
|
+
uses `\p{L}\p{N}`. These agree across all 127 distinct characters in the current
|
|
84
|
+
corpus, measured by `scripts/verify-port-parity.mjs`. They are not guaranteed to
|
|
85
|
+
agree for every code point; the conformance suite is how a divergence gets
|
|
86
|
+
caught.
|
|
87
|
+
|
|
88
|
+
## Scope
|
|
89
|
+
|
|
90
|
+
Deliberately narrow. Tone-mark canonicalisation (`hóa` vs `hòa`) and fuzzy
|
|
91
|
+
matching for unaccented-keyboard input are not here. Sorting is not here
|
|
92
|
+
either -- it is [`vn-collate`](../vn-collate-py), which shares this
|
|
93
|
+
repository's conformance infrastructure. See the
|
|
94
|
+
[root README](../../README.md#what-this-is-not) for the reasoning.
|
|
95
|
+
|
|
96
|
+
## Licence
|
|
97
|
+
|
|
98
|
+
MIT
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
vn_text/__init__.py,sha256=3Qg0FKvvCSb2ZONeigBog2dCkkltzexCip99AfTOUHI,1122
|
|
2
|
+
vn_text/_unicode.py,sha256=TuP33WgIWgbZBoc0MT_xChl3m8jXm-d5qb6K-uAm55I,2922
|
|
3
|
+
vn_text/core.py,sha256=TFvX66DTXF6T9unUbaL24YfJJsA6RdwcvIe3CzPdNHk,7225
|
|
4
|
+
vn_text/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
5
|
+
vn_text-0.1.0.dist-info/METADATA,sha256=B41YqxdVqVW4au5t0WboJ1IKwBKsw6R_f-lZIKPUwc4,3683
|
|
6
|
+
vn_text-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
7
|
+
vn_text-0.1.0.dist-info/licenses/LICENSE,sha256=jBDksFrsccC6mMMWfqpJtTCSLSe1WQXAJW0I1K99mB0,1080
|
|
8
|
+
vn_text-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 vn-toolkit contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|