vn-text 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vn_text-0.1.0/.gitignore +32 -0
- vn_text-0.1.0/LICENSE +21 -0
- vn_text-0.1.0/PKG-INFO +98 -0
- vn_text-0.1.0/README.md +71 -0
- vn_text-0.1.0/pyproject.toml +60 -0
- vn_text-0.1.0/src/vn_text/__init__.py +57 -0
- vn_text-0.1.0/src/vn_text/_unicode.py +76 -0
- vn_text-0.1.0/src/vn_text/core.py +206 -0
- vn_text-0.1.0/src/vn_text/py.typed +0 -0
- vn_text-0.1.0/tests/test_vn_text.py +227 -0
vn_text-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
node_modules/
|
|
2
|
+
dist/
|
|
3
|
+
*.tsbuildinfo
|
|
4
|
+
|
|
5
|
+
# Python
|
|
6
|
+
__pycache__/
|
|
7
|
+
*.py[cod]
|
|
8
|
+
.venv/
|
|
9
|
+
venv/
|
|
10
|
+
*.egg-info/
|
|
11
|
+
.pytest_cache/
|
|
12
|
+
.mypy_cache/
|
|
13
|
+
.ruff_cache/
|
|
14
|
+
|
|
15
|
+
# Editors / OS
|
|
16
|
+
.DS_Store
|
|
17
|
+
Thumbs.db
|
|
18
|
+
.idea/
|
|
19
|
+
.vscode/
|
|
20
|
+
|
|
21
|
+
# Build output of the conformance tooling
|
|
22
|
+
.tmp/
|
|
23
|
+
|
|
24
|
+
# Traction snapshots. Machine-generated history for scripts/track-traction.mjs,
|
|
25
|
+
# not something to review in a diff.
|
|
26
|
+
.traction/
|
|
27
|
+
|
|
28
|
+
# LICENSE copies staged into each package at pack time by
|
|
29
|
+
# scripts/stage-license.mjs. One canonical file at the repository root; these
|
|
30
|
+
# are build output, and a staged one in a diff is a bug, not a licence update.
|
|
31
|
+
packages/*/LICENSE
|
|
32
|
+
packages/*/*/LICENSE
|
vn_text-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 vn-toolkit contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
vn_text-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: vn-text
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Correct Unicode primitives for Vietnamese text: deaccent, fold, normalize, strip_stroke. Standard library only. Ships a cross-language conformance suite shared with the TypeScript port.
|
|
5
|
+
Project-URL: Homepage, https://github.com/leeloc1809/vn-toolkit
|
|
6
|
+
Project-URL: Repository, https://github.com/leeloc1809/vn-toolkit
|
|
7
|
+
Project-URL: Issues, https://github.com/leeloc1809/vn-toolkit/issues
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: diacritics,i18n,nfc,nfd,normalization,search,slugify,unicode,vietnam,vietnamese
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Natural Language :: Vietnamese
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# vn-text
|
|
29
|
+
|
|
30
|
+
Correct Unicode primitives for Vietnamese text. **No runtime dependencies** —
|
|
31
|
+
standard library only.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install vn-text
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from vn_text import fold, deaccent, strip_stroke, is_vietnamese
|
|
39
|
+
|
|
40
|
+
fold('Đặng Minh Anh') # 'dang minh anh'
|
|
41
|
+
fold('Hà Nội, Việt Nam!') # 'ha noi viet nam'
|
|
42
|
+
deaccent('Tiếng Việt') # 'Tieng Viet'
|
|
43
|
+
strip_stroke('Đặng') # 'Dặng'
|
|
44
|
+
is_vietnamese('Nguyễn') # True
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Why
|
|
48
|
+
|
|
49
|
+
Vietnamese `Đ` (U+0110) has no canonical decomposition, so NFD cannot fold it.
|
|
50
|
+
The usual "NFD then strip combining marks" recipe leaves `Đặng Minh` as
|
|
51
|
+
`Đang Minh` — and nobody types `Đang` to find `Đặng`, so search quietly returns
|
|
52
|
+
too few results. The tempting blanket fix also turns the Icelandic letter `Ð`
|
|
53
|
+
(U+00D0) into `D` and corrupts those names.
|
|
54
|
+
|
|
55
|
+
`strip_stroke` handles U+0110 and U+0111 explicitly, and nothing else.
|
|
56
|
+
|
|
57
|
+
## Validation
|
|
58
|
+
|
|
59
|
+
Every function is checked against the shared conformance suite at
|
|
60
|
+
[`conformance/vn-text-1.0.0.json`](../../conformance/vn-text-1.0.0.json) — 156
|
|
61
|
+
hand-authored cases that the TypeScript port also consumes, so the two
|
|
62
|
+
implementations cannot drift apart.
|
|
63
|
+
|
|
64
|
+
The test file runs with or without pytest, because a library whose correctness
|
|
65
|
+
guarantee depends on a test runner is a library you cannot check on a machine
|
|
66
|
+
that has no test runner:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
python packages/vn-text-py/tests/test_vn_text.py # stdlib only, nothing to install
|
|
70
|
+
|
|
71
|
+
pytest packages/vn-text-py/tests -q # if you have it
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## API parity
|
|
75
|
+
|
|
76
|
+
`snake_case` is the Python convention, but `stripStroke` and `isVietnamese` are
|
|
77
|
+
exported as aliases so a call site can move between the Python and TypeScript
|
|
78
|
+
ports without renaming.
|
|
79
|
+
|
|
80
|
+
## Known port risk
|
|
81
|
+
|
|
82
|
+
`fold()` classifies characters using `str.isalnum()`, while the TypeScript port
|
|
83
|
+
uses `\p{L}\p{N}`. These agree across all 127 distinct characters in the current
|
|
84
|
+
corpus, measured by `scripts/verify-port-parity.mjs`. They are not guaranteed to
|
|
85
|
+
agree for every code point; the conformance suite is how a divergence gets
|
|
86
|
+
caught.
|
|
87
|
+
|
|
88
|
+
## Scope
|
|
89
|
+
|
|
90
|
+
Deliberately narrow. Tone-mark canonicalisation (`hóa` vs `hòa`) and fuzzy
|
|
91
|
+
matching for unaccented-keyboard input are not here. Sorting is not here
|
|
92
|
+
either -- it is [`vn-collate`](../vn-collate-py), which shares this
|
|
93
|
+
repository's conformance infrastructure. See the
|
|
94
|
+
[root README](../../README.md#what-this-is-not) for the reasoning.
|
|
95
|
+
|
|
96
|
+
## Licence
|
|
97
|
+
|
|
98
|
+
MIT
|
vn_text-0.1.0/README.md
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
# vn-text
|
|
2
|
+
|
|
3
|
+
Correct Unicode primitives for Vietnamese text. **No runtime dependencies** —
|
|
4
|
+
standard library only.
|
|
5
|
+
|
|
6
|
+
```bash
|
|
7
|
+
pip install vn-text
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
from vn_text import fold, deaccent, strip_stroke, is_vietnamese
|
|
12
|
+
|
|
13
|
+
fold('Đặng Minh Anh') # 'dang minh anh'
|
|
14
|
+
fold('Hà Nội, Việt Nam!') # 'ha noi viet nam'
|
|
15
|
+
deaccent('Tiếng Việt') # 'Tieng Viet'
|
|
16
|
+
strip_stroke('Đặng') # 'Dặng'
|
|
17
|
+
is_vietnamese('Nguyễn') # True
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## Why
|
|
21
|
+
|
|
22
|
+
Vietnamese `Đ` (U+0110) has no canonical decomposition, so NFD cannot fold it.
|
|
23
|
+
The usual "NFD then strip combining marks" recipe leaves `Đặng Minh` as
|
|
24
|
+
`Đang Minh` — and nobody types `Đang` to find `Đặng`, so search quietly returns
|
|
25
|
+
too few results. The tempting blanket fix also turns the Icelandic letter `Ð`
|
|
26
|
+
(U+00D0) into `D` and corrupts those names.
|
|
27
|
+
|
|
28
|
+
`strip_stroke` handles U+0110 and U+0111 explicitly, and nothing else.
|
|
29
|
+
|
|
30
|
+
## Validation
|
|
31
|
+
|
|
32
|
+
Every function is checked against the shared conformance suite at
|
|
33
|
+
[`conformance/vn-text-1.0.0.json`](../../conformance/vn-text-1.0.0.json) — 156
|
|
34
|
+
hand-authored cases that the TypeScript port also consumes, so the two
|
|
35
|
+
implementations cannot drift apart.
|
|
36
|
+
|
|
37
|
+
The test file runs with or without pytest, because a library whose correctness
|
|
38
|
+
guarantee depends on a test runner is a library you cannot check on a machine
|
|
39
|
+
that has no test runner:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
python packages/vn-text-py/tests/test_vn_text.py # stdlib only, nothing to install
|
|
43
|
+
|
|
44
|
+
pytest packages/vn-text-py/tests -q # if you have it
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## API parity
|
|
48
|
+
|
|
49
|
+
`snake_case` is the Python convention, but `stripStroke` and `isVietnamese` are
|
|
50
|
+
exported as aliases so a call site can move between the Python and TypeScript
|
|
51
|
+
ports without renaming.
|
|
52
|
+
|
|
53
|
+
## Known port risk
|
|
54
|
+
|
|
55
|
+
`fold()` classifies characters using `str.isalnum()`, while the TypeScript port
|
|
56
|
+
uses `\p{L}\p{N}`. These agree across all 127 distinct characters in the current
|
|
57
|
+
corpus, measured by `scripts/verify-port-parity.mjs`. They are not guaranteed to
|
|
58
|
+
agree for every code point; the conformance suite is how a divergence gets
|
|
59
|
+
caught.
|
|
60
|
+
|
|
61
|
+
## Scope
|
|
62
|
+
|
|
63
|
+
Deliberately narrow. Tone-mark canonicalisation (`hóa` vs `hòa`) and fuzzy
|
|
64
|
+
matching for unaccented-keyboard input are not here. Sorting is not here
|
|
65
|
+
either -- it is [`vn-collate`](../vn-collate-py), which shares this
|
|
66
|
+
repository's conformance infrastructure. See the
|
|
67
|
+
[root README](../../README.md#what-this-is-not) for the reasoning.
|
|
68
|
+
|
|
69
|
+
## Licence
|
|
70
|
+
|
|
71
|
+
MIT
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.24"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "vn-text"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Correct Unicode primitives for Vietnamese text: deaccent, fold, normalize, strip_stroke. Standard library only. Ships a cross-language conformance suite shared with the TypeScript port."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
license-files = { paths = ["LICENSE"] }
|
|
13
|
+
keywords = [
|
|
14
|
+
"vietnamese",
|
|
15
|
+
"unicode",
|
|
16
|
+
"normalization",
|
|
17
|
+
"search",
|
|
18
|
+
"nfc",
|
|
19
|
+
"nfd",
|
|
20
|
+
"diacritics",
|
|
21
|
+
"slugify",
|
|
22
|
+
"i18n",
|
|
23
|
+
"vietnam",
|
|
24
|
+
]
|
|
25
|
+
classifiers = [
|
|
26
|
+
"Development Status :: 3 - Alpha",
|
|
27
|
+
"Intended Audience :: Developers",
|
|
28
|
+
"License :: OSI Approved :: MIT License",
|
|
29
|
+
"Natural Language :: Vietnamese",
|
|
30
|
+
"Programming Language :: Python :: 3",
|
|
31
|
+
"Programming Language :: Python :: 3.9",
|
|
32
|
+
"Programming Language :: Python :: 3.10",
|
|
33
|
+
"Programming Language :: Python :: 3.11",
|
|
34
|
+
"Programming Language :: Python :: 3.12",
|
|
35
|
+
"Programming Language :: Python :: 3.13",
|
|
36
|
+
"Topic :: Text Processing :: Linguistic",
|
|
37
|
+
"Typing :: Typed",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
# Intentionally empty. A text-normalisation library that drags in a
|
|
41
|
+
# dependency is a dependency every consumer has to audit, and this one has to
|
|
42
|
+
# stay auditable itself.
|
|
43
|
+
dependencies = []
|
|
44
|
+
|
|
45
|
+
[project.optional-dependencies]
|
|
46
|
+
dev = ["pytest>=8.0"]
|
|
47
|
+
|
|
48
|
+
[project.urls]
|
|
49
|
+
Homepage = "https://github.com/leeloc1809/vn-toolkit"
|
|
50
|
+
Repository = "https://github.com/leeloc1809/vn-toolkit"
|
|
51
|
+
Issues = "https://github.com/leeloc1809/vn-toolkit/issues"
|
|
52
|
+
|
|
53
|
+
[tool.hatch.build.targets.wheel]
|
|
54
|
+
packages = ["src/vn_text"]
|
|
55
|
+
|
|
56
|
+
[tool.hatch.build.targets.sdist]
|
|
57
|
+
include = ["src/vn_text", "tests", "README.md", "LICENSE"]
|
|
58
|
+
|
|
59
|
+
[tool.pytest.ini_options]
|
|
60
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""vn-text — correct Unicode primitives for Vietnamese text.
|
|
2
|
+
|
|
3
|
+
Standard library only. Mirrors the TypeScript port function-for-function, and
|
|
4
|
+
both are validated against the shared conformance suite in
|
|
5
|
+
``conformance/vn-text-1.0.0.json``.
|
|
6
|
+
|
|
7
|
+
>>> from vn_text import fold
|
|
8
|
+
>>> fold("Đặng Minh Anh")
|
|
9
|
+
'dang minh anh'
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from ._unicode import (
|
|
15
|
+
COMBINING_MARKS,
|
|
16
|
+
D_WITH_STROKE_LOWER,
|
|
17
|
+
D_WITH_STROKE_UPPER,
|
|
18
|
+
ETH_LOWER,
|
|
19
|
+
ETH_UPPER,
|
|
20
|
+
TONE_MARKS,
|
|
21
|
+
VIETNAMESE_SPECIFIC,
|
|
22
|
+
)
|
|
23
|
+
from .core import (
|
|
24
|
+
decompose,
|
|
25
|
+
deaccent,
|
|
26
|
+
fold,
|
|
27
|
+
is_vietnamese,
|
|
28
|
+
isVietnamese,
|
|
29
|
+
normalize,
|
|
30
|
+
repair_mojibake,
|
|
31
|
+
repairMojibake,
|
|
32
|
+
strip_stroke,
|
|
33
|
+
stripStroke,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
__version__ = "0.1.0"
|
|
37
|
+
|
|
38
|
+
__all__ = [
|
|
39
|
+
"__version__",
|
|
40
|
+
"normalize",
|
|
41
|
+
"decompose",
|
|
42
|
+
"deaccent",
|
|
43
|
+
"strip_stroke",
|
|
44
|
+
"repair_mojibake",
|
|
45
|
+
"fold",
|
|
46
|
+
"is_vietnamese",
|
|
47
|
+
"stripStroke",
|
|
48
|
+
"repairMojibake",
|
|
49
|
+
"isVietnamese",
|
|
50
|
+
"D_WITH_STROKE_UPPER",
|
|
51
|
+
"D_WITH_STROKE_LOWER",
|
|
52
|
+
"ETH_UPPER",
|
|
53
|
+
"ETH_LOWER",
|
|
54
|
+
"TONE_MARKS",
|
|
55
|
+
"COMBINING_MARKS",
|
|
56
|
+
"VIETNAMESE_SPECIFIC",
|
|
57
|
+
]
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Unicode constants that matter for Vietnamese text handling.
|
|
2
|
+
|
|
3
|
+
The trap
|
|
4
|
+
--------
|
|
5
|
+
|
|
6
|
+
Four characters look like a capital or lowercase D with a stroke::
|
|
7
|
+
|
|
8
|
+
U+0110 LATIN CAPITAL LETTER D WITH STROKE -> Vietnamese
|
|
9
|
+
U+0111 LATIN SMALL LETTER D WITH STROKE -> Vietnamese
|
|
10
|
+
U+00D0 LATIN CAPITAL LETTER ETH -> Icelandic, Danish
|
|
11
|
+
U+00F0 LATIN SMALL LETTER ETH -> Icelandic, Danish
|
|
12
|
+
|
|
13
|
+
None has a canonical decomposition, and none has a compatibility decomposition
|
|
14
|
+
either -- verified against both NFD and NFKD. Unicode treats each as a letter in
|
|
15
|
+
its own right rather than a decorated ``D``.
|
|
16
|
+
|
|
17
|
+
Why that breaks search
|
|
18
|
+
----------------------
|
|
19
|
+
|
|
20
|
+
The obvious implementation, normalise to NFD and strip the combining marks,
|
|
21
|
+
silently fails to fold Vietnamese ``Đ``::
|
|
22
|
+
|
|
23
|
+
'Đặng Minh' -> 'Đang Minh'
|
|
24
|
+
|
|
25
|
+
The surviving ``Đ`` means the index key no longer matches anything a user can
|
|
26
|
+
type, because nobody types ``Đang`` to find ``Đặng``. Nothing throws. Search
|
|
27
|
+
just quietly returns fewer results than it should.
|
|
28
|
+
|
|
29
|
+
The obvious repair is worse. A blanket "fold any stroked D" rule also turns ETH
|
|
30
|
+
into ``D``, corrupting Icelandic and Danish names. Transliteration tables that
|
|
31
|
+
treat ``Đ`` as decoration hit exactly that.
|
|
32
|
+
|
|
33
|
+
So the two must be distinguished explicitly. That is what
|
|
34
|
+
:func:`vn_text.strip_stroke` does: U+0110 and U+0111, nothing else.
|
|
35
|
+
|
|
36
|
+
All code points below are written as :func:`chr` calls rather than as literal
|
|
37
|
+
characters or ``\\uXXXX`` escapes. Combining marks are invisible in an editor
|
|
38
|
+
and a stray reformat or copy-paste can silently corrupt a character class that
|
|
39
|
+
happens to still compile. Keeping this module pure ASCII makes that class of
|
|
40
|
+
accident impossible, and keeps diffs readable.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
import re
|
|
46
|
+
|
|
47
|
+
#: LATIN CAPITAL LETTER D WITH STROKE (Vietnamese).
|
|
48
|
+
D_WITH_STROKE_UPPER = chr(0x0110)
|
|
49
|
+
|
|
50
|
+
#: LATIN SMALL LETTER D WITH STROKE (Vietnamese).
|
|
51
|
+
D_WITH_STROKE_LOWER = chr(0x0111)
|
|
52
|
+
|
|
53
|
+
#: LATIN CAPITAL LETTER ETH (Icelandic, Danish). Never fold to D.
|
|
54
|
+
ETH_UPPER = chr(0x00D0)
|
|
55
|
+
|
|
56
|
+
#: LATIN SMALL LETTER ETH (Icelandic, Danish). Never fold to d.
|
|
57
|
+
ETH_LOWER = chr(0x00F0)
|
|
58
|
+
|
|
59
|
+
#: The five Vietnamese tone marks as combining marks, in NFD form.
|
|
60
|
+
#: sắc, huyền, hỏi, ngã, nặng.
|
|
61
|
+
TONE_MARKS: tuple[str, ...] = (
|
|
62
|
+
chr(0x0301), # sắc - COMBINING ACUTE ACCENT
|
|
63
|
+
chr(0x0300), # huyền - COMBINING GRAVE ACCENT
|
|
64
|
+
chr(0x0309), # hỏi - COMBINING HOOK ABOVE
|
|
65
|
+
chr(0x0303), # ngã - COMBINING TILDE
|
|
66
|
+
chr(0x0323), # nặng - COMBINING DOT BELOW
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
#: Full Combining Diacritical Marks block, U+0300 to U+036F.
|
|
70
|
+
COMBINING_MARKS = re.compile(f"[{chr(0x0300)}-{chr(0x036F)}]")
|
|
71
|
+
|
|
72
|
+
#: Vietnamese-specific letters: D WITH STROKE plus the precomposed block
|
|
73
|
+
#: U+1EA0 to U+1EF9 (a-circumflex-dot-below through y-tilde-dot-below).
|
|
74
|
+
VIETNAMESE_SPECIFIC = re.compile(
|
|
75
|
+
f"[{D_WITH_STROKE_UPPER}{D_WITH_STROKE_LOWER}{chr(0x1EA0)}-{chr(0x1EF9)}]"
|
|
76
|
+
)
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Core Vietnamese text transforms.
|
|
2
|
+
|
|
3
|
+
Standard library only, on purpose. A text-normalisation library that drags in
|
|
4
|
+
a dependency is a dependency every consumer has to audit, and this one has to
|
|
5
|
+
stay auditable itself.
|
|
6
|
+
|
|
7
|
+
Every function here has a counterpart in the TypeScript port, and both ports
|
|
8
|
+
are validated against the same conformance file at
|
|
9
|
+
``conformance/vn-text-1.0.0.json``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import unicodedata
|
|
15
|
+
|
|
16
|
+
from ._unicode import (
|
|
17
|
+
COMBINING_MARKS,
|
|
18
|
+
D_WITH_STROKE_LOWER,
|
|
19
|
+
D_WITH_STROKE_UPPER,
|
|
20
|
+
ETH_LOWER,
|
|
21
|
+
ETH_UPPER,
|
|
22
|
+
VIETNAMESE_SPECIFIC,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"normalize",
|
|
27
|
+
"decompose",
|
|
28
|
+
"deaccent",
|
|
29
|
+
"strip_stroke",
|
|
30
|
+
"repair_mojibake",
|
|
31
|
+
"fold",
|
|
32
|
+
"is_vietnamese",
|
|
33
|
+
# camelCase aliases for API parity with the TypeScript port.
|
|
34
|
+
"stripStroke",
|
|
35
|
+
"repairMojibake",
|
|
36
|
+
"isVietnamese",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def normalize(text: str) -> str:
|
|
41
|
+
"""Normalise to Unicode Normalization Form C (composed).
|
|
42
|
+
|
|
43
|
+
The form to use for storage, database keys, equality checks and
|
|
44
|
+
deduplication. Two strings that look identical to a user can differ at the
|
|
45
|
+
code-point level if one came from a Windows-1252 editor and the other from
|
|
46
|
+
a UTF-8 terminal; NFC collapses them.
|
|
47
|
+
|
|
48
|
+
>>> normalize(decompose("Cái gì thế này")) == "Cái gì thế này"
|
|
49
|
+
True
|
|
50
|
+
"""
|
|
51
|
+
return unicodedata.normalize("NFC", text)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def decompose(text: str) -> str:
|
|
55
|
+
"""Normalise to Unicode Normalization Form D (decomposed).
|
|
56
|
+
|
|
57
|
+
Vietnamese letters such as ``ế`` become a base ``e`` plus the combining
|
|
58
|
+
marks U+0302 (circumflex) and U+0301 (acute). Exposed because downstream
|
|
59
|
+
code that walks combining marks needs the decomposed form.
|
|
60
|
+
"""
|
|
61
|
+
return unicodedata.normalize("NFD", text)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def deaccent(text: str) -> str:
|
|
65
|
+
"""Remove every combining diacritical mark, leaving base letters intact.
|
|
66
|
+
|
|
67
|
+
Applies to all of Unicode, not just Vietnamese -- ``café`` becomes
|
|
68
|
+
``cafe``, ``Ѐ`` becomes ``Е``.
|
|
69
|
+
|
|
70
|
+
What this function deliberately does NOT do:
|
|
71
|
+
|
|
72
|
+
It does not fold ``đ`` (U+0111) to ``d``. Vietnamese D-WITH-STROKE has no
|
|
73
|
+
canonical decomposition, so it survives NFD untouched and comes back out
|
|
74
|
+
as ``đ``. That is correct: stripping tone marks should not change which
|
|
75
|
+
letter you are looking at. Use :func:`strip_stroke` or :func:`fold` when
|
|
76
|
+
you need ``đ`` to become ``d``.
|
|
77
|
+
|
|
78
|
+
It also leaves ETH (U+00D0) alone, because ETH is a real letter in
|
|
79
|
+
Icelandic and Danish, not a decorated D.
|
|
80
|
+
|
|
81
|
+
>>> deaccent("Tiếng Việt")
|
|
82
|
+
'Tieng Viet'
|
|
83
|
+
>>> deaccent("Đặng Minh")
|
|
84
|
+
'Đang Minh'
|
|
85
|
+
>>> deaccent("Ðor")
|
|
86
|
+
'Ðor'
|
|
87
|
+
"""
|
|
88
|
+
return unicodedata.normalize(
|
|
89
|
+
"NFC", COMBINING_MARKS.sub("", unicodedata.normalize("NFD", text))
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def strip_stroke(text: str) -> str:
|
|
94
|
+
"""Replace Vietnamese D-WITH-STROKE with plain ``d`` / ``D``.
|
|
95
|
+
|
|
96
|
+
Nothing else is changed. In particular tone marks are left alone -- that is
|
|
97
|
+
:func:`deaccent`'s job -- and ETH (U+00D0, U+00F0) is preserved because it
|
|
98
|
+
is a different letter belonging to another script.
|
|
99
|
+
|
|
100
|
+
>>> strip_stroke("Đặng")
|
|
101
|
+
'Dặng'
|
|
102
|
+
>>> strip_stroke("Ðor")
|
|
103
|
+
'Ðor'
|
|
104
|
+
"""
|
|
105
|
+
return text.replace(D_WITH_STROKE_UPPER, "D").replace(
|
|
106
|
+
D_WITH_STROKE_LOWER, "d"
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def repair_mojibake(text: str) -> str:
|
|
111
|
+
"""Repair the most common corruption in stored Vietnamese text.
|
|
112
|
+
|
|
113
|
+
Maps ``Ð`` (U+00D0) to ``Đ`` (U+0110) and ``ð`` (U+00F0) to ``đ`` (U+0111).
|
|
114
|
+
|
|
115
|
+
This is a **data repair** operation, not a search-key operation. It answers
|
|
116
|
+
a different question from :func:`deaccent` and :func:`fold`:
|
|
117
|
+
|
|
118
|
+
- :func:`fold` produces a search key and never reinterprets which letter you
|
|
119
|
+
have. It leaves ``Ð`` alone, because in Icelandic and Danish it is a real
|
|
120
|
+
letter called eth.
|
|
121
|
+
- ``repair_mojibake`` assumes the text *is* Vietnamese and that ``Ð`` is
|
|
122
|
+
damage. In Vietnamese text produced by legacy systems, ``Ð`` is
|
|
123
|
+
essentially always a mangled ``Đ``, and leaving it in place means a user
|
|
124
|
+
can never find the record again.
|
|
125
|
+
|
|
126
|
+
Compose them when the corpus is Vietnamese and may be damaged::
|
|
127
|
+
|
|
128
|
+
fold(repair_mojibake('Ðảm baỏ')) # 'dam bao' -> finds the record
|
|
129
|
+
|
|
130
|
+
What it deliberately does not do:
|
|
131
|
+
|
|
132
|
+
- **It is Vietnamese-biased by design.** ``repair_mojibake('Ðor')`` returns
|
|
133
|
+
``'Đor'``, which is wrong for an Icelandic name. That is the trade, and it
|
|
134
|
+
is the right one for a Vietnamese library, but do not call it on text you
|
|
135
|
+
know contains Scandinavian or Icelandic content.
|
|
136
|
+
- **It does not recover byte-level mojibake.** Damage of the form ``Ä Ä¡``
|
|
137
|
+
comes from UTF-8 bytes decoded as Windows-1252, and undoing it needs the
|
|
138
|
+
original bytes, not a character mapping. There is a dedicated library for
|
|
139
|
+
that (``ftfy``); guessing at it from a Unicode string is not reliable
|
|
140
|
+
enough to ship.
|
|
141
|
+
- **It does not fix spelling or vowel composition.** ``lựơng`` stays
|
|
142
|
+
``lựơng``. Those are separate problems, and ``VietnameseTextNormalizer``
|
|
143
|
+
and ``underthesea.text_normalize`` are the right tools for them.
|
|
144
|
+
|
|
145
|
+
>>> repair_mojibake('Ðảm baỏ')
|
|
146
|
+
'Đảm bảo'
|
|
147
|
+
"""
|
|
148
|
+
return text.replace(ETH_UPPER, D_WITH_STROKE_UPPER).replace(
|
|
149
|
+
ETH_LOWER, D_WITH_STROKE_LOWER
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def fold(text: str) -> str:
|
|
154
|
+
"""Produce a search key.
|
|
155
|
+
|
|
156
|
+
Accent-free, stroke-free, lower-cased, with every run of punctuation and
|
|
157
|
+
symbols collapsed to a single space. This is the canonical form to index
|
|
158
|
+
under for Vietnamese search: two strings a user would consider the same
|
|
159
|
+
hit produce the same key.
|
|
160
|
+
|
|
161
|
+
>>> fold("Cái gì thế này")
|
|
162
|
+
'cai gi the nay'
|
|
163
|
+
>>> fold("Đặng Minh Anh")
|
|
164
|
+
'dang minh anh'
|
|
165
|
+
>>> fold("Hà Nội, Việt Nam!")
|
|
166
|
+
'ha noi viet nam'
|
|
167
|
+
|
|
168
|
+
Letters from other scripts are preserved rather than dropped, so
|
|
169
|
+
``fold("東京 Tokyo")`` gives ``"東京 tokyo"`` -- the ideographs stay
|
|
170
|
+
searchable even though they carry no diacritics.
|
|
171
|
+
"""
|
|
172
|
+
folded = strip_stroke(deaccent(text)).lower()
|
|
173
|
+
# Lower-casing can reintroduce combining marks, e.g. U+0130 -> i + U+0307.
|
|
174
|
+
folded = COMBINING_MARKS.sub("", unicodedata.normalize("NFD", folded))
|
|
175
|
+
# str.isalnum() is the closest stdlib equivalent of JavaScript's
|
|
176
|
+
# \p{L}\p{N}, covering the L* and N* general categories.
|
|
177
|
+
spaced = "".join(ch if ch.isalnum() else " " for ch in folded)
|
|
178
|
+
return " ".join(spaced.split())
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def is_vietnamese(text: str) -> bool:
|
|
182
|
+
"""True when the string contains at least one Vietnamese-specific character.
|
|
183
|
+
|
|
184
|
+
Useful for deciding whether a string needs Vietnamese-aware handling at
|
|
185
|
+
all.
|
|
186
|
+
|
|
187
|
+
This is a fast heuristic, not a language detector: the common Vietnamese
|
|
188
|
+
surnames ``Nguyen``, ``Tran`` and ``Le`` are pure ASCII and return False.
|
|
189
|
+
|
|
190
|
+
>>> is_vietnamese("Đặng")
|
|
191
|
+
True
|
|
192
|
+
>>> is_vietnamese("Nguyen")
|
|
193
|
+
False
|
|
194
|
+
"""
|
|
195
|
+
return VIETNAMESE_SPECIFIC.search(text) is not None
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
#: API-parity alias, so code can move between the Python and TypeScript ports
|
|
199
|
+
#: without renaming every call site.
|
|
200
|
+
stripStroke = strip_stroke
|
|
201
|
+
|
|
202
|
+
#: API-parity alias, see :func:`stripStroke`.
|
|
203
|
+
repairMojibake = repair_mojibake
|
|
204
|
+
|
|
205
|
+
#: API-parity alias, see :func:`stripStroke`.
|
|
206
|
+
isVietnamese = is_vietnamese
|
|
File without changes
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
"""Run the shared conformance suite against the Python port.
|
|
2
|
+
|
|
3
|
+
This file is deliberately runnable two ways:
|
|
4
|
+
|
|
5
|
+
pytest packages/py/tests -q # full runner, nice reporting
|
|
6
|
+
python packages/vn-text-py/tests/test_vn_text.py # stdlib only, no install
|
|
7
|
+
|
|
8
|
+
The second form exists because the conformance suite is the contract between
|
|
9
|
+
the two ports. Being able to check it with nothing but a Python interpreter
|
|
10
|
+
means the Python side can never silently drift from the TypeScript side.
|
|
11
|
+
|
|
12
|
+
Both ports read the exact same JSON file. There is no second copy to keep in
|
|
13
|
+
sync, which is the whole point.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
import sys
|
|
20
|
+
import unicodedata
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any, Callable
|
|
23
|
+
|
|
24
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
25
|
+
|
|
26
|
+
from vn_text import ( # noqa: E402 (path set up above)
|
|
27
|
+
deaccent,
|
|
28
|
+
decompose,
|
|
29
|
+
fold,
|
|
30
|
+
is_vietnamese,
|
|
31
|
+
normalize,
|
|
32
|
+
repair_mojibake,
|
|
33
|
+
strip_stroke,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
CONFORMANCE_PATH = (
|
|
37
|
+
Path(__file__).resolve().parents[3] / "conformance" / "vn-text-1.0.0.json"
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_package_ships_a_py_typed_marker() -> None:
|
|
42
|
+
# The package declares "Typing :: Typed" on the PyPI page. Without this
|
|
43
|
+
# marker a type checker ignores every annotation in the module, so the claim
|
|
44
|
+
# is false and a consumer running mypy gets "untyped import" from a library
|
|
45
|
+
# that is annotated throughout.
|
|
46
|
+
import vn_text
|
|
47
|
+
|
|
48
|
+
marker = Path(vn_text.__file__).resolve().parent / "py.typed"
|
|
49
|
+
assert marker.exists(), (
|
|
50
|
+
f"{marker} is missing. The wheel must contain it, or drop the "
|
|
51
|
+
f"Typing :: Typed classifier instead of advertising types it hides."
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def load_suite() -> dict[str, Any]:
|
|
56
|
+
with CONFORMANCE_PATH.open(encoding="utf-8") as fh:
|
|
57
|
+
return json.load(fh)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
#: Dispatch table keyed by the function names used in the conformance file.
|
|
61
|
+
#: These are the TypeScript names, so the mapping stays one-to-one.
|
|
62
|
+
IMPLS: dict[str, Callable[[str], str | bool]] = {
|
|
63
|
+
"normalize": normalize,
|
|
64
|
+
"deaccent": deaccent,
|
|
65
|
+
"stripStroke": strip_stroke,
|
|
66
|
+
"repairMojibake": repair_mojibake,
|
|
67
|
+
"fold": fold,
|
|
68
|
+
"isVietnamese": is_vietnamese,
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
SUITE = load_suite()
|
|
72
|
+
CASES = SUITE["cases"]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# --------------------------------------------------------------------------
|
|
76
|
+
# Metadata checks -- fail loudly if the suite itself drifts out of shape.
|
|
77
|
+
# --------------------------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def test_schema_is_understood() -> None:
|
|
81
|
+
assert SUITE["schema"] == "vn-text-conformance/1", SUITE["schema"]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_no_case_references_an_unknown_function() -> None:
|
|
85
|
+
unknown = [f"{c['id']} -> {c['fn']}" for c in CASES if c["fn"] not in IMPLS]
|
|
86
|
+
assert unknown == [], f"unknown functions in conformance file: {unknown}"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_case_ids_are_unique() -> None:
|
|
90
|
+
ids = [c["id"] for c in CASES]
|
|
91
|
+
assert len(set(ids)) == len(ids), "duplicate case ids in conformance file"
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def test_every_declared_function_has_cases() -> None:
|
|
95
|
+
for fn in SUITE["functions"]:
|
|
96
|
+
count = sum(1 for c in CASES if c["fn"] == fn)
|
|
97
|
+
assert count > 0, f'function "{fn}" has no conformance cases'
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
# --------------------------------------------------------------------------
|
|
101
|
+
# The suite itself.
|
|
102
|
+
# --------------------------------------------------------------------------
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def test_conformance_suite() -> None:
|
|
106
|
+
failures: list[str] = []
|
|
107
|
+
for case in CASES:
|
|
108
|
+
impl = IMPLS[case["fn"]]
|
|
109
|
+
actual = impl(case["input"])
|
|
110
|
+
if actual != case["expected"]:
|
|
111
|
+
failures.append(
|
|
112
|
+
f"{case['id']} {case['fn']}({case['input']!r})\n"
|
|
113
|
+
f" expected: {case['expected']!r}\n"
|
|
114
|
+
f" actual: {actual!r}"
|
|
115
|
+
)
|
|
116
|
+
assert not failures, "conformance failures:\n" + "\n".join(failures)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
# --------------------------------------------------------------------------
|
|
120
|
+
# Regression tests for the mistakes that make existing libraries wrong.
|
|
121
|
+
# Duplicated on purpose so the headline behaviour stays visible in the report.
|
|
122
|
+
# --------------------------------------------------------------------------
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def test_d_with_stroke_folds_to_d() -> None:
|
|
126
|
+
assert fold("Đặng Minh") == "dang minh"
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_eth_never_folds_to_d() -> None:
|
|
130
|
+
assert fold("Ðor") == "ðor"
|
|
131
|
+
assert deaccent("Ðor") == "Ðor"
|
|
132
|
+
assert strip_stroke("Ðor") == "Ðor"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def test_deaccent_preserves_letter_identity_but_fold_collapses_it() -> None:
|
|
136
|
+
# Stripping tone marks should not silently rewrite which letter you have.
|
|
137
|
+
assert deaccent("Đặng") == "Đang"
|
|
138
|
+
# A search key has to collapse it, or users cannot find "Đặng" by
|
|
139
|
+
# typing "Dang".
|
|
140
|
+
assert fold("Đặng") == "dang"
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def test_search_equivalence() -> None:
|
|
144
|
+
assert fold("Cai gi the nay") == fold("Cái gì thế này")
|
|
145
|
+
assert fold("Dang") == fold("Đặng")
|
|
146
|
+
assert fold("HA NOI") == fold("Hà Nội")
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_fold_is_idempotent() -> None:
|
|
150
|
+
for s in [
|
|
151
|
+
"Cái gì thế này",
|
|
152
|
+
"Đặng Minh",
|
|
153
|
+
"Hà Nội, Việt Nam!",
|
|
154
|
+
"Ðor",
|
|
155
|
+
"東京 Tokyo",
|
|
156
|
+
]:
|
|
157
|
+
assert fold(s) == fold(fold(s)), s
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_whitespace_and_punctuation_collapse_consistently() -> None:
|
|
161
|
+
assert fold(" Hà Nội ") == fold("Hà-Nội")
|
|
162
|
+
assert fold("TP. Hồ Chí Minh") == "tp ho chi minh"
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_nfd_and_nfc_spellings_agree() -> None:
|
|
166
|
+
nfc = "Cái gì thế này"
|
|
167
|
+
nfd = unicodedata.normalize("NFD", nfc)
|
|
168
|
+
assert nfc != nfd
|
|
169
|
+
assert normalize(nfd) == nfc
|
|
170
|
+
assert fold(nfc) == fold(nfd)
|
|
171
|
+
assert decompose(nfc) == nfd
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def test_uses_nfd_not_nfkd_so_compatibility_ligatures_survive() -> None:
|
|
175
|
+
# NFKD would rewrite this to "fi" and change English text.
|
|
176
|
+
ligature = chr(0xFB01)
|
|
177
|
+
assert deaccent(ligature) == ligature
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def test_repair_mojibake_is_idempotent() -> None:
|
|
181
|
+
for s in ["Ðảm baỏ", "Ðor", "Đặng", "", "Không có gì"]:
|
|
182
|
+
assert repair_mojibake(s) == repair_mojibake(repair_mojibake(s)), s
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def test_repair_then_fold_makes_damaged_text_findable() -> None:
|
|
186
|
+
# The whole point of the function: a record indexed after repair is
|
|
187
|
+
# reachable by the string a user would actually type.
|
|
188
|
+
assert fold(repair_mojibake("Ðặng Minh")) == fold("Đặng Minh") == "dang minh"
|
|
189
|
+
# Without the repair step the record is unreachable, and silently so.
|
|
190
|
+
assert fold("Ðặng Minh") != fold("Đặng Minh")
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def test_repair_and_fold_answer_different_questions() -> None:
|
|
194
|
+
# fold never reinterprets a letter; repair_mojibake assumes the corpus is
|
|
195
|
+
# Vietnamese and that ETH is damage. Both are correct, for different inputs.
|
|
196
|
+
assert fold("Ðor") == "ðor"
|
|
197
|
+
assert repair_mojibake("Ðor") == "Đor"
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
# --------------------------------------------------------------------------
|
|
201
|
+
# Zero-dependency runner, for when pytest is not available.
|
|
202
|
+
# --------------------------------------------------------------------------
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _run_standalone() -> int:
|
|
206
|
+
tests = [
|
|
207
|
+
(name, obj)
|
|
208
|
+
for name, obj in sorted(globals().items())
|
|
209
|
+
if name.startswith("test_") and callable(obj)
|
|
210
|
+
]
|
|
211
|
+
failed = 0
|
|
212
|
+
for name, fn in tests:
|
|
213
|
+
try:
|
|
214
|
+
fn()
|
|
215
|
+
except AssertionError as exc:
|
|
216
|
+
failed += 1
|
|
217
|
+
print(f"FAIL {name}\n{exc}\n")
|
|
218
|
+
else:
|
|
219
|
+
print(f"ok {name}")
|
|
220
|
+
print()
|
|
221
|
+
print(f"{len(tests) - failed}/{len(tests)} test functions passed")
|
|
222
|
+
print(f"{len(CASES)} conformance cases across {len(SUITE['functions'])} functions")
|
|
223
|
+
return 1 if failed else 0
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
if __name__ == "__main__":
|
|
227
|
+
raise SystemExit(_run_standalone())
|