vn-ident 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,32 @@
1
+ node_modules/
2
+ dist/
3
+ *.tsbuildinfo
4
+
5
+ # Python
6
+ __pycache__/
7
+ *.py[cod]
8
+ .venv/
9
+ venv/
10
+ *.egg-info/
11
+ .pytest_cache/
12
+ .mypy_cache/
13
+ .ruff_cache/
14
+
15
+ # Editors / OS
16
+ .DS_Store
17
+ Thumbs.db
18
+ .idea/
19
+ .vscode/
20
+
21
+ # Build output of the conformance tooling
22
+ .tmp/
23
+
24
+ # Traction snapshots. Machine-generated history for scripts/track-traction.mjs,
25
+ # not something to review in a diff.
26
+ .traction/
27
+
28
+ # LICENSE copies staged into each package at pack time by
29
+ # scripts/stage-license.mjs. One canonical file at the repository root; these
30
+ # are build output, and a staged one in a diff is a bug, not a licence update.
31
+ packages/*/LICENSE
32
+ packages/*/*/LICENSE
vn_ident-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 vn-toolkit contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,158 @@
1
+ Metadata-Version: 2.5
2
+ Name: vn-ident
3
+ Version: 0.1.0
4
+ Summary: Vietnamese tax codes, mobile numbers and card numbers, validated against the sources that define them. A tax code has a check digit; a card number does not, and this says so. Standard library only.
5
+ Project-URL: Homepage, https://github.com/leeloc1809/vn-toolkit
6
+ Project-URL: Repository, https://github.com/leeloc1809/vn-toolkit
7
+ Project-URL: Issues, https://github.com/leeloc1809/vn-toolkit/issues
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: carrier,cccd,i18n,id-card,mst,phone,tax-code,validation,vietnam,vietnamese
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Natural Language :: Vietnamese
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Software Development :: Localization
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.9
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest>=8.0; extra == 'dev'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # vn-ident
29
+
30
+ Vietnamese tax codes, mobile numbers and card numbers — validated against the
31
+ sources that define them, not against a regex somebody liked. Standard library
32
+ only, zero runtime dependencies.
33
+
34
+ ```bash
35
+ pip install vn-ident
36
+ ```
37
+
38
+ ```python
39
+ from vn_ident import is_valid_mst, mst_check_digit, normalize_phone, detect_carrier, parse_cccd, classify_id
40
+
41
+ is_valid_mst("0100047516") # True — the tenth digit is computed, not matched
42
+ mst_check_digit("010004751") # 6 — the worked example in the circular
43
+ mst_check_digit("000000000") # None — and None is the interesting part
44
+
45
+ normalize_phone("+84912345678") # '0912345678'
46
+ detect_carrier("0987654321") # 'viettel'
47
+ detect_carrier("0953456780") # None — a valid number that belongs to nobody
48
+
49
+ parse_cccd("001087123456") # CardNumber(province_code='001', province='Hà Nội',
50
+ # century=20, gender='male', birth_year=1987, …)
51
+ classify_id("0912345678") # 'phone'
52
+ ```
53
+
54
+ The TypeScript port is [`@vntoolkit/vn-ident`](https://www.npmjs.com/package/@vntoolkit/vn-ident)
55
+ and the two are held to the same
56
+ [conformance suite](../../conformance/vn-ident-1.0.0.json) — 258 cases, read by
57
+ both. That suite is the contract; this package is one implementation of it.
58
+
59
+ ## The thing to read first
60
+
61
+ **A tax code has a check digit. A card number does not.**
62
+
63
+ A tax code's tenth digit is computed from the first nine, so `is_valid_mst` tells
64
+ you the code is *right*. A card number's last six digits are random, so there is
65
+ nothing to verify: `is_valid_cccd` is a **format check and nothing more** —
66
+ twelve digits, a province code that exists, a fourth digit that is defined. It
67
+ cannot detect a number that is well-formed and wrong, and it will not pretend to.
68
+
69
+ It also cannot tell you which generation of card it is. The old unchipped
70
+ 12-digit number and the current chip number have the same structure, and the 2025
71
+ reduction to 34 provincial units did not renumber any card issued before it.
72
+ `classify_id` returning `"cccd"` does not mean a chip card.
73
+
74
+ ## The zero remainder
75
+
76
+ Ten minus zero is ten, which is not a digit, so the circular skips that sequence
77
+ number rather than assigning a check digit. That is why the function is typed
78
+ `int | None` and not `int`:
79
+
80
+ ```python
81
+ mst_check_digit("000000000") # None
82
+ mst_check_digit("000000007") # 0
83
+ ```
84
+
85
+ A naive implementation returns `10` there and then compares it against a
86
+ character. It rejects valid codes and never says why. No tax code whose first
87
+ nine digits sum to a multiple of 11 can be valid — and one that claims to be is
88
+ not.
89
+
90
+ ## The weights are not a progression
91
+
92
+ `31 29 23 19 17 13 7 5 3` is transcribed from Thông tư 105/2020/TT-BTC, Phụ
93
+ lục 1, and pinned in the suite. That is the reason the parity check generates ten
94
+ thousand tax codes rather than trusting the hand-picked cases: a transposed pair
95
+ of weights still produces a well-formed digit for every code, so nothing
96
+ downstream would ever notice.
97
+
98
+ ```text
99
+ ident parity: 673311/673311 operations identical across both ports
100
+ ```
101
+
102
+ Six port breakages are introduced on purpose to prove the check catches them —
103
+ two transposed weights, the divisible-by-11 case, the Bắc Giang `023`/`024` typo
104
+ that one of the four published province tables actually carries, two swapped
105
+ carrier prefixes, the trunk-zero inference, and a `\d` widened to Unicode:
106
+
107
+ ```bash
108
+ node scripts/prove-ident-parity-fails.mjs
109
+ # 6/6 real port breakages were caught
110
+ ```
111
+
112
+ ## Errors are values, not exceptions
113
+
114
+ Every function returns `False` or `None` on bad input, and every one accepts
115
+ anything at runtime. A validator that crashes on the wrong type has turned a
116
+ validation error into a 500, and form posts and JSON bodies do not respect your
117
+ type hints:
118
+
119
+ ```python
120
+ is_valid_mst(None) # False
121
+ parse_cccd(12345) # None
122
+ classify_id({"a": 1}) # 'unknown'
123
+ ```
124
+
125
+ The one exception is `mst_check_digit`, which raises `TypeError` on a prefix that
126
+ is not nine digits. That is a programming error in the caller rather than bad
127
+ user data, and it should not be a `None` you discover three lines later.
128
+
129
+ ## Reading an amount out of a card number
130
+
131
+ The fourth digit is the century and the sex in one character, so the two year
132
+ digits alone are ambiguous and it is what settles them:
133
+
134
+ ```python
135
+ parse_cccd("001085123456").birth_year # 1985
136
+ parse_cccd("001285123456").birth_year # 2085
137
+ ```
138
+
139
+ A year in the future parses, deliberately: with no check digit nothing can detect
140
+ it, and a date-dependent `valid` would make the conformance suite change its own
141
+ answer over time.
142
+
143
+ ## Known limitations
144
+
145
+ - **No card number verification is possible.** No check digit exists.
146
+ - **No province lookup for a tax code.** `region_code` comes from the tax
147
+ authority's catalogue, not the card-number table, and this package does not
148
+ carry a join between the two.
149
+ - **`detect_carrier` does not normalise.** `detect_carrier("+84912345678")` is
150
+ `None`. Call `normalize_phone` first.
151
+ - **`is_valid_cmnd` is a shape test.** Nine digits, no structure, no check digit.
152
+ It accepts `000000000`, which was never issued to anyone.
153
+ - **Non-ASCII digits are refused everywhere.** Both ports agree on that, and the
154
+ proof script checks that they do.
155
+
156
+ ## Licence
157
+
158
+ MIT
@@ -0,0 +1,131 @@
1
+ # vn-ident
2
+
3
+ Vietnamese tax codes, mobile numbers and card numbers — validated against the
4
+ sources that define them, not against a regex somebody liked. Standard library
5
+ only, zero runtime dependencies.
6
+
7
+ ```bash
8
+ pip install vn-ident
9
+ ```
10
+
11
+ ```python
12
+ from vn_ident import is_valid_mst, mst_check_digit, normalize_phone, detect_carrier, parse_cccd, classify_id
13
+
14
+ is_valid_mst("0100047516") # True — the tenth digit is computed, not matched
15
+ mst_check_digit("010004751") # 6 — the worked example in the circular
16
+ mst_check_digit("000000000") # None — and None is the interesting part
17
+
18
+ normalize_phone("+84912345678") # '0912345678'
19
+ detect_carrier("0987654321") # 'viettel'
20
+ detect_carrier("0953456780") # None — a valid number that belongs to nobody
21
+
22
+ parse_cccd("001087123456") # CardNumber(province_code='001', province='Hà Nội',
23
+ # century=20, gender='male', birth_year=1987, …)
24
+ classify_id("0912345678") # 'phone'
25
+ ```
26
+
27
+ The TypeScript port is [`@vntoolkit/vn-ident`](https://www.npmjs.com/package/@vntoolkit/vn-ident)
28
+ and the two are held to the same
29
+ [conformance suite](../../conformance/vn-ident-1.0.0.json) — 258 cases, read by
30
+ both. That suite is the contract; this package is one implementation of it.
31
+
32
+ ## The thing to read first
33
+
34
+ **A tax code has a check digit. A card number does not.**
35
+
36
+ A tax code's tenth digit is computed from the first nine, so `is_valid_mst` tells
37
+ you the code is *right*. A card number's last six digits are random, so there is
38
+ nothing to verify: `is_valid_cccd` is a **format check and nothing more** —
39
+ twelve digits, a province code that exists, a fourth digit that is defined. It
40
+ cannot detect a number that is well-formed and wrong, and it will not pretend to.
41
+
42
+ It also cannot tell you which generation of card it is. The old unchipped
43
+ 12-digit number and the current chip number have the same structure, and the 2025
44
+ reduction to 34 provincial units did not renumber any card issued before it.
45
+ `classify_id` returning `"cccd"` does not mean a chip card.
46
+
47
+ ## The zero remainder
48
+
49
+ Ten minus zero is ten, which is not a digit, so the circular skips that sequence
50
+ number rather than assigning a check digit. That is why the function is typed
51
+ `int | None` and not `int`:
52
+
53
+ ```python
54
+ mst_check_digit("000000000") # None
55
+ mst_check_digit("000000007") # 0
56
+ ```
57
+
58
+ A naive implementation returns `10` there and then compares it against a
59
+ character. It rejects valid codes and never says why. No tax code whose first
60
+ nine digits sum to a multiple of 11 can be valid — and one that claims to be is
61
+ not.
62
+
63
+ ## The weights are not a progression
64
+
65
+ `31 29 23 19 17 13 7 5 3` is transcribed from Thông tư 105/2020/TT-BTC, Phụ
66
+ lục 1, and pinned in the suite. That is the reason the parity check generates ten
67
+ thousand tax codes rather than trusting the hand-picked cases: a transposed pair
68
+ of weights still produces a well-formed digit for every code, so nothing
69
+ downstream would ever notice.
70
+
71
+ ```text
72
+ ident parity: 673311/673311 operations identical across both ports
73
+ ```
74
+
75
+ Six port breakages are introduced on purpose to prove the check catches them —
76
+ two transposed weights, the divisible-by-11 case, the Bắc Giang `023`/`024` typo
77
+ that one of the four published province tables actually carries, two swapped
78
+ carrier prefixes, the trunk-zero inference, and a `\d` widened to Unicode:
79
+
80
+ ```bash
81
+ node scripts/prove-ident-parity-fails.mjs
82
+ # 6/6 real port breakages were caught
83
+ ```
84
+
85
+ ## Errors are values, not exceptions
86
+
87
+ Every function returns `False` or `None` on bad input, and every one accepts
88
+ anything at runtime. A validator that crashes on the wrong type has turned a
89
+ validation error into a 500, and form posts and JSON bodies do not respect your
90
+ type hints:
91
+
92
+ ```python
93
+ is_valid_mst(None) # False
94
+ parse_cccd(12345) # None
95
+ classify_id({"a": 1}) # 'unknown'
96
+ ```
97
+
98
+ The one exception is `mst_check_digit`, which raises `TypeError` on a prefix that
99
+ is not nine digits. That is a programming error in the caller rather than bad
100
+ user data, and it should not be a `None` you discover three lines later.
101
+
102
+ ## Reading an amount out of a card number
103
+
104
+ The fourth digit is the century and the sex in one character, so the two year
105
+ digits alone are ambiguous and it is what settles them:
106
+
107
+ ```python
108
+ parse_cccd("001085123456").birth_year # 1985
109
+ parse_cccd("001285123456").birth_year # 2085
110
+ ```
111
+
112
+ A year in the future parses, deliberately: with no check digit nothing can detect
113
+ it, and a date-dependent `valid` would make the conformance suite change its own
114
+ answer over time.
115
+
116
+ ## Known limitations
117
+
118
+ - **No card number verification is possible.** No check digit exists.
119
+ - **No province lookup for a tax code.** `region_code` comes from the tax
120
+ authority's catalogue, not the card-number table, and this package does not
121
+ carry a join between the two.
122
+ - **`detect_carrier` does not normalise.** `detect_carrier("+84912345678")` is
123
+ `None`. Call `normalize_phone` first.
124
+ - **`is_valid_cmnd` is a shape test.** Nine digits, no structure, no check digit.
125
+ It accepts `000000000`, which was never issued to anyone.
126
+ - **Non-ASCII digits are refused everywhere.** Both ports agree on that, and the
127
+ proof script checks that they do.
128
+
129
+ ## Licence
130
+
131
+ MIT
@@ -0,0 +1,60 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.24"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "vn-ident"
7
+ version = "0.1.0"
8
+ description = "Vietnamese tax codes, mobile numbers and card numbers, validated against the sources that define them. A tax code has a check digit; a card number does not, and this says so. Standard library only."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ license-files = { paths = ["LICENSE"] }
13
+ keywords = [
14
+ "vietnamese",
15
+ "tax-code",
16
+ "mst",
17
+ "phone",
18
+ "carrier",
19
+ "id-card",
20
+ "cccd",
21
+ "validation",
22
+ "i18n",
23
+ "vietnam",
24
+ ]
25
+ classifiers = [
26
+ "Development Status :: 3 - Alpha",
27
+ "Intended Audience :: Developers",
28
+ "License :: OSI Approved :: MIT License",
29
+ "Natural Language :: Vietnamese",
30
+ "Programming Language :: Python :: 3",
31
+ "Programming Language :: Python :: 3.9",
32
+ "Programming Language :: Python :: 3.10",
33
+ "Programming Language :: Python :: 3.11",
34
+ "Programming Language :: Python :: 3.12",
35
+ "Programming Language :: Python :: 3.13",
36
+ "Topic :: Software Development :: Localization",
37
+ "Typing :: Typed",
38
+ ]
39
+
40
+ # Intentionally empty. A validation library that drags in dependencies is a
41
+ # dependency every consumer has to audit, and this one has to stay auditable
42
+ # itself.
43
+ dependencies = []
44
+
45
+ [project.optional-dependencies]
46
+ dev = ["pytest>=8.0"]
47
+
48
+ [project.urls]
49
+ Homepage = "https://github.com/leeloc1809/vn-toolkit"
50
+ Repository = "https://github.com/leeloc1809/vn-toolkit"
51
+ Issues = "https://github.com/leeloc1809/vn-toolkit/issues"
52
+
53
+ [tool.hatch.build.targets.wheel]
54
+ packages = ["src/vn_ident"]
55
+
56
+ [tool.hatch.build.targets.sdist]
57
+ include = ["src/vn_ident", "tests", "README.md", "LICENSE"]
58
+
59
+ [tool.pytest.ini_options]
60
+ testpaths = ["tests"]
@@ -0,0 +1,73 @@
1
+ """Vietnamese tax codes, mobile numbers and card numbers.
2
+
3
+ Standard library only, and a conformance suite shared with the TypeScript port
4
+ at conformance/vn-ident-1.0.0.json.
5
+
6
+ The one thing to read before using it: a tax code has a check digit and this
7
+ can tell you it is right. A card number does not, and this cannot. Both facts
8
+ are stated in the code, in the suite and in the README, because a validation
9
+ library that overstates what it checks is worse than one that checks less.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from vn_ident.identity import (
15
+ CardNumber,
16
+ IdentifierKind,
17
+ classify_id,
18
+ is_valid_cccd,
19
+ is_valid_cmnd,
20
+ parse_cccd,
21
+ )
22
+ from vn_ident.mst import TaxCode, is_valid_mst, mst_check_digit, parse_mst
23
+ from vn_ident.phone import detect_carrier, is_valid_phone, normalize_phone
24
+ from vn_ident.tables import (
25
+ ASSIGNED_PREFIX_COUNT,
26
+ CARRIERS,
27
+ CENTURY_GENDER,
28
+ MST_MODULUS,
29
+ MST_WEIGHTS,
30
+ PROVINCES,
31
+ Carrier,
32
+ Gender,
33
+ carrier_of_prefix,
34
+ century_of,
35
+ gender_of,
36
+ is_province_code,
37
+ province_name,
38
+ )
39
+
40
+ __all__ = [
41
+ "ASSIGNED_PREFIX_COUNT",
42
+ "CARRIERS",
43
+ "CARRIER_NAMES",
44
+ "CENTURY_GENDER",
45
+ "MST_MODULUS",
46
+ "MST_WEIGHTS",
47
+ "PROVINCES",
48
+ "CardNumber",
49
+ "Carrier",
50
+ "Gender",
51
+ "IdentifierKind",
52
+ "TaxCode",
53
+ "carrier_of_prefix",
54
+ "century_of",
55
+ "classify_id",
56
+ "detect_carrier",
57
+ "gender_of",
58
+ "is_province_code",
59
+ "is_valid_cccd",
60
+ "is_valid_cmnd",
61
+ "is_valid_mst",
62
+ "is_valid_phone",
63
+ "mst_check_digit",
64
+ "normalize_phone",
65
+ "parse_cccd",
66
+ "parse_mst",
67
+ "province_name",
68
+ ]
69
+
70
+ __version__ = "0.1.0"
71
+
72
+ #: The five carrier names, in the order the table lists them.
73
+ CARRIER_NAMES: tuple[Carrier, ...] = tuple(name for name, _ in CARRIERS)
@@ -0,0 +1,125 @@
1
+ """Card numbers, the old paper card, and telling them apart.
2
+
3
+ The honest limit of this module is stated in every function that hits it: a
4
+ card number has no check digit, so it can be checked for shape and not for
5
+ truth.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from typing import Literal, NamedTuple
12
+
13
+ from vn_ident.mst import is_valid_mst
14
+ from vn_ident.phone import is_valid_phone
15
+ from vn_ident.tables import Gender, century_of, gender_of, is_province_code, province_name
16
+
17
+ _CCCD_SHAPE = re.compile(r"[0-9]{12}")
18
+ _CMND_SHAPE = re.compile(r"[0-9]{9}")
19
+
20
+
21
+ def is_valid_cccd(code: str) -> bool:
22
+ """Is this a well-formed card number with a province code that exists?
23
+
24
+ **There is no check digit on a card number.** The last six digits are
25
+ random, so unlike a tax code there is nothing to verify: this cannot detect
26
+ a number that is well-formed and wrong, and pretending otherwise would be
27
+ the most expensive thing this package could do. A card number is a format
28
+ check and nothing more, and the conformance suite says the same.
29
+
30
+ **It also cannot tell you which generation of card it is.** The old
31
+ unchipped 12-digit number and the current chip number have the same
32
+ structure, the 2025 reduction to 34 provincial units did not renumber any
33
+ card, and no function here will claim to distinguish them. If you need to
34
+ know, you need the issuing authority.
35
+ """
36
+ if not isinstance(code, str) or not _CCCD_SHAPE.fullmatch(code):
37
+ return False
38
+ if not is_province_code(code[:3]):
39
+ return False
40
+ # Every fourth digit is defined, so this is a table lookup rather than a
41
+ # range check. It is written out anyway so that a future table which drops
42
+ # a digit fails here instead of silently yielding None.
43
+ return century_of(code[3]) is not None
44
+
45
+
46
+ class CardNumber(NamedTuple):
47
+ """The parts of a card number.
48
+
49
+ ``birth_year`` uses the century digit rather than the two digits alone:
50
+ 85 in the fourth digit 0 is 1985, and in the fourth digit 2 it is 2085.
51
+ That is the whole reason the fourth digit exists.
52
+ """
53
+
54
+ province_code: str
55
+ province: str
56
+ century: int
57
+ gender: Gender
58
+ birth_year: int
59
+ serial: str
60
+
61
+
62
+ def parse_cccd(code: str) -> CardNumber | None:
63
+ """The parts of a card number, or ``None`` if it is not a well-formed one.
64
+
65
+ A birth year in the future parses. There is no check digit, so nothing here
66
+ can detect it, and a date-dependent "valid" would make the conformance
67
+ suite change its own answer over time.
68
+ """
69
+ if not is_valid_cccd(code):
70
+ return None
71
+ century = century_of(code[3])
72
+ gender = gender_of(code[3])
73
+ province = province_name(code[:3])
74
+ if century is None or gender is None or province is None:
75
+ return None
76
+ return CardNumber(
77
+ province_code=code[:3],
78
+ province=province,
79
+ century=century,
80
+ gender=gender,
81
+ birth_year=(century - 1) * 100 + int(code[4:6]),
82
+ serial=code[6:],
83
+ )
84
+
85
+
86
+ def is_valid_cmnd(code: str) -> bool:
87
+ """The old paper card: nine digits.
88
+
89
+ No structure to check and no check digit, so this is a shape test. All
90
+ zeros is well-formed even though no such card was ever issued, and that is
91
+ the honest answer: this function cannot tell you more than that.
92
+ """
93
+ return isinstance(code, str) and bool(_CMND_SHAPE.fullmatch(code))
94
+
95
+
96
+ IdentifierKind = Literal["cccd", "cmnd", "mst", "phone", "unknown"]
97
+
98
+
99
+ def classify_id(value: str) -> IdentifierKind:
100
+ """What is this string, as far as structure goes?
101
+
102
+ Ordered so the specific reading wins: a twelve-digit value with a known
103
+ province code is a card, a ten- or thirteen-digit value is a tax code, a
104
+ ten-digit value starting 0 is a phone number, and nine digits is the old
105
+ paper card.
106
+
107
+ Note what this does **not** say. ``"cccd"`` does not mean a chip card, and
108
+ ``"phone"`` does not mean the number is subscribed to anyone. A result of
109
+ ``"unknown"`` means the structure did not match anything here, which is a
110
+ statement about this package and not about the value.
111
+ """
112
+ if not isinstance(value, str):
113
+ return "unknown"
114
+ s = value.strip()
115
+ if not s:
116
+ return "unknown"
117
+ if is_valid_cccd(s):
118
+ return "cccd"
119
+ if is_valid_mst(s):
120
+ return "mst"
121
+ if is_valid_phone(s):
122
+ return "phone"
123
+ if is_valid_cmnd(s):
124
+ return "cmnd"
125
+ return "unknown"
@@ -0,0 +1,107 @@
1
+ """Tax codes: validate them and take them apart.
2
+
3
+ The tenth digit is computed, not pattern-matched, so a code is either right or
4
+ it is refused. There is one case in the algorithm that a naive implementation
5
+ gets wrong and that this module is explicit about: the remainder being zero.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from typing import NamedTuple
12
+
13
+ from vn_ident.tables import MST_MODULUS, MST_WEIGHTS
14
+
15
+ _DIGITS = re.compile(r"[0-9]+")
16
+ _PREFIX = re.compile(r"[0-9]{9}")
17
+ _CC = re.compile(r"[0-9]{12}")
18
+
19
+
20
+ def _bare(code: str) -> str:
21
+ """Drop the hyphen the circular writes into the 13-digit form."""
22
+ return code.replace("-", "")
23
+
24
+
25
+ def mst_check_digit(nine: str) -> int | None:
26
+ """The tenth digit of a tax code, from its first nine.
27
+
28
+ Thông tư 105/2020/TT-BTC, Phụ lục 1: multiply the nine digits by the nine
29
+ weights, add, divide by 11, and take ten minus the remainder.
30
+
31
+ **The remainder being zero is a case, not an edge case.** Ten minus zero is
32
+ ten, which is not a digit, so the circular skips that sequence number rather
33
+ than assigning a check digit. This returns ``None`` for it, which means no
34
+ tax code whose first nine digits sum to a multiple of 11 can be valid --
35
+ and one that claims to be is not. A naive implementation returns 10 here and
36
+ then compares it against a character, so it rejects a valid code and never
37
+ says why.
38
+
39
+ >>> mst_check_digit("010004751") # the worked example in the circular
40
+ 6
41
+ >>> mst_check_digit("000000000") is None
42
+ True
43
+ """
44
+ if not isinstance(nine, str) or not _PREFIX.fullmatch(nine):
45
+ raise TypeError(f"a tax code prefix is nine digits, got {nine!r}")
46
+
47
+ total = sum(int(digit) * weight for digit, weight in zip(nine, MST_WEIGHTS))
48
+ remainder = total % MST_MODULUS
49
+ # The circular says it in words: ten minus the remainder.
50
+ return None if remainder == 0 else 10 - remainder
51
+
52
+
53
+ def is_valid_mst(code: str) -> bool:
54
+ """Is this a tax code with a check digit that matches?
55
+
56
+ Ten digits for an independent entity, thirteen for a dependent unit, and
57
+ the hyphen between them is accepted because that is how the circular prints
58
+ it. A thirteen-digit code's last three run from 001 to 999, so a branch of
59
+ 000 does not exist and accepting it would let a typo through as a real
60
+ entity.
61
+
62
+ Whitespace is refused rather than trimmed. A paste that arrives with a
63
+ trailing space is a visible error, not a silent success that hides a broken
64
+ integration.
65
+ """
66
+ if not isinstance(code, str):
67
+ return False
68
+ s = _bare(code)
69
+ if not _DIGITS.fullmatch(s):
70
+ return False
71
+ if len(s) not in (10, 13):
72
+ return False
73
+ if len(s) == 13 and not 1 <= int(s[10:]) <= 999:
74
+ return False
75
+ return mst_check_digit(s[:9]) == int(s[9])
76
+
77
+
78
+ class TaxCode(NamedTuple):
79
+ """The parts of a tax code.
80
+
81
+ ``region_code`` is **not** a province. These two digits are the revenue
82
+ code of the provincial tax office, from the Ministry of Finance's own
83
+ catalogue. It is a *different list* from the province codes on a card
84
+ number, and calling this one a province is how two catalogues end up merged
85
+ in somebody's database.
86
+
87
+ ``serial`` is a string because the leading zeros are part of it: 0004751 and
88
+ 4751 are the same serial and this is the representation that says so.
89
+ """
90
+
91
+ region_code: str
92
+ serial: str
93
+ check_digit: int
94
+ branch: str | None
95
+
96
+
97
+ def parse_mst(code: str) -> TaxCode | None:
98
+ """The parts of a tax code, or ``None`` if it is not one."""
99
+ if not is_valid_mst(code):
100
+ return None
101
+ s = _bare(code)
102
+ return TaxCode(
103
+ region_code=s[:2],
104
+ serial=s[2:9],
105
+ check_digit=int(s[9]),
106
+ branch=s[10:] if len(s) == 13 else None,
107
+ )
@@ -0,0 +1,85 @@
1
+ """Mobile numbers: reduce what a person typed, and say whose it is.
2
+
3
+ Everything here is unambiguous except one switch, which exists because a typo
4
+ and a correct number look exactly the same at the boundary.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import re
10
+
11
+ from vn_ident.tables import Carrier, carrier_of_prefix
12
+
13
+ _TRUNK = "0"
14
+ _COUNTRY = "84"
15
+
16
+ #: Already normalised: ten digits, starting with the trunk zero.
17
+ _NATIONAL_SHAPE = re.compile(r"0[0-9]{9}")
18
+ _NON_DIGITS = re.compile(r"\D")
19
+
20
+
21
+ def normalize_phone(text: str, *, assume_trunk_zero: bool = True) -> str | None:
22
+ """Reduce what a person typed to a ten-digit national number, or ``None``.
23
+
24
+ The ``+84`` forms all say the same thing; spaces, dashes, dots and brackets
25
+ are how people write a number down rather than information; and a number
26
+ that is 8, 11 or 12 digits with no country code is refused rather than
27
+ reshaped.
28
+
29
+ ``assume_trunk_zero`` is the one inference in this package, and the one
30
+ place a typo and a correct number look the same. A national number written
31
+ without its leading zero is how it appears in a contact form and how it is
32
+ read aloud, so the default recovers it. Pass ``assume_trunk_zero=False`` to
33
+ refuse instead, which is the right choice if the input came from a system
34
+ that always includes it.
35
+
36
+ >>> normalize_phone("+84912345678")
37
+ '0912345678'
38
+ >>> normalize_phone("0912 345 678")
39
+ '0912345678'
40
+ """
41
+ if not isinstance(text, str):
42
+ return None
43
+
44
+ s = _NON_DIGITS.sub("", text)
45
+ if not s:
46
+ return None
47
+
48
+ rest = s[len(_COUNTRY) :] if s.startswith(_COUNTRY) else s
49
+
50
+ if len(rest) == 9 and not rest.startswith(_TRUNK):
51
+ if not assume_trunk_zero:
52
+ return None
53
+ rest = _TRUNK + rest
54
+
55
+ return rest if _NATIONAL_SHAPE.fullmatch(rest) else None
56
+
57
+
58
+ def detect_carrier(phone: str) -> Carrier | None:
59
+ """Which mobile network is this number on, or ``None``.
60
+
61
+ Only a normalised number is looked up, on purpose. A nine-digit number has
62
+ no carrier until the trunk zero is known, and guessing is how a phone
63
+ number ends up attributed to the wrong network -- which then decides which
64
+ carrier portal the user is sent to.
65
+
66
+ ``None`` covers three different situations, and they are not the same: a
67
+ number that is not a number, a number whose prefix belongs to nobody, and a
68
+ ten-digit number that is not a mobile number at all. Call
69
+ :func:`normalize_phone` first if you need to tell them apart.
70
+ """
71
+ if not isinstance(phone, str):
72
+ return None
73
+ s = _NON_DIGITS.sub("", phone)
74
+ if not _NATIONAL_SHAPE.fullmatch(s):
75
+ return None
76
+ return carrier_of_prefix(s[:3])
77
+
78
+
79
+ def is_valid_phone(text: str) -> bool:
80
+ """Is this something a person could have meant as a Vietnamese phone number?
81
+
82
+ True for a ten-digit number that no carrier owns, which is correct: the
83
+ shape is what is being checked, not the subscription.
84
+ """
85
+ return normalize_phone(text) is not None
File without changes
@@ -0,0 +1,181 @@
1
+ """The tables, and where each one came from.
2
+
3
+ There is no platform to read these out of the way there is for collation and
4
+ currency, so they are transcribed once, here, and the conformance suite records
5
+ the source of each. The suite's job is to make a transcription error visible: a
6
+ transposed pair of tax weights still produces a well-formed check digit, so it
7
+ would validate real tax codes and reject none.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from typing import Literal
13
+
14
+ # --- Tax codes -------------------------------------------------------------
15
+
16
+ #: Thông tư 105/2020/TT-BTC Điều 5 and Phụ lục 1, from the official VBQPPL
17
+ #: portal at moj.gov.vn.
18
+ #:
19
+ #: Nine weights for the nine digits before the check digit. The sums are
20
+ #: deliberately not a neat progression, which is exactly why they have to be
21
+ #: transcribed rather than computed.
22
+ MST_WEIGHTS = (31, 29, 23, 19, 17, 13, 7, 5, 3)
23
+
24
+ #: The circular divides the weighted sum by this.
25
+ MST_MODULUS = 11
26
+
27
+ # --- Province codes --------------------------------------------------------
28
+
29
+ #: The 63 province codes on a card number, from Quyết định 124/2004/QĐ-TTg,
30
+ #: cross-checked against chinhphu.gov.vn, thuvienphapluat.vn and vietnamnet.vn.
31
+ #:
32
+ #: One of those four prints Bắc Giang as 023. The other three, and the decree
33
+ #: itself, say 024, so 024 is what is here. A single-source table would have
34
+ #: carried the typo.
35
+ #:
36
+ #: The codes run 001 to 096 with gaps, and a gap is not an assignment: 003 is
37
+ #: nobody, and a card number starting 003 is not a card number.
38
+ PROVINCES: tuple[tuple[str, str], ...] = (
39
+ ("001", "Hà Nội"),
40
+ ("002", "Hà Giang"),
41
+ ("004", "Cao Bằng"),
42
+ ("006", "Bắc Kạn"),
43
+ ("008", "Tuyên Quang"),
44
+ ("010", "Lào Cai"),
45
+ ("011", "Điện Biên"),
46
+ ("012", "Lai Châu"),
47
+ ("014", "Sơn La"),
48
+ ("015", "Yên Bái"),
49
+ ("017", "Hòa Bình"),
50
+ ("019", "Thái Nguyên"),
51
+ ("020", "Lạng Sơn"),
52
+ ("022", "Quảng Ninh"),
53
+ ("024", "Bắc Giang"),
54
+ ("025", "Phú Thọ"),
55
+ ("026", "Vĩnh Phúc"),
56
+ ("027", "Bắc Ninh"),
57
+ ("030", "Hải Dương"),
58
+ ("031", "Hải Phòng"),
59
+ ("033", "Hưng Yên"),
60
+ ("034", "Thái Bình"),
61
+ ("035", "Hà Nam"),
62
+ ("036", "Nam Định"),
63
+ ("037", "Ninh Bình"),
64
+ ("038", "Thanh Hóa"),
65
+ ("040", "Nghệ An"),
66
+ ("042", "Hà Tĩnh"),
67
+ ("044", "Quảng Bình"),
68
+ ("045", "Quảng Trị"),
69
+ ("046", "Thừa Thiên Huế"),
70
+ ("048", "Đà Nẵng"),
71
+ ("049", "Quảng Nam"),
72
+ ("051", "Quảng Ngãi"),
73
+ ("052", "Bình Định"),
74
+ ("054", "Phú Yên"),
75
+ ("056", "Khánh Hòa"),
76
+ ("058", "Ninh Thuận"),
77
+ ("060", "Bình Thuận"),
78
+ ("062", "Kon Tum"),
79
+ ("064", "Gia Lai"),
80
+ ("066", "Đắk Lắk"),
81
+ ("067", "Đắk Nông"),
82
+ ("068", "Lâm Đồng"),
83
+ ("070", "Bình Phước"),
84
+ ("072", "Tây Ninh"),
85
+ ("074", "Bình Dương"),
86
+ ("075", "Đồng Nai"),
87
+ ("077", "Bà Rịa - Vũng Tàu"),
88
+ ("079", "Hồ Chí Minh"),
89
+ ("080", "Long An"),
90
+ ("082", "Tiền Giang"),
91
+ ("083", "Bến Tre"),
92
+ ("084", "Trà Vinh"),
93
+ ("086", "Vĩnh Long"),
94
+ ("087", "Đồng Tháp"),
95
+ ("089", "An Giang"),
96
+ ("091", "Kiên Giang"),
97
+ ("092", "Cần Thơ"),
98
+ ("093", "Hậu Giang"),
99
+ ("094", "Sóc Trăng"),
100
+ ("095", "Bạc Liêu"),
101
+ ("096", "Cà Mau"),
102
+ )
103
+
104
+ Gender = Literal["male", "female"]
105
+ Carrier = Literal["viettel", "vinaphone", "mobifone", "vietnamobile", "gmobile"]
106
+
107
+ #: The fourth digit: which century, and which sex.
108
+ #:
109
+ #: The pairs run in order, so even digits are male and odd are female, two per
110
+ #: century, from the twentieth to the twenty-fourth.
111
+ CENTURY_GENDER: tuple[tuple[str, int, Gender], ...] = (
112
+ ("0", 20, "male"),
113
+ ("1", 20, "female"),
114
+ ("2", 21, "male"),
115
+ ("3", 21, "female"),
116
+ ("4", 22, "male"),
117
+ ("5", 22, "female"),
118
+ ("6", 23, "male"),
119
+ ("7", 23, "female"),
120
+ ("8", 24, "male"),
121
+ ("9", 24, "female"),
122
+ )
123
+
124
+ # --- Mobile carriers -------------------------------------------------------
125
+
126
+ #: Published three-digit mobile prefixes, 34 in total.
127
+ #:
128
+ #: 34 is not 100, and the gap is the point: 030, 040, 050, 053, 060, 071, 080,
129
+ #: 087 and 095 belong to nobody. A well-formed ten-digit number can therefore
130
+ #: be valid and still have no carrier, and saying "unknown" is a different
131
+ #: answer from saying "invalid".
132
+ CARRIERS: tuple[tuple[Carrier, tuple[str, ...]], ...] = (
133
+ ("viettel", ("032", "033", "034", "035", "036", "037", "038", "039", "086", "096", "097", "098")),
134
+ ("vinaphone", ("081", "082", "083", "084", "085", "088", "091", "094")),
135
+ ("mobifone", ("070", "076", "077", "078", "079", "089", "090", "093")),
136
+ ("vietnamobile", ("052", "056", "058", "092")),
137
+ ("gmobile", ("059", "099")),
138
+ )
139
+
140
+ PROVINCE_NAMES: dict[str, str] = dict(PROVINCES)
141
+ CENTURY_BY_DIGIT: dict[str, int] = {d: c for d, c, _ in CENTURY_GENDER}
142
+ GENDER_BY_DIGIT: dict[str, Gender] = {d: g for d, _, g in CENTURY_GENDER}
143
+
144
+
145
+ def _build_prefix_map() -> dict[str, Carrier]:
146
+ out: dict[str, Carrier] = {}
147
+ for name, prefixes in CARRIERS:
148
+ for prefix in prefixes:
149
+ # A prefix assigned to two carriers is a transcription error, and it
150
+ # would otherwise resolve to whichever table was read first. The
151
+ # conformance generator checks this too, but failing here means it
152
+ # cannot ship.
153
+ if prefix in out:
154
+ raise ValueError(f"prefix {prefix} is assigned to two carriers")
155
+ out[prefix] = name
156
+ return out
157
+
158
+
159
+ PREFIX_TO_CARRIER: dict[str, Carrier] = _build_prefix_map()
160
+
161
+ ASSIGNED_PREFIX_COUNT = len(PREFIX_TO_CARRIER)
162
+
163
+
164
+ def province_name(code: str) -> str | None:
165
+ return PROVINCE_NAMES.get(code)
166
+
167
+
168
+ def is_province_code(code: str) -> bool:
169
+ return code in PROVINCE_NAMES
170
+
171
+
172
+ def century_of(digit: str) -> int | None:
173
+ return CENTURY_BY_DIGIT.get(digit)
174
+
175
+
176
+ def gender_of(digit: str) -> Gender | None:
177
+ return GENDER_BY_DIGIT.get(digit)
178
+
179
+
180
+ def carrier_of_prefix(prefix: str) -> Carrier | None:
181
+ return PREFIX_TO_CARRIER.get(prefix)
@@ -0,0 +1,338 @@
1
+ """Run the shared identity conformance suite against the Python port.
2
+
3
+ Runnable two ways, on purpose:
4
+
5
+ pytest packages/vn-ident-py/tests -q
6
+ python packages/vn-ident-py/tests/test_vn_ident.py
7
+
8
+ The second form needs nothing installed. The conformance suite is the contract
9
+ between the two ports, and being able to check it with nothing but a Python
10
+ interpreter is what stops them drifting apart.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import sys
17
+ from pathlib import Path
18
+ from typing import Any, Callable
19
+
20
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
21
+
22
+ from vn_ident import ( # noqa: E402 (path set up above)
23
+ ASSIGNED_PREFIX_COUNT,
24
+ CARRIERS,
25
+ CENTURY_GENDER,
26
+ MST_MODULUS,
27
+ MST_WEIGHTS,
28
+ PROVINCES,
29
+ classify_id,
30
+ detect_carrier,
31
+ is_valid_cccd,
32
+ is_valid_cmnd,
33
+ is_valid_mst,
34
+ is_valid_phone,
35
+ mst_check_digit,
36
+ normalize_phone,
37
+ parse_cccd,
38
+ parse_mst,
39
+ )
40
+
41
+ CONFORMANCE_PATH = (
42
+ Path(__file__).resolve().parents[3] / "conformance" / "vn-ident-1.0.0.json"
43
+ )
44
+
45
+
46
+ def load_suite() -> dict[str, Any]:
47
+ with CONFORMANCE_PATH.open(encoding="utf-8") as fh:
48
+ return json.load(fh)
49
+
50
+
51
+ SUITE = load_suite()
52
+ CASES = SUITE["cases"]
53
+ TABLES = SUITE["tables"]
54
+
55
+
56
+ def _as_tuple(value: Any) -> tuple[Any, ...]:
57
+ """The JSON suite gives lists; the port gives tuples.
58
+
59
+ Compared as lists, because that is what the file contains and what the
60
+ TypeScript port sees. A runner that quietly normalised here would hide a
61
+ real difference in the table order.
62
+ """
63
+ return value if isinstance(value, list) else list(value)
64
+
65
+
66
+ RUNNERS: dict[str, Callable[[dict[str, Any]], Any]] = {
67
+ "mstCheckDigit": lambda d: mst_check_digit(d["prefix"]),
68
+ "isValidMst": lambda d: is_valid_mst(d["code"]),
69
+ "parseMst": lambda d: (
70
+ None
71
+ if parse_mst(d["code"]) is None
72
+ else {
73
+ "regionCode": parse_mst(d["code"]).region_code,
74
+ "serial": parse_mst(d["code"]).serial,
75
+ "checkDigit": parse_mst(d["code"]).check_digit,
76
+ "branch": parse_mst(d["code"]).branch,
77
+ }
78
+ ),
79
+ "normalizePhone": lambda d: normalize_phone(
80
+ d["input"], assume_trunk_zero=(d.get("options") or {}).get("assumeTrunkZero", True)
81
+ ),
82
+ "detectCarrier": lambda d: detect_carrier(d["phone"]),
83
+ "isValidPhone": lambda d: is_valid_phone(d["input"]),
84
+ "isValidCccd": lambda d: is_valid_cccd(d["code"]),
85
+ "parseCccd": lambda d: (
86
+ None
87
+ if parse_cccd(d["code"]) is None
88
+ else {
89
+ "provinceCode": parse_cccd(d["code"]).province_code,
90
+ "province": parse_cccd(d["code"]).province,
91
+ "century": parse_cccd(d["code"]).century,
92
+ "gender": parse_cccd(d["code"]).gender,
93
+ "birthYear": parse_cccd(d["code"]).birth_year,
94
+ "serial": parse_cccd(d["code"]).serial,
95
+ }
96
+ ),
97
+ "isValidCmnd": lambda d: is_valid_cmnd(d["code"]),
98
+ "classifyId": lambda d: classify_id(d["value"]),
99
+ }
100
+
101
+
102
+ def test_schema_is_understood() -> None:
103
+ assert SUITE["schema"] == "vn-ident-conformance/1", SUITE["schema"]
104
+
105
+
106
+ def test_package_ships_a_py_typed_marker() -> None:
107
+ # The package declares "Typing :: Typed" on the PyPI page. Without this
108
+ # marker a type checker ignores every annotation in the module, so the claim
109
+ # is false and a consumer running mypy gets "untyped import" from a library
110
+ # that is annotated throughout.
111
+ import vn_ident
112
+
113
+ marker = Path(vn_ident.__file__).resolve().parent / "py.typed"
114
+ assert marker.exists(), (
115
+ f"{marker} is missing. The wheel must contain it, or drop the "
116
+ f"Typing :: Typed classifier instead of advertising types it hides."
117
+ )
118
+
119
+
120
+ def test_suite_records_every_source() -> None:
121
+ # Four sources, not one. A single-source transcription of a weight table or
122
+ # a province list is a typo waiting to be shipped.
123
+ derived = SUITE["derivedFrom"]
124
+ assert "105/2020/TT-BTC" in derived, derived
125
+ assert "124/2004" in derived, derived
126
+ assert "cross-checked" in derived, derived
127
+
128
+
129
+ def test_case_ids_are_unique() -> None:
130
+ ids = [c["id"] for c in CASES]
131
+ assert len(set(ids)) == len(ids), "duplicate case ids in conformance file"
132
+
133
+
134
+ def test_every_function_has_cases() -> None:
135
+ for fn in SUITE["functions"]:
136
+ count = sum(1 for c in CASES if c["fn"] == fn)
137
+ assert count > 0, f"no cases for {fn}"
138
+ assert count == SUITE["caseCounts"][fn], (
139
+ f"{fn}: caseCounts says {SUITE['caseCounts'][fn]} but there are {count}"
140
+ )
141
+
142
+
143
+ def test_tables_match_the_suite() -> None:
144
+ # The tables are the contract. A regeneration cannot quietly reorder them
145
+ # without showing up here.
146
+ assert _as_tuple(MST_WEIGHTS) == TABLES["mstWeights"]
147
+ assert MST_MODULUS == TABLES["mstModulus"]
148
+ assert len(MST_WEIGHTS) == 9
149
+
150
+ assert len(PROVINCES) == 63
151
+ assert len(TABLES["provinces"]) == 63
152
+ codes = [code for code, _ in PROVINCES]
153
+ assert len(set(codes)) == len(codes), "duplicate province codes"
154
+ names = [name for _, name in PROVINCES]
155
+ assert len(set(names)) == len(names), "duplicate province names"
156
+
157
+ assert len(CENTURY_GENDER) == 10
158
+ assert sorted(int(d) for d, _, _ in CENTURY_GENDER) == list(range(10))
159
+
160
+ all_prefixes = [p for _, prefixes in CARRIERS for p in prefixes]
161
+ assert len(all_prefixes) == 34
162
+ assert len(set(all_prefixes)) == 34, "a prefix is assigned to two carriers"
163
+ assert ASSIGNED_PREFIX_COUNT == 34
164
+ assert TABLES["carrierCount"] == 34
165
+
166
+
167
+ def test_conformance_suite() -> None:
168
+ failures: list[str] = []
169
+ for case in CASES:
170
+ try:
171
+ actual = RUNNERS[case["fn"]](case["input"])
172
+ except (TypeError, ValueError) as exc:
173
+ actual = {"error": type(exc).__name__}
174
+ if actual != case["expected"]:
175
+ failures.append(
176
+ f"{case['id']} {case['fn']}({case['input']!r})\n"
177
+ f" expected: {case['expected']!r}\n"
178
+ f" actual: {actual!r}\n"
179
+ f" note: {case.get('note', '')}"
180
+ )
181
+ assert not failures, f"{len(failures)} conformance failures:\n" + "\n".join(failures[:10])
182
+
183
+
184
+ # --- Invariants -------------------------------------------------------------
185
+
186
+
187
+ def test_reproduces_the_worksed_example_from_the_circular() -> None:
188
+ # The one anchor that comes from outside this repository. Everything else
189
+ # about the tax rules is a transcription; this is the transcription being
190
+ # checked against its source.
191
+ assert mst_check_digit("010004751") == 6
192
+ assert is_valid_mst("0100047516") is True
193
+
194
+
195
+ def test_zero_remainder_gives_none_not_ten() -> None:
196
+ # 10 - 0 is 10, which is not a digit, so the circular skips the sequence
197
+ # number. A naive implementation returns 10 and then compares it against a
198
+ # character, so it rejects valid codes without ever saying why.
199
+ assert mst_check_digit("000000000") is None
200
+ assert is_valid_mst("0000000000") is False
201
+ for d in range(10):
202
+ assert is_valid_mst(f"000000000{d}") is False, f"check digit {d}"
203
+
204
+
205
+ def test_every_check_digit_value_is_reachable() -> None:
206
+ seen = set()
207
+ for n in range(40000):
208
+ check = mst_check_digit(str(n).zfill(9))
209
+ if check is not None:
210
+ seen.add(check)
211
+ if len(seen) == 10:
212
+ break
213
+ assert sorted(seen) == list(range(10))
214
+
215
+
216
+ def test_exactly_one_tenth_digit_is_accepted() -> None:
217
+ # The property the whole thing rests on: change any digit and the code is
218
+ # refused. If two digits passed, the check digit would not be checking.
219
+ for prefix in ("010004751", "030475101", "790000000", "012345678"):
220
+ accepted = [d for d in range(10) if is_valid_mst(f"{prefix}{d}")]
221
+ assert accepted == [mst_check_digit(prefix)], prefix
222
+
223
+
224
+ def test_whitespace_and_separators_are_refused_not_trimmed() -> None:
225
+ assert is_valid_mst("0100047516") is True
226
+ assert is_valid_mst("0100047516 ") is False
227
+ assert is_valid_mst(" 0100047516") is False
228
+ assert is_valid_mst("01.00047516") is False
229
+ # The hyphen the circular prints is the one exception, in the right place.
230
+ assert is_valid_mst("0100047516-001") is True
231
+
232
+
233
+ def test_a_non_string_is_refused_rather_than_crashing() -> None:
234
+ # The TypeScript port takes `unknown` here. A Python caller passing None or
235
+ # an int must get the same answer, not a traceback.
236
+ for value in (None, 12345, [], {}, 0):
237
+ assert is_valid_mst(value) is False, repr(value)
238
+ assert is_valid_cccd(value) is False, repr(value)
239
+ assert is_valid_cmnd(value) is False, repr(value)
240
+ assert is_valid_phone(value) is False, repr(value)
241
+ assert normalize_phone(value) is None, repr(value)
242
+ assert detect_carrier(value) is None, repr(value)
243
+ assert parse_mst(value) is None, repr(value)
244
+ assert parse_cccd(value) is None, repr(value)
245
+ assert classify_id(value) == "unknown", repr(value)
246
+
247
+
248
+ def test_every_province_code_parses_to_its_own_name() -> None:
249
+ for code, name in PROVINCES:
250
+ parsed = parse_cccd(f"{code}098512345")
251
+ assert parsed is not None, code
252
+ assert parsed.province == name, code
253
+ assert parsed.province_code == code
254
+
255
+
256
+ def test_the_century_digit_is_what_makes_a_birth_year() -> None:
257
+ # 85 in the fourth digit 0 is 1985; in the fourth digit 2 it is 2085. This
258
+ # is the whole reason the fourth digit exists.
259
+ assert parse_cccd("001085123456").birth_year == 1985
260
+ assert parse_cccd("001285123456").birth_year == 2085
261
+ assert parse_cccd("001285123456").century == 21
262
+
263
+
264
+ def test_a_future_year_still_parses() -> None:
265
+ # There is no check digit, so nothing can detect it, and a date-dependent
266
+ # "valid" would make the conformance suite change its own answer over time.
267
+ # The fourth digit 8 is the twenty-fourth century, so 99 there is 2399.
268
+ assert parse_cccd("001899999999").birth_year == 2399
269
+ assert parse_cccd("001000000000").birth_year == 1900
270
+
271
+
272
+ def test_phone_normalisation_is_one_number_many_spellings() -> None:
273
+ expected = "0912345678"
274
+ for text in (
275
+ "0912345678", "+84912345678", "84912345678", "840912345678",
276
+ "0912 345 678", "0912-345-678", "0912.345.678", "(0912) 345678",
277
+ " 0912345678 ",
278
+ ):
279
+ assert normalize_phone(text) == expected, text
280
+
281
+
282
+ def test_the_one_inference_is_switchable() -> None:
283
+ assert normalize_phone("912345678") == "0912345678"
284
+ assert normalize_phone("912345678", assume_trunk_zero=False) is None
285
+
286
+
287
+ def test_a_valid_number_can_belong_to_nobody() -> None:
288
+ # 34 of the 100 three-digit prefixes are assigned. "Unassigned" is a
289
+ # different answer from "invalid", and this is where the two get confused.
290
+ assert is_valid_phone("0953456780") is True
291
+ assert detect_carrier("0953456780") is None
292
+ assert is_valid_phone("0800123456") is True
293
+ assert detect_carrier("0800123456") is None
294
+
295
+
296
+ def test_detect_carrier_does_not_normalise() -> None:
297
+ # Guessing the trunk zero is how a number ends up attributed to the wrong
298
+ # network, which then decides which carrier portal the user is sent to.
299
+ assert detect_carrier("912345678") is None
300
+ assert detect_carrier("+84912345678") is None
301
+
302
+
303
+ def test_classify_id_sorts_out_the_four_shapes() -> None:
304
+ assert classify_id("001098512345") == "cccd"
305
+ assert classify_id("001001001") == "cmnd"
306
+ assert classify_id("0100047516") == "mst"
307
+ assert classify_id("0100047516001") == "mst"
308
+ assert classify_id("0912345678") == "phone"
309
+ assert classify_id("") == "unknown"
310
+ assert classify_id("abc") == "unknown"
311
+
312
+
313
+ def _run_standalone() -> int:
314
+ tests = [
315
+ (name, obj)
316
+ for name, obj in sorted(globals().items())
317
+ if name.startswith("test_") and callable(obj)
318
+ ]
319
+ failed = 0
320
+ for name, fn in tests:
321
+ try:
322
+ fn()
323
+ except AssertionError as exc:
324
+ failed += 1
325
+ print(f"FAIL {name}\n{exc}\n")
326
+ else:
327
+ print(f"ok {name}")
328
+ print()
329
+ print(f"{len(tests) - failed}/{len(tests)} test functions passed")
330
+ counts = SUITE["caseCounts"]
331
+ detail = ", ".join(f"{fn} {counts[fn]}" for fn in SUITE["functions"])
332
+ print(f"{len(CASES)} conformance cases: {detail}")
333
+ print(f"63 provinces, {ASSIGNED_PREFIX_COUNT} carrier prefixes")
334
+ return 1 if failed else 0
335
+
336
+
337
+ if __name__ == "__main__":
338
+ raise SystemExit(_run_standalone())