maskflow-pack-india 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,15 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ venv/
5
+ .env
6
+ node_modules/
7
+ dist/
8
+ build/
9
+ *.egg-info/
10
+ .DS_Store
11
+ .pytest_cache/
12
+ .idea/
13
+ .coverage
14
+ .mypy_cache/
15
+ .ruff_cache/
@@ -0,0 +1,7 @@
1
+ Metadata-Version: 2.5
2
+ Name: maskflow-pack-india
3
+ Version: 0.1.0
4
+ Summary: MaskFlow's Indian PII recognizers (Aadhaar, PAN, GSTIN, IFSC, UPI VPA)
5
+ License: MIT
6
+ Requires-Python: >=3.10
7
+ Requires-Dist: maskflow-core<0.5,>=0.4.0
@@ -0,0 +1,25 @@
1
+ [project]
2
+ name = "maskflow-pack-india"
3
+ version = "0.1.0"
4
+ description = "MaskFlow's Indian PII recognizers (Aadhaar, PAN, GSTIN, IFSC, UPI VPA)"
5
+ requires-python = ">=3.10"
6
+ license = { text = "MIT" }
7
+ dependencies = [
8
+ # register_pattern()'s validator/context_keywords kwargs are all this
9
+ # pack needs -- no NER, so no maskflow-core[nlp] extra required (keeps
10
+ # this pack spaCy-free and its own import cost near zero).
11
+ "maskflow-core>=0.4.0,<0.5",
12
+ ]
13
+
14
+ [tool.uv.sources]
15
+ maskflow-core = { workspace = true }
16
+
17
+ [tool.uv]
18
+ package = true
19
+
20
+ [build-system]
21
+ requires = ["hatchling"]
22
+ build-backend = "hatchling.build"
23
+
24
+ [tool.hatch.build.targets.wheel]
25
+ packages = ["src/maskflow_pack_india"]
@@ -0,0 +1,147 @@
1
+ """Registers MaskFlow's India-specific recognizers (AADHAAR, AADHAAR_MASKED,
2
+ PAN, GSTIN, IFSC, UPI_VPA) against maskflow-core on import. Importing this
3
+ package is the side effect that makes detect()/mask()/unmask() aware of
4
+ these types -- see maskflow_core.registry.register_pattern.
5
+
6
+ Context keywords are positive-only (English, Hindi/Devanagari, and Hinglish
7
+ transliterations) -- maskflow-core's context.apply_context_boost() has no
8
+ negative-context mechanism yet (CLAUDE.md's confidence formula documents one
9
+ as a target, but it isn't implemented in core), so "example/test/dummy"-style
10
+ suppression is out of scope for this pack until core grows that hook.
11
+ """
12
+
13
+ from maskflow_core.registry import register_pattern
14
+
15
+ from . import patterns
16
+
17
+ register_pattern(
18
+ "AADHAAR",
19
+ patterns.AADHAAR_RE,
20
+ 0.5,
21
+ validator=patterns.validate_aadhaar,
22
+ context_keywords=(
23
+ "aadhaar",
24
+ "aadhar",
25
+ "adhaar",
26
+ "uidai",
27
+ "aadhaar number",
28
+ "aadhaar no",
29
+ "aadhar no",
30
+ "aadhaar card",
31
+ "uid number",
32
+ "आधार",
33
+ "आधार कार्ड",
34
+ "आधार संख्या",
35
+ ),
36
+ )
37
+ register_pattern(
38
+ "AADHAAR",
39
+ patterns.AADHAAR_VID_RE,
40
+ 0.5,
41
+ validator=patterns.validate_aadhaar,
42
+ context_keywords=(
43
+ "vid",
44
+ "virtual id",
45
+ "aadhaar vid",
46
+ "aadhaar",
47
+ "aadhar",
48
+ "uidai",
49
+ "आधार",
50
+ "वर्चुअल आईडी",
51
+ ),
52
+ )
53
+ register_pattern(
54
+ "AADHAAR_MASKED",
55
+ patterns.AADHAAR_MASKED_RE,
56
+ # No validator possible (8 of 12 digits are gone) -- 0.45 starts BELOW
57
+ # detection.py's DEFAULT_MIN_CONFIDENCE (0.5), same design as pack-intl's
58
+ # SSN_PLAIN: an unvalidated, structurally-ambiguous match should need a
59
+ # nearby keyword to clear the bar, not pass on shape alone.
60
+ 0.45,
61
+ context_keywords=(
62
+ "aadhaar",
63
+ "aadhar",
64
+ "adhaar",
65
+ "uidai",
66
+ "masked aadhaar",
67
+ "aadhaar ending",
68
+ "आधार",
69
+ ),
70
+ )
71
+
72
+ register_pattern(
73
+ "PAN",
74
+ patterns.PAN_RE,
75
+ 0.6,
76
+ validator=patterns.validate_pan,
77
+ context_keywords=(
78
+ "pan",
79
+ "pan card",
80
+ "pan number",
81
+ "pan no",
82
+ "permanent account number",
83
+ "पैन",
84
+ "पैन कार्ड",
85
+ "पैन नंबर",
86
+ ),
87
+ )
88
+ register_pattern(
89
+ "PAN",
90
+ patterns.PAN_EMBEDDED_IN_GSTIN_RE,
91
+ 0.6,
92
+ validator=patterns.validate_pan,
93
+ )
94
+
95
+ register_pattern(
96
+ "GSTIN",
97
+ patterns.GSTIN_RE,
98
+ 0.6,
99
+ validator=patterns.validate_gstin,
100
+ context_keywords=(
101
+ "gstin",
102
+ "gst number",
103
+ "gst no",
104
+ "gst reg",
105
+ "goods and services tax",
106
+ "gstin number",
107
+ "जीएसटी",
108
+ "जीएसटीआईएन",
109
+ "जीएसटी नंबर",
110
+ ),
111
+ )
112
+
113
+ register_pattern(
114
+ "IFSC",
115
+ patterns.IFSC_RE,
116
+ 0.6,
117
+ validator=patterns.validate_ifsc,
118
+ context_keywords=(
119
+ "ifsc",
120
+ "ifsc code",
121
+ "branch code",
122
+ "bank branch",
123
+ "आईएफएससी",
124
+ "आईएफएससी कोड",
125
+ ),
126
+ )
127
+
128
+ register_pattern(
129
+ "UPI_VPA",
130
+ patterns.UPI_VPA_RE,
131
+ 0.5,
132
+ validator=patterns.validate_upi_vpa,
133
+ context_keywords=(
134
+ "upi",
135
+ "upi id",
136
+ "vpa",
137
+ "pay to",
138
+ "gpay",
139
+ "google pay",
140
+ "phonepe",
141
+ "paytm",
142
+ "यूपीआई",
143
+ "यूपीआई आईडी",
144
+ ),
145
+ )
146
+
147
+ __all__: list[str] = []
@@ -0,0 +1,103 @@
1
+ """Verhoeff (AADHAAR) and GSTIN check-digit algorithms, isolated from regex/
2
+ registration concerns so they can be unit-tested against published test
3
+ vectors on their own (see tests/test_checksums.py).
4
+
5
+ Neither algorithm is invented here -- both are the standard published
6
+ constants/procedures (Verhoeff: Wikipedia "Verhoeff algorithm" / ISO
7
+ reference tables; GSTIN: GSTN's published Luhn-mod-36 check-digit scheme,
8
+ reimplemented independently by several open gstin-validator packages).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ # ---------------------------------------------------------------------------
14
+ # Verhoeff checksum -- dihedral group D5 multiplication/permutation tables.
15
+ # ---------------------------------------------------------------------------
16
+
17
+ _D5_MULT: tuple[tuple[int, ...], ...] = (
18
+ (0, 1, 2, 3, 4, 5, 6, 7, 8, 9),
19
+ (1, 2, 3, 4, 0, 6, 7, 8, 9, 5),
20
+ (2, 3, 4, 0, 1, 7, 8, 9, 5, 6),
21
+ (3, 4, 0, 1, 2, 8, 9, 5, 6, 7),
22
+ (4, 0, 1, 2, 3, 9, 5, 6, 7, 8),
23
+ (5, 9, 8, 7, 6, 0, 4, 3, 2, 1),
24
+ (6, 5, 9, 8, 7, 1, 0, 4, 3, 2),
25
+ (7, 6, 5, 9, 8, 2, 1, 0, 4, 3),
26
+ (8, 7, 6, 5, 9, 3, 2, 1, 0, 4),
27
+ (9, 8, 7, 6, 5, 4, 3, 2, 1, 0),
28
+ )
29
+
30
+ _D5_PERM: tuple[tuple[int, ...], ...] = (
31
+ (0, 1, 2, 3, 4, 5, 6, 7, 8, 9),
32
+ (1, 5, 7, 6, 2, 8, 3, 0, 9, 4),
33
+ (5, 8, 0, 3, 7, 9, 6, 1, 4, 2),
34
+ (8, 9, 1, 6, 0, 4, 3, 5, 2, 7),
35
+ (9, 4, 5, 3, 1, 2, 6, 8, 7, 0),
36
+ (4, 2, 8, 6, 5, 7, 3, 9, 0, 1),
37
+ (2, 7, 9, 3, 8, 0, 6, 4, 1, 5),
38
+ (7, 0, 4, 6, 9, 1, 3, 2, 5, 8),
39
+ )
40
+
41
+ _D5_INV: tuple[int, ...] = (0, 4, 3, 2, 1, 5, 6, 7, 8, 9)
42
+
43
+
44
+ def verhoeff_is_valid(digits: str) -> bool:
45
+ """`digits` is the FULL number (AADHAAR UID or VID) including its
46
+ trailing Verhoeff check digit. Processed right-to-left; valid iff the
47
+ accumulator lands on 0. Caller is responsible for length/digit-only
48
+ checks -- this raises ValueError on a non-digit character rather than
49
+ silently misjudging it, since `int(ch)` would otherwise be the only
50
+ signal something was wrong."""
51
+ c = 0
52
+ for i, ch in enumerate(reversed(digits)):
53
+ c = _D5_MULT[c][_D5_PERM[i % 8][int(ch)]]
54
+ return c == 0
55
+
56
+
57
+ def verhoeff_generate(digits: str) -> str:
58
+ """Given digits WITHOUT a check digit, return the check digit that makes
59
+ `verhoeff_is_valid(digits + check_digit)` True. Not used by the
60
+ recognizer itself (nothing needs to *produce* a checksum at detection
61
+ time) -- exists so tests/fixtures can build synthetic, checksum-valid
62
+ AADHAAR-shaped values without hand-computing them."""
63
+ c = 0
64
+ for i, ch in enumerate(reversed(digits)):
65
+ c = _D5_MULT[c][_D5_PERM[(i + 1) % 8][int(ch)]]
66
+ return str(_D5_INV[c])
67
+
68
+
69
+ # ---------------------------------------------------------------------------
70
+ # GSTIN checksum -- Luhn-mod-36 variant: alternating weights (2, 1, 2, 1, ...)
71
+ # right-to-left over the 36-char alphanumeric alphabet, with base-36
72
+ # digit-sum folding (the base-36 analogue of Luhn's "subtract 9 if > 9").
73
+ # ---------------------------------------------------------------------------
74
+
75
+ _GST_ALPHABET = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ"
76
+ _GST_MOD = 36
77
+
78
+
79
+ def gstin_checksum_char(first_14: str) -> str:
80
+ """Compute the 15th (check) character for the first 14 GSTIN characters."""
81
+ total = 0
82
+ factor = 2
83
+ for ch in reversed(first_14):
84
+ value = _GST_ALPHABET.index(ch)
85
+ product = factor * value
86
+ total += (product // _GST_MOD) + (product % _GST_MOD)
87
+ factor = 1 if factor == 2 else 2
88
+ check_value = (_GST_MOD - (total % _GST_MOD)) % _GST_MOD
89
+ return _GST_ALPHABET[check_value]
90
+
91
+
92
+ def gstin_is_valid(gstin: str) -> bool:
93
+ """`gstin` is the full 15-character value. Caller is responsible for the
94
+ other structural checks (state code range, embedded-PAN shape, the
95
+ literal 'Z') -- this only verifies the 15th-character checksum."""
96
+ if len(gstin) != 15:
97
+ return False
98
+ try:
99
+ return gstin[14] == gstin_checksum_char(gstin[:14])
100
+ except ValueError:
101
+ # gstin[:14] contains a character outside _GST_ALPHABET (e.g.
102
+ # lowercase, punctuation) -- not a valid GSTIN shape, not a bug.
103
+ return False
@@ -0,0 +1,95 @@
1
+ """Bundled RBI-assigned 4-letter bank codes -- the first 4 characters of an
2
+ IFSC (e.g. "HDFC" in HDFC0001234). Curated, NOT exhaustive: it covers major
3
+ scheduled commercial banks, small finance banks, and payments banks
4
+ operating in India as of this file's last refresh. An IFSC whose bank code
5
+ isn't in this set is treated as structurally invalid by validate_ifsc()
6
+ (patterns.py) -- a false negative on an obscure/regional bank is preferred
7
+ over inventing a checksum-like check IFSC doesn't actually have (there is no
8
+ public per-character checksum on an IFSC; the bank-code lookup IS the
9
+ structural check).
10
+
11
+ Refresh procedure:
12
+ 1. RBI publishes the authoritative IFSC master list (bank-wise, all
13
+ branches) at https://www.rbi.org.in -- search "IFSC master list" for
14
+ the current CSV/XLSX download.
15
+ 2. Extract the unique set of 4-character bank-code prefixes (characters
16
+ 0-3 of each IFSC in the list; character 4 is always '0').
17
+ 3. Diff against IFSC_BANK_CODES below, add any new entries (bank mergers
18
+ retire old codes but never reuse them for a different bank -- safe to
19
+ leave retired codes in this set indefinitely).
20
+ 4. Update the "last refreshed" date in this docstring.
21
+
22
+ Last refreshed: 2026-08 (manually curated from public bank-merger and PSP
23
+ records, not a direct RBI CSV import -- treat as a reasonable starting set,
24
+ not a guarantee of completeness).
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ IFSC_BANK_CODES: frozenset[str] = frozenset(
30
+ {
31
+ # Public sector banks
32
+ "SBIN", # State Bank of India
33
+ "PUNB", # Punjab National Bank
34
+ "BKID", # Bank of India
35
+ "CNRB", # Canara Bank
36
+ "UBIN", # Union Bank of India
37
+ "IOBA", # Indian Overseas Bank
38
+ "IDIB", # Indian Bank
39
+ "CBIN", # Central Bank of India
40
+ "MAHB", # Bank of Maharashtra
41
+ "UCBA", # UCO Bank
42
+ "PSIB", # Punjab & Sind Bank
43
+ "BARB", # Bank of Baroda
44
+ # Merged/retired public sector codes (still valid on old branches --
45
+ # never reused for a different bank, see refresh procedure above)
46
+ "ORBC", # Oriental Bank of Commerce (merged into PNB)
47
+ "ANDB", # Andhra Bank (merged into Union Bank)
48
+ "CORP", # Corporation Bank (merged into Union Bank)
49
+ "ALLA", # Allahabad Bank (merged into Indian Bank)
50
+ "SYNB", # Syndicate Bank (merged into Canara Bank)
51
+ "VIJB", # Vijaya Bank (merged into Bank of Baroda)
52
+ # Major private sector banks
53
+ "HDFC", # HDFC Bank
54
+ "ICIC", # ICICI Bank
55
+ "UTIB", # Axis Bank
56
+ "KKBK", # Kotak Mahindra Bank
57
+ "YESB", # Yes Bank
58
+ "INDB", # IndusInd Bank
59
+ "IDFB", # IDFC FIRST Bank
60
+ "RATN", # RBL Bank
61
+ "FDRL", # Federal Bank
62
+ "SIBL", # South Indian Bank
63
+ "KVBL", # Karur Vysya Bank
64
+ "TMBL", # Tamilnad Mercantile Bank
65
+ "DCBB", # DCB Bank
66
+ "CSBK", # CSB Bank
67
+ "KARB", # Karnataka Bank
68
+ "DBSS", # DBS Bank India (incl. merged Lakshmi Vilas Bank branches)
69
+ "BDBL", # Bandhan Bank
70
+ "JAKA", # Jammu & Kashmir Bank
71
+ "NKGS", # NKGSB Co-operative Bank
72
+ "SVCB", # Shamrao Vithal Co-operative Bank
73
+ # Foreign banks operating in India
74
+ "CITI", # Citibank
75
+ "HSBC", # HSBC
76
+ "SCBL", # Standard Chartered Bank
77
+ "DEUT", # Deutsche Bank
78
+ "BOFA", # Bank of America
79
+ # Small finance banks
80
+ "AUBL", # AU Small Finance Bank
81
+ "EQBL", # Equitas Small Finance Bank
82
+ "UJVN", # Ujjivan Small Finance Bank
83
+ "ESFB", # ESAF Small Finance Bank
84
+ "SURY", # Suryoday Small Finance Bank
85
+ "UTKS", # Utkarsh Small Finance Bank
86
+ "JSFB", # Jana Small Finance Bank
87
+ # Payments banks
88
+ "PYTM", # Paytm Payments Bank
89
+ "AIRP", # Airtel Payments Bank
90
+ "FINO", # Fino Payments Bank
91
+ "NSPB", # NSDL Payments Bank
92
+ "IPOS", # India Post Payments Bank
93
+ "JIOP", # Jio Payments Bank
94
+ }
95
+ )
@@ -0,0 +1,114 @@
1
+ """Bundled NPCI-issued UPI PSP handles -- the part of a VPA after '@' (e.g.
2
+ "okhdfcbank" in name@okhdfcbank). Curated, NOT exhaustive: NPCI periodically
3
+ approves new handles for new/rebranded PSPs. A handle not in this set is
4
+ treated as NOT a UPI VPA by validate_upi_vpa() (patterns.py) -- deliberately
5
+ conservative, since a handle-shaped string that happens to have a TLD-like
6
+ look (e.g. "name@company") should fall through to nothing rather than being
7
+ misclassified.
8
+
9
+ Refresh procedure:
10
+ 1. NPCI publishes the list of live PSP handles as part of its UPI
11
+ "member/handle" directory at https://www.npci.org.in -- cross-referenced
12
+ against each PSP's own published VPA format documentation (banks list
13
+ their own handles on their UPI help pages).
14
+ 2. Add any newly observed handle (lowercase, no leading '@') to
15
+ UPI_PSP_HANDLES below.
16
+ 3. Do not remove a retired handle -- old VPAs using it may still appear in
17
+ historical text even after NPCI stops issuing new ones.
18
+ 4. Update the "last refreshed" date in this docstring.
19
+
20
+ Last refreshed: 2026-08 (manually curated from public PSP/bank documentation,
21
+ not a direct NPCI feed import -- treat as a reasonable starting set, not a
22
+ guarantee of completeness).
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ UPI_PSP_HANDLES: frozenset[str] = frozenset(
28
+ {
29
+ # Google Pay (issued per linked bank)
30
+ "okhdfcbank",
31
+ "oksbi",
32
+ "okaxis",
33
+ "okicici",
34
+ "okbizaxis",
35
+ # PhonePe / Yes Bank
36
+ "ybl",
37
+ "yapl",
38
+ # Paytm
39
+ "paytm",
40
+ # Generic / NPCI-operated
41
+ "upi",
42
+ # IDBI Bank (legacy handle)
43
+ "ibl",
44
+ # Axis Bank (incl. merchant handle used by several PSPs)
45
+ "axl",
46
+ "axisbank",
47
+ # Amazon Pay
48
+ "apl",
49
+ "rapl", # Amazon Pay via RBL Bank
50
+ # Federal Bank
51
+ "fbl",
52
+ "federal",
53
+ # IDFC FIRST Bank
54
+ "idfcbank",
55
+ # Jupiter (via Axis Bank)
56
+ "jupiteraxis",
57
+ # Kotak Mahindra Bank
58
+ "kotak",
59
+ # Yes Bank
60
+ "yesbank",
61
+ # State Bank of India
62
+ "sbi",
63
+ # ICICI Bank
64
+ "icici",
65
+ # HDFC Bank
66
+ "hdfcbank",
67
+ # IndusInd Bank
68
+ "indus",
69
+ # Canara Bank
70
+ "cnrb",
71
+ # DBS Bank India
72
+ "dbs",
73
+ # Bank of India
74
+ "boi",
75
+ # Central Bank of India
76
+ "cbin",
77
+ "centralbank",
78
+ # Punjab National Bank
79
+ "pnb",
80
+ # Union Bank of India
81
+ "unionbankofindia",
82
+ "unionbank",
83
+ # HSBC
84
+ "hsbc",
85
+ # Indian Bank
86
+ "indianbank",
87
+ # Indian Overseas Bank
88
+ "iob",
89
+ # Karur Vysya Bank
90
+ "kvb",
91
+ # Karnataka Bank
92
+ "karb",
93
+ # Bank of Baroda
94
+ "barodampay",
95
+ "baroda",
96
+ # Freecharge (via Axis Bank)
97
+ "freecharge",
98
+ # MobiKwik
99
+ "mobikwik",
100
+ # Airtel Payments Bank
101
+ "airtel",
102
+ # Jio Payments Bank
103
+ "jio",
104
+ # Slice
105
+ "slice",
106
+ # Aditya Birla Finance
107
+ "abfspay",
108
+ # WhatsApp Pay (issued per linked bank, "wa" prefix)
109
+ "waaxis",
110
+ "wahdfcbank",
111
+ "waicici",
112
+ "wasbi",
113
+ }
114
+ )
@@ -0,0 +1,144 @@
1
+ """Regex patterns and structural/checksum validators for India-specific PII
2
+ types. Each validator(value) -> float | None returns an adjusted confidence,
3
+ or None to reject the match entirely -- see checksums.py for the Verhoeff
4
+ (AADHAAR) and GSTIN check-digit algorithms these call into. __init__.py
5
+ registers these against maskflow-core via register_pattern().
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+
12
+ from .checksums import gstin_is_valid, verhoeff_is_valid
13
+ from .data.ifsc_bank_codes import IFSC_BANK_CODES
14
+ from .data.upi_handles import UPI_PSP_HANDLES
15
+
16
+ # ---------------------------------------------------------------------------
17
+ # AADHAAR -- 12-digit UID, 16-digit VID, and masked display forms.
18
+ # ---------------------------------------------------------------------------
19
+
20
+ # 12 digits, never starting 0/1 (UIDAI never issues those as a first digit).
21
+ # group(1) is the whole matched run including separators -- detection.py's
22
+ # _scan_pattern uses group(1) as the span whenever the regex has ANY group,
23
+ # so the outer group must span the full value, not just a sub-piece of it.
24
+ # group(2) is the separator, backreferenced via \2 so both gaps must match
25
+ # (both spaced, both hyphenated, or both absent -- never mixed).
26
+ AADHAAR_RE = re.compile(r"(?<!\d)([2-9]\d{3}([ -]?)\d{4}\2\d{4})(?!\d)")
27
+
28
+ # 16-digit Virtual ID (VID) -- same first-digit rule and Verhoeff checksum,
29
+ # one more group of 4 digits than the UID form.
30
+ AADHAAR_VID_RE = re.compile(r"(?<!\d)([2-9]\d{3}([ -]?)\d{4}\2\d{4}\2\d{4})(?!\d)")
31
+
32
+ # Masked display form banks/agencies show back to a user, e.g. "XXXX XXXX
33
+ # 9012" or "xxxxxxxx9012" -- first 8 digits replaced with a mask character
34
+ # (X or *), last 4 digits real. Unverifiable (8 of 12 digits are gone), so
35
+ # this is registered as its own lower-confidence, unvalidated entity type
36
+ # rather than fed through validate_aadhaar().
37
+ AADHAAR_MASKED_RE = re.compile(r"(?<!\w)([xX*]{4}([ -]?)[xX*]{4}\2\d{4})(?!\w)")
38
+
39
+
40
+ def validate_aadhaar(value: str) -> float | None:
41
+ digits = re.sub(r"[ -]", "", value)
42
+ if len(digits) not in (12, 16) or not digits.isdigit():
43
+ return None
44
+ # 0.9 base leaves room for the context boost (CONTEXT_KEYWORDS in
45
+ # __init__.py) to reach MAX_CONFIDENCE without a checksum-valid AADHAAR
46
+ # ever scoring below a checksum-valid but context-free match on some
47
+ # other overlapping type.
48
+ return 0.9 if verhoeff_is_valid(digits) else None
49
+
50
+
51
+ # ---------------------------------------------------------------------------
52
+ # PAN -- 5 letters + 4 digits + 1 letter, 4th letter constrained to a known
53
+ # holder-category set. No public checksum exists for the final letter.
54
+ # ---------------------------------------------------------------------------
55
+
56
+ _PAN_FOURTH_CHAR_CATEGORIES = "PCHFATBLJG"
57
+
58
+ # Standalone PAN: boundaries exclude alnum neighbors on both sides, so a PAN
59
+ # embedded directly inside a longer alnum run (e.g. a GSTIN) does NOT match
60
+ # here -- see PAN_EMBEDDED_IN_GSTIN_RE below for that case specifically.
61
+ PAN_RE = re.compile(r"(?<![A-Za-z0-9])[A-Z]{5}[0-9]{4}[A-Z](?![A-Za-z0-9])")
62
+
63
+ # The PAN embedded in a GSTIN's characters 3-12: preceded by the 2-digit
64
+ # state code, followed by <entity_number><'Z'><checksum>. A valid GSTIN's
65
+ # embedded PAN also surfaces as its own PAN candidate span this way;
66
+ # spanset.py's CONTAINS resolution (CLAUDE.md design decision #1) then picks
67
+ # the longer, equally-validated GSTIN span over the shorter contained PAN.
68
+ PAN_EMBEDDED_IN_GSTIN_RE = re.compile(r"(?<=\d{2})[A-Z]{5}[0-9]{4}[A-Z](?=[0-9A-Z]Z[0-9A-Z])")
69
+
70
+
71
+ def validate_pan(value: str) -> float | None:
72
+ if len(value) != 10:
73
+ return None
74
+ if value[3] not in _PAN_FOURTH_CHAR_CATEGORIES:
75
+ return None
76
+ # 0.85, not higher: this is a structural check (holder-category letter
77
+ # only), not a checksum -- PAN's final letter has no public checksum, so
78
+ # this can never be as confident as a Verhoeff- or mod-97-validated span.
79
+ return 0.85
80
+
81
+
82
+ # ---------------------------------------------------------------------------
83
+ # GSTIN -- 15 chars: state code (01-38) + PAN + entity number + 'Z' + base-36
84
+ # checksum. Reuses validate_pan() for the embedded PAN's structural check.
85
+ # ---------------------------------------------------------------------------
86
+
87
+ GSTIN_RE = re.compile(r"\b\d{2}[A-Z]{5}\d{4}[A-Z][0-9A-Z]Z[0-9A-Z]\b")
88
+
89
+ _MIN_STATE_CODE = 1
90
+ _MAX_STATE_CODE = 38
91
+
92
+
93
+ def validate_gstin(value: str) -> float | None:
94
+ if len(value) != 15:
95
+ return None
96
+ if not (_MIN_STATE_CODE <= int(value[:2]) <= _MAX_STATE_CODE):
97
+ return None
98
+ if validate_pan(value[2:12]) is None:
99
+ return None
100
+ if value[13] != "Z":
101
+ return None
102
+ return 0.95 if gstin_is_valid(value) else None
103
+
104
+
105
+ # ---------------------------------------------------------------------------
106
+ # IFSC -- 4-letter bank code + literal '0' + 6 alnum branch code. Bank code
107
+ # validated against a bundled, periodically-refreshed RBI code list (see
108
+ # data/ifsc_bank_codes.py) -- there is no per-character checksum on an IFSC,
109
+ # the bank-code lookup IS the structural check.
110
+ # ---------------------------------------------------------------------------
111
+
112
+ IFSC_RE = re.compile(r"\b[A-Z]{4}0[A-Z0-9]{6}\b")
113
+
114
+
115
+ def validate_ifsc(value: str) -> float | None:
116
+ if len(value) != 11 or value[4] != "0":
117
+ return None
118
+ if value[:4] not in IFSC_BANK_CODES:
119
+ return None
120
+ return 0.9
121
+
122
+
123
+ # ---------------------------------------------------------------------------
124
+ # UPI_VPA -- handle@psp, PSP validated against a bundled NPCI handle list
125
+ # (see data/upi_handles.py). The trailing `(?!\.[A-Za-z])` keeps this from
126
+ # matching the first label of a real multi-label domain (name@gmail.com):
127
+ # the psp group is letters-only (no dot), so on "name@gmail.com" it can only
128
+ # ever match up to "gmail" -- the lookahead then vetoes that because a
129
+ # dot-then-letter follows, i.e. this is actually someone's email domain, not
130
+ # a UPI handle. A handle NOT in the bundled list is rejected here (returns
131
+ # None) rather than guessed at, so a general-purpose EMAIL recognizer (e.g.
132
+ # maskflow-pack-intl's) gets first claim on anything that looks like mail.
133
+ # ---------------------------------------------------------------------------
134
+
135
+ UPI_VPA_RE = re.compile(r"\b[A-Za-z0-9.\-_]{2,256}@[A-Za-z]{2,64}(?!\.[A-Za-z])\b")
136
+
137
+
138
+ def validate_upi_vpa(value: str) -> float | None:
139
+ handle, _, psp = value.partition("@")
140
+ if not handle or not psp:
141
+ return None
142
+ if psp.lower() not in UPI_PSP_HANDLES:
143
+ return None
144
+ return 0.95
@@ -0,0 +1,5 @@
1
+ from maskflow_core.testing import ( # noqa: F401 -- picked up as pytest hooks by name
2
+ pytest_collection_modifyitems,
3
+ pytest_configure,
4
+ pytest_exception_interact,
5
+ )
File without changes