maskflow-core 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ from .detection import detect
2
+ from .entities import Finding, PIIType
3
+ from .masking import MaskResult, mask, unmask
4
+
5
+ __all__ = ["detect", "mask", "unmask", "Finding", "PIIType", "MaskResult"]
@@ -0,0 +1,36 @@
1
+ """Keyword-proximity confidence boosting.
2
+
3
+ Structural matches (email, credit card w/ Luhn, AWS keys, ...) are already
4
+ high-confidence on their own. Ambiguous matches (a bare 9-digit number, a
5
+ street-shaped line of text) need a nearby keyword to be trusted.
6
+ """
7
+ from .entities import PIIType
8
+
9
+ WINDOW = 40
10
+
11
+ CONTEXT_KEYWORDS: dict[PIIType, tuple[str, ...]] = {
12
+ PIIType.SSN: ("ssn", "social security"),
13
+ PIIType.DATE_OF_BIRTH: ("dob", "date of birth", "born on", "birthdate"),
14
+ PIIType.ADDRESS: ("address", "lives at", "located at", "ship to", "mailing"),
15
+ PIIType.API_KEY: ("api key", "apikey", "secret", "token", "credential"),
16
+ PIIType.PHONE: ("phone", "call", "tel", "mobile", "cell", "contact"),
17
+ PIIType.CREDIT_CARD: ("card number", "credit card", "visa", "mastercard", "cc#"),
18
+ PIIType.PERSON_NAME: ("name is", "name:", "my name", "signed", "regards"),
19
+ }
20
+
21
+ BOOST = 0.4
22
+ MAX_CONFIDENCE = 0.99
23
+
24
+
25
+ def apply_context_boost(text: str, start: int, end: int, pii_type: PIIType, base_confidence: float) -> float:
26
+ keywords = CONTEXT_KEYWORDS.get(pii_type)
27
+ if not keywords:
28
+ return base_confidence
29
+
30
+ window_start = max(0, start - WINDOW)
31
+ window_end = min(len(text), end + WINDOW)
32
+ nearby = text[window_start:start].lower() + " " + text[end:window_end].lower()
33
+
34
+ if any(keyword in nearby for keyword in keywords):
35
+ return min(MAX_CONFIDENCE, base_confidence + BOOST)
36
+ return base_confidence
@@ -0,0 +1,55 @@
1
+ """Orchestrates the regex, context, and NER passes into one deduplicated finding list."""
2
+ from .context import apply_context_boost
3
+ from .entities import Finding, PIIType
4
+ from .ner import detect_ner
5
+ from .patterns import PATTERNS
6
+
7
+ DEFAULT_MIN_CONFIDENCE = 0.5
8
+
9
+
10
+ def _regex_pass(text: str) -> list[Finding]:
11
+ candidates: list[Finding] = []
12
+
13
+ for pii_type, rules in PATTERNS.items():
14
+ for regex, base_confidence, validator in rules:
15
+ for match in regex.finditer(text):
16
+ if match.re.groups:
17
+ start, end = match.span(1)
18
+ value = match.group(1)
19
+ else:
20
+ start, end = match.span(0)
21
+ value = match.group(0)
22
+
23
+ confidence = base_confidence
24
+ if validator is not None:
25
+ adjusted = validator(value)
26
+ if adjusted is None:
27
+ continue
28
+ confidence = adjusted
29
+
30
+ confidence = apply_context_boost(text, start, end, pii_type, confidence)
31
+ candidates.append(Finding(pii_type, value, start, end, round(confidence, 2)))
32
+
33
+ return candidates
34
+
35
+
36
+ def _spans_overlap(a: Finding, b: Finding) -> bool:
37
+ return a.start < b.end and b.start < a.end
38
+
39
+
40
+ def _merge_overlaps(candidates: list[Finding]) -> list[Finding]:
41
+ accepted: list[Finding] = []
42
+ for finding in sorted(candidates, key=lambda f: f.confidence, reverse=True):
43
+ if not any(_spans_overlap(finding, kept) for kept in accepted):
44
+ accepted.append(finding)
45
+ return sorted(accepted, key=lambda f: f.start)
46
+
47
+
48
+ def detect(text: str, min_confidence: float = DEFAULT_MIN_CONFIDENCE) -> list[Finding]:
49
+ """Detect PII in `text`, returning non-overlapping Findings sorted by position."""
50
+ candidates = _regex_pass(text) + detect_ner(text)
51
+ merged = _merge_overlaps(candidates)
52
+ return [f for f in merged if f.confidence >= min_confidence]
53
+
54
+
55
+ __all__ = ["detect", "Finding", "PIIType"]
@@ -0,0 +1,30 @@
1
+ from dataclasses import dataclass
2
+ from enum import Enum
3
+
4
+
5
+ class PIIType(str, Enum):
6
+ EMAIL = "EMAIL"
7
+ PHONE = "PHONE"
8
+ SSN = "SSN"
9
+ CREDIT_CARD = "CREDIT_CARD"
10
+ IP_ADDRESS = "IP_ADDRESS"
11
+ AWS_KEY = "AWS_KEY"
12
+ API_KEY = "API_KEY"
13
+ JWT = "JWT"
14
+ IBAN = "IBAN"
15
+ ADDRESS = "ADDRESS"
16
+ PERSON_NAME = "PERSON_NAME"
17
+ DATE_OF_BIRTH = "DATE_OF_BIRTH"
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class Finding:
22
+ type: PIIType
23
+ value: str
24
+ start: int
25
+ end: int
26
+ confidence: float
27
+
28
+ @property
29
+ def span(self) -> tuple[int, int]:
30
+ return (self.start, self.end)
@@ -0,0 +1,39 @@
1
+ """Pure, stateless mask/unmask. No files, no DB -- the caller owns persistence of the mapping."""
2
+ from typing import NamedTuple
3
+
4
+ from .detection import DEFAULT_MIN_CONFIDENCE, detect
5
+
6
+
7
+ class MaskResult(NamedTuple):
8
+ masked_text: str
9
+ mapping: dict[str, str]
10
+
11
+
12
+ def mask(text: str, min_confidence: float = DEFAULT_MIN_CONFIDENCE) -> MaskResult:
13
+ """Replace detected PII with `<TYPE_n>` tokens, returning the masked text and a
14
+ {token: original_value} mapping the caller can use to unmask a later response."""
15
+ findings = detect(text, min_confidence=min_confidence)
16
+
17
+ mapping: dict[str, str] = {}
18
+ counters: dict[str, int] = {}
19
+ pieces: list[str] = []
20
+ cursor = 0
21
+
22
+ for finding in findings: # detect() returns non-overlapping findings sorted by start
23
+ counters[finding.type.value] = counters.get(finding.type.value, 0) + 1
24
+ token = f"<{finding.type.value}_{counters[finding.type.value]}>"
25
+ mapping[token] = finding.value
26
+ pieces.append(text[cursor:finding.start])
27
+ pieces.append(token)
28
+ cursor = finding.end
29
+
30
+ pieces.append(text[cursor:])
31
+ return MaskResult("".join(pieces), mapping)
32
+
33
+
34
+ def unmask(masked_text: str, mapping: dict[str, str]) -> str:
35
+ """Restore original values from a mapping produced by `mask()`."""
36
+ result = masked_text
37
+ for token, original in mapping.items():
38
+ result = result.replace(token, original)
39
+ return result
maskflow_core/ner.py ADDED
@@ -0,0 +1,49 @@
1
+ """spacy-based detection for entity types regex can't reliably catch: names and dates."""
2
+ from functools import lru_cache
3
+
4
+ import spacy
5
+
6
+ from .context import apply_context_boost
7
+ from .entities import Finding, PIIType
8
+
9
+ MODEL_NAME = "en_core_web_sm"
10
+
11
+ PERSON_BASE_CONFIDENCE = 0.75
12
+ DATE_BASE_CONFIDENCE = 0.3
13
+ DATE_OF_BIRTH_THRESHOLD = 0.6
14
+
15
+
16
+ @lru_cache(maxsize=1)
17
+ def _get_nlp():
18
+ try:
19
+ return spacy.load(MODEL_NAME)
20
+ except OSError as exc:
21
+ raise OSError(
22
+ f"spaCy model '{MODEL_NAME}' isn't installed. Run: "
23
+ f"python -m spacy download {MODEL_NAME}"
24
+ ) from exc
25
+
26
+
27
+ def detect_ner(text: str) -> list[Finding]:
28
+ nlp = _get_nlp()
29
+ doc = nlp(text)
30
+ findings: list[Finding] = []
31
+
32
+ for ent in doc.ents:
33
+ if ent.label_ == "PERSON":
34
+ confidence = apply_context_boost(
35
+ text, ent.start_char, ent.end_char, PIIType.PERSON_NAME, PERSON_BASE_CONFIDENCE
36
+ )
37
+ findings.append(
38
+ Finding(PIIType.PERSON_NAME, ent.text, ent.start_char, ent.end_char, confidence)
39
+ )
40
+ elif ent.label_ == "DATE":
41
+ confidence = apply_context_boost(
42
+ text, ent.start_char, ent.end_char, PIIType.DATE_OF_BIRTH, DATE_BASE_CONFIDENCE
43
+ )
44
+ if confidence >= DATE_OF_BIRTH_THRESHOLD:
45
+ findings.append(
46
+ Finding(PIIType.DATE_OF_BIRTH, ent.text, ent.start_char, ent.end_char, confidence)
47
+ )
48
+
49
+ return findings
@@ -0,0 +1,123 @@
1
+ """Regex patterns and structural validators for each PII type.
2
+
3
+ Each entry in PATTERNS maps a PIIType to (compiled_regex, base_confidence, validator).
4
+ `validator(value) -> float | None` returns an adjusted confidence, or None to reject
5
+ the match entirely (e.g. a 16-digit number that fails the Luhn check).
6
+ """
7
+ import re
8
+
9
+ from .entities import PIIType
10
+
11
+ EMAIL_RE = re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")
12
+
13
+ PHONE_RE = re.compile(
14
+ r"(?<!\d)(?:\+?\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}(?!\d)"
15
+ )
16
+
17
+ SSN_DASHED_RE = re.compile(r"(?<!\d)\d{3}-\d{2}-\d{4}(?!\d)")
18
+ SSN_PLAIN_RE = re.compile(r"(?<!\d)\d{9}(?!\d)")
19
+
20
+ CREDIT_CARD_RE = re.compile(
21
+ r"(?<!\d)(?:\d[ -]?){13,19}(?!\d)"
22
+ )
23
+
24
+ IPV4_RE = re.compile(
25
+ r"\b(?:(?:25[0-5]|2[0-4]\d|1\d\d|[1-9]?\d)\.){3}(?:25[0-5]|2[0-4]\d|1\d\d|[1-9]?\d)\b"
26
+ )
27
+ IPV6_RE = re.compile(r"\b(?:[A-Fa-f0-9]{1,4}:){7}[A-Fa-f0-9]{1,4}\b")
28
+
29
+ AWS_KEY_RE = re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b")
30
+
31
+ API_KEY_RE = re.compile(
32
+ r"\b(?:sk-ant-[A-Za-z0-9_-]{20,}"
33
+ r"|sk-[A-Za-z0-9_-]{20,}"
34
+ r"|gh[pousr]_[A-Za-z0-9]{20,}"
35
+ r"|xox[baprs]-[A-Za-z0-9-]{10,}"
36
+ r"|AIza[A-Za-z0-9_-]{35})\b"
37
+ )
38
+
39
+ GENERIC_SECRET_ASSIGNMENT_RE = re.compile(
40
+ r"(?i)\b\w*(?:key|secret|token|password|credential)\w*"
41
+ r"\s*[:=]\s*['\"]?([A-Za-z0-9_\-/+]{16,})['\"]?"
42
+ )
43
+
44
+ JWT_RE = re.compile(
45
+ r"\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b"
46
+ )
47
+
48
+ IBAN_RE = re.compile(r"\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b")
49
+
50
+ ADDRESS_RE = re.compile(
51
+ r"\b\d{1,6}\s+(?:[A-Z][a-zA-Z]*\s){1,4}"
52
+ r"(?:Street|St|Avenue|Ave|Road|Rd|Boulevard|Blvd|Lane|Ln|"
53
+ r"Drive|Dr|Court|Ct|Way|Place|Pl|Terrace|Ter)\.?\b"
54
+ r"(?:,?\s+(?:Apt|Suite|Ste|Unit)\.?\s*#?\w+)?"
55
+ )
56
+
57
+
58
+ def luhn_is_valid(digits: str) -> bool:
59
+ digits = [int(d) for d in digits]
60
+ checksum = 0
61
+ parity = len(digits) % 2
62
+ for i, d in enumerate(digits):
63
+ if i % 2 == parity:
64
+ d *= 2
65
+ if d > 9:
66
+ d -= 9
67
+ checksum += d
68
+ return checksum % 10 == 0
69
+
70
+
71
+ def _validate_credit_card(value: str) -> float | None:
72
+ digits = re.sub(r"[ -]", "", value)
73
+ if not (13 <= len(digits) <= 19) or not digits.isdigit():
74
+ return None
75
+ return 0.97 if luhn_is_valid(digits) else None
76
+
77
+
78
+ def iban_is_valid(value: str) -> bool:
79
+ value = value.upper()
80
+ rearranged = value[4:] + value[:4]
81
+ converted = "".join(str(int(c, 36)) for c in rearranged)
82
+ try:
83
+ return int(converted) % 97 == 1
84
+ except ValueError:
85
+ return False
86
+
87
+
88
+ def _validate_iban(value: str) -> float | None:
89
+ return 0.9 if iban_is_valid(value) else None
90
+
91
+
92
+ def _validate_ssn_dashed(value: str) -> float | None:
93
+ area = value[:3]
94
+ if area in ("000", "666") or area.startswith("9"):
95
+ return None
96
+ return 0.95
97
+
98
+
99
+ def _validate_ssn_plain(value: str) -> float | None:
100
+ # Bare 9-digit numbers are ambiguous (order IDs, phone numbers without
101
+ # formatting, etc.) -- start low, let context.py decide if it's really an SSN.
102
+ return 0.35
103
+
104
+
105
+ # type -> (regex, base_confidence, validator)
106
+ PATTERNS: dict[PIIType, list[tuple[re.Pattern, float, "callable | None"]]] = {
107
+ PIIType.EMAIL: [(EMAIL_RE, 0.95, None)],
108
+ PIIType.PHONE: [(PHONE_RE, 0.85, None)],
109
+ PIIType.SSN: [
110
+ (SSN_DASHED_RE, 0.95, _validate_ssn_dashed),
111
+ (SSN_PLAIN_RE, 0.35, _validate_ssn_plain),
112
+ ],
113
+ PIIType.CREDIT_CARD: [(CREDIT_CARD_RE, 0.9, _validate_credit_card)],
114
+ PIIType.IP_ADDRESS: [(IPV4_RE, 0.75, None), (IPV6_RE, 0.85, None)],
115
+ PIIType.AWS_KEY: [(AWS_KEY_RE, 0.97, None)],
116
+ PIIType.API_KEY: [
117
+ (API_KEY_RE, 0.95, None),
118
+ (GENERIC_SECRET_ASSIGNMENT_RE, 0.6, None),
119
+ ],
120
+ PIIType.JWT: [(JWT_RE, 0.9, None)],
121
+ PIIType.IBAN: [(IBAN_RE, 0.6, _validate_iban)],
122
+ PIIType.ADDRESS: [(ADDRESS_RE, 0.7, None)],
123
+ }
@@ -0,0 +1,9 @@
1
+ Metadata-Version: 2.4
2
+ Name: maskflow-core
3
+ Version: 0.1.0
4
+ Summary: PII detection and reversible masking engine for MaskFlow
5
+ Requires-Python: >=3.9
6
+ Requires-Dist: click>=8.0
7
+ Requires-Dist: spacy<4,>=3.7
8
+ Provides-Extra: dev
9
+ Requires-Dist: pytest>=8.0; extra == 'dev'
@@ -0,0 +1,10 @@
1
+ maskflow_core/__init__.py,sha256=7ycEq8WPMCAYBaJjU26XHOZzTsMljKyFrHsuqOS4j-U,191
2
+ maskflow_core/context.py,sha256=pYHBoG1w5TQIegzXrPOx2izemI48K9pfEegvKMXgEag,1440
3
+ maskflow_core/detection.py,sha256=pvqxmK3SVK-NcgUstXTm4KpmZyYGscU3wsulLeB73KY,2011
4
+ maskflow_core/entities.py,sha256=MgH_T24nSDKWE2lRhSjNDt0Cza7l0yJvkjdHlmOGTAI,592
5
+ maskflow_core/masking.py,sha256=YHu3j-4ZrC8eE5_yXFXXeis9OHDkWG0GBF78hRjqvYk,1431
6
+ maskflow_core/ner.py,sha256=YXhyK5y2elvuPgDrfIeISiE6ikpaEQoIhdzlVYWb7NE,1507
7
+ maskflow_core/patterns.py,sha256=hfJnTe4RVdXOB8peOs2a-QzvaRUEogg4S5Fv2Bf5PuE,3796
8
+ maskflow_core-0.1.0.dist-info/METADATA,sha256=fPBuMqTJYlqNLVPmQRQWHyYN9T-aMxz_Z9n2lumzTa4,264
9
+ maskflow_core-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
10
+ maskflow_core-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.31.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any