maskflow-core 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- maskflow_core/__init__.py +5 -0
- maskflow_core/context.py +36 -0
- maskflow_core/detection.py +55 -0
- maskflow_core/entities.py +30 -0
- maskflow_core/masking.py +39 -0
- maskflow_core/ner.py +49 -0
- maskflow_core/patterns.py +123 -0
- maskflow_core-0.1.0.dist-info/METADATA +9 -0
- maskflow_core-0.1.0.dist-info/RECORD +10 -0
- maskflow_core-0.1.0.dist-info/WHEEL +4 -0
maskflow_core/context.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Keyword-proximity confidence boosting.
|
|
2
|
+
|
|
3
|
+
Structural matches (email, credit card w/ Luhn, AWS keys, ...) are already
|
|
4
|
+
high-confidence on their own. Ambiguous matches (a bare 9-digit number, a
|
|
5
|
+
street-shaped line of text) need a nearby keyword to be trusted.
|
|
6
|
+
"""
|
|
7
|
+
from .entities import PIIType
|
|
8
|
+
|
|
9
|
+
WINDOW = 40
|
|
10
|
+
|
|
11
|
+
CONTEXT_KEYWORDS: dict[PIIType, tuple[str, ...]] = {
|
|
12
|
+
PIIType.SSN: ("ssn", "social security"),
|
|
13
|
+
PIIType.DATE_OF_BIRTH: ("dob", "date of birth", "born on", "birthdate"),
|
|
14
|
+
PIIType.ADDRESS: ("address", "lives at", "located at", "ship to", "mailing"),
|
|
15
|
+
PIIType.API_KEY: ("api key", "apikey", "secret", "token", "credential"),
|
|
16
|
+
PIIType.PHONE: ("phone", "call", "tel", "mobile", "cell", "contact"),
|
|
17
|
+
PIIType.CREDIT_CARD: ("card number", "credit card", "visa", "mastercard", "cc#"),
|
|
18
|
+
PIIType.PERSON_NAME: ("name is", "name:", "my name", "signed", "regards"),
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
BOOST = 0.4
|
|
22
|
+
MAX_CONFIDENCE = 0.99
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def apply_context_boost(text: str, start: int, end: int, pii_type: PIIType, base_confidence: float) -> float:
|
|
26
|
+
keywords = CONTEXT_KEYWORDS.get(pii_type)
|
|
27
|
+
if not keywords:
|
|
28
|
+
return base_confidence
|
|
29
|
+
|
|
30
|
+
window_start = max(0, start - WINDOW)
|
|
31
|
+
window_end = min(len(text), end + WINDOW)
|
|
32
|
+
nearby = text[window_start:start].lower() + " " + text[end:window_end].lower()
|
|
33
|
+
|
|
34
|
+
if any(keyword in nearby for keyword in keywords):
|
|
35
|
+
return min(MAX_CONFIDENCE, base_confidence + BOOST)
|
|
36
|
+
return base_confidence
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Orchestrates the regex, context, and NER passes into one deduplicated finding list."""
|
|
2
|
+
from .context import apply_context_boost
|
|
3
|
+
from .entities import Finding, PIIType
|
|
4
|
+
from .ner import detect_ner
|
|
5
|
+
from .patterns import PATTERNS
|
|
6
|
+
|
|
7
|
+
DEFAULT_MIN_CONFIDENCE = 0.5
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _regex_pass(text: str) -> list[Finding]:
|
|
11
|
+
candidates: list[Finding] = []
|
|
12
|
+
|
|
13
|
+
for pii_type, rules in PATTERNS.items():
|
|
14
|
+
for regex, base_confidence, validator in rules:
|
|
15
|
+
for match in regex.finditer(text):
|
|
16
|
+
if match.re.groups:
|
|
17
|
+
start, end = match.span(1)
|
|
18
|
+
value = match.group(1)
|
|
19
|
+
else:
|
|
20
|
+
start, end = match.span(0)
|
|
21
|
+
value = match.group(0)
|
|
22
|
+
|
|
23
|
+
confidence = base_confidence
|
|
24
|
+
if validator is not None:
|
|
25
|
+
adjusted = validator(value)
|
|
26
|
+
if adjusted is None:
|
|
27
|
+
continue
|
|
28
|
+
confidence = adjusted
|
|
29
|
+
|
|
30
|
+
confidence = apply_context_boost(text, start, end, pii_type, confidence)
|
|
31
|
+
candidates.append(Finding(pii_type, value, start, end, round(confidence, 2)))
|
|
32
|
+
|
|
33
|
+
return candidates
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _spans_overlap(a: Finding, b: Finding) -> bool:
|
|
37
|
+
return a.start < b.end and b.start < a.end
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _merge_overlaps(candidates: list[Finding]) -> list[Finding]:
|
|
41
|
+
accepted: list[Finding] = []
|
|
42
|
+
for finding in sorted(candidates, key=lambda f: f.confidence, reverse=True):
|
|
43
|
+
if not any(_spans_overlap(finding, kept) for kept in accepted):
|
|
44
|
+
accepted.append(finding)
|
|
45
|
+
return sorted(accepted, key=lambda f: f.start)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def detect(text: str, min_confidence: float = DEFAULT_MIN_CONFIDENCE) -> list[Finding]:
|
|
49
|
+
"""Detect PII in `text`, returning non-overlapping Findings sorted by position."""
|
|
50
|
+
candidates = _regex_pass(text) + detect_ner(text)
|
|
51
|
+
merged = _merge_overlaps(candidates)
|
|
52
|
+
return [f for f in merged if f.confidence >= min_confidence]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
__all__ = ["detect", "Finding", "PIIType"]
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from enum import Enum
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class PIIType(str, Enum):
|
|
6
|
+
EMAIL = "EMAIL"
|
|
7
|
+
PHONE = "PHONE"
|
|
8
|
+
SSN = "SSN"
|
|
9
|
+
CREDIT_CARD = "CREDIT_CARD"
|
|
10
|
+
IP_ADDRESS = "IP_ADDRESS"
|
|
11
|
+
AWS_KEY = "AWS_KEY"
|
|
12
|
+
API_KEY = "API_KEY"
|
|
13
|
+
JWT = "JWT"
|
|
14
|
+
IBAN = "IBAN"
|
|
15
|
+
ADDRESS = "ADDRESS"
|
|
16
|
+
PERSON_NAME = "PERSON_NAME"
|
|
17
|
+
DATE_OF_BIRTH = "DATE_OF_BIRTH"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Finding:
|
|
22
|
+
type: PIIType
|
|
23
|
+
value: str
|
|
24
|
+
start: int
|
|
25
|
+
end: int
|
|
26
|
+
confidence: float
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def span(self) -> tuple[int, int]:
|
|
30
|
+
return (self.start, self.end)
|
maskflow_core/masking.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Pure, stateless mask/unmask. No files, no DB -- the caller owns persistence of the mapping."""
|
|
2
|
+
from typing import NamedTuple
|
|
3
|
+
|
|
4
|
+
from .detection import DEFAULT_MIN_CONFIDENCE, detect
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class MaskResult(NamedTuple):
|
|
8
|
+
masked_text: str
|
|
9
|
+
mapping: dict[str, str]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def mask(text: str, min_confidence: float = DEFAULT_MIN_CONFIDENCE) -> MaskResult:
|
|
13
|
+
"""Replace detected PII with `<TYPE_n>` tokens, returning the masked text and a
|
|
14
|
+
{token: original_value} mapping the caller can use to unmask a later response."""
|
|
15
|
+
findings = detect(text, min_confidence=min_confidence)
|
|
16
|
+
|
|
17
|
+
mapping: dict[str, str] = {}
|
|
18
|
+
counters: dict[str, int] = {}
|
|
19
|
+
pieces: list[str] = []
|
|
20
|
+
cursor = 0
|
|
21
|
+
|
|
22
|
+
for finding in findings: # detect() returns non-overlapping findings sorted by start
|
|
23
|
+
counters[finding.type.value] = counters.get(finding.type.value, 0) + 1
|
|
24
|
+
token = f"<{finding.type.value}_{counters[finding.type.value]}>"
|
|
25
|
+
mapping[token] = finding.value
|
|
26
|
+
pieces.append(text[cursor:finding.start])
|
|
27
|
+
pieces.append(token)
|
|
28
|
+
cursor = finding.end
|
|
29
|
+
|
|
30
|
+
pieces.append(text[cursor:])
|
|
31
|
+
return MaskResult("".join(pieces), mapping)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def unmask(masked_text: str, mapping: dict[str, str]) -> str:
|
|
35
|
+
"""Restore original values from a mapping produced by `mask()`."""
|
|
36
|
+
result = masked_text
|
|
37
|
+
for token, original in mapping.items():
|
|
38
|
+
result = result.replace(token, original)
|
|
39
|
+
return result
|
maskflow_core/ner.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""spacy-based detection for entity types regex can't reliably catch: names and dates."""
|
|
2
|
+
from functools import lru_cache
|
|
3
|
+
|
|
4
|
+
import spacy
|
|
5
|
+
|
|
6
|
+
from .context import apply_context_boost
|
|
7
|
+
from .entities import Finding, PIIType
|
|
8
|
+
|
|
9
|
+
MODEL_NAME = "en_core_web_sm"
|
|
10
|
+
|
|
11
|
+
PERSON_BASE_CONFIDENCE = 0.75
|
|
12
|
+
DATE_BASE_CONFIDENCE = 0.3
|
|
13
|
+
DATE_OF_BIRTH_THRESHOLD = 0.6
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@lru_cache(maxsize=1)
|
|
17
|
+
def _get_nlp():
|
|
18
|
+
try:
|
|
19
|
+
return spacy.load(MODEL_NAME)
|
|
20
|
+
except OSError as exc:
|
|
21
|
+
raise OSError(
|
|
22
|
+
f"spaCy model '{MODEL_NAME}' isn't installed. Run: "
|
|
23
|
+
f"python -m spacy download {MODEL_NAME}"
|
|
24
|
+
) from exc
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def detect_ner(text: str) -> list[Finding]:
|
|
28
|
+
nlp = _get_nlp()
|
|
29
|
+
doc = nlp(text)
|
|
30
|
+
findings: list[Finding] = []
|
|
31
|
+
|
|
32
|
+
for ent in doc.ents:
|
|
33
|
+
if ent.label_ == "PERSON":
|
|
34
|
+
confidence = apply_context_boost(
|
|
35
|
+
text, ent.start_char, ent.end_char, PIIType.PERSON_NAME, PERSON_BASE_CONFIDENCE
|
|
36
|
+
)
|
|
37
|
+
findings.append(
|
|
38
|
+
Finding(PIIType.PERSON_NAME, ent.text, ent.start_char, ent.end_char, confidence)
|
|
39
|
+
)
|
|
40
|
+
elif ent.label_ == "DATE":
|
|
41
|
+
confidence = apply_context_boost(
|
|
42
|
+
text, ent.start_char, ent.end_char, PIIType.DATE_OF_BIRTH, DATE_BASE_CONFIDENCE
|
|
43
|
+
)
|
|
44
|
+
if confidence >= DATE_OF_BIRTH_THRESHOLD:
|
|
45
|
+
findings.append(
|
|
46
|
+
Finding(PIIType.DATE_OF_BIRTH, ent.text, ent.start_char, ent.end_char, confidence)
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
return findings
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Regex patterns and structural validators for each PII type.
|
|
2
|
+
|
|
3
|
+
Each entry in PATTERNS maps a PIIType to (compiled_regex, base_confidence, validator).
|
|
4
|
+
`validator(value) -> float | None` returns an adjusted confidence, or None to reject
|
|
5
|
+
the match entirely (e.g. a 16-digit number that fails the Luhn check).
|
|
6
|
+
"""
|
|
7
|
+
import re
|
|
8
|
+
|
|
9
|
+
from .entities import PIIType
|
|
10
|
+
|
|
11
|
+
EMAIL_RE = re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")
|
|
12
|
+
|
|
13
|
+
PHONE_RE = re.compile(
|
|
14
|
+
r"(?<!\d)(?:\+?\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}(?!\d)"
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
SSN_DASHED_RE = re.compile(r"(?<!\d)\d{3}-\d{2}-\d{4}(?!\d)")
|
|
18
|
+
SSN_PLAIN_RE = re.compile(r"(?<!\d)\d{9}(?!\d)")
|
|
19
|
+
|
|
20
|
+
CREDIT_CARD_RE = re.compile(
|
|
21
|
+
r"(?<!\d)(?:\d[ -]?){13,19}(?!\d)"
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
IPV4_RE = re.compile(
|
|
25
|
+
r"\b(?:(?:25[0-5]|2[0-4]\d|1\d\d|[1-9]?\d)\.){3}(?:25[0-5]|2[0-4]\d|1\d\d|[1-9]?\d)\b"
|
|
26
|
+
)
|
|
27
|
+
IPV6_RE = re.compile(r"\b(?:[A-Fa-f0-9]{1,4}:){7}[A-Fa-f0-9]{1,4}\b")
|
|
28
|
+
|
|
29
|
+
AWS_KEY_RE = re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b")
|
|
30
|
+
|
|
31
|
+
API_KEY_RE = re.compile(
|
|
32
|
+
r"\b(?:sk-ant-[A-Za-z0-9_-]{20,}"
|
|
33
|
+
r"|sk-[A-Za-z0-9_-]{20,}"
|
|
34
|
+
r"|gh[pousr]_[A-Za-z0-9]{20,}"
|
|
35
|
+
r"|xox[baprs]-[A-Za-z0-9-]{10,}"
|
|
36
|
+
r"|AIza[A-Za-z0-9_-]{35})\b"
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
GENERIC_SECRET_ASSIGNMENT_RE = re.compile(
|
|
40
|
+
r"(?i)\b\w*(?:key|secret|token|password|credential)\w*"
|
|
41
|
+
r"\s*[:=]\s*['\"]?([A-Za-z0-9_\-/+]{16,})['\"]?"
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
JWT_RE = re.compile(
|
|
45
|
+
r"\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
IBAN_RE = re.compile(r"\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b")
|
|
49
|
+
|
|
50
|
+
ADDRESS_RE = re.compile(
|
|
51
|
+
r"\b\d{1,6}\s+(?:[A-Z][a-zA-Z]*\s){1,4}"
|
|
52
|
+
r"(?:Street|St|Avenue|Ave|Road|Rd|Boulevard|Blvd|Lane|Ln|"
|
|
53
|
+
r"Drive|Dr|Court|Ct|Way|Place|Pl|Terrace|Ter)\.?\b"
|
|
54
|
+
r"(?:,?\s+(?:Apt|Suite|Ste|Unit)\.?\s*#?\w+)?"
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def luhn_is_valid(digits: str) -> bool:
|
|
59
|
+
digits = [int(d) for d in digits]
|
|
60
|
+
checksum = 0
|
|
61
|
+
parity = len(digits) % 2
|
|
62
|
+
for i, d in enumerate(digits):
|
|
63
|
+
if i % 2 == parity:
|
|
64
|
+
d *= 2
|
|
65
|
+
if d > 9:
|
|
66
|
+
d -= 9
|
|
67
|
+
checksum += d
|
|
68
|
+
return checksum % 10 == 0
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _validate_credit_card(value: str) -> float | None:
|
|
72
|
+
digits = re.sub(r"[ -]", "", value)
|
|
73
|
+
if not (13 <= len(digits) <= 19) or not digits.isdigit():
|
|
74
|
+
return None
|
|
75
|
+
return 0.97 if luhn_is_valid(digits) else None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def iban_is_valid(value: str) -> bool:
|
|
79
|
+
value = value.upper()
|
|
80
|
+
rearranged = value[4:] + value[:4]
|
|
81
|
+
converted = "".join(str(int(c, 36)) for c in rearranged)
|
|
82
|
+
try:
|
|
83
|
+
return int(converted) % 97 == 1
|
|
84
|
+
except ValueError:
|
|
85
|
+
return False
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _validate_iban(value: str) -> float | None:
|
|
89
|
+
return 0.9 if iban_is_valid(value) else None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _validate_ssn_dashed(value: str) -> float | None:
|
|
93
|
+
area = value[:3]
|
|
94
|
+
if area in ("000", "666") or area.startswith("9"):
|
|
95
|
+
return None
|
|
96
|
+
return 0.95
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _validate_ssn_plain(value: str) -> float | None:
|
|
100
|
+
# Bare 9-digit numbers are ambiguous (order IDs, phone numbers without
|
|
101
|
+
# formatting, etc.) -- start low, let context.py decide if it's really an SSN.
|
|
102
|
+
return 0.35
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# type -> (regex, base_confidence, validator)
|
|
106
|
+
PATTERNS: dict[PIIType, list[tuple[re.Pattern, float, "callable | None"]]] = {
|
|
107
|
+
PIIType.EMAIL: [(EMAIL_RE, 0.95, None)],
|
|
108
|
+
PIIType.PHONE: [(PHONE_RE, 0.85, None)],
|
|
109
|
+
PIIType.SSN: [
|
|
110
|
+
(SSN_DASHED_RE, 0.95, _validate_ssn_dashed),
|
|
111
|
+
(SSN_PLAIN_RE, 0.35, _validate_ssn_plain),
|
|
112
|
+
],
|
|
113
|
+
PIIType.CREDIT_CARD: [(CREDIT_CARD_RE, 0.9, _validate_credit_card)],
|
|
114
|
+
PIIType.IP_ADDRESS: [(IPV4_RE, 0.75, None), (IPV6_RE, 0.85, None)],
|
|
115
|
+
PIIType.AWS_KEY: [(AWS_KEY_RE, 0.97, None)],
|
|
116
|
+
PIIType.API_KEY: [
|
|
117
|
+
(API_KEY_RE, 0.95, None),
|
|
118
|
+
(GENERIC_SECRET_ASSIGNMENT_RE, 0.6, None),
|
|
119
|
+
],
|
|
120
|
+
PIIType.JWT: [(JWT_RE, 0.9, None)],
|
|
121
|
+
PIIType.IBAN: [(IBAN_RE, 0.6, _validate_iban)],
|
|
122
|
+
PIIType.ADDRESS: [(ADDRESS_RE, 0.7, None)],
|
|
123
|
+
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: maskflow-core
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: PII detection and reversible masking engine for MaskFlow
|
|
5
|
+
Requires-Python: >=3.9
|
|
6
|
+
Requires-Dist: click>=8.0
|
|
7
|
+
Requires-Dist: spacy<4,>=3.7
|
|
8
|
+
Provides-Extra: dev
|
|
9
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
maskflow_core/__init__.py,sha256=7ycEq8WPMCAYBaJjU26XHOZzTsMljKyFrHsuqOS4j-U,191
|
|
2
|
+
maskflow_core/context.py,sha256=pYHBoG1w5TQIegzXrPOx2izemI48K9pfEegvKMXgEag,1440
|
|
3
|
+
maskflow_core/detection.py,sha256=pvqxmK3SVK-NcgUstXTm4KpmZyYGscU3wsulLeB73KY,2011
|
|
4
|
+
maskflow_core/entities.py,sha256=MgH_T24nSDKWE2lRhSjNDt0Cza7l0yJvkjdHlmOGTAI,592
|
|
5
|
+
maskflow_core/masking.py,sha256=YHu3j-4ZrC8eE5_yXFXXeis9OHDkWG0GBF78hRjqvYk,1431
|
|
6
|
+
maskflow_core/ner.py,sha256=YXhyK5y2elvuPgDrfIeISiE6ikpaEQoIhdzlVYWb7NE,1507
|
|
7
|
+
maskflow_core/patterns.py,sha256=hfJnTe4RVdXOB8peOs2a-QzvaRUEogg4S5Fv2Bf5PuE,3796
|
|
8
|
+
maskflow_core-0.1.0.dist-info/METADATA,sha256=fPBuMqTJYlqNLVPmQRQWHyYN9T-aMxz_Z9n2lumzTa4,264
|
|
9
|
+
maskflow_core-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
10
|
+
maskflow_core-0.1.0.dist-info/RECORD,,
|