maskflow-core 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- maskflow_core-0.1.0/.gitignore +12 -0
- maskflow_core-0.1.0/.python-version +1 -0
- maskflow_core-0.1.0/PKG-INFO +9 -0
- maskflow_core-0.1.0/README.md +53 -0
- maskflow_core-0.1.0/pyproject.toml +24 -0
- maskflow_core-0.1.0/src/maskflow_core/__init__.py +5 -0
- maskflow_core-0.1.0/src/maskflow_core/context.py +36 -0
- maskflow_core-0.1.0/src/maskflow_core/detection.py +55 -0
- maskflow_core-0.1.0/src/maskflow_core/entities.py +30 -0
- maskflow_core-0.1.0/src/maskflow_core/masking.py +39 -0
- maskflow_core-0.1.0/src/maskflow_core/ner.py +49 -0
- maskflow_core-0.1.0/src/maskflow_core/patterns.py +123 -0
- maskflow_core-0.1.0/tests/fixtures/__init__.py +0 -0
- maskflow_core-0.1.0/tests/fixtures/pii_samples.py +268 -0
- maskflow_core-0.1.0/tests/test_detection.py +55 -0
- maskflow_core-0.1.0/tests/test_masking.py +38 -0
- maskflow_core-0.1.0/uv.lock +2020 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.12
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: maskflow-core
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: PII detection and reversible masking engine for MaskFlow
|
|
5
|
+
Requires-Python: >=3.9
|
|
6
|
+
Requires-Dist: click>=8.0
|
|
7
|
+
Requires-Dist: spacy<4,>=3.7
|
|
8
|
+
Provides-Extra: dev
|
|
9
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# maskflow-core
|
|
2
|
+
|
|
3
|
+
PII detection and reversible masking engine — the foundation for every other MaskFlow phase (CLI, GitHub Action, SDK, ...).
|
|
4
|
+
|
|
5
|
+
Detects 12 PII types via a three-layer pipeline: regex + structural validation (Luhn for cards, mod-97 for IBANs) -> keyword-context confidence boosting -> spacy NER for names and dates. Overlapping matches are resolved by confidence, and detection returns non-overlapping `Finding`s sorted by position.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
Requires [uv](https://docs.astral.sh/uv/).
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
cd core
|
|
13
|
+
uv sync --extra dev
|
|
14
|
+
uv run python -m spacy download en_core_web_sm
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
The spaCy model is a separate download rather than a pip dependency -- PyPI doesn't allow packages
|
|
18
|
+
to declare a direct URL as a dependency, and this is the standard pattern for any spaCy-based
|
|
19
|
+
package.
|
|
20
|
+
|
|
21
|
+
## Usage
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
from maskflow_core import detect, mask, unmask
|
|
25
|
+
|
|
26
|
+
detect("Email me at alice@example.com or call 415-555-0132.")
|
|
27
|
+
# [Finding(type=PIIType.EMAIL, value='alice@example.com', ...), Finding(type=PIIType.PHONE, ...)]
|
|
28
|
+
|
|
29
|
+
result = mask("Email me at alice@example.com or call 415-555-0132.")
|
|
30
|
+
result.masked_text # "Email me at <EMAIL_1> or call <PHONE_1>."
|
|
31
|
+
result.mapping # {'<EMAIL_1>': 'alice@example.com', '<PHONE_1>': '415-555-0132'}
|
|
32
|
+
|
|
33
|
+
unmask(result.masked_text, result.mapping) # original text, restored
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
`mask`/`unmask` are pure functions — the engine never writes to disk or a database. Persisting the mapping is the caller's responsibility (the CLI and SDK phases handle that).
|
|
37
|
+
|
|
38
|
+
## PII types (v1)
|
|
39
|
+
|
|
40
|
+
Email, phone, SSN, credit card, IP address (v4/v6), AWS access key, API key / generic secret, JWT, IBAN, street address, person name, date of birth.
|
|
41
|
+
|
|
42
|
+
## Tests
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
uv run pytest
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
`tests/fixtures/pii_samples.py` has 100+ labeled examples; `test_detection.py` enforces a 95% accuracy floor against them and checks that PII-free text produces zero findings.
|
|
49
|
+
|
|
50
|
+
## Known limitations
|
|
51
|
+
|
|
52
|
+
- spacy's NER occasionally flags a capitalized sentence-starter as a `PERSON_NAME` false positive (e.g. "Email me at..." -> "Email" tagged as a name). Not a training data issue — future phases add a `.maskflowrc.yml` exclude-config for exactly this kind of false positive.
|
|
53
|
+
- Phone/address regexes are US-shaped; international formats are a future improvement.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "maskflow-core"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "PII detection and reversible masking engine for MaskFlow"
|
|
5
|
+
requires-python = ">=3.9"
|
|
6
|
+
dependencies = [
|
|
7
|
+
"spacy>=3.7,<4",
|
|
8
|
+
"click>=8.0",
|
|
9
|
+
]
|
|
10
|
+
|
|
11
|
+
[project.optional-dependencies]
|
|
12
|
+
dev = [
|
|
13
|
+
"pytest>=8.0",
|
|
14
|
+
]
|
|
15
|
+
|
|
16
|
+
[tool.uv]
|
|
17
|
+
package = true
|
|
18
|
+
|
|
19
|
+
[build-system]
|
|
20
|
+
requires = ["hatchling"]
|
|
21
|
+
build-backend = "hatchling.build"
|
|
22
|
+
|
|
23
|
+
[tool.hatch.build.targets.wheel]
|
|
24
|
+
packages = ["src/maskflow_core"]
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Keyword-proximity confidence boosting.
|
|
2
|
+
|
|
3
|
+
Structural matches (email, credit card w/ Luhn, AWS keys, ...) are already
|
|
4
|
+
high-confidence on their own. Ambiguous matches (a bare 9-digit number, a
|
|
5
|
+
street-shaped line of text) need a nearby keyword to be trusted.
|
|
6
|
+
"""
|
|
7
|
+
from .entities import PIIType
|
|
8
|
+
|
|
9
|
+
WINDOW = 40
|
|
10
|
+
|
|
11
|
+
CONTEXT_KEYWORDS: dict[PIIType, tuple[str, ...]] = {
|
|
12
|
+
PIIType.SSN: ("ssn", "social security"),
|
|
13
|
+
PIIType.DATE_OF_BIRTH: ("dob", "date of birth", "born on", "birthdate"),
|
|
14
|
+
PIIType.ADDRESS: ("address", "lives at", "located at", "ship to", "mailing"),
|
|
15
|
+
PIIType.API_KEY: ("api key", "apikey", "secret", "token", "credential"),
|
|
16
|
+
PIIType.PHONE: ("phone", "call", "tel", "mobile", "cell", "contact"),
|
|
17
|
+
PIIType.CREDIT_CARD: ("card number", "credit card", "visa", "mastercard", "cc#"),
|
|
18
|
+
PIIType.PERSON_NAME: ("name is", "name:", "my name", "signed", "regards"),
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
BOOST = 0.4
|
|
22
|
+
MAX_CONFIDENCE = 0.99
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def apply_context_boost(text: str, start: int, end: int, pii_type: PIIType, base_confidence: float) -> float:
|
|
26
|
+
keywords = CONTEXT_KEYWORDS.get(pii_type)
|
|
27
|
+
if not keywords:
|
|
28
|
+
return base_confidence
|
|
29
|
+
|
|
30
|
+
window_start = max(0, start - WINDOW)
|
|
31
|
+
window_end = min(len(text), end + WINDOW)
|
|
32
|
+
nearby = text[window_start:start].lower() + " " + text[end:window_end].lower()
|
|
33
|
+
|
|
34
|
+
if any(keyword in nearby for keyword in keywords):
|
|
35
|
+
return min(MAX_CONFIDENCE, base_confidence + BOOST)
|
|
36
|
+
return base_confidence
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Orchestrates the regex, context, and NER passes into one deduplicated finding list."""
|
|
2
|
+
from .context import apply_context_boost
|
|
3
|
+
from .entities import Finding, PIIType
|
|
4
|
+
from .ner import detect_ner
|
|
5
|
+
from .patterns import PATTERNS
|
|
6
|
+
|
|
7
|
+
DEFAULT_MIN_CONFIDENCE = 0.5
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _regex_pass(text: str) -> list[Finding]:
|
|
11
|
+
candidates: list[Finding] = []
|
|
12
|
+
|
|
13
|
+
for pii_type, rules in PATTERNS.items():
|
|
14
|
+
for regex, base_confidence, validator in rules:
|
|
15
|
+
for match in regex.finditer(text):
|
|
16
|
+
if match.re.groups:
|
|
17
|
+
start, end = match.span(1)
|
|
18
|
+
value = match.group(1)
|
|
19
|
+
else:
|
|
20
|
+
start, end = match.span(0)
|
|
21
|
+
value = match.group(0)
|
|
22
|
+
|
|
23
|
+
confidence = base_confidence
|
|
24
|
+
if validator is not None:
|
|
25
|
+
adjusted = validator(value)
|
|
26
|
+
if adjusted is None:
|
|
27
|
+
continue
|
|
28
|
+
confidence = adjusted
|
|
29
|
+
|
|
30
|
+
confidence = apply_context_boost(text, start, end, pii_type, confidence)
|
|
31
|
+
candidates.append(Finding(pii_type, value, start, end, round(confidence, 2)))
|
|
32
|
+
|
|
33
|
+
return candidates
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _spans_overlap(a: Finding, b: Finding) -> bool:
|
|
37
|
+
return a.start < b.end and b.start < a.end
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _merge_overlaps(candidates: list[Finding]) -> list[Finding]:
|
|
41
|
+
accepted: list[Finding] = []
|
|
42
|
+
for finding in sorted(candidates, key=lambda f: f.confidence, reverse=True):
|
|
43
|
+
if not any(_spans_overlap(finding, kept) for kept in accepted):
|
|
44
|
+
accepted.append(finding)
|
|
45
|
+
return sorted(accepted, key=lambda f: f.start)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def detect(text: str, min_confidence: float = DEFAULT_MIN_CONFIDENCE) -> list[Finding]:
|
|
49
|
+
"""Detect PII in `text`, returning non-overlapping Findings sorted by position."""
|
|
50
|
+
candidates = _regex_pass(text) + detect_ner(text)
|
|
51
|
+
merged = _merge_overlaps(candidates)
|
|
52
|
+
return [f for f in merged if f.confidence >= min_confidence]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
__all__ = ["detect", "Finding", "PIIType"]
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from enum import Enum
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class PIIType(str, Enum):
|
|
6
|
+
EMAIL = "EMAIL"
|
|
7
|
+
PHONE = "PHONE"
|
|
8
|
+
SSN = "SSN"
|
|
9
|
+
CREDIT_CARD = "CREDIT_CARD"
|
|
10
|
+
IP_ADDRESS = "IP_ADDRESS"
|
|
11
|
+
AWS_KEY = "AWS_KEY"
|
|
12
|
+
API_KEY = "API_KEY"
|
|
13
|
+
JWT = "JWT"
|
|
14
|
+
IBAN = "IBAN"
|
|
15
|
+
ADDRESS = "ADDRESS"
|
|
16
|
+
PERSON_NAME = "PERSON_NAME"
|
|
17
|
+
DATE_OF_BIRTH = "DATE_OF_BIRTH"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Finding:
|
|
22
|
+
type: PIIType
|
|
23
|
+
value: str
|
|
24
|
+
start: int
|
|
25
|
+
end: int
|
|
26
|
+
confidence: float
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def span(self) -> tuple[int, int]:
|
|
30
|
+
return (self.start, self.end)
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Pure, stateless mask/unmask. No files, no DB -- the caller owns persistence of the mapping."""
|
|
2
|
+
from typing import NamedTuple
|
|
3
|
+
|
|
4
|
+
from .detection import DEFAULT_MIN_CONFIDENCE, detect
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class MaskResult(NamedTuple):
|
|
8
|
+
masked_text: str
|
|
9
|
+
mapping: dict[str, str]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def mask(text: str, min_confidence: float = DEFAULT_MIN_CONFIDENCE) -> MaskResult:
|
|
13
|
+
"""Replace detected PII with `<TYPE_n>` tokens, returning the masked text and a
|
|
14
|
+
{token: original_value} mapping the caller can use to unmask a later response."""
|
|
15
|
+
findings = detect(text, min_confidence=min_confidence)
|
|
16
|
+
|
|
17
|
+
mapping: dict[str, str] = {}
|
|
18
|
+
counters: dict[str, int] = {}
|
|
19
|
+
pieces: list[str] = []
|
|
20
|
+
cursor = 0
|
|
21
|
+
|
|
22
|
+
for finding in findings: # detect() returns non-overlapping findings sorted by start
|
|
23
|
+
counters[finding.type.value] = counters.get(finding.type.value, 0) + 1
|
|
24
|
+
token = f"<{finding.type.value}_{counters[finding.type.value]}>"
|
|
25
|
+
mapping[token] = finding.value
|
|
26
|
+
pieces.append(text[cursor:finding.start])
|
|
27
|
+
pieces.append(token)
|
|
28
|
+
cursor = finding.end
|
|
29
|
+
|
|
30
|
+
pieces.append(text[cursor:])
|
|
31
|
+
return MaskResult("".join(pieces), mapping)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def unmask(masked_text: str, mapping: dict[str, str]) -> str:
|
|
35
|
+
"""Restore original values from a mapping produced by `mask()`."""
|
|
36
|
+
result = masked_text
|
|
37
|
+
for token, original in mapping.items():
|
|
38
|
+
result = result.replace(token, original)
|
|
39
|
+
return result
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""spacy-based detection for entity types regex can't reliably catch: names and dates."""
|
|
2
|
+
from functools import lru_cache
|
|
3
|
+
|
|
4
|
+
import spacy
|
|
5
|
+
|
|
6
|
+
from .context import apply_context_boost
|
|
7
|
+
from .entities import Finding, PIIType
|
|
8
|
+
|
|
9
|
+
MODEL_NAME = "en_core_web_sm"
|
|
10
|
+
|
|
11
|
+
PERSON_BASE_CONFIDENCE = 0.75
|
|
12
|
+
DATE_BASE_CONFIDENCE = 0.3
|
|
13
|
+
DATE_OF_BIRTH_THRESHOLD = 0.6
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@lru_cache(maxsize=1)
|
|
17
|
+
def _get_nlp():
|
|
18
|
+
try:
|
|
19
|
+
return spacy.load(MODEL_NAME)
|
|
20
|
+
except OSError as exc:
|
|
21
|
+
raise OSError(
|
|
22
|
+
f"spaCy model '{MODEL_NAME}' isn't installed. Run: "
|
|
23
|
+
f"python -m spacy download {MODEL_NAME}"
|
|
24
|
+
) from exc
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def detect_ner(text: str) -> list[Finding]:
|
|
28
|
+
nlp = _get_nlp()
|
|
29
|
+
doc = nlp(text)
|
|
30
|
+
findings: list[Finding] = []
|
|
31
|
+
|
|
32
|
+
for ent in doc.ents:
|
|
33
|
+
if ent.label_ == "PERSON":
|
|
34
|
+
confidence = apply_context_boost(
|
|
35
|
+
text, ent.start_char, ent.end_char, PIIType.PERSON_NAME, PERSON_BASE_CONFIDENCE
|
|
36
|
+
)
|
|
37
|
+
findings.append(
|
|
38
|
+
Finding(PIIType.PERSON_NAME, ent.text, ent.start_char, ent.end_char, confidence)
|
|
39
|
+
)
|
|
40
|
+
elif ent.label_ == "DATE":
|
|
41
|
+
confidence = apply_context_boost(
|
|
42
|
+
text, ent.start_char, ent.end_char, PIIType.DATE_OF_BIRTH, DATE_BASE_CONFIDENCE
|
|
43
|
+
)
|
|
44
|
+
if confidence >= DATE_OF_BIRTH_THRESHOLD:
|
|
45
|
+
findings.append(
|
|
46
|
+
Finding(PIIType.DATE_OF_BIRTH, ent.text, ent.start_char, ent.end_char, confidence)
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
return findings
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Regex patterns and structural validators for each PII type.
|
|
2
|
+
|
|
3
|
+
Each entry in PATTERNS maps a PIIType to (compiled_regex, base_confidence, validator).
|
|
4
|
+
`validator(value) -> float | None` returns an adjusted confidence, or None to reject
|
|
5
|
+
the match entirely (e.g. a 16-digit number that fails the Luhn check).
|
|
6
|
+
"""
|
|
7
|
+
import re
|
|
8
|
+
|
|
9
|
+
from .entities import PIIType
|
|
10
|
+
|
|
11
|
+
EMAIL_RE = re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")
|
|
12
|
+
|
|
13
|
+
PHONE_RE = re.compile(
|
|
14
|
+
r"(?<!\d)(?:\+?\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}(?!\d)"
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
SSN_DASHED_RE = re.compile(r"(?<!\d)\d{3}-\d{2}-\d{4}(?!\d)")
|
|
18
|
+
SSN_PLAIN_RE = re.compile(r"(?<!\d)\d{9}(?!\d)")
|
|
19
|
+
|
|
20
|
+
CREDIT_CARD_RE = re.compile(
|
|
21
|
+
r"(?<!\d)(?:\d[ -]?){13,19}(?!\d)"
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
IPV4_RE = re.compile(
|
|
25
|
+
r"\b(?:(?:25[0-5]|2[0-4]\d|1\d\d|[1-9]?\d)\.){3}(?:25[0-5]|2[0-4]\d|1\d\d|[1-9]?\d)\b"
|
|
26
|
+
)
|
|
27
|
+
IPV6_RE = re.compile(r"\b(?:[A-Fa-f0-9]{1,4}:){7}[A-Fa-f0-9]{1,4}\b")
|
|
28
|
+
|
|
29
|
+
AWS_KEY_RE = re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b")
|
|
30
|
+
|
|
31
|
+
API_KEY_RE = re.compile(
|
|
32
|
+
r"\b(?:sk-ant-[A-Za-z0-9_-]{20,}"
|
|
33
|
+
r"|sk-[A-Za-z0-9_-]{20,}"
|
|
34
|
+
r"|gh[pousr]_[A-Za-z0-9]{20,}"
|
|
35
|
+
r"|xox[baprs]-[A-Za-z0-9-]{10,}"
|
|
36
|
+
r"|AIza[A-Za-z0-9_-]{35})\b"
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
GENERIC_SECRET_ASSIGNMENT_RE = re.compile(
|
|
40
|
+
r"(?i)\b\w*(?:key|secret|token|password|credential)\w*"
|
|
41
|
+
r"\s*[:=]\s*['\"]?([A-Za-z0-9_\-/+]{16,})['\"]?"
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
JWT_RE = re.compile(
|
|
45
|
+
r"\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
IBAN_RE = re.compile(r"\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b")
|
|
49
|
+
|
|
50
|
+
ADDRESS_RE = re.compile(
|
|
51
|
+
r"\b\d{1,6}\s+(?:[A-Z][a-zA-Z]*\s){1,4}"
|
|
52
|
+
r"(?:Street|St|Avenue|Ave|Road|Rd|Boulevard|Blvd|Lane|Ln|"
|
|
53
|
+
r"Drive|Dr|Court|Ct|Way|Place|Pl|Terrace|Ter)\.?\b"
|
|
54
|
+
r"(?:,?\s+(?:Apt|Suite|Ste|Unit)\.?\s*#?\w+)?"
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def luhn_is_valid(digits: str) -> bool:
|
|
59
|
+
digits = [int(d) for d in digits]
|
|
60
|
+
checksum = 0
|
|
61
|
+
parity = len(digits) % 2
|
|
62
|
+
for i, d in enumerate(digits):
|
|
63
|
+
if i % 2 == parity:
|
|
64
|
+
d *= 2
|
|
65
|
+
if d > 9:
|
|
66
|
+
d -= 9
|
|
67
|
+
checksum += d
|
|
68
|
+
return checksum % 10 == 0
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _validate_credit_card(value: str) -> float | None:
|
|
72
|
+
digits = re.sub(r"[ -]", "", value)
|
|
73
|
+
if not (13 <= len(digits) <= 19) or not digits.isdigit():
|
|
74
|
+
return None
|
|
75
|
+
return 0.97 if luhn_is_valid(digits) else None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def iban_is_valid(value: str) -> bool:
|
|
79
|
+
value = value.upper()
|
|
80
|
+
rearranged = value[4:] + value[:4]
|
|
81
|
+
converted = "".join(str(int(c, 36)) for c in rearranged)
|
|
82
|
+
try:
|
|
83
|
+
return int(converted) % 97 == 1
|
|
84
|
+
except ValueError:
|
|
85
|
+
return False
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _validate_iban(value: str) -> float | None:
|
|
89
|
+
return 0.9 if iban_is_valid(value) else None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _validate_ssn_dashed(value: str) -> float | None:
|
|
93
|
+
area = value[:3]
|
|
94
|
+
if area in ("000", "666") or area.startswith("9"):
|
|
95
|
+
return None
|
|
96
|
+
return 0.95
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _validate_ssn_plain(value: str) -> float | None:
|
|
100
|
+
# Bare 9-digit numbers are ambiguous (order IDs, phone numbers without
|
|
101
|
+
# formatting, etc.) -- start low, let context.py decide if it's really an SSN.
|
|
102
|
+
return 0.35
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# type -> (regex, base_confidence, validator)
|
|
106
|
+
PATTERNS: dict[PIIType, list[tuple[re.Pattern, float, "callable | None"]]] = {
|
|
107
|
+
PIIType.EMAIL: [(EMAIL_RE, 0.95, None)],
|
|
108
|
+
PIIType.PHONE: [(PHONE_RE, 0.85, None)],
|
|
109
|
+
PIIType.SSN: [
|
|
110
|
+
(SSN_DASHED_RE, 0.95, _validate_ssn_dashed),
|
|
111
|
+
(SSN_PLAIN_RE, 0.35, _validate_ssn_plain),
|
|
112
|
+
],
|
|
113
|
+
PIIType.CREDIT_CARD: [(CREDIT_CARD_RE, 0.9, _validate_credit_card)],
|
|
114
|
+
PIIType.IP_ADDRESS: [(IPV4_RE, 0.75, None), (IPV6_RE, 0.85, None)],
|
|
115
|
+
PIIType.AWS_KEY: [(AWS_KEY_RE, 0.97, None)],
|
|
116
|
+
PIIType.API_KEY: [
|
|
117
|
+
(API_KEY_RE, 0.95, None),
|
|
118
|
+
(GENERIC_SECRET_ASSIGNMENT_RE, 0.6, None),
|
|
119
|
+
],
|
|
120
|
+
PIIType.JWT: [(JWT_RE, 0.9, None)],
|
|
121
|
+
PIIType.IBAN: [(IBAN_RE, 0.6, _validate_iban)],
|
|
122
|
+
PIIType.ADDRESS: [(ADDRESS_RE, 0.7, None)],
|
|
123
|
+
}
|
|
File without changes
|