medguardx-core 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ # Python
2
+ __pycache__/
3
+ *.pyc
4
+ venv/
5
+ .venv/
6
+ *.egg-info/
7
+ build/
8
+ dist/
9
+ .pytest_cache/
10
+ testenv/
11
+
12
+ # Node / Next
13
+ node_modules/
14
+ .next/
15
+ out/
16
+
17
+ # Data & secrets
18
+ *.db
19
+ *.db-wal
20
+ *.db-shm
21
+ .env
22
+ !.env.example
23
+
24
+ # OS / editor
25
+ .DS_Store
26
+ *.mov
27
+
28
+ # TS build info
29
+ *.tsbuildinfo
@@ -0,0 +1,127 @@
1
+ Metadata-Version: 2.5
2
+ Name: medguardx-core
3
+ Version: 1.0.0
4
+ Summary: Context-aware PII/PHI detection and masking engine for healthcare data. Stateless, model-configurable, framework-agnostic.
5
+ Project-URL: Homepage, https://github.com/adarshcod30/MedGuardX
6
+ Project-URL: Repository, https://github.com/adarshcod30/MedGuardX
7
+ Project-URL: Issues, https://github.com/adarshcod30/MedGuardX/issues
8
+ Author-email: Adarsh Dwivedi <23ucs509@lnmiit.ac.in>
9
+ License: Apache-2.0
10
+ Keywords: anonymization,dpdp,gdpr,healthcare,hipaa,masking,phi,pii,presidio
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Healthcare Industry
14
+ Classifier: License :: OSI Approved :: Apache Software License
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.9
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Classifier: Topic :: Security
23
+ Classifier: Topic :: Text Processing :: Linguistic
24
+ Requires-Python: >=3.9
25
+ Requires-Dist: presidio-analyzer>=2.2.354
26
+ Requires-Dist: presidio-anonymizer>=2.2.354
27
+ Requires-Dist: spacy<3.9,>=3.7
28
+ Provides-Extra: dev
29
+ Requires-Dist: build>=1.0; extra == 'dev'
30
+ Requires-Dist: pytest-cov>=4; extra == 'dev'
31
+ Requires-Dist: pytest>=7; extra == 'dev'
32
+ Requires-Dist: twine>=5.0; extra == 'dev'
33
+ Provides-Extra: trf
34
+ Requires-Dist: spacy-transformers>=1.3; extra == 'trf'
35
+ Description-Content-Type: text/markdown
36
+
37
+ # medguardx-core
38
+
39
+ Context-aware **PII/PHI detection and masking** for healthcare data — a stateless,
40
+ model-configurable, framework-agnostic Python engine. This is the reusable core of
41
+ [MedGuardX](https://github.com/adarshcod30/MedGuardX); embed it in any project.
42
+
43
+ ## Why
44
+
45
+ - **Stateless & safe to embed** — text and context in, masked text out. No auth, no
46
+ database, no global state. Your app owns storage, identity, and audit.
47
+ - **Bring your own model** — any installed spaCy English pipeline works
48
+ (`en_core_web_sm` / `md` / `lg` / `trf`). Pick your accuracy/RAM trade-off.
49
+ - **Model-independent structured IDs** — Aadhaar (Verhoeff-validated), PAN, MRN,
50
+ credit cards, IBAN, IP are matched by format, so they work identically on every
51
+ model.
52
+ - **Deterministic, leak-proof masking** — overlapping detections are resolved by a
53
+ fixed priority, and no strategy ever leaves part of a detected identifier visible.
54
+
55
+ ## Install
56
+
57
+ ```bash
58
+ pip install medguardx-core
59
+ python -m spacy download en_core_web_md # recommended default
60
+ # other options: en_core_web_sm (smallest) · en_core_web_lg (best statistical)
61
+ # for the transformer model: pip install "medguardx-core[trf]" && python -m spacy download en_core_web_trf
62
+ ```
63
+
64
+ The engine is model-agnostic — install any spaCy English pipeline and pass its name
65
+ to `EngineConfig(model=...)`. (Model wheels aren't declared as dependencies because
66
+ spaCy models aren't on PyPI; `spacy download` is the standard way to fetch them.)
67
+
68
+ ## Quickstart
69
+
70
+ ```python
71
+ from medguardx import MedGuardEngine, EngineConfig, Role, Purpose
72
+
73
+ engine = MedGuardEngine(EngineConfig(model="en_core_web_md"))
74
+
75
+ result = engine.process(
76
+ "Patient John Smith, Aadhaar 2341 2341 2346, card 4111 1111 1111 1111.",
77
+ role=Role.NURSE, purpose=Purpose.TREATMENT, consent=False,
78
+ )
79
+
80
+ print(result.masking_strategy if False else result.strategy.value) # partial_mask
81
+ print(result.masked_text)
82
+ # Patient [NAME_REDACTED], Aadhaar [AADHAAR_REDACTED], card [CARD ****1111].
83
+ ```
84
+
85
+ Compose the steps yourself when you need to:
86
+
87
+ ```python
88
+ entities = engine.detect(text) # list[PIIEntity]
89
+ strategy, rule = engine.evaluate_policy(Role.DOCTOR, Purpose.RESEARCH, consent=False)
90
+ masked = engine.mask(text, entities, strategy)
91
+ ```
92
+
93
+ ## Configuration
94
+
95
+ `EngineConfig(model=..., score_threshold=..., entities=[...], enable_custom_recognizers=True)`.
96
+
97
+ `DATE_TIME` is intentionally **not** in the default entity set — as a high-confidence
98
+ span it used to shadow phone numbers and Aadhaar. Add it back explicitly if you need
99
+ date masking.
100
+
101
+ ## Custom policy
102
+
103
+ ```python
104
+ from medguardx import PolicyEngine, MedGuardEngine, Role, Purpose, MaskingStrategy
105
+
106
+ rules = {(Role.COMPANY, Purpose.TREATMENT, True): (MaskingStrategy.PARTIAL_MASK, "vendor SLA")}
107
+ engine = MedGuardEngine(policy=PolicyEngine(rules=rules)) # deny-by-default for the rest
108
+ ```
109
+
110
+ ## Masking strategies
111
+
112
+ | Strategy | Behaviour |
113
+ |---|---|
114
+ | `full_access` | text returned unchanged |
115
+ | `partial_mask` | identifiers redacted; card/phone/email keep a minimal safe hint |
116
+ | `full_anonymize` | every identifier replaced with a typed `[TYPE_REDACTED]` token |
117
+ | `deny` | access refused; no data returned |
118
+
119
+ ## Ingestion (optional)
120
+
121
+ `medguardx.ingestion.extract_text(filename, bytes)` pulls text from plain text, PDF
122
+ (`pdfplumber`), images (`pytesseract`), and HL7 (`hl7apy`). Those extractors import
123
+ their heavy deps lazily.
124
+
125
+ ## License
126
+
127
+ Apache-2.0.
@@ -0,0 +1,91 @@
1
+ # medguardx-core
2
+
3
+ Context-aware **PII/PHI detection and masking** for healthcare data — a stateless,
4
+ model-configurable, framework-agnostic Python engine. This is the reusable core of
5
+ [MedGuardX](https://github.com/adarshcod30/MedGuardX); embed it in any project.
6
+
7
+ ## Why
8
+
9
+ - **Stateless & safe to embed** — text and context in, masked text out. No auth, no
10
+ database, no global state. Your app owns storage, identity, and audit.
11
+ - **Bring your own model** — any installed spaCy English pipeline works
12
+ (`en_core_web_sm` / `md` / `lg` / `trf`). Pick your accuracy/RAM trade-off.
13
+ - **Model-independent structured IDs** — Aadhaar (Verhoeff-validated), PAN, MRN,
14
+ credit cards, IBAN, IP are matched by format, so they work identically on every
15
+ model.
16
+ - **Deterministic, leak-proof masking** — overlapping detections are resolved by a
17
+ fixed priority, and no strategy ever leaves part of a detected identifier visible.
18
+
19
+ ## Install
20
+
21
+ ```bash
22
+ pip install medguardx-core
23
+ python -m spacy download en_core_web_md # recommended default
24
+ # other options: en_core_web_sm (smallest) · en_core_web_lg (best statistical)
25
+ # for the transformer model: pip install "medguardx-core[trf]" && python -m spacy download en_core_web_trf
26
+ ```
27
+
28
+ The engine is model-agnostic — install any spaCy English pipeline and pass its name
29
+ to `EngineConfig(model=...)`. (Model wheels aren't declared as dependencies because
30
+ spaCy models aren't on PyPI; `spacy download` is the standard way to fetch them.)
31
+
32
+ ## Quickstart
33
+
34
+ ```python
35
+ from medguardx import MedGuardEngine, EngineConfig, Role, Purpose
36
+
37
+ engine = MedGuardEngine(EngineConfig(model="en_core_web_md"))
38
+
39
+ result = engine.process(
40
+ "Patient John Smith, Aadhaar 2341 2341 2346, card 4111 1111 1111 1111.",
41
+ role=Role.NURSE, purpose=Purpose.TREATMENT, consent=False,
42
+ )
43
+
44
+ print(result.masking_strategy if False else result.strategy.value) # partial_mask
45
+ print(result.masked_text)
46
+ # Patient [NAME_REDACTED], Aadhaar [AADHAAR_REDACTED], card [CARD ****1111].
47
+ ```
48
+
49
+ Compose the steps yourself when you need to:
50
+
51
+ ```python
52
+ entities = engine.detect(text) # list[PIIEntity]
53
+ strategy, rule = engine.evaluate_policy(Role.DOCTOR, Purpose.RESEARCH, consent=False)
54
+ masked = engine.mask(text, entities, strategy)
55
+ ```
56
+
57
+ ## Configuration
58
+
59
+ `EngineConfig(model=..., score_threshold=..., entities=[...], enable_custom_recognizers=True)`.
60
+
61
+ `DATE_TIME` is intentionally **not** in the default entity set — as a high-confidence
62
+ span it used to shadow phone numbers and Aadhaar. Add it back explicitly if you need
63
+ date masking.
64
+
65
+ ## Custom policy
66
+
67
+ ```python
68
+ from medguardx import PolicyEngine, MedGuardEngine, Role, Purpose, MaskingStrategy
69
+
70
+ rules = {(Role.COMPANY, Purpose.TREATMENT, True): (MaskingStrategy.PARTIAL_MASK, "vendor SLA")}
71
+ engine = MedGuardEngine(policy=PolicyEngine(rules=rules)) # deny-by-default for the rest
72
+ ```
73
+
74
+ ## Masking strategies
75
+
76
+ | Strategy | Behaviour |
77
+ |---|---|
78
+ | `full_access` | text returned unchanged |
79
+ | `partial_mask` | identifiers redacted; card/phone/email keep a minimal safe hint |
80
+ | `full_anonymize` | every identifier replaced with a typed `[TYPE_REDACTED]` token |
81
+ | `deny` | access refused; no data returned |
82
+
83
+ ## Ingestion (optional)
84
+
85
+ `medguardx.ingestion.extract_text(filename, bytes)` pulls text from plain text, PDF
86
+ (`pdfplumber`), images (`pytesseract`), and HL7 (`hl7apy`). Those extractors import
87
+ their heavy deps lazily.
88
+
89
+ ## License
90
+
91
+ Apache-2.0.
@@ -0,0 +1,54 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "medguardx-core"
7
+ version = "1.0.0"
8
+ description = "Context-aware PII/PHI detection and masking engine for healthcare data. Stateless, model-configurable, framework-agnostic."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "Apache-2.0" }
12
+ authors = [{ name = "Adarsh Dwivedi", email = "23ucs509@lnmiit.ac.in" }]
13
+ keywords = ["pii", "phi", "healthcare", "masking", "anonymization", "presidio", "hipaa", "dpdp", "gdpr"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Healthcare Industry",
17
+ "Intended Audience :: Developers",
18
+ "License :: OSI Approved :: Apache Software License",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.9",
22
+ "Programming Language :: Python :: 3.10",
23
+ "Programming Language :: Python :: 3.11",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Topic :: Security",
26
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
27
+ "Topic :: Text Processing :: Linguistic",
28
+ ]
29
+
30
+ dependencies = [
31
+ "presidio-analyzer>=2.2.354",
32
+ "presidio-anonymizer>=2.2.354",
33
+ "spacy>=3.7,<3.9",
34
+ ]
35
+
36
+ # The engine is model-agnostic: after install, choose any spaCy English pipeline
37
+ # with `python -m spacy download en_core_web_{sm,md,lg,trf}` and pass its name to
38
+ # EngineConfig(model=...). Model wheels are NOT declared as dependencies because
39
+ # PyPI forbids direct-URL requirements and spaCy models are not on PyPI.
40
+ [project.optional-dependencies]
41
+ # The transformer pipeline (en_core_web_trf) additionally needs spacy-transformers.
42
+ trf = ["spacy-transformers>=1.3"]
43
+ dev = ["pytest>=7", "pytest-cov>=4", "build>=1.0", "twine>=5.0"]
44
+
45
+ [project.urls]
46
+ Homepage = "https://github.com/adarshcod30/MedGuardX"
47
+ Repository = "https://github.com/adarshcod30/MedGuardX"
48
+ Issues = "https://github.com/adarshcod30/MedGuardX/issues"
49
+
50
+ [tool.hatch.build.targets.wheel]
51
+ packages = ["src/medguardx"]
52
+
53
+ [tool.pytest.ini_options]
54
+ testpaths = ["tests"]
@@ -0,0 +1,40 @@
1
+ """MedGuardX core -- context-aware PII/PHI detection and masking.
2
+
3
+ Public API::
4
+
5
+ from medguardx import (
6
+ MedGuardEngine, EngineConfig, ProcessResult,
7
+ PolicyEngine, PIIEntity,
8
+ Role, Purpose, MaskingStrategy,
9
+ )
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from .config import DEFAULT_ENTITIES, EngineConfig
14
+ from .detection import Detector, PIIEntity, resolve_overlaps
15
+ from .engine import MedGuardEngine, ProcessResult
16
+ from .enums import MaskingStrategy, Purpose, Role
17
+ from .ingestion import ExtractionError, extract_text
18
+ from .masking import mask_text
19
+ from .policy import DEFAULT_RULES, PolicyEngine
20
+
21
+ __version__ = "1.0.0"
22
+
23
+ __all__ = [
24
+ "MedGuardEngine",
25
+ "EngineConfig",
26
+ "ProcessResult",
27
+ "Detector",
28
+ "PIIEntity",
29
+ "resolve_overlaps",
30
+ "PolicyEngine",
31
+ "DEFAULT_RULES",
32
+ "DEFAULT_ENTITIES",
33
+ "ExtractionError",
34
+ "extract_text",
35
+ "mask_text",
36
+ "Role",
37
+ "Purpose",
38
+ "MaskingStrategy",
39
+ "__version__",
40
+ ]
@@ -0,0 +1,81 @@
1
+ """Engine configuration.
2
+
3
+ Everything an integrator might want to tune lives here. The most important knob
4
+ is ``model``: any installed spaCy English pipeline works (en_core_web_sm / md /
5
+ lg / trf). Nothing else in the engine changes when you swap it -- structured
6
+ identifiers are matched by regex recognizers that are model-independent.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field
11
+ from typing import List
12
+
13
+ # Entities detected by default.
14
+ #
15
+ # NOTE: DATE_TIME is deliberately EXCLUDED from the default set. In the old build
16
+ # spaCy tagged nearly every digit run (phone numbers, Aadhaar, card numbers) as a
17
+ # high-confidence DATE_TIME, which hijacked the span and defeated the specialized
18
+ # recognizers -- the root cause of "some data gets masked, some not". Integrators
19
+ # who genuinely need date masking can add it back explicitly.
20
+ DEFAULT_ENTITIES: List[str] = [
21
+ "PERSON",
22
+ "PHONE_NUMBER",
23
+ "EMAIL_ADDRESS",
24
+ "CREDIT_CARD",
25
+ "IBAN_CODE",
26
+ "IP_ADDRESS",
27
+ "LOCATION",
28
+ "NRP",
29
+ "MEDICAL_LICENSE",
30
+ "URL",
31
+ "US_SSN",
32
+ # Custom, model-independent recognizers:
33
+ "IN_AADHAAR",
34
+ "IN_PAN",
35
+ "MEDICAL_RECORD_NUMBER",
36
+ ]
37
+
38
+ # Overlap-resolution priority. When two detections overlap, the one with the
39
+ # higher priority wins regardless of raw score. More specific / structured
40
+ # identifiers rank above generic linguistic ones so a phone number never loses to
41
+ # a stray PERSON or DATE_TIME span. Higher number = higher priority.
42
+ ENTITY_PRIORITY = {
43
+ "IN_AADHAAR": 100,
44
+ "IN_PAN": 100,
45
+ "US_SSN": 100,
46
+ "CREDIT_CARD": 95,
47
+ "IBAN_CODE": 95,
48
+ "MEDICAL_RECORD_NUMBER": 90,
49
+ "MEDICAL_LICENSE": 85,
50
+ "EMAIL_ADDRESS": 80,
51
+ "PHONE_NUMBER": 75,
52
+ "IP_ADDRESS": 70,
53
+ "URL": 40,
54
+ "PERSON": 60,
55
+ "LOCATION": 55,
56
+ "NRP": 50,
57
+ "DATE_TIME": 20,
58
+ }
59
+
60
+
61
+ @dataclass
62
+ class EngineConfig:
63
+ """Configuration for a :class:`~medguardx.engine.MedGuardEngine`.
64
+
65
+ Attributes:
66
+ model: Name of the installed spaCy pipeline to use. Pick per your
67
+ accuracy/RAM trade-off: ``en_core_web_sm`` (~50MB, weakest),
68
+ ``en_core_web_md`` (~120MB, recommended default), ``en_core_web_lg``
69
+ (~600MB, best statistical), ``en_core_web_trf`` (transformer, highest
70
+ accuracy, heaviest).
71
+ score_threshold: Minimum confidence for a detection to be kept.
72
+ entities: Which entity types to detect.
73
+ enable_custom_recognizers: Register the Aadhaar/PAN/MRN recognizers.
74
+ default_language: Language code passed to Presidio.
75
+ """
76
+
77
+ model: str = "en_core_web_md"
78
+ score_threshold: float = 0.35
79
+ entities: List[str] = field(default_factory=lambda: list(DEFAULT_ENTITIES))
80
+ enable_custom_recognizers: bool = True
81
+ default_language: str = "en"
@@ -0,0 +1,132 @@
1
+ """PII/PHI detection built on Microsoft Presidio.
2
+
3
+ Two things here fix the old build's inconsistent masking:
4
+
5
+ 1. The spaCy model is configurable (``EngineConfig.model``) instead of hardcoded.
6
+ 2. Overlapping detections are resolved by a deterministic priority pass, so a
7
+ specialized entity (phone, Aadhaar, card) can never be shadowed by a generic
8
+ high-score span. In the old build the winner was whatever Presidio happened to
9
+ score highest, which varied with phrasing -- the source of the flakiness.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import asdict, dataclass
14
+ from typing import Dict, List, Optional
15
+
16
+ from .config import ENTITY_PRIORITY, EngineConfig
17
+
18
+
19
+ @dataclass
20
+ class PIIEntity:
21
+ """A single detected entity with its exact span in the source text."""
22
+
23
+ entity_type: str
24
+ start: int
25
+ end: int
26
+ score: float
27
+ text: str
28
+
29
+ def to_dict(self) -> Dict:
30
+ return asdict(self)
31
+
32
+ @classmethod
33
+ def from_dict(cls, d: Dict) -> "PIIEntity":
34
+ return cls(
35
+ entity_type=d["entity_type"],
36
+ start=int(d["start"]),
37
+ end=int(d["end"]),
38
+ score=float(d["score"]),
39
+ text=d.get("text", ""),
40
+ )
41
+
42
+
43
+ class Detector:
44
+ """Lazily-initialised Presidio analyzer wrapper.
45
+
46
+ The heavy analyzer/model load happens on first use, not at import time, so
47
+ importing :mod:`medguardx` stays cheap.
48
+ """
49
+
50
+ def __init__(self, config: Optional[EngineConfig] = None) -> None:
51
+ self.config = config or EngineConfig()
52
+ self._analyzer = None
53
+
54
+ def _get_analyzer(self):
55
+ if self._analyzer is not None:
56
+ return self._analyzer
57
+
58
+ from presidio_analyzer import AnalyzerEngine, RecognizerRegistry
59
+ from presidio_analyzer.nlp_engine import NlpEngineProvider
60
+
61
+ provider = NlpEngineProvider(
62
+ nlp_configuration={
63
+ "nlp_engine_name": "spacy",
64
+ "models": [{"lang_code": self.config.default_language, "model_name": self.config.model}],
65
+ }
66
+ )
67
+ nlp_engine = provider.create_engine()
68
+
69
+ registry = RecognizerRegistry()
70
+ registry.load_predefined_recognizers(languages=[self.config.default_language])
71
+ if self.config.enable_custom_recognizers:
72
+ from .recognizers import build_custom_recognizers
73
+
74
+ for recognizer in build_custom_recognizers():
75
+ registry.add_recognizer(recognizer)
76
+
77
+ self._analyzer = AnalyzerEngine(
78
+ nlp_engine=nlp_engine,
79
+ registry=registry,
80
+ supported_languages=[self.config.default_language],
81
+ )
82
+ return self._analyzer
83
+
84
+ def detect(self, text: str, entities: Optional[List[str]] = None) -> List[PIIEntity]:
85
+ """Detect PII entities in ``text`` and return a non-overlapping, sorted list."""
86
+ if not text or not text.strip():
87
+ return []
88
+
89
+ analyzer = self._get_analyzer()
90
+ raw = analyzer.analyze(
91
+ text=text,
92
+ language=self.config.default_language,
93
+ entities=entities or self.config.entities,
94
+ score_threshold=self.config.score_threshold,
95
+ )
96
+
97
+ found = [
98
+ PIIEntity(
99
+ entity_type=r.entity_type,
100
+ start=r.start,
101
+ end=r.end,
102
+ score=round(float(r.score), 2),
103
+ text=text[r.start : r.end],
104
+ )
105
+ for r in raw
106
+ ]
107
+ return resolve_overlaps(found)
108
+
109
+
110
+ def resolve_overlaps(entities: List[PIIEntity]) -> List[PIIEntity]:
111
+ """Collapse overlapping spans, keeping the highest-priority entity per region.
112
+
113
+ Priority is decided first by :data:`ENTITY_PRIORITY` (so a PHONE_NUMBER beats
114
+ a DATE_TIME even at lower raw score), then by score, then by span length. The
115
+ result is deterministic: the same text always yields the same masking.
116
+ """
117
+ if not entities:
118
+ return []
119
+
120
+ def rank(e: PIIEntity):
121
+ return (ENTITY_PRIORITY.get(e.entity_type, 30), e.score, e.end - e.start)
122
+
123
+ # Strongest candidates first; greedily accept ones that don't overlap an
124
+ # already-accepted (higher-ranked) span.
125
+ ordered = sorted(entities, key=rank, reverse=True)
126
+ accepted: List[PIIEntity] = []
127
+ for cand in ordered:
128
+ if not any(cand.start < a.end and a.start < cand.end for a in accepted):
129
+ accepted.append(cand)
130
+
131
+ accepted.sort(key=lambda e: e.start)
132
+ return accepted
@@ -0,0 +1,101 @@
1
+ """The public MedGuardX engine -- a stateless facade over detection, policy and masking.
2
+
3
+ Typical use::
4
+
5
+ from medguardx import MedGuardEngine, EngineConfig, Role, Purpose
6
+
7
+ engine = MedGuardEngine(EngineConfig(model="en_core_web_md"))
8
+ result = engine.process(
9
+ "Patient John Smith, Aadhaar 2341 2341 2341, card 4111 1111 1111 1111.",
10
+ role=Role.NURSE, purpose=Purpose.TREATMENT, consent=False,
11
+ )
12
+ print(result.masked_text)
13
+
14
+ The engine holds no data and does no I/O -- integrators own storage, auth and
15
+ audit. That statelessness is deliberate: it is what makes the engine safe to
16
+ embed anywhere.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ from dataclasses import dataclass, field
21
+ from typing import List, Optional
22
+
23
+ from .config import EngineConfig
24
+ from .detection import Detector, PIIEntity
25
+ from .enums import MaskingStrategy, Purpose, Role, coerce_purpose, coerce_role
26
+ from .masking import mask_text
27
+ from .policy import PolicyEngine
28
+
29
+
30
+ @dataclass
31
+ class ProcessResult:
32
+ """Outcome of :meth:`MedGuardEngine.process`."""
33
+
34
+ original_text: str
35
+ masked_text: str
36
+ entities: List[PIIEntity]
37
+ strategy: MaskingStrategy
38
+ policy_rule: str
39
+ denied: bool = field(init=False)
40
+
41
+ def __post_init__(self) -> None:
42
+ self.denied = self.strategy == MaskingStrategy.DENY
43
+
44
+ def to_dict(self) -> dict:
45
+ return {
46
+ "original_text": self.original_text,
47
+ "masked_text": self.masked_text,
48
+ "entities": [e.to_dict() for e in self.entities],
49
+ "masking_strategy": self.strategy.value,
50
+ "policy_rule": self.policy_rule,
51
+ "entities_masked": 0 if self.strategy == MaskingStrategy.FULL_ACCESS else len(self.entities),
52
+ "denied": self.denied,
53
+ }
54
+
55
+
56
+ class MedGuardEngine:
57
+ """Detect -> evaluate policy -> mask, in one call or as separate steps."""
58
+
59
+ def __init__(
60
+ self,
61
+ config: Optional[EngineConfig] = None,
62
+ policy: Optional[PolicyEngine] = None,
63
+ ) -> None:
64
+ self.config = config or EngineConfig()
65
+ self.detector = Detector(self.config)
66
+ self.policy = policy or PolicyEngine()
67
+
68
+ # --- composable steps -------------------------------------------------------
69
+ def detect(self, text: str) -> List[PIIEntity]:
70
+ return self.detector.detect(text)
71
+
72
+ def evaluate_policy(self, role, purpose, consent: bool):
73
+ return self.policy.evaluate(role, purpose, consent)
74
+
75
+ def mask(self, text: str, entities: List[PIIEntity], strategy: MaskingStrategy) -> str:
76
+ return mask_text(text, entities, strategy)
77
+
78
+ # --- one-shot ---------------------------------------------------------------
79
+ def process(
80
+ self,
81
+ text: str,
82
+ role,
83
+ purpose,
84
+ consent: bool = False,
85
+ entities: Optional[List[PIIEntity]] = None,
86
+ ) -> ProcessResult:
87
+ """Run the full pipeline. Pass pre-computed ``entities`` to skip detection."""
88
+ role = coerce_role(role)
89
+ purpose = coerce_purpose(purpose)
90
+ strategy, rule = self.policy.evaluate(role, purpose, consent)
91
+
92
+ if strategy == MaskingStrategy.DENY:
93
+ return ProcessResult(text, "", [], strategy, rule)
94
+
95
+ detected = entities if entities is not None else self.detect(text)
96
+ masked = self.mask(text, detected, strategy)
97
+ return ProcessResult(text, masked, detected, strategy, rule)
98
+
99
+ def warm_up(self) -> None:
100
+ """Eagerly load the model (useful at service startup)."""
101
+ self.detector._get_analyzer()
@@ -0,0 +1,47 @@
1
+ """Core enumerations shared across the MedGuardX engine.
2
+
3
+ These are intentionally plain ``str`` enums so they serialize cleanly to JSON and
4
+ compare equal to their string values -- an integrator can pass ``"nurse"`` or
5
+ ``Role.NURSE`` interchangeably.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from enum import Enum
10
+
11
+
12
+ class Role(str, Enum):
13
+ """Who is requesting the data."""
14
+
15
+ DOCTOR = "doctor"
16
+ NURSE = "nurse"
17
+ RESEARCHER = "researcher"
18
+ PATIENT = "patient"
19
+ COMPANY = "company"
20
+ ADMIN = "admin"
21
+
22
+
23
+ class Purpose(str, Enum):
24
+ """Why the data is being requested."""
25
+
26
+ TREATMENT = "treatment"
27
+ RESEARCH = "research"
28
+ BILLING = "billing"
29
+ LEGAL = "legal"
30
+ PERSONAL = "personal"
31
+
32
+
33
+ class MaskingStrategy(str, Enum):
34
+ """How much of the data the requester is allowed to see."""
35
+
36
+ FULL_ACCESS = "full_access"
37
+ PARTIAL_MASK = "partial_mask"
38
+ FULL_ANONYMIZE = "full_anonymize"
39
+ DENY = "deny"
40
+
41
+
42
+ def coerce_role(value) -> "Role":
43
+ return value if isinstance(value, Role) else Role(str(value).lower())
44
+
45
+
46
+ def coerce_purpose(value) -> "Purpose":
47
+ return value if isinstance(value, Purpose) else Purpose(str(value).lower())
@@ -0,0 +1,104 @@
1
+ """Multi-format text extraction: plain text, PDF, image (OCR), HL7.
2
+
3
+ Optional heavy dependencies (pdfplumber, pytesseract/Pillow, hl7apy) are imported
4
+ lazily and degrade gracefully, so installing the core engine does not force an
5
+ OCR/PDF toolchain on integrators who only mask plain text.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import io
10
+ import os
11
+ from typing import Tuple
12
+
13
+
14
+ class ExtractionError(RuntimeError):
15
+ """Raised when text cannot be extracted from an uploaded file.
16
+
17
+ Distinct from "no text found": this signals a real failure (corrupt file, a
18
+ missing OCR engine, etc.) so callers can reject the upload rather than
19
+ silently storing an error string as the record's content.
20
+ """
21
+
22
+
23
+ def detect_file_type(filename: str, content: bytes) -> str:
24
+ ext = os.path.splitext(filename or "")[1].lower()
25
+ if ext in (".hl7", ".adt"):
26
+ return "hl7"
27
+ if ext == ".pdf":
28
+ return "pdf"
29
+ if ext in (".png", ".jpg", ".jpeg", ".tiff", ".bmp"):
30
+ return "image"
31
+ preview = content[:200].decode("utf-8", errors="ignore")
32
+ if "MSH|" in preview:
33
+ return "hl7"
34
+ return "text"
35
+
36
+
37
+ def extract_text_from_pdf(content: bytes) -> str:
38
+ try:
39
+ import pdfplumber
40
+
41
+ with pdfplumber.open(io.BytesIO(content)) as pdf:
42
+ pages = [page.extract_text() or "" for page in pdf.pages]
43
+ return "\n".join(pages).strip()
44
+ except Exception as exc: # pragma: no cover - depends on optional dep
45
+ raise ExtractionError(f"PDF extraction failed: {exc}") from exc
46
+
47
+
48
+ def extract_text_from_image(content: bytes) -> str:
49
+ try:
50
+ import tempfile
51
+
52
+ import pytesseract
53
+ from PIL import Image
54
+ except Exception as exc: # pragma: no cover - optional deps not installed
55
+ raise ExtractionError("OCR dependencies (pytesseract/Pillow) are not installed.") from exc
56
+
57
+ try:
58
+ img = Image.open(io.BytesIO(content))
59
+ with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as fh:
60
+ img.save(fh.name)
61
+ try:
62
+ return pytesseract.image_to_string(fh.name).strip()
63
+ finally:
64
+ os.unlink(fh.name)
65
+ except pytesseract.TesseractNotFoundError as exc: # pragma: no cover
66
+ raise ExtractionError(
67
+ "OCR engine (tesseract) is not available in this deployment. "
68
+ "Image files are not supported here; use text, PDF, or HL7."
69
+ ) from exc
70
+ except Exception as exc: # pragma: no cover - depends on optional dep
71
+ raise ExtractionError(f"OCR failed: {exc}") from exc
72
+
73
+
74
+ def parse_hl7(content: bytes) -> str:
75
+ text = content.decode("utf-8", errors="ignore")
76
+ try:
77
+ from hl7apy.parser import parse_message
78
+
79
+ msg = parse_message(text.replace("\n", "\r"))
80
+ segments = []
81
+ for seg in msg.children:
82
+ fields = []
83
+ for field in seg.children:
84
+ try:
85
+ fields.append(f"{field.name}: {field.value}")
86
+ except Exception:
87
+ pass
88
+ if fields:
89
+ segments.append(f"[{seg.name}] " + " | ".join(fields))
90
+ return "\n".join(segments) if segments else text
91
+ except Exception: # pragma: no cover - depends on optional dep
92
+ return text
93
+
94
+
95
+ def extract_text(filename: str, content: bytes) -> Tuple[str, str]:
96
+ """Return ``(extracted_text, file_type)`` for an uploaded file."""
97
+ file_type = detect_file_type(filename, content)
98
+ if file_type == "pdf":
99
+ return extract_text_from_pdf(content), file_type
100
+ if file_type == "image":
101
+ return extract_text_from_image(content), file_type
102
+ if file_type == "hl7":
103
+ return parse_hl7(content), file_type
104
+ return content.decode("utf-8", errors="ignore"), file_type
@@ -0,0 +1,107 @@
1
+ """Deterministic, leak-proof masking.
2
+
3
+ The old build delegated to Presidio's anonymizer with a DEFAULT operator that
4
+ masked only the first 8 characters -- so a 19-digit card came back as
5
+ ``********1 1111 1111`` and IPs/IBANs leaked their tails. Here we do span
6
+ replacement ourselves over the already-de-overlapped entity list, replacing from
7
+ right to left so indices stay valid. The invariant is simple and testable: for
8
+ any strategy other than FULL_ACCESS, the original entity substring never survives
9
+ in the output.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from typing import Callable, Dict, List
14
+
15
+ from .detection import PIIEntity
16
+ from .enums import MaskingStrategy
17
+
18
+ # Human-readable label per entity type, used in redaction tokens.
19
+ _LABELS: Dict[str, str] = {
20
+ "PERSON": "NAME",
21
+ "PHONE_NUMBER": "PHONE",
22
+ "EMAIL_ADDRESS": "EMAIL",
23
+ "CREDIT_CARD": "CARD",
24
+ "IBAN_CODE": "IBAN",
25
+ "IP_ADDRESS": "IP",
26
+ "LOCATION": "LOCATION",
27
+ "URL": "URL",
28
+ "NRP": "NRP",
29
+ "US_SSN": "SSN",
30
+ "MEDICAL_LICENSE": "MED_LICENSE",
31
+ "MEDICAL_RECORD_NUMBER": "MRN",
32
+ "IN_AADHAAR": "AADHAAR",
33
+ "IN_PAN": "PAN",
34
+ "DATE_TIME": "DATE",
35
+ }
36
+
37
+ DENY_MESSAGE = "[ACCESS DENIED - Insufficient permissions]"
38
+
39
+
40
+ def _label(entity_type: str) -> str:
41
+ return _LABELS.get(entity_type, entity_type)
42
+
43
+
44
+ def _redact(entity: PIIEntity) -> str:
45
+ return f"[{_label(entity.entity_type)}_REDACTED]"
46
+
47
+
48
+ # --- Partial-mask reveal helpers -------------------------------------------------
49
+ # Partial masking trades a *small, bounded* amount of the original for downstream
50
+ # utility (e.g. matching a card by its last 4). Every helper below reveals at most
51
+ # a safe suffix and masks everything else; none can leave the majority visible.
52
+
53
+ def _reveal_last(value: str, keep: int, ch: str = "*") -> str:
54
+ digits = [c for c in value if c.isalnum()]
55
+ if len(digits) <= keep:
56
+ return ch * len(digits)
57
+ return ch * (len(digits) - keep) + "".join(digits[-keep:])
58
+
59
+
60
+ def _partial_card(entity: PIIEntity) -> str:
61
+ return f"[CARD ****{_reveal_last(entity.text, 4)[-4:]}]"
62
+
63
+
64
+ def _partial_phone(entity: PIIEntity) -> str:
65
+ return f"[PHONE ****{_reveal_last(entity.text, 2)[-2:]}]"
66
+
67
+
68
+ def _partial_email(entity: PIIEntity) -> str:
69
+ local, _, domain = entity.text.partition("@")
70
+ if not domain:
71
+ return "[EMAIL_MASKED]"
72
+ hint = (local[:1] + "***") if local else "***"
73
+ return f"{hint}@{domain}"
74
+
75
+
76
+ # Types that get a partial reveal under PARTIAL_MASK. Everything else is fully
77
+ # redacted even under partial masking -- safe by default.
78
+ _PARTIAL_OPERATORS: Dict[str, Callable[[PIIEntity], str]] = {
79
+ "CREDIT_CARD": _partial_card,
80
+ "PHONE_NUMBER": _partial_phone,
81
+ "EMAIL_ADDRESS": _partial_email,
82
+ }
83
+
84
+
85
+ def _mask_token(entity: PIIEntity, strategy: MaskingStrategy) -> str:
86
+ if strategy == MaskingStrategy.PARTIAL_MASK and entity.entity_type in _PARTIAL_OPERATORS:
87
+ return _PARTIAL_OPERATORS[entity.entity_type](entity)
88
+ return _redact(entity)
89
+
90
+
91
+ def mask_text(text: str, entities: List[PIIEntity], strategy: MaskingStrategy) -> str:
92
+ """Apply ``strategy`` to ``text`` given its detected ``entities``."""
93
+ if strategy == MaskingStrategy.FULL_ACCESS:
94
+ return text
95
+ if strategy == MaskingStrategy.DENY:
96
+ return DENY_MESSAGE
97
+ if not entities:
98
+ return text
99
+
100
+ # Replace right-to-left so earlier spans keep their original indices.
101
+ ordered = sorted(entities, key=lambda e: e.start, reverse=True)
102
+ out = text
103
+ for e in ordered:
104
+ if e.start < 0 or e.end > len(out) or e.start >= e.end:
105
+ continue # stale/invalid span -- skip rather than corrupt the text
106
+ out = out[: e.start] + _mask_token(e, strategy) + out[e.end :]
107
+ return out
@@ -0,0 +1,73 @@
1
+ """Context-aware policy engine.
2
+
3
+ Maps ``(role, purpose, consent)`` to a masking strategy plus a human-readable
4
+ rationale. The default matrix is the healthcare policy MedGuardX ships with;
5
+ integrators can subclass :class:`PolicyEngine` or pass a custom ``rules`` dict to
6
+ encode their own governance without touching detection or masking.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from typing import Dict, Optional, Tuple
11
+
12
+ from .enums import MaskingStrategy, Purpose, Role, coerce_purpose, coerce_role
13
+
14
+ PolicyKey = Tuple[Role, Purpose, bool]
15
+ PolicyValue = Tuple[MaskingStrategy, str]
16
+
17
+ _FA = MaskingStrategy.FULL_ACCESS
18
+ _PM = MaskingStrategy.PARTIAL_MASK
19
+ _AN = MaskingStrategy.FULL_ANONYMIZE
20
+ _DN = MaskingStrategy.DENY
21
+
22
+ DEFAULT_RULES: Dict[PolicyKey, PolicyValue] = {
23
+ # Doctors
24
+ (Role.DOCTOR, Purpose.TREATMENT, True): (_FA, "Doctor requesting treatment records with patient consent: full access granted."),
25
+ (Role.DOCTOR, Purpose.TREATMENT, False): (_PM, "Doctor requesting treatment records without consent: partial access (identifiers masked)."),
26
+ (Role.DOCTOR, Purpose.RESEARCH, True): (_PM, "Doctor requesting research data with consent: partial access (identifiers masked)."),
27
+ (Role.DOCTOR, Purpose.RESEARCH, False): (_AN, "Doctor requesting research data without consent: anonymized access only."),
28
+ # Nurses
29
+ (Role.NURSE, Purpose.TREATMENT, True): (_PM, "Nurse requesting treatment records with consent: partial access (identifiers masked)."),
30
+ (Role.NURSE, Purpose.TREATMENT, False): (_PM, "Nurse requesting treatment records without consent: partial access (identifiers masked)."),
31
+ (Role.NURSE, Purpose.RESEARCH, True): (_AN, "Nurse requesting research data with consent: anonymized access only."),
32
+ # Researchers
33
+ (Role.RESEARCHER, Purpose.RESEARCH, True): (_AN, "Researcher requesting research data with consent: anonymized access only."),
34
+ (Role.RESEARCHER, Purpose.RESEARCH, False): (_AN, "Researcher requesting research data without consent: anonymized access only."),
35
+ # Patients (own records)
36
+ (Role.PATIENT, Purpose.PERSONAL, True): (_FA, "Patient accessing personal records: full access."),
37
+ (Role.PATIENT, Purpose.PERSONAL, False): (_FA, "Patient accessing personal records: full access."),
38
+ (Role.PATIENT, Purpose.TREATMENT, True): (_FA, "Patient accessing treatment records: full access."),
39
+ (Role.PATIENT, Purpose.TREATMENT, False): (_FA, "Patient accessing treatment records: full access."),
40
+ (Role.PATIENT, Purpose.BILLING, True): (_FA, "Patient accessing billing records: full access."),
41
+ (Role.PATIENT, Purpose.BILLING, False): (_FA, "Patient accessing billing records: full access."),
42
+ (Role.PATIENT, Purpose.LEGAL, True): (_FA, "Patient retrieving records for legal reasons: full access."),
43
+ (Role.PATIENT, Purpose.LEGAL, False): (_FA, "Patient retrieving records for legal reasons: full access."),
44
+ # Companies
45
+ (Role.COMPANY, Purpose.RESEARCH, True): (_AN, "Company requesting research data with consent: anonymized access only."),
46
+ (Role.COMPANY, Purpose.BILLING, True): (_PM, "Company requesting billing records with consent: partial access (identifiers masked)."),
47
+ # Admins (governance / break-glass, still logged upstream)
48
+ (Role.ADMIN, Purpose.LEGAL, True): (_FA, "Admin performing authorized legal retrieval: full access."),
49
+ (Role.ADMIN, Purpose.LEGAL, False): (_FA, "Admin performing authorized legal retrieval: full access."),
50
+ }
51
+
52
+
53
+ class PolicyEngine:
54
+ """Evaluates access policy. Deny-by-default for any unmapped combination."""
55
+
56
+ def __init__(self, rules: Optional[Dict[PolicyKey, PolicyValue]] = None) -> None:
57
+ self.rules = rules if rules is not None else dict(DEFAULT_RULES)
58
+
59
+ def evaluate(self, role, purpose, consent: bool) -> PolicyValue:
60
+ role = coerce_role(role)
61
+ purpose = coerce_purpose(purpose)
62
+ consent = bool(consent)
63
+
64
+ hit = self.rules.get((role, purpose, consent))
65
+ if hit is not None:
66
+ return hit
67
+
68
+ consent_str = "even with patient consent" if consent else "without patient consent"
69
+ return (
70
+ _DN,
71
+ f"Access denied: {role.value.capitalize()}s are not authorized to access "
72
+ f"records for {purpose.value} purposes, {consent_str}.",
73
+ )
@@ -0,0 +1,56 @@
1
+ """Model-independent pattern recognizers.
2
+
3
+ These run the same way regardless of which spaCy model is loaded (sm / md / lg /
4
+ trf), because structured identifiers -- Aadhaar, PAN, MRN, credit cards -- are
5
+ matched by format, not by linguistic NER. This is what makes the "Indian PII"
6
+ support actually work: it never depended on the model in the first place.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from typing import List
11
+
12
+ from presidio_analyzer import Pattern, PatternRecognizer
13
+
14
+ from .aadhaar import AadhaarRecognizer
15
+
16
+
17
+ def build_custom_recognizers() -> List[PatternRecognizer]:
18
+ """Return the custom recognizers MedGuardX registers by default."""
19
+ return [
20
+ AadhaarRecognizer(),
21
+ _pan_recognizer(),
22
+ _mrn_recognizer(),
23
+ ]
24
+
25
+
26
+ def _pan_recognizer() -> PatternRecognizer:
27
+ """Indian Permanent Account Number: 5 letters, 4 digits, 1 letter (e.g. ABCDE1234F)."""
28
+ return PatternRecognizer(
29
+ supported_entity="IN_PAN",
30
+ name="in_pan_recognizer",
31
+ patterns=[
32
+ Pattern(name="pan", regex=r"\b[A-Z]{5}[0-9]{4}[A-Z]\b", score=0.85),
33
+ ],
34
+ context=["pan", "permanent account", "income tax"],
35
+ )
36
+
37
+
38
+ def _mrn_recognizer() -> PatternRecognizer:
39
+ """Medical Record Number.
40
+
41
+ MRNs have no universal format, so we anchor on the explicit ``MRN``/``medical
42
+ record`` context label followed by an alphanumeric code. Anchoring on the
43
+ label keeps this precise instead of masking every stray number.
44
+ """
45
+ return PatternRecognizer(
46
+ supported_entity="MEDICAL_RECORD_NUMBER",
47
+ name="mrn_recognizer",
48
+ patterns=[
49
+ Pattern(
50
+ name="mrn_labelled",
51
+ regex=r"\b(?:MRN|Medical\s*Record\s*(?:No\.?|Number|#)?)[:\s#-]*([A-Z0-9][A-Z0-9-]{2,14})\b",
52
+ score=0.8,
53
+ ),
54
+ ],
55
+ context=["mrn", "medical record", "patient id", "chart"],
56
+ )
@@ -0,0 +1,93 @@
1
+ """Aadhaar recognizer with Verhoeff checksum validation.
2
+
3
+ A 12-digit Aadhaar is easy to confuse with a phone number, an amount, or a date
4
+ if you match on shape alone -- that is exactly why the old build mis-tagged
5
+ Aadhaar as ``DATE_TIME``. Validating the Verhoeff checksum lets us assign a high
6
+ confidence score and win overlap resolution against the generic recognizers,
7
+ while keeping false positives low.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from typing import List, Optional
13
+
14
+ from presidio_analyzer import EntityRecognizer, RecognizerResult
15
+ from presidio_analyzer.nlp_engine import NlpArtifacts
16
+
17
+ # Verhoeff algorithm tables.
18
+ _D = [
19
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
20
+ [1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
21
+ [2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
22
+ [3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
23
+ [4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
24
+ [5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
25
+ [6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
26
+ [7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
27
+ [8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
28
+ [9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
29
+ ]
30
+ _P = [
31
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
32
+ [1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
33
+ [5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
34
+ [8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
35
+ [9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
36
+ [4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
37
+ [2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
38
+ [7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
39
+ ]
40
+
41
+
42
+ def _verhoeff_valid(number: str) -> bool:
43
+ c = 0
44
+ for i, digit in enumerate(reversed(number)):
45
+ c = _D[c][_P[i % 8][int(digit)]]
46
+ return c == 0
47
+
48
+
49
+ # 12 digits, optionally grouped 4-4-4 by spaces or hyphens. First digit is 2-9.
50
+ # The lookarounds ensure the 12-digit run is not part of a LONGER grouped number
51
+ # (e.g. the first 12 digits of a 16-digit credit card) -- without them the
52
+ # recognizer would shadow CREDIT_CARD and leak the trailing group.
53
+ _AADHAAR_RE = re.compile(
54
+ r"(?<!\d)(?<![\d][\s-])" # not preceded by a digit or a digit-group separator
55
+ r"[2-9][0-9]{3}[\s-]?[0-9]{4}[\s-]?[0-9]{4}"
56
+ r"(?![\s-]?[0-9])" # not followed by another digit group
57
+ )
58
+
59
+
60
+ class AadhaarRecognizer(EntityRecognizer):
61
+ """Detects Indian Aadhaar numbers, verified with the Verhoeff checksum."""
62
+
63
+ def __init__(self) -> None:
64
+ super().__init__(supported_entities=["IN_AADHAAR"], name="aadhaar_recognizer")
65
+
66
+ def load(self) -> None: # required by the EntityRecognizer contract
67
+ return None
68
+
69
+ def analyze(
70
+ self, text: str, entities: List[str], nlp_artifacts: Optional[NlpArtifacts] = None
71
+ ) -> List[RecognizerResult]:
72
+ if "IN_AADHAAR" not in entities:
73
+ return []
74
+
75
+ results: List[RecognizerResult] = []
76
+ for match in _AADHAAR_RE.finditer(text):
77
+ digits = re.sub(r"[\s-]", "", match.group())
78
+ if len(digits) != 12:
79
+ continue
80
+ # A valid Verhoeff checksum -> high confidence. A well-shaped but
81
+ # unverified match still gets a moderate score so it is masked, just
82
+ # with lower priority in overlap resolution.
83
+ score = 0.9 if _verhoeff_valid(digits) else 0.5
84
+ results.append(
85
+ RecognizerResult(
86
+ entity_type="IN_AADHAAR",
87
+ start=match.start(),
88
+ end=match.end(),
89
+ score=score,
90
+ analysis_explanation=None,
91
+ )
92
+ )
93
+ return results
@@ -0,0 +1,41 @@
1
+ """Aadhaar recognizer: Verhoeff-valid numbers score high; junk is rejected."""
2
+ from medguardx.recognizers.aadhaar import AadhaarRecognizer, _verhoeff_valid
3
+
4
+
5
+ def _run(text):
6
+ return AadhaarRecognizer().analyze(text, entities=["IN_AADHAAR"])
7
+
8
+
9
+ def test_verhoeff_known_valid_and_invalid():
10
+ # 234123412346 has a valid Verhoeff check digit; flipping it invalidates.
11
+ assert _verhoeff_valid("234123412346")
12
+ assert not _verhoeff_valid("234123412340")
13
+
14
+
15
+ def test_detects_spaced_aadhaar():
16
+ res = _run("Aadhaar 2341 2341 2346 on record")
17
+ assert len(res) == 1
18
+ assert res[0].entity_type == "IN_AADHAAR"
19
+ assert res[0].score >= 0.9 # valid checksum -> high confidence
20
+
21
+
22
+ def test_detects_bare_twelve_digit_aadhaar():
23
+ # The exact case the old build missed entirely (returned nothing).
24
+ res = _run("ID 234123412346 filed")
25
+ assert len(res) == 1 and res[0].entity_type == "IN_AADHAAR"
26
+
27
+
28
+ def test_shape_match_without_valid_checksum_still_flagged_lower():
29
+ res = _run("Number 234123412340 here")
30
+ assert len(res) == 1
31
+ assert 0.4 <= res[0].score < 0.9
32
+
33
+
34
+ def test_ignores_non_aadhaar_numbers():
35
+ assert _run("Call 9305597756 today") == [] # 10 digits, not Aadhaar-shaped
36
+
37
+
38
+ def test_does_not_match_inside_a_credit_card():
39
+ # 16-digit card must NOT be picked up as a 12-digit Aadhaar.
40
+ assert _run("card 4111 1111 1111 1111 on file") == []
41
+ assert _run("4111111111111111") == []
@@ -0,0 +1,54 @@
1
+ """Masking must never leak the original substring (the old build's core bug)."""
2
+ from medguardx.detection import PIIEntity
3
+ from medguardx.enums import MaskingStrategy
4
+ from medguardx.masking import mask_text
5
+
6
+
7
+ def _ent(text, start, etype, score=0.9):
8
+ return PIIEntity(entity_type=etype, start=start, end=start + len(text), score=score, text=text)
9
+
10
+
11
+ def test_full_access_returns_text_unchanged():
12
+ txt = "Card 4111 1111 1111 1111"
13
+ ents = [_ent("4111 1111 1111 1111", 5, "CREDIT_CARD")]
14
+ assert mask_text(txt, ents, MaskingStrategy.FULL_ACCESS) == txt
15
+
16
+
17
+ def test_full_anonymize_removes_every_entity_substring():
18
+ txt = "IP 192.168.1.55 and IBAN DE89370400440532013000 here."
19
+ ents = [_ent("192.168.1.55", 3, "IP_ADDRESS"), _ent("DE89370400440532013000", 25, "IBAN_CODE")]
20
+ out = mask_text(txt, ents, MaskingStrategy.FULL_ANONYMIZE)
21
+ assert "192.168.1.55" not in out
22
+ assert "DE89370400440532013000" not in out
23
+ assert "[IP_REDACTED]" in out and "[IBAN_REDACTED]" in out
24
+
25
+
26
+ def test_partial_card_reveals_only_last_four():
27
+ txt = "Card 4111111111111111 on file"
28
+ ents = [_ent("4111111111111111", 5, "CREDIT_CARD")]
29
+ out = mask_text(txt, ents, MaskingStrategy.PARTIAL_MASK)
30
+ assert "4111111111111111" not in out
31
+ assert "1111" in out # last 4 kept for utility
32
+ assert out.count("1") == 4 # nothing else from the PAN survives
33
+
34
+
35
+ def test_partial_ip_and_iban_are_fully_redacted_not_leaked():
36
+ # These have no partial operator -> must be fully redacted, unlike the old
37
+ # build which left "********1.55" and "********00440532013000".
38
+ txt = "IP 192.168.1.55, IBAN DE89370400440532013000"
39
+ ents = [_ent("192.168.1.55", 3, "IP_ADDRESS"), _ent("DE89370400440532013000", 22, "IBAN_CODE")]
40
+ out = mask_text(txt, ents, MaskingStrategy.PARTIAL_MASK)
41
+ assert "1.55" not in out
42
+ assert "440532013000" not in out
43
+
44
+
45
+ def test_deny_returns_denied_placeholder_only():
46
+ out = mask_text("secret", [], MaskingStrategy.DENY)
47
+ assert "secret" not in out and "DENIED" in out.upper()
48
+
49
+
50
+ def test_stale_span_is_skipped_not_crashing():
51
+ txt = "short"
52
+ ents = [_ent("this is longer than the text", 0, "PERSON")]
53
+ # end past len(text): span skipped rather than corrupting output
54
+ assert mask_text(txt, ents, MaskingStrategy.FULL_ANONYMIZE) == txt
@@ -0,0 +1,38 @@
1
+ """Overlap resolution: a specialized entity must beat a generic one deterministically."""
2
+ from medguardx.detection import PIIEntity, resolve_overlaps
3
+
4
+
5
+ def _ent(etype, start, end, score):
6
+ return PIIEntity(entity_type=etype, start=start, end=end, score=score, text="x")
7
+
8
+
9
+ def test_phone_beats_datetime_even_at_lower_score():
10
+ # The exact failure from the old build: DATE_TIME(0.85) shadowed PHONE(0.75).
11
+ ents = [_ent("DATE_TIME", 0, 10, 0.85), _ent("PHONE_NUMBER", 0, 10, 0.75)]
12
+ out = resolve_overlaps(ents)
13
+ assert len(out) == 1
14
+ assert out[0].entity_type == "PHONE_NUMBER"
15
+
16
+
17
+ def test_aadhaar_beats_datetime():
18
+ ents = [_ent("DATE_TIME", 5, 19, 0.85), _ent("IN_AADHAAR", 5, 19, 0.9)]
19
+ out = resolve_overlaps(ents)
20
+ assert len(out) == 1 and out[0].entity_type == "IN_AADHAAR"
21
+
22
+
23
+ def test_email_beats_overlapping_url():
24
+ ents = [_ent("URL", 10, 21, 0.5), _ent("EMAIL_ADDRESS", 5, 21, 1.0)]
25
+ out = resolve_overlaps(ents)
26
+ assert len(out) == 1 and out[0].entity_type == "EMAIL_ADDRESS"
27
+
28
+
29
+ def test_non_overlapping_entities_all_kept_and_sorted():
30
+ ents = [_ent("PERSON", 20, 30, 0.9), _ent("PHONE_NUMBER", 0, 10, 0.8)]
31
+ out = resolve_overlaps(ents)
32
+ assert [e.entity_type for e in out] == ["PHONE_NUMBER", "PERSON"]
33
+ assert out[0].start < out[1].start
34
+
35
+
36
+ def test_deterministic_same_input_same_output():
37
+ ents = [_ent("DATE_TIME", 0, 10, 0.85), _ent("PHONE_NUMBER", 0, 10, 0.75)]
38
+ assert resolve_overlaps(list(ents)) == resolve_overlaps(list(reversed(ents)))
@@ -0,0 +1,39 @@
1
+ """Policy engine: correct strategies and deny-by-default for unmapped combos."""
2
+ from medguardx.enums import MaskingStrategy, Purpose, Role
3
+ from medguardx.policy import PolicyEngine
4
+
5
+
6
+ def setup_function():
7
+ global engine
8
+ engine = PolicyEngine()
9
+
10
+
11
+ def test_doctor_treatment_consent_is_full_access():
12
+ strat, _ = engine.evaluate(Role.DOCTOR, Purpose.TREATMENT, True)
13
+ assert strat == MaskingStrategy.FULL_ACCESS
14
+
15
+
16
+ def test_researcher_research_is_anonymized_regardless_of_consent():
17
+ for consent in (True, False):
18
+ strat, _ = engine.evaluate(Role.RESEARCHER, Purpose.RESEARCH, consent)
19
+ assert strat == MaskingStrategy.FULL_ANONYMIZE
20
+
21
+
22
+ def test_unmapped_combination_is_denied():
23
+ strat, rule = engine.evaluate(Role.COMPANY, Purpose.TREATMENT, False)
24
+ assert strat == MaskingStrategy.DENY
25
+ assert "denied" in rule.lower()
26
+
27
+
28
+ def test_accepts_plain_strings():
29
+ strat, _ = engine.evaluate("nurse", "treatment", False)
30
+ assert strat == MaskingStrategy.PARTIAL_MASK
31
+
32
+
33
+ def test_custom_rules_override_defaults():
34
+ custom = {(Role.COMPANY, Purpose.TREATMENT, True): (MaskingStrategy.FULL_ACCESS, "custom")}
35
+ e = PolicyEngine(rules=custom)
36
+ strat, rule = e.evaluate(Role.COMPANY, Purpose.TREATMENT, True)
37
+ assert strat == MaskingStrategy.FULL_ACCESS and rule == "custom"
38
+ # anything else still deny-by-default
39
+ assert e.evaluate(Role.DOCTOR, Purpose.TREATMENT, True)[0] == MaskingStrategy.DENY