medguardx-core 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- medguardx_core-1.0.0/.gitignore +29 -0
- medguardx_core-1.0.0/PKG-INFO +127 -0
- medguardx_core-1.0.0/README.md +91 -0
- medguardx_core-1.0.0/pyproject.toml +54 -0
- medguardx_core-1.0.0/src/medguardx/__init__.py +40 -0
- medguardx_core-1.0.0/src/medguardx/config.py +81 -0
- medguardx_core-1.0.0/src/medguardx/detection.py +132 -0
- medguardx_core-1.0.0/src/medguardx/engine.py +101 -0
- medguardx_core-1.0.0/src/medguardx/enums.py +47 -0
- medguardx_core-1.0.0/src/medguardx/ingestion.py +104 -0
- medguardx_core-1.0.0/src/medguardx/masking.py +107 -0
- medguardx_core-1.0.0/src/medguardx/policy.py +73 -0
- medguardx_core-1.0.0/src/medguardx/recognizers/__init__.py +56 -0
- medguardx_core-1.0.0/src/medguardx/recognizers/aadhaar.py +93 -0
- medguardx_core-1.0.0/tests/test_aadhaar.py +41 -0
- medguardx_core-1.0.0/tests/test_masking.py +54 -0
- medguardx_core-1.0.0/tests/test_overlap.py +38 -0
- medguardx_core-1.0.0/tests/test_policy.py +39 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.pyc
|
|
4
|
+
venv/
|
|
5
|
+
.venv/
|
|
6
|
+
*.egg-info/
|
|
7
|
+
build/
|
|
8
|
+
dist/
|
|
9
|
+
.pytest_cache/
|
|
10
|
+
testenv/
|
|
11
|
+
|
|
12
|
+
# Node / Next
|
|
13
|
+
node_modules/
|
|
14
|
+
.next/
|
|
15
|
+
out/
|
|
16
|
+
|
|
17
|
+
# Data & secrets
|
|
18
|
+
*.db
|
|
19
|
+
*.db-wal
|
|
20
|
+
*.db-shm
|
|
21
|
+
.env
|
|
22
|
+
!.env.example
|
|
23
|
+
|
|
24
|
+
# OS / editor
|
|
25
|
+
.DS_Store
|
|
26
|
+
*.mov
|
|
27
|
+
|
|
28
|
+
# TS build info
|
|
29
|
+
*.tsbuildinfo
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: medguardx-core
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Context-aware PII/PHI detection and masking engine for healthcare data. Stateless, model-configurable, framework-agnostic.
|
|
5
|
+
Project-URL: Homepage, https://github.com/adarshcod30/MedGuardX
|
|
6
|
+
Project-URL: Repository, https://github.com/adarshcod30/MedGuardX
|
|
7
|
+
Project-URL: Issues, https://github.com/adarshcod30/MedGuardX/issues
|
|
8
|
+
Author-email: Adarsh Dwivedi <23ucs509@lnmiit.ac.in>
|
|
9
|
+
License: Apache-2.0
|
|
10
|
+
Keywords: anonymization,dpdp,gdpr,healthcare,hipaa,masking,phi,pii,presidio
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Healthcare Industry
|
|
14
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Classifier: Topic :: Security
|
|
23
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
24
|
+
Requires-Python: >=3.9
|
|
25
|
+
Requires-Dist: presidio-analyzer>=2.2.354
|
|
26
|
+
Requires-Dist: presidio-anonymizer>=2.2.354
|
|
27
|
+
Requires-Dist: spacy<3.9,>=3.7
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: build>=1.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: pytest-cov>=4; extra == 'dev'
|
|
31
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
32
|
+
Requires-Dist: twine>=5.0; extra == 'dev'
|
|
33
|
+
Provides-Extra: trf
|
|
34
|
+
Requires-Dist: spacy-transformers>=1.3; extra == 'trf'
|
|
35
|
+
Description-Content-Type: text/markdown
|
|
36
|
+
|
|
37
|
+
# medguardx-core
|
|
38
|
+
|
|
39
|
+
Context-aware **PII/PHI detection and masking** for healthcare data — a stateless,
|
|
40
|
+
model-configurable, framework-agnostic Python engine. This is the reusable core of
|
|
41
|
+
[MedGuardX](https://github.com/adarshcod30/MedGuardX); embed it in any project.
|
|
42
|
+
|
|
43
|
+
## Why
|
|
44
|
+
|
|
45
|
+
- **Stateless & safe to embed** — text and context in, masked text out. No auth, no
|
|
46
|
+
database, no global state. Your app owns storage, identity, and audit.
|
|
47
|
+
- **Bring your own model** — any installed spaCy English pipeline works
|
|
48
|
+
(`en_core_web_sm` / `md` / `lg` / `trf`). Pick your accuracy/RAM trade-off.
|
|
49
|
+
- **Model-independent structured IDs** — Aadhaar (Verhoeff-validated), PAN, MRN,
|
|
50
|
+
credit cards, IBAN, IP are matched by format, so they work identically on every
|
|
51
|
+
model.
|
|
52
|
+
- **Deterministic, leak-proof masking** — overlapping detections are resolved by a
|
|
53
|
+
fixed priority, and no strategy ever leaves part of a detected identifier visible.
|
|
54
|
+
|
|
55
|
+
## Install
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install medguardx-core
|
|
59
|
+
python -m spacy download en_core_web_md # recommended default
|
|
60
|
+
# other options: en_core_web_sm (smallest) · en_core_web_lg (best statistical)
|
|
61
|
+
# for the transformer model: pip install "medguardx-core[trf]" && python -m spacy download en_core_web_trf
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
The engine is model-agnostic — install any spaCy English pipeline and pass its name
|
|
65
|
+
to `EngineConfig(model=...)`. (Model wheels aren't declared as dependencies because
|
|
66
|
+
spaCy models aren't on PyPI; `spacy download` is the standard way to fetch them.)
|
|
67
|
+
|
|
68
|
+
## Quickstart
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from medguardx import MedGuardEngine, EngineConfig, Role, Purpose
|
|
72
|
+
|
|
73
|
+
engine = MedGuardEngine(EngineConfig(model="en_core_web_md"))
|
|
74
|
+
|
|
75
|
+
result = engine.process(
|
|
76
|
+
"Patient John Smith, Aadhaar 2341 2341 2346, card 4111 1111 1111 1111.",
|
|
77
|
+
role=Role.NURSE, purpose=Purpose.TREATMENT, consent=False,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
print(result.masking_strategy if False else result.strategy.value) # partial_mask
|
|
81
|
+
print(result.masked_text)
|
|
82
|
+
# Patient [NAME_REDACTED], Aadhaar [AADHAAR_REDACTED], card [CARD ****1111].
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Compose the steps yourself when you need to:
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
entities = engine.detect(text) # list[PIIEntity]
|
|
89
|
+
strategy, rule = engine.evaluate_policy(Role.DOCTOR, Purpose.RESEARCH, consent=False)
|
|
90
|
+
masked = engine.mask(text, entities, strategy)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Configuration
|
|
94
|
+
|
|
95
|
+
`EngineConfig(model=..., score_threshold=..., entities=[...], enable_custom_recognizers=True)`.
|
|
96
|
+
|
|
97
|
+
`DATE_TIME` is intentionally **not** in the default entity set — as a high-confidence
|
|
98
|
+
span it used to shadow phone numbers and Aadhaar. Add it back explicitly if you need
|
|
99
|
+
date masking.
|
|
100
|
+
|
|
101
|
+
## Custom policy
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
from medguardx import PolicyEngine, MedGuardEngine, Role, Purpose, MaskingStrategy
|
|
105
|
+
|
|
106
|
+
rules = {(Role.COMPANY, Purpose.TREATMENT, True): (MaskingStrategy.PARTIAL_MASK, "vendor SLA")}
|
|
107
|
+
engine = MedGuardEngine(policy=PolicyEngine(rules=rules)) # deny-by-default for the rest
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## Masking strategies
|
|
111
|
+
|
|
112
|
+
| Strategy | Behaviour |
|
|
113
|
+
|---|---|
|
|
114
|
+
| `full_access` | text returned unchanged |
|
|
115
|
+
| `partial_mask` | identifiers redacted; card/phone/email keep a minimal safe hint |
|
|
116
|
+
| `full_anonymize` | every identifier replaced with a typed `[TYPE_REDACTED]` token |
|
|
117
|
+
| `deny` | access refused; no data returned |
|
|
118
|
+
|
|
119
|
+
## Ingestion (optional)
|
|
120
|
+
|
|
121
|
+
`medguardx.ingestion.extract_text(filename, bytes)` pulls text from plain text, PDF
|
|
122
|
+
(`pdfplumber`), images (`pytesseract`), and HL7 (`hl7apy`). Those extractors import
|
|
123
|
+
their heavy deps lazily.
|
|
124
|
+
|
|
125
|
+
## License
|
|
126
|
+
|
|
127
|
+
Apache-2.0.
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# medguardx-core
|
|
2
|
+
|
|
3
|
+
Context-aware **PII/PHI detection and masking** for healthcare data — a stateless,
|
|
4
|
+
model-configurable, framework-agnostic Python engine. This is the reusable core of
|
|
5
|
+
[MedGuardX](https://github.com/adarshcod30/MedGuardX); embed it in any project.
|
|
6
|
+
|
|
7
|
+
## Why
|
|
8
|
+
|
|
9
|
+
- **Stateless & safe to embed** — text and context in, masked text out. No auth, no
|
|
10
|
+
database, no global state. Your app owns storage, identity, and audit.
|
|
11
|
+
- **Bring your own model** — any installed spaCy English pipeline works
|
|
12
|
+
(`en_core_web_sm` / `md` / `lg` / `trf`). Pick your accuracy/RAM trade-off.
|
|
13
|
+
- **Model-independent structured IDs** — Aadhaar (Verhoeff-validated), PAN, MRN,
|
|
14
|
+
credit cards, IBAN, IP are matched by format, so they work identically on every
|
|
15
|
+
model.
|
|
16
|
+
- **Deterministic, leak-proof masking** — overlapping detections are resolved by a
|
|
17
|
+
fixed priority, and no strategy ever leaves part of a detected identifier visible.
|
|
18
|
+
|
|
19
|
+
## Install
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
pip install medguardx-core
|
|
23
|
+
python -m spacy download en_core_web_md # recommended default
|
|
24
|
+
# other options: en_core_web_sm (smallest) · en_core_web_lg (best statistical)
|
|
25
|
+
# for the transformer model: pip install "medguardx-core[trf]" && python -m spacy download en_core_web_trf
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
The engine is model-agnostic — install any spaCy English pipeline and pass its name
|
|
29
|
+
to `EngineConfig(model=...)`. (Model wheels aren't declared as dependencies because
|
|
30
|
+
spaCy models aren't on PyPI; `spacy download` is the standard way to fetch them.)
|
|
31
|
+
|
|
32
|
+
## Quickstart
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
from medguardx import MedGuardEngine, EngineConfig, Role, Purpose
|
|
36
|
+
|
|
37
|
+
engine = MedGuardEngine(EngineConfig(model="en_core_web_md"))
|
|
38
|
+
|
|
39
|
+
result = engine.process(
|
|
40
|
+
"Patient John Smith, Aadhaar 2341 2341 2346, card 4111 1111 1111 1111.",
|
|
41
|
+
role=Role.NURSE, purpose=Purpose.TREATMENT, consent=False,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
print(result.masking_strategy if False else result.strategy.value) # partial_mask
|
|
45
|
+
print(result.masked_text)
|
|
46
|
+
# Patient [NAME_REDACTED], Aadhaar [AADHAAR_REDACTED], card [CARD ****1111].
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Compose the steps yourself when you need to:
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
entities = engine.detect(text) # list[PIIEntity]
|
|
53
|
+
strategy, rule = engine.evaluate_policy(Role.DOCTOR, Purpose.RESEARCH, consent=False)
|
|
54
|
+
masked = engine.mask(text, entities, strategy)
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Configuration
|
|
58
|
+
|
|
59
|
+
`EngineConfig(model=..., score_threshold=..., entities=[...], enable_custom_recognizers=True)`.
|
|
60
|
+
|
|
61
|
+
`DATE_TIME` is intentionally **not** in the default entity set — as a high-confidence
|
|
62
|
+
span it used to shadow phone numbers and Aadhaar. Add it back explicitly if you need
|
|
63
|
+
date masking.
|
|
64
|
+
|
|
65
|
+
## Custom policy
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
from medguardx import PolicyEngine, MedGuardEngine, Role, Purpose, MaskingStrategy
|
|
69
|
+
|
|
70
|
+
rules = {(Role.COMPANY, Purpose.TREATMENT, True): (MaskingStrategy.PARTIAL_MASK, "vendor SLA")}
|
|
71
|
+
engine = MedGuardEngine(policy=PolicyEngine(rules=rules)) # deny-by-default for the rest
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Masking strategies
|
|
75
|
+
|
|
76
|
+
| Strategy | Behaviour |
|
|
77
|
+
|---|---|
|
|
78
|
+
| `full_access` | text returned unchanged |
|
|
79
|
+
| `partial_mask` | identifiers redacted; card/phone/email keep a minimal safe hint |
|
|
80
|
+
| `full_anonymize` | every identifier replaced with a typed `[TYPE_REDACTED]` token |
|
|
81
|
+
| `deny` | access refused; no data returned |
|
|
82
|
+
|
|
83
|
+
## Ingestion (optional)
|
|
84
|
+
|
|
85
|
+
`medguardx.ingestion.extract_text(filename, bytes)` pulls text from plain text, PDF
|
|
86
|
+
(`pdfplumber`), images (`pytesseract`), and HL7 (`hl7apy`). Those extractors import
|
|
87
|
+
their heavy deps lazily.
|
|
88
|
+
|
|
89
|
+
## License
|
|
90
|
+
|
|
91
|
+
Apache-2.0.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "medguardx-core"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "Context-aware PII/PHI detection and masking engine for healthcare data. Stateless, model-configurable, framework-agnostic."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "Apache-2.0" }
|
|
12
|
+
authors = [{ name = "Adarsh Dwivedi", email = "23ucs509@lnmiit.ac.in" }]
|
|
13
|
+
keywords = ["pii", "phi", "healthcare", "masking", "anonymization", "presidio", "hipaa", "dpdp", "gdpr"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Healthcare Industry",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"License :: OSI Approved :: Apache Software License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.9",
|
|
22
|
+
"Programming Language :: Python :: 3.10",
|
|
23
|
+
"Programming Language :: Python :: 3.11",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Topic :: Security",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
27
|
+
"Topic :: Text Processing :: Linguistic",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
dependencies = [
|
|
31
|
+
"presidio-analyzer>=2.2.354",
|
|
32
|
+
"presidio-anonymizer>=2.2.354",
|
|
33
|
+
"spacy>=3.7,<3.9",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
# The engine is model-agnostic: after install, choose any spaCy English pipeline
|
|
37
|
+
# with `python -m spacy download en_core_web_{sm,md,lg,trf}` and pass its name to
|
|
38
|
+
# EngineConfig(model=...). Model wheels are NOT declared as dependencies because
|
|
39
|
+
# PyPI forbids direct-URL requirements and spaCy models are not on PyPI.
|
|
40
|
+
[project.optional-dependencies]
|
|
41
|
+
# The transformer pipeline (en_core_web_trf) additionally needs spacy-transformers.
|
|
42
|
+
trf = ["spacy-transformers>=1.3"]
|
|
43
|
+
dev = ["pytest>=7", "pytest-cov>=4", "build>=1.0", "twine>=5.0"]
|
|
44
|
+
|
|
45
|
+
[project.urls]
|
|
46
|
+
Homepage = "https://github.com/adarshcod30/MedGuardX"
|
|
47
|
+
Repository = "https://github.com/adarshcod30/MedGuardX"
|
|
48
|
+
Issues = "https://github.com/adarshcod30/MedGuardX/issues"
|
|
49
|
+
|
|
50
|
+
[tool.hatch.build.targets.wheel]
|
|
51
|
+
packages = ["src/medguardx"]
|
|
52
|
+
|
|
53
|
+
[tool.pytest.ini_options]
|
|
54
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""MedGuardX core -- context-aware PII/PHI detection and masking.
|
|
2
|
+
|
|
3
|
+
Public API::
|
|
4
|
+
|
|
5
|
+
from medguardx import (
|
|
6
|
+
MedGuardEngine, EngineConfig, ProcessResult,
|
|
7
|
+
PolicyEngine, PIIEntity,
|
|
8
|
+
Role, Purpose, MaskingStrategy,
|
|
9
|
+
)
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from .config import DEFAULT_ENTITIES, EngineConfig
|
|
14
|
+
from .detection import Detector, PIIEntity, resolve_overlaps
|
|
15
|
+
from .engine import MedGuardEngine, ProcessResult
|
|
16
|
+
from .enums import MaskingStrategy, Purpose, Role
|
|
17
|
+
from .ingestion import ExtractionError, extract_text
|
|
18
|
+
from .masking import mask_text
|
|
19
|
+
from .policy import DEFAULT_RULES, PolicyEngine
|
|
20
|
+
|
|
21
|
+
__version__ = "1.0.0"
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"MedGuardEngine",
|
|
25
|
+
"EngineConfig",
|
|
26
|
+
"ProcessResult",
|
|
27
|
+
"Detector",
|
|
28
|
+
"PIIEntity",
|
|
29
|
+
"resolve_overlaps",
|
|
30
|
+
"PolicyEngine",
|
|
31
|
+
"DEFAULT_RULES",
|
|
32
|
+
"DEFAULT_ENTITIES",
|
|
33
|
+
"ExtractionError",
|
|
34
|
+
"extract_text",
|
|
35
|
+
"mask_text",
|
|
36
|
+
"Role",
|
|
37
|
+
"Purpose",
|
|
38
|
+
"MaskingStrategy",
|
|
39
|
+
"__version__",
|
|
40
|
+
]
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Engine configuration.
|
|
2
|
+
|
|
3
|
+
Everything an integrator might want to tune lives here. The most important knob
|
|
4
|
+
is ``model``: any installed spaCy English pipeline works (en_core_web_sm / md /
|
|
5
|
+
lg / trf). Nothing else in the engine changes when you swap it -- structured
|
|
6
|
+
identifiers are matched by regex recognizers that are model-independent.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import List
|
|
12
|
+
|
|
13
|
+
# Entities detected by default.
|
|
14
|
+
#
|
|
15
|
+
# NOTE: DATE_TIME is deliberately EXCLUDED from the default set. In the old build
|
|
16
|
+
# spaCy tagged nearly every digit run (phone numbers, Aadhaar, card numbers) as a
|
|
17
|
+
# high-confidence DATE_TIME, which hijacked the span and defeated the specialized
|
|
18
|
+
# recognizers -- the root cause of "some data gets masked, some not". Integrators
|
|
19
|
+
# who genuinely need date masking can add it back explicitly.
|
|
20
|
+
DEFAULT_ENTITIES: List[str] = [
|
|
21
|
+
"PERSON",
|
|
22
|
+
"PHONE_NUMBER",
|
|
23
|
+
"EMAIL_ADDRESS",
|
|
24
|
+
"CREDIT_CARD",
|
|
25
|
+
"IBAN_CODE",
|
|
26
|
+
"IP_ADDRESS",
|
|
27
|
+
"LOCATION",
|
|
28
|
+
"NRP",
|
|
29
|
+
"MEDICAL_LICENSE",
|
|
30
|
+
"URL",
|
|
31
|
+
"US_SSN",
|
|
32
|
+
# Custom, model-independent recognizers:
|
|
33
|
+
"IN_AADHAAR",
|
|
34
|
+
"IN_PAN",
|
|
35
|
+
"MEDICAL_RECORD_NUMBER",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
# Overlap-resolution priority. When two detections overlap, the one with the
|
|
39
|
+
# higher priority wins regardless of raw score. More specific / structured
|
|
40
|
+
# identifiers rank above generic linguistic ones so a phone number never loses to
|
|
41
|
+
# a stray PERSON or DATE_TIME span. Higher number = higher priority.
|
|
42
|
+
ENTITY_PRIORITY = {
|
|
43
|
+
"IN_AADHAAR": 100,
|
|
44
|
+
"IN_PAN": 100,
|
|
45
|
+
"US_SSN": 100,
|
|
46
|
+
"CREDIT_CARD": 95,
|
|
47
|
+
"IBAN_CODE": 95,
|
|
48
|
+
"MEDICAL_RECORD_NUMBER": 90,
|
|
49
|
+
"MEDICAL_LICENSE": 85,
|
|
50
|
+
"EMAIL_ADDRESS": 80,
|
|
51
|
+
"PHONE_NUMBER": 75,
|
|
52
|
+
"IP_ADDRESS": 70,
|
|
53
|
+
"URL": 40,
|
|
54
|
+
"PERSON": 60,
|
|
55
|
+
"LOCATION": 55,
|
|
56
|
+
"NRP": 50,
|
|
57
|
+
"DATE_TIME": 20,
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class EngineConfig:
|
|
63
|
+
"""Configuration for a :class:`~medguardx.engine.MedGuardEngine`.
|
|
64
|
+
|
|
65
|
+
Attributes:
|
|
66
|
+
model: Name of the installed spaCy pipeline to use. Pick per your
|
|
67
|
+
accuracy/RAM trade-off: ``en_core_web_sm`` (~50MB, weakest),
|
|
68
|
+
``en_core_web_md`` (~120MB, recommended default), ``en_core_web_lg``
|
|
69
|
+
(~600MB, best statistical), ``en_core_web_trf`` (transformer, highest
|
|
70
|
+
accuracy, heaviest).
|
|
71
|
+
score_threshold: Minimum confidence for a detection to be kept.
|
|
72
|
+
entities: Which entity types to detect.
|
|
73
|
+
enable_custom_recognizers: Register the Aadhaar/PAN/MRN recognizers.
|
|
74
|
+
default_language: Language code passed to Presidio.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
model: str = "en_core_web_md"
|
|
78
|
+
score_threshold: float = 0.35
|
|
79
|
+
entities: List[str] = field(default_factory=lambda: list(DEFAULT_ENTITIES))
|
|
80
|
+
enable_custom_recognizers: bool = True
|
|
81
|
+
default_language: str = "en"
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""PII/PHI detection built on Microsoft Presidio.
|
|
2
|
+
|
|
3
|
+
Two things here fix the old build's inconsistent masking:
|
|
4
|
+
|
|
5
|
+
1. The spaCy model is configurable (``EngineConfig.model``) instead of hardcoded.
|
|
6
|
+
2. Overlapping detections are resolved by a deterministic priority pass, so a
|
|
7
|
+
specialized entity (phone, Aadhaar, card) can never be shadowed by a generic
|
|
8
|
+
high-score span. In the old build the winner was whatever Presidio happened to
|
|
9
|
+
score highest, which varied with phrasing -- the source of the flakiness.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import asdict, dataclass
|
|
14
|
+
from typing import Dict, List, Optional
|
|
15
|
+
|
|
16
|
+
from .config import ENTITY_PRIORITY, EngineConfig
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class PIIEntity:
|
|
21
|
+
"""A single detected entity with its exact span in the source text."""
|
|
22
|
+
|
|
23
|
+
entity_type: str
|
|
24
|
+
start: int
|
|
25
|
+
end: int
|
|
26
|
+
score: float
|
|
27
|
+
text: str
|
|
28
|
+
|
|
29
|
+
def to_dict(self) -> Dict:
|
|
30
|
+
return asdict(self)
|
|
31
|
+
|
|
32
|
+
@classmethod
|
|
33
|
+
def from_dict(cls, d: Dict) -> "PIIEntity":
|
|
34
|
+
return cls(
|
|
35
|
+
entity_type=d["entity_type"],
|
|
36
|
+
start=int(d["start"]),
|
|
37
|
+
end=int(d["end"]),
|
|
38
|
+
score=float(d["score"]),
|
|
39
|
+
text=d.get("text", ""),
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class Detector:
|
|
44
|
+
"""Lazily-initialised Presidio analyzer wrapper.
|
|
45
|
+
|
|
46
|
+
The heavy analyzer/model load happens on first use, not at import time, so
|
|
47
|
+
importing :mod:`medguardx` stays cheap.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(self, config: Optional[EngineConfig] = None) -> None:
|
|
51
|
+
self.config = config or EngineConfig()
|
|
52
|
+
self._analyzer = None
|
|
53
|
+
|
|
54
|
+
def _get_analyzer(self):
|
|
55
|
+
if self._analyzer is not None:
|
|
56
|
+
return self._analyzer
|
|
57
|
+
|
|
58
|
+
from presidio_analyzer import AnalyzerEngine, RecognizerRegistry
|
|
59
|
+
from presidio_analyzer.nlp_engine import NlpEngineProvider
|
|
60
|
+
|
|
61
|
+
provider = NlpEngineProvider(
|
|
62
|
+
nlp_configuration={
|
|
63
|
+
"nlp_engine_name": "spacy",
|
|
64
|
+
"models": [{"lang_code": self.config.default_language, "model_name": self.config.model}],
|
|
65
|
+
}
|
|
66
|
+
)
|
|
67
|
+
nlp_engine = provider.create_engine()
|
|
68
|
+
|
|
69
|
+
registry = RecognizerRegistry()
|
|
70
|
+
registry.load_predefined_recognizers(languages=[self.config.default_language])
|
|
71
|
+
if self.config.enable_custom_recognizers:
|
|
72
|
+
from .recognizers import build_custom_recognizers
|
|
73
|
+
|
|
74
|
+
for recognizer in build_custom_recognizers():
|
|
75
|
+
registry.add_recognizer(recognizer)
|
|
76
|
+
|
|
77
|
+
self._analyzer = AnalyzerEngine(
|
|
78
|
+
nlp_engine=nlp_engine,
|
|
79
|
+
registry=registry,
|
|
80
|
+
supported_languages=[self.config.default_language],
|
|
81
|
+
)
|
|
82
|
+
return self._analyzer
|
|
83
|
+
|
|
84
|
+
def detect(self, text: str, entities: Optional[List[str]] = None) -> List[PIIEntity]:
|
|
85
|
+
"""Detect PII entities in ``text`` and return a non-overlapping, sorted list."""
|
|
86
|
+
if not text or not text.strip():
|
|
87
|
+
return []
|
|
88
|
+
|
|
89
|
+
analyzer = self._get_analyzer()
|
|
90
|
+
raw = analyzer.analyze(
|
|
91
|
+
text=text,
|
|
92
|
+
language=self.config.default_language,
|
|
93
|
+
entities=entities or self.config.entities,
|
|
94
|
+
score_threshold=self.config.score_threshold,
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
found = [
|
|
98
|
+
PIIEntity(
|
|
99
|
+
entity_type=r.entity_type,
|
|
100
|
+
start=r.start,
|
|
101
|
+
end=r.end,
|
|
102
|
+
score=round(float(r.score), 2),
|
|
103
|
+
text=text[r.start : r.end],
|
|
104
|
+
)
|
|
105
|
+
for r in raw
|
|
106
|
+
]
|
|
107
|
+
return resolve_overlaps(found)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def resolve_overlaps(entities: List[PIIEntity]) -> List[PIIEntity]:
|
|
111
|
+
"""Collapse overlapping spans, keeping the highest-priority entity per region.
|
|
112
|
+
|
|
113
|
+
Priority is decided first by :data:`ENTITY_PRIORITY` (so a PHONE_NUMBER beats
|
|
114
|
+
a DATE_TIME even at lower raw score), then by score, then by span length. The
|
|
115
|
+
result is deterministic: the same text always yields the same masking.
|
|
116
|
+
"""
|
|
117
|
+
if not entities:
|
|
118
|
+
return []
|
|
119
|
+
|
|
120
|
+
def rank(e: PIIEntity):
|
|
121
|
+
return (ENTITY_PRIORITY.get(e.entity_type, 30), e.score, e.end - e.start)
|
|
122
|
+
|
|
123
|
+
# Strongest candidates first; greedily accept ones that don't overlap an
|
|
124
|
+
# already-accepted (higher-ranked) span.
|
|
125
|
+
ordered = sorted(entities, key=rank, reverse=True)
|
|
126
|
+
accepted: List[PIIEntity] = []
|
|
127
|
+
for cand in ordered:
|
|
128
|
+
if not any(cand.start < a.end and a.start < cand.end for a in accepted):
|
|
129
|
+
accepted.append(cand)
|
|
130
|
+
|
|
131
|
+
accepted.sort(key=lambda e: e.start)
|
|
132
|
+
return accepted
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""The public MedGuardX engine -- a stateless facade over detection, policy and masking.
|
|
2
|
+
|
|
3
|
+
Typical use::
|
|
4
|
+
|
|
5
|
+
from medguardx import MedGuardEngine, EngineConfig, Role, Purpose
|
|
6
|
+
|
|
7
|
+
engine = MedGuardEngine(EngineConfig(model="en_core_web_md"))
|
|
8
|
+
result = engine.process(
|
|
9
|
+
"Patient John Smith, Aadhaar 2341 2341 2341, card 4111 1111 1111 1111.",
|
|
10
|
+
role=Role.NURSE, purpose=Purpose.TREATMENT, consent=False,
|
|
11
|
+
)
|
|
12
|
+
print(result.masked_text)
|
|
13
|
+
|
|
14
|
+
The engine holds no data and does no I/O -- integrators own storage, auth and
|
|
15
|
+
audit. That statelessness is deliberate: it is what makes the engine safe to
|
|
16
|
+
embed anywhere.
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from typing import List, Optional
|
|
22
|
+
|
|
23
|
+
from .config import EngineConfig
|
|
24
|
+
from .detection import Detector, PIIEntity
|
|
25
|
+
from .enums import MaskingStrategy, Purpose, Role, coerce_purpose, coerce_role
|
|
26
|
+
from .masking import mask_text
|
|
27
|
+
from .policy import PolicyEngine
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class ProcessResult:
|
|
32
|
+
"""Outcome of :meth:`MedGuardEngine.process`."""
|
|
33
|
+
|
|
34
|
+
original_text: str
|
|
35
|
+
masked_text: str
|
|
36
|
+
entities: List[PIIEntity]
|
|
37
|
+
strategy: MaskingStrategy
|
|
38
|
+
policy_rule: str
|
|
39
|
+
denied: bool = field(init=False)
|
|
40
|
+
|
|
41
|
+
def __post_init__(self) -> None:
|
|
42
|
+
self.denied = self.strategy == MaskingStrategy.DENY
|
|
43
|
+
|
|
44
|
+
def to_dict(self) -> dict:
|
|
45
|
+
return {
|
|
46
|
+
"original_text": self.original_text,
|
|
47
|
+
"masked_text": self.masked_text,
|
|
48
|
+
"entities": [e.to_dict() for e in self.entities],
|
|
49
|
+
"masking_strategy": self.strategy.value,
|
|
50
|
+
"policy_rule": self.policy_rule,
|
|
51
|
+
"entities_masked": 0 if self.strategy == MaskingStrategy.FULL_ACCESS else len(self.entities),
|
|
52
|
+
"denied": self.denied,
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class MedGuardEngine:
|
|
57
|
+
"""Detect -> evaluate policy -> mask, in one call or as separate steps."""
|
|
58
|
+
|
|
59
|
+
def __init__(
|
|
60
|
+
self,
|
|
61
|
+
config: Optional[EngineConfig] = None,
|
|
62
|
+
policy: Optional[PolicyEngine] = None,
|
|
63
|
+
) -> None:
|
|
64
|
+
self.config = config or EngineConfig()
|
|
65
|
+
self.detector = Detector(self.config)
|
|
66
|
+
self.policy = policy or PolicyEngine()
|
|
67
|
+
|
|
68
|
+
# --- composable steps -------------------------------------------------------
|
|
69
|
+
def detect(self, text: str) -> List[PIIEntity]:
|
|
70
|
+
return self.detector.detect(text)
|
|
71
|
+
|
|
72
|
+
def evaluate_policy(self, role, purpose, consent: bool):
|
|
73
|
+
return self.policy.evaluate(role, purpose, consent)
|
|
74
|
+
|
|
75
|
+
def mask(self, text: str, entities: List[PIIEntity], strategy: MaskingStrategy) -> str:
|
|
76
|
+
return mask_text(text, entities, strategy)
|
|
77
|
+
|
|
78
|
+
# --- one-shot ---------------------------------------------------------------
|
|
79
|
+
def process(
|
|
80
|
+
self,
|
|
81
|
+
text: str,
|
|
82
|
+
role,
|
|
83
|
+
purpose,
|
|
84
|
+
consent: bool = False,
|
|
85
|
+
entities: Optional[List[PIIEntity]] = None,
|
|
86
|
+
) -> ProcessResult:
|
|
87
|
+
"""Run the full pipeline. Pass pre-computed ``entities`` to skip detection."""
|
|
88
|
+
role = coerce_role(role)
|
|
89
|
+
purpose = coerce_purpose(purpose)
|
|
90
|
+
strategy, rule = self.policy.evaluate(role, purpose, consent)
|
|
91
|
+
|
|
92
|
+
if strategy == MaskingStrategy.DENY:
|
|
93
|
+
return ProcessResult(text, "", [], strategy, rule)
|
|
94
|
+
|
|
95
|
+
detected = entities if entities is not None else self.detect(text)
|
|
96
|
+
masked = self.mask(text, detected, strategy)
|
|
97
|
+
return ProcessResult(text, masked, detected, strategy, rule)
|
|
98
|
+
|
|
99
|
+
def warm_up(self) -> None:
|
|
100
|
+
"""Eagerly load the model (useful at service startup)."""
|
|
101
|
+
self.detector._get_analyzer()
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Core enumerations shared across the MedGuardX engine.
|
|
2
|
+
|
|
3
|
+
These are intentionally plain ``str`` enums so they serialize cleanly to JSON and
|
|
4
|
+
compare equal to their string values -- an integrator can pass ``"nurse"`` or
|
|
5
|
+
``Role.NURSE`` interchangeably.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from enum import Enum
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Role(str, Enum):
|
|
13
|
+
"""Who is requesting the data."""
|
|
14
|
+
|
|
15
|
+
DOCTOR = "doctor"
|
|
16
|
+
NURSE = "nurse"
|
|
17
|
+
RESEARCHER = "researcher"
|
|
18
|
+
PATIENT = "patient"
|
|
19
|
+
COMPANY = "company"
|
|
20
|
+
ADMIN = "admin"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Purpose(str, Enum):
|
|
24
|
+
"""Why the data is being requested."""
|
|
25
|
+
|
|
26
|
+
TREATMENT = "treatment"
|
|
27
|
+
RESEARCH = "research"
|
|
28
|
+
BILLING = "billing"
|
|
29
|
+
LEGAL = "legal"
|
|
30
|
+
PERSONAL = "personal"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class MaskingStrategy(str, Enum):
|
|
34
|
+
"""How much of the data the requester is allowed to see."""
|
|
35
|
+
|
|
36
|
+
FULL_ACCESS = "full_access"
|
|
37
|
+
PARTIAL_MASK = "partial_mask"
|
|
38
|
+
FULL_ANONYMIZE = "full_anonymize"
|
|
39
|
+
DENY = "deny"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def coerce_role(value) -> "Role":
|
|
43
|
+
return value if isinstance(value, Role) else Role(str(value).lower())
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def coerce_purpose(value) -> "Purpose":
|
|
47
|
+
return value if isinstance(value, Purpose) else Purpose(str(value).lower())
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Multi-format text extraction: plain text, PDF, image (OCR), HL7.
|
|
2
|
+
|
|
3
|
+
Optional heavy dependencies (pdfplumber, pytesseract/Pillow, hl7apy) are imported
|
|
4
|
+
lazily and degrade gracefully, so installing the core engine does not force an
|
|
5
|
+
OCR/PDF toolchain on integrators who only mask plain text.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import io
|
|
10
|
+
import os
|
|
11
|
+
from typing import Tuple
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ExtractionError(RuntimeError):
|
|
15
|
+
"""Raised when text cannot be extracted from an uploaded file.
|
|
16
|
+
|
|
17
|
+
Distinct from "no text found": this signals a real failure (corrupt file, a
|
|
18
|
+
missing OCR engine, etc.) so callers can reject the upload rather than
|
|
19
|
+
silently storing an error string as the record's content.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def detect_file_type(filename: str, content: bytes) -> str:
|
|
24
|
+
ext = os.path.splitext(filename or "")[1].lower()
|
|
25
|
+
if ext in (".hl7", ".adt"):
|
|
26
|
+
return "hl7"
|
|
27
|
+
if ext == ".pdf":
|
|
28
|
+
return "pdf"
|
|
29
|
+
if ext in (".png", ".jpg", ".jpeg", ".tiff", ".bmp"):
|
|
30
|
+
return "image"
|
|
31
|
+
preview = content[:200].decode("utf-8", errors="ignore")
|
|
32
|
+
if "MSH|" in preview:
|
|
33
|
+
return "hl7"
|
|
34
|
+
return "text"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def extract_text_from_pdf(content: bytes) -> str:
|
|
38
|
+
try:
|
|
39
|
+
import pdfplumber
|
|
40
|
+
|
|
41
|
+
with pdfplumber.open(io.BytesIO(content)) as pdf:
|
|
42
|
+
pages = [page.extract_text() or "" for page in pdf.pages]
|
|
43
|
+
return "\n".join(pages).strip()
|
|
44
|
+
except Exception as exc: # pragma: no cover - depends on optional dep
|
|
45
|
+
raise ExtractionError(f"PDF extraction failed: {exc}") from exc
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def extract_text_from_image(content: bytes) -> str:
|
|
49
|
+
try:
|
|
50
|
+
import tempfile
|
|
51
|
+
|
|
52
|
+
import pytesseract
|
|
53
|
+
from PIL import Image
|
|
54
|
+
except Exception as exc: # pragma: no cover - optional deps not installed
|
|
55
|
+
raise ExtractionError("OCR dependencies (pytesseract/Pillow) are not installed.") from exc
|
|
56
|
+
|
|
57
|
+
try:
|
|
58
|
+
img = Image.open(io.BytesIO(content))
|
|
59
|
+
with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as fh:
|
|
60
|
+
img.save(fh.name)
|
|
61
|
+
try:
|
|
62
|
+
return pytesseract.image_to_string(fh.name).strip()
|
|
63
|
+
finally:
|
|
64
|
+
os.unlink(fh.name)
|
|
65
|
+
except pytesseract.TesseractNotFoundError as exc: # pragma: no cover
|
|
66
|
+
raise ExtractionError(
|
|
67
|
+
"OCR engine (tesseract) is not available in this deployment. "
|
|
68
|
+
"Image files are not supported here; use text, PDF, or HL7."
|
|
69
|
+
) from exc
|
|
70
|
+
except Exception as exc: # pragma: no cover - depends on optional dep
|
|
71
|
+
raise ExtractionError(f"OCR failed: {exc}") from exc
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def parse_hl7(content: bytes) -> str:
|
|
75
|
+
text = content.decode("utf-8", errors="ignore")
|
|
76
|
+
try:
|
|
77
|
+
from hl7apy.parser import parse_message
|
|
78
|
+
|
|
79
|
+
msg = parse_message(text.replace("\n", "\r"))
|
|
80
|
+
segments = []
|
|
81
|
+
for seg in msg.children:
|
|
82
|
+
fields = []
|
|
83
|
+
for field in seg.children:
|
|
84
|
+
try:
|
|
85
|
+
fields.append(f"{field.name}: {field.value}")
|
|
86
|
+
except Exception:
|
|
87
|
+
pass
|
|
88
|
+
if fields:
|
|
89
|
+
segments.append(f"[{seg.name}] " + " | ".join(fields))
|
|
90
|
+
return "\n".join(segments) if segments else text
|
|
91
|
+
except Exception: # pragma: no cover - depends on optional dep
|
|
92
|
+
return text
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def extract_text(filename: str, content: bytes) -> Tuple[str, str]:
|
|
96
|
+
"""Return ``(extracted_text, file_type)`` for an uploaded file."""
|
|
97
|
+
file_type = detect_file_type(filename, content)
|
|
98
|
+
if file_type == "pdf":
|
|
99
|
+
return extract_text_from_pdf(content), file_type
|
|
100
|
+
if file_type == "image":
|
|
101
|
+
return extract_text_from_image(content), file_type
|
|
102
|
+
if file_type == "hl7":
|
|
103
|
+
return parse_hl7(content), file_type
|
|
104
|
+
return content.decode("utf-8", errors="ignore"), file_type
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Deterministic, leak-proof masking.
|
|
2
|
+
|
|
3
|
+
The old build delegated to Presidio's anonymizer with a DEFAULT operator that
|
|
4
|
+
masked only the first 8 characters -- so a 19-digit card came back as
|
|
5
|
+
``********1 1111 1111`` and IPs/IBANs leaked their tails. Here we do span
|
|
6
|
+
replacement ourselves over the already-de-overlapped entity list, replacing from
|
|
7
|
+
right to left so indices stay valid. The invariant is simple and testable: for
|
|
8
|
+
any strategy other than FULL_ACCESS, the original entity substring never survives
|
|
9
|
+
in the output.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from typing import Callable, Dict, List
|
|
14
|
+
|
|
15
|
+
from .detection import PIIEntity
|
|
16
|
+
from .enums import MaskingStrategy
|
|
17
|
+
|
|
18
|
+
# Human-readable label per entity type, used in redaction tokens.
|
|
19
|
+
_LABELS: Dict[str, str] = {
|
|
20
|
+
"PERSON": "NAME",
|
|
21
|
+
"PHONE_NUMBER": "PHONE",
|
|
22
|
+
"EMAIL_ADDRESS": "EMAIL",
|
|
23
|
+
"CREDIT_CARD": "CARD",
|
|
24
|
+
"IBAN_CODE": "IBAN",
|
|
25
|
+
"IP_ADDRESS": "IP",
|
|
26
|
+
"LOCATION": "LOCATION",
|
|
27
|
+
"URL": "URL",
|
|
28
|
+
"NRP": "NRP",
|
|
29
|
+
"US_SSN": "SSN",
|
|
30
|
+
"MEDICAL_LICENSE": "MED_LICENSE",
|
|
31
|
+
"MEDICAL_RECORD_NUMBER": "MRN",
|
|
32
|
+
"IN_AADHAAR": "AADHAAR",
|
|
33
|
+
"IN_PAN": "PAN",
|
|
34
|
+
"DATE_TIME": "DATE",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
DENY_MESSAGE = "[ACCESS DENIED - Insufficient permissions]"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _label(entity_type: str) -> str:
|
|
41
|
+
return _LABELS.get(entity_type, entity_type)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _redact(entity: PIIEntity) -> str:
|
|
45
|
+
return f"[{_label(entity.entity_type)}_REDACTED]"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
# --- Partial-mask reveal helpers -------------------------------------------------
|
|
49
|
+
# Partial masking trades a *small, bounded* amount of the original for downstream
|
|
50
|
+
# utility (e.g. matching a card by its last 4). Every helper below reveals at most
|
|
51
|
+
# a safe suffix and masks everything else; none can leave the majority visible.
|
|
52
|
+
|
|
53
|
+
def _reveal_last(value: str, keep: int, ch: str = "*") -> str:
|
|
54
|
+
digits = [c for c in value if c.isalnum()]
|
|
55
|
+
if len(digits) <= keep:
|
|
56
|
+
return ch * len(digits)
|
|
57
|
+
return ch * (len(digits) - keep) + "".join(digits[-keep:])
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _partial_card(entity: PIIEntity) -> str:
|
|
61
|
+
return f"[CARD ****{_reveal_last(entity.text, 4)[-4:]}]"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _partial_phone(entity: PIIEntity) -> str:
|
|
65
|
+
return f"[PHONE ****{_reveal_last(entity.text, 2)[-2:]}]"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _partial_email(entity: PIIEntity) -> str:
|
|
69
|
+
local, _, domain = entity.text.partition("@")
|
|
70
|
+
if not domain:
|
|
71
|
+
return "[EMAIL_MASKED]"
|
|
72
|
+
hint = (local[:1] + "***") if local else "***"
|
|
73
|
+
return f"{hint}@{domain}"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# Types that get a partial reveal under PARTIAL_MASK. Everything else is fully
|
|
77
|
+
# redacted even under partial masking -- safe by default.
|
|
78
|
+
_PARTIAL_OPERATORS: Dict[str, Callable[[PIIEntity], str]] = {
|
|
79
|
+
"CREDIT_CARD": _partial_card,
|
|
80
|
+
"PHONE_NUMBER": _partial_phone,
|
|
81
|
+
"EMAIL_ADDRESS": _partial_email,
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _mask_token(entity: PIIEntity, strategy: MaskingStrategy) -> str:
|
|
86
|
+
if strategy == MaskingStrategy.PARTIAL_MASK and entity.entity_type in _PARTIAL_OPERATORS:
|
|
87
|
+
return _PARTIAL_OPERATORS[entity.entity_type](entity)
|
|
88
|
+
return _redact(entity)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def mask_text(text: str, entities: List[PIIEntity], strategy: MaskingStrategy) -> str:
|
|
92
|
+
"""Apply ``strategy`` to ``text`` given its detected ``entities``."""
|
|
93
|
+
if strategy == MaskingStrategy.FULL_ACCESS:
|
|
94
|
+
return text
|
|
95
|
+
if strategy == MaskingStrategy.DENY:
|
|
96
|
+
return DENY_MESSAGE
|
|
97
|
+
if not entities:
|
|
98
|
+
return text
|
|
99
|
+
|
|
100
|
+
# Replace right-to-left so earlier spans keep their original indices.
|
|
101
|
+
ordered = sorted(entities, key=lambda e: e.start, reverse=True)
|
|
102
|
+
out = text
|
|
103
|
+
for e in ordered:
|
|
104
|
+
if e.start < 0 or e.end > len(out) or e.start >= e.end:
|
|
105
|
+
continue # stale/invalid span -- skip rather than corrupt the text
|
|
106
|
+
out = out[: e.start] + _mask_token(e, strategy) + out[e.end :]
|
|
107
|
+
return out
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Context-aware policy engine.
|
|
2
|
+
|
|
3
|
+
Maps ``(role, purpose, consent)`` to a masking strategy plus a human-readable
|
|
4
|
+
rationale. The default matrix is the healthcare policy MedGuardX ships with;
|
|
5
|
+
integrators can subclass :class:`PolicyEngine` or pass a custom ``rules`` dict to
|
|
6
|
+
encode their own governance without touching detection or masking.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Dict, Optional, Tuple
|
|
11
|
+
|
|
12
|
+
from .enums import MaskingStrategy, Purpose, Role, coerce_purpose, coerce_role
|
|
13
|
+
|
|
14
|
+
PolicyKey = Tuple[Role, Purpose, bool]
|
|
15
|
+
PolicyValue = Tuple[MaskingStrategy, str]
|
|
16
|
+
|
|
17
|
+
_FA = MaskingStrategy.FULL_ACCESS
|
|
18
|
+
_PM = MaskingStrategy.PARTIAL_MASK
|
|
19
|
+
_AN = MaskingStrategy.FULL_ANONYMIZE
|
|
20
|
+
_DN = MaskingStrategy.DENY
|
|
21
|
+
|
|
22
|
+
DEFAULT_RULES: Dict[PolicyKey, PolicyValue] = {
|
|
23
|
+
# Doctors
|
|
24
|
+
(Role.DOCTOR, Purpose.TREATMENT, True): (_FA, "Doctor requesting treatment records with patient consent: full access granted."),
|
|
25
|
+
(Role.DOCTOR, Purpose.TREATMENT, False): (_PM, "Doctor requesting treatment records without consent: partial access (identifiers masked)."),
|
|
26
|
+
(Role.DOCTOR, Purpose.RESEARCH, True): (_PM, "Doctor requesting research data with consent: partial access (identifiers masked)."),
|
|
27
|
+
(Role.DOCTOR, Purpose.RESEARCH, False): (_AN, "Doctor requesting research data without consent: anonymized access only."),
|
|
28
|
+
# Nurses
|
|
29
|
+
(Role.NURSE, Purpose.TREATMENT, True): (_PM, "Nurse requesting treatment records with consent: partial access (identifiers masked)."),
|
|
30
|
+
(Role.NURSE, Purpose.TREATMENT, False): (_PM, "Nurse requesting treatment records without consent: partial access (identifiers masked)."),
|
|
31
|
+
(Role.NURSE, Purpose.RESEARCH, True): (_AN, "Nurse requesting research data with consent: anonymized access only."),
|
|
32
|
+
# Researchers
|
|
33
|
+
(Role.RESEARCHER, Purpose.RESEARCH, True): (_AN, "Researcher requesting research data with consent: anonymized access only."),
|
|
34
|
+
(Role.RESEARCHER, Purpose.RESEARCH, False): (_AN, "Researcher requesting research data without consent: anonymized access only."),
|
|
35
|
+
# Patients (own records)
|
|
36
|
+
(Role.PATIENT, Purpose.PERSONAL, True): (_FA, "Patient accessing personal records: full access."),
|
|
37
|
+
(Role.PATIENT, Purpose.PERSONAL, False): (_FA, "Patient accessing personal records: full access."),
|
|
38
|
+
(Role.PATIENT, Purpose.TREATMENT, True): (_FA, "Patient accessing treatment records: full access."),
|
|
39
|
+
(Role.PATIENT, Purpose.TREATMENT, False): (_FA, "Patient accessing treatment records: full access."),
|
|
40
|
+
(Role.PATIENT, Purpose.BILLING, True): (_FA, "Patient accessing billing records: full access."),
|
|
41
|
+
(Role.PATIENT, Purpose.BILLING, False): (_FA, "Patient accessing billing records: full access."),
|
|
42
|
+
(Role.PATIENT, Purpose.LEGAL, True): (_FA, "Patient retrieving records for legal reasons: full access."),
|
|
43
|
+
(Role.PATIENT, Purpose.LEGAL, False): (_FA, "Patient retrieving records for legal reasons: full access."),
|
|
44
|
+
# Companies
|
|
45
|
+
(Role.COMPANY, Purpose.RESEARCH, True): (_AN, "Company requesting research data with consent: anonymized access only."),
|
|
46
|
+
(Role.COMPANY, Purpose.BILLING, True): (_PM, "Company requesting billing records with consent: partial access (identifiers masked)."),
|
|
47
|
+
# Admins (governance / break-glass, still logged upstream)
|
|
48
|
+
(Role.ADMIN, Purpose.LEGAL, True): (_FA, "Admin performing authorized legal retrieval: full access."),
|
|
49
|
+
(Role.ADMIN, Purpose.LEGAL, False): (_FA, "Admin performing authorized legal retrieval: full access."),
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class PolicyEngine:
|
|
54
|
+
"""Evaluates access policy. Deny-by-default for any unmapped combination."""
|
|
55
|
+
|
|
56
|
+
def __init__(self, rules: Optional[Dict[PolicyKey, PolicyValue]] = None) -> None:
|
|
57
|
+
self.rules = rules if rules is not None else dict(DEFAULT_RULES)
|
|
58
|
+
|
|
59
|
+
def evaluate(self, role, purpose, consent: bool) -> PolicyValue:
|
|
60
|
+
role = coerce_role(role)
|
|
61
|
+
purpose = coerce_purpose(purpose)
|
|
62
|
+
consent = bool(consent)
|
|
63
|
+
|
|
64
|
+
hit = self.rules.get((role, purpose, consent))
|
|
65
|
+
if hit is not None:
|
|
66
|
+
return hit
|
|
67
|
+
|
|
68
|
+
consent_str = "even with patient consent" if consent else "without patient consent"
|
|
69
|
+
return (
|
|
70
|
+
_DN,
|
|
71
|
+
f"Access denied: {role.value.capitalize()}s are not authorized to access "
|
|
72
|
+
f"records for {purpose.value} purposes, {consent_str}.",
|
|
73
|
+
)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Model-independent pattern recognizers.
|
|
2
|
+
|
|
3
|
+
These run the same way regardless of which spaCy model is loaded (sm / md / lg /
|
|
4
|
+
trf), because structured identifiers -- Aadhaar, PAN, MRN, credit cards -- are
|
|
5
|
+
matched by format, not by linguistic NER. This is what makes the "Indian PII"
|
|
6
|
+
support actually work: it never depended on the model in the first place.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import List
|
|
11
|
+
|
|
12
|
+
from presidio_analyzer import Pattern, PatternRecognizer
|
|
13
|
+
|
|
14
|
+
from .aadhaar import AadhaarRecognizer
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def build_custom_recognizers() -> List[PatternRecognizer]:
|
|
18
|
+
"""Return the custom recognizers MedGuardX registers by default."""
|
|
19
|
+
return [
|
|
20
|
+
AadhaarRecognizer(),
|
|
21
|
+
_pan_recognizer(),
|
|
22
|
+
_mrn_recognizer(),
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _pan_recognizer() -> PatternRecognizer:
|
|
27
|
+
"""Indian Permanent Account Number: 5 letters, 4 digits, 1 letter (e.g. ABCDE1234F)."""
|
|
28
|
+
return PatternRecognizer(
|
|
29
|
+
supported_entity="IN_PAN",
|
|
30
|
+
name="in_pan_recognizer",
|
|
31
|
+
patterns=[
|
|
32
|
+
Pattern(name="pan", regex=r"\b[A-Z]{5}[0-9]{4}[A-Z]\b", score=0.85),
|
|
33
|
+
],
|
|
34
|
+
context=["pan", "permanent account", "income tax"],
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _mrn_recognizer() -> PatternRecognizer:
|
|
39
|
+
"""Medical Record Number.
|
|
40
|
+
|
|
41
|
+
MRNs have no universal format, so we anchor on the explicit ``MRN``/``medical
|
|
42
|
+
record`` context label followed by an alphanumeric code. Anchoring on the
|
|
43
|
+
label keeps this precise instead of masking every stray number.
|
|
44
|
+
"""
|
|
45
|
+
return PatternRecognizer(
|
|
46
|
+
supported_entity="MEDICAL_RECORD_NUMBER",
|
|
47
|
+
name="mrn_recognizer",
|
|
48
|
+
patterns=[
|
|
49
|
+
Pattern(
|
|
50
|
+
name="mrn_labelled",
|
|
51
|
+
regex=r"\b(?:MRN|Medical\s*Record\s*(?:No\.?|Number|#)?)[:\s#-]*([A-Z0-9][A-Z0-9-]{2,14})\b",
|
|
52
|
+
score=0.8,
|
|
53
|
+
),
|
|
54
|
+
],
|
|
55
|
+
context=["mrn", "medical record", "patient id", "chart"],
|
|
56
|
+
)
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Aadhaar recognizer with Verhoeff checksum validation.
|
|
2
|
+
|
|
3
|
+
A 12-digit Aadhaar is easy to confuse with a phone number, an amount, or a date
|
|
4
|
+
if you match on shape alone -- that is exactly why the old build mis-tagged
|
|
5
|
+
Aadhaar as ``DATE_TIME``. Validating the Verhoeff checksum lets us assign a high
|
|
6
|
+
confidence score and win overlap resolution against the generic recognizers,
|
|
7
|
+
while keeping false positives low.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from typing import List, Optional
|
|
13
|
+
|
|
14
|
+
from presidio_analyzer import EntityRecognizer, RecognizerResult
|
|
15
|
+
from presidio_analyzer.nlp_engine import NlpArtifacts
|
|
16
|
+
|
|
17
|
+
# Verhoeff algorithm tables.
|
|
18
|
+
_D = [
|
|
19
|
+
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
|
20
|
+
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
|
|
21
|
+
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
|
|
22
|
+
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
|
|
23
|
+
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
|
|
24
|
+
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
|
|
25
|
+
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
|
|
26
|
+
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
|
|
27
|
+
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
|
|
28
|
+
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
|
|
29
|
+
]
|
|
30
|
+
_P = [
|
|
31
|
+
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
|
32
|
+
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
|
|
33
|
+
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
|
|
34
|
+
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
|
|
35
|
+
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
|
|
36
|
+
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
|
|
37
|
+
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
|
|
38
|
+
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _verhoeff_valid(number: str) -> bool:
|
|
43
|
+
c = 0
|
|
44
|
+
for i, digit in enumerate(reversed(number)):
|
|
45
|
+
c = _D[c][_P[i % 8][int(digit)]]
|
|
46
|
+
return c == 0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# 12 digits, optionally grouped 4-4-4 by spaces or hyphens. First digit is 2-9.
|
|
50
|
+
# The lookarounds ensure the 12-digit run is not part of a LONGER grouped number
|
|
51
|
+
# (e.g. the first 12 digits of a 16-digit credit card) -- without them the
|
|
52
|
+
# recognizer would shadow CREDIT_CARD and leak the trailing group.
|
|
53
|
+
_AADHAAR_RE = re.compile(
|
|
54
|
+
r"(?<!\d)(?<![\d][\s-])" # not preceded by a digit or a digit-group separator
|
|
55
|
+
r"[2-9][0-9]{3}[\s-]?[0-9]{4}[\s-]?[0-9]{4}"
|
|
56
|
+
r"(?![\s-]?[0-9])" # not followed by another digit group
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class AadhaarRecognizer(EntityRecognizer):
|
|
61
|
+
"""Detects Indian Aadhaar numbers, verified with the Verhoeff checksum."""
|
|
62
|
+
|
|
63
|
+
def __init__(self) -> None:
|
|
64
|
+
super().__init__(supported_entities=["IN_AADHAAR"], name="aadhaar_recognizer")
|
|
65
|
+
|
|
66
|
+
def load(self) -> None: # required by the EntityRecognizer contract
|
|
67
|
+
return None
|
|
68
|
+
|
|
69
|
+
def analyze(
|
|
70
|
+
self, text: str, entities: List[str], nlp_artifacts: Optional[NlpArtifacts] = None
|
|
71
|
+
) -> List[RecognizerResult]:
|
|
72
|
+
if "IN_AADHAAR" not in entities:
|
|
73
|
+
return []
|
|
74
|
+
|
|
75
|
+
results: List[RecognizerResult] = []
|
|
76
|
+
for match in _AADHAAR_RE.finditer(text):
|
|
77
|
+
digits = re.sub(r"[\s-]", "", match.group())
|
|
78
|
+
if len(digits) != 12:
|
|
79
|
+
continue
|
|
80
|
+
# A valid Verhoeff checksum -> high confidence. A well-shaped but
|
|
81
|
+
# unverified match still gets a moderate score so it is masked, just
|
|
82
|
+
# with lower priority in overlap resolution.
|
|
83
|
+
score = 0.9 if _verhoeff_valid(digits) else 0.5
|
|
84
|
+
results.append(
|
|
85
|
+
RecognizerResult(
|
|
86
|
+
entity_type="IN_AADHAAR",
|
|
87
|
+
start=match.start(),
|
|
88
|
+
end=match.end(),
|
|
89
|
+
score=score,
|
|
90
|
+
analysis_explanation=None,
|
|
91
|
+
)
|
|
92
|
+
)
|
|
93
|
+
return results
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Aadhaar recognizer: Verhoeff-valid numbers score high; junk is rejected."""
|
|
2
|
+
from medguardx.recognizers.aadhaar import AadhaarRecognizer, _verhoeff_valid
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def _run(text):
|
|
6
|
+
return AadhaarRecognizer().analyze(text, entities=["IN_AADHAAR"])
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def test_verhoeff_known_valid_and_invalid():
|
|
10
|
+
# 234123412346 has a valid Verhoeff check digit; flipping it invalidates.
|
|
11
|
+
assert _verhoeff_valid("234123412346")
|
|
12
|
+
assert not _verhoeff_valid("234123412340")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_detects_spaced_aadhaar():
|
|
16
|
+
res = _run("Aadhaar 2341 2341 2346 on record")
|
|
17
|
+
assert len(res) == 1
|
|
18
|
+
assert res[0].entity_type == "IN_AADHAAR"
|
|
19
|
+
assert res[0].score >= 0.9 # valid checksum -> high confidence
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def test_detects_bare_twelve_digit_aadhaar():
|
|
23
|
+
# The exact case the old build missed entirely (returned nothing).
|
|
24
|
+
res = _run("ID 234123412346 filed")
|
|
25
|
+
assert len(res) == 1 and res[0].entity_type == "IN_AADHAAR"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_shape_match_without_valid_checksum_still_flagged_lower():
|
|
29
|
+
res = _run("Number 234123412340 here")
|
|
30
|
+
assert len(res) == 1
|
|
31
|
+
assert 0.4 <= res[0].score < 0.9
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_ignores_non_aadhaar_numbers():
|
|
35
|
+
assert _run("Call 9305597756 today") == [] # 10 digits, not Aadhaar-shaped
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_does_not_match_inside_a_credit_card():
|
|
39
|
+
# 16-digit card must NOT be picked up as a 12-digit Aadhaar.
|
|
40
|
+
assert _run("card 4111 1111 1111 1111 on file") == []
|
|
41
|
+
assert _run("4111111111111111") == []
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Masking must never leak the original substring (the old build's core bug)."""
|
|
2
|
+
from medguardx.detection import PIIEntity
|
|
3
|
+
from medguardx.enums import MaskingStrategy
|
|
4
|
+
from medguardx.masking import mask_text
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def _ent(text, start, etype, score=0.9):
|
|
8
|
+
return PIIEntity(entity_type=etype, start=start, end=start + len(text), score=score, text=text)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_full_access_returns_text_unchanged():
|
|
12
|
+
txt = "Card 4111 1111 1111 1111"
|
|
13
|
+
ents = [_ent("4111 1111 1111 1111", 5, "CREDIT_CARD")]
|
|
14
|
+
assert mask_text(txt, ents, MaskingStrategy.FULL_ACCESS) == txt
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def test_full_anonymize_removes_every_entity_substring():
|
|
18
|
+
txt = "IP 192.168.1.55 and IBAN DE89370400440532013000 here."
|
|
19
|
+
ents = [_ent("192.168.1.55", 3, "IP_ADDRESS"), _ent("DE89370400440532013000", 25, "IBAN_CODE")]
|
|
20
|
+
out = mask_text(txt, ents, MaskingStrategy.FULL_ANONYMIZE)
|
|
21
|
+
assert "192.168.1.55" not in out
|
|
22
|
+
assert "DE89370400440532013000" not in out
|
|
23
|
+
assert "[IP_REDACTED]" in out and "[IBAN_REDACTED]" in out
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def test_partial_card_reveals_only_last_four():
|
|
27
|
+
txt = "Card 4111111111111111 on file"
|
|
28
|
+
ents = [_ent("4111111111111111", 5, "CREDIT_CARD")]
|
|
29
|
+
out = mask_text(txt, ents, MaskingStrategy.PARTIAL_MASK)
|
|
30
|
+
assert "4111111111111111" not in out
|
|
31
|
+
assert "1111" in out # last 4 kept for utility
|
|
32
|
+
assert out.count("1") == 4 # nothing else from the PAN survives
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_partial_ip_and_iban_are_fully_redacted_not_leaked():
|
|
36
|
+
# These have no partial operator -> must be fully redacted, unlike the old
|
|
37
|
+
# build which left "********1.55" and "********00440532013000".
|
|
38
|
+
txt = "IP 192.168.1.55, IBAN DE89370400440532013000"
|
|
39
|
+
ents = [_ent("192.168.1.55", 3, "IP_ADDRESS"), _ent("DE89370400440532013000", 22, "IBAN_CODE")]
|
|
40
|
+
out = mask_text(txt, ents, MaskingStrategy.PARTIAL_MASK)
|
|
41
|
+
assert "1.55" not in out
|
|
42
|
+
assert "440532013000" not in out
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_deny_returns_denied_placeholder_only():
|
|
46
|
+
out = mask_text("secret", [], MaskingStrategy.DENY)
|
|
47
|
+
assert "secret" not in out and "DENIED" in out.upper()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_stale_span_is_skipped_not_crashing():
|
|
51
|
+
txt = "short"
|
|
52
|
+
ents = [_ent("this is longer than the text", 0, "PERSON")]
|
|
53
|
+
# end past len(text): span skipped rather than corrupting output
|
|
54
|
+
assert mask_text(txt, ents, MaskingStrategy.FULL_ANONYMIZE) == txt
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Overlap resolution: a specialized entity must beat a generic one deterministically."""
|
|
2
|
+
from medguardx.detection import PIIEntity, resolve_overlaps
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def _ent(etype, start, end, score):
|
|
6
|
+
return PIIEntity(entity_type=etype, start=start, end=end, score=score, text="x")
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def test_phone_beats_datetime_even_at_lower_score():
|
|
10
|
+
# The exact failure from the old build: DATE_TIME(0.85) shadowed PHONE(0.75).
|
|
11
|
+
ents = [_ent("DATE_TIME", 0, 10, 0.85), _ent("PHONE_NUMBER", 0, 10, 0.75)]
|
|
12
|
+
out = resolve_overlaps(ents)
|
|
13
|
+
assert len(out) == 1
|
|
14
|
+
assert out[0].entity_type == "PHONE_NUMBER"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def test_aadhaar_beats_datetime():
|
|
18
|
+
ents = [_ent("DATE_TIME", 5, 19, 0.85), _ent("IN_AADHAAR", 5, 19, 0.9)]
|
|
19
|
+
out = resolve_overlaps(ents)
|
|
20
|
+
assert len(out) == 1 and out[0].entity_type == "IN_AADHAAR"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_email_beats_overlapping_url():
|
|
24
|
+
ents = [_ent("URL", 10, 21, 0.5), _ent("EMAIL_ADDRESS", 5, 21, 1.0)]
|
|
25
|
+
out = resolve_overlaps(ents)
|
|
26
|
+
assert len(out) == 1 and out[0].entity_type == "EMAIL_ADDRESS"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_non_overlapping_entities_all_kept_and_sorted():
|
|
30
|
+
ents = [_ent("PERSON", 20, 30, 0.9), _ent("PHONE_NUMBER", 0, 10, 0.8)]
|
|
31
|
+
out = resolve_overlaps(ents)
|
|
32
|
+
assert [e.entity_type for e in out] == ["PHONE_NUMBER", "PERSON"]
|
|
33
|
+
assert out[0].start < out[1].start
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_deterministic_same_input_same_output():
|
|
37
|
+
ents = [_ent("DATE_TIME", 0, 10, 0.85), _ent("PHONE_NUMBER", 0, 10, 0.75)]
|
|
38
|
+
assert resolve_overlaps(list(ents)) == resolve_overlaps(list(reversed(ents)))
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Policy engine: correct strategies and deny-by-default for unmapped combos."""
|
|
2
|
+
from medguardx.enums import MaskingStrategy, Purpose, Role
|
|
3
|
+
from medguardx.policy import PolicyEngine
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def setup_function():
|
|
7
|
+
global engine
|
|
8
|
+
engine = PolicyEngine()
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_doctor_treatment_consent_is_full_access():
|
|
12
|
+
strat, _ = engine.evaluate(Role.DOCTOR, Purpose.TREATMENT, True)
|
|
13
|
+
assert strat == MaskingStrategy.FULL_ACCESS
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_researcher_research_is_anonymized_regardless_of_consent():
|
|
17
|
+
for consent in (True, False):
|
|
18
|
+
strat, _ = engine.evaluate(Role.RESEARCHER, Purpose.RESEARCH, consent)
|
|
19
|
+
assert strat == MaskingStrategy.FULL_ANONYMIZE
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def test_unmapped_combination_is_denied():
|
|
23
|
+
strat, rule = engine.evaluate(Role.COMPANY, Purpose.TREATMENT, False)
|
|
24
|
+
assert strat == MaskingStrategy.DENY
|
|
25
|
+
assert "denied" in rule.lower()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_accepts_plain_strings():
|
|
29
|
+
strat, _ = engine.evaluate("nurse", "treatment", False)
|
|
30
|
+
assert strat == MaskingStrategy.PARTIAL_MASK
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_custom_rules_override_defaults():
|
|
34
|
+
custom = {(Role.COMPANY, Purpose.TREATMENT, True): (MaskingStrategy.FULL_ACCESS, "custom")}
|
|
35
|
+
e = PolicyEngine(rules=custom)
|
|
36
|
+
strat, rule = e.evaluate(Role.COMPANY, Purpose.TREATMENT, True)
|
|
37
|
+
assert strat == MaskingStrategy.FULL_ACCESS and rule == "custom"
|
|
38
|
+
# anything else still deny-by-default
|
|
39
|
+
assert e.evaluate(Role.DOCTOR, Purpose.TREATMENT, True)[0] == MaskingStrategy.DENY
|