datafog 4.6.0__tar.gz → 4.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {datafog-4.6.0 → datafog-4.7.0}/PKG-INFO +37 -11
- {datafog-4.6.0 → datafog-4.7.0}/README.md +36 -10
- datafog-4.7.0/datafog/__about__.py +1 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/__init__.py +28 -2
- {datafog-4.6.0 → datafog-4.7.0}/datafog/engine.py +90 -1
- {datafog-4.6.0 → datafog-4.7.0}/datafog/integrations/claude_code.py +41 -5
- {datafog-4.6.0 → datafog-4.7.0}/datafog/integrations/litellm_guardrail.py +35 -11
- datafog-4.7.0/datafog/py.typed +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/PKG-INFO +37 -11
- {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/SOURCES.txt +2 -0
- {datafog-4.6.0 → datafog-4.7.0}/setup.py +1 -0
- datafog-4.7.0/tests/test_allowlist.py +166 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_claude_code_hook.py +30 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_litellm_guardrail.py +14 -0
- datafog-4.6.0/datafog/__about__.py +0 -1
- {datafog-4.6.0 → datafog-4.7.0}/LICENSE +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/__init___lean.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/__init___original.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/agent.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/client.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/config.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/core.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/exceptions.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/integrations/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/main.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/main_lean.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/main_original.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/models/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/models/annotator.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/models/anonymizer.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/models/common.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/models/spacy_nlp.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/donut_processor.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/image_downloader.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/pytesseract_processor.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/spark_processing/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/spark_processing/pyspark_udfs.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/gliner_annotator.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/regex_annotator/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/regex_annotator/regex_annotator.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/spacy_pii_annotator.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/services/__init__.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/services/image_service.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/services/spark_service.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/services/text_service.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/services/text_service_lean.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/services/text_service_original.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog/telemetry.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/dependency_links.txt +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/entry_points.txt +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/requires.txt +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/top_level.txt +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/setup.cfg +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_agent_api.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_anonymizer.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_cli_smoke.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_client.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_de_pii_regex.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_detection_accuracy.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_donut_lazy_import.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_engine_api.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_gliner_annotator.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_image_service.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_install_profiles.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_main.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_no_network_core.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_ocr_integration.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_regex_annotator.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_runtime_dependency_safety.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_spark_integration.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_telemetry.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_text_service.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_text_service_integration.py +0 -0
- {datafog-4.6.0 → datafog-4.7.0}/tests/test_v44_bridge_api.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: datafog
|
|
3
|
-
Version: 4.
|
|
3
|
+
Version: 4.7.0
|
|
4
4
|
Summary: Lightning-fast PII detection and anonymization library with 190x performance advantage
|
|
5
5
|
Author: Sid Mohan
|
|
6
6
|
Author-email: sid@datafog.ai
|
|
@@ -112,17 +112,43 @@ DataFog is a Python library for detecting and redacting personally identifiable
|
|
|
112
112
|
It provides:
|
|
113
113
|
|
|
114
114
|
- Fast structured PII detection via regex
|
|
115
|
+
- An offline PII firewall for AI agents: a Claude Code hook and a LiteLLM
|
|
116
|
+
gateway guardrail (new in 4.6)
|
|
115
117
|
- Optional NER support via spaCy and GLiNER
|
|
116
118
|
- A simple agent-oriented API for LLM applications
|
|
117
119
|
- Backward-compatible `DataFog` and `TextService` classes
|
|
118
120
|
|
|
119
|
-
## 4.
|
|
121
|
+
## Agent & Gateway Firewall (4.6)
|
|
120
122
|
|
|
121
|
-
DataFog 4.
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
123
|
+
DataFog 4.6 adds two ready-made enforcement points that catch PII at the
|
|
124
|
+
moment it would leave your machine — offline, in microseconds, with matched
|
|
125
|
+
values never echoed into logs or transcripts:
|
|
126
|
+
|
|
127
|
+
- **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
|
|
128
|
+
commands, web requests, file writes, MCP tools) and warns the model when
|
|
129
|
+
prompts or tool results carry PII. ~70ms per invocation including process
|
|
130
|
+
startup. Easiest install is the
|
|
131
|
+
[Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
|
|
132
|
+
|
|
133
|
+
```
|
|
134
|
+
/plugin marketplace add DataFog/datafog-claude-plugin
|
|
135
|
+
/plugin install datafog@datafog
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Manual hook setup and limitations: [examples/claude_code_hook/](examples/claude_code_hook/).
|
|
139
|
+
|
|
140
|
+
- **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
|
|
141
|
+
requests and responses at the gateway, for any LiteLLM-proxied provider.
|
|
142
|
+
In-process (~31µs per request), no sidecar service. Setup:
|
|
143
|
+
[examples/litellm_guardrail/](examples/litellm_guardrail/).
|
|
144
|
+
|
|
145
|
+
Both default to the high-precision entity set (`EMAIL`, `PHONE`,
|
|
146
|
+
`CREDIT_CARD`, `SSN`); noisier types are opt-in. Known-safe values can be
|
|
147
|
+
exempted with an allowlist: `scan(text, allowlist=[...])` for exact values,
|
|
148
|
+
`allowlist_patterns=[...]` for full-match regexes (e.g. `^\d{10}$` to stop
|
|
149
|
+
unix timestamps matching as phone numbers) — available in both adapters and
|
|
150
|
+
the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
|
|
151
|
+
`US_SSN`) are accepted as aliases for easy migration.
|
|
126
152
|
|
|
127
153
|
## Installation
|
|
128
154
|
|
|
@@ -149,7 +175,7 @@ pip install datafog[all]
|
|
|
149
175
|
Python 3.13 support is certified for the core SDK, CLI, `nlp`,
|
|
150
176
|
`nlp-advanced`, and `ocr` install profiles. Donut OCR still requires a model
|
|
151
177
|
that is available locally before runtime use. `distributed` and `all` are not
|
|
152
|
-
newly certified on Python 3.13 in the 4.
|
|
178
|
+
newly certified on Python 3.13 in the 4.x line.
|
|
153
179
|
|
|
154
180
|
## Quick Start
|
|
155
181
|
|
|
@@ -224,7 +250,7 @@ Use the engine that matches your accuracy and dependency constraints:
|
|
|
224
250
|
|
|
225
251
|
- `regex`:
|
|
226
252
|
- Fastest and always available.
|
|
227
|
-
- Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE
|
|
253
|
+
- Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE` (`DOB` and `ZIP` are accepted as input aliases).
|
|
228
254
|
- Use `locales=["de"]` for German structured IDs such as `DE_VAT_ID`, `DE_IBAN`, `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
|
|
229
255
|
- `spacy`:
|
|
230
256
|
- Requires `pip install datafog[nlp]`.
|
|
@@ -238,7 +264,7 @@ Use the engine that matches your accuracy and dependency constraints:
|
|
|
238
264
|
|
|
239
265
|
## Optional OCR And Spark Surfaces
|
|
240
266
|
|
|
241
|
-
|
|
267
|
+
The 4.x line keeps the main package story centered on lightweight text PII
|
|
242
268
|
screening. OCR and Spark remain supported optional surfaces for users who
|
|
243
269
|
already rely on them, but they are not required for the core import, default
|
|
244
270
|
scan/redact helpers, or guardrail helpers.
|
|
@@ -258,7 +284,7 @@ scan/redact helpers, or guardrail helpers.
|
|
|
258
284
|
- A Java runtime is required by PySpark.
|
|
259
285
|
|
|
260
286
|
OCR and Spark are not deprecated. Their broader API and packaging overhaul is
|
|
261
|
-
deferred; the 4.
|
|
287
|
+
deferred; the 4.x goal is to keep them explicit, documented, and isolated from
|
|
262
288
|
the lightweight core path.
|
|
263
289
|
|
|
264
290
|
## Backward-Compatible APIs
|
|
@@ -5,17 +5,43 @@ DataFog is a Python library for detecting and redacting personally identifiable
|
|
|
5
5
|
It provides:
|
|
6
6
|
|
|
7
7
|
- Fast structured PII detection via regex
|
|
8
|
+
- An offline PII firewall for AI agents: a Claude Code hook and a LiteLLM
|
|
9
|
+
gateway guardrail (new in 4.6)
|
|
8
10
|
- Optional NER support via spaCy and GLiNER
|
|
9
11
|
- A simple agent-oriented API for LLM applications
|
|
10
12
|
- Backward-compatible `DataFog` and `TextService` classes
|
|
11
13
|
|
|
12
|
-
## 4.
|
|
14
|
+
## Agent & Gateway Firewall (4.6)
|
|
13
15
|
|
|
14
|
-
DataFog 4.
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
16
|
+
DataFog 4.6 adds two ready-made enforcement points that catch PII at the
|
|
17
|
+
moment it would leave your machine — offline, in microseconds, with matched
|
|
18
|
+
values never echoed into logs or transcripts:
|
|
19
|
+
|
|
20
|
+
- **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
|
|
21
|
+
commands, web requests, file writes, MCP tools) and warns the model when
|
|
22
|
+
prompts or tool results carry PII. ~70ms per invocation including process
|
|
23
|
+
startup. Easiest install is the
|
|
24
|
+
[Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
/plugin marketplace add DataFog/datafog-claude-plugin
|
|
28
|
+
/plugin install datafog@datafog
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Manual hook setup and limitations: [examples/claude_code_hook/](examples/claude_code_hook/).
|
|
32
|
+
|
|
33
|
+
- **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
|
|
34
|
+
requests and responses at the gateway, for any LiteLLM-proxied provider.
|
|
35
|
+
In-process (~31µs per request), no sidecar service. Setup:
|
|
36
|
+
[examples/litellm_guardrail/](examples/litellm_guardrail/).
|
|
37
|
+
|
|
38
|
+
Both default to the high-precision entity set (`EMAIL`, `PHONE`,
|
|
39
|
+
`CREDIT_CARD`, `SSN`); noisier types are opt-in. Known-safe values can be
|
|
40
|
+
exempted with an allowlist: `scan(text, allowlist=[...])` for exact values,
|
|
41
|
+
`allowlist_patterns=[...]` for full-match regexes (e.g. `^\d{10}$` to stop
|
|
42
|
+
unix timestamps matching as phone numbers) — available in both adapters and
|
|
43
|
+
the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
|
|
44
|
+
`US_SSN`) are accepted as aliases for easy migration.
|
|
19
45
|
|
|
20
46
|
## Installation
|
|
21
47
|
|
|
@@ -42,7 +68,7 @@ pip install datafog[all]
|
|
|
42
68
|
Python 3.13 support is certified for the core SDK, CLI, `nlp`,
|
|
43
69
|
`nlp-advanced`, and `ocr` install profiles. Donut OCR still requires a model
|
|
44
70
|
that is available locally before runtime use. `distributed` and `all` are not
|
|
45
|
-
newly certified on Python 3.13 in the 4.
|
|
71
|
+
newly certified on Python 3.13 in the 4.x line.
|
|
46
72
|
|
|
47
73
|
## Quick Start
|
|
48
74
|
|
|
@@ -117,7 +143,7 @@ Use the engine that matches your accuracy and dependency constraints:
|
|
|
117
143
|
|
|
118
144
|
- `regex`:
|
|
119
145
|
- Fastest and always available.
|
|
120
|
-
- Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE
|
|
146
|
+
- Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE` (`DOB` and `ZIP` are accepted as input aliases).
|
|
121
147
|
- Use `locales=["de"]` for German structured IDs such as `DE_VAT_ID`, `DE_IBAN`, `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
|
|
122
148
|
- `spacy`:
|
|
123
149
|
- Requires `pip install datafog[nlp]`.
|
|
@@ -131,7 +157,7 @@ Use the engine that matches your accuracy and dependency constraints:
|
|
|
131
157
|
|
|
132
158
|
## Optional OCR And Spark Surfaces
|
|
133
159
|
|
|
134
|
-
|
|
160
|
+
The 4.x line keeps the main package story centered on lightweight text PII
|
|
135
161
|
screening. OCR and Spark remain supported optional surfaces for users who
|
|
136
162
|
already rely on them, but they are not required for the core import, default
|
|
137
163
|
scan/redact helpers, or guardrail helpers.
|
|
@@ -151,7 +177,7 @@ scan/redact helpers, or guardrail helpers.
|
|
|
151
177
|
- A Java runtime is required by PySpark.
|
|
152
178
|
|
|
153
179
|
OCR and Spark are not deprecated. Their broader API and packaging overhaul is
|
|
154
|
-
deferred; the 4.
|
|
180
|
+
deferred; the 4.x goal is to keep them explicit, documented, and isolated from
|
|
155
181
|
the lightweight core path.
|
|
156
182
|
|
|
157
183
|
## Backward-Compatible APIs
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "4.7.0"
|
|
@@ -153,14 +153,28 @@ def scan(
|
|
|
153
153
|
engine: str = "regex",
|
|
154
154
|
entity_types: list[str] | None = None,
|
|
155
155
|
locales: list[str] | None = None,
|
|
156
|
+
allowlist: list[str] | None = None,
|
|
157
|
+
allowlist_patterns: list[str] | None = None,
|
|
156
158
|
) -> ScanResult:
|
|
157
159
|
"""
|
|
158
160
|
v5-preview scan entrypoint.
|
|
159
161
|
|
|
160
162
|
Defaults to the lightweight regex engine so the core install works without
|
|
161
163
|
optional dependency fallback warnings.
|
|
164
|
+
|
|
165
|
+
``allowlist`` exempts exact entity texts (your own support address, doc
|
|
166
|
+
placeholders); ``allowlist_patterns`` exempts entities whose full text
|
|
167
|
+
matches a regex (e.g. ``^\\d{10}$`` so unix timestamps stop matching as
|
|
168
|
+
phone numbers).
|
|
162
169
|
"""
|
|
163
|
-
return _scan(
|
|
170
|
+
return _scan(
|
|
171
|
+
text=text,
|
|
172
|
+
engine=engine,
|
|
173
|
+
entity_types=entity_types,
|
|
174
|
+
locales=locales,
|
|
175
|
+
allowlist=allowlist,
|
|
176
|
+
allowlist_patterns=allowlist_patterns,
|
|
177
|
+
)
|
|
164
178
|
|
|
165
179
|
|
|
166
180
|
def redact(
|
|
@@ -171,12 +185,17 @@ def redact(
|
|
|
171
185
|
strategy: str = "token",
|
|
172
186
|
preset: str | None = None,
|
|
173
187
|
locales: list[str] | None = None,
|
|
188
|
+
allowlist: list[str] | None = None,
|
|
189
|
+
allowlist_patterns: list[str] | None = None,
|
|
174
190
|
) -> RedactResult:
|
|
175
191
|
"""
|
|
176
192
|
v5-preview redaction entrypoint.
|
|
177
193
|
|
|
178
194
|
If entities are provided, redact those spans. Otherwise, scan text first
|
|
179
|
-
using the selected engine and redact the detected entities.
|
|
195
|
+
using the selected engine and redact the detected entities. ``allowlist``
|
|
196
|
+
and ``allowlist_patterns`` exempt findings from redaction (exact text and
|
|
197
|
+
full-text regex match respectively); they apply to the scan path and are
|
|
198
|
+
rejected when explicit ``entities`` are supplied.
|
|
180
199
|
"""
|
|
181
200
|
if preset is not None:
|
|
182
201
|
try:
|
|
@@ -186,6 +205,11 @@ def redact(
|
|
|
186
205
|
raise ValueError(f"preset must be one of: {allowed}") from exc
|
|
187
206
|
|
|
188
207
|
if entities is not None:
|
|
208
|
+
if allowlist or allowlist_patterns:
|
|
209
|
+
raise ValueError(
|
|
210
|
+
"allowlist/allowlist_patterns cannot be combined with explicit "
|
|
211
|
+
"entities; filter the entities before calling redact"
|
|
212
|
+
)
|
|
189
213
|
return _redact_entities(text=text, entities=entities, strategy=strategy)
|
|
190
214
|
|
|
191
215
|
return _scan_and_redact(
|
|
@@ -194,6 +218,8 @@ def redact(
|
|
|
194
218
|
entity_types=entity_types,
|
|
195
219
|
strategy=strategy,
|
|
196
220
|
locales=locales,
|
|
221
|
+
allowlist=allowlist,
|
|
222
|
+
allowlist_patterns=allowlist_patterns,
|
|
197
223
|
)
|
|
198
224
|
|
|
199
225
|
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import hashlib
|
|
6
|
+
import re
|
|
6
7
|
import warnings
|
|
7
8
|
from dataclasses import dataclass
|
|
8
9
|
from functools import lru_cache
|
|
@@ -23,6 +24,9 @@ CANONICAL_TYPE_MAP = {
|
|
|
23
24
|
"SOCIAL_SECURITY_NUMBER": "SSN",
|
|
24
25
|
"CREDIT_CARD_NUMBER": "CREDIT_CARD",
|
|
25
26
|
"DATE_OF_BIRTH": "DATE",
|
|
27
|
+
# Presidio-compatible aliases, so configs migrate without renames.
|
|
28
|
+
"EMAIL_ADDRESS": "EMAIL",
|
|
29
|
+
"US_SSN": "SSN",
|
|
26
30
|
}
|
|
27
31
|
|
|
28
32
|
ALL_ENTITY_TYPES = {
|
|
@@ -277,6 +281,74 @@ def _filter_entity_types(
|
|
|
277
281
|
return [entity for entity in entities if entity.type in allowed]
|
|
278
282
|
|
|
279
283
|
|
|
284
|
+
# Python's re module backtracks; a quantified group containing another
|
|
285
|
+
# quantifier (e.g. ``(a+)+``) can take exponential time on adversarial
|
|
286
|
+
# input, and entity text can be attacker-influenced (LLM messages, tool
|
|
287
|
+
# output). Reject that construct outright rather than matching under it.
|
|
288
|
+
_NESTED_QUANTIFIER = re.compile(
|
|
289
|
+
r"\((?:[^()\\]|\\.)*(?<!\\)[+*}](?:[^()\\]|\\.)*\)\s*[+*{]"
|
|
290
|
+
)
|
|
291
|
+
MAX_ALLOWLIST_PATTERN_LENGTH = 512
|
|
292
|
+
# Entities longer than this skip pattern matching (fail-safe: the finding
|
|
293
|
+
# is kept, never suppressed) so match time stays bounded.
|
|
294
|
+
MAX_PATTERN_SUBJECT_LENGTH = 512
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _compile_allowlist_patterns(
|
|
298
|
+
allowlist_patterns: Optional[list[str]],
|
|
299
|
+
) -> list["re.Pattern[str]"]:
|
|
300
|
+
compiled = []
|
|
301
|
+
for raw in allowlist_patterns or []:
|
|
302
|
+
if len(raw) > MAX_ALLOWLIST_PATTERN_LENGTH:
|
|
303
|
+
raise ValueError(
|
|
304
|
+
"allowlist_patterns entries must be at most "
|
|
305
|
+
f"{MAX_ALLOWLIST_PATTERN_LENGTH} characters"
|
|
306
|
+
)
|
|
307
|
+
if _NESTED_QUANTIFIER.search(raw):
|
|
308
|
+
raise ValueError(
|
|
309
|
+
"allowlist_patterns contains a quantified group with a nested "
|
|
310
|
+
f"quantifier ({raw!r}), which risks catastrophic backtracking; "
|
|
311
|
+
"rewrite the pattern without nesting quantifiers"
|
|
312
|
+
)
|
|
313
|
+
try:
|
|
314
|
+
compiled.append(re.compile(raw))
|
|
315
|
+
except re.error as exc:
|
|
316
|
+
raise ValueError(
|
|
317
|
+
f"allowlist_patterns contains an invalid regex: {raw!r} ({exc})"
|
|
318
|
+
) from None
|
|
319
|
+
return compiled
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _apply_allowlist(
|
|
323
|
+
entities: list[Entity],
|
|
324
|
+
allowlist: Optional[list[str]],
|
|
325
|
+
allowlist_patterns: Optional[list[str]],
|
|
326
|
+
) -> list[Entity]:
|
|
327
|
+
"""Drop entities whose exact text is allowlisted.
|
|
328
|
+
|
|
329
|
+
Matching semantics, deliberately strict for a security boundary:
|
|
330
|
+
exact values are case-sensitive with no Unicode normalization, and
|
|
331
|
+
patterns must fullmatch the entity text, so a partial match never
|
|
332
|
+
suppresses a finding. Allowlist entries and patterns are operator
|
|
333
|
+
configuration; treat them like code and never accept them from end
|
|
334
|
+
users.
|
|
335
|
+
"""
|
|
336
|
+
if not allowlist and not allowlist_patterns:
|
|
337
|
+
return entities
|
|
338
|
+
exact = set(allowlist or [])
|
|
339
|
+
patterns = _compile_allowlist_patterns(allowlist_patterns)
|
|
340
|
+
return [
|
|
341
|
+
entity
|
|
342
|
+
for entity in entities
|
|
343
|
+
if entity.text not in exact
|
|
344
|
+
and not any(
|
|
345
|
+
pattern.fullmatch(entity.text)
|
|
346
|
+
for pattern in patterns
|
|
347
|
+
if len(entity.text) <= MAX_PATTERN_SUBJECT_LENGTH
|
|
348
|
+
)
|
|
349
|
+
]
|
|
350
|
+
|
|
351
|
+
|
|
280
352
|
def _needs_ner(entity_types: Optional[list[str]]) -> bool:
|
|
281
353
|
if entity_types is None:
|
|
282
354
|
return True
|
|
@@ -289,14 +361,25 @@ def scan(
|
|
|
289
361
|
engine: str = "smart",
|
|
290
362
|
entity_types: Optional[list[str]] = None,
|
|
291
363
|
locales: Optional[list[str]] = None,
|
|
364
|
+
allowlist: Optional[list[str]] = None,
|
|
365
|
+
allowlist_patterns: Optional[list[str]] = None,
|
|
292
366
|
) -> ScanResult:
|
|
293
|
-
"""Scan text for PII entities.
|
|
367
|
+
"""Scan text for PII entities.
|
|
368
|
+
|
|
369
|
+
``allowlist`` exempts exact entity texts (e.g. your own support email);
|
|
370
|
+
``allowlist_patterns`` exempts entities whose full text matches a regex
|
|
371
|
+
(e.g. ``^\\d{10}$`` to stop unix timestamps matching as phone numbers).
|
|
372
|
+
"""
|
|
294
373
|
if not isinstance(text, str):
|
|
295
374
|
raise TypeError("text must be a string")
|
|
296
375
|
|
|
297
376
|
if engine not in {"regex", "spacy", "gliner", "smart"}:
|
|
298
377
|
raise ValueError("engine must be one of: regex, spacy, gliner, smart")
|
|
299
378
|
|
|
379
|
+
# Validate patterns up front so config errors fail fast even when the
|
|
380
|
+
# text contains no entities.
|
|
381
|
+
_compile_allowlist_patterns(allowlist_patterns)
|
|
382
|
+
|
|
300
383
|
regex_entities = _regex_entities(
|
|
301
384
|
text,
|
|
302
385
|
entity_types=entity_types,
|
|
@@ -305,6 +388,7 @@ def scan(
|
|
|
305
388
|
|
|
306
389
|
if engine == "regex":
|
|
307
390
|
filtered = _filter_entity_types(regex_entities, entity_types)
|
|
391
|
+
filtered = _apply_allowlist(filtered, allowlist, allowlist_patterns)
|
|
308
392
|
return ScanResult(
|
|
309
393
|
entities=_dedupe_entities(filtered), text=text, engine_used="regex"
|
|
310
394
|
)
|
|
@@ -367,6 +451,7 @@ def scan(
|
|
|
367
451
|
)
|
|
368
452
|
|
|
369
453
|
filtered = _filter_entity_types(combined, entity_types)
|
|
454
|
+
filtered = _apply_allowlist(filtered, allowlist, allowlist_patterns)
|
|
370
455
|
deduped = _dedupe_entities(filtered)
|
|
371
456
|
return ScanResult(
|
|
372
457
|
entities=deduped,
|
|
@@ -437,6 +522,8 @@ def scan_and_redact(
|
|
|
437
522
|
entity_types: Optional[list[str]] = None,
|
|
438
523
|
strategy: str = "token",
|
|
439
524
|
locales: Optional[list[str]] = None,
|
|
525
|
+
allowlist: Optional[list[str]] = None,
|
|
526
|
+
allowlist_patterns: Optional[list[str]] = None,
|
|
440
527
|
) -> RedactResult:
|
|
441
528
|
"""Convenience wrapper: scan then redact."""
|
|
442
529
|
scan_result = scan(
|
|
@@ -444,5 +531,7 @@ def scan_and_redact(
|
|
|
444
531
|
engine=engine,
|
|
445
532
|
entity_types=entity_types,
|
|
446
533
|
locales=locales,
|
|
534
|
+
allowlist=allowlist,
|
|
535
|
+
allowlist_patterns=allowlist_patterns,
|
|
447
536
|
)
|
|
448
537
|
return redact(text=text, entities=scan_result.entities, strategy=strategy)
|
|
@@ -16,6 +16,11 @@ Configuration (environment variables):
|
|
|
16
16
|
- ``DATAFOG_HOOK_ENTITIES``: comma-separated entity types to detect.
|
|
17
17
|
Defaults to the high-precision set; noisy-in-code types (IP_ADDRESS,
|
|
18
18
|
DOB, ZIP) must be opted into.
|
|
19
|
+
- ``DATAFOG_HOOK_ALLOWLIST``: comma-separated exact values to exempt
|
|
20
|
+
(your own support address, documentation placeholders).
|
|
21
|
+
- ``DATAFOG_HOOK_ALLOWLIST_PATTERNS``: comma-separated regexes; findings
|
|
22
|
+
whose full text matches are exempt (note: a pattern containing a comma
|
|
23
|
+
cannot be expressed here).
|
|
19
24
|
|
|
20
25
|
Failure policy: fail open. A hook bug must never brick a Claude Code
|
|
21
26
|
session, so any unexpected error exits non-blocking with no output.
|
|
@@ -59,6 +64,11 @@ def _action(env: Mapping[str, str]) -> str:
|
|
|
59
64
|
return action if action in VALID_ACTIONS else "ask"
|
|
60
65
|
|
|
61
66
|
|
|
67
|
+
def _csv_env(env: Mapping[str, str], name: str) -> list[str]:
|
|
68
|
+
raw = env.get(name, "")
|
|
69
|
+
return [item.strip() for item in raw.split(",") if item.strip()]
|
|
70
|
+
|
|
71
|
+
|
|
62
72
|
def _iter_strings(value: Any) -> Iterator[str]:
|
|
63
73
|
"""Yield every string embedded in a JSON-like structure.
|
|
64
74
|
|
|
@@ -76,7 +86,12 @@ def _iter_strings(value: Any) -> Iterator[str]:
|
|
|
76
86
|
stack.extend(current)
|
|
77
87
|
|
|
78
88
|
|
|
79
|
-
def _scan_findings(
|
|
89
|
+
def _scan_findings(
|
|
90
|
+
value: Any,
|
|
91
|
+
entity_types: list[str],
|
|
92
|
+
allowlist: list[str] | None = None,
|
|
93
|
+
allowlist_patterns: list[str] | None = None,
|
|
94
|
+
) -> dict[str, int]:
|
|
80
95
|
"""Scan all strings in ``value``; return counts per entity type."""
|
|
81
96
|
import datafog
|
|
82
97
|
|
|
@@ -87,7 +102,13 @@ def _scan_findings(value: Any, entity_types: list[str]) -> dict[str, int]:
|
|
|
87
102
|
break
|
|
88
103
|
chunk = text[: min(MAX_SCAN_CHARS, total_budget)]
|
|
89
104
|
total_budget -= len(chunk)
|
|
90
|
-
result = datafog.scan(
|
|
105
|
+
result = datafog.scan(
|
|
106
|
+
chunk,
|
|
107
|
+
engine="regex",
|
|
108
|
+
entity_types=entity_types,
|
|
109
|
+
allowlist=allowlist or None,
|
|
110
|
+
allowlist_patterns=allowlist_patterns or None,
|
|
111
|
+
)
|
|
91
112
|
for entity in result.entities:
|
|
92
113
|
counts[entity.type] = counts.get(entity.type, 0) + 1
|
|
93
114
|
return counts
|
|
@@ -104,7 +125,12 @@ def _emit(event: str, fields: dict[str, Any]) -> str:
|
|
|
104
125
|
|
|
105
126
|
|
|
106
127
|
def _handle_pre_tool_use(payload: dict, env: Mapping[str, str]) -> str:
|
|
107
|
-
counts = _scan_findings(
|
|
128
|
+
counts = _scan_findings(
|
|
129
|
+
payload.get("tool_input"),
|
|
130
|
+
_entity_types(env),
|
|
131
|
+
allowlist=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST"),
|
|
132
|
+
allowlist_patterns=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST_PATTERNS"),
|
|
133
|
+
)
|
|
108
134
|
if not counts:
|
|
109
135
|
return ""
|
|
110
136
|
tool = payload.get("tool_name", "tool")
|
|
@@ -119,7 +145,12 @@ def _handle_pre_tool_use(payload: dict, env: Mapping[str, str]) -> str:
|
|
|
119
145
|
|
|
120
146
|
|
|
121
147
|
def _handle_user_prompt_submit(payload: dict, env: Mapping[str, str]) -> str:
|
|
122
|
-
counts = _scan_findings(
|
|
148
|
+
counts = _scan_findings(
|
|
149
|
+
payload.get("prompt"),
|
|
150
|
+
_entity_types(env),
|
|
151
|
+
allowlist=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST"),
|
|
152
|
+
allowlist_patterns=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST_PATTERNS"),
|
|
153
|
+
)
|
|
123
154
|
if not counts:
|
|
124
155
|
return ""
|
|
125
156
|
context = (
|
|
@@ -130,7 +161,12 @@ def _handle_user_prompt_submit(payload: dict, env: Mapping[str, str]) -> str:
|
|
|
130
161
|
|
|
131
162
|
|
|
132
163
|
def _handle_post_tool_use(payload: dict, env: Mapping[str, str]) -> str:
|
|
133
|
-
counts = _scan_findings(
|
|
164
|
+
counts = _scan_findings(
|
|
165
|
+
payload.get("tool_response"),
|
|
166
|
+
_entity_types(env),
|
|
167
|
+
allowlist=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST"),
|
|
168
|
+
allowlist_patterns=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST_PATTERNS"),
|
|
169
|
+
)
|
|
134
170
|
if not counts:
|
|
135
171
|
return ""
|
|
136
172
|
tool = payload.get("tool_name", "tool")
|
|
@@ -46,11 +46,22 @@ VALID_FAIL_POLICIES = {"open", "closed"}
|
|
|
46
46
|
logger = logging.getLogger(__name__)
|
|
47
47
|
|
|
48
48
|
|
|
49
|
-
def _redact_text(
|
|
49
|
+
def _redact_text(
|
|
50
|
+
text: str,
|
|
51
|
+
entity_types: list[str],
|
|
52
|
+
allowlist: list[str] | None = None,
|
|
53
|
+
allowlist_patterns: list[str] | None = None,
|
|
54
|
+
) -> tuple[str, dict[str, int]]:
|
|
50
55
|
"""Redact ``text``; return (redacted_text, counts per entity type)."""
|
|
51
56
|
import datafog
|
|
52
57
|
|
|
53
|
-
result = datafog.redact(
|
|
58
|
+
result = datafog.redact(
|
|
59
|
+
text,
|
|
60
|
+
engine="regex",
|
|
61
|
+
entity_types=entity_types,
|
|
62
|
+
allowlist=allowlist,
|
|
63
|
+
allowlist_patterns=allowlist_patterns,
|
|
64
|
+
)
|
|
54
65
|
counts: dict[str, int] = {}
|
|
55
66
|
for entity in result.entities:
|
|
56
67
|
counts[entity.type] = counts.get(entity.type, 0) + 1
|
|
@@ -69,6 +80,8 @@ class DataFogGuardrail(CustomGuardrail):
|
|
|
69
80
|
action: str = "redact",
|
|
70
81
|
entity_types: Optional[list[str]] = None,
|
|
71
82
|
fail_policy: str = "open",
|
|
83
|
+
allowlist: Optional[list[str]] = None,
|
|
84
|
+
allowlist_patterns: Optional[list[str]] = None,
|
|
72
85
|
**kwargs: Any,
|
|
73
86
|
) -> None:
|
|
74
87
|
if action not in VALID_ACTIONS:
|
|
@@ -80,13 +93,17 @@ class DataFogGuardrail(CustomGuardrail):
|
|
|
80
93
|
self.action = action
|
|
81
94
|
self.entity_types = entity_types or DEFAULT_ENTITY_TYPES
|
|
82
95
|
self.fail_policy = fail_policy
|
|
96
|
+
self.allowlist = allowlist
|
|
97
|
+
self.allowlist_patterns = allowlist_patterns
|
|
83
98
|
super().__init__(**kwargs)
|
|
84
99
|
|
|
85
100
|
def _process_content(self, content: Any) -> tuple[Any, dict[str, int]]:
|
|
86
101
|
"""Redact a message content value (str or list of content parts)."""
|
|
87
102
|
counts: dict[str, int] = {}
|
|
88
103
|
if isinstance(content, str):
|
|
89
|
-
redacted, counts = _redact_text(
|
|
104
|
+
redacted, counts = _redact_text(
|
|
105
|
+
content, self.entity_types, self.allowlist, self.allowlist_patterns
|
|
106
|
+
)
|
|
90
107
|
return redacted, counts
|
|
91
108
|
if isinstance(content, list):
|
|
92
109
|
new_parts = []
|
|
@@ -94,7 +111,10 @@ class DataFogGuardrail(CustomGuardrail):
|
|
|
94
111
|
for part in content:
|
|
95
112
|
if isinstance(part, dict) and isinstance(part.get("text"), str):
|
|
96
113
|
redacted, part_counts = _redact_text(
|
|
97
|
-
part["text"],
|
|
114
|
+
part["text"],
|
|
115
|
+
self.entity_types,
|
|
116
|
+
self.allowlist,
|
|
117
|
+
self.allowlist_patterns,
|
|
98
118
|
)
|
|
99
119
|
new_parts.append({**part, "text": redacted})
|
|
100
120
|
for etype, n in part_counts.items():
|
|
@@ -160,9 +180,8 @@ class DataFogGuardrail(CustomGuardrail):
|
|
|
160
180
|
if not total_counts:
|
|
161
181
|
return data
|
|
162
182
|
|
|
163
|
-
self._record_guardrail_logging(data, total_counts)
|
|
164
|
-
|
|
165
183
|
if self.action == "block":
|
|
184
|
+
self._record_guardrail_logging(data, total_counts)
|
|
166
185
|
# HTTPException(400) is one of the exception types litellm's
|
|
167
186
|
# _is_guardrail_intervention recognizes, so the block is
|
|
168
187
|
# classified as a policy intervention (not a backend failure)
|
|
@@ -178,7 +197,9 @@ class DataFogGuardrail(CustomGuardrail):
|
|
|
178
197
|
},
|
|
179
198
|
)
|
|
180
199
|
|
|
181
|
-
|
|
200
|
+
new_data = {**data, "messages": new_messages}
|
|
201
|
+
self._record_guardrail_logging(new_data, total_counts)
|
|
202
|
+
return new_data
|
|
182
203
|
|
|
183
204
|
def _record_guardrail_logging(
|
|
184
205
|
self, data: dict, total_counts: dict[str, int]
|
|
@@ -188,9 +209,7 @@ class DataFogGuardrail(CustomGuardrail):
|
|
|
188
209
|
self.add_standard_logging_guardrail_information_to_request_data(
|
|
189
210
|
guardrail_json_response=_summary(total_counts),
|
|
190
211
|
request_data=data,
|
|
191
|
-
guardrail_status=
|
|
192
|
-
"guardrail_intervened" if self.action == "block" else "success"
|
|
193
|
-
),
|
|
212
|
+
guardrail_status="guardrail_intervened",
|
|
194
213
|
masked_entity_count=dict(total_counts),
|
|
195
214
|
)
|
|
196
215
|
except Exception: # noqa: BLE001 — observability must never break traffic
|
|
@@ -217,7 +236,12 @@ class DataFogGuardrail(CustomGuardrail):
|
|
|
217
236
|
for choice in choices:
|
|
218
237
|
message = getattr(choice, "message", None)
|
|
219
238
|
if message is not None and isinstance(message.content, str):
|
|
220
|
-
redacted, counts = _redact_text(
|
|
239
|
+
redacted, counts = _redact_text(
|
|
240
|
+
message.content,
|
|
241
|
+
self.entity_types,
|
|
242
|
+
self.allowlist,
|
|
243
|
+
self.allowlist_patterns,
|
|
244
|
+
)
|
|
221
245
|
if counts:
|
|
222
246
|
message.content = redacted
|
|
223
247
|
elif message is not None and message.content is not None:
|
|
File without changes
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: datafog
|
|
3
|
-
Version: 4.
|
|
3
|
+
Version: 4.7.0
|
|
4
4
|
Summary: Lightning-fast PII detection and anonymization library with 190x performance advantage
|
|
5
5
|
Author: Sid Mohan
|
|
6
6
|
Author-email: sid@datafog.ai
|
|
@@ -112,17 +112,43 @@ DataFog is a Python library for detecting and redacting personally identifiable
|
|
|
112
112
|
It provides:
|
|
113
113
|
|
|
114
114
|
- Fast structured PII detection via regex
|
|
115
|
+
- An offline PII firewall for AI agents: a Claude Code hook and a LiteLLM
|
|
116
|
+
gateway guardrail (new in 4.6)
|
|
115
117
|
- Optional NER support via spaCy and GLiNER
|
|
116
118
|
- A simple agent-oriented API for LLM applications
|
|
117
119
|
- Backward-compatible `DataFog` and `TextService` classes
|
|
118
120
|
|
|
119
|
-
## 4.
|
|
121
|
+
## Agent & Gateway Firewall (4.6)
|
|
120
122
|
|
|
121
|
-
DataFog 4.
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
123
|
+
DataFog 4.6 adds two ready-made enforcement points that catch PII at the
|
|
124
|
+
moment it would leave your machine — offline, in microseconds, with matched
|
|
125
|
+
values never echoed into logs or transcripts:
|
|
126
|
+
|
|
127
|
+
- **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
|
|
128
|
+
commands, web requests, file writes, MCP tools) and warns the model when
|
|
129
|
+
prompts or tool results carry PII. ~70ms per invocation including process
|
|
130
|
+
startup. Easiest install is the
|
|
131
|
+
[Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
|
|
132
|
+
|
|
133
|
+
```
|
|
134
|
+
/plugin marketplace add DataFog/datafog-claude-plugin
|
|
135
|
+
/plugin install datafog@datafog
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Manual hook setup and limitations: [examples/claude_code_hook/](examples/claude_code_hook/).
|
|
139
|
+
|
|
140
|
+
- **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
|
|
141
|
+
requests and responses at the gateway, for any LiteLLM-proxied provider.
|
|
142
|
+
In-process (~31µs per request), no sidecar service. Setup:
|
|
143
|
+
[examples/litellm_guardrail/](examples/litellm_guardrail/).
|
|
144
|
+
|
|
145
|
+
Both default to the high-precision entity set (`EMAIL`, `PHONE`,
|
|
146
|
+
`CREDIT_CARD`, `SSN`); noisier types are opt-in. Known-safe values can be
|
|
147
|
+
exempted with an allowlist: `scan(text, allowlist=[...])` for exact values,
|
|
148
|
+
`allowlist_patterns=[...]` for full-match regexes (e.g. `^\d{10}$` to stop
|
|
149
|
+
unix timestamps matching as phone numbers) — available in both adapters and
|
|
150
|
+
the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
|
|
151
|
+
`US_SSN`) are accepted as aliases for easy migration.
|
|
126
152
|
|
|
127
153
|
## Installation
|
|
128
154
|
|
|
@@ -149,7 +175,7 @@ pip install datafog[all]
|
|
|
149
175
|
Python 3.13 support is certified for the core SDK, CLI, `nlp`,
|
|
150
176
|
`nlp-advanced`, and `ocr` install profiles. Donut OCR still requires a model
|
|
151
177
|
that is available locally before runtime use. `distributed` and `all` are not
|
|
152
|
-
newly certified on Python 3.13 in the 4.
|
|
178
|
+
newly certified on Python 3.13 in the 4.x line.
|
|
153
179
|
|
|
154
180
|
## Quick Start
|
|
155
181
|
|
|
@@ -224,7 +250,7 @@ Use the engine that matches your accuracy and dependency constraints:
|
|
|
224
250
|
|
|
225
251
|
- `regex`:
|
|
226
252
|
- Fastest and always available.
|
|
227
|
-
- Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE
|
|
253
|
+
- Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE` (`DOB` and `ZIP` are accepted as input aliases).
|
|
228
254
|
- Use `locales=["de"]` for German structured IDs such as `DE_VAT_ID`, `DE_IBAN`, `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
|
|
229
255
|
- `spacy`:
|
|
230
256
|
- Requires `pip install datafog[nlp]`.
|
|
@@ -238,7 +264,7 @@ Use the engine that matches your accuracy and dependency constraints:
|
|
|
238
264
|
|
|
239
265
|
## Optional OCR And Spark Surfaces
|
|
240
266
|
|
|
241
|
-
|
|
267
|
+
The 4.x line keeps the main package story centered on lightweight text PII
|
|
242
268
|
screening. OCR and Spark remain supported optional surfaces for users who
|
|
243
269
|
already rely on them, but they are not required for the core import, default
|
|
244
270
|
scan/redact helpers, or guardrail helpers.
|
|
@@ -258,7 +284,7 @@ scan/redact helpers, or guardrail helpers.
|
|
|
258
284
|
- A Java runtime is required by PySpark.
|
|
259
285
|
|
|
260
286
|
OCR and Spark are not deprecated. Their broader API and packaging overhaul is
|
|
261
|
-
deferred; the 4.
|
|
287
|
+
deferred; the 4.x goal is to keep them explicit, documented, and isolated from
|
|
262
288
|
the lightweight core path.
|
|
263
289
|
|
|
264
290
|
## Backward-Compatible APIs
|
|
@@ -14,6 +14,7 @@ datafog/exceptions.py
|
|
|
14
14
|
datafog/main.py
|
|
15
15
|
datafog/main_lean.py
|
|
16
16
|
datafog/main_original.py
|
|
17
|
+
datafog/py.typed
|
|
17
18
|
datafog/telemetry.py
|
|
18
19
|
datafog.egg-info/PKG-INFO
|
|
19
20
|
datafog.egg-info/SOURCES.txt
|
|
@@ -48,6 +49,7 @@ datafog/services/text_service.py
|
|
|
48
49
|
datafog/services/text_service_lean.py
|
|
49
50
|
datafog/services/text_service_original.py
|
|
50
51
|
tests/test_agent_api.py
|
|
52
|
+
tests/test_allowlist.py
|
|
51
53
|
tests/test_anonymizer.py
|
|
52
54
|
tests/test_claude_code_hook.py
|
|
53
55
|
tests/test_cli_smoke.py
|
|
@@ -114,6 +114,7 @@ setup(
|
|
|
114
114
|
long_description=long_description,
|
|
115
115
|
long_description_content_type="text/markdown",
|
|
116
116
|
packages=find_packages(exclude=["tests", "tests.*"]),
|
|
117
|
+
package_data={"datafog": ["py.typed"]},
|
|
117
118
|
install_requires=core_deps,
|
|
118
119
|
extras_require=extras_require,
|
|
119
120
|
python_requires=">=3.10,<3.14",
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Tests for scan/redact allowlist support and presidio-style entity aliases.
|
|
2
|
+
|
|
3
|
+
PII literals are assembled from split parts so write-time scanners
|
|
4
|
+
(including our own Claude Code hook) do not match this source file.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
import datafog
|
|
10
|
+
|
|
11
|
+
EMAIL = "jane.doe@" "example.com"
|
|
12
|
+
OTHER_EMAIL = "sid@" "example.com"
|
|
13
|
+
TIMESTAMP_LIKE = "17830" "25668" # ten digits: matches the PHONE pattern
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class TestExactAllowlist:
|
|
17
|
+
def test_allowlisted_value_is_not_reported(self):
|
|
18
|
+
result = datafog.scan(
|
|
19
|
+
f"mail {EMAIL} and {OTHER_EMAIL}",
|
|
20
|
+
engine="regex",
|
|
21
|
+
allowlist=[OTHER_EMAIL],
|
|
22
|
+
)
|
|
23
|
+
assert [e.text for e in result.entities] == [EMAIL]
|
|
24
|
+
|
|
25
|
+
def test_allowlist_is_exact_not_substring(self):
|
|
26
|
+
result = datafog.scan(f"mail {EMAIL}", engine="regex", allowlist=["jane.doe"])
|
|
27
|
+
assert [e.text for e in result.entities] == [EMAIL]
|
|
28
|
+
|
|
29
|
+
def test_empty_allowlist_is_noop(self):
|
|
30
|
+
result = datafog.scan(f"mail {EMAIL}", engine="regex", allowlist=[])
|
|
31
|
+
assert len(result.entities) == 1
|
|
32
|
+
|
|
33
|
+
def test_redact_respects_allowlist(self):
|
|
34
|
+
result = datafog.redact(
|
|
35
|
+
f"mail {EMAIL} and {OTHER_EMAIL}",
|
|
36
|
+
engine="regex",
|
|
37
|
+
allowlist=[OTHER_EMAIL],
|
|
38
|
+
)
|
|
39
|
+
assert OTHER_EMAIL in result.redacted_text
|
|
40
|
+
assert EMAIL not in result.redacted_text
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class TestPatternAllowlist:
|
|
44
|
+
def test_pattern_suppresses_matching_entities(self):
|
|
45
|
+
# The motivating case: unix timestamps and numeric IDs match the
|
|
46
|
+
# PHONE pattern; a pattern allowlist can exempt all-digit strings.
|
|
47
|
+
noisy = datafog.scan(f"created {TIMESTAMP_LIKE}", engine="regex")
|
|
48
|
+
assert len(noisy.entities) == 1 # sanity: it is detected by default
|
|
49
|
+
|
|
50
|
+
result = datafog.scan(
|
|
51
|
+
f"created {TIMESTAMP_LIKE}",
|
|
52
|
+
engine="regex",
|
|
53
|
+
allowlist_patterns=[r"^\d{10}$"],
|
|
54
|
+
)
|
|
55
|
+
assert result.entities == []
|
|
56
|
+
|
|
57
|
+
def test_pattern_matches_full_entity_text_only(self):
|
|
58
|
+
result = datafog.scan(
|
|
59
|
+
f"mail {EMAIL}", engine="regex", allowlist_patterns=[r"^jane\."]
|
|
60
|
+
)
|
|
61
|
+
assert len(result.entities) == 1 # partial match must not suppress
|
|
62
|
+
|
|
63
|
+
def test_invalid_pattern_raises_value_error(self):
|
|
64
|
+
with pytest.raises(ValueError, match="allowlist_patterns"):
|
|
65
|
+
datafog.scan("text", engine="regex", allowlist_patterns=["("])
|
|
66
|
+
|
|
67
|
+
def test_patterns_and_values_combine(self):
|
|
68
|
+
result = datafog.scan(
|
|
69
|
+
f"{EMAIL} then {TIMESTAMP_LIKE}",
|
|
70
|
+
engine="regex",
|
|
71
|
+
allowlist=[EMAIL],
|
|
72
|
+
allowlist_patterns=[r"^\d{10}$"],
|
|
73
|
+
)
|
|
74
|
+
assert result.entities == []
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class TestReDoSGuards:
|
|
78
|
+
def test_catastrophic_pattern_rejected(self):
|
|
79
|
+
with pytest.raises(ValueError, match="catastrophic backtracking"):
|
|
80
|
+
datafog.scan("text", engine="regex", allowlist_patterns=[r"(a+)+$"])
|
|
81
|
+
|
|
82
|
+
def test_nested_star_rejected(self):
|
|
83
|
+
with pytest.raises(ValueError, match="catastrophic backtracking"):
|
|
84
|
+
datafog.scan("text", engine="regex", allowlist_patterns=[r"(.*)*"])
|
|
85
|
+
|
|
86
|
+
def test_overlong_pattern_rejected(self):
|
|
87
|
+
with pytest.raises(ValueError, match="at most"):
|
|
88
|
+
datafog.scan("text", engine="regex", allowlist_patterns=["a" * 513])
|
|
89
|
+
|
|
90
|
+
def test_overlong_entity_text_skips_patterns_but_is_kept(self):
|
|
91
|
+
# Fail-safe: an entity too long to pattern-match safely must still
|
|
92
|
+
# be reported, never silently suppressed.
|
|
93
|
+
from datafog.engine import Entity, _apply_allowlist
|
|
94
|
+
|
|
95
|
+
long_entity = Entity(
|
|
96
|
+
type="EMAIL",
|
|
97
|
+
text="a" * 600,
|
|
98
|
+
start=0,
|
|
99
|
+
end=600,
|
|
100
|
+
confidence=1.0,
|
|
101
|
+
engine="regex",
|
|
102
|
+
)
|
|
103
|
+
kept = _apply_allowlist([long_entity], None, [r".*"])
|
|
104
|
+
assert kept == [long_entity]
|
|
105
|
+
|
|
106
|
+
def test_overlong_entity_text_still_matches_exact_allowlist(self):
|
|
107
|
+
# The subject-length cap only bounds regex matching; exact string
|
|
108
|
+
# comparison is O(n) and still applies.
|
|
109
|
+
from datafog.engine import Entity, _apply_allowlist
|
|
110
|
+
|
|
111
|
+
long_entity = Entity(
|
|
112
|
+
type="EMAIL",
|
|
113
|
+
text="a" * 600,
|
|
114
|
+
start=0,
|
|
115
|
+
end=600,
|
|
116
|
+
confidence=1.0,
|
|
117
|
+
engine="regex",
|
|
118
|
+
)
|
|
119
|
+
assert _apply_allowlist([long_entity], ["a" * 600], None) == []
|
|
120
|
+
|
|
121
|
+
def test_benign_quantified_group_still_allowed(self):
|
|
122
|
+
result = datafog.scan(
|
|
123
|
+
f"mail {EMAIL}",
|
|
124
|
+
engine="regex",
|
|
125
|
+
allowlist_patterns=[r"(abc)+", r".*@example\.com"],
|
|
126
|
+
)
|
|
127
|
+
assert result.entities == [] # broad pattern suppresses, no rejection
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class TestEnginePaths:
|
|
131
|
+
def test_smart_engine_applies_allowlist(self):
|
|
132
|
+
import warnings as _warnings
|
|
133
|
+
|
|
134
|
+
with _warnings.catch_warnings():
|
|
135
|
+
_warnings.simplefilter("ignore")
|
|
136
|
+
result = datafog.scan(f"mail {EMAIL}", engine="smart", allowlist=[EMAIL])
|
|
137
|
+
assert result.entities == []
|
|
138
|
+
|
|
139
|
+
def test_redact_rejects_allowlist_with_explicit_entities(self):
|
|
140
|
+
scanned = datafog.scan(f"mail {EMAIL}", engine="regex")
|
|
141
|
+
with pytest.raises(ValueError, match="cannot be combined"):
|
|
142
|
+
datafog.redact(
|
|
143
|
+
f"mail {EMAIL}",
|
|
144
|
+
entities=scanned.entities,
|
|
145
|
+
allowlist=[EMAIL],
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class TestPresidioAliases:
|
|
150
|
+
def test_email_address_alias(self):
|
|
151
|
+
result = datafog.scan(
|
|
152
|
+
f"mail {EMAIL}", engine="regex", entity_types=["EMAIL_ADDRESS"]
|
|
153
|
+
)
|
|
154
|
+
assert [e.type for e in result.entities] == ["EMAIL"]
|
|
155
|
+
|
|
156
|
+
def test_us_ssn_alias(self):
|
|
157
|
+
ssn = "856-45-" "6789"
|
|
158
|
+
result = datafog.scan(f"ssn {ssn}", engine="regex", entity_types=["US_SSN"])
|
|
159
|
+
assert [e.type for e in result.entities] == ["SSN"]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class TestPyTyped:
|
|
163
|
+
def test_py_typed_marker_ships_with_package(self):
|
|
164
|
+
import importlib.resources
|
|
165
|
+
|
|
166
|
+
assert importlib.resources.files("datafog").joinpath("py.typed").is_file()
|
|
@@ -72,6 +72,36 @@ class TestPreToolUse:
|
|
|
72
72
|
_, stdout = run(payload, env={"DATAFOG_HOOK_ENTITIES": "IP_ADDRESS"})
|
|
73
73
|
assert "IP_ADDRESS" in _decision(stdout)["permissionDecisionReason"]
|
|
74
74
|
|
|
75
|
+
def test_allowlist_env_exempts_exact_value(self):
|
|
76
|
+
own_email = "sid@" "example.com"
|
|
77
|
+
payload = _pre_tool_use("Bash", {"command": f"echo {own_email}"})
|
|
78
|
+
code, stdout = run(payload, env={"DATAFOG_HOOK_ALLOWLIST": own_email})
|
|
79
|
+
assert code == 0
|
|
80
|
+
assert stdout == ""
|
|
81
|
+
|
|
82
|
+
def test_allowlist_pattern_env_exempts_timestamps(self):
|
|
83
|
+
# Ten-digit numeric IDs and unix timestamps match the PHONE pattern;
|
|
84
|
+
# the pattern allowlist silences that class of false positive.
|
|
85
|
+
payload = _pre_tool_use("Bash", {"command": "echo created 17830" "25668"})
|
|
86
|
+
code, stdout = run(
|
|
87
|
+
payload, env={"DATAFOG_HOOK_ALLOWLIST_PATTERNS": r"^\d{10}$"}
|
|
88
|
+
)
|
|
89
|
+
assert code == 0
|
|
90
|
+
assert stdout == ""
|
|
91
|
+
|
|
92
|
+
def test_allowlist_does_not_exempt_other_values(self):
|
|
93
|
+
own_email = "sid@" "example.com"
|
|
94
|
+
other = "jane.doe@" "example.com"
|
|
95
|
+
payload = _pre_tool_use("Bash", {"command": f"echo {other}"})
|
|
96
|
+
_, stdout = run(payload, env={"DATAFOG_HOOK_ALLOWLIST": own_email})
|
|
97
|
+
assert "EMAIL" in _decision(stdout)["permissionDecisionReason"]
|
|
98
|
+
|
|
99
|
+
def test_invalid_allowlist_pattern_fails_open(self):
|
|
100
|
+
payload = _pre_tool_use("Bash", {"command": "echo jane.doe@" "example.com"})
|
|
101
|
+
code, stdout = run(payload, env={"DATAFOG_HOOK_ALLOWLIST_PATTERNS": "("})
|
|
102
|
+
assert code != 2 # fail-open, never blocking
|
|
103
|
+
assert stdout == ""
|
|
104
|
+
|
|
75
105
|
|
|
76
106
|
class TestUserPromptSubmit:
|
|
77
107
|
def test_pii_in_prompt_adds_context_warning(self):
|
|
@@ -262,6 +262,20 @@ class TestFailPolicy:
|
|
|
262
262
|
call_type="completion",
|
|
263
263
|
)
|
|
264
264
|
|
|
265
|
+
async def test_allowlist_exempts_configured_values(self):
|
|
266
|
+
own = "sid@" "example.com"
|
|
267
|
+
other = "jane.doe@" "example.com"
|
|
268
|
+
guardrail = DataFogGuardrail(guardrail_name="datafog-pii", allowlist=[own])
|
|
269
|
+
data = await guardrail.async_pre_call_hook(
|
|
270
|
+
user_api_key_dict=None,
|
|
271
|
+
cache=None,
|
|
272
|
+
data=_chat_data(f"contact {own} or {other}"),
|
|
273
|
+
call_type="completion",
|
|
274
|
+
)
|
|
275
|
+
content = data["messages"][0]["content"]
|
|
276
|
+
assert own in content
|
|
277
|
+
assert other not in content
|
|
278
|
+
|
|
265
279
|
async def test_invalid_config_rejected(self):
|
|
266
280
|
with pytest.raises(ValueError):
|
|
267
281
|
DataFogGuardrail(guardrail_name="datafog-pii", action="explode")
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "4.6.0"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/pytesseract_processor.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/regex_annotator/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|