datafog 4.6.0__tar.gz → 4.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. {datafog-4.6.0 → datafog-4.7.0}/PKG-INFO +37 -11
  2. {datafog-4.6.0 → datafog-4.7.0}/README.md +36 -10
  3. datafog-4.7.0/datafog/__about__.py +1 -0
  4. {datafog-4.6.0 → datafog-4.7.0}/datafog/__init__.py +28 -2
  5. {datafog-4.6.0 → datafog-4.7.0}/datafog/engine.py +90 -1
  6. {datafog-4.6.0 → datafog-4.7.0}/datafog/integrations/claude_code.py +41 -5
  7. {datafog-4.6.0 → datafog-4.7.0}/datafog/integrations/litellm_guardrail.py +35 -11
  8. datafog-4.7.0/datafog/py.typed +0 -0
  9. {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/PKG-INFO +37 -11
  10. {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/SOURCES.txt +2 -0
  11. {datafog-4.6.0 → datafog-4.7.0}/setup.py +1 -0
  12. datafog-4.7.0/tests/test_allowlist.py +166 -0
  13. {datafog-4.6.0 → datafog-4.7.0}/tests/test_claude_code_hook.py +30 -0
  14. {datafog-4.6.0 → datafog-4.7.0}/tests/test_litellm_guardrail.py +14 -0
  15. datafog-4.6.0/datafog/__about__.py +0 -1
  16. {datafog-4.6.0 → datafog-4.7.0}/LICENSE +0 -0
  17. {datafog-4.6.0 → datafog-4.7.0}/datafog/__init___lean.py +0 -0
  18. {datafog-4.6.0 → datafog-4.7.0}/datafog/__init___original.py +0 -0
  19. {datafog-4.6.0 → datafog-4.7.0}/datafog/agent.py +0 -0
  20. {datafog-4.6.0 → datafog-4.7.0}/datafog/client.py +0 -0
  21. {datafog-4.6.0 → datafog-4.7.0}/datafog/config.py +0 -0
  22. {datafog-4.6.0 → datafog-4.7.0}/datafog/core.py +0 -0
  23. {datafog-4.6.0 → datafog-4.7.0}/datafog/exceptions.py +0 -0
  24. {datafog-4.6.0 → datafog-4.7.0}/datafog/integrations/__init__.py +0 -0
  25. {datafog-4.6.0 → datafog-4.7.0}/datafog/main.py +0 -0
  26. {datafog-4.6.0 → datafog-4.7.0}/datafog/main_lean.py +0 -0
  27. {datafog-4.6.0 → datafog-4.7.0}/datafog/main_original.py +0 -0
  28. {datafog-4.6.0 → datafog-4.7.0}/datafog/models/__init__.py +0 -0
  29. {datafog-4.6.0 → datafog-4.7.0}/datafog/models/annotator.py +0 -0
  30. {datafog-4.6.0 → datafog-4.7.0}/datafog/models/anonymizer.py +0 -0
  31. {datafog-4.6.0 → datafog-4.7.0}/datafog/models/common.py +0 -0
  32. {datafog-4.6.0 → datafog-4.7.0}/datafog/models/spacy_nlp.py +0 -0
  33. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/__init__.py +0 -0
  34. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/__init__.py +0 -0
  35. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/donut_processor.py +0 -0
  36. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/image_downloader.py +0 -0
  37. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/image_processing/pytesseract_processor.py +0 -0
  38. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/spark_processing/__init__.py +0 -0
  39. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/spark_processing/pyspark_udfs.py +0 -0
  40. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/__init__.py +0 -0
  41. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/gliner_annotator.py +0 -0
  42. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/regex_annotator/__init__.py +0 -0
  43. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/regex_annotator/regex_annotator.py +0 -0
  44. {datafog-4.6.0 → datafog-4.7.0}/datafog/processing/text_processing/spacy_pii_annotator.py +0 -0
  45. {datafog-4.6.0 → datafog-4.7.0}/datafog/services/__init__.py +0 -0
  46. {datafog-4.6.0 → datafog-4.7.0}/datafog/services/image_service.py +0 -0
  47. {datafog-4.6.0 → datafog-4.7.0}/datafog/services/spark_service.py +0 -0
  48. {datafog-4.6.0 → datafog-4.7.0}/datafog/services/text_service.py +0 -0
  49. {datafog-4.6.0 → datafog-4.7.0}/datafog/services/text_service_lean.py +0 -0
  50. {datafog-4.6.0 → datafog-4.7.0}/datafog/services/text_service_original.py +0 -0
  51. {datafog-4.6.0 → datafog-4.7.0}/datafog/telemetry.py +0 -0
  52. {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/dependency_links.txt +0 -0
  53. {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/entry_points.txt +0 -0
  54. {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/requires.txt +0 -0
  55. {datafog-4.6.0 → datafog-4.7.0}/datafog.egg-info/top_level.txt +0 -0
  56. {datafog-4.6.0 → datafog-4.7.0}/setup.cfg +0 -0
  57. {datafog-4.6.0 → datafog-4.7.0}/tests/test_agent_api.py +0 -0
  58. {datafog-4.6.0 → datafog-4.7.0}/tests/test_anonymizer.py +0 -0
  59. {datafog-4.6.0 → datafog-4.7.0}/tests/test_cli_smoke.py +0 -0
  60. {datafog-4.6.0 → datafog-4.7.0}/tests/test_client.py +0 -0
  61. {datafog-4.6.0 → datafog-4.7.0}/tests/test_de_pii_regex.py +0 -0
  62. {datafog-4.6.0 → datafog-4.7.0}/tests/test_detection_accuracy.py +0 -0
  63. {datafog-4.6.0 → datafog-4.7.0}/tests/test_donut_lazy_import.py +0 -0
  64. {datafog-4.6.0 → datafog-4.7.0}/tests/test_engine_api.py +0 -0
  65. {datafog-4.6.0 → datafog-4.7.0}/tests/test_gliner_annotator.py +0 -0
  66. {datafog-4.6.0 → datafog-4.7.0}/tests/test_image_service.py +0 -0
  67. {datafog-4.6.0 → datafog-4.7.0}/tests/test_install_profiles.py +0 -0
  68. {datafog-4.6.0 → datafog-4.7.0}/tests/test_main.py +0 -0
  69. {datafog-4.6.0 → datafog-4.7.0}/tests/test_no_network_core.py +0 -0
  70. {datafog-4.6.0 → datafog-4.7.0}/tests/test_ocr_integration.py +0 -0
  71. {datafog-4.6.0 → datafog-4.7.0}/tests/test_regex_annotator.py +0 -0
  72. {datafog-4.6.0 → datafog-4.7.0}/tests/test_runtime_dependency_safety.py +0 -0
  73. {datafog-4.6.0 → datafog-4.7.0}/tests/test_spark_integration.py +0 -0
  74. {datafog-4.6.0 → datafog-4.7.0}/tests/test_telemetry.py +0 -0
  75. {datafog-4.6.0 → datafog-4.7.0}/tests/test_text_service.py +0 -0
  76. {datafog-4.6.0 → datafog-4.7.0}/tests/test_text_service_integration.py +0 -0
  77. {datafog-4.6.0 → datafog-4.7.0}/tests/test_v44_bridge_api.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: datafog
3
- Version: 4.6.0
3
+ Version: 4.7.0
4
4
  Summary: Lightning-fast PII detection and anonymization library with 190x performance advantage
5
5
  Author: Sid Mohan
6
6
  Author-email: sid@datafog.ai
@@ -112,17 +112,43 @@ DataFog is a Python library for detecting and redacting personally identifiable
112
112
  It provides:
113
113
 
114
114
  - Fast structured PII detection via regex
115
+ - An offline PII firewall for AI agents: a Claude Code hook and a LiteLLM
116
+ gateway guardrail (new in 4.6)
115
117
  - Optional NER support via spaCy and GLiNER
116
118
  - A simple agent-oriented API for LLM applications
117
119
  - Backward-compatible `DataFog` and `TextService` classes
118
120
 
119
- ## 4.5 Focus
121
+ ## Agent & Gateway Firewall (4.6)
120
122
 
121
- DataFog 4.5 is focused on lightweight text PII screening: a small core install,
122
- fast regex-based scan/redact helpers, explicit optional extras, and a clearer
123
- path toward future middleware use cases. Dedicated Sentry, OpenTelemetry,
124
- logging-framework, and cloud DLP adapters are future-facing work and are not
125
- part of the 4.5 release.
123
+ DataFog 4.6 adds two ready-made enforcement points that catch PII at the
124
+ moment it would leave your machine — offline, in microseconds, with matched
125
+ values never echoed into logs or transcripts:
126
+
127
+ - **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
128
+ commands, web requests, file writes, MCP tools) and warns the model when
129
+ prompts or tool results carry PII. ~70ms per invocation including process
130
+ startup. Easiest install is the
131
+ [Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
132
+
133
+ ```
134
+ /plugin marketplace add DataFog/datafog-claude-plugin
135
+ /plugin install datafog@datafog
136
+ ```
137
+
138
+ Manual hook setup and limitations: [examples/claude_code_hook/](examples/claude_code_hook/).
139
+
140
+ - **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
141
+ requests and responses at the gateway, for any LiteLLM-proxied provider.
142
+ In-process (~31µs per request), no sidecar service. Setup:
143
+ [examples/litellm_guardrail/](examples/litellm_guardrail/).
144
+
145
+ Both default to the high-precision entity set (`EMAIL`, `PHONE`,
146
+ `CREDIT_CARD`, `SSN`); noisier types are opt-in. Known-safe values can be
147
+ exempted with an allowlist: `scan(text, allowlist=[...])` for exact values,
148
+ `allowlist_patterns=[...]` for full-match regexes (e.g. `^\d{10}$` to stop
149
+ unix timestamps matching as phone numbers) — available in both adapters and
150
+ the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
151
+ `US_SSN`) are accepted as aliases for easy migration.
126
152
 
127
153
  ## Installation
128
154
 
@@ -149,7 +175,7 @@ pip install datafog[all]
149
175
  Python 3.13 support is certified for the core SDK, CLI, `nlp`,
150
176
  `nlp-advanced`, and `ocr` install profiles. Donut OCR still requires a model
151
177
  that is available locally before runtime use. `distributed` and `all` are not
152
- newly certified on Python 3.13 in the 4.5 line.
178
+ newly certified on Python 3.13 in the 4.x line.
153
179
 
154
180
  ## Quick Start
155
181
 
@@ -224,7 +250,7 @@ Use the engine that matches your accuracy and dependency constraints:
224
250
 
225
251
  - `regex`:
226
252
  - Fastest and always available.
227
- - Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE`.
253
+ - Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE` (`DOB` and `ZIP` are accepted as input aliases).
228
254
  - Use `locales=["de"]` for German structured IDs such as `DE_VAT_ID`, `DE_IBAN`, `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
229
255
  - `spacy`:
230
256
  - Requires `pip install datafog[nlp]`.
@@ -238,7 +264,7 @@ Use the engine that matches your accuracy and dependency constraints:
238
264
 
239
265
  ## Optional OCR And Spark Surfaces
240
266
 
241
- DataFog 4.5 keeps the main package story centered on lightweight text PII
267
+ The 4.x line keeps the main package story centered on lightweight text PII
242
268
  screening. OCR and Spark remain supported optional surfaces for users who
243
269
  already rely on them, but they are not required for the core import, default
244
270
  scan/redact helpers, or guardrail helpers.
@@ -258,7 +284,7 @@ scan/redact helpers, or guardrail helpers.
258
284
  - A Java runtime is required by PySpark.
259
285
 
260
286
  OCR and Spark are not deprecated. Their broader API and packaging overhaul is
261
- deferred; the 4.5 goal is to keep them explicit, documented, and isolated from
287
+ deferred; the 4.x goal is to keep them explicit, documented, and isolated from
262
288
  the lightweight core path.
263
289
 
264
290
  ## Backward-Compatible APIs
@@ -5,17 +5,43 @@ DataFog is a Python library for detecting and redacting personally identifiable
5
5
  It provides:
6
6
 
7
7
  - Fast structured PII detection via regex
8
+ - An offline PII firewall for AI agents: a Claude Code hook and a LiteLLM
9
+ gateway guardrail (new in 4.6)
8
10
  - Optional NER support via spaCy and GLiNER
9
11
  - A simple agent-oriented API for LLM applications
10
12
  - Backward-compatible `DataFog` and `TextService` classes
11
13
 
12
- ## 4.5 Focus
14
+ ## Agent & Gateway Firewall (4.6)
13
15
 
14
- DataFog 4.5 is focused on lightweight text PII screening: a small core install,
15
- fast regex-based scan/redact helpers, explicit optional extras, and a clearer
16
- path toward future middleware use cases. Dedicated Sentry, OpenTelemetry,
17
- logging-framework, and cloud DLP adapters are future-facing work and are not
18
- part of the 4.5 release.
16
+ DataFog 4.6 adds two ready-made enforcement points that catch PII at the
17
+ moment it would leave your machine — offline, in microseconds, with matched
18
+ values never echoed into logs or transcripts:
19
+
20
+ - **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
21
+ commands, web requests, file writes, MCP tools) and warns the model when
22
+ prompts or tool results carry PII. ~70ms per invocation including process
23
+ startup. Easiest install is the
24
+ [Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
25
+
26
+ ```
27
+ /plugin marketplace add DataFog/datafog-claude-plugin
28
+ /plugin install datafog@datafog
29
+ ```
30
+
31
+ Manual hook setup and limitations: [examples/claude_code_hook/](examples/claude_code_hook/).
32
+
33
+ - **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
34
+ requests and responses at the gateway, for any LiteLLM-proxied provider.
35
+ In-process (~31µs per request), no sidecar service. Setup:
36
+ [examples/litellm_guardrail/](examples/litellm_guardrail/).
37
+
38
+ Both default to the high-precision entity set (`EMAIL`, `PHONE`,
39
+ `CREDIT_CARD`, `SSN`); noisier types are opt-in. Known-safe values can be
40
+ exempted with an allowlist: `scan(text, allowlist=[...])` for exact values,
41
+ `allowlist_patterns=[...]` for full-match regexes (e.g. `^\d{10}$` to stop
42
+ unix timestamps matching as phone numbers) — available in both adapters and
43
+ the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
44
+ `US_SSN`) are accepted as aliases for easy migration.
19
45
 
20
46
  ## Installation
21
47
 
@@ -42,7 +68,7 @@ pip install datafog[all]
42
68
  Python 3.13 support is certified for the core SDK, CLI, `nlp`,
43
69
  `nlp-advanced`, and `ocr` install profiles. Donut OCR still requires a model
44
70
  that is available locally before runtime use. `distributed` and `all` are not
45
- newly certified on Python 3.13 in the 4.5 line.
71
+ newly certified on Python 3.13 in the 4.x line.
46
72
 
47
73
  ## Quick Start
48
74
 
@@ -117,7 +143,7 @@ Use the engine that matches your accuracy and dependency constraints:
117
143
 
118
144
  - `regex`:
119
145
  - Fastest and always available.
120
- - Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE`.
146
+ - Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE` (`DOB` and `ZIP` are accepted as input aliases).
121
147
  - Use `locales=["de"]` for German structured IDs such as `DE_VAT_ID`, `DE_IBAN`, `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
122
148
  - `spacy`:
123
149
  - Requires `pip install datafog[nlp]`.
@@ -131,7 +157,7 @@ Use the engine that matches your accuracy and dependency constraints:
131
157
 
132
158
  ## Optional OCR And Spark Surfaces
133
159
 
134
- DataFog 4.5 keeps the main package story centered on lightweight text PII
160
+ The 4.x line keeps the main package story centered on lightweight text PII
135
161
  screening. OCR and Spark remain supported optional surfaces for users who
136
162
  already rely on them, but they are not required for the core import, default
137
163
  scan/redact helpers, or guardrail helpers.
@@ -151,7 +177,7 @@ scan/redact helpers, or guardrail helpers.
151
177
  - A Java runtime is required by PySpark.
152
178
 
153
179
  OCR and Spark are not deprecated. Their broader API and packaging overhaul is
154
- deferred; the 4.5 goal is to keep them explicit, documented, and isolated from
180
+ deferred; the 4.x goal is to keep them explicit, documented, and isolated from
155
181
  the lightweight core path.
156
182
 
157
183
  ## Backward-Compatible APIs
@@ -0,0 +1 @@
1
+ __version__ = "4.7.0"
@@ -153,14 +153,28 @@ def scan(
153
153
  engine: str = "regex",
154
154
  entity_types: list[str] | None = None,
155
155
  locales: list[str] | None = None,
156
+ allowlist: list[str] | None = None,
157
+ allowlist_patterns: list[str] | None = None,
156
158
  ) -> ScanResult:
157
159
  """
158
160
  v5-preview scan entrypoint.
159
161
 
160
162
  Defaults to the lightweight regex engine so the core install works without
161
163
  optional dependency fallback warnings.
164
+
165
+ ``allowlist`` exempts exact entity texts (your own support address, doc
166
+ placeholders); ``allowlist_patterns`` exempts entities whose full text
167
+ matches a regex (e.g. ``^\\d{10}$`` so unix timestamps stop matching as
168
+ phone numbers).
162
169
  """
163
- return _scan(text=text, engine=engine, entity_types=entity_types, locales=locales)
170
+ return _scan(
171
+ text=text,
172
+ engine=engine,
173
+ entity_types=entity_types,
174
+ locales=locales,
175
+ allowlist=allowlist,
176
+ allowlist_patterns=allowlist_patterns,
177
+ )
164
178
 
165
179
 
166
180
  def redact(
@@ -171,12 +185,17 @@ def redact(
171
185
  strategy: str = "token",
172
186
  preset: str | None = None,
173
187
  locales: list[str] | None = None,
188
+ allowlist: list[str] | None = None,
189
+ allowlist_patterns: list[str] | None = None,
174
190
  ) -> RedactResult:
175
191
  """
176
192
  v5-preview redaction entrypoint.
177
193
 
178
194
  If entities are provided, redact those spans. Otherwise, scan text first
179
- using the selected engine and redact the detected entities.
195
+ using the selected engine and redact the detected entities. ``allowlist``
196
+ and ``allowlist_patterns`` exempt findings from redaction (exact text and
197
+ full-text regex match respectively); they apply to the scan path and are
198
+ rejected when explicit ``entities`` are supplied.
180
199
  """
181
200
  if preset is not None:
182
201
  try:
@@ -186,6 +205,11 @@ def redact(
186
205
  raise ValueError(f"preset must be one of: {allowed}") from exc
187
206
 
188
207
  if entities is not None:
208
+ if allowlist or allowlist_patterns:
209
+ raise ValueError(
210
+ "allowlist/allowlist_patterns cannot be combined with explicit "
211
+ "entities; filter the entities before calling redact"
212
+ )
189
213
  return _redact_entities(text=text, entities=entities, strategy=strategy)
190
214
 
191
215
  return _scan_and_redact(
@@ -194,6 +218,8 @@ def redact(
194
218
  entity_types=entity_types,
195
219
  strategy=strategy,
196
220
  locales=locales,
221
+ allowlist=allowlist,
222
+ allowlist_patterns=allowlist_patterns,
197
223
  )
198
224
 
199
225
 
@@ -3,6 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import hashlib
6
+ import re
6
7
  import warnings
7
8
  from dataclasses import dataclass
8
9
  from functools import lru_cache
@@ -23,6 +24,9 @@ CANONICAL_TYPE_MAP = {
23
24
  "SOCIAL_SECURITY_NUMBER": "SSN",
24
25
  "CREDIT_CARD_NUMBER": "CREDIT_CARD",
25
26
  "DATE_OF_BIRTH": "DATE",
27
+ # Presidio-compatible aliases, so configs migrate without renames.
28
+ "EMAIL_ADDRESS": "EMAIL",
29
+ "US_SSN": "SSN",
26
30
  }
27
31
 
28
32
  ALL_ENTITY_TYPES = {
@@ -277,6 +281,74 @@ def _filter_entity_types(
277
281
  return [entity for entity in entities if entity.type in allowed]
278
282
 
279
283
 
284
+ # Python's re module backtracks; a quantified group containing another
285
+ # quantifier (e.g. ``(a+)+``) can take exponential time on adversarial
286
+ # input, and entity text can be attacker-influenced (LLM messages, tool
287
+ # output). Reject that construct outright rather than matching under it.
288
+ _NESTED_QUANTIFIER = re.compile(
289
+ r"\((?:[^()\\]|\\.)*(?<!\\)[+*}](?:[^()\\]|\\.)*\)\s*[+*{]"
290
+ )
291
+ MAX_ALLOWLIST_PATTERN_LENGTH = 512
292
+ # Entities longer than this skip pattern matching (fail-safe: the finding
293
+ # is kept, never suppressed) so match time stays bounded.
294
+ MAX_PATTERN_SUBJECT_LENGTH = 512
295
+
296
+
297
+ def _compile_allowlist_patterns(
298
+ allowlist_patterns: Optional[list[str]],
299
+ ) -> list["re.Pattern[str]"]:
300
+ compiled = []
301
+ for raw in allowlist_patterns or []:
302
+ if len(raw) > MAX_ALLOWLIST_PATTERN_LENGTH:
303
+ raise ValueError(
304
+ "allowlist_patterns entries must be at most "
305
+ f"{MAX_ALLOWLIST_PATTERN_LENGTH} characters"
306
+ )
307
+ if _NESTED_QUANTIFIER.search(raw):
308
+ raise ValueError(
309
+ "allowlist_patterns contains a quantified group with a nested "
310
+ f"quantifier ({raw!r}), which risks catastrophic backtracking; "
311
+ "rewrite the pattern without nesting quantifiers"
312
+ )
313
+ try:
314
+ compiled.append(re.compile(raw))
315
+ except re.error as exc:
316
+ raise ValueError(
317
+ f"allowlist_patterns contains an invalid regex: {raw!r} ({exc})"
318
+ ) from None
319
+ return compiled
320
+
321
+
322
+ def _apply_allowlist(
323
+ entities: list[Entity],
324
+ allowlist: Optional[list[str]],
325
+ allowlist_patterns: Optional[list[str]],
326
+ ) -> list[Entity]:
327
+ """Drop entities whose exact text is allowlisted.
328
+
329
+ Matching semantics, deliberately strict for a security boundary:
330
+ exact values are case-sensitive with no Unicode normalization, and
331
+ patterns must fullmatch the entity text, so a partial match never
332
+ suppresses a finding. Allowlist entries and patterns are operator
333
+ configuration; treat them like code and never accept them from end
334
+ users.
335
+ """
336
+ if not allowlist and not allowlist_patterns:
337
+ return entities
338
+ exact = set(allowlist or [])
339
+ patterns = _compile_allowlist_patterns(allowlist_patterns)
340
+ return [
341
+ entity
342
+ for entity in entities
343
+ if entity.text not in exact
344
+ and not any(
345
+ pattern.fullmatch(entity.text)
346
+ for pattern in patterns
347
+ if len(entity.text) <= MAX_PATTERN_SUBJECT_LENGTH
348
+ )
349
+ ]
350
+
351
+
280
352
  def _needs_ner(entity_types: Optional[list[str]]) -> bool:
281
353
  if entity_types is None:
282
354
  return True
@@ -289,14 +361,25 @@ def scan(
289
361
  engine: str = "smart",
290
362
  entity_types: Optional[list[str]] = None,
291
363
  locales: Optional[list[str]] = None,
364
+ allowlist: Optional[list[str]] = None,
365
+ allowlist_patterns: Optional[list[str]] = None,
292
366
  ) -> ScanResult:
293
- """Scan text for PII entities."""
367
+ """Scan text for PII entities.
368
+
369
+ ``allowlist`` exempts exact entity texts (e.g. your own support email);
370
+ ``allowlist_patterns`` exempts entities whose full text matches a regex
371
+ (e.g. ``^\\d{10}$`` to stop unix timestamps matching as phone numbers).
372
+ """
294
373
  if not isinstance(text, str):
295
374
  raise TypeError("text must be a string")
296
375
 
297
376
  if engine not in {"regex", "spacy", "gliner", "smart"}:
298
377
  raise ValueError("engine must be one of: regex, spacy, gliner, smart")
299
378
 
379
+ # Validate patterns up front so config errors fail fast even when the
380
+ # text contains no entities.
381
+ _compile_allowlist_patterns(allowlist_patterns)
382
+
300
383
  regex_entities = _regex_entities(
301
384
  text,
302
385
  entity_types=entity_types,
@@ -305,6 +388,7 @@ def scan(
305
388
 
306
389
  if engine == "regex":
307
390
  filtered = _filter_entity_types(regex_entities, entity_types)
391
+ filtered = _apply_allowlist(filtered, allowlist, allowlist_patterns)
308
392
  return ScanResult(
309
393
  entities=_dedupe_entities(filtered), text=text, engine_used="regex"
310
394
  )
@@ -367,6 +451,7 @@ def scan(
367
451
  )
368
452
 
369
453
  filtered = _filter_entity_types(combined, entity_types)
454
+ filtered = _apply_allowlist(filtered, allowlist, allowlist_patterns)
370
455
  deduped = _dedupe_entities(filtered)
371
456
  return ScanResult(
372
457
  entities=deduped,
@@ -437,6 +522,8 @@ def scan_and_redact(
437
522
  entity_types: Optional[list[str]] = None,
438
523
  strategy: str = "token",
439
524
  locales: Optional[list[str]] = None,
525
+ allowlist: Optional[list[str]] = None,
526
+ allowlist_patterns: Optional[list[str]] = None,
440
527
  ) -> RedactResult:
441
528
  """Convenience wrapper: scan then redact."""
442
529
  scan_result = scan(
@@ -444,5 +531,7 @@ def scan_and_redact(
444
531
  engine=engine,
445
532
  entity_types=entity_types,
446
533
  locales=locales,
534
+ allowlist=allowlist,
535
+ allowlist_patterns=allowlist_patterns,
447
536
  )
448
537
  return redact(text=text, entities=scan_result.entities, strategy=strategy)
@@ -16,6 +16,11 @@ Configuration (environment variables):
16
16
  - ``DATAFOG_HOOK_ENTITIES``: comma-separated entity types to detect.
17
17
  Defaults to the high-precision set; noisy-in-code types (IP_ADDRESS,
18
18
  DOB, ZIP) must be opted into.
19
+ - ``DATAFOG_HOOK_ALLOWLIST``: comma-separated exact values to exempt
20
+ (your own support address, documentation placeholders).
21
+ - ``DATAFOG_HOOK_ALLOWLIST_PATTERNS``: comma-separated regexes; findings
22
+ whose full text matches are exempt (note: a pattern containing a comma
23
+ cannot be expressed here).
19
24
 
20
25
  Failure policy: fail open. A hook bug must never brick a Claude Code
21
26
  session, so any unexpected error exits non-blocking with no output.
@@ -59,6 +64,11 @@ def _action(env: Mapping[str, str]) -> str:
59
64
  return action if action in VALID_ACTIONS else "ask"
60
65
 
61
66
 
67
+ def _csv_env(env: Mapping[str, str], name: str) -> list[str]:
68
+ raw = env.get(name, "")
69
+ return [item.strip() for item in raw.split(",") if item.strip()]
70
+
71
+
62
72
  def _iter_strings(value: Any) -> Iterator[str]:
63
73
  """Yield every string embedded in a JSON-like structure.
64
74
 
@@ -76,7 +86,12 @@ def _iter_strings(value: Any) -> Iterator[str]:
76
86
  stack.extend(current)
77
87
 
78
88
 
79
- def _scan_findings(value: Any, entity_types: list[str]) -> dict[str, int]:
89
+ def _scan_findings(
90
+ value: Any,
91
+ entity_types: list[str],
92
+ allowlist: list[str] | None = None,
93
+ allowlist_patterns: list[str] | None = None,
94
+ ) -> dict[str, int]:
80
95
  """Scan all strings in ``value``; return counts per entity type."""
81
96
  import datafog
82
97
 
@@ -87,7 +102,13 @@ def _scan_findings(value: Any, entity_types: list[str]) -> dict[str, int]:
87
102
  break
88
103
  chunk = text[: min(MAX_SCAN_CHARS, total_budget)]
89
104
  total_budget -= len(chunk)
90
- result = datafog.scan(chunk, engine="regex", entity_types=entity_types)
105
+ result = datafog.scan(
106
+ chunk,
107
+ engine="regex",
108
+ entity_types=entity_types,
109
+ allowlist=allowlist or None,
110
+ allowlist_patterns=allowlist_patterns or None,
111
+ )
91
112
  for entity in result.entities:
92
113
  counts[entity.type] = counts.get(entity.type, 0) + 1
93
114
  return counts
@@ -104,7 +125,12 @@ def _emit(event: str, fields: dict[str, Any]) -> str:
104
125
 
105
126
 
106
127
  def _handle_pre_tool_use(payload: dict, env: Mapping[str, str]) -> str:
107
- counts = _scan_findings(payload.get("tool_input"), _entity_types(env))
128
+ counts = _scan_findings(
129
+ payload.get("tool_input"),
130
+ _entity_types(env),
131
+ allowlist=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST"),
132
+ allowlist_patterns=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST_PATTERNS"),
133
+ )
108
134
  if not counts:
109
135
  return ""
110
136
  tool = payload.get("tool_name", "tool")
@@ -119,7 +145,12 @@ def _handle_pre_tool_use(payload: dict, env: Mapping[str, str]) -> str:
119
145
 
120
146
 
121
147
  def _handle_user_prompt_submit(payload: dict, env: Mapping[str, str]) -> str:
122
- counts = _scan_findings(payload.get("prompt"), _entity_types(env))
148
+ counts = _scan_findings(
149
+ payload.get("prompt"),
150
+ _entity_types(env),
151
+ allowlist=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST"),
152
+ allowlist_patterns=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST_PATTERNS"),
153
+ )
123
154
  if not counts:
124
155
  return ""
125
156
  context = (
@@ -130,7 +161,12 @@ def _handle_user_prompt_submit(payload: dict, env: Mapping[str, str]) -> str:
130
161
 
131
162
 
132
163
  def _handle_post_tool_use(payload: dict, env: Mapping[str, str]) -> str:
133
- counts = _scan_findings(payload.get("tool_response"), _entity_types(env))
164
+ counts = _scan_findings(
165
+ payload.get("tool_response"),
166
+ _entity_types(env),
167
+ allowlist=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST"),
168
+ allowlist_patterns=_csv_env(env, "DATAFOG_HOOK_ALLOWLIST_PATTERNS"),
169
+ )
134
170
  if not counts:
135
171
  return ""
136
172
  tool = payload.get("tool_name", "tool")
@@ -46,11 +46,22 @@ VALID_FAIL_POLICIES = {"open", "closed"}
46
46
  logger = logging.getLogger(__name__)
47
47
 
48
48
 
49
- def _redact_text(text: str, entity_types: list[str]) -> tuple[str, dict[str, int]]:
49
+ def _redact_text(
50
+ text: str,
51
+ entity_types: list[str],
52
+ allowlist: list[str] | None = None,
53
+ allowlist_patterns: list[str] | None = None,
54
+ ) -> tuple[str, dict[str, int]]:
50
55
  """Redact ``text``; return (redacted_text, counts per entity type)."""
51
56
  import datafog
52
57
 
53
- result = datafog.redact(text, engine="regex", entity_types=entity_types)
58
+ result = datafog.redact(
59
+ text,
60
+ engine="regex",
61
+ entity_types=entity_types,
62
+ allowlist=allowlist,
63
+ allowlist_patterns=allowlist_patterns,
64
+ )
54
65
  counts: dict[str, int] = {}
55
66
  for entity in result.entities:
56
67
  counts[entity.type] = counts.get(entity.type, 0) + 1
@@ -69,6 +80,8 @@ class DataFogGuardrail(CustomGuardrail):
69
80
  action: str = "redact",
70
81
  entity_types: Optional[list[str]] = None,
71
82
  fail_policy: str = "open",
83
+ allowlist: Optional[list[str]] = None,
84
+ allowlist_patterns: Optional[list[str]] = None,
72
85
  **kwargs: Any,
73
86
  ) -> None:
74
87
  if action not in VALID_ACTIONS:
@@ -80,13 +93,17 @@ class DataFogGuardrail(CustomGuardrail):
80
93
  self.action = action
81
94
  self.entity_types = entity_types or DEFAULT_ENTITY_TYPES
82
95
  self.fail_policy = fail_policy
96
+ self.allowlist = allowlist
97
+ self.allowlist_patterns = allowlist_patterns
83
98
  super().__init__(**kwargs)
84
99
 
85
100
  def _process_content(self, content: Any) -> tuple[Any, dict[str, int]]:
86
101
  """Redact a message content value (str or list of content parts)."""
87
102
  counts: dict[str, int] = {}
88
103
  if isinstance(content, str):
89
- redacted, counts = _redact_text(content, self.entity_types)
104
+ redacted, counts = _redact_text(
105
+ content, self.entity_types, self.allowlist, self.allowlist_patterns
106
+ )
90
107
  return redacted, counts
91
108
  if isinstance(content, list):
92
109
  new_parts = []
@@ -94,7 +111,10 @@ class DataFogGuardrail(CustomGuardrail):
94
111
  for part in content:
95
112
  if isinstance(part, dict) and isinstance(part.get("text"), str):
96
113
  redacted, part_counts = _redact_text(
97
- part["text"], self.entity_types
114
+ part["text"],
115
+ self.entity_types,
116
+ self.allowlist,
117
+ self.allowlist_patterns,
98
118
  )
99
119
  new_parts.append({**part, "text": redacted})
100
120
  for etype, n in part_counts.items():
@@ -160,9 +180,8 @@ class DataFogGuardrail(CustomGuardrail):
160
180
  if not total_counts:
161
181
  return data
162
182
 
163
- self._record_guardrail_logging(data, total_counts)
164
-
165
183
  if self.action == "block":
184
+ self._record_guardrail_logging(data, total_counts)
166
185
  # HTTPException(400) is one of the exception types litellm's
167
186
  # _is_guardrail_intervention recognizes, so the block is
168
187
  # classified as a policy intervention (not a backend failure)
@@ -178,7 +197,9 @@ class DataFogGuardrail(CustomGuardrail):
178
197
  },
179
198
  )
180
199
 
181
- return {**data, "messages": new_messages}
200
+ new_data = {**data, "messages": new_messages}
201
+ self._record_guardrail_logging(new_data, total_counts)
202
+ return new_data
182
203
 
183
204
  def _record_guardrail_logging(
184
205
  self, data: dict, total_counts: dict[str, int]
@@ -188,9 +209,7 @@ class DataFogGuardrail(CustomGuardrail):
188
209
  self.add_standard_logging_guardrail_information_to_request_data(
189
210
  guardrail_json_response=_summary(total_counts),
190
211
  request_data=data,
191
- guardrail_status=(
192
- "guardrail_intervened" if self.action == "block" else "success"
193
- ),
212
+ guardrail_status="guardrail_intervened",
194
213
  masked_entity_count=dict(total_counts),
195
214
  )
196
215
  except Exception: # noqa: BLE001 — observability must never break traffic
@@ -217,7 +236,12 @@ class DataFogGuardrail(CustomGuardrail):
217
236
  for choice in choices:
218
237
  message = getattr(choice, "message", None)
219
238
  if message is not None and isinstance(message.content, str):
220
- redacted, counts = _redact_text(message.content, self.entity_types)
239
+ redacted, counts = _redact_text(
240
+ message.content,
241
+ self.entity_types,
242
+ self.allowlist,
243
+ self.allowlist_patterns,
244
+ )
221
245
  if counts:
222
246
  message.content = redacted
223
247
  elif message is not None and message.content is not None:
File without changes
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: datafog
3
- Version: 4.6.0
3
+ Version: 4.7.0
4
4
  Summary: Lightning-fast PII detection and anonymization library with 190x performance advantage
5
5
  Author: Sid Mohan
6
6
  Author-email: sid@datafog.ai
@@ -112,17 +112,43 @@ DataFog is a Python library for detecting and redacting personally identifiable
112
112
  It provides:
113
113
 
114
114
  - Fast structured PII detection via regex
115
+ - An offline PII firewall for AI agents: a Claude Code hook and a LiteLLM
116
+ gateway guardrail (new in 4.6)
115
117
  - Optional NER support via spaCy and GLiNER
116
118
  - A simple agent-oriented API for LLM applications
117
119
  - Backward-compatible `DataFog` and `TextService` classes
118
120
 
119
- ## 4.5 Focus
121
+ ## Agent & Gateway Firewall (4.6)
120
122
 
121
- DataFog 4.5 is focused on lightweight text PII screening: a small core install,
122
- fast regex-based scan/redact helpers, explicit optional extras, and a clearer
123
- path toward future middleware use cases. Dedicated Sentry, OpenTelemetry,
124
- logging-framework, and cloud DLP adapters are future-facing work and are not
125
- part of the 4.5 release.
123
+ DataFog 4.6 adds two ready-made enforcement points that catch PII at the
124
+ moment it would leave your machine — offline, in microseconds, with matched
125
+ values never echoed into logs or transcripts:
126
+
127
+ - **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
128
+ commands, web requests, file writes, MCP tools) and warns the model when
129
+ prompts or tool results carry PII. ~70ms per invocation including process
130
+ startup. Easiest install is the
131
+ [Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
132
+
133
+ ```
134
+ /plugin marketplace add DataFog/datafog-claude-plugin
135
+ /plugin install datafog@datafog
136
+ ```
137
+
138
+ Manual hook setup and limitations: [examples/claude_code_hook/](examples/claude_code_hook/).
139
+
140
+ - **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
141
+ requests and responses at the gateway, for any LiteLLM-proxied provider.
142
+ In-process (~31µs per request), no sidecar service. Setup:
143
+ [examples/litellm_guardrail/](examples/litellm_guardrail/).
144
+
145
+ Both default to the high-precision entity set (`EMAIL`, `PHONE`,
146
+ `CREDIT_CARD`, `SSN`); noisier types are opt-in. Known-safe values can be
147
+ exempted with an allowlist: `scan(text, allowlist=[...])` for exact values,
148
+ `allowlist_patterns=[...]` for full-match regexes (e.g. `^\d{10}$` to stop
149
+ unix timestamps matching as phone numbers) — available in both adapters and
150
+ the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
151
+ `US_SSN`) are accepted as aliases for easy migration.
126
152
 
127
153
  ## Installation
128
154
 
@@ -149,7 +175,7 @@ pip install datafog[all]
149
175
  Python 3.13 support is certified for the core SDK, CLI, `nlp`,
150
176
  `nlp-advanced`, and `ocr` install profiles. Donut OCR still requires a model
151
177
  that is available locally before runtime use. `distributed` and `all` are not
152
- newly certified on Python 3.13 in the 4.5 line.
178
+ newly certified on Python 3.13 in the 4.x line.
153
179
 
154
180
  ## Quick Start
155
181
 
@@ -224,7 +250,7 @@ Use the engine that matches your accuracy and dependency constraints:
224
250
 
225
251
  - `regex`:
226
252
  - Fastest and always available.
227
- - Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE`.
253
+ - Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE` (`DOB` and `ZIP` are accepted as input aliases).
228
254
  - Use `locales=["de"]` for German structured IDs such as `DE_VAT_ID`, `DE_IBAN`, `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
229
255
  - `spacy`:
230
256
  - Requires `pip install datafog[nlp]`.
@@ -238,7 +264,7 @@ Use the engine that matches your accuracy and dependency constraints:
238
264
 
239
265
  ## Optional OCR And Spark Surfaces
240
266
 
241
- DataFog 4.5 keeps the main package story centered on lightweight text PII
267
+ The 4.x line keeps the main package story centered on lightweight text PII
242
268
  screening. OCR and Spark remain supported optional surfaces for users who
243
269
  already rely on them, but they are not required for the core import, default
244
270
  scan/redact helpers, or guardrail helpers.
@@ -258,7 +284,7 @@ scan/redact helpers, or guardrail helpers.
258
284
  - A Java runtime is required by PySpark.
259
285
 
260
286
  OCR and Spark are not deprecated. Their broader API and packaging overhaul is
261
- deferred; the 4.5 goal is to keep them explicit, documented, and isolated from
287
+ deferred; the 4.x goal is to keep them explicit, documented, and isolated from
262
288
  the lightweight core path.
263
289
 
264
290
  ## Backward-Compatible APIs
@@ -14,6 +14,7 @@ datafog/exceptions.py
14
14
  datafog/main.py
15
15
  datafog/main_lean.py
16
16
  datafog/main_original.py
17
+ datafog/py.typed
17
18
  datafog/telemetry.py
18
19
  datafog.egg-info/PKG-INFO
19
20
  datafog.egg-info/SOURCES.txt
@@ -48,6 +49,7 @@ datafog/services/text_service.py
48
49
  datafog/services/text_service_lean.py
49
50
  datafog/services/text_service_original.py
50
51
  tests/test_agent_api.py
52
+ tests/test_allowlist.py
51
53
  tests/test_anonymizer.py
52
54
  tests/test_claude_code_hook.py
53
55
  tests/test_cli_smoke.py
@@ -114,6 +114,7 @@ setup(
114
114
  long_description=long_description,
115
115
  long_description_content_type="text/markdown",
116
116
  packages=find_packages(exclude=["tests", "tests.*"]),
117
+ package_data={"datafog": ["py.typed"]},
117
118
  install_requires=core_deps,
118
119
  extras_require=extras_require,
119
120
  python_requires=">=3.10,<3.14",
@@ -0,0 +1,166 @@
1
+ """Tests for scan/redact allowlist support and presidio-style entity aliases.
2
+
3
+ PII literals are assembled from split parts so write-time scanners
4
+ (including our own Claude Code hook) do not match this source file.
5
+ """
6
+
7
+ import pytest
8
+
9
+ import datafog
10
+
11
+ EMAIL = "jane.doe@" "example.com"
12
+ OTHER_EMAIL = "sid@" "example.com"
13
+ TIMESTAMP_LIKE = "17830" "25668" # ten digits: matches the PHONE pattern
14
+
15
+
16
+ class TestExactAllowlist:
17
+ def test_allowlisted_value_is_not_reported(self):
18
+ result = datafog.scan(
19
+ f"mail {EMAIL} and {OTHER_EMAIL}",
20
+ engine="regex",
21
+ allowlist=[OTHER_EMAIL],
22
+ )
23
+ assert [e.text for e in result.entities] == [EMAIL]
24
+
25
+ def test_allowlist_is_exact_not_substring(self):
26
+ result = datafog.scan(f"mail {EMAIL}", engine="regex", allowlist=["jane.doe"])
27
+ assert [e.text for e in result.entities] == [EMAIL]
28
+
29
+ def test_empty_allowlist_is_noop(self):
30
+ result = datafog.scan(f"mail {EMAIL}", engine="regex", allowlist=[])
31
+ assert len(result.entities) == 1
32
+
33
+ def test_redact_respects_allowlist(self):
34
+ result = datafog.redact(
35
+ f"mail {EMAIL} and {OTHER_EMAIL}",
36
+ engine="regex",
37
+ allowlist=[OTHER_EMAIL],
38
+ )
39
+ assert OTHER_EMAIL in result.redacted_text
40
+ assert EMAIL not in result.redacted_text
41
+
42
+
43
+ class TestPatternAllowlist:
44
+ def test_pattern_suppresses_matching_entities(self):
45
+ # The motivating case: unix timestamps and numeric IDs match the
46
+ # PHONE pattern; a pattern allowlist can exempt all-digit strings.
47
+ noisy = datafog.scan(f"created {TIMESTAMP_LIKE}", engine="regex")
48
+ assert len(noisy.entities) == 1 # sanity: it is detected by default
49
+
50
+ result = datafog.scan(
51
+ f"created {TIMESTAMP_LIKE}",
52
+ engine="regex",
53
+ allowlist_patterns=[r"^\d{10}$"],
54
+ )
55
+ assert result.entities == []
56
+
57
+ def test_pattern_matches_full_entity_text_only(self):
58
+ result = datafog.scan(
59
+ f"mail {EMAIL}", engine="regex", allowlist_patterns=[r"^jane\."]
60
+ )
61
+ assert len(result.entities) == 1 # partial match must not suppress
62
+
63
+ def test_invalid_pattern_raises_value_error(self):
64
+ with pytest.raises(ValueError, match="allowlist_patterns"):
65
+ datafog.scan("text", engine="regex", allowlist_patterns=["("])
66
+
67
+ def test_patterns_and_values_combine(self):
68
+ result = datafog.scan(
69
+ f"{EMAIL} then {TIMESTAMP_LIKE}",
70
+ engine="regex",
71
+ allowlist=[EMAIL],
72
+ allowlist_patterns=[r"^\d{10}$"],
73
+ )
74
+ assert result.entities == []
75
+
76
+
77
+ class TestReDoSGuards:
78
+ def test_catastrophic_pattern_rejected(self):
79
+ with pytest.raises(ValueError, match="catastrophic backtracking"):
80
+ datafog.scan("text", engine="regex", allowlist_patterns=[r"(a+)+$"])
81
+
82
+ def test_nested_star_rejected(self):
83
+ with pytest.raises(ValueError, match="catastrophic backtracking"):
84
+ datafog.scan("text", engine="regex", allowlist_patterns=[r"(.*)*"])
85
+
86
+ def test_overlong_pattern_rejected(self):
87
+ with pytest.raises(ValueError, match="at most"):
88
+ datafog.scan("text", engine="regex", allowlist_patterns=["a" * 513])
89
+
90
+ def test_overlong_entity_text_skips_patterns_but_is_kept(self):
91
+ # Fail-safe: an entity too long to pattern-match safely must still
92
+ # be reported, never silently suppressed.
93
+ from datafog.engine import Entity, _apply_allowlist
94
+
95
+ long_entity = Entity(
96
+ type="EMAIL",
97
+ text="a" * 600,
98
+ start=0,
99
+ end=600,
100
+ confidence=1.0,
101
+ engine="regex",
102
+ )
103
+ kept = _apply_allowlist([long_entity], None, [r".*"])
104
+ assert kept == [long_entity]
105
+
106
+ def test_overlong_entity_text_still_matches_exact_allowlist(self):
107
+ # The subject-length cap only bounds regex matching; exact string
108
+ # comparison is O(n) and still applies.
109
+ from datafog.engine import Entity, _apply_allowlist
110
+
111
+ long_entity = Entity(
112
+ type="EMAIL",
113
+ text="a" * 600,
114
+ start=0,
115
+ end=600,
116
+ confidence=1.0,
117
+ engine="regex",
118
+ )
119
+ assert _apply_allowlist([long_entity], ["a" * 600], None) == []
120
+
121
+ def test_benign_quantified_group_still_allowed(self):
122
+ result = datafog.scan(
123
+ f"mail {EMAIL}",
124
+ engine="regex",
125
+ allowlist_patterns=[r"(abc)+", r".*@example\.com"],
126
+ )
127
+ assert result.entities == [] # broad pattern suppresses, no rejection
128
+
129
+
130
+ class TestEnginePaths:
131
+ def test_smart_engine_applies_allowlist(self):
132
+ import warnings as _warnings
133
+
134
+ with _warnings.catch_warnings():
135
+ _warnings.simplefilter("ignore")
136
+ result = datafog.scan(f"mail {EMAIL}", engine="smart", allowlist=[EMAIL])
137
+ assert result.entities == []
138
+
139
+ def test_redact_rejects_allowlist_with_explicit_entities(self):
140
+ scanned = datafog.scan(f"mail {EMAIL}", engine="regex")
141
+ with pytest.raises(ValueError, match="cannot be combined"):
142
+ datafog.redact(
143
+ f"mail {EMAIL}",
144
+ entities=scanned.entities,
145
+ allowlist=[EMAIL],
146
+ )
147
+
148
+
149
+ class TestPresidioAliases:
150
+ def test_email_address_alias(self):
151
+ result = datafog.scan(
152
+ f"mail {EMAIL}", engine="regex", entity_types=["EMAIL_ADDRESS"]
153
+ )
154
+ assert [e.type for e in result.entities] == ["EMAIL"]
155
+
156
+ def test_us_ssn_alias(self):
157
+ ssn = "856-45-" "6789"
158
+ result = datafog.scan(f"ssn {ssn}", engine="regex", entity_types=["US_SSN"])
159
+ assert [e.type for e in result.entities] == ["SSN"]
160
+
161
+
162
+ class TestPyTyped:
163
+ def test_py_typed_marker_ships_with_package(self):
164
+ import importlib.resources
165
+
166
+ assert importlib.resources.files("datafog").joinpath("py.typed").is_file()
@@ -72,6 +72,36 @@ class TestPreToolUse:
72
72
  _, stdout = run(payload, env={"DATAFOG_HOOK_ENTITIES": "IP_ADDRESS"})
73
73
  assert "IP_ADDRESS" in _decision(stdout)["permissionDecisionReason"]
74
74
 
75
+ def test_allowlist_env_exempts_exact_value(self):
76
+ own_email = "sid@" "example.com"
77
+ payload = _pre_tool_use("Bash", {"command": f"echo {own_email}"})
78
+ code, stdout = run(payload, env={"DATAFOG_HOOK_ALLOWLIST": own_email})
79
+ assert code == 0
80
+ assert stdout == ""
81
+
82
+ def test_allowlist_pattern_env_exempts_timestamps(self):
83
+ # Ten-digit numeric IDs and unix timestamps match the PHONE pattern;
84
+ # the pattern allowlist silences that class of false positive.
85
+ payload = _pre_tool_use("Bash", {"command": "echo created 17830" "25668"})
86
+ code, stdout = run(
87
+ payload, env={"DATAFOG_HOOK_ALLOWLIST_PATTERNS": r"^\d{10}$"}
88
+ )
89
+ assert code == 0
90
+ assert stdout == ""
91
+
92
+ def test_allowlist_does_not_exempt_other_values(self):
93
+ own_email = "sid@" "example.com"
94
+ other = "jane.doe@" "example.com"
95
+ payload = _pre_tool_use("Bash", {"command": f"echo {other}"})
96
+ _, stdout = run(payload, env={"DATAFOG_HOOK_ALLOWLIST": own_email})
97
+ assert "EMAIL" in _decision(stdout)["permissionDecisionReason"]
98
+
99
+ def test_invalid_allowlist_pattern_fails_open(self):
100
+ payload = _pre_tool_use("Bash", {"command": "echo jane.doe@" "example.com"})
101
+ code, stdout = run(payload, env={"DATAFOG_HOOK_ALLOWLIST_PATTERNS": "("})
102
+ assert code != 2 # fail-open, never blocking
103
+ assert stdout == ""
104
+
75
105
 
76
106
  class TestUserPromptSubmit:
77
107
  def test_pii_in_prompt_adds_context_warning(self):
@@ -262,6 +262,20 @@ class TestFailPolicy:
262
262
  call_type="completion",
263
263
  )
264
264
 
265
+ async def test_allowlist_exempts_configured_values(self):
266
+ own = "sid@" "example.com"
267
+ other = "jane.doe@" "example.com"
268
+ guardrail = DataFogGuardrail(guardrail_name="datafog-pii", allowlist=[own])
269
+ data = await guardrail.async_pre_call_hook(
270
+ user_api_key_dict=None,
271
+ cache=None,
272
+ data=_chat_data(f"contact {own} or {other}"),
273
+ call_type="completion",
274
+ )
275
+ content = data["messages"][0]["content"]
276
+ assert own in content
277
+ assert other not in content
278
+
265
279
  async def test_invalid_config_rejected(self):
266
280
  with pytest.raises(ValueError):
267
281
  DataFogGuardrail(guardrail_name="datafog-pii", action="explode")
@@ -1 +0,0 @@
1
- __version__ = "4.6.0"
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes