datafog 4.7.0__tar.gz → 4.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {datafog-4.7.0 → datafog-4.8.0}/PKG-INFO +10 -5
- {datafog-4.7.0 → datafog-4.8.0}/README.md +8 -3
- datafog-4.8.0/datafog/__about__.py +1 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/__init__.py +2 -1
- {datafog-4.7.0 → datafog-4.8.0}/datafog/__init___lean.py +2 -1
- {datafog-4.7.0 → datafog-4.8.0}/datafog/engine.py +29 -6
- {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/PKG-INFO +10 -5
- {datafog-4.7.0 → datafog-4.8.0}/setup.py +1 -1
- datafog-4.8.0/tests/test_engine_api.py +268 -0
- datafog-4.7.0/datafog/__about__.py +0 -1
- datafog-4.7.0/tests/test_engine_api.py +0 -131
- {datafog-4.7.0 → datafog-4.8.0}/LICENSE +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/__init___original.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/agent.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/client.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/config.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/core.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/exceptions.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/integrations/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/integrations/claude_code.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/integrations/litellm_guardrail.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/main.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/main_lean.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/main_original.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/models/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/models/annotator.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/models/anonymizer.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/models/common.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/models/spacy_nlp.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/donut_processor.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/image_downloader.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/pytesseract_processor.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/spark_processing/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/spark_processing/pyspark_udfs.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/gliner_annotator.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/regex_annotator/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/regex_annotator/regex_annotator.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/spacy_pii_annotator.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/py.typed +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/services/__init__.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/services/image_service.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/services/spark_service.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/services/text_service.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/services/text_service_lean.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/services/text_service_original.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog/telemetry.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/SOURCES.txt +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/dependency_links.txt +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/entry_points.txt +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/requires.txt +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/top_level.txt +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/setup.cfg +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_agent_api.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_allowlist.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_anonymizer.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_claude_code_hook.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_cli_smoke.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_client.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_de_pii_regex.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_detection_accuracy.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_donut_lazy_import.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_gliner_annotator.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_image_service.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_install_profiles.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_litellm_guardrail.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_main.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_no_network_core.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_ocr_integration.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_regex_annotator.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_runtime_dependency_safety.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_spark_integration.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_telemetry.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_text_service.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_text_service_integration.py +0 -0
- {datafog-4.7.0 → datafog-4.8.0}/tests/test_v44_bridge_api.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: datafog
|
|
3
|
-
Version: 4.
|
|
4
|
-
Summary: Lightning-fast PII detection and anonymization library
|
|
3
|
+
Version: 4.8.0
|
|
4
|
+
Summary: Lightning-fast PII detection and anonymization library, 100x+ faster than NER-based detection (see benchmarks/)
|
|
5
5
|
Author: Sid Mohan
|
|
6
6
|
Author-email: sid@datafog.ai
|
|
7
7
|
Project-URL: Homepage, https://datafog.ai
|
|
@@ -126,8 +126,8 @@ values never echoed into logs or transcripts:
|
|
|
126
126
|
|
|
127
127
|
- **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
|
|
128
128
|
commands, web requests, file writes, MCP tools) and warns the model when
|
|
129
|
-
prompts or tool results carry PII. ~
|
|
130
|
-
startup. Easiest install is the
|
|
129
|
+
prompts or tool results carry PII. ~70–90ms per invocation including
|
|
130
|
+
process startup. Easiest install is the
|
|
131
131
|
[Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
|
|
132
132
|
|
|
133
133
|
```
|
|
@@ -139,7 +139,8 @@ values never echoed into logs or transcripts:
|
|
|
139
139
|
|
|
140
140
|
- **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
|
|
141
141
|
requests and responses at the gateway, for any LiteLLM-proxied provider.
|
|
142
|
-
In-process (~
|
|
142
|
+
In-process (~40µs per message scanned; a request clears the guardrail in
|
|
143
|
+
well under a millisecond), no sidecar service. Setup:
|
|
143
144
|
[examples/litellm_guardrail/](examples/litellm_guardrail/).
|
|
144
145
|
|
|
145
146
|
Both default to the high-precision entity set (`EMAIL`, `PHONE`,
|
|
@@ -150,6 +151,10 @@ unix timestamps matching as phone numbers) — available in both adapters and
|
|
|
150
151
|
the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
|
|
151
152
|
`US_SSN`) are accepted as aliases for easy migration.
|
|
152
153
|
|
|
154
|
+
Every performance number above is reproducible with one command —
|
|
155
|
+
methodology, pinned payloads, and comparisons against Presidio and spaCy
|
|
156
|
+
NER live in [benchmarks/](benchmarks/).
|
|
157
|
+
|
|
153
158
|
## Installation
|
|
154
159
|
|
|
155
160
|
```bash
|
|
@@ -19,8 +19,8 @@ values never echoed into logs or transcripts:
|
|
|
19
19
|
|
|
20
20
|
- **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
|
|
21
21
|
commands, web requests, file writes, MCP tools) and warns the model when
|
|
22
|
-
prompts or tool results carry PII. ~
|
|
23
|
-
startup. Easiest install is the
|
|
22
|
+
prompts or tool results carry PII. ~70–90ms per invocation including
|
|
23
|
+
process startup. Easiest install is the
|
|
24
24
|
[Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
|
|
25
25
|
|
|
26
26
|
```
|
|
@@ -32,7 +32,8 @@ values never echoed into logs or transcripts:
|
|
|
32
32
|
|
|
33
33
|
- **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
|
|
34
34
|
requests and responses at the gateway, for any LiteLLM-proxied provider.
|
|
35
|
-
In-process (~
|
|
35
|
+
In-process (~40µs per message scanned; a request clears the guardrail in
|
|
36
|
+
well under a millisecond), no sidecar service. Setup:
|
|
36
37
|
[examples/litellm_guardrail/](examples/litellm_guardrail/).
|
|
37
38
|
|
|
38
39
|
Both default to the high-precision entity set (`EMAIL`, `PHONE`,
|
|
@@ -43,6 +44,10 @@ unix timestamps matching as phone numbers) — available in both adapters and
|
|
|
43
44
|
the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
|
|
44
45
|
`US_SSN`) are accepted as aliases for easy migration.
|
|
45
46
|
|
|
47
|
+
Every performance number above is reproducible with one command —
|
|
48
|
+
methodology, pinned payloads, and comparisons against Presidio and spaCy
|
|
49
|
+
NER live in [benchmarks/](benchmarks/).
|
|
50
|
+
|
|
46
51
|
## Installation
|
|
47
52
|
|
|
48
53
|
```bash
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "4.8.0"
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
"""
|
|
2
2
|
DataFog: Lightning-fast PII detection and anonymization library.
|
|
3
3
|
|
|
4
|
-
Core package provides regex-based PII detection
|
|
4
|
+
Core package provides regex-based PII detection, 100x+ faster than NER-based
|
|
5
|
+
detection on identical payloads (reproduce: python benchmarks/run.py).
|
|
5
6
|
Optional extras available for advanced features:
|
|
6
7
|
- pip install datafog[nlp] - for spaCy integration
|
|
7
8
|
- pip install datafog[ocr] - for image/OCR processing
|
|
@@ -7,7 +7,8 @@ only as historical reference until legacy cleanup can remove it safely.
|
|
|
7
7
|
|
|
8
8
|
DataFog: Lightning-fast PII detection and anonymization library.
|
|
9
9
|
|
|
10
|
-
Core package provides regex-based PII detection
|
|
10
|
+
Core package provides regex-based PII detection, 100x+ faster than NER-based
|
|
11
|
+
detection on identical payloads (reproduce: python benchmarks/run.py).
|
|
11
12
|
Optional extras available for advanced features:
|
|
12
13
|
- pip install datafog[nlp] - for spaCy integration
|
|
13
14
|
- pip install datafog[ocr] - for image/OCR processing
|
|
@@ -173,6 +173,7 @@ def _suppress_overlapping_entities(entities: list[Entity]) -> list[Entity]:
|
|
|
173
173
|
key=lambda item: (
|
|
174
174
|
-_entity_length(item),
|
|
175
175
|
-ENTITY_TYPE_PRIORITY.get(item.type, 0),
|
|
176
|
+
-item.confidence,
|
|
176
177
|
item.start,
|
|
177
178
|
item.end,
|
|
178
179
|
item.type,
|
|
@@ -465,7 +466,18 @@ def redact(
|
|
|
465
466
|
entities: list[Entity],
|
|
466
467
|
strategy: str = "token",
|
|
467
468
|
) -> RedactResult:
|
|
468
|
-
"""Redact PII entities from text.
|
|
469
|
+
"""Redact PII entities from text.
|
|
470
|
+
|
|
471
|
+
Callers may pass arbitrary entity lists (not just ``scan()`` output), so
|
|
472
|
+
overlapping or duplicate spans are resolved here: for each overlapping
|
|
473
|
+
group only one span is redacted — the longest, breaking ties by entity
|
|
474
|
+
type priority and then confidence. Suppressed spans are omitted from
|
|
475
|
+
``RedactResult.entities`` and ``mapping``.
|
|
476
|
+
|
|
477
|
+
``mapping`` for the ``mask`` strategy is keyed by per-type indexed keys
|
|
478
|
+
(``[EMAIL_MASK_1]``) rather than the mask string, because distinct
|
|
479
|
+
same-length values produce identical masks and would collide.
|
|
480
|
+
"""
|
|
469
481
|
if not isinstance(text, str):
|
|
470
482
|
raise TypeError("text must be a string")
|
|
471
483
|
if strategy not in {"token", "mask", "hash", "pseudonymize"}:
|
|
@@ -481,14 +493,22 @@ def redact(
|
|
|
481
493
|
for entity in entities
|
|
482
494
|
if 0 <= entity.start < entity.end <= len(text) and entity.text
|
|
483
495
|
]
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
)
|
|
496
|
+
# Returns non-overlapping spans sorted by document position.
|
|
497
|
+
valid_entities = _suppress_overlapping_entities(valid_entities)
|
|
487
498
|
|
|
499
|
+
# Assign replacements in document order so the per-type counters used by
|
|
500
|
+
# the token/pseudonymize strategies number the first occurrence _1; spans
|
|
501
|
+
# are applied right-to-left below so earlier offsets stay valid.
|
|
502
|
+
replacements: list[tuple[Entity, str]] = []
|
|
488
503
|
for entity in valid_entities:
|
|
489
|
-
original =
|
|
504
|
+
original = text[entity.start : entity.end]
|
|
505
|
+
mapping_key: Optional[str] = None
|
|
490
506
|
if strategy == "mask":
|
|
491
507
|
replacement = "*" * max(len(original), 1)
|
|
508
|
+
# Distinct same-length values share a mask, so the mask string
|
|
509
|
+
# cannot key the mapping without collisions.
|
|
510
|
+
counters[entity.type] = counters.get(entity.type, 0) + 1
|
|
511
|
+
mapping_key = f"[{entity.type}_MASK_{counters[entity.type]}]"
|
|
492
512
|
elif strategy == "hash":
|
|
493
513
|
digest = hashlib.sha256(original.encode("utf-8")).hexdigest()[:12]
|
|
494
514
|
replacement = f"[{entity.type}_{digest}]"
|
|
@@ -504,10 +524,13 @@ def redact(
|
|
|
504
524
|
counters[entity.type] = counters.get(entity.type, 0) + 1
|
|
505
525
|
replacement = f"[{entity.type}_{counters[entity.type]}]"
|
|
506
526
|
|
|
527
|
+
replacements.append((entity, replacement))
|
|
528
|
+
mapping[mapping_key if mapping_key is not None else replacement] = original
|
|
529
|
+
|
|
530
|
+
for entity, replacement in reversed(replacements):
|
|
507
531
|
redacted_text = (
|
|
508
532
|
redacted_text[: entity.start] + replacement + redacted_text[entity.end :]
|
|
509
533
|
)
|
|
510
|
-
mapping[replacement] = original
|
|
511
534
|
|
|
512
535
|
return RedactResult(
|
|
513
536
|
redacted_text=redacted_text,
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: datafog
|
|
3
|
-
Version: 4.
|
|
4
|
-
Summary: Lightning-fast PII detection and anonymization library
|
|
3
|
+
Version: 4.8.0
|
|
4
|
+
Summary: Lightning-fast PII detection and anonymization library, 100x+ faster than NER-based detection (see benchmarks/)
|
|
5
5
|
Author: Sid Mohan
|
|
6
6
|
Author-email: sid@datafog.ai
|
|
7
7
|
Project-URL: Homepage, https://datafog.ai
|
|
@@ -126,8 +126,8 @@ values never echoed into logs or transcripts:
|
|
|
126
126
|
|
|
127
127
|
- **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
|
|
128
128
|
commands, web requests, file writes, MCP tools) and warns the model when
|
|
129
|
-
prompts or tool results carry PII. ~
|
|
130
|
-
startup. Easiest install is the
|
|
129
|
+
prompts or tool results carry PII. ~70–90ms per invocation including
|
|
130
|
+
process startup. Easiest install is the
|
|
131
131
|
[Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
|
|
132
132
|
|
|
133
133
|
```
|
|
@@ -139,7 +139,8 @@ values never echoed into logs or transcripts:
|
|
|
139
139
|
|
|
140
140
|
- **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
|
|
141
141
|
requests and responses at the gateway, for any LiteLLM-proxied provider.
|
|
142
|
-
In-process (~
|
|
142
|
+
In-process (~40µs per message scanned; a request clears the guardrail in
|
|
143
|
+
well under a millisecond), no sidecar service. Setup:
|
|
143
144
|
[examples/litellm_guardrail/](examples/litellm_guardrail/).
|
|
144
145
|
|
|
145
146
|
Both default to the high-precision entity set (`EMAIL`, `PHONE`,
|
|
@@ -150,6 +151,10 @@ unix timestamps matching as phone numbers) — available in both adapters and
|
|
|
150
151
|
the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
|
|
151
152
|
`US_SSN`) are accepted as aliases for easy migration.
|
|
152
153
|
|
|
154
|
+
Every performance number above is reproducible with one command —
|
|
155
|
+
methodology, pinned payloads, and comparisons against Presidio and spaCy
|
|
156
|
+
NER live in [benchmarks/](benchmarks/).
|
|
157
|
+
|
|
153
158
|
## Installation
|
|
154
159
|
|
|
155
160
|
```bash
|
|
@@ -110,7 +110,7 @@ setup(
|
|
|
110
110
|
version=version,
|
|
111
111
|
author="Sid Mohan",
|
|
112
112
|
author_email="sid@datafog.ai",
|
|
113
|
-
description="Lightning-fast PII detection and anonymization library
|
|
113
|
+
description="Lightning-fast PII detection and anonymization library, 100x+ faster than NER-based detection (see benchmarks/)",
|
|
114
114
|
long_description=long_description,
|
|
115
115
|
long_description_content_type="text/markdown",
|
|
116
116
|
packages=find_packages(exclude=["tests", "tests.*"]),
|
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
"""Tests for the internal engine boundary API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from datafog.engine import Entity, redact, scan, scan_and_redact
|
|
8
|
+
from datafog.exceptions import EngineNotAvailable
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_scan_regex_detects_structured_entities() -> None:
|
|
12
|
+
result = scan("Email john@example.com and SSN 123-45-6789", engine="regex")
|
|
13
|
+
|
|
14
|
+
entity_types = {entity.type for entity in result.entities}
|
|
15
|
+
assert "EMAIL" in entity_types
|
|
16
|
+
assert "SSN" in entity_types
|
|
17
|
+
assert result.engine_used == "regex"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def test_scan_filters_entity_types() -> None:
|
|
21
|
+
result = scan(
|
|
22
|
+
"Email john@example.com and SSN 123-45-6789",
|
|
23
|
+
engine="regex",
|
|
24
|
+
entity_types=["EMAIL"],
|
|
25
|
+
)
|
|
26
|
+
assert result.entities
|
|
27
|
+
assert {entity.type for entity in result.entities} == {"EMAIL"}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_scan_invalid_engine_raises_value_error() -> None:
|
|
31
|
+
with pytest.raises(ValueError, match="engine must be one of"):
|
|
32
|
+
scan("test", engine="invalid")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_scan_non_string_raises_type_error() -> None:
|
|
36
|
+
with pytest.raises(TypeError, match="text must be a string"):
|
|
37
|
+
scan(None, engine="regex") # type: ignore[arg-type]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
|
|
41
|
+
def test_redact_strategies(strategy: str) -> None:
|
|
42
|
+
text = "Contact john@example.com"
|
|
43
|
+
entities = [
|
|
44
|
+
Entity(
|
|
45
|
+
type="EMAIL",
|
|
46
|
+
text="john@example.com",
|
|
47
|
+
start=8,
|
|
48
|
+
end=24,
|
|
49
|
+
confidence=1.0,
|
|
50
|
+
engine="regex",
|
|
51
|
+
)
|
|
52
|
+
]
|
|
53
|
+
|
|
54
|
+
result = redact(text=text, entities=entities, strategy=strategy)
|
|
55
|
+
assert result.redacted_text != text
|
|
56
|
+
assert result.mapping
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _entity_at(text: str, value: str, entity_type: str = "EMAIL") -> Entity:
|
|
60
|
+
start = text.index(value)
|
|
61
|
+
return Entity(
|
|
62
|
+
type=entity_type,
|
|
63
|
+
text=value,
|
|
64
|
+
start=start,
|
|
65
|
+
end=start + len(value),
|
|
66
|
+
confidence=1.0,
|
|
67
|
+
engine="regex",
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_redact_token_numbering_follows_document_order() -> None:
|
|
72
|
+
text = "first alpha@example.com then beta@example.com end"
|
|
73
|
+
entities = [
|
|
74
|
+
_entity_at(text, "alpha@example.com"),
|
|
75
|
+
_entity_at(text, "beta@example.com"),
|
|
76
|
+
]
|
|
77
|
+
|
|
78
|
+
result = redact(text=text, entities=entities, strategy="token")
|
|
79
|
+
|
|
80
|
+
assert result.redacted_text == "first [EMAIL_1] then [EMAIL_2] end"
|
|
81
|
+
assert result.mapping == {
|
|
82
|
+
"[EMAIL_1]": "alpha@example.com",
|
|
83
|
+
"[EMAIL_2]": "beta@example.com",
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def test_redact_pseudonymize_numbering_follows_document_order() -> None:
|
|
88
|
+
text = "first alpha@example.com then beta@example.com end"
|
|
89
|
+
entities = [
|
|
90
|
+
_entity_at(text, "alpha@example.com"),
|
|
91
|
+
_entity_at(text, "beta@example.com"),
|
|
92
|
+
]
|
|
93
|
+
|
|
94
|
+
result = redact(text=text, entities=entities, strategy="pseudonymize")
|
|
95
|
+
|
|
96
|
+
assert result.redacted_text == "first [EMAIL_PSEUDO_1] then [EMAIL_PSEUDO_2] end"
|
|
97
|
+
assert result.mapping == {
|
|
98
|
+
"[EMAIL_PSEUDO_1]": "alpha@example.com",
|
|
99
|
+
"[EMAIL_PSEUDO_2]": "beta@example.com",
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _span(
|
|
104
|
+
text: str,
|
|
105
|
+
start: int,
|
|
106
|
+
end: int,
|
|
107
|
+
entity_type: str = "PHONE",
|
|
108
|
+
confidence: float = 1.0,
|
|
109
|
+
engine: str = "regex",
|
|
110
|
+
) -> Entity:
|
|
111
|
+
return Entity(
|
|
112
|
+
type=entity_type,
|
|
113
|
+
text=text[start:end],
|
|
114
|
+
start=start,
|
|
115
|
+
end=end,
|
|
116
|
+
confidence=confidence,
|
|
117
|
+
engine=engine,
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
|
|
122
|
+
def test_redact_overlapping_spans_match_longest_only(strategy: str) -> None:
|
|
123
|
+
text = "call 555-000-1111 now"
|
|
124
|
+
full = _span(text, 5, 17)
|
|
125
|
+
suffix = _span(text, 9, 17)
|
|
126
|
+
|
|
127
|
+
result = redact(text=text, entities=[full, suffix], strategy=strategy)
|
|
128
|
+
expected = redact(text=text, entities=[full], strategy=strategy)
|
|
129
|
+
|
|
130
|
+
assert result.redacted_text == expected.redacted_text
|
|
131
|
+
assert result.mapping == expected.mapping
|
|
132
|
+
assert result.entities == [full]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
|
|
136
|
+
def test_redact_nested_spans_keep_outer(strategy: str) -> None:
|
|
137
|
+
text = "mail alice@example.com today"
|
|
138
|
+
outer = _span(text, 5, 22, entity_type="EMAIL")
|
|
139
|
+
inner = _span(text, 11, 18, entity_type="EMAIL")
|
|
140
|
+
|
|
141
|
+
result = redact(text=text, entities=[inner, outer], strategy=strategy)
|
|
142
|
+
expected = redact(text=text, entities=[outer], strategy=strategy)
|
|
143
|
+
|
|
144
|
+
assert result.redacted_text == expected.redacted_text
|
|
145
|
+
assert result.mapping == expected.mapping
|
|
146
|
+
assert result.entities == [outer]
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_redact_overlapping_spans_produce_clean_token_output() -> None:
|
|
150
|
+
text = "call 555-000-1111 now"
|
|
151
|
+
entities = [_span(text, 5, 17), _span(text, 9, 17)]
|
|
152
|
+
|
|
153
|
+
result = redact(text=text, entities=entities, strategy="token")
|
|
154
|
+
|
|
155
|
+
assert result.redacted_text == "call [PHONE_1] now"
|
|
156
|
+
assert result.mapping == {"[PHONE_1]": text[5:17]}
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def test_redact_duplicate_spans_applied_once() -> None:
|
|
160
|
+
text = "mail alice@example.com today"
|
|
161
|
+
entity = _entity_at(text, "alice@example.com")
|
|
162
|
+
|
|
163
|
+
result = redact(text=text, entities=[entity, entity], strategy="token")
|
|
164
|
+
|
|
165
|
+
assert result.redacted_text == "mail [EMAIL_1] today"
|
|
166
|
+
assert result.entities == [entity]
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def test_redact_overlap_tiebreak_prefers_higher_confidence() -> None:
|
|
170
|
+
text = "Acme Corporation announced"
|
|
171
|
+
org = _span(text, 0, 16, entity_type="ORGANIZATION", confidence=0.7, engine="spacy")
|
|
172
|
+
person = _span(text, 0, 16, entity_type="PERSON", confidence=0.9, engine="gliner")
|
|
173
|
+
|
|
174
|
+
result = redact(text=text, entities=[org, person], strategy="token")
|
|
175
|
+
|
|
176
|
+
assert result.redacted_text == "[PERSON_1] announced"
|
|
177
|
+
assert result.entities == [person]
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def test_redact_mask_mapping_distinguishes_same_length_values() -> None:
|
|
181
|
+
text = "a alice@example.com b bobby@example.com c"
|
|
182
|
+
entities = [
|
|
183
|
+
_entity_at(text, "alice@example.com"),
|
|
184
|
+
_entity_at(text, "bobby@example.com"),
|
|
185
|
+
]
|
|
186
|
+
|
|
187
|
+
result = redact(text=text, entities=entities, strategy="mask")
|
|
188
|
+
|
|
189
|
+
assert result.redacted_text == "a " + "*" * 17 + " b " + "*" * 17 + " c"
|
|
190
|
+
assert result.mapping == {
|
|
191
|
+
"[EMAIL_MASK_1]": "alice@example.com",
|
|
192
|
+
"[EMAIL_MASK_2]": "bobby@example.com",
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def test_redact_invalid_strategy_raises_value_error() -> None:
|
|
197
|
+
with pytest.raises(ValueError, match="strategy must be one of"):
|
|
198
|
+
redact("test", entities=[], strategy="invalid")
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def test_redact_ignores_invalid_spans() -> None:
|
|
202
|
+
text = "hello"
|
|
203
|
+
entities = [
|
|
204
|
+
Entity(
|
|
205
|
+
type="EMAIL",
|
|
206
|
+
text="x",
|
|
207
|
+
start=-1,
|
|
208
|
+
end=2,
|
|
209
|
+
confidence=1.0,
|
|
210
|
+
engine="regex",
|
|
211
|
+
),
|
|
212
|
+
Entity(
|
|
213
|
+
type="EMAIL",
|
|
214
|
+
text="x",
|
|
215
|
+
start=2,
|
|
216
|
+
end=10,
|
|
217
|
+
confidence=1.0,
|
|
218
|
+
engine="regex",
|
|
219
|
+
),
|
|
220
|
+
]
|
|
221
|
+
|
|
222
|
+
result = redact(text=text, entities=entities, strategy="token")
|
|
223
|
+
assert result.redacted_text == text
|
|
224
|
+
assert result.mapping == {}
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def test_scan_and_redact_combines_operations() -> None:
|
|
228
|
+
text = "Call me at (555) 123-4567"
|
|
229
|
+
result = scan_and_redact(text=text, engine="regex", strategy="token")
|
|
230
|
+
|
|
231
|
+
assert result.entities
|
|
232
|
+
assert "[PHONE_1]" in result.redacted_text
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
@pytest.mark.asyncio
|
|
236
|
+
async def test_scan_from_async_context() -> None:
|
|
237
|
+
"""Verify sync engine API works when called from async code."""
|
|
238
|
+
result = scan("john@example.com", engine="regex")
|
|
239
|
+
assert len(result.entities) >= 1
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def test_gliner_engine_unavailable_raises_clear_error(
|
|
243
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
244
|
+
) -> None:
|
|
245
|
+
def _raise(_: str):
|
|
246
|
+
raise EngineNotAvailable(
|
|
247
|
+
"GLiNER engine requires the nlp-advanced extra. Install with: pip install datafog[nlp-advanced]"
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
|
|
251
|
+
|
|
252
|
+
with pytest.raises(EngineNotAvailable, match="nlp-advanced"):
|
|
253
|
+
scan("john@example.com", engine="gliner")
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def test_smart_engine_degrades_to_regex_with_warning(
|
|
257
|
+
monkeypatch: pytest.MonkeyPatch,
|
|
258
|
+
) -> None:
|
|
259
|
+
def _raise(_: str):
|
|
260
|
+
raise EngineNotAvailable("not installed")
|
|
261
|
+
|
|
262
|
+
monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
|
|
263
|
+
monkeypatch.setattr("datafog.engine._spacy_entities", _raise)
|
|
264
|
+
|
|
265
|
+
with pytest.warns(UserWarning, match="regex only"):
|
|
266
|
+
result = scan("john@example.com", engine="smart")
|
|
267
|
+
|
|
268
|
+
assert any(entity.type == "EMAIL" for entity in result.entities)
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "4.7.0"
|
|
@@ -1,131 +0,0 @@
|
|
|
1
|
-
"""Tests for the internal engine boundary API."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
import pytest
|
|
6
|
-
|
|
7
|
-
from datafog.engine import Entity, redact, scan, scan_and_redact
|
|
8
|
-
from datafog.exceptions import EngineNotAvailable
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
def test_scan_regex_detects_structured_entities() -> None:
|
|
12
|
-
result = scan("Email john@example.com and SSN 123-45-6789", engine="regex")
|
|
13
|
-
|
|
14
|
-
entity_types = {entity.type for entity in result.entities}
|
|
15
|
-
assert "EMAIL" in entity_types
|
|
16
|
-
assert "SSN" in entity_types
|
|
17
|
-
assert result.engine_used == "regex"
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
def test_scan_filters_entity_types() -> None:
|
|
21
|
-
result = scan(
|
|
22
|
-
"Email john@example.com and SSN 123-45-6789",
|
|
23
|
-
engine="regex",
|
|
24
|
-
entity_types=["EMAIL"],
|
|
25
|
-
)
|
|
26
|
-
assert result.entities
|
|
27
|
-
assert {entity.type for entity in result.entities} == {"EMAIL"}
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
def test_scan_invalid_engine_raises_value_error() -> None:
|
|
31
|
-
with pytest.raises(ValueError, match="engine must be one of"):
|
|
32
|
-
scan("test", engine="invalid")
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def test_scan_non_string_raises_type_error() -> None:
|
|
36
|
-
with pytest.raises(TypeError, match="text must be a string"):
|
|
37
|
-
scan(None, engine="regex") # type: ignore[arg-type]
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
|
|
41
|
-
def test_redact_strategies(strategy: str) -> None:
|
|
42
|
-
text = "Contact john@example.com"
|
|
43
|
-
entities = [
|
|
44
|
-
Entity(
|
|
45
|
-
type="EMAIL",
|
|
46
|
-
text="john@example.com",
|
|
47
|
-
start=8,
|
|
48
|
-
end=24,
|
|
49
|
-
confidence=1.0,
|
|
50
|
-
engine="regex",
|
|
51
|
-
)
|
|
52
|
-
]
|
|
53
|
-
|
|
54
|
-
result = redact(text=text, entities=entities, strategy=strategy)
|
|
55
|
-
assert result.redacted_text != text
|
|
56
|
-
assert result.mapping
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
def test_redact_invalid_strategy_raises_value_error() -> None:
|
|
60
|
-
with pytest.raises(ValueError, match="strategy must be one of"):
|
|
61
|
-
redact("test", entities=[], strategy="invalid")
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
def test_redact_ignores_invalid_spans() -> None:
|
|
65
|
-
text = "hello"
|
|
66
|
-
entities = [
|
|
67
|
-
Entity(
|
|
68
|
-
type="EMAIL",
|
|
69
|
-
text="x",
|
|
70
|
-
start=-1,
|
|
71
|
-
end=2,
|
|
72
|
-
confidence=1.0,
|
|
73
|
-
engine="regex",
|
|
74
|
-
),
|
|
75
|
-
Entity(
|
|
76
|
-
type="EMAIL",
|
|
77
|
-
text="x",
|
|
78
|
-
start=2,
|
|
79
|
-
end=10,
|
|
80
|
-
confidence=1.0,
|
|
81
|
-
engine="regex",
|
|
82
|
-
),
|
|
83
|
-
]
|
|
84
|
-
|
|
85
|
-
result = redact(text=text, entities=entities, strategy="token")
|
|
86
|
-
assert result.redacted_text == text
|
|
87
|
-
assert result.mapping == {}
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
def test_scan_and_redact_combines_operations() -> None:
|
|
91
|
-
text = "Call me at (555) 123-4567"
|
|
92
|
-
result = scan_and_redact(text=text, engine="regex", strategy="token")
|
|
93
|
-
|
|
94
|
-
assert result.entities
|
|
95
|
-
assert "[PHONE_1]" in result.redacted_text
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
@pytest.mark.asyncio
|
|
99
|
-
async def test_scan_from_async_context() -> None:
|
|
100
|
-
"""Verify sync engine API works when called from async code."""
|
|
101
|
-
result = scan("john@example.com", engine="regex")
|
|
102
|
-
assert len(result.entities) >= 1
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
def test_gliner_engine_unavailable_raises_clear_error(
|
|
106
|
-
monkeypatch: pytest.MonkeyPatch,
|
|
107
|
-
) -> None:
|
|
108
|
-
def _raise(_: str):
|
|
109
|
-
raise EngineNotAvailable(
|
|
110
|
-
"GLiNER engine requires the nlp-advanced extra. Install with: pip install datafog[nlp-advanced]"
|
|
111
|
-
)
|
|
112
|
-
|
|
113
|
-
monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
|
|
114
|
-
|
|
115
|
-
with pytest.raises(EngineNotAvailable, match="nlp-advanced"):
|
|
116
|
-
scan("john@example.com", engine="gliner")
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
def test_smart_engine_degrades_to_regex_with_warning(
|
|
120
|
-
monkeypatch: pytest.MonkeyPatch,
|
|
121
|
-
) -> None:
|
|
122
|
-
def _raise(_: str):
|
|
123
|
-
raise EngineNotAvailable("not installed")
|
|
124
|
-
|
|
125
|
-
monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
|
|
126
|
-
monkeypatch.setattr("datafog.engine._spacy_entities", _raise)
|
|
127
|
-
|
|
128
|
-
with pytest.warns(UserWarning, match="regex only"):
|
|
129
|
-
result = scan("john@example.com", engine="smart")
|
|
130
|
-
|
|
131
|
-
assert any(entity.type == "EMAIL" for entity in result.entities)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/pytesseract_processor.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/regex_annotator/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|