datafog 4.7.0__tar.gz → 4.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. {datafog-4.7.0 → datafog-4.8.0}/PKG-INFO +10 -5
  2. {datafog-4.7.0 → datafog-4.8.0}/README.md +8 -3
  3. datafog-4.8.0/datafog/__about__.py +1 -0
  4. {datafog-4.7.0 → datafog-4.8.0}/datafog/__init__.py +2 -1
  5. {datafog-4.7.0 → datafog-4.8.0}/datafog/__init___lean.py +2 -1
  6. {datafog-4.7.0 → datafog-4.8.0}/datafog/engine.py +29 -6
  7. {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/PKG-INFO +10 -5
  8. {datafog-4.7.0 → datafog-4.8.0}/setup.py +1 -1
  9. datafog-4.8.0/tests/test_engine_api.py +268 -0
  10. datafog-4.7.0/datafog/__about__.py +0 -1
  11. datafog-4.7.0/tests/test_engine_api.py +0 -131
  12. {datafog-4.7.0 → datafog-4.8.0}/LICENSE +0 -0
  13. {datafog-4.7.0 → datafog-4.8.0}/datafog/__init___original.py +0 -0
  14. {datafog-4.7.0 → datafog-4.8.0}/datafog/agent.py +0 -0
  15. {datafog-4.7.0 → datafog-4.8.0}/datafog/client.py +0 -0
  16. {datafog-4.7.0 → datafog-4.8.0}/datafog/config.py +0 -0
  17. {datafog-4.7.0 → datafog-4.8.0}/datafog/core.py +0 -0
  18. {datafog-4.7.0 → datafog-4.8.0}/datafog/exceptions.py +0 -0
  19. {datafog-4.7.0 → datafog-4.8.0}/datafog/integrations/__init__.py +0 -0
  20. {datafog-4.7.0 → datafog-4.8.0}/datafog/integrations/claude_code.py +0 -0
  21. {datafog-4.7.0 → datafog-4.8.0}/datafog/integrations/litellm_guardrail.py +0 -0
  22. {datafog-4.7.0 → datafog-4.8.0}/datafog/main.py +0 -0
  23. {datafog-4.7.0 → datafog-4.8.0}/datafog/main_lean.py +0 -0
  24. {datafog-4.7.0 → datafog-4.8.0}/datafog/main_original.py +0 -0
  25. {datafog-4.7.0 → datafog-4.8.0}/datafog/models/__init__.py +0 -0
  26. {datafog-4.7.0 → datafog-4.8.0}/datafog/models/annotator.py +0 -0
  27. {datafog-4.7.0 → datafog-4.8.0}/datafog/models/anonymizer.py +0 -0
  28. {datafog-4.7.0 → datafog-4.8.0}/datafog/models/common.py +0 -0
  29. {datafog-4.7.0 → datafog-4.8.0}/datafog/models/spacy_nlp.py +0 -0
  30. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/__init__.py +0 -0
  31. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/__init__.py +0 -0
  32. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/donut_processor.py +0 -0
  33. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/image_downloader.py +0 -0
  34. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/image_processing/pytesseract_processor.py +0 -0
  35. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/spark_processing/__init__.py +0 -0
  36. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/spark_processing/pyspark_udfs.py +0 -0
  37. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/__init__.py +0 -0
  38. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/gliner_annotator.py +0 -0
  39. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/regex_annotator/__init__.py +0 -0
  40. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/regex_annotator/regex_annotator.py +0 -0
  41. {datafog-4.7.0 → datafog-4.8.0}/datafog/processing/text_processing/spacy_pii_annotator.py +0 -0
  42. {datafog-4.7.0 → datafog-4.8.0}/datafog/py.typed +0 -0
  43. {datafog-4.7.0 → datafog-4.8.0}/datafog/services/__init__.py +0 -0
  44. {datafog-4.7.0 → datafog-4.8.0}/datafog/services/image_service.py +0 -0
  45. {datafog-4.7.0 → datafog-4.8.0}/datafog/services/spark_service.py +0 -0
  46. {datafog-4.7.0 → datafog-4.8.0}/datafog/services/text_service.py +0 -0
  47. {datafog-4.7.0 → datafog-4.8.0}/datafog/services/text_service_lean.py +0 -0
  48. {datafog-4.7.0 → datafog-4.8.0}/datafog/services/text_service_original.py +0 -0
  49. {datafog-4.7.0 → datafog-4.8.0}/datafog/telemetry.py +0 -0
  50. {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/SOURCES.txt +0 -0
  51. {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/dependency_links.txt +0 -0
  52. {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/entry_points.txt +0 -0
  53. {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/requires.txt +0 -0
  54. {datafog-4.7.0 → datafog-4.8.0}/datafog.egg-info/top_level.txt +0 -0
  55. {datafog-4.7.0 → datafog-4.8.0}/setup.cfg +0 -0
  56. {datafog-4.7.0 → datafog-4.8.0}/tests/test_agent_api.py +0 -0
  57. {datafog-4.7.0 → datafog-4.8.0}/tests/test_allowlist.py +0 -0
  58. {datafog-4.7.0 → datafog-4.8.0}/tests/test_anonymizer.py +0 -0
  59. {datafog-4.7.0 → datafog-4.8.0}/tests/test_claude_code_hook.py +0 -0
  60. {datafog-4.7.0 → datafog-4.8.0}/tests/test_cli_smoke.py +0 -0
  61. {datafog-4.7.0 → datafog-4.8.0}/tests/test_client.py +0 -0
  62. {datafog-4.7.0 → datafog-4.8.0}/tests/test_de_pii_regex.py +0 -0
  63. {datafog-4.7.0 → datafog-4.8.0}/tests/test_detection_accuracy.py +0 -0
  64. {datafog-4.7.0 → datafog-4.8.0}/tests/test_donut_lazy_import.py +0 -0
  65. {datafog-4.7.0 → datafog-4.8.0}/tests/test_gliner_annotator.py +0 -0
  66. {datafog-4.7.0 → datafog-4.8.0}/tests/test_image_service.py +0 -0
  67. {datafog-4.7.0 → datafog-4.8.0}/tests/test_install_profiles.py +0 -0
  68. {datafog-4.7.0 → datafog-4.8.0}/tests/test_litellm_guardrail.py +0 -0
  69. {datafog-4.7.0 → datafog-4.8.0}/tests/test_main.py +0 -0
  70. {datafog-4.7.0 → datafog-4.8.0}/tests/test_no_network_core.py +0 -0
  71. {datafog-4.7.0 → datafog-4.8.0}/tests/test_ocr_integration.py +0 -0
  72. {datafog-4.7.0 → datafog-4.8.0}/tests/test_regex_annotator.py +0 -0
  73. {datafog-4.7.0 → datafog-4.8.0}/tests/test_runtime_dependency_safety.py +0 -0
  74. {datafog-4.7.0 → datafog-4.8.0}/tests/test_spark_integration.py +0 -0
  75. {datafog-4.7.0 → datafog-4.8.0}/tests/test_telemetry.py +0 -0
  76. {datafog-4.7.0 → datafog-4.8.0}/tests/test_text_service.py +0 -0
  77. {datafog-4.7.0 → datafog-4.8.0}/tests/test_text_service_integration.py +0 -0
  78. {datafog-4.7.0 → datafog-4.8.0}/tests/test_v44_bridge_api.py +0 -0
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: datafog
3
- Version: 4.7.0
4
- Summary: Lightning-fast PII detection and anonymization library with 190x performance advantage
3
+ Version: 4.8.0
4
+ Summary: Lightning-fast PII detection and anonymization library, 100x+ faster than NER-based detection (see benchmarks/)
5
5
  Author: Sid Mohan
6
6
  Author-email: sid@datafog.ai
7
7
  Project-URL: Homepage, https://datafog.ai
@@ -126,8 +126,8 @@ values never echoed into logs or transcripts:
126
126
 
127
127
  - **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
128
128
  commands, web requests, file writes, MCP tools) and warns the model when
129
- prompts or tool results carry PII. ~70ms per invocation including process
130
- startup. Easiest install is the
129
+ prompts or tool results carry PII. ~70–90ms per invocation including
130
+ process startup. Easiest install is the
131
131
  [Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
132
132
 
133
133
  ```
@@ -139,7 +139,8 @@ values never echoed into logs or transcripts:
139
139
 
140
140
  - **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
141
141
  requests and responses at the gateway, for any LiteLLM-proxied provider.
142
- In-process (~31µs per request), no sidecar service. Setup:
142
+ In-process (~40µs per message scanned; a request clears the guardrail in
143
+ well under a millisecond), no sidecar service. Setup:
143
144
  [examples/litellm_guardrail/](examples/litellm_guardrail/).
144
145
 
145
146
  Both default to the high-precision entity set (`EMAIL`, `PHONE`,
@@ -150,6 +151,10 @@ unix timestamps matching as phone numbers) — available in both adapters and
150
151
  the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
151
152
  `US_SSN`) are accepted as aliases for easy migration.
152
153
 
154
+ Every performance number above is reproducible with one command —
155
+ methodology, pinned payloads, and comparisons against Presidio and spaCy
156
+ NER live in [benchmarks/](benchmarks/).
157
+
153
158
  ## Installation
154
159
 
155
160
  ```bash
@@ -19,8 +19,8 @@ values never echoed into logs or transcripts:
19
19
 
20
20
  - **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
21
21
  commands, web requests, file writes, MCP tools) and warns the model when
22
- prompts or tool results carry PII. ~70ms per invocation including process
23
- startup. Easiest install is the
22
+ prompts or tool results carry PII. ~70–90ms per invocation including
23
+ process startup. Easiest install is the
24
24
  [Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
25
25
 
26
26
  ```
@@ -32,7 +32,8 @@ values never echoed into logs or transcripts:
32
32
 
33
33
  - **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
34
34
  requests and responses at the gateway, for any LiteLLM-proxied provider.
35
- In-process (~31µs per request), no sidecar service. Setup:
35
+ In-process (~40µs per message scanned; a request clears the guardrail in
36
+ well under a millisecond), no sidecar service. Setup:
36
37
  [examples/litellm_guardrail/](examples/litellm_guardrail/).
37
38
 
38
39
  Both default to the high-precision entity set (`EMAIL`, `PHONE`,
@@ -43,6 +44,10 @@ unix timestamps matching as phone numbers) — available in both adapters and
43
44
  the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
44
45
  `US_SSN`) are accepted as aliases for easy migration.
45
46
 
47
+ Every performance number above is reproducible with one command —
48
+ methodology, pinned payloads, and comparisons against Presidio and spaCy
49
+ NER live in [benchmarks/](benchmarks/).
50
+
46
51
  ## Installation
47
52
 
48
53
  ```bash
@@ -0,0 +1 @@
1
+ __version__ = "4.8.0"
@@ -1,7 +1,8 @@
1
1
  """
2
2
  DataFog: Lightning-fast PII detection and anonymization library.
3
3
 
4
- Core package provides regex-based PII detection with 190x performance advantage.
4
+ Core package provides regex-based PII detection, 100x+ faster than NER-based
5
+ detection on identical payloads (reproduce: python benchmarks/run.py).
5
6
  Optional extras available for advanced features:
6
7
  - pip install datafog[nlp] - for spaCy integration
7
8
  - pip install datafog[ocr] - for image/OCR processing
@@ -7,7 +7,8 @@ only as historical reference until legacy cleanup can remove it safely.
7
7
 
8
8
  DataFog: Lightning-fast PII detection and anonymization library.
9
9
 
10
- Core package provides regex-based PII detection with 190x performance advantage.
10
+ Core package provides regex-based PII detection, 100x+ faster than NER-based
11
+ detection on identical payloads (reproduce: python benchmarks/run.py).
11
12
  Optional extras available for advanced features:
12
13
  - pip install datafog[nlp] - for spaCy integration
13
14
  - pip install datafog[ocr] - for image/OCR processing
@@ -173,6 +173,7 @@ def _suppress_overlapping_entities(entities: list[Entity]) -> list[Entity]:
173
173
  key=lambda item: (
174
174
  -_entity_length(item),
175
175
  -ENTITY_TYPE_PRIORITY.get(item.type, 0),
176
+ -item.confidence,
176
177
  item.start,
177
178
  item.end,
178
179
  item.type,
@@ -465,7 +466,18 @@ def redact(
465
466
  entities: list[Entity],
466
467
  strategy: str = "token",
467
468
  ) -> RedactResult:
468
- """Redact PII entities from text."""
469
+ """Redact PII entities from text.
470
+
471
+ Callers may pass arbitrary entity lists (not just ``scan()`` output), so
472
+ overlapping or duplicate spans are resolved here: for each overlapping
473
+ group only one span is redacted — the longest, breaking ties by entity
474
+ type priority and then confidence. Suppressed spans are omitted from
475
+ ``RedactResult.entities`` and ``mapping``.
476
+
477
+ ``mapping`` for the ``mask`` strategy is keyed by per-type indexed keys
478
+ (``[EMAIL_MASK_1]``) rather than the mask string, because distinct
479
+ same-length values produce identical masks and would collide.
480
+ """
469
481
  if not isinstance(text, str):
470
482
  raise TypeError("text must be a string")
471
483
  if strategy not in {"token", "mask", "hash", "pseudonymize"}:
@@ -481,14 +493,22 @@ def redact(
481
493
  for entity in entities
482
494
  if 0 <= entity.start < entity.end <= len(text) and entity.text
483
495
  ]
484
- valid_entities = sorted(
485
- valid_entities, key=lambda e: (e.start, e.end), reverse=True
486
- )
496
+ # Returns non-overlapping spans sorted by document position.
497
+ valid_entities = _suppress_overlapping_entities(valid_entities)
487
498
 
499
+ # Assign replacements in document order so the per-type counters used by
500
+ # the token/pseudonymize strategies number the first occurrence _1; spans
501
+ # are applied right-to-left below so earlier offsets stay valid.
502
+ replacements: list[tuple[Entity, str]] = []
488
503
  for entity in valid_entities:
489
- original = redacted_text[entity.start : entity.end]
504
+ original = text[entity.start : entity.end]
505
+ mapping_key: Optional[str] = None
490
506
  if strategy == "mask":
491
507
  replacement = "*" * max(len(original), 1)
508
+ # Distinct same-length values share a mask, so the mask string
509
+ # cannot key the mapping without collisions.
510
+ counters[entity.type] = counters.get(entity.type, 0) + 1
511
+ mapping_key = f"[{entity.type}_MASK_{counters[entity.type]}]"
492
512
  elif strategy == "hash":
493
513
  digest = hashlib.sha256(original.encode("utf-8")).hexdigest()[:12]
494
514
  replacement = f"[{entity.type}_{digest}]"
@@ -504,10 +524,13 @@ def redact(
504
524
  counters[entity.type] = counters.get(entity.type, 0) + 1
505
525
  replacement = f"[{entity.type}_{counters[entity.type]}]"
506
526
 
527
+ replacements.append((entity, replacement))
528
+ mapping[mapping_key if mapping_key is not None else replacement] = original
529
+
530
+ for entity, replacement in reversed(replacements):
507
531
  redacted_text = (
508
532
  redacted_text[: entity.start] + replacement + redacted_text[entity.end :]
509
533
  )
510
- mapping[replacement] = original
511
534
 
512
535
  return RedactResult(
513
536
  redacted_text=redacted_text,
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: datafog
3
- Version: 4.7.0
4
- Summary: Lightning-fast PII detection and anonymization library with 190x performance advantage
3
+ Version: 4.8.0
4
+ Summary: Lightning-fast PII detection and anonymization library, 100x+ faster than NER-based detection (see benchmarks/)
5
5
  Author: Sid Mohan
6
6
  Author-email: sid@datafog.ai
7
7
  Project-URL: Homepage, https://datafog.ai
@@ -126,8 +126,8 @@ values never echoed into logs or transcripts:
126
126
 
127
127
  - **Claude Code hook** (`datafog-hook`): gates agent tool calls (shell
128
128
  commands, web requests, file writes, MCP tools) and warns the model when
129
- prompts or tool results carry PII. ~70ms per invocation including process
130
- startup. Easiest install is the
129
+ prompts or tool results carry PII. ~70–90ms per invocation including
130
+ process startup. Easiest install is the
131
131
  [Claude Code plugin](https://github.com/DataFog/datafog-claude-plugin):
132
132
 
133
133
  ```
@@ -139,7 +139,8 @@ values never echoed into logs or transcripts:
139
139
 
140
140
  - **LiteLLM guardrail** (`DataFogGuardrail`): redacts or blocks PII in
141
141
  requests and responses at the gateway, for any LiteLLM-proxied provider.
142
- In-process (~31µs per request), no sidecar service. Setup:
142
+ In-process (~40µs per message scanned; a request clears the guardrail in
143
+ well under a millisecond), no sidecar service. Setup:
143
144
  [examples/litellm_guardrail/](examples/litellm_guardrail/).
144
145
 
145
146
  Both default to the high-precision entity set (`EMAIL`, `PHONE`,
@@ -150,6 +151,10 @@ unix timestamps matching as phone numbers) — available in both adapters and
150
151
  the API. Presidio-style entity names (`EMAIL_ADDRESS`, `PHONE_NUMBER`,
151
152
  `US_SSN`) are accepted as aliases for easy migration.
152
153
 
154
+ Every performance number above is reproducible with one command —
155
+ methodology, pinned payloads, and comparisons against Presidio and spaCy
156
+ NER live in [benchmarks/](benchmarks/).
157
+
153
158
  ## Installation
154
159
 
155
160
  ```bash
@@ -110,7 +110,7 @@ setup(
110
110
  version=version,
111
111
  author="Sid Mohan",
112
112
  author_email="sid@datafog.ai",
113
- description="Lightning-fast PII detection and anonymization library with 190x performance advantage",
113
+ description="Lightning-fast PII detection and anonymization library, 100x+ faster than NER-based detection (see benchmarks/)",
114
114
  long_description=long_description,
115
115
  long_description_content_type="text/markdown",
116
116
  packages=find_packages(exclude=["tests", "tests.*"]),
@@ -0,0 +1,268 @@
1
+ """Tests for the internal engine boundary API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import pytest
6
+
7
+ from datafog.engine import Entity, redact, scan, scan_and_redact
8
+ from datafog.exceptions import EngineNotAvailable
9
+
10
+
11
+ def test_scan_regex_detects_structured_entities() -> None:
12
+ result = scan("Email john@example.com and SSN 123-45-6789", engine="regex")
13
+
14
+ entity_types = {entity.type for entity in result.entities}
15
+ assert "EMAIL" in entity_types
16
+ assert "SSN" in entity_types
17
+ assert result.engine_used == "regex"
18
+
19
+
20
+ def test_scan_filters_entity_types() -> None:
21
+ result = scan(
22
+ "Email john@example.com and SSN 123-45-6789",
23
+ engine="regex",
24
+ entity_types=["EMAIL"],
25
+ )
26
+ assert result.entities
27
+ assert {entity.type for entity in result.entities} == {"EMAIL"}
28
+
29
+
30
+ def test_scan_invalid_engine_raises_value_error() -> None:
31
+ with pytest.raises(ValueError, match="engine must be one of"):
32
+ scan("test", engine="invalid")
33
+
34
+
35
+ def test_scan_non_string_raises_type_error() -> None:
36
+ with pytest.raises(TypeError, match="text must be a string"):
37
+ scan(None, engine="regex") # type: ignore[arg-type]
38
+
39
+
40
+ @pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
41
+ def test_redact_strategies(strategy: str) -> None:
42
+ text = "Contact john@example.com"
43
+ entities = [
44
+ Entity(
45
+ type="EMAIL",
46
+ text="john@example.com",
47
+ start=8,
48
+ end=24,
49
+ confidence=1.0,
50
+ engine="regex",
51
+ )
52
+ ]
53
+
54
+ result = redact(text=text, entities=entities, strategy=strategy)
55
+ assert result.redacted_text != text
56
+ assert result.mapping
57
+
58
+
59
+ def _entity_at(text: str, value: str, entity_type: str = "EMAIL") -> Entity:
60
+ start = text.index(value)
61
+ return Entity(
62
+ type=entity_type,
63
+ text=value,
64
+ start=start,
65
+ end=start + len(value),
66
+ confidence=1.0,
67
+ engine="regex",
68
+ )
69
+
70
+
71
+ def test_redact_token_numbering_follows_document_order() -> None:
72
+ text = "first alpha@example.com then beta@example.com end"
73
+ entities = [
74
+ _entity_at(text, "alpha@example.com"),
75
+ _entity_at(text, "beta@example.com"),
76
+ ]
77
+
78
+ result = redact(text=text, entities=entities, strategy="token")
79
+
80
+ assert result.redacted_text == "first [EMAIL_1] then [EMAIL_2] end"
81
+ assert result.mapping == {
82
+ "[EMAIL_1]": "alpha@example.com",
83
+ "[EMAIL_2]": "beta@example.com",
84
+ }
85
+
86
+
87
+ def test_redact_pseudonymize_numbering_follows_document_order() -> None:
88
+ text = "first alpha@example.com then beta@example.com end"
89
+ entities = [
90
+ _entity_at(text, "alpha@example.com"),
91
+ _entity_at(text, "beta@example.com"),
92
+ ]
93
+
94
+ result = redact(text=text, entities=entities, strategy="pseudonymize")
95
+
96
+ assert result.redacted_text == "first [EMAIL_PSEUDO_1] then [EMAIL_PSEUDO_2] end"
97
+ assert result.mapping == {
98
+ "[EMAIL_PSEUDO_1]": "alpha@example.com",
99
+ "[EMAIL_PSEUDO_2]": "beta@example.com",
100
+ }
101
+
102
+
103
+ def _span(
104
+ text: str,
105
+ start: int,
106
+ end: int,
107
+ entity_type: str = "PHONE",
108
+ confidence: float = 1.0,
109
+ engine: str = "regex",
110
+ ) -> Entity:
111
+ return Entity(
112
+ type=entity_type,
113
+ text=text[start:end],
114
+ start=start,
115
+ end=end,
116
+ confidence=confidence,
117
+ engine=engine,
118
+ )
119
+
120
+
121
+ @pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
122
+ def test_redact_overlapping_spans_match_longest_only(strategy: str) -> None:
123
+ text = "call 555-000-1111 now"
124
+ full = _span(text, 5, 17)
125
+ suffix = _span(text, 9, 17)
126
+
127
+ result = redact(text=text, entities=[full, suffix], strategy=strategy)
128
+ expected = redact(text=text, entities=[full], strategy=strategy)
129
+
130
+ assert result.redacted_text == expected.redacted_text
131
+ assert result.mapping == expected.mapping
132
+ assert result.entities == [full]
133
+
134
+
135
+ @pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
136
+ def test_redact_nested_spans_keep_outer(strategy: str) -> None:
137
+ text = "mail alice@example.com today"
138
+ outer = _span(text, 5, 22, entity_type="EMAIL")
139
+ inner = _span(text, 11, 18, entity_type="EMAIL")
140
+
141
+ result = redact(text=text, entities=[inner, outer], strategy=strategy)
142
+ expected = redact(text=text, entities=[outer], strategy=strategy)
143
+
144
+ assert result.redacted_text == expected.redacted_text
145
+ assert result.mapping == expected.mapping
146
+ assert result.entities == [outer]
147
+
148
+
149
+ def test_redact_overlapping_spans_produce_clean_token_output() -> None:
150
+ text = "call 555-000-1111 now"
151
+ entities = [_span(text, 5, 17), _span(text, 9, 17)]
152
+
153
+ result = redact(text=text, entities=entities, strategy="token")
154
+
155
+ assert result.redacted_text == "call [PHONE_1] now"
156
+ assert result.mapping == {"[PHONE_1]": text[5:17]}
157
+
158
+
159
+ def test_redact_duplicate_spans_applied_once() -> None:
160
+ text = "mail alice@example.com today"
161
+ entity = _entity_at(text, "alice@example.com")
162
+
163
+ result = redact(text=text, entities=[entity, entity], strategy="token")
164
+
165
+ assert result.redacted_text == "mail [EMAIL_1] today"
166
+ assert result.entities == [entity]
167
+
168
+
169
+ def test_redact_overlap_tiebreak_prefers_higher_confidence() -> None:
170
+ text = "Acme Corporation announced"
171
+ org = _span(text, 0, 16, entity_type="ORGANIZATION", confidence=0.7, engine="spacy")
172
+ person = _span(text, 0, 16, entity_type="PERSON", confidence=0.9, engine="gliner")
173
+
174
+ result = redact(text=text, entities=[org, person], strategy="token")
175
+
176
+ assert result.redacted_text == "[PERSON_1] announced"
177
+ assert result.entities == [person]
178
+
179
+
180
+ def test_redact_mask_mapping_distinguishes_same_length_values() -> None:
181
+ text = "a alice@example.com b bobby@example.com c"
182
+ entities = [
183
+ _entity_at(text, "alice@example.com"),
184
+ _entity_at(text, "bobby@example.com"),
185
+ ]
186
+
187
+ result = redact(text=text, entities=entities, strategy="mask")
188
+
189
+ assert result.redacted_text == "a " + "*" * 17 + " b " + "*" * 17 + " c"
190
+ assert result.mapping == {
191
+ "[EMAIL_MASK_1]": "alice@example.com",
192
+ "[EMAIL_MASK_2]": "bobby@example.com",
193
+ }
194
+
195
+
196
+ def test_redact_invalid_strategy_raises_value_error() -> None:
197
+ with pytest.raises(ValueError, match="strategy must be one of"):
198
+ redact("test", entities=[], strategy="invalid")
199
+
200
+
201
+ def test_redact_ignores_invalid_spans() -> None:
202
+ text = "hello"
203
+ entities = [
204
+ Entity(
205
+ type="EMAIL",
206
+ text="x",
207
+ start=-1,
208
+ end=2,
209
+ confidence=1.0,
210
+ engine="regex",
211
+ ),
212
+ Entity(
213
+ type="EMAIL",
214
+ text="x",
215
+ start=2,
216
+ end=10,
217
+ confidence=1.0,
218
+ engine="regex",
219
+ ),
220
+ ]
221
+
222
+ result = redact(text=text, entities=entities, strategy="token")
223
+ assert result.redacted_text == text
224
+ assert result.mapping == {}
225
+
226
+
227
+ def test_scan_and_redact_combines_operations() -> None:
228
+ text = "Call me at (555) 123-4567"
229
+ result = scan_and_redact(text=text, engine="regex", strategy="token")
230
+
231
+ assert result.entities
232
+ assert "[PHONE_1]" in result.redacted_text
233
+
234
+
235
+ @pytest.mark.asyncio
236
+ async def test_scan_from_async_context() -> None:
237
+ """Verify sync engine API works when called from async code."""
238
+ result = scan("john@example.com", engine="regex")
239
+ assert len(result.entities) >= 1
240
+
241
+
242
+ def test_gliner_engine_unavailable_raises_clear_error(
243
+ monkeypatch: pytest.MonkeyPatch,
244
+ ) -> None:
245
+ def _raise(_: str):
246
+ raise EngineNotAvailable(
247
+ "GLiNER engine requires the nlp-advanced extra. Install with: pip install datafog[nlp-advanced]"
248
+ )
249
+
250
+ monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
251
+
252
+ with pytest.raises(EngineNotAvailable, match="nlp-advanced"):
253
+ scan("john@example.com", engine="gliner")
254
+
255
+
256
+ def test_smart_engine_degrades_to_regex_with_warning(
257
+ monkeypatch: pytest.MonkeyPatch,
258
+ ) -> None:
259
+ def _raise(_: str):
260
+ raise EngineNotAvailable("not installed")
261
+
262
+ monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
263
+ monkeypatch.setattr("datafog.engine._spacy_entities", _raise)
264
+
265
+ with pytest.warns(UserWarning, match="regex only"):
266
+ result = scan("john@example.com", engine="smart")
267
+
268
+ assert any(entity.type == "EMAIL" for entity in result.entities)
@@ -1 +0,0 @@
1
- __version__ = "4.7.0"
@@ -1,131 +0,0 @@
1
- """Tests for the internal engine boundary API."""
2
-
3
- from __future__ import annotations
4
-
5
- import pytest
6
-
7
- from datafog.engine import Entity, redact, scan, scan_and_redact
8
- from datafog.exceptions import EngineNotAvailable
9
-
10
-
11
- def test_scan_regex_detects_structured_entities() -> None:
12
- result = scan("Email john@example.com and SSN 123-45-6789", engine="regex")
13
-
14
- entity_types = {entity.type for entity in result.entities}
15
- assert "EMAIL" in entity_types
16
- assert "SSN" in entity_types
17
- assert result.engine_used == "regex"
18
-
19
-
20
- def test_scan_filters_entity_types() -> None:
21
- result = scan(
22
- "Email john@example.com and SSN 123-45-6789",
23
- engine="regex",
24
- entity_types=["EMAIL"],
25
- )
26
- assert result.entities
27
- assert {entity.type for entity in result.entities} == {"EMAIL"}
28
-
29
-
30
- def test_scan_invalid_engine_raises_value_error() -> None:
31
- with pytest.raises(ValueError, match="engine must be one of"):
32
- scan("test", engine="invalid")
33
-
34
-
35
- def test_scan_non_string_raises_type_error() -> None:
36
- with pytest.raises(TypeError, match="text must be a string"):
37
- scan(None, engine="regex") # type: ignore[arg-type]
38
-
39
-
40
- @pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"])
41
- def test_redact_strategies(strategy: str) -> None:
42
- text = "Contact john@example.com"
43
- entities = [
44
- Entity(
45
- type="EMAIL",
46
- text="john@example.com",
47
- start=8,
48
- end=24,
49
- confidence=1.0,
50
- engine="regex",
51
- )
52
- ]
53
-
54
- result = redact(text=text, entities=entities, strategy=strategy)
55
- assert result.redacted_text != text
56
- assert result.mapping
57
-
58
-
59
- def test_redact_invalid_strategy_raises_value_error() -> None:
60
- with pytest.raises(ValueError, match="strategy must be one of"):
61
- redact("test", entities=[], strategy="invalid")
62
-
63
-
64
- def test_redact_ignores_invalid_spans() -> None:
65
- text = "hello"
66
- entities = [
67
- Entity(
68
- type="EMAIL",
69
- text="x",
70
- start=-1,
71
- end=2,
72
- confidence=1.0,
73
- engine="regex",
74
- ),
75
- Entity(
76
- type="EMAIL",
77
- text="x",
78
- start=2,
79
- end=10,
80
- confidence=1.0,
81
- engine="regex",
82
- ),
83
- ]
84
-
85
- result = redact(text=text, entities=entities, strategy="token")
86
- assert result.redacted_text == text
87
- assert result.mapping == {}
88
-
89
-
90
- def test_scan_and_redact_combines_operations() -> None:
91
- text = "Call me at (555) 123-4567"
92
- result = scan_and_redact(text=text, engine="regex", strategy="token")
93
-
94
- assert result.entities
95
- assert "[PHONE_1]" in result.redacted_text
96
-
97
-
98
- @pytest.mark.asyncio
99
- async def test_scan_from_async_context() -> None:
100
- """Verify sync engine API works when called from async code."""
101
- result = scan("john@example.com", engine="regex")
102
- assert len(result.entities) >= 1
103
-
104
-
105
- def test_gliner_engine_unavailable_raises_clear_error(
106
- monkeypatch: pytest.MonkeyPatch,
107
- ) -> None:
108
- def _raise(_: str):
109
- raise EngineNotAvailable(
110
- "GLiNER engine requires the nlp-advanced extra. Install with: pip install datafog[nlp-advanced]"
111
- )
112
-
113
- monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
114
-
115
- with pytest.raises(EngineNotAvailable, match="nlp-advanced"):
116
- scan("john@example.com", engine="gliner")
117
-
118
-
119
- def test_smart_engine_degrades_to_regex_with_warning(
120
- monkeypatch: pytest.MonkeyPatch,
121
- ) -> None:
122
- def _raise(_: str):
123
- raise EngineNotAvailable("not installed")
124
-
125
- monkeypatch.setattr("datafog.engine._gliner_entities", _raise)
126
- monkeypatch.setattr("datafog.engine._spacy_entities", _raise)
127
-
128
- with pytest.warns(UserWarning, match="regex only"):
129
- result = scan("john@example.com", engine="smart")
130
-
131
- assert any(entity.type == "EMAIL" for entity in result.entities)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes