datafog-core 0.3.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. {datafog_core-0.3.1 → datafog_core-0.4.0}/Cargo.lock +4 -4
  2. {datafog_core-0.3.1 → datafog_core-0.4.0}/PKG-INFO +40 -3
  3. {datafog_core-0.3.1/crates/core → datafog_core-0.4.0}/README.md +39 -2
  4. {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/Cargo.toml +1 -1
  5. {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/src/lib.rs +9 -0
  6. datafog_core-0.4.0/bindings/python/tests/capabilities_conformance.py +86 -0
  7. datafog_core-0.4.0/bindings/python/tests/german_conformance.py +180 -0
  8. datafog_core-0.4.0/bindings/python/tests/jwt_conformance.py +125 -0
  9. datafog_core-0.4.0/bindings/python/tests/npi_conformance.py +124 -0
  10. datafog_core-0.4.0/bindings/python/tests/private_key_conformance.py +122 -0
  11. {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/tests/test_installed.py +22 -0
  12. {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/tests/test_typing.py +10 -4
  13. datafog_core-0.4.0/bindings/python/tests/us_routing_number_conformance.py +124 -0
  14. datafog_core-0.4.0/bindings/python/tests/uuid_conformance.py +129 -0
  15. {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/Cargo.toml +1 -1
  16. {datafog_core-0.3.1 → datafog_core-0.4.0/crates/core}/README.md +39 -2
  17. datafog_core-0.4.0/crates/core/examples/german_benchmark.rs +47 -0
  18. datafog_core-0.4.0/crates/core/src/capabilities.rs +213 -0
  19. datafog_core-0.4.0/crates/core/src/german.rs +90 -0
  20. datafog_core-0.4.0/crates/core/src/jwt.rs +66 -0
  21. {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/lib.rs +297 -11
  22. datafog_core-0.4.0/crates/core/src/npi.rs +40 -0
  23. datafog_core-0.4.0/crates/core/src/private_key.rs +62 -0
  24. {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/structured.rs +20 -5
  25. datafog_core-0.4.0/crates/core/src/us_routing_number.rs +40 -0
  26. datafog_core-0.4.0/crates/core/src/uuid.rs +42 -0
  27. datafog_core-0.4.0/crates/core/tests/capabilities.rs +211 -0
  28. datafog_core-0.4.0/crates/core/tests/german.rs +182 -0
  29. datafog_core-0.4.0/crates/core/tests/jwt.rs +82 -0
  30. datafog_core-0.4.0/crates/core/tests/npi.rs +93 -0
  31. datafog_core-0.4.0/crates/core/tests/private_key.rs +109 -0
  32. datafog_core-0.4.0/crates/core/tests/us_routing_number.rs +93 -0
  33. datafog_core-0.4.0/crates/core/tests/uuid.rs +117 -0
  34. {datafog_core-0.3.1 → datafog_core-0.4.0}/pyproject.toml +1 -1
  35. {datafog_core-0.3.1 → datafog_core-0.4.0}/python/datafog_core/__init__.py +2 -0
  36. {datafog_core-0.3.1 → datafog_core-0.4.0}/python/datafog_core/__init__.pyi +23 -0
  37. {datafog_core-0.3.1 → datafog_core-0.4.0}/Cargo.toml +0 -0
  38. {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/tests/requirements-typing.txt +0 -0
  39. {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/examples/scan_benchmark.rs +0 -0
  40. {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/offsets.rs +0 -0
  41. {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/selection_tests.rs +0 -0
  42. {datafog_core-0.3.1 → datafog_core-0.4.0}/python/datafog_core/py.typed +0 -0
@@ -80,7 +80,7 @@ checksum = "914a755b7c2d4af2bdcff7ce1739e2db9a1b81a9b07123d8015786ae03c0980d"
80
80
 
81
81
  [[package]]
82
82
  name = "datafog-core"
83
- version = "0.3.1"
83
+ version = "0.4.0"
84
84
  dependencies = [
85
85
  "base64",
86
86
  "futures",
@@ -94,7 +94,7 @@ dependencies = [
94
94
 
95
95
  [[package]]
96
96
  name = "datafog-core-python"
97
- version = "0.3.1"
97
+ version = "0.4.0"
98
98
  dependencies = [
99
99
  "datafog-core",
100
100
  "pyo3",
@@ -104,7 +104,7 @@ dependencies = [
104
104
 
105
105
  [[package]]
106
106
  name = "datafog-node"
107
- version = "0.3.1"
107
+ version = "0.4.0"
108
108
  dependencies = [
109
109
  "datafog-core",
110
110
  "napi",
@@ -115,7 +115,7 @@ dependencies = [
115
115
 
116
116
  [[package]]
117
117
  name = "datafog-wasm"
118
- version = "0.3.1"
118
+ version = "0.4.0"
119
119
  dependencies = [
120
120
  "datafog-core",
121
121
  "serde",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: datafog-core
3
- Version: 0.3.1
3
+ Version: 0.4.0
4
4
  Classifier: Development Status :: 3 - Alpha
5
5
  Classifier: License :: OSI Approved :: MIT License
6
6
  Classifier: Programming Language :: Python :: 3
@@ -25,7 +25,7 @@ Project-URL: Repository, https://github.com/DataFog/datafog-core
25
25
 
26
26
  Fast structured PII detection, implemented in Rust and exposed for Rust, Python, Node.js, and browsers.
27
27
 
28
- It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. Every binding returns the same finding information:
28
+ It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. The next release also detects `JWT` tokens by default; see the [JWT reference](docs/reference/jwt.mdx). With an explicit German locale, it also detects `DE_IBAN`, `DE_VAT_ID`, `DE_TAX_ID`, `DE_SOCIAL_SECURITY_NUMBER`, `DE_POSTAL_CODE`, `DE_PASSPORT_NUMBER`, and `DE_RESIDENCE_PERMIT_NUMBER` (unreleased). Complete PEM private-key blocks are also detected as `PRIVATE_KEY` by default (unreleased); see the [private-key reference](docs/reference/private-keys.mdx). Context-labeled `US_ROUTING_NUMBER` detection is also available (unreleased); see [the detector rules](docs/reference/us-routing-number.mdx). Context-labeled `NPI` detection is also available (unreleased); see [the detector rules](docs/reference/npi.mdx). Every binding returns the same finding information:
29
29
 
30
30
  ```text
31
31
  entity type, matched text, byte range, code-point range,
@@ -79,7 +79,7 @@ full-match regex values:
79
79
  }
80
80
  ```
81
81
 
82
- `scan_and_transform` uses `{ scan?: { locale?: string }, transform: ... }` so
82
+ `scan_and_transform` uses `{ scan?: { locale?: string, detect_uuid?: boolean }, transform: ... }` so
83
83
  detection settings remain separate from transformation policy.
84
84
 
85
85
  ## Packages
@@ -297,3 +297,40 @@ fixtures/ Shared conformance fixtures
297
297
 
298
298
  [MIT](LICENSE)
299
299
 
300
+ ## German structured identifiers (unreleased)
301
+
302
+ Pass `{"locale":"de"}` to text or structured scans. Trimmed, ASCII
303
+ case-insensitive `de`, `de-DE`, and `de_DE` activate all seven German detectors;
304
+ omitted locale and recognized `en-US`/`fr` aliases keep base detection only.
305
+ The 0.4.0 candidate rejects unsupported explicit locales.
306
+
307
+ ```python
308
+ from datafog_core import scan_and_transform
309
+
310
+ result = scan_and_transform("IBAN DE44 5001 0517 5407 3249 31", {
311
+ "scan": {"locale": "de"},
312
+ "transform": {"default": {"strategy": "redact"}, "entities": ["DE_IBAN"]},
313
+ })
314
+ assert result.text == "IBAN [DE_IBAN]"
315
+ ```
316
+
317
+ These are format/context detectors, not official identifier validators. IBAN
318
+ checksums and account existence are not checked. Digits are ASCII; permitted
319
+ internal separators are space, tab, NBSP and narrow NBSP at specified group
320
+ boundaries, never newlines. Returned text and offsets preserve the source.
321
+ Passport and residence-permit patterns are legacy heuristics with limited
322
+ coverage. See the [German entity reference](docs/reference/german-entities.mdx)
323
+ and [migration differences](docs/guides/migrating-from-datafog-python.mdx).
324
+ The Python 4.9 adapter requires a subsequently published compatible Core wheel;
325
+ this source change does not update its extra pin or publish a release.
326
+
327
+ ## UUID identifiers (unreleased)
328
+
329
+ Canonical UUID detection is opt-in: pass `{"detect_uuid":true}` to text or
330
+ structured scans, independently of locale. It emits `UUID` findings for versions
331
+ 1–8 with the IETF variant and original casing/ranges. UUID syntax does not imply
332
+ sensitivity. See the [UUID reference](docs/reference/uuid.mdx) for boundaries,
333
+ excluded sentinel forms and transformation examples.
334
+
335
+ The source candidate targets **0.4.0**; publication and downstream Python integration are separate release gates. See the [candidate release checklist](docs/releases/0-4-0.mdx), [runtime capabilities](docs/reference/capabilities.mdx), and [0.4.x compatibility policy](docs/reference/compatibility.mdx).
336
+
@@ -2,7 +2,7 @@
2
2
 
3
3
  Fast structured PII detection, implemented in Rust and exposed for Rust, Python, Node.js, and browsers.
4
4
 
5
- It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. Every binding returns the same finding information:
5
+ It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. The next release also detects `JWT` tokens by default; see the [JWT reference](docs/reference/jwt.mdx). With an explicit German locale, it also detects `DE_IBAN`, `DE_VAT_ID`, `DE_TAX_ID`, `DE_SOCIAL_SECURITY_NUMBER`, `DE_POSTAL_CODE`, `DE_PASSPORT_NUMBER`, and `DE_RESIDENCE_PERMIT_NUMBER` (unreleased). Complete PEM private-key blocks are also detected as `PRIVATE_KEY` by default (unreleased); see the [private-key reference](docs/reference/private-keys.mdx). Context-labeled `US_ROUTING_NUMBER` detection is also available (unreleased); see [the detector rules](docs/reference/us-routing-number.mdx). Context-labeled `NPI` detection is also available (unreleased); see [the detector rules](docs/reference/npi.mdx). Every binding returns the same finding information:
6
6
 
7
7
  ```text
8
8
  entity type, matched text, byte range, code-point range,
@@ -56,7 +56,7 @@ full-match regex values:
56
56
  }
57
57
  ```
58
58
 
59
- `scan_and_transform` uses `{ scan?: { locale?: string }, transform: ... }` so
59
+ `scan_and_transform` uses `{ scan?: { locale?: string, detect_uuid?: boolean }, transform: ... }` so
60
60
  detection settings remain separate from transformation policy.
61
61
 
62
62
  ## Packages
@@ -273,3 +273,40 @@ fixtures/ Shared conformance fixtures
273
273
  ## License
274
274
 
275
275
  [MIT](LICENSE)
276
+
277
+ ## German structured identifiers (unreleased)
278
+
279
+ Pass `{"locale":"de"}` to text or structured scans. Trimmed, ASCII
280
+ case-insensitive `de`, `de-DE`, and `de_DE` activate all seven German detectors;
281
+ omitted locale and recognized `en-US`/`fr` aliases keep base detection only.
282
+ The 0.4.0 candidate rejects unsupported explicit locales.
283
+
284
+ ```python
285
+ from datafog_core import scan_and_transform
286
+
287
+ result = scan_and_transform("IBAN DE44 5001 0517 5407 3249 31", {
288
+ "scan": {"locale": "de"},
289
+ "transform": {"default": {"strategy": "redact"}, "entities": ["DE_IBAN"]},
290
+ })
291
+ assert result.text == "IBAN [DE_IBAN]"
292
+ ```
293
+
294
+ These are format/context detectors, not official identifier validators. IBAN
295
+ checksums and account existence are not checked. Digits are ASCII; permitted
296
+ internal separators are space, tab, NBSP and narrow NBSP at specified group
297
+ boundaries, never newlines. Returned text and offsets preserve the source.
298
+ Passport and residence-permit patterns are legacy heuristics with limited
299
+ coverage. See the [German entity reference](docs/reference/german-entities.mdx)
300
+ and [migration differences](docs/guides/migrating-from-datafog-python.mdx).
301
+ The Python 4.9 adapter requires a subsequently published compatible Core wheel;
302
+ this source change does not update its extra pin or publish a release.
303
+
304
+ ## UUID identifiers (unreleased)
305
+
306
+ Canonical UUID detection is opt-in: pass `{"detect_uuid":true}` to text or
307
+ structured scans, independently of locale. It emits `UUID` findings for versions
308
+ 1–8 with the IETF variant and original casing/ranges. UUID syntax does not imply
309
+ sensitivity. See the [UUID reference](docs/reference/uuid.mdx) for boundaries,
310
+ excluded sentinel forms and transformation examples.
311
+
312
+ The source candidate targets **0.4.0**; publication and downstream Python integration are separate release gates. See the [candidate release checklist](docs/releases/0-4-0.mdx), [runtime capabilities](docs/reference/capabilities.mdx), and [0.4.x compatibility policy](docs/reference/compatibility.mdx).
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "datafog-core-python"
3
- version = "0.3.1"
3
+ version = "0.4.0"
4
4
  edition = "2024"
5
5
  rust-version = "1.88"
6
6
  description = "Python bindings for datafog-core"
@@ -956,6 +956,14 @@ fn privacy_error(py: Python<'_>, error: core::PrivacyError) -> PyErr {
956
956
  exception
957
957
  }
958
958
 
959
+ /// Return an independent, JSON-compatible snapshot of the Core capability contract.
960
+ #[pyfunction]
961
+ fn capabilities(py: Python<'_>) -> PyResult<Py<PyAny>> {
962
+ let json = serde_json::to_string(&core::capabilities())
963
+ .map_err(|_| internal_error(py, "failed to serialize Core capabilities"))?;
964
+ Ok(py.import("json")?.call_method1("loads", (json,))?.unbind())
965
+ }
966
+
959
967
  /// Scan text for supported PII findings.
960
968
  #[pyfunction]
961
969
  #[pyo3(signature = (text, config=None))]
@@ -1300,6 +1308,7 @@ fn datafog_core(module: &Bound<'_, PyModule>) -> PyResult<()> {
1300
1308
  module.add_class::<Restoration>()?;
1301
1309
  module.add_class::<RestoreResult>()?;
1302
1310
  module.add_class::<PrivacyManager>()?;
1311
+ module.add_function(wrap_pyfunction!(capabilities, module)?)?;
1303
1312
  module.add_function(wrap_pyfunction!(scan, module)?)?;
1304
1313
  module.add_function(wrap_pyfunction!(transform, module)?)?;
1305
1314
  module.add_function(wrap_pyfunction!(scan_and_transform, module)?)?;
@@ -0,0 +1,86 @@
1
+ """Capability discovery contract against the installed, compiled Core wheel."""
2
+ import json
3
+ from pathlib import Path
4
+
5
+ import datafog_core as api
6
+
7
+ SUPPORTED = sorted([
8
+ "CREDIT_CARD", "DATE", "DE_IBAN", "DE_PASSPORT_NUMBER", "DE_POSTAL_CODE",
9
+ "DE_RESIDENCE_PERMIT_NUMBER", "DE_SOCIAL_SECURITY_NUMBER", "DE_TAX_ID",
10
+ "DE_VAT_ID", "EMAIL", "IP_ADDRESS", "JWT", "NPI", "PERSON", "PHONE",
11
+ "PRIVATE_KEY", "SSN", "US_ROUTING_NUMBER", "UUID", "ZIP_CODE",
12
+ ])
13
+ GERMAN = [entity for entity in SUPPORTED if entity.startswith("DE_")]
14
+ DEFAULT = [entity for entity in SUPPORTED if entity not in GERMAN + ["PERSON", "UUID"]]
15
+
16
+
17
+ def verify():
18
+ assert callable(api.capabilities)
19
+ try:
20
+ api.capabilities({})
21
+ except TypeError:
22
+ pass
23
+ else:
24
+ raise AssertionError("capabilities accepted arguments")
25
+ snapshot = api.capabilities()
26
+ assert type(snapshot) is dict
27
+ assert snapshot["contract_version"] == 1
28
+ assert json.loads(json.dumps(snapshot)) == snapshot
29
+ assert snapshot["supported_entities"] == SUPPORTED
30
+ assert snapshot["default_entities"] == DEFAULT
31
+ assert sorted(snapshot["entities"]) == SUPPORTED
32
+ for key in ("supported_entities", "default_entities"):
33
+ assert snapshot[key] == sorted(set(snapshot[key]))
34
+ assert set(snapshot["locales"]) == {"de", "de-DE", "de_DE", "en-US", "fr"}
35
+ for locale, metadata in snapshot["locales"].items():
36
+ assert metadata["enabled_entities"] == (GERMAN if locale.startswith("de") else [])
37
+ for label, entity in snapshot["entities"].items():
38
+ assert entity["scopes"] == (["structured"] if label == "PERSON" else ["structured", "text"])
39
+ kind = "locale" if label in GERMAN else "config" if label == "UUID" else "structured" if label == "PERSON" else "default"
40
+ assert entity["activation"]["kind"] == kind
41
+ uuid_config = snapshot["entities"]["UUID"]["activation"]["scan_config"]
42
+ assert uuid_config == {"detect_uuid": True}
43
+ text = "👋 café é 550e8400-e29b-41d4-a716-446655440000"
44
+ assert not any(f.entity_type == "UUID" for f in api.scan(text))
45
+ uuid_findings = [f for f in api.scan(text, uuid_config) if f.entity_type == "UUID"]
46
+ assert len(uuid_findings) == 1
47
+ finding = uuid_findings[0]
48
+ assert text[finding.codepoint_range.start:finding.codepoint_range.end] == finding.matched_text
49
+ assert text.encode()[finding.byte_range.start:finding.byte_range.end].decode() == finding.matched_text
50
+ samples = {}
51
+ for line in (Path(__file__).resolve().parents[3] / "fixtures/german.jsonl").read_text().splitlines():
52
+ row = json.loads(line)
53
+ for expected in row["entities"]:
54
+ samples.setdefault(expected["label"], row["text"])
55
+ for locale in ("de", "de-DE", "de_DE", "DE-de", " De "):
56
+ for label, text in samples.items():
57
+ assert any(f.entity_type == label for f in api.scan(text, {"locale": locale})), (locale, label)
58
+ for label, text in samples.items():
59
+ activation = snapshot["entities"][label]["activation"]["scan_config"]
60
+ assert any(f.entity_type == label for f in api.scan(text, activation))
61
+ for locale in ("en-US", "fr"):
62
+ assert api.scan("jane@example.com", {"locale": locale}) == api.scan("jane@example.com")
63
+ for call, path in [
64
+ (lambda: api.scan("", {"locale": "zz-ZZ"}), "/locale"),
65
+ (lambda: api.scan_structured({}, {"locale": "zz-ZZ"}), "/locale"),
66
+ (lambda: api.scan_and_transform("", {"scan": {"locale": "zz-ZZ"}, "transform": {"default": {"strategy": "redact"}}}), "/scan/locale"),
67
+ (lambda: api.scan_and_transform_structured({}, {"scan": {"locale": "zz-ZZ"}, "transform": {"default": {"strategy": "redact"}}}), "/scan/locale"),
68
+ ]:
69
+ try:
70
+ call()
71
+ except api.DataFogConfigurationError as error:
72
+ assert error.code == "invalid_configuration"
73
+ assert error.reason == "invalid_value"
74
+ assert error.path == path
75
+ else:
76
+ raise AssertionError("unsupported locale accepted")
77
+ assert not any(f.entity_type == "PERSON" for f in api.scan("Jane Doe"))
78
+ assert any(f.finding.entity_type == "PERSON" for f in api.scan_structured({"full_name": "Jane Doe"}).findings)
79
+ mutated = api.capabilities()
80
+ assert mutated is not snapshot
81
+ mutated["supported_entities"].clear()
82
+ mutated["default_entities"].append("FAKE")
83
+ mutated["locales"]["de"]["enabled_entities"].clear()
84
+ mutated["entities"]["EMAIL"]["scopes"].clear()
85
+ mutated["entities"]["UUID"]["activation"]["scan_config"]["detect_uuid"] = False
86
+ assert api.capabilities() == snapshot
@@ -0,0 +1,180 @@
1
+ """Shared German fixtures against the installed wheel, including provider round trips."""
2
+
3
+ import json
4
+ from pathlib import Path
5
+ import datafog_core as api
6
+
7
+ RECORDS = [
8
+ json.loads(line)
9
+ for line in (Path(__file__).resolve().parents[3] / "fixtures/german.jsonl")
10
+ .read_text()
11
+ .splitlines()
12
+ ]
13
+
14
+
15
+ def project(text, findings):
16
+ result = []
17
+ for f in findings:
18
+ if not f.entity_type.startswith("DE_"):
19
+ continue
20
+ assert (
21
+ text.encode()[f.byte_range.start : f.byte_range.end].decode()
22
+ == f.matched_text
23
+ )
24
+ assert text[f.codepoint_range.start : f.codepoint_range.end] == f.matched_text
25
+ assert f.confidence is None
26
+ assert f.detector_name == "datafog-core/" + f.entity_type.lower().replace(
27
+ "_", "-"
28
+ )
29
+ assert f.detector_version
30
+ result.append(
31
+ dict(
32
+ label=f.entity_type,
33
+ text=f.matched_text,
34
+ start=f.codepoint_range.start,
35
+ end=f.codepoint_range.end,
36
+ )
37
+ )
38
+ return result
39
+
40
+
41
+ def verify():
42
+ for row in RECORDS:
43
+ findings = api.scan(row["text"], row["config"])
44
+ assert project(row["text"], findings) == row["entities"], row["id"]
45
+ data = {"a/b": row["text"], "array": [row["text"]], "z": row["text"]}
46
+ located = api.scan_structured(data, row["config"]).findings
47
+ for path in ["/a~1b", "/array/0", "/z"]:
48
+ assert (
49
+ project(row["text"], [f.finding for f in located if f.path == path])
50
+ == row["entities"]
51
+ ), row["id"]
52
+ for case in row.get("transforms", []):
53
+ result = api.scan_and_transform(
54
+ row["text"], {"scan": row["config"], "transform": case["config"]}
55
+ )
56
+ assert result.text == case["text"], row["id"]
57
+ assert api.transform(row["text"], findings, case["config"]) == result
58
+ for t in result.transformations:
59
+ assert (
60
+ result.text.encode()[
61
+ t.output_byte_range.start : t.output_byte_range.end
62
+ ].decode()
63
+ == t.replacement
64
+ )
65
+ assert (
66
+ result.text[
67
+ t.output_codepoint_range.start : t.output_codepoint_range.end
68
+ ]
69
+ == t.replacement
70
+ )
71
+ assert any(
72
+ f.byte_range == t.source_byte_range
73
+ and f.entity_type == t.entity_type
74
+ for f in findings
75
+ )
76
+ assert not hasattr(t, "matched_text") and not hasattr(t, "finding")
77
+ structured = api.scan_and_transform_structured(
78
+ data, {"scan": row["config"], "transform": case["config"]}
79
+ )
80
+ assert structured.data == {
81
+ "a/b": case["text"],
82
+ "array": [case["text"]],
83
+ "z": case["text"],
84
+ }
85
+ explicit = api.transform_structured(data, located, case["config"])
86
+ assert explicit.data == structured.data
87
+ assert [(t.path, t.transformation) for t in explicit.transformations] == [
88
+ (t.path, t.transformation) for t in structured.transformations
89
+ ]
90
+ split = {
91
+ "Steuer-ID": "12345678901",
92
+ "a": "Steuer-ID",
93
+ "b": "12345678901",
94
+ "c": ["Passport", "C12345678"],
95
+ "d": "DE4450010517",
96
+ "e": "5407324931",
97
+ }
98
+ assert not any(
99
+ f.finding.entity_type.startswith("DE_")
100
+ for f in api.scan_structured(split, {"locale": "de"}).findings
101
+ )
102
+ for config in [
103
+ {"locale": ""},
104
+ {"locale": " \t"},
105
+ {"locale": None},
106
+ {"locale": 1},
107
+ {"locales": ["de"]},
108
+ ]:
109
+ for call in [
110
+ lambda: api.scan("", config),
111
+ lambda: api.scan_structured({}, config),
112
+ ]:
113
+ try:
114
+ call()
115
+ except api.DataFogConfigurationError:
116
+ pass
117
+ else:
118
+ raise AssertionError("malformed config accepted")
119
+ assert (
120
+ api.scan_and_transform(
121
+ "DE44500105175407324931",
122
+ {"transform": {"default": {"strategy": "redact"}, "entities": ["DE_IBAN"]}},
123
+ ).text
124
+ == "DE44500105175407324931"
125
+ )
126
+ for text, label, generic, expected in [
127
+ ("DE44\t4111111111111111\t11", "DE_IBAN", "CREDIT_CARD", "[DE_IBAN]"),
128
+ ("DE 123456789", "DE_VAT_ID", "SSN", "[DE_VAT_ID]"),
129
+ ("Steuer-ID 12345678901", "DE_TAX_ID", "PHONE", "Steuer-ID [DE_TAX_ID]"),
130
+ ("DE-10115", "DE_POSTAL_CODE", "ZIP_CODE", "[DE_POSTAL_CODE]"),
131
+ ]:
132
+ findings = api.scan(text, {"locale": "de"})
133
+ assert all(
134
+ any(f.entity_type == entity for f in findings)
135
+ for entity in [label, generic]
136
+ )
137
+ assert (
138
+ api.scan_and_transform(
139
+ text,
140
+ {
141
+ "scan": {"locale": "de"},
142
+ "transform": {"default": {"strategy": "redact"}},
143
+ },
144
+ ).text
145
+ == expected
146
+ )
147
+
148
+
149
+ async def verify_providers(manager, token_manager, context):
150
+ for row in RECORDS:
151
+ if not row.get("sample"):
152
+ continue
153
+ entities = list(dict.fromkeys(e["label"] for e in row["entities"]))
154
+ pseudonyms = await manager.scan_and_transform(
155
+ row["text"],
156
+ {
157
+ "scan": row["config"],
158
+ "transform": {
159
+ "default": {"strategy": "pseudonymize", "key_ref": "german"},
160
+ "entities": entities,
161
+ },
162
+ },
163
+ )
164
+ assert (
165
+ len(pseudonyms.transformations) == len(row["entities"])
166
+ and pseudonyms.text != row["text"]
167
+ )
168
+ tokens = await token_manager.scan_and_transform(
169
+ row["text"],
170
+ {
171
+ "scan": row["config"],
172
+ "transform": {
173
+ "default": {"strategy": "tokenize", "token_ref": "german"},
174
+ "entities": entities,
175
+ },
176
+ },
177
+ context,
178
+ )
179
+ assert len(tokens.transformations) == len(row["entities"])
180
+ assert (await token_manager.restore(tokens.text, context)).text == row["text"]
@@ -0,0 +1,125 @@
1
+ """Shared JWT fixtures against the installed wheel, including provider round trips."""
2
+
3
+ import json
4
+ from pathlib import Path
5
+ import datafog_core as api
6
+
7
+ RECORDS = [
8
+ json.loads(line)
9
+ for line in (Path(__file__).resolve().parents[3] / "fixtures/jwt.jsonl")
10
+ .read_text()
11
+ .splitlines()
12
+ ]
13
+
14
+
15
+ def project(text, findings):
16
+ result = []
17
+ for f in findings:
18
+ if not f.entity_type == "JWT":
19
+ continue
20
+ assert (
21
+ text.encode()[f.byte_range.start : f.byte_range.end].decode()
22
+ == f.matched_text
23
+ )
24
+ assert text[f.codepoint_range.start : f.codepoint_range.end] == f.matched_text
25
+ assert f.confidence is None
26
+ assert f.detector_name == "datafog-core/" + f.entity_type.lower().replace(
27
+ "_", "-"
28
+ )
29
+ assert f.detector_version
30
+ result.append(
31
+ dict(
32
+ label=f.entity_type,
33
+ text=f.matched_text,
34
+ start=f.codepoint_range.start,
35
+ end=f.codepoint_range.end,
36
+ )
37
+ )
38
+ return result
39
+
40
+
41
+ def verify():
42
+ for row in RECORDS:
43
+ findings = api.scan(row["text"], row["config"])
44
+ if "overlap" in row:
45
+ assert any(f.entity_type == row["overlap"] for f in findings)
46
+ assert project(row["text"], findings) == row["entities"], row["id"]
47
+ data = {"a/b": row["text"], "array": [row["text"]], "z": row["text"]}
48
+ located = api.scan_structured(data, row["config"]).findings
49
+ for path in ["/a~1b", "/array/0", "/z"]:
50
+ assert (
51
+ project(row["text"], [f.finding for f in located if f.path == path])
52
+ == row["entities"]
53
+ ), row["id"]
54
+ for case in row.get("transforms", []):
55
+ result = api.scan_and_transform(
56
+ row["text"], {"scan": row["config"], "transform": case["config"]}
57
+ )
58
+ assert result.text == case["text"], row["id"]
59
+ assert api.transform(row["text"], findings, case["config"]) == result
60
+ for t in result.transformations:
61
+ assert (
62
+ result.text.encode()[
63
+ t.output_byte_range.start : t.output_byte_range.end
64
+ ].decode()
65
+ == t.replacement
66
+ )
67
+ assert (
68
+ result.text[
69
+ t.output_codepoint_range.start : t.output_codepoint_range.end
70
+ ]
71
+ == t.replacement
72
+ )
73
+ assert any(
74
+ f.byte_range == t.source_byte_range
75
+ and f.entity_type == t.entity_type
76
+ for f in findings
77
+ )
78
+ assert not hasattr(t, "matched_text") and not hasattr(t, "finding")
79
+ structured = api.scan_and_transform_structured(
80
+ data, {"scan": row["config"], "transform": case["config"]}
81
+ )
82
+ assert structured.data == {
83
+ "a/b": case["text"],
84
+ "array": [case["text"]],
85
+ "z": case["text"],
86
+ }
87
+ explicit = api.transform_structured(data, located, case["config"])
88
+ assert explicit.data == structured.data
89
+ assert [(t.path, t.transformation) for t in explicit.transformations] == [
90
+ (t.path, t.transformation) for t in structured.transformations
91
+ ]
92
+
93
+
94
+ async def verify_providers(manager, token_manager, context):
95
+ for row in RECORDS:
96
+ if not row.get("sample"):
97
+ continue
98
+ entities = list(dict.fromkeys(e["label"] for e in row["entities"]))
99
+ pseudonyms = await manager.scan_and_transform(
100
+ row["text"],
101
+ {
102
+ "scan": row["config"],
103
+ "transform": {
104
+ "default": {"strategy": "pseudonymize", "key_ref": "jwt"},
105
+ "entities": entities,
106
+ },
107
+ },
108
+ )
109
+ assert (
110
+ len(pseudonyms.transformations) == len(row["entities"])
111
+ and pseudonyms.text != row["text"]
112
+ )
113
+ tokens = await token_manager.scan_and_transform(
114
+ row["text"],
115
+ {
116
+ "scan": row["config"],
117
+ "transform": {
118
+ "default": {"strategy": "tokenize", "token_ref": "jwt"},
119
+ "entities": entities,
120
+ },
121
+ },
122
+ context,
123
+ )
124
+ assert len(tokens.transformations) == len(row["entities"])
125
+ assert (await token_manager.restore(tokens.text, context)).text == row["text"]