datafog-core 0.3.1__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {datafog_core-0.3.1 → datafog_core-0.4.0}/Cargo.lock +4 -4
- {datafog_core-0.3.1 → datafog_core-0.4.0}/PKG-INFO +40 -3
- {datafog_core-0.3.1/crates/core → datafog_core-0.4.0}/README.md +39 -2
- {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/Cargo.toml +1 -1
- {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/src/lib.rs +9 -0
- datafog_core-0.4.0/bindings/python/tests/capabilities_conformance.py +86 -0
- datafog_core-0.4.0/bindings/python/tests/german_conformance.py +180 -0
- datafog_core-0.4.0/bindings/python/tests/jwt_conformance.py +125 -0
- datafog_core-0.4.0/bindings/python/tests/npi_conformance.py +124 -0
- datafog_core-0.4.0/bindings/python/tests/private_key_conformance.py +122 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/tests/test_installed.py +22 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/tests/test_typing.py +10 -4
- datafog_core-0.4.0/bindings/python/tests/us_routing_number_conformance.py +124 -0
- datafog_core-0.4.0/bindings/python/tests/uuid_conformance.py +129 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/Cargo.toml +1 -1
- {datafog_core-0.3.1 → datafog_core-0.4.0/crates/core}/README.md +39 -2
- datafog_core-0.4.0/crates/core/examples/german_benchmark.rs +47 -0
- datafog_core-0.4.0/crates/core/src/capabilities.rs +213 -0
- datafog_core-0.4.0/crates/core/src/german.rs +90 -0
- datafog_core-0.4.0/crates/core/src/jwt.rs +66 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/lib.rs +297 -11
- datafog_core-0.4.0/crates/core/src/npi.rs +40 -0
- datafog_core-0.4.0/crates/core/src/private_key.rs +62 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/structured.rs +20 -5
- datafog_core-0.4.0/crates/core/src/us_routing_number.rs +40 -0
- datafog_core-0.4.0/crates/core/src/uuid.rs +42 -0
- datafog_core-0.4.0/crates/core/tests/capabilities.rs +211 -0
- datafog_core-0.4.0/crates/core/tests/german.rs +182 -0
- datafog_core-0.4.0/crates/core/tests/jwt.rs +82 -0
- datafog_core-0.4.0/crates/core/tests/npi.rs +93 -0
- datafog_core-0.4.0/crates/core/tests/private_key.rs +109 -0
- datafog_core-0.4.0/crates/core/tests/us_routing_number.rs +93 -0
- datafog_core-0.4.0/crates/core/tests/uuid.rs +117 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/pyproject.toml +1 -1
- {datafog_core-0.3.1 → datafog_core-0.4.0}/python/datafog_core/__init__.py +2 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/python/datafog_core/__init__.pyi +23 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/Cargo.toml +0 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/bindings/python/tests/requirements-typing.txt +0 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/examples/scan_benchmark.rs +0 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/offsets.rs +0 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/crates/core/src/selection_tests.rs +0 -0
- {datafog_core-0.3.1 → datafog_core-0.4.0}/python/datafog_core/py.typed +0 -0
|
@@ -80,7 +80,7 @@ checksum = "914a755b7c2d4af2bdcff7ce1739e2db9a1b81a9b07123d8015786ae03c0980d"
|
|
|
80
80
|
|
|
81
81
|
[[package]]
|
|
82
82
|
name = "datafog-core"
|
|
83
|
-
version = "0.
|
|
83
|
+
version = "0.4.0"
|
|
84
84
|
dependencies = [
|
|
85
85
|
"base64",
|
|
86
86
|
"futures",
|
|
@@ -94,7 +94,7 @@ dependencies = [
|
|
|
94
94
|
|
|
95
95
|
[[package]]
|
|
96
96
|
name = "datafog-core-python"
|
|
97
|
-
version = "0.
|
|
97
|
+
version = "0.4.0"
|
|
98
98
|
dependencies = [
|
|
99
99
|
"datafog-core",
|
|
100
100
|
"pyo3",
|
|
@@ -104,7 +104,7 @@ dependencies = [
|
|
|
104
104
|
|
|
105
105
|
[[package]]
|
|
106
106
|
name = "datafog-node"
|
|
107
|
-
version = "0.
|
|
107
|
+
version = "0.4.0"
|
|
108
108
|
dependencies = [
|
|
109
109
|
"datafog-core",
|
|
110
110
|
"napi",
|
|
@@ -115,7 +115,7 @@ dependencies = [
|
|
|
115
115
|
|
|
116
116
|
[[package]]
|
|
117
117
|
name = "datafog-wasm"
|
|
118
|
-
version = "0.
|
|
118
|
+
version = "0.4.0"
|
|
119
119
|
dependencies = [
|
|
120
120
|
"datafog-core",
|
|
121
121
|
"serde",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: datafog-core
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Classifier: Development Status :: 3 - Alpha
|
|
5
5
|
Classifier: License :: OSI Approved :: MIT License
|
|
6
6
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -25,7 +25,7 @@ Project-URL: Repository, https://github.com/DataFog/datafog-core
|
|
|
25
25
|
|
|
26
26
|
Fast structured PII detection, implemented in Rust and exposed for Rust, Python, Node.js, and browsers.
|
|
27
27
|
|
|
28
|
-
It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. Every binding returns the same finding information:
|
|
28
|
+
It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. The next release also detects `JWT` tokens by default; see the [JWT reference](docs/reference/jwt.mdx). With an explicit German locale, it also detects `DE_IBAN`, `DE_VAT_ID`, `DE_TAX_ID`, `DE_SOCIAL_SECURITY_NUMBER`, `DE_POSTAL_CODE`, `DE_PASSPORT_NUMBER`, and `DE_RESIDENCE_PERMIT_NUMBER` (unreleased). Complete PEM private-key blocks are also detected as `PRIVATE_KEY` by default (unreleased); see the [private-key reference](docs/reference/private-keys.mdx). Context-labeled `US_ROUTING_NUMBER` detection is also available (unreleased); see [the detector rules](docs/reference/us-routing-number.mdx). Context-labeled `NPI` detection is also available (unreleased); see [the detector rules](docs/reference/npi.mdx). Every binding returns the same finding information:
|
|
29
29
|
|
|
30
30
|
```text
|
|
31
31
|
entity type, matched text, byte range, code-point range,
|
|
@@ -79,7 +79,7 @@ full-match regex values:
|
|
|
79
79
|
}
|
|
80
80
|
```
|
|
81
81
|
|
|
82
|
-
`scan_and_transform` uses `{ scan?: { locale?: string }, transform: ... }` so
|
|
82
|
+
`scan_and_transform` uses `{ scan?: { locale?: string, detect_uuid?: boolean }, transform: ... }` so
|
|
83
83
|
detection settings remain separate from transformation policy.
|
|
84
84
|
|
|
85
85
|
## Packages
|
|
@@ -297,3 +297,40 @@ fixtures/ Shared conformance fixtures
|
|
|
297
297
|
|
|
298
298
|
[MIT](LICENSE)
|
|
299
299
|
|
|
300
|
+
## German structured identifiers (unreleased)
|
|
301
|
+
|
|
302
|
+
Pass `{"locale":"de"}` to text or structured scans. Trimmed, ASCII
|
|
303
|
+
case-insensitive `de`, `de-DE`, and `de_DE` activate all seven German detectors;
|
|
304
|
+
omitted locale and recognized `en-US`/`fr` aliases keep base detection only.
|
|
305
|
+
The 0.4.0 candidate rejects unsupported explicit locales.
|
|
306
|
+
|
|
307
|
+
```python
|
|
308
|
+
from datafog_core import scan_and_transform
|
|
309
|
+
|
|
310
|
+
result = scan_and_transform("IBAN DE44 5001 0517 5407 3249 31", {
|
|
311
|
+
"scan": {"locale": "de"},
|
|
312
|
+
"transform": {"default": {"strategy": "redact"}, "entities": ["DE_IBAN"]},
|
|
313
|
+
})
|
|
314
|
+
assert result.text == "IBAN [DE_IBAN]"
|
|
315
|
+
```
|
|
316
|
+
|
|
317
|
+
These are format/context detectors, not official identifier validators. IBAN
|
|
318
|
+
checksums and account existence are not checked. Digits are ASCII; permitted
|
|
319
|
+
internal separators are space, tab, NBSP and narrow NBSP at specified group
|
|
320
|
+
boundaries, never newlines. Returned text and offsets preserve the source.
|
|
321
|
+
Passport and residence-permit patterns are legacy heuristics with limited
|
|
322
|
+
coverage. See the [German entity reference](docs/reference/german-entities.mdx)
|
|
323
|
+
and [migration differences](docs/guides/migrating-from-datafog-python.mdx).
|
|
324
|
+
The Python 4.9 adapter requires a subsequently published compatible Core wheel;
|
|
325
|
+
this source change does not update its extra pin or publish a release.
|
|
326
|
+
|
|
327
|
+
## UUID identifiers (unreleased)
|
|
328
|
+
|
|
329
|
+
Canonical UUID detection is opt-in: pass `{"detect_uuid":true}` to text or
|
|
330
|
+
structured scans, independently of locale. It emits `UUID` findings for versions
|
|
331
|
+
1–8 with the IETF variant and original casing/ranges. UUID syntax does not imply
|
|
332
|
+
sensitivity. See the [UUID reference](docs/reference/uuid.mdx) for boundaries,
|
|
333
|
+
excluded sentinel forms and transformation examples.
|
|
334
|
+
|
|
335
|
+
The source candidate targets **0.4.0**; publication and downstream Python integration are separate release gates. See the [candidate release checklist](docs/releases/0-4-0.mdx), [runtime capabilities](docs/reference/capabilities.mdx), and [0.4.x compatibility policy](docs/reference/compatibility.mdx).
|
|
336
|
+
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Fast structured PII detection, implemented in Rust and exposed for Rust, Python, Node.js, and browsers.
|
|
4
4
|
|
|
5
|
-
It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. Every binding returns the same finding information:
|
|
5
|
+
It detects `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, and `ZIP_CODE`. The next release also detects `JWT` tokens by default; see the [JWT reference](docs/reference/jwt.mdx). With an explicit German locale, it also detects `DE_IBAN`, `DE_VAT_ID`, `DE_TAX_ID`, `DE_SOCIAL_SECURITY_NUMBER`, `DE_POSTAL_CODE`, `DE_PASSPORT_NUMBER`, and `DE_RESIDENCE_PERMIT_NUMBER` (unreleased). Complete PEM private-key blocks are also detected as `PRIVATE_KEY` by default (unreleased); see the [private-key reference](docs/reference/private-keys.mdx). Context-labeled `US_ROUTING_NUMBER` detection is also available (unreleased); see [the detector rules](docs/reference/us-routing-number.mdx). Context-labeled `NPI` detection is also available (unreleased); see [the detector rules](docs/reference/npi.mdx). Every binding returns the same finding information:
|
|
6
6
|
|
|
7
7
|
```text
|
|
8
8
|
entity type, matched text, byte range, code-point range,
|
|
@@ -56,7 +56,7 @@ full-match regex values:
|
|
|
56
56
|
}
|
|
57
57
|
```
|
|
58
58
|
|
|
59
|
-
`scan_and_transform` uses `{ scan?: { locale?: string }, transform: ... }` so
|
|
59
|
+
`scan_and_transform` uses `{ scan?: { locale?: string, detect_uuid?: boolean }, transform: ... }` so
|
|
60
60
|
detection settings remain separate from transformation policy.
|
|
61
61
|
|
|
62
62
|
## Packages
|
|
@@ -273,3 +273,40 @@ fixtures/ Shared conformance fixtures
|
|
|
273
273
|
## License
|
|
274
274
|
|
|
275
275
|
[MIT](LICENSE)
|
|
276
|
+
|
|
277
|
+
## German structured identifiers (unreleased)
|
|
278
|
+
|
|
279
|
+
Pass `{"locale":"de"}` to text or structured scans. Trimmed, ASCII
|
|
280
|
+
case-insensitive `de`, `de-DE`, and `de_DE` activate all seven German detectors;
|
|
281
|
+
omitted locale and recognized `en-US`/`fr` aliases keep base detection only.
|
|
282
|
+
The 0.4.0 candidate rejects unsupported explicit locales.
|
|
283
|
+
|
|
284
|
+
```python
|
|
285
|
+
from datafog_core import scan_and_transform
|
|
286
|
+
|
|
287
|
+
result = scan_and_transform("IBAN DE44 5001 0517 5407 3249 31", {
|
|
288
|
+
"scan": {"locale": "de"},
|
|
289
|
+
"transform": {"default": {"strategy": "redact"}, "entities": ["DE_IBAN"]},
|
|
290
|
+
})
|
|
291
|
+
assert result.text == "IBAN [DE_IBAN]"
|
|
292
|
+
```
|
|
293
|
+
|
|
294
|
+
These are format/context detectors, not official identifier validators. IBAN
|
|
295
|
+
checksums and account existence are not checked. Digits are ASCII; permitted
|
|
296
|
+
internal separators are space, tab, NBSP and narrow NBSP at specified group
|
|
297
|
+
boundaries, never newlines. Returned text and offsets preserve the source.
|
|
298
|
+
Passport and residence-permit patterns are legacy heuristics with limited
|
|
299
|
+
coverage. See the [German entity reference](docs/reference/german-entities.mdx)
|
|
300
|
+
and [migration differences](docs/guides/migrating-from-datafog-python.mdx).
|
|
301
|
+
The Python 4.9 adapter requires a subsequently published compatible Core wheel;
|
|
302
|
+
this source change does not update its extra pin or publish a release.
|
|
303
|
+
|
|
304
|
+
## UUID identifiers (unreleased)
|
|
305
|
+
|
|
306
|
+
Canonical UUID detection is opt-in: pass `{"detect_uuid":true}` to text or
|
|
307
|
+
structured scans, independently of locale. It emits `UUID` findings for versions
|
|
308
|
+
1–8 with the IETF variant and original casing/ranges. UUID syntax does not imply
|
|
309
|
+
sensitivity. See the [UUID reference](docs/reference/uuid.mdx) for boundaries,
|
|
310
|
+
excluded sentinel forms and transformation examples.
|
|
311
|
+
|
|
312
|
+
The source candidate targets **0.4.0**; publication and downstream Python integration are separate release gates. See the [candidate release checklist](docs/releases/0-4-0.mdx), [runtime capabilities](docs/reference/capabilities.mdx), and [0.4.x compatibility policy](docs/reference/compatibility.mdx).
|
|
@@ -956,6 +956,14 @@ fn privacy_error(py: Python<'_>, error: core::PrivacyError) -> PyErr {
|
|
|
956
956
|
exception
|
|
957
957
|
}
|
|
958
958
|
|
|
959
|
+
/// Return an independent, JSON-compatible snapshot of the Core capability contract.
|
|
960
|
+
#[pyfunction]
|
|
961
|
+
fn capabilities(py: Python<'_>) -> PyResult<Py<PyAny>> {
|
|
962
|
+
let json = serde_json::to_string(&core::capabilities())
|
|
963
|
+
.map_err(|_| internal_error(py, "failed to serialize Core capabilities"))?;
|
|
964
|
+
Ok(py.import("json")?.call_method1("loads", (json,))?.unbind())
|
|
965
|
+
}
|
|
966
|
+
|
|
959
967
|
/// Scan text for supported PII findings.
|
|
960
968
|
#[pyfunction]
|
|
961
969
|
#[pyo3(signature = (text, config=None))]
|
|
@@ -1300,6 +1308,7 @@ fn datafog_core(module: &Bound<'_, PyModule>) -> PyResult<()> {
|
|
|
1300
1308
|
module.add_class::<Restoration>()?;
|
|
1301
1309
|
module.add_class::<RestoreResult>()?;
|
|
1302
1310
|
module.add_class::<PrivacyManager>()?;
|
|
1311
|
+
module.add_function(wrap_pyfunction!(capabilities, module)?)?;
|
|
1303
1312
|
module.add_function(wrap_pyfunction!(scan, module)?)?;
|
|
1304
1313
|
module.add_function(wrap_pyfunction!(transform, module)?)?;
|
|
1305
1314
|
module.add_function(wrap_pyfunction!(scan_and_transform, module)?)?;
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Capability discovery contract against the installed, compiled Core wheel."""
|
|
2
|
+
import json
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
import datafog_core as api
|
|
6
|
+
|
|
7
|
+
SUPPORTED = sorted([
|
|
8
|
+
"CREDIT_CARD", "DATE", "DE_IBAN", "DE_PASSPORT_NUMBER", "DE_POSTAL_CODE",
|
|
9
|
+
"DE_RESIDENCE_PERMIT_NUMBER", "DE_SOCIAL_SECURITY_NUMBER", "DE_TAX_ID",
|
|
10
|
+
"DE_VAT_ID", "EMAIL", "IP_ADDRESS", "JWT", "NPI", "PERSON", "PHONE",
|
|
11
|
+
"PRIVATE_KEY", "SSN", "US_ROUTING_NUMBER", "UUID", "ZIP_CODE",
|
|
12
|
+
])
|
|
13
|
+
GERMAN = [entity for entity in SUPPORTED if entity.startswith("DE_")]
|
|
14
|
+
DEFAULT = [entity for entity in SUPPORTED if entity not in GERMAN + ["PERSON", "UUID"]]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def verify():
|
|
18
|
+
assert callable(api.capabilities)
|
|
19
|
+
try:
|
|
20
|
+
api.capabilities({})
|
|
21
|
+
except TypeError:
|
|
22
|
+
pass
|
|
23
|
+
else:
|
|
24
|
+
raise AssertionError("capabilities accepted arguments")
|
|
25
|
+
snapshot = api.capabilities()
|
|
26
|
+
assert type(snapshot) is dict
|
|
27
|
+
assert snapshot["contract_version"] == 1
|
|
28
|
+
assert json.loads(json.dumps(snapshot)) == snapshot
|
|
29
|
+
assert snapshot["supported_entities"] == SUPPORTED
|
|
30
|
+
assert snapshot["default_entities"] == DEFAULT
|
|
31
|
+
assert sorted(snapshot["entities"]) == SUPPORTED
|
|
32
|
+
for key in ("supported_entities", "default_entities"):
|
|
33
|
+
assert snapshot[key] == sorted(set(snapshot[key]))
|
|
34
|
+
assert set(snapshot["locales"]) == {"de", "de-DE", "de_DE", "en-US", "fr"}
|
|
35
|
+
for locale, metadata in snapshot["locales"].items():
|
|
36
|
+
assert metadata["enabled_entities"] == (GERMAN if locale.startswith("de") else [])
|
|
37
|
+
for label, entity in snapshot["entities"].items():
|
|
38
|
+
assert entity["scopes"] == (["structured"] if label == "PERSON" else ["structured", "text"])
|
|
39
|
+
kind = "locale" if label in GERMAN else "config" if label == "UUID" else "structured" if label == "PERSON" else "default"
|
|
40
|
+
assert entity["activation"]["kind"] == kind
|
|
41
|
+
uuid_config = snapshot["entities"]["UUID"]["activation"]["scan_config"]
|
|
42
|
+
assert uuid_config == {"detect_uuid": True}
|
|
43
|
+
text = "👋 café é 550e8400-e29b-41d4-a716-446655440000"
|
|
44
|
+
assert not any(f.entity_type == "UUID" for f in api.scan(text))
|
|
45
|
+
uuid_findings = [f for f in api.scan(text, uuid_config) if f.entity_type == "UUID"]
|
|
46
|
+
assert len(uuid_findings) == 1
|
|
47
|
+
finding = uuid_findings[0]
|
|
48
|
+
assert text[finding.codepoint_range.start:finding.codepoint_range.end] == finding.matched_text
|
|
49
|
+
assert text.encode()[finding.byte_range.start:finding.byte_range.end].decode() == finding.matched_text
|
|
50
|
+
samples = {}
|
|
51
|
+
for line in (Path(__file__).resolve().parents[3] / "fixtures/german.jsonl").read_text().splitlines():
|
|
52
|
+
row = json.loads(line)
|
|
53
|
+
for expected in row["entities"]:
|
|
54
|
+
samples.setdefault(expected["label"], row["text"])
|
|
55
|
+
for locale in ("de", "de-DE", "de_DE", "DE-de", " De "):
|
|
56
|
+
for label, text in samples.items():
|
|
57
|
+
assert any(f.entity_type == label for f in api.scan(text, {"locale": locale})), (locale, label)
|
|
58
|
+
for label, text in samples.items():
|
|
59
|
+
activation = snapshot["entities"][label]["activation"]["scan_config"]
|
|
60
|
+
assert any(f.entity_type == label for f in api.scan(text, activation))
|
|
61
|
+
for locale in ("en-US", "fr"):
|
|
62
|
+
assert api.scan("jane@example.com", {"locale": locale}) == api.scan("jane@example.com")
|
|
63
|
+
for call, path in [
|
|
64
|
+
(lambda: api.scan("", {"locale": "zz-ZZ"}), "/locale"),
|
|
65
|
+
(lambda: api.scan_structured({}, {"locale": "zz-ZZ"}), "/locale"),
|
|
66
|
+
(lambda: api.scan_and_transform("", {"scan": {"locale": "zz-ZZ"}, "transform": {"default": {"strategy": "redact"}}}), "/scan/locale"),
|
|
67
|
+
(lambda: api.scan_and_transform_structured({}, {"scan": {"locale": "zz-ZZ"}, "transform": {"default": {"strategy": "redact"}}}), "/scan/locale"),
|
|
68
|
+
]:
|
|
69
|
+
try:
|
|
70
|
+
call()
|
|
71
|
+
except api.DataFogConfigurationError as error:
|
|
72
|
+
assert error.code == "invalid_configuration"
|
|
73
|
+
assert error.reason == "invalid_value"
|
|
74
|
+
assert error.path == path
|
|
75
|
+
else:
|
|
76
|
+
raise AssertionError("unsupported locale accepted")
|
|
77
|
+
assert not any(f.entity_type == "PERSON" for f in api.scan("Jane Doe"))
|
|
78
|
+
assert any(f.finding.entity_type == "PERSON" for f in api.scan_structured({"full_name": "Jane Doe"}).findings)
|
|
79
|
+
mutated = api.capabilities()
|
|
80
|
+
assert mutated is not snapshot
|
|
81
|
+
mutated["supported_entities"].clear()
|
|
82
|
+
mutated["default_entities"].append("FAKE")
|
|
83
|
+
mutated["locales"]["de"]["enabled_entities"].clear()
|
|
84
|
+
mutated["entities"]["EMAIL"]["scopes"].clear()
|
|
85
|
+
mutated["entities"]["UUID"]["activation"]["scan_config"]["detect_uuid"] = False
|
|
86
|
+
assert api.capabilities() == snapshot
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
"""Shared German fixtures against the installed wheel, including provider round trips."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import datafog_core as api
|
|
6
|
+
|
|
7
|
+
RECORDS = [
|
|
8
|
+
json.loads(line)
|
|
9
|
+
for line in (Path(__file__).resolve().parents[3] / "fixtures/german.jsonl")
|
|
10
|
+
.read_text()
|
|
11
|
+
.splitlines()
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def project(text, findings):
|
|
16
|
+
result = []
|
|
17
|
+
for f in findings:
|
|
18
|
+
if not f.entity_type.startswith("DE_"):
|
|
19
|
+
continue
|
|
20
|
+
assert (
|
|
21
|
+
text.encode()[f.byte_range.start : f.byte_range.end].decode()
|
|
22
|
+
== f.matched_text
|
|
23
|
+
)
|
|
24
|
+
assert text[f.codepoint_range.start : f.codepoint_range.end] == f.matched_text
|
|
25
|
+
assert f.confidence is None
|
|
26
|
+
assert f.detector_name == "datafog-core/" + f.entity_type.lower().replace(
|
|
27
|
+
"_", "-"
|
|
28
|
+
)
|
|
29
|
+
assert f.detector_version
|
|
30
|
+
result.append(
|
|
31
|
+
dict(
|
|
32
|
+
label=f.entity_type,
|
|
33
|
+
text=f.matched_text,
|
|
34
|
+
start=f.codepoint_range.start,
|
|
35
|
+
end=f.codepoint_range.end,
|
|
36
|
+
)
|
|
37
|
+
)
|
|
38
|
+
return result
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def verify():
|
|
42
|
+
for row in RECORDS:
|
|
43
|
+
findings = api.scan(row["text"], row["config"])
|
|
44
|
+
assert project(row["text"], findings) == row["entities"], row["id"]
|
|
45
|
+
data = {"a/b": row["text"], "array": [row["text"]], "z": row["text"]}
|
|
46
|
+
located = api.scan_structured(data, row["config"]).findings
|
|
47
|
+
for path in ["/a~1b", "/array/0", "/z"]:
|
|
48
|
+
assert (
|
|
49
|
+
project(row["text"], [f.finding for f in located if f.path == path])
|
|
50
|
+
== row["entities"]
|
|
51
|
+
), row["id"]
|
|
52
|
+
for case in row.get("transforms", []):
|
|
53
|
+
result = api.scan_and_transform(
|
|
54
|
+
row["text"], {"scan": row["config"], "transform": case["config"]}
|
|
55
|
+
)
|
|
56
|
+
assert result.text == case["text"], row["id"]
|
|
57
|
+
assert api.transform(row["text"], findings, case["config"]) == result
|
|
58
|
+
for t in result.transformations:
|
|
59
|
+
assert (
|
|
60
|
+
result.text.encode()[
|
|
61
|
+
t.output_byte_range.start : t.output_byte_range.end
|
|
62
|
+
].decode()
|
|
63
|
+
== t.replacement
|
|
64
|
+
)
|
|
65
|
+
assert (
|
|
66
|
+
result.text[
|
|
67
|
+
t.output_codepoint_range.start : t.output_codepoint_range.end
|
|
68
|
+
]
|
|
69
|
+
== t.replacement
|
|
70
|
+
)
|
|
71
|
+
assert any(
|
|
72
|
+
f.byte_range == t.source_byte_range
|
|
73
|
+
and f.entity_type == t.entity_type
|
|
74
|
+
for f in findings
|
|
75
|
+
)
|
|
76
|
+
assert not hasattr(t, "matched_text") and not hasattr(t, "finding")
|
|
77
|
+
structured = api.scan_and_transform_structured(
|
|
78
|
+
data, {"scan": row["config"], "transform": case["config"]}
|
|
79
|
+
)
|
|
80
|
+
assert structured.data == {
|
|
81
|
+
"a/b": case["text"],
|
|
82
|
+
"array": [case["text"]],
|
|
83
|
+
"z": case["text"],
|
|
84
|
+
}
|
|
85
|
+
explicit = api.transform_structured(data, located, case["config"])
|
|
86
|
+
assert explicit.data == structured.data
|
|
87
|
+
assert [(t.path, t.transformation) for t in explicit.transformations] == [
|
|
88
|
+
(t.path, t.transformation) for t in structured.transformations
|
|
89
|
+
]
|
|
90
|
+
split = {
|
|
91
|
+
"Steuer-ID": "12345678901",
|
|
92
|
+
"a": "Steuer-ID",
|
|
93
|
+
"b": "12345678901",
|
|
94
|
+
"c": ["Passport", "C12345678"],
|
|
95
|
+
"d": "DE4450010517",
|
|
96
|
+
"e": "5407324931",
|
|
97
|
+
}
|
|
98
|
+
assert not any(
|
|
99
|
+
f.finding.entity_type.startswith("DE_")
|
|
100
|
+
for f in api.scan_structured(split, {"locale": "de"}).findings
|
|
101
|
+
)
|
|
102
|
+
for config in [
|
|
103
|
+
{"locale": ""},
|
|
104
|
+
{"locale": " \t"},
|
|
105
|
+
{"locale": None},
|
|
106
|
+
{"locale": 1},
|
|
107
|
+
{"locales": ["de"]},
|
|
108
|
+
]:
|
|
109
|
+
for call in [
|
|
110
|
+
lambda: api.scan("", config),
|
|
111
|
+
lambda: api.scan_structured({}, config),
|
|
112
|
+
]:
|
|
113
|
+
try:
|
|
114
|
+
call()
|
|
115
|
+
except api.DataFogConfigurationError:
|
|
116
|
+
pass
|
|
117
|
+
else:
|
|
118
|
+
raise AssertionError("malformed config accepted")
|
|
119
|
+
assert (
|
|
120
|
+
api.scan_and_transform(
|
|
121
|
+
"DE44500105175407324931",
|
|
122
|
+
{"transform": {"default": {"strategy": "redact"}, "entities": ["DE_IBAN"]}},
|
|
123
|
+
).text
|
|
124
|
+
== "DE44500105175407324931"
|
|
125
|
+
)
|
|
126
|
+
for text, label, generic, expected in [
|
|
127
|
+
("DE44\t4111111111111111\t11", "DE_IBAN", "CREDIT_CARD", "[DE_IBAN]"),
|
|
128
|
+
("DE 123456789", "DE_VAT_ID", "SSN", "[DE_VAT_ID]"),
|
|
129
|
+
("Steuer-ID 12345678901", "DE_TAX_ID", "PHONE", "Steuer-ID [DE_TAX_ID]"),
|
|
130
|
+
("DE-10115", "DE_POSTAL_CODE", "ZIP_CODE", "[DE_POSTAL_CODE]"),
|
|
131
|
+
]:
|
|
132
|
+
findings = api.scan(text, {"locale": "de"})
|
|
133
|
+
assert all(
|
|
134
|
+
any(f.entity_type == entity for f in findings)
|
|
135
|
+
for entity in [label, generic]
|
|
136
|
+
)
|
|
137
|
+
assert (
|
|
138
|
+
api.scan_and_transform(
|
|
139
|
+
text,
|
|
140
|
+
{
|
|
141
|
+
"scan": {"locale": "de"},
|
|
142
|
+
"transform": {"default": {"strategy": "redact"}},
|
|
143
|
+
},
|
|
144
|
+
).text
|
|
145
|
+
== expected
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
async def verify_providers(manager, token_manager, context):
|
|
150
|
+
for row in RECORDS:
|
|
151
|
+
if not row.get("sample"):
|
|
152
|
+
continue
|
|
153
|
+
entities = list(dict.fromkeys(e["label"] for e in row["entities"]))
|
|
154
|
+
pseudonyms = await manager.scan_and_transform(
|
|
155
|
+
row["text"],
|
|
156
|
+
{
|
|
157
|
+
"scan": row["config"],
|
|
158
|
+
"transform": {
|
|
159
|
+
"default": {"strategy": "pseudonymize", "key_ref": "german"},
|
|
160
|
+
"entities": entities,
|
|
161
|
+
},
|
|
162
|
+
},
|
|
163
|
+
)
|
|
164
|
+
assert (
|
|
165
|
+
len(pseudonyms.transformations) == len(row["entities"])
|
|
166
|
+
and pseudonyms.text != row["text"]
|
|
167
|
+
)
|
|
168
|
+
tokens = await token_manager.scan_and_transform(
|
|
169
|
+
row["text"],
|
|
170
|
+
{
|
|
171
|
+
"scan": row["config"],
|
|
172
|
+
"transform": {
|
|
173
|
+
"default": {"strategy": "tokenize", "token_ref": "german"},
|
|
174
|
+
"entities": entities,
|
|
175
|
+
},
|
|
176
|
+
},
|
|
177
|
+
context,
|
|
178
|
+
)
|
|
179
|
+
assert len(tokens.transformations) == len(row["entities"])
|
|
180
|
+
assert (await token_manager.restore(tokens.text, context)).text == row["text"]
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Shared JWT fixtures against the installed wheel, including provider round trips."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import datafog_core as api
|
|
6
|
+
|
|
7
|
+
RECORDS = [
|
|
8
|
+
json.loads(line)
|
|
9
|
+
for line in (Path(__file__).resolve().parents[3] / "fixtures/jwt.jsonl")
|
|
10
|
+
.read_text()
|
|
11
|
+
.splitlines()
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def project(text, findings):
|
|
16
|
+
result = []
|
|
17
|
+
for f in findings:
|
|
18
|
+
if not f.entity_type == "JWT":
|
|
19
|
+
continue
|
|
20
|
+
assert (
|
|
21
|
+
text.encode()[f.byte_range.start : f.byte_range.end].decode()
|
|
22
|
+
== f.matched_text
|
|
23
|
+
)
|
|
24
|
+
assert text[f.codepoint_range.start : f.codepoint_range.end] == f.matched_text
|
|
25
|
+
assert f.confidence is None
|
|
26
|
+
assert f.detector_name == "datafog-core/" + f.entity_type.lower().replace(
|
|
27
|
+
"_", "-"
|
|
28
|
+
)
|
|
29
|
+
assert f.detector_version
|
|
30
|
+
result.append(
|
|
31
|
+
dict(
|
|
32
|
+
label=f.entity_type,
|
|
33
|
+
text=f.matched_text,
|
|
34
|
+
start=f.codepoint_range.start,
|
|
35
|
+
end=f.codepoint_range.end,
|
|
36
|
+
)
|
|
37
|
+
)
|
|
38
|
+
return result
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def verify():
|
|
42
|
+
for row in RECORDS:
|
|
43
|
+
findings = api.scan(row["text"], row["config"])
|
|
44
|
+
if "overlap" in row:
|
|
45
|
+
assert any(f.entity_type == row["overlap"] for f in findings)
|
|
46
|
+
assert project(row["text"], findings) == row["entities"], row["id"]
|
|
47
|
+
data = {"a/b": row["text"], "array": [row["text"]], "z": row["text"]}
|
|
48
|
+
located = api.scan_structured(data, row["config"]).findings
|
|
49
|
+
for path in ["/a~1b", "/array/0", "/z"]:
|
|
50
|
+
assert (
|
|
51
|
+
project(row["text"], [f.finding for f in located if f.path == path])
|
|
52
|
+
== row["entities"]
|
|
53
|
+
), row["id"]
|
|
54
|
+
for case in row.get("transforms", []):
|
|
55
|
+
result = api.scan_and_transform(
|
|
56
|
+
row["text"], {"scan": row["config"], "transform": case["config"]}
|
|
57
|
+
)
|
|
58
|
+
assert result.text == case["text"], row["id"]
|
|
59
|
+
assert api.transform(row["text"], findings, case["config"]) == result
|
|
60
|
+
for t in result.transformations:
|
|
61
|
+
assert (
|
|
62
|
+
result.text.encode()[
|
|
63
|
+
t.output_byte_range.start : t.output_byte_range.end
|
|
64
|
+
].decode()
|
|
65
|
+
== t.replacement
|
|
66
|
+
)
|
|
67
|
+
assert (
|
|
68
|
+
result.text[
|
|
69
|
+
t.output_codepoint_range.start : t.output_codepoint_range.end
|
|
70
|
+
]
|
|
71
|
+
== t.replacement
|
|
72
|
+
)
|
|
73
|
+
assert any(
|
|
74
|
+
f.byte_range == t.source_byte_range
|
|
75
|
+
and f.entity_type == t.entity_type
|
|
76
|
+
for f in findings
|
|
77
|
+
)
|
|
78
|
+
assert not hasattr(t, "matched_text") and not hasattr(t, "finding")
|
|
79
|
+
structured = api.scan_and_transform_structured(
|
|
80
|
+
data, {"scan": row["config"], "transform": case["config"]}
|
|
81
|
+
)
|
|
82
|
+
assert structured.data == {
|
|
83
|
+
"a/b": case["text"],
|
|
84
|
+
"array": [case["text"]],
|
|
85
|
+
"z": case["text"],
|
|
86
|
+
}
|
|
87
|
+
explicit = api.transform_structured(data, located, case["config"])
|
|
88
|
+
assert explicit.data == structured.data
|
|
89
|
+
assert [(t.path, t.transformation) for t in explicit.transformations] == [
|
|
90
|
+
(t.path, t.transformation) for t in structured.transformations
|
|
91
|
+
]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
async def verify_providers(manager, token_manager, context):
|
|
95
|
+
for row in RECORDS:
|
|
96
|
+
if not row.get("sample"):
|
|
97
|
+
continue
|
|
98
|
+
entities = list(dict.fromkeys(e["label"] for e in row["entities"]))
|
|
99
|
+
pseudonyms = await manager.scan_and_transform(
|
|
100
|
+
row["text"],
|
|
101
|
+
{
|
|
102
|
+
"scan": row["config"],
|
|
103
|
+
"transform": {
|
|
104
|
+
"default": {"strategy": "pseudonymize", "key_ref": "jwt"},
|
|
105
|
+
"entities": entities,
|
|
106
|
+
},
|
|
107
|
+
},
|
|
108
|
+
)
|
|
109
|
+
assert (
|
|
110
|
+
len(pseudonyms.transformations) == len(row["entities"])
|
|
111
|
+
and pseudonyms.text != row["text"]
|
|
112
|
+
)
|
|
113
|
+
tokens = await token_manager.scan_and_transform(
|
|
114
|
+
row["text"],
|
|
115
|
+
{
|
|
116
|
+
"scan": row["config"],
|
|
117
|
+
"transform": {
|
|
118
|
+
"default": {"strategy": "tokenize", "token_ref": "jwt"},
|
|
119
|
+
"entities": entities,
|
|
120
|
+
},
|
|
121
|
+
},
|
|
122
|
+
context,
|
|
123
|
+
)
|
|
124
|
+
assert len(tokens.transformations) == len(row["entities"])
|
|
125
|
+
assert (await token_manager.restore(tokens.text, context)).text == row["text"]
|