datafog-core 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {datafog_core-0.2.0 → datafog_core-0.3.0}/Cargo.lock +4 -4
- {datafog_core-0.2.0 → datafog_core-0.3.0}/PKG-INFO +29 -1
- {datafog_core-0.2.0/crates/core → datafog_core-0.3.0}/README.md +28 -0
- {datafog_core-0.2.0 → datafog_core-0.3.0}/bindings/python/Cargo.toml +1 -1
- {datafog_core-0.2.0 → datafog_core-0.3.0}/bindings/python/src/lib.rs +411 -8
- {datafog_core-0.2.0 → datafog_core-0.3.0}/bindings/python/tests/test_installed.py +72 -0
- {datafog_core-0.2.0 → datafog_core-0.3.0}/crates/core/Cargo.toml +1 -1
- {datafog_core-0.2.0 → datafog_core-0.3.0/crates/core}/README.md +28 -0
- datafog_core-0.3.0/crates/core/examples/scan_benchmark.rs +88 -0
- {datafog_core-0.2.0 → datafog_core-0.3.0}/crates/core/src/lib.rs +118 -37
- datafog_core-0.3.0/crates/core/src/offsets.rs +203 -0
- datafog_core-0.3.0/crates/core/src/selection_tests.rs +320 -0
- datafog_core-0.3.0/crates/core/src/structured.rs +1039 -0
- {datafog_core-0.2.0 → datafog_core-0.3.0}/pyproject.toml +1 -1
- {datafog_core-0.2.0 → datafog_core-0.3.0}/Cargo.toml +0 -0
|
@@ -80,7 +80,7 @@ checksum = "914a755b7c2d4af2bdcff7ce1739e2db9a1b81a9b07123d8015786ae03c0980d"
|
|
|
80
80
|
|
|
81
81
|
[[package]]
|
|
82
82
|
name = "datafog-core"
|
|
83
|
-
version = "0.
|
|
83
|
+
version = "0.3.0"
|
|
84
84
|
dependencies = [
|
|
85
85
|
"base64",
|
|
86
86
|
"futures",
|
|
@@ -94,7 +94,7 @@ dependencies = [
|
|
|
94
94
|
|
|
95
95
|
[[package]]
|
|
96
96
|
name = "datafog-core-python"
|
|
97
|
-
version = "0.
|
|
97
|
+
version = "0.3.0"
|
|
98
98
|
dependencies = [
|
|
99
99
|
"datafog-core",
|
|
100
100
|
"pyo3",
|
|
@@ -104,7 +104,7 @@ dependencies = [
|
|
|
104
104
|
|
|
105
105
|
[[package]]
|
|
106
106
|
name = "datafog-node"
|
|
107
|
-
version = "0.
|
|
107
|
+
version = "0.3.0"
|
|
108
108
|
dependencies = [
|
|
109
109
|
"datafog-core",
|
|
110
110
|
"napi",
|
|
@@ -115,7 +115,7 @@ dependencies = [
|
|
|
115
115
|
|
|
116
116
|
[[package]]
|
|
117
117
|
name = "datafog-wasm"
|
|
118
|
-
version = "0.
|
|
118
|
+
version = "0.3.0"
|
|
119
119
|
dependencies = [
|
|
120
120
|
"datafog-core",
|
|
121
121
|
"serde",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: datafog-core
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Classifier: Development Status :: 3 - Alpha
|
|
5
5
|
Classifier: License :: OSI Approved :: MIT License
|
|
6
6
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -32,6 +32,12 @@ entity type, matched text, byte range, code-point range,
|
|
|
32
32
|
optional confidence, detector name, optional detector version
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
+
Structured JSON scanning additionally discovers `PERSON` from documented name-field
|
|
36
|
+
aliases or explicit JSON Pointer mappings. It scans every string value with the
|
|
37
|
+
existing detectors and returns field paths plus string-local findings. No model
|
|
38
|
+
or dictionary download is needed. See [person-field discovery](docs/guides/person-discovery.mdx)
|
|
39
|
+
for `scan_structured` / `scanStructured` and structured transformation APIs.
|
|
40
|
+
|
|
35
41
|
Both ranges use zero-based, end-exclusive offsets. The byte range addresses the
|
|
36
42
|
UTF-8 input; the code-point range addresses Unicode scalar values. Rule-based
|
|
37
43
|
detectors currently report no confidence score. Node.js and browser WASM also
|
|
@@ -85,6 +91,28 @@ detection settings remain separate from transformation policy.
|
|
|
85
91
|
| Node.js | `@datafog/node` | `@datafog/node` | Published |
|
|
86
92
|
| Browser/WASM | `@datafog/wasm` | `@datafog/wasm` | Published |
|
|
87
93
|
|
|
94
|
+
## Migrating from DataFog Python
|
|
95
|
+
|
|
96
|
+
DataFog Core is a separate distribution and canonical API, not a drop-in
|
|
97
|
+
replacement for the established `datafog` Python package.
|
|
98
|
+
|
|
99
|
+
| DataFog Python 4.8.x | DataFog Core 0.3.x |
|
|
100
|
+
| --- | --- |
|
|
101
|
+
| `pip install datafog` | `pip install datafog-core` |
|
|
102
|
+
| `from datafog.engine import ...` | `from datafog_core import ...` |
|
|
103
|
+
| `scan(...).entities` | `scan(...)` returns `list[Finding]` |
|
|
104
|
+
| `scan_and_redact(...)` | `scan_and_transform(...)` |
|
|
105
|
+
| `result.redacted_text` | `result.text` |
|
|
106
|
+
| `Entity.type`, `.text`, `.start`, `.end` | `Finding.entity_type`, `.matched_text`, `.byte_range`, `.codepoint_range` |
|
|
107
|
+
|
|
108
|
+
Do not mechanically rename legacy `token` to Core `tokenize`: Core tokenization
|
|
109
|
+
is provider-backed and reversible. For non-reversible output, use `redact`,
|
|
110
|
+
`mask`, or `remove`; use keyed `pseudonymize` when stable linkage is required.
|
|
111
|
+
|
|
112
|
+
See the dedicated [DataFog Python migration
|
|
113
|
+
guide](docs/guides/migrating-from-datafog-python.mdx) for the API mapping,
|
|
114
|
+
strategy differences, entity names, range semantics, and migration checklist.
|
|
115
|
+
|
|
88
116
|
## Quick start
|
|
89
117
|
|
|
90
118
|
### Rust
|
|
@@ -9,6 +9,12 @@ entity type, matched text, byte range, code-point range,
|
|
|
9
9
|
optional confidence, detector name, optional detector version
|
|
10
10
|
```
|
|
11
11
|
|
|
12
|
+
Structured JSON scanning additionally discovers `PERSON` from documented name-field
|
|
13
|
+
aliases or explicit JSON Pointer mappings. It scans every string value with the
|
|
14
|
+
existing detectors and returns field paths plus string-local findings. No model
|
|
15
|
+
or dictionary download is needed. See [person-field discovery](docs/guides/person-discovery.mdx)
|
|
16
|
+
for `scan_structured` / `scanStructured` and structured transformation APIs.
|
|
17
|
+
|
|
12
18
|
Both ranges use zero-based, end-exclusive offsets. The byte range addresses the
|
|
13
19
|
UTF-8 input; the code-point range addresses Unicode scalar values. Rule-based
|
|
14
20
|
detectors currently report no confidence score. Node.js and browser WASM also
|
|
@@ -62,6 +68,28 @@ detection settings remain separate from transformation policy.
|
|
|
62
68
|
| Node.js | `@datafog/node` | `@datafog/node` | Published |
|
|
63
69
|
| Browser/WASM | `@datafog/wasm` | `@datafog/wasm` | Published |
|
|
64
70
|
|
|
71
|
+
## Migrating from DataFog Python
|
|
72
|
+
|
|
73
|
+
DataFog Core is a separate distribution and canonical API, not a drop-in
|
|
74
|
+
replacement for the established `datafog` Python package.
|
|
75
|
+
|
|
76
|
+
| DataFog Python 4.8.x | DataFog Core 0.3.x |
|
|
77
|
+
| --- | --- |
|
|
78
|
+
| `pip install datafog` | `pip install datafog-core` |
|
|
79
|
+
| `from datafog.engine import ...` | `from datafog_core import ...` |
|
|
80
|
+
| `scan(...).entities` | `scan(...)` returns `list[Finding]` |
|
|
81
|
+
| `scan_and_redact(...)` | `scan_and_transform(...)` |
|
|
82
|
+
| `result.redacted_text` | `result.text` |
|
|
83
|
+
| `Entity.type`, `.text`, `.start`, `.end` | `Finding.entity_type`, `.matched_text`, `.byte_range`, `.codepoint_range` |
|
|
84
|
+
|
|
85
|
+
Do not mechanically rename legacy `token` to Core `tokenize`: Core tokenization
|
|
86
|
+
is provider-backed and reversible. For non-reversible output, use `redact`,
|
|
87
|
+
`mask`, or `remove`; use keyed `pseudonymize` when stable linkage is required.
|
|
88
|
+
|
|
89
|
+
See the dedicated [DataFog Python migration
|
|
90
|
+
guide](docs/guides/migrating-from-datafog-python.mdx) for the API mapping,
|
|
91
|
+
strategy differences, entity names, range semantics, and migration checklist.
|
|
92
|
+
|
|
65
93
|
## Quick start
|
|
66
94
|
|
|
67
95
|
### Rust
|
|
@@ -320,14 +320,7 @@ impl From<core::RestoreResult> for RestoreResult {
|
|
|
320
320
|
restorations: result
|
|
321
321
|
.restorations
|
|
322
322
|
.into_iter()
|
|
323
|
-
.map(
|
|
324
|
-
source_byte_range: record.source_byte_range.into(),
|
|
325
|
-
source_codepoint_range: record.source_codepoint_range.into(),
|
|
326
|
-
output_byte_range: record.output_byte_range.into(),
|
|
327
|
-
output_codepoint_range: record.output_codepoint_range.into(),
|
|
328
|
-
token_ref: record.token_ref,
|
|
329
|
-
resolved_token_version: record.resolved_token_version,
|
|
330
|
-
})
|
|
323
|
+
.map(Restoration::from)
|
|
331
324
|
.collect(),
|
|
332
325
|
}
|
|
333
326
|
}
|
|
@@ -571,6 +564,124 @@ struct PrivacyManager {
|
|
|
571
564
|
|
|
572
565
|
#[pymethods]
|
|
573
566
|
impl PrivacyManager {
|
|
567
|
+
#[pyo3(signature = (data, findings, config, context=None))]
|
|
568
|
+
fn transform_structured<'py>(
|
|
569
|
+
&self,
|
|
570
|
+
py: Python<'py>,
|
|
571
|
+
data: Py<PyAny>,
|
|
572
|
+
findings: Vec<Py<StructuredFinding>>,
|
|
573
|
+
config: Py<PyAny>,
|
|
574
|
+
context: Option<Py<PyAny>>,
|
|
575
|
+
) -> PyResult<Bound<'py, PyAny>> {
|
|
576
|
+
let data = structured_data(py, data.bind(py))?;
|
|
577
|
+
let config_value = structured_options(py, config.bind(py), "")?;
|
|
578
|
+
let config = core::parse_transformation_config(&config_value)
|
|
579
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
580
|
+
let findings = findings
|
|
581
|
+
.iter()
|
|
582
|
+
.map(|finding| finding.bind(py).borrow().to_core())
|
|
583
|
+
.collect::<Vec<_>>();
|
|
584
|
+
let context = context
|
|
585
|
+
.map(|context| structured_options(py, context.bind(py), ""))
|
|
586
|
+
.transpose()?
|
|
587
|
+
.map(|value| core::parse_privacy_context(&value))
|
|
588
|
+
.transpose()
|
|
589
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
590
|
+
let key_provider = self
|
|
591
|
+
.key_provider
|
|
592
|
+
.as_ref()
|
|
593
|
+
.map(|provider| provider.clone_ref(py));
|
|
594
|
+
let token_provider = self
|
|
595
|
+
.token_provider
|
|
596
|
+
.as_ref()
|
|
597
|
+
.map(|provider| provider.clone_ref(py));
|
|
598
|
+
pyo3_async_runtimes::tokio::future_into_py(py, async move {
|
|
599
|
+
let manager = core::PrivacyManager::new(PythonKeyProvider {
|
|
600
|
+
provider: key_provider,
|
|
601
|
+
})
|
|
602
|
+
.with_token_provider(PythonTokenProvider {
|
|
603
|
+
provider: token_provider,
|
|
604
|
+
});
|
|
605
|
+
let result = manager
|
|
606
|
+
.transform_structured(&data, &findings, &config, context.as_ref())
|
|
607
|
+
.await
|
|
608
|
+
.map_err(|error| Python::attach(|py| privacy_error(py, error)))?;
|
|
609
|
+
Python::attach(|py| Py::new(py, structured_transform_result(py, result)?))
|
|
610
|
+
})
|
|
611
|
+
}
|
|
612
|
+
#[pyo3(signature = (data, config, context=None))]
|
|
613
|
+
fn scan_and_transform_structured<'py>(
|
|
614
|
+
&self,
|
|
615
|
+
py: Python<'py>,
|
|
616
|
+
data: Py<PyAny>,
|
|
617
|
+
config: Py<PyAny>,
|
|
618
|
+
context: Option<Py<PyAny>>,
|
|
619
|
+
) -> PyResult<Bound<'py, PyAny>> {
|
|
620
|
+
let data = structured_data(py, data.bind(py))?;
|
|
621
|
+
let config_value = structured_options(py, config.bind(py), "")?;
|
|
622
|
+
let config = core::structured::parse_scan_and_transform_config(&config_value)
|
|
623
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
624
|
+
let context = context
|
|
625
|
+
.map(|context| structured_options(py, context.bind(py), ""))
|
|
626
|
+
.transpose()?
|
|
627
|
+
.map(|value| core::parse_privacy_context(&value))
|
|
628
|
+
.transpose()
|
|
629
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
630
|
+
let key_provider = self
|
|
631
|
+
.key_provider
|
|
632
|
+
.as_ref()
|
|
633
|
+
.map(|provider| provider.clone_ref(py));
|
|
634
|
+
let token_provider = self
|
|
635
|
+
.token_provider
|
|
636
|
+
.as_ref()
|
|
637
|
+
.map(|provider| provider.clone_ref(py));
|
|
638
|
+
pyo3_async_runtimes::tokio::future_into_py(py, async move {
|
|
639
|
+
let manager = core::PrivacyManager::new(PythonKeyProvider {
|
|
640
|
+
provider: key_provider,
|
|
641
|
+
})
|
|
642
|
+
.with_token_provider(PythonTokenProvider {
|
|
643
|
+
provider: token_provider,
|
|
644
|
+
});
|
|
645
|
+
let result = manager
|
|
646
|
+
.scan_and_transform_structured(&data, &config, context.as_ref())
|
|
647
|
+
.await
|
|
648
|
+
.map_err(|error| Python::attach(|py| privacy_error(py, error)))?;
|
|
649
|
+
Python::attach(|py| Py::new(py, structured_transform_result(py, result)?))
|
|
650
|
+
})
|
|
651
|
+
}
|
|
652
|
+
fn restore_structured<'py>(
|
|
653
|
+
&self,
|
|
654
|
+
py: Python<'py>,
|
|
655
|
+
data: Py<PyAny>,
|
|
656
|
+
context: Py<PyAny>,
|
|
657
|
+
) -> PyResult<Bound<'py, PyAny>> {
|
|
658
|
+
let data = structured_data(py, data.bind(py))?;
|
|
659
|
+
let context = structured_options(py, context.bind(py), "")?;
|
|
660
|
+
let context =
|
|
661
|
+
core::parse_privacy_context(&context).map_err(|error| privacy_error(py, error))?;
|
|
662
|
+
let key_provider = self
|
|
663
|
+
.key_provider
|
|
664
|
+
.as_ref()
|
|
665
|
+
.map(|provider| provider.clone_ref(py));
|
|
666
|
+
let token_provider = self
|
|
667
|
+
.token_provider
|
|
668
|
+
.as_ref()
|
|
669
|
+
.map(|provider| provider.clone_ref(py));
|
|
670
|
+
pyo3_async_runtimes::tokio::future_into_py(py, async move {
|
|
671
|
+
let manager = core::PrivacyManager::new(PythonKeyProvider {
|
|
672
|
+
provider: key_provider,
|
|
673
|
+
})
|
|
674
|
+
.with_token_provider(PythonTokenProvider {
|
|
675
|
+
provider: token_provider,
|
|
676
|
+
});
|
|
677
|
+
let result = manager
|
|
678
|
+
.restore_structured(&data, &context)
|
|
679
|
+
.await
|
|
680
|
+
.map_err(|error| Python::attach(|py| privacy_error(py, error)))?;
|
|
681
|
+
Python::attach(|py| Py::new(py, structured_restore_result(py, result)?))
|
|
682
|
+
})
|
|
683
|
+
}
|
|
684
|
+
|
|
574
685
|
#[new]
|
|
575
686
|
#[pyo3(signature = (provider=None, token_provider=None))]
|
|
576
687
|
fn new(
|
|
@@ -899,6 +1010,260 @@ fn scan_and_transform(
|
|
|
899
1010
|
.map_err(|error| privacy_error(py, error))
|
|
900
1011
|
}
|
|
901
1012
|
|
|
1013
|
+
#[pyclass(frozen, skip_from_py_object)]
|
|
1014
|
+
#[derive(Clone)]
|
|
1015
|
+
struct FieldMapping {
|
|
1016
|
+
#[pyo3(get)]
|
|
1017
|
+
path: String,
|
|
1018
|
+
#[pyo3(get)]
|
|
1019
|
+
entity_type: String,
|
|
1020
|
+
#[pyo3(get)]
|
|
1021
|
+
source: String,
|
|
1022
|
+
#[pyo3(get)]
|
|
1023
|
+
rule: String,
|
|
1024
|
+
}
|
|
1025
|
+
impl From<core::structured::FieldMapping> for FieldMapping {
|
|
1026
|
+
fn from(mapping: core::structured::FieldMapping) -> Self {
|
|
1027
|
+
Self {
|
|
1028
|
+
path: mapping.path,
|
|
1029
|
+
entity_type: mapping.entity_type,
|
|
1030
|
+
source: mapping.source,
|
|
1031
|
+
rule: mapping.rule,
|
|
1032
|
+
}
|
|
1033
|
+
}
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
#[pyclass(frozen, skip_from_py_object)]
|
|
1037
|
+
#[derive(Clone)]
|
|
1038
|
+
struct StructuredFinding {
|
|
1039
|
+
#[pyo3(get)]
|
|
1040
|
+
path: String,
|
|
1041
|
+
#[pyo3(get)]
|
|
1042
|
+
finding: Finding,
|
|
1043
|
+
}
|
|
1044
|
+
|
|
1045
|
+
#[pyclass(frozen, skip_from_py_object)]
|
|
1046
|
+
struct StructuredScanResult {
|
|
1047
|
+
#[pyo3(get)]
|
|
1048
|
+
mappings: Vec<FieldMapping>,
|
|
1049
|
+
#[pyo3(get)]
|
|
1050
|
+
findings: Vec<StructuredFinding>,
|
|
1051
|
+
}
|
|
1052
|
+
|
|
1053
|
+
fn structured_data(py: Python<'_>, value: &Bound<'_, PyAny>) -> PyResult<serde_json::Value> {
|
|
1054
|
+
// The stdlib encoder rejects cycles/deep recursion safely. Check dictionaries
|
|
1055
|
+
// and sequences first so it cannot silently coerce keys or tuple values.
|
|
1056
|
+
let mut pending = vec![value.clone()];
|
|
1057
|
+
let mut seen = std::collections::BTreeSet::new();
|
|
1058
|
+
while let Some(value) = pending.pop() {
|
|
1059
|
+
if value.is_none()
|
|
1060
|
+
|| value.is_instance_of::<PyBool>()
|
|
1061
|
+
|| value.is_instance_of::<pyo3::types::PyString>()
|
|
1062
|
+
|| value.is_instance_of::<pyo3::types::PyInt>()
|
|
1063
|
+
|| value.is_instance_of::<pyo3::types::PyFloat>()
|
|
1064
|
+
{
|
|
1065
|
+
continue;
|
|
1066
|
+
}
|
|
1067
|
+
if !seen.insert(value.as_ptr() as usize) {
|
|
1068
|
+
// Repeated containers are valid JSON trees after serialization;
|
|
1069
|
+
// the encoder below distinguishes shared references from cycles.
|
|
1070
|
+
continue;
|
|
1071
|
+
}
|
|
1072
|
+
if let Ok(object) = value.cast::<PyDict>() {
|
|
1073
|
+
for (key, child) in object.iter() {
|
|
1074
|
+
if !key.is_instance_of::<pyo3::types::PyString>() {
|
|
1075
|
+
return Err(privacy_error(py, core::structured::invalid_data()));
|
|
1076
|
+
}
|
|
1077
|
+
pending.push(child);
|
|
1078
|
+
}
|
|
1079
|
+
} else if let Ok(array) = value.cast::<PyList>() {
|
|
1080
|
+
pending.extend(array.iter());
|
|
1081
|
+
} else {
|
|
1082
|
+
return Err(privacy_error(py, core::structured::invalid_data()));
|
|
1083
|
+
}
|
|
1084
|
+
}
|
|
1085
|
+
let kwargs = PyDict::new(py);
|
|
1086
|
+
kwargs.set_item("allow_nan", false)?;
|
|
1087
|
+
let json: String = py
|
|
1088
|
+
.import("json")?
|
|
1089
|
+
.call_method("dumps", (value,), Some(&kwargs))
|
|
1090
|
+
.and_then(|value| value.extract())
|
|
1091
|
+
.map_err(|_| privacy_error(py, core::structured::invalid_data()))?;
|
|
1092
|
+
core::structured::parse_document_json(&json).map_err(|error| privacy_error(py, error))
|
|
1093
|
+
}
|
|
1094
|
+
|
|
1095
|
+
fn structured_config(
|
|
1096
|
+
py: Python<'_>,
|
|
1097
|
+
config: Option<&Bound<'_, PyAny>>,
|
|
1098
|
+
) -> PyResult<core::structured::StructuredScanConfig> {
|
|
1099
|
+
match config {
|
|
1100
|
+
Some(config) => core::structured::parse_scan_config(&structured_options(py, config, "")?)
|
|
1101
|
+
.map_err(|error| privacy_error(py, error)),
|
|
1102
|
+
None => Ok(core::structured::StructuredScanConfig::default()),
|
|
1103
|
+
}
|
|
1104
|
+
}
|
|
1105
|
+
|
|
1106
|
+
#[pyfunction]
|
|
1107
|
+
#[pyo3(signature = (data, config=None))]
|
|
1108
|
+
fn discover_fields(
|
|
1109
|
+
py: Python<'_>,
|
|
1110
|
+
data: &Bound<'_, PyAny>,
|
|
1111
|
+
config: Option<&Bound<'_, PyAny>>,
|
|
1112
|
+
) -> PyResult<Vec<FieldMapping>> {
|
|
1113
|
+
core::structured::discover_fields(&structured_data(py, data)?, &structured_config(py, config)?)
|
|
1114
|
+
.map(|mappings| mappings.into_iter().map(FieldMapping::from).collect())
|
|
1115
|
+
.map_err(|error| privacy_error(py, error))
|
|
1116
|
+
}
|
|
1117
|
+
|
|
1118
|
+
#[pyfunction]
|
|
1119
|
+
#[pyo3(signature = (data, config=None))]
|
|
1120
|
+
fn scan_structured(
|
|
1121
|
+
py: Python<'_>,
|
|
1122
|
+
data: &Bound<'_, PyAny>,
|
|
1123
|
+
config: Option<&Bound<'_, PyAny>>,
|
|
1124
|
+
) -> PyResult<StructuredScanResult> {
|
|
1125
|
+
let result =
|
|
1126
|
+
core::structured::scan(&structured_data(py, data)?, &structured_config(py, config)?)
|
|
1127
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
1128
|
+
Ok(StructuredScanResult {
|
|
1129
|
+
mappings: result
|
|
1130
|
+
.mappings
|
|
1131
|
+
.into_iter()
|
|
1132
|
+
.map(FieldMapping::from)
|
|
1133
|
+
.collect(),
|
|
1134
|
+
findings: result
|
|
1135
|
+
.findings
|
|
1136
|
+
.into_iter()
|
|
1137
|
+
.map(|located| StructuredFinding {
|
|
1138
|
+
path: located.path,
|
|
1139
|
+
finding: Finding::from(located.finding),
|
|
1140
|
+
})
|
|
1141
|
+
.collect(),
|
|
1142
|
+
})
|
|
1143
|
+
}
|
|
1144
|
+
|
|
1145
|
+
#[pymethods]
|
|
1146
|
+
impl StructuredFinding {
|
|
1147
|
+
#[new]
|
|
1148
|
+
fn new(path: String, finding: PyRef<'_, Finding>) -> Self {
|
|
1149
|
+
Self {
|
|
1150
|
+
path,
|
|
1151
|
+
finding: finding.clone(),
|
|
1152
|
+
}
|
|
1153
|
+
}
|
|
1154
|
+
}
|
|
1155
|
+
impl StructuredFinding {
|
|
1156
|
+
fn to_core(&self) -> core::structured::StructuredFinding {
|
|
1157
|
+
core::structured::StructuredFinding {
|
|
1158
|
+
path: self.path.clone(),
|
|
1159
|
+
finding: self.finding.to_core(),
|
|
1160
|
+
}
|
|
1161
|
+
}
|
|
1162
|
+
}
|
|
1163
|
+
|
|
1164
|
+
#[pyclass(frozen, skip_from_py_object)]
|
|
1165
|
+
#[derive(Clone)]
|
|
1166
|
+
struct StructuredTransformation {
|
|
1167
|
+
#[pyo3(get)]
|
|
1168
|
+
path: String,
|
|
1169
|
+
#[pyo3(get)]
|
|
1170
|
+
transformation: Transformation,
|
|
1171
|
+
}
|
|
1172
|
+
#[pyclass(frozen, skip_from_py_object)]
|
|
1173
|
+
struct StructuredTransformResult {
|
|
1174
|
+
#[pyo3(get)]
|
|
1175
|
+
data: Py<PyAny>,
|
|
1176
|
+
#[pyo3(get)]
|
|
1177
|
+
transformations: Vec<StructuredTransformation>,
|
|
1178
|
+
}
|
|
1179
|
+
#[pyclass(frozen, skip_from_py_object)]
|
|
1180
|
+
#[derive(Clone)]
|
|
1181
|
+
struct StructuredRestoration {
|
|
1182
|
+
#[pyo3(get)]
|
|
1183
|
+
path: String,
|
|
1184
|
+
#[pyo3(get)]
|
|
1185
|
+
restoration: Restoration,
|
|
1186
|
+
}
|
|
1187
|
+
#[pyclass(frozen, skip_from_py_object)]
|
|
1188
|
+
struct StructuredRestoreResult {
|
|
1189
|
+
#[pyo3(get)]
|
|
1190
|
+
data: Py<PyAny>,
|
|
1191
|
+
#[pyo3(get)]
|
|
1192
|
+
restorations: Vec<StructuredRestoration>,
|
|
1193
|
+
}
|
|
1194
|
+
fn structured_transform_result(
|
|
1195
|
+
py: Python<'_>,
|
|
1196
|
+
result: core::structured::StructuredTransformResult,
|
|
1197
|
+
) -> PyResult<StructuredTransformResult> {
|
|
1198
|
+
Ok(StructuredTransformResult {
|
|
1199
|
+
data: py
|
|
1200
|
+
.import("json")?
|
|
1201
|
+
.call_method1("loads", (result.data.to_string(),))?
|
|
1202
|
+
.unbind(),
|
|
1203
|
+
transformations: result
|
|
1204
|
+
.transformations
|
|
1205
|
+
.into_iter()
|
|
1206
|
+
.map(|record| StructuredTransformation {
|
|
1207
|
+
path: record.path,
|
|
1208
|
+
transformation: Transformation::from(record.transformation),
|
|
1209
|
+
})
|
|
1210
|
+
.collect(),
|
|
1211
|
+
})
|
|
1212
|
+
}
|
|
1213
|
+
fn structured_restore_result(
|
|
1214
|
+
py: Python<'_>,
|
|
1215
|
+
result: core::structured::StructuredRestoreResult,
|
|
1216
|
+
) -> PyResult<StructuredRestoreResult> {
|
|
1217
|
+
Ok(StructuredRestoreResult {
|
|
1218
|
+
data: py
|
|
1219
|
+
.import("json")?
|
|
1220
|
+
.call_method1("loads", (result.data.to_string(),))?
|
|
1221
|
+
.unbind(),
|
|
1222
|
+
restorations: result
|
|
1223
|
+
.restorations
|
|
1224
|
+
.into_iter()
|
|
1225
|
+
.map(|record| StructuredRestoration {
|
|
1226
|
+
path: record.path,
|
|
1227
|
+
restoration: Restoration::from(record.restoration),
|
|
1228
|
+
})
|
|
1229
|
+
.collect(),
|
|
1230
|
+
})
|
|
1231
|
+
}
|
|
1232
|
+
|
|
1233
|
+
#[pyfunction]
|
|
1234
|
+
fn transform_structured(
|
|
1235
|
+
py: Python<'_>,
|
|
1236
|
+
data: &Bound<'_, PyAny>,
|
|
1237
|
+
findings: Vec<Py<StructuredFinding>>,
|
|
1238
|
+
config: &Bound<'_, PyAny>,
|
|
1239
|
+
) -> PyResult<StructuredTransformResult> {
|
|
1240
|
+
let data = structured_data(py, data)?;
|
|
1241
|
+
let config = core::parse_transformation_config(&structured_options(py, config, "")?)
|
|
1242
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
1243
|
+
let findings: Vec<_> = findings
|
|
1244
|
+
.iter()
|
|
1245
|
+
.map(|finding| finding.bind(py).borrow().to_core())
|
|
1246
|
+
.collect();
|
|
1247
|
+
let result = core::structured::transform(&data, &findings, &config)
|
|
1248
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
1249
|
+
structured_transform_result(py, result)
|
|
1250
|
+
}
|
|
1251
|
+
|
|
1252
|
+
#[pyfunction]
|
|
1253
|
+
fn scan_and_transform_structured(
|
|
1254
|
+
py: Python<'_>,
|
|
1255
|
+
data: &Bound<'_, PyAny>,
|
|
1256
|
+
config: &Bound<'_, PyAny>,
|
|
1257
|
+
) -> PyResult<StructuredTransformResult> {
|
|
1258
|
+
let data = structured_data(py, data)?;
|
|
1259
|
+
let config =
|
|
1260
|
+
core::structured::parse_scan_and_transform_config(&structured_options(py, config, "")?)
|
|
1261
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
1262
|
+
let result = core::structured::scan_and_transform(&data, &config)
|
|
1263
|
+
.map_err(|error| privacy_error(py, error))?;
|
|
1264
|
+
structured_transform_result(py, result)
|
|
1265
|
+
}
|
|
1266
|
+
|
|
902
1267
|
#[pymodule]
|
|
903
1268
|
fn datafog_core(module: &Bound<'_, PyModule>) -> PyResult<()> {
|
|
904
1269
|
module.add(
|
|
@@ -917,6 +1282,17 @@ fn datafog_core(module: &Bound<'_, PyModule>) -> PyResult<()> {
|
|
|
917
1282
|
"DataFogKeyProviderError",
|
|
918
1283
|
module.py().get_type::<DataFogKeyProviderError>(),
|
|
919
1284
|
)?;
|
|
1285
|
+
module.add_class::<FieldMapping>()?;
|
|
1286
|
+
module.add_class::<StructuredFinding>()?;
|
|
1287
|
+
module.add_class::<StructuredScanResult>()?;
|
|
1288
|
+
module.add_class::<StructuredTransformation>()?;
|
|
1289
|
+
module.add_class::<StructuredTransformResult>()?;
|
|
1290
|
+
module.add_class::<StructuredRestoration>()?;
|
|
1291
|
+
module.add_class::<StructuredRestoreResult>()?;
|
|
1292
|
+
module.add_function(wrap_pyfunction!(transform_structured, module)?)?;
|
|
1293
|
+
module.add_function(wrap_pyfunction!(scan_and_transform_structured, module)?)?;
|
|
1294
|
+
module.add_function(wrap_pyfunction!(discover_fields, module)?)?;
|
|
1295
|
+
module.add_function(wrap_pyfunction!(scan_structured, module)?)?;
|
|
920
1296
|
module.add_class::<TextRange>()?;
|
|
921
1297
|
module.add_class::<Finding>()?;
|
|
922
1298
|
module.add_class::<Transformation>()?;
|
|
@@ -929,3 +1305,30 @@ fn datafog_core(module: &Bound<'_, PyModule>) -> PyResult<()> {
|
|
|
929
1305
|
module.add_function(wrap_pyfunction!(scan_and_transform, module)?)?;
|
|
930
1306
|
Ok(())
|
|
931
1307
|
}
|
|
1308
|
+
|
|
1309
|
+
impl From<core::Restoration> for Restoration {
|
|
1310
|
+
fn from(record: core::Restoration) -> Self {
|
|
1311
|
+
Restoration {
|
|
1312
|
+
source_byte_range: record.source_byte_range.into(),
|
|
1313
|
+
source_codepoint_range: record.source_codepoint_range.into(),
|
|
1314
|
+
output_byte_range: record.output_byte_range.into(),
|
|
1315
|
+
output_codepoint_range: record.output_codepoint_range.into(),
|
|
1316
|
+
token_ref: record.token_ref,
|
|
1317
|
+
resolved_token_version: record.resolved_token_version,
|
|
1318
|
+
}
|
|
1319
|
+
}
|
|
1320
|
+
}
|
|
1321
|
+
|
|
1322
|
+
fn structured_options(
|
|
1323
|
+
py: Python<'_>,
|
|
1324
|
+
value: &Bound<'_, PyAny>,
|
|
1325
|
+
path: &str,
|
|
1326
|
+
) -> PyResult<serde_json::Value> {
|
|
1327
|
+
structured_data(py, value).map_err(|_| {
|
|
1328
|
+
configuration_conversion_error(
|
|
1329
|
+
py,
|
|
1330
|
+
path,
|
|
1331
|
+
"structured request options must be JSON-compatible",
|
|
1332
|
+
)
|
|
1333
|
+
})
|
|
1334
|
+
}
|
|
@@ -14,6 +14,10 @@ from datafog_core import (
|
|
|
14
14
|
PrivacyManager,
|
|
15
15
|
TextRange,
|
|
16
16
|
scan,
|
|
17
|
+
scan_structured,
|
|
18
|
+
transform_structured,
|
|
19
|
+
scan_and_transform_structured,
|
|
20
|
+
discover_fields,
|
|
17
21
|
scan_and_transform,
|
|
18
22
|
transform,
|
|
19
23
|
)
|
|
@@ -64,7 +68,54 @@ def verify_fixture(name: str) -> None:
|
|
|
64
68
|
verify_contract(record["text"])
|
|
65
69
|
|
|
66
70
|
|
|
71
|
+
def verify_structured() -> None:
|
|
72
|
+
for line in (ROOT / "fixtures" / "structured.jsonl").read_text().splitlines():
|
|
73
|
+
record = json.loads(line)
|
|
74
|
+
result = scan_structured(record["data"], record.get("config"))
|
|
75
|
+
mappings = [dict(path=m.path, entity_type=m.entity_type, source=m.source, rule=m.rule) for m in result.mappings]
|
|
76
|
+
assert mappings == record["mappings"], record["id"]
|
|
77
|
+
discovered = discover_fields(record["data"], record.get("config"))
|
|
78
|
+
assert [m.path for m in discovered] == [m.path for m in result.mappings]
|
|
79
|
+
actual = []
|
|
80
|
+
for located in result.findings:
|
|
81
|
+
text = record["data"]
|
|
82
|
+
for part in located.path[1:].split("/"):
|
|
83
|
+
key = part.replace("~1", "/").replace("~0", "~")
|
|
84
|
+
text = text[int(key)] if isinstance(text, list) else text[key]
|
|
85
|
+
f = located.finding
|
|
86
|
+
assert text.encode()[f.byte_range.start:f.byte_range.end].decode() == f.matched_text
|
|
87
|
+
assert text[f.codepoint_range.start:f.codepoint_range.end] == f.matched_text
|
|
88
|
+
assert f.confidence is None
|
|
89
|
+
actual.append(dict(path=located.path, label=f.entity_type, text=f.matched_text, start=f.codepoint_range.start, end=f.codepoint_range.end))
|
|
90
|
+
assert actual == record["findings"], record["id"]
|
|
91
|
+
for line in (ROOT / "fixtures" / "structured-transform.jsonl").read_text().splitlines():
|
|
92
|
+
record = json.loads(line)
|
|
93
|
+
result = scan_and_transform_structured(record["data"],record["config"])
|
|
94
|
+
explicit = transform_structured(record["data"],scan_structured(record["data"]).findings,record["config"]["transform"])
|
|
95
|
+
assert result.data == record["expected_data"], record["id"]
|
|
96
|
+
assert explicit.data == result.data
|
|
97
|
+
assert all(not hasattr(r.transformation,"matched_text") for r in result.transformations)
|
|
98
|
+
cycle = {}
|
|
99
|
+
cycle["cycle"] = cycle
|
|
100
|
+
try:
|
|
101
|
+
scan_structured({}, cycle)
|
|
102
|
+
except DataFogConfigurationError:
|
|
103
|
+
pass
|
|
104
|
+
else:
|
|
105
|
+
raise AssertionError("cyclic options accepted")
|
|
106
|
+
for data in [None, "secret-value", {1: "name"}, {"n": 2**100}, {"n": float("nan")}, {"n": float("inf")}, {"tuple": (1, 2)}, cycle]:
|
|
107
|
+
try:
|
|
108
|
+
scan_structured(data)
|
|
109
|
+
except DataFogConfigurationError as error:
|
|
110
|
+
assert error.code == "invalid_configuration"
|
|
111
|
+
assert error.path == "/data"
|
|
112
|
+
assert "secret-value" not in str(error)
|
|
113
|
+
else:
|
|
114
|
+
raise AssertionError("invalid structured input accepted")
|
|
115
|
+
|
|
116
|
+
|
|
67
117
|
def main() -> None:
|
|
118
|
+
verify_structured()
|
|
68
119
|
verify_fixture("development.jsonl")
|
|
69
120
|
verify_fixture("final.jsonl")
|
|
70
121
|
emoji_finding = scan("👋 jane@example.com")[0]
|
|
@@ -307,6 +358,27 @@ def main() -> None:
|
|
|
307
358
|
results.append({"id": item["id"], "value": record[3]})
|
|
308
359
|
return results
|
|
309
360
|
|
|
361
|
+
async def structured_round_trip():
|
|
362
|
+
original = {"users":[{"first_name":"👋 José"},{"full_name":"May"}],"count":2}
|
|
363
|
+
provider = Provider()
|
|
364
|
+
pseudonyms = await PrivacyManager(provider).scan_and_transform_structured(original,{"transform":pseudonym_config})
|
|
365
|
+
assert len(provider.calls) == 1
|
|
366
|
+
assert pseudonyms.data != original
|
|
367
|
+
manager = PrivacyManager(None,TokenProvider())
|
|
368
|
+
context = {"scope":"tenant/α"}
|
|
369
|
+
config = {"transform":{"default":{"strategy":"tokenize","token_ref":"names"}}}
|
|
370
|
+
tokens = await manager.scan_and_transform_structured(original,config,context)
|
|
371
|
+
restored = await manager.restore_structured(tokens.data,context)
|
|
372
|
+
assert restored.data == original
|
|
373
|
+
assert len(restored.restorations) == 2
|
|
374
|
+
try:
|
|
375
|
+
await manager.restore_structured(tokens.data,{"scope":"wrong"})
|
|
376
|
+
except DataFogKeyProviderError as error:
|
|
377
|
+
assert error.code == "token_access_denied"
|
|
378
|
+
else:
|
|
379
|
+
raise AssertionError("wrong scope was accepted")
|
|
380
|
+
asyncio.run(structured_round_trip())
|
|
381
|
+
|
|
310
382
|
async def token_round_trip():
|
|
311
383
|
manager = PrivacyManager(None, TokenProvider())
|
|
312
384
|
context = {"scope": "tenant/α"}
|