pseudonymize 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.agents/PRIVATE_RELEASE_PLAN.md +18 -1
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/package.yml +8 -7
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/CHANGELOG.md +29 -1
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/PKG-INFO +1 -1
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/policies.md +5 -4
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releasing.md +9 -1
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/pyproject.toml +2 -2
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/audit_install.py +9 -2
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/ml/onnx.py +47 -44
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/url.py +10 -2
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/engine.py +9 -4
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/formats.py +22 -40
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/normalization.py +4 -5
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/policy.py +2 -1
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/spans.py +13 -5
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/backends/test_onnx.py +52 -21
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/detectors/test_detectors.py +18 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_policy.py +2 -1
- pseudonymize-0.2.1/tests/unit/test_spans.py +39 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/uv.lock +1 -1
- pseudonymize-0.2.0/tests/unit/test_spans.py +0 -15
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.editorconfig +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.gitattributes +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/CODEOWNERS +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/bug.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/config.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/detector-request.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/feature.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/question.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/PUBLICATION_CHECKLIST.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/dependabot.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/pull_request_template.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/ci.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/docs.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/release.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.gitignore +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.pre-commit-config.yaml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/CODE_OF_CONDUCT.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/CONTRIBUTING.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/GEMINI.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/LICENSE +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/README.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/ROADMAP.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/SECURITY.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/SUPPORT.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/VISION.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/README.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/benchmark_detectors.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/benchmark_engine.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/datasets/synthetic_messages.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0001-dependency-free-core.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0002-token-format.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0003-no-reversible-storage.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0004-optional-ml-backend.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/api.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/architecture.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/benchmarks.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/compatibility.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/contributing.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/deployment.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/detectors.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/index.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/limitations.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/llm-gateways.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/llm-payloads.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/migration-a2.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/migration-a3.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/quickstart.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0a2.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0a3.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0b1.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0rc1.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/security.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/threat-model.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/examples/llm_gateway.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/mkdocs.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/models/README.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/smoke_wheel.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/verify_release.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/adapters.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/api.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/base.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/composite.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/ml/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/rules.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/cli.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/base.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/email.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/iban.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/ip_address.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/payment_card.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/phone.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/registry.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/secret.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/document.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/exceptions.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/processing.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/py.typed +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/resolution.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/result.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/alias.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/base.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/hmac.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/mode.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/placeholder.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/redact.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/compatibility/test_b1_contract.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/de.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/en.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/files.json +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/it.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_builtin_files.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_cli.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_cross_platform_corpus.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_file_processing.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_llm_gateway_example.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_public_api.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_release_stress.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_release_verifier.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/property/test_invariants.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_backends.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_cli_coverage.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_document.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_engine.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_formats.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_modes.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_normalization.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_processing.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_transforms.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_types.py +0 -0
|
@@ -92,7 +92,24 @@ PyPI upload verified:
|
|
|
92
92
|
Status:
|
|
93
93
|
```
|
|
94
94
|
|
|
95
|
-
### 0.2.
|
|
95
|
+
### 0.2.1: Performance and default policy patches
|
|
96
|
+
|
|
97
|
+
Implementation started: 2026-08-22
|
|
98
|
+
Starting commit: aabba9a
|
|
99
|
+
Starting PyPI version: 0.2.0
|
|
100
|
+
Baseline downloads day/week/month: 0/0/0
|
|
101
|
+
Baseline stars/forks/watchers: 0/0/0
|
|
102
|
+
Files changed: 13
|
|
103
|
+
Compatibility tests added: Yes
|
|
104
|
+
Artifact smoke environments: Win local isolated wheel
|
|
105
|
+
Known limitations:
|
|
106
|
+
Deferred items:
|
|
107
|
+
Release commit: TBD
|
|
108
|
+
Tag: v0.2.1
|
|
109
|
+
PyPI upload verified: Pending workflow
|
|
110
|
+
7-day metrics:
|
|
111
|
+
28-day metrics:
|
|
112
|
+
Status: Ready
|
|
96
113
|
|
|
97
114
|
Implementation started: 2026-08-21
|
|
98
115
|
Starting commit: 3205c22ef6143aa8ae38fa2db9eea64fc237777d
|
|
@@ -49,16 +49,17 @@ jobs:
|
|
|
49
49
|
- name: Install and audit wheel
|
|
50
50
|
if: runner.os != 'Windows'
|
|
51
51
|
run: |
|
|
52
|
-
uv pip install --python .wheel-venv/bin/python dist
|
|
52
|
+
uv pip install --python .wheel-venv/bin/python dist/*.whl
|
|
53
53
|
uv pip check --python .wheel-venv/bin/python
|
|
54
|
-
.wheel-venv/bin/python -I scripts/audit_install.py
|
|
54
|
+
.wheel-venv/bin/python -I scripts/audit_install.py
|
|
55
55
|
.wheel-venv/bin/python -I scripts/smoke_wheel.py
|
|
56
56
|
.wheel-venv/bin/pseudonymize detectors
|
|
57
57
|
- name: Install and audit wheel on Windows
|
|
58
58
|
if: runner.os == 'Windows'
|
|
59
|
+
shell: bash
|
|
59
60
|
run: |
|
|
60
|
-
uv pip install --python .wheel-venv
|
|
61
|
-
uv pip check --python .wheel-venv
|
|
62
|
-
.wheel-venv
|
|
63
|
-
.wheel-venv
|
|
64
|
-
.wheel-venv
|
|
61
|
+
uv pip install --python .wheel-venv/Scripts/python.exe dist/*.whl
|
|
62
|
+
uv pip check --python .wheel-venv/Scripts/python.exe
|
|
63
|
+
.wheel-venv/Scripts/python.exe -I scripts/audit_install.py
|
|
64
|
+
.wheel-venv/Scripts/python.exe -I scripts/smoke_wheel.py
|
|
65
|
+
.wheel-venv/Scripts/pseudonymize.exe detectors
|
|
@@ -4,6 +4,33 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.2.1] - 2026-08-22
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- `Policy.default()` now includes `URL_CREDENTIAL` and `SECRET`, so the documented secret and
|
|
12
|
+
URL-credential detectors act without opting into `Policy.strict()` or `Policy.llm()`. Callers
|
|
13
|
+
that relied on secrets passing through the default policy must configure an explicit
|
|
14
|
+
`Policy(entity_types=...)`.
|
|
15
|
+
- Overlap resolution now uses an ordered-interval scan instead of a quadratic pairwise check,
|
|
16
|
+
and detected spans are replaced with a single-pass segment join. Texts with thousands of
|
|
17
|
+
detections process in milliseconds instead of seconds.
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
|
|
21
|
+
- `LocalONNXPIIBackend` now merges contiguous token predictions into whole entity spans, so
|
|
22
|
+
multi-word and subword-split names receive a single alias instead of one alias per token.
|
|
23
|
+
- ONNX test artifacts downloaded during testing are now verified against pinned SHA-256
|
|
24
|
+
checksums.
|
|
25
|
+
- URL credential detection now covers the whole userinfo section when it contains extra `@`
|
|
26
|
+
characters and no longer swallows the URL fragment into sensitive query values.
|
|
27
|
+
- Overlap resolution now ranks URL credentials above emails, so a `user:password@host` userinfo
|
|
28
|
+
can no longer lose its `user:` prefix to an overlapping email match.
|
|
29
|
+
- The `ml` extra now declares `numpy`, which the ONNX backend imports directly, and no longer
|
|
30
|
+
pulls in the unused `huggingface-hub` dependency.
|
|
31
|
+
- The optional-import guard in the ONNX backend now type-checks under strict mypy.
|
|
32
|
+
- Normalization now maps generic URL schemes to `http` or `https` for safer alias sharing.
|
|
33
|
+
|
|
7
34
|
## [0.2.0] - 2026-08-21
|
|
8
35
|
|
|
9
36
|
### Added
|
|
@@ -135,7 +162,8 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
|
|
|
135
162
|
- Deterministic processing now requires `mode="deterministic"` in addition to a key.
|
|
136
163
|
- `redact()` now emits `[REDACTED]` by default; generic mode emits typed placeholders.
|
|
137
164
|
|
|
138
|
-
[Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.
|
|
165
|
+
[Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.1...HEAD
|
|
166
|
+
[0.2.1]: https://github.com/ma2za/pseudonymize/compare/v0.2.0...v0.2.1
|
|
139
167
|
[0.2.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.2.0
|
|
140
168
|
[0.1.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0
|
|
141
169
|
[0.1.0rc1]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0rc1
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pseudonymize
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Local-first PII pseudonymization for text, structured data, and LLM payloads.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ma2za/pseudonymize
|
|
6
6
|
Project-URL: Documentation, https://ma2za.github.io/pseudonymize/
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
# Policies
|
|
2
2
|
|
|
3
|
-
`Policy.default()` enables structured identifiers
|
|
4
|
-
when a configured backend can detect them.
|
|
5
|
-
|
|
6
|
-
`Policy.financial()` limits detection to IBANs and payment
|
|
3
|
+
`Policy.default()` enables structured identifiers, secrets, URL credentials, and person,
|
|
4
|
+
organization, and location entities when a configured backend can detect them.
|
|
5
|
+
`Policy.strict()` enables the same entities with a lower confidence threshold. `Policy.llm()`
|
|
6
|
+
enables all available entities. `Policy.financial()` limits detection to IBANs and payment
|
|
7
|
+
cards.
|
|
7
8
|
|
|
8
9
|
Paths are dot-separated segments. `*` matches one segment, so `messages.*.content` matches list
|
|
9
10
|
indices. Exclusions win over inclusions. Dictionary keys are not transformed.
|
|
@@ -29,10 +29,18 @@ After the first upload, verify that the pending publisher became an ordinary pro
|
|
|
29
29
|
uv run ruff check .
|
|
30
30
|
uv run mypy
|
|
31
31
|
uv run pytest
|
|
32
|
-
uv run mkdocs build --strict
|
|
32
|
+
uv run python -m mkdocs build --strict
|
|
33
33
|
uv build
|
|
34
34
|
uv run twine check dist/*
|
|
35
35
|
uv run python scripts/verify_release.py
|
|
36
|
+
|
|
37
|
+
# Verify the built wheel locally in an isolated environment (simulating CI)
|
|
38
|
+
uv venv --python 3.14 .wheel-venv
|
|
39
|
+
uv pip install --python .wheel-venv/bin/python dist/*.whl
|
|
40
|
+
uv pip check --python .wheel-venv/bin/python
|
|
41
|
+
.wheel-venv/bin/python -I scripts/audit_install.py
|
|
42
|
+
.wheel-venv/bin/python -I scripts/smoke_wheel.py
|
|
43
|
+
.wheel-venv/bin/pseudonymize detectors
|
|
36
44
|
```
|
|
37
45
|
|
|
38
46
|
4. Open a pull request and merge only after every required check passes.
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pseudonymize"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.1"
|
|
8
8
|
description = "Local-first PII pseudonymization for text, structured data, and LLM payloads."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11"
|
|
@@ -91,7 +91,7 @@ minversion = "8"
|
|
|
91
91
|
testpaths = ["tests"]
|
|
92
92
|
pythonpath = ["."]
|
|
93
93
|
python_files = ["test_*.py", "benchmark_*.py"]
|
|
94
|
-
addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36"]
|
|
94
|
+
addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36", "--basetemp=.pytest_temp"]
|
|
95
95
|
markers = ["benchmark: performance benchmark", "integration: integration test", "slow: deliberately slow test"]
|
|
96
96
|
filterwarnings = ["error"]
|
|
97
97
|
|
|
@@ -27,11 +27,18 @@ def _blocked_network(*arguments: object, **keywords: object) -> None:
|
|
|
27
27
|
|
|
28
28
|
def main() -> None:
|
|
29
29
|
parser = argparse.ArgumentParser()
|
|
30
|
-
parser.add_argument("--version", required=
|
|
30
|
+
parser.add_argument("--version", required=False)
|
|
31
31
|
arguments = parser.parse_args()
|
|
32
32
|
|
|
33
|
+
expected_version = arguments.version
|
|
34
|
+
if not expected_version:
|
|
35
|
+
import tomllib
|
|
36
|
+
|
|
37
|
+
with open("pyproject.toml", "rb") as stream:
|
|
38
|
+
expected_version = tomllib.load(stream)["project"]["version"]
|
|
39
|
+
|
|
33
40
|
installed = distribution("pseudonymize")
|
|
34
|
-
if installed.version !=
|
|
41
|
+
if installed.version != expected_version:
|
|
35
42
|
raise RuntimeError("installed version does not match release")
|
|
36
43
|
if installed.requires:
|
|
37
44
|
required_deps = [req for req in installed.requires if "extra ==" not in req]
|
|
@@ -25,6 +25,20 @@ else:
|
|
|
25
25
|
ort = None
|
|
26
26
|
Tokenizer = None
|
|
27
27
|
|
|
28
|
+
_LABEL_SUFFIXES: tuple[tuple[tuple[str, ...], EntityType], ...] = (
|
|
29
|
+
# Standard CoNLL-03 suffixes plus Ai4Privacy fine-grained labels.
|
|
30
|
+
(("PER", "FIRSTNAME", "LASTNAME", "MIDDLENAME"), EntityType.PERSON),
|
|
31
|
+
(("ORG", "COMPANYNAME"), EntityType.ORGANIZATION),
|
|
32
|
+
(("LOC", "CITY", "STATE", "COUNTY", "STREET", "ZIPCODE"), EntityType.LOCATION),
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _entity_type_for(label: str) -> EntityType | None:
|
|
37
|
+
for suffixes, entity_type in _LABEL_SUFFIXES:
|
|
38
|
+
if label.endswith(suffixes):
|
|
39
|
+
return entity_type
|
|
40
|
+
return None
|
|
41
|
+
|
|
28
42
|
|
|
29
43
|
class LocalONNXPIIBackend(DetectionBackend):
|
|
30
44
|
def __init__(
|
|
@@ -111,55 +125,44 @@ class LocalONNXPIIBackend(DetectionBackend):
|
|
|
111
125
|
|
|
112
126
|
predictions = np.argmax(logits, axis=-1)
|
|
113
127
|
|
|
114
|
-
|
|
128
|
+
# Token predictions are merged into entity spans: subword continuations
|
|
129
|
+
# (zero gap) and same-type tokens separated by one whitespace character
|
|
130
|
+
# collapse into a single detection so that "John Smith" is one PERSON.
|
|
131
|
+
spans: list[tuple[EntityType, int, int]] = []
|
|
115
132
|
|
|
116
133
|
for idx, label_id in enumerate(predictions):
|
|
117
|
-
|
|
118
|
-
continue
|
|
119
|
-
|
|
120
|
-
label_str = self._id2label.get(label_id)
|
|
134
|
+
label_str = (self._id2label or {}).get(int(label_id))
|
|
121
135
|
if not label_str or label_str == "O":
|
|
122
136
|
continue
|
|
123
|
-
|
|
124
|
-
entity_type
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
if
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
entity_type = EntityType.ORGANIZATION
|
|
139
|
-
elif any(
|
|
140
|
-
label_str.endswith(s) for s in ("CITY", "STATE", "COUNTY", "STREET", "ZIPCODE")
|
|
141
|
-
):
|
|
142
|
-
entity_type = EntityType.LOCATION
|
|
143
|
-
|
|
144
|
-
if entity_type:
|
|
145
|
-
start, end = encoding.offsets[idx]
|
|
146
|
-
if start == 0 and end == 0 and idx not in (0, len(predictions) - 1):
|
|
147
|
-
continue
|
|
148
|
-
if start == end:
|
|
137
|
+
entity_type = _entity_type_for(label_str)
|
|
138
|
+
if entity_type is None:
|
|
139
|
+
continue
|
|
140
|
+
start, end = encoding.offsets[idx]
|
|
141
|
+
if start >= end:
|
|
142
|
+
continue
|
|
143
|
+
if spans:
|
|
144
|
+
previous_type, previous_start, previous_end = spans[-1]
|
|
145
|
+
gap = block.text[previous_end:start]
|
|
146
|
+
if (
|
|
147
|
+
entity_type is previous_type
|
|
148
|
+
and previous_end <= start <= previous_end + 1
|
|
149
|
+
and not gap.strip()
|
|
150
|
+
):
|
|
151
|
+
spans[-1] = (previous_type, previous_start, end)
|
|
149
152
|
continue
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
153
|
+
spans.append((entity_type, start, end))
|
|
154
|
+
|
|
155
|
+
return tuple(
|
|
156
|
+
Detection(
|
|
157
|
+
entity_type=entity_type,
|
|
158
|
+
start=start,
|
|
159
|
+
end=end,
|
|
160
|
+
confidence=0.99,
|
|
161
|
+
backend=self.name,
|
|
162
|
+
detector="onnx",
|
|
163
|
+
)
|
|
164
|
+
for entity_type, start, end in spans
|
|
165
|
+
)
|
|
163
166
|
|
|
164
167
|
except Exception as e:
|
|
165
168
|
raise BackendExecutionError(f"ONNX PII inference failed: {e}") from e
|
|
@@ -10,6 +10,11 @@ _SENSITIVE_QUERY_NAMES = frozenset(
|
|
|
10
10
|
)
|
|
11
11
|
|
|
12
12
|
|
|
13
|
+
def _authority_end(url: str, authority_start: int) -> int:
|
|
14
|
+
indexes = (url.find(character, authority_start) for character in "/?#")
|
|
15
|
+
return min((index for index in indexes if index != -1), default=len(url))
|
|
16
|
+
|
|
17
|
+
|
|
13
18
|
@dataclass(frozen=True, slots=True)
|
|
14
19
|
class UrlDetector:
|
|
15
20
|
name: str = "url"
|
|
@@ -21,7 +26,8 @@ class UrlDetector:
|
|
|
21
26
|
parsed = urllib.parse.urlsplit(url)
|
|
22
27
|
if parsed.username is not None or parsed.password is not None:
|
|
23
28
|
authority_start = url.find("//") + 2
|
|
24
|
-
|
|
29
|
+
authority_end = _authority_end(url, authority_start)
|
|
30
|
+
at = url.rfind("@", authority_start, authority_end)
|
|
25
31
|
detections.append(
|
|
26
32
|
Detection(
|
|
27
33
|
EntityType.URL_CREDENTIAL,
|
|
@@ -34,8 +40,10 @@ class UrlDetector:
|
|
|
34
40
|
query_start = url.find("?") + 1
|
|
35
41
|
if query_start == 0:
|
|
36
42
|
continue
|
|
43
|
+
fragment_start = url.find("#", query_start)
|
|
44
|
+
query_end = fragment_start if fragment_start != -1 else len(url)
|
|
37
45
|
cursor = query_start
|
|
38
|
-
for item in url[query_start:].split("&"):
|
|
46
|
+
for item in url[query_start:query_end].split("&"):
|
|
39
47
|
name, separator, value = item.partition("=")
|
|
40
48
|
if separator and urllib.parse.unquote_plus(name).lower() in _SENSITIVE_QUERY_NAMES:
|
|
41
49
|
value_start = cursor + len(name) + 1
|
|
@@ -265,10 +265,15 @@ class Pseudonymizer:
|
|
|
265
265
|
self.transformer.render(entity, alias)
|
|
266
266
|
for entity, alias in zip(entities, aliases, strict=True)
|
|
267
267
|
)
|
|
268
|
-
|
|
269
|
-
|
|
268
|
+
segments: list[str] = []
|
|
269
|
+
cursor = 0
|
|
270
|
+
for entity, token in zip(entities, tokens, strict=True):
|
|
270
271
|
detection = entity.detection
|
|
271
|
-
|
|
272
|
+
segments.append(text[cursor : detection.start])
|
|
273
|
+
segments.append(token)
|
|
274
|
+
cursor = detection.end
|
|
275
|
+
segments.append(text[cursor:])
|
|
276
|
+
output = "".join(segments)
|
|
272
277
|
replacements = _replacement_reports(entities, tokens)
|
|
273
278
|
reports.extend(_replacement_detection_reports(block, replacements))
|
|
274
279
|
mapping = _mapping(text, entities, aliases, tokens) if include_mapping else None
|
|
@@ -287,10 +292,10 @@ class Pseudonymizer:
|
|
|
287
292
|
if isinstance(data, str):
|
|
288
293
|
block_id = f"block-{block_counter[0]:06d}"
|
|
289
294
|
block_counter[0] += 1
|
|
290
|
-
block = ContentBlock(block_id, data, JSONPathLocation(path))
|
|
291
295
|
if not self.policy.allows_path(tuple(str(part) for part in path)):
|
|
292
296
|
statistics.blocks_processed += 1
|
|
293
297
|
return data
|
|
298
|
+
block = ContentBlock(block_id, data, JSONPathLocation(path))
|
|
294
299
|
return self._process_block(block, context, False, statistics, reports).text
|
|
295
300
|
if data is None or isinstance(data, (bool, int, float)):
|
|
296
301
|
return data
|
|
@@ -242,7 +242,7 @@ def _json_replacements(document: Document) -> dict[tuple[str | int, ...], str]:
|
|
|
242
242
|
|
|
243
243
|
def _render_json(document: Document, state: _AdapterState) -> str:
|
|
244
244
|
value = _replace_json_strings(
|
|
245
|
-
|
|
245
|
+
cast(JSONValue, state.content),
|
|
246
246
|
(),
|
|
247
247
|
_json_replacements(document),
|
|
248
248
|
)
|
|
@@ -250,32 +250,26 @@ def _render_json(document: Document, state: _AdapterState) -> str:
|
|
|
250
250
|
|
|
251
251
|
|
|
252
252
|
def _extract_jsonl(text: str, decoded: _DecodedText) -> tuple[Document, tuple[JSONValue, ...]]:
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
value,
|
|
269
|
-
(line_index,),
|
|
270
|
-
blocks,
|
|
271
|
-
prefix=f"line-{line_index:06d}-block",
|
|
272
|
-
)
|
|
273
|
-
values = tuple(values_list)
|
|
274
|
-
return (
|
|
275
|
-
Document("file", tuple(blocks), _metadata(FileFormat.JSONL, decoded)),
|
|
276
|
-
values,
|
|
253
|
+
values: list[JSONValue] = []
|
|
254
|
+
blocks: list[ContentBlock] = []
|
|
255
|
+
for line_index, line in enumerate(text.splitlines()):
|
|
256
|
+
if not line.strip():
|
|
257
|
+
raise _JSONLLineError(line_index + 1)
|
|
258
|
+
try:
|
|
259
|
+
value = _load_json(line)
|
|
260
|
+
except (TypeError, ValueError):
|
|
261
|
+
raise _JSONLLineError(line_index + 1) from None
|
|
262
|
+
values.append(value)
|
|
263
|
+
_string_blocks(
|
|
264
|
+
value,
|
|
265
|
+
(line_index,),
|
|
266
|
+
blocks,
|
|
267
|
+
prefix=f"line-{line_index:06d}-block",
|
|
277
268
|
)
|
|
278
|
-
return
|
|
269
|
+
return (
|
|
270
|
+
Document("file", tuple(blocks), _metadata(FileFormat.JSONL, decoded)),
|
|
271
|
+
tuple(values),
|
|
272
|
+
)
|
|
279
273
|
|
|
280
274
|
|
|
281
275
|
def _validate_document(document: Document, original: Document) -> None:
|
|
@@ -289,7 +283,7 @@ def _validate_document(document: Document, original: Document) -> None:
|
|
|
289
283
|
|
|
290
284
|
|
|
291
285
|
def _render_jsonl(document: Document, state: _AdapterState) -> str:
|
|
292
|
-
values =
|
|
286
|
+
values = cast(tuple[JSONValue, ...], state.content)
|
|
293
287
|
replacements = _json_replacements(document)
|
|
294
288
|
rendered = (
|
|
295
289
|
json.dumps(
|
|
@@ -332,7 +326,7 @@ def _read_csv(text: str) -> tuple[tuple[str, ...], ...]:
|
|
|
332
326
|
|
|
333
327
|
|
|
334
328
|
def _render_csv(document: Document, state: _AdapterState) -> str:
|
|
335
|
-
rows =
|
|
329
|
+
rows = cast(tuple[tuple[str, ...], ...], state.content)
|
|
336
330
|
replacements = {
|
|
337
331
|
(
|
|
338
332
|
cast(CSVCellLocation, block.location).row,
|
|
@@ -347,15 +341,3 @@ def _render_csv(document: Document, state: _AdapterState) -> str:
|
|
|
347
341
|
output = io.StringIO(newline="")
|
|
348
342
|
csv.writer(output, dialect="excel", lineterminator="\r\n").writerows(transformed)
|
|
349
343
|
return output.getvalue()
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
def _require_json_content(content: object) -> JSONValue:
|
|
353
|
-
return cast(JSONValue, content)
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
def _require_jsonl_content(content: object) -> tuple[JSONValue, ...]:
|
|
357
|
-
return cast(tuple[JSONValue, ...], content)
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
def _require_csv_content(content: object) -> tuple[tuple[str, ...], ...]:
|
|
361
|
-
return cast(tuple[tuple[str, ...], ...], content)
|
|
@@ -11,12 +11,11 @@ def normalize(value: str, entity_type: EntityType) -> str:
|
|
|
11
11
|
if entity_type is EntityType.EMAIL:
|
|
12
12
|
local, separator, domain = value.rpartition("@")
|
|
13
13
|
return f"{local}{separator}{domain.lower()}"
|
|
14
|
-
if entity_type
|
|
14
|
+
if entity_type is EntityType.IBAN:
|
|
15
|
+
return "".join(value.split()).upper()
|
|
16
|
+
if entity_type in {EntityType.PAYMENT_CARD, EntityType.PHONE}:
|
|
15
17
|
prefix = "+" if entity_type is EntityType.PHONE and value.startswith("+") else ""
|
|
16
|
-
|
|
17
|
-
if entity_type is EntityType.IBAN:
|
|
18
|
-
return "".join(value.split()).upper()
|
|
19
|
-
return prefix + compact
|
|
18
|
+
return prefix + "".join(character for character in value if character.isdigit())
|
|
20
19
|
if entity_type is EntityType.IP_ADDRESS:
|
|
21
20
|
return ipaddress.ip_address(value.strip("[]")).compressed
|
|
22
21
|
return value
|
|
@@ -13,6 +13,7 @@ _STRUCTURED = frozenset(
|
|
|
13
13
|
EntityType.PAYMENT_CARD,
|
|
14
14
|
}
|
|
15
15
|
)
|
|
16
|
+
_CREDENTIALS = frozenset({EntityType.URL_CREDENTIAL, EntityType.SECRET})
|
|
16
17
|
_SEMANTIC = frozenset({EntityType.PERSON, EntityType.ORGANIZATION, EntityType.LOCATION})
|
|
17
18
|
|
|
18
19
|
|
|
@@ -24,7 +25,7 @@ class NetworkPolicy(StrEnum):
|
|
|
24
25
|
|
|
25
26
|
@dataclass(frozen=True, slots=True)
|
|
26
27
|
class Policy:
|
|
27
|
-
entity_types: Set[EntityType] = _STRUCTURED | _SEMANTIC
|
|
28
|
+
entity_types: Set[EntityType] = _STRUCTURED | _CREDENTIALS | _SEMANTIC
|
|
28
29
|
minimum_confidence: float = 0.8
|
|
29
30
|
detector_priority: tuple[str, ...] = ()
|
|
30
31
|
include_paths: tuple[str, ...] = ()
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import bisect
|
|
1
2
|
from collections.abc import Iterable
|
|
2
3
|
|
|
3
4
|
from pseudonymize.result import Detection, EntityType
|
|
@@ -5,9 +6,12 @@ from pseudonymize.result import Detection, EntityType
|
|
|
5
6
|
_ENTITY_PRIORITY = {
|
|
6
7
|
EntityType.PAYMENT_CARD: 70,
|
|
7
8
|
EntityType.IBAN: 70,
|
|
9
|
+
# URL credentials outrank emails: a password such as "s3cret" followed by
|
|
10
|
+
# "@host" also matches the email pattern, and letting the email span win
|
|
11
|
+
# would leave the "user:" part of the userinfo unmasked.
|
|
12
|
+
EntityType.URL_CREDENTIAL: 65,
|
|
8
13
|
EntityType.EMAIL: 60,
|
|
9
14
|
EntityType.IP_ADDRESS: 60,
|
|
10
|
-
EntityType.URL_CREDENTIAL: 50,
|
|
11
15
|
EntityType.SECRET: 50,
|
|
12
16
|
EntityType.PHONE: 40,
|
|
13
17
|
EntityType.PERSON: 30,
|
|
@@ -35,12 +39,16 @@ def resolve_overlaps(
|
|
|
35
39
|
detection.backend,
|
|
36
40
|
),
|
|
37
41
|
)
|
|
42
|
+
# Accepted spans are kept sorted and non-overlapping, so a candidate can
|
|
43
|
+
# only collide with the span immediately before its insertion point.
|
|
44
|
+
starts: list[int] = []
|
|
45
|
+
ends: list[int] = []
|
|
38
46
|
selected: list[Detection] = []
|
|
39
47
|
for detection in ranked:
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
for existing in selected
|
|
43
|
-
):
|
|
48
|
+
index = bisect.bisect_left(starts, detection.end)
|
|
49
|
+
if index and ends[index - 1] > detection.start:
|
|
44
50
|
continue
|
|
51
|
+
starts.insert(index, detection.start)
|
|
52
|
+
ends.insert(index, detection.end)
|
|
45
53
|
selected.append(detection)
|
|
46
54
|
return tuple(sorted(selected, key=lambda detection: (detection.start, detection.end)))
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import hashlib
|
|
1
2
|
import urllib.request
|
|
2
3
|
from pathlib import Path
|
|
3
4
|
from typing import Any
|
|
@@ -15,27 +16,39 @@ MODEL_URL_BASE = (
|
|
|
15
16
|
"https://huggingface.co/onnx-community/distilbert_finetuned_ai4privacy_v2-ONNX/resolve/main/"
|
|
16
17
|
)
|
|
17
18
|
MODEL_FILES = {
|
|
18
|
-
"config.json":
|
|
19
|
-
|
|
20
|
-
|
|
19
|
+
"config.json": (
|
|
20
|
+
"config.json",
|
|
21
|
+
"5155e76f303c68ee15d8f01b580550c062cebfbb42dda3f5f3698b2d75424216",
|
|
22
|
+
),
|
|
23
|
+
"tokenizer.json": (
|
|
24
|
+
"tokenizer.json",
|
|
25
|
+
"cb374d6bc042c22455946f4e09a89d29882a199fdaf8fb25be00dc8b8857a448",
|
|
26
|
+
),
|
|
27
|
+
"model_int8.onnx": (
|
|
28
|
+
"onnx/model_int8.onnx",
|
|
29
|
+
"6faa1d7f5b54140bbba18ba87480e11073927b5fff16f69558bd51058a05b305",
|
|
30
|
+
),
|
|
21
31
|
}
|
|
22
32
|
|
|
23
33
|
|
|
24
|
-
def download_file(url: str, dest: Path) -> None:
|
|
25
|
-
if dest.exists():
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
34
|
+
def download_file(url: str, dest: Path, sha256: str) -> None:
|
|
35
|
+
if not dest.exists():
|
|
36
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
37
|
+
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"}) # noqa: S310
|
|
38
|
+
with urllib.request.urlopen(req) as response, open(dest, "wb") as f: # noqa: S310
|
|
39
|
+
f.write(response.read())
|
|
40
|
+
digest = hashlib.sha256(dest.read_bytes()).hexdigest()
|
|
41
|
+
if digest != sha256:
|
|
42
|
+
dest.unlink()
|
|
43
|
+
raise RuntimeError(f"checksum mismatch for {dest.name}: {digest}")
|
|
31
44
|
|
|
32
45
|
|
|
33
46
|
@pytest.fixture(scope="session")
|
|
34
47
|
def distilbert_artifacts() -> tuple[Path, Path, Path]:
|
|
35
48
|
paths = []
|
|
36
|
-
for local_name, remote_path in MODEL_FILES.items():
|
|
49
|
+
for local_name, (remote_path, sha256) in MODEL_FILES.items():
|
|
37
50
|
dest = CACHE_DIR / local_name
|
|
38
|
-
download_file(MODEL_URL_BASE + remote_path, dest)
|
|
51
|
+
download_file(MODEL_URL_BASE + remote_path, dest, sha256)
|
|
39
52
|
paths.append(dest)
|
|
40
53
|
# Returns (config, tokenizer, model)
|
|
41
54
|
return (paths[0], paths[1], paths[2])
|
|
@@ -166,19 +179,15 @@ def test_ml_detect_real_inference_returns_meaningful_detections(
|
|
|
166
179
|
# Sort detections by offset for deterministic assertion
|
|
167
180
|
sorted_detections = sorted(detections, key=lambda d: d.start)
|
|
168
181
|
|
|
169
|
-
#
|
|
170
|
-
# PERSON: "John" (0,
|
|
171
|
-
# LOC: "Seattle" (61,68)
|
|
172
|
-
|
|
173
|
-
# Note: Since the backend returns detections at the token level,
|
|
174
|
-
# we rigorously assert the exact subword spans mapped correctly to their entity type
|
|
175
|
-
assert len(sorted_detections) >= 4
|
|
182
|
+
# Token predictions merge into entity spans, so we expect:
|
|
183
|
+
# PERSON: "John Smith" (0,10) as one span, subwords and the space included
|
|
184
|
+
# LOC: "Seattle" (61,68) and "Washington" (70,80) kept apart by the comma
|
|
185
|
+
assert len(sorted_detections) >= 3
|
|
176
186
|
|
|
177
187
|
# Let's map out the exact expected strings for the entities found
|
|
178
188
|
found_entities = [(d.entity_type, text[d.start : d.end]) for d in sorted_detections]
|
|
179
189
|
|
|
180
|
-
assert (EntityType.PERSON, "John") in found_entities
|
|
181
|
-
assert (EntityType.PERSON, "Smith") in found_entities
|
|
190
|
+
assert (EntityType.PERSON, "John Smith") in found_entities
|
|
182
191
|
assert (EntityType.LOCATION, "Seattle") in found_entities
|
|
183
192
|
assert (EntityType.LOCATION, "Washington") in found_entities
|
|
184
193
|
|
|
@@ -250,6 +259,28 @@ def test_ml_detect_real_inference_returns_meaningful_detections(
|
|
|
250
259
|
assert EntityType.LOCATION in found_types
|
|
251
260
|
|
|
252
261
|
|
|
262
|
+
def test_ml_engine_shares_one_alias_per_merged_entity(
|
|
263
|
+
distilbert_artifacts: tuple[Path, Path, Path],
|
|
264
|
+
) -> None:
|
|
265
|
+
from pseudonymize import Pseudonymizer
|
|
266
|
+
|
|
267
|
+
config_path, tokenizer_path, model_path = distilbert_artifacts
|
|
268
|
+
backend = LocalONNXPIIBackend(
|
|
269
|
+
model_path=model_path, tokenizer_path=tokenizer_path, config_path=config_path
|
|
270
|
+
)
|
|
271
|
+
engine = Pseudonymizer(backends=[backend])
|
|
272
|
+
result = engine.process("John Smith emailed John Smith from Seattle.")
|
|
273
|
+
|
|
274
|
+
assert "John" not in result.text
|
|
275
|
+
assert "Smith" not in result.text
|
|
276
|
+
person_tokens = {
|
|
277
|
+
replacement.token
|
|
278
|
+
for replacement in result.replacements
|
|
279
|
+
if replacement.detection.entity_type is EntityType.PERSON
|
|
280
|
+
}
|
|
281
|
+
assert person_tokens == {"<PERSON_1>"}
|
|
282
|
+
|
|
283
|
+
|
|
253
284
|
def test_ml_detect_raises_on_inference_failure(
|
|
254
285
|
distilbert_artifacts: tuple[Path, Path, Path], monkeypatch: pytest.MonkeyPatch
|
|
255
286
|
) -> None:
|