pseudonymize 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.agents/PRIVATE_RELEASE_PLAN.md +37 -1
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/package.yml +8 -7
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/CHANGELOG.md +37 -1
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/PKG-INFO +7 -1
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/ROADMAP.md +2 -1
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/policies.md +5 -4
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releasing.md +9 -1
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/pyproject.toml +16 -4
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/audit_install.py +9 -2
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/ml/onnx.py +47 -44
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/url.py +10 -2
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/document.py +47 -2
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/engine.py +34 -7
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/formats.py +30 -40
- pseudonymize-0.3.0/src/pseudonymize/inspection/__init__.py +4 -0
- pseudonymize-0.3.0/src/pseudonymize/inspection/office.py +112 -0
- pseudonymize-0.3.0/src/pseudonymize/inspection/pdf.py +61 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/normalization.py +4 -5
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/policy.py +2 -1
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/spans.py +13 -5
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/compatibility/test_b1_contract.py +4 -0
- pseudonymize-0.3.0/tests/integration/test_inspection_adapters.py +141 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/backends/test_onnx.py +52 -21
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/detectors/test_detectors.py +18 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_document.py +18 -1
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_formats.py +1 -1
- pseudonymize-0.3.0/tests/unit/test_inspection.py +60 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_policy.py +2 -1
- pseudonymize-0.3.0/tests/unit/test_spans.py +39 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/uv.lock +439 -2
- pseudonymize-0.2.0/tests/unit/test_spans.py +0 -15
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.editorconfig +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.gitattributes +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/CODEOWNERS +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/bug.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/config.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/detector-request.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/feature.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/question.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/PUBLICATION_CHECKLIST.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/dependabot.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/pull_request_template.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/ci.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/docs.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/release.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.gitignore +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.pre-commit-config.yaml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/CODE_OF_CONDUCT.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/CONTRIBUTING.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/GEMINI.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/LICENSE +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/README.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/SECURITY.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/SUPPORT.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/VISION.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/README.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/benchmark_detectors.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/benchmark_engine.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/datasets/synthetic_messages.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0001-dependency-free-core.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0002-token-format.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0003-no-reversible-storage.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0004-optional-ml-backend.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/api.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/architecture.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/benchmarks.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/compatibility.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/contributing.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/deployment.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/detectors.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/index.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/limitations.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/llm-gateways.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/llm-payloads.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/migration-a2.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/migration-a3.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/quickstart.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0a2.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0a3.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0b1.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0rc1.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/security.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/threat-model.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/examples/llm_gateway.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/mkdocs.yml +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/models/README.md +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/smoke_wheel.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/verify_release.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/adapters.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/api.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/base.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/composite.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/ml/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/rules.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/cli.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/base.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/email.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/iban.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/ip_address.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/payment_card.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/phone.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/registry.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/secret.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/exceptions.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/processing.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/py.typed +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/resolution.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/result.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/__init__.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/alias.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/base.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/hmac.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/mode.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/placeholder.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/redact.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/de.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/en.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/files.json +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/it.jsonl +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_builtin_files.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_cli.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_cross_platform_corpus.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_file_processing.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_llm_gateway_example.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_public_api.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_release_stress.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_release_verifier.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/property/test_invariants.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_backends.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_cli_coverage.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_engine.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_modes.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_normalization.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_processing.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_transforms.py +0 -0
- {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_types.py +0 -0
|
@@ -92,7 +92,24 @@ PyPI upload verified:
|
|
|
92
92
|
Status:
|
|
93
93
|
```
|
|
94
94
|
|
|
95
|
-
### 0.2.
|
|
95
|
+
### 0.2.1: Performance and default policy patches
|
|
96
|
+
|
|
97
|
+
Implementation started: 2026-08-22
|
|
98
|
+
Starting commit: aabba9a
|
|
99
|
+
Starting PyPI version: 0.2.0
|
|
100
|
+
Baseline downloads day/week/month: 0/0/0
|
|
101
|
+
Baseline stars/forks/watchers: 0/0/0
|
|
102
|
+
Files changed: 13
|
|
103
|
+
Compatibility tests added: Yes
|
|
104
|
+
Artifact smoke environments: Win local isolated wheel
|
|
105
|
+
Known limitations:
|
|
106
|
+
Deferred items:
|
|
107
|
+
Release commit: TBD
|
|
108
|
+
Tag: v0.2.1
|
|
109
|
+
PyPI upload verified: Pending workflow
|
|
110
|
+
7-day metrics:
|
|
111
|
+
28-day metrics:
|
|
112
|
+
Status: Released
|
|
96
113
|
|
|
97
114
|
Implementation started: 2026-08-21
|
|
98
115
|
Starting commit: 3205c22ef6143aa8ae38fa2db9eea64fc237777d
|
|
@@ -111,3 +128,22 @@ PyPI upload verified: Pending workflow
|
|
|
111
128
|
28-day metrics:
|
|
112
129
|
Status: Released
|
|
113
130
|
|
|
131
|
+
### 0.3.0: Document inspection
|
|
132
|
+
|
|
133
|
+
Implementation started: 2026-08-22
|
|
134
|
+
Starting commit: 54a8b7d97ef1f61ebae722d902fa2d31797efb0e
|
|
135
|
+
Starting PyPI version: 0.2.1
|
|
136
|
+
Baseline downloads day/week/month: 0/0/0
|
|
137
|
+
Baseline stars/forks/watchers: 0/0/0
|
|
138
|
+
Files changed: 10
|
|
139
|
+
Compatibility tests added: Yes
|
|
140
|
+
Artifact smoke environments: Win local isolated wheel
|
|
141
|
+
Known limitations: Extract-only; no format-preserving rewriting yet.
|
|
142
|
+
Deferred items: Format-preserving rewrite deferred to 0.4.0.
|
|
143
|
+
Release commit: TBD
|
|
144
|
+
Tag: v0.3.0
|
|
145
|
+
PyPI upload verified: Pending workflow
|
|
146
|
+
7-day metrics:
|
|
147
|
+
28-day metrics:
|
|
148
|
+
Status: Ready
|
|
149
|
+
|
|
@@ -49,16 +49,17 @@ jobs:
|
|
|
49
49
|
- name: Install and audit wheel
|
|
50
50
|
if: runner.os != 'Windows'
|
|
51
51
|
run: |
|
|
52
|
-
uv pip install --python .wheel-venv/bin/python dist
|
|
52
|
+
uv pip install --python .wheel-venv/bin/python dist/*.whl
|
|
53
53
|
uv pip check --python .wheel-venv/bin/python
|
|
54
|
-
.wheel-venv/bin/python -I scripts/audit_install.py
|
|
54
|
+
.wheel-venv/bin/python -I scripts/audit_install.py
|
|
55
55
|
.wheel-venv/bin/python -I scripts/smoke_wheel.py
|
|
56
56
|
.wheel-venv/bin/pseudonymize detectors
|
|
57
57
|
- name: Install and audit wheel on Windows
|
|
58
58
|
if: runner.os == 'Windows'
|
|
59
|
+
shell: bash
|
|
59
60
|
run: |
|
|
60
|
-
uv pip install --python .wheel-venv
|
|
61
|
-
uv pip check --python .wheel-venv
|
|
62
|
-
.wheel-venv
|
|
63
|
-
.wheel-venv
|
|
64
|
-
.wheel-venv
|
|
61
|
+
uv pip install --python .wheel-venv/Scripts/python.exe dist/*.whl
|
|
62
|
+
uv pip check --python .wheel-venv/Scripts/python.exe
|
|
63
|
+
.wheel-venv/Scripts/python.exe -I scripts/audit_install.py
|
|
64
|
+
.wheel-venv/Scripts/python.exe -I scripts/smoke_wheel.py
|
|
65
|
+
.wheel-venv/Scripts/pseudonymize.exe detectors
|
|
@@ -4,6 +4,41 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.3.0] - 2026-08-22
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Added `pdf` extra (using `pdfminer.six`) for PDF text and coordinate extraction.
|
|
12
|
+
- Added `office` extra (using `python-docx`, `openpyxl`, `python-pptx`) for DOCX, XLSX, and PPTX inspection.
|
|
13
|
+
- Introduced `CoordinateLocation` and `StructuralLocation` for granular source tracking in complex file formats.
|
|
14
|
+
|
|
15
|
+
## [0.2.1] - 2026-08-22
|
|
16
|
+
|
|
17
|
+
### Changed
|
|
18
|
+
|
|
19
|
+
- `Policy.default()` now includes `URL_CREDENTIAL` and `SECRET`, so the documented secret and
|
|
20
|
+
URL-credential detectors act without opting into `Policy.strict()` or `Policy.llm()`. Callers
|
|
21
|
+
that relied on secrets passing through the default policy must configure an explicit
|
|
22
|
+
`Policy(entity_types=...)`.
|
|
23
|
+
- Overlap resolution now uses an ordered-interval scan instead of a quadratic pairwise check,
|
|
24
|
+
and detected spans are replaced with a single-pass segment join. Texts with thousands of
|
|
25
|
+
detections process in milliseconds instead of seconds.
|
|
26
|
+
|
|
27
|
+
### Fixed
|
|
28
|
+
|
|
29
|
+
- `LocalONNXPIIBackend` now merges contiguous token predictions into whole entity spans, so
|
|
30
|
+
multi-word and subword-split names receive a single alias instead of one alias per token.
|
|
31
|
+
- ONNX test artifacts downloaded during testing are now verified against pinned SHA-256
|
|
32
|
+
checksums.
|
|
33
|
+
- URL credential detection now covers the whole userinfo section when it contains extra `@`
|
|
34
|
+
characters and no longer swallows the URL fragment into sensitive query values.
|
|
35
|
+
- Overlap resolution now ranks URL credentials above emails, so a `user:password@host` userinfo
|
|
36
|
+
can no longer lose its `user:` prefix to an overlapping email match.
|
|
37
|
+
- The `ml` extra now declares `numpy`, which the ONNX backend imports directly, and no longer
|
|
38
|
+
pulls in the unused `huggingface-hub` dependency.
|
|
39
|
+
- The optional-import guard in the ONNX backend now type-checks under strict mypy.
|
|
40
|
+
- Normalization now maps generic URL schemes to `http` or `https` for safer alias sharing.
|
|
41
|
+
|
|
7
42
|
## [0.2.0] - 2026-08-21
|
|
8
43
|
|
|
9
44
|
### Added
|
|
@@ -135,7 +170,8 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
|
|
|
135
170
|
- Deterministic processing now requires `mode="deterministic"` in addition to a key.
|
|
136
171
|
- `redact()` now emits `[REDACTED]` by default; generic mode emits typed placeholders.
|
|
137
172
|
|
|
138
|
-
[Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.
|
|
173
|
+
[Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.1...HEAD
|
|
174
|
+
[0.2.1]: https://github.com/ma2za/pseudonymize/compare/v0.2.0...v0.2.1
|
|
139
175
|
[0.2.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.2.0
|
|
140
176
|
[0.1.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0
|
|
141
177
|
[0.1.0rc1]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0rc1
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pseudonymize
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Local-first PII pseudonymization for text, structured data, and LLM payloads.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ma2za/pseudonymize
|
|
6
6
|
Project-URL: Documentation, https://ma2za.github.io/pseudonymize/
|
|
@@ -28,6 +28,12 @@ Provides-Extra: ml
|
|
|
28
28
|
Requires-Dist: huggingface-hub>=1.28.0; extra == 'ml'
|
|
29
29
|
Requires-Dist: onnxruntime>=1.29.0; extra == 'ml'
|
|
30
30
|
Requires-Dist: tokenizers>=0.23.1; extra == 'ml'
|
|
31
|
+
Provides-Extra: office
|
|
32
|
+
Requires-Dist: openpyxl>=3.1.0; extra == 'office'
|
|
33
|
+
Requires-Dist: python-docx>=1.1.0; extra == 'office'
|
|
34
|
+
Requires-Dist: python-pptx>=1.0.0; extra == 'office'
|
|
35
|
+
Provides-Extra: pdf
|
|
36
|
+
Requires-Dist: pdfminer-six>=20240706; extra == 'pdf'
|
|
31
37
|
Description-Content-Type: text/markdown
|
|
32
38
|
|
|
33
39
|
# Pseudonymize
|
|
@@ -13,7 +13,8 @@ publishable without requiring unfinished later layers.
|
|
|
13
13
|
| `0.1.0b1` | Published | Core API freeze and production-oriented examples |
|
|
14
14
|
| `0.1.0rc1` | Published | External installation and release validation |
|
|
15
15
|
| `0.1.0` | Published | First stable text and machine-readable release |
|
|
16
|
-
| `0.2.0` |
|
|
16
|
+
| `0.2.0` | Published | Optional local machine learning identification for PII |
|
|
17
|
+
| `0.3.0` | Next | Document inspection (PDF, DOCX, XLSX, PPTX) |
|
|
17
18
|
|
|
18
19
|
Alpha releases optimize for the cleanest safe architecture, not backward compatibility. They may
|
|
19
20
|
remove, rename, or replace public APIs without aliases or shims. Material changes are documented,
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
# Policies
|
|
2
2
|
|
|
3
|
-
`Policy.default()` enables structured identifiers
|
|
4
|
-
when a configured backend can detect them.
|
|
5
|
-
|
|
6
|
-
`Policy.financial()` limits detection to IBANs and payment
|
|
3
|
+
`Policy.default()` enables structured identifiers, secrets, URL credentials, and person,
|
|
4
|
+
organization, and location entities when a configured backend can detect them.
|
|
5
|
+
`Policy.strict()` enables the same entities with a lower confidence threshold. `Policy.llm()`
|
|
6
|
+
enables all available entities. `Policy.financial()` limits detection to IBANs and payment
|
|
7
|
+
cards.
|
|
7
8
|
|
|
8
9
|
Paths are dot-separated segments. `*` matches one segment, so `messages.*.content` matches list
|
|
9
10
|
indices. Exclusions win over inclusions. Dictionary keys are not transformed.
|
|
@@ -29,10 +29,18 @@ After the first upload, verify that the pending publisher became an ordinary pro
|
|
|
29
29
|
uv run ruff check .
|
|
30
30
|
uv run mypy
|
|
31
31
|
uv run pytest
|
|
32
|
-
uv run mkdocs build --strict
|
|
32
|
+
uv run python -m mkdocs build --strict
|
|
33
33
|
uv build
|
|
34
34
|
uv run twine check dist/*
|
|
35
35
|
uv run python scripts/verify_release.py
|
|
36
|
+
|
|
37
|
+
# Verify the built wheel locally in an isolated environment (simulating CI)
|
|
38
|
+
uv venv --python 3.14 .wheel-venv
|
|
39
|
+
uv pip install --python .wheel-venv/bin/python dist/*.whl
|
|
40
|
+
uv pip check --python .wheel-venv/bin/python
|
|
41
|
+
.wheel-venv/bin/python -I scripts/audit_install.py
|
|
42
|
+
.wheel-venv/bin/python -I scripts/smoke_wheel.py
|
|
43
|
+
.wheel-venv/bin/pseudonymize detectors
|
|
36
44
|
```
|
|
37
45
|
|
|
38
46
|
4. Open a pull request and merge only after every required check passes.
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pseudonymize"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "Local-first PII pseudonymization for text, structured data, and LLM payloads."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11"
|
|
@@ -43,12 +43,20 @@ ml = [
|
|
|
43
43
|
"onnxruntime>=1.29.0",
|
|
44
44
|
"tokenizers>=0.23.1",
|
|
45
45
|
]
|
|
46
|
+
office = [
|
|
47
|
+
"openpyxl>=3.1.0",
|
|
48
|
+
"python-docx>=1.1.0",
|
|
49
|
+
"python-pptx>=1.0.0",
|
|
50
|
+
]
|
|
51
|
+
pdf = [
|
|
52
|
+
"pdfminer.six>=20240706",
|
|
53
|
+
]
|
|
46
54
|
|
|
47
55
|
[tool.hatch.build.targets.wheel]
|
|
48
56
|
packages = ["src/pseudonymize"]
|
|
49
57
|
|
|
50
58
|
[dependency-groups]
|
|
51
|
-
test = ["hypothesis>=6", "pytest>=8", "pytest-benchmark>=5", "pytest-cov>=5"]
|
|
59
|
+
test = ["hypothesis>=6", "pytest>=8", "pytest-benchmark>=5", "pytest-cov>=5", "fpdf2>=2.7.9"]
|
|
52
60
|
quality = ["mypy>=1.11", "pre-commit>=4", "ruff>=0.9"]
|
|
53
61
|
docs = ["mkdocs-material>=9", "mkdocstrings[python]>=1.0.6"]
|
|
54
62
|
release = ["build>=1", "twine>=6"]
|
|
@@ -81,7 +89,11 @@ files = ["src", "tests", "scripts"]
|
|
|
81
89
|
module = [
|
|
82
90
|
"numpy.*",
|
|
83
91
|
"onnxruntime.*",
|
|
84
|
-
"tokenizers.*"
|
|
92
|
+
"tokenizers.*",
|
|
93
|
+
"pdfminer.*",
|
|
94
|
+
"docx.*",
|
|
95
|
+
"pptx.*",
|
|
96
|
+
"openpyxl.*"
|
|
85
97
|
]
|
|
86
98
|
ignore_missing_imports = true
|
|
87
99
|
follow_imports = "skip"
|
|
@@ -91,7 +103,7 @@ minversion = "8"
|
|
|
91
103
|
testpaths = ["tests"]
|
|
92
104
|
pythonpath = ["."]
|
|
93
105
|
python_files = ["test_*.py", "benchmark_*.py"]
|
|
94
|
-
addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36"]
|
|
106
|
+
addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36", "--basetemp=.pytest_temp"]
|
|
95
107
|
markers = ["benchmark: performance benchmark", "integration: integration test", "slow: deliberately slow test"]
|
|
96
108
|
filterwarnings = ["error"]
|
|
97
109
|
|
|
@@ -27,11 +27,18 @@ def _blocked_network(*arguments: object, **keywords: object) -> None:
|
|
|
27
27
|
|
|
28
28
|
def main() -> None:
|
|
29
29
|
parser = argparse.ArgumentParser()
|
|
30
|
-
parser.add_argument("--version", required=
|
|
30
|
+
parser.add_argument("--version", required=False)
|
|
31
31
|
arguments = parser.parse_args()
|
|
32
32
|
|
|
33
|
+
expected_version = arguments.version
|
|
34
|
+
if not expected_version:
|
|
35
|
+
import tomllib
|
|
36
|
+
|
|
37
|
+
with open("pyproject.toml", "rb") as stream:
|
|
38
|
+
expected_version = tomllib.load(stream)["project"]["version"]
|
|
39
|
+
|
|
33
40
|
installed = distribution("pseudonymize")
|
|
34
|
-
if installed.version !=
|
|
41
|
+
if installed.version != expected_version:
|
|
35
42
|
raise RuntimeError("installed version does not match release")
|
|
36
43
|
if installed.requires:
|
|
37
44
|
required_deps = [req for req in installed.requires if "extra ==" not in req]
|
|
@@ -25,6 +25,20 @@ else:
|
|
|
25
25
|
ort = None
|
|
26
26
|
Tokenizer = None
|
|
27
27
|
|
|
28
|
+
_LABEL_SUFFIXES: tuple[tuple[tuple[str, ...], EntityType], ...] = (
|
|
29
|
+
# Standard CoNLL-03 suffixes plus Ai4Privacy fine-grained labels.
|
|
30
|
+
(("PER", "FIRSTNAME", "LASTNAME", "MIDDLENAME"), EntityType.PERSON),
|
|
31
|
+
(("ORG", "COMPANYNAME"), EntityType.ORGANIZATION),
|
|
32
|
+
(("LOC", "CITY", "STATE", "COUNTY", "STREET", "ZIPCODE"), EntityType.LOCATION),
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _entity_type_for(label: str) -> EntityType | None:
|
|
37
|
+
for suffixes, entity_type in _LABEL_SUFFIXES:
|
|
38
|
+
if label.endswith(suffixes):
|
|
39
|
+
return entity_type
|
|
40
|
+
return None
|
|
41
|
+
|
|
28
42
|
|
|
29
43
|
class LocalONNXPIIBackend(DetectionBackend):
|
|
30
44
|
def __init__(
|
|
@@ -111,55 +125,44 @@ class LocalONNXPIIBackend(DetectionBackend):
|
|
|
111
125
|
|
|
112
126
|
predictions = np.argmax(logits, axis=-1)
|
|
113
127
|
|
|
114
|
-
|
|
128
|
+
# Token predictions are merged into entity spans: subword continuations
|
|
129
|
+
# (zero gap) and same-type tokens separated by one whitespace character
|
|
130
|
+
# collapse into a single detection so that "John Smith" is one PERSON.
|
|
131
|
+
spans: list[tuple[EntityType, int, int]] = []
|
|
115
132
|
|
|
116
133
|
for idx, label_id in enumerate(predictions):
|
|
117
|
-
|
|
118
|
-
continue
|
|
119
|
-
|
|
120
|
-
label_str = self._id2label.get(label_id)
|
|
134
|
+
label_str = (self._id2label or {}).get(int(label_id))
|
|
121
135
|
if not label_str or label_str == "O":
|
|
122
136
|
continue
|
|
123
|
-
|
|
124
|
-
entity_type
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
if
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
entity_type = EntityType.ORGANIZATION
|
|
139
|
-
elif any(
|
|
140
|
-
label_str.endswith(s) for s in ("CITY", "STATE", "COUNTY", "STREET", "ZIPCODE")
|
|
141
|
-
):
|
|
142
|
-
entity_type = EntityType.LOCATION
|
|
143
|
-
|
|
144
|
-
if entity_type:
|
|
145
|
-
start, end = encoding.offsets[idx]
|
|
146
|
-
if start == 0 and end == 0 and idx not in (0, len(predictions) - 1):
|
|
147
|
-
continue
|
|
148
|
-
if start == end:
|
|
137
|
+
entity_type = _entity_type_for(label_str)
|
|
138
|
+
if entity_type is None:
|
|
139
|
+
continue
|
|
140
|
+
start, end = encoding.offsets[idx]
|
|
141
|
+
if start >= end:
|
|
142
|
+
continue
|
|
143
|
+
if spans:
|
|
144
|
+
previous_type, previous_start, previous_end = spans[-1]
|
|
145
|
+
gap = block.text[previous_end:start]
|
|
146
|
+
if (
|
|
147
|
+
entity_type is previous_type
|
|
148
|
+
and previous_end <= start <= previous_end + 1
|
|
149
|
+
and not gap.strip()
|
|
150
|
+
):
|
|
151
|
+
spans[-1] = (previous_type, previous_start, end)
|
|
149
152
|
continue
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
153
|
+
spans.append((entity_type, start, end))
|
|
154
|
+
|
|
155
|
+
return tuple(
|
|
156
|
+
Detection(
|
|
157
|
+
entity_type=entity_type,
|
|
158
|
+
start=start,
|
|
159
|
+
end=end,
|
|
160
|
+
confidence=0.99,
|
|
161
|
+
backend=self.name,
|
|
162
|
+
detector="onnx",
|
|
163
|
+
)
|
|
164
|
+
for entity_type, start, end in spans
|
|
165
|
+
)
|
|
163
166
|
|
|
164
167
|
except Exception as e:
|
|
165
168
|
raise BackendExecutionError(f"ONNX PII inference failed: {e}") from e
|
|
@@ -10,6 +10,11 @@ _SENSITIVE_QUERY_NAMES = frozenset(
|
|
|
10
10
|
)
|
|
11
11
|
|
|
12
12
|
|
|
13
|
+
def _authority_end(url: str, authority_start: int) -> int:
|
|
14
|
+
indexes = (url.find(character, authority_start) for character in "/?#")
|
|
15
|
+
return min((index for index in indexes if index != -1), default=len(url))
|
|
16
|
+
|
|
17
|
+
|
|
13
18
|
@dataclass(frozen=True, slots=True)
|
|
14
19
|
class UrlDetector:
|
|
15
20
|
name: str = "url"
|
|
@@ -21,7 +26,8 @@ class UrlDetector:
|
|
|
21
26
|
parsed = urllib.parse.urlsplit(url)
|
|
22
27
|
if parsed.username is not None or parsed.password is not None:
|
|
23
28
|
authority_start = url.find("//") + 2
|
|
24
|
-
|
|
29
|
+
authority_end = _authority_end(url, authority_start)
|
|
30
|
+
at = url.rfind("@", authority_start, authority_end)
|
|
25
31
|
detections.append(
|
|
26
32
|
Detection(
|
|
27
33
|
EntityType.URL_CREDENTIAL,
|
|
@@ -34,8 +40,10 @@ class UrlDetector:
|
|
|
34
40
|
query_start = url.find("?") + 1
|
|
35
41
|
if query_start == 0:
|
|
36
42
|
continue
|
|
43
|
+
fragment_start = url.find("#", query_start)
|
|
44
|
+
query_end = fragment_start if fragment_start != -1 else len(url)
|
|
37
45
|
cursor = query_start
|
|
38
|
-
for item in url[query_start:].split("&"):
|
|
46
|
+
for item in url[query_start:query_end].split("&"):
|
|
39
47
|
name, separator, value = item.partition("=")
|
|
40
48
|
if separator and urllib.parse.unquote_plus(name).lower() in _SENSITIVE_QUERY_NAMES:
|
|
41
49
|
value_start = cursor + len(name) + 1
|
|
@@ -68,7 +68,43 @@ class CSVCellLocation:
|
|
|
68
68
|
raise ValueError("CSV row and column indexes must be non-negative")
|
|
69
69
|
|
|
70
70
|
|
|
71
|
-
|
|
71
|
+
@dataclass(frozen=True, slots=True)
|
|
72
|
+
class CoordinateLocation:
|
|
73
|
+
page: int
|
|
74
|
+
x0: float
|
|
75
|
+
y0: float
|
|
76
|
+
x1: float
|
|
77
|
+
y1: float
|
|
78
|
+
|
|
79
|
+
def __post_init__(self) -> None:
|
|
80
|
+
if not isinstance(self.page, int) or isinstance(self.page, bool) or self.page < 0:
|
|
81
|
+
raise ValueError("page must be a non-negative integer")
|
|
82
|
+
for val in (self.x0, self.y0, self.x1, self.y1):
|
|
83
|
+
if not isinstance(val, (int, float)) or isinstance(val, bool) or not isfinite(val):
|
|
84
|
+
raise TypeError("coordinates must be finite numbers")
|
|
85
|
+
if self.x0 > self.x1 or self.y0 > self.y1:
|
|
86
|
+
raise ValueError("coordinates must describe a valid bounding box")
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@dataclass(frozen=True, slots=True)
|
|
90
|
+
class StructuralLocation:
|
|
91
|
+
path: tuple[str | int, ...]
|
|
92
|
+
|
|
93
|
+
def __post_init__(self) -> None:
|
|
94
|
+
object.__setattr__(self, "path", tuple(self.path))
|
|
95
|
+
if any(not isinstance(part, (str, int)) or isinstance(part, bool) for part in self.path):
|
|
96
|
+
raise TypeError("structural path parts must be strings or integers")
|
|
97
|
+
if any(isinstance(part, int) and part < 0 for part in self.path):
|
|
98
|
+
raise ValueError("structural path indexes must be non-negative")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
SourceLocation: TypeAlias = (
|
|
102
|
+
TextOffsetLocation
|
|
103
|
+
| JSONPathLocation
|
|
104
|
+
| CSVCellLocation
|
|
105
|
+
| CoordinateLocation
|
|
106
|
+
| StructuralLocation
|
|
107
|
+
)
|
|
72
108
|
|
|
73
109
|
|
|
74
110
|
@dataclass(frozen=True, slots=True)
|
|
@@ -83,7 +119,16 @@ class ContentBlock:
|
|
|
83
119
|
raise ValueError("block id must not be empty")
|
|
84
120
|
if not isinstance(self.text, str):
|
|
85
121
|
raise TypeError("block text must be a string")
|
|
86
|
-
if not isinstance(
|
|
122
|
+
if not isinstance(
|
|
123
|
+
self.location,
|
|
124
|
+
(
|
|
125
|
+
TextOffsetLocation,
|
|
126
|
+
JSONPathLocation,
|
|
127
|
+
CSVCellLocation,
|
|
128
|
+
CoordinateLocation,
|
|
129
|
+
StructuralLocation,
|
|
130
|
+
),
|
|
131
|
+
):
|
|
87
132
|
raise TypeError("block location must be a supported source location")
|
|
88
133
|
object.__setattr__(self, "metadata", _metadata(self.metadata))
|
|
89
134
|
|
|
@@ -265,10 +265,15 @@ class Pseudonymizer:
|
|
|
265
265
|
self.transformer.render(entity, alias)
|
|
266
266
|
for entity, alias in zip(entities, aliases, strict=True)
|
|
267
267
|
)
|
|
268
|
-
|
|
269
|
-
|
|
268
|
+
segments: list[str] = []
|
|
269
|
+
cursor = 0
|
|
270
|
+
for entity, token in zip(entities, tokens, strict=True):
|
|
270
271
|
detection = entity.detection
|
|
271
|
-
|
|
272
|
+
segments.append(text[cursor : detection.start])
|
|
273
|
+
segments.append(token)
|
|
274
|
+
cursor = detection.end
|
|
275
|
+
segments.append(text[cursor:])
|
|
276
|
+
output = "".join(segments)
|
|
272
277
|
replacements = _replacement_reports(entities, tokens)
|
|
273
278
|
reports.extend(_replacement_detection_reports(block, replacements))
|
|
274
279
|
mapping = _mapping(text, entities, aliases, tokens) if include_mapping else None
|
|
@@ -287,10 +292,10 @@ class Pseudonymizer:
|
|
|
287
292
|
if isinstance(data, str):
|
|
288
293
|
block_id = f"block-{block_counter[0]:06d}"
|
|
289
294
|
block_counter[0] += 1
|
|
290
|
-
block = ContentBlock(block_id, data, JSONPathLocation(path))
|
|
291
295
|
if not self.policy.allows_path(tuple(str(part) for part in path)):
|
|
292
296
|
statistics.blocks_processed += 1
|
|
293
297
|
return data
|
|
298
|
+
block = ContentBlock(block_id, data, JSONPathLocation(path))
|
|
294
299
|
return self._process_block(block, context, False, statistics, reports).text
|
|
295
300
|
if data is None or isinstance(data, (bool, int, float)):
|
|
296
301
|
return data
|
|
@@ -462,8 +467,20 @@ def _processing_adapters(
|
|
|
462
467
|
output_adapter: OutputAdapter | None,
|
|
463
468
|
) -> tuple[InputAdapter[Path], OutputAdapter]:
|
|
464
469
|
if input_adapter is None and output_adapter is None:
|
|
465
|
-
|
|
466
|
-
|
|
470
|
+
selected_format = select_file_format(source, format)
|
|
471
|
+
if selected_format == FileFormat.PDF:
|
|
472
|
+
from pseudonymize.inspection.pdf import PDFInspectionAdapter
|
|
473
|
+
|
|
474
|
+
pdf_adapter = PDFInspectionAdapter()
|
|
475
|
+
return pdf_adapter, pdf_adapter
|
|
476
|
+
elif selected_format in (FileFormat.DOCX, FileFormat.XLSX, FileFormat.PPTX):
|
|
477
|
+
from pseudonymize.inspection.office import OfficeInspectionAdapter
|
|
478
|
+
|
|
479
|
+
office_adapter = OfficeInspectionAdapter(selected_format)
|
|
480
|
+
return office_adapter, office_adapter
|
|
481
|
+
else:
|
|
482
|
+
builtin_adapter = BuiltinFileAdapter(selected_format, encoding)
|
|
483
|
+
return builtin_adapter, builtin_adapter
|
|
467
484
|
if input_adapter is None or output_adapter is None:
|
|
468
485
|
raise ValueError("custom file processing requires input and output adapters")
|
|
469
486
|
if format is not None or encoding is not None:
|
|
@@ -478,7 +495,17 @@ def _inspection_adapter(
|
|
|
478
495
|
input_adapter: InputAdapter[Path] | None,
|
|
479
496
|
) -> InputAdapter[Path]:
|
|
480
497
|
if input_adapter is None:
|
|
481
|
-
|
|
498
|
+
selected_format = select_file_format(source, format)
|
|
499
|
+
if selected_format == FileFormat.PDF:
|
|
500
|
+
from pseudonymize.inspection.pdf import PDFInspectionAdapter
|
|
501
|
+
|
|
502
|
+
return PDFInspectionAdapter()
|
|
503
|
+
elif selected_format in (FileFormat.DOCX, FileFormat.XLSX, FileFormat.PPTX):
|
|
504
|
+
from pseudonymize.inspection.office import OfficeInspectionAdapter
|
|
505
|
+
|
|
506
|
+
return OfficeInspectionAdapter(selected_format)
|
|
507
|
+
else:
|
|
508
|
+
return BuiltinFileAdapter(selected_format, encoding)
|
|
482
509
|
if format is not None or encoding is not None:
|
|
483
510
|
raise ValueError("custom adapters cannot be combined with format or encoding")
|
|
484
511
|
return input_adapter
|