pseudonymize 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.agents/PRIVATE_RELEASE_PLAN.md +37 -1
  2. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/package.yml +8 -7
  3. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/CHANGELOG.md +37 -1
  4. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/PKG-INFO +7 -1
  5. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/ROADMAP.md +2 -1
  6. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/policies.md +5 -4
  7. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releasing.md +9 -1
  8. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/pyproject.toml +16 -4
  9. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/audit_install.py +9 -2
  10. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/ml/onnx.py +47 -44
  11. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/url.py +10 -2
  12. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/document.py +47 -2
  13. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/engine.py +34 -7
  14. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/formats.py +30 -40
  15. pseudonymize-0.3.0/src/pseudonymize/inspection/__init__.py +4 -0
  16. pseudonymize-0.3.0/src/pseudonymize/inspection/office.py +112 -0
  17. pseudonymize-0.3.0/src/pseudonymize/inspection/pdf.py +61 -0
  18. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/normalization.py +4 -5
  19. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/policy.py +2 -1
  20. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/spans.py +13 -5
  21. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/compatibility/test_b1_contract.py +4 -0
  22. pseudonymize-0.3.0/tests/integration/test_inspection_adapters.py +141 -0
  23. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/backends/test_onnx.py +52 -21
  24. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/detectors/test_detectors.py +18 -0
  25. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_document.py +18 -1
  26. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_formats.py +1 -1
  27. pseudonymize-0.3.0/tests/unit/test_inspection.py +60 -0
  28. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_policy.py +2 -1
  29. pseudonymize-0.3.0/tests/unit/test_spans.py +39 -0
  30. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/uv.lock +439 -2
  31. pseudonymize-0.2.0/tests/unit/test_spans.py +0 -15
  32. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.editorconfig +0 -0
  33. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.gitattributes +0 -0
  34. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/CODEOWNERS +0 -0
  35. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/bug.yml +0 -0
  36. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/config.yml +0 -0
  37. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/detector-request.yml +0 -0
  38. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/feature.yml +0 -0
  39. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/ISSUE_TEMPLATE/question.yml +0 -0
  40. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/PUBLICATION_CHECKLIST.md +0 -0
  41. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/dependabot.yml +0 -0
  42. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/pull_request_template.md +0 -0
  43. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/ci.yml +0 -0
  44. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/docs.yml +0 -0
  45. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.github/workflows/release.yml +0 -0
  46. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.gitignore +0 -0
  47. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/.pre-commit-config.yaml +0 -0
  48. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/CODE_OF_CONDUCT.md +0 -0
  49. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/CONTRIBUTING.md +0 -0
  50. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/GEMINI.md +0 -0
  51. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/LICENSE +0 -0
  52. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/README.md +0 -0
  53. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/SECURITY.md +0 -0
  54. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/SUPPORT.md +0 -0
  55. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/VISION.md +0 -0
  56. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/README.md +0 -0
  57. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/benchmark_detectors.py +0 -0
  58. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/benchmark_engine.py +0 -0
  59. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/benchmarks/datasets/synthetic_messages.jsonl +0 -0
  60. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0001-dependency-free-core.md +0 -0
  61. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0002-token-format.md +0 -0
  62. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0003-no-reversible-storage.md +0 -0
  63. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/adr/0004-optional-ml-backend.md +0 -0
  64. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/api.md +0 -0
  65. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/architecture.md +0 -0
  66. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/benchmarks.md +0 -0
  67. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/compatibility.md +0 -0
  68. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/contributing.md +0 -0
  69. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/deployment.md +0 -0
  70. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/detectors.md +0 -0
  71. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/index.md +0 -0
  72. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/limitations.md +0 -0
  73. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/llm-gateways.md +0 -0
  74. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/llm-payloads.md +0 -0
  75. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/migration-a2.md +0 -0
  76. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/migration-a3.md +0 -0
  77. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/quickstart.md +0 -0
  78. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0.md +0 -0
  79. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0a2.md +0 -0
  80. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0a3.md +0 -0
  81. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0b1.md +0 -0
  82. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/releases/0.1.0rc1.md +0 -0
  83. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/security.md +0 -0
  84. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/docs/threat-model.md +0 -0
  85. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/examples/llm_gateway.py +0 -0
  86. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/mkdocs.yml +0 -0
  87. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/models/README.md +0 -0
  88. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/__init__.py +0 -0
  89. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/smoke_wheel.py +0 -0
  90. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/scripts/verify_release.py +0 -0
  91. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/__init__.py +0 -0
  92. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/adapters.py +0 -0
  93. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/api.py +0 -0
  94. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/__init__.py +0 -0
  95. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/base.py +0 -0
  96. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/composite.py +0 -0
  97. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/ml/__init__.py +0 -0
  98. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/backends/rules.py +0 -0
  99. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/cli.py +0 -0
  100. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/__init__.py +0 -0
  101. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/base.py +0 -0
  102. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/email.py +0 -0
  103. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/iban.py +0 -0
  104. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/ip_address.py +0 -0
  105. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/payment_card.py +0 -0
  106. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/phone.py +0 -0
  107. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/registry.py +0 -0
  108. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/detectors/secret.py +0 -0
  109. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/exceptions.py +0 -0
  110. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/processing.py +0 -0
  111. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/py.typed +0 -0
  112. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/resolution.py +0 -0
  113. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/result.py +0 -0
  114. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/__init__.py +0 -0
  115. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/alias.py +0 -0
  116. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/base.py +0 -0
  117. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/hmac.py +0 -0
  118. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/mode.py +0 -0
  119. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/placeholder.py +0 -0
  120. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/src/pseudonymize/transforms/redact.py +0 -0
  121. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/de.jsonl +0 -0
  122. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/en.jsonl +0 -0
  123. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/files.json +0 -0
  124. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/corpus/it.jsonl +0 -0
  125. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_builtin_files.py +0 -0
  126. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_cli.py +0 -0
  127. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_cross_platform_corpus.py +0 -0
  128. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_file_processing.py +0 -0
  129. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_llm_gateway_example.py +0 -0
  130. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_public_api.py +0 -0
  131. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_release_stress.py +0 -0
  132. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/integration/test_release_verifier.py +0 -0
  133. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/property/test_invariants.py +0 -0
  134. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_backends.py +0 -0
  135. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_cli_coverage.py +0 -0
  136. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_engine.py +0 -0
  137. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_modes.py +0 -0
  138. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_normalization.py +0 -0
  139. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_processing.py +0 -0
  140. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_transforms.py +0 -0
  141. {pseudonymize-0.2.0 → pseudonymize-0.3.0}/tests/unit/test_types.py +0 -0
@@ -92,7 +92,24 @@ PyPI upload verified:
92
92
  Status:
93
93
  ```
94
94
 
95
- ### 0.2.0: Optional local ML
95
+ ### 0.2.1: Performance and default policy patches
96
+
97
+ Implementation started: 2026-08-22
98
+ Starting commit: aabba9a
99
+ Starting PyPI version: 0.2.0
100
+ Baseline downloads day/week/month: 0/0/0
101
+ Baseline stars/forks/watchers: 0/0/0
102
+ Files changed: 13
103
+ Compatibility tests added: Yes
104
+ Artifact smoke environments: Win local isolated wheel
105
+ Known limitations:
106
+ Deferred items:
107
+ Release commit: TBD
108
+ Tag: v0.2.1
109
+ PyPI upload verified: Pending workflow
110
+ 7-day metrics:
111
+ 28-day metrics:
112
+ Status: Released
96
113
 
97
114
  Implementation started: 2026-08-21
98
115
  Starting commit: 3205c22ef6143aa8ae38fa2db9eea64fc237777d
@@ -111,3 +128,22 @@ PyPI upload verified: Pending workflow
111
128
  28-day metrics:
112
129
  Status: Released
113
130
 
131
+ ### 0.3.0: Document inspection
132
+
133
+ Implementation started: 2026-08-22
134
+ Starting commit: 54a8b7d97ef1f61ebae722d902fa2d31797efb0e
135
+ Starting PyPI version: 0.2.1
136
+ Baseline downloads day/week/month: 0/0/0
137
+ Baseline stars/forks/watchers: 0/0/0
138
+ Files changed: 10
139
+ Compatibility tests added: Yes
140
+ Artifact smoke environments: Win local isolated wheel
141
+ Known limitations: Extract-only; no format-preserving rewriting yet.
142
+ Deferred items: Format-preserving rewrite deferred to 0.4.0.
143
+ Release commit: TBD
144
+ Tag: v0.3.0
145
+ PyPI upload verified: Pending workflow
146
+ 7-day metrics:
147
+ 28-day metrics:
148
+ Status: Ready
149
+
@@ -49,16 +49,17 @@ jobs:
49
49
  - name: Install and audit wheel
50
50
  if: runner.os != 'Windows'
51
51
  run: |
52
- uv pip install --python .wheel-venv/bin/python dist/pseudonymize-0.2.0-py3-none-any.whl
52
+ uv pip install --python .wheel-venv/bin/python dist/*.whl
53
53
  uv pip check --python .wheel-venv/bin/python
54
- .wheel-venv/bin/python -I scripts/audit_install.py --version 0.2.0
54
+ .wheel-venv/bin/python -I scripts/audit_install.py
55
55
  .wheel-venv/bin/python -I scripts/smoke_wheel.py
56
56
  .wheel-venv/bin/pseudonymize detectors
57
57
  - name: Install and audit wheel on Windows
58
58
  if: runner.os == 'Windows'
59
+ shell: bash
59
60
  run: |
60
- uv pip install --python .wheel-venv\Scripts\python.exe dist\pseudonymize-0.2.0-py3-none-any.whl
61
- uv pip check --python .wheel-venv\Scripts\python.exe
62
- .wheel-venv\Scripts\python.exe -I scripts\audit_install.py --version 0.2.0
63
- .wheel-venv\Scripts\python.exe -I scripts\smoke_wheel.py
64
- .wheel-venv\Scripts\pseudonymize.exe detectors
61
+ uv pip install --python .wheel-venv/Scripts/python.exe dist/*.whl
62
+ uv pip check --python .wheel-venv/Scripts/python.exe
63
+ .wheel-venv/Scripts/python.exe -I scripts/audit_install.py
64
+ .wheel-venv/Scripts/python.exe -I scripts/smoke_wheel.py
65
+ .wheel-venv/Scripts/pseudonymize.exe detectors
@@ -4,6 +4,41 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [0.3.0] - 2026-08-22
8
+
9
+ ### Added
10
+
11
+ - Added `pdf` extra (using `pdfminer.six`) for PDF text and coordinate extraction.
12
+ - Added `office` extra (using `python-docx`, `openpyxl`, `python-pptx`) for DOCX, XLSX, and PPTX inspection.
13
+ - Introduced `CoordinateLocation` and `StructuralLocation` for granular source tracking in complex file formats.
14
+
15
+ ## [0.2.1] - 2026-08-22
16
+
17
+ ### Changed
18
+
19
+ - `Policy.default()` now includes `URL_CREDENTIAL` and `SECRET`, so the documented secret and
20
+ URL-credential detectors act without opting into `Policy.strict()` or `Policy.llm()`. Callers
21
+ that relied on secrets passing through the default policy must configure an explicit
22
+ `Policy(entity_types=...)`.
23
+ - Overlap resolution now uses an ordered-interval scan instead of a quadratic pairwise check,
24
+ and detected spans are replaced with a single-pass segment join. Texts with thousands of
25
+ detections process in milliseconds instead of seconds.
26
+
27
+ ### Fixed
28
+
29
+ - `LocalONNXPIIBackend` now merges contiguous token predictions into whole entity spans, so
30
+ multi-word and subword-split names receive a single alias instead of one alias per token.
31
+ - ONNX test artifacts downloaded during testing are now verified against pinned SHA-256
32
+ checksums.
33
+ - URL credential detection now covers the whole userinfo section when it contains extra `@`
34
+ characters and no longer swallows the URL fragment into sensitive query values.
35
+ - Overlap resolution now ranks URL credentials above emails, so a `user:password@host` userinfo
36
+ can no longer lose its `user:` prefix to an overlapping email match.
37
+ - The `ml` extra now declares `numpy`, which the ONNX backend imports directly, and no longer
38
+ pulls in the unused `huggingface-hub` dependency.
39
+ - The optional-import guard in the ONNX backend now type-checks under strict mypy.
40
+ - Normalization now maps generic URL schemes to `http` or `https` for safer alias sharing.
41
+
7
42
  ## [0.2.0] - 2026-08-21
8
43
 
9
44
  ### Added
@@ -135,7 +170,8 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
135
170
  - Deterministic processing now requires `mode="deterministic"` in addition to a key.
136
171
  - `redact()` now emits `[REDACTED]` by default; generic mode emits typed placeholders.
137
172
 
138
- [Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.0...HEAD
173
+ [Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.1...HEAD
174
+ [0.2.1]: https://github.com/ma2za/pseudonymize/compare/v0.2.0...v0.2.1
139
175
  [0.2.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.2.0
140
176
  [0.1.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0
141
177
  [0.1.0rc1]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0rc1
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pseudonymize
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Local-first PII pseudonymization for text, structured data, and LLM payloads.
5
5
  Project-URL: Homepage, https://github.com/ma2za/pseudonymize
6
6
  Project-URL: Documentation, https://ma2za.github.io/pseudonymize/
@@ -28,6 +28,12 @@ Provides-Extra: ml
28
28
  Requires-Dist: huggingface-hub>=1.28.0; extra == 'ml'
29
29
  Requires-Dist: onnxruntime>=1.29.0; extra == 'ml'
30
30
  Requires-Dist: tokenizers>=0.23.1; extra == 'ml'
31
+ Provides-Extra: office
32
+ Requires-Dist: openpyxl>=3.1.0; extra == 'office'
33
+ Requires-Dist: python-docx>=1.1.0; extra == 'office'
34
+ Requires-Dist: python-pptx>=1.0.0; extra == 'office'
35
+ Provides-Extra: pdf
36
+ Requires-Dist: pdfminer-six>=20240706; extra == 'pdf'
31
37
  Description-Content-Type: text/markdown
32
38
 
33
39
  # Pseudonymize
@@ -13,7 +13,8 @@ publishable without requiring unfinished later layers.
13
13
  | `0.1.0b1` | Published | Core API freeze and production-oriented examples |
14
14
  | `0.1.0rc1` | Published | External installation and release validation |
15
15
  | `0.1.0` | Published | First stable text and machine-readable release |
16
- | `0.2.0` | Next | Optional local machine learning identification for PII |
16
+ | `0.2.0` | Published | Optional local machine learning identification for PII |
17
+ | `0.3.0` | Next | Document inspection (PDF, DOCX, XLSX, PPTX) |
17
18
 
18
19
  Alpha releases optimize for the cleanest safe architecture, not backward compatibility. They may
19
20
  remove, rename, or replace public APIs without aliases or shims. Material changes are documented,
@@ -1,9 +1,10 @@
1
1
  # Policies
2
2
 
3
- `Policy.default()` enables structured identifiers plus person, organization, and location entities
4
- when a configured backend can detect them. `Policy.strict()` also enables secrets and URL
5
- credentials with a lower threshold. `Policy.llm()` enables all available entities.
6
- `Policy.financial()` limits detection to IBANs and payment cards.
3
+ `Policy.default()` enables structured identifiers, secrets, URL credentials, and person,
4
+ organization, and location entities when a configured backend can detect them.
5
+ `Policy.strict()` enables the same entities with a lower confidence threshold. `Policy.llm()`
6
+ enables all available entities. `Policy.financial()` limits detection to IBANs and payment
7
+ cards.
7
8
 
8
9
  Paths are dot-separated segments. `*` matches one segment, so `messages.*.content` matches list
9
10
  indices. Exclusions win over inclusions. Dictionary keys are not transformed.
@@ -29,10 +29,18 @@ After the first upload, verify that the pending publisher became an ordinary pro
29
29
  uv run ruff check .
30
30
  uv run mypy
31
31
  uv run pytest
32
- uv run mkdocs build --strict
32
+ uv run python -m mkdocs build --strict
33
33
  uv build
34
34
  uv run twine check dist/*
35
35
  uv run python scripts/verify_release.py
36
+
37
+ # Verify the built wheel locally in an isolated environment (simulating CI)
38
+ uv venv --python 3.14 .wheel-venv
39
+ uv pip install --python .wheel-venv/bin/python dist/*.whl
40
+ uv pip check --python .wheel-venv/bin/python
41
+ .wheel-venv/bin/python -I scripts/audit_install.py
42
+ .wheel-venv/bin/python -I scripts/smoke_wheel.py
43
+ .wheel-venv/bin/pseudonymize detectors
36
44
  ```
37
45
 
38
46
  4. Open a pull request and merge only after every required check passes.
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pseudonymize"
7
- version = "0.2.0"
7
+ version = "0.3.0"
8
8
  description = "Local-first PII pseudonymization for text, structured data, and LLM payloads."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -43,12 +43,20 @@ ml = [
43
43
  "onnxruntime>=1.29.0",
44
44
  "tokenizers>=0.23.1",
45
45
  ]
46
+ office = [
47
+ "openpyxl>=3.1.0",
48
+ "python-docx>=1.1.0",
49
+ "python-pptx>=1.0.0",
50
+ ]
51
+ pdf = [
52
+ "pdfminer.six>=20240706",
53
+ ]
46
54
 
47
55
  [tool.hatch.build.targets.wheel]
48
56
  packages = ["src/pseudonymize"]
49
57
 
50
58
  [dependency-groups]
51
- test = ["hypothesis>=6", "pytest>=8", "pytest-benchmark>=5", "pytest-cov>=5"]
59
+ test = ["hypothesis>=6", "pytest>=8", "pytest-benchmark>=5", "pytest-cov>=5", "fpdf2>=2.7.9"]
52
60
  quality = ["mypy>=1.11", "pre-commit>=4", "ruff>=0.9"]
53
61
  docs = ["mkdocs-material>=9", "mkdocstrings[python]>=1.0.6"]
54
62
  release = ["build>=1", "twine>=6"]
@@ -81,7 +89,11 @@ files = ["src", "tests", "scripts"]
81
89
  module = [
82
90
  "numpy.*",
83
91
  "onnxruntime.*",
84
- "tokenizers.*"
92
+ "tokenizers.*",
93
+ "pdfminer.*",
94
+ "docx.*",
95
+ "pptx.*",
96
+ "openpyxl.*"
85
97
  ]
86
98
  ignore_missing_imports = true
87
99
  follow_imports = "skip"
@@ -91,7 +103,7 @@ minversion = "8"
91
103
  testpaths = ["tests"]
92
104
  pythonpath = ["."]
93
105
  python_files = ["test_*.py", "benchmark_*.py"]
94
- addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36"]
106
+ addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36", "--basetemp=.pytest_temp"]
95
107
  markers = ["benchmark: performance benchmark", "integration: integration test", "slow: deliberately slow test"]
96
108
  filterwarnings = ["error"]
97
109
 
@@ -27,11 +27,18 @@ def _blocked_network(*arguments: object, **keywords: object) -> None:
27
27
 
28
28
  def main() -> None:
29
29
  parser = argparse.ArgumentParser()
30
- parser.add_argument("--version", required=True)
30
+ parser.add_argument("--version", required=False)
31
31
  arguments = parser.parse_args()
32
32
 
33
+ expected_version = arguments.version
34
+ if not expected_version:
35
+ import tomllib
36
+
37
+ with open("pyproject.toml", "rb") as stream:
38
+ expected_version = tomllib.load(stream)["project"]["version"]
39
+
33
40
  installed = distribution("pseudonymize")
34
- if installed.version != arguments.version:
41
+ if installed.version != expected_version:
35
42
  raise RuntimeError("installed version does not match release")
36
43
  if installed.requires:
37
44
  required_deps = [req for req in installed.requires if "extra ==" not in req]
@@ -25,6 +25,20 @@ else:
25
25
  ort = None
26
26
  Tokenizer = None
27
27
 
28
+ _LABEL_SUFFIXES: tuple[tuple[tuple[str, ...], EntityType], ...] = (
29
+ # Standard CoNLL-03 suffixes plus Ai4Privacy fine-grained labels.
30
+ (("PER", "FIRSTNAME", "LASTNAME", "MIDDLENAME"), EntityType.PERSON),
31
+ (("ORG", "COMPANYNAME"), EntityType.ORGANIZATION),
32
+ (("LOC", "CITY", "STATE", "COUNTY", "STREET", "ZIPCODE"), EntityType.LOCATION),
33
+ )
34
+
35
+
36
+ def _entity_type_for(label: str) -> EntityType | None:
37
+ for suffixes, entity_type in _LABEL_SUFFIXES:
38
+ if label.endswith(suffixes):
39
+ return entity_type
40
+ return None
41
+
28
42
 
29
43
  class LocalONNXPIIBackend(DetectionBackend):
30
44
  def __init__(
@@ -111,55 +125,44 @@ class LocalONNXPIIBackend(DetectionBackend):
111
125
 
112
126
  predictions = np.argmax(logits, axis=-1)
113
127
 
114
- detections = []
128
+ # Token predictions are merged into entity spans: subword continuations
129
+ # (zero gap) and same-type tokens separated by one whitespace character
130
+ # collapse into a single detection so that "John Smith" is one PERSON.
131
+ spans: list[tuple[EntityType, int, int]] = []
115
132
 
116
133
  for idx, label_id in enumerate(predictions):
117
- if label_id == 0 or self._id2label is None:
118
- continue
119
-
120
- label_str = self._id2label.get(label_id)
134
+ label_str = (self._id2label or {}).get(int(label_id))
121
135
  if not label_str or label_str == "O":
122
136
  continue
123
-
124
- entity_type = None
125
-
126
- # Standard CoNLL-03 tags
127
- if label_str.endswith("PER"):
128
- entity_type = EntityType.PERSON
129
- elif label_str.endswith("ORG"):
130
- entity_type = EntityType.ORGANIZATION
131
- elif label_str.endswith("LOC"):
132
- entity_type = EntityType.LOCATION
133
-
134
- # Ai4Privacy tags mapping
135
- elif any(label_str.endswith(s) for s in ("FIRSTNAME", "LASTNAME", "MIDDLENAME")):
136
- entity_type = EntityType.PERSON
137
- elif label_str.endswith("COMPANYNAME"):
138
- entity_type = EntityType.ORGANIZATION
139
- elif any(
140
- label_str.endswith(s) for s in ("CITY", "STATE", "COUNTY", "STREET", "ZIPCODE")
141
- ):
142
- entity_type = EntityType.LOCATION
143
-
144
- if entity_type:
145
- start, end = encoding.offsets[idx]
146
- if start == 0 and end == 0 and idx not in (0, len(predictions) - 1):
147
- continue
148
- if start == end:
137
+ entity_type = _entity_type_for(label_str)
138
+ if entity_type is None:
139
+ continue
140
+ start, end = encoding.offsets[idx]
141
+ if start >= end:
142
+ continue
143
+ if spans:
144
+ previous_type, previous_start, previous_end = spans[-1]
145
+ gap = block.text[previous_end:start]
146
+ if (
147
+ entity_type is previous_type
148
+ and previous_end <= start <= previous_end + 1
149
+ and not gap.strip()
150
+ ):
151
+ spans[-1] = (previous_type, previous_start, end)
149
152
  continue
150
-
151
- detections.append(
152
- Detection(
153
- entity_type=entity_type,
154
- start=start,
155
- end=end,
156
- confidence=0.99,
157
- backend=self.name,
158
- detector="onnx",
159
- )
160
- )
161
-
162
- return tuple(detections)
153
+ spans.append((entity_type, start, end))
154
+
155
+ return tuple(
156
+ Detection(
157
+ entity_type=entity_type,
158
+ start=start,
159
+ end=end,
160
+ confidence=0.99,
161
+ backend=self.name,
162
+ detector="onnx",
163
+ )
164
+ for entity_type, start, end in spans
165
+ )
163
166
 
164
167
  except Exception as e:
165
168
  raise BackendExecutionError(f"ONNX PII inference failed: {e}") from e
@@ -10,6 +10,11 @@ _SENSITIVE_QUERY_NAMES = frozenset(
10
10
  )
11
11
 
12
12
 
13
+ def _authority_end(url: str, authority_start: int) -> int:
14
+ indexes = (url.find(character, authority_start) for character in "/?#")
15
+ return min((index for index in indexes if index != -1), default=len(url))
16
+
17
+
13
18
  @dataclass(frozen=True, slots=True)
14
19
  class UrlDetector:
15
20
  name: str = "url"
@@ -21,7 +26,8 @@ class UrlDetector:
21
26
  parsed = urllib.parse.urlsplit(url)
22
27
  if parsed.username is not None or parsed.password is not None:
23
28
  authority_start = url.find("//") + 2
24
- at = url.find("@", authority_start)
29
+ authority_end = _authority_end(url, authority_start)
30
+ at = url.rfind("@", authority_start, authority_end)
25
31
  detections.append(
26
32
  Detection(
27
33
  EntityType.URL_CREDENTIAL,
@@ -34,8 +40,10 @@ class UrlDetector:
34
40
  query_start = url.find("?") + 1
35
41
  if query_start == 0:
36
42
  continue
43
+ fragment_start = url.find("#", query_start)
44
+ query_end = fragment_start if fragment_start != -1 else len(url)
37
45
  cursor = query_start
38
- for item in url[query_start:].split("&"):
46
+ for item in url[query_start:query_end].split("&"):
39
47
  name, separator, value = item.partition("=")
40
48
  if separator and urllib.parse.unquote_plus(name).lower() in _SENSITIVE_QUERY_NAMES:
41
49
  value_start = cursor + len(name) + 1
@@ -68,7 +68,43 @@ class CSVCellLocation:
68
68
  raise ValueError("CSV row and column indexes must be non-negative")
69
69
 
70
70
 
71
- SourceLocation: TypeAlias = TextOffsetLocation | JSONPathLocation | CSVCellLocation
71
+ @dataclass(frozen=True, slots=True)
72
+ class CoordinateLocation:
73
+ page: int
74
+ x0: float
75
+ y0: float
76
+ x1: float
77
+ y1: float
78
+
79
+ def __post_init__(self) -> None:
80
+ if not isinstance(self.page, int) or isinstance(self.page, bool) or self.page < 0:
81
+ raise ValueError("page must be a non-negative integer")
82
+ for val in (self.x0, self.y0, self.x1, self.y1):
83
+ if not isinstance(val, (int, float)) or isinstance(val, bool) or not isfinite(val):
84
+ raise TypeError("coordinates must be finite numbers")
85
+ if self.x0 > self.x1 or self.y0 > self.y1:
86
+ raise ValueError("coordinates must describe a valid bounding box")
87
+
88
+
89
+ @dataclass(frozen=True, slots=True)
90
+ class StructuralLocation:
91
+ path: tuple[str | int, ...]
92
+
93
+ def __post_init__(self) -> None:
94
+ object.__setattr__(self, "path", tuple(self.path))
95
+ if any(not isinstance(part, (str, int)) or isinstance(part, bool) for part in self.path):
96
+ raise TypeError("structural path parts must be strings or integers")
97
+ if any(isinstance(part, int) and part < 0 for part in self.path):
98
+ raise ValueError("structural path indexes must be non-negative")
99
+
100
+
101
+ SourceLocation: TypeAlias = (
102
+ TextOffsetLocation
103
+ | JSONPathLocation
104
+ | CSVCellLocation
105
+ | CoordinateLocation
106
+ | StructuralLocation
107
+ )
72
108
 
73
109
 
74
110
  @dataclass(frozen=True, slots=True)
@@ -83,7 +119,16 @@ class ContentBlock:
83
119
  raise ValueError("block id must not be empty")
84
120
  if not isinstance(self.text, str):
85
121
  raise TypeError("block text must be a string")
86
- if not isinstance(self.location, (TextOffsetLocation, JSONPathLocation, CSVCellLocation)):
122
+ if not isinstance(
123
+ self.location,
124
+ (
125
+ TextOffsetLocation,
126
+ JSONPathLocation,
127
+ CSVCellLocation,
128
+ CoordinateLocation,
129
+ StructuralLocation,
130
+ ),
131
+ ):
87
132
  raise TypeError("block location must be a supported source location")
88
133
  object.__setattr__(self, "metadata", _metadata(self.metadata))
89
134
 
@@ -265,10 +265,15 @@ class Pseudonymizer:
265
265
  self.transformer.render(entity, alias)
266
266
  for entity, alias in zip(entities, aliases, strict=True)
267
267
  )
268
- output = text
269
- for entity, token in reversed(tuple(zip(entities, tokens, strict=True))):
268
+ segments: list[str] = []
269
+ cursor = 0
270
+ for entity, token in zip(entities, tokens, strict=True):
270
271
  detection = entity.detection
271
- output = output[: detection.start] + token + output[detection.end :]
272
+ segments.append(text[cursor : detection.start])
273
+ segments.append(token)
274
+ cursor = detection.end
275
+ segments.append(text[cursor:])
276
+ output = "".join(segments)
272
277
  replacements = _replacement_reports(entities, tokens)
273
278
  reports.extend(_replacement_detection_reports(block, replacements))
274
279
  mapping = _mapping(text, entities, aliases, tokens) if include_mapping else None
@@ -287,10 +292,10 @@ class Pseudonymizer:
287
292
  if isinstance(data, str):
288
293
  block_id = f"block-{block_counter[0]:06d}"
289
294
  block_counter[0] += 1
290
- block = ContentBlock(block_id, data, JSONPathLocation(path))
291
295
  if not self.policy.allows_path(tuple(str(part) for part in path)):
292
296
  statistics.blocks_processed += 1
293
297
  return data
298
+ block = ContentBlock(block_id, data, JSONPathLocation(path))
294
299
  return self._process_block(block, context, False, statistics, reports).text
295
300
  if data is None or isinstance(data, (bool, int, float)):
296
301
  return data
@@ -462,8 +467,20 @@ def _processing_adapters(
462
467
  output_adapter: OutputAdapter | None,
463
468
  ) -> tuple[InputAdapter[Path], OutputAdapter]:
464
469
  if input_adapter is None and output_adapter is None:
465
- adapter = BuiltinFileAdapter(select_file_format(source, format), encoding)
466
- return adapter, adapter
470
+ selected_format = select_file_format(source, format)
471
+ if selected_format == FileFormat.PDF:
472
+ from pseudonymize.inspection.pdf import PDFInspectionAdapter
473
+
474
+ pdf_adapter = PDFInspectionAdapter()
475
+ return pdf_adapter, pdf_adapter
476
+ elif selected_format in (FileFormat.DOCX, FileFormat.XLSX, FileFormat.PPTX):
477
+ from pseudonymize.inspection.office import OfficeInspectionAdapter
478
+
479
+ office_adapter = OfficeInspectionAdapter(selected_format)
480
+ return office_adapter, office_adapter
481
+ else:
482
+ builtin_adapter = BuiltinFileAdapter(selected_format, encoding)
483
+ return builtin_adapter, builtin_adapter
467
484
  if input_adapter is None or output_adapter is None:
468
485
  raise ValueError("custom file processing requires input and output adapters")
469
486
  if format is not None or encoding is not None:
@@ -478,7 +495,17 @@ def _inspection_adapter(
478
495
  input_adapter: InputAdapter[Path] | None,
479
496
  ) -> InputAdapter[Path]:
480
497
  if input_adapter is None:
481
- return BuiltinFileAdapter(select_file_format(source, format), encoding)
498
+ selected_format = select_file_format(source, format)
499
+ if selected_format == FileFormat.PDF:
500
+ from pseudonymize.inspection.pdf import PDFInspectionAdapter
501
+
502
+ return PDFInspectionAdapter()
503
+ elif selected_format in (FileFormat.DOCX, FileFormat.XLSX, FileFormat.PPTX):
504
+ from pseudonymize.inspection.office import OfficeInspectionAdapter
505
+
506
+ return OfficeInspectionAdapter(selected_format)
507
+ else:
508
+ return BuiltinFileAdapter(selected_format, encoding)
482
509
  if format is not None or encoding is not None:
483
510
  raise ValueError("custom adapters cannot be combined with format or encoding")
484
511
  return input_adapter