pseudonymize 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.agents/PRIVATE_RELEASE_PLAN.md +18 -1
  2. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/package.yml +8 -7
  3. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/CHANGELOG.md +29 -1
  4. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/PKG-INFO +1 -1
  5. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/policies.md +5 -4
  6. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releasing.md +9 -1
  7. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/pyproject.toml +2 -2
  8. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/audit_install.py +9 -2
  9. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/ml/onnx.py +47 -44
  10. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/url.py +10 -2
  11. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/engine.py +9 -4
  12. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/formats.py +22 -40
  13. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/normalization.py +4 -5
  14. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/policy.py +2 -1
  15. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/spans.py +13 -5
  16. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/backends/test_onnx.py +52 -21
  17. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/detectors/test_detectors.py +18 -0
  18. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_policy.py +2 -1
  19. pseudonymize-0.2.1/tests/unit/test_spans.py +39 -0
  20. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/uv.lock +1 -1
  21. pseudonymize-0.2.0/tests/unit/test_spans.py +0 -15
  22. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.editorconfig +0 -0
  23. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.gitattributes +0 -0
  24. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/CODEOWNERS +0 -0
  25. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/bug.yml +0 -0
  26. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/config.yml +0 -0
  27. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/detector-request.yml +0 -0
  28. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/feature.yml +0 -0
  29. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/ISSUE_TEMPLATE/question.yml +0 -0
  30. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/PUBLICATION_CHECKLIST.md +0 -0
  31. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/dependabot.yml +0 -0
  32. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/pull_request_template.md +0 -0
  33. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/ci.yml +0 -0
  34. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/docs.yml +0 -0
  35. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.github/workflows/release.yml +0 -0
  36. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.gitignore +0 -0
  37. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/.pre-commit-config.yaml +0 -0
  38. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/CODE_OF_CONDUCT.md +0 -0
  39. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/CONTRIBUTING.md +0 -0
  40. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/GEMINI.md +0 -0
  41. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/LICENSE +0 -0
  42. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/README.md +0 -0
  43. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/ROADMAP.md +0 -0
  44. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/SECURITY.md +0 -0
  45. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/SUPPORT.md +0 -0
  46. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/VISION.md +0 -0
  47. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/README.md +0 -0
  48. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/benchmark_detectors.py +0 -0
  49. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/benchmark_engine.py +0 -0
  50. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/benchmarks/datasets/synthetic_messages.jsonl +0 -0
  51. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0001-dependency-free-core.md +0 -0
  52. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0002-token-format.md +0 -0
  53. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0003-no-reversible-storage.md +0 -0
  54. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/adr/0004-optional-ml-backend.md +0 -0
  55. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/api.md +0 -0
  56. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/architecture.md +0 -0
  57. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/benchmarks.md +0 -0
  58. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/compatibility.md +0 -0
  59. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/contributing.md +0 -0
  60. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/deployment.md +0 -0
  61. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/detectors.md +0 -0
  62. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/index.md +0 -0
  63. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/limitations.md +0 -0
  64. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/llm-gateways.md +0 -0
  65. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/llm-payloads.md +0 -0
  66. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/migration-a2.md +0 -0
  67. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/migration-a3.md +0 -0
  68. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/quickstart.md +0 -0
  69. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0.md +0 -0
  70. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0a2.md +0 -0
  71. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0a3.md +0 -0
  72. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0b1.md +0 -0
  73. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/releases/0.1.0rc1.md +0 -0
  74. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/security.md +0 -0
  75. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/docs/threat-model.md +0 -0
  76. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/examples/llm_gateway.py +0 -0
  77. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/mkdocs.yml +0 -0
  78. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/models/README.md +0 -0
  79. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/__init__.py +0 -0
  80. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/smoke_wheel.py +0 -0
  81. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/scripts/verify_release.py +0 -0
  82. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/__init__.py +0 -0
  83. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/adapters.py +0 -0
  84. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/api.py +0 -0
  85. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/__init__.py +0 -0
  86. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/base.py +0 -0
  87. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/composite.py +0 -0
  88. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/ml/__init__.py +0 -0
  89. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/backends/rules.py +0 -0
  90. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/cli.py +0 -0
  91. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/__init__.py +0 -0
  92. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/base.py +0 -0
  93. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/email.py +0 -0
  94. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/iban.py +0 -0
  95. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/ip_address.py +0 -0
  96. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/payment_card.py +0 -0
  97. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/phone.py +0 -0
  98. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/registry.py +0 -0
  99. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/detectors/secret.py +0 -0
  100. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/document.py +0 -0
  101. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/exceptions.py +0 -0
  102. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/processing.py +0 -0
  103. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/py.typed +0 -0
  104. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/resolution.py +0 -0
  105. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/result.py +0 -0
  106. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/__init__.py +0 -0
  107. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/alias.py +0 -0
  108. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/base.py +0 -0
  109. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/hmac.py +0 -0
  110. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/mode.py +0 -0
  111. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/placeholder.py +0 -0
  112. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/src/pseudonymize/transforms/redact.py +0 -0
  113. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/compatibility/test_b1_contract.py +0 -0
  114. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/de.jsonl +0 -0
  115. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/en.jsonl +0 -0
  116. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/files.json +0 -0
  117. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/corpus/it.jsonl +0 -0
  118. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_builtin_files.py +0 -0
  119. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_cli.py +0 -0
  120. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_cross_platform_corpus.py +0 -0
  121. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_file_processing.py +0 -0
  122. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_llm_gateway_example.py +0 -0
  123. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_public_api.py +0 -0
  124. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_release_stress.py +0 -0
  125. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/integration/test_release_verifier.py +0 -0
  126. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/property/test_invariants.py +0 -0
  127. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_backends.py +0 -0
  128. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_cli_coverage.py +0 -0
  129. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_document.py +0 -0
  130. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_engine.py +0 -0
  131. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_formats.py +0 -0
  132. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_modes.py +0 -0
  133. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_normalization.py +0 -0
  134. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_processing.py +0 -0
  135. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_transforms.py +0 -0
  136. {pseudonymize-0.2.0 → pseudonymize-0.2.1}/tests/unit/test_types.py +0 -0
@@ -92,7 +92,24 @@ PyPI upload verified:
92
92
  Status:
93
93
  ```
94
94
 
95
- ### 0.2.0: Optional local ML
95
+ ### 0.2.1: Performance and default policy patches
96
+
97
+ Implementation started: 2026-08-22
98
+ Starting commit: aabba9a
99
+ Starting PyPI version: 0.2.0
100
+ Baseline downloads day/week/month: 0/0/0
101
+ Baseline stars/forks/watchers: 0/0/0
102
+ Files changed: 13
103
+ Compatibility tests added: Yes
104
+ Artifact smoke environments: Win local isolated wheel
105
+ Known limitations:
106
+ Deferred items:
107
+ Release commit: TBD
108
+ Tag: v0.2.1
109
+ PyPI upload verified: Pending workflow
110
+ 7-day metrics:
111
+ 28-day metrics:
112
+ Status: Ready
96
113
 
97
114
  Implementation started: 2026-08-21
98
115
  Starting commit: 3205c22ef6143aa8ae38fa2db9eea64fc237777d
@@ -49,16 +49,17 @@ jobs:
49
49
  - name: Install and audit wheel
50
50
  if: runner.os != 'Windows'
51
51
  run: |
52
- uv pip install --python .wheel-venv/bin/python dist/pseudonymize-0.2.0-py3-none-any.whl
52
+ uv pip install --python .wheel-venv/bin/python dist/*.whl
53
53
  uv pip check --python .wheel-venv/bin/python
54
- .wheel-venv/bin/python -I scripts/audit_install.py --version 0.2.0
54
+ .wheel-venv/bin/python -I scripts/audit_install.py
55
55
  .wheel-venv/bin/python -I scripts/smoke_wheel.py
56
56
  .wheel-venv/bin/pseudonymize detectors
57
57
  - name: Install and audit wheel on Windows
58
58
  if: runner.os == 'Windows'
59
+ shell: bash
59
60
  run: |
60
- uv pip install --python .wheel-venv\Scripts\python.exe dist\pseudonymize-0.2.0-py3-none-any.whl
61
- uv pip check --python .wheel-venv\Scripts\python.exe
62
- .wheel-venv\Scripts\python.exe -I scripts\audit_install.py --version 0.2.0
63
- .wheel-venv\Scripts\python.exe -I scripts\smoke_wheel.py
64
- .wheel-venv\Scripts\pseudonymize.exe detectors
61
+ uv pip install --python .wheel-venv/Scripts/python.exe dist/*.whl
62
+ uv pip check --python .wheel-venv/Scripts/python.exe
63
+ .wheel-venv/Scripts/python.exe -I scripts/audit_install.py
64
+ .wheel-venv/Scripts/python.exe -I scripts/smoke_wheel.py
65
+ .wheel-venv/Scripts/pseudonymize.exe detectors
@@ -4,6 +4,33 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [0.2.1] - 2026-08-22
8
+
9
+ ### Changed
10
+
11
+ - `Policy.default()` now includes `URL_CREDENTIAL` and `SECRET`, so the documented secret and
12
+ URL-credential detectors act without opting into `Policy.strict()` or `Policy.llm()`. Callers
13
+ that relied on secrets passing through the default policy must configure an explicit
14
+ `Policy(entity_types=...)`.
15
+ - Overlap resolution now uses an ordered-interval scan instead of a quadratic pairwise check,
16
+ and detected spans are replaced with a single-pass segment join. Texts with thousands of
17
+ detections process in milliseconds instead of seconds.
18
+
19
+ ### Fixed
20
+
21
+ - `LocalONNXPIIBackend` now merges contiguous token predictions into whole entity spans, so
22
+ multi-word and subword-split names receive a single alias instead of one alias per token.
23
+ - ONNX test artifacts downloaded during testing are now verified against pinned SHA-256
24
+ checksums.
25
+ - URL credential detection now covers the whole userinfo section when it contains extra `@`
26
+ characters and no longer swallows the URL fragment into sensitive query values.
27
+ - Overlap resolution now ranks URL credentials above emails, so a `user:password@host` userinfo
28
+ can no longer lose its `user:` prefix to an overlapping email match.
29
+ - The `ml` extra now declares `numpy`, which the ONNX backend imports directly, and no longer
30
+ pulls in the unused `huggingface-hub` dependency.
31
+ - The optional-import guard in the ONNX backend now type-checks under strict mypy.
32
+ - Normalization now maps generic URL schemes to `http` or `https` for safer alias sharing.
33
+
7
34
  ## [0.2.0] - 2026-08-21
8
35
 
9
36
  ### Added
@@ -135,7 +162,8 @@ All notable changes follow Keep a Changelog and Semantic Versioning.
135
162
  - Deterministic processing now requires `mode="deterministic"` in addition to a key.
136
163
  - `redact()` now emits `[REDACTED]` by default; generic mode emits typed placeholders.
137
164
 
138
- [Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.0...HEAD
165
+ [Unreleased]: https://github.com/ma2za/pseudonymize/compare/v0.2.1...HEAD
166
+ [0.2.1]: https://github.com/ma2za/pseudonymize/compare/v0.2.0...v0.2.1
139
167
  [0.2.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.2.0
140
168
  [0.1.0]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0
141
169
  [0.1.0rc1]: https://github.com/ma2za/pseudonymize/releases/tag/v0.1.0rc1
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pseudonymize
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Local-first PII pseudonymization for text, structured data, and LLM payloads.
5
5
  Project-URL: Homepage, https://github.com/ma2za/pseudonymize
6
6
  Project-URL: Documentation, https://ma2za.github.io/pseudonymize/
@@ -1,9 +1,10 @@
1
1
  # Policies
2
2
 
3
- `Policy.default()` enables structured identifiers plus person, organization, and location entities
4
- when a configured backend can detect them. `Policy.strict()` also enables secrets and URL
5
- credentials with a lower threshold. `Policy.llm()` enables all available entities.
6
- `Policy.financial()` limits detection to IBANs and payment cards.
3
+ `Policy.default()` enables structured identifiers, secrets, URL credentials, and person,
4
+ organization, and location entities when a configured backend can detect them.
5
+ `Policy.strict()` enables the same entities with a lower confidence threshold. `Policy.llm()`
6
+ enables all available entities. `Policy.financial()` limits detection to IBANs and payment
7
+ cards.
7
8
 
8
9
  Paths are dot-separated segments. `*` matches one segment, so `messages.*.content` matches list
9
10
  indices. Exclusions win over inclusions. Dictionary keys are not transformed.
@@ -29,10 +29,18 @@ After the first upload, verify that the pending publisher became an ordinary pro
29
29
  uv run ruff check .
30
30
  uv run mypy
31
31
  uv run pytest
32
- uv run mkdocs build --strict
32
+ uv run python -m mkdocs build --strict
33
33
  uv build
34
34
  uv run twine check dist/*
35
35
  uv run python scripts/verify_release.py
36
+
37
+ # Verify the built wheel locally in an isolated environment (simulating CI)
38
+ uv venv --python 3.14 .wheel-venv
39
+ uv pip install --python .wheel-venv/bin/python dist/*.whl
40
+ uv pip check --python .wheel-venv/bin/python
41
+ .wheel-venv/bin/python -I scripts/audit_install.py
42
+ .wheel-venv/bin/python -I scripts/smoke_wheel.py
43
+ .wheel-venv/bin/pseudonymize detectors
36
44
  ```
37
45
 
38
46
  4. Open a pull request and merge only after every required check passes.
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pseudonymize"
7
- version = "0.2.0"
7
+ version = "0.2.1"
8
8
  description = "Local-first PII pseudonymization for text, structured data, and LLM payloads."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -91,7 +91,7 @@ minversion = "8"
91
91
  testpaths = ["tests"]
92
92
  pythonpath = ["."]
93
93
  python_files = ["test_*.py", "benchmark_*.py"]
94
- addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36"]
94
+ addopts = ["--strict-config", "--strict-markers", "-ra", "--cov=pseudonymize", "--cov-report=term-missing", "--cov-branch", "--cov-fail-under=99.36", "--basetemp=.pytest_temp"]
95
95
  markers = ["benchmark: performance benchmark", "integration: integration test", "slow: deliberately slow test"]
96
96
  filterwarnings = ["error"]
97
97
 
@@ -27,11 +27,18 @@ def _blocked_network(*arguments: object, **keywords: object) -> None:
27
27
 
28
28
  def main() -> None:
29
29
  parser = argparse.ArgumentParser()
30
- parser.add_argument("--version", required=True)
30
+ parser.add_argument("--version", required=False)
31
31
  arguments = parser.parse_args()
32
32
 
33
+ expected_version = arguments.version
34
+ if not expected_version:
35
+ import tomllib
36
+
37
+ with open("pyproject.toml", "rb") as stream:
38
+ expected_version = tomllib.load(stream)["project"]["version"]
39
+
33
40
  installed = distribution("pseudonymize")
34
- if installed.version != arguments.version:
41
+ if installed.version != expected_version:
35
42
  raise RuntimeError("installed version does not match release")
36
43
  if installed.requires:
37
44
  required_deps = [req for req in installed.requires if "extra ==" not in req]
@@ -25,6 +25,20 @@ else:
25
25
  ort = None
26
26
  Tokenizer = None
27
27
 
28
+ _LABEL_SUFFIXES: tuple[tuple[tuple[str, ...], EntityType], ...] = (
29
+ # Standard CoNLL-03 suffixes plus Ai4Privacy fine-grained labels.
30
+ (("PER", "FIRSTNAME", "LASTNAME", "MIDDLENAME"), EntityType.PERSON),
31
+ (("ORG", "COMPANYNAME"), EntityType.ORGANIZATION),
32
+ (("LOC", "CITY", "STATE", "COUNTY", "STREET", "ZIPCODE"), EntityType.LOCATION),
33
+ )
34
+
35
+
36
+ def _entity_type_for(label: str) -> EntityType | None:
37
+ for suffixes, entity_type in _LABEL_SUFFIXES:
38
+ if label.endswith(suffixes):
39
+ return entity_type
40
+ return None
41
+
28
42
 
29
43
  class LocalONNXPIIBackend(DetectionBackend):
30
44
  def __init__(
@@ -111,55 +125,44 @@ class LocalONNXPIIBackend(DetectionBackend):
111
125
 
112
126
  predictions = np.argmax(logits, axis=-1)
113
127
 
114
- detections = []
128
+ # Token predictions are merged into entity spans: subword continuations
129
+ # (zero gap) and same-type tokens separated by one whitespace character
130
+ # collapse into a single detection so that "John Smith" is one PERSON.
131
+ spans: list[tuple[EntityType, int, int]] = []
115
132
 
116
133
  for idx, label_id in enumerate(predictions):
117
- if label_id == 0 or self._id2label is None:
118
- continue
119
-
120
- label_str = self._id2label.get(label_id)
134
+ label_str = (self._id2label or {}).get(int(label_id))
121
135
  if not label_str or label_str == "O":
122
136
  continue
123
-
124
- entity_type = None
125
-
126
- # Standard CoNLL-03 tags
127
- if label_str.endswith("PER"):
128
- entity_type = EntityType.PERSON
129
- elif label_str.endswith("ORG"):
130
- entity_type = EntityType.ORGANIZATION
131
- elif label_str.endswith("LOC"):
132
- entity_type = EntityType.LOCATION
133
-
134
- # Ai4Privacy tags mapping
135
- elif any(label_str.endswith(s) for s in ("FIRSTNAME", "LASTNAME", "MIDDLENAME")):
136
- entity_type = EntityType.PERSON
137
- elif label_str.endswith("COMPANYNAME"):
138
- entity_type = EntityType.ORGANIZATION
139
- elif any(
140
- label_str.endswith(s) for s in ("CITY", "STATE", "COUNTY", "STREET", "ZIPCODE")
141
- ):
142
- entity_type = EntityType.LOCATION
143
-
144
- if entity_type:
145
- start, end = encoding.offsets[idx]
146
- if start == 0 and end == 0 and idx not in (0, len(predictions) - 1):
147
- continue
148
- if start == end:
137
+ entity_type = _entity_type_for(label_str)
138
+ if entity_type is None:
139
+ continue
140
+ start, end = encoding.offsets[idx]
141
+ if start >= end:
142
+ continue
143
+ if spans:
144
+ previous_type, previous_start, previous_end = spans[-1]
145
+ gap = block.text[previous_end:start]
146
+ if (
147
+ entity_type is previous_type
148
+ and previous_end <= start <= previous_end + 1
149
+ and not gap.strip()
150
+ ):
151
+ spans[-1] = (previous_type, previous_start, end)
149
152
  continue
150
-
151
- detections.append(
152
- Detection(
153
- entity_type=entity_type,
154
- start=start,
155
- end=end,
156
- confidence=0.99,
157
- backend=self.name,
158
- detector="onnx",
159
- )
160
- )
161
-
162
- return tuple(detections)
153
+ spans.append((entity_type, start, end))
154
+
155
+ return tuple(
156
+ Detection(
157
+ entity_type=entity_type,
158
+ start=start,
159
+ end=end,
160
+ confidence=0.99,
161
+ backend=self.name,
162
+ detector="onnx",
163
+ )
164
+ for entity_type, start, end in spans
165
+ )
163
166
 
164
167
  except Exception as e:
165
168
  raise BackendExecutionError(f"ONNX PII inference failed: {e}") from e
@@ -10,6 +10,11 @@ _SENSITIVE_QUERY_NAMES = frozenset(
10
10
  )
11
11
 
12
12
 
13
+ def _authority_end(url: str, authority_start: int) -> int:
14
+ indexes = (url.find(character, authority_start) for character in "/?#")
15
+ return min((index for index in indexes if index != -1), default=len(url))
16
+
17
+
13
18
  @dataclass(frozen=True, slots=True)
14
19
  class UrlDetector:
15
20
  name: str = "url"
@@ -21,7 +26,8 @@ class UrlDetector:
21
26
  parsed = urllib.parse.urlsplit(url)
22
27
  if parsed.username is not None or parsed.password is not None:
23
28
  authority_start = url.find("//") + 2
24
- at = url.find("@", authority_start)
29
+ authority_end = _authority_end(url, authority_start)
30
+ at = url.rfind("@", authority_start, authority_end)
25
31
  detections.append(
26
32
  Detection(
27
33
  EntityType.URL_CREDENTIAL,
@@ -34,8 +40,10 @@ class UrlDetector:
34
40
  query_start = url.find("?") + 1
35
41
  if query_start == 0:
36
42
  continue
43
+ fragment_start = url.find("#", query_start)
44
+ query_end = fragment_start if fragment_start != -1 else len(url)
37
45
  cursor = query_start
38
- for item in url[query_start:].split("&"):
46
+ for item in url[query_start:query_end].split("&"):
39
47
  name, separator, value = item.partition("=")
40
48
  if separator and urllib.parse.unquote_plus(name).lower() in _SENSITIVE_QUERY_NAMES:
41
49
  value_start = cursor + len(name) + 1
@@ -265,10 +265,15 @@ class Pseudonymizer:
265
265
  self.transformer.render(entity, alias)
266
266
  for entity, alias in zip(entities, aliases, strict=True)
267
267
  )
268
- output = text
269
- for entity, token in reversed(tuple(zip(entities, tokens, strict=True))):
268
+ segments: list[str] = []
269
+ cursor = 0
270
+ for entity, token in zip(entities, tokens, strict=True):
270
271
  detection = entity.detection
271
- output = output[: detection.start] + token + output[detection.end :]
272
+ segments.append(text[cursor : detection.start])
273
+ segments.append(token)
274
+ cursor = detection.end
275
+ segments.append(text[cursor:])
276
+ output = "".join(segments)
272
277
  replacements = _replacement_reports(entities, tokens)
273
278
  reports.extend(_replacement_detection_reports(block, replacements))
274
279
  mapping = _mapping(text, entities, aliases, tokens) if include_mapping else None
@@ -287,10 +292,10 @@ class Pseudonymizer:
287
292
  if isinstance(data, str):
288
293
  block_id = f"block-{block_counter[0]:06d}"
289
294
  block_counter[0] += 1
290
- block = ContentBlock(block_id, data, JSONPathLocation(path))
291
295
  if not self.policy.allows_path(tuple(str(part) for part in path)):
292
296
  statistics.blocks_processed += 1
293
297
  return data
298
+ block = ContentBlock(block_id, data, JSONPathLocation(path))
294
299
  return self._process_block(block, context, False, statistics, reports).text
295
300
  if data is None or isinstance(data, (bool, int, float)):
296
301
  return data
@@ -242,7 +242,7 @@ def _json_replacements(document: Document) -> dict[tuple[str | int, ...], str]:
242
242
 
243
243
  def _render_json(document: Document, state: _AdapterState) -> str:
244
244
  value = _replace_json_strings(
245
- _require_json_content(state.content),
245
+ cast(JSONValue, state.content),
246
246
  (),
247
247
  _json_replacements(document),
248
248
  )
@@ -250,32 +250,26 @@ def _render_json(document: Document, state: _AdapterState) -> str:
250
250
 
251
251
 
252
252
  def _extract_jsonl(text: str, decoded: _DecodedText) -> tuple[Document, tuple[JSONValue, ...]]:
253
- if not text:
254
- values: tuple[JSONValue, ...] = ()
255
- else:
256
- lines = text.splitlines()
257
- values_list: list[JSONValue] = []
258
- blocks: list[ContentBlock] = []
259
- for line_index, line in enumerate(lines):
260
- if not line.strip():
261
- raise _JSONLLineError(line_index + 1)
262
- try:
263
- value = _load_json(line)
264
- except (TypeError, ValueError):
265
- raise _JSONLLineError(line_index + 1) from None
266
- values_list.append(value)
267
- _string_blocks(
268
- value,
269
- (line_index,),
270
- blocks,
271
- prefix=f"line-{line_index:06d}-block",
272
- )
273
- values = tuple(values_list)
274
- return (
275
- Document("file", tuple(blocks), _metadata(FileFormat.JSONL, decoded)),
276
- values,
253
+ values: list[JSONValue] = []
254
+ blocks: list[ContentBlock] = []
255
+ for line_index, line in enumerate(text.splitlines()):
256
+ if not line.strip():
257
+ raise _JSONLLineError(line_index + 1)
258
+ try:
259
+ value = _load_json(line)
260
+ except (TypeError, ValueError):
261
+ raise _JSONLLineError(line_index + 1) from None
262
+ values.append(value)
263
+ _string_blocks(
264
+ value,
265
+ (line_index,),
266
+ blocks,
267
+ prefix=f"line-{line_index:06d}-block",
277
268
  )
278
- return Document("file", (), _metadata(FileFormat.JSONL, decoded)), values
269
+ return (
270
+ Document("file", tuple(blocks), _metadata(FileFormat.JSONL, decoded)),
271
+ tuple(values),
272
+ )
279
273
 
280
274
 
281
275
  def _validate_document(document: Document, original: Document) -> None:
@@ -289,7 +283,7 @@ def _validate_document(document: Document, original: Document) -> None:
289
283
 
290
284
 
291
285
  def _render_jsonl(document: Document, state: _AdapterState) -> str:
292
- values = _require_jsonl_content(state.content)
286
+ values = cast(tuple[JSONValue, ...], state.content)
293
287
  replacements = _json_replacements(document)
294
288
  rendered = (
295
289
  json.dumps(
@@ -332,7 +326,7 @@ def _read_csv(text: str) -> tuple[tuple[str, ...], ...]:
332
326
 
333
327
 
334
328
  def _render_csv(document: Document, state: _AdapterState) -> str:
335
- rows = _require_csv_content(state.content)
329
+ rows = cast(tuple[tuple[str, ...], ...], state.content)
336
330
  replacements = {
337
331
  (
338
332
  cast(CSVCellLocation, block.location).row,
@@ -347,15 +341,3 @@ def _render_csv(document: Document, state: _AdapterState) -> str:
347
341
  output = io.StringIO(newline="")
348
342
  csv.writer(output, dialect="excel", lineterminator="\r\n").writerows(transformed)
349
343
  return output.getvalue()
350
-
351
-
352
- def _require_json_content(content: object) -> JSONValue:
353
- return cast(JSONValue, content)
354
-
355
-
356
- def _require_jsonl_content(content: object) -> tuple[JSONValue, ...]:
357
- return cast(tuple[JSONValue, ...], content)
358
-
359
-
360
- def _require_csv_content(content: object) -> tuple[tuple[str, ...], ...]:
361
- return cast(tuple[tuple[str, ...], ...], content)
@@ -11,12 +11,11 @@ def normalize(value: str, entity_type: EntityType) -> str:
11
11
  if entity_type is EntityType.EMAIL:
12
12
  local, separator, domain = value.rpartition("@")
13
13
  return f"{local}{separator}{domain.lower()}"
14
- if entity_type in {EntityType.IBAN, EntityType.PAYMENT_CARD, EntityType.PHONE}:
14
+ if entity_type is EntityType.IBAN:
15
+ return "".join(value.split()).upper()
16
+ if entity_type in {EntityType.PAYMENT_CARD, EntityType.PHONE}:
15
17
  prefix = "+" if entity_type is EntityType.PHONE and value.startswith("+") else ""
16
- compact = "".join(character for character in value if character.isdigit())
17
- if entity_type is EntityType.IBAN:
18
- return "".join(value.split()).upper()
19
- return prefix + compact
18
+ return prefix + "".join(character for character in value if character.isdigit())
20
19
  if entity_type is EntityType.IP_ADDRESS:
21
20
  return ipaddress.ip_address(value.strip("[]")).compressed
22
21
  return value
@@ -13,6 +13,7 @@ _STRUCTURED = frozenset(
13
13
  EntityType.PAYMENT_CARD,
14
14
  }
15
15
  )
16
+ _CREDENTIALS = frozenset({EntityType.URL_CREDENTIAL, EntityType.SECRET})
16
17
  _SEMANTIC = frozenset({EntityType.PERSON, EntityType.ORGANIZATION, EntityType.LOCATION})
17
18
 
18
19
 
@@ -24,7 +25,7 @@ class NetworkPolicy(StrEnum):
24
25
 
25
26
  @dataclass(frozen=True, slots=True)
26
27
  class Policy:
27
- entity_types: Set[EntityType] = _STRUCTURED | _SEMANTIC
28
+ entity_types: Set[EntityType] = _STRUCTURED | _CREDENTIALS | _SEMANTIC
28
29
  minimum_confidence: float = 0.8
29
30
  detector_priority: tuple[str, ...] = ()
30
31
  include_paths: tuple[str, ...] = ()
@@ -1,3 +1,4 @@
1
+ import bisect
1
2
  from collections.abc import Iterable
2
3
 
3
4
  from pseudonymize.result import Detection, EntityType
@@ -5,9 +6,12 @@ from pseudonymize.result import Detection, EntityType
5
6
  _ENTITY_PRIORITY = {
6
7
  EntityType.PAYMENT_CARD: 70,
7
8
  EntityType.IBAN: 70,
9
+ # URL credentials outrank emails: a password such as "s3cret" followed by
10
+ # "@host" also matches the email pattern, and letting the email span win
11
+ # would leave the "user:" part of the userinfo unmasked.
12
+ EntityType.URL_CREDENTIAL: 65,
8
13
  EntityType.EMAIL: 60,
9
14
  EntityType.IP_ADDRESS: 60,
10
- EntityType.URL_CREDENTIAL: 50,
11
15
  EntityType.SECRET: 50,
12
16
  EntityType.PHONE: 40,
13
17
  EntityType.PERSON: 30,
@@ -35,12 +39,16 @@ def resolve_overlaps(
35
39
  detection.backend,
36
40
  ),
37
41
  )
42
+ # Accepted spans are kept sorted and non-overlapping, so a candidate can
43
+ # only collide with the span immediately before its insertion point.
44
+ starts: list[int] = []
45
+ ends: list[int] = []
38
46
  selected: list[Detection] = []
39
47
  for detection in ranked:
40
- if any(
41
- detection.start < existing.end and existing.start < detection.end
42
- for existing in selected
43
- ):
48
+ index = bisect.bisect_left(starts, detection.end)
49
+ if index and ends[index - 1] > detection.start:
44
50
  continue
51
+ starts.insert(index, detection.start)
52
+ ends.insert(index, detection.end)
45
53
  selected.append(detection)
46
54
  return tuple(sorted(selected, key=lambda detection: (detection.start, detection.end)))
@@ -1,3 +1,4 @@
1
+ import hashlib
1
2
  import urllib.request
2
3
  from pathlib import Path
3
4
  from typing import Any
@@ -15,27 +16,39 @@ MODEL_URL_BASE = (
15
16
  "https://huggingface.co/onnx-community/distilbert_finetuned_ai4privacy_v2-ONNX/resolve/main/"
16
17
  )
17
18
  MODEL_FILES = {
18
- "config.json": "config.json",
19
- "tokenizer.json": "tokenizer.json",
20
- "model_int8.onnx": "onnx/model_int8.onnx",
19
+ "config.json": (
20
+ "config.json",
21
+ "5155e76f303c68ee15d8f01b580550c062cebfbb42dda3f5f3698b2d75424216",
22
+ ),
23
+ "tokenizer.json": (
24
+ "tokenizer.json",
25
+ "cb374d6bc042c22455946f4e09a89d29882a199fdaf8fb25be00dc8b8857a448",
26
+ ),
27
+ "model_int8.onnx": (
28
+ "onnx/model_int8.onnx",
29
+ "6faa1d7f5b54140bbba18ba87480e11073927b5fff16f69558bd51058a05b305",
30
+ ),
21
31
  }
22
32
 
23
33
 
24
- def download_file(url: str, dest: Path) -> None:
25
- if dest.exists():
26
- return
27
- dest.parent.mkdir(parents=True, exist_ok=True)
28
- req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"}) # noqa: S310
29
- with urllib.request.urlopen(req) as response, open(dest, "wb") as f: # noqa: S310
30
- f.write(response.read())
34
+ def download_file(url: str, dest: Path, sha256: str) -> None:
35
+ if not dest.exists():
36
+ dest.parent.mkdir(parents=True, exist_ok=True)
37
+ req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"}) # noqa: S310
38
+ with urllib.request.urlopen(req) as response, open(dest, "wb") as f: # noqa: S310
39
+ f.write(response.read())
40
+ digest = hashlib.sha256(dest.read_bytes()).hexdigest()
41
+ if digest != sha256:
42
+ dest.unlink()
43
+ raise RuntimeError(f"checksum mismatch for {dest.name}: {digest}")
31
44
 
32
45
 
33
46
  @pytest.fixture(scope="session")
34
47
  def distilbert_artifacts() -> tuple[Path, Path, Path]:
35
48
  paths = []
36
- for local_name, remote_path in MODEL_FILES.items():
49
+ for local_name, (remote_path, sha256) in MODEL_FILES.items():
37
50
  dest = CACHE_DIR / local_name
38
- download_file(MODEL_URL_BASE + remote_path, dest)
51
+ download_file(MODEL_URL_BASE + remote_path, dest, sha256)
39
52
  paths.append(dest)
40
53
  # Returns (config, tokenizer, model)
41
54
  return (paths[0], paths[1], paths[2])
@@ -166,19 +179,15 @@ def test_ml_detect_real_inference_returns_meaningful_detections(
166
179
  # Sort detections by offset for deterministic assertion
167
180
  sorted_detections = sorted(detections, key=lambda d: d.start)
168
181
 
169
- # We expect a good DistilBERT ML model to find:
170
- # PERSON: "John" (0,4), "Smith" (5,10) - (backend yields token-by-token right now)
171
- # LOC: "Seattle" (61,68), "Washington" (70,80)
172
-
173
- # Note: Since the backend returns detections at the token level,
174
- # we rigorously assert the exact subword spans mapped correctly to their entity type
175
- assert len(sorted_detections) >= 4
182
+ # Token predictions merge into entity spans, so we expect:
183
+ # PERSON: "John Smith" (0,10) as one span, subwords and the space included
184
+ # LOC: "Seattle" (61,68) and "Washington" (70,80) kept apart by the comma
185
+ assert len(sorted_detections) >= 3
176
186
 
177
187
  # Let's map out the exact expected strings for the entities found
178
188
  found_entities = [(d.entity_type, text[d.start : d.end]) for d in sorted_detections]
179
189
 
180
- assert (EntityType.PERSON, "John") in found_entities
181
- assert (EntityType.PERSON, "Smith") in found_entities
190
+ assert (EntityType.PERSON, "John Smith") in found_entities
182
191
  assert (EntityType.LOCATION, "Seattle") in found_entities
183
192
  assert (EntityType.LOCATION, "Washington") in found_entities
184
193
 
@@ -250,6 +259,28 @@ def test_ml_detect_real_inference_returns_meaningful_detections(
250
259
  assert EntityType.LOCATION in found_types
251
260
 
252
261
 
262
+ def test_ml_engine_shares_one_alias_per_merged_entity(
263
+ distilbert_artifacts: tuple[Path, Path, Path],
264
+ ) -> None:
265
+ from pseudonymize import Pseudonymizer
266
+
267
+ config_path, tokenizer_path, model_path = distilbert_artifacts
268
+ backend = LocalONNXPIIBackend(
269
+ model_path=model_path, tokenizer_path=tokenizer_path, config_path=config_path
270
+ )
271
+ engine = Pseudonymizer(backends=[backend])
272
+ result = engine.process("John Smith emailed John Smith from Seattle.")
273
+
274
+ assert "John" not in result.text
275
+ assert "Smith" not in result.text
276
+ person_tokens = {
277
+ replacement.token
278
+ for replacement in result.replacements
279
+ if replacement.detection.entity_type is EntityType.PERSON
280
+ }
281
+ assert person_tokens == {"<PERSON_1>"}
282
+
283
+
253
284
  def test_ml_detect_raises_on_inference_failure(
254
285
  distilbert_artifacts: tuple[Path, Path, Path], monkeypatch: pytest.MonkeyPatch
255
286
  ) -> None: