pdf-strikethrough-detect 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pdf_strikethrough_detect-0.4.0/CHANGELOG.md +5 -0
- pdf_strikethrough_detect-0.4.0/LICENSE +21 -0
- pdf_strikethrough_detect-0.4.0/MANIFEST.in +4 -0
- pdf_strikethrough_detect-0.4.0/PKG-INFO +203 -0
- pdf_strikethrough_detect-0.4.0/README.md +164 -0
- pdf_strikethrough_detect-0.4.0/pyproject.toml +55 -0
- pdf_strikethrough_detect-0.4.0/setup.cfg +4 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/__init__.py +107 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/__main__.py +119 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/cnn.py +171 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/detect.py +274 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/lines.py +291 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/markdown.py +113 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/native.py +252 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/ocr.py +119 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/scanned.py +316 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/strike_verdict_cnn.meta.json +9 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/strike_verdict_cnn.onnx +0 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/PKG-INFO +203 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/SOURCES.txt +23 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/dependency_links.txt +1 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/entry_points.txt +2 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/requires.txt +21 -0
- pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/top_level.txt +1 -0
- pdf_strikethrough_detect-0.4.0/tests/test_smoke.py +274 -0
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.4.0 — first public release
|
|
4
|
+
|
|
5
|
+
- Detect struck-through (deleted) text in born-digital PDFs (exact vector/flag detection) and scanned pages (stroke geometry + OCR + ONNX CNN), with `~~struck~~` markdown, clean text, and grouped passages via a Python API and CLI.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Niles Liu
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pdf-strikethrough-detect
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Detect struck-through (deleted) text in PDFs and scanned document images.
|
|
5
|
+
Author: Niles Liu
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/niles-liu/pdf-strikethrough-detect
|
|
8
|
+
Project-URL: Repository, https://github.com/niles-liu/pdf-strikethrough-detect
|
|
9
|
+
Project-URL: Issues, https://github.com/niles-liu/pdf-strikethrough-detect/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/niles-liu/pdf-strikethrough-detect/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: strikethrough,strikeout,redline,ocr,pdf,document-intelligence,deleted-text,scanned-documents
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
18
|
+
Classifier: Topic :: Text Processing
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: numpy>=1.24
|
|
23
|
+
Requires-Dist: pillow>=9.0
|
|
24
|
+
Requires-Dist: scipy>=1.10
|
|
25
|
+
Requires-Dist: onnxruntime>=1.16
|
|
26
|
+
Requires-Dist: pymupdf>=1.26
|
|
27
|
+
Provides-Extra: markdown
|
|
28
|
+
Requires-Dist: pymupdf4llm>=0.0.27; extra == "markdown"
|
|
29
|
+
Provides-Extra: rapidocr
|
|
30
|
+
Requires-Dist: rapidocr>=3.2; extra == "rapidocr"
|
|
31
|
+
Requires-Dist: onnxruntime>=1.16; extra == "rapidocr"
|
|
32
|
+
Provides-Extra: tesseract
|
|
33
|
+
Requires-Dist: pytesseract>=0.3.10; extra == "tesseract"
|
|
34
|
+
Provides-Extra: torch
|
|
35
|
+
Requires-Dist: torch>=2.0; extra == "torch"
|
|
36
|
+
Provides-Extra: dev
|
|
37
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
38
|
+
Dynamic: license-file
|
|
39
|
+
|
|
40
|
+
# pdf-strikethrough-detect
|
|
41
|
+
|
|
42
|
+
[](https://pypi.org/project/pdf-strikethrough-detect/)
|
|
43
|
+
[](https://pypi.org/project/pdf-strikethrough-detect/)
|
|
44
|
+
[](https://github.com/niles-liu/pdf-strikethrough-detect/actions/workflows/ci.yml)
|
|
45
|
+
[](LICENSE)
|
|
46
|
+
|
|
47
|
+
Detect **struck-through (deleted) text** in PDFs and scanned document images.
|
|
48
|
+
|
|
49
|
+
Strikethrough detection is a surprisingly unserved niche: most "redline"/diff tools assume clean
|
|
50
|
+
born-digital PDFs and fall apart on scans — which is the real-world case. `pdf-strikethrough-detect`
|
|
51
|
+
handles both, does the hard part (scanned images) with a tiny CPU model, and makes no cloud calls.
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
import pdf_strikethrough as st
|
|
55
|
+
|
|
56
|
+
# born-digital PDF — exact, no OCR
|
|
57
|
+
for w in st.strikethroughs_in_pdf("contract.pdf"):
|
|
58
|
+
print(w["page"], repr(w["chars"]), "partial" if w["partial"] else "full")
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Install
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pip install pdf-strikethrough-detect
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Pure pip, no system binaries required: the CNN runs on ONNX Runtime (CPU) and the ~318 KB model
|
|
68
|
+
ships inside the wheel. Extras:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
pip install "pdf-strikethrough-detect[markdown]" # clean_markdown() via pymupdf4llm
|
|
72
|
+
pip install "pdf-strikethrough-detect[rapidocr]" # free scanned-word OCR backend (no binary)
|
|
73
|
+
pip install "pdf-strikethrough-detect[tesseract]" # word-level OCR (also needs the tesseract binary)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Native / born-digital PDFs — exact
|
|
77
|
+
|
|
78
|
+
In a born-digital PDF a strikethrough is a *vector drawing* (a line or thin rect over the text),
|
|
79
|
+
so detection is exact ground truth — no OCR, no model, no guessing. Both vector-rule and
|
|
80
|
+
filled-rect strikethrough styles are handled.
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
import pdf_strikethrough as st
|
|
84
|
+
|
|
85
|
+
for w in st.strikethroughs_in_pdf("contract.pdf"):
|
|
86
|
+
print(w["page"], repr(w["chars"])) # 'chars' = the struck substring
|
|
87
|
+
print(st.clean_markdown("contract.pdf")) # surviving text, deletions removed (needs [markdown])
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Each record: `{page, text, chars, char_span, partial, bbox_frac, coverage, verdict, final}`.
|
|
91
|
+
Partial strikes (`semi-` of `semi-monthly`) are resolved to a char range. `bbox_frac` is in
|
|
92
|
+
fractions of the rendered page (rotation-aware), so it maps directly onto a rendered pixmap.
|
|
93
|
+
|
|
94
|
+
Two native detectors, both **base-PyMuPDF only** (no pymupdf4llm), selected by `method`:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
st.strikethroughs_in_pdf("contract.pdf", method="vector") # stroke geometry (default) —
|
|
98
|
+
# precise partial-char spans
|
|
99
|
+
st.strikethroughs_in_pdf("contract.pdf", method="flag") # MuPDF's FZ_STEXT_STRIKEOUT signal —
|
|
100
|
+
# also catches font-attribute strikes
|
|
101
|
+
st.strikethroughs_in_pdf("contract.pdf", method="both") # union — maximum recall
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
**Validated across domains.** On 12 public redline PDFs (federal & state regulations, court
|
|
105
|
+
rules, procurement clauses, municipal codes, university policy; 33k struck words),
|
|
106
|
+
**99.9–100% of vector detections are independently confirmed by MuPDF's strikeout signal**,
|
|
107
|
+
and the flag method adds ~2% more words (font-attribute strikes and edge cases) — use
|
|
108
|
+
`method="both"` to capture them. `pymupdf4llm` is not used for detection at all; it is only an
|
|
109
|
+
optional `[markdown]` extra for richer layout in `clean_markdown()`.
|
|
110
|
+
|
|
111
|
+
## Any PDF — routed per page, scanned pages use OCR + CNN
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
import pdf_strikethrough as st
|
|
115
|
+
from pdf_strikethrough.ocr import rapidocr_backend
|
|
116
|
+
from pdf_strikethrough.scanned import ScanConfig
|
|
117
|
+
|
|
118
|
+
res = st.detect_pdf("mixed.pdf",
|
|
119
|
+
ocr=rapidocr_backend(), # for scanned pages
|
|
120
|
+
scan_config=ScanConfig.confidence_free())
|
|
121
|
+
|
|
122
|
+
struck = [w for w in res["words"] if w["final"]] # struck words (boxes, char spans)
|
|
123
|
+
markdown = res["markdown"] # deletions as ~~struck~~
|
|
124
|
+
clean = res["clean_text"] # surviving text, deletions removed
|
|
125
|
+
passages = res["passages"] # grouped deletion sections
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
`detect_pdf` classifies each page native-vs-scanned, runs the exact path on native pages
|
|
129
|
+
(`native_method="vector"|"flag"|"both"`) and the geometry→OCR→CNN pipeline on scanned ones, and
|
|
130
|
+
assembles `markdown` / `clean_text` / `passages` for **both** page kinds from its own strike
|
|
131
|
+
decisions (so the text and the word records always agree — no dependence on an external markdown
|
|
132
|
+
engine). Already have an Azure Document Intelligence result? Pass `di_result=...` (the REST JSON
|
|
133
|
+
dict, an `{'analyzeResult': ...}` envelope, or `sdk_result.as_dict()`) to skip re-OCR and use
|
|
134
|
+
DI's word boxes.
|
|
135
|
+
|
|
136
|
+
Scanned pages with no OCR backend raise `OcrRequiredError` by default; pass
|
|
137
|
+
`on_missing_ocr="skip"` to skip them (with a warning in `res["warnings"]`) and still get
|
|
138
|
+
everything from the native pages. Password-protected PDFs raise `EncryptedPdfError`.
|
|
139
|
+
|
|
140
|
+
> `clean_markdown()` remains a separate, higher-fidelity **native-only** path that borrows
|
|
141
|
+
> pymupdf4llm's layout (headings, paragraphs). `detect_pdf`'s `markdown` is layout-plain but works
|
|
142
|
+
> uniformly on scanned pages too.
|
|
143
|
+
|
|
144
|
+
### Choosing an OCR backend
|
|
145
|
+
|
|
146
|
+
The geometry + CNN carry the detection and are **OCR-independent**; OCR only supplies word boxes
|
|
147
|
+
to attribute strikes to, plus a confidence prior. Benchmarked on a heavily-edited document
|
|
148
|
+
(Azure DI as reference):
|
|
149
|
+
|
|
150
|
+
| Backend | Setup | Struck **regions** | Spatial agreement | Word granularity |
|
|
151
|
+
|---|---|---|---|---|
|
|
152
|
+
| Azure Document Intelligence | cloud, paid | reference | — | exact word boxes |
|
|
153
|
+
| **RapidOCR** | `pip`, no binary | **100% covered** | **~99%** | ~4× coarser (phrase-level) |
|
|
154
|
+
| Tesseract | needs system binary | — | — | genuine word-level |
|
|
155
|
+
|
|
156
|
+
Use `ScanConfig.confidence_free()` with RapidOCR (its confidences cluster near 1.0 and don't
|
|
157
|
+
separate struck from clean text); the default `ScanConfig()` is calibrated to Azure DI, whose
|
|
158
|
+
struck words drop to 0.43–0.94. The DI-decoupled classifier reproduces the original Azure-DI
|
|
159
|
+
pipeline to **99.5%** (1477 vs 1484 struck words on the validation doc).
|
|
160
|
+
|
|
161
|
+
## Low-level building blocks
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
gray = st.render_page_gray(doc[0], dpi=200) # HxW grayscale; RGB/float arrays are coerced
|
|
165
|
+
|
|
166
|
+
lines = st.strike_lines(gray, dpi=200) # OCR-free stroke geometry (strike/underline/rule)
|
|
167
|
+
# pass the dpi the image was rendered/scanned at
|
|
168
|
+
|
|
169
|
+
# word boxes are PAGE FRACTIONS in [0,1], origin top-left — not pixels
|
|
170
|
+
p = st.score_word(gray, (0.12, 0.34, 0.38, 0.36)) # CNN strike probability (0..1)
|
|
171
|
+
|
|
172
|
+
from pdf_strikethrough.ocr import Word
|
|
173
|
+
recs = st.detect_scanned_image(gray, [Word("foo", (0.12, 0.34, 0.38, 0.36), 0.6)])
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
## CLI
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
pdf-strikethrough detect contract.pdf # native pages (scanned pages are
|
|
180
|
+
# skipped with a warning)
|
|
181
|
+
pdf-strikethrough detect scan.pdf --ocr rapidocr # include scanned pages
|
|
182
|
+
pdf-strikethrough detect doc.pdf --method both # max-recall native detection
|
|
183
|
+
pdf-strikethrough detect doc.pdf --json out.json # full struck words + passages
|
|
184
|
+
pdf-strikethrough detect doc.pdf --clean-text clean.txt # surviving text, deletions removed
|
|
185
|
+
pdf-strikethrough detect doc.pdf --markdown marked.md # deletions as ~~struck~~
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
## How it works
|
|
189
|
+
|
|
190
|
+
- **Native**: merged horizontal vector strokes through a word's middle band (excludes under/over-
|
|
191
|
+
lines); coverage ≥ 50% → struck, partials resolved to a char range.
|
|
192
|
+
- **Scanned geometry** (`lines.py`): per-angle morphological opening extracts stroke *fragments*,
|
|
193
|
+
collinear fragments are stitched, then strict filters (spine fill, stroke run-thickness,
|
|
194
|
+
angle/length) separate real strikes from bold crossbars and serif-glyph chains.
|
|
195
|
+
- **CNN** (`cnn.py`, StrikeNet, 79k params): resolves pixel-ambiguous cases — a thin strike over
|
|
196
|
+
an ascender-less word is pixel-identical to a glyph chain, and only a learned model tells them
|
|
197
|
+
apart. Ships as ONNX; set `PDF_STRIKETHROUGH_MODEL_DIR` to use your own weights.
|
|
198
|
+
- **Attribution** (`scanned.py`): assigns strokes to OCR words with char spans and full/partial
|
|
199
|
+
resolution, plus a visual-row "orphan" pass for words the detector's stroke evidence missed.
|
|
200
|
+
|
|
201
|
+
## License
|
|
202
|
+
|
|
203
|
+
MIT.
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
# pdf-strikethrough-detect
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/pdf-strikethrough-detect/)
|
|
4
|
+
[](https://pypi.org/project/pdf-strikethrough-detect/)
|
|
5
|
+
[](https://github.com/niles-liu/pdf-strikethrough-detect/actions/workflows/ci.yml)
|
|
6
|
+
[](LICENSE)
|
|
7
|
+
|
|
8
|
+
Detect **struck-through (deleted) text** in PDFs and scanned document images.
|
|
9
|
+
|
|
10
|
+
Strikethrough detection is a surprisingly unserved niche: most "redline"/diff tools assume clean
|
|
11
|
+
born-digital PDFs and fall apart on scans — which is the real-world case. `pdf-strikethrough-detect`
|
|
12
|
+
handles both, does the hard part (scanned images) with a tiny CPU model, and makes no cloud calls.
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
import pdf_strikethrough as st
|
|
16
|
+
|
|
17
|
+
# born-digital PDF — exact, no OCR
|
|
18
|
+
for w in st.strikethroughs_in_pdf("contract.pdf"):
|
|
19
|
+
print(w["page"], repr(w["chars"]), "partial" if w["partial"] else "full")
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install pdf-strikethrough-detect
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Pure pip, no system binaries required: the CNN runs on ONNX Runtime (CPU) and the ~318 KB model
|
|
29
|
+
ships inside the wheel. Extras:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pip install "pdf-strikethrough-detect[markdown]" # clean_markdown() via pymupdf4llm
|
|
33
|
+
pip install "pdf-strikethrough-detect[rapidocr]" # free scanned-word OCR backend (no binary)
|
|
34
|
+
pip install "pdf-strikethrough-detect[tesseract]" # word-level OCR (also needs the tesseract binary)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Native / born-digital PDFs — exact
|
|
38
|
+
|
|
39
|
+
In a born-digital PDF a strikethrough is a *vector drawing* (a line or thin rect over the text),
|
|
40
|
+
so detection is exact ground truth — no OCR, no model, no guessing. Both vector-rule and
|
|
41
|
+
filled-rect strikethrough styles are handled.
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
import pdf_strikethrough as st
|
|
45
|
+
|
|
46
|
+
for w in st.strikethroughs_in_pdf("contract.pdf"):
|
|
47
|
+
print(w["page"], repr(w["chars"])) # 'chars' = the struck substring
|
|
48
|
+
print(st.clean_markdown("contract.pdf")) # surviving text, deletions removed (needs [markdown])
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Each record: `{page, text, chars, char_span, partial, bbox_frac, coverage, verdict, final}`.
|
|
52
|
+
Partial strikes (`semi-` of `semi-monthly`) are resolved to a char range. `bbox_frac` is in
|
|
53
|
+
fractions of the rendered page (rotation-aware), so it maps directly onto a rendered pixmap.
|
|
54
|
+
|
|
55
|
+
Two native detectors, both **base-PyMuPDF only** (no pymupdf4llm), selected by `method`:
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
st.strikethroughs_in_pdf("contract.pdf", method="vector") # stroke geometry (default) —
|
|
59
|
+
# precise partial-char spans
|
|
60
|
+
st.strikethroughs_in_pdf("contract.pdf", method="flag") # MuPDF's FZ_STEXT_STRIKEOUT signal —
|
|
61
|
+
# also catches font-attribute strikes
|
|
62
|
+
st.strikethroughs_in_pdf("contract.pdf", method="both") # union — maximum recall
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
**Validated across domains.** On 12 public redline PDFs (federal & state regulations, court
|
|
66
|
+
rules, procurement clauses, municipal codes, university policy; 33k struck words),
|
|
67
|
+
**99.9–100% of vector detections are independently confirmed by MuPDF's strikeout signal**,
|
|
68
|
+
and the flag method adds ~2% more words (font-attribute strikes and edge cases) — use
|
|
69
|
+
`method="both"` to capture them. `pymupdf4llm` is not used for detection at all; it is only an
|
|
70
|
+
optional `[markdown]` extra for richer layout in `clean_markdown()`.
|
|
71
|
+
|
|
72
|
+
## Any PDF — routed per page, scanned pages use OCR + CNN
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
import pdf_strikethrough as st
|
|
76
|
+
from pdf_strikethrough.ocr import rapidocr_backend
|
|
77
|
+
from pdf_strikethrough.scanned import ScanConfig
|
|
78
|
+
|
|
79
|
+
res = st.detect_pdf("mixed.pdf",
|
|
80
|
+
ocr=rapidocr_backend(), # for scanned pages
|
|
81
|
+
scan_config=ScanConfig.confidence_free())
|
|
82
|
+
|
|
83
|
+
struck = [w for w in res["words"] if w["final"]] # struck words (boxes, char spans)
|
|
84
|
+
markdown = res["markdown"] # deletions as ~~struck~~
|
|
85
|
+
clean = res["clean_text"] # surviving text, deletions removed
|
|
86
|
+
passages = res["passages"] # grouped deletion sections
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
`detect_pdf` classifies each page native-vs-scanned, runs the exact path on native pages
|
|
90
|
+
(`native_method="vector"|"flag"|"both"`) and the geometry→OCR→CNN pipeline on scanned ones, and
|
|
91
|
+
assembles `markdown` / `clean_text` / `passages` for **both** page kinds from its own strike
|
|
92
|
+
decisions (so the text and the word records always agree — no dependence on an external markdown
|
|
93
|
+
engine). Already have an Azure Document Intelligence result? Pass `di_result=...` (the REST JSON
|
|
94
|
+
dict, an `{'analyzeResult': ...}` envelope, or `sdk_result.as_dict()`) to skip re-OCR and use
|
|
95
|
+
DI's word boxes.
|
|
96
|
+
|
|
97
|
+
Scanned pages with no OCR backend raise `OcrRequiredError` by default; pass
|
|
98
|
+
`on_missing_ocr="skip"` to skip them (with a warning in `res["warnings"]`) and still get
|
|
99
|
+
everything from the native pages. Password-protected PDFs raise `EncryptedPdfError`.
|
|
100
|
+
|
|
101
|
+
> `clean_markdown()` remains a separate, higher-fidelity **native-only** path that borrows
|
|
102
|
+
> pymupdf4llm's layout (headings, paragraphs). `detect_pdf`'s `markdown` is layout-plain but works
|
|
103
|
+
> uniformly on scanned pages too.
|
|
104
|
+
|
|
105
|
+
### Choosing an OCR backend
|
|
106
|
+
|
|
107
|
+
The geometry + CNN carry the detection and are **OCR-independent**; OCR only supplies word boxes
|
|
108
|
+
to attribute strikes to, plus a confidence prior. Benchmarked on a heavily-edited document
|
|
109
|
+
(Azure DI as reference):
|
|
110
|
+
|
|
111
|
+
| Backend | Setup | Struck **regions** | Spatial agreement | Word granularity |
|
|
112
|
+
|---|---|---|---|---|
|
|
113
|
+
| Azure Document Intelligence | cloud, paid | reference | — | exact word boxes |
|
|
114
|
+
| **RapidOCR** | `pip`, no binary | **100% covered** | **~99%** | ~4× coarser (phrase-level) |
|
|
115
|
+
| Tesseract | needs system binary | — | — | genuine word-level |
|
|
116
|
+
|
|
117
|
+
Use `ScanConfig.confidence_free()` with RapidOCR (its confidences cluster near 1.0 and don't
|
|
118
|
+
separate struck from clean text); the default `ScanConfig()` is calibrated to Azure DI, whose
|
|
119
|
+
struck words drop to 0.43–0.94. The DI-decoupled classifier reproduces the original Azure-DI
|
|
120
|
+
pipeline to **99.5%** (1477 vs 1484 struck words on the validation doc).
|
|
121
|
+
|
|
122
|
+
## Low-level building blocks
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
gray = st.render_page_gray(doc[0], dpi=200) # HxW grayscale; RGB/float arrays are coerced
|
|
126
|
+
|
|
127
|
+
lines = st.strike_lines(gray, dpi=200) # OCR-free stroke geometry (strike/underline/rule)
|
|
128
|
+
# pass the dpi the image was rendered/scanned at
|
|
129
|
+
|
|
130
|
+
# word boxes are PAGE FRACTIONS in [0,1], origin top-left — not pixels
|
|
131
|
+
p = st.score_word(gray, (0.12, 0.34, 0.38, 0.36)) # CNN strike probability (0..1)
|
|
132
|
+
|
|
133
|
+
from pdf_strikethrough.ocr import Word
|
|
134
|
+
recs = st.detect_scanned_image(gray, [Word("foo", (0.12, 0.34, 0.38, 0.36), 0.6)])
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## CLI
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
pdf-strikethrough detect contract.pdf # native pages (scanned pages are
|
|
141
|
+
# skipped with a warning)
|
|
142
|
+
pdf-strikethrough detect scan.pdf --ocr rapidocr # include scanned pages
|
|
143
|
+
pdf-strikethrough detect doc.pdf --method both # max-recall native detection
|
|
144
|
+
pdf-strikethrough detect doc.pdf --json out.json # full struck words + passages
|
|
145
|
+
pdf-strikethrough detect doc.pdf --clean-text clean.txt # surviving text, deletions removed
|
|
146
|
+
pdf-strikethrough detect doc.pdf --markdown marked.md # deletions as ~~struck~~
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## How it works
|
|
150
|
+
|
|
151
|
+
- **Native**: merged horizontal vector strokes through a word's middle band (excludes under/over-
|
|
152
|
+
lines); coverage ≥ 50% → struck, partials resolved to a char range.
|
|
153
|
+
- **Scanned geometry** (`lines.py`): per-angle morphological opening extracts stroke *fragments*,
|
|
154
|
+
collinear fragments are stitched, then strict filters (spine fill, stroke run-thickness,
|
|
155
|
+
angle/length) separate real strikes from bold crossbars and serif-glyph chains.
|
|
156
|
+
- **CNN** (`cnn.py`, StrikeNet, 79k params): resolves pixel-ambiguous cases — a thin strike over
|
|
157
|
+
an ascender-less word is pixel-identical to a glyph chain, and only a learned model tells them
|
|
158
|
+
apart. Ships as ONNX; set `PDF_STRIKETHROUGH_MODEL_DIR` to use your own weights.
|
|
159
|
+
- **Attribution** (`scanned.py`): assigns strokes to OCR words with char spans and full/partial
|
|
160
|
+
resolution, plus a visual-row "orphan" pass for words the detector's stroke evidence missed.
|
|
161
|
+
|
|
162
|
+
## License
|
|
163
|
+
|
|
164
|
+
MIT.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pdf-strikethrough-detect"
|
|
7
|
+
version = "0.4.0"
|
|
8
|
+
description = "Detect struck-through (deleted) text in PDFs and scanned document images."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Niles Liu" }]
|
|
14
|
+
keywords = ["strikethrough", "strikeout", "redline", "ocr", "pdf", "document-intelligence",
|
|
15
|
+
"deleted-text", "scanned-documents"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.10",
|
|
19
|
+
"Programming Language :: Python :: 3.11",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Programming Language :: Python :: 3.13",
|
|
22
|
+
"Topic :: Scientific/Engineering :: Image Recognition",
|
|
23
|
+
"Topic :: Text Processing",
|
|
24
|
+
]
|
|
25
|
+
dependencies = [
|
|
26
|
+
"numpy>=1.24",
|
|
27
|
+
"pillow>=9.0",
|
|
28
|
+
"scipy>=1.10",
|
|
29
|
+
"onnxruntime>=1.16",
|
|
30
|
+
"pymupdf>=1.26", # >=1.26: TEXT_COLLECT_STYLES + FZ_STEXT_STRIKEOUT (flag detector)
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
markdown = ["pymupdf4llm>=0.0.27"] # clean_markdown(); >=0.0.22 introduced ~~ strike spans,
|
|
35
|
+
# >=0.0.27 matches the pymupdf>=1.26 base floor
|
|
36
|
+
rapidocr = ["rapidocr>=3.2", "onnxruntime>=1.16"] # free, pip-only scanned-word OCR backend
|
|
37
|
+
# (>=3.2: nested word_results shape)
|
|
38
|
+
tesseract = ["pytesseract>=0.3.10"] # needs the tesseract system binary too
|
|
39
|
+
torch = ["torch>=2.0"] # optional .pt fallback for the CNN
|
|
40
|
+
dev = ["pytest>=7.0"]
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
pdf-strikethrough = "pdf_strikethrough.__main__:main"
|
|
44
|
+
|
|
45
|
+
[project.urls]
|
|
46
|
+
Homepage = "https://github.com/niles-liu/pdf-strikethrough-detect"
|
|
47
|
+
Repository = "https://github.com/niles-liu/pdf-strikethrough-detect"
|
|
48
|
+
Issues = "https://github.com/niles-liu/pdf-strikethrough-detect/issues"
|
|
49
|
+
Changelog = "https://github.com/niles-liu/pdf-strikethrough-detect/blob/main/CHANGELOG.md"
|
|
50
|
+
|
|
51
|
+
[tool.setuptools.packages.find]
|
|
52
|
+
where = ["src"]
|
|
53
|
+
|
|
54
|
+
[tool.setuptools.package-data]
|
|
55
|
+
pdf_strikethrough = ["strike_verdict_cnn.onnx", "strike_verdict_cnn.meta.json"]
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""pdf_strikethrough — detect struck-through (deleted) text in PDFs and scanned document images.
|
|
2
|
+
|
|
3
|
+
Quick start
|
|
4
|
+
-----------
|
|
5
|
+
Native / born-digital PDF (strikethroughs are vector drawings — detection is EXACT, no OCR):
|
|
6
|
+
|
|
7
|
+
import pdf_strikethrough as st
|
|
8
|
+
for w in st.strikethroughs_in_pdf("contract.pdf"):
|
|
9
|
+
print(w["page"], repr(w["chars"]), "partial" if w["partial"] else "full")
|
|
10
|
+
print(st.clean_markdown("contract.pdf")) # surviving text, deletions removed
|
|
11
|
+
|
|
12
|
+
Any PDF (routes native/scanned per page; scanned needs an OCR backend):
|
|
13
|
+
|
|
14
|
+
from pdf_strikethrough.ocr import rapidocr_backend
|
|
15
|
+
from pdf_strikethrough.scanned import ScanConfig
|
|
16
|
+
res = st.detect_pdf("scan.pdf", ocr=rapidocr_backend(),
|
|
17
|
+
scan_config=ScanConfig.confidence_free())
|
|
18
|
+
struck = [w for w in res["words"] if w["final"]]
|
|
19
|
+
|
|
20
|
+
Low-level, on your own image (no PDF):
|
|
21
|
+
|
|
22
|
+
lines = st.strike_lines(gray, dpi=200) # OCR-free stroke geometry
|
|
23
|
+
p = st.score_word(gray, (x0, y0, x1, y1)) # CNN strike probability for a word box
|
|
24
|
+
"""
|
|
25
|
+
from . import cnn, detect, lines, markdown, native, ocr, scanned
|
|
26
|
+
from .cnn import (get_model_meta, score_crops, score_word, std_crop, verdict_of, word_crop_px)
|
|
27
|
+
from .detect import (EncryptedPdfError, OcrRequiredError, apply_cnn_verdict,
|
|
28
|
+
classify_page_source, detect_pdf, detect_scanned_image)
|
|
29
|
+
from .lines import ink_mask, strike_lines, to_gray_u8
|
|
30
|
+
from .native import (native_doc_strikes, native_flag_strikes, native_markdown,
|
|
31
|
+
native_page_strikes, page_strikes, strip_struck_markdown)
|
|
32
|
+
from .ocr import (Word, rapidocr_backend, tesseract_backend, words_from_azure_di)
|
|
33
|
+
from .scanned import ScanConfig, analyze_scanned_page
|
|
34
|
+
|
|
35
|
+
__version__ = "0.4.0"
|
|
36
|
+
|
|
37
|
+
__all__ = [
|
|
38
|
+
# high-level
|
|
39
|
+
"strikethroughs_in_pdf", "clean_markdown", "detect_pdf", "detect_scanned_image",
|
|
40
|
+
"open_pdf", "render_page_gray",
|
|
41
|
+
# native
|
|
42
|
+
"native_page_strikes", "native_flag_strikes", "native_doc_strikes", "page_strikes",
|
|
43
|
+
"native_markdown", "strip_struck_markdown",
|
|
44
|
+
# scanned geometry + classifier
|
|
45
|
+
"strike_lines", "ink_mask", "to_gray_u8", "analyze_scanned_page", "ScanConfig",
|
|
46
|
+
"classify_page_source", "apply_cnn_verdict",
|
|
47
|
+
# errors
|
|
48
|
+
"OcrRequiredError", "EncryptedPdfError",
|
|
49
|
+
# OCR
|
|
50
|
+
"Word", "rapidocr_backend", "tesseract_backend", "words_from_azure_di",
|
|
51
|
+
# CNN
|
|
52
|
+
"score_word", "score_crops", "std_crop", "word_crop_px", "verdict_of", "get_model_meta",
|
|
53
|
+
# submodules
|
|
54
|
+
"cnn", "lines", "native", "ocr", "scanned", "detect", "markdown",
|
|
55
|
+
]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def open_pdf(source):
|
|
59
|
+
"""Open `source` (path, bytes, or an already-open fitz document) as a fitz document.
|
|
60
|
+
Raises EncryptedPdfError for password-protected PDFs."""
|
|
61
|
+
import fitz
|
|
62
|
+
if hasattr(source, "page_count"):
|
|
63
|
+
doc = source
|
|
64
|
+
elif isinstance(source, (bytes, bytearray)):
|
|
65
|
+
doc = fitz.open(stream=bytes(source), filetype="pdf")
|
|
66
|
+
else:
|
|
67
|
+
doc = fitz.open(source)
|
|
68
|
+
if getattr(doc, "needs_pass", False):
|
|
69
|
+
raise EncryptedPdfError(
|
|
70
|
+
"PDF is password-protected; open it with fitz and call doc.authenticate(password) "
|
|
71
|
+
"(or save a decrypted copy) before processing")
|
|
72
|
+
return doc
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def render_page_gray(page, dpi=lines.RENDER_DPI):
|
|
76
|
+
"""Render a fitz page to a grayscale uint8 (H, W) numpy array."""
|
|
77
|
+
import fitz
|
|
78
|
+
import numpy as np
|
|
79
|
+
pix = page.get_pixmap(dpi=dpi, colorspace=fitz.csGRAY)
|
|
80
|
+
return np.frombuffer(pix.samples, dtype=np.uint8).reshape(pix.height, pix.width)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def strikethroughs_in_pdf(source, method="vector"):
|
|
84
|
+
"""Struck-word records for a born-digital PDF (path/bytes/fitz doc), all pages, reading order.
|
|
85
|
+
Exact — driven by the PDF's own strike drawings. `method`: 'vector' (stroke geometry, precise
|
|
86
|
+
partial-char spans; default), 'flag' (MuPDF's strikeout span flag), or 'both' (union, maximum
|
|
87
|
+
recall). Returns [] for a scanned PDF; use ``detect_pdf(..., ocr=...)`` for those."""
|
|
88
|
+
doc = open_pdf(source)
|
|
89
|
+
try:
|
|
90
|
+
return native.native_doc_strikes(doc, method)
|
|
91
|
+
finally:
|
|
92
|
+
if not hasattr(source, "page_count"):
|
|
93
|
+
doc.close()
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def clean_markdown(source):
|
|
97
|
+
"""Markdown for a born-digital PDF with struck (deleted) spans removed — the surviving text.
|
|
98
|
+
Requires the ``[markdown]`` extra (pymupdf4llm); raises ImportError with the pip command
|
|
99
|
+
otherwise. For an extra-free equivalent use ``detect_pdf(source)["clean_text"]``."""
|
|
100
|
+
doc = open_pdf(source)
|
|
101
|
+
try:
|
|
102
|
+
if doc.page_count == 0:
|
|
103
|
+
return ""
|
|
104
|
+
return native.strip_struck_markdown(native.native_markdown(doc))
|
|
105
|
+
finally:
|
|
106
|
+
if not hasattr(source, "page_count"):
|
|
107
|
+
doc.close()
|