pdf-strikethrough-detect 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. pdf_strikethrough_detect-0.4.0/CHANGELOG.md +5 -0
  2. pdf_strikethrough_detect-0.4.0/LICENSE +21 -0
  3. pdf_strikethrough_detect-0.4.0/MANIFEST.in +4 -0
  4. pdf_strikethrough_detect-0.4.0/PKG-INFO +203 -0
  5. pdf_strikethrough_detect-0.4.0/README.md +164 -0
  6. pdf_strikethrough_detect-0.4.0/pyproject.toml +55 -0
  7. pdf_strikethrough_detect-0.4.0/setup.cfg +4 -0
  8. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/__init__.py +107 -0
  9. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/__main__.py +119 -0
  10. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/cnn.py +171 -0
  11. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/detect.py +274 -0
  12. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/lines.py +291 -0
  13. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/markdown.py +113 -0
  14. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/native.py +252 -0
  15. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/ocr.py +119 -0
  16. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/scanned.py +316 -0
  17. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/strike_verdict_cnn.meta.json +9 -0
  18. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough/strike_verdict_cnn.onnx +0 -0
  19. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/PKG-INFO +203 -0
  20. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/SOURCES.txt +23 -0
  21. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/dependency_links.txt +1 -0
  22. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/entry_points.txt +2 -0
  23. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/requires.txt +21 -0
  24. pdf_strikethrough_detect-0.4.0/src/pdf_strikethrough_detect.egg-info/top_level.txt +1 -0
  25. pdf_strikethrough_detect-0.4.0/tests/test_smoke.py +274 -0
@@ -0,0 +1,5 @@
1
+ # Changelog
2
+
3
+ ## 0.4.0 — first public release
4
+
5
+ - Detect struck-through (deleted) text in born-digital PDFs (exact vector/flag detection) and scanned pages (stroke geometry + OCR + ONNX CNN), with `~~struck~~` markdown, clean text, and grouped passages via a Python API and CLI.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Niles Liu
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,4 @@
1
+ include LICENSE README.md CHANGELOG.md
2
+ include src/pdf_strikethrough/strike_verdict_cnn.onnx
3
+ include src/pdf_strikethrough/strike_verdict_cnn.meta.json
4
+ recursive-exclude test_docs *
@@ -0,0 +1,203 @@
1
+ Metadata-Version: 2.4
2
+ Name: pdf-strikethrough-detect
3
+ Version: 0.4.0
4
+ Summary: Detect struck-through (deleted) text in PDFs and scanned document images.
5
+ Author: Niles Liu
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/niles-liu/pdf-strikethrough-detect
8
+ Project-URL: Repository, https://github.com/niles-liu/pdf-strikethrough-detect
9
+ Project-URL: Issues, https://github.com/niles-liu/pdf-strikethrough-detect/issues
10
+ Project-URL: Changelog, https://github.com/niles-liu/pdf-strikethrough-detect/blob/main/CHANGELOG.md
11
+ Keywords: strikethrough,strikeout,redline,ocr,pdf,document-intelligence,deleted-text,scanned-documents
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.10
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Topic :: Scientific/Engineering :: Image Recognition
18
+ Classifier: Topic :: Text Processing
19
+ Requires-Python: >=3.10
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: numpy>=1.24
23
+ Requires-Dist: pillow>=9.0
24
+ Requires-Dist: scipy>=1.10
25
+ Requires-Dist: onnxruntime>=1.16
26
+ Requires-Dist: pymupdf>=1.26
27
+ Provides-Extra: markdown
28
+ Requires-Dist: pymupdf4llm>=0.0.27; extra == "markdown"
29
+ Provides-Extra: rapidocr
30
+ Requires-Dist: rapidocr>=3.2; extra == "rapidocr"
31
+ Requires-Dist: onnxruntime>=1.16; extra == "rapidocr"
32
+ Provides-Extra: tesseract
33
+ Requires-Dist: pytesseract>=0.3.10; extra == "tesseract"
34
+ Provides-Extra: torch
35
+ Requires-Dist: torch>=2.0; extra == "torch"
36
+ Provides-Extra: dev
37
+ Requires-Dist: pytest>=7.0; extra == "dev"
38
+ Dynamic: license-file
39
+
40
+ # pdf-strikethrough-detect
41
+
42
+ [![PyPI](https://img.shields.io/pypi/v/pdf-strikethrough-detect.svg)](https://pypi.org/project/pdf-strikethrough-detect/)
43
+ [![Python versions](https://img.shields.io/pypi/pyversions/pdf-strikethrough-detect.svg)](https://pypi.org/project/pdf-strikethrough-detect/)
44
+ [![CI](https://github.com/niles-liu/pdf-strikethrough-detect/actions/workflows/ci.yml/badge.svg)](https://github.com/niles-liu/pdf-strikethrough-detect/actions/workflows/ci.yml)
45
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
46
+
47
+ Detect **struck-through (deleted) text** in PDFs and scanned document images.
48
+
49
+ Strikethrough detection is a surprisingly unserved niche: most "redline"/diff tools assume clean
50
+ born-digital PDFs and fall apart on scans — which is the real-world case. `pdf-strikethrough-detect`
51
+ handles both, does the hard part (scanned images) with a tiny CPU model, and makes no cloud calls.
52
+
53
+ ```python
54
+ import pdf_strikethrough as st
55
+
56
+ # born-digital PDF — exact, no OCR
57
+ for w in st.strikethroughs_in_pdf("contract.pdf"):
58
+ print(w["page"], repr(w["chars"]), "partial" if w["partial"] else "full")
59
+ ```
60
+
61
+ ## Install
62
+
63
+ ```bash
64
+ pip install pdf-strikethrough-detect
65
+ ```
66
+
67
+ Pure pip, no system binaries required: the CNN runs on ONNX Runtime (CPU) and the ~318 KB model
68
+ ships inside the wheel. Extras:
69
+
70
+ ```bash
71
+ pip install "pdf-strikethrough-detect[markdown]" # clean_markdown() via pymupdf4llm
72
+ pip install "pdf-strikethrough-detect[rapidocr]" # free scanned-word OCR backend (no binary)
73
+ pip install "pdf-strikethrough-detect[tesseract]" # word-level OCR (also needs the tesseract binary)
74
+ ```
75
+
76
+ ## Native / born-digital PDFs — exact
77
+
78
+ In a born-digital PDF a strikethrough is a *vector drawing* (a line or thin rect over the text),
79
+ so detection is exact ground truth — no OCR, no model, no guessing. Both vector-rule and
80
+ filled-rect strikethrough styles are handled.
81
+
82
+ ```python
83
+ import pdf_strikethrough as st
84
+
85
+ for w in st.strikethroughs_in_pdf("contract.pdf"):
86
+ print(w["page"], repr(w["chars"])) # 'chars' = the struck substring
87
+ print(st.clean_markdown("contract.pdf")) # surviving text, deletions removed (needs [markdown])
88
+ ```
89
+
90
+ Each record: `{page, text, chars, char_span, partial, bbox_frac, coverage, verdict, final}`.
91
+ Partial strikes (`semi-` of `semi-monthly`) are resolved to a char range. `bbox_frac` is in
92
+ fractions of the rendered page (rotation-aware), so it maps directly onto a rendered pixmap.
93
+
94
+ Two native detectors, both **base-PyMuPDF only** (no pymupdf4llm), selected by `method`:
95
+
96
+ ```python
97
+ st.strikethroughs_in_pdf("contract.pdf", method="vector") # stroke geometry (default) —
98
+ # precise partial-char spans
99
+ st.strikethroughs_in_pdf("contract.pdf", method="flag") # MuPDF's FZ_STEXT_STRIKEOUT signal —
100
+ # also catches font-attribute strikes
101
+ st.strikethroughs_in_pdf("contract.pdf", method="both") # union — maximum recall
102
+ ```
103
+
104
+ **Validated across domains.** On 12 public redline PDFs (federal & state regulations, court
105
+ rules, procurement clauses, municipal codes, university policy; 33k struck words),
106
+ **99.9–100% of vector detections are independently confirmed by MuPDF's strikeout signal**,
107
+ and the flag method adds ~2% more words (font-attribute strikes and edge cases) — use
108
+ `method="both"` to capture them. `pymupdf4llm` is not used for detection at all; it is only an
109
+ optional `[markdown]` extra for richer layout in `clean_markdown()`.
110
+
111
+ ## Any PDF — routed per page, scanned pages use OCR + CNN
112
+
113
+ ```python
114
+ import pdf_strikethrough as st
115
+ from pdf_strikethrough.ocr import rapidocr_backend
116
+ from pdf_strikethrough.scanned import ScanConfig
117
+
118
+ res = st.detect_pdf("mixed.pdf",
119
+ ocr=rapidocr_backend(), # for scanned pages
120
+ scan_config=ScanConfig.confidence_free())
121
+
122
+ struck = [w for w in res["words"] if w["final"]] # struck words (boxes, char spans)
123
+ markdown = res["markdown"] # deletions as ~~struck~~
124
+ clean = res["clean_text"] # surviving text, deletions removed
125
+ passages = res["passages"] # grouped deletion sections
126
+ ```
127
+
128
+ `detect_pdf` classifies each page native-vs-scanned, runs the exact path on native pages
129
+ (`native_method="vector"|"flag"|"both"`) and the geometry→OCR→CNN pipeline on scanned ones, and
130
+ assembles `markdown` / `clean_text` / `passages` for **both** page kinds from its own strike
131
+ decisions (so the text and the word records always agree — no dependence on an external markdown
132
+ engine). Already have an Azure Document Intelligence result? Pass `di_result=...` (the REST JSON
133
+ dict, an `{'analyzeResult': ...}` envelope, or `sdk_result.as_dict()`) to skip re-OCR and use
134
+ DI's word boxes.
135
+
136
+ Scanned pages with no OCR backend raise `OcrRequiredError` by default; pass
137
+ `on_missing_ocr="skip"` to skip them (with a warning in `res["warnings"]`) and still get
138
+ everything from the native pages. Password-protected PDFs raise `EncryptedPdfError`.
139
+
140
+ > `clean_markdown()` remains a separate, higher-fidelity **native-only** path that borrows
141
+ > pymupdf4llm's layout (headings, paragraphs). `detect_pdf`'s `markdown` is layout-plain but works
142
+ > uniformly on scanned pages too.
143
+
144
+ ### Choosing an OCR backend
145
+
146
+ The geometry + CNN carry the detection and are **OCR-independent**; OCR only supplies word boxes
147
+ to attribute strikes to, plus a confidence prior. Benchmarked on a heavily-edited document
148
+ (Azure DI as reference):
149
+
150
+ | Backend | Setup | Struck **regions** | Spatial agreement | Word granularity |
151
+ |---|---|---|---|---|
152
+ | Azure Document Intelligence | cloud, paid | reference | — | exact word boxes |
153
+ | **RapidOCR** | `pip`, no binary | **100% covered** | **~99%** | ~4× coarser (phrase-level) |
154
+ | Tesseract | needs system binary | — | — | genuine word-level |
155
+
156
+ Use `ScanConfig.confidence_free()` with RapidOCR (its confidences cluster near 1.0 and don't
157
+ separate struck from clean text); the default `ScanConfig()` is calibrated to Azure DI, whose
158
+ struck words drop to 0.43–0.94. The DI-decoupled classifier reproduces the original Azure-DI
159
+ pipeline to **99.5%** (1477 vs 1484 struck words on the validation doc).
160
+
161
+ ## Low-level building blocks
162
+
163
+ ```python
164
+ gray = st.render_page_gray(doc[0], dpi=200) # HxW grayscale; RGB/float arrays are coerced
165
+
166
+ lines = st.strike_lines(gray, dpi=200) # OCR-free stroke geometry (strike/underline/rule)
167
+ # pass the dpi the image was rendered/scanned at
168
+
169
+ # word boxes are PAGE FRACTIONS in [0,1], origin top-left — not pixels
170
+ p = st.score_word(gray, (0.12, 0.34, 0.38, 0.36)) # CNN strike probability (0..1)
171
+
172
+ from pdf_strikethrough.ocr import Word
173
+ recs = st.detect_scanned_image(gray, [Word("foo", (0.12, 0.34, 0.38, 0.36), 0.6)])
174
+ ```
175
+
176
+ ## CLI
177
+
178
+ ```bash
179
+ pdf-strikethrough detect contract.pdf # native pages (scanned pages are
180
+ # skipped with a warning)
181
+ pdf-strikethrough detect scan.pdf --ocr rapidocr # include scanned pages
182
+ pdf-strikethrough detect doc.pdf --method both # max-recall native detection
183
+ pdf-strikethrough detect doc.pdf --json out.json # full struck words + passages
184
+ pdf-strikethrough detect doc.pdf --clean-text clean.txt # surviving text, deletions removed
185
+ pdf-strikethrough detect doc.pdf --markdown marked.md # deletions as ~~struck~~
186
+ ```
187
+
188
+ ## How it works
189
+
190
+ - **Native**: merged horizontal vector strokes through a word's middle band (excludes under/over-
191
+ lines); coverage ≥ 50% → struck, partials resolved to a char range.
192
+ - **Scanned geometry** (`lines.py`): per-angle morphological opening extracts stroke *fragments*,
193
+ collinear fragments are stitched, then strict filters (spine fill, stroke run-thickness,
194
+ angle/length) separate real strikes from bold crossbars and serif-glyph chains.
195
+ - **CNN** (`cnn.py`, StrikeNet, 79k params): resolves pixel-ambiguous cases — a thin strike over
196
+ an ascender-less word is pixel-identical to a glyph chain, and only a learned model tells them
197
+ apart. Ships as ONNX; set `PDF_STRIKETHROUGH_MODEL_DIR` to use your own weights.
198
+ - **Attribution** (`scanned.py`): assigns strokes to OCR words with char spans and full/partial
199
+ resolution, plus a visual-row "orphan" pass for words the detector's stroke evidence missed.
200
+
201
+ ## License
202
+
203
+ MIT.
@@ -0,0 +1,164 @@
1
+ # pdf-strikethrough-detect
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/pdf-strikethrough-detect.svg)](https://pypi.org/project/pdf-strikethrough-detect/)
4
+ [![Python versions](https://img.shields.io/pypi/pyversions/pdf-strikethrough-detect.svg)](https://pypi.org/project/pdf-strikethrough-detect/)
5
+ [![CI](https://github.com/niles-liu/pdf-strikethrough-detect/actions/workflows/ci.yml/badge.svg)](https://github.com/niles-liu/pdf-strikethrough-detect/actions/workflows/ci.yml)
6
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
7
+
8
+ Detect **struck-through (deleted) text** in PDFs and scanned document images.
9
+
10
+ Strikethrough detection is a surprisingly unserved niche: most "redline"/diff tools assume clean
11
+ born-digital PDFs and fall apart on scans — which is the real-world case. `pdf-strikethrough-detect`
12
+ handles both, does the hard part (scanned images) with a tiny CPU model, and makes no cloud calls.
13
+
14
+ ```python
15
+ import pdf_strikethrough as st
16
+
17
+ # born-digital PDF — exact, no OCR
18
+ for w in st.strikethroughs_in_pdf("contract.pdf"):
19
+ print(w["page"], repr(w["chars"]), "partial" if w["partial"] else "full")
20
+ ```
21
+
22
+ ## Install
23
+
24
+ ```bash
25
+ pip install pdf-strikethrough-detect
26
+ ```
27
+
28
+ Pure pip, no system binaries required: the CNN runs on ONNX Runtime (CPU) and the ~318 KB model
29
+ ships inside the wheel. Extras:
30
+
31
+ ```bash
32
+ pip install "pdf-strikethrough-detect[markdown]" # clean_markdown() via pymupdf4llm
33
+ pip install "pdf-strikethrough-detect[rapidocr]" # free scanned-word OCR backend (no binary)
34
+ pip install "pdf-strikethrough-detect[tesseract]" # word-level OCR (also needs the tesseract binary)
35
+ ```
36
+
37
+ ## Native / born-digital PDFs — exact
38
+
39
+ In a born-digital PDF a strikethrough is a *vector drawing* (a line or thin rect over the text),
40
+ so detection is exact ground truth — no OCR, no model, no guessing. Both vector-rule and
41
+ filled-rect strikethrough styles are handled.
42
+
43
+ ```python
44
+ import pdf_strikethrough as st
45
+
46
+ for w in st.strikethroughs_in_pdf("contract.pdf"):
47
+ print(w["page"], repr(w["chars"])) # 'chars' = the struck substring
48
+ print(st.clean_markdown("contract.pdf")) # surviving text, deletions removed (needs [markdown])
49
+ ```
50
+
51
+ Each record: `{page, text, chars, char_span, partial, bbox_frac, coverage, verdict, final}`.
52
+ Partial strikes (`semi-` of `semi-monthly`) are resolved to a char range. `bbox_frac` is in
53
+ fractions of the rendered page (rotation-aware), so it maps directly onto a rendered pixmap.
54
+
55
+ Two native detectors, both **base-PyMuPDF only** (no pymupdf4llm), selected by `method`:
56
+
57
+ ```python
58
+ st.strikethroughs_in_pdf("contract.pdf", method="vector") # stroke geometry (default) —
59
+ # precise partial-char spans
60
+ st.strikethroughs_in_pdf("contract.pdf", method="flag") # MuPDF's FZ_STEXT_STRIKEOUT signal —
61
+ # also catches font-attribute strikes
62
+ st.strikethroughs_in_pdf("contract.pdf", method="both") # union — maximum recall
63
+ ```
64
+
65
+ **Validated across domains.** On 12 public redline PDFs (federal & state regulations, court
66
+ rules, procurement clauses, municipal codes, university policy; 33k struck words),
67
+ **99.9–100% of vector detections are independently confirmed by MuPDF's strikeout signal**,
68
+ and the flag method adds ~2% more words (font-attribute strikes and edge cases) — use
69
+ `method="both"` to capture them. `pymupdf4llm` is not used for detection at all; it is only an
70
+ optional `[markdown]` extra for richer layout in `clean_markdown()`.
71
+
72
+ ## Any PDF — routed per page, scanned pages use OCR + CNN
73
+
74
+ ```python
75
+ import pdf_strikethrough as st
76
+ from pdf_strikethrough.ocr import rapidocr_backend
77
+ from pdf_strikethrough.scanned import ScanConfig
78
+
79
+ res = st.detect_pdf("mixed.pdf",
80
+ ocr=rapidocr_backend(), # for scanned pages
81
+ scan_config=ScanConfig.confidence_free())
82
+
83
+ struck = [w for w in res["words"] if w["final"]] # struck words (boxes, char spans)
84
+ markdown = res["markdown"] # deletions as ~~struck~~
85
+ clean = res["clean_text"] # surviving text, deletions removed
86
+ passages = res["passages"] # grouped deletion sections
87
+ ```
88
+
89
+ `detect_pdf` classifies each page native-vs-scanned, runs the exact path on native pages
90
+ (`native_method="vector"|"flag"|"both"`) and the geometry→OCR→CNN pipeline on scanned ones, and
91
+ assembles `markdown` / `clean_text` / `passages` for **both** page kinds from its own strike
92
+ decisions (so the text and the word records always agree — no dependence on an external markdown
93
+ engine). Already have an Azure Document Intelligence result? Pass `di_result=...` (the REST JSON
94
+ dict, an `{'analyzeResult': ...}` envelope, or `sdk_result.as_dict()`) to skip re-OCR and use
95
+ DI's word boxes.
96
+
97
+ Scanned pages with no OCR backend raise `OcrRequiredError` by default; pass
98
+ `on_missing_ocr="skip"` to skip them (with a warning in `res["warnings"]`) and still get
99
+ everything from the native pages. Password-protected PDFs raise `EncryptedPdfError`.
100
+
101
+ > `clean_markdown()` remains a separate, higher-fidelity **native-only** path that borrows
102
+ > pymupdf4llm's layout (headings, paragraphs). `detect_pdf`'s `markdown` is layout-plain but works
103
+ > uniformly on scanned pages too.
104
+
105
+ ### Choosing an OCR backend
106
+
107
+ The geometry + CNN carry the detection and are **OCR-independent**; OCR only supplies word boxes
108
+ to attribute strikes to, plus a confidence prior. Benchmarked on a heavily-edited document
109
+ (Azure DI as reference):
110
+
111
+ | Backend | Setup | Struck **regions** | Spatial agreement | Word granularity |
112
+ |---|---|---|---|---|
113
+ | Azure Document Intelligence | cloud, paid | reference | — | exact word boxes |
114
+ | **RapidOCR** | `pip`, no binary | **100% covered** | **~99%** | ~4× coarser (phrase-level) |
115
+ | Tesseract | needs system binary | — | — | genuine word-level |
116
+
117
+ Use `ScanConfig.confidence_free()` with RapidOCR (its confidences cluster near 1.0 and don't
118
+ separate struck from clean text); the default `ScanConfig()` is calibrated to Azure DI, whose
119
+ struck words drop to 0.43–0.94. The DI-decoupled classifier reproduces the original Azure-DI
120
+ pipeline to **99.5%** (1477 vs 1484 struck words on the validation doc).
121
+
122
+ ## Low-level building blocks
123
+
124
+ ```python
125
+ gray = st.render_page_gray(doc[0], dpi=200) # HxW grayscale; RGB/float arrays are coerced
126
+
127
+ lines = st.strike_lines(gray, dpi=200) # OCR-free stroke geometry (strike/underline/rule)
128
+ # pass the dpi the image was rendered/scanned at
129
+
130
+ # word boxes are PAGE FRACTIONS in [0,1], origin top-left — not pixels
131
+ p = st.score_word(gray, (0.12, 0.34, 0.38, 0.36)) # CNN strike probability (0..1)
132
+
133
+ from pdf_strikethrough.ocr import Word
134
+ recs = st.detect_scanned_image(gray, [Word("foo", (0.12, 0.34, 0.38, 0.36), 0.6)])
135
+ ```
136
+
137
+ ## CLI
138
+
139
+ ```bash
140
+ pdf-strikethrough detect contract.pdf # native pages (scanned pages are
141
+ # skipped with a warning)
142
+ pdf-strikethrough detect scan.pdf --ocr rapidocr # include scanned pages
143
+ pdf-strikethrough detect doc.pdf --method both # max-recall native detection
144
+ pdf-strikethrough detect doc.pdf --json out.json # full struck words + passages
145
+ pdf-strikethrough detect doc.pdf --clean-text clean.txt # surviving text, deletions removed
146
+ pdf-strikethrough detect doc.pdf --markdown marked.md # deletions as ~~struck~~
147
+ ```
148
+
149
+ ## How it works
150
+
151
+ - **Native**: merged horizontal vector strokes through a word's middle band (excludes under/over-
152
+ lines); coverage ≥ 50% → struck, partials resolved to a char range.
153
+ - **Scanned geometry** (`lines.py`): per-angle morphological opening extracts stroke *fragments*,
154
+ collinear fragments are stitched, then strict filters (spine fill, stroke run-thickness,
155
+ angle/length) separate real strikes from bold crossbars and serif-glyph chains.
156
+ - **CNN** (`cnn.py`, StrikeNet, 79k params): resolves pixel-ambiguous cases — a thin strike over
157
+ an ascender-less word is pixel-identical to a glyph chain, and only a learned model tells them
158
+ apart. Ships as ONNX; set `PDF_STRIKETHROUGH_MODEL_DIR` to use your own weights.
159
+ - **Attribution** (`scanned.py`): assigns strokes to OCR words with char spans and full/partial
160
+ resolution, plus a visual-row "orphan" pass for words the detector's stroke evidence missed.
161
+
162
+ ## License
163
+
164
+ MIT.
@@ -0,0 +1,55 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pdf-strikethrough-detect"
7
+ version = "0.4.0"
8
+ description = "Detect struck-through (deleted) text in PDFs and scanned document images."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Niles Liu" }]
14
+ keywords = ["strikethrough", "strikeout", "redline", "ocr", "pdf", "document-intelligence",
15
+ "deleted-text", "scanned-documents"]
16
+ classifiers = [
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.10",
19
+ "Programming Language :: Python :: 3.11",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Programming Language :: Python :: 3.13",
22
+ "Topic :: Scientific/Engineering :: Image Recognition",
23
+ "Topic :: Text Processing",
24
+ ]
25
+ dependencies = [
26
+ "numpy>=1.24",
27
+ "pillow>=9.0",
28
+ "scipy>=1.10",
29
+ "onnxruntime>=1.16",
30
+ "pymupdf>=1.26", # >=1.26: TEXT_COLLECT_STYLES + FZ_STEXT_STRIKEOUT (flag detector)
31
+ ]
32
+
33
+ [project.optional-dependencies]
34
+ markdown = ["pymupdf4llm>=0.0.27"] # clean_markdown(); >=0.0.22 introduced ~~ strike spans,
35
+ # >=0.0.27 matches the pymupdf>=1.26 base floor
36
+ rapidocr = ["rapidocr>=3.2", "onnxruntime>=1.16"] # free, pip-only scanned-word OCR backend
37
+ # (>=3.2: nested word_results shape)
38
+ tesseract = ["pytesseract>=0.3.10"] # needs the tesseract system binary too
39
+ torch = ["torch>=2.0"] # optional .pt fallback for the CNN
40
+ dev = ["pytest>=7.0"]
41
+
42
+ [project.scripts]
43
+ pdf-strikethrough = "pdf_strikethrough.__main__:main"
44
+
45
+ [project.urls]
46
+ Homepage = "https://github.com/niles-liu/pdf-strikethrough-detect"
47
+ Repository = "https://github.com/niles-liu/pdf-strikethrough-detect"
48
+ Issues = "https://github.com/niles-liu/pdf-strikethrough-detect/issues"
49
+ Changelog = "https://github.com/niles-liu/pdf-strikethrough-detect/blob/main/CHANGELOG.md"
50
+
51
+ [tool.setuptools.packages.find]
52
+ where = ["src"]
53
+
54
+ [tool.setuptools.package-data]
55
+ pdf_strikethrough = ["strike_verdict_cnn.onnx", "strike_verdict_cnn.meta.json"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,107 @@
1
+ """pdf_strikethrough — detect struck-through (deleted) text in PDFs and scanned document images.
2
+
3
+ Quick start
4
+ -----------
5
+ Native / born-digital PDF (strikethroughs are vector drawings — detection is EXACT, no OCR):
6
+
7
+ import pdf_strikethrough as st
8
+ for w in st.strikethroughs_in_pdf("contract.pdf"):
9
+ print(w["page"], repr(w["chars"]), "partial" if w["partial"] else "full")
10
+ print(st.clean_markdown("contract.pdf")) # surviving text, deletions removed
11
+
12
+ Any PDF (routes native/scanned per page; scanned needs an OCR backend):
13
+
14
+ from pdf_strikethrough.ocr import rapidocr_backend
15
+ from pdf_strikethrough.scanned import ScanConfig
16
+ res = st.detect_pdf("scan.pdf", ocr=rapidocr_backend(),
17
+ scan_config=ScanConfig.confidence_free())
18
+ struck = [w for w in res["words"] if w["final"]]
19
+
20
+ Low-level, on your own image (no PDF):
21
+
22
+ lines = st.strike_lines(gray, dpi=200) # OCR-free stroke geometry
23
+ p = st.score_word(gray, (x0, y0, x1, y1)) # CNN strike probability for a word box
24
+ """
25
+ from . import cnn, detect, lines, markdown, native, ocr, scanned
26
+ from .cnn import (get_model_meta, score_crops, score_word, std_crop, verdict_of, word_crop_px)
27
+ from .detect import (EncryptedPdfError, OcrRequiredError, apply_cnn_verdict,
28
+ classify_page_source, detect_pdf, detect_scanned_image)
29
+ from .lines import ink_mask, strike_lines, to_gray_u8
30
+ from .native import (native_doc_strikes, native_flag_strikes, native_markdown,
31
+ native_page_strikes, page_strikes, strip_struck_markdown)
32
+ from .ocr import (Word, rapidocr_backend, tesseract_backend, words_from_azure_di)
33
+ from .scanned import ScanConfig, analyze_scanned_page
34
+
35
+ __version__ = "0.4.0"
36
+
37
+ __all__ = [
38
+ # high-level
39
+ "strikethroughs_in_pdf", "clean_markdown", "detect_pdf", "detect_scanned_image",
40
+ "open_pdf", "render_page_gray",
41
+ # native
42
+ "native_page_strikes", "native_flag_strikes", "native_doc_strikes", "page_strikes",
43
+ "native_markdown", "strip_struck_markdown",
44
+ # scanned geometry + classifier
45
+ "strike_lines", "ink_mask", "to_gray_u8", "analyze_scanned_page", "ScanConfig",
46
+ "classify_page_source", "apply_cnn_verdict",
47
+ # errors
48
+ "OcrRequiredError", "EncryptedPdfError",
49
+ # OCR
50
+ "Word", "rapidocr_backend", "tesseract_backend", "words_from_azure_di",
51
+ # CNN
52
+ "score_word", "score_crops", "std_crop", "word_crop_px", "verdict_of", "get_model_meta",
53
+ # submodules
54
+ "cnn", "lines", "native", "ocr", "scanned", "detect", "markdown",
55
+ ]
56
+
57
+
58
+ def open_pdf(source):
59
+ """Open `source` (path, bytes, or an already-open fitz document) as a fitz document.
60
+ Raises EncryptedPdfError for password-protected PDFs."""
61
+ import fitz
62
+ if hasattr(source, "page_count"):
63
+ doc = source
64
+ elif isinstance(source, (bytes, bytearray)):
65
+ doc = fitz.open(stream=bytes(source), filetype="pdf")
66
+ else:
67
+ doc = fitz.open(source)
68
+ if getattr(doc, "needs_pass", False):
69
+ raise EncryptedPdfError(
70
+ "PDF is password-protected; open it with fitz and call doc.authenticate(password) "
71
+ "(or save a decrypted copy) before processing")
72
+ return doc
73
+
74
+
75
+ def render_page_gray(page, dpi=lines.RENDER_DPI):
76
+ """Render a fitz page to a grayscale uint8 (H, W) numpy array."""
77
+ import fitz
78
+ import numpy as np
79
+ pix = page.get_pixmap(dpi=dpi, colorspace=fitz.csGRAY)
80
+ return np.frombuffer(pix.samples, dtype=np.uint8).reshape(pix.height, pix.width)
81
+
82
+
83
+ def strikethroughs_in_pdf(source, method="vector"):
84
+ """Struck-word records for a born-digital PDF (path/bytes/fitz doc), all pages, reading order.
85
+ Exact — driven by the PDF's own strike drawings. `method`: 'vector' (stroke geometry, precise
86
+ partial-char spans; default), 'flag' (MuPDF's strikeout span flag), or 'both' (union, maximum
87
+ recall). Returns [] for a scanned PDF; use ``detect_pdf(..., ocr=...)`` for those."""
88
+ doc = open_pdf(source)
89
+ try:
90
+ return native.native_doc_strikes(doc, method)
91
+ finally:
92
+ if not hasattr(source, "page_count"):
93
+ doc.close()
94
+
95
+
96
+ def clean_markdown(source):
97
+ """Markdown for a born-digital PDF with struck (deleted) spans removed — the surviving text.
98
+ Requires the ``[markdown]`` extra (pymupdf4llm); raises ImportError with the pip command
99
+ otherwise. For an extra-free equivalent use ``detect_pdf(source)["clean_text"]``."""
100
+ doc = open_pdf(source)
101
+ try:
102
+ if doc.page_count == 0:
103
+ return ""
104
+ return native.strip_struck_markdown(native.native_markdown(doc))
105
+ finally:
106
+ if not hasattr(source, "page_count"):
107
+ doc.close()