scanlayer 1.0.0__tar.gz → 1.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scanlayer-1.0.2/PKG-INFO +233 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/pyproject.toml +13 -2
- scanlayer-1.0.2/scanlayer.egg-info/PKG-INFO +233 -0
- scanlayer-1.0.0/PKG-INFO +0 -17
- scanlayer-1.0.0/scanlayer.egg-info/PKG-INFO +0 -17
- {scanlayer-1.0.0 → scanlayer-1.0.2}/LICENSE.md +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/README.md +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/__init__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/__main__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/cli/__init__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/cli/dry_run.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/cli/parser.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/cli/run.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/config.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/fonts/DejaVuSans.ttf +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/layout/__init__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/layout/columns.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/main.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/ocr/__init__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/ocr/engine.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/ocr/export.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/pdf/__init__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/pdf/builder.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/pdf/fonts.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/preprocessing/__init__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/preprocessing/enhance.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/utils/__init__.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/utils/debug_image.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/utils/errors.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/utils/logger.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer/utils/validators.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer.egg-info/SOURCES.txt +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer.egg-info/dependency_links.txt +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer.egg-info/entry_points.txt +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer.egg-info/requires.txt +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/scanlayer.egg-info/top_level.txt +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/setup.cfg +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_columns.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_config.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_dry_run.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_export.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_fonts.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_main_integration.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_pdf_builder.py +0 -0
- {scanlayer-1.0.0 → scanlayer-1.0.2}/tests/test_validators.py +0 -0
scanlayer-1.0.2/PKG-INFO
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scanlayer
|
|
3
|
+
Version: 1.0.2
|
|
4
|
+
Summary: Image (scan/photo) to searchable PDF, or raw OCR export, via Tesseract.
|
|
5
|
+
Author: Hyacinthe-primus
|
|
6
|
+
Author-email: Hyacinthe-primus <hyacintheatho91@gmail.com>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/Hyacinthe-primus/scanlayer
|
|
9
|
+
Project-URL: Documentation, https://Hyacinthe-primus.github.io/scanlayer/
|
|
10
|
+
Project-URL: Repository, https://github.com/Hyacinthe-primus/scanlayer
|
|
11
|
+
Requires-Python: >=3.9
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE.md
|
|
14
|
+
Requires-Dist: pytesseract>=0.3.10
|
|
15
|
+
Requires-Dist: Pillow>=10.0.0
|
|
16
|
+
Requires-Dist: reportlab>=4.0.0
|
|
17
|
+
Requires-Dist: opencv-python-headless>=4.8.0
|
|
18
|
+
Requires-Dist: numpy>=1.24.0
|
|
19
|
+
Requires-Dist: pdf2image>=1.17.0
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest>=7.0; extra == "test"
|
|
22
|
+
Provides-Extra: yaml
|
|
23
|
+
Requires-Dist: PyYAML>=6.0; extra == "yaml"
|
|
24
|
+
Dynamic: license-file
|
|
25
|
+
|
|
26
|
+
<div align="center">
|
|
27
|
+
<img src="scanlayer.png" alt="scanlayer" width="110"/>
|
|
28
|
+
<h1>scanlayer</h1>
|
|
29
|
+
<p><em>Scanned image or photographed document → searchable PDF, or raw OCR text/JSON/TSV/hOCR.</em></p>
|
|
30
|
+
</div>
|
|
31
|
+
|
|
32
|
+
**Turn a scanned image or photographed document into a searchable PDF**, or
|
|
33
|
+
export the raw OCR result as plain text, JSON, TSV, or hOCR. Use it as a
|
|
34
|
+
command-line tool or as a Python library; both are the same engine underneath.
|
|
35
|
+
|
|
36
|
+
[](#)
|
|
37
|
+

|
|
38
|
+
[](https://Hyacinthe-primus.github.io/scanlayer/)
|
|
39
|
+
[](LICENSE.md)
|
|
40
|
+
|
|
41
|
+
---
|
|
42
|
+
|
|
43
|
+
## What it does
|
|
44
|
+
|
|
45
|
+
A scanned invoice, a phone photo of a letter, a stack of photographed pages:
|
|
46
|
+
scanlayer runs it through [Tesseract OCR](https://github.com/tesseract-ocr/tesseract)
|
|
47
|
+
and gives you back either:
|
|
48
|
+
|
|
49
|
+
- a **searchable PDF**: the original page image, with an invisible, precisely
|
|
50
|
+
positioned text layer over it, so you can select and search text exactly
|
|
51
|
+
where it visually appears, or
|
|
52
|
+
- the **raw OCR result** as `txt`, `json`, `tsv`, or `hocr`: text, per-word
|
|
53
|
+
confidence, and bounding boxes, no PDF built at all.
|
|
54
|
+
|
|
55
|
+
Along the way it automatically straightens rotated/skewed pages, corrects
|
|
56
|
+
uneven lighting, denoises and sharpens for OCR accuracy without touching what
|
|
57
|
+
you actually see in the output, reconstructs correct reading order on
|
|
58
|
+
genuine multi-column pages, and races several OCR configurations against
|
|
59
|
+
each other to pick the most confident result.
|
|
60
|
+
|
|
61
|
+
| | |
|
|
62
|
+
|---|---|
|
|
63
|
+
| **Two interfaces, one engine** | `scanlayer` CLI and `import scanlayer` call the exact same pipeline |
|
|
64
|
+
| **Five output formats** | Searchable `pdf`, or raw `txt` / `json` / `tsv` / `hocr` |
|
|
65
|
+
| **Real column detection** | Two-column articles/letters read in correct order, not interleaved |
|
|
66
|
+
| **Batch, merge, native PDF input** | Convert a folder in one call, merge pages into one PDF, or `--dry-run` a batch before spending time on OCR |
|
|
67
|
+
| **One configuration surface** | `configure()`, a JSON/YAML profile, or CLI flags, documented precedence |
|
|
68
|
+
| **Debug overlay** | `--debug-image` draws every word, color-coded by confidence |
|
|
69
|
+
|
|
70
|
+
See the [feature catalog](https://Hyacinthe-primus.github.io/scanlayer/features.html) for the complete list, and
|
|
71
|
+
[Roadmap & Limitations](https://Hyacinthe-primus.github.io/scanlayer/roadmap.html) for what's deliberately out of
|
|
72
|
+
scope or not built yet.
|
|
73
|
+
|
|
74
|
+
## Requirements
|
|
75
|
+
|
|
76
|
+
- Python 3.9+
|
|
77
|
+
- [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) (a separate, system-level install, see below)
|
|
78
|
+
- [poppler](https://poppler.freedesktop.org/) *only if* you feed scanlayer a native `.pdf` file directly
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
# Debian / Ubuntu
|
|
82
|
+
sudo apt install tesseract-ocr poppler-utils
|
|
83
|
+
|
|
84
|
+
# macOS (Homebrew)
|
|
85
|
+
brew install tesseract poppler
|
|
86
|
+
|
|
87
|
+
# Windows: Tesseract -> https://github.com/tesseract-ocr/tesseract/wiki
|
|
88
|
+
# poppler -> download a release, add its bin/ to PATH
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Full detail, including how scanlayer locates the Tesseract binary
|
|
92
|
+
automatically and how to bundle your own, is in
|
|
93
|
+
[Installation](https://Hyacinthe-primus.github.io/scanlayer/installation.html) and [Bundling Tesseract](https://Hyacinthe-primus.github.io/scanlayer/bundling-tesseract.html).
|
|
94
|
+
|
|
95
|
+
## Install
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
pip install scanlayer
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
This installs the `scanlayer` console command and makes `import scanlayer`
|
|
102
|
+
available anywhere on the machine.
|
|
103
|
+
|
|
104
|
+
Working on scanlayer itself, or want to run it straight from a checkout with
|
|
105
|
+
no install at all?
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
git clone https://github.com/Hyacinthe-primus/scanlayer.git
|
|
109
|
+
cd scanlayer # the repo root, which contains requirements.txt
|
|
110
|
+
pip install -r requirements.txt
|
|
111
|
+
python -m scanlayer invoice.jpg -o invoice.pdf # works with no install at all
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full contributor setup
|
|
115
|
+
(editable install and running the test suite).
|
|
116
|
+
|
|
117
|
+
## Quick start
|
|
118
|
+
|
|
119
|
+
**As a CLI:**
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
scanlayer invoice.jpg -o invoice.pdf --lang fra+eng --dpi 300
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
**As a library:**
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
import scanlayer
|
|
129
|
+
|
|
130
|
+
result = scanlayer.convert("invoice.jpg", "invoice.pdf", lang="fra+eng", dpi=300)
|
|
131
|
+
print(f"{result.words_count} words, {result.mean_confidence:.1f}% confidence")
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Every CLI flag and library keyword argument in this project are named to
|
|
135
|
+
match each other (`--lang` ↔ `lang=`, `--dpi` ↔ `dpi=`, and so on), see
|
|
136
|
+
[Examples](https://Hyacinthe-primus.github.io/scanlayer/examples.html) for every
|
|
137
|
+
feature shown both ways, side by side, and
|
|
138
|
+
[CLI Reference](https://Hyacinthe-primus.github.io/scanlayer/cli-reference.html) /
|
|
139
|
+
[Library API](https://Hyacinthe-primus.github.io/scanlayer/library-api.html) for
|
|
140
|
+
the complete details of each.
|
|
141
|
+
|
|
142
|
+
A few more common cases:
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
# Batch-convert a folder
|
|
146
|
+
scanlayer *.jpg -o ./converted/
|
|
147
|
+
|
|
148
|
+
# Merge several photographed pages into one searchable PDF
|
|
149
|
+
scanlayer page1.jpg page2.jpg page3.jpg -o report.pdf --merge
|
|
150
|
+
|
|
151
|
+
# Export raw OCR text/JSON instead of a PDF
|
|
152
|
+
scanlayer invoice.jpg -o invoice.json --format json
|
|
153
|
+
|
|
154
|
+
# See what OCR actually detected, color-coded by confidence
|
|
155
|
+
scanlayer invoice.jpg --debug-image
|
|
156
|
+
|
|
157
|
+
# Validate a batch before spending time on OCR: files exist,
|
|
158
|
+
# Tesseract reachable, output paths writable
|
|
159
|
+
scanlayer *.jpg -o ./converted/ --dry-run
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
```python
|
|
163
|
+
import scanlayer
|
|
164
|
+
|
|
165
|
+
# Batch
|
|
166
|
+
result = scanlayer.convert_batch(["*.jpg"], "./converted/")
|
|
167
|
+
|
|
168
|
+
# Merge
|
|
169
|
+
scanlayer.convert_merge(["page1.jpg", "page2.jpg", "page3.jpg"], "report.pdf")
|
|
170
|
+
|
|
171
|
+
# Raw export
|
|
172
|
+
scanlayer.convert("invoice.jpg", "invoice.json", output_format="json")
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
## Documentation
|
|
176
|
+
|
|
177
|
+
| | |
|
|
178
|
+
|---|---|
|
|
179
|
+
| **[Installation](https://Hyacinthe-primus.github.io/scanlayer/installation.html)** | Tesseract, poppler, and the two ways to install scanlayer itself |
|
|
180
|
+
| **[Examples](https://Hyacinthe-primus.github.io/scanlayer/examples.html)** | Every feature, CLI and library side by side |
|
|
181
|
+
| **[CLI Reference](https://Hyacinthe-primus.github.io/scanlayer/cli-reference.html)** | Every flag and exit code |
|
|
182
|
+
| **[Library API](https://Hyacinthe-primus.github.io/scanlayer/library-api.html)** | `convert()`, `convert_batch()`, `convert_merge()`, exceptions, and the full low-level pipeline API |
|
|
183
|
+
| **[Feature Catalog](https://Hyacinthe-primus.github.io/scanlayer/features.html)** | Everything scanlayer does, by pipeline stage |
|
|
184
|
+
| **[Configuration](https://Hyacinthe-primus.github.io/scanlayer/configuration.html)** | `configure()`, config files, precedence, every tunable |
|
|
185
|
+
| **[Output Formats](https://Hyacinthe-primus.github.io/scanlayer/output-formats.html)** | The `pdf`/`txt`/`json`/`tsv`/`hocr` schemas |
|
|
186
|
+
| **[Multi-Page & Merge](https://Hyacinthe-primus.github.io/scanlayer/multi-page-merge.html)** | Batching, merging, native PDF input |
|
|
187
|
+
| **[Debug Visualization](https://Hyacinthe-primus.github.io/scanlayer/debug-visualization.html)** | Reading the `--debug-image` confidence overlay |
|
|
188
|
+
| **[Bundling Tesseract](https://Hyacinthe-primus.github.io/scanlayer/bundling-tesseract.html)** | Shipping your own Tesseract binary |
|
|
189
|
+
| **[Troubleshooting](https://Hyacinthe-primus.github.io/scanlayer/troubleshooting.html)** | Common errors and fixes |
|
|
190
|
+
| **[Roadmap & Limitations](https://Hyacinthe-primus.github.io/scanlayer/roadmap.html)** | What's missing and what's deliberately out of scope |
|
|
191
|
+
|
|
192
|
+
The site is published from the `gh-pages` branch at
|
|
193
|
+
https://Hyacinthe-primus.github.io/scanlayer/. To browse it locally, open
|
|
194
|
+
`index.html` in a `gh-pages` worktree (e.g. `git worktree add <path> gh-pages`);
|
|
195
|
+
it has no build step or server-side dependency.
|
|
196
|
+
|
|
197
|
+
## Repository layout
|
|
198
|
+
|
|
199
|
+
```
|
|
200
|
+
.
|
|
201
|
+
├── .github/ # FUNDING.yml
|
|
202
|
+
├── scanlayer/ # library source
|
|
203
|
+
│ ├── __init__.py
|
|
204
|
+
│ ├── __main__.py # enables `python -m scanlayer`
|
|
205
|
+
│ ├── main.py # public API: convert()/convert_batch()/convert_merge()
|
|
206
|
+
│ ├── config.py
|
|
207
|
+
│ ├── cli/ # argparse CLI: parser.py, run.py, dry_run.py
|
|
208
|
+
│ ├── fonts/ # bundled DejaVu Sans for the PDF text layer
|
|
209
|
+
│ ├── preprocessing/
|
|
210
|
+
│ ├── ocr/
|
|
211
|
+
│ ├── layout/
|
|
212
|
+
│ ├── pdf/
|
|
213
|
+
│ └── utils/
|
|
214
|
+
├── tests/ # pytest suite
|
|
215
|
+
├── pyproject.toml
|
|
216
|
+
├── requirements.txt
|
|
217
|
+
├── README.md
|
|
218
|
+
├── CONTRIBUTING.md
|
|
219
|
+
├── SECURITY.md
|
|
220
|
+
└── LICENSE.md
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
## Contributing
|
|
224
|
+
|
|
225
|
+
Bug reports, fixes, and feature discussions are welcome: see
|
|
226
|
+
[CONTRIBUTING.md](CONTRIBUTING.md) for how to set up a development
|
|
227
|
+
environment and what to include in a pull request. There is a `tests/`
|
|
228
|
+
directory (now including `utils/validators.py` and the `--dry-run` flag, with
|
|
229
|
+
cross-platform coverage for Tesseract discovery on Windows/macOS/Linux) and it runs with `pytest`, but no CI yet; see [Roadmap & Limitations](https://Hyacinthe-primus.github.io/scanlayer/roadmap.html) and the [Adding tests](CONTRIBUTING.md#adding-tests) section of the contributing guide for where coverage is thinnest.
|
|
230
|
+
|
|
231
|
+
## License
|
|
232
|
+
|
|
233
|
+
[MIT](LICENSE.md).
|
|
@@ -1,12 +1,18 @@
|
|
|
1
1
|
[build-system]
|
|
2
|
-
requires = ["setuptools>=
|
|
2
|
+
requires = ["setuptools>=77.0"]
|
|
3
3
|
build-backend = "setuptools.build_meta"
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "scanlayer"
|
|
7
|
-
version = "1.0.
|
|
7
|
+
version = "1.0.2"
|
|
8
8
|
description = "Image (scan/photo) to searchable PDF, or raw OCR export, via Tesseract."
|
|
9
|
+
readme = "README.md"
|
|
9
10
|
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Hyacinthe-primus" },
|
|
14
|
+
{ name = "Hyacinthe-primus", email = "hyacintheatho91@gmail.com" },
|
|
15
|
+
]
|
|
10
16
|
dependencies = [
|
|
11
17
|
"pytesseract>=0.3.10",
|
|
12
18
|
"Pillow>=10.0.0",
|
|
@@ -23,6 +29,11 @@ yaml = ["PyYAML>=6.0"]
|
|
|
23
29
|
[project.scripts]
|
|
24
30
|
scanlayer = "scanlayer.cli:main"
|
|
25
31
|
|
|
32
|
+
[project.urls]
|
|
33
|
+
Homepage = "https://github.com/Hyacinthe-primus/scanlayer"
|
|
34
|
+
Documentation = "https://Hyacinthe-primus.github.io/scanlayer/"
|
|
35
|
+
Repository = "https://github.com/Hyacinthe-primus/scanlayer"
|
|
36
|
+
|
|
26
37
|
[tool.setuptools.packages.find]
|
|
27
38
|
include = ["scanlayer*"]
|
|
28
39
|
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scanlayer
|
|
3
|
+
Version: 1.0.2
|
|
4
|
+
Summary: Image (scan/photo) to searchable PDF, or raw OCR export, via Tesseract.
|
|
5
|
+
Author: Hyacinthe-primus
|
|
6
|
+
Author-email: Hyacinthe-primus <hyacintheatho91@gmail.com>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/Hyacinthe-primus/scanlayer
|
|
9
|
+
Project-URL: Documentation, https://Hyacinthe-primus.github.io/scanlayer/
|
|
10
|
+
Project-URL: Repository, https://github.com/Hyacinthe-primus/scanlayer
|
|
11
|
+
Requires-Python: >=3.9
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE.md
|
|
14
|
+
Requires-Dist: pytesseract>=0.3.10
|
|
15
|
+
Requires-Dist: Pillow>=10.0.0
|
|
16
|
+
Requires-Dist: reportlab>=4.0.0
|
|
17
|
+
Requires-Dist: opencv-python-headless>=4.8.0
|
|
18
|
+
Requires-Dist: numpy>=1.24.0
|
|
19
|
+
Requires-Dist: pdf2image>=1.17.0
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest>=7.0; extra == "test"
|
|
22
|
+
Provides-Extra: yaml
|
|
23
|
+
Requires-Dist: PyYAML>=6.0; extra == "yaml"
|
|
24
|
+
Dynamic: license-file
|
|
25
|
+
|
|
26
|
+
<div align="center">
|
|
27
|
+
<img src="scanlayer.png" alt="scanlayer" width="110"/>
|
|
28
|
+
<h1>scanlayer</h1>
|
|
29
|
+
<p><em>Scanned image or photographed document → searchable PDF, or raw OCR text/JSON/TSV/hOCR.</em></p>
|
|
30
|
+
</div>
|
|
31
|
+
|
|
32
|
+
**Turn a scanned image or photographed document into a searchable PDF**, or
|
|
33
|
+
export the raw OCR result as plain text, JSON, TSV, or hOCR. Use it as a
|
|
34
|
+
command-line tool or as a Python library; both are the same engine underneath.
|
|
35
|
+
|
|
36
|
+
[](#)
|
|
37
|
+

|
|
38
|
+
[](https://Hyacinthe-primus.github.io/scanlayer/)
|
|
39
|
+
[](LICENSE.md)
|
|
40
|
+
|
|
41
|
+
---
|
|
42
|
+
|
|
43
|
+
## What it does
|
|
44
|
+
|
|
45
|
+
A scanned invoice, a phone photo of a letter, a stack of photographed pages:
|
|
46
|
+
scanlayer runs it through [Tesseract OCR](https://github.com/tesseract-ocr/tesseract)
|
|
47
|
+
and gives you back either:
|
|
48
|
+
|
|
49
|
+
- a **searchable PDF**: the original page image, with an invisible, precisely
|
|
50
|
+
positioned text layer over it, so you can select and search text exactly
|
|
51
|
+
where it visually appears, or
|
|
52
|
+
- the **raw OCR result** as `txt`, `json`, `tsv`, or `hocr`: text, per-word
|
|
53
|
+
confidence, and bounding boxes, no PDF built at all.
|
|
54
|
+
|
|
55
|
+
Along the way it automatically straightens rotated/skewed pages, corrects
|
|
56
|
+
uneven lighting, denoises and sharpens for OCR accuracy without touching what
|
|
57
|
+
you actually see in the output, reconstructs correct reading order on
|
|
58
|
+
genuine multi-column pages, and races several OCR configurations against
|
|
59
|
+
each other to pick the most confident result.
|
|
60
|
+
|
|
61
|
+
| | |
|
|
62
|
+
|---|---|
|
|
63
|
+
| **Two interfaces, one engine** | `scanlayer` CLI and `import scanlayer` call the exact same pipeline |
|
|
64
|
+
| **Five output formats** | Searchable `pdf`, or raw `txt` / `json` / `tsv` / `hocr` |
|
|
65
|
+
| **Real column detection** | Two-column articles/letters read in correct order, not interleaved |
|
|
66
|
+
| **Batch, merge, native PDF input** | Convert a folder in one call, merge pages into one PDF, or `--dry-run` a batch before spending time on OCR |
|
|
67
|
+
| **One configuration surface** | `configure()`, a JSON/YAML profile, or CLI flags, documented precedence |
|
|
68
|
+
| **Debug overlay** | `--debug-image` draws every word, color-coded by confidence |
|
|
69
|
+
|
|
70
|
+
See the [feature catalog](https://Hyacinthe-primus.github.io/scanlayer/features.html) for the complete list, and
|
|
71
|
+
[Roadmap & Limitations](https://Hyacinthe-primus.github.io/scanlayer/roadmap.html) for what's deliberately out of
|
|
72
|
+
scope or not built yet.
|
|
73
|
+
|
|
74
|
+
## Requirements
|
|
75
|
+
|
|
76
|
+
- Python 3.9+
|
|
77
|
+
- [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) (a separate, system-level install, see below)
|
|
78
|
+
- [poppler](https://poppler.freedesktop.org/) *only if* you feed scanlayer a native `.pdf` file directly
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
# Debian / Ubuntu
|
|
82
|
+
sudo apt install tesseract-ocr poppler-utils
|
|
83
|
+
|
|
84
|
+
# macOS (Homebrew)
|
|
85
|
+
brew install tesseract poppler
|
|
86
|
+
|
|
87
|
+
# Windows: Tesseract -> https://github.com/tesseract-ocr/tesseract/wiki
|
|
88
|
+
# poppler -> download a release, add its bin/ to PATH
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Full detail, including how scanlayer locates the Tesseract binary
|
|
92
|
+
automatically and how to bundle your own, is in
|
|
93
|
+
[Installation](https://Hyacinthe-primus.github.io/scanlayer/installation.html) and [Bundling Tesseract](https://Hyacinthe-primus.github.io/scanlayer/bundling-tesseract.html).
|
|
94
|
+
|
|
95
|
+
## Install
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
pip install scanlayer
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
This installs the `scanlayer` console command and makes `import scanlayer`
|
|
102
|
+
available anywhere on the machine.
|
|
103
|
+
|
|
104
|
+
Working on scanlayer itself, or want to run it straight from a checkout with
|
|
105
|
+
no install at all?
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
git clone https://github.com/Hyacinthe-primus/scanlayer.git
|
|
109
|
+
cd scanlayer # the repo root, which contains requirements.txt
|
|
110
|
+
pip install -r requirements.txt
|
|
111
|
+
python -m scanlayer invoice.jpg -o invoice.pdf # works with no install at all
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) for the full contributor setup
|
|
115
|
+
(editable install and running the test suite).
|
|
116
|
+
|
|
117
|
+
## Quick start
|
|
118
|
+
|
|
119
|
+
**As a CLI:**
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
scanlayer invoice.jpg -o invoice.pdf --lang fra+eng --dpi 300
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
**As a library:**
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
import scanlayer
|
|
129
|
+
|
|
130
|
+
result = scanlayer.convert("invoice.jpg", "invoice.pdf", lang="fra+eng", dpi=300)
|
|
131
|
+
print(f"{result.words_count} words, {result.mean_confidence:.1f}% confidence")
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Every CLI flag and library keyword argument in this project are named to
|
|
135
|
+
match each other (`--lang` ↔ `lang=`, `--dpi` ↔ `dpi=`, and so on), see
|
|
136
|
+
[Examples](https://Hyacinthe-primus.github.io/scanlayer/examples.html) for every
|
|
137
|
+
feature shown both ways, side by side, and
|
|
138
|
+
[CLI Reference](https://Hyacinthe-primus.github.io/scanlayer/cli-reference.html) /
|
|
139
|
+
[Library API](https://Hyacinthe-primus.github.io/scanlayer/library-api.html) for
|
|
140
|
+
the complete details of each.
|
|
141
|
+
|
|
142
|
+
A few more common cases:
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
# Batch-convert a folder
|
|
146
|
+
scanlayer *.jpg -o ./converted/
|
|
147
|
+
|
|
148
|
+
# Merge several photographed pages into one searchable PDF
|
|
149
|
+
scanlayer page1.jpg page2.jpg page3.jpg -o report.pdf --merge
|
|
150
|
+
|
|
151
|
+
# Export raw OCR text/JSON instead of a PDF
|
|
152
|
+
scanlayer invoice.jpg -o invoice.json --format json
|
|
153
|
+
|
|
154
|
+
# See what OCR actually detected, color-coded by confidence
|
|
155
|
+
scanlayer invoice.jpg --debug-image
|
|
156
|
+
|
|
157
|
+
# Validate a batch before spending time on OCR: files exist,
|
|
158
|
+
# Tesseract reachable, output paths writable
|
|
159
|
+
scanlayer *.jpg -o ./converted/ --dry-run
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
```python
|
|
163
|
+
import scanlayer
|
|
164
|
+
|
|
165
|
+
# Batch
|
|
166
|
+
result = scanlayer.convert_batch(["*.jpg"], "./converted/")
|
|
167
|
+
|
|
168
|
+
# Merge
|
|
169
|
+
scanlayer.convert_merge(["page1.jpg", "page2.jpg", "page3.jpg"], "report.pdf")
|
|
170
|
+
|
|
171
|
+
# Raw export
|
|
172
|
+
scanlayer.convert("invoice.jpg", "invoice.json", output_format="json")
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
## Documentation
|
|
176
|
+
|
|
177
|
+
| | |
|
|
178
|
+
|---|---|
|
|
179
|
+
| **[Installation](https://Hyacinthe-primus.github.io/scanlayer/installation.html)** | Tesseract, poppler, and the two ways to install scanlayer itself |
|
|
180
|
+
| **[Examples](https://Hyacinthe-primus.github.io/scanlayer/examples.html)** | Every feature, CLI and library side by side |
|
|
181
|
+
| **[CLI Reference](https://Hyacinthe-primus.github.io/scanlayer/cli-reference.html)** | Every flag and exit code |
|
|
182
|
+
| **[Library API](https://Hyacinthe-primus.github.io/scanlayer/library-api.html)** | `convert()`, `convert_batch()`, `convert_merge()`, exceptions, and the full low-level pipeline API |
|
|
183
|
+
| **[Feature Catalog](https://Hyacinthe-primus.github.io/scanlayer/features.html)** | Everything scanlayer does, by pipeline stage |
|
|
184
|
+
| **[Configuration](https://Hyacinthe-primus.github.io/scanlayer/configuration.html)** | `configure()`, config files, precedence, every tunable |
|
|
185
|
+
| **[Output Formats](https://Hyacinthe-primus.github.io/scanlayer/output-formats.html)** | The `pdf`/`txt`/`json`/`tsv`/`hocr` schemas |
|
|
186
|
+
| **[Multi-Page & Merge](https://Hyacinthe-primus.github.io/scanlayer/multi-page-merge.html)** | Batching, merging, native PDF input |
|
|
187
|
+
| **[Debug Visualization](https://Hyacinthe-primus.github.io/scanlayer/debug-visualization.html)** | Reading the `--debug-image` confidence overlay |
|
|
188
|
+
| **[Bundling Tesseract](https://Hyacinthe-primus.github.io/scanlayer/bundling-tesseract.html)** | Shipping your own Tesseract binary |
|
|
189
|
+
| **[Troubleshooting](https://Hyacinthe-primus.github.io/scanlayer/troubleshooting.html)** | Common errors and fixes |
|
|
190
|
+
| **[Roadmap & Limitations](https://Hyacinthe-primus.github.io/scanlayer/roadmap.html)** | What's missing and what's deliberately out of scope |
|
|
191
|
+
|
|
192
|
+
The site is published from the `gh-pages` branch at
|
|
193
|
+
https://Hyacinthe-primus.github.io/scanlayer/. To browse it locally, open
|
|
194
|
+
`index.html` in a `gh-pages` worktree (e.g. `git worktree add <path> gh-pages`);
|
|
195
|
+
it has no build step or server-side dependency.
|
|
196
|
+
|
|
197
|
+
## Repository layout
|
|
198
|
+
|
|
199
|
+
```
|
|
200
|
+
.
|
|
201
|
+
├── .github/ # FUNDING.yml
|
|
202
|
+
├── scanlayer/ # library source
|
|
203
|
+
│ ├── __init__.py
|
|
204
|
+
│ ├── __main__.py # enables `python -m scanlayer`
|
|
205
|
+
│ ├── main.py # public API: convert()/convert_batch()/convert_merge()
|
|
206
|
+
│ ├── config.py
|
|
207
|
+
│ ├── cli/ # argparse CLI: parser.py, run.py, dry_run.py
|
|
208
|
+
│ ├── fonts/ # bundled DejaVu Sans for the PDF text layer
|
|
209
|
+
│ ├── preprocessing/
|
|
210
|
+
│ ├── ocr/
|
|
211
|
+
│ ├── layout/
|
|
212
|
+
│ ├── pdf/
|
|
213
|
+
│ └── utils/
|
|
214
|
+
├── tests/ # pytest suite
|
|
215
|
+
├── pyproject.toml
|
|
216
|
+
├── requirements.txt
|
|
217
|
+
├── README.md
|
|
218
|
+
├── CONTRIBUTING.md
|
|
219
|
+
├── SECURITY.md
|
|
220
|
+
└── LICENSE.md
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
## Contributing
|
|
224
|
+
|
|
225
|
+
Bug reports, fixes, and feature discussions are welcome: see
|
|
226
|
+
[CONTRIBUTING.md](CONTRIBUTING.md) for how to set up a development
|
|
227
|
+
environment and what to include in a pull request. There is a `tests/`
|
|
228
|
+
directory (now including `utils/validators.py` and the `--dry-run` flag, with
|
|
229
|
+
cross-platform coverage for Tesseract discovery on Windows/macOS/Linux) and it runs with `pytest`, but no CI yet; see [Roadmap & Limitations](https://Hyacinthe-primus.github.io/scanlayer/roadmap.html) and the [Adding tests](CONTRIBUTING.md#adding-tests) section of the contributing guide for where coverage is thinnest.
|
|
230
|
+
|
|
231
|
+
## License
|
|
232
|
+
|
|
233
|
+
[MIT](LICENSE.md).
|
scanlayer-1.0.0/PKG-INFO
DELETED
|
@@ -1,17 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: scanlayer
|
|
3
|
-
Version: 1.0.0
|
|
4
|
-
Summary: Image (scan/photo) to searchable PDF, or raw OCR export, via Tesseract.
|
|
5
|
-
Requires-Python: >=3.9
|
|
6
|
-
License-File: LICENSE.md
|
|
7
|
-
Requires-Dist: pytesseract>=0.3.10
|
|
8
|
-
Requires-Dist: Pillow>=10.0.0
|
|
9
|
-
Requires-Dist: reportlab>=4.0.0
|
|
10
|
-
Requires-Dist: opencv-python-headless>=4.8.0
|
|
11
|
-
Requires-Dist: numpy>=1.24.0
|
|
12
|
-
Requires-Dist: pdf2image>=1.17.0
|
|
13
|
-
Provides-Extra: test
|
|
14
|
-
Requires-Dist: pytest>=7.0; extra == "test"
|
|
15
|
-
Provides-Extra: yaml
|
|
16
|
-
Requires-Dist: PyYAML>=6.0; extra == "yaml"
|
|
17
|
-
Dynamic: license-file
|
|
@@ -1,17 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: scanlayer
|
|
3
|
-
Version: 1.0.0
|
|
4
|
-
Summary: Image (scan/photo) to searchable PDF, or raw OCR export, via Tesseract.
|
|
5
|
-
Requires-Python: >=3.9
|
|
6
|
-
License-File: LICENSE.md
|
|
7
|
-
Requires-Dist: pytesseract>=0.3.10
|
|
8
|
-
Requires-Dist: Pillow>=10.0.0
|
|
9
|
-
Requires-Dist: reportlab>=4.0.0
|
|
10
|
-
Requires-Dist: opencv-python-headless>=4.8.0
|
|
11
|
-
Requires-Dist: numpy>=1.24.0
|
|
12
|
-
Requires-Dist: pdf2image>=1.17.0
|
|
13
|
-
Provides-Extra: test
|
|
14
|
-
Requires-Dist: pytest>=7.0; extra == "test"
|
|
15
|
-
Provides-Extra: yaml
|
|
16
|
-
Requires-Dist: PyYAML>=6.0; extra == "yaml"
|
|
17
|
-
Dynamic: license-file
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|