parisaocr 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parisaocr/__init__.py +27 -0
- parisaocr/__main__.py +3 -0
- parisaocr/api.py +54 -0
- parisaocr/bidi.py +37 -0
- parisaocr/cli.py +256 -0
- parisaocr/cut.py +113 -0
- parisaocr/detect.py +87 -0
- parisaocr/detectors.py +117 -0
- parisaocr/fa_text.py +72 -0
- parisaocr/fonts/OFL-Vazirmatn.txt +93 -0
- parisaocr/fonts/Vazirmatn-Regular.ttf +0 -0
- parisaocr/kraken_reader.py +195 -0
- parisaocr/models/parisaocr-fa-0.1.safetensors +0 -0
- parisaocr/models/ppocrv6-det-small.onnx +0 -0
- parisaocr/order.py +29 -0
- parisaocr/output.py +59 -0
- parisaocr/pages.py +211 -0
- parisaocr/pdfout.py +163 -0
- parisaocr/pipeline.py +206 -0
- parisaocr/reader.py +62 -0
- parisaocr-0.2.0.dist-info/METADATA +223 -0
- parisaocr-0.2.0.dist-info/RECORD +27 -0
- parisaocr-0.2.0.dist-info/WHEEL +5 -0
- parisaocr-0.2.0.dist-info/entry_points.txt +2 -0
- parisaocr-0.2.0.dist-info/licenses/LICENSE +73 -0
- parisaocr-0.2.0.dist-info/licenses/NOTICE +23 -0
- parisaocr-0.2.0.dist-info/top_level.txt +1 -0
parisaocr/__init__.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""ParisaOCR: open-source OCR for printed Persian.
|
|
2
|
+
|
|
3
|
+
A text-line detector (PP-OCRv6, run with ONNX Runtime) finds the lines of a
|
|
4
|
+
page; a Kraken recognition model trained for Persian print reads each line; the
|
|
5
|
+
lines are ordered into right-to-left columns and written as text, hOCR (with
|
|
6
|
+
line and word boxes on the original page) or JSONL. Both models ship with the
|
|
7
|
+
package, so it runs offline, on a CPU or a GPU.
|
|
8
|
+
|
|
9
|
+
parisaocr ocr page.png # print the text
|
|
10
|
+
parisaocr ocr book.pdf --out out # out/txt, out/hocr
|
|
11
|
+
parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf, searchable
|
|
12
|
+
|
|
13
|
+
from parisaocr import OCR
|
|
14
|
+
print(OCR().text("page.png"))
|
|
15
|
+
|
|
16
|
+
The recognition model and the detector are replaceable (`--model`, `--detector`):
|
|
17
|
+
Tesseract models and the Surya detector are supported for comparison, and
|
|
18
|
+
`parisaocr cut` cuts the lines of scanned books for building training data.
|
|
19
|
+
"""
|
|
20
|
+
__version__ = "0.2.0"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def __getattr__(name): # the interface imports torch and Kraken; only load them when asked for
|
|
24
|
+
if name == "OCR":
|
|
25
|
+
from .api import OCR
|
|
26
|
+
return OCR
|
|
27
|
+
raise AttributeError(name)
|
parisaocr/__main__.py
ADDED
parisaocr/api.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Python interface: read a page image and get its text or its lines.
|
|
2
|
+
|
|
3
|
+
from parisaocr import OCR
|
|
4
|
+
ocr = OCR() # loads the detector and the recognition model once
|
|
5
|
+
print(ocr.text("page.png")) # the page's text, lines in reading order
|
|
6
|
+
for line in ocr.lines("page.png"):
|
|
7
|
+
print(line.bbox, line.conf, line.text)
|
|
8
|
+
|
|
9
|
+
`OCR(model=..., detector=..., device=...)` takes the same model and detector
|
|
10
|
+
specifications as the command line (`parisaocr ocr --help`).
|
|
11
|
+
"""
|
|
12
|
+
import pathlib
|
|
13
|
+
import tempfile
|
|
14
|
+
from types import SimpleNamespace
|
|
15
|
+
|
|
16
|
+
from PIL import Image
|
|
17
|
+
|
|
18
|
+
from . import output
|
|
19
|
+
from .cli import build_engine
|
|
20
|
+
from .order import order
|
|
21
|
+
from .pipeline import recognize_batch
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class OCR:
|
|
25
|
+
def __init__(self, model="default", detector="ppocr:v6-small:1.3", cpu=False, line_order="rtl"):
|
|
26
|
+
opts = SimpleNamespace(model=model, detector=detector, cpu=cpu, min_width=1600, max_width=2000, batch=1,
|
|
27
|
+
pad=6, pad_frac=0.4, min_height=40, block_tall=True, fallback=True, jobs=8)
|
|
28
|
+
self.detector, self.reader, self.options = build_engine(opts)
|
|
29
|
+
self.line_order = line_order
|
|
30
|
+
|
|
31
|
+
def columns(self, image):
|
|
32
|
+
"""Lines of one page (a path or a PIL image) as reading-order columns of `pipeline.Line`."""
|
|
33
|
+
with tempfile.TemporaryDirectory(prefix="parisaocr-") as tmp:
|
|
34
|
+
tmp = pathlib.Path(tmp)
|
|
35
|
+
if not isinstance(image, (str, pathlib.Path)):
|
|
36
|
+
path = tmp / "page.png"
|
|
37
|
+
image.save(path)
|
|
38
|
+
else:
|
|
39
|
+
path = pathlib.Path(image)
|
|
40
|
+
with Image.open(path) as page:
|
|
41
|
+
im, f = self.detector.prepare(page)
|
|
42
|
+
rec = {"page": "page", "image": str(path), "width": im.width, "height": im.height, "scale": f,
|
|
43
|
+
"orig_width": page.width, "orig_height": page.height}
|
|
44
|
+
rec["lines"] = self.detector.detect([im])[0]
|
|
45
|
+
lines = recognize_batch([rec], self.reader, self.options, tmp)["page"]
|
|
46
|
+
return order(lines, self.line_order)
|
|
47
|
+
|
|
48
|
+
def lines(self, image):
|
|
49
|
+
"""The lines of one page in reading order (text, bbox on the page, mean confidence, words)."""
|
|
50
|
+
return [line for col in self.columns(image) for line in col]
|
|
51
|
+
|
|
52
|
+
def text(self, image):
|
|
53
|
+
"""The text of one page: one line per printed line, columns right to left."""
|
|
54
|
+
return output.text(self.columns(image))
|
parisaocr/bidi.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Reading order <-> display order for one text line, with each line's own base direction.
|
|
2
|
+
|
|
3
|
+
Recognition models see a line image from left to right, so they are trained on
|
|
4
|
+
the display order of the text and their output is turned back into reading
|
|
5
|
+
order. Kraken can do both itself, but (mittagessen/kraken#809) its bidi code
|
|
6
|
+
deletes ZWNJ, the Persian half-space, and it gives every line the same base
|
|
7
|
+
direction, so a Latin-only line such as "1. Vercingétorix" is trained as
|
|
8
|
+
"Vercingétorix .1". Here the Unicode bidi algorithm of kraken.lib.bidi runs
|
|
9
|
+
with ZWNJ protected (swapped for U+00A6, a neutral that stays in place between
|
|
10
|
+
two letters) and with the base direction taken from the line itself.
|
|
11
|
+
"""
|
|
12
|
+
import unicodedata
|
|
13
|
+
|
|
14
|
+
ZWNJ, PLACEHOLDER = "", "¦"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def base_direction(text):
|
|
18
|
+
""""L" for a line with more Latin than Arabic-script letters, else "R"."""
|
|
19
|
+
latin = sum(1 for c in text if c.isalpha() and unicodedata.bidirectional(c) == "L")
|
|
20
|
+
arabic = sum(1 for c in text if unicodedata.bidirectional(c) in ("R", "AL"))
|
|
21
|
+
return "L" if latin > arabic else "R"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def to_display(text, base=None):
|
|
25
|
+
"""Display (left-to-right visual) order of a line in reading order; ZWNJ kept."""
|
|
26
|
+
from kraken.lib.bidi import get_display
|
|
27
|
+
base = base or base_direction(text)
|
|
28
|
+
return get_display(text.replace(ZWNJ, PLACEHOLDER), base_dir=base).replace(PLACEHOLDER, ZWNJ)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def to_logical(display, base=None):
|
|
32
|
+
"""Reading order of a line in display order, and for each character of the result the index
|
|
33
|
+
of the display character it came from (to carry character positions along)."""
|
|
34
|
+
from kraken.lib.bidi import get_display_map
|
|
35
|
+
base = base or base_direction(display)
|
|
36
|
+
text, order = get_display_map(display.replace(ZWNJ, PLACEHOLDER), base_dir=base)
|
|
37
|
+
return text.replace(PLACEHOLDER, ZWNJ), order
|
parisaocr/cli.py
ADDED
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
"""Command line: `parisaocr ocr` and `parisaocr cut`."""
|
|
2
|
+
import argparse
|
|
3
|
+
import os
|
|
4
|
+
import pathlib
|
|
5
|
+
import shlex
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
import time
|
|
9
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
10
|
+
|
|
11
|
+
from . import __version__, output
|
|
12
|
+
from .detect import detect_pages
|
|
13
|
+
from .detectors import make_detector
|
|
14
|
+
from .order import order
|
|
15
|
+
from .pages import collect
|
|
16
|
+
from .pipeline import Options, recognize_batch
|
|
17
|
+
from .reader import TesseractReader
|
|
18
|
+
|
|
19
|
+
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
|
20
|
+
BUNDLED_MODEL = pathlib.Path(__file__).resolve().parent / "models" / "parisaocr-fa-0.1.safetensors"
|
|
21
|
+
DEFAULT_TESSDATA = os.environ.get("PARISAOCR_TESSDATA", str(ROOT / "dist/tessdata_contrib/fas_print/best"))
|
|
22
|
+
DEFAULT_LANG = os.environ.get("PARISAOCR_LANG", "fas_print")
|
|
23
|
+
SMALL_TEXT_PX = 14 # median line height (original page pixels) below which accuracy drops clearly
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def engine_args(p, default_model="default"):
|
|
27
|
+
g = p.add_argument_group("engine")
|
|
28
|
+
g.add_argument("--model", default=default_model,
|
|
29
|
+
help="recognition model: default (the ParisaOCR model shipped with the package), "
|
|
30
|
+
"kraken:MODEL.safetensors (another Kraken model), or TESSDATA_DIR:LANG[:EXTRA ARGS] for Tesseract "
|
|
31
|
+
f"(the cut command defaults to the fas_print Tesseract model, {DEFAULT_TESSDATA}:{DEFAULT_LANG})")
|
|
32
|
+
g.add_argument("--detector", default="ppocr:v6-small:1.3",
|
|
33
|
+
help="line detector: ppocr[:VERSION-SIZE[:UNCLIP]] (PaddleOCR weights via RapidOCR on ONNX Runtime, "
|
|
34
|
+
"Apache-2.0; default ppocr:v6-small:1.3), surya (needs the surya extra; its weights are free "
|
|
35
|
+
"only for research, personal use and small companies), kraken (Kraken's default segmenter)")
|
|
36
|
+
g.add_argument("--min-width", type=int, default=1600, help="upscale narrower pages to this width before detection")
|
|
37
|
+
g.add_argument("--max-width", type=int, default=2000, help="reduce wider pages to this width for detection only (0 = never)")
|
|
38
|
+
g.add_argument("--batch", type=int, default=8, help="pages per detector batch")
|
|
39
|
+
g.add_argument("--cpu", action="store_true", help="run the detector on the CPU")
|
|
40
|
+
g.add_argument("--pad", type=int, default=6, help="least margin around a line crop, in working pixels (vertical: half)")
|
|
41
|
+
g.add_argument("--pad-frac", type=float, default=0.4,
|
|
42
|
+
help="margin as a fraction of the page's median line height; the larger of --pad and this is used")
|
|
43
|
+
g.add_argument("--min-height", type=int, default=40, help="enlarge crops shorter than this many pixels")
|
|
44
|
+
g.add_argument("--no-block-tall", dest="block_tall", action="store_false", help="read tall boxes in raw-line mode too")
|
|
45
|
+
g.add_argument("--no-fallback", dest="fallback", action="store_false", help="no single-line retry of near-empty results")
|
|
46
|
+
g.add_argument("--jobs", type=int, default=os.cpu_count(), help="parallel Tesseract processes")
|
|
47
|
+
g.add_argument("--redo", action="store_true", help="ignore cached detection and existing output")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def build_engine(opts):
|
|
51
|
+
if opts.model == "default" or opts.model.startswith("kraken:"):
|
|
52
|
+
import torch
|
|
53
|
+
from .kraken_reader import KrakenReader
|
|
54
|
+
path = BUNDLED_MODEL if opts.model == "default" else opts.model.split(":", 1)[1]
|
|
55
|
+
reader = KrakenReader(path, "cuda:0" if torch.cuda.is_available() and not opts.cpu else "cpu")
|
|
56
|
+
fallback, short_conf = False, 80.0 # the single-line retry is a Tesseract remedy
|
|
57
|
+
else:
|
|
58
|
+
tessdata, lang, *extra = opts.model.split(":", 2)
|
|
59
|
+
reader = TesseractReader(tessdata, lang, shlex.split(extra[0]) if extra else [])
|
|
60
|
+
fallback, short_conf = opts.fallback, 0.0
|
|
61
|
+
detector = make_detector(opts.detector, opts.min_width, opts.max_width, opts.batch, opts.cpu)
|
|
62
|
+
options = Options(opts.pad, opts.pad_frac, opts.min_height, opts.block_tall, fallback, opts.jobs, short_conf)
|
|
63
|
+
return detector, reader, options
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def small_text_warning(rec):
|
|
67
|
+
"""A note on stderr when a page's text is so small that accuracy drops (median line height in page pixels)."""
|
|
68
|
+
heights = sorted((l["bbox"][3] - l["bbox"][1]) / rec.get("scale", 1.0) for l in rec["lines"])
|
|
69
|
+
if heights and heights[len(heights) // 2] < SMALL_TEXT_PX:
|
|
70
|
+
print(f"parisaocr: {rec['page']}: text lines are only ~{heights[len(heights) // 2]:.0f} px high; "
|
|
71
|
+
"accuracy drops on such small text (a 300 dpi scan of the page helps)", file=sys.stderr, flush=True)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def cmd_ocr(opts):
|
|
75
|
+
if opts.out is None: # no output directory: read into a temporary one and print the text
|
|
76
|
+
with tempfile.TemporaryDirectory(prefix="parisaocr-out-") as tmp:
|
|
77
|
+
opts.out, opts.format, opts.quiet = tmp, "txt", True
|
|
78
|
+
pages = cmd_ocr(opts)
|
|
79
|
+
for i, (pid, _) in enumerate(pages):
|
|
80
|
+
if len(pages) > 1:
|
|
81
|
+
print(("\n" if i else "") + f"# {pid}")
|
|
82
|
+
sys.stdout.write((pathlib.Path(tmp) / "txt" / f"{pid}.txt").read_text(encoding="utf-8"))
|
|
83
|
+
return
|
|
84
|
+
log = (lambda *a, **k: None) if getattr(opts, "quiet", False) else (lambda *a, **k: print(*a, **k, flush=True))
|
|
85
|
+
out = pathlib.Path(opts.out)
|
|
86
|
+
formats = [f.strip() for f in opts.format.split(",") if f.strip()]
|
|
87
|
+
bad = set(formats) - {"txt", "hocr", "jsonl", "pdf"}
|
|
88
|
+
if bad:
|
|
89
|
+
sys.exit(f"parisaocr: unknown format {', '.join(bad)}")
|
|
90
|
+
pages = collect(opts.input, out, opts.pdf, opts.dpi, opts.first, opts.last, opts.redo)
|
|
91
|
+
if not pages:
|
|
92
|
+
sys.exit("parisaocr: no pages to read")
|
|
93
|
+
want_pdf = "pdf" in formats
|
|
94
|
+
formats = [f for f in formats if f != "pdf"]
|
|
95
|
+
if want_pdf and "jsonl" not in formats:
|
|
96
|
+
formats.append("jsonl") # the searchable PDF is built from the per-page lines and boxes
|
|
97
|
+
for f in formats:
|
|
98
|
+
(out / f).mkdir(parents=True, exist_ok=True)
|
|
99
|
+
todo = [(pid, p) for pid, p in pages if opts.redo or not all((out / f / f"{pid}.{f}").exists() for f in formats)]
|
|
100
|
+
log(f"{len(pages)} pages, {len(todo)} to read -> {out}/{{{','.join(formats + ['pdf'] * want_pdf)}}}")
|
|
101
|
+
if not todo:
|
|
102
|
+
if want_pdf:
|
|
103
|
+
write_pdfs(opts, out, pages, log)
|
|
104
|
+
return pages
|
|
105
|
+
detector, reader, options = build_engine(opts)
|
|
106
|
+
started, n_lines, done = time.time(), 0, 0
|
|
107
|
+
|
|
108
|
+
def finish(records, lines_of):
|
|
109
|
+
n = 0
|
|
110
|
+
for rec in records:
|
|
111
|
+
pid = rec["page"]
|
|
112
|
+
small_text_warning(rec)
|
|
113
|
+
cols = order(lines_of[pid], opts.order)
|
|
114
|
+
n += sum(len(c) for c in cols)
|
|
115
|
+
if "txt" in formats:
|
|
116
|
+
(out / "txt" / f"{pid}.txt").write_text(output.text(cols), encoding="utf-8")
|
|
117
|
+
if "hocr" in formats:
|
|
118
|
+
(out / "hocr" / f"{pid}.hocr").write_text(
|
|
119
|
+
output.hocr(pid, pathlib.Path(rec["image"]).name, rec.get("orig_width", rec["width"]),
|
|
120
|
+
rec.get("orig_height", rec["height"]), cols, reader.describe()), encoding="utf-8")
|
|
121
|
+
if "jsonl" in formats:
|
|
122
|
+
(out / "jsonl" / f"{pid}.jsonl").write_text(output.jsonl(pid, cols), encoding="utf-8")
|
|
123
|
+
return n
|
|
124
|
+
|
|
125
|
+
with tempfile.TemporaryDirectory(prefix="parisaocr-") as tmp, ThreadPoolExecutor(1) as cpu:
|
|
126
|
+
tmp = pathlib.Path(tmp)
|
|
127
|
+
pending = None
|
|
128
|
+
# Detection (GPU) of one batch overlaps with recognition (CPU) of the previous one.
|
|
129
|
+
for records in detect_pages(detector, todo, out / "det", opts.redo):
|
|
130
|
+
if pending:
|
|
131
|
+
n_lines += pending.result()
|
|
132
|
+
done += opts.batch
|
|
133
|
+
log(f" {min(done, len(todo))}/{len(todo)} pages, {n_lines} lines, {time.time() - started:.0f} s")
|
|
134
|
+
pending = cpu.submit(lambda recs=records: finish(recs, recognize_batch(recs, reader, options, tmp)))
|
|
135
|
+
if pending:
|
|
136
|
+
n_lines += pending.result()
|
|
137
|
+
log(f"{len(todo)} pages, {n_lines} lines in {time.time() - started:.0f} s")
|
|
138
|
+
if want_pdf:
|
|
139
|
+
write_pdfs(opts, out, pages, log)
|
|
140
|
+
return pages
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def write_pdfs(opts, out, pages, log):
|
|
144
|
+
"""Searchable PDFs in OUT/pdf: the input PDF with a text layer, and/or the image inputs as PDFs."""
|
|
145
|
+
from PIL import Image
|
|
146
|
+
from . import pdfout
|
|
147
|
+
(out / "pdf").mkdir(exist_ok=True)
|
|
148
|
+
pdf_inputs = [pathlib.Path(p) for p in opts.input if pathlib.Path(p).suffix.lower() == ".pdf"]
|
|
149
|
+
pages_dir = (out / "pages").resolve()
|
|
150
|
+
from_pdf = [(pid, p) for pid, p in pages if pathlib.Path(p).resolve().parent == pages_dir]
|
|
151
|
+
images = [(pid, p) for pid, p in pages if (pid, p) not in from_pdf]
|
|
152
|
+
|
|
153
|
+
def lines_of(pid):
|
|
154
|
+
f = out / "jsonl" / f"{pid}.jsonl"
|
|
155
|
+
return pdfout.read_jsonl(f) if f.exists() else []
|
|
156
|
+
|
|
157
|
+
if pdf_inputs and from_pdf:
|
|
158
|
+
src = pdf_inputs[0]
|
|
159
|
+
layers = {}
|
|
160
|
+
for pid, image in from_pdf:
|
|
161
|
+
with Image.open(image) as im:
|
|
162
|
+
layers[int(pid.split("-")[1])] = (im.width, im.height, lines_of(pid))
|
|
163
|
+
dst = out / "pdf" / src.name
|
|
164
|
+
done = pdfout.overlay_pdf(src, layers, dst, keep_text=opts.pdf_text == "skip")
|
|
165
|
+
skipped = len(layers) - len(done)
|
|
166
|
+
log(f"searchable PDF -> {dst} (text layer on {len(done)} pages"
|
|
167
|
+
+ (f"; {skipped} already had text and were left as they are (--pdf-text add to add ours)" if skipped else "")
|
|
168
|
+
+ ")")
|
|
169
|
+
if images:
|
|
170
|
+
if opts.merge_pdf:
|
|
171
|
+
dst = out / "pdf" / (opts.merge_pdf if opts.merge_pdf.lower().endswith(".pdf") else opts.merge_pdf + ".pdf")
|
|
172
|
+
pdfout.images_to_pdf([(p, lines_of(pid)) for pid, p in images], dst)
|
|
173
|
+
log(f"searchable PDF -> {dst} ({len(images)} pages)")
|
|
174
|
+
else:
|
|
175
|
+
for pid, p in images:
|
|
176
|
+
pdfout.images_to_pdf([(p, lines_of(pid))], out / "pdf" / f"{pid}.pdf")
|
|
177
|
+
log(f"searchable PDFs -> {out / 'pdf'} ({len(images)} files)")
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def cmd_pages(opts):
|
|
181
|
+
"""Only turn a PDF into page images (OUT/p-NNN.png), e.g. to prepare a book for `parisaocr cut`."""
|
|
182
|
+
from .pages import pdf_pages
|
|
183
|
+
out = pathlib.Path(opts.out)
|
|
184
|
+
got = pdf_pages(pathlib.Path(opts.pdf), out, opts.pdf_mode, opts.dpi, opts.first, opts.last, opts.redo)
|
|
185
|
+
print(f"{len(got)} pages in {out}")
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def cmd_cut(opts):
|
|
189
|
+
from .cut import cut
|
|
190
|
+
detector, reader, options = build_engine(opts)
|
|
191
|
+
with tempfile.TemporaryDirectory(prefix="parisaocr-") as tmp:
|
|
192
|
+
cut(opts.book_dir, opts.out or pathlib.Path(opts.book_dir) / "lines", detector, reader, options,
|
|
193
|
+
test=opts.test, train=opts.train, exclude=opts.exclude, seed=opts.seed, min_lines=opts.min_lines,
|
|
194
|
+
min_page_conf=opts.min_page_conf, min_conf=opts.min_conf, pad=opts.crop_pad, redo=opts.redo, tmp=pathlib.Path(tmp),
|
|
195
|
+
test_pages=opts.test_pages, train_pages=opts.train_pages)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def main(argv=None):
|
|
199
|
+
ap = argparse.ArgumentParser(prog="parisaocr", description=__doc__)
|
|
200
|
+
ap.add_argument("--version", action="version", version=f"parisaocr {__version__}")
|
|
201
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
202
|
+
|
|
203
|
+
o = sub.add_parser("ocr", help="read pages (images, a directory of images, or one PDF)",
|
|
204
|
+
description="Writes OUT/txt/PAGE.txt, OUT/hocr/PAGE.hocr and/or OUT/jsonl/PAGE.jsonl; "
|
|
205
|
+
"PDF pages are rendered to OUT/pages first; detection is cached in OUT/det.")
|
|
206
|
+
o.add_argument("input", nargs="+")
|
|
207
|
+
o.add_argument("--out", help="output directory; without it the text is printed")
|
|
208
|
+
o.add_argument("--format", default="txt,hocr",
|
|
209
|
+
help="comma-separated: txt, hocr, jsonl, pdf (with --out). pdf: searchable PDF, the input PDF with "
|
|
210
|
+
"an invisible text layer, or one PDF per input image (see --merge-pdf)")
|
|
211
|
+
o.add_argument("--merge-pdf", metavar="NAME", help="with --format pdf: put all input images into one PDF, OUT/pdf/NAME.pdf")
|
|
212
|
+
o.add_argument("--pdf-text", choices=["skip", "add"], default="skip",
|
|
213
|
+
help="with --format pdf and a PDF input: leave pages that already have a text layer alone (skip), "
|
|
214
|
+
"or add ours to them too (add; e.g. over a poor earlier OCR layer)")
|
|
215
|
+
o.add_argument("--order", choices=["rtl", "raster"], default="rtl", help="line order: right-to-left columns, or top to bottom")
|
|
216
|
+
o.add_argument("--pdf", choices=["auto", "extract", "render"], default="auto",
|
|
217
|
+
help="PDF pages: extract the embedded scan images, render at --dpi, or decide per file")
|
|
218
|
+
o.add_argument("--dpi", type=int, default=300)
|
|
219
|
+
o.add_argument("--first", type=int, help="first PDF page")
|
|
220
|
+
o.add_argument("--last", type=int, help="last PDF page")
|
|
221
|
+
engine_args(o)
|
|
222
|
+
o.set_defaults(func=cmd_ocr)
|
|
223
|
+
|
|
224
|
+
p = sub.add_parser("pages", help="extract or render a PDF's pages to OUT/p-NNN.png (no OCR)")
|
|
225
|
+
p.add_argument("pdf")
|
|
226
|
+
p.add_argument("--out", required=True)
|
|
227
|
+
p.add_argument("--pdf-mode", choices=["auto", "extract", "render"], default="auto")
|
|
228
|
+
p.add_argument("--dpi", type=int, default=300)
|
|
229
|
+
p.add_argument("--first", type=int)
|
|
230
|
+
p.add_argument("--last", type=int)
|
|
231
|
+
p.add_argument("--redo", action="store_true")
|
|
232
|
+
p.set_defaults(func=cmd_pages)
|
|
233
|
+
|
|
234
|
+
c = sub.add_parser("cut", help="cut the lines of a book (BOOK_DIR/p-NNN.png) for labelling",
|
|
235
|
+
description="Writes LINES/test|train/p-NNN_LL.png crops and LINES/manifest.json (default LINES = BOOK_DIR/lines).")
|
|
236
|
+
c.add_argument("book_dir")
|
|
237
|
+
c.add_argument("--out", help="lines directory (default BOOK_DIR/lines)")
|
|
238
|
+
c.add_argument("--test", type=int, default=12, help="test pages")
|
|
239
|
+
c.add_argument("--train", type=int, default=24, help="train pages")
|
|
240
|
+
c.add_argument("--exclude", default="", help="page ranges to leave out, e.g. 1-11,93-97")
|
|
241
|
+
c.add_argument("--test-pages", default="", help="use exactly these pages as test, e.g. 30,45 (no random choice, no filters)")
|
|
242
|
+
c.add_argument("--train-pages", default="", help="use exactly these pages as train, e.g. 18,50-52")
|
|
243
|
+
c.add_argument("--seed", type=int, default=0)
|
|
244
|
+
c.add_argument("--min-lines", type=int, default=15, help="pages with fewer detected lines are not candidates")
|
|
245
|
+
c.add_argument("--min-page-conf", type=float, default=80, help="skip pages whose mean word confidence is lower (0 = keep all)")
|
|
246
|
+
c.add_argument("--min-conf", type=float, default=60, help="draft words below this confidence are listed for checking")
|
|
247
|
+
c.add_argument("--crop-pad", type=int, default=10, help="margin of the saved line crops, in page pixels (vertical: half)")
|
|
248
|
+
engine_args(c, default_model=f"{DEFAULT_TESSDATA}:{DEFAULT_LANG}") # page filters are calibrated on Tesseract confidences
|
|
249
|
+
c.set_defaults(func=cmd_cut)
|
|
250
|
+
|
|
251
|
+
opts = ap.parse_args(argv)
|
|
252
|
+
opts.func(opts)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
if __name__ == "__main__":
|
|
256
|
+
main()
|
parisaocr/cut.py
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Cut the text lines of a scanned book for labelling (replaces `correct_lines.py cut`).
|
|
2
|
+
|
|
3
|
+
Pages are BOOK_DIR/p-NNN.png. All pages are detected (cached in BOOK_DIR/det);
|
|
4
|
+
pages with too few lines or in --exclude are dropped, the rest are shuffled
|
|
5
|
+
(--seed) and read one batch at a time until --test and --train pages are
|
|
6
|
+
accepted; a page whose mean word confidence is below --min-page-conf (tables,
|
|
7
|
+
maps, foreign text) is skipped. With --test-pages/--train-pages the given
|
|
8
|
+
pages are used as they are. Every detected line of an accepted page is
|
|
9
|
+
cropped from the original page at native resolution with a small margin into
|
|
10
|
+
LINES/SPLIT/p-NNN_LL.png (LL = line index in detection order) and described in
|
|
11
|
+
LINES/manifest.json with the draft reading and its low-confidence words, in the
|
|
12
|
+
format scripts/correct_lines.py serve, line_sheets.py and dataset.py read.
|
|
13
|
+
"""
|
|
14
|
+
import json
|
|
15
|
+
import pathlib
|
|
16
|
+
import random
|
|
17
|
+
import statistics
|
|
18
|
+
import sys
|
|
19
|
+
|
|
20
|
+
from PIL import Image
|
|
21
|
+
|
|
22
|
+
from .detect import detect_pages
|
|
23
|
+
from .pipeline import expand, line_crop, margins, recognize_batch
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def parse_ranges(spec):
|
|
27
|
+
pages = set()
|
|
28
|
+
for part in filter(None, spec.split(",")):
|
|
29
|
+
a, _, b = part.partition("-")
|
|
30
|
+
pages.update(range(int(a), int(b or a) + 1))
|
|
31
|
+
return pages
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def cut(book_dir, out, detector, reader, opts, test=12, train=24, exclude="", seed=0,
|
|
35
|
+
min_lines=15, min_page_conf=80.0, min_conf=60.0, pad=10, redo=False, tmp=None,
|
|
36
|
+
test_pages="", train_pages=""):
|
|
37
|
+
book_dir, out = pathlib.Path(book_dir), pathlib.Path(out)
|
|
38
|
+
pages = [(p.stem, p) for p in sorted(book_dir.glob("p-*.png"))]
|
|
39
|
+
if not pages:
|
|
40
|
+
sys.exit(f"parisaocr cut: no p-NNN.png pages in {book_dir}")
|
|
41
|
+
# Explicit pages (--test-pages/--train-pages) are taken as given, without the line-count
|
|
42
|
+
# and confidence filters; otherwise pages are drawn at random until the quotas are met.
|
|
43
|
+
explicit = {f"p-{p:03d}": split for split, spec in (("test", test_pages), ("train", train_pages)) for p in parse_ranges(spec)}
|
|
44
|
+
missing = [p for p in explicit if p not in dict(pages)]
|
|
45
|
+
if missing:
|
|
46
|
+
sys.exit(f"parisaocr cut: no such pages in {book_dir}: {' '.join(missing)}")
|
|
47
|
+
if explicit:
|
|
48
|
+
pages = [(pid, path) for pid, path in pages if pid in explicit]
|
|
49
|
+
records = {}
|
|
50
|
+
for batch in detect_pages(detector, pages, book_dir / "det", redo):
|
|
51
|
+
records.update({r["page"]: r for r in batch})
|
|
52
|
+
print(f"detected {len(records)}/{len(pages)} pages", end="\r", flush=True)
|
|
53
|
+
print()
|
|
54
|
+
if explicit:
|
|
55
|
+
order = [pid for pid, _ in pages]
|
|
56
|
+
print(f"{len(order)} pages given explicitly")
|
|
57
|
+
else:
|
|
58
|
+
excluded = parse_ranges(exclude)
|
|
59
|
+
order = [pid for pid, _ in pages if len(records[pid]["lines"]) >= min_lines and int(pid.split("-")[1]) not in excluded]
|
|
60
|
+
random.Random(seed).shuffle(order)
|
|
61
|
+
print(f"{len(order)} of {len(pages)} pages have at least {min_lines} lines and are not excluded")
|
|
62
|
+
|
|
63
|
+
manifest, chosen, skipped = [], {}, []
|
|
64
|
+
quota = {"test": test, "train": train}
|
|
65
|
+
i = 0
|
|
66
|
+
while i < len(order) and (explicit or sum(quota.values()) > 0):
|
|
67
|
+
batch = order[i:i + detector.batch]
|
|
68
|
+
i += len(batch)
|
|
69
|
+
lines_of = recognize_batch([records[p] for p in batch], reader, opts, tmp)
|
|
70
|
+
for pid in batch:
|
|
71
|
+
lines = lines_of[pid]
|
|
72
|
+
if explicit:
|
|
73
|
+
split = explicit[pid]
|
|
74
|
+
else:
|
|
75
|
+
if sum(quota.values()) == 0:
|
|
76
|
+
break # the rest of the batch was read for nothing; a batch is at most 8 pages
|
|
77
|
+
words = [w for l in lines for w in l.words]
|
|
78
|
+
mean = sum(w.conf for w in words) / len(words) if words else 0.0
|
|
79
|
+
if mean < min_page_conf:
|
|
80
|
+
skipped.append((pid, round(mean)))
|
|
81
|
+
continue
|
|
82
|
+
split = "test" if quota["test"] > 0 else "train"
|
|
83
|
+
quota[split] -= 1
|
|
84
|
+
chosen[pid] = split
|
|
85
|
+
(out / split).mkdir(parents=True, exist_ok=True)
|
|
86
|
+
with Image.open(records[pid]["image"]) as page:
|
|
87
|
+
im = page.convert("L") if page.mode not in ("L", "1") else page.copy()
|
|
88
|
+
# The saved crop gets the same margin rule as the recognizer's crop, in page pixels,
|
|
89
|
+
# so training lines look like what the model sees at inference.
|
|
90
|
+
heights = [l.bbox[3] - l.bbox[1] for l in lines]
|
|
91
|
+
px, py = margins(pad, opts.pad_frac, statistics.median(heights) if heights else 0)
|
|
92
|
+
boxes = [l.bbox for l in lines]
|
|
93
|
+
rects = expand(boxes, px, py, im.width, im.height)
|
|
94
|
+
if im.mode == "1":
|
|
95
|
+
im = im.convert("L") # so neighbours can be painted over
|
|
96
|
+
for k, (line, rect) in enumerate(zip(lines, rects)): # detection order; a block box read as several lines gives several names
|
|
97
|
+
name = f"{pid}_{k:02d}"
|
|
98
|
+
line_crop(im, rect, boxes[k], boxes).save(out / split / f"{name}.png")
|
|
99
|
+
manifest.append({"name": name, "split": split, "page": pid, "draft": line.text,
|
|
100
|
+
"check": [[w.text, None] for w in line.words if w.conf < min_conf],
|
|
101
|
+
"bbox": list(line.bbox), "conf": line.conf})
|
|
102
|
+
manifest.sort(key=lambda m: (m["split"] != "test", m["name"]))
|
|
103
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
104
|
+
(out / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=0), encoding="utf-8")
|
|
105
|
+
for split in ("test", "train"):
|
|
106
|
+
ps = sorted(p for p, s in chosen.items() if s == split)
|
|
107
|
+
print(f"{split}: {len(ps)} pages, {sum(m['split'] == split for m in manifest)} lines: " + " ".join(p[2:] for p in ps))
|
|
108
|
+
if skipped:
|
|
109
|
+
print(f"skipped {len(skipped)} pages below {min_page_conf:.0f} mean confidence: "
|
|
110
|
+
+ " ".join(f"{p[2:]}({c})" for p, c in skipped[:20]) + (" ..." if len(skipped) > 20 else ""))
|
|
111
|
+
if not explicit and sum(quota.values()):
|
|
112
|
+
print(f"short of {quota['test']} test and {quota['train']} train pages", file=sys.stderr)
|
|
113
|
+
return manifest
|
parisaocr/detect.py
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Text-line detection. `LineDetector` wraps Surya; replace it to try another detector.
|
|
2
|
+
|
|
3
|
+
A detection record (cached as JSON) holds the page image path, the size of the
|
|
4
|
+
image the detector saw (`width`, `height`), the factor it was scaled by
|
|
5
|
+
(`scale`, 1.0 when the page was left alone) and the line boxes in those
|
|
6
|
+
detection coordinates. Small pages are upscaled to `min_width` first: the
|
|
7
|
+
detector misses lines on low-resolution scans otherwise; pages wider than
|
|
8
|
+
`max_width` are reduced for detection only.
|
|
9
|
+
"""
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
import pathlib
|
|
13
|
+
|
|
14
|
+
from PIL import Image
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class LineDetector:
|
|
18
|
+
def __init__(self, min_width=1600, max_width=2000, batch=8, cpu=False):
|
|
19
|
+
self.min_width, self.max_width, self.batch, self.cpu = min_width, max_width, batch, cpu
|
|
20
|
+
if cpu:
|
|
21
|
+
os.environ["TORCH_DEVICE"] = "cpu"
|
|
22
|
+
self._model = None
|
|
23
|
+
|
|
24
|
+
def model(self):
|
|
25
|
+
if self._model is None:
|
|
26
|
+
from surya.detection import DetectionPredictor # slow import, only when detection is needed
|
|
27
|
+
# local(): this process owns the model. The default constructor talks to a shared
|
|
28
|
+
# server subprocess that outlives the client (it was found still holding GPU memory
|
|
29
|
+
# minutes after a run) and that crashed on batches of large pages.
|
|
30
|
+
self._model = DetectionPredictor.local()
|
|
31
|
+
return self._model
|
|
32
|
+
|
|
33
|
+
def prepare(self, image):
|
|
34
|
+
"""The RGB image the detector will see and the factor it was scaled by.
|
|
35
|
+
|
|
36
|
+
Small pages are enlarged (the detector misses lines on them), very large
|
|
37
|
+
scans are reduced (the detector works at a fixed internal size anyway,
|
|
38
|
+
and a batch of 400-dpi pages overflows its server).
|
|
39
|
+
"""
|
|
40
|
+
im = image.convert("RGB")
|
|
41
|
+
f = 1.0
|
|
42
|
+
if 0 < im.width < self.min_width:
|
|
43
|
+
f = self.min_width / im.width
|
|
44
|
+
elif self.max_width and im.width > self.max_width:
|
|
45
|
+
f = self.max_width / im.width
|
|
46
|
+
if f != 1.0:
|
|
47
|
+
im = im.resize((round(im.width * f), round(im.height * f)), Image.LANCZOS)
|
|
48
|
+
return im, f
|
|
49
|
+
|
|
50
|
+
def detect(self, images):
|
|
51
|
+
"""Line boxes for each RGB image, as dicts with bbox, polygon, confidence (detection coordinates)."""
|
|
52
|
+
out = []
|
|
53
|
+
for i in range(0, len(images), self.batch):
|
|
54
|
+
for res in self.model()(images[i:i + self.batch]):
|
|
55
|
+
out.append([{"bbox": [round(v) for v in b.bbox],
|
|
56
|
+
"polygon": [[round(x), round(y)] for x, y in b.polygon],
|
|
57
|
+
"confidence": round(float(b.confidence), 3)} for b in res.bboxes])
|
|
58
|
+
return out
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def detect_pages(detector, pages, cache_dir=None, redo=False):
|
|
62
|
+
"""Detection records for (page id, image path) pairs, batch by batch, using CACHE_DIR/PID.json when present."""
|
|
63
|
+
if cache_dir:
|
|
64
|
+
cache_dir = pathlib.Path(cache_dir)
|
|
65
|
+
cache_dir.mkdir(parents=True, exist_ok=True)
|
|
66
|
+
for i in range(0, len(pages), detector.batch):
|
|
67
|
+
chunk, records, todo = pages[i:i + detector.batch], {}, []
|
|
68
|
+
for pid, path in chunk:
|
|
69
|
+
cached = cache_dir / f"{pid}.json" if cache_dir else None
|
|
70
|
+
if cached and cached.exists() and not redo:
|
|
71
|
+
records[pid] = json.loads(cached.read_text(encoding="utf-8"))
|
|
72
|
+
else:
|
|
73
|
+
todo.append((pid, path))
|
|
74
|
+
if todo:
|
|
75
|
+
images, recs = [], []
|
|
76
|
+
for pid, path in todo:
|
|
77
|
+
with Image.open(path) as page:
|
|
78
|
+
im, f = detector.prepare(page)
|
|
79
|
+
recs.append({"page": pid, "image": str(path), "width": im.width, "height": im.height, "scale": f,
|
|
80
|
+
"orig_width": page.width, "orig_height": page.height})
|
|
81
|
+
images.append(im)
|
|
82
|
+
for rec, lines in zip(recs, detector.detect(images)):
|
|
83
|
+
rec["lines"] = lines
|
|
84
|
+
records[rec["page"]] = rec
|
|
85
|
+
if cache_dir:
|
|
86
|
+
(cache_dir / f"{rec['page']}.json").write_text(json.dumps(rec, ensure_ascii=False), encoding="utf-8")
|
|
87
|
+
yield [records[pid] for pid, _ in chunk]
|