ocr-util 2.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ocr_util/__init__.py +3 -0
- ocr_util/cli.py +311 -0
- ocr_util/corpus/__init__.py +17 -0
- ocr_util/corpus/common.py +522 -0
- ocr_util/corpus/generate_corpus.py +149 -0
- ocr_util/corpus/load_metadata.py +325 -0
- ocr_util/corpus/template.corpus.xml +24 -0
- ocr_util/eval/__init__.py +34 -0
- ocr_util/eval/aggregation.py +634 -0
- ocr_util/eval/cli.py +827 -0
- ocr_util/eval/dictionary_metrics/__init__.py +0 -0
- ocr_util/eval/dictionary_metrics/common.py +84 -0
- ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +116 -0
- ocr_util/eval/dictionary_metrics/language_tool/Util.py +91 -0
- ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
- ocr_util/eval/dictionary_metrics/language_tool/common.py +49 -0
- ocr_util/eval/evaluation.py +628 -0
- ocr_util/eval/geometry.py +117 -0
- ocr_util/eval/metrics.py +271 -0
- ocr_util/eval/model/common.py +107 -0
- ocr_util/eval/model/digital_object_model.py +257 -0
- ocr_util/eval/model/digital_object_util.py +120 -0
- ocr_util/eval/model/filter.py +130 -0
- ocr_util/eval/model/format_alto_v3_util.py +247 -0
- ocr_util/eval/model/format_page_util.py +267 -0
- ocr_util/eval/model/main.py +16 -0
- ocr_util/eval/model/minidom_util.py +46 -0
- ocr_util/eval/preprocessing.py +439 -0
- ocr_util/eval/resolve.py +55 -0
- ocr_util/show/cli.py +66 -0
- ocr_util/show/ocr_show_segmentation.py +396 -0
- ocr_util/slice/__init__.py +4 -0
- ocr_util/slice/cli.py +225 -0
- ocr_util/slice/gts_pairs.py +702 -0
- ocr_util/slice/pairs_lstmfs.py +77 -0
- ocr_util-2.0.1.dist-info/METADATA +136 -0
- ocr_util-2.0.1.dist-info/RECORD +41 -0
- ocr_util-2.0.1.dist-info/WHEEL +5 -0
- ocr_util-2.0.1.dist-info/entry_points.txt +2 -0
- ocr_util-2.0.1.dist-info/licenses/LICENSE +21 -0
- ocr_util-2.0.1.dist-info/top_level.txt +1 -0
ocr_util/__init__.py
ADDED
ocr_util/cli.py
ADDED
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""OCR Utils"""
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
from pathlib import Path, PurePath
|
|
9
|
+
|
|
10
|
+
import ocr_util
|
|
11
|
+
import ocr_util.eval.model as do
|
|
12
|
+
import ocr_util.eval.model.filter as dofi
|
|
13
|
+
import ocr_util.eval.cli as eval_cli
|
|
14
|
+
import ocr_util.slice.cli as slice_cli
|
|
15
|
+
import ocr_util.show.cli as show_cli
|
|
16
|
+
import ocr_util.corpus.generate_corpus as gc
|
|
17
|
+
|
|
18
|
+
from ocr_util.corpus.common import CorpusArgs
|
|
19
|
+
|
|
20
|
+
# script constants
|
|
21
|
+
DEFAULT_VERBOSITY = 0
|
|
22
|
+
SUB_CMD_FRAME = "frame"
|
|
23
|
+
SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
|
|
24
|
+
CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
|
|
25
|
+
CORPUS_CACHE_DIR = os.path.join(
|
|
26
|
+
os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
SUB_CMD_EVALUATE = "eval"
|
|
30
|
+
SUB_CMD_SLICE = "slice"
|
|
31
|
+
SUB_CMD_SHOW = "show"
|
|
32
|
+
|
|
33
|
+
# Remove this constant as it's now managed by show_cli
|
|
34
|
+
# SUB_CMD_SHOW is kept for backward compatibility with other references
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def points_type(points: str) -> str:
|
|
38
|
+
match: re.Match = re.match(dofi.PolygonFrameFilterUtil.POINT_LIST_PATTERN, points)
|
|
39
|
+
if not match:
|
|
40
|
+
raise argparse.ArgumentTypeError(f"Invalid point coordinates: '{points}'")
|
|
41
|
+
return points
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def start() -> None:
|
|
45
|
+
# Configure logging once, centrally
|
|
46
|
+
logging.basicConfig(
|
|
47
|
+
level=logging.INFO,
|
|
48
|
+
format='%(asctime)s [%(levelname)s][%(name)s] %(message)s',
|
|
49
|
+
datefmt='%Y-%m-%d %H:%M:%S'
|
|
50
|
+
)
|
|
51
|
+
arg_parser: argparse.ArgumentParser = argparse.ArgumentParser(
|
|
52
|
+
prog="ocr-util",
|
|
53
|
+
description=f"OCR Util {ocr_util.__version__} of ULB Sachsen-Anhalt",
|
|
54
|
+
)
|
|
55
|
+
sub_arg_parsers = arg_parser.add_subparsers(
|
|
56
|
+
title="Subkommandos",
|
|
57
|
+
dest="subcommand",
|
|
58
|
+
required=True,
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# frames subcommand
|
|
62
|
+
frame_arg_parser = sub_arg_parsers.add_parser(
|
|
63
|
+
SUB_CMD_FRAME,
|
|
64
|
+
help="Filter Contents of provided ALTO-v3-Data by provided Coordinates, where Coordinates span a rectangular"
|
|
65
|
+
" box with",
|
|
66
|
+
)
|
|
67
|
+
frame_arg_parser.add_argument(
|
|
68
|
+
"-v",
|
|
69
|
+
"--verbosity",
|
|
70
|
+
action="count",
|
|
71
|
+
default=DEFAULT_VERBOSITY,
|
|
72
|
+
required=False,
|
|
73
|
+
help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
|
|
74
|
+
)
|
|
75
|
+
frame_arg_parser.add_argument(
|
|
76
|
+
"-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True
|
|
77
|
+
)
|
|
78
|
+
frame_arg_parser.add_argument(
|
|
79
|
+
"-o",
|
|
80
|
+
"--output-ocr-file",
|
|
81
|
+
help="Path of resulting OCR-Data file",
|
|
82
|
+
required=False,
|
|
83
|
+
default=None,
|
|
84
|
+
)
|
|
85
|
+
frame_arg_parser.add_argument(
|
|
86
|
+
"-p",
|
|
87
|
+
"--points",
|
|
88
|
+
required=True,
|
|
89
|
+
type=points_type,
|
|
90
|
+
help="""
|
|
91
|
+
Frame to slice words/lines/regions from input OCR-Data
|
|
92
|
+
f.e.: --frame "2892,2480 5072,2480 5072,5148 2892,5148"
|
|
93
|
+
""",
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
# groundtruth-corpus subcommand
|
|
97
|
+
groundtruth_corpus_arg_parser = sub_arg_parsers.add_parser(
|
|
98
|
+
SUB_CMD_GROUNDTRUTH_CORPUS,
|
|
99
|
+
help="Create METS file from N ground truth PAGE-XML files with URN identifiers",
|
|
100
|
+
)
|
|
101
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
102
|
+
"-i",
|
|
103
|
+
"--input",
|
|
104
|
+
dest="input_dir",
|
|
105
|
+
help="Path to the input directory containing GT PAGE-XML files",
|
|
106
|
+
required=True,
|
|
107
|
+
)
|
|
108
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
109
|
+
"-o",
|
|
110
|
+
"--output",
|
|
111
|
+
dest="output_dir",
|
|
112
|
+
help="Path to the output directory for generated corpus",
|
|
113
|
+
required=True,
|
|
114
|
+
)
|
|
115
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
116
|
+
"-l",
|
|
117
|
+
"--limit",
|
|
118
|
+
type=int,
|
|
119
|
+
default=0,
|
|
120
|
+
help="Number of files to process (default: 0 = unlimited)",
|
|
121
|
+
required=False,
|
|
122
|
+
)
|
|
123
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
124
|
+
"-t",
|
|
125
|
+
"--temp-dir",
|
|
126
|
+
dest="temp_dir",
|
|
127
|
+
default=CORPUS_CACHE_DIR,
|
|
128
|
+
help=f"Path to temporary directory for caching METS files (default: {CORPUS_CACHE_DIR})",
|
|
129
|
+
required=False,
|
|
130
|
+
)
|
|
131
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
132
|
+
"-v",
|
|
133
|
+
"--verbosity",
|
|
134
|
+
action="count",
|
|
135
|
+
default=DEFAULT_VERBOSITY,
|
|
136
|
+
required=False,
|
|
137
|
+
help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
|
|
138
|
+
)
|
|
139
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
140
|
+
"--corpus-label",
|
|
141
|
+
dest="corpus_label",
|
|
142
|
+
default="Ground Truth Corpus",
|
|
143
|
+
help="Label for the corpus in the METS logical structure (default: 'Ground Truth Corpus')",
|
|
144
|
+
required=False,
|
|
145
|
+
)
|
|
146
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
147
|
+
"--oai-base-url",
|
|
148
|
+
dest="oai_base_url",
|
|
149
|
+
help="Base URL for OAI-PMH requests",
|
|
150
|
+
required=False,
|
|
151
|
+
)
|
|
152
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
153
|
+
"--clear-cache",
|
|
154
|
+
dest="clear_cache",
|
|
155
|
+
action="store_true",
|
|
156
|
+
default=False,
|
|
157
|
+
help="Clear the local cache directory before processing (default: False)",
|
|
158
|
+
required=False,
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
# evaluate subcommand
|
|
162
|
+
evaluate_arg_parser = sub_arg_parsers.add_parser(
|
|
163
|
+
SUB_CMD_EVALUATE,
|
|
164
|
+
help="Evaluate OCR candidates against ground truth data",
|
|
165
|
+
add_help=True,
|
|
166
|
+
)
|
|
167
|
+
eval_cli.register_arguments(evaluate_arg_parser)
|
|
168
|
+
|
|
169
|
+
# slice subcommand
|
|
170
|
+
slice_arg_parser = sub_arg_parsers.add_parser(
|
|
171
|
+
SUB_CMD_SLICE,
|
|
172
|
+
help="Generate pairs of textlines and image frames from OCR and image data",
|
|
173
|
+
add_help=True,
|
|
174
|
+
)
|
|
175
|
+
slice_arg_parser.add_argument(
|
|
176
|
+
"data",
|
|
177
|
+
type=str,
|
|
178
|
+
help="path to local alto|page file corresponding to image",
|
|
179
|
+
)
|
|
180
|
+
slice_arg_parser.add_argument(
|
|
181
|
+
"-i",
|
|
182
|
+
"--image",
|
|
183
|
+
required=True,
|
|
184
|
+
help="path to local image file tif|jpg|png corresponding to ocr",
|
|
185
|
+
)
|
|
186
|
+
slice_arg_parser.add_argument(
|
|
187
|
+
"-o",
|
|
188
|
+
"--output_dir",
|
|
189
|
+
default=slice_cli.DEFAULT_OUTDIR_PREFIX,
|
|
190
|
+
help=f"output directory, re-created if already exists (default: {slice_cli.DEFAULT_OUTDIR_PREFIX})",
|
|
191
|
+
)
|
|
192
|
+
slice_arg_parser.add_argument(
|
|
193
|
+
"--prefix-output",
|
|
194
|
+
required=False,
|
|
195
|
+
help="optional: prefix each pair using this arg (default: '')",
|
|
196
|
+
)
|
|
197
|
+
slice_arg_parser.add_argument(
|
|
198
|
+
"-m",
|
|
199
|
+
"--minchars",
|
|
200
|
+
required=False,
|
|
201
|
+
type=int,
|
|
202
|
+
default=int(slice_cli.DEFAULT_MIN_CHARS),
|
|
203
|
+
help=f"optional: minimum printable chars required for a line to be included into set (default: {slice_cli.DEFAULT_MIN_CHARS})",
|
|
204
|
+
)
|
|
205
|
+
slice_arg_parser.add_argument(
|
|
206
|
+
"-s",
|
|
207
|
+
"--summary",
|
|
208
|
+
required=False,
|
|
209
|
+
action="store_true",
|
|
210
|
+
default=slice_cli.DEFAULT_USE_SUMMARY,
|
|
211
|
+
help=f"optional: print all lines in additional file (default: {slice_cli.DEFAULT_USE_SUMMARY})",
|
|
212
|
+
)
|
|
213
|
+
slice_arg_parser.add_argument(
|
|
214
|
+
"-r",
|
|
215
|
+
"--reorder",
|
|
216
|
+
required=False,
|
|
217
|
+
action="store_true",
|
|
218
|
+
default=slice_cli.DEFAULT_USE_REORDER,
|
|
219
|
+
help=f"optional: re-order word tokens from right-to-left (default: {slice_cli.DEFAULT_USE_REORDER})",
|
|
220
|
+
)
|
|
221
|
+
slice_arg_parser.add_argument(
|
|
222
|
+
"--binarize",
|
|
223
|
+
required=False,
|
|
224
|
+
action="store_true",
|
|
225
|
+
default=slice_cli.DEFAULT_BINARIZE,
|
|
226
|
+
help=f"optional: binarize textline images (default: {slice_cli.DEFAULT_BINARIZE})",
|
|
227
|
+
)
|
|
228
|
+
slice_arg_parser.add_argument(
|
|
229
|
+
"--sanitize",
|
|
230
|
+
required=False,
|
|
231
|
+
type=bool,
|
|
232
|
+
default=slice_cli.DEFAULT_SANITIZE,
|
|
233
|
+
help=f"optional: sanitize textline images (default: {slice_cli.DEFAULT_SANITIZE})",
|
|
234
|
+
)
|
|
235
|
+
slice_arg_parser.add_argument(
|
|
236
|
+
"--no-sanitize", dest="sanitize", action="store_false"
|
|
237
|
+
)
|
|
238
|
+
slice_arg_parser.add_argument(
|
|
239
|
+
"--intrusion-ratio",
|
|
240
|
+
required=False,
|
|
241
|
+
default=slice_cli.DEFAULT_INTRUSION_RATIO,
|
|
242
|
+
help=f"optional: alter threshold for top and bottom ratios for intrusion detection for sanitizing (default: {slice_cli.DEFAULT_INTRUSION_RATIO})",
|
|
243
|
+
)
|
|
244
|
+
slice_arg_parser.add_argument(
|
|
245
|
+
"--rotation-threshold",
|
|
246
|
+
required=False,
|
|
247
|
+
type=float,
|
|
248
|
+
default=slice_cli.DEFAULT_ROTATION_THRESH,
|
|
249
|
+
help=f"optional: alter threshold for rotation of textline image (default: {slice_cli.DEFAULT_ROTATION_THRESH})",
|
|
250
|
+
)
|
|
251
|
+
slice_arg_parser.add_argument(
|
|
252
|
+
"-p",
|
|
253
|
+
"--padding",
|
|
254
|
+
required=False,
|
|
255
|
+
type=int,
|
|
256
|
+
default=slice_cli.DEFAULT_PADDING,
|
|
257
|
+
help=f"optional: additional padding for existing textline image (default: {slice_cli.DEFAULT_PADDING})",
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
# show subcommand
|
|
261
|
+
show_cli.register_arguments(sub_arg_parsers)
|
|
262
|
+
|
|
263
|
+
args = arg_parser.parse_args()
|
|
264
|
+
|
|
265
|
+
verbosity: int = getattr(args, "verbosity", DEFAULT_VERBOSITY)
|
|
266
|
+
|
|
267
|
+
if args.subcommand == SUB_CMD_FRAME:
|
|
268
|
+
input_ocr_file: str = args.input_ocr_file
|
|
269
|
+
output_ocr_file: str = args.output_ocr_file
|
|
270
|
+
points: str = args.points
|
|
271
|
+
if verbosity > 1:
|
|
272
|
+
print(
|
|
273
|
+
f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}"
|
|
274
|
+
)
|
|
275
|
+
polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(
|
|
276
|
+
input_ocr_file, points, verbosity
|
|
277
|
+
)
|
|
278
|
+
piece_result: do.DigitalObjectTree = polygon_frame_filter.process()
|
|
279
|
+
file_result: PurePath = do.from_digital_object(piece_result, output_ocr_file)
|
|
280
|
+
if verbosity > 0:
|
|
281
|
+
print("[INFO ] file_result", file_result)
|
|
282
|
+
|
|
283
|
+
elif args.subcommand == SUB_CMD_GROUNDTRUTH_CORPUS:
|
|
284
|
+
corpus_args = CorpusArgs(
|
|
285
|
+
input_dir=Path(args.input_dir).absolute(),
|
|
286
|
+
output_dir=Path(args.output_dir).absolute(),
|
|
287
|
+
local_cache_dir=Path(args.temp_dir).absolute(),
|
|
288
|
+
limit=int(args.limit),
|
|
289
|
+
corpus_label=args.corpus_label,
|
|
290
|
+
clear_cache=args.clear_cache
|
|
291
|
+
)
|
|
292
|
+
gc.generate(corpus_args)
|
|
293
|
+
|
|
294
|
+
elif args.subcommand == SUB_CMD_EVALUATE:
|
|
295
|
+
eval_args = vars(args)
|
|
296
|
+
eval_args.pop("subcommand", None)
|
|
297
|
+
eval_cli.start_evaluation(eval_args)
|
|
298
|
+
|
|
299
|
+
elif args.subcommand == SUB_CMD_SLICE:
|
|
300
|
+
slice_args = vars(args)
|
|
301
|
+
slice_args.pop("subcommand", None)
|
|
302
|
+
slice_cli.start_slice(slice_args)
|
|
303
|
+
|
|
304
|
+
elif args.subcommand == SUB_CMD_SHOW:
|
|
305
|
+
show_args = vars(args)
|
|
306
|
+
show_args.pop("subcommand", None)
|
|
307
|
+
show_cli.start_show(show_args)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
if __name__ == "__main__":
|
|
311
|
+
start()
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Public API for the OCR ground truth corpus generation package.
|
|
2
|
+
|
|
3
|
+
Import :func:`~ocr_util.corpus.generate_corpus.generate` and
|
|
4
|
+
:class:`~ocr_util.corpus.common.CorpusArgs` to create a METS-based corpus
|
|
5
|
+
from a directory of PAGE-XML ground truth files::
|
|
6
|
+
|
|
7
|
+
from ocr_util.corpus.generate_corpus import generate
|
|
8
|
+
from ocr_util.corpus.common import CorpusArgs
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
result = generate(CorpusArgs(
|
|
12
|
+
input_dir=Path("gt/"),
|
|
13
|
+
output_dir=Path("corpus/"),
|
|
14
|
+
local_cache_dir=Path("/tmp/mets_cache"),
|
|
15
|
+
))
|
|
16
|
+
print(result.file_path, result.n_pages)
|
|
17
|
+
"""
|