ocr-util 2.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. ocr_util-2.0.1/LICENSE +21 -0
  2. ocr_util-2.0.1/PKG-INFO +136 -0
  3. ocr_util-2.0.1/README.md +103 -0
  4. ocr_util-2.0.1/pyproject.toml +81 -0
  5. ocr_util-2.0.1/setup.cfg +4 -0
  6. ocr_util-2.0.1/src/ocr_util/__init__.py +3 -0
  7. ocr_util-2.0.1/src/ocr_util/cli.py +311 -0
  8. ocr_util-2.0.1/src/ocr_util/corpus/__init__.py +17 -0
  9. ocr_util-2.0.1/src/ocr_util/corpus/common.py +522 -0
  10. ocr_util-2.0.1/src/ocr_util/corpus/generate_corpus.py +149 -0
  11. ocr_util-2.0.1/src/ocr_util/corpus/load_metadata.py +325 -0
  12. ocr_util-2.0.1/src/ocr_util/corpus/template.corpus.xml +24 -0
  13. ocr_util-2.0.1/src/ocr_util/eval/__init__.py +34 -0
  14. ocr_util-2.0.1/src/ocr_util/eval/aggregation.py +634 -0
  15. ocr_util-2.0.1/src/ocr_util/eval/cli.py +827 -0
  16. ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/__init__.py +0 -0
  17. ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/common.py +84 -0
  18. ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +116 -0
  19. ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/Util.py +91 -0
  20. ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
  21. ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/common.py +49 -0
  22. ocr_util-2.0.1/src/ocr_util/eval/evaluation.py +628 -0
  23. ocr_util-2.0.1/src/ocr_util/eval/geometry.py +117 -0
  24. ocr_util-2.0.1/src/ocr_util/eval/metrics.py +271 -0
  25. ocr_util-2.0.1/src/ocr_util/eval/model/common.py +107 -0
  26. ocr_util-2.0.1/src/ocr_util/eval/model/digital_object_model.py +257 -0
  27. ocr_util-2.0.1/src/ocr_util/eval/model/digital_object_util.py +120 -0
  28. ocr_util-2.0.1/src/ocr_util/eval/model/filter.py +130 -0
  29. ocr_util-2.0.1/src/ocr_util/eval/model/format_alto_v3_util.py +247 -0
  30. ocr_util-2.0.1/src/ocr_util/eval/model/format_page_util.py +267 -0
  31. ocr_util-2.0.1/src/ocr_util/eval/model/main.py +16 -0
  32. ocr_util-2.0.1/src/ocr_util/eval/model/minidom_util.py +46 -0
  33. ocr_util-2.0.1/src/ocr_util/eval/preprocessing.py +439 -0
  34. ocr_util-2.0.1/src/ocr_util/eval/resolve.py +55 -0
  35. ocr_util-2.0.1/src/ocr_util/show/cli.py +66 -0
  36. ocr_util-2.0.1/src/ocr_util/show/ocr_show_segmentation.py +396 -0
  37. ocr_util-2.0.1/src/ocr_util/slice/__init__.py +4 -0
  38. ocr_util-2.0.1/src/ocr_util/slice/cli.py +225 -0
  39. ocr_util-2.0.1/src/ocr_util/slice/gts_pairs.py +702 -0
  40. ocr_util-2.0.1/src/ocr_util/slice/pairs_lstmfs.py +77 -0
  41. ocr_util-2.0.1/src/ocr_util.egg-info/PKG-INFO +136 -0
  42. ocr_util-2.0.1/src/ocr_util.egg-info/SOURCES.txt +63 -0
  43. ocr_util-2.0.1/src/ocr_util.egg-info/dependency_links.txt +1 -0
  44. ocr_util-2.0.1/src/ocr_util.egg-info/entry_points.txt +2 -0
  45. ocr_util-2.0.1/src/ocr_util.egg-info/requires.txt +22 -0
  46. ocr_util-2.0.1/src/ocr_util.egg-info/top_level.txt +1 -0
  47. ocr_util-2.0.1/tests/test_aggregation.py +939 -0
  48. ocr_util-2.0.1/tests/test_corpus.py +526 -0
  49. ocr_util-2.0.1/tests/test_corpus_examples.py +317 -0
  50. ocr_util-2.0.1/tests/test_corpus_fulltext_generation.py +144 -0
  51. ocr_util-2.0.1/tests/test_corpus_load_metadata.py +183 -0
  52. ocr_util-2.0.1/tests/test_digital_eval_cli.py +524 -0
  53. ocr_util-2.0.1/tests/test_digital_object_model.py +131 -0
  54. ocr_util-2.0.1/tests/test_digital_object_ocr_alto.py +274 -0
  55. ocr_util-2.0.1/tests/test_digital_object_ocr_filter.py +124 -0
  56. ocr_util-2.0.1/tests/test_digital_object_ocr_page.py +230 -0
  57. ocr_util-2.0.1/tests/test_digital_object_ocr_page_eynollah.py +54 -0
  58. ocr_util-2.0.1/tests/test_digital_object_txt.py +25 -0
  59. ocr_util-2.0.1/tests/test_generate_sets.py +635 -0
  60. ocr_util-2.0.1/tests/test_ocr_evaluate.py +498 -0
  61. ocr_util-2.0.1/tests/test_ocr_metrics.py +462 -0
  62. ocr_util-2.0.1/tests/test_ocr_metrics_base.py +105 -0
  63. ocr_util-2.0.1/tests/test_ocr_preprocessing.py +159 -0
  64. ocr_util-2.0.1/tests/test_page_reading_order.py +63 -0
  65. ocr_util-2.0.1/tests/test_show_segmentation.py +453 -0
ocr_util-2.0.1/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2022 Universitäts- und Landesbibliothek Sachsen-Anhalt
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,136 @@
1
+ Metadata-Version: 2.4
2
+ Name: ocr-util
3
+ Version: 2.0.1
4
+ Summary: OCR Utils
5
+ Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
6
+ Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: License :: OSI Approved :: MIT License
9
+ Requires-Python: >=3.10
10
+ Description-Content-Type: text/markdown
11
+ License-File: LICENSE
12
+ Requires-Dist: rapidfuzz>3
13
+ Requires-Dist: nltk
14
+ Requires-Dist: requests
15
+ Requires-Dist: docker
16
+ Requires-Dist: numpy
17
+ Requires-Dist: shapely
18
+ Requires-Dist: lxml
19
+ Requires-Dist: opencv-python-headless
20
+ Requires-Dist: exifread
21
+ Requires-Dist: Pillow
22
+ Provides-Extra: dev
23
+ Requires-Dist: black; extra == "dev"
24
+ Requires-Dist: pylint; extra == "dev"
25
+ Provides-Extra: test
26
+ Requires-Dist: pytest-cov; extra == "test"
27
+ Requires-Dist: coverage-badge; extra == "test"
28
+ Requires-Dist: pytest-xdist; extra == "test"
29
+ Requires-Dist: flake8; extra == "test"
30
+ Requires-Dist: lxml-stubs; extra == "test"
31
+ Requires-Dist: opencv-stubs; extra == "test"
32
+ Dynamic: license-file
33
+
34
+ # OCR Util
35
+
36
+ ![python-app](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml/badge.svg) [![Coverage](./coverage.svg)](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [![PyPi version](https://badgen.net/pypi/v/digital-eval/)](https://pypi.org/project/digital-eval) ![PyPI - Downloads](https://img.shields.io/pypi/dm/digital-eval) ![PyPI - License](https://img.shields.io/pypi/l/digital-eval) ![PyPI - Python Version](https://img.shields.io/pypi/pyversions/digital-eval)
37
+
38
+
39
+ Collection of utils to
40
+ * evaluation of OCR data for the masses
41
+ * generation of extended OCR-Evaluation Corpora
42
+ * generation of pair-wise Trainingdata for OCR-Backends
43
+
44
+ ## Requirements
45
+
46
+ * recent *nix-OS
47
+ * Python3.10+ Environment
48
+
49
+ ## Usage
50
+
51
+ Each section contains detailed usage help instructions:
52
+
53
+ ```bash
54
+ # evaluation
55
+ ocr eval --help
56
+
57
+ # corpus management
58
+ ocr corpus --help
59
+
60
+ # slice image by image + input OCR
61
+ ocr slice --help
62
+
63
+ # render image + input OCR
64
+ ocr show --help
65
+ ```
66
+
67
+ ### Data problems
68
+
69
+ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
70
+
71
+ _Please note_:
72
+ Invalid data files are tried(!) to be excluded from evaluation.
73
+
74
+ ### Evaluation Filter-Then-Aggregate
75
+
76
+ The evaluation CLI supports a single pre-aggregation filter using metadata extractors.
77
+
78
+ Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
79
+
80
+ ```bash
81
+ ocr eval <candidates> \
82
+ --reference <groundtruth> \
83
+ --mets-file <mets.xml> \
84
+ --filter-by "mods:language=ger" \
85
+ --aggregate-by "mods:dateIssued:century"
86
+ ```
87
+
88
+ Multi-language filter values are interpreted as sets:
89
+
90
+ ```bash
91
+ ocr eval <candidates> \
92
+ --reference <groundtruth> \
93
+ --mets-file <mets.xml> \
94
+ --filter-by "mods:language=ger+eng" \
95
+ --aggregate-by "mods:dateIssued:century"
96
+ ```
97
+
98
+ Behavior:
99
+ * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
100
+ * multi-value filter -> all filter values must be present in any order
101
+ * entries missing the filter criterion are reported as WARNING and discarded
102
+
103
+ ## Development
104
+
105
+ Plattform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
106
+
107
+ ```bash
108
+ # clone local
109
+ git clone <repository-url> <local-dir>
110
+ cd <local-dir>
111
+
112
+ # enable virtual python 3 environment (linux)
113
+ # and update pip itself
114
+ python3.10 -m venv venv
115
+ . venv/bin/activate
116
+ python -m pip install -U pip
117
+
118
+ # install with dev dependencies
119
+ python -m pip install -e ".[dev,test]"
120
+
121
+ # run tests with coverage
122
+ python -m pytest --cov=src
123
+
124
+ # run tests faster (parallel, auto worker count)
125
+ python -m pytest -q -n auto
126
+ ```
127
+
128
+ ## Contribution
129
+
130
+ Contributions, suggestions and proposals welcome!
131
+
132
+ ## License
133
+
134
+ Under terms of the [MIT license](https://opensource.org/licenses/MIT).
135
+
136
+ **NOTE**: This software depends on packages that _might_ be licensed under different terms.
@@ -0,0 +1,103 @@
1
+ # OCR Util
2
+
3
+ ![python-app](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml/badge.svg) [![Coverage](./coverage.svg)](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [![PyPi version](https://badgen.net/pypi/v/digital-eval/)](https://pypi.org/project/digital-eval) ![PyPI - Downloads](https://img.shields.io/pypi/dm/digital-eval) ![PyPI - License](https://img.shields.io/pypi/l/digital-eval) ![PyPI - Python Version](https://img.shields.io/pypi/pyversions/digital-eval)
4
+
5
+
6
+ Collection of utils to
7
+ * evaluation of OCR data for the masses
8
+ * generation of extended OCR-Evaluation Corpora
9
+ * generation of pair-wise Trainingdata for OCR-Backends
10
+
11
+ ## Requirements
12
+
13
+ * recent *nix-OS
14
+ * Python3.10+ Environment
15
+
16
+ ## Usage
17
+
18
+ Each section contains detailed usage help instructions:
19
+
20
+ ```bash
21
+ # evaluation
22
+ ocr eval --help
23
+
24
+ # corpus management
25
+ ocr corpus --help
26
+
27
+ # slice image by image + input OCR
28
+ ocr slice --help
29
+
30
+ # render image + input OCR
31
+ ocr show --help
32
+ ```
33
+
34
+ ### Data problems
35
+
36
+ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
37
+
38
+ _Please note_:
39
+ Invalid data files are tried(!) to be excluded from evaluation.
40
+
41
+ ### Evaluation Filter-Then-Aggregate
42
+
43
+ The evaluation CLI supports a single pre-aggregation filter using metadata extractors.
44
+
45
+ Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
46
+
47
+ ```bash
48
+ ocr eval <candidates> \
49
+ --reference <groundtruth> \
50
+ --mets-file <mets.xml> \
51
+ --filter-by "mods:language=ger" \
52
+ --aggregate-by "mods:dateIssued:century"
53
+ ```
54
+
55
+ Multi-language filter values are interpreted as sets:
56
+
57
+ ```bash
58
+ ocr eval <candidates> \
59
+ --reference <groundtruth> \
60
+ --mets-file <mets.xml> \
61
+ --filter-by "mods:language=ger+eng" \
62
+ --aggregate-by "mods:dateIssued:century"
63
+ ```
64
+
65
+ Behavior:
66
+ * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
67
+ * multi-value filter -> all filter values must be present in any order
68
+ * entries missing the filter criterion are reported as WARNING and discarded
69
+
70
+ ## Development
71
+
72
+ Plattform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
73
+
74
+ ```bash
75
+ # clone local
76
+ git clone <repository-url> <local-dir>
77
+ cd <local-dir>
78
+
79
+ # enable virtual python 3 environment (linux)
80
+ # and update pip itself
81
+ python3.10 -m venv venv
82
+ . venv/bin/activate
83
+ python -m pip install -U pip
84
+
85
+ # install with dev dependencies
86
+ python -m pip install -e ".[dev,test]"
87
+
88
+ # run tests with coverage
89
+ python -m pytest --cov=src
90
+
91
+ # run tests faster (parallel, auto worker count)
92
+ python -m pytest -q -n auto
93
+ ```
94
+
95
+ ## Contribution
96
+
97
+ Contributions, suggestions and proposals welcome!
98
+
99
+ ## License
100
+
101
+ Under terms of the [MIT license](https://opensource.org/licenses/MIT).
102
+
103
+ **NOTE**: This software depends on packages that _might_ be licensed under different terms.
@@ -0,0 +1,81 @@
1
+ [project]
2
+ name = "ocr-util"
3
+ dynamic = ["version"]
4
+ description = "OCR Utils"
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ authors = [{name = "Universitäts- und Landesbibliothek Sachsen-Anhalt",email = "development@bibliothek.uni-halle.de"}]
8
+ classifiers = [
9
+ "Programming Language :: Python :: 3",
10
+ "License :: OSI Approved :: MIT License"
11
+ ]
12
+ dependencies = [
13
+ "rapidfuzz>3",
14
+ "nltk",
15
+ "requests",
16
+ "docker",
17
+ "numpy",
18
+ "shapely",
19
+ "lxml",
20
+ "opencv-python-headless",
21
+ "exifread",
22
+ "Pillow"
23
+ ]
24
+
25
+ [project.optional-dependencies]
26
+ dev = [
27
+ "black",
28
+ "pylint",
29
+ ]
30
+ test = [
31
+ "pytest-cov",
32
+ "coverage-badge",
33
+ "pytest-xdist",
34
+ "flake8",
35
+ "lxml-stubs",
36
+ "opencv-stubs",
37
+ ]
38
+
39
+ [project.urls]
40
+ Homepage = "https://github.com/ulb-sachsen-anhalt/ocr-util"
41
+
42
+ [project.scripts]
43
+ ocr-util = "ocr_util.cli:start"
44
+
45
+ [build-system]
46
+ requires = ["setuptools>=61.0.0", "wheel"]
47
+ build-backend = "setuptools.build_meta"
48
+
49
+ [tool.setuptools.dynamic]
50
+ version = {attr = "ocr_util.__version__"}
51
+
52
+ [tool.setuptools.packages.find]
53
+ where = ["src"]
54
+
55
+ [tool.setuptools.package-data]
56
+ ocr_util = ["corpus/*.xml"]
57
+
58
+ [tool.setuptools]
59
+ package-dir = {"" = "src"}
60
+
61
+ [tool.black]
62
+ line-length = 120
63
+ target-version = ["py310", "py311", "py312"]
64
+ include = '\.pyi?$'
65
+
66
+ [tool.pylint.main]
67
+ extension-pkg-allow-list=["lxml.etree"]
68
+
69
+ [tool.pylint.messages_control]
70
+ reportAttributeAccessIssue = "none"
71
+
72
+ [tool.pytest.ini_options]
73
+ testpaths = ["tests"]
74
+ python_files = ["test_*.py"]
75
+ python_classes = ["Test*"]
76
+ python_functions = ["test_*"]
77
+ addopts = "-v --tb=short --cov-report=xml"
78
+
79
+ [tool.coverage.run]
80
+ source = ["ocr_util"]
81
+ omit = ["*/tests/*", "*/__pycache__/*", "*/venv/*", "*/build/*"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,3 @@
1
+ """main API"""
2
+
3
+ __version__ = "2.0.1"
@@ -0,0 +1,311 @@
1
+ # -*- coding: utf-8 -*-
2
+ """OCR Utils"""
3
+
4
+ import argparse
5
+ import logging
6
+ import os
7
+ import re
8
+ from pathlib import Path, PurePath
9
+
10
+ import ocr_util
11
+ import ocr_util.eval.model as do
12
+ import ocr_util.eval.model.filter as dofi
13
+ import ocr_util.eval.cli as eval_cli
14
+ import ocr_util.slice.cli as slice_cli
15
+ import ocr_util.show.cli as show_cli
16
+ import ocr_util.corpus.generate_corpus as gc
17
+
18
+ from ocr_util.corpus.common import CorpusArgs
19
+
20
+ # script constants
21
+ DEFAULT_VERBOSITY = 0
22
+ SUB_CMD_FRAME = "frame"
23
+ SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
24
+ CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
25
+ CORPUS_CACHE_DIR = os.path.join(
26
+ os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME
27
+ )
28
+
29
+ SUB_CMD_EVALUATE = "eval"
30
+ SUB_CMD_SLICE = "slice"
31
+ SUB_CMD_SHOW = "show"
32
+
33
+ # Remove this constant as it's now managed by show_cli
34
+ # SUB_CMD_SHOW is kept for backward compatibility with other references
35
+
36
+
37
+ def points_type(points: str) -> str:
38
+ match: re.Match = re.match(dofi.PolygonFrameFilterUtil.POINT_LIST_PATTERN, points)
39
+ if not match:
40
+ raise argparse.ArgumentTypeError(f"Invalid point coordinates: '{points}'")
41
+ return points
42
+
43
+
44
+ def start() -> None:
45
+ # Configure logging once, centrally
46
+ logging.basicConfig(
47
+ level=logging.INFO,
48
+ format='%(asctime)s [%(levelname)s][%(name)s] %(message)s',
49
+ datefmt='%Y-%m-%d %H:%M:%S'
50
+ )
51
+ arg_parser: argparse.ArgumentParser = argparse.ArgumentParser(
52
+ prog="ocr-util",
53
+ description=f"OCR Util {ocr_util.__version__} of ULB Sachsen-Anhalt",
54
+ )
55
+ sub_arg_parsers = arg_parser.add_subparsers(
56
+ title="Subkommandos",
57
+ dest="subcommand",
58
+ required=True,
59
+ )
60
+
61
+ # frames subcommand
62
+ frame_arg_parser = sub_arg_parsers.add_parser(
63
+ SUB_CMD_FRAME,
64
+ help="Filter Contents of provided ALTO-v3-Data by provided Coordinates, where Coordinates span a rectangular"
65
+ " box with",
66
+ )
67
+ frame_arg_parser.add_argument(
68
+ "-v",
69
+ "--verbosity",
70
+ action="count",
71
+ default=DEFAULT_VERBOSITY,
72
+ required=False,
73
+ help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
74
+ )
75
+ frame_arg_parser.add_argument(
76
+ "-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True
77
+ )
78
+ frame_arg_parser.add_argument(
79
+ "-o",
80
+ "--output-ocr-file",
81
+ help="Path of resulting OCR-Data file",
82
+ required=False,
83
+ default=None,
84
+ )
85
+ frame_arg_parser.add_argument(
86
+ "-p",
87
+ "--points",
88
+ required=True,
89
+ type=points_type,
90
+ help="""
91
+ Frame to slice words/lines/regions from input OCR-Data
92
+ f.e.: --frame "2892,2480 5072,2480 5072,5148 2892,5148"
93
+ """,
94
+ )
95
+
96
+ # groundtruth-corpus subcommand
97
+ groundtruth_corpus_arg_parser = sub_arg_parsers.add_parser(
98
+ SUB_CMD_GROUNDTRUTH_CORPUS,
99
+ help="Create METS file from N ground truth PAGE-XML files with URN identifiers",
100
+ )
101
+ groundtruth_corpus_arg_parser.add_argument(
102
+ "-i",
103
+ "--input",
104
+ dest="input_dir",
105
+ help="Path to the input directory containing GT PAGE-XML files",
106
+ required=True,
107
+ )
108
+ groundtruth_corpus_arg_parser.add_argument(
109
+ "-o",
110
+ "--output",
111
+ dest="output_dir",
112
+ help="Path to the output directory for generated corpus",
113
+ required=True,
114
+ )
115
+ groundtruth_corpus_arg_parser.add_argument(
116
+ "-l",
117
+ "--limit",
118
+ type=int,
119
+ default=0,
120
+ help="Number of files to process (default: 0 = unlimited)",
121
+ required=False,
122
+ )
123
+ groundtruth_corpus_arg_parser.add_argument(
124
+ "-t",
125
+ "--temp-dir",
126
+ dest="temp_dir",
127
+ default=CORPUS_CACHE_DIR,
128
+ help=f"Path to temporary directory for caching METS files (default: {CORPUS_CACHE_DIR})",
129
+ required=False,
130
+ )
131
+ groundtruth_corpus_arg_parser.add_argument(
132
+ "-v",
133
+ "--verbosity",
134
+ action="count",
135
+ default=DEFAULT_VERBOSITY,
136
+ required=False,
137
+ help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
138
+ )
139
+ groundtruth_corpus_arg_parser.add_argument(
140
+ "--corpus-label",
141
+ dest="corpus_label",
142
+ default="Ground Truth Corpus",
143
+ help="Label for the corpus in the METS logical structure (default: 'Ground Truth Corpus')",
144
+ required=False,
145
+ )
146
+ groundtruth_corpus_arg_parser.add_argument(
147
+ "--oai-base-url",
148
+ dest="oai_base_url",
149
+ help="Base URL for OAI-PMH requests",
150
+ required=False,
151
+ )
152
+ groundtruth_corpus_arg_parser.add_argument(
153
+ "--clear-cache",
154
+ dest="clear_cache",
155
+ action="store_true",
156
+ default=False,
157
+ help="Clear the local cache directory before processing (default: False)",
158
+ required=False,
159
+ )
160
+
161
+ # evaluate subcommand
162
+ evaluate_arg_parser = sub_arg_parsers.add_parser(
163
+ SUB_CMD_EVALUATE,
164
+ help="Evaluate OCR candidates against ground truth data",
165
+ add_help=True,
166
+ )
167
+ eval_cli.register_arguments(evaluate_arg_parser)
168
+
169
+ # slice subcommand
170
+ slice_arg_parser = sub_arg_parsers.add_parser(
171
+ SUB_CMD_SLICE,
172
+ help="Generate pairs of textlines and image frames from OCR and image data",
173
+ add_help=True,
174
+ )
175
+ slice_arg_parser.add_argument(
176
+ "data",
177
+ type=str,
178
+ help="path to local alto|page file corresponding to image",
179
+ )
180
+ slice_arg_parser.add_argument(
181
+ "-i",
182
+ "--image",
183
+ required=True,
184
+ help="path to local image file tif|jpg|png corresponding to ocr",
185
+ )
186
+ slice_arg_parser.add_argument(
187
+ "-o",
188
+ "--output_dir",
189
+ default=slice_cli.DEFAULT_OUTDIR_PREFIX,
190
+ help=f"output directory, re-created if already exists (default: {slice_cli.DEFAULT_OUTDIR_PREFIX})",
191
+ )
192
+ slice_arg_parser.add_argument(
193
+ "--prefix-output",
194
+ required=False,
195
+ help="optional: prefix each pair using this arg (default: '')",
196
+ )
197
+ slice_arg_parser.add_argument(
198
+ "-m",
199
+ "--minchars",
200
+ required=False,
201
+ type=int,
202
+ default=int(slice_cli.DEFAULT_MIN_CHARS),
203
+ help=f"optional: minimum printable chars required for a line to be included into set (default: {slice_cli.DEFAULT_MIN_CHARS})",
204
+ )
205
+ slice_arg_parser.add_argument(
206
+ "-s",
207
+ "--summary",
208
+ required=False,
209
+ action="store_true",
210
+ default=slice_cli.DEFAULT_USE_SUMMARY,
211
+ help=f"optional: print all lines in additional file (default: {slice_cli.DEFAULT_USE_SUMMARY})",
212
+ )
213
+ slice_arg_parser.add_argument(
214
+ "-r",
215
+ "--reorder",
216
+ required=False,
217
+ action="store_true",
218
+ default=slice_cli.DEFAULT_USE_REORDER,
219
+ help=f"optional: re-order word tokens from right-to-left (default: {slice_cli.DEFAULT_USE_REORDER})",
220
+ )
221
+ slice_arg_parser.add_argument(
222
+ "--binarize",
223
+ required=False,
224
+ action="store_true",
225
+ default=slice_cli.DEFAULT_BINARIZE,
226
+ help=f"optional: binarize textline images (default: {slice_cli.DEFAULT_BINARIZE})",
227
+ )
228
+ slice_arg_parser.add_argument(
229
+ "--sanitize",
230
+ required=False,
231
+ type=bool,
232
+ default=slice_cli.DEFAULT_SANITIZE,
233
+ help=f"optional: sanitize textline images (default: {slice_cli.DEFAULT_SANITIZE})",
234
+ )
235
+ slice_arg_parser.add_argument(
236
+ "--no-sanitize", dest="sanitize", action="store_false"
237
+ )
238
+ slice_arg_parser.add_argument(
239
+ "--intrusion-ratio",
240
+ required=False,
241
+ default=slice_cli.DEFAULT_INTRUSION_RATIO,
242
+ help=f"optional: alter threshold for top and bottom ratios for intrusion detection for sanitizing (default: {slice_cli.DEFAULT_INTRUSION_RATIO})",
243
+ )
244
+ slice_arg_parser.add_argument(
245
+ "--rotation-threshold",
246
+ required=False,
247
+ type=float,
248
+ default=slice_cli.DEFAULT_ROTATION_THRESH,
249
+ help=f"optional: alter threshold for rotation of textline image (default: {slice_cli.DEFAULT_ROTATION_THRESH})",
250
+ )
251
+ slice_arg_parser.add_argument(
252
+ "-p",
253
+ "--padding",
254
+ required=False,
255
+ type=int,
256
+ default=slice_cli.DEFAULT_PADDING,
257
+ help=f"optional: additional padding for existing textline image (default: {slice_cli.DEFAULT_PADDING})",
258
+ )
259
+
260
+ # show subcommand
261
+ show_cli.register_arguments(sub_arg_parsers)
262
+
263
+ args = arg_parser.parse_args()
264
+
265
+ verbosity: int = getattr(args, "verbosity", DEFAULT_VERBOSITY)
266
+
267
+ if args.subcommand == SUB_CMD_FRAME:
268
+ input_ocr_file: str = args.input_ocr_file
269
+ output_ocr_file: str = args.output_ocr_file
270
+ points: str = args.points
271
+ if verbosity > 1:
272
+ print(
273
+ f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}"
274
+ )
275
+ polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(
276
+ input_ocr_file, points, verbosity
277
+ )
278
+ piece_result: do.DigitalObjectTree = polygon_frame_filter.process()
279
+ file_result: PurePath = do.from_digital_object(piece_result, output_ocr_file)
280
+ if verbosity > 0:
281
+ print("[INFO ] file_result", file_result)
282
+
283
+ elif args.subcommand == SUB_CMD_GROUNDTRUTH_CORPUS:
284
+ corpus_args = CorpusArgs(
285
+ input_dir=Path(args.input_dir).absolute(),
286
+ output_dir=Path(args.output_dir).absolute(),
287
+ local_cache_dir=Path(args.temp_dir).absolute(),
288
+ limit=int(args.limit),
289
+ corpus_label=args.corpus_label,
290
+ clear_cache=args.clear_cache
291
+ )
292
+ gc.generate(corpus_args)
293
+
294
+ elif args.subcommand == SUB_CMD_EVALUATE:
295
+ eval_args = vars(args)
296
+ eval_args.pop("subcommand", None)
297
+ eval_cli.start_evaluation(eval_args)
298
+
299
+ elif args.subcommand == SUB_CMD_SLICE:
300
+ slice_args = vars(args)
301
+ slice_args.pop("subcommand", None)
302
+ slice_cli.start_slice(slice_args)
303
+
304
+ elif args.subcommand == SUB_CMD_SHOW:
305
+ show_args = vars(args)
306
+ show_args.pop("subcommand", None)
307
+ show_cli.start_show(show_args)
308
+
309
+
310
+ if __name__ == "__main__":
311
+ start()
@@ -0,0 +1,17 @@
1
+ """Public API for the OCR ground truth corpus generation package.
2
+
3
+ Import :func:`~ocr_util.corpus.generate_corpus.generate` and
4
+ :class:`~ocr_util.corpus.common.CorpusArgs` to create a METS-based corpus
5
+ from a directory of PAGE-XML ground truth files::
6
+
7
+ from ocr_util.corpus.generate_corpus import generate
8
+ from ocr_util.corpus.common import CorpusArgs
9
+ from pathlib import Path
10
+
11
+ result = generate(CorpusArgs(
12
+ input_dir=Path("gt/"),
13
+ output_dir=Path("corpus/"),
14
+ local_cache_dir=Path("/tmp/mets_cache"),
15
+ ))
16
+ print(result.file_path, result.n_pages)
17
+ """