ocr-util 2.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ocr_util-2.0.1/LICENSE +21 -0
- ocr_util-2.0.1/PKG-INFO +136 -0
- ocr_util-2.0.1/README.md +103 -0
- ocr_util-2.0.1/pyproject.toml +81 -0
- ocr_util-2.0.1/setup.cfg +4 -0
- ocr_util-2.0.1/src/ocr_util/__init__.py +3 -0
- ocr_util-2.0.1/src/ocr_util/cli.py +311 -0
- ocr_util-2.0.1/src/ocr_util/corpus/__init__.py +17 -0
- ocr_util-2.0.1/src/ocr_util/corpus/common.py +522 -0
- ocr_util-2.0.1/src/ocr_util/corpus/generate_corpus.py +149 -0
- ocr_util-2.0.1/src/ocr_util/corpus/load_metadata.py +325 -0
- ocr_util-2.0.1/src/ocr_util/corpus/template.corpus.xml +24 -0
- ocr_util-2.0.1/src/ocr_util/eval/__init__.py +34 -0
- ocr_util-2.0.1/src/ocr_util/eval/aggregation.py +634 -0
- ocr_util-2.0.1/src/ocr_util/eval/cli.py +827 -0
- ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/__init__.py +0 -0
- ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/common.py +84 -0
- ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +116 -0
- ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/Util.py +91 -0
- ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
- ocr_util-2.0.1/src/ocr_util/eval/dictionary_metrics/language_tool/common.py +49 -0
- ocr_util-2.0.1/src/ocr_util/eval/evaluation.py +628 -0
- ocr_util-2.0.1/src/ocr_util/eval/geometry.py +117 -0
- ocr_util-2.0.1/src/ocr_util/eval/metrics.py +271 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/common.py +107 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/digital_object_model.py +257 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/digital_object_util.py +120 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/filter.py +130 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/format_alto_v3_util.py +247 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/format_page_util.py +267 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/main.py +16 -0
- ocr_util-2.0.1/src/ocr_util/eval/model/minidom_util.py +46 -0
- ocr_util-2.0.1/src/ocr_util/eval/preprocessing.py +439 -0
- ocr_util-2.0.1/src/ocr_util/eval/resolve.py +55 -0
- ocr_util-2.0.1/src/ocr_util/show/cli.py +66 -0
- ocr_util-2.0.1/src/ocr_util/show/ocr_show_segmentation.py +396 -0
- ocr_util-2.0.1/src/ocr_util/slice/__init__.py +4 -0
- ocr_util-2.0.1/src/ocr_util/slice/cli.py +225 -0
- ocr_util-2.0.1/src/ocr_util/slice/gts_pairs.py +702 -0
- ocr_util-2.0.1/src/ocr_util/slice/pairs_lstmfs.py +77 -0
- ocr_util-2.0.1/src/ocr_util.egg-info/PKG-INFO +136 -0
- ocr_util-2.0.1/src/ocr_util.egg-info/SOURCES.txt +63 -0
- ocr_util-2.0.1/src/ocr_util.egg-info/dependency_links.txt +1 -0
- ocr_util-2.0.1/src/ocr_util.egg-info/entry_points.txt +2 -0
- ocr_util-2.0.1/src/ocr_util.egg-info/requires.txt +22 -0
- ocr_util-2.0.1/src/ocr_util.egg-info/top_level.txt +1 -0
- ocr_util-2.0.1/tests/test_aggregation.py +939 -0
- ocr_util-2.0.1/tests/test_corpus.py +526 -0
- ocr_util-2.0.1/tests/test_corpus_examples.py +317 -0
- ocr_util-2.0.1/tests/test_corpus_fulltext_generation.py +144 -0
- ocr_util-2.0.1/tests/test_corpus_load_metadata.py +183 -0
- ocr_util-2.0.1/tests/test_digital_eval_cli.py +524 -0
- ocr_util-2.0.1/tests/test_digital_object_model.py +131 -0
- ocr_util-2.0.1/tests/test_digital_object_ocr_alto.py +274 -0
- ocr_util-2.0.1/tests/test_digital_object_ocr_filter.py +124 -0
- ocr_util-2.0.1/tests/test_digital_object_ocr_page.py +230 -0
- ocr_util-2.0.1/tests/test_digital_object_ocr_page_eynollah.py +54 -0
- ocr_util-2.0.1/tests/test_digital_object_txt.py +25 -0
- ocr_util-2.0.1/tests/test_generate_sets.py +635 -0
- ocr_util-2.0.1/tests/test_ocr_evaluate.py +498 -0
- ocr_util-2.0.1/tests/test_ocr_metrics.py +462 -0
- ocr_util-2.0.1/tests/test_ocr_metrics_base.py +105 -0
- ocr_util-2.0.1/tests/test_ocr_preprocessing.py +159 -0
- ocr_util-2.0.1/tests/test_page_reading_order.py +63 -0
- ocr_util-2.0.1/tests/test_show_segmentation.py +453 -0
ocr_util-2.0.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2022 Universitäts- und Landesbibliothek Sachsen-Anhalt
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
ocr_util-2.0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ocr-util
|
|
3
|
+
Version: 2.0.1
|
|
4
|
+
Summary: OCR Utils
|
|
5
|
+
Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
|
|
6
|
+
Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
9
|
+
Requires-Python: >=3.10
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: rapidfuzz>3
|
|
13
|
+
Requires-Dist: nltk
|
|
14
|
+
Requires-Dist: requests
|
|
15
|
+
Requires-Dist: docker
|
|
16
|
+
Requires-Dist: numpy
|
|
17
|
+
Requires-Dist: shapely
|
|
18
|
+
Requires-Dist: lxml
|
|
19
|
+
Requires-Dist: opencv-python-headless
|
|
20
|
+
Requires-Dist: exifread
|
|
21
|
+
Requires-Dist: Pillow
|
|
22
|
+
Provides-Extra: dev
|
|
23
|
+
Requires-Dist: black; extra == "dev"
|
|
24
|
+
Requires-Dist: pylint; extra == "dev"
|
|
25
|
+
Provides-Extra: test
|
|
26
|
+
Requires-Dist: pytest-cov; extra == "test"
|
|
27
|
+
Requires-Dist: coverage-badge; extra == "test"
|
|
28
|
+
Requires-Dist: pytest-xdist; extra == "test"
|
|
29
|
+
Requires-Dist: flake8; extra == "test"
|
|
30
|
+
Requires-Dist: lxml-stubs; extra == "test"
|
|
31
|
+
Requires-Dist: opencv-stubs; extra == "test"
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
34
|
+
# OCR Util
|
|
35
|
+
|
|
36
|
+
 [](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [](https://pypi.org/project/digital-eval)   
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
Collection of utils to
|
|
40
|
+
* evaluation of OCR data for the masses
|
|
41
|
+
* generation of extended OCR-Evaluation Corpora
|
|
42
|
+
* generation of pair-wise Trainingdata for OCR-Backends
|
|
43
|
+
|
|
44
|
+
## Requirements
|
|
45
|
+
|
|
46
|
+
* recent *nix-OS
|
|
47
|
+
* Python3.10+ Environment
|
|
48
|
+
|
|
49
|
+
## Usage
|
|
50
|
+
|
|
51
|
+
Each section contains detailed usage help instructions:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
# evaluation
|
|
55
|
+
ocr eval --help
|
|
56
|
+
|
|
57
|
+
# corpus management
|
|
58
|
+
ocr corpus --help
|
|
59
|
+
|
|
60
|
+
# slice image by image + input OCR
|
|
61
|
+
ocr slice --help
|
|
62
|
+
|
|
63
|
+
# render image + input OCR
|
|
64
|
+
ocr show --help
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### Data problems
|
|
68
|
+
|
|
69
|
+
Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
|
|
70
|
+
|
|
71
|
+
_Please note_:
|
|
72
|
+
Invalid data files are tried(!) to be excluded from evaluation.
|
|
73
|
+
|
|
74
|
+
### Evaluation Filter-Then-Aggregate
|
|
75
|
+
|
|
76
|
+
The evaluation CLI supports a single pre-aggregation filter using metadata extractors.
|
|
77
|
+
|
|
78
|
+
Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
ocr eval <candidates> \
|
|
82
|
+
--reference <groundtruth> \
|
|
83
|
+
--mets-file <mets.xml> \
|
|
84
|
+
--filter-by "mods:language=ger" \
|
|
85
|
+
--aggregate-by "mods:dateIssued:century"
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Multi-language filter values are interpreted as sets:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
ocr eval <candidates> \
|
|
92
|
+
--reference <groundtruth> \
|
|
93
|
+
--mets-file <mets.xml> \
|
|
94
|
+
--filter-by "mods:language=ger+eng" \
|
|
95
|
+
--aggregate-by "mods:dateIssued:century"
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Behavior:
|
|
99
|
+
* single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
|
|
100
|
+
* multi-value filter -> all filter values must be present in any order
|
|
101
|
+
* entries missing the filter criterion are reported as WARNING and discarded
|
|
102
|
+
|
|
103
|
+
## Development
|
|
104
|
+
|
|
105
|
+
Plattform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
# clone local
|
|
109
|
+
git clone <repository-url> <local-dir>
|
|
110
|
+
cd <local-dir>
|
|
111
|
+
|
|
112
|
+
# enable virtual python 3 environment (linux)
|
|
113
|
+
# and update pip itself
|
|
114
|
+
python3.10 -m venv venv
|
|
115
|
+
. venv/bin/activate
|
|
116
|
+
python -m pip install -U pip
|
|
117
|
+
|
|
118
|
+
# install with dev dependencies
|
|
119
|
+
python -m pip install -e ".[dev,test]"
|
|
120
|
+
|
|
121
|
+
# run tests with coverage
|
|
122
|
+
python -m pytest --cov=src
|
|
123
|
+
|
|
124
|
+
# run tests faster (parallel, auto worker count)
|
|
125
|
+
python -m pytest -q -n auto
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
## Contribution
|
|
129
|
+
|
|
130
|
+
Contributions, suggestions and proposals welcome!
|
|
131
|
+
|
|
132
|
+
## License
|
|
133
|
+
|
|
134
|
+
Under terms of the [MIT license](https://opensource.org/licenses/MIT).
|
|
135
|
+
|
|
136
|
+
**NOTE**: This software depends on packages that _might_ be licensed under different terms.
|
ocr_util-2.0.1/README.md
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# OCR Util
|
|
2
|
+
|
|
3
|
+
 [](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [](https://pypi.org/project/digital-eval)   
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
Collection of utils to
|
|
7
|
+
* evaluation of OCR data for the masses
|
|
8
|
+
* generation of extended OCR-Evaluation Corpora
|
|
9
|
+
* generation of pair-wise Trainingdata for OCR-Backends
|
|
10
|
+
|
|
11
|
+
## Requirements
|
|
12
|
+
|
|
13
|
+
* recent *nix-OS
|
|
14
|
+
* Python3.10+ Environment
|
|
15
|
+
|
|
16
|
+
## Usage
|
|
17
|
+
|
|
18
|
+
Each section contains detailed usage help instructions:
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
# evaluation
|
|
22
|
+
ocr eval --help
|
|
23
|
+
|
|
24
|
+
# corpus management
|
|
25
|
+
ocr corpus --help
|
|
26
|
+
|
|
27
|
+
# slice image by image + input OCR
|
|
28
|
+
ocr slice --help
|
|
29
|
+
|
|
30
|
+
# render image + input OCR
|
|
31
|
+
ocr show --help
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
### Data problems
|
|
35
|
+
|
|
36
|
+
Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
|
|
37
|
+
|
|
38
|
+
_Please note_:
|
|
39
|
+
Invalid data files are tried(!) to be excluded from evaluation.
|
|
40
|
+
|
|
41
|
+
### Evaluation Filter-Then-Aggregate
|
|
42
|
+
|
|
43
|
+
The evaluation CLI supports a single pre-aggregation filter using metadata extractors.
|
|
44
|
+
|
|
45
|
+
Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
ocr eval <candidates> \
|
|
49
|
+
--reference <groundtruth> \
|
|
50
|
+
--mets-file <mets.xml> \
|
|
51
|
+
--filter-by "mods:language=ger" \
|
|
52
|
+
--aggregate-by "mods:dateIssued:century"
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Multi-language filter values are interpreted as sets:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
ocr eval <candidates> \
|
|
59
|
+
--reference <groundtruth> \
|
|
60
|
+
--mets-file <mets.xml> \
|
|
61
|
+
--filter-by "mods:language=ger+eng" \
|
|
62
|
+
--aggregate-by "mods:dateIssued:century"
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Behavior:
|
|
66
|
+
* single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
|
|
67
|
+
* multi-value filter -> all filter values must be present in any order
|
|
68
|
+
* entries missing the filter criterion are reported as WARNING and discarded
|
|
69
|
+
|
|
70
|
+
## Development
|
|
71
|
+
|
|
72
|
+
Plattform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
# clone local
|
|
76
|
+
git clone <repository-url> <local-dir>
|
|
77
|
+
cd <local-dir>
|
|
78
|
+
|
|
79
|
+
# enable virtual python 3 environment (linux)
|
|
80
|
+
# and update pip itself
|
|
81
|
+
python3.10 -m venv venv
|
|
82
|
+
. venv/bin/activate
|
|
83
|
+
python -m pip install -U pip
|
|
84
|
+
|
|
85
|
+
# install with dev dependencies
|
|
86
|
+
python -m pip install -e ".[dev,test]"
|
|
87
|
+
|
|
88
|
+
# run tests with coverage
|
|
89
|
+
python -m pytest --cov=src
|
|
90
|
+
|
|
91
|
+
# run tests faster (parallel, auto worker count)
|
|
92
|
+
python -m pytest -q -n auto
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Contribution
|
|
96
|
+
|
|
97
|
+
Contributions, suggestions and proposals welcome!
|
|
98
|
+
|
|
99
|
+
## License
|
|
100
|
+
|
|
101
|
+
Under terms of the [MIT license](https://opensource.org/licenses/MIT).
|
|
102
|
+
|
|
103
|
+
**NOTE**: This software depends on packages that _might_ be licensed under different terms.
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "ocr-util"
|
|
3
|
+
dynamic = ["version"]
|
|
4
|
+
description = "OCR Utils"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
authors = [{name = "Universitäts- und Landesbibliothek Sachsen-Anhalt",email = "development@bibliothek.uni-halle.de"}]
|
|
8
|
+
classifiers = [
|
|
9
|
+
"Programming Language :: Python :: 3",
|
|
10
|
+
"License :: OSI Approved :: MIT License"
|
|
11
|
+
]
|
|
12
|
+
dependencies = [
|
|
13
|
+
"rapidfuzz>3",
|
|
14
|
+
"nltk",
|
|
15
|
+
"requests",
|
|
16
|
+
"docker",
|
|
17
|
+
"numpy",
|
|
18
|
+
"shapely",
|
|
19
|
+
"lxml",
|
|
20
|
+
"opencv-python-headless",
|
|
21
|
+
"exifread",
|
|
22
|
+
"Pillow"
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
[project.optional-dependencies]
|
|
26
|
+
dev = [
|
|
27
|
+
"black",
|
|
28
|
+
"pylint",
|
|
29
|
+
]
|
|
30
|
+
test = [
|
|
31
|
+
"pytest-cov",
|
|
32
|
+
"coverage-badge",
|
|
33
|
+
"pytest-xdist",
|
|
34
|
+
"flake8",
|
|
35
|
+
"lxml-stubs",
|
|
36
|
+
"opencv-stubs",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
[project.urls]
|
|
40
|
+
Homepage = "https://github.com/ulb-sachsen-anhalt/ocr-util"
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
ocr-util = "ocr_util.cli:start"
|
|
44
|
+
|
|
45
|
+
[build-system]
|
|
46
|
+
requires = ["setuptools>=61.0.0", "wheel"]
|
|
47
|
+
build-backend = "setuptools.build_meta"
|
|
48
|
+
|
|
49
|
+
[tool.setuptools.dynamic]
|
|
50
|
+
version = {attr = "ocr_util.__version__"}
|
|
51
|
+
|
|
52
|
+
[tool.setuptools.packages.find]
|
|
53
|
+
where = ["src"]
|
|
54
|
+
|
|
55
|
+
[tool.setuptools.package-data]
|
|
56
|
+
ocr_util = ["corpus/*.xml"]
|
|
57
|
+
|
|
58
|
+
[tool.setuptools]
|
|
59
|
+
package-dir = {"" = "src"}
|
|
60
|
+
|
|
61
|
+
[tool.black]
|
|
62
|
+
line-length = 120
|
|
63
|
+
target-version = ["py310", "py311", "py312"]
|
|
64
|
+
include = '\.pyi?$'
|
|
65
|
+
|
|
66
|
+
[tool.pylint.main]
|
|
67
|
+
extension-pkg-allow-list=["lxml.etree"]
|
|
68
|
+
|
|
69
|
+
[tool.pylint.messages_control]
|
|
70
|
+
reportAttributeAccessIssue = "none"
|
|
71
|
+
|
|
72
|
+
[tool.pytest.ini_options]
|
|
73
|
+
testpaths = ["tests"]
|
|
74
|
+
python_files = ["test_*.py"]
|
|
75
|
+
python_classes = ["Test*"]
|
|
76
|
+
python_functions = ["test_*"]
|
|
77
|
+
addopts = "-v --tb=short --cov-report=xml"
|
|
78
|
+
|
|
79
|
+
[tool.coverage.run]
|
|
80
|
+
source = ["ocr_util"]
|
|
81
|
+
omit = ["*/tests/*", "*/__pycache__/*", "*/venv/*", "*/build/*"]
|
ocr_util-2.0.1/setup.cfg
ADDED
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""OCR Utils"""
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
from pathlib import Path, PurePath
|
|
9
|
+
|
|
10
|
+
import ocr_util
|
|
11
|
+
import ocr_util.eval.model as do
|
|
12
|
+
import ocr_util.eval.model.filter as dofi
|
|
13
|
+
import ocr_util.eval.cli as eval_cli
|
|
14
|
+
import ocr_util.slice.cli as slice_cli
|
|
15
|
+
import ocr_util.show.cli as show_cli
|
|
16
|
+
import ocr_util.corpus.generate_corpus as gc
|
|
17
|
+
|
|
18
|
+
from ocr_util.corpus.common import CorpusArgs
|
|
19
|
+
|
|
20
|
+
# script constants
|
|
21
|
+
DEFAULT_VERBOSITY = 0
|
|
22
|
+
SUB_CMD_FRAME = "frame"
|
|
23
|
+
SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
|
|
24
|
+
CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
|
|
25
|
+
CORPUS_CACHE_DIR = os.path.join(
|
|
26
|
+
os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
SUB_CMD_EVALUATE = "eval"
|
|
30
|
+
SUB_CMD_SLICE = "slice"
|
|
31
|
+
SUB_CMD_SHOW = "show"
|
|
32
|
+
|
|
33
|
+
# Remove this constant as it's now managed by show_cli
|
|
34
|
+
# SUB_CMD_SHOW is kept for backward compatibility with other references
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def points_type(points: str) -> str:
|
|
38
|
+
match: re.Match = re.match(dofi.PolygonFrameFilterUtil.POINT_LIST_PATTERN, points)
|
|
39
|
+
if not match:
|
|
40
|
+
raise argparse.ArgumentTypeError(f"Invalid point coordinates: '{points}'")
|
|
41
|
+
return points
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def start() -> None:
|
|
45
|
+
# Configure logging once, centrally
|
|
46
|
+
logging.basicConfig(
|
|
47
|
+
level=logging.INFO,
|
|
48
|
+
format='%(asctime)s [%(levelname)s][%(name)s] %(message)s',
|
|
49
|
+
datefmt='%Y-%m-%d %H:%M:%S'
|
|
50
|
+
)
|
|
51
|
+
arg_parser: argparse.ArgumentParser = argparse.ArgumentParser(
|
|
52
|
+
prog="ocr-util",
|
|
53
|
+
description=f"OCR Util {ocr_util.__version__} of ULB Sachsen-Anhalt",
|
|
54
|
+
)
|
|
55
|
+
sub_arg_parsers = arg_parser.add_subparsers(
|
|
56
|
+
title="Subkommandos",
|
|
57
|
+
dest="subcommand",
|
|
58
|
+
required=True,
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# frames subcommand
|
|
62
|
+
frame_arg_parser = sub_arg_parsers.add_parser(
|
|
63
|
+
SUB_CMD_FRAME,
|
|
64
|
+
help="Filter Contents of provided ALTO-v3-Data by provided Coordinates, where Coordinates span a rectangular"
|
|
65
|
+
" box with",
|
|
66
|
+
)
|
|
67
|
+
frame_arg_parser.add_argument(
|
|
68
|
+
"-v",
|
|
69
|
+
"--verbosity",
|
|
70
|
+
action="count",
|
|
71
|
+
default=DEFAULT_VERBOSITY,
|
|
72
|
+
required=False,
|
|
73
|
+
help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
|
|
74
|
+
)
|
|
75
|
+
frame_arg_parser.add_argument(
|
|
76
|
+
"-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True
|
|
77
|
+
)
|
|
78
|
+
frame_arg_parser.add_argument(
|
|
79
|
+
"-o",
|
|
80
|
+
"--output-ocr-file",
|
|
81
|
+
help="Path of resulting OCR-Data file",
|
|
82
|
+
required=False,
|
|
83
|
+
default=None,
|
|
84
|
+
)
|
|
85
|
+
frame_arg_parser.add_argument(
|
|
86
|
+
"-p",
|
|
87
|
+
"--points",
|
|
88
|
+
required=True,
|
|
89
|
+
type=points_type,
|
|
90
|
+
help="""
|
|
91
|
+
Frame to slice words/lines/regions from input OCR-Data
|
|
92
|
+
f.e.: --frame "2892,2480 5072,2480 5072,5148 2892,5148"
|
|
93
|
+
""",
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
# groundtruth-corpus subcommand
|
|
97
|
+
groundtruth_corpus_arg_parser = sub_arg_parsers.add_parser(
|
|
98
|
+
SUB_CMD_GROUNDTRUTH_CORPUS,
|
|
99
|
+
help="Create METS file from N ground truth PAGE-XML files with URN identifiers",
|
|
100
|
+
)
|
|
101
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
102
|
+
"-i",
|
|
103
|
+
"--input",
|
|
104
|
+
dest="input_dir",
|
|
105
|
+
help="Path to the input directory containing GT PAGE-XML files",
|
|
106
|
+
required=True,
|
|
107
|
+
)
|
|
108
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
109
|
+
"-o",
|
|
110
|
+
"--output",
|
|
111
|
+
dest="output_dir",
|
|
112
|
+
help="Path to the output directory for generated corpus",
|
|
113
|
+
required=True,
|
|
114
|
+
)
|
|
115
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
116
|
+
"-l",
|
|
117
|
+
"--limit",
|
|
118
|
+
type=int,
|
|
119
|
+
default=0,
|
|
120
|
+
help="Number of files to process (default: 0 = unlimited)",
|
|
121
|
+
required=False,
|
|
122
|
+
)
|
|
123
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
124
|
+
"-t",
|
|
125
|
+
"--temp-dir",
|
|
126
|
+
dest="temp_dir",
|
|
127
|
+
default=CORPUS_CACHE_DIR,
|
|
128
|
+
help=f"Path to temporary directory for caching METS files (default: {CORPUS_CACHE_DIR})",
|
|
129
|
+
required=False,
|
|
130
|
+
)
|
|
131
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
132
|
+
"-v",
|
|
133
|
+
"--verbosity",
|
|
134
|
+
action="count",
|
|
135
|
+
default=DEFAULT_VERBOSITY,
|
|
136
|
+
required=False,
|
|
137
|
+
help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
|
|
138
|
+
)
|
|
139
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
140
|
+
"--corpus-label",
|
|
141
|
+
dest="corpus_label",
|
|
142
|
+
default="Ground Truth Corpus",
|
|
143
|
+
help="Label for the corpus in the METS logical structure (default: 'Ground Truth Corpus')",
|
|
144
|
+
required=False,
|
|
145
|
+
)
|
|
146
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
147
|
+
"--oai-base-url",
|
|
148
|
+
dest="oai_base_url",
|
|
149
|
+
help="Base URL for OAI-PMH requests",
|
|
150
|
+
required=False,
|
|
151
|
+
)
|
|
152
|
+
groundtruth_corpus_arg_parser.add_argument(
|
|
153
|
+
"--clear-cache",
|
|
154
|
+
dest="clear_cache",
|
|
155
|
+
action="store_true",
|
|
156
|
+
default=False,
|
|
157
|
+
help="Clear the local cache directory before processing (default: False)",
|
|
158
|
+
required=False,
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
# evaluate subcommand
|
|
162
|
+
evaluate_arg_parser = sub_arg_parsers.add_parser(
|
|
163
|
+
SUB_CMD_EVALUATE,
|
|
164
|
+
help="Evaluate OCR candidates against ground truth data",
|
|
165
|
+
add_help=True,
|
|
166
|
+
)
|
|
167
|
+
eval_cli.register_arguments(evaluate_arg_parser)
|
|
168
|
+
|
|
169
|
+
# slice subcommand
|
|
170
|
+
slice_arg_parser = sub_arg_parsers.add_parser(
|
|
171
|
+
SUB_CMD_SLICE,
|
|
172
|
+
help="Generate pairs of textlines and image frames from OCR and image data",
|
|
173
|
+
add_help=True,
|
|
174
|
+
)
|
|
175
|
+
slice_arg_parser.add_argument(
|
|
176
|
+
"data",
|
|
177
|
+
type=str,
|
|
178
|
+
help="path to local alto|page file corresponding to image",
|
|
179
|
+
)
|
|
180
|
+
slice_arg_parser.add_argument(
|
|
181
|
+
"-i",
|
|
182
|
+
"--image",
|
|
183
|
+
required=True,
|
|
184
|
+
help="path to local image file tif|jpg|png corresponding to ocr",
|
|
185
|
+
)
|
|
186
|
+
slice_arg_parser.add_argument(
|
|
187
|
+
"-o",
|
|
188
|
+
"--output_dir",
|
|
189
|
+
default=slice_cli.DEFAULT_OUTDIR_PREFIX,
|
|
190
|
+
help=f"output directory, re-created if already exists (default: {slice_cli.DEFAULT_OUTDIR_PREFIX})",
|
|
191
|
+
)
|
|
192
|
+
slice_arg_parser.add_argument(
|
|
193
|
+
"--prefix-output",
|
|
194
|
+
required=False,
|
|
195
|
+
help="optional: prefix each pair using this arg (default: '')",
|
|
196
|
+
)
|
|
197
|
+
slice_arg_parser.add_argument(
|
|
198
|
+
"-m",
|
|
199
|
+
"--minchars",
|
|
200
|
+
required=False,
|
|
201
|
+
type=int,
|
|
202
|
+
default=int(slice_cli.DEFAULT_MIN_CHARS),
|
|
203
|
+
help=f"optional: minimum printable chars required for a line to be included into set (default: {slice_cli.DEFAULT_MIN_CHARS})",
|
|
204
|
+
)
|
|
205
|
+
slice_arg_parser.add_argument(
|
|
206
|
+
"-s",
|
|
207
|
+
"--summary",
|
|
208
|
+
required=False,
|
|
209
|
+
action="store_true",
|
|
210
|
+
default=slice_cli.DEFAULT_USE_SUMMARY,
|
|
211
|
+
help=f"optional: print all lines in additional file (default: {slice_cli.DEFAULT_USE_SUMMARY})",
|
|
212
|
+
)
|
|
213
|
+
slice_arg_parser.add_argument(
|
|
214
|
+
"-r",
|
|
215
|
+
"--reorder",
|
|
216
|
+
required=False,
|
|
217
|
+
action="store_true",
|
|
218
|
+
default=slice_cli.DEFAULT_USE_REORDER,
|
|
219
|
+
help=f"optional: re-order word tokens from right-to-left (default: {slice_cli.DEFAULT_USE_REORDER})",
|
|
220
|
+
)
|
|
221
|
+
slice_arg_parser.add_argument(
|
|
222
|
+
"--binarize",
|
|
223
|
+
required=False,
|
|
224
|
+
action="store_true",
|
|
225
|
+
default=slice_cli.DEFAULT_BINARIZE,
|
|
226
|
+
help=f"optional: binarize textline images (default: {slice_cli.DEFAULT_BINARIZE})",
|
|
227
|
+
)
|
|
228
|
+
slice_arg_parser.add_argument(
|
|
229
|
+
"--sanitize",
|
|
230
|
+
required=False,
|
|
231
|
+
type=bool,
|
|
232
|
+
default=slice_cli.DEFAULT_SANITIZE,
|
|
233
|
+
help=f"optional: sanitize textline images (default: {slice_cli.DEFAULT_SANITIZE})",
|
|
234
|
+
)
|
|
235
|
+
slice_arg_parser.add_argument(
|
|
236
|
+
"--no-sanitize", dest="sanitize", action="store_false"
|
|
237
|
+
)
|
|
238
|
+
slice_arg_parser.add_argument(
|
|
239
|
+
"--intrusion-ratio",
|
|
240
|
+
required=False,
|
|
241
|
+
default=slice_cli.DEFAULT_INTRUSION_RATIO,
|
|
242
|
+
help=f"optional: alter threshold for top and bottom ratios for intrusion detection for sanitizing (default: {slice_cli.DEFAULT_INTRUSION_RATIO})",
|
|
243
|
+
)
|
|
244
|
+
slice_arg_parser.add_argument(
|
|
245
|
+
"--rotation-threshold",
|
|
246
|
+
required=False,
|
|
247
|
+
type=float,
|
|
248
|
+
default=slice_cli.DEFAULT_ROTATION_THRESH,
|
|
249
|
+
help=f"optional: alter threshold for rotation of textline image (default: {slice_cli.DEFAULT_ROTATION_THRESH})",
|
|
250
|
+
)
|
|
251
|
+
slice_arg_parser.add_argument(
|
|
252
|
+
"-p",
|
|
253
|
+
"--padding",
|
|
254
|
+
required=False,
|
|
255
|
+
type=int,
|
|
256
|
+
default=slice_cli.DEFAULT_PADDING,
|
|
257
|
+
help=f"optional: additional padding for existing textline image (default: {slice_cli.DEFAULT_PADDING})",
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
# show subcommand
|
|
261
|
+
show_cli.register_arguments(sub_arg_parsers)
|
|
262
|
+
|
|
263
|
+
args = arg_parser.parse_args()
|
|
264
|
+
|
|
265
|
+
verbosity: int = getattr(args, "verbosity", DEFAULT_VERBOSITY)
|
|
266
|
+
|
|
267
|
+
if args.subcommand == SUB_CMD_FRAME:
|
|
268
|
+
input_ocr_file: str = args.input_ocr_file
|
|
269
|
+
output_ocr_file: str = args.output_ocr_file
|
|
270
|
+
points: str = args.points
|
|
271
|
+
if verbosity > 1:
|
|
272
|
+
print(
|
|
273
|
+
f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}"
|
|
274
|
+
)
|
|
275
|
+
polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(
|
|
276
|
+
input_ocr_file, points, verbosity
|
|
277
|
+
)
|
|
278
|
+
piece_result: do.DigitalObjectTree = polygon_frame_filter.process()
|
|
279
|
+
file_result: PurePath = do.from_digital_object(piece_result, output_ocr_file)
|
|
280
|
+
if verbosity > 0:
|
|
281
|
+
print("[INFO ] file_result", file_result)
|
|
282
|
+
|
|
283
|
+
elif args.subcommand == SUB_CMD_GROUNDTRUTH_CORPUS:
|
|
284
|
+
corpus_args = CorpusArgs(
|
|
285
|
+
input_dir=Path(args.input_dir).absolute(),
|
|
286
|
+
output_dir=Path(args.output_dir).absolute(),
|
|
287
|
+
local_cache_dir=Path(args.temp_dir).absolute(),
|
|
288
|
+
limit=int(args.limit),
|
|
289
|
+
corpus_label=args.corpus_label,
|
|
290
|
+
clear_cache=args.clear_cache
|
|
291
|
+
)
|
|
292
|
+
gc.generate(corpus_args)
|
|
293
|
+
|
|
294
|
+
elif args.subcommand == SUB_CMD_EVALUATE:
|
|
295
|
+
eval_args = vars(args)
|
|
296
|
+
eval_args.pop("subcommand", None)
|
|
297
|
+
eval_cli.start_evaluation(eval_args)
|
|
298
|
+
|
|
299
|
+
elif args.subcommand == SUB_CMD_SLICE:
|
|
300
|
+
slice_args = vars(args)
|
|
301
|
+
slice_args.pop("subcommand", None)
|
|
302
|
+
slice_cli.start_slice(slice_args)
|
|
303
|
+
|
|
304
|
+
elif args.subcommand == SUB_CMD_SHOW:
|
|
305
|
+
show_args = vars(args)
|
|
306
|
+
show_args.pop("subcommand", None)
|
|
307
|
+
show_cli.start_show(show_args)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
if __name__ == "__main__":
|
|
311
|
+
start()
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Public API for the OCR ground truth corpus generation package.
|
|
2
|
+
|
|
3
|
+
Import :func:`~ocr_util.corpus.generate_corpus.generate` and
|
|
4
|
+
:class:`~ocr_util.corpus.common.CorpusArgs` to create a METS-based corpus
|
|
5
|
+
from a directory of PAGE-XML ground truth files::
|
|
6
|
+
|
|
7
|
+
from ocr_util.corpus.generate_corpus import generate
|
|
8
|
+
from ocr_util.corpus.common import CorpusArgs
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
result = generate(CorpusArgs(
|
|
12
|
+
input_dir=Path("gt/"),
|
|
13
|
+
output_dir=Path("corpus/"),
|
|
14
|
+
local_cache_dir=Path("/tmp/mets_cache"),
|
|
15
|
+
))
|
|
16
|
+
print(result.file_path, result.n_pages)
|
|
17
|
+
"""
|