ocr-util 2.2.1__tar.gz → 3.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ocr_util-2.2.1/src/ocr_util.egg-info → ocr_util-3.0.1}/PKG-INFO +22 -39
- {ocr_util-2.2.1 → ocr_util-3.0.1}/README.md +20 -37
- {ocr_util-2.2.1 → ocr_util-3.0.1}/pyproject.toml +2 -2
- ocr_util-3.0.1/src/ocr_util/__init__.py +3 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/cli.py +1 -99
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/aggregation.py +2 -2
- {ocr_util-2.2.1 → ocr_util-3.0.1/src/ocr_util.egg-info}/PKG-INFO +22 -39
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/SOURCES.txt +0 -11
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_aggregation.py +19 -3
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_eval_cli.py +83 -9
- ocr_util-2.2.1/src/ocr_util/__init__.py +0 -3
- ocr_util-2.2.1/src/ocr_util/corpus/__init__.py +0 -17
- ocr_util-2.2.1/src/ocr_util/corpus/analyse.py +0 -173
- ocr_util-2.2.1/src/ocr_util/corpus/common.py +0 -520
- ocr_util-2.2.1/src/ocr_util/corpus/generate_corpus.py +0 -149
- ocr_util-2.2.1/src/ocr_util/corpus/load_metadata.py +0 -324
- ocr_util-2.2.1/src/ocr_util/corpus/template.corpus.xml +0 -24
- ocr_util-2.2.1/tests/test_corpus.py +0 -515
- ocr_util-2.2.1/tests/test_corpus_analyse.py +0 -108
- ocr_util-2.2.1/tests/test_corpus_examples.py +0 -304
- ocr_util-2.2.1/tests/test_corpus_fulltext_generation.py +0 -144
- ocr_util-2.2.1/tests/test_corpus_load_metadata.py +0 -179
- {ocr_util-2.2.1 → ocr_util-3.0.1}/LICENSE +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/setup.cfg +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/__init__.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/cli.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/constants.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/__init__.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/common.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/Util.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/common.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/evaluation.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/geometry.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/metrics.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/common.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/digital_object_model.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/digital_object_util.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/filter.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/format_alto_v3_util.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/format_page_util.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/main.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/minidom_util.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/preprocessing.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/resolve.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/show/cli.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/show/ocr_show_segmentation.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/__init__.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/cli.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/gts_pairs.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/pairs_lstmfs.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/dependency_links.txt +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/entry_points.txt +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/requires.txt +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/top_level.txt +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_model.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_alto.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_filter.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_page.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_page_eynollah.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_txt.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_generate_sets.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_evaluate.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_metrics.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_metrics_base.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_preprocessing.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_page_reading_order.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_show_segmentation.py +0 -0
- {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_version_compatibility.py +0 -0
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ocr-util
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.0.1
|
|
4
4
|
Summary: OCR Utils
|
|
5
5
|
Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
|
|
6
|
+
License-Expression: MIT
|
|
6
7
|
Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
|
|
7
8
|
Classifier: Programming Language :: Python :: 3
|
|
8
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
9
9
|
Requires-Python: >=3.10
|
|
10
10
|
Description-Content-Type: text/markdown
|
|
11
11
|
License-File: LICENSE
|
|
@@ -38,7 +38,6 @@ Dynamic: license-file
|
|
|
38
38
|
|
|
39
39
|
Collection of utils for
|
|
40
40
|
* evaluation of OCR data for the masses
|
|
41
|
-
* generation of extended OCR-Evaluation Corpora
|
|
42
41
|
* generation of pair-wise Trainingdata for OCR-Backends
|
|
43
42
|
|
|
44
43
|
## Requirements
|
|
@@ -54,16 +53,10 @@ Each section contains detailed usage help instructions:
|
|
|
54
53
|
# evaluation
|
|
55
54
|
ocr-util eval --help
|
|
56
55
|
|
|
57
|
-
# corpus management
|
|
58
|
-
ocr-util corpus --help
|
|
59
|
-
|
|
60
|
-
# analyse an existing METS corpus
|
|
61
|
-
ocr-util corpus-analyse --help
|
|
62
|
-
|
|
63
56
|
# slice image by image + input OCR
|
|
64
57
|
ocr-util slice --help
|
|
65
58
|
|
|
66
|
-
# render
|
|
59
|
+
# render input OCR (regions, lines, words) on given image
|
|
67
60
|
ocr-util show --help
|
|
68
61
|
```
|
|
69
62
|
|
|
@@ -73,7 +66,7 @@ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONT
|
|
|
73
66
|
|
|
74
67
|
_Please note_:
|
|
75
68
|
Invalid data files are excluded and reported where possible from evaluation.
|
|
76
|
-
The term 'invalid' refers to
|
|
69
|
+
The term 'invalid' refers to malformed or structurally invalid XML, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
|
|
77
70
|
|
|
78
71
|
### Evaluation Filter-Then-Aggregate
|
|
79
72
|
|
|
@@ -89,6 +82,22 @@ ocr-util eval <candidates> \
|
|
|
89
82
|
--aggregate-by "mods:dateIssued:century"
|
|
90
83
|
```
|
|
91
84
|
|
|
85
|
+
The dimensions can also be reversed: filter by publication century or decade,
|
|
86
|
+
then aggregate by language. For example, keep only publications from the 17th
|
|
87
|
+
century (1601-1700 inclusive), then group their evaluation results by MODS language:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
ocr-util eval <candidates> \
|
|
91
|
+
--reference <groundtruth> \
|
|
92
|
+
--mets-file <mets.xml> \
|
|
93
|
+
--filter-by "mods:dateIssued:century=17th" \
|
|
94
|
+
--aggregate-by "mods:language"
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Filtering still happens before evaluation and aggregation; only the metadata
|
|
98
|
+
dimensions exchange roles. To select a decade instead, replace the filter above
|
|
99
|
+
with `--filter-by "mods:dateIssued:decade=1800s"` (1800-1809 inclusive).
|
|
100
|
+
|
|
92
101
|
Multi-language filter values are interpreted as sets:
|
|
93
102
|
|
|
94
103
|
```bash
|
|
@@ -100,6 +109,8 @@ ocr-util eval <candidates> \
|
|
|
100
109
|
```
|
|
101
110
|
|
|
102
111
|
Behavior:
|
|
112
|
+
* centuries follow calendar numbering: the 18th century is 1701-1800,
|
|
113
|
+
and the 19th century is 1801-1900; decades retain ten-year buckets such as 1800-1809
|
|
103
114
|
* single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
|
|
104
115
|
* multi-value filter -> all filter values must be present in any order
|
|
105
116
|
* entries missing the filter criterion are reported as WARNING and discarded
|
|
@@ -108,34 +119,6 @@ Behavior:
|
|
|
108
119
|
* aggregation coverage reports how many evaluated pairs provide each dimension;
|
|
109
120
|
reported `items` count candidate/ground-truth pairs, not MODS elements
|
|
110
121
|
|
|
111
|
-
### Corpus Analysis
|
|
112
|
-
|
|
113
|
-
Existing METS corpora can be filtered without evaluation candidate data. Matching
|
|
114
|
-
full-text file references are printed one per line:
|
|
115
|
-
|
|
116
|
-
```bash
|
|
117
|
-
ocr-util corpus-analyse <mets.xml> \
|
|
118
|
-
--filter-by "mods:dateIssued:century=16th"
|
|
119
|
-
```
|
|
120
|
-
|
|
121
|
-
Repeat `--filter-by` to combine criteria with AND:
|
|
122
|
-
|
|
123
|
-
```bash
|
|
124
|
-
ocr-util corpus-analyse <mets.xml> \
|
|
125
|
-
--filter-by "mods:dateIssued:century=16th" \
|
|
126
|
-
--filter-by "mods:language=ger"
|
|
127
|
-
```
|
|
128
|
-
|
|
129
|
-
Check that every local file referenced by any METS `FLocat` exists:
|
|
130
|
-
|
|
131
|
-
```bash
|
|
132
|
-
ocr-util corpus-analyse <mets.xml> --check
|
|
133
|
-
```
|
|
134
|
-
|
|
135
|
-
Relative paths are resolved from the METS directory. Missing files and remote
|
|
136
|
-
references that cannot be checked locally produce a non-zero exit status. This
|
|
137
|
-
is a file-presence check; it does not perform XML schema validation.
|
|
138
|
-
|
|
139
122
|
## Development
|
|
140
123
|
|
|
141
124
|
Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
|
|
@@ -5,7 +5,6 @@
|
|
|
5
5
|
|
|
6
6
|
Collection of utils for
|
|
7
7
|
* evaluation of OCR data for the masses
|
|
8
|
-
* generation of extended OCR-Evaluation Corpora
|
|
9
8
|
* generation of pair-wise Trainingdata for OCR-Backends
|
|
10
9
|
|
|
11
10
|
## Requirements
|
|
@@ -21,16 +20,10 @@ Each section contains detailed usage help instructions:
|
|
|
21
20
|
# evaluation
|
|
22
21
|
ocr-util eval --help
|
|
23
22
|
|
|
24
|
-
# corpus management
|
|
25
|
-
ocr-util corpus --help
|
|
26
|
-
|
|
27
|
-
# analyse an existing METS corpus
|
|
28
|
-
ocr-util corpus-analyse --help
|
|
29
|
-
|
|
30
23
|
# slice image by image + input OCR
|
|
31
24
|
ocr-util slice --help
|
|
32
25
|
|
|
33
|
-
# render
|
|
26
|
+
# render input OCR (regions, lines, words) on given image
|
|
34
27
|
ocr-util show --help
|
|
35
28
|
```
|
|
36
29
|
|
|
@@ -40,7 +33,7 @@ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONT
|
|
|
40
33
|
|
|
41
34
|
_Please note_:
|
|
42
35
|
Invalid data files are excluded and reported where possible from evaluation.
|
|
43
|
-
The term 'invalid' refers to
|
|
36
|
+
The term 'invalid' refers to malformed or structurally invalid XML, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
|
|
44
37
|
|
|
45
38
|
### Evaluation Filter-Then-Aggregate
|
|
46
39
|
|
|
@@ -56,6 +49,22 @@ ocr-util eval <candidates> \
|
|
|
56
49
|
--aggregate-by "mods:dateIssued:century"
|
|
57
50
|
```
|
|
58
51
|
|
|
52
|
+
The dimensions can also be reversed: filter by publication century or decade,
|
|
53
|
+
then aggregate by language. For example, keep only publications from the 17th
|
|
54
|
+
century (1601-1700 inclusive), then group their evaluation results by MODS language:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
ocr-util eval <candidates> \
|
|
58
|
+
--reference <groundtruth> \
|
|
59
|
+
--mets-file <mets.xml> \
|
|
60
|
+
--filter-by "mods:dateIssued:century=17th" \
|
|
61
|
+
--aggregate-by "mods:language"
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Filtering still happens before evaluation and aggregation; only the metadata
|
|
65
|
+
dimensions exchange roles. To select a decade instead, replace the filter above
|
|
66
|
+
with `--filter-by "mods:dateIssued:decade=1800s"` (1800-1809 inclusive).
|
|
67
|
+
|
|
59
68
|
Multi-language filter values are interpreted as sets:
|
|
60
69
|
|
|
61
70
|
```bash
|
|
@@ -67,6 +76,8 @@ ocr-util eval <candidates> \
|
|
|
67
76
|
```
|
|
68
77
|
|
|
69
78
|
Behavior:
|
|
79
|
+
* centuries follow calendar numbering: the 18th century is 1701-1800,
|
|
80
|
+
and the 19th century is 1801-1900; decades retain ten-year buckets such as 1800-1809
|
|
70
81
|
* single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
|
|
71
82
|
* multi-value filter -> all filter values must be present in any order
|
|
72
83
|
* entries missing the filter criterion are reported as WARNING and discarded
|
|
@@ -75,34 +86,6 @@ Behavior:
|
|
|
75
86
|
* aggregation coverage reports how many evaluated pairs provide each dimension;
|
|
76
87
|
reported `items` count candidate/ground-truth pairs, not MODS elements
|
|
77
88
|
|
|
78
|
-
### Corpus Analysis
|
|
79
|
-
|
|
80
|
-
Existing METS corpora can be filtered without evaluation candidate data. Matching
|
|
81
|
-
full-text file references are printed one per line:
|
|
82
|
-
|
|
83
|
-
```bash
|
|
84
|
-
ocr-util corpus-analyse <mets.xml> \
|
|
85
|
-
--filter-by "mods:dateIssued:century=16th"
|
|
86
|
-
```
|
|
87
|
-
|
|
88
|
-
Repeat `--filter-by` to combine criteria with AND:
|
|
89
|
-
|
|
90
|
-
```bash
|
|
91
|
-
ocr-util corpus-analyse <mets.xml> \
|
|
92
|
-
--filter-by "mods:dateIssued:century=16th" \
|
|
93
|
-
--filter-by "mods:language=ger"
|
|
94
|
-
```
|
|
95
|
-
|
|
96
|
-
Check that every local file referenced by any METS `FLocat` exists:
|
|
97
|
-
|
|
98
|
-
```bash
|
|
99
|
-
ocr-util corpus-analyse <mets.xml> --check
|
|
100
|
-
```
|
|
101
|
-
|
|
102
|
-
Relative paths are resolved from the METS directory. Missing files and remote
|
|
103
|
-
references that cannot be checked locally produce a non-zero exit status. This
|
|
104
|
-
is a file-presence check; it does not perform XML schema validation.
|
|
105
|
-
|
|
106
89
|
## Development
|
|
107
90
|
|
|
108
91
|
Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
|
|
@@ -7,8 +7,8 @@ requires-python = ">=3.10"
|
|
|
7
7
|
authors = [{name = "Universitäts- und Landesbibliothek Sachsen-Anhalt",email = "development@bibliothek.uni-halle.de"}]
|
|
8
8
|
classifiers = [
|
|
9
9
|
"Programming Language :: Python :: 3",
|
|
10
|
-
"License :: OSI Approved :: MIT License"
|
|
11
10
|
]
|
|
11
|
+
license = "MIT"
|
|
12
12
|
dependencies = [
|
|
13
13
|
"rapidfuzz>3",
|
|
14
14
|
"nltk",
|
|
@@ -53,7 +53,7 @@ version = {attr = "ocr_util.__version__"}
|
|
|
53
53
|
where = ["src"]
|
|
54
54
|
|
|
55
55
|
[tool.setuptools.package-data]
|
|
56
|
-
|
|
56
|
+
ocr_corpus = ["*.xml"]
|
|
57
57
|
|
|
58
58
|
[tool.setuptools]
|
|
59
59
|
package-dir = {"" = "src"}
|
|
@@ -3,9 +3,8 @@
|
|
|
3
3
|
|
|
4
4
|
import argparse
|
|
5
5
|
import logging
|
|
6
|
-
import os
|
|
7
6
|
import re
|
|
8
|
-
from pathlib import
|
|
7
|
+
from pathlib import PurePath
|
|
9
8
|
|
|
10
9
|
import ocr_util
|
|
11
10
|
import ocr_util.eval.model as do
|
|
@@ -13,18 +12,10 @@ import ocr_util.eval.model.filter as dofi
|
|
|
13
12
|
import ocr_util.eval.cli as eval_cli
|
|
14
13
|
import ocr_util.slice.cli as slice_cli
|
|
15
14
|
import ocr_util.show.cli as show_cli
|
|
16
|
-
import ocr_util.corpus.analyse as corpus_analyse
|
|
17
|
-
import ocr_util.corpus.generate_corpus as gc
|
|
18
|
-
|
|
19
|
-
from ocr_util.corpus.common import CorpusArgs
|
|
20
15
|
|
|
21
16
|
# script constants
|
|
22
17
|
DEFAULT_VERBOSITY = 0
|
|
23
18
|
SUB_CMD_FRAME = "frame"
|
|
24
|
-
SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
|
|
25
|
-
SUB_CMD_CORPUS_ANALYSE = "corpus-analyse"
|
|
26
|
-
CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
|
|
27
|
-
CORPUS_CACHE_DIR = os.path.join(os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME)
|
|
28
19
|
|
|
29
20
|
SUB_CMD_EVALUATE = "eval"
|
|
30
21
|
SUB_CMD_SLICE = "slice"
|
|
@@ -89,77 +80,6 @@ def start() -> None:
|
|
|
89
80
|
""",
|
|
90
81
|
)
|
|
91
82
|
|
|
92
|
-
# groundtruth-corpus subcommand
|
|
93
|
-
groundtruth_corpus_arg_parser = sub_arg_parsers.add_parser(
|
|
94
|
-
SUB_CMD_GROUNDTRUTH_CORPUS,
|
|
95
|
-
help="Create METS file from N ground truth PAGE-XML files with URN identifiers",
|
|
96
|
-
)
|
|
97
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
98
|
-
"-i",
|
|
99
|
-
"--input",
|
|
100
|
-
dest="input_dir",
|
|
101
|
-
help="Path to the input directory containing GT PAGE-XML files",
|
|
102
|
-
required=True,
|
|
103
|
-
)
|
|
104
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
105
|
-
"-o",
|
|
106
|
-
"--output",
|
|
107
|
-
dest="output_dir",
|
|
108
|
-
help="Path to the output directory for generated corpus",
|
|
109
|
-
required=True,
|
|
110
|
-
)
|
|
111
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
112
|
-
"-l",
|
|
113
|
-
"--limit",
|
|
114
|
-
type=int,
|
|
115
|
-
default=0,
|
|
116
|
-
help="Number of files to process (default: 0 = unlimited)",
|
|
117
|
-
required=False,
|
|
118
|
-
)
|
|
119
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
120
|
-
"-t",
|
|
121
|
-
"--temp-dir",
|
|
122
|
-
dest="temp_dir",
|
|
123
|
-
default=CORPUS_CACHE_DIR,
|
|
124
|
-
help=f"Path to temporary directory for caching METS files (default: {CORPUS_CACHE_DIR})",
|
|
125
|
-
required=False,
|
|
126
|
-
)
|
|
127
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
128
|
-
"-v",
|
|
129
|
-
"--verbosity",
|
|
130
|
-
action="count",
|
|
131
|
-
default=DEFAULT_VERBOSITY,
|
|
132
|
-
required=False,
|
|
133
|
-
help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
|
|
134
|
-
)
|
|
135
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
136
|
-
"--corpus-label",
|
|
137
|
-
dest="corpus_label",
|
|
138
|
-
default="Ground Truth Corpus",
|
|
139
|
-
help="Label for the corpus in the METS logical structure (default: 'Ground Truth Corpus')",
|
|
140
|
-
required=False,
|
|
141
|
-
)
|
|
142
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
143
|
-
"--oai-base-url",
|
|
144
|
-
dest="oai_base_url",
|
|
145
|
-
help="Base URL for OAI-PMH requests",
|
|
146
|
-
required=False,
|
|
147
|
-
)
|
|
148
|
-
groundtruth_corpus_arg_parser.add_argument(
|
|
149
|
-
"--clear-cache",
|
|
150
|
-
dest="clear_cache",
|
|
151
|
-
action="store_true",
|
|
152
|
-
default=False,
|
|
153
|
-
help="Clear the local cache directory before processing (default: False)",
|
|
154
|
-
required=False,
|
|
155
|
-
)
|
|
156
|
-
|
|
157
|
-
corpus_analyse_parser = sub_arg_parsers.add_parser(
|
|
158
|
-
SUB_CMD_CORPUS_ANALYSE,
|
|
159
|
-
help="List files in an existing METS corpus that match metadata filters",
|
|
160
|
-
)
|
|
161
|
-
corpus_analyse.register_arguments(corpus_analyse_parser)
|
|
162
|
-
|
|
163
83
|
# evaluate subcommand
|
|
164
84
|
evaluate_arg_parser = sub_arg_parsers.add_parser(
|
|
165
85
|
SUB_CMD_EVALUATE,
|
|
@@ -276,24 +196,6 @@ def start() -> None:
|
|
|
276
196
|
if verbosity > 0:
|
|
277
197
|
print("[INFO ] file_result", file_result)
|
|
278
198
|
|
|
279
|
-
elif args.subcommand == SUB_CMD_GROUNDTRUTH_CORPUS:
|
|
280
|
-
corpus_args = CorpusArgs(
|
|
281
|
-
input_dir=Path(args.input_dir).absolute(),
|
|
282
|
-
output_dir=Path(args.output_dir).absolute(),
|
|
283
|
-
local_cache_dir=Path(args.temp_dir).absolute(),
|
|
284
|
-
limit=int(args.limit),
|
|
285
|
-
corpus_label=args.corpus_label,
|
|
286
|
-
clear_cache=args.clear_cache,
|
|
287
|
-
)
|
|
288
|
-
gc.generate(corpus_args)
|
|
289
|
-
|
|
290
|
-
elif args.subcommand == SUB_CMD_CORPUS_ANALYSE:
|
|
291
|
-
analyse_args = vars(args)
|
|
292
|
-
analyse_args.pop("subcommand", None)
|
|
293
|
-
result = corpus_analyse.start_analysis(analyse_args)
|
|
294
|
-
if isinstance(result, corpus_analyse.CorpusCheckResult) and not result.is_valid:
|
|
295
|
-
raise SystemExit(1)
|
|
296
|
-
|
|
297
199
|
elif args.subcommand == SUB_CMD_EVALUATE:
|
|
298
200
|
eval_args = vars(args)
|
|
299
201
|
eval_args.pop("subcommand", None)
|
|
@@ -142,10 +142,10 @@ def decade_transform(value: str) -> typing.Optional[str]:
|
|
|
142
142
|
|
|
143
143
|
|
|
144
144
|
def century_transform(value: str) -> typing.Optional[str]:
|
|
145
|
-
"""
|
|
145
|
+
"""Bucket years by calendar century: 1801 through 1900 belong to the 19th."""
|
|
146
146
|
try:
|
|
147
147
|
year = int(str(value).strip()[:4])
|
|
148
|
-
century = year // 100 + 1
|
|
148
|
+
century = (year - 1) // 100 + 1
|
|
149
149
|
suffix = {1: "st", 2: "nd", 3: "rd"}.get(century % 10 if century % 100 not in (11, 12, 13) else 0, "th")
|
|
150
150
|
return f"{century}{suffix}"
|
|
151
151
|
except (ValueError, TypeError):
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ocr-util
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.0.1
|
|
4
4
|
Summary: OCR Utils
|
|
5
5
|
Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
|
|
6
|
+
License-Expression: MIT
|
|
6
7
|
Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
|
|
7
8
|
Classifier: Programming Language :: Python :: 3
|
|
8
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
9
9
|
Requires-Python: >=3.10
|
|
10
10
|
Description-Content-Type: text/markdown
|
|
11
11
|
License-File: LICENSE
|
|
@@ -38,7 +38,6 @@ Dynamic: license-file
|
|
|
38
38
|
|
|
39
39
|
Collection of utils for
|
|
40
40
|
* evaluation of OCR data for the masses
|
|
41
|
-
* generation of extended OCR-Evaluation Corpora
|
|
42
41
|
* generation of pair-wise Trainingdata for OCR-Backends
|
|
43
42
|
|
|
44
43
|
## Requirements
|
|
@@ -54,16 +53,10 @@ Each section contains detailed usage help instructions:
|
|
|
54
53
|
# evaluation
|
|
55
54
|
ocr-util eval --help
|
|
56
55
|
|
|
57
|
-
# corpus management
|
|
58
|
-
ocr-util corpus --help
|
|
59
|
-
|
|
60
|
-
# analyse an existing METS corpus
|
|
61
|
-
ocr-util corpus-analyse --help
|
|
62
|
-
|
|
63
56
|
# slice image by image + input OCR
|
|
64
57
|
ocr-util slice --help
|
|
65
58
|
|
|
66
|
-
# render
|
|
59
|
+
# render input OCR (regions, lines, words) on given image
|
|
67
60
|
ocr-util show --help
|
|
68
61
|
```
|
|
69
62
|
|
|
@@ -73,7 +66,7 @@ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONT
|
|
|
73
66
|
|
|
74
67
|
_Please note_:
|
|
75
68
|
Invalid data files are excluded and reported where possible from evaluation.
|
|
76
|
-
The term 'invalid' refers to
|
|
69
|
+
The term 'invalid' refers to malformed or structurally invalid XML, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
|
|
77
70
|
|
|
78
71
|
### Evaluation Filter-Then-Aggregate
|
|
79
72
|
|
|
@@ -89,6 +82,22 @@ ocr-util eval <candidates> \
|
|
|
89
82
|
--aggregate-by "mods:dateIssued:century"
|
|
90
83
|
```
|
|
91
84
|
|
|
85
|
+
The dimensions can also be reversed: filter by publication century or decade,
|
|
86
|
+
then aggregate by language. For example, keep only publications from the 17th
|
|
87
|
+
century (1601-1700 inclusive), then group their evaluation results by MODS language:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
ocr-util eval <candidates> \
|
|
91
|
+
--reference <groundtruth> \
|
|
92
|
+
--mets-file <mets.xml> \
|
|
93
|
+
--filter-by "mods:dateIssued:century=17th" \
|
|
94
|
+
--aggregate-by "mods:language"
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Filtering still happens before evaluation and aggregation; only the metadata
|
|
98
|
+
dimensions exchange roles. To select a decade instead, replace the filter above
|
|
99
|
+
with `--filter-by "mods:dateIssued:decade=1800s"` (1800-1809 inclusive).
|
|
100
|
+
|
|
92
101
|
Multi-language filter values are interpreted as sets:
|
|
93
102
|
|
|
94
103
|
```bash
|
|
@@ -100,6 +109,8 @@ ocr-util eval <candidates> \
|
|
|
100
109
|
```
|
|
101
110
|
|
|
102
111
|
Behavior:
|
|
112
|
+
* centuries follow calendar numbering: the 18th century is 1701-1800,
|
|
113
|
+
and the 19th century is 1801-1900; decades retain ten-year buckets such as 1800-1809
|
|
103
114
|
* single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
|
|
104
115
|
* multi-value filter -> all filter values must be present in any order
|
|
105
116
|
* entries missing the filter criterion are reported as WARNING and discarded
|
|
@@ -108,34 +119,6 @@ Behavior:
|
|
|
108
119
|
* aggregation coverage reports how many evaluated pairs provide each dimension;
|
|
109
120
|
reported `items` count candidate/ground-truth pairs, not MODS elements
|
|
110
121
|
|
|
111
|
-
### Corpus Analysis
|
|
112
|
-
|
|
113
|
-
Existing METS corpora can be filtered without evaluation candidate data. Matching
|
|
114
|
-
full-text file references are printed one per line:
|
|
115
|
-
|
|
116
|
-
```bash
|
|
117
|
-
ocr-util corpus-analyse <mets.xml> \
|
|
118
|
-
--filter-by "mods:dateIssued:century=16th"
|
|
119
|
-
```
|
|
120
|
-
|
|
121
|
-
Repeat `--filter-by` to combine criteria with AND:
|
|
122
|
-
|
|
123
|
-
```bash
|
|
124
|
-
ocr-util corpus-analyse <mets.xml> \
|
|
125
|
-
--filter-by "mods:dateIssued:century=16th" \
|
|
126
|
-
--filter-by "mods:language=ger"
|
|
127
|
-
```
|
|
128
|
-
|
|
129
|
-
Check that every local file referenced by any METS `FLocat` exists:
|
|
130
|
-
|
|
131
|
-
```bash
|
|
132
|
-
ocr-util corpus-analyse <mets.xml> --check
|
|
133
|
-
```
|
|
134
|
-
|
|
135
|
-
Relative paths are resolved from the METS directory. Missing files and remote
|
|
136
|
-
references that cannot be checked locally produce a non-zero exit status. This
|
|
137
|
-
is a file-presence check; it does not perform XML schema validation.
|
|
138
|
-
|
|
139
122
|
## Development
|
|
140
123
|
|
|
141
124
|
Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
|
|
@@ -9,12 +9,6 @@ src/ocr_util.egg-info/dependency_links.txt
|
|
|
9
9
|
src/ocr_util.egg-info/entry_points.txt
|
|
10
10
|
src/ocr_util.egg-info/requires.txt
|
|
11
11
|
src/ocr_util.egg-info/top_level.txt
|
|
12
|
-
src/ocr_util/corpus/__init__.py
|
|
13
|
-
src/ocr_util/corpus/analyse.py
|
|
14
|
-
src/ocr_util/corpus/common.py
|
|
15
|
-
src/ocr_util/corpus/generate_corpus.py
|
|
16
|
-
src/ocr_util/corpus/load_metadata.py
|
|
17
|
-
src/ocr_util/corpus/template.corpus.xml
|
|
18
12
|
src/ocr_util/eval/__init__.py
|
|
19
13
|
src/ocr_util/eval/aggregation.py
|
|
20
14
|
src/ocr_util/eval/cli.py
|
|
@@ -45,11 +39,6 @@ src/ocr_util/slice/cli.py
|
|
|
45
39
|
src/ocr_util/slice/gts_pairs.py
|
|
46
40
|
src/ocr_util/slice/pairs_lstmfs.py
|
|
47
41
|
tests/test_aggregation.py
|
|
48
|
-
tests/test_corpus.py
|
|
49
|
-
tests/test_corpus_analyse.py
|
|
50
|
-
tests/test_corpus_examples.py
|
|
51
|
-
tests/test_corpus_fulltext_generation.py
|
|
52
|
-
tests/test_corpus_load_metadata.py
|
|
53
42
|
tests/test_digital_eval_cli.py
|
|
54
43
|
tests/test_digital_object_model.py
|
|
55
44
|
tests/test_digital_object_ocr_alto.py
|
|
@@ -697,9 +697,25 @@ def test_century_transform_21st():
|
|
|
697
697
|
assert digev.century_transform("2026") == "21st"
|
|
698
698
|
|
|
699
699
|
|
|
700
|
-
|
|
701
|
-
"
|
|
702
|
-
|
|
700
|
+
@pytest.mark.parametrize(
|
|
701
|
+
"year, expected",
|
|
702
|
+
[
|
|
703
|
+
("0001", "1st"),
|
|
704
|
+
("0100", "1st"),
|
|
705
|
+
("0101", "2nd"),
|
|
706
|
+
("1700", "17th"),
|
|
707
|
+
("1701", "18th"),
|
|
708
|
+
("1800", "18th"),
|
|
709
|
+
("1801", "19th"),
|
|
710
|
+
("1900", "19th"),
|
|
711
|
+
("1901", "20th"),
|
|
712
|
+
("2000", "20th"),
|
|
713
|
+
("2001", "21st"),
|
|
714
|
+
],
|
|
715
|
+
)
|
|
716
|
+
def test_century_transform_calendar_boundaries(year, expected):
|
|
717
|
+
"""Calendar centuries start in year xx01 and end in xx00, not xx00 through xx99."""
|
|
718
|
+
assert digev.century_transform(year) == expected
|
|
703
719
|
|
|
704
720
|
|
|
705
721
|
def test_century_transform_11th_century_no_spurious_st():
|