ocr-util 2.0.1__tar.gz → 2.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ocr_util-2.0.1/src/ocr_util.egg-info → ocr_util-2.2.1}/PKG-INFO +47 -11
- {ocr_util-2.0.1 → ocr_util-2.2.1}/README.md +46 -10
- {ocr_util-2.0.1 → ocr_util-2.2.1}/pyproject.toml +0 -1
- ocr_util-2.2.1/src/ocr_util/__init__.py +3 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/cli.py +22 -19
- ocr_util-2.2.1/src/ocr_util/corpus/analyse.py +173 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/common.py +1 -3
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/load_metadata.py +0 -1
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/aggregation.py +9 -27
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/cli.py +104 -95
- ocr_util-2.2.1/src/ocr_util/eval/constants.py +27 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +3 -9
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/Util.py +1 -3
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/evaluation.py +107 -65
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/geometry.py +20 -10
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/metrics.py +15 -12
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/common.py +1 -3
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/digital_object_model.py +6 -15
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/digital_object_util.py +3 -9
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/filter.py +1 -4
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/format_alto_v3_util.py +9 -30
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/format_page_util.py +14 -42
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/minidom_util.py +2 -6
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/preprocessing.py +114 -41
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/resolve.py +1 -3
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/show/cli.py +26 -20
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/show/ocr_show_segmentation.py +214 -133
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/cli.py +53 -41
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/gts_pairs.py +109 -114
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/pairs_lstmfs.py +24 -26
- {ocr_util-2.0.1 → ocr_util-2.2.1/src/ocr_util.egg-info}/PKG-INFO +47 -11
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/SOURCES.txt +5 -1
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_aggregation.py +8 -26
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus.py +6 -17
- ocr_util-2.2.1/tests/test_corpus_analyse.py +108 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus_examples.py +16 -29
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus_load_metadata.py +6 -10
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_eval_cli.py +171 -80
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_model.py +12 -18
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_alto.py +56 -61
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_filter.py +43 -41
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_page.py +24 -24
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_page_eynollah.py +5 -5
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_txt.py +1 -2
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_generate_sets.py +131 -133
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_evaluate.py +117 -102
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_metrics.py +46 -43
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_metrics_base.py +4 -4
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_preprocessing.py +28 -8
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_page_reading_order.py +17 -17
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_show_segmentation.py +151 -133
- ocr_util-2.2.1/tests/test_version_compatibility.py +222 -0
- ocr_util-2.0.1/src/ocr_util/__init__.py +0 -3
- {ocr_util-2.0.1 → ocr_util-2.2.1}/LICENSE +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/setup.cfg +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/__init__.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/generate_corpus.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/template.corpus.xml +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/__init__.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/__init__.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/common.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/common.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/main.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/__init__.py +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/dependency_links.txt +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/entry_points.txt +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/requires.txt +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/top_level.txt +0 -0
- {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus_fulltext_generation.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ocr-util
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.2.1
|
|
4
4
|
Summary: OCR Utils
|
|
5
5
|
Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
|
|
6
6
|
Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
|
|
@@ -33,10 +33,10 @@ Dynamic: license-file
|
|
|
33
33
|
|
|
34
34
|
# OCR Util
|
|
35
35
|
|
|
36
|
-
 [ [](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [](https://pypi.org/project/ocr-util)   
|
|
37
37
|
|
|
38
38
|
|
|
39
|
-
Collection of utils
|
|
39
|
+
Collection of utils for
|
|
40
40
|
* evaluation of OCR data for the masses
|
|
41
41
|
* generation of extended OCR-Evaluation Corpora
|
|
42
42
|
* generation of pair-wise Trainingdata for OCR-Backends
|
|
@@ -52,16 +52,19 @@ Each section contains detailed usage help instructions:
|
|
|
52
52
|
|
|
53
53
|
```bash
|
|
54
54
|
# evaluation
|
|
55
|
-
ocr eval --help
|
|
55
|
+
ocr-util eval --help
|
|
56
56
|
|
|
57
57
|
# corpus management
|
|
58
|
-
ocr corpus --help
|
|
58
|
+
ocr-util corpus --help
|
|
59
|
+
|
|
60
|
+
# analyse an existing METS corpus
|
|
61
|
+
ocr-util corpus-analyse --help
|
|
59
62
|
|
|
60
63
|
# slice image by image + input OCR
|
|
61
|
-
ocr slice --help
|
|
64
|
+
ocr-util slice --help
|
|
62
65
|
|
|
63
66
|
# render image + input OCR
|
|
64
|
-
ocr show --help
|
|
67
|
+
ocr-util show --help
|
|
65
68
|
```
|
|
66
69
|
|
|
67
70
|
### Data problems
|
|
@@ -69,7 +72,8 @@ ocr show --help
|
|
|
69
72
|
Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
|
|
70
73
|
|
|
71
74
|
_Please note_:
|
|
72
|
-
Invalid data files are
|
|
75
|
+
Invalid data files are excluded and reported where possible from evaluation.
|
|
76
|
+
The term 'invalid' refers to errors in schemas in structured XML-data, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
|
|
73
77
|
|
|
74
78
|
### Evaluation Filter-Then-Aggregate
|
|
75
79
|
|
|
@@ -78,7 +82,7 @@ The evaluation CLI supports a single pre-aggregation filter using metadata extra
|
|
|
78
82
|
Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
|
|
79
83
|
|
|
80
84
|
```bash
|
|
81
|
-
ocr eval <candidates> \
|
|
85
|
+
ocr-util eval <candidates> \
|
|
82
86
|
--reference <groundtruth> \
|
|
83
87
|
--mets-file <mets.xml> \
|
|
84
88
|
--filter-by "mods:language=ger" \
|
|
@@ -88,7 +92,7 @@ ocr eval <candidates> \
|
|
|
88
92
|
Multi-language filter values are interpreted as sets:
|
|
89
93
|
|
|
90
94
|
```bash
|
|
91
|
-
ocr eval <candidates> \
|
|
95
|
+
ocr-util eval <candidates> \
|
|
92
96
|
--reference <groundtruth> \
|
|
93
97
|
--mets-file <mets.xml> \
|
|
94
98
|
--filter-by "mods:language=ger+eng" \
|
|
@@ -99,10 +103,42 @@ Behavior:
|
|
|
99
103
|
* single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
|
|
100
104
|
* multi-value filter -> all filter values must be present in any order
|
|
101
105
|
* entries missing the filter criterion are reported as WARNING and discarded
|
|
106
|
+
* METS coverage reports how many evaluation pairs map to full-text references
|
|
107
|
+
and how many METS references have no evaluation pair
|
|
108
|
+
* aggregation coverage reports how many evaluated pairs provide each dimension;
|
|
109
|
+
reported `items` count candidate/ground-truth pairs, not MODS elements
|
|
110
|
+
|
|
111
|
+
### Corpus Analysis
|
|
112
|
+
|
|
113
|
+
Existing METS corpora can be filtered without evaluation candidate data. Matching
|
|
114
|
+
full-text file references are printed one per line:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
ocr-util corpus-analyse <mets.xml> \
|
|
118
|
+
--filter-by "mods:dateIssued:century=16th"
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Repeat `--filter-by` to combine criteria with AND:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
ocr-util corpus-analyse <mets.xml> \
|
|
125
|
+
--filter-by "mods:dateIssued:century=16th" \
|
|
126
|
+
--filter-by "mods:language=ger"
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Check that every local file referenced by any METS `FLocat` exists:
|
|
130
|
+
|
|
131
|
+
```bash
|
|
132
|
+
ocr-util corpus-analyse <mets.xml> --check
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
Relative paths are resolved from the METS directory. Missing files and remote
|
|
136
|
+
references that cannot be checked locally produce a non-zero exit status. This
|
|
137
|
+
is a file-presence check; it does not perform XML schema validation.
|
|
102
138
|
|
|
103
139
|
## Development
|
|
104
140
|
|
|
105
|
-
|
|
141
|
+
Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
|
|
106
142
|
|
|
107
143
|
```bash
|
|
108
144
|
# clone local
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
# OCR Util
|
|
2
2
|
|
|
3
|
-
 [ [](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [](https://pypi.org/project/ocr-util)   
|
|
4
4
|
|
|
5
5
|
|
|
6
|
-
Collection of utils
|
|
6
|
+
Collection of utils for
|
|
7
7
|
* evaluation of OCR data for the masses
|
|
8
8
|
* generation of extended OCR-Evaluation Corpora
|
|
9
9
|
* generation of pair-wise Trainingdata for OCR-Backends
|
|
@@ -19,16 +19,19 @@ Each section contains detailed usage help instructions:
|
|
|
19
19
|
|
|
20
20
|
```bash
|
|
21
21
|
# evaluation
|
|
22
|
-
ocr eval --help
|
|
22
|
+
ocr-util eval --help
|
|
23
23
|
|
|
24
24
|
# corpus management
|
|
25
|
-
ocr corpus --help
|
|
25
|
+
ocr-util corpus --help
|
|
26
|
+
|
|
27
|
+
# analyse an existing METS corpus
|
|
28
|
+
ocr-util corpus-analyse --help
|
|
26
29
|
|
|
27
30
|
# slice image by image + input OCR
|
|
28
|
-
ocr slice --help
|
|
31
|
+
ocr-util slice --help
|
|
29
32
|
|
|
30
33
|
# render image + input OCR
|
|
31
|
-
ocr show --help
|
|
34
|
+
ocr-util show --help
|
|
32
35
|
```
|
|
33
36
|
|
|
34
37
|
### Data problems
|
|
@@ -36,7 +39,8 @@ ocr show --help
|
|
|
36
39
|
Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
|
|
37
40
|
|
|
38
41
|
_Please note_:
|
|
39
|
-
Invalid data files are
|
|
42
|
+
Invalid data files are excluded and reported where possible from evaluation.
|
|
43
|
+
The term 'invalid' refers to errors in schemas in structured XML-data, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
|
|
40
44
|
|
|
41
45
|
### Evaluation Filter-Then-Aggregate
|
|
42
46
|
|
|
@@ -45,7 +49,7 @@ The evaluation CLI supports a single pre-aggregation filter using metadata extra
|
|
|
45
49
|
Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
|
|
46
50
|
|
|
47
51
|
```bash
|
|
48
|
-
ocr eval <candidates> \
|
|
52
|
+
ocr-util eval <candidates> \
|
|
49
53
|
--reference <groundtruth> \
|
|
50
54
|
--mets-file <mets.xml> \
|
|
51
55
|
--filter-by "mods:language=ger" \
|
|
@@ -55,7 +59,7 @@ ocr eval <candidates> \
|
|
|
55
59
|
Multi-language filter values are interpreted as sets:
|
|
56
60
|
|
|
57
61
|
```bash
|
|
58
|
-
ocr eval <candidates> \
|
|
62
|
+
ocr-util eval <candidates> \
|
|
59
63
|
--reference <groundtruth> \
|
|
60
64
|
--mets-file <mets.xml> \
|
|
61
65
|
--filter-by "mods:language=ger+eng" \
|
|
@@ -66,10 +70,42 @@ Behavior:
|
|
|
66
70
|
* single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
|
|
67
71
|
* multi-value filter -> all filter values must be present in any order
|
|
68
72
|
* entries missing the filter criterion are reported as WARNING and discarded
|
|
73
|
+
* METS coverage reports how many evaluation pairs map to full-text references
|
|
74
|
+
and how many METS references have no evaluation pair
|
|
75
|
+
* aggregation coverage reports how many evaluated pairs provide each dimension;
|
|
76
|
+
reported `items` count candidate/ground-truth pairs, not MODS elements
|
|
77
|
+
|
|
78
|
+
### Corpus Analysis
|
|
79
|
+
|
|
80
|
+
Existing METS corpora can be filtered without evaluation candidate data. Matching
|
|
81
|
+
full-text file references are printed one per line:
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
ocr-util corpus-analyse <mets.xml> \
|
|
85
|
+
--filter-by "mods:dateIssued:century=16th"
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Repeat `--filter-by` to combine criteria with AND:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
ocr-util corpus-analyse <mets.xml> \
|
|
92
|
+
--filter-by "mods:dateIssued:century=16th" \
|
|
93
|
+
--filter-by "mods:language=ger"
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Check that every local file referenced by any METS `FLocat` exists:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
ocr-util corpus-analyse <mets.xml> --check
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Relative paths are resolved from the METS directory. Missing files and remote
|
|
103
|
+
references that cannot be checked locally produce a non-zero exit status. This
|
|
104
|
+
is a file-presence check; it does not perform XML schema validation.
|
|
69
105
|
|
|
70
106
|
## Development
|
|
71
107
|
|
|
72
|
-
|
|
108
|
+
Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
|
|
73
109
|
|
|
74
110
|
```bash
|
|
75
111
|
# clone local
|
|
@@ -13,6 +13,7 @@ import ocr_util.eval.model.filter as dofi
|
|
|
13
13
|
import ocr_util.eval.cli as eval_cli
|
|
14
14
|
import ocr_util.slice.cli as slice_cli
|
|
15
15
|
import ocr_util.show.cli as show_cli
|
|
16
|
+
import ocr_util.corpus.analyse as corpus_analyse
|
|
16
17
|
import ocr_util.corpus.generate_corpus as gc
|
|
17
18
|
|
|
18
19
|
from ocr_util.corpus.common import CorpusArgs
|
|
@@ -21,10 +22,9 @@ from ocr_util.corpus.common import CorpusArgs
|
|
|
21
22
|
DEFAULT_VERBOSITY = 0
|
|
22
23
|
SUB_CMD_FRAME = "frame"
|
|
23
24
|
SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
|
|
25
|
+
SUB_CMD_CORPUS_ANALYSE = "corpus-analyse"
|
|
24
26
|
CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
|
|
25
|
-
CORPUS_CACHE_DIR = os.path.join(
|
|
26
|
-
os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME
|
|
27
|
-
)
|
|
27
|
+
CORPUS_CACHE_DIR = os.path.join(os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME)
|
|
28
28
|
|
|
29
29
|
SUB_CMD_EVALUATE = "eval"
|
|
30
30
|
SUB_CMD_SLICE = "slice"
|
|
@@ -44,9 +44,7 @@ def points_type(points: str) -> str:
|
|
|
44
44
|
def start() -> None:
|
|
45
45
|
# Configure logging once, centrally
|
|
46
46
|
logging.basicConfig(
|
|
47
|
-
level=logging.INFO,
|
|
48
|
-
format='%(asctime)s [%(levelname)s][%(name)s] %(message)s',
|
|
49
|
-
datefmt='%Y-%m-%d %H:%M:%S'
|
|
47
|
+
level=logging.INFO, format="%(asctime)s [%(levelname)s][%(name)s] %(message)s", datefmt="%Y-%m-%d %H:%M:%S"
|
|
50
48
|
)
|
|
51
49
|
arg_parser: argparse.ArgumentParser = argparse.ArgumentParser(
|
|
52
50
|
prog="ocr-util",
|
|
@@ -72,9 +70,7 @@ def start() -> None:
|
|
|
72
70
|
required=False,
|
|
73
71
|
help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
|
|
74
72
|
)
|
|
75
|
-
frame_arg_parser.add_argument(
|
|
76
|
-
"-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True
|
|
77
|
-
)
|
|
73
|
+
frame_arg_parser.add_argument("-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True)
|
|
78
74
|
frame_arg_parser.add_argument(
|
|
79
75
|
"-o",
|
|
80
76
|
"--output-ocr-file",
|
|
@@ -158,6 +154,12 @@ def start() -> None:
|
|
|
158
154
|
required=False,
|
|
159
155
|
)
|
|
160
156
|
|
|
157
|
+
corpus_analyse_parser = sub_arg_parsers.add_parser(
|
|
158
|
+
SUB_CMD_CORPUS_ANALYSE,
|
|
159
|
+
help="List files in an existing METS corpus that match metadata filters",
|
|
160
|
+
)
|
|
161
|
+
corpus_analyse.register_arguments(corpus_analyse_parser)
|
|
162
|
+
|
|
161
163
|
# evaluate subcommand
|
|
162
164
|
evaluate_arg_parser = sub_arg_parsers.add_parser(
|
|
163
165
|
SUB_CMD_EVALUATE,
|
|
@@ -232,9 +234,7 @@ def start() -> None:
|
|
|
232
234
|
default=slice_cli.DEFAULT_SANITIZE,
|
|
233
235
|
help=f"optional: sanitize textline images (default: {slice_cli.DEFAULT_SANITIZE})",
|
|
234
236
|
)
|
|
235
|
-
slice_arg_parser.add_argument(
|
|
236
|
-
"--no-sanitize", dest="sanitize", action="store_false"
|
|
237
|
-
)
|
|
237
|
+
slice_arg_parser.add_argument("--no-sanitize", dest="sanitize", action="store_false")
|
|
238
238
|
slice_arg_parser.add_argument(
|
|
239
239
|
"--intrusion-ratio",
|
|
240
240
|
required=False,
|
|
@@ -269,12 +269,8 @@ def start() -> None:
|
|
|
269
269
|
output_ocr_file: str = args.output_ocr_file
|
|
270
270
|
points: str = args.points
|
|
271
271
|
if verbosity > 1:
|
|
272
|
-
print(
|
|
273
|
-
|
|
274
|
-
)
|
|
275
|
-
polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(
|
|
276
|
-
input_ocr_file, points, verbosity
|
|
277
|
-
)
|
|
272
|
+
print(f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}")
|
|
273
|
+
polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(input_ocr_file, points, verbosity)
|
|
278
274
|
piece_result: do.DigitalObjectTree = polygon_frame_filter.process()
|
|
279
275
|
file_result: PurePath = do.from_digital_object(piece_result, output_ocr_file)
|
|
280
276
|
if verbosity > 0:
|
|
@@ -287,10 +283,17 @@ def start() -> None:
|
|
|
287
283
|
local_cache_dir=Path(args.temp_dir).absolute(),
|
|
288
284
|
limit=int(args.limit),
|
|
289
285
|
corpus_label=args.corpus_label,
|
|
290
|
-
clear_cache=args.clear_cache
|
|
286
|
+
clear_cache=args.clear_cache,
|
|
291
287
|
)
|
|
292
288
|
gc.generate(corpus_args)
|
|
293
289
|
|
|
290
|
+
elif args.subcommand == SUB_CMD_CORPUS_ANALYSE:
|
|
291
|
+
analyse_args = vars(args)
|
|
292
|
+
analyse_args.pop("subcommand", None)
|
|
293
|
+
result = corpus_analyse.start_analysis(analyse_args)
|
|
294
|
+
if isinstance(result, corpus_analyse.CorpusCheckResult) and not result.is_valid:
|
|
295
|
+
raise SystemExit(1)
|
|
296
|
+
|
|
294
297
|
elif args.subcommand == SUB_CMD_EVALUATE:
|
|
295
298
|
eval_args = vars(args)
|
|
296
299
|
eval_args.pop("subcommand", None)
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""Analyse a METS corpus with evaluation-compatible filters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import dataclasses
|
|
7
|
+
import re
|
|
8
|
+
import typing
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from urllib.parse import unquote, urlparse
|
|
11
|
+
|
|
12
|
+
import lxml.etree as ET
|
|
13
|
+
|
|
14
|
+
import ocr_util.eval as digev
|
|
15
|
+
from ocr_util.eval.cli import _build_filter_spec, _filter_value_matches
|
|
16
|
+
|
|
17
|
+
METS_NAMESPACES = digev.METSModsExtractor.DEFAULT_NAMESPACES
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclasses.dataclass(frozen=True)
|
|
21
|
+
class CorpusCheckResult:
|
|
22
|
+
"""summarize local file references found in a METS corpus."""
|
|
23
|
+
|
|
24
|
+
checked: tuple[Path, ...]
|
|
25
|
+
missing: tuple[Path, ...]
|
|
26
|
+
uncheckable: tuple[str, ...]
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def is_valid(self) -> bool:
|
|
30
|
+
"""return whether every METS file reference is locally present."""
|
|
31
|
+
return not self.missing and not self.uncheckable
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _groundtruth_type(path: Path) -> typing.Optional[str]:
|
|
35
|
+
"""infer the evaluation ground-truth type from a corpus file name."""
|
|
36
|
+
match = re.match(r".*(?:gt\.(\w{3,})|\.(\w{3,})\.gt)\.xml$", path.name)
|
|
37
|
+
label = next((group for group in match.groups() if group), None) if match else None
|
|
38
|
+
if label and label.startswith("art"):
|
|
39
|
+
return "article"
|
|
40
|
+
if label and label.startswith("ann"):
|
|
41
|
+
return "announcement"
|
|
42
|
+
return None
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _corpus_entries(mets_file: Path) -> list[digev.EvalEntry]:
|
|
46
|
+
"""create entry-like objects for full-text files referenced by a METS corpus."""
|
|
47
|
+
tree = ET.parse(str(mets_file))
|
|
48
|
+
entries: list[digev.EvalEntry] = []
|
|
49
|
+
|
|
50
|
+
for file_group in tree.xpath("//mets:fileGrp", namespaces=METS_NAMESPACES):
|
|
51
|
+
if "FULLTEXT" not in (file_group.get("USE") or "").upper():
|
|
52
|
+
continue
|
|
53
|
+
|
|
54
|
+
for href in file_group.xpath("./mets:file/mets:FLocat/@xlink:href", namespaces=METS_NAMESPACES):
|
|
55
|
+
referenced_path = Path(str(href))
|
|
56
|
+
entry = digev.EvalEntry(referenced_path)
|
|
57
|
+
entry.path_groundtruth = referenced_path
|
|
58
|
+
entry.domain_directories = list(reversed(referenced_path.parent.parts))
|
|
59
|
+
entry.gt_type = _groundtruth_type(referenced_path) or entry.gt_type
|
|
60
|
+
entries.append(entry)
|
|
61
|
+
|
|
62
|
+
return entries
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def check_corpus(mets_file: Path) -> CorpusCheckResult:
|
|
66
|
+
"""check whether every file referenced by the METS corpus is present locally."""
|
|
67
|
+
mets_file = Path(mets_file)
|
|
68
|
+
if not mets_file.is_file():
|
|
69
|
+
raise FileNotFoundError(f"METS file not found: {mets_file}")
|
|
70
|
+
|
|
71
|
+
tree = ET.parse(str(mets_file))
|
|
72
|
+
hrefs = tree.xpath("//mets:FLocat/@xlink:href", namespaces=METS_NAMESPACES)
|
|
73
|
+
checked: list[Path] = []
|
|
74
|
+
missing: list[Path] = []
|
|
75
|
+
uncheckable: list[str] = []
|
|
76
|
+
|
|
77
|
+
for href_value in hrefs:
|
|
78
|
+
href = str(href_value)
|
|
79
|
+
parsed = urlparse(href)
|
|
80
|
+
if parsed.scheme not in ("", "file") or (parsed.netloc not in ("", "localhost")):
|
|
81
|
+
uncheckable.append(href)
|
|
82
|
+
continue
|
|
83
|
+
|
|
84
|
+
path = Path(unquote(parsed.path))
|
|
85
|
+
if not path.is_absolute():
|
|
86
|
+
path = mets_file.parent / path
|
|
87
|
+
checked.append(path)
|
|
88
|
+
if not path.is_file():
|
|
89
|
+
missing.append(path)
|
|
90
|
+
|
|
91
|
+
return CorpusCheckResult(tuple(checked), tuple(missing), tuple(uncheckable))
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def analyse(mets_file: Path, filters: typing.Sequence[str]) -> list[str]:
|
|
95
|
+
"""return corpus file references matching all evaluation filter specifications."""
|
|
96
|
+
mets_file = Path(mets_file)
|
|
97
|
+
if not mets_file.is_file():
|
|
98
|
+
raise FileNotFoundError(f"METS file not found: {mets_file}")
|
|
99
|
+
|
|
100
|
+
entries = _corpus_entries(mets_file)
|
|
101
|
+
for filter_by in filters:
|
|
102
|
+
filter_spec = _build_filter_spec(filter_by, mets_file, strict=True)
|
|
103
|
+
if filter_spec is None:
|
|
104
|
+
raise ValueError(f"Invalid filter specification: '{filter_by}'")
|
|
105
|
+
_, extractor, expected_value = filter_spec
|
|
106
|
+
entries = [
|
|
107
|
+
entry
|
|
108
|
+
for entry in entries
|
|
109
|
+
if (value := extractor(entry)) is not None
|
|
110
|
+
and str(value).strip()
|
|
111
|
+
and _filter_value_matches(str(value), expected_value)
|
|
112
|
+
]
|
|
113
|
+
|
|
114
|
+
return [str(entry.path_groundtruth) for entry in entries]
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def register_arguments(parser: argparse.ArgumentParser) -> None:
|
|
118
|
+
"""register command-line arguments for corpus analysis."""
|
|
119
|
+
parser.add_argument("mets_file", type=Path, help="Path to the corpus METS file")
|
|
120
|
+
parser.add_argument(
|
|
121
|
+
"--filter-by",
|
|
122
|
+
action="append",
|
|
123
|
+
default=[],
|
|
124
|
+
help=(
|
|
125
|
+
"Evaluation-compatible filter; repeat to combine filters with AND, "
|
|
126
|
+
"for example 'mods:dateIssued:century=16th'"
|
|
127
|
+
),
|
|
128
|
+
)
|
|
129
|
+
parser.add_argument(
|
|
130
|
+
"--check",
|
|
131
|
+
action="store_true",
|
|
132
|
+
help="Check that every local file referenced by the METS corpus exists",
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def start_analysis(
|
|
137
|
+
args: typing.Mapping[str, typing.Any],
|
|
138
|
+
) -> typing.Union[list[str], CorpusCheckResult]:
|
|
139
|
+
"""analyse a corpus and print each matching file reference to stdout."""
|
|
140
|
+
if args.get("check"):
|
|
141
|
+
result = check_corpus(Path(args["mets_file"]))
|
|
142
|
+
missing = set(result.missing)
|
|
143
|
+
for path in result.checked:
|
|
144
|
+
status = "MISSING" if path in missing else "OK"
|
|
145
|
+
print(f"{status} {path}")
|
|
146
|
+
for href in result.uncheckable:
|
|
147
|
+
print(f"UNCHECKABLE {href}")
|
|
148
|
+
print(
|
|
149
|
+
f"SUMMARY checked={len(result.checked)} missing={len(result.missing)} "
|
|
150
|
+
f"uncheckable={len(result.uncheckable)}"
|
|
151
|
+
)
|
|
152
|
+
return result
|
|
153
|
+
|
|
154
|
+
if not args.get("filter_by"):
|
|
155
|
+
raise ValueError("at least one --filter-by or --check is required")
|
|
156
|
+
|
|
157
|
+
matches = analyse(Path(args["mets_file"]), args["filter_by"])
|
|
158
|
+
for match in matches:
|
|
159
|
+
print(match)
|
|
160
|
+
return matches
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def start() -> None:
|
|
164
|
+
"""run the standalone corpus analysis CLI."""
|
|
165
|
+
parser = argparse.ArgumentParser()
|
|
166
|
+
register_arguments(parser)
|
|
167
|
+
result = start_analysis(vars(parser.parse_args()))
|
|
168
|
+
if isinstance(result, CorpusCheckResult) and not result.is_valid:
|
|
169
|
+
raise SystemExit(1)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
if __name__ == "__main__":
|
|
173
|
+
start()
|
|
@@ -419,9 +419,7 @@ class MetsResourceFile(MetsFile):
|
|
|
419
419
|
assert file_fulltext is not None, f"No fulltext file pointer for page with CONTENTIDS='{page_urn}'"
|
|
420
420
|
return file_image, file_fulltext
|
|
421
421
|
|
|
422
|
-
|
|
423
|
-
def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path,
|
|
424
|
-
out_dir: pathlib.Path) -> ET._Element:
|
|
422
|
+
def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path, out_dir: pathlib.Path) -> ET._Element:
|
|
425
423
|
"""Create a new file element for the fulltext file with the appropriate FLocat child."""
|
|
426
424
|
file_fulltext = ET.Element(
|
|
427
425
|
f'{{{self.nsmap["mets"]}}}file', attrib={"ID": new_id, "MIMETYPE": "application/vnd.prima.page+xml"}
|
|
@@ -146,9 +146,7 @@ def century_transform(value: str) -> typing.Optional[str]:
|
|
|
146
146
|
try:
|
|
147
147
|
year = int(str(value).strip()[:4])
|
|
148
148
|
century = year // 100 + 1
|
|
149
|
-
suffix = {1: "st", 2: "nd", 3: "rd"}.get(
|
|
150
|
-
century % 10 if century % 100 not in (11, 12, 13) else 0, "th"
|
|
151
|
-
)
|
|
149
|
+
suffix = {1: "st", 2: "nd", 3: "rd"}.get(century % 10 if century % 100 not in (11, 12, 13) else 0, "th")
|
|
152
150
|
return f"{century}{suffix}"
|
|
153
151
|
except (ValueError, TypeError):
|
|
154
152
|
return None
|
|
@@ -247,9 +245,7 @@ class METSDivAttrExtractor:
|
|
|
247
245
|
|
|
248
246
|
# physical div ID → [file hrefs]
|
|
249
247
|
phys_to_hrefs: typing.Dict[str, typing.List[str]] = {}
|
|
250
|
-
for pdiv in tree.xpath(
|
|
251
|
-
'//mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ID]', namespaces=ns
|
|
252
|
-
):
|
|
248
|
+
for pdiv in tree.xpath('//mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ID]', namespaces=ns):
|
|
253
249
|
pid = pdiv.get("ID")
|
|
254
250
|
hrefs = [
|
|
255
251
|
file_id_to_href[fid]
|
|
@@ -263,9 +259,7 @@ class METSDivAttrExtractor:
|
|
|
263
259
|
log_to_attr: typing.Dict[str, str] = {}
|
|
264
260
|
# also: DMDID → attribute value (fallback)
|
|
265
261
|
dmdid_to_attr: typing.Dict[str, str] = {}
|
|
266
|
-
for ldiv in tree.xpath(
|
|
267
|
-
'//mets:structMap[@TYPE="LOGICAL"]//mets:div[@ID]', namespaces=ns
|
|
268
|
-
):
|
|
262
|
+
for ldiv in tree.xpath('//mets:structMap[@TYPE="LOGICAL"]//mets:div[@ID]', namespaces=ns):
|
|
269
263
|
lid = ldiv.get("ID")
|
|
270
264
|
attr_val = ldiv.get(self.attribute)
|
|
271
265
|
if lid and attr_val:
|
|
@@ -412,9 +406,7 @@ class METSModsExtractor:
|
|
|
412
406
|
files = tree.xpath("//mets:file", namespaces=self.namespaces)
|
|
413
407
|
for file_elem in files:
|
|
414
408
|
file_id = file_elem.get("ID")
|
|
415
|
-
flocat = file_elem.xpath(
|
|
416
|
-
"./mets:FLocat/@xlink:href", namespaces=self.namespaces
|
|
417
|
-
)
|
|
409
|
+
flocat = file_elem.xpath("./mets:FLocat/@xlink:href", namespaces=self.namespaces)
|
|
418
410
|
if not flocat:
|
|
419
411
|
continue
|
|
420
412
|
|
|
@@ -456,9 +448,7 @@ class METSModsExtractor:
|
|
|
456
448
|
continue
|
|
457
449
|
|
|
458
450
|
hrefs = []
|
|
459
|
-
for file_id in physical_div.xpath(
|
|
460
|
-
"./mets:fptr/@FILEID", namespaces=self.namespaces
|
|
461
|
-
):
|
|
451
|
+
for file_id in physical_div.xpath("./mets:fptr/@FILEID", namespaces=self.namespaces):
|
|
462
452
|
file_href = file_id_to_href.get(file_id)
|
|
463
453
|
if file_href:
|
|
464
454
|
hrefs.append(file_href)
|
|
@@ -467,9 +457,7 @@ class METSModsExtractor:
|
|
|
467
457
|
physical_div_to_hrefs[physical_id] = hrefs
|
|
468
458
|
|
|
469
459
|
xlink_ns = self.namespaces.get("xlink", "http://www.w3.org/1999/xlink")
|
|
470
|
-
for link in tree.xpath(
|
|
471
|
-
"//mets:structLink/mets:smLink", namespaces=self.namespaces
|
|
472
|
-
):
|
|
460
|
+
for link in tree.xpath("//mets:structLink/mets:smLink", namespaces=self.namespaces):
|
|
473
461
|
logical_id = link.get(f"{{{xlink_ns}}}from")
|
|
474
462
|
physical_id = link.get(f"{{{xlink_ns}}}to")
|
|
475
463
|
if not logical_id or not physical_id:
|
|
@@ -510,9 +498,7 @@ class METSModsExtractor:
|
|
|
510
498
|
# Apply user's XPath expression to MODS section
|
|
511
499
|
for mods_section in mods_sections:
|
|
512
500
|
try:
|
|
513
|
-
results = mods_section.xpath(
|
|
514
|
-
self.xpath_expression, namespaces=self.namespaces
|
|
515
|
-
)
|
|
501
|
+
results = mods_section.xpath(self.xpath_expression, namespaces=self.namespaces)
|
|
516
502
|
if results:
|
|
517
503
|
values = []
|
|
518
504
|
for result in results:
|
|
@@ -583,15 +569,11 @@ class AggregationStrategy:
|
|
|
583
569
|
hierarchical: If True, creates hierarchical keys combining all dimensions
|
|
584
570
|
"""
|
|
585
571
|
|
|
586
|
-
def __init__(
|
|
587
|
-
self, dimensions: typing.List[AggregationDimension], hierarchical: bool = False
|
|
588
|
-
):
|
|
572
|
+
def __init__(self, dimensions: typing.List[AggregationDimension], hierarchical: bool = False):
|
|
589
573
|
self.dimensions = dimensions
|
|
590
574
|
self.hierarchical = hierarchical
|
|
591
575
|
|
|
592
|
-
def generate_keys(
|
|
593
|
-
self, entry: typing.Any, metric: digem.OCRMetric
|
|
594
|
-
) -> typing.List[str]:
|
|
576
|
+
def generate_keys(self, entry: typing.Any, metric: digem.OCRMetric) -> typing.List[str]:
|
|
595
577
|
"""Generate aggregation keys for an evaluation entry
|
|
596
578
|
|
|
597
579
|
Args:
|