ocr-util 2.0.1__tar.gz → 2.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. {ocr_util-2.0.1/src/ocr_util.egg-info → ocr_util-2.2.1}/PKG-INFO +47 -11
  2. {ocr_util-2.0.1 → ocr_util-2.2.1}/README.md +46 -10
  3. {ocr_util-2.0.1 → ocr_util-2.2.1}/pyproject.toml +0 -1
  4. ocr_util-2.2.1/src/ocr_util/__init__.py +3 -0
  5. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/cli.py +22 -19
  6. ocr_util-2.2.1/src/ocr_util/corpus/analyse.py +173 -0
  7. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/common.py +1 -3
  8. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/load_metadata.py +0 -1
  9. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/aggregation.py +9 -27
  10. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/cli.py +104 -95
  11. ocr_util-2.2.1/src/ocr_util/eval/constants.py +27 -0
  12. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +3 -9
  13. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/Util.py +1 -3
  14. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/evaluation.py +107 -65
  15. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/geometry.py +20 -10
  16. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/metrics.py +15 -12
  17. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/common.py +1 -3
  18. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/digital_object_model.py +6 -15
  19. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/digital_object_util.py +3 -9
  20. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/filter.py +1 -4
  21. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/format_alto_v3_util.py +9 -30
  22. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/format_page_util.py +14 -42
  23. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/minidom_util.py +2 -6
  24. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/preprocessing.py +114 -41
  25. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/resolve.py +1 -3
  26. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/show/cli.py +26 -20
  27. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/show/ocr_show_segmentation.py +214 -133
  28. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/cli.py +53 -41
  29. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/gts_pairs.py +109 -114
  30. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/pairs_lstmfs.py +24 -26
  31. {ocr_util-2.0.1 → ocr_util-2.2.1/src/ocr_util.egg-info}/PKG-INFO +47 -11
  32. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/SOURCES.txt +5 -1
  33. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_aggregation.py +8 -26
  34. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus.py +6 -17
  35. ocr_util-2.2.1/tests/test_corpus_analyse.py +108 -0
  36. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus_examples.py +16 -29
  37. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus_load_metadata.py +6 -10
  38. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_eval_cli.py +171 -80
  39. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_model.py +12 -18
  40. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_alto.py +56 -61
  41. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_filter.py +43 -41
  42. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_page.py +24 -24
  43. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_page_eynollah.py +5 -5
  44. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_digital_object_txt.py +1 -2
  45. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_generate_sets.py +131 -133
  46. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_evaluate.py +117 -102
  47. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_metrics.py +46 -43
  48. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_metrics_base.py +4 -4
  49. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_ocr_preprocessing.py +28 -8
  50. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_page_reading_order.py +17 -17
  51. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_show_segmentation.py +151 -133
  52. ocr_util-2.2.1/tests/test_version_compatibility.py +222 -0
  53. ocr_util-2.0.1/src/ocr_util/__init__.py +0 -3
  54. {ocr_util-2.0.1 → ocr_util-2.2.1}/LICENSE +0 -0
  55. {ocr_util-2.0.1 → ocr_util-2.2.1}/setup.cfg +0 -0
  56. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/__init__.py +0 -0
  57. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/generate_corpus.py +0 -0
  58. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/corpus/template.corpus.xml +0 -0
  59. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/__init__.py +0 -0
  60. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/__init__.py +0 -0
  61. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/common.py +0 -0
  62. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
  63. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/common.py +0 -0
  64. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/main.py +0 -0
  65. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util/slice/__init__.py +0 -0
  66. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/dependency_links.txt +0 -0
  67. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/entry_points.txt +0 -0
  68. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/requires.txt +0 -0
  69. {ocr_util-2.0.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/top_level.txt +0 -0
  70. {ocr_util-2.0.1 → ocr_util-2.2.1}/tests/test_corpus_fulltext_generation.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ocr-util
3
- Version: 2.0.1
3
+ Version: 2.2.1
4
4
  Summary: OCR Utils
5
5
  Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
6
6
  Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
@@ -33,10 +33,10 @@ Dynamic: license-file
33
33
 
34
34
  # OCR Util
35
35
 
36
- ![python-app](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml/badge.svg) [![Coverage](./coverage.svg)](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [![PyPi version](https://badgen.net/pypi/v/digital-eval/)](https://pypi.org/project/digital-eval) ![PyPI - Downloads](https://img.shields.io/pypi/dm/digital-eval) ![PyPI - License](https://img.shields.io/pypi/l/digital-eval) ![PyPI - Python Version](https://img.shields.io/pypi/pyversions/digital-eval)
36
+ ![python-app](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml/badge.svg) [![Coverage](https://raw.githubusercontent.com/ulb-sachsen-anhalt/ocr-util/main/coverage.svg)](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [![PyPi version](https://badgen.net/pypi/v/ocr-util/)](https://pypi.org/project/ocr-util) ![PyPI - Downloads](https://img.shields.io/pypi/dm/ocr-util) ![PyPI - License](https://img.shields.io/pypi/l/ocr-util) ![PyPI - Python Version](https://img.shields.io/pypi/pyversions/ocr-util)
37
37
 
38
38
 
39
- Collection of utils to
39
+ Collection of utils for
40
40
  * evaluation of OCR data for the masses
41
41
  * generation of extended OCR-Evaluation Corpora
42
42
  * generation of pair-wise Trainingdata for OCR-Backends
@@ -52,16 +52,19 @@ Each section contains detailed usage help instructions:
52
52
 
53
53
  ```bash
54
54
  # evaluation
55
- ocr eval --help
55
+ ocr-util eval --help
56
56
 
57
57
  # corpus management
58
- ocr corpus --help
58
+ ocr-util corpus --help
59
+
60
+ # analyse an existing METS corpus
61
+ ocr-util corpus-analyse --help
59
62
 
60
63
  # slice image by image + input OCR
61
- ocr slice --help
64
+ ocr-util slice --help
62
65
 
63
66
  # render image + input OCR
64
- ocr show --help
67
+ ocr-util show --help
65
68
  ```
66
69
 
67
70
  ### Data problems
@@ -69,7 +72,8 @@ ocr show --help
69
72
  Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
70
73
 
71
74
  _Please note_:
72
- Invalid data files are tried(!) to be excluded from evaluation.
75
+ Invalid data files are excluded and reported where possible from evaluation.
76
+ The term 'invalid' refers to errors in schemas in structured XML-data, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
73
77
 
74
78
  ### Evaluation Filter-Then-Aggregate
75
79
 
@@ -78,7 +82,7 @@ The evaluation CLI supports a single pre-aggregation filter using metadata extra
78
82
  Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
79
83
 
80
84
  ```bash
81
- ocr eval <candidates> \
85
+ ocr-util eval <candidates> \
82
86
  --reference <groundtruth> \
83
87
  --mets-file <mets.xml> \
84
88
  --filter-by "mods:language=ger" \
@@ -88,7 +92,7 @@ ocr eval <candidates> \
88
92
  Multi-language filter values are interpreted as sets:
89
93
 
90
94
  ```bash
91
- ocr eval <candidates> \
95
+ ocr-util eval <candidates> \
92
96
  --reference <groundtruth> \
93
97
  --mets-file <mets.xml> \
94
98
  --filter-by "mods:language=ger+eng" \
@@ -99,10 +103,42 @@ Behavior:
99
103
  * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
100
104
  * multi-value filter -> all filter values must be present in any order
101
105
  * entries missing the filter criterion are reported as WARNING and discarded
106
+ * METS coverage reports how many evaluation pairs map to full-text references
107
+ and how many METS references have no evaluation pair
108
+ * aggregation coverage reports how many evaluated pairs provide each dimension;
109
+ reported `items` count candidate/ground-truth pairs, not MODS elements
110
+
111
+ ### Corpus Analysis
112
+
113
+ Existing METS corpora can be filtered without evaluation candidate data. Matching
114
+ full-text file references are printed one per line:
115
+
116
+ ```bash
117
+ ocr-util corpus-analyse <mets.xml> \
118
+ --filter-by "mods:dateIssued:century=16th"
119
+ ```
120
+
121
+ Repeat `--filter-by` to combine criteria with AND:
122
+
123
+ ```bash
124
+ ocr-util corpus-analyse <mets.xml> \
125
+ --filter-by "mods:dateIssued:century=16th" \
126
+ --filter-by "mods:language=ger"
127
+ ```
128
+
129
+ Check that every local file referenced by any METS `FLocat` exists:
130
+
131
+ ```bash
132
+ ocr-util corpus-analyse <mets.xml> --check
133
+ ```
134
+
135
+ Relative paths are resolved from the METS directory. Missing files and remote
136
+ references that cannot be checked locally produce a non-zero exit status. This
137
+ is a file-presence check; it does not perform XML schema validation.
102
138
 
103
139
  ## Development
104
140
 
105
- Plattform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
141
+ Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
106
142
 
107
143
  ```bash
108
144
  # clone local
@@ -1,9 +1,9 @@
1
1
  # OCR Util
2
2
 
3
- ![python-app](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml/badge.svg) [![Coverage](./coverage.svg)](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [![PyPi version](https://badgen.net/pypi/v/digital-eval/)](https://pypi.org/project/digital-eval) ![PyPI - Downloads](https://img.shields.io/pypi/dm/digital-eval) ![PyPI - License](https://img.shields.io/pypi/l/digital-eval) ![PyPI - Python Version](https://img.shields.io/pypi/pyversions/digital-eval)
3
+ ![python-app](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml/badge.svg) [![Coverage](https://raw.githubusercontent.com/ulb-sachsen-anhalt/ocr-util/main/coverage.svg)](https://github.com/ulb-sachsen-anhalt/ocr-util/actions/workflows/python-app.yml) [![PyPi version](https://badgen.net/pypi/v/ocr-util/)](https://pypi.org/project/ocr-util) ![PyPI - Downloads](https://img.shields.io/pypi/dm/ocr-util) ![PyPI - License](https://img.shields.io/pypi/l/ocr-util) ![PyPI - Python Version](https://img.shields.io/pypi/pyversions/ocr-util)
4
4
 
5
5
 
6
- Collection of utils to
6
+ Collection of utils for
7
7
  * evaluation of OCR data for the masses
8
8
  * generation of extended OCR-Evaluation Corpora
9
9
  * generation of pair-wise Trainingdata for OCR-Backends
@@ -19,16 +19,19 @@ Each section contains detailed usage help instructions:
19
19
 
20
20
  ```bash
21
21
  # evaluation
22
- ocr eval --help
22
+ ocr-util eval --help
23
23
 
24
24
  # corpus management
25
- ocr corpus --help
25
+ ocr-util corpus --help
26
+
27
+ # analyse an existing METS corpus
28
+ ocr-util corpus-analyse --help
26
29
 
27
30
  # slice image by image + input OCR
28
- ocr slice --help
31
+ ocr-util slice --help
29
32
 
30
33
  # render image + input OCR
31
- ocr show --help
34
+ ocr-util show --help
32
35
  ```
33
36
 
34
37
  ### Data problems
@@ -36,7 +39,8 @@ ocr show --help
36
39
  Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONTENT or PAGE without TextEquiv) or invalid geometrical coordinates (less than 3 points or even empty) will lead to evaluation errors if geometry must be respected.
37
40
 
38
41
  _Please note_:
39
- Invalid data files are tried(!) to be excluded from evaluation.
42
+ Invalid data files are excluded and reported where possible from evaluation.
43
+ The term 'invalid' refers to errors in schemas in structured XML-data, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
40
44
 
41
45
  ### Evaluation Filter-Then-Aggregate
42
46
 
@@ -45,7 +49,7 @@ The evaluation CLI supports a single pre-aggregation filter using metadata extra
45
49
  Example: keep only entries where MODS language is exactly German, then aggregate by publication century:
46
50
 
47
51
  ```bash
48
- ocr eval <candidates> \
52
+ ocr-util eval <candidates> \
49
53
  --reference <groundtruth> \
50
54
  --mets-file <mets.xml> \
51
55
  --filter-by "mods:language=ger" \
@@ -55,7 +59,7 @@ ocr eval <candidates> \
55
59
  Multi-language filter values are interpreted as sets:
56
60
 
57
61
  ```bash
58
- ocr eval <candidates> \
62
+ ocr-util eval <candidates> \
59
63
  --reference <groundtruth> \
60
64
  --mets-file <mets.xml> \
61
65
  --filter-by "mods:language=ger+eng" \
@@ -66,10 +70,42 @@ Behavior:
66
70
  * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
67
71
  * multi-value filter -> all filter values must be present in any order
68
72
  * entries missing the filter criterion are reported as WARNING and discarded
73
+ * METS coverage reports how many evaluation pairs map to full-text references
74
+ and how many METS references have no evaluation pair
75
+ * aggregation coverage reports how many evaluated pairs provide each dimension;
76
+ reported `items` count candidate/ground-truth pairs, not MODS elements
77
+
78
+ ### Corpus Analysis
79
+
80
+ Existing METS corpora can be filtered without evaluation candidate data. Matching
81
+ full-text file references are printed one per line:
82
+
83
+ ```bash
84
+ ocr-util corpus-analyse <mets.xml> \
85
+ --filter-by "mods:dateIssued:century=16th"
86
+ ```
87
+
88
+ Repeat `--filter-by` to combine criteria with AND:
89
+
90
+ ```bash
91
+ ocr-util corpus-analyse <mets.xml> \
92
+ --filter-by "mods:dateIssued:century=16th" \
93
+ --filter-by "mods:language=ger"
94
+ ```
95
+
96
+ Check that every local file referenced by any METS `FLocat` exists:
97
+
98
+ ```bash
99
+ ocr-util corpus-analyse <mets.xml> --check
100
+ ```
101
+
102
+ Relative paths are resolved from the METS directory. Missing files and remote
103
+ references that cannot be checked locally produce a non-zero exit status. This
104
+ is a file-presence check; it does not perform XML schema validation.
69
105
 
70
106
  ## Development
71
107
 
72
- Plattform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
108
+ Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
73
109
 
74
110
  ```bash
75
111
  # clone local
@@ -74,7 +74,6 @@ testpaths = ["tests"]
74
74
  python_files = ["test_*.py"]
75
75
  python_classes = ["Test*"]
76
76
  python_functions = ["test_*"]
77
- addopts = "-v --tb=short --cov-report=xml"
78
77
 
79
78
  [tool.coverage.run]
80
79
  source = ["ocr_util"]
@@ -0,0 +1,3 @@
1
+ """main API"""
2
+
3
+ __version__ = "2.2.1"
@@ -13,6 +13,7 @@ import ocr_util.eval.model.filter as dofi
13
13
  import ocr_util.eval.cli as eval_cli
14
14
  import ocr_util.slice.cli as slice_cli
15
15
  import ocr_util.show.cli as show_cli
16
+ import ocr_util.corpus.analyse as corpus_analyse
16
17
  import ocr_util.corpus.generate_corpus as gc
17
18
 
18
19
  from ocr_util.corpus.common import CorpusArgs
@@ -21,10 +22,9 @@ from ocr_util.corpus.common import CorpusArgs
21
22
  DEFAULT_VERBOSITY = 0
22
23
  SUB_CMD_FRAME = "frame"
23
24
  SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
25
+ SUB_CMD_CORPUS_ANALYSE = "corpus-analyse"
24
26
  CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
25
- CORPUS_CACHE_DIR = os.path.join(
26
- os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME
27
- )
27
+ CORPUS_CACHE_DIR = os.path.join(os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME)
28
28
 
29
29
  SUB_CMD_EVALUATE = "eval"
30
30
  SUB_CMD_SLICE = "slice"
@@ -44,9 +44,7 @@ def points_type(points: str) -> str:
44
44
  def start() -> None:
45
45
  # Configure logging once, centrally
46
46
  logging.basicConfig(
47
- level=logging.INFO,
48
- format='%(asctime)s [%(levelname)s][%(name)s] %(message)s',
49
- datefmt='%Y-%m-%d %H:%M:%S'
47
+ level=logging.INFO, format="%(asctime)s [%(levelname)s][%(name)s] %(message)s", datefmt="%Y-%m-%d %H:%M:%S"
50
48
  )
51
49
  arg_parser: argparse.ArgumentParser = argparse.ArgumentParser(
52
50
  prog="ocr-util",
@@ -72,9 +70,7 @@ def start() -> None:
72
70
  required=False,
73
71
  help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
74
72
  )
75
- frame_arg_parser.add_argument(
76
- "-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True
77
- )
73
+ frame_arg_parser.add_argument("-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True)
78
74
  frame_arg_parser.add_argument(
79
75
  "-o",
80
76
  "--output-ocr-file",
@@ -158,6 +154,12 @@ def start() -> None:
158
154
  required=False,
159
155
  )
160
156
 
157
+ corpus_analyse_parser = sub_arg_parsers.add_parser(
158
+ SUB_CMD_CORPUS_ANALYSE,
159
+ help="List files in an existing METS corpus that match metadata filters",
160
+ )
161
+ corpus_analyse.register_arguments(corpus_analyse_parser)
162
+
161
163
  # evaluate subcommand
162
164
  evaluate_arg_parser = sub_arg_parsers.add_parser(
163
165
  SUB_CMD_EVALUATE,
@@ -232,9 +234,7 @@ def start() -> None:
232
234
  default=slice_cli.DEFAULT_SANITIZE,
233
235
  help=f"optional: sanitize textline images (default: {slice_cli.DEFAULT_SANITIZE})",
234
236
  )
235
- slice_arg_parser.add_argument(
236
- "--no-sanitize", dest="sanitize", action="store_false"
237
- )
237
+ slice_arg_parser.add_argument("--no-sanitize", dest="sanitize", action="store_false")
238
238
  slice_arg_parser.add_argument(
239
239
  "--intrusion-ratio",
240
240
  required=False,
@@ -269,12 +269,8 @@ def start() -> None:
269
269
  output_ocr_file: str = args.output_ocr_file
270
270
  points: str = args.points
271
271
  if verbosity > 1:
272
- print(
273
- f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}"
274
- )
275
- polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(
276
- input_ocr_file, points, verbosity
277
- )
272
+ print(f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}")
273
+ polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(input_ocr_file, points, verbosity)
278
274
  piece_result: do.DigitalObjectTree = polygon_frame_filter.process()
279
275
  file_result: PurePath = do.from_digital_object(piece_result, output_ocr_file)
280
276
  if verbosity > 0:
@@ -287,10 +283,17 @@ def start() -> None:
287
283
  local_cache_dir=Path(args.temp_dir).absolute(),
288
284
  limit=int(args.limit),
289
285
  corpus_label=args.corpus_label,
290
- clear_cache=args.clear_cache
286
+ clear_cache=args.clear_cache,
291
287
  )
292
288
  gc.generate(corpus_args)
293
289
 
290
+ elif args.subcommand == SUB_CMD_CORPUS_ANALYSE:
291
+ analyse_args = vars(args)
292
+ analyse_args.pop("subcommand", None)
293
+ result = corpus_analyse.start_analysis(analyse_args)
294
+ if isinstance(result, corpus_analyse.CorpusCheckResult) and not result.is_valid:
295
+ raise SystemExit(1)
296
+
294
297
  elif args.subcommand == SUB_CMD_EVALUATE:
295
298
  eval_args = vars(args)
296
299
  eval_args.pop("subcommand", None)
@@ -0,0 +1,173 @@
1
+ """Analyse a METS corpus with evaluation-compatible filters."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import dataclasses
7
+ import re
8
+ import typing
9
+ from pathlib import Path
10
+ from urllib.parse import unquote, urlparse
11
+
12
+ import lxml.etree as ET
13
+
14
+ import ocr_util.eval as digev
15
+ from ocr_util.eval.cli import _build_filter_spec, _filter_value_matches
16
+
17
+ METS_NAMESPACES = digev.METSModsExtractor.DEFAULT_NAMESPACES
18
+
19
+
20
+ @dataclasses.dataclass(frozen=True)
21
+ class CorpusCheckResult:
22
+ """summarize local file references found in a METS corpus."""
23
+
24
+ checked: tuple[Path, ...]
25
+ missing: tuple[Path, ...]
26
+ uncheckable: tuple[str, ...]
27
+
28
+ @property
29
+ def is_valid(self) -> bool:
30
+ """return whether every METS file reference is locally present."""
31
+ return not self.missing and not self.uncheckable
32
+
33
+
34
+ def _groundtruth_type(path: Path) -> typing.Optional[str]:
35
+ """infer the evaluation ground-truth type from a corpus file name."""
36
+ match = re.match(r".*(?:gt\.(\w{3,})|\.(\w{3,})\.gt)\.xml$", path.name)
37
+ label = next((group for group in match.groups() if group), None) if match else None
38
+ if label and label.startswith("art"):
39
+ return "article"
40
+ if label and label.startswith("ann"):
41
+ return "announcement"
42
+ return None
43
+
44
+
45
+ def _corpus_entries(mets_file: Path) -> list[digev.EvalEntry]:
46
+ """create entry-like objects for full-text files referenced by a METS corpus."""
47
+ tree = ET.parse(str(mets_file))
48
+ entries: list[digev.EvalEntry] = []
49
+
50
+ for file_group in tree.xpath("//mets:fileGrp", namespaces=METS_NAMESPACES):
51
+ if "FULLTEXT" not in (file_group.get("USE") or "").upper():
52
+ continue
53
+
54
+ for href in file_group.xpath("./mets:file/mets:FLocat/@xlink:href", namespaces=METS_NAMESPACES):
55
+ referenced_path = Path(str(href))
56
+ entry = digev.EvalEntry(referenced_path)
57
+ entry.path_groundtruth = referenced_path
58
+ entry.domain_directories = list(reversed(referenced_path.parent.parts))
59
+ entry.gt_type = _groundtruth_type(referenced_path) or entry.gt_type
60
+ entries.append(entry)
61
+
62
+ return entries
63
+
64
+
65
+ def check_corpus(mets_file: Path) -> CorpusCheckResult:
66
+ """check whether every file referenced by the METS corpus is present locally."""
67
+ mets_file = Path(mets_file)
68
+ if not mets_file.is_file():
69
+ raise FileNotFoundError(f"METS file not found: {mets_file}")
70
+
71
+ tree = ET.parse(str(mets_file))
72
+ hrefs = tree.xpath("//mets:FLocat/@xlink:href", namespaces=METS_NAMESPACES)
73
+ checked: list[Path] = []
74
+ missing: list[Path] = []
75
+ uncheckable: list[str] = []
76
+
77
+ for href_value in hrefs:
78
+ href = str(href_value)
79
+ parsed = urlparse(href)
80
+ if parsed.scheme not in ("", "file") or (parsed.netloc not in ("", "localhost")):
81
+ uncheckable.append(href)
82
+ continue
83
+
84
+ path = Path(unquote(parsed.path))
85
+ if not path.is_absolute():
86
+ path = mets_file.parent / path
87
+ checked.append(path)
88
+ if not path.is_file():
89
+ missing.append(path)
90
+
91
+ return CorpusCheckResult(tuple(checked), tuple(missing), tuple(uncheckable))
92
+
93
+
94
+ def analyse(mets_file: Path, filters: typing.Sequence[str]) -> list[str]:
95
+ """return corpus file references matching all evaluation filter specifications."""
96
+ mets_file = Path(mets_file)
97
+ if not mets_file.is_file():
98
+ raise FileNotFoundError(f"METS file not found: {mets_file}")
99
+
100
+ entries = _corpus_entries(mets_file)
101
+ for filter_by in filters:
102
+ filter_spec = _build_filter_spec(filter_by, mets_file, strict=True)
103
+ if filter_spec is None:
104
+ raise ValueError(f"Invalid filter specification: '{filter_by}'")
105
+ _, extractor, expected_value = filter_spec
106
+ entries = [
107
+ entry
108
+ for entry in entries
109
+ if (value := extractor(entry)) is not None
110
+ and str(value).strip()
111
+ and _filter_value_matches(str(value), expected_value)
112
+ ]
113
+
114
+ return [str(entry.path_groundtruth) for entry in entries]
115
+
116
+
117
+ def register_arguments(parser: argparse.ArgumentParser) -> None:
118
+ """register command-line arguments for corpus analysis."""
119
+ parser.add_argument("mets_file", type=Path, help="Path to the corpus METS file")
120
+ parser.add_argument(
121
+ "--filter-by",
122
+ action="append",
123
+ default=[],
124
+ help=(
125
+ "Evaluation-compatible filter; repeat to combine filters with AND, "
126
+ "for example 'mods:dateIssued:century=16th'"
127
+ ),
128
+ )
129
+ parser.add_argument(
130
+ "--check",
131
+ action="store_true",
132
+ help="Check that every local file referenced by the METS corpus exists",
133
+ )
134
+
135
+
136
+ def start_analysis(
137
+ args: typing.Mapping[str, typing.Any],
138
+ ) -> typing.Union[list[str], CorpusCheckResult]:
139
+ """analyse a corpus and print each matching file reference to stdout."""
140
+ if args.get("check"):
141
+ result = check_corpus(Path(args["mets_file"]))
142
+ missing = set(result.missing)
143
+ for path in result.checked:
144
+ status = "MISSING" if path in missing else "OK"
145
+ print(f"{status} {path}")
146
+ for href in result.uncheckable:
147
+ print(f"UNCHECKABLE {href}")
148
+ print(
149
+ f"SUMMARY checked={len(result.checked)} missing={len(result.missing)} "
150
+ f"uncheckable={len(result.uncheckable)}"
151
+ )
152
+ return result
153
+
154
+ if not args.get("filter_by"):
155
+ raise ValueError("at least one --filter-by or --check is required")
156
+
157
+ matches = analyse(Path(args["mets_file"]), args["filter_by"])
158
+ for match in matches:
159
+ print(match)
160
+ return matches
161
+
162
+
163
+ def start() -> None:
164
+ """run the standalone corpus analysis CLI."""
165
+ parser = argparse.ArgumentParser()
166
+ register_arguments(parser)
167
+ result = start_analysis(vars(parser.parse_args()))
168
+ if isinstance(result, CorpusCheckResult) and not result.is_valid:
169
+ raise SystemExit(1)
170
+
171
+
172
+ if __name__ == "__main__":
173
+ start()
@@ -419,9 +419,7 @@ class MetsResourceFile(MetsFile):
419
419
  assert file_fulltext is not None, f"No fulltext file pointer for page with CONTENTIDS='{page_urn}'"
420
420
  return file_image, file_fulltext
421
421
 
422
-
423
- def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path,
424
- out_dir: pathlib.Path) -> ET._Element:
422
+ def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path, out_dir: pathlib.Path) -> ET._Element:
425
423
  """Create a new file element for the fulltext file with the appropriate FLocat child."""
426
424
  file_fulltext = ET.Element(
427
425
  f'{{{self.nsmap["mets"]}}}file', attrib={"ID": new_id, "MIMETYPE": "application/vnd.prima.page+xml"}
@@ -37,7 +37,6 @@ FALLBACK_IDX_BASE_URLS = {
37
37
  }
38
38
 
39
39
 
40
-
41
40
  @dataclasses.dataclass(frozen=True)
42
41
  class RecordResolutionResult:
43
42
  """Normalized result of a handle lookup attempt.
@@ -146,9 +146,7 @@ def century_transform(value: str) -> typing.Optional[str]:
146
146
  try:
147
147
  year = int(str(value).strip()[:4])
148
148
  century = year // 100 + 1
149
- suffix = {1: "st", 2: "nd", 3: "rd"}.get(
150
- century % 10 if century % 100 not in (11, 12, 13) else 0, "th"
151
- )
149
+ suffix = {1: "st", 2: "nd", 3: "rd"}.get(century % 10 if century % 100 not in (11, 12, 13) else 0, "th")
152
150
  return f"{century}{suffix}"
153
151
  except (ValueError, TypeError):
154
152
  return None
@@ -247,9 +245,7 @@ class METSDivAttrExtractor:
247
245
 
248
246
  # physical div ID → [file hrefs]
249
247
  phys_to_hrefs: typing.Dict[str, typing.List[str]] = {}
250
- for pdiv in tree.xpath(
251
- '//mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ID]', namespaces=ns
252
- ):
248
+ for pdiv in tree.xpath('//mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ID]', namespaces=ns):
253
249
  pid = pdiv.get("ID")
254
250
  hrefs = [
255
251
  file_id_to_href[fid]
@@ -263,9 +259,7 @@ class METSDivAttrExtractor:
263
259
  log_to_attr: typing.Dict[str, str] = {}
264
260
  # also: DMDID → attribute value (fallback)
265
261
  dmdid_to_attr: typing.Dict[str, str] = {}
266
- for ldiv in tree.xpath(
267
- '//mets:structMap[@TYPE="LOGICAL"]//mets:div[@ID]', namespaces=ns
268
- ):
262
+ for ldiv in tree.xpath('//mets:structMap[@TYPE="LOGICAL"]//mets:div[@ID]', namespaces=ns):
269
263
  lid = ldiv.get("ID")
270
264
  attr_val = ldiv.get(self.attribute)
271
265
  if lid and attr_val:
@@ -412,9 +406,7 @@ class METSModsExtractor:
412
406
  files = tree.xpath("//mets:file", namespaces=self.namespaces)
413
407
  for file_elem in files:
414
408
  file_id = file_elem.get("ID")
415
- flocat = file_elem.xpath(
416
- "./mets:FLocat/@xlink:href", namespaces=self.namespaces
417
- )
409
+ flocat = file_elem.xpath("./mets:FLocat/@xlink:href", namespaces=self.namespaces)
418
410
  if not flocat:
419
411
  continue
420
412
 
@@ -456,9 +448,7 @@ class METSModsExtractor:
456
448
  continue
457
449
 
458
450
  hrefs = []
459
- for file_id in physical_div.xpath(
460
- "./mets:fptr/@FILEID", namespaces=self.namespaces
461
- ):
451
+ for file_id in physical_div.xpath("./mets:fptr/@FILEID", namespaces=self.namespaces):
462
452
  file_href = file_id_to_href.get(file_id)
463
453
  if file_href:
464
454
  hrefs.append(file_href)
@@ -467,9 +457,7 @@ class METSModsExtractor:
467
457
  physical_div_to_hrefs[physical_id] = hrefs
468
458
 
469
459
  xlink_ns = self.namespaces.get("xlink", "http://www.w3.org/1999/xlink")
470
- for link in tree.xpath(
471
- "//mets:structLink/mets:smLink", namespaces=self.namespaces
472
- ):
460
+ for link in tree.xpath("//mets:structLink/mets:smLink", namespaces=self.namespaces):
473
461
  logical_id = link.get(f"{{{xlink_ns}}}from")
474
462
  physical_id = link.get(f"{{{xlink_ns}}}to")
475
463
  if not logical_id or not physical_id:
@@ -510,9 +498,7 @@ class METSModsExtractor:
510
498
  # Apply user's XPath expression to MODS section
511
499
  for mods_section in mods_sections:
512
500
  try:
513
- results = mods_section.xpath(
514
- self.xpath_expression, namespaces=self.namespaces
515
- )
501
+ results = mods_section.xpath(self.xpath_expression, namespaces=self.namespaces)
516
502
  if results:
517
503
  values = []
518
504
  for result in results:
@@ -583,15 +569,11 @@ class AggregationStrategy:
583
569
  hierarchical: If True, creates hierarchical keys combining all dimensions
584
570
  """
585
571
 
586
- def __init__(
587
- self, dimensions: typing.List[AggregationDimension], hierarchical: bool = False
588
- ):
572
+ def __init__(self, dimensions: typing.List[AggregationDimension], hierarchical: bool = False):
589
573
  self.dimensions = dimensions
590
574
  self.hierarchical = hierarchical
591
575
 
592
- def generate_keys(
593
- self, entry: typing.Any, metric: digem.OCRMetric
594
- ) -> typing.List[str]:
576
+ def generate_keys(self, entry: typing.Any, metric: digem.OCRMetric) -> typing.List[str]:
595
577
  """Generate aggregation keys for an evaluation entry
596
578
 
597
579
  Args: