ocr-util 2.2.1__tar.gz → 3.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. {ocr_util-2.2.1/src/ocr_util.egg-info → ocr_util-3.0.1}/PKG-INFO +22 -39
  2. {ocr_util-2.2.1 → ocr_util-3.0.1}/README.md +20 -37
  3. {ocr_util-2.2.1 → ocr_util-3.0.1}/pyproject.toml +2 -2
  4. ocr_util-3.0.1/src/ocr_util/__init__.py +3 -0
  5. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/cli.py +1 -99
  6. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/aggregation.py +2 -2
  7. {ocr_util-2.2.1 → ocr_util-3.0.1/src/ocr_util.egg-info}/PKG-INFO +22 -39
  8. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/SOURCES.txt +0 -11
  9. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_aggregation.py +19 -3
  10. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_eval_cli.py +83 -9
  11. ocr_util-2.2.1/src/ocr_util/__init__.py +0 -3
  12. ocr_util-2.2.1/src/ocr_util/corpus/__init__.py +0 -17
  13. ocr_util-2.2.1/src/ocr_util/corpus/analyse.py +0 -173
  14. ocr_util-2.2.1/src/ocr_util/corpus/common.py +0 -520
  15. ocr_util-2.2.1/src/ocr_util/corpus/generate_corpus.py +0 -149
  16. ocr_util-2.2.1/src/ocr_util/corpus/load_metadata.py +0 -324
  17. ocr_util-2.2.1/src/ocr_util/corpus/template.corpus.xml +0 -24
  18. ocr_util-2.2.1/tests/test_corpus.py +0 -515
  19. ocr_util-2.2.1/tests/test_corpus_analyse.py +0 -108
  20. ocr_util-2.2.1/tests/test_corpus_examples.py +0 -304
  21. ocr_util-2.2.1/tests/test_corpus_fulltext_generation.py +0 -144
  22. ocr_util-2.2.1/tests/test_corpus_load_metadata.py +0 -179
  23. {ocr_util-2.2.1 → ocr_util-3.0.1}/LICENSE +0 -0
  24. {ocr_util-2.2.1 → ocr_util-3.0.1}/setup.cfg +0 -0
  25. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/__init__.py +0 -0
  26. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/cli.py +0 -0
  27. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/constants.py +0 -0
  28. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/__init__.py +0 -0
  29. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/common.py +0 -0
  30. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +0 -0
  31. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/Util.py +0 -0
  32. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
  33. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/dictionary_metrics/language_tool/common.py +0 -0
  34. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/evaluation.py +0 -0
  35. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/geometry.py +0 -0
  36. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/metrics.py +0 -0
  37. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/common.py +0 -0
  38. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/digital_object_model.py +0 -0
  39. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/digital_object_util.py +0 -0
  40. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/filter.py +0 -0
  41. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/format_alto_v3_util.py +0 -0
  42. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/format_page_util.py +0 -0
  43. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/main.py +0 -0
  44. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/model/minidom_util.py +0 -0
  45. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/preprocessing.py +0 -0
  46. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/eval/resolve.py +0 -0
  47. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/show/cli.py +0 -0
  48. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/show/ocr_show_segmentation.py +0 -0
  49. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/__init__.py +0 -0
  50. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/cli.py +0 -0
  51. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/gts_pairs.py +0 -0
  52. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util/slice/pairs_lstmfs.py +0 -0
  53. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/dependency_links.txt +0 -0
  54. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/entry_points.txt +0 -0
  55. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/requires.txt +0 -0
  56. {ocr_util-2.2.1 → ocr_util-3.0.1}/src/ocr_util.egg-info/top_level.txt +0 -0
  57. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_model.py +0 -0
  58. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_alto.py +0 -0
  59. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_filter.py +0 -0
  60. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_page.py +0 -0
  61. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_ocr_page_eynollah.py +0 -0
  62. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_digital_object_txt.py +0 -0
  63. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_generate_sets.py +0 -0
  64. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_evaluate.py +0 -0
  65. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_metrics.py +0 -0
  66. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_metrics_base.py +0 -0
  67. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_ocr_preprocessing.py +0 -0
  68. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_page_reading_order.py +0 -0
  69. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_show_segmentation.py +0 -0
  70. {ocr_util-2.2.1 → ocr_util-3.0.1}/tests/test_version_compatibility.py +0 -0
@@ -1,11 +1,11 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ocr-util
3
- Version: 2.2.1
3
+ Version: 3.0.1
4
4
  Summary: OCR Utils
5
5
  Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
6
+ License-Expression: MIT
6
7
  Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
7
8
  Classifier: Programming Language :: Python :: 3
8
- Classifier: License :: OSI Approved :: MIT License
9
9
  Requires-Python: >=3.10
10
10
  Description-Content-Type: text/markdown
11
11
  License-File: LICENSE
@@ -38,7 +38,6 @@ Dynamic: license-file
38
38
 
39
39
  Collection of utils for
40
40
  * evaluation of OCR data for the masses
41
- * generation of extended OCR-Evaluation Corpora
42
41
  * generation of pair-wise Trainingdata for OCR-Backends
43
42
 
44
43
  ## Requirements
@@ -54,16 +53,10 @@ Each section contains detailed usage help instructions:
54
53
  # evaluation
55
54
  ocr-util eval --help
56
55
 
57
- # corpus management
58
- ocr-util corpus --help
59
-
60
- # analyse an existing METS corpus
61
- ocr-util corpus-analyse --help
62
-
63
56
  # slice image by image + input OCR
64
57
  ocr-util slice --help
65
58
 
66
- # render image + input OCR
59
+ # render input OCR (regions, lines, words) on given image
67
60
  ocr-util show --help
68
61
  ```
69
62
 
@@ -73,7 +66,7 @@ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONT
73
66
 
74
67
  _Please note_:
75
68
  Invalid data files are excluded and reported where possible from evaluation.
76
- The term 'invalid' refers to errors in schemas in structured XML-data, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
69
+ The term 'invalid' refers to malformed or structurally invalid XML, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
77
70
 
78
71
  ### Evaluation Filter-Then-Aggregate
79
72
 
@@ -89,6 +82,22 @@ ocr-util eval <candidates> \
89
82
  --aggregate-by "mods:dateIssued:century"
90
83
  ```
91
84
 
85
+ The dimensions can also be reversed: filter by publication century or decade,
86
+ then aggregate by language. For example, keep only publications from the 17th
87
+ century (1601-1700 inclusive), then group their evaluation results by MODS language:
88
+
89
+ ```bash
90
+ ocr-util eval <candidates> \
91
+ --reference <groundtruth> \
92
+ --mets-file <mets.xml> \
93
+ --filter-by "mods:dateIssued:century=17th" \
94
+ --aggregate-by "mods:language"
95
+ ```
96
+
97
+ Filtering still happens before evaluation and aggregation; only the metadata
98
+ dimensions exchange roles. To select a decade instead, replace the filter above
99
+ with `--filter-by "mods:dateIssued:decade=1800s"` (1800-1809 inclusive).
100
+
92
101
  Multi-language filter values are interpreted as sets:
93
102
 
94
103
  ```bash
@@ -100,6 +109,8 @@ ocr-util eval <candidates> \
100
109
  ```
101
110
 
102
111
  Behavior:
112
+ * centuries follow calendar numbering: the 18th century is 1701-1800,
113
+ and the 19th century is 1801-1900; decades retain ten-year buckets such as 1800-1809
103
114
  * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
104
115
  * multi-value filter -> all filter values must be present in any order
105
116
  * entries missing the filter criterion are reported as WARNING and discarded
@@ -108,34 +119,6 @@ Behavior:
108
119
  * aggregation coverage reports how many evaluated pairs provide each dimension;
109
120
  reported `items` count candidate/ground-truth pairs, not MODS elements
110
121
 
111
- ### Corpus Analysis
112
-
113
- Existing METS corpora can be filtered without evaluation candidate data. Matching
114
- full-text file references are printed one per line:
115
-
116
- ```bash
117
- ocr-util corpus-analyse <mets.xml> \
118
- --filter-by "mods:dateIssued:century=16th"
119
- ```
120
-
121
- Repeat `--filter-by` to combine criteria with AND:
122
-
123
- ```bash
124
- ocr-util corpus-analyse <mets.xml> \
125
- --filter-by "mods:dateIssued:century=16th" \
126
- --filter-by "mods:language=ger"
127
- ```
128
-
129
- Check that every local file referenced by any METS `FLocat` exists:
130
-
131
- ```bash
132
- ocr-util corpus-analyse <mets.xml> --check
133
- ```
134
-
135
- Relative paths are resolved from the METS directory. Missing files and remote
136
- references that cannot be checked locally produce a non-zero exit status. This
137
- is a file-presence check; it does not perform XML schema validation.
138
-
139
122
  ## Development
140
123
 
141
124
  Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
@@ -5,7 +5,6 @@
5
5
 
6
6
  Collection of utils for
7
7
  * evaluation of OCR data for the masses
8
- * generation of extended OCR-Evaluation Corpora
9
8
  * generation of pair-wise Trainingdata for OCR-Backends
10
9
 
11
10
  ## Requirements
@@ -21,16 +20,10 @@ Each section contains detailed usage help instructions:
21
20
  # evaluation
22
21
  ocr-util eval --help
23
22
 
24
- # corpus management
25
- ocr-util corpus --help
26
-
27
- # analyse an existing METS corpus
28
- ocr-util corpus-analyse --help
29
-
30
23
  # slice image by image + input OCR
31
24
  ocr-util slice --help
32
25
 
33
- # render image + input OCR
26
+ # render input OCR (regions, lines, words) on given image
34
27
  ocr-util show --help
35
28
  ```
36
29
 
@@ -40,7 +33,7 @@ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONT
40
33
 
41
34
  _Please note_:
42
35
  Invalid data files are excluded and reported where possible from evaluation.
43
- The term 'invalid' refers to errors in schemas in structured XML-data, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
36
+ The term 'invalid' refers to malformed or structurally invalid XML, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
44
37
 
45
38
  ### Evaluation Filter-Then-Aggregate
46
39
 
@@ -56,6 +49,22 @@ ocr-util eval <candidates> \
56
49
  --aggregate-by "mods:dateIssued:century"
57
50
  ```
58
51
 
52
+ The dimensions can also be reversed: filter by publication century or decade,
53
+ then aggregate by language. For example, keep only publications from the 17th
54
+ century (1601-1700 inclusive), then group their evaluation results by MODS language:
55
+
56
+ ```bash
57
+ ocr-util eval <candidates> \
58
+ --reference <groundtruth> \
59
+ --mets-file <mets.xml> \
60
+ --filter-by "mods:dateIssued:century=17th" \
61
+ --aggregate-by "mods:language"
62
+ ```
63
+
64
+ Filtering still happens before evaluation and aggregation; only the metadata
65
+ dimensions exchange roles. To select a decade instead, replace the filter above
66
+ with `--filter-by "mods:dateIssued:decade=1800s"` (1800-1809 inclusive).
67
+
59
68
  Multi-language filter values are interpreted as sets:
60
69
 
61
70
  ```bash
@@ -67,6 +76,8 @@ ocr-util eval <candidates> \
67
76
  ```
68
77
 
69
78
  Behavior:
79
+ * centuries follow calendar numbering: the 18th century is 1701-1800,
80
+ and the 19th century is 1801-1900; decades retain ten-year buckets such as 1800-1809
70
81
  * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
71
82
  * multi-value filter -> all filter values must be present in any order
72
83
  * entries missing the filter criterion are reported as WARNING and discarded
@@ -75,34 +86,6 @@ Behavior:
75
86
  * aggregation coverage reports how many evaluated pairs provide each dimension;
76
87
  reported `items` count candidate/ground-truth pairs, not MODS elements
77
88
 
78
- ### Corpus Analysis
79
-
80
- Existing METS corpora can be filtered without evaluation candidate data. Matching
81
- full-text file references are printed one per line:
82
-
83
- ```bash
84
- ocr-util corpus-analyse <mets.xml> \
85
- --filter-by "mods:dateIssued:century=16th"
86
- ```
87
-
88
- Repeat `--filter-by` to combine criteria with AND:
89
-
90
- ```bash
91
- ocr-util corpus-analyse <mets.xml> \
92
- --filter-by "mods:dateIssued:century=16th" \
93
- --filter-by "mods:language=ger"
94
- ```
95
-
96
- Check that every local file referenced by any METS `FLocat` exists:
97
-
98
- ```bash
99
- ocr-util corpus-analyse <mets.xml> --check
100
- ```
101
-
102
- Relative paths are resolved from the METS directory. Missing files and remote
103
- references that cannot be checked locally produce a non-zero exit status. This
104
- is a file-presence check; it does not perform XML schema validation.
105
-
106
89
  ## Development
107
90
 
108
91
  Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
@@ -7,8 +7,8 @@ requires-python = ">=3.10"
7
7
  authors = [{name = "Universitäts- und Landesbibliothek Sachsen-Anhalt",email = "development@bibliothek.uni-halle.de"}]
8
8
  classifiers = [
9
9
  "Programming Language :: Python :: 3",
10
- "License :: OSI Approved :: MIT License"
11
10
  ]
11
+ license = "MIT"
12
12
  dependencies = [
13
13
  "rapidfuzz>3",
14
14
  "nltk",
@@ -53,7 +53,7 @@ version = {attr = "ocr_util.__version__"}
53
53
  where = ["src"]
54
54
 
55
55
  [tool.setuptools.package-data]
56
- ocr_util = ["corpus/*.xml"]
56
+ ocr_corpus = ["*.xml"]
57
57
 
58
58
  [tool.setuptools]
59
59
  package-dir = {"" = "src"}
@@ -0,0 +1,3 @@
1
+ """main API"""
2
+
3
+ __version__ = "3.0.1"
@@ -3,9 +3,8 @@
3
3
 
4
4
  import argparse
5
5
  import logging
6
- import os
7
6
  import re
8
- from pathlib import Path, PurePath
7
+ from pathlib import PurePath
9
8
 
10
9
  import ocr_util
11
10
  import ocr_util.eval.model as do
@@ -13,18 +12,10 @@ import ocr_util.eval.model.filter as dofi
13
12
  import ocr_util.eval.cli as eval_cli
14
13
  import ocr_util.slice.cli as slice_cli
15
14
  import ocr_util.show.cli as show_cli
16
- import ocr_util.corpus.analyse as corpus_analyse
17
- import ocr_util.corpus.generate_corpus as gc
18
-
19
- from ocr_util.corpus.common import CorpusArgs
20
15
 
21
16
  # script constants
22
17
  DEFAULT_VERBOSITY = 0
23
18
  SUB_CMD_FRAME = "frame"
24
- SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
25
- SUB_CMD_CORPUS_ANALYSE = "corpus-analyse"
26
- CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
27
- CORPUS_CACHE_DIR = os.path.join(os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME)
28
19
 
29
20
  SUB_CMD_EVALUATE = "eval"
30
21
  SUB_CMD_SLICE = "slice"
@@ -89,77 +80,6 @@ def start() -> None:
89
80
  """,
90
81
  )
91
82
 
92
- # groundtruth-corpus subcommand
93
- groundtruth_corpus_arg_parser = sub_arg_parsers.add_parser(
94
- SUB_CMD_GROUNDTRUTH_CORPUS,
95
- help="Create METS file from N ground truth PAGE-XML files with URN identifiers",
96
- )
97
- groundtruth_corpus_arg_parser.add_argument(
98
- "-i",
99
- "--input",
100
- dest="input_dir",
101
- help="Path to the input directory containing GT PAGE-XML files",
102
- required=True,
103
- )
104
- groundtruth_corpus_arg_parser.add_argument(
105
- "-o",
106
- "--output",
107
- dest="output_dir",
108
- help="Path to the output directory for generated corpus",
109
- required=True,
110
- )
111
- groundtruth_corpus_arg_parser.add_argument(
112
- "-l",
113
- "--limit",
114
- type=int,
115
- default=0,
116
- help="Number of files to process (default: 0 = unlimited)",
117
- required=False,
118
- )
119
- groundtruth_corpus_arg_parser.add_argument(
120
- "-t",
121
- "--temp-dir",
122
- dest="temp_dir",
123
- default=CORPUS_CACHE_DIR,
124
- help=f"Path to temporary directory for caching METS files (default: {CORPUS_CACHE_DIR})",
125
- required=False,
126
- )
127
- groundtruth_corpus_arg_parser.add_argument(
128
- "-v",
129
- "--verbosity",
130
- action="count",
131
- default=DEFAULT_VERBOSITY,
132
- required=False,
133
- help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
134
- )
135
- groundtruth_corpus_arg_parser.add_argument(
136
- "--corpus-label",
137
- dest="corpus_label",
138
- default="Ground Truth Corpus",
139
- help="Label for the corpus in the METS logical structure (default: 'Ground Truth Corpus')",
140
- required=False,
141
- )
142
- groundtruth_corpus_arg_parser.add_argument(
143
- "--oai-base-url",
144
- dest="oai_base_url",
145
- help="Base URL for OAI-PMH requests",
146
- required=False,
147
- )
148
- groundtruth_corpus_arg_parser.add_argument(
149
- "--clear-cache",
150
- dest="clear_cache",
151
- action="store_true",
152
- default=False,
153
- help="Clear the local cache directory before processing (default: False)",
154
- required=False,
155
- )
156
-
157
- corpus_analyse_parser = sub_arg_parsers.add_parser(
158
- SUB_CMD_CORPUS_ANALYSE,
159
- help="List files in an existing METS corpus that match metadata filters",
160
- )
161
- corpus_analyse.register_arguments(corpus_analyse_parser)
162
-
163
83
  # evaluate subcommand
164
84
  evaluate_arg_parser = sub_arg_parsers.add_parser(
165
85
  SUB_CMD_EVALUATE,
@@ -276,24 +196,6 @@ def start() -> None:
276
196
  if verbosity > 0:
277
197
  print("[INFO ] file_result", file_result)
278
198
 
279
- elif args.subcommand == SUB_CMD_GROUNDTRUTH_CORPUS:
280
- corpus_args = CorpusArgs(
281
- input_dir=Path(args.input_dir).absolute(),
282
- output_dir=Path(args.output_dir).absolute(),
283
- local_cache_dir=Path(args.temp_dir).absolute(),
284
- limit=int(args.limit),
285
- corpus_label=args.corpus_label,
286
- clear_cache=args.clear_cache,
287
- )
288
- gc.generate(corpus_args)
289
-
290
- elif args.subcommand == SUB_CMD_CORPUS_ANALYSE:
291
- analyse_args = vars(args)
292
- analyse_args.pop("subcommand", None)
293
- result = corpus_analyse.start_analysis(analyse_args)
294
- if isinstance(result, corpus_analyse.CorpusCheckResult) and not result.is_valid:
295
- raise SystemExit(1)
296
-
297
199
  elif args.subcommand == SUB_CMD_EVALUATE:
298
200
  eval_args = vars(args)
299
201
  eval_args.pop("subcommand", None)
@@ -142,10 +142,10 @@ def decade_transform(value: str) -> typing.Optional[str]:
142
142
 
143
143
 
144
144
  def century_transform(value: str) -> typing.Optional[str]:
145
- """Transform a 4-digit year string into its century bucket (e.g. '1867' → '19th')."""
145
+ """Bucket years by calendar century: 1801 through 1900 belong to the 19th."""
146
146
  try:
147
147
  year = int(str(value).strip()[:4])
148
- century = year // 100 + 1
148
+ century = (year - 1) // 100 + 1
149
149
  suffix = {1: "st", 2: "nd", 3: "rd"}.get(century % 10 if century % 100 not in (11, 12, 13) else 0, "th")
150
150
  return f"{century}{suffix}"
151
151
  except (ValueError, TypeError):
@@ -1,11 +1,11 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ocr-util
3
- Version: 2.2.1
3
+ Version: 3.0.1
4
4
  Summary: OCR Utils
5
5
  Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
6
+ License-Expression: MIT
6
7
  Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
7
8
  Classifier: Programming Language :: Python :: 3
8
- Classifier: License :: OSI Approved :: MIT License
9
9
  Requires-Python: >=3.10
10
10
  Description-Content-Type: text/markdown
11
11
  License-File: LICENSE
@@ -38,7 +38,6 @@ Dynamic: license-file
38
38
 
39
39
  Collection of utils for
40
40
  * evaluation of OCR data for the masses
41
- * generation of extended OCR-Evaluation Corpora
42
41
  * generation of pair-wise Trainingdata for OCR-Backends
43
42
 
44
43
  ## Requirements
@@ -54,16 +53,10 @@ Each section contains detailed usage help instructions:
54
53
  # evaluation
55
54
  ocr-util eval --help
56
55
 
57
- # corpus management
58
- ocr-util corpus --help
59
-
60
- # analyse an existing METS corpus
61
- ocr-util corpus-analyse --help
62
-
63
56
  # slice image by image + input OCR
64
57
  ocr-util slice --help
65
58
 
66
- # render image + input OCR
59
+ # render input OCR (regions, lines, words) on given image
67
60
  ocr-util show --help
68
61
  ```
69
62
 
@@ -73,7 +66,7 @@ Inconsistent OCR Groundtruth with empty texts (ALTO String elements missing CONT
73
66
 
74
67
  _Please note_:
75
68
  Invalid data files are excluded and reported where possible from evaluation.
76
- The term 'invalid' refers to errors in schemas in structured XML-data, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
69
+ The term 'invalid' refers to malformed or structurally invalid XML, i.e. syntax errors and further if included geometrical information includes inconsistencies like missing points or missmatching shapes.
77
70
 
78
71
  ### Evaluation Filter-Then-Aggregate
79
72
 
@@ -89,6 +82,22 @@ ocr-util eval <candidates> \
89
82
  --aggregate-by "mods:dateIssued:century"
90
83
  ```
91
84
 
85
+ The dimensions can also be reversed: filter by publication century or decade,
86
+ then aggregate by language. For example, keep only publications from the 17th
87
+ century (1601-1700 inclusive), then group their evaluation results by MODS language:
88
+
89
+ ```bash
90
+ ocr-util eval <candidates> \
91
+ --reference <groundtruth> \
92
+ --mets-file <mets.xml> \
93
+ --filter-by "mods:dateIssued:century=17th" \
94
+ --aggregate-by "mods:language"
95
+ ```
96
+
97
+ Filtering still happens before evaluation and aggregation; only the metadata
98
+ dimensions exchange roles. To select a decade instead, replace the filter above
99
+ with `--filter-by "mods:dateIssued:decade=1800s"` (1800-1809 inclusive).
100
+
92
101
  Multi-language filter values are interpreted as sets:
93
102
 
94
103
  ```bash
@@ -100,6 +109,8 @@ ocr-util eval <candidates> \
100
109
  ```
101
110
 
102
111
  Behavior:
112
+ * centuries follow calendar numbering: the 18th century is 1701-1800,
113
+ and the 19th century is 1801-1900; decades retain ten-year buckets such as 1800-1809
103
114
  * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
104
115
  * multi-value filter -> all filter values must be present in any order
105
116
  * entries missing the filter criterion are reported as WARNING and discarded
@@ -108,34 +119,6 @@ Behavior:
108
119
  * aggregation coverage reports how many evaluated pairs provide each dimension;
109
120
  reported `items` count candidate/ground-truth pairs, not MODS elements
110
121
 
111
- ### Corpus Analysis
112
-
113
- Existing METS corpora can be filtered without evaluation candidate data. Matching
114
- full-text file references are printed one per line:
115
-
116
- ```bash
117
- ocr-util corpus-analyse <mets.xml> \
118
- --filter-by "mods:dateIssued:century=16th"
119
- ```
120
-
121
- Repeat `--filter-by` to combine criteria with AND:
122
-
123
- ```bash
124
- ocr-util corpus-analyse <mets.xml> \
125
- --filter-by "mods:dateIssued:century=16th" \
126
- --filter-by "mods:language=ger"
127
- ```
128
-
129
- Check that every local file referenced by any METS `FLocat` exists:
130
-
131
- ```bash
132
- ocr-util corpus-analyse <mets.xml> --check
133
- ```
134
-
135
- Relative paths are resolved from the METS directory. Missing files and remote
136
- references that cannot be checked locally produce a non-zero exit status. This
137
- is a file-presence check; it does not perform XML schema validation.
138
-
139
122
  ## Development
140
123
 
141
124
  Platform: Intel(R) Core(TM) i5-6500 CPU@3.20GHz, 16GB RAM, Ubuntu 22.04 LTS, Python 3.10+
@@ -9,12 +9,6 @@ src/ocr_util.egg-info/dependency_links.txt
9
9
  src/ocr_util.egg-info/entry_points.txt
10
10
  src/ocr_util.egg-info/requires.txt
11
11
  src/ocr_util.egg-info/top_level.txt
12
- src/ocr_util/corpus/__init__.py
13
- src/ocr_util/corpus/analyse.py
14
- src/ocr_util/corpus/common.py
15
- src/ocr_util/corpus/generate_corpus.py
16
- src/ocr_util/corpus/load_metadata.py
17
- src/ocr_util/corpus/template.corpus.xml
18
12
  src/ocr_util/eval/__init__.py
19
13
  src/ocr_util/eval/aggregation.py
20
14
  src/ocr_util/eval/cli.py
@@ -45,11 +39,6 @@ src/ocr_util/slice/cli.py
45
39
  src/ocr_util/slice/gts_pairs.py
46
40
  src/ocr_util/slice/pairs_lstmfs.py
47
41
  tests/test_aggregation.py
48
- tests/test_corpus.py
49
- tests/test_corpus_analyse.py
50
- tests/test_corpus_examples.py
51
- tests/test_corpus_fulltext_generation.py
52
- tests/test_corpus_load_metadata.py
53
42
  tests/test_digital_eval_cli.py
54
43
  tests/test_digital_object_model.py
55
44
  tests/test_digital_object_ocr_alto.py
@@ -697,9 +697,25 @@ def test_century_transform_21st():
697
697
  assert digev.century_transform("2026") == "21st"
698
698
 
699
699
 
700
- def test_century_transform_boundary_1900():
701
- """1900 is in the 20th century."""
702
- assert digev.century_transform("1900") == "20th"
700
+ @pytest.mark.parametrize(
701
+ "year, expected",
702
+ [
703
+ ("0001", "1st"),
704
+ ("0100", "1st"),
705
+ ("0101", "2nd"),
706
+ ("1700", "17th"),
707
+ ("1701", "18th"),
708
+ ("1800", "18th"),
709
+ ("1801", "19th"),
710
+ ("1900", "19th"),
711
+ ("1901", "20th"),
712
+ ("2000", "20th"),
713
+ ("2001", "21st"),
714
+ ],
715
+ )
716
+ def test_century_transform_calendar_boundaries(year, expected):
717
+ """Calendar centuries start in year xx01 and end in xx00, not xx00 through xx99."""
718
+ assert digev.century_transform(year) == expected
703
719
 
704
720
 
705
721
  def test_century_transform_11th_century_no_spurious_st():