ocr-util 2.1.1__tar.gz → 2.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. {ocr_util-2.1.1/src/ocr_util.egg-info → ocr_util-2.2.1}/PKG-INFO +36 -1
  2. {ocr_util-2.1.1 → ocr_util-2.2.1}/README.md +35 -0
  3. ocr_util-2.2.1/src/ocr_util/__init__.py +3 -0
  4. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/cli.py +22 -19
  5. ocr_util-2.2.1/src/ocr_util/corpus/analyse.py +173 -0
  6. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/corpus/common.py +1 -3
  7. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/corpus/load_metadata.py +0 -1
  8. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/aggregation.py +9 -27
  9. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/cli.py +90 -90
  10. ocr_util-2.2.1/src/ocr_util/eval/constants.py +27 -0
  11. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +3 -9
  12. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/Util.py +1 -3
  13. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/evaluation.py +47 -88
  14. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/geometry.py +4 -13
  15. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/metrics.py +9 -16
  16. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/common.py +1 -3
  17. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/digital_object_model.py +6 -15
  18. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/digital_object_util.py +3 -9
  19. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/filter.py +1 -4
  20. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/format_alto_v3_util.py +9 -30
  21. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/format_page_util.py +14 -42
  22. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/minidom_util.py +2 -6
  23. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/preprocessing.py +44 -74
  24. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/resolve.py +1 -3
  25. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/show/cli.py +26 -20
  26. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/show/ocr_show_segmentation.py +214 -133
  27. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/slice/cli.py +53 -41
  28. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/slice/gts_pairs.py +109 -114
  29. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/slice/pairs_lstmfs.py +24 -26
  30. {ocr_util-2.1.1 → ocr_util-2.2.1/src/ocr_util.egg-info}/PKG-INFO +36 -1
  31. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/SOURCES.txt +3 -0
  32. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_aggregation.py +8 -26
  33. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_corpus.py +6 -17
  34. ocr_util-2.2.1/tests/test_corpus_analyse.py +108 -0
  35. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_corpus_examples.py +16 -29
  36. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_corpus_load_metadata.py +6 -10
  37. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_digital_eval_cli.py +148 -98
  38. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_digital_object_model.py +12 -18
  39. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_alto.py +56 -61
  40. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_filter.py +43 -41
  41. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_page.py +24 -24
  42. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_digital_object_ocr_page_eynollah.py +5 -5
  43. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_digital_object_txt.py +1 -2
  44. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_generate_sets.py +131 -133
  45. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_ocr_evaluate.py +88 -101
  46. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_ocr_metrics.py +41 -41
  47. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_ocr_metrics_base.py +4 -4
  48. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_ocr_preprocessing.py +8 -8
  49. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_page_reading_order.py +17 -17
  50. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_show_segmentation.py +151 -133
  51. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_version_compatibility.py +10 -12
  52. ocr_util-2.1.1/src/ocr_util/__init__.py +0 -3
  53. {ocr_util-2.1.1 → ocr_util-2.2.1}/LICENSE +0 -0
  54. {ocr_util-2.1.1 → ocr_util-2.2.1}/pyproject.toml +0 -0
  55. {ocr_util-2.1.1 → ocr_util-2.2.1}/setup.cfg +0 -0
  56. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/corpus/__init__.py +0 -0
  57. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/corpus/generate_corpus.py +0 -0
  58. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/corpus/template.corpus.xml +0 -0
  59. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/__init__.py +0 -0
  60. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/__init__.py +0 -0
  61. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/common.py +0 -0
  62. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
  63. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/dictionary_metrics/language_tool/common.py +0 -0
  64. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/eval/model/main.py +0 -0
  65. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util/slice/__init__.py +0 -0
  66. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/dependency_links.txt +0 -0
  67. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/entry_points.txt +0 -0
  68. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/requires.txt +0 -0
  69. {ocr_util-2.1.1 → ocr_util-2.2.1}/src/ocr_util.egg-info/top_level.txt +0 -0
  70. {ocr_util-2.1.1 → ocr_util-2.2.1}/tests/test_corpus_fulltext_generation.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ocr-util
3
- Version: 2.1.1
3
+ Version: 2.2.1
4
4
  Summary: OCR Utils
5
5
  Author-email: Universitäts- und Landesbibliothek Sachsen-Anhalt <development@bibliothek.uni-halle.de>
6
6
  Project-URL: Homepage, https://github.com/ulb-sachsen-anhalt/ocr-util
@@ -57,6 +57,9 @@ ocr-util eval --help
57
57
  # corpus management
58
58
  ocr-util corpus --help
59
59
 
60
+ # analyse an existing METS corpus
61
+ ocr-util corpus-analyse --help
62
+
60
63
  # slice image by image + input OCR
61
64
  ocr-util slice --help
62
65
 
@@ -100,6 +103,38 @@ Behavior:
100
103
  * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
101
104
  * multi-value filter -> all filter values must be present in any order
102
105
  * entries missing the filter criterion are reported as WARNING and discarded
106
+ * METS coverage reports how many evaluation pairs map to full-text references
107
+ and how many METS references have no evaluation pair
108
+ * aggregation coverage reports how many evaluated pairs provide each dimension;
109
+ reported `items` count candidate/ground-truth pairs, not MODS elements
110
+
111
+ ### Corpus Analysis
112
+
113
+ Existing METS corpora can be filtered without evaluation candidate data. Matching
114
+ full-text file references are printed one per line:
115
+
116
+ ```bash
117
+ ocr-util corpus-analyse <mets.xml> \
118
+ --filter-by "mods:dateIssued:century=16th"
119
+ ```
120
+
121
+ Repeat `--filter-by` to combine criteria with AND:
122
+
123
+ ```bash
124
+ ocr-util corpus-analyse <mets.xml> \
125
+ --filter-by "mods:dateIssued:century=16th" \
126
+ --filter-by "mods:language=ger"
127
+ ```
128
+
129
+ Check that every local file referenced by any METS `FLocat` exists:
130
+
131
+ ```bash
132
+ ocr-util corpus-analyse <mets.xml> --check
133
+ ```
134
+
135
+ Relative paths are resolved from the METS directory. Missing files and remote
136
+ references that cannot be checked locally produce a non-zero exit status. This
137
+ is a file-presence check; it does not perform XML schema validation.
103
138
 
104
139
  ## Development
105
140
 
@@ -24,6 +24,9 @@ ocr-util eval --help
24
24
  # corpus management
25
25
  ocr-util corpus --help
26
26
 
27
+ # analyse an existing METS corpus
28
+ ocr-util corpus-analyse --help
29
+
27
30
  # slice image by image + input OCR
28
31
  ocr-util slice --help
29
32
 
@@ -67,6 +70,38 @@ Behavior:
67
70
  * single filter value -> exact match (e.g. `ger` does not match `ger+eng`)
68
71
  * multi-value filter -> all filter values must be present in any order
69
72
  * entries missing the filter criterion are reported as WARNING and discarded
73
+ * METS coverage reports how many evaluation pairs map to full-text references
74
+ and how many METS references have no evaluation pair
75
+ * aggregation coverage reports how many evaluated pairs provide each dimension;
76
+ reported `items` count candidate/ground-truth pairs, not MODS elements
77
+
78
+ ### Corpus Analysis
79
+
80
+ Existing METS corpora can be filtered without evaluation candidate data. Matching
81
+ full-text file references are printed one per line:
82
+
83
+ ```bash
84
+ ocr-util corpus-analyse <mets.xml> \
85
+ --filter-by "mods:dateIssued:century=16th"
86
+ ```
87
+
88
+ Repeat `--filter-by` to combine criteria with AND:
89
+
90
+ ```bash
91
+ ocr-util corpus-analyse <mets.xml> \
92
+ --filter-by "mods:dateIssued:century=16th" \
93
+ --filter-by "mods:language=ger"
94
+ ```
95
+
96
+ Check that every local file referenced by any METS `FLocat` exists:
97
+
98
+ ```bash
99
+ ocr-util corpus-analyse <mets.xml> --check
100
+ ```
101
+
102
+ Relative paths are resolved from the METS directory. Missing files and remote
103
+ references that cannot be checked locally produce a non-zero exit status. This
104
+ is a file-presence check; it does not perform XML schema validation.
70
105
 
71
106
  ## Development
72
107
 
@@ -0,0 +1,3 @@
1
+ """main API"""
2
+
3
+ __version__ = "2.2.1"
@@ -13,6 +13,7 @@ import ocr_util.eval.model.filter as dofi
13
13
  import ocr_util.eval.cli as eval_cli
14
14
  import ocr_util.slice.cli as slice_cli
15
15
  import ocr_util.show.cli as show_cli
16
+ import ocr_util.corpus.analyse as corpus_analyse
16
17
  import ocr_util.corpus.generate_corpus as gc
17
18
 
18
19
  from ocr_util.corpus.common import CorpusArgs
@@ -21,10 +22,9 @@ from ocr_util.corpus.common import CorpusArgs
21
22
  DEFAULT_VERBOSITY = 0
22
23
  SUB_CMD_FRAME = "frame"
23
24
  SUB_CMD_GROUNDTRUTH_CORPUS = "corpus"
25
+ SUB_CMD_CORPUS_ANALYSE = "corpus-analyse"
24
26
  CORPUS_CACHE_DIR_NAME = "ocr_util_corpus_mets_cache"
25
- CORPUS_CACHE_DIR = os.path.join(
26
- os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME
27
- )
27
+ CORPUS_CACHE_DIR = os.path.join(os.path.expanduser("~"), ".cache", CORPUS_CACHE_DIR_NAME)
28
28
 
29
29
  SUB_CMD_EVALUATE = "eval"
30
30
  SUB_CMD_SLICE = "slice"
@@ -44,9 +44,7 @@ def points_type(points: str) -> str:
44
44
  def start() -> None:
45
45
  # Configure logging once, centrally
46
46
  logging.basicConfig(
47
- level=logging.INFO,
48
- format='%(asctime)s [%(levelname)s][%(name)s] %(message)s',
49
- datefmt='%Y-%m-%d %H:%M:%S'
47
+ level=logging.INFO, format="%(asctime)s [%(levelname)s][%(name)s] %(message)s", datefmt="%Y-%m-%d %H:%M:%S"
50
48
  )
51
49
  arg_parser: argparse.ArgumentParser = argparse.ArgumentParser(
52
50
  prog="ocr-util",
@@ -72,9 +70,7 @@ def start() -> None:
72
70
  required=False,
73
71
  help=f"Verbosity flag. To increase, append multiple 'v's (optional; default: '{DEFAULT_VERBOSITY}')",
74
72
  )
75
- frame_arg_parser.add_argument(
76
- "-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True
77
- )
73
+ frame_arg_parser.add_argument("-i", "--input-ocr-file", help="Path of OCR-Data file to process", required=True)
78
74
  frame_arg_parser.add_argument(
79
75
  "-o",
80
76
  "--output-ocr-file",
@@ -158,6 +154,12 @@ def start() -> None:
158
154
  required=False,
159
155
  )
160
156
 
157
+ corpus_analyse_parser = sub_arg_parsers.add_parser(
158
+ SUB_CMD_CORPUS_ANALYSE,
159
+ help="List files in an existing METS corpus that match metadata filters",
160
+ )
161
+ corpus_analyse.register_arguments(corpus_analyse_parser)
162
+
161
163
  # evaluate subcommand
162
164
  evaluate_arg_parser = sub_arg_parsers.add_parser(
163
165
  SUB_CMD_EVALUATE,
@@ -232,9 +234,7 @@ def start() -> None:
232
234
  default=slice_cli.DEFAULT_SANITIZE,
233
235
  help=f"optional: sanitize textline images (default: {slice_cli.DEFAULT_SANITIZE})",
234
236
  )
235
- slice_arg_parser.add_argument(
236
- "--no-sanitize", dest="sanitize", action="store_false"
237
- )
237
+ slice_arg_parser.add_argument("--no-sanitize", dest="sanitize", action="store_false")
238
238
  slice_arg_parser.add_argument(
239
239
  "--intrusion-ratio",
240
240
  required=False,
@@ -269,12 +269,8 @@ def start() -> None:
269
269
  output_ocr_file: str = args.output_ocr_file
270
270
  points: str = args.points
271
271
  if verbosity > 1:
272
- print(
273
- f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}"
274
- )
275
- polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(
276
- input_ocr_file, points, verbosity
277
- )
272
+ print(f"[DEBUG] args: {input_ocr_file}, {output_ocr_file}, {points}, {verbosity}")
273
+ polygon_frame_filter: dofi.PolygonFrameFilter = dofi.PolygonFrameFilter(input_ocr_file, points, verbosity)
278
274
  piece_result: do.DigitalObjectTree = polygon_frame_filter.process()
279
275
  file_result: PurePath = do.from_digital_object(piece_result, output_ocr_file)
280
276
  if verbosity > 0:
@@ -287,10 +283,17 @@ def start() -> None:
287
283
  local_cache_dir=Path(args.temp_dir).absolute(),
288
284
  limit=int(args.limit),
289
285
  corpus_label=args.corpus_label,
290
- clear_cache=args.clear_cache
286
+ clear_cache=args.clear_cache,
291
287
  )
292
288
  gc.generate(corpus_args)
293
289
 
290
+ elif args.subcommand == SUB_CMD_CORPUS_ANALYSE:
291
+ analyse_args = vars(args)
292
+ analyse_args.pop("subcommand", None)
293
+ result = corpus_analyse.start_analysis(analyse_args)
294
+ if isinstance(result, corpus_analyse.CorpusCheckResult) and not result.is_valid:
295
+ raise SystemExit(1)
296
+
294
297
  elif args.subcommand == SUB_CMD_EVALUATE:
295
298
  eval_args = vars(args)
296
299
  eval_args.pop("subcommand", None)
@@ -0,0 +1,173 @@
1
+ """Analyse a METS corpus with evaluation-compatible filters."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import dataclasses
7
+ import re
8
+ import typing
9
+ from pathlib import Path
10
+ from urllib.parse import unquote, urlparse
11
+
12
+ import lxml.etree as ET
13
+
14
+ import ocr_util.eval as digev
15
+ from ocr_util.eval.cli import _build_filter_spec, _filter_value_matches
16
+
17
+ METS_NAMESPACES = digev.METSModsExtractor.DEFAULT_NAMESPACES
18
+
19
+
20
+ @dataclasses.dataclass(frozen=True)
21
+ class CorpusCheckResult:
22
+ """summarize local file references found in a METS corpus."""
23
+
24
+ checked: tuple[Path, ...]
25
+ missing: tuple[Path, ...]
26
+ uncheckable: tuple[str, ...]
27
+
28
+ @property
29
+ def is_valid(self) -> bool:
30
+ """return whether every METS file reference is locally present."""
31
+ return not self.missing and not self.uncheckable
32
+
33
+
34
+ def _groundtruth_type(path: Path) -> typing.Optional[str]:
35
+ """infer the evaluation ground-truth type from a corpus file name."""
36
+ match = re.match(r".*(?:gt\.(\w{3,})|\.(\w{3,})\.gt)\.xml$", path.name)
37
+ label = next((group for group in match.groups() if group), None) if match else None
38
+ if label and label.startswith("art"):
39
+ return "article"
40
+ if label and label.startswith("ann"):
41
+ return "announcement"
42
+ return None
43
+
44
+
45
+ def _corpus_entries(mets_file: Path) -> list[digev.EvalEntry]:
46
+ """create entry-like objects for full-text files referenced by a METS corpus."""
47
+ tree = ET.parse(str(mets_file))
48
+ entries: list[digev.EvalEntry] = []
49
+
50
+ for file_group in tree.xpath("//mets:fileGrp", namespaces=METS_NAMESPACES):
51
+ if "FULLTEXT" not in (file_group.get("USE") or "").upper():
52
+ continue
53
+
54
+ for href in file_group.xpath("./mets:file/mets:FLocat/@xlink:href", namespaces=METS_NAMESPACES):
55
+ referenced_path = Path(str(href))
56
+ entry = digev.EvalEntry(referenced_path)
57
+ entry.path_groundtruth = referenced_path
58
+ entry.domain_directories = list(reversed(referenced_path.parent.parts))
59
+ entry.gt_type = _groundtruth_type(referenced_path) or entry.gt_type
60
+ entries.append(entry)
61
+
62
+ return entries
63
+
64
+
65
+ def check_corpus(mets_file: Path) -> CorpusCheckResult:
66
+ """check whether every file referenced by the METS corpus is present locally."""
67
+ mets_file = Path(mets_file)
68
+ if not mets_file.is_file():
69
+ raise FileNotFoundError(f"METS file not found: {mets_file}")
70
+
71
+ tree = ET.parse(str(mets_file))
72
+ hrefs = tree.xpath("//mets:FLocat/@xlink:href", namespaces=METS_NAMESPACES)
73
+ checked: list[Path] = []
74
+ missing: list[Path] = []
75
+ uncheckable: list[str] = []
76
+
77
+ for href_value in hrefs:
78
+ href = str(href_value)
79
+ parsed = urlparse(href)
80
+ if parsed.scheme not in ("", "file") or (parsed.netloc not in ("", "localhost")):
81
+ uncheckable.append(href)
82
+ continue
83
+
84
+ path = Path(unquote(parsed.path))
85
+ if not path.is_absolute():
86
+ path = mets_file.parent / path
87
+ checked.append(path)
88
+ if not path.is_file():
89
+ missing.append(path)
90
+
91
+ return CorpusCheckResult(tuple(checked), tuple(missing), tuple(uncheckable))
92
+
93
+
94
+ def analyse(mets_file: Path, filters: typing.Sequence[str]) -> list[str]:
95
+ """return corpus file references matching all evaluation filter specifications."""
96
+ mets_file = Path(mets_file)
97
+ if not mets_file.is_file():
98
+ raise FileNotFoundError(f"METS file not found: {mets_file}")
99
+
100
+ entries = _corpus_entries(mets_file)
101
+ for filter_by in filters:
102
+ filter_spec = _build_filter_spec(filter_by, mets_file, strict=True)
103
+ if filter_spec is None:
104
+ raise ValueError(f"Invalid filter specification: '{filter_by}'")
105
+ _, extractor, expected_value = filter_spec
106
+ entries = [
107
+ entry
108
+ for entry in entries
109
+ if (value := extractor(entry)) is not None
110
+ and str(value).strip()
111
+ and _filter_value_matches(str(value), expected_value)
112
+ ]
113
+
114
+ return [str(entry.path_groundtruth) for entry in entries]
115
+
116
+
117
+ def register_arguments(parser: argparse.ArgumentParser) -> None:
118
+ """register command-line arguments for corpus analysis."""
119
+ parser.add_argument("mets_file", type=Path, help="Path to the corpus METS file")
120
+ parser.add_argument(
121
+ "--filter-by",
122
+ action="append",
123
+ default=[],
124
+ help=(
125
+ "Evaluation-compatible filter; repeat to combine filters with AND, "
126
+ "for example 'mods:dateIssued:century=16th'"
127
+ ),
128
+ )
129
+ parser.add_argument(
130
+ "--check",
131
+ action="store_true",
132
+ help="Check that every local file referenced by the METS corpus exists",
133
+ )
134
+
135
+
136
+ def start_analysis(
137
+ args: typing.Mapping[str, typing.Any],
138
+ ) -> typing.Union[list[str], CorpusCheckResult]:
139
+ """analyse a corpus and print each matching file reference to stdout."""
140
+ if args.get("check"):
141
+ result = check_corpus(Path(args["mets_file"]))
142
+ missing = set(result.missing)
143
+ for path in result.checked:
144
+ status = "MISSING" if path in missing else "OK"
145
+ print(f"{status} {path}")
146
+ for href in result.uncheckable:
147
+ print(f"UNCHECKABLE {href}")
148
+ print(
149
+ f"SUMMARY checked={len(result.checked)} missing={len(result.missing)} "
150
+ f"uncheckable={len(result.uncheckable)}"
151
+ )
152
+ return result
153
+
154
+ if not args.get("filter_by"):
155
+ raise ValueError("at least one --filter-by or --check is required")
156
+
157
+ matches = analyse(Path(args["mets_file"]), args["filter_by"])
158
+ for match in matches:
159
+ print(match)
160
+ return matches
161
+
162
+
163
+ def start() -> None:
164
+ """run the standalone corpus analysis CLI."""
165
+ parser = argparse.ArgumentParser()
166
+ register_arguments(parser)
167
+ result = start_analysis(vars(parser.parse_args()))
168
+ if isinstance(result, CorpusCheckResult) and not result.is_valid:
169
+ raise SystemExit(1)
170
+
171
+
172
+ if __name__ == "__main__":
173
+ start()
@@ -419,9 +419,7 @@ class MetsResourceFile(MetsFile):
419
419
  assert file_fulltext is not None, f"No fulltext file pointer for page with CONTENTIDS='{page_urn}'"
420
420
  return file_image, file_fulltext
421
421
 
422
-
423
- def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path,
424
- out_dir: pathlib.Path) -> ET._Element:
422
+ def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path, out_dir: pathlib.Path) -> ET._Element:
425
423
  """Create a new file element for the fulltext file with the appropriate FLocat child."""
426
424
  file_fulltext = ET.Element(
427
425
  f'{{{self.nsmap["mets"]}}}file', attrib={"ID": new_id, "MIMETYPE": "application/vnd.prima.page+xml"}
@@ -37,7 +37,6 @@ FALLBACK_IDX_BASE_URLS = {
37
37
  }
38
38
 
39
39
 
40
-
41
40
  @dataclasses.dataclass(frozen=True)
42
41
  class RecordResolutionResult:
43
42
  """Normalized result of a handle lookup attempt.
@@ -146,9 +146,7 @@ def century_transform(value: str) -> typing.Optional[str]:
146
146
  try:
147
147
  year = int(str(value).strip()[:4])
148
148
  century = year // 100 + 1
149
- suffix = {1: "st", 2: "nd", 3: "rd"}.get(
150
- century % 10 if century % 100 not in (11, 12, 13) else 0, "th"
151
- )
149
+ suffix = {1: "st", 2: "nd", 3: "rd"}.get(century % 10 if century % 100 not in (11, 12, 13) else 0, "th")
152
150
  return f"{century}{suffix}"
153
151
  except (ValueError, TypeError):
154
152
  return None
@@ -247,9 +245,7 @@ class METSDivAttrExtractor:
247
245
 
248
246
  # physical div ID → [file hrefs]
249
247
  phys_to_hrefs: typing.Dict[str, typing.List[str]] = {}
250
- for pdiv in tree.xpath(
251
- '//mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ID]', namespaces=ns
252
- ):
248
+ for pdiv in tree.xpath('//mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ID]', namespaces=ns):
253
249
  pid = pdiv.get("ID")
254
250
  hrefs = [
255
251
  file_id_to_href[fid]
@@ -263,9 +259,7 @@ class METSDivAttrExtractor:
263
259
  log_to_attr: typing.Dict[str, str] = {}
264
260
  # also: DMDID → attribute value (fallback)
265
261
  dmdid_to_attr: typing.Dict[str, str] = {}
266
- for ldiv in tree.xpath(
267
- '//mets:structMap[@TYPE="LOGICAL"]//mets:div[@ID]', namespaces=ns
268
- ):
262
+ for ldiv in tree.xpath('//mets:structMap[@TYPE="LOGICAL"]//mets:div[@ID]', namespaces=ns):
269
263
  lid = ldiv.get("ID")
270
264
  attr_val = ldiv.get(self.attribute)
271
265
  if lid and attr_val:
@@ -412,9 +406,7 @@ class METSModsExtractor:
412
406
  files = tree.xpath("//mets:file", namespaces=self.namespaces)
413
407
  for file_elem in files:
414
408
  file_id = file_elem.get("ID")
415
- flocat = file_elem.xpath(
416
- "./mets:FLocat/@xlink:href", namespaces=self.namespaces
417
- )
409
+ flocat = file_elem.xpath("./mets:FLocat/@xlink:href", namespaces=self.namespaces)
418
410
  if not flocat:
419
411
  continue
420
412
 
@@ -456,9 +448,7 @@ class METSModsExtractor:
456
448
  continue
457
449
 
458
450
  hrefs = []
459
- for file_id in physical_div.xpath(
460
- "./mets:fptr/@FILEID", namespaces=self.namespaces
461
- ):
451
+ for file_id in physical_div.xpath("./mets:fptr/@FILEID", namespaces=self.namespaces):
462
452
  file_href = file_id_to_href.get(file_id)
463
453
  if file_href:
464
454
  hrefs.append(file_href)
@@ -467,9 +457,7 @@ class METSModsExtractor:
467
457
  physical_div_to_hrefs[physical_id] = hrefs
468
458
 
469
459
  xlink_ns = self.namespaces.get("xlink", "http://www.w3.org/1999/xlink")
470
- for link in tree.xpath(
471
- "//mets:structLink/mets:smLink", namespaces=self.namespaces
472
- ):
460
+ for link in tree.xpath("//mets:structLink/mets:smLink", namespaces=self.namespaces):
473
461
  logical_id = link.get(f"{{{xlink_ns}}}from")
474
462
  physical_id = link.get(f"{{{xlink_ns}}}to")
475
463
  if not logical_id or not physical_id:
@@ -510,9 +498,7 @@ class METSModsExtractor:
510
498
  # Apply user's XPath expression to MODS section
511
499
  for mods_section in mods_sections:
512
500
  try:
513
- results = mods_section.xpath(
514
- self.xpath_expression, namespaces=self.namespaces
515
- )
501
+ results = mods_section.xpath(self.xpath_expression, namespaces=self.namespaces)
516
502
  if results:
517
503
  values = []
518
504
  for result in results:
@@ -583,15 +569,11 @@ class AggregationStrategy:
583
569
  hierarchical: If True, creates hierarchical keys combining all dimensions
584
570
  """
585
571
 
586
- def __init__(
587
- self, dimensions: typing.List[AggregationDimension], hierarchical: bool = False
588
- ):
572
+ def __init__(self, dimensions: typing.List[AggregationDimension], hierarchical: bool = False):
589
573
  self.dimensions = dimensions
590
574
  self.hierarchical = hierarchical
591
575
 
592
- def generate_keys(
593
- self, entry: typing.Any, metric: digem.OCRMetric
594
- ) -> typing.List[str]:
576
+ def generate_keys(self, entry: typing.Any, metric: digem.OCRMetric) -> typing.List[str]:
595
577
  """Generate aggregation keys for an evaluation entry
596
578
 
597
579
  Args: