ocr-util 2.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. ocr_util/__init__.py +3 -0
  2. ocr_util/cli.py +311 -0
  3. ocr_util/corpus/__init__.py +17 -0
  4. ocr_util/corpus/common.py +522 -0
  5. ocr_util/corpus/generate_corpus.py +149 -0
  6. ocr_util/corpus/load_metadata.py +325 -0
  7. ocr_util/corpus/template.corpus.xml +24 -0
  8. ocr_util/eval/__init__.py +34 -0
  9. ocr_util/eval/aggregation.py +634 -0
  10. ocr_util/eval/cli.py +827 -0
  11. ocr_util/eval/dictionary_metrics/__init__.py +0 -0
  12. ocr_util/eval/dictionary_metrics/common.py +84 -0
  13. ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +116 -0
  14. ocr_util/eval/dictionary_metrics/language_tool/Util.py +91 -0
  15. ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
  16. ocr_util/eval/dictionary_metrics/language_tool/common.py +49 -0
  17. ocr_util/eval/evaluation.py +628 -0
  18. ocr_util/eval/geometry.py +117 -0
  19. ocr_util/eval/metrics.py +271 -0
  20. ocr_util/eval/model/common.py +107 -0
  21. ocr_util/eval/model/digital_object_model.py +257 -0
  22. ocr_util/eval/model/digital_object_util.py +120 -0
  23. ocr_util/eval/model/filter.py +130 -0
  24. ocr_util/eval/model/format_alto_v3_util.py +247 -0
  25. ocr_util/eval/model/format_page_util.py +267 -0
  26. ocr_util/eval/model/main.py +16 -0
  27. ocr_util/eval/model/minidom_util.py +46 -0
  28. ocr_util/eval/preprocessing.py +439 -0
  29. ocr_util/eval/resolve.py +55 -0
  30. ocr_util/show/cli.py +66 -0
  31. ocr_util/show/ocr_show_segmentation.py +396 -0
  32. ocr_util/slice/__init__.py +4 -0
  33. ocr_util/slice/cli.py +225 -0
  34. ocr_util/slice/gts_pairs.py +702 -0
  35. ocr_util/slice/pairs_lstmfs.py +77 -0
  36. ocr_util-2.0.1.dist-info/METADATA +136 -0
  37. ocr_util-2.0.1.dist-info/RECORD +41 -0
  38. ocr_util-2.0.1.dist-info/WHEEL +5 -0
  39. ocr_util-2.0.1.dist-info/entry_points.txt +2 -0
  40. ocr_util-2.0.1.dist-info/licenses/LICENSE +21 -0
  41. ocr_util-2.0.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,522 @@
1
+ """Shared data models and METS processing utilities for corpus generation.
2
+
3
+ This module provides the building blocks used by :mod:`ocr_util.corpus.generate_corpus`
4
+ to assemble a METS-based ground truth corpus:
5
+
6
+ * **Data classes** – :class:`CorpusArgs`, :class:`GroundtruthFile`,
7
+ :class:`CorpusPageInput`, :class:`MetsModsSection`,
8
+ :class:`CorpusGeneratorResult`.
9
+ * **METS file abstractions**
10
+
11
+ * :class:`MetsFile` – base class for reading and querying a METS XML document.
12
+ * :class:`MetsCorpusFile` – extends :class:`MetsFile` to represent the
13
+ growing output corpus METS document; exposes :meth:`~MetsCorpusFile.attach`
14
+ and :meth:`~MetsCorpusFile.write`.
15
+ * :class:`MetsResourceFile` – extends :class:`MetsFile` to extract page-level
16
+ METS/MODS sections from a published source METS document.
17
+
18
+ Constants
19
+ ---------
20
+ EASY_URN_XML_PATTERN
21
+ Regular expression that identifies PAGE-XML files carrying a URN-based
22
+ page identifier in their file name
23
+ (e.g. ``urn+nbn+de+gbv+3+5-12345-fp-00000001.xml``).
24
+ METS_MEDIA_TYPES
25
+ Set of METS logical structure ``TYPE`` attribute values that are expected
26
+ to carry a ``DMDID`` reference to the descriptive metadata section.
27
+ """
28
+
29
+ import copy
30
+ import dataclasses
31
+ import hashlib
32
+ import logging
33
+ import os
34
+ import pathlib
35
+ import re
36
+ import shutil
37
+ import typing
38
+
39
+ import lxml.etree as ET
40
+
41
+ logger = logging.getLogger(__name__)
42
+
43
+
44
+ DEFAULT_FULLEXT_FILEGROUP = "FULLTEXT"
45
+ DEFAULT_IMAGE_FILEGROUP = "MAX"
46
+
47
+ LABEL_FILEGROUP_IMAGE = "GT-IMAGE"
48
+ KWARG_LABEL_FILEGROUP_FULLTEXT = "label_filegroup_ocr"
49
+
50
+ GT_TARGET_SUBDIR = "GT-PAGE"
51
+ GT_METS_FILEGROUP_FULLTEXT = "GT-FULLTEXT"
52
+ GT_METS_FILEGROUP_IMAGE = "GT-IMAGE"
53
+ DEFAULT_METS_FILE_NAME = "mets.xml"
54
+ INDENT = 4
55
+
56
+ DMDID_GLUE = "_"
57
+
58
+ EASY_URN_XML_PATTERN = r"^(urn[\+\-\w]+)-fp-(\w+).xml$"
59
+
60
+ METS_MEDIA_TYPES = {
61
+ "monograph",
62
+ "volume",
63
+ "issue",
64
+ "additional",
65
+ }
66
+
67
+
68
+ class CorpusException(Exception):
69
+ """Base exception for corpus-related errors."""
70
+
71
+
72
+ @dataclasses.dataclass
73
+ class CorpusArgs:
74
+ """Arguments for corpus generation."""
75
+
76
+ input_dir: pathlib.Path
77
+ output_dir: pathlib.Path
78
+ local_cache_dir: pathlib.Path
79
+ limit: int = 0
80
+ clear_cache: bool = False
81
+ corpus_label: str = "Ground Truth Corpus"
82
+
83
+
84
+ @dataclasses.dataclass
85
+ class GroundtruthFile:
86
+ """Represents local ground truth file resource."""
87
+
88
+ identifier: str
89
+ file_base_name: str
90
+ file_path: pathlib.Path
91
+ relative_file_path: pathlib.Path
92
+ languages: typing.Optional[typing.List[str]] = None
93
+
94
+ @classmethod
95
+ def from_dir_copy(cls, in_dir: pathlib.Path, out_dir: pathlib.Path, limit: int = 0) -> typing.List:
96
+ """Scan directory for ground truth files, copy to output directory, and return list of resources."""
97
+ gt_resources: typing.List[GroundtruthFile] = cls.from_dir(in_dir, limit)
98
+ for gt_resource in gt_resources:
99
+ src_abs_path: pathlib.Path = gt_resource.file_path
100
+ src_rel_path: pathlib.Path = gt_resource.file_path.relative_to(in_dir)
101
+ dest_abs_path: pathlib.Path = out_dir.joinpath(src_rel_path).absolute()
102
+ os.makedirs(dest_abs_path.parent, exist_ok=True)
103
+ shutil.copy2(src_abs_path, dest_abs_path)
104
+ gt_resource.file_path = dest_abs_path
105
+ gt_resource.relative_file_path = src_rel_path
106
+ return gt_resources
107
+
108
+ @classmethod
109
+ def from_dir(cls, gt_dir: pathlib.Path, limit: int = 0) -> typing.List:
110
+ """Scan directory for ground truth files and return list of resources."""
111
+ resources: typing.List[GroundtruthFile] = []
112
+ current_dir: str
113
+ # child_dirs: typing.List[str]
114
+ files: typing.List[str]
115
+ for current_dir, _, files in os.walk(gt_dir, topdown=False):
116
+ for file in files:
117
+ file_path: pathlib.Path = pathlib.Path(current_dir).joinpath(file)
118
+ the_match: typing.Optional[typing.Match[str]] = re.match(EASY_URN_XML_PATTERN, file_path.name)
119
+ if the_match is not None:
120
+ urn_enc = f"{the_match.group(1)}/fragment/page={the_match.group(2)}"
121
+ urn_dec: str = urn_enc.replace("+", ":").replace("x", "X")
122
+ gt_file = GroundtruthFile(
123
+ identifier=urn_dec,
124
+ file_base_name=the_match.group(1),
125
+ file_path=file_path,
126
+ relative_file_path=file_path.relative_to(gt_dir),
127
+ )
128
+ resources.append(gt_file)
129
+ if 0 < limit <= len(resources):
130
+ break
131
+ else:
132
+ logger.warning(
133
+ "File name '%s' does not match pattern '%s'; ignoring file",
134
+ file_path.name,
135
+ EASY_URN_XML_PATTERN,
136
+ )
137
+ break
138
+ return sorted(resources, key=lambda r: r.file_path.name)
139
+
140
+
141
+ @dataclasses.dataclass
142
+ class MetsModsSection:
143
+ """Represents extracted METS/MODS sections for a given page."""
144
+
145
+ phys_div: ET._Element
146
+ file_image: ET._Element
147
+ file_fulltext: ET._Element
148
+ sm_link: ET._Element
149
+ log_div: ET._Element
150
+ dmd_sec: typing.Optional[ET._Element] = None
151
+ calculated_identifier_hash: typing.Optional[str] = None
152
+
153
+
154
+ class MetsFile:
155
+ """Represents a published METS file resource."""
156
+
157
+ def __init__(self, file_path: pathlib.Path, **kwargs):
158
+ self.file_path = file_path
159
+ self.kwargs = kwargs
160
+ self.label_fgroup_fulltext = self.kwargs.get("label_filegroup_ocr", DEFAULT_FULLEXT_FILEGROUP)
161
+ self.label_fgroup_image = self.kwargs.get("label_filegroup_image", DEFAULT_IMAGE_FILEGROUP)
162
+ self.logical_root: typing.Optional[ET._Element] = None
163
+ self.physical_root: typing.Optional[ET._Element] = None
164
+ self.document = file_path # trigger setter
165
+
166
+ @property
167
+ def document(self) -> ET._ElementTree:
168
+ """Get the METS XML document representing the corpus template."""
169
+ return self.__document
170
+
171
+ @document.setter
172
+ def document(self, path_document: pathlib.Path) -> None:
173
+ """Set the METS XML document and extract relevant sections."""
174
+ self.__document = ET.parse(path_document, ET.XMLParser(remove_blank_text=True))
175
+ the_root = self.__document.getroot()
176
+ # Extract namespaces directly using xpath; filter out empty/None prefixes
177
+ raw_ns = typing.cast(typing.List[typing.Tuple[typing.Optional[str], str]], the_root.xpath("//namespace::*"))
178
+ self.__nsmap = {prefix: uri for prefix, uri in raw_ns if prefix}
179
+ logical_root = self.document.find('.//mets:structMap[@TYPE="LOGICAL"]', self.nsmap)
180
+ if logical_root is None:
181
+ raise CorpusException(f"Missing logical structMap in METS file '{self.file_path}'")
182
+ self.logical_root = logical_root
183
+ physical_root = self.document.find('.//mets:structMap[@TYPE="PHYSICAL"]/mets:div', self.nsmap)
184
+ if physical_root is not None:
185
+ self.physical_root = physical_root
186
+ self.file_path = path_document
187
+
188
+ @property
189
+ def nsmap(self) -> dict:
190
+ """Get the namespace mapping extracted from the METS document."""
191
+ return self.__nsmap
192
+
193
+ def find_dmd_id(self) -> str:
194
+ """Find DMDID in logical structMap of given published METS file."""
195
+
196
+ assert self.logical_root is not None, f"No logical structMap in METS file {self.file_path}"
197
+ logical_children = self.logical_root.findall(".//mets:div", self.nsmap)
198
+ for element in logical_children:
199
+ the_type = element.get("TYPE", None)
200
+ dmd_id = element.get("DMDID", None)
201
+ if dmd_id is not None and the_type in METS_MEDIA_TYPES:
202
+ return dmd_id
203
+ raise CorpusException(f"no DMD_ID in METS file {self.file_path} found")
204
+
205
+
206
+ class MetsCorpusFile(MetsFile):
207
+ """Extends and modifies base MetsFile class to represent a METS file resource
208
+ for a given Corpus Page with additional carries required information."""
209
+
210
+ def __init__(self, file_path: pathlib.Path, **kwargs):
211
+ super().__init__(file_path, **kwargs)
212
+ self.label_fgroup_fulltext = self.kwargs.get("corpus_filegroup_ocr", GT_METS_FILEGROUP_FULLTEXT)
213
+ self.label_fgroup_image = self.kwargs.get("corpus_filegroup_image", GT_METS_FILEGROUP_IMAGE)
214
+ self.corpus_file_name = self.kwargs.get("corpus_file_name", DEFAULT_METS_FILE_NAME)
215
+ # for the corpus METS file, go a little bit further down the logical line
216
+ assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
217
+ self.logical_root = self.logical_root.find("mets:div", self.nsmap)
218
+ assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
219
+ corpus_label = self.kwargs.get("corpus_label")
220
+ if corpus_label:
221
+ self.logical_root.set("LABEL", str(corpus_label))
222
+ self._prepare_filegroups()
223
+ the_root = self.document.getroot()
224
+ prev_phys_root = the_root.findall('.//mets:div[@ID="physroot"]', self.nsmap)
225
+ if len(prev_phys_root) == 0:
226
+ physroot = ET.Element(f'{{{self.nsmap["mets"]}}}div', attrib={"ID": "physroot", "TYPE": "physSequence"})
227
+ assert self.physical_root is not None, f"Missing physical structMap in METS file '{self.file_path}'"
228
+ self.physical_root.append(physroot)
229
+ self.link_root = self.document.find(".//mets:structLink", self.nsmap)
230
+
231
+ def _prepare_filegroups(self):
232
+ xpr_fulltext = f'mets:fileGrp[@USE="{self.label_fgroup_fulltext}"]'
233
+ self.file_group_fulltext = self._get_or_create_filegroup(
234
+ "fileGrp", xpr_fulltext, {"USE": self.label_fgroup_fulltext}
235
+ )
236
+ xpr_image = f'mets:fileGrp[@USE="{self.label_fgroup_image}"]'
237
+ self.file_group_image = self._get_or_create_filegroup("fileGrp", xpr_image, {"USE": self.label_fgroup_image})
238
+
239
+ def _get_or_create_filegroup(self, tag: str, xpr: str, attrib: dict) -> ET._Element:
240
+ """Get or create an XML element with the specified tag and attributes."""
241
+ filesec_root = self.document.getroot().find("mets:fileSec", self.nsmap)
242
+ if filesec_root is None:
243
+ filesec_root = ET.Element(f'{{{self.nsmap["mets"]}}}fileSec')
244
+ self.document.getroot().append(filesec_root)
245
+ file_group = filesec_root.find(xpr, self.nsmap)
246
+ if file_group is None:
247
+ file_group = ET.Element(f'{{{self.nsmap["mets"]}}}{tag}', attrib=attrib)
248
+ filesec_root.append(file_group)
249
+ return file_group
250
+
251
+ def attach(self, extract: MetsModsSection) -> None:
252
+ """Attach extracted METS sections to the corpus METS document
253
+ but attach MODS only if not already present in the corpus METS file."""
254
+
255
+ self.file_group_image.append(extract.file_image)
256
+ self.file_group_fulltext.append(extract.file_fulltext)
257
+ assert self.physical_root is not None, f"Missing physical structMap in METS file '{self.file_path}'"
258
+ self.physical_root.append(extract.phys_div)
259
+ assert self.link_root is not None, f"Missing structLink in METS file '{self.file_path}'"
260
+ self.link_root.append(extract.sm_link)
261
+ assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
262
+ self.logical_root.append(extract.log_div)
263
+ corpus_root = self.document.getroot()
264
+ assert extract.dmd_sec is not None, "DMD section missing in section for attachment"
265
+ xtrct_dmd_id = extract.dmd_sec.get("ID")
266
+ # look for extracted dmd_id in corpus file
267
+ if corpus_root.find(f'.//mets:dmdSec[@ID="{xtrct_dmd_id}"]', namespaces=self.nsmap) is None:
268
+ # if descriptive metadata not yet present in corpus file, append dmd_sec from extract
269
+ logger.info(
270
+ "Attach new DMD section with ID='%s' to corpus METS file",
271
+ xtrct_dmd_id,
272
+ )
273
+ if xtrct_dmd_id is not None and extract.dmd_sec is not None:
274
+ idx: int = len(self.document.findall(".//mets:dmdSec", namespaces=self.nsmap))
275
+ self.document.getroot().insert(idx, extract.dmd_sec)
276
+ else:
277
+ logger.info(
278
+ "DMD section with ID='%s' already present in corpus METS file",
279
+ xtrct_dmd_id,
280
+ )
281
+
282
+ def finalize(self):
283
+ """Re-order attached gt-image containers"""
284
+ the_root = self.document.getroot()
285
+ pages_with_order: list[ET._Element] = the_root.findall(
286
+ './/mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ORDER]', self.nsmap
287
+ )
288
+ for i, elm in enumerate(pages_with_order, 1):
289
+ elm.set("ORDER", f"{i}")
290
+ assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
291
+ log_divs_with_order: list[ET._Element] = self.logical_root.findall(".//mets:div[@ORDER]", self.nsmap)
292
+ for i, elm in enumerate(log_divs_with_order, 1):
293
+ elm.set("ORDER", f"{i}")
294
+
295
+ def write(self, out_dir: pathlib.Path) -> pathlib.Path:
296
+ """Write the METS XML document to the specified file path."""
297
+ encoding = self.document.docinfo.encoding if self.document.docinfo.encoding else "UTF-8"
298
+ ET.indent(self.document.getroot(), space=(" " * INDENT))
299
+ out_file: pathlib.Path = out_dir.joinpath(self.corpus_file_name)
300
+ self.document.write(out_file, xml_declaration=True, pretty_print=True, encoding=encoding)
301
+ return out_file
302
+
303
+
304
+ class MetsResourceFile(MetsFile):
305
+ """Represents a METS file resource for extraction."""
306
+
307
+ def __init__(self, path_mets_file, idx: int, **kwargs):
308
+ super().__init__(path_mets_file, **kwargs)
309
+ self.mdsec_identifier: typing.Optional[str] = None
310
+ self.idx = idx
311
+
312
+ @property
313
+ def identifier_hash(self) -> str:
314
+ """Calculate and return a 8-character hash based on the METS/MODS resource section identifier."""
315
+ if self.mdsec_identifier is None:
316
+ raise CorpusException(f"Identifier for METS resource file {self.file_path} not set")
317
+ return hashlib.sha256(self.mdsec_identifier.encode()).hexdigest()[:8]
318
+
319
+ def extract(
320
+ self,
321
+ page_urn: str,
322
+ out_dir: pathlib.Path,
323
+ gt_file_path: pathlib.Path,
324
+ ) -> MetsModsSection:
325
+ """Grab required information and resolve relationships starting from
326
+ the page div with given CONTENTIDS in the METS file."""
327
+
328
+ the_root = self.document.getroot()
329
+ page_div = the_root.find(f'.//mets:div[@CONTENTIDS="{page_urn}"]', self.nsmap)
330
+ assert page_div is not None, f"no page with CONTENTIDS='{page_urn}' found"
331
+ file_image, file_fulltext = self._set_page_with_files(the_root, page_div, gt_file_path, out_dir)
332
+
333
+ # now resolve relationships to get logical div and dmdSec for the page
334
+ # PHYSICAL
335
+ source_phys_id = page_div.get("ID")
336
+ assert source_phys_id is not None, f"page div with CONTENTIDS='{page_urn}' missing ID attribute"
337
+ # LINK
338
+ sm_link = the_root.find(f'.//mets:smLink[@xlink:to="{source_phys_id}"]', self.nsmap)
339
+ assert sm_link is not None, f"smLink for page div with ID='{source_phys_id}' not found"
340
+ # LOGICAL
341
+ log_id = sm_link.get(f'{{{self.nsmap["xlink"]}}}from')
342
+ assert log_id is not None, f"log ID for smLink with to='{source_phys_id}' not found"
343
+ log_div = the_root.find(f'.//mets:div[@ID="{log_id}"]', self.nsmap)
344
+ assert log_div is not None, f"log div with ID='{log_id}' not found"
345
+ # clean up logical div: remove un-related attributes like "AMDID", remove all child elements
346
+ if log_div.get("AMDID") is not None:
347
+ log_div.attrib.pop("AMDID")
348
+ for child in log_div.getchildren():
349
+ child.getparent().remove(child)
350
+ # fix current DMDID for identifier calculation
351
+ source_dmd_id = self.find_dmd_id()
352
+ source_dmd_sec = the_root.find(f'.//mets:dmdSec[@ID="{source_dmd_id}"]', self.nsmap)
353
+ assert source_dmd_sec is not None, f"DMD section with ID='{source_dmd_id}' not found"
354
+ copy_dmd_sec = self._build_dmd_section(source_dmd_sec)
355
+ section = MetsModsSection(
356
+ phys_div=page_div,
357
+ file_image=file_image,
358
+ file_fulltext=file_fulltext,
359
+ sm_link=sm_link,
360
+ log_div=log_div,
361
+ dmd_sec=copy_dmd_sec,
362
+ calculated_identifier_hash=self.identifier_hash,
363
+ )
364
+ self._reindex(section)
365
+ return section
366
+
367
+ def _set_page_with_files(
368
+ self, the_root: ET._Element, page_div: ET._Element, gt_file_path: pathlib.Path, out_dir: pathlib.Path
369
+ ) -> typing.Tuple[ET._Element, ET._Element]:
370
+ """Re-Create page container with the same attributes as the source page but with modified children."""
371
+
372
+ file_pointers: list[ET._Element] = page_div.findall("mets:fptr", namespaces=self.nsmap)
373
+ # Remove all child elements from actual page
374
+ for el in file_pointers:
375
+ el_pa = el.getparent()
376
+ if el_pa is not None:
377
+ el_pa.remove(el)
378
+ # now re-attach selected file pointers for FULLTEXT and MAX image
379
+ file_image = None
380
+ file_fulltext = None
381
+ for fp in file_pointers:
382
+ the_file = the_root.find(f'.//mets:file[@ID="{fp.get("FILEID")}"]', self.nsmap)
383
+ assert the_file is not None, f"no file with ID='{fp.get('FILEID')}' found in {self.document.docinfo.URL}"
384
+ the_file_group = the_file.xpath("ancestor::mets:fileGrp/@USE", namespaces=self.nsmap)
385
+ assert len(the_file_group) == 1, f"file with ID='{fp.get('FILEID')}' invalid parent fileGrp"
386
+ the_group = the_file_group[0]
387
+ if the_group not in {self.label_fgroup_fulltext, self.label_fgroup_image}:
388
+ continue
389
+
390
+ # here we go
391
+ new_id = f"{the_group}-{(self.idx):04d}"
392
+ if the_group == self.label_fgroup_image:
393
+ prev_img_fileid = fp.get("FILEID")
394
+ file_image = the_root.find(f'.//mets:file[@ID="{prev_img_fileid}"]', self.nsmap)
395
+ assert file_image is not None, f"no file with ID='{prev_img_fileid}' found"
396
+ pre_file_id = file_image.get("ID")
397
+ logger.debug(
398
+ "Re-assigning file ID '%s' to '%s' for page with CONTENTIDS='%s'",
399
+ pre_file_id,
400
+ new_id,
401
+ page_div.get("CONTENTIDS"),
402
+ )
403
+ file_image.set("ID", new_id)
404
+ fp.set("FILEID", new_id)
405
+ page_div.append(fp)
406
+ elif the_group == self.label_fgroup_fulltext:
407
+ file_ptr_fulltext = ET.Element(f'{{{self.nsmap["mets"]}}}fptr', attrib={"FILEID": new_id})
408
+ page_div.append(file_ptr_fulltext)
409
+ file_fulltext = self._create_fulltext_element(new_id, gt_file_path, out_dir)
410
+ # if no file pointer for fulltext exists, generate it since print may do not possess
411
+ if file_fulltext is None:
412
+ new_id = f"{self.label_fgroup_fulltext}-{(self.idx):04d}"
413
+ file_fulltext = self._create_fulltext_element(new_id, gt_file_path, out_dir)
414
+ file_ptr_fulltext = ET.Element(f'{{{self.nsmap["mets"]}}}fptr', attrib={"FILEID": new_id})
415
+ page_div.append(file_ptr_fulltext)
416
+
417
+ page_urn = page_div.get("CONTENTIDS")
418
+ assert file_image is not None, f"No image file pointer for page with CONTENTIDS='{page_urn}'"
419
+ assert file_fulltext is not None, f"No fulltext file pointer for page with CONTENTIDS='{page_urn}'"
420
+ return file_image, file_fulltext
421
+
422
+
423
+ def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path,
424
+ out_dir: pathlib.Path) -> ET._Element:
425
+ """Create a new file element for the fulltext file with the appropriate FLocat child."""
426
+ file_fulltext = ET.Element(
427
+ f'{{{self.nsmap["mets"]}}}file', attrib={"ID": new_id, "MIMETYPE": "application/vnd.prima.page+xml"}
428
+ )
429
+ file_fulltext.append(
430
+ ET.Element(
431
+ f'{{{self.nsmap["mets"]}}}FLocat',
432
+ attrib={
433
+ f'{{{self.nsmap["xlink"]}}}href': str(gt_file_path.relative_to(out_dir)),
434
+ "LOCTYPE": "OTHER",
435
+ "OTHERLOCTYPE": "FILE",
436
+ },
437
+ )
438
+ )
439
+ return file_fulltext
440
+
441
+ def _build_dmd_section(self, source_dmd_sec: ET._Element) -> ET._Element:
442
+
443
+ source_mods_root = source_dmd_sec.find(".//mods:mods", self.nsmap)
444
+ assert source_mods_root is not None, f"no MODS root in DMD section with ID='{source_dmd_sec.get('ID')}' found"
445
+ identifer_elements: list[ET._Element] = source_mods_root.findall("mods:identifier", self.nsmap)
446
+ identifer_urn: typing.List[str] = [i.text for i in identifer_elements if i.get("type") == "urn"]
447
+ if not identifer_urn:
448
+ raise CorpusException(
449
+ f"No identifier with type='urn' in MODS metadata of METS file {self.document.docinfo.URL} found"
450
+ )
451
+ self.mdsec_identifier = identifer_urn[0]
452
+ copy_dmd_sec = copy.deepcopy(
453
+ source_dmd_sec
454
+ ) # create deep copy of dmd_sec to avoid modifying the original METS file
455
+ copy_root = copy_dmd_sec.find(".//mods:mods", self.nsmap)
456
+ assert copy_root is not None, f"no MODS root in DMD section with ID='{copy_dmd_sec.get('ID')}' found"
457
+ # clear copy: remove all copy child elements
458
+ for child in copy_root.getchildren():
459
+ child.getparent().remove(child)
460
+
461
+ copy_root.extend(identifer_elements)
462
+ title_elements: list[ET._Element] = source_mods_root.findall("mods:titleInfo", self.nsmap)
463
+ if len(title_elements) > 0:
464
+ copy_root.extend(title_elements)
465
+ else:
466
+ title_host = source_mods_root.find("mods:relatedItem/mods:titleInfo", self.nsmap)
467
+ if title_host is not None:
468
+ copy_root.append(title_host)
469
+ language_elements: list[ET._Element] = source_mods_root.findall("mods:language", self.nsmap)
470
+ if len(language_elements) > 0:
471
+ copy_root.extend(language_elements)
472
+ genre_elements: list[ET._Element] = source_mods_root.findall("mods:genre", self.nsmap)
473
+ if len(genre_elements) > 0:
474
+ copy_root.extend(genre_elements)
475
+ publication_info = source_mods_root.find('mods:originInfo[@eventType="publication"]', self.nsmap)
476
+ if publication_info is not None:
477
+ copy_root.append(publication_info)
478
+ access_info = source_mods_root.find("mods:accessCondition", self.nsmap)
479
+ if access_info is not None:
480
+ copy_root.append(access_info)
481
+ return copy_dmd_sec
482
+
483
+ def _reindex(self, section: MetsModsSection) -> None:
484
+ """Re-assign IDs for phys_div, sm_link and log_div in the given section."""
485
+ new_phys_id = f"PHYS-{(self.idx):04d}"
486
+ prev_phys_id = section.phys_div.get("ID")
487
+ logger.debug("Re-assigning phys div ID '%s' to '%s'", prev_phys_id, new_phys_id)
488
+ new_log_id = f"LOG-{(self.idx):04d}"
489
+ prev_log_id = section.log_div.get("ID")
490
+ logger.debug("Re-assigning log div ID '%s' to '%s'", prev_log_id, new_log_id)
491
+ section.phys_div.set("ID", new_phys_id)
492
+ section.sm_link.set(f'{{{self.nsmap["xlink"]}}}to', new_phys_id)
493
+ section.sm_link.set(f'{{{self.nsmap["xlink"]}}}from', new_log_id)
494
+ section.log_div.set("ID", new_log_id)
495
+ assert section.dmd_sec is not None, "DMD section missing in section for re-indexing"
496
+ prev_dmdid = section.dmd_sec.get("ID")
497
+ new_dmdid = f"{prev_dmdid}{DMDID_GLUE}{section.calculated_identifier_hash}"
498
+ section.log_div.set("DMDID", new_dmdid)
499
+ section.dmd_sec.set("ID", new_dmdid)
500
+ logger.debug("Re-assigning DMDID '%s' to '%s'", prev_dmdid, new_dmdid)
501
+
502
+
503
+ @dataclasses.dataclass
504
+ class CorpusPageInput:
505
+ """Encapsulate input data required for corpus generation."""
506
+
507
+ identifier_urn: str
508
+ groundtruth_file: GroundtruthFile
509
+ cached_media_mets_file: typing.Optional[pathlib.Path] = None
510
+ metadata: typing.Optional[MetsResourceFile] = None
511
+
512
+ def __init__(self, groundtruth_file: GroundtruthFile):
513
+ self.groundtruth_file = groundtruth_file
514
+ self.identifier_urn = groundtruth_file.identifier
515
+
516
+
517
+ @dataclasses.dataclass
518
+ class CorpusGeneratorResult:
519
+ """Encapsulate final Corpus generation result."""
520
+
521
+ file_path: pathlib.Path
522
+ n_pages: int
@@ -0,0 +1,149 @@
1
+ """Ground truth corpus generation for OCR evaluation.
2
+
3
+ This module orchestrates the assembly of a METS-based ground truth corpus
4
+ from a collection of PAGE-XML files with URN identifiers. The main entry
5
+ point is :func:`generate`, which accepts a :class:`~ocr_util.corpus.common.CorpusArgs`
6
+ configuration object and returns a :class:`~ocr_util.corpus.common.CorpusGeneratorResult`.
7
+
8
+ Typical call chain
9
+ ------------------
10
+ 1. :func:`generate` scans the input directory for PAGE-XML ground truth files.
11
+ 2. METS metadata is fetched (and cached) in parallel via
12
+ :class:`~ocr_util.corpus.load_metadata.RecordMetadataResolver`.
13
+ 3. :class:`Corpus` iterates over the resolved inputs, extracts the relevant
14
+ METS/MODS sections via
15
+ :class:`~ocr_util.corpus.common.MetsResourceFile`, and attaches them to a
16
+ shared corpus METS document built from a template.
17
+ 4. The finished METS document is written to the output directory.
18
+
19
+ """
20
+
21
+ import argparse
22
+ import logging
23
+ import math
24
+ import os
25
+ import pathlib
26
+ import shutil
27
+ import typing
28
+
29
+ from concurrent.futures import ThreadPoolExecutor
30
+ from pathlib import Path
31
+ from typing import Final
32
+
33
+ import ocr_util.corpus.common as cc
34
+ import ocr_util.corpus.load_metadata as lr
35
+
36
+ CPUS = os.cpu_count() or 1
37
+ NUM_THREADS: typing.Final[int] = math.ceil(CPUS * 0.85)
38
+ DEFAULT_TEMP_DIR = pathlib.Path.home().joinpath(".cache", "ocr_util_corpus_metadata")
39
+ DEFAULT_LIMIT = 0
40
+
41
+ logger = logging.getLogger(__name__)
42
+
43
+
44
+ def fetch_resources(cache_path: pathlib.Path, gt_files: list[cc.CorpusPageInput]) -> list[cc.CorpusPageInput]:
45
+ """Fetch required METS metadata parallel for given inputs."""
46
+ if not cache_path.exists():
47
+ cache_path.mkdir()
48
+ resolver: lr.RecordMetadataResolver = lr.RecordMetadataResolver()
49
+ # Use ThreadPoolExecutor directly for parallel execution
50
+ with ThreadPoolExecutor(max_workers=NUM_THREADS) as executor:
51
+ futures = [executor.submit(resolver.fetch, an_input, cache_path=cache_path) for an_input in gt_files]
52
+ return [future.result() for future in futures]
53
+
54
+
55
+ class Corpus:
56
+ """Ground Truth Corpus in METS format."""
57
+
58
+ def __init__(self, cargs: cc.CorpusArgs, inputs: list[cc.CorpusPageInput], **kwargs) -> None:
59
+ self.__cargs = cargs
60
+ self.__out_dir: Final[Path] = self.__cargs.output_dir
61
+ self.__inputs: Final[list[cc.CorpusPageInput]] = inputs
62
+ # Template file is located in the same directory as this module
63
+ template_path = Path(__file__).parent / "template.corpus.xml"
64
+ self.corpus_template = template_path
65
+ self.kwargs = kwargs
66
+ self.corpus_label = self.__cargs.corpus_label
67
+ self.corpus_file = cc.MetsCorpusFile(template_path, corpus_label=self.corpus_label, **kwargs)
68
+
69
+ def build(self) -> cc.CorpusGeneratorResult:
70
+ """Build the corpus by extracting data from original METS files and write new METS file."""
71
+
72
+ total: int = len(self.__inputs)
73
+ assert self.corpus_file is not None, "Corpus template must be set before start building corpus"
74
+ for idx, page_input in enumerate(self.__inputs, 1):
75
+ logger.info("[%d/%d] Process file with identifier %s", idx, total, page_input.identifier_urn)
76
+ try:
77
+ assert (
78
+ page_input.cached_media_mets_file is not None
79
+ ), f"METS file for {page_input.identifier_urn} not found"
80
+ mets_res = cc.MetsResourceFile(page_input.cached_media_mets_file, idx, **self.kwargs)
81
+ page_input.metadata = mets_res
82
+ the_section: cc.MetsModsSection = mets_res.extract(
83
+ page_urn=page_input.identifier_urn,
84
+ out_dir=self.__out_dir,
85
+ gt_file_path=page_input.groundtruth_file.file_path,
86
+ )
87
+ self.corpus_file.attach(the_section)
88
+ except Exception as exc:
89
+ logger.exception(
90
+ "Error processing %s: %s - skip file",
91
+ page_input.cached_media_mets_file,
92
+ exc,
93
+ )
94
+ # re-sort pages
95
+ self.corpus_file.finalize()
96
+ out_path = self.corpus_file.write(self.__out_dir)
97
+ return cc.CorpusGeneratorResult(file_path=out_path, n_pages=len(self.__inputs))
98
+
99
+
100
+ def generate(cargs: cc.CorpusArgs) -> cc.CorpusGeneratorResult:
101
+ """Generate GT corpus in METS format from given corpus arguments."""
102
+ if not cargs.input_dir.exists():
103
+ raise cc.CorpusException(f"Input directory '{cargs.input_dir}' does not exist")
104
+ if cargs.output_dir.exists():
105
+ logger.warning(
106
+ "Output directory '%s' already exists. Refusing to overwrite existing data.",
107
+ cargs.output_dir,
108
+ )
109
+ if cargs.local_cache_dir.exists() and cargs.clear_cache:
110
+ logger.info("Wipe existing cache directory %s", cargs.local_cache_dir)
111
+ shutil.rmtree(cargs.local_cache_dir)
112
+ cargs.local_cache_dir.mkdir(parents=True, exist_ok=True)
113
+ cargs.output_dir.mkdir(parents=True, exist_ok=True)
114
+ gt_files: list[cc.GroundtruthFile] = cc.GroundtruthFile.from_dir_copy(
115
+ in_dir=cargs.input_dir, out_dir=cargs.output_dir.joinpath(cc.GT_TARGET_SUBDIR), limit=cargs.limit
116
+ )
117
+ corpus_inputs = [cc.CorpusPageInput(groundtruth_file=gt_file) for gt_file in gt_files]
118
+ local_cache_path = Path(f"{cargs.local_cache_dir}").joinpath("mets")
119
+ corpus_input_with_resources = fetch_resources(local_cache_path, corpus_inputs)
120
+ corpus_file = Corpus(cargs, corpus_input_with_resources)
121
+ return corpus_file.build()
122
+
123
+
124
+ if __name__ == "__main__":
125
+ parser: argparse.ArgumentParser = argparse.ArgumentParser()
126
+ parser.add_argument("input", help="Path to the input directory")
127
+ parser.add_argument("output", help="Path to the output directory")
128
+ parser.add_argument(
129
+ "-l",
130
+ "--limit",
131
+ help="Number of Files being processed, default = 0 (unlimited)",
132
+ required=False,
133
+ type=int,
134
+ default=DEFAULT_LIMIT,
135
+ )
136
+ parser.add_argument(
137
+ "-t", "--temp-dir", help="Path to the temporary directory", required=False, default=DEFAULT_TEMP_DIR
138
+ )
139
+ args = parser.parse_args()
140
+ corpus_args = cc.CorpusArgs(
141
+ input_dir=Path(args.input).absolute(),
142
+ output_dir=Path(args.output).absolute(),
143
+ local_cache_dir=Path(args.temp_dir).absolute(),
144
+ limit=int(args.limit),
145
+ corpus_label="Ground Truth Corpus",
146
+ )
147
+
148
+ outcome = generate(corpus_args)
149
+ logger.info("Corpus generated at %s with %d pages", outcome.file_path, outcome.n_pages)