ocr-util 2.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ocr_util/__init__.py +3 -0
- ocr_util/cli.py +311 -0
- ocr_util/corpus/__init__.py +17 -0
- ocr_util/corpus/common.py +522 -0
- ocr_util/corpus/generate_corpus.py +149 -0
- ocr_util/corpus/load_metadata.py +325 -0
- ocr_util/corpus/template.corpus.xml +24 -0
- ocr_util/eval/__init__.py +34 -0
- ocr_util/eval/aggregation.py +634 -0
- ocr_util/eval/cli.py +827 -0
- ocr_util/eval/dictionary_metrics/__init__.py +0 -0
- ocr_util/eval/dictionary_metrics/common.py +84 -0
- ocr_util/eval/dictionary_metrics/language_tool/LanguageTool.py +116 -0
- ocr_util/eval/dictionary_metrics/language_tool/Util.py +91 -0
- ocr_util/eval/dictionary_metrics/language_tool/__init__.py +0 -0
- ocr_util/eval/dictionary_metrics/language_tool/common.py +49 -0
- ocr_util/eval/evaluation.py +628 -0
- ocr_util/eval/geometry.py +117 -0
- ocr_util/eval/metrics.py +271 -0
- ocr_util/eval/model/common.py +107 -0
- ocr_util/eval/model/digital_object_model.py +257 -0
- ocr_util/eval/model/digital_object_util.py +120 -0
- ocr_util/eval/model/filter.py +130 -0
- ocr_util/eval/model/format_alto_v3_util.py +247 -0
- ocr_util/eval/model/format_page_util.py +267 -0
- ocr_util/eval/model/main.py +16 -0
- ocr_util/eval/model/minidom_util.py +46 -0
- ocr_util/eval/preprocessing.py +439 -0
- ocr_util/eval/resolve.py +55 -0
- ocr_util/show/cli.py +66 -0
- ocr_util/show/ocr_show_segmentation.py +396 -0
- ocr_util/slice/__init__.py +4 -0
- ocr_util/slice/cli.py +225 -0
- ocr_util/slice/gts_pairs.py +702 -0
- ocr_util/slice/pairs_lstmfs.py +77 -0
- ocr_util-2.0.1.dist-info/METADATA +136 -0
- ocr_util-2.0.1.dist-info/RECORD +41 -0
- ocr_util-2.0.1.dist-info/WHEEL +5 -0
- ocr_util-2.0.1.dist-info/entry_points.txt +2 -0
- ocr_util-2.0.1.dist-info/licenses/LICENSE +21 -0
- ocr_util-2.0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,522 @@
|
|
|
1
|
+
"""Shared data models and METS processing utilities for corpus generation.
|
|
2
|
+
|
|
3
|
+
This module provides the building blocks used by :mod:`ocr_util.corpus.generate_corpus`
|
|
4
|
+
to assemble a METS-based ground truth corpus:
|
|
5
|
+
|
|
6
|
+
* **Data classes** – :class:`CorpusArgs`, :class:`GroundtruthFile`,
|
|
7
|
+
:class:`CorpusPageInput`, :class:`MetsModsSection`,
|
|
8
|
+
:class:`CorpusGeneratorResult`.
|
|
9
|
+
* **METS file abstractions**
|
|
10
|
+
|
|
11
|
+
* :class:`MetsFile` – base class for reading and querying a METS XML document.
|
|
12
|
+
* :class:`MetsCorpusFile` – extends :class:`MetsFile` to represent the
|
|
13
|
+
growing output corpus METS document; exposes :meth:`~MetsCorpusFile.attach`
|
|
14
|
+
and :meth:`~MetsCorpusFile.write`.
|
|
15
|
+
* :class:`MetsResourceFile` – extends :class:`MetsFile` to extract page-level
|
|
16
|
+
METS/MODS sections from a published source METS document.
|
|
17
|
+
|
|
18
|
+
Constants
|
|
19
|
+
---------
|
|
20
|
+
EASY_URN_XML_PATTERN
|
|
21
|
+
Regular expression that identifies PAGE-XML files carrying a URN-based
|
|
22
|
+
page identifier in their file name
|
|
23
|
+
(e.g. ``urn+nbn+de+gbv+3+5-12345-fp-00000001.xml``).
|
|
24
|
+
METS_MEDIA_TYPES
|
|
25
|
+
Set of METS logical structure ``TYPE`` attribute values that are expected
|
|
26
|
+
to carry a ``DMDID`` reference to the descriptive metadata section.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
import copy
|
|
30
|
+
import dataclasses
|
|
31
|
+
import hashlib
|
|
32
|
+
import logging
|
|
33
|
+
import os
|
|
34
|
+
import pathlib
|
|
35
|
+
import re
|
|
36
|
+
import shutil
|
|
37
|
+
import typing
|
|
38
|
+
|
|
39
|
+
import lxml.etree as ET
|
|
40
|
+
|
|
41
|
+
logger = logging.getLogger(__name__)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
DEFAULT_FULLEXT_FILEGROUP = "FULLTEXT"
|
|
45
|
+
DEFAULT_IMAGE_FILEGROUP = "MAX"
|
|
46
|
+
|
|
47
|
+
LABEL_FILEGROUP_IMAGE = "GT-IMAGE"
|
|
48
|
+
KWARG_LABEL_FILEGROUP_FULLTEXT = "label_filegroup_ocr"
|
|
49
|
+
|
|
50
|
+
GT_TARGET_SUBDIR = "GT-PAGE"
|
|
51
|
+
GT_METS_FILEGROUP_FULLTEXT = "GT-FULLTEXT"
|
|
52
|
+
GT_METS_FILEGROUP_IMAGE = "GT-IMAGE"
|
|
53
|
+
DEFAULT_METS_FILE_NAME = "mets.xml"
|
|
54
|
+
INDENT = 4
|
|
55
|
+
|
|
56
|
+
DMDID_GLUE = "_"
|
|
57
|
+
|
|
58
|
+
EASY_URN_XML_PATTERN = r"^(urn[\+\-\w]+)-fp-(\w+).xml$"
|
|
59
|
+
|
|
60
|
+
METS_MEDIA_TYPES = {
|
|
61
|
+
"monograph",
|
|
62
|
+
"volume",
|
|
63
|
+
"issue",
|
|
64
|
+
"additional",
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class CorpusException(Exception):
|
|
69
|
+
"""Base exception for corpus-related errors."""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclasses.dataclass
|
|
73
|
+
class CorpusArgs:
|
|
74
|
+
"""Arguments for corpus generation."""
|
|
75
|
+
|
|
76
|
+
input_dir: pathlib.Path
|
|
77
|
+
output_dir: pathlib.Path
|
|
78
|
+
local_cache_dir: pathlib.Path
|
|
79
|
+
limit: int = 0
|
|
80
|
+
clear_cache: bool = False
|
|
81
|
+
corpus_label: str = "Ground Truth Corpus"
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclasses.dataclass
|
|
85
|
+
class GroundtruthFile:
|
|
86
|
+
"""Represents local ground truth file resource."""
|
|
87
|
+
|
|
88
|
+
identifier: str
|
|
89
|
+
file_base_name: str
|
|
90
|
+
file_path: pathlib.Path
|
|
91
|
+
relative_file_path: pathlib.Path
|
|
92
|
+
languages: typing.Optional[typing.List[str]] = None
|
|
93
|
+
|
|
94
|
+
@classmethod
|
|
95
|
+
def from_dir_copy(cls, in_dir: pathlib.Path, out_dir: pathlib.Path, limit: int = 0) -> typing.List:
|
|
96
|
+
"""Scan directory for ground truth files, copy to output directory, and return list of resources."""
|
|
97
|
+
gt_resources: typing.List[GroundtruthFile] = cls.from_dir(in_dir, limit)
|
|
98
|
+
for gt_resource in gt_resources:
|
|
99
|
+
src_abs_path: pathlib.Path = gt_resource.file_path
|
|
100
|
+
src_rel_path: pathlib.Path = gt_resource.file_path.relative_to(in_dir)
|
|
101
|
+
dest_abs_path: pathlib.Path = out_dir.joinpath(src_rel_path).absolute()
|
|
102
|
+
os.makedirs(dest_abs_path.parent, exist_ok=True)
|
|
103
|
+
shutil.copy2(src_abs_path, dest_abs_path)
|
|
104
|
+
gt_resource.file_path = dest_abs_path
|
|
105
|
+
gt_resource.relative_file_path = src_rel_path
|
|
106
|
+
return gt_resources
|
|
107
|
+
|
|
108
|
+
@classmethod
|
|
109
|
+
def from_dir(cls, gt_dir: pathlib.Path, limit: int = 0) -> typing.List:
|
|
110
|
+
"""Scan directory for ground truth files and return list of resources."""
|
|
111
|
+
resources: typing.List[GroundtruthFile] = []
|
|
112
|
+
current_dir: str
|
|
113
|
+
# child_dirs: typing.List[str]
|
|
114
|
+
files: typing.List[str]
|
|
115
|
+
for current_dir, _, files in os.walk(gt_dir, topdown=False):
|
|
116
|
+
for file in files:
|
|
117
|
+
file_path: pathlib.Path = pathlib.Path(current_dir).joinpath(file)
|
|
118
|
+
the_match: typing.Optional[typing.Match[str]] = re.match(EASY_URN_XML_PATTERN, file_path.name)
|
|
119
|
+
if the_match is not None:
|
|
120
|
+
urn_enc = f"{the_match.group(1)}/fragment/page={the_match.group(2)}"
|
|
121
|
+
urn_dec: str = urn_enc.replace("+", ":").replace("x", "X")
|
|
122
|
+
gt_file = GroundtruthFile(
|
|
123
|
+
identifier=urn_dec,
|
|
124
|
+
file_base_name=the_match.group(1),
|
|
125
|
+
file_path=file_path,
|
|
126
|
+
relative_file_path=file_path.relative_to(gt_dir),
|
|
127
|
+
)
|
|
128
|
+
resources.append(gt_file)
|
|
129
|
+
if 0 < limit <= len(resources):
|
|
130
|
+
break
|
|
131
|
+
else:
|
|
132
|
+
logger.warning(
|
|
133
|
+
"File name '%s' does not match pattern '%s'; ignoring file",
|
|
134
|
+
file_path.name,
|
|
135
|
+
EASY_URN_XML_PATTERN,
|
|
136
|
+
)
|
|
137
|
+
break
|
|
138
|
+
return sorted(resources, key=lambda r: r.file_path.name)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
@dataclasses.dataclass
|
|
142
|
+
class MetsModsSection:
|
|
143
|
+
"""Represents extracted METS/MODS sections for a given page."""
|
|
144
|
+
|
|
145
|
+
phys_div: ET._Element
|
|
146
|
+
file_image: ET._Element
|
|
147
|
+
file_fulltext: ET._Element
|
|
148
|
+
sm_link: ET._Element
|
|
149
|
+
log_div: ET._Element
|
|
150
|
+
dmd_sec: typing.Optional[ET._Element] = None
|
|
151
|
+
calculated_identifier_hash: typing.Optional[str] = None
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class MetsFile:
|
|
155
|
+
"""Represents a published METS file resource."""
|
|
156
|
+
|
|
157
|
+
def __init__(self, file_path: pathlib.Path, **kwargs):
|
|
158
|
+
self.file_path = file_path
|
|
159
|
+
self.kwargs = kwargs
|
|
160
|
+
self.label_fgroup_fulltext = self.kwargs.get("label_filegroup_ocr", DEFAULT_FULLEXT_FILEGROUP)
|
|
161
|
+
self.label_fgroup_image = self.kwargs.get("label_filegroup_image", DEFAULT_IMAGE_FILEGROUP)
|
|
162
|
+
self.logical_root: typing.Optional[ET._Element] = None
|
|
163
|
+
self.physical_root: typing.Optional[ET._Element] = None
|
|
164
|
+
self.document = file_path # trigger setter
|
|
165
|
+
|
|
166
|
+
@property
|
|
167
|
+
def document(self) -> ET._ElementTree:
|
|
168
|
+
"""Get the METS XML document representing the corpus template."""
|
|
169
|
+
return self.__document
|
|
170
|
+
|
|
171
|
+
@document.setter
|
|
172
|
+
def document(self, path_document: pathlib.Path) -> None:
|
|
173
|
+
"""Set the METS XML document and extract relevant sections."""
|
|
174
|
+
self.__document = ET.parse(path_document, ET.XMLParser(remove_blank_text=True))
|
|
175
|
+
the_root = self.__document.getroot()
|
|
176
|
+
# Extract namespaces directly using xpath; filter out empty/None prefixes
|
|
177
|
+
raw_ns = typing.cast(typing.List[typing.Tuple[typing.Optional[str], str]], the_root.xpath("//namespace::*"))
|
|
178
|
+
self.__nsmap = {prefix: uri for prefix, uri in raw_ns if prefix}
|
|
179
|
+
logical_root = self.document.find('.//mets:structMap[@TYPE="LOGICAL"]', self.nsmap)
|
|
180
|
+
if logical_root is None:
|
|
181
|
+
raise CorpusException(f"Missing logical structMap in METS file '{self.file_path}'")
|
|
182
|
+
self.logical_root = logical_root
|
|
183
|
+
physical_root = self.document.find('.//mets:structMap[@TYPE="PHYSICAL"]/mets:div', self.nsmap)
|
|
184
|
+
if physical_root is not None:
|
|
185
|
+
self.physical_root = physical_root
|
|
186
|
+
self.file_path = path_document
|
|
187
|
+
|
|
188
|
+
@property
|
|
189
|
+
def nsmap(self) -> dict:
|
|
190
|
+
"""Get the namespace mapping extracted from the METS document."""
|
|
191
|
+
return self.__nsmap
|
|
192
|
+
|
|
193
|
+
def find_dmd_id(self) -> str:
|
|
194
|
+
"""Find DMDID in logical structMap of given published METS file."""
|
|
195
|
+
|
|
196
|
+
assert self.logical_root is not None, f"No logical structMap in METS file {self.file_path}"
|
|
197
|
+
logical_children = self.logical_root.findall(".//mets:div", self.nsmap)
|
|
198
|
+
for element in logical_children:
|
|
199
|
+
the_type = element.get("TYPE", None)
|
|
200
|
+
dmd_id = element.get("DMDID", None)
|
|
201
|
+
if dmd_id is not None and the_type in METS_MEDIA_TYPES:
|
|
202
|
+
return dmd_id
|
|
203
|
+
raise CorpusException(f"no DMD_ID in METS file {self.file_path} found")
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
class MetsCorpusFile(MetsFile):
|
|
207
|
+
"""Extends and modifies base MetsFile class to represent a METS file resource
|
|
208
|
+
for a given Corpus Page with additional carries required information."""
|
|
209
|
+
|
|
210
|
+
def __init__(self, file_path: pathlib.Path, **kwargs):
|
|
211
|
+
super().__init__(file_path, **kwargs)
|
|
212
|
+
self.label_fgroup_fulltext = self.kwargs.get("corpus_filegroup_ocr", GT_METS_FILEGROUP_FULLTEXT)
|
|
213
|
+
self.label_fgroup_image = self.kwargs.get("corpus_filegroup_image", GT_METS_FILEGROUP_IMAGE)
|
|
214
|
+
self.corpus_file_name = self.kwargs.get("corpus_file_name", DEFAULT_METS_FILE_NAME)
|
|
215
|
+
# for the corpus METS file, go a little bit further down the logical line
|
|
216
|
+
assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
|
|
217
|
+
self.logical_root = self.logical_root.find("mets:div", self.nsmap)
|
|
218
|
+
assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
|
|
219
|
+
corpus_label = self.kwargs.get("corpus_label")
|
|
220
|
+
if corpus_label:
|
|
221
|
+
self.logical_root.set("LABEL", str(corpus_label))
|
|
222
|
+
self._prepare_filegroups()
|
|
223
|
+
the_root = self.document.getroot()
|
|
224
|
+
prev_phys_root = the_root.findall('.//mets:div[@ID="physroot"]', self.nsmap)
|
|
225
|
+
if len(prev_phys_root) == 0:
|
|
226
|
+
physroot = ET.Element(f'{{{self.nsmap["mets"]}}}div', attrib={"ID": "physroot", "TYPE": "physSequence"})
|
|
227
|
+
assert self.physical_root is not None, f"Missing physical structMap in METS file '{self.file_path}'"
|
|
228
|
+
self.physical_root.append(physroot)
|
|
229
|
+
self.link_root = self.document.find(".//mets:structLink", self.nsmap)
|
|
230
|
+
|
|
231
|
+
def _prepare_filegroups(self):
|
|
232
|
+
xpr_fulltext = f'mets:fileGrp[@USE="{self.label_fgroup_fulltext}"]'
|
|
233
|
+
self.file_group_fulltext = self._get_or_create_filegroup(
|
|
234
|
+
"fileGrp", xpr_fulltext, {"USE": self.label_fgroup_fulltext}
|
|
235
|
+
)
|
|
236
|
+
xpr_image = f'mets:fileGrp[@USE="{self.label_fgroup_image}"]'
|
|
237
|
+
self.file_group_image = self._get_or_create_filegroup("fileGrp", xpr_image, {"USE": self.label_fgroup_image})
|
|
238
|
+
|
|
239
|
+
def _get_or_create_filegroup(self, tag: str, xpr: str, attrib: dict) -> ET._Element:
|
|
240
|
+
"""Get or create an XML element with the specified tag and attributes."""
|
|
241
|
+
filesec_root = self.document.getroot().find("mets:fileSec", self.nsmap)
|
|
242
|
+
if filesec_root is None:
|
|
243
|
+
filesec_root = ET.Element(f'{{{self.nsmap["mets"]}}}fileSec')
|
|
244
|
+
self.document.getroot().append(filesec_root)
|
|
245
|
+
file_group = filesec_root.find(xpr, self.nsmap)
|
|
246
|
+
if file_group is None:
|
|
247
|
+
file_group = ET.Element(f'{{{self.nsmap["mets"]}}}{tag}', attrib=attrib)
|
|
248
|
+
filesec_root.append(file_group)
|
|
249
|
+
return file_group
|
|
250
|
+
|
|
251
|
+
def attach(self, extract: MetsModsSection) -> None:
|
|
252
|
+
"""Attach extracted METS sections to the corpus METS document
|
|
253
|
+
but attach MODS only if not already present in the corpus METS file."""
|
|
254
|
+
|
|
255
|
+
self.file_group_image.append(extract.file_image)
|
|
256
|
+
self.file_group_fulltext.append(extract.file_fulltext)
|
|
257
|
+
assert self.physical_root is not None, f"Missing physical structMap in METS file '{self.file_path}'"
|
|
258
|
+
self.physical_root.append(extract.phys_div)
|
|
259
|
+
assert self.link_root is not None, f"Missing structLink in METS file '{self.file_path}'"
|
|
260
|
+
self.link_root.append(extract.sm_link)
|
|
261
|
+
assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
|
|
262
|
+
self.logical_root.append(extract.log_div)
|
|
263
|
+
corpus_root = self.document.getroot()
|
|
264
|
+
assert extract.dmd_sec is not None, "DMD section missing in section for attachment"
|
|
265
|
+
xtrct_dmd_id = extract.dmd_sec.get("ID")
|
|
266
|
+
# look for extracted dmd_id in corpus file
|
|
267
|
+
if corpus_root.find(f'.//mets:dmdSec[@ID="{xtrct_dmd_id}"]', namespaces=self.nsmap) is None:
|
|
268
|
+
# if descriptive metadata not yet present in corpus file, append dmd_sec from extract
|
|
269
|
+
logger.info(
|
|
270
|
+
"Attach new DMD section with ID='%s' to corpus METS file",
|
|
271
|
+
xtrct_dmd_id,
|
|
272
|
+
)
|
|
273
|
+
if xtrct_dmd_id is not None and extract.dmd_sec is not None:
|
|
274
|
+
idx: int = len(self.document.findall(".//mets:dmdSec", namespaces=self.nsmap))
|
|
275
|
+
self.document.getroot().insert(idx, extract.dmd_sec)
|
|
276
|
+
else:
|
|
277
|
+
logger.info(
|
|
278
|
+
"DMD section with ID='%s' already present in corpus METS file",
|
|
279
|
+
xtrct_dmd_id,
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
def finalize(self):
|
|
283
|
+
"""Re-order attached gt-image containers"""
|
|
284
|
+
the_root = self.document.getroot()
|
|
285
|
+
pages_with_order: list[ET._Element] = the_root.findall(
|
|
286
|
+
'.//mets:structMap[@TYPE="PHYSICAL"]//mets:div[@ORDER]', self.nsmap
|
|
287
|
+
)
|
|
288
|
+
for i, elm in enumerate(pages_with_order, 1):
|
|
289
|
+
elm.set("ORDER", f"{i}")
|
|
290
|
+
assert self.logical_root is not None, f"Missing logical structMap in METS file '{self.file_path}'"
|
|
291
|
+
log_divs_with_order: list[ET._Element] = self.logical_root.findall(".//mets:div[@ORDER]", self.nsmap)
|
|
292
|
+
for i, elm in enumerate(log_divs_with_order, 1):
|
|
293
|
+
elm.set("ORDER", f"{i}")
|
|
294
|
+
|
|
295
|
+
def write(self, out_dir: pathlib.Path) -> pathlib.Path:
|
|
296
|
+
"""Write the METS XML document to the specified file path."""
|
|
297
|
+
encoding = self.document.docinfo.encoding if self.document.docinfo.encoding else "UTF-8"
|
|
298
|
+
ET.indent(self.document.getroot(), space=(" " * INDENT))
|
|
299
|
+
out_file: pathlib.Path = out_dir.joinpath(self.corpus_file_name)
|
|
300
|
+
self.document.write(out_file, xml_declaration=True, pretty_print=True, encoding=encoding)
|
|
301
|
+
return out_file
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
class MetsResourceFile(MetsFile):
|
|
305
|
+
"""Represents a METS file resource for extraction."""
|
|
306
|
+
|
|
307
|
+
def __init__(self, path_mets_file, idx: int, **kwargs):
|
|
308
|
+
super().__init__(path_mets_file, **kwargs)
|
|
309
|
+
self.mdsec_identifier: typing.Optional[str] = None
|
|
310
|
+
self.idx = idx
|
|
311
|
+
|
|
312
|
+
@property
|
|
313
|
+
def identifier_hash(self) -> str:
|
|
314
|
+
"""Calculate and return a 8-character hash based on the METS/MODS resource section identifier."""
|
|
315
|
+
if self.mdsec_identifier is None:
|
|
316
|
+
raise CorpusException(f"Identifier for METS resource file {self.file_path} not set")
|
|
317
|
+
return hashlib.sha256(self.mdsec_identifier.encode()).hexdigest()[:8]
|
|
318
|
+
|
|
319
|
+
def extract(
|
|
320
|
+
self,
|
|
321
|
+
page_urn: str,
|
|
322
|
+
out_dir: pathlib.Path,
|
|
323
|
+
gt_file_path: pathlib.Path,
|
|
324
|
+
) -> MetsModsSection:
|
|
325
|
+
"""Grab required information and resolve relationships starting from
|
|
326
|
+
the page div with given CONTENTIDS in the METS file."""
|
|
327
|
+
|
|
328
|
+
the_root = self.document.getroot()
|
|
329
|
+
page_div = the_root.find(f'.//mets:div[@CONTENTIDS="{page_urn}"]', self.nsmap)
|
|
330
|
+
assert page_div is not None, f"no page with CONTENTIDS='{page_urn}' found"
|
|
331
|
+
file_image, file_fulltext = self._set_page_with_files(the_root, page_div, gt_file_path, out_dir)
|
|
332
|
+
|
|
333
|
+
# now resolve relationships to get logical div and dmdSec for the page
|
|
334
|
+
# PHYSICAL
|
|
335
|
+
source_phys_id = page_div.get("ID")
|
|
336
|
+
assert source_phys_id is not None, f"page div with CONTENTIDS='{page_urn}' missing ID attribute"
|
|
337
|
+
# LINK
|
|
338
|
+
sm_link = the_root.find(f'.//mets:smLink[@xlink:to="{source_phys_id}"]', self.nsmap)
|
|
339
|
+
assert sm_link is not None, f"smLink for page div with ID='{source_phys_id}' not found"
|
|
340
|
+
# LOGICAL
|
|
341
|
+
log_id = sm_link.get(f'{{{self.nsmap["xlink"]}}}from')
|
|
342
|
+
assert log_id is not None, f"log ID for smLink with to='{source_phys_id}' not found"
|
|
343
|
+
log_div = the_root.find(f'.//mets:div[@ID="{log_id}"]', self.nsmap)
|
|
344
|
+
assert log_div is not None, f"log div with ID='{log_id}' not found"
|
|
345
|
+
# clean up logical div: remove un-related attributes like "AMDID", remove all child elements
|
|
346
|
+
if log_div.get("AMDID") is not None:
|
|
347
|
+
log_div.attrib.pop("AMDID")
|
|
348
|
+
for child in log_div.getchildren():
|
|
349
|
+
child.getparent().remove(child)
|
|
350
|
+
# fix current DMDID for identifier calculation
|
|
351
|
+
source_dmd_id = self.find_dmd_id()
|
|
352
|
+
source_dmd_sec = the_root.find(f'.//mets:dmdSec[@ID="{source_dmd_id}"]', self.nsmap)
|
|
353
|
+
assert source_dmd_sec is not None, f"DMD section with ID='{source_dmd_id}' not found"
|
|
354
|
+
copy_dmd_sec = self._build_dmd_section(source_dmd_sec)
|
|
355
|
+
section = MetsModsSection(
|
|
356
|
+
phys_div=page_div,
|
|
357
|
+
file_image=file_image,
|
|
358
|
+
file_fulltext=file_fulltext,
|
|
359
|
+
sm_link=sm_link,
|
|
360
|
+
log_div=log_div,
|
|
361
|
+
dmd_sec=copy_dmd_sec,
|
|
362
|
+
calculated_identifier_hash=self.identifier_hash,
|
|
363
|
+
)
|
|
364
|
+
self._reindex(section)
|
|
365
|
+
return section
|
|
366
|
+
|
|
367
|
+
def _set_page_with_files(
|
|
368
|
+
self, the_root: ET._Element, page_div: ET._Element, gt_file_path: pathlib.Path, out_dir: pathlib.Path
|
|
369
|
+
) -> typing.Tuple[ET._Element, ET._Element]:
|
|
370
|
+
"""Re-Create page container with the same attributes as the source page but with modified children."""
|
|
371
|
+
|
|
372
|
+
file_pointers: list[ET._Element] = page_div.findall("mets:fptr", namespaces=self.nsmap)
|
|
373
|
+
# Remove all child elements from actual page
|
|
374
|
+
for el in file_pointers:
|
|
375
|
+
el_pa = el.getparent()
|
|
376
|
+
if el_pa is not None:
|
|
377
|
+
el_pa.remove(el)
|
|
378
|
+
# now re-attach selected file pointers for FULLTEXT and MAX image
|
|
379
|
+
file_image = None
|
|
380
|
+
file_fulltext = None
|
|
381
|
+
for fp in file_pointers:
|
|
382
|
+
the_file = the_root.find(f'.//mets:file[@ID="{fp.get("FILEID")}"]', self.nsmap)
|
|
383
|
+
assert the_file is not None, f"no file with ID='{fp.get('FILEID')}' found in {self.document.docinfo.URL}"
|
|
384
|
+
the_file_group = the_file.xpath("ancestor::mets:fileGrp/@USE", namespaces=self.nsmap)
|
|
385
|
+
assert len(the_file_group) == 1, f"file with ID='{fp.get('FILEID')}' invalid parent fileGrp"
|
|
386
|
+
the_group = the_file_group[0]
|
|
387
|
+
if the_group not in {self.label_fgroup_fulltext, self.label_fgroup_image}:
|
|
388
|
+
continue
|
|
389
|
+
|
|
390
|
+
# here we go
|
|
391
|
+
new_id = f"{the_group}-{(self.idx):04d}"
|
|
392
|
+
if the_group == self.label_fgroup_image:
|
|
393
|
+
prev_img_fileid = fp.get("FILEID")
|
|
394
|
+
file_image = the_root.find(f'.//mets:file[@ID="{prev_img_fileid}"]', self.nsmap)
|
|
395
|
+
assert file_image is not None, f"no file with ID='{prev_img_fileid}' found"
|
|
396
|
+
pre_file_id = file_image.get("ID")
|
|
397
|
+
logger.debug(
|
|
398
|
+
"Re-assigning file ID '%s' to '%s' for page with CONTENTIDS='%s'",
|
|
399
|
+
pre_file_id,
|
|
400
|
+
new_id,
|
|
401
|
+
page_div.get("CONTENTIDS"),
|
|
402
|
+
)
|
|
403
|
+
file_image.set("ID", new_id)
|
|
404
|
+
fp.set("FILEID", new_id)
|
|
405
|
+
page_div.append(fp)
|
|
406
|
+
elif the_group == self.label_fgroup_fulltext:
|
|
407
|
+
file_ptr_fulltext = ET.Element(f'{{{self.nsmap["mets"]}}}fptr', attrib={"FILEID": new_id})
|
|
408
|
+
page_div.append(file_ptr_fulltext)
|
|
409
|
+
file_fulltext = self._create_fulltext_element(new_id, gt_file_path, out_dir)
|
|
410
|
+
# if no file pointer for fulltext exists, generate it since print may do not possess
|
|
411
|
+
if file_fulltext is None:
|
|
412
|
+
new_id = f"{self.label_fgroup_fulltext}-{(self.idx):04d}"
|
|
413
|
+
file_fulltext = self._create_fulltext_element(new_id, gt_file_path, out_dir)
|
|
414
|
+
file_ptr_fulltext = ET.Element(f'{{{self.nsmap["mets"]}}}fptr', attrib={"FILEID": new_id})
|
|
415
|
+
page_div.append(file_ptr_fulltext)
|
|
416
|
+
|
|
417
|
+
page_urn = page_div.get("CONTENTIDS")
|
|
418
|
+
assert file_image is not None, f"No image file pointer for page with CONTENTIDS='{page_urn}'"
|
|
419
|
+
assert file_fulltext is not None, f"No fulltext file pointer for page with CONTENTIDS='{page_urn}'"
|
|
420
|
+
return file_image, file_fulltext
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def _create_fulltext_element(self, new_id: str, gt_file_path: pathlib.Path,
|
|
424
|
+
out_dir: pathlib.Path) -> ET._Element:
|
|
425
|
+
"""Create a new file element for the fulltext file with the appropriate FLocat child."""
|
|
426
|
+
file_fulltext = ET.Element(
|
|
427
|
+
f'{{{self.nsmap["mets"]}}}file', attrib={"ID": new_id, "MIMETYPE": "application/vnd.prima.page+xml"}
|
|
428
|
+
)
|
|
429
|
+
file_fulltext.append(
|
|
430
|
+
ET.Element(
|
|
431
|
+
f'{{{self.nsmap["mets"]}}}FLocat',
|
|
432
|
+
attrib={
|
|
433
|
+
f'{{{self.nsmap["xlink"]}}}href': str(gt_file_path.relative_to(out_dir)),
|
|
434
|
+
"LOCTYPE": "OTHER",
|
|
435
|
+
"OTHERLOCTYPE": "FILE",
|
|
436
|
+
},
|
|
437
|
+
)
|
|
438
|
+
)
|
|
439
|
+
return file_fulltext
|
|
440
|
+
|
|
441
|
+
def _build_dmd_section(self, source_dmd_sec: ET._Element) -> ET._Element:
|
|
442
|
+
|
|
443
|
+
source_mods_root = source_dmd_sec.find(".//mods:mods", self.nsmap)
|
|
444
|
+
assert source_mods_root is not None, f"no MODS root in DMD section with ID='{source_dmd_sec.get('ID')}' found"
|
|
445
|
+
identifer_elements: list[ET._Element] = source_mods_root.findall("mods:identifier", self.nsmap)
|
|
446
|
+
identifer_urn: typing.List[str] = [i.text for i in identifer_elements if i.get("type") == "urn"]
|
|
447
|
+
if not identifer_urn:
|
|
448
|
+
raise CorpusException(
|
|
449
|
+
f"No identifier with type='urn' in MODS metadata of METS file {self.document.docinfo.URL} found"
|
|
450
|
+
)
|
|
451
|
+
self.mdsec_identifier = identifer_urn[0]
|
|
452
|
+
copy_dmd_sec = copy.deepcopy(
|
|
453
|
+
source_dmd_sec
|
|
454
|
+
) # create deep copy of dmd_sec to avoid modifying the original METS file
|
|
455
|
+
copy_root = copy_dmd_sec.find(".//mods:mods", self.nsmap)
|
|
456
|
+
assert copy_root is not None, f"no MODS root in DMD section with ID='{copy_dmd_sec.get('ID')}' found"
|
|
457
|
+
# clear copy: remove all copy child elements
|
|
458
|
+
for child in copy_root.getchildren():
|
|
459
|
+
child.getparent().remove(child)
|
|
460
|
+
|
|
461
|
+
copy_root.extend(identifer_elements)
|
|
462
|
+
title_elements: list[ET._Element] = source_mods_root.findall("mods:titleInfo", self.nsmap)
|
|
463
|
+
if len(title_elements) > 0:
|
|
464
|
+
copy_root.extend(title_elements)
|
|
465
|
+
else:
|
|
466
|
+
title_host = source_mods_root.find("mods:relatedItem/mods:titleInfo", self.nsmap)
|
|
467
|
+
if title_host is not None:
|
|
468
|
+
copy_root.append(title_host)
|
|
469
|
+
language_elements: list[ET._Element] = source_mods_root.findall("mods:language", self.nsmap)
|
|
470
|
+
if len(language_elements) > 0:
|
|
471
|
+
copy_root.extend(language_elements)
|
|
472
|
+
genre_elements: list[ET._Element] = source_mods_root.findall("mods:genre", self.nsmap)
|
|
473
|
+
if len(genre_elements) > 0:
|
|
474
|
+
copy_root.extend(genre_elements)
|
|
475
|
+
publication_info = source_mods_root.find('mods:originInfo[@eventType="publication"]', self.nsmap)
|
|
476
|
+
if publication_info is not None:
|
|
477
|
+
copy_root.append(publication_info)
|
|
478
|
+
access_info = source_mods_root.find("mods:accessCondition", self.nsmap)
|
|
479
|
+
if access_info is not None:
|
|
480
|
+
copy_root.append(access_info)
|
|
481
|
+
return copy_dmd_sec
|
|
482
|
+
|
|
483
|
+
def _reindex(self, section: MetsModsSection) -> None:
|
|
484
|
+
"""Re-assign IDs for phys_div, sm_link and log_div in the given section."""
|
|
485
|
+
new_phys_id = f"PHYS-{(self.idx):04d}"
|
|
486
|
+
prev_phys_id = section.phys_div.get("ID")
|
|
487
|
+
logger.debug("Re-assigning phys div ID '%s' to '%s'", prev_phys_id, new_phys_id)
|
|
488
|
+
new_log_id = f"LOG-{(self.idx):04d}"
|
|
489
|
+
prev_log_id = section.log_div.get("ID")
|
|
490
|
+
logger.debug("Re-assigning log div ID '%s' to '%s'", prev_log_id, new_log_id)
|
|
491
|
+
section.phys_div.set("ID", new_phys_id)
|
|
492
|
+
section.sm_link.set(f'{{{self.nsmap["xlink"]}}}to', new_phys_id)
|
|
493
|
+
section.sm_link.set(f'{{{self.nsmap["xlink"]}}}from', new_log_id)
|
|
494
|
+
section.log_div.set("ID", new_log_id)
|
|
495
|
+
assert section.dmd_sec is not None, "DMD section missing in section for re-indexing"
|
|
496
|
+
prev_dmdid = section.dmd_sec.get("ID")
|
|
497
|
+
new_dmdid = f"{prev_dmdid}{DMDID_GLUE}{section.calculated_identifier_hash}"
|
|
498
|
+
section.log_div.set("DMDID", new_dmdid)
|
|
499
|
+
section.dmd_sec.set("ID", new_dmdid)
|
|
500
|
+
logger.debug("Re-assigning DMDID '%s' to '%s'", prev_dmdid, new_dmdid)
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
@dataclasses.dataclass
|
|
504
|
+
class CorpusPageInput:
|
|
505
|
+
"""Encapsulate input data required for corpus generation."""
|
|
506
|
+
|
|
507
|
+
identifier_urn: str
|
|
508
|
+
groundtruth_file: GroundtruthFile
|
|
509
|
+
cached_media_mets_file: typing.Optional[pathlib.Path] = None
|
|
510
|
+
metadata: typing.Optional[MetsResourceFile] = None
|
|
511
|
+
|
|
512
|
+
def __init__(self, groundtruth_file: GroundtruthFile):
|
|
513
|
+
self.groundtruth_file = groundtruth_file
|
|
514
|
+
self.identifier_urn = groundtruth_file.identifier
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
@dataclasses.dataclass
|
|
518
|
+
class CorpusGeneratorResult:
|
|
519
|
+
"""Encapsulate final Corpus generation result."""
|
|
520
|
+
|
|
521
|
+
file_path: pathlib.Path
|
|
522
|
+
n_pages: int
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Ground truth corpus generation for OCR evaluation.
|
|
2
|
+
|
|
3
|
+
This module orchestrates the assembly of a METS-based ground truth corpus
|
|
4
|
+
from a collection of PAGE-XML files with URN identifiers. The main entry
|
|
5
|
+
point is :func:`generate`, which accepts a :class:`~ocr_util.corpus.common.CorpusArgs`
|
|
6
|
+
configuration object and returns a :class:`~ocr_util.corpus.common.CorpusGeneratorResult`.
|
|
7
|
+
|
|
8
|
+
Typical call chain
|
|
9
|
+
------------------
|
|
10
|
+
1. :func:`generate` scans the input directory for PAGE-XML ground truth files.
|
|
11
|
+
2. METS metadata is fetched (and cached) in parallel via
|
|
12
|
+
:class:`~ocr_util.corpus.load_metadata.RecordMetadataResolver`.
|
|
13
|
+
3. :class:`Corpus` iterates over the resolved inputs, extracts the relevant
|
|
14
|
+
METS/MODS sections via
|
|
15
|
+
:class:`~ocr_util.corpus.common.MetsResourceFile`, and attaches them to a
|
|
16
|
+
shared corpus METS document built from a template.
|
|
17
|
+
4. The finished METS document is written to the output directory.
|
|
18
|
+
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import logging
|
|
23
|
+
import math
|
|
24
|
+
import os
|
|
25
|
+
import pathlib
|
|
26
|
+
import shutil
|
|
27
|
+
import typing
|
|
28
|
+
|
|
29
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
from typing import Final
|
|
32
|
+
|
|
33
|
+
import ocr_util.corpus.common as cc
|
|
34
|
+
import ocr_util.corpus.load_metadata as lr
|
|
35
|
+
|
|
36
|
+
CPUS = os.cpu_count() or 1
|
|
37
|
+
NUM_THREADS: typing.Final[int] = math.ceil(CPUS * 0.85)
|
|
38
|
+
DEFAULT_TEMP_DIR = pathlib.Path.home().joinpath(".cache", "ocr_util_corpus_metadata")
|
|
39
|
+
DEFAULT_LIMIT = 0
|
|
40
|
+
|
|
41
|
+
logger = logging.getLogger(__name__)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def fetch_resources(cache_path: pathlib.Path, gt_files: list[cc.CorpusPageInput]) -> list[cc.CorpusPageInput]:
|
|
45
|
+
"""Fetch required METS metadata parallel for given inputs."""
|
|
46
|
+
if not cache_path.exists():
|
|
47
|
+
cache_path.mkdir()
|
|
48
|
+
resolver: lr.RecordMetadataResolver = lr.RecordMetadataResolver()
|
|
49
|
+
# Use ThreadPoolExecutor directly for parallel execution
|
|
50
|
+
with ThreadPoolExecutor(max_workers=NUM_THREADS) as executor:
|
|
51
|
+
futures = [executor.submit(resolver.fetch, an_input, cache_path=cache_path) for an_input in gt_files]
|
|
52
|
+
return [future.result() for future in futures]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class Corpus:
|
|
56
|
+
"""Ground Truth Corpus in METS format."""
|
|
57
|
+
|
|
58
|
+
def __init__(self, cargs: cc.CorpusArgs, inputs: list[cc.CorpusPageInput], **kwargs) -> None:
|
|
59
|
+
self.__cargs = cargs
|
|
60
|
+
self.__out_dir: Final[Path] = self.__cargs.output_dir
|
|
61
|
+
self.__inputs: Final[list[cc.CorpusPageInput]] = inputs
|
|
62
|
+
# Template file is located in the same directory as this module
|
|
63
|
+
template_path = Path(__file__).parent / "template.corpus.xml"
|
|
64
|
+
self.corpus_template = template_path
|
|
65
|
+
self.kwargs = kwargs
|
|
66
|
+
self.corpus_label = self.__cargs.corpus_label
|
|
67
|
+
self.corpus_file = cc.MetsCorpusFile(template_path, corpus_label=self.corpus_label, **kwargs)
|
|
68
|
+
|
|
69
|
+
def build(self) -> cc.CorpusGeneratorResult:
|
|
70
|
+
"""Build the corpus by extracting data from original METS files and write new METS file."""
|
|
71
|
+
|
|
72
|
+
total: int = len(self.__inputs)
|
|
73
|
+
assert self.corpus_file is not None, "Corpus template must be set before start building corpus"
|
|
74
|
+
for idx, page_input in enumerate(self.__inputs, 1):
|
|
75
|
+
logger.info("[%d/%d] Process file with identifier %s", idx, total, page_input.identifier_urn)
|
|
76
|
+
try:
|
|
77
|
+
assert (
|
|
78
|
+
page_input.cached_media_mets_file is not None
|
|
79
|
+
), f"METS file for {page_input.identifier_urn} not found"
|
|
80
|
+
mets_res = cc.MetsResourceFile(page_input.cached_media_mets_file, idx, **self.kwargs)
|
|
81
|
+
page_input.metadata = mets_res
|
|
82
|
+
the_section: cc.MetsModsSection = mets_res.extract(
|
|
83
|
+
page_urn=page_input.identifier_urn,
|
|
84
|
+
out_dir=self.__out_dir,
|
|
85
|
+
gt_file_path=page_input.groundtruth_file.file_path,
|
|
86
|
+
)
|
|
87
|
+
self.corpus_file.attach(the_section)
|
|
88
|
+
except Exception as exc:
|
|
89
|
+
logger.exception(
|
|
90
|
+
"Error processing %s: %s - skip file",
|
|
91
|
+
page_input.cached_media_mets_file,
|
|
92
|
+
exc,
|
|
93
|
+
)
|
|
94
|
+
# re-sort pages
|
|
95
|
+
self.corpus_file.finalize()
|
|
96
|
+
out_path = self.corpus_file.write(self.__out_dir)
|
|
97
|
+
return cc.CorpusGeneratorResult(file_path=out_path, n_pages=len(self.__inputs))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def generate(cargs: cc.CorpusArgs) -> cc.CorpusGeneratorResult:
|
|
101
|
+
"""Generate GT corpus in METS format from given corpus arguments."""
|
|
102
|
+
if not cargs.input_dir.exists():
|
|
103
|
+
raise cc.CorpusException(f"Input directory '{cargs.input_dir}' does not exist")
|
|
104
|
+
if cargs.output_dir.exists():
|
|
105
|
+
logger.warning(
|
|
106
|
+
"Output directory '%s' already exists. Refusing to overwrite existing data.",
|
|
107
|
+
cargs.output_dir,
|
|
108
|
+
)
|
|
109
|
+
if cargs.local_cache_dir.exists() and cargs.clear_cache:
|
|
110
|
+
logger.info("Wipe existing cache directory %s", cargs.local_cache_dir)
|
|
111
|
+
shutil.rmtree(cargs.local_cache_dir)
|
|
112
|
+
cargs.local_cache_dir.mkdir(parents=True, exist_ok=True)
|
|
113
|
+
cargs.output_dir.mkdir(parents=True, exist_ok=True)
|
|
114
|
+
gt_files: list[cc.GroundtruthFile] = cc.GroundtruthFile.from_dir_copy(
|
|
115
|
+
in_dir=cargs.input_dir, out_dir=cargs.output_dir.joinpath(cc.GT_TARGET_SUBDIR), limit=cargs.limit
|
|
116
|
+
)
|
|
117
|
+
corpus_inputs = [cc.CorpusPageInput(groundtruth_file=gt_file) for gt_file in gt_files]
|
|
118
|
+
local_cache_path = Path(f"{cargs.local_cache_dir}").joinpath("mets")
|
|
119
|
+
corpus_input_with_resources = fetch_resources(local_cache_path, corpus_inputs)
|
|
120
|
+
corpus_file = Corpus(cargs, corpus_input_with_resources)
|
|
121
|
+
return corpus_file.build()
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
if __name__ == "__main__":
|
|
125
|
+
parser: argparse.ArgumentParser = argparse.ArgumentParser()
|
|
126
|
+
parser.add_argument("input", help="Path to the input directory")
|
|
127
|
+
parser.add_argument("output", help="Path to the output directory")
|
|
128
|
+
parser.add_argument(
|
|
129
|
+
"-l",
|
|
130
|
+
"--limit",
|
|
131
|
+
help="Number of Files being processed, default = 0 (unlimited)",
|
|
132
|
+
required=False,
|
|
133
|
+
type=int,
|
|
134
|
+
default=DEFAULT_LIMIT,
|
|
135
|
+
)
|
|
136
|
+
parser.add_argument(
|
|
137
|
+
"-t", "--temp-dir", help="Path to the temporary directory", required=False, default=DEFAULT_TEMP_DIR
|
|
138
|
+
)
|
|
139
|
+
args = parser.parse_args()
|
|
140
|
+
corpus_args = cc.CorpusArgs(
|
|
141
|
+
input_dir=Path(args.input).absolute(),
|
|
142
|
+
output_dir=Path(args.output).absolute(),
|
|
143
|
+
local_cache_dir=Path(args.temp_dir).absolute(),
|
|
144
|
+
limit=int(args.limit),
|
|
145
|
+
corpus_label="Ground Truth Corpus",
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
outcome = generate(corpus_args)
|
|
149
|
+
logger.info("Corpus generated at %s with %d pages", outcome.file_path, outcome.n_pages)
|