marc-bibframe 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- marc_bibframe/__init__.py +175 -0
- marc_bibframe/cli.py +150 -0
- marc_bibframe/py.typed +0 -0
- marc_bibframe/xsl/ConvSpec-001-007.xsl +2276 -0
- marc_bibframe/xsl/ConvSpec-006,008.xsl +1295 -0
- marc_bibframe/xsl/ConvSpec-010-048.xsl +1712 -0
- marc_bibframe/xsl/ConvSpec-050-088.xsl +713 -0
- marc_bibframe/xsl/ConvSpec-1XX,7XX,8XX-names.xsl +1141 -0
- marc_bibframe/xsl/ConvSpec-200-247not240-Titles.xsl +771 -0
- marc_bibframe/xsl/ConvSpec-240andX30-UnifTitle.xsl +535 -0
- marc_bibframe/xsl/ConvSpec-250-270.xsl +283 -0
- marc_bibframe/xsl/ConvSpec-3XX.xsl +2245 -0
- marc_bibframe/xsl/ConvSpec-460-468-SeriesTreat.xsl +135 -0
- marc_bibframe/xsl/ConvSpec-490-510-Links.xsl +145 -0
- marc_bibframe/xsl/ConvSpec-5XX.xsl +1448 -0
- marc_bibframe/xsl/ConvSpec-600-662.xsl +1417 -0
- marc_bibframe/xsl/ConvSpec-720+740to755.xsl +333 -0
- marc_bibframe/xsl/ConvSpec-758.xsl +102 -0
- marc_bibframe/xsl/ConvSpec-760-788-Links.xsl +771 -0
- marc_bibframe/xsl/ConvSpec-841-887.xsl +328 -0
- marc_bibframe/xsl/ConvSpec-880.xsl +119 -0
- marc_bibframe/xsl/ConvSpec-ControlSubfields.xsl +495 -0
- marc_bibframe/xsl/ConvSpec-LDR.xsl +183 -0
- marc_bibframe/xsl/ConvSpec-Preprocess0-Splitting.xsl +765 -0
- marc_bibframe/xsl/ConvSpec-Process6-Series.xsl +627 -0
- marc_bibframe/xsl/ConvSpec-Process8-ProvAct.xsl +1021 -0
- marc_bibframe/xsl/LICENSE +116 -0
- marc_bibframe/xsl/UPSTREAM +3 -0
- marc_bibframe/xsl/conf/abbreviations.xml +13 -0
- marc_bibframe/xsl/conf/code-to-script.xml +399 -0
- marc_bibframe/xsl/conf/codeMaps.xml +723 -0
- marc_bibframe/xsl/conf/exclusions.xml +4 -0
- marc_bibframe/xsl/conf/iso6392-to-1.xml +615 -0
- marc_bibframe/xsl/conf/languageCrosswalk.xml +1503 -0
- marc_bibframe/xsl/conf/map880.xml +175 -0
- marc_bibframe/xsl/conf/scriptCrosswalk.xml +210 -0
- marc_bibframe/xsl/conf/subjectThesaurus.xml +29 -0
- marc_bibframe/xsl/marc2bibframe2.xsl +610 -0
- marc_bibframe/xsl/naco-normalize.xsl +205 -0
- marc_bibframe/xsl/utils.xsl +964 -0
- marc_bibframe/xsl/variables.xsl +108 -0
- marc_bibframe-0.2.1.dist-info/METADATA +145 -0
- marc_bibframe-0.2.1.dist-info/RECORD +46 -0
- marc_bibframe-0.2.1.dist-info/WHEEL +4 -0
- marc_bibframe-0.2.1.dist-info/entry_points.txt +2 -0
- marc_bibframe-0.2.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
"""Convert MARC records to BIBFRAME RDF.
|
|
2
|
+
|
|
3
|
+
A thin Python wrapper around the Library of Congress marc2bibframe2 XSLT
|
|
4
|
+
(https://github.com/lcnetdev/marc2bibframe2), which is vendored in this
|
|
5
|
+
package. See ``src/marc_bibframe/xsl/UPSTREAM`` for the version, and
|
|
6
|
+
``patches/`` for any local changes to it.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import atexit
|
|
12
|
+
import functools
|
|
13
|
+
from contextlib import ExitStack
|
|
14
|
+
from importlib.resources import as_file, files
|
|
15
|
+
from io import BytesIO
|
|
16
|
+
from typing import Any, BinaryIO
|
|
17
|
+
|
|
18
|
+
import lxml.etree as ET
|
|
19
|
+
import pymarc
|
|
20
|
+
from pymarc.marcxml import record_to_xml
|
|
21
|
+
from rdflib import Graph
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"DEFAULT_BASE_URI",
|
|
25
|
+
"marc_to_graph",
|
|
26
|
+
"marc_to_marcxml",
|
|
27
|
+
"marcxml_to_graph",
|
|
28
|
+
"marcxml_to_rdfxml",
|
|
29
|
+
"upstream",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
#: The stylesheet's own default. It is deliberately non-resolvable: URIs the
|
|
33
|
+
#: transform mints for entities that MARC does not identify (agents, topics,
|
|
34
|
+
#: the work and instance themselves) are built from it, and they name nothing.
|
|
35
|
+
DEFAULT_BASE_URI = "http://example.org/"
|
|
36
|
+
|
|
37
|
+
_MARCXML_NS = "http://www.loc.gov/MARC21/slim"
|
|
38
|
+
_COLLECTION_OPEN = (
|
|
39
|
+
b'<?xml version="1.0" encoding="UTF-8"?>'
|
|
40
|
+
b'<collection xmlns="' + _MARCXML_NS.encode() + b'">'
|
|
41
|
+
)
|
|
42
|
+
_COLLECTION_CLOSE = b"</collection>"
|
|
43
|
+
|
|
44
|
+
# The stylesheet xsl:includes ~30 siblings by relative href, so it has to be
|
|
45
|
+
# resolved from a real directory rather than read out of the package as bytes.
|
|
46
|
+
# Holding the ExitStack open for the life of the process keeps that directory
|
|
47
|
+
# around when the package is imported from a zip.
|
|
48
|
+
_files = ExitStack()
|
|
49
|
+
atexit.register(_files.close)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@functools.cache
|
|
53
|
+
def _xsl_dir():
|
|
54
|
+
return _files.enter_context(as_file(files(__package__).joinpath("xsl")))
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@functools.cache
|
|
58
|
+
def _transform() -> ET.XSLT:
|
|
59
|
+
"""Parse and compile the stylesheet, once per process (it is not cheap)."""
|
|
60
|
+
return ET.XSLT(ET.parse(str(_xsl_dir() / "marc2bibframe2.xsl")))
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def upstream() -> dict[str, str]:
|
|
64
|
+
"""The upstream marc2bibframe2 repository, tag and commit that is vendored here."""
|
|
65
|
+
text = (_xsl_dir() / "UPSTREAM").read_text()
|
|
66
|
+
return {
|
|
67
|
+
k.strip(): v.strip()
|
|
68
|
+
for k, _, v in (line.partition(":") for line in text.splitlines())
|
|
69
|
+
if k
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _xslt_params(**kwargs: Any) -> dict[str, Any]:
|
|
74
|
+
"""Build XSLT parameters, dropping any left as None so the stylesheet default wins.
|
|
75
|
+
|
|
76
|
+
Booleans become the XPath expressions true()/false() rather than string
|
|
77
|
+
literals, because every non-empty string is true in XPath -- passing "false"
|
|
78
|
+
as a string would quietly mean the opposite of what was asked for.
|
|
79
|
+
"""
|
|
80
|
+
params = {}
|
|
81
|
+
for name, value in kwargs.items():
|
|
82
|
+
if value is None:
|
|
83
|
+
continue
|
|
84
|
+
params[name] = (
|
|
85
|
+
"true()"
|
|
86
|
+
if value is True
|
|
87
|
+
else "false()"
|
|
88
|
+
if value is False
|
|
89
|
+
else ET.XSLT.strparam(str(value))
|
|
90
|
+
)
|
|
91
|
+
return params
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def marc_to_marcxml(marc: bytes | BinaryIO) -> bytes:
|
|
95
|
+
"""Convert binary MARC21 to a MARCXML ``<collection>`` of one or more records."""
|
|
96
|
+
handle = BytesIO(marc) if isinstance(marc, bytes) else marc
|
|
97
|
+
parts = [_COLLECTION_OPEN]
|
|
98
|
+
for i, record in enumerate(pymarc.MARCReader(handle)):
|
|
99
|
+
if record is None:
|
|
100
|
+
raise ValueError(f"Could not read MARC record at position {i}")
|
|
101
|
+
parts.append(record_to_xml(record, namespace=False))
|
|
102
|
+
parts.append(_COLLECTION_CLOSE)
|
|
103
|
+
return b"".join(parts)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def marcxml_to_rdfxml(
|
|
107
|
+
marcxml: str | bytes,
|
|
108
|
+
*,
|
|
109
|
+
baseuri: str = DEFAULT_BASE_URI,
|
|
110
|
+
idfield: str | None = None,
|
|
111
|
+
idsource: str | None = None,
|
|
112
|
+
localfields: bool | None = None,
|
|
113
|
+
bcp47_inference: bool | None = None,
|
|
114
|
+
generation_datestamp: str | None = None,
|
|
115
|
+
) -> bytes:
|
|
116
|
+
"""Transform MARCXML into BIBFRAME RDF/XML.
|
|
117
|
+
|
|
118
|
+
Accepts a single ``<record>`` or a ``<collection>`` of them.
|
|
119
|
+
|
|
120
|
+
baseuri
|
|
121
|
+
Base for the URIs the transform mints. Every minted URI has the form
|
|
122
|
+
``{baseuri}{record id}#{fragment}``, so they are scoped to the record
|
|
123
|
+
they came from and are not authority URIs -- two records describing the
|
|
124
|
+
same person yield two different agent URIs. Reconciling them against an
|
|
125
|
+
authority is the caller's job.
|
|
126
|
+
idfield
|
|
127
|
+
MARC field holding the record id, ``001`` by default. Suffix a subfield
|
|
128
|
+
code to use one, e.g. ``035a``.
|
|
129
|
+
idsource
|
|
130
|
+
URI identifying the source of the record id, e.g.
|
|
131
|
+
``http://id.loc.gov/vocabulary/organizations/dlc``.
|
|
132
|
+
localfields
|
|
133
|
+
Convert fields LC defines locally (e.g. 859), off by default.
|
|
134
|
+
bcp47_inference
|
|
135
|
+
Omit a BCP-47 script subtag when it can be inferred from the language.
|
|
136
|
+
generation_datestamp
|
|
137
|
+
Override the timestamp recorded in the work's admin metadata, which
|
|
138
|
+
otherwise defaults to now. Set it to make the RDF/XML this function
|
|
139
|
+
returns byte-for-byte reproducible. Note that a Graph serialized by
|
|
140
|
+
rdflib still varies between runs, because rdflib mints fresh blank
|
|
141
|
+
node labels each time.
|
|
142
|
+
"""
|
|
143
|
+
if isinstance(marcxml, str):
|
|
144
|
+
# Encode first: lxml refuses a str carrying an encoding declaration.
|
|
145
|
+
marcxml = marcxml.encode("utf-8")
|
|
146
|
+
params = _xslt_params(
|
|
147
|
+
baseuri=baseuri,
|
|
148
|
+
idfield=idfield,
|
|
149
|
+
idsource=idsource,
|
|
150
|
+
localfields=localfields,
|
|
151
|
+
bcp47inferrence=bcp47_inference,
|
|
152
|
+
pGenerationDatestamp=generation_datestamp,
|
|
153
|
+
)
|
|
154
|
+
result = _transform()(ET.fromstring(marcxml), **params)
|
|
155
|
+
return ET.tostring(
|
|
156
|
+
result, xml_declaration=True, encoding="UTF-8", pretty_print=True
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def marcxml_to_graph(marcxml: str | bytes, **kwargs: Any) -> Graph:
|
|
161
|
+
"""Transform MARCXML into BIBFRAME as an rdflib Graph.
|
|
162
|
+
|
|
163
|
+
Takes the same keyword arguments as :func:`marcxml_to_rdfxml`.
|
|
164
|
+
"""
|
|
165
|
+
graph = Graph()
|
|
166
|
+
graph.parse(data=marcxml_to_rdfxml(marcxml, **kwargs), format="xml")
|
|
167
|
+
return graph
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def marc_to_graph(marc: bytes | BinaryIO, **kwargs: Any) -> Graph:
|
|
171
|
+
"""Convert binary MARC21 straight to BIBFRAME as an rdflib Graph.
|
|
172
|
+
|
|
173
|
+
Takes the same keyword arguments as :func:`marcxml_to_rdfxml`.
|
|
174
|
+
"""
|
|
175
|
+
return marcxml_to_graph(marc_to_marcxml(marc), **kwargs)
|
marc_bibframe/cli.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Command line interface: MARC in, BIBFRAME out."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from marc_bibframe import (
|
|
10
|
+
DEFAULT_BASE_URI,
|
|
11
|
+
marc_to_marcxml,
|
|
12
|
+
marcxml_to_graph,
|
|
13
|
+
marcxml_to_rdfxml,
|
|
14
|
+
upstream,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
# rdflib's name for each, except rdfxml, which we serve from the transform
|
|
18
|
+
# directly rather than round-tripping it through a parse.
|
|
19
|
+
FORMATS = {
|
|
20
|
+
"turtle": "turtle",
|
|
21
|
+
"ttl": "turtle",
|
|
22
|
+
"json-ld": "json-ld",
|
|
23
|
+
"jsonld": "json-ld",
|
|
24
|
+
"ntriples": "nt",
|
|
25
|
+
"nt": "nt",
|
|
26
|
+
"xml": "xml",
|
|
27
|
+
"rdfxml": None,
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def looks_like_marcxml(data: bytes) -> bool:
|
|
32
|
+
return data.lstrip().lstrip(b"\xef\xbb\xbf").startswith(b"<")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
|
36
|
+
parser = argparse.ArgumentParser(
|
|
37
|
+
prog="marc-bibframe",
|
|
38
|
+
description="Convert MARC records to BIBFRAME RDF.",
|
|
39
|
+
epilog="Binary MARC21 and MARCXML are both accepted; the format is "
|
|
40
|
+
"detected from the content. With no INPUT, reads standard input.",
|
|
41
|
+
)
|
|
42
|
+
parser.add_argument(
|
|
43
|
+
"input",
|
|
44
|
+
nargs="?",
|
|
45
|
+
default="-",
|
|
46
|
+
help="MARC file to convert, or - for standard input (the default)",
|
|
47
|
+
)
|
|
48
|
+
parser.add_argument(
|
|
49
|
+
"-o",
|
|
50
|
+
"--output",
|
|
51
|
+
help="write to this file instead of standard output",
|
|
52
|
+
)
|
|
53
|
+
parser.add_argument(
|
|
54
|
+
"-f",
|
|
55
|
+
"--format",
|
|
56
|
+
default="turtle",
|
|
57
|
+
choices=sorted(FORMATS),
|
|
58
|
+
help="output serialization (default: turtle). rdfxml is the "
|
|
59
|
+
"stylesheet's own output, passed through unparsed",
|
|
60
|
+
)
|
|
61
|
+
parser.add_argument(
|
|
62
|
+
"-b",
|
|
63
|
+
"--baseuri",
|
|
64
|
+
default=DEFAULT_BASE_URI,
|
|
65
|
+
metavar="URI",
|
|
66
|
+
help="base for the URIs the transform mints, which are NOT authority "
|
|
67
|
+
f"URIs -- see the README (default: {DEFAULT_BASE_URI})",
|
|
68
|
+
)
|
|
69
|
+
parser.add_argument(
|
|
70
|
+
"--idfield",
|
|
71
|
+
metavar="FIELD",
|
|
72
|
+
help="MARC field holding the record id, 001 by default. Suffix a "
|
|
73
|
+
"subfield code to use one, e.g. 035a",
|
|
74
|
+
)
|
|
75
|
+
parser.add_argument(
|
|
76
|
+
"--idsource",
|
|
77
|
+
metavar="URI",
|
|
78
|
+
help="URI identifying the source of the record id, e.g. "
|
|
79
|
+
"http://id.loc.gov/vocabulary/organizations/dlc",
|
|
80
|
+
)
|
|
81
|
+
parser.add_argument(
|
|
82
|
+
"--local-fields",
|
|
83
|
+
action="store_true",
|
|
84
|
+
default=None,
|
|
85
|
+
help="convert fields the Library of Congress defines locally, e.g. 859",
|
|
86
|
+
)
|
|
87
|
+
parser.add_argument(
|
|
88
|
+
"--no-bcp47-inference",
|
|
89
|
+
dest="bcp47_inference",
|
|
90
|
+
action="store_false",
|
|
91
|
+
default=None,
|
|
92
|
+
help="keep the script subtag in BCP-47 codes even when the language implies it",
|
|
93
|
+
)
|
|
94
|
+
parser.add_argument(
|
|
95
|
+
"--datestamp",
|
|
96
|
+
metavar="TIMESTAMP",
|
|
97
|
+
help="override the timestamp in the work's admin metadata. With "
|
|
98
|
+
"-f rdfxml this makes output byte-for-byte reproducible; other "
|
|
99
|
+
"formats still vary, as rdflib renames blank nodes on each run",
|
|
100
|
+
)
|
|
101
|
+
parser.add_argument(
|
|
102
|
+
"--version",
|
|
103
|
+
action="version",
|
|
104
|
+
version=f"marc-bibframe, marc2bibframe2 {upstream()['tag']}",
|
|
105
|
+
)
|
|
106
|
+
return parser.parse_args(argv)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def main(argv: list[str] | None = None) -> int:
|
|
110
|
+
args = parse_args(argv)
|
|
111
|
+
|
|
112
|
+
if args.input == "-":
|
|
113
|
+
data = sys.stdin.buffer.read()
|
|
114
|
+
else:
|
|
115
|
+
path = Path(args.input)
|
|
116
|
+
if not path.exists():
|
|
117
|
+
sys.exit(f"marc-bibframe: {path}: no such file")
|
|
118
|
+
data = path.read_bytes()
|
|
119
|
+
|
|
120
|
+
if not data.strip():
|
|
121
|
+
sys.exit("marc-bibframe: no input")
|
|
122
|
+
|
|
123
|
+
marcxml = data if looks_like_marcxml(data) else marc_to_marcxml(data)
|
|
124
|
+
|
|
125
|
+
params = {
|
|
126
|
+
"baseuri": args.baseuri,
|
|
127
|
+
"idfield": args.idfield,
|
|
128
|
+
"idsource": args.idsource,
|
|
129
|
+
"localfields": args.local_fields,
|
|
130
|
+
"bcp47_inference": args.bcp47_inference,
|
|
131
|
+
"generation_datestamp": args.datestamp,
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
rdflib_format = FORMATS[args.format]
|
|
135
|
+
if rdflib_format is None:
|
|
136
|
+
out = marcxml_to_rdfxml(marcxml, **params)
|
|
137
|
+
else:
|
|
138
|
+
out = marcxml_to_graph(marcxml, **params).serialize(
|
|
139
|
+
format=rdflib_format, encoding="utf-8"
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
if args.output:
|
|
143
|
+
Path(args.output).write_bytes(out)
|
|
144
|
+
else:
|
|
145
|
+
sys.stdout.buffer.write(out)
|
|
146
|
+
return 0
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
if __name__ == "__main__":
|
|
150
|
+
raise SystemExit(main())
|
marc_bibframe/py.typed
ADDED
|
File without changes
|