marc-bibframe 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. marc_bibframe/__init__.py +175 -0
  2. marc_bibframe/cli.py +150 -0
  3. marc_bibframe/py.typed +0 -0
  4. marc_bibframe/xsl/ConvSpec-001-007.xsl +2276 -0
  5. marc_bibframe/xsl/ConvSpec-006,008.xsl +1295 -0
  6. marc_bibframe/xsl/ConvSpec-010-048.xsl +1712 -0
  7. marc_bibframe/xsl/ConvSpec-050-088.xsl +713 -0
  8. marc_bibframe/xsl/ConvSpec-1XX,7XX,8XX-names.xsl +1141 -0
  9. marc_bibframe/xsl/ConvSpec-200-247not240-Titles.xsl +771 -0
  10. marc_bibframe/xsl/ConvSpec-240andX30-UnifTitle.xsl +535 -0
  11. marc_bibframe/xsl/ConvSpec-250-270.xsl +283 -0
  12. marc_bibframe/xsl/ConvSpec-3XX.xsl +2245 -0
  13. marc_bibframe/xsl/ConvSpec-460-468-SeriesTreat.xsl +135 -0
  14. marc_bibframe/xsl/ConvSpec-490-510-Links.xsl +145 -0
  15. marc_bibframe/xsl/ConvSpec-5XX.xsl +1448 -0
  16. marc_bibframe/xsl/ConvSpec-600-662.xsl +1417 -0
  17. marc_bibframe/xsl/ConvSpec-720+740to755.xsl +333 -0
  18. marc_bibframe/xsl/ConvSpec-758.xsl +102 -0
  19. marc_bibframe/xsl/ConvSpec-760-788-Links.xsl +771 -0
  20. marc_bibframe/xsl/ConvSpec-841-887.xsl +328 -0
  21. marc_bibframe/xsl/ConvSpec-880.xsl +119 -0
  22. marc_bibframe/xsl/ConvSpec-ControlSubfields.xsl +495 -0
  23. marc_bibframe/xsl/ConvSpec-LDR.xsl +183 -0
  24. marc_bibframe/xsl/ConvSpec-Preprocess0-Splitting.xsl +765 -0
  25. marc_bibframe/xsl/ConvSpec-Process6-Series.xsl +627 -0
  26. marc_bibframe/xsl/ConvSpec-Process8-ProvAct.xsl +1021 -0
  27. marc_bibframe/xsl/LICENSE +116 -0
  28. marc_bibframe/xsl/UPSTREAM +3 -0
  29. marc_bibframe/xsl/conf/abbreviations.xml +13 -0
  30. marc_bibframe/xsl/conf/code-to-script.xml +399 -0
  31. marc_bibframe/xsl/conf/codeMaps.xml +723 -0
  32. marc_bibframe/xsl/conf/exclusions.xml +4 -0
  33. marc_bibframe/xsl/conf/iso6392-to-1.xml +615 -0
  34. marc_bibframe/xsl/conf/languageCrosswalk.xml +1503 -0
  35. marc_bibframe/xsl/conf/map880.xml +175 -0
  36. marc_bibframe/xsl/conf/scriptCrosswalk.xml +210 -0
  37. marc_bibframe/xsl/conf/subjectThesaurus.xml +29 -0
  38. marc_bibframe/xsl/marc2bibframe2.xsl +610 -0
  39. marc_bibframe/xsl/naco-normalize.xsl +205 -0
  40. marc_bibframe/xsl/utils.xsl +964 -0
  41. marc_bibframe/xsl/variables.xsl +108 -0
  42. marc_bibframe-0.2.1.dist-info/METADATA +145 -0
  43. marc_bibframe-0.2.1.dist-info/RECORD +46 -0
  44. marc_bibframe-0.2.1.dist-info/WHEEL +4 -0
  45. marc_bibframe-0.2.1.dist-info/entry_points.txt +2 -0
  46. marc_bibframe-0.2.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,175 @@
1
+ """Convert MARC records to BIBFRAME RDF.
2
+
3
+ A thin Python wrapper around the Library of Congress marc2bibframe2 XSLT
4
+ (https://github.com/lcnetdev/marc2bibframe2), which is vendored in this
5
+ package. See ``src/marc_bibframe/xsl/UPSTREAM`` for the version, and
6
+ ``patches/`` for any local changes to it.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import atexit
12
+ import functools
13
+ from contextlib import ExitStack
14
+ from importlib.resources import as_file, files
15
+ from io import BytesIO
16
+ from typing import Any, BinaryIO
17
+
18
+ import lxml.etree as ET
19
+ import pymarc
20
+ from pymarc.marcxml import record_to_xml
21
+ from rdflib import Graph
22
+
23
+ __all__ = [
24
+ "DEFAULT_BASE_URI",
25
+ "marc_to_graph",
26
+ "marc_to_marcxml",
27
+ "marcxml_to_graph",
28
+ "marcxml_to_rdfxml",
29
+ "upstream",
30
+ ]
31
+
32
+ #: The stylesheet's own default. It is deliberately non-resolvable: URIs the
33
+ #: transform mints for entities that MARC does not identify (agents, topics,
34
+ #: the work and instance themselves) are built from it, and they name nothing.
35
+ DEFAULT_BASE_URI = "http://example.org/"
36
+
37
+ _MARCXML_NS = "http://www.loc.gov/MARC21/slim"
38
+ _COLLECTION_OPEN = (
39
+ b'<?xml version="1.0" encoding="UTF-8"?>'
40
+ b'<collection xmlns="' + _MARCXML_NS.encode() + b'">'
41
+ )
42
+ _COLLECTION_CLOSE = b"</collection>"
43
+
44
+ # The stylesheet xsl:includes ~30 siblings by relative href, so it has to be
45
+ # resolved from a real directory rather than read out of the package as bytes.
46
+ # Holding the ExitStack open for the life of the process keeps that directory
47
+ # around when the package is imported from a zip.
48
+ _files = ExitStack()
49
+ atexit.register(_files.close)
50
+
51
+
52
+ @functools.cache
53
+ def _xsl_dir():
54
+ return _files.enter_context(as_file(files(__package__).joinpath("xsl")))
55
+
56
+
57
+ @functools.cache
58
+ def _transform() -> ET.XSLT:
59
+ """Parse and compile the stylesheet, once per process (it is not cheap)."""
60
+ return ET.XSLT(ET.parse(str(_xsl_dir() / "marc2bibframe2.xsl")))
61
+
62
+
63
+ def upstream() -> dict[str, str]:
64
+ """The upstream marc2bibframe2 repository, tag and commit that is vendored here."""
65
+ text = (_xsl_dir() / "UPSTREAM").read_text()
66
+ return {
67
+ k.strip(): v.strip()
68
+ for k, _, v in (line.partition(":") for line in text.splitlines())
69
+ if k
70
+ }
71
+
72
+
73
+ def _xslt_params(**kwargs: Any) -> dict[str, Any]:
74
+ """Build XSLT parameters, dropping any left as None so the stylesheet default wins.
75
+
76
+ Booleans become the XPath expressions true()/false() rather than string
77
+ literals, because every non-empty string is true in XPath -- passing "false"
78
+ as a string would quietly mean the opposite of what was asked for.
79
+ """
80
+ params = {}
81
+ for name, value in kwargs.items():
82
+ if value is None:
83
+ continue
84
+ params[name] = (
85
+ "true()"
86
+ if value is True
87
+ else "false()"
88
+ if value is False
89
+ else ET.XSLT.strparam(str(value))
90
+ )
91
+ return params
92
+
93
+
94
+ def marc_to_marcxml(marc: bytes | BinaryIO) -> bytes:
95
+ """Convert binary MARC21 to a MARCXML ``<collection>`` of one or more records."""
96
+ handle = BytesIO(marc) if isinstance(marc, bytes) else marc
97
+ parts = [_COLLECTION_OPEN]
98
+ for i, record in enumerate(pymarc.MARCReader(handle)):
99
+ if record is None:
100
+ raise ValueError(f"Could not read MARC record at position {i}")
101
+ parts.append(record_to_xml(record, namespace=False))
102
+ parts.append(_COLLECTION_CLOSE)
103
+ return b"".join(parts)
104
+
105
+
106
+ def marcxml_to_rdfxml(
107
+ marcxml: str | bytes,
108
+ *,
109
+ baseuri: str = DEFAULT_BASE_URI,
110
+ idfield: str | None = None,
111
+ idsource: str | None = None,
112
+ localfields: bool | None = None,
113
+ bcp47_inference: bool | None = None,
114
+ generation_datestamp: str | None = None,
115
+ ) -> bytes:
116
+ """Transform MARCXML into BIBFRAME RDF/XML.
117
+
118
+ Accepts a single ``<record>`` or a ``<collection>`` of them.
119
+
120
+ baseuri
121
+ Base for the URIs the transform mints. Every minted URI has the form
122
+ ``{baseuri}{record id}#{fragment}``, so they are scoped to the record
123
+ they came from and are not authority URIs -- two records describing the
124
+ same person yield two different agent URIs. Reconciling them against an
125
+ authority is the caller's job.
126
+ idfield
127
+ MARC field holding the record id, ``001`` by default. Suffix a subfield
128
+ code to use one, e.g. ``035a``.
129
+ idsource
130
+ URI identifying the source of the record id, e.g.
131
+ ``http://id.loc.gov/vocabulary/organizations/dlc``.
132
+ localfields
133
+ Convert fields LC defines locally (e.g. 859), off by default.
134
+ bcp47_inference
135
+ Omit a BCP-47 script subtag when it can be inferred from the language.
136
+ generation_datestamp
137
+ Override the timestamp recorded in the work's admin metadata, which
138
+ otherwise defaults to now. Set it to make the RDF/XML this function
139
+ returns byte-for-byte reproducible. Note that a Graph serialized by
140
+ rdflib still varies between runs, because rdflib mints fresh blank
141
+ node labels each time.
142
+ """
143
+ if isinstance(marcxml, str):
144
+ # Encode first: lxml refuses a str carrying an encoding declaration.
145
+ marcxml = marcxml.encode("utf-8")
146
+ params = _xslt_params(
147
+ baseuri=baseuri,
148
+ idfield=idfield,
149
+ idsource=idsource,
150
+ localfields=localfields,
151
+ bcp47inferrence=bcp47_inference,
152
+ pGenerationDatestamp=generation_datestamp,
153
+ )
154
+ result = _transform()(ET.fromstring(marcxml), **params)
155
+ return ET.tostring(
156
+ result, xml_declaration=True, encoding="UTF-8", pretty_print=True
157
+ )
158
+
159
+
160
+ def marcxml_to_graph(marcxml: str | bytes, **kwargs: Any) -> Graph:
161
+ """Transform MARCXML into BIBFRAME as an rdflib Graph.
162
+
163
+ Takes the same keyword arguments as :func:`marcxml_to_rdfxml`.
164
+ """
165
+ graph = Graph()
166
+ graph.parse(data=marcxml_to_rdfxml(marcxml, **kwargs), format="xml")
167
+ return graph
168
+
169
+
170
+ def marc_to_graph(marc: bytes | BinaryIO, **kwargs: Any) -> Graph:
171
+ """Convert binary MARC21 straight to BIBFRAME as an rdflib Graph.
172
+
173
+ Takes the same keyword arguments as :func:`marcxml_to_rdfxml`.
174
+ """
175
+ return marcxml_to_graph(marc_to_marcxml(marc), **kwargs)
marc_bibframe/cli.py ADDED
@@ -0,0 +1,150 @@
1
+ """Command line interface: MARC in, BIBFRAME out."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ from marc_bibframe import (
10
+ DEFAULT_BASE_URI,
11
+ marc_to_marcxml,
12
+ marcxml_to_graph,
13
+ marcxml_to_rdfxml,
14
+ upstream,
15
+ )
16
+
17
+ # rdflib's name for each, except rdfxml, which we serve from the transform
18
+ # directly rather than round-tripping it through a parse.
19
+ FORMATS = {
20
+ "turtle": "turtle",
21
+ "ttl": "turtle",
22
+ "json-ld": "json-ld",
23
+ "jsonld": "json-ld",
24
+ "ntriples": "nt",
25
+ "nt": "nt",
26
+ "xml": "xml",
27
+ "rdfxml": None,
28
+ }
29
+
30
+
31
+ def looks_like_marcxml(data: bytes) -> bool:
32
+ return data.lstrip().lstrip(b"\xef\xbb\xbf").startswith(b"<")
33
+
34
+
35
+ def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
36
+ parser = argparse.ArgumentParser(
37
+ prog="marc-bibframe",
38
+ description="Convert MARC records to BIBFRAME RDF.",
39
+ epilog="Binary MARC21 and MARCXML are both accepted; the format is "
40
+ "detected from the content. With no INPUT, reads standard input.",
41
+ )
42
+ parser.add_argument(
43
+ "input",
44
+ nargs="?",
45
+ default="-",
46
+ help="MARC file to convert, or - for standard input (the default)",
47
+ )
48
+ parser.add_argument(
49
+ "-o",
50
+ "--output",
51
+ help="write to this file instead of standard output",
52
+ )
53
+ parser.add_argument(
54
+ "-f",
55
+ "--format",
56
+ default="turtle",
57
+ choices=sorted(FORMATS),
58
+ help="output serialization (default: turtle). rdfxml is the "
59
+ "stylesheet's own output, passed through unparsed",
60
+ )
61
+ parser.add_argument(
62
+ "-b",
63
+ "--baseuri",
64
+ default=DEFAULT_BASE_URI,
65
+ metavar="URI",
66
+ help="base for the URIs the transform mints, which are NOT authority "
67
+ f"URIs -- see the README (default: {DEFAULT_BASE_URI})",
68
+ )
69
+ parser.add_argument(
70
+ "--idfield",
71
+ metavar="FIELD",
72
+ help="MARC field holding the record id, 001 by default. Suffix a "
73
+ "subfield code to use one, e.g. 035a",
74
+ )
75
+ parser.add_argument(
76
+ "--idsource",
77
+ metavar="URI",
78
+ help="URI identifying the source of the record id, e.g. "
79
+ "http://id.loc.gov/vocabulary/organizations/dlc",
80
+ )
81
+ parser.add_argument(
82
+ "--local-fields",
83
+ action="store_true",
84
+ default=None,
85
+ help="convert fields the Library of Congress defines locally, e.g. 859",
86
+ )
87
+ parser.add_argument(
88
+ "--no-bcp47-inference",
89
+ dest="bcp47_inference",
90
+ action="store_false",
91
+ default=None,
92
+ help="keep the script subtag in BCP-47 codes even when the language implies it",
93
+ )
94
+ parser.add_argument(
95
+ "--datestamp",
96
+ metavar="TIMESTAMP",
97
+ help="override the timestamp in the work's admin metadata. With "
98
+ "-f rdfxml this makes output byte-for-byte reproducible; other "
99
+ "formats still vary, as rdflib renames blank nodes on each run",
100
+ )
101
+ parser.add_argument(
102
+ "--version",
103
+ action="version",
104
+ version=f"marc-bibframe, marc2bibframe2 {upstream()['tag']}",
105
+ )
106
+ return parser.parse_args(argv)
107
+
108
+
109
+ def main(argv: list[str] | None = None) -> int:
110
+ args = parse_args(argv)
111
+
112
+ if args.input == "-":
113
+ data = sys.stdin.buffer.read()
114
+ else:
115
+ path = Path(args.input)
116
+ if not path.exists():
117
+ sys.exit(f"marc-bibframe: {path}: no such file")
118
+ data = path.read_bytes()
119
+
120
+ if not data.strip():
121
+ sys.exit("marc-bibframe: no input")
122
+
123
+ marcxml = data if looks_like_marcxml(data) else marc_to_marcxml(data)
124
+
125
+ params = {
126
+ "baseuri": args.baseuri,
127
+ "idfield": args.idfield,
128
+ "idsource": args.idsource,
129
+ "localfields": args.local_fields,
130
+ "bcp47_inference": args.bcp47_inference,
131
+ "generation_datestamp": args.datestamp,
132
+ }
133
+
134
+ rdflib_format = FORMATS[args.format]
135
+ if rdflib_format is None:
136
+ out = marcxml_to_rdfxml(marcxml, **params)
137
+ else:
138
+ out = marcxml_to_graph(marcxml, **params).serialize(
139
+ format=rdflib_format, encoding="utf-8"
140
+ )
141
+
142
+ if args.output:
143
+ Path(args.output).write_bytes(out)
144
+ else:
145
+ sys.stdout.buffer.write(out)
146
+ return 0
147
+
148
+
149
+ if __name__ == "__main__":
150
+ raise SystemExit(main())
marc_bibframe/py.typed ADDED
File without changes