sslabdata 3.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sslabdata/__init__.py +40 -0
- sslabdata/assembler.py +452 -0
- sslabdata/cli.py +274 -0
- sslabdata/config.py +443 -0
- sslabdata/diagnostics.py +159 -0
- sslabdata/exporters.py +86 -0
- sslabdata/loaders.py +298 -0
- sslabdata/models.py +342 -0
- sslabdata/parsers/__init__.py +0 -0
- sslabdata/parsers/bibtex.py +1132 -0
- sslabdata/parsers/latex.py +215 -0
- sslabdata/resolver.py +517 -0
- sslabdata/schema/__init__.py +0 -0
- sslabdata/schema/v5/output.schema.json +458 -0
- sslabdata-3.0.0.dist-info/METADATA +409 -0
- sslabdata-3.0.0.dist-info/RECORD +20 -0
- sslabdata-3.0.0.dist-info/WHEEL +5 -0
- sslabdata-3.0.0.dist-info/entry_points.txt +2 -0
- sslabdata-3.0.0.dist-info/licenses/LICENSE +21 -0
- sslabdata-3.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1132 @@
|
|
|
1
|
+
"""
|
|
2
|
+
BibTeX parsing pipeline.
|
|
3
|
+
|
|
4
|
+
pybtex reads the files — @string macros, BibTeX's own name splitting, entry
|
|
5
|
+
order — and this module maps its Entry and Person objects onto sslabdata's
|
|
6
|
+
Work model: per-field LaTeX conversion (latex.py), the structured venue,
|
|
7
|
+
identifiers, links and diagnostics.
|
|
8
|
+
|
|
9
|
+
Together with latex.py this is the adapter: no other module imports pybtex or
|
|
10
|
+
pylatexenc, and nothing here lets a library object or a library message reach
|
|
11
|
+
the rest of sslabdata.
|
|
12
|
+
|
|
13
|
+
Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
|
|
14
|
+
Author: Siddhartha Srinivasa
|
|
15
|
+
MIT License - see LICENSE file for details.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Callable, Dict, List, Optional, Tuple
|
|
21
|
+
from urllib.parse import urlsplit
|
|
22
|
+
|
|
23
|
+
import pybtex.errors
|
|
24
|
+
from pybtex.database import BibliographyData, Entry, Person
|
|
25
|
+
from pybtex.exceptions import PybtexError
|
|
26
|
+
from pybtex.database.output.bibtex import Writer as BibTeXWriter
|
|
27
|
+
from pybtex.database.input.bibtex import (
|
|
28
|
+
LowLevelParser, Parser as PybtexParser, SkipEntry, UndefinedMacro,
|
|
29
|
+
)
|
|
30
|
+
from pybtex.scanner import PybtexSyntaxError
|
|
31
|
+
|
|
32
|
+
from .latex import latex_to_text, strip_braces, unknown_commands
|
|
33
|
+
from ..config import (
|
|
34
|
+
CONTROL_CHARACTER, control_message, nfc, without_control_characters,
|
|
35
|
+
)
|
|
36
|
+
from ..diagnostics import Diagnostic, diagnostic
|
|
37
|
+
from ..models import Author, Contributor, Link, Venue, Work
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# The fields converted from LaTeX to plain text (SPEC.md §2). The repository
|
|
41
|
+
# fields are among them because braces around `{arXiv}` are grouping, not part
|
|
42
|
+
# of the name.
|
|
43
|
+
TEXT_FIELDS = frozenset({
|
|
44
|
+
"title", "abstract", "note", "journal", "booktitle", "school",
|
|
45
|
+
"institution", "type", "series", "publisher", "address", "organization",
|
|
46
|
+
"archiveprefix", "eprinttype",
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
# A name list ending in "and others" means "et al."; it is not an author.
|
|
50
|
+
OTHERS = "others"
|
|
51
|
+
|
|
52
|
+
# The codes this module reports (SPEC.md "Diagnostic codes").
|
|
53
|
+
OTHERS_NOT_LAST = "BIB-OTHERS-NOT-LAST"
|
|
54
|
+
DUPLICATE_CITATION_KEY = "BIB-DUPLICATE-KEY"
|
|
55
|
+
CROSSREF_UNSUPPORTED = "BIB-CROSSREF-UNSUPPORTED"
|
|
56
|
+
YEAR_MISSING = "BIB-YEAR-MISSING"
|
|
57
|
+
YEAR_INVALID = "BIB-YEAR-INVALID"
|
|
58
|
+
YEAR_DIGITS = re.compile(r"[0-9]+")
|
|
59
|
+
DOI_INVALID = "BIB-DOI-INVALID"
|
|
60
|
+
STRING_UNDEFINED = "BIB-STRING-UNDEFINED"
|
|
61
|
+
SYNTAX_ERROR = "BIB-SYNTAX-ERROR"
|
|
62
|
+
VENUE_MISSING = "BIB-VENUE-MISSING"
|
|
63
|
+
ENTRY_TYPE_UNSUPPORTED = "BIB-ENTRY-TYPE-UNSUPPORTED"
|
|
64
|
+
STRING_REDEFINED = "BIB-STRING-REDEFINED"
|
|
65
|
+
PARSER_MESSAGE = "BIB-PARSER-MESSAGE"
|
|
66
|
+
LATEX_CONVERSION_FAILED = "LATEX-CONVERSION-FAILED"
|
|
67
|
+
WRITE_BACK_FAILED = "BIB-WRITE-BACK-FAILED"
|
|
68
|
+
ENCODING_INVALID = "BIB-ENCODING-INVALID"
|
|
69
|
+
LATEX_COMMAND_UNKNOWN = "LATEX-COMMAND-UNKNOWN"
|
|
70
|
+
|
|
71
|
+
# Equal contribution is written as a star on one part of a name, in one of
|
|
72
|
+
# these four forms. It is an annotation rather than part of the name, so it is
|
|
73
|
+
# taken off the part before the name is read and recorded on the author
|
|
74
|
+
# instead. Other author annotations — corresponding author, affiliation
|
|
75
|
+
# numbers, daggers — are not read.
|
|
76
|
+
#
|
|
77
|
+
# A star, caret or dollar written with a backslash in front of it is escaped
|
|
78
|
+
# text rather than the start of a marker: `Brown\*` is not marked, and the
|
|
79
|
+
# ordinary LaTeX conversion then consumes the escaped star. The accent in
|
|
80
|
+
# `C{\^o}t{\'e}$^{*}$` is escaped the same way, and the marker after it is
|
|
81
|
+
# not, which is why that name reads `Côté` and is marked.
|
|
82
|
+
_WRITTEN = r"\$\^\{\*\}\$|\^\{\*\}|\\textsuperscript\s*\{\*\}"
|
|
83
|
+
_MARKER = rf"(?<!\\)(?:{_WRITTEN}|\*)"
|
|
84
|
+
|
|
85
|
+
# A marker at the end of a name part, on its own or in a brace group of its
|
|
86
|
+
# own: BibTeX grouping such as `Brown{$^{*}$}` protects the marker from the
|
|
87
|
+
# name, and does not make it part of it. Only a form that brings its own
|
|
88
|
+
# command is unwrapped, because a lone `{*}` is how any other command takes
|
|
89
|
+
# its argument — the star in `Brown\^{*}` is an accented star, not a marker.
|
|
90
|
+
_ANY_MARKER = rf"(?:(?<!\\)\{{\s*(?:{_WRITTEN})\s*\}}|{_MARKER})"
|
|
91
|
+
EQUAL_CONTRIBUTION = re.compile(rf"{_ANY_MARKER}\s*$")
|
|
92
|
+
|
|
93
|
+
# `\textsuperscript {*}`, with a space before the argument, is the same form:
|
|
94
|
+
# BibTeX splits a name on spaces, so the command and its argument arrive as two
|
|
95
|
+
# name parts and neither is a marker on its own. Only this one command takes
|
|
96
|
+
# its argument back, written as the command and not as an escaped backslash,
|
|
97
|
+
# and only from a part that is the argument and nothing but further markers —
|
|
98
|
+
# so the accent in `Brown\^ {*}`, a `{*}` after any other command, and a part
|
|
99
|
+
# carrying text of its own all keep their own boundary. What the part carries
|
|
100
|
+
# after `{*}` is left to the stripping below, as it is for an unspaced marker.
|
|
101
|
+
_MARKER_COMMAND = re.compile(r"(?<!\\)\\textsuperscript\s*$")
|
|
102
|
+
_MARKER_ARGUMENT = re.compile(rf"\{{\*\}}(?:{_ANY_MARKER})*")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# The command name pybtex is about to read, when that name is `comment`.
|
|
106
|
+
_COMMENT_COMMAND = re.compile(r'\s*comment\s*[{(]', re.IGNORECASE)
|
|
107
|
+
|
|
108
|
+
# The two shapes `BIB-BRACE-MISMATCH` looks for (SPEC.md "Diagnostic codes").
|
|
109
|
+
BRACE_MISMATCH = "BIB-BRACE-MISMATCH"
|
|
110
|
+
_FIELD_IN_VALUE = re.compile(r',\s*([A-Za-z][\w-]*)\s*=\s*[{"]')
|
|
111
|
+
_TEXT_AFTER_ENTRY = re.compile(r'[ \t]*([^\s%@][^\r\n]*)')
|
|
112
|
+
_FIELD_AFTER_ENTRY = re.compile(r'\s*,\s*([A-Za-z][\w-]*)\s*=')
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _redefined_macros(text: str,
|
|
117
|
+
definitions: List[Tuple[str, int]]) -> List[Tuple[str, int]]:
|
|
118
|
+
"""Every redefinition of an ``@string`` macro, as ``(name, line)``, in
|
|
119
|
+
source order: each definition after a macro's first.
|
|
120
|
+
|
|
121
|
+
``definitions`` are the ``(name, offset)`` of every ``@string`` the parser
|
|
122
|
+
actually read from ``text``, so a definition inside an ``@comment`` group
|
|
123
|
+
is not one. Names compare without case, as the parser's macros do.
|
|
124
|
+
"""
|
|
125
|
+
seen: set = set()
|
|
126
|
+
repeated: List[Tuple[str, int]] = []
|
|
127
|
+
for name, offset in definitions:
|
|
128
|
+
name = name.lower()
|
|
129
|
+
if name in seen:
|
|
130
|
+
repeated.append((name, text.count("\n", 0, offset) + 1))
|
|
131
|
+
else:
|
|
132
|
+
seen.add(name)
|
|
133
|
+
return repeated
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def redefined_summary(redefinitions: List[Tuple[str, str, int]]) -> Optional[Diagnostic]:
|
|
137
|
+
"""One `STRING_REDEFINED` line for ``(file, name, line)`` redefinitions,
|
|
138
|
+
in the form SPEC.md §7 shows; None when there is nothing to say."""
|
|
139
|
+
if not redefinitions:
|
|
140
|
+
return None
|
|
141
|
+
names = sorted({name for _, name, _ in redefinitions})
|
|
142
|
+
files = sorted({path for path, _, _ in redefinitions})
|
|
143
|
+
where = ", ".join(f"{path}:{line}" for path, _, line in sorted(
|
|
144
|
+
redefinitions, key=lambda found: (found[0], found[2])))
|
|
145
|
+
count = f"{len(names)} @string macro{'s' if len(names) != 1 else ''}"
|
|
146
|
+
return diagnostic(
|
|
147
|
+
STRING_REDEFINED, files[0] if len(files) == 1 else None, None, None,
|
|
148
|
+
f"{count} redefined (last definition used): {', '.join(names)} "
|
|
149
|
+
f"[{where}]")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class _CommentSkippingParser(LowLevelParser):
|
|
153
|
+
"""pybtex's tokenizer, with a balanced ``@comment{...}`` group stepped over.
|
|
154
|
+
|
|
155
|
+
pybtex raises ``SkipEntry`` for ``@comment`` before reading the body, so
|
|
156
|
+
the scanner resumes just inside the group and an entry written there is a
|
|
157
|
+
real entry to it — BibTeX behaves the same way. sslabdata treats a
|
|
158
|
+
commented-out entry as commented out, so the group is consumed here, at
|
|
159
|
+
the parser's own position and with the parser's own scanner. Nothing else
|
|
160
|
+
in the file is read by sslabdata, which is why the shape of a value or of a
|
|
161
|
+
neighbouring command cannot be got wrong.
|
|
162
|
+
|
|
163
|
+
Only a *balanced* group is consumed. Prose that merely mentions
|
|
164
|
+
``@comment{`` does not close, so the position is put back and pybtex reads
|
|
165
|
+
the rest of the file itself: at worst a commented-out entry stays visible,
|
|
166
|
+
never a real entry disappears.
|
|
167
|
+
"""
|
|
168
|
+
|
|
169
|
+
# The scanner's position, which pybtex's `Scanner` sets.
|
|
170
|
+
pos: int
|
|
171
|
+
lineno: int
|
|
172
|
+
|
|
173
|
+
def __init__(self, *args, **kwargs):
|
|
174
|
+
super().__init__(*args, **kwargs)
|
|
175
|
+
# (macro name, offset of its `@`) for every `@string` read in full.
|
|
176
|
+
self.definitions: List[Tuple[str, int]] = []
|
|
177
|
+
|
|
178
|
+
# The ingress for a citation key: it is put in NFC as the tokenizer reads
|
|
179
|
+
# it, so the key an entry is stored, compared and located under is one
|
|
180
|
+
# string however its accents were written (SPEC.md §2).
|
|
181
|
+
@property
|
|
182
|
+
def current_entry_key(self) -> Optional[str]:
|
|
183
|
+
return self._entry_key
|
|
184
|
+
|
|
185
|
+
@current_entry_key.setter
|
|
186
|
+
def current_entry_key(self, key: Optional[str]) -> None:
|
|
187
|
+
self._entry_key = None if key is None else nfc(key)
|
|
188
|
+
|
|
189
|
+
def parse_string_body(self, body_end):
|
|
190
|
+
"""Read one ``@string`` body, and remember the definition it made."""
|
|
191
|
+
super().parse_string_body(body_end)
|
|
192
|
+
self.definitions.append((self.current_field_name, self.command_start))
|
|
193
|
+
|
|
194
|
+
def parse_command(self):
|
|
195
|
+
# pybtex raises SkipEntry for a @comment, and also for an entry that
|
|
196
|
+
# `wanted_entries` filters out — which leaves the scanner somewhere
|
|
197
|
+
# quite different. Only the first is ours to recover from, so the
|
|
198
|
+
# command is identified before the parser reads it.
|
|
199
|
+
comment = _COMMENT_COMMAND.match(self.text, self.pos) is not None
|
|
200
|
+
try:
|
|
201
|
+
return super().parse_command()
|
|
202
|
+
except SkipEntry:
|
|
203
|
+
if comment:
|
|
204
|
+
position, lineno = self.pos, self.lineno
|
|
205
|
+
if not self._skip_comment_group():
|
|
206
|
+
self.pos, self.lineno = position, lineno
|
|
207
|
+
raise
|
|
208
|
+
|
|
209
|
+
def _skip_comment_group(self) -> bool:
|
|
210
|
+
"""Consume the ``@comment`` body just opened. False if it never closes."""
|
|
211
|
+
closing = self.RBRACE if self.text[self.pos - 1] == "{" else self.RPAREN
|
|
212
|
+
while True:
|
|
213
|
+
token = self.skip_to([closing, self.LBRACE])
|
|
214
|
+
if token is None:
|
|
215
|
+
return False
|
|
216
|
+
if token.pattern is closing:
|
|
217
|
+
return True
|
|
218
|
+
try:
|
|
219
|
+
for _ in self.parse_string(self.RBRACE):
|
|
220
|
+
pass
|
|
221
|
+
except PybtexSyntaxError:
|
|
222
|
+
return False
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class _Parser(PybtexParser):
|
|
226
|
+
"""pybtex's BibTeX parser, reading ``@comment`` groups as comments.
|
|
227
|
+
|
|
228
|
+
``Parser.parse_string`` names ``LowLevelParser`` directly, so swapping the
|
|
229
|
+
tokenizer means restating that loop. It and `_VerbatimWriter` are the two
|
|
230
|
+
places sslabdata touches a pybtex internal, which is why ``pybtex~=0.26``
|
|
231
|
+
is pinned.
|
|
232
|
+
"""
|
|
233
|
+
|
|
234
|
+
def __init__(self, *args, duplicate_keys=None, **kwargs):
|
|
235
|
+
super().__init__(*args, **kwargs)
|
|
236
|
+
self.duplicate_keys = duplicate_keys if duplicate_keys is not None else []
|
|
237
|
+
# (error, entry key, field name, where the command began) for every
|
|
238
|
+
# syntax error, with the tokenizer's position in the file as it was
|
|
239
|
+
# when the error was raised: by the time the file is read, it has
|
|
240
|
+
# moved on.
|
|
241
|
+
self.syntax_errors: List[Tuple[PybtexSyntaxError, Optional[str],
|
|
242
|
+
Optional[str], Optional[int]]] = []
|
|
243
|
+
# The entry key each captured library message was raised while
|
|
244
|
+
# reading, by the message's identity.
|
|
245
|
+
self.message_keys: Dict[int, str] = {}
|
|
246
|
+
# (entry key, field name, characters) for every field value its
|
|
247
|
+
# control characters were removed from.
|
|
248
|
+
self.control_characters: List[Tuple[str, str, List[str]]] = []
|
|
249
|
+
# (entry key, field name, message) for every entry whose braces
|
|
250
|
+
# read it differently from how it was written (`BRACE_MISMATCH`).
|
|
251
|
+
self.brace_mismatches: List[Tuple[str, Optional[str], str]] = []
|
|
252
|
+
|
|
253
|
+
def handle_error(self, error):
|
|
254
|
+
"""Keep a syntax error with where it happened; relay anything else."""
|
|
255
|
+
if isinstance(error, PybtexSyntaxError):
|
|
256
|
+
tokenizer = error.parser
|
|
257
|
+
self.syntax_errors.append((error, tokenizer.current_entry_key,
|
|
258
|
+
tokenizer.current_field_name,
|
|
259
|
+
tokenizer.command_start))
|
|
260
|
+
return
|
|
261
|
+
super().handle_error(error)
|
|
262
|
+
|
|
263
|
+
def process_entry(self, entry_type, key, fields):
|
|
264
|
+
"""Remember duplicate keys before pybtex discards the later entry.
|
|
265
|
+
|
|
266
|
+
``BibliographyData.add_entry`` reports a parser-library warning and
|
|
267
|
+
keeps the first entry. Recording the key here lets sslabdata surface a
|
|
268
|
+
stable, file-qualified diagnostic instead of exposing that wording.
|
|
269
|
+
"""
|
|
270
|
+
if key is not None and key in self.data.entries:
|
|
271
|
+
self.duplicate_keys.append(key)
|
|
272
|
+
# The ingress for field text: control characters are removed here,
|
|
273
|
+
# and the value put in NFC, before pybtex splits a name list or
|
|
274
|
+
# normalises whitespace, and before any LaTeX is read
|
|
275
|
+
# (`CONTROL_CHARACTER`, SPEC.md §2).
|
|
276
|
+
fields = [(name, self._without_controls(key, name, parts))
|
|
277
|
+
for name, parts in fields]
|
|
278
|
+
captured = pybtex.errors.captured_errors
|
|
279
|
+
before = len(captured) if captured is not None else 0
|
|
280
|
+
super().process_entry(entry_type, key, fields)
|
|
281
|
+
for error in (captured or [])[before:]:
|
|
282
|
+
self.message_keys[id(error)] = key
|
|
283
|
+
|
|
284
|
+
def _without_controls(self, key, name: str, parts: List[str]) -> List[str]:
|
|
285
|
+
"""One field's value, joined from its parts, with control characters
|
|
286
|
+
removed and in NFC, and the field recorded when there were any
|
|
287
|
+
control characters. Joined first, because a part can begin with a
|
|
288
|
+
mark that composes with the end of the one before."""
|
|
289
|
+
cleaned = [without_control_characters(part) for part in parts]
|
|
290
|
+
found = list(dict.fromkeys(c for _, chars in cleaned for c in chars))
|
|
291
|
+
if found:
|
|
292
|
+
self.control_characters.append((key, name.lower(), found))
|
|
293
|
+
return [nfc("".join(part for part, _ in cleaned))]
|
|
294
|
+
|
|
295
|
+
def parse_string(self, text: str):
|
|
296
|
+
self.unnamed_entry_counter = 1
|
|
297
|
+
self.command_start = 0
|
|
298
|
+
commands = _CommentSkippingParser(
|
|
299
|
+
text,
|
|
300
|
+
keyless_entries=self.keyless_entries,
|
|
301
|
+
handle_error=self.handle_error,
|
|
302
|
+
want_entry=self.data.want_entry,
|
|
303
|
+
filename=self.filename,
|
|
304
|
+
macros=self.macros,
|
|
305
|
+
)
|
|
306
|
+
self.string_definitions = commands.definitions
|
|
307
|
+
read = 0
|
|
308
|
+
for command, arguments in commands:
|
|
309
|
+
kind = command.lower()
|
|
310
|
+
if kind == "preamble":
|
|
311
|
+
self.process_preamble(*arguments)
|
|
312
|
+
elif kind != "string":
|
|
313
|
+
# An entry with a syntax error already says its values may
|
|
314
|
+
# hold text meant for later fields. An undefined macro is
|
|
315
|
+
# kept as a syntax error too, but says nothing of the kind.
|
|
316
|
+
key = arguments[0]
|
|
317
|
+
if all(error[1] != key or isinstance(error[0], UndefinedMacro)
|
|
318
|
+
for error in self.syntax_errors[read:]):
|
|
319
|
+
self._check_braces(commands, *arguments)
|
|
320
|
+
self.process_entry(command, *arguments)
|
|
321
|
+
read = len(self.syntax_errors)
|
|
322
|
+
return self.data
|
|
323
|
+
|
|
324
|
+
def _check_braces(self, commands, key, fields) -> None:
|
|
325
|
+
"""Record the entry just read when its braces read it differently
|
|
326
|
+
from how it was written (`BRACE_MISMATCH`), at the first field whose
|
|
327
|
+
text did not arrive as written."""
|
|
328
|
+
for name, parts in fields:
|
|
329
|
+
merged = _FIELD_IN_VALUE.search("".join(parts))
|
|
330
|
+
if merged:
|
|
331
|
+
other = merged.group(1)
|
|
332
|
+
self.brace_mismatches.append((key, other.lower(), (
|
|
333
|
+
f"the field '{other}' is read into the value of '{name}', "
|
|
334
|
+
f"because a brace in '{name}' is not closed where it was "
|
|
335
|
+
"meant to be")))
|
|
336
|
+
return
|
|
337
|
+
after = _TEXT_AFTER_ENTRY.match(commands.text, commands.pos)
|
|
338
|
+
if after and fields:
|
|
339
|
+
name = fields[-1][0]
|
|
340
|
+
lost = _FIELD_AFTER_ENTRY.match(commands.text, commands.pos)
|
|
341
|
+
line = commands.text.count("\n", 0, commands.pos) + 1
|
|
342
|
+
self.brace_mismatches.append((
|
|
343
|
+
key, (lost.group(1) if lost else name).lower(),
|
|
344
|
+
f"a closing brace in the value of '{name}' ends the entry at "
|
|
345
|
+
f"line {line}, so " + (
|
|
346
|
+
f"the field '{lost.group(1)}' and any after it are not read"
|
|
347
|
+
if lost else
|
|
348
|
+
f"'{after.group(1).strip()}' after it is not read")))
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _duplicate_key_error(
|
|
352
|
+
path: str,
|
|
353
|
+
key: str,
|
|
354
|
+
first_path: Optional[str] = None,
|
|
355
|
+
first_key: Optional[str] = None,
|
|
356
|
+
) -> Diagnostic:
|
|
357
|
+
"""One stable duplicate-key diagnostic, with both locations when known."""
|
|
358
|
+
message = "duplicate citation key"
|
|
359
|
+
if first_path is not None:
|
|
360
|
+
first_location = f"{first_path}:{first_key or key}:citation_key"
|
|
361
|
+
message += f"; first defined in {first_location}"
|
|
362
|
+
return diagnostic(DUPLICATE_CITATION_KEY, path, key, "citation_key", message)
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def _on_comment_line(text: str, position: Optional[int]) -> bool:
|
|
366
|
+
"""True when ``position`` is on a line that starts with `%`.
|
|
367
|
+
|
|
368
|
+
The parser library reads an `@` anywhere outside an entry as the start of
|
|
369
|
+
a command, so prose on a `%` line that mentions `@article` fails to parse.
|
|
370
|
+
That failure is not reported: the prose was never meant as BibTeX. Only
|
|
371
|
+
the report is suppressed. A well-formed command on such a line is read,
|
|
372
|
+
as the library reads it.
|
|
373
|
+
"""
|
|
374
|
+
if position is None:
|
|
375
|
+
return False
|
|
376
|
+
line_start = text.rfind("\n", 0, position) + 1
|
|
377
|
+
return text[line_start:position].lstrip().startswith("%")
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _syntax_diagnostic(path: str, error: PybtexSyntaxError,
|
|
381
|
+
key: Optional[str], field_name: Optional[str]) -> Diagnostic:
|
|
382
|
+
"""One parser-library syntax error, in sslabdata's voice and located."""
|
|
383
|
+
if isinstance(error, UndefinedMacro):
|
|
384
|
+
macro = str(error).rsplit(": ", 1)[-1]
|
|
385
|
+
return diagnostic(
|
|
386
|
+
STRING_UNDEFINED, path, key, field_name if key else None,
|
|
387
|
+
f"the macro '{macro}' is not defined by any @string; it is read "
|
|
388
|
+
"as empty, and the entry is kept")
|
|
389
|
+
if key is None:
|
|
390
|
+
return diagnostic(
|
|
391
|
+
SYNTAX_ERROR, path, None, None,
|
|
392
|
+
f"the text at line {error.lineno} does not read as BibTeX and is "
|
|
393
|
+
"skipped")
|
|
394
|
+
after = f", after the value of '{field_name}'" if field_name else ""
|
|
395
|
+
return diagnostic(
|
|
396
|
+
SYNTAX_ERROR, path, key, field_name,
|
|
397
|
+
f"the entry stops reading as BibTeX at line {error.lineno}{after}. "
|
|
398
|
+
"It is kept as far as it was read, so that value may hold text meant "
|
|
399
|
+
"for later fields; check its braces and quotes")
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def parse_bibtex_file(
|
|
403
|
+
path: str,
|
|
404
|
+
diagnostics: List[Diagnostic],
|
|
405
|
+
redefinitions: List[Tuple[str, str, int]],
|
|
406
|
+
) -> Dict[str, Entry]:
|
|
407
|
+
"""Parse one BibTeX file into pybtex entries, keyed by citation key.
|
|
408
|
+
|
|
409
|
+
Anything the parser has to say is captured and reported by sslabdata, so no
|
|
410
|
+
library logging reaches the user. Syntax errors, undefined macros and
|
|
411
|
+
repeated citation keys go to ``diagnostics``, located at the entry and
|
|
412
|
+
field they were found in. Redefined ``@string`` macros are added to
|
|
413
|
+
``redefinitions``, for the caller to summarise over a whole run.
|
|
414
|
+
"""
|
|
415
|
+
text = Path(path).read_text(encoding="utf-8-sig")
|
|
416
|
+
|
|
417
|
+
duplicate_keys: List[str] = []
|
|
418
|
+
with pybtex.errors.capture() as errors:
|
|
419
|
+
parser = _Parser(duplicate_keys=duplicate_keys)
|
|
420
|
+
data = parser.parse_string(text)
|
|
421
|
+
|
|
422
|
+
redefinitions.extend((path, name, line) for name, line
|
|
423
|
+
in _redefined_macros(text, parser.string_definitions))
|
|
424
|
+
key: Optional[str]
|
|
425
|
+
field_name: Optional[str]
|
|
426
|
+
for key, field_name, found in parser.control_characters:
|
|
427
|
+
diagnostics.append(diagnostic(CONTROL_CHARACTER, path, key, field_name,
|
|
428
|
+
control_message(found)))
|
|
429
|
+
for key, field_name, message in parser.brace_mismatches:
|
|
430
|
+
diagnostics.append(diagnostic(BRACE_MISMATCH, path, key, field_name,
|
|
431
|
+
f"{message}; check its braces"))
|
|
432
|
+
for error, key, field_name, start in parser.syntax_errors:
|
|
433
|
+
if key is None and _on_comment_line(text, start):
|
|
434
|
+
continue
|
|
435
|
+
diagnostics.append(_syntax_diagnostic(path, error, key, field_name))
|
|
436
|
+
for key in duplicate_keys:
|
|
437
|
+
diagnostics.append(_duplicate_key_error(path, key))
|
|
438
|
+
for error in errors:
|
|
439
|
+
# The duplicate has already been recorded with sslabdata's stable code.
|
|
440
|
+
if str(error).startswith("repeated bibliography entry:"):
|
|
441
|
+
continue
|
|
442
|
+
diagnostics.append(diagnostic(PARSER_MESSAGE, path,
|
|
443
|
+
parser.message_keys.get(id(error)), None,
|
|
444
|
+
str(error)))
|
|
445
|
+
|
|
446
|
+
return data.entries
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
_UNREADABLE_LATEX = ("could not read the LaTeX in this field; keeping the text "
|
|
451
|
+
"as written")
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
class _FieldReport:
|
|
455
|
+
"""Where one field's unknown LaTeX commands, and LaTeX that cannot be read
|
|
456
|
+
at all, are reported: called with a command, or ``failed()``."""
|
|
457
|
+
|
|
458
|
+
def __init__(self, unknown: Callable[[str], None],
|
|
459
|
+
failed: Callable[[], None]):
|
|
460
|
+
self.unknown = unknown
|
|
461
|
+
self.failed = failed
|
|
462
|
+
|
|
463
|
+
def __call__(self, command: str) -> None:
|
|
464
|
+
self.unknown(command)
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _convert(value: str, on_unknown: _FieldReport) -> str:
|
|
468
|
+
"""Convert one field from LaTeX, keeping the raw text if that fails.
|
|
469
|
+
|
|
470
|
+
Each command the converter does not know is passed to ``on_unknown``,
|
|
471
|
+
which knows where the field is, and so is a value it cannot read at all
|
|
472
|
+
(``on_unknown.failed()``).
|
|
473
|
+
|
|
474
|
+
Every converted value leaves here, in NFC: the value was NFC as read, but
|
|
475
|
+
dropping a brace can bring a letter and its mark together, `n{\u0303}`
|
|
476
|
+
(SPEC.md §2).
|
|
477
|
+
"""
|
|
478
|
+
try:
|
|
479
|
+
text = latex_to_text(value)
|
|
480
|
+
except Exception: # noqa: BLE001 - never drop an entry over one field
|
|
481
|
+
on_unknown.failed()
|
|
482
|
+
text = strip_braces(value)
|
|
483
|
+
else:
|
|
484
|
+
for command in unknown_commands(value):
|
|
485
|
+
on_unknown(command)
|
|
486
|
+
return nfc(text)
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def unknown_command_diagnostic(command: str, where: Tuple[str, str, str],
|
|
490
|
+
fields: int) -> Diagnostic:
|
|
491
|
+
"""One `LATEX_COMMAND_UNKNOWN` line for a whole run: how many fields use
|
|
492
|
+
the command, located at ``(file, key, field)``, the first of them."""
|
|
493
|
+
uses = (f"; used in {fields} field{'s' if fields != 1 else ''}, "
|
|
494
|
+
"located at the first")
|
|
495
|
+
return diagnostic(
|
|
496
|
+
LATEX_COMMAND_UNKNOWN, *where,
|
|
497
|
+
f"the LaTeX command '\\{command}' is not one sslabdata converts; it "
|
|
498
|
+
f"is dropped, and a braced argument after it is kept as plain "
|
|
499
|
+
f"text{uses}")
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _unknown_command_reporter(report, file: str, key: str,
|
|
503
|
+
tally: Dict[str, list]):
|
|
504
|
+
"""For one entry: a field name → the ``on_unknown`` for that field.
|
|
505
|
+
|
|
506
|
+
A command is counted once per field, however many times it is used, and
|
|
507
|
+
added to ``tally[command]`` as ``[fields, first location]``, for a caller
|
|
508
|
+
that reports a whole run in one line per command. A field whose LaTeX
|
|
509
|
+
cannot be read is reported to ``report`` at once.
|
|
510
|
+
"""
|
|
511
|
+
reported = set()
|
|
512
|
+
|
|
513
|
+
def in_field(field_name: str) -> _FieldReport:
|
|
514
|
+
def on_unknown(command: str) -> None:
|
|
515
|
+
if (field_name, command) in reported:
|
|
516
|
+
return
|
|
517
|
+
reported.add((field_name, command))
|
|
518
|
+
where = (file, key, field_name)
|
|
519
|
+
if command in tally:
|
|
520
|
+
tally[command][0] += 1
|
|
521
|
+
else:
|
|
522
|
+
tally[command] = [1, where]
|
|
523
|
+
|
|
524
|
+
def failed() -> None:
|
|
525
|
+
report(diagnostic(LATEX_CONVERSION_FAILED, file, key, field_name,
|
|
526
|
+
_UNREADABLE_LATEX))
|
|
527
|
+
return _FieldReport(on_unknown, failed)
|
|
528
|
+
return in_field
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def _name_part_groups(person: Person) -> List[List[str]]:
|
|
532
|
+
"""A pybtex name's parts, grouped as the BibTeX parts that read them.
|
|
533
|
+
|
|
534
|
+
Given and middle names are one group because they are read as one part,
|
|
535
|
+
and because BibTeX puts the first word of a given name in one list and the
|
|
536
|
+
rest in the other — which is a split a marker can land across.
|
|
537
|
+
"""
|
|
538
|
+
return [list(person.first_names) + list(person.middle_names),
|
|
539
|
+
list(person.prelast_names),
|
|
540
|
+
list(person.last_names),
|
|
541
|
+
list(person.lineage_names)]
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _with_marker_joined(parts: List[str]) -> List[str]:
|
|
545
|
+
"""One group of name parts, with a marker split across two of them joined."""
|
|
546
|
+
joined: List[str] = []
|
|
547
|
+
for part in parts:
|
|
548
|
+
if (joined and _MARKER_ARGUMENT.fullmatch(part)
|
|
549
|
+
and _MARKER_COMMAND.search(joined[-1])):
|
|
550
|
+
joined[-1] += part
|
|
551
|
+
else:
|
|
552
|
+
joined.append(part)
|
|
553
|
+
return joined
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
def _name_parts(person: Person) -> List[str]:
|
|
557
|
+
"""Every part of a pybtex name, as written, with split markers joined."""
|
|
558
|
+
return [part for group in _name_part_groups(person)
|
|
559
|
+
for part in _with_marker_joined(group)]
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def _without_marker(part: str) -> str:
|
|
563
|
+
"""One name part with its equal-contribution markers taken off the end.
|
|
564
|
+
|
|
565
|
+
Stripping repeats, because a name written ``Brown$^{*}$*`` carries the
|
|
566
|
+
marker twice and taking one off would leave the other in the name.
|
|
567
|
+
"""
|
|
568
|
+
while True:
|
|
569
|
+
stripped = EQUAL_CONTRIBUTION.sub("", part, count=1)
|
|
570
|
+
if stripped == part:
|
|
571
|
+
return part
|
|
572
|
+
part = stripped
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
def marks_equal_contribution(person: Person) -> bool:
|
|
576
|
+
"""True when any part of this name carries an equal-contribution marker.
|
|
577
|
+
|
|
578
|
+
Given, family, von and suffix are all read: BibTeX splits the name before
|
|
579
|
+
sslabdata sees it, so which part the star landed on is the author's choice
|
|
580
|
+
of where to write it, not a different meaning.
|
|
581
|
+
"""
|
|
582
|
+
return any(_without_marker(part) != part for part in _name_parts(person))
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def _is_others(person: Person) -> bool:
|
|
586
|
+
"""``and others``: BibTeX's "et al.", not a person."""
|
|
587
|
+
return (not person.first_names and not person.middle_names
|
|
588
|
+
and not person.prelast_names and not person.lineage_names
|
|
589
|
+
and [name.lower() for name in person.last_names] == [OTHERS])
|
|
590
|
+
|
|
591
|
+
|
|
592
|
+
def _is_literal(person: Person) -> bool:
|
|
593
|
+
"""A corporate author: one brace-protected unit, with no given name."""
|
|
594
|
+
return (not person.first_names and not person.middle_names
|
|
595
|
+
and not person.prelast_names and not person.lineage_names
|
|
596
|
+
and len(person.last_names) == 1
|
|
597
|
+
and person.last_names[0].startswith("{")
|
|
598
|
+
and person.last_names[0].endswith("}"))
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def person_name_parts(person: Person, on_unknown) -> Dict[str, Optional[str]]:
|
|
602
|
+
"""One pybtex Person as the parts BibTeX split it into, converted to text.
|
|
603
|
+
|
|
604
|
+
``given``, ``von``, ``family`` and ``suffix`` are BibTeX's four parts; a
|
|
605
|
+
corporate name comes back as ``literal`` instead, with the other four
|
|
606
|
+
unset. An empty part is ``None`` rather than ``""``, so the output says
|
|
607
|
+
"this name has no such part" rather than "it is blank".
|
|
608
|
+
|
|
609
|
+
An equal-contribution marker is not part of the name and does not appear
|
|
610
|
+
in any part; ``marks_equal_contribution`` reports it separately.
|
|
611
|
+
"""
|
|
612
|
+
def text(parts) -> Optional[str]:
|
|
613
|
+
joined = " ".join(_convert(_without_marker(part), on_unknown)
|
|
614
|
+
for part in _with_marker_joined(parts)).strip()
|
|
615
|
+
return joined or None
|
|
616
|
+
|
|
617
|
+
if _is_literal(person):
|
|
618
|
+
return {"given": None, "von": None, "family": None, "suffix": None,
|
|
619
|
+
"literal": text(person.last_names)}
|
|
620
|
+
return {
|
|
621
|
+
"given": text(person.first_names + person.middle_names),
|
|
622
|
+
"von": text(person.prelast_names),
|
|
623
|
+
"family": text(person.last_names),
|
|
624
|
+
"suffix": text(person.lineage_names),
|
|
625
|
+
"literal": None,
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
def given_words(name: str) -> List[str]:
|
|
630
|
+
"""The words BibTeX reads as the given name of a plain string: its first
|
|
631
|
+
and middle names, as written. Empty when BibTeX cannot read it."""
|
|
632
|
+
try:
|
|
633
|
+
person = Person(name)
|
|
634
|
+
except PybtexError:
|
|
635
|
+
# pybtex raises for three or more commas, and for a word nested
|
|
636
|
+
# more than 100 braces deep. `declared_form()` passes no comma, so
|
|
637
|
+
# from there only the nesting reaches this.
|
|
638
|
+
return []
|
|
639
|
+
return list(person.first_names) + list(person.middle_names)
|
|
640
|
+
|
|
641
|
+
|
|
642
|
+
def readable_name(parts: Dict[str, Optional[str]]) -> str:
|
|
643
|
+
"""The parts of a name joined in reading order: ``John van Last Jr.``
|
|
644
|
+
(SPEC.md §5).
|
|
645
|
+
|
|
646
|
+
The resolver matches on the structured parts rather than on this string,
|
|
647
|
+
so matching can change without changing what the document displays.
|
|
648
|
+
"""
|
|
649
|
+
if parts["literal"]:
|
|
650
|
+
return parts["literal"]
|
|
651
|
+
ordered = (parts["given"], parts["von"], parts["family"], parts["suffix"])
|
|
652
|
+
return " ".join(part for part in ordered if part)
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def _contributors(entry: Entry, role: str, on_unknown) -> List[Dict]:
|
|
656
|
+
"""The entry's names for one role, in source order, as parts plus position
|
|
657
|
+
(SPEC.md §3).
|
|
658
|
+
|
|
659
|
+
A name that reads as empty is dropped, as ``and others`` is, so
|
|
660
|
+
``position`` counts the names that reach the document and nothing else.
|
|
661
|
+
"""
|
|
662
|
+
found: List[Dict] = []
|
|
663
|
+
for person in entry.persons.get(role, []):
|
|
664
|
+
if _is_others(person):
|
|
665
|
+
continue
|
|
666
|
+
parts = person_name_parts(person, on_unknown)
|
|
667
|
+
name = readable_name(parts)
|
|
668
|
+
if name:
|
|
669
|
+
found.append({"name": name, "position": len(found) + 1,
|
|
670
|
+
"parts": parts, "person": person})
|
|
671
|
+
return found
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def check_others(entry: Entry, bib_id: str, source: str, report) -> None:
|
|
675
|
+
"""Report each author or editor list with ``and others`` before its end."""
|
|
676
|
+
for role in ("author", "editor"):
|
|
677
|
+
persons = entry.persons.get(role, [])
|
|
678
|
+
if any(_is_others(person) for person in persons[:-1]):
|
|
679
|
+
report(diagnostic(
|
|
680
|
+
OTHERS_NOT_LAST, source, bib_id, role,
|
|
681
|
+
f"'and others' is not the last name in the {role} list, where "
|
|
682
|
+
"it would mean 'et al.'; it is dropped, and the names around "
|
|
683
|
+
"it are kept"))
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
def parse_author_list(entry: Entry, on_unknown) -> List[Author]:
|
|
687
|
+
"""The entry's authors, in source order, with no contributor resolved yet;
|
|
688
|
+
matching them to a person is the resolver's."""
|
|
689
|
+
return [Author(name=found["name"],
|
|
690
|
+
position=found["position"],
|
|
691
|
+
equal_contribution=marks_equal_contribution(found["person"]),
|
|
692
|
+
**found["parts"])
|
|
693
|
+
for found in _contributors(entry, "author", on_unknown)]
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
def parse_editor_list(entry: Entry, on_unknown) -> List[Contributor]:
|
|
697
|
+
"""The entry's editors, read by the same machinery as its authors.
|
|
698
|
+
Editing a volume is not an authorship (SPEC.md §5)."""
|
|
699
|
+
return [Contributor(name=found["name"], position=found["position"],
|
|
700
|
+
**found["parts"])
|
|
701
|
+
for found in _contributors(entry, "editor", on_unknown)]
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
def entry_fields(bib_id: str, entry: Entry, unknown_in) -> Dict[str, str]:
|
|
705
|
+
"""The entry's fields, with prose converted from LaTeX.
|
|
706
|
+
|
|
707
|
+
``ENTRYTYPE`` and ``ID`` are included so the rules below read one plain
|
|
708
|
+
dictionary and know nothing about pybtex. No field is filled in from any
|
|
709
|
+
other entry: ``crossref`` is rejected rather than resolved.
|
|
710
|
+
"""
|
|
711
|
+
fields = {name.lower(): value for name, value in entry.fields.items()}
|
|
712
|
+
read = {
|
|
713
|
+
name: _convert(value, unknown_in(name)) if name in TEXT_FIELDS else value
|
|
714
|
+
for name, value in fields.items()
|
|
715
|
+
}
|
|
716
|
+
read["ENTRYTYPE"] = entry.type.lower()
|
|
717
|
+
read["ID"] = bib_id
|
|
718
|
+
return read
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
class _VerbatimWriter(BibTeXWriter):
|
|
722
|
+
"""pybtex's BibTeX writer, writing each value exactly as it was read.
|
|
723
|
+
|
|
724
|
+
The library's writer encodes every value as LaTeX, which escapes `%`, `&`,
|
|
725
|
+
`_` and `#` whether or not they already were: `20\\%` came back as
|
|
726
|
+
`20\\\\%`, a line break and a comment. A value read from a `.bib` file is
|
|
727
|
+
BibTeX already, so it is written as it stands. The braces are still
|
|
728
|
+
checked, so a value that cannot be written back is still reported.
|
|
729
|
+
"""
|
|
730
|
+
|
|
731
|
+
def _encode(self, text):
|
|
732
|
+
return text
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
def format_bibtex(bib_id: str, entry: Entry, source: str,
|
|
736
|
+
report) -> Optional[str]:
|
|
737
|
+
"""The entry written back out as BibTeX, for readers to copy (SPEC.md §5).
|
|
738
|
+
|
|
739
|
+
Nothing in sslabdata reads a value back out of it.
|
|
740
|
+
"""
|
|
741
|
+
try:
|
|
742
|
+
return _VerbatimWriter().to_string(
|
|
743
|
+
BibliographyData(entries={bib_id: entry})).strip()
|
|
744
|
+
except Exception: # noqa: BLE001 - a copyable string is not worth an entry
|
|
745
|
+
report(diagnostic(
|
|
746
|
+
WRITE_BACK_FAILED, source, bib_id, "bibtex",
|
|
747
|
+
"could not write this entry back out as BibTeX; bibtex is null"))
|
|
748
|
+
return None
|
|
749
|
+
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
# The fields that name a work's container, in order of precedence
|
|
753
|
+
# (SPEC.md §5).
|
|
754
|
+
CONTAINER_FIELDS = ("journal", "booktitle", "school", "institution")
|
|
755
|
+
|
|
756
|
+
# The venue kind each container field implies. `booktitle` depends on the
|
|
757
|
+
# entry type, because a proceedings volume and a collection are different
|
|
758
|
+
# kinds of container under one field name.
|
|
759
|
+
CONTAINER_KINDS = {"journal": "journal", "school": "institution",
|
|
760
|
+
"institution": "institution"}
|
|
761
|
+
BOOKTITLE_KINDS = {"inproceedings": "conference", "conference": "conference",
|
|
762
|
+
"proceedings": "conference", "incollection": "book",
|
|
763
|
+
"inbook": "book", "book": "book"}
|
|
764
|
+
OTHER_KIND = "other"
|
|
765
|
+
|
|
766
|
+
# A preprint's venue is the repository `archivePrefix` or `eprinttype` names.
|
|
767
|
+
# arXiv is the default, because a bare `eprint` is read as an arXiv
|
|
768
|
+
# identifier (`build_identifiers`) and linked as one.
|
|
769
|
+
ARXIV = "arXiv"
|
|
770
|
+
REPOSITORY_KIND = "repository"
|
|
771
|
+
|
|
772
|
+
# The bibliographic fields carried flat on the work, under BibTeX's own names
|
|
773
|
+
# and with BibTeX's own meanings. `number` in particular is an issue number
|
|
774
|
+
# for an @article and a report number for a @techreport; reinterpreting it is
|
|
775
|
+
# not sslabdata's job, and `venue.kind` gives a consumer the branch it needs.
|
|
776
|
+
FLAT_FIELDS = ("volume", "number", "pages", "series", "edition", "publisher",
|
|
777
|
+
"address", "organization", "chapter", "month", "howpublished",
|
|
778
|
+
"type")
|
|
779
|
+
|
|
780
|
+
# The identifier schemes sslabdata reads out of an entry, and the field each
|
|
781
|
+
# comes from.
|
|
782
|
+
IDENTIFIER_FIELDS = {"doi": "doi", "isbn": "isbn", "issn": "issn"}
|
|
783
|
+
|
|
784
|
+
# A DOI written as a URL is the resolver plus the DOI; the identifier is the
|
|
785
|
+
# part after it. Stripping exactly these prefixes is not a guess -- they are
|
|
786
|
+
# the registered resolvers -- and it is what makes `identifiers.doi` usable
|
|
787
|
+
# as an identifier rather than as a second copy of the link.
|
|
788
|
+
DOI_RESOLVERS = ("https://doi.org/", "http://doi.org/",
|
|
789
|
+
"https://dx.doi.org/", "http://dx.doi.org/")
|
|
790
|
+
DOI_BASE = "https://doi.org/"
|
|
791
|
+
ARXIV_BASE = "https://arxiv.org/abs/"
|
|
792
|
+
|
|
793
|
+
# Link origins and hosts (SPEC.md §5).
|
|
794
|
+
FROM_INPUT = "input"
|
|
795
|
+
DERIVED = "derived"
|
|
796
|
+
VIDEO_HOSTS = ("youtube.com", "youtu.be", "vimeo.com")
|
|
797
|
+
|
|
798
|
+
UNCHECKED, VERIFIED, MISSING = "unchecked", "verified", "missing"
|
|
799
|
+
|
|
800
|
+
|
|
801
|
+
def build_venue(entry: dict) -> Optional[Venue]:
|
|
802
|
+
"""The container this work appeared in, or None when the entry names none
|
|
803
|
+
(SPEC.md §5)."""
|
|
804
|
+
entry_type = entry.get("ENTRYTYPE", "")
|
|
805
|
+
for field_name in CONTAINER_FIELDS:
|
|
806
|
+
value = (entry.get(field_name) or "").strip()
|
|
807
|
+
if not value:
|
|
808
|
+
continue
|
|
809
|
+
if field_name == "booktitle":
|
|
810
|
+
kind = BOOKTITLE_KINDS.get(entry_type, OTHER_KIND)
|
|
811
|
+
else:
|
|
812
|
+
kind = CONTAINER_KINDS[field_name]
|
|
813
|
+
return Venue(kind=kind, name=value)
|
|
814
|
+
|
|
815
|
+
eprint = (entry.get("eprint") or "").strip()
|
|
816
|
+
if eprint:
|
|
817
|
+
return Venue(kind=REPOSITORY_KIND, name=_archive_prefix(entry))
|
|
818
|
+
return None
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
def _archive_prefix(entry: dict) -> str:
|
|
822
|
+
"""The repository an `eprint` belongs to, as the entry names it.
|
|
823
|
+
|
|
824
|
+
biblatex names it in `eprinttype`, of which `archivePrefix` is an alias;
|
|
825
|
+
an entry carrying both is read from `archivePrefix`.
|
|
826
|
+
"""
|
|
827
|
+
for field_name in ("archiveprefix", "eprinttype"):
|
|
828
|
+
prefix = (entry.get(field_name) or "").strip()
|
|
829
|
+
if prefix:
|
|
830
|
+
return prefix
|
|
831
|
+
return ARXIV
|
|
832
|
+
|
|
833
|
+
|
|
834
|
+
def bare_doi(doi: str) -> str:
|
|
835
|
+
"""One DOI with its resolver prefix taken off, if it was written as a URL."""
|
|
836
|
+
doi = doi.strip()
|
|
837
|
+
for resolver in DOI_RESOLVERS:
|
|
838
|
+
if doi.lower().startswith(resolver):
|
|
839
|
+
return doi[len(resolver):]
|
|
840
|
+
return doi
|
|
841
|
+
|
|
842
|
+
|
|
843
|
+
def build_identifiers(entry: dict, source: str, report) -> Dict[str, List[str]]:
|
|
844
|
+
"""The entry's identifiers, as a map from scheme to a list of identifiers.
|
|
845
|
+
|
|
846
|
+
A list, because ISBN and ISSN genuinely repeat -- a print and an
|
|
847
|
+
electronic one -- even though a BibTeX field holds one value.
|
|
848
|
+
"""
|
|
849
|
+
identifiers: Dict[str, List[str]] = {}
|
|
850
|
+
for scheme, field_name in IDENTIFIER_FIELDS.items():
|
|
851
|
+
value = (entry.get(field_name) or "").strip()
|
|
852
|
+
if not value:
|
|
853
|
+
continue
|
|
854
|
+
if scheme == "doi":
|
|
855
|
+
if not bare_doi(value):
|
|
856
|
+
report(diagnostic(
|
|
857
|
+
DOI_INVALID, source, entry.get("ID"), field_name,
|
|
858
|
+
f"'{value}' is a DOI resolver with no DOI after it; the "
|
|
859
|
+
"work gets no DOI identifier and no DOI link"))
|
|
860
|
+
continue
|
|
861
|
+
value = bare_doi(value)
|
|
862
|
+
identifiers[scheme] = [value]
|
|
863
|
+
|
|
864
|
+
eprint = (entry.get("eprint") or "").strip()
|
|
865
|
+
if eprint:
|
|
866
|
+
identifiers[_archive_prefix(entry).lower()] = [eprint]
|
|
867
|
+
return identifiers
|
|
868
|
+
|
|
869
|
+
|
|
870
|
+
def is_video_url(url: str) -> bool:
|
|
871
|
+
"""True when a URL's host is a video host, or a subdomain of one.
|
|
872
|
+
|
|
873
|
+
Only the parsed hostname counts: a lookalike domain, or a host named in
|
|
874
|
+
the path, query or fragment, does not. A URL that cannot be parsed, or has
|
|
875
|
+
no host, is not a video.
|
|
876
|
+
"""
|
|
877
|
+
try:
|
|
878
|
+
hostname = urlsplit(url).hostname
|
|
879
|
+
except ValueError:
|
|
880
|
+
return False
|
|
881
|
+
return hostname is not None and any(
|
|
882
|
+
hostname == host or hostname.endswith("." + host)
|
|
883
|
+
for host in VIDEO_HOSTS)
|
|
884
|
+
|
|
885
|
+
|
|
886
|
+
def pdf_link(bib_id: str, pdf_base_url: Optional[str]) -> Optional[Link]:
|
|
887
|
+
"""The PDF this work would be at under ``pdf_base_url``, checked only if
|
|
888
|
+
local: a build never fetches (SPEC.md §5)."""
|
|
889
|
+
if not pdf_base_url:
|
|
890
|
+
return None
|
|
891
|
+
base = pdf_base_url.rstrip('/')
|
|
892
|
+
url = f"{base}/{bib_id}.pdf"
|
|
893
|
+
if pdf_base_url.startswith(('http://', 'https://')):
|
|
894
|
+
return Link(url=url, origin=DERIVED, status=UNCHECKED)
|
|
895
|
+
return Link(url=url, origin=DERIVED,
|
|
896
|
+
status=VERIFIED if Path(url).exists() else MISSING)
|
|
897
|
+
|
|
898
|
+
|
|
899
|
+
def build_links(entry: dict, bib_id: str, identifiers: Dict[str, List[str]],
|
|
900
|
+
pdf_base_url: Optional[str]) -> Dict[str, List[Link]]:
|
|
901
|
+
"""Every URL this work can be reached at, filed by kind (SPEC.md §5)."""
|
|
902
|
+
links: Dict[str, List[Link]] = {}
|
|
903
|
+
|
|
904
|
+
def add(kind: str, link: Optional[Link]) -> None:
|
|
905
|
+
if link is not None:
|
|
906
|
+
links.setdefault(kind, []).append(link)
|
|
907
|
+
|
|
908
|
+
url = (entry.get("url") or "").strip()
|
|
909
|
+
if url:
|
|
910
|
+
add("video" if is_video_url(url) else "url",
|
|
911
|
+
Link(url=url, origin=FROM_INPUT, status=UNCHECKED))
|
|
912
|
+
# `video` is always a video, whatever its host, so an entry can name a
|
|
913
|
+
# project website in `url` and its video here.
|
|
914
|
+
video = (entry.get("video") or "").strip()
|
|
915
|
+
if video:
|
|
916
|
+
add("video", Link(url=video, origin=FROM_INPUT, status=UNCHECKED))
|
|
917
|
+
# A `pdf` the entry names is the work's PDF, so it replaces the one
|
|
918
|
+
# guessed from `pdf_base_url` rather than sitting beside it.
|
|
919
|
+
pdf = (entry.get("pdf") or "").strip()
|
|
920
|
+
if pdf:
|
|
921
|
+
add("pdf", Link(url=pdf, origin=FROM_INPUT, status=UNCHECKED))
|
|
922
|
+
else:
|
|
923
|
+
add("pdf", pdf_link(bib_id, pdf_base_url))
|
|
924
|
+
for doi in identifiers.get("doi", []):
|
|
925
|
+
add("doi", Link(url=DOI_BASE + doi, origin=DERIVED, status=UNCHECKED))
|
|
926
|
+
for eprint in identifiers.get(ARXIV.lower(), []):
|
|
927
|
+
add("arxiv", Link(url=ARXIV_BASE + eprint, origin=DERIVED,
|
|
928
|
+
status=UNCHECKED))
|
|
929
|
+
return links
|
|
930
|
+
|
|
931
|
+
|
|
932
|
+
def extract_note(entry: dict) -> Optional[str]:
|
|
933
|
+
"""Extract and format the note field."""
|
|
934
|
+
note = entry.get("note", "").strip().rstrip('. ')
|
|
935
|
+
if not note:
|
|
936
|
+
return None
|
|
937
|
+
return note
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
def parse_project_ids(entry: dict) -> List[str]:
|
|
941
|
+
"""Parse the project field from a BibTeX entry."""
|
|
942
|
+
project_field = entry.get("project", "").strip()
|
|
943
|
+
if not project_field:
|
|
944
|
+
return []
|
|
945
|
+
project_field = project_field.strip('{}')
|
|
946
|
+
return [p.strip() for p in project_field.split(',') if p.strip()]
|
|
947
|
+
|
|
948
|
+
|
|
949
|
+
def entry_year(entry: dict, source: str, report) -> Optional[int]:
|
|
950
|
+
"""The entry's year, or None with a diagnostic when it has none or it is
|
|
951
|
+
not a number.
|
|
952
|
+
|
|
953
|
+
Only an unsigned run of ASCII digits is a number here: `int()` would also
|
|
954
|
+
read `-5`, `+2020`, `2_020` and full-width `2020`.
|
|
955
|
+
"""
|
|
956
|
+
raw = str(entry.get("year", "")).strip()
|
|
957
|
+
if not raw:
|
|
958
|
+
report(diagnostic(YEAR_MISSING, source, entry.get("ID"), "year",
|
|
959
|
+
"entry has no year"))
|
|
960
|
+
return None
|
|
961
|
+
if YEAR_DIGITS.fullmatch(raw):
|
|
962
|
+
return int(raw)
|
|
963
|
+
report(diagnostic(YEAR_INVALID, source, entry.get("ID"), "year",
|
|
964
|
+
f"'{raw}' is not a year, which is written in the "
|
|
965
|
+
"digits 0-9 alone; the work is emitted with year: null "
|
|
966
|
+
"and sorts last"))
|
|
967
|
+
return None
|
|
968
|
+
|
|
969
|
+
|
|
970
|
+
# The container field checked for an entry type, and the entry types
|
|
971
|
+
# sslabdata documents (SPEC.md "Diagnostic codes").
|
|
972
|
+
REQUIRED_CONTAINER = {"article": "journal", "inproceedings": "booktitle"}
|
|
973
|
+
SUPPORTED_TYPES = frozenset({
|
|
974
|
+
"article", "inproceedings", "conference", "proceedings", "incollection",
|
|
975
|
+
"inbook", "book", "phdthesis", "mastersthesis", "techreport", "manual",
|
|
976
|
+
"misc"})
|
|
977
|
+
|
|
978
|
+
|
|
979
|
+
def check_entry_type(entry: dict, source: str, report) -> None:
|
|
980
|
+
"""Report an entry type sslabdata does not document, or a missing container."""
|
|
981
|
+
entry_type, key = entry["ENTRYTYPE"], entry["ID"]
|
|
982
|
+
if entry_type not in SUPPORTED_TYPES:
|
|
983
|
+
report(diagnostic(
|
|
984
|
+
ENTRY_TYPE_UNSUPPORTED, source, key, "entry_type",
|
|
985
|
+
f"@{entry_type} is not an entry type sslabdata documents; the "
|
|
986
|
+
"entry is kept, with its venue read from whichever container "
|
|
987
|
+
"field it carries"))
|
|
988
|
+
return
|
|
989
|
+
required = REQUIRED_CONTAINER.get(entry_type)
|
|
990
|
+
if required and not (entry.get(required) or "").strip():
|
|
991
|
+
report(diagnostic(
|
|
992
|
+
VENUE_MISSING, source, key, required,
|
|
993
|
+
f"@{entry_type} has no {required}; the entry is kept, and its "
|
|
994
|
+
"venue is read from any other container field it carries, or "
|
|
995
|
+
"is null"))
|
|
996
|
+
|
|
997
|
+
|
|
998
|
+
def entry_to_work(
|
|
999
|
+
bib_id: str,
|
|
1000
|
+
entry: Entry,
|
|
1001
|
+
category: str,
|
|
1002
|
+
pdf_base_url: Optional[str],
|
|
1003
|
+
source: str,
|
|
1004
|
+
source_file: str,
|
|
1005
|
+
report,
|
|
1006
|
+
unknown_commands_seen: Dict[str, list],
|
|
1007
|
+
) -> Work:
|
|
1008
|
+
"""Convert one pybtex Entry to a Work dataclass.
|
|
1009
|
+
|
|
1010
|
+
Unknown LaTeX commands are added to ``unknown_commands_seen``
|
|
1011
|
+
(`_unknown_command_reporter`); every other diagnostic goes to ``report``.
|
|
1012
|
+
"""
|
|
1013
|
+
unknown_in = _unknown_command_reporter(report, source, bib_id,
|
|
1014
|
+
unknown_commands_seen)
|
|
1015
|
+
fields = entry_fields(bib_id, entry, unknown_in)
|
|
1016
|
+
identifiers = build_identifiers(fields, source, report)
|
|
1017
|
+
check_entry_type(fields, source, report)
|
|
1018
|
+
check_others(entry, bib_id, source, report)
|
|
1019
|
+
|
|
1020
|
+
return Work(
|
|
1021
|
+
bib_id=bib_id,
|
|
1022
|
+
title=fields.get("title", ""),
|
|
1023
|
+
authors=parse_author_list(entry, unknown_in("author")),
|
|
1024
|
+
editors=parse_editor_list(entry, unknown_in("editor")),
|
|
1025
|
+
year=entry_year(fields, source, report),
|
|
1026
|
+
category=category,
|
|
1027
|
+
entry_type=fields["ENTRYTYPE"],
|
|
1028
|
+
source_file=source_file,
|
|
1029
|
+
venue=build_venue(fields),
|
|
1030
|
+
abstract=fields.get("abstract"),
|
|
1031
|
+
note=extract_note(fields),
|
|
1032
|
+
identifiers=identifiers,
|
|
1033
|
+
links=build_links(fields, bib_id, identifiers, pdf_base_url),
|
|
1034
|
+
project_ids=parse_project_ids(fields),
|
|
1035
|
+
bibtex=format_bibtex(bib_id, entry, source, report),
|
|
1036
|
+
# Empty, as its default is; named so that the mapping below is
|
|
1037
|
+
# type-checked against the flat fields alone.
|
|
1038
|
+
derived={},
|
|
1039
|
+
**{name: fields.get(name) for name in FLAT_FIELDS},
|
|
1040
|
+
)
|
|
1041
|
+
|
|
1042
|
+
|
|
1043
|
+
def _encoding_error(path: str, error: UnicodeDecodeError) -> Diagnostic:
|
|
1044
|
+
"""The one diagnostic for a `.bib` file that is not UTF-8."""
|
|
1045
|
+
line = error.object[:error.start].count(b"\n") + 1
|
|
1046
|
+
return diagnostic(ENCODING_INVALID, path, None, None,
|
|
1047
|
+
f"the file is not UTF-8: byte "
|
|
1048
|
+
f"0x{error.object[error.start]:02x} on line {line} "
|
|
1049
|
+
"cannot be read. Save the file as UTF-8.")
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
def _crossref_error(path: str, bib_id: str, parent: str) -> Diagnostic:
|
|
1053
|
+
"""The one diagnostic for an entry that carries a ``crossref`` field.
|
|
1054
|
+
|
|
1055
|
+
A field that is there but empty is reported as what it is rather than as
|
|
1056
|
+
a parent whose name happens to be blank.
|
|
1057
|
+
"""
|
|
1058
|
+
names = f"names the parent '{parent}'" if parent else "names no parent"
|
|
1059
|
+
return diagnostic(CROSSREF_UNSUPPORTED, path, bib_id, "crossref",
|
|
1060
|
+
f"crossref is not supported; this entry {names}. "
|
|
1061
|
+
"Write the fields out on the entry itself.")
|
|
1062
|
+
|
|
1063
|
+
|
|
1064
|
+
def parse_all_works(
|
|
1065
|
+
bib_dir: str,
|
|
1066
|
+
bib_files: list,
|
|
1067
|
+
diagnostics: List[Diagnostic],
|
|
1068
|
+
pdf_base_url: Optional[str] = None,
|
|
1069
|
+
) -> List[Work]:
|
|
1070
|
+
"""Parse all configured BibTeX files and return a flat list of Works.
|
|
1071
|
+
|
|
1072
|
+
Every file is read first, so a citation key repeated across two of them is
|
|
1073
|
+
reported against both.
|
|
1074
|
+
|
|
1075
|
+
Args:
|
|
1076
|
+
bib_dir: Directory containing the BibTeX files
|
|
1077
|
+
bib_files: List of dicts with 'name' and 'category' keys
|
|
1078
|
+
diagnostics: The list that receives every coded diagnostic, in the
|
|
1079
|
+
order found
|
|
1080
|
+
pdf_base_url: Base URL/path for PDFs
|
|
1081
|
+
|
|
1082
|
+
Returns:
|
|
1083
|
+
List of Work objects, in the order SPEC.md §3 gives.
|
|
1084
|
+
"""
|
|
1085
|
+
read: List[Tuple[str, str, str, Entry, str]] = []
|
|
1086
|
+
first_source: Dict[str, Tuple[str, str]] = {}
|
|
1087
|
+
|
|
1088
|
+
report = diagnostics.append
|
|
1089
|
+
|
|
1090
|
+
# Every file's redefined macros, summarised once when all are read.
|
|
1091
|
+
redefinitions: List[Tuple[str, str, int]] = []
|
|
1092
|
+
for bib_file in bib_files:
|
|
1093
|
+
name = bib_file['name'] if isinstance(bib_file, dict) else bib_file.name
|
|
1094
|
+
category = bib_file['category'] if isinstance(bib_file, dict) else bib_file.category
|
|
1095
|
+
path = f"{bib_dir}/{name}"
|
|
1096
|
+
try:
|
|
1097
|
+
parsed = parse_bibtex_file(path, diagnostics, redefinitions)
|
|
1098
|
+
except UnicodeDecodeError as error:
|
|
1099
|
+
report(_encoding_error(path, error))
|
|
1100
|
+
continue
|
|
1101
|
+
for bib_id, entry in parsed.items():
|
|
1102
|
+
normalized = bib_id.lower()
|
|
1103
|
+
if normalized in first_source:
|
|
1104
|
+
previous_path, previous_key = first_source[normalized]
|
|
1105
|
+
report(_duplicate_key_error(path, bib_id, previous_path, previous_key))
|
|
1106
|
+
else:
|
|
1107
|
+
first_source[normalized] = (path, bib_id)
|
|
1108
|
+
# Rejected on presence, not on value (SPEC.md "Diagnostic codes").
|
|
1109
|
+
crossref = [value for field_name, value in entry.fields.items()
|
|
1110
|
+
if field_name.lower() == "crossref"]
|
|
1111
|
+
if crossref:
|
|
1112
|
+
report(_crossref_error(path, bib_id, str(crossref[0]).strip()))
|
|
1113
|
+
continue
|
|
1114
|
+
read.append((path, name, bib_id, entry, category))
|
|
1115
|
+
|
|
1116
|
+
summary = redefined_summary(redefinitions)
|
|
1117
|
+
if summary:
|
|
1118
|
+
report(summary)
|
|
1119
|
+
|
|
1120
|
+
# One line per unknown command for the whole run: a real bibliography
|
|
1121
|
+
# can use one command in thousands of fields, and a line for each would
|
|
1122
|
+
# bury every other diagnostic.
|
|
1123
|
+
unknown: Dict[str, list] = {}
|
|
1124
|
+
works = [
|
|
1125
|
+
entry_to_work(bib_id, entry, category, pdf_base_url, path, name, report,
|
|
1126
|
+
unknown)
|
|
1127
|
+
for path, name, bib_id, entry, category in read
|
|
1128
|
+
]
|
|
1129
|
+
for command, (fields, where) in unknown.items():
|
|
1130
|
+
report(unknown_command_diagnostic(command, where, fields))
|
|
1131
|
+
works.sort(key=lambda w: (w.year is not None, w.year or 0), reverse=True)
|
|
1132
|
+
return works
|