sslabdata 3.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1132 @@
1
+ """
2
+ BibTeX parsing pipeline.
3
+
4
+ pybtex reads the files — @string macros, BibTeX's own name splitting, entry
5
+ order — and this module maps its Entry and Person objects onto sslabdata's
6
+ Work model: per-field LaTeX conversion (latex.py), the structured venue,
7
+ identifiers, links and diagnostics.
8
+
9
+ Together with latex.py this is the adapter: no other module imports pybtex or
10
+ pylatexenc, and nothing here lets a library object or a library message reach
11
+ the rest of sslabdata.
12
+
13
+ Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
14
+ Author: Siddhartha Srinivasa
15
+ MIT License - see LICENSE file for details.
16
+ """
17
+
18
+ import re
19
+ from pathlib import Path
20
+ from typing import Callable, Dict, List, Optional, Tuple
21
+ from urllib.parse import urlsplit
22
+
23
+ import pybtex.errors
24
+ from pybtex.database import BibliographyData, Entry, Person
25
+ from pybtex.exceptions import PybtexError
26
+ from pybtex.database.output.bibtex import Writer as BibTeXWriter
27
+ from pybtex.database.input.bibtex import (
28
+ LowLevelParser, Parser as PybtexParser, SkipEntry, UndefinedMacro,
29
+ )
30
+ from pybtex.scanner import PybtexSyntaxError
31
+
32
+ from .latex import latex_to_text, strip_braces, unknown_commands
33
+ from ..config import (
34
+ CONTROL_CHARACTER, control_message, nfc, without_control_characters,
35
+ )
36
+ from ..diagnostics import Diagnostic, diagnostic
37
+ from ..models import Author, Contributor, Link, Venue, Work
38
+
39
+
40
+ # The fields converted from LaTeX to plain text (SPEC.md §2). The repository
41
+ # fields are among them because braces around `{arXiv}` are grouping, not part
42
+ # of the name.
43
+ TEXT_FIELDS = frozenset({
44
+ "title", "abstract", "note", "journal", "booktitle", "school",
45
+ "institution", "type", "series", "publisher", "address", "organization",
46
+ "archiveprefix", "eprinttype",
47
+ })
48
+
49
+ # A name list ending in "and others" means "et al."; it is not an author.
50
+ OTHERS = "others"
51
+
52
+ # The codes this module reports (SPEC.md "Diagnostic codes").
53
+ OTHERS_NOT_LAST = "BIB-OTHERS-NOT-LAST"
54
+ DUPLICATE_CITATION_KEY = "BIB-DUPLICATE-KEY"
55
+ CROSSREF_UNSUPPORTED = "BIB-CROSSREF-UNSUPPORTED"
56
+ YEAR_MISSING = "BIB-YEAR-MISSING"
57
+ YEAR_INVALID = "BIB-YEAR-INVALID"
58
+ YEAR_DIGITS = re.compile(r"[0-9]+")
59
+ DOI_INVALID = "BIB-DOI-INVALID"
60
+ STRING_UNDEFINED = "BIB-STRING-UNDEFINED"
61
+ SYNTAX_ERROR = "BIB-SYNTAX-ERROR"
62
+ VENUE_MISSING = "BIB-VENUE-MISSING"
63
+ ENTRY_TYPE_UNSUPPORTED = "BIB-ENTRY-TYPE-UNSUPPORTED"
64
+ STRING_REDEFINED = "BIB-STRING-REDEFINED"
65
+ PARSER_MESSAGE = "BIB-PARSER-MESSAGE"
66
+ LATEX_CONVERSION_FAILED = "LATEX-CONVERSION-FAILED"
67
+ WRITE_BACK_FAILED = "BIB-WRITE-BACK-FAILED"
68
+ ENCODING_INVALID = "BIB-ENCODING-INVALID"
69
+ LATEX_COMMAND_UNKNOWN = "LATEX-COMMAND-UNKNOWN"
70
+
71
+ # Equal contribution is written as a star on one part of a name, in one of
72
+ # these four forms. It is an annotation rather than part of the name, so it is
73
+ # taken off the part before the name is read and recorded on the author
74
+ # instead. Other author annotations — corresponding author, affiliation
75
+ # numbers, daggers — are not read.
76
+ #
77
+ # A star, caret or dollar written with a backslash in front of it is escaped
78
+ # text rather than the start of a marker: `Brown\*` is not marked, and the
79
+ # ordinary LaTeX conversion then consumes the escaped star. The accent in
80
+ # `C{\^o}t{\'e}$^{*}$` is escaped the same way, and the marker after it is
81
+ # not, which is why that name reads `Côté` and is marked.
82
+ _WRITTEN = r"\$\^\{\*\}\$|\^\{\*\}|\\textsuperscript\s*\{\*\}"
83
+ _MARKER = rf"(?<!\\)(?:{_WRITTEN}|\*)"
84
+
85
+ # A marker at the end of a name part, on its own or in a brace group of its
86
+ # own: BibTeX grouping such as `Brown{$^{*}$}` protects the marker from the
87
+ # name, and does not make it part of it. Only a form that brings its own
88
+ # command is unwrapped, because a lone `{*}` is how any other command takes
89
+ # its argument — the star in `Brown\^{*}` is an accented star, not a marker.
90
+ _ANY_MARKER = rf"(?:(?<!\\)\{{\s*(?:{_WRITTEN})\s*\}}|{_MARKER})"
91
+ EQUAL_CONTRIBUTION = re.compile(rf"{_ANY_MARKER}\s*$")
92
+
93
+ # `\textsuperscript {*}`, with a space before the argument, is the same form:
94
+ # BibTeX splits a name on spaces, so the command and its argument arrive as two
95
+ # name parts and neither is a marker on its own. Only this one command takes
96
+ # its argument back, written as the command and not as an escaped backslash,
97
+ # and only from a part that is the argument and nothing but further markers —
98
+ # so the accent in `Brown\^ {*}`, a `{*}` after any other command, and a part
99
+ # carrying text of its own all keep their own boundary. What the part carries
100
+ # after `{*}` is left to the stripping below, as it is for an unspaced marker.
101
+ _MARKER_COMMAND = re.compile(r"(?<!\\)\\textsuperscript\s*$")
102
+ _MARKER_ARGUMENT = re.compile(rf"\{{\*\}}(?:{_ANY_MARKER})*")
103
+
104
+
105
+ # The command name pybtex is about to read, when that name is `comment`.
106
+ _COMMENT_COMMAND = re.compile(r'\s*comment\s*[{(]', re.IGNORECASE)
107
+
108
+ # The two shapes `BIB-BRACE-MISMATCH` looks for (SPEC.md "Diagnostic codes").
109
+ BRACE_MISMATCH = "BIB-BRACE-MISMATCH"
110
+ _FIELD_IN_VALUE = re.compile(r',\s*([A-Za-z][\w-]*)\s*=\s*[{"]')
111
+ _TEXT_AFTER_ENTRY = re.compile(r'[ \t]*([^\s%@][^\r\n]*)')
112
+ _FIELD_AFTER_ENTRY = re.compile(r'\s*,\s*([A-Za-z][\w-]*)\s*=')
113
+
114
+
115
+
116
+ def _redefined_macros(text: str,
117
+ definitions: List[Tuple[str, int]]) -> List[Tuple[str, int]]:
118
+ """Every redefinition of an ``@string`` macro, as ``(name, line)``, in
119
+ source order: each definition after a macro's first.
120
+
121
+ ``definitions`` are the ``(name, offset)`` of every ``@string`` the parser
122
+ actually read from ``text``, so a definition inside an ``@comment`` group
123
+ is not one. Names compare without case, as the parser's macros do.
124
+ """
125
+ seen: set = set()
126
+ repeated: List[Tuple[str, int]] = []
127
+ for name, offset in definitions:
128
+ name = name.lower()
129
+ if name in seen:
130
+ repeated.append((name, text.count("\n", 0, offset) + 1))
131
+ else:
132
+ seen.add(name)
133
+ return repeated
134
+
135
+
136
+ def redefined_summary(redefinitions: List[Tuple[str, str, int]]) -> Optional[Diagnostic]:
137
+ """One `STRING_REDEFINED` line for ``(file, name, line)`` redefinitions,
138
+ in the form SPEC.md §7 shows; None when there is nothing to say."""
139
+ if not redefinitions:
140
+ return None
141
+ names = sorted({name for _, name, _ in redefinitions})
142
+ files = sorted({path for path, _, _ in redefinitions})
143
+ where = ", ".join(f"{path}:{line}" for path, _, line in sorted(
144
+ redefinitions, key=lambda found: (found[0], found[2])))
145
+ count = f"{len(names)} @string macro{'s' if len(names) != 1 else ''}"
146
+ return diagnostic(
147
+ STRING_REDEFINED, files[0] if len(files) == 1 else None, None, None,
148
+ f"{count} redefined (last definition used): {', '.join(names)} "
149
+ f"[{where}]")
150
+
151
+
152
+ class _CommentSkippingParser(LowLevelParser):
153
+ """pybtex's tokenizer, with a balanced ``@comment{...}`` group stepped over.
154
+
155
+ pybtex raises ``SkipEntry`` for ``@comment`` before reading the body, so
156
+ the scanner resumes just inside the group and an entry written there is a
157
+ real entry to it — BibTeX behaves the same way. sslabdata treats a
158
+ commented-out entry as commented out, so the group is consumed here, at
159
+ the parser's own position and with the parser's own scanner. Nothing else
160
+ in the file is read by sslabdata, which is why the shape of a value or of a
161
+ neighbouring command cannot be got wrong.
162
+
163
+ Only a *balanced* group is consumed. Prose that merely mentions
164
+ ``@comment{`` does not close, so the position is put back and pybtex reads
165
+ the rest of the file itself: at worst a commented-out entry stays visible,
166
+ never a real entry disappears.
167
+ """
168
+
169
+ # The scanner's position, which pybtex's `Scanner` sets.
170
+ pos: int
171
+ lineno: int
172
+
173
+ def __init__(self, *args, **kwargs):
174
+ super().__init__(*args, **kwargs)
175
+ # (macro name, offset of its `@`) for every `@string` read in full.
176
+ self.definitions: List[Tuple[str, int]] = []
177
+
178
+ # The ingress for a citation key: it is put in NFC as the tokenizer reads
179
+ # it, so the key an entry is stored, compared and located under is one
180
+ # string however its accents were written (SPEC.md §2).
181
+ @property
182
+ def current_entry_key(self) -> Optional[str]:
183
+ return self._entry_key
184
+
185
+ @current_entry_key.setter
186
+ def current_entry_key(self, key: Optional[str]) -> None:
187
+ self._entry_key = None if key is None else nfc(key)
188
+
189
+ def parse_string_body(self, body_end):
190
+ """Read one ``@string`` body, and remember the definition it made."""
191
+ super().parse_string_body(body_end)
192
+ self.definitions.append((self.current_field_name, self.command_start))
193
+
194
+ def parse_command(self):
195
+ # pybtex raises SkipEntry for a @comment, and also for an entry that
196
+ # `wanted_entries` filters out — which leaves the scanner somewhere
197
+ # quite different. Only the first is ours to recover from, so the
198
+ # command is identified before the parser reads it.
199
+ comment = _COMMENT_COMMAND.match(self.text, self.pos) is not None
200
+ try:
201
+ return super().parse_command()
202
+ except SkipEntry:
203
+ if comment:
204
+ position, lineno = self.pos, self.lineno
205
+ if not self._skip_comment_group():
206
+ self.pos, self.lineno = position, lineno
207
+ raise
208
+
209
+ def _skip_comment_group(self) -> bool:
210
+ """Consume the ``@comment`` body just opened. False if it never closes."""
211
+ closing = self.RBRACE if self.text[self.pos - 1] == "{" else self.RPAREN
212
+ while True:
213
+ token = self.skip_to([closing, self.LBRACE])
214
+ if token is None:
215
+ return False
216
+ if token.pattern is closing:
217
+ return True
218
+ try:
219
+ for _ in self.parse_string(self.RBRACE):
220
+ pass
221
+ except PybtexSyntaxError:
222
+ return False
223
+
224
+
225
+ class _Parser(PybtexParser):
226
+ """pybtex's BibTeX parser, reading ``@comment`` groups as comments.
227
+
228
+ ``Parser.parse_string`` names ``LowLevelParser`` directly, so swapping the
229
+ tokenizer means restating that loop. It and `_VerbatimWriter` are the two
230
+ places sslabdata touches a pybtex internal, which is why ``pybtex~=0.26``
231
+ is pinned.
232
+ """
233
+
234
+ def __init__(self, *args, duplicate_keys=None, **kwargs):
235
+ super().__init__(*args, **kwargs)
236
+ self.duplicate_keys = duplicate_keys if duplicate_keys is not None else []
237
+ # (error, entry key, field name, where the command began) for every
238
+ # syntax error, with the tokenizer's position in the file as it was
239
+ # when the error was raised: by the time the file is read, it has
240
+ # moved on.
241
+ self.syntax_errors: List[Tuple[PybtexSyntaxError, Optional[str],
242
+ Optional[str], Optional[int]]] = []
243
+ # The entry key each captured library message was raised while
244
+ # reading, by the message's identity.
245
+ self.message_keys: Dict[int, str] = {}
246
+ # (entry key, field name, characters) for every field value its
247
+ # control characters were removed from.
248
+ self.control_characters: List[Tuple[str, str, List[str]]] = []
249
+ # (entry key, field name, message) for every entry whose braces
250
+ # read it differently from how it was written (`BRACE_MISMATCH`).
251
+ self.brace_mismatches: List[Tuple[str, Optional[str], str]] = []
252
+
253
+ def handle_error(self, error):
254
+ """Keep a syntax error with where it happened; relay anything else."""
255
+ if isinstance(error, PybtexSyntaxError):
256
+ tokenizer = error.parser
257
+ self.syntax_errors.append((error, tokenizer.current_entry_key,
258
+ tokenizer.current_field_name,
259
+ tokenizer.command_start))
260
+ return
261
+ super().handle_error(error)
262
+
263
+ def process_entry(self, entry_type, key, fields):
264
+ """Remember duplicate keys before pybtex discards the later entry.
265
+
266
+ ``BibliographyData.add_entry`` reports a parser-library warning and
267
+ keeps the first entry. Recording the key here lets sslabdata surface a
268
+ stable, file-qualified diagnostic instead of exposing that wording.
269
+ """
270
+ if key is not None and key in self.data.entries:
271
+ self.duplicate_keys.append(key)
272
+ # The ingress for field text: control characters are removed here,
273
+ # and the value put in NFC, before pybtex splits a name list or
274
+ # normalises whitespace, and before any LaTeX is read
275
+ # (`CONTROL_CHARACTER`, SPEC.md §2).
276
+ fields = [(name, self._without_controls(key, name, parts))
277
+ for name, parts in fields]
278
+ captured = pybtex.errors.captured_errors
279
+ before = len(captured) if captured is not None else 0
280
+ super().process_entry(entry_type, key, fields)
281
+ for error in (captured or [])[before:]:
282
+ self.message_keys[id(error)] = key
283
+
284
+ def _without_controls(self, key, name: str, parts: List[str]) -> List[str]:
285
+ """One field's value, joined from its parts, with control characters
286
+ removed and in NFC, and the field recorded when there were any
287
+ control characters. Joined first, because a part can begin with a
288
+ mark that composes with the end of the one before."""
289
+ cleaned = [without_control_characters(part) for part in parts]
290
+ found = list(dict.fromkeys(c for _, chars in cleaned for c in chars))
291
+ if found:
292
+ self.control_characters.append((key, name.lower(), found))
293
+ return [nfc("".join(part for part, _ in cleaned))]
294
+
295
+ def parse_string(self, text: str):
296
+ self.unnamed_entry_counter = 1
297
+ self.command_start = 0
298
+ commands = _CommentSkippingParser(
299
+ text,
300
+ keyless_entries=self.keyless_entries,
301
+ handle_error=self.handle_error,
302
+ want_entry=self.data.want_entry,
303
+ filename=self.filename,
304
+ macros=self.macros,
305
+ )
306
+ self.string_definitions = commands.definitions
307
+ read = 0
308
+ for command, arguments in commands:
309
+ kind = command.lower()
310
+ if kind == "preamble":
311
+ self.process_preamble(*arguments)
312
+ elif kind != "string":
313
+ # An entry with a syntax error already says its values may
314
+ # hold text meant for later fields. An undefined macro is
315
+ # kept as a syntax error too, but says nothing of the kind.
316
+ key = arguments[0]
317
+ if all(error[1] != key or isinstance(error[0], UndefinedMacro)
318
+ for error in self.syntax_errors[read:]):
319
+ self._check_braces(commands, *arguments)
320
+ self.process_entry(command, *arguments)
321
+ read = len(self.syntax_errors)
322
+ return self.data
323
+
324
+ def _check_braces(self, commands, key, fields) -> None:
325
+ """Record the entry just read when its braces read it differently
326
+ from how it was written (`BRACE_MISMATCH`), at the first field whose
327
+ text did not arrive as written."""
328
+ for name, parts in fields:
329
+ merged = _FIELD_IN_VALUE.search("".join(parts))
330
+ if merged:
331
+ other = merged.group(1)
332
+ self.brace_mismatches.append((key, other.lower(), (
333
+ f"the field '{other}' is read into the value of '{name}', "
334
+ f"because a brace in '{name}' is not closed where it was "
335
+ "meant to be")))
336
+ return
337
+ after = _TEXT_AFTER_ENTRY.match(commands.text, commands.pos)
338
+ if after and fields:
339
+ name = fields[-1][0]
340
+ lost = _FIELD_AFTER_ENTRY.match(commands.text, commands.pos)
341
+ line = commands.text.count("\n", 0, commands.pos) + 1
342
+ self.brace_mismatches.append((
343
+ key, (lost.group(1) if lost else name).lower(),
344
+ f"a closing brace in the value of '{name}' ends the entry at "
345
+ f"line {line}, so " + (
346
+ f"the field '{lost.group(1)}' and any after it are not read"
347
+ if lost else
348
+ f"'{after.group(1).strip()}' after it is not read")))
349
+
350
+
351
+ def _duplicate_key_error(
352
+ path: str,
353
+ key: str,
354
+ first_path: Optional[str] = None,
355
+ first_key: Optional[str] = None,
356
+ ) -> Diagnostic:
357
+ """One stable duplicate-key diagnostic, with both locations when known."""
358
+ message = "duplicate citation key"
359
+ if first_path is not None:
360
+ first_location = f"{first_path}:{first_key or key}:citation_key"
361
+ message += f"; first defined in {first_location}"
362
+ return diagnostic(DUPLICATE_CITATION_KEY, path, key, "citation_key", message)
363
+
364
+
365
+ def _on_comment_line(text: str, position: Optional[int]) -> bool:
366
+ """True when ``position`` is on a line that starts with `%`.
367
+
368
+ The parser library reads an `@` anywhere outside an entry as the start of
369
+ a command, so prose on a `%` line that mentions `@article` fails to parse.
370
+ That failure is not reported: the prose was never meant as BibTeX. Only
371
+ the report is suppressed. A well-formed command on such a line is read,
372
+ as the library reads it.
373
+ """
374
+ if position is None:
375
+ return False
376
+ line_start = text.rfind("\n", 0, position) + 1
377
+ return text[line_start:position].lstrip().startswith("%")
378
+
379
+
380
+ def _syntax_diagnostic(path: str, error: PybtexSyntaxError,
381
+ key: Optional[str], field_name: Optional[str]) -> Diagnostic:
382
+ """One parser-library syntax error, in sslabdata's voice and located."""
383
+ if isinstance(error, UndefinedMacro):
384
+ macro = str(error).rsplit(": ", 1)[-1]
385
+ return diagnostic(
386
+ STRING_UNDEFINED, path, key, field_name if key else None,
387
+ f"the macro '{macro}' is not defined by any @string; it is read "
388
+ "as empty, and the entry is kept")
389
+ if key is None:
390
+ return diagnostic(
391
+ SYNTAX_ERROR, path, None, None,
392
+ f"the text at line {error.lineno} does not read as BibTeX and is "
393
+ "skipped")
394
+ after = f", after the value of '{field_name}'" if field_name else ""
395
+ return diagnostic(
396
+ SYNTAX_ERROR, path, key, field_name,
397
+ f"the entry stops reading as BibTeX at line {error.lineno}{after}. "
398
+ "It is kept as far as it was read, so that value may hold text meant "
399
+ "for later fields; check its braces and quotes")
400
+
401
+
402
+ def parse_bibtex_file(
403
+ path: str,
404
+ diagnostics: List[Diagnostic],
405
+ redefinitions: List[Tuple[str, str, int]],
406
+ ) -> Dict[str, Entry]:
407
+ """Parse one BibTeX file into pybtex entries, keyed by citation key.
408
+
409
+ Anything the parser has to say is captured and reported by sslabdata, so no
410
+ library logging reaches the user. Syntax errors, undefined macros and
411
+ repeated citation keys go to ``diagnostics``, located at the entry and
412
+ field they were found in. Redefined ``@string`` macros are added to
413
+ ``redefinitions``, for the caller to summarise over a whole run.
414
+ """
415
+ text = Path(path).read_text(encoding="utf-8-sig")
416
+
417
+ duplicate_keys: List[str] = []
418
+ with pybtex.errors.capture() as errors:
419
+ parser = _Parser(duplicate_keys=duplicate_keys)
420
+ data = parser.parse_string(text)
421
+
422
+ redefinitions.extend((path, name, line) for name, line
423
+ in _redefined_macros(text, parser.string_definitions))
424
+ key: Optional[str]
425
+ field_name: Optional[str]
426
+ for key, field_name, found in parser.control_characters:
427
+ diagnostics.append(diagnostic(CONTROL_CHARACTER, path, key, field_name,
428
+ control_message(found)))
429
+ for key, field_name, message in parser.brace_mismatches:
430
+ diagnostics.append(diagnostic(BRACE_MISMATCH, path, key, field_name,
431
+ f"{message}; check its braces"))
432
+ for error, key, field_name, start in parser.syntax_errors:
433
+ if key is None and _on_comment_line(text, start):
434
+ continue
435
+ diagnostics.append(_syntax_diagnostic(path, error, key, field_name))
436
+ for key in duplicate_keys:
437
+ diagnostics.append(_duplicate_key_error(path, key))
438
+ for error in errors:
439
+ # The duplicate has already been recorded with sslabdata's stable code.
440
+ if str(error).startswith("repeated bibliography entry:"):
441
+ continue
442
+ diagnostics.append(diagnostic(PARSER_MESSAGE, path,
443
+ parser.message_keys.get(id(error)), None,
444
+ str(error)))
445
+
446
+ return data.entries
447
+
448
+
449
+
450
+ _UNREADABLE_LATEX = ("could not read the LaTeX in this field; keeping the text "
451
+ "as written")
452
+
453
+
454
+ class _FieldReport:
455
+ """Where one field's unknown LaTeX commands, and LaTeX that cannot be read
456
+ at all, are reported: called with a command, or ``failed()``."""
457
+
458
+ def __init__(self, unknown: Callable[[str], None],
459
+ failed: Callable[[], None]):
460
+ self.unknown = unknown
461
+ self.failed = failed
462
+
463
+ def __call__(self, command: str) -> None:
464
+ self.unknown(command)
465
+
466
+
467
+ def _convert(value: str, on_unknown: _FieldReport) -> str:
468
+ """Convert one field from LaTeX, keeping the raw text if that fails.
469
+
470
+ Each command the converter does not know is passed to ``on_unknown``,
471
+ which knows where the field is, and so is a value it cannot read at all
472
+ (``on_unknown.failed()``).
473
+
474
+ Every converted value leaves here, in NFC: the value was NFC as read, but
475
+ dropping a brace can bring a letter and its mark together, `n{\u0303}`
476
+ (SPEC.md §2).
477
+ """
478
+ try:
479
+ text = latex_to_text(value)
480
+ except Exception: # noqa: BLE001 - never drop an entry over one field
481
+ on_unknown.failed()
482
+ text = strip_braces(value)
483
+ else:
484
+ for command in unknown_commands(value):
485
+ on_unknown(command)
486
+ return nfc(text)
487
+
488
+
489
+ def unknown_command_diagnostic(command: str, where: Tuple[str, str, str],
490
+ fields: int) -> Diagnostic:
491
+ """One `LATEX_COMMAND_UNKNOWN` line for a whole run: how many fields use
492
+ the command, located at ``(file, key, field)``, the first of them."""
493
+ uses = (f"; used in {fields} field{'s' if fields != 1 else ''}, "
494
+ "located at the first")
495
+ return diagnostic(
496
+ LATEX_COMMAND_UNKNOWN, *where,
497
+ f"the LaTeX command '\\{command}' is not one sslabdata converts; it "
498
+ f"is dropped, and a braced argument after it is kept as plain "
499
+ f"text{uses}")
500
+
501
+
502
+ def _unknown_command_reporter(report, file: str, key: str,
503
+ tally: Dict[str, list]):
504
+ """For one entry: a field name → the ``on_unknown`` for that field.
505
+
506
+ A command is counted once per field, however many times it is used, and
507
+ added to ``tally[command]`` as ``[fields, first location]``, for a caller
508
+ that reports a whole run in one line per command. A field whose LaTeX
509
+ cannot be read is reported to ``report`` at once.
510
+ """
511
+ reported = set()
512
+
513
+ def in_field(field_name: str) -> _FieldReport:
514
+ def on_unknown(command: str) -> None:
515
+ if (field_name, command) in reported:
516
+ return
517
+ reported.add((field_name, command))
518
+ where = (file, key, field_name)
519
+ if command in tally:
520
+ tally[command][0] += 1
521
+ else:
522
+ tally[command] = [1, where]
523
+
524
+ def failed() -> None:
525
+ report(diagnostic(LATEX_CONVERSION_FAILED, file, key, field_name,
526
+ _UNREADABLE_LATEX))
527
+ return _FieldReport(on_unknown, failed)
528
+ return in_field
529
+
530
+
531
+ def _name_part_groups(person: Person) -> List[List[str]]:
532
+ """A pybtex name's parts, grouped as the BibTeX parts that read them.
533
+
534
+ Given and middle names are one group because they are read as one part,
535
+ and because BibTeX puts the first word of a given name in one list and the
536
+ rest in the other — which is a split a marker can land across.
537
+ """
538
+ return [list(person.first_names) + list(person.middle_names),
539
+ list(person.prelast_names),
540
+ list(person.last_names),
541
+ list(person.lineage_names)]
542
+
543
+
544
+ def _with_marker_joined(parts: List[str]) -> List[str]:
545
+ """One group of name parts, with a marker split across two of them joined."""
546
+ joined: List[str] = []
547
+ for part in parts:
548
+ if (joined and _MARKER_ARGUMENT.fullmatch(part)
549
+ and _MARKER_COMMAND.search(joined[-1])):
550
+ joined[-1] += part
551
+ else:
552
+ joined.append(part)
553
+ return joined
554
+
555
+
556
+ def _name_parts(person: Person) -> List[str]:
557
+ """Every part of a pybtex name, as written, with split markers joined."""
558
+ return [part for group in _name_part_groups(person)
559
+ for part in _with_marker_joined(group)]
560
+
561
+
562
+ def _without_marker(part: str) -> str:
563
+ """One name part with its equal-contribution markers taken off the end.
564
+
565
+ Stripping repeats, because a name written ``Brown$^{*}$*`` carries the
566
+ marker twice and taking one off would leave the other in the name.
567
+ """
568
+ while True:
569
+ stripped = EQUAL_CONTRIBUTION.sub("", part, count=1)
570
+ if stripped == part:
571
+ return part
572
+ part = stripped
573
+
574
+
575
+ def marks_equal_contribution(person: Person) -> bool:
576
+ """True when any part of this name carries an equal-contribution marker.
577
+
578
+ Given, family, von and suffix are all read: BibTeX splits the name before
579
+ sslabdata sees it, so which part the star landed on is the author's choice
580
+ of where to write it, not a different meaning.
581
+ """
582
+ return any(_without_marker(part) != part for part in _name_parts(person))
583
+
584
+
585
+ def _is_others(person: Person) -> bool:
586
+ """``and others``: BibTeX's "et al.", not a person."""
587
+ return (not person.first_names and not person.middle_names
588
+ and not person.prelast_names and not person.lineage_names
589
+ and [name.lower() for name in person.last_names] == [OTHERS])
590
+
591
+
592
+ def _is_literal(person: Person) -> bool:
593
+ """A corporate author: one brace-protected unit, with no given name."""
594
+ return (not person.first_names and not person.middle_names
595
+ and not person.prelast_names and not person.lineage_names
596
+ and len(person.last_names) == 1
597
+ and person.last_names[0].startswith("{")
598
+ and person.last_names[0].endswith("}"))
599
+
600
+
601
+ def person_name_parts(person: Person, on_unknown) -> Dict[str, Optional[str]]:
602
+ """One pybtex Person as the parts BibTeX split it into, converted to text.
603
+
604
+ ``given``, ``von``, ``family`` and ``suffix`` are BibTeX's four parts; a
605
+ corporate name comes back as ``literal`` instead, with the other four
606
+ unset. An empty part is ``None`` rather than ``""``, so the output says
607
+ "this name has no such part" rather than "it is blank".
608
+
609
+ An equal-contribution marker is not part of the name and does not appear
610
+ in any part; ``marks_equal_contribution`` reports it separately.
611
+ """
612
+ def text(parts) -> Optional[str]:
613
+ joined = " ".join(_convert(_without_marker(part), on_unknown)
614
+ for part in _with_marker_joined(parts)).strip()
615
+ return joined or None
616
+
617
+ if _is_literal(person):
618
+ return {"given": None, "von": None, "family": None, "suffix": None,
619
+ "literal": text(person.last_names)}
620
+ return {
621
+ "given": text(person.first_names + person.middle_names),
622
+ "von": text(person.prelast_names),
623
+ "family": text(person.last_names),
624
+ "suffix": text(person.lineage_names),
625
+ "literal": None,
626
+ }
627
+
628
+
629
+ def given_words(name: str) -> List[str]:
630
+ """The words BibTeX reads as the given name of a plain string: its first
631
+ and middle names, as written. Empty when BibTeX cannot read it."""
632
+ try:
633
+ person = Person(name)
634
+ except PybtexError:
635
+ # pybtex raises for three or more commas, and for a word nested
636
+ # more than 100 braces deep. `declared_form()` passes no comma, so
637
+ # from there only the nesting reaches this.
638
+ return []
639
+ return list(person.first_names) + list(person.middle_names)
640
+
641
+
642
+ def readable_name(parts: Dict[str, Optional[str]]) -> str:
643
+ """The parts of a name joined in reading order: ``John van Last Jr.``
644
+ (SPEC.md §5).
645
+
646
+ The resolver matches on the structured parts rather than on this string,
647
+ so matching can change without changing what the document displays.
648
+ """
649
+ if parts["literal"]:
650
+ return parts["literal"]
651
+ ordered = (parts["given"], parts["von"], parts["family"], parts["suffix"])
652
+ return " ".join(part for part in ordered if part)
653
+
654
+
655
+ def _contributors(entry: Entry, role: str, on_unknown) -> List[Dict]:
656
+ """The entry's names for one role, in source order, as parts plus position
657
+ (SPEC.md §3).
658
+
659
+ A name that reads as empty is dropped, as ``and others`` is, so
660
+ ``position`` counts the names that reach the document and nothing else.
661
+ """
662
+ found: List[Dict] = []
663
+ for person in entry.persons.get(role, []):
664
+ if _is_others(person):
665
+ continue
666
+ parts = person_name_parts(person, on_unknown)
667
+ name = readable_name(parts)
668
+ if name:
669
+ found.append({"name": name, "position": len(found) + 1,
670
+ "parts": parts, "person": person})
671
+ return found
672
+
673
+
674
+ def check_others(entry: Entry, bib_id: str, source: str, report) -> None:
675
+ """Report each author or editor list with ``and others`` before its end."""
676
+ for role in ("author", "editor"):
677
+ persons = entry.persons.get(role, [])
678
+ if any(_is_others(person) for person in persons[:-1]):
679
+ report(diagnostic(
680
+ OTHERS_NOT_LAST, source, bib_id, role,
681
+ f"'and others' is not the last name in the {role} list, where "
682
+ "it would mean 'et al.'; it is dropped, and the names around "
683
+ "it are kept"))
684
+
685
+
686
+ def parse_author_list(entry: Entry, on_unknown) -> List[Author]:
687
+ """The entry's authors, in source order, with no contributor resolved yet;
688
+ matching them to a person is the resolver's."""
689
+ return [Author(name=found["name"],
690
+ position=found["position"],
691
+ equal_contribution=marks_equal_contribution(found["person"]),
692
+ **found["parts"])
693
+ for found in _contributors(entry, "author", on_unknown)]
694
+
695
+
696
+ def parse_editor_list(entry: Entry, on_unknown) -> List[Contributor]:
697
+ """The entry's editors, read by the same machinery as its authors.
698
+ Editing a volume is not an authorship (SPEC.md §5)."""
699
+ return [Contributor(name=found["name"], position=found["position"],
700
+ **found["parts"])
701
+ for found in _contributors(entry, "editor", on_unknown)]
702
+
703
+
704
+ def entry_fields(bib_id: str, entry: Entry, unknown_in) -> Dict[str, str]:
705
+ """The entry's fields, with prose converted from LaTeX.
706
+
707
+ ``ENTRYTYPE`` and ``ID`` are included so the rules below read one plain
708
+ dictionary and know nothing about pybtex. No field is filled in from any
709
+ other entry: ``crossref`` is rejected rather than resolved.
710
+ """
711
+ fields = {name.lower(): value for name, value in entry.fields.items()}
712
+ read = {
713
+ name: _convert(value, unknown_in(name)) if name in TEXT_FIELDS else value
714
+ for name, value in fields.items()
715
+ }
716
+ read["ENTRYTYPE"] = entry.type.lower()
717
+ read["ID"] = bib_id
718
+ return read
719
+
720
+
721
+ class _VerbatimWriter(BibTeXWriter):
722
+ """pybtex's BibTeX writer, writing each value exactly as it was read.
723
+
724
+ The library's writer encodes every value as LaTeX, which escapes `%`, `&`,
725
+ `_` and `#` whether or not they already were: `20\\%` came back as
726
+ `20\\\\%`, a line break and a comment. A value read from a `.bib` file is
727
+ BibTeX already, so it is written as it stands. The braces are still
728
+ checked, so a value that cannot be written back is still reported.
729
+ """
730
+
731
+ def _encode(self, text):
732
+ return text
733
+
734
+
735
+ def format_bibtex(bib_id: str, entry: Entry, source: str,
736
+ report) -> Optional[str]:
737
+ """The entry written back out as BibTeX, for readers to copy (SPEC.md §5).
738
+
739
+ Nothing in sslabdata reads a value back out of it.
740
+ """
741
+ try:
742
+ return _VerbatimWriter().to_string(
743
+ BibliographyData(entries={bib_id: entry})).strip()
744
+ except Exception: # noqa: BLE001 - a copyable string is not worth an entry
745
+ report(diagnostic(
746
+ WRITE_BACK_FAILED, source, bib_id, "bibtex",
747
+ "could not write this entry back out as BibTeX; bibtex is null"))
748
+ return None
749
+
750
+
751
+
752
+ # The fields that name a work's container, in order of precedence
753
+ # (SPEC.md §5).
754
+ CONTAINER_FIELDS = ("journal", "booktitle", "school", "institution")
755
+
756
+ # The venue kind each container field implies. `booktitle` depends on the
757
+ # entry type, because a proceedings volume and a collection are different
758
+ # kinds of container under one field name.
759
+ CONTAINER_KINDS = {"journal": "journal", "school": "institution",
760
+ "institution": "institution"}
761
+ BOOKTITLE_KINDS = {"inproceedings": "conference", "conference": "conference",
762
+ "proceedings": "conference", "incollection": "book",
763
+ "inbook": "book", "book": "book"}
764
+ OTHER_KIND = "other"
765
+
766
+ # A preprint's venue is the repository `archivePrefix` or `eprinttype` names.
767
+ # arXiv is the default, because a bare `eprint` is read as an arXiv
768
+ # identifier (`build_identifiers`) and linked as one.
769
+ ARXIV = "arXiv"
770
+ REPOSITORY_KIND = "repository"
771
+
772
+ # The bibliographic fields carried flat on the work, under BibTeX's own names
773
+ # and with BibTeX's own meanings. `number` in particular is an issue number
774
+ # for an @article and a report number for a @techreport; reinterpreting it is
775
+ # not sslabdata's job, and `venue.kind` gives a consumer the branch it needs.
776
+ FLAT_FIELDS = ("volume", "number", "pages", "series", "edition", "publisher",
777
+ "address", "organization", "chapter", "month", "howpublished",
778
+ "type")
779
+
780
+ # The identifier schemes sslabdata reads out of an entry, and the field each
781
+ # comes from.
782
+ IDENTIFIER_FIELDS = {"doi": "doi", "isbn": "isbn", "issn": "issn"}
783
+
784
+ # A DOI written as a URL is the resolver plus the DOI; the identifier is the
785
+ # part after it. Stripping exactly these prefixes is not a guess -- they are
786
+ # the registered resolvers -- and it is what makes `identifiers.doi` usable
787
+ # as an identifier rather than as a second copy of the link.
788
+ DOI_RESOLVERS = ("https://doi.org/", "http://doi.org/",
789
+ "https://dx.doi.org/", "http://dx.doi.org/")
790
+ DOI_BASE = "https://doi.org/"
791
+ ARXIV_BASE = "https://arxiv.org/abs/"
792
+
793
+ # Link origins and hosts (SPEC.md §5).
794
+ FROM_INPUT = "input"
795
+ DERIVED = "derived"
796
+ VIDEO_HOSTS = ("youtube.com", "youtu.be", "vimeo.com")
797
+
798
+ UNCHECKED, VERIFIED, MISSING = "unchecked", "verified", "missing"
799
+
800
+
801
+ def build_venue(entry: dict) -> Optional[Venue]:
802
+ """The container this work appeared in, or None when the entry names none
803
+ (SPEC.md §5)."""
804
+ entry_type = entry.get("ENTRYTYPE", "")
805
+ for field_name in CONTAINER_FIELDS:
806
+ value = (entry.get(field_name) or "").strip()
807
+ if not value:
808
+ continue
809
+ if field_name == "booktitle":
810
+ kind = BOOKTITLE_KINDS.get(entry_type, OTHER_KIND)
811
+ else:
812
+ kind = CONTAINER_KINDS[field_name]
813
+ return Venue(kind=kind, name=value)
814
+
815
+ eprint = (entry.get("eprint") or "").strip()
816
+ if eprint:
817
+ return Venue(kind=REPOSITORY_KIND, name=_archive_prefix(entry))
818
+ return None
819
+
820
+
821
+ def _archive_prefix(entry: dict) -> str:
822
+ """The repository an `eprint` belongs to, as the entry names it.
823
+
824
+ biblatex names it in `eprinttype`, of which `archivePrefix` is an alias;
825
+ an entry carrying both is read from `archivePrefix`.
826
+ """
827
+ for field_name in ("archiveprefix", "eprinttype"):
828
+ prefix = (entry.get(field_name) or "").strip()
829
+ if prefix:
830
+ return prefix
831
+ return ARXIV
832
+
833
+
834
+ def bare_doi(doi: str) -> str:
835
+ """One DOI with its resolver prefix taken off, if it was written as a URL."""
836
+ doi = doi.strip()
837
+ for resolver in DOI_RESOLVERS:
838
+ if doi.lower().startswith(resolver):
839
+ return doi[len(resolver):]
840
+ return doi
841
+
842
+
843
+ def build_identifiers(entry: dict, source: str, report) -> Dict[str, List[str]]:
844
+ """The entry's identifiers, as a map from scheme to a list of identifiers.
845
+
846
+ A list, because ISBN and ISSN genuinely repeat -- a print and an
847
+ electronic one -- even though a BibTeX field holds one value.
848
+ """
849
+ identifiers: Dict[str, List[str]] = {}
850
+ for scheme, field_name in IDENTIFIER_FIELDS.items():
851
+ value = (entry.get(field_name) or "").strip()
852
+ if not value:
853
+ continue
854
+ if scheme == "doi":
855
+ if not bare_doi(value):
856
+ report(diagnostic(
857
+ DOI_INVALID, source, entry.get("ID"), field_name,
858
+ f"'{value}' is a DOI resolver with no DOI after it; the "
859
+ "work gets no DOI identifier and no DOI link"))
860
+ continue
861
+ value = bare_doi(value)
862
+ identifiers[scheme] = [value]
863
+
864
+ eprint = (entry.get("eprint") or "").strip()
865
+ if eprint:
866
+ identifiers[_archive_prefix(entry).lower()] = [eprint]
867
+ return identifiers
868
+
869
+
870
+ def is_video_url(url: str) -> bool:
871
+ """True when a URL's host is a video host, or a subdomain of one.
872
+
873
+ Only the parsed hostname counts: a lookalike domain, or a host named in
874
+ the path, query or fragment, does not. A URL that cannot be parsed, or has
875
+ no host, is not a video.
876
+ """
877
+ try:
878
+ hostname = urlsplit(url).hostname
879
+ except ValueError:
880
+ return False
881
+ return hostname is not None and any(
882
+ hostname == host or hostname.endswith("." + host)
883
+ for host in VIDEO_HOSTS)
884
+
885
+
886
+ def pdf_link(bib_id: str, pdf_base_url: Optional[str]) -> Optional[Link]:
887
+ """The PDF this work would be at under ``pdf_base_url``, checked only if
888
+ local: a build never fetches (SPEC.md §5)."""
889
+ if not pdf_base_url:
890
+ return None
891
+ base = pdf_base_url.rstrip('/')
892
+ url = f"{base}/{bib_id}.pdf"
893
+ if pdf_base_url.startswith(('http://', 'https://')):
894
+ return Link(url=url, origin=DERIVED, status=UNCHECKED)
895
+ return Link(url=url, origin=DERIVED,
896
+ status=VERIFIED if Path(url).exists() else MISSING)
897
+
898
+
899
+ def build_links(entry: dict, bib_id: str, identifiers: Dict[str, List[str]],
900
+ pdf_base_url: Optional[str]) -> Dict[str, List[Link]]:
901
+ """Every URL this work can be reached at, filed by kind (SPEC.md §5)."""
902
+ links: Dict[str, List[Link]] = {}
903
+
904
+ def add(kind: str, link: Optional[Link]) -> None:
905
+ if link is not None:
906
+ links.setdefault(kind, []).append(link)
907
+
908
+ url = (entry.get("url") or "").strip()
909
+ if url:
910
+ add("video" if is_video_url(url) else "url",
911
+ Link(url=url, origin=FROM_INPUT, status=UNCHECKED))
912
+ # `video` is always a video, whatever its host, so an entry can name a
913
+ # project website in `url` and its video here.
914
+ video = (entry.get("video") or "").strip()
915
+ if video:
916
+ add("video", Link(url=video, origin=FROM_INPUT, status=UNCHECKED))
917
+ # A `pdf` the entry names is the work's PDF, so it replaces the one
918
+ # guessed from `pdf_base_url` rather than sitting beside it.
919
+ pdf = (entry.get("pdf") or "").strip()
920
+ if pdf:
921
+ add("pdf", Link(url=pdf, origin=FROM_INPUT, status=UNCHECKED))
922
+ else:
923
+ add("pdf", pdf_link(bib_id, pdf_base_url))
924
+ for doi in identifiers.get("doi", []):
925
+ add("doi", Link(url=DOI_BASE + doi, origin=DERIVED, status=UNCHECKED))
926
+ for eprint in identifiers.get(ARXIV.lower(), []):
927
+ add("arxiv", Link(url=ARXIV_BASE + eprint, origin=DERIVED,
928
+ status=UNCHECKED))
929
+ return links
930
+
931
+
932
+ def extract_note(entry: dict) -> Optional[str]:
933
+ """Extract and format the note field."""
934
+ note = entry.get("note", "").strip().rstrip('. ')
935
+ if not note:
936
+ return None
937
+ return note
938
+
939
+
940
+ def parse_project_ids(entry: dict) -> List[str]:
941
+ """Parse the project field from a BibTeX entry."""
942
+ project_field = entry.get("project", "").strip()
943
+ if not project_field:
944
+ return []
945
+ project_field = project_field.strip('{}')
946
+ return [p.strip() for p in project_field.split(',') if p.strip()]
947
+
948
+
949
+ def entry_year(entry: dict, source: str, report) -> Optional[int]:
950
+ """The entry's year, or None with a diagnostic when it has none or it is
951
+ not a number.
952
+
953
+ Only an unsigned run of ASCII digits is a number here: `int()` would also
954
+ read `-5`, `+2020`, `2_020` and full-width `2020`.
955
+ """
956
+ raw = str(entry.get("year", "")).strip()
957
+ if not raw:
958
+ report(diagnostic(YEAR_MISSING, source, entry.get("ID"), "year",
959
+ "entry has no year"))
960
+ return None
961
+ if YEAR_DIGITS.fullmatch(raw):
962
+ return int(raw)
963
+ report(diagnostic(YEAR_INVALID, source, entry.get("ID"), "year",
964
+ f"'{raw}' is not a year, which is written in the "
965
+ "digits 0-9 alone; the work is emitted with year: null "
966
+ "and sorts last"))
967
+ return None
968
+
969
+
970
+ # The container field checked for an entry type, and the entry types
971
+ # sslabdata documents (SPEC.md "Diagnostic codes").
972
+ REQUIRED_CONTAINER = {"article": "journal", "inproceedings": "booktitle"}
973
+ SUPPORTED_TYPES = frozenset({
974
+ "article", "inproceedings", "conference", "proceedings", "incollection",
975
+ "inbook", "book", "phdthesis", "mastersthesis", "techreport", "manual",
976
+ "misc"})
977
+
978
+
979
+ def check_entry_type(entry: dict, source: str, report) -> None:
980
+ """Report an entry type sslabdata does not document, or a missing container."""
981
+ entry_type, key = entry["ENTRYTYPE"], entry["ID"]
982
+ if entry_type not in SUPPORTED_TYPES:
983
+ report(diagnostic(
984
+ ENTRY_TYPE_UNSUPPORTED, source, key, "entry_type",
985
+ f"@{entry_type} is not an entry type sslabdata documents; the "
986
+ "entry is kept, with its venue read from whichever container "
987
+ "field it carries"))
988
+ return
989
+ required = REQUIRED_CONTAINER.get(entry_type)
990
+ if required and not (entry.get(required) or "").strip():
991
+ report(diagnostic(
992
+ VENUE_MISSING, source, key, required,
993
+ f"@{entry_type} has no {required}; the entry is kept, and its "
994
+ "venue is read from any other container field it carries, or "
995
+ "is null"))
996
+
997
+
998
+ def entry_to_work(
999
+ bib_id: str,
1000
+ entry: Entry,
1001
+ category: str,
1002
+ pdf_base_url: Optional[str],
1003
+ source: str,
1004
+ source_file: str,
1005
+ report,
1006
+ unknown_commands_seen: Dict[str, list],
1007
+ ) -> Work:
1008
+ """Convert one pybtex Entry to a Work dataclass.
1009
+
1010
+ Unknown LaTeX commands are added to ``unknown_commands_seen``
1011
+ (`_unknown_command_reporter`); every other diagnostic goes to ``report``.
1012
+ """
1013
+ unknown_in = _unknown_command_reporter(report, source, bib_id,
1014
+ unknown_commands_seen)
1015
+ fields = entry_fields(bib_id, entry, unknown_in)
1016
+ identifiers = build_identifiers(fields, source, report)
1017
+ check_entry_type(fields, source, report)
1018
+ check_others(entry, bib_id, source, report)
1019
+
1020
+ return Work(
1021
+ bib_id=bib_id,
1022
+ title=fields.get("title", ""),
1023
+ authors=parse_author_list(entry, unknown_in("author")),
1024
+ editors=parse_editor_list(entry, unknown_in("editor")),
1025
+ year=entry_year(fields, source, report),
1026
+ category=category,
1027
+ entry_type=fields["ENTRYTYPE"],
1028
+ source_file=source_file,
1029
+ venue=build_venue(fields),
1030
+ abstract=fields.get("abstract"),
1031
+ note=extract_note(fields),
1032
+ identifiers=identifiers,
1033
+ links=build_links(fields, bib_id, identifiers, pdf_base_url),
1034
+ project_ids=parse_project_ids(fields),
1035
+ bibtex=format_bibtex(bib_id, entry, source, report),
1036
+ # Empty, as its default is; named so that the mapping below is
1037
+ # type-checked against the flat fields alone.
1038
+ derived={},
1039
+ **{name: fields.get(name) for name in FLAT_FIELDS},
1040
+ )
1041
+
1042
+
1043
+ def _encoding_error(path: str, error: UnicodeDecodeError) -> Diagnostic:
1044
+ """The one diagnostic for a `.bib` file that is not UTF-8."""
1045
+ line = error.object[:error.start].count(b"\n") + 1
1046
+ return diagnostic(ENCODING_INVALID, path, None, None,
1047
+ f"the file is not UTF-8: byte "
1048
+ f"0x{error.object[error.start]:02x} on line {line} "
1049
+ "cannot be read. Save the file as UTF-8.")
1050
+
1051
+
1052
+ def _crossref_error(path: str, bib_id: str, parent: str) -> Diagnostic:
1053
+ """The one diagnostic for an entry that carries a ``crossref`` field.
1054
+
1055
+ A field that is there but empty is reported as what it is rather than as
1056
+ a parent whose name happens to be blank.
1057
+ """
1058
+ names = f"names the parent '{parent}'" if parent else "names no parent"
1059
+ return diagnostic(CROSSREF_UNSUPPORTED, path, bib_id, "crossref",
1060
+ f"crossref is not supported; this entry {names}. "
1061
+ "Write the fields out on the entry itself.")
1062
+
1063
+
1064
+ def parse_all_works(
1065
+ bib_dir: str,
1066
+ bib_files: list,
1067
+ diagnostics: List[Diagnostic],
1068
+ pdf_base_url: Optional[str] = None,
1069
+ ) -> List[Work]:
1070
+ """Parse all configured BibTeX files and return a flat list of Works.
1071
+
1072
+ Every file is read first, so a citation key repeated across two of them is
1073
+ reported against both.
1074
+
1075
+ Args:
1076
+ bib_dir: Directory containing the BibTeX files
1077
+ bib_files: List of dicts with 'name' and 'category' keys
1078
+ diagnostics: The list that receives every coded diagnostic, in the
1079
+ order found
1080
+ pdf_base_url: Base URL/path for PDFs
1081
+
1082
+ Returns:
1083
+ List of Work objects, in the order SPEC.md §3 gives.
1084
+ """
1085
+ read: List[Tuple[str, str, str, Entry, str]] = []
1086
+ first_source: Dict[str, Tuple[str, str]] = {}
1087
+
1088
+ report = diagnostics.append
1089
+
1090
+ # Every file's redefined macros, summarised once when all are read.
1091
+ redefinitions: List[Tuple[str, str, int]] = []
1092
+ for bib_file in bib_files:
1093
+ name = bib_file['name'] if isinstance(bib_file, dict) else bib_file.name
1094
+ category = bib_file['category'] if isinstance(bib_file, dict) else bib_file.category
1095
+ path = f"{bib_dir}/{name}"
1096
+ try:
1097
+ parsed = parse_bibtex_file(path, diagnostics, redefinitions)
1098
+ except UnicodeDecodeError as error:
1099
+ report(_encoding_error(path, error))
1100
+ continue
1101
+ for bib_id, entry in parsed.items():
1102
+ normalized = bib_id.lower()
1103
+ if normalized in first_source:
1104
+ previous_path, previous_key = first_source[normalized]
1105
+ report(_duplicate_key_error(path, bib_id, previous_path, previous_key))
1106
+ else:
1107
+ first_source[normalized] = (path, bib_id)
1108
+ # Rejected on presence, not on value (SPEC.md "Diagnostic codes").
1109
+ crossref = [value for field_name, value in entry.fields.items()
1110
+ if field_name.lower() == "crossref"]
1111
+ if crossref:
1112
+ report(_crossref_error(path, bib_id, str(crossref[0]).strip()))
1113
+ continue
1114
+ read.append((path, name, bib_id, entry, category))
1115
+
1116
+ summary = redefined_summary(redefinitions)
1117
+ if summary:
1118
+ report(summary)
1119
+
1120
+ # One line per unknown command for the whole run: a real bibliography
1121
+ # can use one command in thousands of fields, and a line for each would
1122
+ # bury every other diagnostic.
1123
+ unknown: Dict[str, list] = {}
1124
+ works = [
1125
+ entry_to_work(bib_id, entry, category, pdf_base_url, path, name, report,
1126
+ unknown)
1127
+ for path, name, bib_id, entry, category in read
1128
+ ]
1129
+ for command, (fields, where) in unknown.items():
1130
+ report(unknown_command_diagnostic(command, where, fields))
1131
+ works.sort(key=lambda w: (w.year is not None, w.year or 0), reverse=True)
1132
+ return works