sslabdata 3.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sslabdata/__init__.py +40 -0
- sslabdata/assembler.py +452 -0
- sslabdata/cli.py +274 -0
- sslabdata/config.py +443 -0
- sslabdata/diagnostics.py +159 -0
- sslabdata/exporters.py +86 -0
- sslabdata/loaders.py +298 -0
- sslabdata/models.py +342 -0
- sslabdata/parsers/__init__.py +0 -0
- sslabdata/parsers/bibtex.py +1132 -0
- sslabdata/parsers/latex.py +215 -0
- sslabdata/resolver.py +517 -0
- sslabdata/schema/__init__.py +0 -0
- sslabdata/schema/v5/output.schema.json +458 -0
- sslabdata-3.0.0.dist-info/METADATA +409 -0
- sslabdata-3.0.0.dist-info/RECORD +20 -0
- sslabdata-3.0.0.dist-info/WHEEL +5 -0
- sslabdata-3.0.0.dist-info/entry_points.txt +2 -0
- sslabdata-3.0.0.dist-info/licenses/LICENSE +21 -0
- sslabdata-3.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""
|
|
2
|
+
LaTeX → plain Unicode text, via pylatexenc, one field value at a time
|
|
3
|
+
(SPEC.md §2).
|
|
4
|
+
|
|
5
|
+
Together with bibtex.py this is the adapter: no other module imports pybtex or
|
|
6
|
+
pylatexenc.
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2024 Personal Robotics Laboratory, University of Washington
|
|
9
|
+
Author: Siddhartha Srinivasa
|
|
10
|
+
MIT License - see LICENSE file for details.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
from pylatexenc.latex2text import (
|
|
16
|
+
LatexNodes2Text, MacroTextSpec, get_default_latex_context_db,
|
|
17
|
+
)
|
|
18
|
+
from pylatexenc.latexwalker import (
|
|
19
|
+
LatexMacroNode, LatexMathNode, LatexWalker,
|
|
20
|
+
get_default_latex_context_db as get_default_parsing_db,
|
|
21
|
+
)
|
|
22
|
+
from pylatexenc.macrospec import std_macro
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# Common text macros the converter's own table lacks. Without a rule each
|
|
26
|
+
# would be dropped, so `The \TeX{} book 1990\emdash 2000` would read
|
|
27
|
+
# `The book 19902000`.
|
|
28
|
+
_TEXT_MACROS = {
|
|
29
|
+
"TeX": "TeX", "LaTeX": "LaTeX", "LaTeXe": "LaTeX2e", "BibTeX": "BibTeX",
|
|
30
|
+
"emdash": "\u2014", "endash": "\u2013", "slash": "/",
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
# Marks a command that leaves nothing, or a parenthesis, where it stood, so
|
|
34
|
+
# the spaces before it go too. This marker and `_PLACEHOLDER` below are control
|
|
35
|
+
# characters, which cannot collide with the input: those are removed where a
|
|
36
|
+
# value is read, before it is converted (`CONTROL_CHARACTER` in config.py).
|
|
37
|
+
_JOIN = '\x02'
|
|
38
|
+
_JOIN_RE = re.compile(f'[ \t\xa0]*{_JOIN}')
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _footnote(node, l2tobj):
|
|
42
|
+
return f'{_JOIN} ({l2tobj.node_arg_to_text(node, 1).strip()})'
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _item(node, l2tobj):
|
|
46
|
+
if node.nodeoptarg:
|
|
47
|
+
return _JOIN + '\n' + l2tobj.nodelist_to_text([node.nodeoptarg])
|
|
48
|
+
return _JOIN + '\n\u2022 '
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# The converter's own table writes these as syntax rather than text
|
|
52
|
+
# (SPEC.md "What sslabdata converts, and what it does not").
|
|
53
|
+
_PLAIN_TEXT_RULES = [
|
|
54
|
+
# A `\url` whose argument holds braces is not set aside by `_prepared()`.
|
|
55
|
+
MacroTextSpec("url", "%s"),
|
|
56
|
+
MacroTextSpec("footnote", _footnote),
|
|
57
|
+
MacroTextSpec("item", _item),
|
|
58
|
+
MacroTextSpec("textfrac", "%s/%s"),
|
|
59
|
+
] + [MacroTextSpec(name, _JOIN) for name in (
|
|
60
|
+
"cite", "citep", "citet", "ref", "autoref", "cref", "Cref", "eqref",
|
|
61
|
+
"includegraphics",
|
|
62
|
+
)] + [MacroTextSpec("maketitle", "")]
|
|
63
|
+
|
|
64
|
+
# Commands whose argument is the text itself; the converter's own rule for
|
|
65
|
+
# `\title`, `\author` and `\date` drops it.
|
|
66
|
+
_TEXT_ARGUMENT = ("texttt", "textsf", "textmd", "textup", "textnormal",
|
|
67
|
+
"mbox", "fbox", "hbox", "title", "author", "date")
|
|
68
|
+
|
|
69
|
+
# Commands whose arguments are not text: a reference takes the spaces before
|
|
70
|
+
# it, and a setting does not (SPEC.md §2).
|
|
71
|
+
_REFERENCES = ("citealp", "citealt", "citeauthor", "citefullauthor",
|
|
72
|
+
"citenum", "citeyear", "citeyearpar", "citepalias",
|
|
73
|
+
"citetalias", "Citealp", "Citealt", "Citeauthor", "Citep",
|
|
74
|
+
"Citet", "nocite", "label", "pageref", "nameref")
|
|
75
|
+
_SETTINGS = ("color", "colorlet", "definecolor", "providecolor", "pagecolor",
|
|
76
|
+
"nopagecolor", "rowcolors", "documentclass", "usepackage",
|
|
77
|
+
"RequirePackage", "bibliography", "hypersetup", "selectlanguage",
|
|
78
|
+
"setcounter", "addcounter", "setlength", "addlength",
|
|
79
|
+
"defcitealias", "hphantom", "vphantom")
|
|
80
|
+
|
|
81
|
+
# The commands the converter has a rule for. A command outside it is dropped,
|
|
82
|
+
# and what follows it is read as plain text (`_WITHOUT_RULE` below).
|
|
83
|
+
_KNOWN = get_default_latex_context_db()
|
|
84
|
+
_KNOWN.add_context_category(
|
|
85
|
+
"sslabdata-text",
|
|
86
|
+
macros=[MacroTextSpec(name, text) for name, text in _TEXT_MACROS.items()]
|
|
87
|
+
+ _PLAIN_TEXT_RULES
|
|
88
|
+
+ [MacroTextSpec(name, "%s") for name in _TEXT_ARGUMENT]
|
|
89
|
+
+ [MacroTextSpec(name, _JOIN) for name in _REFERENCES]
|
|
90
|
+
+ [MacroTextSpec(name, "") for name in _SETTINGS],
|
|
91
|
+
prepend=True)
|
|
92
|
+
|
|
93
|
+
# How the walker reads each command's arguments. `\textfrac` has a text rule
|
|
94
|
+
# but no argument spec, so without this its rule would print `%s/%s`; nor do
|
|
95
|
+
# `\textnormal`, `\fbox`, `\hbox`, `\nocite`, `\pageref` and `\nameref`.
|
|
96
|
+
_PARSING = get_default_parsing_db()
|
|
97
|
+
_PARSING.add_context_category(
|
|
98
|
+
"sslabdata-arguments",
|
|
99
|
+
macros=[std_macro("textfrac", False, 2)]
|
|
100
|
+
+ [std_macro(name, False, 1) for name in (
|
|
101
|
+
"textnormal", "fbox", "hbox", "nocite", "pageref", "nameref")],
|
|
102
|
+
prepend=True)
|
|
103
|
+
|
|
104
|
+
# A command the walker knows the arguments of but the converter has no rule
|
|
105
|
+
# for is read as taking no arguments, like one the walker does not know at
|
|
106
|
+
# all; otherwise the converter would drop its arguments with it. That includes
|
|
107
|
+
# `\newcommand` and its kin, which would read a set-aside URL as a name.
|
|
108
|
+
_WITHOUT_RULE = sorted({spec.macroname for spec in _PARSING.iter_macro_specs()
|
|
109
|
+
if _KNOWN.get_macro_spec(spec.macroname) is None})
|
|
110
|
+
_PARSING.add_context_category(
|
|
111
|
+
"sslabdata-no-arguments",
|
|
112
|
+
macros=[std_macro(name, False, 0) for name in _WITHOUT_RULE],
|
|
113
|
+
prepend=True)
|
|
114
|
+
|
|
115
|
+
_CONVERTER = LatexNodes2Text(latex_context=_KNOWN, math_mode='verbatim')
|
|
116
|
+
|
|
117
|
+
# Two commands outside that table whose conversion sslabdata documents, so
|
|
118
|
+
# they are not reported as unknown (SPEC.md "Diagnostic codes").
|
|
119
|
+
_DOCUMENTED = frozenset({"textsuperscript", "*"})
|
|
120
|
+
|
|
121
|
+
# pylatexenc 2.11 raises IndexError on every \href, so the link is rewritten
|
|
122
|
+
# to "text (url)" before conversion. The URL itself is set aside first: it is
|
|
123
|
+
# not LaTeX, and characters such as _ or % would not survive the converter.
|
|
124
|
+
_HREF_URL_TEXT = re.compile(r'\\href\s*\{([^{}]*)\}\s*\{((?:[^{}]|\{[^{}]*\})*)\}')
|
|
125
|
+
_HREF_URL_ONLY = re.compile(r'\\href\s*\{([^{}]*)\}')
|
|
126
|
+
# `\url{u}` is emitted as `u` exactly: the URL is set aside the same way, so a
|
|
127
|
+
# `~`, `%`, `_`, `#` or `&` in it is kept rather than converted.
|
|
128
|
+
_URL = re.compile(r'\\url\s*\{([^{}]*)\}')
|
|
129
|
+
|
|
130
|
+
# An unescaped & in a BibTeX field is a literal ampersand, not an alignment tab.
|
|
131
|
+
_BARE_AMPERSAND = re.compile(r'(?<!\\)&')
|
|
132
|
+
# An unescaped % is a literal percent, not a comment: BibTeX keeps the rest of
|
|
133
|
+
# the value. It is escaped when an even run of backslashes (`\\`, a line
|
|
134
|
+
# break) or none comes before it, and left alone after `\%`.
|
|
135
|
+
_BARE_PERCENT = re.compile(r'(?<!\\)((?:\\\\)*)%')
|
|
136
|
+
|
|
137
|
+
_PLACEHOLDER = '\x01'
|
|
138
|
+
_PLACEHOLDER_RE = re.compile(f'{_PLACEHOLDER}(\\d+){_PLACEHOLDER}')
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def latex_to_text(text: str) -> str:
|
|
142
|
+
"""Convert one LaTeX field value to plain Unicode text.
|
|
143
|
+
|
|
144
|
+
Raises whatever pylatexenc raises, and ValueError when a marker survives
|
|
145
|
+
conversion; callers decide what to do with a value that cannot be
|
|
146
|
+
converted.
|
|
147
|
+
"""
|
|
148
|
+
if not text:
|
|
149
|
+
return text
|
|
150
|
+
|
|
151
|
+
verbatim: list = []
|
|
152
|
+
|
|
153
|
+
def set_aside(value: str) -> str:
|
|
154
|
+
verbatim.append(value)
|
|
155
|
+
return f'{_PLACEHOLDER}{len(verbatim) - 1}{_PLACEHOLDER}'
|
|
156
|
+
|
|
157
|
+
converted = _CONVERTER.latex_to_text(_prepared(text, set_aside),
|
|
158
|
+
latex_context=_PARSING)
|
|
159
|
+
converted = _JOIN_RE.sub('', converted)
|
|
160
|
+
converted = _PLACEHOLDER_RE.sub(lambda m: verbatim[int(m.group(1))],
|
|
161
|
+
converted)
|
|
162
|
+
# A command that read part of a placeholder as its argument leaves the
|
|
163
|
+
# rest behind. The input holds no marker, so any left is one of ours.
|
|
164
|
+
if _PLACEHOLDER in converted or _JOIN in converted:
|
|
165
|
+
raise ValueError("a conversion marker survived")
|
|
166
|
+
return converted
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _prepared(text: str, set_aside) -> str:
|
|
170
|
+
"""The value as the converter is given it: links rewritten, bare & and % escaped."""
|
|
171
|
+
text = _HREF_URL_TEXT.sub(lambda m: f'{m.group(2)} ({set_aside(m.group(1))})', text)
|
|
172
|
+
text = _HREF_URL_ONLY.sub(lambda m: set_aside(m.group(1)), text)
|
|
173
|
+
text = _URL.sub(lambda m: set_aside(m.group(1)), text)
|
|
174
|
+
text = _BARE_PERCENT.sub(r'\1\\%', text)
|
|
175
|
+
return _BARE_AMPERSAND.sub(r'\&', text)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def unknown_commands(text: str) -> list:
|
|
179
|
+
"""The LaTeX commands in one value the converter has no rule for, in order.
|
|
180
|
+
|
|
181
|
+
Math is not searched: it is left as TeX for the renderer. A value the
|
|
182
|
+
walker cannot read at all yields nothing here; converting it is what
|
|
183
|
+
reports that.
|
|
184
|
+
"""
|
|
185
|
+
if not text:
|
|
186
|
+
return []
|
|
187
|
+
try:
|
|
188
|
+
nodes, _, _ = LatexWalker(_prepared(text, lambda _: ''),
|
|
189
|
+
latex_context=_PARSING,
|
|
190
|
+
tolerant_parsing=True).get_latex_nodes()
|
|
191
|
+
except Exception: # noqa: BLE001 - finding nothing is the safe answer
|
|
192
|
+
return []
|
|
193
|
+
found: list = []
|
|
194
|
+
|
|
195
|
+
def walk(nodelist):
|
|
196
|
+
for node in nodelist or []:
|
|
197
|
+
if node is None or isinstance(node, LatexMathNode):
|
|
198
|
+
continue
|
|
199
|
+
if (isinstance(node, LatexMacroNode)
|
|
200
|
+
and node.macroname not in _DOCUMENTED
|
|
201
|
+
and _KNOWN.get_macro_spec(node.macroname) is None
|
|
202
|
+
and node.macroname not in found):
|
|
203
|
+
found.append(node.macroname)
|
|
204
|
+
walk(getattr(node, 'nodelist', None))
|
|
205
|
+
arguments = getattr(node, 'nodeargd', None)
|
|
206
|
+
if arguments is not None:
|
|
207
|
+
walk(arguments.argnlist)
|
|
208
|
+
|
|
209
|
+
walk(nodes)
|
|
210
|
+
return found
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def strip_braces(text: str) -> str:
|
|
214
|
+
"""The fallback for a value pylatexenc cannot convert: keep it, lose the braces."""
|
|
215
|
+
return text.replace('{', '').replace('}', '')
|