texdiff 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- texdiff/__init__.py +45 -0
- texdiff/align.py +202 -0
- texdiff/api.py +326 -0
- texdiff/check.py +61 -0
- texdiff/cli.py +87 -0
- texdiff/emit.py +1193 -0
- texdiff/flatten.py +185 -0
- texdiff/nodes.py +147 -0
- texdiff/oldlines.py +49 -0
- texdiff/parse.py +524 -0
- texdiff/preamble.py +388 -0
- texdiff/tables.py +540 -0
- texdiff/textdiff.py +157 -0
- texdiff-0.2.0.dist-info/METADATA +112 -0
- texdiff-0.2.0.dist-info/RECORD +19 -0
- texdiff-0.2.0.dist-info/WHEEL +5 -0
- texdiff-0.2.0.dist-info/entry_points.txt +2 -0
- texdiff-0.2.0.dist-info/licenses/LICENSE +21 -0
- texdiff-0.2.0.dist-info/top_level.txt +1 -0
texdiff/parse.py
ADDED
|
@@ -0,0 +1,524 @@
|
|
|
1
|
+
"""Parser wrapper: LaTeX source → list of :class:`texdiff.nodes.Node`.
|
|
2
|
+
|
|
3
|
+
Built on :mod:`pylatexenc.latexwalker`. This module owns all knowledge
|
|
4
|
+
about pylatexenc: the rest of texdiff works only on our own Node type.
|
|
5
|
+
|
|
6
|
+
Design decisions (v0):
|
|
7
|
+
|
|
8
|
+
* ``\\input``/``\\include`` are NOT resolved here yet - v0 expects the
|
|
9
|
+
caller to compare flattened sources or single files (flattening via
|
|
10
|
+
latexpand stays an external step); see docs/roadmap.
|
|
11
|
+
* Math, verbatim environments, comments and unknown macros become
|
|
12
|
+
ATOMIC nodes: exact-match comparison, never diffed inside.
|
|
13
|
+
* Known text-bearing environments and groups get children so the
|
|
14
|
+
aligner can recurse into them.
|
|
15
|
+
* Source positions recorded by pylatexenc are dropped: ``Node.text``
|
|
16
|
+
carries the exact span already.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import re
|
|
22
|
+
from dataclasses import replace
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
from pylatexenc.latexwalker import (
|
|
26
|
+
LatexCommentNode,
|
|
27
|
+
LatexEnvironmentNode,
|
|
28
|
+
LatexGroupNode,
|
|
29
|
+
LatexMacroNode,
|
|
30
|
+
LatexCharsNode,
|
|
31
|
+
LatexMathNode,
|
|
32
|
+
LatexSpecialsNode,
|
|
33
|
+
LatexWalker,
|
|
34
|
+
get_default_latex_context_db,
|
|
35
|
+
)
|
|
36
|
+
from pylatexenc.macrospec import (
|
|
37
|
+
EnvironmentSpec,
|
|
38
|
+
MacroStandardArgsParser,
|
|
39
|
+
ParsedMacroArgs,
|
|
40
|
+
ParsedVerbatimArgs,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
from .nodes import Node
|
|
44
|
+
|
|
45
|
+
# Environments whose body is text-like: the aligner may recurse into
|
|
46
|
+
# them (marking up changes inside paragraphs). Everything else -
|
|
47
|
+
# verbatim (lstlisting, verbatim, minted...), math displays, tikz
|
|
48
|
+
# pictures - is an atomic block by default.
|
|
49
|
+
TEXT_ENVIRONMENTS = frozenset(
|
|
50
|
+
{
|
|
51
|
+
"document",
|
|
52
|
+
"itemize",
|
|
53
|
+
"enumerate",
|
|
54
|
+
"description",
|
|
55
|
+
"quote",
|
|
56
|
+
"quotation",
|
|
57
|
+
"center",
|
|
58
|
+
"flushleft",
|
|
59
|
+
"flushright",
|
|
60
|
+
}
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
# Environments that must NEVER be diffed internally.
|
|
64
|
+
VERBATIM_ENVIRONMENTS = frozenset(
|
|
65
|
+
{
|
|
66
|
+
"verbatim",
|
|
67
|
+
"verbatim*",
|
|
68
|
+
"lstlisting",
|
|
69
|
+
"minted",
|
|
70
|
+
"alltt",
|
|
71
|
+
"filecontents",
|
|
72
|
+
"filecontents*",
|
|
73
|
+
}
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
# Table environments: atomic in v0 (whole-block replace when their
|
|
77
|
+
# structure changes). Row-granular alignment is a v1 roadmap item.
|
|
78
|
+
TABLE_ENVIRONMENTS = frozenset(
|
|
79
|
+
{"tabular", "tabular*", "tabularx", "longtable", "longtable*", "array"}
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class _NamedVerbatimArgsParser(MacroStandardArgsParser):
|
|
84
|
+
"""Verbatim args parser for named environments.
|
|
85
|
+
|
|
86
|
+
pylatexenc's :class:`VerbatimArgsParser` hardcodes the terminator
|
|
87
|
+
``\\end{verbatim}`` — registering ``lstlisting`` (or any other
|
|
88
|
+
named verbatim environment) with it makes the walker swallow
|
|
89
|
+
everything up to a *literal* ``\\end{verbatim}`` far later in the
|
|
90
|
+
document, merging dozens of unrelated listings and the prose
|
|
91
|
+
between them into one giant node. This subclass scans for the
|
|
92
|
+
environment's own ``\\end{<envname>}``, exactly as LaTeX does.
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
def __init__(self, envname: str, **kwargs):
|
|
96
|
+
super().__init__(argspec="{", **kwargs)
|
|
97
|
+
self.envname = envname
|
|
98
|
+
|
|
99
|
+
def parse_args(self, w, pos, parsing_state=None):
|
|
100
|
+
from pylatexenc import latexwalker
|
|
101
|
+
|
|
102
|
+
endtoken = f"\\end{{{self.envname}}}"
|
|
103
|
+
endpos = w.s.find(endtoken, pos)
|
|
104
|
+
if endpos == -1:
|
|
105
|
+
raise latexwalker.LatexWalkerParseError(
|
|
106
|
+
s=w.s,
|
|
107
|
+
pos=pos,
|
|
108
|
+
msg=f"Cannot find matching {endtoken}",
|
|
109
|
+
)
|
|
110
|
+
len_ = endpos - pos
|
|
111
|
+
argd = ParsedVerbatimArgs(
|
|
112
|
+
verbatim_chars_node=w.make_node(
|
|
113
|
+
latexwalker.LatexCharsNode,
|
|
114
|
+
parsing_state=parsing_state,
|
|
115
|
+
chars=w.s[pos : pos + len_],
|
|
116
|
+
pos=pos,
|
|
117
|
+
len=len_,
|
|
118
|
+
)
|
|
119
|
+
)
|
|
120
|
+
return (argd, pos, len_)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _latex_context():
|
|
124
|
+
"""Default pylatexenc context extended with verbatim environments.
|
|
125
|
+
|
|
126
|
+
pylatexenc's default specs only know ``verbatim`` as a verbatim
|
|
127
|
+
environment; ``lstlisting``, ``minted``, ``alltt`` and friends are
|
|
128
|
+
parsed as *regular* environments, so code listings containing
|
|
129
|
+
unbalanced braces (C++ ``std::tuple<int, ...>`` constructors,
|
|
130
|
+
``#include <x>`` in ``alltt``...) make strict parsing fail and
|
|
131
|
+
tolerant parsing produce garbage error nodes. Registering them
|
|
132
|
+
with a named verbatim args parser makes the walker read their
|
|
133
|
+
bodies verbatim, exactly as LaTeX does.
|
|
134
|
+
"""
|
|
135
|
+
db = get_default_latex_context_db()
|
|
136
|
+
db.add_context_category(
|
|
137
|
+
"texdiff-verbatim",
|
|
138
|
+
macros=[],
|
|
139
|
+
environments=[
|
|
140
|
+
EnvironmentSpec(
|
|
141
|
+
envname,
|
|
142
|
+
args_parser=_NamedVerbatimArgsParser(envname),
|
|
143
|
+
)
|
|
144
|
+
for envname in VERBATIM_ENVIRONMENTS
|
|
145
|
+
],
|
|
146
|
+
specials=[],
|
|
147
|
+
prepend=True,
|
|
148
|
+
)
|
|
149
|
+
return db
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
_LATEX_CONTEXT = _latex_context()
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class ParseError(ValueError):
|
|
156
|
+
"""Raised when LaTeX source cannot be parsed even tolerantly."""
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def parse(source: str) -> list[Node]:
|
|
160
|
+
"""Parse LaTeX source text into texdiff nodes.
|
|
161
|
+
|
|
162
|
+
Args:
|
|
163
|
+
source: LaTeX source of one file (already flattened if the
|
|
164
|
+
document is spread over several files).
|
|
165
|
+
|
|
166
|
+
Returns:
|
|
167
|
+
Top-level node list. Concatenating the ``text`` of all returned
|
|
168
|
+
nodes reproduces ``source`` exactly.
|
|
169
|
+
|
|
170
|
+
Raises:
|
|
171
|
+
ParseError: if the source has unbalanced group delimiters or
|
|
172
|
+
cannot be parsed even tolerantly.
|
|
173
|
+
|
|
174
|
+
Policy: parse strictly if possible (catches most malformed input),
|
|
175
|
+
fall back to *tolerant* parsing when strict mode rejects a construct
|
|
176
|
+
real documents use (e.g. ``\\makecell`` bodies with ``\\\\``). In
|
|
177
|
+
tolerant mode a balance check over the resulting edit script keeps
|
|
178
|
+
unbalanced braces out (pylatexexenc would silently swallow them,
|
|
179
|
+
which must not reach the emitter).
|
|
180
|
+
"""
|
|
181
|
+
nodes: list[Node]
|
|
182
|
+
try:
|
|
183
|
+
nodelist, _pos, _len = LatexWalker(
|
|
184
|
+
source, latex_context=_LATEX_CONTEXT, tolerant_parsing=False
|
|
185
|
+
).get_latex_nodes()
|
|
186
|
+
converted = [_convert(n, source) for n in nodelist]
|
|
187
|
+
except Exception:
|
|
188
|
+
# strict mode failed: real-world construct or genuinely broken?
|
|
189
|
+
try:
|
|
190
|
+
nodelist, _pos, length = LatexWalker(
|
|
191
|
+
source, latex_context=_LATEX_CONTEXT, tolerant_parsing=True
|
|
192
|
+
).get_latex_nodes()
|
|
193
|
+
except Exception as exc:
|
|
194
|
+
raise ParseError(f"cannot parse LaTeX source: {exc}") from exc
|
|
195
|
+
if length == len(source) and _has_error_nodes(nodelist, source):
|
|
196
|
+
raise ParseError(
|
|
197
|
+
"LaTeX source is malformed (tolerant parsing recovered, "
|
|
198
|
+
"but with error tokens)"
|
|
199
|
+
)
|
|
200
|
+
converted = [_convert(n, source) for n in nodelist]
|
|
201
|
+
|
|
202
|
+
if "".join(n.text for n in converted) != source:
|
|
203
|
+
# round-trip broken: reject rather than emit wrong bytes
|
|
204
|
+
raise ParseError("parse round-trip mismatch (node spans do not cover source)")
|
|
205
|
+
return _fold_escapes(converted)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
# --- escape folding -----------------------------------------------------------
|
|
209
|
+
|
|
210
|
+
# Escape macros that typeset a literal character (``\_`` `` \& `` ...
|
|
211
|
+
# ) carry no syntactic meaning: their only effect in prose is to
|
|
212
|
+
# create a node boundary. When one revision writes ``ABCDE`` as one
|
|
213
|
+
# text run and the other writes ``ABCD\_-E`` (the identical string
|
|
214
|
+
# with an escaped underscore), the node lists disagree on
|
|
215
|
+
# granularity: the aligner pairs the whole old run with just the
|
|
216
|
+
# fragment before ``\_``, and the word diff then strikes out the
|
|
217
|
+
# entire unchanged tail of the old run ("file type. When the list
|
|
218
|
+
# ..."). Folding both into plain-text runs with identical structure
|
|
219
|
+
# restores granularity symmetry; the exact source text is kept in
|
|
220
|
+
# each node, so a folded span matching both sides round-trips
|
|
221
|
+
# verbatim.
|
|
222
|
+
_ESCAPE_MACRO_RE = re.compile(
|
|
223
|
+
r"\\(?:_|&|%|#|\$)\s*(?![a-zA-Z])" # \_ \& \% \# \$ not part of \_cmd
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _is_glue_macro(node: Node) -> bool:
|
|
228
|
+
"""True if node is an escape macro like ``\\_`` (no argument)."""
|
|
229
|
+
return (
|
|
230
|
+
node.kind == "macro"
|
|
231
|
+
and node.name in ("_", "&", "%", "#", "$")
|
|
232
|
+
and node.text.strip() == f"\\{node.name}"
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _fold_escapes(nodes: list[Node]) -> list[Node]:
|
|
237
|
+
"""Merge escape macros into an adjacent text node.
|
|
238
|
+
|
|
239
|
+
Recurses into recursable children. An escape macro between two
|
|
240
|
+
text nodes merges left (prose continuation); a leading escape
|
|
241
|
+
macro merges right. Standalone escapes (surrounded by structure)
|
|
242
|
+
stay untouched - text equality still round-trips them.
|
|
243
|
+
"""
|
|
244
|
+
out: list[Node] = []
|
|
245
|
+
for node in nodes:
|
|
246
|
+
if node.children:
|
|
247
|
+
node = replace(node, children=_fold_escapes(node.children))
|
|
248
|
+
if _is_glue_macro(node) and out and out[-1].kind == "text":
|
|
249
|
+
out[-1] = replace(out[-1], text=out[-1].text + node.text)
|
|
250
|
+
continue
|
|
251
|
+
if (
|
|
252
|
+
_is_glue_macro(node)
|
|
253
|
+
and not out
|
|
254
|
+
and len(nodes) > 1
|
|
255
|
+
and nodes[1].kind == "text"
|
|
256
|
+
):
|
|
257
|
+
# leading escape with text following: remember, merge right
|
|
258
|
+
out.append(node)
|
|
259
|
+
continue
|
|
260
|
+
if node.kind == "text" and out and _is_glue_macro(out[-1]):
|
|
261
|
+
out[-1] = replace(out[-1], text=out[-1].text + node.text)
|
|
262
|
+
continue
|
|
263
|
+
if node.kind == "text" and out and out[-1].kind == "text":
|
|
264
|
+
# consecutive text runs (an artefact of escape folding or
|
|
265
|
+
# of comments/verbatim dropped between them in earlier
|
|
266
|
+
# processing): merge so both revisions present prose with
|
|
267
|
+
# identical granularity - the aligner pairs like with like
|
|
268
|
+
out[-1] = replace(out[-1], text=out[-1].text + node.text)
|
|
269
|
+
continue
|
|
270
|
+
out.append(node)
|
|
271
|
+
return out
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def parse_file(path: str | Path) -> list[Node]:
|
|
275
|
+
"""Read and parse a ``.tex`` file (UTF-8)."""
|
|
276
|
+
return parse(Path(path).read_text(encoding="utf-8"))
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
# --- pylatexenc → texdiff conversion ---------------------------------------
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _convert(node: LatexNode, source: str) -> Node:
|
|
283
|
+
"""Convert one pylatexenc node (recursively) to a texdiff Node."""
|
|
284
|
+
text = _span(node, source)
|
|
285
|
+
|
|
286
|
+
if isinstance(node, LatexCharsNode):
|
|
287
|
+
return Node(kind="text", text=text, atom=True)
|
|
288
|
+
|
|
289
|
+
if isinstance(node, LatexCommentNode):
|
|
290
|
+
return Node(kind="comment", text=text, atom=True)
|
|
291
|
+
|
|
292
|
+
if isinstance(node, LatexMathNode):
|
|
293
|
+
return Node(kind="math", text=text, atom=True)
|
|
294
|
+
|
|
295
|
+
if isinstance(node, LatexSpecialsNode):
|
|
296
|
+
return Node(kind="specials", text=text, atom=True)
|
|
297
|
+
|
|
298
|
+
if isinstance(node, LatexGroupNode):
|
|
299
|
+
return Node(
|
|
300
|
+
kind="group",
|
|
301
|
+
text=text,
|
|
302
|
+
children=[_convert(c, source) for c in node.nodelist],
|
|
303
|
+
atom=False,
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
if isinstance(node, LatexMacroNode):
|
|
307
|
+
# macros stay atomic in v0; recursable text macros (\textbf,
|
|
308
|
+
# \makecell, \emph, ...) are a v1 configuration item
|
|
309
|
+
return Node(kind="macro", text=text, name=node.macroname, atom=True)
|
|
310
|
+
|
|
311
|
+
if isinstance(node, LatexEnvironmentNode):
|
|
312
|
+
envname = node.envname
|
|
313
|
+
if envname in VERBATIM_ENVIRONMENTS:
|
|
314
|
+
return Node(kind="env", text=text, name=envname, atom=True)
|
|
315
|
+
if envname in TABLE_ENVIRONMENTS:
|
|
316
|
+
return _table_env_node(node, source, envname)
|
|
317
|
+
recursable = envname in TEXT_ENVIRONMENTS
|
|
318
|
+
return Node(
|
|
319
|
+
kind="env",
|
|
320
|
+
text=text,
|
|
321
|
+
name=envname,
|
|
322
|
+
atom=not recursable,
|
|
323
|
+
children=[_convert(c, source) for c in node.nodelist] if recursable else [],
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
# unknown pylatexenc node class: keep as an opaque atomic block
|
|
327
|
+
return Node(kind="specials", text=text, atom=True) # pragma: no cover
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
# --- table ↔ row nodes -------------------------------------------------------
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _table_env_node(node: LatexEnvironmentNode, source: str, envname: str) -> Node:
|
|
334
|
+
"""Convert a table environment into a recursable env of row nodes.
|
|
335
|
+
|
|
336
|
+
Row splitting is brace-aware (``\\\\makecell{a\\\\\\\\b}`` is one row)
|
|
337
|
+
and turns ``\\\\endfirsthead``/``\\\\endhead``/``\\\\endfoot``/``\\\\endlastfoot``
|
|
338
|
+
boundaries into row-kind segments, so the aligner can keep header
|
|
339
|
+
and body regions separate.
|
|
340
|
+
"""
|
|
341
|
+
children = _split_table_body(node, source)
|
|
342
|
+
return Node(
|
|
343
|
+
kind="env",
|
|
344
|
+
text=source[node.pos : node.pos + node.len],
|
|
345
|
+
name=envname,
|
|
346
|
+
children=children,
|
|
347
|
+
atom=False,
|
|
348
|
+
)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _split_table_body(node: LatexEnvironmentNode, source: str) -> list[Node]:
|
|
352
|
+
"""Split a table body (source span) into row nodes.
|
|
353
|
+
|
|
354
|
+
A *row* ends at a ``\\\\`` which sits outside braces and outside
|
|
355
|
+
comments. Everything between the environment's begin/end markup
|
|
356
|
+
is split at those points; each segment (including its trailing
|
|
357
|
+
``\\\\`` and line breaks) becomes one atomic row node whose
|
|
358
|
+
signature is its normalized content.
|
|
359
|
+
"""
|
|
360
|
+
body_start = _env_body_start(node, source)
|
|
361
|
+
body_end = _env_body_end(node, source)
|
|
362
|
+
body = source[body_start:body_end]
|
|
363
|
+
|
|
364
|
+
rows: list[Node] = []
|
|
365
|
+
seg_start = 0
|
|
366
|
+
depth = 0
|
|
367
|
+
i = 0
|
|
368
|
+
n = len(body)
|
|
369
|
+
while i < n:
|
|
370
|
+
c = body[i]
|
|
371
|
+
if c == "\\":
|
|
372
|
+
# comment: skip to end of line
|
|
373
|
+
if i + 1 < n and body[i + 1] == "%":
|
|
374
|
+
j = body.find("\n", i)
|
|
375
|
+
i = n if j < 0 else j + 1
|
|
376
|
+
continue
|
|
377
|
+
# \\ (row terminator) outside braces
|
|
378
|
+
if i + 1 < n and body[i + 1] == "\\":
|
|
379
|
+
if depth == 0:
|
|
380
|
+
rows.append(_row_node(body, seg_start, i + 2))
|
|
381
|
+
seg_start = i + 2
|
|
382
|
+
i += 2
|
|
383
|
+
continue
|
|
384
|
+
i += 1
|
|
385
|
+
continue
|
|
386
|
+
if c == "%":
|
|
387
|
+
j = body.find("\n", i)
|
|
388
|
+
i = n if j < 0 else j + 1
|
|
389
|
+
continue
|
|
390
|
+
if c == "{":
|
|
391
|
+
depth += 1
|
|
392
|
+
elif c == "}":
|
|
393
|
+
depth = max(0, -1 + depth) if depth else 0
|
|
394
|
+
i += 1
|
|
395
|
+
if seg_start < n:
|
|
396
|
+
rows.append(_row_node(body, seg_start, n))
|
|
397
|
+
return rows
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _row_node(body: str, start: int, end: int) -> Node:
|
|
401
|
+
"""Build one row node from a body slice; content-based signature."""
|
|
402
|
+
text = body[start:end]
|
|
403
|
+
return Node(
|
|
404
|
+
kind="row",
|
|
405
|
+
text=text,
|
|
406
|
+
name=_row_key(text),
|
|
407
|
+
atom=True,
|
|
408
|
+
)
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
# pure table structure: makes or breaks nothing visually by itself and
|
|
412
|
+
# code-generators freely move it across row boundaries (\hline before
|
|
413
|
+
# vs after a row); row signatures must ignore it or rows flip between
|
|
414
|
+
# "deleted" and "added" wholesale
|
|
415
|
+
_STRUCT_TOKEN_RE = re.compile(
|
|
416
|
+
r"\\(?:hline|hdashline|toprule|midrule|bottomrule"
|
|
417
|
+
r"|endfirsthead|endhead|endfoot|endlastfoot"
|
|
418
|
+
r"|cline\s*\{[^{}]*\})"
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def _row_key(text: str) -> str:
|
|
423
|
+
"""Content-based signature of a table row (ordering by content).
|
|
424
|
+
|
|
425
|
+
Structural tokens (``\\\\hline``, ``\\\\``, ``\\\\endhead`` ...)
|
|
426
|
+
are stripped first: two generators may emit ``row \\\\ \\hline``
|
|
427
|
+
and ``\\hline row \\\\`` for the same logical row, and those must
|
|
428
|
+
compare equal or the whole table degrades into delete+add runs.
|
|
429
|
+
"""
|
|
430
|
+
stripped = re.sub(r"%[^\n]*", "", text)
|
|
431
|
+
stripped = _STRUCT_TOKEN_RE.sub(" ", stripped)
|
|
432
|
+
stripped = stripped.replace("\\\\", " ")
|
|
433
|
+
return re.sub(r"\s+", " ", stripped).strip() or "\x00empty"
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def _env_body_start(node: LatexEnvironmentNode, source: str) -> int:
|
|
437
|
+
"""Offset just past ``\\begin{env}`` plus its arguments.
|
|
438
|
+
|
|
439
|
+
Skips the optional ``[...]`` AND the mandatory ``{colspec}`` group
|
|
440
|
+
(brace-depth aware: ``{|W{.25}|W{.07}|}`` nests one level): the
|
|
441
|
+
column specification belongs to the table structure, never to a
|
|
442
|
+
row - markup between ``\\begin{env}`` and its colspec provokes
|
|
443
|
+
``Illegal pream-token`` in the array package.
|
|
444
|
+
"""
|
|
445
|
+
m = re.match(r"\s*\\begin\{[a-zA-Z*]+\}", source[node.pos :])
|
|
446
|
+
if not m: # pragma: no cover - parse guarantees this exists
|
|
447
|
+
return node.pos + 1
|
|
448
|
+
i = node.pos + m.end()
|
|
449
|
+
rest = source[i:]
|
|
450
|
+
m_opt = re.match(r"\s*\[[^\]]*\]", rest)
|
|
451
|
+
if m_opt:
|
|
452
|
+
i += m_opt.end()
|
|
453
|
+
rest = source[i:]
|
|
454
|
+
m_open = re.match(r"\s*\{", rest)
|
|
455
|
+
if m_open:
|
|
456
|
+
start = i + m_open.end()
|
|
457
|
+
depth = 1
|
|
458
|
+
j = start
|
|
459
|
+
while j < len(source) and depth > 0:
|
|
460
|
+
c = source[j]
|
|
461
|
+
if c == "{":
|
|
462
|
+
depth += 1
|
|
463
|
+
elif c == "}":
|
|
464
|
+
depth -= 1
|
|
465
|
+
elif c == "%": # pragma: no cover - degenerate colspec
|
|
466
|
+
nl = source.find("\n", j)
|
|
467
|
+
j = len(source) if nl < 0 else nl
|
|
468
|
+
continue
|
|
469
|
+
j += 1
|
|
470
|
+
return j
|
|
471
|
+
return i
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def _env_body_end(node: LatexEnvironmentNode, source: str) -> int:
|
|
475
|
+
"""Offset of the ``\\end{env}`` (last occurrence)."""
|
|
476
|
+
end = f"\\end{{{node.envname}}}"
|
|
477
|
+
idx = source.rfind(end, node.pos, node.pos + node.len)
|
|
478
|
+
if idx < 0: # pragma: no cover - round-trip check would catch it
|
|
479
|
+
return node.pos + node.len
|
|
480
|
+
return idx
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def _span(node: LatexNode, source: str) -> str:
|
|
484
|
+
"""Exact source span of a pylatexenc node, including delimiters."""
|
|
485
|
+
return source[node.pos : node.pos + node.len]
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _has_error_nodes(nodelist: list[LatexNode], source: str) -> bool:
|
|
489
|
+
"""True if tolerant parsing recovered over malformed input.
|
|
490
|
+
|
|
491
|
+
pylatexenc's tolerant mode inserts error placeholders in some
|
|
492
|
+
cases; the one it *swallows silently* is an unclosed group: the
|
|
493
|
+
returned node claims closing-delimiter ``}`` but the span does not
|
|
494
|
+
contain it. Detect both.
|
|
495
|
+
"""
|
|
496
|
+
for node in _walk_pylatexenc(nodelist):
|
|
497
|
+
cls = type(node).__name__
|
|
498
|
+
if cls == "LatexGroupNodeWithError" or getattr(node, "is_error_node", False):
|
|
499
|
+
return True
|
|
500
|
+
delimiters = getattr(node, "delimiters", None)
|
|
501
|
+
if delimiters and len(delimiters) == 2:
|
|
502
|
+
closing = delimiters[1]
|
|
503
|
+
if closing:
|
|
504
|
+
body = source[node.pos : node.pos + node.len]
|
|
505
|
+
# a real closed group ends with its closing delimiter
|
|
506
|
+
# inside the span (up to trailing whitespace/comments);
|
|
507
|
+
# an unclosed one claims a '}' that is not there
|
|
508
|
+
if not body.rstrip().endswith(closing):
|
|
509
|
+
return True
|
|
510
|
+
if isinstance(node, LatexEnvironmentNode):
|
|
511
|
+
end = f"\\end{{{node.envname}}}"
|
|
512
|
+
body = source[node.pos : node.pos + node.len]
|
|
513
|
+
if end not in body:
|
|
514
|
+
# tolerant parsing produced an environment node whose
|
|
515
|
+
# closing \end is missing - malformed input
|
|
516
|
+
return True
|
|
517
|
+
return False
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def _walk_pylatexenc(nodelist: list[LatexNode]):
|
|
521
|
+
for node in nodelist:
|
|
522
|
+
yield node
|
|
523
|
+
for child in getattr(node, "nodelist", []) or []:
|
|
524
|
+
yield from _walk_pylatexenc([child])
|