texdiff 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
texdiff/parse.py ADDED
@@ -0,0 +1,524 @@
1
+ """Parser wrapper: LaTeX source → list of :class:`texdiff.nodes.Node`.
2
+
3
+ Built on :mod:`pylatexenc.latexwalker`. This module owns all knowledge
4
+ about pylatexenc: the rest of texdiff works only on our own Node type.
5
+
6
+ Design decisions (v0):
7
+
8
+ * ``\\input``/``\\include`` are NOT resolved here yet - v0 expects the
9
+ caller to compare flattened sources or single files (flattening via
10
+ latexpand stays an external step); see docs/roadmap.
11
+ * Math, verbatim environments, comments and unknown macros become
12
+ ATOMIC nodes: exact-match comparison, never diffed inside.
13
+ * Known text-bearing environments and groups get children so the
14
+ aligner can recurse into them.
15
+ * Source positions recorded by pylatexenc are dropped: ``Node.text``
16
+ carries the exact span already.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import re
22
+ from dataclasses import replace
23
+ from pathlib import Path
24
+
25
+ from pylatexenc.latexwalker import (
26
+ LatexCommentNode,
27
+ LatexEnvironmentNode,
28
+ LatexGroupNode,
29
+ LatexMacroNode,
30
+ LatexCharsNode,
31
+ LatexMathNode,
32
+ LatexSpecialsNode,
33
+ LatexWalker,
34
+ get_default_latex_context_db,
35
+ )
36
+ from pylatexenc.macrospec import (
37
+ EnvironmentSpec,
38
+ MacroStandardArgsParser,
39
+ ParsedMacroArgs,
40
+ ParsedVerbatimArgs,
41
+ )
42
+
43
+ from .nodes import Node
44
+
45
+ # Environments whose body is text-like: the aligner may recurse into
46
+ # them (marking up changes inside paragraphs). Everything else -
47
+ # verbatim (lstlisting, verbatim, minted...), math displays, tikz
48
+ # pictures - is an atomic block by default.
49
+ TEXT_ENVIRONMENTS = frozenset(
50
+ {
51
+ "document",
52
+ "itemize",
53
+ "enumerate",
54
+ "description",
55
+ "quote",
56
+ "quotation",
57
+ "center",
58
+ "flushleft",
59
+ "flushright",
60
+ }
61
+ )
62
+
63
+ # Environments that must NEVER be diffed internally.
64
+ VERBATIM_ENVIRONMENTS = frozenset(
65
+ {
66
+ "verbatim",
67
+ "verbatim*",
68
+ "lstlisting",
69
+ "minted",
70
+ "alltt",
71
+ "filecontents",
72
+ "filecontents*",
73
+ }
74
+ )
75
+
76
+ # Table environments: atomic in v0 (whole-block replace when their
77
+ # structure changes). Row-granular alignment is a v1 roadmap item.
78
+ TABLE_ENVIRONMENTS = frozenset(
79
+ {"tabular", "tabular*", "tabularx", "longtable", "longtable*", "array"}
80
+ )
81
+
82
+
83
+ class _NamedVerbatimArgsParser(MacroStandardArgsParser):
84
+ """Verbatim args parser for named environments.
85
+
86
+ pylatexenc's :class:`VerbatimArgsParser` hardcodes the terminator
87
+ ``\\end{verbatim}`` — registering ``lstlisting`` (or any other
88
+ named verbatim environment) with it makes the walker swallow
89
+ everything up to a *literal* ``\\end{verbatim}`` far later in the
90
+ document, merging dozens of unrelated listings and the prose
91
+ between them into one giant node. This subclass scans for the
92
+ environment's own ``\\end{<envname>}``, exactly as LaTeX does.
93
+ """
94
+
95
+ def __init__(self, envname: str, **kwargs):
96
+ super().__init__(argspec="{", **kwargs)
97
+ self.envname = envname
98
+
99
+ def parse_args(self, w, pos, parsing_state=None):
100
+ from pylatexenc import latexwalker
101
+
102
+ endtoken = f"\\end{{{self.envname}}}"
103
+ endpos = w.s.find(endtoken, pos)
104
+ if endpos == -1:
105
+ raise latexwalker.LatexWalkerParseError(
106
+ s=w.s,
107
+ pos=pos,
108
+ msg=f"Cannot find matching {endtoken}",
109
+ )
110
+ len_ = endpos - pos
111
+ argd = ParsedVerbatimArgs(
112
+ verbatim_chars_node=w.make_node(
113
+ latexwalker.LatexCharsNode,
114
+ parsing_state=parsing_state,
115
+ chars=w.s[pos : pos + len_],
116
+ pos=pos,
117
+ len=len_,
118
+ )
119
+ )
120
+ return (argd, pos, len_)
121
+
122
+
123
+ def _latex_context():
124
+ """Default pylatexenc context extended with verbatim environments.
125
+
126
+ pylatexenc's default specs only know ``verbatim`` as a verbatim
127
+ environment; ``lstlisting``, ``minted``, ``alltt`` and friends are
128
+ parsed as *regular* environments, so code listings containing
129
+ unbalanced braces (C++ ``std::tuple<int, ...>`` constructors,
130
+ ``#include <x>`` in ``alltt``...) make strict parsing fail and
131
+ tolerant parsing produce garbage error nodes. Registering them
132
+ with a named verbatim args parser makes the walker read their
133
+ bodies verbatim, exactly as LaTeX does.
134
+ """
135
+ db = get_default_latex_context_db()
136
+ db.add_context_category(
137
+ "texdiff-verbatim",
138
+ macros=[],
139
+ environments=[
140
+ EnvironmentSpec(
141
+ envname,
142
+ args_parser=_NamedVerbatimArgsParser(envname),
143
+ )
144
+ for envname in VERBATIM_ENVIRONMENTS
145
+ ],
146
+ specials=[],
147
+ prepend=True,
148
+ )
149
+ return db
150
+
151
+
152
+ _LATEX_CONTEXT = _latex_context()
153
+
154
+
155
+ class ParseError(ValueError):
156
+ """Raised when LaTeX source cannot be parsed even tolerantly."""
157
+
158
+
159
+ def parse(source: str) -> list[Node]:
160
+ """Parse LaTeX source text into texdiff nodes.
161
+
162
+ Args:
163
+ source: LaTeX source of one file (already flattened if the
164
+ document is spread over several files).
165
+
166
+ Returns:
167
+ Top-level node list. Concatenating the ``text`` of all returned
168
+ nodes reproduces ``source`` exactly.
169
+
170
+ Raises:
171
+ ParseError: if the source has unbalanced group delimiters or
172
+ cannot be parsed even tolerantly.
173
+
174
+ Policy: parse strictly if possible (catches most malformed input),
175
+ fall back to *tolerant* parsing when strict mode rejects a construct
176
+ real documents use (e.g. ``\\makecell`` bodies with ``\\\\``). In
177
+ tolerant mode a balance check over the resulting edit script keeps
178
+ unbalanced braces out (pylatexexenc would silently swallow them,
179
+ which must not reach the emitter).
180
+ """
181
+ nodes: list[Node]
182
+ try:
183
+ nodelist, _pos, _len = LatexWalker(
184
+ source, latex_context=_LATEX_CONTEXT, tolerant_parsing=False
185
+ ).get_latex_nodes()
186
+ converted = [_convert(n, source) for n in nodelist]
187
+ except Exception:
188
+ # strict mode failed: real-world construct or genuinely broken?
189
+ try:
190
+ nodelist, _pos, length = LatexWalker(
191
+ source, latex_context=_LATEX_CONTEXT, tolerant_parsing=True
192
+ ).get_latex_nodes()
193
+ except Exception as exc:
194
+ raise ParseError(f"cannot parse LaTeX source: {exc}") from exc
195
+ if length == len(source) and _has_error_nodes(nodelist, source):
196
+ raise ParseError(
197
+ "LaTeX source is malformed (tolerant parsing recovered, "
198
+ "but with error tokens)"
199
+ )
200
+ converted = [_convert(n, source) for n in nodelist]
201
+
202
+ if "".join(n.text for n in converted) != source:
203
+ # round-trip broken: reject rather than emit wrong bytes
204
+ raise ParseError("parse round-trip mismatch (node spans do not cover source)")
205
+ return _fold_escapes(converted)
206
+
207
+
208
+ # --- escape folding -----------------------------------------------------------
209
+
210
+ # Escape macros that typeset a literal character (``\_`` `` \& `` ...
211
+ # ) carry no syntactic meaning: their only effect in prose is to
212
+ # create a node boundary. When one revision writes ``ABCDE`` as one
213
+ # text run and the other writes ``ABCD\_-E`` (the identical string
214
+ # with an escaped underscore), the node lists disagree on
215
+ # granularity: the aligner pairs the whole old run with just the
216
+ # fragment before ``\_``, and the word diff then strikes out the
217
+ # entire unchanged tail of the old run ("file type. When the list
218
+ # ..."). Folding both into plain-text runs with identical structure
219
+ # restores granularity symmetry; the exact source text is kept in
220
+ # each node, so a folded span matching both sides round-trips
221
+ # verbatim.
222
+ _ESCAPE_MACRO_RE = re.compile(
223
+ r"\\(?:_|&|%|#|\$)\s*(?![a-zA-Z])" # \_ \& \% \# \$ not part of \_cmd
224
+ )
225
+
226
+
227
+ def _is_glue_macro(node: Node) -> bool:
228
+ """True if node is an escape macro like ``\\_`` (no argument)."""
229
+ return (
230
+ node.kind == "macro"
231
+ and node.name in ("_", "&", "%", "#", "$")
232
+ and node.text.strip() == f"\\{node.name}"
233
+ )
234
+
235
+
236
+ def _fold_escapes(nodes: list[Node]) -> list[Node]:
237
+ """Merge escape macros into an adjacent text node.
238
+
239
+ Recurses into recursable children. An escape macro between two
240
+ text nodes merges left (prose continuation); a leading escape
241
+ macro merges right. Standalone escapes (surrounded by structure)
242
+ stay untouched - text equality still round-trips them.
243
+ """
244
+ out: list[Node] = []
245
+ for node in nodes:
246
+ if node.children:
247
+ node = replace(node, children=_fold_escapes(node.children))
248
+ if _is_glue_macro(node) and out and out[-1].kind == "text":
249
+ out[-1] = replace(out[-1], text=out[-1].text + node.text)
250
+ continue
251
+ if (
252
+ _is_glue_macro(node)
253
+ and not out
254
+ and len(nodes) > 1
255
+ and nodes[1].kind == "text"
256
+ ):
257
+ # leading escape with text following: remember, merge right
258
+ out.append(node)
259
+ continue
260
+ if node.kind == "text" and out and _is_glue_macro(out[-1]):
261
+ out[-1] = replace(out[-1], text=out[-1].text + node.text)
262
+ continue
263
+ if node.kind == "text" and out and out[-1].kind == "text":
264
+ # consecutive text runs (an artefact of escape folding or
265
+ # of comments/verbatim dropped between them in earlier
266
+ # processing): merge so both revisions present prose with
267
+ # identical granularity - the aligner pairs like with like
268
+ out[-1] = replace(out[-1], text=out[-1].text + node.text)
269
+ continue
270
+ out.append(node)
271
+ return out
272
+
273
+
274
+ def parse_file(path: str | Path) -> list[Node]:
275
+ """Read and parse a ``.tex`` file (UTF-8)."""
276
+ return parse(Path(path).read_text(encoding="utf-8"))
277
+
278
+
279
+ # --- pylatexenc → texdiff conversion ---------------------------------------
280
+
281
+
282
+ def _convert(node: LatexNode, source: str) -> Node:
283
+ """Convert one pylatexenc node (recursively) to a texdiff Node."""
284
+ text = _span(node, source)
285
+
286
+ if isinstance(node, LatexCharsNode):
287
+ return Node(kind="text", text=text, atom=True)
288
+
289
+ if isinstance(node, LatexCommentNode):
290
+ return Node(kind="comment", text=text, atom=True)
291
+
292
+ if isinstance(node, LatexMathNode):
293
+ return Node(kind="math", text=text, atom=True)
294
+
295
+ if isinstance(node, LatexSpecialsNode):
296
+ return Node(kind="specials", text=text, atom=True)
297
+
298
+ if isinstance(node, LatexGroupNode):
299
+ return Node(
300
+ kind="group",
301
+ text=text,
302
+ children=[_convert(c, source) for c in node.nodelist],
303
+ atom=False,
304
+ )
305
+
306
+ if isinstance(node, LatexMacroNode):
307
+ # macros stay atomic in v0; recursable text macros (\textbf,
308
+ # \makecell, \emph, ...) are a v1 configuration item
309
+ return Node(kind="macro", text=text, name=node.macroname, atom=True)
310
+
311
+ if isinstance(node, LatexEnvironmentNode):
312
+ envname = node.envname
313
+ if envname in VERBATIM_ENVIRONMENTS:
314
+ return Node(kind="env", text=text, name=envname, atom=True)
315
+ if envname in TABLE_ENVIRONMENTS:
316
+ return _table_env_node(node, source, envname)
317
+ recursable = envname in TEXT_ENVIRONMENTS
318
+ return Node(
319
+ kind="env",
320
+ text=text,
321
+ name=envname,
322
+ atom=not recursable,
323
+ children=[_convert(c, source) for c in node.nodelist] if recursable else [],
324
+ )
325
+
326
+ # unknown pylatexenc node class: keep as an opaque atomic block
327
+ return Node(kind="specials", text=text, atom=True) # pragma: no cover
328
+
329
+
330
+ # --- table ↔ row nodes -------------------------------------------------------
331
+
332
+
333
+ def _table_env_node(node: LatexEnvironmentNode, source: str, envname: str) -> Node:
334
+ """Convert a table environment into a recursable env of row nodes.
335
+
336
+ Row splitting is brace-aware (``\\\\makecell{a\\\\\\\\b}`` is one row)
337
+ and turns ``\\\\endfirsthead``/``\\\\endhead``/``\\\\endfoot``/``\\\\endlastfoot``
338
+ boundaries into row-kind segments, so the aligner can keep header
339
+ and body regions separate.
340
+ """
341
+ children = _split_table_body(node, source)
342
+ return Node(
343
+ kind="env",
344
+ text=source[node.pos : node.pos + node.len],
345
+ name=envname,
346
+ children=children,
347
+ atom=False,
348
+ )
349
+
350
+
351
+ def _split_table_body(node: LatexEnvironmentNode, source: str) -> list[Node]:
352
+ """Split a table body (source span) into row nodes.
353
+
354
+ A *row* ends at a ``\\\\`` which sits outside braces and outside
355
+ comments. Everything between the environment's begin/end markup
356
+ is split at those points; each segment (including its trailing
357
+ ``\\\\`` and line breaks) becomes one atomic row node whose
358
+ signature is its normalized content.
359
+ """
360
+ body_start = _env_body_start(node, source)
361
+ body_end = _env_body_end(node, source)
362
+ body = source[body_start:body_end]
363
+
364
+ rows: list[Node] = []
365
+ seg_start = 0
366
+ depth = 0
367
+ i = 0
368
+ n = len(body)
369
+ while i < n:
370
+ c = body[i]
371
+ if c == "\\":
372
+ # comment: skip to end of line
373
+ if i + 1 < n and body[i + 1] == "%":
374
+ j = body.find("\n", i)
375
+ i = n if j < 0 else j + 1
376
+ continue
377
+ # \\ (row terminator) outside braces
378
+ if i + 1 < n and body[i + 1] == "\\":
379
+ if depth == 0:
380
+ rows.append(_row_node(body, seg_start, i + 2))
381
+ seg_start = i + 2
382
+ i += 2
383
+ continue
384
+ i += 1
385
+ continue
386
+ if c == "%":
387
+ j = body.find("\n", i)
388
+ i = n if j < 0 else j + 1
389
+ continue
390
+ if c == "{":
391
+ depth += 1
392
+ elif c == "}":
393
+ depth = max(0, -1 + depth) if depth else 0
394
+ i += 1
395
+ if seg_start < n:
396
+ rows.append(_row_node(body, seg_start, n))
397
+ return rows
398
+
399
+
400
+ def _row_node(body: str, start: int, end: int) -> Node:
401
+ """Build one row node from a body slice; content-based signature."""
402
+ text = body[start:end]
403
+ return Node(
404
+ kind="row",
405
+ text=text,
406
+ name=_row_key(text),
407
+ atom=True,
408
+ )
409
+
410
+
411
+ # pure table structure: makes or breaks nothing visually by itself and
412
+ # code-generators freely move it across row boundaries (\hline before
413
+ # vs after a row); row signatures must ignore it or rows flip between
414
+ # "deleted" and "added" wholesale
415
+ _STRUCT_TOKEN_RE = re.compile(
416
+ r"\\(?:hline|hdashline|toprule|midrule|bottomrule"
417
+ r"|endfirsthead|endhead|endfoot|endlastfoot"
418
+ r"|cline\s*\{[^{}]*\})"
419
+ )
420
+
421
+
422
+ def _row_key(text: str) -> str:
423
+ """Content-based signature of a table row (ordering by content).
424
+
425
+ Structural tokens (``\\\\hline``, ``\\\\``, ``\\\\endhead`` ...)
426
+ are stripped first: two generators may emit ``row \\\\ \\hline``
427
+ and ``\\hline row \\\\`` for the same logical row, and those must
428
+ compare equal or the whole table degrades into delete+add runs.
429
+ """
430
+ stripped = re.sub(r"%[^\n]*", "", text)
431
+ stripped = _STRUCT_TOKEN_RE.sub(" ", stripped)
432
+ stripped = stripped.replace("\\\\", " ")
433
+ return re.sub(r"\s+", " ", stripped).strip() or "\x00empty"
434
+
435
+
436
+ def _env_body_start(node: LatexEnvironmentNode, source: str) -> int:
437
+ """Offset just past ``\\begin{env}`` plus its arguments.
438
+
439
+ Skips the optional ``[...]`` AND the mandatory ``{colspec}`` group
440
+ (brace-depth aware: ``{|W{.25}|W{.07}|}`` nests one level): the
441
+ column specification belongs to the table structure, never to a
442
+ row - markup between ``\\begin{env}`` and its colspec provokes
443
+ ``Illegal pream-token`` in the array package.
444
+ """
445
+ m = re.match(r"\s*\\begin\{[a-zA-Z*]+\}", source[node.pos :])
446
+ if not m: # pragma: no cover - parse guarantees this exists
447
+ return node.pos + 1
448
+ i = node.pos + m.end()
449
+ rest = source[i:]
450
+ m_opt = re.match(r"\s*\[[^\]]*\]", rest)
451
+ if m_opt:
452
+ i += m_opt.end()
453
+ rest = source[i:]
454
+ m_open = re.match(r"\s*\{", rest)
455
+ if m_open:
456
+ start = i + m_open.end()
457
+ depth = 1
458
+ j = start
459
+ while j < len(source) and depth > 0:
460
+ c = source[j]
461
+ if c == "{":
462
+ depth += 1
463
+ elif c == "}":
464
+ depth -= 1
465
+ elif c == "%": # pragma: no cover - degenerate colspec
466
+ nl = source.find("\n", j)
467
+ j = len(source) if nl < 0 else nl
468
+ continue
469
+ j += 1
470
+ return j
471
+ return i
472
+
473
+
474
+ def _env_body_end(node: LatexEnvironmentNode, source: str) -> int:
475
+ """Offset of the ``\\end{env}`` (last occurrence)."""
476
+ end = f"\\end{{{node.envname}}}"
477
+ idx = source.rfind(end, node.pos, node.pos + node.len)
478
+ if idx < 0: # pragma: no cover - round-trip check would catch it
479
+ return node.pos + node.len
480
+ return idx
481
+
482
+
483
+ def _span(node: LatexNode, source: str) -> str:
484
+ """Exact source span of a pylatexenc node, including delimiters."""
485
+ return source[node.pos : node.pos + node.len]
486
+
487
+
488
+ def _has_error_nodes(nodelist: list[LatexNode], source: str) -> bool:
489
+ """True if tolerant parsing recovered over malformed input.
490
+
491
+ pylatexenc's tolerant mode inserts error placeholders in some
492
+ cases; the one it *swallows silently* is an unclosed group: the
493
+ returned node claims closing-delimiter ``}`` but the span does not
494
+ contain it. Detect both.
495
+ """
496
+ for node in _walk_pylatexenc(nodelist):
497
+ cls = type(node).__name__
498
+ if cls == "LatexGroupNodeWithError" or getattr(node, "is_error_node", False):
499
+ return True
500
+ delimiters = getattr(node, "delimiters", None)
501
+ if delimiters and len(delimiters) == 2:
502
+ closing = delimiters[1]
503
+ if closing:
504
+ body = source[node.pos : node.pos + node.len]
505
+ # a real closed group ends with its closing delimiter
506
+ # inside the span (up to trailing whitespace/comments);
507
+ # an unclosed one claims a '}' that is not there
508
+ if not body.rstrip().endswith(closing):
509
+ return True
510
+ if isinstance(node, LatexEnvironmentNode):
511
+ end = f"\\end{{{node.envname}}}"
512
+ body = source[node.pos : node.pos + node.len]
513
+ if end not in body:
514
+ # tolerant parsing produced an environment node whose
515
+ # closing \end is missing - malformed input
516
+ return True
517
+ return False
518
+
519
+
520
+ def _walk_pylatexenc(nodelist: list[LatexNode]):
521
+ for node in nodelist:
522
+ yield node
523
+ for child in getattr(node, "nodelist", []) or []:
524
+ yield from _walk_pylatexenc([child])