texdiff 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- texdiff/__init__.py +45 -0
- texdiff/align.py +202 -0
- texdiff/api.py +326 -0
- texdiff/check.py +61 -0
- texdiff/cli.py +87 -0
- texdiff/emit.py +1193 -0
- texdiff/flatten.py +185 -0
- texdiff/nodes.py +147 -0
- texdiff/oldlines.py +49 -0
- texdiff/parse.py +524 -0
- texdiff/preamble.py +388 -0
- texdiff/tables.py +540 -0
- texdiff/textdiff.py +157 -0
- texdiff-0.2.0.dist-info/METADATA +112 -0
- texdiff-0.2.0.dist-info/RECORD +19 -0
- texdiff-0.2.0.dist-info/WHEEL +5 -0
- texdiff-0.2.0.dist-info/entry_points.txt +2 -0
- texdiff-0.2.0.dist-info/licenses/LICENSE +21 -0
- texdiff-0.2.0.dist-info/top_level.txt +1 -0
texdiff/flatten.py
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""\\input/\\include expansion: multi-file document → single source.
|
|
2
|
+
|
|
3
|
+
The equivalent of ``latexdiff --flatten`` / ``latexpand``, inside the
|
|
4
|
+
tool. texdiff parses per file, then expands references so the aligner
|
|
5
|
+
sees one document tree.
|
|
6
|
+
|
|
7
|
+
Design:
|
|
8
|
+
|
|
9
|
+
* textual (linear) expansion, matching latexpand's contract: no macro
|
|
10
|
+
semantics, just file inclusion - what ``\\input`` does in TeX;
|
|
11
|
+
* files are resolved relative to the INCLUDING file's directory
|
|
12
|
+
(TeX's kpathsea semantics simplified: the main file's dir, then the
|
|
13
|
+
including chain);
|
|
14
|
+
* cycle protection: a file already on the expansion stack is kept
|
|
15
|
+
verbatim instead of being recursed into (TeX would loop forever);
|
|
16
|
+
* missing files are kept verbatim rather than fatal - a broken
|
|
17
|
+
``\\input`` must not block the diff, and keeping the reference makes
|
|
18
|
+
the output still compilable;
|
|
19
|
+
* comment lines and verbatim environments are skipped: a commented
|
|
20
|
+
``% \\input{x}`` is not an inclusion, and verbatim bodies must never
|
|
21
|
+
be expanded.
|
|
22
|
+
|
|
23
|
+
The emitter later re-uses the skip logic for the same purpose (never
|
|
24
|
+
mark up inside verbatim).
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import re
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
# \input{name} / \include{name}; name may lack the .tex extension
|
|
33
|
+
# (TeX appends it), may contain letters, /, -, _, . and spaces
|
|
34
|
+
_INPUT_RE = re.compile(r"\\(input|include)\*?\s*\{([^{}]+)\}")
|
|
35
|
+
|
|
36
|
+
# block of lines which must never be expanded: verbatim-like envs
|
|
37
|
+
_VERBATIM_ENVS = (
|
|
38
|
+
"verbatim",
|
|
39
|
+
"verbatim*",
|
|
40
|
+
"lstlisting",
|
|
41
|
+
"minted",
|
|
42
|
+
"alltt",
|
|
43
|
+
"filecontents",
|
|
44
|
+
"filecontents*",
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
_MARKER_PREFIX = "% texdiff-flatten:"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class Flattener:
|
|
51
|
+
"""Expand ``\\input``/``\\include`` in a source string.
|
|
52
|
+
|
|
53
|
+
Attributes:
|
|
54
|
+
base_dir: directory files are resolved against.
|
|
55
|
+
expanded: number of references successfully expanded.
|
|
56
|
+
missing: number of references whose file did not exist
|
|
57
|
+
(kept verbatim).
|
|
58
|
+
markers: insert ``% texdiff-flatten:`` annotation comments
|
|
59
|
+
around expanded content (for debugging / provenance).
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
def __init__(self, base_dir: Path | str | None = None, markers: bool = False):
|
|
63
|
+
self.base_dir = Path(base_dir) if base_dir is not None else Path.cwd()
|
|
64
|
+
self.markers = markers
|
|
65
|
+
self.expanded = 0
|
|
66
|
+
self.missing = 0
|
|
67
|
+
|
|
68
|
+
def flatten(self, source: str) -> str:
|
|
69
|
+
"""Expand all \\input/\\include references in ``source``."""
|
|
70
|
+
return self._expand(source, self.base_dir, stack=frozenset())
|
|
71
|
+
|
|
72
|
+
def _expand(self, source: str, current_dir: Path, stack: frozenset[Path]) -> str:
|
|
73
|
+
out: list[str] = []
|
|
74
|
+
pos = 0
|
|
75
|
+
for m in self._iter_skipping(source):
|
|
76
|
+
out.append(source[pos : m.start()])
|
|
77
|
+
cmd, name = m.group(1), m.group(2)
|
|
78
|
+
target = self._resolve(name, current_dir)
|
|
79
|
+
if target is None:
|
|
80
|
+
self.missing += 1
|
|
81
|
+
out.append(m.group(0))
|
|
82
|
+
else:
|
|
83
|
+
resolved = target.resolve()
|
|
84
|
+
if resolved in stack:
|
|
85
|
+
# cycle: keep verbatim, do not recurse
|
|
86
|
+
out.append(m.group(0))
|
|
87
|
+
self.expanded += 1
|
|
88
|
+
pos = m.end()
|
|
89
|
+
continue
|
|
90
|
+
content = target.read_text(encoding="utf-8", errors="replace")
|
|
91
|
+
beginning = len(out)
|
|
92
|
+
if self.markers:
|
|
93
|
+
out.append(f"{_MARKER_PREFIX} begin {target.name}\n")
|
|
94
|
+
out.append(self._expand(content, resolved.parent, stack | {resolved}))
|
|
95
|
+
if self.markers:
|
|
96
|
+
out.append(f"\n{_MARKER_PREFIX} end {target.name}\n")
|
|
97
|
+
self.expanded += 1
|
|
98
|
+
del beginning
|
|
99
|
+
pos = m.end()
|
|
100
|
+
out.append(source[pos:])
|
|
101
|
+
return "".join(out)
|
|
102
|
+
|
|
103
|
+
def _resolve(self, name: str, current_dir: Path) -> Path | None:
|
|
104
|
+
"""Resolve an \\input name to a file, TeX-style (with .tex)."""
|
|
105
|
+
candidate = name.strip()
|
|
106
|
+
if candidate.startswith("/"):
|
|
107
|
+
# absolute: as-is (with .tex fallback)
|
|
108
|
+
p = Path(candidate)
|
|
109
|
+
return p if p.is_file() else self._with_tex(p)
|
|
110
|
+
|
|
111
|
+
for base in (current_dir, self.base_dir):
|
|
112
|
+
p = base / candidate
|
|
113
|
+
if p.is_file():
|
|
114
|
+
return p
|
|
115
|
+
p = p.with_suffix(".tex")
|
|
116
|
+
if p.is_file():
|
|
117
|
+
with_tex = base / f"{candidate}.tex"
|
|
118
|
+
if with_tex.is_file():
|
|
119
|
+
return with_tex
|
|
120
|
+
return None
|
|
121
|
+
|
|
122
|
+
def _with_tex(self, p: Path) -> Path | None:
|
|
123
|
+
q = p.with_suffix(".tex")
|
|
124
|
+
return p if p.is_file() else (q if q.is_file() else None)
|
|
125
|
+
|
|
126
|
+
def _iter_skipping(self, source: str) -> list[re.Match]:
|
|
127
|
+
"""Matches of _INPUT_RE outside comments and verbatim bodies."""
|
|
128
|
+
results: list[re.Match] = []
|
|
129
|
+
i = 0
|
|
130
|
+
n = len(source)
|
|
131
|
+
while i < n:
|
|
132
|
+
# comment: up to end of line (comment includes it)
|
|
133
|
+
if source[i] == "%":
|
|
134
|
+
nl = source.find("\n", i)
|
|
135
|
+
i = n if nl < 0 else nl + 1
|
|
136
|
+
continue
|
|
137
|
+
# verbatim-like environment: skip whole body
|
|
138
|
+
m_verb = re.compile(
|
|
139
|
+
rf"\\begin\{{({'|'.join(re.escape(e) for e in _VERBATIM_ENVS)})\}}"
|
|
140
|
+
).match(source, i)
|
|
141
|
+
if m_verb:
|
|
142
|
+
end = re.compile(rf"\\end\{{{m_verb.group(1)}\}}")
|
|
143
|
+
m_end = end.search(source, m_verb.end())
|
|
144
|
+
skip_to = m_end.end() if m_end else n
|
|
145
|
+
results_adjacent = _INPUT_RE.finditer(source, m_verb.end(), skip_to)
|
|
146
|
+
# inside verbatim: do not expand - skip entirely
|
|
147
|
+
i = skip_to
|
|
148
|
+
continue
|
|
149
|
+
m = _INPUT_RE.match(source, i)
|
|
150
|
+
if m:
|
|
151
|
+
results.append(m)
|
|
152
|
+
i = m.end()
|
|
153
|
+
continue
|
|
154
|
+
i += 1
|
|
155
|
+
return results
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def flatten_source(
|
|
159
|
+
source: str,
|
|
160
|
+
base_dir: Path | str | None = None,
|
|
161
|
+
markers: bool = False,
|
|
162
|
+
) -> str:
|
|
163
|
+
"""Expand \\input/\\include in a source string (base_dir=cwd)."""
|
|
164
|
+
return Flattener(base_dir=base_dir, markers=markers).flatten(source)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def flatten_file(
|
|
168
|
+
source_path: Path | str,
|
|
169
|
+
base_dir: Path | str | None = None,
|
|
170
|
+
markers: bool = False,
|
|
171
|
+
) -> str:
|
|
172
|
+
"""Read a file and expand all its \\input/\\include references.
|
|
173
|
+
|
|
174
|
+
Args:
|
|
175
|
+
source_path: the main .tex file.
|
|
176
|
+
base_dir: fallback directory for unresolvable-relative names
|
|
177
|
+
(default: the main file's directory).
|
|
178
|
+
markers: insert provenance comments around expansions.
|
|
179
|
+
|
|
180
|
+
Returns:
|
|
181
|
+
The flattened source text.
|
|
182
|
+
"""
|
|
183
|
+
sp = Path(source_path)
|
|
184
|
+
f = Flattener(base_dir=base_dir if base_dir is not None else sp.parent, markers=markers)
|
|
185
|
+
return f.flatten(sp.read_text(encoding="utf-8", errors="replace"))
|
texdiff/nodes.py
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Nodes - the intermediate representation produced by parsing.
|
|
2
|
+
|
|
3
|
+
texdiff never operates on pylatexenc node classes directly: we convert
|
|
4
|
+
them to our own lightweight :class:`Node` records first. Each node
|
|
5
|
+
|
|
6
|
+
* carries ``text`` - its EXACT source span, so unchanged nodes round
|
|
7
|
+
trip byte-identically (critical for listings, attachments and
|
|
8
|
+
generated tables);
|
|
9
|
+
* carries ``children`` - a list of sub-nodes (only for group-like
|
|
10
|
+
nodes: environments, braced groups, macro arguments);
|
|
11
|
+
* can be *atomic* (math, verbatim, unknown macros): no children, never
|
|
12
|
+
diffed internally, compared on exact source equality;
|
|
13
|
+
* knows whether it may carry change markup inside it (``atom`` False
|
|
14
|
+
only for text-ish nodes; markup is NEVER injected into atomic
|
|
15
|
+
nodes - that injector is what breaks markup in latexdiff).
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import re
|
|
21
|
+
from dataclasses import dataclass, field
|
|
22
|
+
from typing import Iterator, Optional
|
|
23
|
+
|
|
24
|
+
# sectioning macros whose {\title} becomes part of the signature
|
|
25
|
+
_SECTIONING_RE = re.compile(
|
|
26
|
+
r"\s*\\(?:chapter|section|subsection|subsubsection|paragraph|subparagraph)\*?\s*\{"
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class Node:
|
|
32
|
+
"""One node of the concrete syntax tree.
|
|
33
|
+
|
|
34
|
+
Attributes:
|
|
35
|
+
kind: node type discriminator, e.g. ``"env"`` (environment),
|
|
36
|
+
``"group"`` (braced group / macro argument), ``"macro"``,
|
|
37
|
+
``"text"`` (plain characters run), ``"math"`` (inline or
|
|
38
|
+
display math), ``"comment"``, ``"specials"``, ``"verb"``
|
|
39
|
+
(verbatim-like environment, atomic).
|
|
40
|
+
text: exact source text of the node INCLUDING delimiters
|
|
41
|
+
(``\\begin{...}...\\end{...}`` for environments, the braces
|
|
42
|
+
for groups). Unchanged nodes are emitted using this text
|
|
43
|
+
verbatim, so formatting survives the diff round trip.
|
|
44
|
+
name: environment or macro name (``None`` for text).
|
|
45
|
+
children: sub-nodes for group-like nodes; empty otherwise.
|
|
46
|
+
atom: if True the node is atomic - never diffed internally and
|
|
47
|
+
never receives inline markup (math, verbatim, unknown
|
|
48
|
+
macros, specials).
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
kind: str
|
|
52
|
+
text: str
|
|
53
|
+
name: Optional[str] = None
|
|
54
|
+
children: list[Node] = field(default_factory=list)
|
|
55
|
+
atom: bool = True
|
|
56
|
+
|
|
57
|
+
def walk(self) -> Iterator["Node"]:
|
|
58
|
+
"""Yield this node and all descendants, depth first."""
|
|
59
|
+
yield self
|
|
60
|
+
for child in self.children:
|
|
61
|
+
yield from child.walk()
|
|
62
|
+
|
|
63
|
+
def signature(self) -> str:
|
|
64
|
+
"""Return a canonical key used to align two nodes.
|
|
65
|
+
|
|
66
|
+
Two nodes are aligned candidate-wise when their signatures are
|
|
67
|
+
equal and their exact source differs (then we diff inside);
|
|
68
|
+
nodes with equal signature AND equal text are identical.
|
|
69
|
+
|
|
70
|
+
For group-like nodes the signature carries a *content anchor*:
|
|
71
|
+
the key of the first table row found inside (typically the
|
|
72
|
+
``\\caption`` row of a ``longtable``). Without it every
|
|
73
|
+
``{\\scriptsize \\begin{longtable}...}`` group would share the
|
|
74
|
+
bare signature ``group`` and a shifted positional match could
|
|
75
|
+
pair two *different* generated tables - whose column counts
|
|
76
|
+
then disagree and the marked-up document no longer compiles.
|
|
77
|
+
The anchor ties each table to its true counterpart as long as
|
|
78
|
+
its caption (or first row) is stable, which holds even when
|
|
79
|
+
every data row changed.
|
|
80
|
+
"""
|
|
81
|
+
if self.name and self.kind not in ("group", "env"):
|
|
82
|
+
if self.kind == "macro" and _SECTIONING_RE.match(self.text or ""):
|
|
83
|
+
# sectioning macros: carry the title. Without it every
|
|
84
|
+
# \subsubsection* shares one signature and a heading
|
|
85
|
+
# inserted mid-document shifts every later heading by
|
|
86
|
+
# one - the aligner then retires the old titles just
|
|
87
|
+
# to re-add them as insertions further down.
|
|
88
|
+
title = self._heading_title()
|
|
89
|
+
if title:
|
|
90
|
+
return f"{self.kind}:{self.name}:{title}"
|
|
91
|
+
return f"{self.kind}:{self.name}"
|
|
92
|
+
if self.kind in ("group", "env"):
|
|
93
|
+
base = f"{self.kind}:{self.name}" if self.name else self.kind
|
|
94
|
+
anchor = self._row_anchor()
|
|
95
|
+
if anchor:
|
|
96
|
+
return f"{base}:{anchor}"
|
|
97
|
+
return base
|
|
98
|
+
return self.kind
|
|
99
|
+
|
|
100
|
+
def _heading_title(self) -> str | None:
|
|
101
|
+
"""Normalised title argument of a sectioning macro, if any."""
|
|
102
|
+
m = re.match(r"\\[a-zA-Z]+\*?\s*(\[[^\]]*\])?\s*\{", self.text or "")
|
|
103
|
+
if not m:
|
|
104
|
+
return None
|
|
105
|
+
title = (self.text or "")[m.end() :] # after the opening brace
|
|
106
|
+
depth = 1
|
|
107
|
+
for i, ch in enumerate(title):
|
|
108
|
+
if ch == "{" and (i == 0 or title[i - 1] != "\\"):
|
|
109
|
+
depth += 1
|
|
110
|
+
elif ch == "}" and (i == 0 or title[i - 1] != "\\"):
|
|
111
|
+
depth -= 1
|
|
112
|
+
if depth == 0:
|
|
113
|
+
return re.sub(r"\s+", " ", title[:i]).strip() or None
|
|
114
|
+
return None
|
|
115
|
+
|
|
116
|
+
def _row_anchor(self) -> str | None:
|
|
117
|
+
"""Key of the first table row in this subtree, if any."""
|
|
118
|
+
for node in self.walk():
|
|
119
|
+
if node.kind == "row" and node.name:
|
|
120
|
+
return node.name[:64]
|
|
121
|
+
return None
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def text_node(s: str) -> Node:
|
|
125
|
+
"""Build a plain text run node."""
|
|
126
|
+
return Node(kind="text", text=s, atom=True)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def env(
|
|
130
|
+
name: str,
|
|
131
|
+
*children: Node,
|
|
132
|
+
text: Optional[str] = None,
|
|
133
|
+
atom: bool = False,
|
|
134
|
+
) -> Node:
|
|
135
|
+
"""Build an environment node (test helper / prototyping)."""
|
|
136
|
+
if text is None:
|
|
137
|
+
body = "".join(c.text for c in children)
|
|
138
|
+
text = f"\\begin{{{name}}}{body}\\end{{{name}}}"
|
|
139
|
+
return Node(kind="env", text=text, name=name, children=list(children), atom=atom)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def group(*children: Node, name: Optional[str] = None, text: Optional[str] = None) -> Node:
|
|
143
|
+
"""Build a braced group node (test helper / prototyping)."""
|
|
144
|
+
if text is None:
|
|
145
|
+
body = "".join(c.text for c in children)
|
|
146
|
+
text = "{" + body + "}"
|
|
147
|
+
return Node(kind="group", text=text, name=name, children=list(children), atom=False)
|
texdiff/oldlines.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""CROSS-REVISION CONTEXT FOR MARKUP EMISSION.
|
|
2
|
+
|
|
3
|
+
refine-diff semantics: an added line is only "genuinely new" - and
|
|
4
|
+
rendered blue - when its normalised content did not already exist in
|
|
5
|
+
the OLD revision of the document. Re-emitted table cells (makecell
|
|
6
|
+
bodies that moved between rows, reformatted config listings) carry
|
|
7
|
+
lines that existed verbatim on the old side; marking those blue makes
|
|
8
|
+
the diff noisy and wrong. This module provides global access to a set
|
|
9
|
+
of all old-revision line norms for consult during rendering, sparing
|
|
10
|
+
us a large refactor to thread it through the render call chain.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
|
|
17
|
+
# module-level mutable state (populated from api.diff_documents):
|
|
18
|
+
# the old revision as one blob of concatenated line norms - refine-diff
|
|
19
|
+
# matches by SUBSTRING containment, so a re-emitted row line whose old
|
|
20
|
+
# counterpart carries extra trailing structure (``... \\ \\hline``)
|
|
21
|
+
# still counts as existing
|
|
22
|
+
old_blob: str = ""
|
|
23
|
+
old_lines: set[str] = set()
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def norm_line(line: str) -> str:
|
|
27
|
+
"""Normalise a source line the way the reference build does."""
|
|
28
|
+
clean = re.sub(r"%DIF.*$", "", line)
|
|
29
|
+
clean = re.sub(r"[%\\{}\[\]&]", "", clean)
|
|
30
|
+
return re.sub(r"\s+", "", clean)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def mark_old_lines(source: str) -> None:
|
|
34
|
+
"""Record all content line norms of the OLD revision."""
|
|
35
|
+
global old_blob, old_lines
|
|
36
|
+
norms = [norm_line(line) for line in source.split("\n")]
|
|
37
|
+
old_lines = {n for n in norms if len(n) >= 10}
|
|
38
|
+
old_blob = "".join(n + "\n" for n in norms if len(n) >= 10)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def in_old(line: str) -> bool:
|
|
42
|
+
"""True when a normalised version of `line` existed in the old revision."""
|
|
43
|
+
norm = norm_line(line)
|
|
44
|
+
return len(norm) >= 10 and norm in old_blob
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def is_struct_line(line: str) -> bool:
|
|
48
|
+
"""Bare structure lines never take a colour declaration."""
|
|
49
|
+
return bool(re.match(r"^\s*\\(?:rowcolor|hline|begin|end|caption)\b", line))
|