texdiff 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
texdiff/flatten.py ADDED
@@ -0,0 +1,185 @@
1
+ """\\input/\\include expansion: multi-file document → single source.
2
+
3
+ The equivalent of ``latexdiff --flatten`` / ``latexpand``, inside the
4
+ tool. texdiff parses per file, then expands references so the aligner
5
+ sees one document tree.
6
+
7
+ Design:
8
+
9
+ * textual (linear) expansion, matching latexpand's contract: no macro
10
+ semantics, just file inclusion - what ``\\input`` does in TeX;
11
+ * files are resolved relative to the INCLUDING file's directory
12
+ (TeX's kpathsea semantics simplified: the main file's dir, then the
13
+ including chain);
14
+ * cycle protection: a file already on the expansion stack is kept
15
+ verbatim instead of being recursed into (TeX would loop forever);
16
+ * missing files are kept verbatim rather than fatal - a broken
17
+ ``\\input`` must not block the diff, and keeping the reference makes
18
+ the output still compilable;
19
+ * comment lines and verbatim environments are skipped: a commented
20
+ ``% \\input{x}`` is not an inclusion, and verbatim bodies must never
21
+ be expanded.
22
+
23
+ The emitter later re-uses the skip logic for the same purpose (never
24
+ mark up inside verbatim).
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import re
30
+ from pathlib import Path
31
+
32
+ # \input{name} / \include{name}; name may lack the .tex extension
33
+ # (TeX appends it), may contain letters, /, -, _, . and spaces
34
+ _INPUT_RE = re.compile(r"\\(input|include)\*?\s*\{([^{}]+)\}")
35
+
36
+ # block of lines which must never be expanded: verbatim-like envs
37
+ _VERBATIM_ENVS = (
38
+ "verbatim",
39
+ "verbatim*",
40
+ "lstlisting",
41
+ "minted",
42
+ "alltt",
43
+ "filecontents",
44
+ "filecontents*",
45
+ )
46
+
47
+ _MARKER_PREFIX = "% texdiff-flatten:"
48
+
49
+
50
+ class Flattener:
51
+ """Expand ``\\input``/``\\include`` in a source string.
52
+
53
+ Attributes:
54
+ base_dir: directory files are resolved against.
55
+ expanded: number of references successfully expanded.
56
+ missing: number of references whose file did not exist
57
+ (kept verbatim).
58
+ markers: insert ``% texdiff-flatten:`` annotation comments
59
+ around expanded content (for debugging / provenance).
60
+ """
61
+
62
+ def __init__(self, base_dir: Path | str | None = None, markers: bool = False):
63
+ self.base_dir = Path(base_dir) if base_dir is not None else Path.cwd()
64
+ self.markers = markers
65
+ self.expanded = 0
66
+ self.missing = 0
67
+
68
+ def flatten(self, source: str) -> str:
69
+ """Expand all \\input/\\include references in ``source``."""
70
+ return self._expand(source, self.base_dir, stack=frozenset())
71
+
72
+ def _expand(self, source: str, current_dir: Path, stack: frozenset[Path]) -> str:
73
+ out: list[str] = []
74
+ pos = 0
75
+ for m in self._iter_skipping(source):
76
+ out.append(source[pos : m.start()])
77
+ cmd, name = m.group(1), m.group(2)
78
+ target = self._resolve(name, current_dir)
79
+ if target is None:
80
+ self.missing += 1
81
+ out.append(m.group(0))
82
+ else:
83
+ resolved = target.resolve()
84
+ if resolved in stack:
85
+ # cycle: keep verbatim, do not recurse
86
+ out.append(m.group(0))
87
+ self.expanded += 1
88
+ pos = m.end()
89
+ continue
90
+ content = target.read_text(encoding="utf-8", errors="replace")
91
+ beginning = len(out)
92
+ if self.markers:
93
+ out.append(f"{_MARKER_PREFIX} begin {target.name}\n")
94
+ out.append(self._expand(content, resolved.parent, stack | {resolved}))
95
+ if self.markers:
96
+ out.append(f"\n{_MARKER_PREFIX} end {target.name}\n")
97
+ self.expanded += 1
98
+ del beginning
99
+ pos = m.end()
100
+ out.append(source[pos:])
101
+ return "".join(out)
102
+
103
+ def _resolve(self, name: str, current_dir: Path) -> Path | None:
104
+ """Resolve an \\input name to a file, TeX-style (with .tex)."""
105
+ candidate = name.strip()
106
+ if candidate.startswith("/"):
107
+ # absolute: as-is (with .tex fallback)
108
+ p = Path(candidate)
109
+ return p if p.is_file() else self._with_tex(p)
110
+
111
+ for base in (current_dir, self.base_dir):
112
+ p = base / candidate
113
+ if p.is_file():
114
+ return p
115
+ p = p.with_suffix(".tex")
116
+ if p.is_file():
117
+ with_tex = base / f"{candidate}.tex"
118
+ if with_tex.is_file():
119
+ return with_tex
120
+ return None
121
+
122
+ def _with_tex(self, p: Path) -> Path | None:
123
+ q = p.with_suffix(".tex")
124
+ return p if p.is_file() else (q if q.is_file() else None)
125
+
126
+ def _iter_skipping(self, source: str) -> list[re.Match]:
127
+ """Matches of _INPUT_RE outside comments and verbatim bodies."""
128
+ results: list[re.Match] = []
129
+ i = 0
130
+ n = len(source)
131
+ while i < n:
132
+ # comment: up to end of line (comment includes it)
133
+ if source[i] == "%":
134
+ nl = source.find("\n", i)
135
+ i = n if nl < 0 else nl + 1
136
+ continue
137
+ # verbatim-like environment: skip whole body
138
+ m_verb = re.compile(
139
+ rf"\\begin\{{({'|'.join(re.escape(e) for e in _VERBATIM_ENVS)})\}}"
140
+ ).match(source, i)
141
+ if m_verb:
142
+ end = re.compile(rf"\\end\{{{m_verb.group(1)}\}}")
143
+ m_end = end.search(source, m_verb.end())
144
+ skip_to = m_end.end() if m_end else n
145
+ results_adjacent = _INPUT_RE.finditer(source, m_verb.end(), skip_to)
146
+ # inside verbatim: do not expand - skip entirely
147
+ i = skip_to
148
+ continue
149
+ m = _INPUT_RE.match(source, i)
150
+ if m:
151
+ results.append(m)
152
+ i = m.end()
153
+ continue
154
+ i += 1
155
+ return results
156
+
157
+
158
+ def flatten_source(
159
+ source: str,
160
+ base_dir: Path | str | None = None,
161
+ markers: bool = False,
162
+ ) -> str:
163
+ """Expand \\input/\\include in a source string (base_dir=cwd)."""
164
+ return Flattener(base_dir=base_dir, markers=markers).flatten(source)
165
+
166
+
167
+ def flatten_file(
168
+ source_path: Path | str,
169
+ base_dir: Path | str | None = None,
170
+ markers: bool = False,
171
+ ) -> str:
172
+ """Read a file and expand all its \\input/\\include references.
173
+
174
+ Args:
175
+ source_path: the main .tex file.
176
+ base_dir: fallback directory for unresolvable-relative names
177
+ (default: the main file's directory).
178
+ markers: insert provenance comments around expansions.
179
+
180
+ Returns:
181
+ The flattened source text.
182
+ """
183
+ sp = Path(source_path)
184
+ f = Flattener(base_dir=base_dir if base_dir is not None else sp.parent, markers=markers)
185
+ return f.flatten(sp.read_text(encoding="utf-8", errors="replace"))
texdiff/nodes.py ADDED
@@ -0,0 +1,147 @@
1
+ """Nodes - the intermediate representation produced by parsing.
2
+
3
+ texdiff never operates on pylatexenc node classes directly: we convert
4
+ them to our own lightweight :class:`Node` records first. Each node
5
+
6
+ * carries ``text`` - its EXACT source span, so unchanged nodes round
7
+ trip byte-identically (critical for listings, attachments and
8
+ generated tables);
9
+ * carries ``children`` - a list of sub-nodes (only for group-like
10
+ nodes: environments, braced groups, macro arguments);
11
+ * can be *atomic* (math, verbatim, unknown macros): no children, never
12
+ diffed internally, compared on exact source equality;
13
+ * knows whether it may carry change markup inside it (``atom`` False
14
+ only for text-ish nodes; markup is NEVER injected into atomic
15
+ nodes - that injector is what breaks markup in latexdiff).
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import re
21
+ from dataclasses import dataclass, field
22
+ from typing import Iterator, Optional
23
+
24
+ # sectioning macros whose {\title} becomes part of the signature
25
+ _SECTIONING_RE = re.compile(
26
+ r"\s*\\(?:chapter|section|subsection|subsubsection|paragraph|subparagraph)\*?\s*\{"
27
+ )
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class Node:
32
+ """One node of the concrete syntax tree.
33
+
34
+ Attributes:
35
+ kind: node type discriminator, e.g. ``"env"`` (environment),
36
+ ``"group"`` (braced group / macro argument), ``"macro"``,
37
+ ``"text"`` (plain characters run), ``"math"`` (inline or
38
+ display math), ``"comment"``, ``"specials"``, ``"verb"``
39
+ (verbatim-like environment, atomic).
40
+ text: exact source text of the node INCLUDING delimiters
41
+ (``\\begin{...}...\\end{...}`` for environments, the braces
42
+ for groups). Unchanged nodes are emitted using this text
43
+ verbatim, so formatting survives the diff round trip.
44
+ name: environment or macro name (``None`` for text).
45
+ children: sub-nodes for group-like nodes; empty otherwise.
46
+ atom: if True the node is atomic - never diffed internally and
47
+ never receives inline markup (math, verbatim, unknown
48
+ macros, specials).
49
+ """
50
+
51
+ kind: str
52
+ text: str
53
+ name: Optional[str] = None
54
+ children: list[Node] = field(default_factory=list)
55
+ atom: bool = True
56
+
57
+ def walk(self) -> Iterator["Node"]:
58
+ """Yield this node and all descendants, depth first."""
59
+ yield self
60
+ for child in self.children:
61
+ yield from child.walk()
62
+
63
+ def signature(self) -> str:
64
+ """Return a canonical key used to align two nodes.
65
+
66
+ Two nodes are aligned candidate-wise when their signatures are
67
+ equal and their exact source differs (then we diff inside);
68
+ nodes with equal signature AND equal text are identical.
69
+
70
+ For group-like nodes the signature carries a *content anchor*:
71
+ the key of the first table row found inside (typically the
72
+ ``\\caption`` row of a ``longtable``). Without it every
73
+ ``{\\scriptsize \\begin{longtable}...}`` group would share the
74
+ bare signature ``group`` and a shifted positional match could
75
+ pair two *different* generated tables - whose column counts
76
+ then disagree and the marked-up document no longer compiles.
77
+ The anchor ties each table to its true counterpart as long as
78
+ its caption (or first row) is stable, which holds even when
79
+ every data row changed.
80
+ """
81
+ if self.name and self.kind not in ("group", "env"):
82
+ if self.kind == "macro" and _SECTIONING_RE.match(self.text or ""):
83
+ # sectioning macros: carry the title. Without it every
84
+ # \subsubsection* shares one signature and a heading
85
+ # inserted mid-document shifts every later heading by
86
+ # one - the aligner then retires the old titles just
87
+ # to re-add them as insertions further down.
88
+ title = self._heading_title()
89
+ if title:
90
+ return f"{self.kind}:{self.name}:{title}"
91
+ return f"{self.kind}:{self.name}"
92
+ if self.kind in ("group", "env"):
93
+ base = f"{self.kind}:{self.name}" if self.name else self.kind
94
+ anchor = self._row_anchor()
95
+ if anchor:
96
+ return f"{base}:{anchor}"
97
+ return base
98
+ return self.kind
99
+
100
+ def _heading_title(self) -> str | None:
101
+ """Normalised title argument of a sectioning macro, if any."""
102
+ m = re.match(r"\\[a-zA-Z]+\*?\s*(\[[^\]]*\])?\s*\{", self.text or "")
103
+ if not m:
104
+ return None
105
+ title = (self.text or "")[m.end() :] # after the opening brace
106
+ depth = 1
107
+ for i, ch in enumerate(title):
108
+ if ch == "{" and (i == 0 or title[i - 1] != "\\"):
109
+ depth += 1
110
+ elif ch == "}" and (i == 0 or title[i - 1] != "\\"):
111
+ depth -= 1
112
+ if depth == 0:
113
+ return re.sub(r"\s+", " ", title[:i]).strip() or None
114
+ return None
115
+
116
+ def _row_anchor(self) -> str | None:
117
+ """Key of the first table row in this subtree, if any."""
118
+ for node in self.walk():
119
+ if node.kind == "row" and node.name:
120
+ return node.name[:64]
121
+ return None
122
+
123
+
124
+ def text_node(s: str) -> Node:
125
+ """Build a plain text run node."""
126
+ return Node(kind="text", text=s, atom=True)
127
+
128
+
129
+ def env(
130
+ name: str,
131
+ *children: Node,
132
+ text: Optional[str] = None,
133
+ atom: bool = False,
134
+ ) -> Node:
135
+ """Build an environment node (test helper / prototyping)."""
136
+ if text is None:
137
+ body = "".join(c.text for c in children)
138
+ text = f"\\begin{{{name}}}{body}\\end{{{name}}}"
139
+ return Node(kind="env", text=text, name=name, children=list(children), atom=atom)
140
+
141
+
142
+ def group(*children: Node, name: Optional[str] = None, text: Optional[str] = None) -> Node:
143
+ """Build a braced group node (test helper / prototyping)."""
144
+ if text is None:
145
+ body = "".join(c.text for c in children)
146
+ text = "{" + body + "}"
147
+ return Node(kind="group", text=text, name=name, children=list(children), atom=False)
texdiff/oldlines.py ADDED
@@ -0,0 +1,49 @@
1
+ """CROSS-REVISION CONTEXT FOR MARKUP EMISSION.
2
+
3
+ refine-diff semantics: an added line is only "genuinely new" - and
4
+ rendered blue - when its normalised content did not already exist in
5
+ the OLD revision of the document. Re-emitted table cells (makecell
6
+ bodies that moved between rows, reformatted config listings) carry
7
+ lines that existed verbatim on the old side; marking those blue makes
8
+ the diff noisy and wrong. This module provides global access to a set
9
+ of all old-revision line norms for consult during rendering, sparing
10
+ us a large refactor to thread it through the render call chain.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import re
16
+
17
+ # module-level mutable state (populated from api.diff_documents):
18
+ # the old revision as one blob of concatenated line norms - refine-diff
19
+ # matches by SUBSTRING containment, so a re-emitted row line whose old
20
+ # counterpart carries extra trailing structure (``... \\ \\hline``)
21
+ # still counts as existing
22
+ old_blob: str = ""
23
+ old_lines: set[str] = set()
24
+
25
+
26
+ def norm_line(line: str) -> str:
27
+ """Normalise a source line the way the reference build does."""
28
+ clean = re.sub(r"%DIF.*$", "", line)
29
+ clean = re.sub(r"[%\\{}\[\]&]", "", clean)
30
+ return re.sub(r"\s+", "", clean)
31
+
32
+
33
+ def mark_old_lines(source: str) -> None:
34
+ """Record all content line norms of the OLD revision."""
35
+ global old_blob, old_lines
36
+ norms = [norm_line(line) for line in source.split("\n")]
37
+ old_lines = {n for n in norms if len(n) >= 10}
38
+ old_blob = "".join(n + "\n" for n in norms if len(n) >= 10)
39
+
40
+
41
+ def in_old(line: str) -> bool:
42
+ """True when a normalised version of `line` existed in the old revision."""
43
+ norm = norm_line(line)
44
+ return len(norm) >= 10 and norm in old_blob
45
+
46
+
47
+ def is_struct_line(line: str) -> bool:
48
+ """Bare structure lines never take a colour declaration."""
49
+ return bool(re.match(r"^\s*\\(?:rowcolor|hline|begin|end|caption)\b", line))