texdiff 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
texdiff/__init__.py ADDED
@@ -0,0 +1,45 @@
1
+ """texdiff - AST-driven semantic diff for LaTeX documents.
2
+
3
+ The pipeline IS::
4
+
5
+ old.tex ─┐
6
+ ├─> parse → align trees → emit markup → diff.tex
7
+ new.tex ─┘
8
+
9
+ latexdiff diffs characters and breaks structure;
10
+ texdiff diffs structure and never breaks characters.
11
+ """
12
+
13
+ from .align import Delete, Edit, Insert, Match, Modify, align
14
+ from .api import DiffResult, DiffStats, diff_documents, diff_files
15
+ from .emit import LatexdiffMarkup, render
16
+ from .flatten import Flattener, flatten_file, flatten_source
17
+ from .nodes import Node
18
+ from .parse import ParseError, parse, parse_file
19
+ from .textdiff import Chunk, word_diff
20
+
21
+ __all__ = [
22
+ "align",
23
+ "Chunk",
24
+ "Delete",
25
+ "diff_documents",
26
+ "diff_files",
27
+ "DiffResult",
28
+ "DiffStats",
29
+ "Edit",
30
+ "Flattener",
31
+ "flatten_file",
32
+ "flatten_source",
33
+ "Insert",
34
+ "LatexdiffMarkup",
35
+ "Match",
36
+ "Modify",
37
+ "Node",
38
+ "parse",
39
+ "ParseError",
40
+ "parse_file",
41
+ "render",
42
+ "word_diff",
43
+ ]
44
+
45
+ __version__ = "0.1.0"
texdiff/align.py ADDED
@@ -0,0 +1,202 @@
1
+ """Tree alignment: match two node lists into aligned pairs and edits.
2
+
3
+ The aligner is where latexdiff's heuristics become data. It produces,
4
+ for a pair of node lists, an edit script of three shapes:
5
+
6
+ * ``match(old, new)`` - same signature, identical text → keep verbatim
7
+ * ``modify(old, new)`` - same signature, different text → recurse if
8
+ both sides have children (block diff), else whole-block replacement
9
+ * ``insert(new)`` / ``delete(old)`` - no counterpart
10
+
11
+ Uses :class:`difflib.SequenceMatcher` over node signatures; runs of
12
+ equal signature with different text become ``modify`` pairs aligned
13
+ positionally within the run.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from dataclasses import dataclass
19
+ from difflib import SequenceMatcher
20
+
21
+ from .nodes import Node
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class Match:
26
+ """A node kept verbatim (identical on both sides)."""
27
+
28
+ node: Node
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class Modify:
33
+ """A node pair to be diffed further (recursed) - or replaced.
34
+
35
+ Attributes:
36
+ old: the node from the old revision.
37
+ new: the node from the new revision.
38
+ inner: the recursive edit script of the children, when both
39
+ sides are recursable; ``None`` means whole-block replace.
40
+ """
41
+
42
+ old: Node
43
+ new: Node
44
+ inner: "list[Edit] | None" = None
45
+
46
+
47
+ @dataclass(frozen=True)
48
+ class Insert:
49
+ """A node present only in the new version."""
50
+
51
+ new: Node
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class Delete:
56
+ """A node present only in the old version."""
57
+
58
+ old: Node
59
+
60
+
61
+ Edit = Match | Modify | Insert | Delete
62
+
63
+
64
+ def align(old: list[Node], new: list[Node]) -> list[Edit]:
65
+ """Align two node lists into an edit script.
66
+
67
+ The returned script is in document order, interleaving old and new
68
+ positions like a unified diff.
69
+
70
+ Algorithm:
71
+
72
+ 1. an equal *prefix* and *suffix* of signatures anchors the
73
+ alignment first: ``SequenceMatcher`` finds the globally
74
+ longest matching block, which can let a later identical run
75
+ steal the match from the earliest identical nodes - an
76
+ inserted chapter between two near-identical passages then
77
+ drags unchanged leading lines into the insertion (they
78
+ render as added);
79
+ 2. ``SequenceMatcher`` over the remaining node *signatures*
80
+ yields runs of same-signature positions deemed "matching";
81
+ 3. within a run, nodes are paired positionally: identical source
82
+ text becomes :class:`Match`, differing text :class:`Modify`;
83
+ 4. gap regions become ``Delete`` (old-only) then ``Insert``
84
+ (new-only) sequences, preserving document order.
85
+ """
86
+
87
+ def _pair(o: Node, n: Node) -> Edit:
88
+ if o.text == n.text:
89
+ return Match(node=o)
90
+ if o.kind == "row" and n.kind == "row" and o.name == n.name:
91
+ # equal row signature (content key) with differing text
92
+ # means the difference is purely structural: \hline on
93
+ # the other side of the row, whitespace, comments.
94
+ # Duplicating such a pair (del + add) would emit the
95
+ # table header material (\endfirsthead, \endhead ...)
96
+ # twice inside one longtable, which can throw longtable
97
+ # into an infinite loop. Keep the old row verbatim.
98
+ return Match(node=o)
99
+ return Modify(old=o, new=n)
100
+
101
+ # 1. common prefix, paired positionally (identical to what a
102
+ # matching-block run does - only the anchoring is stronger)
103
+ pre = 0
104
+ while (
105
+ pre < len(old)
106
+ and pre < len(new)
107
+ and old[pre].signature() == new[pre].signature()
108
+ ):
109
+ pre += 1
110
+ # common suffix (may not overlap the prefix)
111
+ suf = 0
112
+ while (
113
+ suf < len(old) - pre
114
+ and suf < len(new) - pre
115
+ and old[len(old) - 1 - suf].signature() == new[len(new) - 1 - suf].signature()
116
+ ):
117
+ suf += 1
118
+ mid_old = old[pre : len(old) - suf]
119
+ mid_new = new[pre : len(new) - suf]
120
+
121
+ sm = SequenceMatcher(
122
+ a=[n.signature() for n in mid_old],
123
+ b=[n.signature() for n in mid_new],
124
+ autojunk=False,
125
+ )
126
+
127
+ edits: list[Edit] = [_pair(o, n) for o, n in zip(old[:pre], new[:pre])]
128
+ prev_a = prev_b = 0
129
+ for block in sm.get_matching_blocks():
130
+ # gap before this matching block: deletions then insertions.
131
+ # Block-head refinement: ``SequenceMatcher`` extends a matching
132
+ # block as far as equal *signatures* reach - with the wildcard
133
+ # ``text`` signature, a run may begin by pairing two text nodes
134
+ # whose contents share nothing, while the node that should pair
135
+ # with the old head sits at the START of the insertion gap just
136
+ # before the block. When that pairing is clearly better, consume
137
+ # the gap head into a Modify pair, emit the rest of the gap as
138
+ # inserts, and let the block start one position later on both
139
+ # sides (its first new-side node then lands at the gap end).
140
+ new_gap = mid_new[prev_b : block.b]
141
+ head_a, head_b = block.a, block.b
142
+ if (
143
+ block.size
144
+ and new_gap
145
+ and mid_old[block.a].kind == "text"
146
+ and new_gap[0].kind == "text"
147
+ and mid_new[block.b].kind == "text"
148
+ ):
149
+ o_head = mid_old[block.a]
150
+ best = max(
151
+ range(min(len(new_gap), 4)),
152
+ key=lambda k: _text_similarity(o_head.text, new_gap[k].text),
153
+ )
154
+ n_head = new_gap[best]
155
+ if _text_similarity(o_head.text, n_head.text) > (
156
+ _text_similarity(o_head.text, mid_new[block.b].text) + 0.2
157
+ ):
158
+ edits.extend(Delete(old=n) for n in mid_old[prev_a : block.a])
159
+ edits.extend(Insert(new=n) for n in new_gap[:best])
160
+ edits.append(_pair(o_head, n_head))
161
+ edits.extend(Insert(new=n) for n in new_gap[best + 1 :])
162
+ edits.append(Insert(new=mid_new[block.b]))
163
+ head_a, head_b = block.a + 1, block.b + 1
164
+ for i in range(1, block.size):
165
+ edits.append(
166
+ _pair(mid_old[block.a + i], mid_new[block.b + i])
167
+ )
168
+ prev_a, prev_b = block.a + block.size, block.b + block.size
169
+ continue
170
+ edits.extend(Delete(old=n) for n in mid_old[prev_a : block.a])
171
+ edits.extend(Insert(new=n) for n in new_gap)
172
+ # the matching run: pair positionally, decide Match vs Modify
173
+ for i in range(head_a, head_a + block.size):
174
+ edits.append(
175
+ _pair(mid_old[i], mid_new[head_b + (i - head_a)])
176
+ )
177
+ prev_a, prev_b = head_a + block.size, head_b + block.size
178
+ edits.extend(Delete(old=n) for n in mid_old[prev_a:])
179
+ edits.extend(Insert(new=n) for n in mid_new[prev_b:])
180
+ # common suffix, in document order
181
+ edits.extend(
182
+ _pair(o, n)
183
+ for o, n in zip(
184
+ old[len(old) - suf :], new[len(new) - suf :]
185
+ )
186
+ )
187
+ return edits
188
+
189
+
190
+ def _text_similarity(a: str, b: str) -> float:
191
+ """Quick content similarity of two text runs, in ``[0, 1]``.
192
+
193
+ Used only to compare pairing candidates at matching-block
194
+ boundaries, so a cheap ``SequenceMatcher.ratio`` over word
195
+ tokens (not raw characters) is enough - and stays stable for
196
+ long runs where character noise would dominate.
197
+ """
198
+ wa = [t for t in a.split() if t]
199
+ wb = [t for t in b.split() if t]
200
+ if not wa or not wb:
201
+ return 0.0
202
+ return SequenceMatcher(a=wa, b=wb, autojunk=False).ratio()
texdiff/api.py ADDED
@@ -0,0 +1,326 @@
1
+ """Public API of texdiff.
2
+
3
+ Three functions make up the pipeline::
4
+
5
+ from texdiff import parse, align, render
6
+ from texdiff import diff_documents
7
+
8
+ result = diff_documents(old_source, new_source)
9
+ print(result.marked_up) # the diff.tex source
10
+ print(result.stats) # counts of edits
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from dataclasses import dataclass
17
+
18
+ from .align import Delete, Edit, Insert, Match, Modify, align
19
+
20
+ # word-similarity floor below which a paired text run is retired and
21
+ # re-added wholesale instead of word-marked: two sentences sharing
22
+ # less than half their words are a rewrite, not an edit
23
+ _REPLACE_WORD_SIMILARITY = 0.5
24
+ from .emit import LatexdiffMarkup, render
25
+ from .flatten import Flattener, flatten_file, flatten_source
26
+ from .nodes import Node, text_node
27
+ from .parse import parse
28
+ from .preamble import (
29
+ count_preamble_changes,
30
+ mark_preamble_macro_changes,
31
+ split_preamble,
32
+ )
33
+ from .textdiff import DELETE, EQUAL, Chunk, word_diff
34
+
35
+
36
+ @dataclass(frozen=True)
37
+ class DiffStats:
38
+ """Aggregated change counts of one diff."""
39
+
40
+ matches: int = 0
41
+ modifications: int = 0
42
+ insertions: int = 0
43
+ deletions: int = 0
44
+ preamble_changes: int = 0
45
+
46
+ @property
47
+ def changed(self) -> int:
48
+ """Number of changed blocks (modify+insert+delete)."""
49
+ return self.modifications + self.insertions + self.deletions
50
+
51
+ def __add__(self, other: "DiffStats") -> "DiffStats":
52
+ """Sum two stats (used to layer preamble counts onto body counts)."""
53
+ return DiffStats(
54
+ matches=self.matches + other.matches,
55
+ modifications=self.modifications + other.modifications,
56
+ insertions=self.insertions + other.insertions,
57
+ deletions=self.deletions + other.deletions,
58
+ preamble_changes=self.preamble_changes + other.preamble_changes,
59
+ )
60
+
61
+ def __str__(self) -> str: # pragma: no cover - cosmetic
62
+ pre = (
63
+ f", {self.preamble_changes} preamble lines changed"
64
+ if self.preamble_changes
65
+ else ""
66
+ )
67
+ return (
68
+ f"{self.matches} unchanged, {self.modifications} modified, "
69
+ f"{self.insertions} added, {self.deletions} deleted{pre}"
70
+ )
71
+
72
+
73
+ @dataclass(frozen=True)
74
+ class DiffResult:
75
+ """Outcome of diffing two documents.
76
+
77
+ Attributes:
78
+ marked_up: the LaTeX source with \\DIFadd/\\DIFdel markup.
79
+ stats: change counts.
80
+ edits: the full edit script (for tools/UIs on top).
81
+ """
82
+
83
+ marked_up: str
84
+ stats: DiffStats
85
+ edits: list[Edit]
86
+
87
+
88
+ def diff_documents(
89
+ old_source: str,
90
+ new_source: str,
91
+ markup: LatexdiffMarkup = LatexdiffMarkup(),
92
+ inject_preamble: bool = True,
93
+ ) -> DiffResult:
94
+ """Diff two LaTeX documents (flattened sources).
95
+
96
+ Args:
97
+ old_source: old revision LaTeX source.
98
+ new_source: new revision LaTeX source.
99
+ markup: markup style for additions/deletions.
100
+ inject_preamble: insert the markup macro definitions
101
+ (``\\DIFadd`` etc.) before ``\\begin{document}`` so the
102
+ output compiles standalone; skipped when the document
103
+ defines them already or has no document body.
104
+
105
+ Returns:
106
+ :class:`DiffResult` with the marked-up document.
107
+ """
108
+ # preamble policy: the output uses the NEW preamble; preamble
109
+ # differences are counted, never marked up (see texdiff.preamble)
110
+ pre_old, body_old, post_old = split_preamble(old_source)
111
+ pre_new, body_new, post_new = split_preamble(new_source)
112
+ preamble_changes = count_preamble_changes(old_source, new_source)
113
+
114
+ # cross-revision context: added lines that already existed in the
115
+ # old revision render black, not blue (refine-diff semantics)
116
+ from . import oldlines
117
+
118
+ oldlines.mark_old_lines(old_source)
119
+
120
+ edits = _diff_nodes(parse(body_old), parse(body_new))
121
+ body_markup = render(edits, markup)
122
+ stats = _count(edits) + DiffStats(preamble_changes=preamble_changes)
123
+ marked_up = pre_new + body_markup + post_new if pre_new else body_markup
124
+ marked_up = _fix_listing_languages(marked_up)
125
+ if pre_new:
126
+ # change-record and other body-invoked preamble macros:
127
+ # their typeset content would otherwise silently swallow
128
+ # additions (preamble policy keeps the new revision as-is)
129
+ marked_up = mark_preamble_macro_changes(marked_up, old_source, new_source)
130
+ if inject_preamble and stats.changed:
131
+ marked_up = _inject_preamble(marked_up, new_source)
132
+ return DiffResult(marked_up=marked_up, stats=stats, edits=edits)
133
+
134
+
135
+ def _fix_listing_languages(marked_up: str) -> str:
136
+ """Drop ``language=<lang>,`` from document-level ``\\lstset`` calls.
137
+
138
+ The marked-up listings carry ``[alsolanguage=DIFcode]`` so the
139
+ ``%DIF`` line markers render as strikeout/blue instead of literal
140
+ text. But a primary language with C-style *block comments*
141
+ (``/** ... */``) swallows the markers inside comment regions -
142
+ listings stops processing delimiters once a block comment opens.
143
+ The reference pipeline strips ``language=C++`` for exactly this
144
+ reason; keyword colouring is a visual nicety, marker visibility
145
+ is the point of a diff document.
146
+ """
147
+ import re
148
+
149
+ return re.sub(r"(\\lstset\{)language=[A-Za-z0-9+#]*,", r"\1", marked_up)
150
+
151
+
152
+ def _inject_preamble(marked_up: str, source: str) -> str:
153
+ """Insert PREAMBLE_TEMPLATE before ``\\begin{document}``.
154
+
155
+ Follows latexdiff's convention; skips injection when the source
156
+ already defines ``\\DIFadd`` or has no ``\\begin{document}```
157
+ (a fragment - the caller provides definitions).
158
+ """
159
+ if "\\DIFadd" in source:
160
+ return marked_up
161
+ idx = marked_up.find("\\begin{document}")
162
+ if idx < 0:
163
+ return marked_up
164
+ from .emit import PREAMBLE_TEMPLATE
165
+
166
+ return marked_up[:idx] + PREAMBLE_TEMPLATE + marked_up[idx:]
167
+
168
+
169
+ def diff_files(
170
+ old_path: str,
171
+ new_path: str,
172
+ flatten: bool = True,
173
+ **kwargs,
174
+ ) -> DiffResult:
175
+ """Diff two LaTeX files (UTF-8).
176
+
177
+ By default the sources are flattened first (``\\input``/``\\include``
178
+ expanded), like ``latexdiff --flatten``; pass ``flatten=False`` to
179
+ compare already-flattened sources.
180
+ """
181
+ from pathlib import Path
182
+
183
+ def read(p):
184
+ return Path(p).read_text(encoding="utf-8")
185
+
186
+ old_source = flatten_file(old_path) if flatten else read(old_path)
187
+ new_source = flatten_file(new_path) if flatten else read(new_path)
188
+ return diff_documents(old_source, new_source, **kwargs)
189
+
190
+
191
+ def _diff_nodes(old: list[Node], new: list[Node]) -> list[Edit]:
192
+ """Align, then recurse into modified pairable nodes.
193
+
194
+ A ``Modify`` pair whose both sides have children is re-aligned at
195
+ the next level down (e.g. a changed paragraph lives inside the
196
+ ``document`` environment); the recursion result is carried on the
197
+ ``Modify`` (``inner``) and rendered in place of the whole-block
198
+ del/add replacement.
199
+
200
+ A ``Modify`` pair of two *text* nodes is refined to a word-level
201
+ diff so individual words - not whole paragraphs - get marked up.
202
+ """
203
+ edits = align(old, new)
204
+ result: list[Edit] = []
205
+ for edit in edits:
206
+ if (
207
+ isinstance(edit, Modify)
208
+ and edit.old.children
209
+ and edit.new.children
210
+ ):
211
+ inner = _diff_nodes(edit.old.children, edit.new.children)
212
+ result.append(Modify(old=edit.old, new=edit.new, inner=inner))
213
+ elif (
214
+ isinstance(edit, Modify)
215
+ and edit.old.kind == "text"
216
+ and edit.new.kind == "text"
217
+ ):
218
+ # heavily rewritten paragraphs: interleaving word marks
219
+ # between two sentences sharing less than half their words
220
+ # reads as word salad. Such paragraphs retire wholesale -
221
+ # delete + re-add, the track-changes convention - while
222
+ # well-matched paragraphs of the same run keep their
223
+ # word-level treatment.
224
+ para_edits = _paragraph_edits(edit.old.text, edit.new.text)
225
+ if para_edits is None:
226
+ result.append(edit)
227
+ else:
228
+ result.extend(para_edits)
229
+ else:
230
+ result.append(edit)
231
+ return result
232
+
233
+
234
+ def _run_similarity(a: str, b: str) -> float:
235
+ """Word-level similarity of two text runs, in ``[0, 1]``.
236
+
237
+ Tokens are compared with edge punctuation stripped, so
238
+ ``text.`` matches ``text`` - the word differ itself works at
239
+ that granularity and the retire/re-add decision must agree
240
+ with it, or trivially reworded sentences (``Body text.`` to
241
+ ``Body text changed.``) would look like rewrites.
242
+ """
243
+ from difflib import SequenceMatcher
244
+
245
+ wa = [w for w in (t.strip(".,;:!?()\"'`") for t in a.split()) if w]
246
+ wb = [w for w in (t.strip(".,;:!?()\"'`") for t in b.split()) if w]
247
+ if not wa or not wb:
248
+ return 0.0
249
+ return SequenceMatcher(a=wa, b=wb, autojunk=False).ratio()
250
+
251
+
252
+ def _paragraph_edits(old: str, new: str) -> list[Edit] | None:
253
+ """Word-refine a text run paragraph by paragraph.
254
+
255
+ A text run spans everything up to the next macro/environment -
256
+ often several paragraphs. Blank-line-separated paragraphs pair
257
+ positionally; each pair either takes the word-level diff or, when
258
+ the two paragraphs share less than half their words, the
259
+ whole-paragraph delete + re-add. Whitespace around and between
260
+ paragraphs keeps the NEW bytes - separator differences are
261
+ invisible and must not produce marks.
262
+
263
+ Returns ``None`` when paragraph counts do not match or the run
264
+ is a wholesale rewrite: the caller then keeps the block-level
265
+ ``Modify`` (whole-run retire + re-add).
266
+ """
267
+
268
+ def split_run(s: str) -> tuple[str, list[str], str]:
269
+ lead = s[: len(s) - len(s.lstrip())]
270
+ trail = s[len(s.rstrip()) :]
271
+ core = s[len(lead) : len(s) - len(trail)]
272
+ return lead, re.split(r"\n\s*\n", core), trail
273
+
274
+ lead_a, paras_a, trail_a = split_run(old)
275
+ lead_b, paras_b, trail_b = split_run(new)
276
+ if len(paras_a) != len(paras_b) or any(
277
+ not p.strip() for p in paras_a + paras_b
278
+ ):
279
+ # mismatched paragraph structure: whole-run judgement
280
+ return (
281
+ None
282
+ if _run_similarity(old, new) < _REPLACE_WORD_SIMILARITY
283
+ else _chunks_to_edits(word_diff(old, new))
284
+ )
285
+
286
+ def sep(text: str) -> Edit:
287
+ # paragraph separator, kept from the NEW side when it exists
288
+ return Match(node=text_node(text if text.strip() else "\n\n"))
289
+
290
+ out: list[Edit] = [sep(lead_b or lead_a)]
291
+ for k, (ca, cb) in enumerate(zip(paras_a, paras_b)):
292
+ if _run_similarity(ca, cb) < _REPLACE_WORD_SIMILARITY:
293
+ out.append(Delete(old=text_node(ca)))
294
+ out.append(Insert(new=text_node(cb)))
295
+ else:
296
+ out.extend(_chunks_to_edits(word_diff(ca, cb)))
297
+ if k + 1 < len(paras_a):
298
+ out.append(sep("\n\n"))
299
+ out.extend(_chunks_to_edits(word_diff(trail_a, trail_b)))
300
+ return out
301
+
302
+
303
+ def _chunks_to_edits(chunks: list[Chunk]) -> list[Edit]:
304
+ """Convert word-diff chunks to a flat list of atomic edits."""
305
+ out: list[Edit] = []
306
+ for chunk in chunks:
307
+ if chunk.op == EQUAL:
308
+ out.append(Match(node=text_node(chunk.text)))
309
+ elif chunk.op == DELETE:
310
+ out.append(Delete(old=text_node(chunk.text)))
311
+ else: # insert
312
+ out.append(Insert(new=text_node(chunk.text)))
313
+ return out
314
+
315
+ def _count(edits: list[Edit]) -> DiffStats:
316
+ m = i = d = n = 0
317
+ for e in edits:
318
+ if isinstance(e, Match):
319
+ m += 1
320
+ elif isinstance(e, Modify):
321
+ n += 1
322
+ elif isinstance(e, Insert):
323
+ i += 1
324
+ elif isinstance(e, Delete):
325
+ d += 1
326
+ return DiffStats(matches=m, modifications=n, insertions=i, deletions=d)
texdiff/check.py ADDED
@@ -0,0 +1,61 @@
1
+ """--check: compile the marked-up diff once, report/fail loudly.
2
+
3
+ The whole point of texdiff is diffs that compile. ``--check`` makes
4
+ that property *verified* rather than trusted: it writes the marked-up
5
+ document into a scratch directory, runs one ``pdflatex`` pass, and
6
+ returns the verdict with the log tail on failure.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import subprocess
12
+ import tempfile
13
+ from pathlib import Path
14
+
15
+ from .api import diff_files
16
+
17
+ _PDFLATEX_TIMEOUT = 120
18
+
19
+
20
+ def run_pdflatex(tex_path: Path, work_dir: Path) -> subprocess.CompletedProcess:
21
+ """Run one nonstop pdflatex pass in ``work_dir``."""
22
+ return subprocess.run(
23
+ [
24
+ "pdflatex",
25
+ "-interaction=nonstopmode",
26
+ "-halt-on-error",
27
+ "-output-directory",
28
+ str(work_dir),
29
+ str(tex_path),
30
+ ],
31
+ cwd=work_dir,
32
+ capture_output=True,
33
+ text=True,
34
+ timeout=_PDFLATEX_TIMEOUT,
35
+ )
36
+
37
+
38
+ def check_compiles(old_path, new_path, work_dir: Path | None = None) -> tuple[bool, str]:
39
+ """Diff two files and verify the marked-up result compiles.
40
+
41
+ Returns:
42
+ (ok, log): ok is True when pdflatex exited 0; log is the
43
+ pdflatex output (empty string when pdflatex is unavailable).
44
+ """
45
+ if work_dir is None:
46
+ work_dir = Path(tempfile.mkdtemp(prefix="texdiff-check-"))
47
+ else:
48
+ work_dir = Path(work_dir)
49
+ work_dir.mkdir(parents=True, exist_ok=True)
50
+
51
+ result = diff_files(str(old_path), str(new_path))
52
+ tex = work_dir / "texdiff-check.tex"
53
+ tex.write_text(result.marked_up, encoding="utf-8")
54
+
55
+ from shutil import which
56
+
57
+ if which("pdflatex") is None: # pragma: no cover - environment
58
+ return True, ""
59
+
60
+ cp = run_pdflatex(tex, work_dir)
61
+ return cp.returncode == 0, cp.stdout + cp.stderr