texdiff 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. texdiff-0.2.0/LICENSE +21 -0
  2. texdiff-0.2.0/PKG-INFO +112 -0
  3. texdiff-0.2.0/README.md +61 -0
  4. texdiff-0.2.0/pyproject.toml +64 -0
  5. texdiff-0.2.0/setup.cfg +4 -0
  6. texdiff-0.2.0/src/texdiff/__init__.py +45 -0
  7. texdiff-0.2.0/src/texdiff/align.py +202 -0
  8. texdiff-0.2.0/src/texdiff/api.py +326 -0
  9. texdiff-0.2.0/src/texdiff/check.py +61 -0
  10. texdiff-0.2.0/src/texdiff/cli.py +87 -0
  11. texdiff-0.2.0/src/texdiff/emit.py +1193 -0
  12. texdiff-0.2.0/src/texdiff/flatten.py +185 -0
  13. texdiff-0.2.0/src/texdiff/nodes.py +147 -0
  14. texdiff-0.2.0/src/texdiff/oldlines.py +49 -0
  15. texdiff-0.2.0/src/texdiff/parse.py +524 -0
  16. texdiff-0.2.0/src/texdiff/preamble.py +388 -0
  17. texdiff-0.2.0/src/texdiff/tables.py +540 -0
  18. texdiff-0.2.0/src/texdiff/textdiff.py +157 -0
  19. texdiff-0.2.0/src/texdiff.egg-info/PKG-INFO +112 -0
  20. texdiff-0.2.0/src/texdiff.egg-info/SOURCES.txt +36 -0
  21. texdiff-0.2.0/src/texdiff.egg-info/dependency_links.txt +1 -0
  22. texdiff-0.2.0/src/texdiff.egg-info/entry_points.txt +2 -0
  23. texdiff-0.2.0/src/texdiff.egg-info/requires.txt +11 -0
  24. texdiff-0.2.0/src/texdiff.egg-info/top_level.txt +1 -0
  25. texdiff-0.2.0/tests/test_align.py +71 -0
  26. texdiff-0.2.0/tests/test_alignment.py +586 -0
  27. texdiff-0.2.0/tests/test_api.py +135 -0
  28. texdiff-0.2.0/tests/test_check.py +125 -0
  29. texdiff-0.2.0/tests/test_cli.py +64 -0
  30. texdiff-0.2.0/tests/test_corpus.py +111 -0
  31. texdiff-0.2.0/tests/test_emit.py +119 -0
  32. texdiff-0.2.0/tests/test_flatten.py +159 -0
  33. texdiff-0.2.0/tests/test_parse.py +132 -0
  34. texdiff-0.2.0/tests/test_preamble.py +83 -0
  35. texdiff-0.2.0/tests/test_tables.py +160 -0
  36. texdiff-0.2.0/tests/test_tables_render.py +324 -0
  37. texdiff-0.2.0/tests/test_textdiff.py +61 -0
  38. texdiff-0.2.0/tests/test_verbatim.py +207 -0
texdiff-0.2.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 GoBobr
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
texdiff-0.2.0/PKG-INFO ADDED
@@ -0,0 +1,112 @@
1
+ Metadata-Version: 2.4
2
+ Name: texdiff
3
+ Version: 0.2.0
4
+ Summary: AST-driven semantic diff for LaTeX documents
5
+ Author: GoBobr
6
+ License: MIT License
7
+
8
+ Copyright (c) 2026 GoBobr
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Project-URL: Homepage, https://github.com/GoBobr/texdiff
29
+ Keywords: latex,diff,latexdiff,ast,track-changes
30
+ Classifier: Development Status :: 2 - Pre-Alpha
31
+ Classifier: Intended Audience :: Science/Research
32
+ Classifier: License :: OSI Approved :: MIT License
33
+ Classifier: Programming Language :: Python :: 3
34
+ Classifier: Programming Language :: Python :: 3.10
35
+ Classifier: Programming Language :: Python :: 3.11
36
+ Classifier: Programming Language :: Python :: 3.12
37
+ Classifier: Topic :: Text Processing :: Markup :: LaTeX
38
+ Requires-Python: >=3.10
39
+ Description-Content-Type: text/markdown
40
+ License-File: LICENSE
41
+ Requires-Dist: pylatexenc>=2.10
42
+ Provides-Extra: dev
43
+ Requires-Dist: pytest>=7; extra == "dev"
44
+ Requires-Dist: pytest-cov>=4; extra == "dev"
45
+ Provides-Extra: docs
46
+ Requires-Dist: mkdocs>=1.5; extra == "docs"
47
+ Requires-Dist: mkdocs-material>=9; extra == "docs"
48
+ Requires-Dist: mkdocstrings[python]>=0.24; extra == "docs"
49
+ Requires-Dist: pymdown-extensions>=10; extra == "docs"
50
+ Dynamic: license-file
51
+
52
+ # texdiff
53
+
54
+ **AST-driven semantic diff for LaTeX documents.**
55
+
56
+ `texdiff` compares two revisions of a LaTeX document and produces a single
57
+ marked-up `diff.tex` that compiles with `pdflatex` — added text in blue,
58
+ deleted text struck out — while preserving the original document's styling,
59
+ cross-references and custom classes.
60
+
61
+ ## Why
62
+
63
+ [`latexdiff`](https://www.ctan.org/pkg/latexdiff) is the de-facto standard for
64
+ review-marked LaTeX, but it diffs *token soup*: it guesses word boundaries with
65
+ regexes and re-injects markup into the merged text, hoping braces still
66
+ balance. On real-world documents the result is markup that crosses brace /
67
+ environment boundaries, glues old and new table cells together, and corrupts
68
+ listings — which is why every serious latexdiff user eventually ends up
69
+ maintaining a pile of fragile pre/post-processing scripts around it.
70
+
71
+ `texdiff` takes the other road: **diff the tree, never the text.**
72
+
73
+ 1. Parse old and new documents into a LaTeX syntax tree
74
+ ([pylatexenc](https://github.com/phfaist/pylatexenc) — a faithful concrete
75
+ syntax tree carrying exact source spans, no macro expansion).
76
+ 2. Align the trees level by level (document → environments → groups → text
77
+ runs) with difflib-style sequence matching. A change can never leak across
78
+ a node boundary.
79
+ 3. Walk the merged tree and emit `\DIFadd{...}` / `\DIFdel{...}` markup —
80
+ latexdiff-compatible rendering, valid by construction.
81
+
82
+ Restructured tables (re-ordered rows, changed column layouts), math,
83
+ listings and verbatim environments are handled as atomic or row-aligned
84
+ blocks instead of being shredded by a word diff.
85
+
86
+ Multi-file sources are flattened automatically (`\input`/`\include`
87
+ resolved relative to each file, like `latexdiff --flatten`), and the
88
+ preamble of the new revision is used verbatim so custom classes and
89
+ macros keep working.
90
+
91
+ ## Usage
92
+
93
+ ```sh
94
+ texdiff old/main.tex new/main.tex > diff.tex # \input expansion built in
95
+ pdflatex diff.tex # done - no pre/postprocessing
96
+ texdiff old.tex new.tex -o diff.tex --check # + fail loudly if it won't compile
97
+ texdiff old.tex new.tex --stats # "611 unchanged, 29 added, ..."
98
+ ```
99
+
100
+ ```sh
101
+ pip install texdiff # or from a checkout:
102
+ pip install -e ".[dev]" && pytest
103
+ ```
104
+
105
+ Status: **v0.2.0** — core pipeline plus flattening, row-granular table
106
+ alignment, preamble policy, `--check` and a compile-guaranteed corpus in
107
+ CI; validated on a 65-page generated longtable-heavy document. See the
108
+ [roadmap](https://gobobr.github.io/texdiff/roadmap/) for what is next.
109
+
110
+ ## License
111
+
112
+ MIT
@@ -0,0 +1,61 @@
1
+ # texdiff
2
+
3
+ **AST-driven semantic diff for LaTeX documents.**
4
+
5
+ `texdiff` compares two revisions of a LaTeX document and produces a single
6
+ marked-up `diff.tex` that compiles with `pdflatex` — added text in blue,
7
+ deleted text struck out — while preserving the original document's styling,
8
+ cross-references and custom classes.
9
+
10
+ ## Why
11
+
12
+ [`latexdiff`](https://www.ctan.org/pkg/latexdiff) is the de-facto standard for
13
+ review-marked LaTeX, but it diffs *token soup*: it guesses word boundaries with
14
+ regexes and re-injects markup into the merged text, hoping braces still
15
+ balance. On real-world documents the result is markup that crosses brace /
16
+ environment boundaries, glues old and new table cells together, and corrupts
17
+ listings — which is why every serious latexdiff user eventually ends up
18
+ maintaining a pile of fragile pre/post-processing scripts around it.
19
+
20
+ `texdiff` takes the other road: **diff the tree, never the text.**
21
+
22
+ 1. Parse old and new documents into a LaTeX syntax tree
23
+ ([pylatexenc](https://github.com/phfaist/pylatexenc) — a faithful concrete
24
+ syntax tree carrying exact source spans, no macro expansion).
25
+ 2. Align the trees level by level (document → environments → groups → text
26
+ runs) with difflib-style sequence matching. A change can never leak across
27
+ a node boundary.
28
+ 3. Walk the merged tree and emit `\DIFadd{...}` / `\DIFdel{...}` markup —
29
+ latexdiff-compatible rendering, valid by construction.
30
+
31
+ Restructured tables (re-ordered rows, changed column layouts), math,
32
+ listings and verbatim environments are handled as atomic or row-aligned
33
+ blocks instead of being shredded by a word diff.
34
+
35
+ Multi-file sources are flattened automatically (`\input`/`\include`
36
+ resolved relative to each file, like `latexdiff --flatten`), and the
37
+ preamble of the new revision is used verbatim so custom classes and
38
+ macros keep working.
39
+
40
+ ## Usage
41
+
42
+ ```sh
43
+ texdiff old/main.tex new/main.tex > diff.tex # \input expansion built in
44
+ pdflatex diff.tex # done - no pre/postprocessing
45
+ texdiff old.tex new.tex -o diff.tex --check # + fail loudly if it won't compile
46
+ texdiff old.tex new.tex --stats # "611 unchanged, 29 added, ..."
47
+ ```
48
+
49
+ ```sh
50
+ pip install texdiff # or from a checkout:
51
+ pip install -e ".[dev]" && pytest
52
+ ```
53
+
54
+ Status: **v0.2.0** — core pipeline plus flattening, row-granular table
55
+ alignment, preamble policy, `--check` and a compile-guaranteed corpus in
56
+ CI; validated on a 65-page generated longtable-heavy document. See the
57
+ [roadmap](https://gobobr.github.io/texdiff/roadmap/) for what is next.
58
+
59
+ ## License
60
+
61
+ MIT
@@ -0,0 +1,64 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "texdiff"
7
+ version = "0.2.0"
8
+ description = "AST-driven semantic diff for LaTeX documents"
9
+ readme = "README.md"
10
+ license = { file = "LICENSE" }
11
+ requires-python = ">=3.10"
12
+ authors = [{ name = "GoBobr" }]
13
+ keywords = ["latex", "diff", "latexdiff", "ast", "track-changes"]
14
+ classifiers = [
15
+ "Development Status :: 2 - Pre-Alpha",
16
+ "Intended Audience :: Science/Research",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Topic :: Text Processing :: Markup :: LaTeX",
23
+ ]
24
+ dependencies = [
25
+ "pylatexenc>=2.10",
26
+ ]
27
+
28
+ [project.optional-dependencies]
29
+ dev = [
30
+ "pytest>=7",
31
+ "pytest-cov>=4",
32
+ ]
33
+ docs = [
34
+ "mkdocs>=1.5",
35
+ "mkdocs-material>=9",
36
+ "mkdocstrings[python]>=0.24",
37
+ "pymdown-extensions>=10",
38
+ ]
39
+
40
+ [project.scripts]
41
+ texdiff = "texdiff.cli:main"
42
+
43
+ [project.urls]
44
+ Homepage = "https://github.com/GoBobr/texdiff"
45
+
46
+ [tool.setuptools.packages.find]
47
+ where = ["src"]
48
+
49
+ [tool.setuptools.package-data]
50
+ texdiff = ["py.typed"]
51
+
52
+ [tool.pytest.ini_options]
53
+ testpaths = ["tests"]
54
+ addopts = "--cov=texdiff --cov-report=term-missing --cov-fail-under=80"
55
+
56
+ [tool.coverage.run]
57
+ source = ["texdiff"]
58
+
59
+ [tool.coverage.report]
60
+ exclude_lines = [
61
+ "pragma: no cover",
62
+ "if TYPE_CHECKING:",
63
+ "raise NotImplementedError",
64
+ ]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,45 @@
1
+ """texdiff - AST-driven semantic diff for LaTeX documents.
2
+
3
+ The pipeline IS::
4
+
5
+ old.tex ─┐
6
+ ├─> parse → align trees → emit markup → diff.tex
7
+ new.tex ─┘
8
+
9
+ latexdiff diffs characters and breaks structure;
10
+ texdiff diffs structure and never breaks characters.
11
+ """
12
+
13
+ from .align import Delete, Edit, Insert, Match, Modify, align
14
+ from .api import DiffResult, DiffStats, diff_documents, diff_files
15
+ from .emit import LatexdiffMarkup, render
16
+ from .flatten import Flattener, flatten_file, flatten_source
17
+ from .nodes import Node
18
+ from .parse import ParseError, parse, parse_file
19
+ from .textdiff import Chunk, word_diff
20
+
21
+ __all__ = [
22
+ "align",
23
+ "Chunk",
24
+ "Delete",
25
+ "diff_documents",
26
+ "diff_files",
27
+ "DiffResult",
28
+ "DiffStats",
29
+ "Edit",
30
+ "Flattener",
31
+ "flatten_file",
32
+ "flatten_source",
33
+ "Insert",
34
+ "LatexdiffMarkup",
35
+ "Match",
36
+ "Modify",
37
+ "Node",
38
+ "parse",
39
+ "ParseError",
40
+ "parse_file",
41
+ "render",
42
+ "word_diff",
43
+ ]
44
+
45
+ __version__ = "0.1.0"
@@ -0,0 +1,202 @@
1
+ """Tree alignment: match two node lists into aligned pairs and edits.
2
+
3
+ The aligner is where latexdiff's heuristics become data. It produces,
4
+ for a pair of node lists, an edit script of three shapes:
5
+
6
+ * ``match(old, new)`` - same signature, identical text → keep verbatim
7
+ * ``modify(old, new)`` - same signature, different text → recurse if
8
+ both sides have children (block diff), else whole-block replacement
9
+ * ``insert(new)`` / ``delete(old)`` - no counterpart
10
+
11
+ Uses :class:`difflib.SequenceMatcher` over node signatures; runs of
12
+ equal signature with different text become ``modify`` pairs aligned
13
+ positionally within the run.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from dataclasses import dataclass
19
+ from difflib import SequenceMatcher
20
+
21
+ from .nodes import Node
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class Match:
26
+ """A node kept verbatim (identical on both sides)."""
27
+
28
+ node: Node
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class Modify:
33
+ """A node pair to be diffed further (recursed) - or replaced.
34
+
35
+ Attributes:
36
+ old: the node from the old revision.
37
+ new: the node from the new revision.
38
+ inner: the recursive edit script of the children, when both
39
+ sides are recursable; ``None`` means whole-block replace.
40
+ """
41
+
42
+ old: Node
43
+ new: Node
44
+ inner: "list[Edit] | None" = None
45
+
46
+
47
+ @dataclass(frozen=True)
48
+ class Insert:
49
+ """A node present only in the new version."""
50
+
51
+ new: Node
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class Delete:
56
+ """A node present only in the old version."""
57
+
58
+ old: Node
59
+
60
+
61
+ Edit = Match | Modify | Insert | Delete
62
+
63
+
64
+ def align(old: list[Node], new: list[Node]) -> list[Edit]:
65
+ """Align two node lists into an edit script.
66
+
67
+ The returned script is in document order, interleaving old and new
68
+ positions like a unified diff.
69
+
70
+ Algorithm:
71
+
72
+ 1. an equal *prefix* and *suffix* of signatures anchors the
73
+ alignment first: ``SequenceMatcher`` finds the globally
74
+ longest matching block, which can let a later identical run
75
+ steal the match from the earliest identical nodes - an
76
+ inserted chapter between two near-identical passages then
77
+ drags unchanged leading lines into the insertion (they
78
+ render as added);
79
+ 2. ``SequenceMatcher`` over the remaining node *signatures*
80
+ yields runs of same-signature positions deemed "matching";
81
+ 3. within a run, nodes are paired positionally: identical source
82
+ text becomes :class:`Match`, differing text :class:`Modify`;
83
+ 4. gap regions become ``Delete`` (old-only) then ``Insert``
84
+ (new-only) sequences, preserving document order.
85
+ """
86
+
87
+ def _pair(o: Node, n: Node) -> Edit:
88
+ if o.text == n.text:
89
+ return Match(node=o)
90
+ if o.kind == "row" and n.kind == "row" and o.name == n.name:
91
+ # equal row signature (content key) with differing text
92
+ # means the difference is purely structural: \hline on
93
+ # the other side of the row, whitespace, comments.
94
+ # Duplicating such a pair (del + add) would emit the
95
+ # table header material (\endfirsthead, \endhead ...)
96
+ # twice inside one longtable, which can throw longtable
97
+ # into an infinite loop. Keep the old row verbatim.
98
+ return Match(node=o)
99
+ return Modify(old=o, new=n)
100
+
101
+ # 1. common prefix, paired positionally (identical to what a
102
+ # matching-block run does - only the anchoring is stronger)
103
+ pre = 0
104
+ while (
105
+ pre < len(old)
106
+ and pre < len(new)
107
+ and old[pre].signature() == new[pre].signature()
108
+ ):
109
+ pre += 1
110
+ # common suffix (may not overlap the prefix)
111
+ suf = 0
112
+ while (
113
+ suf < len(old) - pre
114
+ and suf < len(new) - pre
115
+ and old[len(old) - 1 - suf].signature() == new[len(new) - 1 - suf].signature()
116
+ ):
117
+ suf += 1
118
+ mid_old = old[pre : len(old) - suf]
119
+ mid_new = new[pre : len(new) - suf]
120
+
121
+ sm = SequenceMatcher(
122
+ a=[n.signature() for n in mid_old],
123
+ b=[n.signature() for n in mid_new],
124
+ autojunk=False,
125
+ )
126
+
127
+ edits: list[Edit] = [_pair(o, n) for o, n in zip(old[:pre], new[:pre])]
128
+ prev_a = prev_b = 0
129
+ for block in sm.get_matching_blocks():
130
+ # gap before this matching block: deletions then insertions.
131
+ # Block-head refinement: ``SequenceMatcher`` extends a matching
132
+ # block as far as equal *signatures* reach - with the wildcard
133
+ # ``text`` signature, a run may begin by pairing two text nodes
134
+ # whose contents share nothing, while the node that should pair
135
+ # with the old head sits at the START of the insertion gap just
136
+ # before the block. When that pairing is clearly better, consume
137
+ # the gap head into a Modify pair, emit the rest of the gap as
138
+ # inserts, and let the block start one position later on both
139
+ # sides (its first new-side node then lands at the gap end).
140
+ new_gap = mid_new[prev_b : block.b]
141
+ head_a, head_b = block.a, block.b
142
+ if (
143
+ block.size
144
+ and new_gap
145
+ and mid_old[block.a].kind == "text"
146
+ and new_gap[0].kind == "text"
147
+ and mid_new[block.b].kind == "text"
148
+ ):
149
+ o_head = mid_old[block.a]
150
+ best = max(
151
+ range(min(len(new_gap), 4)),
152
+ key=lambda k: _text_similarity(o_head.text, new_gap[k].text),
153
+ )
154
+ n_head = new_gap[best]
155
+ if _text_similarity(o_head.text, n_head.text) > (
156
+ _text_similarity(o_head.text, mid_new[block.b].text) + 0.2
157
+ ):
158
+ edits.extend(Delete(old=n) for n in mid_old[prev_a : block.a])
159
+ edits.extend(Insert(new=n) for n in new_gap[:best])
160
+ edits.append(_pair(o_head, n_head))
161
+ edits.extend(Insert(new=n) for n in new_gap[best + 1 :])
162
+ edits.append(Insert(new=mid_new[block.b]))
163
+ head_a, head_b = block.a + 1, block.b + 1
164
+ for i in range(1, block.size):
165
+ edits.append(
166
+ _pair(mid_old[block.a + i], mid_new[block.b + i])
167
+ )
168
+ prev_a, prev_b = block.a + block.size, block.b + block.size
169
+ continue
170
+ edits.extend(Delete(old=n) for n in mid_old[prev_a : block.a])
171
+ edits.extend(Insert(new=n) for n in new_gap)
172
+ # the matching run: pair positionally, decide Match vs Modify
173
+ for i in range(head_a, head_a + block.size):
174
+ edits.append(
175
+ _pair(mid_old[i], mid_new[head_b + (i - head_a)])
176
+ )
177
+ prev_a, prev_b = head_a + block.size, head_b + block.size
178
+ edits.extend(Delete(old=n) for n in mid_old[prev_a:])
179
+ edits.extend(Insert(new=n) for n in mid_new[prev_b:])
180
+ # common suffix, in document order
181
+ edits.extend(
182
+ _pair(o, n)
183
+ for o, n in zip(
184
+ old[len(old) - suf :], new[len(new) - suf :]
185
+ )
186
+ )
187
+ return edits
188
+
189
+
190
+ def _text_similarity(a: str, b: str) -> float:
191
+ """Quick content similarity of two text runs, in ``[0, 1]``.
192
+
193
+ Used only to compare pairing candidates at matching-block
194
+ boundaries, so a cheap ``SequenceMatcher.ratio`` over word
195
+ tokens (not raw characters) is enough - and stays stable for
196
+ long runs where character noise would dominate.
197
+ """
198
+ wa = [t for t in a.split() if t]
199
+ wb = [t for t in b.split() if t]
200
+ if not wa or not wb:
201
+ return 0.0
202
+ return SequenceMatcher(a=wa, b=wb, autojunk=False).ratio()