texdiff 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- texdiff-0.2.0/LICENSE +21 -0
- texdiff-0.2.0/PKG-INFO +112 -0
- texdiff-0.2.0/README.md +61 -0
- texdiff-0.2.0/pyproject.toml +64 -0
- texdiff-0.2.0/setup.cfg +4 -0
- texdiff-0.2.0/src/texdiff/__init__.py +45 -0
- texdiff-0.2.0/src/texdiff/align.py +202 -0
- texdiff-0.2.0/src/texdiff/api.py +326 -0
- texdiff-0.2.0/src/texdiff/check.py +61 -0
- texdiff-0.2.0/src/texdiff/cli.py +87 -0
- texdiff-0.2.0/src/texdiff/emit.py +1193 -0
- texdiff-0.2.0/src/texdiff/flatten.py +185 -0
- texdiff-0.2.0/src/texdiff/nodes.py +147 -0
- texdiff-0.2.0/src/texdiff/oldlines.py +49 -0
- texdiff-0.2.0/src/texdiff/parse.py +524 -0
- texdiff-0.2.0/src/texdiff/preamble.py +388 -0
- texdiff-0.2.0/src/texdiff/tables.py +540 -0
- texdiff-0.2.0/src/texdiff/textdiff.py +157 -0
- texdiff-0.2.0/src/texdiff.egg-info/PKG-INFO +112 -0
- texdiff-0.2.0/src/texdiff.egg-info/SOURCES.txt +36 -0
- texdiff-0.2.0/src/texdiff.egg-info/dependency_links.txt +1 -0
- texdiff-0.2.0/src/texdiff.egg-info/entry_points.txt +2 -0
- texdiff-0.2.0/src/texdiff.egg-info/requires.txt +11 -0
- texdiff-0.2.0/src/texdiff.egg-info/top_level.txt +1 -0
- texdiff-0.2.0/tests/test_align.py +71 -0
- texdiff-0.2.0/tests/test_alignment.py +586 -0
- texdiff-0.2.0/tests/test_api.py +135 -0
- texdiff-0.2.0/tests/test_check.py +125 -0
- texdiff-0.2.0/tests/test_cli.py +64 -0
- texdiff-0.2.0/tests/test_corpus.py +111 -0
- texdiff-0.2.0/tests/test_emit.py +119 -0
- texdiff-0.2.0/tests/test_flatten.py +159 -0
- texdiff-0.2.0/tests/test_parse.py +132 -0
- texdiff-0.2.0/tests/test_preamble.py +83 -0
- texdiff-0.2.0/tests/test_tables.py +160 -0
- texdiff-0.2.0/tests/test_tables_render.py +324 -0
- texdiff-0.2.0/tests/test_textdiff.py +61 -0
- texdiff-0.2.0/tests/test_verbatim.py +207 -0
texdiff-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 GoBobr
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
texdiff-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: texdiff
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: AST-driven semantic diff for LaTeX documents
|
|
5
|
+
Author: GoBobr
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2026 GoBobr
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Project-URL: Homepage, https://github.com/GoBobr/texdiff
|
|
29
|
+
Keywords: latex,diff,latexdiff,ast,track-changes
|
|
30
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
31
|
+
Classifier: Intended Audience :: Science/Research
|
|
32
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
33
|
+
Classifier: Programming Language :: Python :: 3
|
|
34
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
37
|
+
Classifier: Topic :: Text Processing :: Markup :: LaTeX
|
|
38
|
+
Requires-Python: >=3.10
|
|
39
|
+
Description-Content-Type: text/markdown
|
|
40
|
+
License-File: LICENSE
|
|
41
|
+
Requires-Dist: pylatexenc>=2.10
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest-cov>=4; extra == "dev"
|
|
45
|
+
Provides-Extra: docs
|
|
46
|
+
Requires-Dist: mkdocs>=1.5; extra == "docs"
|
|
47
|
+
Requires-Dist: mkdocs-material>=9; extra == "docs"
|
|
48
|
+
Requires-Dist: mkdocstrings[python]>=0.24; extra == "docs"
|
|
49
|
+
Requires-Dist: pymdown-extensions>=10; extra == "docs"
|
|
50
|
+
Dynamic: license-file
|
|
51
|
+
|
|
52
|
+
# texdiff
|
|
53
|
+
|
|
54
|
+
**AST-driven semantic diff for LaTeX documents.**
|
|
55
|
+
|
|
56
|
+
`texdiff` compares two revisions of a LaTeX document and produces a single
|
|
57
|
+
marked-up `diff.tex` that compiles with `pdflatex` — added text in blue,
|
|
58
|
+
deleted text struck out — while preserving the original document's styling,
|
|
59
|
+
cross-references and custom classes.
|
|
60
|
+
|
|
61
|
+
## Why
|
|
62
|
+
|
|
63
|
+
[`latexdiff`](https://www.ctan.org/pkg/latexdiff) is the de-facto standard for
|
|
64
|
+
review-marked LaTeX, but it diffs *token soup*: it guesses word boundaries with
|
|
65
|
+
regexes and re-injects markup into the merged text, hoping braces still
|
|
66
|
+
balance. On real-world documents the result is markup that crosses brace /
|
|
67
|
+
environment boundaries, glues old and new table cells together, and corrupts
|
|
68
|
+
listings — which is why every serious latexdiff user eventually ends up
|
|
69
|
+
maintaining a pile of fragile pre/post-processing scripts around it.
|
|
70
|
+
|
|
71
|
+
`texdiff` takes the other road: **diff the tree, never the text.**
|
|
72
|
+
|
|
73
|
+
1. Parse old and new documents into a LaTeX syntax tree
|
|
74
|
+
([pylatexenc](https://github.com/phfaist/pylatexenc) — a faithful concrete
|
|
75
|
+
syntax tree carrying exact source spans, no macro expansion).
|
|
76
|
+
2. Align the trees level by level (document → environments → groups → text
|
|
77
|
+
runs) with difflib-style sequence matching. A change can never leak across
|
|
78
|
+
a node boundary.
|
|
79
|
+
3. Walk the merged tree and emit `\DIFadd{...}` / `\DIFdel{...}` markup —
|
|
80
|
+
latexdiff-compatible rendering, valid by construction.
|
|
81
|
+
|
|
82
|
+
Restructured tables (re-ordered rows, changed column layouts), math,
|
|
83
|
+
listings and verbatim environments are handled as atomic or row-aligned
|
|
84
|
+
blocks instead of being shredded by a word diff.
|
|
85
|
+
|
|
86
|
+
Multi-file sources are flattened automatically (`\input`/`\include`
|
|
87
|
+
resolved relative to each file, like `latexdiff --flatten`), and the
|
|
88
|
+
preamble of the new revision is used verbatim so custom classes and
|
|
89
|
+
macros keep working.
|
|
90
|
+
|
|
91
|
+
## Usage
|
|
92
|
+
|
|
93
|
+
```sh
|
|
94
|
+
texdiff old/main.tex new/main.tex > diff.tex # \input expansion built in
|
|
95
|
+
pdflatex diff.tex # done - no pre/postprocessing
|
|
96
|
+
texdiff old.tex new.tex -o diff.tex --check # + fail loudly if it won't compile
|
|
97
|
+
texdiff old.tex new.tex --stats # "611 unchanged, 29 added, ..."
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
```sh
|
|
101
|
+
pip install texdiff # or from a checkout:
|
|
102
|
+
pip install -e ".[dev]" && pytest
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Status: **v0.2.0** — core pipeline plus flattening, row-granular table
|
|
106
|
+
alignment, preamble policy, `--check` and a compile-guaranteed corpus in
|
|
107
|
+
CI; validated on a 65-page generated longtable-heavy document. See the
|
|
108
|
+
[roadmap](https://gobobr.github.io/texdiff/roadmap/) for what is next.
|
|
109
|
+
|
|
110
|
+
## License
|
|
111
|
+
|
|
112
|
+
MIT
|
texdiff-0.2.0/README.md
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
# texdiff
|
|
2
|
+
|
|
3
|
+
**AST-driven semantic diff for LaTeX documents.**
|
|
4
|
+
|
|
5
|
+
`texdiff` compares two revisions of a LaTeX document and produces a single
|
|
6
|
+
marked-up `diff.tex` that compiles with `pdflatex` — added text in blue,
|
|
7
|
+
deleted text struck out — while preserving the original document's styling,
|
|
8
|
+
cross-references and custom classes.
|
|
9
|
+
|
|
10
|
+
## Why
|
|
11
|
+
|
|
12
|
+
[`latexdiff`](https://www.ctan.org/pkg/latexdiff) is the de-facto standard for
|
|
13
|
+
review-marked LaTeX, but it diffs *token soup*: it guesses word boundaries with
|
|
14
|
+
regexes and re-injects markup into the merged text, hoping braces still
|
|
15
|
+
balance. On real-world documents the result is markup that crosses brace /
|
|
16
|
+
environment boundaries, glues old and new table cells together, and corrupts
|
|
17
|
+
listings — which is why every serious latexdiff user eventually ends up
|
|
18
|
+
maintaining a pile of fragile pre/post-processing scripts around it.
|
|
19
|
+
|
|
20
|
+
`texdiff` takes the other road: **diff the tree, never the text.**
|
|
21
|
+
|
|
22
|
+
1. Parse old and new documents into a LaTeX syntax tree
|
|
23
|
+
([pylatexenc](https://github.com/phfaist/pylatexenc) — a faithful concrete
|
|
24
|
+
syntax tree carrying exact source spans, no macro expansion).
|
|
25
|
+
2. Align the trees level by level (document → environments → groups → text
|
|
26
|
+
runs) with difflib-style sequence matching. A change can never leak across
|
|
27
|
+
a node boundary.
|
|
28
|
+
3. Walk the merged tree and emit `\DIFadd{...}` / `\DIFdel{...}` markup —
|
|
29
|
+
latexdiff-compatible rendering, valid by construction.
|
|
30
|
+
|
|
31
|
+
Restructured tables (re-ordered rows, changed column layouts), math,
|
|
32
|
+
listings and verbatim environments are handled as atomic or row-aligned
|
|
33
|
+
blocks instead of being shredded by a word diff.
|
|
34
|
+
|
|
35
|
+
Multi-file sources are flattened automatically (`\input`/`\include`
|
|
36
|
+
resolved relative to each file, like `latexdiff --flatten`), and the
|
|
37
|
+
preamble of the new revision is used verbatim so custom classes and
|
|
38
|
+
macros keep working.
|
|
39
|
+
|
|
40
|
+
## Usage
|
|
41
|
+
|
|
42
|
+
```sh
|
|
43
|
+
texdiff old/main.tex new/main.tex > diff.tex # \input expansion built in
|
|
44
|
+
pdflatex diff.tex # done - no pre/postprocessing
|
|
45
|
+
texdiff old.tex new.tex -o diff.tex --check # + fail loudly if it won't compile
|
|
46
|
+
texdiff old.tex new.tex --stats # "611 unchanged, 29 added, ..."
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
```sh
|
|
50
|
+
pip install texdiff # or from a checkout:
|
|
51
|
+
pip install -e ".[dev]" && pytest
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Status: **v0.2.0** — core pipeline plus flattening, row-granular table
|
|
55
|
+
alignment, preamble policy, `--check` and a compile-guaranteed corpus in
|
|
56
|
+
CI; validated on a 65-page generated longtable-heavy document. See the
|
|
57
|
+
[roadmap](https://gobobr.github.io/texdiff/roadmap/) for what is next.
|
|
58
|
+
|
|
59
|
+
## License
|
|
60
|
+
|
|
61
|
+
MIT
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "texdiff"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "AST-driven semantic diff for LaTeX documents"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { file = "LICENSE" }
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
authors = [{ name = "GoBobr" }]
|
|
13
|
+
keywords = ["latex", "diff", "latexdiff", "ast", "track-changes"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Topic :: Text Processing :: Markup :: LaTeX",
|
|
23
|
+
]
|
|
24
|
+
dependencies = [
|
|
25
|
+
"pylatexenc>=2.10",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
dev = [
|
|
30
|
+
"pytest>=7",
|
|
31
|
+
"pytest-cov>=4",
|
|
32
|
+
]
|
|
33
|
+
docs = [
|
|
34
|
+
"mkdocs>=1.5",
|
|
35
|
+
"mkdocs-material>=9",
|
|
36
|
+
"mkdocstrings[python]>=0.24",
|
|
37
|
+
"pymdown-extensions>=10",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
[project.scripts]
|
|
41
|
+
texdiff = "texdiff.cli:main"
|
|
42
|
+
|
|
43
|
+
[project.urls]
|
|
44
|
+
Homepage = "https://github.com/GoBobr/texdiff"
|
|
45
|
+
|
|
46
|
+
[tool.setuptools.packages.find]
|
|
47
|
+
where = ["src"]
|
|
48
|
+
|
|
49
|
+
[tool.setuptools.package-data]
|
|
50
|
+
texdiff = ["py.typed"]
|
|
51
|
+
|
|
52
|
+
[tool.pytest.ini_options]
|
|
53
|
+
testpaths = ["tests"]
|
|
54
|
+
addopts = "--cov=texdiff --cov-report=term-missing --cov-fail-under=80"
|
|
55
|
+
|
|
56
|
+
[tool.coverage.run]
|
|
57
|
+
source = ["texdiff"]
|
|
58
|
+
|
|
59
|
+
[tool.coverage.report]
|
|
60
|
+
exclude_lines = [
|
|
61
|
+
"pragma: no cover",
|
|
62
|
+
"if TYPE_CHECKING:",
|
|
63
|
+
"raise NotImplementedError",
|
|
64
|
+
]
|
texdiff-0.2.0/setup.cfg
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""texdiff - AST-driven semantic diff for LaTeX documents.
|
|
2
|
+
|
|
3
|
+
The pipeline IS::
|
|
4
|
+
|
|
5
|
+
old.tex ─┐
|
|
6
|
+
├─> parse → align trees → emit markup → diff.tex
|
|
7
|
+
new.tex ─┘
|
|
8
|
+
|
|
9
|
+
latexdiff diffs characters and breaks structure;
|
|
10
|
+
texdiff diffs structure and never breaks characters.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from .align import Delete, Edit, Insert, Match, Modify, align
|
|
14
|
+
from .api import DiffResult, DiffStats, diff_documents, diff_files
|
|
15
|
+
from .emit import LatexdiffMarkup, render
|
|
16
|
+
from .flatten import Flattener, flatten_file, flatten_source
|
|
17
|
+
from .nodes import Node
|
|
18
|
+
from .parse import ParseError, parse, parse_file
|
|
19
|
+
from .textdiff import Chunk, word_diff
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
"align",
|
|
23
|
+
"Chunk",
|
|
24
|
+
"Delete",
|
|
25
|
+
"diff_documents",
|
|
26
|
+
"diff_files",
|
|
27
|
+
"DiffResult",
|
|
28
|
+
"DiffStats",
|
|
29
|
+
"Edit",
|
|
30
|
+
"Flattener",
|
|
31
|
+
"flatten_file",
|
|
32
|
+
"flatten_source",
|
|
33
|
+
"Insert",
|
|
34
|
+
"LatexdiffMarkup",
|
|
35
|
+
"Match",
|
|
36
|
+
"Modify",
|
|
37
|
+
"Node",
|
|
38
|
+
"parse",
|
|
39
|
+
"ParseError",
|
|
40
|
+
"parse_file",
|
|
41
|
+
"render",
|
|
42
|
+
"word_diff",
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""Tree alignment: match two node lists into aligned pairs and edits.
|
|
2
|
+
|
|
3
|
+
The aligner is where latexdiff's heuristics become data. It produces,
|
|
4
|
+
for a pair of node lists, an edit script of three shapes:
|
|
5
|
+
|
|
6
|
+
* ``match(old, new)`` - same signature, identical text → keep verbatim
|
|
7
|
+
* ``modify(old, new)`` - same signature, different text → recurse if
|
|
8
|
+
both sides have children (block diff), else whole-block replacement
|
|
9
|
+
* ``insert(new)`` / ``delete(old)`` - no counterpart
|
|
10
|
+
|
|
11
|
+
Uses :class:`difflib.SequenceMatcher` over node signatures; runs of
|
|
12
|
+
equal signature with different text become ``modify`` pairs aligned
|
|
13
|
+
positionally within the run.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from difflib import SequenceMatcher
|
|
20
|
+
|
|
21
|
+
from .nodes import Node
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class Match:
|
|
26
|
+
"""A node kept verbatim (identical on both sides)."""
|
|
27
|
+
|
|
28
|
+
node: Node
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class Modify:
|
|
33
|
+
"""A node pair to be diffed further (recursed) - or replaced.
|
|
34
|
+
|
|
35
|
+
Attributes:
|
|
36
|
+
old: the node from the old revision.
|
|
37
|
+
new: the node from the new revision.
|
|
38
|
+
inner: the recursive edit script of the children, when both
|
|
39
|
+
sides are recursable; ``None`` means whole-block replace.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
old: Node
|
|
43
|
+
new: Node
|
|
44
|
+
inner: "list[Edit] | None" = None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True)
|
|
48
|
+
class Insert:
|
|
49
|
+
"""A node present only in the new version."""
|
|
50
|
+
|
|
51
|
+
new: Node
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass(frozen=True)
|
|
55
|
+
class Delete:
|
|
56
|
+
"""A node present only in the old version."""
|
|
57
|
+
|
|
58
|
+
old: Node
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
Edit = Match | Modify | Insert | Delete
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def align(old: list[Node], new: list[Node]) -> list[Edit]:
|
|
65
|
+
"""Align two node lists into an edit script.
|
|
66
|
+
|
|
67
|
+
The returned script is in document order, interleaving old and new
|
|
68
|
+
positions like a unified diff.
|
|
69
|
+
|
|
70
|
+
Algorithm:
|
|
71
|
+
|
|
72
|
+
1. an equal *prefix* and *suffix* of signatures anchors the
|
|
73
|
+
alignment first: ``SequenceMatcher`` finds the globally
|
|
74
|
+
longest matching block, which can let a later identical run
|
|
75
|
+
steal the match from the earliest identical nodes - an
|
|
76
|
+
inserted chapter between two near-identical passages then
|
|
77
|
+
drags unchanged leading lines into the insertion (they
|
|
78
|
+
render as added);
|
|
79
|
+
2. ``SequenceMatcher`` over the remaining node *signatures*
|
|
80
|
+
yields runs of same-signature positions deemed "matching";
|
|
81
|
+
3. within a run, nodes are paired positionally: identical source
|
|
82
|
+
text becomes :class:`Match`, differing text :class:`Modify`;
|
|
83
|
+
4. gap regions become ``Delete`` (old-only) then ``Insert``
|
|
84
|
+
(new-only) sequences, preserving document order.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
def _pair(o: Node, n: Node) -> Edit:
|
|
88
|
+
if o.text == n.text:
|
|
89
|
+
return Match(node=o)
|
|
90
|
+
if o.kind == "row" and n.kind == "row" and o.name == n.name:
|
|
91
|
+
# equal row signature (content key) with differing text
|
|
92
|
+
# means the difference is purely structural: \hline on
|
|
93
|
+
# the other side of the row, whitespace, comments.
|
|
94
|
+
# Duplicating such a pair (del + add) would emit the
|
|
95
|
+
# table header material (\endfirsthead, \endhead ...)
|
|
96
|
+
# twice inside one longtable, which can throw longtable
|
|
97
|
+
# into an infinite loop. Keep the old row verbatim.
|
|
98
|
+
return Match(node=o)
|
|
99
|
+
return Modify(old=o, new=n)
|
|
100
|
+
|
|
101
|
+
# 1. common prefix, paired positionally (identical to what a
|
|
102
|
+
# matching-block run does - only the anchoring is stronger)
|
|
103
|
+
pre = 0
|
|
104
|
+
while (
|
|
105
|
+
pre < len(old)
|
|
106
|
+
and pre < len(new)
|
|
107
|
+
and old[pre].signature() == new[pre].signature()
|
|
108
|
+
):
|
|
109
|
+
pre += 1
|
|
110
|
+
# common suffix (may not overlap the prefix)
|
|
111
|
+
suf = 0
|
|
112
|
+
while (
|
|
113
|
+
suf < len(old) - pre
|
|
114
|
+
and suf < len(new) - pre
|
|
115
|
+
and old[len(old) - 1 - suf].signature() == new[len(new) - 1 - suf].signature()
|
|
116
|
+
):
|
|
117
|
+
suf += 1
|
|
118
|
+
mid_old = old[pre : len(old) - suf]
|
|
119
|
+
mid_new = new[pre : len(new) - suf]
|
|
120
|
+
|
|
121
|
+
sm = SequenceMatcher(
|
|
122
|
+
a=[n.signature() for n in mid_old],
|
|
123
|
+
b=[n.signature() for n in mid_new],
|
|
124
|
+
autojunk=False,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
edits: list[Edit] = [_pair(o, n) for o, n in zip(old[:pre], new[:pre])]
|
|
128
|
+
prev_a = prev_b = 0
|
|
129
|
+
for block in sm.get_matching_blocks():
|
|
130
|
+
# gap before this matching block: deletions then insertions.
|
|
131
|
+
# Block-head refinement: ``SequenceMatcher`` extends a matching
|
|
132
|
+
# block as far as equal *signatures* reach - with the wildcard
|
|
133
|
+
# ``text`` signature, a run may begin by pairing two text nodes
|
|
134
|
+
# whose contents share nothing, while the node that should pair
|
|
135
|
+
# with the old head sits at the START of the insertion gap just
|
|
136
|
+
# before the block. When that pairing is clearly better, consume
|
|
137
|
+
# the gap head into a Modify pair, emit the rest of the gap as
|
|
138
|
+
# inserts, and let the block start one position later on both
|
|
139
|
+
# sides (its first new-side node then lands at the gap end).
|
|
140
|
+
new_gap = mid_new[prev_b : block.b]
|
|
141
|
+
head_a, head_b = block.a, block.b
|
|
142
|
+
if (
|
|
143
|
+
block.size
|
|
144
|
+
and new_gap
|
|
145
|
+
and mid_old[block.a].kind == "text"
|
|
146
|
+
and new_gap[0].kind == "text"
|
|
147
|
+
and mid_new[block.b].kind == "text"
|
|
148
|
+
):
|
|
149
|
+
o_head = mid_old[block.a]
|
|
150
|
+
best = max(
|
|
151
|
+
range(min(len(new_gap), 4)),
|
|
152
|
+
key=lambda k: _text_similarity(o_head.text, new_gap[k].text),
|
|
153
|
+
)
|
|
154
|
+
n_head = new_gap[best]
|
|
155
|
+
if _text_similarity(o_head.text, n_head.text) > (
|
|
156
|
+
_text_similarity(o_head.text, mid_new[block.b].text) + 0.2
|
|
157
|
+
):
|
|
158
|
+
edits.extend(Delete(old=n) for n in mid_old[prev_a : block.a])
|
|
159
|
+
edits.extend(Insert(new=n) for n in new_gap[:best])
|
|
160
|
+
edits.append(_pair(o_head, n_head))
|
|
161
|
+
edits.extend(Insert(new=n) for n in new_gap[best + 1 :])
|
|
162
|
+
edits.append(Insert(new=mid_new[block.b]))
|
|
163
|
+
head_a, head_b = block.a + 1, block.b + 1
|
|
164
|
+
for i in range(1, block.size):
|
|
165
|
+
edits.append(
|
|
166
|
+
_pair(mid_old[block.a + i], mid_new[block.b + i])
|
|
167
|
+
)
|
|
168
|
+
prev_a, prev_b = block.a + block.size, block.b + block.size
|
|
169
|
+
continue
|
|
170
|
+
edits.extend(Delete(old=n) for n in mid_old[prev_a : block.a])
|
|
171
|
+
edits.extend(Insert(new=n) for n in new_gap)
|
|
172
|
+
# the matching run: pair positionally, decide Match vs Modify
|
|
173
|
+
for i in range(head_a, head_a + block.size):
|
|
174
|
+
edits.append(
|
|
175
|
+
_pair(mid_old[i], mid_new[head_b + (i - head_a)])
|
|
176
|
+
)
|
|
177
|
+
prev_a, prev_b = head_a + block.size, head_b + block.size
|
|
178
|
+
edits.extend(Delete(old=n) for n in mid_old[prev_a:])
|
|
179
|
+
edits.extend(Insert(new=n) for n in mid_new[prev_b:])
|
|
180
|
+
# common suffix, in document order
|
|
181
|
+
edits.extend(
|
|
182
|
+
_pair(o, n)
|
|
183
|
+
for o, n in zip(
|
|
184
|
+
old[len(old) - suf :], new[len(new) - suf :]
|
|
185
|
+
)
|
|
186
|
+
)
|
|
187
|
+
return edits
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _text_similarity(a: str, b: str) -> float:
|
|
191
|
+
"""Quick content similarity of two text runs, in ``[0, 1]``.
|
|
192
|
+
|
|
193
|
+
Used only to compare pairing candidates at matching-block
|
|
194
|
+
boundaries, so a cheap ``SequenceMatcher.ratio`` over word
|
|
195
|
+
tokens (not raw characters) is enough - and stays stable for
|
|
196
|
+
long runs where character noise would dominate.
|
|
197
|
+
"""
|
|
198
|
+
wa = [t for t in a.split() if t]
|
|
199
|
+
wb = [t for t in b.split() if t]
|
|
200
|
+
if not wa or not wb:
|
|
201
|
+
return 0.0
|
|
202
|
+
return SequenceMatcher(a=wa, b=wb, autojunk=False).ratio()
|