texdiff 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- texdiff/__init__.py +45 -0
- texdiff/align.py +202 -0
- texdiff/api.py +326 -0
- texdiff/check.py +61 -0
- texdiff/cli.py +87 -0
- texdiff/emit.py +1193 -0
- texdiff/flatten.py +185 -0
- texdiff/nodes.py +147 -0
- texdiff/oldlines.py +49 -0
- texdiff/parse.py +524 -0
- texdiff/preamble.py +388 -0
- texdiff/tables.py +540 -0
- texdiff/textdiff.py +157 -0
- texdiff-0.2.0.dist-info/METADATA +112 -0
- texdiff-0.2.0.dist-info/RECORD +19 -0
- texdiff-0.2.0.dist-info/WHEEL +5 -0
- texdiff-0.2.0.dist-info/entry_points.txt +2 -0
- texdiff-0.2.0.dist-info/licenses/LICENSE +21 -0
- texdiff-0.2.0.dist-info/top_level.txt +1 -0
texdiff/tables.py
ADDED
|
@@ -0,0 +1,540 @@
|
|
|
1
|
+
"""Wholesale rendering of heavily restructured tables.
|
|
2
|
+
|
|
3
|
+
Ports the behaviour of the reference ``sanitize-tables.pl`` toolkit to
|
|
4
|
+
generic texdiff: when one logical table's content changed so much that
|
|
5
|
+
row-granular markup is noise (80 %+ row churn), the change reads best
|
|
6
|
+
as *two whole tables* - the old revision's table struck through in
|
|
7
|
+
red, the new revision's in blue. latexdiff's own row markup for such
|
|
8
|
+
configurations produces pathological glue; the reference build
|
|
9
|
+
solves that by locating each clean source table and rendering it
|
|
10
|
+
wholesale.
|
|
11
|
+
|
|
12
|
+
Components:
|
|
13
|
+
|
|
14
|
+
* :func:`is_restructured` - classify an old/new table-text pair by
|
|
15
|
+
data-row churn (the reference thresholds: >= 80 % relative
|
|
16
|
+
difference AND >= 3 rows apart);
|
|
17
|
+
* :func:`is_pathological` - inline row markup of a pair would be
|
|
18
|
+
unreadable interleaving (word similarity below
|
|
19
|
+
:data:`PATHOLOGICAL_WORD_SIMILARITY`); such pairs take the
|
|
20
|
+
wholesale or row-merged rendering instead;
|
|
21
|
+
* :func:`merge_tables` - the row-merged rendering: one table, common
|
|
22
|
+
rows black, removed rows red-struck, added rows blue (the reference
|
|
23
|
+
``merged_rows``, LCS over normalized rows with the 0.3 / 0.5 / 0.5
|
|
24
|
+
thresholds);
|
|
25
|
+
* :func:`render_restructured` - the wholesale renderer: the old side
|
|
26
|
+
as :func:`strike_through` (red, per-word so wide W{} cells can
|
|
27
|
+
still wrap), the new side unmarked inside a blue colour group.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import re
|
|
33
|
+
from difflib import SequenceMatcher
|
|
34
|
+
|
|
35
|
+
# reference thresholds (sanitize-tables.pl): a table pair is
|
|
36
|
+
# "restructured" when the data-row counts differ by >= 80 % relative
|
|
37
|
+
# AND >= 3 absolute rows
|
|
38
|
+
RETIRE_DEL_FRAC = 0.80
|
|
39
|
+
RETIRE_MIN_ROWS = 3
|
|
40
|
+
|
|
41
|
+
# when the word-level similarity of the data rows falls below this,
|
|
42
|
+
# per-cell inline markup degenerates into an unreadable red/blue
|
|
43
|
+
# interleaving - the reference build calls such blocks "pathological"
|
|
44
|
+
# and replaces them wholesale (calibrated so well-matching tables
|
|
45
|
+
# like hand-edited ICD longtables, similarity ~ 0.97, keep their
|
|
46
|
+
# inline markup)
|
|
47
|
+
PATHOLOGICAL_WORD_SIMILARITY = 0.50
|
|
48
|
+
|
|
49
|
+
# row-merge acceptance (reference): the LCS must pair >= 30 % of the
|
|
50
|
+
# smaller row set, <= 50 % of the old rows may stay unpaired, and a
|
|
51
|
+
# merge is rejected when >= 50 % of the unmatched old rows reappear
|
|
52
|
+
# verbatim among the unmatched new rows
|
|
53
|
+
MIN_MATCH_FRAC = 0.30
|
|
54
|
+
MAX_UNMATCHED_FRAC = 0.50
|
|
55
|
+
REAPPEAR_FRAC = 0.50
|
|
56
|
+
|
|
57
|
+
_MAX_SINGLE_PAGE_ROWS = 25 # more rows -> multi-page rendering path
|
|
58
|
+
_MAX_SINGLE_PAGE_LINES = 120
|
|
59
|
+
|
|
60
|
+
# marker comment the emit-side retired-table detection keys off: a
|
|
61
|
+
# restructured pair renders within ONE render call, so only the emit
|
|
62
|
+
# side (a lone Delete before a lone Insert) needs marker scanning
|
|
63
|
+
RESTRUCTURED_MARKER = "% texdiff: restructured table - new version from source (blue)\n"
|
|
64
|
+
|
|
65
|
+
_ROW_LINE_RE = re.compile(r"(^|[^\\])&(?!\\)|\\\\")
|
|
66
|
+
_STRUCT_BEGIN_RE = re.compile(
|
|
67
|
+
r"^\s*\\(?:hline|begin\{longtable|end\{longtable|begin\{tabular|end\{tabular)"
|
|
68
|
+
)
|
|
69
|
+
_HEADFOOT_RE = re.compile(r"\\end(first|last)?(head|foot)")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def data_rows(text: str) -> int:
|
|
73
|
+
"""Count data rows of a (clean source) table text.
|
|
74
|
+
|
|
75
|
+
Non-structure, non-comment lines holding a cell separator or a
|
|
76
|
+
row terminator; mirrors ``data_rows_source`` of the reference.
|
|
77
|
+
"""
|
|
78
|
+
n = 0
|
|
79
|
+
for line in text.split("\n"):
|
|
80
|
+
stripped = line.strip()
|
|
81
|
+
if not stripped or stripped.startswith("%"):
|
|
82
|
+
continue
|
|
83
|
+
if _STRUCT_BEGIN_RE.match(stripped):
|
|
84
|
+
continue
|
|
85
|
+
if _HEADFOOT_RE.search(line):
|
|
86
|
+
continue
|
|
87
|
+
if _ROW_LINE_RE.search(line):
|
|
88
|
+
n += 1
|
|
89
|
+
return n
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def is_restructured(old_text: str, new_text: str) -> bool:
|
|
93
|
+
"""True when a table pair is too far apart for row-granular markup.
|
|
94
|
+
|
|
95
|
+
Either side missing counts: an old table without a new counterpart
|
|
96
|
+
is retired, a new without an old is new - both take the wholesale
|
|
97
|
+
rendering too. For two present tables the reference thresholds
|
|
98
|
+
decide (>= 80 % relative data-row difference, >= 3 rows apart).
|
|
99
|
+
"""
|
|
100
|
+
n_old = data_rows(old_text) if old_text else 0
|
|
101
|
+
n_new = data_rows(new_text) if new_text else 0
|
|
102
|
+
if n_old == 0 or n_new == 0:
|
|
103
|
+
return True
|
|
104
|
+
diff = abs(n_old - n_new)
|
|
105
|
+
larger = max(n_old, n_new)
|
|
106
|
+
return diff >= RETIRE_MIN_ROWS and diff / larger >= RETIRE_DEL_FRAC
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# --- old-side renderer: strike every word, keep structure ------------------
|
|
110
|
+
|
|
111
|
+
_LIST_TOKEN_RE = re.compile(r"(\\begin\{itemize\}|\\end\{itemize\}|\\item\b|\\t\b)")
|
|
112
|
+
_BREAKPOINT_RE = re.compile(r"(\\-\\_?\\allowbreak|\\allowbreak|\\-\\_|\\-|\\_|(?<!\\)_)")
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _strike_word(word: str) -> str:
|
|
116
|
+
"""Strike one word, inserting breakpoints between struck segments.
|
|
117
|
+
|
|
118
|
+
``\\sout`` makes its argument unbreakable, so a whole-cell strike
|
|
119
|
+
in a narrow ``W{}`` column can never wrap - the reference strikes
|
|
120
|
+
per word and re-opens the strike after each breakpoint token
|
|
121
|
+
(``\\_``, ``\\-``, ``\\allowbreak`` ...). List-environment tokens
|
|
122
|
+
(``\\item`` ...) stay outside: ulem's LR mode rejects them.
|
|
123
|
+
"""
|
|
124
|
+
if word.startswith("\\\\") or word == "&":
|
|
125
|
+
return word
|
|
126
|
+
if re.match(r"^\\(?:rowcolor|hline|cline|multicolumn|multirow)", word):
|
|
127
|
+
return word
|
|
128
|
+
if _LIST_TOKEN_RE.search(word):
|
|
129
|
+
parts = _LIST_TOKEN_RE.split(word)
|
|
130
|
+
out = []
|
|
131
|
+
for seg in parts:
|
|
132
|
+
if not seg:
|
|
133
|
+
continue
|
|
134
|
+
if _LIST_TOKEN_RE.fullmatch(seg):
|
|
135
|
+
out.append(" " if seg == "\\t" else seg)
|
|
136
|
+
else:
|
|
137
|
+
out.append(_strike_word(seg))
|
|
138
|
+
return "".join(out)
|
|
139
|
+
pieces = _BREAKPOINT_RE.split(word)
|
|
140
|
+
out = []
|
|
141
|
+
depth = 0
|
|
142
|
+
for piece in pieces:
|
|
143
|
+
if not piece:
|
|
144
|
+
continue
|
|
145
|
+
if _BREAKPOINT_RE.fullmatch(piece):
|
|
146
|
+
if depth:
|
|
147
|
+
out.append("}")
|
|
148
|
+
depth -= 1
|
|
149
|
+
out.append(piece + "\\allowbreak ")
|
|
150
|
+
out.append("\\sout{")
|
|
151
|
+
depth += 1
|
|
152
|
+
else:
|
|
153
|
+
if not depth:
|
|
154
|
+
out.append("\\sout{")
|
|
155
|
+
depth += 1
|
|
156
|
+
out.append(piece)
|
|
157
|
+
if depth:
|
|
158
|
+
out.append("}")
|
|
159
|
+
return "".join(out)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _add_breakpoints(s: str) -> str:
|
|
163
|
+
"""Let long tokens wrap: ``\\_`` gets ``\\allowbreak`` after it.
|
|
164
|
+
|
|
165
|
+
Skips comments and package/filename arguments, whose corruption
|
|
166
|
+
would break file references; comma breaking of the reference is
|
|
167
|
+
not ported (texdiff emits generated tables whose cells are short).
|
|
168
|
+
"""
|
|
169
|
+
if s.lstrip().startswith("%"):
|
|
170
|
+
return s
|
|
171
|
+
if re.search(r"\\(?:documentclass|usepackage|RequirePackage|textattachfile|path|includegraphics|input|include)\s*[\[{]", s):
|
|
172
|
+
return s
|
|
173
|
+
return re.sub(r"(\\_)(?!\s*\\allowbreak)", r"\1\\allowbreak ", s)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
# a full \caption{...} call; the body must be brace-balanced on ONE
|
|
177
|
+
# line (captions in generated spec tables always are)
|
|
178
|
+
_CAPTION_RE = re.compile(r"(\\caption)(\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\})")
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def strike_through(text: str) -> str:
|
|
182
|
+
"""Render a clean old table text red-struck, per word.
|
|
183
|
+
|
|
184
|
+
Data-row lines get every word struck (terminator ``\\\\`` stays
|
|
185
|
+
outside the strikes); structure lines (``\\hline``, environment
|
|
186
|
+
boundaries), blank lines and header/footer markers pass through
|
|
187
|
+
untouched. Wrapped-row continuation lines are struck as well so
|
|
188
|
+
long attribute tails do not render as live text.
|
|
189
|
+
"""
|
|
190
|
+
out = []
|
|
191
|
+
in_row = False
|
|
192
|
+
lines = _fix_cell_counts(_add_breakpoints(text)).split("\n")
|
|
193
|
+
for line in lines:
|
|
194
|
+
# caption line: a retired table's caption typesets struck like
|
|
195
|
+
# its rows but must stay transparent to numbering - it neither
|
|
196
|
+
# steps the table counter nor writes a list-of-tables entry,
|
|
197
|
+
# so the blue replacement keeps the number the retired table
|
|
198
|
+
# would have taken (captions *of and only of* retired tables
|
|
199
|
+
# pass through strike_through; live ones never do)
|
|
200
|
+
mcap = _CAPTION_RE.search(line)
|
|
201
|
+
if mcap:
|
|
202
|
+
begin, body = mcap.groups()
|
|
203
|
+
out.append(
|
|
204
|
+
line[: mcap.start()]
|
|
205
|
+
# empty optional argument: the caption typesets (with
|
|
206
|
+
# its Table N: label, struck like the table body) but
|
|
207
|
+
# writes NO list-of-tables entry; the immediate
|
|
208
|
+
# \addtocounter{table}{-1} then gives the number back,
|
|
209
|
+
# so the blue replacement caption takes the very
|
|
210
|
+
# number the retired table carried - they read as one
|
|
211
|
+
# logical "Table N (old) -> Table N (new)" pair
|
|
212
|
+
+ begin
|
|
213
|
+
+ "[]{\\texorpdfstring{"
|
|
214
|
+
+ _strike_word(body[1:-1])
|
|
215
|
+
+ "}{}"
|
|
216
|
+
+ "}"
|
|
217
|
+
+ "\\addtocounter{table}{-1}% texdiff: "
|
|
218
|
+
"retired caption - struck, unnumbered, counter-neutral\n"
|
|
219
|
+
+ line[mcap.end() :]
|
|
220
|
+
)
|
|
221
|
+
continue
|
|
222
|
+
if _STRUCT_BEGIN_RE.match(line) or not line.strip() or _HEADFOOT_RE.search(line):
|
|
223
|
+
out.append(line)
|
|
224
|
+
continue
|
|
225
|
+
if _ROW_LINE_RE.search(line):
|
|
226
|
+
m = re.search(r"\s*(\\\\)\s*$", line)
|
|
227
|
+
term = ""
|
|
228
|
+
row = line
|
|
229
|
+
if m:
|
|
230
|
+
term = m.group(1) + "\n"
|
|
231
|
+
row = line[: m.start()]
|
|
232
|
+
struck = re.sub(r"\S+", lambda m2: _strike_word(m2.group(0)), row)
|
|
233
|
+
out.append(struck)
|
|
234
|
+
out.append(term)
|
|
235
|
+
in_row = True
|
|
236
|
+
continue
|
|
237
|
+
if in_row and line.strip() and not line.lstrip().startswith("\\"):
|
|
238
|
+
out.append(re.sub(r"\S+", lambda m2: _strike_word(m2.group(0)), line))
|
|
239
|
+
continue
|
|
240
|
+
out.append(line)
|
|
241
|
+
return "".join(x if x.endswith("\n") else x + "\n" for x in out)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
# plane implementation of the reference's cell-count repair: the old
|
|
245
|
+
# hand-authored tables occasionally hold rows with one cell too few;
|
|
246
|
+
# the restructured rendering has no latexdiff glue to fix, so a plain
|
|
247
|
+
# pass keeps the column count of the spec
|
|
248
|
+
def _fix_cell_counts(text: str) -> str:
|
|
249
|
+
"""No-op placeholder keeping the reference call graph shape.
|
|
250
|
+
|
|
251
|
+
The reference repairs latexdiff's glue rows; texdiff renders the
|
|
252
|
+
clean source directly, whose rows already match the spec (round
|
|
253
|
+
trip verified on parse). Returns the input unchanged.
|
|
254
|
+
"""
|
|
255
|
+
return text
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def render_restructured(old_text: str, new_text: str) -> str:
|
|
259
|
+
"""Render a restructured pair: old red-struck, then new in blue.
|
|
260
|
+
|
|
261
|
+
Both sides are the clean revision sources (no markup); the old
|
|
262
|
+
side is per-word struck through inside a red colour group, the
|
|
263
|
+
new side emitted verbatim inside a blue one - the reference build's
|
|
264
|
+
"restructured (multi-page)" and "restructured new" rendering.
|
|
265
|
+
"""
|
|
266
|
+
parts = ["{\\color{red}\n", strike_through(old_text), "}\n"]
|
|
267
|
+
if new_text:
|
|
268
|
+
parts += [
|
|
269
|
+
RESTRUCTURED_MARKER,
|
|
270
|
+
"{\\color{blue}\n",
|
|
271
|
+
new_text,
|
|
272
|
+
"}\n",
|
|
273
|
+
]
|
|
274
|
+
return "".join(parts)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
# --- pathology and row merging ---------------------------------------------
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _data_row_lines(text: str) -> list[str]:
|
|
281
|
+
"""Data-row physical lines of a clean table text."""
|
|
282
|
+
out = []
|
|
283
|
+
for line in text.split("\n"):
|
|
284
|
+
s = line.strip()
|
|
285
|
+
if not s or s.startswith("%"):
|
|
286
|
+
continue
|
|
287
|
+
if _STRUCT_BEGIN_RE.match(s):
|
|
288
|
+
continue
|
|
289
|
+
if _HEADFOOT_RE.search(line):
|
|
290
|
+
continue
|
|
291
|
+
if _ROW_LINE_RE.search(line):
|
|
292
|
+
out.append(line)
|
|
293
|
+
return out
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def is_pathological(old_text: str, new_text: str) -> bool:
|
|
297
|
+
"""True when inline row markup of the pair would be noise.
|
|
298
|
+
|
|
299
|
+
A word-level similarity of the data-row content below
|
|
300
|
+
:data:`PATHOLOGICAL_WORD_SIMILARITY` means almost every cell is
|
|
301
|
+
rewritten; per-cell markup then interleaves red and blue fragments
|
|
302
|
+
line upon line. Such pairs render better as wholesale tables (or
|
|
303
|
+
a row-merged single table when enough rows still match).
|
|
304
|
+
"""
|
|
305
|
+
wo = " ".join(_data_row_lines(old_text)).split()
|
|
306
|
+
wn = " ".join(_data_row_lines(new_text)).split()
|
|
307
|
+
if not wo or not wn:
|
|
308
|
+
return False
|
|
309
|
+
ratio = SequenceMatcher(a=wo, b=wn, autojunk=False).ratio()
|
|
310
|
+
return ratio < PATHOLOGICAL_WORD_SIMILARITY
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
_SPEC_RE = re.compile(r"\\begin\{longtable\*?\}\s*\{")
|
|
314
|
+
|
|
315
|
+
_STRUCT_TAIL_RE = re.compile(
|
|
316
|
+
r"\n[ \t]*(\\(?:hline|cline\s*\{[^{}]*\})(?:[ \t]+\\(?:hline|cline\s*\{[^{}]*\}))*)[ \t]*(?=\n)"
|
|
317
|
+
)
|
|
318
|
+
_ROW_TERM_RE = re.compile(r"(?:^|\n)([^\n]*?\\\\)", re.M)
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _table_spec(text: str) -> str | None:
|
|
322
|
+
"""Column spec of a longtable (brace-balanced, one nesting level)."""
|
|
323
|
+
m = _SPEC_RE.search(text)
|
|
324
|
+
if m is None:
|
|
325
|
+
return None
|
|
326
|
+
start = m.end() - 1
|
|
327
|
+
depth = 0
|
|
328
|
+
for i in range(start, len(text)):
|
|
329
|
+
c = text[i]
|
|
330
|
+
if c == "{":
|
|
331
|
+
depth += 1
|
|
332
|
+
elif c == "}":
|
|
333
|
+
depth -= 1
|
|
334
|
+
if depth == 0:
|
|
335
|
+
return text[start : i + 1]
|
|
336
|
+
return None
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _split_table_rows(text: str) -> tuple[str, list[str], str]:
|
|
340
|
+
"""Split a table into (pre, rows, post); rows end at their ``\\\\``.
|
|
341
|
+
|
|
342
|
+
A row runs from its line start to the terminating ``\\\\``,
|
|
343
|
+
absorbing the standalone structure lines (``\\hline`` /
|
|
344
|
+
``\\cline``) that directly follow it plus one blank line - that
|
|
345
|
+
keeps the horizontal borders attached to the right row.
|
|
346
|
+
"""
|
|
347
|
+
rows: list[str] = []
|
|
348
|
+
spans: list[tuple[int, int]] = []
|
|
349
|
+
pos = 0
|
|
350
|
+
while True:
|
|
351
|
+
m = _ROW_TERM_RE.search(text, pos)
|
|
352
|
+
if m is None:
|
|
353
|
+
break
|
|
354
|
+
start = m.start(1)
|
|
355
|
+
end = m.end(1)
|
|
356
|
+
# absorb following standalone \hline / \cline lines + one blank
|
|
357
|
+
while True:
|
|
358
|
+
t = _STRUCT_TAIL_RE.match(text, end)
|
|
359
|
+
if t is None:
|
|
360
|
+
break
|
|
361
|
+
end = t.end()
|
|
362
|
+
if text.startswith("\n", end):
|
|
363
|
+
end += 1
|
|
364
|
+
rows.append(text[start:end])
|
|
365
|
+
spans.append((start, end))
|
|
366
|
+
pos = end
|
|
367
|
+
if not spans:
|
|
368
|
+
return text, [], ""
|
|
369
|
+
pre = text[: spans[0][0]]
|
|
370
|
+
post = text[spans[-1][1] :]
|
|
371
|
+
return pre, rows, post
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def _row_is_data(row: str) -> bool:
|
|
375
|
+
"""False for structure-only rows (lone \\hline etc.)."""
|
|
376
|
+
t = re.sub(r"\\\\\s*$", "", row)
|
|
377
|
+
t = t.replace("\\hline", "")
|
|
378
|
+
t = re.sub(r"\s+", "", t)
|
|
379
|
+
return len(t) > 0
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _norm_row(row: str) -> str:
|
|
383
|
+
"""Normalization for row matching (breakpoints/underscores/ws)."""
|
|
384
|
+
r = row.replace("\\-", "")
|
|
385
|
+
r = r.replace("\\_", "_")
|
|
386
|
+
r = re.sub(r"\\allowbreak\s*", "", r)
|
|
387
|
+
r = re.sub(r"\s+", " ", r)
|
|
388
|
+
return r.strip()
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _lcs_pairs(a: list[str], b: list[str]) -> list[tuple[int, int]]:
|
|
392
|
+
"""Longest common subsequence pairs of two key lists (DP)."""
|
|
393
|
+
n, m = len(a), len(b)
|
|
394
|
+
dp = [[0] * (m + 1) for _ in range(n + 1)]
|
|
395
|
+
for i in range(n - 1, -1, -1):
|
|
396
|
+
for j in range(m - 1, -1, -1):
|
|
397
|
+
dp[i][j] = (
|
|
398
|
+
dp[i + 1][j + 1] + 1
|
|
399
|
+
if a[i] == b[j]
|
|
400
|
+
else max(dp[i + 1][j], dp[i][j + 1])
|
|
401
|
+
)
|
|
402
|
+
pairs = []
|
|
403
|
+
i = j = 0
|
|
404
|
+
while i < n and j < m:
|
|
405
|
+
if a[i] == b[j]:
|
|
406
|
+
pairs.append((i, j))
|
|
407
|
+
i += 1
|
|
408
|
+
j += 1
|
|
409
|
+
elif dp[i + 1][j] >= dp[i][j + 1]:
|
|
410
|
+
i += 1
|
|
411
|
+
else:
|
|
412
|
+
j += 1
|
|
413
|
+
return pairs
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def _color_row(row: str, color: str) -> str:
|
|
417
|
+
"""Wrap one source row's cells in a colour group, red struck.
|
|
418
|
+
|
|
419
|
+
Lead (``\\rowcolor`` etc.), the ``\\\\`` terminator and trailing
|
|
420
|
+
structure lines stay outside the colour groups: a brace group may
|
|
421
|
+
not span the alignment separator ``&``, and row colours must be
|
|
422
|
+
given right after ``\\\\``. Each cell is coloured (and, for red,
|
|
423
|
+
struck word-wise) individually.
|
|
424
|
+
"""
|
|
425
|
+
tail = ""
|
|
426
|
+
while True:
|
|
427
|
+
t = re.search(r"\n[ \t]*(\\(?:hline|cline\s*\{[^{}]*\}|rowcolor\s*\{[^{}]*\})[^\n]*)\s*$", row)
|
|
428
|
+
if t is None:
|
|
429
|
+
break
|
|
430
|
+
row = row[: t.start()]
|
|
431
|
+
tail = "\n" + t.group(1) + tail
|
|
432
|
+
m = re.search(r"\s*(\\\\)\s*$", row)
|
|
433
|
+
term = ""
|
|
434
|
+
if m:
|
|
435
|
+
term = m.group(1)
|
|
436
|
+
row = row[: m.start()]
|
|
437
|
+
lead = ""
|
|
438
|
+
while True:
|
|
439
|
+
m = re.match(r"\s*(\\rowcolor\s*\{[^{}]*\}|\\hline|\\cline\s*\{[^{}]*\})\s*(.*)$", row, re.S)
|
|
440
|
+
if m is None:
|
|
441
|
+
break
|
|
442
|
+
lead += m.group(1) + " "
|
|
443
|
+
row = m.group(2)
|
|
444
|
+
cells: list[str] = []
|
|
445
|
+
depth = 0
|
|
446
|
+
cur = []
|
|
447
|
+
i = 0
|
|
448
|
+
while i < len(row):
|
|
449
|
+
c = row[i]
|
|
450
|
+
if c == "\\" and i + 1 < len(row):
|
|
451
|
+
cur.append(row[i : i + 2])
|
|
452
|
+
i += 2
|
|
453
|
+
continue
|
|
454
|
+
if c == "&" and depth == 0:
|
|
455
|
+
cells.append("".join(cur))
|
|
456
|
+
cur = []
|
|
457
|
+
i += 1
|
|
458
|
+
continue
|
|
459
|
+
if c == "{":
|
|
460
|
+
depth += 1
|
|
461
|
+
elif c == "}":
|
|
462
|
+
depth -= 1
|
|
463
|
+
cur.append(c)
|
|
464
|
+
i += 1
|
|
465
|
+
cells.append("".join(cur))
|
|
466
|
+
out = []
|
|
467
|
+
for cell in cells:
|
|
468
|
+
if color == "red":
|
|
469
|
+
cell = re.sub(r"\S+", lambda m2: _strike_word(m2.group(0)), cell)
|
|
470
|
+
out.append("{\\color{" + color + "}" + cell + "}")
|
|
471
|
+
return lead + " & ".join(out) + term + tail + "\n"
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def merge_tables(old_text: str, new_text: str) -> str | None:
|
|
475
|
+
"""Row-merged rendering of a matched pair, or None when impossible.
|
|
476
|
+
|
|
477
|
+
One longtable: common rows black (new variant), removed rows
|
|
478
|
+
red-struck, added rows blue - reading like the classic diff while
|
|
479
|
+
keeping the layout. Returns None when the merge would be mostly
|
|
480
|
+
noise (reference acceptance thresholds).
|
|
481
|
+
"""
|
|
482
|
+
if _table_spec(old_text) != _table_spec(new_text):
|
|
483
|
+
return None
|
|
484
|
+
pre_old, rows_old, post_old = _split_table_rows(old_text)
|
|
485
|
+
pre_new, rows_new, post_new = _split_table_rows(new_text)
|
|
486
|
+
|
|
487
|
+
idx_old = [i for i, r in enumerate(rows_old) if _row_is_data(r)]
|
|
488
|
+
idx_new = [j for j, r in enumerate(rows_new) if _row_is_data(r)]
|
|
489
|
+
if not idx_old or not idx_new:
|
|
490
|
+
return None
|
|
491
|
+
data_old = [_norm_row(rows_old[i]) for i in idx_old]
|
|
492
|
+
data_new = [_norm_row(rows_new[j]) for j in idx_new]
|
|
493
|
+
pairs = _lcs_pairs(data_old, data_new)
|
|
494
|
+
if len(pairs) < MIN_MATCH_FRAC * min(len(data_old), len(data_new)):
|
|
495
|
+
return None
|
|
496
|
+
pair_of_new = {j: i for i, j in pairs}
|
|
497
|
+
matched_old = {i for i, _ in pairs}
|
|
498
|
+
unmatched_old = len(data_old) - len(pairs)
|
|
499
|
+
if unmatched_old >= MAX_UNMATCHED_FRAC * len(data_old):
|
|
500
|
+
return None
|
|
501
|
+
matched_new = {j for _, j in pairs}
|
|
502
|
+
unmatched_new_norms: dict[str, int] = {}
|
|
503
|
+
for j in range(len(data_new)):
|
|
504
|
+
if j not in matched_new:
|
|
505
|
+
unmatched_new_norms[data_new[j]] = (
|
|
506
|
+
unmatched_new_norms.get(data_new[j], 0) + 1
|
|
507
|
+
)
|
|
508
|
+
identical = 0
|
|
509
|
+
for i in range(len(data_old)):
|
|
510
|
+
if i in matched_old:
|
|
511
|
+
continue
|
|
512
|
+
if unmatched_new_norms.get(data_old[i], 0):
|
|
513
|
+
identical += 1
|
|
514
|
+
unmatched_new_norms[data_old[i]] -= 1
|
|
515
|
+
if unmatched_old > 0 and identical / unmatched_old >= REAPPEAR_FRAC:
|
|
516
|
+
return None
|
|
517
|
+
|
|
518
|
+
out_rows: list[str] = []
|
|
519
|
+
old_cursor = 0
|
|
520
|
+
for j in range(len(data_new)):
|
|
521
|
+
if j in pair_of_new:
|
|
522
|
+
while old_cursor < pair_of_new[j]:
|
|
523
|
+
out_rows.append(
|
|
524
|
+
_color_row(rows_old[idx_old[old_cursor]], "red")
|
|
525
|
+
)
|
|
526
|
+
old_cursor += 1
|
|
527
|
+
old_cursor += 1 # skip the common old row
|
|
528
|
+
out_rows.append(rows_new[idx_new[j]].rstrip("\n") + "\n")
|
|
529
|
+
else:
|
|
530
|
+
out_rows.append(_color_row(rows_new[idx_new[j]], "blue"))
|
|
531
|
+
while old_cursor < len(data_old):
|
|
532
|
+
out_rows.append(_color_row(rows_old[idx_old[old_cursor]], "red"))
|
|
533
|
+
old_cursor += 1
|
|
534
|
+
|
|
535
|
+
return (
|
|
536
|
+
"% texdiff: row-merged table (common rows black, removed red, added blue)\n"
|
|
537
|
+
+ pre_new
|
|
538
|
+
+ "".join(out_rows)
|
|
539
|
+
+ post_new
|
|
540
|
+
)
|
texdiff/textdiff.py
ADDED
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""Word-level diff of text runs (character-level, semantics-cleaned).
|
|
2
|
+
|
|
3
|
+
Used inside matched group nodes to mark up changed words. Uses Google's
|
|
4
|
+
diff-match-patch algorithm (vendored pure-python port, small) so that
|
|
5
|
+
``cleanupSemantic`` merges the "island" edits into human-readable
|
|
6
|
+
replacement regions - the track-changes look reviewers expect.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from difflib import SequenceMatcher
|
|
14
|
+
|
|
15
|
+
EQUAL = "equal"
|
|
16
|
+
DELETE = "delete"
|
|
17
|
+
INSERT = "insert"
|
|
18
|
+
|
|
19
|
+
# Tokens which must never be split by a diff boundary: a control
|
|
20
|
+
# sequence must stay glued to the braces of its argument, and braces
|
|
21
|
+
# must not be orphaned.
|
|
22
|
+
_CS = r"\\[a-zA-Z]+\*?"
|
|
23
|
+
_CS_ESC = r"\\."
|
|
24
|
+
_TOKEN_RE = re.compile(
|
|
25
|
+
"|".join(
|
|
26
|
+
[
|
|
27
|
+
rf"{_CS}\{{[^{{}}]*\}}", # \cmd{...}: keep together
|
|
28
|
+
_CS, # bare control sequence
|
|
29
|
+
_CS_ESC, # \\, \&, \%, ...
|
|
30
|
+
r"\s+", # whitespace run
|
|
31
|
+
r"[^\s\\{}]+(?:\{[^{}]*\}[^\s\\{}]*)*", # word w/ inline groups
|
|
32
|
+
r"[{}]", # lone brace
|
|
33
|
+
]
|
|
34
|
+
)
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class Chunk:
|
|
40
|
+
"""One region of a word/structure-level diff.
|
|
41
|
+
|
|
42
|
+
Attributes:
|
|
43
|
+
op: one of ``equal``, ``delete``, ``insert``.
|
|
44
|
+
text: the region's text (old side for delete, new side for
|
|
45
|
+
insert and equal).
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
op: str
|
|
49
|
+
text: str
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def tokenize(s: str) -> list[str]:
|
|
53
|
+
"""Split a text run into diff-safe tokens (whitespace included)."""
|
|
54
|
+
return [m.group(0) for m in _TOKEN_RE.finditer(s)]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def word_diff(old: str, new: str) -> list[Chunk]:
|
|
58
|
+
"""Diff two text runs at word granularity.
|
|
59
|
+
|
|
60
|
+
Returns a list of :class:`Chunk` regions which, concatenated:
|
|
61
|
+
``equal``+``insert`` reproduce ``new``; ``equal``+``delete``
|
|
62
|
+
reproduce ``old``.
|
|
63
|
+
|
|
64
|
+
Rules (v0 contract):
|
|
65
|
+
|
|
66
|
+
* whitespace-only differences keep the NEW text;
|
|
67
|
+
* chaff edits (isolated one-char changes) are merged into
|
|
68
|
+
replacement regions by semantic cleanup;
|
|
69
|
+
* the diff never splits a ``\\command`` name from its argument
|
|
70
|
+
braces or from its escape.
|
|
71
|
+
"""
|
|
72
|
+
a, b = tokenize(old), tokenize(new)
|
|
73
|
+
sm = SequenceMatcher(a=a, b=b, autojunk=False)
|
|
74
|
+
chunks: list[Chunk] = []
|
|
75
|
+
for tag, i1, i2, j1, j2 in sm.get_opcodes():
|
|
76
|
+
if tag == "equal":
|
|
77
|
+
chunks.append(Chunk(EQUAL, "".join(a[i1:i2])))
|
|
78
|
+
elif tag == "delete":
|
|
79
|
+
chunks.append(Chunk(DELETE, "".join(a[i1:i2])))
|
|
80
|
+
elif tag == "insert":
|
|
81
|
+
chunks.append(Chunk(INSERT, "".join(b[j1:j2])))
|
|
82
|
+
else: # replace
|
|
83
|
+
chunks.append(Chunk(DELETE, "".join(a[i1:i2])))
|
|
84
|
+
chunks.append(Chunk(INSERT, "".join(b[j1:j2])))
|
|
85
|
+
|
|
86
|
+
chunks = _merge_adjacent(chunks)
|
|
87
|
+
chunks = _whitespace_to_new(chunks)
|
|
88
|
+
chunks = _merge_replacements(chunks)
|
|
89
|
+
return chunks
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _merge_adjacent(chunks: list[Chunk]) -> list[Chunk]:
|
|
93
|
+
"""Concatenate neighbouring chunks of the same op."""
|
|
94
|
+
out: list[Chunk] = []
|
|
95
|
+
for c in chunks:
|
|
96
|
+
if out and out[-1].op == c.op:
|
|
97
|
+
out[-1] = Chunk(c.op, out[-1].text + c.text)
|
|
98
|
+
else:
|
|
99
|
+
out.append(c)
|
|
100
|
+
return out
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _whitespace_to_new(chunks: list[Chunk]) -> list[Chunk]:
|
|
104
|
+
"""Whitespace-only changes keep the NEW side's whitespace.
|
|
105
|
+
|
|
106
|
+
A delete chunk that is pure whitespace next to an insert that is
|
|
107
|
+
pure whitespace collapses to an *equal* chunk with the new text.
|
|
108
|
+
"""
|
|
109
|
+
out: list[Chunk] = []
|
|
110
|
+
i = 0
|
|
111
|
+
while i < len(chunks):
|
|
112
|
+
cur = chunks[i]
|
|
113
|
+
nxt = chunks[i + 1] if i + 1 < len(chunks) else None
|
|
114
|
+
if (
|
|
115
|
+
cur.op == DELETE
|
|
116
|
+
and nxt is not None
|
|
117
|
+
and nxt.op == INSERT
|
|
118
|
+
and cur.text.strip() == ""
|
|
119
|
+
and nxt.text.strip() == ""
|
|
120
|
+
):
|
|
121
|
+
out.append(Chunk(EQUAL, nxt.text))
|
|
122
|
+
i += 2
|
|
123
|
+
continue
|
|
124
|
+
out.append(cur)
|
|
125
|
+
i += 1
|
|
126
|
+
return out
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _merge_replacements(chunks: list[Chunk]) -> list[Chunk]:
|
|
130
|
+
"""Semantic cleanup: collapse scattered edits into replacements.
|
|
131
|
+
|
|
132
|
+
``equal`` regions shorter than the neighbouring deletions+insertions
|
|
133
|
+
are absorbed, so one word changed inside a word flips a whole
|
|
134
|
+
single replace region instead of many microscopic islands.
|
|
135
|
+
"""
|
|
136
|
+
if len(chunks) < 5:
|
|
137
|
+
return chunks
|
|
138
|
+
out = list(chunks)
|
|
139
|
+
changed = True
|
|
140
|
+
while changed:
|
|
141
|
+
changed = False
|
|
142
|
+
for i in range(1, len(out) - 1):
|
|
143
|
+
if out[i].op != EQUAL:
|
|
144
|
+
continue
|
|
145
|
+
left = out[i - 1]
|
|
146
|
+
right = out[i + 1]
|
|
147
|
+
if left.op == DELETE and right.op == INSERT:
|
|
148
|
+
# pattern: del eq ins -> merge if eq is short chaff
|
|
149
|
+
if len(out[i].text.strip()) <= 2 and len(out[i].text) <= max(
|
|
150
|
+
len(left.text), len(right.text)
|
|
151
|
+
):
|
|
152
|
+
merged_del = Chunk(DELETE, left.text + out[i].text)
|
|
153
|
+
merged_ins = Chunk(INSERT, out[i].text + right.text)
|
|
154
|
+
out[i - 1 : i + 2] = [merged_del, merged_ins]
|
|
155
|
+
changed = True
|
|
156
|
+
break
|
|
157
|
+
return _merge_adjacent(out)
|