texdiff 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
texdiff/tables.py ADDED
@@ -0,0 +1,540 @@
1
+ """Wholesale rendering of heavily restructured tables.
2
+
3
+ Ports the behaviour of the reference ``sanitize-tables.pl`` toolkit to
4
+ generic texdiff: when one logical table's content changed so much that
5
+ row-granular markup is noise (80 %+ row churn), the change reads best
6
+ as *two whole tables* - the old revision's table struck through in
7
+ red, the new revision's in blue. latexdiff's own row markup for such
8
+ configurations produces pathological glue; the reference build
9
+ solves that by locating each clean source table and rendering it
10
+ wholesale.
11
+
12
+ Components:
13
+
14
+ * :func:`is_restructured` - classify an old/new table-text pair by
15
+ data-row churn (the reference thresholds: >= 80 % relative
16
+ difference AND >= 3 rows apart);
17
+ * :func:`is_pathological` - inline row markup of a pair would be
18
+ unreadable interleaving (word similarity below
19
+ :data:`PATHOLOGICAL_WORD_SIMILARITY`); such pairs take the
20
+ wholesale or row-merged rendering instead;
21
+ * :func:`merge_tables` - the row-merged rendering: one table, common
22
+ rows black, removed rows red-struck, added rows blue (the reference
23
+ ``merged_rows``, LCS over normalized rows with the 0.3 / 0.5 / 0.5
24
+ thresholds);
25
+ * :func:`render_restructured` - the wholesale renderer: the old side
26
+ as :func:`strike_through` (red, per-word so wide W{} cells can
27
+ still wrap), the new side unmarked inside a blue colour group.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import re
33
+ from difflib import SequenceMatcher
34
+
35
+ # reference thresholds (sanitize-tables.pl): a table pair is
36
+ # "restructured" when the data-row counts differ by >= 80 % relative
37
+ # AND >= 3 absolute rows
38
+ RETIRE_DEL_FRAC = 0.80
39
+ RETIRE_MIN_ROWS = 3
40
+
41
+ # when the word-level similarity of the data rows falls below this,
42
+ # per-cell inline markup degenerates into an unreadable red/blue
43
+ # interleaving - the reference build calls such blocks "pathological"
44
+ # and replaces them wholesale (calibrated so well-matching tables
45
+ # like hand-edited ICD longtables, similarity ~ 0.97, keep their
46
+ # inline markup)
47
+ PATHOLOGICAL_WORD_SIMILARITY = 0.50
48
+
49
+ # row-merge acceptance (reference): the LCS must pair >= 30 % of the
50
+ # smaller row set, <= 50 % of the old rows may stay unpaired, and a
51
+ # merge is rejected when >= 50 % of the unmatched old rows reappear
52
+ # verbatim among the unmatched new rows
53
+ MIN_MATCH_FRAC = 0.30
54
+ MAX_UNMATCHED_FRAC = 0.50
55
+ REAPPEAR_FRAC = 0.50
56
+
57
+ _MAX_SINGLE_PAGE_ROWS = 25 # more rows -> multi-page rendering path
58
+ _MAX_SINGLE_PAGE_LINES = 120
59
+
60
+ # marker comment the emit-side retired-table detection keys off: a
61
+ # restructured pair renders within ONE render call, so only the emit
62
+ # side (a lone Delete before a lone Insert) needs marker scanning
63
+ RESTRUCTURED_MARKER = "% texdiff: restructured table - new version from source (blue)\n"
64
+
65
+ _ROW_LINE_RE = re.compile(r"(^|[^\\])&(?!\\)|\\\\")
66
+ _STRUCT_BEGIN_RE = re.compile(
67
+ r"^\s*\\(?:hline|begin\{longtable|end\{longtable|begin\{tabular|end\{tabular)"
68
+ )
69
+ _HEADFOOT_RE = re.compile(r"\\end(first|last)?(head|foot)")
70
+
71
+
72
+ def data_rows(text: str) -> int:
73
+ """Count data rows of a (clean source) table text.
74
+
75
+ Non-structure, non-comment lines holding a cell separator or a
76
+ row terminator; mirrors ``data_rows_source`` of the reference.
77
+ """
78
+ n = 0
79
+ for line in text.split("\n"):
80
+ stripped = line.strip()
81
+ if not stripped or stripped.startswith("%"):
82
+ continue
83
+ if _STRUCT_BEGIN_RE.match(stripped):
84
+ continue
85
+ if _HEADFOOT_RE.search(line):
86
+ continue
87
+ if _ROW_LINE_RE.search(line):
88
+ n += 1
89
+ return n
90
+
91
+
92
+ def is_restructured(old_text: str, new_text: str) -> bool:
93
+ """True when a table pair is too far apart for row-granular markup.
94
+
95
+ Either side missing counts: an old table without a new counterpart
96
+ is retired, a new without an old is new - both take the wholesale
97
+ rendering too. For two present tables the reference thresholds
98
+ decide (>= 80 % relative data-row difference, >= 3 rows apart).
99
+ """
100
+ n_old = data_rows(old_text) if old_text else 0
101
+ n_new = data_rows(new_text) if new_text else 0
102
+ if n_old == 0 or n_new == 0:
103
+ return True
104
+ diff = abs(n_old - n_new)
105
+ larger = max(n_old, n_new)
106
+ return diff >= RETIRE_MIN_ROWS and diff / larger >= RETIRE_DEL_FRAC
107
+
108
+
109
+ # --- old-side renderer: strike every word, keep structure ------------------
110
+
111
+ _LIST_TOKEN_RE = re.compile(r"(\\begin\{itemize\}|\\end\{itemize\}|\\item\b|\\t\b)")
112
+ _BREAKPOINT_RE = re.compile(r"(\\-\\_?\\allowbreak|\\allowbreak|\\-\\_|\\-|\\_|(?<!\\)_)")
113
+
114
+
115
+ def _strike_word(word: str) -> str:
116
+ """Strike one word, inserting breakpoints between struck segments.
117
+
118
+ ``\\sout`` makes its argument unbreakable, so a whole-cell strike
119
+ in a narrow ``W{}`` column can never wrap - the reference strikes
120
+ per word and re-opens the strike after each breakpoint token
121
+ (``\\_``, ``\\-``, ``\\allowbreak`` ...). List-environment tokens
122
+ (``\\item`` ...) stay outside: ulem's LR mode rejects them.
123
+ """
124
+ if word.startswith("\\\\") or word == "&":
125
+ return word
126
+ if re.match(r"^\\(?:rowcolor|hline|cline|multicolumn|multirow)", word):
127
+ return word
128
+ if _LIST_TOKEN_RE.search(word):
129
+ parts = _LIST_TOKEN_RE.split(word)
130
+ out = []
131
+ for seg in parts:
132
+ if not seg:
133
+ continue
134
+ if _LIST_TOKEN_RE.fullmatch(seg):
135
+ out.append(" " if seg == "\\t" else seg)
136
+ else:
137
+ out.append(_strike_word(seg))
138
+ return "".join(out)
139
+ pieces = _BREAKPOINT_RE.split(word)
140
+ out = []
141
+ depth = 0
142
+ for piece in pieces:
143
+ if not piece:
144
+ continue
145
+ if _BREAKPOINT_RE.fullmatch(piece):
146
+ if depth:
147
+ out.append("}")
148
+ depth -= 1
149
+ out.append(piece + "\\allowbreak ")
150
+ out.append("\\sout{")
151
+ depth += 1
152
+ else:
153
+ if not depth:
154
+ out.append("\\sout{")
155
+ depth += 1
156
+ out.append(piece)
157
+ if depth:
158
+ out.append("}")
159
+ return "".join(out)
160
+
161
+
162
+ def _add_breakpoints(s: str) -> str:
163
+ """Let long tokens wrap: ``\\_`` gets ``\\allowbreak`` after it.
164
+
165
+ Skips comments and package/filename arguments, whose corruption
166
+ would break file references; comma breaking of the reference is
167
+ not ported (texdiff emits generated tables whose cells are short).
168
+ """
169
+ if s.lstrip().startswith("%"):
170
+ return s
171
+ if re.search(r"\\(?:documentclass|usepackage|RequirePackage|textattachfile|path|includegraphics|input|include)\s*[\[{]", s):
172
+ return s
173
+ return re.sub(r"(\\_)(?!\s*\\allowbreak)", r"\1\\allowbreak ", s)
174
+
175
+
176
+ # a full \caption{...} call; the body must be brace-balanced on ONE
177
+ # line (captions in generated spec tables always are)
178
+ _CAPTION_RE = re.compile(r"(\\caption)(\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\})")
179
+
180
+
181
+ def strike_through(text: str) -> str:
182
+ """Render a clean old table text red-struck, per word.
183
+
184
+ Data-row lines get every word struck (terminator ``\\\\`` stays
185
+ outside the strikes); structure lines (``\\hline``, environment
186
+ boundaries), blank lines and header/footer markers pass through
187
+ untouched. Wrapped-row continuation lines are struck as well so
188
+ long attribute tails do not render as live text.
189
+ """
190
+ out = []
191
+ in_row = False
192
+ lines = _fix_cell_counts(_add_breakpoints(text)).split("\n")
193
+ for line in lines:
194
+ # caption line: a retired table's caption typesets struck like
195
+ # its rows but must stay transparent to numbering - it neither
196
+ # steps the table counter nor writes a list-of-tables entry,
197
+ # so the blue replacement keeps the number the retired table
198
+ # would have taken (captions *of and only of* retired tables
199
+ # pass through strike_through; live ones never do)
200
+ mcap = _CAPTION_RE.search(line)
201
+ if mcap:
202
+ begin, body = mcap.groups()
203
+ out.append(
204
+ line[: mcap.start()]
205
+ # empty optional argument: the caption typesets (with
206
+ # its Table N: label, struck like the table body) but
207
+ # writes NO list-of-tables entry; the immediate
208
+ # \addtocounter{table}{-1} then gives the number back,
209
+ # so the blue replacement caption takes the very
210
+ # number the retired table carried - they read as one
211
+ # logical "Table N (old) -> Table N (new)" pair
212
+ + begin
213
+ + "[]{\\texorpdfstring{"
214
+ + _strike_word(body[1:-1])
215
+ + "}{}"
216
+ + "}"
217
+ + "\\addtocounter{table}{-1}% texdiff: "
218
+ "retired caption - struck, unnumbered, counter-neutral\n"
219
+ + line[mcap.end() :]
220
+ )
221
+ continue
222
+ if _STRUCT_BEGIN_RE.match(line) or not line.strip() or _HEADFOOT_RE.search(line):
223
+ out.append(line)
224
+ continue
225
+ if _ROW_LINE_RE.search(line):
226
+ m = re.search(r"\s*(\\\\)\s*$", line)
227
+ term = ""
228
+ row = line
229
+ if m:
230
+ term = m.group(1) + "\n"
231
+ row = line[: m.start()]
232
+ struck = re.sub(r"\S+", lambda m2: _strike_word(m2.group(0)), row)
233
+ out.append(struck)
234
+ out.append(term)
235
+ in_row = True
236
+ continue
237
+ if in_row and line.strip() and not line.lstrip().startswith("\\"):
238
+ out.append(re.sub(r"\S+", lambda m2: _strike_word(m2.group(0)), line))
239
+ continue
240
+ out.append(line)
241
+ return "".join(x if x.endswith("\n") else x + "\n" for x in out)
242
+
243
+
244
+ # plane implementation of the reference's cell-count repair: the old
245
+ # hand-authored tables occasionally hold rows with one cell too few;
246
+ # the restructured rendering has no latexdiff glue to fix, so a plain
247
+ # pass keeps the column count of the spec
248
+ def _fix_cell_counts(text: str) -> str:
249
+ """No-op placeholder keeping the reference call graph shape.
250
+
251
+ The reference repairs latexdiff's glue rows; texdiff renders the
252
+ clean source directly, whose rows already match the spec (round
253
+ trip verified on parse). Returns the input unchanged.
254
+ """
255
+ return text
256
+
257
+
258
+ def render_restructured(old_text: str, new_text: str) -> str:
259
+ """Render a restructured pair: old red-struck, then new in blue.
260
+
261
+ Both sides are the clean revision sources (no markup); the old
262
+ side is per-word struck through inside a red colour group, the
263
+ new side emitted verbatim inside a blue one - the reference build's
264
+ "restructured (multi-page)" and "restructured new" rendering.
265
+ """
266
+ parts = ["{\\color{red}\n", strike_through(old_text), "}\n"]
267
+ if new_text:
268
+ parts += [
269
+ RESTRUCTURED_MARKER,
270
+ "{\\color{blue}\n",
271
+ new_text,
272
+ "}\n",
273
+ ]
274
+ return "".join(parts)
275
+
276
+
277
+ # --- pathology and row merging ---------------------------------------------
278
+
279
+
280
+ def _data_row_lines(text: str) -> list[str]:
281
+ """Data-row physical lines of a clean table text."""
282
+ out = []
283
+ for line in text.split("\n"):
284
+ s = line.strip()
285
+ if not s or s.startswith("%"):
286
+ continue
287
+ if _STRUCT_BEGIN_RE.match(s):
288
+ continue
289
+ if _HEADFOOT_RE.search(line):
290
+ continue
291
+ if _ROW_LINE_RE.search(line):
292
+ out.append(line)
293
+ return out
294
+
295
+
296
+ def is_pathological(old_text: str, new_text: str) -> bool:
297
+ """True when inline row markup of the pair would be noise.
298
+
299
+ A word-level similarity of the data-row content below
300
+ :data:`PATHOLOGICAL_WORD_SIMILARITY` means almost every cell is
301
+ rewritten; per-cell markup then interleaves red and blue fragments
302
+ line upon line. Such pairs render better as wholesale tables (or
303
+ a row-merged single table when enough rows still match).
304
+ """
305
+ wo = " ".join(_data_row_lines(old_text)).split()
306
+ wn = " ".join(_data_row_lines(new_text)).split()
307
+ if not wo or not wn:
308
+ return False
309
+ ratio = SequenceMatcher(a=wo, b=wn, autojunk=False).ratio()
310
+ return ratio < PATHOLOGICAL_WORD_SIMILARITY
311
+
312
+
313
+ _SPEC_RE = re.compile(r"\\begin\{longtable\*?\}\s*\{")
314
+
315
+ _STRUCT_TAIL_RE = re.compile(
316
+ r"\n[ \t]*(\\(?:hline|cline\s*\{[^{}]*\})(?:[ \t]+\\(?:hline|cline\s*\{[^{}]*\}))*)[ \t]*(?=\n)"
317
+ )
318
+ _ROW_TERM_RE = re.compile(r"(?:^|\n)([^\n]*?\\\\)", re.M)
319
+
320
+
321
+ def _table_spec(text: str) -> str | None:
322
+ """Column spec of a longtable (brace-balanced, one nesting level)."""
323
+ m = _SPEC_RE.search(text)
324
+ if m is None:
325
+ return None
326
+ start = m.end() - 1
327
+ depth = 0
328
+ for i in range(start, len(text)):
329
+ c = text[i]
330
+ if c == "{":
331
+ depth += 1
332
+ elif c == "}":
333
+ depth -= 1
334
+ if depth == 0:
335
+ return text[start : i + 1]
336
+ return None
337
+
338
+
339
+ def _split_table_rows(text: str) -> tuple[str, list[str], str]:
340
+ """Split a table into (pre, rows, post); rows end at their ``\\\\``.
341
+
342
+ A row runs from its line start to the terminating ``\\\\``,
343
+ absorbing the standalone structure lines (``\\hline`` /
344
+ ``\\cline``) that directly follow it plus one blank line - that
345
+ keeps the horizontal borders attached to the right row.
346
+ """
347
+ rows: list[str] = []
348
+ spans: list[tuple[int, int]] = []
349
+ pos = 0
350
+ while True:
351
+ m = _ROW_TERM_RE.search(text, pos)
352
+ if m is None:
353
+ break
354
+ start = m.start(1)
355
+ end = m.end(1)
356
+ # absorb following standalone \hline / \cline lines + one blank
357
+ while True:
358
+ t = _STRUCT_TAIL_RE.match(text, end)
359
+ if t is None:
360
+ break
361
+ end = t.end()
362
+ if text.startswith("\n", end):
363
+ end += 1
364
+ rows.append(text[start:end])
365
+ spans.append((start, end))
366
+ pos = end
367
+ if not spans:
368
+ return text, [], ""
369
+ pre = text[: spans[0][0]]
370
+ post = text[spans[-1][1] :]
371
+ return pre, rows, post
372
+
373
+
374
+ def _row_is_data(row: str) -> bool:
375
+ """False for structure-only rows (lone \\hline etc.)."""
376
+ t = re.sub(r"\\\\\s*$", "", row)
377
+ t = t.replace("\\hline", "")
378
+ t = re.sub(r"\s+", "", t)
379
+ return len(t) > 0
380
+
381
+
382
+ def _norm_row(row: str) -> str:
383
+ """Normalization for row matching (breakpoints/underscores/ws)."""
384
+ r = row.replace("\\-", "")
385
+ r = r.replace("\\_", "_")
386
+ r = re.sub(r"\\allowbreak\s*", "", r)
387
+ r = re.sub(r"\s+", " ", r)
388
+ return r.strip()
389
+
390
+
391
+ def _lcs_pairs(a: list[str], b: list[str]) -> list[tuple[int, int]]:
392
+ """Longest common subsequence pairs of two key lists (DP)."""
393
+ n, m = len(a), len(b)
394
+ dp = [[0] * (m + 1) for _ in range(n + 1)]
395
+ for i in range(n - 1, -1, -1):
396
+ for j in range(m - 1, -1, -1):
397
+ dp[i][j] = (
398
+ dp[i + 1][j + 1] + 1
399
+ if a[i] == b[j]
400
+ else max(dp[i + 1][j], dp[i][j + 1])
401
+ )
402
+ pairs = []
403
+ i = j = 0
404
+ while i < n and j < m:
405
+ if a[i] == b[j]:
406
+ pairs.append((i, j))
407
+ i += 1
408
+ j += 1
409
+ elif dp[i + 1][j] >= dp[i][j + 1]:
410
+ i += 1
411
+ else:
412
+ j += 1
413
+ return pairs
414
+
415
+
416
+ def _color_row(row: str, color: str) -> str:
417
+ """Wrap one source row's cells in a colour group, red struck.
418
+
419
+ Lead (``\\rowcolor`` etc.), the ``\\\\`` terminator and trailing
420
+ structure lines stay outside the colour groups: a brace group may
421
+ not span the alignment separator ``&``, and row colours must be
422
+ given right after ``\\\\``. Each cell is coloured (and, for red,
423
+ struck word-wise) individually.
424
+ """
425
+ tail = ""
426
+ while True:
427
+ t = re.search(r"\n[ \t]*(\\(?:hline|cline\s*\{[^{}]*\}|rowcolor\s*\{[^{}]*\})[^\n]*)\s*$", row)
428
+ if t is None:
429
+ break
430
+ row = row[: t.start()]
431
+ tail = "\n" + t.group(1) + tail
432
+ m = re.search(r"\s*(\\\\)\s*$", row)
433
+ term = ""
434
+ if m:
435
+ term = m.group(1)
436
+ row = row[: m.start()]
437
+ lead = ""
438
+ while True:
439
+ m = re.match(r"\s*(\\rowcolor\s*\{[^{}]*\}|\\hline|\\cline\s*\{[^{}]*\})\s*(.*)$", row, re.S)
440
+ if m is None:
441
+ break
442
+ lead += m.group(1) + " "
443
+ row = m.group(2)
444
+ cells: list[str] = []
445
+ depth = 0
446
+ cur = []
447
+ i = 0
448
+ while i < len(row):
449
+ c = row[i]
450
+ if c == "\\" and i + 1 < len(row):
451
+ cur.append(row[i : i + 2])
452
+ i += 2
453
+ continue
454
+ if c == "&" and depth == 0:
455
+ cells.append("".join(cur))
456
+ cur = []
457
+ i += 1
458
+ continue
459
+ if c == "{":
460
+ depth += 1
461
+ elif c == "}":
462
+ depth -= 1
463
+ cur.append(c)
464
+ i += 1
465
+ cells.append("".join(cur))
466
+ out = []
467
+ for cell in cells:
468
+ if color == "red":
469
+ cell = re.sub(r"\S+", lambda m2: _strike_word(m2.group(0)), cell)
470
+ out.append("{\\color{" + color + "}" + cell + "}")
471
+ return lead + " & ".join(out) + term + tail + "\n"
472
+
473
+
474
+ def merge_tables(old_text: str, new_text: str) -> str | None:
475
+ """Row-merged rendering of a matched pair, or None when impossible.
476
+
477
+ One longtable: common rows black (new variant), removed rows
478
+ red-struck, added rows blue - reading like the classic diff while
479
+ keeping the layout. Returns None when the merge would be mostly
480
+ noise (reference acceptance thresholds).
481
+ """
482
+ if _table_spec(old_text) != _table_spec(new_text):
483
+ return None
484
+ pre_old, rows_old, post_old = _split_table_rows(old_text)
485
+ pre_new, rows_new, post_new = _split_table_rows(new_text)
486
+
487
+ idx_old = [i for i, r in enumerate(rows_old) if _row_is_data(r)]
488
+ idx_new = [j for j, r in enumerate(rows_new) if _row_is_data(r)]
489
+ if not idx_old or not idx_new:
490
+ return None
491
+ data_old = [_norm_row(rows_old[i]) for i in idx_old]
492
+ data_new = [_norm_row(rows_new[j]) for j in idx_new]
493
+ pairs = _lcs_pairs(data_old, data_new)
494
+ if len(pairs) < MIN_MATCH_FRAC * min(len(data_old), len(data_new)):
495
+ return None
496
+ pair_of_new = {j: i for i, j in pairs}
497
+ matched_old = {i for i, _ in pairs}
498
+ unmatched_old = len(data_old) - len(pairs)
499
+ if unmatched_old >= MAX_UNMATCHED_FRAC * len(data_old):
500
+ return None
501
+ matched_new = {j for _, j in pairs}
502
+ unmatched_new_norms: dict[str, int] = {}
503
+ for j in range(len(data_new)):
504
+ if j not in matched_new:
505
+ unmatched_new_norms[data_new[j]] = (
506
+ unmatched_new_norms.get(data_new[j], 0) + 1
507
+ )
508
+ identical = 0
509
+ for i in range(len(data_old)):
510
+ if i in matched_old:
511
+ continue
512
+ if unmatched_new_norms.get(data_old[i], 0):
513
+ identical += 1
514
+ unmatched_new_norms[data_old[i]] -= 1
515
+ if unmatched_old > 0 and identical / unmatched_old >= REAPPEAR_FRAC:
516
+ return None
517
+
518
+ out_rows: list[str] = []
519
+ old_cursor = 0
520
+ for j in range(len(data_new)):
521
+ if j in pair_of_new:
522
+ while old_cursor < pair_of_new[j]:
523
+ out_rows.append(
524
+ _color_row(rows_old[idx_old[old_cursor]], "red")
525
+ )
526
+ old_cursor += 1
527
+ old_cursor += 1 # skip the common old row
528
+ out_rows.append(rows_new[idx_new[j]].rstrip("\n") + "\n")
529
+ else:
530
+ out_rows.append(_color_row(rows_new[idx_new[j]], "blue"))
531
+ while old_cursor < len(data_old):
532
+ out_rows.append(_color_row(rows_old[idx_old[old_cursor]], "red"))
533
+ old_cursor += 1
534
+
535
+ return (
536
+ "% texdiff: row-merged table (common rows black, removed red, added blue)\n"
537
+ + pre_new
538
+ + "".join(out_rows)
539
+ + post_new
540
+ )
texdiff/textdiff.py ADDED
@@ -0,0 +1,157 @@
1
+ """Word-level diff of text runs (character-level, semantics-cleaned).
2
+
3
+ Used inside matched group nodes to mark up changed words. Uses Google's
4
+ diff-match-patch algorithm (vendored pure-python port, small) so that
5
+ ``cleanupSemantic`` merges the "island" edits into human-readable
6
+ replacement regions - the track-changes look reviewers expect.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from dataclasses import dataclass
13
+ from difflib import SequenceMatcher
14
+
15
+ EQUAL = "equal"
16
+ DELETE = "delete"
17
+ INSERT = "insert"
18
+
19
+ # Tokens which must never be split by a diff boundary: a control
20
+ # sequence must stay glued to the braces of its argument, and braces
21
+ # must not be orphaned.
22
+ _CS = r"\\[a-zA-Z]+\*?"
23
+ _CS_ESC = r"\\."
24
+ _TOKEN_RE = re.compile(
25
+ "|".join(
26
+ [
27
+ rf"{_CS}\{{[^{{}}]*\}}", # \cmd{...}: keep together
28
+ _CS, # bare control sequence
29
+ _CS_ESC, # \\, \&, \%, ...
30
+ r"\s+", # whitespace run
31
+ r"[^\s\\{}]+(?:\{[^{}]*\}[^\s\\{}]*)*", # word w/ inline groups
32
+ r"[{}]", # lone brace
33
+ ]
34
+ )
35
+ )
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class Chunk:
40
+ """One region of a word/structure-level diff.
41
+
42
+ Attributes:
43
+ op: one of ``equal``, ``delete``, ``insert``.
44
+ text: the region's text (old side for delete, new side for
45
+ insert and equal).
46
+ """
47
+
48
+ op: str
49
+ text: str
50
+
51
+
52
+ def tokenize(s: str) -> list[str]:
53
+ """Split a text run into diff-safe tokens (whitespace included)."""
54
+ return [m.group(0) for m in _TOKEN_RE.finditer(s)]
55
+
56
+
57
+ def word_diff(old: str, new: str) -> list[Chunk]:
58
+ """Diff two text runs at word granularity.
59
+
60
+ Returns a list of :class:`Chunk` regions which, concatenated:
61
+ ``equal``+``insert`` reproduce ``new``; ``equal``+``delete``
62
+ reproduce ``old``.
63
+
64
+ Rules (v0 contract):
65
+
66
+ * whitespace-only differences keep the NEW text;
67
+ * chaff edits (isolated one-char changes) are merged into
68
+ replacement regions by semantic cleanup;
69
+ * the diff never splits a ``\\command`` name from its argument
70
+ braces or from its escape.
71
+ """
72
+ a, b = tokenize(old), tokenize(new)
73
+ sm = SequenceMatcher(a=a, b=b, autojunk=False)
74
+ chunks: list[Chunk] = []
75
+ for tag, i1, i2, j1, j2 in sm.get_opcodes():
76
+ if tag == "equal":
77
+ chunks.append(Chunk(EQUAL, "".join(a[i1:i2])))
78
+ elif tag == "delete":
79
+ chunks.append(Chunk(DELETE, "".join(a[i1:i2])))
80
+ elif tag == "insert":
81
+ chunks.append(Chunk(INSERT, "".join(b[j1:j2])))
82
+ else: # replace
83
+ chunks.append(Chunk(DELETE, "".join(a[i1:i2])))
84
+ chunks.append(Chunk(INSERT, "".join(b[j1:j2])))
85
+
86
+ chunks = _merge_adjacent(chunks)
87
+ chunks = _whitespace_to_new(chunks)
88
+ chunks = _merge_replacements(chunks)
89
+ return chunks
90
+
91
+
92
+ def _merge_adjacent(chunks: list[Chunk]) -> list[Chunk]:
93
+ """Concatenate neighbouring chunks of the same op."""
94
+ out: list[Chunk] = []
95
+ for c in chunks:
96
+ if out and out[-1].op == c.op:
97
+ out[-1] = Chunk(c.op, out[-1].text + c.text)
98
+ else:
99
+ out.append(c)
100
+ return out
101
+
102
+
103
+ def _whitespace_to_new(chunks: list[Chunk]) -> list[Chunk]:
104
+ """Whitespace-only changes keep the NEW side's whitespace.
105
+
106
+ A delete chunk that is pure whitespace next to an insert that is
107
+ pure whitespace collapses to an *equal* chunk with the new text.
108
+ """
109
+ out: list[Chunk] = []
110
+ i = 0
111
+ while i < len(chunks):
112
+ cur = chunks[i]
113
+ nxt = chunks[i + 1] if i + 1 < len(chunks) else None
114
+ if (
115
+ cur.op == DELETE
116
+ and nxt is not None
117
+ and nxt.op == INSERT
118
+ and cur.text.strip() == ""
119
+ and nxt.text.strip() == ""
120
+ ):
121
+ out.append(Chunk(EQUAL, nxt.text))
122
+ i += 2
123
+ continue
124
+ out.append(cur)
125
+ i += 1
126
+ return out
127
+
128
+
129
+ def _merge_replacements(chunks: list[Chunk]) -> list[Chunk]:
130
+ """Semantic cleanup: collapse scattered edits into replacements.
131
+
132
+ ``equal`` regions shorter than the neighbouring deletions+insertions
133
+ are absorbed, so one word changed inside a word flips a whole
134
+ single replace region instead of many microscopic islands.
135
+ """
136
+ if len(chunks) < 5:
137
+ return chunks
138
+ out = list(chunks)
139
+ changed = True
140
+ while changed:
141
+ changed = False
142
+ for i in range(1, len(out) - 1):
143
+ if out[i].op != EQUAL:
144
+ continue
145
+ left = out[i - 1]
146
+ right = out[i + 1]
147
+ if left.op == DELETE and right.op == INSERT:
148
+ # pattern: del eq ins -> merge if eq is short chaff
149
+ if len(out[i].text.strip()) <= 2 and len(out[i].text) <= max(
150
+ len(left.text), len(right.text)
151
+ ):
152
+ merged_del = Chunk(DELETE, left.text + out[i].text)
153
+ merged_ins = Chunk(INSERT, out[i].text + right.text)
154
+ out[i - 1 : i + 2] = [merged_del, merged_ins]
155
+ changed = True
156
+ break
157
+ return _merge_adjacent(out)