texdiff 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {texdiff-0.2.2/src/texdiff.egg-info → texdiff-0.2.3}/PKG-INFO +1 -1
  2. {texdiff-0.2.2 → texdiff-0.2.3}/pyproject.toml +1 -1
  3. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/align.py +35 -3
  4. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/api.py +35 -3
  5. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/emit.py +160 -2
  6. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/preamble.py +43 -31
  7. {texdiff-0.2.2 → texdiff-0.2.3/src/texdiff.egg-info}/PKG-INFO +1 -1
  8. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_align.py +20 -0
  9. texdiff-0.2.3/tests/test_api.py +247 -0
  10. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_preamble.py +54 -0
  11. texdiff-0.2.2/tests/test_api.py +0 -135
  12. {texdiff-0.2.2 → texdiff-0.2.3}/LICENSE +0 -0
  13. {texdiff-0.2.2 → texdiff-0.2.3}/README.md +0 -0
  14. {texdiff-0.2.2 → texdiff-0.2.3}/setup.cfg +0 -0
  15. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/__init__.py +0 -0
  16. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/check.py +0 -0
  17. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/cli.py +0 -0
  18. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/flatten.py +0 -0
  19. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/newlines.py +0 -0
  20. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/nodes.py +0 -0
  21. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/oldlines.py +0 -0
  22. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/parse.py +0 -0
  23. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/tables.py +0 -0
  24. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff/textdiff.py +0 -0
  25. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff.egg-info/SOURCES.txt +0 -0
  26. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff.egg-info/dependency_links.txt +0 -0
  27. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff.egg-info/entry_points.txt +0 -0
  28. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff.egg-info/requires.txt +0 -0
  29. {texdiff-0.2.2 → texdiff-0.2.3}/src/texdiff.egg-info/top_level.txt +0 -0
  30. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_alignment.py +0 -0
  31. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_check.py +0 -0
  32. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_cli.py +0 -0
  33. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_corpus.py +0 -0
  34. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_emit.py +0 -0
  35. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_flatten.py +0 -0
  36. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_parse.py +0 -0
  37. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_table_head_order.py +0 -0
  38. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_tables.py +0 -0
  39. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_tables_render.py +0 -0
  40. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_textdiff.py +0 -0
  41. {texdiff-0.2.2 → texdiff-0.2.3}/tests/test_verbatim.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: texdiff
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: AST-driven semantic diff for LaTeX documents
5
5
  Author: GoBobr
6
6
  License: MIT License
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "texdiff"
7
- version = "0.2.2"
7
+ version = "0.2.3"
8
8
  description = "AST-driven semantic diff for LaTeX documents"
9
9
  readme = "README.md"
10
10
  license = { file = "LICENSE" }
@@ -120,12 +120,17 @@ def _align_plain(old: list[Node], new: list[Node]) -> list[Edit]:
120
120
  return Modify(old=o, new=n)
121
121
 
122
122
  # 1. common prefix, paired positionally (identical to what a
123
- # matching-block run does - only the anchoring is stronger)
123
+ # matching-block run does - only the anchoring is stronger).
124
+ # Wildcard signatures additionally need equal text: for them
125
+ # signature equality means only "same kind of node", so an
126
+ # inserted sibling shifts every following wildcard by one and a
127
+ # signature-only prefix would pair the unchanged old item with
128
+ # the newly inserted neighbour.
124
129
  pre = 0
125
130
  while (
126
131
  pre < len(old)
127
132
  and pre < len(new)
128
- and old[pre].signature() == new[pre].signature()
133
+ and _anchorable(old[pre], new[pre])
129
134
  ):
130
135
  pre += 1
131
136
  # common suffix (may not overlap the prefix)
@@ -133,7 +138,9 @@ def _align_plain(old: list[Node], new: list[Node]) -> list[Edit]:
133
138
  while (
134
139
  suf < len(old) - pre
135
140
  and suf < len(new) - pre
136
- and old[len(old) - 1 - suf].signature() == new[len(new) - 1 - suf].signature()
141
+ and _anchorable(
142
+ old[len(old) - 1 - suf], new[len(new) - 1 - suf]
143
+ )
137
144
  ):
138
145
  suf += 1
139
146
  mid_old = old[pre : len(old) - suf]
@@ -612,6 +619,31 @@ def _content_similarity(a: "Node", b: "Node") -> float:
612
619
  return _text_similarity(a.text or "", b.text or "")
613
620
 
614
621
 
622
+ # wildcard signatures: equality means "same node kind", not
623
+ # "counterpart" - prefix/suffix anchoring demands equal text for them
624
+ _PLAIN_SIGS = ("text", "group", "env", "macro:label")
625
+
626
+
627
+ def _anchorable(o: Node, n: Node) -> bool:
628
+ """Whether two nodes may extend the prefix/suffix anchor.
629
+
630
+ Equal signature is a real anchor for SPECIFIC signatures (macro
631
+ names, sectioning titles, table row keys). The wildcard
632
+ signatures (``text``, ``group``, ...) say only "same kind of
633
+ node": an inserted sibling item shifts every following wildcard
634
+ by one, and a signature-only anchor then pairs the unchanged old
635
+ item with the newly inserted neighbour - spurious inline diffs -
636
+ while the old item's byte-identical counterpart ends up among
637
+ the inserts. For wildcards we therefore also demand equal text;
638
+ a genuinely rewritten paragraph simply anchors the diff at its
639
+ first differing node, which the sequence alignment over the
640
+ remaining middle then pairs as before.
641
+ """
642
+ if o.signature() != n.signature():
643
+ return False
644
+ return o.text == n.text or o.signature() not in _PLAIN_SIGS
645
+
646
+
615
647
  def _sibling_swap_rescue(edits: list[Edit], _pair) -> list[Edit]:
616
648
  """Re-pair a Modify whose new side is the WRONG same-sig sibling.
617
649
 
@@ -256,6 +256,27 @@ def _run_similarity(a: str, b: str) -> float:
256
256
  return SequenceMatcher(a=wa, b=wb, autojunk=False).ratio()
257
257
 
258
258
 
259
+ def _extends_by_words(short: str, long: str) -> bool:
260
+ """Whether one run is the other with words appended (or removed).
261
+
262
+ Text appended to a sentence (``... compression.`` growing into
263
+ ``... compression: the file structure ...``) drives the word
264
+ overlap ratio far below the replace threshold - the shared
265
+ prefix is drowned by the additions - yet the change IS an edit
266
+ of the same sentence: the word differ renders exactly the right
267
+ thing (struck final period, inserted tail). Prefix/suffix word
268
+ containment therefore overrides the ratio before a paragraph is
269
+ judged a wholesale rewrite.
270
+ """
271
+ wa = [w for w in (t.strip(".,;:!?()\"'`") for t in short.split()) if w]
272
+ wb = [w for w in (t.strip(".,;:!?()\"'`") for t in long.split()) if w]
273
+ if len(wa) >= len(wb):
274
+ wa, wb = wb, wa
275
+ if not wa or len(wb) - len(wa) < 3:
276
+ return False
277
+ return wb[: len(wa)] == wa or wb[len(wb) - len(wa):] == wa
278
+
279
+
259
280
  def _paragraph_edits(old: str, new: str) -> list[Edit] | None:
260
281
  """Word-refine a text run paragraph by paragraph.
261
282
 
@@ -283,10 +304,13 @@ def _paragraph_edits(old: str, new: str) -> list[Edit] | None:
283
304
  if len(paras_a) != len(paras_b) or any(
284
305
  not p.strip() for p in paras_a + paras_b
285
306
  ):
286
- # mismatched paragraph structure: whole-run judgement
307
+ # mismatched paragraph structure: whole-run judgement.
308
+ # (A word-prefix run is still an edit of the same sentence,
309
+ # never a rewrite - keep the word-level treatment.)
287
310
  return (
288
311
  None
289
312
  if _run_similarity(old, new) < _REPLACE_WORD_SIMILARITY
313
+ and not _extends_by_words(old, new)
290
314
  else _chunks_to_edits(word_diff(old, new))
291
315
  )
292
316
 
@@ -294,9 +318,17 @@ def _paragraph_edits(old: str, new: str) -> list[Edit] | None:
294
318
  # paragraph separator, kept from the NEW side when it exists
295
319
  return Match(node=text_node(text if text.strip() else "\n\n"))
296
320
 
297
- out: list[Edit] = [sep(lead_b or lead_a)]
321
+ # the lead is leading whitespace of the run (often empty): an
322
+ # EMPTY lead must not fabricate a "\n\n" paragraph break that
323
+ # neither revision had - it would typeset a blank line at the
324
+ # start of the marked run (e.g. right after a list item's bold
325
+ # label, splitting the item into "label." + "rest" lines)
326
+ out: list[Edit] = [Match(node=text_node(lead_b or lead_a))]
298
327
  for k, (ca, cb) in enumerate(zip(paras_a, paras_b)):
299
- if _run_similarity(ca, cb) < _REPLACE_WORD_SIMILARITY:
328
+ if (
329
+ _run_similarity(ca, cb) < _REPLACE_WORD_SIMILARITY
330
+ and not _extends_by_words(ca, cb)
331
+ ):
300
332
  out.append(Delete(old=text_node(ca)))
301
333
  out.append(Insert(new=text_node(cb)))
302
334
  else:
@@ -162,11 +162,125 @@ def render(edits: list[Edit], markup: LatexdiffMarkup = LatexdiffMarkup()) -> st
162
162
  else:
163
163
  out.append(_wrap_node(edit.old, markup, added=False))
164
164
  out.append(_wrap_node(edit.new, markup, added=True))
165
- return _collapse_repeated_hlines(
166
- _normalize_endmark_colours("".join(out))
165
+ return _strike_deleted_blocks(
166
+ _collapse_repeated_hlines(
167
+ _collapse_marker_glue(
168
+ _hoist_markers_off_macros(_normalize_endmark_colours("".join(out)))
169
+ )
170
+ )
167
171
  )
168
172
 
169
173
 
174
+ # a block marker sitting directly after a control-sequence name.
175
+ # TeX expands tokens while scanning a macro's argument, so a marker
176
+ # glued after an argument-taking macro (\sphinxcode\DIFaddend{...})
177
+ # executes INSIDE the argument's group scope: the colour switch it
178
+ # performs never unbalances at the outer level and the added colour
179
+ # bleeds over all following black material. Hoisting the marker in
180
+ # front of the macro makes it execute at the outer level where the
181
+ # emitter intended it. The macro part must not itself be a DIF
182
+ # marker (adjacent markers may not commute).
183
+ _MARKER_AFTER_MACRO_RE = re.compile(
184
+ r"(\\(?!DIF(?:add|del))[a-zA-Z]+\*?)"
185
+ r"((?:\\DIF(?:add|del)(?:begin|end)(?:FL)?)[ \t]*\n?)"
186
+ )
187
+
188
+
189
+ def _hoist_markers_off_macros(text: str) -> str:
190
+ """Move block markers from after a macro name to before it."""
191
+ if "\\DIFadd" not in text and "\\DIFdel" not in text:
192
+ return text
193
+ return _MARKER_AFTER_MACRO_RE.sub(r"\2\1", text)
194
+
195
+
196
+ # a whole deleted block: text between the block markers. Only spans
197
+ # that do not nest and do not already carry word marks are struck;
198
+ # within them, a strike candidate is a text fragment WITHOUT any
199
+ # backslash (macro), brace, ampersand, comment or math shift -
200
+ # prose fragments between inline macro calls strike, entire lines
201
+ # carrying macros stay red-unstruck
202
+ _STRIKE_FRAGMENT_RE = re.compile(r"[^\\{}&%$]*[A-Za-z]{3,}[^\\{}&%$]*")
203
+
204
+ # a del-block span, non-greedy, not nested (del begin..end never
205
+ # nests another del begin by construction of the emitter)
206
+ _DEL_SPAN_RE = re.compile(
207
+ r"(\\DIFdelbegin(?:FL)?)(.*?)(\\DIFdelend(?:FL)?)", re.S
208
+ )
209
+
210
+
211
+ def _strike_deleted_blocks(text: str) -> str:
212
+ """Strike plain text lines inside whole deleted blocks.
213
+
214
+ A block deletion (a retired environment or list item) colours
215
+ its span red via ``\\DIFdelbegin`` but never strikes it: the
216
+ ``\\sout`` of a single \\DIFdel{..} wrapper cannot span the
217
+ environment/structure inside. The plain prose between the
218
+ structural commands CAN be struck though, line by line, one
219
+ ``\\sout{..}`` per word (ulem cannot span the line break
220
+ itself). Lines containing anything TeX-structural (a macro,
221
+ brace, ampersand, comment, math) keep the plain red colour -
222
+ striking those would not compile or would mangle the markup.
223
+ Lines already carrying an inline \\DIFdel{..} word mark (mixed
224
+ word-level diffs inside a block region) stay untouched.
225
+ """
226
+ if "\\DIFdelbegin" not in text:
227
+ return text
228
+
229
+ def _strike_span(m: "re.Match") -> str:
230
+ body = m.group(2)
231
+ if "\\DIFdel{" in body or "\\DIFadd{" in body:
232
+ return m.group(0) # already word-marked content
233
+ if "\\sout{" in body:
234
+ return m.group(0) # a per-word strike already present
235
+ if not re.search(r"[A-Za-z]{3,}", body):
236
+ return m.group(0) # structural content only, all red
237
+ # strike the longest safe fragment of each line: everything
238
+ # up to the first TeX-structural character; the remainder
239
+ # (macro calls, braces) keeps the plain red colour
240
+ out_lines = []
241
+ for line in body.split("\n"):
242
+ frag = _STRIKE_FRAGMENT_RE.match(line)
243
+ if not frag or not frag.group(0).strip():
244
+ out_lines.append(line)
245
+ continue
246
+ head = frag.group(0)
247
+ words = head.split()
248
+ struck = " ".join(f"\\sout{{{w}}}" for w in words)
249
+ out_lines.append(struck + line[len(head):])
250
+ return m.group(1) + "\n".join(out_lines) + m.group(3)
251
+
252
+ return _DEL_SPAN_RE.sub(_strike_span, text)
253
+
254
+
255
+ # markup-glue blank lines: a whitespace-only line between block
256
+ # markers of the same type is fabricated by the marker's own
257
+ # trailing newline plus matched glue - TeX reads it as \par and
258
+ # visibly splits an inserted list item's number from its content
259
+ # ("8." alone, blank line, then the blue text). Whitespace-only
260
+ # content carries no information: no revision ever had a blank line
261
+ # between markers with nothing renderable between them.
262
+ _MARKER_GLUE_RE = re.compile(
263
+ r"(\\DIF(?:add|del)(?:begin|end)(?:FL)?)"
264
+ r"((?:(?:[ \t]*\n)+[ \t]*|\\DIFadd\{\{\}\})+)"
265
+ r"(\\DIF(?:add|del)(?:begin|end)(?:FL)?)"
266
+ )
267
+
268
+
269
+ def _collapse_marker_glue(text: str) -> str:
270
+ """Merge blank-line runs between adjacent block markers.
271
+
272
+ Only glue that would typeset nothing - whitespace-only lines or
273
+ an empty ``\\DIFadd{{}}`` wrapper - is collapsed between markers
274
+ of the same family; real content between markers keeps its
275
+ paragraphs untouched.
276
+ """
277
+ while True:
278
+ new = _MARKER_GLUE_RE.sub(r"\1\3", text)
279
+ if new == text:
280
+ return new
281
+ text = new
282
+
283
+
170
284
  # inline end-marker tokens of longtable HEAD material; NOT preceded
171
285
  # by an existing colour declaration (the lookbehind avoids doubling
172
286
  # up on lines the row renderers already reset). Foot markers
@@ -861,10 +975,37 @@ def _wrap_node(node: Node, markup: LatexdiffMarkup, added: bool) -> str:
861
975
  # outside markup - \DIFdel{a & b} is illegal in alignment
862
976
  return _wrap_row(node, markup, added)
863
977
  if _needs_block(node):
978
+ if _SAFE_DECOR_RE.fullmatch(node.text.strip()) and _is_safe_inline(
979
+ _SAFE_DECOR_RE.fullmatch(node.text.strip()).group(2)
980
+ ):
981
+ # a decoration macro WITH its complete argument
982
+ # (\textbf{...} from Sphinx's \sphinxstylestrong etc.)
983
+ # is LR-safe to wrap as a whole: the markup braces
984
+ # cannot steal the macro's argument because the node
985
+ # text carries it. The block markers' trailing newlines
986
+ # would instead insert a paragraph break after the
987
+ # wrapped label (a lone \DIFaddend\n between label and
988
+ # text leaves a blank line at the item's first row).
989
+ pad_l = node.text[: len(node.text) - len(node.text.lstrip())]
990
+ pad_r = node.text[len(node.text.rstrip()) :]
991
+ core = node.text.strip()
992
+ if added:
993
+ return pad_l + _wrap(core, markup.add_open, markup.add_close) + pad_r
994
+ return pad_l + _wrap(core, markup.del_open, markup.del_close) + pad_r
864
995
  if added:
865
996
  body = _mark_added_listings(node.text)
866
997
  body = _mark_heading_args_in_run(body)
867
998
  return f"{markup.block_add_open}{body}{markup.block_add_close}"
999
+ if _DEFINES_MACRO_RE.search(node.text):
1000
+ # a deleted macro definition would not *render* in the
1001
+ # red-strike region - it would EXECUTE at typeset time
1002
+ # and silently redefine the macro, overriding whatever
1003
+ # the (possibly marked-up) preamble or an earlier part
1004
+ # of the body established. latexdiff's convention for
1005
+ # deleted commands applies: comment the definition out
1006
+ # (%DIFDELCMD), invisible in the output, reviewable in
1007
+ # the source.
1008
+ return _comment_out(node.text)
868
1009
  return f"{markup.block_del_open}{node.text}{markup.block_del_close}"
869
1010
  if added:
870
1011
  body = _mark_heading_args_in_run(node.text)
@@ -1052,6 +1193,16 @@ def _wrap_row(node: Node, markup: LatexdiffMarkup, added: bool) -> str:
1052
1193
  # list/paragraph primitives that cannot appear inside \uwave/\sout
1053
1194
  _LIST_ITEM_RE = re.compile(r"\\(?:item|par|newline|linebreak|cr)\b")
1054
1195
 
1196
+ # a macro definition inside a deleted block: \def/\gdef/\edef/\xdef
1197
+ # directly followed by the defined name, or a \newcommand/\renewcommand
1198
+ # whose first argument is the defined name. Matching only the OPENING
1199
+ # token keeps false positives near zero (\definedcolor etc. do not
1200
+ # parse as \def + name).
1201
+ _DEFINES_MACRO_RE = re.compile(
1202
+ r"\\(?:gdef|edef|xdef|def)\s*\\[a-zA-Z]+\s*\{"
1203
+ r"|\\(?:re)?newcommand\*?\s*\{\s*\\[a-zA-Z]+\s*\}"
1204
+ )
1205
+
1055
1206
 
1056
1207
  def _comment_out(text: str) -> str:
1057
1208
  """Comment out each line, latexdiff ``%DIFDELCMD <`` convention.
@@ -1570,6 +1721,13 @@ def _render_row_region(edits: list[Edit], markup: LatexdiffMarkup) -> str:
1570
1721
  # single edit: normal render path, but recurse for inner lists
1571
1722
  if isinstance(e, Modify) and e.inner is not None:
1572
1723
  out.append(_render_recursed(e, markup))
1724
+ elif isinstance(e, Modify):
1725
+ # a plain Modify (no children to recurse into) used to
1726
+ # fall through every branch and vanish from the output -
1727
+ # e.g. a reworded \textbf{label} macro inside a modified
1728
+ # list item. Render both sides with the normal wrap.
1729
+ out.append(_wrap_node(e.old, markup, added=False))
1730
+ out.append(_wrap_node(e.new, markup, added=True))
1573
1731
  elif isinstance(e, Match):
1574
1732
  txt = e.node.text
1575
1733
  _record_tail(txt, emitted_tails)
@@ -190,7 +190,10 @@ def mark_preamble_macro_changes(
190
190
  per-cell colour treatment the emitter uses for unsafe added runs
191
191
  - ``\\color{blue}`` re-started after each ``&``. Deleted lines are
192
192
  commented out (``%DIF <``, latexdiff preamble convention) so the
193
- old rows stay visible in the source without typesetting.
193
+ old rows stay visible in the source without typesetting. The
194
+ definition and pure closing lines of a tracked macro stay
195
+ verbatim, so the colour declaration cannot leak out of the cell
196
+ groups (see :func:`_color_body_line`).
194
197
 
195
198
  The operation is textual and strictly contained in the preamble;
196
199
  it cannot affect compilation because ``\\color`` inside a macro
@@ -200,6 +203,11 @@ def mark_preamble_macro_changes(
200
203
  pre_new, body_new, _post_new = split_preamble(new_source)
201
204
  if not pre_new:
202
205
  return marked_up
206
+ # The old counterpart of a preamble macro may be defined in the
207
+ # old BODY (macros can migrate between revisions): search the
208
+ # whole old document, so a moved definition keeps its unchanged
209
+ # rows unmarked instead of painting the entire table blue.
210
+ old_def = old_source
203
211
  names = _invoked_preamble_macros(pre_new, body_new)
204
212
  # only macro bodies that are typeset as TABLES may carry the
205
213
  # per-cell colour markup: a \color is only meaningful - and only
@@ -216,35 +224,13 @@ def mark_preamble_macro_changes(
216
224
  if not changed and not deleted:
217
225
  return marked_up
218
226
  return (
219
- _apply_macro_markup(pre_old, pre_new, changed, deleted, names)
227
+ _apply_macro_markup(old_def, pre_new, changed, deleted, names)
220
228
  + marked_up[len(pre_new) :]
221
229
  )
222
230
 
223
231
 
224
232
  def _apply_macro_markup(
225
- old_pre: str,
226
- new_pre: str,
227
- changed: list[str],
228
- deleted: list[str],
229
- names: set[str],
230
- ) -> str:
231
- """Merge the old and new tracked macro bodies, marked per line.
232
-
233
- Emits the NEW preamble with, inside each tracked macro body:
234
-
235
- * genuinely new lines prefixed ``\\color{blue}`` per cell;
236
- * lines that existed only in the old body re-inserted at their
237
- original position as ``%DIF <`` comments (latexdiff preamble
238
- convention: invisible in the typeset output, reviewable in
239
- source).
240
-
241
- The merge aligns the old and new body lines on their normalized
242
- content (SequenceMatcher over normalized lines): equal lines are
243
- kept verbatim, old-only lines become comments, new-only lines
244
- are coloured blue.
245
- """
246
- def _apply_macro_markup(
247
- old_pre: str,
233
+ old_def: str,
248
234
  new_pre: str,
249
235
  changed: list[str],
250
236
  deleted: list[str],
@@ -262,15 +248,17 @@ def _apply_macro_markup(
262
248
 
263
249
  Each macro's old and new body lines are aligned on their
264
250
  normalized content (SequenceMatcher): equal lines kept verbatim,
265
- old-only lines commented, new-only lines coloured blue. Bodies
266
- are spliced back in reverse document order so earlier line
267
- indices stay valid.
251
+ old-only lines commented, new-only lines coloured blue.
252
+
253
+ ``old_def`` carries the OLD side text the macro bodies are read
254
+ from - usually the old preamble, but a whole document when the
255
+ macro was defined in the old body (migration case).
268
256
  """
269
257
  pre_out = new_pre.splitlines()
270
258
  spliced = False
271
259
  for name, rng in _body_ranges(new_pre, names):
272
260
  new_body = _macro_body(new_pre, name)
273
- old_body = _macro_body(old_pre, name)
261
+ old_body = _macro_body(old_def, name)
274
262
  sm = difflib.SequenceMatcher(
275
263
  a=[_norm_line(l) for l in old_body],
276
264
  b=[_norm_line(l) for l in new_body],
@@ -290,10 +278,10 @@ def _apply_macro_markup(
290
278
  elif tag == "delete":
291
279
  merged.extend(f"%DIF < {l}" for l in old_body[i1:i2])
292
280
  elif tag == "insert":
293
- merged.extend(_color_line(l) for l in new_body[j1:j2])
281
+ merged.extend(_color_body_line(l) for l in new_body[j1:j2])
294
282
  else: # replace
295
283
  merged.extend(f"%DIF < {l}" for l in old_body[i1:i2])
296
- merged.extend(_color_line(l) for l in new_body[j1:j2])
284
+ merged.extend(_color_body_line(l) for l in new_body[j1:j2])
297
285
  if merged != new_body or len(merged) != len(new_body):
298
286
  pre_out[rng[0] : rng[1]] = merged
299
287
  spliced = True
@@ -389,3 +377,27 @@ def _color_line(line: str) -> str:
389
377
  """Prepend \\color{blue} per table cell (after every &)."""
390
378
  line = f"\\color{{blue}} {line}"
391
379
  return re.sub(r"(?<!\\)&", r"& \\color{blue} ", line)
380
+
381
+
382
+ def _color_body_line(line: str) -> str:
383
+ """Colour one macro-body line blue, unless it must stay verbatim.
384
+
385
+ Two kinds of lines are returned untouched:
386
+
387
+ * any definition/newcommand opener (``\\def\\name{...``) - a
388
+ ``\\color`` painted on the macro's own definition line
389
+ executes while the PREAMBLE is being read, before any group
390
+ scopes it, and the declaration then leaks over the entire
391
+ document (table captions, running headers and page numbers
392
+ all render blue);
393
+ * a line that is nothing but braces/whitespace/comment (a pure
394
+ closing ``}``) - it carries no content and no cell group to
395
+ contain the declaration.
396
+ """
397
+ code = _strip_comment(line)
398
+ if _DEF_RE.search(code) or _NEWCOMMAND_RE.search(code):
399
+ return line
400
+ if _is_closing_line(line):
401
+ return line
402
+ line = f"\\color{{blue}} {line}"
403
+ return re.sub(r"(?<!\\)&", r"& \\color{blue} ", line)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: texdiff
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: AST-driven semantic diff for LaTeX documents
5
5
  Author: GoBobr
6
6
  License: MIT License
@@ -69,3 +69,23 @@ class TestSameSignatureDifferentText:
69
69
  mods = [e for e in edits if isinstance(e, Modify)]
70
70
  assert len(mods) == 1
71
71
  assert mods[0].new.children[0].text == "B"
72
+
73
+ def test_inserted_sibling_does_not_steal_wildcard_anchor(self):
74
+ # one item inserted into a list whose entries are wildcard-
75
+ # signature nodes (texts and plain groups): the unchanged
76
+ # old tail must Match verbatim, not inline-diff against the
77
+ # inserted neighbour
78
+ old = [text_node("a\n\n"), text_node("b long body\n\n"), text_node("c\n\n")]
79
+ new = [text_node("a\n\n"), text_node("NEW inserted\n\n"), text_node("b long body\n\n"), text_node("c\n\n")]
80
+ edits = align(old, new)
81
+ assert [type(e).__name__ for e in edits] == [
82
+ "Match", "Insert", "Match", "Match",
83
+ ]
84
+
85
+ def test_inserted_group_sibling_leaves_equal_groups_matched(self):
86
+ # same scenario with plain groups (e.g. {\sphinxupquote{...}}
87
+ # label groups): equal-text groups anchor, unequal ones do not
88
+ old = [group(text_node("handlers")), text_node("tail\n\n")]
89
+ new = [group(text_node("schemas")), group(text_node("handlers")), text_node("tail\n\n")]
90
+ edits = align(old, new)
91
+ assert [type(e).__name__ for e in edits] == ["Insert", "Match", "Match"]
@@ -0,0 +1,247 @@
1
+ """End-to-end contract tests for the public API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import pytest
6
+
7
+ from texdiff import DiffResult, diff_documents, diff_files
8
+
9
+
10
+ OLD_DOC = """\
11
+ \\documentclass{article}
12
+ \\begin{document}
13
+ The sensor measures temperature.
14
+ \\begin{itemize}
15
+ \\item accuracy
16
+ \\item range
17
+ \\end{itemize}
18
+ Formula: $E = mc^2$
19
+ \\end{document}
20
+ """
21
+
22
+ NEW_DOC = """\
23
+ \\documentclass{article}
24
+ \\begin{document}
25
+ The instrument measures temperature.
26
+ \\begin{itemize}
27
+ \\item accuracy
28
+ \\item dynamic range
29
+ \\item stability
30
+ \\end{itemize}
31
+ Formula: $E = mc^2$
32
+ \\end{document}
33
+ """
34
+
35
+
36
+ class TestDiffDocuments:
37
+ def test_returns_result_object(self):
38
+ r = diff_documents(OLD_DOC, NEW_DOC)
39
+ assert isinstance(r, DiffResult)
40
+
41
+ def test_stats_count_changes(self):
42
+ r = diff_documents(OLD_DOC, NEW_DOC)
43
+ assert r.stats.changed > 0
44
+
45
+ def test_identical_docs_no_changes(self):
46
+ r = diff_documents(OLD_DOC, OLD_DOC)
47
+ assert r.stats.changed == 0
48
+ assert r.marked_up == OLD_DOC
49
+
50
+ def test_markup_present(self):
51
+ r = diff_documents(OLD_DOC, NEW_DOC)
52
+ assert "\\DIFadd{" in r.marked_up or "\\DIFdel{" in r.marked_up
53
+
54
+ def test_unchanged_lines_verbatim(self):
55
+ r = diff_documents(OLD_DOC, NEW_DOC)
56
+ assert "Formula: $E = mc^2$" in r.marked_up
57
+
58
+ def test_preamble_injected_before_begin_document(self):
59
+ r = diff_documents(OLD_DOC, NEW_DOC)
60
+ assert "\\providecommand{\\DIFadd}" in r.marked_up
61
+ # injected before \begin{document}, i.e. into the preamble
62
+ assert r.marked_up.index("\\providecommand{\\DIFadd}") < r.marked_up.index(
63
+ "\\begin{document}"
64
+ )
65
+
66
+ def test_preamble_not_injected_when_already_defined(self):
67
+ doc = OLD_DOC.replace(
68
+ "\\begin{document}",
69
+ "\\providecommand{\\DIFadd}[1]{#1}\n\\providecommand{\\DIFdel}[1]{#1}\n\\begin{document}",
70
+ )
71
+ r = diff_documents(doc, doc)
72
+ marked = r.marked_up
73
+ assert marked.count("\\providecommand{\\DIFadd}") == 1
74
+
75
+ def test_no_preamble_for_fragments(self):
76
+ r = diff_documents("plain fragment old\n", "plain fragment new\n")
77
+ assert "\\providecommand" not in r.marked_up
78
+
79
+
80
+ class TestDiffFiles:
81
+ def test_file_api(self, tmp_path):
82
+ old = tmp_path / "old.tex"
83
+ new = tmp_path / "new.tex"
84
+ old.write_text(OLD_DOC, encoding="utf-8")
85
+ new.write_text(NEW_DOC, encoding="utf-8")
86
+ r = diff_files(str(old), str(new))
87
+ assert "\\DIFadd{" in r.marked_up or "\\DIFdel{" in r.marked_up
88
+
89
+
90
+ class TestTableScenario:
91
+ """The scenario that breaks latexdiff: restructured longtable rows."""
92
+
93
+ OLD_TABLE = """\
94
+ \\begin{longtable}{|l|l|}
95
+ \\hline
96
+ name & type \\\\
97
+ \\hline
98
+ longitude & float32 \\\\
99
+ latitude & float32 \\\\
100
+ \\hline
101
+ \\end{longtable}
102
+ """
103
+
104
+ NEW_TABLE = """\
105
+ \\begin{longtable}{|l|l|}
106
+ \\hline
107
+ name & type \\\\
108
+ \\hline
109
+ latitude & float32 \\\\
110
+ longitude & float32 \\\\
111
+ \\hline
112
+ \\end{longtable}
113
+ """
114
+
115
+ def test_row_reorder_produces_valid_markup(self):
116
+ r = diff_documents(self.OLD_TABLE, self.NEW_TABLE)
117
+ # rows reordered: aligned/merged, never glued together
118
+ assert "longitude" not in r.marked_up.split("latitude")[0].split("\\DIFdel")[0] or True
119
+ assert "\\begin{longtable}" in r.marked_up
120
+ assert "\\end{longtable}" in r.marked_up
121
+
122
+ def test_row_insertion_marks_row(self):
123
+ old = self.OLD_TABLE
124
+ extra = old.replace(
125
+ "latitude & float32 \\\\\n",
126
+ "latitude & float32 \\\\\npolarised & float \\\\\n",
127
+ )
128
+ r = diff_documents(old, extra)
129
+ # v0 contract: tables are atomic (row-granular alignment is
130
+ # v1), so the whole table is block-replaced - but the output
131
+ # must still contain the added row and compile-safe markers
132
+ assert "\\DIFaddbegin" in r.marked_up
133
+ assert "polarised" in r.marked_up
134
+ assert "\\begin{longtable}" in r.marked_up
135
+ assert "\\end{longtable}" in r.marked_up
136
+
137
+
138
+ class TestModifiedLabelledItem:
139
+ """A reworded item of a labelled list (bold head + text body).
140
+
141
+ Sphinx list items render as ``\\item {}`` + ``\\par`` +
142
+ ``\\sphinxstylestrong{Label}: body``; the bold head is an atomic
143
+ macro node and the body a separate text node. Both used to take
144
+ block markers whose trailing newlines - plus a fabricated ``\\n\\n``
145
+ lead from the paragraph refine - typeset a blank line between the
146
+ item number and the (struck/added) content, breaking the item
147
+ across two visual lines.
148
+ """
149
+
150
+ OLD = """\
151
+ \\documentclass{article}
152
+ \\begin{document}
153
+ \\begin{enumerate}
154
+ \\item {}
155
+ \\par
156
+ \\textbf{Output packaging}: Processed files packaged by the IOHandler.
157
+ \\end{enumerate}
158
+ \\end{document}
159
+ """
160
+
161
+ def _diff(self, new_body: str) -> str:
162
+ new = self.OLD.replace(
163
+ "\\textbf{Output packaging}: Processed files packaged by the IOHandler.",
164
+ new_body,
165
+ )
166
+ return diff_documents(self.OLD, new, inject_preamble=False).marked_up
167
+
168
+ def test_no_paragraph_break_after_item_label(self):
169
+ out = self._diff(
170
+ "\\textbf{Schema-driven output}: Each file created by the writer."
171
+ )
172
+ # the label keeps the inline wrap (no block markers with
173
+ # their trailing newlines) and stays glued to the body:
174
+ # no blank line between the item label and the text run
175
+ assert "\\DIFdelbegin\n" not in out
176
+ assert "\\textbf{Output packaging}\\DIFdelend" not in out
177
+ assert "\\DIFdel{\\textbf{Output packaging}}" in out
178
+ assert "\\DIFadd{\\textbf{Schema-driven output}}" in out
179
+ assert "\\DIFdel{\\textbf{Output packaging}}\\DIFadd{" in out
180
+
181
+ def test_reworded_plain_item_stays_one_paragraph(self):
182
+ out = self._diff(
183
+ "\\textbf{Output packaging}: Each file is packaged into SAFE containers."
184
+ )
185
+ # same label, reworded body: the label stays visible right
186
+ # after \par and no blank line opens between it and the
187
+ # marked-up body
188
+ assert "\\textbf{Output packaging}" in out
189
+ assert "par\n\\textbf" in out
190
+ assert "\\textbf{Output packaging}\n\n" not in out
191
+ assert "\\DIFdel{" in out and "\\DIFadd{" in out
192
+
193
+
194
+ class TestAppendedClause:
195
+ """A sentence extended with a new clause stays word-diffed.
196
+
197
+ The overlap ratio of "... compression." growing into
198
+ "... compression: the file structure ..." falls below the
199
+ replace threshold, yet the change is an edit of the same
200
+ sentence - not a wholesale rewrite. The word differ renders it
201
+ as a struck final period plus an inserted tail.
202
+ """
203
+
204
+ OLD = (
205
+ "\\documentclass{article}\n\\begin{document}\n"
206
+ "Write output to NetCDF4 with standardised metadata and compression.\n"
207
+ "\\end{document}\n"
208
+ )
209
+
210
+ def _diff(self, new_body: str) -> str:
211
+ new = self.OLD.replace(
212
+ "Write output to NetCDF4 with standardised metadata and compression.",
213
+ new_body,
214
+ )
215
+ return diff_documents(self.OLD, new, inject_preamble=False).marked_up
216
+
217
+ def test_appended_clause_is_word_diffed(self):
218
+ out = self._diff(
219
+ "Write output to NetCDF4 with standardised metadata and "
220
+ "compression: the file structure (dimensions, variables, "
221
+ "types, fill values, attributes) is defined by the NCML "
222
+ "product schema for the output type and written through "
223
+ "the generic schema-driven writer."
224
+ )
225
+ # the shared sentence is kept and the change is inline word
226
+ # markup (both families present), NOT a whole-paragraph
227
+ # retire + re-add, which would strike the full old sentence
228
+ assert "Write output to NetCDF4 with standardised metadata" in out
229
+ assert "\\DIFadd{" in out
230
+ assert "\\DIFdel{" in out
231
+ # the struck material is punctuation/small words only - the
232
+ # sentence body is never deleted wholesale
233
+ import re as _re
234
+
235
+ for m in _re.finditer(r"\\DIFdel\{([^{}]*)\}", out):
236
+ words = [w for w in m.group(1).split() if w.isalpha()]
237
+ assert len(words) <= 2, m.group(1)
238
+
239
+ def test_genuine_rewrite_still_retires(self):
240
+ out = self._diff(
241
+ "Completely different content that shares no words at all here."
242
+ )
243
+ # unrelated sentence: whole-paragraph retire + re-add - the
244
+ # full old sentence is struck in one piece (block or whole-
245
+ # paragraph inline form, never word fragments)
246
+ assert "\\DIFdel{Write output to NetCDF4" in out
247
+ assert "\\DIFadd{Completely different" in out
@@ -14,6 +14,8 @@ from __future__ import annotations
14
14
 
15
15
  import pytest
16
16
 
17
+ import re
18
+
17
19
  from texdiff.api import diff_documents
18
20
  from texdiff.preamble import new_preamble, preamble_tuple, split_preamble
19
21
 
@@ -151,3 +153,55 @@ class TestMacroBodyMarkup:
151
153
  out = self._run(old_rows, new_rows)
152
154
  assert out.count("2 & 2023 & & DCR1") == 1
153
155
  assert "\\hline}" in out
156
+
157
+ def test_definition_line_never_coloured(self):
158
+ # a \color painted on the macro's own definition line would
159
+ # execute while the PREAMBLE is being read - no group scopes
160
+ # it there - leaking blue over the entire document (table
161
+ # captions, running headers, page numbers)
162
+ old_rows = "1 & 2022 & & Init \\\\\n\\hline"
163
+ new_rows = "1 & 2022 & & Init \\\\\n2 & 2023 & & DCR1 \\\\\n\\hline"
164
+ out = self._run(old_rows, new_rows)
165
+ assert "\\color{blue} \\def\\changerecord{" not in out
166
+ start = out.index("\\def\\changerecord{")
167
+ assert not out[start:].startswith("\\color")
168
+
169
+ def test_migrated_macro_keeps_unchanged_rows_black(self):
170
+ # the macro defined in the old BODY and in the new PREAMBLE:
171
+ # the old body must be searched for the counterpart rows, or
172
+ # every row (including unchanged ones) is painted blue
173
+ old = (
174
+ "\\documentclass{article}\n"
175
+ "\\begin{document}\n"
176
+ "\\def\\changerecord{%\n"
177
+ "1 & 2022 & & Init \\\\\n"
178
+ "}\n"
179
+ "\\begin{longtable}{ll}\n\\changerecord\n\\end{longtable}\n"
180
+ "\\end{document}\n"
181
+ )
182
+ new = (
183
+ "\\documentclass{article}\n"
184
+ "\\def\\changerecord{%\n"
185
+ "1 & 2022 & & Init \\\\\n"
186
+ "2 & 2023 & & DCR1 \\\\\n"
187
+ "}\n"
188
+ "\\begin{document}\n"
189
+ "\\begin{longtable}{ll}\n\\changerecord\n\\end{longtable}\n"
190
+ "\\end{document}\n"
191
+ )
192
+ out = diff_documents(old, new).marked_up
193
+ pre = out[: out.index("\\begin{document}")]
194
+ # the unchanged row 1 keeps no colour declaration ahead of it
195
+ assert "\\color{blue} 1 & 2022" not in pre
196
+ # the new row 2 is coloured per cell
197
+ assert re.search(
198
+ r"\\color\{blue\} 2 & \\color\{blue\}\s+2023", pre
199
+ )
200
+ # the old body-side definition no longer typesets: it would
201
+ # silently redefine the macro and override the marked one
202
+ # (it must survive only inside %DIFDELCMD comments)
203
+ body = out[out.index("\\begin{document}") :]
204
+ body_code = "\n".join(
205
+ l for l in body.splitlines() if not l.lstrip().startswith("%")
206
+ )
207
+ assert "\\def\\changerecord{" not in body_code
@@ -1,135 +0,0 @@
1
- """End-to-end contract tests for the public API."""
2
-
3
- from __future__ import annotations
4
-
5
- import pytest
6
-
7
- from texdiff import DiffResult, diff_documents, diff_files
8
-
9
-
10
- OLD_DOC = """\
11
- \\documentclass{article}
12
- \\begin{document}
13
- The sensor measures temperature.
14
- \\begin{itemize}
15
- \\item accuracy
16
- \\item range
17
- \\end{itemize}
18
- Formula: $E = mc^2$
19
- \\end{document}
20
- """
21
-
22
- NEW_DOC = """\
23
- \\documentclass{article}
24
- \\begin{document}
25
- The instrument measures temperature.
26
- \\begin{itemize}
27
- \\item accuracy
28
- \\item dynamic range
29
- \\item stability
30
- \\end{itemize}
31
- Formula: $E = mc^2$
32
- \\end{document}
33
- """
34
-
35
-
36
- class TestDiffDocuments:
37
- def test_returns_result_object(self):
38
- r = diff_documents(OLD_DOC, NEW_DOC)
39
- assert isinstance(r, DiffResult)
40
-
41
- def test_stats_count_changes(self):
42
- r = diff_documents(OLD_DOC, NEW_DOC)
43
- assert r.stats.changed > 0
44
-
45
- def test_identical_docs_no_changes(self):
46
- r = diff_documents(OLD_DOC, OLD_DOC)
47
- assert r.stats.changed == 0
48
- assert r.marked_up == OLD_DOC
49
-
50
- def test_markup_present(self):
51
- r = diff_documents(OLD_DOC, NEW_DOC)
52
- assert "\\DIFadd{" in r.marked_up or "\\DIFdel{" in r.marked_up
53
-
54
- def test_unchanged_lines_verbatim(self):
55
- r = diff_documents(OLD_DOC, NEW_DOC)
56
- assert "Formula: $E = mc^2$" in r.marked_up
57
-
58
- def test_preamble_injected_before_begin_document(self):
59
- r = diff_documents(OLD_DOC, NEW_DOC)
60
- assert "\\providecommand{\\DIFadd}" in r.marked_up
61
- # injected before \begin{document}, i.e. into the preamble
62
- assert r.marked_up.index("\\providecommand{\\DIFadd}") < r.marked_up.index(
63
- "\\begin{document}"
64
- )
65
-
66
- def test_preamble_not_injected_when_already_defined(self):
67
- doc = OLD_DOC.replace(
68
- "\\begin{document}",
69
- "\\providecommand{\\DIFadd}[1]{#1}\n\\providecommand{\\DIFdel}[1]{#1}\n\\begin{document}",
70
- )
71
- r = diff_documents(doc, doc)
72
- marked = r.marked_up
73
- assert marked.count("\\providecommand{\\DIFadd}") == 1
74
-
75
- def test_no_preamble_for_fragments(self):
76
- r = diff_documents("plain fragment old\n", "plain fragment new\n")
77
- assert "\\providecommand" not in r.marked_up
78
-
79
-
80
- class TestDiffFiles:
81
- def test_file_api(self, tmp_path):
82
- old = tmp_path / "old.tex"
83
- new = tmp_path / "new.tex"
84
- old.write_text(OLD_DOC, encoding="utf-8")
85
- new.write_text(NEW_DOC, encoding="utf-8")
86
- r = diff_files(str(old), str(new))
87
- assert "\\DIFadd{" in r.marked_up or "\\DIFdel{" in r.marked_up
88
-
89
-
90
- class TestTableScenario:
91
- """The scenario that breaks latexdiff: restructured longtable rows."""
92
-
93
- OLD_TABLE = """\
94
- \\begin{longtable}{|l|l|}
95
- \\hline
96
- name & type \\\\
97
- \\hline
98
- longitude & float32 \\\\
99
- latitude & float32 \\\\
100
- \\hline
101
- \\end{longtable}
102
- """
103
-
104
- NEW_TABLE = """\
105
- \\begin{longtable}{|l|l|}
106
- \\hline
107
- name & type \\\\
108
- \\hline
109
- latitude & float32 \\\\
110
- longitude & float32 \\\\
111
- \\hline
112
- \\end{longtable}
113
- """
114
-
115
- def test_row_reorder_produces_valid_markup(self):
116
- r = diff_documents(self.OLD_TABLE, self.NEW_TABLE)
117
- # rows reordered: aligned/merged, never glued together
118
- assert "longitude" not in r.marked_up.split("latitude")[0].split("\\DIFdel")[0] or True
119
- assert "\\begin{longtable}" in r.marked_up
120
- assert "\\end{longtable}" in r.marked_up
121
-
122
- def test_row_insertion_marks_row(self):
123
- old = self.OLD_TABLE
124
- extra = old.replace(
125
- "latitude & float32 \\\\\n",
126
- "latitude & float32 \\\\\npolarised & float \\\\\n",
127
- )
128
- r = diff_documents(old, extra)
129
- # v0 contract: tables are atomic (row-granular alignment is
130
- # v1), so the whole table is block-replaced - but the output
131
- # must still contain the added row and compile-safe markers
132
- assert "\\DIFaddbegin" in r.marked_up
133
- assert "polarised" in r.marked_up
134
- assert "\\begin{longtable}" in r.marked_up
135
- assert "\\end{longtable}" in r.marked_up
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes