patchahead 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. patchahead/__init__.py +8 -0
  2. patchahead/analysis/__init__.py +52 -0
  3. patchahead/analysis/edits.py +143 -0
  4. patchahead/analysis/index.py +203 -0
  5. patchahead/analysis/python_ast.py +457 -0
  6. patchahead/apidiff/__init__.py +23 -0
  7. patchahead/apidiff/compare.py +366 -0
  8. patchahead/apidiff/download.py +95 -0
  9. patchahead/apidiff/surface.py +337 -0
  10. patchahead/ci.py +301 -0
  11. patchahead/cli.py +627 -0
  12. patchahead/config.py +284 -0
  13. patchahead/demo/__init__.py +256 -0
  14. patchahead/demo/fixtures/changes/field-rename.md +14 -0
  15. patchahead/demo/fixtures/changes/invoice-field-rename.md +21 -0
  16. patchahead/demo/fixtures/changes/kwarg-rename.md +14 -0
  17. patchahead/demo/fixtures/changes/method-rename.md +12 -0
  18. patchahead/demo/fixtures/changes/pagination-cursor.json +24 -0
  19. patchahead/demo/fixtures/changes/pagination-cursor.md +20 -0
  20. patchahead/demo/fixtures/changes/sdk-v2.md +31 -0
  21. patchahead/demo/fixtures/orders-service/README.md +51 -0
  22. patchahead/demo/fixtures/orders-service/app/__init__.py +0 -0
  23. patchahead/demo/fixtures/orders-service/app/client.py +15 -0
  24. patchahead/demo/fixtures/orders-service/app/models.py +10 -0
  25. patchahead/demo/fixtures/orders-service/app/order_report.py +24 -0
  26. patchahead/demo/fixtures/orders-service/app/order_sync.py +21 -0
  27. patchahead/demo/fixtures/orders-service/conftest.py +6 -0
  28. patchahead/demo/fixtures/orders-service/pyproject.toml +16 -0
  29. patchahead/demo/fixtures/orders-service/tests/test_client.py +14 -0
  30. patchahead/demo/fixtures/orders-service/tests/test_order_report.py +24 -0
  31. patchahead/demo/fixtures/orders-service/tests/test_order_sync.py +11 -0
  32. patchahead/demo/fixtures/orders-service/upstream/__init__.py +0 -0
  33. patchahead/demo/fixtures/orders-service/upstream/api_v1.py +34 -0
  34. patchahead/demo/fixtures/orders-service/upstream/api_v2.py +56 -0
  35. patchahead/demo/serve.py +189 -0
  36. patchahead/domain/__init__.py +67 -0
  37. patchahead/domain/change.py +269 -0
  38. patchahead/domain/completeness.py +91 -0
  39. patchahead/domain/impact.py +248 -0
  40. patchahead/domain/patch.py +81 -0
  41. patchahead/domain/plan.py +170 -0
  42. patchahead/domain/result.py +210 -0
  43. patchahead/domain/validation.py +200 -0
  44. patchahead/engine.py +609 -0
  45. patchahead/handlers/__init__.py +35 -0
  46. patchahead/handlers/base.py +211 -0
  47. patchahead/handlers/field_rename.py +425 -0
  48. patchahead/handlers/kwarg_rename.py +201 -0
  49. patchahead/handlers/method_rename.py +608 -0
  50. patchahead/handlers/pagination.py +582 -0
  51. patchahead/ingest/__init__.py +32 -0
  52. patchahead/ingest/base.py +102 -0
  53. patchahead/ingest/markdown.py +1138 -0
  54. patchahead/ingest/structured.py +218 -0
  55. patchahead/llm/__init__.py +28 -0
  56. patchahead/llm/client.py +152 -0
  57. patchahead/llm/proposer.py +620 -0
  58. patchahead/observability.py +223 -0
  59. patchahead/reporting.py +451 -0
  60. patchahead/testing/__init__.py +22 -0
  61. patchahead/testing/discovery.py +113 -0
  62. patchahead/testing/runner.py +138 -0
  63. patchahead/validation/__init__.py +5 -0
  64. patchahead/validation/completeness.py +265 -0
  65. patchahead/validation/engine.py +531 -0
  66. patchahead/web/__init__.py +13 -0
  67. patchahead/web/server.py +279 -0
  68. patchahead/web/static/index.html +650 -0
  69. patchahead/workspace.py +382 -0
  70. patchahead-0.3.0.dist-info/METADATA +368 -0
  71. patchahead-0.3.0.dist-info/RECORD +75 -0
  72. patchahead-0.3.0.dist-info/WHEEL +5 -0
  73. patchahead-0.3.0.dist-info/entry_points.txt +2 -0
  74. patchahead-0.3.0.dist-info/licenses/LICENSE +21 -0
  75. patchahead-0.3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1138 @@
1
+ """Parsing release notes written as Markdown or plain text.
2
+
3
+ This is the format most upstream providers actually publish, and the least
4
+ structured. Two design rules follow from that:
5
+
6
+ **Classification is scored, not first-match.** Each migration family has
7
+ weighted signal phrases; the highest total wins, and the winning score and the
8
+ phrases that produced it are recorded on
9
+ :attr:`~patchahead.domain.change.BreakingChange.classification_reason` so a user
10
+ can see why a document was read the way it was.
11
+
12
+ **Uncertainty is represented, not smoothed over.** A document that matches
13
+ nothing yields ``ChangeKind.UNKNOWN``; a recognized-but-unhandled change yields
14
+ ``ChangeKind.UNSUPPORTED``; a rename whose symbols could not be extracted keeps
15
+ ``Confidence.LOW``. The prototype instead defaulted unparseable documents to the
16
+ demo's pagination text, which made every failure look like a confident success
17
+ (``docs/assessment.md`` §2.5).
18
+
19
+ **A section can hold several changes.** Vendors list renames as bullets, table
20
+ rows, or two clauses of one sentence under a single heading. Each statement is
21
+ read on its own, so a table of four renamed methods yields four changes, and a
22
+ bullet about an endpoint move is reported as unsupported rather than dropped.
23
+ A section that states one rename is still read as a whole, which is what lets a
24
+ Before/After example under it supply an owner.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import html
30
+ import logging
31
+ import re
32
+ from dataclasses import dataclass
33
+
34
+ from patchahead.domain.change import (
35
+ BreakingChange,
36
+ ChangeKind,
37
+ Confidence,
38
+ Evidence,
39
+ PaginationContract,
40
+ Severity,
41
+ SymbolTarget,
42
+ )
43
+ from patchahead.ingest.base import ChangeDocument, ChangeParser, register
44
+
45
+ log = logging.getLogger(__name__)
46
+
47
+ #: Signal phrases per change kind, with weights. Phrases are matched as regular
48
+ #: expressions against the lowercased section text. Weights are relative: a
49
+ #: phrase that names the construct ("keyword argument") outranks one that merely
50
+ #: suggests it ("renamed").
51
+ _SIGNALS: dict[ChangeKind, list[tuple[str, int]]] = {
52
+ ChangeKind.KWARG_RENAME: [
53
+ (r"keyword argument", 5),
54
+ (r"\bkwarg\b", 5),
55
+ (r"\bparameter\b.{0,40}\brenamed\b", 4),
56
+ (r"\brenamed\b.{0,40}\bparameter\b", 4),
57
+ (r"\bargument\b.{0,40}\brenamed\b", 4),
58
+ (r"\brenamed\b.{0,40}\bargument\b", 4),
59
+ (r"`\w+=`", 3),
60
+ ],
61
+ ChangeKind.METHOD_RENAME: [
62
+ (r"method (?:was )?renamed", 5),
63
+ (r"renamed method", 5),
64
+ (r"function (?:was )?renamed", 5),
65
+ (r"renamed function", 5),
66
+ (r"method renamed", 5),
67
+ (r"deprecated method", 3),
68
+ (r"`\w+\(\)`.{0,30}(?:->|→|renamed to)", 3),
69
+ (r"\bmethod\b", 1),
70
+ ],
71
+ ChangeKind.FIELD_RENAME: [
72
+ (r"field (?:was )?renamed", 5),
73
+ (r"renamed field", 5),
74
+ (r"field renamed", 5),
75
+ (r"attribute (?:was )?renamed", 4),
76
+ (r"property (?:was )?renamed", 4),
77
+ (r"\bfield\b.{0,40}(?:->|→)", 3),
78
+ (r"\bfield\b", 1),
79
+ ],
80
+ ChangeKind.PAGINATION_PAGE_TO_CURSOR: [
81
+ (r"cursor-based", 5),
82
+ (r"\bnext_cursor\b", 4),
83
+ (r"\btotal_pages\b", 4),
84
+ (r"page-based", 4),
85
+ (r"\bhas_more\b", 3),
86
+ (r"\bpagination\b", 3),
87
+ (r"\bcursor\b", 2),
88
+ ],
89
+ }
90
+
91
+ #: Changes PatchAhead can recognize but has no v1 handler for. Detecting them
92
+ #: explicitly lets the CLI say "this is an endpoint move, which v1 cannot
93
+ #: migrate" instead of misclassifying it as something it can.
94
+ _UNSUPPORTED_SIGNALS: list[tuple[str, int, str]] = [
95
+ (r"endpoint (?:was )?(?:moved|changed|removed)", 5, "endpoint change"),
96
+ (r"moved to a (?:new|different) endpoint", 5, "endpoint change"),
97
+ (r"response (?:shape|format|schema) (?:has )?changed", 5, "response shape change"),
98
+ (r"schema (?:was )?changed", 4, "response shape change"),
99
+ (r"authentication (?:has )?changed", 4, "authentication change"),
100
+ (r"\brate limit(?:ing)? (?:has )?changed", 4, "rate limiting change"),
101
+ (r"now returns? an? (?:list|array|object) instead", 4, "response shape change"),
102
+ (r"replaced by an? `[^`]+` (?:dict|dictionary|object|class|mapping)", 5, "restructuring"),
103
+ ]
104
+
105
+ #: Minimum score before a classification is trusted at all.
106
+ _MIN_SCORE = 3
107
+ #: Score at or above which a classification is considered confident.
108
+ _STRONG_SCORE = 5
109
+
110
+ _HEADING = re.compile(r"^(#{2,4})\s+(.+?)\s*#*$", re.MULTILINE)
111
+ _NON_BREAKING_HEADING = re.compile(
112
+ r"^#{1,4}\s+(?:non[- ]breaking|additions?|added|improvements?|enhancements?|"
113
+ r"fixes|fixed|bug ?fixes|(?:new )?features?)\b",
114
+ re.IGNORECASE,
115
+ )
116
+
117
+ #: Nouns a release note puts between a renamed name and the word "to": "renamed
118
+ #: the `retries` keyword argument to `max_retries`". Spelled out rather than
119
+ #: allowed as "any two words", because `\w+` would also match "renamed the
120
+ #: `client` argument passed to `fetch_orders`" -- which names two symbols and
121
+ #: renames neither of them.
122
+ _RENAME_NOUN = (
123
+ r"(?:keyword|kwarg|argument|arg|parameter|param|field|attribute|property"
124
+ r"|option|setting|method|function)s?"
125
+ )
126
+
127
+
128
+ def _name(group: str) -> str:
129
+ """A backticked name, recording whether it was written as a call or a keyword."""
130
+ return rf"`(?P<{group}>\.?[A-Za-z_][\w.]*)(?P<{group}_call>\(\))?(?P<{group}_kw>=)?`"
131
+
132
+
133
+ _OLD, _NEW = _name("old"), _name("new")
134
+ #: "of `get()` and `post()`", "on the `Charge` object": a clause naming what the
135
+ #: renamed thing belongs to, which may sit between the name and the verb.
136
+ _QUALIFIER = (
137
+ r"(?:\s+(?:of|on|for|in)\s+(?:the\s+|each\s+|all\s+)?"
138
+ r"(?:`[^`\n]+`(?:\s*(?:,|and|or)\s*`[^`\n]+`)*|[A-Z]\w*)(?:\s+[a-z]+)?)?"
139
+ )
140
+ _VERB = (
141
+ r"(?:(?:(?:was|is|are|were|has\s+been|have\s+been)\s+)?(?:renamed|changed)\s+to"
142
+ r"|(?:was|is|are|were|has\s+been|have\s+been)\s+replaced\s+(?:by|with)"
143
+ r"|(?:is|are)\s+now(?:\s+(?:called|named|spelled|returned\s+as|exposed\s+as))?"
144
+ r"|(?:is|was|has\s+been)\s+deprecated\s+in\s+favou?r\s+of)"
145
+ )
146
+ #: How each notation states a rename, most trustworthy first. ``\`a\` ->
147
+ #: \`b\``` is unambiguous; a subject-verb sentence can pick up the wrong name
148
+ #: from an intervening clause, which is why the qualifier is matched explicitly.
149
+ _RENAME_PATTERNS: tuple[re.Pattern[str], ...] = (
150
+ re.compile(rf"{_OLD}\s*(?:->|→|=>)\s*{_NEW}"),
151
+ re.compile(
152
+ rf"renamed\s+(?:from\s+)?(?:the\s+|an?\s+)?{_OLD}(?:\s+{_RENAME_NOUN}){{0,2}}"
153
+ rf"\s+to\s+{_NEW}",
154
+ re.IGNORECASE,
155
+ ),
156
+ re.compile(rf"{_OLD}(?:\s+[A-Za-z]+){{0,3}}?{_QUALIFIER}\s+{_VERB}\s+{_NEW}", re.IGNORECASE),
157
+ re.compile(rf"\b(?:use|call)\s+{_NEW}\s+instead\s+of\s+{_OLD}", re.IGNORECASE),
158
+ re.compile(rf"\breplace\s+{_OLD}\s+with\s+{_NEW}", re.IGNORECASE),
159
+ )
160
+ #: "Renamed `a()` to `b()` and `c()` to `d()`": the second pair, only read in a
161
+ #: statement that already says "renamed".
162
+ _AND_TO = re.compile(rf"(?:,|\band)\s+{_OLD}\s+to\s+{_NEW}")
163
+ #: Words that look like a new name after "is now" but describe a type or a state.
164
+ _NOT_A_NAME = {
165
+ "int",
166
+ "float",
167
+ "str",
168
+ "bool",
169
+ "list",
170
+ "dict",
171
+ "tuple",
172
+ "set",
173
+ "bytes",
174
+ "none",
175
+ "null",
176
+ "true",
177
+ "false",
178
+ "optional",
179
+ "required",
180
+ "datetime",
181
+ "date",
182
+ "decimal",
183
+ "any",
184
+ "object",
185
+ }
186
+ #: The construct a noun names, for deciding what kind of rename a statement is.
187
+ _NOUN_KINDS: tuple[tuple[re.Pattern[str], ChangeKind], ...] = (
188
+ (
189
+ re.compile(
190
+ r"\b(?:keyword\s+arguments?|kwargs?|arguments?|args|parameters?|params?)\b", re.I
191
+ ),
192
+ ChangeKind.KWARG_RENAME,
193
+ ),
194
+ (re.compile(r"\b(?:methods?|functions?)\b", re.I), ChangeKind.METHOD_RENAME),
195
+ (
196
+ re.compile(r"\b(?:fields?|propert(?:y|ies)|attributes?|keys?)\b", re.I),
197
+ ChangeKind.FIELD_RENAME,
198
+ ),
199
+ )
200
+ _BOLD_FIELD = re.compile(r"\*\*(?P<label>[A-Za-z][\w /-]*?)\s*:?\*\*[:\s]*(?P<value>.+)")
201
+ _RISK = re.compile(r"risk:?\s*(high|medium|low)", re.IGNORECASE)
202
+
203
+ _STOPWORDS = {
204
+ "the",
205
+ "a",
206
+ "an",
207
+ "is",
208
+ "was",
209
+ "were",
210
+ "to",
211
+ "from",
212
+ "in",
213
+ "on",
214
+ "of",
215
+ "and",
216
+ "or",
217
+ "now",
218
+ "new",
219
+ "old",
220
+ "this",
221
+ "that",
222
+ "it",
223
+ "be",
224
+ "been",
225
+ "before",
226
+ "after",
227
+ "migration",
228
+ "risk",
229
+ "breaking",
230
+ "change",
231
+ "changes",
232
+ }
233
+
234
+
235
+ @dataclass
236
+ class _Section:
237
+ """One ``###``-delimited chunk of a document."""
238
+
239
+ title: str
240
+ text: str
241
+ start_line: int
242
+
243
+
244
+ _HTML_BLOCK = re.compile(r"<(?:h[1-6]|li|ul|ol|p|code|blockquote|details)\b", re.IGNORECASE)
245
+ _HTML_HEADING = re.compile(r"<h([1-6])[^>]*>(.*?)</h\1\s*>", re.IGNORECASE | re.DOTALL)
246
+ _COMMITS_BLOCK = re.compile(
247
+ r"<details>\s*<summary>\s*Commits\s*</summary>.*?</details>", re.IGNORECASE | re.DOTALL
248
+ )
249
+
250
+
251
+ def _from_html(text: str) -> str:
252
+ """Rewrite HTML release notes -- what Dependabot puts in a pull request -- as Markdown.
253
+
254
+ Headings become ``##``/``###``, list items become bullets, ``<code>``
255
+ becomes backticks, and every other tag is dropped. A Dependabot "Commits"
256
+ list is removed first: it is commit subjects, not release notes.
257
+ """
258
+ text = _COMMITS_BLOCK.sub("", text)
259
+ text = _HTML_HEADING.sub(
260
+ lambda m: f"\n{'#' * min(max(int(m.group(1)), 2), 4)} {m.group(2).strip()}\n", text
261
+ )
262
+ text = re.sub(r"<code>(.*?)</code>", r"`\1`", text, flags=re.IGNORECASE | re.DOTALL)
263
+ text = re.sub(r"<li[^>]*>", "\n- ", text, flags=re.IGNORECASE)
264
+ text = re.sub(r"<br\s*/?>|</p\s*>|</li\s*>|</?[uo]l[^>]*>", "\n", text, flags=re.IGNORECASE)
265
+ text = re.sub(r"<[^>]+>", "", text)
266
+ text = html.unescape(text)
267
+ return re.sub(r"\n{3,}", "\n\n", text)
268
+
269
+
270
+ _UNDERLINE = re.compile(r"^([=\-~^\"'*+#])\1{2,}\s*$")
271
+
272
+
273
+ #: Sphinx cross-references to code: :meth:`.assertTrue`, :class:`~unittest.TestCase`,
274
+ #: :func:`Title <pkg.func>`. Other roles (:gh:`123`, :ref:`...`) are left as text.
275
+ _ROLE = re.compile(
276
+ r":(?:py:)?(?P<role>meth|func|attr|class|mod|data|exc|obj|const):"
277
+ r"`(?P<title>[^`<]*?)(?:\s*<(?P<target>[^>`]+)>)?`"
278
+ )
279
+ #: A reStructuredText simple-table border: columns of ``=`` separated by spaces.
280
+ _RST_BORDER = re.compile(r"^(?P<indent>\s*)=+(?: +=+)+\s*$")
281
+
282
+
283
+ def _sphinx_role(match: re.Match[str]) -> str:
284
+ """:meth:`.assertTrue` -> `assertTrue()`; :class:`~unittest.TestCase` -> `TestCase`."""
285
+ name = (match.group("target") or match.group("title")).strip().lstrip("!")
286
+ if name.startswith("~"):
287
+ name = name[1:].rsplit(".", 1)[-1]
288
+ name = name.lstrip(".").removesuffix("()")
289
+ return f"`{name}()`" if match.group("role") in ("meth", "func") else f"`{name}`"
290
+
291
+
292
+ def _rst_tables(lines: list[str]) -> list[str]:
293
+ """Rewrite reStructuredText simple tables as Markdown pipe tables, in place.
294
+
295
+ Columns are where the border's runs of ``=`` are, so cells are cut by
296
+ position before any markup inside them changes width. Borders become blank
297
+ lines, keeping line numbers.
298
+ """
299
+ index = 0
300
+ while index < len(lines):
301
+ top = _RST_BORDER.match(lines[index])
302
+ if not top:
303
+ index += 1
304
+ continue
305
+ borders = [i for i in range(index, len(lines)) if _RST_BORDER.match(lines[i])][:3]
306
+ if len(borders) < 3:
307
+ break
308
+ spans = [m.start() for m in re.finditer(r"=+", lines[index])]
309
+
310
+ def cells(line: str, spans: list[int] = spans) -> str:
311
+ ends = spans[1:] + [len(line)]
312
+ return (
313
+ "| "
314
+ + " | ".join(line[a:b].strip() for a, b in zip(spans, ends, strict=True))
315
+ + " |"
316
+ )
317
+
318
+ indent = top.group("indent")
319
+ first, rule, last = borders
320
+ for row in range(first + 1, last):
321
+ if row == rule:
322
+ lines[row] = indent + "|" + "|".join("---" for _ in spans) + "|"
323
+ elif lines[row].strip():
324
+ lines[row] = indent + cells(lines[row])
325
+ lines[first] = lines[last] = ""
326
+ index = last + 1
327
+ return lines
328
+
329
+
330
+ def _normalize(text: str) -> str:
331
+ """Rewrite HTML, reStructuredText, and Sphinx markup as plain Markdown.
332
+
333
+ Simple tables become pipe tables, Sphinx cross-references to code become
334
+ backticked names (with ``()`` for a method or function, which is how the
335
+ parser tells a call from a field), a double-backtick literal becomes a
336
+ single-backtick one, and a title underlined with ``====`` or ``----``
337
+ becomes a ``##`` heading -- levels assigned in order of first appearance,
338
+ as reStructuredText does. Underlines and table borders become blank lines,
339
+ so line numbers in evidence still point at the original document.
340
+ """
341
+ if _HTML_BLOCK.search(text):
342
+ text = _from_html(text)
343
+ lines = _rst_tables(text.splitlines())
344
+ text = "\n".join(lines) + ("\n" if text.endswith("\n") else "")
345
+ text = _ROLE.sub(_sphinx_role, text)
346
+ text = re.sub(r"``([^`\n]+)``", r"`\1`", text)
347
+ lines = text.splitlines()
348
+ levels: dict[str, int] = {}
349
+ for index in range(1, len(lines)):
350
+ match = _UNDERLINE.match(lines[index])
351
+ title = lines[index - 1].strip()
352
+ if not match or not title or _UNDERLINE.match(title) or title.startswith(("-", "*", "|")):
353
+ continue
354
+ level = levels.setdefault(match.group(1), min(2 + len(levels), 4))
355
+ lines[index - 1] = f"{'#' * level} {title}"
356
+ lines[index] = ""
357
+ return "\n".join(lines) + ("\n" if text.endswith("\n") else "")
358
+
359
+
360
+ def _split_sections(document: ChangeDocument) -> list[_Section]:
361
+ """Split a document at its headings, dropping non-breaking-change sections.
362
+
363
+ A release note usually describes several changes. Treating the whole file as
364
+ one blob lets a "Non-breaking changes" bullet contribute signal words to a
365
+ breaking change's classification.
366
+ """
367
+ text = _normalize(document.text)
368
+ matches = list(_HEADING.finditer(text))
369
+ if not matches:
370
+ return [_Section(title=_infer_title(text), text=text, start_line=1)]
371
+
372
+ sections: list[_Section] = []
373
+ suppressing = False
374
+ for position, match in enumerate(matches):
375
+ heading_line = text[: match.start()].count("\n") + 1
376
+ heading_text = text[match.start() : match.end()]
377
+ level = len(match.group(1))
378
+ body_end = matches[position + 1].start() if position + 1 < len(matches) else len(text)
379
+ body = text[match.end() : body_end]
380
+
381
+ if _NON_BREAKING_HEADING.match(heading_text.strip()):
382
+ # Suppress this heading and any sub-headings beneath it.
383
+ suppressing = True
384
+ suppress_level = level
385
+ continue
386
+ if suppressing and level > suppress_level:
387
+ continue
388
+ suppressing = False
389
+
390
+ title = match.group(2).strip()
391
+ if _looks_like_container_heading(title, body):
392
+ continue
393
+ sections.append(_Section(title=title, text=heading_text + body, start_line=heading_line))
394
+
395
+ if not sections:
396
+ return [_Section(title=_infer_title(text), text=text, start_line=1)]
397
+ return sections
398
+
399
+
400
+ def _looks_like_container_heading(title: str, body: str) -> bool:
401
+ """Whether a heading is a wrapper (``## Breaking changes``) with no content."""
402
+ if not re.match(r"(?i)^(breaking changes?|changes?|release notes?)$", title.strip()):
403
+ return False
404
+ return not body.strip() or bool(_HEADING.search(body))
405
+
406
+
407
+ def _infer_title(text: str) -> str:
408
+ for line in text.splitlines():
409
+ stripped = line.strip().lstrip("#").strip()
410
+ if stripped:
411
+ return stripped[:120]
412
+ return "Upstream breaking change"
413
+
414
+
415
+ def classify(text: str) -> tuple[ChangeKind, int, str]:
416
+ """Classify a block of release-note text.
417
+
418
+ Returns the kind, the winning score, and a human-readable reason naming the
419
+ phrases that decided it.
420
+ """
421
+ low = text.lower()
422
+
423
+ scores: dict[ChangeKind, tuple[int, list[str]]] = {}
424
+ for kind, signals in _SIGNALS.items():
425
+ total = 0
426
+ hits: list[str] = []
427
+ for pattern, weight in signals:
428
+ if re.search(pattern, low):
429
+ total += weight
430
+ hits.append(pattern)
431
+ if total:
432
+ scores[kind] = (total, hits)
433
+
434
+ unsupported_score = 0
435
+ unsupported_label = ""
436
+ for pattern, weight, label in _UNSUPPORTED_SIGNALS:
437
+ if weight > unsupported_score and re.search(pattern, low):
438
+ unsupported_score, unsupported_label = weight, label
439
+
440
+ if not scores and not unsupported_score:
441
+ return ChangeKind.UNKNOWN, 0, "no recognized breaking-change signals in the document"
442
+
443
+ best_kind, (best_score, best_hits) = (
444
+ max(scores.items(), key=lambda item: item[1][0])
445
+ if scores
446
+ else (ChangeKind.UNKNOWN, (0, []))
447
+ )
448
+
449
+ if unsupported_score > best_score:
450
+ return (
451
+ ChangeKind.UNSUPPORTED,
452
+ unsupported_score,
453
+ f"looks like a {unsupported_label}, which PatchAhead v1 cannot migrate",
454
+ )
455
+ if best_score < _MIN_SCORE:
456
+ return (
457
+ ChangeKind.UNKNOWN,
458
+ best_score,
459
+ f"weak signals only (score {best_score}, need {_MIN_SCORE}); "
460
+ f"closest match was {best_kind.value}",
461
+ )
462
+
463
+ phrases = ", ".join(sorted(hit.replace("\\b", "") for hit in best_hits)[:4])
464
+ return best_kind, best_score, f"matched {best_kind.value} signals: {phrases}"
465
+
466
+
467
+ @dataclass(frozen=True)
468
+ class _Rename:
469
+ """One rename statement: ``old`` -> ``new`` as written, and where it was found."""
470
+
471
+ old: str
472
+ new: str
473
+ #: How the names were written: a call ``x()`` or a keyword ``x=``.
474
+ call: bool
475
+ keyword: bool
476
+ #: Notation rank (lower is more trustworthy), then position in the text.
477
+ priority: int
478
+ start: int
479
+ end: int
480
+
481
+ @property
482
+ def symbol(self) -> str:
483
+ return self.old.rsplit(".", 1)[-1]
484
+
485
+ @property
486
+ def replacement(self) -> str:
487
+ return self.new.rsplit(".", 1)[-1]
488
+
489
+ @property
490
+ def qualifier(self) -> str:
491
+ """The receiver prefix of a dotted old name -- an *asserted* owner."""
492
+ return self.old.rsplit(".", 1)[0].lstrip(".") if "." in self.old.lstrip(".") else ""
493
+
494
+ @property
495
+ def new_qualifier(self) -> str:
496
+ return self.new.rsplit(".", 1)[0].lstrip(".") if "." in self.new.lstrip(".") else ""
497
+
498
+ @property
499
+ def is_move(self) -> bool:
500
+ """Whether the thing changes where it lives, not just what it is called.
501
+
502
+ ``a.X.create`` -> ``b.y.create`` keeps the name. ``imp.find_module()``
503
+ -> ``importlib.util.find_spec()`` changes both: renaming the call in
504
+ place would write ``imp.find_spec()``, which does not exist. A new name
505
+ under a different owner is a move; ``Client.fetch_all`` ->
506
+ ``Client.list_all``, or ``Client.fetch_all`` -> ``list_all``, is not.
507
+ """
508
+ if self.symbol == self.replacement:
509
+ return True
510
+ new_home = self.new_qualifier.lower()
511
+ return bool(new_home) and new_home != self.qualifier.lower()
512
+
513
+
514
+ def _find_renames(text: str) -> list[_Rename]:
515
+ """Every rename a piece of text states, in document order.
516
+
517
+ Several notations can match the same words, and they are not equally
518
+ trustworthy: in "The `timeout_seconds` keyword argument on `fetch_orders`
519
+ was renamed to `timeout`", a bare "`x` was renamed to `y`" reading would
520
+ pick ``fetch_orders`` -> ``timeout``. Matches are therefore taken best
521
+ notation first, and a match overlapping one already taken is discarded.
522
+ """
523
+ candidates: list[_Rename] = []
524
+ for priority, pattern in enumerate(_RENAME_PATTERNS):
525
+ candidates.extend(_rename(match, priority) for match in pattern.finditer(text))
526
+ if re.search(r"\brenamed\b", text, re.IGNORECASE):
527
+ candidates.extend(_rename(match, len(_RENAME_PATTERNS)) for match in _AND_TO.finditer(text))
528
+
529
+ taken: list[_Rename] = []
530
+ for candidate in sorted(candidates, key=lambda r: (r.priority, r.start)):
531
+ if not _plausible(candidate):
532
+ continue
533
+ if any(candidate.start < r.end and r.start < candidate.end for r in taken):
534
+ continue
535
+ taken.append(candidate)
536
+ return sorted(taken, key=lambda r: r.start)
537
+
538
+
539
+ def _rename(match: re.Match[str], priority: int) -> _Rename:
540
+ return _Rename(
541
+ old=match.group("old"),
542
+ new=match.group("new"),
543
+ call=bool(match.group("old_call") or match.group("new_call")),
544
+ keyword=bool(match.group("old_kw") or match.group("new_kw")),
545
+ priority=priority,
546
+ start=match.start(),
547
+ end=match.end(),
548
+ )
549
+
550
+
551
+ def _plausible(rename: _Rename) -> bool:
552
+ old, new = rename.symbol.lower(), rename.replacement.lower()
553
+ if not old or not new or rename.old == rename.new:
554
+ return False
555
+ if old in _STOPWORDS or new in _STOPWORDS:
556
+ return False
557
+ # "`timeout` is now `float`" changes a type, not a name. A type is written
558
+ # bare; `list()` or `dict=` is a call or a keyword, and a real new name.
559
+ return rename.call or rename.keyword or new not in _NOT_A_NAME
560
+
561
+
562
+ @dataclass
563
+ class _Statement:
564
+ """One bullet, table row, paragraph or heading inside a section."""
565
+
566
+ text: str
567
+ line: int
568
+
569
+
570
+ _BULLET = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+")
571
+ _TABLE_RULE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)*\|?\s*$")
572
+ _OLD_COLUMN = re.compile(r"(?i)\b(?:old|before|previous(?:ly)?|from|deprecated|removed)\b")
573
+ _NEW_COLUMN = re.compile(r"(?i)\b(?:new|after|replacement|now|to|use)\b")
574
+
575
+
576
+ def _cells(row: str) -> list[str]:
577
+ return [cell.strip() for cell in row.strip().strip("|").split("|")]
578
+
579
+
580
+ def _statements(section: _Section) -> tuple[list[_Statement], str]:
581
+ """Split a section into statements, plus the prose that frames them.
582
+
583
+ A table row becomes ``old → new`` followed by its remaining cells, with the
584
+ columns chosen by their headers ("Old"/"New", "Before"/"After", "v1"/"v2").
585
+ The returned context -- the heading and any prose that is not a bullet or a
586
+ row -- is what each statement is read against for the nouns and owners a
587
+ row or terse bullet leaves out.
588
+ """
589
+ lines = section.text.splitlines()
590
+ statements: list[_Statement] = []
591
+ context: list[str] = []
592
+ in_fence = False
593
+ index = 0
594
+ while index < len(lines):
595
+ line = lines[index]
596
+ number = section.start_line + index
597
+ if line.strip().startswith("```"):
598
+ in_fence = not in_fence
599
+ index += 1
600
+ continue
601
+ if in_fence or not line.strip():
602
+ index += 1
603
+ continue
604
+
605
+ if (
606
+ line.lstrip().startswith("|")
607
+ and index + 1 < len(lines)
608
+ and _TABLE_RULE.match(lines[index + 1])
609
+ ):
610
+ header = _cells(line)
611
+ old_col = next((i for i, c in enumerate(header) if _OLD_COLUMN.search(c)), 0)
612
+ new_col = next(
613
+ (i for i, c in enumerate(header) if i != old_col and _NEW_COLUMN.search(c)),
614
+ 1 if old_col == 0 else 0,
615
+ )
616
+ index += 2
617
+ while index < len(lines) and lines[index].lstrip().startswith("|"):
618
+ cells = _cells(lines[index])
619
+ if max(old_col, new_col) < len(cells):
620
+ rest = [c for i, c in enumerate(cells) if i not in (old_col, new_col)]
621
+ text = f"{cells[old_col]} → {cells[new_col]} {' '.join(rest)}".strip()
622
+ statements.append(_Statement(text, section.start_line + index))
623
+ index += 1
624
+ continue
625
+
626
+ if _BULLET.match(line):
627
+ parts = [_BULLET.sub("", line, count=1)]
628
+ index += 1
629
+ while (
630
+ index < len(lines)
631
+ and lines[index].strip()
632
+ and not _BULLET.match(lines[index])
633
+ and lines[index].startswith((" ", "\t"))
634
+ ):
635
+ parts.append(lines[index].strip())
636
+ index += 1
637
+ statements.append(_Statement(" ".join(parts), number))
638
+ continue
639
+
640
+ # A heading or a prose paragraph: a statement in its own right, and
641
+ # the frame the bullets and rows are read against.
642
+ parts = [line.strip().lstrip("#").strip()]
643
+ index += 1
644
+ while (
645
+ not line.lstrip().startswith("#")
646
+ and index < len(lines)
647
+ and lines[index].strip()
648
+ and not _BULLET.match(lines[index])
649
+ and not lines[index].lstrip().startswith(("|", "#", "```"))
650
+ ):
651
+ parts.append(lines[index].strip())
652
+ index += 1
653
+ paragraph = " ".join(parts)
654
+ statements.append(_Statement(paragraph, number))
655
+ context.append(paragraph)
656
+ return statements, "\n".join(context)
657
+
658
+
659
+ def _noun_kind(text: str, near: int | None = None) -> ChangeKind | None:
660
+ """The construct a statement's nouns name; the noun nearest ``near`` wins."""
661
+ best: tuple[int, ChangeKind] | None = None
662
+ for pattern, kind in _NOUN_KINDS:
663
+ for match in pattern.finditer(text):
664
+ distance = abs(match.start() - near) if near is not None else 0
665
+ if best is None or distance < best[0]:
666
+ best = (distance, kind)
667
+ if near is None and best is not None:
668
+ kinds = {kind for pattern, kind in _NOUN_KINDS if pattern.search(text)}
669
+ return best[1] if len(kinds) == 1 else None
670
+ return best[1] if best else None
671
+
672
+
673
+ def _rename_kind(rename: _Rename, statement: str, context: str) -> tuple[ChangeKind, int, str]:
674
+ """Decide what kind of rename one statement states.
675
+
676
+ The way the names are written is the strongest evidence (``x()`` is called,
677
+ ``x=`` is passed); then a noun beside the old name ("the `x` keyword
678
+ argument"); then the scored signals of the statement; then the nouns of the
679
+ heading and prose around it.
680
+ """
681
+ if rename.keyword:
682
+ return ChangeKind.KWARG_RENAME, _STRONG_SCORE, "written as a keyword argument (`name=`)"
683
+ if rename.call:
684
+ return ChangeKind.METHOD_RENAME, _STRONG_SCORE, "written as a call (`name()`)"
685
+ noun = _noun_kind(statement, near=rename.start)
686
+ if noun is not None:
687
+ return noun, _STRONG_SCORE, f"named as a {noun.value.split('_')[0]} in the statement"
688
+ kind, score, reason = classify(statement)
689
+ if kind.is_actionable and kind is not ChangeKind.PAGINATION_PAGE_TO_CURSOR:
690
+ return kind, score, reason
691
+ shape = _example_kind(rename.symbol, context)
692
+ if shape is not None:
693
+ return shape, _STRONG_SCORE, f"used as a {shape.value.split('_')[0]} in the examples"
694
+ noun = _noun_kind(context)
695
+ if noun is not None:
696
+ return noun, _STRONG_SCORE, f"named as a {noun.value.split('_')[0]} in the section"
697
+ return ChangeKind.UNKNOWN, 0, "a rename, but nothing says of what"
698
+
699
+
700
+ def _example_kind(symbol: str, text: str) -> ChangeKind | None:
701
+ """How the old name is used in the document's code examples, if only one way.
702
+
703
+ ``client.fetch_all(limit=10)`` shows a call; ``fetch(timeout_seconds=5)`` a
704
+ keyword; ``order["total"]`` a field. Two different uses decide nothing.
705
+ """
706
+ escaped = re.escape(symbol)
707
+ uses = {
708
+ kind
709
+ for kind, pattern in (
710
+ (ChangeKind.METHOD_RENAME, rf"(?<![\w\[\"'])\.?{escaped}\s*\("),
711
+ (ChangeKind.KWARG_RENAME, rf"[(,]\s*{escaped}\s*=(?!=)"),
712
+ (ChangeKind.FIELD_RENAME, rf"\[\s*[\"']{escaped}[\"']\s*\]"),
713
+ )
714
+ if re.search(pattern, text)
715
+ }
716
+ return uses.pop() if len(uses) == 1 else None
717
+
718
+
719
+ def _unsupported_label(text: str) -> str:
720
+ low = text.lower()
721
+ for pattern, _weight, label in _UNSUPPORTED_SIGNALS:
722
+ if re.search(pattern, low):
723
+ return label
724
+ return ""
725
+
726
+
727
+ def _extract_owner(text: str, symbol: str, kind: ChangeKind) -> tuple[str, bool]:
728
+ """Find the object or function the symbol belongs to.
729
+
730
+ The owner is what keeps a rename scoped. Without it a field rename of
731
+ ``total`` is a repository-wide search for the word "total"; with it, the
732
+ search is for ``order["total"]``.
733
+
734
+ Returns ``(owner, is_explicit)``. Phrasings that *assert* ownership -- "on
735
+ each `order` object", "the keyword argument on `fetch_orders`" -- are
736
+ explicit, and a receiver mismatch then refuses to patch. A receiver merely
737
+ scraped out of an illustrative snippet is not: the vendor writing
738
+ ``client.fetch_orders(limit=10)`` is naming their own example variable, not
739
+ making a claim about what a downstream repository calls its client.
740
+ """
741
+ if not symbol:
742
+ return "", False
743
+ escaped = re.escape(symbol)
744
+
745
+ if kind is ChangeKind.KWARG_RENAME:
746
+ # Which function an argument belongs to is definitional rather than a
747
+ # naming coincidence, so both spellings count as assertions.
748
+ match = re.search(rf"(\w+)\s*\([^)]*\b{escaped}\s*=", text)
749
+ if match:
750
+ return match.group(1), True
751
+ # "the `timeout_seconds` keyword argument on `fetch_orders`"
752
+ # "... of `Client.get()`" -- one function only; "of `get()` and `post()`"
753
+ # names two, and neither alone is the owner.
754
+ match = re.search(
755
+ rf"`{escaped}=?`[^`\n]{{0,60}}?\b(?:on|of|for)\s+(?:the\s+)?"
756
+ r"`(?:[\w.]*\.)?(\w+)(?:\(\))?`(?!\s*(?:,|and|or)\s*`)",
757
+ text,
758
+ re.IGNORECASE,
759
+ )
760
+ if match:
761
+ return match.group(1), True
762
+ return "", False
763
+
764
+ if kind is ChangeKind.METHOD_RENAME:
765
+ # "the method on `client` was renamed" asserts a receiver.
766
+ match = re.search(
767
+ rf"\bon\s+(?:the\s+)?`(\w+)`[^`\n]{{0,60}}?`?{escaped}`?", text, re.IGNORECASE
768
+ )
769
+ if match and match.group(1).lower() not in _STOPWORDS:
770
+ return match.group(1), True
771
+ match = re.search(
772
+ rf"`{escaped}(?:\(\))?`[^\n]{{0,60}}?\bon\s+(?:the\s+)?`(\w+)`", text, re.IGNORECASE
773
+ )
774
+ if match and match.group(1).lower() not in _STOPWORDS:
775
+ return match.group(1), True
776
+ # `client.fetch_orders(...)` in a Before/After example: a hint, not a
777
+ # constraint. See the docstring.
778
+ match = re.search(rf"`?(\w+)\.{escaped}\s*\(", text)
779
+ if match and match.group(1).lower() not in _STOPWORDS:
780
+ return match.group(1), False
781
+ return "", False
782
+
783
+ # Field rename. "on each `order` object" asserts which object owns the
784
+ # field; a bare `order["total"]` snippet only illustrates it.
785
+ for pattern in (
786
+ r"\bon\s+(?:(?:each|the|an|a|all)\s+)?`(\w+)`\s+objects?",
787
+ # An API reference's capitalized object name, written without backticks.
788
+ r"\bon\s+(?:(?:each|the|an|a|all)\s+)?([A-Z]\w*)\s+objects?",
789
+ rf"`(\w+)`\s+object[^.\n]{{0,60}}?`{escaped}`",
790
+ ):
791
+ match = re.search(pattern, text, re.IGNORECASE)
792
+ if match and match.group(1).lower() not in _STOPWORDS:
793
+ return match.group(1), True
794
+ for pattern in (
795
+ rf"(\w+)\s*\[\s*[\"']{escaped}[\"']\s*\]",
796
+ rf"`(\w+)\.{escaped}`",
797
+ ):
798
+ match = re.search(pattern, text, re.IGNORECASE)
799
+ if match and match.group(1).lower() not in _STOPWORDS:
800
+ return match.group(1), False
801
+ return "", False
802
+
803
+
804
+ def _extract_labeled(text: str, *labels: str) -> str:
805
+ """Pull the value of a ``**Label:** value`` bullet."""
806
+ for match in _BOLD_FIELD.finditer(text):
807
+ label = match.group("label").strip().lower()
808
+ for wanted in labels:
809
+ if label.startswith(wanted.lower()):
810
+ return match.group("value").strip().rstrip(".")
811
+ return ""
812
+
813
+
814
+ def _collect_evidence(section: _Section, kind: ChangeKind, symbol: str) -> list[Evidence]:
815
+ """Quote the lines that justify the classification, with line numbers."""
816
+ interesting = [
817
+ "renamed",
818
+ "removed",
819
+ "migration",
820
+ "breaking",
821
+ "deprecated",
822
+ "before",
823
+ "after",
824
+ "instead of",
825
+ ]
826
+ if symbol:
827
+ interesting.append(symbol.lower())
828
+ if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
829
+ interesting.extend(["cursor", "total_pages", "has_more", "page"])
830
+
831
+ evidence: list[Evidence] = []
832
+ for offset, line in enumerate(section.text.splitlines()):
833
+ stripped = _plain(line.strip().lstrip("#-*> ")).strip()
834
+ if not stripped:
835
+ continue
836
+ low = stripped.lower()
837
+ matched = [word for word in interesting if word in low]
838
+ if not matched:
839
+ continue
840
+ evidence.append(
841
+ Evidence(
842
+ quote=stripped[:240],
843
+ line=section.start_line + offset,
844
+ note=f"mentions {matched[0]}",
845
+ )
846
+ )
847
+ if len(evidence) >= 5:
848
+ break
849
+ return evidence
850
+
851
+
852
+ def _pagination_contract(text: str) -> PaginationContract:
853
+ """Read non-default pagination field names out of the notes."""
854
+ contract = PaginationContract()
855
+ # In document order, backticked names first: they are the vendor's literal
856
+ # spelling. The first candidate for each field wins. This used to iterate a
857
+ # `set`, so with several candidates the field chosen depended on the hash
858
+ # seed -- the same note could produce a different migration on each run.
859
+ identifiers = dict.fromkeys(
860
+ re.findall(r"`(\w+)`", text) + re.findall(r"\b(\w*(?:cursor|page|more)\w*)\b", text.lower())
861
+ )
862
+ assigned: set[str] = set()
863
+ for candidate in identifiers:
864
+ low = candidate.lower()
865
+ if low.endswith("_pages") or low == "total_pages":
866
+ key = "total_pages_key"
867
+ elif low.startswith("next") and "cursor" in low:
868
+ key = "next_cursor_key"
869
+ elif low in ("has_more", "hasmore", "more"):
870
+ key = "has_more_key"
871
+ else:
872
+ continue
873
+ if key not in assigned:
874
+ assigned.add(key)
875
+ setattr(contract, key, candidate)
876
+ return contract
877
+
878
+
879
+ def _severity(text: str) -> Severity:
880
+ match = _RISK.search(text)
881
+ if match:
882
+ return Severity.parse(match.group(1), Severity.MEDIUM)
883
+ return Severity.MEDIUM
884
+
885
+
886
+ def _confidence(kind: ChangeKind, score: int, target: SymbolTarget) -> Confidence:
887
+ """Grade how sure we are that we read this document correctly.
888
+
889
+ High requires both a strong classification signal *and* successful symbol
890
+ extraction, because a rename we cannot name is a rename we cannot perform.
891
+ """
892
+ if not kind.is_actionable:
893
+ return Confidence.LOW
894
+ if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
895
+ return Confidence.HIGH if score >= _STRONG_SCORE + 4 else Confidence.MEDIUM
896
+ if not target.is_rename:
897
+ return Confidence.LOW
898
+ if score >= _STRONG_SCORE and target.owner:
899
+ return Confidence.HIGH
900
+ if score >= _STRONG_SCORE:
901
+ return Confidence.MEDIUM
902
+ return Confidence.LOW
903
+
904
+
905
+ def parse_section(section: _Section, source: str = "markdown") -> list[BreakingChange]:
906
+ """Turn one document section into the breaking changes it states."""
907
+ kind, score, reason = classify(section.text)
908
+ if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
909
+ return [_pagination_change(section, score, reason, source)]
910
+
911
+ statements, context = _statements(section)
912
+ renames: list[tuple[_Rename, _Statement]] = []
913
+ seen: set[tuple[str, str]] = set()
914
+ unsupported: list[tuple[_Statement, str]] = []
915
+ for statement in statements:
916
+ found = _find_renames(statement.text)
917
+ for rename in found:
918
+ key = (rename.old.lstrip("."), rename.new.lstrip("."))
919
+ if key not in seen:
920
+ seen.add(key)
921
+ renames.append((rename, statement))
922
+ label = "" if found else _unsupported_label(statement.text)
923
+ if label:
924
+ unsupported.append((statement, label))
925
+
926
+ if len(renames) == 1 and not unsupported:
927
+ # One rename: the whole section describes it, so its Before/After
928
+ # examples and prose all count as evidence for owner and kind.
929
+ rename, statement = renames[0]
930
+ return [_rename_change(rename, statement.text, section.text, section, source)]
931
+ if renames or len(unsupported) > 1:
932
+ changes = [
933
+ _rename_change(
934
+ rename, statement.text, f"{statement.text}\n{context}", section, source, statement
935
+ )
936
+ for rename, statement in renames
937
+ ]
938
+ changes.extend(
939
+ _unsupported_change(statement, label, section, source)
940
+ for statement, label in unsupported
941
+ )
942
+ return changes
943
+ return [_section_change(section, kind, score, reason, source)]
944
+
945
+
946
+ def _rename_change(
947
+ rename: _Rename,
948
+ statement: str,
949
+ local: str,
950
+ section: _Section,
951
+ source: str,
952
+ located: _Statement | None = None,
953
+ ) -> BreakingChange:
954
+ """A change for one rename statement, read against ``local`` for its owner."""
955
+ title = section.title if located is None else _plain(statement)
956
+ evidence = (
957
+ _collect_evidence(section, ChangeKind.UNKNOWN, rename.symbol)
958
+ if located is None
959
+ else [Evidence(quote=_plain(statement)[:240], line=located.line, note="states the rename")]
960
+ )
961
+ if rename.is_move:
962
+ return BreakingChange(
963
+ title=title,
964
+ kind=ChangeKind.UNSUPPORTED,
965
+ severity=_severity(section.text),
966
+ confidence=Confidence.LOW,
967
+ evidence=evidence,
968
+ source=source,
969
+ classification_reason=(
970
+ f"`{rename.old}` -> `{rename.new}` keeps the name `{rename.symbol}` and "
971
+ f"changes where it lives -- a move to a different module or object, "
972
+ f"which PatchAhead v1 cannot migrate"
973
+ if rename.symbol == rename.replacement
974
+ else f"`{rename.old}` -> `{rename.new}` moves to a different module or "
975
+ f"object as well as changing its name; renaming it in place would "
976
+ f"write a name the old owner does not have"
977
+ ),
978
+ )
979
+
980
+ context = local if located is None else local.split("\n", 1)[-1]
981
+ kind, score, reason = _rename_kind(rename, statement, context)
982
+ if kind is ChangeKind.UNKNOWN and located is None:
983
+ kind, score, reason = classify(section.text)
984
+ if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
985
+ kind = ChangeKind.UNKNOWN
986
+ owner, owner_is_explicit = _extract_owner(local, rename.symbol, kind)
987
+ if rename.qualifier and kind is not ChangeKind.KWARG_RENAME:
988
+ # A dotted rename statement outranks anything scraped from prose.
989
+ owner, owner_is_explicit = rename.qualifier, True
990
+ target = SymbolTarget(
991
+ symbol=rename.symbol,
992
+ replacement=rename.replacement,
993
+ owner=owner,
994
+ owner_is_explicit=owner_is_explicit,
995
+ )
996
+ text = section.text if located is None else statement
997
+ return BreakingChange(
998
+ title=title,
999
+ kind=kind,
1000
+ target=target,
1001
+ old_behavior=_extract_labeled(text, "before", "old", "previously"),
1002
+ new_behavior=_extract_labeled(text, "after", "new", "now"),
1003
+ migration_hint=_extract_labeled(text, "migration", "action", "fix"),
1004
+ severity=_severity(section.text),
1005
+ confidence=_confidence(kind, score, target),
1006
+ evidence=evidence
1007
+ if located is not None
1008
+ else _collect_evidence(section, kind, rename.symbol),
1009
+ source=source,
1010
+ classification_reason=f"`{rename.old}` -> `{rename.new}`: {reason}",
1011
+ )
1012
+
1013
+
1014
+ def _unsupported_change(
1015
+ statement: _Statement, label: str, section: _Section, source: str
1016
+ ) -> BreakingChange:
1017
+ return BreakingChange(
1018
+ title=_plain(statement.text),
1019
+ kind=ChangeKind.UNSUPPORTED,
1020
+ severity=_severity(section.text),
1021
+ confidence=Confidence.LOW,
1022
+ evidence=[Evidence(quote=_plain(statement.text)[:240], line=statement.line)],
1023
+ source=source,
1024
+ classification_reason=f"looks like a {label}, which PatchAhead v1 cannot migrate",
1025
+ )
1026
+
1027
+
1028
+ def _section_change(
1029
+ section: _Section, kind: ChangeKind, score: int, reason: str, source: str
1030
+ ) -> BreakingChange:
1031
+ """A section that states no rename: classified as a whole, as before."""
1032
+ if kind.is_actionable:
1033
+ reason = (
1034
+ f"{reason}; could not extract the old and new names, so no migration can be planned"
1035
+ )
1036
+ target = SymbolTarget()
1037
+ return BreakingChange(
1038
+ title=section.title,
1039
+ kind=kind,
1040
+ target=target,
1041
+ old_behavior=_extract_labeled(section.text, "before", "old", "previously"),
1042
+ new_behavior=_extract_labeled(section.text, "after", "new", "now"),
1043
+ migration_hint=_extract_labeled(section.text, "migration", "action", "fix"),
1044
+ severity=_severity(section.text),
1045
+ confidence=_confidence(kind, score, target),
1046
+ evidence=_collect_evidence(section, kind, ""),
1047
+ source=source,
1048
+ classification_reason=reason,
1049
+ )
1050
+
1051
+
1052
+ def _pagination_change(section: _Section, score: int, reason: str, source: str) -> BreakingChange:
1053
+ """A page-to-cursor change, if the note names the cursor it moves to.
1054
+
1055
+ The handler writes the cursor parameter and response fields into the
1056
+ repository. A note that describes cursor pagination without naming the
1057
+ parameter -- `starting_after`, `NextToken` -- would have PatchAhead write
1058
+ default names the vendor never used, so it is reported instead.
1059
+ """
1060
+ names_cursor = re.search(r"`cursor=?`|\bcursor=", section.text)
1061
+ change = BreakingChange(
1062
+ title=section.title,
1063
+ kind=ChangeKind.PAGINATION_PAGE_TO_CURSOR if names_cursor else ChangeKind.UNSUPPORTED,
1064
+ old_behavior=_extract_labeled(section.text, "before", "old", "previously"),
1065
+ new_behavior=_extract_labeled(section.text, "after", "new", "now"),
1066
+ migration_hint=_extract_labeled(section.text, "migration", "action", "fix"),
1067
+ severity=_severity(section.text),
1068
+ evidence=_collect_evidence(section, ChangeKind.PAGINATION_PAGE_TO_CURSOR, ""),
1069
+ source=source,
1070
+ classification_reason=reason
1071
+ if names_cursor
1072
+ else (
1073
+ "a move to cursor-based pagination, but the note never names a `cursor` "
1074
+ "parameter, so the names PatchAhead would write are a guess"
1075
+ ),
1076
+ )
1077
+ change.confidence = _confidence(change.kind, score, change.target)
1078
+ if names_cursor:
1079
+ change.pagination = _pagination_contract(section.text)
1080
+ return change
1081
+
1082
+
1083
+ def _plain(text: str) -> str:
1084
+ """Statement text without Markdown emphasis, for titles and quotes."""
1085
+ return re.sub(r"\*\*|__", "", text).strip()
1086
+
1087
+
1088
+ def _distinct(changes: list[BreakingChange]) -> list[BreakingChange]:
1089
+ """Drop a rename the document states twice, keeping the first statement.
1090
+
1091
+ Release notes repeat themselves: a GitHub release and the project's
1092
+ changelog, both quoted in one Dependabot pull request, describe the same
1093
+ change. Patching it twice would report the second as "no impact".
1094
+ """
1095
+ seen: set[tuple[str, str, str, str]] = set()
1096
+ kept: list[BreakingChange] = []
1097
+ for change in changes:
1098
+ target = change.target
1099
+ key = (change.kind.value, target.symbol, target.replacement, target.owner)
1100
+ if target.is_rename and key in seen:
1101
+ continue
1102
+ seen.add(key)
1103
+ kept.append(change)
1104
+ return kept
1105
+
1106
+
1107
+ class MarkdownChangeParser(ChangeParser):
1108
+ """Parses Markdown, reStructuredText, and plain-text release notes."""
1109
+
1110
+ name = "markdown"
1111
+ suffixes = (".md", ".markdown", ".txt", ".rst", ".text", "")
1112
+
1113
+ def supports(self, document: ChangeDocument) -> bool:
1114
+ return document.suffix in self.suffixes
1115
+
1116
+ def parse(self, document: ChangeDocument) -> list[BreakingChange]:
1117
+ sections = _split_sections(document)
1118
+ changes = [
1119
+ change for section in sections for change in parse_section(section, source=self.name)
1120
+ ]
1121
+
1122
+ # A multi-section document usually has prose sections that classify as
1123
+ # UNKNOWN. Drop those *only if* at least one real change was found, so a
1124
+ # document with nothing in it still reports honestly rather than
1125
+ # returning an empty list that reads like "no breaking changes".
1126
+ actionable = _distinct([c for c in changes if c.kind is not ChangeKind.UNKNOWN])
1127
+ result = actionable or changes[:1]
1128
+ log.debug(
1129
+ "parsed %s: %d section(s) -> %d change(s) [%s]",
1130
+ document.path,
1131
+ len(sections),
1132
+ len(result),
1133
+ ", ".join(c.kind.value for c in result),
1134
+ )
1135
+ return result
1136
+
1137
+
1138
+ register(MarkdownChangeParser())