patchahead 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- patchahead/__init__.py +8 -0
- patchahead/analysis/__init__.py +52 -0
- patchahead/analysis/edits.py +143 -0
- patchahead/analysis/index.py +203 -0
- patchahead/analysis/python_ast.py +457 -0
- patchahead/apidiff/__init__.py +23 -0
- patchahead/apidiff/compare.py +366 -0
- patchahead/apidiff/download.py +95 -0
- patchahead/apidiff/surface.py +337 -0
- patchahead/ci.py +301 -0
- patchahead/cli.py +627 -0
- patchahead/config.py +284 -0
- patchahead/demo/__init__.py +256 -0
- patchahead/demo/fixtures/changes/field-rename.md +14 -0
- patchahead/demo/fixtures/changes/invoice-field-rename.md +21 -0
- patchahead/demo/fixtures/changes/kwarg-rename.md +14 -0
- patchahead/demo/fixtures/changes/method-rename.md +12 -0
- patchahead/demo/fixtures/changes/pagination-cursor.json +24 -0
- patchahead/demo/fixtures/changes/pagination-cursor.md +20 -0
- patchahead/demo/fixtures/changes/sdk-v2.md +31 -0
- patchahead/demo/fixtures/orders-service/README.md +51 -0
- patchahead/demo/fixtures/orders-service/app/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/app/client.py +15 -0
- patchahead/demo/fixtures/orders-service/app/models.py +10 -0
- patchahead/demo/fixtures/orders-service/app/order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/app/order_sync.py +21 -0
- patchahead/demo/fixtures/orders-service/conftest.py +6 -0
- patchahead/demo/fixtures/orders-service/pyproject.toml +16 -0
- patchahead/demo/fixtures/orders-service/tests/test_client.py +14 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_sync.py +11 -0
- patchahead/demo/fixtures/orders-service/upstream/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v1.py +34 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v2.py +56 -0
- patchahead/demo/serve.py +189 -0
- patchahead/domain/__init__.py +67 -0
- patchahead/domain/change.py +269 -0
- patchahead/domain/completeness.py +91 -0
- patchahead/domain/impact.py +248 -0
- patchahead/domain/patch.py +81 -0
- patchahead/domain/plan.py +170 -0
- patchahead/domain/result.py +210 -0
- patchahead/domain/validation.py +200 -0
- patchahead/engine.py +609 -0
- patchahead/handlers/__init__.py +35 -0
- patchahead/handlers/base.py +211 -0
- patchahead/handlers/field_rename.py +425 -0
- patchahead/handlers/kwarg_rename.py +201 -0
- patchahead/handlers/method_rename.py +608 -0
- patchahead/handlers/pagination.py +582 -0
- patchahead/ingest/__init__.py +32 -0
- patchahead/ingest/base.py +102 -0
- patchahead/ingest/markdown.py +1138 -0
- patchahead/ingest/structured.py +218 -0
- patchahead/llm/__init__.py +28 -0
- patchahead/llm/client.py +152 -0
- patchahead/llm/proposer.py +620 -0
- patchahead/observability.py +223 -0
- patchahead/reporting.py +451 -0
- patchahead/testing/__init__.py +22 -0
- patchahead/testing/discovery.py +113 -0
- patchahead/testing/runner.py +138 -0
- patchahead/validation/__init__.py +5 -0
- patchahead/validation/completeness.py +265 -0
- patchahead/validation/engine.py +531 -0
- patchahead/web/__init__.py +13 -0
- patchahead/web/server.py +279 -0
- patchahead/web/static/index.html +650 -0
- patchahead/workspace.py +382 -0
- patchahead-0.3.0.dist-info/METADATA +368 -0
- patchahead-0.3.0.dist-info/RECORD +75 -0
- patchahead-0.3.0.dist-info/WHEEL +5 -0
- patchahead-0.3.0.dist-info/entry_points.txt +2 -0
- patchahead-0.3.0.dist-info/licenses/LICENSE +21 -0
- patchahead-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1138 @@
|
|
|
1
|
+
"""Parsing release notes written as Markdown or plain text.
|
|
2
|
+
|
|
3
|
+
This is the format most upstream providers actually publish, and the least
|
|
4
|
+
structured. Two design rules follow from that:
|
|
5
|
+
|
|
6
|
+
**Classification is scored, not first-match.** Each migration family has
|
|
7
|
+
weighted signal phrases; the highest total wins, and the winning score and the
|
|
8
|
+
phrases that produced it are recorded on
|
|
9
|
+
:attr:`~patchahead.domain.change.BreakingChange.classification_reason` so a user
|
|
10
|
+
can see why a document was read the way it was.
|
|
11
|
+
|
|
12
|
+
**Uncertainty is represented, not smoothed over.** A document that matches
|
|
13
|
+
nothing yields ``ChangeKind.UNKNOWN``; a recognized-but-unhandled change yields
|
|
14
|
+
``ChangeKind.UNSUPPORTED``; a rename whose symbols could not be extracted keeps
|
|
15
|
+
``Confidence.LOW``. The prototype instead defaulted unparseable documents to the
|
|
16
|
+
demo's pagination text, which made every failure look like a confident success
|
|
17
|
+
(``docs/assessment.md`` §2.5).
|
|
18
|
+
|
|
19
|
+
**A section can hold several changes.** Vendors list renames as bullets, table
|
|
20
|
+
rows, or two clauses of one sentence under a single heading. Each statement is
|
|
21
|
+
read on its own, so a table of four renamed methods yields four changes, and a
|
|
22
|
+
bullet about an endpoint move is reported as unsupported rather than dropped.
|
|
23
|
+
A section that states one rename is still read as a whole, which is what lets a
|
|
24
|
+
Before/After example under it supply an owner.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import html
|
|
30
|
+
import logging
|
|
31
|
+
import re
|
|
32
|
+
from dataclasses import dataclass
|
|
33
|
+
|
|
34
|
+
from patchahead.domain.change import (
|
|
35
|
+
BreakingChange,
|
|
36
|
+
ChangeKind,
|
|
37
|
+
Confidence,
|
|
38
|
+
Evidence,
|
|
39
|
+
PaginationContract,
|
|
40
|
+
Severity,
|
|
41
|
+
SymbolTarget,
|
|
42
|
+
)
|
|
43
|
+
from patchahead.ingest.base import ChangeDocument, ChangeParser, register
|
|
44
|
+
|
|
45
|
+
log = logging.getLogger(__name__)
|
|
46
|
+
|
|
47
|
+
#: Signal phrases per change kind, with weights. Phrases are matched as regular
|
|
48
|
+
#: expressions against the lowercased section text. Weights are relative: a
|
|
49
|
+
#: phrase that names the construct ("keyword argument") outranks one that merely
|
|
50
|
+
#: suggests it ("renamed").
|
|
51
|
+
_SIGNALS: dict[ChangeKind, list[tuple[str, int]]] = {
|
|
52
|
+
ChangeKind.KWARG_RENAME: [
|
|
53
|
+
(r"keyword argument", 5),
|
|
54
|
+
(r"\bkwarg\b", 5),
|
|
55
|
+
(r"\bparameter\b.{0,40}\brenamed\b", 4),
|
|
56
|
+
(r"\brenamed\b.{0,40}\bparameter\b", 4),
|
|
57
|
+
(r"\bargument\b.{0,40}\brenamed\b", 4),
|
|
58
|
+
(r"\brenamed\b.{0,40}\bargument\b", 4),
|
|
59
|
+
(r"`\w+=`", 3),
|
|
60
|
+
],
|
|
61
|
+
ChangeKind.METHOD_RENAME: [
|
|
62
|
+
(r"method (?:was )?renamed", 5),
|
|
63
|
+
(r"renamed method", 5),
|
|
64
|
+
(r"function (?:was )?renamed", 5),
|
|
65
|
+
(r"renamed function", 5),
|
|
66
|
+
(r"method renamed", 5),
|
|
67
|
+
(r"deprecated method", 3),
|
|
68
|
+
(r"`\w+\(\)`.{0,30}(?:->|→|renamed to)", 3),
|
|
69
|
+
(r"\bmethod\b", 1),
|
|
70
|
+
],
|
|
71
|
+
ChangeKind.FIELD_RENAME: [
|
|
72
|
+
(r"field (?:was )?renamed", 5),
|
|
73
|
+
(r"renamed field", 5),
|
|
74
|
+
(r"field renamed", 5),
|
|
75
|
+
(r"attribute (?:was )?renamed", 4),
|
|
76
|
+
(r"property (?:was )?renamed", 4),
|
|
77
|
+
(r"\bfield\b.{0,40}(?:->|→)", 3),
|
|
78
|
+
(r"\bfield\b", 1),
|
|
79
|
+
],
|
|
80
|
+
ChangeKind.PAGINATION_PAGE_TO_CURSOR: [
|
|
81
|
+
(r"cursor-based", 5),
|
|
82
|
+
(r"\bnext_cursor\b", 4),
|
|
83
|
+
(r"\btotal_pages\b", 4),
|
|
84
|
+
(r"page-based", 4),
|
|
85
|
+
(r"\bhas_more\b", 3),
|
|
86
|
+
(r"\bpagination\b", 3),
|
|
87
|
+
(r"\bcursor\b", 2),
|
|
88
|
+
],
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
#: Changes PatchAhead can recognize but has no v1 handler for. Detecting them
|
|
92
|
+
#: explicitly lets the CLI say "this is an endpoint move, which v1 cannot
|
|
93
|
+
#: migrate" instead of misclassifying it as something it can.
|
|
94
|
+
_UNSUPPORTED_SIGNALS: list[tuple[str, int, str]] = [
|
|
95
|
+
(r"endpoint (?:was )?(?:moved|changed|removed)", 5, "endpoint change"),
|
|
96
|
+
(r"moved to a (?:new|different) endpoint", 5, "endpoint change"),
|
|
97
|
+
(r"response (?:shape|format|schema) (?:has )?changed", 5, "response shape change"),
|
|
98
|
+
(r"schema (?:was )?changed", 4, "response shape change"),
|
|
99
|
+
(r"authentication (?:has )?changed", 4, "authentication change"),
|
|
100
|
+
(r"\brate limit(?:ing)? (?:has )?changed", 4, "rate limiting change"),
|
|
101
|
+
(r"now returns? an? (?:list|array|object) instead", 4, "response shape change"),
|
|
102
|
+
(r"replaced by an? `[^`]+` (?:dict|dictionary|object|class|mapping)", 5, "restructuring"),
|
|
103
|
+
]
|
|
104
|
+
|
|
105
|
+
#: Minimum score before a classification is trusted at all.
|
|
106
|
+
_MIN_SCORE = 3
|
|
107
|
+
#: Score at or above which a classification is considered confident.
|
|
108
|
+
_STRONG_SCORE = 5
|
|
109
|
+
|
|
110
|
+
_HEADING = re.compile(r"^(#{2,4})\s+(.+?)\s*#*$", re.MULTILINE)
|
|
111
|
+
_NON_BREAKING_HEADING = re.compile(
|
|
112
|
+
r"^#{1,4}\s+(?:non[- ]breaking|additions?|added|improvements?|enhancements?|"
|
|
113
|
+
r"fixes|fixed|bug ?fixes|(?:new )?features?)\b",
|
|
114
|
+
re.IGNORECASE,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
#: Nouns a release note puts between a renamed name and the word "to": "renamed
|
|
118
|
+
#: the `retries` keyword argument to `max_retries`". Spelled out rather than
|
|
119
|
+
#: allowed as "any two words", because `\w+` would also match "renamed the
|
|
120
|
+
#: `client` argument passed to `fetch_orders`" -- which names two symbols and
|
|
121
|
+
#: renames neither of them.
|
|
122
|
+
_RENAME_NOUN = (
|
|
123
|
+
r"(?:keyword|kwarg|argument|arg|parameter|param|field|attribute|property"
|
|
124
|
+
r"|option|setting|method|function)s?"
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _name(group: str) -> str:
|
|
129
|
+
"""A backticked name, recording whether it was written as a call or a keyword."""
|
|
130
|
+
return rf"`(?P<{group}>\.?[A-Za-z_][\w.]*)(?P<{group}_call>\(\))?(?P<{group}_kw>=)?`"
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
_OLD, _NEW = _name("old"), _name("new")
|
|
134
|
+
#: "of `get()` and `post()`", "on the `Charge` object": a clause naming what the
|
|
135
|
+
#: renamed thing belongs to, which may sit between the name and the verb.
|
|
136
|
+
_QUALIFIER = (
|
|
137
|
+
r"(?:\s+(?:of|on|for|in)\s+(?:the\s+|each\s+|all\s+)?"
|
|
138
|
+
r"(?:`[^`\n]+`(?:\s*(?:,|and|or)\s*`[^`\n]+`)*|[A-Z]\w*)(?:\s+[a-z]+)?)?"
|
|
139
|
+
)
|
|
140
|
+
_VERB = (
|
|
141
|
+
r"(?:(?:(?:was|is|are|were|has\s+been|have\s+been)\s+)?(?:renamed|changed)\s+to"
|
|
142
|
+
r"|(?:was|is|are|were|has\s+been|have\s+been)\s+replaced\s+(?:by|with)"
|
|
143
|
+
r"|(?:is|are)\s+now(?:\s+(?:called|named|spelled|returned\s+as|exposed\s+as))?"
|
|
144
|
+
r"|(?:is|was|has\s+been)\s+deprecated\s+in\s+favou?r\s+of)"
|
|
145
|
+
)
|
|
146
|
+
#: How each notation states a rename, most trustworthy first. ``\`a\` ->
|
|
147
|
+
#: \`b\``` is unambiguous; a subject-verb sentence can pick up the wrong name
|
|
148
|
+
#: from an intervening clause, which is why the qualifier is matched explicitly.
|
|
149
|
+
_RENAME_PATTERNS: tuple[re.Pattern[str], ...] = (
|
|
150
|
+
re.compile(rf"{_OLD}\s*(?:->|→|=>)\s*{_NEW}"),
|
|
151
|
+
re.compile(
|
|
152
|
+
rf"renamed\s+(?:from\s+)?(?:the\s+|an?\s+)?{_OLD}(?:\s+{_RENAME_NOUN}){{0,2}}"
|
|
153
|
+
rf"\s+to\s+{_NEW}",
|
|
154
|
+
re.IGNORECASE,
|
|
155
|
+
),
|
|
156
|
+
re.compile(rf"{_OLD}(?:\s+[A-Za-z]+){{0,3}}?{_QUALIFIER}\s+{_VERB}\s+{_NEW}", re.IGNORECASE),
|
|
157
|
+
re.compile(rf"\b(?:use|call)\s+{_NEW}\s+instead\s+of\s+{_OLD}", re.IGNORECASE),
|
|
158
|
+
re.compile(rf"\breplace\s+{_OLD}\s+with\s+{_NEW}", re.IGNORECASE),
|
|
159
|
+
)
|
|
160
|
+
#: "Renamed `a()` to `b()` and `c()` to `d()`": the second pair, only read in a
|
|
161
|
+
#: statement that already says "renamed".
|
|
162
|
+
_AND_TO = re.compile(rf"(?:,|\band)\s+{_OLD}\s+to\s+{_NEW}")
|
|
163
|
+
#: Words that look like a new name after "is now" but describe a type or a state.
|
|
164
|
+
_NOT_A_NAME = {
|
|
165
|
+
"int",
|
|
166
|
+
"float",
|
|
167
|
+
"str",
|
|
168
|
+
"bool",
|
|
169
|
+
"list",
|
|
170
|
+
"dict",
|
|
171
|
+
"tuple",
|
|
172
|
+
"set",
|
|
173
|
+
"bytes",
|
|
174
|
+
"none",
|
|
175
|
+
"null",
|
|
176
|
+
"true",
|
|
177
|
+
"false",
|
|
178
|
+
"optional",
|
|
179
|
+
"required",
|
|
180
|
+
"datetime",
|
|
181
|
+
"date",
|
|
182
|
+
"decimal",
|
|
183
|
+
"any",
|
|
184
|
+
"object",
|
|
185
|
+
}
|
|
186
|
+
#: The construct a noun names, for deciding what kind of rename a statement is.
|
|
187
|
+
_NOUN_KINDS: tuple[tuple[re.Pattern[str], ChangeKind], ...] = (
|
|
188
|
+
(
|
|
189
|
+
re.compile(
|
|
190
|
+
r"\b(?:keyword\s+arguments?|kwargs?|arguments?|args|parameters?|params?)\b", re.I
|
|
191
|
+
),
|
|
192
|
+
ChangeKind.KWARG_RENAME,
|
|
193
|
+
),
|
|
194
|
+
(re.compile(r"\b(?:methods?|functions?)\b", re.I), ChangeKind.METHOD_RENAME),
|
|
195
|
+
(
|
|
196
|
+
re.compile(r"\b(?:fields?|propert(?:y|ies)|attributes?|keys?)\b", re.I),
|
|
197
|
+
ChangeKind.FIELD_RENAME,
|
|
198
|
+
),
|
|
199
|
+
)
|
|
200
|
+
_BOLD_FIELD = re.compile(r"\*\*(?P<label>[A-Za-z][\w /-]*?)\s*:?\*\*[:\s]*(?P<value>.+)")
|
|
201
|
+
_RISK = re.compile(r"risk:?\s*(high|medium|low)", re.IGNORECASE)
|
|
202
|
+
|
|
203
|
+
_STOPWORDS = {
|
|
204
|
+
"the",
|
|
205
|
+
"a",
|
|
206
|
+
"an",
|
|
207
|
+
"is",
|
|
208
|
+
"was",
|
|
209
|
+
"were",
|
|
210
|
+
"to",
|
|
211
|
+
"from",
|
|
212
|
+
"in",
|
|
213
|
+
"on",
|
|
214
|
+
"of",
|
|
215
|
+
"and",
|
|
216
|
+
"or",
|
|
217
|
+
"now",
|
|
218
|
+
"new",
|
|
219
|
+
"old",
|
|
220
|
+
"this",
|
|
221
|
+
"that",
|
|
222
|
+
"it",
|
|
223
|
+
"be",
|
|
224
|
+
"been",
|
|
225
|
+
"before",
|
|
226
|
+
"after",
|
|
227
|
+
"migration",
|
|
228
|
+
"risk",
|
|
229
|
+
"breaking",
|
|
230
|
+
"change",
|
|
231
|
+
"changes",
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
@dataclass
|
|
236
|
+
class _Section:
|
|
237
|
+
"""One ``###``-delimited chunk of a document."""
|
|
238
|
+
|
|
239
|
+
title: str
|
|
240
|
+
text: str
|
|
241
|
+
start_line: int
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
_HTML_BLOCK = re.compile(r"<(?:h[1-6]|li|ul|ol|p|code|blockquote|details)\b", re.IGNORECASE)
|
|
245
|
+
_HTML_HEADING = re.compile(r"<h([1-6])[^>]*>(.*?)</h\1\s*>", re.IGNORECASE | re.DOTALL)
|
|
246
|
+
_COMMITS_BLOCK = re.compile(
|
|
247
|
+
r"<details>\s*<summary>\s*Commits\s*</summary>.*?</details>", re.IGNORECASE | re.DOTALL
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _from_html(text: str) -> str:
|
|
252
|
+
"""Rewrite HTML release notes -- what Dependabot puts in a pull request -- as Markdown.
|
|
253
|
+
|
|
254
|
+
Headings become ``##``/``###``, list items become bullets, ``<code>``
|
|
255
|
+
becomes backticks, and every other tag is dropped. A Dependabot "Commits"
|
|
256
|
+
list is removed first: it is commit subjects, not release notes.
|
|
257
|
+
"""
|
|
258
|
+
text = _COMMITS_BLOCK.sub("", text)
|
|
259
|
+
text = _HTML_HEADING.sub(
|
|
260
|
+
lambda m: f"\n{'#' * min(max(int(m.group(1)), 2), 4)} {m.group(2).strip()}\n", text
|
|
261
|
+
)
|
|
262
|
+
text = re.sub(r"<code>(.*?)</code>", r"`\1`", text, flags=re.IGNORECASE | re.DOTALL)
|
|
263
|
+
text = re.sub(r"<li[^>]*>", "\n- ", text, flags=re.IGNORECASE)
|
|
264
|
+
text = re.sub(r"<br\s*/?>|</p\s*>|</li\s*>|</?[uo]l[^>]*>", "\n", text, flags=re.IGNORECASE)
|
|
265
|
+
text = re.sub(r"<[^>]+>", "", text)
|
|
266
|
+
text = html.unescape(text)
|
|
267
|
+
return re.sub(r"\n{3,}", "\n\n", text)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
_UNDERLINE = re.compile(r"^([=\-~^\"'*+#])\1{2,}\s*$")
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
#: Sphinx cross-references to code: :meth:`.assertTrue`, :class:`~unittest.TestCase`,
|
|
274
|
+
#: :func:`Title <pkg.func>`. Other roles (:gh:`123`, :ref:`...`) are left as text.
|
|
275
|
+
_ROLE = re.compile(
|
|
276
|
+
r":(?:py:)?(?P<role>meth|func|attr|class|mod|data|exc|obj|const):"
|
|
277
|
+
r"`(?P<title>[^`<]*?)(?:\s*<(?P<target>[^>`]+)>)?`"
|
|
278
|
+
)
|
|
279
|
+
#: A reStructuredText simple-table border: columns of ``=`` separated by spaces.
|
|
280
|
+
_RST_BORDER = re.compile(r"^(?P<indent>\s*)=+(?: +=+)+\s*$")
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _sphinx_role(match: re.Match[str]) -> str:
|
|
284
|
+
""":meth:`.assertTrue` -> `assertTrue()`; :class:`~unittest.TestCase` -> `TestCase`."""
|
|
285
|
+
name = (match.group("target") or match.group("title")).strip().lstrip("!")
|
|
286
|
+
if name.startswith("~"):
|
|
287
|
+
name = name[1:].rsplit(".", 1)[-1]
|
|
288
|
+
name = name.lstrip(".").removesuffix("()")
|
|
289
|
+
return f"`{name}()`" if match.group("role") in ("meth", "func") else f"`{name}`"
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _rst_tables(lines: list[str]) -> list[str]:
|
|
293
|
+
"""Rewrite reStructuredText simple tables as Markdown pipe tables, in place.
|
|
294
|
+
|
|
295
|
+
Columns are where the border's runs of ``=`` are, so cells are cut by
|
|
296
|
+
position before any markup inside them changes width. Borders become blank
|
|
297
|
+
lines, keeping line numbers.
|
|
298
|
+
"""
|
|
299
|
+
index = 0
|
|
300
|
+
while index < len(lines):
|
|
301
|
+
top = _RST_BORDER.match(lines[index])
|
|
302
|
+
if not top:
|
|
303
|
+
index += 1
|
|
304
|
+
continue
|
|
305
|
+
borders = [i for i in range(index, len(lines)) if _RST_BORDER.match(lines[i])][:3]
|
|
306
|
+
if len(borders) < 3:
|
|
307
|
+
break
|
|
308
|
+
spans = [m.start() for m in re.finditer(r"=+", lines[index])]
|
|
309
|
+
|
|
310
|
+
def cells(line: str, spans: list[int] = spans) -> str:
|
|
311
|
+
ends = spans[1:] + [len(line)]
|
|
312
|
+
return (
|
|
313
|
+
"| "
|
|
314
|
+
+ " | ".join(line[a:b].strip() for a, b in zip(spans, ends, strict=True))
|
|
315
|
+
+ " |"
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
indent = top.group("indent")
|
|
319
|
+
first, rule, last = borders
|
|
320
|
+
for row in range(first + 1, last):
|
|
321
|
+
if row == rule:
|
|
322
|
+
lines[row] = indent + "|" + "|".join("---" for _ in spans) + "|"
|
|
323
|
+
elif lines[row].strip():
|
|
324
|
+
lines[row] = indent + cells(lines[row])
|
|
325
|
+
lines[first] = lines[last] = ""
|
|
326
|
+
index = last + 1
|
|
327
|
+
return lines
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _normalize(text: str) -> str:
|
|
331
|
+
"""Rewrite HTML, reStructuredText, and Sphinx markup as plain Markdown.
|
|
332
|
+
|
|
333
|
+
Simple tables become pipe tables, Sphinx cross-references to code become
|
|
334
|
+
backticked names (with ``()`` for a method or function, which is how the
|
|
335
|
+
parser tells a call from a field), a double-backtick literal becomes a
|
|
336
|
+
single-backtick one, and a title underlined with ``====`` or ``----``
|
|
337
|
+
becomes a ``##`` heading -- levels assigned in order of first appearance,
|
|
338
|
+
as reStructuredText does. Underlines and table borders become blank lines,
|
|
339
|
+
so line numbers in evidence still point at the original document.
|
|
340
|
+
"""
|
|
341
|
+
if _HTML_BLOCK.search(text):
|
|
342
|
+
text = _from_html(text)
|
|
343
|
+
lines = _rst_tables(text.splitlines())
|
|
344
|
+
text = "\n".join(lines) + ("\n" if text.endswith("\n") else "")
|
|
345
|
+
text = _ROLE.sub(_sphinx_role, text)
|
|
346
|
+
text = re.sub(r"``([^`\n]+)``", r"`\1`", text)
|
|
347
|
+
lines = text.splitlines()
|
|
348
|
+
levels: dict[str, int] = {}
|
|
349
|
+
for index in range(1, len(lines)):
|
|
350
|
+
match = _UNDERLINE.match(lines[index])
|
|
351
|
+
title = lines[index - 1].strip()
|
|
352
|
+
if not match or not title or _UNDERLINE.match(title) or title.startswith(("-", "*", "|")):
|
|
353
|
+
continue
|
|
354
|
+
level = levels.setdefault(match.group(1), min(2 + len(levels), 4))
|
|
355
|
+
lines[index - 1] = f"{'#' * level} {title}"
|
|
356
|
+
lines[index] = ""
|
|
357
|
+
return "\n".join(lines) + ("\n" if text.endswith("\n") else "")
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def _split_sections(document: ChangeDocument) -> list[_Section]:
|
|
361
|
+
"""Split a document at its headings, dropping non-breaking-change sections.
|
|
362
|
+
|
|
363
|
+
A release note usually describes several changes. Treating the whole file as
|
|
364
|
+
one blob lets a "Non-breaking changes" bullet contribute signal words to a
|
|
365
|
+
breaking change's classification.
|
|
366
|
+
"""
|
|
367
|
+
text = _normalize(document.text)
|
|
368
|
+
matches = list(_HEADING.finditer(text))
|
|
369
|
+
if not matches:
|
|
370
|
+
return [_Section(title=_infer_title(text), text=text, start_line=1)]
|
|
371
|
+
|
|
372
|
+
sections: list[_Section] = []
|
|
373
|
+
suppressing = False
|
|
374
|
+
for position, match in enumerate(matches):
|
|
375
|
+
heading_line = text[: match.start()].count("\n") + 1
|
|
376
|
+
heading_text = text[match.start() : match.end()]
|
|
377
|
+
level = len(match.group(1))
|
|
378
|
+
body_end = matches[position + 1].start() if position + 1 < len(matches) else len(text)
|
|
379
|
+
body = text[match.end() : body_end]
|
|
380
|
+
|
|
381
|
+
if _NON_BREAKING_HEADING.match(heading_text.strip()):
|
|
382
|
+
# Suppress this heading and any sub-headings beneath it.
|
|
383
|
+
suppressing = True
|
|
384
|
+
suppress_level = level
|
|
385
|
+
continue
|
|
386
|
+
if suppressing and level > suppress_level:
|
|
387
|
+
continue
|
|
388
|
+
suppressing = False
|
|
389
|
+
|
|
390
|
+
title = match.group(2).strip()
|
|
391
|
+
if _looks_like_container_heading(title, body):
|
|
392
|
+
continue
|
|
393
|
+
sections.append(_Section(title=title, text=heading_text + body, start_line=heading_line))
|
|
394
|
+
|
|
395
|
+
if not sections:
|
|
396
|
+
return [_Section(title=_infer_title(text), text=text, start_line=1)]
|
|
397
|
+
return sections
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _looks_like_container_heading(title: str, body: str) -> bool:
|
|
401
|
+
"""Whether a heading is a wrapper (``## Breaking changes``) with no content."""
|
|
402
|
+
if not re.match(r"(?i)^(breaking changes?|changes?|release notes?)$", title.strip()):
|
|
403
|
+
return False
|
|
404
|
+
return not body.strip() or bool(_HEADING.search(body))
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _infer_title(text: str) -> str:
|
|
408
|
+
for line in text.splitlines():
|
|
409
|
+
stripped = line.strip().lstrip("#").strip()
|
|
410
|
+
if stripped:
|
|
411
|
+
return stripped[:120]
|
|
412
|
+
return "Upstream breaking change"
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def classify(text: str) -> tuple[ChangeKind, int, str]:
|
|
416
|
+
"""Classify a block of release-note text.
|
|
417
|
+
|
|
418
|
+
Returns the kind, the winning score, and a human-readable reason naming the
|
|
419
|
+
phrases that decided it.
|
|
420
|
+
"""
|
|
421
|
+
low = text.lower()
|
|
422
|
+
|
|
423
|
+
scores: dict[ChangeKind, tuple[int, list[str]]] = {}
|
|
424
|
+
for kind, signals in _SIGNALS.items():
|
|
425
|
+
total = 0
|
|
426
|
+
hits: list[str] = []
|
|
427
|
+
for pattern, weight in signals:
|
|
428
|
+
if re.search(pattern, low):
|
|
429
|
+
total += weight
|
|
430
|
+
hits.append(pattern)
|
|
431
|
+
if total:
|
|
432
|
+
scores[kind] = (total, hits)
|
|
433
|
+
|
|
434
|
+
unsupported_score = 0
|
|
435
|
+
unsupported_label = ""
|
|
436
|
+
for pattern, weight, label in _UNSUPPORTED_SIGNALS:
|
|
437
|
+
if weight > unsupported_score and re.search(pattern, low):
|
|
438
|
+
unsupported_score, unsupported_label = weight, label
|
|
439
|
+
|
|
440
|
+
if not scores and not unsupported_score:
|
|
441
|
+
return ChangeKind.UNKNOWN, 0, "no recognized breaking-change signals in the document"
|
|
442
|
+
|
|
443
|
+
best_kind, (best_score, best_hits) = (
|
|
444
|
+
max(scores.items(), key=lambda item: item[1][0])
|
|
445
|
+
if scores
|
|
446
|
+
else (ChangeKind.UNKNOWN, (0, []))
|
|
447
|
+
)
|
|
448
|
+
|
|
449
|
+
if unsupported_score > best_score:
|
|
450
|
+
return (
|
|
451
|
+
ChangeKind.UNSUPPORTED,
|
|
452
|
+
unsupported_score,
|
|
453
|
+
f"looks like a {unsupported_label}, which PatchAhead v1 cannot migrate",
|
|
454
|
+
)
|
|
455
|
+
if best_score < _MIN_SCORE:
|
|
456
|
+
return (
|
|
457
|
+
ChangeKind.UNKNOWN,
|
|
458
|
+
best_score,
|
|
459
|
+
f"weak signals only (score {best_score}, need {_MIN_SCORE}); "
|
|
460
|
+
f"closest match was {best_kind.value}",
|
|
461
|
+
)
|
|
462
|
+
|
|
463
|
+
phrases = ", ".join(sorted(hit.replace("\\b", "") for hit in best_hits)[:4])
|
|
464
|
+
return best_kind, best_score, f"matched {best_kind.value} signals: {phrases}"
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
@dataclass(frozen=True)
|
|
468
|
+
class _Rename:
|
|
469
|
+
"""One rename statement: ``old`` -> ``new`` as written, and where it was found."""
|
|
470
|
+
|
|
471
|
+
old: str
|
|
472
|
+
new: str
|
|
473
|
+
#: How the names were written: a call ``x()`` or a keyword ``x=``.
|
|
474
|
+
call: bool
|
|
475
|
+
keyword: bool
|
|
476
|
+
#: Notation rank (lower is more trustworthy), then position in the text.
|
|
477
|
+
priority: int
|
|
478
|
+
start: int
|
|
479
|
+
end: int
|
|
480
|
+
|
|
481
|
+
@property
|
|
482
|
+
def symbol(self) -> str:
|
|
483
|
+
return self.old.rsplit(".", 1)[-1]
|
|
484
|
+
|
|
485
|
+
@property
|
|
486
|
+
def replacement(self) -> str:
|
|
487
|
+
return self.new.rsplit(".", 1)[-1]
|
|
488
|
+
|
|
489
|
+
@property
|
|
490
|
+
def qualifier(self) -> str:
|
|
491
|
+
"""The receiver prefix of a dotted old name -- an *asserted* owner."""
|
|
492
|
+
return self.old.rsplit(".", 1)[0].lstrip(".") if "." in self.old.lstrip(".") else ""
|
|
493
|
+
|
|
494
|
+
@property
|
|
495
|
+
def new_qualifier(self) -> str:
|
|
496
|
+
return self.new.rsplit(".", 1)[0].lstrip(".") if "." in self.new.lstrip(".") else ""
|
|
497
|
+
|
|
498
|
+
@property
|
|
499
|
+
def is_move(self) -> bool:
|
|
500
|
+
"""Whether the thing changes where it lives, not just what it is called.
|
|
501
|
+
|
|
502
|
+
``a.X.create`` -> ``b.y.create`` keeps the name. ``imp.find_module()``
|
|
503
|
+
-> ``importlib.util.find_spec()`` changes both: renaming the call in
|
|
504
|
+
place would write ``imp.find_spec()``, which does not exist. A new name
|
|
505
|
+
under a different owner is a move; ``Client.fetch_all`` ->
|
|
506
|
+
``Client.list_all``, or ``Client.fetch_all`` -> ``list_all``, is not.
|
|
507
|
+
"""
|
|
508
|
+
if self.symbol == self.replacement:
|
|
509
|
+
return True
|
|
510
|
+
new_home = self.new_qualifier.lower()
|
|
511
|
+
return bool(new_home) and new_home != self.qualifier.lower()
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def _find_renames(text: str) -> list[_Rename]:
|
|
515
|
+
"""Every rename a piece of text states, in document order.
|
|
516
|
+
|
|
517
|
+
Several notations can match the same words, and they are not equally
|
|
518
|
+
trustworthy: in "The `timeout_seconds` keyword argument on `fetch_orders`
|
|
519
|
+
was renamed to `timeout`", a bare "`x` was renamed to `y`" reading would
|
|
520
|
+
pick ``fetch_orders`` -> ``timeout``. Matches are therefore taken best
|
|
521
|
+
notation first, and a match overlapping one already taken is discarded.
|
|
522
|
+
"""
|
|
523
|
+
candidates: list[_Rename] = []
|
|
524
|
+
for priority, pattern in enumerate(_RENAME_PATTERNS):
|
|
525
|
+
candidates.extend(_rename(match, priority) for match in pattern.finditer(text))
|
|
526
|
+
if re.search(r"\brenamed\b", text, re.IGNORECASE):
|
|
527
|
+
candidates.extend(_rename(match, len(_RENAME_PATTERNS)) for match in _AND_TO.finditer(text))
|
|
528
|
+
|
|
529
|
+
taken: list[_Rename] = []
|
|
530
|
+
for candidate in sorted(candidates, key=lambda r: (r.priority, r.start)):
|
|
531
|
+
if not _plausible(candidate):
|
|
532
|
+
continue
|
|
533
|
+
if any(candidate.start < r.end and r.start < candidate.end for r in taken):
|
|
534
|
+
continue
|
|
535
|
+
taken.append(candidate)
|
|
536
|
+
return sorted(taken, key=lambda r: r.start)
|
|
537
|
+
|
|
538
|
+
|
|
539
|
+
def _rename(match: re.Match[str], priority: int) -> _Rename:
|
|
540
|
+
return _Rename(
|
|
541
|
+
old=match.group("old"),
|
|
542
|
+
new=match.group("new"),
|
|
543
|
+
call=bool(match.group("old_call") or match.group("new_call")),
|
|
544
|
+
keyword=bool(match.group("old_kw") or match.group("new_kw")),
|
|
545
|
+
priority=priority,
|
|
546
|
+
start=match.start(),
|
|
547
|
+
end=match.end(),
|
|
548
|
+
)
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def _plausible(rename: _Rename) -> bool:
|
|
552
|
+
old, new = rename.symbol.lower(), rename.replacement.lower()
|
|
553
|
+
if not old or not new or rename.old == rename.new:
|
|
554
|
+
return False
|
|
555
|
+
if old in _STOPWORDS or new in _STOPWORDS:
|
|
556
|
+
return False
|
|
557
|
+
# "`timeout` is now `float`" changes a type, not a name. A type is written
|
|
558
|
+
# bare; `list()` or `dict=` is a call or a keyword, and a real new name.
|
|
559
|
+
return rename.call or rename.keyword or new not in _NOT_A_NAME
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
@dataclass
|
|
563
|
+
class _Statement:
|
|
564
|
+
"""One bullet, table row, paragraph or heading inside a section."""
|
|
565
|
+
|
|
566
|
+
text: str
|
|
567
|
+
line: int
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
_BULLET = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+")
|
|
571
|
+
_TABLE_RULE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)*\|?\s*$")
|
|
572
|
+
_OLD_COLUMN = re.compile(r"(?i)\b(?:old|before|previous(?:ly)?|from|deprecated|removed)\b")
|
|
573
|
+
_NEW_COLUMN = re.compile(r"(?i)\b(?:new|after|replacement|now|to|use)\b")
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
def _cells(row: str) -> list[str]:
|
|
577
|
+
return [cell.strip() for cell in row.strip().strip("|").split("|")]
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
def _statements(section: _Section) -> tuple[list[_Statement], str]:
|
|
581
|
+
"""Split a section into statements, plus the prose that frames them.
|
|
582
|
+
|
|
583
|
+
A table row becomes ``old → new`` followed by its remaining cells, with the
|
|
584
|
+
columns chosen by their headers ("Old"/"New", "Before"/"After", "v1"/"v2").
|
|
585
|
+
The returned context -- the heading and any prose that is not a bullet or a
|
|
586
|
+
row -- is what each statement is read against for the nouns and owners a
|
|
587
|
+
row or terse bullet leaves out.
|
|
588
|
+
"""
|
|
589
|
+
lines = section.text.splitlines()
|
|
590
|
+
statements: list[_Statement] = []
|
|
591
|
+
context: list[str] = []
|
|
592
|
+
in_fence = False
|
|
593
|
+
index = 0
|
|
594
|
+
while index < len(lines):
|
|
595
|
+
line = lines[index]
|
|
596
|
+
number = section.start_line + index
|
|
597
|
+
if line.strip().startswith("```"):
|
|
598
|
+
in_fence = not in_fence
|
|
599
|
+
index += 1
|
|
600
|
+
continue
|
|
601
|
+
if in_fence or not line.strip():
|
|
602
|
+
index += 1
|
|
603
|
+
continue
|
|
604
|
+
|
|
605
|
+
if (
|
|
606
|
+
line.lstrip().startswith("|")
|
|
607
|
+
and index + 1 < len(lines)
|
|
608
|
+
and _TABLE_RULE.match(lines[index + 1])
|
|
609
|
+
):
|
|
610
|
+
header = _cells(line)
|
|
611
|
+
old_col = next((i for i, c in enumerate(header) if _OLD_COLUMN.search(c)), 0)
|
|
612
|
+
new_col = next(
|
|
613
|
+
(i for i, c in enumerate(header) if i != old_col and _NEW_COLUMN.search(c)),
|
|
614
|
+
1 if old_col == 0 else 0,
|
|
615
|
+
)
|
|
616
|
+
index += 2
|
|
617
|
+
while index < len(lines) and lines[index].lstrip().startswith("|"):
|
|
618
|
+
cells = _cells(lines[index])
|
|
619
|
+
if max(old_col, new_col) < len(cells):
|
|
620
|
+
rest = [c for i, c in enumerate(cells) if i not in (old_col, new_col)]
|
|
621
|
+
text = f"{cells[old_col]} → {cells[new_col]} {' '.join(rest)}".strip()
|
|
622
|
+
statements.append(_Statement(text, section.start_line + index))
|
|
623
|
+
index += 1
|
|
624
|
+
continue
|
|
625
|
+
|
|
626
|
+
if _BULLET.match(line):
|
|
627
|
+
parts = [_BULLET.sub("", line, count=1)]
|
|
628
|
+
index += 1
|
|
629
|
+
while (
|
|
630
|
+
index < len(lines)
|
|
631
|
+
and lines[index].strip()
|
|
632
|
+
and not _BULLET.match(lines[index])
|
|
633
|
+
and lines[index].startswith((" ", "\t"))
|
|
634
|
+
):
|
|
635
|
+
parts.append(lines[index].strip())
|
|
636
|
+
index += 1
|
|
637
|
+
statements.append(_Statement(" ".join(parts), number))
|
|
638
|
+
continue
|
|
639
|
+
|
|
640
|
+
# A heading or a prose paragraph: a statement in its own right, and
|
|
641
|
+
# the frame the bullets and rows are read against.
|
|
642
|
+
parts = [line.strip().lstrip("#").strip()]
|
|
643
|
+
index += 1
|
|
644
|
+
while (
|
|
645
|
+
not line.lstrip().startswith("#")
|
|
646
|
+
and index < len(lines)
|
|
647
|
+
and lines[index].strip()
|
|
648
|
+
and not _BULLET.match(lines[index])
|
|
649
|
+
and not lines[index].lstrip().startswith(("|", "#", "```"))
|
|
650
|
+
):
|
|
651
|
+
parts.append(lines[index].strip())
|
|
652
|
+
index += 1
|
|
653
|
+
paragraph = " ".join(parts)
|
|
654
|
+
statements.append(_Statement(paragraph, number))
|
|
655
|
+
context.append(paragraph)
|
|
656
|
+
return statements, "\n".join(context)
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def _noun_kind(text: str, near: int | None = None) -> ChangeKind | None:
|
|
660
|
+
"""The construct a statement's nouns name; the noun nearest ``near`` wins."""
|
|
661
|
+
best: tuple[int, ChangeKind] | None = None
|
|
662
|
+
for pattern, kind in _NOUN_KINDS:
|
|
663
|
+
for match in pattern.finditer(text):
|
|
664
|
+
distance = abs(match.start() - near) if near is not None else 0
|
|
665
|
+
if best is None or distance < best[0]:
|
|
666
|
+
best = (distance, kind)
|
|
667
|
+
if near is None and best is not None:
|
|
668
|
+
kinds = {kind for pattern, kind in _NOUN_KINDS if pattern.search(text)}
|
|
669
|
+
return best[1] if len(kinds) == 1 else None
|
|
670
|
+
return best[1] if best else None
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
def _rename_kind(rename: _Rename, statement: str, context: str) -> tuple[ChangeKind, int, str]:
|
|
674
|
+
"""Decide what kind of rename one statement states.
|
|
675
|
+
|
|
676
|
+
The way the names are written is the strongest evidence (``x()`` is called,
|
|
677
|
+
``x=`` is passed); then a noun beside the old name ("the `x` keyword
|
|
678
|
+
argument"); then the scored signals of the statement; then the nouns of the
|
|
679
|
+
heading and prose around it.
|
|
680
|
+
"""
|
|
681
|
+
if rename.keyword:
|
|
682
|
+
return ChangeKind.KWARG_RENAME, _STRONG_SCORE, "written as a keyword argument (`name=`)"
|
|
683
|
+
if rename.call:
|
|
684
|
+
return ChangeKind.METHOD_RENAME, _STRONG_SCORE, "written as a call (`name()`)"
|
|
685
|
+
noun = _noun_kind(statement, near=rename.start)
|
|
686
|
+
if noun is not None:
|
|
687
|
+
return noun, _STRONG_SCORE, f"named as a {noun.value.split('_')[0]} in the statement"
|
|
688
|
+
kind, score, reason = classify(statement)
|
|
689
|
+
if kind.is_actionable and kind is not ChangeKind.PAGINATION_PAGE_TO_CURSOR:
|
|
690
|
+
return kind, score, reason
|
|
691
|
+
shape = _example_kind(rename.symbol, context)
|
|
692
|
+
if shape is not None:
|
|
693
|
+
return shape, _STRONG_SCORE, f"used as a {shape.value.split('_')[0]} in the examples"
|
|
694
|
+
noun = _noun_kind(context)
|
|
695
|
+
if noun is not None:
|
|
696
|
+
return noun, _STRONG_SCORE, f"named as a {noun.value.split('_')[0]} in the section"
|
|
697
|
+
return ChangeKind.UNKNOWN, 0, "a rename, but nothing says of what"
|
|
698
|
+
|
|
699
|
+
|
|
700
|
+
def _example_kind(symbol: str, text: str) -> ChangeKind | None:
|
|
701
|
+
"""How the old name is used in the document's code examples, if only one way.
|
|
702
|
+
|
|
703
|
+
``client.fetch_all(limit=10)`` shows a call; ``fetch(timeout_seconds=5)`` a
|
|
704
|
+
keyword; ``order["total"]`` a field. Two different uses decide nothing.
|
|
705
|
+
"""
|
|
706
|
+
escaped = re.escape(symbol)
|
|
707
|
+
uses = {
|
|
708
|
+
kind
|
|
709
|
+
for kind, pattern in (
|
|
710
|
+
(ChangeKind.METHOD_RENAME, rf"(?<![\w\[\"'])\.?{escaped}\s*\("),
|
|
711
|
+
(ChangeKind.KWARG_RENAME, rf"[(,]\s*{escaped}\s*=(?!=)"),
|
|
712
|
+
(ChangeKind.FIELD_RENAME, rf"\[\s*[\"']{escaped}[\"']\s*\]"),
|
|
713
|
+
)
|
|
714
|
+
if re.search(pattern, text)
|
|
715
|
+
}
|
|
716
|
+
return uses.pop() if len(uses) == 1 else None
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def _unsupported_label(text: str) -> str:
|
|
720
|
+
low = text.lower()
|
|
721
|
+
for pattern, _weight, label in _UNSUPPORTED_SIGNALS:
|
|
722
|
+
if re.search(pattern, low):
|
|
723
|
+
return label
|
|
724
|
+
return ""
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
def _extract_owner(text: str, symbol: str, kind: ChangeKind) -> tuple[str, bool]:
|
|
728
|
+
"""Find the object or function the symbol belongs to.
|
|
729
|
+
|
|
730
|
+
The owner is what keeps a rename scoped. Without it a field rename of
|
|
731
|
+
``total`` is a repository-wide search for the word "total"; with it, the
|
|
732
|
+
search is for ``order["total"]``.
|
|
733
|
+
|
|
734
|
+
Returns ``(owner, is_explicit)``. Phrasings that *assert* ownership -- "on
|
|
735
|
+
each `order` object", "the keyword argument on `fetch_orders`" -- are
|
|
736
|
+
explicit, and a receiver mismatch then refuses to patch. A receiver merely
|
|
737
|
+
scraped out of an illustrative snippet is not: the vendor writing
|
|
738
|
+
``client.fetch_orders(limit=10)`` is naming their own example variable, not
|
|
739
|
+
making a claim about what a downstream repository calls its client.
|
|
740
|
+
"""
|
|
741
|
+
if not symbol:
|
|
742
|
+
return "", False
|
|
743
|
+
escaped = re.escape(symbol)
|
|
744
|
+
|
|
745
|
+
if kind is ChangeKind.KWARG_RENAME:
|
|
746
|
+
# Which function an argument belongs to is definitional rather than a
|
|
747
|
+
# naming coincidence, so both spellings count as assertions.
|
|
748
|
+
match = re.search(rf"(\w+)\s*\([^)]*\b{escaped}\s*=", text)
|
|
749
|
+
if match:
|
|
750
|
+
return match.group(1), True
|
|
751
|
+
# "the `timeout_seconds` keyword argument on `fetch_orders`"
|
|
752
|
+
# "... of `Client.get()`" -- one function only; "of `get()` and `post()`"
|
|
753
|
+
# names two, and neither alone is the owner.
|
|
754
|
+
match = re.search(
|
|
755
|
+
rf"`{escaped}=?`[^`\n]{{0,60}}?\b(?:on|of|for)\s+(?:the\s+)?"
|
|
756
|
+
r"`(?:[\w.]*\.)?(\w+)(?:\(\))?`(?!\s*(?:,|and|or)\s*`)",
|
|
757
|
+
text,
|
|
758
|
+
re.IGNORECASE,
|
|
759
|
+
)
|
|
760
|
+
if match:
|
|
761
|
+
return match.group(1), True
|
|
762
|
+
return "", False
|
|
763
|
+
|
|
764
|
+
if kind is ChangeKind.METHOD_RENAME:
|
|
765
|
+
# "the method on `client` was renamed" asserts a receiver.
|
|
766
|
+
match = re.search(
|
|
767
|
+
rf"\bon\s+(?:the\s+)?`(\w+)`[^`\n]{{0,60}}?`?{escaped}`?", text, re.IGNORECASE
|
|
768
|
+
)
|
|
769
|
+
if match and match.group(1).lower() not in _STOPWORDS:
|
|
770
|
+
return match.group(1), True
|
|
771
|
+
match = re.search(
|
|
772
|
+
rf"`{escaped}(?:\(\))?`[^\n]{{0,60}}?\bon\s+(?:the\s+)?`(\w+)`", text, re.IGNORECASE
|
|
773
|
+
)
|
|
774
|
+
if match and match.group(1).lower() not in _STOPWORDS:
|
|
775
|
+
return match.group(1), True
|
|
776
|
+
# `client.fetch_orders(...)` in a Before/After example: a hint, not a
|
|
777
|
+
# constraint. See the docstring.
|
|
778
|
+
match = re.search(rf"`?(\w+)\.{escaped}\s*\(", text)
|
|
779
|
+
if match and match.group(1).lower() not in _STOPWORDS:
|
|
780
|
+
return match.group(1), False
|
|
781
|
+
return "", False
|
|
782
|
+
|
|
783
|
+
# Field rename. "on each `order` object" asserts which object owns the
|
|
784
|
+
# field; a bare `order["total"]` snippet only illustrates it.
|
|
785
|
+
for pattern in (
|
|
786
|
+
r"\bon\s+(?:(?:each|the|an|a|all)\s+)?`(\w+)`\s+objects?",
|
|
787
|
+
# An API reference's capitalized object name, written without backticks.
|
|
788
|
+
r"\bon\s+(?:(?:each|the|an|a|all)\s+)?([A-Z]\w*)\s+objects?",
|
|
789
|
+
rf"`(\w+)`\s+object[^.\n]{{0,60}}?`{escaped}`",
|
|
790
|
+
):
|
|
791
|
+
match = re.search(pattern, text, re.IGNORECASE)
|
|
792
|
+
if match and match.group(1).lower() not in _STOPWORDS:
|
|
793
|
+
return match.group(1), True
|
|
794
|
+
for pattern in (
|
|
795
|
+
rf"(\w+)\s*\[\s*[\"']{escaped}[\"']\s*\]",
|
|
796
|
+
rf"`(\w+)\.{escaped}`",
|
|
797
|
+
):
|
|
798
|
+
match = re.search(pattern, text, re.IGNORECASE)
|
|
799
|
+
if match and match.group(1).lower() not in _STOPWORDS:
|
|
800
|
+
return match.group(1), False
|
|
801
|
+
return "", False
|
|
802
|
+
|
|
803
|
+
|
|
804
|
+
def _extract_labeled(text: str, *labels: str) -> str:
|
|
805
|
+
"""Pull the value of a ``**Label:** value`` bullet."""
|
|
806
|
+
for match in _BOLD_FIELD.finditer(text):
|
|
807
|
+
label = match.group("label").strip().lower()
|
|
808
|
+
for wanted in labels:
|
|
809
|
+
if label.startswith(wanted.lower()):
|
|
810
|
+
return match.group("value").strip().rstrip(".")
|
|
811
|
+
return ""
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
def _collect_evidence(section: _Section, kind: ChangeKind, symbol: str) -> list[Evidence]:
|
|
815
|
+
"""Quote the lines that justify the classification, with line numbers."""
|
|
816
|
+
interesting = [
|
|
817
|
+
"renamed",
|
|
818
|
+
"removed",
|
|
819
|
+
"migration",
|
|
820
|
+
"breaking",
|
|
821
|
+
"deprecated",
|
|
822
|
+
"before",
|
|
823
|
+
"after",
|
|
824
|
+
"instead of",
|
|
825
|
+
]
|
|
826
|
+
if symbol:
|
|
827
|
+
interesting.append(symbol.lower())
|
|
828
|
+
if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
|
|
829
|
+
interesting.extend(["cursor", "total_pages", "has_more", "page"])
|
|
830
|
+
|
|
831
|
+
evidence: list[Evidence] = []
|
|
832
|
+
for offset, line in enumerate(section.text.splitlines()):
|
|
833
|
+
stripped = _plain(line.strip().lstrip("#-*> ")).strip()
|
|
834
|
+
if not stripped:
|
|
835
|
+
continue
|
|
836
|
+
low = stripped.lower()
|
|
837
|
+
matched = [word for word in interesting if word in low]
|
|
838
|
+
if not matched:
|
|
839
|
+
continue
|
|
840
|
+
evidence.append(
|
|
841
|
+
Evidence(
|
|
842
|
+
quote=stripped[:240],
|
|
843
|
+
line=section.start_line + offset,
|
|
844
|
+
note=f"mentions {matched[0]}",
|
|
845
|
+
)
|
|
846
|
+
)
|
|
847
|
+
if len(evidence) >= 5:
|
|
848
|
+
break
|
|
849
|
+
return evidence
|
|
850
|
+
|
|
851
|
+
|
|
852
|
+
def _pagination_contract(text: str) -> PaginationContract:
|
|
853
|
+
"""Read non-default pagination field names out of the notes."""
|
|
854
|
+
contract = PaginationContract()
|
|
855
|
+
# In document order, backticked names first: they are the vendor's literal
|
|
856
|
+
# spelling. The first candidate for each field wins. This used to iterate a
|
|
857
|
+
# `set`, so with several candidates the field chosen depended on the hash
|
|
858
|
+
# seed -- the same note could produce a different migration on each run.
|
|
859
|
+
identifiers = dict.fromkeys(
|
|
860
|
+
re.findall(r"`(\w+)`", text) + re.findall(r"\b(\w*(?:cursor|page|more)\w*)\b", text.lower())
|
|
861
|
+
)
|
|
862
|
+
assigned: set[str] = set()
|
|
863
|
+
for candidate in identifiers:
|
|
864
|
+
low = candidate.lower()
|
|
865
|
+
if low.endswith("_pages") or low == "total_pages":
|
|
866
|
+
key = "total_pages_key"
|
|
867
|
+
elif low.startswith("next") and "cursor" in low:
|
|
868
|
+
key = "next_cursor_key"
|
|
869
|
+
elif low in ("has_more", "hasmore", "more"):
|
|
870
|
+
key = "has_more_key"
|
|
871
|
+
else:
|
|
872
|
+
continue
|
|
873
|
+
if key not in assigned:
|
|
874
|
+
assigned.add(key)
|
|
875
|
+
setattr(contract, key, candidate)
|
|
876
|
+
return contract
|
|
877
|
+
|
|
878
|
+
|
|
879
|
+
def _severity(text: str) -> Severity:
|
|
880
|
+
match = _RISK.search(text)
|
|
881
|
+
if match:
|
|
882
|
+
return Severity.parse(match.group(1), Severity.MEDIUM)
|
|
883
|
+
return Severity.MEDIUM
|
|
884
|
+
|
|
885
|
+
|
|
886
|
+
def _confidence(kind: ChangeKind, score: int, target: SymbolTarget) -> Confidence:
|
|
887
|
+
"""Grade how sure we are that we read this document correctly.
|
|
888
|
+
|
|
889
|
+
High requires both a strong classification signal *and* successful symbol
|
|
890
|
+
extraction, because a rename we cannot name is a rename we cannot perform.
|
|
891
|
+
"""
|
|
892
|
+
if not kind.is_actionable:
|
|
893
|
+
return Confidence.LOW
|
|
894
|
+
if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
|
|
895
|
+
return Confidence.HIGH if score >= _STRONG_SCORE + 4 else Confidence.MEDIUM
|
|
896
|
+
if not target.is_rename:
|
|
897
|
+
return Confidence.LOW
|
|
898
|
+
if score >= _STRONG_SCORE and target.owner:
|
|
899
|
+
return Confidence.HIGH
|
|
900
|
+
if score >= _STRONG_SCORE:
|
|
901
|
+
return Confidence.MEDIUM
|
|
902
|
+
return Confidence.LOW
|
|
903
|
+
|
|
904
|
+
|
|
905
|
+
def parse_section(section: _Section, source: str = "markdown") -> list[BreakingChange]:
|
|
906
|
+
"""Turn one document section into the breaking changes it states."""
|
|
907
|
+
kind, score, reason = classify(section.text)
|
|
908
|
+
if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
|
|
909
|
+
return [_pagination_change(section, score, reason, source)]
|
|
910
|
+
|
|
911
|
+
statements, context = _statements(section)
|
|
912
|
+
renames: list[tuple[_Rename, _Statement]] = []
|
|
913
|
+
seen: set[tuple[str, str]] = set()
|
|
914
|
+
unsupported: list[tuple[_Statement, str]] = []
|
|
915
|
+
for statement in statements:
|
|
916
|
+
found = _find_renames(statement.text)
|
|
917
|
+
for rename in found:
|
|
918
|
+
key = (rename.old.lstrip("."), rename.new.lstrip("."))
|
|
919
|
+
if key not in seen:
|
|
920
|
+
seen.add(key)
|
|
921
|
+
renames.append((rename, statement))
|
|
922
|
+
label = "" if found else _unsupported_label(statement.text)
|
|
923
|
+
if label:
|
|
924
|
+
unsupported.append((statement, label))
|
|
925
|
+
|
|
926
|
+
if len(renames) == 1 and not unsupported:
|
|
927
|
+
# One rename: the whole section describes it, so its Before/After
|
|
928
|
+
# examples and prose all count as evidence for owner and kind.
|
|
929
|
+
rename, statement = renames[0]
|
|
930
|
+
return [_rename_change(rename, statement.text, section.text, section, source)]
|
|
931
|
+
if renames or len(unsupported) > 1:
|
|
932
|
+
changes = [
|
|
933
|
+
_rename_change(
|
|
934
|
+
rename, statement.text, f"{statement.text}\n{context}", section, source, statement
|
|
935
|
+
)
|
|
936
|
+
for rename, statement in renames
|
|
937
|
+
]
|
|
938
|
+
changes.extend(
|
|
939
|
+
_unsupported_change(statement, label, section, source)
|
|
940
|
+
for statement, label in unsupported
|
|
941
|
+
)
|
|
942
|
+
return changes
|
|
943
|
+
return [_section_change(section, kind, score, reason, source)]
|
|
944
|
+
|
|
945
|
+
|
|
946
|
+
def _rename_change(
|
|
947
|
+
rename: _Rename,
|
|
948
|
+
statement: str,
|
|
949
|
+
local: str,
|
|
950
|
+
section: _Section,
|
|
951
|
+
source: str,
|
|
952
|
+
located: _Statement | None = None,
|
|
953
|
+
) -> BreakingChange:
|
|
954
|
+
"""A change for one rename statement, read against ``local`` for its owner."""
|
|
955
|
+
title = section.title if located is None else _plain(statement)
|
|
956
|
+
evidence = (
|
|
957
|
+
_collect_evidence(section, ChangeKind.UNKNOWN, rename.symbol)
|
|
958
|
+
if located is None
|
|
959
|
+
else [Evidence(quote=_plain(statement)[:240], line=located.line, note="states the rename")]
|
|
960
|
+
)
|
|
961
|
+
if rename.is_move:
|
|
962
|
+
return BreakingChange(
|
|
963
|
+
title=title,
|
|
964
|
+
kind=ChangeKind.UNSUPPORTED,
|
|
965
|
+
severity=_severity(section.text),
|
|
966
|
+
confidence=Confidence.LOW,
|
|
967
|
+
evidence=evidence,
|
|
968
|
+
source=source,
|
|
969
|
+
classification_reason=(
|
|
970
|
+
f"`{rename.old}` -> `{rename.new}` keeps the name `{rename.symbol}` and "
|
|
971
|
+
f"changes where it lives -- a move to a different module or object, "
|
|
972
|
+
f"which PatchAhead v1 cannot migrate"
|
|
973
|
+
if rename.symbol == rename.replacement
|
|
974
|
+
else f"`{rename.old}` -> `{rename.new}` moves to a different module or "
|
|
975
|
+
f"object as well as changing its name; renaming it in place would "
|
|
976
|
+
f"write a name the old owner does not have"
|
|
977
|
+
),
|
|
978
|
+
)
|
|
979
|
+
|
|
980
|
+
context = local if located is None else local.split("\n", 1)[-1]
|
|
981
|
+
kind, score, reason = _rename_kind(rename, statement, context)
|
|
982
|
+
if kind is ChangeKind.UNKNOWN and located is None:
|
|
983
|
+
kind, score, reason = classify(section.text)
|
|
984
|
+
if kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
|
|
985
|
+
kind = ChangeKind.UNKNOWN
|
|
986
|
+
owner, owner_is_explicit = _extract_owner(local, rename.symbol, kind)
|
|
987
|
+
if rename.qualifier and kind is not ChangeKind.KWARG_RENAME:
|
|
988
|
+
# A dotted rename statement outranks anything scraped from prose.
|
|
989
|
+
owner, owner_is_explicit = rename.qualifier, True
|
|
990
|
+
target = SymbolTarget(
|
|
991
|
+
symbol=rename.symbol,
|
|
992
|
+
replacement=rename.replacement,
|
|
993
|
+
owner=owner,
|
|
994
|
+
owner_is_explicit=owner_is_explicit,
|
|
995
|
+
)
|
|
996
|
+
text = section.text if located is None else statement
|
|
997
|
+
return BreakingChange(
|
|
998
|
+
title=title,
|
|
999
|
+
kind=kind,
|
|
1000
|
+
target=target,
|
|
1001
|
+
old_behavior=_extract_labeled(text, "before", "old", "previously"),
|
|
1002
|
+
new_behavior=_extract_labeled(text, "after", "new", "now"),
|
|
1003
|
+
migration_hint=_extract_labeled(text, "migration", "action", "fix"),
|
|
1004
|
+
severity=_severity(section.text),
|
|
1005
|
+
confidence=_confidence(kind, score, target),
|
|
1006
|
+
evidence=evidence
|
|
1007
|
+
if located is not None
|
|
1008
|
+
else _collect_evidence(section, kind, rename.symbol),
|
|
1009
|
+
source=source,
|
|
1010
|
+
classification_reason=f"`{rename.old}` -> `{rename.new}`: {reason}",
|
|
1011
|
+
)
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def _unsupported_change(
|
|
1015
|
+
statement: _Statement, label: str, section: _Section, source: str
|
|
1016
|
+
) -> BreakingChange:
|
|
1017
|
+
return BreakingChange(
|
|
1018
|
+
title=_plain(statement.text),
|
|
1019
|
+
kind=ChangeKind.UNSUPPORTED,
|
|
1020
|
+
severity=_severity(section.text),
|
|
1021
|
+
confidence=Confidence.LOW,
|
|
1022
|
+
evidence=[Evidence(quote=_plain(statement.text)[:240], line=statement.line)],
|
|
1023
|
+
source=source,
|
|
1024
|
+
classification_reason=f"looks like a {label}, which PatchAhead v1 cannot migrate",
|
|
1025
|
+
)
|
|
1026
|
+
|
|
1027
|
+
|
|
1028
|
+
def _section_change(
|
|
1029
|
+
section: _Section, kind: ChangeKind, score: int, reason: str, source: str
|
|
1030
|
+
) -> BreakingChange:
|
|
1031
|
+
"""A section that states no rename: classified as a whole, as before."""
|
|
1032
|
+
if kind.is_actionable:
|
|
1033
|
+
reason = (
|
|
1034
|
+
f"{reason}; could not extract the old and new names, so no migration can be planned"
|
|
1035
|
+
)
|
|
1036
|
+
target = SymbolTarget()
|
|
1037
|
+
return BreakingChange(
|
|
1038
|
+
title=section.title,
|
|
1039
|
+
kind=kind,
|
|
1040
|
+
target=target,
|
|
1041
|
+
old_behavior=_extract_labeled(section.text, "before", "old", "previously"),
|
|
1042
|
+
new_behavior=_extract_labeled(section.text, "after", "new", "now"),
|
|
1043
|
+
migration_hint=_extract_labeled(section.text, "migration", "action", "fix"),
|
|
1044
|
+
severity=_severity(section.text),
|
|
1045
|
+
confidence=_confidence(kind, score, target),
|
|
1046
|
+
evidence=_collect_evidence(section, kind, ""),
|
|
1047
|
+
source=source,
|
|
1048
|
+
classification_reason=reason,
|
|
1049
|
+
)
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
def _pagination_change(section: _Section, score: int, reason: str, source: str) -> BreakingChange:
|
|
1053
|
+
"""A page-to-cursor change, if the note names the cursor it moves to.
|
|
1054
|
+
|
|
1055
|
+
The handler writes the cursor parameter and response fields into the
|
|
1056
|
+
repository. A note that describes cursor pagination without naming the
|
|
1057
|
+
parameter -- `starting_after`, `NextToken` -- would have PatchAhead write
|
|
1058
|
+
default names the vendor never used, so it is reported instead.
|
|
1059
|
+
"""
|
|
1060
|
+
names_cursor = re.search(r"`cursor=?`|\bcursor=", section.text)
|
|
1061
|
+
change = BreakingChange(
|
|
1062
|
+
title=section.title,
|
|
1063
|
+
kind=ChangeKind.PAGINATION_PAGE_TO_CURSOR if names_cursor else ChangeKind.UNSUPPORTED,
|
|
1064
|
+
old_behavior=_extract_labeled(section.text, "before", "old", "previously"),
|
|
1065
|
+
new_behavior=_extract_labeled(section.text, "after", "new", "now"),
|
|
1066
|
+
migration_hint=_extract_labeled(section.text, "migration", "action", "fix"),
|
|
1067
|
+
severity=_severity(section.text),
|
|
1068
|
+
evidence=_collect_evidence(section, ChangeKind.PAGINATION_PAGE_TO_CURSOR, ""),
|
|
1069
|
+
source=source,
|
|
1070
|
+
classification_reason=reason
|
|
1071
|
+
if names_cursor
|
|
1072
|
+
else (
|
|
1073
|
+
"a move to cursor-based pagination, but the note never names a `cursor` "
|
|
1074
|
+
"parameter, so the names PatchAhead would write are a guess"
|
|
1075
|
+
),
|
|
1076
|
+
)
|
|
1077
|
+
change.confidence = _confidence(change.kind, score, change.target)
|
|
1078
|
+
if names_cursor:
|
|
1079
|
+
change.pagination = _pagination_contract(section.text)
|
|
1080
|
+
return change
|
|
1081
|
+
|
|
1082
|
+
|
|
1083
|
+
def _plain(text: str) -> str:
|
|
1084
|
+
"""Statement text without Markdown emphasis, for titles and quotes."""
|
|
1085
|
+
return re.sub(r"\*\*|__", "", text).strip()
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
def _distinct(changes: list[BreakingChange]) -> list[BreakingChange]:
|
|
1089
|
+
"""Drop a rename the document states twice, keeping the first statement.
|
|
1090
|
+
|
|
1091
|
+
Release notes repeat themselves: a GitHub release and the project's
|
|
1092
|
+
changelog, both quoted in one Dependabot pull request, describe the same
|
|
1093
|
+
change. Patching it twice would report the second as "no impact".
|
|
1094
|
+
"""
|
|
1095
|
+
seen: set[tuple[str, str, str, str]] = set()
|
|
1096
|
+
kept: list[BreakingChange] = []
|
|
1097
|
+
for change in changes:
|
|
1098
|
+
target = change.target
|
|
1099
|
+
key = (change.kind.value, target.symbol, target.replacement, target.owner)
|
|
1100
|
+
if target.is_rename and key in seen:
|
|
1101
|
+
continue
|
|
1102
|
+
seen.add(key)
|
|
1103
|
+
kept.append(change)
|
|
1104
|
+
return kept
|
|
1105
|
+
|
|
1106
|
+
|
|
1107
|
+
class MarkdownChangeParser(ChangeParser):
|
|
1108
|
+
"""Parses Markdown, reStructuredText, and plain-text release notes."""
|
|
1109
|
+
|
|
1110
|
+
name = "markdown"
|
|
1111
|
+
suffixes = (".md", ".markdown", ".txt", ".rst", ".text", "")
|
|
1112
|
+
|
|
1113
|
+
def supports(self, document: ChangeDocument) -> bool:
|
|
1114
|
+
return document.suffix in self.suffixes
|
|
1115
|
+
|
|
1116
|
+
def parse(self, document: ChangeDocument) -> list[BreakingChange]:
|
|
1117
|
+
sections = _split_sections(document)
|
|
1118
|
+
changes = [
|
|
1119
|
+
change for section in sections for change in parse_section(section, source=self.name)
|
|
1120
|
+
]
|
|
1121
|
+
|
|
1122
|
+
# A multi-section document usually has prose sections that classify as
|
|
1123
|
+
# UNKNOWN. Drop those *only if* at least one real change was found, so a
|
|
1124
|
+
# document with nothing in it still reports honestly rather than
|
|
1125
|
+
# returning an empty list that reads like "no breaking changes".
|
|
1126
|
+
actionable = _distinct([c for c in changes if c.kind is not ChangeKind.UNKNOWN])
|
|
1127
|
+
result = actionable or changes[:1]
|
|
1128
|
+
log.debug(
|
|
1129
|
+
"parsed %s: %d section(s) -> %d change(s) [%s]",
|
|
1130
|
+
document.path,
|
|
1131
|
+
len(sections),
|
|
1132
|
+
len(result),
|
|
1133
|
+
", ".join(c.kind.value for c in result),
|
|
1134
|
+
)
|
|
1135
|
+
return result
|
|
1136
|
+
|
|
1137
|
+
|
|
1138
|
+
register(MarkdownChangeParser())
|