patchahead 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- patchahead/__init__.py +8 -0
- patchahead/analysis/__init__.py +52 -0
- patchahead/analysis/edits.py +143 -0
- patchahead/analysis/index.py +203 -0
- patchahead/analysis/python_ast.py +457 -0
- patchahead/apidiff/__init__.py +23 -0
- patchahead/apidiff/compare.py +366 -0
- patchahead/apidiff/download.py +95 -0
- patchahead/apidiff/surface.py +337 -0
- patchahead/ci.py +301 -0
- patchahead/cli.py +627 -0
- patchahead/config.py +284 -0
- patchahead/demo/__init__.py +256 -0
- patchahead/demo/fixtures/changes/field-rename.md +14 -0
- patchahead/demo/fixtures/changes/invoice-field-rename.md +21 -0
- patchahead/demo/fixtures/changes/kwarg-rename.md +14 -0
- patchahead/demo/fixtures/changes/method-rename.md +12 -0
- patchahead/demo/fixtures/changes/pagination-cursor.json +24 -0
- patchahead/demo/fixtures/changes/pagination-cursor.md +20 -0
- patchahead/demo/fixtures/changes/sdk-v2.md +31 -0
- patchahead/demo/fixtures/orders-service/README.md +51 -0
- patchahead/demo/fixtures/orders-service/app/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/app/client.py +15 -0
- patchahead/demo/fixtures/orders-service/app/models.py +10 -0
- patchahead/demo/fixtures/orders-service/app/order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/app/order_sync.py +21 -0
- patchahead/demo/fixtures/orders-service/conftest.py +6 -0
- patchahead/demo/fixtures/orders-service/pyproject.toml +16 -0
- patchahead/demo/fixtures/orders-service/tests/test_client.py +14 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_sync.py +11 -0
- patchahead/demo/fixtures/orders-service/upstream/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v1.py +34 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v2.py +56 -0
- patchahead/demo/serve.py +189 -0
- patchahead/domain/__init__.py +67 -0
- patchahead/domain/change.py +269 -0
- patchahead/domain/completeness.py +91 -0
- patchahead/domain/impact.py +248 -0
- patchahead/domain/patch.py +81 -0
- patchahead/domain/plan.py +170 -0
- patchahead/domain/result.py +210 -0
- patchahead/domain/validation.py +200 -0
- patchahead/engine.py +609 -0
- patchahead/handlers/__init__.py +35 -0
- patchahead/handlers/base.py +211 -0
- patchahead/handlers/field_rename.py +425 -0
- patchahead/handlers/kwarg_rename.py +201 -0
- patchahead/handlers/method_rename.py +608 -0
- patchahead/handlers/pagination.py +582 -0
- patchahead/ingest/__init__.py +32 -0
- patchahead/ingest/base.py +102 -0
- patchahead/ingest/markdown.py +1138 -0
- patchahead/ingest/structured.py +218 -0
- patchahead/llm/__init__.py +28 -0
- patchahead/llm/client.py +152 -0
- patchahead/llm/proposer.py +620 -0
- patchahead/observability.py +223 -0
- patchahead/reporting.py +451 -0
- patchahead/testing/__init__.py +22 -0
- patchahead/testing/discovery.py +113 -0
- patchahead/testing/runner.py +138 -0
- patchahead/validation/__init__.py +5 -0
- patchahead/validation/completeness.py +265 -0
- patchahead/validation/engine.py +531 -0
- patchahead/web/__init__.py +13 -0
- patchahead/web/server.py +279 -0
- patchahead/web/static/index.html +650 -0
- patchahead/workspace.py +382 -0
- patchahead-0.3.0.dist-info/METADATA +368 -0
- patchahead-0.3.0.dist-info/RECORD +75 -0
- patchahead-0.3.0.dist-info/WHEEL +5 -0
- patchahead-0.3.0.dist-info/entry_points.txt +2 -0
- patchahead-0.3.0.dist-info/licenses/LICENSE +21 -0
- patchahead-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,457 @@
|
|
|
1
|
+
"""AST analysis of Python source.
|
|
2
|
+
|
|
3
|
+
The prototype matched source text with regular expressions, which produced more
|
|
4
|
+
false positives than true ones (see ``docs/assessment.md`` §2.2). This module
|
|
5
|
+
replaces that with a single pass over the syntax tree that records the four
|
|
6
|
+
construct types the v1 migration families care about:
|
|
7
|
+
|
|
8
|
+
* subscripts with a constant string key -- ``order["total"]``
|
|
9
|
+
* ``.get("key")`` calls -- ``order.get("total")``
|
|
10
|
+
* attribute access -- ``order.total``
|
|
11
|
+
* calls and their keyword arguments -- ``fetch(timeout_seconds=30)``
|
|
12
|
+
|
|
13
|
+
Each record carries an exact source range, so patching can edit that range
|
|
14
|
+
instead of regenerating the file. That is what keeps diffs minimal and leaves
|
|
15
|
+
comments and formatting untouched.
|
|
16
|
+
|
|
17
|
+
**Columns are converted from bytes to characters here.** ``ast`` reports
|
|
18
|
+
``col_offset`` as an offset into the line's *UTF-8 encoding*, while Python
|
|
19
|
+
string slicing is by *character*. On a line containing any non-ASCII text the
|
|
20
|
+
two disagree, and slicing with a raw ``col_offset`` cuts the source at the wrong
|
|
21
|
+
place -- ``order["total"]`` after ``name = "Jos\u00e9"`` became ``order[""amount"``,
|
|
22
|
+
which is not even valid Python. Every ``SourceRange`` this module produces is
|
|
23
|
+
character-based, so no consumer downstream has to know that ``ast`` counts
|
|
24
|
+
differently.
|
|
25
|
+
|
|
26
|
+
The tree is walked once per file and the result is cached by
|
|
27
|
+
:class:`patchahead.analysis.index.RepoIndex`.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import ast
|
|
33
|
+
import re
|
|
34
|
+
from collections.abc import Iterator
|
|
35
|
+
from dataclasses import dataclass, field
|
|
36
|
+
|
|
37
|
+
#: Node types that introduce a new enclosing scope for symbol naming.
|
|
38
|
+
_SCOPE_NODES = (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)
|
|
39
|
+
|
|
40
|
+
MODULE_SCOPE = "<module>"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ParseError(Exception):
|
|
44
|
+
"""Raised when a file cannot be parsed as Python."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class ColumnMap:
|
|
48
|
+
"""Converts ``ast`` byte columns to Python string character columns.
|
|
49
|
+
|
|
50
|
+
``ast`` reports columns as offsets into the UTF-8 encoding of the line;
|
|
51
|
+
``str`` indexes by character. They coincide only while a line is pure ASCII,
|
|
52
|
+
so the conversion has to happen before any range is used to slice source.
|
|
53
|
+
|
|
54
|
+
Tables are built lazily and only for lines that actually contain non-ASCII
|
|
55
|
+
text, which is the overwhelming minority, so this costs nothing on a typical
|
|
56
|
+
file.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
def __init__(self, source: str) -> None:
|
|
60
|
+
self._lines = source.splitlines()
|
|
61
|
+
self._tables: dict[int, dict[int, int]] = {}
|
|
62
|
+
|
|
63
|
+
def char_col(self, line: int, byte_col: int) -> int:
|
|
64
|
+
"""The character offset for a 1-indexed line and a byte column."""
|
|
65
|
+
if byte_col <= 0:
|
|
66
|
+
return 0
|
|
67
|
+
if not (1 <= line <= len(self._lines)):
|
|
68
|
+
return byte_col
|
|
69
|
+
text = self._lines[line - 1]
|
|
70
|
+
if text.isascii():
|
|
71
|
+
return byte_col
|
|
72
|
+
|
|
73
|
+
table = self._tables.get(line)
|
|
74
|
+
if table is None:
|
|
75
|
+
table = {}
|
|
76
|
+
offset = 0
|
|
77
|
+
for index, char in enumerate(text):
|
|
78
|
+
table[offset] = index
|
|
79
|
+
offset += len(char.encode("utf-8"))
|
|
80
|
+
table[offset] = len(text)
|
|
81
|
+
self._tables[line] = table
|
|
82
|
+
|
|
83
|
+
if byte_col in table:
|
|
84
|
+
return table[byte_col]
|
|
85
|
+
# A column inside a multi-byte character should not happen for a real
|
|
86
|
+
# node boundary. Clamp to the nearest boundary at or below it rather
|
|
87
|
+
# than returning a position that would split the character.
|
|
88
|
+
candidates = [b for b in table if b <= byte_col]
|
|
89
|
+
return table[max(candidates)] if candidates else 0
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
#: Used where a range is built from source that is already character-indexed.
|
|
93
|
+
IDENTITY_COLUMNS = None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass(frozen=True)
|
|
97
|
+
class SourceRange:
|
|
98
|
+
"""An exact half-open range in a source file.
|
|
99
|
+
|
|
100
|
+
``line``/``end_line`` are 1-indexed; ``col``/``end_col`` are 0-indexed
|
|
101
|
+
**character** offsets and ``end_col`` is exclusive. Note the difference from
|
|
102
|
+
raw ``ast`` attributes, whose columns are UTF-8 byte offsets -- see
|
|
103
|
+
:class:`ColumnMap`.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
line: int
|
|
107
|
+
col: int
|
|
108
|
+
end_line: int
|
|
109
|
+
end_col: int
|
|
110
|
+
|
|
111
|
+
@classmethod
|
|
112
|
+
def of(cls, node: ast.AST, columns: ColumnMap | None = None) -> SourceRange:
|
|
113
|
+
line = node.lineno
|
|
114
|
+
end_line = getattr(node, "end_lineno", None) or line
|
|
115
|
+
col = node.col_offset
|
|
116
|
+
end_col = getattr(node, "end_col_offset", None) or col
|
|
117
|
+
if columns is not None:
|
|
118
|
+
col = columns.char_col(line, col)
|
|
119
|
+
end_col = columns.char_col(end_line, end_col)
|
|
120
|
+
return cls(line=line, col=col, end_line=end_line, end_col=end_col)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def receiver_name(node: ast.AST) -> str:
|
|
124
|
+
"""Best-effort dotted name of the expression a construct hangs off.
|
|
125
|
+
|
|
126
|
+
``order`` for ``order["total"]``, ``self.cache`` for ``self.cache["total"]``,
|
|
127
|
+
``get_orders()`` for ``get_orders()["total"]``, ``""`` when the receiver is
|
|
128
|
+
something we cannot name (a comprehension, a subscript chain, a literal).
|
|
129
|
+
|
|
130
|
+
This is the main signal for scoping a rename. It is a *name*, not a type:
|
|
131
|
+
PatchAhead does no type inference, and this function must not be read as
|
|
132
|
+
though it does. It narrows candidate sites; confidence grading handles the
|
|
133
|
+
residual uncertainty.
|
|
134
|
+
"""
|
|
135
|
+
if isinstance(node, ast.Name):
|
|
136
|
+
return node.id
|
|
137
|
+
if isinstance(node, ast.Attribute):
|
|
138
|
+
base = receiver_name(node.value)
|
|
139
|
+
return f"{base}.{node.attr}" if base else node.attr
|
|
140
|
+
if isinstance(node, ast.Call):
|
|
141
|
+
base = receiver_name(node.func)
|
|
142
|
+
return f"{base}()" if base else ""
|
|
143
|
+
return ""
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def base_name(dotted: str) -> str:
|
|
147
|
+
"""The leftmost segment of a dotted receiver name (``a.b.c`` -> ``a``)."""
|
|
148
|
+
return dotted.split(".", 1)[0].removesuffix("()") if dotted else ""
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def receiver_matches_owner(receiver: str, owner: str) -> bool:
|
|
152
|
+
"""Whether a receiver expression names the object the change document names.
|
|
153
|
+
|
|
154
|
+
Matches on the leftmost or the rightmost segment, so both ``order`` and
|
|
155
|
+
``self.order`` match an owner of ``order`` while ``customer`` and
|
|
156
|
+
``self.cache`` do not. Case and naming style are ignored, so an owner of
|
|
157
|
+
``Charge`` or ``PaymentIntent`` matches ``charge`` and ``payment_intent``,
|
|
158
|
+
and a qualifying prefix is allowed: ``api_client`` and ``self._client`` are
|
|
159
|
+
instances of ``Client``, while ``client_config`` is not.
|
|
160
|
+
Deliberately narrow otherwise: this is the predicate that decides whether a
|
|
161
|
+
rename is applied automatically, and a loose match here is exactly how
|
|
162
|
+
unrelated code gets corrupted.
|
|
163
|
+
|
|
164
|
+
An unnameable receiver -- a comprehension, a chained subscript -- yields
|
|
165
|
+
``""`` and never matches, so it fails closed.
|
|
166
|
+
"""
|
|
167
|
+
if not receiver or not owner:
|
|
168
|
+
return False
|
|
169
|
+
segments = [segment.removesuffix("()") for segment in receiver.split(".")]
|
|
170
|
+
# An API reference names the class (`Charge`, `PaymentIntent`); code names
|
|
171
|
+
# the instance (`charge`, `payment_intent`) -- often with a qualifier in
|
|
172
|
+
# front (`api_client`, `self._client` for a `Client`). Same object.
|
|
173
|
+
wanted = _snake(owner)
|
|
174
|
+
|
|
175
|
+
def names_owner(segment: str) -> bool:
|
|
176
|
+
name = _snake(segment).lstrip("_")
|
|
177
|
+
return name == wanted or name.endswith(f"_{wanted}")
|
|
178
|
+
|
|
179
|
+
return names_owner(segments[0]) or names_owner(segments[-1])
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _snake(name: str) -> str:
|
|
183
|
+
"""``PaymentIntent`` -> ``payment_intent``; ``order`` is unchanged."""
|
|
184
|
+
return re.sub(r"(?<=[a-z0-9])(?=[A-Z])", "_", name).lower()
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def iter_own_scope(node: ast.AST) -> Iterator[ast.AST]:
|
|
188
|
+
"""Walk ``node`` without descending into nested function or class bodies.
|
|
189
|
+
|
|
190
|
+
``ast.walk`` crosses scope boundaries, so a ``while`` loop written inside a
|
|
191
|
+
nested ``def`` is also found when walking the enclosing function. For the
|
|
192
|
+
pagination handler that meant the same loop was matched twice -- once as
|
|
193
|
+
``outer`` and once as ``outer.inner`` -- producing two overlapping sets of
|
|
194
|
+
edits for one loop.
|
|
195
|
+
|
|
196
|
+
Lambdas and comprehensions are *not* skipped: they are expressions within
|
|
197
|
+
the enclosing statement, and a name they use is a real use of that name in
|
|
198
|
+
code the handler is reasoning about.
|
|
199
|
+
"""
|
|
200
|
+
stack: list[ast.AST] = [node]
|
|
201
|
+
while stack:
|
|
202
|
+
current = stack.pop()
|
|
203
|
+
yield current
|
|
204
|
+
for child in ast.iter_child_nodes(current):
|
|
205
|
+
if child is not node and isinstance(child, _SCOPE_NODES):
|
|
206
|
+
continue
|
|
207
|
+
stack.append(child)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def called_name(node: ast.Call) -> str:
|
|
211
|
+
"""The name being called: ``list_orders`` for ``client.list_orders(...)``."""
|
|
212
|
+
if isinstance(node.func, ast.Attribute):
|
|
213
|
+
return node.func.attr
|
|
214
|
+
if isinstance(node.func, ast.Name):
|
|
215
|
+
return node.func.id
|
|
216
|
+
return ""
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
@dataclass
|
|
220
|
+
class SubscriptAccess:
|
|
221
|
+
"""``<receiver>["<key>"]`` with a constant string key."""
|
|
222
|
+
|
|
223
|
+
key: str
|
|
224
|
+
receiver: str
|
|
225
|
+
#: Range of the whole ``x["k"]`` expression.
|
|
226
|
+
range: SourceRange
|
|
227
|
+
#: Range of just the ``"k"`` literal, including its quotes.
|
|
228
|
+
key_range: SourceRange
|
|
229
|
+
symbol: str
|
|
230
|
+
#: True for a store or delete context (``x["k"] = v``), which a rename
|
|
231
|
+
#: must still handle but which reads differently in a report.
|
|
232
|
+
is_write: bool = False
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
@dataclass
|
|
236
|
+
class GetCallAccess:
|
|
237
|
+
"""``<receiver>.get("<key>")`` or ``.get("<key>", default)``."""
|
|
238
|
+
|
|
239
|
+
key: str
|
|
240
|
+
receiver: str
|
|
241
|
+
range: SourceRange
|
|
242
|
+
key_range: SourceRange
|
|
243
|
+
symbol: str
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
@dataclass
|
|
247
|
+
class AttributeAccess:
|
|
248
|
+
"""``<receiver>.<attr>`` outside of a call position."""
|
|
249
|
+
|
|
250
|
+
attr: str
|
|
251
|
+
receiver: str
|
|
252
|
+
range: SourceRange
|
|
253
|
+
#: Range of just the attribute name, excluding the dot.
|
|
254
|
+
attr_range: SourceRange
|
|
255
|
+
symbol: str
|
|
256
|
+
is_write: bool = False
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
@dataclass
|
|
260
|
+
class CallSite:
|
|
261
|
+
"""A function or method call, with its keyword arguments."""
|
|
262
|
+
|
|
263
|
+
name: str
|
|
264
|
+
receiver: str
|
|
265
|
+
range: SourceRange
|
|
266
|
+
#: Range of just the callee name.
|
|
267
|
+
name_range: SourceRange
|
|
268
|
+
symbol: str
|
|
269
|
+
#: keyword name -> range of the ``name=`` token (name only, not the value).
|
|
270
|
+
keywords: dict[str, SourceRange] = field(default_factory=dict)
|
|
271
|
+
node: ast.Call | None = None
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
@dataclass
|
|
275
|
+
class ModuleAnalysis:
|
|
276
|
+
"""Everything one pass over one module recorded."""
|
|
277
|
+
|
|
278
|
+
path: str
|
|
279
|
+
source: str
|
|
280
|
+
tree: ast.Module
|
|
281
|
+
#: Byte-to-character column converter for this module's source. Handlers
|
|
282
|
+
#: that build ranges from raw ``ast`` nodes must pass this to
|
|
283
|
+
#: :meth:`SourceRange.of`, or their columns will be wrong on any line
|
|
284
|
+
#: containing non-ASCII text.
|
|
285
|
+
columns: ColumnMap | None = None
|
|
286
|
+
subscripts: list[SubscriptAccess] = field(default_factory=list)
|
|
287
|
+
get_calls: list[GetCallAccess] = field(default_factory=list)
|
|
288
|
+
attributes: list[AttributeAccess] = field(default_factory=list)
|
|
289
|
+
calls: list[CallSite] = field(default_factory=list)
|
|
290
|
+
|
|
291
|
+
def line_text(self, line: int) -> str:
|
|
292
|
+
"""The stripped text of a 1-indexed source line, or ``""``."""
|
|
293
|
+
lines = self.source.splitlines()
|
|
294
|
+
if 1 <= line <= len(lines):
|
|
295
|
+
return lines[line - 1].strip()
|
|
296
|
+
return ""
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
class _Collector(ast.NodeVisitor):
|
|
300
|
+
"""Single-pass collector. Tracks the enclosing scope as it descends."""
|
|
301
|
+
|
|
302
|
+
def __init__(self, analysis: ModuleAnalysis, columns: ColumnMap) -> None:
|
|
303
|
+
self.analysis = analysis
|
|
304
|
+
self.columns = columns
|
|
305
|
+
self._scope: list[str] = []
|
|
306
|
+
# Attribute nodes that are a call's callee, so they are recorded as
|
|
307
|
+
# calls rather than double-counted as plain attribute reads.
|
|
308
|
+
self._callee_nodes: set[int] = set()
|
|
309
|
+
|
|
310
|
+
@property
|
|
311
|
+
def symbol(self) -> str:
|
|
312
|
+
return ".".join(self._scope) if self._scope else MODULE_SCOPE
|
|
313
|
+
|
|
314
|
+
def _enter(self, node: ast.AST) -> None:
|
|
315
|
+
self._scope.append(node.name) # type: ignore[attr-defined]
|
|
316
|
+
self.generic_visit(node)
|
|
317
|
+
self._scope.pop()
|
|
318
|
+
|
|
319
|
+
visit_FunctionDef = _enter
|
|
320
|
+
visit_AsyncFunctionDef = _enter
|
|
321
|
+
visit_ClassDef = _enter
|
|
322
|
+
|
|
323
|
+
def visit_Call(self, node: ast.Call) -> None:
|
|
324
|
+
name = called_name(node)
|
|
325
|
+
if name:
|
|
326
|
+
if isinstance(node.func, ast.Attribute):
|
|
327
|
+
self._callee_nodes.add(id(node.func))
|
|
328
|
+
name_range = _attribute_name_range(node.func, self.columns)
|
|
329
|
+
receiver = receiver_name(node.func.value)
|
|
330
|
+
else:
|
|
331
|
+
name_range = SourceRange.of(node.func, self.columns)
|
|
332
|
+
receiver = ""
|
|
333
|
+
|
|
334
|
+
keywords: dict[str, SourceRange] = {}
|
|
335
|
+
for keyword in node.keywords:
|
|
336
|
+
if keyword.arg is None: # `**kwargs`
|
|
337
|
+
continue
|
|
338
|
+
keywords[keyword.arg] = _keyword_name_range(keyword, self.columns)
|
|
339
|
+
|
|
340
|
+
self.analysis.calls.append(
|
|
341
|
+
CallSite(
|
|
342
|
+
name=name,
|
|
343
|
+
receiver=receiver,
|
|
344
|
+
range=SourceRange.of(node, self.columns),
|
|
345
|
+
name_range=name_range,
|
|
346
|
+
symbol=self.symbol,
|
|
347
|
+
keywords=keywords,
|
|
348
|
+
node=node,
|
|
349
|
+
)
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
# `x.get("k")` is a distinct access shape from a generic call.
|
|
353
|
+
if (
|
|
354
|
+
name == "get"
|
|
355
|
+
and isinstance(node.func, ast.Attribute)
|
|
356
|
+
and node.args
|
|
357
|
+
and isinstance(node.args[0], ast.Constant)
|
|
358
|
+
and isinstance(node.args[0].value, str)
|
|
359
|
+
):
|
|
360
|
+
self.analysis.get_calls.append(
|
|
361
|
+
GetCallAccess(
|
|
362
|
+
key=node.args[0].value,
|
|
363
|
+
receiver=receiver,
|
|
364
|
+
range=SourceRange.of(node, self.columns),
|
|
365
|
+
key_range=SourceRange.of(node.args[0], self.columns),
|
|
366
|
+
symbol=self.symbol,
|
|
367
|
+
)
|
|
368
|
+
)
|
|
369
|
+
self.generic_visit(node)
|
|
370
|
+
|
|
371
|
+
def visit_Subscript(self, node: ast.Subscript) -> None:
|
|
372
|
+
key_node = node.slice
|
|
373
|
+
if isinstance(key_node, ast.Constant) and isinstance(key_node.value, str):
|
|
374
|
+
self.analysis.subscripts.append(
|
|
375
|
+
SubscriptAccess(
|
|
376
|
+
key=key_node.value,
|
|
377
|
+
receiver=receiver_name(node.value),
|
|
378
|
+
range=SourceRange.of(node, self.columns),
|
|
379
|
+
key_range=SourceRange.of(key_node, self.columns),
|
|
380
|
+
symbol=self.symbol,
|
|
381
|
+
is_write=not isinstance(node.ctx, ast.Load),
|
|
382
|
+
)
|
|
383
|
+
)
|
|
384
|
+
self.generic_visit(node)
|
|
385
|
+
|
|
386
|
+
def visit_Attribute(self, node: ast.Attribute) -> None:
|
|
387
|
+
if id(node) not in self._callee_nodes:
|
|
388
|
+
self.analysis.attributes.append(
|
|
389
|
+
AttributeAccess(
|
|
390
|
+
attr=node.attr,
|
|
391
|
+
receiver=receiver_name(node.value),
|
|
392
|
+
range=SourceRange.of(node, self.columns),
|
|
393
|
+
attr_range=_attribute_name_range(node, self.columns),
|
|
394
|
+
symbol=self.symbol,
|
|
395
|
+
is_write=not isinstance(node.ctx, ast.Load),
|
|
396
|
+
)
|
|
397
|
+
)
|
|
398
|
+
self.generic_visit(node)
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def _attribute_name_range(node: ast.Attribute, columns: ColumnMap) -> SourceRange:
|
|
402
|
+
"""Range of just the attribute name in ``value.attr``.
|
|
403
|
+
|
|
404
|
+
``ast`` gives no direct position for the name, but the node ends exactly at
|
|
405
|
+
the end of the attribute, so the name occupies the final bytes of the node.
|
|
406
|
+
This holds even across a line continuation, because ``end_lineno`` is the
|
|
407
|
+
line the name is on.
|
|
408
|
+
|
|
409
|
+
The subtraction happens in byte space, because that is the space
|
|
410
|
+
``end_col_offset`` and ``str.encode()`` are both in; only the result is
|
|
411
|
+
converted to a character column.
|
|
412
|
+
"""
|
|
413
|
+
end_line = node.end_lineno or node.lineno
|
|
414
|
+
end_byte = node.end_col_offset or node.col_offset
|
|
415
|
+
start_byte = max(0, end_byte - len(node.attr.encode("utf-8")))
|
|
416
|
+
return SourceRange(
|
|
417
|
+
line=end_line,
|
|
418
|
+
col=columns.char_col(end_line, start_byte),
|
|
419
|
+
end_line=end_line,
|
|
420
|
+
end_col=columns.char_col(end_line, end_byte),
|
|
421
|
+
)
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _keyword_name_range(node: ast.keyword, columns: ColumnMap) -> SourceRange:
|
|
425
|
+
"""Range of the ``name`` token in ``name=value``.
|
|
426
|
+
|
|
427
|
+
The keyword node starts at the name, so the name occupies the first bytes of
|
|
428
|
+
the node. As above, the arithmetic is done in byte space.
|
|
429
|
+
"""
|
|
430
|
+
name = node.arg or ""
|
|
431
|
+
start_byte = node.col_offset
|
|
432
|
+
end_byte = start_byte + len(name.encode("utf-8"))
|
|
433
|
+
return SourceRange(
|
|
434
|
+
line=node.lineno,
|
|
435
|
+
col=columns.char_col(node.lineno, start_byte),
|
|
436
|
+
end_line=node.lineno,
|
|
437
|
+
end_col=columns.char_col(node.lineno, end_byte),
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def analyze_source(source: str, path: str) -> ModuleAnalysis:
|
|
442
|
+
"""Parse and collect from one module.
|
|
443
|
+
|
|
444
|
+
Raises :class:`ParseError` with the offending line for unparseable input,
|
|
445
|
+
so callers can report which file was skipped and why.
|
|
446
|
+
"""
|
|
447
|
+
try:
|
|
448
|
+
tree = ast.parse(source, filename=path)
|
|
449
|
+
except SyntaxError as exc:
|
|
450
|
+
raise ParseError(f"{path}:{exc.lineno or '?'}: {exc.msg}") from exc
|
|
451
|
+
except ValueError as exc: # e.g. source containing null bytes
|
|
452
|
+
raise ParseError(f"{path}: {exc}") from exc
|
|
453
|
+
|
|
454
|
+
analysis = ModuleAnalysis(path=path, source=source, tree=tree)
|
|
455
|
+
analysis.columns = ColumnMap(source)
|
|
456
|
+
_Collector(analysis, analysis.columns).visit(tree)
|
|
457
|
+
return analysis
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Breaking changes read from two versions of a library, not from its release notes.
|
|
2
|
+
|
|
3
|
+
``surface.read`` parses a library's public API without importing it;
|
|
4
|
+
``compare.compare`` turns two of those into :class:`BreakingChange` objects that
|
|
5
|
+
the rest of PatchAhead migrates like any other; ``download.fetch`` gets a published
|
|
6
|
+
version from PyPI as a wheel, which is unpacked and read -- never installed.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from patchahead.apidiff.compare import ApiDiff, compare
|
|
10
|
+
from patchahead.apidiff.download import ApiDiffError, fetch, unpack
|
|
11
|
+
from patchahead.apidiff.surface import Member, Param, Surface, read
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"ApiDiff",
|
|
15
|
+
"ApiDiffError",
|
|
16
|
+
"Member",
|
|
17
|
+
"Param",
|
|
18
|
+
"Surface",
|
|
19
|
+
"compare",
|
|
20
|
+
"fetch",
|
|
21
|
+
"read",
|
|
22
|
+
"unpack",
|
|
23
|
+
]
|