patchahead 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. patchahead/__init__.py +8 -0
  2. patchahead/analysis/__init__.py +52 -0
  3. patchahead/analysis/edits.py +143 -0
  4. patchahead/analysis/index.py +203 -0
  5. patchahead/analysis/python_ast.py +457 -0
  6. patchahead/apidiff/__init__.py +23 -0
  7. patchahead/apidiff/compare.py +366 -0
  8. patchahead/apidiff/download.py +95 -0
  9. patchahead/apidiff/surface.py +337 -0
  10. patchahead/ci.py +301 -0
  11. patchahead/cli.py +627 -0
  12. patchahead/config.py +284 -0
  13. patchahead/demo/__init__.py +256 -0
  14. patchahead/demo/fixtures/changes/field-rename.md +14 -0
  15. patchahead/demo/fixtures/changes/invoice-field-rename.md +21 -0
  16. patchahead/demo/fixtures/changes/kwarg-rename.md +14 -0
  17. patchahead/demo/fixtures/changes/method-rename.md +12 -0
  18. patchahead/demo/fixtures/changes/pagination-cursor.json +24 -0
  19. patchahead/demo/fixtures/changes/pagination-cursor.md +20 -0
  20. patchahead/demo/fixtures/changes/sdk-v2.md +31 -0
  21. patchahead/demo/fixtures/orders-service/README.md +51 -0
  22. patchahead/demo/fixtures/orders-service/app/__init__.py +0 -0
  23. patchahead/demo/fixtures/orders-service/app/client.py +15 -0
  24. patchahead/demo/fixtures/orders-service/app/models.py +10 -0
  25. patchahead/demo/fixtures/orders-service/app/order_report.py +24 -0
  26. patchahead/demo/fixtures/orders-service/app/order_sync.py +21 -0
  27. patchahead/demo/fixtures/orders-service/conftest.py +6 -0
  28. patchahead/demo/fixtures/orders-service/pyproject.toml +16 -0
  29. patchahead/demo/fixtures/orders-service/tests/test_client.py +14 -0
  30. patchahead/demo/fixtures/orders-service/tests/test_order_report.py +24 -0
  31. patchahead/demo/fixtures/orders-service/tests/test_order_sync.py +11 -0
  32. patchahead/demo/fixtures/orders-service/upstream/__init__.py +0 -0
  33. patchahead/demo/fixtures/orders-service/upstream/api_v1.py +34 -0
  34. patchahead/demo/fixtures/orders-service/upstream/api_v2.py +56 -0
  35. patchahead/demo/serve.py +189 -0
  36. patchahead/domain/__init__.py +67 -0
  37. patchahead/domain/change.py +269 -0
  38. patchahead/domain/completeness.py +91 -0
  39. patchahead/domain/impact.py +248 -0
  40. patchahead/domain/patch.py +81 -0
  41. patchahead/domain/plan.py +170 -0
  42. patchahead/domain/result.py +210 -0
  43. patchahead/domain/validation.py +200 -0
  44. patchahead/engine.py +609 -0
  45. patchahead/handlers/__init__.py +35 -0
  46. patchahead/handlers/base.py +211 -0
  47. patchahead/handlers/field_rename.py +425 -0
  48. patchahead/handlers/kwarg_rename.py +201 -0
  49. patchahead/handlers/method_rename.py +608 -0
  50. patchahead/handlers/pagination.py +582 -0
  51. patchahead/ingest/__init__.py +32 -0
  52. patchahead/ingest/base.py +102 -0
  53. patchahead/ingest/markdown.py +1138 -0
  54. patchahead/ingest/structured.py +218 -0
  55. patchahead/llm/__init__.py +28 -0
  56. patchahead/llm/client.py +152 -0
  57. patchahead/llm/proposer.py +620 -0
  58. patchahead/observability.py +223 -0
  59. patchahead/reporting.py +451 -0
  60. patchahead/testing/__init__.py +22 -0
  61. patchahead/testing/discovery.py +113 -0
  62. patchahead/testing/runner.py +138 -0
  63. patchahead/validation/__init__.py +5 -0
  64. patchahead/validation/completeness.py +265 -0
  65. patchahead/validation/engine.py +531 -0
  66. patchahead/web/__init__.py +13 -0
  67. patchahead/web/server.py +279 -0
  68. patchahead/web/static/index.html +650 -0
  69. patchahead/workspace.py +382 -0
  70. patchahead-0.3.0.dist-info/METADATA +368 -0
  71. patchahead-0.3.0.dist-info/RECORD +75 -0
  72. patchahead-0.3.0.dist-info/WHEEL +5 -0
  73. patchahead-0.3.0.dist-info/entry_points.txt +2 -0
  74. patchahead-0.3.0.dist-info/licenses/LICENSE +21 -0
  75. patchahead-0.3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,457 @@
1
+ """AST analysis of Python source.
2
+
3
+ The prototype matched source text with regular expressions, which produced more
4
+ false positives than true ones (see ``docs/assessment.md`` §2.2). This module
5
+ replaces that with a single pass over the syntax tree that records the four
6
+ construct types the v1 migration families care about:
7
+
8
+ * subscripts with a constant string key -- ``order["total"]``
9
+ * ``.get("key")`` calls -- ``order.get("total")``
10
+ * attribute access -- ``order.total``
11
+ * calls and their keyword arguments -- ``fetch(timeout_seconds=30)``
12
+
13
+ Each record carries an exact source range, so patching can edit that range
14
+ instead of regenerating the file. That is what keeps diffs minimal and leaves
15
+ comments and formatting untouched.
16
+
17
+ **Columns are converted from bytes to characters here.** ``ast`` reports
18
+ ``col_offset`` as an offset into the line's *UTF-8 encoding*, while Python
19
+ string slicing is by *character*. On a line containing any non-ASCII text the
20
+ two disagree, and slicing with a raw ``col_offset`` cuts the source at the wrong
21
+ place -- ``order["total"]`` after ``name = "Jos\u00e9"`` became ``order[""amount"``,
22
+ which is not even valid Python. Every ``SourceRange`` this module produces is
23
+ character-based, so no consumer downstream has to know that ``ast`` counts
24
+ differently.
25
+
26
+ The tree is walked once per file and the result is cached by
27
+ :class:`patchahead.analysis.index.RepoIndex`.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import ast
33
+ import re
34
+ from collections.abc import Iterator
35
+ from dataclasses import dataclass, field
36
+
37
+ #: Node types that introduce a new enclosing scope for symbol naming.
38
+ _SCOPE_NODES = (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)
39
+
40
+ MODULE_SCOPE = "<module>"
41
+
42
+
43
+ class ParseError(Exception):
44
+ """Raised when a file cannot be parsed as Python."""
45
+
46
+
47
+ class ColumnMap:
48
+ """Converts ``ast`` byte columns to Python string character columns.
49
+
50
+ ``ast`` reports columns as offsets into the UTF-8 encoding of the line;
51
+ ``str`` indexes by character. They coincide only while a line is pure ASCII,
52
+ so the conversion has to happen before any range is used to slice source.
53
+
54
+ Tables are built lazily and only for lines that actually contain non-ASCII
55
+ text, which is the overwhelming minority, so this costs nothing on a typical
56
+ file.
57
+ """
58
+
59
+ def __init__(self, source: str) -> None:
60
+ self._lines = source.splitlines()
61
+ self._tables: dict[int, dict[int, int]] = {}
62
+
63
+ def char_col(self, line: int, byte_col: int) -> int:
64
+ """The character offset for a 1-indexed line and a byte column."""
65
+ if byte_col <= 0:
66
+ return 0
67
+ if not (1 <= line <= len(self._lines)):
68
+ return byte_col
69
+ text = self._lines[line - 1]
70
+ if text.isascii():
71
+ return byte_col
72
+
73
+ table = self._tables.get(line)
74
+ if table is None:
75
+ table = {}
76
+ offset = 0
77
+ for index, char in enumerate(text):
78
+ table[offset] = index
79
+ offset += len(char.encode("utf-8"))
80
+ table[offset] = len(text)
81
+ self._tables[line] = table
82
+
83
+ if byte_col in table:
84
+ return table[byte_col]
85
+ # A column inside a multi-byte character should not happen for a real
86
+ # node boundary. Clamp to the nearest boundary at or below it rather
87
+ # than returning a position that would split the character.
88
+ candidates = [b for b in table if b <= byte_col]
89
+ return table[max(candidates)] if candidates else 0
90
+
91
+
92
+ #: Used where a range is built from source that is already character-indexed.
93
+ IDENTITY_COLUMNS = None
94
+
95
+
96
+ @dataclass(frozen=True)
97
+ class SourceRange:
98
+ """An exact half-open range in a source file.
99
+
100
+ ``line``/``end_line`` are 1-indexed; ``col``/``end_col`` are 0-indexed
101
+ **character** offsets and ``end_col`` is exclusive. Note the difference from
102
+ raw ``ast`` attributes, whose columns are UTF-8 byte offsets -- see
103
+ :class:`ColumnMap`.
104
+ """
105
+
106
+ line: int
107
+ col: int
108
+ end_line: int
109
+ end_col: int
110
+
111
+ @classmethod
112
+ def of(cls, node: ast.AST, columns: ColumnMap | None = None) -> SourceRange:
113
+ line = node.lineno
114
+ end_line = getattr(node, "end_lineno", None) or line
115
+ col = node.col_offset
116
+ end_col = getattr(node, "end_col_offset", None) or col
117
+ if columns is not None:
118
+ col = columns.char_col(line, col)
119
+ end_col = columns.char_col(end_line, end_col)
120
+ return cls(line=line, col=col, end_line=end_line, end_col=end_col)
121
+
122
+
123
+ def receiver_name(node: ast.AST) -> str:
124
+ """Best-effort dotted name of the expression a construct hangs off.
125
+
126
+ ``order`` for ``order["total"]``, ``self.cache`` for ``self.cache["total"]``,
127
+ ``get_orders()`` for ``get_orders()["total"]``, ``""`` when the receiver is
128
+ something we cannot name (a comprehension, a subscript chain, a literal).
129
+
130
+ This is the main signal for scoping a rename. It is a *name*, not a type:
131
+ PatchAhead does no type inference, and this function must not be read as
132
+ though it does. It narrows candidate sites; confidence grading handles the
133
+ residual uncertainty.
134
+ """
135
+ if isinstance(node, ast.Name):
136
+ return node.id
137
+ if isinstance(node, ast.Attribute):
138
+ base = receiver_name(node.value)
139
+ return f"{base}.{node.attr}" if base else node.attr
140
+ if isinstance(node, ast.Call):
141
+ base = receiver_name(node.func)
142
+ return f"{base}()" if base else ""
143
+ return ""
144
+
145
+
146
+ def base_name(dotted: str) -> str:
147
+ """The leftmost segment of a dotted receiver name (``a.b.c`` -> ``a``)."""
148
+ return dotted.split(".", 1)[0].removesuffix("()") if dotted else ""
149
+
150
+
151
+ def receiver_matches_owner(receiver: str, owner: str) -> bool:
152
+ """Whether a receiver expression names the object the change document names.
153
+
154
+ Matches on the leftmost or the rightmost segment, so both ``order`` and
155
+ ``self.order`` match an owner of ``order`` while ``customer`` and
156
+ ``self.cache`` do not. Case and naming style are ignored, so an owner of
157
+ ``Charge`` or ``PaymentIntent`` matches ``charge`` and ``payment_intent``,
158
+ and a qualifying prefix is allowed: ``api_client`` and ``self._client`` are
159
+ instances of ``Client``, while ``client_config`` is not.
160
+ Deliberately narrow otherwise: this is the predicate that decides whether a
161
+ rename is applied automatically, and a loose match here is exactly how
162
+ unrelated code gets corrupted.
163
+
164
+ An unnameable receiver -- a comprehension, a chained subscript -- yields
165
+ ``""`` and never matches, so it fails closed.
166
+ """
167
+ if not receiver or not owner:
168
+ return False
169
+ segments = [segment.removesuffix("()") for segment in receiver.split(".")]
170
+ # An API reference names the class (`Charge`, `PaymentIntent`); code names
171
+ # the instance (`charge`, `payment_intent`) -- often with a qualifier in
172
+ # front (`api_client`, `self._client` for a `Client`). Same object.
173
+ wanted = _snake(owner)
174
+
175
+ def names_owner(segment: str) -> bool:
176
+ name = _snake(segment).lstrip("_")
177
+ return name == wanted or name.endswith(f"_{wanted}")
178
+
179
+ return names_owner(segments[0]) or names_owner(segments[-1])
180
+
181
+
182
+ def _snake(name: str) -> str:
183
+ """``PaymentIntent`` -> ``payment_intent``; ``order`` is unchanged."""
184
+ return re.sub(r"(?<=[a-z0-9])(?=[A-Z])", "_", name).lower()
185
+
186
+
187
+ def iter_own_scope(node: ast.AST) -> Iterator[ast.AST]:
188
+ """Walk ``node`` without descending into nested function or class bodies.
189
+
190
+ ``ast.walk`` crosses scope boundaries, so a ``while`` loop written inside a
191
+ nested ``def`` is also found when walking the enclosing function. For the
192
+ pagination handler that meant the same loop was matched twice -- once as
193
+ ``outer`` and once as ``outer.inner`` -- producing two overlapping sets of
194
+ edits for one loop.
195
+
196
+ Lambdas and comprehensions are *not* skipped: they are expressions within
197
+ the enclosing statement, and a name they use is a real use of that name in
198
+ code the handler is reasoning about.
199
+ """
200
+ stack: list[ast.AST] = [node]
201
+ while stack:
202
+ current = stack.pop()
203
+ yield current
204
+ for child in ast.iter_child_nodes(current):
205
+ if child is not node and isinstance(child, _SCOPE_NODES):
206
+ continue
207
+ stack.append(child)
208
+
209
+
210
+ def called_name(node: ast.Call) -> str:
211
+ """The name being called: ``list_orders`` for ``client.list_orders(...)``."""
212
+ if isinstance(node.func, ast.Attribute):
213
+ return node.func.attr
214
+ if isinstance(node.func, ast.Name):
215
+ return node.func.id
216
+ return ""
217
+
218
+
219
+ @dataclass
220
+ class SubscriptAccess:
221
+ """``<receiver>["<key>"]`` with a constant string key."""
222
+
223
+ key: str
224
+ receiver: str
225
+ #: Range of the whole ``x["k"]`` expression.
226
+ range: SourceRange
227
+ #: Range of just the ``"k"`` literal, including its quotes.
228
+ key_range: SourceRange
229
+ symbol: str
230
+ #: True for a store or delete context (``x["k"] = v``), which a rename
231
+ #: must still handle but which reads differently in a report.
232
+ is_write: bool = False
233
+
234
+
235
+ @dataclass
236
+ class GetCallAccess:
237
+ """``<receiver>.get("<key>")`` or ``.get("<key>", default)``."""
238
+
239
+ key: str
240
+ receiver: str
241
+ range: SourceRange
242
+ key_range: SourceRange
243
+ symbol: str
244
+
245
+
246
+ @dataclass
247
+ class AttributeAccess:
248
+ """``<receiver>.<attr>`` outside of a call position."""
249
+
250
+ attr: str
251
+ receiver: str
252
+ range: SourceRange
253
+ #: Range of just the attribute name, excluding the dot.
254
+ attr_range: SourceRange
255
+ symbol: str
256
+ is_write: bool = False
257
+
258
+
259
+ @dataclass
260
+ class CallSite:
261
+ """A function or method call, with its keyword arguments."""
262
+
263
+ name: str
264
+ receiver: str
265
+ range: SourceRange
266
+ #: Range of just the callee name.
267
+ name_range: SourceRange
268
+ symbol: str
269
+ #: keyword name -> range of the ``name=`` token (name only, not the value).
270
+ keywords: dict[str, SourceRange] = field(default_factory=dict)
271
+ node: ast.Call | None = None
272
+
273
+
274
+ @dataclass
275
+ class ModuleAnalysis:
276
+ """Everything one pass over one module recorded."""
277
+
278
+ path: str
279
+ source: str
280
+ tree: ast.Module
281
+ #: Byte-to-character column converter for this module's source. Handlers
282
+ #: that build ranges from raw ``ast`` nodes must pass this to
283
+ #: :meth:`SourceRange.of`, or their columns will be wrong on any line
284
+ #: containing non-ASCII text.
285
+ columns: ColumnMap | None = None
286
+ subscripts: list[SubscriptAccess] = field(default_factory=list)
287
+ get_calls: list[GetCallAccess] = field(default_factory=list)
288
+ attributes: list[AttributeAccess] = field(default_factory=list)
289
+ calls: list[CallSite] = field(default_factory=list)
290
+
291
+ def line_text(self, line: int) -> str:
292
+ """The stripped text of a 1-indexed source line, or ``""``."""
293
+ lines = self.source.splitlines()
294
+ if 1 <= line <= len(lines):
295
+ return lines[line - 1].strip()
296
+ return ""
297
+
298
+
299
+ class _Collector(ast.NodeVisitor):
300
+ """Single-pass collector. Tracks the enclosing scope as it descends."""
301
+
302
+ def __init__(self, analysis: ModuleAnalysis, columns: ColumnMap) -> None:
303
+ self.analysis = analysis
304
+ self.columns = columns
305
+ self._scope: list[str] = []
306
+ # Attribute nodes that are a call's callee, so they are recorded as
307
+ # calls rather than double-counted as plain attribute reads.
308
+ self._callee_nodes: set[int] = set()
309
+
310
+ @property
311
+ def symbol(self) -> str:
312
+ return ".".join(self._scope) if self._scope else MODULE_SCOPE
313
+
314
+ def _enter(self, node: ast.AST) -> None:
315
+ self._scope.append(node.name) # type: ignore[attr-defined]
316
+ self.generic_visit(node)
317
+ self._scope.pop()
318
+
319
+ visit_FunctionDef = _enter
320
+ visit_AsyncFunctionDef = _enter
321
+ visit_ClassDef = _enter
322
+
323
+ def visit_Call(self, node: ast.Call) -> None:
324
+ name = called_name(node)
325
+ if name:
326
+ if isinstance(node.func, ast.Attribute):
327
+ self._callee_nodes.add(id(node.func))
328
+ name_range = _attribute_name_range(node.func, self.columns)
329
+ receiver = receiver_name(node.func.value)
330
+ else:
331
+ name_range = SourceRange.of(node.func, self.columns)
332
+ receiver = ""
333
+
334
+ keywords: dict[str, SourceRange] = {}
335
+ for keyword in node.keywords:
336
+ if keyword.arg is None: # `**kwargs`
337
+ continue
338
+ keywords[keyword.arg] = _keyword_name_range(keyword, self.columns)
339
+
340
+ self.analysis.calls.append(
341
+ CallSite(
342
+ name=name,
343
+ receiver=receiver,
344
+ range=SourceRange.of(node, self.columns),
345
+ name_range=name_range,
346
+ symbol=self.symbol,
347
+ keywords=keywords,
348
+ node=node,
349
+ )
350
+ )
351
+
352
+ # `x.get("k")` is a distinct access shape from a generic call.
353
+ if (
354
+ name == "get"
355
+ and isinstance(node.func, ast.Attribute)
356
+ and node.args
357
+ and isinstance(node.args[0], ast.Constant)
358
+ and isinstance(node.args[0].value, str)
359
+ ):
360
+ self.analysis.get_calls.append(
361
+ GetCallAccess(
362
+ key=node.args[0].value,
363
+ receiver=receiver,
364
+ range=SourceRange.of(node, self.columns),
365
+ key_range=SourceRange.of(node.args[0], self.columns),
366
+ symbol=self.symbol,
367
+ )
368
+ )
369
+ self.generic_visit(node)
370
+
371
+ def visit_Subscript(self, node: ast.Subscript) -> None:
372
+ key_node = node.slice
373
+ if isinstance(key_node, ast.Constant) and isinstance(key_node.value, str):
374
+ self.analysis.subscripts.append(
375
+ SubscriptAccess(
376
+ key=key_node.value,
377
+ receiver=receiver_name(node.value),
378
+ range=SourceRange.of(node, self.columns),
379
+ key_range=SourceRange.of(key_node, self.columns),
380
+ symbol=self.symbol,
381
+ is_write=not isinstance(node.ctx, ast.Load),
382
+ )
383
+ )
384
+ self.generic_visit(node)
385
+
386
+ def visit_Attribute(self, node: ast.Attribute) -> None:
387
+ if id(node) not in self._callee_nodes:
388
+ self.analysis.attributes.append(
389
+ AttributeAccess(
390
+ attr=node.attr,
391
+ receiver=receiver_name(node.value),
392
+ range=SourceRange.of(node, self.columns),
393
+ attr_range=_attribute_name_range(node, self.columns),
394
+ symbol=self.symbol,
395
+ is_write=not isinstance(node.ctx, ast.Load),
396
+ )
397
+ )
398
+ self.generic_visit(node)
399
+
400
+
401
+ def _attribute_name_range(node: ast.Attribute, columns: ColumnMap) -> SourceRange:
402
+ """Range of just the attribute name in ``value.attr``.
403
+
404
+ ``ast`` gives no direct position for the name, but the node ends exactly at
405
+ the end of the attribute, so the name occupies the final bytes of the node.
406
+ This holds even across a line continuation, because ``end_lineno`` is the
407
+ line the name is on.
408
+
409
+ The subtraction happens in byte space, because that is the space
410
+ ``end_col_offset`` and ``str.encode()`` are both in; only the result is
411
+ converted to a character column.
412
+ """
413
+ end_line = node.end_lineno or node.lineno
414
+ end_byte = node.end_col_offset or node.col_offset
415
+ start_byte = max(0, end_byte - len(node.attr.encode("utf-8")))
416
+ return SourceRange(
417
+ line=end_line,
418
+ col=columns.char_col(end_line, start_byte),
419
+ end_line=end_line,
420
+ end_col=columns.char_col(end_line, end_byte),
421
+ )
422
+
423
+
424
+ def _keyword_name_range(node: ast.keyword, columns: ColumnMap) -> SourceRange:
425
+ """Range of the ``name`` token in ``name=value``.
426
+
427
+ The keyword node starts at the name, so the name occupies the first bytes of
428
+ the node. As above, the arithmetic is done in byte space.
429
+ """
430
+ name = node.arg or ""
431
+ start_byte = node.col_offset
432
+ end_byte = start_byte + len(name.encode("utf-8"))
433
+ return SourceRange(
434
+ line=node.lineno,
435
+ col=columns.char_col(node.lineno, start_byte),
436
+ end_line=node.lineno,
437
+ end_col=columns.char_col(node.lineno, end_byte),
438
+ )
439
+
440
+
441
+ def analyze_source(source: str, path: str) -> ModuleAnalysis:
442
+ """Parse and collect from one module.
443
+
444
+ Raises :class:`ParseError` with the offending line for unparseable input,
445
+ so callers can report which file was skipped and why.
446
+ """
447
+ try:
448
+ tree = ast.parse(source, filename=path)
449
+ except SyntaxError as exc:
450
+ raise ParseError(f"{path}:{exc.lineno or '?'}: {exc.msg}") from exc
451
+ except ValueError as exc: # e.g. source containing null bytes
452
+ raise ParseError(f"{path}: {exc}") from exc
453
+
454
+ analysis = ModuleAnalysis(path=path, source=source, tree=tree)
455
+ analysis.columns = ColumnMap(source)
456
+ _Collector(analysis, analysis.columns).visit(tree)
457
+ return analysis
@@ -0,0 +1,23 @@
1
+ """Breaking changes read from two versions of a library, not from its release notes.
2
+
3
+ ``surface.read`` parses a library's public API without importing it;
4
+ ``compare.compare`` turns two of those into :class:`BreakingChange` objects that
5
+ the rest of PatchAhead migrates like any other; ``download.fetch`` gets a published
6
+ version from PyPI as a wheel, which is unpacked and read -- never installed.
7
+ """
8
+
9
+ from patchahead.apidiff.compare import ApiDiff, compare
10
+ from patchahead.apidiff.download import ApiDiffError, fetch, unpack
11
+ from patchahead.apidiff.surface import Member, Param, Surface, read
12
+
13
+ __all__ = [
14
+ "ApiDiff",
15
+ "ApiDiffError",
16
+ "Member",
17
+ "Param",
18
+ "Surface",
19
+ "compare",
20
+ "fetch",
21
+ "read",
22
+ "unpack",
23
+ ]