patchahead 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- patchahead/__init__.py +8 -0
- patchahead/analysis/__init__.py +52 -0
- patchahead/analysis/edits.py +143 -0
- patchahead/analysis/index.py +203 -0
- patchahead/analysis/python_ast.py +457 -0
- patchahead/apidiff/__init__.py +23 -0
- patchahead/apidiff/compare.py +366 -0
- patchahead/apidiff/download.py +95 -0
- patchahead/apidiff/surface.py +337 -0
- patchahead/ci.py +301 -0
- patchahead/cli.py +627 -0
- patchahead/config.py +284 -0
- patchahead/demo/__init__.py +256 -0
- patchahead/demo/fixtures/changes/field-rename.md +14 -0
- patchahead/demo/fixtures/changes/invoice-field-rename.md +21 -0
- patchahead/demo/fixtures/changes/kwarg-rename.md +14 -0
- patchahead/demo/fixtures/changes/method-rename.md +12 -0
- patchahead/demo/fixtures/changes/pagination-cursor.json +24 -0
- patchahead/demo/fixtures/changes/pagination-cursor.md +20 -0
- patchahead/demo/fixtures/changes/sdk-v2.md +31 -0
- patchahead/demo/fixtures/orders-service/README.md +51 -0
- patchahead/demo/fixtures/orders-service/app/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/app/client.py +15 -0
- patchahead/demo/fixtures/orders-service/app/models.py +10 -0
- patchahead/demo/fixtures/orders-service/app/order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/app/order_sync.py +21 -0
- patchahead/demo/fixtures/orders-service/conftest.py +6 -0
- patchahead/demo/fixtures/orders-service/pyproject.toml +16 -0
- patchahead/demo/fixtures/orders-service/tests/test_client.py +14 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_sync.py +11 -0
- patchahead/demo/fixtures/orders-service/upstream/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v1.py +34 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v2.py +56 -0
- patchahead/demo/serve.py +189 -0
- patchahead/domain/__init__.py +67 -0
- patchahead/domain/change.py +269 -0
- patchahead/domain/completeness.py +91 -0
- patchahead/domain/impact.py +248 -0
- patchahead/domain/patch.py +81 -0
- patchahead/domain/plan.py +170 -0
- patchahead/domain/result.py +210 -0
- patchahead/domain/validation.py +200 -0
- patchahead/engine.py +609 -0
- patchahead/handlers/__init__.py +35 -0
- patchahead/handlers/base.py +211 -0
- patchahead/handlers/field_rename.py +425 -0
- patchahead/handlers/kwarg_rename.py +201 -0
- patchahead/handlers/method_rename.py +608 -0
- patchahead/handlers/pagination.py +582 -0
- patchahead/ingest/__init__.py +32 -0
- patchahead/ingest/base.py +102 -0
- patchahead/ingest/markdown.py +1138 -0
- patchahead/ingest/structured.py +218 -0
- patchahead/llm/__init__.py +28 -0
- patchahead/llm/client.py +152 -0
- patchahead/llm/proposer.py +620 -0
- patchahead/observability.py +223 -0
- patchahead/reporting.py +451 -0
- patchahead/testing/__init__.py +22 -0
- patchahead/testing/discovery.py +113 -0
- patchahead/testing/runner.py +138 -0
- patchahead/validation/__init__.py +5 -0
- patchahead/validation/completeness.py +265 -0
- patchahead/validation/engine.py +531 -0
- patchahead/web/__init__.py +13 -0
- patchahead/web/server.py +279 -0
- patchahead/web/static/index.html +650 -0
- patchahead/workspace.py +382 -0
- patchahead-0.3.0.dist-info/METADATA +368 -0
- patchahead-0.3.0.dist-info/RECORD +75 -0
- patchahead-0.3.0.dist-info/WHEEL +5 -0
- patchahead-0.3.0.dist-info/entry_points.txt +2 -0
- patchahead-0.3.0.dist-info/licenses/LICENSE +21 -0
- patchahead-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Finding the tests that exercise a given module.
|
|
2
|
+
|
|
3
|
+
Used for two things: telling a reviewer which tests cover the code being
|
|
4
|
+
changed, and giving the targeted-tests validation gate a narrow, fast subset to
|
|
5
|
+
run before the full suite.
|
|
6
|
+
|
|
7
|
+
The strategy is name-based and deliberately simple: a module ``app/client.py``
|
|
8
|
+
is matched to ``tests/test_client.py``, ``tests/app/test_client.py``,
|
|
9
|
+
``tests/test_app_client.py``, and so on. It does not trace imports, so it is
|
|
10
|
+
*best effort*. That is why the regression gate always runs the full suite as
|
|
11
|
+
well -- the targeted gate exists to fail fast and to tell a reviewer where to
|
|
12
|
+
look, not to replace running everything.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import logging
|
|
18
|
+
import shlex
|
|
19
|
+
from pathlib import PurePosixPath
|
|
20
|
+
|
|
21
|
+
from patchahead.analysis.index import RepoIndex, is_test_path
|
|
22
|
+
|
|
23
|
+
log = logging.getLogger(__name__)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def module_stem(path: str) -> str:
|
|
27
|
+
"""The importable stem of a module path: ``app/order_sync.py`` -> ``order_sync``."""
|
|
28
|
+
return PurePosixPath(path).stem
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _candidate_names(source_path: str) -> set[str]:
|
|
32
|
+
"""Test-module basenames that would conventionally cover ``source_path``."""
|
|
33
|
+
pure = PurePosixPath(source_path)
|
|
34
|
+
stem = pure.stem
|
|
35
|
+
if stem == "__init__":
|
|
36
|
+
stem = pure.parent.name or stem
|
|
37
|
+
|
|
38
|
+
names = {f"test_{stem}.py", f"{stem}_test.py"}
|
|
39
|
+
# tests/test_app_client.py for app/client.py
|
|
40
|
+
parts = [part for part in pure.parts[:-1] if part not in (".", "src")]
|
|
41
|
+
if parts:
|
|
42
|
+
names.add(f"test_{'_'.join(parts[-1:] + [stem])}.py")
|
|
43
|
+
names.add(f"test_{'_'.join(parts + [stem])}.py")
|
|
44
|
+
return names
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def tests_for_path(index: RepoIndex, source_path: str) -> list[str]:
|
|
48
|
+
"""Test modules that look like they cover ``source_path``.
|
|
49
|
+
|
|
50
|
+
Ordered by how specific the match is, so the most likely test comes first.
|
|
51
|
+
"""
|
|
52
|
+
if is_test_path(source_path):
|
|
53
|
+
return []
|
|
54
|
+
|
|
55
|
+
candidates = _candidate_names(source_path)
|
|
56
|
+
stem = module_stem(source_path)
|
|
57
|
+
exact: list[str] = []
|
|
58
|
+
fuzzy: list[str] = []
|
|
59
|
+
|
|
60
|
+
for test_path in index.test_paths():
|
|
61
|
+
name = PurePosixPath(test_path).name
|
|
62
|
+
if name in candidates:
|
|
63
|
+
exact.append(test_path)
|
|
64
|
+
elif stem and stem in name:
|
|
65
|
+
fuzzy.append(test_path)
|
|
66
|
+
|
|
67
|
+
return exact + fuzzy
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def tests_for_paths(index: RepoIndex, source_paths: list[str]) -> list[str]:
|
|
71
|
+
"""The union of tests covering several modules, order-preserving.
|
|
72
|
+
|
|
73
|
+
Callers pass one path per finding, so the same module can arrive thousands
|
|
74
|
+
of times; each distinct module is matched once.
|
|
75
|
+
"""
|
|
76
|
+
found: dict[str, None] = {}
|
|
77
|
+
for source_path in dict.fromkeys(source_paths):
|
|
78
|
+
found.update(dict.fromkeys(tests_for_path(index, source_path)))
|
|
79
|
+
return list(found)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def tests_by_file(index: RepoIndex, source_paths: list[str]) -> dict[str, list[str]]:
|
|
83
|
+
"""A ``source path -> test paths`` mapping, for the impact graph."""
|
|
84
|
+
mapping: dict[str, list[str]] = {}
|
|
85
|
+
for source_path in dict.fromkeys(source_paths):
|
|
86
|
+
tests = tests_for_path(index, source_path)
|
|
87
|
+
if tests:
|
|
88
|
+
mapping[source_path] = tests
|
|
89
|
+
return mapping
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def has_any_tests(index: RepoIndex) -> bool:
|
|
93
|
+
return bool(index.test_paths())
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def scoped_command(test_command: str, test_paths: list[str]) -> str:
|
|
97
|
+
"""Build a command that runs only ``test_paths``.
|
|
98
|
+
|
|
99
|
+
Only attempted for pytest-shaped commands, where appending paths is a
|
|
100
|
+
well-defined way to narrow a run. For any other runner the full command is
|
|
101
|
+
returned unchanged and the caller reports the targeted gate as covering the
|
|
102
|
+
whole suite -- guessing at another runner's path syntax would produce a gate
|
|
103
|
+
that silently tests nothing.
|
|
104
|
+
"""
|
|
105
|
+
if not test_paths:
|
|
106
|
+
return test_command
|
|
107
|
+
if "pytest" not in test_command:
|
|
108
|
+
return test_command
|
|
109
|
+
# Refuse to append to a command that already has shell plumbing; appending
|
|
110
|
+
# after a pipe or redirect would change what the arguments apply to.
|
|
111
|
+
if any(token in test_command for token in ("|", ">", "<", "&&", ";")):
|
|
112
|
+
return test_command
|
|
113
|
+
return " ".join([test_command] + [shlex.quote(path) for path in test_paths])
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""Running a test command inside a workspace and structuring the result.
|
|
2
|
+
|
|
3
|
+
Parsing test output is inherently runner-specific. This module understands
|
|
4
|
+
pytest's summary lines well and degrades to "the command's exit code" for
|
|
5
|
+
anything else, which is the part that actually decides a gate.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
import re
|
|
12
|
+
import subprocess
|
|
13
|
+
import time
|
|
14
|
+
|
|
15
|
+
from patchahead.domain.validation import TestRun
|
|
16
|
+
from patchahead.workspace import Workspace
|
|
17
|
+
|
|
18
|
+
log = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
_FAILED_LINE = re.compile(r"^(?:FAILED|ERROR)\s+(\S+)", re.MULTILINE)
|
|
21
|
+
_SHORT_TALLY = re.compile(r"(\d+ (?:passed|failed|error|skipped)[^\n=]*)")
|
|
22
|
+
_ASSERTION = re.compile(r"^E\s+(\w*(?:Error|Exception|AssertionError).*)$", re.MULTILINE)
|
|
23
|
+
_NO_TESTS = re.compile(r"no tests ran|collected 0 items", re.IGNORECASE)
|
|
24
|
+
_MISSING_RUNNER = re.compile(
|
|
25
|
+
r"No module named (\S+)|(?:command )?not found|is not recognized as an internal"
|
|
26
|
+
)
|
|
27
|
+
#: POSIX exit codes for "the shell could not run this at all": 127 is
|
|
28
|
+
#: command-not-found, 126 is found-but-not-executable. Neither is a test
|
|
29
|
+
#: result, and neither depends on how a particular shell words its error --
|
|
30
|
+
#: bash says "command not found", dash says "not found", and a Windows shell
|
|
31
|
+
#: says something else again. Matching on the status rather than the prose is
|
|
32
|
+
#: what makes this reliable: relying on the wording meant `sh` reporting
|
|
33
|
+
#: "not found" was read as a test failure and reported as a regression.
|
|
34
|
+
CANNOT_START_CODES = (126, 127)
|
|
35
|
+
#: Appended wherever a gate has to explain that nothing ran. A message that
|
|
36
|
+
#: says only "could not run" leaves the reader with no next step.
|
|
37
|
+
RUNNER_ADVICE = (
|
|
38
|
+
"Check `test_command` in the repository's PatchAhead configuration, and "
|
|
39
|
+
"that the test runner is installed in the environment PatchAhead is "
|
|
40
|
+
"running in."
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def could_not_start(output: str, returncode: int) -> bool:
|
|
45
|
+
"""Whether the command never got as far as running a test.
|
|
46
|
+
|
|
47
|
+
Two independent signals, because either alone misses cases: the exit status
|
|
48
|
+
(127/126, which no test runner produces) and the shell's message (which
|
|
49
|
+
covers a runner that exits non-zero while reporting a missing module).
|
|
50
|
+
"""
|
|
51
|
+
if returncode in CANNOT_START_CODES:
|
|
52
|
+
return True
|
|
53
|
+
return returncode != 0 and bool(_MISSING_RUNNER.search(output))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _summarize(output: str, returncode: int) -> str:
|
|
57
|
+
"""One line describing what happened, for reports and logs."""
|
|
58
|
+
if could_not_start(output, returncode):
|
|
59
|
+
missing = _MISSING_RUNNER.search(output)
|
|
60
|
+
reason = missing.group(0).strip() if missing else f"exit status {returncode}"
|
|
61
|
+
return f"the test command could not start ({reason}). {RUNNER_ADVICE}"
|
|
62
|
+
if _NO_TESTS.search(output):
|
|
63
|
+
return "no tests ran"
|
|
64
|
+
tally = _SHORT_TALLY.search(output)
|
|
65
|
+
if returncode == 0:
|
|
66
|
+
return tally.group(1).strip() if tally else "tests passed"
|
|
67
|
+
assertion = _ASSERTION.search(output)
|
|
68
|
+
if assertion:
|
|
69
|
+
detail = assertion.group(1).strip()
|
|
70
|
+
return f"{tally.group(1).strip()} ({detail})" if tally else detail
|
|
71
|
+
if tally:
|
|
72
|
+
return tally.group(1).strip()
|
|
73
|
+
return f"test command exited {returncode}"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def run_tests(
|
|
77
|
+
workspace: Workspace,
|
|
78
|
+
command: str,
|
|
79
|
+
timeout: int = 300,
|
|
80
|
+
) -> TestRun:
|
|
81
|
+
"""Run ``command`` in ``workspace`` and return a structured result.
|
|
82
|
+
|
|
83
|
+
A command that cannot be executed at all -- not found, or killed by the
|
|
84
|
+
timeout -- is returned with ``errored=True`` rather than raising, so a gate
|
|
85
|
+
can report "could not verify" distinctly from "verified and failed". These
|
|
86
|
+
are different facts and the prototype's boolean could not tell them apart.
|
|
87
|
+
"""
|
|
88
|
+
start = time.perf_counter()
|
|
89
|
+
try:
|
|
90
|
+
completed = workspace.run(command, timeout=timeout)
|
|
91
|
+
except subprocess.TimeoutExpired:
|
|
92
|
+
duration = int((time.perf_counter() - start) * 1000)
|
|
93
|
+
log.warning("test command timed out after %ds: %s", timeout, command)
|
|
94
|
+
return TestRun(
|
|
95
|
+
command=command,
|
|
96
|
+
returncode=-1,
|
|
97
|
+
summary=f"test command timed out after {timeout}s",
|
|
98
|
+
duration_ms=duration,
|
|
99
|
+
errored=True,
|
|
100
|
+
)
|
|
101
|
+
except OSError as exc:
|
|
102
|
+
duration = int((time.perf_counter() - start) * 1000)
|
|
103
|
+
log.warning("test command could not be run: %s", exc)
|
|
104
|
+
return TestRun(
|
|
105
|
+
command=command,
|
|
106
|
+
returncode=-1,
|
|
107
|
+
stderr=str(exc),
|
|
108
|
+
summary=f"could not run the test command: {exc}",
|
|
109
|
+
duration_ms=duration,
|
|
110
|
+
errored=True,
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
duration = int((time.perf_counter() - start) * 1000)
|
|
114
|
+
output = completed.stdout + completed.stderr
|
|
115
|
+
failing = sorted(set(_FAILED_LINE.findall(output)))
|
|
116
|
+
|
|
117
|
+
# pytest exit code 5 is "no tests collected", which is not a test failure.
|
|
118
|
+
# Reporting it as one would make an empty repository look broken. A runner
|
|
119
|
+
# that could not start at all is also "could not verify", not "verified and
|
|
120
|
+
# failed" -- the gates treat those differently on purpose, and every gate
|
|
121
|
+
# that executes tests has to read it the same way or the same fact gets two
|
|
122
|
+
# different verdicts depending on which gate saw it.
|
|
123
|
+
errored = (completed.returncode == 5 and bool(_NO_TESTS.search(output))) or could_not_start(
|
|
124
|
+
output, completed.returncode
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
run = TestRun(
|
|
128
|
+
command=command,
|
|
129
|
+
returncode=completed.returncode,
|
|
130
|
+
stdout=completed.stdout,
|
|
131
|
+
stderr=completed.stderr,
|
|
132
|
+
failing_tests=failing,
|
|
133
|
+
summary=_summarize(output, completed.returncode),
|
|
134
|
+
duration_ms=duration,
|
|
135
|
+
errored=errored,
|
|
136
|
+
)
|
|
137
|
+
log.debug("test run finished in %dms: rc=%d, %s", duration, completed.returncode, run.summary)
|
|
138
|
+
return run
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
"""Scanning the patched copy for what is left of the old API.
|
|
2
|
+
|
|
3
|
+
Runs after the gates, against the workspace as patched. It is a report, not a
|
|
4
|
+
gate: a migration can be verified by its tests and still leave the old name
|
|
5
|
+
behind somewhere no test reaches, and saying where is the point. ``--require-
|
|
6
|
+
complete`` lets a CI run treat unfinished residuals as a failure.
|
|
7
|
+
|
|
8
|
+
Three sources, cheapest to read first:
|
|
9
|
+
|
|
10
|
+
1. **The handler's own analysis, re-run on the patched code.** Whatever it still
|
|
11
|
+
finds was not rewritten -- below the confidence threshold, an unfamiliar
|
|
12
|
+
shape, or another object that shares the name.
|
|
13
|
+
2. **Python tokens.** Strings and comments that mention the name, a string
|
|
14
|
+
handed to ``getattr`` (an access no static rewrite can see), tests that
|
|
15
|
+
still use the old name, and -- for a method rename -- any other use of the
|
|
16
|
+
name, such as the ``from sdk import fetch_orders`` that the call-site rewrite
|
|
17
|
+
leaves behind.
|
|
18
|
+
3. **Configuration and documentation files** that mention it.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import io
|
|
24
|
+
import logging
|
|
25
|
+
import re
|
|
26
|
+
import tokenize
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
|
|
29
|
+
from patchahead.analysis import index as repo_index
|
|
30
|
+
from patchahead.analysis.edits import read_source
|
|
31
|
+
from patchahead.config import Config
|
|
32
|
+
from patchahead.domain.change import BreakingChange, ChangeKind
|
|
33
|
+
from patchahead.domain.completeness import CompletenessReport, Residual, ResidualKind
|
|
34
|
+
from patchahead.domain.impact import ImpactReport
|
|
35
|
+
|
|
36
|
+
log = logging.getLogger(__name__)
|
|
37
|
+
|
|
38
|
+
#: Text files worth reading for mentions, by suffix.
|
|
39
|
+
TEXT_FILES: dict[str, ResidualKind] = {
|
|
40
|
+
".yaml": ResidualKind.CONFIG,
|
|
41
|
+
".yml": ResidualKind.CONFIG,
|
|
42
|
+
".json": ResidualKind.CONFIG,
|
|
43
|
+
".toml": ResidualKind.CONFIG,
|
|
44
|
+
".ini": ResidualKind.CONFIG,
|
|
45
|
+
".cfg": ResidualKind.CONFIG,
|
|
46
|
+
".env": ResidualKind.CONFIG,
|
|
47
|
+
".md": ResidualKind.DOCS,
|
|
48
|
+
".rst": ResidualKind.DOCS,
|
|
49
|
+
".txt": ResidualKind.DOCS,
|
|
50
|
+
}
|
|
51
|
+
#: A text file larger than this is skipped rather than read.
|
|
52
|
+
MAX_TEXT_BYTES = 1_000_000
|
|
53
|
+
_DYNAMIC_ACCESS = re.compile(r"\b(?:getattr|hasattr|setattr|delattr)\s*\(")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def scan(
|
|
57
|
+
change: BreakingChange,
|
|
58
|
+
remaining: ImpactReport,
|
|
59
|
+
index: repo_index.RepoIndex,
|
|
60
|
+
root: Path,
|
|
61
|
+
config: Config,
|
|
62
|
+
) -> CompletenessReport:
|
|
63
|
+
"""Every place ``change``'s old name survives in the patched tree at ``root``.
|
|
64
|
+
|
|
65
|
+
``remaining`` is the handler's analysis re-run on the patched code and
|
|
66
|
+
``index`` the patched tree's index.
|
|
67
|
+
"""
|
|
68
|
+
old, new = _names(change)
|
|
69
|
+
report = CompletenessReport(old=old, new=new)
|
|
70
|
+
seen: set[tuple[str, int]] = set()
|
|
71
|
+
|
|
72
|
+
for finding in remaining.findings:
|
|
73
|
+
if finding.other_object:
|
|
74
|
+
kind, reason = ResidualKind.OTHER_OBJECT, finding.unpatchable_reason
|
|
75
|
+
elif finding.patchable:
|
|
76
|
+
kind = ResidualKind.CODE
|
|
77
|
+
reason = (
|
|
78
|
+
f"confidence {finding.confidence.value} is below the "
|
|
79
|
+
f"`{config.min_confidence.value}` threshold"
|
|
80
|
+
)
|
|
81
|
+
else:
|
|
82
|
+
kind, reason = ResidualKind.CODE, finding.unpatchable_reason or finding.reason
|
|
83
|
+
_add(
|
|
84
|
+
report,
|
|
85
|
+
seen,
|
|
86
|
+
finding.path,
|
|
87
|
+
finding.reference.line,
|
|
88
|
+
kind,
|
|
89
|
+
finding.reference.snippet,
|
|
90
|
+
reason,
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
symbol = change.target.symbol
|
|
94
|
+
if not symbol:
|
|
95
|
+
return report
|
|
96
|
+
word = re.compile(rf"\b{re.escape(symbol)}\b")
|
|
97
|
+
|
|
98
|
+
for path in index.paths():
|
|
99
|
+
module = index.modules[path]
|
|
100
|
+
_scan_python(report, seen, change, path, module.source, word)
|
|
101
|
+
|
|
102
|
+
for path in _text_files(root, config):
|
|
103
|
+
try:
|
|
104
|
+
text = read_source(root / path)
|
|
105
|
+
except (OSError, UnicodeDecodeError):
|
|
106
|
+
continue
|
|
107
|
+
for number, line in enumerate(text.splitlines(), start=1):
|
|
108
|
+
if word.search(line):
|
|
109
|
+
kind = TEXT_FILES[Path(path).suffix.lower()]
|
|
110
|
+
_add(report, seen, path, number, kind, line.strip(), f"mentions `{symbol}`")
|
|
111
|
+
|
|
112
|
+
report.residuals.sort(key=lambda r: (list(ResidualKind).index(r.kind), r.path, r.line))
|
|
113
|
+
log.debug(
|
|
114
|
+
"completeness for `%s`: %d residual(s), %d unfinished",
|
|
115
|
+
old,
|
|
116
|
+
len(report.residuals),
|
|
117
|
+
len(report.unfinished),
|
|
118
|
+
)
|
|
119
|
+
return report
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _names(change: BreakingChange) -> tuple[str, str]:
|
|
123
|
+
if change.kind is ChangeKind.PAGINATION_PAGE_TO_CURSOR:
|
|
124
|
+
return change.pagination.page_param, change.pagination.cursor_param
|
|
125
|
+
return change.target.symbol, change.target.replacement
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _scan_python(
|
|
129
|
+
report: CompletenessReport,
|
|
130
|
+
seen: set[tuple[str, int]],
|
|
131
|
+
change: BreakingChange,
|
|
132
|
+
path: str,
|
|
133
|
+
source: str,
|
|
134
|
+
word: re.Pattern[str],
|
|
135
|
+
) -> None:
|
|
136
|
+
symbol = change.target.symbol
|
|
137
|
+
is_test = repo_index.is_test_path(path)
|
|
138
|
+
lines = source.splitlines()
|
|
139
|
+
try:
|
|
140
|
+
tokens = list(tokenize.generate_tokens(io.StringIO(source).readline))
|
|
141
|
+
except (tokenize.TokenError, SyntaxError):
|
|
142
|
+
return
|
|
143
|
+
|
|
144
|
+
for token in tokens:
|
|
145
|
+
line = token.start[0]
|
|
146
|
+
text = lines[line - 1].strip() if 0 < line <= len(lines) else ""
|
|
147
|
+
if token.type == tokenize.NAME and token.string == symbol:
|
|
148
|
+
if is_test:
|
|
149
|
+
_add(
|
|
150
|
+
report,
|
|
151
|
+
seen,
|
|
152
|
+
path,
|
|
153
|
+
line,
|
|
154
|
+
ResidualKind.TEST,
|
|
155
|
+
text,
|
|
156
|
+
"a test still uses the old name",
|
|
157
|
+
)
|
|
158
|
+
elif change.kind is ChangeKind.METHOD_RENAME:
|
|
159
|
+
# Field and keyword names are ordinary variable names too; a
|
|
160
|
+
# method's name appearing as a name is an import, a reference,
|
|
161
|
+
# or a definition -- each worth a look.
|
|
162
|
+
_add(report, seen, path, line, ResidualKind.CODE, text, _name_reason(text, symbol))
|
|
163
|
+
elif token.type == tokenize.STRING and word.search(token.string):
|
|
164
|
+
kind, reason = _string_residual(change, token, lines, text, is_test)
|
|
165
|
+
_add(report, seen, path, line, kind, text, reason)
|
|
166
|
+
elif token.type == tokenize.COMMENT and word.search(token.string):
|
|
167
|
+
_add(
|
|
168
|
+
report,
|
|
169
|
+
seen,
|
|
170
|
+
path,
|
|
171
|
+
line,
|
|
172
|
+
ResidualKind.COMMENT,
|
|
173
|
+
text,
|
|
174
|
+
f"a comment mentions `{symbol}`",
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _string_residual(
|
|
179
|
+
change: BreakingChange,
|
|
180
|
+
token: tokenize.TokenInfo,
|
|
181
|
+
lines: list[str],
|
|
182
|
+
text: str,
|
|
183
|
+
is_test: bool,
|
|
184
|
+
) -> tuple[ResidualKind, str]:
|
|
185
|
+
"""Classify a string literal that contains the old name."""
|
|
186
|
+
symbol = change.target.symbol
|
|
187
|
+
literal = token.string.strip("rbuRBUfF").strip("'\"")
|
|
188
|
+
if literal == symbol and _DYNAMIC_ACCESS.search(text):
|
|
189
|
+
return (
|
|
190
|
+
ResidualKind.DYNAMIC,
|
|
191
|
+
"the old name is passed as a string, which no static rewrite can follow",
|
|
192
|
+
)
|
|
193
|
+
single_line = token.start[0] == token.end[0]
|
|
194
|
+
if literal == symbol and change.kind is ChangeKind.FIELD_RENAME and single_line:
|
|
195
|
+
role = _key_role(lines[token.start[0] - 1], token.start[1], token.end[1])
|
|
196
|
+
if role:
|
|
197
|
+
return (
|
|
198
|
+
(ResidualKind.TEST, f"a test {role}")
|
|
199
|
+
if is_test
|
|
200
|
+
else (
|
|
201
|
+
ResidualKind.CODE,
|
|
202
|
+
f"code {role}",
|
|
203
|
+
)
|
|
204
|
+
)
|
|
205
|
+
return ResidualKind.STRING, f"a string mentions `{symbol}`"
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _key_role(line: str, start: int, end: int) -> str:
|
|
209
|
+
"""How a string literal equal to a field name is used, if as a key at all.
|
|
210
|
+
|
|
211
|
+
``x["total"]`` and ``x.get("total")`` read the field; ``{"total": 5}``
|
|
212
|
+
builds a payload or fixture with it. Any other string -- a label, a
|
|
213
|
+
message -- only mentions the word.
|
|
214
|
+
"""
|
|
215
|
+
before, after = line[:start].rstrip(), line[end:].lstrip()
|
|
216
|
+
if before.endswith("[") and after.startswith("]"):
|
|
217
|
+
return "reads the old field by key"
|
|
218
|
+
if before.endswith(".get("):
|
|
219
|
+
return "reads the old field with .get()"
|
|
220
|
+
if after.startswith(":") and before.endswith(("{", ",")):
|
|
221
|
+
return "builds a dictionary with the old field name"
|
|
222
|
+
return ""
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _name_reason(line: str, symbol: str) -> str:
|
|
226
|
+
if re.match(r"(?:from\s+\S+\s+)?import\b", line):
|
|
227
|
+
return "an import of the old name; call sites were rewritten, imports are not"
|
|
228
|
+
if re.match(rf"(?:async\s+)?def\s+{re.escape(symbol)}\b", line):
|
|
229
|
+
return "a definition with the old name in this repository"
|
|
230
|
+
return "a use of the old name that is not a call PatchAhead could rewrite"
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _add(
|
|
234
|
+
report: CompletenessReport,
|
|
235
|
+
seen: set[tuple[str, int]],
|
|
236
|
+
path: str,
|
|
237
|
+
line: int,
|
|
238
|
+
kind: ResidualKind,
|
|
239
|
+
snippet: str,
|
|
240
|
+
reason: str,
|
|
241
|
+
) -> None:
|
|
242
|
+
"""Record a residual once per line: the first, most specific reading wins."""
|
|
243
|
+
if (path, line) in seen:
|
|
244
|
+
return
|
|
245
|
+
seen.add((path, line))
|
|
246
|
+
report.residuals.append(
|
|
247
|
+
Residual(path=path, line=line, kind=kind, snippet=snippet[:200], reason=reason)
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _text_files(root: Path, config: Config) -> list[str]:
|
|
252
|
+
found: list[str] = []
|
|
253
|
+
for path in sorted(root.rglob("*")):
|
|
254
|
+
if path.suffix.lower() not in TEXT_FILES or not path.is_file():
|
|
255
|
+
continue
|
|
256
|
+
relative = path.relative_to(root).as_posix()
|
|
257
|
+
if repo_index.is_excluded(relative, config.exclude):
|
|
258
|
+
continue
|
|
259
|
+
try:
|
|
260
|
+
if path.stat().st_size > MAX_TEXT_BYTES:
|
|
261
|
+
continue
|
|
262
|
+
except OSError:
|
|
263
|
+
continue
|
|
264
|
+
found.append(relative)
|
|
265
|
+
return found
|