diffcone 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffcone/__init__.py +13 -0
- diffcone/cache.py +528 -0
- diffcone/check.py +399 -0
- diffcone/classify.py +255 -0
- diffcone/cli.py +886 -0
- diffcone/collect.py +846 -0
- diffcone/cython.py +757 -0
- diffcone/declarations.py +111 -0
- diffcone/discovery/__init__.py +121 -0
- diffcone/discovery/asv_static.py +494 -0
- diffcone/discovery/common.py +130 -0
- diffcone/discovery/pytest_static.py +2680 -0
- diffcone/evidence.py +738 -0
- diffcone/evidence_plan.py +1487 -0
- diffcone/execution.py +1681 -0
- diffcone/indexer/__init__.py +48 -0
- diffcone/indexer/core.py +286 -0
- diffcone/indexer/definitions.py +292 -0
- diffcone/indexer/dynamics.py +406 -0
- diffcone/indexer/facts.py +373 -0
- diffcone/indexer/literals.py +388 -0
- diffcone/indexer/references.py +895 -0
- diffcone/indexer/resolver.py +1001 -0
- diffcone/indexer/scopes.py +276 -0
- diffcone/indexer/state.py +83 -0
- diffcone/indexer/symbols.py +441 -0
- diffcone/indexer/syntax.py +173 -0
- diffcone/manifest.py +166 -0
- diffcone/model.py +191 -0
- diffcone/planner.py +1453 -0
- diffcone/report.py +256 -0
- diffcone/selection.py +133 -0
- diffcone/snapshot.py +739 -0
- diffcone/testing.py +200 -0
- diffcone-0.1.0.dist-info/METADATA +133 -0
- diffcone-0.1.0.dist-info/RECORD +39 -0
- diffcone-0.1.0.dist-info/WHEEL +4 -0
- diffcone-0.1.0.dist-info/entry_points.txt +3 -0
- diffcone-0.1.0.dist-info/licenses/LICENSE +21 -0
diffcone/check.py
ADDED
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
"""Check a plan against what a runner found (roadmap item 9).
|
|
2
|
+
|
|
3
|
+
JUnit XML from a full pytest run says which tests failed; every one must be
|
|
4
|
+
among the plan's selected targets, or it is a *miss*. A failure that a
|
|
5
|
+
baseline run (the nightly run at the evidence commit) also had is reported
|
|
6
|
+
as already failing instead: the change did not cause it.
|
|
7
|
+
|
|
8
|
+
Other selective runs are read the same way, the tests in their JUnit being
|
|
9
|
+
the tests they ran: diffcone's own run of the plan (its outcomes should agree
|
|
10
|
+
with the full run's) and another selector such as pytest-testmon, compared
|
|
11
|
+
on the same failures.
|
|
12
|
+
|
|
13
|
+
pytest's junitxml names a test by ``classname`` and ``name``, built from the
|
|
14
|
+
node ID (``_pytest.junitxml.mangle_test_address``): the file path with ``/``
|
|
15
|
+
as ``.`` and ``.py`` dropped, then the classes, joined by ``.``; ``name`` is
|
|
16
|
+
the function with its parameters. A collection error is a case whose
|
|
17
|
+
``classname`` is empty and whose ``name`` is the dotted file. This module
|
|
18
|
+
inverts that against the plan's targets (functions, parameters folded); a
|
|
19
|
+
case matching no target is reported as unmatched, and a failing one is a
|
|
20
|
+
miss (the plan did not know the test).
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import json
|
|
26
|
+
import xml.etree.ElementTree as ET
|
|
27
|
+
from collections import defaultdict
|
|
28
|
+
from dataclasses import dataclass, field
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
PASSED, FAILED, ERROR, SKIPPED = "passed", "failed", "error", "skipped"
|
|
32
|
+
BROKEN = frozenset({FAILED, ERROR})
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class CheckError(Exception):
|
|
36
|
+
"""An input that cannot be read."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True)
|
|
40
|
+
class Case:
|
|
41
|
+
classname: str
|
|
42
|
+
name: str
|
|
43
|
+
outcome: str
|
|
44
|
+
time: float = 0.0
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def key(self) -> str:
|
|
48
|
+
return f"{self.classname}::{self.name}" if self.classname else self.name
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def read_junit(path: str | Path) -> list[Case]:
|
|
52
|
+
"""Every test case in a JUnit XML file (pytest's, xunit1 or xunit2)."""
|
|
53
|
+
try:
|
|
54
|
+
root = ET.parse(path).getroot()
|
|
55
|
+
except (OSError, ET.ParseError) as exc:
|
|
56
|
+
raise CheckError(f"cannot read JUnit XML {path}: {exc}") from exc
|
|
57
|
+
cases = []
|
|
58
|
+
for node in root.iter("testcase"):
|
|
59
|
+
tags = {child.tag for child in node}
|
|
60
|
+
outcome = (
|
|
61
|
+
ERROR
|
|
62
|
+
if "error" in tags
|
|
63
|
+
else FAILED
|
|
64
|
+
if "failure" in tags
|
|
65
|
+
else SKIPPED
|
|
66
|
+
if "skipped" in tags
|
|
67
|
+
else PASSED
|
|
68
|
+
)
|
|
69
|
+
try:
|
|
70
|
+
seconds = float(node.get("time") or 0)
|
|
71
|
+
except ValueError:
|
|
72
|
+
seconds = 0.0
|
|
73
|
+
cases.append(Case(node.get("classname") or "", node.get("name") or "", outcome, seconds))
|
|
74
|
+
return cases
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _mangle(runner_id: str) -> tuple[str, str, str]:
|
|
78
|
+
"""(classname, name, dotted file) pytest's junitxml gives a node ID."""
|
|
79
|
+
path, _, _params = runner_id.partition("[")
|
|
80
|
+
names = path.split("::")
|
|
81
|
+
dotted = names[0].replace("/", ".")
|
|
82
|
+
if dotted.endswith(".py"):
|
|
83
|
+
dotted = dotted[: -len(".py")]
|
|
84
|
+
return ".".join([dotted, *names[1:-1]]), names[-1], dotted
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass
|
|
88
|
+
class _Targets:
|
|
89
|
+
"""The plan's pytest targets, findable from a JUnit case."""
|
|
90
|
+
|
|
91
|
+
selected: set[str]
|
|
92
|
+
all: set[str]
|
|
93
|
+
by_case: dict[tuple[str, str], list[str]]
|
|
94
|
+
by_file: dict[str, list[str]]
|
|
95
|
+
|
|
96
|
+
@classmethod
|
|
97
|
+
def from_plan(cls, plan: dict) -> _Targets:
|
|
98
|
+
selected: set[str] = set()
|
|
99
|
+
every: set[str] = set()
|
|
100
|
+
by_case: dict[tuple[str, str], list[str]] = defaultdict(list)
|
|
101
|
+
by_file: dict[str, list[str]] = defaultdict(list)
|
|
102
|
+
for key, chosen in (("selected_targets", True), ("unselected_targets", False)):
|
|
103
|
+
for target in plan.get(key, ()):
|
|
104
|
+
if target.get("runner") != "pytest":
|
|
105
|
+
continue
|
|
106
|
+
runner_id = target["runner_id"]
|
|
107
|
+
classname, name, dotted = _mangle(runner_id)
|
|
108
|
+
by_case[(classname, name)].append(runner_id)
|
|
109
|
+
by_file[dotted].append(runner_id)
|
|
110
|
+
every.add(runner_id)
|
|
111
|
+
if chosen:
|
|
112
|
+
selected.add(runner_id)
|
|
113
|
+
return cls(selected, every, dict(by_case), dict(by_file))
|
|
114
|
+
|
|
115
|
+
def match(self, case: Case) -> list[str]:
|
|
116
|
+
"""The targets a case is a run of; for a collection error, every
|
|
117
|
+
target of its file."""
|
|
118
|
+
if not case.classname:
|
|
119
|
+
return self.by_file.get(case.name, [])
|
|
120
|
+
return self.by_case.get((case.classname, case.name.partition("[")[0]), [])
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@dataclass(frozen=True)
|
|
124
|
+
class Failure:
|
|
125
|
+
test: str # a target, or an unmatched case's key
|
|
126
|
+
outcome: str
|
|
127
|
+
kind: str # test, collection (a file failed to collect) or unknown (no target)
|
|
128
|
+
selected: bool
|
|
129
|
+
already: bool # the baseline run failed it too
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# The ``--run`` name of diffcone's own selective run: its misses fail the check.
|
|
133
|
+
OWN_RUN = "diffcone"
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
@dataclass
|
|
137
|
+
class RunReport:
|
|
138
|
+
name: str
|
|
139
|
+
cases: int
|
|
140
|
+
time: float
|
|
141
|
+
ran: int # targets with at least one case in the run
|
|
142
|
+
misses: list[str] = field(default_factory=list) # new failures it did not run
|
|
143
|
+
disagreements: list[str] = field(default_factory=list) # ran, broke in only one run
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
@dataclass
|
|
147
|
+
class CheckReport:
|
|
148
|
+
plan_status: str
|
|
149
|
+
discovery_incomplete: bool
|
|
150
|
+
targets: int
|
|
151
|
+
selected: int
|
|
152
|
+
full_cases: int
|
|
153
|
+
full_time: float
|
|
154
|
+
selected_time: float # the full run's time for the selected targets' cases
|
|
155
|
+
unmatched: int # full-run cases matching no target
|
|
156
|
+
failures: list[Failure]
|
|
157
|
+
runs: list[RunReport]
|
|
158
|
+
|
|
159
|
+
@property
|
|
160
|
+
def misses(self) -> list[Failure]:
|
|
161
|
+
return [f for f in self.failures if not f.selected and not f.already]
|
|
162
|
+
|
|
163
|
+
@property
|
|
164
|
+
def not_run(self) -> list[str]:
|
|
165
|
+
"""New failures diffcone's own selective run (``--run diffcone=...``)
|
|
166
|
+
did not run: the plan may have selected them, but they were never
|
|
167
|
+
executed (collection disagreed, a fallback ran something else)."""
|
|
168
|
+
return [t for r in self.runs if r.name == OWN_RUN for t in r.misses]
|
|
169
|
+
|
|
170
|
+
@property
|
|
171
|
+
def ok(self) -> bool:
|
|
172
|
+
return not self.misses and not self.not_run
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _broken(cases: list[Case], targets: _Targets) -> dict[str, tuple[str, str]]:
|
|
176
|
+
"""Test -> (outcome, kind) for every broken case, by target where one
|
|
177
|
+
matches."""
|
|
178
|
+
out: dict[str, tuple[str, str]] = {}
|
|
179
|
+
for case in cases:
|
|
180
|
+
if case.outcome not in BROKEN:
|
|
181
|
+
continue
|
|
182
|
+
matched = targets.match(case)
|
|
183
|
+
if not case.classname and matched:
|
|
184
|
+
# A file that failed to collect: one failure for the file, caught
|
|
185
|
+
# when any of its targets is selected.
|
|
186
|
+
out.setdefault(case.name, (case.outcome, "collection"))
|
|
187
|
+
elif matched:
|
|
188
|
+
for target in matched:
|
|
189
|
+
out.setdefault(target, (case.outcome, "test"))
|
|
190
|
+
else:
|
|
191
|
+
out.setdefault(case.key, (case.outcome, "unknown"))
|
|
192
|
+
return out
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _ran(cases: list[Case], targets: _Targets) -> set[str]:
|
|
196
|
+
"""The targets a run ran (and the keys of cases matching none)."""
|
|
197
|
+
ran: set[str] = set()
|
|
198
|
+
for case in cases:
|
|
199
|
+
matched = targets.match(case)
|
|
200
|
+
if not matched:
|
|
201
|
+
ran.add(case.key)
|
|
202
|
+
elif case.classname:
|
|
203
|
+
ran.update(matched)
|
|
204
|
+
return ran
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def check(
|
|
208
|
+
plan: dict,
|
|
209
|
+
full: list[Case],
|
|
210
|
+
*,
|
|
211
|
+
baseline: list[Case] | None = None,
|
|
212
|
+
runs: dict[str, list[Case]] | None = None,
|
|
213
|
+
) -> CheckReport:
|
|
214
|
+
targets = _Targets.from_plan(plan)
|
|
215
|
+
broken = _broken(full, targets)
|
|
216
|
+
already = set(_broken(baseline, targets)) if baseline is not None else set()
|
|
217
|
+
|
|
218
|
+
def selected(test: str, kind: str) -> bool:
|
|
219
|
+
if kind == "collection":
|
|
220
|
+
return any(t in targets.selected for t in targets.by_file.get(test, ()))
|
|
221
|
+
return test in targets.selected
|
|
222
|
+
|
|
223
|
+
failures = [
|
|
224
|
+
Failure(test, outcome, kind, selected(test, kind), test in already)
|
|
225
|
+
for test, (outcome, kind) in sorted(broken.items())
|
|
226
|
+
]
|
|
227
|
+
new = {f.test for f in failures if not f.already}
|
|
228
|
+
run_reports = []
|
|
229
|
+
for name, cases in (runs or {}).items():
|
|
230
|
+
ran = _ran(cases, targets)
|
|
231
|
+
run_broken = _broken(cases, targets)
|
|
232
|
+
files_ran = {f for f, ts in targets.by_file.items() if ran.intersection(ts)}
|
|
233
|
+
missed = sorted(
|
|
234
|
+
f.test
|
|
235
|
+
for f in failures
|
|
236
|
+
if f.test in new and f.test not in (files_ran if f.kind == "collection" else ran)
|
|
237
|
+
)
|
|
238
|
+
# A test both runs ran that broke in only one: flaky, or dependent on
|
|
239
|
+
# what ran before it.
|
|
240
|
+
disagreements = sorted(
|
|
241
|
+
t for t in ran & targets.all if (t in run_broken) != (t in broken) and t not in already
|
|
242
|
+
)
|
|
243
|
+
run_reports.append(
|
|
244
|
+
RunReport(
|
|
245
|
+
name,
|
|
246
|
+
len(cases),
|
|
247
|
+
sum(c.time for c in cases),
|
|
248
|
+
len(ran & targets.all),
|
|
249
|
+
missed,
|
|
250
|
+
disagreements,
|
|
251
|
+
)
|
|
252
|
+
)
|
|
253
|
+
selected_time = sum(
|
|
254
|
+
c.time for c in full if c.classname and any(t in targets.selected for t in targets.match(c))
|
|
255
|
+
)
|
|
256
|
+
return CheckReport(
|
|
257
|
+
plan_status=plan.get("status", "complete"),
|
|
258
|
+
discovery_incomplete=bool(plan.get("discovery_incomplete")),
|
|
259
|
+
targets=len(targets.all),
|
|
260
|
+
selected=len(targets.selected),
|
|
261
|
+
full_cases=len(full),
|
|
262
|
+
full_time=sum(c.time for c in full),
|
|
263
|
+
selected_time=selected_time,
|
|
264
|
+
unmatched=sum(1 for c in full if not targets.match(c)),
|
|
265
|
+
failures=failures,
|
|
266
|
+
runs=run_reports,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def load_plan(path: str | Path) -> dict:
|
|
271
|
+
try:
|
|
272
|
+
data = json.loads(Path(path).read_text("utf-8"))
|
|
273
|
+
except (OSError, ValueError) as exc:
|
|
274
|
+
raise CheckError(f"cannot read plan {path}: {exc}") from exc
|
|
275
|
+
if not isinstance(data, dict) or "selected_targets" not in data:
|
|
276
|
+
raise CheckError(f"{path} is not a diffcone plan (JSON format)")
|
|
277
|
+
return data
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
# --------------------------------------------------------------------------- reports
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def to_dict(report: CheckReport) -> dict:
|
|
284
|
+
return {
|
|
285
|
+
"ok": report.ok,
|
|
286
|
+
"not_run": report.not_run,
|
|
287
|
+
"plan": {
|
|
288
|
+
"status": report.plan_status,
|
|
289
|
+
"discovery_incomplete": report.discovery_incomplete,
|
|
290
|
+
"targets": report.targets,
|
|
291
|
+
"selected": report.selected,
|
|
292
|
+
},
|
|
293
|
+
"full_run": {
|
|
294
|
+
"cases": report.full_cases,
|
|
295
|
+
"test_time": round(report.full_time, 2),
|
|
296
|
+
"selected_test_time": round(report.selected_time, 2),
|
|
297
|
+
"unmatched_cases": report.unmatched,
|
|
298
|
+
},
|
|
299
|
+
"failures": [
|
|
300
|
+
{
|
|
301
|
+
"test": f.test,
|
|
302
|
+
"outcome": f.outcome,
|
|
303
|
+
"kind": f.kind,
|
|
304
|
+
"selected": f.selected,
|
|
305
|
+
"already_failing": f.already,
|
|
306
|
+
}
|
|
307
|
+
for f in report.failures
|
|
308
|
+
],
|
|
309
|
+
"misses": [f.test for f in report.misses],
|
|
310
|
+
"runs": [
|
|
311
|
+
{
|
|
312
|
+
"name": r.name,
|
|
313
|
+
"cases": r.cases,
|
|
314
|
+
"test_time": round(r.time, 2),
|
|
315
|
+
"targets_run": r.ran,
|
|
316
|
+
"misses": r.misses,
|
|
317
|
+
"disagreements": r.disagreements,
|
|
318
|
+
}
|
|
319
|
+
for r in report.runs
|
|
320
|
+
],
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _share(part: float, whole: float) -> str:
|
|
325
|
+
return f"{100 * part / whole:.1f} %" if whole else "-"
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def to_text(report: CheckReport) -> str:
|
|
329
|
+
lines = [
|
|
330
|
+
f"plan: {report.selected} of {report.targets} targets selected "
|
|
331
|
+
f"({_share(report.selected, report.targets)}), status {report.plan_status}"
|
|
332
|
+
+ (", discovery may be incomplete" if report.discovery_incomplete else ""),
|
|
333
|
+
f"full run: {report.full_cases} cases, {report.full_time:.0f} s of test time; "
|
|
334
|
+
f"the selected targets took {report.selected_time:.0f} s "
|
|
335
|
+
f"({_share(report.selected_time, report.full_time)})",
|
|
336
|
+
]
|
|
337
|
+
if report.unmatched:
|
|
338
|
+
lines.append(f" {report.unmatched} cases match no target of the plan")
|
|
339
|
+
new = [f for f in report.failures if not f.already]
|
|
340
|
+
lines.append(
|
|
341
|
+
f"failures: {len(new)} new, {len(report.failures) - len(new)} already failing at "
|
|
342
|
+
f"the baseline; missed by the plan: {len(report.misses)}"
|
|
343
|
+
)
|
|
344
|
+
for f in report.misses:
|
|
345
|
+
lines.append(f" MISSED {f.test} ({f.outcome}, {f.kind})")
|
|
346
|
+
for r in report.runs:
|
|
347
|
+
lines.append(
|
|
348
|
+
f"run {r.name}: {r.cases} cases, {r.time:.0f} s of test time, "
|
|
349
|
+
f"{len(r.misses)} new failures not run, {len(r.disagreements)} outcome "
|
|
350
|
+
"disagreements with the full run"
|
|
351
|
+
)
|
|
352
|
+
lines.extend(f" NOT RUN {t}" for t in r.misses)
|
|
353
|
+
lines.extend(f" DISAGREES {t}" for t in r.disagreements)
|
|
354
|
+
if report.not_run:
|
|
355
|
+
lines.append(
|
|
356
|
+
f"diffcone's run did not run {len(report.not_run)} new failure(s) (see NOT RUN above)"
|
|
357
|
+
)
|
|
358
|
+
lines.append("OK: every new failure was selected and run" if report.ok else "MISS")
|
|
359
|
+
return "\n".join(lines) + "\n"
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def to_markdown(report: CheckReport) -> str:
|
|
363
|
+
new = [f for f in report.failures if not f.already]
|
|
364
|
+
missed = len({f.test for f in report.misses} | set(report.not_run))
|
|
365
|
+
verdict = "no miss" if report.ok else f"**{missed} missed**"
|
|
366
|
+
lines = [
|
|
367
|
+
f"### diffcone check: {verdict}",
|
|
368
|
+
"",
|
|
369
|
+
"| | tests | test time | new failures not covered |",
|
|
370
|
+
"|---|---|---|---|",
|
|
371
|
+
f"| full run | {report.full_cases} cases | {report.full_time:.0f} s | - |",
|
|
372
|
+
f"| diffcone plan | {report.selected} of {report.targets} targets "
|
|
373
|
+
f"({_share(report.selected, report.targets)}) | {report.selected_time:.0f} s "
|
|
374
|
+
f"({_share(report.selected_time, report.full_time)}) | {len(report.misses)} |",
|
|
375
|
+
]
|
|
376
|
+
for r in report.runs:
|
|
377
|
+
lines.append(
|
|
378
|
+
f"| {r.name} run | {r.cases} cases | {r.time:.0f} s "
|
|
379
|
+
f"({_share(r.time, report.full_time)}) | {len(r.misses)} |"
|
|
380
|
+
)
|
|
381
|
+
lines += [
|
|
382
|
+
"",
|
|
383
|
+
f"{len(new)} new failures in the full run, "
|
|
384
|
+
f"{len(report.failures) - len(new)} already failing at the baseline.",
|
|
385
|
+
]
|
|
386
|
+
if report.plan_status != "complete" or report.discovery_incomplete:
|
|
387
|
+
lines.append(
|
|
388
|
+
f"Plan status: {report.plan_status}"
|
|
389
|
+
+ ("; discovery may be incomplete." if report.discovery_incomplete else ".")
|
|
390
|
+
)
|
|
391
|
+
if report.misses:
|
|
392
|
+
lines += ["", "Missed by the plan:", ""]
|
|
393
|
+
lines += [f"- `{f.test}` ({f.outcome}, {f.kind})" for f in report.misses]
|
|
394
|
+
for r in report.runs:
|
|
395
|
+
if r.misses or r.disagreements:
|
|
396
|
+
lines += ["", f"{r.name} run:", ""]
|
|
397
|
+
lines += [f"- not run: `{t}`" for t in r.misses]
|
|
398
|
+
lines += [f"- outcome differs from the full run: `{t}`" for t in r.disagreements]
|
|
399
|
+
return "\n".join(lines) + "\n"
|
diffcone/classify.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""Change classifier: compares two source indexes symbol by symbol."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from diffcone.model import CLASS, DEFINED_IN, IMPORTS, IMPORTS_NAME, MODULE, SourceIndex, Symbol
|
|
9
|
+
|
|
10
|
+
ADDED = "added"
|
|
11
|
+
DELETED = "deleted"
|
|
12
|
+
BODY_CHANGED = "body_changed"
|
|
13
|
+
DEFINITION_CHANGED = "definition_changed"
|
|
14
|
+
IMPORTS_ADDED = "imports_added" # modules: new import bindings only
|
|
15
|
+
DEPENDENCIES_CHANGED = "dependencies_changed" # an edge was removed or redirected
|
|
16
|
+
# Edges were only added. A name the symbol uses now resolves to something
|
|
17
|
+
# (a missing import added, a fallback import that now succeeds): its own
|
|
18
|
+
# behaviour changed. For a module, only when its import-time code gained a
|
|
19
|
+
# reference, or an added import brings in-scope modules into its import
|
|
20
|
+
# closure whose import-time code did not run before; new name bindings alone
|
|
21
|
+
# are ``imports_added``, their users carrying their own added edges.
|
|
22
|
+
DEPENDENCIES_ADDED = "dependencies_added"
|
|
23
|
+
DOCSTRING_CHANGED = "docstring_changed" # only the docstring differs
|
|
24
|
+
# Only a function's annotations differ and they are never evaluated at import
|
|
25
|
+
# (see Symbol.deferred_annotations): behaviour-level, not structural, not
|
|
26
|
+
# import-time (introspection such as typer's happens when called).
|
|
27
|
+
ANNOTATIONS_CHANGED = "annotations_changed"
|
|
28
|
+
|
|
29
|
+
# Changes that invalidate everything defined inside the symbol (and, for
|
|
30
|
+
# deletions, everything that imports it), not just direct references.
|
|
31
|
+
# Pure additions (a new import binding, a new dependency edge) cannot break
|
|
32
|
+
# an existing member: a member whose own resolution changed because of the
|
|
33
|
+
# addition carries its own dependencies_changed.
|
|
34
|
+
STRUCTURAL = frozenset({ADDED, DELETED, DEFINITION_CHANGED, DEPENDENCIES_CHANGED})
|
|
35
|
+
|
|
36
|
+
# Changes that are reported but carry no impact of their own: nothing an
|
|
37
|
+
# existing dependent can observe differs (a member whose resolution moved
|
|
38
|
+
# because of the addition carries its own dependencies_added or
|
|
39
|
+
# dependencies_changed).
|
|
40
|
+
NON_IMPACT = frozenset({IMPORTS_ADDED, DOCSTRING_CHANGED})
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True, order=True)
|
|
44
|
+
class SymbolChange:
|
|
45
|
+
id: str
|
|
46
|
+
kind: str
|
|
47
|
+
changes: tuple[str, ...]
|
|
48
|
+
base: Symbol | None
|
|
49
|
+
head: Symbol | None
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def symbol(self) -> Symbol:
|
|
53
|
+
"""The symbol as it is in head, or as it was in base when deleted:
|
|
54
|
+
a change always has one side."""
|
|
55
|
+
symbol = self.head or self.base
|
|
56
|
+
assert symbol is not None, f"{self.id} has neither side"
|
|
57
|
+
return symbol
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def carries_impact(self) -> bool:
|
|
61
|
+
return not set(self.changes) <= NON_IMPACT
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def structural(self) -> bool:
|
|
65
|
+
"""Whether the change invalidates every member of the symbol.
|
|
66
|
+
|
|
67
|
+
Class bodies count: class-level attributes (ASV ``params``, pytest
|
|
68
|
+
marks, registries) shape how every method runs even when no method
|
|
69
|
+
references them textually. Module bodies do not; members that use
|
|
70
|
+
module state carry their own edges (see internal/design.md).
|
|
71
|
+
"""
|
|
72
|
+
if STRUCTURAL & set(self.changes):
|
|
73
|
+
return True
|
|
74
|
+
return self.kind == CLASS and BODY_CHANGED in self.changes
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# Dependency-signature kinds for references outside the edge graph.
|
|
78
|
+
UNRESOLVED_SIG = "unresolved"
|
|
79
|
+
EXTERNAL_SIG = "external"
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _dependency_signatures(index: SourceIndex) -> dict[str, frozenset[tuple[str, str, str]]]:
|
|
83
|
+
"""Each symbol's outgoing edges, and its unresolved and external
|
|
84
|
+
references: a name that starts resolving to a third-party module (the
|
|
85
|
+
``from json import dumps`` that fixes a NameError) changes what the
|
|
86
|
+
symbol does as much as a new edge does."""
|
|
87
|
+
grouped: dict[str, set[tuple[str, str, str]]] = defaultdict(set)
|
|
88
|
+
for e in index.edges:
|
|
89
|
+
if e.kind != DEFINED_IN:
|
|
90
|
+
grouped[e.source].add((e.kind, e.target, e.detail))
|
|
91
|
+
for u in index.unresolved:
|
|
92
|
+
grouped[u.symbol].add((UNRESOLVED_SIG, f"{u.kind}:{u.name}", ""))
|
|
93
|
+
for x in index.external:
|
|
94
|
+
grouped[x.symbol].add((EXTERNAL_SIG, x.module, ""))
|
|
95
|
+
return {source: frozenset(sig) for source, sig in grouped.items()}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _is_subsequence(before: tuple[str, ...], after: tuple[str, ...]) -> bool:
|
|
99
|
+
"""Whether ``after`` is ``before`` with entries inserted: every import
|
|
100
|
+
kept its block and its order relative to the others."""
|
|
101
|
+
remaining = iter(after)
|
|
102
|
+
return all(entry in remaining for entry in before)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _runtime_imports(symbol: Symbol, modules: set[str]) -> set[str]:
|
|
106
|
+
"""The in-scope modules a module's own top-level code imports when it
|
|
107
|
+
runs, from its import layout: not imports in function bodies (not in the
|
|
108
|
+
layout) or under ``if TYPE_CHECKING:``, which never run."""
|
|
109
|
+
found: set[str] = set()
|
|
110
|
+
for entry in symbol.import_layout:
|
|
111
|
+
context, _, statement = entry.rpartition("|")
|
|
112
|
+
if "TYPE_CHECKING" in context:
|
|
113
|
+
continue
|
|
114
|
+
words = statement.split()
|
|
115
|
+
if words[:1] == ["import"] and len(words) >= 2:
|
|
116
|
+
names = [words[1]]
|
|
117
|
+
elif words[:1] == ["from"] and len(words) >= 4:
|
|
118
|
+
names = [words[1], f"{words[1]}.{words[3]}"]
|
|
119
|
+
else:
|
|
120
|
+
continue
|
|
121
|
+
for name in names:
|
|
122
|
+
# ``import a.b.c`` runs ``a`` and ``a.b`` too.
|
|
123
|
+
parts = name.split(".")
|
|
124
|
+
for i in range(1, len(parts) + 1):
|
|
125
|
+
if ".".join(parts[:i]) in modules:
|
|
126
|
+
found.add(".".join(parts[:i]))
|
|
127
|
+
return found
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _import_closures(index: SourceIndex) -> dict[str, set[str]]:
|
|
131
|
+
"""Each module's transitive closure of the modules importing it runs
|
|
132
|
+
(itself included)."""
|
|
133
|
+
modules = {s.id for s in index.symbols.values() if s.kind == MODULE}
|
|
134
|
+
imports = {m: _runtime_imports(index.symbols[m], modules) for m in modules}
|
|
135
|
+
closures: dict[str, set[str]] = {}
|
|
136
|
+
|
|
137
|
+
def closure(module: str) -> set[str]:
|
|
138
|
+
if module not in closures:
|
|
139
|
+
seen = {module}
|
|
140
|
+
stack = [module]
|
|
141
|
+
while stack:
|
|
142
|
+
for target in imports.get(stack.pop(), ()):
|
|
143
|
+
if target not in seen:
|
|
144
|
+
seen.add(target)
|
|
145
|
+
stack.append(target)
|
|
146
|
+
closures[module] = seen
|
|
147
|
+
return closures[module]
|
|
148
|
+
|
|
149
|
+
return {module: closure(module) for module in modules}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _module_additions_matter(
|
|
153
|
+
module: str,
|
|
154
|
+
added: frozenset[tuple[str, str, str]],
|
|
155
|
+
base_closures: dict[str, set[str]],
|
|
156
|
+
base: SourceIndex,
|
|
157
|
+
b: Symbol,
|
|
158
|
+
h: Symbol,
|
|
159
|
+
) -> bool:
|
|
160
|
+
"""Whether a module's additions change what importing it does: its
|
|
161
|
+
import-time code references something new, or an import that now runs
|
|
162
|
+
at import reaches an in-scope module importing it did not run before (a
|
|
163
|
+
registration import such as ``import pkg.json_handler``, one moved out of
|
|
164
|
+
a function or out of ``if TYPE_CHECKING:``). A module new in head is
|
|
165
|
+
added, and reached through its own change."""
|
|
166
|
+
closure = base_closures.get(module, {module})
|
|
167
|
+
for kind, _target, _detail in added:
|
|
168
|
+
# A new third-party import is not import-time code of the project's
|
|
169
|
+
# (installed code is assumed unchanged); anything else is.
|
|
170
|
+
if kind not in (IMPORTS, IMPORTS_NAME, EXTERNAL_SIG):
|
|
171
|
+
return True
|
|
172
|
+
modules = {s.id for s in base.symbols.values() if s.kind == MODULE}
|
|
173
|
+
newly_run = _runtime_imports(h, modules) - _runtime_imports(b, modules)
|
|
174
|
+
return bool(newly_run - closure)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def classify(base: SourceIndex, head: SourceIndex) -> list[SymbolChange]:
|
|
178
|
+
"""Return every symbol that differs between the two revisions.
|
|
179
|
+
|
|
180
|
+
Symbols whose module failed to parse in one revision are skipped: their
|
|
181
|
+
status is unknown, and the planner handles that through the analysis
|
|
182
|
+
error fallback instead of inventing additions or deletions.
|
|
183
|
+
"""
|
|
184
|
+
changes: list[SymbolChange] = []
|
|
185
|
+
base_deps = _dependency_signatures(base)
|
|
186
|
+
head_deps = _dependency_signatures(head)
|
|
187
|
+
base_closures = _import_closures(base)
|
|
188
|
+
empty: frozenset[tuple[str, str, str]] = frozenset()
|
|
189
|
+
for symbol_id in sorted(set(base.symbols) | set(head.symbols)):
|
|
190
|
+
b = base.symbols.get(symbol_id)
|
|
191
|
+
h = head.symbols.get(symbol_id)
|
|
192
|
+
if b is None:
|
|
193
|
+
assert h is not None # the id came from one of the two
|
|
194
|
+
if h.module not in base.failed_modules:
|
|
195
|
+
changes.append(SymbolChange(symbol_id, h.kind, (ADDED,), None, h))
|
|
196
|
+
continue
|
|
197
|
+
if h is None:
|
|
198
|
+
if b.module in head.failed_modules:
|
|
199
|
+
continue
|
|
200
|
+
changes.append(SymbolChange(symbol_id, b.kind, (DELETED,), b, None))
|
|
201
|
+
continue
|
|
202
|
+
kinds: list[str] = []
|
|
203
|
+
if b.body_hash != h.body_hash:
|
|
204
|
+
kinds.append(BODY_CHANGED)
|
|
205
|
+
elif b.docstring_hash != h.docstring_hash:
|
|
206
|
+
kinds.append(DOCSTRING_CHANGED)
|
|
207
|
+
if b.kind != h.kind:
|
|
208
|
+
kinds.append(DEFINITION_CHANGED)
|
|
209
|
+
elif b.definition_hash != h.definition_hash:
|
|
210
|
+
future = {i for i in set(b.imports) ^ set(h.imports) if i.startswith("from __future__")}
|
|
211
|
+
if (
|
|
212
|
+
b.kind == MODULE
|
|
213
|
+
and set(b.imports) <= set(h.imports)
|
|
214
|
+
and _is_subsequence(b.import_layout, h.import_layout)
|
|
215
|
+
and not future
|
|
216
|
+
):
|
|
217
|
+
kinds.append(IMPORTS_ADDED)
|
|
218
|
+
else:
|
|
219
|
+
kinds.append(DEFINITION_CHANGED)
|
|
220
|
+
if b.annotation_hash != h.annotation_hash and DEFINITION_CHANGED not in kinds:
|
|
221
|
+
deferred = b.deferred_annotations and h.deferred_annotations
|
|
222
|
+
kinds.append(ANNOTATIONS_CHANGED if deferred else DEFINITION_CHANGED)
|
|
223
|
+
before = base_deps.get(symbol_id, empty)
|
|
224
|
+
after = head_deps.get(symbol_id, empty)
|
|
225
|
+
if BODY_CHANGED in kinds or DEFINITION_CHANGED in kinds:
|
|
226
|
+
# The edit itself explains new unresolved and external references;
|
|
227
|
+
# they matter on their own only when a binding moved under them.
|
|
228
|
+
before = frozenset(d for d in before if d[0] not in (UNRESOLVED_SIG, EXTERNAL_SIG))
|
|
229
|
+
after = frozenset(d for d in after if d[0] not in (UNRESOLVED_SIG, EXTERNAL_SIG))
|
|
230
|
+
# A name that stopped being unresolved now resolves: an addition, not
|
|
231
|
+
# a redirect (which is a resolved dependency lost or moved).
|
|
232
|
+
lost = {d for d in before - after if d[0] != UNRESOLVED_SIG}
|
|
233
|
+
resolved_now = any(d[0] == UNRESOLVED_SIG for d in before - after)
|
|
234
|
+
if before != after:
|
|
235
|
+
if lost:
|
|
236
|
+
kinds.append(DEPENDENCIES_CHANGED)
|
|
237
|
+
elif (
|
|
238
|
+
h.kind != MODULE
|
|
239
|
+
or resolved_now
|
|
240
|
+
or _module_additions_matter(symbol_id, after - before, base_closures, base, b, h)
|
|
241
|
+
):
|
|
242
|
+
kinds.append(DEPENDENCIES_ADDED)
|
|
243
|
+
elif IMPORTS_ADDED not in kinds:
|
|
244
|
+
kinds.append(IMPORTS_ADDED)
|
|
245
|
+
elif (
|
|
246
|
+
h.kind == MODULE
|
|
247
|
+
and IMPORTS_ADDED in kinds
|
|
248
|
+
and _module_additions_matter(symbol_id, frozenset(), base_closures, base, b, h)
|
|
249
|
+
):
|
|
250
|
+
# No new edge (a TYPE_CHECKING copy already had it), but the
|
|
251
|
+
# import now runs.
|
|
252
|
+
kinds.append(DEPENDENCIES_ADDED)
|
|
253
|
+
if kinds:
|
|
254
|
+
changes.append(SymbolChange(symbol_id, h.kind, tuple(kinds), b, h))
|
|
255
|
+
return changes
|