diffcone 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffcone/__init__.py +13 -0
- diffcone/cache.py +528 -0
- diffcone/check.py +399 -0
- diffcone/classify.py +255 -0
- diffcone/cli.py +886 -0
- diffcone/collect.py +846 -0
- diffcone/cython.py +757 -0
- diffcone/declarations.py +111 -0
- diffcone/discovery/__init__.py +121 -0
- diffcone/discovery/asv_static.py +494 -0
- diffcone/discovery/common.py +130 -0
- diffcone/discovery/pytest_static.py +2680 -0
- diffcone/evidence.py +738 -0
- diffcone/evidence_plan.py +1487 -0
- diffcone/execution.py +1681 -0
- diffcone/indexer/__init__.py +48 -0
- diffcone/indexer/core.py +286 -0
- diffcone/indexer/definitions.py +292 -0
- diffcone/indexer/dynamics.py +406 -0
- diffcone/indexer/facts.py +373 -0
- diffcone/indexer/literals.py +388 -0
- diffcone/indexer/references.py +895 -0
- diffcone/indexer/resolver.py +1001 -0
- diffcone/indexer/scopes.py +276 -0
- diffcone/indexer/state.py +83 -0
- diffcone/indexer/symbols.py +441 -0
- diffcone/indexer/syntax.py +173 -0
- diffcone/manifest.py +166 -0
- diffcone/model.py +191 -0
- diffcone/planner.py +1453 -0
- diffcone/report.py +256 -0
- diffcone/selection.py +133 -0
- diffcone/snapshot.py +739 -0
- diffcone/testing.py +200 -0
- diffcone-0.1.0.dist-info/METADATA +133 -0
- diffcone-0.1.0.dist-info/RECORD +39 -0
- diffcone-0.1.0.dist-info/WHEEL +4 -0
- diffcone-0.1.0.dist-info/entry_points.txt +3 -0
- diffcone-0.1.0.dist-info/licenses/LICENSE +21 -0
diffcone/execution.py
ADDED
|
@@ -0,0 +1,1681 @@
|
|
|
1
|
+
"""Execution integration: run selected targets, and validate plans.
|
|
2
|
+
|
|
3
|
+
Planning never executes project code. Everything in this module runs
|
|
4
|
+
*after* a plan exists and is invoked only by the ``run`` and ``validate``
|
|
5
|
+
commands, in a subprocess, with a command the user controls.
|
|
6
|
+
|
|
7
|
+
``run`` executes the selected targets of a plan with the runner's CLI.
|
|
8
|
+
``validate`` runs the full pytest suite at both snapshots (in temporary
|
|
9
|
+
``git worktree`` checkouts for commits, in place for WORKTREE)
|
|
10
|
+
and checks that every test whose outcome changed was selected.
|
|
11
|
+
This is outcome-based validation: a test whose behaviour
|
|
12
|
+
changed without changing its pass/fail outcome is not detected.
|
|
13
|
+
With ``coverage=True`` it also runs the head suite under
|
|
14
|
+
pytest-cov with per-test contexts and checks that every test
|
|
15
|
+
that *executed* a changed symbol was selected, which measures
|
|
16
|
+
recall (and reports precision) against dynamic ground truth.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import posixpath
|
|
24
|
+
import re
|
|
25
|
+
import shlex
|
|
26
|
+
import shutil
|
|
27
|
+
import sqlite3
|
|
28
|
+
import subprocess
|
|
29
|
+
import tempfile
|
|
30
|
+
import threading
|
|
31
|
+
from collections import defaultdict
|
|
32
|
+
from collections.abc import Iterator, Sequence
|
|
33
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
34
|
+
from contextlib import contextmanager
|
|
35
|
+
from dataclasses import dataclass, field
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
|
|
38
|
+
from diffcone.cache import IndexCache
|
|
39
|
+
from diffcone.discovery.asv_static import strip_json_comments
|
|
40
|
+
from diffcone.evidence import EVIDENCE_DIR, Evidence, EvidenceError, advance, fold, write_store
|
|
41
|
+
from diffcone.manifest import Target
|
|
42
|
+
from diffcone.model import KIND_COMMIT, KIND_WORKTREE, MODULE, SourceIndex
|
|
43
|
+
from diffcone.planner import Plan, _index_snapshot
|
|
44
|
+
from diffcone.snapshot import GitError, _git, is_bytecode, resolve_commit, split_root
|
|
45
|
+
|
|
46
|
+
DEFAULT_COMMANDS = {"pytest": "python -m pytest", "asv": "asv run --python=same"}
|
|
47
|
+
|
|
48
|
+
# Parameter ids may contain spaces, pipes and nested brackets
|
|
49
|
+
# (``test_x[choices4-[TEXT: a|b]] PASSED [ 12%]``), so the node id is
|
|
50
|
+
# everything up to the outcome token.
|
|
51
|
+
_PYTEST_LINE = re.compile(
|
|
52
|
+
r"^(?P<nodeid>\S+::.*?) (?P<outcome>PASSED|FAILED|ERROR|SKIPPED|XFAIL|XPASS)(?:\s|$)"
|
|
53
|
+
)
|
|
54
|
+
# pytest-xdist's report of a worker that died mid-run, and the interpreter's
|
|
55
|
+
# own last words.
|
|
56
|
+
_CRASH_LINE = re.compile(
|
|
57
|
+
r"node down|crashed while running|replacing crashed worker|Fatal Python error|"
|
|
58
|
+
r"Segmentation fault|Bus error|Killed"
|
|
59
|
+
)
|
|
60
|
+
# pytest-xdist puts the worker and the outcome first: ``[gw3] PASSED a.py::t``
|
|
61
|
+
# (``[gw3] [ 12%] PASSED ...`` in some versions).
|
|
62
|
+
_XDIST_LINE = re.compile(
|
|
63
|
+
r"^\[gw\d+\](?: \[\s*\d+%\])? (?P<outcome>PASSED|FAILED|ERROR|SKIPPED|XFAIL|XPASS) "
|
|
64
|
+
r"(?P<nodeid>\S+::.*?)\s*$"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# --------------------------------------------------------------------------- run
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def build_command(
|
|
72
|
+
runner: str, targets: list[Target], command: str | None, extra: list[str]
|
|
73
|
+
) -> list[str]:
|
|
74
|
+
"""The command line that runs exactly ``targets`` with ``runner``."""
|
|
75
|
+
base = shlex.split(command or DEFAULT_COMMANDS[runner])
|
|
76
|
+
ids = [t.runner_id for t in sorted(targets)]
|
|
77
|
+
if runner == "pytest":
|
|
78
|
+
return [*base, *extra, *ids]
|
|
79
|
+
if runner == "asv":
|
|
80
|
+
# ASV matches --bench against the benchmark's name, and for a
|
|
81
|
+
# parameterised one against ``name(param0, param1, ...)`` instead, so
|
|
82
|
+
# a pattern anchored with ``$`` selects none of those: the name must
|
|
83
|
+
# be followed by the end of the string or its parameter list.
|
|
84
|
+
pattern = "^(" + "|".join(re.escape(i) for i in ids) + r")($|\()"
|
|
85
|
+
return [*base, *extra, "--bench", pattern]
|
|
86
|
+
raise ValueError(f"unknown runner {runner!r}")
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def resolve_command(command: str | None, repo: Path) -> str | None:
|
|
90
|
+
"""Make a relative executable path in ``command`` (``.venv/bin/python``)
|
|
91
|
+
absolute, against the current directory and then the repository, so
|
|
92
|
+
the command still works from a temporary worktree. Symlinks are kept:
|
|
93
|
+
a venv's ``python`` is a link to the base interpreter, and following it
|
|
94
|
+
would run outside the venv."""
|
|
95
|
+
if command is None:
|
|
96
|
+
return None
|
|
97
|
+
argv = shlex.split(command)
|
|
98
|
+
if not argv or os.path.isabs(argv[0]) or os.sep not in argv[0]:
|
|
99
|
+
return command
|
|
100
|
+
for base in (Path.cwd(), repo):
|
|
101
|
+
candidate = base / argv[0]
|
|
102
|
+
if candidate.exists():
|
|
103
|
+
return shlex.join([os.path.abspath(candidate), *argv[1:]])
|
|
104
|
+
return command
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass
|
|
108
|
+
class RunResult:
|
|
109
|
+
runner: str
|
|
110
|
+
command: list[str]
|
|
111
|
+
selected: list[Target]
|
|
112
|
+
total: int
|
|
113
|
+
returncode: int | None # None when nothing was run (dry run or empty selection)
|
|
114
|
+
# Selected pytest targets that pytest did not collect (discovery and
|
|
115
|
+
# collection disagree): a node id on the command line used to fail loudly.
|
|
116
|
+
missing: list[str] = field(default_factory=list)
|
|
117
|
+
# Collected tests that are no target of the plan: kept and run (see
|
|
118
|
+
# selection.py), and reported.
|
|
119
|
+
unknown: list[str] = field(default_factory=list)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def worktree_mismatch(repo: Path, plan: Plan) -> str | None:
|
|
123
|
+
"""Why the working tree is not what the plan analysed, if it is not.
|
|
124
|
+
|
|
125
|
+
``run`` executes the checkout, so a plan made from a commit only
|
|
126
|
+
describes what will run while the checkout *is* that commit, unmodified.
|
|
127
|
+
Running a different tree turns the plan's guarantee into a guess about
|
|
128
|
+
code that was never analysed.
|
|
129
|
+
"""
|
|
130
|
+
if plan.head.kind == KIND_WORKTREE:
|
|
131
|
+
return None
|
|
132
|
+
# What matters is the code, not the commit: a checkout carrying an extra
|
|
133
|
+
# commit that only adds a manifest runs the analysed code, while one
|
|
134
|
+
# uncommitted edit to a source file does not.
|
|
135
|
+
against = ["diff", "--name-only"] + (
|
|
136
|
+
[plan.head.commit] if plan.head.kind == KIND_COMMIT else []
|
|
137
|
+
)
|
|
138
|
+
try:
|
|
139
|
+
differing = _git(repo, against).decode("utf-8", "surrogateescape").split("\n")
|
|
140
|
+
untracked = (
|
|
141
|
+
_git(repo, ["ls-files", "--others", "--exclude-standard"])
|
|
142
|
+
.decode("utf-8", "surrogateescape")
|
|
143
|
+
.split("\n")
|
|
144
|
+
)
|
|
145
|
+
except GitError as exc:
|
|
146
|
+
return f"cannot tell what the working tree holds: {exc}"
|
|
147
|
+
roots = [split_root(r)[0] for r in plan.source_roots]
|
|
148
|
+
in_scope = sorted(
|
|
149
|
+
path
|
|
150
|
+
for path in {p.strip() for p in differing + untracked if p.strip()}
|
|
151
|
+
if path.endswith(".py")
|
|
152
|
+
and any(root in ("", ".") or path.startswith(root + "/") for root in roots)
|
|
153
|
+
)
|
|
154
|
+
if not in_scope:
|
|
155
|
+
return None
|
|
156
|
+
analysed = (
|
|
157
|
+
f"commit {plan.head.commit[:12]}" if plan.head.kind == KIND_COMMIT else "the staged index"
|
|
158
|
+
)
|
|
159
|
+
return (
|
|
160
|
+
f"the plan analysed {analysed}, and the working tree it would run has "
|
|
161
|
+
f"{len(in_scope)} differing Python file(s) under the source roots (e.g. {in_scope[0]})"
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def run_selected(
|
|
166
|
+
plan: Plan,
|
|
167
|
+
runner: str,
|
|
168
|
+
*,
|
|
169
|
+
cwd: Path,
|
|
170
|
+
command: str | None = None,
|
|
171
|
+
extra: list[str] | None = None,
|
|
172
|
+
dry_run: bool = False,
|
|
173
|
+
env: dict[str, str] | None = None,
|
|
174
|
+
) -> RunResult:
|
|
175
|
+
"""Run the selected targets of ``runner``. ``--dry-run`` reports the
|
|
176
|
+
equivalent command (pytest node ids appended); a real pytest run instead
|
|
177
|
+
collects from pytest's own starting points and keeps the selection with
|
|
178
|
+
``-p diffcone_select`` (``selection.py`` says why), so conftests load as
|
|
179
|
+
in a full run."""
|
|
180
|
+
selected = [d.target for d in plan.decisions if d.selected and d.target.runner == runner]
|
|
181
|
+
total = sum(1 for d in plan.decisions if d.target.runner == runner)
|
|
182
|
+
argv = build_command(runner, selected, command, list(extra or [])) if selected else []
|
|
183
|
+
result = RunResult(runner, argv, selected, total, None)
|
|
184
|
+
if dry_run or not selected:
|
|
185
|
+
return result
|
|
186
|
+
if runner != "pytest":
|
|
187
|
+
result.returncode = subprocess.run(argv, cwd=cwd, env=env).returncode
|
|
188
|
+
return result
|
|
189
|
+
with tempfile.TemporaryDirectory(prefix="diffcone-select-") as tmp:
|
|
190
|
+
own = env is None
|
|
191
|
+
run_env = plugin_environment(dict(os.environ), None, cwd) if own else dict(env)
|
|
192
|
+
run_env["DIFFCONE_SELECT"] = tmp
|
|
193
|
+
Path(tmp, "selected").write_text(
|
|
194
|
+
"".join(f"{t.runner_id}\n" for t in selected), encoding="utf-8"
|
|
195
|
+
)
|
|
196
|
+
every = [d.target for d in plan.decisions if d.target.runner == runner]
|
|
197
|
+
Path(tmp, "targets").write_text(
|
|
198
|
+
"".join(f"{t.runner_id}\n" for t in every), encoding="utf-8"
|
|
199
|
+
)
|
|
200
|
+
result.command = [
|
|
201
|
+
*shlex.split(command or DEFAULT_COMMANDS["pytest"]),
|
|
202
|
+
"-p",
|
|
203
|
+
SELECT_PLUGIN,
|
|
204
|
+
*(extra or []),
|
|
205
|
+
]
|
|
206
|
+
try:
|
|
207
|
+
result.returncode = subprocess.run(result.command, cwd=cwd, env=run_env).returncode
|
|
208
|
+
finally:
|
|
209
|
+
if own:
|
|
210
|
+
shutil.rmtree(run_env["PYTHONPATH"].split(os.pathsep)[-1], ignore_errors=True)
|
|
211
|
+
reports = [
|
|
212
|
+
set(f.read_text("utf-8").split("\n")) - {""} for f in Path(tmp).glob("missing-*")
|
|
213
|
+
]
|
|
214
|
+
if reports:
|
|
215
|
+
result.missing = sorted(set.intersection(*reports))
|
|
216
|
+
result.unknown = sorted(
|
|
217
|
+
{
|
|
218
|
+
line
|
|
219
|
+
for f in Path(tmp).glob("unknown-*")
|
|
220
|
+
for line in f.read_text("utf-8").split("\n")
|
|
221
|
+
if line
|
|
222
|
+
}
|
|
223
|
+
)
|
|
224
|
+
return result
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
# --------------------------------------------------------------------------- validate
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
@dataclass(frozen=True)
|
|
231
|
+
class TargetOutcome:
|
|
232
|
+
runner_id: str
|
|
233
|
+
base: str | None # None: absent at that snapshot
|
|
234
|
+
head: str | None
|
|
235
|
+
selected: bool
|
|
236
|
+
known_at_head: bool = True # a target of the plan (exists in the head snapshot)
|
|
237
|
+
|
|
238
|
+
@property
|
|
239
|
+
def changed(self) -> bool:
|
|
240
|
+
return self.base != self.head
|
|
241
|
+
|
|
242
|
+
@property
|
|
243
|
+
def removed(self) -> bool:
|
|
244
|
+
"""Ran at base, gone at head: nothing to select, so never a miss."""
|
|
245
|
+
return self.head is None and not self.known_at_head
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
@dataclass(frozen=True)
|
|
249
|
+
class CoverageHit:
|
|
250
|
+
runner_id: str
|
|
251
|
+
selected: bool
|
|
252
|
+
executed_changed: tuple[str, ...] # changed symbols whose lines the test ran
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
@dataclass
|
|
256
|
+
class CoverageValidation:
|
|
257
|
+
hits: list[CoverageHit] = field(default_factory=list) # one per test seen under coverage
|
|
258
|
+
changed_symbols: tuple[str, ...] = ()
|
|
259
|
+
log: str = ""
|
|
260
|
+
|
|
261
|
+
@property
|
|
262
|
+
def affected(self) -> list[CoverageHit]:
|
|
263
|
+
return [h for h in self.hits if h.executed_changed]
|
|
264
|
+
|
|
265
|
+
@property
|
|
266
|
+
def caught(self) -> list[CoverageHit]:
|
|
267
|
+
return [h for h in self.affected if h.selected]
|
|
268
|
+
|
|
269
|
+
@property
|
|
270
|
+
def missed(self) -> list[CoverageHit]:
|
|
271
|
+
return [h for h in self.affected if not h.selected]
|
|
272
|
+
|
|
273
|
+
@property
|
|
274
|
+
def recall(self) -> float | None:
|
|
275
|
+
return len(self.caught) / len(self.affected) if self.affected else None
|
|
276
|
+
|
|
277
|
+
@property
|
|
278
|
+
def precision(self) -> float | None:
|
|
279
|
+
selected = [h for h in self.hits if h.selected]
|
|
280
|
+
return len(self.caught) / len(selected) if selected else None
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
@dataclass
|
|
284
|
+
class Validation:
|
|
285
|
+
runner: str
|
|
286
|
+
command: list[str]
|
|
287
|
+
outcomes: list[TargetOutcome] = field(default_factory=list)
|
|
288
|
+
base_log: str = ""
|
|
289
|
+
head_log: str = ""
|
|
290
|
+
coverage: CoverageValidation | None = None
|
|
291
|
+
|
|
292
|
+
@property
|
|
293
|
+
def caught(self) -> list[TargetOutcome]:
|
|
294
|
+
return [o for o in self.outcomes if o.changed and o.selected]
|
|
295
|
+
|
|
296
|
+
@property
|
|
297
|
+
def missed(self) -> list[TargetOutcome]:
|
|
298
|
+
return [o for o in self.outcomes if o.changed and not o.selected and not o.removed]
|
|
299
|
+
|
|
300
|
+
@property
|
|
301
|
+
def removed(self) -> list[TargetOutcome]:
|
|
302
|
+
return [o for o in self.outcomes if o.removed]
|
|
303
|
+
|
|
304
|
+
@property
|
|
305
|
+
def selected_count(self) -> int:
|
|
306
|
+
return sum(1 for o in self.outcomes if o.selected)
|
|
307
|
+
|
|
308
|
+
@property
|
|
309
|
+
def ok(self) -> bool:
|
|
310
|
+
return not self.missed and (self.coverage is None or not self.coverage.missed)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def fold_nodeid(nodeid: str) -> str:
|
|
314
|
+
"""Drop the parameter case (``[...]``) so a node id names the test function."""
|
|
315
|
+
return re.sub(r"\[.*\]$", "", nodeid)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def parse_pytest_verbose(output: str) -> dict[str, str]:
|
|
319
|
+
"""Map pytest node ids (parameter cases folded into their function) to
|
|
320
|
+
the outcomes observed for that function, in a fixed order and joined
|
|
321
|
+
with ``+`` (``PASSED+SKIPPED``). The fold must not depend on the order
|
|
322
|
+
the cases ran in: under pytest-xdist that order varies from run to run,
|
|
323
|
+
and a fold that kept the first or last case would report changes that
|
|
324
|
+
never happened."""
|
|
325
|
+
order = ("PASSED", "SKIPPED", "XFAIL", "XPASS", "FAILED", "ERROR")
|
|
326
|
+
seen: dict[str, set[str]] = {}
|
|
327
|
+
for line in output.splitlines():
|
|
328
|
+
m = _PYTEST_LINE.match(line.strip()) or _XDIST_LINE.match(line.strip())
|
|
329
|
+
if not m:
|
|
330
|
+
continue
|
|
331
|
+
seen.setdefault(fold_nodeid(m.group("nodeid")), set()).add(m.group("outcome"))
|
|
332
|
+
return {nodeid: "+".join(o for o in order if o in found) for nodeid, found in seen.items()}
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def parse_outcome_lines(lines: list[str]) -> dict[str, str]:
|
|
336
|
+
"""``parse_pytest_verbose`` for the selection plugin's outcome lines
|
|
337
|
+
(``nodeid<TAB>OUTCOME``): the same fold, the same order."""
|
|
338
|
+
order = ("PASSED", "SKIPPED", "XFAIL", "XPASS", "FAILED", "ERROR")
|
|
339
|
+
seen: dict[str, set[str]] = {}
|
|
340
|
+
for line in lines:
|
|
341
|
+
nodeid, _, outcome = line.rpartition("\t")
|
|
342
|
+
if nodeid and outcome in order:
|
|
343
|
+
seen.setdefault(fold_nodeid(nodeid), set()).add(outcome)
|
|
344
|
+
return {nodeid: "+".join(o for o in order if o in found) for nodeid, found in seen.items()}
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
class _Checkout:
|
|
348
|
+
"""A directory holding a snapshot: a temporary detached worktree for a
|
|
349
|
+
commit, or the repository itself for WORKTREE."""
|
|
350
|
+
|
|
351
|
+
def __init__(
|
|
352
|
+
self, repo: Path, kind: str, commit: str, setup_command: str | None = None
|
|
353
|
+
) -> None:
|
|
354
|
+
self.repo = repo
|
|
355
|
+
self.kind = kind
|
|
356
|
+
self.setup_command = setup_command
|
|
357
|
+
self.commit = commit
|
|
358
|
+
self.path = repo
|
|
359
|
+
self._tmp: tempfile.TemporaryDirectory[str] | None = None
|
|
360
|
+
|
|
361
|
+
def __enter__(self) -> Path:
|
|
362
|
+
if self.kind == KIND_COMMIT:
|
|
363
|
+
self._tmp = tempfile.TemporaryDirectory(prefix="diffcone-validate-")
|
|
364
|
+
# A name of its own: git names the worktree's admin directory
|
|
365
|
+
# (.git/worktrees/<name>) after it, and parallel corpus jobs that
|
|
366
|
+
# all used "checkout" raced on creating it.
|
|
367
|
+
self.path = Path(self._tmp.name) / f"checkout-{Path(self._tmp.name).name}"
|
|
368
|
+
_git(self.repo, ["worktree", "add", "--detach", "-q", str(self.path), self.commit])
|
|
369
|
+
elif self.kind != KIND_WORKTREE:
|
|
370
|
+
raise GitError("validate supports commit and WORKTREE snapshots, not INDEX")
|
|
371
|
+
if self.setup_command:
|
|
372
|
+
# Build-generated, git-ignored files (a setuptools-scm _version.py,
|
|
373
|
+
# compiled extensions) are absent from a fresh checkout; the user's
|
|
374
|
+
# setup command recreates what the suite needs.
|
|
375
|
+
proc = subprocess.run(
|
|
376
|
+
self.setup_command, shell=True, cwd=self.path, capture_output=True, text=True
|
|
377
|
+
)
|
|
378
|
+
if proc.returncode != 0:
|
|
379
|
+
self.__exit__()
|
|
380
|
+
raise GitError(
|
|
381
|
+
f"setup command failed in checkout of {self.commit[:12]} "
|
|
382
|
+
f"(exit {proc.returncode}):\n{(proc.stdout + proc.stderr)[-2000:]}"
|
|
383
|
+
)
|
|
384
|
+
return self.path
|
|
385
|
+
|
|
386
|
+
def __exit__(self, *exc: object) -> None:
|
|
387
|
+
if self._tmp is not None:
|
|
388
|
+
try:
|
|
389
|
+
_git(self.repo, ["worktree", "remove", "--force", str(self.path)])
|
|
390
|
+
except GitError:
|
|
391
|
+
pass
|
|
392
|
+
self._tmp.cleanup()
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
@dataclass
|
|
396
|
+
class _SuiteRun:
|
|
397
|
+
outcomes: dict[str, str]
|
|
398
|
+
log: str
|
|
399
|
+
returncode: int
|
|
400
|
+
coverage_db: Path | None = None
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _checkout_env(cwd: Path, source_roots: list[str]) -> dict[str, str]:
|
|
404
|
+
"""Environment that makes the checkout's own code win over an installed
|
|
405
|
+
(typically editable) copy of the project: its source roots go first on
|
|
406
|
+
PYTHONPATH. With a ``src`` layout the current directory alone would not
|
|
407
|
+
do it, and the suite would silently test the installed code."""
|
|
408
|
+
env = dict(os.environ)
|
|
409
|
+
roots = []
|
|
410
|
+
for spec in source_roots:
|
|
411
|
+
root = split_root(spec)[0]
|
|
412
|
+
candidate = (cwd / root).resolve() if root else cwd.resolve()
|
|
413
|
+
if candidate.is_dir() and str(candidate) not in roots:
|
|
414
|
+
roots.append(str(candidate))
|
|
415
|
+
existing = env.get("PYTHONPATH")
|
|
416
|
+
env["PYTHONPATH"] = os.pathsep.join([*roots, *([existing] if existing else [])])
|
|
417
|
+
return env
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
@contextmanager
|
|
421
|
+
def _run_full_pytest(
|
|
422
|
+
cwd: Path,
|
|
423
|
+
command: str | None,
|
|
424
|
+
*,
|
|
425
|
+
coverage: bool = False,
|
|
426
|
+
source_roots: Sequence[str] = (),
|
|
427
|
+
hash_seed: str | None = None,
|
|
428
|
+
coverage_include: list[str] | None = None,
|
|
429
|
+
) -> Iterator[_SuiteRun]:
|
|
430
|
+
"""Run the whole suite once with ``-v``; with ``coverage`` the same run
|
|
431
|
+
also records per-test coverage contexts into a temporary database that
|
|
432
|
+
lives for the duration of the context."""
|
|
433
|
+
argv = [
|
|
434
|
+
*shlex.split(command or DEFAULT_COMMANDS["pytest"]),
|
|
435
|
+
"-v",
|
|
436
|
+
"-p",
|
|
437
|
+
"no:cacheprovider",
|
|
438
|
+
"-p",
|
|
439
|
+
SELECT_PLUGIN,
|
|
440
|
+
"--no-header",
|
|
441
|
+
"-rN",
|
|
442
|
+
]
|
|
443
|
+
with tempfile.TemporaryDirectory(prefix="diffcone-cov-") as tmp:
|
|
444
|
+
# Outcomes come from the selection plugin, not the console: a project
|
|
445
|
+
# whose addopts carry ``-q`` prints no per-test lines at all.
|
|
446
|
+
env = plugin_environment(_checkout_env(cwd, list(source_roots)), None, cwd)
|
|
447
|
+
outcomes_dir = Path(tmp) / "outcomes"
|
|
448
|
+
outcomes_dir.mkdir()
|
|
449
|
+
env["DIFFCONE_OUTCOMES"] = str(outcomes_dir)
|
|
450
|
+
if hash_seed is not None:
|
|
451
|
+
# An evidence plan assumed the recorded hash seed; check it under that.
|
|
452
|
+
env["PYTHONHASHSEED"] = hash_seed
|
|
453
|
+
db: Path | None = None
|
|
454
|
+
if coverage:
|
|
455
|
+
db = Path(tmp) / ".coverage"
|
|
456
|
+
# The project's own addopts stay in force so this run collects the
|
|
457
|
+
# same tests as a plain run; only the coverage options are added.
|
|
458
|
+
# The project's coverage config is replaced: its ``source``/``omit``
|
|
459
|
+
# (typically excluding tests) would blind attribution, and
|
|
460
|
+
# ``parallel``/``branch`` change the database layout.
|
|
461
|
+
rc = Path(tmp) / "coveragerc"
|
|
462
|
+
config = "[run]\nbranch = false\nparallel = false\nrelative_files = false\n"
|
|
463
|
+
if coverage_include is not None:
|
|
464
|
+
# Measure only these checkout-relative files: on a suite the
|
|
465
|
+
# size of pandas, per-test contexts over every file make a
|
|
466
|
+
# database of many gigabytes. (``include`` is ignored when a
|
|
467
|
+
# source is given, so ``--cov`` names none.)
|
|
468
|
+
paths = "".join(f"\n {cwd.resolve() / p}" for p in coverage_include)
|
|
469
|
+
config += f"include ={paths}\n"
|
|
470
|
+
argv += ["--cov"]
|
|
471
|
+
else:
|
|
472
|
+
argv += ["--cov=."]
|
|
473
|
+
rc.write_text(config)
|
|
474
|
+
argv += ["--cov-context=test", "--cov-report=", f"--cov-config={rc}"]
|
|
475
|
+
env["COVERAGE_FILE"] = str(db)
|
|
476
|
+
# coverage.py 7.x defaults to the sys.monitoring core on Python
|
|
477
|
+
# 3.12+, which disables a line after its first hit: with per-test
|
|
478
|
+
# contexts only the *first* test to run a line gets credit for it.
|
|
479
|
+
# coverage avoids that core only for contexts set in its own config,
|
|
480
|
+
# not for pytest-cov's switch_context(), so force the C tracer (it
|
|
481
|
+
# falls back to the pure-Python tracer when unavailable).
|
|
482
|
+
env.setdefault("COVERAGE_CORE", "ctrace")
|
|
483
|
+
try:
|
|
484
|
+
proc = subprocess.run(argv, cwd=cwd, capture_output=True, text=True, env=env)
|
|
485
|
+
except OSError as exc:
|
|
486
|
+
raise GitError(
|
|
487
|
+
f"cannot run {argv[0]!r}: {exc}; a relative path in --command is resolved "
|
|
488
|
+
"against the current directory and the repository"
|
|
489
|
+
) from exc
|
|
490
|
+
finally:
|
|
491
|
+
shutil.rmtree(env["PYTHONPATH"].split(os.pathsep)[-1], ignore_errors=True)
|
|
492
|
+
log = proc.stdout + proc.stderr
|
|
493
|
+
if coverage and not (db and db.exists()):
|
|
494
|
+
raise GitError(
|
|
495
|
+
"coverage validation produced no coverage database; is pytest-cov installed in "
|
|
496
|
+
f"the environment that runs {argv[0]!r} (and not disabled by --no-cov)?\n"
|
|
497
|
+
f"exit code {proc.returncode}\n{log[-2000:]}"
|
|
498
|
+
)
|
|
499
|
+
# pytest exits 0 when every test passed and 1 when some failed; both
|
|
500
|
+
# are suites that ran. Anything else means it did not: 2 a collection
|
|
501
|
+
# error or an interrupt, 3 an internal error, 4 a usage error, 5 no
|
|
502
|
+
# tests at all. Comparing the outcomes of a suite that never ran finds
|
|
503
|
+
# no missed outcome change and would report success.
|
|
504
|
+
if proc.returncode not in (0, 1):
|
|
505
|
+
raise GitError(
|
|
506
|
+
f"the suite did not run: {argv[0]!r} exited {proc.returncode} "
|
|
507
|
+
f"({PYTEST_EXIT.get(proc.returncode, 'unknown')}), so there are no outcomes to "
|
|
508
|
+
f"compare\n{log[-2000:]}"
|
|
509
|
+
)
|
|
510
|
+
lines = [
|
|
511
|
+
line
|
|
512
|
+
for f in sorted(outcomes_dir.glob("outcomes-*"))
|
|
513
|
+
for line in f.read_text("utf-8").splitlines()
|
|
514
|
+
]
|
|
515
|
+
if not lines:
|
|
516
|
+
raise GitError(
|
|
517
|
+
f"the suite ran ({argv[0]!r} exited {proc.returncode}) but reported no test "
|
|
518
|
+
"outcomes: diffcone's outcome plugin did not load, so there is nothing to "
|
|
519
|
+
f"compare\n{log[-2000:]}"
|
|
520
|
+
)
|
|
521
|
+
yield _SuiteRun(parse_outcome_lines(lines), log, proc.returncode, db)
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
# --------------------------------------------------------------------------- coverage
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
PYTEST_EXIT = {
|
|
528
|
+
2: "interrupted, usually a collection error",
|
|
529
|
+
3: "internal error",
|
|
530
|
+
4: "usage error",
|
|
531
|
+
5: "no tests collected",
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def _numbits_to_lines(blob: bytes) -> list[int]:
|
|
536
|
+
"""Decode coverage.py's numbits bitmap: bit n set means line n executed."""
|
|
537
|
+
lines: list[int] = []
|
|
538
|
+
for byte_index, byte in enumerate(blob):
|
|
539
|
+
for bit in range(8):
|
|
540
|
+
if byte & (1 << bit):
|
|
541
|
+
lines.append(byte_index * 8 + bit)
|
|
542
|
+
return lines
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def read_coverage_contexts(
|
|
546
|
+
db_path: Path, root: Path, outside: set[str] | None = None
|
|
547
|
+
) -> dict[str, dict[str, set[int]]]:
|
|
548
|
+
"""Per pytest node id (parameter cases and setup/run/teardown phases
|
|
549
|
+
folded), the executed lines per checkout-relative file. Measured files
|
|
550
|
+
that lie outside the checkout are collected into ``outside`` when given.
|
|
551
|
+
|
|
552
|
+
Reads both the ``line_bits`` table (line coverage) and the ``arc`` table
|
|
553
|
+
(branch coverage, which coverage.py uses *instead* when ``branch = True``
|
|
554
|
+
is configured). Paths may be stored relative when the project sets
|
|
555
|
+
``relative_files``; they are resolved against the checkout root.
|
|
556
|
+
"""
|
|
557
|
+
result: dict[str, dict[str, set[int]]] = defaultdict(lambda: defaultdict(set))
|
|
558
|
+
root = root.resolve()
|
|
559
|
+
|
|
560
|
+
def rel_path(path: str) -> str | None:
|
|
561
|
+
p = Path(path)
|
|
562
|
+
if not p.is_absolute():
|
|
563
|
+
p = root / p
|
|
564
|
+
try:
|
|
565
|
+
return str(p.resolve().relative_to(root))
|
|
566
|
+
except ValueError:
|
|
567
|
+
if outside is not None:
|
|
568
|
+
outside.add(str(p))
|
|
569
|
+
return None
|
|
570
|
+
|
|
571
|
+
con = sqlite3.connect(db_path)
|
|
572
|
+
try:
|
|
573
|
+
line_rows = con.execute(
|
|
574
|
+
"SELECT context.context, file.path, line_bits.numbits FROM line_bits "
|
|
575
|
+
"JOIN context ON context.id = line_bits.context_id "
|
|
576
|
+
"JOIN file ON file.id = line_bits.file_id"
|
|
577
|
+
).fetchall()
|
|
578
|
+
arc_rows = con.execute(
|
|
579
|
+
"SELECT context.context, file.path, arc.fromno, arc.tono FROM arc "
|
|
580
|
+
"JOIN context ON context.id = arc.context_id "
|
|
581
|
+
"JOIN file ON file.id = arc.file_id"
|
|
582
|
+
).fetchall()
|
|
583
|
+
finally:
|
|
584
|
+
con.close()
|
|
585
|
+
for context, path, numbits in line_rows:
|
|
586
|
+
rel = rel_path(path) if context else None
|
|
587
|
+
if rel is not None:
|
|
588
|
+
result[fold_nodeid(context.rsplit("|", 1)[0])][rel].update(_numbits_to_lines(numbits))
|
|
589
|
+
for context, path, fromno, tono in arc_rows:
|
|
590
|
+
rel = rel_path(path) if context else None
|
|
591
|
+
if rel is not None:
|
|
592
|
+
lines = result[fold_nodeid(context.rsplit("|", 1)[0])][rel]
|
|
593
|
+
# Negative numbers mark entry/exit arcs; abs() gives the real line.
|
|
594
|
+
lines.update(n for n in (abs(fromno), abs(tono)) if n > 0)
|
|
595
|
+
return result
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def _line_owner_index(
|
|
599
|
+
plan: Plan, side: str = "head"
|
|
600
|
+
) -> tuple[dict[str, dict[int, set[str]]], tuple[str, ...]]:
|
|
601
|
+
"""``{path: {line: changed symbol ids}}`` for the changed symbols as they
|
|
602
|
+
are at ``side`` (a deleted symbol has lines at base only).
|
|
603
|
+
|
|
604
|
+
A container (module or class) owns only the lines outside its members'
|
|
605
|
+
definitions, mirroring how the planner treats body changes; when its
|
|
606
|
+
change is structural (which invalidates every member) all its lines
|
|
607
|
+
count, mirroring the ``defined_in`` propagation rule.
|
|
608
|
+
"""
|
|
609
|
+
index_ = plan.head_index if side == "head" else plan.base_index
|
|
610
|
+
members_of: dict[str, list[tuple[int, int]]] = defaultdict(list)
|
|
611
|
+
for symbol in index_.symbols.values():
|
|
612
|
+
if symbol.container is not None:
|
|
613
|
+
members_of[symbol.container].extend(symbol.line_ranges)
|
|
614
|
+
index: dict[str, dict[int, set[str]]] = defaultdict(lambda: defaultdict(set))
|
|
615
|
+
changed_ids: list[str] = []
|
|
616
|
+
for change in plan.changes:
|
|
617
|
+
symbol = change.head if side == "head" else change.base
|
|
618
|
+
# Additive-only changes carry no impact for the planner and mean no
|
|
619
|
+
# behaviour change for the code executed, so they are not ground truth.
|
|
620
|
+
if symbol is None or not symbol.line_ranges or not change.carries_impact:
|
|
621
|
+
continue
|
|
622
|
+
changed_ids.append(symbol.id)
|
|
623
|
+
excluded: set[int] = set()
|
|
624
|
+
if not change.structural:
|
|
625
|
+
for start, end in members_of.get(symbol.id, ()):
|
|
626
|
+
excluded.update(range(start, end + 1))
|
|
627
|
+
for start, end in symbol.line_ranges:
|
|
628
|
+
for line in range(start, end + 1):
|
|
629
|
+
if line not in excluded:
|
|
630
|
+
index[symbol.path][line].add(symbol.id)
|
|
631
|
+
return index, tuple(sorted(changed_ids))
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
def shadowed_files(plan: Plan, outside: set[str]) -> list[tuple[str, str]]:
|
|
635
|
+
"""Measured files outside the checkout that have the same source-root-
|
|
636
|
+
relative path as a file inside it: the suite imported an installed copy
|
|
637
|
+
of the project instead of the checkout."""
|
|
638
|
+
suffixes: dict[str, str] = {}
|
|
639
|
+
for symbol in plan.head_index.symbols.values():
|
|
640
|
+
rel = symbol.path
|
|
641
|
+
suffixes["/" + rel] = rel
|
|
642
|
+
for spec in plan.source_roots:
|
|
643
|
+
root = split_root(spec)[0]
|
|
644
|
+
if root and rel.startswith(root + "/"):
|
|
645
|
+
suffixes["/" + rel[len(root) + 1 :]] = rel
|
|
646
|
+
found: list[tuple[str, str]] = []
|
|
647
|
+
for path in sorted(outside):
|
|
648
|
+
for suffix, rel in suffixes.items():
|
|
649
|
+
if path.endswith(suffix):
|
|
650
|
+
found.append((path, rel))
|
|
651
|
+
break
|
|
652
|
+
return found
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def coverage_validation(
|
|
656
|
+
plan: Plan, run: _SuiteRun, checkout: Path, selected: set[str], side: str = "head"
|
|
657
|
+
) -> CoverageValidation:
|
|
658
|
+
"""Attribute one suite run's per-test coverage to the changed symbols as
|
|
659
|
+
they are at ``side`` of the plan."""
|
|
660
|
+
assert run.coverage_db is not None
|
|
661
|
+
outside: set[str] = set()
|
|
662
|
+
contexts = read_coverage_contexts(run.coverage_db, checkout, outside)
|
|
663
|
+
shadowing = shadowed_files(plan, outside)
|
|
664
|
+
if shadowing:
|
|
665
|
+
listing = "\n".join(f" {path} (shadows {rel})" for path, rel in shadowing[:5])
|
|
666
|
+
raise GitError(
|
|
667
|
+
"the suite imported project code from outside the checkout, so the validation "
|
|
668
|
+
"would test the wrong revision:\n"
|
|
669
|
+
f"{listing}\nPut the checkout's source roots first on PYTHONPATH (diffcone does "
|
|
670
|
+
"this for --source-root entries; pass the roots that hold the package) or run "
|
|
671
|
+
"against an environment without an installed copy of the project."
|
|
672
|
+
)
|
|
673
|
+
if not contexts:
|
|
674
|
+
raise GitError(
|
|
675
|
+
f"coverage validation recorded no per-test contexts: the {side} suite collected "
|
|
676
|
+
f"no tests under coverage (exit code {run.returncode})\n{run.log[-2000:]}"
|
|
677
|
+
)
|
|
678
|
+
owners, changed_ids = _line_owner_index(plan, side)
|
|
679
|
+
hits: list[CoverageHit] = []
|
|
680
|
+
for nodeid in sorted(contexts):
|
|
681
|
+
executed: set[str] = set()
|
|
682
|
+
for rel, lines in contexts[nodeid].items():
|
|
683
|
+
by_line = owners.get(rel)
|
|
684
|
+
if by_line:
|
|
685
|
+
for line in lines:
|
|
686
|
+
executed.update(by_line.get(line, ()))
|
|
687
|
+
hits.append(CoverageHit(nodeid, nodeid in selected, tuple(sorted(executed))))
|
|
688
|
+
return CoverageValidation(hits, changed_ids, run.log)
|
|
689
|
+
|
|
690
|
+
|
|
691
|
+
def merge_coverage(
|
|
692
|
+
head: CoverageValidation, base: CoverageValidation | None, head_tests: set[str]
|
|
693
|
+
) -> CoverageValidation:
|
|
694
|
+
"""One record per test with the changed symbols it executed at either
|
|
695
|
+
side. A test that no longer exists at head is ``removed`` in the outcome
|
|
696
|
+
comparison and nothing could select it, so its base-side hits are
|
|
697
|
+
dropped rather than counted as misses."""
|
|
698
|
+
if base is None:
|
|
699
|
+
return head
|
|
700
|
+
hits = {h.runner_id: h for h in head.hits}
|
|
701
|
+
for hit in base.hits:
|
|
702
|
+
if hit.runner_id not in hits and hit.runner_id not in head_tests:
|
|
703
|
+
continue
|
|
704
|
+
existing = hits.get(hit.runner_id)
|
|
705
|
+
executed = set(hit.executed_changed) | set(existing.executed_changed if existing else ())
|
|
706
|
+
hits[hit.runner_id] = CoverageHit(
|
|
707
|
+
hit.runner_id,
|
|
708
|
+
existing.selected if existing else hit.selected,
|
|
709
|
+
tuple(sorted(executed)),
|
|
710
|
+
)
|
|
711
|
+
return CoverageValidation(
|
|
712
|
+
[hits[k] for k in sorted(hits)],
|
|
713
|
+
tuple(sorted(set(head.changed_symbols) | set(base.changed_symbols))),
|
|
714
|
+
head.log,
|
|
715
|
+
)
|
|
716
|
+
|
|
717
|
+
|
|
718
|
+
# (commit sha, ran under coverage, PYTHONHASHSEED an evidence plan pinned)
|
|
719
|
+
_OutcomeKey = tuple[str, bool, str | None]
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
class OutcomeCache(dict[_OutcomeKey, dict[str, str]]):
|
|
723
|
+
"""(commit sha, ran under coverage, hash seed) -> per-test outcomes, plus the
|
|
724
|
+
coverage database of every snapshot that ran under coverage (copied into
|
|
725
|
+
a temporary directory that lives until ``close``), so a base that is not
|
|
726
|
+
run again can still be attributed for a later pair.
|
|
727
|
+
|
|
728
|
+
Outcomes measured under the coverage tracer are only comparable with
|
|
729
|
+
each other (tests that depend on recursion depth or timing can flip under
|
|
730
|
+
``sys.settrace``), hence the flag in the key. Shared between parallel
|
|
731
|
+
corpus jobs with a no-wait policy: a snapshot that is already present is
|
|
732
|
+
reused, otherwise the requester runs it itself and offers the result
|
|
733
|
+
(set-if-absent). Nobody ever waits on another job, because in a linear
|
|
734
|
+
history every pair's base is the previous pair's head and waiting would
|
|
735
|
+
serialise the whole corpus; the price is at most one extra suite run per
|
|
736
|
+
pair when two jobs need the same snapshot at the same time.
|
|
737
|
+
"""
|
|
738
|
+
|
|
739
|
+
def __init__(self) -> None:
|
|
740
|
+
super().__init__()
|
|
741
|
+
self._lock = threading.Lock()
|
|
742
|
+
self._databases: dict[_OutcomeKey, tuple[Path, Path]] = {} # key -> (db, checkout)
|
|
743
|
+
self._directory: tempfile.TemporaryDirectory | None = None
|
|
744
|
+
|
|
745
|
+
def keep_coverage(self, key: _OutcomeKey, db: Path, checkout: Path) -> None:
|
|
746
|
+
"""Copy a run's coverage database so it outlives its worktree
|
|
747
|
+
(``checkout`` is remembered to resolve the paths it recorded)."""
|
|
748
|
+
with self._lock:
|
|
749
|
+
if key in self._databases:
|
|
750
|
+
return
|
|
751
|
+
if self._directory is None:
|
|
752
|
+
self._directory = tempfile.TemporaryDirectory(prefix="diffcone-basecov-")
|
|
753
|
+
copy = Path(self._directory.name) / f"{key[0]}-{int(key[1])}-{key[2]}.coverage"
|
|
754
|
+
shutil.copyfile(db, copy)
|
|
755
|
+
self._databases[key] = (copy, checkout)
|
|
756
|
+
|
|
757
|
+
def coverage_of(self, key: _OutcomeKey) -> tuple[Path, Path] | None:
|
|
758
|
+
with self._lock:
|
|
759
|
+
return self._databases.get(key)
|
|
760
|
+
|
|
761
|
+
def close(self) -> None:
|
|
762
|
+
with self._lock:
|
|
763
|
+
if self._directory is not None:
|
|
764
|
+
self._directory.cleanup()
|
|
765
|
+
self._directory = None
|
|
766
|
+
self._databases.clear()
|
|
767
|
+
|
|
768
|
+
def lookup(self, key: _OutcomeKey) -> dict[str, str] | None:
|
|
769
|
+
with self._lock:
|
|
770
|
+
return self.get(key)
|
|
771
|
+
|
|
772
|
+
def offer(self, key: _OutcomeKey, value: dict[str, str]) -> None:
|
|
773
|
+
with self._lock:
|
|
774
|
+
self.setdefault(key, value)
|
|
775
|
+
|
|
776
|
+
def reuse_or_run(self, key: _OutcomeKey, run) -> dict[str, str]:
|
|
777
|
+
value = self.lookup(key)
|
|
778
|
+
if value is not None:
|
|
779
|
+
return value
|
|
780
|
+
value = run()
|
|
781
|
+
self.offer(key, value)
|
|
782
|
+
return value
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
def validate_pytest(
|
|
786
|
+
plan: Plan,
|
|
787
|
+
*,
|
|
788
|
+
repo: Path,
|
|
789
|
+
command: str | None = None,
|
|
790
|
+
coverage: bool = False,
|
|
791
|
+
outcome_cache: OutcomeCache | None = None,
|
|
792
|
+
setup_command: str | None = None,
|
|
793
|
+
) -> Validation:
|
|
794
|
+
"""Run the full suite at base and head and compare outcome changes with
|
|
795
|
+
the plan's pytest selection; with ``coverage`` the head run also records
|
|
796
|
+
per-test coverage and every test that executed a changed symbol must be
|
|
797
|
+
selected. ``outcome_cache`` lets consecutive validations (a corpus) reuse
|
|
798
|
+
a committed snapshot's outcomes instead of running its suite again."""
|
|
799
|
+
if plan.head.kind == KIND_WORKTREE and plan.base.kind == KIND_WORKTREE:
|
|
800
|
+
raise GitError("validate needs at least one committed snapshot")
|
|
801
|
+
command = resolve_command(command, repo)
|
|
802
|
+
selected = {
|
|
803
|
+
d.target.runner_id for d in plan.decisions if d.selected and d.target.runner == "pytest"
|
|
804
|
+
}
|
|
805
|
+
cache = outcome_cache if outcome_cache is not None else OutcomeCache()
|
|
806
|
+
try:
|
|
807
|
+
return _validate_pytest(plan, repo, command, coverage, cache, setup_command, selected)
|
|
808
|
+
finally:
|
|
809
|
+
if outcome_cache is None:
|
|
810
|
+
cache.close()
|
|
811
|
+
|
|
812
|
+
|
|
813
|
+
def _validate_pytest(
|
|
814
|
+
plan: Plan,
|
|
815
|
+
repo: Path,
|
|
816
|
+
command: str | None,
|
|
817
|
+
coverage: bool,
|
|
818
|
+
cache: OutcomeCache,
|
|
819
|
+
setup_command: str | None,
|
|
820
|
+
selected: set[str],
|
|
821
|
+
) -> Validation:
|
|
822
|
+
seed = (plan.evidence or {}).get("hash_seed")
|
|
823
|
+
base_key = (plan.base.commit, coverage, seed)
|
|
824
|
+
base_log = ""
|
|
825
|
+
|
|
826
|
+
def run_base() -> dict[str, str]:
|
|
827
|
+
# Same instrumentation on both sides: with --coverage the base suite
|
|
828
|
+
# also runs under the tracer, and its database is kept so tests that
|
|
829
|
+
# executed a symbol deleted in head can be attributed too.
|
|
830
|
+
nonlocal base_log
|
|
831
|
+
with _Checkout(repo, plan.base.kind, plan.base.commit, setup_command) as base_dir:
|
|
832
|
+
with _run_full_pytest(
|
|
833
|
+
base_dir,
|
|
834
|
+
command,
|
|
835
|
+
coverage=coverage,
|
|
836
|
+
source_roots=plan.source_roots,
|
|
837
|
+
hash_seed=seed,
|
|
838
|
+
) as base_run:
|
|
839
|
+
base_log = base_run.log
|
|
840
|
+
if coverage and base_run.coverage_db is not None:
|
|
841
|
+
cache.keep_coverage(base_key, base_run.coverage_db, base_dir)
|
|
842
|
+
return base_run.outcomes
|
|
843
|
+
|
|
844
|
+
if plan.base.kind == KIND_COMMIT:
|
|
845
|
+
base_outcomes = cache.reuse_or_run(base_key, run_base)
|
|
846
|
+
else:
|
|
847
|
+
base_outcomes = run_base()
|
|
848
|
+
cov: CoverageValidation | None = None
|
|
849
|
+
# The head always runs here (its coverage database is needed); its
|
|
850
|
+
# outcomes are offered to the cache for pairs that use it as a base.
|
|
851
|
+
with _Checkout(repo, plan.head.kind, plan.head.commit, setup_command) as head_dir:
|
|
852
|
+
with _run_full_pytest(
|
|
853
|
+
head_dir,
|
|
854
|
+
command,
|
|
855
|
+
coverage=coverage,
|
|
856
|
+
source_roots=plan.source_roots,
|
|
857
|
+
hash_seed=seed,
|
|
858
|
+
) as head_run:
|
|
859
|
+
head_outcomes, head_log = head_run.outcomes, head_run.log
|
|
860
|
+
if coverage:
|
|
861
|
+
cov = coverage_validation(plan, head_run, head_dir, selected)
|
|
862
|
+
if head_run.coverage_db is not None and plan.head.kind == KIND_COMMIT:
|
|
863
|
+
cache.keep_coverage(
|
|
864
|
+
(plan.head.commit, coverage, seed), head_run.coverage_db, head_dir
|
|
865
|
+
)
|
|
866
|
+
if plan.head.kind == KIND_COMMIT:
|
|
867
|
+
cache.offer((plan.head.commit, coverage, seed), head_outcomes)
|
|
868
|
+
if cov is not None:
|
|
869
|
+
kept = cache.coverage_of(base_key)
|
|
870
|
+
base_cov: CoverageValidation | None = None
|
|
871
|
+
if kept is not None:
|
|
872
|
+
db, base_checkout = kept
|
|
873
|
+
base_run = _SuiteRun(base_outcomes, base_log, 0, db)
|
|
874
|
+
base_cov = coverage_validation(plan, base_run, base_checkout, selected, side="base")
|
|
875
|
+
head_tests = {fold_nodeid(n) for n in head_outcomes}
|
|
876
|
+
cov = merge_coverage(cov, base_cov, head_tests)
|
|
877
|
+
known = {d.target.runner_id for d in plan.decisions if d.target.runner == "pytest"}
|
|
878
|
+
ids = sorted(known | set(base_outcomes) | set(head_outcomes))
|
|
879
|
+
validation = Validation(
|
|
880
|
+
"pytest",
|
|
881
|
+
[*shlex.split(command or DEFAULT_COMMANDS["pytest"]), "-v"],
|
|
882
|
+
base_log=base_log,
|
|
883
|
+
head_log=head_log,
|
|
884
|
+
coverage=cov,
|
|
885
|
+
)
|
|
886
|
+
for runner_id in ids:
|
|
887
|
+
validation.outcomes.append(
|
|
888
|
+
TargetOutcome(
|
|
889
|
+
runner_id,
|
|
890
|
+
base_outcomes.get(runner_id),
|
|
891
|
+
head_outcomes.get(runner_id),
|
|
892
|
+
runner_id in selected,
|
|
893
|
+
known_at_head=runner_id in known,
|
|
894
|
+
)
|
|
895
|
+
)
|
|
896
|
+
return validation
|
|
897
|
+
|
|
898
|
+
|
|
899
|
+
def validation_to_dict(v: Validation) -> dict:
|
|
900
|
+
return {
|
|
901
|
+
"runner": v.runner,
|
|
902
|
+
"ok": v.ok,
|
|
903
|
+
"counts": {
|
|
904
|
+
"targets": len(v.outcomes),
|
|
905
|
+
"selected": v.selected_count,
|
|
906
|
+
"outcome_changed": sum(1 for o in v.outcomes if o.changed),
|
|
907
|
+
"caught": len(v.caught),
|
|
908
|
+
"missed": len(v.missed),
|
|
909
|
+
},
|
|
910
|
+
"missed": [{"runner_id": o.runner_id, "base": o.base, "head": o.head} for o in v.missed],
|
|
911
|
+
"removed": [o.runner_id for o in v.removed],
|
|
912
|
+
"caught": [{"runner_id": o.runner_id, "base": o.base, "head": o.head} for o in v.caught],
|
|
913
|
+
"outcomes": [
|
|
914
|
+
{"runner_id": o.runner_id, "base": o.base, "head": o.head, "selected": o.selected}
|
|
915
|
+
for o in v.outcomes
|
|
916
|
+
],
|
|
917
|
+
"coverage": None
|
|
918
|
+
if v.coverage is None
|
|
919
|
+
else {
|
|
920
|
+
"changed_symbols": list(v.coverage.changed_symbols),
|
|
921
|
+
"counts": {
|
|
922
|
+
"tests": len(v.coverage.hits),
|
|
923
|
+
"executed_a_changed_symbol": len(v.coverage.affected),
|
|
924
|
+
"caught": len(v.coverage.caught),
|
|
925
|
+
"missed": len(v.coverage.missed),
|
|
926
|
+
},
|
|
927
|
+
"recall": v.coverage.recall,
|
|
928
|
+
"precision": v.coverage.precision,
|
|
929
|
+
"missed": [
|
|
930
|
+
{"runner_id": h.runner_id, "executed": list(h.executed_changed)}
|
|
931
|
+
for h in v.coverage.missed
|
|
932
|
+
],
|
|
933
|
+
"hits": [
|
|
934
|
+
{
|
|
935
|
+
"runner_id": h.runner_id,
|
|
936
|
+
"selected": h.selected,
|
|
937
|
+
"executed": list(h.executed_changed),
|
|
938
|
+
}
|
|
939
|
+
for h in v.coverage.hits
|
|
940
|
+
],
|
|
941
|
+
},
|
|
942
|
+
}
|
|
943
|
+
|
|
944
|
+
|
|
945
|
+
def _status_label(v: Validation) -> str:
|
|
946
|
+
if v.ok:
|
|
947
|
+
return "OK"
|
|
948
|
+
reasons = []
|
|
949
|
+
if v.missed:
|
|
950
|
+
reasons.append("outcome")
|
|
951
|
+
if v.coverage is not None and v.coverage.missed:
|
|
952
|
+
reasons.append("coverage")
|
|
953
|
+
return f"MISSED ({', '.join(reasons)})"
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
def validation_to_text(v: Validation) -> str:
|
|
957
|
+
lines = [
|
|
958
|
+
f"validation ({v.runner}): {_status_label(v)}",
|
|
959
|
+
f" targets: {len(v.outcomes)}, selected: {v.selected_count}, "
|
|
960
|
+
f"outcome changed: {sum(1 for o in v.outcomes if o.changed)} "
|
|
961
|
+
f"(caught {len(v.caught)}, missed {len(v.missed)})",
|
|
962
|
+
]
|
|
963
|
+
for o in v.missed:
|
|
964
|
+
lines.append(f" MISSED {o.runner_id}: {o.base} -> {o.head}")
|
|
965
|
+
if v.removed:
|
|
966
|
+
lines.append(f" removed at head (not misses): {len(v.removed)}")
|
|
967
|
+
for o in v.caught:
|
|
968
|
+
lines.append(f" caught {o.runner_id}: {o.base} -> {o.head}")
|
|
969
|
+
if v.coverage is None:
|
|
970
|
+
lines.append(
|
|
971
|
+
" note: outcome-based only; behaviour changes that keep the same pass/fail outcome "
|
|
972
|
+
"are invisible to this check (use --coverage)"
|
|
973
|
+
)
|
|
974
|
+
else:
|
|
975
|
+
c = v.coverage
|
|
976
|
+
fmt = lambda x: "n/a" if x is None else f"{x:.0%}" # noqa: E731
|
|
977
|
+
lines.append(
|
|
978
|
+
f"coverage: {len(c.affected)} of {len(c.hits)} test(s) executed a changed symbol; "
|
|
979
|
+
f"recall {fmt(c.recall)} (caught {len(c.caught)}, missed {len(c.missed)}), "
|
|
980
|
+
f"precision {fmt(c.precision)}"
|
|
981
|
+
)
|
|
982
|
+
for h in c.missed:
|
|
983
|
+
lines.append(f" MISSED {h.runner_id}: executed {', '.join(h.executed_changed)}")
|
|
984
|
+
return "\n".join(lines) + "\n"
|
|
985
|
+
|
|
986
|
+
|
|
987
|
+
# --------------------------------------------------------------------------- corpus
|
|
988
|
+
|
|
989
|
+
|
|
990
|
+
@dataclass
|
|
991
|
+
class CorpusEntry:
|
|
992
|
+
commit: str
|
|
993
|
+
parent: str
|
|
994
|
+
subject: str
|
|
995
|
+
skipped: str | None = None # reason, when no validation ran
|
|
996
|
+
error: str | None = None
|
|
997
|
+
changed_symbols: int = 0
|
|
998
|
+
targets: int = 0
|
|
999
|
+
selected: int = 0
|
|
1000
|
+
degraded: bool = False
|
|
1001
|
+
validation: Validation | None = None
|
|
1002
|
+
|
|
1003
|
+
@property
|
|
1004
|
+
def savings(self) -> float | None:
|
|
1005
|
+
return 1 - self.selected / self.targets if self.targets else None
|
|
1006
|
+
|
|
1007
|
+
|
|
1008
|
+
@dataclass
|
|
1009
|
+
class CorpusReport:
|
|
1010
|
+
repo: str
|
|
1011
|
+
revision_range: str
|
|
1012
|
+
coverage: bool
|
|
1013
|
+
entries: list[CorpusEntry] = field(default_factory=list)
|
|
1014
|
+
|
|
1015
|
+
@property
|
|
1016
|
+
def validated(self) -> list[CorpusEntry]:
|
|
1017
|
+
return [e for e in self.entries if e.validation is not None]
|
|
1018
|
+
|
|
1019
|
+
def _sum(self, pick) -> int:
|
|
1020
|
+
return sum(pick(e.validation) for e in self.validated)
|
|
1021
|
+
|
|
1022
|
+
@property
|
|
1023
|
+
def outcome_changed(self) -> int:
|
|
1024
|
+
return self._sum(lambda v: sum(1 for o in v.outcomes if o.changed))
|
|
1025
|
+
|
|
1026
|
+
@property
|
|
1027
|
+
def outcome_missed(self) -> int:
|
|
1028
|
+
return self._sum(lambda v: len(v.missed))
|
|
1029
|
+
|
|
1030
|
+
@property
|
|
1031
|
+
def coverage_affected(self) -> int:
|
|
1032
|
+
return self._sum(lambda v: len(v.coverage.affected) if v.coverage else 0)
|
|
1033
|
+
|
|
1034
|
+
@property
|
|
1035
|
+
def coverage_caught(self) -> int:
|
|
1036
|
+
return self._sum(lambda v: len(v.coverage.caught) if v.coverage else 0)
|
|
1037
|
+
|
|
1038
|
+
@property
|
|
1039
|
+
def coverage_selected(self) -> int:
|
|
1040
|
+
return self._sum(
|
|
1041
|
+
lambda v: sum(1 for h in v.coverage.hits if h.selected) if v.coverage else 0
|
|
1042
|
+
)
|
|
1043
|
+
|
|
1044
|
+
@property
|
|
1045
|
+
def recall(self) -> float | None:
|
|
1046
|
+
return self.coverage_caught / self.coverage_affected if self.coverage_affected else None
|
|
1047
|
+
|
|
1048
|
+
@property
|
|
1049
|
+
def precision(self) -> float | None:
|
|
1050
|
+
return self.coverage_caught / self.coverage_selected if self.coverage_selected else None
|
|
1051
|
+
|
|
1052
|
+
@property
|
|
1053
|
+
def mean_savings(self) -> float | None:
|
|
1054
|
+
values = [e.savings for e in self.validated if e.savings is not None]
|
|
1055
|
+
return sum(values) / len(values) if values else None
|
|
1056
|
+
|
|
1057
|
+
@property
|
|
1058
|
+
def ok(self) -> bool:
|
|
1059
|
+
return all(e.validation.ok for e in self.validated if e.validation) and not any(
|
|
1060
|
+
e.error for e in self.entries
|
|
1061
|
+
)
|
|
1062
|
+
|
|
1063
|
+
|
|
1064
|
+
def _commits_in_range(repo: Path, revision_range: str) -> list[tuple[str, str, str]]:
|
|
1065
|
+
"""(commit, first parent, subject) for each commit in ``A..B``, oldest first."""
|
|
1066
|
+
out = _git(
|
|
1067
|
+
repo,
|
|
1068
|
+
[
|
|
1069
|
+
"rev-list",
|
|
1070
|
+
"--reverse",
|
|
1071
|
+
"--first-parent",
|
|
1072
|
+
"--format=%H %P%x00%s",
|
|
1073
|
+
"--no-commit-header",
|
|
1074
|
+
revision_range,
|
|
1075
|
+
],
|
|
1076
|
+
)
|
|
1077
|
+
entries: list[tuple[str, str, str]] = []
|
|
1078
|
+
for line in out.decode("utf-8", "replace").splitlines():
|
|
1079
|
+
if not line.strip():
|
|
1080
|
+
continue
|
|
1081
|
+
ids, _, subject = line.partition("\0")
|
|
1082
|
+
parts = ids.split()
|
|
1083
|
+
if len(parts) < 2:
|
|
1084
|
+
continue # a root commit has no parent to compare against
|
|
1085
|
+
entries.append((parts[0], parts[1], subject.strip()))
|
|
1086
|
+
return entries
|
|
1087
|
+
|
|
1088
|
+
|
|
1089
|
+
def _touches_python(repo: Path, parent: str, commit: str) -> bool:
|
|
1090
|
+
out = _git(repo, ["diff", "--name-only", parent, commit])
|
|
1091
|
+
return any(line.endswith(".py") for line in out.decode("utf-8", "replace").splitlines())
|
|
1092
|
+
|
|
1093
|
+
|
|
1094
|
+
def corpus_validation(
|
|
1095
|
+
repo: Path,
|
|
1096
|
+
revision_range: str,
|
|
1097
|
+
make_plan,
|
|
1098
|
+
*,
|
|
1099
|
+
command: str | None = None,
|
|
1100
|
+
coverage: bool = False,
|
|
1101
|
+
only_python_changes: bool = True,
|
|
1102
|
+
max_commits: int | None = None,
|
|
1103
|
+
progress=None,
|
|
1104
|
+
setup_command: str | None = None,
|
|
1105
|
+
jobs: int = 1,
|
|
1106
|
+
) -> CorpusReport:
|
|
1107
|
+
"""Plan and validate every ``parent -> commit`` pair in ``revision_range``.
|
|
1108
|
+
|
|
1109
|
+
``make_plan(base, head)`` builds the plan (the CLI binds discovery,
|
|
1110
|
+
manifest and source roots into it). Suites run once per commit thanks to
|
|
1111
|
+
the shared outcome cache; coverage runs are per pair.
|
|
1112
|
+
"""
|
|
1113
|
+
report = CorpusReport(str(repo), revision_range, coverage)
|
|
1114
|
+
cache = OutcomeCache()
|
|
1115
|
+
commits = _commits_in_range(repo, revision_range)
|
|
1116
|
+
if max_commits is not None:
|
|
1117
|
+
commits = commits[-max_commits:]
|
|
1118
|
+
entries: list[CorpusEntry] = []
|
|
1119
|
+
for commit, parent, subject in commits:
|
|
1120
|
+
entry = CorpusEntry(commit, parent, subject)
|
|
1121
|
+
report.entries.append(entry)
|
|
1122
|
+
if only_python_changes and not _touches_python(repo, parent, commit):
|
|
1123
|
+
entry.skipped = "no .py files changed"
|
|
1124
|
+
if progress:
|
|
1125
|
+
progress(entry) # reported as skipped, in order
|
|
1126
|
+
else:
|
|
1127
|
+
entries.append(entry)
|
|
1128
|
+
|
|
1129
|
+
def validate_entry(entry: CorpusEntry) -> None:
|
|
1130
|
+
if progress:
|
|
1131
|
+
progress(entry)
|
|
1132
|
+
try:
|
|
1133
|
+
plan = make_plan(entry.parent, entry.commit)
|
|
1134
|
+
entry.changed_symbols = len(plan.changes)
|
|
1135
|
+
entry.targets = sum(1 for d in plan.decisions if d.target.runner == "pytest")
|
|
1136
|
+
entry.selected = sum(
|
|
1137
|
+
1 for d in plan.decisions if d.selected and d.target.runner == "pytest"
|
|
1138
|
+
)
|
|
1139
|
+
entry.degraded = plan.degraded
|
|
1140
|
+
entry.validation = validate_pytest(
|
|
1141
|
+
plan,
|
|
1142
|
+
repo=repo,
|
|
1143
|
+
command=command,
|
|
1144
|
+
coverage=coverage,
|
|
1145
|
+
outcome_cache=cache,
|
|
1146
|
+
setup_command=setup_command,
|
|
1147
|
+
)
|
|
1148
|
+
except GitError as exc:
|
|
1149
|
+
entry.error = str(exc)
|
|
1150
|
+
|
|
1151
|
+
try:
|
|
1152
|
+
if jobs <= 1:
|
|
1153
|
+
for entry in entries:
|
|
1154
|
+
validate_entry(entry)
|
|
1155
|
+
else:
|
|
1156
|
+
# Pairs are independent: each validates in its own temporary
|
|
1157
|
+
# worktrees and coverage database; the outcome cache is shared with
|
|
1158
|
+
# a no-wait policy (see OutcomeCache). Any exception, including
|
|
1159
|
+
# Ctrl-C, cancels the pairs that have not started instead of
|
|
1160
|
+
# letting them run on.
|
|
1161
|
+
pool = ThreadPoolExecutor(max_workers=jobs)
|
|
1162
|
+
try:
|
|
1163
|
+
futures = [pool.submit(validate_entry, entry) for entry in entries]
|
|
1164
|
+
for future in futures:
|
|
1165
|
+
future.result()
|
|
1166
|
+
except BaseException:
|
|
1167
|
+
pool.shutdown(wait=False, cancel_futures=True)
|
|
1168
|
+
raise
|
|
1169
|
+
pool.shutdown(wait=True)
|
|
1170
|
+
finally:
|
|
1171
|
+
cache.close() # the kept coverage databases
|
|
1172
|
+
return report
|
|
1173
|
+
|
|
1174
|
+
|
|
1175
|
+
def corpus_to_dict(report: CorpusReport) -> dict:
|
|
1176
|
+
def entry(e: CorpusEntry) -> dict:
|
|
1177
|
+
d: dict = {
|
|
1178
|
+
"commit": e.commit,
|
|
1179
|
+
"parent": e.parent,
|
|
1180
|
+
"subject": e.subject,
|
|
1181
|
+
"skipped": e.skipped,
|
|
1182
|
+
"error": e.error,
|
|
1183
|
+
"changed_symbols": e.changed_symbols,
|
|
1184
|
+
"targets": e.targets,
|
|
1185
|
+
"selected": e.selected,
|
|
1186
|
+
"savings": e.savings,
|
|
1187
|
+
"degraded": e.degraded,
|
|
1188
|
+
}
|
|
1189
|
+
if e.validation is not None:
|
|
1190
|
+
v = validation_to_dict(e.validation)
|
|
1191
|
+
d["ok"] = v["ok"]
|
|
1192
|
+
d["outcome"] = v["counts"]
|
|
1193
|
+
d["coverage"] = v["coverage"] and {
|
|
1194
|
+
k: v["coverage"][k] for k in ("counts", "recall", "precision", "missed")
|
|
1195
|
+
}
|
|
1196
|
+
d["missed"] = v["missed"]
|
|
1197
|
+
return d
|
|
1198
|
+
|
|
1199
|
+
return {
|
|
1200
|
+
"repo": report.repo,
|
|
1201
|
+
"range": report.revision_range,
|
|
1202
|
+
"coverage": report.coverage,
|
|
1203
|
+
"ok": report.ok,
|
|
1204
|
+
"totals": {
|
|
1205
|
+
"commits": len(report.entries),
|
|
1206
|
+
"validated": len(report.validated),
|
|
1207
|
+
"skipped": sum(1 for e in report.entries if e.skipped),
|
|
1208
|
+
"errors": sum(1 for e in report.entries if e.error),
|
|
1209
|
+
"outcome_changed": report.outcome_changed,
|
|
1210
|
+
"outcome_missed": report.outcome_missed,
|
|
1211
|
+
"coverage_affected": report.coverage_affected,
|
|
1212
|
+
"coverage_caught": report.coverage_caught,
|
|
1213
|
+
"recall": report.recall,
|
|
1214
|
+
"precision": report.precision,
|
|
1215
|
+
"mean_savings": report.mean_savings,
|
|
1216
|
+
},
|
|
1217
|
+
"entries": [entry(e) for e in report.entries],
|
|
1218
|
+
}
|
|
1219
|
+
|
|
1220
|
+
|
|
1221
|
+
def _corpus_verdict(report: CorpusReport) -> str:
|
|
1222
|
+
"""OK, MISSES when a plan missed something, ERRORS when nothing was
|
|
1223
|
+
missed but some commits could not be validated (a suite that does not
|
|
1224
|
+
run in this environment is not a miss, and not a pass either)."""
|
|
1225
|
+
if report.ok:
|
|
1226
|
+
return "OK"
|
|
1227
|
+
missed = report.outcome_missed or any(
|
|
1228
|
+
e.validation is not None and not e.validation.ok for e in report.entries
|
|
1229
|
+
)
|
|
1230
|
+
return "MISSES" if missed else "ERRORS"
|
|
1231
|
+
|
|
1232
|
+
|
|
1233
|
+
def corpus_to_text(report: CorpusReport) -> str:
|
|
1234
|
+
def pct(x: float | None) -> str:
|
|
1235
|
+
return "n/a" if x is None else f"{x:.0%}"
|
|
1236
|
+
|
|
1237
|
+
lines = [
|
|
1238
|
+
f"corpus {report.revision_range}: {len(report.validated)} validated, "
|
|
1239
|
+
f"{sum(1 for e in report.entries if e.skipped)} skipped, "
|
|
1240
|
+
f"{sum(1 for e in report.entries if e.error)} error(s); "
|
|
1241
|
+
f"{_corpus_verdict(report)}",
|
|
1242
|
+
f" outcome changes: {report.outcome_changed} (missed {report.outcome_missed}); "
|
|
1243
|
+
f"mean savings {pct(report.mean_savings)}",
|
|
1244
|
+
]
|
|
1245
|
+
if report.coverage:
|
|
1246
|
+
lines.append(
|
|
1247
|
+
f" coverage: recall {pct(report.recall)} ({report.coverage_caught} of "
|
|
1248
|
+
f"{report.coverage_affected}), precision {pct(report.precision)}"
|
|
1249
|
+
)
|
|
1250
|
+
for e in report.entries:
|
|
1251
|
+
short = e.commit[:10]
|
|
1252
|
+
if e.skipped:
|
|
1253
|
+
lines.append(f" {short} skipped: {e.skipped} {e.subject}")
|
|
1254
|
+
elif e.error:
|
|
1255
|
+
lines.append(f" {short} ERROR: {e.error.splitlines()[0]} {e.subject}")
|
|
1256
|
+
else:
|
|
1257
|
+
v = e.validation
|
|
1258
|
+
assert v is not None
|
|
1259
|
+
status = "ok" if v.ok else "MISSED"
|
|
1260
|
+
cov = ""
|
|
1261
|
+
if v.coverage is not None:
|
|
1262
|
+
cov = f", recall {pct(v.coverage.recall)}, precision {pct(v.coverage.precision)}"
|
|
1263
|
+
lines.append(
|
|
1264
|
+
f" {short} {status}: {e.selected}/{e.targets} selected "
|
|
1265
|
+
f"(savings {pct(e.savings)}), outcome misses {len(v.missed)}{cov}"
|
|
1266
|
+
f"{' [degraded]' if e.degraded else ''} {e.subject}"
|
|
1267
|
+
)
|
|
1268
|
+
for o in v.missed:
|
|
1269
|
+
lines.append(f" MISSED outcome {o.runner_id}: {o.base} -> {o.head}")
|
|
1270
|
+
if v.coverage is not None:
|
|
1271
|
+
for h in v.coverage.missed:
|
|
1272
|
+
lines.append(
|
|
1273
|
+
f" MISSED coverage {h.runner_id}: executed "
|
|
1274
|
+
f"{', '.join(h.executed_changed)}"
|
|
1275
|
+
)
|
|
1276
|
+
return "\n".join(lines) + "\n"
|
|
1277
|
+
|
|
1278
|
+
|
|
1279
|
+
# --------------------------------------------------------------------------- evidence
|
|
1280
|
+
|
|
1281
|
+
# The recorder's module name inside the project's process (see plugin_environment).
|
|
1282
|
+
PLUGIN = "diffcone_collect"
|
|
1283
|
+
SELECT_PLUGIN = "diffcone_select"
|
|
1284
|
+
|
|
1285
|
+
|
|
1286
|
+
def _python_module_names(index: SourceIndex, source_roots: list[str]) -> set[str]:
|
|
1287
|
+
"""The names Python imports the indexed modules by (a prefixed source
|
|
1288
|
+
root names modules diffcone's way, not Python's)."""
|
|
1289
|
+
names = set()
|
|
1290
|
+
dirs = sorted((split_root(r)[0] for r in source_roots), key=len, reverse=True)
|
|
1291
|
+
for symbol in index.symbols.values():
|
|
1292
|
+
if symbol.kind != MODULE:
|
|
1293
|
+
continue
|
|
1294
|
+
for d in dirs:
|
|
1295
|
+
if d in ("", ".") or symbol.path.startswith(d + "/"):
|
|
1296
|
+
rel = symbol.path if d in ("", ".") else symbol.path[len(d) + 1 :]
|
|
1297
|
+
parts = rel[: -len(".py")].split("/")
|
|
1298
|
+
if parts[-1] == "__init__":
|
|
1299
|
+
parts.pop()
|
|
1300
|
+
if parts:
|
|
1301
|
+
names.add(".".join(parts))
|
|
1302
|
+
break
|
|
1303
|
+
return names
|
|
1304
|
+
|
|
1305
|
+
|
|
1306
|
+
def _collect_argv(command: str | None, extra: list[str] | None) -> list[str]:
|
|
1307
|
+
"""The suite under the recorder, as a store's ``command`` records it."""
|
|
1308
|
+
return [
|
|
1309
|
+
*shlex.split(command or DEFAULT_COMMANDS["pytest"]),
|
|
1310
|
+
"-p",
|
|
1311
|
+
PLUGIN,
|
|
1312
|
+
"-p",
|
|
1313
|
+
"no:cacheprovider",
|
|
1314
|
+
*(extra or []),
|
|
1315
|
+
]
|
|
1316
|
+
|
|
1317
|
+
|
|
1318
|
+
def advance_refusal(
|
|
1319
|
+
repo: Path, plan: Plan, evidence: Evidence, command: str | None, extra: list[str] | None
|
|
1320
|
+
) -> str | None:
|
|
1321
|
+
"""Why ``run --collect`` cannot advance ``evidence`` to the plan's head,
|
|
1322
|
+
if it cannot (roadmap item 6). A store names a commit, so head must be
|
|
1323
|
+
the clean checkout; and a fresh record replaces an old one only if the
|
|
1324
|
+
same arguments selected the same cases."""
|
|
1325
|
+
dirty = _dirty(repo)
|
|
1326
|
+
if dirty:
|
|
1327
|
+
return (
|
|
1328
|
+
f"the working tree has {len(dirty)} change(s) (e.g. {dirty[0]}); a recording "
|
|
1329
|
+
"describes a commit, so commit them first"
|
|
1330
|
+
)
|
|
1331
|
+
if plan.head.kind == KIND_COMMIT and plan.head.commit != resolve_commit(repo, "HEAD"):
|
|
1332
|
+
return f"head {plan.head.revision} is not the checked-out commit"
|
|
1333
|
+
if sorted(plan.source_roots) != sorted(evidence.source_roots):
|
|
1334
|
+
return (
|
|
1335
|
+
f"the plan's source roots {plan.source_roots} differ from the recording's "
|
|
1336
|
+
f"{evidence.source_roots}"
|
|
1337
|
+
)
|
|
1338
|
+
wanted = shlex.join(_collect_argv(command, extra))
|
|
1339
|
+
if wanted != evidence.command:
|
|
1340
|
+
return (
|
|
1341
|
+
"the pytest command and arguments differ from the ones the recording was "
|
|
1342
|
+
f"made with, so a new record could cover other cases:\n recording: "
|
|
1343
|
+
f"{evidence.command}\n now: {wanted}"
|
|
1344
|
+
)
|
|
1345
|
+
return None
|
|
1346
|
+
|
|
1347
|
+
|
|
1348
|
+
def _dirty(repo: Path) -> list[str]:
|
|
1349
|
+
"""Changes in the working tree that make it differ from its commit.
|
|
1350
|
+
Untracked output nobody runs does not count: Python's bytecode (in a
|
|
1351
|
+
repository that does not ignore it) and ASV's results, environments and
|
|
1352
|
+
HTML beside its configuration."""
|
|
1353
|
+
status = _git(repo, ["status", "--porcelain", "--untracked-files=normal"]).decode(
|
|
1354
|
+
"utf-8", "surrogateescape"
|
|
1355
|
+
)
|
|
1356
|
+
outputs = _asv_output_dirs(repo)
|
|
1357
|
+
found = []
|
|
1358
|
+
for line in status.splitlines():
|
|
1359
|
+
path = line[3:]
|
|
1360
|
+
if not path or path.startswith(str(EVIDENCE_DIR.parts[0]) + "/"):
|
|
1361
|
+
continue
|
|
1362
|
+
if line.startswith("??") and (
|
|
1363
|
+
is_bytecode(path) or any(path.startswith(d + "/") for d in outputs)
|
|
1364
|
+
):
|
|
1365
|
+
continue
|
|
1366
|
+
found.append(path)
|
|
1367
|
+
return found
|
|
1368
|
+
|
|
1369
|
+
|
|
1370
|
+
def _asv_output_dirs(repo: Path) -> list[str]:
|
|
1371
|
+
"""Where ASV writes beside each tracked ``asv.conf.json``: its
|
|
1372
|
+
``results_dir``, ``env_dir`` and ``html_dir`` (``results``, ``env`` and
|
|
1373
|
+
``html`` by default), repository-relative."""
|
|
1374
|
+
listed = _git(repo, ["ls-files", "-z", "--", "asv.conf.json", "*/asv.conf.json"])
|
|
1375
|
+
dirs: list[str] = []
|
|
1376
|
+
for raw in listed.split(b"\0"):
|
|
1377
|
+
if not raw:
|
|
1378
|
+
continue
|
|
1379
|
+
path = raw.decode("utf-8", "surrogateescape")
|
|
1380
|
+
here = posixpath.dirname(path)
|
|
1381
|
+
try:
|
|
1382
|
+
text = (repo / path).read_text("utf-8", "replace")
|
|
1383
|
+
data = json.loads(strip_json_comments(text))
|
|
1384
|
+
except (OSError, ValueError):
|
|
1385
|
+
data = {}
|
|
1386
|
+
if not isinstance(data, dict):
|
|
1387
|
+
data = {}
|
|
1388
|
+
for key, default in (("results_dir", "results"), ("env_dir", "env"), ("html_dir", "html")):
|
|
1389
|
+
value = data.get(key, default)
|
|
1390
|
+
if isinstance(value, str) and value.strip():
|
|
1391
|
+
dirs.append(posixpath.normpath(posixpath.join(here, value.strip())))
|
|
1392
|
+
return dirs
|
|
1393
|
+
|
|
1394
|
+
|
|
1395
|
+
def plugin_environment(base: dict[str, str], out: Path | None, root: Path) -> dict[str, str]:
|
|
1396
|
+
"""``base`` plus what loading ``-p diffcone_collect`` needs: the recorder
|
|
1397
|
+
importable as a module of its own (a link to ``collect.py`` in a
|
|
1398
|
+
directory of its own, so neither diffcone's package nor anything else of
|
|
1399
|
+
its environment is imported into the project's process), and the
|
|
1400
|
+
recorder's settings. The caller removes the directory (the last
|
|
1401
|
+
``PYTHONPATH`` entry) when the run ends."""
|
|
1402
|
+
env = dict(base)
|
|
1403
|
+
link_dir = Path(tempfile.mkdtemp(prefix="diffcone-plugin-"))
|
|
1404
|
+
(link_dir / f"{PLUGIN}.py").symlink_to(Path(__file__).parent / "collect.py")
|
|
1405
|
+
(link_dir / f"{SELECT_PLUGIN}.py").symlink_to(Path(__file__).parent / "selection.py")
|
|
1406
|
+
existing = env.get("PYTHONPATH")
|
|
1407
|
+
env["PYTHONPATH"] = os.pathsep.join([*([existing] if existing else []), str(link_dir)])
|
|
1408
|
+
# Absolute and with symlinks resolved, as pytest reports a test's path:
|
|
1409
|
+
# a relative or linked ``--repo`` would otherwise match nothing.
|
|
1410
|
+
env["DIFFCONE_COLLECT_ROOT"] = str(Path(root).resolve())
|
|
1411
|
+
if out is not None:
|
|
1412
|
+
env["DIFFCONE_COLLECT_OUT"] = str(out)
|
|
1413
|
+
return env
|
|
1414
|
+
|
|
1415
|
+
|
|
1416
|
+
@dataclass
|
|
1417
|
+
class CollectResult:
|
|
1418
|
+
evidence: Evidence
|
|
1419
|
+
store: Path
|
|
1420
|
+
command: list[str]
|
|
1421
|
+
returncode: int
|
|
1422
|
+
log: str
|
|
1423
|
+
|
|
1424
|
+
|
|
1425
|
+
def collect_evidence(
|
|
1426
|
+
repo: Path,
|
|
1427
|
+
*,
|
|
1428
|
+
command: str | None,
|
|
1429
|
+
source_roots: list[str],
|
|
1430
|
+
rev: str | None = None,
|
|
1431
|
+
setup_command: str | None = None,
|
|
1432
|
+
reverse_check: bool = False,
|
|
1433
|
+
extra: list[str] | None = None,
|
|
1434
|
+
cache: IndexCache | None = None,
|
|
1435
|
+
env_variables: list[str] | None = None,
|
|
1436
|
+
) -> CollectResult:
|
|
1437
|
+
"""Run the whole suite under the recorder at a commit and write a store.
|
|
1438
|
+
|
|
1439
|
+
Without ``rev`` the repository itself is run, and it must be clean: the
|
|
1440
|
+
evidence is keyed by a commit, so it must describe that commit's code.
|
|
1441
|
+
With ``rev`` a temporary worktree is checked out (and ``setup_command``
|
|
1442
|
+
run in it). ``PYTHONHASHSEED`` is pinned to 0 when unset, and recorded.
|
|
1443
|
+
With ``reverse_check`` the suite runs a second time in reverse order;
|
|
1444
|
+
tests whose records differ are marked unstable and always selected."""
|
|
1445
|
+
commit = resolve_commit(repo, rev or "HEAD")
|
|
1446
|
+
if rev is None:
|
|
1447
|
+
dirty = _dirty(repo)
|
|
1448
|
+
if dirty:
|
|
1449
|
+
raise GitError(
|
|
1450
|
+
f"the working tree has {len(dirty)} change(s) (e.g. {dirty[0]}); evidence "
|
|
1451
|
+
"describes a commit, so collect from a clean checkout or pass --rev"
|
|
1452
|
+
)
|
|
1453
|
+
index, _ = _index_snapshot(repo, commit, source_roots, with_config=False, cache=cache)
|
|
1454
|
+
if index.errors:
|
|
1455
|
+
raise EvidenceError(
|
|
1456
|
+
f"{len(index.errors)} analysis error(s) at {commit[:12]} (e.g. "
|
|
1457
|
+
f"{index.errors[0].path}: {index.errors[0].message}); code in those modules "
|
|
1458
|
+
"could not be mapped to symbols"
|
|
1459
|
+
)
|
|
1460
|
+
modules = _python_module_names(index, source_roots)
|
|
1461
|
+
argv = _collect_argv(command, extra)
|
|
1462
|
+
kind = KIND_COMMIT if rev is not None else KIND_WORKTREE
|
|
1463
|
+
with (
|
|
1464
|
+
_Checkout(repo, kind, commit, setup_command) as cwd,
|
|
1465
|
+
tempfile.TemporaryDirectory(prefix="diffcone-collect-") as tmp,
|
|
1466
|
+
):
|
|
1467
|
+
base_env = _checkout_env(cwd, source_roots)
|
|
1468
|
+
base_env["DIFFCONE_COLLECT_REPO"] = str(Path(repo).resolve())
|
|
1469
|
+
base_env.setdefault("PYTHONHASHSEED", "0")
|
|
1470
|
+
if env_variables:
|
|
1471
|
+
base_env["DIFFCONE_ENV_VARIABLES"] = ",".join(sorted(set(env_variables)))
|
|
1472
|
+
base_env["DIFFCONE_COLLECT_PACKAGES"] = ",".join(sorted({m.split(".")[0] for m in modules}))
|
|
1473
|
+
runs = [Path(tmp) / "forward"] + ([Path(tmp) / "reverse"] if reverse_check else [])
|
|
1474
|
+
logs, returncode = [], 0
|
|
1475
|
+
for out in runs:
|
|
1476
|
+
env = plugin_environment(base_env, out, cwd)
|
|
1477
|
+
if out.name == "reverse":
|
|
1478
|
+
env["DIFFCONE_COLLECT_REVERSE"] = "1"
|
|
1479
|
+
try:
|
|
1480
|
+
proc = subprocess.run(argv, cwd=cwd, capture_output=True, text=True, env=env)
|
|
1481
|
+
except OSError as exc:
|
|
1482
|
+
raise GitError(f"cannot run {argv[0]!r}: {exc}") from exc
|
|
1483
|
+
finally:
|
|
1484
|
+
shutil.rmtree(Path(env["PYTHONPATH"].split(os.pathsep)[-1]), ignore_errors=True)
|
|
1485
|
+
log = proc.stdout + proc.stderr
|
|
1486
|
+
logs.append(log)
|
|
1487
|
+
if proc.returncode not in (0, 1):
|
|
1488
|
+
raise EvidenceError(
|
|
1489
|
+
f"the suite did not run: {argv[0]!r} exited {proc.returncode} "
|
|
1490
|
+
f"({PYTEST_EXIT.get(proc.returncode, 'unknown')})\n{log[-2000:]}"
|
|
1491
|
+
)
|
|
1492
|
+
returncode = max(returncode, proc.returncode)
|
|
1493
|
+
try:
|
|
1494
|
+
evidence = fold(
|
|
1495
|
+
runs,
|
|
1496
|
+
index,
|
|
1497
|
+
commit=commit,
|
|
1498
|
+
source_roots=source_roots,
|
|
1499
|
+
command=shlex.join(argv),
|
|
1500
|
+
project_modules=modules,
|
|
1501
|
+
)
|
|
1502
|
+
except EvidenceError as exc:
|
|
1503
|
+
# What the runner said about a process that did not finish.
|
|
1504
|
+
crashes = [
|
|
1505
|
+
line for log in logs for line in log.splitlines() if _CRASH_LINE.search(line)
|
|
1506
|
+
]
|
|
1507
|
+
if crashes:
|
|
1508
|
+
raise EvidenceError(f"{exc}\n" + "\n".join(crashes[:10])) from exc
|
|
1509
|
+
raise
|
|
1510
|
+
store = write_store(evidence, repo / EVIDENCE_DIR)
|
|
1511
|
+
return CollectResult(evidence, store, argv, returncode, "\n".join(logs))
|
|
1512
|
+
|
|
1513
|
+
|
|
1514
|
+
@dataclass
|
|
1515
|
+
class EvidenceRun:
|
|
1516
|
+
result: RunResult
|
|
1517
|
+
# The environment the run met, when it differed from the evidence's:
|
|
1518
|
+
# the evidence plan was then not run, and the static one was.
|
|
1519
|
+
mismatch: dict | None = None
|
|
1520
|
+
static: RunResult | None = None
|
|
1521
|
+
# The static plan was incomplete, so the whole suite ran instead.
|
|
1522
|
+
static_whole: bool = False
|
|
1523
|
+
|
|
1524
|
+
@property
|
|
1525
|
+
def ran(self) -> RunResult:
|
|
1526
|
+
"""What actually ran: the static fallback when the environment differed."""
|
|
1527
|
+
return self.static if self.static is not None else self.result
|
|
1528
|
+
|
|
1529
|
+
# With ``advance``: the store written for head, or why none was.
|
|
1530
|
+
advanced: Path | None = None
|
|
1531
|
+
not_advanced: str | None = None
|
|
1532
|
+
|
|
1533
|
+
|
|
1534
|
+
def evidence_env(plan: Plan) -> dict[str, str]:
|
|
1535
|
+
"""What a run relying on an evidence plan changes in the environment:
|
|
1536
|
+
``PYTHONHASHSEED`` as recorded (the records assumed it), and the names of
|
|
1537
|
+
the variables the recording included, so the check compares the same."""
|
|
1538
|
+
env = dict(os.environ)
|
|
1539
|
+
seed = (plan.evidence or {}).get("hash_seed")
|
|
1540
|
+
if seed is not None:
|
|
1541
|
+
env["PYTHONHASHSEED"] = seed
|
|
1542
|
+
variables = (plan.evidence or {}).get("variables")
|
|
1543
|
+
if variables:
|
|
1544
|
+
env["DIFFCONE_ENV_VARIABLES"] = ",".join(variables)
|
|
1545
|
+
return env
|
|
1546
|
+
|
|
1547
|
+
|
|
1548
|
+
def run_with_evidence(
|
|
1549
|
+
plan: Plan,
|
|
1550
|
+
static_plan,
|
|
1551
|
+
*,
|
|
1552
|
+
cwd: Path,
|
|
1553
|
+
command: str | None = None,
|
|
1554
|
+
extra: list[str] | None = None,
|
|
1555
|
+
dry_run: bool = False,
|
|
1556
|
+
advance_from: Evidence | None = None,
|
|
1557
|
+
allow_incomplete: bool = False,
|
|
1558
|
+
) -> EvidenceRun:
|
|
1559
|
+
"""Run an evidence plan's pytest selection with the recorder in check
|
|
1560
|
+
mode: before any test runs, it compares the environment with the one the
|
|
1561
|
+
evidence was recorded in. On a mismatch the session stops, the evidence
|
|
1562
|
+
says nothing about this environment, and ``static_plan()`` (a static plan
|
|
1563
|
+
of the same snapshots) is run instead.
|
|
1564
|
+
|
|
1565
|
+
With ``advance_from`` (the plan's store; the caller has checked
|
|
1566
|
+
:func:`advance_refusal`) the recorder also records, and the store is
|
|
1567
|
+
advanced to head: the run's records for the selected tests, the old ones
|
|
1568
|
+
for the rest."""
|
|
1569
|
+
assert plan.evidence is not None
|
|
1570
|
+
with tempfile.TemporaryDirectory(prefix="diffcone-check-") as tmp:
|
|
1571
|
+
report = Path(tmp) / "environment.json"
|
|
1572
|
+
out = Path(tmp) / "records" if advance_from is not None else None
|
|
1573
|
+
env = plugin_environment(evidence_env(plan), out, cwd)
|
|
1574
|
+
env["DIFFCONE_CHECK_ENV"] = plan.evidence["environment_hash"]
|
|
1575
|
+
env["DIFFCONE_CHECK_REPORT"] = str(report)
|
|
1576
|
+
plugins = ["-p", PLUGIN]
|
|
1577
|
+
if advance_from is not None:
|
|
1578
|
+
modules = _python_module_names(plan.head_index, plan.source_roots)
|
|
1579
|
+
env["DIFFCONE_COLLECT_PACKAGES"] = ",".join(sorted({m.split(".")[0] for m in modules}))
|
|
1580
|
+
plugins += ["-p", "no:cacheprovider"]
|
|
1581
|
+
try:
|
|
1582
|
+
result = run_selected(
|
|
1583
|
+
plan,
|
|
1584
|
+
"pytest",
|
|
1585
|
+
cwd=cwd,
|
|
1586
|
+
command=command,
|
|
1587
|
+
extra=[*plugins, *(extra or [])],
|
|
1588
|
+
dry_run=dry_run,
|
|
1589
|
+
env=env,
|
|
1590
|
+
)
|
|
1591
|
+
if not dry_run and not result.selected:
|
|
1592
|
+
# Nothing selected is a verdict of the evidence too: check
|
|
1593
|
+
# that it applies to this environment before trusting it.
|
|
1594
|
+
argv = build_command("pytest", [], command, ["-p", PLUGIN, *(extra or [])])
|
|
1595
|
+
subprocess.run(argv, cwd=cwd, env={**env, "DIFFCONE_CHECK_ONLY": "1"})
|
|
1596
|
+
finally:
|
|
1597
|
+
shutil.rmtree(Path(env["PYTHONPATH"].split(os.pathsep)[-1]), ignore_errors=True)
|
|
1598
|
+
checked = json.loads(report.read_text("utf-8")) if report.exists() else None
|
|
1599
|
+
if dry_run or (checked is not None and checked.get("match")):
|
|
1600
|
+
run = EvidenceRun(result)
|
|
1601
|
+
if advance_from is not None and out is not None and not dry_run:
|
|
1602
|
+
_advance(run, plan, advance_from, out, cwd)
|
|
1603
|
+
return run
|
|
1604
|
+
# A mismatch, or no report at all (the plugin did not load, a wrapper
|
|
1605
|
+
# dropped the variables): the evidence does not vouch for this run.
|
|
1606
|
+
met = (
|
|
1607
|
+
checked["environment"]
|
|
1608
|
+
if checked is not None
|
|
1609
|
+
else {"unchecked": "the environment check did not run"}
|
|
1610
|
+
)
|
|
1611
|
+
static = static_plan()
|
|
1612
|
+
if static.incomplete_discovery and not allow_incomplete:
|
|
1613
|
+
# The evidence may have settled what the static plan cannot: pytest
|
|
1614
|
+
# may collect tests that are not targets, so run them all.
|
|
1615
|
+
return EvidenceRun(
|
|
1616
|
+
result, met, run_whole(static, cwd=cwd, command=command, extra=extra), True
|
|
1617
|
+
)
|
|
1618
|
+
fallback = run_selected(static, "pytest", cwd=cwd, command=command, extra=extra)
|
|
1619
|
+
return EvidenceRun(result, met, fallback)
|
|
1620
|
+
|
|
1621
|
+
|
|
1622
|
+
def run_whole(
|
|
1623
|
+
plan: Plan, *, cwd: Path, command: str | None = None, extra: list[str] | None = None
|
|
1624
|
+
) -> RunResult:
|
|
1625
|
+
"""Run the whole pytest suite (every target counts as selected)."""
|
|
1626
|
+
targets = [d.target for d in plan.decisions if d.target.runner == "pytest"]
|
|
1627
|
+
argv = build_command("pytest", [], command, list(extra or []))
|
|
1628
|
+
returncode = subprocess.run(argv, cwd=cwd).returncode
|
|
1629
|
+
return RunResult("pytest", argv, targets, len(targets), returncode)
|
|
1630
|
+
|
|
1631
|
+
|
|
1632
|
+
def _advance(run: EvidenceRun, plan: Plan, previous: Evidence, out: Path, repo: Path) -> None:
|
|
1633
|
+
"""Write the store for head after ``run`` (roadmap item 6), or say why not."""
|
|
1634
|
+
result = run.result
|
|
1635
|
+
if result.selected and result.returncode not in (0, 1):
|
|
1636
|
+
run.not_advanced = (
|
|
1637
|
+
f"pytest exited {result.returncode} "
|
|
1638
|
+
f"({PYTEST_EXIT.get(result.returncode or 0, 'unknown')}), so the records may be partial"
|
|
1639
|
+
)
|
|
1640
|
+
return
|
|
1641
|
+
if result.selected and not any(out.glob("process-*.json")):
|
|
1642
|
+
run.not_advanced = (
|
|
1643
|
+
f"pytest exited {result.returncode} before any test process finished its "
|
|
1644
|
+
"session (see its output above), so it recorded nothing"
|
|
1645
|
+
)
|
|
1646
|
+
return
|
|
1647
|
+
head = resolve_commit(repo, "HEAD")
|
|
1648
|
+
rerun = {t.runner_id for t in result.selected}
|
|
1649
|
+
fresh = None
|
|
1650
|
+
try:
|
|
1651
|
+
if result.selected:
|
|
1652
|
+
fresh = fold(
|
|
1653
|
+
[out],
|
|
1654
|
+
plan.head_index,
|
|
1655
|
+
commit=head,
|
|
1656
|
+
source_roots=previous.source_roots,
|
|
1657
|
+
command=previous.command,
|
|
1658
|
+
project_modules=_python_module_names(plan.head_index, plan.source_roots),
|
|
1659
|
+
)
|
|
1660
|
+
alive = {d.target.runner_id for d in plan.decisions if d.target.runner == "pytest"}
|
|
1661
|
+
advanced = advance(previous, fresh, rerun, head, alive)
|
|
1662
|
+
run.advanced = write_store(advanced, repo / EVIDENCE_DIR)
|
|
1663
|
+
except EvidenceError as exc:
|
|
1664
|
+
run.not_advanced = str(exc)
|
|
1665
|
+
|
|
1666
|
+
|
|
1667
|
+
def environment_differences(recorded: dict, met: dict) -> list[str]:
|
|
1668
|
+
"""Human-readable differences between two recorder environments."""
|
|
1669
|
+
out = []
|
|
1670
|
+
for key in ("implementation", "python", "platform", "machine"):
|
|
1671
|
+
if recorded.get(key) != met.get(key):
|
|
1672
|
+
out.append(f"{key}: {recorded.get(key)!r} recorded, {met.get(key)!r} now")
|
|
1673
|
+
before, after = set(recorded.get("distributions", ())), set(met.get("distributions", ()))
|
|
1674
|
+
for dist in sorted(before - after)[:5]:
|
|
1675
|
+
out.append(f"distribution {dist} recorded, not installed now")
|
|
1676
|
+
for dist in sorted(after - before)[:5]:
|
|
1677
|
+
out.append(f"distribution {dist} installed now, not recorded")
|
|
1678
|
+
for name, value in sorted(recorded.get("variables", {}).items()):
|
|
1679
|
+
if met.get("variables", {}).get(name) != value:
|
|
1680
|
+
out.append(f"{name}: {value!r} recorded, {met['variables'].get(name)!r} now")
|
|
1681
|
+
return out
|