diffcone 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
diffcone/execution.py ADDED
@@ -0,0 +1,1681 @@
1
+ """Execution integration: run selected targets, and validate plans.
2
+
3
+ Planning never executes project code. Everything in this module runs
4
+ *after* a plan exists and is invoked only by the ``run`` and ``validate``
5
+ commands, in a subprocess, with a command the user controls.
6
+
7
+ ``run`` executes the selected targets of a plan with the runner's CLI.
8
+ ``validate`` runs the full pytest suite at both snapshots (in temporary
9
+ ``git worktree`` checkouts for commits, in place for WORKTREE)
10
+ and checks that every test whose outcome changed was selected.
11
+ This is outcome-based validation: a test whose behaviour
12
+ changed without changing its pass/fail outcome is not detected.
13
+ With ``coverage=True`` it also runs the head suite under
14
+ pytest-cov with per-test contexts and checks that every test
15
+ that *executed* a changed symbol was selected, which measures
16
+ recall (and reports precision) against dynamic ground truth.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import os
23
+ import posixpath
24
+ import re
25
+ import shlex
26
+ import shutil
27
+ import sqlite3
28
+ import subprocess
29
+ import tempfile
30
+ import threading
31
+ from collections import defaultdict
32
+ from collections.abc import Iterator, Sequence
33
+ from concurrent.futures import ThreadPoolExecutor
34
+ from contextlib import contextmanager
35
+ from dataclasses import dataclass, field
36
+ from pathlib import Path
37
+
38
+ from diffcone.cache import IndexCache
39
+ from diffcone.discovery.asv_static import strip_json_comments
40
+ from diffcone.evidence import EVIDENCE_DIR, Evidence, EvidenceError, advance, fold, write_store
41
+ from diffcone.manifest import Target
42
+ from diffcone.model import KIND_COMMIT, KIND_WORKTREE, MODULE, SourceIndex
43
+ from diffcone.planner import Plan, _index_snapshot
44
+ from diffcone.snapshot import GitError, _git, is_bytecode, resolve_commit, split_root
45
+
46
+ DEFAULT_COMMANDS = {"pytest": "python -m pytest", "asv": "asv run --python=same"}
47
+
48
+ # Parameter ids may contain spaces, pipes and nested brackets
49
+ # (``test_x[choices4-[TEXT: a|b]] PASSED [ 12%]``), so the node id is
50
+ # everything up to the outcome token.
51
+ _PYTEST_LINE = re.compile(
52
+ r"^(?P<nodeid>\S+::.*?) (?P<outcome>PASSED|FAILED|ERROR|SKIPPED|XFAIL|XPASS)(?:\s|$)"
53
+ )
54
+ # pytest-xdist's report of a worker that died mid-run, and the interpreter's
55
+ # own last words.
56
+ _CRASH_LINE = re.compile(
57
+ r"node down|crashed while running|replacing crashed worker|Fatal Python error|"
58
+ r"Segmentation fault|Bus error|Killed"
59
+ )
60
+ # pytest-xdist puts the worker and the outcome first: ``[gw3] PASSED a.py::t``
61
+ # (``[gw3] [ 12%] PASSED ...`` in some versions).
62
+ _XDIST_LINE = re.compile(
63
+ r"^\[gw\d+\](?: \[\s*\d+%\])? (?P<outcome>PASSED|FAILED|ERROR|SKIPPED|XFAIL|XPASS) "
64
+ r"(?P<nodeid>\S+::.*?)\s*$"
65
+ )
66
+
67
+
68
+ # --------------------------------------------------------------------------- run
69
+
70
+
71
+ def build_command(
72
+ runner: str, targets: list[Target], command: str | None, extra: list[str]
73
+ ) -> list[str]:
74
+ """The command line that runs exactly ``targets`` with ``runner``."""
75
+ base = shlex.split(command or DEFAULT_COMMANDS[runner])
76
+ ids = [t.runner_id for t in sorted(targets)]
77
+ if runner == "pytest":
78
+ return [*base, *extra, *ids]
79
+ if runner == "asv":
80
+ # ASV matches --bench against the benchmark's name, and for a
81
+ # parameterised one against ``name(param0, param1, ...)`` instead, so
82
+ # a pattern anchored with ``$`` selects none of those: the name must
83
+ # be followed by the end of the string or its parameter list.
84
+ pattern = "^(" + "|".join(re.escape(i) for i in ids) + r")($|\()"
85
+ return [*base, *extra, "--bench", pattern]
86
+ raise ValueError(f"unknown runner {runner!r}")
87
+
88
+
89
+ def resolve_command(command: str | None, repo: Path) -> str | None:
90
+ """Make a relative executable path in ``command`` (``.venv/bin/python``)
91
+ absolute, against the current directory and then the repository, so
92
+ the command still works from a temporary worktree. Symlinks are kept:
93
+ a venv's ``python`` is a link to the base interpreter, and following it
94
+ would run outside the venv."""
95
+ if command is None:
96
+ return None
97
+ argv = shlex.split(command)
98
+ if not argv or os.path.isabs(argv[0]) or os.sep not in argv[0]:
99
+ return command
100
+ for base in (Path.cwd(), repo):
101
+ candidate = base / argv[0]
102
+ if candidate.exists():
103
+ return shlex.join([os.path.abspath(candidate), *argv[1:]])
104
+ return command
105
+
106
+
107
+ @dataclass
108
+ class RunResult:
109
+ runner: str
110
+ command: list[str]
111
+ selected: list[Target]
112
+ total: int
113
+ returncode: int | None # None when nothing was run (dry run or empty selection)
114
+ # Selected pytest targets that pytest did not collect (discovery and
115
+ # collection disagree): a node id on the command line used to fail loudly.
116
+ missing: list[str] = field(default_factory=list)
117
+ # Collected tests that are no target of the plan: kept and run (see
118
+ # selection.py), and reported.
119
+ unknown: list[str] = field(default_factory=list)
120
+
121
+
122
+ def worktree_mismatch(repo: Path, plan: Plan) -> str | None:
123
+ """Why the working tree is not what the plan analysed, if it is not.
124
+
125
+ ``run`` executes the checkout, so a plan made from a commit only
126
+ describes what will run while the checkout *is* that commit, unmodified.
127
+ Running a different tree turns the plan's guarantee into a guess about
128
+ code that was never analysed.
129
+ """
130
+ if plan.head.kind == KIND_WORKTREE:
131
+ return None
132
+ # What matters is the code, not the commit: a checkout carrying an extra
133
+ # commit that only adds a manifest runs the analysed code, while one
134
+ # uncommitted edit to a source file does not.
135
+ against = ["diff", "--name-only"] + (
136
+ [plan.head.commit] if plan.head.kind == KIND_COMMIT else []
137
+ )
138
+ try:
139
+ differing = _git(repo, against).decode("utf-8", "surrogateescape").split("\n")
140
+ untracked = (
141
+ _git(repo, ["ls-files", "--others", "--exclude-standard"])
142
+ .decode("utf-8", "surrogateescape")
143
+ .split("\n")
144
+ )
145
+ except GitError as exc:
146
+ return f"cannot tell what the working tree holds: {exc}"
147
+ roots = [split_root(r)[0] for r in plan.source_roots]
148
+ in_scope = sorted(
149
+ path
150
+ for path in {p.strip() for p in differing + untracked if p.strip()}
151
+ if path.endswith(".py")
152
+ and any(root in ("", ".") or path.startswith(root + "/") for root in roots)
153
+ )
154
+ if not in_scope:
155
+ return None
156
+ analysed = (
157
+ f"commit {plan.head.commit[:12]}" if plan.head.kind == KIND_COMMIT else "the staged index"
158
+ )
159
+ return (
160
+ f"the plan analysed {analysed}, and the working tree it would run has "
161
+ f"{len(in_scope)} differing Python file(s) under the source roots (e.g. {in_scope[0]})"
162
+ )
163
+
164
+
165
+ def run_selected(
166
+ plan: Plan,
167
+ runner: str,
168
+ *,
169
+ cwd: Path,
170
+ command: str | None = None,
171
+ extra: list[str] | None = None,
172
+ dry_run: bool = False,
173
+ env: dict[str, str] | None = None,
174
+ ) -> RunResult:
175
+ """Run the selected targets of ``runner``. ``--dry-run`` reports the
176
+ equivalent command (pytest node ids appended); a real pytest run instead
177
+ collects from pytest's own starting points and keeps the selection with
178
+ ``-p diffcone_select`` (``selection.py`` says why), so conftests load as
179
+ in a full run."""
180
+ selected = [d.target for d in plan.decisions if d.selected and d.target.runner == runner]
181
+ total = sum(1 for d in plan.decisions if d.target.runner == runner)
182
+ argv = build_command(runner, selected, command, list(extra or [])) if selected else []
183
+ result = RunResult(runner, argv, selected, total, None)
184
+ if dry_run or not selected:
185
+ return result
186
+ if runner != "pytest":
187
+ result.returncode = subprocess.run(argv, cwd=cwd, env=env).returncode
188
+ return result
189
+ with tempfile.TemporaryDirectory(prefix="diffcone-select-") as tmp:
190
+ own = env is None
191
+ run_env = plugin_environment(dict(os.environ), None, cwd) if own else dict(env)
192
+ run_env["DIFFCONE_SELECT"] = tmp
193
+ Path(tmp, "selected").write_text(
194
+ "".join(f"{t.runner_id}\n" for t in selected), encoding="utf-8"
195
+ )
196
+ every = [d.target for d in plan.decisions if d.target.runner == runner]
197
+ Path(tmp, "targets").write_text(
198
+ "".join(f"{t.runner_id}\n" for t in every), encoding="utf-8"
199
+ )
200
+ result.command = [
201
+ *shlex.split(command or DEFAULT_COMMANDS["pytest"]),
202
+ "-p",
203
+ SELECT_PLUGIN,
204
+ *(extra or []),
205
+ ]
206
+ try:
207
+ result.returncode = subprocess.run(result.command, cwd=cwd, env=run_env).returncode
208
+ finally:
209
+ if own:
210
+ shutil.rmtree(run_env["PYTHONPATH"].split(os.pathsep)[-1], ignore_errors=True)
211
+ reports = [
212
+ set(f.read_text("utf-8").split("\n")) - {""} for f in Path(tmp).glob("missing-*")
213
+ ]
214
+ if reports:
215
+ result.missing = sorted(set.intersection(*reports))
216
+ result.unknown = sorted(
217
+ {
218
+ line
219
+ for f in Path(tmp).glob("unknown-*")
220
+ for line in f.read_text("utf-8").split("\n")
221
+ if line
222
+ }
223
+ )
224
+ return result
225
+
226
+
227
+ # --------------------------------------------------------------------------- validate
228
+
229
+
230
+ @dataclass(frozen=True)
231
+ class TargetOutcome:
232
+ runner_id: str
233
+ base: str | None # None: absent at that snapshot
234
+ head: str | None
235
+ selected: bool
236
+ known_at_head: bool = True # a target of the plan (exists in the head snapshot)
237
+
238
+ @property
239
+ def changed(self) -> bool:
240
+ return self.base != self.head
241
+
242
+ @property
243
+ def removed(self) -> bool:
244
+ """Ran at base, gone at head: nothing to select, so never a miss."""
245
+ return self.head is None and not self.known_at_head
246
+
247
+
248
+ @dataclass(frozen=True)
249
+ class CoverageHit:
250
+ runner_id: str
251
+ selected: bool
252
+ executed_changed: tuple[str, ...] # changed symbols whose lines the test ran
253
+
254
+
255
+ @dataclass
256
+ class CoverageValidation:
257
+ hits: list[CoverageHit] = field(default_factory=list) # one per test seen under coverage
258
+ changed_symbols: tuple[str, ...] = ()
259
+ log: str = ""
260
+
261
+ @property
262
+ def affected(self) -> list[CoverageHit]:
263
+ return [h for h in self.hits if h.executed_changed]
264
+
265
+ @property
266
+ def caught(self) -> list[CoverageHit]:
267
+ return [h for h in self.affected if h.selected]
268
+
269
+ @property
270
+ def missed(self) -> list[CoverageHit]:
271
+ return [h for h in self.affected if not h.selected]
272
+
273
+ @property
274
+ def recall(self) -> float | None:
275
+ return len(self.caught) / len(self.affected) if self.affected else None
276
+
277
+ @property
278
+ def precision(self) -> float | None:
279
+ selected = [h for h in self.hits if h.selected]
280
+ return len(self.caught) / len(selected) if selected else None
281
+
282
+
283
+ @dataclass
284
+ class Validation:
285
+ runner: str
286
+ command: list[str]
287
+ outcomes: list[TargetOutcome] = field(default_factory=list)
288
+ base_log: str = ""
289
+ head_log: str = ""
290
+ coverage: CoverageValidation | None = None
291
+
292
+ @property
293
+ def caught(self) -> list[TargetOutcome]:
294
+ return [o for o in self.outcomes if o.changed and o.selected]
295
+
296
+ @property
297
+ def missed(self) -> list[TargetOutcome]:
298
+ return [o for o in self.outcomes if o.changed and not o.selected and not o.removed]
299
+
300
+ @property
301
+ def removed(self) -> list[TargetOutcome]:
302
+ return [o for o in self.outcomes if o.removed]
303
+
304
+ @property
305
+ def selected_count(self) -> int:
306
+ return sum(1 for o in self.outcomes if o.selected)
307
+
308
+ @property
309
+ def ok(self) -> bool:
310
+ return not self.missed and (self.coverage is None or not self.coverage.missed)
311
+
312
+
313
+ def fold_nodeid(nodeid: str) -> str:
314
+ """Drop the parameter case (``[...]``) so a node id names the test function."""
315
+ return re.sub(r"\[.*\]$", "", nodeid)
316
+
317
+
318
+ def parse_pytest_verbose(output: str) -> dict[str, str]:
319
+ """Map pytest node ids (parameter cases folded into their function) to
320
+ the outcomes observed for that function, in a fixed order and joined
321
+ with ``+`` (``PASSED+SKIPPED``). The fold must not depend on the order
322
+ the cases ran in: under pytest-xdist that order varies from run to run,
323
+ and a fold that kept the first or last case would report changes that
324
+ never happened."""
325
+ order = ("PASSED", "SKIPPED", "XFAIL", "XPASS", "FAILED", "ERROR")
326
+ seen: dict[str, set[str]] = {}
327
+ for line in output.splitlines():
328
+ m = _PYTEST_LINE.match(line.strip()) or _XDIST_LINE.match(line.strip())
329
+ if not m:
330
+ continue
331
+ seen.setdefault(fold_nodeid(m.group("nodeid")), set()).add(m.group("outcome"))
332
+ return {nodeid: "+".join(o for o in order if o in found) for nodeid, found in seen.items()}
333
+
334
+
335
+ def parse_outcome_lines(lines: list[str]) -> dict[str, str]:
336
+ """``parse_pytest_verbose`` for the selection plugin's outcome lines
337
+ (``nodeid<TAB>OUTCOME``): the same fold, the same order."""
338
+ order = ("PASSED", "SKIPPED", "XFAIL", "XPASS", "FAILED", "ERROR")
339
+ seen: dict[str, set[str]] = {}
340
+ for line in lines:
341
+ nodeid, _, outcome = line.rpartition("\t")
342
+ if nodeid and outcome in order:
343
+ seen.setdefault(fold_nodeid(nodeid), set()).add(outcome)
344
+ return {nodeid: "+".join(o for o in order if o in found) for nodeid, found in seen.items()}
345
+
346
+
347
+ class _Checkout:
348
+ """A directory holding a snapshot: a temporary detached worktree for a
349
+ commit, or the repository itself for WORKTREE."""
350
+
351
+ def __init__(
352
+ self, repo: Path, kind: str, commit: str, setup_command: str | None = None
353
+ ) -> None:
354
+ self.repo = repo
355
+ self.kind = kind
356
+ self.setup_command = setup_command
357
+ self.commit = commit
358
+ self.path = repo
359
+ self._tmp: tempfile.TemporaryDirectory[str] | None = None
360
+
361
+ def __enter__(self) -> Path:
362
+ if self.kind == KIND_COMMIT:
363
+ self._tmp = tempfile.TemporaryDirectory(prefix="diffcone-validate-")
364
+ # A name of its own: git names the worktree's admin directory
365
+ # (.git/worktrees/<name>) after it, and parallel corpus jobs that
366
+ # all used "checkout" raced on creating it.
367
+ self.path = Path(self._tmp.name) / f"checkout-{Path(self._tmp.name).name}"
368
+ _git(self.repo, ["worktree", "add", "--detach", "-q", str(self.path), self.commit])
369
+ elif self.kind != KIND_WORKTREE:
370
+ raise GitError("validate supports commit and WORKTREE snapshots, not INDEX")
371
+ if self.setup_command:
372
+ # Build-generated, git-ignored files (a setuptools-scm _version.py,
373
+ # compiled extensions) are absent from a fresh checkout; the user's
374
+ # setup command recreates what the suite needs.
375
+ proc = subprocess.run(
376
+ self.setup_command, shell=True, cwd=self.path, capture_output=True, text=True
377
+ )
378
+ if proc.returncode != 0:
379
+ self.__exit__()
380
+ raise GitError(
381
+ f"setup command failed in checkout of {self.commit[:12]} "
382
+ f"(exit {proc.returncode}):\n{(proc.stdout + proc.stderr)[-2000:]}"
383
+ )
384
+ return self.path
385
+
386
+ def __exit__(self, *exc: object) -> None:
387
+ if self._tmp is not None:
388
+ try:
389
+ _git(self.repo, ["worktree", "remove", "--force", str(self.path)])
390
+ except GitError:
391
+ pass
392
+ self._tmp.cleanup()
393
+
394
+
395
+ @dataclass
396
+ class _SuiteRun:
397
+ outcomes: dict[str, str]
398
+ log: str
399
+ returncode: int
400
+ coverage_db: Path | None = None
401
+
402
+
403
+ def _checkout_env(cwd: Path, source_roots: list[str]) -> dict[str, str]:
404
+ """Environment that makes the checkout's own code win over an installed
405
+ (typically editable) copy of the project: its source roots go first on
406
+ PYTHONPATH. With a ``src`` layout the current directory alone would not
407
+ do it, and the suite would silently test the installed code."""
408
+ env = dict(os.environ)
409
+ roots = []
410
+ for spec in source_roots:
411
+ root = split_root(spec)[0]
412
+ candidate = (cwd / root).resolve() if root else cwd.resolve()
413
+ if candidate.is_dir() and str(candidate) not in roots:
414
+ roots.append(str(candidate))
415
+ existing = env.get("PYTHONPATH")
416
+ env["PYTHONPATH"] = os.pathsep.join([*roots, *([existing] if existing else [])])
417
+ return env
418
+
419
+
420
+ @contextmanager
421
+ def _run_full_pytest(
422
+ cwd: Path,
423
+ command: str | None,
424
+ *,
425
+ coverage: bool = False,
426
+ source_roots: Sequence[str] = (),
427
+ hash_seed: str | None = None,
428
+ coverage_include: list[str] | None = None,
429
+ ) -> Iterator[_SuiteRun]:
430
+ """Run the whole suite once with ``-v``; with ``coverage`` the same run
431
+ also records per-test coverage contexts into a temporary database that
432
+ lives for the duration of the context."""
433
+ argv = [
434
+ *shlex.split(command or DEFAULT_COMMANDS["pytest"]),
435
+ "-v",
436
+ "-p",
437
+ "no:cacheprovider",
438
+ "-p",
439
+ SELECT_PLUGIN,
440
+ "--no-header",
441
+ "-rN",
442
+ ]
443
+ with tempfile.TemporaryDirectory(prefix="diffcone-cov-") as tmp:
444
+ # Outcomes come from the selection plugin, not the console: a project
445
+ # whose addopts carry ``-q`` prints no per-test lines at all.
446
+ env = plugin_environment(_checkout_env(cwd, list(source_roots)), None, cwd)
447
+ outcomes_dir = Path(tmp) / "outcomes"
448
+ outcomes_dir.mkdir()
449
+ env["DIFFCONE_OUTCOMES"] = str(outcomes_dir)
450
+ if hash_seed is not None:
451
+ # An evidence plan assumed the recorded hash seed; check it under that.
452
+ env["PYTHONHASHSEED"] = hash_seed
453
+ db: Path | None = None
454
+ if coverage:
455
+ db = Path(tmp) / ".coverage"
456
+ # The project's own addopts stay in force so this run collects the
457
+ # same tests as a plain run; only the coverage options are added.
458
+ # The project's coverage config is replaced: its ``source``/``omit``
459
+ # (typically excluding tests) would blind attribution, and
460
+ # ``parallel``/``branch`` change the database layout.
461
+ rc = Path(tmp) / "coveragerc"
462
+ config = "[run]\nbranch = false\nparallel = false\nrelative_files = false\n"
463
+ if coverage_include is not None:
464
+ # Measure only these checkout-relative files: on a suite the
465
+ # size of pandas, per-test contexts over every file make a
466
+ # database of many gigabytes. (``include`` is ignored when a
467
+ # source is given, so ``--cov`` names none.)
468
+ paths = "".join(f"\n {cwd.resolve() / p}" for p in coverage_include)
469
+ config += f"include ={paths}\n"
470
+ argv += ["--cov"]
471
+ else:
472
+ argv += ["--cov=."]
473
+ rc.write_text(config)
474
+ argv += ["--cov-context=test", "--cov-report=", f"--cov-config={rc}"]
475
+ env["COVERAGE_FILE"] = str(db)
476
+ # coverage.py 7.x defaults to the sys.monitoring core on Python
477
+ # 3.12+, which disables a line after its first hit: with per-test
478
+ # contexts only the *first* test to run a line gets credit for it.
479
+ # coverage avoids that core only for contexts set in its own config,
480
+ # not for pytest-cov's switch_context(), so force the C tracer (it
481
+ # falls back to the pure-Python tracer when unavailable).
482
+ env.setdefault("COVERAGE_CORE", "ctrace")
483
+ try:
484
+ proc = subprocess.run(argv, cwd=cwd, capture_output=True, text=True, env=env)
485
+ except OSError as exc:
486
+ raise GitError(
487
+ f"cannot run {argv[0]!r}: {exc}; a relative path in --command is resolved "
488
+ "against the current directory and the repository"
489
+ ) from exc
490
+ finally:
491
+ shutil.rmtree(env["PYTHONPATH"].split(os.pathsep)[-1], ignore_errors=True)
492
+ log = proc.stdout + proc.stderr
493
+ if coverage and not (db and db.exists()):
494
+ raise GitError(
495
+ "coverage validation produced no coverage database; is pytest-cov installed in "
496
+ f"the environment that runs {argv[0]!r} (and not disabled by --no-cov)?\n"
497
+ f"exit code {proc.returncode}\n{log[-2000:]}"
498
+ )
499
+ # pytest exits 0 when every test passed and 1 when some failed; both
500
+ # are suites that ran. Anything else means it did not: 2 a collection
501
+ # error or an interrupt, 3 an internal error, 4 a usage error, 5 no
502
+ # tests at all. Comparing the outcomes of a suite that never ran finds
503
+ # no missed outcome change and would report success.
504
+ if proc.returncode not in (0, 1):
505
+ raise GitError(
506
+ f"the suite did not run: {argv[0]!r} exited {proc.returncode} "
507
+ f"({PYTEST_EXIT.get(proc.returncode, 'unknown')}), so there are no outcomes to "
508
+ f"compare\n{log[-2000:]}"
509
+ )
510
+ lines = [
511
+ line
512
+ for f in sorted(outcomes_dir.glob("outcomes-*"))
513
+ for line in f.read_text("utf-8").splitlines()
514
+ ]
515
+ if not lines:
516
+ raise GitError(
517
+ f"the suite ran ({argv[0]!r} exited {proc.returncode}) but reported no test "
518
+ "outcomes: diffcone's outcome plugin did not load, so there is nothing to "
519
+ f"compare\n{log[-2000:]}"
520
+ )
521
+ yield _SuiteRun(parse_outcome_lines(lines), log, proc.returncode, db)
522
+
523
+
524
+ # --------------------------------------------------------------------------- coverage
525
+
526
+
527
+ PYTEST_EXIT = {
528
+ 2: "interrupted, usually a collection error",
529
+ 3: "internal error",
530
+ 4: "usage error",
531
+ 5: "no tests collected",
532
+ }
533
+
534
+
535
+ def _numbits_to_lines(blob: bytes) -> list[int]:
536
+ """Decode coverage.py's numbits bitmap: bit n set means line n executed."""
537
+ lines: list[int] = []
538
+ for byte_index, byte in enumerate(blob):
539
+ for bit in range(8):
540
+ if byte & (1 << bit):
541
+ lines.append(byte_index * 8 + bit)
542
+ return lines
543
+
544
+
545
+ def read_coverage_contexts(
546
+ db_path: Path, root: Path, outside: set[str] | None = None
547
+ ) -> dict[str, dict[str, set[int]]]:
548
+ """Per pytest node id (parameter cases and setup/run/teardown phases
549
+ folded), the executed lines per checkout-relative file. Measured files
550
+ that lie outside the checkout are collected into ``outside`` when given.
551
+
552
+ Reads both the ``line_bits`` table (line coverage) and the ``arc`` table
553
+ (branch coverage, which coverage.py uses *instead* when ``branch = True``
554
+ is configured). Paths may be stored relative when the project sets
555
+ ``relative_files``; they are resolved against the checkout root.
556
+ """
557
+ result: dict[str, dict[str, set[int]]] = defaultdict(lambda: defaultdict(set))
558
+ root = root.resolve()
559
+
560
+ def rel_path(path: str) -> str | None:
561
+ p = Path(path)
562
+ if not p.is_absolute():
563
+ p = root / p
564
+ try:
565
+ return str(p.resolve().relative_to(root))
566
+ except ValueError:
567
+ if outside is not None:
568
+ outside.add(str(p))
569
+ return None
570
+
571
+ con = sqlite3.connect(db_path)
572
+ try:
573
+ line_rows = con.execute(
574
+ "SELECT context.context, file.path, line_bits.numbits FROM line_bits "
575
+ "JOIN context ON context.id = line_bits.context_id "
576
+ "JOIN file ON file.id = line_bits.file_id"
577
+ ).fetchall()
578
+ arc_rows = con.execute(
579
+ "SELECT context.context, file.path, arc.fromno, arc.tono FROM arc "
580
+ "JOIN context ON context.id = arc.context_id "
581
+ "JOIN file ON file.id = arc.file_id"
582
+ ).fetchall()
583
+ finally:
584
+ con.close()
585
+ for context, path, numbits in line_rows:
586
+ rel = rel_path(path) if context else None
587
+ if rel is not None:
588
+ result[fold_nodeid(context.rsplit("|", 1)[0])][rel].update(_numbits_to_lines(numbits))
589
+ for context, path, fromno, tono in arc_rows:
590
+ rel = rel_path(path) if context else None
591
+ if rel is not None:
592
+ lines = result[fold_nodeid(context.rsplit("|", 1)[0])][rel]
593
+ # Negative numbers mark entry/exit arcs; abs() gives the real line.
594
+ lines.update(n for n in (abs(fromno), abs(tono)) if n > 0)
595
+ return result
596
+
597
+
598
+ def _line_owner_index(
599
+ plan: Plan, side: str = "head"
600
+ ) -> tuple[dict[str, dict[int, set[str]]], tuple[str, ...]]:
601
+ """``{path: {line: changed symbol ids}}`` for the changed symbols as they
602
+ are at ``side`` (a deleted symbol has lines at base only).
603
+
604
+ A container (module or class) owns only the lines outside its members'
605
+ definitions, mirroring how the planner treats body changes; when its
606
+ change is structural (which invalidates every member) all its lines
607
+ count, mirroring the ``defined_in`` propagation rule.
608
+ """
609
+ index_ = plan.head_index if side == "head" else plan.base_index
610
+ members_of: dict[str, list[tuple[int, int]]] = defaultdict(list)
611
+ for symbol in index_.symbols.values():
612
+ if symbol.container is not None:
613
+ members_of[symbol.container].extend(symbol.line_ranges)
614
+ index: dict[str, dict[int, set[str]]] = defaultdict(lambda: defaultdict(set))
615
+ changed_ids: list[str] = []
616
+ for change in plan.changes:
617
+ symbol = change.head if side == "head" else change.base
618
+ # Additive-only changes carry no impact for the planner and mean no
619
+ # behaviour change for the code executed, so they are not ground truth.
620
+ if symbol is None or not symbol.line_ranges or not change.carries_impact:
621
+ continue
622
+ changed_ids.append(symbol.id)
623
+ excluded: set[int] = set()
624
+ if not change.structural:
625
+ for start, end in members_of.get(symbol.id, ()):
626
+ excluded.update(range(start, end + 1))
627
+ for start, end in symbol.line_ranges:
628
+ for line in range(start, end + 1):
629
+ if line not in excluded:
630
+ index[symbol.path][line].add(symbol.id)
631
+ return index, tuple(sorted(changed_ids))
632
+
633
+
634
+ def shadowed_files(plan: Plan, outside: set[str]) -> list[tuple[str, str]]:
635
+ """Measured files outside the checkout that have the same source-root-
636
+ relative path as a file inside it: the suite imported an installed copy
637
+ of the project instead of the checkout."""
638
+ suffixes: dict[str, str] = {}
639
+ for symbol in plan.head_index.symbols.values():
640
+ rel = symbol.path
641
+ suffixes["/" + rel] = rel
642
+ for spec in plan.source_roots:
643
+ root = split_root(spec)[0]
644
+ if root and rel.startswith(root + "/"):
645
+ suffixes["/" + rel[len(root) + 1 :]] = rel
646
+ found: list[tuple[str, str]] = []
647
+ for path in sorted(outside):
648
+ for suffix, rel in suffixes.items():
649
+ if path.endswith(suffix):
650
+ found.append((path, rel))
651
+ break
652
+ return found
653
+
654
+
655
+ def coverage_validation(
656
+ plan: Plan, run: _SuiteRun, checkout: Path, selected: set[str], side: str = "head"
657
+ ) -> CoverageValidation:
658
+ """Attribute one suite run's per-test coverage to the changed symbols as
659
+ they are at ``side`` of the plan."""
660
+ assert run.coverage_db is not None
661
+ outside: set[str] = set()
662
+ contexts = read_coverage_contexts(run.coverage_db, checkout, outside)
663
+ shadowing = shadowed_files(plan, outside)
664
+ if shadowing:
665
+ listing = "\n".join(f" {path} (shadows {rel})" for path, rel in shadowing[:5])
666
+ raise GitError(
667
+ "the suite imported project code from outside the checkout, so the validation "
668
+ "would test the wrong revision:\n"
669
+ f"{listing}\nPut the checkout's source roots first on PYTHONPATH (diffcone does "
670
+ "this for --source-root entries; pass the roots that hold the package) or run "
671
+ "against an environment without an installed copy of the project."
672
+ )
673
+ if not contexts:
674
+ raise GitError(
675
+ f"coverage validation recorded no per-test contexts: the {side} suite collected "
676
+ f"no tests under coverage (exit code {run.returncode})\n{run.log[-2000:]}"
677
+ )
678
+ owners, changed_ids = _line_owner_index(plan, side)
679
+ hits: list[CoverageHit] = []
680
+ for nodeid in sorted(contexts):
681
+ executed: set[str] = set()
682
+ for rel, lines in contexts[nodeid].items():
683
+ by_line = owners.get(rel)
684
+ if by_line:
685
+ for line in lines:
686
+ executed.update(by_line.get(line, ()))
687
+ hits.append(CoverageHit(nodeid, nodeid in selected, tuple(sorted(executed))))
688
+ return CoverageValidation(hits, changed_ids, run.log)
689
+
690
+
691
+ def merge_coverage(
692
+ head: CoverageValidation, base: CoverageValidation | None, head_tests: set[str]
693
+ ) -> CoverageValidation:
694
+ """One record per test with the changed symbols it executed at either
695
+ side. A test that no longer exists at head is ``removed`` in the outcome
696
+ comparison and nothing could select it, so its base-side hits are
697
+ dropped rather than counted as misses."""
698
+ if base is None:
699
+ return head
700
+ hits = {h.runner_id: h for h in head.hits}
701
+ for hit in base.hits:
702
+ if hit.runner_id not in hits and hit.runner_id not in head_tests:
703
+ continue
704
+ existing = hits.get(hit.runner_id)
705
+ executed = set(hit.executed_changed) | set(existing.executed_changed if existing else ())
706
+ hits[hit.runner_id] = CoverageHit(
707
+ hit.runner_id,
708
+ existing.selected if existing else hit.selected,
709
+ tuple(sorted(executed)),
710
+ )
711
+ return CoverageValidation(
712
+ [hits[k] for k in sorted(hits)],
713
+ tuple(sorted(set(head.changed_symbols) | set(base.changed_symbols))),
714
+ head.log,
715
+ )
716
+
717
+
718
+ # (commit sha, ran under coverage, PYTHONHASHSEED an evidence plan pinned)
719
+ _OutcomeKey = tuple[str, bool, str | None]
720
+
721
+
722
+ class OutcomeCache(dict[_OutcomeKey, dict[str, str]]):
723
+ """(commit sha, ran under coverage, hash seed) -> per-test outcomes, plus the
724
+ coverage database of every snapshot that ran under coverage (copied into
725
+ a temporary directory that lives until ``close``), so a base that is not
726
+ run again can still be attributed for a later pair.
727
+
728
+ Outcomes measured under the coverage tracer are only comparable with
729
+ each other (tests that depend on recursion depth or timing can flip under
730
+ ``sys.settrace``), hence the flag in the key. Shared between parallel
731
+ corpus jobs with a no-wait policy: a snapshot that is already present is
732
+ reused, otherwise the requester runs it itself and offers the result
733
+ (set-if-absent). Nobody ever waits on another job, because in a linear
734
+ history every pair's base is the previous pair's head and waiting would
735
+ serialise the whole corpus; the price is at most one extra suite run per
736
+ pair when two jobs need the same snapshot at the same time.
737
+ """
738
+
739
+ def __init__(self) -> None:
740
+ super().__init__()
741
+ self._lock = threading.Lock()
742
+ self._databases: dict[_OutcomeKey, tuple[Path, Path]] = {} # key -> (db, checkout)
743
+ self._directory: tempfile.TemporaryDirectory | None = None
744
+
745
+ def keep_coverage(self, key: _OutcomeKey, db: Path, checkout: Path) -> None:
746
+ """Copy a run's coverage database so it outlives its worktree
747
+ (``checkout`` is remembered to resolve the paths it recorded)."""
748
+ with self._lock:
749
+ if key in self._databases:
750
+ return
751
+ if self._directory is None:
752
+ self._directory = tempfile.TemporaryDirectory(prefix="diffcone-basecov-")
753
+ copy = Path(self._directory.name) / f"{key[0]}-{int(key[1])}-{key[2]}.coverage"
754
+ shutil.copyfile(db, copy)
755
+ self._databases[key] = (copy, checkout)
756
+
757
+ def coverage_of(self, key: _OutcomeKey) -> tuple[Path, Path] | None:
758
+ with self._lock:
759
+ return self._databases.get(key)
760
+
761
+ def close(self) -> None:
762
+ with self._lock:
763
+ if self._directory is not None:
764
+ self._directory.cleanup()
765
+ self._directory = None
766
+ self._databases.clear()
767
+
768
+ def lookup(self, key: _OutcomeKey) -> dict[str, str] | None:
769
+ with self._lock:
770
+ return self.get(key)
771
+
772
+ def offer(self, key: _OutcomeKey, value: dict[str, str]) -> None:
773
+ with self._lock:
774
+ self.setdefault(key, value)
775
+
776
+ def reuse_or_run(self, key: _OutcomeKey, run) -> dict[str, str]:
777
+ value = self.lookup(key)
778
+ if value is not None:
779
+ return value
780
+ value = run()
781
+ self.offer(key, value)
782
+ return value
783
+
784
+
785
+ def validate_pytest(
786
+ plan: Plan,
787
+ *,
788
+ repo: Path,
789
+ command: str | None = None,
790
+ coverage: bool = False,
791
+ outcome_cache: OutcomeCache | None = None,
792
+ setup_command: str | None = None,
793
+ ) -> Validation:
794
+ """Run the full suite at base and head and compare outcome changes with
795
+ the plan's pytest selection; with ``coverage`` the head run also records
796
+ per-test coverage and every test that executed a changed symbol must be
797
+ selected. ``outcome_cache`` lets consecutive validations (a corpus) reuse
798
+ a committed snapshot's outcomes instead of running its suite again."""
799
+ if plan.head.kind == KIND_WORKTREE and plan.base.kind == KIND_WORKTREE:
800
+ raise GitError("validate needs at least one committed snapshot")
801
+ command = resolve_command(command, repo)
802
+ selected = {
803
+ d.target.runner_id for d in plan.decisions if d.selected and d.target.runner == "pytest"
804
+ }
805
+ cache = outcome_cache if outcome_cache is not None else OutcomeCache()
806
+ try:
807
+ return _validate_pytest(plan, repo, command, coverage, cache, setup_command, selected)
808
+ finally:
809
+ if outcome_cache is None:
810
+ cache.close()
811
+
812
+
813
+ def _validate_pytest(
814
+ plan: Plan,
815
+ repo: Path,
816
+ command: str | None,
817
+ coverage: bool,
818
+ cache: OutcomeCache,
819
+ setup_command: str | None,
820
+ selected: set[str],
821
+ ) -> Validation:
822
+ seed = (plan.evidence or {}).get("hash_seed")
823
+ base_key = (plan.base.commit, coverage, seed)
824
+ base_log = ""
825
+
826
+ def run_base() -> dict[str, str]:
827
+ # Same instrumentation on both sides: with --coverage the base suite
828
+ # also runs under the tracer, and its database is kept so tests that
829
+ # executed a symbol deleted in head can be attributed too.
830
+ nonlocal base_log
831
+ with _Checkout(repo, plan.base.kind, plan.base.commit, setup_command) as base_dir:
832
+ with _run_full_pytest(
833
+ base_dir,
834
+ command,
835
+ coverage=coverage,
836
+ source_roots=plan.source_roots,
837
+ hash_seed=seed,
838
+ ) as base_run:
839
+ base_log = base_run.log
840
+ if coverage and base_run.coverage_db is not None:
841
+ cache.keep_coverage(base_key, base_run.coverage_db, base_dir)
842
+ return base_run.outcomes
843
+
844
+ if plan.base.kind == KIND_COMMIT:
845
+ base_outcomes = cache.reuse_or_run(base_key, run_base)
846
+ else:
847
+ base_outcomes = run_base()
848
+ cov: CoverageValidation | None = None
849
+ # The head always runs here (its coverage database is needed); its
850
+ # outcomes are offered to the cache for pairs that use it as a base.
851
+ with _Checkout(repo, plan.head.kind, plan.head.commit, setup_command) as head_dir:
852
+ with _run_full_pytest(
853
+ head_dir,
854
+ command,
855
+ coverage=coverage,
856
+ source_roots=plan.source_roots,
857
+ hash_seed=seed,
858
+ ) as head_run:
859
+ head_outcomes, head_log = head_run.outcomes, head_run.log
860
+ if coverage:
861
+ cov = coverage_validation(plan, head_run, head_dir, selected)
862
+ if head_run.coverage_db is not None and plan.head.kind == KIND_COMMIT:
863
+ cache.keep_coverage(
864
+ (plan.head.commit, coverage, seed), head_run.coverage_db, head_dir
865
+ )
866
+ if plan.head.kind == KIND_COMMIT:
867
+ cache.offer((plan.head.commit, coverage, seed), head_outcomes)
868
+ if cov is not None:
869
+ kept = cache.coverage_of(base_key)
870
+ base_cov: CoverageValidation | None = None
871
+ if kept is not None:
872
+ db, base_checkout = kept
873
+ base_run = _SuiteRun(base_outcomes, base_log, 0, db)
874
+ base_cov = coverage_validation(plan, base_run, base_checkout, selected, side="base")
875
+ head_tests = {fold_nodeid(n) for n in head_outcomes}
876
+ cov = merge_coverage(cov, base_cov, head_tests)
877
+ known = {d.target.runner_id for d in plan.decisions if d.target.runner == "pytest"}
878
+ ids = sorted(known | set(base_outcomes) | set(head_outcomes))
879
+ validation = Validation(
880
+ "pytest",
881
+ [*shlex.split(command or DEFAULT_COMMANDS["pytest"]), "-v"],
882
+ base_log=base_log,
883
+ head_log=head_log,
884
+ coverage=cov,
885
+ )
886
+ for runner_id in ids:
887
+ validation.outcomes.append(
888
+ TargetOutcome(
889
+ runner_id,
890
+ base_outcomes.get(runner_id),
891
+ head_outcomes.get(runner_id),
892
+ runner_id in selected,
893
+ known_at_head=runner_id in known,
894
+ )
895
+ )
896
+ return validation
897
+
898
+
899
+ def validation_to_dict(v: Validation) -> dict:
900
+ return {
901
+ "runner": v.runner,
902
+ "ok": v.ok,
903
+ "counts": {
904
+ "targets": len(v.outcomes),
905
+ "selected": v.selected_count,
906
+ "outcome_changed": sum(1 for o in v.outcomes if o.changed),
907
+ "caught": len(v.caught),
908
+ "missed": len(v.missed),
909
+ },
910
+ "missed": [{"runner_id": o.runner_id, "base": o.base, "head": o.head} for o in v.missed],
911
+ "removed": [o.runner_id for o in v.removed],
912
+ "caught": [{"runner_id": o.runner_id, "base": o.base, "head": o.head} for o in v.caught],
913
+ "outcomes": [
914
+ {"runner_id": o.runner_id, "base": o.base, "head": o.head, "selected": o.selected}
915
+ for o in v.outcomes
916
+ ],
917
+ "coverage": None
918
+ if v.coverage is None
919
+ else {
920
+ "changed_symbols": list(v.coverage.changed_symbols),
921
+ "counts": {
922
+ "tests": len(v.coverage.hits),
923
+ "executed_a_changed_symbol": len(v.coverage.affected),
924
+ "caught": len(v.coverage.caught),
925
+ "missed": len(v.coverage.missed),
926
+ },
927
+ "recall": v.coverage.recall,
928
+ "precision": v.coverage.precision,
929
+ "missed": [
930
+ {"runner_id": h.runner_id, "executed": list(h.executed_changed)}
931
+ for h in v.coverage.missed
932
+ ],
933
+ "hits": [
934
+ {
935
+ "runner_id": h.runner_id,
936
+ "selected": h.selected,
937
+ "executed": list(h.executed_changed),
938
+ }
939
+ for h in v.coverage.hits
940
+ ],
941
+ },
942
+ }
943
+
944
+
945
+ def _status_label(v: Validation) -> str:
946
+ if v.ok:
947
+ return "OK"
948
+ reasons = []
949
+ if v.missed:
950
+ reasons.append("outcome")
951
+ if v.coverage is not None and v.coverage.missed:
952
+ reasons.append("coverage")
953
+ return f"MISSED ({', '.join(reasons)})"
954
+
955
+
956
+ def validation_to_text(v: Validation) -> str:
957
+ lines = [
958
+ f"validation ({v.runner}): {_status_label(v)}",
959
+ f" targets: {len(v.outcomes)}, selected: {v.selected_count}, "
960
+ f"outcome changed: {sum(1 for o in v.outcomes if o.changed)} "
961
+ f"(caught {len(v.caught)}, missed {len(v.missed)})",
962
+ ]
963
+ for o in v.missed:
964
+ lines.append(f" MISSED {o.runner_id}: {o.base} -> {o.head}")
965
+ if v.removed:
966
+ lines.append(f" removed at head (not misses): {len(v.removed)}")
967
+ for o in v.caught:
968
+ lines.append(f" caught {o.runner_id}: {o.base} -> {o.head}")
969
+ if v.coverage is None:
970
+ lines.append(
971
+ " note: outcome-based only; behaviour changes that keep the same pass/fail outcome "
972
+ "are invisible to this check (use --coverage)"
973
+ )
974
+ else:
975
+ c = v.coverage
976
+ fmt = lambda x: "n/a" if x is None else f"{x:.0%}" # noqa: E731
977
+ lines.append(
978
+ f"coverage: {len(c.affected)} of {len(c.hits)} test(s) executed a changed symbol; "
979
+ f"recall {fmt(c.recall)} (caught {len(c.caught)}, missed {len(c.missed)}), "
980
+ f"precision {fmt(c.precision)}"
981
+ )
982
+ for h in c.missed:
983
+ lines.append(f" MISSED {h.runner_id}: executed {', '.join(h.executed_changed)}")
984
+ return "\n".join(lines) + "\n"
985
+
986
+
987
+ # --------------------------------------------------------------------------- corpus
988
+
989
+
990
+ @dataclass
991
+ class CorpusEntry:
992
+ commit: str
993
+ parent: str
994
+ subject: str
995
+ skipped: str | None = None # reason, when no validation ran
996
+ error: str | None = None
997
+ changed_symbols: int = 0
998
+ targets: int = 0
999
+ selected: int = 0
1000
+ degraded: bool = False
1001
+ validation: Validation | None = None
1002
+
1003
+ @property
1004
+ def savings(self) -> float | None:
1005
+ return 1 - self.selected / self.targets if self.targets else None
1006
+
1007
+
1008
+ @dataclass
1009
+ class CorpusReport:
1010
+ repo: str
1011
+ revision_range: str
1012
+ coverage: bool
1013
+ entries: list[CorpusEntry] = field(default_factory=list)
1014
+
1015
+ @property
1016
+ def validated(self) -> list[CorpusEntry]:
1017
+ return [e for e in self.entries if e.validation is not None]
1018
+
1019
+ def _sum(self, pick) -> int:
1020
+ return sum(pick(e.validation) for e in self.validated)
1021
+
1022
+ @property
1023
+ def outcome_changed(self) -> int:
1024
+ return self._sum(lambda v: sum(1 for o in v.outcomes if o.changed))
1025
+
1026
+ @property
1027
+ def outcome_missed(self) -> int:
1028
+ return self._sum(lambda v: len(v.missed))
1029
+
1030
+ @property
1031
+ def coverage_affected(self) -> int:
1032
+ return self._sum(lambda v: len(v.coverage.affected) if v.coverage else 0)
1033
+
1034
+ @property
1035
+ def coverage_caught(self) -> int:
1036
+ return self._sum(lambda v: len(v.coverage.caught) if v.coverage else 0)
1037
+
1038
+ @property
1039
+ def coverage_selected(self) -> int:
1040
+ return self._sum(
1041
+ lambda v: sum(1 for h in v.coverage.hits if h.selected) if v.coverage else 0
1042
+ )
1043
+
1044
+ @property
1045
+ def recall(self) -> float | None:
1046
+ return self.coverage_caught / self.coverage_affected if self.coverage_affected else None
1047
+
1048
+ @property
1049
+ def precision(self) -> float | None:
1050
+ return self.coverage_caught / self.coverage_selected if self.coverage_selected else None
1051
+
1052
+ @property
1053
+ def mean_savings(self) -> float | None:
1054
+ values = [e.savings for e in self.validated if e.savings is not None]
1055
+ return sum(values) / len(values) if values else None
1056
+
1057
+ @property
1058
+ def ok(self) -> bool:
1059
+ return all(e.validation.ok for e in self.validated if e.validation) and not any(
1060
+ e.error for e in self.entries
1061
+ )
1062
+
1063
+
1064
+ def _commits_in_range(repo: Path, revision_range: str) -> list[tuple[str, str, str]]:
1065
+ """(commit, first parent, subject) for each commit in ``A..B``, oldest first."""
1066
+ out = _git(
1067
+ repo,
1068
+ [
1069
+ "rev-list",
1070
+ "--reverse",
1071
+ "--first-parent",
1072
+ "--format=%H %P%x00%s",
1073
+ "--no-commit-header",
1074
+ revision_range,
1075
+ ],
1076
+ )
1077
+ entries: list[tuple[str, str, str]] = []
1078
+ for line in out.decode("utf-8", "replace").splitlines():
1079
+ if not line.strip():
1080
+ continue
1081
+ ids, _, subject = line.partition("\0")
1082
+ parts = ids.split()
1083
+ if len(parts) < 2:
1084
+ continue # a root commit has no parent to compare against
1085
+ entries.append((parts[0], parts[1], subject.strip()))
1086
+ return entries
1087
+
1088
+
1089
+ def _touches_python(repo: Path, parent: str, commit: str) -> bool:
1090
+ out = _git(repo, ["diff", "--name-only", parent, commit])
1091
+ return any(line.endswith(".py") for line in out.decode("utf-8", "replace").splitlines())
1092
+
1093
+
1094
+ def corpus_validation(
1095
+ repo: Path,
1096
+ revision_range: str,
1097
+ make_plan,
1098
+ *,
1099
+ command: str | None = None,
1100
+ coverage: bool = False,
1101
+ only_python_changes: bool = True,
1102
+ max_commits: int | None = None,
1103
+ progress=None,
1104
+ setup_command: str | None = None,
1105
+ jobs: int = 1,
1106
+ ) -> CorpusReport:
1107
+ """Plan and validate every ``parent -> commit`` pair in ``revision_range``.
1108
+
1109
+ ``make_plan(base, head)`` builds the plan (the CLI binds discovery,
1110
+ manifest and source roots into it). Suites run once per commit thanks to
1111
+ the shared outcome cache; coverage runs are per pair.
1112
+ """
1113
+ report = CorpusReport(str(repo), revision_range, coverage)
1114
+ cache = OutcomeCache()
1115
+ commits = _commits_in_range(repo, revision_range)
1116
+ if max_commits is not None:
1117
+ commits = commits[-max_commits:]
1118
+ entries: list[CorpusEntry] = []
1119
+ for commit, parent, subject in commits:
1120
+ entry = CorpusEntry(commit, parent, subject)
1121
+ report.entries.append(entry)
1122
+ if only_python_changes and not _touches_python(repo, parent, commit):
1123
+ entry.skipped = "no .py files changed"
1124
+ if progress:
1125
+ progress(entry) # reported as skipped, in order
1126
+ else:
1127
+ entries.append(entry)
1128
+
1129
+ def validate_entry(entry: CorpusEntry) -> None:
1130
+ if progress:
1131
+ progress(entry)
1132
+ try:
1133
+ plan = make_plan(entry.parent, entry.commit)
1134
+ entry.changed_symbols = len(plan.changes)
1135
+ entry.targets = sum(1 for d in plan.decisions if d.target.runner == "pytest")
1136
+ entry.selected = sum(
1137
+ 1 for d in plan.decisions if d.selected and d.target.runner == "pytest"
1138
+ )
1139
+ entry.degraded = plan.degraded
1140
+ entry.validation = validate_pytest(
1141
+ plan,
1142
+ repo=repo,
1143
+ command=command,
1144
+ coverage=coverage,
1145
+ outcome_cache=cache,
1146
+ setup_command=setup_command,
1147
+ )
1148
+ except GitError as exc:
1149
+ entry.error = str(exc)
1150
+
1151
+ try:
1152
+ if jobs <= 1:
1153
+ for entry in entries:
1154
+ validate_entry(entry)
1155
+ else:
1156
+ # Pairs are independent: each validates in its own temporary
1157
+ # worktrees and coverage database; the outcome cache is shared with
1158
+ # a no-wait policy (see OutcomeCache). Any exception, including
1159
+ # Ctrl-C, cancels the pairs that have not started instead of
1160
+ # letting them run on.
1161
+ pool = ThreadPoolExecutor(max_workers=jobs)
1162
+ try:
1163
+ futures = [pool.submit(validate_entry, entry) for entry in entries]
1164
+ for future in futures:
1165
+ future.result()
1166
+ except BaseException:
1167
+ pool.shutdown(wait=False, cancel_futures=True)
1168
+ raise
1169
+ pool.shutdown(wait=True)
1170
+ finally:
1171
+ cache.close() # the kept coverage databases
1172
+ return report
1173
+
1174
+
1175
+ def corpus_to_dict(report: CorpusReport) -> dict:
1176
+ def entry(e: CorpusEntry) -> dict:
1177
+ d: dict = {
1178
+ "commit": e.commit,
1179
+ "parent": e.parent,
1180
+ "subject": e.subject,
1181
+ "skipped": e.skipped,
1182
+ "error": e.error,
1183
+ "changed_symbols": e.changed_symbols,
1184
+ "targets": e.targets,
1185
+ "selected": e.selected,
1186
+ "savings": e.savings,
1187
+ "degraded": e.degraded,
1188
+ }
1189
+ if e.validation is not None:
1190
+ v = validation_to_dict(e.validation)
1191
+ d["ok"] = v["ok"]
1192
+ d["outcome"] = v["counts"]
1193
+ d["coverage"] = v["coverage"] and {
1194
+ k: v["coverage"][k] for k in ("counts", "recall", "precision", "missed")
1195
+ }
1196
+ d["missed"] = v["missed"]
1197
+ return d
1198
+
1199
+ return {
1200
+ "repo": report.repo,
1201
+ "range": report.revision_range,
1202
+ "coverage": report.coverage,
1203
+ "ok": report.ok,
1204
+ "totals": {
1205
+ "commits": len(report.entries),
1206
+ "validated": len(report.validated),
1207
+ "skipped": sum(1 for e in report.entries if e.skipped),
1208
+ "errors": sum(1 for e in report.entries if e.error),
1209
+ "outcome_changed": report.outcome_changed,
1210
+ "outcome_missed": report.outcome_missed,
1211
+ "coverage_affected": report.coverage_affected,
1212
+ "coverage_caught": report.coverage_caught,
1213
+ "recall": report.recall,
1214
+ "precision": report.precision,
1215
+ "mean_savings": report.mean_savings,
1216
+ },
1217
+ "entries": [entry(e) for e in report.entries],
1218
+ }
1219
+
1220
+
1221
+ def _corpus_verdict(report: CorpusReport) -> str:
1222
+ """OK, MISSES when a plan missed something, ERRORS when nothing was
1223
+ missed but some commits could not be validated (a suite that does not
1224
+ run in this environment is not a miss, and not a pass either)."""
1225
+ if report.ok:
1226
+ return "OK"
1227
+ missed = report.outcome_missed or any(
1228
+ e.validation is not None and not e.validation.ok for e in report.entries
1229
+ )
1230
+ return "MISSES" if missed else "ERRORS"
1231
+
1232
+
1233
+ def corpus_to_text(report: CorpusReport) -> str:
1234
+ def pct(x: float | None) -> str:
1235
+ return "n/a" if x is None else f"{x:.0%}"
1236
+
1237
+ lines = [
1238
+ f"corpus {report.revision_range}: {len(report.validated)} validated, "
1239
+ f"{sum(1 for e in report.entries if e.skipped)} skipped, "
1240
+ f"{sum(1 for e in report.entries if e.error)} error(s); "
1241
+ f"{_corpus_verdict(report)}",
1242
+ f" outcome changes: {report.outcome_changed} (missed {report.outcome_missed}); "
1243
+ f"mean savings {pct(report.mean_savings)}",
1244
+ ]
1245
+ if report.coverage:
1246
+ lines.append(
1247
+ f" coverage: recall {pct(report.recall)} ({report.coverage_caught} of "
1248
+ f"{report.coverage_affected}), precision {pct(report.precision)}"
1249
+ )
1250
+ for e in report.entries:
1251
+ short = e.commit[:10]
1252
+ if e.skipped:
1253
+ lines.append(f" {short} skipped: {e.skipped} {e.subject}")
1254
+ elif e.error:
1255
+ lines.append(f" {short} ERROR: {e.error.splitlines()[0]} {e.subject}")
1256
+ else:
1257
+ v = e.validation
1258
+ assert v is not None
1259
+ status = "ok" if v.ok else "MISSED"
1260
+ cov = ""
1261
+ if v.coverage is not None:
1262
+ cov = f", recall {pct(v.coverage.recall)}, precision {pct(v.coverage.precision)}"
1263
+ lines.append(
1264
+ f" {short} {status}: {e.selected}/{e.targets} selected "
1265
+ f"(savings {pct(e.savings)}), outcome misses {len(v.missed)}{cov}"
1266
+ f"{' [degraded]' if e.degraded else ''} {e.subject}"
1267
+ )
1268
+ for o in v.missed:
1269
+ lines.append(f" MISSED outcome {o.runner_id}: {o.base} -> {o.head}")
1270
+ if v.coverage is not None:
1271
+ for h in v.coverage.missed:
1272
+ lines.append(
1273
+ f" MISSED coverage {h.runner_id}: executed "
1274
+ f"{', '.join(h.executed_changed)}"
1275
+ )
1276
+ return "\n".join(lines) + "\n"
1277
+
1278
+
1279
+ # --------------------------------------------------------------------------- evidence
1280
+
1281
+ # The recorder's module name inside the project's process (see plugin_environment).
1282
+ PLUGIN = "diffcone_collect"
1283
+ SELECT_PLUGIN = "diffcone_select"
1284
+
1285
+
1286
+ def _python_module_names(index: SourceIndex, source_roots: list[str]) -> set[str]:
1287
+ """The names Python imports the indexed modules by (a prefixed source
1288
+ root names modules diffcone's way, not Python's)."""
1289
+ names = set()
1290
+ dirs = sorted((split_root(r)[0] for r in source_roots), key=len, reverse=True)
1291
+ for symbol in index.symbols.values():
1292
+ if symbol.kind != MODULE:
1293
+ continue
1294
+ for d in dirs:
1295
+ if d in ("", ".") or symbol.path.startswith(d + "/"):
1296
+ rel = symbol.path if d in ("", ".") else symbol.path[len(d) + 1 :]
1297
+ parts = rel[: -len(".py")].split("/")
1298
+ if parts[-1] == "__init__":
1299
+ parts.pop()
1300
+ if parts:
1301
+ names.add(".".join(parts))
1302
+ break
1303
+ return names
1304
+
1305
+
1306
+ def _collect_argv(command: str | None, extra: list[str] | None) -> list[str]:
1307
+ """The suite under the recorder, as a store's ``command`` records it."""
1308
+ return [
1309
+ *shlex.split(command or DEFAULT_COMMANDS["pytest"]),
1310
+ "-p",
1311
+ PLUGIN,
1312
+ "-p",
1313
+ "no:cacheprovider",
1314
+ *(extra or []),
1315
+ ]
1316
+
1317
+
1318
+ def advance_refusal(
1319
+ repo: Path, plan: Plan, evidence: Evidence, command: str | None, extra: list[str] | None
1320
+ ) -> str | None:
1321
+ """Why ``run --collect`` cannot advance ``evidence`` to the plan's head,
1322
+ if it cannot (roadmap item 6). A store names a commit, so head must be
1323
+ the clean checkout; and a fresh record replaces an old one only if the
1324
+ same arguments selected the same cases."""
1325
+ dirty = _dirty(repo)
1326
+ if dirty:
1327
+ return (
1328
+ f"the working tree has {len(dirty)} change(s) (e.g. {dirty[0]}); a recording "
1329
+ "describes a commit, so commit them first"
1330
+ )
1331
+ if plan.head.kind == KIND_COMMIT and plan.head.commit != resolve_commit(repo, "HEAD"):
1332
+ return f"head {plan.head.revision} is not the checked-out commit"
1333
+ if sorted(plan.source_roots) != sorted(evidence.source_roots):
1334
+ return (
1335
+ f"the plan's source roots {plan.source_roots} differ from the recording's "
1336
+ f"{evidence.source_roots}"
1337
+ )
1338
+ wanted = shlex.join(_collect_argv(command, extra))
1339
+ if wanted != evidence.command:
1340
+ return (
1341
+ "the pytest command and arguments differ from the ones the recording was "
1342
+ f"made with, so a new record could cover other cases:\n recording: "
1343
+ f"{evidence.command}\n now: {wanted}"
1344
+ )
1345
+ return None
1346
+
1347
+
1348
+ def _dirty(repo: Path) -> list[str]:
1349
+ """Changes in the working tree that make it differ from its commit.
1350
+ Untracked output nobody runs does not count: Python's bytecode (in a
1351
+ repository that does not ignore it) and ASV's results, environments and
1352
+ HTML beside its configuration."""
1353
+ status = _git(repo, ["status", "--porcelain", "--untracked-files=normal"]).decode(
1354
+ "utf-8", "surrogateescape"
1355
+ )
1356
+ outputs = _asv_output_dirs(repo)
1357
+ found = []
1358
+ for line in status.splitlines():
1359
+ path = line[3:]
1360
+ if not path or path.startswith(str(EVIDENCE_DIR.parts[0]) + "/"):
1361
+ continue
1362
+ if line.startswith("??") and (
1363
+ is_bytecode(path) or any(path.startswith(d + "/") for d in outputs)
1364
+ ):
1365
+ continue
1366
+ found.append(path)
1367
+ return found
1368
+
1369
+
1370
+ def _asv_output_dirs(repo: Path) -> list[str]:
1371
+ """Where ASV writes beside each tracked ``asv.conf.json``: its
1372
+ ``results_dir``, ``env_dir`` and ``html_dir`` (``results``, ``env`` and
1373
+ ``html`` by default), repository-relative."""
1374
+ listed = _git(repo, ["ls-files", "-z", "--", "asv.conf.json", "*/asv.conf.json"])
1375
+ dirs: list[str] = []
1376
+ for raw in listed.split(b"\0"):
1377
+ if not raw:
1378
+ continue
1379
+ path = raw.decode("utf-8", "surrogateescape")
1380
+ here = posixpath.dirname(path)
1381
+ try:
1382
+ text = (repo / path).read_text("utf-8", "replace")
1383
+ data = json.loads(strip_json_comments(text))
1384
+ except (OSError, ValueError):
1385
+ data = {}
1386
+ if not isinstance(data, dict):
1387
+ data = {}
1388
+ for key, default in (("results_dir", "results"), ("env_dir", "env"), ("html_dir", "html")):
1389
+ value = data.get(key, default)
1390
+ if isinstance(value, str) and value.strip():
1391
+ dirs.append(posixpath.normpath(posixpath.join(here, value.strip())))
1392
+ return dirs
1393
+
1394
+
1395
+ def plugin_environment(base: dict[str, str], out: Path | None, root: Path) -> dict[str, str]:
1396
+ """``base`` plus what loading ``-p diffcone_collect`` needs: the recorder
1397
+ importable as a module of its own (a link to ``collect.py`` in a
1398
+ directory of its own, so neither diffcone's package nor anything else of
1399
+ its environment is imported into the project's process), and the
1400
+ recorder's settings. The caller removes the directory (the last
1401
+ ``PYTHONPATH`` entry) when the run ends."""
1402
+ env = dict(base)
1403
+ link_dir = Path(tempfile.mkdtemp(prefix="diffcone-plugin-"))
1404
+ (link_dir / f"{PLUGIN}.py").symlink_to(Path(__file__).parent / "collect.py")
1405
+ (link_dir / f"{SELECT_PLUGIN}.py").symlink_to(Path(__file__).parent / "selection.py")
1406
+ existing = env.get("PYTHONPATH")
1407
+ env["PYTHONPATH"] = os.pathsep.join([*([existing] if existing else []), str(link_dir)])
1408
+ # Absolute and with symlinks resolved, as pytest reports a test's path:
1409
+ # a relative or linked ``--repo`` would otherwise match nothing.
1410
+ env["DIFFCONE_COLLECT_ROOT"] = str(Path(root).resolve())
1411
+ if out is not None:
1412
+ env["DIFFCONE_COLLECT_OUT"] = str(out)
1413
+ return env
1414
+
1415
+
1416
+ @dataclass
1417
+ class CollectResult:
1418
+ evidence: Evidence
1419
+ store: Path
1420
+ command: list[str]
1421
+ returncode: int
1422
+ log: str
1423
+
1424
+
1425
+ def collect_evidence(
1426
+ repo: Path,
1427
+ *,
1428
+ command: str | None,
1429
+ source_roots: list[str],
1430
+ rev: str | None = None,
1431
+ setup_command: str | None = None,
1432
+ reverse_check: bool = False,
1433
+ extra: list[str] | None = None,
1434
+ cache: IndexCache | None = None,
1435
+ env_variables: list[str] | None = None,
1436
+ ) -> CollectResult:
1437
+ """Run the whole suite under the recorder at a commit and write a store.
1438
+
1439
+ Without ``rev`` the repository itself is run, and it must be clean: the
1440
+ evidence is keyed by a commit, so it must describe that commit's code.
1441
+ With ``rev`` a temporary worktree is checked out (and ``setup_command``
1442
+ run in it). ``PYTHONHASHSEED`` is pinned to 0 when unset, and recorded.
1443
+ With ``reverse_check`` the suite runs a second time in reverse order;
1444
+ tests whose records differ are marked unstable and always selected."""
1445
+ commit = resolve_commit(repo, rev or "HEAD")
1446
+ if rev is None:
1447
+ dirty = _dirty(repo)
1448
+ if dirty:
1449
+ raise GitError(
1450
+ f"the working tree has {len(dirty)} change(s) (e.g. {dirty[0]}); evidence "
1451
+ "describes a commit, so collect from a clean checkout or pass --rev"
1452
+ )
1453
+ index, _ = _index_snapshot(repo, commit, source_roots, with_config=False, cache=cache)
1454
+ if index.errors:
1455
+ raise EvidenceError(
1456
+ f"{len(index.errors)} analysis error(s) at {commit[:12]} (e.g. "
1457
+ f"{index.errors[0].path}: {index.errors[0].message}); code in those modules "
1458
+ "could not be mapped to symbols"
1459
+ )
1460
+ modules = _python_module_names(index, source_roots)
1461
+ argv = _collect_argv(command, extra)
1462
+ kind = KIND_COMMIT if rev is not None else KIND_WORKTREE
1463
+ with (
1464
+ _Checkout(repo, kind, commit, setup_command) as cwd,
1465
+ tempfile.TemporaryDirectory(prefix="diffcone-collect-") as tmp,
1466
+ ):
1467
+ base_env = _checkout_env(cwd, source_roots)
1468
+ base_env["DIFFCONE_COLLECT_REPO"] = str(Path(repo).resolve())
1469
+ base_env.setdefault("PYTHONHASHSEED", "0")
1470
+ if env_variables:
1471
+ base_env["DIFFCONE_ENV_VARIABLES"] = ",".join(sorted(set(env_variables)))
1472
+ base_env["DIFFCONE_COLLECT_PACKAGES"] = ",".join(sorted({m.split(".")[0] for m in modules}))
1473
+ runs = [Path(tmp) / "forward"] + ([Path(tmp) / "reverse"] if reverse_check else [])
1474
+ logs, returncode = [], 0
1475
+ for out in runs:
1476
+ env = plugin_environment(base_env, out, cwd)
1477
+ if out.name == "reverse":
1478
+ env["DIFFCONE_COLLECT_REVERSE"] = "1"
1479
+ try:
1480
+ proc = subprocess.run(argv, cwd=cwd, capture_output=True, text=True, env=env)
1481
+ except OSError as exc:
1482
+ raise GitError(f"cannot run {argv[0]!r}: {exc}") from exc
1483
+ finally:
1484
+ shutil.rmtree(Path(env["PYTHONPATH"].split(os.pathsep)[-1]), ignore_errors=True)
1485
+ log = proc.stdout + proc.stderr
1486
+ logs.append(log)
1487
+ if proc.returncode not in (0, 1):
1488
+ raise EvidenceError(
1489
+ f"the suite did not run: {argv[0]!r} exited {proc.returncode} "
1490
+ f"({PYTEST_EXIT.get(proc.returncode, 'unknown')})\n{log[-2000:]}"
1491
+ )
1492
+ returncode = max(returncode, proc.returncode)
1493
+ try:
1494
+ evidence = fold(
1495
+ runs,
1496
+ index,
1497
+ commit=commit,
1498
+ source_roots=source_roots,
1499
+ command=shlex.join(argv),
1500
+ project_modules=modules,
1501
+ )
1502
+ except EvidenceError as exc:
1503
+ # What the runner said about a process that did not finish.
1504
+ crashes = [
1505
+ line for log in logs for line in log.splitlines() if _CRASH_LINE.search(line)
1506
+ ]
1507
+ if crashes:
1508
+ raise EvidenceError(f"{exc}\n" + "\n".join(crashes[:10])) from exc
1509
+ raise
1510
+ store = write_store(evidence, repo / EVIDENCE_DIR)
1511
+ return CollectResult(evidence, store, argv, returncode, "\n".join(logs))
1512
+
1513
+
1514
+ @dataclass
1515
+ class EvidenceRun:
1516
+ result: RunResult
1517
+ # The environment the run met, when it differed from the evidence's:
1518
+ # the evidence plan was then not run, and the static one was.
1519
+ mismatch: dict | None = None
1520
+ static: RunResult | None = None
1521
+ # The static plan was incomplete, so the whole suite ran instead.
1522
+ static_whole: bool = False
1523
+
1524
+ @property
1525
+ def ran(self) -> RunResult:
1526
+ """What actually ran: the static fallback when the environment differed."""
1527
+ return self.static if self.static is not None else self.result
1528
+
1529
+ # With ``advance``: the store written for head, or why none was.
1530
+ advanced: Path | None = None
1531
+ not_advanced: str | None = None
1532
+
1533
+
1534
+ def evidence_env(plan: Plan) -> dict[str, str]:
1535
+ """What a run relying on an evidence plan changes in the environment:
1536
+ ``PYTHONHASHSEED`` as recorded (the records assumed it), and the names of
1537
+ the variables the recording included, so the check compares the same."""
1538
+ env = dict(os.environ)
1539
+ seed = (plan.evidence or {}).get("hash_seed")
1540
+ if seed is not None:
1541
+ env["PYTHONHASHSEED"] = seed
1542
+ variables = (plan.evidence or {}).get("variables")
1543
+ if variables:
1544
+ env["DIFFCONE_ENV_VARIABLES"] = ",".join(variables)
1545
+ return env
1546
+
1547
+
1548
+ def run_with_evidence(
1549
+ plan: Plan,
1550
+ static_plan,
1551
+ *,
1552
+ cwd: Path,
1553
+ command: str | None = None,
1554
+ extra: list[str] | None = None,
1555
+ dry_run: bool = False,
1556
+ advance_from: Evidence | None = None,
1557
+ allow_incomplete: bool = False,
1558
+ ) -> EvidenceRun:
1559
+ """Run an evidence plan's pytest selection with the recorder in check
1560
+ mode: before any test runs, it compares the environment with the one the
1561
+ evidence was recorded in. On a mismatch the session stops, the evidence
1562
+ says nothing about this environment, and ``static_plan()`` (a static plan
1563
+ of the same snapshots) is run instead.
1564
+
1565
+ With ``advance_from`` (the plan's store; the caller has checked
1566
+ :func:`advance_refusal`) the recorder also records, and the store is
1567
+ advanced to head: the run's records for the selected tests, the old ones
1568
+ for the rest."""
1569
+ assert plan.evidence is not None
1570
+ with tempfile.TemporaryDirectory(prefix="diffcone-check-") as tmp:
1571
+ report = Path(tmp) / "environment.json"
1572
+ out = Path(tmp) / "records" if advance_from is not None else None
1573
+ env = plugin_environment(evidence_env(plan), out, cwd)
1574
+ env["DIFFCONE_CHECK_ENV"] = plan.evidence["environment_hash"]
1575
+ env["DIFFCONE_CHECK_REPORT"] = str(report)
1576
+ plugins = ["-p", PLUGIN]
1577
+ if advance_from is not None:
1578
+ modules = _python_module_names(plan.head_index, plan.source_roots)
1579
+ env["DIFFCONE_COLLECT_PACKAGES"] = ",".join(sorted({m.split(".")[0] for m in modules}))
1580
+ plugins += ["-p", "no:cacheprovider"]
1581
+ try:
1582
+ result = run_selected(
1583
+ plan,
1584
+ "pytest",
1585
+ cwd=cwd,
1586
+ command=command,
1587
+ extra=[*plugins, *(extra or [])],
1588
+ dry_run=dry_run,
1589
+ env=env,
1590
+ )
1591
+ if not dry_run and not result.selected:
1592
+ # Nothing selected is a verdict of the evidence too: check
1593
+ # that it applies to this environment before trusting it.
1594
+ argv = build_command("pytest", [], command, ["-p", PLUGIN, *(extra or [])])
1595
+ subprocess.run(argv, cwd=cwd, env={**env, "DIFFCONE_CHECK_ONLY": "1"})
1596
+ finally:
1597
+ shutil.rmtree(Path(env["PYTHONPATH"].split(os.pathsep)[-1]), ignore_errors=True)
1598
+ checked = json.loads(report.read_text("utf-8")) if report.exists() else None
1599
+ if dry_run or (checked is not None and checked.get("match")):
1600
+ run = EvidenceRun(result)
1601
+ if advance_from is not None and out is not None and not dry_run:
1602
+ _advance(run, plan, advance_from, out, cwd)
1603
+ return run
1604
+ # A mismatch, or no report at all (the plugin did not load, a wrapper
1605
+ # dropped the variables): the evidence does not vouch for this run.
1606
+ met = (
1607
+ checked["environment"]
1608
+ if checked is not None
1609
+ else {"unchecked": "the environment check did not run"}
1610
+ )
1611
+ static = static_plan()
1612
+ if static.incomplete_discovery and not allow_incomplete:
1613
+ # The evidence may have settled what the static plan cannot: pytest
1614
+ # may collect tests that are not targets, so run them all.
1615
+ return EvidenceRun(
1616
+ result, met, run_whole(static, cwd=cwd, command=command, extra=extra), True
1617
+ )
1618
+ fallback = run_selected(static, "pytest", cwd=cwd, command=command, extra=extra)
1619
+ return EvidenceRun(result, met, fallback)
1620
+
1621
+
1622
+ def run_whole(
1623
+ plan: Plan, *, cwd: Path, command: str | None = None, extra: list[str] | None = None
1624
+ ) -> RunResult:
1625
+ """Run the whole pytest suite (every target counts as selected)."""
1626
+ targets = [d.target for d in plan.decisions if d.target.runner == "pytest"]
1627
+ argv = build_command("pytest", [], command, list(extra or []))
1628
+ returncode = subprocess.run(argv, cwd=cwd).returncode
1629
+ return RunResult("pytest", argv, targets, len(targets), returncode)
1630
+
1631
+
1632
+ def _advance(run: EvidenceRun, plan: Plan, previous: Evidence, out: Path, repo: Path) -> None:
1633
+ """Write the store for head after ``run`` (roadmap item 6), or say why not."""
1634
+ result = run.result
1635
+ if result.selected and result.returncode not in (0, 1):
1636
+ run.not_advanced = (
1637
+ f"pytest exited {result.returncode} "
1638
+ f"({PYTEST_EXIT.get(result.returncode or 0, 'unknown')}), so the records may be partial"
1639
+ )
1640
+ return
1641
+ if result.selected and not any(out.glob("process-*.json")):
1642
+ run.not_advanced = (
1643
+ f"pytest exited {result.returncode} before any test process finished its "
1644
+ "session (see its output above), so it recorded nothing"
1645
+ )
1646
+ return
1647
+ head = resolve_commit(repo, "HEAD")
1648
+ rerun = {t.runner_id for t in result.selected}
1649
+ fresh = None
1650
+ try:
1651
+ if result.selected:
1652
+ fresh = fold(
1653
+ [out],
1654
+ plan.head_index,
1655
+ commit=head,
1656
+ source_roots=previous.source_roots,
1657
+ command=previous.command,
1658
+ project_modules=_python_module_names(plan.head_index, plan.source_roots),
1659
+ )
1660
+ alive = {d.target.runner_id for d in plan.decisions if d.target.runner == "pytest"}
1661
+ advanced = advance(previous, fresh, rerun, head, alive)
1662
+ run.advanced = write_store(advanced, repo / EVIDENCE_DIR)
1663
+ except EvidenceError as exc:
1664
+ run.not_advanced = str(exc)
1665
+
1666
+
1667
+ def environment_differences(recorded: dict, met: dict) -> list[str]:
1668
+ """Human-readable differences between two recorder environments."""
1669
+ out = []
1670
+ for key in ("implementation", "python", "platform", "machine"):
1671
+ if recorded.get(key) != met.get(key):
1672
+ out.append(f"{key}: {recorded.get(key)!r} recorded, {met.get(key)!r} now")
1673
+ before, after = set(recorded.get("distributions", ())), set(met.get("distributions", ()))
1674
+ for dist in sorted(before - after)[:5]:
1675
+ out.append(f"distribution {dist} recorded, not installed now")
1676
+ for dist in sorted(after - before)[:5]:
1677
+ out.append(f"distribution {dist} installed now, not recorded")
1678
+ for name, value in sorted(recorded.get("variables", {}).items()):
1679
+ if met.get("variables", {}).get(name) != value:
1680
+ out.append(f"{name}: {value!r} recorded, {met['variables'].get(name)!r} now")
1681
+ return out