diffcone 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
diffcone/planner.py ADDED
@@ -0,0 +1,1453 @@
1
+ """Impact planner.
2
+
3
+ Operates purely on the explicit data structures produced by the snapshot
4
+ reader, indexer and classifier. It knows nothing about pytest or ASV: targets
5
+ are opaque (runner, runner_id, entry symbol, lifecycle dependencies) records
6
+ supplied by the manifest.
7
+
8
+ Propagation rules (a dependency edge ``X -> Y`` carries impact from Y to X):
9
+
10
+ * ``references``/``entry``/``lifecycle``/``unresolved_name_match``: any impact
11
+ on Y affects X (behaviour-level).
12
+ * ``defined_in``: X is affected only when its container Y is *structurally*
13
+ affected (added, deleted, definition or dependencies changed, a class body
14
+ change, or itself structurally invalidated by its own container). A plain
15
+ body change of a module does not invalidate every member; members that use
16
+ module state carry their own ``references`` edges.
17
+ * ``imports``: importing a module runs its import-time code, so any impact
18
+ on the imported module reaches the importer (behaviour-level); deleting
19
+ it invalidates the importer and every member of it (structural).
20
+ * ``imports_name``: only a deletion of the imported name propagates
21
+ (structural); the accompanying ``imports`` edge to the module carries
22
+ import-time impact.
23
+
24
+ A change that runs at import (a module body change, a variable, a class, a
25
+ function's decorators or defaults, an added or deleted definition) also
26
+ seeds its module, so every module that transitively imports it is reached.
27
+
28
+ Every selection is backed by a concrete edge path or an explicit fallback
29
+ rule. See internal/design.md.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import functools
35
+ import gc
36
+ from collections import defaultdict, deque
37
+ from collections.abc import Iterable
38
+ from dataclasses import dataclass, field, replace
39
+ from pathlib import Path, PurePosixPath
40
+ from typing import TypeVar
41
+
42
+ from diffcone.cache import IndexCache
43
+ from diffcone.classify import (
44
+ ADDED,
45
+ DEFINITION_CHANGED,
46
+ DELETED,
47
+ DOCSTRING_CHANGED,
48
+ SymbolChange,
49
+ classify,
50
+ )
51
+ from diffcone.declarations import FILENAME as DECLARATION_FILE
52
+ from diffcone.declarations import Declaration
53
+ from diffcone.declarations import load as load_declarations
54
+ from diffcone.discovery import (
55
+ INCOMPLETE_NOTE_KINDS,
56
+ RUNNER_MODULES,
57
+ DiscoveryNote,
58
+ DiscoveryOptions,
59
+ DiscoveryResult,
60
+ discover,
61
+ )
62
+ from diffcone.evidence import Evidence, EvidenceError, has_commit
63
+ from diffcone.indexer import build_index
64
+ from diffcone.manifest import Manifest, Target
65
+ from diffcone.model import (
66
+ CLASS,
67
+ DECLARED,
68
+ DEFINED_IN,
69
+ ENTRY,
70
+ IMPORTS,
71
+ IMPORTS_NAME,
72
+ LIFECYCLE,
73
+ METHOD,
74
+ MODULE,
75
+ REFERENCES,
76
+ UNRESOLVED_DYNAMIC,
77
+ UNRESOLVED_NAME_MATCH,
78
+ VARIABLE,
79
+ AnalysisError,
80
+ Edge,
81
+ SnapshotInfo,
82
+ SourceIndex,
83
+ UnresolvedReference,
84
+ )
85
+ from diffcone.snapshot import (
86
+ CONFIG_FILES,
87
+ INDEX,
88
+ WORKTREE,
89
+ GitError,
90
+ Snapshot,
91
+ changed_paths,
92
+ commit_description,
93
+ file_id,
94
+ read_snapshot,
95
+ resolve_commit,
96
+ split_root,
97
+ )
98
+
99
+ # Impact modes, weakest first. REGISTERED: what a registry holds changed (a
100
+ # function ``@app.command`` registered): code that later calls the registry
101
+ # behaves differently, the import-time code that registers does not.
102
+ REGISTERED = 1
103
+ BEHAVIOR = 2
104
+ STRUCTURAL = 3
105
+
106
+ RULE_DEPENDENCY = "dependency"
107
+ RULE_UNRESOLVED_NAME_MATCH = "unresolved_name_match"
108
+ RULE_DYNAMIC_REFERENCE = "dynamic_reference"
109
+ RULE_ENTRY_UNRESOLVED = "entry_symbol_unresolved"
110
+ RULE_LIFECYCLE_UNRESOLVED = "lifecycle_dependency_unresolved"
111
+ RULE_ANALYSIS_ERROR = "analysis_error"
112
+ RULE_RUNNER_DEPENDENCY = "runner_dependency"
113
+ RULE_ENTRY_DOCSTRING = "entry_docstring_changed"
114
+ RULE_DECLARED_DEPENDENCY = "declared_dependency"
115
+ RULE_NEW_TARGET = "new_target"
116
+ RULE_UNANALYSED_FILE = "unanalysed_file_changed"
117
+ # Evidence mode (evidence_plan.py). ``escalated`` is a selection made by
118
+ # static planning of a change that evidence cannot bound.
119
+ RULE_ESCALATED = "escalated"
120
+ RULE_EXECUTED_CHANGED = "executed_changed"
121
+ RULE_EXECUTED_READER = "executed_reader"
122
+ RULE_LOOKUP_SITE = "lookup_site"
123
+ RULE_TOUCHED_FILE = "touched_file"
124
+ RULE_TEST_SCOPE = "test_scope"
125
+ RULE_CHANGED_TARGET = "changed_target"
126
+ RULE_NO_EVIDENCE = "no_evidence"
127
+ RULE_UNSTABLE = "unstable"
128
+ RULE_SUBPROCESS = "subprocess"
129
+ RULE_PYTEST_HOOK = "pytest_hook_changed"
130
+ RULE_UNOBSERVED_FILE = "unobserved_file_changed"
131
+ RULE_UNINDEXED_IMPORT = "unindexed_import"
132
+ # A test executed a Cython function that names a changed nogil or cpdef one,
133
+ # which a profiled build does not report itself (roadmap item 7).
134
+ RULE_CYTHON_CALLER = "cython_caller"
135
+ # A declared endpoint that names a container stands for everything in it;
136
+ # beyond this many pairs the declaration is too coarse to be useful.
137
+ DECLARATION_FANOUT = 5000
138
+
139
+ # A lifecycle dependency ``dynamic:<module>`` says the target runs code with
140
+ # that module's globals (a doctest): it is affected by any impact-carrying
141
+ # change in the module's import closure, as a dynamic reference is.
142
+ DYNAMIC_DEPENDENCY = "dynamic:"
143
+
144
+ CONSERVATIVE_RULES = frozenset(
145
+ {
146
+ RULE_UNRESOLVED_NAME_MATCH,
147
+ RULE_DYNAMIC_REFERENCE,
148
+ RULE_ENTRY_UNRESOLVED,
149
+ RULE_LIFECYCLE_UNRESOLVED,
150
+ RULE_ANALYSIS_ERROR,
151
+ RULE_RUNNER_DEPENDENCY,
152
+ RULE_UNANALYSED_FILE,
153
+ RULE_LOOKUP_SITE,
154
+ RULE_NO_EVIDENCE,
155
+ RULE_UNSTABLE,
156
+ RULE_SUBPROCESS,
157
+ RULE_PYTEST_HOOK,
158
+ RULE_UNOBSERVED_FILE,
159
+ RULE_UNINDEXED_IMPORT,
160
+ }
161
+ )
162
+
163
+ # Files under the source roots that are diffcone's own, not the project's:
164
+ # the cache, and the declarations the planner already reads from both
165
+ # revisions.
166
+ OWN_FILES = ("diffcone.toml",)
167
+ # Build scripts at the repository root (see plan_from_indexes).
168
+ BUILD_SCRIPTS = frozenset({"setup.py", "hatch_build.py", "build.py", "pdm_build.py"})
169
+ OWN_DIRS = (".diffcone/",)
170
+ UNANALYSED_PATHS_SHOWN = 5
171
+
172
+
173
+ def _changed_unanalysed_files(base: SourceIndex, head: SourceIndex) -> list[str]:
174
+ """Non-Python files under the source roots whose content differs between
175
+ the snapshots (added, deleted or edited). The index reads none of them,
176
+ so it cannot say who depends on one: a data file the code opens, a
177
+ compiled extension's source, a configuration file."""
178
+ paths = base.other_files.keys() | head.other_files.keys()
179
+ return sorted(
180
+ p
181
+ for p in paths
182
+ if base.other_files.get(p) != head.other_files.get(p)
183
+ and p not in OWN_FILES
184
+ and not p.startswith(OWN_DIRS)
185
+ )
186
+
187
+
188
+ def _runner_files_outside_roots(
189
+ repo: Path, base: SourceIndex, head: SourceIndex, roots: list[str]
190
+ ) -> list[str]:
191
+ """Changed files outside the source roots that decide how the tests run:
192
+ the runners' configuration at the repository root (``pyproject.toml``'s
193
+ pytest table, ``tox.ini``, ``asv.conf.json`` anywhere), conftests and
194
+ the build scripts. Inside a root such a file is an unanalysed file; out
195
+ of every root nothing else would see it change."""
196
+ if "" in (split_root(r)[0] for r in roots):
197
+ return []
198
+ if base.snapshot.committed:
199
+ fixed, other = base.snapshot, head.snapshot
200
+ elif head.snapshot.committed:
201
+ fixed, other = head.snapshot, base.snapshot
202
+ else:
203
+ return []
204
+ changed = changed_paths(repo, fixed.commit, other.commit, other.kind)
205
+ dirs = [split_root(r)[0] for r in roots]
206
+ return sorted(
207
+ path
208
+ for path in changed
209
+ if not any(path.startswith(d + "/") for d in dirs)
210
+ and (
211
+ path in CONFIG_FILES
212
+ or path in BUILD_SCRIPTS
213
+ or PurePosixPath(path).name in ("conftest.py", "asv.conf.json")
214
+ )
215
+ )
216
+
217
+
218
+ @dataclass(frozen=True)
219
+ class Seeds:
220
+ """Where a restricted search starts (evidence mode's escalation): the
221
+ changes to plan from, plus other nodes with the reason each one is a
222
+ starting point (a module whose import ran changed code). A restricted
223
+ search adds no dynamic-reference pseudo-seeds, no unanalysed-file
224
+ fallback and no target-level rules (new target, entry docstring,
225
+ ``dynamic:`` dependencies): the caller decides those."""
226
+
227
+ changes: frozenset[str] = frozenset()
228
+ nodes: dict[str, str] = field(default_factory=dict)
229
+
230
+
231
+ @dataclass(frozen=True)
232
+ class Step:
233
+ source: str
234
+ target: str
235
+ kind: str
236
+ detail: str
237
+ revisions: tuple[str, ...]
238
+
239
+
240
+ @dataclass(frozen=True)
241
+ class Reason:
242
+ rule: str
243
+ detail: str
244
+ path: tuple[Step, ...] = ()
245
+ changed_symbol: str | None = None
246
+ changes: tuple[str, ...] = ()
247
+
248
+ @property
249
+ def conservative(self) -> bool:
250
+ return self.rule in CONSERVATIVE_RULES
251
+
252
+
253
+ @dataclass
254
+ class Decision:
255
+ target: Target
256
+ selected: bool
257
+ reasons: list[Reason]
258
+ affected_dependencies: list[str]
259
+ unselected_reason: str | None = None
260
+
261
+
262
+ @dataclass(frozen=True)
263
+ class Fallback:
264
+ rule: str
265
+ scope: str # "all_targets" | "target"
266
+ detail: str
267
+ target: str | None = None
268
+
269
+
270
+ @dataclass(frozen=True)
271
+ class UnresolvedRecord:
272
+ symbol: str
273
+ kind: str
274
+ name: str
275
+ detail: str
276
+ revisions: tuple[str, ...]
277
+ matched_affected_symbols: tuple[str, ...]
278
+
279
+
280
+ @dataclass
281
+ class Plan:
282
+ repo: str
283
+ source_roots: list[str]
284
+ changes: list[SymbolChange]
285
+ decisions: list[Decision]
286
+ fallbacks: list[Fallback]
287
+ unresolved: list[UnresolvedRecord]
288
+ errors: list[AnalysisError]
289
+ base_index: SourceIndex = field(repr=False)
290
+ head_index: SourceIndex = field(repr=False)
291
+ discovery: list[DiscoveryResult] = field(default_factory=list)
292
+ targets: list[Target] = field(default_factory=list)
293
+ declarations: list[Declaration] = field(default_factory=list)
294
+ # Evidence mode: which store planned the pytest targets, and from where
295
+ # (evidence_plan._summary); None for a static plan.
296
+ evidence: dict | None = None
297
+
298
+ @property
299
+ def selected(self) -> list[Decision]:
300
+ return [d for d in self.decisions if d.selected]
301
+
302
+ @property
303
+ def unselected(self) -> list[Decision]:
304
+ return [d for d in self.decisions if not d.selected]
305
+
306
+ @property
307
+ def degraded(self) -> bool:
308
+ return bool(self.errors)
309
+
310
+ @property
311
+ def incomplete_discovery(self) -> list:
312
+ """Discovery notes saying a runner may collect tests that are not
313
+ targets. A degraded plan runs too much; this runs too little, so the
314
+ two are reported (and exited) separately."""
315
+ return [n for d in self.discovery for n in d.incomplete]
316
+
317
+ @property
318
+ def base(self) -> SnapshotInfo:
319
+ return self.base_index.snapshot
320
+
321
+ @property
322
+ def head(self) -> SnapshotInfo:
323
+ return self.head_index.snapshot
324
+
325
+ @property
326
+ def uncommitted_analyzed(self) -> bool:
327
+ """True when either snapshot is the index or the working tree."""
328
+ return not (self.base.committed and self.head.committed)
329
+
330
+ @property
331
+ def working_tree_analyzed(self) -> bool:
332
+ return self.base.is_worktree or self.head.is_worktree
333
+
334
+ @property
335
+ def scope_statement(self) -> str:
336
+ """One sentence saying exactly what was analysed."""
337
+ if not self.uncommitted_analyzed:
338
+ return "two committed snapshots; the working tree was not analyzed"
339
+ sides = [
340
+ name for name, info in (("base", self.base), ("head", self.head)) if not info.committed
341
+ ]
342
+ return f"UNCOMMITTED state was analyzed as {' and '.join(sides)}; see base/head"
343
+
344
+
345
+ # --------------------------------------------------------------------------- graph
346
+
347
+
348
+ @dataclass
349
+ class _Graph:
350
+ """Union of both revisions' edges, indexed for backward traversal."""
351
+
352
+ reverse: dict[str, list[tuple[str, Edge, tuple[str, ...]]]] = field(
353
+ default_factory=lambda: defaultdict(list)
354
+ )
355
+ forward: dict[str, list[tuple[str, Edge, tuple[str, ...]]]] = field(
356
+ default_factory=lambda: defaultdict(list)
357
+ )
358
+
359
+ def add(self, edge: Edge, revisions: tuple[str, ...]) -> None:
360
+ self.reverse[edge.target].append((edge.source, edge, revisions))
361
+ self.forward[edge.source].append((edge.target, edge, revisions))
362
+
363
+ def freeze(self) -> None:
364
+ # A plain tuple key: dataclass ordering on Edge is far slower.
365
+ for adj in (self.reverse, self.forward):
366
+ for key in adj:
367
+ adj[key].sort(
368
+ key=lambda item: (item[0], item[1].kind, item[1].detail, item[1].target)
369
+ )
370
+
371
+
372
+ T = TypeVar("T")
373
+
374
+
375
+ def _union(base_items: Iterable[T], head_items: Iterable[T]) -> dict[T, tuple[str, ...]]:
376
+ """Merge two revisions' items, tagging each with the revisions it appears in.
377
+
378
+ Insertion order is not significant: adjacency lists are sorted in
379
+ ``_Graph.freeze`` and records are sorted before reporting, so the items
380
+ are not sorted here (comparing tens of thousands of dataclasses was a
381
+ third of planning time)."""
382
+ revs: dict[T, list[str]] = defaultdict(list)
383
+ for item in base_items:
384
+ revs[item].append("base")
385
+ for item in head_items:
386
+ revs[item].append("head")
387
+ return {item: tuple(r) for item, r in revs.items()}
388
+
389
+
390
+ def _propagate(
391
+ edge: Edge, target_mode: int, target_change: SymbolChange | None, source_is_module: bool
392
+ ) -> int | None:
393
+ if edge.kind == DEFINED_IN:
394
+ return STRUCTURAL if target_mode == STRUCTURAL else None
395
+ if edge.detail == "registers":
396
+ return REGISTERED
397
+ if target_mode == REGISTERED and source_is_module:
398
+ return None # registering ran at import, unchanged
399
+ if edge.kind in (IMPORTS, IMPORTS_NAME):
400
+ if target_change is not None and DELETED in target_change.changes:
401
+ return STRUCTURAL
402
+ # Importing a module runs its import-time code.
403
+ return BEHAVIOR if edge.kind == IMPORTS else None
404
+ return BEHAVIOR
405
+
406
+
407
+ def _runs_at_import(change: SymbolChange) -> bool:
408
+ """Module bodies, variable initialisers and class bodies run at import;
409
+ so does a ``def`` statement's decorators, defaults and (when evaluated
410
+ eagerly) annotations. A function body does not (what import-time code
411
+ calls is reached through its edges), and an inert ``def`` at both
412
+ revisions only binds a name (removing or rebinding it reaches its users
413
+ through deletion and resolution edges)."""
414
+ symbol = change.symbol
415
+ if symbol.kind in (MODULE, CLASS):
416
+ return True
417
+ if symbol.kind == VARIABLE:
418
+ # Binding a literal runs nothing; a computed value does, and so does
419
+ # an eagerly evaluated annotation that changed (definition_changed).
420
+ if DEFINITION_CHANGED in change.changes:
421
+ return True
422
+ return not all(s.inert_definition for s in (change.base, change.head) if s is not None)
423
+ if not {ADDED, DELETED, DEFINITION_CHANGED} & set(change.changes):
424
+ return False
425
+ return not all(s.inert_definition for s in (change.base, change.head) if s is not None)
426
+
427
+
428
+ # --------------------------------------------------------------------------- planning
429
+
430
+
431
+ def merge_targets(manifest: Manifest | None, discovered: list[DiscoveryResult]) -> list[Target]:
432
+ """Manifest targets win over discovered ones with the same runner and id."""
433
+ merged: dict[tuple[str, str], Target] = {}
434
+ for result in discovered:
435
+ for target in result.targets:
436
+ merged.setdefault((target.runner, target.runner_id), target)
437
+ if manifest is not None:
438
+ for target in manifest.targets:
439
+ merged[(target.runner, target.runner_id)] = target
440
+ return sorted(merged.values())
441
+
442
+
443
+ def _members_by_container(symbols: set[str]) -> dict[str, list[str]]:
444
+ """Symbols inside each module or class, by qualified name. Identity is the
445
+ dotted path, so a container's members are the symbols under its prefix."""
446
+ members: dict[str, list[str]] = defaultdict(list)
447
+ for symbol in symbols:
448
+ prefix = symbol
449
+ while "." in prefix:
450
+ prefix = prefix.rsplit(".", 1)[0]
451
+ if prefix in symbols:
452
+ members[prefix].append(symbol)
453
+ return members
454
+
455
+
456
+ def _runner_only_classes(
457
+ base: SourceIndex,
458
+ head: SourceIndex,
459
+ discovered: list[DiscoveryResult],
460
+ edges: dict[Edge, tuple[str, ...]],
461
+ ) -> set[str]:
462
+ """Test classes whose instances only the test runner ever holds.
463
+
464
+ A call ``obj.m()`` reaches ``C.m`` only if ``obj`` is an instance of C or
465
+ of a subclass. pytest instantiates a test class to run its tests; if the
466
+ analysed code never constructs it, never refers to it as a value and
467
+ never hands an instance on, those are the only instances, and they never
468
+ leave the class's own methods. Nothing outside it can be holding one, so
469
+ no name-matched call from outside can land on its members. (Calls on
470
+ ``self`` inside it are resolved through the MRO and are not name matches.)
471
+
472
+ pandas is why this matters: its library code says ``x.dtype``,
473
+ ``x.index``, ``x.copy`` thousands of times, and every one of those matched
474
+ a fixture or helper of the same name on some test class -- one of which
475
+ held a dynamic reference that every test reached through ``DataFrame``.
476
+
477
+ A class counts as runner-instantiated when it owns the entry of a pytest
478
+ target, or is the collecting class a target lists among its lifecycle
479
+ dependencies. It is *held* -- and keeps its members as candidates -- when
480
+ an instance or the class is passed on (``escaped_classes``), when a
481
+ symbol outside every runner class refers to it, or when a held class
482
+ inherits from it. A read of an attribute named ``instance`` anywhere is
483
+ pytest's ``request.instance``, the one channel that hands a test instance
484
+ to other code, and turns the rule off. (Evidence mode has its own
485
+ version, evidence_plan._runner_only.)"""
486
+ symbols = {**base.symbols, **head.symbols}
487
+ runner: set[str] = set()
488
+ for result in discovered:
489
+ if result.runner != "pytest":
490
+ continue
491
+ for target in result.targets:
492
+ entry = symbols.get(target.entry_symbol)
493
+ if entry is not None and entry.kind == METHOD and entry.container:
494
+ runner.add(entry.container)
495
+ for dep in target.lifecycle_dependencies:
496
+ owner = symbols.get(dep)
497
+ if owner is not None and owner.kind == CLASS:
498
+ runner.add(dep)
499
+ if not runner:
500
+ return set()
501
+ for index in (base, head):
502
+ if any(u.kind != UNRESOLVED_DYNAMIC and u.name == "instance" for u in index.unresolved):
503
+ return set()
504
+
505
+ def within_runner(symbol_id: str) -> bool:
506
+ return _inside(symbol_id, runner, base, head)
507
+
508
+ held = (base.escaped_classes | head.escaped_classes) & runner
509
+ for edge in edges:
510
+ if edge.kind == REFERENCES and edge.target in runner and not within_runner(edge.source):
511
+ held.add(edge.target)
512
+ # A held subclass holds its bases too: its instances carry their methods.
513
+ changed = True
514
+ while changed:
515
+ changed = False
516
+ for edge in edges:
517
+ if (
518
+ edge.kind == REFERENCES
519
+ and edge.source in held
520
+ and edge.target in runner
521
+ and edge.target not in held
522
+ ):
523
+ held.add(edge.target)
524
+ changed = True
525
+ return runner - held
526
+
527
+
528
+ def _inside(symbol_id: str, classes: set[str], base: SourceIndex, head: SourceIndex) -> bool:
529
+ """Whether ``symbol_id`` is one of ``classes`` or sits inside one."""
530
+ current: str | None = symbol_id
531
+ while current:
532
+ if current in classes:
533
+ return True
534
+ symbol = head.symbols.get(current) or base.symbols.get(current)
535
+ current = symbol.container if symbol is not None else None
536
+ return False
537
+
538
+
539
+ def plan_from_indexes(
540
+ base: SourceIndex,
541
+ head: SourceIndex,
542
+ manifest: Manifest | None,
543
+ *,
544
+ repo: str = "",
545
+ source_roots: list[str] | None = None,
546
+ discovered: list[DiscoveryResult] | None = None,
547
+ declarations: list[Declaration] | None = None,
548
+ base_target_ids: set[str] | None = None,
549
+ seeds: Seeds | None = None,
550
+ runner_files: Iterable[str] = (),
551
+ ) -> Plan:
552
+ """``runner_files``: changed runner configuration outside the source
553
+ roots (_runner_files_outside_roots), which selects every target."""
554
+ discovered = list(discovered or []) + _manifest_notes(manifest, discovered or [])
555
+ discovered_ids = {t.runner_id for result in discovered for t in result.targets}
556
+ declared = list(declarations or [])
557
+ targets = merge_targets(manifest, discovered)
558
+ changes = classify(base, head)
559
+ change_by_id = {c.id: c for c in changes}
560
+ known_symbols = set(base.symbols) | set(head.symbols)
561
+ fallbacks: list[Fallback] = []
562
+ errors = sorted(base.errors + head.errors)
563
+
564
+ graph = _Graph()
565
+ union = _union(base.edges, head.edges)
566
+ for edge, revs in union.items():
567
+ graph.add(edge, revs)
568
+
569
+ # An instance handed to someone else can have any attribute read off it by
570
+ # a name nothing resolves (``invoke(obj, name)``), and no static rule can
571
+ # say which. So referring to such a class depends on its members, not only
572
+ # on its structure.
573
+ class_members = _members_by_container(set(base.symbols) | set(head.symbols))
574
+ for cls in sorted(base.escaped_classes | head.escaped_classes):
575
+ for member in class_members.get(cls, ()):
576
+ graph.add(
577
+ Edge(cls, member, REFERENCES, "attribute of a class passed to other code"),
578
+ ("base", "head"),
579
+ )
580
+
581
+ # Dependencies the project declares (diffcone.toml): the analysis cannot
582
+ # see them, and they only add edges, so they widen selection and never
583
+ # narrow it. An endpoint that is in neither revision is an analysis
584
+ # error: a declaration that silently does nothing is worth failing on.
585
+ # An endpoint that names a module or a class means everything in it, so
586
+ # it expands to that container's members -- a change to one of them is
587
+ # what the declaration is about, and the container node alone would
588
+ # never see it.
589
+ members = _members_by_container(known_symbols)
590
+ for decl in declared:
591
+ missing = [e for e in (decl.source, decl.target) if e not in known_symbols]
592
+ if missing:
593
+ errors.append(
594
+ AnalysisError(
595
+ revision=head.snapshot.revision,
596
+ path=DECLARATION_FILE,
597
+ message=(
598
+ f"declared edge {decl.source!r} -> {decl.target!r} names "
599
+ f"{' and '.join(repr(m) for m in missing)}, which is in neither revision"
600
+ ),
601
+ )
602
+ )
603
+ continue
604
+ from_side = [decl.source, *members.get(decl.source, ())]
605
+ to_side = [decl.target, *members.get(decl.target, ())]
606
+ if len(from_side) * len(to_side) > DECLARATION_FANOUT:
607
+ errors.append(
608
+ AnalysisError(
609
+ revision=head.snapshot.revision,
610
+ path=DECLARATION_FILE,
611
+ message=(
612
+ f"declared edge {decl.source!r} -> {decl.target!r} joins "
613
+ f"{len(from_side)} and {len(to_side)} symbols, more than "
614
+ f"{DECLARATION_FANOUT} pairs; declare the symbols that depend "
615
+ "on each other instead"
616
+ ),
617
+ )
618
+ )
619
+ continue
620
+ for source in from_side:
621
+ for target in to_side:
622
+ if source != target:
623
+ graph.add(Edge(source, target, DECLARED, decl.detail), ("declared",))
624
+
625
+ # Conservative edges from unresolved references: ``obj.run()`` may be any
626
+ # known ``run`` (function, method or class) in either revision, and
627
+ # impact flows through the graph as usual (matching only *changed*
628
+ # symbols would miss a ``run`` that is unchanged but calls something that
629
+ # changed). Each name gets one pseudo-node ``name:<n>`` so the edge count
630
+ # is linear in references plus symbols. Dunder names (``__init__``,
631
+ # ``__eq__``) are excluded: they exist on nearly every class and bound
632
+ # nothing; constructors are reached through explicit class references.
633
+ # Dynamic references are pseudo-seeds.
634
+ symbols_by_name: dict[str, list[str]] = defaultdict(list)
635
+ runner_only = _runner_only_classes(base, head, discovered, union)
636
+ for symbol_id in sorted(known_symbols):
637
+ symbol = head.symbols.get(symbol_id) or base.symbols[symbol_id]
638
+ if symbol.kind != MODULE and not _is_dunder(symbol.name):
639
+ if _inside(symbol_id, runner_only, base, head):
640
+ continue # only the test runner can hold an instance: see below
641
+ symbols_by_name[symbol.name].append(symbol_id)
642
+ for name, symbols in symbols_by_name.items():
643
+ for symbol_id in symbols:
644
+ graph.add(Edge(_name_node(name), symbol_id, UNRESOLVED_NAME_MATCH), ("both",))
645
+ pending_unresolved: list[tuple[UnresolvedReference, tuple[str, ...]]] = []
646
+ dynamic_symbols: dict[str, tuple[str, ...]] = {}
647
+ unbounded_dynamic: set[str] = set() # dynamic *imports*: reach anything
648
+ for ref, revs in _union(base.unresolved, head.unresolved).items():
649
+ if ref.kind == UNRESOLVED_DYNAMIC:
650
+ dynamic_symbols.setdefault(ref.symbol, revs)
651
+ if "import" in ref.detail:
652
+ unbounded_dynamic.add(ref.symbol)
653
+ elif ref.name in symbols_by_name and not _is_dunder(ref.name):
654
+ graph.add(
655
+ Edge(ref.symbol, _name_node(ref.name), UNRESOLVED_NAME_MATCH, ref.detail), revs
656
+ )
657
+ pending_unresolved.append((ref, revs))
658
+
659
+ # Targets join the graph as nodes with explicit dependency edges.
660
+ dynamic_deps: dict[str, list[str]] = defaultdict(list)
661
+ # Target -> its entry's module, when the manifest does not list it.
662
+ entry_modules: dict[str, str] = {}
663
+ for target in targets:
664
+ if target.entry_symbol in known_symbols:
665
+ graph.add(Edge(target.node_id, target.entry_symbol, ENTRY), ("manifest",))
666
+ entry = head.symbols.get(target.entry_symbol) or base.symbols.get(target.entry_symbol)
667
+ if entry is not None and entry.module != target.entry_symbol:
668
+ if entry.module not in target.lifecycle_dependencies:
669
+ entry_modules[target.node_id] = entry.module
670
+ else:
671
+ fallbacks.append(
672
+ Fallback(
673
+ RULE_ENTRY_UNRESOLVED,
674
+ "target",
675
+ f"entry symbol {target.entry_symbol!r} was not found in either revision",
676
+ target=target.node_id,
677
+ )
678
+ )
679
+ for dep in target.lifecycle_dependencies:
680
+ module = dep[len(DYNAMIC_DEPENDENCY) :] if dep.startswith(DYNAMIC_DEPENDENCY) else None
681
+ if module is not None and module in known_symbols:
682
+ dynamic_deps[target.node_id].append(module)
683
+ elif dep in known_symbols:
684
+ graph.add(Edge(target.node_id, dep, LIFECYCLE), ("manifest",))
685
+ else:
686
+ fallbacks.append(
687
+ Fallback(
688
+ RULE_LIFECYCLE_UNRESOLVED,
689
+ "target",
690
+ f"lifecycle dependency {dep!r} was not found in either revision",
691
+ target=target.node_id,
692
+ )
693
+ )
694
+ graph.freeze()
695
+ seeded = changes if seeds is None else [c for c in changes if c.id in seeds.changes]
696
+ fallbacks += _runner_dependency_fallbacks(targets, seeded, base, head)
697
+ unanalysed = _changed_unanalysed_files(base, head) if seeds is None else []
698
+ if seeds is None:
699
+ # A build script is Python nobody imports, but it decides what is
700
+ # compiled and installed: a change to it is as unbounded as a changed
701
+ # compiled source.
702
+ unanalysed += sorted({c.symbol.path for c in changes if c.symbol.path in BUILD_SCRIPTS})
703
+ if unanalysed:
704
+ shown = ", ".join(unanalysed[:UNANALYSED_PATHS_SHOWN])
705
+ more = len(unanalysed) - UNANALYSED_PATHS_SHOWN
706
+ fallbacks.append(
707
+ Fallback(
708
+ RULE_UNANALYSED_FILE,
709
+ "all_targets",
710
+ f"{len(unanalysed)} file(s) the analysis does not read changed under the source "
711
+ f"roots ({shown}{f', and {more} more' if more > 0 else ''}); code or tests may "
712
+ "read them, so every supplied target is selected",
713
+ )
714
+ )
715
+ outside = sorted(runner_files) if seeds is None else []
716
+ if outside:
717
+ shown = ", ".join(outside[:UNANALYSED_PATHS_SHOWN])
718
+ more = len(outside) - UNANALYSED_PATHS_SHOWN
719
+ fallbacks.append(
720
+ Fallback(
721
+ RULE_UNANALYSED_FILE,
722
+ "all_targets",
723
+ f"{len(outside)} runner configuration or build file(s) outside the source roots "
724
+ f"changed ({shown}{f', and {more} more' if more > 0 else ''}); they decide how "
725
+ "the tests run, so every supplied target is selected",
726
+ )
727
+ )
728
+
729
+ # Backward reachability from every changed symbol.
730
+ mode: dict[str, int] = {}
731
+ via: dict[str, tuple[Edge, tuple[str, ...], str] | None] = {}
732
+ queue: deque[str] = deque()
733
+ impacting = [c for c in seeded if c.carries_impact]
734
+ for change in impacting:
735
+ mode[change.id] = STRUCTURAL if change.structural else BEHAVIOR
736
+ via[change.id] = None
737
+ queue.append(change.id)
738
+ # A change that runs when its module is imported affects the module's
739
+ # import, hence (through ``imports`` edges) every importer.
740
+ for change in impacting:
741
+ symbol = change.symbol
742
+ if symbol.module not in mode and _runs_at_import(change):
743
+ mode[symbol.module] = BEHAVIOR
744
+ via[symbol.module] = None
745
+ queue.append(symbol.module)
746
+ # A symbol reading docstrings (``f.__doc__``, ``getdoc(cls)``, its
747
+ # module's ``__doc__``) sees a docstring-only change of what it references.
748
+ documented = {c.id for c in seeded if DOCSTRING_CHANGED in c.changes}
749
+ # A docstring its decorator reads (pandas' ``@doc`` formats it) changes
750
+ # what the decorator does when the module is imported.
751
+ decorated = documented & (base.doc_decorated | head.doc_decorated)
752
+ for change in seeded:
753
+ if change.id in decorated:
754
+ for node in (change.id, change.symbol.module):
755
+ if node not in mode:
756
+ mode[node] = BEHAVIOR
757
+ via[node] = None
758
+ queue.append(node)
759
+ if documented:
760
+ for index in (base, head):
761
+ read: dict[str, set[str]] = {
762
+ s.id: {s.module} for s in index.symbols.values() if s.reads_docstrings
763
+ }
764
+ for e in index.edges:
765
+ if e.source in read and e.kind != DEFINED_IN:
766
+ read[e.source].add(e.target)
767
+ for reader, referenced in sorted(read.items()):
768
+ if reader not in mode and referenced & documented:
769
+ mode[reader] = BEHAVIOR
770
+ via[reader] = None
771
+ queue.append(reader)
772
+ seed_reasons = dict(seeds.nodes) if seeds is not None else {}
773
+ for node in sorted(seed_reasons):
774
+ if node not in mode:
775
+ mode[node] = BEHAVIOR
776
+ via[node] = None
777
+ queue.append(node)
778
+ # A dynamic reference (eval/exec/getattr with an unbounded name) can reach
779
+ # whatever its module's globals can reach: the module itself and every
780
+ # module it imports, transitively. A dynamic *import* can reach anything.
781
+ # That bound holds only while the object read is one of those globals.
782
+ # ``def invoke(obj, name): getattr(obj, name)`` reads an object a caller
783
+ # supplied, which can belong to any module; the index binds that read at
784
+ # the other end instead (a class whose instances are handed around gains
785
+ # an edge to each of its members), and the closure check stays as well.
786
+ changed_modules = {c.symbol.module for c in impacting}
787
+ reach = _ImportReach(base, head)
788
+
789
+ if impacting and seeds is None:
790
+ for symbol in sorted(dynamic_symbols):
791
+ if symbol in mode:
792
+ continue
793
+ # A symbol can hold several dynamic references; the widest one
794
+ # decides, so an import that names anything is not narrowed by a
795
+ # getattr beside it.
796
+ if symbol in unbounded_dynamic or reach.closure_of(symbol) & changed_modules:
797
+ mode[symbol] = BEHAVIOR
798
+ via[symbol] = None
799
+ queue.append(symbol)
800
+ module_nodes = {
801
+ s.id for index in (base, head) for s in index.symbols.values() if s.kind == MODULE
802
+ }
803
+ while queue:
804
+ node = queue.popleft()
805
+ node_mode = mode[node]
806
+ for source, edge, revs in graph.reverse.get(node, ()):
807
+ new_mode = _propagate(edge, node_mode, change_by_id.get(node), source in module_nodes)
808
+ if new_mode is None or new_mode <= mode.get(source, 0):
809
+ continue
810
+ mode[source] = new_mode
811
+ # A node whose mode rises keeps its old explanation when the new
812
+ # one would lead back through itself (REG -> register -> REG): the
813
+ # explanation must end at a change, not loop.
814
+ if source not in via or not _leads_to(via, node, source):
815
+ via[source] = (edge, revs, node)
816
+ queue.append(source)
817
+ # The runner imports a target's module to reach it, so the module's
818
+ # import-time code runs before the target whether or not a hand-written
819
+ # manifest lists the module (discovered targets always do). Applied after
820
+ # the search, so a target that a more specific path reaches keeps it.
821
+ for node_id, module in sorted(entry_modules.items()):
822
+ if node_id not in mode and module in mode:
823
+ mode[node_id] = BEHAVIOR
824
+ via[node_id] = (Edge(node_id, module, LIFECYCLE), ("manifest",), module)
825
+
826
+ affected_by_name = {
827
+ name: tuple(s for s in symbols if s in mode) for name, symbols in symbols_by_name.items()
828
+ }
829
+ unresolved_records = [
830
+ UnresolvedRecord(
831
+ ref.symbol,
832
+ ref.kind,
833
+ ref.name,
834
+ ref.detail,
835
+ revs,
836
+ tuple(s for s in affected_by_name.get(ref.name, ()) if s != ref.symbol)
837
+ if ref.kind != UNRESOLVED_DYNAMIC
838
+ else (),
839
+ )
840
+ for ref, revs in pending_unresolved
841
+ ]
842
+
843
+ if errors:
844
+ fallbacks.append(
845
+ Fallback(
846
+ RULE_ANALYSIS_ERROR,
847
+ "all_targets",
848
+ f"{len(errors)} analysis error(s); the dependency graph is incomplete, "
849
+ "so every supplied target is selected",
850
+ )
851
+ )
852
+ target_fallbacks: dict[str, list[Fallback]] = defaultdict(list)
853
+ global_fallbacks: list[Fallback] = []
854
+ for fb in fallbacks:
855
+ if fb.scope == "target" and fb.target:
856
+ target_fallbacks[fb.target].append(fb)
857
+ else:
858
+ global_fallbacks.append(fb)
859
+
860
+ decisions: list[Decision] = []
861
+ for target in targets:
862
+ reasons: list[Reason] = []
863
+ if target.node_id in mode:
864
+ reasons.append(
865
+ _explain(
866
+ target.node_id,
867
+ via,
868
+ change_by_id,
869
+ dynamic_symbols,
870
+ unbounded_dynamic,
871
+ seed_reasons,
872
+ )
873
+ )
874
+ for fb in target_fallbacks.get(target.node_id, ()):
875
+ reasons.append(Reason(fb.rule, fb.detail))
876
+ for fb in global_fallbacks:
877
+ reasons.append(Reason(fb.rule, fb.detail))
878
+ if seeds is not None:
879
+ decisions.append(_decision(target, reasons, mode))
880
+ continue
881
+ for module in dynamic_deps.get(target.node_id, ()):
882
+ if reach.closure_of(module) & changed_modules:
883
+ reasons.append(
884
+ Reason(
885
+ RULE_DYNAMIC_REFERENCE,
886
+ f"the target runs code with the globals of {module}, and a change lies "
887
+ "in that module's import closure",
888
+ )
889
+ )
890
+ # A discovered target the base snapshot did not have is new, whatever
891
+ # its entry symbol did: ``from support import test_shared as
892
+ # test_new`` adds a test whose entry is untouched. Only discovery can
893
+ # answer this, and only because it runs at the base as well; a
894
+ # manifest names targets without saying when they appeared.
895
+ if (
896
+ base_target_ids is not None
897
+ and target.runner_id in discovered_ids
898
+ and target.runner_id not in base_target_ids
899
+ ):
900
+ reasons.append(
901
+ Reason(
902
+ RULE_NEW_TARGET,
903
+ f"{target.runner_id} is not in the base snapshot: a new target is selected "
904
+ "whatever its entry symbol did",
905
+ )
906
+ )
907
+ entry_change = change_by_id.get(target.entry_symbol)
908
+ if entry_change is not None and DOCSTRING_CHANGED in entry_change.changes:
909
+ # For a doctest the docstring is the test; for anything else
910
+ # re-running a target whose own docstring changed is cheap.
911
+ reasons.append(
912
+ Reason(
913
+ RULE_ENTRY_DOCSTRING,
914
+ f"the docstring of the entry symbol {target.entry_symbol} changed",
915
+ )
916
+ )
917
+ decisions.append(_decision(target, reasons, mode))
918
+
919
+ return Plan(
920
+ repo=repo,
921
+ source_roots=list(source_roots or []),
922
+ changes=changes,
923
+ decisions=decisions,
924
+ fallbacks=fallbacks,
925
+ unresolved=sorted(unresolved_records, key=lambda r: (r.symbol, r.kind, r.name, r.detail)),
926
+ errors=errors,
927
+ declarations=sorted(declared),
928
+ base_index=base,
929
+ head_index=head,
930
+ discovery=discovered,
931
+ targets=targets,
932
+ )
933
+
934
+
935
+ def _decision(target: Target, reasons: list[Reason], mode: dict[str, int]) -> Decision:
936
+ affected = sorted(
937
+ dep for dep in (target.entry_symbol, *target.lifecycle_dependencies) if dep in mode
938
+ )
939
+ selected = bool(reasons)
940
+ return Decision(
941
+ target=target,
942
+ selected=selected,
943
+ reasons=reasons,
944
+ affected_dependencies=affected,
945
+ unselected_reason=None
946
+ if selected
947
+ else "no dependency path from this target to a changed symbol in either revision",
948
+ )
949
+
950
+
951
+ def _runner_dependency_fallbacks(
952
+ targets: list[Target], changes: list[SymbolChange], base: SourceIndex, head: SourceIndex
953
+ ) -> list[Fallback]:
954
+ """A runner whose own process imports modules that are in the source
955
+ roots (pytest runs pluggy's hooks for every test) runs project code for
956
+ every target: a change there selects all of that runner's targets. The
957
+ modules are discovery's RUNNER_MODULES, those under them, and their
958
+ import closure."""
959
+ impacting = [c for c in changes if c.carries_impact]
960
+ if not impacting:
961
+ return []
962
+ modules = base.modules | head.modules
963
+ reach = _ImportReach(base, head)
964
+ fallbacks: list[Fallback] = []
965
+ for runner in sorted({t.runner for t in targets}):
966
+ entries = RUNNER_MODULES.get(runner, ())
967
+ roots = {m for m in modules if any(m == e or m.startswith(e + ".") for e in entries)}
968
+ if not roots:
969
+ continue
970
+ runner_modules: set[str] = set()
971
+ for module in roots:
972
+ runner_modules |= reach.closure_of(module)
973
+ hit = sorted(c.id for c in impacting if c.symbol.module in runner_modules)
974
+ if not hit:
975
+ continue
976
+ shown = ", ".join(hit[:3]) + (f" and {len(hit) - 3} more" if len(hit) > 3 else "")
977
+ detail = (
978
+ f"{shown} changed in code the {runner} runner itself imports and runs for every "
979
+ "target, so every target of that runner is selected"
980
+ )
981
+ for target in targets:
982
+ if target.runner == runner:
983
+ fallbacks.append(
984
+ Fallback(RULE_RUNNER_DEPENDENCY, "target", detail, target=target.node_id)
985
+ )
986
+ return fallbacks
987
+
988
+
989
+ class _ImportReach:
990
+ """Transitive import closure of modules, over both revisions."""
991
+
992
+ def __init__(self, base: SourceIndex, head: SourceIndex) -> None:
993
+ self.module_of: dict[str, str] = {}
994
+ self.imports_of: dict[str, set[str]] = defaultdict(set)
995
+ for index in (base, head):
996
+ for symbol in index.symbols.values():
997
+ self.module_of[symbol.id] = symbol.module
998
+ for index in (base, head):
999
+ for edge in index.edges:
1000
+ if edge.kind == IMPORTS and edge.target in self.module_of:
1001
+ source_module = self.module_of.get(edge.source)
1002
+ if source_module is not None:
1003
+ self.imports_of[source_module].add(edge.target)
1004
+ self._closures: dict[str, set[str]] = {}
1005
+
1006
+ def closure_of(self, symbol_id: str) -> set[str]:
1007
+ module = self.module_of.get(symbol_id)
1008
+ if module is None:
1009
+ return set()
1010
+ if module not in self._closures:
1011
+ seen = {module}
1012
+ stack = [module]
1013
+ while stack:
1014
+ for target in self.imports_of.get(stack.pop(), ()):
1015
+ if target not in seen:
1016
+ seen.add(target)
1017
+ stack.append(target)
1018
+ self._closures[module] = seen
1019
+ return self._closures[module]
1020
+
1021
+
1022
+ def _name_node(name: str) -> str:
1023
+ return f"name:{name}"
1024
+
1025
+
1026
+ def _is_name_node(node_id: str) -> bool:
1027
+ return node_id.startswith("name:")
1028
+
1029
+
1030
+ def _is_dunder(name: str) -> bool:
1031
+ return len(name) > 4 and name.startswith("__") and name.endswith("__")
1032
+
1033
+
1034
+ def _leads_to(
1035
+ via: dict[str, tuple[Edge, tuple[str, ...], str] | None], start: str, target: str
1036
+ ) -> bool:
1037
+ """Whether the explanation chain from ``start`` passes through ``target``."""
1038
+ seen: set[str] = set()
1039
+ current: str | None = start
1040
+ while current is not None and current not in seen:
1041
+ if current == target:
1042
+ return True
1043
+ seen.add(current)
1044
+ link = via.get(current)
1045
+ current = link[2] if link is not None else None
1046
+ return False
1047
+
1048
+
1049
+ def _explain(
1050
+ node: str,
1051
+ via: dict[str, tuple[Edge, tuple[str, ...], str] | None],
1052
+ change_by_id: dict[str, SymbolChange],
1053
+ dynamic_symbols: dict[str, tuple[str, ...]],
1054
+ unbounded_dynamic: set[str],
1055
+ seed_reasons: dict[str, str] | None = None,
1056
+ ) -> Reason:
1057
+ steps: list[Step] = []
1058
+ current = node
1059
+ rule = RULE_DEPENDENCY
1060
+ pending: tuple[str, str, tuple[str, ...]] | None = None # (source, detail, revs)
1061
+ walked: set[str] = set()
1062
+ while True:
1063
+ link = via[current]
1064
+ if link is None or current in walked:
1065
+ break
1066
+ walked.add(current)
1067
+ edge, revs, nxt = link
1068
+ if edge.kind == DECLARED and rule == RULE_DEPENDENCY:
1069
+ # The path only holds because the project said so; say which rule
1070
+ # carried it rather than calling it an ordinary dependency. A name
1071
+ # match anywhere on the path outranks it: that one is our guess,
1072
+ # this one is the project's statement.
1073
+ rule = RULE_DECLARED_DEPENDENCY
1074
+ if edge.kind == UNRESOLVED_NAME_MATCH:
1075
+ rule = RULE_UNRESOLVED_NAME_MATCH # outranks a declared edge
1076
+ if _is_name_node(edge.target):
1077
+ pending = (edge.source, edge.detail, revs) # collapse the pseudo-node
1078
+ current = nxt
1079
+ continue
1080
+ if pending is not None:
1081
+ source, detail, revs = pending
1082
+ steps.append(Step(source, edge.target, UNRESOLVED_NAME_MATCH, detail, revs))
1083
+ pending = None
1084
+ current = nxt
1085
+ continue
1086
+ steps.append(Step(edge.source, edge.target, edge.kind, edge.detail, revs))
1087
+ current = nxt
1088
+ change = change_by_id.get(current)
1089
+ if change is not None:
1090
+ detail = f"{current} {'/'.join(change.changes)}"
1091
+ return Reason(rule, detail, tuple(steps), current, change.changes)
1092
+ if seed_reasons and current in seed_reasons:
1093
+ return Reason(RULE_ESCALATED, seed_reasons[current], tuple(steps))
1094
+ # Pseudo-seed: a symbol with a dynamic reference.
1095
+ revs = dynamic_symbols.get(current, ())
1096
+ if current in unbounded_dynamic:
1097
+ return Reason(
1098
+ RULE_DYNAMIC_REFERENCE,
1099
+ f"{current} imports a module named at runtime ({', '.join(revs)}); any module in "
1100
+ "scope may be behind it, so its dependencies cannot be bounded statically",
1101
+ tuple(steps),
1102
+ )
1103
+ return Reason(
1104
+ RULE_DYNAMIC_REFERENCE,
1105
+ f"{current} uses a dynamic import/attribute access ({', '.join(revs)}); "
1106
+ "a change is reachable from its module's imports, so its dependencies cannot be "
1107
+ "bounded statically",
1108
+ tuple(steps),
1109
+ None,
1110
+ (),
1111
+ )
1112
+
1113
+
1114
+ def _index_snapshot(
1115
+ repo_path: Path,
1116
+ revision: str,
1117
+ roots: list[str],
1118
+ *,
1119
+ with_config: bool,
1120
+ cache: IndexCache | None,
1121
+ ) -> tuple[SourceIndex, Snapshot | None]:
1122
+ """Index a snapshot, serving committed snapshots from the cache. Returns
1123
+ the snapshot too when it had to be read (discovery needs its files)."""
1124
+ if cache is not None and revision not in (WORKTREE, INDEX) and not with_config:
1125
+ try:
1126
+ commit = resolve_commit(repo_path, revision)
1127
+ except GitError:
1128
+ commit = None
1129
+ if commit is not None:
1130
+ cached = cache.load(commit, roots)
1131
+ if cached is not None:
1132
+ # The cached index was built for whatever spelling of this
1133
+ # commit came first; the report shows this one.
1134
+ info = replace(
1135
+ cached.snapshot,
1136
+ revision=revision,
1137
+ description=commit_description(commit, revision),
1138
+ )
1139
+ cached = replace(
1140
+ cached,
1141
+ snapshot=info,
1142
+ errors=[replace(e, revision=revision) for e in cached.errors],
1143
+ )
1144
+ return cached, None
1145
+ snapshot = read_snapshot(repo_path, revision, roots, with_config=with_config)
1146
+ index = build_index(snapshot, module_cache=cache.modules if cache is not None else None)
1147
+ if cache is not None and snapshot.info.committed:
1148
+ cache.store(index, roots)
1149
+ return index, snapshot
1150
+
1151
+
1152
+ def _cacheable_commit(repo_path: Path, revision: str) -> str | None:
1153
+ """The commit a revision names, or None for ``WORKTREE``, ``INDEX`` and
1154
+ anything that does not resolve: what may be served from a cache."""
1155
+ if revision in (WORKTREE, INDEX):
1156
+ return None
1157
+ try:
1158
+ return resolve_commit(repo_path, revision)
1159
+ except GitError:
1160
+ return None
1161
+
1162
+
1163
+ def without_cyclic_gc(func):
1164
+ """Run ``func`` with Python's cyclic garbage collector suspended, and
1165
+ restore its state after. A plan holds two whole indexes (on pandas,
1166
+ millions of objects) while it allocates syntax trees and sets, so each
1167
+ collection walks the whole heap: on pandas the collector took two thirds
1168
+ of a warm plan (46-56 s with it, 18 s without). Planning makes almost
1169
+ no reference cycles; they are collected once the collector resumes."""
1170
+
1171
+ @functools.wraps(func)
1172
+ def inner(*args, **kwargs):
1173
+ enabled = gc.isenabled()
1174
+ gc.disable()
1175
+ try:
1176
+ return func(*args, **kwargs)
1177
+ finally:
1178
+ if enabled:
1179
+ gc.enable()
1180
+
1181
+ return inner
1182
+
1183
+
1184
+ def _settle_discovery(
1185
+ repo: Path,
1186
+ head: str,
1187
+ evidence: Evidence,
1188
+ evidence_index: SourceIndex,
1189
+ discovered: list[DiscoveryResult],
1190
+ roots: list[str],
1191
+ options: DiscoveryOptions,
1192
+ cache: IndexCache | None,
1193
+ ) -> list[DiscoveryResult]:
1194
+ """What the recording says about pytest discovery's completeness
1195
+ (roadmap item 9). It ran pytest's real collection at C, in the
1196
+ environment ``run`` checks before any test. A note saying a plugin may
1197
+ collect tests that are not targets stops counting when the same note was
1198
+ there at C, its file is unchanged since C, and pytest collected nothing
1199
+ at C that was not a target then. A test it did collect that was not a
1200
+ target is a gap the recording proves, noted as incomplete itself."""
1201
+ if evidence.collected is None or not any(d.runner == "pytest" for d in discovered):
1202
+ return discovered
1203
+ at_c = (
1204
+ cache.discovery.load(evidence.commit, roots, "pytest", options)
1205
+ if cache is not None
1206
+ else None
1207
+ )
1208
+ if at_c is None:
1209
+ snapshot = read_snapshot(repo, evidence.commit, roots, with_config=True)
1210
+ at_c = discover("pytest", snapshot, evidence_index, options)
1211
+ if cache is not None:
1212
+ cache.discovery.store(at_c, evidence.commit, roots, options)
1213
+ targets_at_c = {t.runner_id for t in at_c.targets}
1214
+ extra = sorted(evidence.collected - targets_at_c)
1215
+ notes_at_c = {(n.kind, n.detail) for n in at_c.notes}
1216
+ short = evidence.commit[:12]
1217
+ out = []
1218
+ for result in discovered:
1219
+ if result.runner != "pytest":
1220
+ out.append(result)
1221
+ continue
1222
+ notes = []
1223
+ for note in result.notes:
1224
+ if (
1225
+ note.kind in INCOMPLETE_NOTE_KINDS
1226
+ and not extra
1227
+ and note.path
1228
+ and (note.kind, note.detail) in notes_at_c
1229
+ and file_id(repo, evidence.commit, note.path) == file_id(repo, head, note.path)
1230
+ ):
1231
+ note = DiscoveryNote(
1232
+ note.runner,
1233
+ "settled_by_evidence",
1234
+ f"{note.detail} [settled: at {short}, where this note stood too, pytest "
1235
+ f"collected no test that was not a target, and {note.path} is unchanged "
1236
+ "since]",
1237
+ note.path,
1238
+ )
1239
+ notes.append(note)
1240
+ if extra:
1241
+ notes.append(
1242
+ DiscoveryNote(
1243
+ "pytest",
1244
+ "collected_not_target",
1245
+ f"the recording at {short} collected {len(extra)} test(s) that were not "
1246
+ f"targets there, e.g. {', '.join(extra[:3])}",
1247
+ )
1248
+ )
1249
+ out.append(DiscoveryResult(result.runner, result.targets, notes, result.config))
1250
+ return out
1251
+
1252
+
1253
+ def _check_roots_match(
1254
+ roots: list[str], base: str, base_index: SourceIndex, head: str, head_index: SourceIndex
1255
+ ) -> None:
1256
+ """A source root holding no Python file in either revision is almost
1257
+ certainly a mistake (a typo, a path outside the repository), and would
1258
+ plan nothing as complete: make it an analysis error, so the plan
1259
+ selects everything and says why."""
1260
+ paths = [
1261
+ symbol.path
1262
+ for index in (base_index, head_index)
1263
+ for symbol in index.symbols.values()
1264
+ if symbol.kind == MODULE
1265
+ ]
1266
+ for root in roots:
1267
+ directory = split_root(root)[0]
1268
+ if not directory:
1269
+ continue
1270
+ if not any(path.startswith(directory + "/") for path in paths):
1271
+ head_index.errors.append(
1272
+ AnalysisError(
1273
+ revision=head,
1274
+ path=directory,
1275
+ message=(
1276
+ f"source root {root!r} holds no Python file at {base} or {head}; "
1277
+ "check the path (it is relative to the repository)"
1278
+ ),
1279
+ )
1280
+ )
1281
+
1282
+
1283
+ def _manifest_notes(
1284
+ manifest: Manifest | None, discovered: list[DiscoveryResult]
1285
+ ) -> list[DiscoveryResult]:
1286
+ """The notes ``discover`` wrote into a manifest, for runners this plan
1287
+ did not discover itself: a target list it said may be short keeps the
1288
+ plan at exit code 3."""
1289
+ if manifest is None or not manifest.notes:
1290
+ return []
1291
+ fresh = {result.runner for result in discovered}
1292
+ by_runner: dict[str, list[DiscoveryNote]] = defaultdict(list)
1293
+ for runner, kind, detail, path in manifest.notes:
1294
+ if runner not in fresh:
1295
+ by_runner[runner].append(DiscoveryNote(runner, kind, detail, path))
1296
+ return [
1297
+ DiscoveryResult(runner, notes=notes, config={"from_manifest": True})
1298
+ for runner, notes in sorted(by_runner.items())
1299
+ ]
1300
+
1301
+
1302
+ def _with_base_lifecycle(
1303
+ discovered: list[DiscoveryResult], base_lifecycle: dict[tuple[str, str], tuple[str, ...]]
1304
+ ) -> list[DiscoveryResult]:
1305
+ """Head targets with the lifecycle dependencies the same target had at
1306
+ the base added: a fixture, conftest or setup the test used before the
1307
+ change and no longer does (deleted with its autouse fixture, a removed
1308
+ override) still decides whether the change reaches it, as every other
1309
+ edge counts in both revisions."""
1310
+ merged: list[DiscoveryResult] = []
1311
+ for result in discovered:
1312
+ targets = []
1313
+ for target in result.targets:
1314
+ before = base_lifecycle.get((target.runner, target.runner_id), ())
1315
+ extra = [d for d in before if d not in target.lifecycle_dependencies]
1316
+ if extra:
1317
+ deps = tuple(sorted({*target.lifecycle_dependencies, *extra}))
1318
+ target = replace(target, lifecycle_dependencies=deps)
1319
+ targets.append(target)
1320
+ merged.append(replace(result, targets=targets))
1321
+ return merged
1322
+
1323
+
1324
+ @without_cyclic_gc
1325
+ def plan(
1326
+ repo: str | Path,
1327
+ base: str,
1328
+ head: str,
1329
+ manifest: Manifest | None = None,
1330
+ source_roots: list[str] | None = None,
1331
+ discover_runners: Iterable[str] = (),
1332
+ discovery_options: DiscoveryOptions | None = None,
1333
+ cache: IndexCache | None = None,
1334
+ evidence: Evidence | None = None,
1335
+ ) -> Plan:
1336
+ """Analyse two snapshots and produce a selection plan.
1337
+
1338
+ Targets come from the manifest, from static discovery of the head
1339
+ snapshot for each runner in ``discover_runners``, or both. ``base`` and
1340
+ ``head`` are git revisions, ``INDEX`` or ``WORKTREE``; the plan records
1341
+ which kind each one was. With ``evidence`` (recorded by ``diffcone
1342
+ collect``) pytest targets are selected on what each test executed
1343
+ (evidence_plan.py); the evidence's source roots must match.
1344
+ """
1345
+ repo_path = Path(repo)
1346
+ manifest_roots = manifest.source_roots if manifest is not None else None
1347
+ roots = list(source_roots or manifest_roots or ["."])
1348
+ runners = list(discover_runners)
1349
+ options = discovery_options or DiscoveryOptions()
1350
+ base_index, _ = _index_snapshot(repo_path, base, roots, with_config=False, cache=cache)
1351
+ # Discovery of a committed head may come from the cache, and then so may
1352
+ # the head index; otherwise discovery needs the head snapshot's files.
1353
+ discovery_cache = cache.discovery if cache is not None else None
1354
+ head_commit = _cacheable_commit(repo_path, head) if discovery_cache is not None else None
1355
+ cached_head: list[DiscoveryResult] | None = None
1356
+ if runners and head_commit is not None and discovery_cache is not None:
1357
+ found = [discovery_cache.load(head_commit, roots, r, options) for r in runners]
1358
+ hits = [result for result in found if result is not None]
1359
+ if len(hits) == len(runners):
1360
+ cached_head = hits
1361
+ head_index, head_snapshot = _index_snapshot(
1362
+ repo_path, head, roots, with_config=bool(runners) and cached_head is None, cache=cache
1363
+ )
1364
+ # Both revisions, as every other edge is: a commit that deletes a
1365
+ # declaration while changing what it pointed at must still select.
1366
+ # Discovery at the base too, so a target the base did not have is known
1367
+ # to be new. Only the snapshot is read again (0.1 s on the largest
1368
+ # repositories); the base index still comes from the cache.
1369
+ base_target_ids: set[str] | None = None
1370
+ base_lifecycle: dict[tuple[str, str], tuple[str, ...]] = {}
1371
+ if runners:
1372
+ base_commit = _cacheable_commit(repo_path, base) if discovery_cache is not None else None
1373
+ base_target_ids = set()
1374
+ base_snapshot: Snapshot | None = None
1375
+ for runner in runners:
1376
+ result = (
1377
+ discovery_cache.load(base_commit, roots, runner, options)
1378
+ if base_commit is not None and discovery_cache is not None
1379
+ else None
1380
+ )
1381
+ if result is None:
1382
+ if base_snapshot is None:
1383
+ base_snapshot = read_snapshot(repo_path, base, roots, with_config=True)
1384
+ result = discover(runner, base_snapshot, base_index, options)
1385
+ if base_commit is not None and discovery_cache is not None:
1386
+ discovery_cache.store(result, base_commit, roots, options)
1387
+ base_target_ids |= {target.runner_id for target in result.targets}
1388
+ for target in result.targets:
1389
+ base_lifecycle[(target.runner, target.runner_id)] = target.lifecycle_dependencies
1390
+ _check_roots_match(roots, base, base_index, head, head_index)
1391
+ declared: list[Declaration] = []
1392
+ for revision, index in ((base, base_index), (head, head_index)):
1393
+ found, problems = load_declarations(repo_path, revision)
1394
+ declared += found
1395
+ for problem in problems:
1396
+ index.errors.append(
1397
+ AnalysisError(revision=revision, path=DECLARATION_FILE, message=problem)
1398
+ )
1399
+ declared = sorted(set(declared))
1400
+ discovered: list[DiscoveryResult] = []
1401
+ if cached_head is not None:
1402
+ discovered = cached_head
1403
+ elif runners:
1404
+ if head_snapshot is None: # pragma: no cover
1405
+ raise GitError("discovery needs the head snapshot's files")
1406
+ discovered = [discover(runner, head_snapshot, head_index, options) for runner in runners]
1407
+ if head_commit is not None and discovery_cache is not None:
1408
+ for result in discovered:
1409
+ discovery_cache.store(result, head_commit, roots, options)
1410
+ discovered = _with_base_lifecycle(discovered, base_lifecycle)
1411
+ if evidence is not None:
1412
+ from diffcone.evidence_plan import plan_with_evidence
1413
+
1414
+ if not has_commit(repo_path, evidence.commit):
1415
+ raise EvidenceError(
1416
+ f"the evidence was recorded at {evidence.commit[:12]}, which this checkout does "
1417
+ f"not have (a shallow clone?); fetch it: git fetch --depth=1 origin "
1418
+ f"{evidence.commit}"
1419
+ )
1420
+ if sorted(evidence.source_roots) != sorted(roots):
1421
+ raise EvidenceError(
1422
+ f"the evidence was recorded with source roots {evidence.source_roots}, the plan "
1423
+ f"uses {roots}: symbols would not line up"
1424
+ )
1425
+ evidence_index, _ = _index_snapshot(
1426
+ repo_path, evidence.commit, roots, with_config=False, cache=cache
1427
+ )
1428
+ discovered = _settle_discovery(
1429
+ repo_path, head, evidence, evidence_index, discovered, roots, options, cache
1430
+ )
1431
+ return plan_with_evidence(
1432
+ base_index,
1433
+ head_index,
1434
+ evidence,
1435
+ evidence_index,
1436
+ manifest,
1437
+ repo=str(repo_path),
1438
+ source_roots=roots,
1439
+ discovered=discovered,
1440
+ declarations=declared,
1441
+ base_target_ids=base_target_ids,
1442
+ )
1443
+ return plan_from_indexes(
1444
+ base_index,
1445
+ head_index,
1446
+ manifest,
1447
+ repo=str(repo_path),
1448
+ source_roots=roots,
1449
+ discovered=discovered,
1450
+ declarations=declared,
1451
+ base_target_ids=base_target_ids,
1452
+ runner_files=_runner_files_outside_roots(repo_path, base_index, head_index, roots),
1453
+ )