diffcone 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
diffcone/snapshot.py ADDED
@@ -0,0 +1,739 @@
1
+ """Git snapshot reader.
2
+
3
+ Three kinds of snapshot can be read, and every one says what it is:
4
+
5
+ * ``commit`` (any git revision): sources come straight from the object store;
6
+ nothing is checked out and the working tree is not touched.
7
+ * ``INDEX``: the staged content of every tracked file (what ``git commit``
8
+ would record right now).
9
+ * ``WORKTREE``: the files on disk, tracked or untracked, excluding ignored
10
+ ones; tracked files deleted from disk are absent.
11
+
12
+ The last two are always reported as uncommitted state on top of ``HEAD``.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import os
18
+ import posixpath
19
+ import subprocess
20
+ from collections.abc import Callable
21
+ from dataclasses import dataclass, field
22
+ from pathlib import Path
23
+
24
+ from diffcone.cython import is_cython
25
+ from diffcone.model import (
26
+ KIND_COMMIT,
27
+ KIND_INDEX,
28
+ KIND_WORKTREE,
29
+ AnalysisError,
30
+ SnapshotInfo,
31
+ )
32
+
33
+
34
+ class GitError(Exception):
35
+ """Raised when git cannot supply the requested snapshot."""
36
+
37
+
38
+ CONFIG_FILES = (
39
+ "pytest.toml",
40
+ ".pytest.toml",
41
+ "pytest.ini",
42
+ ".pytest.ini",
43
+ "pyproject.toml",
44
+ "tox.ini",
45
+ "setup.cfg",
46
+ "asv.conf.json",
47
+ )
48
+ # ASV projects usually keep their configuration beside the benchmarks rather
49
+ # than at the repository root (numpy and networkx use ``benchmarks/``, pandas
50
+ # ``asv_bench/``), and ``benchmark_dir`` is relative to it, so nested copies
51
+ # are read too -- shallowest first, and only a few levels down.
52
+ ASV_CONFIG = "asv.conf.json"
53
+ ASV_CONFIG_DEPTH = 3
54
+ NESTED_CONFIGS = (ASV_CONFIG, "pyproject.toml", "setup.cfg")
55
+
56
+ WORKTREE = "WORKTREE"
57
+ INDEX = "INDEX"
58
+
59
+
60
+ @dataclass
61
+ class Snapshot:
62
+ info: SnapshotInfo
63
+ source_roots: list[str]
64
+ files: dict[str, bytes] = field(default_factory=dict) # repo-relative path -> content
65
+ # Root-level runner configuration files, when present (see CONFIG_FILES).
66
+ config_files: dict[str, bytes] = field(default_factory=dict)
67
+ # Problems reading the snapshot itself (e.g. unmerged index entries).
68
+ errors: list[AnalysisError] = field(default_factory=list)
69
+ # The other (non-Python) files under the source roots: path -> git blob
70
+ # id, so a change to one is visible without reading it; and (read only
71
+ # with ``with_config``) the content of the text files among them that
72
+ # pytest could collect as doctests.
73
+ other_files: dict[str, str] = field(default_factory=dict)
74
+ # The content of the Cython sources among them (diffcone.cython).
75
+ cython_files: dict[str, bytes] = field(default_factory=dict)
76
+ text_files: dict[str, bytes] = field(default_factory=dict)
77
+ # Every ``.py`` path in the whole tree, roots or not (read only with
78
+ # ``with_config``): discovery reports test files pytest would collect
79
+ # outside the source roots instead of silently missing them.
80
+ python_paths: tuple[str, ...] = ()
81
+
82
+ @property
83
+ def revision(self) -> str:
84
+ return self.info.revision
85
+
86
+ @property
87
+ def commit(self) -> str:
88
+ return self.info.commit
89
+
90
+ @property
91
+ def kind(self) -> str:
92
+ return self.info.kind
93
+
94
+ @property
95
+ def description(self) -> str:
96
+ return self.info.description
97
+
98
+
99
+ def _git(repo: Path, args: list[str], stdin: bytes | None = None) -> bytes:
100
+ try:
101
+ proc = subprocess.run(
102
+ ["git", *args],
103
+ cwd=repo,
104
+ input=stdin,
105
+ capture_output=True,
106
+ check=False,
107
+ )
108
+ except FileNotFoundError as exc: # pragma: no cover - environment problem
109
+ raise GitError("git executable not found") from exc
110
+ if proc.returncode != 0:
111
+ message = proc.stderr.decode("utf-8", "replace").strip() or "git command failed"
112
+ raise GitError(f"git {' '.join(args[:2])}: {message}")
113
+ return proc.stdout
114
+
115
+
116
+ def file_id(repo: Path, revision: str, path: str) -> str | None:
117
+ """The git blob id of ``path`` in a snapshot (a revision, ``INDEX`` or
118
+ ``WORKTREE``), or None when it is not there."""
119
+ try:
120
+ if revision == WORKTREE:
121
+ if not (repo / path).is_file():
122
+ return None
123
+ out = _git(repo, ["hash-object", "--", path])
124
+ else:
125
+ spec = f":{path}" if revision == INDEX else f"{revision}:{path}"
126
+ out = _git(repo, ["rev-parse", "--verify", "--quiet", spec])
127
+ except GitError:
128
+ return None
129
+ return out.decode().strip() or None
130
+
131
+
132
+ def changed_paths(repo: Path, commit: str, revision: str, kind: str) -> dict[str, str]:
133
+ """Every path whose content differs between ``commit`` and a snapshot
134
+ (``kind``: a commit, the index or the working tree, untracked files
135
+ included), as path -> "added", "deleted" or "edited"."""
136
+ if kind == KIND_WORKTREE:
137
+ args = ["diff", "--name-status", "-z", "--no-renames", commit]
138
+ elif kind == KIND_INDEX:
139
+ args = ["diff", "--cached", "--name-status", "-z", "--no-renames", commit]
140
+ else:
141
+ args = ["diff", "--name-status", "-z", "--no-renames", commit, revision]
142
+ fields = _git(repo, args).decode("utf-8", "surrogateescape").split("\0")
143
+ out: dict[str, str] = {}
144
+ for status, path in zip(fields[0::2], fields[1::2], strict=False):
145
+ if path:
146
+ out[path] = {"A": "added", "D": "deleted"}.get(status[:1], "edited")
147
+ if kind == KIND_WORKTREE:
148
+ untracked = _git(repo, ["ls-files", "-z", "--others", "--exclude-standard"])
149
+ for raw in untracked.split(b"\0"):
150
+ path = raw.decode("utf-8", "surrogateescape")
151
+ if path and not is_bytecode(path):
152
+ out.setdefault(path, "added")
153
+ return out
154
+
155
+
156
+ def resolve_commit(repo: Path, revision: str) -> str:
157
+ try:
158
+ out = _git(repo, ["rev-parse", "--verify", "--quiet", f"{revision}^{{commit}}"])
159
+ except GitError:
160
+ out = b""
161
+ commit = out.decode().strip()
162
+ if not commit:
163
+ hint = ""
164
+ if is_shallow(repo):
165
+ hint = (
166
+ "; this clone is shallow and may not have it: fetch more history (git fetch "
167
+ "--deepen=1 for a parent such as HEAD^1, or git fetch origin <branch>; in GitHub "
168
+ "Actions, actions/checkout with fetch-depth: 2 or 0)"
169
+ )
170
+ raise GitError(f"revision {revision!r} does not name a commit in {repo}{hint}")
171
+ return commit
172
+
173
+
174
+ def is_bytecode(path: str) -> bool:
175
+ """Whether ``path`` is compiled bytecode Python writes beside the code
176
+ (``__pycache__``, ``.pyc``): never part of a snapshot, even in a
177
+ repository that does not ignore it."""
178
+ return "__pycache__" in path.rstrip("/").split("/") or path.endswith((".pyc", ".pyo"))
179
+
180
+
181
+ def is_shallow(repo: Path) -> bool:
182
+ """Whether ``repo`` is a shallow clone (its history is cut off)."""
183
+ try:
184
+ return _git(repo, ["rev-parse", "--is-shallow-repository"]).strip() == b"true"
185
+ except GitError:
186
+ return False
187
+
188
+
189
+ def split_root(spec: str) -> tuple[str, str]:
190
+ """``(directory, module prefix)`` of a source-root spec.
191
+
192
+ A spec is a repo-relative directory, optionally followed by ``=PREFIX``:
193
+ modules under the directory are then named ``PREFIX.<path>`` instead of
194
+ ``<path>``. That gives a monorepo's per-package test trees, whose files
195
+ share names (``opentelemetry-api/tests/trace/test_globals.py`` and
196
+ ``opentelemetry-sdk/tests/trace/test_globals.py``), distinct identities
197
+ when one pytest session collects them (``--import-mode=importlib``). A
198
+ prefixed name is diffcone's, never Python's: it is not used to resolve
199
+ imports. The directory is normalised (``.`` and ``""`` are the root).
200
+ """
201
+ directory, sep, prefix = spec.partition("=")
202
+ directory = directory.strip()
203
+ while directory.startswith("./"):
204
+ directory = directory[2:]
205
+ directory = directory.strip("/")
206
+ directory = "" if directory in ("", ".") else directory
207
+ prefix = prefix.strip() if sep else ""
208
+ if sep and (not prefix or not all(p.isidentifier() for p in prefix.split("."))):
209
+ raise ValueError(f"source root {spec!r}: the module prefix must be a dotted identifier")
210
+ return directory, prefix
211
+
212
+
213
+ def _normalise_root(root: str) -> str:
214
+ return split_root(root)[0]
215
+
216
+
217
+ SYMLINK_MODE = "120000"
218
+
219
+
220
+ def _ls_tree_ids(repo: Path, commit: str, pathspecs: list[str]) -> list[tuple[str, str, str]]:
221
+ """(mode, blob id, path) of every blob under ``pathspecs`` (all when empty)."""
222
+ args = ["ls-tree", "-r", "-z", "--full-tree", commit]
223
+ if pathspecs:
224
+ args += ["--", *pathspecs]
225
+ entries: list[tuple[str, str, str]] = []
226
+ for record in _git(repo, args).split(b"\0"):
227
+ if not record:
228
+ continue
229
+ meta, _, path = record.decode("utf-8", "surrogateescape").partition("\t")
230
+ mode, _, oid = meta.split()
231
+ entries.append((mode, oid, path))
232
+ return entries
233
+
234
+
235
+ def _ls_tree(repo: Path, commit: str, pathspecs: list[str]) -> list[tuple[str, str]]:
236
+ """(mode, path) of every blob under ``pathspecs`` (all when empty)."""
237
+ return [(mode, path) for mode, _, path in _ls_tree_ids(repo, commit, pathspecs)]
238
+
239
+
240
+ def _root_pathspecs(source_roots: list[str]) -> list[str]:
241
+ roots = [_normalise_root(r) for r in source_roots]
242
+ return [] if "" in roots else [r for r in roots if r]
243
+
244
+
245
+ # Suffixes of files pytest's ``--doctest-glob`` commonly collects.
246
+ TEXT_DOCTEST_SUFFIXES = (".txt", ".rst", ".md")
247
+
248
+
249
+ def _text_paths(paths: list[str] | tuple[str, ...]) -> list[str]:
250
+ return [p for p in paths if p.endswith(TEXT_DOCTEST_SUFFIXES)]
251
+
252
+
253
+ def _link_target(link: str, target: str) -> str | None:
254
+ """The repository path a relative symlink at ``link`` points to, or None
255
+ when it is absolute, leaves the repository or is the repository root."""
256
+ if not target or target.startswith("/"):
257
+ return None
258
+ real = posixpath.normpath(posixpath.join(posixpath.dirname(link), target))
259
+ if real in (".", "..") or real.startswith("../"):
260
+ return None
261
+ return real
262
+
263
+
264
+ def expand_symlinks(
265
+ links: dict[str, str], files_under: Callable[[str], list[str]]
266
+ ) -> dict[str, str]:
267
+ """Python files reached through tracked symlinks inside the repository:
268
+ ``{path through the link: real path}``. A file link maps itself; a
269
+ directory link maps every file under its target to the same relative
270
+ path under the link (pytest collects them there). ``files_under`` lists
271
+ the non-link files at or under a real path, so links inside an expanded
272
+ tree are not followed again."""
273
+ aliases: dict[str, str] = {}
274
+ for link, target in sorted(links.items()):
275
+ real = _link_target(link, target)
276
+ if real is None:
277
+ continue
278
+ for path in files_under(real):
279
+ if path == real:
280
+ alias = link
281
+ elif path.startswith(real + "/"):
282
+ alias = link + path[len(real) :]
283
+ else:
284
+ continue
285
+ if alias.endswith(".py"):
286
+ aliases.setdefault(alias, path)
287
+ return aliases
288
+
289
+
290
+ def read_files(
291
+ repo: Path, commit: str, paths: list[str], *, label: str | None = None
292
+ ) -> dict[str, bytes]:
293
+ """Read blobs ``<commit>:<path>``; ``commit=""`` reads the index (``:path``)."""
294
+ if not paths:
295
+ return {}
296
+ label = label or commit
297
+ request = "".join(f"{commit}:{p}\n" for p in paths).encode("utf-8", "surrogateescape")
298
+ out = _git(repo, ["cat-file", "--batch"], stdin=request)
299
+ files: dict[str, bytes] = {}
300
+ pos = 0
301
+ for path in paths:
302
+ newline = out.index(b"\n", pos)
303
+ header = out[pos:newline].decode("utf-8", "replace")
304
+ pos = newline + 1
305
+ parts = header.split()
306
+ if len(parts) < 3 or parts[-1] == "missing":
307
+ raise GitError(f"cannot read {path} at {label}: {header}")
308
+ size = int(parts[2])
309
+ files[path] = out[pos : pos + size]
310
+ pos += size + 1 # trailing newline after each object
311
+ return files
312
+
313
+
314
+ def list_root_files(repo: Path, commit: str) -> set[str]:
315
+ out = _git(repo, ["ls-tree", "--name-only", "-z", commit])
316
+ return {p.decode("utf-8", "surrogateescape") for p in out.split(b"\0") if p}
317
+
318
+
319
+ def _pathspec(source_roots: list[str]) -> list[str]:
320
+ roots = [_normalise_root(r) for r in source_roots]
321
+ if "" in roots:
322
+ return []
323
+ return ["--", *[r for r in roots if r]]
324
+
325
+
326
+ # ``git ls-files -t`` tags: H cached, S skip-worktree (sparse checkout), M unmerged,
327
+ # ? untracked (with --others).
328
+ TAG_CACHED, TAG_SKIP_WORKTREE, TAG_UNMERGED, TAG_OTHER = "H", "S", "M", "?"
329
+
330
+
331
+ def _ls_files_tagged(repo: Path, args: list[str], source_roots: list[str]) -> list[tuple[str, str]]:
332
+ out = _git(repo, ["ls-files", "-z", "-t", *args, *_pathspec(source_roots)])
333
+ entries: set[tuple[str, str]] = set()
334
+ for record in out.split(b"\0"):
335
+ if not record:
336
+ continue
337
+ tag, _, path = record.decode("utf-8", "surrogateescape").partition(" ")
338
+ entries.add((tag, path))
339
+ return sorted(entries, key=lambda e: (e[1], e[0]))
340
+
341
+
342
+ def _ls_files_staged_ids(repo: Path, source_roots: list[str]) -> dict[str, tuple[str, str]]:
343
+ """Stage-0 index entries under the roots: path -> (mode, blob id)."""
344
+ out = _git(repo, ["ls-files", "-z", "--stage", *_pathspec(source_roots)])
345
+ entries: dict[str, tuple[str, str]] = {}
346
+ for record in out.split(b"\0"):
347
+ if not record:
348
+ continue
349
+ meta, _, path = record.decode("utf-8", "surrogateescape").partition("\t")
350
+ mode, oid, stage = meta.split()
351
+ if stage == "0":
352
+ entries[path] = (mode, oid)
353
+ return entries
354
+
355
+
356
+ def _ls_files_staged(repo: Path, source_roots: list[str]) -> dict[str, str]:
357
+ """Stage-0 index entries under the roots: path -> mode."""
358
+ return {path: mode for path, (mode, _) in _ls_files_staged_ids(repo, source_roots).items()}
359
+
360
+
361
+ def _worktree_blob_ids(repo: Path, paths: list[str], source_roots: list[str]) -> dict[str, str]:
362
+ """The git blob id each of ``paths`` would have if added now. A file git
363
+ reports unmodified has its staged id; a modified or untracked one is
364
+ hashed by ``git hash-object``, which applies the same filters (line
365
+ endings, clean filters) and object format as ``git add`` would, so equal
366
+ content gives the id a commit has. A symbolic link is the id of its
367
+ target path, as git stores it."""
368
+ staged = _ls_files_staged_ids(repo, source_roots)
369
+ listed = _git(
370
+ repo, ["ls-files", "-z", "-m", "--others", "--exclude-standard", *_pathspec(source_roots)]
371
+ )
372
+ dirty = {p.decode("utf-8", "surrogateescape") for p in listed.split(b"\0") if p}
373
+ ids: dict[str, str] = {}
374
+ to_hash: list[str] = []
375
+ for path in paths:
376
+ full = repo / path
377
+ if full.is_symlink():
378
+ target = os.readlink(full).encode("utf-8", "surrogateescape")
379
+ ids[path] = _git(repo, ["hash-object", "--stdin"], stdin=target).decode().strip()
380
+ elif path in staged and path not in dirty:
381
+ ids[path] = staged[path][1]
382
+ elif "\n" in path: # ``--stdin-paths`` is line-based
383
+ args = ["hash-object", "--stdin", f"--path={path}"]
384
+ ids[path] = _git(repo, args, stdin=full.read_bytes()).decode().strip()
385
+ else:
386
+ to_hash.append(path)
387
+ if to_hash:
388
+ request = "\n".join(to_hash).encode("utf-8", "surrogateescape") + b"\n"
389
+ out = _git(repo, ["hash-object", "--stdin-paths"], stdin=request).decode().split()
390
+ ids.update(zip(to_hash, out, strict=True))
391
+ return ids
392
+
393
+
394
+ def _nested_configs(listing: bytes) -> list[str]:
395
+ """Paths of ``asv.conf.json``, and of the package metadata a sibling
396
+ package declares its pytest plugins in (``pyproject.toml``,
397
+ ``setup.cfg``), below the root in a newline-separated file listing,
398
+ shallowest first and no deeper than ASV_CONFIG_DEPTH."""
399
+ found = []
400
+ for raw in listing.split(b"\n"):
401
+ path = raw.decode("utf-8", "surrogateescape").strip()
402
+ parts = path.split("/")
403
+ if len(parts) > 1 and parts[-1] in NESTED_CONFIGS and len(parts) <= ASV_CONFIG_DEPTH:
404
+ found.append(path)
405
+ return sorted(found, key=lambda p: (p.count("/"), p))
406
+
407
+
408
+ def _python_paths(listing: bytes) -> tuple[str, ...]:
409
+ """The ``.py`` paths in a newline-separated file listing, sorted."""
410
+ paths = (raw.decode("utf-8", "surrogateescape").strip() for raw in listing.split(b"\n"))
411
+ return tuple(sorted(p for p in paths if p.endswith(".py")))
412
+
413
+
414
+ def _staged_config_files(repo: Path) -> dict[str, bytes]:
415
+ out = _git(repo, ["ls-files", "-z", "--cached", "--", *CONFIG_FILES])
416
+ names = [p.decode("utf-8", "surrogateescape") for p in out.split(b"\0") if p]
417
+ return read_files(repo, "", names, label=INDEX)
418
+
419
+
420
+ def commit_description(commit: str, revision: str) -> str:
421
+ """How a report describes a committed snapshot."""
422
+ return f"commit {commit[:12]} ({revision})"
423
+
424
+
425
+ def read_commit_snapshot(
426
+ repo: Path, revision: str, source_roots: list[str], *, with_config: bool = False
427
+ ) -> Snapshot:
428
+ commit = resolve_commit(repo, revision)
429
+ entries_ids = _ls_tree_ids(repo, commit, _root_pathspecs(source_roots))
430
+ entries = [(m, p) for m, _, p in entries_ids]
431
+ paths = sorted(p for m, p in entries if p.endswith(".py") and m != SYMLINK_MODE)
432
+ link_paths = [p for m, p in entries if m == SYMLINK_MODE]
433
+ targets = read_files(repo, commit, link_paths)
434
+
435
+ tree: list[str] | None = None
436
+
437
+ def files_under(real: str) -> list[str]:
438
+ nonlocal tree
439
+ if tree is None: # one listing of the whole tree, only when there are links
440
+ tree = [p for m, p in _ls_tree(repo, commit, []) if m != SYMLINK_MODE]
441
+ return [p for p in tree if p == real or p.startswith(real + "/")]
442
+
443
+ aliases = expand_symlinks(
444
+ {k: v.decode("utf-8", "surrogateescape") for k, v in targets.items()}, files_under
445
+ )
446
+ listed = set(paths)
447
+ aliases = {a: r for a, r in aliases.items() if a not in listed}
448
+ files = read_files(repo, commit, paths)
449
+ real_files = read_files(repo, commit, sorted(set(aliases.values())))
450
+ files.update({alias: real_files[real] for alias, real in aliases.items()})
451
+ config_files: dict[str, bytes] = {}
452
+ python_paths: tuple[str, ...] = ()
453
+ if with_config:
454
+ root = list_root_files(repo, commit)
455
+ names = [n for n in CONFIG_FILES if n in root]
456
+ whole_tree = _git(repo, ["ls-tree", "-r", "--name-only", commit])
457
+ nested = _nested_configs(whole_tree)
458
+ config_files = read_files(repo, commit, names + nested)
459
+ python_paths = _python_paths(whole_tree)
460
+ return Snapshot(
461
+ info=SnapshotInfo(
462
+ revision=revision,
463
+ commit=commit,
464
+ kind=KIND_COMMIT,
465
+ description=commit_description(commit, revision),
466
+ ),
467
+ source_roots=list(source_roots),
468
+ files=dict(sorted(files.items())),
469
+ config_files=config_files,
470
+ python_paths=python_paths,
471
+ other_files={
472
+ p: oid for _, oid, p in sorted(entries_ids, key=lambda e: e[2]) if not p.endswith(".py")
473
+ },
474
+ cython_files=read_files(
475
+ repo, commit, sorted(p for m, p in entries if is_cython(p) and m != SYMLINK_MODE)
476
+ ),
477
+ text_files=read_files(
478
+ repo, commit, _text_paths([p for m, p in entries if m != SYMLINK_MODE])
479
+ )
480
+ if with_config
481
+ else {},
482
+ )
483
+
484
+
485
+ def read_index_snapshot(
486
+ repo: Path, source_roots: list[str], *, with_config: bool = False
487
+ ) -> Snapshot:
488
+ """The staged content of tracked files (git's index).
489
+
490
+ Unmerged paths (a merge in progress) have no stage-0 blob; they are
491
+ recorded as analysis errors so the plan degrades instead of failing.
492
+ """
493
+ head = resolve_commit(repo, "HEAD")
494
+ errors: list[AnalysisError] = []
495
+ paths: list[str] = []
496
+ for tag, path in _ls_files_tagged(repo, ["--cached"], source_roots):
497
+ if not path.endswith(".py"):
498
+ continue
499
+ if tag == TAG_UNMERGED:
500
+ if path not in paths and not any(e.path == path for e in errors):
501
+ errors.append(
502
+ AnalysisError(
503
+ INDEX, path, "unmerged in the index (merge in progress); no staged content"
504
+ )
505
+ )
506
+ elif path not in paths:
507
+ paths.append(path)
508
+ paths = [p for p in paths if not any(e.path == p for e in errors)]
509
+ staged = _ls_files_staged(repo, source_roots)
510
+ link_paths = [p for p, m in staged.items() if m == SYMLINK_MODE]
511
+ paths = [p for p in paths if staged.get(p) != SYMLINK_MODE]
512
+ targets = read_files(repo, "", link_paths, label=INDEX)
513
+
514
+ staged_all: list[str] | None = None
515
+
516
+ def files_under(real: str) -> list[str]:
517
+ nonlocal staged_all
518
+ if staged_all is None:
519
+ staged_all = [p for p, m in _ls_files_staged(repo, []).items() if m != SYMLINK_MODE]
520
+ return [p for p in staged_all if p == real or p.startswith(real + "/")]
521
+
522
+ aliases = expand_symlinks(
523
+ {k: v.decode("utf-8", "surrogateescape") for k, v in targets.items()}, files_under
524
+ )
525
+ listed = set(paths)
526
+ aliases = {a: r for a, r in aliases.items() if a not in listed}
527
+ files = read_files(repo, "", paths, label=INDEX)
528
+ real_files = read_files(repo, "", sorted(set(aliases.values())), label=INDEX)
529
+ files.update({alias: real_files[real] for alias, real in aliases.items()})
530
+ config_files = _staged_config_files(repo) if with_config else {}
531
+ python_paths: tuple[str, ...] = ()
532
+ if with_config:
533
+ listing = _git(repo, ["ls-files", "-z", "--cached"]).replace(b"\0", b"\n")
534
+ nested = _nested_configs(listing)
535
+ config_files.update(read_files(repo, "", nested, label=INDEX))
536
+ python_paths = _python_paths(listing)
537
+ return Snapshot(
538
+ info=SnapshotInfo(
539
+ revision=INDEX,
540
+ commit=head,
541
+ kind=KIND_INDEX,
542
+ description=f"git index (staged content) on top of commit {head[:12]}; uncommitted",
543
+ ),
544
+ source_roots=list(source_roots),
545
+ files=dict(sorted(files.items())),
546
+ config_files=config_files,
547
+ python_paths=python_paths,
548
+ errors=errors,
549
+ other_files={
550
+ p: oid
551
+ for p, (_, oid) in sorted(_ls_files_staged_ids(repo, source_roots).items())
552
+ if not p.endswith(".py")
553
+ },
554
+ cython_files=read_files(
555
+ repo,
556
+ "",
557
+ sorted(p for p, m in staged.items() if is_cython(p) and m != SYMLINK_MODE),
558
+ label=INDEX,
559
+ ),
560
+ text_files=read_files(
561
+ repo, "", _text_paths([p for p, m in staged.items() if m != SYMLINK_MODE]), label=INDEX
562
+ )
563
+ if with_config
564
+ else {},
565
+ )
566
+
567
+
568
+ def read_worktree_snapshot(
569
+ repo: Path, source_roots: list[str], *, with_config: bool = False
570
+ ) -> Snapshot:
571
+ """Files on disk: tracked and untracked, minus ignored ones.
572
+
573
+ Skip-worktree entries (sparse checkouts) are not on disk by design and
574
+ are read from the index instead of being treated as deletions.
575
+ """
576
+ head = resolve_commit(repo, "HEAD")
577
+ listed = _ls_files_tagged(repo, ["--cached", "--others", "--exclude-standard"], source_roots)
578
+ files: dict[str, bytes] = {}
579
+ from_index: list[str] = []
580
+ for tag, path in listed:
581
+ if not path.endswith(".py") or path in files or path in from_index:
582
+ continue
583
+ full = repo / path
584
+ if full.is_file():
585
+ files[path] = full.read_bytes()
586
+ elif tag == TAG_SKIP_WORKTREE:
587
+ from_index.append(path)
588
+ # otherwise: a tracked file deleted on disk is absent from the snapshot
589
+ files.update(read_files(repo, "", from_index, label=INDEX))
590
+ # Symlinked directories (a file link was read through above).
591
+ links = {
592
+ path: os.readlink(repo / path)
593
+ for _, path in listed
594
+ if (repo / path).is_symlink() and (repo / path).is_dir()
595
+ }
596
+
597
+ def files_under(real: str) -> list[str]:
598
+ found = _ls_files_tagged(repo, ["--cached", "--others", "--exclude-standard"], [real])
599
+ return sorted({p for _, p in found if (repo / p).is_file() and not (repo / p).is_symlink()})
600
+
601
+ for alias, real in expand_symlinks(links, files_under).items():
602
+ if alias not in files:
603
+ files[alias] = (repo / real).read_bytes()
604
+ files = dict(sorted(files.items()))
605
+ config_files: dict[str, bytes] = {}
606
+ python_paths: tuple[str, ...] = ()
607
+ if with_config:
608
+ for name in CONFIG_FILES:
609
+ full = repo / name
610
+ if full.is_file():
611
+ config_files[name] = full.read_bytes()
612
+ listing = _git(repo, ["ls-files", "-z", "--cached", "--others", "--exclude-standard"])
613
+ for name in _nested_configs(listing.replace(b"\0", b"\n")):
614
+ full = repo / name
615
+ if full.is_file():
616
+ config_files[name] = full.read_bytes()
617
+ python_paths = tuple(
618
+ p for p in _python_paths(listing.replace(b"\0", b"\n")) if (repo / p).is_file()
619
+ )
620
+ return Snapshot(
621
+ info=SnapshotInfo(
622
+ revision=WORKTREE,
623
+ commit=head,
624
+ kind=KIND_WORKTREE,
625
+ description=(
626
+ "working tree (tracked and untracked files, ignored files excluded) on top of "
627
+ f"commit {head[:12]}; uncommitted"
628
+ ),
629
+ ),
630
+ source_roots=list(source_roots),
631
+ files=files,
632
+ config_files=config_files,
633
+ python_paths=python_paths,
634
+ other_files=_worktree_blob_ids(
635
+ repo,
636
+ sorted(
637
+ {
638
+ p
639
+ for tag, p in listed
640
+ if not p.endswith(".py")
641
+ and not (tag == TAG_OTHER and is_bytecode(p))
642
+ and (
643
+ (repo / p).is_file()
644
+ or (repo / p).is_symlink()
645
+ or tag == TAG_SKIP_WORKTREE # not on disk by design: the staged id
646
+ )
647
+ }
648
+ ),
649
+ source_roots,
650
+ ),
651
+ cython_files={
652
+ p: (repo / p).read_bytes()
653
+ for p in sorted({p for _, p in listed if is_cython(p)})
654
+ if (repo / p).is_file() and not (repo / p).is_symlink()
655
+ },
656
+ text_files={
657
+ p: (repo / p).read_bytes()
658
+ for p in _text_paths(sorted({p for _, p in listed}))
659
+ if (repo / p).is_file()
660
+ }
661
+ if with_config
662
+ else {},
663
+ )
664
+
665
+
666
+ def read_snapshot(
667
+ repo: Path, revision: str, source_roots: list[str], *, with_config: bool = False
668
+ ) -> Snapshot:
669
+ """Read the Python sources at ``revision``: a commit, ``INDEX`` or ``WORKTREE``.
670
+
671
+ ``with_config`` also reads the root-level runner configuration files;
672
+ only discovery needs them, so the base snapshot skips the extra work.
673
+ """
674
+ if revision == WORKTREE:
675
+ return read_worktree_snapshot(repo, source_roots, with_config=with_config)
676
+ if revision == INDEX:
677
+ return read_index_snapshot(repo, source_roots, with_config=with_config)
678
+ return read_commit_snapshot(repo, revision, source_roots, with_config=with_config)
679
+
680
+
681
+ def module_name_for(path: str, source_roots: list[str]) -> str | None:
682
+ """Map a repo-relative ``.py`` path to a dotted module name.
683
+
684
+ The longest matching source root wins. Returns ``None`` when the path lies
685
+ outside every root.
686
+ """
687
+ best: tuple[str, str] | None = None
688
+ best_len = -1
689
+ for raw in source_roots:
690
+ root, prefix = split_root(raw)
691
+ if root == "":
692
+ rel = path
693
+ elif path.startswith(root + "/"):
694
+ rel = path[len(root) + 1 :]
695
+ else:
696
+ continue
697
+ if len(root) > best_len:
698
+ best, best_len = (rel, prefix), len(root)
699
+ if best is None:
700
+ return None
701
+ rel, prefix = best
702
+ rel = rel[: -len(".py")]
703
+ parts = rel.split("/")
704
+ if parts[-1] == "__init__":
705
+ parts = parts[:-1]
706
+ if (not parts and not prefix) or not all(p.isidentifier() for p in parts):
707
+ return None
708
+ return ".".join([*prefix.split("."), *parts] if prefix else parts)
709
+
710
+
711
+ def child_modules(snapshot: Snapshot) -> dict[str, frozenset[str]]:
712
+ """Immediate submodule names per module, from the files present
713
+ (whether or not they parse): what a package binding can shadow.
714
+ Computed once per snapshot: discovery parses modules one at a time and
715
+ asks for every one (about 100 times a plan on pandas)."""
716
+ cached = snapshot.__dict__.get("_child_modules")
717
+ if cached is not None:
718
+ return cached
719
+ children: dict[str, set[str]] = {}
720
+ for path in snapshot.files:
721
+ module = module_name_for(path, snapshot.source_roots) if path.endswith(".py") else None
722
+ if module is None:
723
+ continue
724
+ parent, _, child = module.rpartition(".")
725
+ if parent:
726
+ children.setdefault(parent, set()).add(child)
727
+ result = {k: frozenset(v) for k, v in children.items()}
728
+ snapshot.__dict__["_child_modules"] = result
729
+ return result
730
+
731
+
732
+ def member_symbol_id(module: str, name: str, submodules: frozenset[str] | set[str]) -> str:
733
+ """Identity of the top-level binding ``name`` of ``module``. When the
734
+ module is a package with a submodule of that name (``pkg/__init__.py``
735
+ defining ``retry`` next to ``pkg/retry.py``) the module keeps
736
+ ``pkg.retry`` and the binding is ``pkg.__init__.retry``."""
737
+ if name in submodules:
738
+ return f"{module}.__init__.{name}"
739
+ return f"{module}.{name}"