diffcone 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
diffcone/__init__.py ADDED
@@ -0,0 +1,13 @@
1
+ """Diffcone: static-first, function-level change-impact engine for Python."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ from diffcone.manifest import Manifest, Target, load_manifest
6
+ from diffcone.planner import Plan, plan
7
+
8
+ try:
9
+ __version__ = version("diffcone")
10
+ except PackageNotFoundError: # pragma: no cover - running from a source tree
11
+ __version__ = "0+unknown"
12
+
13
+ __all__ = ["Manifest", "Plan", "Target", "__version__", "load_manifest", "plan"]
diffcone/cache.py ADDED
@@ -0,0 +1,528 @@
1
+ """Per-commit index cache.
2
+
3
+ A committed snapshot's :class:`SourceIndex` is a pure function of the commit,
4
+ the source roots and the indexer version, so it can be stored and reused.
5
+ The working tree and the git index are never cached whole (their content is
6
+ not identified by a commit). A cache hit must produce a byte-identical plan
7
+ to a cache miss; the cache is an optimisation only.
8
+
9
+ Layout: ``<cache_dir>/index/<key>.json`` where ``cache_dir`` defaults to
10
+ ``<repo>/.diffcone/cache`` (add ``.diffcone/`` to ``.gitignore``).
11
+
12
+ Beside it, the :class:`ModuleCache` keeps per-module results for every
13
+ snapshot kind, including the working tree: a module's first-pass facts are a
14
+ pure function of its file, and its second-pass resolution a pure function of
15
+ the file plus a fingerprint of what other modules expose (see
16
+ ``Indexer.build``). A warm working-tree plan after a one-line edit then
17
+ re-parses and re-resolves only the edited module.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import hashlib
23
+ import json
24
+ import os
25
+ import re
26
+ import shutil
27
+ import sqlite3
28
+ import sys
29
+ import tempfile
30
+ from dataclasses import asdict, dataclass
31
+ from pathlib import Path
32
+ from typing import Any
33
+
34
+ from diffcone.cython import CythonFunction, CythonModule, CythonStatement
35
+ from diffcone.discovery import DiscoveryNote, DiscoveryOptions, DiscoveryResult
36
+ from diffcone.manifest import Target
37
+ from diffcone.model import (
38
+ AnalysisError,
39
+ Edge,
40
+ ExternalReference,
41
+ SnapshotInfo,
42
+ SourceIndex,
43
+ Symbol,
44
+ UnresolvedReference,
45
+ )
46
+
47
+ # Bump whenever the indexer's output for the same input can change.
48
+ INDEX_FORMAT = 24 # 24: docstring decorators; 23: open classes; 22: import layout
49
+
50
+
51
+ def _indexer_fingerprint() -> str:
52
+ """Hash of the modules that determine an index's content, so any change
53
+ to them invalidates cached indexes without anyone remembering to bump
54
+ INDEX_FORMAT (a stale base index once produced phantom changed symbols)."""
55
+ here = Path(__file__).parent
56
+ h = hashlib.sha256()
57
+ # ``ast.dump`` output (hence every hash) may differ between Python versions.
58
+ h.update(f"python{sys.version_info[0]}.{sys.version_info[1]}:".encode())
59
+ sources = [here / name for name in ("model.py", "snapshot.py", "cython.py")]
60
+ sources += sorted((here / "indexer").glob("*.py"))
61
+ for path in sources:
62
+ h.update(path.relative_to(here).as_posix().encode() + b"\0")
63
+ h.update(path.read_bytes())
64
+ return h.hexdigest()[:16]
65
+
66
+
67
+ INDEXER_FINGERPRINT = _indexer_fingerprint()
68
+
69
+
70
+ def make_own_dir(directory: Path) -> None:
71
+ """Create a directory of diffcone's, and give the ``.diffcone`` it sits in
72
+ a ``.gitignore`` of its own (as pytest does for ``.pytest_cache``), so the
73
+ cache and recordings never show up as untracked files of the project."""
74
+ directory.mkdir(parents=True, exist_ok=True)
75
+ for parent in (directory, *directory.parents):
76
+ if parent.name == ".diffcone":
77
+ ignore = parent / ".gitignore"
78
+ if not ignore.exists():
79
+ try:
80
+ ignore.write_text("# created by diffcone\n*\n", "utf-8")
81
+ except OSError:
82
+ pass
83
+ break
84
+
85
+
86
+ def default_cache_dir(repo: Path) -> Path:
87
+ return repo / ".diffcone" / "cache"
88
+
89
+
90
+ def index_key(commit: str, source_roots: list[str]) -> str:
91
+ material = json.dumps([INDEX_FORMAT, INDEXER_FINGERPRINT, commit, sorted(source_roots)])
92
+ return hashlib.sha256(material.encode()).hexdigest()
93
+
94
+
95
+ def index_to_dict(index: SourceIndex) -> dict:
96
+ return {
97
+ "format": INDEX_FORMAT,
98
+ "snapshot": asdict(index.snapshot),
99
+ "modules": sorted(index.modules),
100
+ "failed_modules": sorted(index.failed_modules),
101
+ "escaped_classes": sorted(index.escaped_classes),
102
+ "other_files": dict(sorted(index.other_files.items())),
103
+ "symbols": [asdict(s) for _, s in sorted(index.symbols.items())],
104
+ "edges": [asdict(e) for e in sorted(index.edges)],
105
+ "unresolved": [asdict(u) for u in sorted(index.unresolved)],
106
+ "external": [asdict(x) for x in sorted(index.external)],
107
+ "errors": [asdict(e) for e in sorted(index.errors)],
108
+ "reflection": sorted(list(r) for r in index.reflection),
109
+ "class_attributes": {
110
+ c: dict(sorted(a.items())) for c, a in sorted(index.class_attributes.items())
111
+ },
112
+ "class_bases": {c: list(b) for c, b in sorted(index.class_bases.items())},
113
+ "open_classes": sorted(index.open_classes),
114
+ "doc_decorated": sorted(index.doc_decorated),
115
+ "cython": {
116
+ path: {
117
+ "functions": [{**asdict(f), "names": sorted(f.names)} for f in module.functions],
118
+ "outside_hash": module.outside_hash,
119
+ "statements": None
120
+ if module.statements is None
121
+ else [asdict(st) for st in module.statements],
122
+ }
123
+ for path, module in sorted(index.cython.items())
124
+ },
125
+ }
126
+
127
+
128
+ def index_from_dict(data: dict) -> SourceIndex:
129
+ if data.get("format") != INDEX_FORMAT:
130
+ raise ValueError("unsupported index format")
131
+ symbols = {}
132
+ for s in data["symbols"]:
133
+ s = dict(s)
134
+ s["line_ranges"] = tuple(tuple(r) for r in s["line_ranges"])
135
+ s["imports"] = tuple(s["imports"])
136
+ s["import_layout"] = tuple(s["import_layout"])
137
+ symbol = Symbol(**s)
138
+ symbols[symbol.id] = symbol
139
+ return SourceIndex(
140
+ snapshot=SnapshotInfo(**data["snapshot"]),
141
+ modules=set(data["modules"]),
142
+ symbols=symbols,
143
+ edges={Edge(**e) for e in data["edges"]},
144
+ unresolved={UnresolvedReference(**u) for u in data["unresolved"]},
145
+ external={ExternalReference(**x) for x in data["external"]},
146
+ errors=[AnalysisError(**e) for e in data["errors"]],
147
+ failed_modules=set(data["failed_modules"]),
148
+ escaped_classes=set(data["escaped_classes"]),
149
+ other_files=dict(data["other_files"]),
150
+ reflection={(s, d) for s, d in data["reflection"]},
151
+ class_attributes={c: dict(a) for c, a in data["class_attributes"].items()},
152
+ class_bases={c: tuple(b) for c, b in data["class_bases"].items()},
153
+ open_classes=set(data["open_classes"]),
154
+ doc_decorated=set(data["doc_decorated"]),
155
+ cython={
156
+ path: CythonModule(
157
+ path,
158
+ tuple(_cython_function(f) for f in module["functions"]),
159
+ module["outside_hash"],
160
+ None
161
+ if module["statements"] is None
162
+ else tuple(_cython_statement(st) for st in module["statements"]),
163
+ )
164
+ for path, module in data["cython"].items()
165
+ },
166
+ )
167
+
168
+
169
+ def _cython_function(data: dict[str, Any]) -> CythonFunction:
170
+ fields = dict(data)
171
+ fields["names"] = frozenset(data["names"])
172
+ return CythonFunction(**fields)
173
+
174
+
175
+ def _cython_statement(data: dict[str, Any]) -> CythonStatement:
176
+ fields = dict(data)
177
+ fields["names"] = tuple(data["names"])
178
+ fields["bases"] = tuple(data["bases"])
179
+ return CythonStatement(**fields)
180
+
181
+
182
+ class ModuleCache:
183
+ """Per-module records in one SQLite file (``<cache_dir>/modules.sqlite``):
184
+ first-pass facts under ``(key, "")`` and second-pass outputs under
185
+ ``(key, fingerprint)``, where ``key`` identifies the module name, path,
186
+ file content and indexer version, and ``fingerprint`` the environment the
187
+ module was resolved against. Records are plain JSON produced by the
188
+ indexer, which treats a malformed one as a miss; a hit must be
189
+ indistinguishable from a miss. One file rather than one per record
190
+ because a warm plan on a large tree loads thousands of records, and
191
+ opening that many files costs more than resolving.
192
+
193
+ Facts rows are kept for every distinct file content seen. Resolution
194
+ rows are kept only for the latest fingerprint each file was resolved
195
+ against: a store drops the other fingerprints' rows for the files of
196
+ the snapshot being stored, so the table is bounded by the facts rows.
197
+
198
+ Every operation opens its own connection, so one instance may be shared
199
+ by threads (``corpus --jobs``) and by concurrent processes (SQLite locks;
200
+ a writer that cannot get the lock in time gives up silently). Reads open
201
+ the file read-only and work on a directory that cannot be written."""
202
+
203
+ def __init__(self, directory: Path) -> None:
204
+ self.path = directory / "modules.sqlite"
205
+ self.facts_hits = 0
206
+ self.facts_misses = 0
207
+ self.resolved_hits = 0
208
+ self.resolved_misses = 0
209
+
210
+ @staticmethod
211
+ def key(module: str, path: str, content: bytes) -> str:
212
+ h = hashlib.sha256()
213
+ h.update(f"{INDEX_FORMAT}:{INDEXER_FINGERPRINT}:{module}:{path}:".encode())
214
+ h.update(content)
215
+ return h.hexdigest()
216
+
217
+ def _connect(self, write: bool) -> sqlite3.Connection | None:
218
+ try:
219
+ if not write:
220
+ if not self.path.exists():
221
+ return None
222
+ return sqlite3.connect(f"{self.path.as_uri()}?mode=ro", uri=True, timeout=10)
223
+ make_own_dir(self.path.parent)
224
+ conn = sqlite3.connect(self.path, timeout=10)
225
+ conn.execute("PRAGMA journal_mode=WAL")
226
+ conn.execute("PRAGMA synchronous=NORMAL")
227
+ conn.execute(
228
+ "CREATE TABLE IF NOT EXISTS records ("
229
+ "key TEXT NOT NULL, fingerprint TEXT NOT NULL, data TEXT NOT NULL, "
230
+ "PRIMARY KEY (key, fingerprint))"
231
+ )
232
+ return conn
233
+ except (sqlite3.Error, OSError, ValueError):
234
+ return None
235
+
236
+ def _load(self, keys: list[str], fingerprint: str) -> dict[str, dict]:
237
+ found: dict[str, dict] = {}
238
+ if not keys:
239
+ return found
240
+ conn = self._connect(write=False)
241
+ if conn is None:
242
+ return found
243
+ try:
244
+ for i in range(0, len(keys), 500):
245
+ chunk = keys[i : i + 500]
246
+ marks = ",".join("?" * len(chunk))
247
+ rows = conn.execute(
248
+ f"SELECT key, data FROM records WHERE fingerprint = ? AND key IN ({marks})",
249
+ [fingerprint, *chunk],
250
+ ).fetchall()
251
+ for key, data in rows:
252
+ try:
253
+ record = json.loads(data)
254
+ except ValueError:
255
+ continue
256
+ if isinstance(record, dict):
257
+ found[key] = record
258
+ except sqlite3.Error:
259
+ return {}
260
+ finally:
261
+ conn.close()
262
+ return found
263
+
264
+ def load_facts(self, keys: list[str]) -> dict[str, dict]:
265
+ found = self._load(keys, "")
266
+ self.facts_hits += len(found)
267
+ self.facts_misses += len(keys) - len(found)
268
+ return found
269
+
270
+ def load_resolved(self, keys: list[str], fingerprint: str) -> dict[str, dict]:
271
+ found = self._load(keys, fingerprint)
272
+ self.resolved_hits += len(found)
273
+ self.resolved_misses += len(keys) - len(found)
274
+ return found
275
+
276
+ def store(
277
+ self,
278
+ facts: dict[str, dict],
279
+ resolved: dict[str, dict],
280
+ fingerprint: str,
281
+ snapshot_keys: list[str],
282
+ ) -> None:
283
+ """Store new records in one transaction and evict the resolution rows
284
+ of ``snapshot_keys`` (every module of the snapshot) that belong to
285
+ another fingerprint."""
286
+ conn = self._connect(write=True)
287
+ if conn is None:
288
+ return
289
+ try:
290
+ with conn:
291
+ conn.executemany(
292
+ "INSERT OR REPLACE INTO records (key, fingerprint, data) VALUES (?, ?, ?)",
293
+ [(key, "", json.dumps(record)) for key, record in facts.items()]
294
+ + [(key, fingerprint, json.dumps(record)) for key, record in resolved.items()],
295
+ )
296
+ for i in range(0, len(snapshot_keys), 500):
297
+ chunk = snapshot_keys[i : i + 500]
298
+ marks = ",".join("?" * len(chunk))
299
+ conn.execute(
300
+ "DELETE FROM records WHERE fingerprint NOT IN ('', ?) "
301
+ f"AND key IN ({marks})",
302
+ [fingerprint, *chunk],
303
+ )
304
+ except sqlite3.Error:
305
+ pass # a cache write failure is never an error
306
+ finally:
307
+ conn.close()
308
+
309
+
310
+ DISCOVERY_FORMAT = 1
311
+
312
+
313
+ def _discovery_fingerprint() -> str:
314
+ """The indexer's fingerprint (discovery reads the index) and the
315
+ discovery package's source: any change invalidates cached results."""
316
+ here = Path(__file__).parent
317
+ h = hashlib.sha256(INDEXER_FINGERPRINT.encode())
318
+ for path in sorted((here / "discovery").glob("*.py")) + [here / "manifest.py"]:
319
+ h.update(path.name.encode())
320
+ h.update(path.read_bytes())
321
+ return h.hexdigest()[:16]
322
+
323
+
324
+ DISCOVERY_FINGERPRINT = _discovery_fingerprint()
325
+
326
+
327
+ class DiscoveryCache:
328
+ """Static discovery results per committed snapshot. Discovery reads only
329
+ the snapshot and its index, so a commit, source roots, runner and
330
+ options determine the result; ``WORKTREE`` and ``INDEX`` are never
331
+ cached. On pandas discovering both sides was the largest part of a warm
332
+ plan, and a cached head also lets the head index come from the cache."""
333
+
334
+ def __init__(self, directory: Path) -> None:
335
+ self.directory = directory / "discovery"
336
+
337
+ def _path(self, commit: str, roots: list[str], runner: str, options: DiscoveryOptions) -> Path:
338
+ settings = {
339
+ k: sorted(v) if isinstance(v, (set, frozenset)) else v
340
+ for k, v in sorted(asdict(options).items())
341
+ }
342
+ material = json.dumps(
343
+ [DISCOVERY_FORMAT, DISCOVERY_FINGERPRINT, commit, sorted(roots), runner, settings]
344
+ )
345
+ return self.directory / f"{hashlib.sha256(material.encode()).hexdigest()}.json"
346
+
347
+ def load(
348
+ self, commit: str, roots: list[str], runner: str, options: DiscoveryOptions
349
+ ) -> DiscoveryResult | None:
350
+ try:
351
+ data = json.loads(self._path(commit, roots, runner, options).read_text("utf-8"))
352
+ if data.get("format") != DISCOVERY_FORMAT or data.get("commit") != commit:
353
+ return None
354
+ return DiscoveryResult(
355
+ runner=data["runner"],
356
+ targets=[
357
+ Target(
358
+ t["runner"],
359
+ t["runner_id"],
360
+ t["entry_symbol"],
361
+ tuple(t["lifecycle_dependencies"]),
362
+ )
363
+ for t in data["targets"]
364
+ ],
365
+ notes=[DiscoveryNote(**n) for n in data["notes"]],
366
+ config=data["config"],
367
+ )
368
+ except (OSError, ValueError, KeyError, TypeError):
369
+ return None
370
+
371
+ def store(
372
+ self,
373
+ result: DiscoveryResult,
374
+ commit: str,
375
+ roots: list[str],
376
+ options: DiscoveryOptions,
377
+ ) -> None:
378
+ path = self._path(commit, roots, result.runner, options)
379
+ data = {
380
+ "format": DISCOVERY_FORMAT,
381
+ "commit": commit,
382
+ "fingerprint": DISCOVERY_FINGERPRINT,
383
+ "runner": result.runner,
384
+ "targets": [asdict(t) for t in result.targets],
385
+ "notes": [asdict(n) for n in result.notes],
386
+ "config": result.config,
387
+ }
388
+ try:
389
+ make_own_dir(path.parent)
390
+ fd, tmp = tempfile.mkstemp(dir=path.parent, suffix=".tmp")
391
+ with os.fdopen(fd, "w", encoding="utf-8") as f:
392
+ json.dump(data, f)
393
+ os.replace(tmp, path)
394
+ except (OSError, TypeError, ValueError):
395
+ pass # a cache write failure is never an error
396
+
397
+
398
+ class IndexCache:
399
+ def __init__(self, directory: Path) -> None:
400
+ self.directory = directory
401
+ self.hits = 0
402
+ self.misses = 0
403
+ self.modules = ModuleCache(directory)
404
+ self.discovery = DiscoveryCache(directory)
405
+ # The per-file hash cache this replaced left ``hashes/`` behind.
406
+ shutil.rmtree(directory / "hashes", ignore_errors=True)
407
+
408
+ def _path(self, commit: str, source_roots: list[str]) -> Path:
409
+ return self.directory / "index" / f"{index_key(commit, source_roots)}.json"
410
+
411
+ def load(self, commit: str, source_roots: list[str]) -> SourceIndex | None:
412
+ path = self._path(commit, source_roots)
413
+ try:
414
+ data = json.loads(path.read_text("utf-8"))
415
+ index = index_from_dict(data)
416
+ except (OSError, ValueError, KeyError, TypeError):
417
+ self.misses += 1
418
+ return None
419
+ if index.snapshot.commit != commit:
420
+ self.misses += 1
421
+ return None
422
+ self.hits += 1
423
+ return index
424
+
425
+ def store(self, index: SourceIndex, source_roots: list[str]) -> None:
426
+ path = self._path(index.snapshot.commit, source_roots)
427
+ try:
428
+ make_own_dir(path.parent)
429
+ # Write atomically so a concurrent reader never sees a partial file.
430
+ fd, tmp = tempfile.mkstemp(dir=path.parent, suffix=".tmp")
431
+ with os.fdopen(fd, "w", encoding="utf-8") as f:
432
+ json.dump(index_to_dict(index), f)
433
+ os.replace(tmp, path)
434
+ except OSError:
435
+ pass # a cache write failure is never an error
436
+
437
+
438
+ # --------------------------------------------------------------------------- pruning
439
+
440
+ _COMMIT = re.compile(rb'"commit": "([0-9a-f]{40})"')
441
+ _FINGERPRINT = re.compile(rb'"fingerprint": "([0-9a-f]+)"')
442
+
443
+
444
+ def _head(path: Path) -> bytes:
445
+ try:
446
+ with path.open("rb") as f:
447
+ return f.read(4096)
448
+ except OSError:
449
+ return b""
450
+
451
+
452
+ def _recorded_commit(path: Path) -> str | None:
453
+ """The commit an index or discovery file was stored for: the first
454
+ ``"commit"`` key, which both write near the start."""
455
+ match = _COMMIT.search(_head(path))
456
+ return match.group(1).decode() if match else None
457
+
458
+
459
+ @dataclass
460
+ class PruneResult:
461
+ files_removed: int = 0
462
+ files_kept: int = 0
463
+ rows_removed: int = 0
464
+ rows_kept: int = 0
465
+ bytes_before: int = 0
466
+ bytes_after: int = 0
467
+
468
+
469
+ def _size(directory: Path) -> int:
470
+ return sum(p.stat().st_size for p in directory.rglob("*") if p.is_file())
471
+
472
+
473
+ def prune(
474
+ directory: Path, commits: set[str], source_roots: list[str], module_keys: set[str]
475
+ ) -> PruneResult:
476
+ """Keep only what planning at ``commits`` with ``source_roots`` reads:
477
+ their whole indexes and discovery results as this version of diffcone
478
+ names them, and the per-module rows for their files' ``module_keys``
479
+ (``ModuleCache.key``), which also serve a later commit or a working tree
480
+ sharing those files. Everything else goes: other commits, entries an
481
+ older diffcone wrote, partial writes. The cache stays an optimisation: a
482
+ pruned entry is a miss, never a different plan."""
483
+ result = PruneResult(bytes_before=_size(directory) if directory.exists() else 0)
484
+ indexes = {f"{index_key(c, source_roots)}.json" for c in commits}
485
+
486
+ def wanted(sub: str, path: Path) -> bool:
487
+ if sub == "index":
488
+ return path.name in indexes
489
+ found = _FINGERPRINT.search(_head(path))
490
+ return (
491
+ path.suffix == ".json"
492
+ and _recorded_commit(path) in commits
493
+ and found is not None
494
+ and found.group(1).decode() == DISCOVERY_FINGERPRINT
495
+ )
496
+
497
+ for sub in ("index", "discovery"):
498
+ for path in sorted((directory / sub).glob("*")):
499
+ if wanted(sub, path):
500
+ result.files_kept += 1
501
+ continue
502
+ try:
503
+ path.unlink()
504
+ result.files_removed += 1
505
+ except OSError:
506
+ pass
507
+ modules = directory / "modules.sqlite"
508
+ if modules.exists():
509
+ try:
510
+ conn = sqlite3.connect(modules, timeout=30)
511
+ try:
512
+ with conn:
513
+ conn.execute("CREATE TEMP TABLE keep (key TEXT PRIMARY KEY)")
514
+ conn.executemany(
515
+ "INSERT OR IGNORE INTO keep VALUES (?)", [(k,) for k in module_keys]
516
+ )
517
+ result.rows_removed = conn.execute(
518
+ "DELETE FROM records WHERE key NOT IN (SELECT key FROM keep)"
519
+ ).rowcount
520
+ result.rows_kept = conn.execute("SELECT COUNT(*) FROM records").fetchone()[0]
521
+ conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
522
+ conn.execute("VACUUM")
523
+ finally:
524
+ conn.close()
525
+ except sqlite3.Error:
526
+ pass # a cache that cannot be pruned is still a valid cache
527
+ result.bytes_after = _size(directory) if directory.exists() else 0
528
+ return result