crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/gitio.py ADDED
@@ -0,0 +1,504 @@
1
+ """Git shell layer: the tracked-file universe and the current commit."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ import shutil
6
+ import subprocess
7
+ import threading
8
+ from collections.abc import Iterator
9
+ from contextlib import contextmanager
10
+ from pathlib import Path
11
+
12
+ from .errors import GitError
13
+
14
+ _OBJECT_NAME = re.compile(r"[0-9a-f]{40}|[0-9a-f]{64}")
15
+
16
+
17
+ def _git(root: Path, *args: str) -> str:
18
+ try:
19
+ res = subprocess.run(["git", *args], cwd=root, capture_output=True, text=True, encoding="utf-8")
20
+ except FileNotFoundError as exc:
21
+ raise GitError("git executable not found") from exc
22
+ if res.returncode != 0:
23
+ raise GitError(f"git {' '.join(args)} failed in {root}: {res.stderr.strip()}")
24
+ return res.stdout
25
+
26
+
27
+ def _git_lines(root: Path, *args: str) -> Iterator[str]:
28
+ """Same contract as _git, streamed: the caller sees one line at a time.
29
+
30
+ A failing command yields nothing and raises at the end of iteration, so the
31
+ consumer never mistakes an empty stream for an empty history.
32
+ """
33
+ try:
34
+ proc = subprocess.Popen(["git", *args], cwd=root, stdout=subprocess.PIPE,
35
+ stderr=subprocess.PIPE, text=True, encoding="utf-8")
36
+ except FileNotFoundError as exc:
37
+ raise GitError("git executable not found") from exc
38
+ with proc:
39
+ yield from proc.stdout
40
+ stderr = proc.stderr.read()
41
+ if proc.returncode != 0:
42
+ raise GitError(f"git {' '.join(args)} failed in {root}: {stderr.strip()}")
43
+
44
+
45
+ def stage_path(root: Path, rel_path: str) -> None:
46
+ """git add one path — the hook override's ratchet debt must land IN the commit."""
47
+ _git(root, "add", "--", rel_path)
48
+
49
+
50
+ def ls_files(root: Path) -> list[str]:
51
+ out = _git(root, "ls-files", "-z")
52
+ return [p for p in out.split("\0") if p]
53
+
54
+
55
+ def untracked_files(root: Path) -> list[str]:
56
+ """Paths `git add` would pick up: untracked and not ignored.
57
+
58
+ git applies the ignore rules, so a build directory never reads as source
59
+ somebody forgot to add.
60
+ """
61
+ out = _git(root, "ls-files", "--others", "--exclude-standard", "-z")
62
+ return [p for p in out.split("\0") if p]
63
+
64
+
65
+ def config_value(root: Path, key: str) -> str:
66
+ """One `git config` value, or "" when it is unset.
67
+
68
+ `git config --get` exits 1 on an unset key, which is an answer rather than a
69
+ failure — every caller here asks about a setting the repo need not have.
70
+ """
71
+ try:
72
+ return _git(root, "config", "--get", key).strip()
73
+ except GitError:
74
+ return ""
75
+
76
+
77
+ def index_modes(root: Path, pathspec: str) -> dict[str, str]:
78
+ """path -> index mode for everything git tracks under `pathspec`.
79
+
80
+ `git ls-files -s` is the only place the executable bit is readable on
81
+ Windows, where the filesystem has no such bit and the working copy always
82
+ looks 0644.
83
+ """
84
+ out = _git(root, "ls-files", "-s", "-z", "--", pathspec)
85
+ modes = {}
86
+ for record in out.split("\0"):
87
+ if record:
88
+ meta, _, path = record.partition("\t")
89
+ modes[path.replace("\\", "/")] = meta.split(" ", 1)[0]
90
+ return modes
91
+
92
+
93
+ def staged_diff(root: Path) -> str:
94
+ # --no-renames: a renamed file becomes delete+add, so a rename stays a
95
+ # touched file and its functions still face the gate (and the old ratchet
96
+ # entry's drop is matched by fresh gating at the new path).
97
+ return _git(root, "diff", "--cached", "-U0", "--no-renames")
98
+
99
+
100
+ def unstaged_paths(root: Path) -> set[str]:
101
+ """Tracked files whose working-tree content differs from the index.
102
+
103
+ git decides it, through its own filters. Comparing a staged blob to the
104
+ file's raw bytes reads every file as different under `core.autocrlf=true` —
105
+ git-for-windows' installer default — because the blob holds LF and the
106
+ checkout holds CRLF by design.
107
+ """
108
+ out = _git(root, "diff", "--name-only", "--no-renames")
109
+ return {line.strip().replace("\\", "/") for line in out.splitlines() if line.strip()}
110
+
111
+
112
+ def diff_since(root: Path, commit: str) -> str:
113
+ return _git(root, "diff", commit, "-U0", "--no-renames")
114
+
115
+
116
+ def diff_names_since(root: Path, commit: str) -> list[str]:
117
+ """Files with committed changes between a commit and HEAD."""
118
+ out = _git(root, "diff", "--name-only", "--no-renames", commit, "HEAD")
119
+ return [line.strip().replace("\\", "/") for line in out.splitlines() if line.strip()]
120
+
121
+
122
+ def _rename_pairs(fields: list[str]) -> dict[str, str]:
123
+ """Walk `--name-status -z` records: a status field, then one path — two for R and C.
124
+
125
+ Consuming one path per record would read a rename's destination as the next
126
+ record's status and shift every entry after it.
127
+ """
128
+ pairs: dict[str, str] = {}
129
+ i = 0
130
+ while i < len(fields) and fields[i]:
131
+ status = fields[i]
132
+ paths = 2 if status[0] in ("R", "C") else 1
133
+ if status[0] == "R":
134
+ pairs[fields[i + 1].replace("\\", "/")] = fields[i + 2].replace("\\", "/")
135
+ i += 1 + paths
136
+ return pairs
137
+
138
+
139
+ def renamed_paths(root: Path, since: str, *, similarity: int = 50) -> dict[str, str]:
140
+ """old path -> new path for files git reads as renamed between `since` and HEAD.
141
+
142
+ Tree-to-tree, not a walk of history: a rename here is content similarity
143
+ between the two endpoints, so widening the window costs nothing extra and
144
+ cannot invent a pairing git does not already see. Copies are excluded — the
145
+ source still exists, so nothing about it moved.
146
+ """
147
+ out = _git(root, "diff", "--name-status", f"-M{similarity}", "-z", since, "HEAD")
148
+ return _rename_pairs(out.split("\0"))
149
+
150
+
151
+ def status_names(root: Path) -> list[str]:
152
+ """Files with uncommitted (staged or unstaged) changes in the working tree."""
153
+ out = _git(root, "status", "--porcelain", "--untracked-files=no")
154
+ return [line[3:].strip().replace("\\", "/") for line in out.splitlines() if len(line) > 3]
155
+
156
+
157
+ def merge_base(root: Path, ref: str) -> str:
158
+ """The commit REF and HEAD forked from — a branch's real diff basis, which
159
+ is what a mid-branch run's own commit is not."""
160
+ out = _git(root, "merge-base", ref, "HEAD").strip()
161
+ if not out:
162
+ raise GitError(f"no merge base between {ref} and HEAD in {root}")
163
+ return out
164
+
165
+
166
+ def is_ancestor(root: Path, commit: str, other: str = "HEAD") -> bool:
167
+ """True when `commit` is at or behind `other`; git counts a commit as its own
168
+ ancestor, which is what "at or behind" needs."""
169
+ try:
170
+ res = subprocess.run(["git", "merge-base", "--is-ancestor", commit, other],
171
+ cwd=root, capture_output=True)
172
+ except FileNotFoundError as exc:
173
+ raise GitError("git executable not found") from exc
174
+ return res.returncode == 0
175
+
176
+
177
+ def _batch_stream(root: Path, requests: bytes) -> bytes:
178
+ try:
179
+ res = subprocess.run(["git", "cat-file", "--batch"], cwd=root,
180
+ input=requests, capture_output=True)
181
+ except FileNotFoundError as exc:
182
+ raise GitError("git executable not found") from exc
183
+ if res.returncode != 0:
184
+ raise GitError(f"git cat-file --batch failed in {root}: "
185
+ f"{res.stderr.decode('utf-8', 'replace').strip()}")
186
+ return res.stdout
187
+
188
+
189
+ def _framed_blob(stream: bytes, pos: int) -> tuple[bytes, int]:
190
+ """One record: `<oid> blob <size>` header line, exactly size bytes, then LF.
191
+
192
+ Sliced by the declared byte count, never split on newlines — blobs are
193
+ binary. A path absent from the index gets a `<request> missing` line and a
194
+ zero exit, so the absent case is detected here, not from a return code.
195
+ """
196
+ end = stream.index(b"\n", pos)
197
+ header = stream[pos:end].decode("utf-8", "replace")
198
+ if header.endswith(" missing"):
199
+ raise GitError(f"git cat-file --batch: {header[:-len(' missing')]} is not in the index")
200
+ body_at = end + 1
201
+ size = int(header.rsplit(" ", 1)[1])
202
+ return stream[body_at:body_at + size], body_at + size + 1
203
+
204
+
205
+ def _framed_blobs(stream: bytes, rel_paths: list[str]) -> dict[str, bytes]:
206
+ """The batch answer split back into one blob per requested path, in order."""
207
+ blobs: dict[str, bytes] = {}
208
+ pos = 0
209
+ for rel in rel_paths:
210
+ blobs[rel], pos = _framed_blob(stream, pos)
211
+ return blobs
212
+
213
+
214
+ def staged_blobs(root: Path, rel_paths: list[str]) -> dict[str, bytes]:
215
+ """Every staged blob from one `git cat-file --batch` process.
216
+
217
+ One `git show` per path costs ~22ms of process spawn each and dominates the
218
+ hook; batching makes the fetch flat in file count.
219
+ """
220
+ if not rel_paths:
221
+ return {}
222
+ requests = "".join(f":{rel}\n" for rel in rel_paths).encode("utf-8")
223
+ return _framed_blobs(_batch_stream(root, requests), rel_paths)
224
+
225
+
226
+ class _Started:
227
+ """A git process started now and read later.
228
+
229
+ communicate() writes the request and reads the answer in one call, so a
230
+ request stream larger than a pipe buffer cannot deadlock against the child's
231
+ own output. text=True mirrors _git exactly, universal newlines included: a
232
+ caller that switched to this must not start seeing CR at the end of every
233
+ diff line.
234
+ """
235
+
236
+ def __init__(self, root: Path, args: tuple[str, ...], *, text: bool, stdin: bool) -> None:
237
+ self._args, self._root, self._text = args, root, text
238
+ try:
239
+ self._proc = subprocess.Popen(
240
+ ["git", *args], cwd=root,
241
+ stdin=subprocess.PIPE if stdin else subprocess.DEVNULL,
242
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE,
243
+ text=text, encoding="utf-8" if text else None)
244
+ except FileNotFoundError as exc:
245
+ raise GitError("git executable not found") from exc
246
+
247
+ def result(self, payload=None):
248
+ out, err = self._proc.communicate(payload)
249
+ if self._proc.returncode != 0:
250
+ text = err if self._text else err.decode("utf-8", "replace")
251
+ raise GitError(f"git {' '.join(self._args)} failed in {self._root}: {text.strip()}")
252
+ return out
253
+
254
+ def close(self) -> None:
255
+ """Shut down a process nothing ever read. A process already read to the
256
+ end has a return code, and closing that one is nothing at all."""
257
+ if self._proc.returncode is None:
258
+ self._proc.kill()
259
+ self._proc.communicate()
260
+
261
+
262
+ class GitReads:
263
+ """Where the pre-commit gate's staged bytes come from: one git process each,
264
+ spawned at the moment the gate asks for it."""
265
+
266
+ def __init__(self, root: Path) -> None:
267
+ self.root = root
268
+
269
+ def staged_diff(self) -> str:
270
+ return staged_diff(self.root)
271
+
272
+ def staged_blobs(self, rel_paths: list[str]) -> dict[str, bytes]:
273
+ return staged_blobs(self.root, rel_paths)
274
+
275
+
276
+ class _StartedReads:
277
+ """The same two answers, from processes that are already running.
278
+
279
+ The gate cannot use either one until lizard is imported, and that import
280
+ costs more than both spawns together, so the spawns belong underneath it.
281
+ """
282
+
283
+ def __init__(self, root: Path) -> None:
284
+ self._diff = _Started(root, ("diff", "--cached", "-U0", "--no-renames"),
285
+ text=True, stdin=False)
286
+ self._batch = _Started(root, ("cat-file", "--batch"), text=False, stdin=True)
287
+
288
+ def staged_diff(self) -> str:
289
+ return self._diff.result()
290
+
291
+ def staged_blobs(self, rel_paths: list[str]) -> dict[str, bytes]:
292
+ if not rel_paths:
293
+ return {}
294
+ stream = self._batch.result("".join(f":{rel}\n" for rel in rel_paths).encode("utf-8"))
295
+ return _framed_blobs(stream, rel_paths)
296
+
297
+ def close(self) -> None:
298
+ self._diff.close()
299
+ self._batch.close()
300
+
301
+
302
+ @contextmanager
303
+ def staged_reads(root: Path):
304
+ """The gate's two git reads, started before the caller needs either.
305
+
306
+ Both processes are shut down on the way out, whichever of them the caller got
307
+ around to reading: a commit with nothing staged never asks for a blob, and a
308
+ machine with no lizard never asks for anything at all.
309
+ """
310
+ reads = _StartedReads(root)
311
+ try:
312
+ yield reads
313
+ finally:
314
+ reads.close()
315
+
316
+
317
+ def file_log_patches(root: Path, rel_path: str) -> list[tuple[int, str]]:
318
+ """(commit timestamp, unified patch) per commit touching one file, oldest first.
319
+
320
+ -U0: the only reader is ratchet_report, which looks at +/- lines alone, so
321
+ context lines are pipe traffic that grows with the ratchet file.
322
+ No --follow: rename detection cost 0.6s of a 1.14s `ratchet report` on a
323
+ 72k-commit history and found nothing. The cost is real, since --follow also
324
+ gives up the commit-graph path filtering the plain log gets (measured on a
325
+ 30k-commit synthetic: 0.436s vs 0.257s, same events either way). The price
326
+ is that renaming the ratchet file restarts its burn-down history at the
327
+ rename.
328
+ """
329
+ out = _git(root, "log", "--reverse", "--format=%x01%at", "-p", "-U0", "--", rel_path)
330
+ patches = []
331
+ for block in out.split("\x01"):
332
+ if not block.strip():
333
+ continue
334
+ head, _, patch = block.partition("\n")
335
+ patches.append((int(head.strip()), patch))
336
+ return patches
337
+
338
+
339
+ def churn_log_lines(root: Path, months: int) -> Iterator[str]:
340
+ """The churn window's log, streamed. On a big repo this is 21 MB of text and
341
+ the single most expensive call crapkit makes, so it is never held whole."""
342
+ return _git_lines(root, "log", f"--since={months} months ago",
343
+ "--format=%x01%an%x02%at", "--name-only")
344
+
345
+
346
+ def worktree_add(root: Path, path: Path) -> None:
347
+ """A detached checkout of HEAD at `path`: a second working tree that shares
348
+ the object store, so a worker can edit files without touching the real one."""
349
+ _git(root, "worktree", "add", "--detach", str(path))
350
+
351
+
352
+ def worktree_remove(root: Path, path: Path) -> None:
353
+ """Teardown. --force because the worker's tree is dirty by construction, and
354
+ it never raises: a cleanup error must not mask the failure that caused it.
355
+ `prune` is the fallback that drops the admin entry a stuck directory leaves."""
356
+ try:
357
+ _git(root, "worktree", "remove", "--force", str(path))
358
+ except GitError:
359
+ shutil.rmtree(path, ignore_errors=True)
360
+ _prune_quietly(root)
361
+
362
+
363
+ def _prune_quietly(root: Path) -> None:
364
+ try:
365
+ _git(root, "worktree", "prune")
366
+ except GitError:
367
+ pass
368
+
369
+
370
+ def _git_dir(root: Path) -> Path | None:
371
+ """.git is a directory in a normal clone and a `gitdir:` pointer in a linked
372
+ worktree or a submodule. The pointer may be relative to the working tree."""
373
+ dot = root / ".git"
374
+ if dot.is_dir():
375
+ return dot
376
+ text = _file_text(dot)
377
+ if not text.startswith("gitdir:"):
378
+ return None
379
+ named = Path(text[len("gitdir:"):].strip())
380
+ return named if named.is_absolute() else (root / named)
381
+
382
+
383
+ def _file_text(path: Path) -> str:
384
+ """A ref file's content, or "" when it is not there or not readable.
385
+
386
+ Unreadable is an answer here, not a failure: every caller's fallback is the
387
+ git process, which is what read the file correctly in the first place.
388
+ """
389
+ try:
390
+ return path.read_text(encoding="utf-8").strip()
391
+ except (OSError, UnicodeDecodeError):
392
+ return ""
393
+
394
+
395
+ def _sha_or_none(text: str) -> str | None:
396
+ """Object names only: 40 hex for sha1 repos, 64 for sha256 ones."""
397
+ return text if _OBJECT_NAME.fullmatch(text) else None
398
+
399
+
400
+ def _packed_sha(gitdir: Path, ref: str) -> str | None:
401
+ """The ref's line in packed-refs, where `git pack-refs` puts it once the
402
+ loose file is gone. `^`-prefixed lines are peeled tags and never a match."""
403
+ for line in _file_text(gitdir / "packed-refs").splitlines():
404
+ sha, _, name = line.partition(" ")
405
+ if name.strip() == ref:
406
+ return _sha_or_none(sha)
407
+ return None
408
+
409
+
410
+ def _common_dir(gitdir: Path) -> Path:
411
+ """A linked worktree keeps HEAD in its own admin directory and shares
412
+ refs/heads with the repository it was made from."""
413
+ common = _file_text(gitdir / "commondir")
414
+ return (gitdir / common).resolve() if common else gitdir
415
+
416
+
417
+ def _ref_sha(gitdir: Path, ref: str) -> str | None:
418
+ loose = _sha_or_none(_file_text(gitdir / ref))
419
+ if loose:
420
+ return loose
421
+ shared = _common_dir(gitdir)
422
+ return _sha_or_none(_file_text(shared / ref)) or _packed_sha(shared, ref)
423
+
424
+
425
+ def head_from_refs(root: Path) -> str | None:
426
+ """HEAD read straight out of .git, or None meaning "ask git".
427
+
428
+ Every command opens by asking where HEAD is, and the answer is a 40-character
429
+ string in a file: on Windows the spawn that fetches it costs ~20ms. Anything
430
+ unexpected — a symref chain, a ref this does not find, a torn write — returns
431
+ None rather than a guess, and the caller pays for the process instead.
432
+ """
433
+ gitdir = _git_dir(root)
434
+ if gitdir is None:
435
+ return None
436
+ head = _file_text(gitdir / "HEAD")
437
+ if not head.startswith("ref: "):
438
+ return _sha_or_none(head) # detached: HEAD holds the object name itself
439
+ return _ref_sha(gitdir, head[len("ref: "):].strip())
440
+
441
+
442
+ def head_commit(root: Path) -> str:
443
+ fast = head_from_refs(root)
444
+ if fast:
445
+ return fast
446
+ out = _git(root, "rev-parse", "HEAD").strip()
447
+ if not out:
448
+ raise GitError(f"no HEAD commit in {root}")
449
+ return out
450
+
451
+
452
+ class GitFacts:
453
+ """One command's answers to the three questions every lane asks.
454
+
455
+ HEAD, the dirty-file set and a diff against a stamp commit are the same for
456
+ every lane in a run, but each lane used to pay its own `git` spawn for all
457
+ three. Build one of these per command and pass it down.
458
+
459
+ Asking once also FIXES the answer at the moment the run started, which is
460
+ what the reuse decision wants: a lane command writes into the working tree,
461
+ so a later lane re-asking git would judge itself against another lane's
462
+ output. Errors are not memoized — GitError propagates on every call, so a
463
+ non-git sandbox keeps behaving like one.
464
+
465
+ Parallel lanes share one of these, so the lazy fills take a lock: the first
466
+ caller pays the spawn and the rest wait for its answer instead of racing to
467
+ ask git the same question again.
468
+ """
469
+
470
+ def __init__(self, root: Path) -> None:
471
+ self.root = root
472
+ self._lock = threading.Lock()
473
+ self._head: str | None = None
474
+ self._status: tuple[str, ...] | None = None
475
+ self._diffs: dict[str, tuple[str, ...]] = {}
476
+ self._ancestry: dict[tuple[str, str], bool] = {}
477
+
478
+ def head_commit(self) -> str:
479
+ with self._lock:
480
+ if self._head is None:
481
+ self._head = head_commit(self.root)
482
+ return self._head
483
+
484
+ def status_names(self) -> tuple[str, ...]:
485
+ with self._lock:
486
+ if self._status is None:
487
+ self._status = tuple(status_names(self.root))
488
+ return self._status
489
+
490
+ def diff_names_since(self, commit: str) -> tuple[str, ...]:
491
+ with self._lock:
492
+ if commit not in self._diffs:
493
+ self._diffs[commit] = tuple(diff_names_since(self.root, commit))
494
+ return self._diffs[commit]
495
+
496
+ def is_ancestor(self, commit: str, other: str = "HEAD") -> bool:
497
+ """Memoized per (commit, other) the way the diffs are: verify asks about
498
+ the same commit once per lane, once per open claim and once for the
499
+ baseline, and history does not move under a running command."""
500
+ with self._lock:
501
+ key = (commit, other)
502
+ if key not in self._ancestry:
503
+ self._ancestry[key] = is_ancestor(self.root, commit, other)
504
+ return self._ancestry[key]
crapkit/hook.py ADDED
@@ -0,0 +1,167 @@
1
+ """Pre-commit gate: min-CCN over the target on functions touched by the staged diff.
2
+
3
+ Checks STAGED blobs, never the working tree, so unstaged noise cannot block a
4
+ clean commit and a dirty checkout cannot sneak past one.
5
+
6
+ The hook is a hot path measured in what a developer waits for at every `git
7
+ commit`, so it holds three rules the batch commands do not:
8
+
9
+ - One `git cat-file --batch` for all staged blobs, and each blob fetched once.
10
+ Per-file `git show` spawns cost ~22ms each and made the floor scale with the
11
+ commit size.
12
+ - No repo-wide analysis cache. Loading and rewriting it cost a flat 0.45s on an
13
+ 18 MB cache while the analysis it could save is milliseconds: a commit's worth
14
+ of blobs is cheaper to analyze outright than to look up.
15
+ - A commit's worth of files is analyzed in this process, from the blobs already
16
+ in memory: no worker pool below the crossover, and no temp tree to read back.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import codecs
21
+ import io
22
+ import tempfile
23
+ from pathlib import Path
24
+ from typing import NamedTuple
25
+
26
+ from .analyze import analyze_jobs, analyze_source
27
+ from .config import Config
28
+ from .diffparse import changed_ranges
29
+ from .gitio import GitReads
30
+ from .merge import FunctionRecord
31
+ from .universe import _source_extensions, assign_files, exclude_matcher, excluded
32
+
33
+ # A commit's worth of files, not an inventory's, and never more workers than a
34
+ # commit can keep busy: each worker re-imports lizard, which is the whole cost of
35
+ # a small pool. Measured here at 9 reps a point (serial vs pooled medians, ms):
36
+ # 4 files 17/120, 8 files 52/158, 12 files 112/175, 16 files 178/191, 20 files
37
+ # 197/184, 60 files 505/228. The arms only cross at 16, so that is the threshold;
38
+ # below it the pool spends ~100ms of spawn to save single-digit milliseconds.
39
+ _HOOK_POOL_THRESHOLD = 16
40
+ _HOOK_MAX_WORKERS = 8
41
+
42
+
43
+ class Violation(NamedTuple):
44
+ path: str
45
+ long_name: str
46
+ start: int
47
+ ccn: int
48
+
49
+
50
+ class StagedGate(NamedTuple):
51
+ """The verdict on the staged blobs."""
52
+ violations: list[Violation]
53
+ unscoped: list[str] = [] # staged source files no scope claims: ungated, but never silently
54
+
55
+
56
+ def _touches(record: FunctionRecord, ranges: list[tuple[int, int]]) -> bool:
57
+ return any(not (hi < record.start or lo > record.end) for lo, hi in ranges)
58
+
59
+
60
+ def _materialized(tmp: Path, blobs: dict[str, bytes]) -> list[tuple[str, str]]:
61
+ """Write each staged blob under its own repo-relative path.
62
+
63
+ The repo path, not the basename: src/a/index.ts and src/b/index.ts are one
64
+ file at basename granularity, and analyzing one twice would gate the wrong
65
+ content.
66
+ """
67
+ jobs = []
68
+ for rel, blob in sorted(blobs.items()):
69
+ staged = tmp / rel
70
+ staged.parent.mkdir(parents=True, exist_ok=True)
71
+ staged.write_bytes(blob)
72
+ jobs.append((str(staged), rel))
73
+ return jobs
74
+
75
+
76
+ def _text_of(blob: bytes) -> str:
77
+ """The blob as lizard's own auto_read would have read it back from a file.
78
+
79
+ Three things have to match or the records move: the UTF-8 BOM selects
80
+ utf-8-sig, everything else takes the same default encoding `io.open` takes,
81
+ and text mode translates line endings. A plain `blob.decode()` skips the
82
+ translation, so a CR-only file arrives as one line and lizard reports no
83
+ functions in it at all — the gate would then pass a file it never judged.
84
+ """
85
+ encoding = "utf-8-sig" if blob.startswith(codecs.BOM_UTF8) else None
86
+ return io.TextIOWrapper(io.BytesIO(blob), encoding=encoding, newline=None).read()
87
+
88
+
89
+ def _decoded(blob: bytes) -> str:
90
+ """auto_read's last resort too: bytes no decoder accepts lose the bad ones
91
+ rather than failing the commit."""
92
+ try:
93
+ return _text_of(blob)
94
+ except UnicodeDecodeError:
95
+ return blob.decode("utf-8", "ignore")
96
+
97
+
98
+ def staged_records(blobs: dict[str, bytes]) -> dict[str, list]:
99
+ """Records for the staged blobs, pooled once a commit touches enough files.
100
+
101
+ Below the pool threshold lizard is handed the blob text directly: the bytes
102
+ are already in memory from `git cat-file --batch`, and a temp tree only to
103
+ read them back costs a write and a read per file. The pooled arm still
104
+ materializes, because a worker process reads its own files.
105
+ """
106
+ if len(blobs) < _HOOK_POOL_THRESHOLD:
107
+ return {rel: analyze_source(rel, _decoded(blob)) for rel, blob in sorted(blobs.items())}
108
+ with tempfile.TemporaryDirectory() as tmp:
109
+ jobs = _materialized(Path(tmp), blobs)
110
+ return analyze_jobs(jobs, workers=min(len(jobs), _HOOK_MAX_WORKERS),
111
+ pool_threshold=_HOOK_POOL_THRESHOLD, chunksize=1)
112
+
113
+
114
+ def file_ceilings(cfg, in_scope, checked_files) -> dict[str, int]:
115
+ """The ccn ceiling each file is judged against: its scope's target, else the
116
+ repo's. `rescore --gate` decides on this same map, so a mid-session verdict
117
+ and the commit's cannot disagree."""
118
+ ceilings = cfg.scope_targets
119
+ scope_of = {f: scope for scope, files in in_scope.items() for f in files}
120
+ return {rel: ceilings.get(scope_of.get(rel, ""), cfg.target) for rel in checked_files}
121
+
122
+
123
+ def _touched_over_ceiling(records_by_path, ranges_by_path, checked_files, cfg, in_scope) -> list[Violation]:
124
+ by_file = file_ceilings(cfg, in_scope, checked_files)
125
+ violations = []
126
+ for rel in checked_files:
127
+ ceiling = by_file[rel]
128
+ for rec in records_by_path[rel]:
129
+ if rec.ccn > ceiling and _touches(rec, ranges_by_path[rel]):
130
+ violations.append(Violation(rel, rec.long_name, rec.start, rec.ccn))
131
+ violations.sort(key=lambda v: (-v.ccn, v.path, v.start))
132
+ return violations
133
+
134
+
135
+ def _gate_blind_to(path: str, checked: set[str], exts: tuple, match) -> bool:
136
+ """A staged file the gate cannot judge but a scope language claims by
137
+ extension. Files the config EXCLUDES (test trees, exclude globs) are
138
+ outside scopes on purpose and never a hole."""
139
+ return path not in checked and path.endswith(exts) and not excluded(path, match)
140
+
141
+
142
+ def _unscoped_sources(staged: list[str], checked: set[str], cfg: Config) -> list[str]:
143
+ """The first new top-level directory a repo grows commits ungated; this is
144
+ how that hole stays visible instead of silent."""
145
+ exts = tuple(e for scope in cfg.scopes for e in _source_extensions(scope.languages))
146
+ match = exclude_matcher(cfg.exclude_globs)
147
+ return sorted(f for f in staged if _gate_blind_to(f, checked, exts, match))
148
+
149
+
150
+ def gate_staged(root: Path, cfg: Config, reads=None) -> StagedGate:
151
+ """`reads` is where the staged bytes come from: git processes the caller
152
+ already started, or a spawn-on-demand pair when nobody did. The verdict is
153
+ the same either way."""
154
+ reads = reads or GitReads(root)
155
+ ranges_by_path = changed_ranges(reads.staged_diff())
156
+ if not ranges_by_path:
157
+ return StagedGate([])
158
+ in_scope = assign_files(sorted(ranges_by_path), cfg)
159
+ checked_files = sorted({f for files in in_scope.values() for f in files})
160
+ unscoped = _unscoped_sources(sorted(ranges_by_path), set(checked_files), cfg)
161
+ if not checked_files:
162
+ return StagedGate([], unscoped)
163
+ records_by_path = staged_records(reads.staged_blobs(checked_files))
164
+ return StagedGate(
165
+ _touched_over_ceiling(records_by_path, ranges_by_path, checked_files, cfg, in_scope),
166
+ unscoped
167
+ )