crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/gitio.py
ADDED
|
@@ -0,0 +1,504 @@
|
|
|
1
|
+
"""Git shell layer: the tracked-file universe and the current commit."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
import shutil
|
|
6
|
+
import subprocess
|
|
7
|
+
import threading
|
|
8
|
+
from collections.abc import Iterator
|
|
9
|
+
from contextlib import contextmanager
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from .errors import GitError
|
|
13
|
+
|
|
14
|
+
_OBJECT_NAME = re.compile(r"[0-9a-f]{40}|[0-9a-f]{64}")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _git(root: Path, *args: str) -> str:
|
|
18
|
+
try:
|
|
19
|
+
res = subprocess.run(["git", *args], cwd=root, capture_output=True, text=True, encoding="utf-8")
|
|
20
|
+
except FileNotFoundError as exc:
|
|
21
|
+
raise GitError("git executable not found") from exc
|
|
22
|
+
if res.returncode != 0:
|
|
23
|
+
raise GitError(f"git {' '.join(args)} failed in {root}: {res.stderr.strip()}")
|
|
24
|
+
return res.stdout
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _git_lines(root: Path, *args: str) -> Iterator[str]:
|
|
28
|
+
"""Same contract as _git, streamed: the caller sees one line at a time.
|
|
29
|
+
|
|
30
|
+
A failing command yields nothing and raises at the end of iteration, so the
|
|
31
|
+
consumer never mistakes an empty stream for an empty history.
|
|
32
|
+
"""
|
|
33
|
+
try:
|
|
34
|
+
proc = subprocess.Popen(["git", *args], cwd=root, stdout=subprocess.PIPE,
|
|
35
|
+
stderr=subprocess.PIPE, text=True, encoding="utf-8")
|
|
36
|
+
except FileNotFoundError as exc:
|
|
37
|
+
raise GitError("git executable not found") from exc
|
|
38
|
+
with proc:
|
|
39
|
+
yield from proc.stdout
|
|
40
|
+
stderr = proc.stderr.read()
|
|
41
|
+
if proc.returncode != 0:
|
|
42
|
+
raise GitError(f"git {' '.join(args)} failed in {root}: {stderr.strip()}")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def stage_path(root: Path, rel_path: str) -> None:
|
|
46
|
+
"""git add one path — the hook override's ratchet debt must land IN the commit."""
|
|
47
|
+
_git(root, "add", "--", rel_path)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def ls_files(root: Path) -> list[str]:
|
|
51
|
+
out = _git(root, "ls-files", "-z")
|
|
52
|
+
return [p for p in out.split("\0") if p]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def untracked_files(root: Path) -> list[str]:
|
|
56
|
+
"""Paths `git add` would pick up: untracked and not ignored.
|
|
57
|
+
|
|
58
|
+
git applies the ignore rules, so a build directory never reads as source
|
|
59
|
+
somebody forgot to add.
|
|
60
|
+
"""
|
|
61
|
+
out = _git(root, "ls-files", "--others", "--exclude-standard", "-z")
|
|
62
|
+
return [p for p in out.split("\0") if p]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def config_value(root: Path, key: str) -> str:
|
|
66
|
+
"""One `git config` value, or "" when it is unset.
|
|
67
|
+
|
|
68
|
+
`git config --get` exits 1 on an unset key, which is an answer rather than a
|
|
69
|
+
failure — every caller here asks about a setting the repo need not have.
|
|
70
|
+
"""
|
|
71
|
+
try:
|
|
72
|
+
return _git(root, "config", "--get", key).strip()
|
|
73
|
+
except GitError:
|
|
74
|
+
return ""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def index_modes(root: Path, pathspec: str) -> dict[str, str]:
|
|
78
|
+
"""path -> index mode for everything git tracks under `pathspec`.
|
|
79
|
+
|
|
80
|
+
`git ls-files -s` is the only place the executable bit is readable on
|
|
81
|
+
Windows, where the filesystem has no such bit and the working copy always
|
|
82
|
+
looks 0644.
|
|
83
|
+
"""
|
|
84
|
+
out = _git(root, "ls-files", "-s", "-z", "--", pathspec)
|
|
85
|
+
modes = {}
|
|
86
|
+
for record in out.split("\0"):
|
|
87
|
+
if record:
|
|
88
|
+
meta, _, path = record.partition("\t")
|
|
89
|
+
modes[path.replace("\\", "/")] = meta.split(" ", 1)[0]
|
|
90
|
+
return modes
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def staged_diff(root: Path) -> str:
|
|
94
|
+
# --no-renames: a renamed file becomes delete+add, so a rename stays a
|
|
95
|
+
# touched file and its functions still face the gate (and the old ratchet
|
|
96
|
+
# entry's drop is matched by fresh gating at the new path).
|
|
97
|
+
return _git(root, "diff", "--cached", "-U0", "--no-renames")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def unstaged_paths(root: Path) -> set[str]:
|
|
101
|
+
"""Tracked files whose working-tree content differs from the index.
|
|
102
|
+
|
|
103
|
+
git decides it, through its own filters. Comparing a staged blob to the
|
|
104
|
+
file's raw bytes reads every file as different under `core.autocrlf=true` —
|
|
105
|
+
git-for-windows' installer default — because the blob holds LF and the
|
|
106
|
+
checkout holds CRLF by design.
|
|
107
|
+
"""
|
|
108
|
+
out = _git(root, "diff", "--name-only", "--no-renames")
|
|
109
|
+
return {line.strip().replace("\\", "/") for line in out.splitlines() if line.strip()}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def diff_since(root: Path, commit: str) -> str:
|
|
113
|
+
return _git(root, "diff", commit, "-U0", "--no-renames")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def diff_names_since(root: Path, commit: str) -> list[str]:
|
|
117
|
+
"""Files with committed changes between a commit and HEAD."""
|
|
118
|
+
out = _git(root, "diff", "--name-only", "--no-renames", commit, "HEAD")
|
|
119
|
+
return [line.strip().replace("\\", "/") for line in out.splitlines() if line.strip()]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _rename_pairs(fields: list[str]) -> dict[str, str]:
|
|
123
|
+
"""Walk `--name-status -z` records: a status field, then one path — two for R and C.
|
|
124
|
+
|
|
125
|
+
Consuming one path per record would read a rename's destination as the next
|
|
126
|
+
record's status and shift every entry after it.
|
|
127
|
+
"""
|
|
128
|
+
pairs: dict[str, str] = {}
|
|
129
|
+
i = 0
|
|
130
|
+
while i < len(fields) and fields[i]:
|
|
131
|
+
status = fields[i]
|
|
132
|
+
paths = 2 if status[0] in ("R", "C") else 1
|
|
133
|
+
if status[0] == "R":
|
|
134
|
+
pairs[fields[i + 1].replace("\\", "/")] = fields[i + 2].replace("\\", "/")
|
|
135
|
+
i += 1 + paths
|
|
136
|
+
return pairs
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def renamed_paths(root: Path, since: str, *, similarity: int = 50) -> dict[str, str]:
|
|
140
|
+
"""old path -> new path for files git reads as renamed between `since` and HEAD.
|
|
141
|
+
|
|
142
|
+
Tree-to-tree, not a walk of history: a rename here is content similarity
|
|
143
|
+
between the two endpoints, so widening the window costs nothing extra and
|
|
144
|
+
cannot invent a pairing git does not already see. Copies are excluded — the
|
|
145
|
+
source still exists, so nothing about it moved.
|
|
146
|
+
"""
|
|
147
|
+
out = _git(root, "diff", "--name-status", f"-M{similarity}", "-z", since, "HEAD")
|
|
148
|
+
return _rename_pairs(out.split("\0"))
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def status_names(root: Path) -> list[str]:
|
|
152
|
+
"""Files with uncommitted (staged or unstaged) changes in the working tree."""
|
|
153
|
+
out = _git(root, "status", "--porcelain", "--untracked-files=no")
|
|
154
|
+
return [line[3:].strip().replace("\\", "/") for line in out.splitlines() if len(line) > 3]
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def merge_base(root: Path, ref: str) -> str:
|
|
158
|
+
"""The commit REF and HEAD forked from — a branch's real diff basis, which
|
|
159
|
+
is what a mid-branch run's own commit is not."""
|
|
160
|
+
out = _git(root, "merge-base", ref, "HEAD").strip()
|
|
161
|
+
if not out:
|
|
162
|
+
raise GitError(f"no merge base between {ref} and HEAD in {root}")
|
|
163
|
+
return out
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def is_ancestor(root: Path, commit: str, other: str = "HEAD") -> bool:
|
|
167
|
+
"""True when `commit` is at or behind `other`; git counts a commit as its own
|
|
168
|
+
ancestor, which is what "at or behind" needs."""
|
|
169
|
+
try:
|
|
170
|
+
res = subprocess.run(["git", "merge-base", "--is-ancestor", commit, other],
|
|
171
|
+
cwd=root, capture_output=True)
|
|
172
|
+
except FileNotFoundError as exc:
|
|
173
|
+
raise GitError("git executable not found") from exc
|
|
174
|
+
return res.returncode == 0
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _batch_stream(root: Path, requests: bytes) -> bytes:
|
|
178
|
+
try:
|
|
179
|
+
res = subprocess.run(["git", "cat-file", "--batch"], cwd=root,
|
|
180
|
+
input=requests, capture_output=True)
|
|
181
|
+
except FileNotFoundError as exc:
|
|
182
|
+
raise GitError("git executable not found") from exc
|
|
183
|
+
if res.returncode != 0:
|
|
184
|
+
raise GitError(f"git cat-file --batch failed in {root}: "
|
|
185
|
+
f"{res.stderr.decode('utf-8', 'replace').strip()}")
|
|
186
|
+
return res.stdout
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _framed_blob(stream: bytes, pos: int) -> tuple[bytes, int]:
|
|
190
|
+
"""One record: `<oid> blob <size>` header line, exactly size bytes, then LF.
|
|
191
|
+
|
|
192
|
+
Sliced by the declared byte count, never split on newlines — blobs are
|
|
193
|
+
binary. A path absent from the index gets a `<request> missing` line and a
|
|
194
|
+
zero exit, so the absent case is detected here, not from a return code.
|
|
195
|
+
"""
|
|
196
|
+
end = stream.index(b"\n", pos)
|
|
197
|
+
header = stream[pos:end].decode("utf-8", "replace")
|
|
198
|
+
if header.endswith(" missing"):
|
|
199
|
+
raise GitError(f"git cat-file --batch: {header[:-len(' missing')]} is not in the index")
|
|
200
|
+
body_at = end + 1
|
|
201
|
+
size = int(header.rsplit(" ", 1)[1])
|
|
202
|
+
return stream[body_at:body_at + size], body_at + size + 1
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _framed_blobs(stream: bytes, rel_paths: list[str]) -> dict[str, bytes]:
|
|
206
|
+
"""The batch answer split back into one blob per requested path, in order."""
|
|
207
|
+
blobs: dict[str, bytes] = {}
|
|
208
|
+
pos = 0
|
|
209
|
+
for rel in rel_paths:
|
|
210
|
+
blobs[rel], pos = _framed_blob(stream, pos)
|
|
211
|
+
return blobs
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def staged_blobs(root: Path, rel_paths: list[str]) -> dict[str, bytes]:
|
|
215
|
+
"""Every staged blob from one `git cat-file --batch` process.
|
|
216
|
+
|
|
217
|
+
One `git show` per path costs ~22ms of process spawn each and dominates the
|
|
218
|
+
hook; batching makes the fetch flat in file count.
|
|
219
|
+
"""
|
|
220
|
+
if not rel_paths:
|
|
221
|
+
return {}
|
|
222
|
+
requests = "".join(f":{rel}\n" for rel in rel_paths).encode("utf-8")
|
|
223
|
+
return _framed_blobs(_batch_stream(root, requests), rel_paths)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
class _Started:
|
|
227
|
+
"""A git process started now and read later.
|
|
228
|
+
|
|
229
|
+
communicate() writes the request and reads the answer in one call, so a
|
|
230
|
+
request stream larger than a pipe buffer cannot deadlock against the child's
|
|
231
|
+
own output. text=True mirrors _git exactly, universal newlines included: a
|
|
232
|
+
caller that switched to this must not start seeing CR at the end of every
|
|
233
|
+
diff line.
|
|
234
|
+
"""
|
|
235
|
+
|
|
236
|
+
def __init__(self, root: Path, args: tuple[str, ...], *, text: bool, stdin: bool) -> None:
|
|
237
|
+
self._args, self._root, self._text = args, root, text
|
|
238
|
+
try:
|
|
239
|
+
self._proc = subprocess.Popen(
|
|
240
|
+
["git", *args], cwd=root,
|
|
241
|
+
stdin=subprocess.PIPE if stdin else subprocess.DEVNULL,
|
|
242
|
+
stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
|
243
|
+
text=text, encoding="utf-8" if text else None)
|
|
244
|
+
except FileNotFoundError as exc:
|
|
245
|
+
raise GitError("git executable not found") from exc
|
|
246
|
+
|
|
247
|
+
def result(self, payload=None):
|
|
248
|
+
out, err = self._proc.communicate(payload)
|
|
249
|
+
if self._proc.returncode != 0:
|
|
250
|
+
text = err if self._text else err.decode("utf-8", "replace")
|
|
251
|
+
raise GitError(f"git {' '.join(self._args)} failed in {self._root}: {text.strip()}")
|
|
252
|
+
return out
|
|
253
|
+
|
|
254
|
+
def close(self) -> None:
|
|
255
|
+
"""Shut down a process nothing ever read. A process already read to the
|
|
256
|
+
end has a return code, and closing that one is nothing at all."""
|
|
257
|
+
if self._proc.returncode is None:
|
|
258
|
+
self._proc.kill()
|
|
259
|
+
self._proc.communicate()
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
class GitReads:
|
|
263
|
+
"""Where the pre-commit gate's staged bytes come from: one git process each,
|
|
264
|
+
spawned at the moment the gate asks for it."""
|
|
265
|
+
|
|
266
|
+
def __init__(self, root: Path) -> None:
|
|
267
|
+
self.root = root
|
|
268
|
+
|
|
269
|
+
def staged_diff(self) -> str:
|
|
270
|
+
return staged_diff(self.root)
|
|
271
|
+
|
|
272
|
+
def staged_blobs(self, rel_paths: list[str]) -> dict[str, bytes]:
|
|
273
|
+
return staged_blobs(self.root, rel_paths)
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
class _StartedReads:
|
|
277
|
+
"""The same two answers, from processes that are already running.
|
|
278
|
+
|
|
279
|
+
The gate cannot use either one until lizard is imported, and that import
|
|
280
|
+
costs more than both spawns together, so the spawns belong underneath it.
|
|
281
|
+
"""
|
|
282
|
+
|
|
283
|
+
def __init__(self, root: Path) -> None:
|
|
284
|
+
self._diff = _Started(root, ("diff", "--cached", "-U0", "--no-renames"),
|
|
285
|
+
text=True, stdin=False)
|
|
286
|
+
self._batch = _Started(root, ("cat-file", "--batch"), text=False, stdin=True)
|
|
287
|
+
|
|
288
|
+
def staged_diff(self) -> str:
|
|
289
|
+
return self._diff.result()
|
|
290
|
+
|
|
291
|
+
def staged_blobs(self, rel_paths: list[str]) -> dict[str, bytes]:
|
|
292
|
+
if not rel_paths:
|
|
293
|
+
return {}
|
|
294
|
+
stream = self._batch.result("".join(f":{rel}\n" for rel in rel_paths).encode("utf-8"))
|
|
295
|
+
return _framed_blobs(stream, rel_paths)
|
|
296
|
+
|
|
297
|
+
def close(self) -> None:
|
|
298
|
+
self._diff.close()
|
|
299
|
+
self._batch.close()
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
@contextmanager
|
|
303
|
+
def staged_reads(root: Path):
|
|
304
|
+
"""The gate's two git reads, started before the caller needs either.
|
|
305
|
+
|
|
306
|
+
Both processes are shut down on the way out, whichever of them the caller got
|
|
307
|
+
around to reading: a commit with nothing staged never asks for a blob, and a
|
|
308
|
+
machine with no lizard never asks for anything at all.
|
|
309
|
+
"""
|
|
310
|
+
reads = _StartedReads(root)
|
|
311
|
+
try:
|
|
312
|
+
yield reads
|
|
313
|
+
finally:
|
|
314
|
+
reads.close()
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def file_log_patches(root: Path, rel_path: str) -> list[tuple[int, str]]:
|
|
318
|
+
"""(commit timestamp, unified patch) per commit touching one file, oldest first.
|
|
319
|
+
|
|
320
|
+
-U0: the only reader is ratchet_report, which looks at +/- lines alone, so
|
|
321
|
+
context lines are pipe traffic that grows with the ratchet file.
|
|
322
|
+
No --follow: rename detection cost 0.6s of a 1.14s `ratchet report` on a
|
|
323
|
+
72k-commit history and found nothing. The cost is real, since --follow also
|
|
324
|
+
gives up the commit-graph path filtering the plain log gets (measured on a
|
|
325
|
+
30k-commit synthetic: 0.436s vs 0.257s, same events either way). The price
|
|
326
|
+
is that renaming the ratchet file restarts its burn-down history at the
|
|
327
|
+
rename.
|
|
328
|
+
"""
|
|
329
|
+
out = _git(root, "log", "--reverse", "--format=%x01%at", "-p", "-U0", "--", rel_path)
|
|
330
|
+
patches = []
|
|
331
|
+
for block in out.split("\x01"):
|
|
332
|
+
if not block.strip():
|
|
333
|
+
continue
|
|
334
|
+
head, _, patch = block.partition("\n")
|
|
335
|
+
patches.append((int(head.strip()), patch))
|
|
336
|
+
return patches
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def churn_log_lines(root: Path, months: int) -> Iterator[str]:
|
|
340
|
+
"""The churn window's log, streamed. On a big repo this is 21 MB of text and
|
|
341
|
+
the single most expensive call crapkit makes, so it is never held whole."""
|
|
342
|
+
return _git_lines(root, "log", f"--since={months} months ago",
|
|
343
|
+
"--format=%x01%an%x02%at", "--name-only")
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def worktree_add(root: Path, path: Path) -> None:
|
|
347
|
+
"""A detached checkout of HEAD at `path`: a second working tree that shares
|
|
348
|
+
the object store, so a worker can edit files without touching the real one."""
|
|
349
|
+
_git(root, "worktree", "add", "--detach", str(path))
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def worktree_remove(root: Path, path: Path) -> None:
|
|
353
|
+
"""Teardown. --force because the worker's tree is dirty by construction, and
|
|
354
|
+
it never raises: a cleanup error must not mask the failure that caused it.
|
|
355
|
+
`prune` is the fallback that drops the admin entry a stuck directory leaves."""
|
|
356
|
+
try:
|
|
357
|
+
_git(root, "worktree", "remove", "--force", str(path))
|
|
358
|
+
except GitError:
|
|
359
|
+
shutil.rmtree(path, ignore_errors=True)
|
|
360
|
+
_prune_quietly(root)
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _prune_quietly(root: Path) -> None:
|
|
364
|
+
try:
|
|
365
|
+
_git(root, "worktree", "prune")
|
|
366
|
+
except GitError:
|
|
367
|
+
pass
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _git_dir(root: Path) -> Path | None:
|
|
371
|
+
""".git is a directory in a normal clone and a `gitdir:` pointer in a linked
|
|
372
|
+
worktree or a submodule. The pointer may be relative to the working tree."""
|
|
373
|
+
dot = root / ".git"
|
|
374
|
+
if dot.is_dir():
|
|
375
|
+
return dot
|
|
376
|
+
text = _file_text(dot)
|
|
377
|
+
if not text.startswith("gitdir:"):
|
|
378
|
+
return None
|
|
379
|
+
named = Path(text[len("gitdir:"):].strip())
|
|
380
|
+
return named if named.is_absolute() else (root / named)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _file_text(path: Path) -> str:
|
|
384
|
+
"""A ref file's content, or "" when it is not there or not readable.
|
|
385
|
+
|
|
386
|
+
Unreadable is an answer here, not a failure: every caller's fallback is the
|
|
387
|
+
git process, which is what read the file correctly in the first place.
|
|
388
|
+
"""
|
|
389
|
+
try:
|
|
390
|
+
return path.read_text(encoding="utf-8").strip()
|
|
391
|
+
except (OSError, UnicodeDecodeError):
|
|
392
|
+
return ""
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _sha_or_none(text: str) -> str | None:
|
|
396
|
+
"""Object names only: 40 hex for sha1 repos, 64 for sha256 ones."""
|
|
397
|
+
return text if _OBJECT_NAME.fullmatch(text) else None
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _packed_sha(gitdir: Path, ref: str) -> str | None:
|
|
401
|
+
"""The ref's line in packed-refs, where `git pack-refs` puts it once the
|
|
402
|
+
loose file is gone. `^`-prefixed lines are peeled tags and never a match."""
|
|
403
|
+
for line in _file_text(gitdir / "packed-refs").splitlines():
|
|
404
|
+
sha, _, name = line.partition(" ")
|
|
405
|
+
if name.strip() == ref:
|
|
406
|
+
return _sha_or_none(sha)
|
|
407
|
+
return None
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def _common_dir(gitdir: Path) -> Path:
|
|
411
|
+
"""A linked worktree keeps HEAD in its own admin directory and shares
|
|
412
|
+
refs/heads with the repository it was made from."""
|
|
413
|
+
common = _file_text(gitdir / "commondir")
|
|
414
|
+
return (gitdir / common).resolve() if common else gitdir
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _ref_sha(gitdir: Path, ref: str) -> str | None:
|
|
418
|
+
loose = _sha_or_none(_file_text(gitdir / ref))
|
|
419
|
+
if loose:
|
|
420
|
+
return loose
|
|
421
|
+
shared = _common_dir(gitdir)
|
|
422
|
+
return _sha_or_none(_file_text(shared / ref)) or _packed_sha(shared, ref)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def head_from_refs(root: Path) -> str | None:
|
|
426
|
+
"""HEAD read straight out of .git, or None meaning "ask git".
|
|
427
|
+
|
|
428
|
+
Every command opens by asking where HEAD is, and the answer is a 40-character
|
|
429
|
+
string in a file: on Windows the spawn that fetches it costs ~20ms. Anything
|
|
430
|
+
unexpected — a symref chain, a ref this does not find, a torn write — returns
|
|
431
|
+
None rather than a guess, and the caller pays for the process instead.
|
|
432
|
+
"""
|
|
433
|
+
gitdir = _git_dir(root)
|
|
434
|
+
if gitdir is None:
|
|
435
|
+
return None
|
|
436
|
+
head = _file_text(gitdir / "HEAD")
|
|
437
|
+
if not head.startswith("ref: "):
|
|
438
|
+
return _sha_or_none(head) # detached: HEAD holds the object name itself
|
|
439
|
+
return _ref_sha(gitdir, head[len("ref: "):].strip())
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def head_commit(root: Path) -> str:
|
|
443
|
+
fast = head_from_refs(root)
|
|
444
|
+
if fast:
|
|
445
|
+
return fast
|
|
446
|
+
out = _git(root, "rev-parse", "HEAD").strip()
|
|
447
|
+
if not out:
|
|
448
|
+
raise GitError(f"no HEAD commit in {root}")
|
|
449
|
+
return out
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
class GitFacts:
|
|
453
|
+
"""One command's answers to the three questions every lane asks.
|
|
454
|
+
|
|
455
|
+
HEAD, the dirty-file set and a diff against a stamp commit are the same for
|
|
456
|
+
every lane in a run, but each lane used to pay its own `git` spawn for all
|
|
457
|
+
three. Build one of these per command and pass it down.
|
|
458
|
+
|
|
459
|
+
Asking once also FIXES the answer at the moment the run started, which is
|
|
460
|
+
what the reuse decision wants: a lane command writes into the working tree,
|
|
461
|
+
so a later lane re-asking git would judge itself against another lane's
|
|
462
|
+
output. Errors are not memoized — GitError propagates on every call, so a
|
|
463
|
+
non-git sandbox keeps behaving like one.
|
|
464
|
+
|
|
465
|
+
Parallel lanes share one of these, so the lazy fills take a lock: the first
|
|
466
|
+
caller pays the spawn and the rest wait for its answer instead of racing to
|
|
467
|
+
ask git the same question again.
|
|
468
|
+
"""
|
|
469
|
+
|
|
470
|
+
def __init__(self, root: Path) -> None:
|
|
471
|
+
self.root = root
|
|
472
|
+
self._lock = threading.Lock()
|
|
473
|
+
self._head: str | None = None
|
|
474
|
+
self._status: tuple[str, ...] | None = None
|
|
475
|
+
self._diffs: dict[str, tuple[str, ...]] = {}
|
|
476
|
+
self._ancestry: dict[tuple[str, str], bool] = {}
|
|
477
|
+
|
|
478
|
+
def head_commit(self) -> str:
|
|
479
|
+
with self._lock:
|
|
480
|
+
if self._head is None:
|
|
481
|
+
self._head = head_commit(self.root)
|
|
482
|
+
return self._head
|
|
483
|
+
|
|
484
|
+
def status_names(self) -> tuple[str, ...]:
|
|
485
|
+
with self._lock:
|
|
486
|
+
if self._status is None:
|
|
487
|
+
self._status = tuple(status_names(self.root))
|
|
488
|
+
return self._status
|
|
489
|
+
|
|
490
|
+
def diff_names_since(self, commit: str) -> tuple[str, ...]:
|
|
491
|
+
with self._lock:
|
|
492
|
+
if commit not in self._diffs:
|
|
493
|
+
self._diffs[commit] = tuple(diff_names_since(self.root, commit))
|
|
494
|
+
return self._diffs[commit]
|
|
495
|
+
|
|
496
|
+
def is_ancestor(self, commit: str, other: str = "HEAD") -> bool:
|
|
497
|
+
"""Memoized per (commit, other) the way the diffs are: verify asks about
|
|
498
|
+
the same commit once per lane, once per open claim and once for the
|
|
499
|
+
baseline, and history does not move under a running command."""
|
|
500
|
+
with self._lock:
|
|
501
|
+
key = (commit, other)
|
|
502
|
+
if key not in self._ancestry:
|
|
503
|
+
self._ancestry[key] = is_ancestor(self.root, commit, other)
|
|
504
|
+
return self._ancestry[key]
|
crapkit/hook.py
ADDED
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"""Pre-commit gate: min-CCN over the target on functions touched by the staged diff.
|
|
2
|
+
|
|
3
|
+
Checks STAGED blobs, never the working tree, so unstaged noise cannot block a
|
|
4
|
+
clean commit and a dirty checkout cannot sneak past one.
|
|
5
|
+
|
|
6
|
+
The hook is a hot path measured in what a developer waits for at every `git
|
|
7
|
+
commit`, so it holds three rules the batch commands do not:
|
|
8
|
+
|
|
9
|
+
- One `git cat-file --batch` for all staged blobs, and each blob fetched once.
|
|
10
|
+
Per-file `git show` spawns cost ~22ms each and made the floor scale with the
|
|
11
|
+
commit size.
|
|
12
|
+
- No repo-wide analysis cache. Loading and rewriting it cost a flat 0.45s on an
|
|
13
|
+
18 MB cache while the analysis it could save is milliseconds: a commit's worth
|
|
14
|
+
of blobs is cheaper to analyze outright than to look up.
|
|
15
|
+
- A commit's worth of files is analyzed in this process, from the blobs already
|
|
16
|
+
in memory: no worker pool below the crossover, and no temp tree to read back.
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import codecs
|
|
21
|
+
import io
|
|
22
|
+
import tempfile
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import NamedTuple
|
|
25
|
+
|
|
26
|
+
from .analyze import analyze_jobs, analyze_source
|
|
27
|
+
from .config import Config
|
|
28
|
+
from .diffparse import changed_ranges
|
|
29
|
+
from .gitio import GitReads
|
|
30
|
+
from .merge import FunctionRecord
|
|
31
|
+
from .universe import _source_extensions, assign_files, exclude_matcher, excluded
|
|
32
|
+
|
|
33
|
+
# A commit's worth of files, not an inventory's, and never more workers than a
|
|
34
|
+
# commit can keep busy: each worker re-imports lizard, which is the whole cost of
|
|
35
|
+
# a small pool. Measured here at 9 reps a point (serial vs pooled medians, ms):
|
|
36
|
+
# 4 files 17/120, 8 files 52/158, 12 files 112/175, 16 files 178/191, 20 files
|
|
37
|
+
# 197/184, 60 files 505/228. The arms only cross at 16, so that is the threshold;
|
|
38
|
+
# below it the pool spends ~100ms of spawn to save single-digit milliseconds.
|
|
39
|
+
_HOOK_POOL_THRESHOLD = 16
|
|
40
|
+
_HOOK_MAX_WORKERS = 8
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class Violation(NamedTuple):
|
|
44
|
+
path: str
|
|
45
|
+
long_name: str
|
|
46
|
+
start: int
|
|
47
|
+
ccn: int
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class StagedGate(NamedTuple):
|
|
51
|
+
"""The verdict on the staged blobs."""
|
|
52
|
+
violations: list[Violation]
|
|
53
|
+
unscoped: list[str] = [] # staged source files no scope claims: ungated, but never silently
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _touches(record: FunctionRecord, ranges: list[tuple[int, int]]) -> bool:
|
|
57
|
+
return any(not (hi < record.start or lo > record.end) for lo, hi in ranges)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _materialized(tmp: Path, blobs: dict[str, bytes]) -> list[tuple[str, str]]:
|
|
61
|
+
"""Write each staged blob under its own repo-relative path.
|
|
62
|
+
|
|
63
|
+
The repo path, not the basename: src/a/index.ts and src/b/index.ts are one
|
|
64
|
+
file at basename granularity, and analyzing one twice would gate the wrong
|
|
65
|
+
content.
|
|
66
|
+
"""
|
|
67
|
+
jobs = []
|
|
68
|
+
for rel, blob in sorted(blobs.items()):
|
|
69
|
+
staged = tmp / rel
|
|
70
|
+
staged.parent.mkdir(parents=True, exist_ok=True)
|
|
71
|
+
staged.write_bytes(blob)
|
|
72
|
+
jobs.append((str(staged), rel))
|
|
73
|
+
return jobs
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _text_of(blob: bytes) -> str:
|
|
77
|
+
"""The blob as lizard's own auto_read would have read it back from a file.
|
|
78
|
+
|
|
79
|
+
Three things have to match or the records move: the UTF-8 BOM selects
|
|
80
|
+
utf-8-sig, everything else takes the same default encoding `io.open` takes,
|
|
81
|
+
and text mode translates line endings. A plain `blob.decode()` skips the
|
|
82
|
+
translation, so a CR-only file arrives as one line and lizard reports no
|
|
83
|
+
functions in it at all — the gate would then pass a file it never judged.
|
|
84
|
+
"""
|
|
85
|
+
encoding = "utf-8-sig" if blob.startswith(codecs.BOM_UTF8) else None
|
|
86
|
+
return io.TextIOWrapper(io.BytesIO(blob), encoding=encoding, newline=None).read()
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _decoded(blob: bytes) -> str:
|
|
90
|
+
"""auto_read's last resort too: bytes no decoder accepts lose the bad ones
|
|
91
|
+
rather than failing the commit."""
|
|
92
|
+
try:
|
|
93
|
+
return _text_of(blob)
|
|
94
|
+
except UnicodeDecodeError:
|
|
95
|
+
return blob.decode("utf-8", "ignore")
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def staged_records(blobs: dict[str, bytes]) -> dict[str, list]:
|
|
99
|
+
"""Records for the staged blobs, pooled once a commit touches enough files.
|
|
100
|
+
|
|
101
|
+
Below the pool threshold lizard is handed the blob text directly: the bytes
|
|
102
|
+
are already in memory from `git cat-file --batch`, and a temp tree only to
|
|
103
|
+
read them back costs a write and a read per file. The pooled arm still
|
|
104
|
+
materializes, because a worker process reads its own files.
|
|
105
|
+
"""
|
|
106
|
+
if len(blobs) < _HOOK_POOL_THRESHOLD:
|
|
107
|
+
return {rel: analyze_source(rel, _decoded(blob)) for rel, blob in sorted(blobs.items())}
|
|
108
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
109
|
+
jobs = _materialized(Path(tmp), blobs)
|
|
110
|
+
return analyze_jobs(jobs, workers=min(len(jobs), _HOOK_MAX_WORKERS),
|
|
111
|
+
pool_threshold=_HOOK_POOL_THRESHOLD, chunksize=1)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def file_ceilings(cfg, in_scope, checked_files) -> dict[str, int]:
|
|
115
|
+
"""The ccn ceiling each file is judged against: its scope's target, else the
|
|
116
|
+
repo's. `rescore --gate` decides on this same map, so a mid-session verdict
|
|
117
|
+
and the commit's cannot disagree."""
|
|
118
|
+
ceilings = cfg.scope_targets
|
|
119
|
+
scope_of = {f: scope for scope, files in in_scope.items() for f in files}
|
|
120
|
+
return {rel: ceilings.get(scope_of.get(rel, ""), cfg.target) for rel in checked_files}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _touched_over_ceiling(records_by_path, ranges_by_path, checked_files, cfg, in_scope) -> list[Violation]:
|
|
124
|
+
by_file = file_ceilings(cfg, in_scope, checked_files)
|
|
125
|
+
violations = []
|
|
126
|
+
for rel in checked_files:
|
|
127
|
+
ceiling = by_file[rel]
|
|
128
|
+
for rec in records_by_path[rel]:
|
|
129
|
+
if rec.ccn > ceiling and _touches(rec, ranges_by_path[rel]):
|
|
130
|
+
violations.append(Violation(rel, rec.long_name, rec.start, rec.ccn))
|
|
131
|
+
violations.sort(key=lambda v: (-v.ccn, v.path, v.start))
|
|
132
|
+
return violations
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _gate_blind_to(path: str, checked: set[str], exts: tuple, match) -> bool:
|
|
136
|
+
"""A staged file the gate cannot judge but a scope language claims by
|
|
137
|
+
extension. Files the config EXCLUDES (test trees, exclude globs) are
|
|
138
|
+
outside scopes on purpose and never a hole."""
|
|
139
|
+
return path not in checked and path.endswith(exts) and not excluded(path, match)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _unscoped_sources(staged: list[str], checked: set[str], cfg: Config) -> list[str]:
|
|
143
|
+
"""The first new top-level directory a repo grows commits ungated; this is
|
|
144
|
+
how that hole stays visible instead of silent."""
|
|
145
|
+
exts = tuple(e for scope in cfg.scopes for e in _source_extensions(scope.languages))
|
|
146
|
+
match = exclude_matcher(cfg.exclude_globs)
|
|
147
|
+
return sorted(f for f in staged if _gate_blind_to(f, checked, exts, match))
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def gate_staged(root: Path, cfg: Config, reads=None) -> StagedGate:
|
|
151
|
+
"""`reads` is where the staged bytes come from: git processes the caller
|
|
152
|
+
already started, or a spawn-on-demand pair when nobody did. The verdict is
|
|
153
|
+
the same either way."""
|
|
154
|
+
reads = reads or GitReads(root)
|
|
155
|
+
ranges_by_path = changed_ranges(reads.staged_diff())
|
|
156
|
+
if not ranges_by_path:
|
|
157
|
+
return StagedGate([])
|
|
158
|
+
in_scope = assign_files(sorted(ranges_by_path), cfg)
|
|
159
|
+
checked_files = sorted({f for files in in_scope.values() for f in files})
|
|
160
|
+
unscoped = _unscoped_sources(sorted(ranges_by_path), set(checked_files), cfg)
|
|
161
|
+
if not checked_files:
|
|
162
|
+
return StagedGate([], unscoped)
|
|
163
|
+
records_by_path = staged_records(reads.staged_blobs(checked_files))
|
|
164
|
+
return StagedGate(
|
|
165
|
+
_touched_over_ceiling(records_by_path, ranges_by_path, checked_files, cfg, in_scope),
|
|
166
|
+
unscoped
|
|
167
|
+
)
|