crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/coverage_py.py
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""coverage.py JSON report parser. Pure: report text in, per-file function coverage out.
|
|
2
|
+
|
|
3
|
+
Requires the per-function regions coverage.py has emitted since 7.6.0 and the
|
|
4
|
+
start_line key from 7.13.1; requires branch data (the consuming lane must run
|
|
5
|
+
with branch coverage on) so the coverage term measures the same structure the
|
|
6
|
+
complexity term counts. Function spans come from start_line and the maximum
|
|
7
|
+
executed/missing line, the closest thing the report offers to an end line.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
|
|
13
|
+
from .coverage_istanbul import FnCoverage
|
|
14
|
+
from .errors import ToolError
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _fn_coverage(name: str, fn: dict) -> FnCoverage:
|
|
18
|
+
summary = fn.get("summary", {})
|
|
19
|
+
lines = list(fn.get("executed_lines", ())) + list(fn.get("missing_lines", ()))
|
|
20
|
+
start = fn.get("start_line") or (min(lines) if lines else 0)
|
|
21
|
+
end = max(lines) if lines else start
|
|
22
|
+
return FnCoverage(name=name, start=start, end=end,
|
|
23
|
+
invoked=summary.get("covered_lines", 0) > 0,
|
|
24
|
+
branches_total=summary.get("num_branches", 0),
|
|
25
|
+
branches_covered=summary.get("covered_branches", 0),
|
|
26
|
+
statements_total=summary.get("num_statements", 0),
|
|
27
|
+
statements_covered=summary.get("covered_lines", 0))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _file_functions(raw_path: str, data: dict) -> list[FnCoverage]:
|
|
31
|
+
functions = data.get("functions")
|
|
32
|
+
if functions is None:
|
|
33
|
+
raise ToolError(
|
|
34
|
+
f"coverage.py report has no function regions for {raw_path!r} — needs coverage >= 7.6")
|
|
35
|
+
# the "" key is the "(no function)" module-level bucket
|
|
36
|
+
fns = [_fn_coverage(name, fn) for name, fn in functions.items() if name]
|
|
37
|
+
return sorted(fns, key=lambda f: f.start)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def parse_coveragepy_missing(text: str, *, path_prefix: str) -> dict[str, set[int]]:
|
|
41
|
+
"""Per measured file, the lines coverage.py reports as never run."""
|
|
42
|
+
try:
|
|
43
|
+
report = json.loads(text)
|
|
44
|
+
prefix = (path_prefix.rstrip("/") + "/") if path_prefix else ""
|
|
45
|
+
return {prefix + p.replace("\\", "/"): set(data.get("missing_lines", ()))
|
|
46
|
+
for p, data in report.get("files", {}).items()}
|
|
47
|
+
except Exception as exc:
|
|
48
|
+
raise ToolError(f"unparseable coverage.py report: {exc}") from exc
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _line_contexts(raw: dict) -> dict[int, list[str]]:
|
|
52
|
+
out = {}
|
|
53
|
+
for line, contexts in raw.items():
|
|
54
|
+
tests = sorted({c.split("|")[0] for c in contexts if c})
|
|
55
|
+
if tests:
|
|
56
|
+
out[int(line)] = tests
|
|
57
|
+
return out
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def parse_coveragepy_contexts(text: str, *, path_prefix: str) -> dict[str, dict[int, list[str]]]:
|
|
61
|
+
"""line -> test ids per file, from a report made with --show-contexts and
|
|
62
|
+
dynamic_context = test_function. The empty module-import context is not a test."""
|
|
63
|
+
try:
|
|
64
|
+
report = json.loads(text)
|
|
65
|
+
prefix = (path_prefix.rstrip("/") + "/") if path_prefix else ""
|
|
66
|
+
out = {}
|
|
67
|
+
for p, data in report.get("files", {}).items():
|
|
68
|
+
contexts = _line_contexts(data.get("contexts", {}))
|
|
69
|
+
if contexts:
|
|
70
|
+
out[prefix + p.replace("\\", "/")] = contexts
|
|
71
|
+
return out
|
|
72
|
+
except Exception as exc:
|
|
73
|
+
raise ToolError(f"unparseable coverage.py report: {exc}") from exc
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def parse_coveragepy(text: str, *, path_prefix: str) -> dict[str, list[FnCoverage]]:
|
|
77
|
+
try:
|
|
78
|
+
report = json.loads(text)
|
|
79
|
+
if not report.get("meta", {}).get("branch_coverage"):
|
|
80
|
+
raise ToolError("coverage.py report lacks branch data — run the lane with branch coverage on")
|
|
81
|
+
prefix = (path_prefix.rstrip("/") + "/") if path_prefix else ""
|
|
82
|
+
return {prefix + raw_path.replace("\\", "/"): _file_functions(raw_path, data)
|
|
83
|
+
for raw_path, data in report.get("files", {}).items()}
|
|
84
|
+
except ToolError:
|
|
85
|
+
raise
|
|
86
|
+
except Exception as exc:
|
|
87
|
+
raise ToolError(f"unparseable coverage.py report: {exc}") from exc
|
crapkit/covstream.py
ADDED
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
"""Coverage artifacts read off the file instead of out of a string.
|
|
2
|
+
|
|
3
|
+
The whole-document parsers take `text`, so the caller must already hold the
|
|
4
|
+
artifact: `read_bytes()` plus its UTF-8 decode put two copies of a 150 MB
|
|
5
|
+
artifact on the heap before a single function is attributed. The splitter in
|
|
6
|
+
coverage_istanbul already decodes one member at a time; this module gives it a
|
|
7
|
+
window that refills from a handle instead of a string that holds everything.
|
|
8
|
+
|
|
9
|
+
Peak becomes O(chunk + largest member) rather than O(artifact): 322.6 -> 52.1 MB
|
|
10
|
+
on a 150 MB istanbul artifact, for byte-identical output and the same sha256.
|
|
11
|
+
|
|
12
|
+
Both shapes are split the same way. An istanbul artifact IS the {path: coverage}
|
|
13
|
+
object, so its members are files. A coverage.py report wraps them one level down
|
|
14
|
+
in "files", so the walk descends into that member and hands the rest back whole.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import codecs
|
|
19
|
+
import hashlib
|
|
20
|
+
import json
|
|
21
|
+
import re
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import IO, Iterator
|
|
24
|
+
|
|
25
|
+
from .coverage_istanbul import (_CLOSE_RE, _DECODER, _MEMBER_RE, _OPEN_RE,
|
|
26
|
+
FnCoverage, _dead_lines, _file_coverage, _rel_path)
|
|
27
|
+
from .errors import ToolError
|
|
28
|
+
|
|
29
|
+
CHUNK = 1 << 20
|
|
30
|
+
|
|
31
|
+
# The inner close: one object ending inside a larger document, with no claim
|
|
32
|
+
# about what follows. _CLOSE_RE anchors at the end of the text and is the outer
|
|
33
|
+
# document's business.
|
|
34
|
+
_CLOSE_INNER = re.compile(r"\s*\}")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class _Window:
|
|
38
|
+
"""A sliding decoded window over a byte stream, plus the sha256 of the bytes
|
|
39
|
+
that went past. Offsets stay valid across a refill because refilling only
|
|
40
|
+
appends; only drop() ever moves them, and it says so."""
|
|
41
|
+
|
|
42
|
+
def __init__(self, handle: IO[bytes], chunk: int = CHUNK):
|
|
43
|
+
self._handle = handle
|
|
44
|
+
self._chunk = max(chunk, 1)
|
|
45
|
+
self._decoder = codecs.getincrementaldecoder("utf-8")()
|
|
46
|
+
self.hasher = hashlib.sha256()
|
|
47
|
+
self.buf = ""
|
|
48
|
+
self.pos = 0
|
|
49
|
+
self.eof = False
|
|
50
|
+
|
|
51
|
+
def refill(self) -> bool:
|
|
52
|
+
"""Pull one more chunk into the window. False once the stream is spent."""
|
|
53
|
+
if self.eof:
|
|
54
|
+
return False
|
|
55
|
+
raw = self._handle.read(self._chunk)
|
|
56
|
+
if not raw:
|
|
57
|
+
self.eof = True
|
|
58
|
+
self.buf += self._decoder.decode(b"", True)
|
|
59
|
+
return False
|
|
60
|
+
self.hasher.update(raw)
|
|
61
|
+
self.buf += self._decoder.decode(raw)
|
|
62
|
+
return True
|
|
63
|
+
|
|
64
|
+
def drop(self, i: int) -> None:
|
|
65
|
+
"""Consume through offset `i`. Compaction is amortized: slicing the
|
|
66
|
+
window on every member copies the whole tail each time, so the offset
|
|
67
|
+
moves and the copy happens only once the consumed prefix is a chunk."""
|
|
68
|
+
self.pos = i
|
|
69
|
+
if self.pos >= self._chunk:
|
|
70
|
+
self.buf = self.buf[self.pos:]
|
|
71
|
+
self.pos = 0
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
# --- window-driven splitting ----------------------------------------------
|
|
75
|
+
|
|
76
|
+
def _usable(w: _Window, member) -> bool:
|
|
77
|
+
"""A member header the window can be trusted on. One that runs to the very
|
|
78
|
+
edge is not trustworthy mid-stream: the key string, or the whitespace after
|
|
79
|
+
the colon, may continue in bytes not read yet."""
|
|
80
|
+
return member is not None and (member.end() < len(w.buf) or w.eof)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _next_member(w: _Window):
|
|
84
|
+
"""The next member header, or None once the object closed or the stream ran
|
|
85
|
+
out. Refills only while the window can neither produce a header nor prove
|
|
86
|
+
the object ended, so a closing brace does not drag the rest of the file in."""
|
|
87
|
+
while True:
|
|
88
|
+
member = _MEMBER_RE.match(w.buf, w.pos)
|
|
89
|
+
if _usable(w, member):
|
|
90
|
+
return member
|
|
91
|
+
if _CLOSE_INNER.match(w.buf, w.pos) is not None:
|
|
92
|
+
return None
|
|
93
|
+
if not w.refill():
|
|
94
|
+
return None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _whole_value(w: _Window, start: int):
|
|
98
|
+
"""The decoded value and its end offset, or None while the window may still
|
|
99
|
+
be hiding more of it. A value ending exactly at the edge is not whole: a
|
|
100
|
+
bare number would otherwise decode as its own truncated prefix."""
|
|
101
|
+
try:
|
|
102
|
+
value, end = _DECODER.raw_decode(w.buf, start)
|
|
103
|
+
except ValueError:
|
|
104
|
+
return None
|
|
105
|
+
return (value, end) if end < len(w.buf) or w.eof else None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _decode_value(w: _Window, start: int):
|
|
109
|
+
"""raw_decode at `start`, growing the window until the value is whole."""
|
|
110
|
+
while True:
|
|
111
|
+
whole = _whole_value(w, start)
|
|
112
|
+
if whole is not None:
|
|
113
|
+
return whole
|
|
114
|
+
if not w.refill():
|
|
115
|
+
return _DECODER.raw_decode(w.buf, start)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _enter_object(w: _Window, what: str) -> None:
|
|
119
|
+
while _OPEN_RE.match(w.buf, w.pos) is None and w.refill():
|
|
120
|
+
pass
|
|
121
|
+
opening = _OPEN_RE.match(w.buf, w.pos)
|
|
122
|
+
if opening is None:
|
|
123
|
+
raise ValueError(f"{what} is not a JSON object")
|
|
124
|
+
w.drop(opening.end())
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _expect_document_end(w: _Window) -> None:
|
|
128
|
+
while w.refill():
|
|
129
|
+
pass
|
|
130
|
+
if _CLOSE_RE.match(w.buf, w.pos) is None:
|
|
131
|
+
raise ValueError(f"unexpected content at {w.buf[w.pos:w.pos + 80]!r}")
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _take_member(w: _Window, member) -> tuple[str, object]:
|
|
135
|
+
"""Decode one member's value and step the window past it."""
|
|
136
|
+
key, start = member.group(1), member.end()
|
|
137
|
+
value, end = _decode_value(w, start)
|
|
138
|
+
w.drop(end)
|
|
139
|
+
return json.loads(key), value
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def split_window(w: _Window) -> Iterator[tuple[str, object]]:
|
|
143
|
+
"""(key, value) per member of the outer object, one value live at a time.
|
|
144
|
+
The same pairs in the same order as coverage_istanbul.split_top_level."""
|
|
145
|
+
_enter_object(w, "istanbul artifact")
|
|
146
|
+
while True:
|
|
147
|
+
member = _next_member(w)
|
|
148
|
+
if member is None:
|
|
149
|
+
_expect_document_end(w)
|
|
150
|
+
return
|
|
151
|
+
yield _take_member(w, member)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
# --- coverage.py: the same walk, one level down ---------------------------
|
|
155
|
+
|
|
156
|
+
def _leave_object(w: _Window, what: str) -> None:
|
|
157
|
+
close = _CLOSE_INNER.match(w.buf, w.pos)
|
|
158
|
+
if close is None:
|
|
159
|
+
raise ValueError(f"unterminated {what}")
|
|
160
|
+
w.drop(close.end())
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _walk_nested(w: _Window, start: int) -> Iterator[tuple[str, object, str]]:
|
|
164
|
+
"""The members of the object at `start`, one at a time."""
|
|
165
|
+
w.pos = start
|
|
166
|
+
_enter_object(w, "coverage.py report: 'files'")
|
|
167
|
+
while True:
|
|
168
|
+
member = _next_member(w)
|
|
169
|
+
if member is None:
|
|
170
|
+
_leave_object(w, "coverage.py report: 'files' object")
|
|
171
|
+
return
|
|
172
|
+
key, value = _take_member(w, member)
|
|
173
|
+
yield key, value, "sub"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def walk_report(w: _Window, target: str) -> Iterator[tuple[str, object, str]]:
|
|
177
|
+
"""(key, value, kind) per top-level member. kind is "member" for an ordinary
|
|
178
|
+
decoded value and "sub" for one member of the `target` object, so meta and
|
|
179
|
+
totals arrive whole and "files" arrives one file at a time."""
|
|
180
|
+
_enter_object(w, "coverage.py report")
|
|
181
|
+
while True:
|
|
182
|
+
member = _next_member(w)
|
|
183
|
+
if member is None:
|
|
184
|
+
# A walk that just stops at the first unreadable byte reports zero
|
|
185
|
+
# dark lines, which is indistinguishable from a fully covered repo.
|
|
186
|
+
_expect_document_end(w)
|
|
187
|
+
return
|
|
188
|
+
if json.loads(member.group(1)) == target:
|
|
189
|
+
yield from _walk_nested(w, member.end())
|
|
190
|
+
continue
|
|
191
|
+
key, value = _take_member(w, member)
|
|
192
|
+
yield key, value, "member"
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
# --- public readers --------------------------------------------------------
|
|
196
|
+
|
|
197
|
+
def _window(path: Path | str, chunk: int) -> tuple[_Window, IO[bytes]]:
|
|
198
|
+
handle = open(path, "rb")
|
|
199
|
+
return _Window(handle, chunk), handle
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _guarded(work, message: str):
|
|
203
|
+
"""Run a walk, reporting any parse failure the way the whole-document
|
|
204
|
+
parsers do. A ToolError the walk raised itself is already the right error
|
|
205
|
+
and keeps its own wording."""
|
|
206
|
+
try:
|
|
207
|
+
return work()
|
|
208
|
+
except ToolError:
|
|
209
|
+
raise
|
|
210
|
+
except Exception as exc:
|
|
211
|
+
raise ToolError(f"{message}: {exc}") from exc
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
_BAD_ISTANBUL = "unparseable istanbul artifact"
|
|
215
|
+
_BAD_REPORT = "unparseable coverage.py report"
|
|
216
|
+
_NO_BRANCH = "coverage.py report lacks branch data — run the lane with branch coverage on"
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _istanbul_map(w: _Window, repo_root: str, per_file) -> dict:
|
|
220
|
+
out = {}
|
|
221
|
+
for abs_path, cov in split_window(w):
|
|
222
|
+
out[_rel_path(abs_path, repo_root)] = per_file(cov)
|
|
223
|
+
return out
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def parse_istanbul_file(path: Path | str, *, repo_root: str, chunk: int = CHUNK
|
|
227
|
+
) -> tuple[dict[str, list[FnCoverage]], str]:
|
|
228
|
+
"""Per-file function coverage plus the sha256 of the artifact's own bytes.
|
|
229
|
+
Same result as parse_istanbul(path.read_text(), ...), same digest as
|
|
230
|
+
sha256(path.read_bytes()), without either whole copy ever existing."""
|
|
231
|
+
w, handle = _window(path, chunk)
|
|
232
|
+
with handle:
|
|
233
|
+
per_file = _guarded(lambda: _istanbul_map(w, repo_root, _file_coverage),
|
|
234
|
+
_BAD_ISTANBUL)
|
|
235
|
+
if not per_file:
|
|
236
|
+
raise ToolError(
|
|
237
|
+
"istanbul artifact is empty (zero files) — the coverage run measured nothing")
|
|
238
|
+
return per_file, w.hasher.hexdigest()
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def parse_istanbul_missing_file(path: Path | str, *, repo_root: str,
|
|
242
|
+
chunk: int = CHUNK) -> dict[str, set[int]]:
|
|
243
|
+
"""Per measured file, the lines whose statement never ran."""
|
|
244
|
+
w, handle = _window(path, chunk)
|
|
245
|
+
with handle:
|
|
246
|
+
return _guarded(lambda: _istanbul_map(w, repo_root, _dead_lines), _BAD_ISTANBUL)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _prefix(path_prefix: str) -> str:
|
|
250
|
+
return (path_prefix.rstrip("/") + "/") if path_prefix else ""
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _meta_has_branch(key: str, value: object) -> bool:
|
|
254
|
+
if key != "meta" or not isinstance(value, dict):
|
|
255
|
+
return False
|
|
256
|
+
return bool(value.get("branch_coverage"))
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _require_branch(seen: bool) -> None:
|
|
260
|
+
if not seen:
|
|
261
|
+
raise ToolError(_NO_BRANCH)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _add_file(out: dict, prefix: str, raw_path: str, data: dict, to_functions):
|
|
265
|
+
"""Record one file's functions, or hand back the error it raised.
|
|
266
|
+
|
|
267
|
+
The error is CARRIED, not thrown: the whole-document parser saw the report
|
|
268
|
+
before it read a single file, so it always refused a report with no branch
|
|
269
|
+
data first. Members are not ordered — json.dump(sort_keys=True) writes
|
|
270
|
+
"files" ahead of "meta" — so streaming only keeps that precedence by
|
|
271
|
+
finishing the walk before it decides which complaint wins.
|
|
272
|
+
"""
|
|
273
|
+
try:
|
|
274
|
+
out[prefix + raw_path.replace("\\", "/")] = to_functions(raw_path, data)
|
|
275
|
+
return None
|
|
276
|
+
except Exception as exc:
|
|
277
|
+
return exc
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _coveragepy_functions(w: _Window, prefix: str) -> dict:
|
|
281
|
+
"""path -> function coverage, refusing a report with no branch data."""
|
|
282
|
+
from .coverage_py import _file_functions
|
|
283
|
+
|
|
284
|
+
out: dict = {}
|
|
285
|
+
branch, failure = False, None
|
|
286
|
+
for key, value, kind in walk_report(w, "files"):
|
|
287
|
+
if kind == "member":
|
|
288
|
+
branch = branch or _meta_has_branch(key, value)
|
|
289
|
+
continue
|
|
290
|
+
failure = failure or _add_file(out, prefix, key, value, _file_functions)
|
|
291
|
+
_require_branch(branch)
|
|
292
|
+
if failure is not None:
|
|
293
|
+
raise failure
|
|
294
|
+
return out
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def parse_coveragepy_file(path: Path | str, *, path_prefix: str, chunk: int = CHUNK
|
|
298
|
+
) -> tuple[dict[str, list[FnCoverage]], str]:
|
|
299
|
+
"""Per-file function coverage plus the sha256 of the report's own bytes."""
|
|
300
|
+
w, handle = _window(path, chunk)
|
|
301
|
+
with handle:
|
|
302
|
+
per_file = _guarded(lambda: _coveragepy_functions(w, _prefix(path_prefix)),
|
|
303
|
+
_BAD_REPORT)
|
|
304
|
+
return per_file, w.hasher.hexdigest()
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _coveragepy_missing(w: _Window, prefix: str) -> dict[str, set[int]]:
|
|
308
|
+
out: dict[str, set[int]] = {}
|
|
309
|
+
for key, value, kind in walk_report(w, "files"):
|
|
310
|
+
if kind == "sub":
|
|
311
|
+
out[prefix + key.replace("\\", "/")] = set(value.get("missing_lines", ()))
|
|
312
|
+
return out
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def parse_coveragepy_missing_file(path: Path | str, *, path_prefix: str,
|
|
316
|
+
chunk: int = CHUNK) -> dict[str, set[int]]:
|
|
317
|
+
"""Per measured file, the lines coverage.py reports as never run."""
|
|
318
|
+
w, handle = _window(path, chunk)
|
|
319
|
+
with handle:
|
|
320
|
+
return _guarded(lambda: _coveragepy_missing(w, _prefix(path_prefix)), _BAD_REPORT)
|
crapkit/diffparse.py
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Parse `git diff -U0` output into per-file changed line ranges. Pure.
|
|
2
|
+
|
|
3
|
+
Ranges are new-side. A pure deletion (zero new lines) still marks the line it
|
|
4
|
+
happened at, so a function shrunk by an edit is still a touched function.
|
|
5
|
+
Paths come from the +++ header (the new side survives renames) and decode
|
|
6
|
+
git's C-style quoting via the churn module's helper.
|
|
7
|
+
|
|
8
|
+
Hunk body lines are consumed by the counts the @@ header declares, never
|
|
9
|
+
pattern-matched: an added source line whose text starts with "++ " arrives as
|
|
10
|
+
"+++ <text>" and would otherwise read as a file header, repointing every later
|
|
11
|
+
hunk at a phantom path the gate never checks.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
|
|
17
|
+
from .churn import _unquote_git_path
|
|
18
|
+
|
|
19
|
+
_HUNK = re.compile(r"^@@ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _spend_body_line(line: str, rem_old: int, rem_new: int) -> tuple[int, int] | None:
|
|
23
|
+
"""Charge one line to the hunk body still owed, returning what remains.
|
|
24
|
+
|
|
25
|
+
None means the line is diff structure, not body: either nothing is owed or
|
|
26
|
+
the hunk ended early.
|
|
27
|
+
"""
|
|
28
|
+
if rem_old <= 0 and rem_new <= 0:
|
|
29
|
+
return None
|
|
30
|
+
if line.startswith("\\"): # ""
|
|
31
|
+
return rem_old, rem_new
|
|
32
|
+
if line.startswith("-"):
|
|
33
|
+
return rem_old - 1, rem_new
|
|
34
|
+
if line.startswith("+"):
|
|
35
|
+
return rem_old, rem_new - 1
|
|
36
|
+
# -U0 hunks contain only +/- lines; anything else means the hunk ended
|
|
37
|
+
# early (truncated or synthetic diff) - reparse it normally.
|
|
38
|
+
return None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _open_file(line: str, ranges: dict[str, list[tuple[int, int]]]) -> str | None:
|
|
42
|
+
"""Point the parser at the file a `+++ ` header names; None for /dev/null."""
|
|
43
|
+
target = line[4:].strip()
|
|
44
|
+
if target == "/dev/null":
|
|
45
|
+
return None
|
|
46
|
+
target = _unquote_git_path(target)
|
|
47
|
+
path = target[2:] if target.startswith("b/") else target
|
|
48
|
+
path = path.replace("\\", "/")
|
|
49
|
+
ranges.setdefault(path, [])
|
|
50
|
+
return path
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _new_side_range(start: int, count: int) -> tuple[int, int]:
|
|
54
|
+
"""New-side span a hunk covers."""
|
|
55
|
+
if count == 0: # pure deletion: mark the touch point
|
|
56
|
+
return max(start, 1), max(start, 1)
|
|
57
|
+
return start, start + count - 1
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _record_hunk(
|
|
61
|
+
m: re.Match[str], current: str | None, ranges: dict[str, list[tuple[int, int]]]
|
|
62
|
+
) -> tuple[int, int]:
|
|
63
|
+
"""Land the hunk's range on the current file and return the body counts it owes.
|
|
64
|
+
|
|
65
|
+
An omitted count in the @@ header means one line. A hunk with no current
|
|
66
|
+
file still declares a body to consume.
|
|
67
|
+
"""
|
|
68
|
+
rem_old = int(m.group(2)) if m.group(2) is not None else 1
|
|
69
|
+
rem_new = int(m.group(4)) if m.group(4) is not None else 1
|
|
70
|
+
if current is not None:
|
|
71
|
+
ranges[current].append(_new_side_range(int(m.group(3)), rem_new))
|
|
72
|
+
return rem_old, rem_new
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _files_with_hunks(
|
|
76
|
+
ranges: dict[str, list[tuple[int, int]]],
|
|
77
|
+
) -> dict[str, list[tuple[int, int]]]:
|
|
78
|
+
"""Drop files a header introduced but no hunk ever touched."""
|
|
79
|
+
return {p: r for p, r in ranges.items() if r}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def changed_ranges(diff_text: str) -> dict[str, list[tuple[int, int]]]:
|
|
83
|
+
ranges: dict[str, list[tuple[int, int]]] = {}
|
|
84
|
+
current: str | None = None
|
|
85
|
+
rem_old = rem_new = 0
|
|
86
|
+
for line in diff_text.splitlines():
|
|
87
|
+
owed = _spend_body_line(line, rem_old, rem_new)
|
|
88
|
+
if owed is not None:
|
|
89
|
+
rem_old, rem_new = owed
|
|
90
|
+
continue
|
|
91
|
+
rem_old = rem_new = 0
|
|
92
|
+
if line.startswith("+++ "):
|
|
93
|
+
current = _open_file(line, ranges)
|
|
94
|
+
continue
|
|
95
|
+
m = _HUNK.match(line)
|
|
96
|
+
if m:
|
|
97
|
+
rem_old, rem_new = _record_hunk(m, current, ranges)
|
|
98
|
+
return _files_with_hunks(ranges)
|
crapkit/digest.py
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""Digest and trend totals. Pure.
|
|
2
|
+
|
|
3
|
+
The digest speaks only on change (an unchanged week is silence, per the
|
|
4
|
+
empty-channel rule) and keeps to a handful of numbers: totals delta, the worst
|
|
5
|
+
per-function regressions, improvements, and new over-target functions.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import NamedTuple
|
|
10
|
+
|
|
11
|
+
from .score import ScoredRow, grade
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Totals(NamedTuple):
|
|
15
|
+
functions: int
|
|
16
|
+
over_target: int
|
|
17
|
+
crap_load: float
|
|
18
|
+
avg: float
|
|
19
|
+
pct_over: float # normalized: absolute counts grow with the repo; this does not
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class Digest(NamedTuple):
|
|
23
|
+
quiet: bool
|
|
24
|
+
lines: list[str]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _over_count(rows, target: int, scope_targets) -> int:
|
|
28
|
+
return sum(1 for r in rows if r.crap > (scope_targets or {}).get(r.scope, target))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def totals_from_counts(functions: int, over_target: int, load: float) -> Totals:
|
|
32
|
+
"""The rounding rule, in one place. A caller that already has the three sums
|
|
33
|
+
(store.run_totals adds them up inside the scan) must round them exactly the
|
|
34
|
+
way a caller holding the rows does, or the same run reads two ways."""
|
|
35
|
+
return Totals(
|
|
36
|
+
functions=functions,
|
|
37
|
+
over_target=over_target,
|
|
38
|
+
crap_load=round(load, 2),
|
|
39
|
+
avg=round(load / functions, 4) if functions else 0.0,
|
|
40
|
+
pct_over=round(100.0 * over_target / functions, 2) if functions else 0.0,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def totals(rows: list[ScoredRow], *, target: int,
|
|
45
|
+
scope_targets: dict[str, int] | None = None) -> Totals:
|
|
46
|
+
return totals_from_counts(len(rows), _over_count(rows, target, scope_targets),
|
|
47
|
+
sum(r.crap for r in rows))
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def scope_totals(rows: list[ScoredRow], *, target: int,
|
|
51
|
+
scope_targets: dict[str, int] | None = None) -> dict[str, Totals]:
|
|
52
|
+
by_scope: dict[str, list[ScoredRow]] = {}
|
|
53
|
+
for r in rows:
|
|
54
|
+
by_scope.setdefault(r.scope, []).append(r)
|
|
55
|
+
return {scope: totals(srows, target=target, scope_targets=scope_targets)
|
|
56
|
+
for scope, srows in sorted(by_scope.items())}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def scope_rollup(by_scope: dict[str, Totals]) -> dict[str, dict]:
|
|
60
|
+
"""The per-scope block coverage --json and trend --json both publish.
|
|
61
|
+
|
|
62
|
+
One shaping in one place: the two commands reach their Totals differently
|
|
63
|
+
(rows in hand vs a GROUP BY), and a second shaping would let the same run
|
|
64
|
+
read two ways depending on which command asked.
|
|
65
|
+
"""
|
|
66
|
+
return {scope: {"functions": t.functions, "over_target": t.over_target,
|
|
67
|
+
"crap_load": t.crap_load, "grade": grade(t.over_target, t.functions)}
|
|
68
|
+
for scope, t in by_scope.items()}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def latest_comparable_pair(runs: list[dict]) -> tuple[dict, dict] | None:
|
|
72
|
+
"""The newest pair of runs with IDENTICAL lane sets.
|
|
73
|
+
|
|
74
|
+
Comparing a --lane subset run against a full run flips every out-of-subset
|
|
75
|
+
function to no-lane and manufactures phantom regressions; only like-for-like
|
|
76
|
+
pairs digest.
|
|
77
|
+
"""
|
|
78
|
+
for i in range(len(runs) - 1, 0, -1):
|
|
79
|
+
cur = runs[i]
|
|
80
|
+
cur_lanes = set(cur["lanes"])
|
|
81
|
+
for j in range(i - 1, -1, -1):
|
|
82
|
+
if set(runs[j]["lanes"]) == cur_lanes:
|
|
83
|
+
return runs[j], cur
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def skipped_runs(runs: list[dict], pair: tuple[dict, dict] | None) -> list[dict]:
|
|
88
|
+
"""The runs the pair stepped over: everything after its older half except
|
|
89
|
+
its newer half. Empty exactly when the pair is the newest run and the one
|
|
90
|
+
before it.
|
|
91
|
+
|
|
92
|
+
Stepping over a run is the right move — a subset run never compares — but a
|
|
93
|
+
digest that says nothing about it reports an older delta in this week's
|
|
94
|
+
voice. On the flagship consumer that printed runs 17 -> 19 while run 22 sat
|
|
95
|
+
in the store, and nothing on any surface said 22 existed.
|
|
96
|
+
"""
|
|
97
|
+
if pair is None:
|
|
98
|
+
return []
|
|
99
|
+
prev, cur = pair
|
|
100
|
+
return [r for r in runs[runs.index(prev) + 1:] if r["id"] != cur["id"]]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
_Key = tuple[str, str] # (path, long_name): what identifies a function across runs
|
|
104
|
+
_Move = tuple[float, ScoredRow, ScoredRow] # (delta, before, after)
|
|
105
|
+
_Delta = tuple[float, ScoredRow]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class _Changes(NamedTuple):
|
|
109
|
+
"""What moved between two runs, bucketed the way the digest reports it."""
|
|
110
|
+
regressions: list[_Delta]
|
|
111
|
+
improvements: list[_Delta]
|
|
112
|
+
appeared: list[ScoredRow]
|
|
113
|
+
|
|
114
|
+
def nothing_moved(self) -> bool:
|
|
115
|
+
return not (self.regressions or self.improvements or self.appeared)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _by_key(rows: list[ScoredRow]) -> dict[_Key, ScoredRow]:
|
|
119
|
+
return {(r.path, r.long_name): r for r in rows}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _moves(prev_by_key: dict[_Key, ScoredRow],
|
|
123
|
+
cur_by_key: dict[_Key, ScoredRow]) -> list[_Move]:
|
|
124
|
+
"""Every function measured in both runs, paired with how far its CRAP moved."""
|
|
125
|
+
return [(row.crap - prev_by_key[key].crap, prev_by_key[key], row)
|
|
126
|
+
for key, row in cur_by_key.items() if key in prev_by_key]
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _regressions(moves: list[_Move]) -> list[_Delta]:
|
|
130
|
+
return [(delta, after) for delta, _before, after in moves if delta > 0.01]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _improvements(moves: list[_Move], target: int) -> list[_Delta]:
|
|
134
|
+
"""Only a function that WAS over target improves; drift below it is not news."""
|
|
135
|
+
return [(delta, after) for delta, before, after in moves
|
|
136
|
+
if delta < -0.01 and before.crap > target]
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _appeared(prev_by_key: dict[_Key, ScoredRow], cur_by_key: dict[_Key, ScoredRow],
|
|
140
|
+
target: int) -> list[ScoredRow]:
|
|
141
|
+
"""Functions the previous run never saw; new code under target is not news."""
|
|
142
|
+
return [row for key, row in cur_by_key.items()
|
|
143
|
+
if key not in prev_by_key and row.crap > target]
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _changes_between(prev: list[ScoredRow], cur: list[ScoredRow], target: int) -> _Changes:
|
|
147
|
+
prev_by_key, cur_by_key = _by_key(prev), _by_key(cur)
|
|
148
|
+
moves = _moves(prev_by_key, cur_by_key)
|
|
149
|
+
return _Changes(
|
|
150
|
+
regressions=_regressions(moves),
|
|
151
|
+
improvements=_improvements(moves, target),
|
|
152
|
+
appeared=_appeared(prev_by_key, cur_by_key, target),
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _totals_line(t_prev: Totals, t_cur: Totals) -> str:
|
|
157
|
+
return (f"CRAP load {t_prev.crap_load} -> {t_cur.crap_load}; "
|
|
158
|
+
f"over target {t_prev.over_target} -> {t_cur.over_target}; "
|
|
159
|
+
f"functions {t_prev.functions} -> {t_cur.functions}")
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _fn_line(prefix: str, row: ScoredRow) -> str:
|
|
163
|
+
return f"{prefix}: {row.path} {row.long_name} (crap {row.crap:.1f})"
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _regression_lines(regressions: list[_Delta], top: int) -> list[str]:
|
|
167
|
+
return [_fn_line(f"regressed +{delta:.1f}", row)
|
|
168
|
+
for delta, row in sorted(regressions, key=lambda x: -x[0])[:top]]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _appeared_lines(appeared: list[ScoredRow], top: int) -> list[str]:
|
|
172
|
+
return [_fn_line("new over target", row)
|
|
173
|
+
for row in sorted(appeared, key=lambda r: -r.crap)[:top]]
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _improvement_lines(improvements: list[_Delta], top: int) -> list[str]:
|
|
177
|
+
return [_fn_line(f"improved {delta:.1f}", row)
|
|
178
|
+
for delta, row in sorted(improvements, key=lambda x: x[0])[:top]]
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def build_digest(prev: list[ScoredRow], cur: list[ScoredRow], *, target: int, top: int = 5) -> Digest:
|
|
182
|
+
t_prev, t_cur = totals(prev, target=target), totals(cur, target=target)
|
|
183
|
+
changed = _changes_between(prev, cur, target)
|
|
184
|
+
if t_prev == t_cur and changed.nothing_moved():
|
|
185
|
+
return Digest(quiet=True, lines=[])
|
|
186
|
+
return Digest(quiet=False, lines=[
|
|
187
|
+
_totals_line(t_prev, t_cur),
|
|
188
|
+
*_regression_lines(changed.regressions, top),
|
|
189
|
+
*_appeared_lines(changed.appeared, top),
|
|
190
|
+
*_improvement_lines(changed.improvements, top),
|
|
191
|
+
])
|