crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/coverage_py.py ADDED
@@ -0,0 +1,87 @@
1
+ """coverage.py JSON report parser. Pure: report text in, per-file function coverage out.
2
+
3
+ Requires the per-function regions coverage.py has emitted since 7.6.0 and the
4
+ start_line key from 7.13.1; requires branch data (the consuming lane must run
5
+ with branch coverage on) so the coverage term measures the same structure the
6
+ complexity term counts. Function spans come from start_line and the maximum
7
+ executed/missing line, the closest thing the report offers to an end line.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import json
12
+
13
+ from .coverage_istanbul import FnCoverage
14
+ from .errors import ToolError
15
+
16
+
17
+ def _fn_coverage(name: str, fn: dict) -> FnCoverage:
18
+ summary = fn.get("summary", {})
19
+ lines = list(fn.get("executed_lines", ())) + list(fn.get("missing_lines", ()))
20
+ start = fn.get("start_line") or (min(lines) if lines else 0)
21
+ end = max(lines) if lines else start
22
+ return FnCoverage(name=name, start=start, end=end,
23
+ invoked=summary.get("covered_lines", 0) > 0,
24
+ branches_total=summary.get("num_branches", 0),
25
+ branches_covered=summary.get("covered_branches", 0),
26
+ statements_total=summary.get("num_statements", 0),
27
+ statements_covered=summary.get("covered_lines", 0))
28
+
29
+
30
+ def _file_functions(raw_path: str, data: dict) -> list[FnCoverage]:
31
+ functions = data.get("functions")
32
+ if functions is None:
33
+ raise ToolError(
34
+ f"coverage.py report has no function regions for {raw_path!r} — needs coverage >= 7.6")
35
+ # the "" key is the "(no function)" module-level bucket
36
+ fns = [_fn_coverage(name, fn) for name, fn in functions.items() if name]
37
+ return sorted(fns, key=lambda f: f.start)
38
+
39
+
40
+ def parse_coveragepy_missing(text: str, *, path_prefix: str) -> dict[str, set[int]]:
41
+ """Per measured file, the lines coverage.py reports as never run."""
42
+ try:
43
+ report = json.loads(text)
44
+ prefix = (path_prefix.rstrip("/") + "/") if path_prefix else ""
45
+ return {prefix + p.replace("\\", "/"): set(data.get("missing_lines", ()))
46
+ for p, data in report.get("files", {}).items()}
47
+ except Exception as exc:
48
+ raise ToolError(f"unparseable coverage.py report: {exc}") from exc
49
+
50
+
51
+ def _line_contexts(raw: dict) -> dict[int, list[str]]:
52
+ out = {}
53
+ for line, contexts in raw.items():
54
+ tests = sorted({c.split("|")[0] for c in contexts if c})
55
+ if tests:
56
+ out[int(line)] = tests
57
+ return out
58
+
59
+
60
+ def parse_coveragepy_contexts(text: str, *, path_prefix: str) -> dict[str, dict[int, list[str]]]:
61
+ """line -> test ids per file, from a report made with --show-contexts and
62
+ dynamic_context = test_function. The empty module-import context is not a test."""
63
+ try:
64
+ report = json.loads(text)
65
+ prefix = (path_prefix.rstrip("/") + "/") if path_prefix else ""
66
+ out = {}
67
+ for p, data in report.get("files", {}).items():
68
+ contexts = _line_contexts(data.get("contexts", {}))
69
+ if contexts:
70
+ out[prefix + p.replace("\\", "/")] = contexts
71
+ return out
72
+ except Exception as exc:
73
+ raise ToolError(f"unparseable coverage.py report: {exc}") from exc
74
+
75
+
76
+ def parse_coveragepy(text: str, *, path_prefix: str) -> dict[str, list[FnCoverage]]:
77
+ try:
78
+ report = json.loads(text)
79
+ if not report.get("meta", {}).get("branch_coverage"):
80
+ raise ToolError("coverage.py report lacks branch data — run the lane with branch coverage on")
81
+ prefix = (path_prefix.rstrip("/") + "/") if path_prefix else ""
82
+ return {prefix + raw_path.replace("\\", "/"): _file_functions(raw_path, data)
83
+ for raw_path, data in report.get("files", {}).items()}
84
+ except ToolError:
85
+ raise
86
+ except Exception as exc:
87
+ raise ToolError(f"unparseable coverage.py report: {exc}") from exc
crapkit/covstream.py ADDED
@@ -0,0 +1,320 @@
1
+ """Coverage artifacts read off the file instead of out of a string.
2
+
3
+ The whole-document parsers take `text`, so the caller must already hold the
4
+ artifact: `read_bytes()` plus its UTF-8 decode put two copies of a 150 MB
5
+ artifact on the heap before a single function is attributed. The splitter in
6
+ coverage_istanbul already decodes one member at a time; this module gives it a
7
+ window that refills from a handle instead of a string that holds everything.
8
+
9
+ Peak becomes O(chunk + largest member) rather than O(artifact): 322.6 -> 52.1 MB
10
+ on a 150 MB istanbul artifact, for byte-identical output and the same sha256.
11
+
12
+ Both shapes are split the same way. An istanbul artifact IS the {path: coverage}
13
+ object, so its members are files. A coverage.py report wraps them one level down
14
+ in "files", so the walk descends into that member and hands the rest back whole.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import codecs
19
+ import hashlib
20
+ import json
21
+ import re
22
+ from pathlib import Path
23
+ from typing import IO, Iterator
24
+
25
+ from .coverage_istanbul import (_CLOSE_RE, _DECODER, _MEMBER_RE, _OPEN_RE,
26
+ FnCoverage, _dead_lines, _file_coverage, _rel_path)
27
+ from .errors import ToolError
28
+
29
+ CHUNK = 1 << 20
30
+
31
+ # The inner close: one object ending inside a larger document, with no claim
32
+ # about what follows. _CLOSE_RE anchors at the end of the text and is the outer
33
+ # document's business.
34
+ _CLOSE_INNER = re.compile(r"\s*\}")
35
+
36
+
37
+ class _Window:
38
+ """A sliding decoded window over a byte stream, plus the sha256 of the bytes
39
+ that went past. Offsets stay valid across a refill because refilling only
40
+ appends; only drop() ever moves them, and it says so."""
41
+
42
+ def __init__(self, handle: IO[bytes], chunk: int = CHUNK):
43
+ self._handle = handle
44
+ self._chunk = max(chunk, 1)
45
+ self._decoder = codecs.getincrementaldecoder("utf-8")()
46
+ self.hasher = hashlib.sha256()
47
+ self.buf = ""
48
+ self.pos = 0
49
+ self.eof = False
50
+
51
+ def refill(self) -> bool:
52
+ """Pull one more chunk into the window. False once the stream is spent."""
53
+ if self.eof:
54
+ return False
55
+ raw = self._handle.read(self._chunk)
56
+ if not raw:
57
+ self.eof = True
58
+ self.buf += self._decoder.decode(b"", True)
59
+ return False
60
+ self.hasher.update(raw)
61
+ self.buf += self._decoder.decode(raw)
62
+ return True
63
+
64
+ def drop(self, i: int) -> None:
65
+ """Consume through offset `i`. Compaction is amortized: slicing the
66
+ window on every member copies the whole tail each time, so the offset
67
+ moves and the copy happens only once the consumed prefix is a chunk."""
68
+ self.pos = i
69
+ if self.pos >= self._chunk:
70
+ self.buf = self.buf[self.pos:]
71
+ self.pos = 0
72
+
73
+
74
+ # --- window-driven splitting ----------------------------------------------
75
+
76
+ def _usable(w: _Window, member) -> bool:
77
+ """A member header the window can be trusted on. One that runs to the very
78
+ edge is not trustworthy mid-stream: the key string, or the whitespace after
79
+ the colon, may continue in bytes not read yet."""
80
+ return member is not None and (member.end() < len(w.buf) or w.eof)
81
+
82
+
83
+ def _next_member(w: _Window):
84
+ """The next member header, or None once the object closed or the stream ran
85
+ out. Refills only while the window can neither produce a header nor prove
86
+ the object ended, so a closing brace does not drag the rest of the file in."""
87
+ while True:
88
+ member = _MEMBER_RE.match(w.buf, w.pos)
89
+ if _usable(w, member):
90
+ return member
91
+ if _CLOSE_INNER.match(w.buf, w.pos) is not None:
92
+ return None
93
+ if not w.refill():
94
+ return None
95
+
96
+
97
+ def _whole_value(w: _Window, start: int):
98
+ """The decoded value and its end offset, or None while the window may still
99
+ be hiding more of it. A value ending exactly at the edge is not whole: a
100
+ bare number would otherwise decode as its own truncated prefix."""
101
+ try:
102
+ value, end = _DECODER.raw_decode(w.buf, start)
103
+ except ValueError:
104
+ return None
105
+ return (value, end) if end < len(w.buf) or w.eof else None
106
+
107
+
108
+ def _decode_value(w: _Window, start: int):
109
+ """raw_decode at `start`, growing the window until the value is whole."""
110
+ while True:
111
+ whole = _whole_value(w, start)
112
+ if whole is not None:
113
+ return whole
114
+ if not w.refill():
115
+ return _DECODER.raw_decode(w.buf, start)
116
+
117
+
118
+ def _enter_object(w: _Window, what: str) -> None:
119
+ while _OPEN_RE.match(w.buf, w.pos) is None and w.refill():
120
+ pass
121
+ opening = _OPEN_RE.match(w.buf, w.pos)
122
+ if opening is None:
123
+ raise ValueError(f"{what} is not a JSON object")
124
+ w.drop(opening.end())
125
+
126
+
127
+ def _expect_document_end(w: _Window) -> None:
128
+ while w.refill():
129
+ pass
130
+ if _CLOSE_RE.match(w.buf, w.pos) is None:
131
+ raise ValueError(f"unexpected content at {w.buf[w.pos:w.pos + 80]!r}")
132
+
133
+
134
+ def _take_member(w: _Window, member) -> tuple[str, object]:
135
+ """Decode one member's value and step the window past it."""
136
+ key, start = member.group(1), member.end()
137
+ value, end = _decode_value(w, start)
138
+ w.drop(end)
139
+ return json.loads(key), value
140
+
141
+
142
+ def split_window(w: _Window) -> Iterator[tuple[str, object]]:
143
+ """(key, value) per member of the outer object, one value live at a time.
144
+ The same pairs in the same order as coverage_istanbul.split_top_level."""
145
+ _enter_object(w, "istanbul artifact")
146
+ while True:
147
+ member = _next_member(w)
148
+ if member is None:
149
+ _expect_document_end(w)
150
+ return
151
+ yield _take_member(w, member)
152
+
153
+
154
+ # --- coverage.py: the same walk, one level down ---------------------------
155
+
156
+ def _leave_object(w: _Window, what: str) -> None:
157
+ close = _CLOSE_INNER.match(w.buf, w.pos)
158
+ if close is None:
159
+ raise ValueError(f"unterminated {what}")
160
+ w.drop(close.end())
161
+
162
+
163
+ def _walk_nested(w: _Window, start: int) -> Iterator[tuple[str, object, str]]:
164
+ """The members of the object at `start`, one at a time."""
165
+ w.pos = start
166
+ _enter_object(w, "coverage.py report: 'files'")
167
+ while True:
168
+ member = _next_member(w)
169
+ if member is None:
170
+ _leave_object(w, "coverage.py report: 'files' object")
171
+ return
172
+ key, value = _take_member(w, member)
173
+ yield key, value, "sub"
174
+
175
+
176
+ def walk_report(w: _Window, target: str) -> Iterator[tuple[str, object, str]]:
177
+ """(key, value, kind) per top-level member. kind is "member" for an ordinary
178
+ decoded value and "sub" for one member of the `target` object, so meta and
179
+ totals arrive whole and "files" arrives one file at a time."""
180
+ _enter_object(w, "coverage.py report")
181
+ while True:
182
+ member = _next_member(w)
183
+ if member is None:
184
+ # A walk that just stops at the first unreadable byte reports zero
185
+ # dark lines, which is indistinguishable from a fully covered repo.
186
+ _expect_document_end(w)
187
+ return
188
+ if json.loads(member.group(1)) == target:
189
+ yield from _walk_nested(w, member.end())
190
+ continue
191
+ key, value = _take_member(w, member)
192
+ yield key, value, "member"
193
+
194
+
195
+ # --- public readers --------------------------------------------------------
196
+
197
+ def _window(path: Path | str, chunk: int) -> tuple[_Window, IO[bytes]]:
198
+ handle = open(path, "rb")
199
+ return _Window(handle, chunk), handle
200
+
201
+
202
+ def _guarded(work, message: str):
203
+ """Run a walk, reporting any parse failure the way the whole-document
204
+ parsers do. A ToolError the walk raised itself is already the right error
205
+ and keeps its own wording."""
206
+ try:
207
+ return work()
208
+ except ToolError:
209
+ raise
210
+ except Exception as exc:
211
+ raise ToolError(f"{message}: {exc}") from exc
212
+
213
+
214
+ _BAD_ISTANBUL = "unparseable istanbul artifact"
215
+ _BAD_REPORT = "unparseable coverage.py report"
216
+ _NO_BRANCH = "coverage.py report lacks branch data — run the lane with branch coverage on"
217
+
218
+
219
+ def _istanbul_map(w: _Window, repo_root: str, per_file) -> dict:
220
+ out = {}
221
+ for abs_path, cov in split_window(w):
222
+ out[_rel_path(abs_path, repo_root)] = per_file(cov)
223
+ return out
224
+
225
+
226
+ def parse_istanbul_file(path: Path | str, *, repo_root: str, chunk: int = CHUNK
227
+ ) -> tuple[dict[str, list[FnCoverage]], str]:
228
+ """Per-file function coverage plus the sha256 of the artifact's own bytes.
229
+ Same result as parse_istanbul(path.read_text(), ...), same digest as
230
+ sha256(path.read_bytes()), without either whole copy ever existing."""
231
+ w, handle = _window(path, chunk)
232
+ with handle:
233
+ per_file = _guarded(lambda: _istanbul_map(w, repo_root, _file_coverage),
234
+ _BAD_ISTANBUL)
235
+ if not per_file:
236
+ raise ToolError(
237
+ "istanbul artifact is empty (zero files) — the coverage run measured nothing")
238
+ return per_file, w.hasher.hexdigest()
239
+
240
+
241
+ def parse_istanbul_missing_file(path: Path | str, *, repo_root: str,
242
+ chunk: int = CHUNK) -> dict[str, set[int]]:
243
+ """Per measured file, the lines whose statement never ran."""
244
+ w, handle = _window(path, chunk)
245
+ with handle:
246
+ return _guarded(lambda: _istanbul_map(w, repo_root, _dead_lines), _BAD_ISTANBUL)
247
+
248
+
249
+ def _prefix(path_prefix: str) -> str:
250
+ return (path_prefix.rstrip("/") + "/") if path_prefix else ""
251
+
252
+
253
+ def _meta_has_branch(key: str, value: object) -> bool:
254
+ if key != "meta" or not isinstance(value, dict):
255
+ return False
256
+ return bool(value.get("branch_coverage"))
257
+
258
+
259
+ def _require_branch(seen: bool) -> None:
260
+ if not seen:
261
+ raise ToolError(_NO_BRANCH)
262
+
263
+
264
+ def _add_file(out: dict, prefix: str, raw_path: str, data: dict, to_functions):
265
+ """Record one file's functions, or hand back the error it raised.
266
+
267
+ The error is CARRIED, not thrown: the whole-document parser saw the report
268
+ before it read a single file, so it always refused a report with no branch
269
+ data first. Members are not ordered — json.dump(sort_keys=True) writes
270
+ "files" ahead of "meta" — so streaming only keeps that precedence by
271
+ finishing the walk before it decides which complaint wins.
272
+ """
273
+ try:
274
+ out[prefix + raw_path.replace("\\", "/")] = to_functions(raw_path, data)
275
+ return None
276
+ except Exception as exc:
277
+ return exc
278
+
279
+
280
+ def _coveragepy_functions(w: _Window, prefix: str) -> dict:
281
+ """path -> function coverage, refusing a report with no branch data."""
282
+ from .coverage_py import _file_functions
283
+
284
+ out: dict = {}
285
+ branch, failure = False, None
286
+ for key, value, kind in walk_report(w, "files"):
287
+ if kind == "member":
288
+ branch = branch or _meta_has_branch(key, value)
289
+ continue
290
+ failure = failure or _add_file(out, prefix, key, value, _file_functions)
291
+ _require_branch(branch)
292
+ if failure is not None:
293
+ raise failure
294
+ return out
295
+
296
+
297
+ def parse_coveragepy_file(path: Path | str, *, path_prefix: str, chunk: int = CHUNK
298
+ ) -> tuple[dict[str, list[FnCoverage]], str]:
299
+ """Per-file function coverage plus the sha256 of the report's own bytes."""
300
+ w, handle = _window(path, chunk)
301
+ with handle:
302
+ per_file = _guarded(lambda: _coveragepy_functions(w, _prefix(path_prefix)),
303
+ _BAD_REPORT)
304
+ return per_file, w.hasher.hexdigest()
305
+
306
+
307
+ def _coveragepy_missing(w: _Window, prefix: str) -> dict[str, set[int]]:
308
+ out: dict[str, set[int]] = {}
309
+ for key, value, kind in walk_report(w, "files"):
310
+ if kind == "sub":
311
+ out[prefix + key.replace("\\", "/")] = set(value.get("missing_lines", ()))
312
+ return out
313
+
314
+
315
+ def parse_coveragepy_missing_file(path: Path | str, *, path_prefix: str,
316
+ chunk: int = CHUNK) -> dict[str, set[int]]:
317
+ """Per measured file, the lines coverage.py reports as never run."""
318
+ w, handle = _window(path, chunk)
319
+ with handle:
320
+ return _guarded(lambda: _coveragepy_missing(w, _prefix(path_prefix)), _BAD_REPORT)
crapkit/diffparse.py ADDED
@@ -0,0 +1,98 @@
1
+ """Parse `git diff -U0` output into per-file changed line ranges. Pure.
2
+
3
+ Ranges are new-side. A pure deletion (zero new lines) still marks the line it
4
+ happened at, so a function shrunk by an edit is still a touched function.
5
+ Paths come from the +++ header (the new side survives renames) and decode
6
+ git's C-style quoting via the churn module's helper.
7
+
8
+ Hunk body lines are consumed by the counts the @@ header declares, never
9
+ pattern-matched: an added source line whose text starts with "++ " arrives as
10
+ "+++ <text>" and would otherwise read as a file header, repointing every later
11
+ hunk at a phantom path the gate never checks.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import re
16
+
17
+ from .churn import _unquote_git_path
18
+
19
+ _HUNK = re.compile(r"^@@ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@")
20
+
21
+
22
+ def _spend_body_line(line: str, rem_old: int, rem_new: int) -> tuple[int, int] | None:
23
+ """Charge one line to the hunk body still owed, returning what remains.
24
+
25
+ None means the line is diff structure, not body: either nothing is owed or
26
+ the hunk ended early.
27
+ """
28
+ if rem_old <= 0 and rem_new <= 0:
29
+ return None
30
+ if line.startswith("\\"): # ""
31
+ return rem_old, rem_new
32
+ if line.startswith("-"):
33
+ return rem_old - 1, rem_new
34
+ if line.startswith("+"):
35
+ return rem_old, rem_new - 1
36
+ # -U0 hunks contain only +/- lines; anything else means the hunk ended
37
+ # early (truncated or synthetic diff) - reparse it normally.
38
+ return None
39
+
40
+
41
+ def _open_file(line: str, ranges: dict[str, list[tuple[int, int]]]) -> str | None:
42
+ """Point the parser at the file a `+++ ` header names; None for /dev/null."""
43
+ target = line[4:].strip()
44
+ if target == "/dev/null":
45
+ return None
46
+ target = _unquote_git_path(target)
47
+ path = target[2:] if target.startswith("b/") else target
48
+ path = path.replace("\\", "/")
49
+ ranges.setdefault(path, [])
50
+ return path
51
+
52
+
53
+ def _new_side_range(start: int, count: int) -> tuple[int, int]:
54
+ """New-side span a hunk covers."""
55
+ if count == 0: # pure deletion: mark the touch point
56
+ return max(start, 1), max(start, 1)
57
+ return start, start + count - 1
58
+
59
+
60
+ def _record_hunk(
61
+ m: re.Match[str], current: str | None, ranges: dict[str, list[tuple[int, int]]]
62
+ ) -> tuple[int, int]:
63
+ """Land the hunk's range on the current file and return the body counts it owes.
64
+
65
+ An omitted count in the @@ header means one line. A hunk with no current
66
+ file still declares a body to consume.
67
+ """
68
+ rem_old = int(m.group(2)) if m.group(2) is not None else 1
69
+ rem_new = int(m.group(4)) if m.group(4) is not None else 1
70
+ if current is not None:
71
+ ranges[current].append(_new_side_range(int(m.group(3)), rem_new))
72
+ return rem_old, rem_new
73
+
74
+
75
+ def _files_with_hunks(
76
+ ranges: dict[str, list[tuple[int, int]]],
77
+ ) -> dict[str, list[tuple[int, int]]]:
78
+ """Drop files a header introduced but no hunk ever touched."""
79
+ return {p: r for p, r in ranges.items() if r}
80
+
81
+
82
+ def changed_ranges(diff_text: str) -> dict[str, list[tuple[int, int]]]:
83
+ ranges: dict[str, list[tuple[int, int]]] = {}
84
+ current: str | None = None
85
+ rem_old = rem_new = 0
86
+ for line in diff_text.splitlines():
87
+ owed = _spend_body_line(line, rem_old, rem_new)
88
+ if owed is not None:
89
+ rem_old, rem_new = owed
90
+ continue
91
+ rem_old = rem_new = 0
92
+ if line.startswith("+++ "):
93
+ current = _open_file(line, ranges)
94
+ continue
95
+ m = _HUNK.match(line)
96
+ if m:
97
+ rem_old, rem_new = _record_hunk(m, current, ranges)
98
+ return _files_with_hunks(ranges)
crapkit/digest.py ADDED
@@ -0,0 +1,191 @@
1
+ """Digest and trend totals. Pure.
2
+
3
+ The digest speaks only on change (an unchanged week is silence, per the
4
+ empty-channel rule) and keeps to a handful of numbers: totals delta, the worst
5
+ per-function regressions, improvements, and new over-target functions.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from typing import NamedTuple
10
+
11
+ from .score import ScoredRow, grade
12
+
13
+
14
+ class Totals(NamedTuple):
15
+ functions: int
16
+ over_target: int
17
+ crap_load: float
18
+ avg: float
19
+ pct_over: float # normalized: absolute counts grow with the repo; this does not
20
+
21
+
22
+ class Digest(NamedTuple):
23
+ quiet: bool
24
+ lines: list[str]
25
+
26
+
27
+ def _over_count(rows, target: int, scope_targets) -> int:
28
+ return sum(1 for r in rows if r.crap > (scope_targets or {}).get(r.scope, target))
29
+
30
+
31
+ def totals_from_counts(functions: int, over_target: int, load: float) -> Totals:
32
+ """The rounding rule, in one place. A caller that already has the three sums
33
+ (store.run_totals adds them up inside the scan) must round them exactly the
34
+ way a caller holding the rows does, or the same run reads two ways."""
35
+ return Totals(
36
+ functions=functions,
37
+ over_target=over_target,
38
+ crap_load=round(load, 2),
39
+ avg=round(load / functions, 4) if functions else 0.0,
40
+ pct_over=round(100.0 * over_target / functions, 2) if functions else 0.0,
41
+ )
42
+
43
+
44
+ def totals(rows: list[ScoredRow], *, target: int,
45
+ scope_targets: dict[str, int] | None = None) -> Totals:
46
+ return totals_from_counts(len(rows), _over_count(rows, target, scope_targets),
47
+ sum(r.crap for r in rows))
48
+
49
+
50
+ def scope_totals(rows: list[ScoredRow], *, target: int,
51
+ scope_targets: dict[str, int] | None = None) -> dict[str, Totals]:
52
+ by_scope: dict[str, list[ScoredRow]] = {}
53
+ for r in rows:
54
+ by_scope.setdefault(r.scope, []).append(r)
55
+ return {scope: totals(srows, target=target, scope_targets=scope_targets)
56
+ for scope, srows in sorted(by_scope.items())}
57
+
58
+
59
+ def scope_rollup(by_scope: dict[str, Totals]) -> dict[str, dict]:
60
+ """The per-scope block coverage --json and trend --json both publish.
61
+
62
+ One shaping in one place: the two commands reach their Totals differently
63
+ (rows in hand vs a GROUP BY), and a second shaping would let the same run
64
+ read two ways depending on which command asked.
65
+ """
66
+ return {scope: {"functions": t.functions, "over_target": t.over_target,
67
+ "crap_load": t.crap_load, "grade": grade(t.over_target, t.functions)}
68
+ for scope, t in by_scope.items()}
69
+
70
+
71
+ def latest_comparable_pair(runs: list[dict]) -> tuple[dict, dict] | None:
72
+ """The newest pair of runs with IDENTICAL lane sets.
73
+
74
+ Comparing a --lane subset run against a full run flips every out-of-subset
75
+ function to no-lane and manufactures phantom regressions; only like-for-like
76
+ pairs digest.
77
+ """
78
+ for i in range(len(runs) - 1, 0, -1):
79
+ cur = runs[i]
80
+ cur_lanes = set(cur["lanes"])
81
+ for j in range(i - 1, -1, -1):
82
+ if set(runs[j]["lanes"]) == cur_lanes:
83
+ return runs[j], cur
84
+ return None
85
+
86
+
87
+ def skipped_runs(runs: list[dict], pair: tuple[dict, dict] | None) -> list[dict]:
88
+ """The runs the pair stepped over: everything after its older half except
89
+ its newer half. Empty exactly when the pair is the newest run and the one
90
+ before it.
91
+
92
+ Stepping over a run is the right move — a subset run never compares — but a
93
+ digest that says nothing about it reports an older delta in this week's
94
+ voice. On the flagship consumer that printed runs 17 -> 19 while run 22 sat
95
+ in the store, and nothing on any surface said 22 existed.
96
+ """
97
+ if pair is None:
98
+ return []
99
+ prev, cur = pair
100
+ return [r for r in runs[runs.index(prev) + 1:] if r["id"] != cur["id"]]
101
+
102
+
103
+ _Key = tuple[str, str] # (path, long_name): what identifies a function across runs
104
+ _Move = tuple[float, ScoredRow, ScoredRow] # (delta, before, after)
105
+ _Delta = tuple[float, ScoredRow]
106
+
107
+
108
+ class _Changes(NamedTuple):
109
+ """What moved between two runs, bucketed the way the digest reports it."""
110
+ regressions: list[_Delta]
111
+ improvements: list[_Delta]
112
+ appeared: list[ScoredRow]
113
+
114
+ def nothing_moved(self) -> bool:
115
+ return not (self.regressions or self.improvements or self.appeared)
116
+
117
+
118
+ def _by_key(rows: list[ScoredRow]) -> dict[_Key, ScoredRow]:
119
+ return {(r.path, r.long_name): r for r in rows}
120
+
121
+
122
+ def _moves(prev_by_key: dict[_Key, ScoredRow],
123
+ cur_by_key: dict[_Key, ScoredRow]) -> list[_Move]:
124
+ """Every function measured in both runs, paired with how far its CRAP moved."""
125
+ return [(row.crap - prev_by_key[key].crap, prev_by_key[key], row)
126
+ for key, row in cur_by_key.items() if key in prev_by_key]
127
+
128
+
129
+ def _regressions(moves: list[_Move]) -> list[_Delta]:
130
+ return [(delta, after) for delta, _before, after in moves if delta > 0.01]
131
+
132
+
133
+ def _improvements(moves: list[_Move], target: int) -> list[_Delta]:
134
+ """Only a function that WAS over target improves; drift below it is not news."""
135
+ return [(delta, after) for delta, before, after in moves
136
+ if delta < -0.01 and before.crap > target]
137
+
138
+
139
+ def _appeared(prev_by_key: dict[_Key, ScoredRow], cur_by_key: dict[_Key, ScoredRow],
140
+ target: int) -> list[ScoredRow]:
141
+ """Functions the previous run never saw; new code under target is not news."""
142
+ return [row for key, row in cur_by_key.items()
143
+ if key not in prev_by_key and row.crap > target]
144
+
145
+
146
+ def _changes_between(prev: list[ScoredRow], cur: list[ScoredRow], target: int) -> _Changes:
147
+ prev_by_key, cur_by_key = _by_key(prev), _by_key(cur)
148
+ moves = _moves(prev_by_key, cur_by_key)
149
+ return _Changes(
150
+ regressions=_regressions(moves),
151
+ improvements=_improvements(moves, target),
152
+ appeared=_appeared(prev_by_key, cur_by_key, target),
153
+ )
154
+
155
+
156
+ def _totals_line(t_prev: Totals, t_cur: Totals) -> str:
157
+ return (f"CRAP load {t_prev.crap_load} -> {t_cur.crap_load}; "
158
+ f"over target {t_prev.over_target} -> {t_cur.over_target}; "
159
+ f"functions {t_prev.functions} -> {t_cur.functions}")
160
+
161
+
162
+ def _fn_line(prefix: str, row: ScoredRow) -> str:
163
+ return f"{prefix}: {row.path} {row.long_name} (crap {row.crap:.1f})"
164
+
165
+
166
+ def _regression_lines(regressions: list[_Delta], top: int) -> list[str]:
167
+ return [_fn_line(f"regressed +{delta:.1f}", row)
168
+ for delta, row in sorted(regressions, key=lambda x: -x[0])[:top]]
169
+
170
+
171
+ def _appeared_lines(appeared: list[ScoredRow], top: int) -> list[str]:
172
+ return [_fn_line("new over target", row)
173
+ for row in sorted(appeared, key=lambda r: -r.crap)[:top]]
174
+
175
+
176
+ def _improvement_lines(improvements: list[_Delta], top: int) -> list[str]:
177
+ return [_fn_line(f"improved {delta:.1f}", row)
178
+ for delta, row in sorted(improvements, key=lambda x: x[0])[:top]]
179
+
180
+
181
+ def build_digest(prev: list[ScoredRow], cur: list[ScoredRow], *, target: int, top: int = 5) -> Digest:
182
+ t_prev, t_cur = totals(prev, target=target), totals(cur, target=target)
183
+ changed = _changes_between(prev, cur, target)
184
+ if t_prev == t_cur and changed.nothing_moved():
185
+ return Digest(quiet=True, lines=[])
186
+ return Digest(quiet=False, lines=[
187
+ _totals_line(t_prev, t_cur),
188
+ *_regression_lines(changed.regressions, top),
189
+ *_appeared_lines(changed.appeared, top),
190
+ *_improvement_lines(changed.improvements, top),
191
+ ])