crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/ratchet.py
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""The committed ratchet: per-function high-water CRAP marks for functions above target.
|
|
2
|
+
|
|
3
|
+
Pure text <-> entries. Marks only ever tighten: an improvement lowers or drops
|
|
4
|
+
an entry on update; a regression NEVER raises one (it surfaces as a verdict
|
|
5
|
+
failure instead); brand-new debt enters only through the audited override.
|
|
6
|
+
Identity is (path, long_name): spans drift with every edit, names survive.
|
|
7
|
+
|
|
8
|
+
The file opens with a metric stamp comment naming the analysis version and the
|
|
9
|
+
lizard that produced the numbers. Marks measured under different rules are not
|
|
10
|
+
comparable, and without the stamp that reads as a clean run.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import NamedTuple
|
|
15
|
+
|
|
16
|
+
from .score import ScoredRow
|
|
17
|
+
|
|
18
|
+
_HEADER = "path\tlong_name\tcrap"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class RatchetEntry(NamedTuple):
|
|
22
|
+
path: str
|
|
23
|
+
long_name: str
|
|
24
|
+
crap: float
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def stamp_text(analysis_version: int, lizard_version: str) -> str:
|
|
28
|
+
"""The metric identity a set of marks was measured under, as one line."""
|
|
29
|
+
return f"crapkit-analysis={analysis_version} lizard={lizard_version}"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def metric_version() -> str:
|
|
33
|
+
# Imported here, not at module scope: the git merge driver runs this module
|
|
34
|
+
# in a temp dir with no analysis stack, and it passes its own stamp through.
|
|
35
|
+
import lizard
|
|
36
|
+
|
|
37
|
+
from .analyze import ANALYSIS_VERSION
|
|
38
|
+
return stamp_text(ANALYSIS_VERSION, lizard.version)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def read_stamp(text: str) -> str:
|
|
42
|
+
"""The metric a marks file was written under; "" for one written before stamping."""
|
|
43
|
+
for line in text.splitlines():
|
|
44
|
+
if line.startswith("#"):
|
|
45
|
+
return line[1:].strip()
|
|
46
|
+
if line.strip():
|
|
47
|
+
return ""
|
|
48
|
+
return ""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def stamp_conflict(recorded: str, current: str) -> str | None:
|
|
52
|
+
"""The refusal when marks and the running metric disagree; None when they compare.
|
|
53
|
+
|
|
54
|
+
An unstamped file has nothing to disagree with — the caller warns instead.
|
|
55
|
+
"""
|
|
56
|
+
if not recorded or recorded == current:
|
|
57
|
+
return None
|
|
58
|
+
return (f"ratchet marks were recorded under [{recorded}] but this run measures "
|
|
59
|
+
f"[{current}] — CRAP scores are not comparable across metric versions; "
|
|
60
|
+
"re-baseline with `crapkit ratchet seed`")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _is_skippable(line: str) -> bool:
|
|
64
|
+
"""Blank lines, the header and comment lines (the metric stamp) carry no mark."""
|
|
65
|
+
return not line.strip() or line == _HEADER or line.startswith("#")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def load_ratchet(text: str) -> list[RatchetEntry]:
|
|
69
|
+
entries = []
|
|
70
|
+
for i, line in enumerate(text.splitlines()):
|
|
71
|
+
if _is_skippable(line):
|
|
72
|
+
continue
|
|
73
|
+
parts = line.split("\t")
|
|
74
|
+
if len(parts) != 3:
|
|
75
|
+
raise ValueError(f"ratchet line {i + 1} has {len(parts)} fields, expected 3: {line!r}")
|
|
76
|
+
entries.append(RatchetEntry(parts[0], parts[1], float(parts[2])))
|
|
77
|
+
return entries
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def mark_for(entries: list[RatchetEntry], path: str, long_name: str) -> float | None:
|
|
81
|
+
"""One function's recorded high-water mark, or None when it carries no mark."""
|
|
82
|
+
for e in entries:
|
|
83
|
+
if e.path == path and e.long_name == long_name:
|
|
84
|
+
return e.crap
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def dump_ratchet(entries: list[RatchetEntry], *, stamp: str | None = None) -> str:
|
|
89
|
+
"""`stamp` None takes the running metric; a version string is written verbatim
|
|
90
|
+
and "" writes none, which is how the merge driver keeps two legacy sides legacy."""
|
|
91
|
+
version = metric_version() if stamp is None else stamp
|
|
92
|
+
lines = [f"# {version}"] if version else []
|
|
93
|
+
lines.append(_HEADER)
|
|
94
|
+
for e in sorted(entries, key=lambda e: (e.path, e.long_name)):
|
|
95
|
+
lines.append(f"{e.path}\t{e.long_name}\t{e.crap:.4f}")
|
|
96
|
+
return "\n".join(lines) + "\n"
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _merge_key(b: float | None, o: float | None, t: float | None) -> float | None:
|
|
100
|
+
"""git 3-way semantics per key: the changed side wins over the unchanged one.
|
|
101
|
+
Both changed: keep the mark (it can only fall; prune is re-runnable) at min."""
|
|
102
|
+
if o == t:
|
|
103
|
+
return o
|
|
104
|
+
if o == b:
|
|
105
|
+
return t
|
|
106
|
+
if t == b:
|
|
107
|
+
return o
|
|
108
|
+
# both changed differently; both-None is impossible past the o == t check
|
|
109
|
+
return min(x for x in (o, t) if x is not None)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def merge_ratchets(base: list[RatchetEntry], ours: list[RatchetEntry],
|
|
113
|
+
theirs: list[RatchetEntry]) -> list[RatchetEntry]:
|
|
114
|
+
b = {(e.path, e.long_name): e.crap for e in base}
|
|
115
|
+
o = {(e.path, e.long_name): e.crap for e in ours}
|
|
116
|
+
t = {(e.path, e.long_name): e.crap for e in theirs}
|
|
117
|
+
merged = []
|
|
118
|
+
for key in sorted(set(b) | set(o) | set(t)):
|
|
119
|
+
crap = _merge_key(b.get(key), o.get(key), t.get(key))
|
|
120
|
+
if crap is not None:
|
|
121
|
+
merged.append(RatchetEntry(key[0], key[1], crap))
|
|
122
|
+
return merged
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def seed_ratchet(prior: list[RatchetEntry], fresh: list[ScoredRow], *, target: int,
|
|
126
|
+
scope_targets: dict[str, int] | None = None) -> tuple[list[RatchetEntry], int, int]:
|
|
127
|
+
"""First-class mark entry: record every over-ceiling function at its current
|
|
128
|
+
CRAP. A mark never rises, seeding included — an existing lower mark stays."""
|
|
129
|
+
from .verify import worst_twins
|
|
130
|
+
|
|
131
|
+
ceilings = scope_targets or {}
|
|
132
|
+
marks = {(e.path, e.long_name): e for e in prior}
|
|
133
|
+
added = tightened = 0
|
|
134
|
+
for key, row in worst_twins(fresh).items():
|
|
135
|
+
if row.crap <= ceilings.get(row.scope, target):
|
|
136
|
+
continue
|
|
137
|
+
a, t = _seed_mark(marks, key, row.crap)
|
|
138
|
+
added += a
|
|
139
|
+
tightened += t
|
|
140
|
+
return sorted(marks.values(), key=lambda e: (e.path, e.long_name)), added, tightened
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _seed_mark(marks: dict, key: tuple, crap: float) -> tuple[int, int]:
|
|
144
|
+
"""Place one mark; returns (added, tightened) deltas. A mark never rises."""
|
|
145
|
+
old = marks.get(key)
|
|
146
|
+
mark = round(crap, 4)
|
|
147
|
+
if old is not None and mark >= old.crap:
|
|
148
|
+
return 0, 0
|
|
149
|
+
marks[key] = RatchetEntry(key[0], key[1], mark)
|
|
150
|
+
return (1, 0) if old is None else (0, 1)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _repath(entries: list[RatchetEntry], dest_of) -> tuple[list[RatchetEntry], int]:
|
|
154
|
+
"""Rewrite paths by a per-mark chooser (None leaves a mark alone); values never move."""
|
|
155
|
+
out = []
|
|
156
|
+
moved = 0
|
|
157
|
+
for e in entries:
|
|
158
|
+
dest = dest_of(e)
|
|
159
|
+
if dest is None:
|
|
160
|
+
out.append(e)
|
|
161
|
+
continue
|
|
162
|
+
out.append(e._replace(path=dest))
|
|
163
|
+
moved += 1
|
|
164
|
+
return sorted(out, key=lambda e: (e.path, e.long_name)), moved
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _moved_path(path: str, old: str, new: str) -> str | None:
|
|
168
|
+
"""One mark's destination under an explicit move, or None when `old` misses it."""
|
|
169
|
+
if old.endswith("/"):
|
|
170
|
+
return f"{new.rstrip('/')}/{path[len(old):]}" if path.startswith(old) else None
|
|
171
|
+
return new if path == old else None
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def move_marks(entries: list[RatchetEntry], old: str,
|
|
175
|
+
new: str) -> tuple[list[RatchetEntry], int]:
|
|
176
|
+
"""Re-path marks at their recorded values. A trailing "/" on `old` moves a whole
|
|
177
|
+
directory; anything else matches one exact path."""
|
|
178
|
+
return _repath(entries, lambda e: _moved_path(e.path, old, new))
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _rename_target(entry: RatchetEntry, present: set, renames: dict[str, str]) -> str | None:
|
|
182
|
+
"""Where a mark should follow a renamed file, or None to leave it alone.
|
|
183
|
+
|
|
184
|
+
Three conditions, all required: the function is gone from its recorded path,
|
|
185
|
+
git calls that path renamed, and the SAME long_name exists at the new one. A
|
|
186
|
+
copy fails the first (the source survives), so its mark never travels.
|
|
187
|
+
"""
|
|
188
|
+
if (entry.path, entry.long_name) in present:
|
|
189
|
+
return None
|
|
190
|
+
dest = renames.get(entry.path)
|
|
191
|
+
if dest is None or (dest, entry.long_name) not in present:
|
|
192
|
+
return None
|
|
193
|
+
return dest
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def follow_renames(prior: list[RatchetEntry], fresh: list[ScoredRow],
|
|
197
|
+
renames: dict[str, str]) -> tuple[list[RatchetEntry], int]:
|
|
198
|
+
"""Marks for renamed files, re-pathed. Runs BEFORE prune so a rename reads as a
|
|
199
|
+
move, not as code that left the repo and forfeits its high-water mark."""
|
|
200
|
+
from .verify import worst_twins
|
|
201
|
+
|
|
202
|
+
present = set(worst_twins(fresh))
|
|
203
|
+
return _repath(prior, lambda e: _rename_target(e, present, renames))
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def prune_ratchet(prior: list[RatchetEntry],
|
|
207
|
+
fresh: list[ScoredRow]) -> tuple[list[RatchetEntry], int]:
|
|
208
|
+
"""Deliberate mark exit: drop entries whose function is absent from the run.
|
|
209
|
+
The automatic update keeps them (an exclude glob or a lane outage also removes
|
|
210
|
+
rows); prune is the human confirming the code is really gone."""
|
|
211
|
+
from .verify import worst_twins
|
|
212
|
+
|
|
213
|
+
present = set(worst_twins(fresh))
|
|
214
|
+
kept = [e for e in prior if (e.path, e.long_name) in present]
|
|
215
|
+
return kept, len(prior) - len(kept)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def update_ratchet(prior: list[RatchetEntry], fresh: list[ScoredRow], *, target: int,
|
|
219
|
+
scope_targets: dict[str, int] | None = None) -> list[RatchetEntry]:
|
|
220
|
+
from .verify import worst_twins
|
|
221
|
+
|
|
222
|
+
fresh_by_key = worst_twins(fresh)
|
|
223
|
+
updated = []
|
|
224
|
+
for entry in prior:
|
|
225
|
+
row = fresh_by_key.get((entry.path, entry.long_name))
|
|
226
|
+
if row is None:
|
|
227
|
+
# Absent from the scored rows is NOT proof the code is gone — an
|
|
228
|
+
# exclude glob or a lane outage also removes it, and dropping the
|
|
229
|
+
# entry would erase an audited override's only diff-visible record.
|
|
230
|
+
# Stale entries are inert (verify checks only present functions).
|
|
231
|
+
updated.append(entry)
|
|
232
|
+
continue
|
|
233
|
+
if row.crap <= (scope_targets or {}).get(row.scope, target):
|
|
234
|
+
continue # fixed for real: below the scope's ceiling needs no mark
|
|
235
|
+
updated.append(RatchetEntry(entry.path, entry.long_name, min(entry.crap, round(row.crap, 4))))
|
|
236
|
+
return sorted(updated, key=lambda e: (e.path, e.long_name))
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Burn-down analytics from the ratchet file's own git history. Pure.
|
|
2
|
+
|
|
3
|
+
No timestamps live in the TSV; the commits that changed it carry them, so a
|
|
4
|
+
fixed history reports deterministically. Ages and velocity anchor on the
|
|
5
|
+
NEWEST commit in the history, never the wall clock.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
DAY = 86400
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _mark_line(line: str) -> tuple | None:
|
|
13
|
+
"""((path, long_name), crap) when a +/- patch line is a mark row; else None."""
|
|
14
|
+
body = line[1:]
|
|
15
|
+
if body.startswith(("++ ", "-- ", "path\t")):
|
|
16
|
+
return None
|
|
17
|
+
parts = body.split("\t")
|
|
18
|
+
if len(parts) != 3:
|
|
19
|
+
return None
|
|
20
|
+
try:
|
|
21
|
+
return (parts[0], parts[1]), float(parts[2])
|
|
22
|
+
except ValueError:
|
|
23
|
+
return None
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _commit_delta(patch: str) -> tuple[dict, dict]:
|
|
27
|
+
"""The marks one commit's patch added and removed, keyed."""
|
|
28
|
+
added: dict = {}
|
|
29
|
+
removed: dict = {}
|
|
30
|
+
for line in patch.splitlines():
|
|
31
|
+
if not line or line[0] not in "+-":
|
|
32
|
+
continue
|
|
33
|
+
mark = _mark_line(line)
|
|
34
|
+
if mark is None:
|
|
35
|
+
continue
|
|
36
|
+
(added if line[0] == "+" else removed)[mark[0]] = mark[1]
|
|
37
|
+
return added, removed
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def mark_events(patches: list[tuple[int, str]]) -> list[tuple]:
|
|
41
|
+
"""(ts, key, 'added'|'dropped', crap) per commit. A key removed and re-added
|
|
42
|
+
in the same commit is a tightening and emits nothing — only entry and
|
|
43
|
+
repayment count."""
|
|
44
|
+
events = []
|
|
45
|
+
for ts, patch in patches:
|
|
46
|
+
added, removed = _commit_delta(patch)
|
|
47
|
+
for key in sorted(set(added) - set(removed)):
|
|
48
|
+
events.append((ts, key, "added", added[key]))
|
|
49
|
+
for key in sorted(set(removed) - set(added)):
|
|
50
|
+
events.append((ts, key, "dropped", removed[key]))
|
|
51
|
+
return events
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _drop_velocity(dropped: list[int], anchor: int) -> dict:
|
|
55
|
+
return {"dropped_last_30d": sum(1 for ts in dropped if anchor - ts <= 30 * DAY),
|
|
56
|
+
"dropped_last_90d": sum(1 for ts in dropped if anchor - ts <= 90 * DAY)}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _expired_marks(report: dict, max_age_months: int | None) -> list[str]:
|
|
60
|
+
if max_age_months is None:
|
|
61
|
+
return []
|
|
62
|
+
limit = max_age_months * 30
|
|
63
|
+
# "oldest" is age-sorted, so any expired mark appears in it; a backlog of
|
|
64
|
+
# more than its cap still reports the worst offenders
|
|
65
|
+
return [f"mark {e['long_name']} in {e['path']} is {e['age_days']}d old (limit {limit}d)"
|
|
66
|
+
for e in report["oldest"] if e["age_days"] > limit]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _stalled_repayment(report: dict, min_repaid_30d: int | None) -> list[str]:
|
|
70
|
+
if min_repaid_30d is None or not report["open"]:
|
|
71
|
+
return []
|
|
72
|
+
if report["dropped_last_30d"] >= min_repaid_30d:
|
|
73
|
+
return []
|
|
74
|
+
return [f"repayment stalled: {report['dropped_last_30d']} mark(s) repaid in 30d "
|
|
75
|
+
f"(policy wants {min_repaid_30d})"]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def policy_violations(report: dict, max_age_months: int | None,
|
|
79
|
+
min_repaid_30d: int | None) -> list[str]:
|
|
80
|
+
"""Findings against the debt policy knobs; no knobs, no findings."""
|
|
81
|
+
return _expired_marks(report, max_age_months) + _stalled_repayment(report, min_repaid_30d)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _replay(events: list[tuple]) -> tuple[dict, dict, list[int]]:
|
|
85
|
+
"""The committed state: when each surviving mark entered, what it is worth,
|
|
86
|
+
and the timestamp of every repayment."""
|
|
87
|
+
entered: dict = {}
|
|
88
|
+
crap: dict = {}
|
|
89
|
+
dropped: list[int] = []
|
|
90
|
+
for ts, key, kind, value in events:
|
|
91
|
+
if kind == "added":
|
|
92
|
+
entered[key] = ts
|
|
93
|
+
crap[key] = value
|
|
94
|
+
else:
|
|
95
|
+
entered.pop(key, None)
|
|
96
|
+
crap.pop(key, None)
|
|
97
|
+
dropped.append(ts)
|
|
98
|
+
return entered, crap, dropped
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _open_marks(entered: dict, working: dict | None, anchor: int) -> dict:
|
|
102
|
+
"""Open marks keyed to when they entered. The working tree says WHICH marks
|
|
103
|
+
are open, history says how old each one is; a mark with no commit behind it
|
|
104
|
+
entered at the anchor and reports 0d."""
|
|
105
|
+
if working is None:
|
|
106
|
+
return entered
|
|
107
|
+
return {key: entered.get(key, anchor) for key in working}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _uncommitted(crap: dict, working: dict | None) -> int:
|
|
111
|
+
"""Marks the working tree and the newest committed version disagree on:
|
|
112
|
+
added, repaid or tightened on disk and not committed yet."""
|
|
113
|
+
if working is None:
|
|
114
|
+
return 0
|
|
115
|
+
return sum(1 for key in set(crap) | set(working) if crap.get(key) != working.get(key))
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _age_rows(opened: dict, anchor: int) -> list[dict]:
|
|
119
|
+
return sorted(({"path": k[0], "long_name": k[1], "age_days": (anchor - ts) // DAY}
|
|
120
|
+
for k, ts in opened.items()),
|
|
121
|
+
key=lambda e: (-e["age_days"], e["path"], e["long_name"]))
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def report_from_events(events: list[tuple], working: dict | None = None) -> dict:
|
|
125
|
+
"""`working` is the marks on disk, keyed (path, long_name) -> crap. It decides
|
|
126
|
+
which marks are open, because a seed that has not been committed is still debt
|
|
127
|
+
somebody owes. Repayment counts and velocity stay on committed history, which
|
|
128
|
+
is the only place a timestamp exists. None reports the committed state alone.
|
|
129
|
+
"""
|
|
130
|
+
entered, crap, dropped = _replay(events)
|
|
131
|
+
anchor = max((ts for ts, *_ in events), default=0)
|
|
132
|
+
opened = _open_marks(entered, working, anchor)
|
|
133
|
+
return {"open": len(opened), "dropped_total": len(dropped), "anchor_ts": anchor,
|
|
134
|
+
"uncommitted": _uncommitted(crap, working),
|
|
135
|
+
"oldest": _age_rows(opened, anchor)[:20], **_drop_velocity(dropped, anchor)}
|
crapkit/sarif.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""SARIF 2.1.0 and GitHub workflow-command emission. Pure builders.
|
|
2
|
+
|
|
3
|
+
Code-scanning UIs and PR annotation bots consume this; ruleIds, levels, and
|
|
4
|
+
locations are contract. Every uri is repo-relative with forward slashes.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from . import __version__
|
|
9
|
+
|
|
10
|
+
_RULES = (
|
|
11
|
+
{"id": "crapkit/over-target",
|
|
12
|
+
"shortDescription": {"text": "CRAP score above the scope ceiling"}},
|
|
13
|
+
{"id": "crapkit/gate",
|
|
14
|
+
"shortDescription": {"text": "touched function over the complexity gate"}},
|
|
15
|
+
{"id": "crapkit/ratchet-regression",
|
|
16
|
+
"shortDescription": {"text": "a recorded CRAP mark got worse"}},
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _result(rule_id: str, level: str, path: str, line: int, text: str) -> dict:
|
|
21
|
+
return {
|
|
22
|
+
"ruleId": rule_id, "level": level,
|
|
23
|
+
"message": {"text": text},
|
|
24
|
+
"locations": [{"physicalLocation": {
|
|
25
|
+
"artifactLocation": {"uri": path.replace("\\", "/")},
|
|
26
|
+
"region": {"startLine": line},
|
|
27
|
+
}}],
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def over_target_results(scored, scope_targets: dict, target: int) -> list[dict]:
|
|
32
|
+
out = []
|
|
33
|
+
for r in scored:
|
|
34
|
+
ceiling = scope_targets.get(r.scope, target)
|
|
35
|
+
if r.crap <= ceiling:
|
|
36
|
+
continue
|
|
37
|
+
out.append(_result(
|
|
38
|
+
"crapkit/over-target", "warning", r.path, r.start,
|
|
39
|
+
f"{r.long_name}: CRAP {r.crap:.1f} over ceiling {ceiling} "
|
|
40
|
+
f"(ccn {r.ccn}, cov {r.cov:.0%}) -> {r.remedy}"))
|
|
41
|
+
return out
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def gate_results(violations) -> list[dict]:
|
|
45
|
+
return [_result("crapkit/gate", "error", v.path, v.start,
|
|
46
|
+
f"{v.long_name}: CRAP {v.crap:.1f} (ccn {v.ccn}, cov {v.cov:.0%}) -> {v.remedy}")
|
|
47
|
+
for v in violations]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def regression_results(regressions) -> list[dict]:
|
|
51
|
+
# the ratchet stores no line numbers; line 1 anchors the file-level finding
|
|
52
|
+
return [_result("crapkit/ratchet-regression", "error", r.path, 1,
|
|
53
|
+
f"{r.long_name}: recorded {r.recorded} -> fresh {r.fresh_crap}")
|
|
54
|
+
for r in regressions]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def sarif_document(results: list[dict]) -> dict:
|
|
58
|
+
return {
|
|
59
|
+
"$schema": "https://raw.githubusercontent.com/oasis-tcs/sarif-spec/"
|
|
60
|
+
"master/Schemata/sarif-schema-2.1.0.json",
|
|
61
|
+
"version": "2.1.0",
|
|
62
|
+
"runs": [{
|
|
63
|
+
"tool": {"driver": {
|
|
64
|
+
"name": "crapkit",
|
|
65
|
+
"version": __version__,
|
|
66
|
+
"informationUri": "https://github.com/JeanFrancoisGagne/crapkit",
|
|
67
|
+
"rules": [dict(r) for r in _RULES],
|
|
68
|
+
}},
|
|
69
|
+
"results": results,
|
|
70
|
+
}],
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _esc(text: str) -> str:
|
|
75
|
+
return text.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def github_annotation(result: dict) -> str:
|
|
79
|
+
loc = result["locations"][0]["physicalLocation"]
|
|
80
|
+
return (f"::{result['level']} file={loc['artifactLocation']['uri']},"
|
|
81
|
+
f"line={loc['region']['startLine']},title={result['ruleId']}"
|
|
82
|
+
f"::{_esc(result['message']['text'])}")
|
crapkit/sarifio.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Writing a SARIF document to disk, one finding at a time.
|
|
2
|
+
|
|
3
|
+
json.dump reaches CPython's pure-Python encoder whenever indent is set, so a
|
|
4
|
+
36,767-finding report was serialized character by character in Python: 718 ms,
|
|
5
|
+
against 171 ms for the same findings handed to the C encoder. Nothing in the
|
|
6
|
+
SARIF spec asks for indentation and no consumer of this file is a human
|
|
7
|
+
scrolling 17 MB, so the document is written compact and streamed.
|
|
8
|
+
|
|
9
|
+
The skeleton comes from the builder itself, encoded with an empty results
|
|
10
|
+
array and split at that array. Findings are then C-encoded one at a time into
|
|
11
|
+
the gap, which lands byte for byte where json.dumps of the whole document
|
|
12
|
+
would have put them — and never builds that document as a string.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
from .sarif import sarif_document
|
|
20
|
+
|
|
21
|
+
_EMPTY_RESULTS = '"results": []'
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _frame(document: dict) -> tuple[str, str]:
|
|
25
|
+
"""The document either side of its results array."""
|
|
26
|
+
text = json.dumps(document, sort_keys=True)
|
|
27
|
+
head, found, tail = text.partition(_EMPTY_RESULTS)
|
|
28
|
+
if not found:
|
|
29
|
+
raise ValueError("sarif document has no results array to stream into")
|
|
30
|
+
return head + '"results": [', "]" + tail
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _write_results(handle, results) -> None:
|
|
34
|
+
# ", " is json.dumps' own item separator, which is what keeps the spliced
|
|
35
|
+
# bytes identical to a whole-document encode.
|
|
36
|
+
separator = ""
|
|
37
|
+
for result in results:
|
|
38
|
+
handle.write(separator + json.dumps(result, sort_keys=True))
|
|
39
|
+
separator = ", "
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def write_sarif(path: Path | str, results: list[dict]) -> None:
|
|
43
|
+
"""Write the SARIF report for these findings. newline="\\n" is the
|
|
44
|
+
determinism contract: the bytes must not pick up the host's separator."""
|
|
45
|
+
head, tail = _frame(sarif_document([]))
|
|
46
|
+
with open(path, "w", encoding="utf-8", newline="\n") as handle:
|
|
47
|
+
handle.write(head)
|
|
48
|
+
_write_results(handle, results)
|
|
49
|
+
handle.write(tail)
|