crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/dup.py
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
"""Near-duplicate function detection. Pure: inventory rows + file texts in,
|
|
2
|
+
ranked pairs out.
|
|
3
|
+
|
|
4
|
+
Normalized line shingles with CONTAINMENT scoring (shared / smaller set), so a
|
|
5
|
+
copy-paste that later grew a few lines still surfaces. An inverted shingle
|
|
6
|
+
index keeps a 14k-function repo tractable: only pairs that actually share a
|
|
7
|
+
shingle are ever compared. Tiny functions are structural noise and stay out.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from .snapshot import InventoryRow
|
|
12
|
+
|
|
13
|
+
WINDOW = 4 # consecutive normalized lines per shingle
|
|
14
|
+
_COMMENT_PREFIXES = ("#", "//", "/*", "*", '"""', "'''")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _normalized_lines(file_lines: list[str], start: int, end: int) -> list[str]:
|
|
18
|
+
picked = []
|
|
19
|
+
for raw in file_lines[start - 1:end]:
|
|
20
|
+
line = "".join(raw.split()) # whitespace never distinguishes a clone
|
|
21
|
+
if line and not raw.strip().startswith(_COMMENT_PREFIXES):
|
|
22
|
+
picked.append(line)
|
|
23
|
+
return picked
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _shingles(lines: list[str]) -> set[int]:
|
|
27
|
+
return {hash(tuple(lines[i:i + WINDOW])) for i in range(len(lines) - WINDOW + 1)}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _split_once(path: str, sources: dict[str, str]) -> list[str] | None:
|
|
31
|
+
text = sources.get(path)
|
|
32
|
+
return None if text is None else text.splitlines()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _row_shingles(r: InventoryRow, file_lines: list[str], min_lines: int) -> set[int] | None:
|
|
36
|
+
lines = _normalized_lines(file_lines, r.start, r.end)
|
|
37
|
+
return _shingles(lines) if len(lines) >= min_lines else None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _function_shingles(rows: list[InventoryRow], sources: dict[str, str],
|
|
41
|
+
min_lines: int) -> list[tuple[InventoryRow, set[int]]]:
|
|
42
|
+
# Rows arrive ordered by (scope, path, start), so every row of a file is
|
|
43
|
+
# contiguous: a ONE-ENTRY cache splits each source once instead of once per
|
|
44
|
+
# function in it (measured 3 GB of re-split text on a 104 MB repo). A file
|
|
45
|
+
# whose rows are NOT contiguous still scores identically, just re-split.
|
|
46
|
+
out = []
|
|
47
|
+
cached_path, file_lines = None, None
|
|
48
|
+
for r in rows:
|
|
49
|
+
if r.path != cached_path:
|
|
50
|
+
cached_path, file_lines = r.path, _split_once(r.path, sources)
|
|
51
|
+
if file_lines is None:
|
|
52
|
+
continue
|
|
53
|
+
shingles = _row_shingles(r, file_lines, min_lines)
|
|
54
|
+
if shingles is not None:
|
|
55
|
+
out.append((r, shingles))
|
|
56
|
+
return out
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _owners_by_shingle(indexed: list[tuple[InventoryRow, set[int]]]) -> dict[int, int | list[int]]:
|
|
60
|
+
"""Inverted shingle index: a lone owner stays a bare int, a list starts at two.
|
|
61
|
+
|
|
62
|
+
84% of shingles in a real repo have exactly one owner (measured: 1,210,103
|
|
63
|
+
of 1,443,450) and can never produce a pair, so a singleton never costs a
|
|
64
|
+
one-element list object; 1.21M of those were 77 MB of pure overhead.
|
|
65
|
+
"""
|
|
66
|
+
by_shingle: dict[int, int | list[int]] = {}
|
|
67
|
+
for idx, (_, shingles) in enumerate(indexed):
|
|
68
|
+
for s in shingles:
|
|
69
|
+
prev = by_shingle.get(s)
|
|
70
|
+
if prev is None:
|
|
71
|
+
by_shingle[s] = idx
|
|
72
|
+
elif type(prev) is int:
|
|
73
|
+
by_shingle[s] = [prev, idx]
|
|
74
|
+
else:
|
|
75
|
+
prev.append(idx)
|
|
76
|
+
return by_shingle
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _pairable_owners(indexed: list[tuple[InventoryRow, set[int]]]) -> list[list[int]]:
|
|
80
|
+
"""Only the shingles two or more functions share. Keeping just these lets
|
|
81
|
+
the whole index go before the pair counting below allocates anything."""
|
|
82
|
+
return [o for o in _owners_by_shingle(indexed).values() if type(o) is list]
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _shared_counts(indexed: list[tuple[InventoryRow, set[int]]]) -> dict[tuple[int, int], int]:
|
|
86
|
+
shared: dict[tuple[int, int], int] = {}
|
|
87
|
+
for owners in _pairable_owners(indexed):
|
|
88
|
+
for i, a in enumerate(owners):
|
|
89
|
+
for b in owners[i + 1:]:
|
|
90
|
+
shared[(a, b)] = shared.get((a, b), 0) + 1
|
|
91
|
+
return shared
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _pair_payload(a: InventoryRow, b: InventoryRow, similarity: float) -> dict:
|
|
95
|
+
functions = sorted(({"path": r.path, "long_name": r.long_name, "start": r.start,
|
|
96
|
+
"end": r.end, "nloc": r.nloc} for r in (a, b)),
|
|
97
|
+
key=lambda f: (f["path"], f["start"]))
|
|
98
|
+
return {"functions": functions, "similarity": round(similarity, 4)}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _twin_payload(r: InventoryRow, similarity: float, contained: bool) -> dict:
|
|
102
|
+
return {"path": r.path, "long_name": r.long_name, "start": r.start,
|
|
103
|
+
"end": r.end, "nloc": r.nloc, "similarity": round(similarity, 4),
|
|
104
|
+
"contained": contained}
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _encloses(outer, inner) -> bool:
|
|
108
|
+
return outer.start <= inner.start and inner.end <= outer.end
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _nested_spans(a, b) -> bool:
|
|
112
|
+
"""One span inside the other: nesting, not a clone.
|
|
113
|
+
|
|
114
|
+
A nested function's normalized lines are a subset of its enclosing
|
|
115
|
+
function's, so the pair scores 100% and reads as a perfect duplicate that
|
|
116
|
+
nobody can deduplicate. Only meaningful inside one file — the line numbers
|
|
117
|
+
of two different files never nest.
|
|
118
|
+
"""
|
|
119
|
+
return a.path == b.path and (_encloses(a, b) or _encloses(b, a))
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _is_self(r: InventoryRow, target) -> bool:
|
|
123
|
+
"""Path plus start line: no two functions in a file open on the same line."""
|
|
124
|
+
return r.path == target.path and r.start == target.start
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _target_shingles(target, sources: dict[str, str], min_lines: int) -> set[int] | None:
|
|
128
|
+
lines = _split_once(target.path, sources)
|
|
129
|
+
return None if lines is None else _row_shingles(target, lines, min_lines)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _twin_scores(mine: set[int], target, rows: list[InventoryRow],
|
|
133
|
+
sources: dict[str, str], min_lines: int) -> list[dict]:
|
|
134
|
+
return [_twin_payload(r, len(mine & other) / min(len(mine), len(other)),
|
|
135
|
+
_nested_spans(r, target))
|
|
136
|
+
for r, other in _function_shingles(rows, sources, min_lines)
|
|
137
|
+
if not _is_self(r, target)]
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def find_twins(target, rows: list[InventoryRow], sources: dict[str, str], *,
|
|
141
|
+
min_lines: int = 8, similarity: float = 0.8, top: int = 10) -> list[dict]:
|
|
142
|
+
"""The near-duplicates of ONE function, scored exactly as find_duplicates
|
|
143
|
+
scores the pair it would appear in.
|
|
144
|
+
|
|
145
|
+
One function's shingles against every other function's, so a brief costs a
|
|
146
|
+
single row's comparisons instead of the whole repo's pair counting.
|
|
147
|
+
"""
|
|
148
|
+
mine = _target_shingles(target, sources, min_lines)
|
|
149
|
+
if not mine:
|
|
150
|
+
return []
|
|
151
|
+
kept = [t for t in _twin_scores(mine, target, rows, sources, min_lines)
|
|
152
|
+
if t["similarity"] >= similarity]
|
|
153
|
+
kept.sort(key=lambda t: (-t["similarity"], t["path"], t["start"]))
|
|
154
|
+
return kept[:top]
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def find_duplicates(rows: list[InventoryRow], load_sources, *,
|
|
158
|
+
min_lines: int = 8, similarity: float = 0.8,
|
|
159
|
+
top: int = 50) -> list[dict]:
|
|
160
|
+
"""Every near-duplicate pair in a snapshot, best containment first.
|
|
161
|
+
|
|
162
|
+
`load_sources` is a CALLABLE returning {path: text}, not the texts
|
|
163
|
+
themselves. Shingles are ints: a file's text is dead the moment its rows are
|
|
164
|
+
indexed, and the pair counting below is where the heap actually goes. A
|
|
165
|
+
caller that bound those texts to a name would pin all of them across it —
|
|
166
|
+
measured 523 MB peak on a 104 MB repo against 377 MB with them released.
|
|
167
|
+
Passing the loader's dict straight into _function_shingles is what releases
|
|
168
|
+
it: nothing here ever holds a reference, so the index outlives the texts.
|
|
169
|
+
"""
|
|
170
|
+
indexed = _function_shingles(rows, load_sources(), min_lines)
|
|
171
|
+
pairs = []
|
|
172
|
+
for (ia, ib), count in _shared_counts(indexed).items():
|
|
173
|
+
a, b = indexed[ia], indexed[ib]
|
|
174
|
+
score = count / min(len(a[1]), len(b[1]))
|
|
175
|
+
if score >= similarity:
|
|
176
|
+
pairs.append(_pair_payload(a[0], b[0], score))
|
|
177
|
+
pairs.sort(key=lambda p: (-p["similarity"], p["functions"][0]["path"],
|
|
178
|
+
p["functions"][0]["start"]))
|
|
179
|
+
return pairs[:top]
|
crapkit/errors.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Failure classes with distinct exit codes. A broken pipeline never renders as a healthy zero."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class CrapkitError(Exception):
|
|
6
|
+
exit_code = 1
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class ConfigError(CrapkitError):
|
|
10
|
+
exit_code = 3
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class GitError(CrapkitError):
|
|
14
|
+
exit_code = 4
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ToolError(CrapkitError):
|
|
18
|
+
exit_code = 5
|