crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/dup.py ADDED
@@ -0,0 +1,179 @@
1
+ """Near-duplicate function detection. Pure: inventory rows + file texts in,
2
+ ranked pairs out.
3
+
4
+ Normalized line shingles with CONTAINMENT scoring (shared / smaller set), so a
5
+ copy-paste that later grew a few lines still surfaces. An inverted shingle
6
+ index keeps a 14k-function repo tractable: only pairs that actually share a
7
+ shingle are ever compared. Tiny functions are structural noise and stay out.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from .snapshot import InventoryRow
12
+
13
+ WINDOW = 4 # consecutive normalized lines per shingle
14
+ _COMMENT_PREFIXES = ("#", "//", "/*", "*", '"""', "'''")
15
+
16
+
17
+ def _normalized_lines(file_lines: list[str], start: int, end: int) -> list[str]:
18
+ picked = []
19
+ for raw in file_lines[start - 1:end]:
20
+ line = "".join(raw.split()) # whitespace never distinguishes a clone
21
+ if line and not raw.strip().startswith(_COMMENT_PREFIXES):
22
+ picked.append(line)
23
+ return picked
24
+
25
+
26
+ def _shingles(lines: list[str]) -> set[int]:
27
+ return {hash(tuple(lines[i:i + WINDOW])) for i in range(len(lines) - WINDOW + 1)}
28
+
29
+
30
+ def _split_once(path: str, sources: dict[str, str]) -> list[str] | None:
31
+ text = sources.get(path)
32
+ return None if text is None else text.splitlines()
33
+
34
+
35
+ def _row_shingles(r: InventoryRow, file_lines: list[str], min_lines: int) -> set[int] | None:
36
+ lines = _normalized_lines(file_lines, r.start, r.end)
37
+ return _shingles(lines) if len(lines) >= min_lines else None
38
+
39
+
40
+ def _function_shingles(rows: list[InventoryRow], sources: dict[str, str],
41
+ min_lines: int) -> list[tuple[InventoryRow, set[int]]]:
42
+ # Rows arrive ordered by (scope, path, start), so every row of a file is
43
+ # contiguous: a ONE-ENTRY cache splits each source once instead of once per
44
+ # function in it (measured 3 GB of re-split text on a 104 MB repo). A file
45
+ # whose rows are NOT contiguous still scores identically, just re-split.
46
+ out = []
47
+ cached_path, file_lines = None, None
48
+ for r in rows:
49
+ if r.path != cached_path:
50
+ cached_path, file_lines = r.path, _split_once(r.path, sources)
51
+ if file_lines is None:
52
+ continue
53
+ shingles = _row_shingles(r, file_lines, min_lines)
54
+ if shingles is not None:
55
+ out.append((r, shingles))
56
+ return out
57
+
58
+
59
+ def _owners_by_shingle(indexed: list[tuple[InventoryRow, set[int]]]) -> dict[int, int | list[int]]:
60
+ """Inverted shingle index: a lone owner stays a bare int, a list starts at two.
61
+
62
+ 84% of shingles in a real repo have exactly one owner (measured: 1,210,103
63
+ of 1,443,450) and can never produce a pair, so a singleton never costs a
64
+ one-element list object; 1.21M of those were 77 MB of pure overhead.
65
+ """
66
+ by_shingle: dict[int, int | list[int]] = {}
67
+ for idx, (_, shingles) in enumerate(indexed):
68
+ for s in shingles:
69
+ prev = by_shingle.get(s)
70
+ if prev is None:
71
+ by_shingle[s] = idx
72
+ elif type(prev) is int:
73
+ by_shingle[s] = [prev, idx]
74
+ else:
75
+ prev.append(idx)
76
+ return by_shingle
77
+
78
+
79
+ def _pairable_owners(indexed: list[tuple[InventoryRow, set[int]]]) -> list[list[int]]:
80
+ """Only the shingles two or more functions share. Keeping just these lets
81
+ the whole index go before the pair counting below allocates anything."""
82
+ return [o for o in _owners_by_shingle(indexed).values() if type(o) is list]
83
+
84
+
85
+ def _shared_counts(indexed: list[tuple[InventoryRow, set[int]]]) -> dict[tuple[int, int], int]:
86
+ shared: dict[tuple[int, int], int] = {}
87
+ for owners in _pairable_owners(indexed):
88
+ for i, a in enumerate(owners):
89
+ for b in owners[i + 1:]:
90
+ shared[(a, b)] = shared.get((a, b), 0) + 1
91
+ return shared
92
+
93
+
94
+ def _pair_payload(a: InventoryRow, b: InventoryRow, similarity: float) -> dict:
95
+ functions = sorted(({"path": r.path, "long_name": r.long_name, "start": r.start,
96
+ "end": r.end, "nloc": r.nloc} for r in (a, b)),
97
+ key=lambda f: (f["path"], f["start"]))
98
+ return {"functions": functions, "similarity": round(similarity, 4)}
99
+
100
+
101
+ def _twin_payload(r: InventoryRow, similarity: float, contained: bool) -> dict:
102
+ return {"path": r.path, "long_name": r.long_name, "start": r.start,
103
+ "end": r.end, "nloc": r.nloc, "similarity": round(similarity, 4),
104
+ "contained": contained}
105
+
106
+
107
+ def _encloses(outer, inner) -> bool:
108
+ return outer.start <= inner.start and inner.end <= outer.end
109
+
110
+
111
+ def _nested_spans(a, b) -> bool:
112
+ """One span inside the other: nesting, not a clone.
113
+
114
+ A nested function's normalized lines are a subset of its enclosing
115
+ function's, so the pair scores 100% and reads as a perfect duplicate that
116
+ nobody can deduplicate. Only meaningful inside one file — the line numbers
117
+ of two different files never nest.
118
+ """
119
+ return a.path == b.path and (_encloses(a, b) or _encloses(b, a))
120
+
121
+
122
+ def _is_self(r: InventoryRow, target) -> bool:
123
+ """Path plus start line: no two functions in a file open on the same line."""
124
+ return r.path == target.path and r.start == target.start
125
+
126
+
127
+ def _target_shingles(target, sources: dict[str, str], min_lines: int) -> set[int] | None:
128
+ lines = _split_once(target.path, sources)
129
+ return None if lines is None else _row_shingles(target, lines, min_lines)
130
+
131
+
132
+ def _twin_scores(mine: set[int], target, rows: list[InventoryRow],
133
+ sources: dict[str, str], min_lines: int) -> list[dict]:
134
+ return [_twin_payload(r, len(mine & other) / min(len(mine), len(other)),
135
+ _nested_spans(r, target))
136
+ for r, other in _function_shingles(rows, sources, min_lines)
137
+ if not _is_self(r, target)]
138
+
139
+
140
+ def find_twins(target, rows: list[InventoryRow], sources: dict[str, str], *,
141
+ min_lines: int = 8, similarity: float = 0.8, top: int = 10) -> list[dict]:
142
+ """The near-duplicates of ONE function, scored exactly as find_duplicates
143
+ scores the pair it would appear in.
144
+
145
+ One function's shingles against every other function's, so a brief costs a
146
+ single row's comparisons instead of the whole repo's pair counting.
147
+ """
148
+ mine = _target_shingles(target, sources, min_lines)
149
+ if not mine:
150
+ return []
151
+ kept = [t for t in _twin_scores(mine, target, rows, sources, min_lines)
152
+ if t["similarity"] >= similarity]
153
+ kept.sort(key=lambda t: (-t["similarity"], t["path"], t["start"]))
154
+ return kept[:top]
155
+
156
+
157
+ def find_duplicates(rows: list[InventoryRow], load_sources, *,
158
+ min_lines: int = 8, similarity: float = 0.8,
159
+ top: int = 50) -> list[dict]:
160
+ """Every near-duplicate pair in a snapshot, best containment first.
161
+
162
+ `load_sources` is a CALLABLE returning {path: text}, not the texts
163
+ themselves. Shingles are ints: a file's text is dead the moment its rows are
164
+ indexed, and the pair counting below is where the heap actually goes. A
165
+ caller that bound those texts to a name would pin all of them across it —
166
+ measured 523 MB peak on a 104 MB repo against 377 MB with them released.
167
+ Passing the loader's dict straight into _function_shingles is what releases
168
+ it: nothing here ever holds a reference, so the index outlives the texts.
169
+ """
170
+ indexed = _function_shingles(rows, load_sources(), min_lines)
171
+ pairs = []
172
+ for (ia, ib), count in _shared_counts(indexed).items():
173
+ a, b = indexed[ia], indexed[ib]
174
+ score = count / min(len(a[1]), len(b[1]))
175
+ if score >= similarity:
176
+ pairs.append(_pair_payload(a[0], b[0], score))
177
+ pairs.sort(key=lambda p: (-p["similarity"], p["functions"][0]["path"],
178
+ p["functions"][0]["start"]))
179
+ return pairs[:top]
crapkit/errors.py ADDED
@@ -0,0 +1,18 @@
1
+ """Failure classes with distinct exit codes. A broken pipeline never renders as a healthy zero."""
2
+ from __future__ import annotations
3
+
4
+
5
+ class CrapkitError(Exception):
6
+ exit_code = 1
7
+
8
+
9
+ class ConfigError(CrapkitError):
10
+ exit_code = 3
11
+
12
+
13
+ class GitError(CrapkitError):
14
+ exit_code = 4
15
+
16
+
17
+ class ToolError(CrapkitError):
18
+ exit_code = 5