crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/worklist.py ADDED
@@ -0,0 +1,290 @@
1
+ """Ranked worklist, dormant-risk list, session claims and batch splitting. Pure.
2
+
3
+ Active = churned files ranked by scored complexity (CRAP once coverage lanes
4
+ exist; min-CCN until then) with a floor so trivial functions never appear.
5
+ Dormant = the same scoring over files with zero churn in the window, kept
6
+ separate so sleeping hazards are recorded without clogging the queue.
7
+
8
+ `Admission` is the one rule behind both this list and next-item's queue. It
9
+ answers a single question — is this row a candidate? — and the ceiling half of
10
+ the answer outranks the floor: debt over its ceiling is admissible at any ccn.
11
+
12
+ Shared admission, different views. This list ranks every candidate by risk,
13
+ finished rows and no-lane rows included, so it never empties and never hides a
14
+ hazard. next-item ranks by CRAP descending, drops what no lane measures, and
15
+ reports empty once nothing it ranks has work left. Each entry carries the run's
16
+ `flag` and `remedy` so a row says which view it belongs to.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ from collections.abc import Mapping
21
+ from types import MappingProxyType
22
+ from typing import NamedTuple
23
+
24
+ from .churn import FileChurn
25
+ from .snapshot import InventoryRow
26
+
27
+
28
+ HOT_MIN_CCN = 3 # hot promotion reaches no lower than this, whatever the floor
29
+
30
+
31
+ def over_target_floor(min_ceiling: int) -> int:
32
+ """The lowest ccn that can possibly score over `min_ceiling`.
33
+
34
+ CRAP is ccn^2 * (1 - cov)^3 + ccn, worst at cov 0 where it is ccn^2 + ccn.
35
+ A ccn whose worst case lands at or under the smallest ceiling in the config
36
+ can never be debt, whatever its coverage; every larger ccn can.
37
+ """
38
+ if min_ceiling < 1:
39
+ raise ValueError(f"ceiling must be >= 1, got {min_ceiling}")
40
+ ccn = 1
41
+ while ccn * ccn + ccn <= min_ceiling:
42
+ ccn += 1
43
+ return ccn
44
+
45
+
46
+ def sql_floor(floor: int, min_ceiling: int | None = None) -> int:
47
+ """The lowest ccn any admission rule here can let through.
48
+
49
+ A reader may leave everything below this in SQLite: no rule can admit it, so
50
+ fetching it is pure cost. `min_ceiling` is the smallest ceiling the caller
51
+ can score a row against, and None only for a caller that has no scored run
52
+ to read a remedy from. Passing it is load-bearing: a floor of 5 pushed into
53
+ a read hid every ccn-4 function at 0% coverage, and those score CRAP 20
54
+ against a target of 6. Both queues pass it — the worklist left it out, and
55
+ its rows were gone before the admission rule ever saw them.
56
+ """
57
+ reach = [floor, HOT_MIN_CCN]
58
+ if min_ceiling is not None:
59
+ reach.append(over_target_floor(min_ceiling))
60
+ return min(reach)
61
+
62
+
63
+ _NO_CHURN = FileChurn(commits=0, authors=0)
64
+
65
+
66
+ class Admission(NamedTuple):
67
+ """What counts as a queue candidate, bound to one run's churn.
68
+
69
+ The worklist and next-item both ask this object, so the two can never drift
70
+ on the floor, on hot promotion, or on debt — which is what let the worklist
71
+ rank a function next-item was reporting as `below_floor`.
72
+ """
73
+ churn: dict[str, FileChurn]
74
+ floor: int
75
+ hot: float | None
76
+
77
+ def of(self, path: str) -> FileChurn:
78
+ return self.churn.get(path, _NO_CHURN)
79
+
80
+ def admits(self, path: str, ccn: int, *, over_target: bool) -> bool:
81
+ """Over-target debt is work at any complexity: a ccn-4 function at 0%
82
+ coverage scores CRAP 20 against a ceiling of 6. The floor and the hot
83
+ promotion only ADD candidates the ceiling did not catch; neither may
84
+ veto a function that is already over its ceiling.
85
+ """
86
+ if over_target:
87
+ return True
88
+ if self.hot is not None and self.of(path).weight >= self.hot and ccn >= HOT_MIN_CCN:
89
+ return True # hot simple code: the floor must not hide it
90
+ return ccn >= self.floor
91
+
92
+
93
+ def admission(churn: dict[str, FileChurn], floor: int) -> Admission:
94
+ return Admission(churn, floor, _hot_threshold(churn))
95
+
96
+
97
+ class WorklistEntry(NamedTuple):
98
+ scope: str
99
+ path: str
100
+ long_name: str
101
+ start: int
102
+ end: int
103
+ ccn: int
104
+ ccn_std: int
105
+ nloc: int
106
+ commits: int
107
+ authors: int
108
+ weight: float
109
+ risk: float # ccn x recency-weighted churn: the composite hotspot rank
110
+ # The scoring run's verdict on this row, so the risk map can say which of
111
+ # its rows the burn-down queue will never hand out and which are finished.
112
+ # Both None on an inventory-only run, which scored no verdict to report.
113
+ flag: str | None = None
114
+ remedy: str | None = None
115
+
116
+
117
+ class Worklist(NamedTuple):
118
+ active: list[WorklistEntry]
119
+ dormant: list[WorklistEntry]
120
+
121
+
122
+ def _entry(r: InventoryRow, churn: FileChurn, mark: tuple) -> WorklistEntry:
123
+ return WorklistEntry(r.scope, r.path, r.long_name, r.start, r.end,
124
+ r.ccn, r.ccn_std, r.nloc, churn.commits, churn.authors,
125
+ churn.weight, round(r.ccn * churn.weight, 4), *mark)
126
+
127
+
128
+ def _rank_key(e: WorklistEntry):
129
+ # Composite hotspot rank; equal risk (e.g. weightless churn data) falls
130
+ # back to the old ccn-then-commits order.
131
+ return (-e.risk, -e.ccn, -e.commits, e.path, e.start)
132
+
133
+
134
+ def _hot_threshold(churn: dict[str, FileChurn]) -> float | None:
135
+ """The 90th-percentile file weight: candidacy promotion for hot simple code.
136
+
137
+ Applying the ccn floor before churn hides a heavily-edited simple file
138
+ (Tornhill's ordering). Needs real weights and enough files to mean anything.
139
+ """
140
+ weights = sorted(c.weight for c in churn.values() if c.weight > 0)
141
+ if len(weights) < 5:
142
+ return None
143
+ return weights[int(0.9 * (len(weights) - 1))]
144
+
145
+
146
+ def _at_ceiling(scored, target: int, scope_targets: dict[str, int] | None) -> set:
147
+ """(path, long_name) of every scored function at or under its own ceiling."""
148
+ ceilings = scope_targets or {}
149
+ return {(r.path, r.long_name) for r in scored
150
+ if r.crap is not None and r.crap <= ceilings.get(r.scope, target)}
151
+
152
+
153
+ def closable_claims(claims: list[dict], scored: list, *, target: int,
154
+ scope_targets: dict[str, int] | None, stale_commits: set[str]) -> list[int]:
155
+ """Claim ids a verify may release, sorted.
156
+
157
+ Two ways to be done, and no third: the function now scores at or under its
158
+ scope ceiling, or the commit the claim was taken on is no longer an ancestor
159
+ of HEAD (an amend or rebase took the session's tree with it). A function the
160
+ run never scored keeps its claim — absence is not evidence of a fix.
161
+ """
162
+ done = _at_ceiling(scored, target, scope_targets)
163
+ return sorted(c["id"] for c in claims
164
+ if (c["path"], c["long_name"]) in done or c["commit"] in stale_commits)
165
+
166
+
167
+ _NO_MARK = (None, None)
168
+
169
+
170
+ def build_worklist(
171
+ rows: list[InventoryRow],
172
+ churn: dict[str, FileChurn],
173
+ *,
174
+ floor: int,
175
+ top: int,
176
+ marks: Mapping[tuple[str, str], tuple[str, str]] = MappingProxyType({}),
177
+ ) -> Worklist:
178
+ """The risk map: every admitted function, ranked, whatever its score.
179
+
180
+ `marks` is (flag, remedy) per function the same run scored —
181
+ `SnapshotStore.read_marks`. Inventory rows carry no coverage, so the ceiling
182
+ half of the admission cannot be decided from `rows` alone, and an
183
+ inventory-only run passes nothing here. The marks ride onto the entries as
184
+ well: this list ranks rows the burn-down queue declines, and a row that says
185
+ nothing about which it is sends an agent to work a wiring gap.
186
+ """
187
+ if top < 1:
188
+ raise ValueError(f"worklist top must be >= 1, got {top}")
189
+ adm = admission(churn, floor)
190
+ active, dormant = [], []
191
+ for r in rows:
192
+ c, mark = adm.of(r.path), marks.get((r.path, r.long_name), _NO_MARK)
193
+ if not adm.admits(r.path, r.ccn, over_target=mark[1] not in (None, "ok")):
194
+ continue
195
+ (active if c.commits > 0 else dormant).append(_entry(r, c, mark))
196
+ active.sort(key=_rank_key)
197
+ dormant.sort(key=_rank_key)
198
+ return Worklist(active=active[:top], dormant=dormant)
199
+
200
+
201
+ BATCH_CONTAINMENT = 0.5 # half of one file's commits also touch the other
202
+ BATCH_PAIR_LIMIT = 1000 # coupled pairs read; far more than worklist_top files can bind
203
+
204
+
205
+ class Batch(NamedTuple):
206
+ files: list[str]
207
+ entries: list[WorklistEntry]
208
+
209
+
210
+ def _find(rep: dict[str, str], f: str) -> str:
211
+ while rep.get(f, f) != f:
212
+ f = rep[f]
213
+ return f
214
+
215
+
216
+ def _merge(rep: dict[str, str], a: str, b: str) -> None:
217
+ ra, rb = _find(rep, a), _find(rep, b)
218
+ if ra == rb:
219
+ return
220
+ lo, hi = sorted((ra, rb))
221
+ rep[hi] = lo # the lexically first path represents the group, whatever order pairs arrive in
222
+
223
+
224
+ def _coupling_groups(pairs: list[dict]) -> dict[str, str]:
225
+ """file -> group representative, merging every pair at or above the threshold.
226
+
227
+ Coupling is transitive here: a-b and b-c put all three in one batch. That is
228
+ deliberate, and it is why the threshold is high — a chain of weak links would
229
+ drag the whole repo into a single unit.
230
+ """
231
+ rep: dict[str, str] = {}
232
+ for p in pairs:
233
+ if p["confidence"] >= BATCH_CONTAINMENT:
234
+ _merge(rep, *p["files"])
235
+ return rep
236
+
237
+
238
+ def _unit_key(item) -> tuple:
239
+ key, entries = item
240
+ return (-max(e.risk for e in entries), key)
241
+
242
+
243
+ def _units(active: list[WorklistEntry], rep: dict[str, str]) -> list[list[WorklistEntry]]:
244
+ """Entries grouped by coupling group, heaviest first: what a batch takes whole."""
245
+ groups: dict[str, list] = {}
246
+ for e in active:
247
+ groups.setdefault(_find(rep, e.path), []).append(e)
248
+ return [entries for _, entries in sorted(groups.items(), key=_unit_key)]
249
+
250
+
251
+ def _risk(entries: list[WorklistEntry]) -> float:
252
+ return sum(e.risk for e in entries)
253
+
254
+
255
+ def _bin_key(entries: list[WorklistEntry]) -> tuple:
256
+ """Least risk wins; on a tie the emptiest batch does, so a worklist whose
257
+ risks all round to zero still spreads instead of piling into the first."""
258
+ return (_risk(entries), len(entries))
259
+
260
+
261
+ def _empty_bins(n: int) -> list[list]:
262
+ return [[] for _ in range(n)]
263
+
264
+
265
+ def _as_batch(entries: list[WorklistEntry]) -> Batch:
266
+ return Batch(files=sorted({e.path for e in entries}),
267
+ entries=sorted(entries, key=_rank_key))
268
+
269
+
270
+ def _batch_key(b: Batch) -> tuple:
271
+ return (-_risk(b.entries), b.files)
272
+
273
+
274
+ def split_batches(active: list[WorklistEntry], pairs: list[dict], *,
275
+ batches: int) -> list[Batch]:
276
+ """The active worklist cut into at most `batches` sets of whole files.
277
+
278
+ Two agents editing one file collide in the worktree whatever the ranking
279
+ says, so a file is indivisible and so is a coupled group of files: the batch
280
+ holds every entry from every file in the group. Units go to the lightest
281
+ batch in risk order, which is LPT scheduling, and the whole thing is a pure
282
+ function of the worklist and the coupling pairs — same store, same history,
283
+ same batches. Empty batches are dropped rather than handed to an agent.
284
+ """
285
+ if batches < 1:
286
+ raise ValueError(f"batches must be >= 1, got {batches}")
287
+ bins = _empty_bins(batches)
288
+ for unit in _units(active, _coupling_groups(pairs)):
289
+ min(bins, key=_bin_key).extend(unit)
290
+ return sorted((_as_batch(b) for b in bins if b), key=_batch_key)