crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/worklist.py
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
"""Ranked worklist, dormant-risk list, session claims and batch splitting. Pure.
|
|
2
|
+
|
|
3
|
+
Active = churned files ranked by scored complexity (CRAP once coverage lanes
|
|
4
|
+
exist; min-CCN until then) with a floor so trivial functions never appear.
|
|
5
|
+
Dormant = the same scoring over files with zero churn in the window, kept
|
|
6
|
+
separate so sleeping hazards are recorded without clogging the queue.
|
|
7
|
+
|
|
8
|
+
`Admission` is the one rule behind both this list and next-item's queue. It
|
|
9
|
+
answers a single question — is this row a candidate? — and the ceiling half of
|
|
10
|
+
the answer outranks the floor: debt over its ceiling is admissible at any ccn.
|
|
11
|
+
|
|
12
|
+
Shared admission, different views. This list ranks every candidate by risk,
|
|
13
|
+
finished rows and no-lane rows included, so it never empties and never hides a
|
|
14
|
+
hazard. next-item ranks by CRAP descending, drops what no lane measures, and
|
|
15
|
+
reports empty once nothing it ranks has work left. Each entry carries the run's
|
|
16
|
+
`flag` and `remedy` so a row says which view it belongs to.
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from collections.abc import Mapping
|
|
21
|
+
from types import MappingProxyType
|
|
22
|
+
from typing import NamedTuple
|
|
23
|
+
|
|
24
|
+
from .churn import FileChurn
|
|
25
|
+
from .snapshot import InventoryRow
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
HOT_MIN_CCN = 3 # hot promotion reaches no lower than this, whatever the floor
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def over_target_floor(min_ceiling: int) -> int:
|
|
32
|
+
"""The lowest ccn that can possibly score over `min_ceiling`.
|
|
33
|
+
|
|
34
|
+
CRAP is ccn^2 * (1 - cov)^3 + ccn, worst at cov 0 where it is ccn^2 + ccn.
|
|
35
|
+
A ccn whose worst case lands at or under the smallest ceiling in the config
|
|
36
|
+
can never be debt, whatever its coverage; every larger ccn can.
|
|
37
|
+
"""
|
|
38
|
+
if min_ceiling < 1:
|
|
39
|
+
raise ValueError(f"ceiling must be >= 1, got {min_ceiling}")
|
|
40
|
+
ccn = 1
|
|
41
|
+
while ccn * ccn + ccn <= min_ceiling:
|
|
42
|
+
ccn += 1
|
|
43
|
+
return ccn
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def sql_floor(floor: int, min_ceiling: int | None = None) -> int:
|
|
47
|
+
"""The lowest ccn any admission rule here can let through.
|
|
48
|
+
|
|
49
|
+
A reader may leave everything below this in SQLite: no rule can admit it, so
|
|
50
|
+
fetching it is pure cost. `min_ceiling` is the smallest ceiling the caller
|
|
51
|
+
can score a row against, and None only for a caller that has no scored run
|
|
52
|
+
to read a remedy from. Passing it is load-bearing: a floor of 5 pushed into
|
|
53
|
+
a read hid every ccn-4 function at 0% coverage, and those score CRAP 20
|
|
54
|
+
against a target of 6. Both queues pass it — the worklist left it out, and
|
|
55
|
+
its rows were gone before the admission rule ever saw them.
|
|
56
|
+
"""
|
|
57
|
+
reach = [floor, HOT_MIN_CCN]
|
|
58
|
+
if min_ceiling is not None:
|
|
59
|
+
reach.append(over_target_floor(min_ceiling))
|
|
60
|
+
return min(reach)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
_NO_CHURN = FileChurn(commits=0, authors=0)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class Admission(NamedTuple):
|
|
67
|
+
"""What counts as a queue candidate, bound to one run's churn.
|
|
68
|
+
|
|
69
|
+
The worklist and next-item both ask this object, so the two can never drift
|
|
70
|
+
on the floor, on hot promotion, or on debt — which is what let the worklist
|
|
71
|
+
rank a function next-item was reporting as `below_floor`.
|
|
72
|
+
"""
|
|
73
|
+
churn: dict[str, FileChurn]
|
|
74
|
+
floor: int
|
|
75
|
+
hot: float | None
|
|
76
|
+
|
|
77
|
+
def of(self, path: str) -> FileChurn:
|
|
78
|
+
return self.churn.get(path, _NO_CHURN)
|
|
79
|
+
|
|
80
|
+
def admits(self, path: str, ccn: int, *, over_target: bool) -> bool:
|
|
81
|
+
"""Over-target debt is work at any complexity: a ccn-4 function at 0%
|
|
82
|
+
coverage scores CRAP 20 against a ceiling of 6. The floor and the hot
|
|
83
|
+
promotion only ADD candidates the ceiling did not catch; neither may
|
|
84
|
+
veto a function that is already over its ceiling.
|
|
85
|
+
"""
|
|
86
|
+
if over_target:
|
|
87
|
+
return True
|
|
88
|
+
if self.hot is not None and self.of(path).weight >= self.hot and ccn >= HOT_MIN_CCN:
|
|
89
|
+
return True # hot simple code: the floor must not hide it
|
|
90
|
+
return ccn >= self.floor
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def admission(churn: dict[str, FileChurn], floor: int) -> Admission:
|
|
94
|
+
return Admission(churn, floor, _hot_threshold(churn))
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class WorklistEntry(NamedTuple):
|
|
98
|
+
scope: str
|
|
99
|
+
path: str
|
|
100
|
+
long_name: str
|
|
101
|
+
start: int
|
|
102
|
+
end: int
|
|
103
|
+
ccn: int
|
|
104
|
+
ccn_std: int
|
|
105
|
+
nloc: int
|
|
106
|
+
commits: int
|
|
107
|
+
authors: int
|
|
108
|
+
weight: float
|
|
109
|
+
risk: float # ccn x recency-weighted churn: the composite hotspot rank
|
|
110
|
+
# The scoring run's verdict on this row, so the risk map can say which of
|
|
111
|
+
# its rows the burn-down queue will never hand out and which are finished.
|
|
112
|
+
# Both None on an inventory-only run, which scored no verdict to report.
|
|
113
|
+
flag: str | None = None
|
|
114
|
+
remedy: str | None = None
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class Worklist(NamedTuple):
|
|
118
|
+
active: list[WorklistEntry]
|
|
119
|
+
dormant: list[WorklistEntry]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _entry(r: InventoryRow, churn: FileChurn, mark: tuple) -> WorklistEntry:
|
|
123
|
+
return WorklistEntry(r.scope, r.path, r.long_name, r.start, r.end,
|
|
124
|
+
r.ccn, r.ccn_std, r.nloc, churn.commits, churn.authors,
|
|
125
|
+
churn.weight, round(r.ccn * churn.weight, 4), *mark)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _rank_key(e: WorklistEntry):
|
|
129
|
+
# Composite hotspot rank; equal risk (e.g. weightless churn data) falls
|
|
130
|
+
# back to the old ccn-then-commits order.
|
|
131
|
+
return (-e.risk, -e.ccn, -e.commits, e.path, e.start)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _hot_threshold(churn: dict[str, FileChurn]) -> float | None:
|
|
135
|
+
"""The 90th-percentile file weight: candidacy promotion for hot simple code.
|
|
136
|
+
|
|
137
|
+
Applying the ccn floor before churn hides a heavily-edited simple file
|
|
138
|
+
(Tornhill's ordering). Needs real weights and enough files to mean anything.
|
|
139
|
+
"""
|
|
140
|
+
weights = sorted(c.weight for c in churn.values() if c.weight > 0)
|
|
141
|
+
if len(weights) < 5:
|
|
142
|
+
return None
|
|
143
|
+
return weights[int(0.9 * (len(weights) - 1))]
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _at_ceiling(scored, target: int, scope_targets: dict[str, int] | None) -> set:
|
|
147
|
+
"""(path, long_name) of every scored function at or under its own ceiling."""
|
|
148
|
+
ceilings = scope_targets or {}
|
|
149
|
+
return {(r.path, r.long_name) for r in scored
|
|
150
|
+
if r.crap is not None and r.crap <= ceilings.get(r.scope, target)}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def closable_claims(claims: list[dict], scored: list, *, target: int,
|
|
154
|
+
scope_targets: dict[str, int] | None, stale_commits: set[str]) -> list[int]:
|
|
155
|
+
"""Claim ids a verify may release, sorted.
|
|
156
|
+
|
|
157
|
+
Two ways to be done, and no third: the function now scores at or under its
|
|
158
|
+
scope ceiling, or the commit the claim was taken on is no longer an ancestor
|
|
159
|
+
of HEAD (an amend or rebase took the session's tree with it). A function the
|
|
160
|
+
run never scored keeps its claim — absence is not evidence of a fix.
|
|
161
|
+
"""
|
|
162
|
+
done = _at_ceiling(scored, target, scope_targets)
|
|
163
|
+
return sorted(c["id"] for c in claims
|
|
164
|
+
if (c["path"], c["long_name"]) in done or c["commit"] in stale_commits)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
_NO_MARK = (None, None)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def build_worklist(
|
|
171
|
+
rows: list[InventoryRow],
|
|
172
|
+
churn: dict[str, FileChurn],
|
|
173
|
+
*,
|
|
174
|
+
floor: int,
|
|
175
|
+
top: int,
|
|
176
|
+
marks: Mapping[tuple[str, str], tuple[str, str]] = MappingProxyType({}),
|
|
177
|
+
) -> Worklist:
|
|
178
|
+
"""The risk map: every admitted function, ranked, whatever its score.
|
|
179
|
+
|
|
180
|
+
`marks` is (flag, remedy) per function the same run scored —
|
|
181
|
+
`SnapshotStore.read_marks`. Inventory rows carry no coverage, so the ceiling
|
|
182
|
+
half of the admission cannot be decided from `rows` alone, and an
|
|
183
|
+
inventory-only run passes nothing here. The marks ride onto the entries as
|
|
184
|
+
well: this list ranks rows the burn-down queue declines, and a row that says
|
|
185
|
+
nothing about which it is sends an agent to work a wiring gap.
|
|
186
|
+
"""
|
|
187
|
+
if top < 1:
|
|
188
|
+
raise ValueError(f"worklist top must be >= 1, got {top}")
|
|
189
|
+
adm = admission(churn, floor)
|
|
190
|
+
active, dormant = [], []
|
|
191
|
+
for r in rows:
|
|
192
|
+
c, mark = adm.of(r.path), marks.get((r.path, r.long_name), _NO_MARK)
|
|
193
|
+
if not adm.admits(r.path, r.ccn, over_target=mark[1] not in (None, "ok")):
|
|
194
|
+
continue
|
|
195
|
+
(active if c.commits > 0 else dormant).append(_entry(r, c, mark))
|
|
196
|
+
active.sort(key=_rank_key)
|
|
197
|
+
dormant.sort(key=_rank_key)
|
|
198
|
+
return Worklist(active=active[:top], dormant=dormant)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
BATCH_CONTAINMENT = 0.5 # half of one file's commits also touch the other
|
|
202
|
+
BATCH_PAIR_LIMIT = 1000 # coupled pairs read; far more than worklist_top files can bind
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
class Batch(NamedTuple):
|
|
206
|
+
files: list[str]
|
|
207
|
+
entries: list[WorklistEntry]
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _find(rep: dict[str, str], f: str) -> str:
|
|
211
|
+
while rep.get(f, f) != f:
|
|
212
|
+
f = rep[f]
|
|
213
|
+
return f
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _merge(rep: dict[str, str], a: str, b: str) -> None:
|
|
217
|
+
ra, rb = _find(rep, a), _find(rep, b)
|
|
218
|
+
if ra == rb:
|
|
219
|
+
return
|
|
220
|
+
lo, hi = sorted((ra, rb))
|
|
221
|
+
rep[hi] = lo # the lexically first path represents the group, whatever order pairs arrive in
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _coupling_groups(pairs: list[dict]) -> dict[str, str]:
|
|
225
|
+
"""file -> group representative, merging every pair at or above the threshold.
|
|
226
|
+
|
|
227
|
+
Coupling is transitive here: a-b and b-c put all three in one batch. That is
|
|
228
|
+
deliberate, and it is why the threshold is high — a chain of weak links would
|
|
229
|
+
drag the whole repo into a single unit.
|
|
230
|
+
"""
|
|
231
|
+
rep: dict[str, str] = {}
|
|
232
|
+
for p in pairs:
|
|
233
|
+
if p["confidence"] >= BATCH_CONTAINMENT:
|
|
234
|
+
_merge(rep, *p["files"])
|
|
235
|
+
return rep
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _unit_key(item) -> tuple:
|
|
239
|
+
key, entries = item
|
|
240
|
+
return (-max(e.risk for e in entries), key)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _units(active: list[WorklistEntry], rep: dict[str, str]) -> list[list[WorklistEntry]]:
|
|
244
|
+
"""Entries grouped by coupling group, heaviest first: what a batch takes whole."""
|
|
245
|
+
groups: dict[str, list] = {}
|
|
246
|
+
for e in active:
|
|
247
|
+
groups.setdefault(_find(rep, e.path), []).append(e)
|
|
248
|
+
return [entries for _, entries in sorted(groups.items(), key=_unit_key)]
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _risk(entries: list[WorklistEntry]) -> float:
|
|
252
|
+
return sum(e.risk for e in entries)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _bin_key(entries: list[WorklistEntry]) -> tuple:
|
|
256
|
+
"""Least risk wins; on a tie the emptiest batch does, so a worklist whose
|
|
257
|
+
risks all round to zero still spreads instead of piling into the first."""
|
|
258
|
+
return (_risk(entries), len(entries))
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _empty_bins(n: int) -> list[list]:
|
|
262
|
+
return [[] for _ in range(n)]
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _as_batch(entries: list[WorklistEntry]) -> Batch:
|
|
266
|
+
return Batch(files=sorted({e.path for e in entries}),
|
|
267
|
+
entries=sorted(entries, key=_rank_key))
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _batch_key(b: Batch) -> tuple:
|
|
271
|
+
return (-_risk(b.entries), b.files)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def split_batches(active: list[WorklistEntry], pairs: list[dict], *,
|
|
275
|
+
batches: int) -> list[Batch]:
|
|
276
|
+
"""The active worklist cut into at most `batches` sets of whole files.
|
|
277
|
+
|
|
278
|
+
Two agents editing one file collide in the worktree whatever the ranking
|
|
279
|
+
says, so a file is indivisible and so is a coupled group of files: the batch
|
|
280
|
+
holds every entry from every file in the group. Units go to the lightest
|
|
281
|
+
batch in risk order, which is LPT scheduling, and the whole thing is a pure
|
|
282
|
+
function of the worklist and the coupling pairs — same store, same history,
|
|
283
|
+
same batches. Empty batches are dropped rather than handed to an agent.
|
|
284
|
+
"""
|
|
285
|
+
if batches < 1:
|
|
286
|
+
raise ValueError(f"batches must be >= 1, got {batches}")
|
|
287
|
+
bins = _empty_bins(batches)
|
|
288
|
+
for unit in _units(active, _coupling_groups(pairs)):
|
|
289
|
+
min(bins, key=_bin_key).extend(unit)
|
|
290
|
+
return sorted((_as_batch(b) for b in bins if b), key=_batch_key)
|