crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/scaffold.py
ADDED
|
@@ -0,0 +1,361 @@
|
|
|
1
|
+
"""Repo sniffing for `crapkit init`: tracked files in, a starter crapkit.toml out. Pure."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
from typing import NamedTuple
|
|
6
|
+
|
|
7
|
+
from .universe import LANGUAGE_EXTENSIONS, exclude_matcher, excluded
|
|
8
|
+
|
|
9
|
+
_EXT_LANGUAGE = {ext: lang for lang, exts in LANGUAGE_EXTENSIONS.items() for ext in exts}
|
|
10
|
+
|
|
11
|
+
DEFAULT_EXCLUDES = (
|
|
12
|
+
"**/node_modules/**", "**/dist/**", "**/build/**", "**/vendor/**",
|
|
13
|
+
"**/*.test.*", "**/*.spec.*", "**/test_*.py", "**/*_test.py", "**/conftest.py",
|
|
14
|
+
# runner config files the docs themselves tell users to create; globs are
|
|
15
|
+
# whole-path, so the root form and the nested form are both required
|
|
16
|
+
"*.config.ts", "*.config.js", "*.config.mts",
|
|
17
|
+
"**/*.config.ts", "**/*.config.js", "**/*.config.mts",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
_MATCH_DEFAULT_EXCLUDE = exclude_matcher(DEFAULT_EXCLUDES)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _language_of(path: str) -> str | None:
|
|
24
|
+
for ext, lang in _EXT_LANGUAGE.items():
|
|
25
|
+
if path.endswith(ext):
|
|
26
|
+
return lang
|
|
27
|
+
return None
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _scoped_source(path: str) -> str | None:
|
|
31
|
+
"""The top-level dir when this is a scopeable source file, else None.
|
|
32
|
+
Root-level loose files and dot-dirs make bad scopes; the excludes drop
|
|
33
|
+
generated trees and tests the same way inventory later will."""
|
|
34
|
+
top, _, rest = path.partition("/")
|
|
35
|
+
if not rest or top.startswith(".") or excluded(path, _MATCH_DEFAULT_EXCLUDE):
|
|
36
|
+
return None
|
|
37
|
+
return top if _language_of(path) else None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def source_candidates(files: list[str]) -> list[str]:
|
|
41
|
+
"""The paths init would put in a scope. Counting them over the UNTRACKED set
|
|
42
|
+
is how init tells "you are in the wrong directory" from "you never ran
|
|
43
|
+
`git add`" — crapkit reads `git ls-files` and sees neither case any other way.
|
|
44
|
+
"""
|
|
45
|
+
paths = (raw.replace("\\", "/") for raw in files)
|
|
46
|
+
return [p for p in paths if _scoped_source(p)]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def sniff_scopes(files: list[str]) -> dict[str, tuple[str, ...]]:
|
|
50
|
+
"""Top-level source dirs -> their languages, sorted both ways for stable output."""
|
|
51
|
+
langs: dict[str, set[str]] = {}
|
|
52
|
+
for raw in files:
|
|
53
|
+
path = raw.replace("\\", "/")
|
|
54
|
+
top = _scoped_source(path)
|
|
55
|
+
if top is None:
|
|
56
|
+
continue
|
|
57
|
+
langs.setdefault(top, set()).add(_language_of(path))
|
|
58
|
+
return {d: tuple(sorted(v)) for d, v in sorted(langs.items())}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class LaneSpec(NamedTuple):
|
|
62
|
+
"""A lane init can write live, plus the scope languages it measures."""
|
|
63
|
+
name: str
|
|
64
|
+
command: str
|
|
65
|
+
artifact: str
|
|
66
|
+
parser: str
|
|
67
|
+
languages: tuple[str, ...]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
PYTEST_MARKERS = ("pyproject.toml", "pytest.ini", "setup.cfg")
|
|
71
|
+
|
|
72
|
+
_PY_LANGUAGES = ("python",)
|
|
73
|
+
_JS_LANGUAGES = ("javascript", "tsx", "typescript")
|
|
74
|
+
_JS_RUNNER_COMMAND = {"jest": "npx jest --coverage", "vitest": "npx vitest run --coverage"}
|
|
75
|
+
|
|
76
|
+
# Every artifact a scaffolded lane writes lands under .crapkit/, which init
|
|
77
|
+
# ignores in the same breath. A 14-lane repo that let each runner default grew
|
|
78
|
+
# fifteen coverage-* directories and seven junit files at its root, one per
|
|
79
|
+
# lane, and nothing in the tree said which lane owned which.
|
|
80
|
+
_COV_DIR = ".crapkit/cov"
|
|
81
|
+
_PY_ARTIFACT = f"{_COV_DIR}/py.json"
|
|
82
|
+
_JS_COV_DIR = f"{_COV_DIR}/js"
|
|
83
|
+
_JS_ARTIFACT = f"{_JS_COV_DIR}/coverage-final.json"
|
|
84
|
+
_JS_DEFAULT_ARTIFACT = "coverage/coverage-final.json"
|
|
85
|
+
|
|
86
|
+
# Each runner spells "write the coverage report here" its own way and rejects
|
|
87
|
+
# the other's spelling outright.
|
|
88
|
+
_JS_REPORTS_DIR_FLAG = {"jest": "--coverageDirectory=",
|
|
89
|
+
"vitest": "--coverage.reportsDirectory="}
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _pytest_lane(markers: frozenset[str], interpreter: str) -> LaneSpec | None:
|
|
93
|
+
if not markers.intersection(PYTEST_MARKERS):
|
|
94
|
+
return None
|
|
95
|
+
return LaneSpec("py", f"{interpreter} -m pytest --cov --cov-branch "
|
|
96
|
+
f"--cov-report=json:{_PY_ARTIFACT}",
|
|
97
|
+
_PY_ARTIFACT, "coveragepy", _PY_LANGUAGES)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _npm_test_script(scripts: dict) -> str | None:
|
|
101
|
+
"""The script name npm would run tests with: "test" when it exists, else the
|
|
102
|
+
alphabetically first name starting with "test", so the choice never moves."""
|
|
103
|
+
if "test" in scripts:
|
|
104
|
+
return "test"
|
|
105
|
+
named = sorted(name for name in scripts if name.startswith("test"))
|
|
106
|
+
return named[0] if named else None
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _js_runner_command(dev_dependencies: dict) -> str | None:
|
|
110
|
+
for runner in sorted(_JS_RUNNER_COMMAND):
|
|
111
|
+
if runner in dev_dependencies:
|
|
112
|
+
return _JS_RUNNER_COMMAND[runner]
|
|
113
|
+
return None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _load_json(text: str) -> dict:
|
|
117
|
+
try:
|
|
118
|
+
data = json.loads(text)
|
|
119
|
+
except ValueError:
|
|
120
|
+
return {}
|
|
121
|
+
return data if isinstance(data, dict) else {}
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _js_routing(dev_dependencies: dict) -> str:
|
|
125
|
+
"""The flag that moves this runner's coverage directory under .crapkit/, or
|
|
126
|
+
"" when package.json names neither runner or both.
|
|
127
|
+
|
|
128
|
+
`npm test` can run anything, so devDependencies is the only thing that names
|
|
129
|
+
the runner — and naming it wrong is worse than the litter: jest exits on
|
|
130
|
+
vitest's `--coverage.reportsDirectory` and vitest exits on jest's
|
|
131
|
+
`--coverageDirectory`. Unresolved, the lane keeps the directory its runner
|
|
132
|
+
already defaults to and doctor says so.
|
|
133
|
+
"""
|
|
134
|
+
named = [runner for runner in sorted(_JS_REPORTS_DIR_FLAG) if runner in dev_dependencies]
|
|
135
|
+
return f" {_JS_REPORTS_DIR_FLAG[named[0]]}{_JS_COV_DIR}" if len(named) == 1 else ""
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _js_lane(package_json: str) -> LaneSpec | None:
|
|
139
|
+
"""A test script, or vitest/jest in devDependencies: either says the repo
|
|
140
|
+
already knows how to produce istanbul coverage."""
|
|
141
|
+
package = _load_json(package_json)
|
|
142
|
+
script = _npm_test_script(package.get("scripts", {}))
|
|
143
|
+
command = (f"npm run {script} -- --coverage" if script
|
|
144
|
+
else _js_runner_command(package.get("devDependencies", {})))
|
|
145
|
+
if command is None:
|
|
146
|
+
return None
|
|
147
|
+
routing = _js_routing(package.get("devDependencies", {}))
|
|
148
|
+
artifact = _JS_ARTIFACT if routing else _JS_DEFAULT_ARTIFACT
|
|
149
|
+
return LaneSpec("js", command + routing, artifact, "istanbul", _JS_LANGUAGES)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def detect_lanes(markers: frozenset[str], package_json: str, *,
|
|
153
|
+
interpreter: str = "python") -> tuple[LaneSpec, ...]:
|
|
154
|
+
"""The lanes this repo can already run, decided from files alone.
|
|
155
|
+
|
|
156
|
+
Nothing is executed and nothing is imported: presence of a pytest marker
|
|
157
|
+
file, and what package.json says about itself, are the whole signal.
|
|
158
|
+
"""
|
|
159
|
+
found = (_pytest_lane(markers, interpreter), _js_lane(package_json))
|
|
160
|
+
return tuple(lane for lane in found if lane)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _quoted(names) -> str:
|
|
164
|
+
return ", ".join(f'"{name}"' for name in names)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _scope_stanza(name: str, languages: tuple[str, ...]) -> list[str]:
|
|
168
|
+
return ["[[scope]]", f'name = "{name}"', f'paths = ["{name}"]',
|
|
169
|
+
f"languages = [{_quoted(languages)}]", ""]
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _exclude_stanza() -> list[str]:
|
|
173
|
+
return ["[exclude]", f"globs = [{_quoted(DEFAULT_EXCLUDES)}]", ""]
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _lane_stanza(lane: LaneSpec, scope_names: tuple[str, ...]) -> list[str]:
|
|
177
|
+
return ["[[lane]]", f'name = "{lane.name}"', f'command = "{lane.command}"',
|
|
178
|
+
f'artifact = "{lane.artifact}"', f'parser = "{lane.parser}"',
|
|
179
|
+
f"scopes = [{_quoted(scope_names)}]", ""]
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _scopes_for(lane: LaneSpec, scopes: dict[str, tuple[str, ...]]) -> tuple[str, ...]:
|
|
183
|
+
return tuple(name for name, languages in scopes.items()
|
|
184
|
+
if set(languages).intersection(lane.languages))
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def live_lanes(lanes: tuple[LaneSpec, ...],
|
|
188
|
+
scopes: dict[str, tuple[str, ...]]) -> tuple[LaneSpec, ...]:
|
|
189
|
+
"""The detected lanes init actually writes: the ones with a scope to measure.
|
|
190
|
+
|
|
191
|
+
A lane with an empty scopes list measures nothing, so it goes back to being a
|
|
192
|
+
template rather than papering over the gap — and it leaves no artifact behind
|
|
193
|
+
either, which is what the .gitignore side of init needs to know.
|
|
194
|
+
"""
|
|
195
|
+
return tuple(lane for lane in lanes if _scopes_for(lane, scopes))
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _live_lanes(lanes: tuple[LaneSpec, ...],
|
|
199
|
+
scopes: dict[str, tuple[str, ...]]) -> tuple[list[str], set[str]]:
|
|
200
|
+
"""Stanzas for the lanes init writes, and the parsers they cover."""
|
|
201
|
+
lines: list[str] = []
|
|
202
|
+
covered = set()
|
|
203
|
+
for lane in live_lanes(lanes, scopes):
|
|
204
|
+
lines += _lane_stanza(lane, _scopes_for(lane, scopes))
|
|
205
|
+
covered.add(lane.parser)
|
|
206
|
+
return lines, covered
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
_TEMPLATES = {
|
|
210
|
+
"coveragepy": ("# [[lane]]", '# name = "py"',
|
|
211
|
+
'# command = "python -m pytest --cov --cov-branch '
|
|
212
|
+
f'--cov-report=json:{_PY_ARTIFACT}"',
|
|
213
|
+
f'# artifact = "{_PY_ARTIFACT}"', '# parser = "coveragepy"'),
|
|
214
|
+
"istanbul": ("# [[lane]]", '# name = "js"',
|
|
215
|
+
'# command = "npx vitest run --coverage '
|
|
216
|
+
f'{_JS_REPORTS_DIR_FLAG["vitest"]}{_JS_COV_DIR}"',
|
|
217
|
+
f'# artifact = "{_JS_ARTIFACT}"', '# parser = "istanbul"'),
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# `crapkit test-scoped FILES` runs one of these per scope. A scope with no
|
|
222
|
+
# template exits 3, so the stub is written for every scope init found; the value
|
|
223
|
+
# is the runner that scope's language usually uses, and the whole block stays
|
|
224
|
+
# commented because only the repo knows whether that command is the right one.
|
|
225
|
+
_SCOPED_TEST_COMMANDS = {
|
|
226
|
+
"python": "python -m pytest {files} -q -p no:cacheprovider",
|
|
227
|
+
"javascript": "npx vitest run {files}",
|
|
228
|
+
"tsx": "npx vitest run {files}",
|
|
229
|
+
"typescript": "npx vitest run {files}",
|
|
230
|
+
}
|
|
231
|
+
_SCOPED_TEST_PLACEHOLDER = "<your test command> {files}"
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _scoped_test_command(languages: tuple[str, ...]) -> str:
|
|
235
|
+
for language in languages:
|
|
236
|
+
if language in _SCOPED_TEST_COMMANDS:
|
|
237
|
+
return _SCOPED_TEST_COMMANDS[language]
|
|
238
|
+
return _SCOPED_TEST_PLACEHOLDER
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _runner_confirmed(languages: tuple[str, ...], confirmed: frozenset[str]) -> bool:
|
|
242
|
+
return any(lang in confirmed and lang in _SCOPED_TEST_COMMANDS for lang in languages)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _confirmed_languages(lanes: tuple[LaneSpec, ...]) -> frozenset[str]:
|
|
246
|
+
"""Languages whose scoped command a detected lane already proves.
|
|
247
|
+
|
|
248
|
+
Only pytest: the presence signal that wrote the py coverage lane makes
|
|
249
|
+
`python -m pytest {files}` known-good. The js runners stay unconfirmed on
|
|
250
|
+
purpose — which vitest or jest config a file-scoped run needs is exactly
|
|
251
|
+
what presence detection cannot see."""
|
|
252
|
+
return frozenset({"python"} if any(" -m pytest " in ln.command for ln in lanes) else ())
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _scoped_entry_lines(scopes: dict[str, tuple[str, ...]], live: bool) -> list[str]:
|
|
256
|
+
prefix = "" if live else "# "
|
|
257
|
+
return [f'{prefix}{name} = "{_scoped_test_command(languages)}"'
|
|
258
|
+
for name, languages in scopes.items()]
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _scoped_tests_stub(scopes: dict[str, tuple[str, ...]],
|
|
262
|
+
confirmed: frozenset[str] = frozenset()) -> list[str]:
|
|
263
|
+
"""The [crapkit.scoped_tests] block: live entries for scopes whose runner a
|
|
264
|
+
detected lane proves, commented templates for the rest.
|
|
265
|
+
|
|
266
|
+
A confirmed runner written commented would hand doctor a warning about a
|
|
267
|
+
gap init could have closed. Every commented line still uncomments as
|
|
268
|
+
written: a stub a reader has to rewrite before it parses is no better than
|
|
269
|
+
the nothing that used to be here.
|
|
270
|
+
"""
|
|
271
|
+
live = {n: l for n, l in scopes.items() if _runner_confirmed(l, confirmed)}
|
|
272
|
+
rest = {n: l for n, l in scopes.items() if n not in live}
|
|
273
|
+
intro = ["# `crapkit test-scoped FILES` runs one command per scope, with {files}",
|
|
274
|
+
"# replaced by that scope's files, each quoted."]
|
|
275
|
+
return intro + _live_block(live) + _commented_block(rest, bool(live)) + [""]
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _live_block(live: dict[str, tuple[str, ...]]) -> list[str]:
|
|
279
|
+
if not live:
|
|
280
|
+
return []
|
|
281
|
+
return ["[crapkit.scoped_tests]"] + _scoped_entry_lines(live, True)
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _commented_block(rest: dict[str, tuple[str, ...]], has_live: bool) -> list[str]:
|
|
285
|
+
if not rest:
|
|
286
|
+
return []
|
|
287
|
+
header = ["# Uncomment what fits:"] + ([] if has_live else ["# [crapkit.scoped_tests]"])
|
|
288
|
+
return header + _scoped_entry_lines(rest, False)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _template_lines(covered: set[str], scope_list: str) -> list[str]:
|
|
292
|
+
# the template's scope is a placeholder on purpose: writing a real scope
|
|
293
|
+
# name pointed a TS lane template at a python project's sources
|
|
294
|
+
del scope_list
|
|
295
|
+
lines: list[str] = []
|
|
296
|
+
for parser in sorted(_TEMPLATES):
|
|
297
|
+
if parser not in covered:
|
|
298
|
+
lines += [*_TEMPLATES[parser], '# scopes = ["<your-scope>"]', ""]
|
|
299
|
+
if not lines:
|
|
300
|
+
return []
|
|
301
|
+
return ["# Declare one [[lane]] per coverage command, then run `crapkit coverage`.", *lines]
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def starter_toml(scopes: dict[str, tuple[str, ...]], lanes: tuple[LaneSpec, ...] = ()) -> str:
|
|
305
|
+
lines = ["[crapkit]", "target = 6", ""]
|
|
306
|
+
for name, languages in scopes.items():
|
|
307
|
+
lines += _scope_stanza(name, languages)
|
|
308
|
+
lines += _exclude_stanza()
|
|
309
|
+
live, covered = _live_lanes(lanes, scopes)
|
|
310
|
+
return "\n".join(lines + live + _template_lines(covered, _quoted(scopes))
|
|
311
|
+
+ _scoped_tests_stub(scopes, _confirmed_languages(lanes)))
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
_STORE_IGNORE = ".crapkit/"
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def _artifact_ignore(artifact: str) -> str:
|
|
318
|
+
"""What to ignore for one lane's artifact. A file inside a directory ignores
|
|
319
|
+
the DIRECTORY: istanbul writes a whole coverage/ tree beside
|
|
320
|
+
coverage-final.json, and ignoring the one file leaves the rest untracked."""
|
|
321
|
+
top, sep, _ = artifact.partition("/")
|
|
322
|
+
return f"{top}/" if sep else top
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _runner_droppings(lane: LaneSpec) -> list[str]:
|
|
326
|
+
"""What the lane's runner leaves beside the artifact; a pytest lane drops
|
|
327
|
+
coverage's data file and bytecode caches into the consumer's tree."""
|
|
328
|
+
return [".coverage", "__pycache__/"] if lane.parser == "coveragepy" else []
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def gitignore_entries(lanes: tuple[LaneSpec, ...]) -> list[str]:
|
|
332
|
+
"""Everything adopting crapkit will drop in the consumer's tree: its own
|
|
333
|
+
store, the artifact of each lane init wrote, and the runner's droppings.
|
|
334
|
+
Order is stable and duplicates collapse, so two lanes sharing a directory
|
|
335
|
+
ignore it once."""
|
|
336
|
+
entries = [_STORE_IGNORE, *(entry for lane in lanes
|
|
337
|
+
for entry in (_artifact_ignore(lane.artifact),
|
|
338
|
+
*_runner_droppings(lane)))]
|
|
339
|
+
return list(dict.fromkeys(entries))
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _appended(current: str, entries: list[str]) -> str:
|
|
343
|
+
"""The new entries under their own heading, after whatever was already there."""
|
|
344
|
+
block = "# crapkit\n" + "".join(f"{entry}\n" for entry in entries)
|
|
345
|
+
if not current:
|
|
346
|
+
return block
|
|
347
|
+
separator = "\n" if current.endswith("\n") else "\n\n"
|
|
348
|
+
return current + separator + block
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def gitignore_update(current: str, lanes: tuple[LaneSpec, ...]) -> tuple[str, list[str]]:
|
|
352
|
+
"""The .gitignore this repo needs, and the entries it gained. Pure.
|
|
353
|
+
|
|
354
|
+
Idempotent: an entry the file already carries is never written twice, so a
|
|
355
|
+
repo that adopted crapkit by hand gets no duplicate lines.
|
|
356
|
+
"""
|
|
357
|
+
present = {line.strip() for line in current.splitlines()}
|
|
358
|
+
added = [entry for entry in gitignore_entries(lanes) if entry not in present]
|
|
359
|
+
if not added:
|
|
360
|
+
return current, []
|
|
361
|
+
return _appended(current, added), added
|
crapkit/score.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""CRAP scoring and the coverage join. Pure.
|
|
2
|
+
|
|
3
|
+
The inventory is the master list: every function gets a scored row. Coverage
|
|
4
|
+
joins by path plus span overlap. Flags never conflate: measured (a lane's
|
|
5
|
+
artifact spoke about the file), untested (lane covers the scope, artifact
|
|
6
|
+
silent on this function), no-lane (no lane covers the scope at all), cc-only
|
|
7
|
+
(the scope declares coverage_optional, so no coverage number can exist).
|
|
8
|
+
The three zero-coverage flags all score cov=0; the flag says whether the
|
|
9
|
+
missing number is a testing gap, a tooling gap, or by design.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from collections.abc import Iterator
|
|
14
|
+
from typing import NamedTuple
|
|
15
|
+
|
|
16
|
+
from .coverage_istanbul import FnCoverage
|
|
17
|
+
from .snapshot import InventoryRow
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def crap(ccn: int, cov: float) -> float:
|
|
21
|
+
return ccn * ccn * (1.0 - cov) ** 3 + ccn
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
_GRADES = ((0.02, "A"), (0.05, "B"), (0.10, "C"), (0.20, "D"))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def grade(over_target: int, total: int) -> str:
|
|
28
|
+
"""One letter for over-target density; A+ is reserved for zero debt."""
|
|
29
|
+
if over_target == 0:
|
|
30
|
+
return "A+"
|
|
31
|
+
ratio = over_target / total
|
|
32
|
+
for bound, letter in _GRADES:
|
|
33
|
+
if ratio < bound:
|
|
34
|
+
return letter
|
|
35
|
+
return "F"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ScoredRow(NamedTuple):
|
|
39
|
+
scope: str
|
|
40
|
+
path: str
|
|
41
|
+
long_name: str
|
|
42
|
+
start: int
|
|
43
|
+
end: int
|
|
44
|
+
ccn_std: int
|
|
45
|
+
ccn_mod: int
|
|
46
|
+
ccn: int
|
|
47
|
+
nloc: int
|
|
48
|
+
params: int
|
|
49
|
+
nesting: int
|
|
50
|
+
cov: float
|
|
51
|
+
flag: str
|
|
52
|
+
crap: float
|
|
53
|
+
remedy: str
|
|
54
|
+
cognitive: int = 0 # Sonar-spec cognitive complexity; reporting only, never gated
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
_FIELD_TYPES = (str, str, str, int, int, int, int, int, int, int, int,
|
|
58
|
+
float, str, float, str, int)
|
|
59
|
+
_SCORED_HEADER = "\t".join(ScoredRow._fields)
|
|
60
|
+
# One %s per field, built once. %s of every field is str() of it, so the bytes
|
|
61
|
+
# do not move; a row is a tuple, so it IS the argument list and the per-row
|
|
62
|
+
# generator, the join and the concatenation all disappear (179 -> 116 ms on
|
|
63
|
+
# 140,922 rows). Derived from _fields rather than written out, so a new column
|
|
64
|
+
# cannot leave the template a field short.
|
|
65
|
+
_SCORED_ROW = "\t".join(["%s"] * len(ScoredRow._fields)) + "\n"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def scored_tsv_lines(rows: list[ScoredRow]) -> Iterator[str]:
|
|
69
|
+
"""Header then one newline-terminated line per row. With no rows the header
|
|
70
|
+
is the empty string, so the file stays the single newline it always was."""
|
|
71
|
+
yield (_SCORED_HEADER if rows else "") + "\n"
|
|
72
|
+
for r in rows:
|
|
73
|
+
yield _SCORED_ROW % r
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def parse_scored_row(line: str) -> ScoredRow:
|
|
77
|
+
"""One exported line back to a row. str() of every field round-trips through
|
|
78
|
+
its own constructor, floats included, so the re-emitted bytes are identical."""
|
|
79
|
+
parts = line.split("\t")
|
|
80
|
+
if len(parts) != len(_FIELD_TYPES):
|
|
81
|
+
raise ValueError(f"scored row has {len(parts)} fields, expected {len(_FIELD_TYPES)}: {line!r}")
|
|
82
|
+
return ScoredRow(*[cast(part) for cast, part in zip(_FIELD_TYPES, parts)])
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def parse_scored_tsv(text: str) -> list[ScoredRow]:
|
|
86
|
+
return [parse_scored_row(line) for line in text.splitlines()
|
|
87
|
+
if line.strip() and line != _SCORED_HEADER]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _overlap(a_start: int, a_end: int, b_start: int, b_end: int) -> int:
|
|
91
|
+
return max(0, min(a_end, b_end) - max(a_start, b_start) + 1)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _best_match(row: InventoryRow, candidates: list[FnCoverage]) -> FnCoverage | None:
|
|
95
|
+
# Exact start beats raw overlap, and on remaining ties the tightest span wins:
|
|
96
|
+
# a nested function must join its own entry, never its enclosing function's
|
|
97
|
+
# (an enclosing match would inherit the parent's coverage and understate risk).
|
|
98
|
+
# Identical-span twins (two lanes measuring the same file) keep the BETTER
|
|
99
|
+
# measurement — the true union of branch hits is at least the max, and lane
|
|
100
|
+
# declaration order must never change a score.
|
|
101
|
+
best, best_key = None, None
|
|
102
|
+
for fn in candidates:
|
|
103
|
+
o = _overlap(row.start, row.end, fn.start, fn.end)
|
|
104
|
+
if o <= 0:
|
|
105
|
+
continue
|
|
106
|
+
key = (fn.start == row.start, o, -(fn.end - fn.start), fn.coverage)
|
|
107
|
+
if best_key is None or key > best_key:
|
|
108
|
+
best, best_key = fn, key
|
|
109
|
+
return best
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _remedy(ccn: int, score: float, ceiling: int) -> str:
|
|
113
|
+
if ccn > ceiling:
|
|
114
|
+
return "decompose"
|
|
115
|
+
return "ok" if score <= ceiling else "add-tests"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _finish(row, cov: float, flag: str, *, target: int, scope_targets) -> ScoredRow:
|
|
119
|
+
# cc-only is the pre-commit hook's rule: crap IS ccn, so _remedy can only
|
|
120
|
+
# answer ok or decompose. Feeding it cov=0 through the formula would say
|
|
121
|
+
# add-tests about code no test can reach.
|
|
122
|
+
score = float(row.ccn) if flag == "cc-only" else crap(row.ccn, cov)
|
|
123
|
+
ceiling = scope_targets.get(row.scope, target) if scope_targets else target
|
|
124
|
+
# Positional, and NOT *row: cognitive is last in both tuples with four
|
|
125
|
+
# fields between, so splicing the row in whole lands it in cov. Building
|
|
126
|
+
# this row is a third of the join's cost at 140,922 rows — **row._asdict()
|
|
127
|
+
# built a throwaway dict per row and looked every field up by name.
|
|
128
|
+
return ScoredRow(row[0], row[1], row[2], row[3], row[4], row[5], row[6], row[7],
|
|
129
|
+
row[8], row[9], row[10],
|
|
130
|
+
cov, flag, score, _remedy(row[7], score, ceiling), row[11])
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _named_overlay_cov(row, by_key: dict) -> tuple[float, str]:
|
|
134
|
+
# by_key is grouped on (path, long_name) up front: filtering the whole
|
|
135
|
+
# path's candidates per row made this O(rows x candidates) per file
|
|
136
|
+
# (measured 2,998,554 comparisons on a whole-repo rescore). The nearest-start
|
|
137
|
+
# tie-break is unchanged, and min() still keeps the FIRST nearest twin
|
|
138
|
+
# because the group holds the baseline's own order.
|
|
139
|
+
named = by_key.get((row.path, row.long_name))
|
|
140
|
+
if named:
|
|
141
|
+
return min(named, key=lambda c: abs(c.start - row.start)).cov, "measured"
|
|
142
|
+
return 0.0, "untested"
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _cov_without_join(row, lane_scopes: set, cc_only_scopes) -> tuple[float, str] | None:
|
|
146
|
+
"""The verdict for a row no coverage artifact can speak about, else None.
|
|
147
|
+
|
|
148
|
+
coverage_optional is checked FIRST: such a scope needs no lane, so the
|
|
149
|
+
no-lane fallback would otherwise hide it behind a tooling gap it does
|
|
150
|
+
not have.
|
|
151
|
+
"""
|
|
152
|
+
if row.scope in cc_only_scopes:
|
|
153
|
+
return 0.0, "cc-only"
|
|
154
|
+
if row.scope not in lane_scopes:
|
|
155
|
+
return 0.0, "no-lane"
|
|
156
|
+
return None
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _start_index(coverage_by_path: dict) -> dict[str, dict[int, list[FnCoverage]]]:
|
|
160
|
+
"""path -> {start line: the candidates declaring it}, in candidate order.
|
|
161
|
+
|
|
162
|
+
91.5% of joinable rows share a start line with some candidate, and the scan
|
|
163
|
+
that found it compared every candidate on the path: 936,818 pairwise
|
|
164
|
+
comparisons on the consumer repo. Bucket order is the candidates' own order, which
|
|
165
|
+
is what keeps a dead-even tie resolving to the same twin.
|
|
166
|
+
"""
|
|
167
|
+
index = {}
|
|
168
|
+
for path, candidates in coverage_by_path.items():
|
|
169
|
+
buckets: dict[int, list[FnCoverage]] = {}
|
|
170
|
+
for fn in candidates:
|
|
171
|
+
buckets.setdefault(fn.start, []).append(fn)
|
|
172
|
+
index[path] = buckets
|
|
173
|
+
return index
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _best_exact(row, bucket) -> FnCoverage | None:
|
|
177
|
+
"""The winner among candidates whose start EQUALS the row's, or None when
|
|
178
|
+
none of them overlaps it.
|
|
179
|
+
|
|
180
|
+
_best_match's key leads with (fn.start == row.start): True here and False
|
|
181
|
+
for every candidate outside this bucket, so a winner here is the winner
|
|
182
|
+
over the whole path. The remaining terms are that key's tail, and
|
|
183
|
+
max(a_start, b_start) is row.start by construction.
|
|
184
|
+
"""
|
|
185
|
+
best, best_key = None, None
|
|
186
|
+
for fn in bucket:
|
|
187
|
+
o = min(row.end, fn.end) - row.start + 1
|
|
188
|
+
if o <= 0:
|
|
189
|
+
continue
|
|
190
|
+
key = (o, -(fn.end - fn.start), fn.coverage)
|
|
191
|
+
if best_key is None or key > best_key:
|
|
192
|
+
best, best_key = fn, key
|
|
193
|
+
return best
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _span_join_cov(row, coverage_by_path: dict, start_index: dict) -> tuple[float, str]:
|
|
197
|
+
candidates = coverage_by_path.get(row.path)
|
|
198
|
+
if candidates is None:
|
|
199
|
+
return 0.0, "untested"
|
|
200
|
+
# The bucket answers for most rows; the scan is the fallback for a row that
|
|
201
|
+
# starts where no candidate does, or whose bucket overlaps it nowhere.
|
|
202
|
+
match = _best_exact(row, start_index[row.path].get(row.start, ()))
|
|
203
|
+
if match is None:
|
|
204
|
+
match = _best_match(row, candidates)
|
|
205
|
+
if match is None:
|
|
206
|
+
return 0.0, "untested"
|
|
207
|
+
return match.coverage, "measured"
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def overlay_stale_coverage(
|
|
211
|
+
rows: list[InventoryRow],
|
|
212
|
+
baseline_scored: list["ScoredRow"],
|
|
213
|
+
*,
|
|
214
|
+
lane_scopes: set[str],
|
|
215
|
+
target: int = 6,
|
|
216
|
+
scope_targets: dict[str, int] | None = None,
|
|
217
|
+
cc_only_scopes: frozenset[str] = frozenset(),
|
|
218
|
+
) -> list[ScoredRow]:
|
|
219
|
+
"""Rescore fresh complexity against a BASELINE run's coverage.
|
|
220
|
+
|
|
221
|
+
Joins by function NAME only (edits shift spans, names survive), nearest
|
|
222
|
+
start among same-name twins. A renamed or new function joins NOTHING —
|
|
223
|
+
a span join here would hand it a neighbour's stale number and mislead
|
|
224
|
+
the preview. Coverage values are the baseline's; the caller labels them
|
|
225
|
+
stale.
|
|
226
|
+
"""
|
|
227
|
+
by_key: dict[tuple[str, str], list[ScoredRow]] = {}
|
|
228
|
+
for r in baseline_scored:
|
|
229
|
+
if r.flag == "measured":
|
|
230
|
+
by_key.setdefault((r.path, r.long_name), []).append(r)
|
|
231
|
+
|
|
232
|
+
scored = []
|
|
233
|
+
for row in rows:
|
|
234
|
+
cov, flag = (_cov_without_join(row, lane_scopes, cc_only_scopes)
|
|
235
|
+
or _named_overlay_cov(row, by_key))
|
|
236
|
+
scored.append(_finish(row, cov, flag, target=target, scope_targets=scope_targets))
|
|
237
|
+
return scored
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def score_rows(
|
|
241
|
+
rows: list[InventoryRow],
|
|
242
|
+
coverage_by_path: dict[str, list[FnCoverage]],
|
|
243
|
+
*,
|
|
244
|
+
lane_scopes: set[str],
|
|
245
|
+
target: int = 6,
|
|
246
|
+
scope_targets: dict[str, int] | None = None,
|
|
247
|
+
cc_only_scopes: frozenset[str] = frozenset(),
|
|
248
|
+
) -> list[ScoredRow]:
|
|
249
|
+
start_index = _start_index(coverage_by_path)
|
|
250
|
+
scored = []
|
|
251
|
+
for r in rows:
|
|
252
|
+
cov, flag = (_cov_without_join(r, lane_scopes, cc_only_scopes)
|
|
253
|
+
or _span_join_cov(r, coverage_by_path, start_index))
|
|
254
|
+
scored.append(_finish(r, cov, flag, target=target, scope_targets=scope_targets))
|
|
255
|
+
return scored
|
crapkit/snapshot.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Canonical inventory rows and exports. Pure.
|
|
2
|
+
|
|
3
|
+
Scored rows carry no timestamps and sort totally, so identical inputs produce
|
|
4
|
+
byte-identical exports; run metadata lives on the run row in the store, never here.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from collections.abc import Iterable, Iterator
|
|
9
|
+
from typing import NamedTuple
|
|
10
|
+
|
|
11
|
+
from .merge import FunctionRecord
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class InventoryRow(NamedTuple):
|
|
15
|
+
scope: str
|
|
16
|
+
path: str
|
|
17
|
+
long_name: str
|
|
18
|
+
start: int
|
|
19
|
+
end: int
|
|
20
|
+
ccn_std: int
|
|
21
|
+
ccn_mod: int
|
|
22
|
+
ccn: int
|
|
23
|
+
nloc: int
|
|
24
|
+
params: int
|
|
25
|
+
nesting: int
|
|
26
|
+
cognitive: int = 0 # Sonar-spec cognitive complexity; reporting only, never gated
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def build_inventory_rows(by_scope: dict[str, list[FunctionRecord]]) -> list[InventoryRow]:
|
|
30
|
+
rows = [
|
|
31
|
+
InventoryRow(scope, r.path, r.long_name, r.start, r.end,
|
|
32
|
+
r.ccn_std, r.ccn_mod, r.ccn, r.nloc, r.params, r.nesting,
|
|
33
|
+
r.cognitive)
|
|
34
|
+
for scope, records in by_scope.items()
|
|
35
|
+
for r in records
|
|
36
|
+
]
|
|
37
|
+
rows.sort(key=lambda r: (r.scope, r.path, r.start, r.end, r.long_name))
|
|
38
|
+
return rows
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def tsv_lines(rows: Iterable[InventoryRow]) -> Iterator[str]:
|
|
42
|
+
"""The export document, one newline-terminated line at a time.
|
|
43
|
+
|
|
44
|
+
A generator, not a joined string: at 140k functions the list of lines plus
|
|
45
|
+
the joined document cost 44 MiB of transient copies of rows that already
|
|
46
|
+
exist. The caller writes these straight to a file opened with
|
|
47
|
+
newline="\\n", which is where the byte-identical guarantee is kept.
|
|
48
|
+
"""
|
|
49
|
+
yield "\t".join(InventoryRow._fields) + "\n"
|
|
50
|
+
for r in rows:
|
|
51
|
+
yield "\t".join(str(v) for v in r) + "\n"
|