crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/discover.py
ADDED
|
@@ -0,0 +1,365 @@
|
|
|
1
|
+
"""Discovery helpers for the start-editing packet: who calls a function, which
|
|
2
|
+
tests sit beside a file, and which guide files govern it.
|
|
3
|
+
|
|
4
|
+
Three standalone questions, answered off the working tree and git's index. Every
|
|
5
|
+
answer is sorted and every lookup is batched: one `git grep` per call whatever
|
|
6
|
+
the length of the identifier list, one directory listing per directory whatever
|
|
7
|
+
the number of glob patterns. Nothing here imports the CLI.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import fnmatch
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
import subprocess
|
|
15
|
+
from collections.abc import Iterable, Mapping
|
|
16
|
+
from pathlib import Path, PurePosixPath
|
|
17
|
+
from typing import NamedTuple
|
|
18
|
+
|
|
19
|
+
from .errors import GitError
|
|
20
|
+
|
|
21
|
+
CALLER_CAP = 20
|
|
22
|
+
_GUIDES = ("AGENTS.md", "CLAUDE.md", "CONTRIBUTING.md")
|
|
23
|
+
_SUPPORT_PATTERNS = ("conftest.py", "*.test-support.*")
|
|
24
|
+
_SUPPORT_DIRS = ("fixtures", "factories")
|
|
25
|
+
_EXPORT_STARTS = ("export ", "export{", "import ", "import{", "from ",
|
|
26
|
+
"module.exports", "exports.")
|
|
27
|
+
_DEF_MODIFIERS = (r"(?:@|export\s+|default\s+|public\s+|private\s+|protected\s+"
|
|
28
|
+
r"|static\s+|async\s+)*")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class _Hit(NamedTuple):
|
|
32
|
+
path: str
|
|
33
|
+
line: int
|
|
34
|
+
text: str
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _relative(path) -> PurePosixPath:
|
|
38
|
+
"""Repo-relative and posix, whichever slash the caller happened to hold."""
|
|
39
|
+
return PurePosixPath(str(path).replace("\\", "/"))
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _read(root: Path, path: str) -> str:
|
|
43
|
+
"""One file's text, or "" when it is gone.
|
|
44
|
+
|
|
45
|
+
utf-8 explicitly: a bare open() reads cp1252 on Windows and utf-8 in a
|
|
46
|
+
container, which is one repo answering two ways.
|
|
47
|
+
"""
|
|
48
|
+
try:
|
|
49
|
+
return (root / path).read_text(encoding="utf-8", errors="replace")
|
|
50
|
+
except OSError:
|
|
51
|
+
return ""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _upward(directory: PurePosixPath) -> list[PurePosixPath]:
|
|
55
|
+
"""The directory and every parent up to the repo root, nearest first."""
|
|
56
|
+
return list(dict.fromkeys([directory, *directory.parents]))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _guides_in(root: Path, directory: PurePosixPath) -> list[str]:
|
|
60
|
+
return [str(directory / name) for name in _GUIDES
|
|
61
|
+
if (root / directory / name).is_file()]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def nearest_guides(root, path) -> list[str]:
|
|
65
|
+
"""The AGENTS.md, CLAUDE.md and CONTRIBUTING.md files governing `path`.
|
|
66
|
+
|
|
67
|
+
Walks up from the file's own directory to the repo root, nearest first, so
|
|
68
|
+
the closest instruction an editor has to obey comes first. Within one
|
|
69
|
+
directory the fixed name order breaks the tie.
|
|
70
|
+
"""
|
|
71
|
+
root = Path(root)
|
|
72
|
+
found: list[str] = []
|
|
73
|
+
for directory in _upward(_relative(path).parent):
|
|
74
|
+
found += _guides_in(root, directory)
|
|
75
|
+
return found
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# --- the tests and support files beside one source ----------------------------
|
|
79
|
+
|
|
80
|
+
def _entries(directory: Path) -> list[str]:
|
|
81
|
+
"""The file names in one directory, sorted; a directory that is not there is
|
|
82
|
+
an empty answer, not a failure.
|
|
83
|
+
|
|
84
|
+
THE batching seam. Every glob pattern is matched against this one list, so a
|
|
85
|
+
directory is read once per call however many patterns ask about it.
|
|
86
|
+
"""
|
|
87
|
+
try:
|
|
88
|
+
with os.scandir(directory) as scan:
|
|
89
|
+
return sorted(e.name for e in scan if e.is_file())
|
|
90
|
+
except OSError:
|
|
91
|
+
return []
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _matching(names: list[str], patterns: Iterable[str]) -> list[str]:
|
|
95
|
+
return [n for n in names if any(fnmatch.fnmatchcase(n, p) for p in patterns)]
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _under(directory: PurePosixPath, names: list[str]) -> list[str]:
|
|
99
|
+
return [str(directory / n) for n in names]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _test_patterns(stem: str, suffix: str) -> tuple[str, ...]:
|
|
103
|
+
"""The four sibling namings, in the source's own extension: send.ts asks
|
|
104
|
+
about send*.test.ts and send*.spec.ts, parse.py about test_parse*.py and
|
|
105
|
+
parse*_test.py. None of them can match the source itself."""
|
|
106
|
+
return (f"{stem}*.test{suffix}", f"{stem}*.spec{suffix}",
|
|
107
|
+
f"test_{stem}*{suffix}", f"{stem}*_test{suffix}")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _nested_tests(root: Path, directory: PurePosixPath, stem: str) -> list[str]:
|
|
111
|
+
"""__tests__/send* — the sibling directory JS projects keep tests in."""
|
|
112
|
+
nested = directory / "__tests__"
|
|
113
|
+
return _under(nested, _matching(_entries(root / nested), (f"{stem}*",)))
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _support(root: Path, directory: PurePosixPath, own: list[str]) -> list[str]:
|
|
117
|
+
"""What a test in this directory leans on: the same-directory support names,
|
|
118
|
+
plus everything one level down in fixtures/ and factories/."""
|
|
119
|
+
found = _under(directory, _matching(own, _SUPPORT_PATTERNS))
|
|
120
|
+
for name in _SUPPORT_DIRS:
|
|
121
|
+
found += _under(directory / name, _entries(root / directory / name))
|
|
122
|
+
return sorted(found)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def related_tests(root, path) -> dict[str, list[str]]:
|
|
126
|
+
"""The tests that sit beside `path`, and the support files they lean on.
|
|
127
|
+
|
|
128
|
+
Sibling globs on the source stem in the source's own extension, plus the
|
|
129
|
+
__tests__ directory next to it. Both lists are repo-relative and sorted, so
|
|
130
|
+
two runs of a packet agree.
|
|
131
|
+
"""
|
|
132
|
+
root = Path(root)
|
|
133
|
+
rel = _relative(path)
|
|
134
|
+
directory = rel.parent
|
|
135
|
+
own = _entries(root / directory)
|
|
136
|
+
tests = _under(directory, _matching(own, _test_patterns(rel.stem, rel.suffix)))
|
|
137
|
+
return {"tests": sorted(tests + _nested_tests(root, directory, rel.stem)),
|
|
138
|
+
"support": _support(root, directory, own)}
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
# --- who calls one identifier -------------------------------------------------
|
|
142
|
+
|
|
143
|
+
def _grep_output(root: Path, args: list[str]) -> str:
|
|
144
|
+
"""One `git grep`, run for its output.
|
|
145
|
+
|
|
146
|
+
Exit 1 is git's way of saying nothing matched, which is an answer rather
|
|
147
|
+
than a failure; anything above that is a broken command.
|
|
148
|
+
"""
|
|
149
|
+
try:
|
|
150
|
+
proc = subprocess.run(["git", "grep", *args], cwd=root, capture_output=True,
|
|
151
|
+
text=True, encoding="utf-8", errors="replace")
|
|
152
|
+
except FileNotFoundError as exc:
|
|
153
|
+
raise GitError("git executable not found") from exc
|
|
154
|
+
if proc.returncode > 1:
|
|
155
|
+
raise GitError(f"git grep failed in {root}: {proc.stderr.strip()}")
|
|
156
|
+
return proc.stdout
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _flat_paths(scope_paths) -> list[str]:
|
|
160
|
+
"""crapkit's config holds scope paths as {scope: (path, ...)}; a caller with
|
|
161
|
+
a plain list of paths means the same thing."""
|
|
162
|
+
if isinstance(scope_paths, Mapping):
|
|
163
|
+
return [p for paths in scope_paths.values() for p in paths]
|
|
164
|
+
return list(scope_paths or ())
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _pathspecs(scope_paths) -> list[str]:
|
|
168
|
+
return sorted(set(_flat_paths(scope_paths)))
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _grep_args(identifiers: list[str], scope_paths) -> list[str]:
|
|
172
|
+
"""-F because an identifier is a literal, and one -e per name because fixed
|
|
173
|
+
strings have no alternation. However long the list, this is ONE command."""
|
|
174
|
+
patterns = [arg for name in identifiers for arg in ("-e", name)]
|
|
175
|
+
scoped = _pathspecs(scope_paths)
|
|
176
|
+
limit = ["--", *scoped] if scoped else []
|
|
177
|
+
return ["-n", "-I", "-F", "--full-name", "--no-color", *patterns, *limit]
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _is_test_path(path: str) -> bool:
|
|
181
|
+
"""Test file or not, by the rule cli.verifying._is_test_path applies: a
|
|
182
|
+
test/tests/__tests__ directory segment, a test_ prefix, or a .test./.spec.
|
|
183
|
+
infix. Written twice on purpose — this module imports nothing from the CLI —
|
|
184
|
+
and pinned by a test that routes the same paths through both.
|
|
185
|
+
"""
|
|
186
|
+
parts = path.lower().split("/")
|
|
187
|
+
return (any(p in ("test", "tests", "__tests__") for p in parts[:-1])
|
|
188
|
+
or parts[-1].startswith("test_")
|
|
189
|
+
or ".test." in parts[-1] or ".spec." in parts[-1])
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _parse_hit(raw: str) -> _Hit | None:
|
|
193
|
+
"""`path:line:text`, the one shape `git grep -n` emits."""
|
|
194
|
+
path, _, rest = raw.partition(":")
|
|
195
|
+
number, _, text = rest.partition(":")
|
|
196
|
+
return _Hit(path.replace("\\", "/"), int(number), text) if number.isdigit() else None
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _hits(root: Path, identifiers: list[str], scope_paths) -> list[_Hit]:
|
|
200
|
+
"""Every match for every name, out of ONE `git grep`.
|
|
201
|
+
|
|
202
|
+
git names no pattern in its output, so which name a line answers is settled
|
|
203
|
+
by the word match below, not by the command. Test files are dropped here: a
|
|
204
|
+
call from a test is not a caller a change has to answer to.
|
|
205
|
+
"""
|
|
206
|
+
lines = _grep_output(root, _grep_args(identifiers, scope_paths)).splitlines()
|
|
207
|
+
parsed = [_parse_hit(line) for line in lines]
|
|
208
|
+
return [h for h in parsed if h is not None and not _is_test_path(h.path)]
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _word(identifier: str) -> re.Pattern[str]:
|
|
212
|
+
return re.compile(r"(?<![A-Za-z0-9_])" + re.escape(identifier) + r"(?![A-Za-z0-9_])")
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _indent(line: str) -> int:
|
|
216
|
+
return len(line) - len(line.lstrip())
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _def_pattern(identifier: str) -> re.Pattern[str]:
|
|
220
|
+
"""A line that DEFINES the name: the keyword forms of both language
|
|
221
|
+
families, plus the method shorthand a class body uses."""
|
|
222
|
+
name = re.escape(identifier)
|
|
223
|
+
return re.compile(r"^\s*" + _DEF_MODIFIERS
|
|
224
|
+
+ r"(?:(?:def|class|function|const|let|var)\s+" + name + r"\b"
|
|
225
|
+
+ r"|" + name + r"\s*\([^)]*\)\s*[:{])")
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _def_line(lines: list[str], identifier: str) -> int | None:
|
|
229
|
+
pattern = _def_pattern(identifier)
|
|
230
|
+
for number, line in enumerate(lines, start=1):
|
|
231
|
+
if pattern.search(line):
|
|
232
|
+
return number
|
|
233
|
+
return None
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _span_end(lines: list[str], start: int) -> int:
|
|
237
|
+
"""The last line of the block opened at `start`: everything up to the next
|
|
238
|
+
non-blank line indented no deeper than the definition.
|
|
239
|
+
|
|
240
|
+
Indentation ends a Python body and a braced body alike — the closing brace
|
|
241
|
+
sits at the definition's own indent, and a closing brace holds no
|
|
242
|
+
identifier, so stopping just before it costs nothing.
|
|
243
|
+
"""
|
|
244
|
+
opened = _indent(lines[start - 1])
|
|
245
|
+
for i in range(start, len(lines)):
|
|
246
|
+
if lines[i].strip() and _indent(lines[i]) <= opened:
|
|
247
|
+
return i
|
|
248
|
+
return len(lines)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _defining_span(lines: list[str], identifier: str) -> tuple[int, int] | None:
|
|
252
|
+
"""The 1-based line range this file defines the name over, or None when it
|
|
253
|
+
does not define it at all."""
|
|
254
|
+
start = _def_line(lines, identifier)
|
|
255
|
+
return None if start is None else (start, _span_end(lines, start))
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _is_export_line(stripped: str) -> bool:
|
|
259
|
+
return stripped.startswith(_EXPORT_STARTS) or "__all__" in stripped
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _opens_a_list(stripped: str) -> bool:
|
|
263
|
+
"""A declaration whose names are on the FOLLOWING lines.
|
|
264
|
+
|
|
265
|
+
The bracket is the last thing on the line and no argument list closed just
|
|
266
|
+
before it, which is what separates `export {` and `__all__ = [` from
|
|
267
|
+
`export function send(x) {`. Counting brackets instead swallowed every
|
|
268
|
+
function body under an export line, and every local name in one read as
|
|
269
|
+
exported.
|
|
270
|
+
"""
|
|
271
|
+
return (stripped.endswith(("[", "(", "{"))
|
|
272
|
+
and not stripped.rstrip("[({ ").endswith(")"))
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _closes_a_list(stripped: str) -> bool:
|
|
276
|
+
"""The closer opens its own line in every formatter that wraps a list."""
|
|
277
|
+
return stripped.startswith(("]", ")", "}"))
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _export_lines(lines: list[str]) -> list[str]:
|
|
281
|
+
"""The file's export surface: the export and import statements, plus the
|
|
282
|
+
body of a list one of them opens. Computed once per call and reused by every
|
|
283
|
+
identifier in the batch."""
|
|
284
|
+
out: list[str] = []
|
|
285
|
+
inside = False
|
|
286
|
+
for raw in lines:
|
|
287
|
+
stripped = raw.strip()
|
|
288
|
+
if not inside and not _is_export_line(stripped):
|
|
289
|
+
continue
|
|
290
|
+
out.append(stripped)
|
|
291
|
+
inside = _opens_a_list(stripped) if not inside else not _closes_a_list(stripped)
|
|
292
|
+
return out
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _exported(export_lines: list[str], identifier: str) -> bool:
|
|
296
|
+
word = _word(identifier)
|
|
297
|
+
return any(word.search(line) for line in export_lines)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _in_span(hit: _Hit, path: str, span: tuple[int, int] | None) -> bool:
|
|
301
|
+
"""A definition and its own body are not calls to itself."""
|
|
302
|
+
return span is not None and hit.path == path and span[0] <= hit.line <= span[1]
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _found(hits: list[_Hit], identifier: str, path: str,
|
|
306
|
+
span: tuple[int, int] | None) -> list[dict]:
|
|
307
|
+
word = _word(identifier)
|
|
308
|
+
kept = [h for h in hits if word.search(h.text) and not _in_span(h, path, span)]
|
|
309
|
+
kept.sort(key=lambda h: (h.path, h.line))
|
|
310
|
+
return [{"path": h.path, "line": h.line} for h in kept]
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _answer(hits: list[_Hit], identifier: str, path: str, lines: list[str],
|
|
314
|
+
exports: list[str]) -> dict:
|
|
315
|
+
found = _found(hits, identifier, path, _defining_span(lines, identifier))
|
|
316
|
+
return {"count": len(found), "callers": found[:CALLER_CAP],
|
|
317
|
+
"exported": _exported(exports, identifier)}
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def _names(identifier) -> list[str]:
|
|
321
|
+
"""The names to ask about, deduplicated in asked-about order.
|
|
322
|
+
|
|
323
|
+
A blank one never gets through: `git grep -e ""` matches every line in the
|
|
324
|
+
repo, and thousands of callers is a wrong answer rather than an error.
|
|
325
|
+
"""
|
|
326
|
+
raw = [identifier] if isinstance(identifier, str) else list(identifier)
|
|
327
|
+
return list(dict.fromkeys(n for n in raw if n and n.strip()))
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _empty_answer() -> dict:
|
|
331
|
+
return {"count": 0, "callers": [], "exported": False}
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _answers(root: Path, scope_paths, path: str, names: list[str]) -> dict[str, dict]:
|
|
335
|
+
"""One grep and one read of the defining file, however many names.
|
|
336
|
+
|
|
337
|
+
Nothing to ask about spawns nothing: `git grep` with no -e is a usage error.
|
|
338
|
+
"""
|
|
339
|
+
if not names:
|
|
340
|
+
return {}
|
|
341
|
+
lines = _read(root, path).splitlines()
|
|
342
|
+
exports = _export_lines(lines)
|
|
343
|
+
hits = _hits(root, names, scope_paths)
|
|
344
|
+
return {n: _answer(hits, n, path, lines, exports) for n in names}
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def callers(root, scope_paths, path, identifier):
|
|
348
|
+
"""Who calls `identifier`, which `path` defines.
|
|
349
|
+
|
|
350
|
+
ONE `git grep` per call whatever `identifier` is: a str asks about one name
|
|
351
|
+
and returns that name's answer, a list asks about many and returns
|
|
352
|
+
{name: answer}. The list form IS the cache a packet wants — twelve functions
|
|
353
|
+
of one file cost one process and one read of that file, not twelve of each,
|
|
354
|
+
and every name reads the same grep output.
|
|
355
|
+
|
|
356
|
+
Neither test files nor the definition's own body count as callers, and each
|
|
357
|
+
name is measured against its OWN span, so a helper called from the function
|
|
358
|
+
above it in the same file is a caller. `count` is the whole number found;
|
|
359
|
+
`callers` stops at 20.
|
|
360
|
+
"""
|
|
361
|
+
answers = _answers(Path(root), scope_paths, str(_relative(path)),
|
|
362
|
+
_names(identifier))
|
|
363
|
+
if not isinstance(identifier, str):
|
|
364
|
+
return answers
|
|
365
|
+
return answers.get(identifier, _empty_answer())
|
crapkit/doctor.py
ADDED
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
"""`crapkit doctor` checks: does crapkit.toml still describe THIS repo? Pure."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from typing import NamedTuple
|
|
6
|
+
|
|
7
|
+
_KNOWN = {
|
|
8
|
+
"": {"crapkit", "scope", "lane", "exclude"},
|
|
9
|
+
"crapkit": {"target", "churn_window_months", "worklist_floor", "worklist_top",
|
|
10
|
+
"ratchet_file", "alert_command", "scoped_tests", "notes",
|
|
11
|
+
"mutation_command", "mutation_timeout_seconds", "mutation_workers",
|
|
12
|
+
"diff_uncovered_max", "debt_max_age_months", "repayment_min_per_30d",
|
|
13
|
+
"max_parallel_lanes", "analysis_workers"},
|
|
14
|
+
"scope": {"name", "paths", "languages", "target", "coverage_optional", "notes"},
|
|
15
|
+
"lane": {"name", "command", "artifact", "parser", "scopes", "cwd", "path_prefix", "env",
|
|
16
|
+
"full_suite", "container_ok", "results_artifact", "timeout_seconds", "retries",
|
|
17
|
+
"retest_command"},
|
|
18
|
+
"exclude": {"globs", "max_file_bytes"},
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
_ARRAY_TABLES = frozenset({"scope", "lane"})
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class UnknownKey(NamedTuple):
|
|
26
|
+
"""One ignored key. The table travels with it: without it the reader cannot
|
|
27
|
+
be told which spellings would have been accepted."""
|
|
28
|
+
path: str # dotted, as printed: "crapkit.churn_windo_months"
|
|
29
|
+
table: str # a key of _KNOWN; "" is the top level
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# Sorted once at import, not once per rejected key: every unknown-key message
|
|
33
|
+
# quotes its whole table, so a config with twenty typos in [crapkit] sorted the
|
|
34
|
+
# same fifteen strings twenty times.
|
|
35
|
+
_VALID_KEYS = {table: tuple(sorted(keys)) for table, keys in _KNOWN.items()}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def valid_keys(table: str) -> tuple[str, ...]:
|
|
39
|
+
"""Everything crapkit reads in one table, sorted so a message never moves."""
|
|
40
|
+
return _VALID_KEYS[table]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def table_label(table: str) -> str:
|
|
44
|
+
"""How the table is spelled in crapkit.toml. Arrays of tables double their
|
|
45
|
+
brackets, so a suggestion can be pasted as written."""
|
|
46
|
+
if not table:
|
|
47
|
+
return "crapkit.toml"
|
|
48
|
+
return f"[[{table}]]" if table in _ARRAY_TABLES else f"[{table}]"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _unknown_in(table: str, mapping: dict, label: str) -> list[UnknownKey]:
|
|
52
|
+
return [UnknownKey(f"{label}.{key}" if label else key, table)
|
|
53
|
+
for key in mapping if key not in _KNOWN[table]]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def unknown_key_findings(raw: dict) -> list[UnknownKey]:
|
|
57
|
+
"""Keys crapkit silently ignores — usually typos — each with its table. The
|
|
58
|
+
loader stays lenient (a config must survive version skew); doctor is where
|
|
59
|
+
typos die."""
|
|
60
|
+
problems = _unknown_in("", raw, "")
|
|
61
|
+
for table in ("crapkit", "exclude"):
|
|
62
|
+
problems += _unknown_in(table, raw.get(table, {}), table)
|
|
63
|
+
for table in ("scope", "lane"):
|
|
64
|
+
for row in raw.get(table, ()):
|
|
65
|
+
problems += _unknown_in(table, row, f"{table} {row.get('name', '?')!r}")
|
|
66
|
+
return problems
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class Finding(NamedTuple):
|
|
70
|
+
"""One doctor line. FAIL decides the exit code, WARN never does, and an
|
|
71
|
+
empty level is a continuation line (the file list under a scope)."""
|
|
72
|
+
level: str
|
|
73
|
+
text: str
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
_NO_TEMPLATE = (
|
|
77
|
+
"scope {name!r} has a lane but no [crapkit.scoped_tests] template — "
|
|
78
|
+
"`crapkit test-scoped` exits 3 on its files, so whoever edits them is handed "
|
|
79
|
+
'no command to run their tests; add {name} = "<test command>" under '
|
|
80
|
+
"[crapkit.scoped_tests]"
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def scoped_test_gaps(lanes, scoped_tests) -> tuple[Finding, ...]:
|
|
85
|
+
"""Scopes a lane measures that `crapkit test-scoped` cannot run, sorted. WARN.
|
|
86
|
+
|
|
87
|
+
Nothing is broken here: the lane runs, coverage lands, the gate holds. What
|
|
88
|
+
is missing is the one command a start-editing packet can hand over, and the
|
|
89
|
+
gap otherwise surfaces as an exit 3 in the middle of somebody's edit.
|
|
90
|
+
|
|
91
|
+
The lanes are walked once, not once per scope: a config declaring a lane per
|
|
92
|
+
package has more lanes than scopes, and this is a doctor line, not a survey.
|
|
93
|
+
"""
|
|
94
|
+
templated = {scope for scope, _template in scoped_tests}
|
|
95
|
+
laned = {scope for lane in lanes for scope in lane.scopes}
|
|
96
|
+
return tuple(Finding("WARN", _NO_TEMPLATE.format(name=name))
|
|
97
|
+
for name in sorted(laned - templated))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class Knobs(NamedTuple):
|
|
101
|
+
max_parallel_lanes: int
|
|
102
|
+
analysis_workers: int
|
|
103
|
+
mutation_workers: int
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def suggest_knobs(*, cpus: int, lanes: int) -> Knobs:
|
|
107
|
+
"""Advisory parallelism for this machine.
|
|
108
|
+
|
|
109
|
+
One core stays for the shell watching the run. A lane and a mutation worker
|
|
110
|
+
each hold a whole test suite in memory, so they get a quarter of the box
|
|
111
|
+
rather than a core apiece, and there is never a reason to run more lane
|
|
112
|
+
slots than there are lanes.
|
|
113
|
+
"""
|
|
114
|
+
return Knobs(max_parallel_lanes=max(1, min(lanes, cpus // 4)),
|
|
115
|
+
analysis_workers=max(1, cpus - 1),
|
|
116
|
+
mutation_workers=max(1, cpus // 4))
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def parallel_seconds(durations: tuple[float, ...], slots: int) -> float:
|
|
120
|
+
"""Makespan of these lanes on `slots` runners, longest first (LPT).
|
|
121
|
+
|
|
122
|
+
A bound, never a promise: lanes contend for the same cores and disk. It is
|
|
123
|
+
exact for the handful of lanes a real config declares.
|
|
124
|
+
"""
|
|
125
|
+
ends = [0.0] * max(1, slots)
|
|
126
|
+
for seconds in sorted(durations, reverse=True):
|
|
127
|
+
ends[ends.index(min(ends))] += seconds
|
|
128
|
+
return max(ends)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _cost_line(slots: int, durations: tuple[float, ...]) -> str:
|
|
132
|
+
if not durations:
|
|
133
|
+
return "# lane cost: no durations recorded yet — suggested from the cpu count alone"
|
|
134
|
+
return (f"# lane cost: {sum(durations):.1f}s serial -> "
|
|
135
|
+
f"~{parallel_seconds(durations, slots):.1f}s across {slots} lane slot(s)")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def tune_lines(*, cpus: int, knobs: Knobs, durations: tuple[float, ...]) -> list[str]:
|
|
139
|
+
"""Paste-ready [crapkit] knob lines plus what the suggestion was based on."""
|
|
140
|
+
return [f"# doctor --tune: suggestions for {cpus} cpu(s); nothing was written",
|
|
141
|
+
"[crapkit]",
|
|
142
|
+
f"max_parallel_lanes = {knobs.max_parallel_lanes}",
|
|
143
|
+
f"analysis_workers = {knobs.analysis_workers}",
|
|
144
|
+
f"mutation_workers = {knobs.mutation_workers}",
|
|
145
|
+
_cost_line(knobs.max_parallel_lanes, durations)]
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
class ArtifactLitter(NamedTuple):
|
|
149
|
+
"""One lane output written outside .crapkit/. The lane travels with the path
|
|
150
|
+
because a tree with fifteen coverage directories in it cannot say which lane
|
|
151
|
+
made which."""
|
|
152
|
+
lane: str
|
|
153
|
+
path: str
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
_STORE_DIR = ".crapkit"
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _first_part(path: str) -> str:
|
|
160
|
+
return path.replace("\\", "/").partition("/")[0]
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def scope_top_dirs(scopes) -> frozenset[str]:
|
|
164
|
+
"""The top-level directory of every declared scope path.
|
|
165
|
+
|
|
166
|
+
A lane that writes inside a package it measures (web/coverage/ beside
|
|
167
|
+
web/src) is that package's business, not root litter.
|
|
168
|
+
"""
|
|
169
|
+
return frozenset(_first_part(path) for scope in scopes for path in scope.paths)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _artifact_top(path: str) -> str:
|
|
173
|
+
"""The top-level directory this artifact lands in, or "" for a repo-root
|
|
174
|
+
file — which has no directory to be excused by."""
|
|
175
|
+
normalized = path.replace("\\", "/")
|
|
176
|
+
return _first_part(normalized) if "/" in normalized else ""
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _lane_outputs(lane) -> tuple[str, ...]:
|
|
180
|
+
return tuple(path for path in (lane.artifact, lane.results_artifact) if path)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def artifact_litter(lanes, scope_tops: frozenset[str]) -> tuple[ArtifactLitter, ...]:
|
|
184
|
+
"""Lane outputs that dirty the consumer's tree: a file at the repo root, or a
|
|
185
|
+
top-level directory that is neither .crapkit/ nor a scope's own tree.
|
|
186
|
+
|
|
187
|
+
Reported per path, in declaration order: a lane arguing about its coverage
|
|
188
|
+
file usually drops a junit report beside it, and folding the two into one
|
|
189
|
+
finding leaves the second one unnamed.
|
|
190
|
+
"""
|
|
191
|
+
clean = {_STORE_DIR, *scope_tops}
|
|
192
|
+
return tuple(ArtifactLitter(lane.name, path) for lane in lanes
|
|
193
|
+
for path in _lane_outputs(lane) if _artifact_top(path) not in clean)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
_PLAIN_FILE_MODE = "100644" # git's non-executable file; 100755 is the armed one
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def non_executable_hooks(modes: dict[str, str]) -> tuple[str, ...]:
|
|
200
|
+
"""Committed hook files git will silently skip, in path order.
|
|
201
|
+
|
|
202
|
+
A hook whose index mode is 100644 does not run on Linux or macOS, so
|
|
203
|
+
`core.hooksPath` installs a gate that never fires. Only plain files are
|
|
204
|
+
reported: a symlink (120000) or a gitlink (160000) is not a mode
|
|
205
|
+
`git update-index --chmod=+x` would fix.
|
|
206
|
+
"""
|
|
207
|
+
return tuple(sorted(path for path, mode in modes.items()
|
|
208
|
+
if mode == _PLAIN_FILE_MODE))
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
class UnmeasuredDir(NamedTuple):
|
|
212
|
+
directory: str
|
|
213
|
+
functions: int
|
|
214
|
+
example_test: str
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
@dataclass
|
|
218
|
+
class _DirStats:
|
|
219
|
+
functions: int = 0
|
|
220
|
+
flags: set = field(default_factory=set)
|
|
221
|
+
stems: set = field(default_factory=set)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
_TEST_DIR_PARTS = frozenset({"test", "tests", "__tests__", "spec", "specs"})
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _dir_of(path: str) -> str:
|
|
228
|
+
return path.rpartition("/")[0]
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _stem_of(path: str) -> str:
|
|
232
|
+
return path.rsplit("/", 1)[-1].split(".")[0]
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _subject_stem(path: str) -> str | None:
|
|
236
|
+
"""The source stem a test file names, or None when the name is not a test.
|
|
237
|
+
Four conventions cover every runner crapkit parses: foo.test.ts, foo.spec.ts,
|
|
238
|
+
test_foo.py, foo_test.py."""
|
|
239
|
+
name = path.rsplit("/", 1)[-1]
|
|
240
|
+
stem = _stem_of(path)
|
|
241
|
+
if ".test." in name or ".spec." in name:
|
|
242
|
+
return stem
|
|
243
|
+
if stem.startswith("test_"):
|
|
244
|
+
return stem[len("test_"):]
|
|
245
|
+
if stem.endswith("_test"):
|
|
246
|
+
return stem[:-len("_test")]
|
|
247
|
+
return None
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _path_parts(path: str) -> tuple[str, ...]:
|
|
251
|
+
return tuple(p for p in path.split("/") if p)
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _mirrored_parts(test_dir: str) -> tuple[str, ...]:
|
|
255
|
+
return tuple(p for p in _path_parts(test_dir) if p not in _TEST_DIR_PARTS)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _mirrors(test_dir: str, source_dir: str) -> bool:
|
|
259
|
+
"""A tests/ mirror: the test directory, with its test-named components
|
|
260
|
+
dropped, is a path suffix of the source directory. tests/api mirrors src/api;
|
|
261
|
+
a flat tests/ mirrors nothing, or it would claim the whole repo."""
|
|
262
|
+
parts = _mirrored_parts(test_dir)
|
|
263
|
+
return parts == _path_parts(source_dir)[-len(parts):] if parts else False
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _test_files(tracked: list[str]) -> list[str]:
|
|
267
|
+
return sorted(p for p in tracked if _subject_stem(p))
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _matching_test(directory: str, stems: set, test_files: list[str]) -> str | None:
|
|
271
|
+
"""The first tracked test file that names this directory's code: a same-stem
|
|
272
|
+
test anywhere in the repo (tests/test_parser.py for core/parser.py),
|
|
273
|
+
a sibling, or a tests/ mirror of the directory."""
|
|
274
|
+
for path in test_files:
|
|
275
|
+
if _subject_stem(path) in stems or _mirrors(_dir_of(path), directory):
|
|
276
|
+
return path
|
|
277
|
+
return None
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _group_dirs(rows, skip_scopes: frozenset[str]) -> dict[str, _DirStats]:
|
|
281
|
+
stats: dict[str, _DirStats] = {}
|
|
282
|
+
for row in rows:
|
|
283
|
+
if row.scope in skip_scopes:
|
|
284
|
+
continue
|
|
285
|
+
entry = stats.setdefault(_dir_of(row.path), _DirStats())
|
|
286
|
+
entry.functions += 1
|
|
287
|
+
entry.flags.add(row.flag)
|
|
288
|
+
entry.stems.add(_stem_of(row.path))
|
|
289
|
+
return stats
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def unmeasured_directories(rows, tracked: list[str], *,
|
|
293
|
+
skip_scopes: frozenset[str] = frozenset()) -> tuple[UnmeasuredDir, ...]:
|
|
294
|
+
"""Directories where EVERY scored function is flag "untested" and a test file
|
|
295
|
+
for that directory exists anyway.
|
|
296
|
+
|
|
297
|
+
That combination is a tooling gap, not a testing gap: the lane runs, the
|
|
298
|
+
tests pass, and the lane's own include list never looks at this code. One
|
|
299
|
+
measured function anywhere in the directory clears it.
|
|
300
|
+
"""
|
|
301
|
+
test_files = _test_files(tracked)
|
|
302
|
+
found = []
|
|
303
|
+
for directory, stats in sorted(_group_dirs(rows, skip_scopes).items()):
|
|
304
|
+
example = _matching_test(directory, stats.stems, test_files) \
|
|
305
|
+
if stats.flags == {"untested"} else None
|
|
306
|
+
if example:
|
|
307
|
+
found.append(UnmeasuredDir(directory, stats.functions, example))
|
|
308
|
+
return tuple(found)
|