crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/uncovered.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Dark lines per function: which lines inside a span no lane ever ran.
|
|
2
|
+
|
|
3
|
+
The line-level truth is the one the diff-coverage check already reads — every
|
|
4
|
+
lane's missing-line set, intersected, so a line stays dark only when NO lane
|
|
5
|
+
ran it. What this module adds is a verdict on whether the artifacts on disk
|
|
6
|
+
still describe the working tree. Line numbers from an artifact built before the
|
|
7
|
+
last edit point at code that has moved, which is worse than no numbers at all:
|
|
8
|
+
then the lines are [] and a note names the lane to rerun.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import NamedTuple
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class MissingLines(NamedTuple):
|
|
17
|
+
"""Per-file dead lines, plus the reason there are none to report.
|
|
18
|
+
|
|
19
|
+
A populated `note` overrides everything: no span gets lines, because the
|
|
20
|
+
only lines available would be the wrong ones.
|
|
21
|
+
"""
|
|
22
|
+
by_path: dict[str, set[int]]
|
|
23
|
+
note: str
|
|
24
|
+
|
|
25
|
+
def in_span(self, path: str, start: int, end: int) -> list[int]:
|
|
26
|
+
if self.note:
|
|
27
|
+
return []
|
|
28
|
+
return sorted(n for n in self.by_path.get(path, ()) if start <= n <= end)
|
|
29
|
+
|
|
30
|
+
def note_for(self, path: str, flag: str = "", scope: str = "") -> str:
|
|
31
|
+
"""Why this path has no dark lines, or "" when the artifacts answered.
|
|
32
|
+
|
|
33
|
+
A file no artifact mentioned is not a file with full coverage, and an
|
|
34
|
+
empty list with no note is exactly how that lie would read.
|
|
35
|
+
|
|
36
|
+
Three causes read the same on the surface and want different moves, so
|
|
37
|
+
the note names which one it is. cc-only is decided first and outranks
|
|
38
|
+
everything: the scope asked for no coverage, so no artifact was ever
|
|
39
|
+
going to speak for it and no lane is worth naming. A stale or missing
|
|
40
|
+
artifact is `self.note`, and rerunning coverage on a settled tree clears
|
|
41
|
+
it. A file absent from every artifact is `flag: untested`: nothing
|
|
42
|
+
imports it, so coverage never emitted a record for it.
|
|
43
|
+
"""
|
|
44
|
+
if flag == "cc-only":
|
|
45
|
+
return _cc_only_note(path, scope)
|
|
46
|
+
if self.note:
|
|
47
|
+
return self.note
|
|
48
|
+
if path in self.by_path:
|
|
49
|
+
return ""
|
|
50
|
+
return _absent_note(path, flag)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _cc_only_note(path: str, scope: str) -> str:
|
|
54
|
+
"""The note for a scope that declared coverage_optional.
|
|
55
|
+
|
|
56
|
+
It names that setting rather than a lane: the stale-artifact note used to
|
|
57
|
+
win here and sent readers to commit and rerun coverage for a scope no lane
|
|
58
|
+
covers, which changes nothing.
|
|
59
|
+
"""
|
|
60
|
+
return (f"scope {scope!r} sets coverage_optional = true, so no artifact "
|
|
61
|
+
f"can name uncovered lines for {path}")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _absent_note(path: str, flag: str) -> str:
|
|
65
|
+
"""The note for a file no artifact mentioned, told apart by the score's flag."""
|
|
66
|
+
absent = f"no lane artifact measured {path}"
|
|
67
|
+
if flag != "untested":
|
|
68
|
+
return absent
|
|
69
|
+
return (f"{absent} (flag untested: no test imports it, so coverage records "
|
|
70
|
+
f"nothing for it; write the first test that imports {path})")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def missing_by_path(root: Path, cfg) -> dict[str, set[int]]:
|
|
74
|
+
"""Union of the lanes' line-level truth; a file two lanes measured keeps a
|
|
75
|
+
line dead only when NO lane ran it."""
|
|
76
|
+
from . import covstream
|
|
77
|
+
|
|
78
|
+
missing: dict[str, set[int]] = {}
|
|
79
|
+
for lane in cfg.lanes:
|
|
80
|
+
artifact = root / lane.artifact
|
|
81
|
+
if not artifact.is_file():
|
|
82
|
+
continue
|
|
83
|
+
# Off the file, not out of a string: every declared lane's artifact
|
|
84
|
+
# would otherwise be decoded whole, one after another, on one heap.
|
|
85
|
+
parsed = (covstream.parse_istanbul_missing_file(artifact, repo_root=str(root))
|
|
86
|
+
if lane.parser == "istanbul"
|
|
87
|
+
else covstream.parse_coveragepy_missing_file(
|
|
88
|
+
artifact, path_prefix=lane.path_prefix))
|
|
89
|
+
for path, lines in parsed.items():
|
|
90
|
+
missing[path] = missing[path] & lines if path in missing else set(lines)
|
|
91
|
+
return missing
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _artifact_state(root: Path, lane, scope_paths: dict, git) -> str:
|
|
95
|
+
"""What stops this lane's artifact from naming line numbers, or "" when nothing does."""
|
|
96
|
+
from .lanes import lane_unchanged
|
|
97
|
+
|
|
98
|
+
if not (root / lane.artifact).is_file():
|
|
99
|
+
return f"lane {lane.name!r}: no artifact at {lane.artifact}"
|
|
100
|
+
if not lane_unchanged(root, lane, scope_paths, git):
|
|
101
|
+
return (f"lane {lane.name!r}: files in its scopes changed since {lane.artifact} "
|
|
102
|
+
"was written (uncommitted edits count), so its line numbers are stale — "
|
|
103
|
+
"commit or revert them, then rerun `crapkit coverage`")
|
|
104
|
+
return ""
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _staleness_note(root: Path, cfg, git) -> str:
|
|
108
|
+
return "; ".join(state for state in
|
|
109
|
+
(_artifact_state(root, lane, cfg.scope_paths, git) for lane in cfg.lanes)
|
|
110
|
+
if state)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def load_uncovered(root: Path, cfg, git=None) -> MissingLines:
|
|
114
|
+
"""The dark lines the lane artifacts on disk report, or the note saying why not.
|
|
115
|
+
|
|
116
|
+
A half-written artifact degrades to a note here, unlike in verify: naming a
|
|
117
|
+
function's dark lines is a convenience nothing gates on, and it must never
|
|
118
|
+
turn a question about the worklist into a tooling exit code.
|
|
119
|
+
"""
|
|
120
|
+
from .errors import ToolError
|
|
121
|
+
from .gitio import GitFacts
|
|
122
|
+
|
|
123
|
+
if not cfg.lanes:
|
|
124
|
+
return MissingLines({}, "no [[lane]] declared, so no artifact can say which lines are dark")
|
|
125
|
+
note = _staleness_note(root, cfg, git or GitFacts(root))
|
|
126
|
+
if note:
|
|
127
|
+
return MissingLines({}, note)
|
|
128
|
+
try:
|
|
129
|
+
return MissingLines(missing_by_path(root, cfg), "")
|
|
130
|
+
except ToolError as exc:
|
|
131
|
+
return MissingLines({}, f"unreadable lane artifact: {exc}")
|
crapkit/universe.py
ADDED
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""Assign tracked files to scopes, apply exclusions, name what fell through. Pure.
|
|
2
|
+
|
|
3
|
+
The file universe itself comes from `git ls-files` in the shell layer; lizard is
|
|
4
|
+
always fed these explicit lists because its own directory walking descends
|
|
5
|
+
nested node_modules (measured hang).
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import fnmatch
|
|
10
|
+
import re
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
from typing import NamedTuple
|
|
13
|
+
|
|
14
|
+
from .config import Config, Scope
|
|
15
|
+
|
|
16
|
+
LANGUAGE_EXTENSIONS = {
|
|
17
|
+
"typescript": (".ts",),
|
|
18
|
+
"tsx": (".tsx",),
|
|
19
|
+
"javascript": (".js", ".jsx", ".mjs", ".cjs"),
|
|
20
|
+
"python": (".py",),
|
|
21
|
+
"swift": (".swift",),
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# Test directories are excluded case-insensitively: Swift convention capitalizes Tests/.
|
|
25
|
+
_TEST_DIR = re.compile(r"(^|/)(tests?|__tests__)(/|$)", re.IGNORECASE)
|
|
26
|
+
|
|
27
|
+
_NEVER = re.compile(r"(?!x)x").match # an empty glob list must exclude NOTHING
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def exclude_matcher(globs: tuple[str, ...]) -> Callable[[str], re.Match[str] | None]:
|
|
31
|
+
"""The exclude globs as ONE compiled alternation, matched against a lowered path.
|
|
32
|
+
|
|
33
|
+
`fnmatch.fnmatch` normcases BOTH arguments on every call, and on Windows
|
|
34
|
+
that is an LCMapStringEx syscall per path per glob (measured: 593k calls,
|
|
35
|
+
0.44s on a 31.6k-file repo) for a lowering the caller already did. Compile
|
|
36
|
+
once, match once. Build it OUTSIDE any per-file loop.
|
|
37
|
+
"""
|
|
38
|
+
if not globs:
|
|
39
|
+
return _NEVER
|
|
40
|
+
return re.compile("|".join(fnmatch.translate(g.lower()) for g in globs)).match
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def excluded(path: str, match_glob: Callable[[str], re.Match[str] | None]) -> bool:
|
|
44
|
+
return bool(_TEST_DIR.search(path) or match_glob(path.lower()))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _source_extensions(languages: tuple[str, ...]) -> tuple[str, ...]:
|
|
48
|
+
return tuple(e for lang in languages for e in LANGUAGE_EXTENSIONS[lang])
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class _ScopeMatch(NamedTuple):
|
|
52
|
+
"""One scope's prefix and extension tests, precomputed once per run."""
|
|
53
|
+
name: str
|
|
54
|
+
exact: frozenset[str]
|
|
55
|
+
prefixes: tuple[str, ...]
|
|
56
|
+
extensions: tuple[str, ...]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _scope_matchers(scopes: tuple[Scope, ...]) -> tuple[_ScopeMatch, ...]:
|
|
60
|
+
return tuple(
|
|
61
|
+
_ScopeMatch(s.name, frozenset(s.paths),
|
|
62
|
+
tuple(p.rstrip("/") + "/" for p in s.paths),
|
|
63
|
+
_source_extensions(s.languages))
|
|
64
|
+
for s in scopes
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _owning_scope(path: str, matchers: tuple[_ScopeMatch, ...]) -> str | None:
|
|
69
|
+
"""Name of the scope that claims path, or None when no scope does."""
|
|
70
|
+
# First scope whose path prefix AND language extensions both match wins;
|
|
71
|
+
# a prefix-only match must not stop the search or shared-prefix scopes
|
|
72
|
+
# silently black-hole each other's files.
|
|
73
|
+
for m in matchers:
|
|
74
|
+
if path in m.exact or path.startswith(m.prefixes):
|
|
75
|
+
if path.endswith(m.extensions):
|
|
76
|
+
return m.name
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class Universe(NamedTuple):
|
|
81
|
+
"""Every candidate file's verdict.
|
|
82
|
+
|
|
83
|
+
A candidate is a non-excluded path some scope's language claims by
|
|
84
|
+
extension. `unclaimed` is the set that used to vanish here: source in a
|
|
85
|
+
declared language that no scope PATH owns, which then commits with zero
|
|
86
|
+
gating. `oversized` names what the byte ceiling cut, with the sizes, so
|
|
87
|
+
the skip is reported rather than silent.
|
|
88
|
+
"""
|
|
89
|
+
by_scope: dict[str, list[str]]
|
|
90
|
+
unclaimed: tuple[str, ...]
|
|
91
|
+
oversized: tuple[tuple[str, int], ...]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _candidate(path: str, matchers: tuple[_ScopeMatch, ...]) -> tuple[str | None, bool]:
|
|
95
|
+
"""(owning scope or None, whether any scope's language claims the extension)."""
|
|
96
|
+
owner = _owning_scope(path, matchers)
|
|
97
|
+
if owner is not None:
|
|
98
|
+
return owner, True
|
|
99
|
+
return None, any(path.endswith(m.extensions) for m in matchers)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _candidates(files: list[str], cfg: Config,
|
|
103
|
+
matchers: tuple[_ScopeMatch, ...]) -> list[tuple[str, str | None]]:
|
|
104
|
+
match_glob = exclude_matcher(cfg.exclude_globs)
|
|
105
|
+
out = []
|
|
106
|
+
for raw_path in files:
|
|
107
|
+
path = raw_path.replace("\\", "/")
|
|
108
|
+
if excluded(path, match_glob):
|
|
109
|
+
continue
|
|
110
|
+
owner, known_language = _candidate(path, matchers)
|
|
111
|
+
if known_language:
|
|
112
|
+
out.append((path, owner))
|
|
113
|
+
return out
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _oversize(path: str, max_bytes: int | None, size_of) -> int | None:
|
|
117
|
+
"""The file's size when the byte ceiling puts it out of reach, else None.
|
|
118
|
+
|
|
119
|
+
No limit means no stat: the size lookup is a syscall per file, and a repo
|
|
120
|
+
without max_file_bytes must not pay for a rule it never set.
|
|
121
|
+
"""
|
|
122
|
+
if max_bytes is None or size_of is None:
|
|
123
|
+
return None
|
|
124
|
+
size = size_of(path)
|
|
125
|
+
return size if size > max_bytes else None
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _partition(candidates: list[tuple[str, str | None]], scopes: tuple[Scope, ...],
|
|
129
|
+
max_bytes: int | None, size_of):
|
|
130
|
+
assigned: dict[str, list[str]] = {s.name: [] for s in scopes}
|
|
131
|
+
unclaimed, oversized = [], []
|
|
132
|
+
for path, owner in candidates:
|
|
133
|
+
size = _oversize(path, max_bytes, size_of)
|
|
134
|
+
if size is not None:
|
|
135
|
+
oversized.append((path, size))
|
|
136
|
+
elif owner is not None:
|
|
137
|
+
assigned[owner].append(path)
|
|
138
|
+
else:
|
|
139
|
+
unclaimed.append(path)
|
|
140
|
+
return assigned, unclaimed, oversized
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def scan_files(files: list[str], cfg: Config, *,
|
|
144
|
+
size_of: Callable[[str], int] | None = None) -> Universe:
|
|
145
|
+
"""The whole verdict. `size_of` is injected so this stays pure; the shell
|
|
146
|
+
layer passes a working-tree stat, and callers with no tree pass nothing."""
|
|
147
|
+
assigned, unclaimed, oversized = _partition(
|
|
148
|
+
_candidates(files, cfg, _scope_matchers(cfg.scopes)),
|
|
149
|
+
cfg.scopes, cfg.max_file_bytes, size_of)
|
|
150
|
+
return Universe({name: sorted(paths) for name, paths in assigned.items()},
|
|
151
|
+
tuple(sorted(unclaimed)), tuple(sorted(oversized)))
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def assign_files(files: list[str], cfg: Config, *,
|
|
155
|
+
size_of: Callable[[str], int] | None = None) -> dict[str, list[str]]:
|
|
156
|
+
"""Just the per-scope mapping, for callers with no use for the dropped sets."""
|
|
157
|
+
return scan_files(files, cfg, size_of=size_of).by_scope
|
crapkit/verify.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""The verdict. Pure: fresh scored rows + changed ranges + ratchet + failure sets in, Verdict out.
|
|
2
|
+
|
|
3
|
+
Three independent checks, all must hold:
|
|
4
|
+
- Gate: every function a change touched sits at CRAP <= target (coverage cannot
|
|
5
|
+
save cc > target; that is the target's design).
|
|
6
|
+
- Ratchet: no function above target scores worse than its recorded high-water
|
|
7
|
+
mark, touched or not (coverage rot regresses functions nobody edited).
|
|
8
|
+
- Failures: the fresh failure set adds nothing over the baseline's (the suite
|
|
9
|
+
is never assumed green; 98 pre-existing failures measured on day one).
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from bisect import bisect_left, bisect_right
|
|
14
|
+
from collections.abc import Iterator
|
|
15
|
+
from typing import NamedTuple
|
|
16
|
+
|
|
17
|
+
from .ratchet import RatchetEntry
|
|
18
|
+
from .score import ScoredRow, parse_scored_tsv, scored_tsv_lines
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def diff_uncovered(changed_ranges: dict, missing: dict) -> list[tuple[str, int]]:
|
|
22
|
+
"""Changed lines (new-file coordinates) whose statement never ran — where
|
|
23
|
+
the next bug ships. Files no lane measured stay silent (absent from missing)."""
|
|
24
|
+
out = []
|
|
25
|
+
for path, ranges in sorted(changed_ranges.items()):
|
|
26
|
+
dead = missing.get(path)
|
|
27
|
+
if not dead:
|
|
28
|
+
continue
|
|
29
|
+
# Sort the file's dead lines ONCE: sorting (and linearly scanning) them
|
|
30
|
+
# per hunk was O(hunks x dead log dead) for a file whose dead set never
|
|
31
|
+
# changes. Sorted, each hunk is two bisects and a slice.
|
|
32
|
+
ordered = sorted(dead)
|
|
33
|
+
for start, end in ranges:
|
|
34
|
+
out.extend((path, line)
|
|
35
|
+
for line in ordered[bisect_left(ordered, start):bisect_right(ordered, end)])
|
|
36
|
+
return out
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class PortableBaseline(NamedTuple):
|
|
40
|
+
commit: str
|
|
41
|
+
kind: str
|
|
42
|
+
rows: list[ScoredRow]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def baseline_tsv_lines(commit: str, kind: str, rows: list[ScoredRow]) -> Iterator[str]:
|
|
46
|
+
"""A baseline run as a file the repo can carry: a commit stamp, then the
|
|
47
|
+
run's scored export. The store lives in a gitignored .crapkit/, so a fresh
|
|
48
|
+
clone has nothing else to name what it is being measured against."""
|
|
49
|
+
yield f"# commit={commit} run_kind={kind}\n"
|
|
50
|
+
yield from scored_tsv_lines(rows)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _stamp_fields(stamp: str) -> dict[str, str]:
|
|
54
|
+
return dict(part.split("=", 1) for part in stamp.removeprefix("# ").split() if "=" in part)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def parse_baseline_tsv(text: str) -> PortableBaseline:
|
|
58
|
+
stamp, _, body = text.partition("\n")
|
|
59
|
+
fields = _stamp_fields(stamp)
|
|
60
|
+
if "commit" not in fields or "run_kind" not in fields:
|
|
61
|
+
raise ValueError(
|
|
62
|
+
f"a baseline file starts with `# commit=<sha> run_kind=<kind>`, got {stamp!r}")
|
|
63
|
+
return PortableBaseline(fields["commit"], fields["run_kind"], parse_scored_tsv(body))
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class GateViolation(NamedTuple):
|
|
67
|
+
path: str
|
|
68
|
+
long_name: str
|
|
69
|
+
start: int
|
|
70
|
+
ccn: int
|
|
71
|
+
cov: float
|
|
72
|
+
crap: float
|
|
73
|
+
remedy: str
|
|
74
|
+
dirty: bool = False
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class RatchetRegression(NamedTuple):
|
|
78
|
+
path: str
|
|
79
|
+
long_name: str
|
|
80
|
+
recorded: float
|
|
81
|
+
fresh_crap: float
|
|
82
|
+
dirty: bool = False
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class Verdict(NamedTuple):
|
|
86
|
+
ok: bool
|
|
87
|
+
gate_violations: list[GateViolation]
|
|
88
|
+
ratchet_regressions: list[RatchetRegression]
|
|
89
|
+
new_failures: list[str]
|
|
90
|
+
dirty_failures: list[str]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _id_forms(path: str) -> tuple[str, str]:
|
|
94
|
+
"""The two shapes a junit classname takes for one file: the repo-relative
|
|
95
|
+
path (vitest, and pytest's `file` fallback) and pytest's dotted module."""
|
|
96
|
+
stem = path[:-3] if path.endswith(".py") else path
|
|
97
|
+
return path, stem.replace("/", ".")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def dirty_failure_ids(new_failures: list[str], dirty_paths: set[str]) -> list[str]:
|
|
101
|
+
"""New failures whose test id names a file with uncommitted edits."""
|
|
102
|
+
forms = {form for path in dirty_paths for form in _id_forms(path)}
|
|
103
|
+
return [f for f in new_failures if f.split("::")[0] in forms]
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def dirty_counts(verdict: Verdict) -> tuple[int, int]:
|
|
107
|
+
"""(committed, dirty) over every finding kind, so one line says how much of
|
|
108
|
+
a verdict belongs to the tree as committed and how much to somebody's edits."""
|
|
109
|
+
dirty_ids = set(verdict.dirty_failures)
|
|
110
|
+
flags = ([v.dirty for v in verdict.gate_violations]
|
|
111
|
+
+ [r.dirty for r in verdict.ratchet_regressions]
|
|
112
|
+
+ [f in dirty_ids for f in verdict.new_failures])
|
|
113
|
+
dirty = sum(flags)
|
|
114
|
+
return len(flags) - dirty, dirty
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _touched(row: ScoredRow, ranges: dict[str, list[tuple[int, int]]]) -> bool:
|
|
118
|
+
spans = ranges.get(row.path)
|
|
119
|
+
if not spans:
|
|
120
|
+
return False
|
|
121
|
+
return any(not (hi < row.start or lo > row.end) for lo, hi in spans)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def touched_rows(rows: list[ScoredRow],
|
|
125
|
+
changed_ranges: dict[str, list[tuple[int, int]]]) -> list[ScoredRow]:
|
|
126
|
+
"""The gate's selection without its policy: rows whose span a change overlaps.
|
|
127
|
+
|
|
128
|
+
Every gate in crapkit judges touched functions only — untouched debt is the
|
|
129
|
+
ratchet's business. `rescore --gate` reuses this so its verdict and the
|
|
130
|
+
pre-commit hook's cannot disagree about which functions were even in scope.
|
|
131
|
+
"""
|
|
132
|
+
return [r for r in rows if _touched(r, changed_ranges)]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def worst_twins(fresh: list[ScoredRow]) -> dict[tuple[str, str], ScoredRow]:
|
|
136
|
+
"""Twins share (path, long_name); the WORST twin represents the key, so a
|
|
137
|
+
regression can never hide behind (nor a clean sibling tighten past) it."""
|
|
138
|
+
worst: dict[tuple[str, str], ScoredRow] = {}
|
|
139
|
+
for r in fresh:
|
|
140
|
+
key = (r.path, r.long_name)
|
|
141
|
+
if key not in worst or r.crap > worst[key].crap:
|
|
142
|
+
worst[key] = r
|
|
143
|
+
return worst
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _gate_violations(fresh, changed_ranges, target, scope_targets, dirty) -> list[GateViolation]:
|
|
147
|
+
gate = [
|
|
148
|
+
GateViolation(r.path, r.long_name, r.start, r.ccn, r.cov, r.crap, r.remedy,
|
|
149
|
+
r.path in dirty)
|
|
150
|
+
for r in fresh
|
|
151
|
+
if r.crap > (scope_targets or {}).get(r.scope, target) and _touched(r, changed_ranges)
|
|
152
|
+
]
|
|
153
|
+
gate.sort(key=lambda v: (-v.crap, v.path, v.start))
|
|
154
|
+
return gate
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _ratchet_regressions(fresh, ratchet, dirty) -> list[RatchetRegression]:
|
|
158
|
+
worst_by_key = worst_twins(fresh)
|
|
159
|
+
regressions = []
|
|
160
|
+
for entry in ratchet:
|
|
161
|
+
row = worst_by_key.get((entry.path, entry.long_name))
|
|
162
|
+
# Compare at the precision the mark is STORED at: marks live as 4dp
|
|
163
|
+
# strings, and cov = covered/total makes longer decimals routine — an
|
|
164
|
+
# unrounded compare wedges an unchanged tree against its own mark.
|
|
165
|
+
if row is not None and round(row.crap, 4) > entry.crap:
|
|
166
|
+
regressions.append(RatchetRegression(entry.path, entry.long_name, entry.crap,
|
|
167
|
+
round(row.crap, 4), entry.path in dirty))
|
|
168
|
+
regressions.sort(key=lambda r: (-(r.fresh_crap - r.recorded), r.path))
|
|
169
|
+
return regressions
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def evaluate(
|
|
173
|
+
*,
|
|
174
|
+
fresh: list[ScoredRow],
|
|
175
|
+
changed_ranges: dict[str, list[tuple[int, int]]],
|
|
176
|
+
ratchet: list[RatchetEntry],
|
|
177
|
+
baseline_failures: set[str],
|
|
178
|
+
fresh_failures: set[str],
|
|
179
|
+
target: int,
|
|
180
|
+
scope_targets: dict[str, int] | None = None,
|
|
181
|
+
dirty_paths: set[str] | None = None,
|
|
182
|
+
) -> Verdict:
|
|
183
|
+
dirty = dirty_paths or set()
|
|
184
|
+
gate = _gate_violations(fresh, changed_ranges, target, scope_targets, dirty)
|
|
185
|
+
regressions = _ratchet_regressions(fresh, ratchet, dirty)
|
|
186
|
+
new_failures = sorted(fresh_failures - baseline_failures)
|
|
187
|
+
|
|
188
|
+
return Verdict(
|
|
189
|
+
ok=not gate and not regressions and not new_failures,
|
|
190
|
+
gate_violations=gate,
|
|
191
|
+
ratchet_regressions=regressions,
|
|
192
|
+
new_failures=new_failures,
|
|
193
|
+
dirty_failures=dirty_failure_ids(new_failures, dirty),
|
|
194
|
+
)
|
crapkit/watch.py
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Watch-mode core. Pure: mtime snapshots in, changed paths out.
|
|
2
|
+
|
|
3
|
+
The polling loop in the CLI stays a thin shell around this; stdlib mtimes,
|
|
4
|
+
no filesystem-event dependency, works the same on every host.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
import stat
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _stat_mtime(path: Path) -> float | None:
|
|
14
|
+
"""One stat, or None for anything that is not a readable regular file.
|
|
15
|
+
|
|
16
|
+
`is_file()` followed by `stat()` paid the syscall TWICE per tracked file,
|
|
17
|
+
every poll interval. The S_ISREG check keeps directories out, which is the
|
|
18
|
+
only thing is_file() was buying.
|
|
19
|
+
"""
|
|
20
|
+
try:
|
|
21
|
+
st = os.stat(path)
|
|
22
|
+
except OSError:
|
|
23
|
+
return None
|
|
24
|
+
return st.st_mtime if stat.S_ISREG(st.st_mode) else None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _entry_mtime(entry: os.DirEntry) -> float | None:
|
|
28
|
+
"""The same answer for an already-listed name. On Windows the times came
|
|
29
|
+
with the listing, so this costs no syscall at all; the guard is for the
|
|
30
|
+
hosts where it does one, and for a name that died between the two."""
|
|
31
|
+
try:
|
|
32
|
+
st = entry.stat()
|
|
33
|
+
except OSError:
|
|
34
|
+
return None
|
|
35
|
+
return st.st_mtime if stat.S_ISREG(st.st_mode) else None
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _by_directory(files: list[str]) -> dict[str, dict[str, str]]:
|
|
39
|
+
"""{directory: {basename: path}} — the grouping one listing can answer.
|
|
40
|
+
|
|
41
|
+
Paths are the repo-relative, slash-separated ones git reports. Anything
|
|
42
|
+
shaped otherwise simply groups under the root and misses its listing, which
|
|
43
|
+
costs it a stat and nothing else.
|
|
44
|
+
"""
|
|
45
|
+
grouped: dict[str, dict[str, str]] = {}
|
|
46
|
+
for rel in files:
|
|
47
|
+
parent, _, base = rel.rpartition("/")
|
|
48
|
+
grouped.setdefault(parent, {})[base] = rel
|
|
49
|
+
return grouped
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _keep_wanted(entry: os.DirEntry, names: dict[str, str], out: dict[str, float]) -> None:
|
|
53
|
+
"""A listing walks a whole directory; only the tracked names are wanted,
|
|
54
|
+
and only they are worth a stat on the hosts where one is charged."""
|
|
55
|
+
rel = names.get(entry.name)
|
|
56
|
+
if rel is None:
|
|
57
|
+
return
|
|
58
|
+
mtime = _entry_mtime(entry)
|
|
59
|
+
if mtime is not None:
|
|
60
|
+
out[rel] = mtime
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _list_directory(directory: Path, names: dict[str, str], out: dict[str, float]) -> None:
|
|
64
|
+
"""Everything one listing can answer. A directory that cannot be read (gone,
|
|
65
|
+
refused, replaced by a file) answers nothing and leaves every name in it to
|
|
66
|
+
its own stat, so it costs speed and never an entry."""
|
|
67
|
+
try:
|
|
68
|
+
with os.scandir(directory) as entries:
|
|
69
|
+
for entry in entries:
|
|
70
|
+
_keep_wanted(entry, names, out)
|
|
71
|
+
except OSError:
|
|
72
|
+
return
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _listed_mtimes(root: Path, grouped: dict[str, dict[str, str]]) -> dict[str, float]:
|
|
76
|
+
out: dict[str, float] = {}
|
|
77
|
+
for parent, names in grouped.items():
|
|
78
|
+
_list_directory(root / parent if parent else root, names, out)
|
|
79
|
+
return out
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def snapshot_mtimes(root: Path, files: list[str]) -> dict[str, float]:
|
|
83
|
+
"""The mtime of every tracked file that is one; the rest simply absent.
|
|
84
|
+
|
|
85
|
+
One os.scandir per DIRECTORY, not one os.stat per FILE. On Windows the
|
|
86
|
+
times ride in the listing itself, so a 14,152-file tree polled every two
|
|
87
|
+
seconds dropped from 275 ms of stats to 46 ms of listings; elsewhere the
|
|
88
|
+
listing at least answers from a directory handle instead of walking the
|
|
89
|
+
whole path again once per file. Whatever the listing did not answer for
|
|
90
|
+
still gets its own stat, so a newborn file, a name the filesystem spells
|
|
91
|
+
with different case, and an unreadable directory all behave exactly as they
|
|
92
|
+
did when every file was stat-ed.
|
|
93
|
+
|
|
94
|
+
The result is built in `files` order, not listing order: two polls of one
|
|
95
|
+
unchanged tree have to produce the same mapping, and no filesystem promises
|
|
96
|
+
the order it enumerates in.
|
|
97
|
+
"""
|
|
98
|
+
listed = _listed_mtimes(root, _by_directory(files))
|
|
99
|
+
out: dict[str, float] = {}
|
|
100
|
+
for rel in files:
|
|
101
|
+
mtime = listed.get(rel)
|
|
102
|
+
if mtime is None:
|
|
103
|
+
mtime = _stat_mtime(root / rel)
|
|
104
|
+
if mtime is not None:
|
|
105
|
+
out[rel] = mtime
|
|
106
|
+
return out
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def changed_paths(before: dict[str, float], after: dict[str, float]) -> list[str]:
|
|
110
|
+
"""New, modified, and deleted paths between two snapshots, sorted."""
|
|
111
|
+
moved = {p for p, m in after.items() if before.get(p) != m}
|
|
112
|
+
return sorted(moved | (set(before) - set(after)))
|