crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/uncovered.py ADDED
@@ -0,0 +1,131 @@
1
+ """Dark lines per function: which lines inside a span no lane ever ran.
2
+
3
+ The line-level truth is the one the diff-coverage check already reads — every
4
+ lane's missing-line set, intersected, so a line stays dark only when NO lane
5
+ ran it. What this module adds is a verdict on whether the artifacts on disk
6
+ still describe the working tree. Line numbers from an artifact built before the
7
+ last edit point at code that has moved, which is worse than no numbers at all:
8
+ then the lines are [] and a note names the lane to rerun.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from pathlib import Path
13
+ from typing import NamedTuple
14
+
15
+
16
+ class MissingLines(NamedTuple):
17
+ """Per-file dead lines, plus the reason there are none to report.
18
+
19
+ A populated `note` overrides everything: no span gets lines, because the
20
+ only lines available would be the wrong ones.
21
+ """
22
+ by_path: dict[str, set[int]]
23
+ note: str
24
+
25
+ def in_span(self, path: str, start: int, end: int) -> list[int]:
26
+ if self.note:
27
+ return []
28
+ return sorted(n for n in self.by_path.get(path, ()) if start <= n <= end)
29
+
30
+ def note_for(self, path: str, flag: str = "", scope: str = "") -> str:
31
+ """Why this path has no dark lines, or "" when the artifacts answered.
32
+
33
+ A file no artifact mentioned is not a file with full coverage, and an
34
+ empty list with no note is exactly how that lie would read.
35
+
36
+ Three causes read the same on the surface and want different moves, so
37
+ the note names which one it is. cc-only is decided first and outranks
38
+ everything: the scope asked for no coverage, so no artifact was ever
39
+ going to speak for it and no lane is worth naming. A stale or missing
40
+ artifact is `self.note`, and rerunning coverage on a settled tree clears
41
+ it. A file absent from every artifact is `flag: untested`: nothing
42
+ imports it, so coverage never emitted a record for it.
43
+ """
44
+ if flag == "cc-only":
45
+ return _cc_only_note(path, scope)
46
+ if self.note:
47
+ return self.note
48
+ if path in self.by_path:
49
+ return ""
50
+ return _absent_note(path, flag)
51
+
52
+
53
+ def _cc_only_note(path: str, scope: str) -> str:
54
+ """The note for a scope that declared coverage_optional.
55
+
56
+ It names that setting rather than a lane: the stale-artifact note used to
57
+ win here and sent readers to commit and rerun coverage for a scope no lane
58
+ covers, which changes nothing.
59
+ """
60
+ return (f"scope {scope!r} sets coverage_optional = true, so no artifact "
61
+ f"can name uncovered lines for {path}")
62
+
63
+
64
+ def _absent_note(path: str, flag: str) -> str:
65
+ """The note for a file no artifact mentioned, told apart by the score's flag."""
66
+ absent = f"no lane artifact measured {path}"
67
+ if flag != "untested":
68
+ return absent
69
+ return (f"{absent} (flag untested: no test imports it, so coverage records "
70
+ f"nothing for it; write the first test that imports {path})")
71
+
72
+
73
+ def missing_by_path(root: Path, cfg) -> dict[str, set[int]]:
74
+ """Union of the lanes' line-level truth; a file two lanes measured keeps a
75
+ line dead only when NO lane ran it."""
76
+ from . import covstream
77
+
78
+ missing: dict[str, set[int]] = {}
79
+ for lane in cfg.lanes:
80
+ artifact = root / lane.artifact
81
+ if not artifact.is_file():
82
+ continue
83
+ # Off the file, not out of a string: every declared lane's artifact
84
+ # would otherwise be decoded whole, one after another, on one heap.
85
+ parsed = (covstream.parse_istanbul_missing_file(artifact, repo_root=str(root))
86
+ if lane.parser == "istanbul"
87
+ else covstream.parse_coveragepy_missing_file(
88
+ artifact, path_prefix=lane.path_prefix))
89
+ for path, lines in parsed.items():
90
+ missing[path] = missing[path] & lines if path in missing else set(lines)
91
+ return missing
92
+
93
+
94
+ def _artifact_state(root: Path, lane, scope_paths: dict, git) -> str:
95
+ """What stops this lane's artifact from naming line numbers, or "" when nothing does."""
96
+ from .lanes import lane_unchanged
97
+
98
+ if not (root / lane.artifact).is_file():
99
+ return f"lane {lane.name!r}: no artifact at {lane.artifact}"
100
+ if not lane_unchanged(root, lane, scope_paths, git):
101
+ return (f"lane {lane.name!r}: files in its scopes changed since {lane.artifact} "
102
+ "was written (uncommitted edits count), so its line numbers are stale — "
103
+ "commit or revert them, then rerun `crapkit coverage`")
104
+ return ""
105
+
106
+
107
+ def _staleness_note(root: Path, cfg, git) -> str:
108
+ return "; ".join(state for state in
109
+ (_artifact_state(root, lane, cfg.scope_paths, git) for lane in cfg.lanes)
110
+ if state)
111
+
112
+
113
+ def load_uncovered(root: Path, cfg, git=None) -> MissingLines:
114
+ """The dark lines the lane artifacts on disk report, or the note saying why not.
115
+
116
+ A half-written artifact degrades to a note here, unlike in verify: naming a
117
+ function's dark lines is a convenience nothing gates on, and it must never
118
+ turn a question about the worklist into a tooling exit code.
119
+ """
120
+ from .errors import ToolError
121
+ from .gitio import GitFacts
122
+
123
+ if not cfg.lanes:
124
+ return MissingLines({}, "no [[lane]] declared, so no artifact can say which lines are dark")
125
+ note = _staleness_note(root, cfg, git or GitFacts(root))
126
+ if note:
127
+ return MissingLines({}, note)
128
+ try:
129
+ return MissingLines(missing_by_path(root, cfg), "")
130
+ except ToolError as exc:
131
+ return MissingLines({}, f"unreadable lane artifact: {exc}")
crapkit/universe.py ADDED
@@ -0,0 +1,157 @@
1
+ """Assign tracked files to scopes, apply exclusions, name what fell through. Pure.
2
+
3
+ The file universe itself comes from `git ls-files` in the shell layer; lizard is
4
+ always fed these explicit lists because its own directory walking descends
5
+ nested node_modules (measured hang).
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import fnmatch
10
+ import re
11
+ from collections.abc import Callable
12
+ from typing import NamedTuple
13
+
14
+ from .config import Config, Scope
15
+
16
+ LANGUAGE_EXTENSIONS = {
17
+ "typescript": (".ts",),
18
+ "tsx": (".tsx",),
19
+ "javascript": (".js", ".jsx", ".mjs", ".cjs"),
20
+ "python": (".py",),
21
+ "swift": (".swift",),
22
+ }
23
+
24
+ # Test directories are excluded case-insensitively: Swift convention capitalizes Tests/.
25
+ _TEST_DIR = re.compile(r"(^|/)(tests?|__tests__)(/|$)", re.IGNORECASE)
26
+
27
+ _NEVER = re.compile(r"(?!x)x").match # an empty glob list must exclude NOTHING
28
+
29
+
30
+ def exclude_matcher(globs: tuple[str, ...]) -> Callable[[str], re.Match[str] | None]:
31
+ """The exclude globs as ONE compiled alternation, matched against a lowered path.
32
+
33
+ `fnmatch.fnmatch` normcases BOTH arguments on every call, and on Windows
34
+ that is an LCMapStringEx syscall per path per glob (measured: 593k calls,
35
+ 0.44s on a 31.6k-file repo) for a lowering the caller already did. Compile
36
+ once, match once. Build it OUTSIDE any per-file loop.
37
+ """
38
+ if not globs:
39
+ return _NEVER
40
+ return re.compile("|".join(fnmatch.translate(g.lower()) for g in globs)).match
41
+
42
+
43
+ def excluded(path: str, match_glob: Callable[[str], re.Match[str] | None]) -> bool:
44
+ return bool(_TEST_DIR.search(path) or match_glob(path.lower()))
45
+
46
+
47
+ def _source_extensions(languages: tuple[str, ...]) -> tuple[str, ...]:
48
+ return tuple(e for lang in languages for e in LANGUAGE_EXTENSIONS[lang])
49
+
50
+
51
+ class _ScopeMatch(NamedTuple):
52
+ """One scope's prefix and extension tests, precomputed once per run."""
53
+ name: str
54
+ exact: frozenset[str]
55
+ prefixes: tuple[str, ...]
56
+ extensions: tuple[str, ...]
57
+
58
+
59
+ def _scope_matchers(scopes: tuple[Scope, ...]) -> tuple[_ScopeMatch, ...]:
60
+ return tuple(
61
+ _ScopeMatch(s.name, frozenset(s.paths),
62
+ tuple(p.rstrip("/") + "/" for p in s.paths),
63
+ _source_extensions(s.languages))
64
+ for s in scopes
65
+ )
66
+
67
+
68
+ def _owning_scope(path: str, matchers: tuple[_ScopeMatch, ...]) -> str | None:
69
+ """Name of the scope that claims path, or None when no scope does."""
70
+ # First scope whose path prefix AND language extensions both match wins;
71
+ # a prefix-only match must not stop the search or shared-prefix scopes
72
+ # silently black-hole each other's files.
73
+ for m in matchers:
74
+ if path in m.exact or path.startswith(m.prefixes):
75
+ if path.endswith(m.extensions):
76
+ return m.name
77
+ return None
78
+
79
+
80
+ class Universe(NamedTuple):
81
+ """Every candidate file's verdict.
82
+
83
+ A candidate is a non-excluded path some scope's language claims by
84
+ extension. `unclaimed` is the set that used to vanish here: source in a
85
+ declared language that no scope PATH owns, which then commits with zero
86
+ gating. `oversized` names what the byte ceiling cut, with the sizes, so
87
+ the skip is reported rather than silent.
88
+ """
89
+ by_scope: dict[str, list[str]]
90
+ unclaimed: tuple[str, ...]
91
+ oversized: tuple[tuple[str, int], ...]
92
+
93
+
94
+ def _candidate(path: str, matchers: tuple[_ScopeMatch, ...]) -> tuple[str | None, bool]:
95
+ """(owning scope or None, whether any scope's language claims the extension)."""
96
+ owner = _owning_scope(path, matchers)
97
+ if owner is not None:
98
+ return owner, True
99
+ return None, any(path.endswith(m.extensions) for m in matchers)
100
+
101
+
102
+ def _candidates(files: list[str], cfg: Config,
103
+ matchers: tuple[_ScopeMatch, ...]) -> list[tuple[str, str | None]]:
104
+ match_glob = exclude_matcher(cfg.exclude_globs)
105
+ out = []
106
+ for raw_path in files:
107
+ path = raw_path.replace("\\", "/")
108
+ if excluded(path, match_glob):
109
+ continue
110
+ owner, known_language = _candidate(path, matchers)
111
+ if known_language:
112
+ out.append((path, owner))
113
+ return out
114
+
115
+
116
+ def _oversize(path: str, max_bytes: int | None, size_of) -> int | None:
117
+ """The file's size when the byte ceiling puts it out of reach, else None.
118
+
119
+ No limit means no stat: the size lookup is a syscall per file, and a repo
120
+ without max_file_bytes must not pay for a rule it never set.
121
+ """
122
+ if max_bytes is None or size_of is None:
123
+ return None
124
+ size = size_of(path)
125
+ return size if size > max_bytes else None
126
+
127
+
128
+ def _partition(candidates: list[tuple[str, str | None]], scopes: tuple[Scope, ...],
129
+ max_bytes: int | None, size_of):
130
+ assigned: dict[str, list[str]] = {s.name: [] for s in scopes}
131
+ unclaimed, oversized = [], []
132
+ for path, owner in candidates:
133
+ size = _oversize(path, max_bytes, size_of)
134
+ if size is not None:
135
+ oversized.append((path, size))
136
+ elif owner is not None:
137
+ assigned[owner].append(path)
138
+ else:
139
+ unclaimed.append(path)
140
+ return assigned, unclaimed, oversized
141
+
142
+
143
+ def scan_files(files: list[str], cfg: Config, *,
144
+ size_of: Callable[[str], int] | None = None) -> Universe:
145
+ """The whole verdict. `size_of` is injected so this stays pure; the shell
146
+ layer passes a working-tree stat, and callers with no tree pass nothing."""
147
+ assigned, unclaimed, oversized = _partition(
148
+ _candidates(files, cfg, _scope_matchers(cfg.scopes)),
149
+ cfg.scopes, cfg.max_file_bytes, size_of)
150
+ return Universe({name: sorted(paths) for name, paths in assigned.items()},
151
+ tuple(sorted(unclaimed)), tuple(sorted(oversized)))
152
+
153
+
154
+ def assign_files(files: list[str], cfg: Config, *,
155
+ size_of: Callable[[str], int] | None = None) -> dict[str, list[str]]:
156
+ """Just the per-scope mapping, for callers with no use for the dropped sets."""
157
+ return scan_files(files, cfg, size_of=size_of).by_scope
crapkit/verify.py ADDED
@@ -0,0 +1,194 @@
1
+ """The verdict. Pure: fresh scored rows + changed ranges + ratchet + failure sets in, Verdict out.
2
+
3
+ Three independent checks, all must hold:
4
+ - Gate: every function a change touched sits at CRAP <= target (coverage cannot
5
+ save cc > target; that is the target's design).
6
+ - Ratchet: no function above target scores worse than its recorded high-water
7
+ mark, touched or not (coverage rot regresses functions nobody edited).
8
+ - Failures: the fresh failure set adds nothing over the baseline's (the suite
9
+ is never assumed green; 98 pre-existing failures measured on day one).
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from bisect import bisect_left, bisect_right
14
+ from collections.abc import Iterator
15
+ from typing import NamedTuple
16
+
17
+ from .ratchet import RatchetEntry
18
+ from .score import ScoredRow, parse_scored_tsv, scored_tsv_lines
19
+
20
+
21
+ def diff_uncovered(changed_ranges: dict, missing: dict) -> list[tuple[str, int]]:
22
+ """Changed lines (new-file coordinates) whose statement never ran — where
23
+ the next bug ships. Files no lane measured stay silent (absent from missing)."""
24
+ out = []
25
+ for path, ranges in sorted(changed_ranges.items()):
26
+ dead = missing.get(path)
27
+ if not dead:
28
+ continue
29
+ # Sort the file's dead lines ONCE: sorting (and linearly scanning) them
30
+ # per hunk was O(hunks x dead log dead) for a file whose dead set never
31
+ # changes. Sorted, each hunk is two bisects and a slice.
32
+ ordered = sorted(dead)
33
+ for start, end in ranges:
34
+ out.extend((path, line)
35
+ for line in ordered[bisect_left(ordered, start):bisect_right(ordered, end)])
36
+ return out
37
+
38
+
39
+ class PortableBaseline(NamedTuple):
40
+ commit: str
41
+ kind: str
42
+ rows: list[ScoredRow]
43
+
44
+
45
+ def baseline_tsv_lines(commit: str, kind: str, rows: list[ScoredRow]) -> Iterator[str]:
46
+ """A baseline run as a file the repo can carry: a commit stamp, then the
47
+ run's scored export. The store lives in a gitignored .crapkit/, so a fresh
48
+ clone has nothing else to name what it is being measured against."""
49
+ yield f"# commit={commit} run_kind={kind}\n"
50
+ yield from scored_tsv_lines(rows)
51
+
52
+
53
+ def _stamp_fields(stamp: str) -> dict[str, str]:
54
+ return dict(part.split("=", 1) for part in stamp.removeprefix("# ").split() if "=" in part)
55
+
56
+
57
+ def parse_baseline_tsv(text: str) -> PortableBaseline:
58
+ stamp, _, body = text.partition("\n")
59
+ fields = _stamp_fields(stamp)
60
+ if "commit" not in fields or "run_kind" not in fields:
61
+ raise ValueError(
62
+ f"a baseline file starts with `# commit=<sha> run_kind=<kind>`, got {stamp!r}")
63
+ return PortableBaseline(fields["commit"], fields["run_kind"], parse_scored_tsv(body))
64
+
65
+
66
+ class GateViolation(NamedTuple):
67
+ path: str
68
+ long_name: str
69
+ start: int
70
+ ccn: int
71
+ cov: float
72
+ crap: float
73
+ remedy: str
74
+ dirty: bool = False
75
+
76
+
77
+ class RatchetRegression(NamedTuple):
78
+ path: str
79
+ long_name: str
80
+ recorded: float
81
+ fresh_crap: float
82
+ dirty: bool = False
83
+
84
+
85
+ class Verdict(NamedTuple):
86
+ ok: bool
87
+ gate_violations: list[GateViolation]
88
+ ratchet_regressions: list[RatchetRegression]
89
+ new_failures: list[str]
90
+ dirty_failures: list[str]
91
+
92
+
93
+ def _id_forms(path: str) -> tuple[str, str]:
94
+ """The two shapes a junit classname takes for one file: the repo-relative
95
+ path (vitest, and pytest's `file` fallback) and pytest's dotted module."""
96
+ stem = path[:-3] if path.endswith(".py") else path
97
+ return path, stem.replace("/", ".")
98
+
99
+
100
+ def dirty_failure_ids(new_failures: list[str], dirty_paths: set[str]) -> list[str]:
101
+ """New failures whose test id names a file with uncommitted edits."""
102
+ forms = {form for path in dirty_paths for form in _id_forms(path)}
103
+ return [f for f in new_failures if f.split("::")[0] in forms]
104
+
105
+
106
+ def dirty_counts(verdict: Verdict) -> tuple[int, int]:
107
+ """(committed, dirty) over every finding kind, so one line says how much of
108
+ a verdict belongs to the tree as committed and how much to somebody's edits."""
109
+ dirty_ids = set(verdict.dirty_failures)
110
+ flags = ([v.dirty for v in verdict.gate_violations]
111
+ + [r.dirty for r in verdict.ratchet_regressions]
112
+ + [f in dirty_ids for f in verdict.new_failures])
113
+ dirty = sum(flags)
114
+ return len(flags) - dirty, dirty
115
+
116
+
117
+ def _touched(row: ScoredRow, ranges: dict[str, list[tuple[int, int]]]) -> bool:
118
+ spans = ranges.get(row.path)
119
+ if not spans:
120
+ return False
121
+ return any(not (hi < row.start or lo > row.end) for lo, hi in spans)
122
+
123
+
124
+ def touched_rows(rows: list[ScoredRow],
125
+ changed_ranges: dict[str, list[tuple[int, int]]]) -> list[ScoredRow]:
126
+ """The gate's selection without its policy: rows whose span a change overlaps.
127
+
128
+ Every gate in crapkit judges touched functions only — untouched debt is the
129
+ ratchet's business. `rescore --gate` reuses this so its verdict and the
130
+ pre-commit hook's cannot disagree about which functions were even in scope.
131
+ """
132
+ return [r for r in rows if _touched(r, changed_ranges)]
133
+
134
+
135
+ def worst_twins(fresh: list[ScoredRow]) -> dict[tuple[str, str], ScoredRow]:
136
+ """Twins share (path, long_name); the WORST twin represents the key, so a
137
+ regression can never hide behind (nor a clean sibling tighten past) it."""
138
+ worst: dict[tuple[str, str], ScoredRow] = {}
139
+ for r in fresh:
140
+ key = (r.path, r.long_name)
141
+ if key not in worst or r.crap > worst[key].crap:
142
+ worst[key] = r
143
+ return worst
144
+
145
+
146
+ def _gate_violations(fresh, changed_ranges, target, scope_targets, dirty) -> list[GateViolation]:
147
+ gate = [
148
+ GateViolation(r.path, r.long_name, r.start, r.ccn, r.cov, r.crap, r.remedy,
149
+ r.path in dirty)
150
+ for r in fresh
151
+ if r.crap > (scope_targets or {}).get(r.scope, target) and _touched(r, changed_ranges)
152
+ ]
153
+ gate.sort(key=lambda v: (-v.crap, v.path, v.start))
154
+ return gate
155
+
156
+
157
+ def _ratchet_regressions(fresh, ratchet, dirty) -> list[RatchetRegression]:
158
+ worst_by_key = worst_twins(fresh)
159
+ regressions = []
160
+ for entry in ratchet:
161
+ row = worst_by_key.get((entry.path, entry.long_name))
162
+ # Compare at the precision the mark is STORED at: marks live as 4dp
163
+ # strings, and cov = covered/total makes longer decimals routine — an
164
+ # unrounded compare wedges an unchanged tree against its own mark.
165
+ if row is not None and round(row.crap, 4) > entry.crap:
166
+ regressions.append(RatchetRegression(entry.path, entry.long_name, entry.crap,
167
+ round(row.crap, 4), entry.path in dirty))
168
+ regressions.sort(key=lambda r: (-(r.fresh_crap - r.recorded), r.path))
169
+ return regressions
170
+
171
+
172
+ def evaluate(
173
+ *,
174
+ fresh: list[ScoredRow],
175
+ changed_ranges: dict[str, list[tuple[int, int]]],
176
+ ratchet: list[RatchetEntry],
177
+ baseline_failures: set[str],
178
+ fresh_failures: set[str],
179
+ target: int,
180
+ scope_targets: dict[str, int] | None = None,
181
+ dirty_paths: set[str] | None = None,
182
+ ) -> Verdict:
183
+ dirty = dirty_paths or set()
184
+ gate = _gate_violations(fresh, changed_ranges, target, scope_targets, dirty)
185
+ regressions = _ratchet_regressions(fresh, ratchet, dirty)
186
+ new_failures = sorted(fresh_failures - baseline_failures)
187
+
188
+ return Verdict(
189
+ ok=not gate and not regressions and not new_failures,
190
+ gate_violations=gate,
191
+ ratchet_regressions=regressions,
192
+ new_failures=new_failures,
193
+ dirty_failures=dirty_failure_ids(new_failures, dirty),
194
+ )
crapkit/watch.py ADDED
@@ -0,0 +1,112 @@
1
+ """Watch-mode core. Pure: mtime snapshots in, changed paths out.
2
+
3
+ The polling loop in the CLI stays a thin shell around this; stdlib mtimes,
4
+ no filesystem-event dependency, works the same on every host.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import os
9
+ import stat
10
+ from pathlib import Path
11
+
12
+
13
+ def _stat_mtime(path: Path) -> float | None:
14
+ """One stat, or None for anything that is not a readable regular file.
15
+
16
+ `is_file()` followed by `stat()` paid the syscall TWICE per tracked file,
17
+ every poll interval. The S_ISREG check keeps directories out, which is the
18
+ only thing is_file() was buying.
19
+ """
20
+ try:
21
+ st = os.stat(path)
22
+ except OSError:
23
+ return None
24
+ return st.st_mtime if stat.S_ISREG(st.st_mode) else None
25
+
26
+
27
+ def _entry_mtime(entry: os.DirEntry) -> float | None:
28
+ """The same answer for an already-listed name. On Windows the times came
29
+ with the listing, so this costs no syscall at all; the guard is for the
30
+ hosts where it does one, and for a name that died between the two."""
31
+ try:
32
+ st = entry.stat()
33
+ except OSError:
34
+ return None
35
+ return st.st_mtime if stat.S_ISREG(st.st_mode) else None
36
+
37
+
38
+ def _by_directory(files: list[str]) -> dict[str, dict[str, str]]:
39
+ """{directory: {basename: path}} — the grouping one listing can answer.
40
+
41
+ Paths are the repo-relative, slash-separated ones git reports. Anything
42
+ shaped otherwise simply groups under the root and misses its listing, which
43
+ costs it a stat and nothing else.
44
+ """
45
+ grouped: dict[str, dict[str, str]] = {}
46
+ for rel in files:
47
+ parent, _, base = rel.rpartition("/")
48
+ grouped.setdefault(parent, {})[base] = rel
49
+ return grouped
50
+
51
+
52
+ def _keep_wanted(entry: os.DirEntry, names: dict[str, str], out: dict[str, float]) -> None:
53
+ """A listing walks a whole directory; only the tracked names are wanted,
54
+ and only they are worth a stat on the hosts where one is charged."""
55
+ rel = names.get(entry.name)
56
+ if rel is None:
57
+ return
58
+ mtime = _entry_mtime(entry)
59
+ if mtime is not None:
60
+ out[rel] = mtime
61
+
62
+
63
+ def _list_directory(directory: Path, names: dict[str, str], out: dict[str, float]) -> None:
64
+ """Everything one listing can answer. A directory that cannot be read (gone,
65
+ refused, replaced by a file) answers nothing and leaves every name in it to
66
+ its own stat, so it costs speed and never an entry."""
67
+ try:
68
+ with os.scandir(directory) as entries:
69
+ for entry in entries:
70
+ _keep_wanted(entry, names, out)
71
+ except OSError:
72
+ return
73
+
74
+
75
+ def _listed_mtimes(root: Path, grouped: dict[str, dict[str, str]]) -> dict[str, float]:
76
+ out: dict[str, float] = {}
77
+ for parent, names in grouped.items():
78
+ _list_directory(root / parent if parent else root, names, out)
79
+ return out
80
+
81
+
82
+ def snapshot_mtimes(root: Path, files: list[str]) -> dict[str, float]:
83
+ """The mtime of every tracked file that is one; the rest simply absent.
84
+
85
+ One os.scandir per DIRECTORY, not one os.stat per FILE. On Windows the
86
+ times ride in the listing itself, so a 14,152-file tree polled every two
87
+ seconds dropped from 275 ms of stats to 46 ms of listings; elsewhere the
88
+ listing at least answers from a directory handle instead of walking the
89
+ whole path again once per file. Whatever the listing did not answer for
90
+ still gets its own stat, so a newborn file, a name the filesystem spells
91
+ with different case, and an unreadable directory all behave exactly as they
92
+ did when every file was stat-ed.
93
+
94
+ The result is built in `files` order, not listing order: two polls of one
95
+ unchanged tree have to produce the same mapping, and no filesystem promises
96
+ the order it enumerates in.
97
+ """
98
+ listed = _listed_mtimes(root, _by_directory(files))
99
+ out: dict[str, float] = {}
100
+ for rel in files:
101
+ mtime = listed.get(rel)
102
+ if mtime is None:
103
+ mtime = _stat_mtime(root / rel)
104
+ if mtime is not None:
105
+ out[rel] = mtime
106
+ return out
107
+
108
+
109
+ def changed_paths(before: dict[str, float], after: dict[str, float]) -> list[str]:
110
+ """New, modified, and deleted paths between two snapshots, sorted."""
111
+ moved = {p for p, m in after.items() if before.get(p) != m}
112
+ return sorted(moved | (set(before) - set(after)))