crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/config.py
ADDED
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
"""crapkit.toml parsing. Pure: text in, Config out; every rejection is a ConfigError (exit 3)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import tomllib
|
|
5
|
+
from typing import NamedTuple
|
|
6
|
+
|
|
7
|
+
from .errors import ConfigError
|
|
8
|
+
|
|
9
|
+
SUPPORTED_LANGUAGES = frozenset({"typescript", "tsx", "javascript", "python", "swift"})
|
|
10
|
+
SUPPORTED_PARSERS = frozenset({"istanbul", "coveragepy"})
|
|
11
|
+
DEFAULT_TARGET = 6
|
|
12
|
+
|
|
13
|
+
_SOURCE_SUFFIXES = (".ts", ".tsx", ".mts", ".js", ".jsx", ".mjs", ".py")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class Scope(NamedTuple):
|
|
17
|
+
name: str
|
|
18
|
+
paths: tuple[str, ...]
|
|
19
|
+
languages: tuple[str, ...]
|
|
20
|
+
target: int | None = None # per-scope ceiling; None = the repo default
|
|
21
|
+
# Code no test can reach (production-only scripts, generated shims): scored
|
|
22
|
+
# cc-only, and no lane has to claim it.
|
|
23
|
+
coverage_optional: bool = False
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class Lane(NamedTuple):
|
|
27
|
+
name: str
|
|
28
|
+
command: str
|
|
29
|
+
artifact: str
|
|
30
|
+
parser: str
|
|
31
|
+
scopes: tuple[str, ...]
|
|
32
|
+
cwd: str = ""
|
|
33
|
+
path_prefix: str = ""
|
|
34
|
+
env: tuple[tuple[str, str], ...] = ()
|
|
35
|
+
full_suite: bool = True
|
|
36
|
+
container_ok: bool = False
|
|
37
|
+
results_artifact: str = ""
|
|
38
|
+
timeout_seconds: int = 0 # 0 = no crapkit-owned timeout
|
|
39
|
+
retries: int = 0
|
|
40
|
+
retest_command: str = "" # {tests} template for the flake retry before exit 8
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _validate_coveragepy_command(name: str, command: str) -> None:
|
|
44
|
+
# Subset coverage under a suite with cross-file pollution is run-order-dependent;
|
|
45
|
+
# a full-suite lane refuses positional narrowing. Scoped suites opt out with
|
|
46
|
+
# full_suite = false, an explicit and reviewable decision.
|
|
47
|
+
tokens = command.split()
|
|
48
|
+
if "pytest" not in " ".join(tokens):
|
|
49
|
+
return
|
|
50
|
+
seen_pytest = False
|
|
51
|
+
for tok in tokens:
|
|
52
|
+
if tok.endswith("pytest"):
|
|
53
|
+
seen_pytest = True
|
|
54
|
+
continue
|
|
55
|
+
if seen_pytest and not tok.startswith("-"):
|
|
56
|
+
raise ConfigError(
|
|
57
|
+
f"lane {name!r}: positional argument {tok!r} narrows a full-suite coverage run; "
|
|
58
|
+
f"drop it or set full_suite = false deliberately")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _asks_for_coverage(tokens: list[str]) -> bool:
|
|
62
|
+
return any(t == "--coverage" or t.startswith("--coverage") for t in tokens)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _first_filter_position(tokens: list[str]) -> int:
|
|
66
|
+
# Only tokens after the test runner's `run` subcommand can be positional file
|
|
67
|
+
# filters; the runner script path itself (node scripts/run-vitest.mjs ...) is not.
|
|
68
|
+
return tokens.index("run") + 1 if "run" in tokens else 0
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _is_file_filter(tok: str, preceding: str) -> bool:
|
|
72
|
+
if tok.startswith("-"):
|
|
73
|
+
return False # a flag (e.g. --coverage.exclude=**/*.test.ts) is never a positional filter
|
|
74
|
+
return tok.endswith(_SOURCE_SUFFIXES) and preceding != "--config"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _validate_istanbul_command(name: str, command: str) -> None:
|
|
78
|
+
# The measured vitest trap: any file filter passed beside --coverage silently
|
|
79
|
+
# narrows the coverage include set. A lane command is fixed configuration, so
|
|
80
|
+
# the combination is a config error, not a runtime surprise.
|
|
81
|
+
tokens = command.split()
|
|
82
|
+
if not _asks_for_coverage(tokens):
|
|
83
|
+
return
|
|
84
|
+
for i in range(_first_filter_position(tokens), len(tokens)):
|
|
85
|
+
if _is_file_filter(tokens[i], tokens[i - 1]):
|
|
86
|
+
raise ConfigError(
|
|
87
|
+
f"lane {name!r}: file filter {tokens[i]!r} combined with --coverage silently narrows "
|
|
88
|
+
f"the coverage include set; drop the filter or use a dedicated config")
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class Config(NamedTuple):
|
|
92
|
+
target: int = DEFAULT_TARGET
|
|
93
|
+
|
|
94
|
+
@property
|
|
95
|
+
def scope_targets(self) -> dict[str, int]:
|
|
96
|
+
"""Every scope's effective ceiling: its own target or the repo default."""
|
|
97
|
+
return {s.name: (s.target if s.target is not None else self.target) for s in self.scopes}
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def coverage_optional_scopes(self) -> frozenset[str]:
|
|
101
|
+
"""The scopes scored cc-only: no coverage join, and no lane required."""
|
|
102
|
+
return frozenset(s.name for s in self.scopes if s.coverage_optional)
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def scope_paths(self) -> dict[str, tuple[str, ...]]:
|
|
106
|
+
"""Every scope's declared paths, by name — what a lane's scopes resolve to."""
|
|
107
|
+
return {s.name: s.paths for s in self.scopes}
|
|
108
|
+
scopes: tuple[Scope, ...] = ()
|
|
109
|
+
exclude_globs: tuple[str, ...] = ()
|
|
110
|
+
max_file_bytes: int | None = None # files bigger than this leave the corpus; None = no limit
|
|
111
|
+
churn_window_months: int = 12
|
|
112
|
+
worklist_floor: int = 5
|
|
113
|
+
worklist_top: int = 50
|
|
114
|
+
lanes: tuple[Lane, ...] = ()
|
|
115
|
+
ratchet_file: str = "crapkit-ratchet.tsv"
|
|
116
|
+
alert_command: str = ""
|
|
117
|
+
scoped_tests: tuple[tuple[str, str], ...] = ()
|
|
118
|
+
mutation_command: str = "" # the suite run once per mutant; nonzero exit = killed
|
|
119
|
+
mutation_timeout_seconds: int = 300 # a mutant that loops forever counts as killed
|
|
120
|
+
mutation_workers: int = 1 # >1 runs mutants in that many detached git worktrees
|
|
121
|
+
diff_uncovered_max: int | None = None # verify exit 9 past this many dead changed lines
|
|
122
|
+
debt_max_age_months: int | None = None # ratchet report --enforce flags older marks
|
|
123
|
+
repayment_min_per_30d: int | None = None # --enforce flags a stalled burn-down
|
|
124
|
+
max_parallel_lanes: int = 1 # lanes running at once; 1 = strictly serial
|
|
125
|
+
analysis_workers: int = 0 # lizard pool size; 0 = one worker per core
|
|
126
|
+
# Operational traps the repo learned the hard way. They lived as TOML
|
|
127
|
+
# comments, which the parser drops, so no payload could ever quote them.
|
|
128
|
+
notes: tuple[str, ...] = ()
|
|
129
|
+
# Only the scopes that wrote one, so a scope with nothing to say costs a
|
|
130
|
+
# payload no key. A plain dict, because these end up in --json output.
|
|
131
|
+
scope_notes: dict[str, tuple[str, ...]] = {}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def load_config_text(text: str) -> Config:
|
|
135
|
+
try:
|
|
136
|
+
raw = tomllib.loads(text)
|
|
137
|
+
except tomllib.TOMLDecodeError as exc:
|
|
138
|
+
raise ConfigError(f"crapkit.toml does not parse: {exc}") from exc
|
|
139
|
+
try:
|
|
140
|
+
return _build_config(raw)
|
|
141
|
+
except KeyError as exc:
|
|
142
|
+
raise ConfigError(f"crapkit.toml is missing a required key: {exc}") from exc
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _parse_scope(row: dict) -> Scope:
|
|
146
|
+
languages = tuple(row.get("languages", ()))
|
|
147
|
+
unknown = set(languages) - SUPPORTED_LANGUAGES
|
|
148
|
+
if unknown:
|
|
149
|
+
raise ConfigError(f"unsupported language(s) {sorted(unknown)} in scope {row.get('name')!r}")
|
|
150
|
+
scope_target = row.get("target")
|
|
151
|
+
if scope_target is not None and (not isinstance(scope_target, int) or scope_target < 1):
|
|
152
|
+
raise ConfigError(f"scope {row.get('name')!r}: target must be a positive int, got {scope_target!r}")
|
|
153
|
+
return Scope(name=row["name"], paths=tuple(row["paths"]), languages=languages,
|
|
154
|
+
target=scope_target,
|
|
155
|
+
coverage_optional=bool(row.get("coverage_optional", False)))
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _notes(row: dict, where: str) -> tuple[str, ...]:
|
|
159
|
+
"""The `notes` list, rejected unless every entry is a string.
|
|
160
|
+
|
|
161
|
+
A bare `notes = "..."` is the trap TOML sets: it is iterable, so it would
|
|
162
|
+
load as one note per letter and every reader would print them that way.
|
|
163
|
+
"""
|
|
164
|
+
raw = row.get("notes", [])
|
|
165
|
+
if not isinstance(raw, list) or not all(isinstance(item, str) for item in raw):
|
|
166
|
+
raise ConfigError(f"{where}: notes must be a list of strings, got {raw!r}")
|
|
167
|
+
return tuple(raw)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _parse_scopes(rows) -> tuple[tuple[Scope, ...], dict[str, tuple[str, ...]]]:
|
|
171
|
+
"""Every [[scope]] row and its notes, off ONE walk of the rows.
|
|
172
|
+
|
|
173
|
+
Notes hang on the same rows the scopes come from, so collecting them in a
|
|
174
|
+
second pass would re-read and re-validate every row for nothing — and, when
|
|
175
|
+
the rows arrive as an iterator, would find none of them.
|
|
176
|
+
"""
|
|
177
|
+
scopes: list[Scope] = []
|
|
178
|
+
notes: dict[str, tuple[str, ...]] = {}
|
|
179
|
+
for row in rows:
|
|
180
|
+
scope = _parse_scope(row)
|
|
181
|
+
scopes.append(scope)
|
|
182
|
+
row_notes = _notes(row, f"scope {scope.name!r}")
|
|
183
|
+
if row_notes:
|
|
184
|
+
notes[scope.name] = row_notes
|
|
185
|
+
return tuple(scopes), notes
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _validate_lane_command(parser: str, full_suite: bool, name: str, command: str) -> None:
|
|
189
|
+
if parser == "istanbul":
|
|
190
|
+
_validate_istanbul_command(name, command)
|
|
191
|
+
if parser == "coveragepy" and full_suite:
|
|
192
|
+
_validate_coveragepy_command(name, command)
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _parse_lane(row: dict, scope_names: set) -> Lane:
|
|
196
|
+
parser = row["parser"]
|
|
197
|
+
if parser not in SUPPORTED_PARSERS:
|
|
198
|
+
raise ConfigError(f"lane {row.get('name')!r}: unsupported parser {parser!r}")
|
|
199
|
+
lane_scopes = tuple(row.get("scopes", ()))
|
|
200
|
+
unknown_scopes = set(lane_scopes) - scope_names
|
|
201
|
+
if unknown_scopes:
|
|
202
|
+
raise ConfigError(f"lane {row.get('name')!r} references undeclared scope(s) {sorted(unknown_scopes)}")
|
|
203
|
+
full_suite = bool(row.get("full_suite", True))
|
|
204
|
+
_validate_lane_command(parser, full_suite, row.get("name", "?"), row["command"])
|
|
205
|
+
return Lane(name=row["name"], command=row["command"], artifact=row["artifact"],
|
|
206
|
+
parser=parser, scopes=lane_scopes,
|
|
207
|
+
cwd=row.get("cwd", ""), path_prefix=row.get("path_prefix", ""),
|
|
208
|
+
env=tuple(sorted((str(k), str(v)) for k, v in row.get("env", {}).items())),
|
|
209
|
+
full_suite=full_suite, container_ok=bool(row.get("container_ok", False)),
|
|
210
|
+
results_artifact=row.get("results_artifact", ""),
|
|
211
|
+
timeout_seconds=_nonneg_int(row, "timeout_seconds"),
|
|
212
|
+
retries=_nonneg_int(row, "retries"),
|
|
213
|
+
retest_command=row.get("retest_command", ""))
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _nonneg_int(row: dict, key: str) -> int:
|
|
217
|
+
value = row.get(key, 0)
|
|
218
|
+
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
|
|
219
|
+
raise ConfigError(
|
|
220
|
+
f"lane {row.get('name', '?')!r}: {key} must be a non-negative int, got {value!r}")
|
|
221
|
+
return value
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _reject_shared_artifacts(lanes: list) -> None:
|
|
225
|
+
seen_artifacts: dict[str, str] = {}
|
|
226
|
+
for lane in lanes:
|
|
227
|
+
for artifact in filter(None, (lane.artifact, lane.results_artifact)):
|
|
228
|
+
if artifact in seen_artifacts and seen_artifacts[artifact] != lane.name:
|
|
229
|
+
raise ConfigError(
|
|
230
|
+
f"lanes {seen_artifacts[artifact]!r} and {lane.name!r} share the artifact path "
|
|
231
|
+
f"{artifact!r}; reused paths cross-attribute coverage under --reuse-artifacts")
|
|
232
|
+
seen_artifacts[artifact] = lane.name
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _build_config(raw: dict) -> Config:
|
|
236
|
+
scope_rows = raw.get("scope", [])
|
|
237
|
+
if not scope_rows:
|
|
238
|
+
raise ConfigError("crapkit.toml declares no [[scope]] — nothing to analyze")
|
|
239
|
+
scopes, scope_notes = _parse_scopes(scope_rows)
|
|
240
|
+
scope_names = {s.name for s in scopes}
|
|
241
|
+
lanes = [_parse_lane(row, scope_names) for row in raw.get("lane", [])]
|
|
242
|
+
_reject_shared_artifacts(lanes)
|
|
243
|
+
main = raw.get("crapkit", {})
|
|
244
|
+
return Config(
|
|
245
|
+
target=int(main.get("target", DEFAULT_TARGET)),
|
|
246
|
+
scopes=scopes,
|
|
247
|
+
exclude_globs=tuple(raw.get("exclude", {}).get("globs", ())),
|
|
248
|
+
max_file_bytes=_optional_int(raw.get("exclude", {}), "max_file_bytes"),
|
|
249
|
+
churn_window_months=int(main.get("churn_window_months", 12)),
|
|
250
|
+
worklist_floor=int(main.get("worklist_floor", 5)),
|
|
251
|
+
worklist_top=int(main.get("worklist_top", 50)),
|
|
252
|
+
lanes=tuple(lanes),
|
|
253
|
+
ratchet_file=main.get("ratchet_file", "crapkit-ratchet.tsv"),
|
|
254
|
+
alert_command=main.get("alert_command", ""),
|
|
255
|
+
scoped_tests=tuple(sorted((str(k), str(v)) for k, v in main.get("scoped_tests", {}).items())),
|
|
256
|
+
mutation_command=main.get("mutation_command", ""),
|
|
257
|
+
mutation_timeout_seconds=int(main.get("mutation_timeout_seconds", 300)),
|
|
258
|
+
mutation_workers=_positive_int(main, "mutation_workers", 1),
|
|
259
|
+
diff_uncovered_max=_optional_int(main, "diff_uncovered_max"),
|
|
260
|
+
debt_max_age_months=_optional_int(main, "debt_max_age_months"),
|
|
261
|
+
repayment_min_per_30d=_optional_int(main, "repayment_min_per_30d"),
|
|
262
|
+
max_parallel_lanes=_bounded_int(main, "max_parallel_lanes", default=1, minimum=1),
|
|
263
|
+
analysis_workers=_bounded_int(main, "analysis_workers", default=0, minimum=0),
|
|
264
|
+
notes=_notes(main, "[crapkit]"),
|
|
265
|
+
scope_notes=scope_notes,
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _bounded_int(main: dict, key: str, *, default: int, minimum: int) -> int:
|
|
270
|
+
value = main.get(key, default)
|
|
271
|
+
if not isinstance(value, int) or isinstance(value, bool) or value < minimum:
|
|
272
|
+
raise ConfigError(f"{key} must be an int >= {minimum}, got {value!r}")
|
|
273
|
+
return value
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _positive_int(main: dict, key: str, default: int) -> int:
|
|
277
|
+
value = main.get(key, default)
|
|
278
|
+
if not isinstance(value, int) or isinstance(value, bool) or value < 1:
|
|
279
|
+
raise ConfigError(f"{key} must be a positive int, got {value!r}")
|
|
280
|
+
return value
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _optional_int(main: dict, key: str) -> int | None:
|
|
284
|
+
value = main.get(key)
|
|
285
|
+
if value is None:
|
|
286
|
+
return None
|
|
287
|
+
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
|
|
288
|
+
raise ConfigError(f"{key} must be a non-negative int, got {value!r}")
|
|
289
|
+
return value
|
crapkit/coupling.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Change coupling from the churn git log. Pure.
|
|
2
|
+
|
|
3
|
+
Files that keep landing in the same commits are coupled, whatever the import
|
|
4
|
+
graph says — the hidden dependency the compiler cannot show. Bulk commits are
|
|
5
|
+
skipped for pairing: a 40-file sweep says nothing about any particular pair.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Iterable, Iterator
|
|
10
|
+
from itertools import combinations
|
|
11
|
+
|
|
12
|
+
MAX_COMMIT_FILES = 30
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _commit_file_sets(lines: Iterable[str]) -> Iterator[set[str]]:
|
|
16
|
+
"""One file set per commit, yielded as the log streams past.
|
|
17
|
+
|
|
18
|
+
A %x01 line opens a commit and every non-blank line until the next one is a
|
|
19
|
+
path. Empty sets are yielded rather than skipped: they add no file counts
|
|
20
|
+
and form no pairs, so the caller cannot tell them from a skip.
|
|
21
|
+
"""
|
|
22
|
+
files: set[str] = set()
|
|
23
|
+
past_header = False
|
|
24
|
+
for raw in lines:
|
|
25
|
+
line = raw.strip()
|
|
26
|
+
if line.startswith("\x01"):
|
|
27
|
+
yield files
|
|
28
|
+
files, past_header = set(), True
|
|
29
|
+
continue
|
|
30
|
+
if past_header and line:
|
|
31
|
+
files.add(line.replace("\\", "/"))
|
|
32
|
+
past_header = True # a log starting mid-commit opens on a severed header
|
|
33
|
+
yield files
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _rank_pairs(file_counts: dict, pair_counts: dict, min_support: int,
|
|
37
|
+
min_confidence: float, top: int | None) -> list[dict]:
|
|
38
|
+
out = []
|
|
39
|
+
for (a, b), support in pair_counts.items():
|
|
40
|
+
if support < min_support:
|
|
41
|
+
continue
|
|
42
|
+
confidence = max(support / file_counts[a], support / file_counts[b])
|
|
43
|
+
if confidence < min_confidence:
|
|
44
|
+
continue
|
|
45
|
+
out.append({"files": [a, b], "support": support, "confidence": round(confidence, 4)})
|
|
46
|
+
out.sort(key=lambda p: (-p["support"] * p["confidence"], p["files"]))
|
|
47
|
+
return out if top is None else out[:top]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _partner(pair: dict, path: str) -> dict:
|
|
51
|
+
a, b = pair["files"]
|
|
52
|
+
return {"path": b if a == path else a, "support": pair["support"],
|
|
53
|
+
"confidence": pair["confidence"]}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def partners(lines: Iterable[str], path: str, *, min_support: int = 5,
|
|
57
|
+
min_confidence: float = 0.5, top: int = 5) -> list[dict]:
|
|
58
|
+
"""One file's coupled partners, best first.
|
|
59
|
+
|
|
60
|
+
Ranked over EVERY qualifying pair before the cut: a global top applied first
|
|
61
|
+
would drop a quiet file's own partners behind the repo's noisiest pairs and
|
|
62
|
+
report it as uncoupled.
|
|
63
|
+
"""
|
|
64
|
+
ranked = change_coupling_lines(lines, min_support=min_support,
|
|
65
|
+
min_confidence=min_confidence, top=None)
|
|
66
|
+
return [_partner(p, path) for p in ranked if path in p["files"]][:top]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def change_coupling(log_text: str, *, min_support: int = 5, min_confidence: float = 0.5,
|
|
70
|
+
top: int | None = 50) -> list[dict]:
|
|
71
|
+
"""Whole-text entrypoint: the log already in hand."""
|
|
72
|
+
return change_coupling_lines(log_text.splitlines(), min_support=min_support,
|
|
73
|
+
min_confidence=min_confidence, top=top)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def change_coupling_lines(lines: Iterable[str], *, min_support: int = 5,
|
|
77
|
+
min_confidence: float = 0.5, top: int | None = 50) -> list[dict]:
|
|
78
|
+
"""Streaming entrypoint: the log is consumed once, one commit at a time."""
|
|
79
|
+
commits = _commit_file_sets(lines)
|
|
80
|
+
file_counts: dict[str, int] = {}
|
|
81
|
+
pair_counts: dict[tuple[str, str], int] = {}
|
|
82
|
+
for files in commits:
|
|
83
|
+
for f in files:
|
|
84
|
+
file_counts[f] = file_counts.get(f, 0) + 1
|
|
85
|
+
if len(files) > MAX_COMMIT_FILES:
|
|
86
|
+
continue
|
|
87
|
+
for pair in combinations(sorted(files), 2):
|
|
88
|
+
pair_counts[pair] = pair_counts.get(pair, 0) + 1
|
|
89
|
+
return _rank_pairs(file_counts, pair_counts, min_support, min_confidence, top)
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""Istanbul coverage-final.json parser. Pure: JSON text in, per-file function coverage out.
|
|
2
|
+
|
|
3
|
+
Branch hits map into function spans by line containment. A function with no
|
|
4
|
+
branches inside its span falls back to STATEMENT coverage in that span, and
|
|
5
|
+
only with no statements either to invocation (hit or not) — a straight-line
|
|
6
|
+
function half-executed must not read as fully covered. Written for the
|
|
7
|
+
AST-remapped output of @vitest/coverage-v8 >= 3.2, which is istanbul-schema-identical.
|
|
8
|
+
|
|
9
|
+
The artifact is a flat {abs_path: coverage} object and only one file's coverage
|
|
10
|
+
is ever needed at a time, so the outer object is SPLIT rather than parsed:
|
|
11
|
+
split_top_level walks the members and decodes each value on its own. A
|
|
12
|
+
whole-document json.loads on a 157 MB artifact peaked at 1,432 MB against
|
|
13
|
+
335 MB for the split, for the same output.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import heapq
|
|
18
|
+
import json
|
|
19
|
+
import re
|
|
20
|
+
from typing import Iterator, NamedTuple
|
|
21
|
+
|
|
22
|
+
from .errors import ToolError
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class FnCoverage(NamedTuple):
|
|
26
|
+
name: str
|
|
27
|
+
start: int
|
|
28
|
+
end: int
|
|
29
|
+
invoked: bool
|
|
30
|
+
branches_total: int
|
|
31
|
+
branches_covered: int
|
|
32
|
+
statements_total: int = 0
|
|
33
|
+
statements_covered: int = 0
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def coverage(self) -> float:
|
|
37
|
+
if self.branches_total > 0:
|
|
38
|
+
return self.branches_covered / self.branches_total
|
|
39
|
+
if self.statements_total > 0:
|
|
40
|
+
return self.statements_covered / self.statements_total
|
|
41
|
+
return 1.0 if self.invoked else 0.0
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _rel_path(abs_path: str, repo_root: str) -> str:
|
|
45
|
+
norm = abs_path.replace("\\", "/")
|
|
46
|
+
root = repo_root.replace("\\", "/").rstrip("/") + "/"
|
|
47
|
+
return norm[len(root):] if norm.startswith(root) else norm
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# --- top-level splitter ----------------------------------------------------
|
|
51
|
+
# One member's opening: whitespace, the separating comma, the key string, the
|
|
52
|
+
# colon. The key pattern is string-aware — an escaped quote inside a Windows
|
|
53
|
+
# path must not end it — and its alternation is unambiguous (a character is
|
|
54
|
+
# either not a quote/backslash, or an escape pair), so it scans linearly and
|
|
55
|
+
# cannot backtrack.
|
|
56
|
+
_MEMBER_RE = re.compile(r'\s*,?\s*("(?:[^"\\]|\\.)*")\s*:\s*', re.DOTALL)
|
|
57
|
+
_OPEN_RE = re.compile(r"\s*\{")
|
|
58
|
+
_CLOSE_RE = re.compile(r"\s*\}\s*\Z")
|
|
59
|
+
|
|
60
|
+
# raw_decode reads ONE value at a position and reports where it ended, which is
|
|
61
|
+
# what makes the split possible without writing a second parser. Hand-rolled
|
|
62
|
+
# brace matching finds the same boundary but pays a Python loop iteration per
|
|
63
|
+
# structural character: 4.75s of a 5.4s parse on a 21.6 MB artifact, against
|
|
64
|
+
# 0.23s here, because this scanner is the C one.
|
|
65
|
+
_DECODER = json.JSONDecoder()
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def split_top_level(text: str) -> Iterator[tuple[str, object]]:
|
|
69
|
+
"""Yield (key, value) per member of the outer object, decoding one value at
|
|
70
|
+
a time. Equivalent to json.loads(text).items() except that the whole
|
|
71
|
+
document is never a live dict: each value is dropped as the caller steps
|
|
72
|
+
past it, so peak memory tracks the LARGEST file, not their sum."""
|
|
73
|
+
opening = _OPEN_RE.match(text)
|
|
74
|
+
if opening is None:
|
|
75
|
+
raise ValueError("istanbul artifact is not a JSON object")
|
|
76
|
+
i = opening.end()
|
|
77
|
+
while True:
|
|
78
|
+
member = _MEMBER_RE.match(text, i)
|
|
79
|
+
if member is None:
|
|
80
|
+
_expect_end(text, i)
|
|
81
|
+
return
|
|
82
|
+
value, i = _DECODER.raw_decode(text, member.end())
|
|
83
|
+
yield json.loads(member.group(1)), value
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _expect_end(text: str, i: int) -> None:
|
|
87
|
+
if _CLOSE_RE.match(text, i) is None:
|
|
88
|
+
raise ValueError(f"unexpected content at offset {i}")
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _iter_files(text: str, repo_root: str) -> Iterator[tuple[str, dict]]:
|
|
92
|
+
"""(repo-relative path, coverage object) per artifact member. One file's
|
|
93
|
+
coverage is live at a time; the previous one is unreferenced on the next step."""
|
|
94
|
+
for abs_path, cov in split_top_level(text):
|
|
95
|
+
yield _rel_path(abs_path, repo_root), cov
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# --- span attribution ------------------------------------------------------
|
|
99
|
+
# mutable span layout while attributing: [name, start, end, invoked, b_total, b_cov, s_total, s_cov]
|
|
100
|
+
_B_TOTAL, _B_COV, _S_TOTAL, _S_COV = 4, 5, 6, 7
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _fn_spans(cov: dict) -> list[list]:
|
|
104
|
+
spans = []
|
|
105
|
+
for fid, fn in cov.get("fnMap", {}).items():
|
|
106
|
+
start = fn["decl"]["start"]["line"]
|
|
107
|
+
end = fn.get("loc", {}).get("end", {}).get("line") or start
|
|
108
|
+
invoked = cov.get("f", {}).get(fid, 0) > 0
|
|
109
|
+
spans.append([fn.get("name") or "(anonymous)", start, end, invoked, 0, 0, 0, 0])
|
|
110
|
+
spans.sort(key=lambda s: s[1])
|
|
111
|
+
return spans
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _branch_line(branch: dict) -> int | None:
|
|
115
|
+
return branch.get("loc", {}).get("start", {}).get("line")
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _stmt_line(stmt: dict) -> int | None:
|
|
119
|
+
return stmt.get("start", {}).get("line")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _query_lines(cov: dict) -> set[int]:
|
|
123
|
+
"""Every line the attribution will ask about, branches and statements both."""
|
|
124
|
+
lines = {_branch_line(b) for b in cov.get("branchMap", {}).values()}
|
|
125
|
+
lines |= {_stmt_line(s) for s in cov.get("statementMap", {}).values()}
|
|
126
|
+
lines.discard(None)
|
|
127
|
+
return lines
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _push_started(heap: list, ordered: list[list], nxt: int, line: int) -> int:
|
|
131
|
+
while nxt < len(ordered) and ordered[nxt][1] <= line:
|
|
132
|
+
span = ordered[nxt]
|
|
133
|
+
heapq.heappush(heap, (span[2] - span[1], -span[1], nxt, span))
|
|
134
|
+
nxt += 1
|
|
135
|
+
return nxt
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _drop_ended(heap: list, line: int) -> None:
|
|
139
|
+
"""Discard spans that closed before this line. Safe to do lazily and only at
|
|
140
|
+
the top: query lines only increase, so anything popped here can never
|
|
141
|
+
contain a later line either."""
|
|
142
|
+
while heap and heap[0][3][2] < line:
|
|
143
|
+
heapq.heappop(heap)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _span_owners(fn_spans: list[list], lines: set[int]) -> dict[int, list | None]:
|
|
147
|
+
"""line -> innermost containing span. A hit inside a nested function belongs
|
|
148
|
+
to that function, never to its encloser — else the nested one reads through
|
|
149
|
+
its encloser and the encloser answers for lines it can't fix.
|
|
150
|
+
|
|
151
|
+
Sweeping spans by start into a heap keyed (length, -start, index) settles
|
|
152
|
+
that in O((F + Q) log F) instead of a scan per query. The index term is
|
|
153
|
+
load-bearing: it is the sorted position, so an exact tie on (length, -start)
|
|
154
|
+
resolves to the span the old linear scan met first."""
|
|
155
|
+
ordered = sorted(fn_spans, key=lambda s: s[1])
|
|
156
|
+
heap: list[tuple] = []
|
|
157
|
+
owners: dict[int, list | None] = {}
|
|
158
|
+
nxt = 0
|
|
159
|
+
for line in sorted(lines):
|
|
160
|
+
nxt = _push_started(heap, ordered, nxt, line)
|
|
161
|
+
_drop_ended(heap, line)
|
|
162
|
+
owners[line] = heap[0][3] if heap else None
|
|
163
|
+
return owners
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _attach_branches(owners: dict[int, list | None], cov: dict) -> None:
|
|
167
|
+
hits_by_id = cov.get("b", {})
|
|
168
|
+
for bid, branch in cov.get("branchMap", {}).items():
|
|
169
|
+
best = owners.get(_branch_line(branch))
|
|
170
|
+
if best is not None:
|
|
171
|
+
hits = hits_by_id.get(bid, [])
|
|
172
|
+
best[_B_TOTAL] += len(hits)
|
|
173
|
+
best[_B_COV] += sum(1 for h in hits if h > 0)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _attach_statements(owners: dict[int, list | None], cov: dict) -> None:
|
|
177
|
+
hits_by_id = cov.get("s", {})
|
|
178
|
+
for sid, stmt in cov.get("statementMap", {}).items():
|
|
179
|
+
best = owners.get(_stmt_line(stmt))
|
|
180
|
+
if best is not None:
|
|
181
|
+
best[_S_TOTAL] += 1
|
|
182
|
+
best[_S_COV] += 1 if hits_by_id.get(sid, 0) > 0 else 0
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _file_coverage(cov: dict) -> list[FnCoverage]:
|
|
186
|
+
fn_spans = _fn_spans(cov)
|
|
187
|
+
owners = _span_owners(fn_spans, _query_lines(cov))
|
|
188
|
+
_attach_branches(owners, cov)
|
|
189
|
+
_attach_statements(owners, cov)
|
|
190
|
+
return [FnCoverage(*s) for s in fn_spans]
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _dead_lines(cov: dict) -> set[int]:
|
|
194
|
+
hits_by_id = cov.get("s", {})
|
|
195
|
+
dead = {_stmt_line(stmt)
|
|
196
|
+
for sid, stmt in cov.get("statementMap", {}).items()
|
|
197
|
+
if hits_by_id.get(sid, 0) == 0}
|
|
198
|
+
dead.discard(None)
|
|
199
|
+
return dead
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def parse_istanbul_missing(text: str, *, repo_root: str) -> dict[str, set[int]]:
|
|
203
|
+
"""Per measured file, the lines whose statement never ran — the
|
|
204
|
+
diff-coverage ground truth. Files with everything executed map to set()."""
|
|
205
|
+
try:
|
|
206
|
+
missing: dict[str, set[int]] = {}
|
|
207
|
+
for rel_path, cov in _iter_files(text, repo_root):
|
|
208
|
+
missing[rel_path] = _dead_lines(cov)
|
|
209
|
+
return missing
|
|
210
|
+
except Exception as exc:
|
|
211
|
+
raise ToolError(f"unparseable istanbul artifact: {exc}") from exc
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def parse_istanbul(text: str, *, repo_root: str) -> dict[str, list[FnCoverage]]:
|
|
215
|
+
try:
|
|
216
|
+
per_file: dict[str, list[FnCoverage]] = {}
|
|
217
|
+
for rel_path, cov in _iter_files(text, repo_root):
|
|
218
|
+
per_file[rel_path] = _file_coverage(cov)
|
|
219
|
+
if not per_file:
|
|
220
|
+
raise ToolError("istanbul artifact is empty (zero files) — the coverage run measured nothing")
|
|
221
|
+
return per_file
|
|
222
|
+
except ToolError:
|
|
223
|
+
raise
|
|
224
|
+
except Exception as exc:
|
|
225
|
+
raise ToolError(f"unparseable istanbul artifact: {exc}") from exc
|