crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/config.py ADDED
@@ -0,0 +1,289 @@
1
+ """crapkit.toml parsing. Pure: text in, Config out; every rejection is a ConfigError (exit 3)."""
2
+ from __future__ import annotations
3
+
4
+ import tomllib
5
+ from typing import NamedTuple
6
+
7
+ from .errors import ConfigError
8
+
9
+ SUPPORTED_LANGUAGES = frozenset({"typescript", "tsx", "javascript", "python", "swift"})
10
+ SUPPORTED_PARSERS = frozenset({"istanbul", "coveragepy"})
11
+ DEFAULT_TARGET = 6
12
+
13
+ _SOURCE_SUFFIXES = (".ts", ".tsx", ".mts", ".js", ".jsx", ".mjs", ".py")
14
+
15
+
16
+ class Scope(NamedTuple):
17
+ name: str
18
+ paths: tuple[str, ...]
19
+ languages: tuple[str, ...]
20
+ target: int | None = None # per-scope ceiling; None = the repo default
21
+ # Code no test can reach (production-only scripts, generated shims): scored
22
+ # cc-only, and no lane has to claim it.
23
+ coverage_optional: bool = False
24
+
25
+
26
+ class Lane(NamedTuple):
27
+ name: str
28
+ command: str
29
+ artifact: str
30
+ parser: str
31
+ scopes: tuple[str, ...]
32
+ cwd: str = ""
33
+ path_prefix: str = ""
34
+ env: tuple[tuple[str, str], ...] = ()
35
+ full_suite: bool = True
36
+ container_ok: bool = False
37
+ results_artifact: str = ""
38
+ timeout_seconds: int = 0 # 0 = no crapkit-owned timeout
39
+ retries: int = 0
40
+ retest_command: str = "" # {tests} template for the flake retry before exit 8
41
+
42
+
43
+ def _validate_coveragepy_command(name: str, command: str) -> None:
44
+ # Subset coverage under a suite with cross-file pollution is run-order-dependent;
45
+ # a full-suite lane refuses positional narrowing. Scoped suites opt out with
46
+ # full_suite = false, an explicit and reviewable decision.
47
+ tokens = command.split()
48
+ if "pytest" not in " ".join(tokens):
49
+ return
50
+ seen_pytest = False
51
+ for tok in tokens:
52
+ if tok.endswith("pytest"):
53
+ seen_pytest = True
54
+ continue
55
+ if seen_pytest and not tok.startswith("-"):
56
+ raise ConfigError(
57
+ f"lane {name!r}: positional argument {tok!r} narrows a full-suite coverage run; "
58
+ f"drop it or set full_suite = false deliberately")
59
+
60
+
61
+ def _asks_for_coverage(tokens: list[str]) -> bool:
62
+ return any(t == "--coverage" or t.startswith("--coverage") for t in tokens)
63
+
64
+
65
+ def _first_filter_position(tokens: list[str]) -> int:
66
+ # Only tokens after the test runner's `run` subcommand can be positional file
67
+ # filters; the runner script path itself (node scripts/run-vitest.mjs ...) is not.
68
+ return tokens.index("run") + 1 if "run" in tokens else 0
69
+
70
+
71
+ def _is_file_filter(tok: str, preceding: str) -> bool:
72
+ if tok.startswith("-"):
73
+ return False # a flag (e.g. --coverage.exclude=**/*.test.ts) is never a positional filter
74
+ return tok.endswith(_SOURCE_SUFFIXES) and preceding != "--config"
75
+
76
+
77
+ def _validate_istanbul_command(name: str, command: str) -> None:
78
+ # The measured vitest trap: any file filter passed beside --coverage silently
79
+ # narrows the coverage include set. A lane command is fixed configuration, so
80
+ # the combination is a config error, not a runtime surprise.
81
+ tokens = command.split()
82
+ if not _asks_for_coverage(tokens):
83
+ return
84
+ for i in range(_first_filter_position(tokens), len(tokens)):
85
+ if _is_file_filter(tokens[i], tokens[i - 1]):
86
+ raise ConfigError(
87
+ f"lane {name!r}: file filter {tokens[i]!r} combined with --coverage silently narrows "
88
+ f"the coverage include set; drop the filter or use a dedicated config")
89
+
90
+
91
+ class Config(NamedTuple):
92
+ target: int = DEFAULT_TARGET
93
+
94
+ @property
95
+ def scope_targets(self) -> dict[str, int]:
96
+ """Every scope's effective ceiling: its own target or the repo default."""
97
+ return {s.name: (s.target if s.target is not None else self.target) for s in self.scopes}
98
+
99
+ @property
100
+ def coverage_optional_scopes(self) -> frozenset[str]:
101
+ """The scopes scored cc-only: no coverage join, and no lane required."""
102
+ return frozenset(s.name for s in self.scopes if s.coverage_optional)
103
+
104
+ @property
105
+ def scope_paths(self) -> dict[str, tuple[str, ...]]:
106
+ """Every scope's declared paths, by name — what a lane's scopes resolve to."""
107
+ return {s.name: s.paths for s in self.scopes}
108
+ scopes: tuple[Scope, ...] = ()
109
+ exclude_globs: tuple[str, ...] = ()
110
+ max_file_bytes: int | None = None # files bigger than this leave the corpus; None = no limit
111
+ churn_window_months: int = 12
112
+ worklist_floor: int = 5
113
+ worklist_top: int = 50
114
+ lanes: tuple[Lane, ...] = ()
115
+ ratchet_file: str = "crapkit-ratchet.tsv"
116
+ alert_command: str = ""
117
+ scoped_tests: tuple[tuple[str, str], ...] = ()
118
+ mutation_command: str = "" # the suite run once per mutant; nonzero exit = killed
119
+ mutation_timeout_seconds: int = 300 # a mutant that loops forever counts as killed
120
+ mutation_workers: int = 1 # >1 runs mutants in that many detached git worktrees
121
+ diff_uncovered_max: int | None = None # verify exit 9 past this many dead changed lines
122
+ debt_max_age_months: int | None = None # ratchet report --enforce flags older marks
123
+ repayment_min_per_30d: int | None = None # --enforce flags a stalled burn-down
124
+ max_parallel_lanes: int = 1 # lanes running at once; 1 = strictly serial
125
+ analysis_workers: int = 0 # lizard pool size; 0 = one worker per core
126
+ # Operational traps the repo learned the hard way. They lived as TOML
127
+ # comments, which the parser drops, so no payload could ever quote them.
128
+ notes: tuple[str, ...] = ()
129
+ # Only the scopes that wrote one, so a scope with nothing to say costs a
130
+ # payload no key. A plain dict, because these end up in --json output.
131
+ scope_notes: dict[str, tuple[str, ...]] = {}
132
+
133
+
134
+ def load_config_text(text: str) -> Config:
135
+ try:
136
+ raw = tomllib.loads(text)
137
+ except tomllib.TOMLDecodeError as exc:
138
+ raise ConfigError(f"crapkit.toml does not parse: {exc}") from exc
139
+ try:
140
+ return _build_config(raw)
141
+ except KeyError as exc:
142
+ raise ConfigError(f"crapkit.toml is missing a required key: {exc}") from exc
143
+
144
+
145
+ def _parse_scope(row: dict) -> Scope:
146
+ languages = tuple(row.get("languages", ()))
147
+ unknown = set(languages) - SUPPORTED_LANGUAGES
148
+ if unknown:
149
+ raise ConfigError(f"unsupported language(s) {sorted(unknown)} in scope {row.get('name')!r}")
150
+ scope_target = row.get("target")
151
+ if scope_target is not None and (not isinstance(scope_target, int) or scope_target < 1):
152
+ raise ConfigError(f"scope {row.get('name')!r}: target must be a positive int, got {scope_target!r}")
153
+ return Scope(name=row["name"], paths=tuple(row["paths"]), languages=languages,
154
+ target=scope_target,
155
+ coverage_optional=bool(row.get("coverage_optional", False)))
156
+
157
+
158
+ def _notes(row: dict, where: str) -> tuple[str, ...]:
159
+ """The `notes` list, rejected unless every entry is a string.
160
+
161
+ A bare `notes = "..."` is the trap TOML sets: it is iterable, so it would
162
+ load as one note per letter and every reader would print them that way.
163
+ """
164
+ raw = row.get("notes", [])
165
+ if not isinstance(raw, list) or not all(isinstance(item, str) for item in raw):
166
+ raise ConfigError(f"{where}: notes must be a list of strings, got {raw!r}")
167
+ return tuple(raw)
168
+
169
+
170
+ def _parse_scopes(rows) -> tuple[tuple[Scope, ...], dict[str, tuple[str, ...]]]:
171
+ """Every [[scope]] row and its notes, off ONE walk of the rows.
172
+
173
+ Notes hang on the same rows the scopes come from, so collecting them in a
174
+ second pass would re-read and re-validate every row for nothing — and, when
175
+ the rows arrive as an iterator, would find none of them.
176
+ """
177
+ scopes: list[Scope] = []
178
+ notes: dict[str, tuple[str, ...]] = {}
179
+ for row in rows:
180
+ scope = _parse_scope(row)
181
+ scopes.append(scope)
182
+ row_notes = _notes(row, f"scope {scope.name!r}")
183
+ if row_notes:
184
+ notes[scope.name] = row_notes
185
+ return tuple(scopes), notes
186
+
187
+
188
+ def _validate_lane_command(parser: str, full_suite: bool, name: str, command: str) -> None:
189
+ if parser == "istanbul":
190
+ _validate_istanbul_command(name, command)
191
+ if parser == "coveragepy" and full_suite:
192
+ _validate_coveragepy_command(name, command)
193
+
194
+
195
+ def _parse_lane(row: dict, scope_names: set) -> Lane:
196
+ parser = row["parser"]
197
+ if parser not in SUPPORTED_PARSERS:
198
+ raise ConfigError(f"lane {row.get('name')!r}: unsupported parser {parser!r}")
199
+ lane_scopes = tuple(row.get("scopes", ()))
200
+ unknown_scopes = set(lane_scopes) - scope_names
201
+ if unknown_scopes:
202
+ raise ConfigError(f"lane {row.get('name')!r} references undeclared scope(s) {sorted(unknown_scopes)}")
203
+ full_suite = bool(row.get("full_suite", True))
204
+ _validate_lane_command(parser, full_suite, row.get("name", "?"), row["command"])
205
+ return Lane(name=row["name"], command=row["command"], artifact=row["artifact"],
206
+ parser=parser, scopes=lane_scopes,
207
+ cwd=row.get("cwd", ""), path_prefix=row.get("path_prefix", ""),
208
+ env=tuple(sorted((str(k), str(v)) for k, v in row.get("env", {}).items())),
209
+ full_suite=full_suite, container_ok=bool(row.get("container_ok", False)),
210
+ results_artifact=row.get("results_artifact", ""),
211
+ timeout_seconds=_nonneg_int(row, "timeout_seconds"),
212
+ retries=_nonneg_int(row, "retries"),
213
+ retest_command=row.get("retest_command", ""))
214
+
215
+
216
+ def _nonneg_int(row: dict, key: str) -> int:
217
+ value = row.get(key, 0)
218
+ if not isinstance(value, int) or isinstance(value, bool) or value < 0:
219
+ raise ConfigError(
220
+ f"lane {row.get('name', '?')!r}: {key} must be a non-negative int, got {value!r}")
221
+ return value
222
+
223
+
224
+ def _reject_shared_artifacts(lanes: list) -> None:
225
+ seen_artifacts: dict[str, str] = {}
226
+ for lane in lanes:
227
+ for artifact in filter(None, (lane.artifact, lane.results_artifact)):
228
+ if artifact in seen_artifacts and seen_artifacts[artifact] != lane.name:
229
+ raise ConfigError(
230
+ f"lanes {seen_artifacts[artifact]!r} and {lane.name!r} share the artifact path "
231
+ f"{artifact!r}; reused paths cross-attribute coverage under --reuse-artifacts")
232
+ seen_artifacts[artifact] = lane.name
233
+
234
+
235
+ def _build_config(raw: dict) -> Config:
236
+ scope_rows = raw.get("scope", [])
237
+ if not scope_rows:
238
+ raise ConfigError("crapkit.toml declares no [[scope]] — nothing to analyze")
239
+ scopes, scope_notes = _parse_scopes(scope_rows)
240
+ scope_names = {s.name for s in scopes}
241
+ lanes = [_parse_lane(row, scope_names) for row in raw.get("lane", [])]
242
+ _reject_shared_artifacts(lanes)
243
+ main = raw.get("crapkit", {})
244
+ return Config(
245
+ target=int(main.get("target", DEFAULT_TARGET)),
246
+ scopes=scopes,
247
+ exclude_globs=tuple(raw.get("exclude", {}).get("globs", ())),
248
+ max_file_bytes=_optional_int(raw.get("exclude", {}), "max_file_bytes"),
249
+ churn_window_months=int(main.get("churn_window_months", 12)),
250
+ worklist_floor=int(main.get("worklist_floor", 5)),
251
+ worklist_top=int(main.get("worklist_top", 50)),
252
+ lanes=tuple(lanes),
253
+ ratchet_file=main.get("ratchet_file", "crapkit-ratchet.tsv"),
254
+ alert_command=main.get("alert_command", ""),
255
+ scoped_tests=tuple(sorted((str(k), str(v)) for k, v in main.get("scoped_tests", {}).items())),
256
+ mutation_command=main.get("mutation_command", ""),
257
+ mutation_timeout_seconds=int(main.get("mutation_timeout_seconds", 300)),
258
+ mutation_workers=_positive_int(main, "mutation_workers", 1),
259
+ diff_uncovered_max=_optional_int(main, "diff_uncovered_max"),
260
+ debt_max_age_months=_optional_int(main, "debt_max_age_months"),
261
+ repayment_min_per_30d=_optional_int(main, "repayment_min_per_30d"),
262
+ max_parallel_lanes=_bounded_int(main, "max_parallel_lanes", default=1, minimum=1),
263
+ analysis_workers=_bounded_int(main, "analysis_workers", default=0, minimum=0),
264
+ notes=_notes(main, "[crapkit]"),
265
+ scope_notes=scope_notes,
266
+ )
267
+
268
+
269
+ def _bounded_int(main: dict, key: str, *, default: int, minimum: int) -> int:
270
+ value = main.get(key, default)
271
+ if not isinstance(value, int) or isinstance(value, bool) or value < minimum:
272
+ raise ConfigError(f"{key} must be an int >= {minimum}, got {value!r}")
273
+ return value
274
+
275
+
276
+ def _positive_int(main: dict, key: str, default: int) -> int:
277
+ value = main.get(key, default)
278
+ if not isinstance(value, int) or isinstance(value, bool) or value < 1:
279
+ raise ConfigError(f"{key} must be a positive int, got {value!r}")
280
+ return value
281
+
282
+
283
+ def _optional_int(main: dict, key: str) -> int | None:
284
+ value = main.get(key)
285
+ if value is None:
286
+ return None
287
+ if not isinstance(value, int) or isinstance(value, bool) or value < 0:
288
+ raise ConfigError(f"{key} must be a non-negative int, got {value!r}")
289
+ return value
crapkit/coupling.py ADDED
@@ -0,0 +1,89 @@
1
+ """Change coupling from the churn git log. Pure.
2
+
3
+ Files that keep landing in the same commits are coupled, whatever the import
4
+ graph says — the hidden dependency the compiler cannot show. Bulk commits are
5
+ skipped for pairing: a 40-file sweep says nothing about any particular pair.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from collections.abc import Iterable, Iterator
10
+ from itertools import combinations
11
+
12
+ MAX_COMMIT_FILES = 30
13
+
14
+
15
+ def _commit_file_sets(lines: Iterable[str]) -> Iterator[set[str]]:
16
+ """One file set per commit, yielded as the log streams past.
17
+
18
+ A %x01 line opens a commit and every non-blank line until the next one is a
19
+ path. Empty sets are yielded rather than skipped: they add no file counts
20
+ and form no pairs, so the caller cannot tell them from a skip.
21
+ """
22
+ files: set[str] = set()
23
+ past_header = False
24
+ for raw in lines:
25
+ line = raw.strip()
26
+ if line.startswith("\x01"):
27
+ yield files
28
+ files, past_header = set(), True
29
+ continue
30
+ if past_header and line:
31
+ files.add(line.replace("\\", "/"))
32
+ past_header = True # a log starting mid-commit opens on a severed header
33
+ yield files
34
+
35
+
36
+ def _rank_pairs(file_counts: dict, pair_counts: dict, min_support: int,
37
+ min_confidence: float, top: int | None) -> list[dict]:
38
+ out = []
39
+ for (a, b), support in pair_counts.items():
40
+ if support < min_support:
41
+ continue
42
+ confidence = max(support / file_counts[a], support / file_counts[b])
43
+ if confidence < min_confidence:
44
+ continue
45
+ out.append({"files": [a, b], "support": support, "confidence": round(confidence, 4)})
46
+ out.sort(key=lambda p: (-p["support"] * p["confidence"], p["files"]))
47
+ return out if top is None else out[:top]
48
+
49
+
50
+ def _partner(pair: dict, path: str) -> dict:
51
+ a, b = pair["files"]
52
+ return {"path": b if a == path else a, "support": pair["support"],
53
+ "confidence": pair["confidence"]}
54
+
55
+
56
+ def partners(lines: Iterable[str], path: str, *, min_support: int = 5,
57
+ min_confidence: float = 0.5, top: int = 5) -> list[dict]:
58
+ """One file's coupled partners, best first.
59
+
60
+ Ranked over EVERY qualifying pair before the cut: a global top applied first
61
+ would drop a quiet file's own partners behind the repo's noisiest pairs and
62
+ report it as uncoupled.
63
+ """
64
+ ranked = change_coupling_lines(lines, min_support=min_support,
65
+ min_confidence=min_confidence, top=None)
66
+ return [_partner(p, path) for p in ranked if path in p["files"]][:top]
67
+
68
+
69
+ def change_coupling(log_text: str, *, min_support: int = 5, min_confidence: float = 0.5,
70
+ top: int | None = 50) -> list[dict]:
71
+ """Whole-text entrypoint: the log already in hand."""
72
+ return change_coupling_lines(log_text.splitlines(), min_support=min_support,
73
+ min_confidence=min_confidence, top=top)
74
+
75
+
76
+ def change_coupling_lines(lines: Iterable[str], *, min_support: int = 5,
77
+ min_confidence: float = 0.5, top: int | None = 50) -> list[dict]:
78
+ """Streaming entrypoint: the log is consumed once, one commit at a time."""
79
+ commits = _commit_file_sets(lines)
80
+ file_counts: dict[str, int] = {}
81
+ pair_counts: dict[tuple[str, str], int] = {}
82
+ for files in commits:
83
+ for f in files:
84
+ file_counts[f] = file_counts.get(f, 0) + 1
85
+ if len(files) > MAX_COMMIT_FILES:
86
+ continue
87
+ for pair in combinations(sorted(files), 2):
88
+ pair_counts[pair] = pair_counts.get(pair, 0) + 1
89
+ return _rank_pairs(file_counts, pair_counts, min_support, min_confidence, top)
@@ -0,0 +1,225 @@
1
+ """Istanbul coverage-final.json parser. Pure: JSON text in, per-file function coverage out.
2
+
3
+ Branch hits map into function spans by line containment. A function with no
4
+ branches inside its span falls back to STATEMENT coverage in that span, and
5
+ only with no statements either to invocation (hit or not) — a straight-line
6
+ function half-executed must not read as fully covered. Written for the
7
+ AST-remapped output of @vitest/coverage-v8 >= 3.2, which is istanbul-schema-identical.
8
+
9
+ The artifact is a flat {abs_path: coverage} object and only one file's coverage
10
+ is ever needed at a time, so the outer object is SPLIT rather than parsed:
11
+ split_top_level walks the members and decodes each value on its own. A
12
+ whole-document json.loads on a 157 MB artifact peaked at 1,432 MB against
13
+ 335 MB for the split, for the same output.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import heapq
18
+ import json
19
+ import re
20
+ from typing import Iterator, NamedTuple
21
+
22
+ from .errors import ToolError
23
+
24
+
25
+ class FnCoverage(NamedTuple):
26
+ name: str
27
+ start: int
28
+ end: int
29
+ invoked: bool
30
+ branches_total: int
31
+ branches_covered: int
32
+ statements_total: int = 0
33
+ statements_covered: int = 0
34
+
35
+ @property
36
+ def coverage(self) -> float:
37
+ if self.branches_total > 0:
38
+ return self.branches_covered / self.branches_total
39
+ if self.statements_total > 0:
40
+ return self.statements_covered / self.statements_total
41
+ return 1.0 if self.invoked else 0.0
42
+
43
+
44
+ def _rel_path(abs_path: str, repo_root: str) -> str:
45
+ norm = abs_path.replace("\\", "/")
46
+ root = repo_root.replace("\\", "/").rstrip("/") + "/"
47
+ return norm[len(root):] if norm.startswith(root) else norm
48
+
49
+
50
+ # --- top-level splitter ----------------------------------------------------
51
+ # One member's opening: whitespace, the separating comma, the key string, the
52
+ # colon. The key pattern is string-aware — an escaped quote inside a Windows
53
+ # path must not end it — and its alternation is unambiguous (a character is
54
+ # either not a quote/backslash, or an escape pair), so it scans linearly and
55
+ # cannot backtrack.
56
+ _MEMBER_RE = re.compile(r'\s*,?\s*("(?:[^"\\]|\\.)*")\s*:\s*', re.DOTALL)
57
+ _OPEN_RE = re.compile(r"\s*\{")
58
+ _CLOSE_RE = re.compile(r"\s*\}\s*\Z")
59
+
60
+ # raw_decode reads ONE value at a position and reports where it ended, which is
61
+ # what makes the split possible without writing a second parser. Hand-rolled
62
+ # brace matching finds the same boundary but pays a Python loop iteration per
63
+ # structural character: 4.75s of a 5.4s parse on a 21.6 MB artifact, against
64
+ # 0.23s here, because this scanner is the C one.
65
+ _DECODER = json.JSONDecoder()
66
+
67
+
68
+ def split_top_level(text: str) -> Iterator[tuple[str, object]]:
69
+ """Yield (key, value) per member of the outer object, decoding one value at
70
+ a time. Equivalent to json.loads(text).items() except that the whole
71
+ document is never a live dict: each value is dropped as the caller steps
72
+ past it, so peak memory tracks the LARGEST file, not their sum."""
73
+ opening = _OPEN_RE.match(text)
74
+ if opening is None:
75
+ raise ValueError("istanbul artifact is not a JSON object")
76
+ i = opening.end()
77
+ while True:
78
+ member = _MEMBER_RE.match(text, i)
79
+ if member is None:
80
+ _expect_end(text, i)
81
+ return
82
+ value, i = _DECODER.raw_decode(text, member.end())
83
+ yield json.loads(member.group(1)), value
84
+
85
+
86
+ def _expect_end(text: str, i: int) -> None:
87
+ if _CLOSE_RE.match(text, i) is None:
88
+ raise ValueError(f"unexpected content at offset {i}")
89
+
90
+
91
+ def _iter_files(text: str, repo_root: str) -> Iterator[tuple[str, dict]]:
92
+ """(repo-relative path, coverage object) per artifact member. One file's
93
+ coverage is live at a time; the previous one is unreferenced on the next step."""
94
+ for abs_path, cov in split_top_level(text):
95
+ yield _rel_path(abs_path, repo_root), cov
96
+
97
+
98
+ # --- span attribution ------------------------------------------------------
99
+ # mutable span layout while attributing: [name, start, end, invoked, b_total, b_cov, s_total, s_cov]
100
+ _B_TOTAL, _B_COV, _S_TOTAL, _S_COV = 4, 5, 6, 7
101
+
102
+
103
+ def _fn_spans(cov: dict) -> list[list]:
104
+ spans = []
105
+ for fid, fn in cov.get("fnMap", {}).items():
106
+ start = fn["decl"]["start"]["line"]
107
+ end = fn.get("loc", {}).get("end", {}).get("line") or start
108
+ invoked = cov.get("f", {}).get(fid, 0) > 0
109
+ spans.append([fn.get("name") or "(anonymous)", start, end, invoked, 0, 0, 0, 0])
110
+ spans.sort(key=lambda s: s[1])
111
+ return spans
112
+
113
+
114
+ def _branch_line(branch: dict) -> int | None:
115
+ return branch.get("loc", {}).get("start", {}).get("line")
116
+
117
+
118
+ def _stmt_line(stmt: dict) -> int | None:
119
+ return stmt.get("start", {}).get("line")
120
+
121
+
122
+ def _query_lines(cov: dict) -> set[int]:
123
+ """Every line the attribution will ask about, branches and statements both."""
124
+ lines = {_branch_line(b) for b in cov.get("branchMap", {}).values()}
125
+ lines |= {_stmt_line(s) for s in cov.get("statementMap", {}).values()}
126
+ lines.discard(None)
127
+ return lines
128
+
129
+
130
+ def _push_started(heap: list, ordered: list[list], nxt: int, line: int) -> int:
131
+ while nxt < len(ordered) and ordered[nxt][1] <= line:
132
+ span = ordered[nxt]
133
+ heapq.heappush(heap, (span[2] - span[1], -span[1], nxt, span))
134
+ nxt += 1
135
+ return nxt
136
+
137
+
138
+ def _drop_ended(heap: list, line: int) -> None:
139
+ """Discard spans that closed before this line. Safe to do lazily and only at
140
+ the top: query lines only increase, so anything popped here can never
141
+ contain a later line either."""
142
+ while heap and heap[0][3][2] < line:
143
+ heapq.heappop(heap)
144
+
145
+
146
+ def _span_owners(fn_spans: list[list], lines: set[int]) -> dict[int, list | None]:
147
+ """line -> innermost containing span. A hit inside a nested function belongs
148
+ to that function, never to its encloser — else the nested one reads through
149
+ its encloser and the encloser answers for lines it can't fix.
150
+
151
+ Sweeping spans by start into a heap keyed (length, -start, index) settles
152
+ that in O((F + Q) log F) instead of a scan per query. The index term is
153
+ load-bearing: it is the sorted position, so an exact tie on (length, -start)
154
+ resolves to the span the old linear scan met first."""
155
+ ordered = sorted(fn_spans, key=lambda s: s[1])
156
+ heap: list[tuple] = []
157
+ owners: dict[int, list | None] = {}
158
+ nxt = 0
159
+ for line in sorted(lines):
160
+ nxt = _push_started(heap, ordered, nxt, line)
161
+ _drop_ended(heap, line)
162
+ owners[line] = heap[0][3] if heap else None
163
+ return owners
164
+
165
+
166
+ def _attach_branches(owners: dict[int, list | None], cov: dict) -> None:
167
+ hits_by_id = cov.get("b", {})
168
+ for bid, branch in cov.get("branchMap", {}).items():
169
+ best = owners.get(_branch_line(branch))
170
+ if best is not None:
171
+ hits = hits_by_id.get(bid, [])
172
+ best[_B_TOTAL] += len(hits)
173
+ best[_B_COV] += sum(1 for h in hits if h > 0)
174
+
175
+
176
+ def _attach_statements(owners: dict[int, list | None], cov: dict) -> None:
177
+ hits_by_id = cov.get("s", {})
178
+ for sid, stmt in cov.get("statementMap", {}).items():
179
+ best = owners.get(_stmt_line(stmt))
180
+ if best is not None:
181
+ best[_S_TOTAL] += 1
182
+ best[_S_COV] += 1 if hits_by_id.get(sid, 0) > 0 else 0
183
+
184
+
185
+ def _file_coverage(cov: dict) -> list[FnCoverage]:
186
+ fn_spans = _fn_spans(cov)
187
+ owners = _span_owners(fn_spans, _query_lines(cov))
188
+ _attach_branches(owners, cov)
189
+ _attach_statements(owners, cov)
190
+ return [FnCoverage(*s) for s in fn_spans]
191
+
192
+
193
+ def _dead_lines(cov: dict) -> set[int]:
194
+ hits_by_id = cov.get("s", {})
195
+ dead = {_stmt_line(stmt)
196
+ for sid, stmt in cov.get("statementMap", {}).items()
197
+ if hits_by_id.get(sid, 0) == 0}
198
+ dead.discard(None)
199
+ return dead
200
+
201
+
202
+ def parse_istanbul_missing(text: str, *, repo_root: str) -> dict[str, set[int]]:
203
+ """Per measured file, the lines whose statement never ran — the
204
+ diff-coverage ground truth. Files with everything executed map to set()."""
205
+ try:
206
+ missing: dict[str, set[int]] = {}
207
+ for rel_path, cov in _iter_files(text, repo_root):
208
+ missing[rel_path] = _dead_lines(cov)
209
+ return missing
210
+ except Exception as exc:
211
+ raise ToolError(f"unparseable istanbul artifact: {exc}") from exc
212
+
213
+
214
+ def parse_istanbul(text: str, *, repo_root: str) -> dict[str, list[FnCoverage]]:
215
+ try:
216
+ per_file: dict[str, list[FnCoverage]] = {}
217
+ for rel_path, cov in _iter_files(text, repo_root):
218
+ per_file[rel_path] = _file_coverage(cov)
219
+ if not per_file:
220
+ raise ToolError("istanbul artifact is empty (zero files) — the coverage run measured nothing")
221
+ return per_file
222
+ except ToolError:
223
+ raise
224
+ except Exception as exc:
225
+ raise ToolError(f"unparseable istanbul artifact: {exc}") from exc