crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/scaffold.py ADDED
@@ -0,0 +1,361 @@
1
+ """Repo sniffing for `crapkit init`: tracked files in, a starter crapkit.toml out. Pure."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ from typing import NamedTuple
6
+
7
+ from .universe import LANGUAGE_EXTENSIONS, exclude_matcher, excluded
8
+
9
+ _EXT_LANGUAGE = {ext: lang for lang, exts in LANGUAGE_EXTENSIONS.items() for ext in exts}
10
+
11
+ DEFAULT_EXCLUDES = (
12
+ "**/node_modules/**", "**/dist/**", "**/build/**", "**/vendor/**",
13
+ "**/*.test.*", "**/*.spec.*", "**/test_*.py", "**/*_test.py", "**/conftest.py",
14
+ # runner config files the docs themselves tell users to create; globs are
15
+ # whole-path, so the root form and the nested form are both required
16
+ "*.config.ts", "*.config.js", "*.config.mts",
17
+ "**/*.config.ts", "**/*.config.js", "**/*.config.mts",
18
+ )
19
+
20
+ _MATCH_DEFAULT_EXCLUDE = exclude_matcher(DEFAULT_EXCLUDES)
21
+
22
+
23
+ def _language_of(path: str) -> str | None:
24
+ for ext, lang in _EXT_LANGUAGE.items():
25
+ if path.endswith(ext):
26
+ return lang
27
+ return None
28
+
29
+
30
+ def _scoped_source(path: str) -> str | None:
31
+ """The top-level dir when this is a scopeable source file, else None.
32
+ Root-level loose files and dot-dirs make bad scopes; the excludes drop
33
+ generated trees and tests the same way inventory later will."""
34
+ top, _, rest = path.partition("/")
35
+ if not rest or top.startswith(".") or excluded(path, _MATCH_DEFAULT_EXCLUDE):
36
+ return None
37
+ return top if _language_of(path) else None
38
+
39
+
40
+ def source_candidates(files: list[str]) -> list[str]:
41
+ """The paths init would put in a scope. Counting them over the UNTRACKED set
42
+ is how init tells "you are in the wrong directory" from "you never ran
43
+ `git add`" — crapkit reads `git ls-files` and sees neither case any other way.
44
+ """
45
+ paths = (raw.replace("\\", "/") for raw in files)
46
+ return [p for p in paths if _scoped_source(p)]
47
+
48
+
49
+ def sniff_scopes(files: list[str]) -> dict[str, tuple[str, ...]]:
50
+ """Top-level source dirs -> their languages, sorted both ways for stable output."""
51
+ langs: dict[str, set[str]] = {}
52
+ for raw in files:
53
+ path = raw.replace("\\", "/")
54
+ top = _scoped_source(path)
55
+ if top is None:
56
+ continue
57
+ langs.setdefault(top, set()).add(_language_of(path))
58
+ return {d: tuple(sorted(v)) for d, v in sorted(langs.items())}
59
+
60
+
61
+ class LaneSpec(NamedTuple):
62
+ """A lane init can write live, plus the scope languages it measures."""
63
+ name: str
64
+ command: str
65
+ artifact: str
66
+ parser: str
67
+ languages: tuple[str, ...]
68
+
69
+
70
+ PYTEST_MARKERS = ("pyproject.toml", "pytest.ini", "setup.cfg")
71
+
72
+ _PY_LANGUAGES = ("python",)
73
+ _JS_LANGUAGES = ("javascript", "tsx", "typescript")
74
+ _JS_RUNNER_COMMAND = {"jest": "npx jest --coverage", "vitest": "npx vitest run --coverage"}
75
+
76
+ # Every artifact a scaffolded lane writes lands under .crapkit/, which init
77
+ # ignores in the same breath. A 14-lane repo that let each runner default grew
78
+ # fifteen coverage-* directories and seven junit files at its root, one per
79
+ # lane, and nothing in the tree said which lane owned which.
80
+ _COV_DIR = ".crapkit/cov"
81
+ _PY_ARTIFACT = f"{_COV_DIR}/py.json"
82
+ _JS_COV_DIR = f"{_COV_DIR}/js"
83
+ _JS_ARTIFACT = f"{_JS_COV_DIR}/coverage-final.json"
84
+ _JS_DEFAULT_ARTIFACT = "coverage/coverage-final.json"
85
+
86
+ # Each runner spells "write the coverage report here" its own way and rejects
87
+ # the other's spelling outright.
88
+ _JS_REPORTS_DIR_FLAG = {"jest": "--coverageDirectory=",
89
+ "vitest": "--coverage.reportsDirectory="}
90
+
91
+
92
+ def _pytest_lane(markers: frozenset[str], interpreter: str) -> LaneSpec | None:
93
+ if not markers.intersection(PYTEST_MARKERS):
94
+ return None
95
+ return LaneSpec("py", f"{interpreter} -m pytest --cov --cov-branch "
96
+ f"--cov-report=json:{_PY_ARTIFACT}",
97
+ _PY_ARTIFACT, "coveragepy", _PY_LANGUAGES)
98
+
99
+
100
+ def _npm_test_script(scripts: dict) -> str | None:
101
+ """The script name npm would run tests with: "test" when it exists, else the
102
+ alphabetically first name starting with "test", so the choice never moves."""
103
+ if "test" in scripts:
104
+ return "test"
105
+ named = sorted(name for name in scripts if name.startswith("test"))
106
+ return named[0] if named else None
107
+
108
+
109
+ def _js_runner_command(dev_dependencies: dict) -> str | None:
110
+ for runner in sorted(_JS_RUNNER_COMMAND):
111
+ if runner in dev_dependencies:
112
+ return _JS_RUNNER_COMMAND[runner]
113
+ return None
114
+
115
+
116
+ def _load_json(text: str) -> dict:
117
+ try:
118
+ data = json.loads(text)
119
+ except ValueError:
120
+ return {}
121
+ return data if isinstance(data, dict) else {}
122
+
123
+
124
+ def _js_routing(dev_dependencies: dict) -> str:
125
+ """The flag that moves this runner's coverage directory under .crapkit/, or
126
+ "" when package.json names neither runner or both.
127
+
128
+ `npm test` can run anything, so devDependencies is the only thing that names
129
+ the runner — and naming it wrong is worse than the litter: jest exits on
130
+ vitest's `--coverage.reportsDirectory` and vitest exits on jest's
131
+ `--coverageDirectory`. Unresolved, the lane keeps the directory its runner
132
+ already defaults to and doctor says so.
133
+ """
134
+ named = [runner for runner in sorted(_JS_REPORTS_DIR_FLAG) if runner in dev_dependencies]
135
+ return f" {_JS_REPORTS_DIR_FLAG[named[0]]}{_JS_COV_DIR}" if len(named) == 1 else ""
136
+
137
+
138
+ def _js_lane(package_json: str) -> LaneSpec | None:
139
+ """A test script, or vitest/jest in devDependencies: either says the repo
140
+ already knows how to produce istanbul coverage."""
141
+ package = _load_json(package_json)
142
+ script = _npm_test_script(package.get("scripts", {}))
143
+ command = (f"npm run {script} -- --coverage" if script
144
+ else _js_runner_command(package.get("devDependencies", {})))
145
+ if command is None:
146
+ return None
147
+ routing = _js_routing(package.get("devDependencies", {}))
148
+ artifact = _JS_ARTIFACT if routing else _JS_DEFAULT_ARTIFACT
149
+ return LaneSpec("js", command + routing, artifact, "istanbul", _JS_LANGUAGES)
150
+
151
+
152
+ def detect_lanes(markers: frozenset[str], package_json: str, *,
153
+ interpreter: str = "python") -> tuple[LaneSpec, ...]:
154
+ """The lanes this repo can already run, decided from files alone.
155
+
156
+ Nothing is executed and nothing is imported: presence of a pytest marker
157
+ file, and what package.json says about itself, are the whole signal.
158
+ """
159
+ found = (_pytest_lane(markers, interpreter), _js_lane(package_json))
160
+ return tuple(lane for lane in found if lane)
161
+
162
+
163
+ def _quoted(names) -> str:
164
+ return ", ".join(f'"{name}"' for name in names)
165
+
166
+
167
+ def _scope_stanza(name: str, languages: tuple[str, ...]) -> list[str]:
168
+ return ["[[scope]]", f'name = "{name}"', f'paths = ["{name}"]',
169
+ f"languages = [{_quoted(languages)}]", ""]
170
+
171
+
172
+ def _exclude_stanza() -> list[str]:
173
+ return ["[exclude]", f"globs = [{_quoted(DEFAULT_EXCLUDES)}]", ""]
174
+
175
+
176
+ def _lane_stanza(lane: LaneSpec, scope_names: tuple[str, ...]) -> list[str]:
177
+ return ["[[lane]]", f'name = "{lane.name}"', f'command = "{lane.command}"',
178
+ f'artifact = "{lane.artifact}"', f'parser = "{lane.parser}"',
179
+ f"scopes = [{_quoted(scope_names)}]", ""]
180
+
181
+
182
+ def _scopes_for(lane: LaneSpec, scopes: dict[str, tuple[str, ...]]) -> tuple[str, ...]:
183
+ return tuple(name for name, languages in scopes.items()
184
+ if set(languages).intersection(lane.languages))
185
+
186
+
187
+ def live_lanes(lanes: tuple[LaneSpec, ...],
188
+ scopes: dict[str, tuple[str, ...]]) -> tuple[LaneSpec, ...]:
189
+ """The detected lanes init actually writes: the ones with a scope to measure.
190
+
191
+ A lane with an empty scopes list measures nothing, so it goes back to being a
192
+ template rather than papering over the gap — and it leaves no artifact behind
193
+ either, which is what the .gitignore side of init needs to know.
194
+ """
195
+ return tuple(lane for lane in lanes if _scopes_for(lane, scopes))
196
+
197
+
198
+ def _live_lanes(lanes: tuple[LaneSpec, ...],
199
+ scopes: dict[str, tuple[str, ...]]) -> tuple[list[str], set[str]]:
200
+ """Stanzas for the lanes init writes, and the parsers they cover."""
201
+ lines: list[str] = []
202
+ covered = set()
203
+ for lane in live_lanes(lanes, scopes):
204
+ lines += _lane_stanza(lane, _scopes_for(lane, scopes))
205
+ covered.add(lane.parser)
206
+ return lines, covered
207
+
208
+
209
+ _TEMPLATES = {
210
+ "coveragepy": ("# [[lane]]", '# name = "py"',
211
+ '# command = "python -m pytest --cov --cov-branch '
212
+ f'--cov-report=json:{_PY_ARTIFACT}"',
213
+ f'# artifact = "{_PY_ARTIFACT}"', '# parser = "coveragepy"'),
214
+ "istanbul": ("# [[lane]]", '# name = "js"',
215
+ '# command = "npx vitest run --coverage '
216
+ f'{_JS_REPORTS_DIR_FLAG["vitest"]}{_JS_COV_DIR}"',
217
+ f'# artifact = "{_JS_ARTIFACT}"', '# parser = "istanbul"'),
218
+ }
219
+
220
+
221
+ # `crapkit test-scoped FILES` runs one of these per scope. A scope with no
222
+ # template exits 3, so the stub is written for every scope init found; the value
223
+ # is the runner that scope's language usually uses, and the whole block stays
224
+ # commented because only the repo knows whether that command is the right one.
225
+ _SCOPED_TEST_COMMANDS = {
226
+ "python": "python -m pytest {files} -q -p no:cacheprovider",
227
+ "javascript": "npx vitest run {files}",
228
+ "tsx": "npx vitest run {files}",
229
+ "typescript": "npx vitest run {files}",
230
+ }
231
+ _SCOPED_TEST_PLACEHOLDER = "<your test command> {files}"
232
+
233
+
234
+ def _scoped_test_command(languages: tuple[str, ...]) -> str:
235
+ for language in languages:
236
+ if language in _SCOPED_TEST_COMMANDS:
237
+ return _SCOPED_TEST_COMMANDS[language]
238
+ return _SCOPED_TEST_PLACEHOLDER
239
+
240
+
241
+ def _runner_confirmed(languages: tuple[str, ...], confirmed: frozenset[str]) -> bool:
242
+ return any(lang in confirmed and lang in _SCOPED_TEST_COMMANDS for lang in languages)
243
+
244
+
245
+ def _confirmed_languages(lanes: tuple[LaneSpec, ...]) -> frozenset[str]:
246
+ """Languages whose scoped command a detected lane already proves.
247
+
248
+ Only pytest: the presence signal that wrote the py coverage lane makes
249
+ `python -m pytest {files}` known-good. The js runners stay unconfirmed on
250
+ purpose — which vitest or jest config a file-scoped run needs is exactly
251
+ what presence detection cannot see."""
252
+ return frozenset({"python"} if any(" -m pytest " in ln.command for ln in lanes) else ())
253
+
254
+
255
+ def _scoped_entry_lines(scopes: dict[str, tuple[str, ...]], live: bool) -> list[str]:
256
+ prefix = "" if live else "# "
257
+ return [f'{prefix}{name} = "{_scoped_test_command(languages)}"'
258
+ for name, languages in scopes.items()]
259
+
260
+
261
+ def _scoped_tests_stub(scopes: dict[str, tuple[str, ...]],
262
+ confirmed: frozenset[str] = frozenset()) -> list[str]:
263
+ """The [crapkit.scoped_tests] block: live entries for scopes whose runner a
264
+ detected lane proves, commented templates for the rest.
265
+
266
+ A confirmed runner written commented would hand doctor a warning about a
267
+ gap init could have closed. Every commented line still uncomments as
268
+ written: a stub a reader has to rewrite before it parses is no better than
269
+ the nothing that used to be here.
270
+ """
271
+ live = {n: l for n, l in scopes.items() if _runner_confirmed(l, confirmed)}
272
+ rest = {n: l for n, l in scopes.items() if n not in live}
273
+ intro = ["# `crapkit test-scoped FILES` runs one command per scope, with {files}",
274
+ "# replaced by that scope's files, each quoted."]
275
+ return intro + _live_block(live) + _commented_block(rest, bool(live)) + [""]
276
+
277
+
278
+ def _live_block(live: dict[str, tuple[str, ...]]) -> list[str]:
279
+ if not live:
280
+ return []
281
+ return ["[crapkit.scoped_tests]"] + _scoped_entry_lines(live, True)
282
+
283
+
284
+ def _commented_block(rest: dict[str, tuple[str, ...]], has_live: bool) -> list[str]:
285
+ if not rest:
286
+ return []
287
+ header = ["# Uncomment what fits:"] + ([] if has_live else ["# [crapkit.scoped_tests]"])
288
+ return header + _scoped_entry_lines(rest, False)
289
+
290
+
291
+ def _template_lines(covered: set[str], scope_list: str) -> list[str]:
292
+ # the template's scope is a placeholder on purpose: writing a real scope
293
+ # name pointed a TS lane template at a python project's sources
294
+ del scope_list
295
+ lines: list[str] = []
296
+ for parser in sorted(_TEMPLATES):
297
+ if parser not in covered:
298
+ lines += [*_TEMPLATES[parser], '# scopes = ["<your-scope>"]', ""]
299
+ if not lines:
300
+ return []
301
+ return ["# Declare one [[lane]] per coverage command, then run `crapkit coverage`.", *lines]
302
+
303
+
304
+ def starter_toml(scopes: dict[str, tuple[str, ...]], lanes: tuple[LaneSpec, ...] = ()) -> str:
305
+ lines = ["[crapkit]", "target = 6", ""]
306
+ for name, languages in scopes.items():
307
+ lines += _scope_stanza(name, languages)
308
+ lines += _exclude_stanza()
309
+ live, covered = _live_lanes(lanes, scopes)
310
+ return "\n".join(lines + live + _template_lines(covered, _quoted(scopes))
311
+ + _scoped_tests_stub(scopes, _confirmed_languages(lanes)))
312
+
313
+
314
+ _STORE_IGNORE = ".crapkit/"
315
+
316
+
317
+ def _artifact_ignore(artifact: str) -> str:
318
+ """What to ignore for one lane's artifact. A file inside a directory ignores
319
+ the DIRECTORY: istanbul writes a whole coverage/ tree beside
320
+ coverage-final.json, and ignoring the one file leaves the rest untracked."""
321
+ top, sep, _ = artifact.partition("/")
322
+ return f"{top}/" if sep else top
323
+
324
+
325
+ def _runner_droppings(lane: LaneSpec) -> list[str]:
326
+ """What the lane's runner leaves beside the artifact; a pytest lane drops
327
+ coverage's data file and bytecode caches into the consumer's tree."""
328
+ return [".coverage", "__pycache__/"] if lane.parser == "coveragepy" else []
329
+
330
+
331
+ def gitignore_entries(lanes: tuple[LaneSpec, ...]) -> list[str]:
332
+ """Everything adopting crapkit will drop in the consumer's tree: its own
333
+ store, the artifact of each lane init wrote, and the runner's droppings.
334
+ Order is stable and duplicates collapse, so two lanes sharing a directory
335
+ ignore it once."""
336
+ entries = [_STORE_IGNORE, *(entry for lane in lanes
337
+ for entry in (_artifact_ignore(lane.artifact),
338
+ *_runner_droppings(lane)))]
339
+ return list(dict.fromkeys(entries))
340
+
341
+
342
+ def _appended(current: str, entries: list[str]) -> str:
343
+ """The new entries under their own heading, after whatever was already there."""
344
+ block = "# crapkit\n" + "".join(f"{entry}\n" for entry in entries)
345
+ if not current:
346
+ return block
347
+ separator = "\n" if current.endswith("\n") else "\n\n"
348
+ return current + separator + block
349
+
350
+
351
+ def gitignore_update(current: str, lanes: tuple[LaneSpec, ...]) -> tuple[str, list[str]]:
352
+ """The .gitignore this repo needs, and the entries it gained. Pure.
353
+
354
+ Idempotent: an entry the file already carries is never written twice, so a
355
+ repo that adopted crapkit by hand gets no duplicate lines.
356
+ """
357
+ present = {line.strip() for line in current.splitlines()}
358
+ added = [entry for entry in gitignore_entries(lanes) if entry not in present]
359
+ if not added:
360
+ return current, []
361
+ return _appended(current, added), added
crapkit/score.py ADDED
@@ -0,0 +1,255 @@
1
+ """CRAP scoring and the coverage join. Pure.
2
+
3
+ The inventory is the master list: every function gets a scored row. Coverage
4
+ joins by path plus span overlap. Flags never conflate: measured (a lane's
5
+ artifact spoke about the file), untested (lane covers the scope, artifact
6
+ silent on this function), no-lane (no lane covers the scope at all), cc-only
7
+ (the scope declares coverage_optional, so no coverage number can exist).
8
+ The three zero-coverage flags all score cov=0; the flag says whether the
9
+ missing number is a testing gap, a tooling gap, or by design.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from collections.abc import Iterator
14
+ from typing import NamedTuple
15
+
16
+ from .coverage_istanbul import FnCoverage
17
+ from .snapshot import InventoryRow
18
+
19
+
20
+ def crap(ccn: int, cov: float) -> float:
21
+ return ccn * ccn * (1.0 - cov) ** 3 + ccn
22
+
23
+
24
+ _GRADES = ((0.02, "A"), (0.05, "B"), (0.10, "C"), (0.20, "D"))
25
+
26
+
27
+ def grade(over_target: int, total: int) -> str:
28
+ """One letter for over-target density; A+ is reserved for zero debt."""
29
+ if over_target == 0:
30
+ return "A+"
31
+ ratio = over_target / total
32
+ for bound, letter in _GRADES:
33
+ if ratio < bound:
34
+ return letter
35
+ return "F"
36
+
37
+
38
+ class ScoredRow(NamedTuple):
39
+ scope: str
40
+ path: str
41
+ long_name: str
42
+ start: int
43
+ end: int
44
+ ccn_std: int
45
+ ccn_mod: int
46
+ ccn: int
47
+ nloc: int
48
+ params: int
49
+ nesting: int
50
+ cov: float
51
+ flag: str
52
+ crap: float
53
+ remedy: str
54
+ cognitive: int = 0 # Sonar-spec cognitive complexity; reporting only, never gated
55
+
56
+
57
+ _FIELD_TYPES = (str, str, str, int, int, int, int, int, int, int, int,
58
+ float, str, float, str, int)
59
+ _SCORED_HEADER = "\t".join(ScoredRow._fields)
60
+ # One %s per field, built once. %s of every field is str() of it, so the bytes
61
+ # do not move; a row is a tuple, so it IS the argument list and the per-row
62
+ # generator, the join and the concatenation all disappear (179 -> 116 ms on
63
+ # 140,922 rows). Derived from _fields rather than written out, so a new column
64
+ # cannot leave the template a field short.
65
+ _SCORED_ROW = "\t".join(["%s"] * len(ScoredRow._fields)) + "\n"
66
+
67
+
68
+ def scored_tsv_lines(rows: list[ScoredRow]) -> Iterator[str]:
69
+ """Header then one newline-terminated line per row. With no rows the header
70
+ is the empty string, so the file stays the single newline it always was."""
71
+ yield (_SCORED_HEADER if rows else "") + "\n"
72
+ for r in rows:
73
+ yield _SCORED_ROW % r
74
+
75
+
76
+ def parse_scored_row(line: str) -> ScoredRow:
77
+ """One exported line back to a row. str() of every field round-trips through
78
+ its own constructor, floats included, so the re-emitted bytes are identical."""
79
+ parts = line.split("\t")
80
+ if len(parts) != len(_FIELD_TYPES):
81
+ raise ValueError(f"scored row has {len(parts)} fields, expected {len(_FIELD_TYPES)}: {line!r}")
82
+ return ScoredRow(*[cast(part) for cast, part in zip(_FIELD_TYPES, parts)])
83
+
84
+
85
+ def parse_scored_tsv(text: str) -> list[ScoredRow]:
86
+ return [parse_scored_row(line) for line in text.splitlines()
87
+ if line.strip() and line != _SCORED_HEADER]
88
+
89
+
90
+ def _overlap(a_start: int, a_end: int, b_start: int, b_end: int) -> int:
91
+ return max(0, min(a_end, b_end) - max(a_start, b_start) + 1)
92
+
93
+
94
+ def _best_match(row: InventoryRow, candidates: list[FnCoverage]) -> FnCoverage | None:
95
+ # Exact start beats raw overlap, and on remaining ties the tightest span wins:
96
+ # a nested function must join its own entry, never its enclosing function's
97
+ # (an enclosing match would inherit the parent's coverage and understate risk).
98
+ # Identical-span twins (two lanes measuring the same file) keep the BETTER
99
+ # measurement — the true union of branch hits is at least the max, and lane
100
+ # declaration order must never change a score.
101
+ best, best_key = None, None
102
+ for fn in candidates:
103
+ o = _overlap(row.start, row.end, fn.start, fn.end)
104
+ if o <= 0:
105
+ continue
106
+ key = (fn.start == row.start, o, -(fn.end - fn.start), fn.coverage)
107
+ if best_key is None or key > best_key:
108
+ best, best_key = fn, key
109
+ return best
110
+
111
+
112
+ def _remedy(ccn: int, score: float, ceiling: int) -> str:
113
+ if ccn > ceiling:
114
+ return "decompose"
115
+ return "ok" if score <= ceiling else "add-tests"
116
+
117
+
118
+ def _finish(row, cov: float, flag: str, *, target: int, scope_targets) -> ScoredRow:
119
+ # cc-only is the pre-commit hook's rule: crap IS ccn, so _remedy can only
120
+ # answer ok or decompose. Feeding it cov=0 through the formula would say
121
+ # add-tests about code no test can reach.
122
+ score = float(row.ccn) if flag == "cc-only" else crap(row.ccn, cov)
123
+ ceiling = scope_targets.get(row.scope, target) if scope_targets else target
124
+ # Positional, and NOT *row: cognitive is last in both tuples with four
125
+ # fields between, so splicing the row in whole lands it in cov. Building
126
+ # this row is a third of the join's cost at 140,922 rows — **row._asdict()
127
+ # built a throwaway dict per row and looked every field up by name.
128
+ return ScoredRow(row[0], row[1], row[2], row[3], row[4], row[5], row[6], row[7],
129
+ row[8], row[9], row[10],
130
+ cov, flag, score, _remedy(row[7], score, ceiling), row[11])
131
+
132
+
133
+ def _named_overlay_cov(row, by_key: dict) -> tuple[float, str]:
134
+ # by_key is grouped on (path, long_name) up front: filtering the whole
135
+ # path's candidates per row made this O(rows x candidates) per file
136
+ # (measured 2,998,554 comparisons on a whole-repo rescore). The nearest-start
137
+ # tie-break is unchanged, and min() still keeps the FIRST nearest twin
138
+ # because the group holds the baseline's own order.
139
+ named = by_key.get((row.path, row.long_name))
140
+ if named:
141
+ return min(named, key=lambda c: abs(c.start - row.start)).cov, "measured"
142
+ return 0.0, "untested"
143
+
144
+
145
+ def _cov_without_join(row, lane_scopes: set, cc_only_scopes) -> tuple[float, str] | None:
146
+ """The verdict for a row no coverage artifact can speak about, else None.
147
+
148
+ coverage_optional is checked FIRST: such a scope needs no lane, so the
149
+ no-lane fallback would otherwise hide it behind a tooling gap it does
150
+ not have.
151
+ """
152
+ if row.scope in cc_only_scopes:
153
+ return 0.0, "cc-only"
154
+ if row.scope not in lane_scopes:
155
+ return 0.0, "no-lane"
156
+ return None
157
+
158
+
159
+ def _start_index(coverage_by_path: dict) -> dict[str, dict[int, list[FnCoverage]]]:
160
+ """path -> {start line: the candidates declaring it}, in candidate order.
161
+
162
+ 91.5% of joinable rows share a start line with some candidate, and the scan
163
+ that found it compared every candidate on the path: 936,818 pairwise
164
+ comparisons on the consumer repo. Bucket order is the candidates' own order, which
165
+ is what keeps a dead-even tie resolving to the same twin.
166
+ """
167
+ index = {}
168
+ for path, candidates in coverage_by_path.items():
169
+ buckets: dict[int, list[FnCoverage]] = {}
170
+ for fn in candidates:
171
+ buckets.setdefault(fn.start, []).append(fn)
172
+ index[path] = buckets
173
+ return index
174
+
175
+
176
+ def _best_exact(row, bucket) -> FnCoverage | None:
177
+ """The winner among candidates whose start EQUALS the row's, or None when
178
+ none of them overlaps it.
179
+
180
+ _best_match's key leads with (fn.start == row.start): True here and False
181
+ for every candidate outside this bucket, so a winner here is the winner
182
+ over the whole path. The remaining terms are that key's tail, and
183
+ max(a_start, b_start) is row.start by construction.
184
+ """
185
+ best, best_key = None, None
186
+ for fn in bucket:
187
+ o = min(row.end, fn.end) - row.start + 1
188
+ if o <= 0:
189
+ continue
190
+ key = (o, -(fn.end - fn.start), fn.coverage)
191
+ if best_key is None or key > best_key:
192
+ best, best_key = fn, key
193
+ return best
194
+
195
+
196
+ def _span_join_cov(row, coverage_by_path: dict, start_index: dict) -> tuple[float, str]:
197
+ candidates = coverage_by_path.get(row.path)
198
+ if candidates is None:
199
+ return 0.0, "untested"
200
+ # The bucket answers for most rows; the scan is the fallback for a row that
201
+ # starts where no candidate does, or whose bucket overlaps it nowhere.
202
+ match = _best_exact(row, start_index[row.path].get(row.start, ()))
203
+ if match is None:
204
+ match = _best_match(row, candidates)
205
+ if match is None:
206
+ return 0.0, "untested"
207
+ return match.coverage, "measured"
208
+
209
+
210
+ def overlay_stale_coverage(
211
+ rows: list[InventoryRow],
212
+ baseline_scored: list["ScoredRow"],
213
+ *,
214
+ lane_scopes: set[str],
215
+ target: int = 6,
216
+ scope_targets: dict[str, int] | None = None,
217
+ cc_only_scopes: frozenset[str] = frozenset(),
218
+ ) -> list[ScoredRow]:
219
+ """Rescore fresh complexity against a BASELINE run's coverage.
220
+
221
+ Joins by function NAME only (edits shift spans, names survive), nearest
222
+ start among same-name twins. A renamed or new function joins NOTHING —
223
+ a span join here would hand it a neighbour's stale number and mislead
224
+ the preview. Coverage values are the baseline's; the caller labels them
225
+ stale.
226
+ """
227
+ by_key: dict[tuple[str, str], list[ScoredRow]] = {}
228
+ for r in baseline_scored:
229
+ if r.flag == "measured":
230
+ by_key.setdefault((r.path, r.long_name), []).append(r)
231
+
232
+ scored = []
233
+ for row in rows:
234
+ cov, flag = (_cov_without_join(row, lane_scopes, cc_only_scopes)
235
+ or _named_overlay_cov(row, by_key))
236
+ scored.append(_finish(row, cov, flag, target=target, scope_targets=scope_targets))
237
+ return scored
238
+
239
+
240
+ def score_rows(
241
+ rows: list[InventoryRow],
242
+ coverage_by_path: dict[str, list[FnCoverage]],
243
+ *,
244
+ lane_scopes: set[str],
245
+ target: int = 6,
246
+ scope_targets: dict[str, int] | None = None,
247
+ cc_only_scopes: frozenset[str] = frozenset(),
248
+ ) -> list[ScoredRow]:
249
+ start_index = _start_index(coverage_by_path)
250
+ scored = []
251
+ for r in rows:
252
+ cov, flag = (_cov_without_join(r, lane_scopes, cc_only_scopes)
253
+ or _span_join_cov(r, coverage_by_path, start_index))
254
+ scored.append(_finish(r, cov, flag, target=target, scope_targets=scope_targets))
255
+ return scored
crapkit/snapshot.py ADDED
@@ -0,0 +1,51 @@
1
+ """Canonical inventory rows and exports. Pure.
2
+
3
+ Scored rows carry no timestamps and sort totally, so identical inputs produce
4
+ byte-identical exports; run metadata lives on the run row in the store, never here.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from collections.abc import Iterable, Iterator
9
+ from typing import NamedTuple
10
+
11
+ from .merge import FunctionRecord
12
+
13
+
14
+ class InventoryRow(NamedTuple):
15
+ scope: str
16
+ path: str
17
+ long_name: str
18
+ start: int
19
+ end: int
20
+ ccn_std: int
21
+ ccn_mod: int
22
+ ccn: int
23
+ nloc: int
24
+ params: int
25
+ nesting: int
26
+ cognitive: int = 0 # Sonar-spec cognitive complexity; reporting only, never gated
27
+
28
+
29
+ def build_inventory_rows(by_scope: dict[str, list[FunctionRecord]]) -> list[InventoryRow]:
30
+ rows = [
31
+ InventoryRow(scope, r.path, r.long_name, r.start, r.end,
32
+ r.ccn_std, r.ccn_mod, r.ccn, r.nloc, r.params, r.nesting,
33
+ r.cognitive)
34
+ for scope, records in by_scope.items()
35
+ for r in records
36
+ ]
37
+ rows.sort(key=lambda r: (r.scope, r.path, r.start, r.end, r.long_name))
38
+ return rows
39
+
40
+
41
+ def tsv_lines(rows: Iterable[InventoryRow]) -> Iterator[str]:
42
+ """The export document, one newline-terminated line at a time.
43
+
44
+ A generator, not a joined string: at 140k functions the list of lines plus
45
+ the joined document cost 44 MiB of transient copies of rows that already
46
+ exist. The caller writes these straight to a file opened with
47
+ newline="\\n", which is where the byte-identical guarantee is kept.
48
+ """
49
+ yield "\t".join(InventoryRow._fields) + "\n"
50
+ for r in rows:
51
+ yield "\t".join(str(v) for v in r) + "\n"