crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/mutate_pool.py ADDED
@@ -0,0 +1,152 @@
1
+ """Where mutants actually run. One worker (the default) mutates the live
2
+ working tree, exactly as this command always has. `mutation_workers = N` gives
3
+ each worker its own detached git worktree instead, because a mutant is a write
4
+ to a source file: two of them in one tree would read each other's edits.
5
+
6
+ Two things the parallel path owes the serial one. Results merge by the mutant's
7
+ position in the list, never by who finished first, so the same tree reports the
8
+ same JSON at any worker count. And the worktree is a checkout of HEAD while
9
+ `mutate` is diff-scoped against the working tree, so the targeted files are
10
+ copied in as they are on disk — uncommitted lines are the ones being mutated.
11
+
12
+ The command runs with cwd set to the worktree. A consumer whose test command
13
+ resolves the code under test from somewhere else (an editable install pointing
14
+ at the main checkout, a global site-packages copy) would measure unmutated
15
+ code and score every mutant a survivor: keep the command cwd-relative.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import os
20
+ import shutil
21
+ import subprocess
22
+ import tempfile
23
+ import threading
24
+ from concurrent.futures import ThreadPoolExecutor
25
+ from contextlib import contextmanager
26
+ from pathlib import Path
27
+
28
+ from .gitio import worktree_add, worktree_remove
29
+ from .mutate import apply_mutant
30
+
31
+
32
+ def run_one(tree: Path, cfg, mutant) -> bool:
33
+ """True = killed. The original file ALWAYS comes back, whatever happens."""
34
+ p = tree / mutant.path
35
+ original = p.read_bytes()
36
+ # Python validates .pyc files by source SIZE + mtime in WHOLE SECONDS: two
37
+ # same-size mutants applied within one second would reuse the first one's
38
+ # stale bytecode and read as false survivors. Kill the cache, write none.
39
+ shutil.rmtree(p.parent / "__pycache__", ignore_errors=True)
40
+ env = {**os.environ, "PYTHONDONTWRITEBYTECODE": "1"}
41
+ try:
42
+ p.write_text(apply_mutant(original.decode("utf-8", "replace"), mutant),
43
+ encoding="utf-8", newline="")
44
+ proc = subprocess.run(cfg.mutation_command, shell=True, cwd=tree, capture_output=True,
45
+ env=env, timeout=cfg.mutation_timeout_seconds)
46
+ return proc.returncode != 0
47
+ except subprocess.TimeoutExpired:
48
+ return True # a mutant that loops forever is dead
49
+ finally:
50
+ p.write_bytes(original)
51
+
52
+
53
+ def _shards(indexed: list, workers: int) -> list[list]:
54
+ """Round-robin: worker w takes mutants w, w+K, w+2K. Even to within one
55
+ mutant however the list is shaped, and each shard keeps list order."""
56
+ return [indexed[w::workers] for w in range(workers)]
57
+
58
+
59
+ def _merge(done: list) -> list[bool]:
60
+ """(index, killed) pairs from every worker back into mutant order."""
61
+ return [killed for _, killed in sorted(done)]
62
+
63
+
64
+ def _run_shard(tree: Path, cfg, shard: list, report) -> list:
65
+ out = []
66
+ for index, mutant in shard:
67
+ killed = run_one(tree, cfg, mutant)
68
+ report(index, mutant, killed)
69
+ out.append((index, killed))
70
+ return out
71
+
72
+
73
+ def _seed(root: Path, tree: Path, rel_paths: list) -> None:
74
+ """The worktree checked out HEAD; the mutants were grown from the working
75
+ tree. Copy the targeted files over so both agree on what line 40 is."""
76
+ for rel in rel_paths:
77
+ dst = tree / rel
78
+ dst.parent.mkdir(parents=True, exist_ok=True)
79
+ shutil.copyfile(root / rel, dst)
80
+
81
+
82
+ def _on_every_tree(action, root: Path, trees: list) -> None:
83
+ """`action(root, tree)` on one thread per tree, and WAIT FOR THEM ALL, even
84
+ once one has raised. Leaving a checkout in flight is what turns a failed add
85
+ into a leaked worktree: the cleanup walks the list, finds nothing at that
86
+ path yet, and the abandoned thread creates it a moment later."""
87
+ with ThreadPoolExecutor(max_workers=len(trees)) as pool:
88
+ for done in [pool.submit(action, root, tree) for tree in trees]:
89
+ done.result()
90
+
91
+
92
+ @contextmanager
93
+ def _worktrees(root: Path, count: int):
94
+ """N detached checkouts, all gone on the way out — after a clean run, after
95
+ an exception, and after an add that failed halfway through the set.
96
+
97
+ The adds run together, and so do the removes. Each one is a full checkout
98
+ that spends its time waiting on the disk rather than on a core, so four of
99
+ them serialized cost 54.5 s on a 31,620-file repo against 25.1 s overlapped.
100
+ """
101
+ base = Path(tempfile.mkdtemp(prefix="crapkit-mutate-"))
102
+ trees = [base / f"w{i}" for i in range(count)]
103
+ try:
104
+ _on_every_tree(worktree_add, root, trees)
105
+ yield trees
106
+ finally:
107
+ _teardown(root, base, trees)
108
+
109
+
110
+ def _teardown(root: Path, base: Path, trees: list) -> None:
111
+ _on_every_tree(worktree_remove, root, trees)
112
+ shutil.rmtree(base, ignore_errors=True)
113
+
114
+
115
+ def _fan_out(cfg, trees: list, shards: list, report) -> list:
116
+ with ThreadPoolExecutor(max_workers=len(trees)) as pool:
117
+ futures = [pool.submit(_run_shard, tree, cfg, shard, report)
118
+ for tree, shard in zip(trees, shards)]
119
+ return [pair for f in futures for pair in f.result()]
120
+
121
+
122
+ def _run_parallel(root: Path, cfg, mutants: list, workers: int, report) -> list[bool]:
123
+ shards = _shards(list(enumerate(mutants)), workers)
124
+ targets = sorted({m.path for m in mutants})
125
+ with _worktrees(root, workers) as trees:
126
+ for tree in trees:
127
+ _seed(root, tree, targets)
128
+ done = _fan_out(cfg, trees, shards, report)
129
+ return _merge(done)
130
+
131
+
132
+ def run_mutants(root: Path, cfg, mutants: list, report) -> list[bool]:
133
+ """One killed flag per mutant, in mutant order. Workers past the mutant
134
+ count would only pay for empty worktrees."""
135
+ workers = min(cfg.mutation_workers, len(mutants))
136
+ if workers > 1:
137
+ return _run_parallel(root, cfg, mutants, workers, report)
138
+ return _merge(_run_shard(root, cfg, list(enumerate(mutants)), report))
139
+
140
+
141
+ def reporter(total: int, stream):
142
+ """Progress lines, one per finished mutant. Workers race, so the write is
143
+ locked: a half-written line read as a survivor list is worse than no line."""
144
+ lock = threading.Lock()
145
+
146
+ def report(index: int, mutant, killed: bool) -> None:
147
+ verdict = "killed" if killed else "SURVIVED"
148
+ with lock:
149
+ print(f" mutant {index + 1}/{total} {mutant.path}:{mutant.line} "
150
+ f"[{mutant.op}] {verdict}", file=stream)
151
+
152
+ return report
crapkit/override.py ADDED
@@ -0,0 +1,94 @@
1
+ """The audited override: three records or nothing.
2
+
3
+ An exemption exists only if all three surfaces carry it: the alert (a human
4
+ channel sees one line), the committed ratchet (the debt is diff-visible), and
5
+ the snapshot store (the run remembers). The alert fires first because it is the
6
+ step most likely to fail; a partial override fails loudly and grants nothing.
7
+ No environment-variable or silent bypass exists anywhere in crapkit.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import subprocess
12
+ from pathlib import Path
13
+
14
+ from .errors import ConfigError, ToolError
15
+ from .ratchet import RatchetEntry, dump_ratchet, load_ratchet
16
+ from .store import SnapshotStore
17
+ from .verify import GateViolation
18
+
19
+
20
+ def record_override(
21
+ *,
22
+ store: SnapshotStore,
23
+ run_id: int,
24
+ root: Path,
25
+ ratchet_file: str,
26
+ alert_command: str,
27
+ violations: list[GateViolation],
28
+ reason: str,
29
+ raise_marks: bool = True,
30
+ ) -> None:
31
+ _require_auditable_override(reason, alert_command)
32
+ _alert_or_refuse(alert_command, root, violations, reason)
33
+
34
+ # Audit before grant: the snapshot record lands BEFORE the ratchet write,
35
+ # because the ratchet entry is the functional exemption. A failure between
36
+ # the two leaves an audit trail with no grant, never a grant with no trail.
37
+ store.write_overrides(run_id, [(v.path, v.long_name, v.crap, reason) for v in violations])
38
+
39
+ _grant_ratchet_debt(root / ratchet_file, violations, raise_marks=raise_marks)
40
+
41
+
42
+ def _require_auditable_override(reason: str, alert_command: str) -> None:
43
+ """Refuse an override that could not be audited even if every step succeeded."""
44
+ if not reason.strip():
45
+ raise ConfigError("an override requires a non-empty reason")
46
+ if not alert_command.strip():
47
+ raise ConfigError(
48
+ "no alert_command configured — the override requires a visible alert line; "
49
+ "set [crapkit] alert_command in crapkit.toml")
50
+
51
+
52
+ def _alert_or_refuse(alert_command: str, root: Path, violations: list[GateViolation],
53
+ reason: str) -> None:
54
+ """Put the debt in front of a human first; a silent alert grants nothing."""
55
+ summary = "; ".join(f"{v.path}:{v.start} {v.long_name} crap={v.crap:.1f}" for v in violations)
56
+ line = f"crapkit OVERRIDE ({reason}): {summary}"
57
+ # The line reaches the alert command on stdin, never interpolated into the
58
+ # shell string: function names come from analyzed source and are not shell-safe.
59
+ proc = subprocess.run(alert_command, shell=True, cwd=root, input=line + "\n",
60
+ capture_output=True, text=True, encoding="utf-8", errors="replace")
61
+ if proc.returncode != 0:
62
+ raise ToolError(
63
+ f"override alert command failed (exit {proc.returncode}): "
64
+ f"{(proc.stderr or proc.stdout).strip()[-300:]} — no alert, no override")
65
+
66
+
67
+ def _grant_ratchet_debt(ratchet_path: Path, violations: list[GateViolation], *,
68
+ raise_marks: bool) -> None:
69
+ """The functional exemption: the debt enters the committed ratchet, diff-visible."""
70
+ by_key = _marks_by_key(ratchet_path)
71
+ for v in violations:
72
+ mark = _override_mark(by_key.get((v.path, v.long_name)), v.crap, raise_marks=raise_marks)
73
+ by_key[(v.path, v.long_name)] = RatchetEntry(v.path, v.long_name, round(mark, 4))
74
+ ratchet_path.write_text(dump_ratchet(list(by_key.values())), encoding="utf-8", newline="\n")
75
+
76
+
77
+ def _marks_by_key(ratchet_path: Path) -> dict[tuple[str, str], RatchetEntry]:
78
+ """Prior marks by (path, long_name); an absent ratchet file is simply no marks."""
79
+ existing = load_ratchet(ratchet_path.read_text(encoding="utf-8")) if ratchet_path.is_file() else []
80
+ return {(e.path, e.long_name): e for e in existing}
81
+
82
+
83
+ def _override_mark(prior: RatchetEntry | None, crap: float, *, raise_marks: bool) -> float:
84
+ """The mark this override records for one function."""
85
+ # raise_marks=False is the hook path: it synthesizes worst-case crap (no
86
+ # coverage data), and letting that raise a measured mark would blind the
87
+ # ratchet to a later real coverage collapse. The prior tighter mark stays,
88
+ # so the NEXT verify still demands repayment; the override only lets this
89
+ # one commit through.
90
+ if prior is None:
91
+ return crap
92
+ if raise_marks:
93
+ return max(prior.crap, crap)
94
+ return prior.crap
crapkit/packet.py ADDED
@@ -0,0 +1,343 @@
1
+ """The start-editing packet: everything a session needs before it opens the file.
2
+
3
+ `brief` answered what one function scores. A session then read the file to find
4
+ the other functions in it, guessed which ceiling the gate would apply, hunted for
5
+ the lane that measures the scope, and re-derived the commands to run. Each of
6
+ those is a value some caller already holds, so each is a field here instead of a
7
+ round trip.
8
+
9
+ Every function in this module is pure: values in, a dict or a list out. The
10
+ reads that feed them — the store, git, the config, the file texts — belong to
11
+ the caller, which is what lets one batch of packets pay for them once. Nothing
12
+ here removes or retypes a field `brief --json` already published; the packet is
13
+ what was added around it.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ from .ratchet_report import DAY
18
+
19
+ # What the gate actually enforces, said once. A session that reads a ceiling of
20
+ # 6 beside a standing mark of 72 otherwise reads a contradiction and either
21
+ # refuses to start or "fixes" debt nobody asked it to touch.
22
+ GATE_BINDS = ("changed functions only; a ratchet mark pardons standing debt "
23
+ "at or under it")
24
+
25
+ _OPENERS = "([{<"
26
+ _CLOSERS = ")]}>"
27
+
28
+ # What lizard calls a function it could not name. Every anonymous function in a
29
+ # file prints the same string, which is why the handle below exists.
30
+ ANONYMOUS = "(anonymous)"
31
+
32
+ # `stale` clears when a run lands on the current commit and never before. The
33
+ # packet used to answer its own staleness warning with another `brief`, which
34
+ # re-reads the snapshot that is already stale. `--reuse-unchanged` reruns only
35
+ # the lanes whose scope files moved and parses the rest off the artifacts they
36
+ # already have, so it is the cheapest call that still writes a run.
37
+ REFRESH = "python -m crapkit coverage --reuse-unchanged"
38
+
39
+
40
+ def function_source(text: str | None, start: int, end: int) -> str | None:
41
+ """One function's lines out of the file text the caller already read.
42
+
43
+ None means nobody read the file, which is not the same as a function whose
44
+ span holds no lines.
45
+ """
46
+ if text is None:
47
+ return None
48
+ return "\n".join(text.splitlines()[start - 1:end])
49
+
50
+
51
+ def file_functions(rows) -> list[dict]:
52
+ """Every scored row in the file, not just the one the brief is about.
53
+
54
+ A decomposition lands in the neighbours: the helper it extracts into, the
55
+ twin beside it, the row that is already at its ceiling and must stay there.
56
+ """
57
+ return [{"function": r.long_name, "start": r.start, "end": r.end, "ccn": r.ccn,
58
+ "crap": r.crap, "remedy": r.remedy} for r in rows]
59
+
60
+
61
+ def file_totals(rows, scope_targets: dict, target: int) -> dict:
62
+ """The file's own numbers, each row judged against ITS scope's ceiling.
63
+
64
+ A file can hold rows from two scopes; scoring the whole file against one
65
+ ceiling would report debt a per-scope target deliberately allows.
66
+ """
67
+ over = sum(1 for r in rows if r.crap > scope_targets.get(r.scope, target))
68
+ return {"functions": len(rows), "over_target": over,
69
+ "crap_load": round(sum(r.crap for r in rows), 2)}
70
+
71
+
72
+ def gate_rule(*, ceiling: int, mark: float | None, mark_age_days: int | None,
73
+ diff_uncovered_max: int | None) -> dict:
74
+ """The rule this function will be judged by, spelled out rather than implied."""
75
+ return {"ceiling": ceiling, "binds": GATE_BINDS, "ratchet_mark": mark,
76
+ "mark_age_days": mark_age_days, "diff_uncovered_max": diff_uncovered_max}
77
+
78
+
79
+ def mark_age_days(events: list[tuple], key: tuple) -> int | None:
80
+ """How long this function's mark has stood, in the ratchet history's own time.
81
+
82
+ Anchored on the newest commit in the history, never the wall clock, so a
83
+ fixed history reports the same age forever. A mark that was repaid and later
84
+ re-added is aged from its return: the debt is the one standing now.
85
+ """
86
+ entered = None
87
+ anchor = 0
88
+ for ts, event_key, kind, _ in events:
89
+ anchor = max(anchor, ts)
90
+ if event_key == key:
91
+ entered = ts if kind == "added" else None
92
+ return None if entered is None else (anchor - entered) // DAY
93
+
94
+
95
+ def lane_for(scope: str | None, lanes):
96
+ """The first lane claiming this scope, or None when no lane measures it."""
97
+ if scope is None:
98
+ return None
99
+ return next((lane for lane in lanes if scope in lane.scopes), None)
100
+
101
+
102
+ def lane_record(lane) -> dict | None:
103
+ """The lane verbatim: what ran, where, and how long it is allowed to take.
104
+
105
+ A session that reruns the lane by hand needs the cwd and the env as declared;
106
+ reconstructing them from the command string is how the reruns drift.
107
+ """
108
+ if lane is None:
109
+ return None
110
+ return {"name": lane.name, "command": lane.command, "artifact": lane.artifact,
111
+ "parser": lane.parser, "cwd": lane.cwd, "env": dict(lane.env),
112
+ "timeout_seconds": lane.timeout_seconds}
113
+
114
+
115
+ def commands(path: str, scoped: str | None, note: str = "") -> dict:
116
+ """The four commands a session runs next, with the paths already filled in.
117
+
118
+ `refresh_writes_run` rides beside `refresh` because the other three change
119
+ nothing on disk and that one does: a read-only session, or one holding a
120
+ tree it is not allowed to score, has to know which of the four it may run.
121
+ """
122
+ out = {"gate": f"python -m crapkit rescore {path} --gate",
123
+ "scoped_tests": scoped,
124
+ "verify": "python -m crapkit verify",
125
+ "refresh": REFRESH,
126
+ "refresh_writes_run": True}
127
+ if scoped is None and note:
128
+ out["scoped_tests_note"] = note
129
+ return out
130
+
131
+
132
+ def bare_name(long_name: str) -> str:
133
+ """The identifier a long_name opens with, before its parameter list.
134
+
135
+ Empty for a function lizard could not name: both `(anonymous)` and
136
+ `(anonymous) ( z )` open with the parenthesis, so an empty prefix IS the
137
+ test for anonymity, with no second string to keep in step.
138
+ """
139
+ return long_name.split("(")[0].strip()
140
+
141
+
142
+ def anonymous_starts(rows) -> list[int]:
143
+ """Where the file's anonymous functions open, in file order."""
144
+ return sorted(r.start for r in rows if not bare_name(r.long_name))
145
+
146
+
147
+ def handles(rows) -> dict[int, str]:
148
+ """The handle for every row in one file, keyed by the line it opens on.
149
+
150
+ A named function is its own handle. An anonymous one is `(anonymous)#N`,
151
+ counted over the file's anonymous functions in start order — a position, not
152
+ a line, so the string a session copies out of a packet still names the same
153
+ function after an edit above it moves every line below.
154
+
155
+ Keyed by start because no two functions in a file open on the same line,
156
+ which makes it the one per-file key a row already carries.
157
+ """
158
+ ordinals = {start: n for n, start in enumerate(anonymous_starts(rows), 1)}
159
+ return {r.start: bare_name(r.long_name) or f"{ANONYMOUS}#{ordinals[r.start]}"
160
+ for r in rows}
161
+
162
+
163
+ def handle_names(rows) -> list[str]:
164
+ """Every anonymous handle this file offers, in order.
165
+
166
+ What an out-of-range ordinal is reported against: a session that guessed #5
167
+ needs the two that exist, the same way a wrong bare name gets the file's
168
+ real names back.
169
+ """
170
+ return [f"{ANONYMOUS}#{n}" for n in range(1, len(anonymous_starts(rows)) + 1)]
171
+
172
+
173
+ def handle_ordinal(name: str) -> int | None:
174
+ """The N in `(anonymous)#N`, or None when `name` is some other name form.
175
+
176
+ None rather than an error: this is the question "is that string a handle",
177
+ asked before the other name forms get their turn.
178
+ """
179
+ head, sep, tail = name.partition("#")
180
+ if not sep or head.strip() != ANONYMOUS or not tail.isdigit():
181
+ return None
182
+ return int(tail)
183
+
184
+
185
+ def budget(row, ceiling: int) -> dict:
186
+ """What the work costs: pieces a decomposition needs, decision paths no test
187
+ walks.
188
+
189
+ One definition for both readers. `next-item` published these and `brief` did
190
+ not, so a session that opened on a packet re-derived numbers the queue had
191
+ already computed — and two derivations of one formula drift with nothing to
192
+ catch it.
193
+ """
194
+ return {"est_splits": 0 if row.ccn <= ceiling else -(-row.ccn // ceiling),
195
+ "est_uncovered_paths": max(0, round((1 - row.cov) * row.ccn))}
196
+
197
+
198
+ def regrowth(history: list[dict]) -> dict:
199
+ """Whether this function's complexity fell and then came back.
200
+
201
+ A function somebody already decomposed once, back over its ceiling, is a
202
+ different job from one that has always been big: the decomposition that was
203
+ tried is on record and did not hold.
204
+ """
205
+ return {"regrown": _fell_then_rose([h["ccn"] for h in history]),
206
+ "history": [[h["run_id"], h["ccn"]] for h in history]}
207
+
208
+
209
+ def _fell_then_rose(ccns: list[int]) -> bool:
210
+ """True once a drop is followed anywhere later by a climb."""
211
+ fell = False
212
+ for before, after in zip(ccns, ccns[1:]):
213
+ if fell and after > before:
214
+ return True
215
+ fell = fell or after < before
216
+ return False
217
+
218
+
219
+ def params(long_name: str) -> list[dict]:
220
+ """The parameter list out of lizard's long_name, name first.
221
+
222
+ lizard prints the signature it parsed: `f( a , b = 1 , c : int = 2 )` in
223
+ Python, `dispatch ( a , b Record , c )` in TypeScript. The name leads in
224
+ both; whatever follows it is the type annotation as lizard printed it.
225
+ Anything this cannot read is an empty list, never a guess.
226
+ """
227
+ inner = _param_text(long_name)
228
+ if inner is None:
229
+ return []
230
+ return [_one_param(part) for part in _split_top(inner) if part]
231
+
232
+
233
+ def _param_text(long_name: str) -> str | None:
234
+ """What sits inside the LAST balanced parentheses, or None when there are none.
235
+
236
+ Not the first `(`: lizard names an anonymous function `(anonymous) ( z )`,
237
+ where the first one belongs to the name and the parameter list is the group
238
+ that closes the string.
239
+ """
240
+ closed = long_name.rfind(")")
241
+ opened = _matching_open(long_name, closed)
242
+ return None if opened is None else long_name[opened + 1:closed]
243
+
244
+
245
+ def _matching_open(text: str, closed: int) -> int | None:
246
+ """The index of the `(` that opens the group closing at `closed`."""
247
+ depth = 0
248
+ for i in range(closed, -1, -1):
249
+ depth += (text[i] == ")") - (text[i] == "(")
250
+ if depth == 0 and text[i] == "(":
251
+ return i
252
+ return None
253
+
254
+
255
+ def _split_top(text: str) -> list[str]:
256
+ """Split on commas that are not inside brackets, so `Map<a , b>` stays one."""
257
+ parts = []
258
+ depth = 0
259
+ start = 0
260
+ for i, ch in enumerate(text):
261
+ if ch in _OPENERS:
262
+ depth += 1
263
+ elif ch in _CLOSERS:
264
+ depth -= 1
265
+ elif ch == "," and depth == 0:
266
+ parts.append(text[start:i].strip())
267
+ start = i + 1
268
+ parts.append(text[start:].strip())
269
+ return parts
270
+
271
+
272
+ def _one_param(part: str) -> dict:
273
+ """One parameter as {name, type}. The default value is not part of either."""
274
+ head = part.split("=")[0].strip()
275
+ if ":" in head:
276
+ name, _, annotated = head.partition(":")
277
+ return {"name": name.strip(), "type": annotated.strip() or None}
278
+ if head.startswith("*"):
279
+ return {"name": "".join(head.split()), "type": None} # lizard prints `* args`
280
+ name, _, trailing = head.partition(" ")
281
+ return {"name": name, "type": trailing.strip() or None}
282
+
283
+
284
+ def coupling_partners(ranked: list[dict], path: str, is_test, top: int = 5) -> list[dict]:
285
+ """One file's coupled partners out of the ranking every path shares.
286
+
287
+ The ranking is global on purpose — a quiet file's own partners must not fall
288
+ behind the repo's noisiest pairs — so it is computed once for a whole batch
289
+ and cut per path here. `is_test` marks the partner that is a test file,
290
+ which is the partner an agent edits rather than reads.
291
+ """
292
+ out = []
293
+ for pair in ranked:
294
+ first, second = pair["files"]
295
+ if path not in pair["files"]:
296
+ continue
297
+ other = second if first == path else first
298
+ out.append({"path": other, "support": pair["support"],
299
+ "confidence": pair["confidence"], "is_test": is_test(other)})
300
+ return out[:top]
301
+
302
+
303
+ def with_contained(twins: list[dict]) -> list[dict]:
304
+ """Twins, each saying whether it is wholly contained in the target.
305
+
306
+ A twin the duplication pass did not flag reads as not contained rather than
307
+ as unknown: `contained` is a claim about the shingles, and no claim is False.
308
+ """
309
+ return [{**t, "contained": bool(t.get("contained", False))} for t in twins]
310
+
311
+
312
+ def notes(cfg, scope) -> dict:
313
+ """The prose the config carries for this repo and this scope, or nulls.
314
+
315
+ The config's own scope_notes table is the source of truth for a scope;
316
+ the record's attribute is the fallback. Read defensively: a config that
317
+ declares no notes at all is the ordinary case, and the packet must not
318
+ depend on any of these keys existing.
319
+ """
320
+ table = dict(getattr(cfg, "scope_notes", None) or {})
321
+ scoped = list(table.get(_scope_name(scope)) or ()) or _note_of(scope)
322
+ return {"repo": _note_of(cfg), "scope": scoped or None}
323
+
324
+
325
+ def _scope_name(scope) -> str | None:
326
+ named = getattr(scope, "name", None)
327
+ return named or (scope if isinstance(scope, str) else None)
328
+
329
+
330
+ def _note_of(holder) -> list[str] | str | None:
331
+ found = getattr(holder, "notes", None) or getattr(holder, "note", None)
332
+ if found is None:
333
+ return None
334
+ return list(found) if isinstance(found, tuple) else found
335
+
336
+
337
+ def versions_block(report: dict, analysis_version: int) -> dict:
338
+ """What produced these numbers: the tools, plus the metric's own version.
339
+
340
+ A packet outlives the run it describes. Without the analysis version, marks
341
+ and scores from two metric generations read as one series.
342
+ """
343
+ return {**report, "analysis_version": analysis_version}