crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/mutate_pool.py
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""Where mutants actually run. One worker (the default) mutates the live
|
|
2
|
+
working tree, exactly as this command always has. `mutation_workers = N` gives
|
|
3
|
+
each worker its own detached git worktree instead, because a mutant is a write
|
|
4
|
+
to a source file: two of them in one tree would read each other's edits.
|
|
5
|
+
|
|
6
|
+
Two things the parallel path owes the serial one. Results merge by the mutant's
|
|
7
|
+
position in the list, never by who finished first, so the same tree reports the
|
|
8
|
+
same JSON at any worker count. And the worktree is a checkout of HEAD while
|
|
9
|
+
`mutate` is diff-scoped against the working tree, so the targeted files are
|
|
10
|
+
copied in as they are on disk — uncommitted lines are the ones being mutated.
|
|
11
|
+
|
|
12
|
+
The command runs with cwd set to the worktree. A consumer whose test command
|
|
13
|
+
resolves the code under test from somewhere else (an editable install pointing
|
|
14
|
+
at the main checkout, a global site-packages copy) would measure unmutated
|
|
15
|
+
code and score every mutant a survivor: keep the command cwd-relative.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import os
|
|
20
|
+
import shutil
|
|
21
|
+
import subprocess
|
|
22
|
+
import tempfile
|
|
23
|
+
import threading
|
|
24
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
25
|
+
from contextlib import contextmanager
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
from .gitio import worktree_add, worktree_remove
|
|
29
|
+
from .mutate import apply_mutant
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def run_one(tree: Path, cfg, mutant) -> bool:
|
|
33
|
+
"""True = killed. The original file ALWAYS comes back, whatever happens."""
|
|
34
|
+
p = tree / mutant.path
|
|
35
|
+
original = p.read_bytes()
|
|
36
|
+
# Python validates .pyc files by source SIZE + mtime in WHOLE SECONDS: two
|
|
37
|
+
# same-size mutants applied within one second would reuse the first one's
|
|
38
|
+
# stale bytecode and read as false survivors. Kill the cache, write none.
|
|
39
|
+
shutil.rmtree(p.parent / "__pycache__", ignore_errors=True)
|
|
40
|
+
env = {**os.environ, "PYTHONDONTWRITEBYTECODE": "1"}
|
|
41
|
+
try:
|
|
42
|
+
p.write_text(apply_mutant(original.decode("utf-8", "replace"), mutant),
|
|
43
|
+
encoding="utf-8", newline="")
|
|
44
|
+
proc = subprocess.run(cfg.mutation_command, shell=True, cwd=tree, capture_output=True,
|
|
45
|
+
env=env, timeout=cfg.mutation_timeout_seconds)
|
|
46
|
+
return proc.returncode != 0
|
|
47
|
+
except subprocess.TimeoutExpired:
|
|
48
|
+
return True # a mutant that loops forever is dead
|
|
49
|
+
finally:
|
|
50
|
+
p.write_bytes(original)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _shards(indexed: list, workers: int) -> list[list]:
|
|
54
|
+
"""Round-robin: worker w takes mutants w, w+K, w+2K. Even to within one
|
|
55
|
+
mutant however the list is shaped, and each shard keeps list order."""
|
|
56
|
+
return [indexed[w::workers] for w in range(workers)]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _merge(done: list) -> list[bool]:
|
|
60
|
+
"""(index, killed) pairs from every worker back into mutant order."""
|
|
61
|
+
return [killed for _, killed in sorted(done)]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _run_shard(tree: Path, cfg, shard: list, report) -> list:
|
|
65
|
+
out = []
|
|
66
|
+
for index, mutant in shard:
|
|
67
|
+
killed = run_one(tree, cfg, mutant)
|
|
68
|
+
report(index, mutant, killed)
|
|
69
|
+
out.append((index, killed))
|
|
70
|
+
return out
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _seed(root: Path, tree: Path, rel_paths: list) -> None:
|
|
74
|
+
"""The worktree checked out HEAD; the mutants were grown from the working
|
|
75
|
+
tree. Copy the targeted files over so both agree on what line 40 is."""
|
|
76
|
+
for rel in rel_paths:
|
|
77
|
+
dst = tree / rel
|
|
78
|
+
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
shutil.copyfile(root / rel, dst)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _on_every_tree(action, root: Path, trees: list) -> None:
|
|
83
|
+
"""`action(root, tree)` on one thread per tree, and WAIT FOR THEM ALL, even
|
|
84
|
+
once one has raised. Leaving a checkout in flight is what turns a failed add
|
|
85
|
+
into a leaked worktree: the cleanup walks the list, finds nothing at that
|
|
86
|
+
path yet, and the abandoned thread creates it a moment later."""
|
|
87
|
+
with ThreadPoolExecutor(max_workers=len(trees)) as pool:
|
|
88
|
+
for done in [pool.submit(action, root, tree) for tree in trees]:
|
|
89
|
+
done.result()
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@contextmanager
|
|
93
|
+
def _worktrees(root: Path, count: int):
|
|
94
|
+
"""N detached checkouts, all gone on the way out — after a clean run, after
|
|
95
|
+
an exception, and after an add that failed halfway through the set.
|
|
96
|
+
|
|
97
|
+
The adds run together, and so do the removes. Each one is a full checkout
|
|
98
|
+
that spends its time waiting on the disk rather than on a core, so four of
|
|
99
|
+
them serialized cost 54.5 s on a 31,620-file repo against 25.1 s overlapped.
|
|
100
|
+
"""
|
|
101
|
+
base = Path(tempfile.mkdtemp(prefix="crapkit-mutate-"))
|
|
102
|
+
trees = [base / f"w{i}" for i in range(count)]
|
|
103
|
+
try:
|
|
104
|
+
_on_every_tree(worktree_add, root, trees)
|
|
105
|
+
yield trees
|
|
106
|
+
finally:
|
|
107
|
+
_teardown(root, base, trees)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _teardown(root: Path, base: Path, trees: list) -> None:
|
|
111
|
+
_on_every_tree(worktree_remove, root, trees)
|
|
112
|
+
shutil.rmtree(base, ignore_errors=True)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _fan_out(cfg, trees: list, shards: list, report) -> list:
|
|
116
|
+
with ThreadPoolExecutor(max_workers=len(trees)) as pool:
|
|
117
|
+
futures = [pool.submit(_run_shard, tree, cfg, shard, report)
|
|
118
|
+
for tree, shard in zip(trees, shards)]
|
|
119
|
+
return [pair for f in futures for pair in f.result()]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _run_parallel(root: Path, cfg, mutants: list, workers: int, report) -> list[bool]:
|
|
123
|
+
shards = _shards(list(enumerate(mutants)), workers)
|
|
124
|
+
targets = sorted({m.path for m in mutants})
|
|
125
|
+
with _worktrees(root, workers) as trees:
|
|
126
|
+
for tree in trees:
|
|
127
|
+
_seed(root, tree, targets)
|
|
128
|
+
done = _fan_out(cfg, trees, shards, report)
|
|
129
|
+
return _merge(done)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def run_mutants(root: Path, cfg, mutants: list, report) -> list[bool]:
|
|
133
|
+
"""One killed flag per mutant, in mutant order. Workers past the mutant
|
|
134
|
+
count would only pay for empty worktrees."""
|
|
135
|
+
workers = min(cfg.mutation_workers, len(mutants))
|
|
136
|
+
if workers > 1:
|
|
137
|
+
return _run_parallel(root, cfg, mutants, workers, report)
|
|
138
|
+
return _merge(_run_shard(root, cfg, list(enumerate(mutants)), report))
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def reporter(total: int, stream):
|
|
142
|
+
"""Progress lines, one per finished mutant. Workers race, so the write is
|
|
143
|
+
locked: a half-written line read as a survivor list is worse than no line."""
|
|
144
|
+
lock = threading.Lock()
|
|
145
|
+
|
|
146
|
+
def report(index: int, mutant, killed: bool) -> None:
|
|
147
|
+
verdict = "killed" if killed else "SURVIVED"
|
|
148
|
+
with lock:
|
|
149
|
+
print(f" mutant {index + 1}/{total} {mutant.path}:{mutant.line} "
|
|
150
|
+
f"[{mutant.op}] {verdict}", file=stream)
|
|
151
|
+
|
|
152
|
+
return report
|
crapkit/override.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""The audited override: three records or nothing.
|
|
2
|
+
|
|
3
|
+
An exemption exists only if all three surfaces carry it: the alert (a human
|
|
4
|
+
channel sees one line), the committed ratchet (the debt is diff-visible), and
|
|
5
|
+
the snapshot store (the run remembers). The alert fires first because it is the
|
|
6
|
+
step most likely to fail; a partial override fails loudly and grants nothing.
|
|
7
|
+
No environment-variable or silent bypass exists anywhere in crapkit.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import subprocess
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from .errors import ConfigError, ToolError
|
|
15
|
+
from .ratchet import RatchetEntry, dump_ratchet, load_ratchet
|
|
16
|
+
from .store import SnapshotStore
|
|
17
|
+
from .verify import GateViolation
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def record_override(
|
|
21
|
+
*,
|
|
22
|
+
store: SnapshotStore,
|
|
23
|
+
run_id: int,
|
|
24
|
+
root: Path,
|
|
25
|
+
ratchet_file: str,
|
|
26
|
+
alert_command: str,
|
|
27
|
+
violations: list[GateViolation],
|
|
28
|
+
reason: str,
|
|
29
|
+
raise_marks: bool = True,
|
|
30
|
+
) -> None:
|
|
31
|
+
_require_auditable_override(reason, alert_command)
|
|
32
|
+
_alert_or_refuse(alert_command, root, violations, reason)
|
|
33
|
+
|
|
34
|
+
# Audit before grant: the snapshot record lands BEFORE the ratchet write,
|
|
35
|
+
# because the ratchet entry is the functional exemption. A failure between
|
|
36
|
+
# the two leaves an audit trail with no grant, never a grant with no trail.
|
|
37
|
+
store.write_overrides(run_id, [(v.path, v.long_name, v.crap, reason) for v in violations])
|
|
38
|
+
|
|
39
|
+
_grant_ratchet_debt(root / ratchet_file, violations, raise_marks=raise_marks)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _require_auditable_override(reason: str, alert_command: str) -> None:
|
|
43
|
+
"""Refuse an override that could not be audited even if every step succeeded."""
|
|
44
|
+
if not reason.strip():
|
|
45
|
+
raise ConfigError("an override requires a non-empty reason")
|
|
46
|
+
if not alert_command.strip():
|
|
47
|
+
raise ConfigError(
|
|
48
|
+
"no alert_command configured — the override requires a visible alert line; "
|
|
49
|
+
"set [crapkit] alert_command in crapkit.toml")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _alert_or_refuse(alert_command: str, root: Path, violations: list[GateViolation],
|
|
53
|
+
reason: str) -> None:
|
|
54
|
+
"""Put the debt in front of a human first; a silent alert grants nothing."""
|
|
55
|
+
summary = "; ".join(f"{v.path}:{v.start} {v.long_name} crap={v.crap:.1f}" for v in violations)
|
|
56
|
+
line = f"crapkit OVERRIDE ({reason}): {summary}"
|
|
57
|
+
# The line reaches the alert command on stdin, never interpolated into the
|
|
58
|
+
# shell string: function names come from analyzed source and are not shell-safe.
|
|
59
|
+
proc = subprocess.run(alert_command, shell=True, cwd=root, input=line + "\n",
|
|
60
|
+
capture_output=True, text=True, encoding="utf-8", errors="replace")
|
|
61
|
+
if proc.returncode != 0:
|
|
62
|
+
raise ToolError(
|
|
63
|
+
f"override alert command failed (exit {proc.returncode}): "
|
|
64
|
+
f"{(proc.stderr or proc.stdout).strip()[-300:]} — no alert, no override")
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _grant_ratchet_debt(ratchet_path: Path, violations: list[GateViolation], *,
|
|
68
|
+
raise_marks: bool) -> None:
|
|
69
|
+
"""The functional exemption: the debt enters the committed ratchet, diff-visible."""
|
|
70
|
+
by_key = _marks_by_key(ratchet_path)
|
|
71
|
+
for v in violations:
|
|
72
|
+
mark = _override_mark(by_key.get((v.path, v.long_name)), v.crap, raise_marks=raise_marks)
|
|
73
|
+
by_key[(v.path, v.long_name)] = RatchetEntry(v.path, v.long_name, round(mark, 4))
|
|
74
|
+
ratchet_path.write_text(dump_ratchet(list(by_key.values())), encoding="utf-8", newline="\n")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _marks_by_key(ratchet_path: Path) -> dict[tuple[str, str], RatchetEntry]:
|
|
78
|
+
"""Prior marks by (path, long_name); an absent ratchet file is simply no marks."""
|
|
79
|
+
existing = load_ratchet(ratchet_path.read_text(encoding="utf-8")) if ratchet_path.is_file() else []
|
|
80
|
+
return {(e.path, e.long_name): e for e in existing}
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _override_mark(prior: RatchetEntry | None, crap: float, *, raise_marks: bool) -> float:
|
|
84
|
+
"""The mark this override records for one function."""
|
|
85
|
+
# raise_marks=False is the hook path: it synthesizes worst-case crap (no
|
|
86
|
+
# coverage data), and letting that raise a measured mark would blind the
|
|
87
|
+
# ratchet to a later real coverage collapse. The prior tighter mark stays,
|
|
88
|
+
# so the NEXT verify still demands repayment; the override only lets this
|
|
89
|
+
# one commit through.
|
|
90
|
+
if prior is None:
|
|
91
|
+
return crap
|
|
92
|
+
if raise_marks:
|
|
93
|
+
return max(prior.crap, crap)
|
|
94
|
+
return prior.crap
|
crapkit/packet.py
ADDED
|
@@ -0,0 +1,343 @@
|
|
|
1
|
+
"""The start-editing packet: everything a session needs before it opens the file.
|
|
2
|
+
|
|
3
|
+
`brief` answered what one function scores. A session then read the file to find
|
|
4
|
+
the other functions in it, guessed which ceiling the gate would apply, hunted for
|
|
5
|
+
the lane that measures the scope, and re-derived the commands to run. Each of
|
|
6
|
+
those is a value some caller already holds, so each is a field here instead of a
|
|
7
|
+
round trip.
|
|
8
|
+
|
|
9
|
+
Every function in this module is pure: values in, a dict or a list out. The
|
|
10
|
+
reads that feed them — the store, git, the config, the file texts — belong to
|
|
11
|
+
the caller, which is what lets one batch of packets pay for them once. Nothing
|
|
12
|
+
here removes or retypes a field `brief --json` already published; the packet is
|
|
13
|
+
what was added around it.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from .ratchet_report import DAY
|
|
18
|
+
|
|
19
|
+
# What the gate actually enforces, said once. A session that reads a ceiling of
|
|
20
|
+
# 6 beside a standing mark of 72 otherwise reads a contradiction and either
|
|
21
|
+
# refuses to start or "fixes" debt nobody asked it to touch.
|
|
22
|
+
GATE_BINDS = ("changed functions only; a ratchet mark pardons standing debt "
|
|
23
|
+
"at or under it")
|
|
24
|
+
|
|
25
|
+
_OPENERS = "([{<"
|
|
26
|
+
_CLOSERS = ")]}>"
|
|
27
|
+
|
|
28
|
+
# What lizard calls a function it could not name. Every anonymous function in a
|
|
29
|
+
# file prints the same string, which is why the handle below exists.
|
|
30
|
+
ANONYMOUS = "(anonymous)"
|
|
31
|
+
|
|
32
|
+
# `stale` clears when a run lands on the current commit and never before. The
|
|
33
|
+
# packet used to answer its own staleness warning with another `brief`, which
|
|
34
|
+
# re-reads the snapshot that is already stale. `--reuse-unchanged` reruns only
|
|
35
|
+
# the lanes whose scope files moved and parses the rest off the artifacts they
|
|
36
|
+
# already have, so it is the cheapest call that still writes a run.
|
|
37
|
+
REFRESH = "python -m crapkit coverage --reuse-unchanged"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def function_source(text: str | None, start: int, end: int) -> str | None:
|
|
41
|
+
"""One function's lines out of the file text the caller already read.
|
|
42
|
+
|
|
43
|
+
None means nobody read the file, which is not the same as a function whose
|
|
44
|
+
span holds no lines.
|
|
45
|
+
"""
|
|
46
|
+
if text is None:
|
|
47
|
+
return None
|
|
48
|
+
return "\n".join(text.splitlines()[start - 1:end])
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def file_functions(rows) -> list[dict]:
|
|
52
|
+
"""Every scored row in the file, not just the one the brief is about.
|
|
53
|
+
|
|
54
|
+
A decomposition lands in the neighbours: the helper it extracts into, the
|
|
55
|
+
twin beside it, the row that is already at its ceiling and must stay there.
|
|
56
|
+
"""
|
|
57
|
+
return [{"function": r.long_name, "start": r.start, "end": r.end, "ccn": r.ccn,
|
|
58
|
+
"crap": r.crap, "remedy": r.remedy} for r in rows]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def file_totals(rows, scope_targets: dict, target: int) -> dict:
|
|
62
|
+
"""The file's own numbers, each row judged against ITS scope's ceiling.
|
|
63
|
+
|
|
64
|
+
A file can hold rows from two scopes; scoring the whole file against one
|
|
65
|
+
ceiling would report debt a per-scope target deliberately allows.
|
|
66
|
+
"""
|
|
67
|
+
over = sum(1 for r in rows if r.crap > scope_targets.get(r.scope, target))
|
|
68
|
+
return {"functions": len(rows), "over_target": over,
|
|
69
|
+
"crap_load": round(sum(r.crap for r in rows), 2)}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def gate_rule(*, ceiling: int, mark: float | None, mark_age_days: int | None,
|
|
73
|
+
diff_uncovered_max: int | None) -> dict:
|
|
74
|
+
"""The rule this function will be judged by, spelled out rather than implied."""
|
|
75
|
+
return {"ceiling": ceiling, "binds": GATE_BINDS, "ratchet_mark": mark,
|
|
76
|
+
"mark_age_days": mark_age_days, "diff_uncovered_max": diff_uncovered_max}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def mark_age_days(events: list[tuple], key: tuple) -> int | None:
|
|
80
|
+
"""How long this function's mark has stood, in the ratchet history's own time.
|
|
81
|
+
|
|
82
|
+
Anchored on the newest commit in the history, never the wall clock, so a
|
|
83
|
+
fixed history reports the same age forever. A mark that was repaid and later
|
|
84
|
+
re-added is aged from its return: the debt is the one standing now.
|
|
85
|
+
"""
|
|
86
|
+
entered = None
|
|
87
|
+
anchor = 0
|
|
88
|
+
for ts, event_key, kind, _ in events:
|
|
89
|
+
anchor = max(anchor, ts)
|
|
90
|
+
if event_key == key:
|
|
91
|
+
entered = ts if kind == "added" else None
|
|
92
|
+
return None if entered is None else (anchor - entered) // DAY
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def lane_for(scope: str | None, lanes):
|
|
96
|
+
"""The first lane claiming this scope, or None when no lane measures it."""
|
|
97
|
+
if scope is None:
|
|
98
|
+
return None
|
|
99
|
+
return next((lane for lane in lanes if scope in lane.scopes), None)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def lane_record(lane) -> dict | None:
|
|
103
|
+
"""The lane verbatim: what ran, where, and how long it is allowed to take.
|
|
104
|
+
|
|
105
|
+
A session that reruns the lane by hand needs the cwd and the env as declared;
|
|
106
|
+
reconstructing them from the command string is how the reruns drift.
|
|
107
|
+
"""
|
|
108
|
+
if lane is None:
|
|
109
|
+
return None
|
|
110
|
+
return {"name": lane.name, "command": lane.command, "artifact": lane.artifact,
|
|
111
|
+
"parser": lane.parser, "cwd": lane.cwd, "env": dict(lane.env),
|
|
112
|
+
"timeout_seconds": lane.timeout_seconds}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def commands(path: str, scoped: str | None, note: str = "") -> dict:
|
|
116
|
+
"""The four commands a session runs next, with the paths already filled in.
|
|
117
|
+
|
|
118
|
+
`refresh_writes_run` rides beside `refresh` because the other three change
|
|
119
|
+
nothing on disk and that one does: a read-only session, or one holding a
|
|
120
|
+
tree it is not allowed to score, has to know which of the four it may run.
|
|
121
|
+
"""
|
|
122
|
+
out = {"gate": f"python -m crapkit rescore {path} --gate",
|
|
123
|
+
"scoped_tests": scoped,
|
|
124
|
+
"verify": "python -m crapkit verify",
|
|
125
|
+
"refresh": REFRESH,
|
|
126
|
+
"refresh_writes_run": True}
|
|
127
|
+
if scoped is None and note:
|
|
128
|
+
out["scoped_tests_note"] = note
|
|
129
|
+
return out
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def bare_name(long_name: str) -> str:
|
|
133
|
+
"""The identifier a long_name opens with, before its parameter list.
|
|
134
|
+
|
|
135
|
+
Empty for a function lizard could not name: both `(anonymous)` and
|
|
136
|
+
`(anonymous) ( z )` open with the parenthesis, so an empty prefix IS the
|
|
137
|
+
test for anonymity, with no second string to keep in step.
|
|
138
|
+
"""
|
|
139
|
+
return long_name.split("(")[0].strip()
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def anonymous_starts(rows) -> list[int]:
|
|
143
|
+
"""Where the file's anonymous functions open, in file order."""
|
|
144
|
+
return sorted(r.start for r in rows if not bare_name(r.long_name))
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def handles(rows) -> dict[int, str]:
|
|
148
|
+
"""The handle for every row in one file, keyed by the line it opens on.
|
|
149
|
+
|
|
150
|
+
A named function is its own handle. An anonymous one is `(anonymous)#N`,
|
|
151
|
+
counted over the file's anonymous functions in start order — a position, not
|
|
152
|
+
a line, so the string a session copies out of a packet still names the same
|
|
153
|
+
function after an edit above it moves every line below.
|
|
154
|
+
|
|
155
|
+
Keyed by start because no two functions in a file open on the same line,
|
|
156
|
+
which makes it the one per-file key a row already carries.
|
|
157
|
+
"""
|
|
158
|
+
ordinals = {start: n for n, start in enumerate(anonymous_starts(rows), 1)}
|
|
159
|
+
return {r.start: bare_name(r.long_name) or f"{ANONYMOUS}#{ordinals[r.start]}"
|
|
160
|
+
for r in rows}
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def handle_names(rows) -> list[str]:
|
|
164
|
+
"""Every anonymous handle this file offers, in order.
|
|
165
|
+
|
|
166
|
+
What an out-of-range ordinal is reported against: a session that guessed #5
|
|
167
|
+
needs the two that exist, the same way a wrong bare name gets the file's
|
|
168
|
+
real names back.
|
|
169
|
+
"""
|
|
170
|
+
return [f"{ANONYMOUS}#{n}" for n in range(1, len(anonymous_starts(rows)) + 1)]
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def handle_ordinal(name: str) -> int | None:
|
|
174
|
+
"""The N in `(anonymous)#N`, or None when `name` is some other name form.
|
|
175
|
+
|
|
176
|
+
None rather than an error: this is the question "is that string a handle",
|
|
177
|
+
asked before the other name forms get their turn.
|
|
178
|
+
"""
|
|
179
|
+
head, sep, tail = name.partition("#")
|
|
180
|
+
if not sep or head.strip() != ANONYMOUS or not tail.isdigit():
|
|
181
|
+
return None
|
|
182
|
+
return int(tail)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def budget(row, ceiling: int) -> dict:
|
|
186
|
+
"""What the work costs: pieces a decomposition needs, decision paths no test
|
|
187
|
+
walks.
|
|
188
|
+
|
|
189
|
+
One definition for both readers. `next-item` published these and `brief` did
|
|
190
|
+
not, so a session that opened on a packet re-derived numbers the queue had
|
|
191
|
+
already computed — and two derivations of one formula drift with nothing to
|
|
192
|
+
catch it.
|
|
193
|
+
"""
|
|
194
|
+
return {"est_splits": 0 if row.ccn <= ceiling else -(-row.ccn // ceiling),
|
|
195
|
+
"est_uncovered_paths": max(0, round((1 - row.cov) * row.ccn))}
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def regrowth(history: list[dict]) -> dict:
|
|
199
|
+
"""Whether this function's complexity fell and then came back.
|
|
200
|
+
|
|
201
|
+
A function somebody already decomposed once, back over its ceiling, is a
|
|
202
|
+
different job from one that has always been big: the decomposition that was
|
|
203
|
+
tried is on record and did not hold.
|
|
204
|
+
"""
|
|
205
|
+
return {"regrown": _fell_then_rose([h["ccn"] for h in history]),
|
|
206
|
+
"history": [[h["run_id"], h["ccn"]] for h in history]}
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _fell_then_rose(ccns: list[int]) -> bool:
|
|
210
|
+
"""True once a drop is followed anywhere later by a climb."""
|
|
211
|
+
fell = False
|
|
212
|
+
for before, after in zip(ccns, ccns[1:]):
|
|
213
|
+
if fell and after > before:
|
|
214
|
+
return True
|
|
215
|
+
fell = fell or after < before
|
|
216
|
+
return False
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def params(long_name: str) -> list[dict]:
|
|
220
|
+
"""The parameter list out of lizard's long_name, name first.
|
|
221
|
+
|
|
222
|
+
lizard prints the signature it parsed: `f( a , b = 1 , c : int = 2 )` in
|
|
223
|
+
Python, `dispatch ( a , b Record , c )` in TypeScript. The name leads in
|
|
224
|
+
both; whatever follows it is the type annotation as lizard printed it.
|
|
225
|
+
Anything this cannot read is an empty list, never a guess.
|
|
226
|
+
"""
|
|
227
|
+
inner = _param_text(long_name)
|
|
228
|
+
if inner is None:
|
|
229
|
+
return []
|
|
230
|
+
return [_one_param(part) for part in _split_top(inner) if part]
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _param_text(long_name: str) -> str | None:
|
|
234
|
+
"""What sits inside the LAST balanced parentheses, or None when there are none.
|
|
235
|
+
|
|
236
|
+
Not the first `(`: lizard names an anonymous function `(anonymous) ( z )`,
|
|
237
|
+
where the first one belongs to the name and the parameter list is the group
|
|
238
|
+
that closes the string.
|
|
239
|
+
"""
|
|
240
|
+
closed = long_name.rfind(")")
|
|
241
|
+
opened = _matching_open(long_name, closed)
|
|
242
|
+
return None if opened is None else long_name[opened + 1:closed]
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _matching_open(text: str, closed: int) -> int | None:
|
|
246
|
+
"""The index of the `(` that opens the group closing at `closed`."""
|
|
247
|
+
depth = 0
|
|
248
|
+
for i in range(closed, -1, -1):
|
|
249
|
+
depth += (text[i] == ")") - (text[i] == "(")
|
|
250
|
+
if depth == 0 and text[i] == "(":
|
|
251
|
+
return i
|
|
252
|
+
return None
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _split_top(text: str) -> list[str]:
|
|
256
|
+
"""Split on commas that are not inside brackets, so `Map<a , b>` stays one."""
|
|
257
|
+
parts = []
|
|
258
|
+
depth = 0
|
|
259
|
+
start = 0
|
|
260
|
+
for i, ch in enumerate(text):
|
|
261
|
+
if ch in _OPENERS:
|
|
262
|
+
depth += 1
|
|
263
|
+
elif ch in _CLOSERS:
|
|
264
|
+
depth -= 1
|
|
265
|
+
elif ch == "," and depth == 0:
|
|
266
|
+
parts.append(text[start:i].strip())
|
|
267
|
+
start = i + 1
|
|
268
|
+
parts.append(text[start:].strip())
|
|
269
|
+
return parts
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _one_param(part: str) -> dict:
|
|
273
|
+
"""One parameter as {name, type}. The default value is not part of either."""
|
|
274
|
+
head = part.split("=")[0].strip()
|
|
275
|
+
if ":" in head:
|
|
276
|
+
name, _, annotated = head.partition(":")
|
|
277
|
+
return {"name": name.strip(), "type": annotated.strip() or None}
|
|
278
|
+
if head.startswith("*"):
|
|
279
|
+
return {"name": "".join(head.split()), "type": None} # lizard prints `* args`
|
|
280
|
+
name, _, trailing = head.partition(" ")
|
|
281
|
+
return {"name": name, "type": trailing.strip() or None}
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def coupling_partners(ranked: list[dict], path: str, is_test, top: int = 5) -> list[dict]:
|
|
285
|
+
"""One file's coupled partners out of the ranking every path shares.
|
|
286
|
+
|
|
287
|
+
The ranking is global on purpose — a quiet file's own partners must not fall
|
|
288
|
+
behind the repo's noisiest pairs — so it is computed once for a whole batch
|
|
289
|
+
and cut per path here. `is_test` marks the partner that is a test file,
|
|
290
|
+
which is the partner an agent edits rather than reads.
|
|
291
|
+
"""
|
|
292
|
+
out = []
|
|
293
|
+
for pair in ranked:
|
|
294
|
+
first, second = pair["files"]
|
|
295
|
+
if path not in pair["files"]:
|
|
296
|
+
continue
|
|
297
|
+
other = second if first == path else first
|
|
298
|
+
out.append({"path": other, "support": pair["support"],
|
|
299
|
+
"confidence": pair["confidence"], "is_test": is_test(other)})
|
|
300
|
+
return out[:top]
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def with_contained(twins: list[dict]) -> list[dict]:
|
|
304
|
+
"""Twins, each saying whether it is wholly contained in the target.
|
|
305
|
+
|
|
306
|
+
A twin the duplication pass did not flag reads as not contained rather than
|
|
307
|
+
as unknown: `contained` is a claim about the shingles, and no claim is False.
|
|
308
|
+
"""
|
|
309
|
+
return [{**t, "contained": bool(t.get("contained", False))} for t in twins]
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def notes(cfg, scope) -> dict:
|
|
313
|
+
"""The prose the config carries for this repo and this scope, or nulls.
|
|
314
|
+
|
|
315
|
+
The config's own scope_notes table is the source of truth for a scope;
|
|
316
|
+
the record's attribute is the fallback. Read defensively: a config that
|
|
317
|
+
declares no notes at all is the ordinary case, and the packet must not
|
|
318
|
+
depend on any of these keys existing.
|
|
319
|
+
"""
|
|
320
|
+
table = dict(getattr(cfg, "scope_notes", None) or {})
|
|
321
|
+
scoped = list(table.get(_scope_name(scope)) or ()) or _note_of(scope)
|
|
322
|
+
return {"repo": _note_of(cfg), "scope": scoped or None}
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _scope_name(scope) -> str | None:
|
|
326
|
+
named = getattr(scope, "name", None)
|
|
327
|
+
return named or (scope if isinstance(scope, str) else None)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _note_of(holder) -> list[str] | str | None:
|
|
331
|
+
found = getattr(holder, "notes", None) or getattr(holder, "note", None)
|
|
332
|
+
if found is None:
|
|
333
|
+
return None
|
|
334
|
+
return list(found) if isinstance(found, tuple) else found
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def versions_block(report: dict, analysis_version: int) -> dict:
|
|
338
|
+
"""What produced these numbers: the tools, plus the metric's own version.
|
|
339
|
+
|
|
340
|
+
A packet outlives the run it describes. Without the analysis version, marks
|
|
341
|
+
and scores from two metric generations read as one series.
|
|
342
|
+
"""
|
|
343
|
+
return {**report, "analysis_version": analysis_version}
|