gitmole 0.6.5__tar.gz → 0.6.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gitmole-0.6.5 → gitmole-0.6.7}/PKG-INFO +1 -1
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/__init__.py +1 -1
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/cli.py +3 -1
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/filetypes.py +62 -2
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/findings.py +35 -12
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/identity.py +19 -1
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/load.py +1 -1
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/maat.py +17 -2
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/render.py +25 -2
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/run.py +1 -1
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/watch.py +2 -2
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole.egg-info/PKG-INFO +1 -1
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_cli.py +22 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_filetypes.py +30 -2
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_findings.py +49 -3
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_identity.py +13 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_load.py +5 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_maat.py +12 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_render.py +41 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_run.py +2 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_watch.py +8 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/LICENSE +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/README.md +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/__main__.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/backtest.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/banner.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/blame.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/clean.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/coupling.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/functions.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/hotspots.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/knowledge.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/leaks.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/loss.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/textfmt.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole/trend.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole.egg-info/SOURCES.txt +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole.egg-info/dependency_links.txt +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole.egg-info/entry_points.txt +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole.egg-info/requires.txt +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/gitmole.egg-info/top_level.txt +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/pyproject.toml +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/setup.cfg +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_backtest.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_banner.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_blame.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_clean.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_coupling.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_functions.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_golden.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_hotspots.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_knowledge.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_leaks.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_loss.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_packaging.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_textfmt.py +0 -0
- {gitmole-0.6.5 → gitmole-0.6.7}/tests/test_trend.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.7
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -15,7 +15,7 @@ from rich.live import Live
|
|
|
15
15
|
from rich.spinner import Spinner
|
|
16
16
|
from rich.text import Text
|
|
17
17
|
|
|
18
|
-
from . import __version__, banner, filetypes, findings, load, loss, run
|
|
18
|
+
from . import __version__, banner, blame, filetypes, findings, load, loss, run
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
def parse_args(argv):
|
|
@@ -294,6 +294,8 @@ def _meta_for_run(repo_dir: str, args, estimate, age_ok: bool, plots_ok: bool, p
|
|
|
294
294
|
meta = run.collect_meta(repo_dir, since=args.since_date)
|
|
295
295
|
meta["file_types"] = types_spec # the loader filters scc's size data the way every other step was filtered
|
|
296
296
|
meta["gone_months"] = args.gone
|
|
297
|
+
ignore = list(run.DATA_IGNORES if args.ignore_data else []) + list(args.ignore)
|
|
298
|
+
meta["generated"] = filetypes.generated_files(repo_dir, blame.text_files(repo_dir, ignore)) # hidden from the tables, out of the findings
|
|
297
299
|
if args.since_date and meta["commits"] == 0:
|
|
298
300
|
raise NoCommits(f"no commits since {args.since_date}; widen --since")
|
|
299
301
|
if args.now:
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
Standalone so blame.py and maat.py can import it as scripts."""
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
import fnmatch
|
|
6
|
+
import os
|
|
5
7
|
import re
|
|
6
8
|
import subprocess
|
|
7
9
|
from collections import Counter
|
|
@@ -74,15 +76,73 @@ def is_sample_path(path: str) -> bool:
|
|
|
74
76
|
return bool(_SAMPLE_PATH.search(path))
|
|
75
77
|
|
|
76
78
|
|
|
77
|
-
_VENDOR_PATH = re.compile(r"(^|/)(_?vendor|node_modules|third_?party|external)(/|$)", re.I)
|
|
79
|
+
_VENDOR_PATH = re.compile(r"(^|/)(_?vendor|vendored|node_modules|third_?party|external)(/|$)|^[^/]+/packages/", re.I)
|
|
78
80
|
|
|
79
81
|
|
|
80
82
|
def is_vendor_path(path: str) -> bool:
|
|
81
83
|
"""Vendored and third-party trees: somebody else's code, so its complexity and its single
|
|
82
|
-
importer are not this repository's risk.
|
|
84
|
+
importer are not this repository's risk. A `packages/` inside a package (requests/packages/,
|
|
85
|
+
the Python vendoring convention) counts; a monorepo's own `packages/` at the root does not."""
|
|
83
86
|
return bool(_VENDOR_PATH.search(path))
|
|
84
87
|
|
|
85
88
|
|
|
89
|
+
_RELEASE_NAMES = {"version", "version.rb", "version.py", "version.go", "version.rs", "version.txt", "__version__.py", "package.json",
|
|
90
|
+
"package-lock.json", "yarn.lock", "pnpm-lock.yaml", "gemfile", "gemfile.lock", "cargo.toml", "cargo.lock",
|
|
91
|
+
"pyproject.toml", "setup.py", "setup.cfg", "poetry.lock", "uv.lock", "go.mod", "go.sum", "composer.json", "composer.lock"}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def is_release_path(path: str) -> bool:
|
|
95
|
+
"""Release plumbing: version files, manifests, lock files and changelogs. Two of them changing
|
|
96
|
+
together is a release commit, not a dependency between them."""
|
|
97
|
+
name = path.rsplit("/", 1)[-1].lower()
|
|
98
|
+
return name in _RELEASE_NAMES or name.endswith(".gemspec") or name.startswith(("changelog", "changes.", "history.", "news."))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# What a generated file says about itself in its first lines: protoc, ajv, code generators of every kind.
|
|
102
|
+
_GENERATED = re.compile(r"auto[- ]?generated|generated (by|from|file|code|automatically|with)|do not (edit|modify)|@generated|code generated", re.I)
|
|
103
|
+
GENERATED_HEAD_LINES = 5
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _generated_patterns(repo: str) -> list:
|
|
107
|
+
"""The .gitattributes patterns marked linguist-generated at the repository root."""
|
|
108
|
+
try:
|
|
109
|
+
with open(os.path.join(repo, ".gitattributes"), encoding="utf-8", errors="replace") as fh:
|
|
110
|
+
lines = fh.read().splitlines()
|
|
111
|
+
except OSError:
|
|
112
|
+
return []
|
|
113
|
+
out = []
|
|
114
|
+
for line in lines:
|
|
115
|
+
parts = line.split()
|
|
116
|
+
if len(parts) >= 2 and any(p in ("linguist-generated", "linguist-generated=true") for p in parts[1:]):
|
|
117
|
+
out.append(parts[0].lstrip("/"))
|
|
118
|
+
return out
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _attribute_match(path: str, pattern: str) -> bool:
|
|
122
|
+
if "/" in pattern:
|
|
123
|
+
return fnmatch.fnmatchcase(path, pattern) or fnmatch.fnmatchcase(path, pattern.rstrip("/") + "/*")
|
|
124
|
+
return fnmatch.fnmatchcase(path.rsplit("/", 1)[-1], pattern)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def generated_files(repo: str, paths: list) -> list:
|
|
128
|
+
"""The tracked files that are generated: marked linguist-generated in .gitattributes, or saying so
|
|
129
|
+
in their first lines. Their complexity and churn are the generator's, not the repository's."""
|
|
130
|
+
patterns = _generated_patterns(repo)
|
|
131
|
+
out = []
|
|
132
|
+
for path in paths:
|
|
133
|
+
if any(_attribute_match(path, p) for p in patterns):
|
|
134
|
+
out.append(path)
|
|
135
|
+
continue
|
|
136
|
+
try:
|
|
137
|
+
with open(os.path.join(repo, path), "rb") as fh:
|
|
138
|
+
head = fh.read(2048)
|
|
139
|
+
except OSError:
|
|
140
|
+
continue
|
|
141
|
+
if any(_GENERATED.search(line) for line in head.decode("utf-8", "replace").splitlines()[:GENERATED_HEAD_LINES]):
|
|
142
|
+
out.append(path)
|
|
143
|
+
return sorted(out)
|
|
144
|
+
|
|
145
|
+
|
|
86
146
|
def key(path: str) -> str:
|
|
87
147
|
"""The lowercased extension, or the whole lowercased name when there is none."""
|
|
88
148
|
name = path.rsplit("/", 1)[-1].lower()
|
|
@@ -47,7 +47,8 @@ def secrets_found(report: dict) -> list:
|
|
|
47
47
|
groups = leaks.group(report.get("secrets") or [])
|
|
48
48
|
|
|
49
49
|
def in_source(g):
|
|
50
|
-
return any(not (filetypes.is_test_path(f) or filetypes.is_doc_path(f) or filetypes.is_sample_path(f)
|
|
50
|
+
return any(not (filetypes.is_test_path(f) or filetypes.is_doc_path(f) or filetypes.is_sample_path(f) or filetypes.is_vendor_path(f))
|
|
51
|
+
for f in g["files"])
|
|
51
52
|
source = [g for g in groups if in_source(g)]
|
|
52
53
|
aside = [g for g in groups if not in_source(g)]
|
|
53
54
|
ignore = "Add the fingerprint of any false positive from secrets.json to .betterleaksignore in the repository."
|
|
@@ -56,7 +57,7 @@ def secrets_found(report: dict) -> list:
|
|
|
56
57
|
out.append(_f("critical", f"{len(source)} secret(s) in history", _secret_statement(source),
|
|
57
58
|
f"Rotate them; deleting the file does not remove them from git. {ignore}"))
|
|
58
59
|
if aside:
|
|
59
|
-
out.append(_f("warning", f"{len(aside)} secret(s) only in test, example or documentation files", _secret_statement(aside),
|
|
60
|
+
out.append(_f("warning", f"{len(aside)} secret(s) only in test, example, vendored or documentation files", _secret_statement(aside),
|
|
60
61
|
f"Confirm they are fixtures or templates, not live keys. {ignore}"))
|
|
61
62
|
return out
|
|
62
63
|
|
|
@@ -173,8 +174,10 @@ def sizer_concerns(report: dict) -> list:
|
|
|
173
174
|
|
|
174
175
|
|
|
175
176
|
def hotspot_dominance(report: dict, ratio: float = 2.0, minimum: int = 20) -> list:
|
|
176
|
-
"""One source file takes most of the churn. Test files are left out: they change with everything.
|
|
177
|
-
|
|
177
|
+
"""One source file takes most of the churn. Test files are left out: they change with everything.
|
|
178
|
+
So is release plumbing: a version file or a manifest changes on every release by design."""
|
|
179
|
+
revs = sorted((r for r in report.get("revisions") or [] if not (filetypes.is_test_path(r["entity"]) or filetypes.is_release_path(r["entity"]))),
|
|
180
|
+
key=lambda r: -r["n-revs"])
|
|
178
181
|
if len(revs) < 2 or revs[0]["n-revs"] < minimum or revs[0]["n-revs"] < ratio * revs[1]["n-revs"]:
|
|
179
182
|
return []
|
|
180
183
|
top, nxt = revs[0], revs[1]
|
|
@@ -185,10 +188,12 @@ def hotspot_dominance(report: dict, ratio: float = 2.0, minimum: int = 20) -> li
|
|
|
185
188
|
|
|
186
189
|
def tight_coupling(report: dict, min_degree: int = 80, min_revs: int = 5) -> list:
|
|
187
190
|
"""A file and its test are expected to change together, so pairs with a test file on either side are
|
|
188
|
-
left out; so are pairs where either file is no longer in the tree, which are history, not a dependency
|
|
191
|
+
left out; so are pairs where either file is no longer in the tree, which are history, not a dependency,
|
|
192
|
+
and pairs of release plumbing (two version files, a manifest and its lock file), which are a release."""
|
|
189
193
|
tree = _tree(report)
|
|
190
194
|
pairs = [p for p in report.get("coupling") or [] if p["degree"] >= min_degree and p["average-revs"] >= min_revs
|
|
191
195
|
and not (filetypes.is_test_path(p["entity"]) or filetypes.is_test_path(p["coupled"]))
|
|
196
|
+
and not (filetypes.is_release_path(p["entity"]) and filetypes.is_release_path(p["coupled"]))
|
|
192
197
|
and not (tree and (p["entity"] not in tree or p["coupled"] not in tree))]
|
|
193
198
|
if not pairs:
|
|
194
199
|
return []
|
|
@@ -229,8 +234,10 @@ def stale_files(report: dict, months: int = 12, share: float = 0.3) -> list:
|
|
|
229
234
|
|
|
230
235
|
|
|
231
236
|
def bug_magnets(report: dict, min_recent: int = 3, warn_at: int = 5) -> list:
|
|
232
|
-
"""Source files with a run of recent fix commits. Test files are left out: they change with every fix.
|
|
233
|
-
|
|
237
|
+
"""Source files with a run of recent fix commits. Test files are left out: they change with every fix.
|
|
238
|
+
So is release plumbing: a manifest touched by every fix release is not where the bug was."""
|
|
239
|
+
hot = [f for f in report.get("fixes") or [] if f["recent-fixes"] >= min_recent
|
|
240
|
+
and not (filetypes.is_test_path(f["entity"]) or filetypes.is_release_path(f["entity"]))]
|
|
234
241
|
if not hot:
|
|
235
242
|
return []
|
|
236
243
|
hot.sort(key=lambda f: (-f["recent-fixes"], -f["n-fixes"], f["entity"]))
|
|
@@ -386,20 +393,36 @@ def _partial_functions(report: dict) -> str:
|
|
|
386
393
|
|
|
387
394
|
|
|
388
395
|
def brain_methods(report: dict, min_ccn: int = 15, min_lines: int = 100) -> list:
|
|
389
|
-
"""Functions that are both long and complex, in this repository's own source files: test files
|
|
390
|
-
vendored code are left out. A warning when one sits in a hotspot."""
|
|
396
|
+
"""Functions that are both long and complex, in this repository's own source files: test files,
|
|
397
|
+
vendored code and generated files are left out. A warning when one sits in a hotspot."""
|
|
398
|
+
generated = _generated(report)
|
|
391
399
|
big = [f for f in report.get("functions") or [] if f["ccn"] >= min_ccn and f["nloc"] >= min_lines
|
|
392
|
-
and not (filetypes.is_test_path(f["file"]) or filetypes.is_vendor_path(f["file"]))]
|
|
400
|
+
and not (filetypes.is_test_path(f["file"]) or filetypes.is_vendor_path(f["file"]) or f["file"] in generated)]
|
|
393
401
|
if not big:
|
|
394
402
|
return []
|
|
395
403
|
big.sort(key=lambda f: (-f["ccn"], -f["nloc"], f["file"], f["function"], f["start"]))
|
|
396
404
|
hot = hotspots.top(report)
|
|
397
405
|
sev = "warning" if any(f["file"] in hot for f in big) else "info"
|
|
398
|
-
listed = "; ".join(f"{f['function']} ({f
|
|
406
|
+
listed = "; ".join(f"{f['function']} ({_place(f)}) complexity {f['ccn']}, {f['nloc']} lines, {f['params']} params" for f in big[:5])
|
|
399
407
|
more = f" and {len(big) - 5} more" if len(big) > 5 else ""
|
|
408
|
+
first = big[0]
|
|
409
|
+
which = f"the anonymous function at {_place(first)}" if first["function"] == ANONYMOUS else f"{first['function']} in {first['file']}"
|
|
400
410
|
return [_f(sev, "Brain methods",
|
|
401
411
|
f"{len(big)} function(s) are both long and complex: {listed}{more}.{_partial_functions(report)}",
|
|
402
|
-
f"Split {
|
|
412
|
+
f"Split {which} first, before the next change lands there.")]
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
ANONYMOUS = "(anonymous)"
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _place(f: dict) -> str:
|
|
419
|
+
"""Where a function is: its file, or file:line when it has no name to find it by."""
|
|
420
|
+
return f"{f['file']}:{f['start']}" if f["function"] == ANONYMOUS else f["file"]
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def _generated(report: dict) -> set:
|
|
424
|
+
"""Files the run found to be generated (a header marker or a linguist-generated attribute)."""
|
|
425
|
+
return set((report.get("meta") or {}).get("generated") or [])
|
|
403
426
|
|
|
404
427
|
|
|
405
428
|
def complexity_growth(report: dict, min_growers: int = 3, min_pct: int = 25, top_n: int = 10) -> list:
|
|
@@ -20,8 +20,26 @@ def _tokens(name: str) -> set:
|
|
|
20
20
|
return {t for t in re.split(r"[^a-z0-9]+", name.lower()) if len(t) >= 3}
|
|
21
21
|
|
|
22
22
|
|
|
23
|
+
# A bare first name under two emails may be two people; anything else spelled identically is one.
|
|
24
|
+
_COMMON_FIRST_NAMES = {
|
|
25
|
+
"adam", "alex", "alexander", "andrew", "andy", "ann", "anna", "ben", "bob", "chris", "dan", "daniel", "dave", "david", "ed",
|
|
26
|
+
"eric", "frank", "george", "jack", "james", "jan", "jean", "jim", "joe", "john", "jon", "josh", "kevin", "lee", "li", "luke",
|
|
27
|
+
"mark", "martin", "matt", "max", "michael", "mike", "nick", "paul", "pete", "peter", "phil", "rob", "robert", "ryan", "sam",
|
|
28
|
+
"scott", "steve", "tim", "tom", "tony", "will",
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _plain(name: str) -> str:
|
|
33
|
+
return " ".join(name.lower().split())
|
|
34
|
+
|
|
35
|
+
|
|
23
36
|
def same_person(a: dict, b: dict) -> bool:
|
|
24
|
-
|
|
37
|
+
"""Same email, two shared name tokens, or the same name spelled identically (a handle such as
|
|
38
|
+
KaKa under three emails), unless that name is a bare common first name."""
|
|
39
|
+
if a["email"].lower() == b["email"].lower() or len(_tokens(a["name"]) & _tokens(b["name"])) >= 2:
|
|
40
|
+
return True
|
|
41
|
+
name = _plain(a["name"])
|
|
42
|
+
return bool(name) and name == _plain(b["name"]) and name not in _COMMON_FIRST_NAMES
|
|
25
43
|
|
|
26
44
|
|
|
27
45
|
def merge(identities: list) -> list:
|
|
@@ -151,7 +151,7 @@ def parse_functions(text: str) -> list:
|
|
|
151
151
|
for r in csv.reader(io.StringIO(text)):
|
|
152
152
|
if len(r) < 11:
|
|
153
153
|
continue
|
|
154
|
-
rows.append({"file": _rel(r[6]), "function": r[7], "ccn": _num(r[1]), "nloc": _num(r[0]), "params": _num(r[3]),
|
|
154
|
+
rows.append({"file": _rel(r[6]), "function": r[7] or "(anonymous)", "ccn": _num(r[1]), "nloc": _num(r[0]), "params": _num(r[3]),
|
|
155
155
|
"start": _num(r[9]), "end": _num(r[10])})
|
|
156
156
|
return rows
|
|
157
157
|
|
|
@@ -4,7 +4,9 @@
|
|
|
4
4
|
Standalone on purpose: gitmole runs it as a pipeline step with
|
|
5
5
|
`python3 maat.py LOG OUT_DIR [--aliases META_JSON]` and it must not need the
|
|
6
6
|
package on sys.path. Input is `git log --all --numstat --date=short
|
|
7
|
-
--pretty=format:--%h--%ad--%aN
|
|
7
|
+
--pretty=format:--%h--%ad--%aN -M`: renames are followed, so a moved file
|
|
8
|
+
is one entity under its new path and a pure move adds and deletes nothing.
|
|
9
|
+
Whoever moved a directory to src/ did not write it.
|
|
8
10
|
"""
|
|
9
11
|
from __future__ import annotations
|
|
10
12
|
|
|
@@ -40,13 +42,26 @@ def parse_log(text: str, aliases: dict = None, types=None) -> list:
|
|
|
40
42
|
commits.append(current)
|
|
41
43
|
elif line.strip() and current is not None:
|
|
42
44
|
added, deleted, path = line.split("\t", 2)
|
|
43
|
-
path = filetypes.unquote(path)
|
|
45
|
+
path = _renamed_to(filetypes.unquote(path))
|
|
44
46
|
if not filetypes.matches(path, types):
|
|
45
47
|
continue
|
|
46
48
|
current["files"].append((path, int(added) if added.isdigit() else 0, int(deleted) if deleted.isdigit() else 0))
|
|
47
49
|
return commits
|
|
48
50
|
|
|
49
51
|
|
|
52
|
+
_BRACED_RENAME = re.compile(r"\{([^{}]*) => ([^{}]*)\}")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _renamed_to(path: str) -> str:
|
|
56
|
+
"""The new path of a rename as `git log -M --numstat` spells it: `{old => new}/rest`,
|
|
57
|
+
`dir/{a => b}` or `old => new` for a whole path. A path without ' => ' is itself."""
|
|
58
|
+
if " => " not in path:
|
|
59
|
+
return path
|
|
60
|
+
if "{" in path:
|
|
61
|
+
return _BRACED_RENAME.sub(lambda m: m.group(2), path).replace("//", "/")
|
|
62
|
+
return path.split(" => ", 1)[1]
|
|
63
|
+
|
|
64
|
+
|
|
50
65
|
_FIX_CONVENTIONAL = re.compile(r"^(fix|hotfix|bugfix)(\([^)]*\))?!?:", re.I)
|
|
51
66
|
_FIX_WORDS = re.compile(r"\b(fix|fixes|fixed|fixing|bugfix|hotfix|bug|bugs|regression|crash|crashes)\b", re.I)
|
|
52
67
|
|
|
@@ -111,6 +111,24 @@ def _hide_vendor(rows: list, path_of, full, noun="file in vendored code", plural
|
|
|
111
111
|
return _hide_rows(rows, path_of, full, filetypes.is_vendor_path, noun, plural)
|
|
112
112
|
|
|
113
113
|
|
|
114
|
+
def _hide_generated(rows: list, path_of, report: dict, full, noun="generated file", plural=None) -> tuple:
|
|
115
|
+
"""Generated files (a header marker or a linguist-generated attribute, found at run time): the
|
|
116
|
+
generator's churn and complexity, not the repository's."""
|
|
117
|
+
generated = set((report.get("meta") or {}).get("generated") or [])
|
|
118
|
+
return _hide_rows(rows, path_of, full, lambda p: p in generated, noun, plural)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _hide_release(pairs: list, full) -> tuple:
|
|
122
|
+
"""Coupled pairs where both files are release plumbing (version files, manifests, lock files,
|
|
123
|
+
changelogs): they change together because a release touches them all, not because one depends on
|
|
124
|
+
the other. A version file paired with real code stays."""
|
|
125
|
+
if full is True:
|
|
126
|
+
return pairs, None
|
|
127
|
+
kept = [p for p in pairs if not (filetypes.is_release_path(p["entity"]) and filetypes.is_release_path(p["coupled"]))]
|
|
128
|
+
hidden = len(pairs) - len(kept)
|
|
129
|
+
return kept, (f"{hidden} release pair{'s' if hidden != 1 else ''} hidden{HIDDEN_SUFFIX}" if hidden else None)
|
|
130
|
+
|
|
131
|
+
|
|
114
132
|
def _join_hidden(*notes) -> str:
|
|
115
133
|
"""Several hidden-row notes as one caption phrase: 'A hidden; B hidden; --full shows them'."""
|
|
116
134
|
parts = [n[:-len(HIDDEN_SUFFIX)] if n.endswith(HIDDEN_SUFFIX) else n for n in notes if n]
|
|
@@ -376,7 +394,9 @@ def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
376
394
|
scored = hotspots.ranked(report)
|
|
377
395
|
scored, hidden_note = _hide_tests(scored, lambda h: h["entity"], full)
|
|
378
396
|
scored, deleted_note = _hide_deleted(scored, report, full)
|
|
379
|
-
|
|
397
|
+
scored, generated_note = _hide_generated(scored, lambda h: h["entity"], report, full)
|
|
398
|
+
scored, release_note = _hide_rows(scored, lambda h: h["entity"], full, filetypes.is_release_path, "release file")
|
|
399
|
+
hidden_note = _join_hidden(hidden_note, deleted_note, generated_note, release_note)
|
|
380
400
|
title = "Hotspots (score = revisions × lines of code)" if full is True else "Hotspots"
|
|
381
401
|
limit = _limit("Hotspots", full)
|
|
382
402
|
series = (report.get("trend") or {}).get("files") or {}
|
|
@@ -408,6 +428,8 @@ def coupling_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
408
428
|
pairs = sorted((p for p in report.get("coupling") or [] if p["average-revs"] >= 5), key=lambda p: (-p["degree"], -p["average-revs"]))
|
|
409
429
|
pairs, hidden_note = _hide_tests(pairs, lambda p: (p["entity"], p["coupled"]), full, noun="test pair")
|
|
410
430
|
pairs, gone_note = _hide_gone(pairs, report, full)
|
|
431
|
+
pairs, release_note = _hide_release(pairs, full)
|
|
432
|
+
gone_note = _join_hidden(gone_note, release_note)
|
|
411
433
|
groups, cluster_note = [], None
|
|
412
434
|
if full is not True:
|
|
413
435
|
# a directory whose files all change together is one row; --full lists every pair
|
|
@@ -474,7 +496,8 @@ def functions_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
474
496
|
funcs = sorted((f for f in measured if f["ccn"] >= CCN_FLOOR), key=lambda f: (-f["ccn"], -f["nloc"], f["file"], f["function"], f["start"]))
|
|
475
497
|
funcs, hidden_note = _hide_tests(funcs, lambda f: f["file"], full, noun="function in a test file", plural="functions in test files")
|
|
476
498
|
funcs, vendor_note = _hide_vendor(funcs, lambda f: f["file"], full, noun="function in vendored code", plural="functions in vendored code")
|
|
477
|
-
|
|
499
|
+
funcs, generated_note = _hide_generated(funcs, lambda f: f["file"], report, full, noun="function in a generated file", plural="functions in generated files")
|
|
500
|
+
hidden_note = _join_hidden(hidden_note, vendor_note, generated_note)
|
|
478
501
|
limit = _limit("Complex functions", full)
|
|
479
502
|
rows = [(f["function"], f["file"], f["ccn"], f["nloc"], f["params"]) for f in funcs[:limit]]
|
|
480
503
|
columns = [("function", {"overflow": "fold"}), ("file", PATH), ("ccn", RIGHT), ("lines", RIGHT), ("params", RIGHT)]
|
|
@@ -172,7 +172,7 @@ def plan(repo_dir: str, out_dir: str, branch: str = "HEAD", age: bool = True, pl
|
|
|
172
172
|
{"name": "scc", "argv": ["scc", "--by-file", "--format", "json"], "stdout": o("size.json"), "deps": []},
|
|
173
173
|
{"name": "git-sizer", "argv": ["git-sizer", "--verbose"], "stdout": o("repo-health.txt"), "deps": []},
|
|
174
174
|
{"name": "betterleaks", "argv": [sys.executable, LEAKS_SCRIPT, o("secrets.json")], "stdout": None, "deps": []}, # hashes the values before anything is written
|
|
175
|
-
{"name": "git-log", "argv": [*filetypes.GIT, "log", "--all", "--use-mailmap", "--numstat", "--date=iso-strict", "--pretty=format:--%h--%ad--%aN--%s", "
|
|
175
|
+
{"name": "git-log", "argv": [*filetypes.GIT, "log", "--all", "--use-mailmap", "--numstat", "--date=iso-strict", "--pretty=format:--%h--%ad--%aN--%s", "-M"], "stdout": log, "deps": []}, # -M: a move is not an edit
|
|
176
176
|
{"name": "change analysis", "argv": [sys.executable, MAAT_SCRIPT, log, out_dir, *type_args, *(["--now", now] if now else []), *(["--since", since] if since else []), "--aliases", o("meta.json")], "stdout": None, "deps": ["git-log"]},
|
|
177
177
|
]
|
|
178
178
|
if lizard:
|
|
@@ -66,8 +66,8 @@ def risks(report: dict, min_revs: int = 2) -> list:
|
|
|
66
66
|
|
|
67
67
|
rows = []
|
|
68
68
|
for h in hotspots.ranked(report):
|
|
69
|
-
if h["code"] is None or h["revs"] < min_revs or filetypes.is_test_path(h["entity"]):
|
|
70
|
-
continue
|
|
69
|
+
if h["code"] is None or h["revs"] < min_revs or filetypes.is_test_path(h["entity"]) or filetypes.is_release_path(h["entity"]):
|
|
70
|
+
continue # a version file or a manifest changes on every release, not where the next bug lands
|
|
71
71
|
fx = fixes.get(h["entity"], {})
|
|
72
72
|
own = owners.get(h["entity"]) or Counter()
|
|
73
73
|
owner, owner_lines = (own.most_common(1)[0] if own else (None, 0))
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.7
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -592,6 +592,28 @@ class Arguments(unittest.TestCase):
|
|
|
592
592
|
self.assertNotIn("jar", text)
|
|
593
593
|
|
|
594
594
|
|
|
595
|
+
class GeneratedFiles(unittest.TestCase):
|
|
596
|
+
def test_a_run_records_the_generated_files_in_meta(self):
|
|
597
|
+
with tempfile.TemporaryDirectory() as d:
|
|
598
|
+
_tiny_repo(d)
|
|
599
|
+
os.makedirs(os.path.join(d, "lib"))
|
|
600
|
+
with open(os.path.join(d, "lib", "validator.js"), "w") as fh:
|
|
601
|
+
fh.write("// This file is autogenerated by build/build.js, do not edit\nmodule.exports = 1\n")
|
|
602
|
+
with open(os.path.join(d, "lib", "app.js"), "w") as fh:
|
|
603
|
+
fh.write("module.exports = 2\n")
|
|
604
|
+
import subprocess
|
|
605
|
+
subprocess.run(["git", "-C", d, "add", "-A"], check=True)
|
|
606
|
+
subprocess.run(["git", "-C", d, "-c", "user.name=T", "-c", "user.email=t@x.com", "commit", "-q", "-m", "files"], check=True)
|
|
607
|
+
out = os.path.join(d, "out")
|
|
608
|
+
planner = lambda repo, o, branch="HEAD", **kw: [{"name": "q", "argv": ["true"], "stdout": None, "deps": []}]
|
|
609
|
+
rc = cli.main([d, "--out", out], console=console(), tool_check=lambda **kw: [], planner=planner,
|
|
610
|
+
estimator=lambda repo, interval, **kw: {"files": 2, "samples": 1, "blames": 2})
|
|
611
|
+
with open(os.path.join(out, "meta.json")) as fh:
|
|
612
|
+
meta = json.load(fh)
|
|
613
|
+
self.assertEqual(rc, 0)
|
|
614
|
+
self.assertEqual(meta["generated"], ["lib/validator.js"])
|
|
615
|
+
|
|
616
|
+
|
|
595
617
|
class Clean(unittest.TestCase):
|
|
596
618
|
"""--clean lists what gitmole left behind and deletes on a yes. TMPDIR is pointed at a scratch dir so the
|
|
597
619
|
real temp folder is never listed or touched."""
|
|
@@ -104,11 +104,39 @@ class TestPaths(unittest.TestCase):
|
|
|
104
104
|
for path in ("app/settings.py", "examplesite/app.py", "src/rulesets/a.go", "sampler/x.py", "config/betterleaks.toml"):
|
|
105
105
|
self.assertFalse(filetypes.is_sample_path(path), path)
|
|
106
106
|
|
|
107
|
+
def test_release_plumbing_files(self):
|
|
108
|
+
for path in ("lib/sinatra/version.rb", "VERSION", "src/pkg/__version__.py", "package.json", "package-lock.json", "Gemfile.lock",
|
|
109
|
+
"Cargo.toml", "pyproject.toml", "go.sum", "CHANGELOG.md", "CHANGES.rst", "sinatra.gemspec", "uv.lock"):
|
|
110
|
+
self.assertTrue(filetypes.is_release_path(path), path)
|
|
111
|
+
for path in ("lib/version_check.py", "src/app.py", "docs/versions.md", "Makefile", "lib/sinatra/base.rb"):
|
|
112
|
+
self.assertFalse(filetypes.is_release_path(path), path)
|
|
113
|
+
|
|
114
|
+
def test_generated_files_by_header_marker_or_attribute(self):
|
|
115
|
+
with tempfile.TemporaryDirectory() as d:
|
|
116
|
+
files = {
|
|
117
|
+
"lib/config-validator.js": "// This file is autogenerated by build/build-validation.js, do not edit\n'use strict'\n",
|
|
118
|
+
"pb/api.pb.go": "// Code generated by protoc-gen-go. DO NOT EDIT.\npackage pb\n",
|
|
119
|
+
"gen/schema.py": "# @generated\nx = 1\n",
|
|
120
|
+
"src/app.py": "# the app; it generated reports once\ndef main():\n pass\n",
|
|
121
|
+
"docs/notes.md": "generated notes are the best notes\n",
|
|
122
|
+
"dist/bundle.js": "var a = 1;\n",
|
|
123
|
+
"src/late.py": "\n" * 10 + "# generated by hand, do not edit\n", # a marker past the first lines does not count
|
|
124
|
+
}
|
|
125
|
+
for path, text in files.items():
|
|
126
|
+
os.makedirs(os.path.join(d, os.path.dirname(path)), exist_ok=True)
|
|
127
|
+
with open(os.path.join(d, path), "w") as fh:
|
|
128
|
+
fh.write(text)
|
|
129
|
+
with open(os.path.join(d, ".gitattributes"), "w") as fh:
|
|
130
|
+
fh.write("* text=auto\ndist/* linguist-generated=true\n*.min.js linguist-generated\n")
|
|
131
|
+
found = filetypes.generated_files(d, sorted(files))
|
|
132
|
+
self.assertEqual(found, ["dist/bundle.js", "gen/schema.py", "lib/config-validator.js", "pb/api.pb.go"])
|
|
133
|
+
|
|
107
134
|
def test_vendored_trees(self):
|
|
108
135
|
for path in ("vendor/github.com/x/y.go", "web/node_modules/a/index.js", "third_party/z/a.c", "thirdparty/a.c", "_vendor/a.py",
|
|
109
|
-
"external/lib/a.cpp"):
|
|
136
|
+
"external/lib/a.cpp", "requests/packages/urllib3/a.py", "pip/_vendor/six.py", "botocore/vendored/requests/a.py"):
|
|
110
137
|
self.assertTrue(filetypes.is_vendor_path(path), path)
|
|
111
|
-
for path in ("vendors.py", "src/vendoring/a.py", "node/a.js", "externals.txt", "app/main.go"
|
|
138
|
+
for path in ("vendors.py", "src/vendoring/a.py", "node/a.js", "externals.txt", "app/main.go",
|
|
139
|
+
"packages/runtime-core/src/renderer.ts", "packages-private/x.ts"): # a monorepo's own packages/ at the root
|
|
112
140
|
self.assertFalse(filetypes.is_vendor_path(path), path)
|
|
113
141
|
|
|
114
142
|
|
|
@@ -47,7 +47,7 @@ class SecretsFound(unittest.TestCase):
|
|
|
47
47
|
self.assertIn("1 distinct value in 2 places: generic-api-key in app/settings.py (c1, c2)", crit["detail"])
|
|
48
48
|
self.assertIn("Rotate", crit["advice"])
|
|
49
49
|
self.assertIn(".betterleaksignore", crit["advice"])
|
|
50
|
-
self.assertEqual(warn["title"], "2 secret(s) only in test, example or documentation files")
|
|
50
|
+
self.assertEqual(warn["title"], "2 secret(s) only in test, example, vendored or documentation files")
|
|
51
51
|
self.assertIn("2 distinct values in 3 places", warn["detail"])
|
|
52
52
|
self.assertIn("tests/data/a.html and 1 other file", warn["detail"])
|
|
53
53
|
self.assertIn(".betterleaksignore", warn["advice"])
|
|
@@ -62,15 +62,22 @@ class SecretsFound(unittest.TestCase):
|
|
|
62
62
|
self.row("h3", "pkg/testdata/creds.yaml", "c3")])
|
|
63
63
|
f = findings.secrets_found(r)
|
|
64
64
|
self.assertEqual([x["severity"] for x in f], ["warning"])
|
|
65
|
-
self.assertEqual(f[0]["title"], "3 secret(s) only in test, example or documentation files")
|
|
65
|
+
self.assertEqual(f[0]["title"], "3 secret(s) only in test, example, vendored or documentation files")
|
|
66
66
|
r = report(secrets=[self.row("h1", "examples/app.py", "c1"), self.row("h1", "app/config.py", "c2")])
|
|
67
67
|
self.assertEqual([x["severity"] for x in findings.secrets_found(r)], ["critical"], "the same value in source is a leak")
|
|
68
68
|
|
|
69
|
+
def test_a_value_only_in_vendored_code_is_a_warning(self):
|
|
70
|
+
# oauthlib's RFC test vectors inside requests/packages/: upstream's specimen, not this repository's credential
|
|
71
|
+
r = report(secrets=[self.row("h1", "requests/packages/oauthlib/oauth1/rfc5849/parameters.py", "9576518")])
|
|
72
|
+
f = findings.secrets_found(r)
|
|
73
|
+
self.assertEqual([x["severity"] for x in f], ["warning"])
|
|
74
|
+
self.assertIn("vendored", f[0]["title"])
|
|
75
|
+
|
|
69
76
|
def test_a_value_only_in_documentation_is_a_warning_that_says_template(self):
|
|
70
77
|
r = report(secrets=[self.row("h1", "docs/GA4-API-INTEGRATION.md", "e8c0508")])
|
|
71
78
|
f = findings.secrets_found(r)
|
|
72
79
|
self.assertEqual([x["severity"] for x in f], ["warning"])
|
|
73
|
-
self.assertEqual(f[0]["title"], "1 secret(s) only in test, example or documentation files")
|
|
80
|
+
self.assertEqual(f[0]["title"], "1 secret(s) only in test, example, vendored or documentation files")
|
|
74
81
|
self.assertIn("fixtures or templates", f[0]["advice"])
|
|
75
82
|
r = report(secrets=[self.row("h1", "docs/GA4-API-INTEGRATION.md", "e8c0508"), self.row("h1", "app/config.py", "c2")])
|
|
76
83
|
self.assertEqual([x["severity"] for x in findings.secrets_found(r)], ["critical"], "the same value in source is a leak")
|
|
@@ -254,6 +261,15 @@ class TightCoupling(unittest.TestCase):
|
|
|
254
261
|
f = findings.tight_coupling(report(coupling=pairs))
|
|
255
262
|
self.assertIn("2 pairs", f[0]["detail"], "without a tree listing every pair counts")
|
|
256
263
|
|
|
264
|
+
def test_release_plumbing_pairs_are_not_a_dependency(self):
|
|
265
|
+
pairs = [{"entity": "lib/sinatra/version.rb", "coupled": "rack-protection/lib/rack/protection/version.rb", "degree": 100, "average-revs": 60},
|
|
266
|
+
{"entity": "package.json", "coupled": "package-lock.json", "degree": 95, "average-revs": 40},
|
|
267
|
+
{"entity": "lib/sinatra/version.rb", "coupled": "lib/sinatra/base.rb", "degree": 85, "average-revs": 10}]
|
|
268
|
+
f = findings.tight_coupling(report(coupling=pairs))
|
|
269
|
+
self.assertIn("1 pair changes together", f[0]["detail"], "a version file paired with real code still counts")
|
|
270
|
+
self.assertNotIn("package.json", f[0]["detail"])
|
|
271
|
+
self.assertEqual(findings.tight_coupling(report(coupling=pairs[:2])), [])
|
|
272
|
+
|
|
257
273
|
def test_single_pair_reads_grammatically(self):
|
|
258
274
|
pairs = [{"entity": "a", "coupled": "b", "degree": 100, "average-revs": 10}]
|
|
259
275
|
f = findings.tight_coupling(report(coupling=pairs))
|
|
@@ -274,6 +290,21 @@ class TightCoupling(unittest.TestCase):
|
|
|
274
290
|
self.assertEqual(findings.tight_coupling(report()), [])
|
|
275
291
|
|
|
276
292
|
|
|
293
|
+
class ReleasePlumbing(unittest.TestCase):
|
|
294
|
+
def test_a_version_file_or_manifest_does_not_dominate_the_churn(self):
|
|
295
|
+
revs = [{"entity": "setup.py", "n-revs": 184}, {"entity": "requests/models.py", "n-revs": 60}, {"entity": "requests/api.py", "n-revs": 20}]
|
|
296
|
+
f = findings.hotspot_dominance(report(revisions=revs))
|
|
297
|
+
self.assertIn("requests/models.py changed 60 times", f[0]["detail"])
|
|
298
|
+
self.assertNotIn("setup.py", f[0]["detail"])
|
|
299
|
+
|
|
300
|
+
def test_a_manifest_is_not_a_bug_magnet(self):
|
|
301
|
+
fixes = [{"entity": "package.json", "n-fixes": 20, "last-fix": "2026-09-01", "recent-fixes": 6},
|
|
302
|
+
{"entity": "lib/reply.js", "n-fixes": 10, "last-fix": "2026-09-01", "recent-fixes": 4}]
|
|
303
|
+
f = findings.bug_magnets(report(fixes=fixes))
|
|
304
|
+
self.assertIn("lib/reply.js", f[0]["detail"])
|
|
305
|
+
self.assertNotIn("package.json", f[0]["detail"])
|
|
306
|
+
|
|
307
|
+
|
|
277
308
|
class StaleFiles(unittest.TestCase):
|
|
278
309
|
def test_info_when_a_third_untouched_for_a_year(self):
|
|
279
310
|
age = [{"entity": f"f{i}", "age-months": 12} for i in range(4)] + [{"entity": "g", "age-months": 0} for _ in range(6)]
|
|
@@ -374,6 +405,21 @@ class BrainMethods(unittest.TestCase):
|
|
|
374
405
|
self.assertNotIn("test_all", f[0]["detail"])
|
|
375
406
|
self.assertEqual(findings.brain_methods(report(functions=fns[:1])), [])
|
|
376
407
|
|
|
408
|
+
def test_an_anonymous_function_is_named_by_its_place(self):
|
|
409
|
+
fns = [{"file": "completions.go", "function": "(anonymous)", "ccn": 47, "nloc": 136, "params": 1, "start": 316, "end": 585}]
|
|
410
|
+
f = findings.brain_methods(report(functions=fns))
|
|
411
|
+
self.assertIn("(anonymous) (completions.go:316) complexity 47, 136 lines, 1 params", f[0]["detail"])
|
|
412
|
+
self.assertEqual(f[0]["advice"], "Split the anonymous function at completions.go:316 first, before the next change lands there.")
|
|
413
|
+
|
|
414
|
+
def test_generated_files_are_not_brain_methods(self):
|
|
415
|
+
fns = [{"file": "lib/config-validator.js", "function": "validate10", "ccn": 373, "nloc": 1150, "params": 5, "start": 1, "end": 1150},
|
|
416
|
+
{"file": "lib/reply.js", "function": "onSendEnd", "ccn": 34, "nloc": 180, "params": 2, "start": 1, "end": 180}]
|
|
417
|
+
r = report(functions=fns)
|
|
418
|
+
r["meta"]["generated"] = ["lib/config-validator.js"]
|
|
419
|
+
f = findings.brain_methods(r)
|
|
420
|
+
self.assertEqual(f[0]["advice"], "Split onSendEnd in lib/reply.js first, before the next change lands there.")
|
|
421
|
+
self.assertNotIn("validate10", f[0]["detail"])
|
|
422
|
+
|
|
377
423
|
def test_vendored_functions_are_not_brain_methods(self):
|
|
378
424
|
fns = [{"file": "vendor/github.com/google/jsonschema-go/jsonschema/validate.go", "function": "validate", "ccn": 179, "nloc": 424, "params": 3, "start": 1, "end": 424},
|
|
379
425
|
{"file": "processor/workers.go", "function": "countLoopGeneric", "ccn": 56, "nloc": 164, "params": 8, "start": 1, "end": 164}]
|
|
@@ -18,6 +18,19 @@ class Merge(unittest.TestCase):
|
|
|
18
18
|
names = [m["name"] for m in merged]
|
|
19
19
|
self.assertEqual(names, ["Bob", "Grzegorz Bankosz", "Ann"])
|
|
20
20
|
|
|
21
|
+
def test_an_identical_handle_under_several_emails_is_one_person(self):
|
|
22
|
+
ids = [{"name": "KaKa", "email": "kaka@a.com", "commits": 57}, {"name": "KaKa", "email": "23028015+climba@users.noreply.github.com", "commits": 56},
|
|
23
|
+
{"name": "kaka", "email": "climba@b.com", "commits": 10}, {"name": "namusyaka", "email": "n@a.com", "commits": 180},
|
|
24
|
+
{"name": "namusyaka", "email": "n@b.com", "commits": 8}, {"name": "Li Yu", "email": "li@a.com", "commits": 5},
|
|
25
|
+
{"name": "Li Yu", "email": "li@b.com", "commits": 3}]
|
|
26
|
+
merged = {m["name"]: m["commits"] for m in identity.merge(ids)}
|
|
27
|
+
self.assertEqual(merged, {"KaKa": 123, "namusyaka": 188, "Li Yu": 8})
|
|
28
|
+
|
|
29
|
+
def test_a_bare_common_first_name_is_not_enough(self):
|
|
30
|
+
ids = [{"name": "Jean", "email": "jean@a.com", "commits": 24}, {"name": "Jean", "email": "jean@b.com", "commits": 18},
|
|
31
|
+
{"name": "Alex", "email": "alex@a.com", "commits": 3}, {"name": "alex", "email": "alex@b.com", "commits": 2}]
|
|
32
|
+
self.assertEqual(len(identity.merge(ids)), 4, "two Jeans and two Alexes may be four people")
|
|
33
|
+
|
|
21
34
|
def test_merged_row_sums_commits_and_lists_aliases(self):
|
|
22
35
|
merged = {m["name"]: m for m in identity.merge(IDS)}
|
|
23
36
|
self.assertEqual(merged["Grzegorz Bankosz"]["commits"], 41)
|
|
@@ -138,6 +138,11 @@ class ParseFunctions(unittest.TestCase):
|
|
|
138
138
|
def test_empty(self):
|
|
139
139
|
self.assertEqual(load.parse_functions(""), [])
|
|
140
140
|
|
|
141
|
+
def test_a_nameless_function_is_called_anonymous(self):
|
|
142
|
+
# lizard names Go function literals with an empty string where it names JavaScript's "(anonymous)"
|
|
143
|
+
rows = load.parse_functions('136,47,926,1,270,"@316-585@completions.go","completions.go",""," c * Command",316,585\n')
|
|
144
|
+
self.assertEqual((rows[0]["function"], rows[0]["start"]), ("(anonymous)", 316))
|
|
145
|
+
|
|
141
146
|
def test_a_row_cut_short_by_a_killed_step_does_not_abort_the_report(self):
|
|
142
147
|
rows = load.parse_functions(self.CSV + '5,3,40,1,5,"g@1-5@a.py","a.py","g","g( )",1,\n')
|
|
143
148
|
self.assertEqual(len(rows), 3)
|
|
@@ -48,6 +48,18 @@ class ParseLog(unittest.TestCase):
|
|
|
48
48
|
self.assertEqual(commits[0]["files"], [("src/a.py", 3, 1), ("src/b.py", 2, 0)])
|
|
49
49
|
self.assertEqual(commits[1]["files"][2], ("img/logo.png", 0, 0))
|
|
50
50
|
|
|
51
|
+
def test_renames_are_followed_to_the_new_path_and_a_pure_move_adds_no_lines(self):
|
|
52
|
+
# `git log -M --numstat` spells a rename three ways; the mover is not the owner of what moved
|
|
53
|
+
log = ("--d63e94f5--2023-08-13T10:00:00+00:00--Nate--Move to src layout\n"
|
|
54
|
+
"0\t0\t{requests => src/requests}/__init__.py\n"
|
|
55
|
+
"4\t7\tSECURITY.md => .github/SECURITY.md\n"
|
|
56
|
+
"0\t0\tCODE_OF_CONDUCT.md => .github/CODE_OF_CONDUCT.md\n"
|
|
57
|
+
"2\t0\tsrc/requests/{models.py => models_v2.py}\n"
|
|
58
|
+
"1\t1\tMakefile\n")
|
|
59
|
+
commits = maat.parse_log(log, types=None)
|
|
60
|
+
self.assertEqual(commits[0]["files"], [("src/requests/__init__.py", 0, 0), (".github/SECURITY.md", 4, 7),
|
|
61
|
+
(".github/CODE_OF_CONDUCT.md", 0, 0), ("src/requests/models_v2.py", 2, 0), ("Makefile", 1, 1)])
|
|
62
|
+
|
|
51
63
|
def test_subjects_with_exotic_line_break_characters_do_not_split_the_log(self):
|
|
52
64
|
# U+2028 and form feed are line breaks to str.splitlines but not to git
|
|
53
65
|
text = "--x--2026-05-04T10:00:00+00:00--Ann--Fix\u2028broken\x0cthing\n1\t0\tf.py\n"
|
|
@@ -318,6 +318,47 @@ class Report(unittest.TestCase):
|
|
|
318
318
|
full = _section_text(rendered(r, [], width=200, full=True), "Complex functions")
|
|
319
319
|
self.assertIn("vendor/github.com/x/y.go", full)
|
|
320
320
|
|
|
321
|
+
def test_default_tables_hide_generated_files_and_say_so(self):
|
|
322
|
+
r = sample_report()
|
|
323
|
+
r["meta"]["generated"] = ["lib/config-validator.js"]
|
|
324
|
+
r["size"]["files"]["lib/config-validator.js"] = {"code": 1153, "complexity": 373}
|
|
325
|
+
r["revisions"].append({"entity": "lib/config-validator.js", "n-revs": 8})
|
|
326
|
+
r["functions"].append({"file": "lib/config-validator.js", "function": "validate10", "ccn": 373, "nloc": 1150, "params": 5, "start": 1, "end": 1150})
|
|
327
|
+
text = rendered(r, [], width=200)
|
|
328
|
+
hot = text[text.index("◆ Hotspots"):text.index("Change coupling")]
|
|
329
|
+
self.assertNotIn("config-validator", hot)
|
|
330
|
+
self.assertIn("1 generated file hidden; --full shows them", hot)
|
|
331
|
+
fn = _section_text(text, "Complex functions")
|
|
332
|
+
self.assertNotIn("validate10", fn)
|
|
333
|
+
self.assertIn("1 function in a generated file hidden; --full shows them", fn)
|
|
334
|
+
full = rendered(r, [], width=200, full=True)
|
|
335
|
+
self.assertIn("validate10", full)
|
|
336
|
+
|
|
337
|
+
def test_default_hotspots_hide_release_plumbing_and_say_so(self):
|
|
338
|
+
r = sample_report()
|
|
339
|
+
r["size"]["files"].update({"setup.py": {"code": 6, "complexity": 0}, "version.go": {"code": 2, "complexity": 0}})
|
|
340
|
+
r["revisions"] += [{"entity": "setup.py", "n-revs": 184}, {"entity": "version.go", "n-revs": 29}]
|
|
341
|
+
hot = rendered(r, [], width=200)
|
|
342
|
+
hot = hot[hot.index("◆ Hotspots"):hot.index("Change coupling")]
|
|
343
|
+
self.assertNotIn("setup.py", hot)
|
|
344
|
+
self.assertIn("2 release files hidden; --full shows them", hot)
|
|
345
|
+
full = rendered(r, [], width=200, full=True)
|
|
346
|
+
self.assertIn("setup.py", full[full.index("◆ Hotspots"):])
|
|
347
|
+
|
|
348
|
+
def test_default_coupling_hides_release_plumbing_pairs_and_says_so(self):
|
|
349
|
+
r = sample_report()
|
|
350
|
+
for f in ("lib/version.rb", "contrib/version.rb", "Gemfile", "Gemfile.lock"):
|
|
351
|
+
r["size"]["files"][f] = {"code": 3, "complexity": 0}
|
|
352
|
+
r["coupling"] = [{"entity": "lib/version.rb", "coupled": "contrib/version.rb", "degree": 64, "average-revs": 60},
|
|
353
|
+
{"entity": "Gemfile", "coupled": "Gemfile.lock", "degree": 90, "average-revs": 20},
|
|
354
|
+
{"entity": "static/index.html", "coupled": "static/apps-metadata.json", "degree": 90, "average-revs": 11}]
|
|
355
|
+
coupling = _section_text(rendered(r, [], width=200), "Change coupling")
|
|
356
|
+
self.assertIn("static/index.html", coupling)
|
|
357
|
+
self.assertNotIn("version.rb", coupling)
|
|
358
|
+
self.assertIn("2 release pairs hidden; --full shows them", coupling)
|
|
359
|
+
full = _section_text(rendered(r, [], width=200, full=True), "Change coupling")
|
|
360
|
+
self.assertIn("version.rb", full)
|
|
361
|
+
|
|
321
362
|
def test_hotspots_with_only_test_files_say_what_was_hidden(self):
|
|
322
363
|
r = sample_report()
|
|
323
364
|
r["revisions"] = [{"entity": "tests/test_a.py", "n-revs": 200}]
|
|
@@ -192,6 +192,8 @@ class Plan(unittest.TestCase):
|
|
|
192
192
|
self.assertEqual(by["scc"]["stdout"], "/o/size.json")
|
|
193
193
|
self.assertIn("--by-file", by["scc"]["argv"])
|
|
194
194
|
self.assertIn("--use-mailmap", by["git-log"]["argv"])
|
|
195
|
+
self.assertIn("-M", by["git-log"]["argv"], "renames are followed so a move to src/ credits nobody with the moved lines")
|
|
196
|
+
self.assertNotIn("--no-renames", by["git-log"]["argv"])
|
|
195
197
|
self.assertEqual(by["git-log"]["argv"][:4], ["git", "-c", "core.quotePath=false", "log"], "non-ASCII paths must not be octal-escaped and quoted")
|
|
196
198
|
|
|
197
199
|
def test_function_metrics_step_is_optional_and_runs_the_bundled_script(self):
|
|
@@ -53,6 +53,14 @@ class Risks(unittest.TestCase):
|
|
|
53
53
|
self.assertNotIn("core/gone.py", files, "no longer in the tree")
|
|
54
54
|
self.assertNotIn("core/once.py", files, "changed once")
|
|
55
55
|
|
|
56
|
+
def test_release_plumbing_is_not_on_the_list(self):
|
|
57
|
+
r = report()
|
|
58
|
+
r["size"]["files"].update({"setup.py": {"code": 6, "complexity": 0}, "version.go": {"code": 2, "complexity": 0}, "Makefile": {"code": 21, "complexity": 0}})
|
|
59
|
+
r["revisions"] = [{"entity": "setup.py", "n-revs": 184}, {"entity": "version.go", "n-revs": 29}, {"entity": "Makefile", "n-revs": 131},
|
|
60
|
+
{"entity": "core/parser.py", "n-revs": 40}]
|
|
61
|
+
files = sorted(x["file"] for x in watch.risks(r))
|
|
62
|
+
self.assertEqual(files, ["Makefile", "core/parser.py"], "a version file or a manifest changes on every release, not where the next bug lands")
|
|
63
|
+
|
|
56
64
|
def test_test_companions_and_weak_pairs_are_not_reasons(self):
|
|
57
65
|
top = watch.risks(report())[0]
|
|
58
66
|
coupling = [r for r in top["reasons"] if r.startswith("changes with")][0]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|