gitmole 0.6.8__tar.gz → 0.6.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. {gitmole-0.6.8 → gitmole-0.6.10}/PKG-INFO +2 -2
  2. {gitmole-0.6.8 → gitmole-0.6.10}/README.md +1 -1
  3. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/__init__.py +1 -1
  4. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/filetypes.py +18 -5
  5. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/findings.py +4 -2
  6. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/functions.py +12 -2
  7. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/identity.py +7 -1
  8. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/load.py +12 -2
  9. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/maat.py +14 -0
  10. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/render.py +2 -1
  11. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/run.py +15 -3
  12. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/watch.py +2 -1
  13. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole.egg-info/PKG-INFO +2 -2
  14. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_filetypes.py +7 -4
  15. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_findings.py +6 -0
  16. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_functions.py +11 -0
  17. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_identity.py +6 -0
  18. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_load.py +15 -0
  19. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_maat.py +26 -2
  20. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_render.py +10 -0
  21. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_run.py +27 -2
  22. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_watch.py +7 -0
  23. {gitmole-0.6.8 → gitmole-0.6.10}/LICENSE +0 -0
  24. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/__main__.py +0 -0
  25. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/backtest.py +0 -0
  26. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/banner.py +0 -0
  27. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/blame.py +0 -0
  28. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/clean.py +0 -0
  29. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/cli.py +0 -0
  30. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/coupling.py +0 -0
  31. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/hotspots.py +0 -0
  32. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/knowledge.py +0 -0
  33. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/leaks.py +0 -0
  34. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/loss.py +0 -0
  35. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/textfmt.py +0 -0
  36. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole/trend.py +0 -0
  37. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole.egg-info/SOURCES.txt +0 -0
  38. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole.egg-info/dependency_links.txt +0 -0
  39. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole.egg-info/entry_points.txt +0 -0
  40. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole.egg-info/requires.txt +0 -0
  41. {gitmole-0.6.8 → gitmole-0.6.10}/gitmole.egg-info/top_level.txt +0 -0
  42. {gitmole-0.6.8 → gitmole-0.6.10}/pyproject.toml +0 -0
  43. {gitmole-0.6.8 → gitmole-0.6.10}/setup.cfg +0 -0
  44. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_backtest.py +0 -0
  45. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_banner.py +0 -0
  46. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_blame.py +0 -0
  47. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_clean.py +0 -0
  48. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_cli.py +0 -0
  49. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_coupling.py +0 -0
  50. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_golden.py +0 -0
  51. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_hotspots.py +0 -0
  52. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_knowledge.py +0 -0
  53. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_leaks.py +0 -0
  54. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_loss.py +0 -0
  55. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_packaging.py +0 -0
  56. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_textfmt.py +0 -0
  57. {gitmole-0.6.8 → gitmole-0.6.10}/tests/test_trend.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.6.8
3
+ Version: 0.6.10
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -55,7 +55,7 @@ Linux package names, the release binaries, `--plots` and the pip caveats:
55
55
  ```bash
56
56
  gitmole . # the clone you are in
57
57
  gitmole /path/to/clone # any local clone
58
- gitmole owner/repo # clones with gh into a temp dir first
58
+ gitmole owner/repo # clones into a temp dir first, with gh or plain git
59
59
  gitmole 'owner/*' # every non-archived repo of a user or org, one summary table
60
60
 
61
61
  gitmole . --markdown report.md # the same report as a Markdown document
@@ -35,7 +35,7 @@ Linux package names, the release binaries, `--plots` and the pip caveats:
35
35
  ```bash
36
36
  gitmole . # the clone you are in
37
37
  gitmole /path/to/clone # any local clone
38
- gitmole owner/repo # clones with gh into a temp dir first
38
+ gitmole owner/repo # clones into a temp dir first, with gh or plain git
39
39
  gitmole 'owner/*' # every non-archived repo of a user or org, one summary table
40
40
 
41
41
  gitmole . --markdown report.md # the same report as a Markdown document
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.6.8"
3
+ __version__ = "0.6.10"
@@ -50,20 +50,22 @@ def parse(spec):
50
50
  return {t.strip().lstrip(".").lower() for t in spec.split(",") if t.strip()}
51
51
 
52
52
 
53
- _TEST_PATH = re.compile(r"(^|/)(tests?|spec|specs|__tests__|testing)(/|$)|(^|/)(test_[^/]*|[^/]*_test\.[^/]+|[^/]*\.spec\.[^/]+|[^/]*\.test\.[^/]+)$", re.I)
53
+ _TEST_PATH = re.compile(r"(^|/)(tests?|spec|specs|__tests__|testing|snapshots?|__snapshots__|[\w-]+[_-]tests?|tests?[_-][\w-]+)(/|$)"
54
+ r"|(^|/)(test_[^/]*|[^/]*_test\.[^/]+|[^/]*\.spec\.[^/]+|[^/]*\.test\.[^/]+|[^/]*\.snap)$", re.I)
54
55
 
55
56
 
56
57
  def is_test_path(path: str) -> bool:
57
- """A test file or anything under a tests directory: changes with every fix, so not a signal on its own."""
58
+ """A test file or anything under a tests directory (tests/, pending_tests/, e2e-tests/, test_utils/,
59
+ snapshots/ and .snap files): changes with every fix, so not a signal on its own."""
58
60
  return bool(_TEST_PATH.search(path))
59
61
 
60
62
 
61
- _DOC_PATH = re.compile(r"(^|/)docs?(/|$)|\.(md|markdown|rst|txt|adoc)$", re.I)
63
+ _DOC_PATH = re.compile(r"(^|/)docs?([-_][\w-]+)?(/|$)|\.(md|markdown|rst|txt|adoc)$", re.I)
62
64
 
63
65
 
64
66
  def is_doc_path(path: str) -> bool:
65
- """Documentation: prose formats anywhere, or anything under docs/. A key in a planning document
66
- is far more often a template than a leak."""
67
+ """Documentation: prose formats anywhere, or anything under docs/, doc/, docs_src/, docs-site/. A
68
+ key in a planning document or a tutorial is far more often a template than a leak."""
67
69
  return bool(_DOC_PATH.search(path))
68
70
 
69
71
 
@@ -98,6 +100,17 @@ def is_release_path(path: str) -> bool:
98
100
  return name in _RELEASE_NAMES or name.endswith(".gemspec") or name.startswith(("changelog", "changes.", "history.", "news."))
99
101
 
100
102
 
103
+ def plumbing_paths(report: dict) -> set:
104
+ """Files the change log showed to be release plumbing by behaviour rather than by name: nearly
105
+ every commit touching them changed a line or two (a version constant in __init__.py)."""
106
+ return {r["entity"] for r in report.get("plumbing") or []}
107
+
108
+
109
+ def is_release(path: str, plumbing=frozenset()) -> bool:
110
+ """is_release_path, or a path the change log showed to be plumbing (see plumbing_paths)."""
111
+ return is_release_path(path) or path in plumbing
112
+
113
+
101
114
  # What a generated file says about itself in its first lines: protoc, ajv, code generators of every kind.
102
115
  _GENERATED = re.compile(r"auto[- ]?generated|generated (by|from|file|code|automatically|with)|do not (edit|modify)|@generated|code generated", re.I)
103
116
  GENERATED_HEAD_LINES = 5
@@ -176,7 +176,8 @@ def sizer_concerns(report: dict) -> list:
176
176
  def hotspot_dominance(report: dict, ratio: float = 2.0, minimum: int = 20) -> list:
177
177
  """One source file takes most of the churn. Test files are left out: they change with everything.
178
178
  So is release plumbing: a version file or a manifest changes on every release by design."""
179
- revs = sorted((r for r in report.get("revisions") or [] if not (filetypes.is_test_path(r["entity"]) or filetypes.is_release_path(r["entity"]))),
179
+ plumb = filetypes.plumbing_paths(report)
180
+ revs = sorted((r for r in report.get("revisions") or [] if not (filetypes.is_test_path(r["entity"]) or filetypes.is_release(r["entity"], plumb))),
180
181
  key=lambda r: -r["n-revs"])
181
182
  if len(revs) < 2 or revs[0]["n-revs"] < minimum or revs[0]["n-revs"] < ratio * revs[1]["n-revs"]:
182
183
  return []
@@ -266,8 +267,9 @@ def stale_files(report: dict, months: int = 12, share: float = 0.3) -> list:
266
267
  def bug_magnets(report: dict, min_recent: int = 3, warn_at: int = 5) -> list:
267
268
  """Source files with a run of recent fix commits. Test files are left out: they change with every fix.
268
269
  So is release plumbing: a manifest touched by every fix release is not where the bug was."""
270
+ plumb = filetypes.plumbing_paths(report)
269
271
  hot = [f for f in report.get("fixes") or [] if f["recent-fixes"] >= min_recent
270
- and not (filetypes.is_test_path(f["entity"]) or filetypes.is_release_path(f["entity"]))]
272
+ and not (filetypes.is_test_path(f["entity"]) or filetypes.is_release(f["entity"], plumb))]
271
273
  if not hot:
272
274
  return []
273
275
  hot.sort(key=lambda f: (-f["recent-fixes"], -f["n-fixes"], f["entity"]))
@@ -30,10 +30,20 @@ def select_files(repo: str, ignore=(), types_spec: str = None) -> list:
30
30
  return [f for f in files if lizard.get_reader_for(f) is not None]
31
31
 
32
32
 
33
+ NAME_CAP = 200 # a deeply nested fixture gives lizard a dotted name of megabytes; nobody reads past this
34
+ LONG_NAME_CAP = 500
35
+
36
+
37
+ def _cut(text: str, cap: int) -> str:
38
+ return text if len(text) <= cap else text[:cap - 1] + "…"
39
+
40
+
33
41
  def csv_row(info, fn) -> list:
34
- """The columns `lizard --csv` prints, so the loader does not care which produced the file."""
42
+ """The columns `lizard --csv` prints, so the loader does not care which produced the file. Names
43
+ are cut to what a table can show, so one pathological fixture cannot make the file unreadable."""
44
+ name = _cut(fn.name, NAME_CAP)
35
45
  return [fn.nloc, fn.cyclomatic_complexity, fn.token_count, fn.parameter_count, fn.length,
36
- f"{fn.name}@{fn.start_line}-{fn.end_line}@{info.filename}", info.filename, fn.name, fn.long_name, fn.start_line, fn.end_line]
46
+ f"{name}@{fn.start_line}-{fn.end_line}@{info.filename}", info.filename, name, _cut(fn.long_name, LONG_NAME_CAP), fn.start_line, fn.end_line]
37
47
 
38
48
 
39
49
  def write_duplicates(dup: Duplicates, fh) -> None:
@@ -53,7 +53,13 @@ def same_person(a: dict, b: dict) -> bool:
53
53
  for handle, full in ((na, tb), (nb, ta)):
54
54
  if " " not in handle and handle in full and len(full) >= 2 and _distinctive(handle):
55
55
  return True
56
- return False
56
+ # RobinMalfait and Robin Malfait: the full name run together, six letters or more so it is not anyone
57
+ sa, sb = _squash(a["name"]), _squash(b["name"])
58
+ return bool(sa) and sa == sb and len(sa) >= 6 and (len(ta) >= 2 or len(tb) >= 2)
59
+
60
+
61
+ def _squash(name: str) -> str:
62
+ return re.sub(r"[^a-z0-9]", "", name.lower())
57
63
 
58
64
 
59
65
  def merge(identities: list) -> list:
@@ -3,6 +3,7 @@ from __future__ import annotations
3
3
 
4
4
  import csv
5
5
  import io
6
+ import sys
6
7
  import json
7
8
  import os
8
9
  import re
@@ -64,7 +65,7 @@ def parse_scc(text: str, types=None) -> dict:
64
65
  }
65
66
 
66
67
 
67
- NUMERIC_COLUMNS = {"n-revs", "degree", "average-revs", "n-authors", "age-months", "added", "deleted", "n-fixes", "recent-fixes"}
68
+ NUMERIC_COLUMNS = {"n-revs", "degree", "average-revs", "n-authors", "age-months", "added", "deleted", "n-fixes", "recent-fixes", "tiny-revs"}
68
69
 
69
70
 
70
71
  def parse_maat_csv(text: str) -> list:
@@ -145,13 +146,21 @@ def parse_authors_log(text: str) -> list:
145
146
  ]
146
147
 
147
148
 
149
+ NAME_CAP = 200 # a function name a table can show; deeply nested fixtures give lizard dotted names of megabytes
150
+ csv.field_size_limit(min(sys.maxsize, 2**31 - 1)) # an older functions.csv may still carry such a name
151
+
152
+
153
+ def _cut(name: str, cap: int = NAME_CAP) -> str:
154
+ return name if len(name) <= cap else name[:cap - 1] + "…"
155
+
156
+
148
157
  def parse_functions(text: str) -> list:
149
158
  """lizard --csv rows: nloc, ccn, tokens, params, length, location, file, function, long name, start, end."""
150
159
  rows = []
151
160
  for r in csv.reader(io.StringIO(text)):
152
161
  if len(r) < 11:
153
162
  continue
154
- rows.append({"file": _rel(r[6]), "function": r[7] or "(anonymous)", "ccn": _num(r[1]), "nloc": _num(r[0]), "params": _num(r[3]),
163
+ rows.append({"file": _rel(r[6]), "function": _cut(r[7]) or "(anonymous)", "ccn": _num(r[1]), "nloc": _num(r[0]), "params": _num(r[3]),
155
164
  "start": _num(r[9]), "end": _num(r[10])})
156
165
  return rows
157
166
 
@@ -235,6 +244,7 @@ def load_report(out_dir: str, nested: bool = True) -> dict:
235
244
  # was measured unfiltered, so it is re-rendered unfiltered rather than with a guessed list
236
245
  "size": parse_scc(_read(out_dir, "size.json"), filetypes.parse(meta["file_types"]) if "file_types" in meta else None),
237
246
  "revisions": parse_maat_csv(_read(out_dir, "maat-revisions.csv")),
247
+ "plumbing": parse_maat_csv(_read(out_dir, "maat-plumbing.csv")),
238
248
  "coupling": parse_maat_csv(_read(out_dir, "maat-coupling.csv")),
239
249
  "authors": parse_maat_csv(_read(out_dir, "maat-authors.csv")),
240
250
  "age": parse_maat_csv(_read(out_dir, "maat-age.csv")),
@@ -221,8 +221,22 @@ def activity(commits: list) -> dict:
221
221
  "reverted": dict(sorted(reverted.items(), key=lambda kv: (-kv[1], kv[0])))}
222
222
 
223
223
 
224
+ def plumbing(commits: list, min_revs: int = 20, share: float = 0.8, max_lines: int = 2) -> list:
225
+ """Files whose commits nearly always change a line or two: a version constant in __init__.py, a
226
+ date in a header. Their churn is the release cadence, not where the next bug lands. Needs enough
227
+ commits to judge by."""
228
+ revs, tiny = Counter(), Counter()
229
+ for c in commits:
230
+ for path, added, deleted in c["files"]:
231
+ revs[path] += 1
232
+ if added + deleted <= max_lines:
233
+ tiny[path] += 1
234
+ return [{"entity": e, "n-revs": n, "tiny-revs": tiny[e]} for e, n in sorted(revs.items()) if n >= min_revs and tiny[e] / n >= share]
235
+
236
+
224
237
  ANALYSES = {
225
238
  "revisions": (revisions, ["entity", "n-revs"]),
239
+ "plumbing": (plumbing, ["entity", "n-revs", "tiny-revs"]),
226
240
  "coupling": (coupling, ["entity", "coupled", "degree", "average-revs"]),
227
241
  "authors": (authors, ["entity", "n-authors", "n-revs"]),
228
242
  "age": (age, ["entity", "age-months"]),
@@ -395,7 +395,8 @@ def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
395
395
  scored, hidden_note = _hide_tests(scored, lambda h: h["entity"], full)
396
396
  scored, deleted_note = _hide_deleted(scored, report, full)
397
397
  scored, generated_note = _hide_generated(scored, lambda h: h["entity"], report, full)
398
- scored, release_note = _hide_rows(scored, lambda h: h["entity"], full, filetypes.is_release_path, "release file")
398
+ plumb = filetypes.plumbing_paths(report)
399
+ scored, release_note = _hide_rows(scored, lambda h: h["entity"], full, lambda p: filetypes.is_release(p, plumb), "release file")
399
400
  hidden_note = _join_hidden(hidden_note, deleted_note, generated_note, release_note)
400
401
  title = "Hotspots (score = revisions × lines of code)" if full is True else "Hotspots"
401
402
  limit = _limit("Hotspots", full)
@@ -96,10 +96,22 @@ def list_repos(owner: str, lister=_gh) -> list:
96
96
  return sorted(set(out.split()))
97
97
 
98
98
 
99
- def clone(target: str, dest_parent: str, runner=_gh) -> str:
100
- """Clone a remote target with gh (so private repos use the existing auth). Raises GhError."""
99
+ def clone_url(target: str) -> str:
100
+ """The URL plain git can clone: a URL as given, owner/repo on github.com."""
101
+ return target if _URL.match(target) else f"https://github.com/{target}.git"
102
+
103
+
104
+ def clone(target: str, dest_parent: str, runner=_gh, git_runner=_gh) -> str:
105
+ """Clone a remote target with gh (so private repos use the existing auth), or with plain git when
106
+ gh is missing or fails: a public repository needs no token. Raises GhError naming both failures."""
101
107
  dest = os.path.join(dest_parent, repo_name(target))
102
- _wrap(runner, ["gh", "repo", "clone", target, dest, "--", "--quiet"])
108
+ try:
109
+ _wrap(runner, ["gh", "repo", "clone", target, dest, "--", "--quiet"])
110
+ except GhError as gh_error:
111
+ try:
112
+ _wrap(git_runner, [*filetypes.GIT, "clone", "--quiet", clone_url(target), dest])
113
+ except GhError as git_error:
114
+ raise GhError(f"{gh_error}; git clone also failed: {git_error}") from None
103
115
  return dest
104
116
 
105
117
 
@@ -64,9 +64,10 @@ def risks(report: dict, min_revs: int = 2) -> list:
64
64
  fixes = {f["entity"]: f for f in report.get("fixes") or []}
65
65
  n_authors = {a["entity"]: a["n-authors"] for a in report.get("authors") or []}
66
66
 
67
+ plumb = filetypes.plumbing_paths(report)
67
68
  rows = []
68
69
  for h in hotspots.ranked(report):
69
- if h["code"] is None or h["revs"] < min_revs or filetypes.is_test_path(h["entity"]) or filetypes.is_release_path(h["entity"]):
70
+ if h["code"] is None or h["revs"] < min_revs or filetypes.is_test_path(h["entity"]) or filetypes.is_release(h["entity"], plumb):
70
71
  continue # a version file or a manifest changes on every release, not where the next bug lands
71
72
  fx = fixes.get(h["entity"], {})
72
73
  own = owners.get(h["entity"]) or Counter()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.6.8
3
+ Version: 0.6.10
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -55,7 +55,7 @@ Linux package names, the release binaries, `--plots` and the pip caveats:
55
55
  ```bash
56
56
  gitmole . # the clone you are in
57
57
  gitmole /path/to/clone # any local clone
58
- gitmole owner/repo # clones with gh into a temp dir first
58
+ gitmole owner/repo # clones into a temp dir first, with gh or plain git
59
59
  gitmole 'owner/*' # every non-archived repo of a user or org, one summary table
60
60
 
61
61
  gitmole . --markdown report.md # the same report as a Markdown document
@@ -86,15 +86,18 @@ class Discover(unittest.TestCase):
86
86
 
87
87
  class TestPaths(unittest.TestCase):
88
88
  def test_test_files_and_directories(self):
89
- for path in ("tests/test_a.py", "a/spec/b.rb", "src/__tests__/x.js", "x/y_test.go", "app.spec.ts", "app.test.tsx", "test_x.py"):
89
+ for path in ("tests/test_a.py", "a/spec/b.rb", "src/__tests__/x.js", "x/y_test.go", "app.spec.ts", "app.test.tsx", "test_x.py",
90
+ "pending_tests/main.py", "e2e-tests/login.ts", "src/test_utils/helpers.py", "crates/x/snapshots/rule__S105.py.snap",
91
+ "src/__snapshots__/a.js.snap", "lib/render.snap"):
90
92
  self.assertTrue(filetypes.is_test_path(path), path)
91
- for path in ("src/contest.py", "gitmole/render.py", "attest/x.py", "latest.md"):
93
+ for path in ("src/contest.py", "gitmole/render.py", "attest/x.py", "latest.md", "src/testimony.py", "snapshot.py"):
92
94
  self.assertFalse(filetypes.is_test_path(path), path)
93
95
 
94
96
  def test_documentation_files_and_directories(self):
95
- for path in ("README.md", "docs/GA4-API-INTEGRATION.md", "doc/guide.rst", "NOTES.txt", "a/b/CHANGELOG.markdown", "docs/conf.py", "x.adoc"):
97
+ for path in ("README.md", "docs/GA4-API-INTEGRATION.md", "doc/guide.rst", "NOTES.txt", "a/b/CHANGELOG.markdown", "docs/conf.py", "x.adoc",
98
+ "docs_src/security/tutorial004.py", "docs-site/app.js", "doc_examples/x.py"):
96
99
  self.assertTrue(filetypes.is_doc_path(path), path)
97
- for path in ("app/settings.py", "static/index.html", "docsite/app.js", "mdx/a.py", "config.yaml"):
100
+ for path in ("app/settings.py", "static/index.html", "docsite/app.js", "mdx/a.py", "config.yaml", "doctor/a.py"):
98
101
  self.assertFalse(filetypes.is_doc_path(path), path)
99
102
 
100
103
  def test_example_fixture_and_rule_directories(self):
@@ -297,6 +297,12 @@ class ReleasePlumbing(unittest.TestCase):
297
297
  self.assertIn("requests/models.py changed 60 times", f[0]["detail"])
298
298
  self.assertNotIn("setup.py", f[0]["detail"])
299
299
 
300
+ def test_a_file_the_change_log_shows_as_plumbing_does_not_dominate_the_churn(self):
301
+ revs = [{"entity": "fastapi/__init__.py", "n-revs": 331}, {"entity": "fastapi/routing.py", "n-revs": 187}, {"entity": "fastapi/utils.py", "n-revs": 70}]
302
+ r = report(revisions=revs, plumbing=[{"entity": "fastapi/__init__.py", "n-revs": 331, "tiny-revs": 300}])
303
+ f = findings.hotspot_dominance(r)
304
+ self.assertIn("fastapi/routing.py changed 187 times", f[0]["detail"])
305
+
300
306
  def test_a_manifest_is_not_a_bug_magnet(self):
301
307
  fixes = [{"entity": "package.json", "n-fixes": 20, "last-fix": "2026-09-01", "recent-fixes": 6},
302
308
  {"entity": "lib/reply.js", "n-fixes": 10, "last-fix": "2026-09-01", "recent-fixes": 4}]
@@ -67,6 +67,17 @@ class FunctionsScript(unittest.TestCase):
67
67
  self.assertIn('"tracked"', csv, "functions are still measured")
68
68
  self.assertIsNone(dup, "no duplicates.txt: the finder did not run, so nothing pretends it did")
69
69
 
70
+ def test_a_huge_nested_name_is_cut_before_it_is_written(self):
71
+ from types import SimpleNamespace
72
+ name = ".".join("a" for _ in range(100_000))
73
+ fn = SimpleNamespace(nloc=5, cyclomatic_complexity=3, token_count=40, parameter_count=1, length=5, name=name,
74
+ long_name=name + "( )", start_line=1, end_line=5)
75
+ row = functions.csv_row(SimpleNamespace(filename="a.py"), fn)
76
+ self.assertEqual(len(row[7]), functions.NAME_CAP)
77
+ self.assertTrue(row[7].endswith("…"))
78
+ self.assertLessEqual(len(row[8]), functions.LONG_NAME_CAP)
79
+ self.assertLessEqual(len(row[5]), functions.NAME_CAP + len("@1-5@a.py"))
80
+
70
81
  def test_csv_matches_lizards_own_layout(self):
71
82
  with tempfile.TemporaryDirectory() as d:
72
83
  make_repo(d)
@@ -34,6 +34,12 @@ class Merge(unittest.TestCase):
34
34
  self.assertEqual(merged, {"Junegunn Choi": 3047, "Sam Altman": 5, "sam": 2, "Kevin Brown": 76, "Kevin": 21},
35
35
  "a short or common first name is not distinctive enough")
36
36
 
37
+ def test_a_handle_that_is_the_full_name_run_together_is_the_same_person(self):
38
+ ids = [{"name": "Robin Malfait", "email": "malfait.robin@a.com", "commits": 1271}, {"name": "RobinMalfait", "email": "1834413+RobinMalfait@users.noreply.github.com", "commits": 4},
39
+ {"name": "Jo Li", "email": "jo@a.com", "commits": 3}, {"name": "joli", "email": "x@b.com", "commits": 1}]
40
+ merged = {m["name"]: m["commits"] for m in identity.merge(ids)}
41
+ self.assertEqual(merged, {"Robin Malfait": 1275, "Jo Li": 3, "joli": 1}, "a run-together name shorter than six letters could be anyone")
42
+
37
43
  def test_a_bare_common_first_name_is_not_enough(self):
38
44
  ids = [{"name": "Jean", "email": "jean@a.com", "commits": 24}, {"name": "Jean", "email": "jean@b.com", "commits": 18},
39
45
  {"name": "Alex", "email": "alex@a.com", "commits": 3}, {"name": "alex", "email": "alex@b.com", "commits": 2}]
@@ -138,6 +138,19 @@ class ParseFunctions(unittest.TestCase):
138
138
  def test_empty(self):
139
139
  self.assertEqual(load.parse_functions(""), [])
140
140
 
141
+ def test_a_row_with_a_huge_long_name_does_not_abort_the_report(self):
142
+ # ruff: lizard wrote a long name of several hundred kilobytes for one function, over csv's default field limit
143
+ huge = "f( " + "a, " * 100_000 + ")"
144
+ rows = load.parse_functions(f'5,3,40,1,5,"f@1-5@a.rs","a.rs","f","{huge}",1,5\n')
145
+ self.assertEqual((rows[0]["function"], rows[0]["file"], rows[0]["ccn"]), ("f", "a.rs", 3))
146
+
147
+ def test_a_huge_function_name_is_cut_to_something_a_table_can_show(self):
148
+ # a test fixture of deeply nested functions gives lizard a dotted name of megabytes
149
+ name = ".".join("a" for _ in range(100_000))
150
+ rows = load.parse_functions(f'5,3,40,1,5,"{name}@1-5@a.py","a.py","{name}","{name}( )",1,5\n')
151
+ self.assertEqual(len(rows[0]["function"]), load.NAME_CAP)
152
+ self.assertTrue(rows[0]["function"].endswith("…"))
153
+
141
154
  def test_a_nameless_function_is_called_anonymous(self):
142
155
  # lizard names Go function literals with an empty string where it names JavaScript's "(anonymous)"
143
156
  rows = load.parse_functions('136,47,926,1,270,"@316-585@completions.go","completions.go",""," c * Command",316,585\n')
@@ -224,6 +237,7 @@ class LoadReport(unittest.TestCase):
224
237
  "meta.json": json.dumps({"name": "demo", "commits": 3, "identities": []}),
225
238
  "size.json": json.dumps([{"Name": "Python", "Count": 1, "Code": 10, "Comment": 0, "Blank": 0, "Complexity": 1}]),
226
239
  "maat-revisions.csv": "entity,n-revs\na.py,3\n",
240
+ "maat-plumbing.csv": "entity,n-revs,tiny-revs\npkg/__init__.py,25,24\n",
227
241
  "maat-coupling.csv": "entity,coupled,degree,average-revs\n",
228
242
  "maat-age.csv": "entity,age-months\na.py,0\n",
229
243
  "maat-authors.csv": "entity,n-authors,n-revs\na.py,1,3\n",
@@ -243,6 +257,7 @@ class LoadReport(unittest.TestCase):
243
257
  self.assertEqual(r["meta"]["name"], "demo")
244
258
  self.assertEqual(r["size"]["total_code"], 10)
245
259
  self.assertEqual(r["revisions"][0]["entity"], "a.py")
260
+ self.assertEqual(r["plumbing"], [{"entity": "pkg/__init__.py", "n-revs": 25, "tiny-revs": 24}])
246
261
  self.assertEqual(r["coupling"], [])
247
262
  self.assertEqual(r["fixes"][0]["recent-fixes"], 1)
248
263
  self.assertEqual(r["cohorts"], {"Code added in 2026": 10})
@@ -99,6 +99,29 @@ class Revisions(unittest.TestCase):
99
99
  self.assertEqual(dict((r["entity"], r["n-revs"]) for r in rows)["src/c.py"], 1)
100
100
 
101
101
 
102
+ class Plumbing(unittest.TestCase):
103
+ def _commits(self, path, tiny, big):
104
+ out = [{"hash": f"t{i}", "date": "2026-01-01", "time": "", "author": "A", "subject": "bump", "files": [(path, 1, 1)]} for i in range(tiny)]
105
+ out += [{"hash": f"b{i}", "date": "2026-01-01", "time": "", "author": "A", "subject": "work", "files": [(path, 40, 12)]} for i in range(big)]
106
+ return out
107
+
108
+ def test_a_file_whose_commits_nearly_always_change_a_line_or_two_is_plumbing(self):
109
+ commits = self._commits("fastapi/__init__.py", tiny=300, big=31) + self._commits("fastapi/routing.py", tiny=10, big=90) + self._commits("VERSION", tiny=5, big=0)
110
+ rows = maat.plumbing(commits)
111
+ self.assertEqual(rows, [{"entity": "fastapi/__init__.py", "n-revs": 331, "tiny-revs": 300}],
112
+ "routing.py has real edits; VERSION has too few commits to judge")
113
+
114
+ def test_written_alongside_the_other_analyses(self):
115
+ with tempfile.TemporaryDirectory() as d:
116
+ log = os.path.join(d, "log.txt")
117
+ with open(log, "w", encoding="utf-8") as fh:
118
+ fh.write("".join(f"--h{i}--2026-01-{1 + i % 28:02d}T10:00:00+00:00--Ann--bump\n1\t1\tpkg/__init__.py\n" for i in range(25)))
119
+ maat.write_all(log, d, types=None)
120
+ with open(os.path.join(d, "maat-plumbing.csv"), encoding="utf-8") as fh:
121
+ text = fh.read()
122
+ self.assertEqual(text.splitlines(), ["entity,n-revs,tiny-revs", "pkg/__init__.py,25,25"])
123
+
124
+
102
125
  class Coupling(unittest.TestCase):
103
126
  def test_degree_is_shared_over_average_revisions(self):
104
127
  rows = maat.coupling(maat.parse_log(LOG))
@@ -311,14 +334,15 @@ class SinceWindow(unittest.TestCase):
311
334
 
312
335
 
313
336
  class WriteAll(unittest.TestCase):
314
- def test_writes_the_five_csv_files_in_code_maat_layout(self):
337
+ def test_writes_the_csv_files_in_code_maat_layout_plus_plumbing(self):
315
338
  with tempfile.TemporaryDirectory() as d:
316
339
  log = os.path.join(d, "log.txt")
317
340
  with open(log, "w") as fh:
318
341
  fh.write(LOG)
319
342
  maat.write_all(log, d)
320
343
  names = sorted(n for n in os.listdir(d) if n.startswith("maat-"))
321
- self.assertEqual(names, ["maat-age.csv", "maat-authors.csv", "maat-coupling.csv", "maat-entity-ownership.csv", "maat-fixes.csv", "maat-revisions.csv"])
344
+ self.assertEqual(names, ["maat-age.csv", "maat-authors.csv", "maat-coupling.csv", "maat-entity-ownership.csv", "maat-fixes.csv",
345
+ "maat-plumbing.csv", "maat-revisions.csv"])
322
346
  self.assertTrue(os.path.isfile(os.path.join(d, "activity.json")))
323
347
  with open(os.path.join(d, "maat-revisions.csv")) as fh:
324
348
  self.assertEqual(fh.readline().strip(), "entity,n-revs")
@@ -345,6 +345,16 @@ class Report(unittest.TestCase):
345
345
  full = rendered(r, [], width=200, full=True)
346
346
  self.assertIn("setup.py", full[full.index("◆ Hotspots"):])
347
347
 
348
+ def test_default_hotspots_hide_files_the_change_log_shows_as_plumbing(self):
349
+ r = sample_report()
350
+ r["size"]["files"]["pkg/__init__.py"] = {"code": 40, "complexity": 0}
351
+ r["revisions"].append({"entity": "pkg/__init__.py", "n-revs": 331})
352
+ r["plumbing"] = [{"entity": "pkg/__init__.py", "n-revs": 331, "tiny-revs": 300}]
353
+ hot = rendered(r, [], width=200)
354
+ hot = hot[hot.index("◆ Hotspots"):hot.index("Change coupling")]
355
+ self.assertNotIn("pkg/__init__.py", hot)
356
+ self.assertIn("1 release file hidden; --full shows them", hot)
357
+
348
358
  def test_default_coupling_hides_release_plumbing_pairs_and_says_so(self):
349
359
  r = sample_report()
350
360
  for f in ("lib/version.rb", "contrib/version.rb", "Gemfile", "Gemfile.lock"):
@@ -77,11 +77,36 @@ class GhErrors(unittest.TestCase):
77
77
  run.list_repos("acme", lister=lister)
78
78
  self.assertIn("failed to verify certificate", str(ctx.exception))
79
79
 
80
- def test_clone_wraps_gh_failure_with_its_stderr(self):
80
+ def test_clone_falls_back_to_plain_git_when_gh_is_missing_or_fails(self):
81
+ # a public repository needs no gh and no token: git alone can clone it
82
+ calls = []
83
+ def no_gh(argv):
84
+ raise FileNotFoundError("gh")
85
+ def git(argv):
86
+ calls.append(argv)
87
+ return ""
88
+ with tempfile.TemporaryDirectory() as d:
89
+ dest = run.clone("acme/widgets", d, runner=no_gh, git_runner=git)
90
+ self.assertEqual(dest, os.path.join(d, "widgets"))
91
+ self.assertEqual(calls, [["git", "-c", "core.quotePath=false", "clone", "--quiet", "https://github.com/acme/widgets.git", dest]])
92
+ calls.clear()
93
+ run.clone("https://github.com/acme/widgets.git", d, runner=lambda argv: (_ for _ in ()).throw(subprocess.CalledProcessError(1, argv, stderr="tls")), git_runner=git)
94
+ self.assertEqual(calls[0][5], "https://github.com/acme/widgets.git", "a URL is cloned as given")
95
+
96
+ def test_clone_reports_both_failures_when_git_fails_too(self):
81
97
  with tempfile.TemporaryDirectory() as d:
82
98
  with self.assertRaises(run.GhError) as ctx:
83
- run.clone("acme/definitely-missing-repo-xyz", d, runner=lambda argv: (_ for _ in ()).throw(subprocess.CalledProcessError(1, argv, stderr="repository not found")))
99
+ run.clone("acme/definitely-missing-repo-xyz", d,
100
+ runner=lambda argv: (_ for _ in ()).throw(subprocess.CalledProcessError(1, argv, stderr="repository not found")),
101
+ git_runner=lambda argv: (_ for _ in ()).throw(subprocess.CalledProcessError(128, argv, stderr="fatal: not found")))
84
102
  self.assertIn("repository not found", str(ctx.exception))
103
+ self.assertIn("git clone also failed: fatal: not found", str(ctx.exception))
104
+
105
+ def test_clone_does_not_touch_git_when_gh_succeeds(self):
106
+ calls = []
107
+ with tempfile.TemporaryDirectory() as d:
108
+ run.clone("acme/widgets", d, runner=lambda argv: "", git_runner=lambda argv: calls.append(argv))
109
+ self.assertEqual(calls, [])
85
110
 
86
111
 
87
112
  class ParseSince(unittest.TestCase):
@@ -61,6 +61,13 @@ class Risks(unittest.TestCase):
61
61
  files = sorted(x["file"] for x in watch.risks(r))
62
62
  self.assertEqual(files, ["Makefile", "core/parser.py"], "a version file or a manifest changes on every release, not where the next bug lands")
63
63
 
64
+ def test_a_file_the_change_log_shows_as_plumbing_is_not_on_the_list(self):
65
+ r = report()
66
+ r["size"]["files"]["pkg/__init__.py"] = {"code": 40, "complexity": 0}
67
+ r["revisions"] = [{"entity": "pkg/__init__.py", "n-revs": 331}, {"entity": "core/parser.py", "n-revs": 40}]
68
+ r["plumbing"] = [{"entity": "pkg/__init__.py", "n-revs": 331, "tiny-revs": 300}]
69
+ self.assertEqual([x["file"] for x in watch.risks(r)], ["core/parser.py"])
70
+
64
71
  def test_test_companions_and_weak_pairs_are_not_reasons(self):
65
72
  top = watch.risks(report())[0]
66
73
  coupling = [r for r in top["reasons"] if r.startswith("changes with")][0]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes