gitmole 0.6.3__tar.gz → 0.6.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. {gitmole-0.6.3 → gitmole-0.6.5}/PKG-INFO +1 -1
  2. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/__init__.py +1 -1
  3. gitmole-0.6.5/gitmole/coupling.py +57 -0
  4. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/filetypes.py +18 -0
  5. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/findings.py +42 -17
  6. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/knowledge.py +24 -0
  7. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/leaks.py +4 -1
  8. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/render.py +61 -9
  9. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole.egg-info/PKG-INFO +1 -1
  10. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole.egg-info/SOURCES.txt +2 -0
  11. gitmole-0.6.5/tests/test_coupling.py +68 -0
  12. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_filetypes.py +14 -0
  13. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_findings.py +81 -2
  14. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_knowledge.py +30 -0
  15. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_leaks.py +7 -0
  16. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_render.py +69 -2
  17. {gitmole-0.6.3 → gitmole-0.6.5}/LICENSE +0 -0
  18. {gitmole-0.6.3 → gitmole-0.6.5}/README.md +0 -0
  19. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/__main__.py +0 -0
  20. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/backtest.py +0 -0
  21. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/banner.py +0 -0
  22. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/blame.py +0 -0
  23. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/clean.py +0 -0
  24. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/cli.py +0 -0
  25. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/functions.py +0 -0
  26. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/hotspots.py +0 -0
  27. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/identity.py +0 -0
  28. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/load.py +0 -0
  29. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/loss.py +0 -0
  30. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/maat.py +0 -0
  31. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/run.py +0 -0
  32. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/textfmt.py +0 -0
  33. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/trend.py +0 -0
  34. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole/watch.py +0 -0
  35. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole.egg-info/dependency_links.txt +0 -0
  36. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole.egg-info/entry_points.txt +0 -0
  37. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole.egg-info/requires.txt +0 -0
  38. {gitmole-0.6.3 → gitmole-0.6.5}/gitmole.egg-info/top_level.txt +0 -0
  39. {gitmole-0.6.3 → gitmole-0.6.5}/pyproject.toml +0 -0
  40. {gitmole-0.6.3 → gitmole-0.6.5}/setup.cfg +0 -0
  41. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_backtest.py +0 -0
  42. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_banner.py +0 -0
  43. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_blame.py +0 -0
  44. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_clean.py +0 -0
  45. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_cli.py +0 -0
  46. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_functions.py +0 -0
  47. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_golden.py +0 -0
  48. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_hotspots.py +0 -0
  49. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_identity.py +0 -0
  50. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_load.py +0 -0
  51. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_loss.py +0 -0
  52. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_maat.py +0 -0
  53. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_packaging.py +0 -0
  54. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_run.py +0 -0
  55. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_textfmt.py +0 -0
  56. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_trend.py +0 -0
  57. {gitmole-0.6.3 → gitmole-0.6.5}/tests/test_watch.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.6.3
3
+ Version: 0.6.5
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.6.3"
3
+ __version__ = "0.6.5"
@@ -0,0 +1,57 @@
1
+ """Change-coupling pairs grouped into clusters: a directory whose files all change together (generated
2
+ tables, one-per-version data files) is one fact, not a page of pairs."""
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from collections import defaultdict
7
+
8
+ ROOT = "(root files)"
9
+
10
+
11
+ def _dir(path: str) -> str:
12
+ head = os.path.dirname(path)
13
+ return head + "/" if head else ROOT
14
+
15
+
16
+ def _components(pairs: list) -> list:
17
+ """Connected groups of files among `pairs` (union-find), each as the list of its pairs."""
18
+ parent = {}
19
+
20
+ def find(x):
21
+ parent.setdefault(x, x)
22
+ while parent[x] != x:
23
+ parent[x] = parent[parent[x]]
24
+ x = parent[x]
25
+ return x
26
+
27
+ for p in pairs:
28
+ parent[find(p["entity"])] = find(p["coupled"])
29
+ groups = defaultdict(list)
30
+ for p in pairs:
31
+ groups[find(p["entity"])].append(p)
32
+ return list(groups.values())
33
+
34
+
35
+ def clusters(pairs: list, min_files: int = 4, min_density: float = 0.8) -> tuple:
36
+ """Split `pairs` into (groups, rest). Pairs whose two files share a directory are gathered per
37
+ directory and then into connected groups; a group of at least `min_files` distinct files with at
38
+ least `min_density` of the possible pairs present becomes one cluster with the file and pair
39
+ counts, the weakest degree and the mean of the pairs' average revisions. Every other pair comes
40
+ back unchanged, in its original order. Two unrelated pairs in one directory, or a chain of pairs,
41
+ are not a cluster: "each other" has to be true. Clusters are largest first."""
42
+ by_dir = defaultdict(list)
43
+ for p in pairs:
44
+ if _dir(p["entity"]) == _dir(p["coupled"]):
45
+ by_dir[_dir(p["entity"])].append(p)
46
+ groups, taken = [], set()
47
+ for directory, ps in by_dir.items():
48
+ for component in _components(ps):
49
+ files = {p["entity"] for p in component} | {p["coupled"] for p in component}
50
+ possible = len(files) * (len(files) - 1) / 2
51
+ if len(files) < min_files or len(component) < min_density * possible:
52
+ continue
53
+ groups.append({"dir": directory, "files": len(files), "pairs": len(component), "degree": min(p["degree"] for p in component),
54
+ "average-revs": round(sum(p["average-revs"] for p in component) / len(component))})
55
+ taken.update(id(p) for p in component)
56
+ groups.sort(key=lambda g: (-g["files"], -g["degree"], g["dir"]))
57
+ return groups, [p for p in pairs if id(p) not in taken]
@@ -65,6 +65,24 @@ def is_doc_path(path: str) -> bool:
65
65
  return bool(_DOC_PATH.search(path))
66
66
 
67
67
 
68
+ _SAMPLE_PATH = re.compile(r"(^|/)(examples?|samples?|fixtures?|testdata|demos?|rules)(/|$)", re.I)
69
+
70
+
71
+ def is_sample_path(path: str) -> bool:
72
+ """Example, sample, fixture, demo and rule directories: a value there is a specimen (a language
73
+ sample, a scanner's own rule definitions), not a credential in use."""
74
+ return bool(_SAMPLE_PATH.search(path))
75
+
76
+
77
+ _VENDOR_PATH = re.compile(r"(^|/)(_?vendor|node_modules|third_?party|external)(/|$)", re.I)
78
+
79
+
80
+ def is_vendor_path(path: str) -> bool:
81
+ """Vendored and third-party trees: somebody else's code, so its complexity and its single
82
+ importer are not this repository's risk."""
83
+ return bool(_VENDOR_PATH.search(path))
84
+
85
+
68
86
  def key(path: str) -> str:
69
87
  """The lowercased extension, or the whole lowercased name when there is none."""
70
88
  name = path.rsplit("/", 1)[-1].lower()
@@ -3,7 +3,7 @@ from __future__ import annotations
3
3
 
4
4
  import re
5
5
 
6
- from . import filetypes, hotspots, knowledge, leaks, loss, textfmt, trend
6
+ from . import coupling, filetypes, hotspots, knowledge, leaks, loss, textfmt, trend
7
7
 
8
8
  SEVERITIES = ["critical", "warning", "info"]
9
9
 
@@ -40,13 +40,14 @@ def _secret_statement(groups: list) -> str:
40
40
 
41
41
  def secrets_found(report: dict) -> list:
42
42
  """Secrets grouped by value. A value anywhere in source is critical; one that only ever appears in
43
- test files (fixtures, saved pages) or documentation (templates, samples) is a warning, so a
44
- critical gate does not trip on test data or a planning document. Version strings, template markers
45
- and key blocks without key material were flagged as placeholders and are not a finding."""
43
+ test files (fixtures, saved pages), example or rule directories (language samples, a scanner's own
44
+ rules) or documentation (templates) is a warning, so a critical gate does not trip on test data or a
45
+ planning document. Version strings, template markers and key blocks without key material were
46
+ flagged as placeholders and are not a finding."""
46
47
  groups = leaks.group(report.get("secrets") or [])
47
48
 
48
49
  def in_source(g):
49
- return any(not (filetypes.is_test_path(f) or filetypes.is_doc_path(f)) for f in g["files"])
50
+ return any(not (filetypes.is_test_path(f) or filetypes.is_doc_path(f) or filetypes.is_sample_path(f)) for f in g["files"])
50
51
  source = [g for g in groups if in_source(g)]
51
52
  aside = [g for g in groups if not in_source(g)]
52
53
  ignore = "Add the fingerprint of any false positive from secrets.json to .betterleaksignore in the repository."
@@ -55,7 +56,7 @@ def secrets_found(report: dict) -> list:
55
56
  out.append(_f("critical", f"{len(source)} secret(s) in history", _secret_statement(source),
56
57
  f"Rotate them; deleting the file does not remove them from git. {ignore}"))
57
58
  if aside:
58
- out.append(_f("warning", f"{len(aside)} secret(s) only in test or documentation files", _secret_statement(aside),
59
+ out.append(_f("warning", f"{len(aside)} secret(s) only in test, example or documentation files", _secret_statement(aside),
59
60
  f"Confirm they are fixtures or templates, not live keys. {ignore}"))
60
61
  return out
61
62
 
@@ -93,9 +94,18 @@ def placeholder_identity(report: dict, min_share: float = 0.01) -> list:
93
94
 
94
95
 
95
96
  def _source_ownership(report: dict) -> list:
96
- """Ownership rows for source files. Test files are left out of every rule that names a next
97
- step: owning the tests is not the knowledge risk. The default tables leave them out too."""
98
- return [r for r in report.get("ownership") or [] if not filetypes.is_test_path(r["entity"])]
97
+ """Ownership rows for source files. Test files and vendored trees are left out of every rule that
98
+ names a next step: owning the tests is not the knowledge risk, and whoever imported vendor/ did
99
+ not write it. The default tables leave test files out too."""
100
+ return [r for r in report.get("ownership") or [] if not (filetypes.is_test_path(r["entity"]) or filetypes.is_vendor_path(r["entity"]))]
101
+
102
+
103
+ def _present_areas(report: dict, rows: list, build=knowledge.areas) -> list:
104
+ """Areas built from the ownership rows of directories that still exist, then only those areas that
105
+ exist themselves: a directory the history knows but HEAD does not (the layout before a move to
106
+ src/ or crates/) is nowhere to pair anyone on. `build` is knowledge.areas or a wrapper of it."""
107
+ tree = _tree(report)
108
+ return [a for a in build(knowledge.present_rows(rows, tree)) if knowledge.in_tree(a["area"], tree)]
99
109
 
100
110
 
101
111
  def bus_factor(report: dict, threshold: float = 0.7, min_lines: int = 200) -> list:
@@ -109,7 +119,7 @@ def bus_factor(report: dict, threshold: float = 0.7, min_lines: int = 200) -> li
109
119
  if lines / total <= threshold:
110
120
  return []
111
121
  theirs = []
112
- for a in knowledge.areas(_source_ownership(report)):
122
+ for a in _present_areas(report, _source_ownership(report)):
113
123
  owned = dict(a["owners"]).get(name, 0)
114
124
  if a["lines"] >= min_lines and owned / a["lines"] >= 0.8:
115
125
  theirs.append((a["area"], round(100 * owned / a["lines"])))
@@ -183,11 +193,21 @@ def tight_coupling(report: dict, min_degree: int = 80, min_revs: int = 5) -> lis
183
193
  if not pairs:
184
194
  return []
185
195
  pairs.sort(key=lambda p: (-p["degree"], -p["average-revs"]))
196
+ groups, pairs = coupling.clusters(pairs)
197
+ when = f"together at least {min_degree}% of the time"
186
198
  top = "; ".join(f"{p['entity']} + {p['coupled']} ({p['degree']}%)" for p in pairs[:3])
199
+ if groups:
200
+ # a directory of files that change as one is a generator or a shared layout, said once
201
+ named = ", ".join(f"{g['files']} files in {g['dir']}" for g in groups[:2]) + (f" and {len(groups) - 2} more directories" if len(groups) > 2 else "")
202
+ rest = (f", and {_plural(len(pairs), 'more pair')} {'does' if len(pairs) == 1 else 'do'}: {top}." if pairs
203
+ else f", {_plural(sum(g['pairs'] for g in groups), 'pair')} in all.")
204
+ first = groups[0]
205
+ return [_f("info", "Files that always change together", f"{named} change {when}{rest}",
206
+ f"Review {first['dir']} first: {first['files']} files change as one; a generator or a shared layout links them.")]
187
207
  count = f"{len(pairs)} pair changes" if len(pairs) == 1 else f"{len(pairs)} pairs change"
188
208
  first = pairs[0]
189
209
  return [_f("info", "Files that always change together",
190
- f"{count} together at least {min_degree}% of the time, e.g. {top}.",
210
+ f"{count} {when}, e.g. {top}.",
191
211
  f"Review {first['entity']} and {first['coupled']} first: a shared layout or a hidden dependency links them.")]
192
212
 
193
213
 
@@ -251,7 +271,9 @@ def reverts(report: dict, min_share: float = 0.05, min_count: int = 5, warn_shar
251
271
 
252
272
 
253
273
  def knowledge_islands(report: dict, min_lines: int = 200, min_share: float = 0.9) -> list:
254
- areas = knowledge.areas(_source_ownership(report))
274
+ """Areas of the tree written almost entirely by one person. Areas that no longer exist are left
275
+ out, of the islands and of the total they are measured against."""
276
+ areas = _present_areas(report, _source_ownership(report))
255
277
  islands = knowledge.islands(areas, min_lines=min_lines, min_share=min_share)
256
278
  if not islands:
257
279
  return []
@@ -317,9 +339,10 @@ def _loss_people(by_person: dict, total: int) -> str:
317
339
 
318
340
 
319
341
  def _loss_areas(report: dict, names: set, source_rows: list) -> list:
320
- """Areas at 200+ lines where 80%+ of the surviving code is theirs, tagged live or not and
321
- sorted live-first: that is where the gap bites soonest."""
322
- theirs = [a for a in loss.areas(source_rows, names) if a["lines"] >= 200 and a["lost_share"] >= 0.8]
342
+ """Areas at 200+ lines where 80%+ of the surviving code is theirs, still in the tree, tagged live
343
+ or not and sorted live-first: that is where the gap bites soonest."""
344
+ theirs = [a for a in _present_areas(report, source_rows, build=lambda rows: loss.areas(rows, names))
345
+ if a["lines"] >= 200 and a["lost_share"] >= 0.8]
323
346
  for a in theirs:
324
347
  a["live"] = _is_live(a["area"], report.get("age") or [])
325
348
  theirs.sort(key=lambda a: (not a["live"], -a["lines"], a["area"])) # a live area first: that is where the gap bites
@@ -363,8 +386,10 @@ def _partial_functions(report: dict) -> str:
363
386
 
364
387
 
365
388
  def brain_methods(report: dict, min_ccn: int = 15, min_lines: int = 100) -> list:
366
- """Functions that are both long and complex, in source files. A warning when one sits in a hotspot."""
367
- big = [f for f in report.get("functions") or [] if f["ccn"] >= min_ccn and f["nloc"] >= min_lines and not filetypes.is_test_path(f["file"])]
389
+ """Functions that are both long and complex, in this repository's own source files: test files and
390
+ vendored code are left out. A warning when one sits in a hotspot."""
391
+ big = [f for f in report.get("functions") or [] if f["ccn"] >= min_ccn and f["nloc"] >= min_lines
392
+ and not (filetypes.is_test_path(f["file"]) or filetypes.is_vendor_path(f["file"]))]
368
393
  if not big:
369
394
  return []
370
395
  big.sort(key=lambda f: (-f["ccn"], -f["nloc"], f["file"], f["function"], f["start"]))
@@ -7,6 +7,16 @@ from collections import Counter, defaultdict
7
7
  ROOT = "(root files)"
8
8
 
9
9
 
10
+ def in_tree(area: str, tree: dict) -> bool:
11
+ """Whether any tracked file sits under `area` (a directory prefix ending in "/", or ROOT). With
12
+ no tree listing every area counts: there is nothing to judge by."""
13
+ if not tree:
14
+ return True
15
+ if area == ROOT:
16
+ return any("/" not in path for path in tree)
17
+ return any(path.startswith(area) for path in tree)
18
+
19
+
10
20
  def _area(entity: str, depth: int) -> str:
11
21
  dirs = entity.split("/")[:-1]
12
22
  if not dirs:
@@ -14,6 +24,20 @@ def _area(entity: str, depth: int) -> str:
14
24
  return "/".join(dirs[:depth]) + "/"
15
25
 
16
26
 
27
+ def top_area(entity: str) -> str:
28
+ """The top-level directory of a path, or ROOT."""
29
+ return _area(entity, 1)
30
+
31
+
32
+ def present_rows(rows: list, tree: dict) -> list:
33
+ """Ownership rows for files whose top-level directory still exists in the tree. Filtering the rows
34
+ before areas are built keeps a vanished layout (the src/ before a move to crates/) from inflating
35
+ the total and hiding that one directory now holds almost everything."""
36
+ if not tree:
37
+ return rows
38
+ return [r for r in rows if in_tree(top_area(r["entity"]), tree)]
39
+
40
+
17
41
  def _aggregate(rows: list, depth: int) -> list:
18
42
  lines, per_author = Counter(), defaultdict(Counter)
19
43
  for r in rows:
@@ -33,10 +33,13 @@ RAW_FIELDS = ("Secret", "Match", "Line", "Message", "Attributes")
33
33
 
34
34
  # Shapes that cannot be a live secret: a version string (5.0.0-1667386184.dfbbb54), a token shortened
35
35
  # with an ellipsis, a whole-value template marker (your-project-id, <your-token>, XXXX-XXXX, changeme),
36
+ # a whole value that is one of the words every example uses (`password: 'hello'` in a doc comment),
36
37
  # and a key block whose body holds no key material. Every rule is about the whole value; nothing is
37
38
  # skipped by prefix, since a public and a private key of the same service often share one.
38
39
  _VERSION = re.compile(r"^\d+\.\d+\.\d+(?:[-+][0-9A-Za-z.-]+)?$")
39
40
  _MARKER = re.compile(r"^(<[^<>]+>|x+|(?:x{2,}[-_ ]?)+|your[-_][\w-]+|change[-_]?me|replace[-_]?me)$", re.I)
41
+ _EXAMPLE_WORDS = {"password", "passwd", "pass", "secret", "hello", "hey", "test", "example", "sample", "dummy", "foo", "bar",
42
+ "baz", "admin", "root", "user", "123456", "12345678", "123456789", "abc123", "qwerty", "letmein", "welcome"}
40
43
  _KEY_BLOCK = re.compile(r"-----BEGIN [A-Z ]*KEY-----(.*?)-----END [A-Z ]*KEY-----", re.S)
41
44
  _KEY_MATERIAL = 64 # a real body is hundreds of base64 characters; a template has dots or a few x's
42
45
 
@@ -54,7 +57,7 @@ def digest(value: str, key: bytes) -> str:
54
57
 
55
58
  def is_placeholder(value: str) -> bool:
56
59
  value = (value or "").strip()
57
- if _VERSION.match(value) or value.endswith("...") or value.endswith("…") or _MARKER.match(value):
60
+ if _VERSION.match(value) or value.endswith("...") or value.endswith("…") or _MARKER.match(value) or value.lower() in _EXAMPLE_WORDS:
58
61
  return True
59
62
  m = _KEY_BLOCK.search(value)
60
63
  if m:
@@ -13,7 +13,7 @@ from rich.panel import Panel
13
13
  from rich.table import Table
14
14
  from rich.text import Text
15
15
 
16
- from . import filetypes, hotspots, identity, knowledge, leaks, loss, textfmt, trend, watch
16
+ from . import coupling, filetypes, hotspots, identity, knowledge, leaks, loss, textfmt, trend, watch
17
17
 
18
18
  SEVERITY_STYLE = {"critical": "bold red", "warning": "yellow", "info": "cyan"}
19
19
 
@@ -78,8 +78,11 @@ def _more(total: int, limit) -> str:
78
78
  return f"and {total - limit} more" if limit is not None and total > limit else None
79
79
 
80
80
 
81
- def _hide_tests(rows: list, path_of, full, noun="test file", plural=None) -> tuple:
82
- """Drop rows whose path (or any of whose paths) is a test path, unless `full` is True.
81
+ HIDDEN_SUFFIX = "; --full shows them"
82
+
83
+
84
+ def _hide_rows(rows: list, path_of, full, pred, noun: str, plural=None) -> tuple:
85
+ """Drop rows whose path (or any of whose paths) satisfies `pred`, unless `full` is True.
83
86
  `path_of(row)` returns a single path or a tuple of paths to check. `noun` names one hidden row
84
87
  and `plural` names several, `noun + "s"` by default (a multi-word noun gives its own plural:
85
88
  "function in a test file" -> "functions in test files"). Returns (rows, note), `note` being
@@ -90,14 +93,41 @@ def _hide_tests(rows: list, path_of, full, noun="test file", plural=None) -> tup
90
93
  for row in rows:
91
94
  paths = path_of(row)
92
95
  paths = (paths,) if isinstance(paths, str) else paths
93
- if any(filetypes.is_test_path(p) for p in paths):
96
+ if any(pred(p) for p in paths):
94
97
  hidden += 1
95
98
  else:
96
99
  kept.append(row)
97
- note = f"{hidden} {noun if hidden == 1 else plural or noun + 's'} hidden; --full shows them" if hidden else None
100
+ note = f"{hidden} {noun if hidden == 1 else plural or noun + 's'} hidden{HIDDEN_SUFFIX}" if hidden else None
98
101
  return kept, note
99
102
 
100
103
 
104
+ def _hide_tests(rows: list, path_of, full, noun="test file", plural=None) -> tuple:
105
+ """Test files: they change with every fix, so they are not a signal on their own."""
106
+ return _hide_rows(rows, path_of, full, filetypes.is_test_path, noun, plural)
107
+
108
+
109
+ def _hide_vendor(rows: list, path_of, full, noun="file in vendored code", plural="files in vendored code") -> tuple:
110
+ """Vendored trees: somebody else's code, not this repository's risk."""
111
+ return _hide_rows(rows, path_of, full, filetypes.is_vendor_path, noun, plural)
112
+
113
+
114
+ def _join_hidden(*notes) -> str:
115
+ """Several hidden-row notes as one caption phrase: 'A hidden; B hidden; --full shows them'."""
116
+ parts = [n[:-len(HIDDEN_SUFFIX)] if n.endswith(HIDDEN_SUFFIX) else n for n in notes if n]
117
+ return "; ".join(parts) + HIDDEN_SUFFIX if parts else None
118
+
119
+
120
+ def _hide_deleted(rows: list, report: dict, full) -> tuple:
121
+ """Drop hotspot rows for files no longer in the tree, unless `full` is True or there is no tree
122
+ listing to judge by: a deleted file's churn is history. Returns (rows, note) like _hide_tests."""
123
+ tree = (report.get("size") or {}).get("files") or {}
124
+ if full is True or not tree:
125
+ return rows, None
126
+ kept = [h for h in rows if h["code"] is not None]
127
+ hidden = len(rows) - len(kept)
128
+ return kept, (f"{hidden} deleted file{'s' if hidden != 1 else ''} hidden{HIDDEN_SUFFIX}" if hidden else None)
129
+
130
+
101
131
  def _hide_gone(pairs: list, report: dict, full) -> tuple:
102
132
  """Drop coupled pairs where either file is no longer in the tree, unless `full` is True: they
103
133
  describe a layout that no longer exists. Returns (pairs, note) like _hide_tests."""
@@ -345,6 +375,8 @@ def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
345
375
  fixes = {f["entity"]: f["n-fixes"] for f in report.get("fixes") or []}
346
376
  scored = hotspots.ranked(report)
347
377
  scored, hidden_note = _hide_tests(scored, lambda h: h["entity"], full)
378
+ scored, deleted_note = _hide_deleted(scored, report, full)
379
+ hidden_note = _join_hidden(hidden_note, deleted_note)
348
380
  title = "Hotspots (score = revisions × lines of code)" if full is True else "Hotspots"
349
381
  limit = _limit("Hotspots", full)
350
382
  series = (report.get("trend") or {}).get("files") or {}
@@ -376,9 +408,18 @@ def coupling_section(report: dict, full: bool = True, width=None) -> dict:
376
408
  pairs = sorted((p for p in report.get("coupling") or [] if p["average-revs"] >= 5), key=lambda p: (-p["degree"], -p["average-revs"]))
377
409
  pairs, hidden_note = _hide_tests(pairs, lambda p: (p["entity"], p["coupled"]), full, noun="test pair")
378
410
  pairs, gone_note = _hide_gone(pairs, report, full)
379
- hidden_note = "; ".join(n for n in (hidden_note, gone_note) if n) or None
411
+ groups, cluster_note = [], None
412
+ if full is not True:
413
+ # a directory whose files all change together is one row; --full lists every pair
414
+ groups, pairs = coupling.clusters(pairs)
415
+ if groups:
416
+ n_pairs, n_dirs = sum(g["pairs"] for g in groups), len(groups)
417
+ cluster_note = (f"{n_pairs} pairs in {n_dirs} director{'y' if n_dirs == 1 else 'ies'} shown as "
418
+ f"{'one row' if n_dirs == 1 else 'one row each'}{HIDDEN_SUFFIX}")
419
+ hidden_note = _join_hidden(hidden_note, gone_note, cluster_note)
380
420
  limit = _limit("Change coupling", full)
381
- rows = [(p["entity"], p["coupled"], f"{p['degree']}%", p["average-revs"]) for p in pairs[:limit]]
421
+ rows = [(f"{g['dir']} ({g['files']} files)", "each other", f"≥{g['degree']}%", g["average-revs"]) for g in groups]
422
+ rows += [(p["entity"], p["coupled"], f"{p['degree']}%", p["average-revs"]) for p in pairs[:max(limit - len(groups), 0) if limit else None]]
382
423
  columns = [("file", PATH), ("changes with", PATH), ("degree", RIGHT), ("avg revs", RIGHT)]
383
424
  if full is not True:
384
425
  columns, rows = _keep(columns, rows, ["file", "changes with", "degree"])
@@ -432,6 +473,8 @@ def functions_section(report: dict, full: bool = True, width=None) -> dict:
432
473
  measured = report.get("functions") or []
433
474
  funcs = sorted((f for f in measured if f["ccn"] >= CCN_FLOOR), key=lambda f: (-f["ccn"], -f["nloc"], f["file"], f["function"], f["start"]))
434
475
  funcs, hidden_note = _hide_tests(funcs, lambda f: f["file"], full, noun="function in a test file", plural="functions in test files")
476
+ funcs, vendor_note = _hide_vendor(funcs, lambda f: f["file"], full, noun="function in vendored code", plural="functions in vendored code")
477
+ hidden_note = _join_hidden(hidden_note, vendor_note)
435
478
  limit = _limit("Complex functions", full)
436
479
  rows = [(f["function"], f["file"], f["ccn"], f["nloc"], f["params"]) for f in funcs[:limit]]
437
480
  columns = [("function", {"overflow": "fold"}), ("file", PATH), ("ccn", RIGHT), ("lines", RIGHT), ("params", RIGHT)]
@@ -459,7 +502,16 @@ def knowledge_section(report: dict, full: bool = True, width=None) -> dict:
459
502
  """Ownership by area of the tree: who wrote most of each directory, gone owners marked."""
460
503
  months = report["meta"].get("gone_months", loss.DEFAULT_MONTHS)
461
504
  gone = {g["name"] for g in loss.gone(report, months)}
462
- areas = loss.areas(report.get("ownership") or [], gone) # every area the map showed before, tests included
505
+ rows_all = report.get("ownership") or [] # every area the map showed before, tests included
506
+ areas = loss.areas(rows_all, gone)
507
+ hidden_note = None
508
+ tree = (report.get("size") or {}).get("files") or {}
509
+ if full is not True and tree:
510
+ # a directory the history knows but HEAD does not is a layout that no longer exists; the rows are
511
+ # filtered before the areas are built so a vanished layout cannot hide that one directory now dominates
512
+ areas = [a for a in loss.areas(knowledge.present_rows(rows_all, tree), gone) if knowledge.in_tree(a["area"], tree)]
513
+ hidden = len({knowledge.top_area(r["entity"]) for r in rows_all if not knowledge.in_tree(knowledge.top_area(r["entity"]), tree)})
514
+ hidden_note = f"{hidden} historical area{'s' if hidden != 1 else ''} hidden{HIDDEN_SUFFIX}" if hidden else None
463
515
  limit = _limit("Knowledge map", full)
464
516
  rows = []
465
517
  for a in areas[:limit]:
@@ -469,7 +521,7 @@ def knowledge_section(report: dict, full: bool = True, width=None) -> dict:
469
521
  columns = [("area", PATH), ("lines added", RIGHT), ("authors", RIGHT), ("lost", RIGHT), ("main owner", {}), ("second", {})]
470
522
  if full is not True:
471
523
  columns, rows = _keep(columns, rows, ["area", "lines added", "main owner", "second"])
472
- notes = [c for c in (_more(len(areas), limit),) if c]
524
+ notes = [c for c in (_more(len(areas), limit), hidden_note) if c]
473
525
  if gone:
474
526
  notes.append(f"gone = no commits in the {months} months before {report['meta'].get('last_date')}"
475
527
  + ("; gone and lost are measured over the whole history" if report["meta"].get("since") else ""))
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.6.3
3
+ Version: 0.6.5
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -8,6 +8,7 @@ gitmole/banner.py
8
8
  gitmole/blame.py
9
9
  gitmole/clean.py
10
10
  gitmole/cli.py
11
+ gitmole/coupling.py
11
12
  gitmole/filetypes.py
12
13
  gitmole/findings.py
13
14
  gitmole/functions.py
@@ -34,6 +35,7 @@ tests/test_banner.py
34
35
  tests/test_blame.py
35
36
  tests/test_clean.py
36
37
  tests/test_cli.py
38
+ tests/test_coupling.py
37
39
  tests/test_filetypes.py
38
40
  tests/test_findings.py
39
41
  tests/test_functions.py
@@ -0,0 +1,68 @@
1
+ import unittest
2
+
3
+ from gitmole import coupling
4
+
5
+
6
+ def pair(a, b, degree=100, revs=10):
7
+ return {"entity": a, "coupled": b, "degree": degree, "average-revs": revs}
8
+
9
+
10
+ def clique(directory, names, degree=100, revs=10):
11
+ files = [f"{directory}/{n}" if directory else n for n in names]
12
+ return [pair(a, b, degree, revs) for i, a in enumerate(files) for b in files[i + 1:]]
13
+
14
+
15
+ class Clusters(unittest.TestCase):
16
+ def test_a_directory_of_four_or_more_files_becomes_one_cluster(self):
17
+ pairs = clique("rich/_unicode_data", ["u10.py", "u11.py", "u12.py", "u13.py"]) + [pair("a.py", "b.py", 90)]
18
+ groups, rest = coupling.clusters(pairs)
19
+ self.assertEqual(groups, [{"dir": "rich/_unicode_data/", "files": 4, "pairs": 6, "degree": 100, "average-revs": 10}])
20
+ self.assertEqual(rest, [pair("a.py", "b.py", 90)])
21
+
22
+ def test_the_cluster_degree_is_the_weakest_pair(self):
23
+ pairs = clique("d", ["a", "b", "c", "e"], degree=100)
24
+ pairs[2]["degree"] = 83
25
+ pairs[4]["average-revs"] = 40
26
+ [g], _ = coupling.clusters(pairs)
27
+ self.assertEqual(g["degree"], 83)
28
+ self.assertEqual(g["average-revs"], 15) # the mean of the pairs' averages, rounded
29
+
30
+ def test_three_files_stay_as_pairs(self):
31
+ pairs = clique("d", ["a", "b", "c"])
32
+ groups, rest = coupling.clusters(pairs)
33
+ self.assertEqual(groups, [])
34
+ self.assertEqual(rest, pairs)
35
+
36
+ def test_pairs_across_directories_never_cluster(self):
37
+ pairs = [pair("x/a", "y/b"), pair("x/c", "y/d"), pair("x/e", "y/f"), pair("x/g", "y/h")]
38
+ groups, rest = coupling.clusters(pairs)
39
+ self.assertEqual(groups, [])
40
+ self.assertEqual(len(rest), 4)
41
+
42
+ def test_root_files_cluster_under_their_own_label_and_largest_cluster_first(self):
43
+ pairs = clique("", ["a", "b", "c", "d"]) + clique("lib", ["p", "q", "r", "s", "t"], degree=90)
44
+ groups, rest = coupling.clusters(pairs)
45
+ self.assertEqual([g["dir"] for g in groups], ["lib/", "(root files)"])
46
+ self.assertEqual(rest, [])
47
+
48
+ def test_a_chain_is_not_a_cluster(self):
49
+ pairs = [pair("d/a", "d/b"), pair("d/b", "d/c"), pair("d/c", "d/e")] # 4 files, 3 of 6 possible pairs: connected, not a clique
50
+ groups, rest = coupling.clusters(pairs)
51
+ self.assertEqual(groups, [])
52
+ self.assertEqual(rest, pairs)
53
+
54
+ def test_a_near_complete_clique_is_a_cluster(self):
55
+ pairs = clique("d", ["a", "b", "c", "e", "f"]) # 5 files, 10 pairs
56
+ eight = pairs[:8]
57
+ [g], rest = coupling.clusters(eight)
58
+ self.assertEqual((g["files"], g["pairs"]), (5, 8))
59
+ self.assertEqual(rest, [])
60
+ groups, rest = coupling.clusters(pairs[:7]) # 7 of 10 is under the 80% floor
61
+ self.assertEqual(groups, [])
62
+ self.assertEqual(len(rest), 7)
63
+
64
+ def test_two_unrelated_pairs_in_one_directory_are_not_a_cluster(self):
65
+ pairs = [pair("a", "b"), pair("c", "d")]
66
+ groups, rest = coupling.clusters(pairs)
67
+ self.assertEqual(groups, [])
68
+ self.assertEqual(rest, pairs)
@@ -97,6 +97,20 @@ class TestPaths(unittest.TestCase):
97
97
  for path in ("app/settings.py", "static/index.html", "docsite/app.js", "mdx/a.py", "config.yaml"):
98
98
  self.assertFalse(filetypes.is_doc_path(path), path)
99
99
 
100
+ def test_example_fixture_and_rule_directories(self):
101
+ for path in ("examples/language/bru.bru", "example/app.py", "samples/x.json", "sample/x.json", "fixtures/keys.pem",
102
+ "src/fixture/a.txt", "pkg/testdata/creds.yaml", "demo/x.py", "demos/x.py", "config/generate/rules/slack.go"):
103
+ self.assertTrue(filetypes.is_sample_path(path), path)
104
+ for path in ("app/settings.py", "examplesite/app.py", "src/rulesets/a.go", "sampler/x.py", "config/betterleaks.toml"):
105
+ self.assertFalse(filetypes.is_sample_path(path), path)
106
+
107
+ def test_vendored_trees(self):
108
+ for path in ("vendor/github.com/x/y.go", "web/node_modules/a/index.js", "third_party/z/a.c", "thirdparty/a.c", "_vendor/a.py",
109
+ "external/lib/a.cpp"):
110
+ self.assertTrue(filetypes.is_vendor_path(path), path)
111
+ for path in ("vendors.py", "src/vendoring/a.py", "node/a.js", "externals.txt", "app/main.go"):
112
+ self.assertFalse(filetypes.is_vendor_path(path), path)
113
+
100
114
 
101
115
  if __name__ == "__main__":
102
116
  unittest.main()
@@ -47,7 +47,7 @@ class SecretsFound(unittest.TestCase):
47
47
  self.assertIn("1 distinct value in 2 places: generic-api-key in app/settings.py (c1, c2)", crit["detail"])
48
48
  self.assertIn("Rotate", crit["advice"])
49
49
  self.assertIn(".betterleaksignore", crit["advice"])
50
- self.assertEqual(warn["title"], "2 secret(s) only in test or documentation files")
50
+ self.assertEqual(warn["title"], "2 secret(s) only in test, example or documentation files")
51
51
  self.assertIn("2 distinct values in 3 places", warn["detail"])
52
52
  self.assertIn("tests/data/a.html and 1 other file", warn["detail"])
53
53
  self.assertIn(".betterleaksignore", warn["advice"])
@@ -56,11 +56,21 @@ class SecretsFound(unittest.TestCase):
56
56
  r = report(secrets=[self.row("h3", "tests/t.py")])
57
57
  self.assertEqual([f["severity"] for f in findings.secrets_found(r)], ["warning"])
58
58
 
59
+ def test_a_value_only_in_an_example_fixture_or_rules_directory_is_a_warning(self):
60
+ r = report(secrets=[self.row("h1", "examples/language/bru.bru", "57d82e9", rule="generic-password"),
61
+ self.row("h2", "config/generate/rules/slack.go", "04bdee4", rule="slack-bot-token"),
62
+ self.row("h3", "pkg/testdata/creds.yaml", "c3")])
63
+ f = findings.secrets_found(r)
64
+ self.assertEqual([x["severity"] for x in f], ["warning"])
65
+ self.assertEqual(f[0]["title"], "3 secret(s) only in test, example or documentation files")
66
+ r = report(secrets=[self.row("h1", "examples/app.py", "c1"), self.row("h1", "app/config.py", "c2")])
67
+ self.assertEqual([x["severity"] for x in findings.secrets_found(r)], ["critical"], "the same value in source is a leak")
68
+
59
69
  def test_a_value_only_in_documentation_is_a_warning_that_says_template(self):
60
70
  r = report(secrets=[self.row("h1", "docs/GA4-API-INTEGRATION.md", "e8c0508")])
61
71
  f = findings.secrets_found(r)
62
72
  self.assertEqual([x["severity"] for x in f], ["warning"])
63
- self.assertEqual(f[0]["title"], "1 secret(s) only in test or documentation files")
73
+ self.assertEqual(f[0]["title"], "1 secret(s) only in test, example or documentation files")
64
74
  self.assertIn("fixtures or templates", f[0]["advice"])
65
75
  r = report(secrets=[self.row("h1", "docs/GA4-API-INTEGRATION.md", "e8c0508"), self.row("h1", "app/config.py", "c2")])
66
76
  self.assertEqual([x["severity"] for x in findings.secrets_found(r)], ["critical"], "the same value in source is a leak")
@@ -135,6 +145,21 @@ class BusFactor(unittest.TestCase):
135
145
  self.assertEqual(findings.bus_factor(r)[0]["advice"], "Pair someone with Ann on core/ first; it is 95% theirs since 2025-01-01.",
136
146
  "ownership is windowed while the headline share is not")
137
147
 
148
+ def test_areas_no_longer_in_the_tree_are_not_named_in_the_advice(self):
149
+ own = [{"entity": "flask/a.py", "author": "Ann", "added": 5000, "deleted": 0}, # the pre-src/ layout
150
+ {"entity": "src/a.py", "author": "Ann", "added": 900, "deleted": 0},
151
+ {"entity": "src/b.py", "author": "Bob", "added": 50, "deleted": 0}]
152
+ tree = {"files": {"src/a.py": {"code": 1, "complexity": 0}, "src/b.py": {"code": 1, "complexity": 0}}}
153
+ f = findings.bus_factor(report(theseus_authors={"Ann": 79, "Bob": 21}, ownership=own, size=tree))
154
+ self.assertEqual(f[0]["advice"], "Pair someone with Ann on src/ first; it is 95% theirs.")
155
+
156
+ def test_vendored_trees_are_not_named_in_the_advice(self):
157
+ own = [{"entity": "vendor/github.com/x/a.go", "author": "Ann", "added": 500000, "deleted": 0},
158
+ {"entity": "core/a.py", "author": "Ann", "added": 900, "deleted": 0},
159
+ {"entity": "core/b.py", "author": "Bob", "added": 50, "deleted": 0}]
160
+ f = findings.bus_factor(report(theseus_authors={"Ann": 79, "Bob": 21}, ownership=own))
161
+ self.assertEqual(f[0]["advice"], "Pair someone with Ann on core/ first; it is 95% theirs.")
162
+
138
163
  def test_nothing_when_spread(self):
139
164
  self.assertEqual(findings.bus_factor(report()), [])
140
165
 
@@ -234,6 +259,17 @@ class TightCoupling(unittest.TestCase):
234
259
  f = findings.tight_coupling(report(coupling=pairs))
235
260
  self.assertIn("1 pair changes together", f[0]["detail"])
236
261
 
262
+ def test_a_directory_of_files_that_change_as_one_is_one_cluster(self):
263
+ files = [f"rich/_unicode_data/unicode{n}.py" for n in ("10", "11", "12", "13")]
264
+ pairs = [{"entity": a, "coupled": b, "degree": 100, "average-revs": 5} for i, a in enumerate(files) for b in files[i + 1:]]
265
+ pairs.append({"entity": "rich/a.py", "coupled": "rich/b.py", "degree": 90, "average-revs": 8})
266
+ f = findings.tight_coupling(report(coupling=pairs))
267
+ self.assertIn("4 files in rich/_unicode_data/ change together at least 80% of the time, and 1 more pair does: rich/a.py + rich/b.py (90%).", f[0]["detail"])
268
+ self.assertNotIn("unicode10", f[0]["detail"])
269
+ self.assertTrue(f[0]["detail"].endswith("Review rich/_unicode_data/ first: 4 files change as one; a generator or a shared layout links them."), f[0]["detail"])
270
+ f = findings.tight_coupling(report(coupling=pairs[:-1]))
271
+ self.assertIn("4 files in rich/_unicode_data/ change together at least 80% of the time, 6 pairs in all.", f[0]["detail"])
272
+
237
273
  def test_nothing_when_no_tight_pairs(self):
238
274
  self.assertEqual(findings.tight_coupling(report()), [])
239
275
 
@@ -338,6 +374,14 @@ class BrainMethods(unittest.TestCase):
338
374
  self.assertNotIn("test_all", f[0]["detail"])
339
375
  self.assertEqual(findings.brain_methods(report(functions=fns[:1])), [])
340
376
 
377
+ def test_vendored_functions_are_not_brain_methods(self):
378
+ fns = [{"file": "vendor/github.com/google/jsonschema-go/jsonschema/validate.go", "function": "validate", "ccn": 179, "nloc": 424, "params": 3, "start": 1, "end": 424},
379
+ {"file": "processor/workers.go", "function": "countLoopGeneric", "ccn": 56, "nloc": 164, "params": 8, "start": 1, "end": 164}]
380
+ f = findings.brain_methods(report(functions=fns))
381
+ self.assertEqual(f[0]["advice"], "Split countLoopGeneric in processor/workers.go first, before the next change lands there.")
382
+ self.assertNotIn("vendor/", f[0]["detail"])
383
+ self.assertEqual(findings.brain_methods(report(functions=fns[:1])), [])
384
+
341
385
  def test_a_partial_run_says_there_may_be_more(self):
342
386
  r = report(functions=self.FUNCS)
343
387
  self.assertNotIn("part way", findings.brain_methods(r)[0]["detail"])
@@ -402,6 +446,29 @@ class KnowledgeIslands(unittest.TestCase):
402
446
  self.assertEqual(f[0]["advice"], "Pair someone with Bob on core/ first; it is the largest at 300 lines.")
403
447
  self.assertNotIn("tests/", f[0]["detail"])
404
448
 
449
+ def test_areas_no_longer_in_the_tree_are_not_islands(self):
450
+ own = [{"entity": "src/a.rs", "author": "Ann", "added": 30000, "deleted": 0}, # moved to crates/ years ago
451
+ {"entity": "grep-printer/a.rs", "author": "Ann", "added": 20000, "deleted": 0},
452
+ {"entity": "crates/core/a.rs", "author": "Bob", "added": 300, "deleted": 0}]
453
+ tree = {"files": {"crates/core/a.rs": {"code": 300, "complexity": 1}}}
454
+ f = findings.knowledge_islands(report(ownership=own, size=tree))
455
+ self.assertEqual(f[0]["advice"], "Pair someone with Bob on crates/core/ first; it is the largest at 300 lines.",
456
+ "with the vanished directories gone, crates/ holds everything and the map descends into it")
457
+ self.assertNotIn("src/", f[0]["detail"])
458
+ self.assertIn("100% of all lines added", f[0]["detail"], "lines in vanished directories are not in the denominator")
459
+ f = findings.knowledge_islands(report(ownership=own))
460
+ self.assertIn("src/", f[0]["detail"], "without a tree listing every area counts")
461
+
462
+ def test_vendored_trees_are_not_islands(self):
463
+ own = [{"entity": "vendor/github.com/x/a.go", "author": "Ann", "added": 500000, "deleted": 0},
464
+ {"entity": "web/node_modules/y/b.js", "author": "Ann", "added": 90000, "deleted": 0},
465
+ {"entity": "core/a.py", "author": "Bob", "added": 300, "deleted": 0}]
466
+ f = findings.knowledge_islands(report(ownership=own))
467
+ self.assertEqual(f[0]["advice"], "Pair someone with Bob on core/ first; it is the largest at 300 lines.")
468
+ self.assertNotIn("vendor/", f[0]["detail"])
469
+ self.assertNotIn("web/", f[0]["detail"])
470
+ self.assertIn("100% of all lines added", f[0]["detail"], "vendored lines are not in the denominator either")
471
+
405
472
  def test_nothing_when_shared(self):
406
473
  self.assertEqual(findings.knowledge_islands(report(ownership=self.OWN[2:])), [])
407
474
  self.assertEqual(findings.knowledge_islands(report()), [])
@@ -465,6 +532,18 @@ class KnowledgeLoss(unittest.TestCase):
465
532
  self.assertIn("Areas mostly theirs: old/ (100%), docs/ (100%)", f[0]["detail"])
466
533
  self.assertEqual(f[0]["advice"], "Pair someone on old/ first; nobody who wrote it is around to ask.")
467
534
 
535
+ def test_areas_no_longer_in_the_tree_are_not_named(self):
536
+ r = self._report(theseus_authors={"Ann": 60, "Bob": 40},
537
+ age=[{"entity": "flask/a.py", "age-months": 2}, {"entity": "src/x.py", "age-months": 2}],
538
+ ownership=[{"entity": "flask/a.py", "author": "Bob", "added": 8000, "deleted": 0}, # the old layout, all Bob's
539
+ {"entity": "src/x.py", "author": "Bob", "added": 300, "deleted": 0},
540
+ {"entity": "app/b.py", "author": "Ann", "added": 900, "deleted": 0}],
541
+ size={"files": {"src/x.py": {"code": 1, "complexity": 0}, "app/b.py": {"code": 1, "complexity": 0}}})
542
+ f = findings.knowledge_loss(r)
543
+ self.assertIn("Areas mostly theirs: src/ (100%).", f[0]["detail"])
544
+ self.assertNotIn("flask/", f[0]["detail"])
545
+ self.assertEqual(f[0]["advice"], "Pair someone on src/ first; nobody who wrote it is around to ask.")
546
+
468
547
  def test_a_live_area_is_preferred_over_a_bigger_idle_one(self):
469
548
  r = self._report(theseus_authors={"Ann": 60, "Bob": 40},
470
549
  age=[{"entity": "old/a.py", "age-months": 30}, {"entity": "live/b.py", "age-months": 3}],
@@ -48,6 +48,36 @@ class OddNames(unittest.TestCase):
48
48
  self.assertEqual(areas[0]["lines"], 1000)
49
49
 
50
50
 
51
+ class InTree(unittest.TestCase):
52
+ TREE = {"crates/core/a.rs": {}, "crates/ignore/src/b.rs": {}, "build.rs": {}}
53
+
54
+ def test_an_area_is_in_the_tree_when_any_tracked_file_sits_under_it(self):
55
+ self.assertTrue(knowledge.in_tree("crates/", self.TREE))
56
+ self.assertTrue(knowledge.in_tree("crates/ignore/", self.TREE))
57
+ self.assertFalse(knowledge.in_tree("src/", self.TREE))
58
+ self.assertFalse(knowledge.in_tree("crate/", self.TREE), "a prefix of a directory name is not that directory")
59
+
60
+ def test_root_files_are_in_the_tree_when_any_file_has_no_directory(self):
61
+ self.assertTrue(knowledge.in_tree(knowledge.ROOT, self.TREE))
62
+ self.assertFalse(knowledge.in_tree(knowledge.ROOT, {"crates/core/a.rs": {}}))
63
+
64
+ def test_no_tree_listing_means_every_area_counts(self):
65
+ self.assertTrue(knowledge.in_tree("src/", {}))
66
+
67
+ def test_present_rows_drop_vanished_top_level_directories_so_the_survivor_can_dominate(self):
68
+ rows = [{"entity": "src/a.rs", "author": "Ann", "added": 30000, "deleted": 0}, # the layout before crates/
69
+ {"entity": "crates/core/a.rs", "author": "Ann", "added": 9000, "deleted": 0},
70
+ {"entity": "crates/ignore/b.rs", "author": "Bob", "added": 900, "deleted": 0},
71
+ {"entity": "ci/x.sh", "author": "Ann", "added": 500, "deleted": 0}]
72
+ tree = {"crates/core/a.rs": {}, "crates/ignore/b.rs": {}, "ci/x.sh": {}}
73
+ kept = knowledge.present_rows(rows, tree)
74
+ self.assertEqual([r["entity"] for r in kept], ["crates/core/a.rs", "crates/ignore/b.rs", "ci/x.sh"])
75
+ self.assertEqual([a["area"] for a in knowledge.areas(kept)], ["crates/core/", "crates/ignore/", "ci/"],
76
+ "with src/ gone, crates/ holds over 80% and the map descends into it")
77
+ self.assertEqual([a["area"] for a in knowledge.areas(rows)], ["src/", "crates/", "ci/"], "the vanished src/ used to hide that")
78
+ self.assertEqual(knowledge.present_rows(rows, {}), rows)
79
+
80
+
51
81
  class Islands(unittest.TestCase):
52
82
  def test_areas_dominated_by_one_author(self):
53
83
  areas = knowledge.areas(OWNERSHIP)
@@ -59,6 +59,13 @@ class Placeholder(unittest.TestCase):
59
59
  for value in ["AKIA" + "X" * 16, "6L" + "x" * 38, "ghp_" + "a1" * 18, "yourkey" + "9" * 20]:
60
60
  self.assertFalse(leaks.is_placeholder(value), "a marker inside real-looking material is not enough: " + value)
61
61
 
62
+ def test_common_example_words_are_placeholders(self):
63
+ # `password: 'hello'` in a doc comment, `secret` in a sample config: the words every example uses
64
+ for value in ["hello", "Hello", "secret", "password", "PASSWORD", "example", "123456", "qwerty", "letmein", "foo", "dummy"]:
65
+ self.assertTrue(leaks.is_placeholder(value), value)
66
+ for value in ["hello123", "secret-9f8a7b6c5d4e", "s3cr3t!Passw0rd", "foobarbaz2024"]:
67
+ self.assertFalse(leaks.is_placeholder(value), "a word inside other material is not a placeholder: " + value)
68
+
62
69
  def test_anything_else_is_taken_seriously(self):
63
70
  # built at runtime: a literal in these shapes would trip secret scanners on this very file
64
71
  key_id, long_key = "AKIA" + "X" * 16, "6L" + "x" * 38
@@ -212,6 +212,7 @@ class Report(unittest.TestCase):
212
212
  def test_knowledge_map_caption_says_gone_is_measured_over_the_whole_history(self):
213
213
  r = sample_report()
214
214
  r["meta"].update({"last_date": "2026-09-10", "bots": []})
215
+ r["size"]["files"]["tests/t.py"] = {"code": 1, "complexity": 0} # every area in the tree, so nothing is hidden
215
216
  r["activity"]["authors"] = {"Ann": {"commits": 1, "added": 0, "deleted": 0, "first": "2025-01-01", "last": "2026-09-01"},
216
217
  "Bob": {"commits": 1, "added": 0, "deleted": 0, "first": "2025-01-01", "last": "2025-01-01"}}
217
218
  def caption(rep):
@@ -244,6 +245,7 @@ class Report(unittest.TestCase):
244
245
  self.assertEqual(caption(r, True), "trend sampled for the top 10 hotspots")
245
246
  self.assertIsNone(caption(r, False), "the tight report keeps its captions short")
246
247
  r["revisions"] = [{"entity": f"f{i}.py", "n-revs": 100 - i} for i in range(60)]
248
+ r["size"]["files"].update({f"f{i}.py": {"code": 10, "complexity": 0} for i in range(60)}) # in the tree, so not hidden as deleted
247
249
  self.assertEqual(caption(r, "markdown"), "and 10 more; trend sampled for the top 10 hotspots")
248
250
 
249
251
  def test_watch_list_caption_reports_the_backtest_or_why_not(self):
@@ -289,10 +291,10 @@ class Report(unittest.TestCase):
289
291
  def test_default_coupling_hides_test_pairs_and_says_so(self):
290
292
  r = sample_report()
291
293
  r["coupling"].append({"entity": "static/tax.html", "coupled": "tests/test_tax.py", "degree": 100, "average-revs": 11})
292
- text = rendered(r, [])
294
+ text = rendered(r, [], width=200)
293
295
  coupling = text[text.index("Change coupling"):]
294
296
  self.assertNotIn("tests/test_tax.py", coupling)
295
- self.assertIn("1 test pair hidden; --full shows them", coupling)
297
+ self.assertIn("1 test pair hidden; 1 historical pair hidden; --full shows them", coupling, "one suffix for every hidden count")
296
298
  full_text = rendered(r, [], full=True)
297
299
  self.assertIn("tests/test_tax.py", full_text[full_text.index("Change coupling"):])
298
300
 
@@ -306,6 +308,16 @@ class Report(unittest.TestCase):
306
308
  full_text = rendered(r, [], full=True)
307
309
  self.assertIn("tests/test_a.py", full_text[full_text.index("Complex functions"):])
308
310
 
311
+ def test_default_complex_functions_hide_vendored_code_and_say_so(self):
312
+ r = sample_report()
313
+ r["functions"].append({"file": "vendor/github.com/x/y.go", "function": "validate", "ccn": 179, "nloc": 424, "params": 3, "start": 1, "end": 424})
314
+ r["functions"].append({"file": "tests/test_a.py", "function": "test_thing", "ccn": 40, "nloc": 50, "params": 0, "start": 1, "end": 50})
315
+ fn = _section_text(rendered(r, [], width=200), "Complex functions")
316
+ self.assertNotIn("vendor/", fn)
317
+ self.assertIn("1 function in a test file hidden; 1 function in vendored code hidden; --full shows them", fn)
318
+ full = _section_text(rendered(r, [], width=200, full=True), "Complex functions")
319
+ self.assertIn("vendor/github.com/x/y.go", full)
320
+
309
321
  def test_hotspots_with_only_test_files_say_what_was_hidden(self):
310
322
  r = sample_report()
311
323
  r["revisions"] = [{"entity": "tests/test_a.py", "n-revs": 200}]
@@ -325,6 +337,41 @@ class Report(unittest.TestCase):
325
337
  self.assertIn("static/tax.html", full)
326
338
  self.assertNotIn("hidden", full)
327
339
 
340
+ def test_default_hotspots_hide_deleted_files_and_say_so(self):
341
+ r = sample_report() # the tree holds static/index.html and static/apps-metadata.json only
342
+ r["revisions"].append({"entity": "src/sizes/old.go", "n-revs": 40})
343
+ hot = _section_text(rendered(r, [], width=200), "◆ Hotspots")
344
+ self.assertNotIn("src/sizes/old.go", hot)
345
+ self.assertIn("1 deleted file hidden; --full shows them", hot)
346
+ full = _section_text(rendered(r, [], width=200, full=True), "◆ Hotspots")
347
+ self.assertIn("src/sizes/old.go", full)
348
+ self.assertNotIn("hidden", full)
349
+
350
+ def test_hotspots_without_a_tree_listing_hide_nothing(self):
351
+ r = sample_report()
352
+ r["size"]["files"] = {}
353
+ hot = _section_text(rendered(r, [], width=200), "◆ Hotspots")
354
+ self.assertIn("static/index.html", hot)
355
+ self.assertNotIn("deleted", hot)
356
+
357
+ def test_default_coupling_collapses_a_directory_that_changes_as_one(self):
358
+ r = sample_report()
359
+ files = [f"rich/_unicode_data/unicode{n}.py" for n in ("10", "11", "12", "13")]
360
+ for f in files:
361
+ r["size"]["files"][f] = {"code": 600, "complexity": 0}
362
+ r["coupling"] = [{"entity": a, "coupled": b, "degree": 100, "average-revs": 5} for i, a in enumerate(files) for b in files[i + 1:]]
363
+ r["coupling"].append({"entity": "static/index.html", "coupled": "static/apps-metadata.json", "degree": 90, "average-revs": 11})
364
+ coupling = _section_text(rendered(r, [], width=200), "Change coupling")
365
+ self.assertIn("rich/_unicode_data/ (4 files)", coupling)
366
+ self.assertIn("each other", coupling)
367
+ self.assertIn("≥100%", coupling)
368
+ self.assertNotIn("unicode10", coupling)
369
+ self.assertIn("static/index.html", coupling)
370
+ self.assertIn("6 pairs in 1 directory shown as one row; --full shows them", coupling)
371
+ full = _section_text(rendered(r, [], width=200, full=True), "Change coupling")
372
+ self.assertIn("unicode10", full)
373
+ self.assertNotIn("each other", full)
374
+
328
375
  def test_coupling_with_only_test_pairs_says_what_was_hidden(self):
329
376
  r = sample_report()
330
377
  r["coupling"] = [{"entity": "static/tax.html", "coupled": "tests/test_tax.py", "degree": 100, "average-revs": 11}]
@@ -590,6 +637,25 @@ class KnowledgeMap(unittest.TestCase):
590
637
  r["ownership"] = []
591
638
  self.assertIn("no ownership data", rendered(r, []))
592
639
 
640
+ def test_default_map_hides_areas_no_longer_in_the_tree_and_says_so(self):
641
+ r = sample_report() # the tree holds static/ files only; tests/t.py in the ownership rows is history
642
+ r["ownership"].append({"entity": "flask/app.py", "author": "Ann", "added": 4000, "deleted": 0})
643
+ km = _section_text(rendered(r, [], width=200), "Knowledge map")
644
+ self.assertIn("static/", km)
645
+ self.assertNotIn("flask/", km)
646
+ self.assertNotIn("tests/", km)
647
+ self.assertIn("2 historical areas hidden; --full shows them", km)
648
+ full = _section_text(rendered(r, [], width=200, full=True), "Knowledge map")
649
+ self.assertIn("flask/", full)
650
+ self.assertNotIn("hidden", full)
651
+
652
+ def test_map_without_a_tree_listing_hides_nothing(self):
653
+ r = sample_report()
654
+ r["size"]["files"] = {}
655
+ km = _section_text(rendered(r, [], width=200), "Knowledge map")
656
+ self.assertIn("tests/", km)
657
+ self.assertNotIn("historical", km)
658
+
593
659
 
594
660
  class Timeline(unittest.TestCase):
595
661
  def test_last_twelve_months_per_author_with_dots_for_zero(self):
@@ -866,6 +932,7 @@ class Markdown(unittest.TestCase):
866
932
  r = sample_report()
867
933
  r["coupling"] = []
868
934
  r["revisions"] = [{"entity": "weird|name.py", "n-revs": 3}]
935
+ r["size"]["files"]["weird|name.py"] = {"code": 5, "complexity": 0} # in the tree, so not hidden as deleted
869
936
  md = render.markdown(r, [])
870
937
  self.assertIn("weird\\|name.py", md)
871
938
  self.assertIn("_no pairs with 5+ shared revisions_", md)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes