gitmole 0.18.0__tar.gz → 0.19.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. {gitmole-0.18.0 → gitmole-0.19.0}/PKG-INFO +1 -1
  2. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/__init__.py +1 -1
  3. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/findings.py +109 -1
  4. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/knowledge.py +34 -0
  5. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/load.py +14 -2
  6. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/maat.py +100 -1
  7. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/render.py +14 -2
  8. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/watch.py +22 -0
  9. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/PKG-INFO +1 -1
  10. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_findings.py +56 -1
  11. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_maat.py +53 -2
  12. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_render.py +12 -5
  13. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_watch.py +15 -0
  14. {gitmole-0.18.0 → gitmole-0.19.0}/LICENSE +0 -0
  15. {gitmole-0.18.0 → gitmole-0.19.0}/README.md +0 -0
  16. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/__main__.py +0 -0
  17. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/backtest.py +0 -0
  18. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/banner.py +0 -0
  19. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/blame.py +0 -0
  20. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/classify.py +0 -0
  21. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/clean.py +0 -0
  22. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/cli.py +0 -0
  23. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/compare.py +0 -0
  24. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/coupling.py +0 -0
  25. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/deps.py +0 -0
  26. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/duplicates.py +0 -0
  27. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/evaluate.py +0 -0
  28. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/filetypes.py +0 -0
  29. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/functions.py +0 -0
  30. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/hook.py +0 -0
  31. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/hotspots.py +0 -0
  32. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/hygiene.py +0 -0
  33. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/identity.py +0 -0
  34. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/leaks.py +0 -0
  35. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/loss.py +0 -0
  36. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/provenance.py +0 -0
  37. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/run.py +0 -0
  38. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/sarif.py +0 -0
  39. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/signing.py +0 -0
  40. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/structure.py +0 -0
  41. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/szz.py +0 -0
  42. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/textfmt.py +0 -0
  43. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/trend.py +0 -0
  44. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/SOURCES.txt +0 -0
  45. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/dependency_links.txt +0 -0
  46. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/entry_points.txt +0 -0
  47. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/requires.txt +0 -0
  48. {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/top_level.txt +0 -0
  49. {gitmole-0.18.0 → gitmole-0.19.0}/pyproject.toml +0 -0
  50. {gitmole-0.18.0 → gitmole-0.19.0}/setup.cfg +0 -0
  51. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_backtest.py +0 -0
  52. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_banner.py +0 -0
  53. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_blame.py +0 -0
  54. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_classify.py +0 -0
  55. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_clean.py +0 -0
  56. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_cli.py +0 -0
  57. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_compare.py +0 -0
  58. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_coupling.py +0 -0
  59. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_deps.py +0 -0
  60. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_duplicates.py +0 -0
  61. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_evaluate.py +0 -0
  62. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_filetypes.py +0 -0
  63. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_functions.py +0 -0
  64. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_golden.py +0 -0
  65. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_hook.py +0 -0
  66. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_hotspots.py +0 -0
  67. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_hygiene.py +0 -0
  68. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_identity.py +0 -0
  69. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_knowledge.py +0 -0
  70. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_leaks.py +0 -0
  71. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_load.py +0 -0
  72. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_loss.py +0 -0
  73. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_packaging.py +0 -0
  74. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_provenance.py +0 -0
  75. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_render_examples.py +0 -0
  76. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_run.py +0 -0
  77. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_sarif.py +0 -0
  78. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_signing.py +0 -0
  79. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_structure.py +0 -0
  80. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_szz.py +0 -0
  81. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_textfmt.py +0 -0
  82. {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_trend.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.18.0
3
+ Version: 0.19.0
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.18.0"
3
+ __version__ = "0.19.0"
@@ -1101,10 +1101,118 @@ def signoff_by_co_author(report: dict, min_commits: int = 2) -> list:
1101
1101
  evidence={"identities": rows[:10]})]
1102
1102
 
1103
1103
 
1104
+ def _pool_files(report: dict) -> list:
1105
+ """The source files still in the tree that no classifier reason sets aside."""
1106
+ from . import classify
1107
+ cls = classify.Classifier(report)
1108
+ return sorted(f for f in _tree(report) if cls.reason(f) is None)
1109
+
1110
+
1111
+ def _authors_of(report: dict, files: list, key: str = "is_author") -> dict:
1112
+ wanted = set(files)
1113
+ out = {f: set() for f in files}
1114
+ for r in report.get("doa") or []:
1115
+ if r["entity"] in wanted and r.get(key):
1116
+ out[r["entity"]].add(r["author"])
1117
+ return out
1118
+
1119
+
1120
+ def truck_factor(report: dict, min_files: int = 20, area_files: int = 10) -> list:
1121
+ """Avelino et al.'s truck factor over the degree of authorship: how many people have to leave before
1122
+ more than half the source files have no author. One is a warning, two a note. Changes rather than
1123
+ lines, and a creator's bonus, so it can disagree with the surviving-code share, which the bus-factor
1124
+ finding reads; the finding says so when it does. Also per area, and with knowledge halving every
1125
+ five months."""
1126
+ if not report.get("doa"):
1127
+ return []
1128
+ files = _pool_files(report)
1129
+ authored = {f: a for f, a in _authors_of(report, files).items()}
1130
+ if len(files) < min_files:
1131
+ return []
1132
+ tf, removed, share = knowledge.truck_factor(authored)
1133
+ tf_d, removed_d, _ = knowledge.truck_factor(_authors_of(report, files, "is_author_decayed"))
1134
+ depth = knowledge.depth_for(files)
1135
+ areas = {}
1136
+ for f in files:
1137
+ areas.setdefault(knowledge._area(f, depth), []).append(f)
1138
+ lone = []
1139
+ for area, fs in sorted(areas.items()):
1140
+ if len(fs) >= area_files and area != knowledge.ROOT:
1141
+ n, who, _ = knowledge.truck_factor({f: authored[f] for f in fs})
1142
+ if n == 1:
1143
+ lone.append((area, who[0]))
1144
+ if tf > 2 and not lone:
1145
+ return []
1146
+ orphans = round(share * len(files))
1147
+ statement = (f"Truck factor {tf}: without {textfmt.join_and(removed)}, {orphans} of the {len(files)} source files ({_pct(orphans, len(files))}) "
1148
+ f"have no author left.")
1149
+ if tf_d != tf:
1150
+ statement += f" With knowledge halving every five months it is {tf_d} ({textfmt.join_and(removed_d)})."
1151
+ if lone:
1152
+ statement += " Areas with a truck factor of one: " + ", ".join(f"{a} ({w})" for a, w in lone[:5]) + (f" and {len(lone) - 5} more" if len(lone) > 5 else "") + "."
1153
+ shares = report.get("theseus_authors") or {}
1154
+ if shares and removed:
1155
+ top, lines = max(shares.items(), key=lambda kv: kv[1])
1156
+ if top != removed[0]:
1157
+ statement += f" The surviving code's largest share is {top}'s ({_pct(lines, sum(shares.values()))}), which the bus-factor finding reads."
1158
+ first_area = next((a for a, w in lone if w == removed[0]), lone[0][0] if lone else None)
1159
+ advice = f"Pair someone with {removed[0]}" + (f" on {first_area}" if first_area else "") + " first; they author most of what would be left without an author."
1160
+ return [_f("warning" if tf == 1 else "info", "Truck factor", statement, advice,
1161
+ rule={"id": "truck_factor", "doa_author_share": 0.75, "doa_floor": 3.293, "orphan_share": 0.5, "decay_months": 5,
1162
+ "ref": "Avelino et al., ICPC 2016"},
1163
+ evidence={"truck_factor": tf, "removed": removed, "truck_factor_decayed": tf_d, "removed_decayed": removed_d,
1164
+ "files": len(files), "orphaned": orphans, "areas": [{"area": a, "author": w} for a, w in lone[:10]]})]
1165
+
1166
+
1167
+ def authors_gone(report: dict, min_files: int = 5) -> list:
1168
+ """Files whose every author by degree of authorship has stopped committing, while others still
1169
+ change them: "creator left, editors remain", knowledge the blame share cannot show."""
1170
+ if not report.get("doa"):
1171
+ return []
1172
+ months = report["meta"].get("gone_months", loss.DEFAULT_MONTHS)
1173
+ gone = {g["name"] for g in loss.gone(report, months)}
1174
+ fresh = {a["entity"] for a in report.get("age") or [] if a["age-months"] < 12}
1175
+ files = [f for f in _pool_files(report) if f in fresh]
1176
+ authored = _authors_of(report, files)
1177
+ left = [(f, sorted(a)) for f, a in authored.items() if a and a <= gone]
1178
+ if len(left) < min_files:
1179
+ return []
1180
+ listed = "; ".join(f"{f} ({textfmt.join_and(a)})" for f, a in left[:5]) + (f" and {len(left) - 5} more" if len(left) > 5 else "")
1181
+ return [_f("info", "Files whose authors have left", f"{len(left)} source files changed in the last year have no author still committing: {listed}.",
1182
+ f"Make the people who edit {left[0][0]} its authors: review its design with them and write down what only {left[0][1][0]} knew.",
1183
+ rule={"id": "authors_gone", "gone_months": months, "min_files": min_files, "ref": "Avelino et al., ICPC 2016"},
1184
+ evidence={"count": len(left), "files": [{"file": f, "authors": a} for f, a in left[:10]]})]
1185
+
1186
+
1187
+ def component_coupling(report: dict, min_degree: int = 30) -> list:
1188
+ """Components (top-level directories, or the level below a lone src/) that change together in a
1189
+ large share of their changes: coupling at the level of the architecture, where two files in one
1190
+ directory is only a layout."""
1191
+ rows = report.get("components") or []
1192
+ if not rows:
1193
+ return []
1194
+ depth = knowledge.depth_for(list(_tree(report)) or [r["entity"] + "x" for r in rows])
1195
+
1196
+ def aside(c):
1197
+ probe = c + "x.py"
1198
+ return filetypes.is_test_path(probe) or filetypes.is_sample_path(probe) or filetypes.is_doc_path(probe) or filetypes.is_vendor_path(probe)
1199
+ pairs = [r for r in rows if r["depth"] == depth and r["degree"] >= min_degree and not aside(r["entity"]) and not aside(r["coupled"])]
1200
+ if not pairs:
1201
+ return []
1202
+ listed = "; ".join(f"{p['entity']} and {p['coupled']} change together in {p['degree']}% of their changes ({p['shared']} shared)" for p in pairs[:3])
1203
+ more = f" ({len(pairs) - 3} more pairs)" if len(pairs) > 3 else ""
1204
+ first = pairs[0]
1205
+ return [_f("info", "Components that change together", f"{listed}{more}.",
1206
+ f"Look at what {first['entity']} and {first['coupled']} share: a change that keeps landing in both is an interface nobody named.",
1207
+ rule={"id": "component_coupling", "min_degree": min_degree, "depth": depth, "ref": "Tornhill, Your Code as a Crime Scene, 2024"},
1208
+ evidence={"pairs": [{"a": p["entity"], "b": p["coupled"], "degree": p["degree"], "shared": p["shared"]} for p in pairs[:10]]})]
1209
+
1210
+
1104
1211
  RULES = [dormant, secrets_found, credential_files, vulnerable_dependencies, placeholder_identity, bus_factor, sizer_concerns, hotspot_dominance, bug_magnets,
1105
1212
  minor_contributors, reverts, brain_methods, complexity_growth, tight_coupling, duplication, stale_files, knowledge_islands, knowledge_loss,
1106
1213
  sweeping_commits, tangled_commits, hygiene_findings, debt_in_hotspots, deep_nesting, hidden_coupling, unreferenced_files,
1107
- agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author]
1214
+ agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author,
1215
+ truck_factor, authors_gone, component_coupling]
1108
1216
 
1109
1217
 
1110
1218
  def evaluate(report: dict) -> list:
@@ -78,3 +78,37 @@ def islands(areas_list: list, min_lines: int = 200, min_share: float = 0.9) -> l
78
78
  if n / a["lines"] >= min_share:
79
79
  out.append({"area": a["area"], "owner": owner, "share": round(100 * n / a["lines"]), "lines": a["lines"]})
80
80
  return out
81
+
82
+
83
+ def truck_factor(authors_of: dict, orphan_share: float = 0.5) -> tuple:
84
+ """Avelino et al.'s truck factor: remove the person who authors the most files, again and again, until
85
+ more than half the files have no author left. (the number removed, their names in order, the share
86
+ orphaned at the end). authors_of: {file: set of names}."""
87
+ files = list(authors_of)
88
+ if not files:
89
+ return 0, [], 0.0
90
+ remaining = {f: set(a) for f, a in authors_of.items()}
91
+ removed = []
92
+
93
+ def orphaned():
94
+ return sum(1 for a in remaining.values() if not a) / len(files)
95
+ while orphaned() <= orphan_share:
96
+ counts = {}
97
+ for a in remaining.values():
98
+ for who in a:
99
+ counts[who] = counts.get(who, 0) + 1
100
+ if not counts:
101
+ break
102
+ top = min(counts, key=lambda w: (-counts[w], w))
103
+ removed.append(top)
104
+ for a in remaining.values():
105
+ a.discard(top)
106
+ return len(removed), removed, orphaned()
107
+
108
+
109
+ def depth_for(paths: list, dominant: float = 0.8) -> int:
110
+ """1 for top-level directories, 2 when one top-level directory holds `dominant` of the files (a lone
111
+ src/), as the knowledge map chooses."""
112
+ from collections import Counter
113
+ tops = Counter(_area(p, 1) for p in paths)
114
+ return 2 if tops and tops.most_common(1)[0][1] >= dominant * len(paths) and tops.most_common(1)[0][0] != ROOT else 1
@@ -82,7 +82,9 @@ def parse_scc(text: str, types=None) -> dict:
82
82
 
83
83
 
84
84
  NUMERIC_COLUMNS = {"n-revs", "degree", "average-revs", "n-authors", "age-months", "added", "deleted", "n-fixes", "recent-fixes", "tiny-revs",
85
- "minor", "soc", "partners", "n-sets", "with-tests", "periods"}
85
+ "minor", "soc", "partners", "n-sets", "with-tests", "periods", "fa", "dl", "ac", "is_author", "is_author_decayed", "late",
86
+ "depth", "shared"}
87
+ FLOAT_COLUMNS = {"doa", "doa_decayed", "hcm"}
86
88
 
87
89
 
88
90
  def parse_maat_csv(text: str) -> list:
@@ -91,10 +93,17 @@ def parse_maat_csv(text: str) -> list:
91
93
  return []
92
94
  out = []
93
95
  for row in csv.DictReader(io.StringIO(text)):
94
- out.append({k: (_num(v) if k in NUMERIC_COLUMNS else v) for k, v in row.items()})
96
+ out.append({k: (_num(v) if k in NUMERIC_COLUMNS else _float(v) if k in FLOAT_COLUMNS else v) for k, v in row.items()})
95
97
  return out
96
98
 
97
99
 
100
+ def _float(v):
101
+ try:
102
+ return float(v)
103
+ except (TypeError, ValueError):
104
+ return 0.0
105
+
106
+
98
107
  def _num(v):
99
108
  """An int for a numeric cell; 0 for a missing, empty or garbage one (a row cut short by a killed step)."""
100
109
  try:
@@ -318,6 +327,9 @@ def load_report(out_dir: str, nested: bool = True) -> dict:
318
327
  "soc": parse_maat_csv(_read(out_dir, "maat-soc.csv")), # sum of coupling; empty for an output directory from before 0.11
319
328
  "tests": parse_maat_csv(_read(out_dir, "maat-tests.csv")), # test co-change per production file; empty before 0.12
320
329
  "entropy": parse_maat_csv(_read(out_dir, "maat-entropy.csv")), # Hassan's change entropy per file; empty before 0.13
330
+ "doa": parse_maat_csv(_read(out_dir, "maat-doa.csv")), # degree of authorship per file and person; empty before 0.19
331
+ "latenight": parse_maat_csv(_read(out_dir, "maat-latenight.csv")),
332
+ "components": parse_maat_csv(_read(out_dir, "maat-components.csv")),
321
333
  "authors": parse_maat_csv(_read(out_dir, "maat-authors.csv")),
322
334
  "age": parse_maat_csv(_read(out_dir, "maat-age.csv")),
323
335
  "ownership": ownership,
@@ -381,6 +381,102 @@ def entropy(commits: list, now: str = None, decay: float = ENTROPY_DECAY) -> lis
381
381
  return rows
382
382
 
383
383
 
384
+ DOA_DECAY_MONTHS = 5 # JetBrains' Bus Factor Explorer: knowledge halves every five months
385
+ DOA_AUTHOR_SHARE = 0.75
386
+ DOA_FLOOR = 3.293
387
+
388
+
389
+ def _doa(fa: int, dl: float, ac: float) -> float:
390
+ """Avelino et al.'s degree of authorship: a creator's bonus, the author's own changes, and a
391
+ logarithmic dilution by everyone else's."""
392
+ return 3.293 + 1.098 * fa + 0.164 * dl - 0.321 * math.log(1 + ac)
393
+
394
+
395
+ def doa(commits: list, now: str = None) -> list:
396
+ """Per file and person: created it (the first commit that added lines to it; a pure move creates
397
+ nothing), their changes, others' changes, the degree of authorship, and whether they count as an
398
+ author of it (DOA at least three quarters of the file's highest and at least 3.293), undecayed and
399
+ with knowledge halving every five months. Changes, not lines, so a reformat transfers nothing."""
400
+ now = dt.date.fromisoformat(now or dt.date.today().isoformat())
401
+ changes, decayed, first = defaultdict(Counter), defaultdict(Counter), {}
402
+ for c in commits:
403
+ weight = 0.5 ** (max(0, (now - dt.date.fromisoformat(c["date"])).days) / 30.44 / DOA_DECAY_MONTHS)
404
+ for p, added, _ in c["files"]:
405
+ for who in people(c):
406
+ changes[p][who] += 1
407
+ decayed[p][who] += weight
408
+ if added > 0 and (p not in first or (c["date"], c.get("time", "")) < first[p][0]):
409
+ first[p] = ((c["date"], c.get("time", "")), c["author"])
410
+ rows = []
411
+ for p, per in changes.items():
412
+ total, total_d = sum(per.values()), sum(decayed[p].values())
413
+ creator = first.get(p, (None, None))[1]
414
+ scores = {}
415
+ for who, n in per.items():
416
+ fa = int(who == creator)
417
+ scores[who] = (fa, n, total - n, _doa(fa, n, total - n), _doa(fa, decayed[p][who], total_d - decayed[p][who]))
418
+ top, top_d = max(s[3] for s in scores.values()), max(s[4] for s in scores.values())
419
+ for who, (fa, n, others, value, value_d) in sorted(scores.items()):
420
+ rows.append({"entity": p, "author": who, "fa": fa, "dl": n, "ac": others, "doa": round(value, 4), "doa_decayed": round(value_d, 4),
421
+ "is_author": int(value >= DOA_FLOOR and value >= DOA_AUTHOR_SHARE * top),
422
+ "is_author_decayed": int(value_d >= DOA_FLOOR and value_d >= DOA_AUTHOR_SHARE * top_d)})
423
+ rows.sort(key=lambda r: (r["entity"], r["author"]))
424
+ return rows
425
+
426
+
427
+ def _local_hour(stamp: str):
428
+ try:
429
+ return dt.datetime.fromisoformat(stamp[:-1] + "+00:00" if stamp.endswith("Z") else stamp).hour if len(stamp) > 10 else None
430
+ except ValueError:
431
+ return None
432
+
433
+
434
+ LATE_HOURS = range(0, 4) # Eyolfson, Tan and Lam: commits between midnight and 4 am, in the author's own time, were buggier
435
+
436
+
437
+ def latenight(commits: list) -> list:
438
+ """Per file: its revisions, and how many were committed between midnight and 4 am in the author's
439
+ own offset. A reason beside a file, never a rank: the effect is far weaker than churn or ownership."""
440
+ revs, late = Counter(), Counter()
441
+ for c in commits:
442
+ hour = _local_hour(c.get("time") or "")
443
+ for p, _, _ in c["files"]:
444
+ revs[p] += 1
445
+ late[p] += hour is not None and hour in LATE_HOURS
446
+ rows = [{"entity": p, "n-revs": n, "late": late[p]} for p, n in revs.items()]
447
+ rows.sort(key=lambda r: (-r["late"], r["entity"]))
448
+ return rows
449
+
450
+
451
+ def component(path: str, depth: int) -> str:
452
+ dirs = path.split("/")[:-1]
453
+ return "/".join(dirs[:depth]) + "/" if dirs else "(root files)"
454
+
455
+
456
+ def components(commits: list, min_shared: int = 10, min_degree: int = 20, max_components: int = 10) -> list:
457
+ """Coupling between components, the files truncated to their first one and two directories, over
458
+ the logical changes: two files in one directory changing together is a layout, `auth/` and
459
+ `billing/` changing together 40% of the time is architecture. A change that spans more than
460
+ `max_components` components is a sweep and couples nothing."""
461
+ out = []
462
+ for depth in (1, 2):
463
+ revs, shared = Counter(), Counter()
464
+ for c in changesets(commits):
465
+ comps = sorted({component(p, depth) for p, _, _ in c["files"]} - {"(root files)"})
466
+ if not comps or len(comps) > max_components:
467
+ continue
468
+ revs.update(comps)
469
+ for a, b in itertools.combinations(comps, 2):
470
+ shared[(a, b)] += 1
471
+ for (a, b), n in shared.items():
472
+ avg = (revs[a] + revs[b]) / 2
473
+ degree = int(math.floor(100 * n / avg + 0.5))
474
+ if n >= min_shared and degree >= min_degree:
475
+ out.append({"depth": depth, "entity": a, "coupled": b, "degree": degree, "shared": n, "average-revs": int(math.floor(avg + 0.5))})
476
+ out.sort(key=lambda r: (r["depth"], -r["degree"], -r["shared"], r["entity"], r["coupled"]))
477
+ return out
478
+
479
+
384
480
  RECENT_MONTHS = 6
385
481
  OVERSIZED_PERCENTILE = 0.99 # a fix changing more lines than this share of the history's commits credits nothing
386
482
  OVERSIZED_FLOOR = 500 # ...and never under this many lines, so a small repository's percentile does not bite
@@ -559,8 +655,11 @@ ANALYSES = {
559
655
  "entity-ownership": (entity_ownership, ["entity", "author", "added", "deleted"]),
560
656
  "fixes": (fixes, ["entity", "n-fixes", "last-fix", "recent-fixes"]),
561
657
  "entropy": (entropy, ["entity", "periods", "hcm"]),
658
+ "doa": (doa, ["entity", "author", "fa", "dl", "ac", "doa", "doa_decayed", "is_author", "is_author_decayed"]),
659
+ "latenight": (latenight, ["entity", "n-revs", "late"]),
660
+ "components": (components, ["depth", "entity", "coupled", "degree", "shared", "average-revs"]),
562
661
  }
563
- NEEDS_NOW = {"age", "fixes", "entropy"}
662
+ NEEDS_NOW = {"age", "fixes", "entropy", "doa"}
564
663
 
565
664
 
566
665
  def aliases_from_meta(path: str) -> dict:
@@ -525,6 +525,16 @@ def signing_section(report: dict, full: bool = True, width=None) -> dict:
525
525
  return _section("Signing by year", columns, rows, caption="; ".join(parts))
526
526
 
527
527
 
528
+ def watch_by_component_section(report: dict, full: bool = True, width=None) -> dict:
529
+ """The watch list's top files within each component: --full and Markdown only."""
530
+ groups = watch.by_component(watch.risks(report))
531
+ rows = [(g["component"], f"{g['share']:.0f}%", " · ".join(x["file"] for x in g["files"]))
532
+ for g in groups]
533
+ columns = [("component", PATH), ("share", RIGHT), ("top files", {"overflow": "fold", "ratio": 3})]
534
+ return _section("Watch list by component", columns, rows, note=None if rows else "no component holds 5% of the list's score",
535
+ caption="each component's share of the watch list's revisions × lines of code, and its own top files" if rows else None)
536
+
537
+
528
538
  def trailers_section(report: dict, full: bool = True, width=None) -> dict:
529
539
  """The trailer keys the history carries, with the cohort comparison and the neutral commit-shape
530
540
  descriptors below: --full and Markdown only. Read, never inferred; nothing is labelled."""
@@ -772,11 +782,11 @@ def compare_section(result: dict) -> dict:
772
782
  return _section("Since last report", columns, rows, note=note, caption="\n".join(lines))
773
783
 
774
784
 
775
- BUILDERS = [watch_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
785
+ BUILDERS = [watch_section, watch_by_component_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
776
786
  hotspots_section, coupling_section, signing_section, trailers_section, age_section, functions_section, health_section]
777
787
  # `--full` and Markdown only: Size, Activity and Code age are interesting once and rarely change what you
778
788
  # do next; Hotspots ranks the files the watch list already leads with, by the same product.
779
- FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers"}
789
+ FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers", "watch_by_component"}
780
790
 
781
791
 
782
792
  def sections(report: dict, full: bool = True, width=None) -> list:
@@ -1132,6 +1142,8 @@ def to_json(report: dict, findings: list, risk: dict = None, compare: dict = Non
1132
1142
  out = {**{k: v for k, v in report.items() if k != "backtest"}, "findings": findings, # the sub-report is a report of its own
1133
1143
  "watch": [{k: v for k, v in r.items() if k != "function"} | {"function": r["function"]["function"] if r["function"] else None}
1134
1144
  for r in watch.risks(report)[:WATCH_FULL]]}
1145
+ out["watch_by_component"] = [{"component": g["component"], "share": round(g["share"], 3), "files": [x["file"] for x in g["files"]]}
1146
+ for g in watch.by_component(watch.risks(report))]
1135
1147
  bt = watch.backtest(report)
1136
1148
  if bt is not None:
1137
1149
  out["watch_backtest"] = bt
@@ -31,6 +31,7 @@ PARTNERS_FLOOR = 20 # this many files it shares five or more commits with is
31
31
  DEBT_FLOOR = 3 # TODO/FIXME/XXX/HACK comments worth naming in a hot file
32
32
  NESTING_FLOOR = 5 # a function nested this deep is worth naming (CodeScene flags from 4)
33
33
  GOD_FILE = 60 # top-level functions, classes and methods in one file: a god file
34
+ LATE_FLOOR, LATE_SHARE = 3, 0.25 # commits between midnight and 4 am, the author's own time: this many, and this share
34
35
  PERIODS_FLOOR = 12 # changes in this many different months: scattered, Hassan's entropy signal, a reason and never a rank
35
36
  TESTED_SETS = 5 # this many changes before the share of them that moved a test says anything
36
37
  TESTED_SHARE = 0.2 # a test moved with at most this share of the file's changes: a hot file whose tests do not follow it
@@ -83,6 +84,7 @@ def risks(report: dict, min_revs: int = 2) -> list:
83
84
  has_tests = any(filetypes.is_test_path(p) for p in ((report.get("size") or {}).get("files") or {}))
84
85
  tested = {t["entity"]: (t["n-sets"], t["with-tests"]) for t in report.get("tests") or []} if has_tests else {}
85
86
  periods = {e["entity"]: e["periods"] for e in report.get("entropy") or []} # absent before 0.14
87
+ late = {e["entity"]: (e["late"], e["n-revs"]) for e in report.get("latenight") or []} # absent before 0.19
86
88
  shape = (report.get("structure") or {}).get("files") or {} # tree-sitter, with gitmole[structure]
87
89
  nested = {}
88
90
  for f in (report.get("structure") or {}).get("functions") or []:
@@ -105,6 +107,7 @@ def risks(report: dict, min_revs: int = 2) -> list:
105
107
  "authors": n_authors.get(h["entity"]), "owner": owner, "owner_share": share,
106
108
  "minor": minors.get(h["entity"], 0), "partners": partners.get(h["entity"], 0),
107
109
  "periods": periods.get(h["entity"]),
110
+ "late": late.get(h["entity"], (0, 0))[0], "late_revs": late.get(h["entity"], (0, 0))[1],
108
111
  "debt": (shape.get(h["entity"]) or {}).get("debt", 0), "definitions": (shape.get(h["entity"]) or {}).get("definitions", 0),
109
112
  "deepest": nested.get(h["entity"]),
110
113
  "changes": tested.get(h["entity"], (None, None))[0], "with_tests": tested.get(h["entity"], (None, None))[1],
@@ -185,11 +188,30 @@ def _reasons(r: dict) -> list:
185
188
  out.append(f"changes alongside {r['partners']} other files") # sum of coupling: weakly coupled to everything
186
189
  if r.get("definitions", 0) >= GOD_FILE:
187
190
  out.append(f"defines {r['definitions']} functions and classes")
191
+ if r.get("late", 0) >= LATE_FLOOR and r["late"] / max(1, r.get("late_revs") or 1) >= LATE_SHARE:
192
+ out.append(f"{round(100 * r['late'] / r['late_revs'])}% of its changes made between midnight and 4 am") # Eyolfson et al.: a tie-breaker, never a rank
188
193
  if (r.get("periods") or 0) >= PERIODS_FLOOR:
189
194
  out.append(f"changed in {r['periods']} different months") # Hassan's scatter: lost on the backtest, so a reason, not a rank
190
195
  return out
191
196
 
192
197
 
198
+ def by_component(rows: list, top: int = 3, min_share: float = 5.0, limit: int = 8) -> list:
199
+ """The watch list within each component (top-level directory, or the next level down when one holds
200
+ most of the files): one busy subtree otherwise takes the whole list. Components holding at least
201
+ `min_share` percent of the pool's score, largest first, each with its own top files."""
202
+ from . import knowledge
203
+ from .maat import component
204
+ if not rows:
205
+ return []
206
+ depth = knowledge.depth_for([r["file"] for r in rows])
207
+ groups = {}
208
+ for r in rows:
209
+ groups.setdefault(component(r["file"], depth), []).append(r)
210
+ out = [{"component": c, "share": sum(x["score"] for x in rs), "files": rs[:top]} for c, rs in groups.items()]
211
+ out.sort(key=lambda g: (-g["share"], g["component"]))
212
+ return [g for g in out if g["share"] >= min_share][:limit]
213
+
214
+
193
215
  REASONS_SHOWN = 6 # the default terminal report's cap per row; --full, Markdown and the JSON carry every reason
194
216
 
195
217
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.18.0
3
+ Version: 0.19.0
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -1260,7 +1260,8 @@ class References(unittest.TestCase):
1260
1260
  for rule, ref in expected.items():
1261
1261
  self.assertEqual(findings.REFS[rule], ref, rule)
1262
1262
  import os
1263
- page = open(os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "docs", "references.md")).read()
1263
+ with open(os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "docs", "references.md")) as fh:
1264
+ page = fh.read()
1264
1265
  for ref in expected.values():
1265
1266
  surname = ref.split(",")[0].split(" and ")[0].split(" et al.")[0]
1266
1267
  self.assertIn(surname, page, f"{ref} is on the references page")
@@ -1268,3 +1269,57 @@ class References(unittest.TestCase):
1268
1269
  def test_the_ref_reaches_the_rule_dict(self):
1269
1270
  found = findings.tight_coupling(report(coupling=[{"entity": "a.py", "coupled": "b.py", "degree": 90, "average-revs": 10}]))
1270
1271
  self.assertEqual(found[0]["rule"]["ref"], "Gall, Hajek and Jazayeri, ICSM 1998")
1272
+
1273
+
1274
+ class TruckFactor(unittest.TestCase):
1275
+ def rep(self, doa, **over):
1276
+ files = sorted({r["entity"] for r in doa})
1277
+ base = dict(size={"files": {f: {"code": 10, "complexity": 1} for f in files}}, doa=doa,
1278
+ revisions=[{"entity": f, "n-revs": 3} for f in files],
1279
+ meta={"name": "r", "commits": 100, "identities": [], "last_date": "2026-09-01", "gone_months": 12},
1280
+ activity={"authors_all": {"Ann": {"last": "2026-08-01"}, "Bob": {"last": "2026-08-01"}, "Cat": {"last": "2024-01-01"}}},
1281
+ age=[{"entity": f, "age-months": 1} for f in files])
1282
+ base.update(over)
1283
+ return report(**base)
1284
+
1285
+ def row(self, f, who, author=1, decayed=None):
1286
+ return {"entity": f, "author": who, "fa": 0, "dl": 1, "ac": 0, "doa": 4.0, "doa_decayed": 4.0, "is_author": author,
1287
+ "is_author_decayed": author if decayed is None else decayed}
1288
+
1289
+ def test_one_person_whose_departure_orphans_most_files(self):
1290
+ doa = [self.row(f"core/a{i}.py", "Ann") for i in range(20)] + [self.row(f"web/b{i}.py", "Bob") for i in range(8)]
1291
+ doa += [self.row("core/a0.py", "Bob", author=0)]
1292
+ found = {f["rule"]["id"]: f for f in findings.evaluate(self.rep(doa))}
1293
+ f = found["truck_factor"]
1294
+ self.assertEqual(f["severity"], "warning")
1295
+ self.assertIn("Truck factor 1: without Ann, 20 of the 28 source files (71%) have no author left", f["detail"])
1296
+ self.assertIn("core/ (Ann)", f["detail"], "an area whose own truck factor is one")
1297
+ self.assertEqual(f["rule"]["ref"], "Avelino et al., ICPC 2016")
1298
+ self.assertEqual(f["evidence"]["truck_factor"], 1)
1299
+
1300
+ def test_a_shared_codebase_has_none(self):
1301
+ doa = [self.row(f"core/a{i}.py", who) for i in range(30) for who in ("Ann", "Bob", "Cat")]
1302
+ self.assertNotIn("truck_factor", {f["rule"]["id"] for f in findings.evaluate(self.rep(doa))})
1303
+
1304
+ def test_files_whose_authors_all_left_while_others_still_edit_them(self):
1305
+ doa = [self.row(f"core/a{i}.py", "Cat") for i in range(6)] + [self.row(f"core/a{i}.py", "Bob", author=0) for i in range(6)]
1306
+ doa += [self.row(f"web/b{i}.py", who) for i in range(20) for who in ("Ann", "Bob")]
1307
+ f = {x["rule"]["id"]: x for x in findings.evaluate(self.rep(doa))}["authors_gone"]
1308
+ self.assertEqual(f["severity"], "info")
1309
+ self.assertIn("6 source files changed in the last year have no author still committing", f["detail"])
1310
+ self.assertIn("core/a0.py (Cat)", f["detail"])
1311
+
1312
+
1313
+ class ComponentCoupling(unittest.TestCase):
1314
+ def test_pairs_of_components_that_change_together(self):
1315
+ files = {f"{d}/f{i}.py": {"code": 10, "complexity": 1} for d in ("auth", "billing", "tests", "web") for i in range(5)}
1316
+ r = report(size={"files": files},
1317
+ components=[{"depth": 1, "entity": "auth/", "coupled": "billing/", "degree": 45, "shared": 30, "average-revs": 66},
1318
+ {"depth": 1, "entity": "auth/", "coupled": "tests/", "degree": 80, "shared": 50, "average-revs": 60},
1319
+ {"depth": 1, "entity": "billing/", "coupled": "web/", "degree": 22, "shared": 12, "average-revs": 50},
1320
+ {"depth": 2, "entity": "auth/x/", "coupled": "billing/y/", "degree": 90, "shared": 20, "average-revs": 22}])
1321
+ f = {x["rule"]["id"]: x for x in findings.evaluate(r)}["component_coupling"]
1322
+ self.assertIn("auth/ and billing/ change together in 45% of their changes (30 shared)", f["detail"])
1323
+ self.assertNotIn("tests/", f["detail"], "a component of tests changes with what it tests")
1324
+ self.assertNotIn("web/", f["detail"], "under the 30% floor")
1325
+ self.assertNotIn("auth/x/", f["detail"], "the depth is the one the tree's layout asks for")
@@ -1,4 +1,5 @@
1
1
  import json
2
+ import math
2
3
  import os
3
4
  import tempfile
4
5
  import unittest
@@ -349,8 +350,9 @@ class WriteAll(unittest.TestCase):
349
350
  fh.write(LOG)
350
351
  maat.write_all(log, d)
351
352
  names = sorted(n for n in os.listdir(d) if n.startswith("maat-"))
352
- self.assertEqual(names, ["maat-age.csv", "maat-authors.csv", "maat-coupling.csv", "maat-entity-ownership.csv", "maat-entropy.csv",
353
- "maat-fixes.csv", "maat-plumbing.csv", "maat-revisions.csv", "maat-soc.csv", "maat-tests.csv"])
353
+ self.assertEqual(names, ["maat-age.csv", "maat-authors.csv", "maat-components.csv", "maat-coupling.csv", "maat-doa.csv",
354
+ "maat-entity-ownership.csv", "maat-entropy.csv", "maat-fixes.csv", "maat-latenight.csv", "maat-plumbing.csv",
355
+ "maat-revisions.csv", "maat-soc.csv", "maat-tests.csv"])
354
356
  self.assertTrue(os.path.isfile(os.path.join(d, "activity.json")))
355
357
  with open(os.path.join(d, "maat-revisions.csv")) as fh:
356
358
  self.assertEqual(fh.readline().strip(), "entity,n-revs")
@@ -632,3 +634,52 @@ class ChangeEntropy(unittest.TestCase):
632
634
  maat.write_all(log, d, now="2026-09-15")
633
635
  with open(os.path.join(d, "maat-entropy.csv")) as fh:
634
636
  self.assertEqual(fh.readline().strip(), "entity,periods,hcm")
637
+
638
+
639
+ class DegreeOfAuthorship(unittest.TestCase):
640
+ def test_avelinos_doa_with_the_creator_bonus_and_dilution_by_others(self):
641
+ commits = [_commit("c1", [("a.py", 10, 0)], author="Ann", date="2026-01-01"),
642
+ _commit("c2", [("a.py", 2, 1)], author="Ann", date="2026-01-02"),
643
+ _commit("c3", [("a.py", 1, 1)], author="Bob", date="2026-01-03"),
644
+ _commit("m1", [("b.py", 0, 0)], author="Mover", date="2026-01-04"),
645
+ _commit("c4", [("b.py", 5, 0)], author="Cat", date="2026-01-05")]
646
+ rows = {(r["entity"], r["author"]): r for r in maat.doa(commits, now="2026-01-10")}
647
+ ann = rows[("a.py", "Ann")]
648
+ self.assertEqual((ann["fa"], ann["dl"], ann["ac"]), (1, 2, 1))
649
+ self.assertAlmostEqual(ann["doa"], 3.293 + 1.098 + 0.164 * 2 - 0.321 * math.log(2), places=3)
650
+ self.assertEqual(rows[("a.py", "Bob")]["fa"], 0)
651
+ self.assertEqual((rows[("b.py", "Cat")]["fa"], rows[("b.py", "Mover")]["fa"]), (1, 0), "a pure move adds no lines and creates nothing")
652
+ self.assertEqual(ann["is_author"], 1)
653
+ self.assertEqual(rows[("a.py", "Bob")]["is_author"], 0, "Bob's DOA is under the 3.293 floor")
654
+
655
+ def test_decay_halves_knowledge_every_five_months(self):
656
+ old = [_commit(f"o{i}", [("a.py", 1, 0)], author="Ann", date="2024-01-01") for i in range(10)]
657
+ new = [_commit(f"n{i}", [("a.py", 1, 0)], author="Bob", date="2026-01-01") for i in range(3)]
658
+ rows = {r["author"]: r for r in maat.doa(old + new, now="2026-01-01")}
659
+ self.assertEqual(rows["Ann"]["is_author"], 1, "undecayed, ten changes and creation outweigh three")
660
+ self.assertGreater(rows["Bob"]["doa_decayed"], rows["Ann"]["doa_decayed"] - 1.098, "decayed, Ann's two-year-old changes count for little")
661
+ self.assertEqual(rows["Bob"]["is_author_decayed"], 1)
662
+
663
+
664
+ class LateNight(unittest.TestCase):
665
+ def test_commits_between_midnight_and_four_in_the_authors_own_time(self):
666
+ commits = [dict(_commit("a", [("x.py", 1, 0)]), time="2026-01-05T01:30:00+09:00"),
667
+ dict(_commit("b", [("x.py", 1, 0)]), time="2026-01-05T03:59:00-05:00"),
668
+ dict(_commit("c", [("x.py", 1, 0)]), time="2026-01-05T04:00:00+00:00"),
669
+ dict(_commit("d", [("x.py", 1, 0), ("y.py", 1, 0)]), time="2026-01-05T23:00:00+00:00")]
670
+ rows = {r["entity"]: r for r in maat.latenight(commits)}
671
+ self.assertEqual(rows["x.py"], {"entity": "x.py", "n-revs": 4, "late": 2})
672
+ self.assertEqual(rows["y.py"]["late"], 0)
673
+
674
+
675
+ class Components(unittest.TestCase):
676
+ def test_coupling_between_top_level_components_over_logical_changes(self):
677
+ commits = []
678
+ for i in range(12):
679
+ commits.append(_commit(f"a{i}", [("auth/login.py", 1, 0), ("billing/charge.py", 1, 0)], date=f"2026-01-{1 + i:02d}"))
680
+ for i in range(12):
681
+ commits.append(_commit(f"b{i}", [("auth/token.py", 1, 0)], author="Bob", date=f"2026-02-{1 + i:02d}"))
682
+ commits.append(_commit("c", [("docs/x.md", 1, 0), ("auth/login.py", 1, 0)], date="2026-03-01"))
683
+ rows = [r for r in maat.components(commits) if r["depth"] == 1]
684
+ self.assertEqual(rows, [{"depth": 1, "entity": "auth/", "coupled": "billing/", "degree": 65, "shared": 12, "average-revs": 19}],
685
+ "12 shared changes over an average of (25 + 12) / 2; docs/ shares one change, under the floor")
@@ -292,6 +292,13 @@ class Report(unittest.TestCase):
292
292
  text = (sec.get("caption") or "") + (sec.get("note") or "")
293
293
  self.assertIn("the vulnerability database changed between the runs (2026-09-01 to 2026-09-17), so a dependency finding can move with no change to the code", text)
294
294
 
295
+ def test_the_watch_list_by_component_is_a_full_only_section(self):
296
+ r = sample_report()
297
+ self.assertNotIn("Watch list by component", rendered(r, [], width=200))
298
+ self.assertIn("Watch list by component", rendered(r, [], width=200, full=True))
299
+ self.assertIn("## Watch list by component", render.markdown(r, []))
300
+ self.assertIn("watch_by_component", render.to_json(r, []))
301
+
295
302
  def test_the_json_is_the_same_bytes_for_the_same_clone_whatever_the_run(self):
296
303
  import copy
297
304
  a = sample_report()
@@ -1307,13 +1314,13 @@ class Sections(unittest.TestCase):
1307
1314
  def test_sections_carry_title_columns_and_rows_in_report_order(self):
1308
1315
  secs = render.sections(sample_report(), full=True)
1309
1316
  titles = [x["title"] for x in secs]
1310
- self.assertEqual(titles[:5], ["Watch list", "Size by language", "People", "Knowledge map", "Activity"])
1311
- self.assertTrue(titles[5].startswith("Timeline"))
1312
- self.assertTrue(titles[6].startswith("Hotspots"))
1317
+ self.assertEqual(titles[:6], ["Watch list", "Watch list by component", "Size by language", "People", "Knowledge map", "Activity"])
1318
+ self.assertTrue(titles[6].startswith("Timeline"))
1319
+ self.assertTrue(titles[7].startswith("Hotspots"))
1313
1320
  self.assertEqual(titles[-2], "Complex functions")
1314
1321
  self.assertEqual(titles[-1], "Repo health (git-sizer concerns)")
1315
- self.assertEqual([x["id"] for x in secs][:4], ["watch", "size", "people", "knowledge"])
1316
- size = secs[1]
1322
+ self.assertEqual([x["id"] for x in secs][:5], ["watch", "watch_by_component", "size", "people", "knowledge"])
1323
+ size = secs[2]
1317
1324
  self.assertEqual(size["columns"][:3], ["language", "files", "code"])
1318
1325
  self.assertEqual(size["rows"][0][0], "HTML")
1319
1326
 
@@ -100,6 +100,21 @@ class Risks(unittest.TestCase):
100
100
  self.assertIn("defines 72 functions and classes", reasons)
101
101
  self.assertFalse([x for x in by["core/util.py"]["reasons"] if "TODO" in x or "nested" in x or "defines" in x])
102
102
 
103
+ def test_late_night_changes_are_a_reason_and_never_a_rank(self):
104
+ r = report()
105
+ r["latenight"] = [{"entity": "core/parser.py", "n-revs": 40, "late": 12}, {"entity": "core/util.py", "n-revs": 30, "late": 2}]
106
+ by = {x["file"]: x for x in watch.risks(r)}
107
+ self.assertIn("30% of its changes made between midnight and 4 am", by["core/parser.py"]["reasons"])
108
+ self.assertNotIn("midnight", " ".join(by["core/util.py"]["reasons"]))
109
+ self.assertEqual([x["file"] for x in watch.risks(r)], [x["file"] for x in watch.risks(report())], "the rank does not move")
110
+
111
+ def test_the_watch_list_by_component(self):
112
+ r = report()
113
+ groups = watch.by_component(watch.risks(r), top=2)
114
+ self.assertEqual([(g["component"], [x["file"] for x in g["files"]]) for g in groups],
115
+ [("web/", ["web/index.html"]), ("core/", ["core/parser.py", "core/util.py"])])
116
+ self.assertAlmostEqual(sum(g["share"] for g in groups), 100.0)
117
+
103
118
  def test_tests_that_never_move_with_a_file_are_a_reason(self):
104
119
  r = report()
105
120
  r["tests"] = [{"entity": "core/parser.py", "n-sets": 38, "with-tests": 0}, {"entity": "core/util.py", "n-sets": 28, "with-tests": 4},
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes