gitmole 0.18.0__tar.gz → 0.19.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gitmole-0.18.0 → gitmole-0.19.0}/PKG-INFO +1 -1
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/__init__.py +1 -1
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/findings.py +109 -1
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/knowledge.py +34 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/load.py +14 -2
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/maat.py +100 -1
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/render.py +14 -2
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/watch.py +22 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/PKG-INFO +1 -1
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_findings.py +56 -1
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_maat.py +53 -2
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_render.py +12 -5
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_watch.py +15 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/LICENSE +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/README.md +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/__main__.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/backtest.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/banner.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/blame.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/classify.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/clean.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/cli.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/compare.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/coupling.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/deps.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/duplicates.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/evaluate.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/filetypes.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/functions.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/hook.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/hotspots.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/hygiene.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/identity.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/leaks.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/loss.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/provenance.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/run.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/sarif.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/signing.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/structure.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/szz.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/textfmt.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole/trend.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/SOURCES.txt +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/dependency_links.txt +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/entry_points.txt +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/requires.txt +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/gitmole.egg-info/top_level.txt +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/pyproject.toml +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/setup.cfg +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_backtest.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_banner.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_blame.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_classify.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_clean.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_cli.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_compare.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_coupling.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_deps.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_duplicates.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_evaluate.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_filetypes.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_functions.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_golden.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_hook.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_hotspots.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_hygiene.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_identity.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_knowledge.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_leaks.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_load.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_loss.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_packaging.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_provenance.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_render_examples.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_run.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_sarif.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_signing.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_structure.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_szz.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_textfmt.py +0 -0
- {gitmole-0.18.0 → gitmole-0.19.0}/tests/test_trend.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.19.0
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -1101,10 +1101,118 @@ def signoff_by_co_author(report: dict, min_commits: int = 2) -> list:
|
|
|
1101
1101
|
evidence={"identities": rows[:10]})]
|
|
1102
1102
|
|
|
1103
1103
|
|
|
1104
|
+
def _pool_files(report: dict) -> list:
|
|
1105
|
+
"""The source files still in the tree that no classifier reason sets aside."""
|
|
1106
|
+
from . import classify
|
|
1107
|
+
cls = classify.Classifier(report)
|
|
1108
|
+
return sorted(f for f in _tree(report) if cls.reason(f) is None)
|
|
1109
|
+
|
|
1110
|
+
|
|
1111
|
+
def _authors_of(report: dict, files: list, key: str = "is_author") -> dict:
|
|
1112
|
+
wanted = set(files)
|
|
1113
|
+
out = {f: set() for f in files}
|
|
1114
|
+
for r in report.get("doa") or []:
|
|
1115
|
+
if r["entity"] in wanted and r.get(key):
|
|
1116
|
+
out[r["entity"]].add(r["author"])
|
|
1117
|
+
return out
|
|
1118
|
+
|
|
1119
|
+
|
|
1120
|
+
def truck_factor(report: dict, min_files: int = 20, area_files: int = 10) -> list:
|
|
1121
|
+
"""Avelino et al.'s truck factor over the degree of authorship: how many people have to leave before
|
|
1122
|
+
more than half the source files have no author. One is a warning, two a note. Changes rather than
|
|
1123
|
+
lines, and a creator's bonus, so it can disagree with the surviving-code share, which the bus-factor
|
|
1124
|
+
finding reads; the finding says so when it does. Also per area, and with knowledge halving every
|
|
1125
|
+
five months."""
|
|
1126
|
+
if not report.get("doa"):
|
|
1127
|
+
return []
|
|
1128
|
+
files = _pool_files(report)
|
|
1129
|
+
authored = {f: a for f, a in _authors_of(report, files).items()}
|
|
1130
|
+
if len(files) < min_files:
|
|
1131
|
+
return []
|
|
1132
|
+
tf, removed, share = knowledge.truck_factor(authored)
|
|
1133
|
+
tf_d, removed_d, _ = knowledge.truck_factor(_authors_of(report, files, "is_author_decayed"))
|
|
1134
|
+
depth = knowledge.depth_for(files)
|
|
1135
|
+
areas = {}
|
|
1136
|
+
for f in files:
|
|
1137
|
+
areas.setdefault(knowledge._area(f, depth), []).append(f)
|
|
1138
|
+
lone = []
|
|
1139
|
+
for area, fs in sorted(areas.items()):
|
|
1140
|
+
if len(fs) >= area_files and area != knowledge.ROOT:
|
|
1141
|
+
n, who, _ = knowledge.truck_factor({f: authored[f] for f in fs})
|
|
1142
|
+
if n == 1:
|
|
1143
|
+
lone.append((area, who[0]))
|
|
1144
|
+
if tf > 2 and not lone:
|
|
1145
|
+
return []
|
|
1146
|
+
orphans = round(share * len(files))
|
|
1147
|
+
statement = (f"Truck factor {tf}: without {textfmt.join_and(removed)}, {orphans} of the {len(files)} source files ({_pct(orphans, len(files))}) "
|
|
1148
|
+
f"have no author left.")
|
|
1149
|
+
if tf_d != tf:
|
|
1150
|
+
statement += f" With knowledge halving every five months it is {tf_d} ({textfmt.join_and(removed_d)})."
|
|
1151
|
+
if lone:
|
|
1152
|
+
statement += " Areas with a truck factor of one: " + ", ".join(f"{a} ({w})" for a, w in lone[:5]) + (f" and {len(lone) - 5} more" if len(lone) > 5 else "") + "."
|
|
1153
|
+
shares = report.get("theseus_authors") or {}
|
|
1154
|
+
if shares and removed:
|
|
1155
|
+
top, lines = max(shares.items(), key=lambda kv: kv[1])
|
|
1156
|
+
if top != removed[0]:
|
|
1157
|
+
statement += f" The surviving code's largest share is {top}'s ({_pct(lines, sum(shares.values()))}), which the bus-factor finding reads."
|
|
1158
|
+
first_area = next((a for a, w in lone if w == removed[0]), lone[0][0] if lone else None)
|
|
1159
|
+
advice = f"Pair someone with {removed[0]}" + (f" on {first_area}" if first_area else "") + " first; they author most of what would be left without an author."
|
|
1160
|
+
return [_f("warning" if tf == 1 else "info", "Truck factor", statement, advice,
|
|
1161
|
+
rule={"id": "truck_factor", "doa_author_share": 0.75, "doa_floor": 3.293, "orphan_share": 0.5, "decay_months": 5,
|
|
1162
|
+
"ref": "Avelino et al., ICPC 2016"},
|
|
1163
|
+
evidence={"truck_factor": tf, "removed": removed, "truck_factor_decayed": tf_d, "removed_decayed": removed_d,
|
|
1164
|
+
"files": len(files), "orphaned": orphans, "areas": [{"area": a, "author": w} for a, w in lone[:10]]})]
|
|
1165
|
+
|
|
1166
|
+
|
|
1167
|
+
def authors_gone(report: dict, min_files: int = 5) -> list:
|
|
1168
|
+
"""Files whose every author by degree of authorship has stopped committing, while others still
|
|
1169
|
+
change them: "creator left, editors remain", knowledge the blame share cannot show."""
|
|
1170
|
+
if not report.get("doa"):
|
|
1171
|
+
return []
|
|
1172
|
+
months = report["meta"].get("gone_months", loss.DEFAULT_MONTHS)
|
|
1173
|
+
gone = {g["name"] for g in loss.gone(report, months)}
|
|
1174
|
+
fresh = {a["entity"] for a in report.get("age") or [] if a["age-months"] < 12}
|
|
1175
|
+
files = [f for f in _pool_files(report) if f in fresh]
|
|
1176
|
+
authored = _authors_of(report, files)
|
|
1177
|
+
left = [(f, sorted(a)) for f, a in authored.items() if a and a <= gone]
|
|
1178
|
+
if len(left) < min_files:
|
|
1179
|
+
return []
|
|
1180
|
+
listed = "; ".join(f"{f} ({textfmt.join_and(a)})" for f, a in left[:5]) + (f" and {len(left) - 5} more" if len(left) > 5 else "")
|
|
1181
|
+
return [_f("info", "Files whose authors have left", f"{len(left)} source files changed in the last year have no author still committing: {listed}.",
|
|
1182
|
+
f"Make the people who edit {left[0][0]} its authors: review its design with them and write down what only {left[0][1][0]} knew.",
|
|
1183
|
+
rule={"id": "authors_gone", "gone_months": months, "min_files": min_files, "ref": "Avelino et al., ICPC 2016"},
|
|
1184
|
+
evidence={"count": len(left), "files": [{"file": f, "authors": a} for f, a in left[:10]]})]
|
|
1185
|
+
|
|
1186
|
+
|
|
1187
|
+
def component_coupling(report: dict, min_degree: int = 30) -> list:
|
|
1188
|
+
"""Components (top-level directories, or the level below a lone src/) that change together in a
|
|
1189
|
+
large share of their changes: coupling at the level of the architecture, where two files in one
|
|
1190
|
+
directory is only a layout."""
|
|
1191
|
+
rows = report.get("components") or []
|
|
1192
|
+
if not rows:
|
|
1193
|
+
return []
|
|
1194
|
+
depth = knowledge.depth_for(list(_tree(report)) or [r["entity"] + "x" for r in rows])
|
|
1195
|
+
|
|
1196
|
+
def aside(c):
|
|
1197
|
+
probe = c + "x.py"
|
|
1198
|
+
return filetypes.is_test_path(probe) or filetypes.is_sample_path(probe) or filetypes.is_doc_path(probe) or filetypes.is_vendor_path(probe)
|
|
1199
|
+
pairs = [r for r in rows if r["depth"] == depth and r["degree"] >= min_degree and not aside(r["entity"]) and not aside(r["coupled"])]
|
|
1200
|
+
if not pairs:
|
|
1201
|
+
return []
|
|
1202
|
+
listed = "; ".join(f"{p['entity']} and {p['coupled']} change together in {p['degree']}% of their changes ({p['shared']} shared)" for p in pairs[:3])
|
|
1203
|
+
more = f" ({len(pairs) - 3} more pairs)" if len(pairs) > 3 else ""
|
|
1204
|
+
first = pairs[0]
|
|
1205
|
+
return [_f("info", "Components that change together", f"{listed}{more}.",
|
|
1206
|
+
f"Look at what {first['entity']} and {first['coupled']} share: a change that keeps landing in both is an interface nobody named.",
|
|
1207
|
+
rule={"id": "component_coupling", "min_degree": min_degree, "depth": depth, "ref": "Tornhill, Your Code as a Crime Scene, 2024"},
|
|
1208
|
+
evidence={"pairs": [{"a": p["entity"], "b": p["coupled"], "degree": p["degree"], "shared": p["shared"]} for p in pairs[:10]]})]
|
|
1209
|
+
|
|
1210
|
+
|
|
1104
1211
|
RULES = [dormant, secrets_found, credential_files, vulnerable_dependencies, placeholder_identity, bus_factor, sizer_concerns, hotspot_dominance, bug_magnets,
|
|
1105
1212
|
minor_contributors, reverts, brain_methods, complexity_growth, tight_coupling, duplication, stale_files, knowledge_islands, knowledge_loss,
|
|
1106
1213
|
sweeping_commits, tangled_commits, hygiene_findings, debt_in_hotspots, deep_nesting, hidden_coupling, unreferenced_files,
|
|
1107
|
-
agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author
|
|
1214
|
+
agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author,
|
|
1215
|
+
truck_factor, authors_gone, component_coupling]
|
|
1108
1216
|
|
|
1109
1217
|
|
|
1110
1218
|
def evaluate(report: dict) -> list:
|
|
@@ -78,3 +78,37 @@ def islands(areas_list: list, min_lines: int = 200, min_share: float = 0.9) -> l
|
|
|
78
78
|
if n / a["lines"] >= min_share:
|
|
79
79
|
out.append({"area": a["area"], "owner": owner, "share": round(100 * n / a["lines"]), "lines": a["lines"]})
|
|
80
80
|
return out
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def truck_factor(authors_of: dict, orphan_share: float = 0.5) -> tuple:
|
|
84
|
+
"""Avelino et al.'s truck factor: remove the person who authors the most files, again and again, until
|
|
85
|
+
more than half the files have no author left. (the number removed, their names in order, the share
|
|
86
|
+
orphaned at the end). authors_of: {file: set of names}."""
|
|
87
|
+
files = list(authors_of)
|
|
88
|
+
if not files:
|
|
89
|
+
return 0, [], 0.0
|
|
90
|
+
remaining = {f: set(a) for f, a in authors_of.items()}
|
|
91
|
+
removed = []
|
|
92
|
+
|
|
93
|
+
def orphaned():
|
|
94
|
+
return sum(1 for a in remaining.values() if not a) / len(files)
|
|
95
|
+
while orphaned() <= orphan_share:
|
|
96
|
+
counts = {}
|
|
97
|
+
for a in remaining.values():
|
|
98
|
+
for who in a:
|
|
99
|
+
counts[who] = counts.get(who, 0) + 1
|
|
100
|
+
if not counts:
|
|
101
|
+
break
|
|
102
|
+
top = min(counts, key=lambda w: (-counts[w], w))
|
|
103
|
+
removed.append(top)
|
|
104
|
+
for a in remaining.values():
|
|
105
|
+
a.discard(top)
|
|
106
|
+
return len(removed), removed, orphaned()
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def depth_for(paths: list, dominant: float = 0.8) -> int:
|
|
110
|
+
"""1 for top-level directories, 2 when one top-level directory holds `dominant` of the files (a lone
|
|
111
|
+
src/), as the knowledge map chooses."""
|
|
112
|
+
from collections import Counter
|
|
113
|
+
tops = Counter(_area(p, 1) for p in paths)
|
|
114
|
+
return 2 if tops and tops.most_common(1)[0][1] >= dominant * len(paths) and tops.most_common(1)[0][0] != ROOT else 1
|
|
@@ -82,7 +82,9 @@ def parse_scc(text: str, types=None) -> dict:
|
|
|
82
82
|
|
|
83
83
|
|
|
84
84
|
NUMERIC_COLUMNS = {"n-revs", "degree", "average-revs", "n-authors", "age-months", "added", "deleted", "n-fixes", "recent-fixes", "tiny-revs",
|
|
85
|
-
"minor", "soc", "partners", "n-sets", "with-tests", "periods"
|
|
85
|
+
"minor", "soc", "partners", "n-sets", "with-tests", "periods", "fa", "dl", "ac", "is_author", "is_author_decayed", "late",
|
|
86
|
+
"depth", "shared"}
|
|
87
|
+
FLOAT_COLUMNS = {"doa", "doa_decayed", "hcm"}
|
|
86
88
|
|
|
87
89
|
|
|
88
90
|
def parse_maat_csv(text: str) -> list:
|
|
@@ -91,10 +93,17 @@ def parse_maat_csv(text: str) -> list:
|
|
|
91
93
|
return []
|
|
92
94
|
out = []
|
|
93
95
|
for row in csv.DictReader(io.StringIO(text)):
|
|
94
|
-
out.append({k: (_num(v) if k in NUMERIC_COLUMNS else v) for k, v in row.items()})
|
|
96
|
+
out.append({k: (_num(v) if k in NUMERIC_COLUMNS else _float(v) if k in FLOAT_COLUMNS else v) for k, v in row.items()})
|
|
95
97
|
return out
|
|
96
98
|
|
|
97
99
|
|
|
100
|
+
def _float(v):
|
|
101
|
+
try:
|
|
102
|
+
return float(v)
|
|
103
|
+
except (TypeError, ValueError):
|
|
104
|
+
return 0.0
|
|
105
|
+
|
|
106
|
+
|
|
98
107
|
def _num(v):
|
|
99
108
|
"""An int for a numeric cell; 0 for a missing, empty or garbage one (a row cut short by a killed step)."""
|
|
100
109
|
try:
|
|
@@ -318,6 +327,9 @@ def load_report(out_dir: str, nested: bool = True) -> dict:
|
|
|
318
327
|
"soc": parse_maat_csv(_read(out_dir, "maat-soc.csv")), # sum of coupling; empty for an output directory from before 0.11
|
|
319
328
|
"tests": parse_maat_csv(_read(out_dir, "maat-tests.csv")), # test co-change per production file; empty before 0.12
|
|
320
329
|
"entropy": parse_maat_csv(_read(out_dir, "maat-entropy.csv")), # Hassan's change entropy per file; empty before 0.13
|
|
330
|
+
"doa": parse_maat_csv(_read(out_dir, "maat-doa.csv")), # degree of authorship per file and person; empty before 0.19
|
|
331
|
+
"latenight": parse_maat_csv(_read(out_dir, "maat-latenight.csv")),
|
|
332
|
+
"components": parse_maat_csv(_read(out_dir, "maat-components.csv")),
|
|
321
333
|
"authors": parse_maat_csv(_read(out_dir, "maat-authors.csv")),
|
|
322
334
|
"age": parse_maat_csv(_read(out_dir, "maat-age.csv")),
|
|
323
335
|
"ownership": ownership,
|
|
@@ -381,6 +381,102 @@ def entropy(commits: list, now: str = None, decay: float = ENTROPY_DECAY) -> lis
|
|
|
381
381
|
return rows
|
|
382
382
|
|
|
383
383
|
|
|
384
|
+
DOA_DECAY_MONTHS = 5 # JetBrains' Bus Factor Explorer: knowledge halves every five months
|
|
385
|
+
DOA_AUTHOR_SHARE = 0.75
|
|
386
|
+
DOA_FLOOR = 3.293
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _doa(fa: int, dl: float, ac: float) -> float:
|
|
390
|
+
"""Avelino et al.'s degree of authorship: a creator's bonus, the author's own changes, and a
|
|
391
|
+
logarithmic dilution by everyone else's."""
|
|
392
|
+
return 3.293 + 1.098 * fa + 0.164 * dl - 0.321 * math.log(1 + ac)
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def doa(commits: list, now: str = None) -> list:
|
|
396
|
+
"""Per file and person: created it (the first commit that added lines to it; a pure move creates
|
|
397
|
+
nothing), their changes, others' changes, the degree of authorship, and whether they count as an
|
|
398
|
+
author of it (DOA at least three quarters of the file's highest and at least 3.293), undecayed and
|
|
399
|
+
with knowledge halving every five months. Changes, not lines, so a reformat transfers nothing."""
|
|
400
|
+
now = dt.date.fromisoformat(now or dt.date.today().isoformat())
|
|
401
|
+
changes, decayed, first = defaultdict(Counter), defaultdict(Counter), {}
|
|
402
|
+
for c in commits:
|
|
403
|
+
weight = 0.5 ** (max(0, (now - dt.date.fromisoformat(c["date"])).days) / 30.44 / DOA_DECAY_MONTHS)
|
|
404
|
+
for p, added, _ in c["files"]:
|
|
405
|
+
for who in people(c):
|
|
406
|
+
changes[p][who] += 1
|
|
407
|
+
decayed[p][who] += weight
|
|
408
|
+
if added > 0 and (p not in first or (c["date"], c.get("time", "")) < first[p][0]):
|
|
409
|
+
first[p] = ((c["date"], c.get("time", "")), c["author"])
|
|
410
|
+
rows = []
|
|
411
|
+
for p, per in changes.items():
|
|
412
|
+
total, total_d = sum(per.values()), sum(decayed[p].values())
|
|
413
|
+
creator = first.get(p, (None, None))[1]
|
|
414
|
+
scores = {}
|
|
415
|
+
for who, n in per.items():
|
|
416
|
+
fa = int(who == creator)
|
|
417
|
+
scores[who] = (fa, n, total - n, _doa(fa, n, total - n), _doa(fa, decayed[p][who], total_d - decayed[p][who]))
|
|
418
|
+
top, top_d = max(s[3] for s in scores.values()), max(s[4] for s in scores.values())
|
|
419
|
+
for who, (fa, n, others, value, value_d) in sorted(scores.items()):
|
|
420
|
+
rows.append({"entity": p, "author": who, "fa": fa, "dl": n, "ac": others, "doa": round(value, 4), "doa_decayed": round(value_d, 4),
|
|
421
|
+
"is_author": int(value >= DOA_FLOOR and value >= DOA_AUTHOR_SHARE * top),
|
|
422
|
+
"is_author_decayed": int(value_d >= DOA_FLOOR and value_d >= DOA_AUTHOR_SHARE * top_d)})
|
|
423
|
+
rows.sort(key=lambda r: (r["entity"], r["author"]))
|
|
424
|
+
return rows
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def _local_hour(stamp: str):
|
|
428
|
+
try:
|
|
429
|
+
return dt.datetime.fromisoformat(stamp[:-1] + "+00:00" if stamp.endswith("Z") else stamp).hour if len(stamp) > 10 else None
|
|
430
|
+
except ValueError:
|
|
431
|
+
return None
|
|
432
|
+
|
|
433
|
+
|
|
434
|
+
LATE_HOURS = range(0, 4) # Eyolfson, Tan and Lam: commits between midnight and 4 am, in the author's own time, were buggier
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def latenight(commits: list) -> list:
|
|
438
|
+
"""Per file: its revisions, and how many were committed between midnight and 4 am in the author's
|
|
439
|
+
own offset. A reason beside a file, never a rank: the effect is far weaker than churn or ownership."""
|
|
440
|
+
revs, late = Counter(), Counter()
|
|
441
|
+
for c in commits:
|
|
442
|
+
hour = _local_hour(c.get("time") or "")
|
|
443
|
+
for p, _, _ in c["files"]:
|
|
444
|
+
revs[p] += 1
|
|
445
|
+
late[p] += hour is not None and hour in LATE_HOURS
|
|
446
|
+
rows = [{"entity": p, "n-revs": n, "late": late[p]} for p, n in revs.items()]
|
|
447
|
+
rows.sort(key=lambda r: (-r["late"], r["entity"]))
|
|
448
|
+
return rows
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def component(path: str, depth: int) -> str:
|
|
452
|
+
dirs = path.split("/")[:-1]
|
|
453
|
+
return "/".join(dirs[:depth]) + "/" if dirs else "(root files)"
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def components(commits: list, min_shared: int = 10, min_degree: int = 20, max_components: int = 10) -> list:
|
|
457
|
+
"""Coupling between components, the files truncated to their first one and two directories, over
|
|
458
|
+
the logical changes: two files in one directory changing together is a layout, `auth/` and
|
|
459
|
+
`billing/` changing together 40% of the time is architecture. A change that spans more than
|
|
460
|
+
`max_components` components is a sweep and couples nothing."""
|
|
461
|
+
out = []
|
|
462
|
+
for depth in (1, 2):
|
|
463
|
+
revs, shared = Counter(), Counter()
|
|
464
|
+
for c in changesets(commits):
|
|
465
|
+
comps = sorted({component(p, depth) for p, _, _ in c["files"]} - {"(root files)"})
|
|
466
|
+
if not comps or len(comps) > max_components:
|
|
467
|
+
continue
|
|
468
|
+
revs.update(comps)
|
|
469
|
+
for a, b in itertools.combinations(comps, 2):
|
|
470
|
+
shared[(a, b)] += 1
|
|
471
|
+
for (a, b), n in shared.items():
|
|
472
|
+
avg = (revs[a] + revs[b]) / 2
|
|
473
|
+
degree = int(math.floor(100 * n / avg + 0.5))
|
|
474
|
+
if n >= min_shared and degree >= min_degree:
|
|
475
|
+
out.append({"depth": depth, "entity": a, "coupled": b, "degree": degree, "shared": n, "average-revs": int(math.floor(avg + 0.5))})
|
|
476
|
+
out.sort(key=lambda r: (r["depth"], -r["degree"], -r["shared"], r["entity"], r["coupled"]))
|
|
477
|
+
return out
|
|
478
|
+
|
|
479
|
+
|
|
384
480
|
RECENT_MONTHS = 6
|
|
385
481
|
OVERSIZED_PERCENTILE = 0.99 # a fix changing more lines than this share of the history's commits credits nothing
|
|
386
482
|
OVERSIZED_FLOOR = 500 # ...and never under this many lines, so a small repository's percentile does not bite
|
|
@@ -559,8 +655,11 @@ ANALYSES = {
|
|
|
559
655
|
"entity-ownership": (entity_ownership, ["entity", "author", "added", "deleted"]),
|
|
560
656
|
"fixes": (fixes, ["entity", "n-fixes", "last-fix", "recent-fixes"]),
|
|
561
657
|
"entropy": (entropy, ["entity", "periods", "hcm"]),
|
|
658
|
+
"doa": (doa, ["entity", "author", "fa", "dl", "ac", "doa", "doa_decayed", "is_author", "is_author_decayed"]),
|
|
659
|
+
"latenight": (latenight, ["entity", "n-revs", "late"]),
|
|
660
|
+
"components": (components, ["depth", "entity", "coupled", "degree", "shared", "average-revs"]),
|
|
562
661
|
}
|
|
563
|
-
NEEDS_NOW = {"age", "fixes", "entropy"}
|
|
662
|
+
NEEDS_NOW = {"age", "fixes", "entropy", "doa"}
|
|
564
663
|
|
|
565
664
|
|
|
566
665
|
def aliases_from_meta(path: str) -> dict:
|
|
@@ -525,6 +525,16 @@ def signing_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
525
525
|
return _section("Signing by year", columns, rows, caption="; ".join(parts))
|
|
526
526
|
|
|
527
527
|
|
|
528
|
+
def watch_by_component_section(report: dict, full: bool = True, width=None) -> dict:
|
|
529
|
+
"""The watch list's top files within each component: --full and Markdown only."""
|
|
530
|
+
groups = watch.by_component(watch.risks(report))
|
|
531
|
+
rows = [(g["component"], f"{g['share']:.0f}%", " · ".join(x["file"] for x in g["files"]))
|
|
532
|
+
for g in groups]
|
|
533
|
+
columns = [("component", PATH), ("share", RIGHT), ("top files", {"overflow": "fold", "ratio": 3})]
|
|
534
|
+
return _section("Watch list by component", columns, rows, note=None if rows else "no component holds 5% of the list's score",
|
|
535
|
+
caption="each component's share of the watch list's revisions × lines of code, and its own top files" if rows else None)
|
|
536
|
+
|
|
537
|
+
|
|
528
538
|
def trailers_section(report: dict, full: bool = True, width=None) -> dict:
|
|
529
539
|
"""The trailer keys the history carries, with the cohort comparison and the neutral commit-shape
|
|
530
540
|
descriptors below: --full and Markdown only. Read, never inferred; nothing is labelled."""
|
|
@@ -772,11 +782,11 @@ def compare_section(result: dict) -> dict:
|
|
|
772
782
|
return _section("Since last report", columns, rows, note=note, caption="\n".join(lines))
|
|
773
783
|
|
|
774
784
|
|
|
775
|
-
BUILDERS = [watch_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
|
|
785
|
+
BUILDERS = [watch_section, watch_by_component_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
|
|
776
786
|
hotspots_section, coupling_section, signing_section, trailers_section, age_section, functions_section, health_section]
|
|
777
787
|
# `--full` and Markdown only: Size, Activity and Code age are interesting once and rarely change what you
|
|
778
788
|
# do next; Hotspots ranks the files the watch list already leads with, by the same product.
|
|
779
|
-
FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers"}
|
|
789
|
+
FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers", "watch_by_component"}
|
|
780
790
|
|
|
781
791
|
|
|
782
792
|
def sections(report: dict, full: bool = True, width=None) -> list:
|
|
@@ -1132,6 +1142,8 @@ def to_json(report: dict, findings: list, risk: dict = None, compare: dict = Non
|
|
|
1132
1142
|
out = {**{k: v for k, v in report.items() if k != "backtest"}, "findings": findings, # the sub-report is a report of its own
|
|
1133
1143
|
"watch": [{k: v for k, v in r.items() if k != "function"} | {"function": r["function"]["function"] if r["function"] else None}
|
|
1134
1144
|
for r in watch.risks(report)[:WATCH_FULL]]}
|
|
1145
|
+
out["watch_by_component"] = [{"component": g["component"], "share": round(g["share"], 3), "files": [x["file"] for x in g["files"]]}
|
|
1146
|
+
for g in watch.by_component(watch.risks(report))]
|
|
1135
1147
|
bt = watch.backtest(report)
|
|
1136
1148
|
if bt is not None:
|
|
1137
1149
|
out["watch_backtest"] = bt
|
|
@@ -31,6 +31,7 @@ PARTNERS_FLOOR = 20 # this many files it shares five or more commits with is
|
|
|
31
31
|
DEBT_FLOOR = 3 # TODO/FIXME/XXX/HACK comments worth naming in a hot file
|
|
32
32
|
NESTING_FLOOR = 5 # a function nested this deep is worth naming (CodeScene flags from 4)
|
|
33
33
|
GOD_FILE = 60 # top-level functions, classes and methods in one file: a god file
|
|
34
|
+
LATE_FLOOR, LATE_SHARE = 3, 0.25 # commits between midnight and 4 am, the author's own time: this many, and this share
|
|
34
35
|
PERIODS_FLOOR = 12 # changes in this many different months: scattered, Hassan's entropy signal, a reason and never a rank
|
|
35
36
|
TESTED_SETS = 5 # this many changes before the share of them that moved a test says anything
|
|
36
37
|
TESTED_SHARE = 0.2 # a test moved with at most this share of the file's changes: a hot file whose tests do not follow it
|
|
@@ -83,6 +84,7 @@ def risks(report: dict, min_revs: int = 2) -> list:
|
|
|
83
84
|
has_tests = any(filetypes.is_test_path(p) for p in ((report.get("size") or {}).get("files") or {}))
|
|
84
85
|
tested = {t["entity"]: (t["n-sets"], t["with-tests"]) for t in report.get("tests") or []} if has_tests else {}
|
|
85
86
|
periods = {e["entity"]: e["periods"] for e in report.get("entropy") or []} # absent before 0.14
|
|
87
|
+
late = {e["entity"]: (e["late"], e["n-revs"]) for e in report.get("latenight") or []} # absent before 0.19
|
|
86
88
|
shape = (report.get("structure") or {}).get("files") or {} # tree-sitter, with gitmole[structure]
|
|
87
89
|
nested = {}
|
|
88
90
|
for f in (report.get("structure") or {}).get("functions") or []:
|
|
@@ -105,6 +107,7 @@ def risks(report: dict, min_revs: int = 2) -> list:
|
|
|
105
107
|
"authors": n_authors.get(h["entity"]), "owner": owner, "owner_share": share,
|
|
106
108
|
"minor": minors.get(h["entity"], 0), "partners": partners.get(h["entity"], 0),
|
|
107
109
|
"periods": periods.get(h["entity"]),
|
|
110
|
+
"late": late.get(h["entity"], (0, 0))[0], "late_revs": late.get(h["entity"], (0, 0))[1],
|
|
108
111
|
"debt": (shape.get(h["entity"]) or {}).get("debt", 0), "definitions": (shape.get(h["entity"]) or {}).get("definitions", 0),
|
|
109
112
|
"deepest": nested.get(h["entity"]),
|
|
110
113
|
"changes": tested.get(h["entity"], (None, None))[0], "with_tests": tested.get(h["entity"], (None, None))[1],
|
|
@@ -185,11 +188,30 @@ def _reasons(r: dict) -> list:
|
|
|
185
188
|
out.append(f"changes alongside {r['partners']} other files") # sum of coupling: weakly coupled to everything
|
|
186
189
|
if r.get("definitions", 0) >= GOD_FILE:
|
|
187
190
|
out.append(f"defines {r['definitions']} functions and classes")
|
|
191
|
+
if r.get("late", 0) >= LATE_FLOOR and r["late"] / max(1, r.get("late_revs") or 1) >= LATE_SHARE:
|
|
192
|
+
out.append(f"{round(100 * r['late'] / r['late_revs'])}% of its changes made between midnight and 4 am") # Eyolfson et al.: a tie-breaker, never a rank
|
|
188
193
|
if (r.get("periods") or 0) >= PERIODS_FLOOR:
|
|
189
194
|
out.append(f"changed in {r['periods']} different months") # Hassan's scatter: lost on the backtest, so a reason, not a rank
|
|
190
195
|
return out
|
|
191
196
|
|
|
192
197
|
|
|
198
|
+
def by_component(rows: list, top: int = 3, min_share: float = 5.0, limit: int = 8) -> list:
|
|
199
|
+
"""The watch list within each component (top-level directory, or the next level down when one holds
|
|
200
|
+
most of the files): one busy subtree otherwise takes the whole list. Components holding at least
|
|
201
|
+
`min_share` percent of the pool's score, largest first, each with its own top files."""
|
|
202
|
+
from . import knowledge
|
|
203
|
+
from .maat import component
|
|
204
|
+
if not rows:
|
|
205
|
+
return []
|
|
206
|
+
depth = knowledge.depth_for([r["file"] for r in rows])
|
|
207
|
+
groups = {}
|
|
208
|
+
for r in rows:
|
|
209
|
+
groups.setdefault(component(r["file"], depth), []).append(r)
|
|
210
|
+
out = [{"component": c, "share": sum(x["score"] for x in rs), "files": rs[:top]} for c, rs in groups.items()]
|
|
211
|
+
out.sort(key=lambda g: (-g["share"], g["component"]))
|
|
212
|
+
return [g for g in out if g["share"] >= min_share][:limit]
|
|
213
|
+
|
|
214
|
+
|
|
193
215
|
REASONS_SHOWN = 6 # the default terminal report's cap per row; --full, Markdown and the JSON carry every reason
|
|
194
216
|
|
|
195
217
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.19.0
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -1260,7 +1260,8 @@ class References(unittest.TestCase):
|
|
|
1260
1260
|
for rule, ref in expected.items():
|
|
1261
1261
|
self.assertEqual(findings.REFS[rule], ref, rule)
|
|
1262
1262
|
import os
|
|
1263
|
-
|
|
1263
|
+
with open(os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "docs", "references.md")) as fh:
|
|
1264
|
+
page = fh.read()
|
|
1264
1265
|
for ref in expected.values():
|
|
1265
1266
|
surname = ref.split(",")[0].split(" and ")[0].split(" et al.")[0]
|
|
1266
1267
|
self.assertIn(surname, page, f"{ref} is on the references page")
|
|
@@ -1268,3 +1269,57 @@ class References(unittest.TestCase):
|
|
|
1268
1269
|
def test_the_ref_reaches_the_rule_dict(self):
|
|
1269
1270
|
found = findings.tight_coupling(report(coupling=[{"entity": "a.py", "coupled": "b.py", "degree": 90, "average-revs": 10}]))
|
|
1270
1271
|
self.assertEqual(found[0]["rule"]["ref"], "Gall, Hajek and Jazayeri, ICSM 1998")
|
|
1272
|
+
|
|
1273
|
+
|
|
1274
|
+
class TruckFactor(unittest.TestCase):
|
|
1275
|
+
def rep(self, doa, **over):
|
|
1276
|
+
files = sorted({r["entity"] for r in doa})
|
|
1277
|
+
base = dict(size={"files": {f: {"code": 10, "complexity": 1} for f in files}}, doa=doa,
|
|
1278
|
+
revisions=[{"entity": f, "n-revs": 3} for f in files],
|
|
1279
|
+
meta={"name": "r", "commits": 100, "identities": [], "last_date": "2026-09-01", "gone_months": 12},
|
|
1280
|
+
activity={"authors_all": {"Ann": {"last": "2026-08-01"}, "Bob": {"last": "2026-08-01"}, "Cat": {"last": "2024-01-01"}}},
|
|
1281
|
+
age=[{"entity": f, "age-months": 1} for f in files])
|
|
1282
|
+
base.update(over)
|
|
1283
|
+
return report(**base)
|
|
1284
|
+
|
|
1285
|
+
def row(self, f, who, author=1, decayed=None):
|
|
1286
|
+
return {"entity": f, "author": who, "fa": 0, "dl": 1, "ac": 0, "doa": 4.0, "doa_decayed": 4.0, "is_author": author,
|
|
1287
|
+
"is_author_decayed": author if decayed is None else decayed}
|
|
1288
|
+
|
|
1289
|
+
def test_one_person_whose_departure_orphans_most_files(self):
|
|
1290
|
+
doa = [self.row(f"core/a{i}.py", "Ann") for i in range(20)] + [self.row(f"web/b{i}.py", "Bob") for i in range(8)]
|
|
1291
|
+
doa += [self.row("core/a0.py", "Bob", author=0)]
|
|
1292
|
+
found = {f["rule"]["id"]: f for f in findings.evaluate(self.rep(doa))}
|
|
1293
|
+
f = found["truck_factor"]
|
|
1294
|
+
self.assertEqual(f["severity"], "warning")
|
|
1295
|
+
self.assertIn("Truck factor 1: without Ann, 20 of the 28 source files (71%) have no author left", f["detail"])
|
|
1296
|
+
self.assertIn("core/ (Ann)", f["detail"], "an area whose own truck factor is one")
|
|
1297
|
+
self.assertEqual(f["rule"]["ref"], "Avelino et al., ICPC 2016")
|
|
1298
|
+
self.assertEqual(f["evidence"]["truck_factor"], 1)
|
|
1299
|
+
|
|
1300
|
+
def test_a_shared_codebase_has_none(self):
|
|
1301
|
+
doa = [self.row(f"core/a{i}.py", who) for i in range(30) for who in ("Ann", "Bob", "Cat")]
|
|
1302
|
+
self.assertNotIn("truck_factor", {f["rule"]["id"] for f in findings.evaluate(self.rep(doa))})
|
|
1303
|
+
|
|
1304
|
+
def test_files_whose_authors_all_left_while_others_still_edit_them(self):
|
|
1305
|
+
doa = [self.row(f"core/a{i}.py", "Cat") for i in range(6)] + [self.row(f"core/a{i}.py", "Bob", author=0) for i in range(6)]
|
|
1306
|
+
doa += [self.row(f"web/b{i}.py", who) for i in range(20) for who in ("Ann", "Bob")]
|
|
1307
|
+
f = {x["rule"]["id"]: x for x in findings.evaluate(self.rep(doa))}["authors_gone"]
|
|
1308
|
+
self.assertEqual(f["severity"], "info")
|
|
1309
|
+
self.assertIn("6 source files changed in the last year have no author still committing", f["detail"])
|
|
1310
|
+
self.assertIn("core/a0.py (Cat)", f["detail"])
|
|
1311
|
+
|
|
1312
|
+
|
|
1313
|
+
class ComponentCoupling(unittest.TestCase):
|
|
1314
|
+
def test_pairs_of_components_that_change_together(self):
|
|
1315
|
+
files = {f"{d}/f{i}.py": {"code": 10, "complexity": 1} for d in ("auth", "billing", "tests", "web") for i in range(5)}
|
|
1316
|
+
r = report(size={"files": files},
|
|
1317
|
+
components=[{"depth": 1, "entity": "auth/", "coupled": "billing/", "degree": 45, "shared": 30, "average-revs": 66},
|
|
1318
|
+
{"depth": 1, "entity": "auth/", "coupled": "tests/", "degree": 80, "shared": 50, "average-revs": 60},
|
|
1319
|
+
{"depth": 1, "entity": "billing/", "coupled": "web/", "degree": 22, "shared": 12, "average-revs": 50},
|
|
1320
|
+
{"depth": 2, "entity": "auth/x/", "coupled": "billing/y/", "degree": 90, "shared": 20, "average-revs": 22}])
|
|
1321
|
+
f = {x["rule"]["id"]: x for x in findings.evaluate(r)}["component_coupling"]
|
|
1322
|
+
self.assertIn("auth/ and billing/ change together in 45% of their changes (30 shared)", f["detail"])
|
|
1323
|
+
self.assertNotIn("tests/", f["detail"], "a component of tests changes with what it tests")
|
|
1324
|
+
self.assertNotIn("web/", f["detail"], "under the 30% floor")
|
|
1325
|
+
self.assertNotIn("auth/x/", f["detail"], "the depth is the one the tree's layout asks for")
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import json
|
|
2
|
+
import math
|
|
2
3
|
import os
|
|
3
4
|
import tempfile
|
|
4
5
|
import unittest
|
|
@@ -349,8 +350,9 @@ class WriteAll(unittest.TestCase):
|
|
|
349
350
|
fh.write(LOG)
|
|
350
351
|
maat.write_all(log, d)
|
|
351
352
|
names = sorted(n for n in os.listdir(d) if n.startswith("maat-"))
|
|
352
|
-
self.assertEqual(names, ["maat-age.csv", "maat-authors.csv", "maat-
|
|
353
|
-
"maat-
|
|
353
|
+
self.assertEqual(names, ["maat-age.csv", "maat-authors.csv", "maat-components.csv", "maat-coupling.csv", "maat-doa.csv",
|
|
354
|
+
"maat-entity-ownership.csv", "maat-entropy.csv", "maat-fixes.csv", "maat-latenight.csv", "maat-plumbing.csv",
|
|
355
|
+
"maat-revisions.csv", "maat-soc.csv", "maat-tests.csv"])
|
|
354
356
|
self.assertTrue(os.path.isfile(os.path.join(d, "activity.json")))
|
|
355
357
|
with open(os.path.join(d, "maat-revisions.csv")) as fh:
|
|
356
358
|
self.assertEqual(fh.readline().strip(), "entity,n-revs")
|
|
@@ -632,3 +634,52 @@ class ChangeEntropy(unittest.TestCase):
|
|
|
632
634
|
maat.write_all(log, d, now="2026-09-15")
|
|
633
635
|
with open(os.path.join(d, "maat-entropy.csv")) as fh:
|
|
634
636
|
self.assertEqual(fh.readline().strip(), "entity,periods,hcm")
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
class DegreeOfAuthorship(unittest.TestCase):
|
|
640
|
+
def test_avelinos_doa_with_the_creator_bonus_and_dilution_by_others(self):
|
|
641
|
+
commits = [_commit("c1", [("a.py", 10, 0)], author="Ann", date="2026-01-01"),
|
|
642
|
+
_commit("c2", [("a.py", 2, 1)], author="Ann", date="2026-01-02"),
|
|
643
|
+
_commit("c3", [("a.py", 1, 1)], author="Bob", date="2026-01-03"),
|
|
644
|
+
_commit("m1", [("b.py", 0, 0)], author="Mover", date="2026-01-04"),
|
|
645
|
+
_commit("c4", [("b.py", 5, 0)], author="Cat", date="2026-01-05")]
|
|
646
|
+
rows = {(r["entity"], r["author"]): r for r in maat.doa(commits, now="2026-01-10")}
|
|
647
|
+
ann = rows[("a.py", "Ann")]
|
|
648
|
+
self.assertEqual((ann["fa"], ann["dl"], ann["ac"]), (1, 2, 1))
|
|
649
|
+
self.assertAlmostEqual(ann["doa"], 3.293 + 1.098 + 0.164 * 2 - 0.321 * math.log(2), places=3)
|
|
650
|
+
self.assertEqual(rows[("a.py", "Bob")]["fa"], 0)
|
|
651
|
+
self.assertEqual((rows[("b.py", "Cat")]["fa"], rows[("b.py", "Mover")]["fa"]), (1, 0), "a pure move adds no lines and creates nothing")
|
|
652
|
+
self.assertEqual(ann["is_author"], 1)
|
|
653
|
+
self.assertEqual(rows[("a.py", "Bob")]["is_author"], 0, "Bob's DOA is under the 3.293 floor")
|
|
654
|
+
|
|
655
|
+
def test_decay_halves_knowledge_every_five_months(self):
|
|
656
|
+
old = [_commit(f"o{i}", [("a.py", 1, 0)], author="Ann", date="2024-01-01") for i in range(10)]
|
|
657
|
+
new = [_commit(f"n{i}", [("a.py", 1, 0)], author="Bob", date="2026-01-01") for i in range(3)]
|
|
658
|
+
rows = {r["author"]: r for r in maat.doa(old + new, now="2026-01-01")}
|
|
659
|
+
self.assertEqual(rows["Ann"]["is_author"], 1, "undecayed, ten changes and creation outweigh three")
|
|
660
|
+
self.assertGreater(rows["Bob"]["doa_decayed"], rows["Ann"]["doa_decayed"] - 1.098, "decayed, Ann's two-year-old changes count for little")
|
|
661
|
+
self.assertEqual(rows["Bob"]["is_author_decayed"], 1)
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
class LateNight(unittest.TestCase):
|
|
665
|
+
def test_commits_between_midnight_and_four_in_the_authors_own_time(self):
|
|
666
|
+
commits = [dict(_commit("a", [("x.py", 1, 0)]), time="2026-01-05T01:30:00+09:00"),
|
|
667
|
+
dict(_commit("b", [("x.py", 1, 0)]), time="2026-01-05T03:59:00-05:00"),
|
|
668
|
+
dict(_commit("c", [("x.py", 1, 0)]), time="2026-01-05T04:00:00+00:00"),
|
|
669
|
+
dict(_commit("d", [("x.py", 1, 0), ("y.py", 1, 0)]), time="2026-01-05T23:00:00+00:00")]
|
|
670
|
+
rows = {r["entity"]: r for r in maat.latenight(commits)}
|
|
671
|
+
self.assertEqual(rows["x.py"], {"entity": "x.py", "n-revs": 4, "late": 2})
|
|
672
|
+
self.assertEqual(rows["y.py"]["late"], 0)
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
class Components(unittest.TestCase):
|
|
676
|
+
def test_coupling_between_top_level_components_over_logical_changes(self):
|
|
677
|
+
commits = []
|
|
678
|
+
for i in range(12):
|
|
679
|
+
commits.append(_commit(f"a{i}", [("auth/login.py", 1, 0), ("billing/charge.py", 1, 0)], date=f"2026-01-{1 + i:02d}"))
|
|
680
|
+
for i in range(12):
|
|
681
|
+
commits.append(_commit(f"b{i}", [("auth/token.py", 1, 0)], author="Bob", date=f"2026-02-{1 + i:02d}"))
|
|
682
|
+
commits.append(_commit("c", [("docs/x.md", 1, 0), ("auth/login.py", 1, 0)], date="2026-03-01"))
|
|
683
|
+
rows = [r for r in maat.components(commits) if r["depth"] == 1]
|
|
684
|
+
self.assertEqual(rows, [{"depth": 1, "entity": "auth/", "coupled": "billing/", "degree": 65, "shared": 12, "average-revs": 19}],
|
|
685
|
+
"12 shared changes over an average of (25 + 12) / 2; docs/ shares one change, under the floor")
|
|
@@ -292,6 +292,13 @@ class Report(unittest.TestCase):
|
|
|
292
292
|
text = (sec.get("caption") or "") + (sec.get("note") or "")
|
|
293
293
|
self.assertIn("the vulnerability database changed between the runs (2026-09-01 to 2026-09-17), so a dependency finding can move with no change to the code", text)
|
|
294
294
|
|
|
295
|
+
def test_the_watch_list_by_component_is_a_full_only_section(self):
|
|
296
|
+
r = sample_report()
|
|
297
|
+
self.assertNotIn("Watch list by component", rendered(r, [], width=200))
|
|
298
|
+
self.assertIn("Watch list by component", rendered(r, [], width=200, full=True))
|
|
299
|
+
self.assertIn("## Watch list by component", render.markdown(r, []))
|
|
300
|
+
self.assertIn("watch_by_component", render.to_json(r, []))
|
|
301
|
+
|
|
295
302
|
def test_the_json_is_the_same_bytes_for_the_same_clone_whatever_the_run(self):
|
|
296
303
|
import copy
|
|
297
304
|
a = sample_report()
|
|
@@ -1307,13 +1314,13 @@ class Sections(unittest.TestCase):
|
|
|
1307
1314
|
def test_sections_carry_title_columns_and_rows_in_report_order(self):
|
|
1308
1315
|
secs = render.sections(sample_report(), full=True)
|
|
1309
1316
|
titles = [x["title"] for x in secs]
|
|
1310
|
-
self.assertEqual(titles[:
|
|
1311
|
-
self.assertTrue(titles[
|
|
1312
|
-
self.assertTrue(titles[
|
|
1317
|
+
self.assertEqual(titles[:6], ["Watch list", "Watch list by component", "Size by language", "People", "Knowledge map", "Activity"])
|
|
1318
|
+
self.assertTrue(titles[6].startswith("Timeline"))
|
|
1319
|
+
self.assertTrue(titles[7].startswith("Hotspots"))
|
|
1313
1320
|
self.assertEqual(titles[-2], "Complex functions")
|
|
1314
1321
|
self.assertEqual(titles[-1], "Repo health (git-sizer concerns)")
|
|
1315
|
-
self.assertEqual([x["id"] for x in secs][:
|
|
1316
|
-
size = secs[
|
|
1322
|
+
self.assertEqual([x["id"] for x in secs][:5], ["watch", "watch_by_component", "size", "people", "knowledge"])
|
|
1323
|
+
size = secs[2]
|
|
1317
1324
|
self.assertEqual(size["columns"][:3], ["language", "files", "code"])
|
|
1318
1325
|
self.assertEqual(size["rows"][0][0], "HTML")
|
|
1319
1326
|
|
|
@@ -100,6 +100,21 @@ class Risks(unittest.TestCase):
|
|
|
100
100
|
self.assertIn("defines 72 functions and classes", reasons)
|
|
101
101
|
self.assertFalse([x for x in by["core/util.py"]["reasons"] if "TODO" in x or "nested" in x or "defines" in x])
|
|
102
102
|
|
|
103
|
+
def test_late_night_changes_are_a_reason_and_never_a_rank(self):
|
|
104
|
+
r = report()
|
|
105
|
+
r["latenight"] = [{"entity": "core/parser.py", "n-revs": 40, "late": 12}, {"entity": "core/util.py", "n-revs": 30, "late": 2}]
|
|
106
|
+
by = {x["file"]: x for x in watch.risks(r)}
|
|
107
|
+
self.assertIn("30% of its changes made between midnight and 4 am", by["core/parser.py"]["reasons"])
|
|
108
|
+
self.assertNotIn("midnight", " ".join(by["core/util.py"]["reasons"]))
|
|
109
|
+
self.assertEqual([x["file"] for x in watch.risks(r)], [x["file"] for x in watch.risks(report())], "the rank does not move")
|
|
110
|
+
|
|
111
|
+
def test_the_watch_list_by_component(self):
|
|
112
|
+
r = report()
|
|
113
|
+
groups = watch.by_component(watch.risks(r), top=2)
|
|
114
|
+
self.assertEqual([(g["component"], [x["file"] for x in g["files"]]) for g in groups],
|
|
115
|
+
[("web/", ["web/index.html"]), ("core/", ["core/parser.py", "core/util.py"])])
|
|
116
|
+
self.assertAlmostEqual(sum(g["share"] for g in groups), 100.0)
|
|
117
|
+
|
|
103
118
|
def test_tests_that_never_move_with_a_file_are_a_reason(self):
|
|
104
119
|
r = report()
|
|
105
120
|
r["tests"] = [{"entity": "core/parser.py", "n-sets": 38, "with-tests": 0}, {"entity": "core/util.py", "n-sets": 28, "with-tests": 4},
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|