gitmole 0.8.0__tar.gz → 0.9.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gitmole-0.8.0 → gitmole-0.9.1}/PKG-INFO +7 -7
- {gitmole-0.8.0 → gitmole-0.9.1}/README.md +6 -6
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/__init__.py +1 -1
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/cli.py +1 -1
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/evaluate.py +49 -4
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/findings.py +1 -1
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/load.py +2 -6
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/render.py +44 -12
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/textfmt.py +5 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/trend.py +3 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/watch.py +26 -65
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/PKG-INFO +7 -7
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_evaluate.py +34 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_render.py +115 -48
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_textfmt.py +11 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_watch.py +32 -35
- {gitmole-0.8.0 → gitmole-0.9.1}/LICENSE +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/__main__.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/backtest.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/banner.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/blame.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/clean.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/coupling.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/deps.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/duplicates.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/filetypes.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/functions.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/hotspots.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/identity.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/knowledge.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/leaks.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/loss.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/maat.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/run.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/SOURCES.txt +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/dependency_links.txt +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/entry_points.txt +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/requires.txt +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/top_level.txt +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/pyproject.toml +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/setup.cfg +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_backtest.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_banner.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_blame.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_clean.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_cli.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_coupling.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_deps.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_duplicates.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_filetypes.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_findings.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_functions.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_golden.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_hotspots.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_identity.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_knowledge.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_leaks.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_load.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_loss.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_maat.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_packaging.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_render_examples.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_run.py +0 -0
- {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_trend.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.1
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -96,7 +96,7 @@ since 2013, at a pinned commit:
|
|
|
96
96
|
ork.js commitLayoutEffectOnFiber() complexity 72
|
|
97
97
|
packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
|
|
98
98
|
rk.js beginWork() complexity 52
|
|
99
|
-
ranked by revisions × lines of code; the reasons say what
|
|
99
|
+
ranked by revisions × lines of code alone; the reasons say what to look at there
|
|
100
100
|
6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
|
|
101
101
|
files that had changed more than once would name 0.3; the 15 most changed would name 7)
|
|
102
102
|
```
|
|
@@ -108,11 +108,11 @@ ranks by revisions × lines of code: measured at six cut-offs on three
|
|
|
108
108
|
repositories
|
|
109
109
|
([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
|
|
110
110
|
that named more of the files fixed next than churn alone, size alone or a
|
|
111
|
-
weighted product of fixes, complexity and ownership. Between the header and
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
111
|
+
weighted product of fixes, complexity and ownership. Between the header and that
|
|
112
|
+
list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
|
|
113
|
+
notes); below it, tables for people, the knowledge map, the timeline, change
|
|
114
|
+
coupling, complex functions and repo health; `--full` adds the hotspots table
|
|
115
|
+
behind the list, size, activity and code age. Every section is explained in
|
|
116
116
|
[docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
|
|
117
117
|
|
|
118
118
|
Reports on repositories you know, each at a pinned commit with a fixed
|
|
@@ -76,7 +76,7 @@ since 2013, at a pinned commit:
|
|
|
76
76
|
ork.js commitLayoutEffectOnFiber() complexity 72
|
|
77
77
|
packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
|
|
78
78
|
rk.js beginWork() complexity 52
|
|
79
|
-
ranked by revisions × lines of code; the reasons say what
|
|
79
|
+
ranked by revisions × lines of code alone; the reasons say what to look at there
|
|
80
80
|
6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
|
|
81
81
|
files that had changed more than once would name 0.3; the 15 most changed would name 7)
|
|
82
82
|
```
|
|
@@ -88,11 +88,11 @@ ranks by revisions × lines of code: measured at six cut-offs on three
|
|
|
88
88
|
repositories
|
|
89
89
|
([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
|
|
90
90
|
that named more of the files fixed next than churn alone, size alone or a
|
|
91
|
-
weighted product of fixes, complexity and ownership. Between the header and
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
91
|
+
weighted product of fixes, complexity and ownership. Between the header and that
|
|
92
|
+
list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
|
|
93
|
+
notes); below it, tables for people, the knowledge map, the timeline, change
|
|
94
|
+
coupling, complex functions and repo health; `--full` adds the hotspots table
|
|
95
|
+
behind the list, size, activity and code age. Every section is explained in
|
|
96
96
|
[docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
|
|
97
97
|
|
|
98
98
|
Reports on repositories you know, each at a pinned commit with a fixed
|
|
@@ -38,7 +38,7 @@ def parse_args(argv):
|
|
|
38
38
|
p.add_argument("--clean", action="store_true", help="list the directories gitmole created (temp clones, analysis-* under the target) and delete them after a y/N question, then exit")
|
|
39
39
|
p.add_argument("--yes", action="store_true", help="with --clean: delete without asking")
|
|
40
40
|
p.add_argument("--duplicates", action="store_true", help=argparse.SUPPRESS) # duplicates always run now; kept so older scripts still parse
|
|
41
|
-
p.add_argument("--full", action="store_true", help="every column and
|
|
41
|
+
p.add_argument("--full", action="store_true", help="every section, column and row in the terminal report: adds hotspots, size, activity and code age, and the test files the default tables hide (the default is the tighter, readable one)")
|
|
42
42
|
p.add_argument("--json", metavar="PATH", help="write the report and findings as JSON to PATH, or - for stdout")
|
|
43
43
|
p.add_argument("--markdown", metavar="PATH", help="write the report as Markdown to PATH, or - for stdout")
|
|
44
44
|
p.add_argument("--fail-on", choices=findings.SEVERITIES, help="exit 3 if any finding is at this severity or worse")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
"""How the watch list would have done at several cut-off dates, next to the factor products it
|
|
3
|
-
replaced and the simpler baselines.
|
|
3
|
+
replaced (computed here, over the rows watch.risks returns) and the simpler baselines.
|
|
4
4
|
|
|
5
5
|
A development tool, not a pipeline step: `python -m gitmole.evaluate REPO OUT_DIR [--windows 6]
|
|
6
6
|
[--horizon 6] [--top 15]`, where OUT_DIR is a finished gitmole output directory for REPO (its log.txt
|
|
@@ -13,6 +13,7 @@ column per cut-off, and the total; then how many commits `--all` adds to HEAD's.
|
|
|
13
13
|
from __future__ import annotations
|
|
14
14
|
|
|
15
15
|
import argparse
|
|
16
|
+
import bisect
|
|
16
17
|
import calendar
|
|
17
18
|
import datetime as dt
|
|
18
19
|
import os
|
|
@@ -52,12 +53,56 @@ def report_at(commits: list, t: str, size: dict, meta: dict) -> dict:
|
|
|
52
53
|
"fixes": maat.fixes(past, now=t), "coupling": [], "functions": []}
|
|
53
54
|
|
|
54
55
|
|
|
56
|
+
SOLO_WEIGHT = 1.5 # how much single ownership lifts a factor product
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _by_max(values: list, inclusive: bool):
|
|
60
|
+
"""x as a share of the largest value: the scaling gitmole 0.7 shipped. One outlier moves everyone;
|
|
61
|
+
inclusive is ignored here, since a share of the largest value has no edge to choose."""
|
|
62
|
+
top = max(values)
|
|
63
|
+
return lambda x: x / top if top else 0.0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _by_rank(values: list, inclusive: bool):
|
|
67
|
+
"""x as the share of the scored files at or below it (inclusive), or strictly below it. Churn is
|
|
68
|
+
inclusive, so the most-changed file is 1 and no file is 0; fixes and complexity are strict, so a
|
|
69
|
+
file with none of either gets no lift, as under _by_max. An outlier is one more file, not a new
|
|
70
|
+
scale."""
|
|
71
|
+
ordered = sorted(values)
|
|
72
|
+
cut = bisect.bisect_right if inclusive else bisect.bisect_left
|
|
73
|
+
return lambda x: cut(ordered, x) / len(ordered)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
SCALINGS = {"max": _by_max, "rank": _by_rank}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def factor_scores(rows: list, scaling: str) -> dict:
|
|
80
|
+
"""file -> churn × (1 + recent fixes) × (1 + complexity) × (1.5 if single-owned), what the watch
|
|
81
|
+
list ranked by before 0.8, over the rows watch.risks returns. Kept here, not in watch.py, because
|
|
82
|
+
only this comparison still needs it."""
|
|
83
|
+
if not rows:
|
|
84
|
+
return {}
|
|
85
|
+
scale = SCALINGS[scaling]
|
|
86
|
+
churn = scale([r["revs"] for r in rows], True)
|
|
87
|
+
fixed = scale([r["recent_fixes"] for r in rows], False)
|
|
88
|
+
cplx = scale([r["complexity"] for r in rows], False)
|
|
89
|
+
return {r["file"]: churn(r["revs"]) * (1 + fixed(r["recent_fixes"])) * (1 + cplx(r["complexity"])) * (SOLO_WEIGHT if r["solo"] else 1)
|
|
90
|
+
for r in rows}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def factor_product(rows: list, scaling: str) -> list:
|
|
94
|
+
"""File names by factor_scores, best first; ties by revisions, then by name, as the list itself breaks them."""
|
|
95
|
+
scores = factor_scores(rows, scaling)
|
|
96
|
+
return [r["file"] for r in sorted(rows, key=lambda r: (-scores[r["file"]], -r["revs"], r["file"]))]
|
|
97
|
+
|
|
98
|
+
|
|
55
99
|
def variants(report: dict) -> dict:
|
|
56
|
-
"""variant -> file names, best first, every one drawn from the pool the watch list draws from.
|
|
100
|
+
"""variant -> file names, best first, every one drawn from the pool the watch list draws from. The
|
|
101
|
+
factor products it used to be ranked by are computed here, next to the watch list's own ranking."""
|
|
57
102
|
rows = watch.risks(report)
|
|
58
103
|
out = {"watch list (hotspot)": [r["file"] for r in rows],
|
|
59
|
-
"factor product (max-scaled)":
|
|
60
|
-
"factor product (rank-scaled)":
|
|
104
|
+
"factor product (max-scaled)": factor_product(rows, "max"),
|
|
105
|
+
"factor product (rank-scaled)": factor_product(rows, "rank")}
|
|
61
106
|
for name, key in watch.BASELINES.items():
|
|
62
107
|
out[name] = watch.ranked_by(rows, key)
|
|
63
108
|
out["recent fixes"] = watch.ranked_by(rows, lambda r: (r["recent_fixes"], r["revs"]))
|
|
@@ -521,7 +521,7 @@ def _generated(report: dict) -> set:
|
|
|
521
521
|
return hotspots.derived(report)
|
|
522
522
|
|
|
523
523
|
|
|
524
|
-
def complexity_growth(report: dict, min_growers: int = 3, min_pct: int =
|
|
524
|
+
def complexity_growth(report: dict, min_growers: int = 3, min_pct: int = trend.GROWTH_FLOOR, top_n: int = 10) -> list:
|
|
525
525
|
"""The top_n source hotspots whose complexity grew over the last year, from the trend samples.
|
|
526
526
|
Test files are left out: a growing test file is not the problem the finding is about."""
|
|
527
527
|
series = (report.get("trend") or {}).get("files") or {}
|
|
@@ -9,7 +9,7 @@ import os
|
|
|
9
9
|
import re
|
|
10
10
|
from collections import Counter, OrderedDict
|
|
11
11
|
|
|
12
|
-
from . import filetypes, identity, leaks
|
|
12
|
+
from . import filetypes, identity, leaks, textfmt
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
def _rel(path: str) -> str:
|
|
@@ -168,10 +168,6 @@ NAME_CAP = 200 # a function name a table can show; deeply nested fixtures give
|
|
|
168
168
|
csv.field_size_limit(min(sys.maxsize, 2**31 - 1)) # an older functions.csv may still carry such a name
|
|
169
169
|
|
|
170
170
|
|
|
171
|
-
def _cut(name: str, cap: int = NAME_CAP) -> str:
|
|
172
|
-
return name if len(name) <= cap else name[:cap - 1] + "…"
|
|
173
|
-
|
|
174
|
-
|
|
175
171
|
def parse_functions(text: str) -> list:
|
|
176
172
|
"""lizard --csv rows: nloc, ccn, tokens, params, length, location, file, function, long name, start, end;
|
|
177
173
|
then, from gitmole's own step, a label for a nameless function (its start line) and why the span
|
|
@@ -183,7 +179,7 @@ def parse_functions(text: str) -> list:
|
|
|
183
179
|
continue
|
|
184
180
|
name, label, suspect = r[7], r[11] if len(r) > 11 else "", r[12] if len(r) > 12 else ""
|
|
185
181
|
anonymous = name in ("", "(anonymous)")
|
|
186
|
-
rows.append({"file": _rel(r[6]), "function":
|
|
182
|
+
rows.append({"file": _rel(r[6]), "function": textfmt.cut(label if anonymous and label else name, NAME_CAP) or "(anonymous)", "anonymous": anonymous,
|
|
187
183
|
"ccn": _num(r[1]), "nloc": _num(r[0]), "params": _num(r[3]), "start": _num(r[9]), "end": _num(r[10]), "suspect": suspect})
|
|
188
184
|
return rows
|
|
189
185
|
|
|
@@ -26,6 +26,18 @@ WARM = "#ff9ee0" # values worth a glance
|
|
|
26
26
|
ROW_STYLES = ["", "on #1c2230"]
|
|
27
27
|
SIDE_BY_SIDE_MIN_WIDTH = 100
|
|
28
28
|
|
|
29
|
+
# the Timeline's month columns: each is 3 characters wide plus 2 of column padding, plus the 1-column
|
|
30
|
+
# gap rich reserves between every pair of columns even with the box's edges hidden (verified against
|
|
31
|
+
# rich.table.Table._calculate_column_widths, whose "n columns - 1" extra width cancels the gap saved
|
|
32
|
+
# on the last column, leaving a clean 6 per month). The section itself is indented by 2. FLOOR is the
|
|
33
|
+
# fewest months shown even when a name leaves almost no room. Once FLOOR is reached the months keep
|
|
34
|
+
# their full width and the name gives way instead, cut to whatever room is left; NAME_FLOOR is the
|
|
35
|
+
# fewest characters of a name still shown before the ellipsis, even if the months leave less room than
|
|
36
|
+
# that (eight is enough to keep most short names, and the start of longer ones, still recognisable).
|
|
37
|
+
# The section needs INDENT + NAME_FLOOR + FLOOR × MONTH_WIDTH = 28 columns; below that rich starves
|
|
38
|
+
# the month cells, which no real terminal reaches.
|
|
39
|
+
MONTH_WIDTH, INDENT, FLOOR, NAME_FLOOR = 6, 2, 3, 8
|
|
40
|
+
|
|
29
41
|
SYMBOLS = {"Size by language": "▤", "People": "◉", "Activity": "◔", "Timeline": "▦", "Hotspots": "◆", "Change coupling": "⟷",
|
|
30
42
|
"Surviving code by year written": "◷", "Net lines added by year": "◷", "Paths in history by year last changed": "◷",
|
|
31
43
|
"Knowledge map": "⌂", "Repo health": "✚", "Portfolio": "▣", "File types": "▥", "Complex functions": "λ", "Watch list": "◎",
|
|
@@ -40,8 +52,9 @@ RIGHT = {"justify": "right"}
|
|
|
40
52
|
FOLD = {"overflow": "fold"}
|
|
41
53
|
PATH = {"overflow": "fold", "no_wrap": False}
|
|
42
54
|
|
|
43
|
-
# rows shown by default; `full` lifts the caps. Markdown gets a looser cap of its own.
|
|
44
|
-
|
|
55
|
+
# rows shown by default; `full` lifts the caps. Markdown gets a looser cap of its own. Hotspots has
|
|
56
|
+
# no entry: it is `--full`/Markdown only now, so its row count is never decided by this table.
|
|
57
|
+
CAPS = {"People": 6, "Change coupling": 5, "Knowledge map": 6, "Size by language": 8, "Timeline": 8, "Complex functions": 8}
|
|
45
58
|
MARKDOWN_CAP = 50
|
|
46
59
|
TREND_TOP = 10 # the trend step's own --top default: only those files have samples
|
|
47
60
|
WATCH_CAP, WATCH_FULL = 5, 15 # the watch list is a short list by design; `full` and Markdown get a longer one, never all files
|
|
@@ -284,10 +297,14 @@ def watch_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
284
297
|
rows = [(r["file"], " · ".join(r["reasons"])) for r in ranked[:limit]]
|
|
285
298
|
columns = [("file", PATH), ("why", {"overflow": "fold", "ratio": 3})]
|
|
286
299
|
since = report["meta"].get("since")
|
|
287
|
-
|
|
300
|
+
# "alone": the reasons never move a file; a reader who sees fixes and ownership beside each row
|
|
301
|
+
# would otherwise take them for the ranking
|
|
302
|
+
notes = ["ranked by revisions × lines of code alone; the reasons say what to look at there" + (f"; commits since {since}" if since else "")]
|
|
288
303
|
bt = watch.backtest(report)
|
|
289
304
|
status = report["meta"].get("backtest") or {}
|
|
290
|
-
if bt:
|
|
305
|
+
if bt and not bt["fixed"]:
|
|
306
|
+
notes.append("nothing has been fixed since the cut-off six months ago, so there is nothing to score the list against")
|
|
307
|
+
elif bt:
|
|
291
308
|
notes.append(f"6 months ago this list would have named {bt['hits']} of the {bt['fixed']} files fixed since "
|
|
292
309
|
f"(a random {bt['listed']} of the {bt['pool']} files that had changed more than once would name {bt['expected']}; "
|
|
293
310
|
f"the {bt['listed']} most changed would name {bt['baselines']['churn']})"
|
|
@@ -394,6 +411,12 @@ def _month_label(ym: str) -> str:
|
|
|
394
411
|
|
|
395
412
|
|
|
396
413
|
def timeline_section(report: dict, full: bool = True, width=None, months: int = 12) -> dict:
|
|
414
|
+
"""Commits per author, one column per month. Names never fold: when the year does not fit the
|
|
415
|
+
terminal width, the oldest months are dropped (down to FLOOR) instead. If a name is still too long
|
|
416
|
+
for the room FLOOR leaves, the name gives way, not the months: it is shown cut with an ellipsis
|
|
417
|
+
(never fewer than NAME_FLOOR characters), so the months a reader came for stay full width. The
|
|
418
|
+
title names the months actually shown; ranking, bots filtering and the row's key are all still the
|
|
419
|
+
real name, only the displayed cell is cut. With no width (the Markdown export) nothing is trimmed."""
|
|
397
420
|
tl = (report.get("activity") or {}).get("timeline") or {}
|
|
398
421
|
if not tl:
|
|
399
422
|
return _section("Timeline", [("author", {})], [], note="no timeline data")
|
|
@@ -402,18 +425,25 @@ def timeline_section(report: dict, full: bool = True, width=None, months: int =
|
|
|
402
425
|
since = report["meta"].get("since")
|
|
403
426
|
if since:
|
|
404
427
|
span = [m for m in span if m >= since[:7]] or span[-1:]
|
|
405
|
-
columns = [("author", {"overflow": "fold"})] + [(MONTHS[int(m[5:7]) - 1], RIGHT) for m in span]
|
|
406
428
|
in_window = {a: sum(per.get(m, 0) for m in span) for a, per in tl.items()}
|
|
407
429
|
# the run decided who is a bot from name and email; the timeline only has the name, so it asks the run
|
|
408
430
|
bots = {b["name"] for b in report["meta"].get("bots") or []}
|
|
409
431
|
ranked = [a for a in sorted(in_window, key=lambda a: -in_window[a]) if in_window[a] > 0 and a not in bots and not identity.is_bot(a)]
|
|
410
432
|
limit = _limit("Timeline", full)
|
|
411
|
-
|
|
433
|
+
if width:
|
|
434
|
+
name = max([len("author")] + [len(a) for a in ranked[:limit]])
|
|
435
|
+
span = span[-max(FLOOR, min(len(span), (width - INDENT - name) // MONTH_WIDTH)):]
|
|
436
|
+
columns = [("author", {"no_wrap": True})] + [(MONTHS[int(m[5:7]) - 1], RIGHT) for m in span]
|
|
437
|
+
room = width - INDENT - MONTH_WIDTH * len(span) if width else None
|
|
438
|
+
rows = [(textfmt.cut(a, max(NAME_FLOOR, room)) if width else a, *[tl[a].get(m) or "·" for m in span]) for a in ranked[:limit]]
|
|
412
439
|
return _section(f"Timeline ({_month_label(span[0])} → {_month_label(span[-1])})", columns, rows, caption=_more(len(ranked), limit))
|
|
413
440
|
|
|
414
441
|
|
|
415
442
|
def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
|
|
416
|
-
"""Change frequency times size, Tornhill-style. Files no longer in the tree sort last.
|
|
443
|
+
"""Change frequency times size, Tornhill-style. Files no longer in the tree sort last. Drawn
|
|
444
|
+
under `--full` and in the Markdown export only; the default terminal report leaves it to the
|
|
445
|
+
watch list, which ranks the same files. Built only for those two, it has no row cap of its
|
|
446
|
+
own outside Markdown's."""
|
|
417
447
|
authors = {a["entity"]: a["n-authors"] for a in report.get("authors") or []}
|
|
418
448
|
ages = {a["entity"]: a["age-months"] for a in report.get("age") or []}
|
|
419
449
|
fixes = {f["entity"]: f["n-fixes"] for f in report.get("fixes") or []}
|
|
@@ -443,10 +473,9 @@ def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
443
473
|
("trend", RIGHT)]
|
|
444
474
|
if full is not True:
|
|
445
475
|
columns, rows = _keep(columns, rows, ["file", "revs", "lines", "fixes", "authors", "trend"])
|
|
446
|
-
rows = _shorten(rows, width, columns)
|
|
447
476
|
note = None if rows else _empty_note(None, hidden_note, "no source hotspots")
|
|
448
477
|
notes = [c for c in (_more(len(scored), limit), None if note else hidden_note) if c]
|
|
449
|
-
if series
|
|
478
|
+
if series:
|
|
450
479
|
notes.append(f"trend sampled for the top {TREND_TOP} hotspots") # the rest of the column is empty by design
|
|
451
480
|
return _section(title, columns, rows, note=note, caption="; ".join(notes) or None)
|
|
452
481
|
|
|
@@ -602,16 +631,19 @@ def health_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
602
631
|
|
|
603
632
|
BUILDERS = [watch_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
|
|
604
633
|
hotspots_section, coupling_section, age_section, functions_section, health_section]
|
|
605
|
-
|
|
634
|
+
# `--full` and Markdown only: Size, Activity and Code age are interesting once and rarely change what you
|
|
635
|
+
# do next; Hotspots ranks the files the watch list already leads with, by the same product.
|
|
636
|
+
FULL_ONLY = {"size", "activity", "age", "hotspots"}
|
|
606
637
|
|
|
607
638
|
|
|
608
639
|
def sections(report: dict, full: bool = True, width=None) -> list:
|
|
609
640
|
"""Every section as a dict with an `id` (the builder's name without _section). The default terminal
|
|
610
|
-
report (`full` False) leaves the
|
|
641
|
+
report (`full` False) leaves out the sections in FULL_ONLY (size, activity, code age and
|
|
642
|
+
hotspots); `full` True and Markdown keep them."""
|
|
611
643
|
out = []
|
|
612
644
|
for b in BUILDERS:
|
|
613
645
|
sid = b.__name__[:-len("_section")]
|
|
614
|
-
if full is False and sid in
|
|
646
|
+
if full is False and sid in FULL_ONLY:
|
|
615
647
|
continue
|
|
616
648
|
sec = b(report, full, width)
|
|
617
649
|
sec["id"] = sid
|
|
@@ -22,6 +22,11 @@ def shorten_path(path: str, max_len: int) -> str:
|
|
|
22
22
|
return candidates[-1]
|
|
23
23
|
|
|
24
24
|
|
|
25
|
+
def cut(name: str, cap: int) -> str:
|
|
26
|
+
"""`name`, unchanged if it fits in `cap` characters, else cut to exactly `cap` ending in the ellipsis."""
|
|
27
|
+
return name if len(name) <= cap else name[:cap - 1] + ELLIPSIS
|
|
28
|
+
|
|
29
|
+
|
|
25
30
|
def times(n: int) -> str:
|
|
26
31
|
"""How often something happened, in words for the small numbers: once, twice, 3 times."""
|
|
27
32
|
return {1: "once", 2: "twice"}.get(n, f"{n} times")
|
|
@@ -33,6 +33,9 @@ def sample_dates(first: str, last: str, n: int) -> list:
|
|
|
33
33
|
return out
|
|
34
34
|
|
|
35
35
|
|
|
36
|
+
GROWTH_FLOOR = 25 # percent in a year: below it a hotspot's complexity is not said to be growing
|
|
37
|
+
|
|
38
|
+
|
|
36
39
|
def change_over_year(series: list, last_date: str) -> str:
|
|
37
40
|
if len(series) < 2:
|
|
38
41
|
return "-"
|
|
@@ -1,35 +1,26 @@
|
|
|
1
|
-
"""The watch list: the source files most likely to be fixed next, and what
|
|
1
|
+
"""The watch list: the source files most likely to be fixed next, and what to look at in each.
|
|
2
2
|
|
|
3
3
|
Each source file that is still in the tree and changed more than once is ranked by revisions × lines
|
|
4
4
|
of code, the product the Hotspots table uses: measured at six cut-offs on three repositories
|
|
5
5
|
(docs/validation.md), that product named more of the files fixed in the following six months than
|
|
6
6
|
any weighting of fixes, complexity and ownership did. Those signals are the reasons printed beside
|
|
7
|
-
each file: how often it was fixed lately, who alone owns it, its most complex function,
|
|
8
|
-
always changes with. A file's score is its share, in
|
|
9
|
-
of code, so the scores of the whole list add up to
|
|
10
|
-
|
|
11
|
-
others' only by what it adds to the whole. The
|
|
12
|
-
|
|
13
|
-
largest value ("max") or a rank among the scored files ("rank"), stay selectable so gitmole.evaluate
|
|
14
|
-
can keep comparing them.
|
|
7
|
+
each file: how often it was fixed lately, who alone owns it, its most complex function, how much its
|
|
8
|
+
complexity grew in the last year, what it always changes with. A file's score is its share, in
|
|
9
|
+
percent, of all scored files' revisions × lines of code, so the scores of the whole list add up to
|
|
10
|
+
100 and a change's `--risk` total is the share of that mass the change touches; one enormous file
|
|
11
|
+
takes a large share, as it should, and lowers the others' only by what it adds to the whole. The
|
|
12
|
+
factor products the list used to rank by live in gitmole.evaluate, which still compares them with it.
|
|
15
13
|
Complexity is scc's per-file total, which exists for every file on one scale; lizard's most complex
|
|
16
14
|
function in the file is what the reasons name, since lizard has no reader for shell, Terraform,
|
|
17
15
|
Makefiles and the like."""
|
|
18
16
|
from __future__ import annotations
|
|
19
17
|
|
|
20
|
-
import bisect
|
|
21
18
|
from collections import Counter, defaultdict
|
|
22
19
|
|
|
23
|
-
|
|
24
|
-
from . import filetypes, hotspots, textfmt
|
|
25
|
-
except ImportError: # pragma: no cover - not run as a script, but keep the package pattern
|
|
26
|
-
import filetypes
|
|
27
|
-
import hotspots
|
|
28
|
-
import textfmt
|
|
20
|
+
from . import filetypes, hotspots, textfmt, trend
|
|
29
21
|
|
|
30
22
|
CCN_FLOOR = 10 # lizard's own "complex" threshold: below it a function is not worth naming
|
|
31
23
|
SOLO_SHARE = 0.9 # one author wrote at least this much of the file: single ownership
|
|
32
|
-
SOLO_WEIGHT = 1.5 # how much single ownership lifts the factor-product scores
|
|
33
24
|
COMPANION_DEGREE = 50 # a coupling worth mentioning
|
|
34
25
|
COMPANION_REVS = 5 # ...over enough shared revisions to be a pattern
|
|
35
26
|
|
|
@@ -67,39 +58,16 @@ def _worst_function(report: dict) -> dict:
|
|
|
67
58
|
return worst
|
|
68
59
|
|
|
69
60
|
|
|
70
|
-
def
|
|
71
|
-
"""
|
|
72
|
-
|
|
73
|
-
top = max(values)
|
|
74
|
-
return lambda x: x / top if top else 0.0
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
def _by_rank(values: list, inclusive: bool):
|
|
78
|
-
"""x as the share of the scored files at or below it (inclusive), or strictly below it. Churn is
|
|
79
|
-
inclusive, so the most-changed file is 1 and no file is 0; fixes and complexity are strict, so a
|
|
80
|
-
file with none of either gets no lift, as under _by_max. An outlier is one more file, not a new
|
|
81
|
-
scale."""
|
|
82
|
-
ordered = sorted(values)
|
|
83
|
-
cut = bisect.bisect_right if inclusive else bisect.bisect_left
|
|
84
|
-
return lambda x: cut(ordered, x) / len(ordered)
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
SCALINGS = {"max": _by_max, "rank": _by_rank}
|
|
88
|
-
SCORINGS = ("hotspot", *SCALINGS)
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
def risks(report: dict, min_revs: int = 2, scoring: str = "hotspot") -> list:
|
|
92
|
-
"""The watch list: every scored file with its reasons, worst first. `scoring` is "hotspot" (the
|
|
93
|
-
default: each file's score is its percentage share of the pool's revisions × lines of code), or
|
|
94
|
-
one of the two factor products, "rank" and "max", kept so gitmole.evaluate can compare them with
|
|
95
|
-
it."""
|
|
96
|
-
if scoring not in SCORINGS:
|
|
97
|
-
raise ValueError(f"scoring must be one of {', '.join(SCORINGS)}, got {scoring!r}")
|
|
61
|
+
def risks(report: dict, min_revs: int = 2) -> list:
|
|
62
|
+
"""The watch list: every scored file with its reasons, worst first. A file's score is its
|
|
63
|
+
percentage share of the pool's revisions × lines of code."""
|
|
98
64
|
owners = _owners(report)
|
|
99
65
|
companions = _companions(report)
|
|
100
66
|
worst = _worst_function(report)
|
|
101
67
|
fixes = {f["entity"]: f for f in report.get("fixes") or []}
|
|
102
68
|
n_authors = {a["entity"]: a["n-authors"] for a in report.get("authors") or []}
|
|
69
|
+
series = (report.get("trend") or {}).get("files") or {}
|
|
70
|
+
last = (report.get("meta") or {}).get("last_date") or ""
|
|
103
71
|
|
|
104
72
|
plumb, derived = filetypes.plumbing_paths(report), hotspots.derived(report)
|
|
105
73
|
rows = []
|
|
@@ -115,34 +83,24 @@ def risks(report: dict, min_revs: int = 2, scoring: str = "hotspot") -> list:
|
|
|
115
83
|
rows.append({"file": h["entity"], "revs": h["revs"], "recent_fixes": fx.get("recent-fixes", 0), "fixes": fx.get("n-fixes", 0),
|
|
116
84
|
"authors": n_authors.get(h["entity"]), "owner": owner, "owner_share": share,
|
|
117
85
|
"complexity": h["complexity"] or 0, "code": h["code"],
|
|
118
|
-
"function": fn, "companions": companions.get(h["entity"], [])
|
|
86
|
+
"function": fn, "companions": companions.get(h["entity"], []),
|
|
87
|
+
"trend": trend.change_over_year(series[h["entity"]], last) if last and h["entity"] in series else None})
|
|
119
88
|
if not rows:
|
|
120
89
|
return []
|
|
121
90
|
|
|
122
|
-
|
|
123
|
-
pool = sum(r["revs"] * r["code"] for r in rows)
|
|
124
|
-
|
|
125
|
-
def score(r):
|
|
126
|
-
return 100 * (r["revs"] * r["code"]) / pool if pool else 0.0
|
|
127
|
-
else:
|
|
128
|
-
scale = SCALINGS[scoring]
|
|
129
|
-
churn = scale([r["revs"] for r in rows], True)
|
|
130
|
-
fixed = scale([r["recent_fixes"] for r in rows], False)
|
|
131
|
-
cplx = scale([r["complexity"] for r in rows], False)
|
|
132
|
-
|
|
133
|
-
def score(r):
|
|
134
|
-
return churn(r["revs"]) * (1 + fixed(r["recent_fixes"])) * (1 + cplx(r["complexity"])) * (SOLO_WEIGHT if r["solo"] else 1)
|
|
91
|
+
pool = sum(r["revs"] * r["code"] for r in rows)
|
|
135
92
|
for r in rows:
|
|
136
93
|
r["solo"] = r["authors"] == 1 or r["owner_share"] >= SOLO_SHARE
|
|
137
|
-
r["score"] =
|
|
94
|
+
r["score"] = 100 * (r["revs"] * r["code"]) / pool if pool else 0.0
|
|
138
95
|
r["reasons"] = _reasons(r)
|
|
139
96
|
rows.sort(key=lambda r: (-r["score"], -r["revs"], r["file"]))
|
|
140
97
|
return rows
|
|
141
98
|
|
|
142
99
|
|
|
143
100
|
def why_empty(report: dict, min_revs: int = 2) -> str:
|
|
144
|
-
"""Why risks() came back empty, for the report's one-line note: the honest reason, since
|
|
145
|
-
|
|
101
|
+
"""Why risks() came back empty, for the report's one-line note: the honest reason, since files
|
|
102
|
+
can well have changed even though none of them scored, and a flat "nothing changed" would be
|
|
103
|
+
a lie about them."""
|
|
146
104
|
churned = [h for h in hotspots.ranked(report) if h["revs"] >= min_revs]
|
|
147
105
|
if not churned:
|
|
148
106
|
return "nothing changed more than once"
|
|
@@ -167,6 +125,9 @@ def _reasons(r: dict) -> list:
|
|
|
167
125
|
if fn and fn["ccn"] >= CCN_FLOOR:
|
|
168
126
|
named = f"the function at line {fn['start']}" if fn.get("anonymous") else f"{fn['function']}()"
|
|
169
127
|
out.append(f"{named} complexity {fn['ccn']}")
|
|
128
|
+
grown = r.get("trend") or ""
|
|
129
|
+
if grown.startswith("+") and int(grown[1:-1]) >= trend.GROWTH_FLOOR:
|
|
130
|
+
out.append(f"complexity {grown} in a year") # the Hotspots table's trend column, which the default report no longer shows
|
|
170
131
|
if r["companions"]:
|
|
171
132
|
other, degree = r["companions"][0]
|
|
172
133
|
more = len(r["companions"]) - 1
|
|
@@ -179,9 +140,9 @@ WATCH_TOP = 15 # the same cap the report's --full watch list uses
|
|
|
179
140
|
|
|
180
141
|
|
|
181
142
|
def change_risk(report: dict, files: list) -> dict:
|
|
182
|
-
"""The watch score of each touched file, and their sum:
|
|
183
|
-
|
|
184
|
-
|
|
143
|
+
"""The watch score of each touched file, and their sum: that total is a percentage of the
|
|
144
|
+
repository's revisions × lines of code. Files the watch list never scored get 0 and one reason
|
|
145
|
+
saying why."""
|
|
185
146
|
ranked = risks(report)
|
|
186
147
|
by_file = {r["file"]: r for r in ranked}
|
|
187
148
|
watched = {r["file"] for r in ranked[:WATCH_TOP]}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.1
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -96,7 +96,7 @@ since 2013, at a pinned commit:
|
|
|
96
96
|
ork.js commitLayoutEffectOnFiber() complexity 72
|
|
97
97
|
packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
|
|
98
98
|
rk.js beginWork() complexity 52
|
|
99
|
-
ranked by revisions × lines of code; the reasons say what
|
|
99
|
+
ranked by revisions × lines of code alone; the reasons say what to look at there
|
|
100
100
|
6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
|
|
101
101
|
files that had changed more than once would name 0.3; the 15 most changed would name 7)
|
|
102
102
|
```
|
|
@@ -108,11 +108,11 @@ ranks by revisions × lines of code: measured at six cut-offs on three
|
|
|
108
108
|
repositories
|
|
109
109
|
([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
|
|
110
110
|
that named more of the files fixed next than churn alone, size alone or a
|
|
111
|
-
weighted product of fixes, complexity and ownership. Between the header and
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
111
|
+
weighted product of fixes, complexity and ownership. Between the header and that
|
|
112
|
+
list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
|
|
113
|
+
notes); below it, tables for people, the knowledge map, the timeline, change
|
|
114
|
+
coupling, complex functions and repo health; `--full` adds the hotspots table
|
|
115
|
+
behind the list, size, activity and code age. Every section is explained in
|
|
116
116
|
[docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
|
|
117
117
|
|
|
118
118
|
Reports on repositories you know, each at a pinned commit with a fixed
|
|
@@ -54,6 +54,40 @@ class Score(unittest.TestCase):
|
|
|
54
54
|
self.assertEqual(evaluate.report_at(commits, "2025-06-01", SIZE, {})["ownership"], [])
|
|
55
55
|
|
|
56
56
|
|
|
57
|
+
ROWS = [
|
|
58
|
+
{"file": "core/parser.py", "revs": 40, "recent_fixes": 5, "complexity": 40, "solo": True, "code": 800},
|
|
59
|
+
{"file": "web/index.html", "revs": 60, "recent_fixes": 0, "complexity": 0, "solo": False, "code": 4000},
|
|
60
|
+
{"file": "core/util.py", "revs": 30, "recent_fixes": 0, "complexity": 5, "solo": True, "code": 200},
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class FactorProduct(unittest.TestCase):
|
|
65
|
+
def test_both_scalings_lead_with_the_fixed_complex_single_owned_file(self):
|
|
66
|
+
for scaling in ("max", "rank"):
|
|
67
|
+
self.assertEqual(evaluate.factor_product(ROWS, scaling), ["core/parser.py", "web/index.html", "core/util.py"], scaling)
|
|
68
|
+
|
|
69
|
+
def test_a_file_never_fixed_gets_no_lift_from_fixes_under_either_scaling(self):
|
|
70
|
+
for scaling in ("max", "rank"):
|
|
71
|
+
self.assertEqual(evaluate.factor_scores(ROWS, scaling)["web/index.html"], 1.0, f"{scaling}: most changed, no fixes, no complexity, shared")
|
|
72
|
+
|
|
73
|
+
def test_under_rank_scaling_an_outlier_does_not_rescale_the_other_files(self):
|
|
74
|
+
def util(outlier_revs, scaling):
|
|
75
|
+
big = {"file": "core/big.py", "revs": outlier_revs, "recent_fixes": 0, "complexity": 0, "solo": False, "code": 10}
|
|
76
|
+
return evaluate.factor_scores(ROWS + [big], scaling)["core/util.py"]
|
|
77
|
+
self.assertEqual(util(100, "rank"), util(10000, "rank"))
|
|
78
|
+
self.assertNotEqual(util(100, "max"), util(10000, "max"), "what the rank scaling is for")
|
|
79
|
+
|
|
80
|
+
def test_two_files_that_differ_only_in_complexity(self):
|
|
81
|
+
pair = [{"file": "ops/deploy.sh", "revs": 40, "recent_fixes": 0, "complexity": 80, "solo": False, "code": 300},
|
|
82
|
+
{"file": "ops/plain.sh", "revs": 40, "recent_fixes": 0, "complexity": 0, "solo": False, "code": 300}]
|
|
83
|
+
for scaling in ("max", "rank"):
|
|
84
|
+
scores = evaluate.factor_scores(ROWS + pair, scaling)
|
|
85
|
+
self.assertGreater(scores["ops/deploy.sh"], scores["ops/plain.sh"], f"{scaling}: complexity lifts the factor product")
|
|
86
|
+
|
|
87
|
+
def test_no_rows_is_no_list(self):
|
|
88
|
+
self.assertEqual(evaluate.factor_product([], "rank"), [])
|
|
89
|
+
|
|
90
|
+
|
|
57
91
|
class Table(unittest.TestCase):
|
|
58
92
|
def test_one_row_per_variant_one_column_per_cut_off_and_a_total(self):
|
|
59
93
|
text = evaluate.table([("2025-02-28", 3, 40, {"churn": 1, "random (expected)": 0.4}),
|