gitmole 0.8.0__tar.gz → 0.9.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. {gitmole-0.8.0 → gitmole-0.9.1}/PKG-INFO +7 -7
  2. {gitmole-0.8.0 → gitmole-0.9.1}/README.md +6 -6
  3. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/__init__.py +1 -1
  4. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/cli.py +1 -1
  5. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/evaluate.py +49 -4
  6. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/findings.py +1 -1
  7. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/load.py +2 -6
  8. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/render.py +44 -12
  9. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/textfmt.py +5 -0
  10. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/trend.py +3 -0
  11. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/watch.py +26 -65
  12. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/PKG-INFO +7 -7
  13. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_evaluate.py +34 -0
  14. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_render.py +115 -48
  15. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_textfmt.py +11 -0
  16. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_watch.py +32 -35
  17. {gitmole-0.8.0 → gitmole-0.9.1}/LICENSE +0 -0
  18. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/__main__.py +0 -0
  19. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/backtest.py +0 -0
  20. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/banner.py +0 -0
  21. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/blame.py +0 -0
  22. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/clean.py +0 -0
  23. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/coupling.py +0 -0
  24. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/deps.py +0 -0
  25. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/duplicates.py +0 -0
  26. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/filetypes.py +0 -0
  27. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/functions.py +0 -0
  28. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/hotspots.py +0 -0
  29. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/identity.py +0 -0
  30. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/knowledge.py +0 -0
  31. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/leaks.py +0 -0
  32. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/loss.py +0 -0
  33. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/maat.py +0 -0
  34. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole/run.py +0 -0
  35. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/SOURCES.txt +0 -0
  36. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/dependency_links.txt +0 -0
  37. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/entry_points.txt +0 -0
  38. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/requires.txt +0 -0
  39. {gitmole-0.8.0 → gitmole-0.9.1}/gitmole.egg-info/top_level.txt +0 -0
  40. {gitmole-0.8.0 → gitmole-0.9.1}/pyproject.toml +0 -0
  41. {gitmole-0.8.0 → gitmole-0.9.1}/setup.cfg +0 -0
  42. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_backtest.py +0 -0
  43. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_banner.py +0 -0
  44. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_blame.py +0 -0
  45. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_clean.py +0 -0
  46. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_cli.py +0 -0
  47. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_coupling.py +0 -0
  48. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_deps.py +0 -0
  49. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_duplicates.py +0 -0
  50. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_filetypes.py +0 -0
  51. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_findings.py +0 -0
  52. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_functions.py +0 -0
  53. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_golden.py +0 -0
  54. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_hotspots.py +0 -0
  55. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_identity.py +0 -0
  56. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_knowledge.py +0 -0
  57. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_leaks.py +0 -0
  58. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_load.py +0 -0
  59. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_loss.py +0 -0
  60. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_maat.py +0 -0
  61. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_packaging.py +0 -0
  62. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_render_examples.py +0 -0
  63. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_run.py +0 -0
  64. {gitmole-0.8.0 → gitmole-0.9.1}/tests/test_trend.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.8.0
3
+ Version: 0.9.1
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -96,7 +96,7 @@ since 2013, at a pinned commit:
96
96
  ork.js commitLayoutEffectOnFiber() complexity 72
97
97
  packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
98
98
  rk.js beginWork() complexity 52
99
- ranked by revisions × lines of code; the reasons say what else counts against each file
99
+ ranked by revisions × lines of code alone; the reasons say what to look at there
100
100
  6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
101
101
  files that had changed more than once would name 0.3; the 15 most changed would name 7)
102
102
  ```
@@ -108,11 +108,11 @@ ranks by revisions × lines of code: measured at six cut-offs on three
108
108
  repositories
109
109
  ([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
110
110
  that named more of the files fixed next than churn alone, size alone or a
111
- weighted product of fixes, complexity and ownership. Between the header and
112
- that list the full report puts its findings, 16 for react (1 critical, 6
113
- warnings, 9 notes); below it, tables for people, the knowledge map, the
114
- timeline, hotspots with their complexity trend, change coupling, complex
115
- functions and repo health. Every section is explained in
111
+ weighted product of fixes, complexity and ownership. Between the header and that
112
+ list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
113
+ notes); below it, tables for people, the knowledge map, the timeline, change
114
+ coupling, complex functions and repo health; `--full` adds the hotspots table
115
+ behind the list, size, activity and code age. Every section is explained in
116
116
  [docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
117
117
 
118
118
  Reports on repositories you know, each at a pinned commit with a fixed
@@ -76,7 +76,7 @@ since 2013, at a pinned commit:
76
76
  ork.js commitLayoutEffectOnFiber() complexity 72
77
77
  packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
78
78
  rk.js beginWork() complexity 52
79
- ranked by revisions × lines of code; the reasons say what else counts against each file
79
+ ranked by revisions × lines of code alone; the reasons say what to look at there
80
80
  6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
81
81
  files that had changed more than once would name 0.3; the 15 most changed would name 7)
82
82
  ```
@@ -88,11 +88,11 @@ ranks by revisions × lines of code: measured at six cut-offs on three
88
88
  repositories
89
89
  ([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
90
90
  that named more of the files fixed next than churn alone, size alone or a
91
- weighted product of fixes, complexity and ownership. Between the header and
92
- that list the full report puts its findings, 16 for react (1 critical, 6
93
- warnings, 9 notes); below it, tables for people, the knowledge map, the
94
- timeline, hotspots with their complexity trend, change coupling, complex
95
- functions and repo health. Every section is explained in
91
+ weighted product of fixes, complexity and ownership. Between the header and that
92
+ list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
93
+ notes); below it, tables for people, the knowledge map, the timeline, change
94
+ coupling, complex functions and repo health; `--full` adds the hotspots table
95
+ behind the list, size, activity and code age. Every section is explained in
96
96
  [docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
97
97
 
98
98
  Reports on repositories you know, each at a pinned commit with a fixed
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.8.0"
3
+ __version__ = "0.9.1"
@@ -38,7 +38,7 @@ def parse_args(argv):
38
38
  p.add_argument("--clean", action="store_true", help="list the directories gitmole created (temp clones, analysis-* under the target) and delete them after a y/N question, then exit")
39
39
  p.add_argument("--yes", action="store_true", help="with --clean: delete without asking")
40
40
  p.add_argument("--duplicates", action="store_true", help=argparse.SUPPRESS) # duplicates always run now; kept so older scripts still parse
41
- p.add_argument("--full", action="store_true", help="every column and every row in the terminal report, and the test files the default tables hide (the default is the tighter, readable one)")
41
+ p.add_argument("--full", action="store_true", help="every section, column and row in the terminal report: adds hotspots, size, activity and code age, and the test files the default tables hide (the default is the tighter, readable one)")
42
42
  p.add_argument("--json", metavar="PATH", help="write the report and findings as JSON to PATH, or - for stdout")
43
43
  p.add_argument("--markdown", metavar="PATH", help="write the report as Markdown to PATH, or - for stdout")
44
44
  p.add_argument("--fail-on", choices=findings.SEVERITIES, help="exit 3 if any finding is at this severity or worse")
@@ -1,6 +1,6 @@
1
1
  #!/usr/bin/env python3
2
2
  """How the watch list would have done at several cut-off dates, next to the factor products it
3
- replaced and the simpler baselines.
3
+ replaced (computed here, over the rows watch.risks returns) and the simpler baselines.
4
4
 
5
5
  A development tool, not a pipeline step: `python -m gitmole.evaluate REPO OUT_DIR [--windows 6]
6
6
  [--horizon 6] [--top 15]`, where OUT_DIR is a finished gitmole output directory for REPO (its log.txt
@@ -13,6 +13,7 @@ column per cut-off, and the total; then how many commits `--all` adds to HEAD's.
13
13
  from __future__ import annotations
14
14
 
15
15
  import argparse
16
+ import bisect
16
17
  import calendar
17
18
  import datetime as dt
18
19
  import os
@@ -52,12 +53,56 @@ def report_at(commits: list, t: str, size: dict, meta: dict) -> dict:
52
53
  "fixes": maat.fixes(past, now=t), "coupling": [], "functions": []}
53
54
 
54
55
 
56
+ SOLO_WEIGHT = 1.5 # how much single ownership lifts a factor product
57
+
58
+
59
+ def _by_max(values: list, inclusive: bool):
60
+ """x as a share of the largest value: the scaling gitmole 0.7 shipped. One outlier moves everyone;
61
+ inclusive is ignored here, since a share of the largest value has no edge to choose."""
62
+ top = max(values)
63
+ return lambda x: x / top if top else 0.0
64
+
65
+
66
+ def _by_rank(values: list, inclusive: bool):
67
+ """x as the share of the scored files at or below it (inclusive), or strictly below it. Churn is
68
+ inclusive, so the most-changed file is 1 and no file is 0; fixes and complexity are strict, so a
69
+ file with none of either gets no lift, as under _by_max. An outlier is one more file, not a new
70
+ scale."""
71
+ ordered = sorted(values)
72
+ cut = bisect.bisect_right if inclusive else bisect.bisect_left
73
+ return lambda x: cut(ordered, x) / len(ordered)
74
+
75
+
76
+ SCALINGS = {"max": _by_max, "rank": _by_rank}
77
+
78
+
79
+ def factor_scores(rows: list, scaling: str) -> dict:
80
+ """file -> churn × (1 + recent fixes) × (1 + complexity) × (1.5 if single-owned), what the watch
81
+ list ranked by before 0.8, over the rows watch.risks returns. Kept here, not in watch.py, because
82
+ only this comparison still needs it."""
83
+ if not rows:
84
+ return {}
85
+ scale = SCALINGS[scaling]
86
+ churn = scale([r["revs"] for r in rows], True)
87
+ fixed = scale([r["recent_fixes"] for r in rows], False)
88
+ cplx = scale([r["complexity"] for r in rows], False)
89
+ return {r["file"]: churn(r["revs"]) * (1 + fixed(r["recent_fixes"])) * (1 + cplx(r["complexity"])) * (SOLO_WEIGHT if r["solo"] else 1)
90
+ for r in rows}
91
+
92
+
93
+ def factor_product(rows: list, scaling: str) -> list:
94
+ """File names by factor_scores, best first; ties by revisions, then by name, as the list itself breaks them."""
95
+ scores = factor_scores(rows, scaling)
96
+ return [r["file"] for r in sorted(rows, key=lambda r: (-scores[r["file"]], -r["revs"], r["file"]))]
97
+
98
+
55
99
  def variants(report: dict) -> dict:
56
- """variant -> file names, best first, every one drawn from the pool the watch list draws from."""
100
+ """variant -> file names, best first, every one drawn from the pool the watch list draws from. The
101
+ factor products it used to be ranked by are computed here, next to the watch list's own ranking."""
57
102
  rows = watch.risks(report)
58
103
  out = {"watch list (hotspot)": [r["file"] for r in rows],
59
- "factor product (max-scaled)": [r["file"] for r in watch.risks(report, scoring="max")],
60
- "factor product (rank-scaled)": [r["file"] for r in watch.risks(report, scoring="rank")]}
104
+ "factor product (max-scaled)": factor_product(rows, "max"),
105
+ "factor product (rank-scaled)": factor_product(rows, "rank")}
61
106
  for name, key in watch.BASELINES.items():
62
107
  out[name] = watch.ranked_by(rows, key)
63
108
  out["recent fixes"] = watch.ranked_by(rows, lambda r: (r["recent_fixes"], r["revs"]))
@@ -521,7 +521,7 @@ def _generated(report: dict) -> set:
521
521
  return hotspots.derived(report)
522
522
 
523
523
 
524
- def complexity_growth(report: dict, min_growers: int = 3, min_pct: int = 25, top_n: int = 10) -> list:
524
+ def complexity_growth(report: dict, min_growers: int = 3, min_pct: int = trend.GROWTH_FLOOR, top_n: int = 10) -> list:
525
525
  """The top_n source hotspots whose complexity grew over the last year, from the trend samples.
526
526
  Test files are left out: a growing test file is not the problem the finding is about."""
527
527
  series = (report.get("trend") or {}).get("files") or {}
@@ -9,7 +9,7 @@ import os
9
9
  import re
10
10
  from collections import Counter, OrderedDict
11
11
 
12
- from . import filetypes, identity, leaks
12
+ from . import filetypes, identity, leaks, textfmt
13
13
 
14
14
 
15
15
  def _rel(path: str) -> str:
@@ -168,10 +168,6 @@ NAME_CAP = 200 # a function name a table can show; deeply nested fixtures give
168
168
  csv.field_size_limit(min(sys.maxsize, 2**31 - 1)) # an older functions.csv may still carry such a name
169
169
 
170
170
 
171
- def _cut(name: str, cap: int = NAME_CAP) -> str:
172
- return name if len(name) <= cap else name[:cap - 1] + "…"
173
-
174
-
175
171
  def parse_functions(text: str) -> list:
176
172
  """lizard --csv rows: nloc, ccn, tokens, params, length, location, file, function, long name, start, end;
177
173
  then, from gitmole's own step, a label for a nameless function (its start line) and why the span
@@ -183,7 +179,7 @@ def parse_functions(text: str) -> list:
183
179
  continue
184
180
  name, label, suspect = r[7], r[11] if len(r) > 11 else "", r[12] if len(r) > 12 else ""
185
181
  anonymous = name in ("", "(anonymous)")
186
- rows.append({"file": _rel(r[6]), "function": _cut(label if anonymous and label else name) or "(anonymous)", "anonymous": anonymous,
182
+ rows.append({"file": _rel(r[6]), "function": textfmt.cut(label if anonymous and label else name, NAME_CAP) or "(anonymous)", "anonymous": anonymous,
187
183
  "ccn": _num(r[1]), "nloc": _num(r[0]), "params": _num(r[3]), "start": _num(r[9]), "end": _num(r[10]), "suspect": suspect})
188
184
  return rows
189
185
 
@@ -26,6 +26,18 @@ WARM = "#ff9ee0" # values worth a glance
26
26
  ROW_STYLES = ["", "on #1c2230"]
27
27
  SIDE_BY_SIDE_MIN_WIDTH = 100
28
28
 
29
+ # the Timeline's month columns: each is 3 characters wide plus 2 of column padding, plus the 1-column
30
+ # gap rich reserves between every pair of columns even with the box's edges hidden (verified against
31
+ # rich.table.Table._calculate_column_widths, whose "n columns - 1" extra width cancels the gap saved
32
+ # on the last column, leaving a clean 6 per month). The section itself is indented by 2. FLOOR is the
33
+ # fewest months shown even when a name leaves almost no room. Once FLOOR is reached the months keep
34
+ # their full width and the name gives way instead, cut to whatever room is left; NAME_FLOOR is the
35
+ # fewest characters of a name still shown before the ellipsis, even if the months leave less room than
36
+ # that (eight is enough to keep most short names, and the start of longer ones, still recognisable).
37
+ # The section needs INDENT + NAME_FLOOR + FLOOR × MONTH_WIDTH = 28 columns; below that rich starves
38
+ # the month cells, which no real terminal reaches.
39
+ MONTH_WIDTH, INDENT, FLOOR, NAME_FLOOR = 6, 2, 3, 8
40
+
29
41
  SYMBOLS = {"Size by language": "▤", "People": "◉", "Activity": "◔", "Timeline": "▦", "Hotspots": "◆", "Change coupling": "⟷",
30
42
  "Surviving code by year written": "◷", "Net lines added by year": "◷", "Paths in history by year last changed": "◷",
31
43
  "Knowledge map": "⌂", "Repo health": "✚", "Portfolio": "▣", "File types": "▥", "Complex functions": "λ", "Watch list": "◎",
@@ -40,8 +52,9 @@ RIGHT = {"justify": "right"}
40
52
  FOLD = {"overflow": "fold"}
41
53
  PATH = {"overflow": "fold", "no_wrap": False}
42
54
 
43
- # rows shown by default; `full` lifts the caps. Markdown gets a looser cap of its own.
44
- CAPS = {"People": 6, "Hotspots": 8, "Change coupling": 5, "Knowledge map": 6, "Size by language": 8, "Timeline": 8, "Complex functions": 8}
55
+ # rows shown by default; `full` lifts the caps. Markdown gets a looser cap of its own. Hotspots has
56
+ # no entry: it is `--full`/Markdown only now, so its row count is never decided by this table.
57
+ CAPS = {"People": 6, "Change coupling": 5, "Knowledge map": 6, "Size by language": 8, "Timeline": 8, "Complex functions": 8}
45
58
  MARKDOWN_CAP = 50
46
59
  TREND_TOP = 10 # the trend step's own --top default: only those files have samples
47
60
  WATCH_CAP, WATCH_FULL = 5, 15 # the watch list is a short list by design; `full` and Markdown get a longer one, never all files
@@ -284,10 +297,14 @@ def watch_section(report: dict, full: bool = True, width=None) -> dict:
284
297
  rows = [(r["file"], " · ".join(r["reasons"])) for r in ranked[:limit]]
285
298
  columns = [("file", PATH), ("why", {"overflow": "fold", "ratio": 3})]
286
299
  since = report["meta"].get("since")
287
- notes = ["ranked by revisions × lines of code; the reasons say what else counts against each file" + (f"; commits since {since}" if since else "")]
300
+ # "alone": the reasons never move a file; a reader who sees fixes and ownership beside each row
301
+ # would otherwise take them for the ranking
302
+ notes = ["ranked by revisions × lines of code alone; the reasons say what to look at there" + (f"; commits since {since}" if since else "")]
288
303
  bt = watch.backtest(report)
289
304
  status = report["meta"].get("backtest") or {}
290
- if bt:
305
+ if bt and not bt["fixed"]:
306
+ notes.append("nothing has been fixed since the cut-off six months ago, so there is nothing to score the list against")
307
+ elif bt:
291
308
  notes.append(f"6 months ago this list would have named {bt['hits']} of the {bt['fixed']} files fixed since "
292
309
  f"(a random {bt['listed']} of the {bt['pool']} files that had changed more than once would name {bt['expected']}; "
293
310
  f"the {bt['listed']} most changed would name {bt['baselines']['churn']})"
@@ -394,6 +411,12 @@ def _month_label(ym: str) -> str:
394
411
 
395
412
 
396
413
  def timeline_section(report: dict, full: bool = True, width=None, months: int = 12) -> dict:
414
+ """Commits per author, one column per month. Names never fold: when the year does not fit the
415
+ terminal width, the oldest months are dropped (down to FLOOR) instead. If a name is still too long
416
+ for the room FLOOR leaves, the name gives way, not the months: it is shown cut with an ellipsis
417
+ (never fewer than NAME_FLOOR characters), so the months a reader came for stay full width. The
418
+ title names the months actually shown; ranking, bots filtering and the row's key are all still the
419
+ real name, only the displayed cell is cut. With no width (the Markdown export) nothing is trimmed."""
397
420
  tl = (report.get("activity") or {}).get("timeline") or {}
398
421
  if not tl:
399
422
  return _section("Timeline", [("author", {})], [], note="no timeline data")
@@ -402,18 +425,25 @@ def timeline_section(report: dict, full: bool = True, width=None, months: int =
402
425
  since = report["meta"].get("since")
403
426
  if since:
404
427
  span = [m for m in span if m >= since[:7]] or span[-1:]
405
- columns = [("author", {"overflow": "fold"})] + [(MONTHS[int(m[5:7]) - 1], RIGHT) for m in span]
406
428
  in_window = {a: sum(per.get(m, 0) for m in span) for a, per in tl.items()}
407
429
  # the run decided who is a bot from name and email; the timeline only has the name, so it asks the run
408
430
  bots = {b["name"] for b in report["meta"].get("bots") or []}
409
431
  ranked = [a for a in sorted(in_window, key=lambda a: -in_window[a]) if in_window[a] > 0 and a not in bots and not identity.is_bot(a)]
410
432
  limit = _limit("Timeline", full)
411
- rows = [(a, *[tl[a].get(m) or "·" for m in span]) for a in ranked[:limit]]
433
+ if width:
434
+ name = max([len("author")] + [len(a) for a in ranked[:limit]])
435
+ span = span[-max(FLOOR, min(len(span), (width - INDENT - name) // MONTH_WIDTH)):]
436
+ columns = [("author", {"no_wrap": True})] + [(MONTHS[int(m[5:7]) - 1], RIGHT) for m in span]
437
+ room = width - INDENT - MONTH_WIDTH * len(span) if width else None
438
+ rows = [(textfmt.cut(a, max(NAME_FLOOR, room)) if width else a, *[tl[a].get(m) or "·" for m in span]) for a in ranked[:limit]]
412
439
  return _section(f"Timeline ({_month_label(span[0])} → {_month_label(span[-1])})", columns, rows, caption=_more(len(ranked), limit))
413
440
 
414
441
 
415
442
  def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
416
- """Change frequency times size, Tornhill-style. Files no longer in the tree sort last."""
443
+ """Change frequency times size, Tornhill-style. Files no longer in the tree sort last. Drawn
444
+ under `--full` and in the Markdown export only; the default terminal report leaves it to the
445
+ watch list, which ranks the same files. Built only for those two, it has no row cap of its
446
+ own outside Markdown's."""
417
447
  authors = {a["entity"]: a["n-authors"] for a in report.get("authors") or []}
418
448
  ages = {a["entity"]: a["age-months"] for a in report.get("age") or []}
419
449
  fixes = {f["entity"]: f["n-fixes"] for f in report.get("fixes") or []}
@@ -443,10 +473,9 @@ def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
443
473
  ("trend", RIGHT)]
444
474
  if full is not True:
445
475
  columns, rows = _keep(columns, rows, ["file", "revs", "lines", "fixes", "authors", "trend"])
446
- rows = _shorten(rows, width, columns)
447
476
  note = None if rows else _empty_note(None, hidden_note, "no source hotspots")
448
477
  notes = [c for c in (_more(len(scored), limit), None if note else hidden_note) if c]
449
- if series and full is not False: # the tight report keeps its captions short
478
+ if series:
450
479
  notes.append(f"trend sampled for the top {TREND_TOP} hotspots") # the rest of the column is empty by design
451
480
  return _section(title, columns, rows, note=note, caption="; ".join(notes) or None)
452
481
 
@@ -602,16 +631,19 @@ def health_section(report: dict, full: bool = True, width=None) -> dict:
602
631
 
603
632
  BUILDERS = [watch_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
604
633
  hotspots_section, coupling_section, age_section, functions_section, health_section]
605
- DESCRIPTIVE = {"size", "activity", "age"} # interesting once, rarely change what you do next: `--full` only
634
+ # `--full` and Markdown only: Size, Activity and Code age are interesting once and rarely change what you
635
+ # do next; Hotspots ranks the files the watch list already leads with, by the same product.
636
+ FULL_ONLY = {"size", "activity", "age", "hotspots"}
606
637
 
607
638
 
608
639
  def sections(report: dict, full: bool = True, width=None) -> list:
609
640
  """Every section as a dict with an `id` (the builder's name without _section). The default terminal
610
- report (`full` False) leaves the descriptive ones out; `full` True and Markdown keep them."""
641
+ report (`full` False) leaves out the sections in FULL_ONLY (size, activity, code age and
642
+ hotspots); `full` True and Markdown keep them."""
611
643
  out = []
612
644
  for b in BUILDERS:
613
645
  sid = b.__name__[:-len("_section")]
614
- if full is False and sid in DESCRIPTIVE:
646
+ if full is False and sid in FULL_ONLY:
615
647
  continue
616
648
  sec = b(report, full, width)
617
649
  sec["id"] = sid
@@ -22,6 +22,11 @@ def shorten_path(path: str, max_len: int) -> str:
22
22
  return candidates[-1]
23
23
 
24
24
 
25
+ def cut(name: str, cap: int) -> str:
26
+ """`name`, unchanged if it fits in `cap` characters, else cut to exactly `cap` ending in the ellipsis."""
27
+ return name if len(name) <= cap else name[:cap - 1] + ELLIPSIS
28
+
29
+
25
30
  def times(n: int) -> str:
26
31
  """How often something happened, in words for the small numbers: once, twice, 3 times."""
27
32
  return {1: "once", 2: "twice"}.get(n, f"{n} times")
@@ -33,6 +33,9 @@ def sample_dates(first: str, last: str, n: int) -> list:
33
33
  return out
34
34
 
35
35
 
36
+ GROWTH_FLOOR = 25 # percent in a year: below it a hotspot's complexity is not said to be growing
37
+
38
+
36
39
  def change_over_year(series: list, last_date: str) -> str:
37
40
  if len(series) < 2:
38
41
  return "-"
@@ -1,35 +1,26 @@
1
- """The watch list: the source files most likely to be fixed next, and what else counts against each.
1
+ """The watch list: the source files most likely to be fixed next, and what to look at in each.
2
2
 
3
3
  Each source file that is still in the tree and changed more than once is ranked by revisions × lines
4
4
  of code, the product the Hotspots table uses: measured at six cut-offs on three repositories
5
5
  (docs/validation.md), that product named more of the files fixed in the following six months than
6
6
  any weighting of fixes, complexity and ownership did. Those signals are the reasons printed beside
7
- each file: how often it was fixed lately, who alone owns it, its most complex function, what it
8
- always changes with. A file's score is its share, in percent, of all scored files' revisions × lines
9
- of code, so the scores of the whole list add up to 100 and a change's `--risk` total is the share of
10
- that mass the change touches; one enormous file takes a large share, as it should, and lowers the
11
- others' only by what it adds to the whole. The two factor-product scorings the list used to rank by,
12
- churn × (1 + recent fixes) × (1 + complexity) × (1.5 if single-owned) with each factor a share of the
13
- largest value ("max") or a rank among the scored files ("rank"), stay selectable so gitmole.evaluate
14
- can keep comparing them.
7
+ each file: how often it was fixed lately, who alone owns it, its most complex function, how much its
8
+ complexity grew in the last year, what it always changes with. A file's score is its share, in
9
+ percent, of all scored files' revisions × lines of code, so the scores of the whole list add up to
10
+ 100 and a change's `--risk` total is the share of that mass the change touches; one enormous file
11
+ takes a large share, as it should, and lowers the others' only by what it adds to the whole. The
12
+ factor products the list used to rank by live in gitmole.evaluate, which still compares them with it.
15
13
  Complexity is scc's per-file total, which exists for every file on one scale; lizard's most complex
16
14
  function in the file is what the reasons name, since lizard has no reader for shell, Terraform,
17
15
  Makefiles and the like."""
18
16
  from __future__ import annotations
19
17
 
20
- import bisect
21
18
  from collections import Counter, defaultdict
22
19
 
23
- try:
24
- from . import filetypes, hotspots, textfmt
25
- except ImportError: # pragma: no cover - not run as a script, but keep the package pattern
26
- import filetypes
27
- import hotspots
28
- import textfmt
20
+ from . import filetypes, hotspots, textfmt, trend
29
21
 
30
22
  CCN_FLOOR = 10 # lizard's own "complex" threshold: below it a function is not worth naming
31
23
  SOLO_SHARE = 0.9 # one author wrote at least this much of the file: single ownership
32
- SOLO_WEIGHT = 1.5 # how much single ownership lifts the factor-product scores
33
24
  COMPANION_DEGREE = 50 # a coupling worth mentioning
34
25
  COMPANION_REVS = 5 # ...over enough shared revisions to be a pattern
35
26
 
@@ -67,39 +58,16 @@ def _worst_function(report: dict) -> dict:
67
58
  return worst
68
59
 
69
60
 
70
- def _by_max(values: list, inclusive: bool):
71
- """x as a share of the largest value: the scaling the factor product first shipped with. One outlier
72
- moves everyone; inclusive is ignored here, since a share of the largest value has no edge to choose."""
73
- top = max(values)
74
- return lambda x: x / top if top else 0.0
75
-
76
-
77
- def _by_rank(values: list, inclusive: bool):
78
- """x as the share of the scored files at or below it (inclusive), or strictly below it. Churn is
79
- inclusive, so the most-changed file is 1 and no file is 0; fixes and complexity are strict, so a
80
- file with none of either gets no lift, as under _by_max. An outlier is one more file, not a new
81
- scale."""
82
- ordered = sorted(values)
83
- cut = bisect.bisect_right if inclusive else bisect.bisect_left
84
- return lambda x: cut(ordered, x) / len(ordered)
85
-
86
-
87
- SCALINGS = {"max": _by_max, "rank": _by_rank}
88
- SCORINGS = ("hotspot", *SCALINGS)
89
-
90
-
91
- def risks(report: dict, min_revs: int = 2, scoring: str = "hotspot") -> list:
92
- """The watch list: every scored file with its reasons, worst first. `scoring` is "hotspot" (the
93
- default: each file's score is its percentage share of the pool's revisions × lines of code), or
94
- one of the two factor products, "rank" and "max", kept so gitmole.evaluate can compare them with
95
- it."""
96
- if scoring not in SCORINGS:
97
- raise ValueError(f"scoring must be one of {', '.join(SCORINGS)}, got {scoring!r}")
61
+ def risks(report: dict, min_revs: int = 2) -> list:
62
+ """The watch list: every scored file with its reasons, worst first. A file's score is its
63
+ percentage share of the pool's revisions × lines of code."""
98
64
  owners = _owners(report)
99
65
  companions = _companions(report)
100
66
  worst = _worst_function(report)
101
67
  fixes = {f["entity"]: f for f in report.get("fixes") or []}
102
68
  n_authors = {a["entity"]: a["n-authors"] for a in report.get("authors") or []}
69
+ series = (report.get("trend") or {}).get("files") or {}
70
+ last = (report.get("meta") or {}).get("last_date") or ""
103
71
 
104
72
  plumb, derived = filetypes.plumbing_paths(report), hotspots.derived(report)
105
73
  rows = []
@@ -115,34 +83,24 @@ def risks(report: dict, min_revs: int = 2, scoring: str = "hotspot") -> list:
115
83
  rows.append({"file": h["entity"], "revs": h["revs"], "recent_fixes": fx.get("recent-fixes", 0), "fixes": fx.get("n-fixes", 0),
116
84
  "authors": n_authors.get(h["entity"]), "owner": owner, "owner_share": share,
117
85
  "complexity": h["complexity"] or 0, "code": h["code"],
118
- "function": fn, "companions": companions.get(h["entity"], [])})
86
+ "function": fn, "companions": companions.get(h["entity"], []),
87
+ "trend": trend.change_over_year(series[h["entity"]], last) if last and h["entity"] in series else None})
119
88
  if not rows:
120
89
  return []
121
90
 
122
- if scoring == "hotspot":
123
- pool = sum(r["revs"] * r["code"] for r in rows)
124
-
125
- def score(r):
126
- return 100 * (r["revs"] * r["code"]) / pool if pool else 0.0
127
- else:
128
- scale = SCALINGS[scoring]
129
- churn = scale([r["revs"] for r in rows], True)
130
- fixed = scale([r["recent_fixes"] for r in rows], False)
131
- cplx = scale([r["complexity"] for r in rows], False)
132
-
133
- def score(r):
134
- return churn(r["revs"]) * (1 + fixed(r["recent_fixes"])) * (1 + cplx(r["complexity"])) * (SOLO_WEIGHT if r["solo"] else 1)
91
+ pool = sum(r["revs"] * r["code"] for r in rows)
135
92
  for r in rows:
136
93
  r["solo"] = r["authors"] == 1 or r["owner_share"] >= SOLO_SHARE
137
- r["score"] = score(r)
94
+ r["score"] = 100 * (r["revs"] * r["code"]) / pool if pool else 0.0
138
95
  r["reasons"] = _reasons(r)
139
96
  rows.sort(key=lambda r: (-r["score"], -r["revs"], r["file"]))
140
97
  return rows
141
98
 
142
99
 
143
100
  def why_empty(report: dict, min_revs: int = 2) -> str:
144
- """Why risks() came back empty, for the report's one-line note: the honest reason, since
145
- "nothing changed" above a hotspots table full of revisions would be a lie."""
101
+ """Why risks() came back empty, for the report's one-line note: the honest reason, since files
102
+ can well have changed even though none of them scored, and a flat "nothing changed" would be
103
+ a lie about them."""
146
104
  churned = [h for h in hotspots.ranked(report) if h["revs"] >= min_revs]
147
105
  if not churned:
148
106
  return "nothing changed more than once"
@@ -167,6 +125,9 @@ def _reasons(r: dict) -> list:
167
125
  if fn and fn["ccn"] >= CCN_FLOOR:
168
126
  named = f"the function at line {fn['start']}" if fn.get("anonymous") else f"{fn['function']}()"
169
127
  out.append(f"{named} complexity {fn['ccn']}")
128
+ grown = r.get("trend") or ""
129
+ if grown.startswith("+") and int(grown[1:-1]) >= trend.GROWTH_FLOOR:
130
+ out.append(f"complexity {grown} in a year") # the Hotspots table's trend column, which the default report no longer shows
170
131
  if r["companions"]:
171
132
  other, degree = r["companions"][0]
172
133
  more = len(r["companions"]) - 1
@@ -179,9 +140,9 @@ WATCH_TOP = 15 # the same cap the report's --full watch list uses
179
140
 
180
141
 
181
142
  def change_risk(report: dict, files: list) -> dict:
182
- """The watch score of each touched file, and their sum: under the default "hotspot" scoring, that
183
- total is a percentage of the repository's revisions × lines of code. Files the watch list never
184
- scored get 0 and one reason saying why."""
143
+ """The watch score of each touched file, and their sum: that total is a percentage of the
144
+ repository's revisions × lines of code. Files the watch list never scored get 0 and one reason
145
+ saying why."""
185
146
  ranked = risks(report)
186
147
  by_file = {r["file"]: r for r in ranked}
187
148
  watched = {r["file"] for r in ranked[:WATCH_TOP]}
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.8.0
3
+ Version: 0.9.1
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -96,7 +96,7 @@ since 2013, at a pinned commit:
96
96
  ork.js commitLayoutEffectOnFiber() complexity 72
97
97
  packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
98
98
  rk.js beginWork() complexity 52
99
- ranked by revisions × lines of code; the reasons say what else counts against each file
99
+ ranked by revisions × lines of code alone; the reasons say what to look at there
100
100
  6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
101
101
  files that had changed more than once would name 0.3; the 15 most changed would name 7)
102
102
  ```
@@ -108,11 +108,11 @@ ranks by revisions × lines of code: measured at six cut-offs on three
108
108
  repositories
109
109
  ([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
110
110
  that named more of the files fixed next than churn alone, size alone or a
111
- weighted product of fixes, complexity and ownership. Between the header and
112
- that list the full report puts its findings, 16 for react (1 critical, 6
113
- warnings, 9 notes); below it, tables for people, the knowledge map, the
114
- timeline, hotspots with their complexity trend, change coupling, complex
115
- functions and repo health. Every section is explained in
111
+ weighted product of fixes, complexity and ownership. Between the header and that
112
+ list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
113
+ notes); below it, tables for people, the knowledge map, the timeline, change
114
+ coupling, complex functions and repo health; `--full` adds the hotspots table
115
+ behind the list, size, activity and code age. Every section is explained in
116
116
  [docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
117
117
 
118
118
  Reports on repositories you know, each at a pinned commit with a fixed
@@ -54,6 +54,40 @@ class Score(unittest.TestCase):
54
54
  self.assertEqual(evaluate.report_at(commits, "2025-06-01", SIZE, {})["ownership"], [])
55
55
 
56
56
 
57
+ ROWS = [
58
+ {"file": "core/parser.py", "revs": 40, "recent_fixes": 5, "complexity": 40, "solo": True, "code": 800},
59
+ {"file": "web/index.html", "revs": 60, "recent_fixes": 0, "complexity": 0, "solo": False, "code": 4000},
60
+ {"file": "core/util.py", "revs": 30, "recent_fixes": 0, "complexity": 5, "solo": True, "code": 200},
61
+ ]
62
+
63
+
64
+ class FactorProduct(unittest.TestCase):
65
+ def test_both_scalings_lead_with_the_fixed_complex_single_owned_file(self):
66
+ for scaling in ("max", "rank"):
67
+ self.assertEqual(evaluate.factor_product(ROWS, scaling), ["core/parser.py", "web/index.html", "core/util.py"], scaling)
68
+
69
+ def test_a_file_never_fixed_gets_no_lift_from_fixes_under_either_scaling(self):
70
+ for scaling in ("max", "rank"):
71
+ self.assertEqual(evaluate.factor_scores(ROWS, scaling)["web/index.html"], 1.0, f"{scaling}: most changed, no fixes, no complexity, shared")
72
+
73
+ def test_under_rank_scaling_an_outlier_does_not_rescale_the_other_files(self):
74
+ def util(outlier_revs, scaling):
75
+ big = {"file": "core/big.py", "revs": outlier_revs, "recent_fixes": 0, "complexity": 0, "solo": False, "code": 10}
76
+ return evaluate.factor_scores(ROWS + [big], scaling)["core/util.py"]
77
+ self.assertEqual(util(100, "rank"), util(10000, "rank"))
78
+ self.assertNotEqual(util(100, "max"), util(10000, "max"), "what the rank scaling is for")
79
+
80
+ def test_two_files_that_differ_only_in_complexity(self):
81
+ pair = [{"file": "ops/deploy.sh", "revs": 40, "recent_fixes": 0, "complexity": 80, "solo": False, "code": 300},
82
+ {"file": "ops/plain.sh", "revs": 40, "recent_fixes": 0, "complexity": 0, "solo": False, "code": 300}]
83
+ for scaling in ("max", "rank"):
84
+ scores = evaluate.factor_scores(ROWS + pair, scaling)
85
+ self.assertGreater(scores["ops/deploy.sh"], scores["ops/plain.sh"], f"{scaling}: complexity lifts the factor product")
86
+
87
+ def test_no_rows_is_no_list(self):
88
+ self.assertEqual(evaluate.factor_product([], "rank"), [])
89
+
90
+
57
91
  class Table(unittest.TestCase):
58
92
  def test_one_row_per_variant_one_column_per_cut_off_and_a_total(self):
59
93
  text = evaluate.table([("2025-02-28", 3, 40, {"churn": 1, "random (expected)": 0.4}),