gitmole 0.9.0__tar.gz → 0.10.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. {gitmole-0.9.0 → gitmole-0.10.0}/PKG-INFO +9 -54
  2. {gitmole-0.9.0 → gitmole-0.10.0}/README.md +8 -53
  3. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/__init__.py +1 -1
  4. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/backtest.py +17 -6
  5. gitmole-0.10.0/gitmole/classify.py +88 -0
  6. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/cli.py +44 -9
  7. gitmole-0.10.0/gitmole/compare.py +68 -0
  8. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/evaluate.py +7 -5
  9. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/filetypes.py +84 -41
  10. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/findings.py +14 -2
  11. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/render.py +127 -55
  12. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/run.py +40 -0
  13. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/textfmt.py +6 -0
  14. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/watch.py +26 -21
  15. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/PKG-INFO +9 -54
  16. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/SOURCES.txt +4 -0
  17. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_backtest.py +34 -4
  18. gitmole-0.10.0/tests/test_classify.py +71 -0
  19. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_cli.py +81 -2
  20. gitmole-0.10.0/tests/test_compare.py +62 -0
  21. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_evaluate.py +3 -3
  22. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_filetypes.py +73 -9
  23. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_findings.py +13 -0
  24. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_render.py +83 -6
  25. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_render_examples.py +2 -0
  26. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_run.py +32 -1
  27. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_textfmt.py +8 -0
  28. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_watch.py +30 -6
  29. {gitmole-0.9.0 → gitmole-0.10.0}/LICENSE +0 -0
  30. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/__main__.py +0 -0
  31. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/banner.py +0 -0
  32. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/blame.py +0 -0
  33. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/clean.py +0 -0
  34. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/coupling.py +0 -0
  35. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/deps.py +0 -0
  36. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/duplicates.py +0 -0
  37. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/functions.py +0 -0
  38. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/hotspots.py +0 -0
  39. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/identity.py +0 -0
  40. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/knowledge.py +0 -0
  41. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/leaks.py +0 -0
  42. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/load.py +0 -0
  43. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/loss.py +0 -0
  44. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/maat.py +0 -0
  45. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/trend.py +0 -0
  46. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/dependency_links.txt +0 -0
  47. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/entry_points.txt +0 -0
  48. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/requires.txt +0 -0
  49. {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/top_level.txt +0 -0
  50. {gitmole-0.9.0 → gitmole-0.10.0}/pyproject.toml +0 -0
  51. {gitmole-0.9.0 → gitmole-0.10.0}/setup.cfg +0 -0
  52. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_banner.py +0 -0
  53. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_blame.py +0 -0
  54. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_clean.py +0 -0
  55. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_coupling.py +0 -0
  56. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_deps.py +0 -0
  57. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_duplicates.py +0 -0
  58. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_functions.py +0 -0
  59. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_golden.py +0 -0
  60. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_hotspots.py +0 -0
  61. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_identity.py +0 -0
  62. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_knowledge.py +0 -0
  63. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_leaks.py +0 -0
  64. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_load.py +0 -0
  65. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_loss.py +0 -0
  66. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_maat.py +0 -0
  67. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_packaging.py +0 -0
  68. {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_trend.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.9.0
3
+ Version: 0.10.0
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -32,7 +32,6 @@ Free. Any Stack. Local. Offline. Deterministic. Fast.
32
32
  - **Any stack.** It reads what every repository has: the git log, git blame and the files themselves.
33
33
  - **Local & Offline.** Everything runs against a clone on your machine. Nothing is uploaded, nothing phones home; the vulnerability database is a copy you download once.
34
34
  - **Deterministic.** No AI at runtime. Every finding is a plain rule over counts you can recompute by hand. The JSON export carries each finding's rule and the numbers it fired on. The same clone gives the same report every time.
35
- - **Fast.** A 4,400-commit repository takes under thirty seconds.
36
35
 
37
36
  ## Install
38
37
 
@@ -71,59 +70,15 @@ blocks on secrets in source files and still posts the report. Every option:
71
70
 
72
71
  ## What you get
73
72
 
74
- The opening of the report for [react](https://github.com/facebook/react), 21,703 commits
75
- since 2013, at a pinned commit:
76
-
77
- ```text
78
- ╭─ react ──────────────────────────────────────────────────────────────────────────────────────────╮
79
- │ 21703 commits · 2013-05-28 → 2026-09-16 · 1843 identities · branch main │
80
- │ 681,078 lines in 4781 files · JavaScript, TypeScript, Rust, CSS │
81
- │ most commits on Wed at 15:00 · 14% of commits are fixes · 1% of commits are reverts · 19% │
82
- │ of surviving code from 2026 │
83
- │ 1 critical, 6 warnings, 9 notes │
84
- ╰──────────────────────────────────────────────────────────────────────────────────────────────────╯
85
-
86
- ◎ Watch list
87
- file why
88
- ──────────────────────────────────────────────────────────────────────────────────────────────────
89
- packages/react-server/src/ReactFlightServer.js changed 319 times · fixed once in six months ·
90
- visitAsyncNodeImpl() complexity 46
91
- packages/react-server/src/ReactFizzServer.js changed 297 times · fixed twice in six months ·
92
- retryNode() complexity 41
93
- packages/react-reconciler/src/ReactFiberWorkLoo changed 312 times · fixed 4 times in six months
94
- p.js · flushSpawnedWork() complexity 48
95
- packages/react-reconciler/src/ReactFiberCommitW changed 284 times · fixed 28 times ·
96
- ork.js commitLayoutEffectOnFiber() complexity 72
97
- packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
98
- rk.js beginWork() complexity 52
99
- ranked by revisions × lines of code; the reasons say what else counts against each file
100
- 6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
101
- files that had changed more than once would name 0.3; the 15 most changed would name 7)
102
- ```
73
+ Reports on repositories you know, each at a pinned commit, published as gitmole wrote them:
74
+
75
+ | Repository | Commit | Commits | Lines | gitmole run |
76
+ |---|---|---:|---:|---:|
77
+ | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 61 s |
78
+ | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 153 s |
79
+ | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 138 s |
103
80
 
104
- The watch list is the point: the five source files most likely to need a fix
105
- next, the reasons in words, and a backtest that says how the same list, drawn
106
- six months earlier, would have done against the fixes that followed. The list
107
- ranks by revisions × lines of code: measured at six cut-offs on three
108
- repositories
109
- ([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
110
- that named more of the files fixed next than churn alone, size alone or a
111
- weighted product of fixes, complexity and ownership. Between the header and that
112
- list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
113
- notes); below it, tables for people, the knowledge map, the timeline, change
114
- coupling, complex functions and repo health; `--full` adds the hotspots table
115
- behind the list, size, activity and code age. Every section is explained in
116
- [docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
117
-
118
- Reports on repositories you know, each at a pinned commit with a fixed
119
- reference date, published as gitmole wrote them; the repo-health numbers
120
- come from git-sizer over the whole clone, so a fresh clone can differ there:
121
-
122
- | Repository | Commits | Lines | Watch list backtest |
123
- |---|---:|---:|---|
124
- | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | 39,758 | 247,179 | named 15 of the 238 files fixed in the next six months; a random pick would name 4.9, the 15 most changed 15 |
125
- | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | 34,933 | 431,749 | named 15 of the 213 files fixed in the next six months; a random pick would name 3.1, the 15 most changed 13 |
126
- | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | 21,703 | 681,078 | named 11 of the 46 files fixed in the next six months; a random pick would name 0.3, the 15 most changed 7 |
81
+ Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
127
82
 
128
83
  ## The tool set
129
84
 
@@ -12,7 +12,6 @@ Free. Any Stack. Local. Offline. Deterministic. Fast.
12
12
  - **Any stack.** It reads what every repository has: the git log, git blame and the files themselves.
13
13
  - **Local & Offline.** Everything runs against a clone on your machine. Nothing is uploaded, nothing phones home; the vulnerability database is a copy you download once.
14
14
  - **Deterministic.** No AI at runtime. Every finding is a plain rule over counts you can recompute by hand. The JSON export carries each finding's rule and the numbers it fired on. The same clone gives the same report every time.
15
- - **Fast.** A 4,400-commit repository takes under thirty seconds.
16
15
 
17
16
  ## Install
18
17
 
@@ -51,59 +50,15 @@ blocks on secrets in source files and still posts the report. Every option:
51
50
 
52
51
  ## What you get
53
52
 
54
- The opening of the report for [react](https://github.com/facebook/react), 21,703 commits
55
- since 2013, at a pinned commit:
56
-
57
- ```text
58
- ╭─ react ──────────────────────────────────────────────────────────────────────────────────────────╮
59
- │ 21703 commits · 2013-05-28 → 2026-09-16 · 1843 identities · branch main │
60
- │ 681,078 lines in 4781 files · JavaScript, TypeScript, Rust, CSS │
61
- │ most commits on Wed at 15:00 · 14% of commits are fixes · 1% of commits are reverts · 19% │
62
- │ of surviving code from 2026 │
63
- │ 1 critical, 6 warnings, 9 notes │
64
- ╰──────────────────────────────────────────────────────────────────────────────────────────────────╯
65
-
66
- ◎ Watch list
67
- file why
68
- ──────────────────────────────────────────────────────────────────────────────────────────────────
69
- packages/react-server/src/ReactFlightServer.js changed 319 times · fixed once in six months ·
70
- visitAsyncNodeImpl() complexity 46
71
- packages/react-server/src/ReactFizzServer.js changed 297 times · fixed twice in six months ·
72
- retryNode() complexity 41
73
- packages/react-reconciler/src/ReactFiberWorkLoo changed 312 times · fixed 4 times in six months
74
- p.js · flushSpawnedWork() complexity 48
75
- packages/react-reconciler/src/ReactFiberCommitW changed 284 times · fixed 28 times ·
76
- ork.js commitLayoutEffectOnFiber() complexity 72
77
- packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
78
- rk.js beginWork() complexity 52
79
- ranked by revisions × lines of code; the reasons say what else counts against each file
80
- 6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
81
- files that had changed more than once would name 0.3; the 15 most changed would name 7)
82
- ```
53
+ Reports on repositories you know, each at a pinned commit, published as gitmole wrote them:
54
+
55
+ | Repository | Commit | Commits | Lines | gitmole run |
56
+ |---|---|---:|---:|---:|
57
+ | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 61 s |
58
+ | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 153 s |
59
+ | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 138 s |
83
60
 
84
- The watch list is the point: the five source files most likely to need a fix
85
- next, the reasons in words, and a backtest that says how the same list, drawn
86
- six months earlier, would have done against the fixes that followed. The list
87
- ranks by revisions × lines of code: measured at six cut-offs on three
88
- repositories
89
- ([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
90
- that named more of the files fixed next than churn alone, size alone or a
91
- weighted product of fixes, complexity and ownership. Between the header and that
92
- list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
93
- notes); below it, tables for people, the knowledge map, the timeline, change
94
- coupling, complex functions and repo health; `--full` adds the hotspots table
95
- behind the list, size, activity and code age. Every section is explained in
96
- [docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
97
-
98
- Reports on repositories you know, each at a pinned commit with a fixed
99
- reference date, published as gitmole wrote them; the repo-health numbers
100
- come from git-sizer over the whole clone, so a fresh clone can differ there:
101
-
102
- | Repository | Commits | Lines | Watch list backtest |
103
- |---|---:|---:|---|
104
- | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | 39,758 | 247,179 | named 15 of the 238 files fixed in the next six months; a random pick would name 4.9, the 15 most changed 15 |
105
- | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | 34,933 | 431,749 | named 15 of the 213 files fixed in the next six months; a random pick would name 3.1, the 15 most changed 13 |
106
- | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | 21,703 | 681,078 | named 11 of the 46 files fixed in the next six months; a random pick would name 0.3, the 15 most changed 7 |
61
+ Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
107
62
 
108
63
  ## The tool set
109
64
 
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.9.0"
3
+ __version__ = "0.10.0"
@@ -22,9 +22,11 @@ import tempfile
22
22
  from . import filetypes, load, maat, trend
23
23
 
24
24
 
25
- def size_at(repo: str, rev: str, out_dir: str) -> str:
26
- """scc's --by-file JSON over the tree at rev, exported through a temporary index so no
27
- archive is held in memory and export-ignore attributes do not thin the tree.
25
+ def snapshot_at(repo: str, rev: str, out_dir: str) -> tuple:
26
+ """The tree at rev, measured and classified as it was then: scc's --by-file JSON, the generated files
27
+ and the vendored paths. Exported through a temporary index so no archive is held in memory and
28
+ export-ignore attributes do not thin the tree; the same index lets git check-attr read that tree's
29
+ .gitattributes, so the classification is the cut-off's, not HEAD's.
28
30
 
29
31
  The tree is written under `out_dir`, not the system temp directory: a SIGKILL cannot run the
30
32
  cleanup, and a checkout left next to the report is one the next run clears away."""
@@ -34,7 +36,15 @@ def size_at(repo: str, rev: str, out_dir: str) -> str:
34
36
  env = dict(os.environ, GIT_INDEX_FILE=os.path.join(tmp, "index"))
35
37
  subprocess.run(["git", "read-tree", rev], cwd=repo, env=env, check=True, capture_output=True, text=True)
36
38
  subprocess.run(["git", "checkout-index", "-a", f"--prefix={tree}/"], cwd=repo, env=env, check=True, capture_output=True, text=True)
37
- return subprocess.run(["scc", "--by-file", "--format", "json"], cwd=tree, capture_output=True, text=True, check=True).stdout
39
+ size = subprocess.run(["scc", "--by-file", "--format", "json"], cwd=tree, capture_output=True, text=True, check=True).stdout
40
+ # the text files of that tree, as blame.text_files lists HEAD's: git grep prints "rev:path"
41
+ proc = subprocess.run([*filetypes.GIT, "grep", "-I", "--name-only", "-z", "-e", "", rev], cwd=repo, capture_output=True)
42
+ if proc.returncode not in (0, 1): # 1 is grep's "no match" (an empty tree), not a failure
43
+ raise subprocess.CalledProcessError(proc.returncode, proc.args, proc.stdout.decode("utf-8", "replace"),
44
+ proc.stderr.decode("utf-8", "replace"))
45
+ paths = sorted(p.decode("utf-8", "surrogateescape").split(":", 1)[1] for p in proc.stdout.split(b"\0") if p)
46
+ attrs = filetypes.attributes(repo, paths, cached=True, env=env)
47
+ return size, filetypes.generated_files(tree, paths, attrs), filetypes.vendored_paths(tree, paths, attrs)
38
48
 
39
49
 
40
50
  def main(argv=None) -> int:
@@ -66,7 +76,7 @@ def main(argv=None) -> int:
66
76
  types = filetypes.parse(meta.get("file_types"))
67
77
  maat.write_all(log_path, sub, os.path.join(args.out, "meta.json") if "aliases" in meta else None, types, now=until, until=until)
68
78
  try:
69
- size = size_at(args.repo, rev, args.out)
79
+ size, generated, vendored = snapshot_at(args.repo, rev, args.out)
70
80
  except subprocess.CalledProcessError as e:
71
81
  first = ((e.stderr or "").strip().splitlines() or [f"{' '.join(e.cmd)} exited {e.returncode}"])[0]
72
82
  print(f"backtest: {first}", file=sys.stderr)
@@ -74,7 +84,8 @@ def main(argv=None) -> int:
74
84
  with open(os.path.join(sub, "size.json"), "w", encoding="utf-8") as fh:
75
85
  fh.write(size)
76
86
  with open(os.path.join(sub, "meta.json"), "w", encoding="utf-8") as fh:
77
- json.dump({"now": until, "last_date": until, "file_types": meta.get("file_types"), "aliases": meta.get("aliases", {})}, fh)
87
+ json.dump({"now": until, "last_date": until, "file_types": meta.get("file_types"), "aliases": meta.get("aliases", {}),
88
+ "generated": generated, "vendored": vendored}, fh)
78
89
  return 0
79
90
 
80
91
 
@@ -0,0 +1,88 @@
1
+ """Why a file is out of the scored pool: one answer for every table, the watch list and --risk.
2
+
3
+ Built once per report, so the type filter, scc's rows, the generated and vendored lists, the amalgamations
4
+ and the plumbing are read once and every section asks the same object. Not in filetypes, which blame.py
5
+ and maat.py import as scripts and which must stay free of package imports."""
6
+ from __future__ import annotations
7
+
8
+ from collections import Counter
9
+
10
+ from . import filetypes, hotspots
11
+
12
+ # In descriptive order: what a reader would call the file first. package.json is a release file before
13
+ # it is not a source type; vendor/x_test.go is vendored before it is a test file; a generated test is
14
+ # generated. The order is part of the contract: reason() is the first that applies.
15
+ REASONS = ("generated", "vendored", "test file", "example code", "release file", "amalgamation", "not a source type", "not in the tree")
16
+
17
+ # The coverage line's nouns, by count.
18
+ _NOUNS = {"scored": ("scored", "scored"), "generated": ("generated", "generated"), "vendored": ("vendored", "vendored"),
19
+ "test file": ("test file", "test files"), "example code": ("example code", "example code"),
20
+ "release file": ("release file", "release files"), "amalgamation": ("amalgamation", "amalgamations"),
21
+ "not a source type": ("not a source type", "not a source type"), "not counted by scc": ("not counted by scc", "not counted by scc")}
22
+
23
+
24
+ class Classifier:
25
+ def __init__(self, report: dict):
26
+ meta = report.get("meta") or {}
27
+ # a run records its --file-types spec (None for the default list); a run from before that record
28
+ # was measured unfiltered and is classified unfiltered, as load.load_report filters scc
29
+ self.types = filetypes.parse(meta["file_types"]) if "file_types" in meta else None
30
+ self.tree = (report.get("size") or {}).get("files") or {}
31
+ self.generated = set(meta.get("generated") or [])
32
+ self.vendored = filetypes.vendor_dirs(report)
33
+ self.amalgamations = hotspots.amalgamations(report)
34
+ self.plumbing = filetypes.plumbing_paths(report)
35
+ self._reasons = {} # memoised per instance: every hide pass in a report asks the same paths again
36
+
37
+ def reasons(self, path: str) -> tuple:
38
+ """Every reason that applies, in REASONS order. `not in the tree` needs a tree to judge by: a run
39
+ whose scc step was killed classifies nothing as gone, so it cannot empty every table. Cached per
40
+ path and returned as a tuple so callers cannot mutate the cached result."""
41
+ if path in self._reasons:
42
+ return self._reasons[path]
43
+ out = []
44
+ if path in self.generated:
45
+ out.append("generated")
46
+ if filetypes.is_vendored(path, self.vendored):
47
+ out.append("vendored")
48
+ if filetypes.is_test_path(path):
49
+ out.append("test file")
50
+ if filetypes.is_sample_path(path):
51
+ out.append("example code")
52
+ if filetypes.is_release(path, self.plumbing):
53
+ out.append("release file")
54
+ if path in self.amalgamations:
55
+ out.append("amalgamation")
56
+ if not filetypes.matches(path, self.types):
57
+ out.append("not a source type")
58
+ if self.tree and path not in self.tree:
59
+ out.append("not in the tree")
60
+ self._reasons[path] = tuple(out)
61
+ return self._reasons[path]
62
+
63
+ def reason(self, path: str):
64
+ """The first reason, or None for a file in the scored pool."""
65
+ found = self.reasons(path)
66
+ return found[0] if found else None
67
+
68
+ def excluded(self, path: str, reasons) -> bool:
69
+ """Whether a table that hides `reasons` hides this file: any of its reasons is enough, so a table
70
+ that hides tests still hides a vendored test."""
71
+ return any(r in reasons for r in self.reasons(path))
72
+
73
+
74
+ def coverage(classifier: Classifier, tracked: list) -> dict:
75
+ """How many tracked files each reason claims, `scored` for none. A tracked file with no scc row is a
76
+ type scc does not classify rather than a deleted one, so it counts as `not counted by scc`."""
77
+ counts = Counter()
78
+ for path in tracked:
79
+ reason = classifier.reason(path)
80
+ counts["scored" if reason is None else "not counted by scc" if reason == "not in the tree" else reason] += 1
81
+ return dict(counts)
82
+
83
+
84
+ def coverage_line(cov: dict) -> str:
85
+ """'4,781 files: 3,900 scored · 610 test files · …', the buckets in REASONS order."""
86
+ order = ["scored", *REASONS[:-1], "not counted by scc"]
87
+ parts = [f"{cov[k]:,} {_NOUNS[k][0 if cov[k] == 1 else 1]}" for k in order if cov.get(k)]
88
+ return f"{sum(cov.values()):,} files: " + " · ".join(parts)
@@ -44,6 +44,7 @@ def parse_args(argv):
44
44
  p.add_argument("--fail-on", choices=findings.SEVERITIES, help="exit 3 if any finding is at this severity or worse")
45
45
  p.add_argument("--risk", metavar="BASE", help="score the files changed since BASE (merge base with HEAD) by their share of the repository's revisions × lines of code; needs a local path")
46
46
  p.add_argument("--risk-threshold", type=float, metavar="N", help="with --risk: exit 3 when the changed files hold more than N percent of the repository's revisions × lines of code")
47
+ p.add_argument("--compare", metavar="BEFORE_JSON", help="add a 'Since last report' section against an earlier --json export of the same clone")
47
48
  p.add_argument("--version", action="version", version=f"gitmole {__version__}")
48
49
  return p.parse_args(argv)
49
50
 
@@ -127,11 +128,13 @@ def _check_args(args, err, kind=None) -> int | None:
127
128
  if kind is None:
128
129
  bad = ("--yes needs --clean" if args.yes and not args.clean else
129
130
  "target required" if args.target is None and not args.clean else
130
- "--risk-threshold needs --risk" if args.risk_threshold is not None and not args.risk else None)
131
+ "--risk-threshold needs --risk" if args.risk_threshold is not None and not args.risk else
132
+ "--compare: no such file: " + args.compare if args.compare and not os.path.isfile(args.compare) else None)
131
133
  elif kind == "path":
132
134
  bad = None
133
135
  else:
134
- bad = ("--risk needs a local path" if args.risk else
136
+ bad = ("--compare needs one repository, not owner/*" if args.compare and kind == "org" else
137
+ "--risk needs a local path" if args.risk else
135
138
  "--list-file-types needs a local path" if args.list_file_types else None)
136
139
  if bad:
137
140
  err.print(f"[red]{bad}[/red]")
@@ -297,12 +300,14 @@ def _meta_for_run(repo_dir: str, args, estimate, age_ok: bool, plots_ok: bool, p
297
300
  types_spec = _types_spec(args.file_types)
298
301
 
299
302
  meta = run.collect_meta(repo_dir, since=args.since_date)
303
+ meta["run"] = run.manifest(repo_dir, args) # what produced this report: commit, gitmole and tool versions, the options
300
304
  meta["file_types"] = types_spec # the loader filters scc's size data the way every other step was filtered
301
305
  meta["gone_months"] = args.gone
302
- ignore = list(run.DATA_IGNORES if args.ignore_data else []) + list(args.ignore)
303
- tracked = blame.text_files(repo_dir, ignore)
304
- meta["generated"] = filetypes.generated_files(repo_dir, tracked) # hidden from the tables, out of the findings
305
- meta["vendored"] = filetypes.vendored_dirs(repo_dir, tracked) # somebody else's code, by the licence it carries
306
+ tracked = blame.text_files(repo_dir) # every tracked text file: --ignore shapes blame, functions and duplicates, never what a file is
307
+ attrs = filetypes.attributes(repo_dir, tracked) # one git check-attr pass, shared by the two lists below
308
+ meta["generated"] = filetypes.generated_files(repo_dir, tracked, attrs=attrs) # hidden from the tables, out of the findings
309
+ meta["vendored"] = filetypes.vendored_paths(repo_dir, tracked, attrs=attrs) # somebody else's code, by the licence it carries or the attribute it declares
310
+ meta["credential_files"] = filetypes.credential_files(filetypes.git_paths(repo_dir, "ls-files")) # by name, over every tracked file
306
311
  if args.since_date and meta["commits"] == 0:
307
312
  raise NoCommits(f"no commits since {args.since_date}; widen --since")
308
313
  if args.now:
@@ -348,6 +353,17 @@ def _record_statuses(meta, results, age_ok: bool, plots_ok: bool, lizard_ok: boo
348
353
  meta["steps"] = {name: "run" if rc == 0 else (rc if isinstance(rc, str) else "failed") for name, rc in results.items()}
349
354
 
350
355
 
356
+ def _coverage(repo_dir: str, out_dir: str) -> dict:
357
+ """How many tracked text files each reason claims, from the report as the steps left it. An
358
+ unreadable output directory (a killed run) records nothing rather than failing the run."""
359
+ from . import classify
360
+ try:
361
+ report = load.load_report(out_dir, nested=False)
362
+ except load.Unreadable:
363
+ return {}
364
+ return classify.coverage(classify.Classifier(report), blame.text_files(repo_dir))
365
+
366
+
351
367
  def _analyse(repo_dir: str, out_dir: str, args, ui: Console, planner, estimator) -> None:
352
368
  """Run the whole pipeline for one repository into out_dir."""
353
369
  os.makedirs(os.path.join(out_dir, "theseus"), exist_ok=True)
@@ -372,6 +388,7 @@ def _analyse(repo_dir: str, out_dir: str, args, ui: Console, planner, estimator)
372
388
  raise Interrupted()
373
389
 
374
390
  _record_statuses(meta, results, age_ok, plots_ok, lizard_ok, cut, duplicates_ok)
391
+ meta["coverage"] = _coverage(repo_dir, out_dir)
375
392
  run.save_meta(meta, out_dir)
376
393
 
377
394
  failed = [n for n, rc in results.items() if rc != 0]
@@ -505,12 +522,30 @@ def _render(out_dir: str, console: Console, ui: Console, args, err: Console) ->
505
522
  return 2
506
523
  from . import watch
507
524
  risk = {"base": args.risk, **watch.change_risk(report, files)}
525
+ comparison = None
526
+ if args.compare:
527
+ from . import compare as _compare
528
+ try:
529
+ with open(args.compare, encoding="utf-8") as fh:
530
+ before = json.load(fh)
531
+ except (OSError, ValueError) as e:
532
+ err.print(f"[red]--compare {args.compare}:[/red] {e}", soft_wrap=True)
533
+ return 2
534
+ if not _compare.is_export(before):
535
+ err.print(f"[red]--compare {args.compare}:[/red] not a gitmole --json export (it needs meta, findings with rule ids, and watch; "
536
+ "exports from before 0.8.0 have no rule ids)", soft_wrap=True)
537
+ return 2
538
+ if before["meta"].get("name") != report["meta"].get("name"):
539
+ err.print(f"[red]--compare {args.compare}:[/red] it describes {before['meta'].get('name')}, this run describes {report['meta'].get('name')}; "
540
+ "the two exports must be of the same clone", soft_wrap=True)
541
+ return 2
542
+ comparison = _compare.compare(before, report, found)
508
543
  if args.json:
509
- _write(json.dumps(render.to_json(report, found, risk=risk), indent=2) + "\n", args.json, console)
544
+ _write(json.dumps(render.to_json(report, found, risk=risk, compare=comparison), indent=2) + "\n", args.json, console)
510
545
  if args.markdown:
511
- _write(render.markdown(report, found, full=args.full, risk=risk, base=args.risk), args.markdown, console)
546
+ _write(render.markdown(report, found, full=args.full, risk=risk, base=args.risk, compare=comparison), args.markdown, console)
512
547
  if "-" not in (args.json, args.markdown):
513
- render.report(report, found, console, full=args.full, risk=risk, base=args.risk)
548
+ render.report(report, found, console, full=args.full, risk=risk, base=args.risk, compare=comparison)
514
549
  if args.fail_on and any(findings.SEVERITIES.index(f["severity"]) <= findings.SEVERITIES.index(args.fail_on) for f in found):
515
550
  return 3
516
551
  if risk is not None and args.risk_threshold is not None and risk["total"] > args.risk_threshold:
@@ -0,0 +1,68 @@
1
+ """What changed since an earlier --json export: findings that are new, resolved or persisting, files that
2
+ entered or left the watch list, the tally before and after. Pure over two report dicts."""
3
+ from __future__ import annotations
4
+
5
+ from . import findings, watch
6
+
7
+ # A rule emits one finding per report, except these two, which emit one per row; the evidence field
8
+ # that tells the rows apart joins the rule id in the key.
9
+ KEY_FIELDS = {"repo_health": "metric", "placeholder_identity": "email"}
10
+
11
+
12
+ def key(finding: dict) -> tuple:
13
+ """A finding's identity across two reports: the rule id alone, unless the rule is one of KEY_FIELDS,
14
+ where one report can hold several findings for the same rule and the evidence field is what tells
15
+ them apart (a metric name, an email)."""
16
+ rid = finding["rule"]["id"]
17
+ field = KEY_FIELDS.get(rid)
18
+ return (rid, (finding.get("evidence") or {}).get(field)) if field else (rid,)
19
+
20
+
21
+ def is_export(data) -> bool:
22
+ """A gitmole --json export: a report with its findings and watch list. Every finding must carry a
23
+ rule id and a severity, the shape key() and _tally() read without a default — exports from before
24
+ 0.8.0 have findings without a `rule`, and would otherwise pass this check and crash later on a bare
25
+ KeyError instead of being refused here."""
26
+ if not (isinstance(data, dict) and isinstance(data.get("meta"), dict) and "findings" in data and "watch" in data):
27
+ return False
28
+ found, watch_rows = data.get("findings"), data.get("watch")
29
+ return (isinstance(found, list) and isinstance(watch_rows, list) and all(isinstance(r, dict) for r in watch_rows) and
30
+ all(isinstance(f, dict) and isinstance(f.get("rule"), dict) and "id" in f["rule"] and "severity" in f for f in found))
31
+
32
+
33
+ def _tally(found: list) -> dict:
34
+ counts = {s: 0 for s in findings.SEVERITIES}
35
+ for f in found:
36
+ counts[f["severity"]] += 1
37
+ return counts
38
+
39
+
40
+ def _ordered(found: list) -> list:
41
+ return sorted(found, key=lambda f: (findings.SEVERITIES.index(f["severity"]), f["title"], str(key(f))))
42
+
43
+
44
+ def _options_differ(before_meta: dict, after_meta: dict) -> list:
45
+ """The options that change what a run sees: --since and --file-types from meta's top level, --ignore and
46
+ --ignore-data from the manifest when both exports have one. --deep is recorded there too but only
47
+ decides whether code age, plots and duplicates ran, none of which reach the findings or the watch list."""
48
+ out = [name for name in ("since", "file_types") if before_meta.get(name) != after_meta.get(name)]
49
+ b, a = (before_meta.get("run") or {}).get("options"), (after_meta.get("run") or {}).get("options")
50
+ if b is not None and a is not None:
51
+ out += [name for name in ("ignore", "ignore_data") if b.get(name) != a.get(name)]
52
+ return out
53
+
54
+
55
+ def compare(before: dict, report: dict, found: list, top: int = watch.WATCH_TOP) -> dict:
56
+ """before: an earlier export; report and found: this run's loaded report and its findings."""
57
+ b = {key(f): f for f in before.get("findings") or []}
58
+ a = {key(f): f for f in found}
59
+ persisting = [{**a[k], "was": b[k]["severity"]} for k in a if k in b]
60
+ before_watch = [f for f in (r.get("file") for r in (before.get("watch") or [])[:top]) if f]
61
+ after_watch = [r["file"] for r in watch.risks(report)[:top]]
62
+ meta_b, meta_a = before.get("meta") or {}, report.get("meta") or {}
63
+ return {"new": _ordered([a[k] for k in a if k not in b]), "resolved": _ordered([b[k] for k in b if k not in a]),
64
+ "persisting": _ordered(persisting),
65
+ "watch_entered": [f for f in after_watch if f not in before_watch], "watch_left": [f for f in before_watch if f not in after_watch],
66
+ "tally": {"before": _tally(before.get("findings") or []), "after": _tally(found)},
67
+ "before": {"commit": (meta_b.get("run") or {}).get("commit"), "date": meta_b.get("last_date"),
68
+ "options_differ": _options_differ(meta_b, meta_a)}}
@@ -42,13 +42,14 @@ def fixed_between(commits: list, start: str, end: str) -> set:
42
42
  for p, _, _ in c["files"] if not filetypes.is_test_path(p)}
43
43
 
44
44
 
45
- def report_at(commits: list, t: str, size: dict, meta: dict) -> dict:
46
- """The report watch.risks reads, from the commits before `t` and scc's listing of the tree at `t`.
45
+ def report_at(commits: list, t: str, size: dict, meta: dict, generated: list, vendored: list) -> dict:
46
+ """The report watch.risks reads, from the commits before `t`, scc's listing of the tree at `t`, and
47
+ that tree's own generated and vendored files (snapshot_at classified the cut-off, not HEAD).
47
48
  No coupling and no functions: neither enters the score, and the pipeline's own backtest has no functions either."""
48
49
  past = maat.in_window(commits, until=t)
49
50
  bots = {b["name"] for b in meta.get("bots") or []}
50
51
  ownership = [r for r in maat.entity_ownership(past) if r["author"] not in bots and not identity.is_bot(r["author"])]
51
- return {"meta": {"now": t, "generated": meta.get("generated") or []}, "size": size, "revisions": maat.revisions(past),
52
+ return {"meta": {"now": t, "generated": generated, "vendored": vendored}, "size": size, "revisions": maat.revisions(past),
52
53
  "plumbing": maat.plumbing(past), "authors": maat.authors(past), "ownership": ownership,
53
54
  "fixes": maat.fixes(past, now=t), "coupling": [], "functions": []}
54
55
 
@@ -163,8 +164,9 @@ def main(argv=None) -> int:
163
164
  rev = trend.rev_before(args.repo, t, end_of_day=False)
164
165
  if not rev:
165
166
  continue # the history does not reach back this far
166
- size = load.parse_scc(backtest.size_at(args.repo, rev, args.out), types)
167
- report = report_at(commits, t, size, meta)
167
+ size_json, generated, vendored = backtest.snapshot_at(args.repo, rev, args.out)
168
+ size = load.parse_scc(size_json, types)
169
+ report = report_at(commits, t, size, meta, generated, vendored)
168
170
  fixed = fixed_between(commits, t, months_after(t, args.horizon))
169
171
  pool = set(variants(report)["churn"])
170
172
  results.append((t, len(fixed & pool), len(pool), score(report, fixed, args.top)))