gitmole 0.9.0__tar.gz → 0.10.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gitmole-0.9.0 → gitmole-0.10.0}/PKG-INFO +9 -54
- {gitmole-0.9.0 → gitmole-0.10.0}/README.md +8 -53
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/__init__.py +1 -1
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/backtest.py +17 -6
- gitmole-0.10.0/gitmole/classify.py +88 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/cli.py +44 -9
- gitmole-0.10.0/gitmole/compare.py +68 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/evaluate.py +7 -5
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/filetypes.py +84 -41
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/findings.py +14 -2
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/render.py +127 -55
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/run.py +40 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/textfmt.py +6 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/watch.py +26 -21
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/PKG-INFO +9 -54
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/SOURCES.txt +4 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_backtest.py +34 -4
- gitmole-0.10.0/tests/test_classify.py +71 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_cli.py +81 -2
- gitmole-0.10.0/tests/test_compare.py +62 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_evaluate.py +3 -3
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_filetypes.py +73 -9
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_findings.py +13 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_render.py +83 -6
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_render_examples.py +2 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_run.py +32 -1
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_textfmt.py +8 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_watch.py +30 -6
- {gitmole-0.9.0 → gitmole-0.10.0}/LICENSE +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/__main__.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/banner.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/blame.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/clean.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/coupling.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/deps.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/duplicates.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/functions.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/hotspots.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/identity.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/knowledge.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/leaks.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/load.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/loss.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/maat.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole/trend.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/dependency_links.txt +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/entry_points.txt +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/requires.txt +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/gitmole.egg-info/top_level.txt +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/pyproject.toml +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/setup.cfg +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_banner.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_blame.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_clean.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_coupling.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_deps.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_duplicates.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_functions.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_golden.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_hotspots.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_identity.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_knowledge.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_leaks.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_load.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_loss.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_maat.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_packaging.py +0 -0
- {gitmole-0.9.0 → gitmole-0.10.0}/tests/test_trend.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.10.0
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -32,7 +32,6 @@ Free. Any Stack. Local. Offline. Deterministic. Fast.
|
|
|
32
32
|
- **Any stack.** It reads what every repository has: the git log, git blame and the files themselves.
|
|
33
33
|
- **Local & Offline.** Everything runs against a clone on your machine. Nothing is uploaded, nothing phones home; the vulnerability database is a copy you download once.
|
|
34
34
|
- **Deterministic.** No AI at runtime. Every finding is a plain rule over counts you can recompute by hand. The JSON export carries each finding's rule and the numbers it fired on. The same clone gives the same report every time.
|
|
35
|
-
- **Fast.** A 4,400-commit repository takes under thirty seconds.
|
|
36
35
|
|
|
37
36
|
## Install
|
|
38
37
|
|
|
@@ -71,59 +70,15 @@ blocks on secrets in source files and still posts the report. Every option:
|
|
|
71
70
|
|
|
72
71
|
## What you get
|
|
73
72
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
│ most commits on Wed at 15:00 · 14% of commits are fixes · 1% of commits are reverts · 19% │
|
|
82
|
-
│ of surviving code from 2026 │
|
|
83
|
-
│ 1 critical, 6 warnings, 9 notes │
|
|
84
|
-
╰──────────────────────────────────────────────────────────────────────────────────────────────────╯
|
|
85
|
-
|
|
86
|
-
◎ Watch list
|
|
87
|
-
file why
|
|
88
|
-
──────────────────────────────────────────────────────────────────────────────────────────────────
|
|
89
|
-
packages/react-server/src/ReactFlightServer.js changed 319 times · fixed once in six months ·
|
|
90
|
-
visitAsyncNodeImpl() complexity 46
|
|
91
|
-
packages/react-server/src/ReactFizzServer.js changed 297 times · fixed twice in six months ·
|
|
92
|
-
retryNode() complexity 41
|
|
93
|
-
packages/react-reconciler/src/ReactFiberWorkLoo changed 312 times · fixed 4 times in six months
|
|
94
|
-
p.js · flushSpawnedWork() complexity 48
|
|
95
|
-
packages/react-reconciler/src/ReactFiberCommitW changed 284 times · fixed 28 times ·
|
|
96
|
-
ork.js commitLayoutEffectOnFiber() complexity 72
|
|
97
|
-
packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
|
|
98
|
-
rk.js beginWork() complexity 52
|
|
99
|
-
ranked by revisions × lines of code; the reasons say what else counts against each file
|
|
100
|
-
6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
|
|
101
|
-
files that had changed more than once would name 0.3; the 15 most changed would name 7)
|
|
102
|
-
```
|
|
73
|
+
Reports on repositories you know, each at a pinned commit, published as gitmole wrote them:
|
|
74
|
+
|
|
75
|
+
| Repository | Commit | Commits | Lines | gitmole run |
|
|
76
|
+
|---|---|---:|---:|---:|
|
|
77
|
+
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 61 s |
|
|
78
|
+
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 153 s |
|
|
79
|
+
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 138 s |
|
|
103
80
|
|
|
104
|
-
|
|
105
|
-
next, the reasons in words, and a backtest that says how the same list, drawn
|
|
106
|
-
six months earlier, would have done against the fixes that followed. The list
|
|
107
|
-
ranks by revisions × lines of code: measured at six cut-offs on three
|
|
108
|
-
repositories
|
|
109
|
-
([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
|
|
110
|
-
that named more of the files fixed next than churn alone, size alone or a
|
|
111
|
-
weighted product of fixes, complexity and ownership. Between the header and that
|
|
112
|
-
list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
|
|
113
|
-
notes); below it, tables for people, the knowledge map, the timeline, change
|
|
114
|
-
coupling, complex functions and repo health; `--full` adds the hotspots table
|
|
115
|
-
behind the list, size, activity and code age. Every section is explained in
|
|
116
|
-
[docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
|
|
117
|
-
|
|
118
|
-
Reports on repositories you know, each at a pinned commit with a fixed
|
|
119
|
-
reference date, published as gitmole wrote them; the repo-health numbers
|
|
120
|
-
come from git-sizer over the whole clone, so a fresh clone can differ there:
|
|
121
|
-
|
|
122
|
-
| Repository | Commits | Lines | Watch list backtest |
|
|
123
|
-
|---|---:|---:|---|
|
|
124
|
-
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | 39,758 | 247,179 | named 15 of the 238 files fixed in the next six months; a random pick would name 4.9, the 15 most changed 15 |
|
|
125
|
-
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | 34,933 | 431,749 | named 15 of the 213 files fixed in the next six months; a random pick would name 3.1, the 15 most changed 13 |
|
|
126
|
-
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | 21,703 | 681,078 | named 11 of the 46 files fixed in the next six months; a random pick would name 0.3, the 15 most changed 7 |
|
|
81
|
+
Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
|
|
127
82
|
|
|
128
83
|
## The tool set
|
|
129
84
|
|
|
@@ -12,7 +12,6 @@ Free. Any Stack. Local. Offline. Deterministic. Fast.
|
|
|
12
12
|
- **Any stack.** It reads what every repository has: the git log, git blame and the files themselves.
|
|
13
13
|
- **Local & Offline.** Everything runs against a clone on your machine. Nothing is uploaded, nothing phones home; the vulnerability database is a copy you download once.
|
|
14
14
|
- **Deterministic.** No AI at runtime. Every finding is a plain rule over counts you can recompute by hand. The JSON export carries each finding's rule and the numbers it fired on. The same clone gives the same report every time.
|
|
15
|
-
- **Fast.** A 4,400-commit repository takes under thirty seconds.
|
|
16
15
|
|
|
17
16
|
## Install
|
|
18
17
|
|
|
@@ -51,59 +50,15 @@ blocks on secrets in source files and still posts the report. Every option:
|
|
|
51
50
|
|
|
52
51
|
## What you get
|
|
53
52
|
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
│ most commits on Wed at 15:00 · 14% of commits are fixes · 1% of commits are reverts · 19% │
|
|
62
|
-
│ of surviving code from 2026 │
|
|
63
|
-
│ 1 critical, 6 warnings, 9 notes │
|
|
64
|
-
╰──────────────────────────────────────────────────────────────────────────────────────────────────╯
|
|
65
|
-
|
|
66
|
-
◎ Watch list
|
|
67
|
-
file why
|
|
68
|
-
──────────────────────────────────────────────────────────────────────────────────────────────────
|
|
69
|
-
packages/react-server/src/ReactFlightServer.js changed 319 times · fixed once in six months ·
|
|
70
|
-
visitAsyncNodeImpl() complexity 46
|
|
71
|
-
packages/react-server/src/ReactFizzServer.js changed 297 times · fixed twice in six months ·
|
|
72
|
-
retryNode() complexity 41
|
|
73
|
-
packages/react-reconciler/src/ReactFiberWorkLoo changed 312 times · fixed 4 times in six months
|
|
74
|
-
p.js · flushSpawnedWork() complexity 48
|
|
75
|
-
packages/react-reconciler/src/ReactFiberCommitW changed 284 times · fixed 28 times ·
|
|
76
|
-
ork.js commitLayoutEffectOnFiber() complexity 72
|
|
77
|
-
packages/react-reconciler/src/ReactFiberBeginWo changed 361 times · fixed once in six months ·
|
|
78
|
-
rk.js beginWork() complexity 52
|
|
79
|
-
ranked by revisions × lines of code; the reasons say what else counts against each file
|
|
80
|
-
6 months ago this list would have named 11 of the 46 files fixed since (a random 15 of the 1802
|
|
81
|
-
files that had changed more than once would name 0.3; the 15 most changed would name 7)
|
|
82
|
-
```
|
|
53
|
+
Reports on repositories you know, each at a pinned commit, published as gitmole wrote them:
|
|
54
|
+
|
|
55
|
+
| Repository | Commit | Commits | Lines | gitmole run |
|
|
56
|
+
|---|---|---:|---:|---:|
|
|
57
|
+
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 61 s |
|
|
58
|
+
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 153 s |
|
|
59
|
+
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 138 s |
|
|
83
60
|
|
|
84
|
-
|
|
85
|
-
next, the reasons in words, and a backtest that says how the same list, drawn
|
|
86
|
-
six months earlier, would have done against the fixes that followed. The list
|
|
87
|
-
ranks by revisions × lines of code: measured at six cut-offs on three
|
|
88
|
-
repositories
|
|
89
|
-
([validation](https://github.com/antvinni/gitmole/blob/main/docs/validation.md)),
|
|
90
|
-
that named more of the files fixed next than churn alone, size alone or a
|
|
91
|
-
weighted product of fixes, complexity and ownership. Between the header and that
|
|
92
|
-
list the full report puts its findings, 16 for react (1 critical, 6 warnings, 9
|
|
93
|
-
notes); below it, tables for people, the knowledge map, the timeline, change
|
|
94
|
-
coupling, complex functions and repo health; `--full` adds the hotspots table
|
|
95
|
-
behind the list, size, activity and code age. Every section is explained in
|
|
96
|
-
[docs/output.md](https://github.com/antvinni/gitmole/blob/main/docs/output.md).
|
|
97
|
-
|
|
98
|
-
Reports on repositories you know, each at a pinned commit with a fixed
|
|
99
|
-
reference date, published as gitmole wrote them; the repo-health numbers
|
|
100
|
-
come from git-sizer over the whole clone, so a fresh clone can differ there:
|
|
101
|
-
|
|
102
|
-
| Repository | Commits | Lines | Watch list backtest |
|
|
103
|
-
|---|---:|---:|---|
|
|
104
|
-
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | 39,758 | 247,179 | named 15 of the 238 files fixed in the next six months; a random pick would name 4.9, the 15 most changed 15 |
|
|
105
|
-
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | 34,933 | 431,749 | named 15 of the 213 files fixed in the next six months; a random pick would name 3.1, the 15 most changed 13 |
|
|
106
|
-
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | 21,703 | 681,078 | named 11 of the 46 files fixed in the next six months; a random pick would name 0.3, the 15 most changed 7 |
|
|
61
|
+
Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
|
|
107
62
|
|
|
108
63
|
## The tool set
|
|
109
64
|
|
|
@@ -22,9 +22,11 @@ import tempfile
|
|
|
22
22
|
from . import filetypes, load, maat, trend
|
|
23
23
|
|
|
24
24
|
|
|
25
|
-
def
|
|
26
|
-
"""scc's --by-file JSON
|
|
27
|
-
|
|
25
|
+
def snapshot_at(repo: str, rev: str, out_dir: str) -> tuple:
|
|
26
|
+
"""The tree at rev, measured and classified as it was then: scc's --by-file JSON, the generated files
|
|
27
|
+
and the vendored paths. Exported through a temporary index so no archive is held in memory and
|
|
28
|
+
export-ignore attributes do not thin the tree; the same index lets git check-attr read that tree's
|
|
29
|
+
.gitattributes, so the classification is the cut-off's, not HEAD's.
|
|
28
30
|
|
|
29
31
|
The tree is written under `out_dir`, not the system temp directory: a SIGKILL cannot run the
|
|
30
32
|
cleanup, and a checkout left next to the report is one the next run clears away."""
|
|
@@ -34,7 +36,15 @@ def size_at(repo: str, rev: str, out_dir: str) -> str:
|
|
|
34
36
|
env = dict(os.environ, GIT_INDEX_FILE=os.path.join(tmp, "index"))
|
|
35
37
|
subprocess.run(["git", "read-tree", rev], cwd=repo, env=env, check=True, capture_output=True, text=True)
|
|
36
38
|
subprocess.run(["git", "checkout-index", "-a", f"--prefix={tree}/"], cwd=repo, env=env, check=True, capture_output=True, text=True)
|
|
37
|
-
|
|
39
|
+
size = subprocess.run(["scc", "--by-file", "--format", "json"], cwd=tree, capture_output=True, text=True, check=True).stdout
|
|
40
|
+
# the text files of that tree, as blame.text_files lists HEAD's: git grep prints "rev:path"
|
|
41
|
+
proc = subprocess.run([*filetypes.GIT, "grep", "-I", "--name-only", "-z", "-e", "", rev], cwd=repo, capture_output=True)
|
|
42
|
+
if proc.returncode not in (0, 1): # 1 is grep's "no match" (an empty tree), not a failure
|
|
43
|
+
raise subprocess.CalledProcessError(proc.returncode, proc.args, proc.stdout.decode("utf-8", "replace"),
|
|
44
|
+
proc.stderr.decode("utf-8", "replace"))
|
|
45
|
+
paths = sorted(p.decode("utf-8", "surrogateescape").split(":", 1)[1] for p in proc.stdout.split(b"\0") if p)
|
|
46
|
+
attrs = filetypes.attributes(repo, paths, cached=True, env=env)
|
|
47
|
+
return size, filetypes.generated_files(tree, paths, attrs), filetypes.vendored_paths(tree, paths, attrs)
|
|
38
48
|
|
|
39
49
|
|
|
40
50
|
def main(argv=None) -> int:
|
|
@@ -66,7 +76,7 @@ def main(argv=None) -> int:
|
|
|
66
76
|
types = filetypes.parse(meta.get("file_types"))
|
|
67
77
|
maat.write_all(log_path, sub, os.path.join(args.out, "meta.json") if "aliases" in meta else None, types, now=until, until=until)
|
|
68
78
|
try:
|
|
69
|
-
size =
|
|
79
|
+
size, generated, vendored = snapshot_at(args.repo, rev, args.out)
|
|
70
80
|
except subprocess.CalledProcessError as e:
|
|
71
81
|
first = ((e.stderr or "").strip().splitlines() or [f"{' '.join(e.cmd)} exited {e.returncode}"])[0]
|
|
72
82
|
print(f"backtest: {first}", file=sys.stderr)
|
|
@@ -74,7 +84,8 @@ def main(argv=None) -> int:
|
|
|
74
84
|
with open(os.path.join(sub, "size.json"), "w", encoding="utf-8") as fh:
|
|
75
85
|
fh.write(size)
|
|
76
86
|
with open(os.path.join(sub, "meta.json"), "w", encoding="utf-8") as fh:
|
|
77
|
-
json.dump({"now": until, "last_date": until, "file_types": meta.get("file_types"), "aliases": meta.get("aliases", {})
|
|
87
|
+
json.dump({"now": until, "last_date": until, "file_types": meta.get("file_types"), "aliases": meta.get("aliases", {}),
|
|
88
|
+
"generated": generated, "vendored": vendored}, fh)
|
|
78
89
|
return 0
|
|
79
90
|
|
|
80
91
|
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Why a file is out of the scored pool: one answer for every table, the watch list and --risk.
|
|
2
|
+
|
|
3
|
+
Built once per report, so the type filter, scc's rows, the generated and vendored lists, the amalgamations
|
|
4
|
+
and the plumbing are read once and every section asks the same object. Not in filetypes, which blame.py
|
|
5
|
+
and maat.py import as scripts and which must stay free of package imports."""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from collections import Counter
|
|
9
|
+
|
|
10
|
+
from . import filetypes, hotspots
|
|
11
|
+
|
|
12
|
+
# In descriptive order: what a reader would call the file first. package.json is a release file before
|
|
13
|
+
# it is not a source type; vendor/x_test.go is vendored before it is a test file; a generated test is
|
|
14
|
+
# generated. The order is part of the contract: reason() is the first that applies.
|
|
15
|
+
REASONS = ("generated", "vendored", "test file", "example code", "release file", "amalgamation", "not a source type", "not in the tree")
|
|
16
|
+
|
|
17
|
+
# The coverage line's nouns, by count.
|
|
18
|
+
_NOUNS = {"scored": ("scored", "scored"), "generated": ("generated", "generated"), "vendored": ("vendored", "vendored"),
|
|
19
|
+
"test file": ("test file", "test files"), "example code": ("example code", "example code"),
|
|
20
|
+
"release file": ("release file", "release files"), "amalgamation": ("amalgamation", "amalgamations"),
|
|
21
|
+
"not a source type": ("not a source type", "not a source type"), "not counted by scc": ("not counted by scc", "not counted by scc")}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Classifier:
|
|
25
|
+
def __init__(self, report: dict):
|
|
26
|
+
meta = report.get("meta") or {}
|
|
27
|
+
# a run records its --file-types spec (None for the default list); a run from before that record
|
|
28
|
+
# was measured unfiltered and is classified unfiltered, as load.load_report filters scc
|
|
29
|
+
self.types = filetypes.parse(meta["file_types"]) if "file_types" in meta else None
|
|
30
|
+
self.tree = (report.get("size") or {}).get("files") or {}
|
|
31
|
+
self.generated = set(meta.get("generated") or [])
|
|
32
|
+
self.vendored = filetypes.vendor_dirs(report)
|
|
33
|
+
self.amalgamations = hotspots.amalgamations(report)
|
|
34
|
+
self.plumbing = filetypes.plumbing_paths(report)
|
|
35
|
+
self._reasons = {} # memoised per instance: every hide pass in a report asks the same paths again
|
|
36
|
+
|
|
37
|
+
def reasons(self, path: str) -> tuple:
|
|
38
|
+
"""Every reason that applies, in REASONS order. `not in the tree` needs a tree to judge by: a run
|
|
39
|
+
whose scc step was killed classifies nothing as gone, so it cannot empty every table. Cached per
|
|
40
|
+
path and returned as a tuple so callers cannot mutate the cached result."""
|
|
41
|
+
if path in self._reasons:
|
|
42
|
+
return self._reasons[path]
|
|
43
|
+
out = []
|
|
44
|
+
if path in self.generated:
|
|
45
|
+
out.append("generated")
|
|
46
|
+
if filetypes.is_vendored(path, self.vendored):
|
|
47
|
+
out.append("vendored")
|
|
48
|
+
if filetypes.is_test_path(path):
|
|
49
|
+
out.append("test file")
|
|
50
|
+
if filetypes.is_sample_path(path):
|
|
51
|
+
out.append("example code")
|
|
52
|
+
if filetypes.is_release(path, self.plumbing):
|
|
53
|
+
out.append("release file")
|
|
54
|
+
if path in self.amalgamations:
|
|
55
|
+
out.append("amalgamation")
|
|
56
|
+
if not filetypes.matches(path, self.types):
|
|
57
|
+
out.append("not a source type")
|
|
58
|
+
if self.tree and path not in self.tree:
|
|
59
|
+
out.append("not in the tree")
|
|
60
|
+
self._reasons[path] = tuple(out)
|
|
61
|
+
return self._reasons[path]
|
|
62
|
+
|
|
63
|
+
def reason(self, path: str):
|
|
64
|
+
"""The first reason, or None for a file in the scored pool."""
|
|
65
|
+
found = self.reasons(path)
|
|
66
|
+
return found[0] if found else None
|
|
67
|
+
|
|
68
|
+
def excluded(self, path: str, reasons) -> bool:
|
|
69
|
+
"""Whether a table that hides `reasons` hides this file: any of its reasons is enough, so a table
|
|
70
|
+
that hides tests still hides a vendored test."""
|
|
71
|
+
return any(r in reasons for r in self.reasons(path))
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def coverage(classifier: Classifier, tracked: list) -> dict:
|
|
75
|
+
"""How many tracked files each reason claims, `scored` for none. A tracked file with no scc row is a
|
|
76
|
+
type scc does not classify rather than a deleted one, so it counts as `not counted by scc`."""
|
|
77
|
+
counts = Counter()
|
|
78
|
+
for path in tracked:
|
|
79
|
+
reason = classifier.reason(path)
|
|
80
|
+
counts["scored" if reason is None else "not counted by scc" if reason == "not in the tree" else reason] += 1
|
|
81
|
+
return dict(counts)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def coverage_line(cov: dict) -> str:
|
|
85
|
+
"""'4,781 files: 3,900 scored · 610 test files · …', the buckets in REASONS order."""
|
|
86
|
+
order = ["scored", *REASONS[:-1], "not counted by scc"]
|
|
87
|
+
parts = [f"{cov[k]:,} {_NOUNS[k][0 if cov[k] == 1 else 1]}" for k in order if cov.get(k)]
|
|
88
|
+
return f"{sum(cov.values()):,} files: " + " · ".join(parts)
|
|
@@ -44,6 +44,7 @@ def parse_args(argv):
|
|
|
44
44
|
p.add_argument("--fail-on", choices=findings.SEVERITIES, help="exit 3 if any finding is at this severity or worse")
|
|
45
45
|
p.add_argument("--risk", metavar="BASE", help="score the files changed since BASE (merge base with HEAD) by their share of the repository's revisions × lines of code; needs a local path")
|
|
46
46
|
p.add_argument("--risk-threshold", type=float, metavar="N", help="with --risk: exit 3 when the changed files hold more than N percent of the repository's revisions × lines of code")
|
|
47
|
+
p.add_argument("--compare", metavar="BEFORE_JSON", help="add a 'Since last report' section against an earlier --json export of the same clone")
|
|
47
48
|
p.add_argument("--version", action="version", version=f"gitmole {__version__}")
|
|
48
49
|
return p.parse_args(argv)
|
|
49
50
|
|
|
@@ -127,11 +128,13 @@ def _check_args(args, err, kind=None) -> int | None:
|
|
|
127
128
|
if kind is None:
|
|
128
129
|
bad = ("--yes needs --clean" if args.yes and not args.clean else
|
|
129
130
|
"target required" if args.target is None and not args.clean else
|
|
130
|
-
"--risk-threshold needs --risk" if args.risk_threshold is not None and not args.risk else
|
|
131
|
+
"--risk-threshold needs --risk" if args.risk_threshold is not None and not args.risk else
|
|
132
|
+
"--compare: no such file: " + args.compare if args.compare and not os.path.isfile(args.compare) else None)
|
|
131
133
|
elif kind == "path":
|
|
132
134
|
bad = None
|
|
133
135
|
else:
|
|
134
|
-
bad = ("--
|
|
136
|
+
bad = ("--compare needs one repository, not owner/*" if args.compare and kind == "org" else
|
|
137
|
+
"--risk needs a local path" if args.risk else
|
|
135
138
|
"--list-file-types needs a local path" if args.list_file_types else None)
|
|
136
139
|
if bad:
|
|
137
140
|
err.print(f"[red]{bad}[/red]")
|
|
@@ -297,12 +300,14 @@ def _meta_for_run(repo_dir: str, args, estimate, age_ok: bool, plots_ok: bool, p
|
|
|
297
300
|
types_spec = _types_spec(args.file_types)
|
|
298
301
|
|
|
299
302
|
meta = run.collect_meta(repo_dir, since=args.since_date)
|
|
303
|
+
meta["run"] = run.manifest(repo_dir, args) # what produced this report: commit, gitmole and tool versions, the options
|
|
300
304
|
meta["file_types"] = types_spec # the loader filters scc's size data the way every other step was filtered
|
|
301
305
|
meta["gone_months"] = args.gone
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
meta["generated"] = filetypes.generated_files(repo_dir, tracked) # hidden from the tables, out of the findings
|
|
305
|
-
meta["vendored"] = filetypes.
|
|
306
|
+
tracked = blame.text_files(repo_dir) # every tracked text file: --ignore shapes blame, functions and duplicates, never what a file is
|
|
307
|
+
attrs = filetypes.attributes(repo_dir, tracked) # one git check-attr pass, shared by the two lists below
|
|
308
|
+
meta["generated"] = filetypes.generated_files(repo_dir, tracked, attrs=attrs) # hidden from the tables, out of the findings
|
|
309
|
+
meta["vendored"] = filetypes.vendored_paths(repo_dir, tracked, attrs=attrs) # somebody else's code, by the licence it carries or the attribute it declares
|
|
310
|
+
meta["credential_files"] = filetypes.credential_files(filetypes.git_paths(repo_dir, "ls-files")) # by name, over every tracked file
|
|
306
311
|
if args.since_date and meta["commits"] == 0:
|
|
307
312
|
raise NoCommits(f"no commits since {args.since_date}; widen --since")
|
|
308
313
|
if args.now:
|
|
@@ -348,6 +353,17 @@ def _record_statuses(meta, results, age_ok: bool, plots_ok: bool, lizard_ok: boo
|
|
|
348
353
|
meta["steps"] = {name: "run" if rc == 0 else (rc if isinstance(rc, str) else "failed") for name, rc in results.items()}
|
|
349
354
|
|
|
350
355
|
|
|
356
|
+
def _coverage(repo_dir: str, out_dir: str) -> dict:
|
|
357
|
+
"""How many tracked text files each reason claims, from the report as the steps left it. An
|
|
358
|
+
unreadable output directory (a killed run) records nothing rather than failing the run."""
|
|
359
|
+
from . import classify
|
|
360
|
+
try:
|
|
361
|
+
report = load.load_report(out_dir, nested=False)
|
|
362
|
+
except load.Unreadable:
|
|
363
|
+
return {}
|
|
364
|
+
return classify.coverage(classify.Classifier(report), blame.text_files(repo_dir))
|
|
365
|
+
|
|
366
|
+
|
|
351
367
|
def _analyse(repo_dir: str, out_dir: str, args, ui: Console, planner, estimator) -> None:
|
|
352
368
|
"""Run the whole pipeline for one repository into out_dir."""
|
|
353
369
|
os.makedirs(os.path.join(out_dir, "theseus"), exist_ok=True)
|
|
@@ -372,6 +388,7 @@ def _analyse(repo_dir: str, out_dir: str, args, ui: Console, planner, estimator)
|
|
|
372
388
|
raise Interrupted()
|
|
373
389
|
|
|
374
390
|
_record_statuses(meta, results, age_ok, plots_ok, lizard_ok, cut, duplicates_ok)
|
|
391
|
+
meta["coverage"] = _coverage(repo_dir, out_dir)
|
|
375
392
|
run.save_meta(meta, out_dir)
|
|
376
393
|
|
|
377
394
|
failed = [n for n, rc in results.items() if rc != 0]
|
|
@@ -505,12 +522,30 @@ def _render(out_dir: str, console: Console, ui: Console, args, err: Console) ->
|
|
|
505
522
|
return 2
|
|
506
523
|
from . import watch
|
|
507
524
|
risk = {"base": args.risk, **watch.change_risk(report, files)}
|
|
525
|
+
comparison = None
|
|
526
|
+
if args.compare:
|
|
527
|
+
from . import compare as _compare
|
|
528
|
+
try:
|
|
529
|
+
with open(args.compare, encoding="utf-8") as fh:
|
|
530
|
+
before = json.load(fh)
|
|
531
|
+
except (OSError, ValueError) as e:
|
|
532
|
+
err.print(f"[red]--compare {args.compare}:[/red] {e}", soft_wrap=True)
|
|
533
|
+
return 2
|
|
534
|
+
if not _compare.is_export(before):
|
|
535
|
+
err.print(f"[red]--compare {args.compare}:[/red] not a gitmole --json export (it needs meta, findings with rule ids, and watch; "
|
|
536
|
+
"exports from before 0.8.0 have no rule ids)", soft_wrap=True)
|
|
537
|
+
return 2
|
|
538
|
+
if before["meta"].get("name") != report["meta"].get("name"):
|
|
539
|
+
err.print(f"[red]--compare {args.compare}:[/red] it describes {before['meta'].get('name')}, this run describes {report['meta'].get('name')}; "
|
|
540
|
+
"the two exports must be of the same clone", soft_wrap=True)
|
|
541
|
+
return 2
|
|
542
|
+
comparison = _compare.compare(before, report, found)
|
|
508
543
|
if args.json:
|
|
509
|
-
_write(json.dumps(render.to_json(report, found, risk=risk), indent=2) + "\n", args.json, console)
|
|
544
|
+
_write(json.dumps(render.to_json(report, found, risk=risk, compare=comparison), indent=2) + "\n", args.json, console)
|
|
510
545
|
if args.markdown:
|
|
511
|
-
_write(render.markdown(report, found, full=args.full, risk=risk, base=args.risk), args.markdown, console)
|
|
546
|
+
_write(render.markdown(report, found, full=args.full, risk=risk, base=args.risk, compare=comparison), args.markdown, console)
|
|
512
547
|
if "-" not in (args.json, args.markdown):
|
|
513
|
-
render.report(report, found, console, full=args.full, risk=risk, base=args.risk)
|
|
548
|
+
render.report(report, found, console, full=args.full, risk=risk, base=args.risk, compare=comparison)
|
|
514
549
|
if args.fail_on and any(findings.SEVERITIES.index(f["severity"]) <= findings.SEVERITIES.index(args.fail_on) for f in found):
|
|
515
550
|
return 3
|
|
516
551
|
if risk is not None and args.risk_threshold is not None and risk["total"] > args.risk_threshold:
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""What changed since an earlier --json export: findings that are new, resolved or persisting, files that
|
|
2
|
+
entered or left the watch list, the tally before and after. Pure over two report dicts."""
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from . import findings, watch
|
|
6
|
+
|
|
7
|
+
# A rule emits one finding per report, except these two, which emit one per row; the evidence field
|
|
8
|
+
# that tells the rows apart joins the rule id in the key.
|
|
9
|
+
KEY_FIELDS = {"repo_health": "metric", "placeholder_identity": "email"}
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def key(finding: dict) -> tuple:
|
|
13
|
+
"""A finding's identity across two reports: the rule id alone, unless the rule is one of KEY_FIELDS,
|
|
14
|
+
where one report can hold several findings for the same rule and the evidence field is what tells
|
|
15
|
+
them apart (a metric name, an email)."""
|
|
16
|
+
rid = finding["rule"]["id"]
|
|
17
|
+
field = KEY_FIELDS.get(rid)
|
|
18
|
+
return (rid, (finding.get("evidence") or {}).get(field)) if field else (rid,)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def is_export(data) -> bool:
|
|
22
|
+
"""A gitmole --json export: a report with its findings and watch list. Every finding must carry a
|
|
23
|
+
rule id and a severity, the shape key() and _tally() read without a default — exports from before
|
|
24
|
+
0.8.0 have findings without a `rule`, and would otherwise pass this check and crash later on a bare
|
|
25
|
+
KeyError instead of being refused here."""
|
|
26
|
+
if not (isinstance(data, dict) and isinstance(data.get("meta"), dict) and "findings" in data and "watch" in data):
|
|
27
|
+
return False
|
|
28
|
+
found, watch_rows = data.get("findings"), data.get("watch")
|
|
29
|
+
return (isinstance(found, list) and isinstance(watch_rows, list) and all(isinstance(r, dict) for r in watch_rows) and
|
|
30
|
+
all(isinstance(f, dict) and isinstance(f.get("rule"), dict) and "id" in f["rule"] and "severity" in f for f in found))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _tally(found: list) -> dict:
|
|
34
|
+
counts = {s: 0 for s in findings.SEVERITIES}
|
|
35
|
+
for f in found:
|
|
36
|
+
counts[f["severity"]] += 1
|
|
37
|
+
return counts
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _ordered(found: list) -> list:
|
|
41
|
+
return sorted(found, key=lambda f: (findings.SEVERITIES.index(f["severity"]), f["title"], str(key(f))))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _options_differ(before_meta: dict, after_meta: dict) -> list:
|
|
45
|
+
"""The options that change what a run sees: --since and --file-types from meta's top level, --ignore and
|
|
46
|
+
--ignore-data from the manifest when both exports have one. --deep is recorded there too but only
|
|
47
|
+
decides whether code age, plots and duplicates ran, none of which reach the findings or the watch list."""
|
|
48
|
+
out = [name for name in ("since", "file_types") if before_meta.get(name) != after_meta.get(name)]
|
|
49
|
+
b, a = (before_meta.get("run") or {}).get("options"), (after_meta.get("run") or {}).get("options")
|
|
50
|
+
if b is not None and a is not None:
|
|
51
|
+
out += [name for name in ("ignore", "ignore_data") if b.get(name) != a.get(name)]
|
|
52
|
+
return out
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def compare(before: dict, report: dict, found: list, top: int = watch.WATCH_TOP) -> dict:
|
|
56
|
+
"""before: an earlier export; report and found: this run's loaded report and its findings."""
|
|
57
|
+
b = {key(f): f for f in before.get("findings") or []}
|
|
58
|
+
a = {key(f): f for f in found}
|
|
59
|
+
persisting = [{**a[k], "was": b[k]["severity"]} for k in a if k in b]
|
|
60
|
+
before_watch = [f for f in (r.get("file") for r in (before.get("watch") or [])[:top]) if f]
|
|
61
|
+
after_watch = [r["file"] for r in watch.risks(report)[:top]]
|
|
62
|
+
meta_b, meta_a = before.get("meta") or {}, report.get("meta") or {}
|
|
63
|
+
return {"new": _ordered([a[k] for k in a if k not in b]), "resolved": _ordered([b[k] for k in b if k not in a]),
|
|
64
|
+
"persisting": _ordered(persisting),
|
|
65
|
+
"watch_entered": [f for f in after_watch if f not in before_watch], "watch_left": [f for f in before_watch if f not in after_watch],
|
|
66
|
+
"tally": {"before": _tally(before.get("findings") or []), "after": _tally(found)},
|
|
67
|
+
"before": {"commit": (meta_b.get("run") or {}).get("commit"), "date": meta_b.get("last_date"),
|
|
68
|
+
"options_differ": _options_differ(meta_b, meta_a)}}
|
|
@@ -42,13 +42,14 @@ def fixed_between(commits: list, start: str, end: str) -> set:
|
|
|
42
42
|
for p, _, _ in c["files"] if not filetypes.is_test_path(p)}
|
|
43
43
|
|
|
44
44
|
|
|
45
|
-
def report_at(commits: list, t: str, size: dict, meta: dict) -> dict:
|
|
46
|
-
"""The report watch.risks reads, from the commits before `t
|
|
45
|
+
def report_at(commits: list, t: str, size: dict, meta: dict, generated: list, vendored: list) -> dict:
|
|
46
|
+
"""The report watch.risks reads, from the commits before `t`, scc's listing of the tree at `t`, and
|
|
47
|
+
that tree's own generated and vendored files (snapshot_at classified the cut-off, not HEAD).
|
|
47
48
|
No coupling and no functions: neither enters the score, and the pipeline's own backtest has no functions either."""
|
|
48
49
|
past = maat.in_window(commits, until=t)
|
|
49
50
|
bots = {b["name"] for b in meta.get("bots") or []}
|
|
50
51
|
ownership = [r for r in maat.entity_ownership(past) if r["author"] not in bots and not identity.is_bot(r["author"])]
|
|
51
|
-
return {"meta": {"now": t, "generated":
|
|
52
|
+
return {"meta": {"now": t, "generated": generated, "vendored": vendored}, "size": size, "revisions": maat.revisions(past),
|
|
52
53
|
"plumbing": maat.plumbing(past), "authors": maat.authors(past), "ownership": ownership,
|
|
53
54
|
"fixes": maat.fixes(past, now=t), "coupling": [], "functions": []}
|
|
54
55
|
|
|
@@ -163,8 +164,9 @@ def main(argv=None) -> int:
|
|
|
163
164
|
rev = trend.rev_before(args.repo, t, end_of_day=False)
|
|
164
165
|
if not rev:
|
|
165
166
|
continue # the history does not reach back this far
|
|
166
|
-
|
|
167
|
-
|
|
167
|
+
size_json, generated, vendored = backtest.snapshot_at(args.repo, rev, args.out)
|
|
168
|
+
size = load.parse_scc(size_json, types)
|
|
169
|
+
report = report_at(commits, t, size, meta, generated, vendored)
|
|
168
170
|
fixed = fixed_between(commits, t, months_after(t, args.horizon))
|
|
169
171
|
pool = set(variants(report)["churn"])
|
|
170
172
|
results.append((t, len(fixed & pool), len(pool), score(report, fixed, args.top)))
|