gitmole 0.20.0__tar.gz → 0.22.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gitmole-0.20.0 → gitmole-0.22.0}/PKG-INFO +5 -4
- {gitmole-0.20.0 → gitmole-0.22.0}/README.md +4 -3
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/__init__.py +1 -1
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/evaluate.py +39 -3
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/findings.py +73 -1
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/provenance.py +135 -4
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/render.py +24 -3
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/run.py +1 -1
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/structure.py +133 -4
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/PKG-INFO +5 -4
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/SOURCES.txt +1 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_evaluate.py +10 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_provenance.py +41 -1
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_render.py +15 -0
- gitmole-0.22.0/tests/test_shapes.py +88 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/LICENSE +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/__main__.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/backtest.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/banner.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/blame.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/classify.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/clean.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/cli.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/compare.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/coupling.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/deps.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/duplicates.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/filetypes.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/functions.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/hook.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/hotspots.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/hygiene.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/identity.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/imports.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/knowledge.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/leaks.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/licences.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/load.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/loss.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/maat.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/osps.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/sarif.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/sbom.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/signing.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/szz.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/textfmt.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/trend.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/watch.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/dependency_links.txt +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/entry_points.txt +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/requires.txt +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/top_level.txt +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/pyproject.toml +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/setup.cfg +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_backtest.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_banner.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_blame.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_classify.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_clean.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_cli.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_compare.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_coupling.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_declared.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_deps.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_duplicates.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_filetypes.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_findings.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_functions.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_golden.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_hook.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_hotspots.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_hygiene.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_identity.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_knowledge.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_leaks.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_load.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_loss.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_maat.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_packaging.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_render_examples.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_run.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_sarif.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_signing.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_structure.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_szz.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_textfmt.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_trend.py +0 -0
- {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_watch.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.22.0
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -74,6 +74,7 @@ gitmole . --json report.json # every table, the watch list and the fin
|
|
|
74
74
|
gitmole . --fail-on warning # exit 3 if any finding is a warning or worse
|
|
75
75
|
gitmole . --risk main --risk-threshold 10 # exit 3 if the files changed since main hold over 10% of the risk
|
|
76
76
|
gitmole . --sarif gitmole.sarif # the findings for GitHub code scanning or GitLab
|
|
77
|
+
gitmole . --sbom sbom.cdx.json # a CycloneDX SBOM of every package the lock files pin
|
|
77
78
|
gitmole . --compare last.json # what changed since an earlier --json export
|
|
78
79
|
gitmole analysis-repo --no-run --hook # a coding agent's edit hook: history's view of the files it just touched
|
|
79
80
|
gitmole . --since 2y --full # the current team, every row and column
|
|
@@ -92,9 +93,9 @@ Reports on repositories you know, each at a pinned commit, published as gitmole
|
|
|
92
93
|
|
|
93
94
|
| Repository | Commit | Commits | Lines | gitmole run |
|
|
94
95
|
|---|---|---:|---:|---:|
|
|
95
|
-
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 |
|
|
96
|
-
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 |
|
|
97
|
-
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 |
|
|
96
|
+
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 59 s |
|
|
97
|
+
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 135 s |
|
|
98
|
+
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 157 s |
|
|
98
99
|
|
|
99
100
|
Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
|
|
100
101
|
|
|
@@ -41,6 +41,7 @@ gitmole . --json report.json # every table, the watch list and the fin
|
|
|
41
41
|
gitmole . --fail-on warning # exit 3 if any finding is a warning or worse
|
|
42
42
|
gitmole . --risk main --risk-threshold 10 # exit 3 if the files changed since main hold over 10% of the risk
|
|
43
43
|
gitmole . --sarif gitmole.sarif # the findings for GitHub code scanning or GitLab
|
|
44
|
+
gitmole . --sbom sbom.cdx.json # a CycloneDX SBOM of every package the lock files pin
|
|
44
45
|
gitmole . --compare last.json # what changed since an earlier --json export
|
|
45
46
|
gitmole analysis-repo --no-run --hook # a coding agent's edit hook: history's view of the files it just touched
|
|
46
47
|
gitmole . --since 2y --full # the current team, every row and column
|
|
@@ -59,9 +60,9 @@ Reports on repositories you know, each at a pinned commit, published as gitmole
|
|
|
59
60
|
|
|
60
61
|
| Repository | Commit | Commits | Lines | gitmole run |
|
|
61
62
|
|---|---|---:|---:|---:|
|
|
62
|
-
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 |
|
|
63
|
-
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 |
|
|
64
|
-
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 |
|
|
63
|
+
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 59 s |
|
|
64
|
+
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 135 s |
|
|
65
|
+
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 157 s |
|
|
65
66
|
|
|
66
67
|
Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
|
|
67
68
|
|
|
@@ -62,8 +62,14 @@ def labelled_between(commits: list, labels: dict, start: str, end: str) -> set:
|
|
|
62
62
|
before `end` (ApacheJIT, Defectors: see szz.read_labels): the paths the label names, or every
|
|
63
63
|
source file the commit touched. Either side may be abbreviated."""
|
|
64
64
|
out = set()
|
|
65
|
+
short = [k for k in labels if len(k) < 40] # abbreviated labels need a prefix match; full hashes are a lookup
|
|
65
66
|
for c in maat.in_window(commits, start, end):
|
|
66
|
-
|
|
67
|
+
if c["hash"] in labels:
|
|
68
|
+
paths = labels[c["hash"]]
|
|
69
|
+
elif len(c["hash"]) < 40: # an abbreviated hash in the log: scan for the label it abbreviates
|
|
70
|
+
paths = next((v for k, v in labels.items() if k.startswith(c["hash"])), "none")
|
|
71
|
+
else:
|
|
72
|
+
paths = next((labels[k] for k in short if c["hash"].startswith(k)), "none")
|
|
67
73
|
if paths == "none":
|
|
68
74
|
continue
|
|
69
75
|
out.update(p for p in (paths if paths is not None else [p for p, _, _ in c["files"]]) if not filetypes.is_test_path(p))
|
|
@@ -151,6 +157,32 @@ def score(report: dict, fixed: set, top: int) -> dict:
|
|
|
151
157
|
return out
|
|
152
158
|
|
|
153
159
|
|
|
160
|
+
def effort(report: dict, outcome: set, top: int) -> dict:
|
|
161
|
+
"""variant -> (IFA, lines): how many of its first `top` files come before the first one in the
|
|
162
|
+
outcome (initial false alarms; `top` when none is), and the lines of code those files hold at the
|
|
163
|
+
cut-off, the inspection budget. A list that ranks small files first can look good on hits per line
|
|
164
|
+
and still send a reviewer through many files before one matters, so both are shown."""
|
|
165
|
+
files = (report.get("size") or {}).get("files") or {}
|
|
166
|
+
out = {}
|
|
167
|
+
for name, ranked in variants(report).items():
|
|
168
|
+
head = ranked[:top]
|
|
169
|
+
ifa = next((i for i, f in enumerate(head) if f in outcome), len(head))
|
|
170
|
+
out[name] = (ifa, sum((files.get(f) or {}).get("code", 0) for f in head))
|
|
171
|
+
return out
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def effort_table(efforts: list) -> str:
|
|
175
|
+
"""efforts: [{variant: (ifa, lines)}] per cut-off -> Markdown with the median of each over the cut-offs."""
|
|
176
|
+
import statistics
|
|
177
|
+
names = list(efforts[0]) if efforts else []
|
|
178
|
+
rows = ["| variant | IFA, median | lines of code in the list, median |", "|---|---:|---:|"]
|
|
179
|
+
for name in names:
|
|
180
|
+
ifas = [e[name][0] for e in efforts]
|
|
181
|
+
lines = [e[name][1] for e in efforts]
|
|
182
|
+
rows.append(f"| {name} | {statistics.median(ifas):g} | {int(statistics.median(lines)):,} |")
|
|
183
|
+
return "\n".join(rows)
|
|
184
|
+
|
|
185
|
+
|
|
154
186
|
def table(results: list, noun: str = "fixed") -> str:
|
|
155
187
|
"""results: [(t, files fixed that were in the pool, pool size, {variant: hits})], oldest first -> Markdown.
|
|
156
188
|
`noun` says what the outcome is: fixed, bug-inducing (R-SZZ) or labelled."""
|
|
@@ -184,6 +216,7 @@ def main(argv=None) -> int:
|
|
|
184
216
|
p.add_argument("--top", type=int, default=watch.WATCH_TOP)
|
|
185
217
|
p.add_argument("--szz", action="store_true", help="also score against R-SZZ bug-inducing commits (one git blame per fix and file: minutes)")
|
|
186
218
|
p.add_argument("--labels", metavar="CSV", help="also score against independent bug-inducing labels (ApacheJIT's CSV, Defectors' file rows, or one hash per line)")
|
|
219
|
+
p.add_argument("--end", metavar="YYYY-MM-DD", help="count the cut-offs back from this date rather than the last commit: a labelled dataset that stops earlier (ApacheJIT ends in 2019)")
|
|
187
220
|
args = p.parse_args(argv)
|
|
188
221
|
meta = load._read_json(args.out, "meta.json", {})
|
|
189
222
|
log_path = os.path.join(args.out, "log.txt")
|
|
@@ -201,8 +234,8 @@ def main(argv=None) -> int:
|
|
|
201
234
|
if args.labels and not labels:
|
|
202
235
|
print(f"evaluate: no bug-inducing commits read from {args.labels}", file=sys.stderr)
|
|
203
236
|
return 2
|
|
204
|
-
results, induced_results, labelled_results = [], [], []
|
|
205
|
-
for t in cutoffs(meta["last_date"], args.windows, args.horizon):
|
|
237
|
+
results, induced_results, labelled_results, labelled_effort = [], [], [], []
|
|
238
|
+
for t in cutoffs(args.end or meta["last_date"], args.windows, args.horizon):
|
|
206
239
|
rev = trend.rev_before(args.repo, t, end_of_day=False)
|
|
207
240
|
if not rev:
|
|
208
241
|
continue # the history does not reach back this far
|
|
@@ -220,6 +253,7 @@ def main(argv=None) -> int:
|
|
|
220
253
|
if labels:
|
|
221
254
|
marked = labelled_between(commits, labels, t, end)
|
|
222
255
|
labelled_results.append((t, len(marked & pool), len(pool), score(report, marked, args.top)))
|
|
256
|
+
labelled_effort.append(effort(report, marked, args.top))
|
|
223
257
|
print(f"evaluate: {t} done", file=sys.stderr)
|
|
224
258
|
if not results:
|
|
225
259
|
print("evaluate: no cut-off falls inside the history", file=sys.stderr)
|
|
@@ -233,6 +267,8 @@ def main(argv=None) -> int:
|
|
|
233
267
|
if labelled_results:
|
|
234
268
|
print(f"\nAgainst the files the bug-inducing commits labelled in {os.path.basename(args.labels)} touched inside each window:\n")
|
|
235
269
|
print(table(labelled_results, noun="labelled"))
|
|
270
|
+
print(f"\nWhat each list costs a reviewer against the same labels: initial false alarms before the first labelled file, and the lines of code in its top {args.top}:\n")
|
|
271
|
+
print(effort_table(labelled_effort))
|
|
236
272
|
print(f"\n`--all` exports {spread['all']:,} commits ({spread['fix_all']:,} fixes); HEAD reaches {spread['head']:,} ({spread['fix_head']:,} fixes).")
|
|
237
273
|
return 0
|
|
238
274
|
|
|
@@ -1009,6 +1009,78 @@ def debt_in_hotspots(report: dict, min_markers: int = 3, min_files: int = 2, top
|
|
|
1009
1009
|
evidence={"files": [{"file": f, "markers": n} for f, n in flagged[:10]]})]
|
|
1010
1010
|
|
|
1011
1011
|
|
|
1012
|
+
def _shape_files(report: dict, key: str) -> list:
|
|
1013
|
+
"""[(file, shapes)] for this repository's own source files holding a shape: tests, examples,
|
|
1014
|
+
documentation, vendored and generated files left out, since a fixture or a sample may do on purpose
|
|
1015
|
+
what the source should not."""
|
|
1016
|
+
s = _structure(report)
|
|
1017
|
+
if not s:
|
|
1018
|
+
return []
|
|
1019
|
+
generated, vendored = _generated(report), filetypes.vendor_dirs(report)
|
|
1020
|
+
return [(p, v["shapes"]) for p, v in sorted((s.get("files") or {}).items()) if (v.get("shapes") or {}).get(key)
|
|
1021
|
+
and not (filetypes.is_test_path(p) or filetypes.is_sample_path(p) or filetypes.is_doc_path(p)
|
|
1022
|
+
or filetypes.is_vendored(p, vendored) or p in generated)]
|
|
1023
|
+
|
|
1024
|
+
|
|
1025
|
+
def swallowed_errors(report: dict, min_count: int = 5, top_n: int = 10) -> list:
|
|
1026
|
+
"""Catch, except and rescue blocks that do nothing and say nothing: no statement, no comment. In
|
|
1027
|
+
Python only a bare `except:` or one catching Exception or BaseException counts, since `except
|
|
1028
|
+
KeyError: pass` is the language's idiom. A warning when one sits in a top hotspot, where an error
|
|
1029
|
+
that vanishes is the hardest to trace."""
|
|
1030
|
+
rows = _shape_files(report, "empty_catch")
|
|
1031
|
+
total = sum(sh.get("empty_catch_count", len(sh["empty_catch"])) for _, sh in rows)
|
|
1032
|
+
if total < min_count:
|
|
1033
|
+
return []
|
|
1034
|
+
rows.sort(key=lambda r: (-r[1].get("empty_catch_count", 0), r[0]))
|
|
1035
|
+
top = set(_scored_top(report, top_n))
|
|
1036
|
+
hot = [p for p, _ in rows if p in top]
|
|
1037
|
+
listed = _files_list([f"{p}:{sh['empty_catch'][0]}" + (f" and {sh['empty_catch_count'] - 1} more there" if sh.get("empty_catch_count", 1) > 1 else "") for p, sh in rows])
|
|
1038
|
+
bare = sum(sh.get("bare_except_count", 0) for _, sh in rows)
|
|
1039
|
+
first = hot[0] if hot else rows[0][0]
|
|
1040
|
+
return [_f("warning" if hot else "info", "Errors caught and dropped",
|
|
1041
|
+
f"{_plural(total, 'empty catch block')} in {_plural(len(rows), 'source file')}"
|
|
1042
|
+
+ (f", {bare} of them a bare except" if bare else "") + f": {listed}."
|
|
1043
|
+
+ (f" {textfmt.join_and(hot[:3])} {'is a top hotspot' if len(hot) == 1 else 'are top hotspots'}." if hot else ""),
|
|
1044
|
+
f"Log or rethrow in {first} first, or say in a comment why the error is ignored; an error dropped without a trace is the hardest kind to find.",
|
|
1045
|
+
rule={"id": "swallowed_errors", "min_count": min_count, "measure": "tree-sitter", "python": "bare, Exception or BaseException only"},
|
|
1046
|
+
evidence={"count": total, "bare_except": bare, "hotspots": hot[:10],
|
|
1047
|
+
"files": [{"file": p, "start": sh["empty_catch"][0], "count": sh.get("empty_catch_count", 1)} for p, sh in rows[:10]]})]
|
|
1048
|
+
|
|
1049
|
+
|
|
1050
|
+
def hardcoded_addresses(report: dict) -> list:
|
|
1051
|
+
"""IPv4 addresses written into string literals in source files: a host that moves, or an environment
|
|
1052
|
+
wired into the code. Loopback, unspecified, broadcast, netmask-shaped, documentation-range (RFC 5737)
|
|
1053
|
+
and object-identifier-shaped values are not counted (structure.py)."""
|
|
1054
|
+
rows = _shape_files(report, "addresses")
|
|
1055
|
+
if not rows:
|
|
1056
|
+
return []
|
|
1057
|
+
total = sum(sh.get("addresses_count", len(sh["addresses"])) for _, sh in rows)
|
|
1058
|
+
places = [(p, a) for p, sh in rows for a in sh["addresses"]]
|
|
1059
|
+
listed = _files_list([f"{a['value']} at {p}:{a['line']}" for p, a in places])
|
|
1060
|
+
return [_f("info", "Addresses written into the code",
|
|
1061
|
+
f"{_plural(total, 'IPv4 address')} in string literals in {_plural(len(rows), 'source file')}: {listed}.",
|
|
1062
|
+
f"Move {places[0][1]['value']} in {places[0][0]} into configuration, or a name that DNS resolves; an address in code has to be edited and shipped to change.",
|
|
1063
|
+
rule={"id": "hardcoded_addresses", "measure": "tree-sitter", "left_out": "loopback, 0.0.0.0, broadcast, RFC 5737, x.x.x.0, first octet 0-2"},
|
|
1064
|
+
evidence={"count": total, "files": [{"file": p, "start": a["line"], "value": a["value"]} for p, a in places[:10]]})]
|
|
1065
|
+
|
|
1066
|
+
|
|
1067
|
+
def commented_out_code(report: dict, min_lines: int = 10) -> list:
|
|
1068
|
+
"""Source files with ten or more lines of code left in comments: blocks of line or block comments in
|
|
1069
|
+
which four lines in five read as statements and one starts right at the comment marker. Worked
|
|
1070
|
+
examples in prose and documentation comments are not counted. The history already keeps old code."""
|
|
1071
|
+
rows = [(p, sh) for p, sh in _shape_files(report, "commented_code") if sh["commented_code"] >= min_lines]
|
|
1072
|
+
if not rows:
|
|
1073
|
+
return []
|
|
1074
|
+
rows.sort(key=lambda r: (-r[1]["commented_code"], r[0]))
|
|
1075
|
+
listed = _files_list([f"{p} ({sh['commented_code']} lines from line {sh['commented_sample'][0]})" for p, sh in rows])
|
|
1076
|
+
first = rows[0]
|
|
1077
|
+
return [_f("info", "Code left in comments",
|
|
1078
|
+
f"{_plural(len(rows), 'source file')} {'holds' if len(rows) == 1 else 'hold'} {min_lines} or more lines of commented-out code: {listed}.",
|
|
1079
|
+
f"Delete the block at {first[0]}:{first[1]['commented_sample'][0]}; git keeps the old version, and a reader cannot tell whether it is meant to come back.",
|
|
1080
|
+
rule={"id": "commented_out_code", "min_lines": min_lines, "measure": "tree-sitter", "code_share": 0.8},
|
|
1081
|
+
evidence={"files": [{"file": p, "start": sh["commented_sample"][0], "lines": sh["commented_code"]} for p, sh in rows[:10]]})]
|
|
1082
|
+
|
|
1083
|
+
|
|
1012
1084
|
def deep_nesting(report: dict, min_nesting: int = 5, min_bumps: int = 3, top_n: int = 10) -> list:
|
|
1013
1085
|
"""Functions nested five levels or more, or with three or more separate chunks of nested logic (a
|
|
1014
1086
|
bumpy road), in this repository's own source: CodeScene's nesting and bumpy-road factors, measured
|
|
@@ -1270,7 +1342,7 @@ RULES = [dormant, secrets_found, credential_files, vulnerable_dependencies, plac
|
|
|
1270
1342
|
minor_contributors, reverts, brain_methods, complexity_growth, tight_coupling, duplication, stale_files, knowledge_islands, knowledge_loss,
|
|
1271
1343
|
sweeping_commits, tangled_commits, hygiene_findings, debt_in_hotspots, deep_nesting, hidden_coupling, unreferenced_files,
|
|
1272
1344
|
agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author,
|
|
1273
|
-
truck_factor, authors_gone, component_coupling]
|
|
1345
|
+
truck_factor, authors_gone, component_coupling, swallowed_errors, hardcoded_addresses, commented_out_code]
|
|
1274
1346
|
|
|
1275
1347
|
|
|
1276
1348
|
def evaluate(report: dict) -> list:
|
|
@@ -12,6 +12,12 @@ writes provenance.json:
|
|
|
12
12
|
changed again by another commit within two weeks. This repository against itself, with the share
|
|
13
13
|
of commits the cohort covers beside it; no prior from elsewhere, since the best-controlled study
|
|
14
14
|
found the spread between agents larger than the pooled difference.
|
|
15
|
+
- lines: the lines added to code files in the last year and the year before, with the share git's
|
|
16
|
+
own moved-code detection marks as moved (`--color-moved`, blocks of twenty or more characters) and
|
|
17
|
+
the share deleted again within two weeks, in the same file with the same text: GitClear's moved and
|
|
18
|
+
churned lines, as a direction over this repository rather than a comparison with anyone else's.
|
|
19
|
+
The same two numbers per cohort, and each cohort's watch-list hit rate: the share of its commits
|
|
20
|
+
touching a file on the watch list's top fifteen.
|
|
15
21
|
- shape: neutral descriptors (commits landing in bursts, conventional-commit subjects, how many hours
|
|
16
22
|
of the day commits come in). Every one has a benign cause, and none is labelled.
|
|
17
23
|
- agents: the agent instruction files by path convention (AGENTS.md, CLAUDE.md, GEMINI.md,
|
|
@@ -28,7 +34,7 @@ import os
|
|
|
28
34
|
import re
|
|
29
35
|
import subprocess
|
|
30
36
|
import sys
|
|
31
|
-
from collections import Counter
|
|
37
|
+
from collections import Counter, deque
|
|
32
38
|
|
|
33
39
|
try:
|
|
34
40
|
from . import filetypes, leaks
|
|
@@ -44,6 +50,16 @@ _IDENT = re.compile(r"^\s*(?P<name>[^<]*?)\s*<(?P<email>[^>]*)>\s*$")
|
|
|
44
50
|
_CONVENTIONAL = re.compile(r"^(feat|fix|docs|style|refactor|perf|test|build|ci|chore|revert)(\([^)]*\))?!?: \S")
|
|
45
51
|
BURST_SIZE, BURST_SECONDS = 5, 600
|
|
46
52
|
RETOUCH_DAYS = 14
|
|
53
|
+
YEAR = 365 * 86400
|
|
54
|
+
WATCH_TOP = 15
|
|
55
|
+
# data that inflates line counts without being code (the same set the code-age pass leaves out)
|
|
56
|
+
DATA_EXCLUDES = [":(exclude,glob)**/*.json", ":(exclude,glob)**/*.lock", ":(exclude,glob)**/*.min.js", ":(exclude,glob)**/*.min.css",
|
|
57
|
+
":(exclude,glob)**/*.svg", ":(exclude,glob)**/*.map", ":(exclude,glob)**/*.csv", ":(exclude,glob)**/*.snap"]
|
|
58
|
+
_COLORS = ["-c", "color.diff.new=green", "-c", "color.diff.newMoved=cyan", "-c", "color.diff.old=red", "-c", "color.diff.oldMoved=magenta",
|
|
59
|
+
"-c", "color.diff.meta=normal", "-c", "color.diff.frag=normal", "-c", "color.diff.func=normal", "-c", "color.diff.context=normal",
|
|
60
|
+
"-c", "color.diff.whitespace=normal", "-c", "color.diff.commit=normal"]
|
|
61
|
+
_ADDED, _MOVED, _DELETED, _DELETED_MOVED = "\x1b[32m+", "\x1b[36m+", "\x1b[31m-", "\x1b[35m-"
|
|
62
|
+
_ANSI = re.compile(r"\x1b\[[0-9;]*m")
|
|
47
63
|
|
|
48
64
|
|
|
49
65
|
def read_commits(repo: str) -> list:
|
|
@@ -94,8 +110,8 @@ def trailers(commits: list) -> dict:
|
|
|
94
110
|
"with_any": sum(1 for c in commits if c["trailers"]), "never_author": listing[:50], "signoff_by_co_author": signoff[:50]}
|
|
95
111
|
|
|
96
112
|
|
|
97
|
-
def
|
|
98
|
-
"""The
|
|
113
|
+
def marker(inventory: dict):
|
|
114
|
+
"""The predicate that marks a commit: an `Assisted-by` trailer, or a co-author who never authors."""
|
|
99
115
|
marked_emails = {x["email"] for x in inventory["never_author"]}
|
|
100
116
|
|
|
101
117
|
def marked(c):
|
|
@@ -106,6 +122,13 @@ def cohort(commits: list, inventory: dict) -> dict:
|
|
|
106
122
|
if k.lower() == "co-authored-by" and ident and ident[1] in marked_emails:
|
|
107
123
|
return True
|
|
108
124
|
return False
|
|
125
|
+
return marked
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def cohort(commits: list, inventory: dict, watch_files=None) -> dict:
|
|
129
|
+
"""The commits an `Assisted-by` trailer or a never-authoring co-author marks, against the rest; with
|
|
130
|
+
`watch_files`, how many of each touched a file on the watch list."""
|
|
131
|
+
marked = marker(inventory)
|
|
109
132
|
|
|
110
133
|
reverted = {c["subject"][len('Revert "'):-1] for c in commits if c["subject"].startswith('Revert "') and c["subject"].endswith('"')}
|
|
111
134
|
by_file = {}
|
|
@@ -125,12 +148,107 @@ def cohort(commits: list, inventory: dict) -> dict:
|
|
|
125
148
|
again = True
|
|
126
149
|
break
|
|
127
150
|
s["retouched"] += again
|
|
151
|
+
if watch_files:
|
|
152
|
+
s["watch"] += any(f in watch_files for f in c["files"])
|
|
128
153
|
total = len(commits)
|
|
129
154
|
return {"definition": "an Assisted-by trailer, or a co-author who never authors a commit here",
|
|
130
155
|
"share": round(stats[True]["commits"] / total, 3) if total else 0.0,
|
|
131
156
|
"cohort": dict(stats[True]) or {"commits": 0}, "rest": dict(stats[False]) or {"commits": 0}}
|
|
132
157
|
|
|
133
158
|
|
|
159
|
+
def _code_path(path: str, generated: set, vendored) -> bool:
|
|
160
|
+
return filetypes.matches(path, filetypes.DEFAULT) and path not in generated and not filetypes.is_vendored(path, vendored)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def lines(repo: str, end: int, marked_hashes: set, generated=frozenset(), vendored=()) -> dict:
|
|
164
|
+
"""Added, moved and churned lines in code files over the two years before `end` (a timestamp), from
|
|
165
|
+
one `git log -p` with git's moved-code colouring. A line is churned when a later commit, within two
|
|
166
|
+
weeks, deletes a line with the same text from the same file; blank lines and lines without three
|
|
167
|
+
letters or digits (a lone brace) are not matched, since any brace would pair with any other."""
|
|
168
|
+
start = end - 2 * YEAR
|
|
169
|
+
argv = ["git", *_COLORS, "-c", "core.quotePath=false", "log", "HEAD", "--reverse", "--no-merges", "-p", "-U0", "-M", "--color=always",
|
|
170
|
+
"--color-moved=blocks", "--color-moved-ws=allow-indentation-change", f"--since=@{start}", f"--until=@{end}",
|
|
171
|
+
f"--format={END}%H{SEP}%at", "--", ".", *DATA_EXCLUDES]
|
|
172
|
+
proc = subprocess.Popen(argv, cwd=repo, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL)
|
|
173
|
+
windows = {w: Counter() for w in ("last", "before")}
|
|
174
|
+
groups = {True: Counter(), False: Counter()}
|
|
175
|
+
pending = {} # (path, text) -> [(time, window, marked)] of additions not yet churned
|
|
176
|
+
order = deque() # (time, key) in the order added, so what is past two weeks can be dropped and memory stays bounded
|
|
177
|
+
t, window, is_marked, path, keep = 0, "last", False, None, False
|
|
178
|
+
for raw in proc.stdout:
|
|
179
|
+
line = raw.decode("utf-8", "replace").rstrip("\n")
|
|
180
|
+
if line.startswith(END):
|
|
181
|
+
h, _, at = line[1:].partition(SEP)
|
|
182
|
+
t = int(at or 0)
|
|
183
|
+
window = "last" if t > end - YEAR else "before"
|
|
184
|
+
is_marked = h in marked_hashes
|
|
185
|
+
windows[window]["commits"] += 1
|
|
186
|
+
groups[is_marked]["commits"] += 1
|
|
187
|
+
path = None
|
|
188
|
+
while order and t - order[0][0] > RETOUCH_DAYS * 86400:
|
|
189
|
+
old_t, key = order.popleft()
|
|
190
|
+
adds = pending.get(key)
|
|
191
|
+
if adds and adds[0][0] == old_t:
|
|
192
|
+
adds.pop(0)
|
|
193
|
+
if not adds:
|
|
194
|
+
pending.pop(key, None)
|
|
195
|
+
continue
|
|
196
|
+
if line.startswith("diff --git "):
|
|
197
|
+
plain = _ANSI.sub("", line) # git ends even an uncoloured header with a reset
|
|
198
|
+
path = filetypes.unquote(plain.rsplit(" b/", 1)[-1]) if " b/" in plain else None
|
|
199
|
+
keep = bool(path) and _code_path(path, generated, vendored)
|
|
200
|
+
continue
|
|
201
|
+
if not keep:
|
|
202
|
+
continue
|
|
203
|
+
if line.startswith((_ADDED, _MOVED)):
|
|
204
|
+
text = _ANSI.sub("", line)[1:].strip()
|
|
205
|
+
moved = line.startswith(_MOVED)
|
|
206
|
+
for c in (windows[window], groups[is_marked]):
|
|
207
|
+
c["added"] += 1
|
|
208
|
+
c["moved"] += moved
|
|
209
|
+
if sum(ch.isalnum() for ch in text) >= 3:
|
|
210
|
+
pending.setdefault((path, text), []).append((t, window, is_marked))
|
|
211
|
+
order.append((t, (path, text)))
|
|
212
|
+
elif line.startswith((_DELETED, _DELETED_MOVED)):
|
|
213
|
+
text = _ANSI.sub("", line)[1:].strip()
|
|
214
|
+
adds = pending.get((path, text))
|
|
215
|
+
while adds and t - adds[0][0] > RETOUCH_DAYS * 86400:
|
|
216
|
+
adds.pop(0) # too old to count, and older than any later deletion will reach
|
|
217
|
+
if adds and adds[0][0] < t:
|
|
218
|
+
at_, w, m = adds.pop(0)
|
|
219
|
+
windows[w]["churned"] += 1
|
|
220
|
+
groups[m]["churned"] += 1
|
|
221
|
+
proc.stdout.close()
|
|
222
|
+
proc.wait()
|
|
223
|
+
|
|
224
|
+
def shares(c):
|
|
225
|
+
added = c.get("added", 0)
|
|
226
|
+
return {"commits": c.get("commits", 0), "added": added, "moved": c.get("moved", 0), "churned": c.get("churned", 0),
|
|
227
|
+
"moved_share": round(c.get("moved", 0) / added, 4) if added else None,
|
|
228
|
+
"churn_share": round(c.get("churned", 0) / added, 4) if added else None}
|
|
229
|
+
day = lambda ts: dt.datetime.fromtimestamp(ts, dt.timezone.utc).date().isoformat() # noqa: E731
|
|
230
|
+
return {"windows": [{"label": "last year", "from": day(end - YEAR), "to": day(end), **shares(windows["last"])},
|
|
231
|
+
{"label": "the year before", "from": day(start), "to": day(end - YEAR), **shares(windows["before"])}],
|
|
232
|
+
"cohort": {"marked": shares(groups[True]), "rest": shares(groups[False])},
|
|
233
|
+
"churn_days": RETOUCH_DAYS, "moved": "git --color-moved=blocks"}
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _watch_files(out_dir: str):
|
|
237
|
+
"""The watch list's top files, when the change analysis and scc have written their outputs; None
|
|
238
|
+
otherwise, or when this runs as a script outside the package."""
|
|
239
|
+
if not (os.path.exists(os.path.join(out_dir, "size.json")) and os.path.exists(os.path.join(out_dir, "maat-revisions.csv"))):
|
|
240
|
+
return None
|
|
241
|
+
try:
|
|
242
|
+
from . import load, watch
|
|
243
|
+
except ImportError:
|
|
244
|
+
return None
|
|
245
|
+
try:
|
|
246
|
+
report = load.load_report(out_dir, nested=False)
|
|
247
|
+
except load.Unreadable:
|
|
248
|
+
return None
|
|
249
|
+
return {r["file"] for r in watch.risks(report)[:WATCH_TOP]}
|
|
250
|
+
|
|
251
|
+
|
|
134
252
|
def shape(commits: list) -> dict:
|
|
135
253
|
"""How commits arrive, as neutral numbers: the share landing in runs of five or more by one author
|
|
136
254
|
within ten minutes of each other, the share with conventional-commit subjects, and how many hours
|
|
@@ -231,7 +349,20 @@ def main(argv=None) -> int:
|
|
|
231
349
|
print(f"provenance.py: {(e.stderr or b'').decode('utf-8', 'replace').strip() or e}", file=sys.stderr)
|
|
232
350
|
return 1
|
|
233
351
|
inventory = trailers(commits)
|
|
234
|
-
|
|
352
|
+
watch_files = _watch_files(args[0])
|
|
353
|
+
meta = {}
|
|
354
|
+
try:
|
|
355
|
+
with open(os.path.join(args[0], "meta.json"), encoding="utf-8") as fh:
|
|
356
|
+
meta = json.load(fh)
|
|
357
|
+
except (OSError, ValueError):
|
|
358
|
+
pass
|
|
359
|
+
marked = marker(inventory)
|
|
360
|
+
result = {"trailers": inventory, "cohort": cohort(commits, inventory, watch_files), "shape": shape(commits), "agents": agents(repo)}
|
|
361
|
+
if watch_files is not None:
|
|
362
|
+
result["cohort"]["watch_top"] = WATCH_TOP
|
|
363
|
+
if commits:
|
|
364
|
+
result["lines"] = lines(repo, commits[-1]["time"], {c["hash"] for c in commits if marked(c)}, set(meta.get("generated") or []),
|
|
365
|
+
filetypes.vendor_dirs({"meta": meta}))
|
|
235
366
|
with open(os.path.join(args[0], "provenance.json"), "w", encoding="utf-8") as fh:
|
|
236
367
|
json.dump(result, fh)
|
|
237
368
|
return 0
|
|
@@ -548,14 +548,35 @@ def trailers_section(report: dict, full: bool = True, width=None) -> dict:
|
|
|
548
548
|
if marked.get("commits"):
|
|
549
549
|
def pair(key):
|
|
550
550
|
return f"{_pct(marked.get(key, 0), marked['commits'])} against {_pct(rest.get(key, 0), rest.get('commits') or 0)}"
|
|
551
|
+
watch_part = f", touched a file on the watch list's top {co['watch_top']} {pair('watch')}" if co.get("watch_top") else ""
|
|
551
552
|
notes.append(f"marked commits ({co.get('definition')}): {marked['commits']:,}, {round(100 * co.get('share', 0))}% of the history; "
|
|
552
|
-
f"reverted {pair('reverted')} for the rest, fixes {pair('fixes')}, a file changed again within two weeks {pair('retouched')}")
|
|
553
|
+
f"reverted {pair('reverted')} for the rest, fixes {pair('fixes')}, a file changed again within two weeks {pair('retouched')}{watch_part}")
|
|
553
554
|
if sh:
|
|
554
555
|
notes.append(f"{round(100 * sh.get('burst_share', 0))}% of commits land in bursts of five or more within ten minutes; "
|
|
555
556
|
f"{round(100 * sh.get('conventional_share', 0))}% have conventional-commit subjects; commits come in {sh.get('hours_used', 0)} hours of the day")
|
|
556
557
|
return _section("Trailers", columns, rows, note=None if rows else "no trailers", caption="\n".join(notes) or None)
|
|
557
558
|
|
|
558
559
|
|
|
560
|
+
def lines_section(report: dict, full: bool = True, width=None) -> dict:
|
|
561
|
+
"""Lines added to code files in the last year and the year before, the share git marks as moved and
|
|
562
|
+
the share deleted again within two weeks, and the same for the marked cohort against the rest:
|
|
563
|
+
--full and Markdown only. A direction for this repository, not a score."""
|
|
564
|
+
ln = (report.get("provenance") or {}).get("lines") or {}
|
|
565
|
+
|
|
566
|
+
def share(x):
|
|
567
|
+
return "-" if x is None else f"{100 * x:.1f}%"
|
|
568
|
+
rows = [(f"{w['label']} ({w['from']} to {w['to']})", w["commits"], w["added"], share(w.get("moved_share")), share(w.get("churn_share")))
|
|
569
|
+
for w in ln.get("windows") or []]
|
|
570
|
+
co = ln.get("cohort") or {}
|
|
571
|
+
if (co.get("marked") or {}).get("commits"):
|
|
572
|
+
rows += [(label, c["commits"], c["added"], share(c.get("moved_share")), share(c.get("churn_share")))
|
|
573
|
+
for label, c in (("marked commits, both years", co["marked"]), ("the rest, both years", co["rest"]))]
|
|
574
|
+
columns = [("period", {"overflow": "fold"}), ("commits", RIGHT), ("lines added", RIGHT), ("moved", RIGHT), (f"churned in {ln.get('churn_days', 14)} days", RIGHT)]
|
|
575
|
+
return _section("Changed lines", columns, rows, note=None if rows else "no history in the last two years",
|
|
576
|
+
caption="code files only; moved: lines git's moved-code detection marks (--color-moved=blocks); churned: deleted again "
|
|
577
|
+
"within two weeks from the same file with the same text" if rows else None)
|
|
578
|
+
|
|
579
|
+
|
|
559
580
|
def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
|
|
560
581
|
"""Change frequency times size, Tornhill-style. Files no longer in the tree sort last. Drawn
|
|
561
582
|
under `--full` and in the Markdown export only; the default terminal report leaves it to the
|
|
@@ -791,10 +812,10 @@ def compare_section(result: dict) -> dict:
|
|
|
791
812
|
|
|
792
813
|
|
|
793
814
|
BUILDERS = [watch_section, watch_by_component_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
|
|
794
|
-
hotspots_section, coupling_section, signing_section, trailers_section, age_section, functions_section, health_section, osps_section]
|
|
815
|
+
hotspots_section, coupling_section, signing_section, trailers_section, lines_section, age_section, functions_section, health_section, osps_section]
|
|
795
816
|
# `--full` and Markdown only: Size, Activity and Code age are interesting once and rarely change what you
|
|
796
817
|
# do next; Hotspots ranks the files the watch list already leads with, by the same product.
|
|
797
|
-
FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers", "watch_by_component", "osps"}
|
|
818
|
+
FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers", "lines", "watch_by_component", "osps"}
|
|
798
819
|
|
|
799
820
|
|
|
800
821
|
def sections(report: dict, full: bool = True, width=None) -> list:
|
|
@@ -263,7 +263,7 @@ def plan(repo_dir: str, out_dir: str, branch: str = "HEAD", age: bool = True, pl
|
|
|
263
263
|
{"name": "change analysis", "argv": [sys.executable, MAAT_SCRIPT, log, out_dir, *type_args, *(["--now", now] if now else []), *(["--since", since] if since else []), "--aliases", o("meta.json"), *revs_args], "stdout": None, "deps": ["git-log"]},
|
|
264
264
|
{"name": "signing", "argv": [sys.executable, "-m", "gitmole.signing", out_dir], "stdout": None, "deps": []}, # the gpgsig headers, no keyring
|
|
265
265
|
{"name": "hygiene", "argv": [sys.executable, "-m", "gitmole.hygiene", out_dir], "stdout": None, "deps": []}, # the Scorecard checks, from the clone
|
|
266
|
-
{"name": "provenance", "argv": [sys.executable, "-m", "gitmole.provenance", out_dir], "stdout": None, "deps": []}, # trailers, cohorts, agent files
|
|
266
|
+
{"name": "provenance", "argv": [sys.executable, "-m", "gitmole.provenance", out_dir], "stdout": None, "deps": ["scc", "change analysis"]}, # trailers, cohorts, agent files; the watch list for the hit rate
|
|
267
267
|
]
|
|
268
268
|
workers = procs or blame.default_procs()
|
|
269
269
|
if lizard:
|
|
@@ -36,7 +36,7 @@ try:
|
|
|
36
36
|
except ImportError: # run as a script: the package directory is sys.path[0]
|
|
37
37
|
import filetypes
|
|
38
38
|
|
|
39
|
-
ANALYSER = "
|
|
39
|
+
ANALYSER = "3" # bump whenever what a file yields changes (a metric, an import's shape): the cache key carries it
|
|
40
40
|
MAX_BYTES = 1_000_000
|
|
41
41
|
FUNCTIONS_KEPT = 3000
|
|
42
42
|
|
|
@@ -73,6 +73,89 @@ NO_INCREMENT = {"try_statement"} # the try nests its body; the catch is what S
|
|
|
73
73
|
LOGICAL = {"&&", "||", "and", "or"}
|
|
74
74
|
DEBT = re.compile(r"\b(TODO|FIXME|XXX|HACK)\b")
|
|
75
75
|
|
|
76
|
+
# --- shapes: an error swallowed, an address in a literal, code left in a comment -----------------
|
|
77
|
+
CATCH = {"catch_clause", "except_clause", "rescue"}
|
|
78
|
+
BODY = {"block", "statement_block", "compound_statement", "then"}
|
|
79
|
+
EMPTY_STATEMENTS = {"pass_statement", "empty_statement"}
|
|
80
|
+
STRING = {"string", "string_literal", "interpreted_string_literal", "raw_string_literal", "template_string", "encapsed_string"}
|
|
81
|
+
ATTRIBUTE = {"attribute", "attribute_list", "annotation", "marker_annotation", "decorator", "attribute_item"}
|
|
82
|
+
_QUOTED = re.compile(r"""^[A-Za-z@$]*(["'`]+)(.*?)\1$""", re.S)
|
|
83
|
+
_IPV4 = re.compile(r"^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})(?::\d{1,5})?$")
|
|
84
|
+
# a comment line that reads as a statement: an assignment, a call, a keyword that opens one, a brace
|
|
85
|
+
_CODE_LINE = re.compile(r"""^(?:(?:return|if|elif|else|for|while|import|from|var|let|const|def|function|class|#include|throw|raise|break|continue|await)\b.*[;:{})\]]|[A-Za-z_$][\w.$\[\]]*\s*(?:=|\+=|-=|\|=|:=)\s*\S.*|[\w.$]+(?:\.[\w$]+)*\(.*\)\s*;?|[{}]\s*[;)]?|.*[;{]|\}\s*else\b.*)$""")
|
|
86
|
+
_COMMENT_MARK = re.compile(r"^\s*(?://+|/\*+|\*+/?|#+|--|;+) ?")
|
|
87
|
+
CODE_SHARE = 0.8 # of a comment block's lines that read as code, for the block to count as code left in a comment
|
|
88
|
+
_DIRECTIVE = re.compile(r"^(?:eslint|prettier|istanbul|noqa|type:|pylint|@ts-|jshint|global |c8 |nolint|NOLINT|clang-format|fmt:|pragma|region|endregion|-\*-|SPDX-|Copyright|http)", re.I)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
BROAD_PYTHON = {"Exception", "BaseException"} # the language's own root classes
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _is_empty_catch(node, src: bytes = b"") -> bool:
|
|
95
|
+
"""A catch, except or rescue whose body does nothing and says nothing: no statement but `pass`, and
|
|
96
|
+
no comment, since a comment is the author saying the error is ignored on purpose (Sonar's S108).
|
|
97
|
+
Python's `except SomeError: pass` is the language's idiom for an expected failure (EAFP), so there
|
|
98
|
+
only a bare `except:` or one that catches Exception or BaseException counts."""
|
|
99
|
+
if node.type == "except_clause":
|
|
100
|
+
caught = [c for c in node.named_children if c.type not in BODY and c.type != "comment"]
|
|
101
|
+
if caught and not all(_text(src, c).strip("() ") in BROAD_PYTHON or _text(src, c).split(" as ")[0].strip("() ") in BROAD_PYTHON for c in caught):
|
|
102
|
+
return False
|
|
103
|
+
body = None
|
|
104
|
+
for c in node.named_children:
|
|
105
|
+
if c.type == "comment":
|
|
106
|
+
return False
|
|
107
|
+
if c.type in BODY:
|
|
108
|
+
body = c
|
|
109
|
+
if body is None:
|
|
110
|
+
return node.type == "rescue" # Ruby: an empty rescue has no `then` at all
|
|
111
|
+
for c in body.named_children:
|
|
112
|
+
if c.type == "comment" or c.type not in EMPTY_STATEMENTS:
|
|
113
|
+
return False
|
|
114
|
+
return True
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _address(text: str):
|
|
118
|
+
"""The IPv4 address a string literal is, with its port, or None: loopback, the unspecified and
|
|
119
|
+
broadcast addresses, netmasks, the documentation ranges (RFC 5737), a trailing .0 (a network, or a
|
|
120
|
+
four-part version like 1.0.0.0), and a first octet of 0, 1 or 2, which is how an ASN.1 object
|
|
121
|
+
identifier (2.5.4.3) starts, are left out."""
|
|
122
|
+
m = _QUOTED.match(text.strip())
|
|
123
|
+
value = (m.group(2) if m else text).strip()
|
|
124
|
+
ip = _IPV4.match(value)
|
|
125
|
+
if not ip:
|
|
126
|
+
return None
|
|
127
|
+
octets = [int(x) for x in ip.groups()]
|
|
128
|
+
if any(o > 255 for o in octets) or octets[0] <= 2 or octets[0] in (127, 255) or octets[3] in (0, 255): # 0-2: an object identifier's first arc
|
|
129
|
+
return None
|
|
130
|
+
if octets[:3] in ([192, 0, 2], [198, 51, 100], [203, 0, 113]):
|
|
131
|
+
return None
|
|
132
|
+
return value
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _code_like(line: str) -> bool:
|
|
136
|
+
s = line.strip()
|
|
137
|
+
return bool(s) and not _DIRECTIVE.match(s) and not s.endswith(".") and bool(_CODE_LINE.match(s)) and not re.match(r"^[A-Za-z]+(?: [a-z]+){3,}", s)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def commented_code_lines(text: str) -> int:
|
|
141
|
+
"""How many lines of one comment block (a block comment, or a run of line comments on consecutive
|
|
142
|
+
lines) read as code; 0 unless four in five of its lines do and one of them starts right after the
|
|
143
|
+
comment marker, as an editor's comment-out leaves it. Prose with a worked example under it (indented,
|
|
144
|
+
or introduced by a line ending in a colon), and documentation comments, whose examples are meant to
|
|
145
|
+
be there, are not code left behind."""
|
|
146
|
+
if text.lstrip().startswith(("/**", "///", "//!", "#!", "/*!")):
|
|
147
|
+
return 0
|
|
148
|
+
lines = [re.sub(r"\s*\*/\s*$", "", _COMMENT_MARK.sub("", l)) for l in text.split("\n")]
|
|
149
|
+
lines = [l.rstrip() for l in lines if l.strip() and l.strip() not in ("*/", "*")]
|
|
150
|
+
if not lines:
|
|
151
|
+
return 0
|
|
152
|
+
code = [l for l in lines if _code_like(l)]
|
|
153
|
+
if len(code) < CODE_SHARE * len(lines) or not any(not l[:1].isspace() for l in code):
|
|
154
|
+
return 0
|
|
155
|
+
if any(l.strip().endswith(":") and not _code_like(l) for l in lines):
|
|
156
|
+
return 0 # "Build the dispatcher function:" introduces an example
|
|
157
|
+
return len(code)
|
|
158
|
+
|
|
76
159
|
|
|
77
160
|
def cache_root() -> str | None:
|
|
78
161
|
explicit = os.environ.get("GITMOLE_CACHE")
|
|
@@ -217,6 +300,17 @@ def analyse(src: bytes, lang_name: str, language) -> dict:
|
|
|
217
300
|
funcs, stack, done = [], [], [] # stack: one frame per ancestor on the way down
|
|
218
301
|
comments = comment_lines = definitions = 0
|
|
219
302
|
debt, imports = [], []
|
|
303
|
+
shapes = {"empty_catch": [], "bare_except": [], "addresses": [], "commented_code": 0, "commented_sample": []}
|
|
304
|
+
block = [] # the open comment block: [first line, last line, text, made of line comments]
|
|
305
|
+
|
|
306
|
+
def flush():
|
|
307
|
+
if block:
|
|
308
|
+
code = commented_code_lines(block[2])
|
|
309
|
+
if code:
|
|
310
|
+
shapes["commented_code"] += code
|
|
311
|
+
if len(shapes["commented_sample"]) < 5:
|
|
312
|
+
shapes["commented_sample"].append(block[0] + 1)
|
|
313
|
+
block.clear()
|
|
220
314
|
main_guard = False
|
|
221
315
|
chains = [] # open runs of logical operators: [count]
|
|
222
316
|
while True:
|
|
@@ -263,6 +357,22 @@ def analyse(src: bytes, lang_name: str, language) -> dict:
|
|
|
263
357
|
m = DEBT.search(text)
|
|
264
358
|
if m:
|
|
265
359
|
debt.append({"line": node.start_point[0] + 1, "tag": m.group(1), "text": " ".join(text.split())[:100]})
|
|
360
|
+
own_line = not src[src.rfind(b"\n", 0, node.start_byte) + 1:node.start_byte].strip() # not trailing code
|
|
361
|
+
line_comment = text.startswith(("//", "#", "--")) and "\n" not in text.rstrip()
|
|
362
|
+
if block and line_comment and block[3] and own_line and node.start_point[0] == block[1] + 1:
|
|
363
|
+
block[1], block[2] = node.end_point[0], block[2] + "\n" + text
|
|
364
|
+
else:
|
|
365
|
+
flush()
|
|
366
|
+
if own_line:
|
|
367
|
+
block.extend([node.start_point[0], node.end_point[0], text, line_comment])
|
|
368
|
+
elif t in CATCH and _is_empty_catch(node, src):
|
|
369
|
+
shapes["empty_catch"].append(node.start_point[0] + 1)
|
|
370
|
+
if t == "except_clause" and not any(c.type not in BODY and c.type != "comment" for c in node.named_children):
|
|
371
|
+
shapes["bare_except"].append(node.start_point[0] + 1)
|
|
372
|
+
elif t in STRING and node.end_byte - node.start_byte <= 30 and not any(s[0] in ATTRIBUTE for s in stack):
|
|
373
|
+
value = _address(_text(src, node))
|
|
374
|
+
if value:
|
|
375
|
+
shapes["addresses"].append({"line": node.start_point[0] + 1, "value": value})
|
|
266
376
|
found = _import(node, src, lang_name)
|
|
267
377
|
if found:
|
|
268
378
|
imports.extend(found)
|
|
@@ -278,7 +388,8 @@ def analyse(src: bytes, lang_name: str, language) -> dict:
|
|
|
278
388
|
if cursor.goto_next_sibling():
|
|
279
389
|
break
|
|
280
390
|
if not cursor.goto_parent():
|
|
281
|
-
|
|
391
|
+
flush()
|
|
392
|
+
return _result(done, comments, comment_lines, definitions, debt, imports, main_guard, src, tree, shapes)
|
|
282
393
|
|
|
283
394
|
|
|
284
395
|
def _in_chain(node) -> bool:
|
|
@@ -304,10 +415,10 @@ def _leave(frame, funcs, done, chains):
|
|
|
304
415
|
done.append(funcs.pop())
|
|
305
416
|
|
|
306
417
|
|
|
307
|
-
def _result(done, comments, comment_lines, definitions, debt, imports, main_guard, src, tree) -> dict:
|
|
418
|
+
def _result(done, comments, comment_lines, definitions, debt, imports, main_guard, src, tree, shapes) -> dict:
|
|
308
419
|
lines = src.count(b"\n") + (1 if src and not src.endswith(b"\n") else 0)
|
|
309
420
|
return {"lines": lines, "comments": comments, "comment_lines": comment_lines, "definitions": definitions,
|
|
310
|
-
"debt": debt, "imports": imports, "main": main_guard or src.startswith(b"#!"), "errors": tree.root_node.has_error,
|
|
421
|
+
"debt": debt, "imports": imports, "main": main_guard or src.startswith(b"#!"), "errors": tree.root_node.has_error, "shapes": shapes,
|
|
311
422
|
"functions": [{"name": f.name, "start": f.start, "end": f.end, "nesting": f.max_nesting, "cognitive": f.cognitive,
|
|
312
423
|
"complex_conditions": f.complex, "bumps": f.bumps} for f in done]}
|
|
313
424
|
|
|
@@ -518,6 +629,23 @@ def unreferenced(files: dict, edges: dict, resolved: dict, entries: set) -> list
|
|
|
518
629
|
return [p for p in out if files[p]["language"] not in loud]
|
|
519
630
|
|
|
520
631
|
|
|
632
|
+
def _slim_shapes(s: dict) -> dict:
|
|
633
|
+
"""The shapes one file holds, as structure.json keeps them: counts and the first few lines; empty
|
|
634
|
+
lists and zero counts are left out so a clean file costs nothing."""
|
|
635
|
+
out = {}
|
|
636
|
+
for key in ("empty_catch", "bare_except"):
|
|
637
|
+
if s.get(key):
|
|
638
|
+
out[key] = s[key][:5]
|
|
639
|
+
out[key + "_count"] = len(s[key])
|
|
640
|
+
if s.get("addresses"):
|
|
641
|
+
out["addresses"] = s["addresses"][:5]
|
|
642
|
+
out["addresses_count"] = len(s["addresses"])
|
|
643
|
+
if s.get("commented_code"):
|
|
644
|
+
out["commented_code"] = s["commented_code"]
|
|
645
|
+
out["commented_sample"] = s.get("commented_sample") or []
|
|
646
|
+
return out
|
|
647
|
+
|
|
648
|
+
|
|
521
649
|
def entries_dirs(entries: set) -> set:
|
|
522
650
|
return {e for e in entries if e.endswith("package.json")}
|
|
523
651
|
|
|
@@ -576,6 +704,7 @@ def collect(repo: str, procs: int = None, vendored=()) -> dict:
|
|
|
576
704
|
slim = {p: {"language": v["language"], "lines": v.get("lines", 0), "comments": v.get("comments", 0), "comment_lines": v.get("comment_lines", 0),
|
|
577
705
|
"definitions": v.get("definitions", 0), "debt": len(v.get("debt") or []), "debt_sample": (v.get("debt") or [])[:5],
|
|
578
706
|
"main": v.get("main", False), "imports": edges.get(p, []), "errors": v.get("errors", False),
|
|
707
|
+
"shapes": _slim_shapes(v.get("shapes") or {}),
|
|
579
708
|
"max_nesting": max((f["nesting"] for f in v.get("functions") or []), default=0),
|
|
580
709
|
"max_cognitive": max((f["cognitive"] for f in v.get("functions") or []), default=0)}
|
|
581
710
|
for p, v in files.items() if "failed" not in v}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gitmole
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.22.0
|
|
4
4
|
Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/antvinni/gitmole
|
|
@@ -74,6 +74,7 @@ gitmole . --json report.json # every table, the watch list and the fin
|
|
|
74
74
|
gitmole . --fail-on warning # exit 3 if any finding is a warning or worse
|
|
75
75
|
gitmole . --risk main --risk-threshold 10 # exit 3 if the files changed since main hold over 10% of the risk
|
|
76
76
|
gitmole . --sarif gitmole.sarif # the findings for GitHub code scanning or GitLab
|
|
77
|
+
gitmole . --sbom sbom.cdx.json # a CycloneDX SBOM of every package the lock files pin
|
|
77
78
|
gitmole . --compare last.json # what changed since an earlier --json export
|
|
78
79
|
gitmole analysis-repo --no-run --hook # a coding agent's edit hook: history's view of the files it just touched
|
|
79
80
|
gitmole . --since 2y --full # the current team, every row and column
|
|
@@ -92,9 +93,9 @@ Reports on repositories you know, each at a pinned commit, published as gitmole
|
|
|
92
93
|
|
|
93
94
|
| Repository | Commit | Commits | Lines | gitmole run |
|
|
94
95
|
|---|---|---:|---:|---:|
|
|
95
|
-
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 |
|
|
96
|
-
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 |
|
|
97
|
-
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 |
|
|
96
|
+
| [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 59 s |
|
|
97
|
+
| [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 135 s |
|
|
98
|
+
| [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 157 s |
|
|
98
99
|
|
|
99
100
|
Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
|
|
100
101
|
|
|
@@ -70,6 +70,16 @@ class Induced(unittest.TestCase):
|
|
|
70
70
|
|
|
71
71
|
|
|
72
72
|
class Score(unittest.TestCase):
|
|
73
|
+
def test_effort_is_the_false_alarms_before_the_first_hit_and_the_lines_read(self):
|
|
74
|
+
r = evaluate.report_at(COMMITS, "2025-06-01", SIZE, {"bots": []}, [], [])
|
|
75
|
+
ranked = evaluate.variants(r)["watch list (hotspot)"]
|
|
76
|
+
out = evaluate.effort(r, {ranked[1]}, 15)["watch list (hotspot)"]
|
|
77
|
+
self.assertEqual(out[0], 1, "one file before the first labelled one")
|
|
78
|
+
self.assertEqual(out[1], sum(SIZE["files"][f]["code"] for f in ranked[:15]))
|
|
79
|
+
self.assertEqual(evaluate.effort(r, set(), 15)["watch list (hotspot)"][0], len(ranked[:15]), "no hit: every file was a false alarm")
|
|
80
|
+
table = evaluate.effort_table([{"a": (1, 100)}, {"a": (3, 300)}])
|
|
81
|
+
self.assertIn("| a | 2 | 200 |", table)
|
|
82
|
+
|
|
73
83
|
def test_the_report_at_t_knows_nothing_after_t(self):
|
|
74
84
|
r = evaluate.report_at(COMMITS, "2025-06-01", SIZE, {"bots": []}, [], [])
|
|
75
85
|
self.assertEqual({x["entity"]: x["n-revs"] for x in r["revisions"]}, {"core/a.py": 3, "core/b.py": 2, "tests/test_a.py": 1})
|
|
@@ -110,6 +110,46 @@ class Agents(unittest.TestCase):
|
|
|
110
110
|
self.assertNotIn("p4ss", json.dumps(out))
|
|
111
111
|
|
|
112
112
|
|
|
113
|
+
class Lines(unittest.TestCase):
|
|
114
|
+
def _write(self, r, path, text, message, date):
|
|
115
|
+
with open(os.path.join(r.d, path), "w") as fh:
|
|
116
|
+
fh.write(text)
|
|
117
|
+
r.git("add", "-A", date=date)
|
|
118
|
+
r.git("commit", "-q", "-m", message, date=date)
|
|
119
|
+
|
|
120
|
+
def test_moved_and_churned_lines_by_year_and_cohort(self):
|
|
121
|
+
body = "".join(f"def function_number_{i}(argument):\n return argument * {i} + compute_offset({i})\n" for i in range(6))
|
|
122
|
+
with tempfile.TemporaryDirectory() as d:
|
|
123
|
+
r = Repo(d)
|
|
124
|
+
self._write(r, "c.py", "unrelated_value = compute_something()\n", "start", "2024-06-01T10:00:00") # the year before
|
|
125
|
+
self._write(r, "a.py", body + "keep_this_line = 1\ntemporary_line = 2\n", "add a", "2025-06-01T10:00:00")
|
|
126
|
+
self._write(r, "a.py", body + "keep_this_line = 1\n", "drop the temporary line\n\nAssisted-by: Tool", "2025-06-05T10:00:00")
|
|
127
|
+
with open(os.path.join(d, "c.py"), "a") as fh:
|
|
128
|
+
fh.write(body)
|
|
129
|
+
self._write(r, "a.py", "keep_this_line = 1\n", "move the functions into c.py", "2025-07-01T10:00:00")
|
|
130
|
+
commits = provenance.read_commits(d)
|
|
131
|
+
marked = provenance.marker(provenance.trailers(commits))
|
|
132
|
+
out = provenance.lines(d, commits[-1]["time"], {c["hash"] for c in commits if marked(c)})
|
|
133
|
+
last, before = out["windows"]
|
|
134
|
+
self.assertEqual((before["commits"], before["added"]), (1, 1))
|
|
135
|
+
self.assertEqual(last["added"], len(body.splitlines()) * 2 + 2)
|
|
136
|
+
self.assertEqual(last["churned"], 1, "the temporary line went within four days; the kept line and the moved functions are not churn")
|
|
137
|
+
self.assertEqual(last["moved"], len(body.splitlines()), "git marks the functions as moved from a.py to b.py")
|
|
138
|
+
self.assertEqual(out["cohort"]["marked"]["commits"], 1)
|
|
139
|
+
self.assertEqual(out["cohort"]["rest"]["churned"], 1, "the churn belongs to the commit that added the line")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class WatchHits(unittest.TestCase):
|
|
143
|
+
def test_each_cohort_counts_its_commits_touching_a_watched_file(self):
|
|
144
|
+
with tempfile.TemporaryDirectory() as d:
|
|
145
|
+
history(d)
|
|
146
|
+
commits = provenance.read_commits(d)
|
|
147
|
+
inv = provenance.trailers(commits)
|
|
148
|
+
out = provenance.cohort(commits, inv, {"a.py"})
|
|
149
|
+
self.assertEqual((out["cohort"]["watch"], out["rest"]["watch"]), (1, 2), "the helper commit is marked; start and the fix are not")
|
|
150
|
+
self.assertNotIn("watch", provenance.cohort(commits, inv)["rest"], "no watch list, no count")
|
|
151
|
+
|
|
152
|
+
|
|
113
153
|
class Step(unittest.TestCase):
|
|
114
154
|
def test_the_step_writes_provenance_json(self):
|
|
115
155
|
with tempfile.TemporaryDirectory() as d:
|
|
@@ -121,7 +161,7 @@ class Step(unittest.TestCase):
|
|
|
121
161
|
self.assertEqual(p.returncode, 0, p.stderr)
|
|
122
162
|
with open(os.path.join(out, "provenance.json")) as fh:
|
|
123
163
|
data = json.load(fh)
|
|
124
|
-
self.assertEqual(set(data), {"trailers", "cohort", "shape", "agents"})
|
|
164
|
+
self.assertEqual(set(data), {"trailers", "cohort", "shape", "agents", "lines"})
|
|
125
165
|
|
|
126
166
|
|
|
127
167
|
if __name__ == "__main__":
|
|
@@ -1502,3 +1502,18 @@ class Excerpt(unittest.TestCase):
|
|
|
1502
1502
|
|
|
1503
1503
|
if __name__ == "__main__":
|
|
1504
1504
|
unittest.main()
|
|
1505
|
+
|
|
1506
|
+
|
|
1507
|
+
class ChangedLines(unittest.TestCase):
|
|
1508
|
+
def test_two_windows_and_the_cohorts_when_there_are_marked_commits(self):
|
|
1509
|
+
w = {"commits": 10, "added": 200, "moved": 20, "churned": 10, "moved_share": 0.1, "churn_share": 0.05}
|
|
1510
|
+
rep = {"provenance": {"lines": {"windows": [{"label": "last year", "from": "2025-01-01", "to": "2026-01-01", **w},
|
|
1511
|
+
{"label": "the year before", "from": "2024-01-01", "to": "2025-01-01", **w, "added": 0, "moved_share": None, "churn_share": None}],
|
|
1512
|
+
"cohort": {"marked": {**w, "commits": 2}, "rest": w}, "churn_days": 14},
|
|
1513
|
+
"cohort": {"cohort": {"commits": 2, "watch": 1}, "rest": {"commits": 8, "watch": 2}, "watch_top": 15, "share": 0.2}}}
|
|
1514
|
+
sec = render.lines_section(rep)
|
|
1515
|
+
self.assertEqual(sec["rows"][0], ["last year (2025-01-01 to 2026-01-01)", "10", "200", "10.0%", "5.0%"])
|
|
1516
|
+
self.assertEqual(sec["rows"][1][3:], ["-", "-"])
|
|
1517
|
+
self.assertEqual([r[0] for r in sec["rows"][2:]], ["marked commits, both years", "the rest, both years"])
|
|
1518
|
+
self.assertIn("lines", render.FULL_ONLY)
|
|
1519
|
+
self.assertIn("touched a file on the watch list's top 15 50% against 25%", render.trailers_section(rep)["caption"] or render.trailers_section(rep)["note"] or "")
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""The shape rules on the tree-sitter pass: an error caught and dropped, an address written into a string
|
|
2
|
+
literal, code left in a comment. The parsing tests need gitmole[structure]; the findings do not."""
|
|
3
|
+
import unittest
|
|
4
|
+
|
|
5
|
+
from gitmole import findings, structure
|
|
6
|
+
from tests.test_findings import report
|
|
7
|
+
from tests.test_structure import HAVE, parse
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class CommentedCode(unittest.TestCase):
|
|
11
|
+
"""The comment classifier is plain text, no grammar needed."""
|
|
12
|
+
|
|
13
|
+
def test_code_left_by_an_editor_counts(self):
|
|
14
|
+
self.assertEqual(structure.commented_code_lines("// const old = legacy(x);\n// console.log(state);\n// if (a) {\n// b();\n// }"), 5)
|
|
15
|
+
self.assertEqual(structure.commented_code_lines("/* legacy(x);\n other(y); */"), 2)
|
|
16
|
+
self.assertEqual(structure.commented_code_lines("# x = compute(a, b)"), 1)
|
|
17
|
+
|
|
18
|
+
def test_prose_examples_directives_and_doc_comments_do_not(self):
|
|
19
|
+
self.assertEqual(structure.commented_code_lines("# Some examples:\n# SomeModel.objects.annotate(x)\n# foo(bar)"), 0)
|
|
20
|
+
self.assertEqual(structure.commented_code_lines("// function f(fn) {\n// return g();\n// }"), 0, "indented under the marker: an example")
|
|
21
|
+
self.assertEqual(structure.commented_code_lines("// Build the dispatcher:\n// function Foo(a) {\n// return x;\n// }\n// }"), 0)
|
|
22
|
+
self.assertEqual(structure.commented_code_lines("/** @example foo(1); */"), 0)
|
|
23
|
+
self.assertEqual(structure.commented_code_lines("// eslint-disable-next-line no-console"), 0)
|
|
24
|
+
self.assertEqual(structure.commented_code_lines("/* 35 = OBSOLETE */"), 0)
|
|
25
|
+
self.assertEqual(structure.commented_code_lines("# This explains why we do it."), 0)
|
|
26
|
+
|
|
27
|
+
def test_addresses_leave_out_what_is_not_a_host(self):
|
|
28
|
+
self.assertEqual(structure._address("'10.1.2.3'"), "10.1.2.3")
|
|
29
|
+
self.assertEqual(structure._address("`203.0.114.9:8080`"), "203.0.114.9:8080")
|
|
30
|
+
for literal in ("'127.0.0.1'", "'0.0.0.0'", "'255.255.255.0'", "'1.0.0.0'", "'2.5.4.3'", "'192.0.2.10'", "'300.1.1.1'", "'v1.2.3.4'"):
|
|
31
|
+
self.assertIsNone(structure._address(literal), literal)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@unittest.skipUnless(HAVE, "gitmole[structure] not installed")
|
|
35
|
+
class Shapes(unittest.TestCase):
|
|
36
|
+
def test_python_counts_only_the_broad_except_that_does_nothing(self):
|
|
37
|
+
s = parse(".py", "try:\n x()\nexcept:\n pass\ntry:\n y()\nexcept ValueError:\n pass\n"
|
|
38
|
+
"try:\n z()\nexcept Exception as e:\n pass\ntry:\n w()\nexcept Exception:\n # ignored on purpose\n pass\n")["shapes"]
|
|
39
|
+
self.assertEqual(s["empty_catch"], [3, 11])
|
|
40
|
+
self.assertEqual(s["bare_except"], [3])
|
|
41
|
+
|
|
42
|
+
def test_empty_catch_in_other_languages_and_a_comment_saves_it(self):
|
|
43
|
+
self.assertEqual(parse(".js", "try { a() } catch (e) {}\ntry { b() } catch { /* ok */ }\n")["shapes"]["empty_catch"], [1])
|
|
44
|
+
self.assertEqual(parse(".java", "class A { void f() { try { g(); } catch (Exception e) { } } }\n")["shapes"]["empty_catch"], [1])
|
|
45
|
+
self.assertEqual(parse(".rb", "begin\n x\nrescue\nend\nbegin\n y\nrescue => e\n # ok\nend\n")["shapes"]["empty_catch"], [3])
|
|
46
|
+
|
|
47
|
+
def test_addresses_in_literals_but_not_in_attributes(self):
|
|
48
|
+
s = parse(".cs", '[assembly: AssemblyVersion("1.2.3.4")]\nclass A { void F() { var s = "8.8.4.4"; } }\n')["shapes"]
|
|
49
|
+
self.assertEqual(s["addresses"], [{"line": 2, "value": "8.8.4.4"}])
|
|
50
|
+
|
|
51
|
+
def test_line_comments_on_consecutive_lines_are_one_block_and_trailing_comments_are_not_code(self):
|
|
52
|
+
s = parse(".py", "x = 1 # y = 2\n# a = f(b)\n# c = g(d)\n\n# prose about it\n")["shapes"]
|
|
53
|
+
self.assertEqual((s["commented_code"], s["commented_sample"]), (2, [2]))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _report(files, **over):
|
|
57
|
+
return report(structure={"status": "run", "files": files}, **over)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class ShapeFindings(unittest.TestCase):
|
|
61
|
+
def test_swallowed_errors_from_five_and_never_in_tests(self):
|
|
62
|
+
files = {"src/a.py": {"shapes": {"empty_catch": [3, 9], "empty_catch_count": 4, "bare_except": [3], "bare_except_count": 1}},
|
|
63
|
+
"src/b.js": {"shapes": {"empty_catch": [7], "empty_catch_count": 1}},
|
|
64
|
+
"tests/test_a.py": {"shapes": {"empty_catch": [1], "empty_catch_count": 9}}}
|
|
65
|
+
f = findings.swallowed_errors(_report(files))
|
|
66
|
+
self.assertEqual([(x["rule"]["id"], x["severity"]) for x in f], [("swallowed_errors", "info")])
|
|
67
|
+
self.assertIn("5 empty catch blocks in 2 source files, 1 of them a bare except", f[0]["detail"])
|
|
68
|
+
self.assertEqual(f[0]["evidence"]["files"][0], {"file": "src/a.py", "start": 3, "count": 4})
|
|
69
|
+
del files["src/b.js"]
|
|
70
|
+
self.assertEqual(findings.swallowed_errors(_report(files)), [], "four in source is below the floor")
|
|
71
|
+
|
|
72
|
+
def test_addresses_and_commented_code(self):
|
|
73
|
+
files = {"src/net.go": {"shapes": {"addresses": [{"line": 4, "value": "10.0.0.7"}], "addresses_count": 1}},
|
|
74
|
+
"src/old.js": {"shapes": {"commented_code": 12, "commented_sample": [40, 90]}},
|
|
75
|
+
"src/some.js": {"shapes": {"commented_code": 3, "commented_sample": [5]}}}
|
|
76
|
+
f = findings.hardcoded_addresses(_report(files))
|
|
77
|
+
self.assertIn("10.0.0.7 at src/net.go:4", f[0]["detail"])
|
|
78
|
+
f = findings.commented_out_code(_report(files))
|
|
79
|
+
self.assertEqual([x["file"] for x in f[0]["evidence"]["files"]], ["src/old.js"])
|
|
80
|
+
self.assertIn("src/old.js (12 lines from line 40)", f[0]["detail"])
|
|
81
|
+
|
|
82
|
+
def test_nothing_without_the_structure_step(self):
|
|
83
|
+
for rule in (findings.swallowed_errors, findings.hardcoded_addresses, findings.commented_out_code):
|
|
84
|
+
self.assertEqual(rule(report()), [])
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
if __name__ == "__main__":
|
|
88
|
+
unittest.main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|