gitmole 0.20.0__tar.gz → 0.22.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {gitmole-0.20.0 → gitmole-0.22.0}/PKG-INFO +5 -4
  2. {gitmole-0.20.0 → gitmole-0.22.0}/README.md +4 -3
  3. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/__init__.py +1 -1
  4. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/evaluate.py +39 -3
  5. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/findings.py +73 -1
  6. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/provenance.py +135 -4
  7. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/render.py +24 -3
  8. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/run.py +1 -1
  9. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/structure.py +133 -4
  10. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/PKG-INFO +5 -4
  11. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/SOURCES.txt +1 -0
  12. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_evaluate.py +10 -0
  13. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_provenance.py +41 -1
  14. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_render.py +15 -0
  15. gitmole-0.22.0/tests/test_shapes.py +88 -0
  16. {gitmole-0.20.0 → gitmole-0.22.0}/LICENSE +0 -0
  17. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/__main__.py +0 -0
  18. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/backtest.py +0 -0
  19. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/banner.py +0 -0
  20. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/blame.py +0 -0
  21. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/classify.py +0 -0
  22. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/clean.py +0 -0
  23. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/cli.py +0 -0
  24. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/compare.py +0 -0
  25. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/coupling.py +0 -0
  26. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/deps.py +0 -0
  27. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/duplicates.py +0 -0
  28. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/filetypes.py +0 -0
  29. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/functions.py +0 -0
  30. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/hook.py +0 -0
  31. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/hotspots.py +0 -0
  32. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/hygiene.py +0 -0
  33. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/identity.py +0 -0
  34. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/imports.py +0 -0
  35. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/knowledge.py +0 -0
  36. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/leaks.py +0 -0
  37. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/licences.py +0 -0
  38. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/load.py +0 -0
  39. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/loss.py +0 -0
  40. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/maat.py +0 -0
  41. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/osps.py +0 -0
  42. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/sarif.py +0 -0
  43. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/sbom.py +0 -0
  44. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/signing.py +0 -0
  45. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/szz.py +0 -0
  46. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/textfmt.py +0 -0
  47. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/trend.py +0 -0
  48. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole/watch.py +0 -0
  49. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/dependency_links.txt +0 -0
  50. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/entry_points.txt +0 -0
  51. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/requires.txt +0 -0
  52. {gitmole-0.20.0 → gitmole-0.22.0}/gitmole.egg-info/top_level.txt +0 -0
  53. {gitmole-0.20.0 → gitmole-0.22.0}/pyproject.toml +0 -0
  54. {gitmole-0.20.0 → gitmole-0.22.0}/setup.cfg +0 -0
  55. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_backtest.py +0 -0
  56. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_banner.py +0 -0
  57. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_blame.py +0 -0
  58. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_classify.py +0 -0
  59. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_clean.py +0 -0
  60. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_cli.py +0 -0
  61. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_compare.py +0 -0
  62. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_coupling.py +0 -0
  63. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_declared.py +0 -0
  64. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_deps.py +0 -0
  65. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_duplicates.py +0 -0
  66. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_filetypes.py +0 -0
  67. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_findings.py +0 -0
  68. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_functions.py +0 -0
  69. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_golden.py +0 -0
  70. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_hook.py +0 -0
  71. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_hotspots.py +0 -0
  72. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_hygiene.py +0 -0
  73. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_identity.py +0 -0
  74. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_knowledge.py +0 -0
  75. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_leaks.py +0 -0
  76. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_load.py +0 -0
  77. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_loss.py +0 -0
  78. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_maat.py +0 -0
  79. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_packaging.py +0 -0
  80. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_render_examples.py +0 -0
  81. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_run.py +0 -0
  82. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_sarif.py +0 -0
  83. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_signing.py +0 -0
  84. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_structure.py +0 -0
  85. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_szz.py +0 -0
  86. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_textfmt.py +0 -0
  87. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_trend.py +0 -0
  88. {gitmole-0.20.0 → gitmole-0.22.0}/tests/test_watch.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.20.0
3
+ Version: 0.22.0
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -74,6 +74,7 @@ gitmole . --json report.json # every table, the watch list and the fin
74
74
  gitmole . --fail-on warning # exit 3 if any finding is a warning or worse
75
75
  gitmole . --risk main --risk-threshold 10 # exit 3 if the files changed since main hold over 10% of the risk
76
76
  gitmole . --sarif gitmole.sarif # the findings for GitHub code scanning or GitLab
77
+ gitmole . --sbom sbom.cdx.json # a CycloneDX SBOM of every package the lock files pin
77
78
  gitmole . --compare last.json # what changed since an earlier --json export
78
79
  gitmole analysis-repo --no-run --hook # a coding agent's edit hook: history's view of the files it just touched
79
80
  gitmole . --since 2y --full # the current team, every row and column
@@ -92,9 +93,9 @@ Reports on repositories you know, each at a pinned commit, published as gitmole
92
93
 
93
94
  | Repository | Commit | Commits | Lines | gitmole run |
94
95
  |---|---|---:|---:|---:|
95
- | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 57 s |
96
- | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 132 s |
97
- | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 151 s |
96
+ | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 59 s |
97
+ | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 135 s |
98
+ | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 157 s |
98
99
 
99
100
  Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
100
101
 
@@ -41,6 +41,7 @@ gitmole . --json report.json # every table, the watch list and the fin
41
41
  gitmole . --fail-on warning # exit 3 if any finding is a warning or worse
42
42
  gitmole . --risk main --risk-threshold 10 # exit 3 if the files changed since main hold over 10% of the risk
43
43
  gitmole . --sarif gitmole.sarif # the findings for GitHub code scanning or GitLab
44
+ gitmole . --sbom sbom.cdx.json # a CycloneDX SBOM of every package the lock files pin
44
45
  gitmole . --compare last.json # what changed since an earlier --json export
45
46
  gitmole analysis-repo --no-run --hook # a coding agent's edit hook: history's view of the files it just touched
46
47
  gitmole . --since 2y --full # the current team, every row and column
@@ -59,9 +60,9 @@ Reports on repositories you know, each at a pinned commit, published as gitmole
59
60
 
60
61
  | Repository | Commit | Commits | Lines | gitmole run |
61
62
  |---|---|---:|---:|---:|
62
- | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 57 s |
63
- | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 132 s |
64
- | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 151 s |
63
+ | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 59 s |
64
+ | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 135 s |
65
+ | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 157 s |
65
66
 
66
67
  Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
67
68
 
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.20.0"
3
+ __version__ = "0.22.0"
@@ -62,8 +62,14 @@ def labelled_between(commits: list, labels: dict, start: str, end: str) -> set:
62
62
  before `end` (ApacheJIT, Defectors: see szz.read_labels): the paths the label names, or every
63
63
  source file the commit touched. Either side may be abbreviated."""
64
64
  out = set()
65
+ short = [k for k in labels if len(k) < 40] # abbreviated labels need a prefix match; full hashes are a lookup
65
66
  for c in maat.in_window(commits, start, end):
66
- paths = next((v for k, v in labels.items() if k.startswith(c["hash"]) or c["hash"].startswith(k)), "none")
67
+ if c["hash"] in labels:
68
+ paths = labels[c["hash"]]
69
+ elif len(c["hash"]) < 40: # an abbreviated hash in the log: scan for the label it abbreviates
70
+ paths = next((v for k, v in labels.items() if k.startswith(c["hash"])), "none")
71
+ else:
72
+ paths = next((labels[k] for k in short if c["hash"].startswith(k)), "none")
67
73
  if paths == "none":
68
74
  continue
69
75
  out.update(p for p in (paths if paths is not None else [p for p, _, _ in c["files"]]) if not filetypes.is_test_path(p))
@@ -151,6 +157,32 @@ def score(report: dict, fixed: set, top: int) -> dict:
151
157
  return out
152
158
 
153
159
 
160
+ def effort(report: dict, outcome: set, top: int) -> dict:
161
+ """variant -> (IFA, lines): how many of its first `top` files come before the first one in the
162
+ outcome (initial false alarms; `top` when none is), and the lines of code those files hold at the
163
+ cut-off, the inspection budget. A list that ranks small files first can look good on hits per line
164
+ and still send a reviewer through many files before one matters, so both are shown."""
165
+ files = (report.get("size") or {}).get("files") or {}
166
+ out = {}
167
+ for name, ranked in variants(report).items():
168
+ head = ranked[:top]
169
+ ifa = next((i for i, f in enumerate(head) if f in outcome), len(head))
170
+ out[name] = (ifa, sum((files.get(f) or {}).get("code", 0) for f in head))
171
+ return out
172
+
173
+
174
+ def effort_table(efforts: list) -> str:
175
+ """efforts: [{variant: (ifa, lines)}] per cut-off -> Markdown with the median of each over the cut-offs."""
176
+ import statistics
177
+ names = list(efforts[0]) if efforts else []
178
+ rows = ["| variant | IFA, median | lines of code in the list, median |", "|---|---:|---:|"]
179
+ for name in names:
180
+ ifas = [e[name][0] for e in efforts]
181
+ lines = [e[name][1] for e in efforts]
182
+ rows.append(f"| {name} | {statistics.median(ifas):g} | {int(statistics.median(lines)):,} |")
183
+ return "\n".join(rows)
184
+
185
+
154
186
  def table(results: list, noun: str = "fixed") -> str:
155
187
  """results: [(t, files fixed that were in the pool, pool size, {variant: hits})], oldest first -> Markdown.
156
188
  `noun` says what the outcome is: fixed, bug-inducing (R-SZZ) or labelled."""
@@ -184,6 +216,7 @@ def main(argv=None) -> int:
184
216
  p.add_argument("--top", type=int, default=watch.WATCH_TOP)
185
217
  p.add_argument("--szz", action="store_true", help="also score against R-SZZ bug-inducing commits (one git blame per fix and file: minutes)")
186
218
  p.add_argument("--labels", metavar="CSV", help="also score against independent bug-inducing labels (ApacheJIT's CSV, Defectors' file rows, or one hash per line)")
219
+ p.add_argument("--end", metavar="YYYY-MM-DD", help="count the cut-offs back from this date rather than the last commit: a labelled dataset that stops earlier (ApacheJIT ends in 2019)")
187
220
  args = p.parse_args(argv)
188
221
  meta = load._read_json(args.out, "meta.json", {})
189
222
  log_path = os.path.join(args.out, "log.txt")
@@ -201,8 +234,8 @@ def main(argv=None) -> int:
201
234
  if args.labels and not labels:
202
235
  print(f"evaluate: no bug-inducing commits read from {args.labels}", file=sys.stderr)
203
236
  return 2
204
- results, induced_results, labelled_results = [], [], []
205
- for t in cutoffs(meta["last_date"], args.windows, args.horizon):
237
+ results, induced_results, labelled_results, labelled_effort = [], [], [], []
238
+ for t in cutoffs(args.end or meta["last_date"], args.windows, args.horizon):
206
239
  rev = trend.rev_before(args.repo, t, end_of_day=False)
207
240
  if not rev:
208
241
  continue # the history does not reach back this far
@@ -220,6 +253,7 @@ def main(argv=None) -> int:
220
253
  if labels:
221
254
  marked = labelled_between(commits, labels, t, end)
222
255
  labelled_results.append((t, len(marked & pool), len(pool), score(report, marked, args.top)))
256
+ labelled_effort.append(effort(report, marked, args.top))
223
257
  print(f"evaluate: {t} done", file=sys.stderr)
224
258
  if not results:
225
259
  print("evaluate: no cut-off falls inside the history", file=sys.stderr)
@@ -233,6 +267,8 @@ def main(argv=None) -> int:
233
267
  if labelled_results:
234
268
  print(f"\nAgainst the files the bug-inducing commits labelled in {os.path.basename(args.labels)} touched inside each window:\n")
235
269
  print(table(labelled_results, noun="labelled"))
270
+ print(f"\nWhat each list costs a reviewer against the same labels: initial false alarms before the first labelled file, and the lines of code in its top {args.top}:\n")
271
+ print(effort_table(labelled_effort))
236
272
  print(f"\n`--all` exports {spread['all']:,} commits ({spread['fix_all']:,} fixes); HEAD reaches {spread['head']:,} ({spread['fix_head']:,} fixes).")
237
273
  return 0
238
274
 
@@ -1009,6 +1009,78 @@ def debt_in_hotspots(report: dict, min_markers: int = 3, min_files: int = 2, top
1009
1009
  evidence={"files": [{"file": f, "markers": n} for f, n in flagged[:10]]})]
1010
1010
 
1011
1011
 
1012
+ def _shape_files(report: dict, key: str) -> list:
1013
+ """[(file, shapes)] for this repository's own source files holding a shape: tests, examples,
1014
+ documentation, vendored and generated files left out, since a fixture or a sample may do on purpose
1015
+ what the source should not."""
1016
+ s = _structure(report)
1017
+ if not s:
1018
+ return []
1019
+ generated, vendored = _generated(report), filetypes.vendor_dirs(report)
1020
+ return [(p, v["shapes"]) for p, v in sorted((s.get("files") or {}).items()) if (v.get("shapes") or {}).get(key)
1021
+ and not (filetypes.is_test_path(p) or filetypes.is_sample_path(p) or filetypes.is_doc_path(p)
1022
+ or filetypes.is_vendored(p, vendored) or p in generated)]
1023
+
1024
+
1025
+ def swallowed_errors(report: dict, min_count: int = 5, top_n: int = 10) -> list:
1026
+ """Catch, except and rescue blocks that do nothing and say nothing: no statement, no comment. In
1027
+ Python only a bare `except:` or one catching Exception or BaseException counts, since `except
1028
+ KeyError: pass` is the language's idiom. A warning when one sits in a top hotspot, where an error
1029
+ that vanishes is the hardest to trace."""
1030
+ rows = _shape_files(report, "empty_catch")
1031
+ total = sum(sh.get("empty_catch_count", len(sh["empty_catch"])) for _, sh in rows)
1032
+ if total < min_count:
1033
+ return []
1034
+ rows.sort(key=lambda r: (-r[1].get("empty_catch_count", 0), r[0]))
1035
+ top = set(_scored_top(report, top_n))
1036
+ hot = [p for p, _ in rows if p in top]
1037
+ listed = _files_list([f"{p}:{sh['empty_catch'][0]}" + (f" and {sh['empty_catch_count'] - 1} more there" if sh.get("empty_catch_count", 1) > 1 else "") for p, sh in rows])
1038
+ bare = sum(sh.get("bare_except_count", 0) for _, sh in rows)
1039
+ first = hot[0] if hot else rows[0][0]
1040
+ return [_f("warning" if hot else "info", "Errors caught and dropped",
1041
+ f"{_plural(total, 'empty catch block')} in {_plural(len(rows), 'source file')}"
1042
+ + (f", {bare} of them a bare except" if bare else "") + f": {listed}."
1043
+ + (f" {textfmt.join_and(hot[:3])} {'is a top hotspot' if len(hot) == 1 else 'are top hotspots'}." if hot else ""),
1044
+ f"Log or rethrow in {first} first, or say in a comment why the error is ignored; an error dropped without a trace is the hardest kind to find.",
1045
+ rule={"id": "swallowed_errors", "min_count": min_count, "measure": "tree-sitter", "python": "bare, Exception or BaseException only"},
1046
+ evidence={"count": total, "bare_except": bare, "hotspots": hot[:10],
1047
+ "files": [{"file": p, "start": sh["empty_catch"][0], "count": sh.get("empty_catch_count", 1)} for p, sh in rows[:10]]})]
1048
+
1049
+
1050
+ def hardcoded_addresses(report: dict) -> list:
1051
+ """IPv4 addresses written into string literals in source files: a host that moves, or an environment
1052
+ wired into the code. Loopback, unspecified, broadcast, netmask-shaped, documentation-range (RFC 5737)
1053
+ and object-identifier-shaped values are not counted (structure.py)."""
1054
+ rows = _shape_files(report, "addresses")
1055
+ if not rows:
1056
+ return []
1057
+ total = sum(sh.get("addresses_count", len(sh["addresses"])) for _, sh in rows)
1058
+ places = [(p, a) for p, sh in rows for a in sh["addresses"]]
1059
+ listed = _files_list([f"{a['value']} at {p}:{a['line']}" for p, a in places])
1060
+ return [_f("info", "Addresses written into the code",
1061
+ f"{_plural(total, 'IPv4 address')} in string literals in {_plural(len(rows), 'source file')}: {listed}.",
1062
+ f"Move {places[0][1]['value']} in {places[0][0]} into configuration, or a name that DNS resolves; an address in code has to be edited and shipped to change.",
1063
+ rule={"id": "hardcoded_addresses", "measure": "tree-sitter", "left_out": "loopback, 0.0.0.0, broadcast, RFC 5737, x.x.x.0, first octet 0-2"},
1064
+ evidence={"count": total, "files": [{"file": p, "start": a["line"], "value": a["value"]} for p, a in places[:10]]})]
1065
+
1066
+
1067
+ def commented_out_code(report: dict, min_lines: int = 10) -> list:
1068
+ """Source files with ten or more lines of code left in comments: blocks of line or block comments in
1069
+ which four lines in five read as statements and one starts right at the comment marker. Worked
1070
+ examples in prose and documentation comments are not counted. The history already keeps old code."""
1071
+ rows = [(p, sh) for p, sh in _shape_files(report, "commented_code") if sh["commented_code"] >= min_lines]
1072
+ if not rows:
1073
+ return []
1074
+ rows.sort(key=lambda r: (-r[1]["commented_code"], r[0]))
1075
+ listed = _files_list([f"{p} ({sh['commented_code']} lines from line {sh['commented_sample'][0]})" for p, sh in rows])
1076
+ first = rows[0]
1077
+ return [_f("info", "Code left in comments",
1078
+ f"{_plural(len(rows), 'source file')} {'holds' if len(rows) == 1 else 'hold'} {min_lines} or more lines of commented-out code: {listed}.",
1079
+ f"Delete the block at {first[0]}:{first[1]['commented_sample'][0]}; git keeps the old version, and a reader cannot tell whether it is meant to come back.",
1080
+ rule={"id": "commented_out_code", "min_lines": min_lines, "measure": "tree-sitter", "code_share": 0.8},
1081
+ evidence={"files": [{"file": p, "start": sh["commented_sample"][0], "lines": sh["commented_code"]} for p, sh in rows[:10]]})]
1082
+
1083
+
1012
1084
  def deep_nesting(report: dict, min_nesting: int = 5, min_bumps: int = 3, top_n: int = 10) -> list:
1013
1085
  """Functions nested five levels or more, or with three or more separate chunks of nested logic (a
1014
1086
  bumpy road), in this repository's own source: CodeScene's nesting and bumpy-road factors, measured
@@ -1270,7 +1342,7 @@ RULES = [dormant, secrets_found, credential_files, vulnerable_dependencies, plac
1270
1342
  minor_contributors, reverts, brain_methods, complexity_growth, tight_coupling, duplication, stale_files, knowledge_islands, knowledge_loss,
1271
1343
  sweeping_commits, tangled_commits, hygiene_findings, debt_in_hotspots, deep_nesting, hidden_coupling, unreferenced_files,
1272
1344
  agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author,
1273
- truck_factor, authors_gone, component_coupling]
1345
+ truck_factor, authors_gone, component_coupling, swallowed_errors, hardcoded_addresses, commented_out_code]
1274
1346
 
1275
1347
 
1276
1348
  def evaluate(report: dict) -> list:
@@ -12,6 +12,12 @@ writes provenance.json:
12
12
  changed again by another commit within two weeks. This repository against itself, with the share
13
13
  of commits the cohort covers beside it; no prior from elsewhere, since the best-controlled study
14
14
  found the spread between agents larger than the pooled difference.
15
+ - lines: the lines added to code files in the last year and the year before, with the share git's
16
+ own moved-code detection marks as moved (`--color-moved`, blocks of twenty or more characters) and
17
+ the share deleted again within two weeks, in the same file with the same text: GitClear's moved and
18
+ churned lines, as a direction over this repository rather than a comparison with anyone else's.
19
+ The same two numbers per cohort, and each cohort's watch-list hit rate: the share of its commits
20
+ touching a file on the watch list's top fifteen.
15
21
  - shape: neutral descriptors (commits landing in bursts, conventional-commit subjects, how many hours
16
22
  of the day commits come in). Every one has a benign cause, and none is labelled.
17
23
  - agents: the agent instruction files by path convention (AGENTS.md, CLAUDE.md, GEMINI.md,
@@ -28,7 +34,7 @@ import os
28
34
  import re
29
35
  import subprocess
30
36
  import sys
31
- from collections import Counter
37
+ from collections import Counter, deque
32
38
 
33
39
  try:
34
40
  from . import filetypes, leaks
@@ -44,6 +50,16 @@ _IDENT = re.compile(r"^\s*(?P<name>[^<]*?)\s*<(?P<email>[^>]*)>\s*$")
44
50
  _CONVENTIONAL = re.compile(r"^(feat|fix|docs|style|refactor|perf|test|build|ci|chore|revert)(\([^)]*\))?!?: \S")
45
51
  BURST_SIZE, BURST_SECONDS = 5, 600
46
52
  RETOUCH_DAYS = 14
53
+ YEAR = 365 * 86400
54
+ WATCH_TOP = 15
55
+ # data that inflates line counts without being code (the same set the code-age pass leaves out)
56
+ DATA_EXCLUDES = [":(exclude,glob)**/*.json", ":(exclude,glob)**/*.lock", ":(exclude,glob)**/*.min.js", ":(exclude,glob)**/*.min.css",
57
+ ":(exclude,glob)**/*.svg", ":(exclude,glob)**/*.map", ":(exclude,glob)**/*.csv", ":(exclude,glob)**/*.snap"]
58
+ _COLORS = ["-c", "color.diff.new=green", "-c", "color.diff.newMoved=cyan", "-c", "color.diff.old=red", "-c", "color.diff.oldMoved=magenta",
59
+ "-c", "color.diff.meta=normal", "-c", "color.diff.frag=normal", "-c", "color.diff.func=normal", "-c", "color.diff.context=normal",
60
+ "-c", "color.diff.whitespace=normal", "-c", "color.diff.commit=normal"]
61
+ _ADDED, _MOVED, _DELETED, _DELETED_MOVED = "\x1b[32m+", "\x1b[36m+", "\x1b[31m-", "\x1b[35m-"
62
+ _ANSI = re.compile(r"\x1b\[[0-9;]*m")
47
63
 
48
64
 
49
65
  def read_commits(repo: str) -> list:
@@ -94,8 +110,8 @@ def trailers(commits: list) -> dict:
94
110
  "with_any": sum(1 for c in commits if c["trailers"]), "never_author": listing[:50], "signoff_by_co_author": signoff[:50]}
95
111
 
96
112
 
97
- def cohort(commits: list, inventory: dict) -> dict:
98
- """The commits an `Assisted-by` trailer or a never-authoring co-author marks, against the rest."""
113
+ def marker(inventory: dict):
114
+ """The predicate that marks a commit: an `Assisted-by` trailer, or a co-author who never authors."""
99
115
  marked_emails = {x["email"] for x in inventory["never_author"]}
100
116
 
101
117
  def marked(c):
@@ -106,6 +122,13 @@ def cohort(commits: list, inventory: dict) -> dict:
106
122
  if k.lower() == "co-authored-by" and ident and ident[1] in marked_emails:
107
123
  return True
108
124
  return False
125
+ return marked
126
+
127
+
128
+ def cohort(commits: list, inventory: dict, watch_files=None) -> dict:
129
+ """The commits an `Assisted-by` trailer or a never-authoring co-author marks, against the rest; with
130
+ `watch_files`, how many of each touched a file on the watch list."""
131
+ marked = marker(inventory)
109
132
 
110
133
  reverted = {c["subject"][len('Revert "'):-1] for c in commits if c["subject"].startswith('Revert "') and c["subject"].endswith('"')}
111
134
  by_file = {}
@@ -125,12 +148,107 @@ def cohort(commits: list, inventory: dict) -> dict:
125
148
  again = True
126
149
  break
127
150
  s["retouched"] += again
151
+ if watch_files:
152
+ s["watch"] += any(f in watch_files for f in c["files"])
128
153
  total = len(commits)
129
154
  return {"definition": "an Assisted-by trailer, or a co-author who never authors a commit here",
130
155
  "share": round(stats[True]["commits"] / total, 3) if total else 0.0,
131
156
  "cohort": dict(stats[True]) or {"commits": 0}, "rest": dict(stats[False]) or {"commits": 0}}
132
157
 
133
158
 
159
+ def _code_path(path: str, generated: set, vendored) -> bool:
160
+ return filetypes.matches(path, filetypes.DEFAULT) and path not in generated and not filetypes.is_vendored(path, vendored)
161
+
162
+
163
+ def lines(repo: str, end: int, marked_hashes: set, generated=frozenset(), vendored=()) -> dict:
164
+ """Added, moved and churned lines in code files over the two years before `end` (a timestamp), from
165
+ one `git log -p` with git's moved-code colouring. A line is churned when a later commit, within two
166
+ weeks, deletes a line with the same text from the same file; blank lines and lines without three
167
+ letters or digits (a lone brace) are not matched, since any brace would pair with any other."""
168
+ start = end - 2 * YEAR
169
+ argv = ["git", *_COLORS, "-c", "core.quotePath=false", "log", "HEAD", "--reverse", "--no-merges", "-p", "-U0", "-M", "--color=always",
170
+ "--color-moved=blocks", "--color-moved-ws=allow-indentation-change", f"--since=@{start}", f"--until=@{end}",
171
+ f"--format={END}%H{SEP}%at", "--", ".", *DATA_EXCLUDES]
172
+ proc = subprocess.Popen(argv, cwd=repo, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL)
173
+ windows = {w: Counter() for w in ("last", "before")}
174
+ groups = {True: Counter(), False: Counter()}
175
+ pending = {} # (path, text) -> [(time, window, marked)] of additions not yet churned
176
+ order = deque() # (time, key) in the order added, so what is past two weeks can be dropped and memory stays bounded
177
+ t, window, is_marked, path, keep = 0, "last", False, None, False
178
+ for raw in proc.stdout:
179
+ line = raw.decode("utf-8", "replace").rstrip("\n")
180
+ if line.startswith(END):
181
+ h, _, at = line[1:].partition(SEP)
182
+ t = int(at or 0)
183
+ window = "last" if t > end - YEAR else "before"
184
+ is_marked = h in marked_hashes
185
+ windows[window]["commits"] += 1
186
+ groups[is_marked]["commits"] += 1
187
+ path = None
188
+ while order and t - order[0][0] > RETOUCH_DAYS * 86400:
189
+ old_t, key = order.popleft()
190
+ adds = pending.get(key)
191
+ if adds and adds[0][0] == old_t:
192
+ adds.pop(0)
193
+ if not adds:
194
+ pending.pop(key, None)
195
+ continue
196
+ if line.startswith("diff --git "):
197
+ plain = _ANSI.sub("", line) # git ends even an uncoloured header with a reset
198
+ path = filetypes.unquote(plain.rsplit(" b/", 1)[-1]) if " b/" in plain else None
199
+ keep = bool(path) and _code_path(path, generated, vendored)
200
+ continue
201
+ if not keep:
202
+ continue
203
+ if line.startswith((_ADDED, _MOVED)):
204
+ text = _ANSI.sub("", line)[1:].strip()
205
+ moved = line.startswith(_MOVED)
206
+ for c in (windows[window], groups[is_marked]):
207
+ c["added"] += 1
208
+ c["moved"] += moved
209
+ if sum(ch.isalnum() for ch in text) >= 3:
210
+ pending.setdefault((path, text), []).append((t, window, is_marked))
211
+ order.append((t, (path, text)))
212
+ elif line.startswith((_DELETED, _DELETED_MOVED)):
213
+ text = _ANSI.sub("", line)[1:].strip()
214
+ adds = pending.get((path, text))
215
+ while adds and t - adds[0][0] > RETOUCH_DAYS * 86400:
216
+ adds.pop(0) # too old to count, and older than any later deletion will reach
217
+ if adds and adds[0][0] < t:
218
+ at_, w, m = adds.pop(0)
219
+ windows[w]["churned"] += 1
220
+ groups[m]["churned"] += 1
221
+ proc.stdout.close()
222
+ proc.wait()
223
+
224
+ def shares(c):
225
+ added = c.get("added", 0)
226
+ return {"commits": c.get("commits", 0), "added": added, "moved": c.get("moved", 0), "churned": c.get("churned", 0),
227
+ "moved_share": round(c.get("moved", 0) / added, 4) if added else None,
228
+ "churn_share": round(c.get("churned", 0) / added, 4) if added else None}
229
+ day = lambda ts: dt.datetime.fromtimestamp(ts, dt.timezone.utc).date().isoformat() # noqa: E731
230
+ return {"windows": [{"label": "last year", "from": day(end - YEAR), "to": day(end), **shares(windows["last"])},
231
+ {"label": "the year before", "from": day(start), "to": day(end - YEAR), **shares(windows["before"])}],
232
+ "cohort": {"marked": shares(groups[True]), "rest": shares(groups[False])},
233
+ "churn_days": RETOUCH_DAYS, "moved": "git --color-moved=blocks"}
234
+
235
+
236
+ def _watch_files(out_dir: str):
237
+ """The watch list's top files, when the change analysis and scc have written their outputs; None
238
+ otherwise, or when this runs as a script outside the package."""
239
+ if not (os.path.exists(os.path.join(out_dir, "size.json")) and os.path.exists(os.path.join(out_dir, "maat-revisions.csv"))):
240
+ return None
241
+ try:
242
+ from . import load, watch
243
+ except ImportError:
244
+ return None
245
+ try:
246
+ report = load.load_report(out_dir, nested=False)
247
+ except load.Unreadable:
248
+ return None
249
+ return {r["file"] for r in watch.risks(report)[:WATCH_TOP]}
250
+
251
+
134
252
  def shape(commits: list) -> dict:
135
253
  """How commits arrive, as neutral numbers: the share landing in runs of five or more by one author
136
254
  within ten minutes of each other, the share with conventional-commit subjects, and how many hours
@@ -231,7 +349,20 @@ def main(argv=None) -> int:
231
349
  print(f"provenance.py: {(e.stderr or b'').decode('utf-8', 'replace').strip() or e}", file=sys.stderr)
232
350
  return 1
233
351
  inventory = trailers(commits)
234
- result = {"trailers": inventory, "cohort": cohort(commits, inventory), "shape": shape(commits), "agents": agents(repo)}
352
+ watch_files = _watch_files(args[0])
353
+ meta = {}
354
+ try:
355
+ with open(os.path.join(args[0], "meta.json"), encoding="utf-8") as fh:
356
+ meta = json.load(fh)
357
+ except (OSError, ValueError):
358
+ pass
359
+ marked = marker(inventory)
360
+ result = {"trailers": inventory, "cohort": cohort(commits, inventory, watch_files), "shape": shape(commits), "agents": agents(repo)}
361
+ if watch_files is not None:
362
+ result["cohort"]["watch_top"] = WATCH_TOP
363
+ if commits:
364
+ result["lines"] = lines(repo, commits[-1]["time"], {c["hash"] for c in commits if marked(c)}, set(meta.get("generated") or []),
365
+ filetypes.vendor_dirs({"meta": meta}))
235
366
  with open(os.path.join(args[0], "provenance.json"), "w", encoding="utf-8") as fh:
236
367
  json.dump(result, fh)
237
368
  return 0
@@ -548,14 +548,35 @@ def trailers_section(report: dict, full: bool = True, width=None) -> dict:
548
548
  if marked.get("commits"):
549
549
  def pair(key):
550
550
  return f"{_pct(marked.get(key, 0), marked['commits'])} against {_pct(rest.get(key, 0), rest.get('commits') or 0)}"
551
+ watch_part = f", touched a file on the watch list's top {co['watch_top']} {pair('watch')}" if co.get("watch_top") else ""
551
552
  notes.append(f"marked commits ({co.get('definition')}): {marked['commits']:,}, {round(100 * co.get('share', 0))}% of the history; "
552
- f"reverted {pair('reverted')} for the rest, fixes {pair('fixes')}, a file changed again within two weeks {pair('retouched')}")
553
+ f"reverted {pair('reverted')} for the rest, fixes {pair('fixes')}, a file changed again within two weeks {pair('retouched')}{watch_part}")
553
554
  if sh:
554
555
  notes.append(f"{round(100 * sh.get('burst_share', 0))}% of commits land in bursts of five or more within ten minutes; "
555
556
  f"{round(100 * sh.get('conventional_share', 0))}% have conventional-commit subjects; commits come in {sh.get('hours_used', 0)} hours of the day")
556
557
  return _section("Trailers", columns, rows, note=None if rows else "no trailers", caption="\n".join(notes) or None)
557
558
 
558
559
 
560
+ def lines_section(report: dict, full: bool = True, width=None) -> dict:
561
+ """Lines added to code files in the last year and the year before, the share git marks as moved and
562
+ the share deleted again within two weeks, and the same for the marked cohort against the rest:
563
+ --full and Markdown only. A direction for this repository, not a score."""
564
+ ln = (report.get("provenance") or {}).get("lines") or {}
565
+
566
+ def share(x):
567
+ return "-" if x is None else f"{100 * x:.1f}%"
568
+ rows = [(f"{w['label']} ({w['from']} to {w['to']})", w["commits"], w["added"], share(w.get("moved_share")), share(w.get("churn_share")))
569
+ for w in ln.get("windows") or []]
570
+ co = ln.get("cohort") or {}
571
+ if (co.get("marked") or {}).get("commits"):
572
+ rows += [(label, c["commits"], c["added"], share(c.get("moved_share")), share(c.get("churn_share")))
573
+ for label, c in (("marked commits, both years", co["marked"]), ("the rest, both years", co["rest"]))]
574
+ columns = [("period", {"overflow": "fold"}), ("commits", RIGHT), ("lines added", RIGHT), ("moved", RIGHT), (f"churned in {ln.get('churn_days', 14)} days", RIGHT)]
575
+ return _section("Changed lines", columns, rows, note=None if rows else "no history in the last two years",
576
+ caption="code files only; moved: lines git's moved-code detection marks (--color-moved=blocks); churned: deleted again "
577
+ "within two weeks from the same file with the same text" if rows else None)
578
+
579
+
559
580
  def hotspots_section(report: dict, full: bool = True, width=None) -> dict:
560
581
  """Change frequency times size, Tornhill-style. Files no longer in the tree sort last. Drawn
561
582
  under `--full` and in the Markdown export only; the default terminal report leaves it to the
@@ -791,10 +812,10 @@ def compare_section(result: dict) -> dict:
791
812
 
792
813
 
793
814
  BUILDERS = [watch_section, watch_by_component_section, size_section, people_section, knowledge_section, activity_section, timeline_section,
794
- hotspots_section, coupling_section, signing_section, trailers_section, age_section, functions_section, health_section, osps_section]
815
+ hotspots_section, coupling_section, signing_section, trailers_section, lines_section, age_section, functions_section, health_section, osps_section]
795
816
  # `--full` and Markdown only: Size, Activity and Code age are interesting once and rarely change what you
796
817
  # do next; Hotspots ranks the files the watch list already leads with, by the same product.
797
- FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers", "watch_by_component", "osps"}
818
+ FULL_ONLY = {"size", "activity", "age", "hotspots", "signing", "trailers", "lines", "watch_by_component", "osps"}
798
819
 
799
820
 
800
821
  def sections(report: dict, full: bool = True, width=None) -> list:
@@ -263,7 +263,7 @@ def plan(repo_dir: str, out_dir: str, branch: str = "HEAD", age: bool = True, pl
263
263
  {"name": "change analysis", "argv": [sys.executable, MAAT_SCRIPT, log, out_dir, *type_args, *(["--now", now] if now else []), *(["--since", since] if since else []), "--aliases", o("meta.json"), *revs_args], "stdout": None, "deps": ["git-log"]},
264
264
  {"name": "signing", "argv": [sys.executable, "-m", "gitmole.signing", out_dir], "stdout": None, "deps": []}, # the gpgsig headers, no keyring
265
265
  {"name": "hygiene", "argv": [sys.executable, "-m", "gitmole.hygiene", out_dir], "stdout": None, "deps": []}, # the Scorecard checks, from the clone
266
- {"name": "provenance", "argv": [sys.executable, "-m", "gitmole.provenance", out_dir], "stdout": None, "deps": []}, # trailers, cohorts, agent files
266
+ {"name": "provenance", "argv": [sys.executable, "-m", "gitmole.provenance", out_dir], "stdout": None, "deps": ["scc", "change analysis"]}, # trailers, cohorts, agent files; the watch list for the hit rate
267
267
  ]
268
268
  workers = procs or blame.default_procs()
269
269
  if lizard:
@@ -36,7 +36,7 @@ try:
36
36
  except ImportError: # run as a script: the package directory is sys.path[0]
37
37
  import filetypes
38
38
 
39
- ANALYSER = "2" # bump whenever what a file yields changes (a metric, an import's shape): the cache key carries it
39
+ ANALYSER = "3" # bump whenever what a file yields changes (a metric, an import's shape): the cache key carries it
40
40
  MAX_BYTES = 1_000_000
41
41
  FUNCTIONS_KEPT = 3000
42
42
 
@@ -73,6 +73,89 @@ NO_INCREMENT = {"try_statement"} # the try nests its body; the catch is what S
73
73
  LOGICAL = {"&&", "||", "and", "or"}
74
74
  DEBT = re.compile(r"\b(TODO|FIXME|XXX|HACK)\b")
75
75
 
76
+ # --- shapes: an error swallowed, an address in a literal, code left in a comment -----------------
77
+ CATCH = {"catch_clause", "except_clause", "rescue"}
78
+ BODY = {"block", "statement_block", "compound_statement", "then"}
79
+ EMPTY_STATEMENTS = {"pass_statement", "empty_statement"}
80
+ STRING = {"string", "string_literal", "interpreted_string_literal", "raw_string_literal", "template_string", "encapsed_string"}
81
+ ATTRIBUTE = {"attribute", "attribute_list", "annotation", "marker_annotation", "decorator", "attribute_item"}
82
+ _QUOTED = re.compile(r"""^[A-Za-z@$]*(["'`]+)(.*?)\1$""", re.S)
83
+ _IPV4 = re.compile(r"^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})(?::\d{1,5})?$")
84
+ # a comment line that reads as a statement: an assignment, a call, a keyword that opens one, a brace
85
+ _CODE_LINE = re.compile(r"""^(?:(?:return|if|elif|else|for|while|import|from|var|let|const|def|function|class|#include|throw|raise|break|continue|await)\b.*[;:{})\]]|[A-Za-z_$][\w.$\[\]]*\s*(?:=|\+=|-=|\|=|:=)\s*\S.*|[\w.$]+(?:\.[\w$]+)*\(.*\)\s*;?|[{}]\s*[;)]?|.*[;{]|\}\s*else\b.*)$""")
86
+ _COMMENT_MARK = re.compile(r"^\s*(?://+|/\*+|\*+/?|#+|--|;+) ?")
87
+ CODE_SHARE = 0.8 # of a comment block's lines that read as code, for the block to count as code left in a comment
88
+ _DIRECTIVE = re.compile(r"^(?:eslint|prettier|istanbul|noqa|type:|pylint|@ts-|jshint|global |c8 |nolint|NOLINT|clang-format|fmt:|pragma|region|endregion|-\*-|SPDX-|Copyright|http)", re.I)
89
+
90
+
91
+ BROAD_PYTHON = {"Exception", "BaseException"} # the language's own root classes
92
+
93
+
94
+ def _is_empty_catch(node, src: bytes = b"") -> bool:
95
+ """A catch, except or rescue whose body does nothing and says nothing: no statement but `pass`, and
96
+ no comment, since a comment is the author saying the error is ignored on purpose (Sonar's S108).
97
+ Python's `except SomeError: pass` is the language's idiom for an expected failure (EAFP), so there
98
+ only a bare `except:` or one that catches Exception or BaseException counts."""
99
+ if node.type == "except_clause":
100
+ caught = [c for c in node.named_children if c.type not in BODY and c.type != "comment"]
101
+ if caught and not all(_text(src, c).strip("() ") in BROAD_PYTHON or _text(src, c).split(" as ")[0].strip("() ") in BROAD_PYTHON for c in caught):
102
+ return False
103
+ body = None
104
+ for c in node.named_children:
105
+ if c.type == "comment":
106
+ return False
107
+ if c.type in BODY:
108
+ body = c
109
+ if body is None:
110
+ return node.type == "rescue" # Ruby: an empty rescue has no `then` at all
111
+ for c in body.named_children:
112
+ if c.type == "comment" or c.type not in EMPTY_STATEMENTS:
113
+ return False
114
+ return True
115
+
116
+
117
+ def _address(text: str):
118
+ """The IPv4 address a string literal is, with its port, or None: loopback, the unspecified and
119
+ broadcast addresses, netmasks, the documentation ranges (RFC 5737), a trailing .0 (a network, or a
120
+ four-part version like 1.0.0.0), and a first octet of 0, 1 or 2, which is how an ASN.1 object
121
+ identifier (2.5.4.3) starts, are left out."""
122
+ m = _QUOTED.match(text.strip())
123
+ value = (m.group(2) if m else text).strip()
124
+ ip = _IPV4.match(value)
125
+ if not ip:
126
+ return None
127
+ octets = [int(x) for x in ip.groups()]
128
+ if any(o > 255 for o in octets) or octets[0] <= 2 or octets[0] in (127, 255) or octets[3] in (0, 255): # 0-2: an object identifier's first arc
129
+ return None
130
+ if octets[:3] in ([192, 0, 2], [198, 51, 100], [203, 0, 113]):
131
+ return None
132
+ return value
133
+
134
+
135
+ def _code_like(line: str) -> bool:
136
+ s = line.strip()
137
+ return bool(s) and not _DIRECTIVE.match(s) and not s.endswith(".") and bool(_CODE_LINE.match(s)) and not re.match(r"^[A-Za-z]+(?: [a-z]+){3,}", s)
138
+
139
+
140
+ def commented_code_lines(text: str) -> int:
141
+ """How many lines of one comment block (a block comment, or a run of line comments on consecutive
142
+ lines) read as code; 0 unless four in five of its lines do and one of them starts right after the
143
+ comment marker, as an editor's comment-out leaves it. Prose with a worked example under it (indented,
144
+ or introduced by a line ending in a colon), and documentation comments, whose examples are meant to
145
+ be there, are not code left behind."""
146
+ if text.lstrip().startswith(("/**", "///", "//!", "#!", "/*!")):
147
+ return 0
148
+ lines = [re.sub(r"\s*\*/\s*$", "", _COMMENT_MARK.sub("", l)) for l in text.split("\n")]
149
+ lines = [l.rstrip() for l in lines if l.strip() and l.strip() not in ("*/", "*")]
150
+ if not lines:
151
+ return 0
152
+ code = [l for l in lines if _code_like(l)]
153
+ if len(code) < CODE_SHARE * len(lines) or not any(not l[:1].isspace() for l in code):
154
+ return 0
155
+ if any(l.strip().endswith(":") and not _code_like(l) for l in lines):
156
+ return 0 # "Build the dispatcher function:" introduces an example
157
+ return len(code)
158
+
76
159
 
77
160
  def cache_root() -> str | None:
78
161
  explicit = os.environ.get("GITMOLE_CACHE")
@@ -217,6 +300,17 @@ def analyse(src: bytes, lang_name: str, language) -> dict:
217
300
  funcs, stack, done = [], [], [] # stack: one frame per ancestor on the way down
218
301
  comments = comment_lines = definitions = 0
219
302
  debt, imports = [], []
303
+ shapes = {"empty_catch": [], "bare_except": [], "addresses": [], "commented_code": 0, "commented_sample": []}
304
+ block = [] # the open comment block: [first line, last line, text, made of line comments]
305
+
306
+ def flush():
307
+ if block:
308
+ code = commented_code_lines(block[2])
309
+ if code:
310
+ shapes["commented_code"] += code
311
+ if len(shapes["commented_sample"]) < 5:
312
+ shapes["commented_sample"].append(block[0] + 1)
313
+ block.clear()
220
314
  main_guard = False
221
315
  chains = [] # open runs of logical operators: [count]
222
316
  while True:
@@ -263,6 +357,22 @@ def analyse(src: bytes, lang_name: str, language) -> dict:
263
357
  m = DEBT.search(text)
264
358
  if m:
265
359
  debt.append({"line": node.start_point[0] + 1, "tag": m.group(1), "text": " ".join(text.split())[:100]})
360
+ own_line = not src[src.rfind(b"\n", 0, node.start_byte) + 1:node.start_byte].strip() # not trailing code
361
+ line_comment = text.startswith(("//", "#", "--")) and "\n" not in text.rstrip()
362
+ if block and line_comment and block[3] and own_line and node.start_point[0] == block[1] + 1:
363
+ block[1], block[2] = node.end_point[0], block[2] + "\n" + text
364
+ else:
365
+ flush()
366
+ if own_line:
367
+ block.extend([node.start_point[0], node.end_point[0], text, line_comment])
368
+ elif t in CATCH and _is_empty_catch(node, src):
369
+ shapes["empty_catch"].append(node.start_point[0] + 1)
370
+ if t == "except_clause" and not any(c.type not in BODY and c.type != "comment" for c in node.named_children):
371
+ shapes["bare_except"].append(node.start_point[0] + 1)
372
+ elif t in STRING and node.end_byte - node.start_byte <= 30 and not any(s[0] in ATTRIBUTE for s in stack):
373
+ value = _address(_text(src, node))
374
+ if value:
375
+ shapes["addresses"].append({"line": node.start_point[0] + 1, "value": value})
266
376
  found = _import(node, src, lang_name)
267
377
  if found:
268
378
  imports.extend(found)
@@ -278,7 +388,8 @@ def analyse(src: bytes, lang_name: str, language) -> dict:
278
388
  if cursor.goto_next_sibling():
279
389
  break
280
390
  if not cursor.goto_parent():
281
- return _result(done, comments, comment_lines, definitions, debt, imports, main_guard, src, tree)
391
+ flush()
392
+ return _result(done, comments, comment_lines, definitions, debt, imports, main_guard, src, tree, shapes)
282
393
 
283
394
 
284
395
  def _in_chain(node) -> bool:
@@ -304,10 +415,10 @@ def _leave(frame, funcs, done, chains):
304
415
  done.append(funcs.pop())
305
416
 
306
417
 
307
- def _result(done, comments, comment_lines, definitions, debt, imports, main_guard, src, tree) -> dict:
418
+ def _result(done, comments, comment_lines, definitions, debt, imports, main_guard, src, tree, shapes) -> dict:
308
419
  lines = src.count(b"\n") + (1 if src and not src.endswith(b"\n") else 0)
309
420
  return {"lines": lines, "comments": comments, "comment_lines": comment_lines, "definitions": definitions,
310
- "debt": debt, "imports": imports, "main": main_guard or src.startswith(b"#!"), "errors": tree.root_node.has_error,
421
+ "debt": debt, "imports": imports, "main": main_guard or src.startswith(b"#!"), "errors": tree.root_node.has_error, "shapes": shapes,
311
422
  "functions": [{"name": f.name, "start": f.start, "end": f.end, "nesting": f.max_nesting, "cognitive": f.cognitive,
312
423
  "complex_conditions": f.complex, "bumps": f.bumps} for f in done]}
313
424
 
@@ -518,6 +629,23 @@ def unreferenced(files: dict, edges: dict, resolved: dict, entries: set) -> list
518
629
  return [p for p in out if files[p]["language"] not in loud]
519
630
 
520
631
 
632
+ def _slim_shapes(s: dict) -> dict:
633
+ """The shapes one file holds, as structure.json keeps them: counts and the first few lines; empty
634
+ lists and zero counts are left out so a clean file costs nothing."""
635
+ out = {}
636
+ for key in ("empty_catch", "bare_except"):
637
+ if s.get(key):
638
+ out[key] = s[key][:5]
639
+ out[key + "_count"] = len(s[key])
640
+ if s.get("addresses"):
641
+ out["addresses"] = s["addresses"][:5]
642
+ out["addresses_count"] = len(s["addresses"])
643
+ if s.get("commented_code"):
644
+ out["commented_code"] = s["commented_code"]
645
+ out["commented_sample"] = s.get("commented_sample") or []
646
+ return out
647
+
648
+
521
649
  def entries_dirs(entries: set) -> set:
522
650
  return {e for e in entries if e.endswith("package.json")}
523
651
 
@@ -576,6 +704,7 @@ def collect(repo: str, procs: int = None, vendored=()) -> dict:
576
704
  slim = {p: {"language": v["language"], "lines": v.get("lines", 0), "comments": v.get("comments", 0), "comment_lines": v.get("comment_lines", 0),
577
705
  "definitions": v.get("definitions", 0), "debt": len(v.get("debt") or []), "debt_sample": (v.get("debt") or [])[:5],
578
706
  "main": v.get("main", False), "imports": edges.get(p, []), "errors": v.get("errors", False),
707
+ "shapes": _slim_shapes(v.get("shapes") or {}),
579
708
  "max_nesting": max((f["nesting"] for f in v.get("functions") or []), default=0),
580
709
  "max_cognitive": max((f["cognitive"] for f in v.get("functions") or []), default=0)}
581
710
  for p, v in files.items() if "failed" not in v}
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.20.0
3
+ Version: 0.22.0
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -74,6 +74,7 @@ gitmole . --json report.json # every table, the watch list and the fin
74
74
  gitmole . --fail-on warning # exit 3 if any finding is a warning or worse
75
75
  gitmole . --risk main --risk-threshold 10 # exit 3 if the files changed since main hold over 10% of the risk
76
76
  gitmole . --sarif gitmole.sarif # the findings for GitHub code scanning or GitLab
77
+ gitmole . --sbom sbom.cdx.json # a CycloneDX SBOM of every package the lock files pin
77
78
  gitmole . --compare last.json # what changed since an earlier --json export
78
79
  gitmole analysis-repo --no-run --hook # a coding agent's edit hook: history's view of the files it just touched
79
80
  gitmole . --since 2y --full # the current team, every row and column
@@ -92,9 +93,9 @@ Reports on repositories you know, each at a pinned commit, published as gitmole
92
93
 
93
94
  | Repository | Commit | Commits | Lines | gitmole run |
94
95
  |---|---|---:|---:|---:|
95
- | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 57 s |
96
- | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 132 s |
97
- | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 151 s |
96
+ | [curl](https://github.com/antvinni/gitmole/blob/main/docs/examples/curl.md) | [`540ee5b5`](https://github.com/curl/curl/commit/540ee5b560cc6e775e11317048a13cc7e355bf91) | 39,758 | 247,179 | 59 s |
97
+ | [django](https://github.com/antvinni/gitmole/blob/main/docs/examples/django.md) | [`8cbdd4a8`](https://github.com/django/django/commit/8cbdd4a814397f81adf0129288f32b615bd1f94f) | 34,933 | 431,749 | 135 s |
98
+ | [react](https://github.com/antvinni/gitmole/blob/main/docs/examples/react.md) | [`2b19aecd`](https://github.com/facebook/react/commit/2b19aecd0e9111b774fad0fad9862e50bcb5bc8a) | 21,703 | 681,078 | 157 s |
98
99
 
99
100
  Run times are one `gitmole CLONE` with every default step, on a MacBook Pro (M4, 16 GB).
100
101
 
@@ -77,6 +77,7 @@ tests/test_render.py
77
77
  tests/test_render_examples.py
78
78
  tests/test_run.py
79
79
  tests/test_sarif.py
80
+ tests/test_shapes.py
80
81
  tests/test_signing.py
81
82
  tests/test_structure.py
82
83
  tests/test_szz.py
@@ -70,6 +70,16 @@ class Induced(unittest.TestCase):
70
70
 
71
71
 
72
72
  class Score(unittest.TestCase):
73
+ def test_effort_is_the_false_alarms_before_the_first_hit_and_the_lines_read(self):
74
+ r = evaluate.report_at(COMMITS, "2025-06-01", SIZE, {"bots": []}, [], [])
75
+ ranked = evaluate.variants(r)["watch list (hotspot)"]
76
+ out = evaluate.effort(r, {ranked[1]}, 15)["watch list (hotspot)"]
77
+ self.assertEqual(out[0], 1, "one file before the first labelled one")
78
+ self.assertEqual(out[1], sum(SIZE["files"][f]["code"] for f in ranked[:15]))
79
+ self.assertEqual(evaluate.effort(r, set(), 15)["watch list (hotspot)"][0], len(ranked[:15]), "no hit: every file was a false alarm")
80
+ table = evaluate.effort_table([{"a": (1, 100)}, {"a": (3, 300)}])
81
+ self.assertIn("| a | 2 | 200 |", table)
82
+
73
83
  def test_the_report_at_t_knows_nothing_after_t(self):
74
84
  r = evaluate.report_at(COMMITS, "2025-06-01", SIZE, {"bots": []}, [], [])
75
85
  self.assertEqual({x["entity"]: x["n-revs"] for x in r["revisions"]}, {"core/a.py": 3, "core/b.py": 2, "tests/test_a.py": 1})
@@ -110,6 +110,46 @@ class Agents(unittest.TestCase):
110
110
  self.assertNotIn("p4ss", json.dumps(out))
111
111
 
112
112
 
113
+ class Lines(unittest.TestCase):
114
+ def _write(self, r, path, text, message, date):
115
+ with open(os.path.join(r.d, path), "w") as fh:
116
+ fh.write(text)
117
+ r.git("add", "-A", date=date)
118
+ r.git("commit", "-q", "-m", message, date=date)
119
+
120
+ def test_moved_and_churned_lines_by_year_and_cohort(self):
121
+ body = "".join(f"def function_number_{i}(argument):\n return argument * {i} + compute_offset({i})\n" for i in range(6))
122
+ with tempfile.TemporaryDirectory() as d:
123
+ r = Repo(d)
124
+ self._write(r, "c.py", "unrelated_value = compute_something()\n", "start", "2024-06-01T10:00:00") # the year before
125
+ self._write(r, "a.py", body + "keep_this_line = 1\ntemporary_line = 2\n", "add a", "2025-06-01T10:00:00")
126
+ self._write(r, "a.py", body + "keep_this_line = 1\n", "drop the temporary line\n\nAssisted-by: Tool", "2025-06-05T10:00:00")
127
+ with open(os.path.join(d, "c.py"), "a") as fh:
128
+ fh.write(body)
129
+ self._write(r, "a.py", "keep_this_line = 1\n", "move the functions into c.py", "2025-07-01T10:00:00")
130
+ commits = provenance.read_commits(d)
131
+ marked = provenance.marker(provenance.trailers(commits))
132
+ out = provenance.lines(d, commits[-1]["time"], {c["hash"] for c in commits if marked(c)})
133
+ last, before = out["windows"]
134
+ self.assertEqual((before["commits"], before["added"]), (1, 1))
135
+ self.assertEqual(last["added"], len(body.splitlines()) * 2 + 2)
136
+ self.assertEqual(last["churned"], 1, "the temporary line went within four days; the kept line and the moved functions are not churn")
137
+ self.assertEqual(last["moved"], len(body.splitlines()), "git marks the functions as moved from a.py to b.py")
138
+ self.assertEqual(out["cohort"]["marked"]["commits"], 1)
139
+ self.assertEqual(out["cohort"]["rest"]["churned"], 1, "the churn belongs to the commit that added the line")
140
+
141
+
142
+ class WatchHits(unittest.TestCase):
143
+ def test_each_cohort_counts_its_commits_touching_a_watched_file(self):
144
+ with tempfile.TemporaryDirectory() as d:
145
+ history(d)
146
+ commits = provenance.read_commits(d)
147
+ inv = provenance.trailers(commits)
148
+ out = provenance.cohort(commits, inv, {"a.py"})
149
+ self.assertEqual((out["cohort"]["watch"], out["rest"]["watch"]), (1, 2), "the helper commit is marked; start and the fix are not")
150
+ self.assertNotIn("watch", provenance.cohort(commits, inv)["rest"], "no watch list, no count")
151
+
152
+
113
153
  class Step(unittest.TestCase):
114
154
  def test_the_step_writes_provenance_json(self):
115
155
  with tempfile.TemporaryDirectory() as d:
@@ -121,7 +161,7 @@ class Step(unittest.TestCase):
121
161
  self.assertEqual(p.returncode, 0, p.stderr)
122
162
  with open(os.path.join(out, "provenance.json")) as fh:
123
163
  data = json.load(fh)
124
- self.assertEqual(set(data), {"trailers", "cohort", "shape", "agents"})
164
+ self.assertEqual(set(data), {"trailers", "cohort", "shape", "agents", "lines"})
125
165
 
126
166
 
127
167
  if __name__ == "__main__":
@@ -1502,3 +1502,18 @@ class Excerpt(unittest.TestCase):
1502
1502
 
1503
1503
  if __name__ == "__main__":
1504
1504
  unittest.main()
1505
+
1506
+
1507
+ class ChangedLines(unittest.TestCase):
1508
+ def test_two_windows_and_the_cohorts_when_there_are_marked_commits(self):
1509
+ w = {"commits": 10, "added": 200, "moved": 20, "churned": 10, "moved_share": 0.1, "churn_share": 0.05}
1510
+ rep = {"provenance": {"lines": {"windows": [{"label": "last year", "from": "2025-01-01", "to": "2026-01-01", **w},
1511
+ {"label": "the year before", "from": "2024-01-01", "to": "2025-01-01", **w, "added": 0, "moved_share": None, "churn_share": None}],
1512
+ "cohort": {"marked": {**w, "commits": 2}, "rest": w}, "churn_days": 14},
1513
+ "cohort": {"cohort": {"commits": 2, "watch": 1}, "rest": {"commits": 8, "watch": 2}, "watch_top": 15, "share": 0.2}}}
1514
+ sec = render.lines_section(rep)
1515
+ self.assertEqual(sec["rows"][0], ["last year (2025-01-01 to 2026-01-01)", "10", "200", "10.0%", "5.0%"])
1516
+ self.assertEqual(sec["rows"][1][3:], ["-", "-"])
1517
+ self.assertEqual([r[0] for r in sec["rows"][2:]], ["marked commits, both years", "the rest, both years"])
1518
+ self.assertIn("lines", render.FULL_ONLY)
1519
+ self.assertIn("touched a file on the watch list's top 15 50% against 25%", render.trailers_section(rep)["caption"] or render.trailers_section(rep)["note"] or "")
@@ -0,0 +1,88 @@
1
+ """The shape rules on the tree-sitter pass: an error caught and dropped, an address written into a string
2
+ literal, code left in a comment. The parsing tests need gitmole[structure]; the findings do not."""
3
+ import unittest
4
+
5
+ from gitmole import findings, structure
6
+ from tests.test_findings import report
7
+ from tests.test_structure import HAVE, parse
8
+
9
+
10
+ class CommentedCode(unittest.TestCase):
11
+ """The comment classifier is plain text, no grammar needed."""
12
+
13
+ def test_code_left_by_an_editor_counts(self):
14
+ self.assertEqual(structure.commented_code_lines("// const old = legacy(x);\n// console.log(state);\n// if (a) {\n// b();\n// }"), 5)
15
+ self.assertEqual(structure.commented_code_lines("/* legacy(x);\n other(y); */"), 2)
16
+ self.assertEqual(structure.commented_code_lines("# x = compute(a, b)"), 1)
17
+
18
+ def test_prose_examples_directives_and_doc_comments_do_not(self):
19
+ self.assertEqual(structure.commented_code_lines("# Some examples:\n# SomeModel.objects.annotate(x)\n# foo(bar)"), 0)
20
+ self.assertEqual(structure.commented_code_lines("// function f(fn) {\n// return g();\n// }"), 0, "indented under the marker: an example")
21
+ self.assertEqual(structure.commented_code_lines("// Build the dispatcher:\n// function Foo(a) {\n// return x;\n// }\n// }"), 0)
22
+ self.assertEqual(structure.commented_code_lines("/** @example foo(1); */"), 0)
23
+ self.assertEqual(structure.commented_code_lines("// eslint-disable-next-line no-console"), 0)
24
+ self.assertEqual(structure.commented_code_lines("/* 35 = OBSOLETE */"), 0)
25
+ self.assertEqual(structure.commented_code_lines("# This explains why we do it."), 0)
26
+
27
+ def test_addresses_leave_out_what_is_not_a_host(self):
28
+ self.assertEqual(structure._address("'10.1.2.3'"), "10.1.2.3")
29
+ self.assertEqual(structure._address("`203.0.114.9:8080`"), "203.0.114.9:8080")
30
+ for literal in ("'127.0.0.1'", "'0.0.0.0'", "'255.255.255.0'", "'1.0.0.0'", "'2.5.4.3'", "'192.0.2.10'", "'300.1.1.1'", "'v1.2.3.4'"):
31
+ self.assertIsNone(structure._address(literal), literal)
32
+
33
+
34
+ @unittest.skipUnless(HAVE, "gitmole[structure] not installed")
35
+ class Shapes(unittest.TestCase):
36
+ def test_python_counts_only_the_broad_except_that_does_nothing(self):
37
+ s = parse(".py", "try:\n x()\nexcept:\n pass\ntry:\n y()\nexcept ValueError:\n pass\n"
38
+ "try:\n z()\nexcept Exception as e:\n pass\ntry:\n w()\nexcept Exception:\n # ignored on purpose\n pass\n")["shapes"]
39
+ self.assertEqual(s["empty_catch"], [3, 11])
40
+ self.assertEqual(s["bare_except"], [3])
41
+
42
+ def test_empty_catch_in_other_languages_and_a_comment_saves_it(self):
43
+ self.assertEqual(parse(".js", "try { a() } catch (e) {}\ntry { b() } catch { /* ok */ }\n")["shapes"]["empty_catch"], [1])
44
+ self.assertEqual(parse(".java", "class A { void f() { try { g(); } catch (Exception e) { } } }\n")["shapes"]["empty_catch"], [1])
45
+ self.assertEqual(parse(".rb", "begin\n x\nrescue\nend\nbegin\n y\nrescue => e\n # ok\nend\n")["shapes"]["empty_catch"], [3])
46
+
47
+ def test_addresses_in_literals_but_not_in_attributes(self):
48
+ s = parse(".cs", '[assembly: AssemblyVersion("1.2.3.4")]\nclass A { void F() { var s = "8.8.4.4"; } }\n')["shapes"]
49
+ self.assertEqual(s["addresses"], [{"line": 2, "value": "8.8.4.4"}])
50
+
51
+ def test_line_comments_on_consecutive_lines_are_one_block_and_trailing_comments_are_not_code(self):
52
+ s = parse(".py", "x = 1 # y = 2\n# a = f(b)\n# c = g(d)\n\n# prose about it\n")["shapes"]
53
+ self.assertEqual((s["commented_code"], s["commented_sample"]), (2, [2]))
54
+
55
+
56
+ def _report(files, **over):
57
+ return report(structure={"status": "run", "files": files}, **over)
58
+
59
+
60
+ class ShapeFindings(unittest.TestCase):
61
+ def test_swallowed_errors_from_five_and_never_in_tests(self):
62
+ files = {"src/a.py": {"shapes": {"empty_catch": [3, 9], "empty_catch_count": 4, "bare_except": [3], "bare_except_count": 1}},
63
+ "src/b.js": {"shapes": {"empty_catch": [7], "empty_catch_count": 1}},
64
+ "tests/test_a.py": {"shapes": {"empty_catch": [1], "empty_catch_count": 9}}}
65
+ f = findings.swallowed_errors(_report(files))
66
+ self.assertEqual([(x["rule"]["id"], x["severity"]) for x in f], [("swallowed_errors", "info")])
67
+ self.assertIn("5 empty catch blocks in 2 source files, 1 of them a bare except", f[0]["detail"])
68
+ self.assertEqual(f[0]["evidence"]["files"][0], {"file": "src/a.py", "start": 3, "count": 4})
69
+ del files["src/b.js"]
70
+ self.assertEqual(findings.swallowed_errors(_report(files)), [], "four in source is below the floor")
71
+
72
+ def test_addresses_and_commented_code(self):
73
+ files = {"src/net.go": {"shapes": {"addresses": [{"line": 4, "value": "10.0.0.7"}], "addresses_count": 1}},
74
+ "src/old.js": {"shapes": {"commented_code": 12, "commented_sample": [40, 90]}},
75
+ "src/some.js": {"shapes": {"commented_code": 3, "commented_sample": [5]}}}
76
+ f = findings.hardcoded_addresses(_report(files))
77
+ self.assertIn("10.0.0.7 at src/net.go:4", f[0]["detail"])
78
+ f = findings.commented_out_code(_report(files))
79
+ self.assertEqual([x["file"] for x in f[0]["evidence"]["files"]], ["src/old.js"])
80
+ self.assertIn("src/old.js (12 lines from line 40)", f[0]["detail"])
81
+
82
+ def test_nothing_without_the_structure_step(self):
83
+ for rule in (findings.swallowed_errors, findings.hardcoded_addresses, findings.commented_out_code):
84
+ self.assertEqual(rule(report()), [])
85
+
86
+
87
+ if __name__ == "__main__":
88
+ unittest.main()
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes