gitmole 0.18.0__tar.gz → 0.20.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. {gitmole-0.18.0 → gitmole-0.20.0}/PKG-INFO +1 -1
  2. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/__init__.py +1 -1
  3. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/cli.py +12 -2
  4. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/deps.py +74 -2
  5. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/findings.py +171 -5
  6. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/hygiene.py +13 -6
  7. gitmole-0.20.0/gitmole/imports.py +267 -0
  8. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/knowledge.py +34 -0
  9. gitmole-0.20.0/gitmole/licences.py +329 -0
  10. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/load.py +14 -2
  11. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/maat.py +100 -1
  12. gitmole-0.20.0/gitmole/osps.py +114 -0
  13. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/render.py +25 -3
  14. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/run.py +1 -1
  15. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/sarif.py +1 -1
  16. gitmole-0.20.0/gitmole/sbom.py +98 -0
  17. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/watch.py +22 -0
  18. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole.egg-info/PKG-INFO +1 -1
  19. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole.egg-info/SOURCES.txt +5 -0
  20. gitmole-0.20.0/tests/test_declared.py +265 -0
  21. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_deps.py +21 -3
  22. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_findings.py +56 -1
  23. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_hygiene.py +2 -2
  24. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_maat.py +53 -2
  25. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_render.py +15 -7
  26. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_watch.py +15 -0
  27. {gitmole-0.18.0 → gitmole-0.20.0}/LICENSE +0 -0
  28. {gitmole-0.18.0 → gitmole-0.20.0}/README.md +0 -0
  29. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/__main__.py +0 -0
  30. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/backtest.py +0 -0
  31. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/banner.py +0 -0
  32. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/blame.py +0 -0
  33. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/classify.py +0 -0
  34. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/clean.py +0 -0
  35. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/compare.py +0 -0
  36. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/coupling.py +0 -0
  37. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/duplicates.py +0 -0
  38. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/evaluate.py +0 -0
  39. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/filetypes.py +0 -0
  40. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/functions.py +0 -0
  41. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/hook.py +0 -0
  42. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/hotspots.py +0 -0
  43. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/identity.py +0 -0
  44. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/leaks.py +0 -0
  45. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/loss.py +0 -0
  46. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/provenance.py +0 -0
  47. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/signing.py +0 -0
  48. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/structure.py +0 -0
  49. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/szz.py +0 -0
  50. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/textfmt.py +0 -0
  51. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole/trend.py +0 -0
  52. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole.egg-info/dependency_links.txt +0 -0
  53. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole.egg-info/entry_points.txt +0 -0
  54. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole.egg-info/requires.txt +0 -0
  55. {gitmole-0.18.0 → gitmole-0.20.0}/gitmole.egg-info/top_level.txt +0 -0
  56. {gitmole-0.18.0 → gitmole-0.20.0}/pyproject.toml +0 -0
  57. {gitmole-0.18.0 → gitmole-0.20.0}/setup.cfg +0 -0
  58. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_backtest.py +0 -0
  59. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_banner.py +0 -0
  60. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_blame.py +0 -0
  61. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_classify.py +0 -0
  62. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_clean.py +0 -0
  63. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_cli.py +0 -0
  64. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_compare.py +0 -0
  65. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_coupling.py +0 -0
  66. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_duplicates.py +0 -0
  67. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_evaluate.py +0 -0
  68. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_filetypes.py +0 -0
  69. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_functions.py +0 -0
  70. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_golden.py +0 -0
  71. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_hook.py +0 -0
  72. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_hotspots.py +0 -0
  73. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_identity.py +0 -0
  74. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_knowledge.py +0 -0
  75. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_leaks.py +0 -0
  76. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_load.py +0 -0
  77. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_loss.py +0 -0
  78. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_packaging.py +0 -0
  79. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_provenance.py +0 -0
  80. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_render_examples.py +0 -0
  81. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_run.py +0 -0
  82. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_sarif.py +0 -0
  83. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_signing.py +0 -0
  84. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_structure.py +0 -0
  85. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_szz.py +0 -0
  86. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_textfmt.py +0 -0
  87. {gitmole-0.18.0 → gitmole-0.20.0}/tests/test_trend.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.18.0
3
+ Version: 0.20.0
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.18.0"
3
+ __version__ = "0.20.0"
@@ -47,6 +47,7 @@ def parse_args(argv):
47
47
  p.add_argument("--fail-on", choices=findings.SEVERITIES, help="exit 3 if any finding is at this severity or worse")
48
48
  p.add_argument("--risk", metavar="BASE", help="score the files changed since BASE (merge base with HEAD) by their share of the repository's revisions × lines of code; needs a local path")
49
49
  p.add_argument("--risk-threshold", type=float, metavar="N", help="with --risk: exit 3 when the changed files hold more than N percent of the repository's revisions × lines of code")
50
+ p.add_argument("--sbom", metavar="PATH", help="write a CycloneDX 1.6 SBOM of every locked package to PATH, or - for stdout, from the osv-scanner step's package list")
50
51
  p.add_argument("--compare", metavar="BEFORE_JSON", help="add a 'Since last report' section against an earlier --json export of the same clone")
51
52
  p.add_argument("--hook", action="store_true", help="with --no-run: read an agent hook's JSON on stdin (or files after --), score the files it names like --risk, "
52
53
  "print a summary the agent reads back, exit 2 when --risk-threshold is exceeded")
@@ -84,7 +85,7 @@ def main(argv=None, console: Console = None, tool_check=run.missing_tools, plann
84
85
  if args.clean:
85
86
  return _clean(args, console, ask or (lambda q: console.input(q, markup=False)))
86
87
  # When an export goes to stdout, everything else (banner, progress, report) moves to stderr.
87
- quiet = "-" in (args.json, args.markdown, args.sarif)
88
+ quiet = "-" in (args.json, args.markdown, args.sarif, args.sbom)
88
89
  ui = Console(stderr=True) if quiet else console
89
90
 
90
91
  rc, now = _resolve_time(args, err, ui)
@@ -148,6 +149,7 @@ def _check_args(args, err, kind=None) -> int | None:
148
149
  bad = None
149
150
  else:
150
151
  bad = ("--compare needs one repository, not owner/*" if args.compare and kind == "org" else
152
+ "--sbom needs one repository, not owner/*" if args.sbom and kind == "org" else
151
153
  "--risk needs a local path" if args.risk else
152
154
  "--list-file-types needs a local path" if args.list_file_types else None)
153
155
  if bad:
@@ -598,7 +600,15 @@ def _render(out_dir: str, console: Console, ui: Console, args, err: Console) ->
598
600
  if args.sarif:
599
601
  from . import sarif
600
602
  _write(sarif.dumps(report, found, scope=args.sarif_scope), args.sarif, console)
601
- if "-" not in (args.json, args.markdown, args.sarif):
603
+ if args.sbom:
604
+ from . import sbom
605
+ packages = sbom.read_packages(out_dir)
606
+ if packages is None:
607
+ err.print("[red]--sbom:[/red] no package list in the output directory; the osv-scanner step writes it, and did not "
608
+ f"(dependencies: {(report.get('dependencies') or {}).get('status', 'not-run')}; run.log says why)", soft_wrap=True)
609
+ return 2
610
+ _write(sbom.dumps(report, packages), args.sbom, console)
611
+ if "-" not in (args.json, args.markdown, args.sarif, args.sbom):
602
612
  render.report(report, found, console, full=args.full, risk=risk, base=args.risk, compare=comparison)
603
613
  if args.fail_on and any(findings.SEVERITIES.index(f["severity"]) <= findings.SEVERITIES.index(args.fail_on) for f in found):
604
614
  return 3
@@ -21,6 +21,12 @@ import subprocess
21
21
  import sys
22
22
  import tempfile
23
23
 
24
+ try:
25
+ from . import imports, licences
26
+ except ImportError: # run as a script: the package directory is sys.path[0]
27
+ import imports
28
+ import licences
29
+
24
30
  ARGV = ["osv-scanner", "scan", "source", "-r", "--offline", "--format", "json", "--all-packages", "."]
25
31
  NO_SOURCES = 128 # osv-scanner: no lock file found
26
32
  NO_DATABASE = "no offline version of the OSV database" # its message when the local copy is missing
@@ -141,8 +147,8 @@ def _label(score, vulns: list) -> str:
141
147
 
142
148
  def _relative(path: str, cwd: str) -> str:
143
149
  if os.path.isabs(path):
144
- try:
145
- rel = os.path.relpath(path, cwd)
150
+ try: # real paths on both sides: osv-scanner may report /tmp/... for a cwd of /private/tmp/...
151
+ rel = os.path.relpath(os.path.realpath(path), os.path.realpath(cwd))
146
152
  except ValueError:
147
153
  return path
148
154
  return path if rel.startswith("..") else rel
@@ -174,6 +180,64 @@ def summarise(data: dict, cwd: str) -> dict:
174
180
  return {"status": "scanned", "sources": sources, "packages": packages, "vulnerable": vulnerable}
175
181
 
176
182
 
183
+ PACKAGES = "packages.json" # every locked package, for --sbom; not part of the report, which keeps the vulnerable ones
184
+
185
+
186
+ def packages(data: dict, cwd: str) -> list:
187
+ """Every package the lock files pin, once per ecosystem, name and version, with the lock files that
188
+ pin it and the licence a lock file declares for it (package-lock.json, composer.lock), sorted."""
189
+ declared = {}
190
+ try:
191
+ for d in licences.lock_licences(cwd, [_relative(p, cwd) for p in _lock_paths(data)], every=True):
192
+ declared.setdefault((d["ecosystem"], d["name"], d["version"]), d["expression"])
193
+ except OSError:
194
+ pass
195
+ by_key = {}
196
+ for r in data.get("results") or []:
197
+ path = _relative((r.get("source") or {}).get("path") or "", cwd)
198
+ for p in r.get("packages") or []:
199
+ info = p.get("package") or {}
200
+ key = (info.get("ecosystem", ""), info.get("name", ""), info.get("version", ""))
201
+ row = by_key.setdefault(key, {"ecosystem": key[0], "name": key[1], "version": key[2], "sources": []})
202
+ if path not in row["sources"]:
203
+ row["sources"].append(path)
204
+ out = []
205
+ for key in sorted(by_key):
206
+ row = by_key[key]
207
+ row["sources"].sort()
208
+ if key in declared:
209
+ row["license"] = declared[key]
210
+ out.append(row)
211
+ return out
212
+
213
+
214
+ def _lock_paths(data: dict) -> list:
215
+ return [(r.get("source") or {}).get("path") or "" for r in data.get("results") or []]
216
+
217
+
218
+ # The same scan with the vulnerability matcher switched off: it reads every lock file and matches nothing,
219
+ # so it needs no database. The flag is osv-scanner's experimental plugin switch (2.x); when it fails, there
220
+ # is no package list and --sbom says so.
221
+ LIST_ARGV = ARGV[:-1] + ["--experimental-disable-plugins", "vulnmatch/osvlocal", "."]
222
+
223
+
224
+ def list_packages():
225
+ proc = subprocess.run(LIST_ARGV, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
226
+ if proc.returncode not in (0, 1):
227
+ print(f"deps.py: listing the packages without the database: osv-scanner exited {proc.returncode}", file=sys.stderr)
228
+ return None
229
+ try:
230
+ data = json.loads(proc.stdout.decode("utf-8", "replace") or "{}")
231
+ except json.JSONDecodeError:
232
+ return None
233
+ return data if isinstance(data, dict) else None
234
+
235
+
236
+ def write_packages(rows: list, target: str) -> None:
237
+ with open(target, "w", encoding="utf-8") as fh:
238
+ json.dump({"packages": rows}, fh, indent=0, sort_keys=True)
239
+
240
+
177
241
  def write(result: dict, target: str) -> None:
178
242
  fd, tmp = tempfile.mkstemp(dir=os.path.dirname(os.path.abspath(target)), prefix=".dependencies-", suffix=".json")
179
243
  try:
@@ -198,6 +262,9 @@ def main(argv=None) -> int:
198
262
  result = {"status": "no-sources"}
199
263
  elif NO_DATABASE in err:
200
264
  result = {"status": "no-database", "download": DOWNLOAD}
265
+ listed = list_packages() # the package list for --sbom needs no database
266
+ if listed is not None:
267
+ write_packages(packages(listed, os.getcwd()), os.path.join(os.path.dirname(os.path.abspath(target)), PACKAGES))
201
268
  elif proc.returncode in (0, 1): # 1: packages with vulnerabilities were found
202
269
  text = proc.stdout.decode("utf-8", "replace").strip()
203
270
  try:
@@ -208,6 +275,11 @@ def main(argv=None) -> int:
208
275
  result = summarise(data, os.getcwd())
209
276
  result["database_date"] = database_date()
210
277
  result["database_digest"] = database_digest()
278
+ try:
279
+ imports.annotate(result["vulnerable"], os.getcwd())
280
+ except OSError as e: # the rows stand without it
281
+ print(f"deps.py: imports: {e}", file=sys.stderr)
282
+ write_packages(packages(data, os.getcwd()), os.path.join(os.path.dirname(os.path.abspath(target)), PACKAGES))
211
283
  else:
212
284
  print(f"deps.py: osv-scanner exited {proc.returncode}; no report written", file=sys.stderr)
213
285
  return proc.returncode
@@ -4,7 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  import re
6
6
 
7
- from . import classify, coupling, filetypes, hotspots, knowledge, leaks, loss, maat, textfmt, trend
7
+ from . import classify, coupling, filetypes, hotspots, knowledge, leaks, licences, loss, maat, osps, textfmt, trend
8
8
 
9
9
  SEVERITIES = ["critical", "warning", "info"]
10
10
 
@@ -29,6 +29,8 @@ def _f(severity: str, title: str, statement: str, advice: str, rule: dict, evide
29
29
  between them a reader of the JSON can check the finding without reading this file."""
30
30
  if rule.get("id") in REFS and "ref" not in rule:
31
31
  rule = {**rule, "ref": REFS[rule["id"]]}
32
+ if rule.get("id") in osps.RULE_CONTROLS and "osps" not in rule: # the OSPS Baseline controls it gives evidence for
33
+ rule = {**rule, "osps": list(osps.RULE_CONTROLS[rule["id"]])}
32
34
  return {"severity": severity, "title": title, "detail": f"{statement.rstrip()} {advice}", "advice": advice,
33
35
  "rule": rule, "evidence": evidence}
34
36
 
@@ -698,7 +700,8 @@ def _vuln_statement(rows: list, sources: int) -> str:
698
700
  if r.get("malicious"):
699
701
  score = ", malicious"
700
702
  fixed = f", fixed in {r['fixed']}" if r.get("fixed") else ", no fix yet"
701
- return f"{r['name']} {r['version']} ({ref}{score}{fixed}) in {r['source']}"
703
+ loaded = ", imported by no tracked source" if r.get("imported") is False else ""
704
+ return f"{r['name']} {r['version']} ({ref}{score}{fixed}{loaded}) in {r['source']}"
702
705
  listed = "; ".join(one(r) for r in rows[:3])
703
706
  more = f" and {len(rows) - 3} more" if len(rows) > 3 else ""
704
707
  return f"{_plural(len(rows), 'vulnerable package')} in {_plural(sources, 'lock file')}: {listed}{more}."
@@ -740,7 +743,8 @@ def vulnerable_dependencies(report: dict) -> list:
740
743
  evidence={"lock_files": sources,
741
744
  "packages": [{"name": r["name"], "version": r["version"], "source": r["source"], "score": r.get("score"),
742
745
  "fixed": r.get("fixed") or None, "ids": list(r.get("ids") or []),
743
- "aliases": list(r.get("aliases") or []), "malicious": bool(r.get("malicious"))} for r in group[:10]]}))
746
+ "aliases": list(r.get("aliases") or []), "malicious": bool(r.get("malicious")),
747
+ "imported": r.get("imported", "unknown")} for r in group[:10]]}))
744
748
  return out
745
749
 
746
750
 
@@ -753,7 +757,8 @@ def hygiene_findings(report: dict) -> list:
753
757
  stands in for without the GitHub API. Nothing for an output directory from before the step."""
754
758
  h = report.get("hygiene") or {}
755
759
  out = []
756
- for check in (_hygiene_actions, _hygiene_lockfiles, _hygiene_updates, _hygiene_presence, _hygiene_confusion, _hygiene_install, _hygiene_binaries, _hygiene_submodules, _hygiene_symlinks, _hygiene_trojan):
760
+ for check in (_hygiene_actions, _hygiene_lockfiles, _hygiene_updates, _hygiene_presence, _hygiene_confusion, _hygiene_install, _hygiene_binaries, _hygiene_submodules, _hygiene_symlinks, _hygiene_trojan,
761
+ _hygiene_unused, _hygiene_licence, _hygiene_copyleft):
757
762
  check(h, out)
758
763
  return out
759
764
 
@@ -908,6 +913,59 @@ def _hygiene_trojan(h: dict, out: list) -> None:
908
913
  evidence={"bidi": (tj.get("bidi") or [])[:10], "mixed_script": (tj.get("mixed_script") or [])[:10]}))
909
914
 
910
915
 
916
+ def _hygiene_unused(h: dict, out: list) -> None:
917
+ im = h.get("imports") or {}
918
+ rows = im.get("unused") or []
919
+ if not rows:
920
+ return
921
+ by_manifest = {}
922
+ for r in rows:
923
+ by_manifest.setdefault(r["manifest"], []).append(r["package"])
924
+ listed = "; ".join(f"{_files_list(v)} in {k}" for k, v in list(by_manifest.items())[:3])
925
+ n = im.get("count", len(rows))
926
+ first = rows[0]
927
+ out.append(_f("info", "Declared dependencies nothing imports",
928
+ f"{n} runtime {'dependency is' if n == 1 else 'dependencies are'} declared and never imported by a tracked file, nor named in a script or configuration: {listed}.",
929
+ f"Remove {first['package']} from {first['manifest']} if nothing loads it at run time; an unused dependency is still installed, scanned and updated.",
930
+ rule={"id": "unused_dependencies", "reads": "package.json dependencies, go.mod direct requirements, Cargo.toml [dependencies]"},
931
+ evidence={"count": n, "manifests": im.get("manifests", 0), "unused": rows[:10]}))
932
+
933
+
934
+ def _hygiene_licence(h: dict, out: list) -> None:
935
+ lic = h.get("licences") or {}
936
+ declared = lic.get("declared") or []
937
+ if lic.get("approved") is False or lic.get("mismatch"):
938
+ named = "; ".join(f"{d['source']} declares {d['expression']}" for d in declared)
939
+ parts = []
940
+ if lic.get("mismatch"):
941
+ parts.append(f"{named}, but {lic['files'][0]} is the {lic['file_licence']} text")
942
+ if lic.get("approved") is False:
943
+ parts.append(f"the declared licence ({textfmt.join_and([d['expression'] for d in declared] or [lic.get('file_licence') or ''])}) is not OSI- or FSF-approved")
944
+ out.append(_f("warning" if lic.get("approved") is False else "info", "Project licence as declared", "; ".join(parts) + ".",
945
+ "Make the manifests and the licence file name the same licence; a packager reads the manifest, a lawyer the file." if lic.get("mismatch")
946
+ else "Say so plainly in the README if the project is source-available rather than open source.",
947
+ rule={"id": "project_licence", "reads": "root manifests and the licence file"},
948
+ evidence={"declared": declared, "file": (lic.get("files") or [None])[0], "file_licence": lic.get("file_licence"),
949
+ "approved": lic.get("approved"), "mismatch": bool(lic.get("mismatch"))}))
950
+
951
+
952
+ def _hygiene_copyleft(h: dict, out: list) -> None:
953
+ lic = h.get("licences") or {}
954
+ strong = lic.get("strong") or []
955
+ if not strong or lic.get("project") != licences.PERMISSIVE:
956
+ return
957
+ n = lic.get("strong_count", len(strong))
958
+ listed = _files_list([f"{d['name']} {d['version']} ({d['expression']})" for d in strong])
959
+ own = textfmt.join_and(sorted({d["expression"] for d in lic.get("declared") or []} | ({lic["file_licence"]} if lic.get("file_licence") else set())))
960
+ weak = f" {lic['weak_count']} more declare{'s' if lic.get('weak_count') == 1 else ''} weak copyleft (LGPL, MPL, EPL), which a dependency usually may." if lic.get("weak_count") else ""
961
+ out.append(_f("warning", "Copyleft dependencies in a permissive project",
962
+ f"The project declares {own}, and {n} runtime {'dependency' if n == 1 else 'dependencies'} in {textfmt.join_and(sorted({d['lockfile'] for d in strong}))} "
963
+ f"{'declares' if n == 1 else 'declare'} a strong copyleft licence: {listed}.{weak} Declared, as the lock file records it, not read from the package's files.",
964
+ f"Check whether {strong[0]['name']} is distributed with the project; if it is, its licence terms reach the whole work.",
965
+ rule={"id": "copyleft_dependencies", "reads": "package-lock.json and composer.lock licence fields, runtime packages only"},
966
+ evidence={"project": own, "count": n, "weak": lic.get("weak_count", 0), "dependencies": lic.get("dependencies", 0), "strong": strong[:10]}))
967
+
968
+
911
969
  def _aside_path(path: str) -> bool:
912
970
  return filetypes.is_test_path(path) or filetypes.is_sample_path(path) or filetypes.is_vendor_path(path)
913
971
 
@@ -1101,10 +1159,118 @@ def signoff_by_co_author(report: dict, min_commits: int = 2) -> list:
1101
1159
  evidence={"identities": rows[:10]})]
1102
1160
 
1103
1161
 
1162
+ def _pool_files(report: dict) -> list:
1163
+ """The source files still in the tree that no classifier reason sets aside."""
1164
+ from . import classify
1165
+ cls = classify.Classifier(report)
1166
+ return sorted(f for f in _tree(report) if cls.reason(f) is None)
1167
+
1168
+
1169
+ def _authors_of(report: dict, files: list, key: str = "is_author") -> dict:
1170
+ wanted = set(files)
1171
+ out = {f: set() for f in files}
1172
+ for r in report.get("doa") or []:
1173
+ if r["entity"] in wanted and r.get(key):
1174
+ out[r["entity"]].add(r["author"])
1175
+ return out
1176
+
1177
+
1178
+ def truck_factor(report: dict, min_files: int = 20, area_files: int = 10) -> list:
1179
+ """Avelino et al.'s truck factor over the degree of authorship: how many people have to leave before
1180
+ more than half the source files have no author. One is a warning, two a note. Changes rather than
1181
+ lines, and a creator's bonus, so it can disagree with the surviving-code share, which the bus-factor
1182
+ finding reads; the finding says so when it does. Also per area, and with knowledge halving every
1183
+ five months."""
1184
+ if not report.get("doa"):
1185
+ return []
1186
+ files = _pool_files(report)
1187
+ authored = {f: a for f, a in _authors_of(report, files).items()}
1188
+ if len(files) < min_files:
1189
+ return []
1190
+ tf, removed, share = knowledge.truck_factor(authored)
1191
+ tf_d, removed_d, _ = knowledge.truck_factor(_authors_of(report, files, "is_author_decayed"))
1192
+ depth = knowledge.depth_for(files)
1193
+ areas = {}
1194
+ for f in files:
1195
+ areas.setdefault(knowledge._area(f, depth), []).append(f)
1196
+ lone = []
1197
+ for area, fs in sorted(areas.items()):
1198
+ if len(fs) >= area_files and area != knowledge.ROOT:
1199
+ n, who, _ = knowledge.truck_factor({f: authored[f] for f in fs})
1200
+ if n == 1:
1201
+ lone.append((area, who[0]))
1202
+ if tf > 2 and not lone:
1203
+ return []
1204
+ orphans = round(share * len(files))
1205
+ statement = (f"Truck factor {tf}: without {textfmt.join_and(removed)}, {orphans} of the {len(files)} source files ({_pct(orphans, len(files))}) "
1206
+ f"have no author left.")
1207
+ if tf_d != tf:
1208
+ statement += f" With knowledge halving every five months it is {tf_d} ({textfmt.join_and(removed_d)})."
1209
+ if lone:
1210
+ statement += " Areas with a truck factor of one: " + ", ".join(f"{a} ({w})" for a, w in lone[:5]) + (f" and {len(lone) - 5} more" if len(lone) > 5 else "") + "."
1211
+ shares = report.get("theseus_authors") or {}
1212
+ if shares and removed:
1213
+ top, lines = max(shares.items(), key=lambda kv: kv[1])
1214
+ if top != removed[0]:
1215
+ statement += f" The surviving code's largest share is {top}'s ({_pct(lines, sum(shares.values()))}), which the bus-factor finding reads."
1216
+ first_area = next((a for a, w in lone if w == removed[0]), lone[0][0] if lone else None)
1217
+ advice = f"Pair someone with {removed[0]}" + (f" on {first_area}" if first_area else "") + " first; they author most of what would be left without an author."
1218
+ return [_f("warning" if tf == 1 else "info", "Truck factor", statement, advice,
1219
+ rule={"id": "truck_factor", "doa_author_share": 0.75, "doa_floor": 3.293, "orphan_share": 0.5, "decay_months": 5,
1220
+ "ref": "Avelino et al., ICPC 2016"},
1221
+ evidence={"truck_factor": tf, "removed": removed, "truck_factor_decayed": tf_d, "removed_decayed": removed_d,
1222
+ "files": len(files), "orphaned": orphans, "areas": [{"area": a, "author": w} for a, w in lone[:10]]})]
1223
+
1224
+
1225
+ def authors_gone(report: dict, min_files: int = 5) -> list:
1226
+ """Files whose every author by degree of authorship has stopped committing, while others still
1227
+ change them: "creator left, editors remain", knowledge the blame share cannot show."""
1228
+ if not report.get("doa"):
1229
+ return []
1230
+ months = report["meta"].get("gone_months", loss.DEFAULT_MONTHS)
1231
+ gone = {g["name"] for g in loss.gone(report, months)}
1232
+ fresh = {a["entity"] for a in report.get("age") or [] if a["age-months"] < 12}
1233
+ files = [f for f in _pool_files(report) if f in fresh]
1234
+ authored = _authors_of(report, files)
1235
+ left = [(f, sorted(a)) for f, a in authored.items() if a and a <= gone]
1236
+ if len(left) < min_files:
1237
+ return []
1238
+ listed = "; ".join(f"{f} ({textfmt.join_and(a)})" for f, a in left[:5]) + (f" and {len(left) - 5} more" if len(left) > 5 else "")
1239
+ return [_f("info", "Files whose authors have left", f"{len(left)} source files changed in the last year have no author still committing: {listed}.",
1240
+ f"Make the people who edit {left[0][0]} its authors: review its design with them and write down what only {left[0][1][0]} knew.",
1241
+ rule={"id": "authors_gone", "gone_months": months, "min_files": min_files, "ref": "Avelino et al., ICPC 2016"},
1242
+ evidence={"count": len(left), "files": [{"file": f, "authors": a} for f, a in left[:10]]})]
1243
+
1244
+
1245
+ def component_coupling(report: dict, min_degree: int = 30) -> list:
1246
+ """Components (top-level directories, or the level below a lone src/) that change together in a
1247
+ large share of their changes: coupling at the level of the architecture, where two files in one
1248
+ directory is only a layout."""
1249
+ rows = report.get("components") or []
1250
+ if not rows:
1251
+ return []
1252
+ depth = knowledge.depth_for(list(_tree(report)) or [r["entity"] + "x" for r in rows])
1253
+
1254
+ def aside(c):
1255
+ probe = c + "x.py"
1256
+ return filetypes.is_test_path(probe) or filetypes.is_sample_path(probe) or filetypes.is_doc_path(probe) or filetypes.is_vendor_path(probe)
1257
+ pairs = [r for r in rows if r["depth"] == depth and r["degree"] >= min_degree and not aside(r["entity"]) and not aside(r["coupled"])]
1258
+ if not pairs:
1259
+ return []
1260
+ listed = "; ".join(f"{p['entity']} and {p['coupled']} change together in {p['degree']}% of their changes ({p['shared']} shared)" for p in pairs[:3])
1261
+ more = f" ({len(pairs) - 3} more pairs)" if len(pairs) > 3 else ""
1262
+ first = pairs[0]
1263
+ return [_f("info", "Components that change together", f"{listed}{more}.",
1264
+ f"Look at what {first['entity']} and {first['coupled']} share: a change that keeps landing in both is an interface nobody named.",
1265
+ rule={"id": "component_coupling", "min_degree": min_degree, "depth": depth, "ref": "Tornhill, Your Code as a Crime Scene, 2024"},
1266
+ evidence={"pairs": [{"a": p["entity"], "b": p["coupled"], "degree": p["degree"], "shared": p["shared"]} for p in pairs[:10]]})]
1267
+
1268
+
1104
1269
  RULES = [dormant, secrets_found, credential_files, vulnerable_dependencies, placeholder_identity, bus_factor, sizer_concerns, hotspot_dominance, bug_magnets,
1105
1270
  minor_contributors, reverts, brain_methods, complexity_growth, tight_coupling, duplication, stale_files, knowledge_islands, knowledge_loss,
1106
1271
  sweeping_commits, tangled_commits, hygiene_findings, debt_in_hotspots, deep_nesting, hidden_coupling, unreferenced_files,
1107
- agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author]
1272
+ agent_approval_disabled, agent_local_settings, mcp_literal_env, agent_instructions_drift, signoff_by_co_author,
1273
+ truck_factor, authors_gone, component_coupling]
1108
1274
 
1109
1275
 
1110
1276
  def evaluate(report: dict) -> list:
@@ -10,7 +10,8 @@ workflow's `uses:`, a lock file's name, `.gitmodules`, a symlink's mode) or on t
10
10
  - actions: `uses: owner/repo@ref` in .github/workflows where the ref is not a full commit SHA;
11
11
  - lockfiles: a manifest whose last commit is newer than its lock file's, or a manifest with none;
12
12
  - updates: the ecosystems whose lock files are tracked that dependabot.yml does not cover;
13
- - presence: a licence, a security policy, CODEOWNERS and the CODEOWNERS paths that match nothing;
13
+ - presence: a licence, a security policy, a contribution guide, CODEOWNERS and the CODEOWNERS paths
14
+ that match nothing;
14
15
  - confusion: a scoped npm package resolved from a registry other than the one .npmrc declares for
15
16
  its scope, several registries in one lock file, a pip `extra-index-url`;
16
17
  - install: packages with install scripts in package-lock.json, lifecycle scripts in a tracked
@@ -20,7 +21,9 @@ workflow's `uses:`, a lock file's name, `.gitmodules`, a symlink's mode) or on t
20
21
  - submodules: plain http:// or git:// URLs, credentials in a URL, relative URLs, `branch =`;
21
22
  - symlinks: links that resolve outside the tree or into .git/;
22
23
  - trojan: bidirectional control characters (CVE-2021-42574) and identifiers that mix Latin with
23
- Cyrillic, Greek or another confusable script, in source files."""
24
+ Cyrillic, Greek or another confusable script, in source files;
25
+ - licences: the licences the project and its locked dependencies declare (licences.py);
26
+ - imports: declared dependencies that nothing tracked imports (imports.py)."""
24
27
  from __future__ import annotations
25
28
 
26
29
  import ast
@@ -34,9 +37,11 @@ from collections import defaultdict
34
37
  from urllib.parse import urlsplit
35
38
 
36
39
  try:
37
- from . import filetypes
40
+ from . import filetypes, imports, licences
38
41
  except ImportError: # run as a script: the package directory is sys.path[0]
39
42
  import filetypes
43
+ import imports
44
+ import licences
40
45
 
41
46
  CAP = 50 # rows kept per list: the count says how many there were
42
47
 
@@ -197,8 +202,10 @@ def presence(repo: str) -> dict:
197
202
  if p.startswith(prefix) and "/" not in p[len(prefix):] and re.match(names, p[len(prefix):], re.I):
198
203
  return p
199
204
  return None
200
- licence = next((p for p in tracked if "/" not in p and re.match(r"^(licen[cs]e|copying)(\.|-|$)", p, re.I)), None)
205
+ licence = next((p for p in tracked if "/" not in p and re.match(r"^(licen[cs]e|copying)(\.|-|$)", p, re.I)), None) \
206
+ or next(("LICENSES/" for p in tracked if p.startswith("LICENSES/")), None) # the REUSE layout
201
207
  policy = first(r"^security(\.md|\.txt|\.rst)?$")
208
+ contributing = first(r"^contributing(\.md|\.txt|\.rst|\.adoc)?$")
202
209
  owners = first(r"^codeowners$")
203
210
  missing = []
204
211
  if owners:
@@ -209,7 +216,7 @@ def presence(repo: str) -> dict:
209
216
  pattern = line.split()[0]
210
217
  if not _codeowners_matches(pattern, tracked):
211
218
  missing.append(pattern)
212
- return {"license": licence, "security_policy": policy, "codeowners": owners, "codeowners_missing": missing[:CAP]}
219
+ return {"license": licence, "security_policy": policy, "contributing": contributing, "codeowners": owners, "codeowners_missing": missing[:CAP]}
213
220
 
214
221
 
215
222
  # --- dependency confusion -----------------------------------------------------------------------
@@ -483,7 +490,7 @@ def trojan_source(repo: str) -> dict:
483
490
 
484
491
  CHECKS = {"actions": actions_pinning, "lockfiles": lockfiles, "updates": dependency_updates, "presence": presence,
485
492
  "confusion": dependency_confusion, "install": install_scripts, "binaries": binaries, "submodules": submodules,
486
- "symlinks": symlinks, "trojan": trojan_source}
493
+ "symlinks": symlinks, "trojan": trojan_source, "licences": licences.check, "imports": imports.unused}
487
494
 
488
495
 
489
496
  def main(argv=None) -> int: