gitmole 0.19.0__tar.gz → 0.20.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. {gitmole-0.19.0 → gitmole-0.20.0}/PKG-INFO +1 -1
  2. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/__init__.py +1 -1
  3. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/cli.py +12 -2
  4. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/deps.py +74 -2
  5. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/findings.py +62 -4
  6. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/hygiene.py +13 -6
  7. gitmole-0.20.0/gitmole/imports.py +267 -0
  8. gitmole-0.20.0/gitmole/licences.py +329 -0
  9. gitmole-0.20.0/gitmole/osps.py +114 -0
  10. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/render.py +12 -2
  11. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/run.py +1 -1
  12. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/sarif.py +1 -1
  13. gitmole-0.20.0/gitmole/sbom.py +98 -0
  14. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole.egg-info/PKG-INFO +1 -1
  15. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole.egg-info/SOURCES.txt +5 -0
  16. gitmole-0.20.0/tests/test_declared.py +265 -0
  17. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_deps.py +21 -3
  18. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_hygiene.py +2 -2
  19. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_render.py +3 -2
  20. {gitmole-0.19.0 → gitmole-0.20.0}/LICENSE +0 -0
  21. {gitmole-0.19.0 → gitmole-0.20.0}/README.md +0 -0
  22. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/__main__.py +0 -0
  23. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/backtest.py +0 -0
  24. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/banner.py +0 -0
  25. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/blame.py +0 -0
  26. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/classify.py +0 -0
  27. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/clean.py +0 -0
  28. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/compare.py +0 -0
  29. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/coupling.py +0 -0
  30. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/duplicates.py +0 -0
  31. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/evaluate.py +0 -0
  32. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/filetypes.py +0 -0
  33. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/functions.py +0 -0
  34. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/hook.py +0 -0
  35. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/hotspots.py +0 -0
  36. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/identity.py +0 -0
  37. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/knowledge.py +0 -0
  38. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/leaks.py +0 -0
  39. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/load.py +0 -0
  40. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/loss.py +0 -0
  41. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/maat.py +0 -0
  42. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/provenance.py +0 -0
  43. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/signing.py +0 -0
  44. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/structure.py +0 -0
  45. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/szz.py +0 -0
  46. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/textfmt.py +0 -0
  47. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/trend.py +0 -0
  48. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole/watch.py +0 -0
  49. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole.egg-info/dependency_links.txt +0 -0
  50. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole.egg-info/entry_points.txt +0 -0
  51. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole.egg-info/requires.txt +0 -0
  52. {gitmole-0.19.0 → gitmole-0.20.0}/gitmole.egg-info/top_level.txt +0 -0
  53. {gitmole-0.19.0 → gitmole-0.20.0}/pyproject.toml +0 -0
  54. {gitmole-0.19.0 → gitmole-0.20.0}/setup.cfg +0 -0
  55. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_backtest.py +0 -0
  56. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_banner.py +0 -0
  57. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_blame.py +0 -0
  58. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_classify.py +0 -0
  59. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_clean.py +0 -0
  60. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_cli.py +0 -0
  61. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_compare.py +0 -0
  62. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_coupling.py +0 -0
  63. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_duplicates.py +0 -0
  64. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_evaluate.py +0 -0
  65. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_filetypes.py +0 -0
  66. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_findings.py +0 -0
  67. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_functions.py +0 -0
  68. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_golden.py +0 -0
  69. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_hook.py +0 -0
  70. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_hotspots.py +0 -0
  71. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_identity.py +0 -0
  72. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_knowledge.py +0 -0
  73. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_leaks.py +0 -0
  74. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_load.py +0 -0
  75. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_loss.py +0 -0
  76. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_maat.py +0 -0
  77. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_packaging.py +0 -0
  78. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_provenance.py +0 -0
  79. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_render_examples.py +0 -0
  80. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_run.py +0 -0
  81. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_sarif.py +0 -0
  82. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_signing.py +0 -0
  83. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_structure.py +0 -0
  84. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_szz.py +0 -0
  85. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_textfmt.py +0 -0
  86. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_trend.py +0 -0
  87. {gitmole-0.19.0 → gitmole-0.20.0}/tests/test_watch.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gitmole
3
- Version: 0.19.0
3
+ Version: 0.20.0
4
4
  Summary: Offline git repository analysis with a terminal report: hotspots, coupling, ownership, code age, secrets, repo health.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/antvinni/gitmole
@@ -1,3 +1,3 @@
1
1
  """gitmole: offline git repository analysis with a terminal report."""
2
2
 
3
- __version__ = "0.19.0"
3
+ __version__ = "0.20.0"
@@ -47,6 +47,7 @@ def parse_args(argv):
47
47
  p.add_argument("--fail-on", choices=findings.SEVERITIES, help="exit 3 if any finding is at this severity or worse")
48
48
  p.add_argument("--risk", metavar="BASE", help="score the files changed since BASE (merge base with HEAD) by their share of the repository's revisions × lines of code; needs a local path")
49
49
  p.add_argument("--risk-threshold", type=float, metavar="N", help="with --risk: exit 3 when the changed files hold more than N percent of the repository's revisions × lines of code")
50
+ p.add_argument("--sbom", metavar="PATH", help="write a CycloneDX 1.6 SBOM of every locked package to PATH, or - for stdout, from the osv-scanner step's package list")
50
51
  p.add_argument("--compare", metavar="BEFORE_JSON", help="add a 'Since last report' section against an earlier --json export of the same clone")
51
52
  p.add_argument("--hook", action="store_true", help="with --no-run: read an agent hook's JSON on stdin (or files after --), score the files it names like --risk, "
52
53
  "print a summary the agent reads back, exit 2 when --risk-threshold is exceeded")
@@ -84,7 +85,7 @@ def main(argv=None, console: Console = None, tool_check=run.missing_tools, plann
84
85
  if args.clean:
85
86
  return _clean(args, console, ask or (lambda q: console.input(q, markup=False)))
86
87
  # When an export goes to stdout, everything else (banner, progress, report) moves to stderr.
87
- quiet = "-" in (args.json, args.markdown, args.sarif)
88
+ quiet = "-" in (args.json, args.markdown, args.sarif, args.sbom)
88
89
  ui = Console(stderr=True) if quiet else console
89
90
 
90
91
  rc, now = _resolve_time(args, err, ui)
@@ -148,6 +149,7 @@ def _check_args(args, err, kind=None) -> int | None:
148
149
  bad = None
149
150
  else:
150
151
  bad = ("--compare needs one repository, not owner/*" if args.compare and kind == "org" else
152
+ "--sbom needs one repository, not owner/*" if args.sbom and kind == "org" else
151
153
  "--risk needs a local path" if args.risk else
152
154
  "--list-file-types needs a local path" if args.list_file_types else None)
153
155
  if bad:
@@ -598,7 +600,15 @@ def _render(out_dir: str, console: Console, ui: Console, args, err: Console) ->
598
600
  if args.sarif:
599
601
  from . import sarif
600
602
  _write(sarif.dumps(report, found, scope=args.sarif_scope), args.sarif, console)
601
- if "-" not in (args.json, args.markdown, args.sarif):
603
+ if args.sbom:
604
+ from . import sbom
605
+ packages = sbom.read_packages(out_dir)
606
+ if packages is None:
607
+ err.print("[red]--sbom:[/red] no package list in the output directory; the osv-scanner step writes it, and did not "
608
+ f"(dependencies: {(report.get('dependencies') or {}).get('status', 'not-run')}; run.log says why)", soft_wrap=True)
609
+ return 2
610
+ _write(sbom.dumps(report, packages), args.sbom, console)
611
+ if "-" not in (args.json, args.markdown, args.sarif, args.sbom):
602
612
  render.report(report, found, console, full=args.full, risk=risk, base=args.risk, compare=comparison)
603
613
  if args.fail_on and any(findings.SEVERITIES.index(f["severity"]) <= findings.SEVERITIES.index(args.fail_on) for f in found):
604
614
  return 3
@@ -21,6 +21,12 @@ import subprocess
21
21
  import sys
22
22
  import tempfile
23
23
 
24
+ try:
25
+ from . import imports, licences
26
+ except ImportError: # run as a script: the package directory is sys.path[0]
27
+ import imports
28
+ import licences
29
+
24
30
  ARGV = ["osv-scanner", "scan", "source", "-r", "--offline", "--format", "json", "--all-packages", "."]
25
31
  NO_SOURCES = 128 # osv-scanner: no lock file found
26
32
  NO_DATABASE = "no offline version of the OSV database" # its message when the local copy is missing
@@ -141,8 +147,8 @@ def _label(score, vulns: list) -> str:
141
147
 
142
148
  def _relative(path: str, cwd: str) -> str:
143
149
  if os.path.isabs(path):
144
- try:
145
- rel = os.path.relpath(path, cwd)
150
+ try: # real paths on both sides: osv-scanner may report /tmp/... for a cwd of /private/tmp/...
151
+ rel = os.path.relpath(os.path.realpath(path), os.path.realpath(cwd))
146
152
  except ValueError:
147
153
  return path
148
154
  return path if rel.startswith("..") else rel
@@ -174,6 +180,64 @@ def summarise(data: dict, cwd: str) -> dict:
174
180
  return {"status": "scanned", "sources": sources, "packages": packages, "vulnerable": vulnerable}
175
181
 
176
182
 
183
+ PACKAGES = "packages.json" # every locked package, for --sbom; not part of the report, which keeps the vulnerable ones
184
+
185
+
186
+ def packages(data: dict, cwd: str) -> list:
187
+ """Every package the lock files pin, once per ecosystem, name and version, with the lock files that
188
+ pin it and the licence a lock file declares for it (package-lock.json, composer.lock), sorted."""
189
+ declared = {}
190
+ try:
191
+ for d in licences.lock_licences(cwd, [_relative(p, cwd) for p in _lock_paths(data)], every=True):
192
+ declared.setdefault((d["ecosystem"], d["name"], d["version"]), d["expression"])
193
+ except OSError:
194
+ pass
195
+ by_key = {}
196
+ for r in data.get("results") or []:
197
+ path = _relative((r.get("source") or {}).get("path") or "", cwd)
198
+ for p in r.get("packages") or []:
199
+ info = p.get("package") or {}
200
+ key = (info.get("ecosystem", ""), info.get("name", ""), info.get("version", ""))
201
+ row = by_key.setdefault(key, {"ecosystem": key[0], "name": key[1], "version": key[2], "sources": []})
202
+ if path not in row["sources"]:
203
+ row["sources"].append(path)
204
+ out = []
205
+ for key in sorted(by_key):
206
+ row = by_key[key]
207
+ row["sources"].sort()
208
+ if key in declared:
209
+ row["license"] = declared[key]
210
+ out.append(row)
211
+ return out
212
+
213
+
214
+ def _lock_paths(data: dict) -> list:
215
+ return [(r.get("source") or {}).get("path") or "" for r in data.get("results") or []]
216
+
217
+
218
+ # The same scan with the vulnerability matcher switched off: it reads every lock file and matches nothing,
219
+ # so it needs no database. The flag is osv-scanner's experimental plugin switch (2.x); when it fails, there
220
+ # is no package list and --sbom says so.
221
+ LIST_ARGV = ARGV[:-1] + ["--experimental-disable-plugins", "vulnmatch/osvlocal", "."]
222
+
223
+
224
+ def list_packages():
225
+ proc = subprocess.run(LIST_ARGV, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
226
+ if proc.returncode not in (0, 1):
227
+ print(f"deps.py: listing the packages without the database: osv-scanner exited {proc.returncode}", file=sys.stderr)
228
+ return None
229
+ try:
230
+ data = json.loads(proc.stdout.decode("utf-8", "replace") or "{}")
231
+ except json.JSONDecodeError:
232
+ return None
233
+ return data if isinstance(data, dict) else None
234
+
235
+
236
+ def write_packages(rows: list, target: str) -> None:
237
+ with open(target, "w", encoding="utf-8") as fh:
238
+ json.dump({"packages": rows}, fh, indent=0, sort_keys=True)
239
+
240
+
177
241
  def write(result: dict, target: str) -> None:
178
242
  fd, tmp = tempfile.mkstemp(dir=os.path.dirname(os.path.abspath(target)), prefix=".dependencies-", suffix=".json")
179
243
  try:
@@ -198,6 +262,9 @@ def main(argv=None) -> int:
198
262
  result = {"status": "no-sources"}
199
263
  elif NO_DATABASE in err:
200
264
  result = {"status": "no-database", "download": DOWNLOAD}
265
+ listed = list_packages() # the package list for --sbom needs no database
266
+ if listed is not None:
267
+ write_packages(packages(listed, os.getcwd()), os.path.join(os.path.dirname(os.path.abspath(target)), PACKAGES))
201
268
  elif proc.returncode in (0, 1): # 1: packages with vulnerabilities were found
202
269
  text = proc.stdout.decode("utf-8", "replace").strip()
203
270
  try:
@@ -208,6 +275,11 @@ def main(argv=None) -> int:
208
275
  result = summarise(data, os.getcwd())
209
276
  result["database_date"] = database_date()
210
277
  result["database_digest"] = database_digest()
278
+ try:
279
+ imports.annotate(result["vulnerable"], os.getcwd())
280
+ except OSError as e: # the rows stand without it
281
+ print(f"deps.py: imports: {e}", file=sys.stderr)
282
+ write_packages(packages(data, os.getcwd()), os.path.join(os.path.dirname(os.path.abspath(target)), PACKAGES))
211
283
  else:
212
284
  print(f"deps.py: osv-scanner exited {proc.returncode}; no report written", file=sys.stderr)
213
285
  return proc.returncode
@@ -4,7 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  import re
6
6
 
7
- from . import classify, coupling, filetypes, hotspots, knowledge, leaks, loss, maat, textfmt, trend
7
+ from . import classify, coupling, filetypes, hotspots, knowledge, leaks, licences, loss, maat, osps, textfmt, trend
8
8
 
9
9
  SEVERITIES = ["critical", "warning", "info"]
10
10
 
@@ -29,6 +29,8 @@ def _f(severity: str, title: str, statement: str, advice: str, rule: dict, evide
29
29
  between them a reader of the JSON can check the finding without reading this file."""
30
30
  if rule.get("id") in REFS and "ref" not in rule:
31
31
  rule = {**rule, "ref": REFS[rule["id"]]}
32
+ if rule.get("id") in osps.RULE_CONTROLS and "osps" not in rule: # the OSPS Baseline controls it gives evidence for
33
+ rule = {**rule, "osps": list(osps.RULE_CONTROLS[rule["id"]])}
32
34
  return {"severity": severity, "title": title, "detail": f"{statement.rstrip()} {advice}", "advice": advice,
33
35
  "rule": rule, "evidence": evidence}
34
36
 
@@ -698,7 +700,8 @@ def _vuln_statement(rows: list, sources: int) -> str:
698
700
  if r.get("malicious"):
699
701
  score = ", malicious"
700
702
  fixed = f", fixed in {r['fixed']}" if r.get("fixed") else ", no fix yet"
701
- return f"{r['name']} {r['version']} ({ref}{score}{fixed}) in {r['source']}"
703
+ loaded = ", imported by no tracked source" if r.get("imported") is False else ""
704
+ return f"{r['name']} {r['version']} ({ref}{score}{fixed}{loaded}) in {r['source']}"
702
705
  listed = "; ".join(one(r) for r in rows[:3])
703
706
  more = f" and {len(rows) - 3} more" if len(rows) > 3 else ""
704
707
  return f"{_plural(len(rows), 'vulnerable package')} in {_plural(sources, 'lock file')}: {listed}{more}."
@@ -740,7 +743,8 @@ def vulnerable_dependencies(report: dict) -> list:
740
743
  evidence={"lock_files": sources,
741
744
  "packages": [{"name": r["name"], "version": r["version"], "source": r["source"], "score": r.get("score"),
742
745
  "fixed": r.get("fixed") or None, "ids": list(r.get("ids") or []),
743
- "aliases": list(r.get("aliases") or []), "malicious": bool(r.get("malicious"))} for r in group[:10]]}))
746
+ "aliases": list(r.get("aliases") or []), "malicious": bool(r.get("malicious")),
747
+ "imported": r.get("imported", "unknown")} for r in group[:10]]}))
744
748
  return out
745
749
 
746
750
 
@@ -753,7 +757,8 @@ def hygiene_findings(report: dict) -> list:
753
757
  stands in for without the GitHub API. Nothing for an output directory from before the step."""
754
758
  h = report.get("hygiene") or {}
755
759
  out = []
756
- for check in (_hygiene_actions, _hygiene_lockfiles, _hygiene_updates, _hygiene_presence, _hygiene_confusion, _hygiene_install, _hygiene_binaries, _hygiene_submodules, _hygiene_symlinks, _hygiene_trojan):
760
+ for check in (_hygiene_actions, _hygiene_lockfiles, _hygiene_updates, _hygiene_presence, _hygiene_confusion, _hygiene_install, _hygiene_binaries, _hygiene_submodules, _hygiene_symlinks, _hygiene_trojan,
761
+ _hygiene_unused, _hygiene_licence, _hygiene_copyleft):
757
762
  check(h, out)
758
763
  return out
759
764
 
@@ -908,6 +913,59 @@ def _hygiene_trojan(h: dict, out: list) -> None:
908
913
  evidence={"bidi": (tj.get("bidi") or [])[:10], "mixed_script": (tj.get("mixed_script") or [])[:10]}))
909
914
 
910
915
 
916
+ def _hygiene_unused(h: dict, out: list) -> None:
917
+ im = h.get("imports") or {}
918
+ rows = im.get("unused") or []
919
+ if not rows:
920
+ return
921
+ by_manifest = {}
922
+ for r in rows:
923
+ by_manifest.setdefault(r["manifest"], []).append(r["package"])
924
+ listed = "; ".join(f"{_files_list(v)} in {k}" for k, v in list(by_manifest.items())[:3])
925
+ n = im.get("count", len(rows))
926
+ first = rows[0]
927
+ out.append(_f("info", "Declared dependencies nothing imports",
928
+ f"{n} runtime {'dependency is' if n == 1 else 'dependencies are'} declared and never imported by a tracked file, nor named in a script or configuration: {listed}.",
929
+ f"Remove {first['package']} from {first['manifest']} if nothing loads it at run time; an unused dependency is still installed, scanned and updated.",
930
+ rule={"id": "unused_dependencies", "reads": "package.json dependencies, go.mod direct requirements, Cargo.toml [dependencies]"},
931
+ evidence={"count": n, "manifests": im.get("manifests", 0), "unused": rows[:10]}))
932
+
933
+
934
+ def _hygiene_licence(h: dict, out: list) -> None:
935
+ lic = h.get("licences") or {}
936
+ declared = lic.get("declared") or []
937
+ if lic.get("approved") is False or lic.get("mismatch"):
938
+ named = "; ".join(f"{d['source']} declares {d['expression']}" for d in declared)
939
+ parts = []
940
+ if lic.get("mismatch"):
941
+ parts.append(f"{named}, but {lic['files'][0]} is the {lic['file_licence']} text")
942
+ if lic.get("approved") is False:
943
+ parts.append(f"the declared licence ({textfmt.join_and([d['expression'] for d in declared] or [lic.get('file_licence') or ''])}) is not OSI- or FSF-approved")
944
+ out.append(_f("warning" if lic.get("approved") is False else "info", "Project licence as declared", "; ".join(parts) + ".",
945
+ "Make the manifests and the licence file name the same licence; a packager reads the manifest, a lawyer the file." if lic.get("mismatch")
946
+ else "Say so plainly in the README if the project is source-available rather than open source.",
947
+ rule={"id": "project_licence", "reads": "root manifests and the licence file"},
948
+ evidence={"declared": declared, "file": (lic.get("files") or [None])[0], "file_licence": lic.get("file_licence"),
949
+ "approved": lic.get("approved"), "mismatch": bool(lic.get("mismatch"))}))
950
+
951
+
952
+ def _hygiene_copyleft(h: dict, out: list) -> None:
953
+ lic = h.get("licences") or {}
954
+ strong = lic.get("strong") or []
955
+ if not strong or lic.get("project") != licences.PERMISSIVE:
956
+ return
957
+ n = lic.get("strong_count", len(strong))
958
+ listed = _files_list([f"{d['name']} {d['version']} ({d['expression']})" for d in strong])
959
+ own = textfmt.join_and(sorted({d["expression"] for d in lic.get("declared") or []} | ({lic["file_licence"]} if lic.get("file_licence") else set())))
960
+ weak = f" {lic['weak_count']} more declare{'s' if lic.get('weak_count') == 1 else ''} weak copyleft (LGPL, MPL, EPL), which a dependency usually may." if lic.get("weak_count") else ""
961
+ out.append(_f("warning", "Copyleft dependencies in a permissive project",
962
+ f"The project declares {own}, and {n} runtime {'dependency' if n == 1 else 'dependencies'} in {textfmt.join_and(sorted({d['lockfile'] for d in strong}))} "
963
+ f"{'declares' if n == 1 else 'declare'} a strong copyleft licence: {listed}.{weak} Declared, as the lock file records it, not read from the package's files.",
964
+ f"Check whether {strong[0]['name']} is distributed with the project; if it is, its licence terms reach the whole work.",
965
+ rule={"id": "copyleft_dependencies", "reads": "package-lock.json and composer.lock licence fields, runtime packages only"},
966
+ evidence={"project": own, "count": n, "weak": lic.get("weak_count", 0), "dependencies": lic.get("dependencies", 0), "strong": strong[:10]}))
967
+
968
+
911
969
  def _aside_path(path: str) -> bool:
912
970
  return filetypes.is_test_path(path) or filetypes.is_sample_path(path) or filetypes.is_vendor_path(path)
913
971
 
@@ -10,7 +10,8 @@ workflow's `uses:`, a lock file's name, `.gitmodules`, a symlink's mode) or on t
10
10
  - actions: `uses: owner/repo@ref` in .github/workflows where the ref is not a full commit SHA;
11
11
  - lockfiles: a manifest whose last commit is newer than its lock file's, or a manifest with none;
12
12
  - updates: the ecosystems whose lock files are tracked that dependabot.yml does not cover;
13
- - presence: a licence, a security policy, CODEOWNERS and the CODEOWNERS paths that match nothing;
13
+ - presence: a licence, a security policy, a contribution guide, CODEOWNERS and the CODEOWNERS paths
14
+ that match nothing;
14
15
  - confusion: a scoped npm package resolved from a registry other than the one .npmrc declares for
15
16
  its scope, several registries in one lock file, a pip `extra-index-url`;
16
17
  - install: packages with install scripts in package-lock.json, lifecycle scripts in a tracked
@@ -20,7 +21,9 @@ workflow's `uses:`, a lock file's name, `.gitmodules`, a symlink's mode) or on t
20
21
  - submodules: plain http:// or git:// URLs, credentials in a URL, relative URLs, `branch =`;
21
22
  - symlinks: links that resolve outside the tree or into .git/;
22
23
  - trojan: bidirectional control characters (CVE-2021-42574) and identifiers that mix Latin with
23
- Cyrillic, Greek or another confusable script, in source files."""
24
+ Cyrillic, Greek or another confusable script, in source files;
25
+ - licences: the licences the project and its locked dependencies declare (licences.py);
26
+ - imports: declared dependencies that nothing tracked imports (imports.py)."""
24
27
  from __future__ import annotations
25
28
 
26
29
  import ast
@@ -34,9 +37,11 @@ from collections import defaultdict
34
37
  from urllib.parse import urlsplit
35
38
 
36
39
  try:
37
- from . import filetypes
40
+ from . import filetypes, imports, licences
38
41
  except ImportError: # run as a script: the package directory is sys.path[0]
39
42
  import filetypes
43
+ import imports
44
+ import licences
40
45
 
41
46
  CAP = 50 # rows kept per list: the count says how many there were
42
47
 
@@ -197,8 +202,10 @@ def presence(repo: str) -> dict:
197
202
  if p.startswith(prefix) and "/" not in p[len(prefix):] and re.match(names, p[len(prefix):], re.I):
198
203
  return p
199
204
  return None
200
- licence = next((p for p in tracked if "/" not in p and re.match(r"^(licen[cs]e|copying)(\.|-|$)", p, re.I)), None)
205
+ licence = next((p for p in tracked if "/" not in p and re.match(r"^(licen[cs]e|copying)(\.|-|$)", p, re.I)), None) \
206
+ or next(("LICENSES/" for p in tracked if p.startswith("LICENSES/")), None) # the REUSE layout
201
207
  policy = first(r"^security(\.md|\.txt|\.rst)?$")
208
+ contributing = first(r"^contributing(\.md|\.txt|\.rst|\.adoc)?$")
202
209
  owners = first(r"^codeowners$")
203
210
  missing = []
204
211
  if owners:
@@ -209,7 +216,7 @@ def presence(repo: str) -> dict:
209
216
  pattern = line.split()[0]
210
217
  if not _codeowners_matches(pattern, tracked):
211
218
  missing.append(pattern)
212
- return {"license": licence, "security_policy": policy, "codeowners": owners, "codeowners_missing": missing[:CAP]}
219
+ return {"license": licence, "security_policy": policy, "contributing": contributing, "codeowners": owners, "codeowners_missing": missing[:CAP]}
213
220
 
214
221
 
215
222
  # --- dependency confusion -----------------------------------------------------------------------
@@ -483,7 +490,7 @@ def trojan_source(repo: str) -> dict:
483
490
 
484
491
  CHECKS = {"actions": actions_pinning, "lockfiles": lockfiles, "updates": dependency_updates, "presence": presence,
485
492
  "confusion": dependency_confusion, "install": install_scripts, "binaries": binaries, "submodules": submodules,
486
- "symlinks": symlinks, "trojan": trojan_source}
493
+ "symlinks": symlinks, "trojan": trojan_source, "licences": licences.check, "imports": imports.unused}
487
494
 
488
495
 
489
496
  def main(argv=None) -> int:
@@ -0,0 +1,267 @@
1
+ """Which packages the tracked source imports: for a vulnerable package, whether anything loads it at
2
+ all, and the declared dependencies nothing loads. Not reachability (osv-scanner's call analysis needs a
3
+ buildable tree and a toolchain); a textual read of import statements, so it says `imported` and never
4
+ `reachable`, and nothing is suppressed on it.
5
+
6
+ Each ecosystem keys on its own import syntax: `import`/`require` specifiers in JavaScript and
7
+ TypeScript, import paths in Go, `crate::` paths and `extern crate` in Rust, `import`/`from` in Python,
8
+ `require` in Ruby. Where the import name is the package name by the ecosystem's rules (npm, Go
9
+ modules, Rust crates with `-` read as `_`), an absent import is `false`. Where it need not be (a Python
10
+ distribution's modules, a gem's files), a match is `true` and no match is `unknown`: gitmole keeps no
11
+ table of distribution names, so it cannot say PyYAML is imported as yaml. Declared-but-never-imported
12
+ is read for package.json `dependencies`, go.mod direct requirements and Cargo.toml `[dependencies]`
13
+ only, for the same reason."""
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import os
18
+ import re
19
+
20
+ try:
21
+ from . import filetypes
22
+ except ImportError: # run inside a script: the package directory is sys.path[0]
23
+ import filetypes
24
+
25
+ LIMIT = 1_000_000 # bytes read per file: a larger file is generated or data, not code that imports
26
+ JS = (".js", ".jsx", ".mjs", ".cjs", ".ts", ".tsx", ".mts", ".cts", ".vue", ".svelte", ".astro")
27
+ EXTENSIONS = {"npm": JS, "PyPI": (".py", ".pyi"), "Go": (".go",), "crates.io": (".rs",), "RubyGems": (".rb", ".rake", ".gemspec")}
28
+ EXACT = {"npm", "Go", "crates.io"} # the import name is the package name: an absent import is false
29
+
30
+ _JS_SPEC = re.compile(r"""(?:\bfrom\s*|\bimport\s*\(\s*|\bimport\s+|\brequire\s*\(\s*|\brequire\.resolve\s*\(\s*|\bexport\s*\*\s*from\s*)['"]([^'"\s]+)['"]""")
31
+ _PY_IMPORT = re.compile(r"^[ \t]*import[ \t]+([\w., \t]+)", re.M)
32
+ _PY_FROM = re.compile(r"^[ \t]*from[ \t]+(\w[\w.]*)[ \t]+import\b", re.M)
33
+ _GO_BLOCK = re.compile(r"^import\s*\((.*?)^\)", re.M | re.S)
34
+ _GO_ONE = re.compile(r"""^import\s+(?:[\w.]+\s+)?"([^"]+)\"""", re.M)
35
+ _GO_QUOTED = re.compile(r'"([^"]+)"')
36
+ _GO_GENERATE = re.compile(r"^//go:generate\s+(.*)$", re.M)
37
+ _RS_PATH = re.compile(r"\b([A-Za-z_][A-Za-z0-9_]*)::")
38
+ _RS_CRATE = re.compile(r"\b(?:extern\s+crate|use)\s+(?:::)?([A-Za-z_][A-Za-z0-9_]*)")
39
+ _RB_REQUIRE = re.compile(r"""\brequire(?:_relative)?\s*\(?\s*['"]([^'"]+)['"]""")
40
+
41
+
42
+ def js_package(spec: str) -> str | None:
43
+ """The npm package a specifier loads: `@scope/name` or `name`, without the subpath; None for a
44
+ relative path, an absolute one, a URL or a node: builtin."""
45
+ if not spec or spec[0] in "./#~" or ":" in spec.split("/", 1)[0]:
46
+ return None
47
+ parts = spec.split("/")
48
+ if spec.startswith("@"):
49
+ return "/".join(parts[:2]) if len(parts) > 1 else None
50
+ return parts[0]
51
+
52
+
53
+ def python_name(name: str) -> str:
54
+ """PEP 503's normal form, with `_` as the separator so a module name compares with it."""
55
+ return re.sub(r"[-_.]+", "_", name).lower()
56
+
57
+
58
+ def rust_name(name: str) -> str:
59
+ return name.replace("-", "_")
60
+
61
+
62
+ def _read(repo: str, path: str) -> str:
63
+ try:
64
+ with open(os.path.join(repo, path), "rb") as fh:
65
+ data = fh.read(LIMIT + 1)
66
+ except OSError:
67
+ return ""
68
+ if len(data) > LIMIT or b"\0" in data[:8000]:
69
+ return ""
70
+ return data.decode("utf-8", "replace")
71
+
72
+
73
+ def scan_text(ecosystem: str, text: str) -> set:
74
+ """The names one file imports, in the form the ecosystem's packages are compared in."""
75
+ found = set()
76
+ if ecosystem == "npm":
77
+ found.update(p for p in (js_package(s) for s in _JS_SPEC.findall(text)) if p)
78
+ elif ecosystem == "PyPI":
79
+ for group in _PY_IMPORT.findall(text):
80
+ for item in group.split(","):
81
+ word = item.strip().split(" ")[0].split(".")[0]
82
+ if word:
83
+ found.add(python_name(word))
84
+ found.update(python_name(m.split(".")[0]) for m in _PY_FROM.findall(text))
85
+ elif ecosystem == "Go":
86
+ for block in _GO_BLOCK.findall(text):
87
+ found.update(_GO_QUOTED.findall(block))
88
+ found.update(_GO_ONE.findall(text))
89
+ for line in _GO_GENERATE.findall(text): # a tool run by go:generate is a use of its module
90
+ found.update(w for w in line.split() if "." in w.split("/")[0] and "/" in w)
91
+ elif ecosystem == "crates.io":
92
+ found.update(_RS_PATH.findall(text))
93
+ found.update(_RS_CRATE.findall(text))
94
+ elif ecosystem == "RubyGems":
95
+ found.update(_RB_REQUIRE.findall(text))
96
+ return found
97
+
98
+
99
+ def scan(repo: str, ecosystems, paths: list = None) -> dict:
100
+ """{ecosystem: the names imported anywhere in the tracked files of that ecosystem's extensions}.
101
+ node_modules is left out: somebody else's imports."""
102
+ paths = filetypes.git_paths(repo, "ls-files") if paths is None else paths
103
+ wanted = {e: EXTENSIONS[e] for e in ecosystems if e in EXTENSIONS}
104
+ out = {e: set() for e in wanted}
105
+ for path in paths:
106
+ if "node_modules/" in path:
107
+ continue
108
+ low = path.lower()
109
+ for eco, exts in wanted.items():
110
+ if low.endswith(exts):
111
+ out[eco] |= scan_text(eco, _read(repo, path))
112
+ return out
113
+
114
+
115
+ def _go_match(module: str, imported: set) -> bool:
116
+ return any(p == module or p.startswith(module + "/") for p in imported)
117
+
118
+
119
+ def is_imported(ecosystem: str, name: str, imported: dict):
120
+ """True, False or None (unknown) for one package against scan()'s result."""
121
+ names = imported.get(ecosystem)
122
+ if names is None:
123
+ return None
124
+ if ecosystem == "npm":
125
+ hit = name in names
126
+ elif ecosystem == "Go":
127
+ if name in ("stdlib", "toolchain"):
128
+ return None
129
+ hit = _go_match(name, names)
130
+ elif ecosystem == "crates.io":
131
+ hit = rust_name(name) in names
132
+ elif ecosystem == "PyPI":
133
+ hit = python_name(name) in names
134
+ elif ecosystem == "RubyGems":
135
+ hit = any(r == name or r.split("/")[0] == name or r == name.replace("-", "/") for r in names)
136
+ else:
137
+ return None
138
+ if hit:
139
+ return True
140
+ return False if ecosystem in EXACT else None
141
+
142
+
143
+ def annotate(rows: list, repo: str, paths: list = None) -> None:
144
+ """Set `imported` to true, false or "unknown" on each vulnerable-package row, in place."""
145
+ if not rows:
146
+ return
147
+ imported = scan(repo, {r.get("ecosystem") for r in rows}, paths)
148
+ for r in rows:
149
+ value = is_imported(r.get("ecosystem"), r.get("name", ""), imported)
150
+ r["imported"] = "unknown" if value is None else value
151
+
152
+
153
+ # --- declared, never imported -------------------------------------------------------------------
154
+
155
+ def _aside(path: str) -> bool:
156
+ return ("node_modules/" in path or filetypes.is_test_path(path) or filetypes.is_sample_path(path)
157
+ or filetypes.is_vendor_path(path) or filetypes.is_doc_path(path))
158
+
159
+
160
+ def npm_declared(text: str) -> tuple:
161
+ """(the runtime `dependencies` of a package.json, its scripts' text). Type-only packages
162
+ (`@types/`) are left out: nothing imports them by name."""
163
+ try:
164
+ data = json.loads(text)
165
+ except ValueError:
166
+ return [], ""
167
+ if not isinstance(data, dict):
168
+ return [], ""
169
+ deps = data.get("dependencies") or {}
170
+ scripts = data.get("scripts") or {}
171
+ names = sorted(n for n in deps if isinstance(n, str) and not n.startswith("@types/")) if isinstance(deps, dict) else []
172
+ return names, " ".join(str(v) for v in scripts.values()) if isinstance(scripts, dict) else ""
173
+
174
+
175
+ def go_declared(text: str) -> list:
176
+ """go.mod's direct requirements: the `require` lines without `// indirect`."""
177
+ out, block = [], False
178
+ for raw in text.split("\n"):
179
+ line = raw.strip()
180
+ if line.startswith("require ("):
181
+ block = True
182
+ continue
183
+ if block and line.startswith(")"):
184
+ block = False
185
+ continue
186
+ body = line[len("require "):] if line.startswith("require ") else line if block else None
187
+ if body is None or not body or body.startswith("//") or "// indirect" in body:
188
+ continue
189
+ parts = body.split()
190
+ if len(parts) >= 2:
191
+ out.append(parts[0])
192
+ return sorted(set(out))
193
+
194
+
195
+ _TOML_HEADER = re.compile(r"^\s*\[+\s*([^\]]+?)\s*\]+\s*(?:#.*)?$")
196
+ _TOML_KEY = re.compile(r"""^\s*("?)([A-Za-z0-9_.-]+)\1\s*=""")
197
+
198
+
199
+ def cargo_declared(text: str) -> list:
200
+ """Cargo.toml's `[dependencies]`, target-specific ones included, as the names the code uses: the
201
+ table key, which a `package =` rename makes the import name. Build and dev dependencies are left
202
+ out, and so are `-sys` crates, which are linked for their native library and not named in code."""
203
+ out, table = [], ""
204
+ for line in text.split("\n"):
205
+ h = _TOML_HEADER.match(line)
206
+ if h:
207
+ table = h.group(1).replace('"', "").replace("'", "")
208
+ if re.match(r"^(target\..+\.)?dependencies\.[A-Za-z0-9_-]+$", table): # [dependencies.name]
209
+ out.append(table.rsplit(".", 1)[1])
210
+ continue
211
+ if re.match(r"^(target\..+\.)?dependencies$", table):
212
+ k = _TOML_KEY.match(line)
213
+ if k:
214
+ out.append(k.group(2).split(".")[0]) # `name.workspace = true` is a dotted key
215
+ return sorted({n for n in out if not n.endswith(("-sys", "_sys"))})
216
+
217
+
218
+ def _quoted_heads(text: str) -> set:
219
+ """Every quoted string's package head, for configuration files that name a package without
220
+ importing it (a Babel preset, an ESLint plugin, a Jest environment)."""
221
+ out = set()
222
+ for s in re.findall(r"""['"`]([@\w][\w@./-]*)['"`]""", text):
223
+ p = js_package(s)
224
+ if p:
225
+ out.add(p)
226
+ return out
227
+
228
+
229
+ def _manifest_or_lock(path: str) -> bool:
230
+ """A manifest or lock file names every dependency in quotes, so it cannot count as a use of one."""
231
+ base = os.path.basename(path).lower()
232
+ return base in ("package.json", "composer.json", "deno.json") or "lock" in base or base == "npm-shrinkwrap.json"
233
+
234
+
235
+ def unused(repo: str, paths: list = None) -> dict:
236
+ """Declared runtime dependencies that no tracked file imports: for npm, not imported, not named in
237
+ a quoted string of any JavaScript, TypeScript, JSON or YAML file other than a manifest or lock file,
238
+ and not a word in the manifest's scripts; for Go, no import path or go:generate line under the module; for Rust, no `name::` path,
239
+ `use name` or `extern crate name`. Manifests under tests, examples, documentation and vendored
240
+ code are left out."""
241
+ paths = filetypes.git_paths(repo, "ls-files") if paths is None else paths
242
+ manifests = [p for p in paths if os.path.basename(p) in ("package.json", "go.mod", "Cargo.toml") and not _aside(p)]
243
+ if not manifests:
244
+ return {"manifests": 0, "unused": [], "count": 0}
245
+ kinds = {os.path.basename(p) for p in manifests}
246
+ wanted = ({"npm"} if "package.json" in kinds else set()) | ({"Go"} if "go.mod" in kinds else set()) | ({"crates.io"} if "Cargo.toml" in kinds else set())
247
+ imported = scan(repo, wanted, paths)
248
+ if "npm" in wanted:
249
+ named = set()
250
+ for p in paths:
251
+ if "node_modules/" not in p and p.lower().endswith(JS + (".json", ".jsonc", ".json5", ".yml", ".yaml")) and not _manifest_or_lock(p):
252
+ named |= _quoted_heads(_read(repo, p))
253
+ imported["npm"] = imported["npm"] | named
254
+ out = []
255
+ for m in sorted(manifests):
256
+ text = _read(repo, m)
257
+ base = os.path.basename(m)
258
+ if base == "package.json":
259
+ names, scripts = npm_declared(text)
260
+ words = set(re.findall(r"[\w@./-]+", scripts))
261
+ out += [{"manifest": m, "ecosystem": "npm", "package": n} for n in names
262
+ if n not in imported["npm"] and n not in words and n.rsplit("/", 1)[-1] not in words]
263
+ elif base == "go.mod":
264
+ out += [{"manifest": m, "ecosystem": "Go", "package": n} for n in go_declared(text) if not _go_match(n, imported["Go"])]
265
+ else:
266
+ out += [{"manifest": m, "ecosystem": "crates.io", "package": n} for n in cargo_declared(text) if rust_name(n) not in imported["crates.io"]]
267
+ return {"manifests": len(manifests), "unused": out[:50], "count": len(out)}