@yottameta/yotta-chain 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/LICENSE +21 -0
- package/NOTICE +11 -0
- package/README.md +151 -0
- package/README.zh-CN.md +151 -0
- package/SKILL.md +110 -0
- package/assets/banner.png +0 -0
- package/bin/install.js +163 -0
- package/install.sh +132 -0
- package/package.json +33 -0
- package/references/rules.md +124 -0
- package/scripts/test_yotta_chain.py +616 -0
- package/scripts/yotta_chain.py +1794 -0
|
@@ -0,0 +1,1794 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""yotta-chain (yuan lian) - supply chain dependency validation engine.
|
|
4
|
+
|
|
5
|
+
Zero-dependency Python 3.8+ implementation (standard library only).
|
|
6
|
+
|
|
7
|
+
Scope (v0.1.0):
|
|
8
|
+
- npm : package.json + package-lock.json (v1/v2/v3) + npm-shrinkwrap.json + .npmrc
|
|
9
|
+
- python: requirements*.txt + pyproject.toml (PEP 621 / poetry) + poetry.lock
|
|
10
|
+
+ Pipfile / Pipfile.lock
|
|
11
|
+
- maven : pom.xml (basic: unpinned / SNAPSHOT / suspicious repository URLs)
|
|
12
|
+
|
|
13
|
+
Checks:
|
|
14
|
+
- dependency confusion : registry config vs lockfile resolved URLs, mixed
|
|
15
|
+
registries, suspicious (http / IP-literal / localhost) registry URLs,
|
|
16
|
+
extra-index fallback mixing
|
|
17
|
+
- lockfile consistency : manifest entry missing, range unsatisfied, root
|
|
18
|
+
mismatch, dangling references, missing integrity, duplicate conflicts
|
|
19
|
+
- hygiene : missing lockfile, unpinned versions, SNAPSHOT
|
|
20
|
+
- typosquat : name resembles a popular package (edit distance)
|
|
21
|
+
|
|
22
|
+
No online CVE lookups; fully local and offline. SBOM-lite output is a
|
|
23
|
+
CycloneDX 1.5 JSON subset.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
import argparse
|
|
27
|
+
import json
|
|
28
|
+
import os
|
|
29
|
+
import re
|
|
30
|
+
import sys
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
try:
|
|
34
|
+
sys.stdout.reconfigure(encoding="utf-8")
|
|
35
|
+
except Exception:
|
|
36
|
+
pass
|
|
37
|
+
|
|
38
|
+
VERSION = "0.1.0"
|
|
39
|
+
|
|
40
|
+
SEVERITY_ORDER = ["info", "low", "medium", "high"]
|
|
41
|
+
SEVERITY_RANK = {s: i for i, s in enumerate(SEVERITY_ORDER)}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class Finding:
|
|
45
|
+
"""A single validation finding."""
|
|
46
|
+
|
|
47
|
+
__slots__ = ("rule", "severity", "file", "line", "package", "message", "detail", "ecosystem")
|
|
48
|
+
|
|
49
|
+
def __init__(self, rule, severity, file, package, message, detail="", line=None, ecosystem=""):
|
|
50
|
+
self.rule = rule
|
|
51
|
+
self.severity = severity
|
|
52
|
+
self.file = file
|
|
53
|
+
self.line = line
|
|
54
|
+
self.package = package
|
|
55
|
+
self.message = message
|
|
56
|
+
self.detail = detail
|
|
57
|
+
self.ecosystem = ecosystem
|
|
58
|
+
|
|
59
|
+
def to_dict(self):
|
|
60
|
+
return {
|
|
61
|
+
"rule": self.rule,
|
|
62
|
+
"severity": self.severity,
|
|
63
|
+
"file": self.file,
|
|
64
|
+
"line": self.line,
|
|
65
|
+
"package": self.package,
|
|
66
|
+
"message": self.message,
|
|
67
|
+
"detail": self.detail,
|
|
68
|
+
"ecosystem": self.ecosystem,
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# --------------------------------------------------------------------------
|
|
73
|
+
# small text / url helpers
|
|
74
|
+
# --------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
def _read_text(path):
|
|
77
|
+
try:
|
|
78
|
+
return Path(path).read_text(encoding="utf-8", errors="replace")
|
|
79
|
+
except Exception:
|
|
80
|
+
return ""
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _url_host(url):
|
|
84
|
+
"""Return (scheme, host, port) for a registry/package URL, else None."""
|
|
85
|
+
if not url:
|
|
86
|
+
return None
|
|
87
|
+
m = re.match(r"^(https?)://([^/:]+)(?::(\d+))?", url.strip())
|
|
88
|
+
if not m:
|
|
89
|
+
return None
|
|
90
|
+
return m.group(1), m.group(2), m.group(3)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
_IPV4_RE = re.compile(r"^\d{1,3}(\.\d{1,3}){3}$")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _is_suspicious_url(url):
|
|
97
|
+
"""Return (bool, reason) for http://, IP-literal, localhost hosts."""
|
|
98
|
+
if not url:
|
|
99
|
+
return False, ""
|
|
100
|
+
host = _url_host(url)
|
|
101
|
+
if host is None:
|
|
102
|
+
return False, ""
|
|
103
|
+
scheme, h, _port = host
|
|
104
|
+
if scheme != "https":
|
|
105
|
+
return True, "非 HTTPS 的仓库地址(http://),传输与完整性易被篡改"
|
|
106
|
+
if h in ("localhost", "127.0.0.1", "::1", "0.0.0.0"):
|
|
107
|
+
return True, "仓库地址指向本机(%s),正常发布环境不应使用" % h
|
|
108
|
+
if _IPV4_RE.match(h):
|
|
109
|
+
return True, "仓库地址使用 IP 字面量(%s),建议改用域名并校验来源" % h
|
|
110
|
+
return False, ""
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _norm_pkg(name):
|
|
114
|
+
return (name or "").strip().lower()
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# --------------------------------------------------------------------------
|
|
118
|
+
# Damerau-Levenshtein (for typosquat)
|
|
119
|
+
# --------------------------------------------------------------------------
|
|
120
|
+
|
|
121
|
+
def _damerau_levenshtein(a, b):
|
|
122
|
+
"""Return edit distance with adjacent transpositions (DL distance)."""
|
|
123
|
+
la, lb = len(a), len(b)
|
|
124
|
+
if abs(la - lb) > 2:
|
|
125
|
+
return abs(la - lb) + 1
|
|
126
|
+
d = [[0] * (lb + 1) for _ in range(la + 1)]
|
|
127
|
+
for i in range(la + 1):
|
|
128
|
+
d[i][0] = i
|
|
129
|
+
for j in range(lb + 1):
|
|
130
|
+
d[0][j] = j
|
|
131
|
+
for i in range(1, la + 1):
|
|
132
|
+
for j in range(1, lb + 1):
|
|
133
|
+
cost = 0 if a[i - 1] == b[j - 1] else 1
|
|
134
|
+
d[i][j] = min(
|
|
135
|
+
d[i - 1][j] + 1,
|
|
136
|
+
d[i][j - 1] + 1,
|
|
137
|
+
d[i - 1][j - 1] + cost,
|
|
138
|
+
)
|
|
139
|
+
if i > 1 and j > 1 and a[i - 1] == b[j - 2] and a[i - 2] == b[j - 1]:
|
|
140
|
+
d[i][j] = min(d[i][j], d[i - 2][j - 2] + 1)
|
|
141
|
+
return d[la][lb]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
POPULAR_NPM = [
|
|
145
|
+
"lodash", "express", "react", "request", "axios", "chalk", "commander",
|
|
146
|
+
"debug", "dotenv", "eslint", "moment", "uuid", "yargs", "webpack",
|
|
147
|
+
"typescript", "jest", "babel-core", "vue", "react-dom", "bluebird",
|
|
148
|
+
"underscore", "minimist", "semver", "rimraf", "mkdirp", "glob", "async",
|
|
149
|
+
"q", "inherits", "readable-stream", "string-width", "ansi-regex",
|
|
150
|
+
"is-number", "isarray", "tslib", "core-js", "node-fetch", "fast-glob",
|
|
151
|
+
"postcss", "autoprefixer", "esbuild", "rollup", "vite", "next", "gulp",
|
|
152
|
+
"grunt", "webpack-cli", "cross-env", "concurrently", "nodemon",
|
|
153
|
+
"prettier", "husky", "lint-staged", "js-yaml", "yaml", "fs-extra",
|
|
154
|
+
"supports-color", "picocolors", "schema-utils", "serialize-javascript",
|
|
155
|
+
"source-map", "path-exists", "p-limit", "safe-buffer", "util-deprecate",
|
|
156
|
+
"brace-expansion", "balanced-match", "has-flag", "color-convert",
|
|
157
|
+
]
|
|
158
|
+
|
|
159
|
+
POPULAR_PYPI = [
|
|
160
|
+
"requests", "urllib3", "flask", "django", "numpy", "pandas", "scipy",
|
|
161
|
+
"pytest", "setuptools", "pip", "wheel", "cryptography", "pyyaml",
|
|
162
|
+
"beautifulsoup4", "lxml", "jinja2", "sqlalchemy", "fastapi", "tornado",
|
|
163
|
+
"aiohttp", "click", "tqdm", "matplotlib", "pillow", "six",
|
|
164
|
+
"python-dateutil", "idna", "certifi", "charset-normalizer",
|
|
165
|
+
"typing-extensions", "pydantic", "starlette", "uvicorn", "greenlet",
|
|
166
|
+
"markupsafe", "packaging", "docutils", "sphinx", "celery", "redis",
|
|
167
|
+
"boto3", "botocore", "kubernetes", "grpcio", "protobuf", "docker",
|
|
168
|
+
"openpyxl", "xlsxwriter", "mypy", "black", "ruff", "coverage", "isort",
|
|
169
|
+
]
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def find_typosquat(name, popular):
|
|
173
|
+
"""Return (legit_name, distance) if name resembles a popular package."""
|
|
174
|
+
n = _norm_pkg(name)
|
|
175
|
+
if not n or n in popular:
|
|
176
|
+
return None
|
|
177
|
+
best = None
|
|
178
|
+
best_d = 99
|
|
179
|
+
for p in popular:
|
|
180
|
+
d = _damerau_levenshtein(n, p)
|
|
181
|
+
if d < best_d:
|
|
182
|
+
best_d = d
|
|
183
|
+
best = p
|
|
184
|
+
if best_d <= 1:
|
|
185
|
+
break
|
|
186
|
+
if best is not None and best_d <= 2 and abs(len(n) - len(best)) <= 2:
|
|
187
|
+
return best, best_d
|
|
188
|
+
# direct concatenation variant: name == popular without hyphen separators
|
|
189
|
+
for p in popular:
|
|
190
|
+
if n == p.replace("-", "") and len(p) >= 5:
|
|
191
|
+
return p, 1
|
|
192
|
+
return None
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
# --------------------------------------------------------------------------
|
|
196
|
+
# npm semver (subset sufficient for range checks)
|
|
197
|
+
# --------------------------------------------------------------------------
|
|
198
|
+
|
|
199
|
+
_RE_SEMVER = re.compile(
|
|
200
|
+
r"^v?(\d+)(?:\.(\d+))?(?:\.(\d+))?(?:-([0-9A-Za-z.-]+))?(?:\+[0-9A-Za-z.-]+)?$"
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def parse_version(v):
|
|
205
|
+
"""Parse an npm-style version into (major, minor, patch, prerelease)."""
|
|
206
|
+
if not v:
|
|
207
|
+
return None
|
|
208
|
+
m = _RE_SEMVER.match(v.strip())
|
|
209
|
+
if not m:
|
|
210
|
+
return None
|
|
211
|
+
return (
|
|
212
|
+
int(m.group(1)),
|
|
213
|
+
int(m.group(2) or 0),
|
|
214
|
+
int(m.group(3) or 0),
|
|
215
|
+
m.group(4),
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _pre_key(pre):
|
|
220
|
+
if pre is None:
|
|
221
|
+
return (1 << 30,)
|
|
222
|
+
parts = []
|
|
223
|
+
for p in pre.split("."):
|
|
224
|
+
parts.append(int(p) if p.isdigit() else p)
|
|
225
|
+
return tuple(parts)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _ver_key(v):
|
|
229
|
+
return (v[0], v[1], v[2], _pre_key(v[3]))
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _cmp_ver(a, b):
|
|
233
|
+
if a is None or b is None:
|
|
234
|
+
return 0
|
|
235
|
+
ka, kb = _ver_key(a), _ver_key(b)
|
|
236
|
+
return (ka > kb) - (ka < kb)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _parse_ver_part(s):
|
|
240
|
+
"""Parse a version token that may be partial (1, 1.2, 1.2.x, *) or exact (1.2.3)."""
|
|
241
|
+
s = s.strip().lower()
|
|
242
|
+
if s in ("*", "x", "latest", ""):
|
|
243
|
+
return "any", None
|
|
244
|
+
if re.fullmatch(r"\d+", s):
|
|
245
|
+
return "range", (int(s), None, None)
|
|
246
|
+
if re.fullmatch(r"\d+\.\d+", s):
|
|
247
|
+
mj, mn = s.split(".")
|
|
248
|
+
return "range", (int(mj), int(mn), None)
|
|
249
|
+
if re.fullmatch(r"\d+\.\d+\.\d+", s):
|
|
250
|
+
return "exact", parse_version(s)
|
|
251
|
+
m = re.fullmatch(r"(\d+)(?:\.(\d+))?\.(x|\*)", s)
|
|
252
|
+
if m:
|
|
253
|
+
mj = int(m.group(1))
|
|
254
|
+
mn = int(m.group(2)) if m.group(2) else None
|
|
255
|
+
return "range", (mj, mn, None)
|
|
256
|
+
return "exact", parse_version(s)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _cmp_partial(ver, part):
|
|
261
|
+
"""Compare version tuple against a partial (mj, mn, pt|None) tuple."""
|
|
262
|
+
mj, mn, pt = part
|
|
263
|
+
if ver[0] != mj:
|
|
264
|
+
return (ver[0] > mj) - (ver[0] < mj)
|
|
265
|
+
if mn is not None:
|
|
266
|
+
if ver[1] != mn:
|
|
267
|
+
return (ver[1] > mn) - (ver[1] < mn)
|
|
268
|
+
if pt is not None and ver[2] != pt:
|
|
269
|
+
return (ver[2] > pt) - (ver[2] < pt)
|
|
270
|
+
return 0
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _lower_full(base):
|
|
274
|
+
mj, mn, pt = base
|
|
275
|
+
return (mj, mn if mn is not None else 0, pt if pt is not None else 0, None)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _caret_upper(base):
|
|
279
|
+
mj, mn, pt = base
|
|
280
|
+
if mn is None:
|
|
281
|
+
return (mj + 1, 0, 0)
|
|
282
|
+
if mj > 0:
|
|
283
|
+
return (mj + 1, 0, 0)
|
|
284
|
+
if pt is None:
|
|
285
|
+
return (0, mn + 1, 0)
|
|
286
|
+
if mn > 0:
|
|
287
|
+
return (0, mn + 1, 0)
|
|
288
|
+
return (0, 0, pt + 1)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _tilde_upper(base):
|
|
292
|
+
mj, mn, _pt = base
|
|
293
|
+
if mn is None:
|
|
294
|
+
return (mj + 1, 0, 0)
|
|
295
|
+
return (mj, mn + 1, 0)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _semver_satisfies_one(ver, tok):
|
|
299
|
+
"""Satisfy a single comparator token (already expanded, no OR)."""
|
|
300
|
+
tok = tok.strip()
|
|
301
|
+
if not tok:
|
|
302
|
+
return True
|
|
303
|
+
m = re.match(r"^(>=|<=|>|<|=|~|\^)?\s*(.+)$", tok)
|
|
304
|
+
op = m.group(1) or ""
|
|
305
|
+
rest = m.group(2).strip()
|
|
306
|
+
kind, base = _parse_ver_part(rest)
|
|
307
|
+
if kind == "any":
|
|
308
|
+
return True
|
|
309
|
+
if kind == "range":
|
|
310
|
+
# partial version: 1 / 1.2 / 1.x / 1.2.x
|
|
311
|
+
if op == "^":
|
|
312
|
+
return _cmp_ver(ver, _lower_full(base)) >= 0 and _cmp_partial(ver, _caret_upper(base)) < 0
|
|
313
|
+
if op == "~":
|
|
314
|
+
return _cmp_ver(ver, _lower_full(base)) >= 0 and _cmp_partial(ver, _tilde_upper(base)) < 0
|
|
315
|
+
if op in ("", "="):
|
|
316
|
+
return _cmp_partial(ver, base) == 0
|
|
317
|
+
if op == ">=":
|
|
318
|
+
return _cmp_ver(ver, _lower_full(base)) >= 0
|
|
319
|
+
if op == ">":
|
|
320
|
+
return _cmp_ver(ver, _lower_full(base)) > 0
|
|
321
|
+
if op == "<=":
|
|
322
|
+
return _cmp_partial(ver, base) <= 0
|
|
323
|
+
if op == "<":
|
|
324
|
+
return _cmp_partial(ver, base) < 0
|
|
325
|
+
return _cmp_partial(ver, base) == 0
|
|
326
|
+
if kind == "exact":
|
|
327
|
+
if base is None:
|
|
328
|
+
return False
|
|
329
|
+
if op in ("", "="):
|
|
330
|
+
return _cmp_ver(ver, base) == 0
|
|
331
|
+
if op == ">=":
|
|
332
|
+
return _cmp_ver(ver, base) >= 0
|
|
333
|
+
if op == "<=":
|
|
334
|
+
return _cmp_ver(ver, base) <= 0
|
|
335
|
+
if op == ">":
|
|
336
|
+
return _cmp_ver(ver, base) > 0
|
|
337
|
+
if op == "<":
|
|
338
|
+
return _cmp_ver(ver, base) < 0
|
|
339
|
+
if op == "~":
|
|
340
|
+
return _cmp_ver(ver, base) >= 0 and _cmp_partial(ver, _tilde_upper(base[:3])) < 0
|
|
341
|
+
if op == "^":
|
|
342
|
+
return _cmp_ver(ver, base) >= 0 and _cmp_partial(ver, _caret_upper(base[:3])) < 0
|
|
343
|
+
return False
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def semver_satisfies(version, range_str):
|
|
348
|
+
"""npm-style range satisfaction (^ ~ >= <= > < =, x-ranges, ||, hyphen)."""
|
|
349
|
+
if version is None:
|
|
350
|
+
return False
|
|
351
|
+
ver = parse_version(version)
|
|
352
|
+
if ver is None:
|
|
353
|
+
return False
|
|
354
|
+
range_str = (range_str or "").strip()
|
|
355
|
+
if not range_str or range_str in ("*", "latest", "x", ""):
|
|
356
|
+
return True
|
|
357
|
+
for alt in range_str.split("||"):
|
|
358
|
+
if _semver_satisfies_alt(ver, alt.strip()):
|
|
359
|
+
return True
|
|
360
|
+
return False
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _semver_satisfies_alt(ver, alt):
|
|
364
|
+
# hyphen range: "1.2.3 - 2.3.4"
|
|
365
|
+
m = re.match(r"^(\S+)\s+-\s+(\S+)$", alt)
|
|
366
|
+
if m:
|
|
367
|
+
lo = parse_version(m.group(1))
|
|
368
|
+
hi = parse_version(m.group(2))
|
|
369
|
+
if lo and hi:
|
|
370
|
+
return _cmp_ver(ver, lo) >= 0 and _cmp_ver(ver, hi) <= 0
|
|
371
|
+
toks = re.split(r"[\s,]+", alt)
|
|
372
|
+
toks = [t for t in toks if t and t != "-"]
|
|
373
|
+
if not toks:
|
|
374
|
+
return True
|
|
375
|
+
# npm rule: prerelease versions are excluded unless a comparator has
|
|
376
|
+
# a prerelease on the same [major, minor, patch].
|
|
377
|
+
if ver[3]:
|
|
378
|
+
same = False
|
|
379
|
+
for t in toks:
|
|
380
|
+
mm = re.match(r"^(>=|<=|>|<|=|~|\^)?\s*(.+)$", t)
|
|
381
|
+
rest = mm.group(2).strip() if mm else t
|
|
382
|
+
b = parse_version(rest)
|
|
383
|
+
if b and b[3] and b[0:3] == ver[0:3]:
|
|
384
|
+
same = True
|
|
385
|
+
break
|
|
386
|
+
if not same:
|
|
387
|
+
return False
|
|
388
|
+
return all(_semver_satisfies_one(ver, t) for t in toks)
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
# --------------------------------------------------------------------------
|
|
392
|
+
# PEP 440 (python versions / specifiers, basic)
|
|
393
|
+
# --------------------------------------------------------------------------
|
|
394
|
+
|
|
395
|
+
_RE_PEP440 = re.compile(
|
|
396
|
+
r"^v?(\d+)(?:\.(\d+))?(?:\.(\d+))?(?:\.(\d+))?"
|
|
397
|
+
r"(?:(?:a|b|rc)(\d+))?(?:\.?(?:post|rev|r)(\d+))?"
|
|
398
|
+
r"(?:\.?dev(\d+))?(?:\+[0-9A-Za-z.-]+)?$"
|
|
399
|
+
)
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def parse_pep440(v):
|
|
403
|
+
"""Parse a PEP 440 version into (release, pre, post, dev)."""
|
|
404
|
+
if not v:
|
|
405
|
+
return None
|
|
406
|
+
s = v.strip().lstrip("v")
|
|
407
|
+
m = _RE_PEP440.match(s)
|
|
408
|
+
if not m:
|
|
409
|
+
return None
|
|
410
|
+
rel = tuple(int(x or 0) for x in m.group(1, 2, 3, 4))
|
|
411
|
+
pre = m.group(5)
|
|
412
|
+
post = m.group(6)
|
|
413
|
+
dev = m.group(7)
|
|
414
|
+
return rel, pre, post, dev
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _pep_key(v):
|
|
418
|
+
rel, pre, post, dev = v
|
|
419
|
+
if dev is not None:
|
|
420
|
+
return (rel, -3, int(dev), 0)
|
|
421
|
+
if pre is not None:
|
|
422
|
+
return (rel, -2, int(pre), 0)
|
|
423
|
+
if post is not None:
|
|
424
|
+
return (rel, 1, int(post), 0)
|
|
425
|
+
return (rel, 0, 0, 0)
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _pep_cmp(a, b):
|
|
429
|
+
if a is None or b is None:
|
|
430
|
+
return 0
|
|
431
|
+
ka, kb = _pep_key(a), _pep_key(b)
|
|
432
|
+
return (ka > kb) - (ka < kb)
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def _pep_match_operator(ver, op, rest):
|
|
436
|
+
rest = rest.strip()
|
|
437
|
+
if op == "==" and rest.endswith(".*"):
|
|
438
|
+
prefix = rest[:-2].strip().lstrip("v")
|
|
439
|
+
parts = [p for p in prefix.split(".") if p != ""]
|
|
440
|
+
rel = ver[0]
|
|
441
|
+
if len(parts) > len(rel):
|
|
442
|
+
return False
|
|
443
|
+
for i, p in enumerate(parts):
|
|
444
|
+
if not p.isdigit() or int(p) != rel[i]:
|
|
445
|
+
return False
|
|
446
|
+
return True
|
|
447
|
+
b = parse_pep440(rest)
|
|
448
|
+
if b is None:
|
|
449
|
+
return False
|
|
450
|
+
c = _pep_cmp(ver, b)
|
|
451
|
+
if op in ("==", ""):
|
|
452
|
+
return c == 0
|
|
453
|
+
if op == "!=":
|
|
454
|
+
return c != 0
|
|
455
|
+
if op == ">=":
|
|
456
|
+
return c >= 0
|
|
457
|
+
if op == "<=":
|
|
458
|
+
return c <= 0
|
|
459
|
+
if op == ">":
|
|
460
|
+
return c > 0
|
|
461
|
+
if op == "<":
|
|
462
|
+
return c < 0
|
|
463
|
+
if op == "~=":
|
|
464
|
+
# compatible release: >= b, and first two segments equal
|
|
465
|
+
rel = b[0]
|
|
466
|
+
if len(rel) >= 2:
|
|
467
|
+
return c >= 0 and rel[:2] == ver[0][:2]
|
|
468
|
+
return c >= 0
|
|
469
|
+
return False
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
def pep440_satisfies(version, spec):
|
|
473
|
+
"""PEP 440 specifier satisfaction (==,!=,<=,>=,<,>,~=,===, comma AND, || OR)."""
|
|
474
|
+
ver = parse_pep440(version)
|
|
475
|
+
if ver is None:
|
|
476
|
+
return False
|
|
477
|
+
spec = (spec or "").strip()
|
|
478
|
+
if not spec:
|
|
479
|
+
return True
|
|
480
|
+
for alt in spec.split("||"):
|
|
481
|
+
ok = True
|
|
482
|
+
for part in alt.split(","):
|
|
483
|
+
part = part.strip()
|
|
484
|
+
if not part:
|
|
485
|
+
continue
|
|
486
|
+
m = re.match(r"^(===|==|!=|<=|>=|~=|>|<|\s*)(.+)$", part)
|
|
487
|
+
op = (m.group(1) or "").strip()
|
|
488
|
+
rest = m.group(2) if m else part
|
|
489
|
+
if not _pep_match_operator(ver, op, rest):
|
|
490
|
+
ok = False
|
|
491
|
+
break
|
|
492
|
+
if ok:
|
|
493
|
+
return True
|
|
494
|
+
return False
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
# --------------------------------------------------------------------------
|
|
498
|
+
# minimal TOML parser (subsets used by pyproject.toml / Pipfile / poetry.lock)
|
|
499
|
+
# --------------------------------------------------------------------------
|
|
500
|
+
|
|
501
|
+
def _find_eq_outside_string(line):
|
|
502
|
+
in_str = False
|
|
503
|
+
q = None
|
|
504
|
+
for i, ch in enumerate(line):
|
|
505
|
+
if in_str:
|
|
506
|
+
if ch == q:
|
|
507
|
+
in_str = False
|
|
508
|
+
else:
|
|
509
|
+
if ch in ('"', "'"):
|
|
510
|
+
in_str = True
|
|
511
|
+
q = ch
|
|
512
|
+
elif ch == "=":
|
|
513
|
+
return i
|
|
514
|
+
return None
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def _toml_unescape(s):
|
|
518
|
+
out = []
|
|
519
|
+
i = 0
|
|
520
|
+
while i < len(s):
|
|
521
|
+
ch = s[i]
|
|
522
|
+
if ch == "\\" and i + 1 < len(s):
|
|
523
|
+
nxt = s[i + 1]
|
|
524
|
+
mp = {"n": "\n", "t": "\t", "r": "\r", '"': '"', "\\": "\\", "/": "/", "b": "\b", "f": "\f"}
|
|
525
|
+
out.append(mp.get(nxt, nxt))
|
|
526
|
+
i += 2
|
|
527
|
+
else:
|
|
528
|
+
out.append(ch)
|
|
529
|
+
i += 1
|
|
530
|
+
return "".join(out)
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def _toml_string(s):
|
|
534
|
+
s = s.strip()
|
|
535
|
+
if s.startswith('"""') and s.endswith('"""') and len(s) >= 6:
|
|
536
|
+
return _toml_unescape(s[3:-3].strip("\n"))
|
|
537
|
+
if s.startswith("'''") and s.endswith("'''") and len(s) >= 6:
|
|
538
|
+
return s[3:-3].strip("\n")
|
|
539
|
+
if s.startswith('"') and s.endswith('"') and len(s) >= 2:
|
|
540
|
+
return _toml_unescape(s[1:-1])
|
|
541
|
+
if s.startswith("'") and s.endswith("'") and len(s) >= 2:
|
|
542
|
+
return s[1:-1]
|
|
543
|
+
return None
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _split_top_level(s, sep=","):
|
|
547
|
+
"""Split s on sep, respecting quotes and brackets."""
|
|
548
|
+
parts = []
|
|
549
|
+
depth = 0
|
|
550
|
+
cur = []
|
|
551
|
+
in_str = False
|
|
552
|
+
q = None
|
|
553
|
+
for ch in s:
|
|
554
|
+
if in_str:
|
|
555
|
+
cur.append(ch)
|
|
556
|
+
if ch == q:
|
|
557
|
+
in_str = False
|
|
558
|
+
continue
|
|
559
|
+
if ch in ('"', "'"):
|
|
560
|
+
in_str = True
|
|
561
|
+
q = ch
|
|
562
|
+
cur.append(ch)
|
|
563
|
+
elif ch in "[{":
|
|
564
|
+
depth += 1
|
|
565
|
+
cur.append(ch)
|
|
566
|
+
elif ch in "]}":
|
|
567
|
+
depth -= 1
|
|
568
|
+
cur.append(ch)
|
|
569
|
+
elif ch == sep and depth == 0:
|
|
570
|
+
parts.append("".join(cur).strip())
|
|
571
|
+
cur = []
|
|
572
|
+
else:
|
|
573
|
+
cur.append(ch)
|
|
574
|
+
if cur:
|
|
575
|
+
parts.append("".join(cur).strip())
|
|
576
|
+
return [p for p in parts if p]
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def _strip_toml_comment(s):
|
|
580
|
+
"""Strip a trailing inline TOML comment (# ...) that sits outside strings/brackets."""
|
|
581
|
+
depth = 0
|
|
582
|
+
in_str = False
|
|
583
|
+
q = None
|
|
584
|
+
i = 0
|
|
585
|
+
while i < len(s) - 1:
|
|
586
|
+
ch = s[i]
|
|
587
|
+
if in_str:
|
|
588
|
+
if ch == q:
|
|
589
|
+
in_str = False
|
|
590
|
+
i += 1
|
|
591
|
+
continue
|
|
592
|
+
if ch in ('"', "'"):
|
|
593
|
+
in_str = True
|
|
594
|
+
q = ch
|
|
595
|
+
elif ch in "[{":
|
|
596
|
+
depth += 1
|
|
597
|
+
elif ch in "]}":
|
|
598
|
+
depth -= 1
|
|
599
|
+
elif ch == "#" and depth == 0 and (i == 0 or s[i - 1] in " \t"):
|
|
600
|
+
return s[:i].rstrip()
|
|
601
|
+
i += 1
|
|
602
|
+
return s
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
def _toml_value(s):
|
|
606
|
+
s = _strip_toml_comment(s).strip()
|
|
607
|
+
if not s:
|
|
608
|
+
return None
|
|
609
|
+
if s == "true":
|
|
610
|
+
return True
|
|
611
|
+
if s == "false":
|
|
612
|
+
return False
|
|
613
|
+
if s.startswith("["):
|
|
614
|
+
return [_toml_value(x) for x in _split_top_level(s[1:-1].strip())]
|
|
615
|
+
if s.startswith("{"):
|
|
616
|
+
d = {}
|
|
617
|
+
for pair in _split_top_level(s[1:-1].strip()):
|
|
618
|
+
eq = _find_eq_outside_string(pair)
|
|
619
|
+
if eq is None:
|
|
620
|
+
continue
|
|
621
|
+
k = _toml_key(pair[:eq].strip())
|
|
622
|
+
d[".".join(k)] = _toml_value(pair[eq + 1:])
|
|
623
|
+
return d
|
|
624
|
+
st = _toml_string(s)
|
|
625
|
+
if st is not None:
|
|
626
|
+
return st
|
|
627
|
+
if re.fullmatch(r"[+-]?\d+(_\d+)*", s):
|
|
628
|
+
return int(s.replace("_", ""))
|
|
629
|
+
if re.fullmatch(r"[+-]?(\d+_)*\d+\.\d+", s):
|
|
630
|
+
return float(s.replace("_", ""))
|
|
631
|
+
return s
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def _toml_key(s):
|
|
636
|
+
parts = []
|
|
637
|
+
for p in _split_top_level(s, "."):
|
|
638
|
+
st = _toml_string(p)
|
|
639
|
+
parts.append(st if st is not None else p.strip())
|
|
640
|
+
return parts
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
def _toml_nav(root, parts):
|
|
644
|
+
"""Navigate to the dict for a path, descending through array-of-tables lists."""
|
|
645
|
+
cur = root
|
|
646
|
+
for p in parts:
|
|
647
|
+
nxt = cur.get(p)
|
|
648
|
+
if isinstance(nxt, list):
|
|
649
|
+
if not nxt:
|
|
650
|
+
return None
|
|
651
|
+
cur = nxt[-1]
|
|
652
|
+
continue
|
|
653
|
+
if not isinstance(nxt, dict):
|
|
654
|
+
nxt = {}
|
|
655
|
+
cur[p] = nxt
|
|
656
|
+
cur = nxt
|
|
657
|
+
return cur
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
def _toml_table(root, parts):
|
|
661
|
+
return _toml_nav(root, parts)
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
def _toml_set(root, parts, value):
|
|
665
|
+
cur = _toml_nav(root, parts[:-1])
|
|
666
|
+
if cur is not None:
|
|
667
|
+
cur[parts[-1]] = value
|
|
668
|
+
|
|
669
|
+
|
|
670
|
+
|
|
671
|
+
def _toml_aot(root, parts):
|
|
672
|
+
"""Return the next dict for an array-of-tables header, appending when needed."""
|
|
673
|
+
cur = root
|
|
674
|
+
for p in parts[:-1]:
|
|
675
|
+
nxt = cur.get(p)
|
|
676
|
+
if not isinstance(nxt, dict):
|
|
677
|
+
nxt = {}
|
|
678
|
+
cur[p] = nxt
|
|
679
|
+
cur = nxt
|
|
680
|
+
arr = cur.get(parts[-1])
|
|
681
|
+
if not isinstance(arr, list):
|
|
682
|
+
arr = []
|
|
683
|
+
cur[parts[-1]] = arr
|
|
684
|
+
d = {}
|
|
685
|
+
arr.append(d)
|
|
686
|
+
return d
|
|
687
|
+
|
|
688
|
+
|
|
689
|
+
def _bracket_depth(s):
|
|
690
|
+
depth = 0
|
|
691
|
+
in_str = False
|
|
692
|
+
q = None
|
|
693
|
+
for ch in s:
|
|
694
|
+
if in_str:
|
|
695
|
+
if ch == q:
|
|
696
|
+
in_str = False
|
|
697
|
+
continue
|
|
698
|
+
if ch in ('"', "'"):
|
|
699
|
+
in_str = True
|
|
700
|
+
q = ch
|
|
701
|
+
elif ch == "[":
|
|
702
|
+
depth += 1
|
|
703
|
+
elif ch == "]":
|
|
704
|
+
depth -= 1
|
|
705
|
+
return depth
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
def _toml_needs_more(text, lines, i, n):
|
|
709
|
+
"""Whether a value continues on the following lines."""
|
|
710
|
+
for q in ('"""', "'''"):
|
|
711
|
+
if text.startswith(q):
|
|
712
|
+
return q not in text[3:]
|
|
713
|
+
if text.startswith("["):
|
|
714
|
+
return _bracket_depth(text) > 0
|
|
715
|
+
if text.startswith("{"):
|
|
716
|
+
return _bracket_depth(text) > 0
|
|
717
|
+
return False
|
|
718
|
+
|
|
719
|
+
|
|
720
|
+
def parse_toml(text):
|
|
721
|
+
"""Parse a TOML document (subset) into nested dict/list structures.
|
|
722
|
+
|
|
723
|
+
Supports tables, arrays of tables, dotted keys, basic/literal/triple-quoted
|
|
724
|
+
strings, arrays (multi-line), inline tables and inline comments - the
|
|
725
|
+
subset used by pyproject.toml / Pipfile / poetry.lock.
|
|
726
|
+
"""
|
|
727
|
+
root = {}
|
|
728
|
+
cur = root
|
|
729
|
+
lines = (text or "").lstrip("\ufeff").splitlines()
|
|
730
|
+
i = 0
|
|
731
|
+
n = len(lines)
|
|
732
|
+
while i < n:
|
|
733
|
+
line = lines[i]
|
|
734
|
+
s = line.strip()
|
|
735
|
+
if not s or s.startswith("#"):
|
|
736
|
+
i += 1
|
|
737
|
+
continue
|
|
738
|
+
if s.startswith("[[") and s.endswith("]]"):
|
|
739
|
+
cur = _toml_aot(root, _toml_key(s[2:-2].strip()))
|
|
740
|
+
i += 1
|
|
741
|
+
continue
|
|
742
|
+
if s.startswith("[") and s.endswith("]"):
|
|
743
|
+
cur = _toml_table(root, _toml_key(s[1:-1].strip()))
|
|
744
|
+
i += 1
|
|
745
|
+
continue
|
|
746
|
+
eq = _find_eq_outside_string(line)
|
|
747
|
+
if eq is None:
|
|
748
|
+
i += 1
|
|
749
|
+
continue
|
|
750
|
+
key = _toml_key(line[:eq].strip())
|
|
751
|
+
val_text = line[eq + 1:].strip()
|
|
752
|
+
while _toml_needs_more(val_text, lines, i, n):
|
|
753
|
+
i += 1
|
|
754
|
+
if i >= n:
|
|
755
|
+
break
|
|
756
|
+
val_text += "\n" + lines[i]
|
|
757
|
+
val = _toml_value(val_text.strip())
|
|
758
|
+
_toml_set(cur, key, val)
|
|
759
|
+
i += 1
|
|
760
|
+
return root
|
|
761
|
+
|
|
762
|
+
|
|
763
|
+
# --------------------------------------------------------------------------
|
|
764
|
+
# npm ecosystem
|
|
765
|
+
# --------------------------------------------------------------------------
|
|
766
|
+
|
|
767
|
+
PUBLIC_NPM_HOSTS = ("registry.npmjs.org", "registry.yarnpkg.com")
|
|
768
|
+
|
|
769
|
+
|
|
770
|
+
def _scope_of(name):
|
|
771
|
+
if name.startswith("@"):
|
|
772
|
+
return name.split("/")[0].lstrip("@")
|
|
773
|
+
return None
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
def parse_npmrc(text):
|
|
777
|
+
"""Parse .npmrc into {"registry": url|None, "scopes": {scope: url}}."""
|
|
778
|
+
cfg = {"registry": None, "scopes": {}}
|
|
779
|
+
for raw in (text or "").splitlines():
|
|
780
|
+
line = raw.strip()
|
|
781
|
+
if not line or line.startswith(("#", ";")):
|
|
782
|
+
continue
|
|
783
|
+
if "=" not in line:
|
|
784
|
+
continue
|
|
785
|
+
k, _, v = line.partition("=")
|
|
786
|
+
k = k.strip()
|
|
787
|
+
v = re.sub(r"\s+[#;].*$", "", v).strip()
|
|
788
|
+
if k == "registry":
|
|
789
|
+
cfg["registry"] = v or None
|
|
790
|
+
elif k.startswith("@") and k.endswith(":registry"):
|
|
791
|
+
cfg["scopes"][k[1:-9]] = v or None
|
|
792
|
+
return cfg
|
|
793
|
+
|
|
794
|
+
|
|
795
|
+
def parse_package_json(text):
|
|
796
|
+
try:
|
|
797
|
+
d = json.loads(text or "{}")
|
|
798
|
+
return d if isinstance(d, dict) else {}
|
|
799
|
+
except Exception:
|
|
800
|
+
return {}
|
|
801
|
+
|
|
802
|
+
|
|
803
|
+
def _lock_pkg_name(key):
|
|
804
|
+
if not key:
|
|
805
|
+
return ""
|
|
806
|
+
idx = key.rfind("node_modules/")
|
|
807
|
+
if idx >= 0:
|
|
808
|
+
return key[idx + len("node_modules/"):]
|
|
809
|
+
return key
|
|
810
|
+
|
|
811
|
+
|
|
812
|
+
def _v1_name(key):
|
|
813
|
+
if key.startswith("@"):
|
|
814
|
+
slash = key.find("/")
|
|
815
|
+
at = key.find("@", slash)
|
|
816
|
+
return key if at == -1 else key[:at]
|
|
817
|
+
return key.split("@")[0]
|
|
818
|
+
|
|
819
|
+
|
|
820
|
+
def parse_package_lock(text):
|
|
821
|
+
"""Parse package-lock.json / npm-shrinkwrap.json.
|
|
822
|
+
|
|
823
|
+
Returns {"lockfileVersion", "root", "packages": {name: [entry,...]}} or None.
|
|
824
|
+
"""
|
|
825
|
+
try:
|
|
826
|
+
data = json.loads(text or "{}")
|
|
827
|
+
except Exception:
|
|
828
|
+
return None
|
|
829
|
+
if not isinstance(data, dict):
|
|
830
|
+
return None
|
|
831
|
+
lv = data.get("lockfileVersion")
|
|
832
|
+
root = {
|
|
833
|
+
"name": data.get("name"),
|
|
834
|
+
"version": data.get("version"),
|
|
835
|
+
"deps": data.get("dependencies") or {},
|
|
836
|
+
}
|
|
837
|
+
packages = {}
|
|
838
|
+
if isinstance(lv, int) and lv >= 2:
|
|
839
|
+
for key, ent in (data.get("packages") or {}).items():
|
|
840
|
+
if not isinstance(ent, dict):
|
|
841
|
+
continue
|
|
842
|
+
name = _lock_pkg_name(key)
|
|
843
|
+
if not name:
|
|
844
|
+
continue
|
|
845
|
+
deps = {}
|
|
846
|
+
for dk in ("dependencies", "optionalDependencies", "peerDependencies"):
|
|
847
|
+
for dn, dv in (ent.get(dk) or {}).items():
|
|
848
|
+
deps.setdefault(dn, dv)
|
|
849
|
+
packages.setdefault(name, []).append({
|
|
850
|
+
"version": ent.get("version"),
|
|
851
|
+
"resolved": ent.get("resolved"),
|
|
852
|
+
"integrity": ent.get("integrity"),
|
|
853
|
+
"dev": bool(ent.get("dev") or ent.get("devOptional")),
|
|
854
|
+
"optional": bool(ent.get("optional")),
|
|
855
|
+
"deps": deps,
|
|
856
|
+
"key": key,
|
|
857
|
+
})
|
|
858
|
+
else:
|
|
859
|
+
for key, ent in (data.get("dependencies") or {}).items():
|
|
860
|
+
if not isinstance(ent, dict):
|
|
861
|
+
continue
|
|
862
|
+
name = _v1_name(key)
|
|
863
|
+
packages.setdefault(name, []).append({
|
|
864
|
+
"version": ent.get("version"),
|
|
865
|
+
"resolved": ent.get("resolved"),
|
|
866
|
+
"integrity": ent.get("integrity"),
|
|
867
|
+
"dev": bool(ent.get("dev")),
|
|
868
|
+
"optional": bool(ent.get("optional")),
|
|
869
|
+
"deps": dict(ent.get("requires") or {}),
|
|
870
|
+
"key": key,
|
|
871
|
+
})
|
|
872
|
+
return {"lockfileVersion": lv, "root": root, "packages": packages}
|
|
873
|
+
|
|
874
|
+
|
|
875
|
+
def check_npm(project_dir, findings, sbom_pkgs):
|
|
876
|
+
"""Check the npm ecosystem in project_dir. Returns True if npm detected."""
|
|
877
|
+
base = Path(project_dir)
|
|
878
|
+
pj_path = base / "package.json"
|
|
879
|
+
if not pj_path.exists():
|
|
880
|
+
return False
|
|
881
|
+
pj = parse_package_json(_read_text(pj_path))
|
|
882
|
+
if not pj:
|
|
883
|
+
return False
|
|
884
|
+
|
|
885
|
+
manifest_deps = {}
|
|
886
|
+
for group in ("dependencies", "devDependencies", "optionalDependencies", "peerDependencies"):
|
|
887
|
+
d = pj.get(group) or {}
|
|
888
|
+
if isinstance(d, dict):
|
|
889
|
+
for n, r in d.items():
|
|
890
|
+
manifest_deps.setdefault(n, {"range": r, "group": group})
|
|
891
|
+
|
|
892
|
+
npmrc = parse_npmrc(_read_text(base / ".npmrc"))
|
|
893
|
+
pub_cfg = pj.get("publishConfig") or {}
|
|
894
|
+
if isinstance(pub_cfg, dict) and pub_cfg.get("registry"):
|
|
895
|
+
npmrc["registry"] = npmrc["registry"] or pub_cfg["registry"]
|
|
896
|
+
|
|
897
|
+
lock_path = None
|
|
898
|
+
for cand in ("package-lock.json", "npm-shrinkwrap.json"):
|
|
899
|
+
if (base / cand).exists():
|
|
900
|
+
lock_path = base / cand
|
|
901
|
+
break
|
|
902
|
+
|
|
903
|
+
if lock_path is None:
|
|
904
|
+
if manifest_deps:
|
|
905
|
+
findings.append(Finding(
|
|
906
|
+
"missing_lockfile", "medium", "package.json", None,
|
|
907
|
+
"缺少锁文件(package-lock.json / npm-shrinkwrap.json)",
|
|
908
|
+
"依赖版本未锁定,安装结果不可复现;建议提交 package-lock.json 并使用 npm ci", ecosystem="npm"))
|
|
909
|
+
else:
|
|
910
|
+
lock = parse_package_lock(_read_text(lock_path))
|
|
911
|
+
if lock is None:
|
|
912
|
+
findings.append(Finding(
|
|
913
|
+
"lockfile_parse_error", "medium", lock_path.name, None,
|
|
914
|
+
"锁文件解析失败(JSON 不合法或结构异常)",
|
|
915
|
+
"请用 npm install 重新生成锁文件", ecosystem="npm"))
|
|
916
|
+
else:
|
|
917
|
+
_check_npm_lock(base, lock_path, lock, pj, manifest_deps, npmrc, findings, sbom_pkgs)
|
|
918
|
+
|
|
919
|
+
for name, info in manifest_deps.items():
|
|
920
|
+
rng = (info["range"] or "").strip()
|
|
921
|
+
if rng in ("", "*", "latest", "x"):
|
|
922
|
+
findings.append(Finding(
|
|
923
|
+
"unpinned", "low", "package.json", name,
|
|
924
|
+
"依赖 %s 未固定版本(%s),每次安装可能拉到不同版本" % (name, rng or "未指定"),
|
|
925
|
+
"建议给出 ^/~/精确版本并提交锁文件", ecosystem="npm"))
|
|
926
|
+
if "/" not in name:
|
|
927
|
+
ts = find_typosquat(name, POPULAR_NPM)
|
|
928
|
+
if ts:
|
|
929
|
+
findings.append(Finding(
|
|
930
|
+
"typosquat", "low", "package.json", name,
|
|
931
|
+
"依赖名 %s 与知名包 %s 高度相似(编辑距离 %d),请核对是否为拼写仿冒" % (name, ts[0], ts[1]),
|
|
932
|
+
"typosquat 是供应链投毒常见手法,发布前请人工确认包名与来源", ecosystem="npm"))
|
|
933
|
+
return True
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
def _check_npm_lock(base, lock_path, lock, pj, manifest_deps, npmrc, findings, sbom_pkgs):
|
|
937
|
+
lv = lock["lockfileVersion"]
|
|
938
|
+
lock_name = lock["root"].get("name")
|
|
939
|
+
lock_version = lock["root"].get("version")
|
|
940
|
+
pj_name = pj.get("name")
|
|
941
|
+
if pj_name and lock_name and lock_name != pj_name:
|
|
942
|
+
findings.append(Finding(
|
|
943
|
+
"lockfile_root_mismatch", "medium", lock_path.name, lock_name,
|
|
944
|
+
"锁文件根包名 %s 与 package.json name %s 不一致" % (lock_name, pj_name),
|
|
945
|
+
"锁文件与清单不对应,可能是复制或生成错误", ecosystem="npm"))
|
|
946
|
+
if pj.get("version") and lock_version and str(lock_version) != str(pj.get("version")):
|
|
947
|
+
findings.append(Finding(
|
|
948
|
+
"lockfile_root_mismatch", "medium", lock_path.name, lock_name or "",
|
|
949
|
+
"锁文件根版本 %s 与 package.json version %s 不一致" % (lock_version, pj.get("version")),
|
|
950
|
+
"锁文件与清单不对应,请重新生成", ecosystem="npm"))
|
|
951
|
+
|
|
952
|
+
packages = lock["packages"]
|
|
953
|
+
|
|
954
|
+
for name, info in manifest_deps.items():
|
|
955
|
+
entries = packages.get(name)
|
|
956
|
+
if not entries:
|
|
957
|
+
findings.append(Finding(
|
|
958
|
+
"lockfile_missing_entry", "high", lock_path.name, name,
|
|
959
|
+
"package.json 声明了 %s@%s,但锁文件中没有该包" % (name, info["range"]),
|
|
960
|
+
"锁文件过期或手工改动,运行 npm install 重新生成", ecosystem="npm"))
|
|
961
|
+
continue
|
|
962
|
+
matched = any(e.get("version") and semver_satisfies(e["version"], info["range"])
|
|
963
|
+
for e in entries)
|
|
964
|
+
if not matched:
|
|
965
|
+
versions = ", ".join(sorted({str(e.get("version")) for e in entries if e.get("version")}))
|
|
966
|
+
findings.append(Finding(
|
|
967
|
+
"lockfile_range_unsatisfied", "high", lock_path.name, name,
|
|
968
|
+
"锁文件中 %s 的版本(%s)不满足 package.json 声明范围 %s" % (name, versions or "无版本", info["range"]),
|
|
969
|
+
"声明与锁定不一致,安装可能拉取意外版本", ecosystem="npm"))
|
|
970
|
+
|
|
971
|
+
all_names = set(packages.keys())
|
|
972
|
+
for name, entries in packages.items():
|
|
973
|
+
for e in entries:
|
|
974
|
+
for dn in e["deps"]:
|
|
975
|
+
if dn not in all_names:
|
|
976
|
+
findings.append(Finding(
|
|
977
|
+
"lockfile_dangling_ref", "high", lock_path.name, name,
|
|
978
|
+
"锁文件里 %s 依赖的 %s 不存在于锁文件包列表" % (name, dn),
|
|
979
|
+
"依赖图断裂,安装可能失败或行为异常", ecosystem="npm"))
|
|
980
|
+
|
|
981
|
+
if lv is not None and lv >= 2:
|
|
982
|
+
for name, entries in packages.items():
|
|
983
|
+
for e in entries:
|
|
984
|
+
if e.get("version") and not e.get("integrity"):
|
|
985
|
+
res = e.get("resolved") or ""
|
|
986
|
+
if "github.com" in res or res.startswith("file:") or "git+" in res:
|
|
987
|
+
continue
|
|
988
|
+
findings.append(Finding(
|
|
989
|
+
"lockfile_integrity_missing", "medium", lock_path.name, name,
|
|
990
|
+
"锁文件中 %s@%s 缺少 integrity 校验值" % (name, e.get("version")),
|
|
991
|
+
"缺少完整性校验,安装无法防篡改", ecosystem="npm"))
|
|
992
|
+
|
|
993
|
+
for name, entries in packages.items():
|
|
994
|
+
sig = {}
|
|
995
|
+
for e in entries:
|
|
996
|
+
if not e.get("version"):
|
|
997
|
+
continue
|
|
998
|
+
key = (name, e["version"])
|
|
999
|
+
s = (e.get("resolved"), e.get("integrity"))
|
|
1000
|
+
if key in sig and sig[key] != s:
|
|
1001
|
+
findings.append(Finding(
|
|
1002
|
+
"lockfile_duplicate_conflict", "high", lock_path.name, name,
|
|
1003
|
+
"锁文件中 %s@%s 存在多个不同来源(resolved/integrity 不一致)" % (name, e["version"]),
|
|
1004
|
+
"同一版本解析到不同 tarball,存在被替换风险", ecosystem="npm"))
|
|
1005
|
+
break
|
|
1006
|
+
sig[key] = s
|
|
1007
|
+
|
|
1008
|
+
# registry / dependency-confusion signals
|
|
1009
|
+
cfg_default_host = None
|
|
1010
|
+
if npmrc.get("registry"):
|
|
1011
|
+
cfg_default_host = _url_host(npmrc["registry"])
|
|
1012
|
+
host_by_name = {}
|
|
1013
|
+
for name, entries in packages.items():
|
|
1014
|
+
for e in entries:
|
|
1015
|
+
h = _url_host(e.get("resolved"))
|
|
1016
|
+
if h:
|
|
1017
|
+
host_by_name.setdefault(name, set()).add(h[1])
|
|
1018
|
+
|
|
1019
|
+
for name, hosts in host_by_name.items():
|
|
1020
|
+
if len(hosts) > 1:
|
|
1021
|
+
findings.append(Finding(
|
|
1022
|
+
"confusion_mixed_registry", "high", lock_path.name, name,
|
|
1023
|
+
"包 %s 被解析自多个不同仓库主机:%s" % (name, ", ".join(sorted(hosts))),
|
|
1024
|
+
"同一依赖来源不一致,可能是依赖混淆或镜像污染", ecosystem="npm"))
|
|
1025
|
+
scope = _scope_of(name)
|
|
1026
|
+
scope_cfg = npmrc["scopes"].get(scope) if scope else None
|
|
1027
|
+
for h in sorted(hosts):
|
|
1028
|
+
if scope_cfg:
|
|
1029
|
+
sc_host = _url_host(scope_cfg)
|
|
1030
|
+
if sc_host and h != sc_host[1]:
|
|
1031
|
+
findings.append(Finding(
|
|
1032
|
+
"confusion_scope_registry", "high", lock_path.name, name,
|
|
1033
|
+
"作用域 %s 在 .npmrc 配置了私有仓库 %s,但 %s 实际解析自公共仓库 %s" % (scope, scope_cfg, name, h),
|
|
1034
|
+
"私有包名可能被公共仓库同名抢占(依赖混淆)", ecosystem="npm"))
|
|
1035
|
+
if cfg_default_host and cfg_default_host[1] != h and h in PUBLIC_NPM_HOSTS and not scope_cfg:
|
|
1036
|
+
findings.append(Finding(
|
|
1037
|
+
"confusion_registry_mismatch", "medium", lock_path.name, name,
|
|
1038
|
+
"项目配置了默认仓库 %s,但 %s 实际解析自公共仓库 %s" % (npmrc["registry"], name, h),
|
|
1039
|
+
"配置与锁定来源不一致,私有包名可能被公共仓库抢占(依赖混淆)", ecosystem="npm"))
|
|
1040
|
+
|
|
1041
|
+
for name, entries in packages.items():
|
|
1042
|
+
seen = set()
|
|
1043
|
+
for e in entries:
|
|
1044
|
+
res = e.get("resolved")
|
|
1045
|
+
if not res or res in seen:
|
|
1046
|
+
continue
|
|
1047
|
+
seen.add(res)
|
|
1048
|
+
suspicious, reason = _is_suspicious_url(res)
|
|
1049
|
+
if suspicious:
|
|
1050
|
+
findings.append(Finding(
|
|
1051
|
+
"confusion_suspicious_registry", "medium", lock_path.name, name,
|
|
1052
|
+
"包 %s 的解析地址可疑:%s(%s)" % (name, res, reason),
|
|
1053
|
+
"请核对仓库地址来源", ecosystem="npm"))
|
|
1054
|
+
|
|
1055
|
+
# collect SBOM packages
|
|
1056
|
+
for name, entries in packages.items():
|
|
1057
|
+
for e in entries:
|
|
1058
|
+
scope = "optional" if (e.get("dev") or e.get("optional")) else "required"
|
|
1059
|
+
sbom_pkgs.append({
|
|
1060
|
+
"ecosystem": "npm",
|
|
1061
|
+
"name": name,
|
|
1062
|
+
"version": e.get("version") or "",
|
|
1063
|
+
"resolved": e.get("resolved") or "",
|
|
1064
|
+
"integrity": e.get("integrity") or "",
|
|
1065
|
+
"scope": scope,
|
|
1066
|
+
"direct": name in manifest_deps,
|
|
1067
|
+
"deps": sorted(e["deps"].keys()),
|
|
1068
|
+
})
|
|
1069
|
+
return True
|
|
1070
|
+
|
|
1071
|
+
|
|
1072
|
+
# --------------------------------------------------------------------------
|
|
1073
|
+
# python ecosystem
|
|
1074
|
+
# --------------------------------------------------------------------------
|
|
1075
|
+
|
|
1076
|
+
_RE_PEP508 = re.compile(r"^([A-Za-z0-9][A-Za-z0-9._-]*)(?:\[[^]]*\])?\s*([^;]*)$")
|
|
1077
|
+
|
|
1078
|
+
|
|
1079
|
+
def _pep508_split(d):
|
|
1080
|
+
if not isinstance(d, str):
|
|
1081
|
+
return (str(d), "")
|
|
1082
|
+
m = _RE_PEP508.match(d.strip())
|
|
1083
|
+
if not m:
|
|
1084
|
+
return (d.strip(), "")
|
|
1085
|
+
return (m.group(1), m.group(2).strip())
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
def parse_requirements(text, base_dir=None, depth=0, out=None):
|
|
1089
|
+
"""Parse requirements.txt content.
|
|
1090
|
+
|
|
1091
|
+
Returns {"packages": {name: {"spec", "has_hash"}}, "index": url|None,
|
|
1092
|
+
"extra": [url,...]}. Follows -r includes (bounded depth).
|
|
1093
|
+
"""
|
|
1094
|
+
out = out if out is not None else {"packages": {}, "index": None, "extra": []}
|
|
1095
|
+
if depth > 5:
|
|
1096
|
+
return out
|
|
1097
|
+
for raw in (text or "").splitlines():
|
|
1098
|
+
line = raw.strip()
|
|
1099
|
+
if not line or line.startswith("#"):
|
|
1100
|
+
continue
|
|
1101
|
+
low = line.lower()
|
|
1102
|
+
if low.startswith("--index-url") or low.startswith("-i "):
|
|
1103
|
+
parts = line.split(None, 1)
|
|
1104
|
+
if len(parts) > 1:
|
|
1105
|
+
out["index"] = parts[1].strip().split()[0]
|
|
1106
|
+
continue
|
|
1107
|
+
if low.startswith("--extra-index-url"):
|
|
1108
|
+
parts = line.split(None, 1)
|
|
1109
|
+
if len(parts) > 1:
|
|
1110
|
+
out["extra"].append(parts[1].strip().split()[0])
|
|
1111
|
+
continue
|
|
1112
|
+
if low.startswith("-r ") or low.startswith("--requirement "):
|
|
1113
|
+
parts = line.split(None, 1)
|
|
1114
|
+
if len(parts) > 1 and base_dir:
|
|
1115
|
+
sub = parts[1].strip().split()[0]
|
|
1116
|
+
parse_requirements(_read_text(Path(base_dir) / sub), Path(base_dir), depth + 1, out)
|
|
1117
|
+
continue
|
|
1118
|
+
if (low.startswith("-e ") or line.startswith("git+") or line.startswith("hg+")
|
|
1119
|
+
or line.startswith("svn+") or line.startswith("http://") or line.startswith("https://")):
|
|
1120
|
+
continue
|
|
1121
|
+
name_spec = line.partition(";")[0].strip()
|
|
1122
|
+
if not name_spec:
|
|
1123
|
+
continue
|
|
1124
|
+
m = _RE_PEP508.match(name_spec)
|
|
1125
|
+
if not m:
|
|
1126
|
+
continue
|
|
1127
|
+
name = m.group(1)
|
|
1128
|
+
spec = m.group(2).strip()
|
|
1129
|
+
has_hash = "--hash" in line
|
|
1130
|
+
spec = re.sub(r"\s+--hash=[^\s]+", "", spec).strip()
|
|
1131
|
+
out["packages"].setdefault(name, {"spec": spec, "has_hash": has_hash})
|
|
1132
|
+
return out
|
|
1133
|
+
|
|
1134
|
+
|
|
1135
|
+
def pyproject_deps(py):
|
|
1136
|
+
"""Return (project_deps, poetry_deps) lists of (name, spec)."""
|
|
1137
|
+
proj_deps = []
|
|
1138
|
+
poetry_deps = []
|
|
1139
|
+
proj = py.get("project") or {}
|
|
1140
|
+
for d in (proj.get("dependencies") or []):
|
|
1141
|
+
proj_deps.append(_pep508_split(d))
|
|
1142
|
+
opt = proj.get("optional-dependencies") or {}
|
|
1143
|
+
if isinstance(opt, dict):
|
|
1144
|
+
for lst in opt.values():
|
|
1145
|
+
if isinstance(lst, list):
|
|
1146
|
+
for d in lst:
|
|
1147
|
+
proj_deps.append(_pep508_split(d))
|
|
1148
|
+
tool = py.get("tool") or {}
|
|
1149
|
+
poetry = tool.get("poetry") or {}
|
|
1150
|
+
pdep = poetry.get("dependencies") or {}
|
|
1151
|
+
if isinstance(pdep, dict):
|
|
1152
|
+
for name, spec in pdep.items():
|
|
1153
|
+
if name == "python":
|
|
1154
|
+
continue
|
|
1155
|
+
if isinstance(spec, dict):
|
|
1156
|
+
spec = spec.get("version", "*")
|
|
1157
|
+
poetry_deps.append((name, str(spec or "*")))
|
|
1158
|
+
g = poetry.get("dev-dependencies") or {}
|
|
1159
|
+
if isinstance(g, dict):
|
|
1160
|
+
for name, spec in g.items():
|
|
1161
|
+
poetry_deps.append((name, str(spec) if not isinstance(spec, dict) else "*"))
|
|
1162
|
+
for gname, gdata in (poetry.get("group") or {}).items():
|
|
1163
|
+
if isinstance(gdata, dict):
|
|
1164
|
+
for name, spec in (gdata.get("dependencies") or {}).items():
|
|
1165
|
+
poetry_deps.append((name, str(spec) if not isinstance(spec, dict) else "*"))
|
|
1166
|
+
return proj_deps, poetry_deps
|
|
1167
|
+
|
|
1168
|
+
|
|
1169
|
+
def poetry_sources(py):
|
|
1170
|
+
out = []
|
|
1171
|
+
tool = py.get("tool") or {}
|
|
1172
|
+
poetry = tool.get("poetry") or {}
|
|
1173
|
+
srcs = poetry.get("source") or []
|
|
1174
|
+
if isinstance(srcs, dict):
|
|
1175
|
+
srcs = [srcs]
|
|
1176
|
+
for s in srcs:
|
|
1177
|
+
if isinstance(s, dict):
|
|
1178
|
+
out.append({
|
|
1179
|
+
"name": s.get("name"),
|
|
1180
|
+
"url": s.get("url"),
|
|
1181
|
+
"default": bool(s.get("default")),
|
|
1182
|
+
"secondary": bool(s.get("secondary")),
|
|
1183
|
+
})
|
|
1184
|
+
return out
|
|
1185
|
+
|
|
1186
|
+
|
|
1187
|
+
def _pip_index_urls(py):
|
|
1188
|
+
"""Collect index URLs from [tool.pip.index] / [[tool.pip.index]] / [tool.uv]."""
|
|
1189
|
+
urls = []
|
|
1190
|
+
tool = py.get("tool") or {}
|
|
1191
|
+
for tname in ("pip", "uv"):
|
|
1192
|
+
sec = tool.get(tname) or {}
|
|
1193
|
+
idx = sec.get("index")
|
|
1194
|
+
items = idx if isinstance(idx, list) else ([idx] if idx else [])
|
|
1195
|
+
for it in items:
|
|
1196
|
+
if isinstance(it, dict) and it.get("url"):
|
|
1197
|
+
urls.append(it["url"])
|
|
1198
|
+
if isinstance(sec, dict) and sec.get("index-url"):
|
|
1199
|
+
urls.append(sec["index-url"])
|
|
1200
|
+
return urls
|
|
1201
|
+
|
|
1202
|
+
|
|
1203
|
+
def check_python(project_dir, findings, sbom_pkgs):
|
|
1204
|
+
"""Check the python ecosystem. Returns True if python detected."""
|
|
1205
|
+
base = Path(project_dir)
|
|
1206
|
+
touched = False
|
|
1207
|
+
|
|
1208
|
+
for rf in sorted(base.glob("requirements*.txt")):
|
|
1209
|
+
touched = True
|
|
1210
|
+
parsed = parse_requirements(_read_text(rf), base_dir=rf.parent)
|
|
1211
|
+
for name, info in parsed["packages"].items():
|
|
1212
|
+
spec = info["spec"]
|
|
1213
|
+
if not spec or spec in ("*", ""):
|
|
1214
|
+
findings.append(Finding(
|
|
1215
|
+
"unpinned", "low", rf.name, name,
|
|
1216
|
+
"依赖 %s 未固定版本(%s),每次安装可能拉到不同版本" % (name, spec or "未指定"),
|
|
1217
|
+
"建议给出 ==/>= 等约束并配合锁文件(pip-tools / poetry / pipenv)", ecosystem="python"))
|
|
1218
|
+
ts = find_typosquat(name, POPULAR_PYPI)
|
|
1219
|
+
if ts:
|
|
1220
|
+
findings.append(Finding(
|
|
1221
|
+
"typosquat", "low", rf.name, name,
|
|
1222
|
+
"依赖名 %s 与知名包 %s 高度相似(编辑距离 %d),请核对是否为拼写仿冒" % (name, ts[0], ts[1]),
|
|
1223
|
+
"typosquat 是供应链投毒常见手法,发布前请人工确认包名与来源", ecosystem="python"))
|
|
1224
|
+
idx = parsed["index"]
|
|
1225
|
+
extras = parsed["extra"]
|
|
1226
|
+
if idx and extras:
|
|
1227
|
+
findings.append(Finding(
|
|
1228
|
+
"confusion_extra_index", "medium", rf.name, None,
|
|
1229
|
+
"同时配置了主仓库 %s 与 extra-index %s,公共仓库成为回退源,私有包名可能被公共仓库同名抢占(依赖混淆)" % (idx, ", ".join(extras)),
|
|
1230
|
+
"依赖混淆高危配置:建议仅使用单一可信仓库并锁定私有包名", ecosystem="python"))
|
|
1231
|
+
for url in ([idx] if idx else []) + extras:
|
|
1232
|
+
suspicious, reason = _is_suspicious_url(url)
|
|
1233
|
+
if suspicious:
|
|
1234
|
+
findings.append(Finding(
|
|
1235
|
+
"confusion_suspicious_registry", "medium", rf.name, None,
|
|
1236
|
+
"索引地址可疑:%s(%s)" % (url, reason),
|
|
1237
|
+
"请核对仓库地址来源", ecosystem="python"))
|
|
1238
|
+
|
|
1239
|
+
pp_path = base / "pyproject.toml"
|
|
1240
|
+
if pp_path.exists():
|
|
1241
|
+
touched = True
|
|
1242
|
+
py = parse_toml(_read_text(pp_path))
|
|
1243
|
+
proj_deps, poetry_deps = pyproject_deps(py)
|
|
1244
|
+
declared = proj_deps + poetry_deps
|
|
1245
|
+
lock_path = base / "poetry.lock"
|
|
1246
|
+
lock = None
|
|
1247
|
+
if lock_path.exists():
|
|
1248
|
+
lock = parse_toml(_read_text(lock_path))
|
|
1249
|
+
if declared and lock is None:
|
|
1250
|
+
findings.append(Finding(
|
|
1251
|
+
"missing_lockfile", "medium", "pyproject.toml", None,
|
|
1252
|
+
"pyproject.toml 声明了 %d 个依赖但没有 poetry.lock" % len(declared),
|
|
1253
|
+
"依赖版本未锁定,安装结果不可复现;建议 poetry lock 并提交 poetry.lock", ecosystem="python"))
|
|
1254
|
+
elif lock is not None:
|
|
1255
|
+
_check_poetry_lock(lock, declared, findings, sbom_pkgs)
|
|
1256
|
+
for src in poetry_sources(py):
|
|
1257
|
+
url = src.get("url")
|
|
1258
|
+
if not url:
|
|
1259
|
+
continue
|
|
1260
|
+
suspicious, reason = _is_suspicious_url(url)
|
|
1261
|
+
if suspicious:
|
|
1262
|
+
findings.append(Finding(
|
|
1263
|
+
"confusion_suspicious_registry", "medium", "pyproject.toml", None,
|
|
1264
|
+
"poetry 源 %s 地址可疑:%s(%s)" % (src.get("name") or "?", url, reason),
|
|
1265
|
+
"请核对仓库地址来源", ecosystem="python"))
|
|
1266
|
+
if src.get("secondary") and "pypi.org" not in url:
|
|
1267
|
+
findings.append(Finding(
|
|
1268
|
+
"confusion_extra_index", "medium", "pyproject.toml", None,
|
|
1269
|
+
"poetry 源 %s(%s)标记为 secondary,公共 PyPI 仍是回退源,私有包名存在依赖混淆风险" % (src.get("name") or "?", url),
|
|
1270
|
+
"依赖混淆高危配置:建议把私有源设为 default 并禁用公共回退", ecosystem="python"))
|
|
1271
|
+
for url in _pip_index_urls(py):
|
|
1272
|
+
suspicious, reason = _is_suspicious_url(url)
|
|
1273
|
+
if suspicious:
|
|
1274
|
+
findings.append(Finding(
|
|
1275
|
+
"confusion_suspicious_registry", "medium", "pyproject.toml", None,
|
|
1276
|
+
"索引地址可疑:%s(%s)" % (url, reason),
|
|
1277
|
+
"请核对仓库地址来源", ecosystem="python"))
|
|
1278
|
+
|
|
1279
|
+
if check_pipfile(project_dir, findings, sbom_pkgs):
|
|
1280
|
+
touched = True
|
|
1281
|
+
return touched
|
|
1282
|
+
|
|
1283
|
+
|
|
1284
|
+
def _check_poetry_lock(lock, declared, findings, sbom_pkgs):
|
|
1285
|
+
pkgs = {}
|
|
1286
|
+
for p in (lock.get("package") or []):
|
|
1287
|
+
if isinstance(p, dict) and p.get("name"):
|
|
1288
|
+
pkgs[p["name"]] = p
|
|
1289
|
+
for name, spec in declared:
|
|
1290
|
+
p = pkgs.get(name)
|
|
1291
|
+
if p is None:
|
|
1292
|
+
findings.append(Finding(
|
|
1293
|
+
"lockfile_missing_entry", "high", "poetry.lock", name,
|
|
1294
|
+
"pyproject.toml 声明了 %s(%s),但 poetry.lock 中没有该包" % (name, spec),
|
|
1295
|
+
"锁文件过期或手工改动,运行 poetry lock 重新生成", ecosystem="python"))
|
|
1296
|
+
continue
|
|
1297
|
+
ver = p.get("version")
|
|
1298
|
+
if ver and not pep440_satisfies(ver, spec):
|
|
1299
|
+
findings.append(Finding(
|
|
1300
|
+
"lockfile_range_unsatisfied", "high", "poetry.lock", name,
|
|
1301
|
+
"poetry.lock 中 %s=%s 不满足 pyproject.toml 声明 %s" % (name, ver, spec),
|
|
1302
|
+
"声明与锁定不一致,安装可能拉取意外版本", ecosystem="python"))
|
|
1303
|
+
for name, p in pkgs.items():
|
|
1304
|
+
for dn in (p.get("dependencies") or {}):
|
|
1305
|
+
if dn not in pkgs:
|
|
1306
|
+
findings.append(Finding(
|
|
1307
|
+
"lockfile_dangling_ref", "high", "poetry.lock", name,
|
|
1308
|
+
"poetry.lock 里 %s 依赖的 %s 不存在于锁文件包列表" % (name, dn),
|
|
1309
|
+
"依赖图断裂,安装可能失败或行为异常", ecosystem="python"))
|
|
1310
|
+
files = p.get("files") or []
|
|
1311
|
+
if not files:
|
|
1312
|
+
findings.append(Finding(
|
|
1313
|
+
"lockfile_integrity_missing", "medium", "poetry.lock", name,
|
|
1314
|
+
"poetry.lock 中 %s 缺少文件哈希(files 为空)" % name,
|
|
1315
|
+
"缺少完整性校验,安装无法防篡改", ecosystem="python"))
|
|
1316
|
+
sbom_pkgs.append({
|
|
1317
|
+
"ecosystem": "python",
|
|
1318
|
+
"name": name,
|
|
1319
|
+
"version": str(p.get("version") or ""),
|
|
1320
|
+
"resolved": "",
|
|
1321
|
+
"integrity": (files[0].get("hash") if files and isinstance(files[0], dict) else "") or "",
|
|
1322
|
+
"scope": "optional" if p.get("optional") else "required",
|
|
1323
|
+
"direct": name in dict(declared),
|
|
1324
|
+
"deps": sorted((p.get("dependencies") or {}).keys()),
|
|
1325
|
+
})
|
|
1326
|
+
|
|
1327
|
+
|
|
1328
|
+
def check_pipfile(project_dir, findings, sbom_pkgs):
|
|
1329
|
+
base = Path(project_dir)
|
|
1330
|
+
pf = base / "Pipfile"
|
|
1331
|
+
if not pf.exists():
|
|
1332
|
+
return False
|
|
1333
|
+
doc = parse_toml(_read_text(pf))
|
|
1334
|
+
packages = {}
|
|
1335
|
+
for grp in ("packages", "dev-packages"):
|
|
1336
|
+
d = doc.get(grp) or {}
|
|
1337
|
+
if isinstance(d, dict):
|
|
1338
|
+
for name, spec in d.items():
|
|
1339
|
+
if isinstance(spec, dict):
|
|
1340
|
+
spec = spec.get("version", "*")
|
|
1341
|
+
packages.setdefault(name, {"spec": str(spec or "*"), "group": grp})
|
|
1342
|
+
sources = doc.get("source") or []
|
|
1343
|
+
if isinstance(sources, dict):
|
|
1344
|
+
sources = [sources]
|
|
1345
|
+
lock_path = base / "Pipfile.lock"
|
|
1346
|
+
lock = None
|
|
1347
|
+
if lock_path.exists():
|
|
1348
|
+
try:
|
|
1349
|
+
lock = json.loads(_read_text(lock_path))
|
|
1350
|
+
except Exception:
|
|
1351
|
+
lock = None
|
|
1352
|
+
if packages and lock is None:
|
|
1353
|
+
findings.append(Finding(
|
|
1354
|
+
"missing_lockfile", "medium", "Pipfile", None,
|
|
1355
|
+
"Pipfile 声明了 %d 个依赖但没有 Pipfile.lock" % len(packages),
|
|
1356
|
+
"依赖版本未锁定,安装结果不可复现;建议 pipenv lock 并提交 Pipfile.lock", ecosystem="python"))
|
|
1357
|
+
elif lock is not None:
|
|
1358
|
+
entries = {}
|
|
1359
|
+
for grp in ("default", "develop"):
|
|
1360
|
+
d = lock.get(grp) or {}
|
|
1361
|
+
if isinstance(d, dict):
|
|
1362
|
+
for name, ent in d.items():
|
|
1363
|
+
if isinstance(ent, dict):
|
|
1364
|
+
entries.setdefault(name, ent)
|
|
1365
|
+
for name, info in packages.items():
|
|
1366
|
+
ent = entries.get(name)
|
|
1367
|
+
if ent is None:
|
|
1368
|
+
findings.append(Finding(
|
|
1369
|
+
"lockfile_missing_entry", "high", "Pipfile.lock", name,
|
|
1370
|
+
"Pipfile 声明了 %s(%s),但 Pipfile.lock 中没有该包" % (name, info["spec"]),
|
|
1371
|
+
"锁文件过期或手工改动,运行 pipenv lock 重新生成", ecosystem="python"))
|
|
1372
|
+
continue
|
|
1373
|
+
ver = str(ent.get("version") or "").lstrip("==")
|
|
1374
|
+
if ver and info["spec"] not in ("*", "") and not pep440_satisfies(ver, info["spec"]):
|
|
1375
|
+
findings.append(Finding(
|
|
1376
|
+
"lockfile_range_unsatisfied", "high", "Pipfile.lock", name,
|
|
1377
|
+
"Pipfile.lock 中 %s=%s 不满足 Pipfile 声明 %s" % (name, ver, info["spec"]),
|
|
1378
|
+
"声明与锁定不一致,安装可能拉取意外版本", ecosystem="python"))
|
|
1379
|
+
hashes = ent.get("hashes") or []
|
|
1380
|
+
if not hashes:
|
|
1381
|
+
findings.append(Finding(
|
|
1382
|
+
"lockfile_integrity_missing", "medium", "Pipfile.lock", name,
|
|
1383
|
+
"Pipfile.lock 中 %s 缺少 hashes 校验" % name,
|
|
1384
|
+
"缺少完整性校验,安装无法防篡改", ecosystem="python"))
|
|
1385
|
+
sbom_pkgs.append({
|
|
1386
|
+
"ecosystem": "python",
|
|
1387
|
+
"name": name,
|
|
1388
|
+
"version": ver,
|
|
1389
|
+
"resolved": "",
|
|
1390
|
+
"integrity": (hashes[0] if hashes else ""),
|
|
1391
|
+
"scope": "optional" if info["group"] == "dev-packages" else "required",
|
|
1392
|
+
"direct": True,
|
|
1393
|
+
"deps": [],
|
|
1394
|
+
})
|
|
1395
|
+
pypi_urls = [str(s.get("url")) for s in sources if s.get("url") and "pypi.org" in str(s.get("url"))]
|
|
1396
|
+
private_urls = [str(s.get("url")) for s in sources if s.get("url") and "pypi.org" not in str(s.get("url"))]
|
|
1397
|
+
if pypi_urls and private_urls:
|
|
1398
|
+
findings.append(Finding(
|
|
1399
|
+
"confusion_extra_index", "medium", "Pipfile", None,
|
|
1400
|
+
"Pipfile 同时配置了公共 PyPI 与私有源(%s),私有包名存在依赖混淆风险" % ", ".join(private_urls),
|
|
1401
|
+
"依赖混淆高危配置:建议仅使用单一可信源并锁定私有包名", ecosystem="python"))
|
|
1402
|
+
for src in sources:
|
|
1403
|
+
url = src.get("url")
|
|
1404
|
+
if not url:
|
|
1405
|
+
continue
|
|
1406
|
+
suspicious, reason = _is_suspicious_url(url)
|
|
1407
|
+
if suspicious:
|
|
1408
|
+
findings.append(Finding(
|
|
1409
|
+
"confusion_suspicious_registry", "medium", "Pipfile", None,
|
|
1410
|
+
"源地址可疑:%s(%s)" % (url, reason),
|
|
1411
|
+
"请核对仓库地址来源", ecosystem="python"))
|
|
1412
|
+
return True
|
|
1413
|
+
|
|
1414
|
+
|
|
1415
|
+
# --------------------------------------------------------------------------
|
|
1416
|
+
# maven ecosystem (basic)
|
|
1417
|
+
# --------------------------------------------------------------------------
|
|
1418
|
+
|
|
1419
|
+
def check_maven(project_dir, findings, sbom_pkgs):
|
|
1420
|
+
base = Path(project_dir)
|
|
1421
|
+
pom = base / "pom.xml"
|
|
1422
|
+
if not pom.exists():
|
|
1423
|
+
return False
|
|
1424
|
+
import xml.etree.ElementTree as ET
|
|
1425
|
+
|
|
1426
|
+
def local(tag):
|
|
1427
|
+
return tag.rsplit("}", 1)[-1]
|
|
1428
|
+
|
|
1429
|
+
def child(el, name):
|
|
1430
|
+
for c in el:
|
|
1431
|
+
if local(c.tag) == name:
|
|
1432
|
+
return c
|
|
1433
|
+
return None
|
|
1434
|
+
|
|
1435
|
+
def children(el, name):
|
|
1436
|
+
return [c for c in el if local(c.tag) == name]
|
|
1437
|
+
|
|
1438
|
+
try:
|
|
1439
|
+
root = ET.fromstring(_read_text(pom))
|
|
1440
|
+
except Exception:
|
|
1441
|
+
findings.append(Finding(
|
|
1442
|
+
"lockfile_parse_error", "medium", "pom.xml", None,
|
|
1443
|
+
"pom.xml 解析失败(XML 不合法)",
|
|
1444
|
+
"请修复 pom.xml 格式", ecosystem="maven"))
|
|
1445
|
+
return True
|
|
1446
|
+
|
|
1447
|
+
props = {}
|
|
1448
|
+
pe = child(root, "properties")
|
|
1449
|
+
if pe is not None:
|
|
1450
|
+
for c in pe:
|
|
1451
|
+
props[local(c.tag)] = (c.text or "").strip()
|
|
1452
|
+
|
|
1453
|
+
managed = {}
|
|
1454
|
+
dm = child(root, "dependencyManagement")
|
|
1455
|
+
if dm is not None:
|
|
1456
|
+
dme = child(dm, "dependencies")
|
|
1457
|
+
if dme is not None:
|
|
1458
|
+
for d in children(dme, "dependency"):
|
|
1459
|
+
g = child(d, "groupId")
|
|
1460
|
+
a = child(d, "artifactId")
|
|
1461
|
+
v = child(d, "version")
|
|
1462
|
+
if g is not None and a is not None:
|
|
1463
|
+
managed[(g.text or "").strip(), (a.text or "").strip()] = (v.text or "").strip() if v is not None else None
|
|
1464
|
+
|
|
1465
|
+
deps = []
|
|
1466
|
+
deps_el = child(root, "dependencies")
|
|
1467
|
+
if deps_el is not None:
|
|
1468
|
+
for d in children(deps_el, "dependency"):
|
|
1469
|
+
g = child(d, "groupId")
|
|
1470
|
+
a = child(d, "artifactId")
|
|
1471
|
+
v = child(d, "version")
|
|
1472
|
+
sc = child(d, "scope")
|
|
1473
|
+
ga = ((g.text or "").strip(), (a.text or "").strip())
|
|
1474
|
+
ver = (v.text or "").strip() if v is not None else None
|
|
1475
|
+
if ver and ver.startswith("$" + "{"):
|
|
1476
|
+
ver = props.get(ver[2:-1])
|
|
1477
|
+
if not ver:
|
|
1478
|
+
ver = managed.get(ga)
|
|
1479
|
+
scope = (sc.text or "").strip() if sc is not None else "compile"
|
|
1480
|
+
deps.append({"group": ga[0], "artifact": ga[1], "version": ver, "scope": scope})
|
|
1481
|
+
|
|
1482
|
+
for d in deps:
|
|
1483
|
+
key = "%s:%s" % (d["group"], d["artifact"])
|
|
1484
|
+
if not d["version"]:
|
|
1485
|
+
findings.append(Finding(
|
|
1486
|
+
"unpinned", "medium", "pom.xml", key,
|
|
1487
|
+
"依赖 %s 未固定版本(未声明 <version> 且不在 dependencyManagement)" % key,
|
|
1488
|
+
"版本浮动可能被替换为恶意版本,建议显式固定", ecosystem="maven"))
|
|
1489
|
+
elif d["version"].endswith("-SNAPSHOT"):
|
|
1490
|
+
findings.append(Finding(
|
|
1491
|
+
"snapshot", "low", "pom.xml", key,
|
|
1492
|
+
"依赖 %s 使用 SNAPSHOT 版本 %s" % (key, d["version"]),
|
|
1493
|
+
"SNAPSHOT 版本可变,发布物不应依赖", ecosystem="maven"))
|
|
1494
|
+
sbom_pkgs.append({
|
|
1495
|
+
"ecosystem": "maven",
|
|
1496
|
+
"name": key,
|
|
1497
|
+
"version": d["version"] or "",
|
|
1498
|
+
"resolved": "",
|
|
1499
|
+
"integrity": "",
|
|
1500
|
+
"scope": "optional" if d["scope"] not in ("compile", "runtime") else "required",
|
|
1501
|
+
"direct": True,
|
|
1502
|
+
"deps": [],
|
|
1503
|
+
})
|
|
1504
|
+
|
|
1505
|
+
rep_el = child(root, "repositories")
|
|
1506
|
+
if rep_el is not None:
|
|
1507
|
+
for r in children(rep_el, "repository"):
|
|
1508
|
+
u = child(r, "url")
|
|
1509
|
+
if u is not None and (u.text or "").strip():
|
|
1510
|
+
url = u.text.strip()
|
|
1511
|
+
suspicious, reason = _is_suspicious_url(url)
|
|
1512
|
+
if suspicious:
|
|
1513
|
+
findings.append(Finding(
|
|
1514
|
+
"confusion_suspicious_registry", "medium", "pom.xml", None,
|
|
1515
|
+
"仓库地址可疑:%s(%s)" % (url, reason),
|
|
1516
|
+
"请核对仓库地址来源", ecosystem="maven"))
|
|
1517
|
+
return True
|
|
1518
|
+
|
|
1519
|
+
|
|
1520
|
+
# --------------------------------------------------------------------------
|
|
1521
|
+
# SBOM-lite (CycloneDX 1.5 subset)
|
|
1522
|
+
# --------------------------------------------------------------------------
|
|
1523
|
+
|
|
1524
|
+
def _purl(ecosystem, name, version):
|
|
1525
|
+
if ecosystem == "npm":
|
|
1526
|
+
n = name
|
|
1527
|
+
if n.startswith("@"):
|
|
1528
|
+
n = "%40" + n[1:]
|
|
1529
|
+
return ("pkg:npm/%s@%s" % (n, version)) if version else ("pkg:npm/%s" % n)
|
|
1530
|
+
if ecosystem == "python":
|
|
1531
|
+
return ("pkg:pypi/%s@%s" % (name, version)) if version else ("pkg:pypi/%s" % name)
|
|
1532
|
+
if ecosystem == "maven" and ":" in name:
|
|
1533
|
+
g, a = name.split(":", 1)
|
|
1534
|
+
return ("pkg:maven/%s/%s@%s" % (g, a, version)) if version else ("pkg:maven/%s/%s" % (g, a))
|
|
1535
|
+
return ("pkg:generic/%s@%s" % (name, version)) if version else ("pkg:generic/%s" % name)
|
|
1536
|
+
|
|
1537
|
+
|
|
1538
|
+
def build_sbom(sbom_pkgs, include_dev=True, root_component=None):
|
|
1539
|
+
"""Build a CycloneDX 1.5 subset JSON from collected packages."""
|
|
1540
|
+
seen = set()
|
|
1541
|
+
uniq = []
|
|
1542
|
+
for p in sbom_pkgs:
|
|
1543
|
+
key = (p["ecosystem"], p["name"], p["version"], p.get("resolved", ""))
|
|
1544
|
+
if key in seen:
|
|
1545
|
+
continue
|
|
1546
|
+
seen.add(key)
|
|
1547
|
+
uniq.append(p)
|
|
1548
|
+
if not include_dev:
|
|
1549
|
+
uniq = [p for p in uniq if p.get("scope") != "optional"]
|
|
1550
|
+
|
|
1551
|
+
name_purl = {}
|
|
1552
|
+
for p in uniq:
|
|
1553
|
+
npk = (p["ecosystem"], p["name"])
|
|
1554
|
+
if npk not in name_purl:
|
|
1555
|
+
name_purl[npk] = _purl(p["ecosystem"], p["name"], p["version"])
|
|
1556
|
+
|
|
1557
|
+
components = []
|
|
1558
|
+
for p in sorted(uniq, key=lambda x: (x["ecosystem"], x["name"].lower(), x["version"])):
|
|
1559
|
+
props = []
|
|
1560
|
+
if p.get("resolved"):
|
|
1561
|
+
props.append({"name": "yotta-chain:resolved", "value": p["resolved"]})
|
|
1562
|
+
if p.get("integrity"):
|
|
1563
|
+
props.append({"name": "yotta-chain:integrity", "value": p["integrity"]})
|
|
1564
|
+
if p.get("direct"):
|
|
1565
|
+
props.append({"name": "yotta-chain:direct", "value": "true"})
|
|
1566
|
+
components.append({
|
|
1567
|
+
"type": "library",
|
|
1568
|
+
"name": p["name"],
|
|
1569
|
+
"version": p["version"],
|
|
1570
|
+
"scope": p.get("scope", "required"),
|
|
1571
|
+
"purl": _purl(p["ecosystem"], p["name"], p["version"]),
|
|
1572
|
+
"properties": props,
|
|
1573
|
+
})
|
|
1574
|
+
|
|
1575
|
+
dependencies = []
|
|
1576
|
+
if root_component and root_component.get("name"):
|
|
1577
|
+
root_ref = _purl(root_component.get("ecosystem", "generic"), root_component["name"], root_component.get("version") or "")
|
|
1578
|
+
direct_refs = sorted({
|
|
1579
|
+
_purl(p["ecosystem"], p["name"], p["version"])
|
|
1580
|
+
for p in uniq if p.get("direct")
|
|
1581
|
+
})
|
|
1582
|
+
dependencies.append({"ref": root_ref, "dependsOn": direct_refs})
|
|
1583
|
+
for p in uniq:
|
|
1584
|
+
purl = _purl(p["ecosystem"], p["name"], p["version"])
|
|
1585
|
+
refs = []
|
|
1586
|
+
for dn in p.get("deps") or []:
|
|
1587
|
+
target = name_purl.get((p["ecosystem"], dn))
|
|
1588
|
+
if target and target != purl:
|
|
1589
|
+
refs.append(target)
|
|
1590
|
+
dependencies.append({"ref": purl, "dependsOn": sorted(set(refs))})
|
|
1591
|
+
|
|
1592
|
+
bom = {
|
|
1593
|
+
"bomFormat": "CycloneDX",
|
|
1594
|
+
"specVersion": "1.5",
|
|
1595
|
+
"version": 1,
|
|
1596
|
+
"serialNumber": "urn:uuid:" + _uuid(),
|
|
1597
|
+
"metadata": {
|
|
1598
|
+
"timestamp": _now_iso(),
|
|
1599
|
+
"tools": [{"vendor": "YottaMeta", "name": "yotta-chain", "version": VERSION}],
|
|
1600
|
+
},
|
|
1601
|
+
"components": components,
|
|
1602
|
+
"dependencies": dependencies,
|
|
1603
|
+
}
|
|
1604
|
+
if root_component and root_component.get("name"):
|
|
1605
|
+
bom["metadata"]["component"] = {
|
|
1606
|
+
"type": "application",
|
|
1607
|
+
"name": root_component["name"],
|
|
1608
|
+
"version": root_component.get("version") or "",
|
|
1609
|
+
}
|
|
1610
|
+
return bom
|
|
1611
|
+
|
|
1612
|
+
|
|
1613
|
+
def _uuid():
|
|
1614
|
+
import uuid
|
|
1615
|
+
return str(uuid.uuid4())
|
|
1616
|
+
|
|
1617
|
+
|
|
1618
|
+
def _now_iso():
|
|
1619
|
+
import datetime
|
|
1620
|
+
return datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
1621
|
+
|
|
1622
|
+
|
|
1623
|
+
def render_sbom_text(bom):
|
|
1624
|
+
lines = ["元链 yotta-chain — SBOM-lite(CycloneDX 1.5 子集)"]
|
|
1625
|
+
meta = bom.get("metadata", {})
|
|
1626
|
+
comp = meta.get("component") or {}
|
|
1627
|
+
if comp:
|
|
1628
|
+
lines.append("项目:%s %s" % (comp.get("name"), comp.get("version")))
|
|
1629
|
+
lines.append("")
|
|
1630
|
+
lines.append("%-10s %-32s %-18s %-6s %s" % ("生态", "包名", "版本", "直接", "来源"))
|
|
1631
|
+
lines.append("-" * 100)
|
|
1632
|
+
for c in bom.get("components", []):
|
|
1633
|
+
props = {p["name"]: p["value"] for p in c.get("properties", [])}
|
|
1634
|
+
direct = "是" if props.get("yotta-chain:direct") == "true" else "否"
|
|
1635
|
+
resolved = props.get("yotta-chain:resolved", "")
|
|
1636
|
+
if len(resolved) > 64:
|
|
1637
|
+
resolved = resolved[:61] + "..."
|
|
1638
|
+
purl = c.get("purl", "")
|
|
1639
|
+
eco = "npm" if "pkg:npm" in purl else ("pypi" if "pkg:pypi" in purl else ("maven" if "pkg:maven" in purl else "?"))
|
|
1640
|
+
lines.append("%-10s %-32s %-18s %-6s %s" % (eco, c.get("name", ""), c.get("version", ""), direct, resolved))
|
|
1641
|
+
lines.append("")
|
|
1642
|
+
lines.append("组件数:%d" % len(bom.get("components", [])))
|
|
1643
|
+
return "\n".join(lines)
|
|
1644
|
+
|
|
1645
|
+
|
|
1646
|
+
# --------------------------------------------------------------------------
|
|
1647
|
+
# CLI
|
|
1648
|
+
# --------------------------------------------------------------------------
|
|
1649
|
+
|
|
1650
|
+
def _csv_escape(v):
|
|
1651
|
+
s = str(v)
|
|
1652
|
+
if any(c in s for c in ',"\n\r'):
|
|
1653
|
+
return '"' + s.replace('"', '""') + '"'
|
|
1654
|
+
return s
|
|
1655
|
+
|
|
1656
|
+
|
|
1657
|
+
def _collect(project_dir, findings, sbom_pkgs, root_component):
|
|
1658
|
+
"""Run all ecosystem checks; returns sorted ecosystem list."""
|
|
1659
|
+
base = Path(project_dir)
|
|
1660
|
+
eco = []
|
|
1661
|
+
if check_npm(project_dir, findings, sbom_pkgs):
|
|
1662
|
+
eco.append("npm")
|
|
1663
|
+
pj = parse_package_json(_read_text(base / "package.json"))
|
|
1664
|
+
if pj.get("name"):
|
|
1665
|
+
root_component.update({"ecosystem": "npm", "name": pj["name"], "version": str(pj.get("version") or "")})
|
|
1666
|
+
if check_python(project_dir, findings, sbom_pkgs):
|
|
1667
|
+
eco.append("python")
|
|
1668
|
+
py = parse_toml(_read_text(base / "pyproject.toml"))
|
|
1669
|
+
proj = py.get("project") or {}
|
|
1670
|
+
if proj.get("name"):
|
|
1671
|
+
root_component.update({"ecosystem": "python", "name": proj["name"], "version": str(proj.get("version") or "")})
|
|
1672
|
+
if check_maven(project_dir, findings, sbom_pkgs):
|
|
1673
|
+
eco.append("maven")
|
|
1674
|
+
return eco
|
|
1675
|
+
|
|
1676
|
+
|
|
1677
|
+
def cmd_scan(args):
|
|
1678
|
+
base = Path(args.path)
|
|
1679
|
+
if not base.is_dir():
|
|
1680
|
+
print("错误:路径不存在或不是目录:%s" % base, file=sys.stderr)
|
|
1681
|
+
return 4
|
|
1682
|
+
findings = []
|
|
1683
|
+
sbom_pkgs = []
|
|
1684
|
+
root_component = {}
|
|
1685
|
+
eco = _collect(args.path, findings, sbom_pkgs, root_component)
|
|
1686
|
+
if not eco:
|
|
1687
|
+
print("错误:%s 下未发现支持的依赖清单/锁文件(package.json / requirements*.txt / pyproject.toml / Pipfile / pom.xml)" % base, file=sys.stderr)
|
|
1688
|
+
return 4
|
|
1689
|
+
|
|
1690
|
+
min_rank = SEVERITY_RANK.get(args.level, 0)
|
|
1691
|
+
shown = [f for f in findings if SEVERITY_RANK.get(f.severity, 0) >= min_rank]
|
|
1692
|
+
shown.sort(key=lambda f: (SEVERITY_RANK.get(f.severity, 0), f.rule, f.package or ""))
|
|
1693
|
+
gate_rank = SEVERITY_RANK.get(args.gate, 0)
|
|
1694
|
+
gate_hits = [f for f in findings if SEVERITY_RANK.get(f.severity, 0) >= gate_rank]
|
|
1695
|
+
|
|
1696
|
+
if args.format == "json":
|
|
1697
|
+
out = {
|
|
1698
|
+
"tool": "yotta-chain",
|
|
1699
|
+
"version": VERSION,
|
|
1700
|
+
"project": str(base),
|
|
1701
|
+
"ecosystems": eco,
|
|
1702
|
+
"files": sorted({f.file for f in findings}),
|
|
1703
|
+
"summary": {s: sum(1 for f in findings if f.severity == s) for s in SEVERITY_ORDER},
|
|
1704
|
+
"findings": [f.to_dict() for f in shown],
|
|
1705
|
+
}
|
|
1706
|
+
text = json.dumps(out, ensure_ascii=False, indent=2)
|
|
1707
|
+
elif args.format == "csv":
|
|
1708
|
+
lines = ["rule,severity,ecosystem,file,line,package,message,detail"]
|
|
1709
|
+
for f in shown:
|
|
1710
|
+
ln = f.line if f.line is not None else ""
|
|
1711
|
+
lines.append(",".join(_csv_escape(x) for x in (
|
|
1712
|
+
f.rule, f.severity, f.ecosystem, f.file, ln, f.package or "", f.message, f.detail)))
|
|
1713
|
+
text = "\n".join(lines)
|
|
1714
|
+
else:
|
|
1715
|
+
lines = ["元链 yotta-chain %s — 供应链依赖校验" % VERSION]
|
|
1716
|
+
lines.append("项目:%s 生态:%s" % (base, ", ".join(eco)))
|
|
1717
|
+
if not shown:
|
|
1718
|
+
lines.append("未发现 %s 及以上风险项" % args.level)
|
|
1719
|
+
for s in SEVERITY_ORDER:
|
|
1720
|
+
for f in [x for x in shown if x.severity == s]:
|
|
1721
|
+
loc = ("%s:%s" % (f.file, f.line)) if f.line is not None else f.file
|
|
1722
|
+
pkg = (" %s" % f.package) if f.package else ""
|
|
1723
|
+
lines.append("[%s] %s%s (%s)" % (s, f.rule, pkg, loc))
|
|
1724
|
+
lines.append(" %s" % f.message)
|
|
1725
|
+
if f.detail:
|
|
1726
|
+
lines.append(" → %s" % f.detail)
|
|
1727
|
+
lines.append("")
|
|
1728
|
+
counts = ", ".join("%s %d" % (s, sum(1 for f in findings if f.severity == s)) for s in SEVERITY_ORDER)
|
|
1729
|
+
lines.append("共 %d 项(%s);达到 gate=%s 的有 %d 项" % (len(findings), counts, args.gate, len(gate_hits)))
|
|
1730
|
+
text = "\n".join(lines)
|
|
1731
|
+
|
|
1732
|
+
if args.output:
|
|
1733
|
+
Path(args.output).write_text(text, encoding="utf-8")
|
|
1734
|
+
print("报告已写入 %s" % args.output)
|
|
1735
|
+
else:
|
|
1736
|
+
print(text)
|
|
1737
|
+
return 1 if gate_hits else 0
|
|
1738
|
+
|
|
1739
|
+
|
|
1740
|
+
def cmd_sbom(args):
|
|
1741
|
+
base = Path(args.path)
|
|
1742
|
+
if not base.is_dir():
|
|
1743
|
+
print("错误:路径不存在或不是目录:%s" % base, file=sys.stderr)
|
|
1744
|
+
return 4
|
|
1745
|
+
findings = []
|
|
1746
|
+
sbom_pkgs = []
|
|
1747
|
+
root_component = {}
|
|
1748
|
+
eco = _collect(args.path, findings, sbom_pkgs, root_component)
|
|
1749
|
+
if not eco:
|
|
1750
|
+
print("错误:%s 下未发现支持的依赖清单/锁文件" % base, file=sys.stderr)
|
|
1751
|
+
return 4
|
|
1752
|
+
bom = build_sbom(sbom_pkgs, include_dev=not args.exclude_dev, root_component=root_component)
|
|
1753
|
+
if args.format == "text":
|
|
1754
|
+
text = render_sbom_text(bom)
|
|
1755
|
+
else:
|
|
1756
|
+
text = json.dumps(bom, ensure_ascii=False, indent=2)
|
|
1757
|
+
if args.output:
|
|
1758
|
+
Path(args.output).write_text(text, encoding="utf-8")
|
|
1759
|
+
print("SBOM 已写入 %s" % args.output)
|
|
1760
|
+
else:
|
|
1761
|
+
print(text)
|
|
1762
|
+
return 0
|
|
1763
|
+
|
|
1764
|
+
|
|
1765
|
+
def main(argv=None):
|
|
1766
|
+
ap = argparse.ArgumentParser(
|
|
1767
|
+
prog="yotta-chain",
|
|
1768
|
+
description="元链 yotta-chain — 供应链依赖校验引擎(零依赖 / 纯本地 / 不做在线 CVE)")
|
|
1769
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
1770
|
+
|
|
1771
|
+
p_scan = sub.add_parser("scan", help="扫描依赖混淆 / lockfile 一致性 / 缺失锁文件 / typo-squat")
|
|
1772
|
+
p_scan.add_argument("--path", default=".", help="项目目录(默认当前目录)")
|
|
1773
|
+
p_scan.add_argument("--format", choices=["text", "json", "csv"], default="text", help="输出格式")
|
|
1774
|
+
p_scan.add_argument("--level", choices=SEVERITY_ORDER, default="info", help="只显示 >= 该级别的发现")
|
|
1775
|
+
p_scan.add_argument("--gate", choices=SEVERITY_ORDER, default="info", help="达到该级别即退出码 1(CI 用)")
|
|
1776
|
+
p_scan.add_argument("--output", "-o", default=None, help="写入文件")
|
|
1777
|
+
p_scan.set_defaults(func=cmd_scan)
|
|
1778
|
+
|
|
1779
|
+
p_sbom = sub.add_parser("sbom", help="生成 SBOM-lite(CycloneDX 1.5 子集 JSON)")
|
|
1780
|
+
p_sbom.add_argument("--path", default=".")
|
|
1781
|
+
p_sbom.add_argument("--format", choices=["cyclonedx", "text"], default="cyclonedx")
|
|
1782
|
+
p_sbom.add_argument("--exclude-dev", action="store_true", help="不包含 dev/optional 依赖")
|
|
1783
|
+
p_sbom.add_argument("--output", "-o", default=None)
|
|
1784
|
+
p_sbom.set_defaults(func=cmd_sbom)
|
|
1785
|
+
|
|
1786
|
+
p_ver = sub.add_parser("version", help="显示版本")
|
|
1787
|
+
p_ver.set_defaults(func=lambda a: (print(VERSION) or 0))
|
|
1788
|
+
|
|
1789
|
+
args = ap.parse_args(argv)
|
|
1790
|
+
return args.func(args)
|
|
1791
|
+
|
|
1792
|
+
|
|
1793
|
+
if __name__ == "__main__":
|
|
1794
|
+
sys.exit(main())
|