sast-eval 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sast_eval/__init__.py +5 -0
- sast_eval/adapters/__init__.py +0 -0
- sast_eval/adapters/common.py +66 -0
- sast_eval/adapters/owasp_adapter.py +129 -0
- sast_eval/adapters/sastbench_adapter.py +231 -0
- sast_eval/cli.py +190 -0
- sast_eval/exploit/__init__.py +5 -0
- sast_eval/exploit/oracle.py +319 -0
- sast_eval/exploit/oracles/__init__.py +15 -0
- sast_eval/exploit/oracles/bountytasks_oracle.py +32 -0
- sast_eval/exploit/oracles/cybergym_oracle.py +33 -0
- sast_eval/importers/FORMATS.md +129 -0
- sast_eval/importers/__init__.py +0 -0
- sast_eval/importers/bountytasks_importer.py +192 -0
- sast_eval/importers/cwebench_importer.py +206 -0
- sast_eval/importers/cybergym_importer.py +160 -0
- sast_eval/matching/__init__.py +0 -0
- sast_eval/matching/cwe_map.json +9 -0
- sast_eval/matching/matcher.py +470 -0
- sast_eval/scoring/__init__.py +0 -0
- sast_eval/scoring/metrics.py +645 -0
- sast_eval/tools/__init__.py +0 -0
- sast_eval/tools/fetch_sources.py +267 -0
- sast_eval/tools/package_codebases.py +221 -0
- sast_eval-0.1.0.dist-info/METADATA +262 -0
- sast_eval-0.1.0.dist-info/RECORD +28 -0
- sast_eval-0.1.0.dist-info/WHEEL +4 -0
- sast_eval-0.1.0.dist-info/entry_points.txt +2 -0
sast_eval/__init__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Shared helpers for adapters and importers.
|
|
2
|
+
|
|
3
|
+
CWE normalization (§2.1) and bounty-value parsing (reusing the rules of
|
|
4
|
+
corpus/bountytasks/calculate_bounties.py).
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
from typing import Optional
|
|
10
|
+
|
|
11
|
+
# Matches the first integer in a CWE-ish string, with or without the "CWE-" prefix.
|
|
12
|
+
_CWE_INT = re.compile(r"(\d+)")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def normalize_cwe(raw: Optional[str]) -> str:
|
|
16
|
+
"""Normalize a raw CWE string to canonical ``CWE-<int>`` form.
|
|
17
|
+
|
|
18
|
+
Per §2.1: extract the integer -> ``CWE-<int>`` (zero-padded to 3 for
|
|
19
|
+
display). Empty/unparseable -> ``CWE-UNKNOWN``.
|
|
20
|
+
|
|
21
|
+
Examples (from §2.1):
|
|
22
|
+
"CWE-22: Path Traversal" -> "CWE-022"
|
|
23
|
+
"400: Denial of Service" -> "CWE-400"
|
|
24
|
+
"CWE-29: Path Traversal: '\\..\\filename'" -> "CWE-029"
|
|
25
|
+
"" -> "CWE-UNKNOWN"
|
|
26
|
+
"22" -> "CWE-022"
|
|
27
|
+
None -> "CWE-UNKNOWN"
|
|
28
|
+
"""
|
|
29
|
+
if raw is None:
|
|
30
|
+
return "CWE-UNKNOWN"
|
|
31
|
+
s = str(raw).strip()
|
|
32
|
+
if not s:
|
|
33
|
+
return "CWE-UNKNOWN"
|
|
34
|
+
m = _CWE_INT.search(s)
|
|
35
|
+
if not m:
|
|
36
|
+
return "CWE-UNKNOWN"
|
|
37
|
+
return f"CWE-{int(m.group(1)):03d}"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def parse_bounty_value(value) -> float:
|
|
41
|
+
"""Parse a bounty value from various formats.
|
|
42
|
+
|
|
43
|
+
Reuses the rules of ``corpus/bountytasks/calculate_bounties.py``:
|
|
44
|
+
- None / "" -> 0.0
|
|
45
|
+
- already a number -> float
|
|
46
|
+
- leading "$" stripped
|
|
47
|
+
- "X to Y" range -> average of the two ends
|
|
48
|
+
- otherwise best-effort float(), 0.0 on failure
|
|
49
|
+
"""
|
|
50
|
+
if value is None or value == "":
|
|
51
|
+
return 0.0
|
|
52
|
+
if isinstance(value, (int, float)):
|
|
53
|
+
return float(value)
|
|
54
|
+
if isinstance(value, str) and value.startswith("$"):
|
|
55
|
+
value = value[1:]
|
|
56
|
+
if isinstance(value, str) and " to " in value:
|
|
57
|
+
parts = value.split(" to ")
|
|
58
|
+
if len(parts) == 2:
|
|
59
|
+
try:
|
|
60
|
+
return (float(parts[0]) + float(parts[1])) / 2
|
|
61
|
+
except ValueError:
|
|
62
|
+
pass
|
|
63
|
+
try:
|
|
64
|
+
return float(value)
|
|
65
|
+
except (ValueError, TypeError):
|
|
66
|
+
return 0.0
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""OWASP BenchmarkJava adapter — the only end-to-end adapter in this repo (§3.2).
|
|
2
|
+
|
|
3
|
+
Parses ``expectedresults-1.2.csv`` and emits one unified task record (§2) per row
|
|
4
|
+
to ``tasks/owasp.jsonl``.
|
|
5
|
+
|
|
6
|
+
Per §3.2 / Step 3b:
|
|
7
|
+
- One task per CSV row. Columns: test name, category, real vulnerability, cwe.
|
|
8
|
+
- ``source_root`` = the benchmark repo root (tools scan the whole repo; matching
|
|
9
|
+
is per test-case file).
|
|
10
|
+
- ``vulnerable_files`` = ``src/main/java/org/owasp/benchmark/testcode/<TestName>.java``
|
|
11
|
+
(directory is ``testcode``, not ``testcases``).
|
|
12
|
+
- ``real=false`` -> ``fp_trap: true`` (~1,000 of 2,740 rows are FP traps).
|
|
13
|
+
- Category code -> CWE via the fixed OWASP mapping; the CSV's own ``cwe``
|
|
14
|
+
column is preferred when present (it is authoritative per-row).
|
|
15
|
+
- No weights (all tasks equal).
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import argparse
|
|
20
|
+
import csv
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
from sast_eval.adapters.common import normalize_cwe
|
|
26
|
+
|
|
27
|
+
# Fixed OWASP category -> CWE mapping (§3.2). Used as a fallback when the CSV's
|
|
28
|
+
# per-row ``cwe`` column is absent/empty. Values are zero-padded to 3 for display.
|
|
29
|
+
OWASP_CATEGORY_TO_CWE = {
|
|
30
|
+
"sqli": "CWE-089",
|
|
31
|
+
"xss": "CWE-079",
|
|
32
|
+
"cmdi": "CWE-078",
|
|
33
|
+
"pathtraver": "CWE-022",
|
|
34
|
+
"ldapi": "CWE-090",
|
|
35
|
+
"hash": "CWE-328",
|
|
36
|
+
"weakrand": "CWE-330",
|
|
37
|
+
"trustbound": "CWE-501",
|
|
38
|
+
"securecookie": "CWE-614",
|
|
39
|
+
"crypto": "CWE-327",
|
|
40
|
+
"xpathi": "CWE-643",
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
TESTCODE_REL = "src/main/java/org/owasp/benchmark/testcode"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _category_to_cwe(category: str) -> str:
|
|
47
|
+
return OWASP_CATEGORY_TO_CWE.get(category, "CWE-UNKNOWN")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def build_tasks(root: str) -> list[dict]:
|
|
51
|
+
root_path = Path(root).resolve()
|
|
52
|
+
csv_path = root_path / "expectedresults-1.2.csv"
|
|
53
|
+
if not csv_path.exists():
|
|
54
|
+
raise FileNotFoundError(f"expectedresults-1.2.csv not found at {csv_path}")
|
|
55
|
+
|
|
56
|
+
tasks: list[dict] = []
|
|
57
|
+
with open(csv_path, newline="", encoding="utf-8") as f:
|
|
58
|
+
reader = csv.reader(f)
|
|
59
|
+
header = next(reader, None) # the '#' comment line
|
|
60
|
+
if not header or not header[0].lstrip().startswith("#"):
|
|
61
|
+
# Defensive: if the comment line is missing, the first row is data.
|
|
62
|
+
# Re-open and treat all rows as data. (The known file always has it.)
|
|
63
|
+
f.seek(0)
|
|
64
|
+
reader = csv.reader(f)
|
|
65
|
+
|
|
66
|
+
for row in reader:
|
|
67
|
+
if not row or all(not c.strip() for c in row):
|
|
68
|
+
continue
|
|
69
|
+
# Columns: # test name, category, real vulnerability, cwe
|
|
70
|
+
test_name = row[0].strip()
|
|
71
|
+
category = row[1].strip() if len(row) > 1 else ""
|
|
72
|
+
real = row[2].strip().lower() if len(row) > 2 else "true"
|
|
73
|
+
cwe_raw = row[3].strip() if len(row) > 3 else ""
|
|
74
|
+
|
|
75
|
+
# Prefer the CSV's per-row cwe column (authoritative); fall back to
|
|
76
|
+
# the fixed category mapping.
|
|
77
|
+
cwe = normalize_cwe(cwe_raw) if cwe_raw else _category_to_cwe(category)
|
|
78
|
+
|
|
79
|
+
fp_trap = (real == "false")
|
|
80
|
+
vuln_file = f"{TESTCODE_REL}/{test_name}.java"
|
|
81
|
+
|
|
82
|
+
task = {
|
|
83
|
+
"task_id": f"owasp/{test_name}",
|
|
84
|
+
"benchmark": "owasp",
|
|
85
|
+
"language": "java",
|
|
86
|
+
"source_root": str(root_path),
|
|
87
|
+
"vcs": {
|
|
88
|
+
"type": "git",
|
|
89
|
+
"commit": "", # filled by PROVENANCE from corpus hash
|
|
90
|
+
"checked_out": True,
|
|
91
|
+
},
|
|
92
|
+
"ground_truth": {
|
|
93
|
+
"cwe": cwe,
|
|
94
|
+
"cwe_raw": cwe_raw or category,
|
|
95
|
+
"cve": None,
|
|
96
|
+
"vulnerable_files": [vuln_file],
|
|
97
|
+
"vulnerable_methods": [],
|
|
98
|
+
"fp_trap": fp_trap,
|
|
99
|
+
"notes": "OWASP BenchmarkJava v1.2; category=%s" % category,
|
|
100
|
+
},
|
|
101
|
+
"weights": {},
|
|
102
|
+
"meta": {
|
|
103
|
+
"category": category,
|
|
104
|
+
"real_vulnerability": not fp_trap,
|
|
105
|
+
},
|
|
106
|
+
}
|
|
107
|
+
tasks.append(task)
|
|
108
|
+
return tasks
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def main(argv: list[str] | None = None) -> int:
|
|
112
|
+
parser = argparse.ArgumentParser(description="Build OWASP BenchmarkJava task records")
|
|
113
|
+
parser.add_argument("--root", required=True, help="Path to BenchmarkJava repo root")
|
|
114
|
+
parser.add_argument("--out", required=True, help="Output JSONL path (tasks/owasp.jsonl)")
|
|
115
|
+
args = parser.parse_args(argv)
|
|
116
|
+
|
|
117
|
+
tasks = build_tasks(args.root)
|
|
118
|
+
os.makedirs(os.path.dirname(os.path.abspath(args.out)), exist_ok=True)
|
|
119
|
+
with open(args.out, "w", encoding="utf-8") as f:
|
|
120
|
+
for t in tasks:
|
|
121
|
+
f.write(json.dumps(t, ensure_ascii=False, sort_keys=True) + "\n")
|
|
122
|
+
|
|
123
|
+
# Idempotency note: deterministic output (sorted keys, CSV order preserved).
|
|
124
|
+
print(f"Wrote {len(tasks)} OWASP task records to {args.out}")
|
|
125
|
+
return 0
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
if __name__ == "__main__":
|
|
129
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""SASTbench adapter — end-to-end (this repo owns matching/scoring).
|
|
2
|
+
|
|
3
|
+
SASTbench evaluates SAST tools on agentic codebases. Its cases are
|
|
4
|
+
self-contained JSON files (``cases/<track>/<caseType>/<id>/case.json``) with
|
|
5
|
+
region-level ground truth: each case annotates one or more regions in the
|
|
6
|
+
source, labelled ``vulnerable`` (scanners must detect) or ``capability_safe``
|
|
7
|
+
(scanners must NOT flag — guarded dangerous code is an FP trap).
|
|
8
|
+
|
|
9
|
+
Per the SASTbench case schema (``schema/case.schema.json``):
|
|
10
|
+
- ``canonicalKind`` is one of 6 SASTbench kinds (command_injection,
|
|
11
|
+
path_traversal, ssrf, auth_bypass, authz_bypass, sql_injection). We map each
|
|
12
|
+
to its primary CWE via ``taxonomy/canonical_kinds.json`` (``cweMappings[0]``).
|
|
13
|
+
- ``regions`` carry ``path``, ``startLine``/``endLine``, ``label``,
|
|
14
|
+
``acceptedKinds`` (for vulnerable regions) and ``requiredGuards``/``capability``
|
|
15
|
+
(for capability_safe regions).
|
|
16
|
+
- ``expectedOutcome.mustDetectRegionIds`` / ``mustNotFlagRegionIds`` state the
|
|
17
|
+
ground truth explicitly.
|
|
18
|
+
- ``files.root`` is the scannable project root (``project/`` for core,
|
|
19
|
+
``../../../../.repos/<snapshot>/`` for full track).
|
|
20
|
+
|
|
21
|
+
Mapping to the unified task schema (§2):
|
|
22
|
+
- ``task_id`` = ``sastbench/<case-id>`` (e.g. ``sastbench/SB-PY-SV-001``).
|
|
23
|
+
- ``source_root`` = absolute path to ``<case_dir>/<files.root>``.
|
|
24
|
+
- ``ground_truth.cwe`` = primary CWE for ``canonicalKind``.
|
|
25
|
+
- ``ground_truth.vulnerable_files`` = paths of ``vulnerable`` regions.
|
|
26
|
+
- ``ground_truth.vulnerable_regions`` = full region list (NEW field; the
|
|
27
|
+
matcher uses this for region-level overlap matching — see
|
|
28
|
+
``matching/matcher.py``).
|
|
29
|
+
- ``ground_truth.fp_trap`` = ``True`` for ``capability_safe`` cases (no
|
|
30
|
+
vulnerable regions; any finding is a FP). For ``mixed_intent`` cases this is
|
|
31
|
+
``False`` (they have vulnerable regions), but their ``capability_safe``
|
|
32
|
+
regions are still FP traps at the region level — handled by the matcher.
|
|
33
|
+
- ``meta`` carries ``track``, ``case_type``, ``canonical_kind``, ``agentic``,
|
|
34
|
+
``profile``, and (for real-world cases) ``repo``, ``cve``, ``ghsa``,
|
|
35
|
+
disclosure dates for knowledge-cutoff gating.
|
|
36
|
+
- ``vcs.checked_out`` = ``False`` + ``meta.source_missing`` = ``True`` when
|
|
37
|
+
``source_root`` doesn't exist on disk (Full Track snapshots live under
|
|
38
|
+
``.repos/`` and must be fetched via ``scripts/setup_repos.py`` — see
|
|
39
|
+
``tools/fetch_sources.py``).
|
|
40
|
+
"""
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import argparse
|
|
44
|
+
import json
|
|
45
|
+
import os
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
|
|
48
|
+
from sast_eval.adapters.common import normalize_cwe
|
|
49
|
+
|
|
50
|
+
# SASTbench canonical kind -> primary CWE (first entry in
|
|
51
|
+
# taxonomy/canonical_kinds.json ``cweMappings``). Used for the unified
|
|
52
|
+
# ``ground_truth.cwe`` field. Region-level ``acceptedKinds`` are matched
|
|
53
|
+
# against the finding's CWE via the same map in the matcher.
|
|
54
|
+
KIND_TO_PRIMARY_CWE: dict[str, str] = {
|
|
55
|
+
"command_injection": "CWE-078",
|
|
56
|
+
"path_traversal": "CWE-022",
|
|
57
|
+
"ssrf": "CWE-918",
|
|
58
|
+
"auth_bypass": "CWE-287",
|
|
59
|
+
"authz_bypass": "CWE-862",
|
|
60
|
+
"sql_injection": "CWE-089",
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
# Reverse: CWE -> SASTbench kind, for matching a finding's CWE against a
|
|
64
|
+
# region's acceptedKinds. A finding counts as matching a region if its CWE
|
|
65
|
+
# maps to any of the region's acceptedKinds. Built from the full
|
|
66
|
+
# cweMappings in taxonomy/canonical_kinds.json at load time (see below).
|
|
67
|
+
_CWE_TO_KINDS: dict[str, list[str]] = {}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _load_kind_cwe_map(taxonomy_dir: Path) -> None:
|
|
71
|
+
"""Populate ``_CWE_TO_KINDS`` from taxonomy/canonical_kinds.json."""
|
|
72
|
+
global _CWE_TO_KINDS
|
|
73
|
+
if _CWE_TO_KINDS:
|
|
74
|
+
return
|
|
75
|
+
kinds_file = taxonomy_dir / "canonical_kinds.json"
|
|
76
|
+
if not kinds_file.is_file():
|
|
77
|
+
return
|
|
78
|
+
data = json.loads(kinds_file.read_text(encoding="utf-8"))
|
|
79
|
+
for kind in data.get("canonicalKinds", []):
|
|
80
|
+
kid = kind.get("id", "")
|
|
81
|
+
for cwe in kind.get("cweMappings", []):
|
|
82
|
+
_CWE_TO_KINDS.setdefault(normalize_cwe(cwe), []).append(kid)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def cwe_to_kinds(cwe: str) -> list[str]:
|
|
86
|
+
"""Return the SASTbench kinds that a CWE maps to (for region matching)."""
|
|
87
|
+
return _CWE_TO_KINDS.get(cwe, [])
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _language_map(raw: str) -> str:
|
|
91
|
+
"""Map SASTbench language labels to our language field."""
|
|
92
|
+
m = {
|
|
93
|
+
"python": "python",
|
|
94
|
+
"typescript": "typescript",
|
|
95
|
+
"rust": "rust",
|
|
96
|
+
"swift": "swift",
|
|
97
|
+
"go": "go",
|
|
98
|
+
"java": "java",
|
|
99
|
+
"clojure": "clojure",
|
|
100
|
+
}
|
|
101
|
+
return m.get((raw or "").strip().lower(), (raw or "").strip().lower() or "unknown")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _profile_for(case: dict) -> str:
|
|
105
|
+
"""Return the SASTbench profile: 'agentic', 'generic', or 'all'."""
|
|
106
|
+
if case.get("caseType") == "real_world_generic":
|
|
107
|
+
return "generic"
|
|
108
|
+
return "agentic" if case.get("agentic", True) else "generic"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _build_region(r: dict) -> dict:
|
|
112
|
+
"""Normalize one SASTbench region into our ground-truth region record."""
|
|
113
|
+
return {
|
|
114
|
+
"id": r.get("id", ""),
|
|
115
|
+
"path": r.get("path", ""),
|
|
116
|
+
"start": r.get("startLine", 0),
|
|
117
|
+
"end": r.get("endLine", 0),
|
|
118
|
+
"label": r.get("label", ""), # "vulnerable" | "capability_safe"
|
|
119
|
+
"accepted_kinds": r.get("acceptedKinds", []),
|
|
120
|
+
"capability": r.get("capability", ""),
|
|
121
|
+
"required_guards": r.get("requiredGuards", []),
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def build_tasks(root: str) -> list[dict]:
|
|
126
|
+
"""Build unified task records from SASTbench case definitions.
|
|
127
|
+
|
|
128
|
+
``root`` is the sast-bench repo root (containing ``cases/`` and
|
|
129
|
+
``taxonomy/``).
|
|
130
|
+
"""
|
|
131
|
+
root_path = Path(root).resolve()
|
|
132
|
+
cases_dir = root_path / "cases"
|
|
133
|
+
if not cases_dir.is_dir():
|
|
134
|
+
raise FileNotFoundError(f"cases/ not found at {cases_dir}")
|
|
135
|
+
|
|
136
|
+
_load_kind_cwe_map(root_path / "taxonomy")
|
|
137
|
+
|
|
138
|
+
tasks: list[dict] = []
|
|
139
|
+
for case_json in sorted(cases_dir.rglob("case.json")):
|
|
140
|
+
case = json.loads(case_json.read_text(encoding="utf-8"))
|
|
141
|
+
case_dir = case_json.parent
|
|
142
|
+
case_id = case.get("id", "")
|
|
143
|
+
if not case_id:
|
|
144
|
+
continue
|
|
145
|
+
|
|
146
|
+
kind = case.get("canonicalKind", "")
|
|
147
|
+
cwe = KIND_TO_PRIMARY_CWE.get(kind, "CWE-UNKNOWN")
|
|
148
|
+
|
|
149
|
+
regions = [_build_region(r) for r in case.get("regions", [])]
|
|
150
|
+
vuln_regions = [r for r in regions if r["label"] == "vulnerable"]
|
|
151
|
+
cap_safe_regions = [r for r in regions if r["label"] == "capability_safe"]
|
|
152
|
+
vuln_files = sorted({r["path"] for r in vuln_regions if r["path"]})
|
|
153
|
+
|
|
154
|
+
# capability_safe cases have no vulnerable regions — any finding is a FP.
|
|
155
|
+
# mixed_intent cases have both; they are NOT fp_trap at the task level
|
|
156
|
+
# (they have real vulns to find), but their capability_safe regions are
|
|
157
|
+
# FP traps at the region level (handled by the matcher).
|
|
158
|
+
fp_trap = (case.get("caseType") == "capability_safe")
|
|
159
|
+
|
|
160
|
+
files_root = case.get("files", {}).get("root", "")
|
|
161
|
+
source_root = (case_dir / files_root).resolve() if files_root else case_dir
|
|
162
|
+
source_missing = not source_root.is_dir()
|
|
163
|
+
|
|
164
|
+
real_world = case.get("realWorld") or {}
|
|
165
|
+
disclosure = real_world.get("disclosure") or {}
|
|
166
|
+
|
|
167
|
+
meta = {
|
|
168
|
+
"track": case.get("track", ""),
|
|
169
|
+
"case_type": case.get("caseType", ""),
|
|
170
|
+
"canonical_kind": kind,
|
|
171
|
+
"agentic": case.get("agentic", True),
|
|
172
|
+
"profile": _profile_for(case),
|
|
173
|
+
"title": case.get("title", ""),
|
|
174
|
+
"source_missing": source_missing,
|
|
175
|
+
}
|
|
176
|
+
if real_world:
|
|
177
|
+
meta["repo"] = real_world.get("repo", "")
|
|
178
|
+
meta["cve"] = real_world.get("cve")
|
|
179
|
+
meta["ghsa"] = real_world.get("ghsa")
|
|
180
|
+
meta["vulnerable_commit"] = real_world.get("vulnerableCommit", "")
|
|
181
|
+
meta["fix_commit"] = real_world.get("fixCommit", "")
|
|
182
|
+
meta["disclosure_ghsa_published"] = disclosure.get("ghsaPublished")
|
|
183
|
+
meta["disclosure_fix_commit_date"] = disclosure.get("fixCommitDate")
|
|
184
|
+
meta["disclosure_cve_published"] = disclosure.get("cvePublished")
|
|
185
|
+
|
|
186
|
+
task = {
|
|
187
|
+
"task_id": f"sastbench/{case_id}",
|
|
188
|
+
"benchmark": "sastbench",
|
|
189
|
+
"language": _language_map(case.get("language", "")),
|
|
190
|
+
"source_root": str(source_root),
|
|
191
|
+
"vcs": {
|
|
192
|
+
"type": "git",
|
|
193
|
+
"commit": real_world.get("vulnerableCommit", ""),
|
|
194
|
+
"checked_out": not source_missing,
|
|
195
|
+
},
|
|
196
|
+
"ground_truth": {
|
|
197
|
+
"cwe": cwe,
|
|
198
|
+
"cwe_raw": kind,
|
|
199
|
+
"cve": real_world.get("cve"),
|
|
200
|
+
"vulnerable_files": vuln_files,
|
|
201
|
+
"vulnerable_methods": [],
|
|
202
|
+
"vulnerable_regions": regions,
|
|
203
|
+
"fp_trap": fp_trap,
|
|
204
|
+
"notes": case.get("description", ""),
|
|
205
|
+
},
|
|
206
|
+
"weights": {},
|
|
207
|
+
"meta": meta,
|
|
208
|
+
}
|
|
209
|
+
tasks.append(task)
|
|
210
|
+
return tasks
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def main(argv: list[str] | None = None) -> int:
|
|
214
|
+
parser = argparse.ArgumentParser(description="Build SASTbench task records from case definitions")
|
|
215
|
+
parser.add_argument("--root", required=True, help="Path to sast-bench repo root (contains cases/ and taxonomy/)")
|
|
216
|
+
parser.add_argument("--out", required=True, help="Output JSONL path (tasks/sastbench.jsonl)")
|
|
217
|
+
args = parser.parse_args(argv)
|
|
218
|
+
|
|
219
|
+
tasks = build_tasks(args.root)
|
|
220
|
+
os.makedirs(os.path.dirname(os.path.abspath(args.out)), exist_ok=True)
|
|
221
|
+
with open(args.out, "w", encoding="utf-8") as f:
|
|
222
|
+
for t in tasks:
|
|
223
|
+
f.write(json.dumps(t, ensure_ascii=False, sort_keys=True) + "\n")
|
|
224
|
+
|
|
225
|
+
# Idempotency note: deterministic output (sorted keys, rglob case order).
|
|
226
|
+
print(f"Wrote {len(tasks)} SASTbench task records to {args.out}")
|
|
227
|
+
return 0
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
if __name__ == "__main__":
|
|
231
|
+
raise SystemExit(main())
|
sast_eval/cli.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Unified CLI for the sast-eval harness.
|
|
2
|
+
|
|
3
|
+
Subcommands map to the existing module mains (each accepts ``argv: list[str]``):
|
|
4
|
+
|
|
5
|
+
sast-eval build build all 5 benchmarks' tasks (tasks/*.jsonl)
|
|
6
|
+
sast-eval fetch fetch source for bountytasks/CWE-Bench/CyberGym/SASTbench
|
|
7
|
+
sast-eval package build per-task .tar.gz codebases for SAST analysis
|
|
8
|
+
sast-eval match match SARIF results against ground truth
|
|
9
|
+
sast-eval exploit run exploit-validation oracles on matched results
|
|
10
|
+
sast-eval score render the scorecard from matched + exploit results
|
|
11
|
+
sast-eval all build + fetch + package + match + exploit + score
|
|
12
|
+
|
|
13
|
+
Run ``sast-eval <subcommand> --help`` for per-subcommand options.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
# Default layout (matches the repo's Makefile defaults).
|
|
22
|
+
CORPUS = "corpus"
|
|
23
|
+
TASKS = "tasks"
|
|
24
|
+
IMPORTED = "imported"
|
|
25
|
+
RESULTS = "results"
|
|
26
|
+
REPORTS = "reports"
|
|
27
|
+
CODEBASES = "codebases"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _run(module: str, func: str, argv: list[str]) -> int:
|
|
31
|
+
"""Import module.func and call it with argv."""
|
|
32
|
+
import importlib
|
|
33
|
+
|
|
34
|
+
mod = importlib.import_module(module)
|
|
35
|
+
fn = getattr(mod, func)
|
|
36
|
+
return fn(argv)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _add_corpus_args(p: argparse.ArgumentParser) -> None:
|
|
40
|
+
"""Add --corpus/--tasks/--imported/--results/--reports/--codebases overrides."""
|
|
41
|
+
p.add_argument("--corpus", default=CORPUS, help="Corpus repos root (default: corpus)")
|
|
42
|
+
p.add_argument("--tasks", default=TASKS, help="Tasks dir (default: tasks)")
|
|
43
|
+
p.add_argument("--imported", default=IMPORTED, help="Imported results dir (default: imported)")
|
|
44
|
+
p.add_argument("--results", default=RESULTS, help="Results root (default: results)")
|
|
45
|
+
p.add_argument("--reports", default=REPORTS, help="Reports dir (default: reports)")
|
|
46
|
+
p.add_argument("--codebases", default=CODEBASES, help="Codebases dir (default: codebases)")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def cmd_build(args: argparse.Namespace) -> int:
|
|
50
|
+
rc = 0
|
|
51
|
+
Path(args.tasks).mkdir(parents=True, exist_ok=True)
|
|
52
|
+
Path(args.imported).mkdir(parents=True, exist_ok=True)
|
|
53
|
+
rc |= _run("sast_eval.adapters.owasp_adapter", "main",
|
|
54
|
+
["--root", str(Path(args.corpus) / "BenchmarkJava"), "--out", f"{args.tasks}/owasp.jsonl"])
|
|
55
|
+
rc |= _run("sast_eval.importers.bountytasks_importer", "main",
|
|
56
|
+
["--metadata-root", str(Path(args.corpus) / "bountytasks"),
|
|
57
|
+
"--tasks-out", f"{args.tasks}/bountytasks.jsonl",
|
|
58
|
+
"--imported-out", f"{args.imported}/bountytasks.jsonl"])
|
|
59
|
+
rc |= _run("sast_eval.importers.cwebench_importer", "main",
|
|
60
|
+
["--root", str(Path(args.corpus) / "cwe-bench-java"),
|
|
61
|
+
"--tasks-out", f"{args.tasks}/cwebench.jsonl",
|
|
62
|
+
"--imported-out", f"{args.imported}/cwebench.jsonl"])
|
|
63
|
+
rc |= _run("sast_eval.importers.cybergym_importer", "main",
|
|
64
|
+
["--root", str(Path(args.corpus) / "cybergym"),
|
|
65
|
+
"--tasks-out", f"{args.tasks}/cybergym.jsonl",
|
|
66
|
+
"--imported-out", f"{args.imported}/cybergym.jsonl"])
|
|
67
|
+
rc |= _run("sast_eval.adapters.sastbench_adapter", "main",
|
|
68
|
+
["--root", str(Path(args.corpus) / "sast-bench"), "--out", f"{args.tasks}/sastbench.jsonl"])
|
|
69
|
+
return rc
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def cmd_fetch(args: argparse.Namespace) -> int:
|
|
73
|
+
argv = [
|
|
74
|
+
"--bountytasks", str(Path(args.corpus) / "bountytasks"),
|
|
75
|
+
"--cwebench", str(Path(args.corpus) / "cwe-bench-java"),
|
|
76
|
+
"--cybergym", str(Path(args.corpus) / "cybergym"),
|
|
77
|
+
"--sastbench", str(Path(args.corpus) / "sast-bench"),
|
|
78
|
+
]
|
|
79
|
+
if args.cybergym_limit is not None:
|
|
80
|
+
argv += ["--cybergym-limit", str(args.cybergym_limit)]
|
|
81
|
+
return _run("sast_eval.tools.fetch_sources", "main", argv)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def cmd_package(args: argparse.Namespace) -> int:
|
|
85
|
+
Path(args.codebases).mkdir(parents=True, exist_ok=True)
|
|
86
|
+
return _run("sast_eval.tools.package_codebases", "main",
|
|
87
|
+
["--tasks", args.tasks, "--out", args.codebases])
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def cmd_match(args: argparse.Namespace) -> int:
|
|
91
|
+
out = f"{args.results}/matched/{args.tool}"
|
|
92
|
+
Path(out).mkdir(parents=True, exist_ok=True)
|
|
93
|
+
argv = ["--tasks", args.tasks, "--results", f"{args.results}/raw/{args.tool}", "--out", out, "--tool", args.tool]
|
|
94
|
+
rules_path = Path(f"tools/{args.tool}/rules.json")
|
|
95
|
+
if rules_path.exists():
|
|
96
|
+
argv += ["--rules", str(rules_path)]
|
|
97
|
+
return _run("sast_eval.matching.matcher", "main", argv)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def cmd_exploit(args: argparse.Namespace) -> int:
|
|
101
|
+
out = f"{args.results}/exploits/{args.tool}"
|
|
102
|
+
Path(out).mkdir(parents=True, exist_ok=True)
|
|
103
|
+
return _run("sast_eval.exploit.oracle", "main",
|
|
104
|
+
["--matched", f"{args.results}/matched/{args.tool}",
|
|
105
|
+
"--tasks", args.tasks, "--out", out, "--codebases", args.codebases])
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def cmd_score(args: argparse.Namespace) -> int:
|
|
109
|
+
Path(args.reports).mkdir(parents=True, exist_ok=True)
|
|
110
|
+
out = f"{args.reports}/scorecard.md"
|
|
111
|
+
argv = ["--matched", f"{args.results}/matched/{args.tool}",
|
|
112
|
+
"--imported", args.imported, "--tasks", args.tasks, "--out", out]
|
|
113
|
+
exploits_dir = Path(f"{args.results}/exploits")
|
|
114
|
+
if exploits_dir.is_dir():
|
|
115
|
+
argv += ["--exploits", str(exploits_dir)]
|
|
116
|
+
return _run("sast_eval.scoring.metrics", "main", argv)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def cmd_all(args: argparse.Namespace) -> int:
|
|
120
|
+
rc = cmd_build(args)
|
|
121
|
+
rc |= cmd_fetch(args)
|
|
122
|
+
rc |= cmd_package(args)
|
|
123
|
+
Path(f"{args.results}/raw/{args.tool}").mkdir(parents=True, exist_ok=True)
|
|
124
|
+
rc |= cmd_match(args)
|
|
125
|
+
rc |= cmd_exploit(args)
|
|
126
|
+
rc |= cmd_score(args)
|
|
127
|
+
return rc
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
131
|
+
parser = argparse.ArgumentParser(
|
|
132
|
+
prog="sast-eval",
|
|
133
|
+
description="Unified SAST evaluation harness across 5 vulnerability benchmarks.",
|
|
134
|
+
)
|
|
135
|
+
sub = parser.add_subparsers(dest="cmd", required=True)
|
|
136
|
+
|
|
137
|
+
# build
|
|
138
|
+
p = sub.add_parser("build", help="Build all 5 benchmarks' tasks (tasks/*.jsonl)")
|
|
139
|
+
_add_corpus_args(p)
|
|
140
|
+
p.set_defaults(func=cmd_build)
|
|
141
|
+
|
|
142
|
+
# fetch
|
|
143
|
+
p = sub.add_parser("fetch", help="Fetch source for bountytasks/CWE-Bench/CyberGym/SASTbench")
|
|
144
|
+
_add_corpus_args(p)
|
|
145
|
+
p.add_argument("--cybergym-limit", type=int, default=None,
|
|
146
|
+
help="Cap CyberGym tasks fetched from HuggingFace (empty = all)")
|
|
147
|
+
p.set_defaults(func=cmd_fetch)
|
|
148
|
+
|
|
149
|
+
# package
|
|
150
|
+
p = sub.add_parser("package", help="Build per-task .tar.gz codebases for SAST analysis")
|
|
151
|
+
_add_corpus_args(p)
|
|
152
|
+
p.set_defaults(func=cmd_package)
|
|
153
|
+
|
|
154
|
+
# match
|
|
155
|
+
p = sub.add_parser("match", help="Match SARIF results against ground truth")
|
|
156
|
+
_add_corpus_args(p)
|
|
157
|
+
p.add_argument("--tool", required=True, help="Tool name (results/raw/<tool>/)")
|
|
158
|
+
p.set_defaults(func=cmd_match)
|
|
159
|
+
|
|
160
|
+
# exploit
|
|
161
|
+
p = sub.add_parser("exploit", help="Run exploit-validation oracles on matched results")
|
|
162
|
+
_add_corpus_args(p)
|
|
163
|
+
p.add_argument("--tool", required=True, help="Tool name (results/matched/<tool>/)")
|
|
164
|
+
p.set_defaults(func=cmd_exploit)
|
|
165
|
+
|
|
166
|
+
# score
|
|
167
|
+
p = sub.add_parser("score", help="Render the scorecard from matched + exploit results")
|
|
168
|
+
_add_corpus_args(p)
|
|
169
|
+
p.add_argument("--tool", required=True, help="Tool name (results/matched/<tool>/)")
|
|
170
|
+
p.set_defaults(func=cmd_score)
|
|
171
|
+
|
|
172
|
+
# all
|
|
173
|
+
p = sub.add_parser("all", help="build + fetch + package + match + exploit + score")
|
|
174
|
+
_add_corpus_args(p)
|
|
175
|
+
p.add_argument("--tool", required=True, help="Tool name")
|
|
176
|
+
p.add_argument("--cybergym-limit", type=int, default=None,
|
|
177
|
+
help="Cap CyberGym tasks fetched from HuggingFace (empty = all)")
|
|
178
|
+
p.set_defaults(func=cmd_all)
|
|
179
|
+
|
|
180
|
+
return parser
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def main(argv: list[str] | None = None) -> int:
|
|
184
|
+
parser = build_parser()
|
|
185
|
+
args = parser.parse_args(argv)
|
|
186
|
+
return args.func(args)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
if __name__ == "__main__":
|
|
190
|
+
sys.exit(main())
|