sast-eval 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sast_eval/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """sast-eval — unified SAST evaluation harness across 5 vulnerability benchmarks.
2
+
3
+ Public entry point: ``sast_eval.cli:main`` (the ``sast-eval`` console script).
4
+ """
5
+ __version__ = "0.1.0"
File without changes
@@ -0,0 +1,66 @@
1
+ """Shared helpers for adapters and importers.
2
+
3
+ CWE normalization (§2.1) and bounty-value parsing (reusing the rules of
4
+ corpus/bountytasks/calculate_bounties.py).
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import re
9
+ from typing import Optional
10
+
11
+ # Matches the first integer in a CWE-ish string, with or without the "CWE-" prefix.
12
+ _CWE_INT = re.compile(r"(\d+)")
13
+
14
+
15
+ def normalize_cwe(raw: Optional[str]) -> str:
16
+ """Normalize a raw CWE string to canonical ``CWE-<int>`` form.
17
+
18
+ Per §2.1: extract the integer -> ``CWE-<int>`` (zero-padded to 3 for
19
+ display). Empty/unparseable -> ``CWE-UNKNOWN``.
20
+
21
+ Examples (from §2.1):
22
+ "CWE-22: Path Traversal" -> "CWE-022"
23
+ "400: Denial of Service" -> "CWE-400"
24
+ "CWE-29: Path Traversal: '\\..\\filename'" -> "CWE-029"
25
+ "" -> "CWE-UNKNOWN"
26
+ "22" -> "CWE-022"
27
+ None -> "CWE-UNKNOWN"
28
+ """
29
+ if raw is None:
30
+ return "CWE-UNKNOWN"
31
+ s = str(raw).strip()
32
+ if not s:
33
+ return "CWE-UNKNOWN"
34
+ m = _CWE_INT.search(s)
35
+ if not m:
36
+ return "CWE-UNKNOWN"
37
+ return f"CWE-{int(m.group(1)):03d}"
38
+
39
+
40
+ def parse_bounty_value(value) -> float:
41
+ """Parse a bounty value from various formats.
42
+
43
+ Reuses the rules of ``corpus/bountytasks/calculate_bounties.py``:
44
+ - None / "" -> 0.0
45
+ - already a number -> float
46
+ - leading "$" stripped
47
+ - "X to Y" range -> average of the two ends
48
+ - otherwise best-effort float(), 0.0 on failure
49
+ """
50
+ if value is None or value == "":
51
+ return 0.0
52
+ if isinstance(value, (int, float)):
53
+ return float(value)
54
+ if isinstance(value, str) and value.startswith("$"):
55
+ value = value[1:]
56
+ if isinstance(value, str) and " to " in value:
57
+ parts = value.split(" to ")
58
+ if len(parts) == 2:
59
+ try:
60
+ return (float(parts[0]) + float(parts[1])) / 2
61
+ except ValueError:
62
+ pass
63
+ try:
64
+ return float(value)
65
+ except (ValueError, TypeError):
66
+ return 0.0
@@ -0,0 +1,129 @@
1
+ """OWASP BenchmarkJava adapter — the only end-to-end adapter in this repo (§3.2).
2
+
3
+ Parses ``expectedresults-1.2.csv`` and emits one unified task record (§2) per row
4
+ to ``tasks/owasp.jsonl``.
5
+
6
+ Per §3.2 / Step 3b:
7
+ - One task per CSV row. Columns: test name, category, real vulnerability, cwe.
8
+ - ``source_root`` = the benchmark repo root (tools scan the whole repo; matching
9
+ is per test-case file).
10
+ - ``vulnerable_files`` = ``src/main/java/org/owasp/benchmark/testcode/<TestName>.java``
11
+ (directory is ``testcode``, not ``testcases``).
12
+ - ``real=false`` -> ``fp_trap: true`` (~1,000 of 2,740 rows are FP traps).
13
+ - Category code -> CWE via the fixed OWASP mapping; the CSV's own ``cwe``
14
+ column is preferred when present (it is authoritative per-row).
15
+ - No weights (all tasks equal).
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import argparse
20
+ import csv
21
+ import json
22
+ import os
23
+ from pathlib import Path
24
+
25
+ from sast_eval.adapters.common import normalize_cwe
26
+
27
+ # Fixed OWASP category -> CWE mapping (§3.2). Used as a fallback when the CSV's
28
+ # per-row ``cwe`` column is absent/empty. Values are zero-padded to 3 for display.
29
+ OWASP_CATEGORY_TO_CWE = {
30
+ "sqli": "CWE-089",
31
+ "xss": "CWE-079",
32
+ "cmdi": "CWE-078",
33
+ "pathtraver": "CWE-022",
34
+ "ldapi": "CWE-090",
35
+ "hash": "CWE-328",
36
+ "weakrand": "CWE-330",
37
+ "trustbound": "CWE-501",
38
+ "securecookie": "CWE-614",
39
+ "crypto": "CWE-327",
40
+ "xpathi": "CWE-643",
41
+ }
42
+
43
+ TESTCODE_REL = "src/main/java/org/owasp/benchmark/testcode"
44
+
45
+
46
+ def _category_to_cwe(category: str) -> str:
47
+ return OWASP_CATEGORY_TO_CWE.get(category, "CWE-UNKNOWN")
48
+
49
+
50
+ def build_tasks(root: str) -> list[dict]:
51
+ root_path = Path(root).resolve()
52
+ csv_path = root_path / "expectedresults-1.2.csv"
53
+ if not csv_path.exists():
54
+ raise FileNotFoundError(f"expectedresults-1.2.csv not found at {csv_path}")
55
+
56
+ tasks: list[dict] = []
57
+ with open(csv_path, newline="", encoding="utf-8") as f:
58
+ reader = csv.reader(f)
59
+ header = next(reader, None) # the '#' comment line
60
+ if not header or not header[0].lstrip().startswith("#"):
61
+ # Defensive: if the comment line is missing, the first row is data.
62
+ # Re-open and treat all rows as data. (The known file always has it.)
63
+ f.seek(0)
64
+ reader = csv.reader(f)
65
+
66
+ for row in reader:
67
+ if not row or all(not c.strip() for c in row):
68
+ continue
69
+ # Columns: # test name, category, real vulnerability, cwe
70
+ test_name = row[0].strip()
71
+ category = row[1].strip() if len(row) > 1 else ""
72
+ real = row[2].strip().lower() if len(row) > 2 else "true"
73
+ cwe_raw = row[3].strip() if len(row) > 3 else ""
74
+
75
+ # Prefer the CSV's per-row cwe column (authoritative); fall back to
76
+ # the fixed category mapping.
77
+ cwe = normalize_cwe(cwe_raw) if cwe_raw else _category_to_cwe(category)
78
+
79
+ fp_trap = (real == "false")
80
+ vuln_file = f"{TESTCODE_REL}/{test_name}.java"
81
+
82
+ task = {
83
+ "task_id": f"owasp/{test_name}",
84
+ "benchmark": "owasp",
85
+ "language": "java",
86
+ "source_root": str(root_path),
87
+ "vcs": {
88
+ "type": "git",
89
+ "commit": "", # filled by PROVENANCE from corpus hash
90
+ "checked_out": True,
91
+ },
92
+ "ground_truth": {
93
+ "cwe": cwe,
94
+ "cwe_raw": cwe_raw or category,
95
+ "cve": None,
96
+ "vulnerable_files": [vuln_file],
97
+ "vulnerable_methods": [],
98
+ "fp_trap": fp_trap,
99
+ "notes": "OWASP BenchmarkJava v1.2; category=%s" % category,
100
+ },
101
+ "weights": {},
102
+ "meta": {
103
+ "category": category,
104
+ "real_vulnerability": not fp_trap,
105
+ },
106
+ }
107
+ tasks.append(task)
108
+ return tasks
109
+
110
+
111
+ def main(argv: list[str] | None = None) -> int:
112
+ parser = argparse.ArgumentParser(description="Build OWASP BenchmarkJava task records")
113
+ parser.add_argument("--root", required=True, help="Path to BenchmarkJava repo root")
114
+ parser.add_argument("--out", required=True, help="Output JSONL path (tasks/owasp.jsonl)")
115
+ args = parser.parse_args(argv)
116
+
117
+ tasks = build_tasks(args.root)
118
+ os.makedirs(os.path.dirname(os.path.abspath(args.out)), exist_ok=True)
119
+ with open(args.out, "w", encoding="utf-8") as f:
120
+ for t in tasks:
121
+ f.write(json.dumps(t, ensure_ascii=False, sort_keys=True) + "\n")
122
+
123
+ # Idempotency note: deterministic output (sorted keys, CSV order preserved).
124
+ print(f"Wrote {len(tasks)} OWASP task records to {args.out}")
125
+ return 0
126
+
127
+
128
+ if __name__ == "__main__":
129
+ raise SystemExit(main())
@@ -0,0 +1,231 @@
1
+ """SASTbench adapter — end-to-end (this repo owns matching/scoring).
2
+
3
+ SASTbench evaluates SAST tools on agentic codebases. Its cases are
4
+ self-contained JSON files (``cases/<track>/<caseType>/<id>/case.json``) with
5
+ region-level ground truth: each case annotates one or more regions in the
6
+ source, labelled ``vulnerable`` (scanners must detect) or ``capability_safe``
7
+ (scanners must NOT flag — guarded dangerous code is an FP trap).
8
+
9
+ Per the SASTbench case schema (``schema/case.schema.json``):
10
+ - ``canonicalKind`` is one of 6 SASTbench kinds (command_injection,
11
+ path_traversal, ssrf, auth_bypass, authz_bypass, sql_injection). We map each
12
+ to its primary CWE via ``taxonomy/canonical_kinds.json`` (``cweMappings[0]``).
13
+ - ``regions`` carry ``path``, ``startLine``/``endLine``, ``label``,
14
+ ``acceptedKinds`` (for vulnerable regions) and ``requiredGuards``/``capability``
15
+ (for capability_safe regions).
16
+ - ``expectedOutcome.mustDetectRegionIds`` / ``mustNotFlagRegionIds`` state the
17
+ ground truth explicitly.
18
+ - ``files.root`` is the scannable project root (``project/`` for core,
19
+ ``../../../../.repos/<snapshot>/`` for full track).
20
+
21
+ Mapping to the unified task schema (§2):
22
+ - ``task_id`` = ``sastbench/<case-id>`` (e.g. ``sastbench/SB-PY-SV-001``).
23
+ - ``source_root`` = absolute path to ``<case_dir>/<files.root>``.
24
+ - ``ground_truth.cwe`` = primary CWE for ``canonicalKind``.
25
+ - ``ground_truth.vulnerable_files`` = paths of ``vulnerable`` regions.
26
+ - ``ground_truth.vulnerable_regions`` = full region list (NEW field; the
27
+ matcher uses this for region-level overlap matching — see
28
+ ``matching/matcher.py``).
29
+ - ``ground_truth.fp_trap`` = ``True`` for ``capability_safe`` cases (no
30
+ vulnerable regions; any finding is a FP). For ``mixed_intent`` cases this is
31
+ ``False`` (they have vulnerable regions), but their ``capability_safe``
32
+ regions are still FP traps at the region level — handled by the matcher.
33
+ - ``meta`` carries ``track``, ``case_type``, ``canonical_kind``, ``agentic``,
34
+ ``profile``, and (for real-world cases) ``repo``, ``cve``, ``ghsa``,
35
+ disclosure dates for knowledge-cutoff gating.
36
+ - ``vcs.checked_out`` = ``False`` + ``meta.source_missing`` = ``True`` when
37
+ ``source_root`` doesn't exist on disk (Full Track snapshots live under
38
+ ``.repos/`` and must be fetched via ``scripts/setup_repos.py`` — see
39
+ ``tools/fetch_sources.py``).
40
+ """
41
+ from __future__ import annotations
42
+
43
+ import argparse
44
+ import json
45
+ import os
46
+ from pathlib import Path
47
+
48
+ from sast_eval.adapters.common import normalize_cwe
49
+
50
+ # SASTbench canonical kind -> primary CWE (first entry in
51
+ # taxonomy/canonical_kinds.json ``cweMappings``). Used for the unified
52
+ # ``ground_truth.cwe`` field. Region-level ``acceptedKinds`` are matched
53
+ # against the finding's CWE via the same map in the matcher.
54
+ KIND_TO_PRIMARY_CWE: dict[str, str] = {
55
+ "command_injection": "CWE-078",
56
+ "path_traversal": "CWE-022",
57
+ "ssrf": "CWE-918",
58
+ "auth_bypass": "CWE-287",
59
+ "authz_bypass": "CWE-862",
60
+ "sql_injection": "CWE-089",
61
+ }
62
+
63
+ # Reverse: CWE -> SASTbench kind, for matching a finding's CWE against a
64
+ # region's acceptedKinds. A finding counts as matching a region if its CWE
65
+ # maps to any of the region's acceptedKinds. Built from the full
66
+ # cweMappings in taxonomy/canonical_kinds.json at load time (see below).
67
+ _CWE_TO_KINDS: dict[str, list[str]] = {}
68
+
69
+
70
+ def _load_kind_cwe_map(taxonomy_dir: Path) -> None:
71
+ """Populate ``_CWE_TO_KINDS`` from taxonomy/canonical_kinds.json."""
72
+ global _CWE_TO_KINDS
73
+ if _CWE_TO_KINDS:
74
+ return
75
+ kinds_file = taxonomy_dir / "canonical_kinds.json"
76
+ if not kinds_file.is_file():
77
+ return
78
+ data = json.loads(kinds_file.read_text(encoding="utf-8"))
79
+ for kind in data.get("canonicalKinds", []):
80
+ kid = kind.get("id", "")
81
+ for cwe in kind.get("cweMappings", []):
82
+ _CWE_TO_KINDS.setdefault(normalize_cwe(cwe), []).append(kid)
83
+
84
+
85
+ def cwe_to_kinds(cwe: str) -> list[str]:
86
+ """Return the SASTbench kinds that a CWE maps to (for region matching)."""
87
+ return _CWE_TO_KINDS.get(cwe, [])
88
+
89
+
90
+ def _language_map(raw: str) -> str:
91
+ """Map SASTbench language labels to our language field."""
92
+ m = {
93
+ "python": "python",
94
+ "typescript": "typescript",
95
+ "rust": "rust",
96
+ "swift": "swift",
97
+ "go": "go",
98
+ "java": "java",
99
+ "clojure": "clojure",
100
+ }
101
+ return m.get((raw or "").strip().lower(), (raw or "").strip().lower() or "unknown")
102
+
103
+
104
+ def _profile_for(case: dict) -> str:
105
+ """Return the SASTbench profile: 'agentic', 'generic', or 'all'."""
106
+ if case.get("caseType") == "real_world_generic":
107
+ return "generic"
108
+ return "agentic" if case.get("agentic", True) else "generic"
109
+
110
+
111
+ def _build_region(r: dict) -> dict:
112
+ """Normalize one SASTbench region into our ground-truth region record."""
113
+ return {
114
+ "id": r.get("id", ""),
115
+ "path": r.get("path", ""),
116
+ "start": r.get("startLine", 0),
117
+ "end": r.get("endLine", 0),
118
+ "label": r.get("label", ""), # "vulnerable" | "capability_safe"
119
+ "accepted_kinds": r.get("acceptedKinds", []),
120
+ "capability": r.get("capability", ""),
121
+ "required_guards": r.get("requiredGuards", []),
122
+ }
123
+
124
+
125
+ def build_tasks(root: str) -> list[dict]:
126
+ """Build unified task records from SASTbench case definitions.
127
+
128
+ ``root`` is the sast-bench repo root (containing ``cases/`` and
129
+ ``taxonomy/``).
130
+ """
131
+ root_path = Path(root).resolve()
132
+ cases_dir = root_path / "cases"
133
+ if not cases_dir.is_dir():
134
+ raise FileNotFoundError(f"cases/ not found at {cases_dir}")
135
+
136
+ _load_kind_cwe_map(root_path / "taxonomy")
137
+
138
+ tasks: list[dict] = []
139
+ for case_json in sorted(cases_dir.rglob("case.json")):
140
+ case = json.loads(case_json.read_text(encoding="utf-8"))
141
+ case_dir = case_json.parent
142
+ case_id = case.get("id", "")
143
+ if not case_id:
144
+ continue
145
+
146
+ kind = case.get("canonicalKind", "")
147
+ cwe = KIND_TO_PRIMARY_CWE.get(kind, "CWE-UNKNOWN")
148
+
149
+ regions = [_build_region(r) for r in case.get("regions", [])]
150
+ vuln_regions = [r for r in regions if r["label"] == "vulnerable"]
151
+ cap_safe_regions = [r for r in regions if r["label"] == "capability_safe"]
152
+ vuln_files = sorted({r["path"] for r in vuln_regions if r["path"]})
153
+
154
+ # capability_safe cases have no vulnerable regions — any finding is a FP.
155
+ # mixed_intent cases have both; they are NOT fp_trap at the task level
156
+ # (they have real vulns to find), but their capability_safe regions are
157
+ # FP traps at the region level (handled by the matcher).
158
+ fp_trap = (case.get("caseType") == "capability_safe")
159
+
160
+ files_root = case.get("files", {}).get("root", "")
161
+ source_root = (case_dir / files_root).resolve() if files_root else case_dir
162
+ source_missing = not source_root.is_dir()
163
+
164
+ real_world = case.get("realWorld") or {}
165
+ disclosure = real_world.get("disclosure") or {}
166
+
167
+ meta = {
168
+ "track": case.get("track", ""),
169
+ "case_type": case.get("caseType", ""),
170
+ "canonical_kind": kind,
171
+ "agentic": case.get("agentic", True),
172
+ "profile": _profile_for(case),
173
+ "title": case.get("title", ""),
174
+ "source_missing": source_missing,
175
+ }
176
+ if real_world:
177
+ meta["repo"] = real_world.get("repo", "")
178
+ meta["cve"] = real_world.get("cve")
179
+ meta["ghsa"] = real_world.get("ghsa")
180
+ meta["vulnerable_commit"] = real_world.get("vulnerableCommit", "")
181
+ meta["fix_commit"] = real_world.get("fixCommit", "")
182
+ meta["disclosure_ghsa_published"] = disclosure.get("ghsaPublished")
183
+ meta["disclosure_fix_commit_date"] = disclosure.get("fixCommitDate")
184
+ meta["disclosure_cve_published"] = disclosure.get("cvePublished")
185
+
186
+ task = {
187
+ "task_id": f"sastbench/{case_id}",
188
+ "benchmark": "sastbench",
189
+ "language": _language_map(case.get("language", "")),
190
+ "source_root": str(source_root),
191
+ "vcs": {
192
+ "type": "git",
193
+ "commit": real_world.get("vulnerableCommit", ""),
194
+ "checked_out": not source_missing,
195
+ },
196
+ "ground_truth": {
197
+ "cwe": cwe,
198
+ "cwe_raw": kind,
199
+ "cve": real_world.get("cve"),
200
+ "vulnerable_files": vuln_files,
201
+ "vulnerable_methods": [],
202
+ "vulnerable_regions": regions,
203
+ "fp_trap": fp_trap,
204
+ "notes": case.get("description", ""),
205
+ },
206
+ "weights": {},
207
+ "meta": meta,
208
+ }
209
+ tasks.append(task)
210
+ return tasks
211
+
212
+
213
+ def main(argv: list[str] | None = None) -> int:
214
+ parser = argparse.ArgumentParser(description="Build SASTbench task records from case definitions")
215
+ parser.add_argument("--root", required=True, help="Path to sast-bench repo root (contains cases/ and taxonomy/)")
216
+ parser.add_argument("--out", required=True, help="Output JSONL path (tasks/sastbench.jsonl)")
217
+ args = parser.parse_args(argv)
218
+
219
+ tasks = build_tasks(args.root)
220
+ os.makedirs(os.path.dirname(os.path.abspath(args.out)), exist_ok=True)
221
+ with open(args.out, "w", encoding="utf-8") as f:
222
+ for t in tasks:
223
+ f.write(json.dumps(t, ensure_ascii=False, sort_keys=True) + "\n")
224
+
225
+ # Idempotency note: deterministic output (sorted keys, rglob case order).
226
+ print(f"Wrote {len(tasks)} SASTbench task records to {args.out}")
227
+ return 0
228
+
229
+
230
+ if __name__ == "__main__":
231
+ raise SystemExit(main())
sast_eval/cli.py ADDED
@@ -0,0 +1,190 @@
1
+ """Unified CLI for the sast-eval harness.
2
+
3
+ Subcommands map to the existing module mains (each accepts ``argv: list[str]``):
4
+
5
+ sast-eval build build all 5 benchmarks' tasks (tasks/*.jsonl)
6
+ sast-eval fetch fetch source for bountytasks/CWE-Bench/CyberGym/SASTbench
7
+ sast-eval package build per-task .tar.gz codebases for SAST analysis
8
+ sast-eval match match SARIF results against ground truth
9
+ sast-eval exploit run exploit-validation oracles on matched results
10
+ sast-eval score render the scorecard from matched + exploit results
11
+ sast-eval all build + fetch + package + match + exploit + score
12
+
13
+ Run ``sast-eval <subcommand> --help`` for per-subcommand options.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import argparse
18
+ import sys
19
+ from pathlib import Path
20
+
21
+ # Default layout (matches the repo's Makefile defaults).
22
+ CORPUS = "corpus"
23
+ TASKS = "tasks"
24
+ IMPORTED = "imported"
25
+ RESULTS = "results"
26
+ REPORTS = "reports"
27
+ CODEBASES = "codebases"
28
+
29
+
30
+ def _run(module: str, func: str, argv: list[str]) -> int:
31
+ """Import module.func and call it with argv."""
32
+ import importlib
33
+
34
+ mod = importlib.import_module(module)
35
+ fn = getattr(mod, func)
36
+ return fn(argv)
37
+
38
+
39
+ def _add_corpus_args(p: argparse.ArgumentParser) -> None:
40
+ """Add --corpus/--tasks/--imported/--results/--reports/--codebases overrides."""
41
+ p.add_argument("--corpus", default=CORPUS, help="Corpus repos root (default: corpus)")
42
+ p.add_argument("--tasks", default=TASKS, help="Tasks dir (default: tasks)")
43
+ p.add_argument("--imported", default=IMPORTED, help="Imported results dir (default: imported)")
44
+ p.add_argument("--results", default=RESULTS, help="Results root (default: results)")
45
+ p.add_argument("--reports", default=REPORTS, help="Reports dir (default: reports)")
46
+ p.add_argument("--codebases", default=CODEBASES, help="Codebases dir (default: codebases)")
47
+
48
+
49
+ def cmd_build(args: argparse.Namespace) -> int:
50
+ rc = 0
51
+ Path(args.tasks).mkdir(parents=True, exist_ok=True)
52
+ Path(args.imported).mkdir(parents=True, exist_ok=True)
53
+ rc |= _run("sast_eval.adapters.owasp_adapter", "main",
54
+ ["--root", str(Path(args.corpus) / "BenchmarkJava"), "--out", f"{args.tasks}/owasp.jsonl"])
55
+ rc |= _run("sast_eval.importers.bountytasks_importer", "main",
56
+ ["--metadata-root", str(Path(args.corpus) / "bountytasks"),
57
+ "--tasks-out", f"{args.tasks}/bountytasks.jsonl",
58
+ "--imported-out", f"{args.imported}/bountytasks.jsonl"])
59
+ rc |= _run("sast_eval.importers.cwebench_importer", "main",
60
+ ["--root", str(Path(args.corpus) / "cwe-bench-java"),
61
+ "--tasks-out", f"{args.tasks}/cwebench.jsonl",
62
+ "--imported-out", f"{args.imported}/cwebench.jsonl"])
63
+ rc |= _run("sast_eval.importers.cybergym_importer", "main",
64
+ ["--root", str(Path(args.corpus) / "cybergym"),
65
+ "--tasks-out", f"{args.tasks}/cybergym.jsonl",
66
+ "--imported-out", f"{args.imported}/cybergym.jsonl"])
67
+ rc |= _run("sast_eval.adapters.sastbench_adapter", "main",
68
+ ["--root", str(Path(args.corpus) / "sast-bench"), "--out", f"{args.tasks}/sastbench.jsonl"])
69
+ return rc
70
+
71
+
72
+ def cmd_fetch(args: argparse.Namespace) -> int:
73
+ argv = [
74
+ "--bountytasks", str(Path(args.corpus) / "bountytasks"),
75
+ "--cwebench", str(Path(args.corpus) / "cwe-bench-java"),
76
+ "--cybergym", str(Path(args.corpus) / "cybergym"),
77
+ "--sastbench", str(Path(args.corpus) / "sast-bench"),
78
+ ]
79
+ if args.cybergym_limit is not None:
80
+ argv += ["--cybergym-limit", str(args.cybergym_limit)]
81
+ return _run("sast_eval.tools.fetch_sources", "main", argv)
82
+
83
+
84
+ def cmd_package(args: argparse.Namespace) -> int:
85
+ Path(args.codebases).mkdir(parents=True, exist_ok=True)
86
+ return _run("sast_eval.tools.package_codebases", "main",
87
+ ["--tasks", args.tasks, "--out", args.codebases])
88
+
89
+
90
+ def cmd_match(args: argparse.Namespace) -> int:
91
+ out = f"{args.results}/matched/{args.tool}"
92
+ Path(out).mkdir(parents=True, exist_ok=True)
93
+ argv = ["--tasks", args.tasks, "--results", f"{args.results}/raw/{args.tool}", "--out", out, "--tool", args.tool]
94
+ rules_path = Path(f"tools/{args.tool}/rules.json")
95
+ if rules_path.exists():
96
+ argv += ["--rules", str(rules_path)]
97
+ return _run("sast_eval.matching.matcher", "main", argv)
98
+
99
+
100
+ def cmd_exploit(args: argparse.Namespace) -> int:
101
+ out = f"{args.results}/exploits/{args.tool}"
102
+ Path(out).mkdir(parents=True, exist_ok=True)
103
+ return _run("sast_eval.exploit.oracle", "main",
104
+ ["--matched", f"{args.results}/matched/{args.tool}",
105
+ "--tasks", args.tasks, "--out", out, "--codebases", args.codebases])
106
+
107
+
108
+ def cmd_score(args: argparse.Namespace) -> int:
109
+ Path(args.reports).mkdir(parents=True, exist_ok=True)
110
+ out = f"{args.reports}/scorecard.md"
111
+ argv = ["--matched", f"{args.results}/matched/{args.tool}",
112
+ "--imported", args.imported, "--tasks", args.tasks, "--out", out]
113
+ exploits_dir = Path(f"{args.results}/exploits")
114
+ if exploits_dir.is_dir():
115
+ argv += ["--exploits", str(exploits_dir)]
116
+ return _run("sast_eval.scoring.metrics", "main", argv)
117
+
118
+
119
+ def cmd_all(args: argparse.Namespace) -> int:
120
+ rc = cmd_build(args)
121
+ rc |= cmd_fetch(args)
122
+ rc |= cmd_package(args)
123
+ Path(f"{args.results}/raw/{args.tool}").mkdir(parents=True, exist_ok=True)
124
+ rc |= cmd_match(args)
125
+ rc |= cmd_exploit(args)
126
+ rc |= cmd_score(args)
127
+ return rc
128
+
129
+
130
+ def build_parser() -> argparse.ArgumentParser:
131
+ parser = argparse.ArgumentParser(
132
+ prog="sast-eval",
133
+ description="Unified SAST evaluation harness across 5 vulnerability benchmarks.",
134
+ )
135
+ sub = parser.add_subparsers(dest="cmd", required=True)
136
+
137
+ # build
138
+ p = sub.add_parser("build", help="Build all 5 benchmarks' tasks (tasks/*.jsonl)")
139
+ _add_corpus_args(p)
140
+ p.set_defaults(func=cmd_build)
141
+
142
+ # fetch
143
+ p = sub.add_parser("fetch", help="Fetch source for bountytasks/CWE-Bench/CyberGym/SASTbench")
144
+ _add_corpus_args(p)
145
+ p.add_argument("--cybergym-limit", type=int, default=None,
146
+ help="Cap CyberGym tasks fetched from HuggingFace (empty = all)")
147
+ p.set_defaults(func=cmd_fetch)
148
+
149
+ # package
150
+ p = sub.add_parser("package", help="Build per-task .tar.gz codebases for SAST analysis")
151
+ _add_corpus_args(p)
152
+ p.set_defaults(func=cmd_package)
153
+
154
+ # match
155
+ p = sub.add_parser("match", help="Match SARIF results against ground truth")
156
+ _add_corpus_args(p)
157
+ p.add_argument("--tool", required=True, help="Tool name (results/raw/<tool>/)")
158
+ p.set_defaults(func=cmd_match)
159
+
160
+ # exploit
161
+ p = sub.add_parser("exploit", help="Run exploit-validation oracles on matched results")
162
+ _add_corpus_args(p)
163
+ p.add_argument("--tool", required=True, help="Tool name (results/matched/<tool>/)")
164
+ p.set_defaults(func=cmd_exploit)
165
+
166
+ # score
167
+ p = sub.add_parser("score", help="Render the scorecard from matched + exploit results")
168
+ _add_corpus_args(p)
169
+ p.add_argument("--tool", required=True, help="Tool name (results/matched/<tool>/)")
170
+ p.set_defaults(func=cmd_score)
171
+
172
+ # all
173
+ p = sub.add_parser("all", help="build + fetch + package + match + exploit + score")
174
+ _add_corpus_args(p)
175
+ p.add_argument("--tool", required=True, help="Tool name")
176
+ p.add_argument("--cybergym-limit", type=int, default=None,
177
+ help="Cap CyberGym tasks fetched from HuggingFace (empty = all)")
178
+ p.set_defaults(func=cmd_all)
179
+
180
+ return parser
181
+
182
+
183
+ def main(argv: list[str] | None = None) -> int:
184
+ parser = build_parser()
185
+ args = parser.parse_args(argv)
186
+ return args.func(args)
187
+
188
+
189
+ if __name__ == "__main__":
190
+ sys.exit(main())
@@ -0,0 +1,5 @@
1
+ """Exploit-validation oracles (§10) — 4-tier oracle engine.
2
+
3
+ Importing this package registers the generic Tier 1 reachability oracle and
4
+ makes the bespoke oracle subpackage importable.
5
+ """