secure-code-agent 0.8.0__tar.gz → 0.10.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {secure_code_agent-0.8.0/src/secure_code_agent.egg-info → secure_code_agent-0.10.0}/PKG-INFO +1 -1
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0/src/secure_code_agent.egg-info}/PKG-INFO +1 -1
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/__init__.py +1 -1
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/baseline.py +31 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/cli.py +32 -4
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/config.py +87 -3
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/git_tools.py +31 -4
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/pillar.py +53 -2
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scoring.py +161 -9
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/standards.py +60 -3
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/triage.py +129 -1
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/LICENSE +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/README.md +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/pyproject.toml +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/setup.cfg +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/SOURCES.txt +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/dependency_links.txt +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/entry_points.txt +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/requires.txt +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/top_level.txt +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/data/semgrep-offline.yaml +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/findings.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/history.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/instructions.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/practice.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/remediation.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/renderers.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/ruleset.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/sarif.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanner_status.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/__init__.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/bandit_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/base.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/builtin_rules.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/checkov_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/floor.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/gitleaks_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/gosec_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/hadolint_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/njsscan_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/npm_audit_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/osv_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/pip_audit_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/rubocop_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/scorecard_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/semgrep_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/trivy_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/trufflehog_scanner.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/suppressions.py +0 -0
- {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/verify.py +0 -0
|
@@ -10,6 +10,7 @@ operator has acknowledged at a point in time. On the next run:
|
|
|
10
10
|
from __future__ import annotations
|
|
11
11
|
|
|
12
12
|
import datetime
|
|
13
|
+
import enum
|
|
13
14
|
import json
|
|
14
15
|
import shutil
|
|
15
16
|
import subprocess
|
|
@@ -32,6 +33,36 @@ class BaselineEntry:
|
|
|
32
33
|
notes: str = ""
|
|
33
34
|
|
|
34
35
|
|
|
36
|
+
class State(enum.Enum):
|
|
37
|
+
"""Why the baseline is the shape it is.
|
|
38
|
+
|
|
39
|
+
`load` returns an empty mapping for a baseline that is absent, one that
|
|
40
|
+
is unreadable, and one that is genuinely empty. Downstream those look
|
|
41
|
+
identical and every finding reads as new — which is correct for the
|
|
42
|
+
first case, a silent failure for the second, and worth saying out loud
|
|
43
|
+
in all three once `fail_on_new` gates by default.
|
|
44
|
+
|
|
45
|
+
On a first run the honest message is "there is nothing to compare
|
|
46
|
+
against yet", not "2 new findings since baseline", which claims a
|
|
47
|
+
baseline exists and that these appeared after it.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
ABSENT = "absent"
|
|
51
|
+
UNREADABLE = "unreadable"
|
|
52
|
+
PRESENT = "present"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def state(path: Path) -> State:
|
|
56
|
+
"""Distinguish a missing baseline from a broken one."""
|
|
57
|
+
if not path.exists():
|
|
58
|
+
return State.ABSENT
|
|
59
|
+
try:
|
|
60
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
61
|
+
except (OSError, json.JSONDecodeError):
|
|
62
|
+
return State.UNREADABLE
|
|
63
|
+
return State.PRESENT if isinstance(raw, dict) else State.UNREADABLE
|
|
64
|
+
|
|
65
|
+
|
|
35
66
|
def load(path: Path) -> dict[str, BaselineEntry]:
|
|
36
67
|
if not path.exists():
|
|
37
68
|
return {}
|
|
@@ -160,7 +160,11 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
160
160
|
"--target",
|
|
161
161
|
action="append",
|
|
162
162
|
default=[],
|
|
163
|
-
help=
|
|
163
|
+
help=(
|
|
164
|
+
"Agent name for --init-agent-standards (codex, claude-code, cursor, "
|
|
165
|
+
"copilot, windsurf, generic). NOT the repository to audit — pass that "
|
|
166
|
+
"as a positional path."
|
|
167
|
+
),
|
|
164
168
|
)
|
|
165
169
|
p.add_argument(
|
|
166
170
|
"--instructions-output-dir",
|
|
@@ -183,6 +187,22 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
183
187
|
if args.init_agent_standards:
|
|
184
188
|
return _do_init_standards(args)
|
|
185
189
|
|
|
190
|
+
if args.target:
|
|
191
|
+
# `--target` names an *agent* for --init-agent-standards, and it reads
|
|
192
|
+
# exactly like the flag for "the repository to audit". It was accepted
|
|
193
|
+
# and silently discarded on an audit run, so `--target /some/repo`
|
|
194
|
+
# audited the current directory instead and reported a clean result
|
|
195
|
+
# for a repository nobody had looked at. Found by using the tool: a
|
|
196
|
+
# 136k-line control was audited from inside itself and appeared to
|
|
197
|
+
# honour the flag, which is the kind of coincidence that keeps a bug.
|
|
198
|
+
sys.stderr.write(
|
|
199
|
+
"ERROR: --target names an agent for --init-agent-standards, not a "
|
|
200
|
+
"repository to audit.\n"
|
|
201
|
+
f" To audit a repository, pass it as a path: "
|
|
202
|
+
f"secure-code-agent {args.target[0]}\n"
|
|
203
|
+
)
|
|
204
|
+
return 2
|
|
205
|
+
|
|
186
206
|
try:
|
|
187
207
|
return _do_preflight(args) if args.preflight else _do_audit(args)
|
|
188
208
|
except ValueError as exc:
|
|
@@ -338,6 +358,7 @@ def _do_audit(args: argparse.Namespace) -> int:
|
|
|
338
358
|
# ----- baseline -----
|
|
339
359
|
baseline_path = _under_root(root, args.baseline or cfg.outputs["baseline_path"])
|
|
340
360
|
baseline = baseline_mod.load(baseline_path)
|
|
361
|
+
baseline_state = baseline_mod.state(baseline_path)
|
|
341
362
|
all_findings = baseline_mod.mark_new(all_findings, baseline)
|
|
342
363
|
|
|
343
364
|
# ----- scoring -----
|
|
@@ -377,8 +398,9 @@ def _do_audit(args: argparse.Namespace) -> int:
|
|
|
377
398
|
if cfg.loc_for_scoring:
|
|
378
399
|
loc = int(cfg.loc_for_scoring.get("value", 0))
|
|
379
400
|
test_loc = 0
|
|
401
|
+
docs_loc = 0
|
|
380
402
|
else:
|
|
381
|
-
loc, test_loc = loc_under(
|
|
403
|
+
loc, test_loc, docs_loc = loc_under(
|
|
382
404
|
target,
|
|
383
405
|
cfg.include_extensions,
|
|
384
406
|
cfg.exclude_patterns,
|
|
@@ -386,6 +408,7 @@ def _do_audit(args: argparse.Namespace) -> int:
|
|
|
386
408
|
# Same set the findings were filtered against. Numerator and
|
|
387
409
|
# denominator have to describe the same repository.
|
|
388
410
|
own_artifacts,
|
|
411
|
+
cfg.docs_patterns,
|
|
389
412
|
)
|
|
390
413
|
# Dependencies come off the code-condition score and onto their own axis.
|
|
391
414
|
# A CVE in a pinned dependency is fixed with a version bump; an injection
|
|
@@ -405,7 +428,7 @@ def _do_audit(args: argparse.Namespace) -> int:
|
|
|
405
428
|
score = score_findings(scored_findings, loc, measurable)
|
|
406
429
|
axes = (
|
|
407
430
|
summarize_axis("test tree", test_findings, test_loc),
|
|
408
|
-
summarize_axis("documentation", docs_findings),
|
|
431
|
+
summarize_axis("documentation", docs_findings, docs_loc or None),
|
|
409
432
|
summarize_axis("dependencies", dependency_findings),
|
|
410
433
|
)
|
|
411
434
|
# Naming an import on the command line asserts that it contributes coverage,
|
|
@@ -425,7 +448,12 @@ def _do_audit(args: argparse.Namespace) -> int:
|
|
|
425
448
|
]
|
|
426
449
|
coverage = evaluate_coverage(executions, required)
|
|
427
450
|
# Gates see the dependency advisories; the score does not.
|
|
428
|
-
gate
|
|
451
|
+
# The gate needs to know *why* the baseline is empty to describe a first
|
|
452
|
+
# run truthfully. Passed in the config dict rather than as a parameter so
|
|
453
|
+
# the gate signature stays the one every check shares.
|
|
454
|
+
gate = evaluate_gates(
|
|
455
|
+
gated, score, {**cfg.gates, "_baseline_state": baseline_state.value}, coverage
|
|
456
|
+
)
|
|
429
457
|
verdict = build_verdict(score, cfg.gates, coverage)
|
|
430
458
|
|
|
431
459
|
# ----- write outputs -----
|
|
@@ -8,7 +8,7 @@ from __future__ import annotations
|
|
|
8
8
|
|
|
9
9
|
import json
|
|
10
10
|
from dataclasses import dataclass, field
|
|
11
|
-
from pathlib import Path
|
|
11
|
+
from pathlib import Path, PurePosixPath
|
|
12
12
|
from typing import Any
|
|
13
13
|
|
|
14
14
|
DEFAULT_CONFIG_PATH = Path("secure-code-agent.json")
|
|
@@ -26,6 +26,53 @@ DEFAULT_EXCLUDES: tuple[str, ...] = (
|
|
|
26
26
|
".mypy_cache/",
|
|
27
27
|
"**/*.min.js",
|
|
28
28
|
"**/*.lock",
|
|
29
|
+
# Lockfiles that are not named `.lock`. `**/*.lock` catches
|
|
30
|
+
# `poetry.lock`, `Gemfile.lock`, `Cargo.lock` and `yarn.lock` and misses
|
|
31
|
+
# every lockfile the JavaScript ecosystem actually ships:
|
|
32
|
+
# `package-lock.json` alone was 9,699 of axios's 17,532 non-code lines
|
|
33
|
+
# and 5,845 of lodash's 6,222. A generated dependency manifest is not
|
|
34
|
+
# source, and counting it inflates the denominator that decides the
|
|
35
|
+
# grade.
|
|
36
|
+
"**/package-lock.json",
|
|
37
|
+
"**/npm-shrinkwrap.json",
|
|
38
|
+
"**/pnpm-lock.yaml",
|
|
39
|
+
"**/bun.lockb",
|
|
40
|
+
# --- stored analysis output -------------------------------------------
|
|
41
|
+
#
|
|
42
|
+
# This tool's own output is never its input, *wherever* it is stored.
|
|
43
|
+
#
|
|
44
|
+
# `cli._own_artifacts` already removes the paths the current run is about
|
|
45
|
+
# to write, which is what stopped an audit scoring the report it had just
|
|
46
|
+
# produced. It cannot see a *copy* kept somewhere else, and two
|
|
47
|
+
# independent reports of that landed on the same day:
|
|
48
|
+
#
|
|
49
|
+
# - this repository scanned `calibration/.corpus` — fourteen cloned
|
|
50
|
+
# third-party projects, 556,808 LOC and 550 findings, all about code
|
|
51
|
+
# that is not ours;
|
|
52
|
+
# - `maintainability-agent` scanned `tools/validation/reports/` —
|
|
53
|
+
# 957,219 LOC of stored audit output *about other repositories*,
|
|
54
|
+
# 4,929 findings, which diluted five genuine criticals to an A-.
|
|
55
|
+
#
|
|
56
|
+
# A stored report is the worst possible input: it quotes findings
|
|
57
|
+
# verbatim, including the code snippets and the redacted secrets that
|
|
58
|
+
# produced them, so it manufactures findings about findings and inflates
|
|
59
|
+
# the denominator at the same time.
|
|
60
|
+
#
|
|
61
|
+
# Filename patterns rather than directory names, because the directory is
|
|
62
|
+
# whatever the operator chose and the filenames are ours. Derived from
|
|
63
|
+
# `DEFAULT_OUTPUTS` below so the two cannot drift — see
|
|
64
|
+
# `_own_output_globs`.
|
|
65
|
+
#
|
|
66
|
+
# Deliberately NOT here: `vendor/`, `third_party/` and their kin. Vendored
|
|
67
|
+
# code is deployed code, and excluding it by default would hide real
|
|
68
|
+
# vulnerabilities in exactly the place nobody is reading. Stored analysis
|
|
69
|
+
# output is not code at all; that is the whole difference.
|
|
70
|
+
".secure-code/",
|
|
71
|
+
"**/.secure-code/",
|
|
72
|
+
# maintainability-agent's state and output directory. Same argument: its
|
|
73
|
+
# reports quote findings, and its history is append-only JSONL.
|
|
74
|
+
".maintainability/",
|
|
75
|
+
"**/.maintainability/",
|
|
29
76
|
)
|
|
30
77
|
|
|
31
78
|
#: Conventional test-tree locations across the languages the floor reads.
|
|
@@ -96,6 +143,24 @@ DEFAULT_OUTPUTS: dict[str, str] = {
|
|
|
96
143
|
"history_path": ".secure-code/history.jsonl",
|
|
97
144
|
}
|
|
98
145
|
|
|
146
|
+
|
|
147
|
+
def _own_output_globs() -> tuple[str, ...]:
|
|
148
|
+
"""Match this tool's default output filenames anywhere in a tree.
|
|
149
|
+
|
|
150
|
+
Kept as a function over `DEFAULT_OUTPUTS` rather than a hand-written list
|
|
151
|
+
so that adding an output cannot leave a file this tool writes readable by
|
|
152
|
+
the next run. `history_path` is already covered by the `.secure-code/`
|
|
153
|
+
directory entries; matching its basename anywhere would be wrong, since
|
|
154
|
+
`history.jsonl` is not a name this project owns.
|
|
155
|
+
"""
|
|
156
|
+
names = {
|
|
157
|
+
PurePosixPath(path).name for key, path in DEFAULT_OUTPUTS.items() if key != "history_path"
|
|
158
|
+
}
|
|
159
|
+
return tuple(sorted(f"**/{name}" for name in names))
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
DEFAULT_EXCLUDES = DEFAULT_EXCLUDES + _own_output_globs()
|
|
163
|
+
|
|
99
164
|
_SEVERITIES = {"critical", "high", "medium", "low", "informational"}
|
|
100
165
|
_CATEGORIES = {
|
|
101
166
|
"secrets",
|
|
@@ -169,7 +234,24 @@ class Config:
|
|
|
169
234
|
scanners: dict[str, ScannerConfig] = field(default_factory=dict)
|
|
170
235
|
severity_overrides: dict[str, str] = field(default_factory=dict)
|
|
171
236
|
category_overrides: dict[str, str] = field(default_factory=dict)
|
|
172
|
-
|
|
237
|
+
#: Default policy: **ratchet on regressions**, not on absolute state.
|
|
238
|
+
#:
|
|
239
|
+
#: `fail_on_new` is the one gate measured to work. Severity-based
|
|
240
|
+
#: defaults cannot: `fail_on_severity: ["critical"]` caught none of four
|
|
241
|
+
#: known-vulnerable control repositories, and `["critical","high"]`
|
|
242
|
+
#: failed five of ten well-maintained ones while still missing SQL
|
|
243
|
+
#: injection and `pickle.loads`, which Bandit rates *medium*. See D15.
|
|
244
|
+
#:
|
|
245
|
+
#: The ratchet reads no severity at all, so it inherits none of that. A
|
|
246
|
+
#: false positive is baselined once and never asked about again, which
|
|
247
|
+
#: is the property a threshold cannot have. Measured across an adoption
|
|
248
|
+
#: lifecycle: first run fails (everything is new), `--bump-baseline`
|
|
249
|
+
#: accepts existing debt, an introduced SQL injection fails, reverting
|
|
250
|
+
#: passes.
|
|
251
|
+
#:
|
|
252
|
+
#: An empty `{}` was the previous default and provided no floor at all —
|
|
253
|
+
#: an absent gate cannot trip, so every audit "passed".
|
|
254
|
+
gates: dict[str, Any] = field(default_factory=lambda: {"fail_on_new": True})
|
|
173
255
|
outputs: dict[str, str] = field(default_factory=lambda: dict(DEFAULT_OUTPUTS))
|
|
174
256
|
suppressions_file: str = ".scignore.yaml"
|
|
175
257
|
loc_for_scoring: dict[str, Any] | None = None
|
|
@@ -343,7 +425,9 @@ def _from_dict(raw: dict[str, Any]) -> Config:
|
|
|
343
425
|
raise ValueError(f"invalid severity override: {', '.join(invalid)}")
|
|
344
426
|
if invalid := sorted(set(cfg.category_overrides.values()) - _CATEGORIES):
|
|
345
427
|
raise ValueError(f"invalid category override: {', '.join(invalid)}")
|
|
346
|
-
|
|
428
|
+
# An operator who writes a `gates` block chooses their own policy
|
|
429
|
+
# entirely; the ratchet default applies only when they write none.
|
|
430
|
+
cfg.gates = _validate_gates(raw["gates"]) if "gates" in raw else dict(cfg.gates)
|
|
347
431
|
|
|
348
432
|
outputs = raw.get("outputs", {})
|
|
349
433
|
if not isinstance(outputs, dict):
|
|
@@ -58,9 +58,25 @@ def _matches(rel: str, name: str, pat: str) -> bool:
|
|
|
58
58
|
`router/context_test.go` and never `context_test.go`. Gin keeps its tests
|
|
59
59
|
beside the code they test, so three of its four "production" secrets were
|
|
60
60
|
test fixtures at the repository root, and that alone held it at F.
|
|
61
|
+
|
|
62
|
+
**The `**/` strip has to happen on the directory branch too.** It did not,
|
|
63
|
+
and so `**/__pycache__/` matched *nothing*: the branch searched for a
|
|
64
|
+
literal `/**/__pycache__/` inside the path, and no real path contains
|
|
65
|
+
`/**/`. A bare `__pycache__/` already matches at any depth, so the two
|
|
66
|
+
spellings differed by everything — one worked and the one this
|
|
67
|
+
repository's own config used was inert. `.pyc` files were scanned as
|
|
68
|
+
source the whole time, and CI caught it only because a compiled test
|
|
69
|
+
fixture tripped a secrets rule.
|
|
70
|
+
|
|
71
|
+
An inert exclude pattern is the worst kind of configuration defect: it
|
|
72
|
+
reads as intent, it never errors, and the only symptom is findings the
|
|
73
|
+
operator believed they had excluded.
|
|
61
74
|
"""
|
|
62
75
|
if pat.endswith("/"):
|
|
63
|
-
|
|
76
|
+
bare = pat[3:] if pat.startswith("**/") else pat
|
|
77
|
+
if not bare: # a lone `**/` would otherwise exclude the entire tree
|
|
78
|
+
return False
|
|
79
|
+
return rel.startswith(bare) or f"/{bare}" in f"/{rel}/"
|
|
64
80
|
bare = pat[3:] if pat.startswith("**/") else pat
|
|
65
81
|
return fnmatch.fnmatch(rel, pat) or fnmatch.fnmatch(rel, bare) or fnmatch.fnmatch(name, bare)
|
|
66
82
|
|
|
@@ -97,8 +113,9 @@ def loc_under(
|
|
|
97
113
|
excludes: Iterable[str],
|
|
98
114
|
test_patterns: Iterable[str] = (),
|
|
99
115
|
skip: Iterable[Path] = (),
|
|
100
|
-
|
|
101
|
-
|
|
116
|
+
docs_patterns: Iterable[str] = (),
|
|
117
|
+
) -> tuple[int, int, int]:
|
|
118
|
+
"""Non-blank in-scope lines, split into (primary, test, docs).
|
|
102
119
|
|
|
103
120
|
The split exists because the score's denominator has to move with its
|
|
104
121
|
numerator. Scoring primary-tree findings over a LOC count that included the
|
|
@@ -106,6 +123,12 @@ def loc_under(
|
|
|
106
123
|
tested — the same numerator/denominator mismatch that `exclude_patterns`
|
|
107
124
|
already caused once, arriving by a different door.
|
|
108
125
|
|
|
126
|
+
Documentation is split for the same reason, and was not: its *findings*
|
|
127
|
+
move to their own axis and out of the score, while its *lines* stayed in
|
|
128
|
+
the primary denominator. FastAPI carries 7,160 lines of `docs/en/data/`
|
|
129
|
+
— translator and contributor lists — diluting the count its code is
|
|
130
|
+
graded against. Third occurrence of one mismatch.
|
|
131
|
+
|
|
109
132
|
`skip` names the run's own artifacts — the report, the baseline, the
|
|
110
133
|
suppressions file. Dropping their *findings* without dropping their
|
|
111
134
|
*lines* is that same mismatch a third time: an audit that wrote a
|
|
@@ -114,9 +137,11 @@ def loc_under(
|
|
|
114
137
|
unchanged repository returned 0.00 and 4.25.
|
|
115
138
|
"""
|
|
116
139
|
test_patterns = tuple(test_patterns)
|
|
140
|
+
docs_patterns = tuple(docs_patterns)
|
|
117
141
|
skip = {p.resolve() for p in skip}
|
|
118
142
|
primary = 0
|
|
119
143
|
test = 0
|
|
144
|
+
docs = 0
|
|
120
145
|
# `rglob` on a file yields nothing, so a single-file audit reported zero
|
|
121
146
|
# lines — and a zero denominator is not normalised at all, so the grade
|
|
122
147
|
# became the raw subtotal. Auditing one file is supported; it should be
|
|
@@ -138,6 +163,8 @@ def loc_under(
|
|
|
138
163
|
lines = sum(1 for line in text.splitlines() if line.strip())
|
|
139
164
|
if test_patterns and is_test_path(path, root, test_patterns):
|
|
140
165
|
test += lines
|
|
166
|
+
elif docs_patterns and is_test_path(path, root, docs_patterns):
|
|
167
|
+
docs += lines
|
|
141
168
|
else:
|
|
142
169
|
primary += lines
|
|
143
|
-
return primary, test
|
|
170
|
+
return primary, test, docs
|
|
@@ -5,6 +5,41 @@ declares Security a `DELEGATED` pillar naming this tool, and reports it as
|
|
|
5
5
|
`NotApplicable` so a reader never mistakes silence for safety. This module is
|
|
6
6
|
the other half: the thing that makes that entry unnecessary.
|
|
7
7
|
|
|
8
|
+
**Changing this document's shape is a two-repository release, in order.**
|
|
9
|
+
|
|
10
|
+
MA refuses an unknown `schema` or `schema_version` outright and reports no
|
|
11
|
+
delegated pillar rather than a partial one — deliberately, because "the schema
|
|
12
|
+
string is the producer's promise about the shape, and guessing past it is how
|
|
13
|
+
a consumer starts reporting fields that mean something different." That is the
|
|
14
|
+
safe failure and a *silent* one.
|
|
15
|
+
|
|
16
|
+
So a schema bump ships in this sequence and no other:
|
|
17
|
+
|
|
18
|
+
1. specify the new shape and send it to MA;
|
|
19
|
+
2. **MA lands its reader first**, accepting the old version and the new;
|
|
20
|
+
3. only then does this tool emit the new version.
|
|
21
|
+
|
|
22
|
+
Emitting first leaves the pillar unmeasured for the entire window between the
|
|
23
|
+
two releases, with nothing on either side reporting why. MA will not
|
|
24
|
+
pre-accept an unspecified shape, which is correct for the same reason this
|
|
25
|
+
rule exists. Agreed with the MA maintainer 2026-09-11; see D18, D19 and MA's
|
|
26
|
+
D155.
|
|
27
|
+
|
|
28
|
+
**v2 is live and was cut this way.** It is v1 plus one field, `scoring_model`
|
|
29
|
+
— every v1 key keeps its name, type and meaning. MA's reader accepts v1 and
|
|
30
|
+
v2 and keys its trend on `scoring_model` when present, falling back to the
|
|
31
|
+
producer version when absent, so its reader could land before this tool
|
|
32
|
+
emitted anything and no document was ever refused.
|
|
33
|
+
|
|
34
|
+
**`scoring_model` is load-bearing and is not ours alone.** MA keys trend
|
|
35
|
+
comparability on it, because a delegated pillar can change its scoring model
|
|
36
|
+
without changing its schema — which is exactly what D16 and D17 did. A wrong
|
|
37
|
+
value here silently splices two scoring models into one trend and presents it
|
|
38
|
+
as knowledge, so `tests/integration/test_scoring_drift.py` digests the weights,
|
|
39
|
+
the bands and the real `score()` output and fails when the model moves without
|
|
40
|
+
the integer. `producer.version` remains pinned to `__version__` for the same
|
|
41
|
+
reason, since it is still the key for any v1 document.
|
|
42
|
+
|
|
8
43
|
**Everything structural here is MA's and is reused deliberately.** The scope
|
|
9
44
|
vocabulary, the two-axis split, the posture matrix and its thresholds all come
|
|
10
45
|
from `_pillars.py`. Two tools reporting "level 3" or "healthy" about the same
|
|
@@ -41,7 +76,7 @@ from typing import Any
|
|
|
41
76
|
from secure_code_audit import __version__
|
|
42
77
|
from secure_code_audit.practice import PracticeLevel
|
|
43
78
|
from secure_code_audit.scanner_status import CoverageReport, CoverageStatus
|
|
44
|
-
from secure_code_audit.scoring import AxisReport, ScoreReport, Verdict
|
|
79
|
+
from secure_code_audit.scoring import SCORING_MODEL, AxisReport, ScoreReport, Verdict
|
|
45
80
|
|
|
46
81
|
#: MA's matrix thresholds, imported by value because the two tools must agree
|
|
47
82
|
#: on where the cells fall. `_pillars.py` holds the originals.
|
|
@@ -151,8 +186,24 @@ def to_dict(pillar: SecurityPillar) -> dict[str, Any]:
|
|
|
151
186
|
"""The document MA reads. Both axes present, their mean absent."""
|
|
152
187
|
return {
|
|
153
188
|
"schema": "secure-code-agent/security-pillar",
|
|
154
|
-
|
|
189
|
+
# v2 = v1 plus `scoring_model`. Nothing else moved: every v1 key keeps
|
|
190
|
+
# its name, its type and its meaning.
|
|
191
|
+
"schema_version": 2,
|
|
155
192
|
"producer": {"tool": "secure-code-agent", "version": __version__},
|
|
193
|
+
# Which scoring model produced `condition`, for a consumer keeping a
|
|
194
|
+
# trend. Deliberately top-level rather than inside `producer`:
|
|
195
|
+
# `producer` says *who*, this says *what model*, and a consumer keying
|
|
196
|
+
# on it should not have to reach through an identity block.
|
|
197
|
+
#
|
|
198
|
+
# An integer, not a version string, because the only question is "same
|
|
199
|
+
# or different" — an integer cannot be padded, or compared as text, or
|
|
200
|
+
# read as ordering that means more than it does.
|
|
201
|
+
#
|
|
202
|
+
# MA keyed this on our release version before, which is correct and far
|
|
203
|
+
# too broad: a new series opened on every release, including ones that
|
|
204
|
+
# changed no scoring, and a signal that fires constantly teaches people
|
|
205
|
+
# to ignore it.
|
|
206
|
+
"scoring_model": SCORING_MODEL,
|
|
156
207
|
"generated": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
157
208
|
"pillar": "security",
|
|
158
209
|
"scope": SCOPE,
|
|
@@ -3,8 +3,8 @@
|
|
|
3
3
|
Implements the model documented in docs/scoring.md:
|
|
4
4
|
· finding_score = severity × confidence × category × top25_bonus
|
|
5
5
|
· category_subtotal = Σ finding_score per category
|
|
6
|
-
· category_normalized = subtotal /
|
|
7
|
-
· category_grade = clamp(5.0 - (normalized ×
|
|
6
|
+
· category_normalized = subtotal / (LOC / 1000)
|
|
7
|
+
· category_grade = clamp(5.0 - (normalized × 1.5), 0, 5)
|
|
8
8
|
· overall = min(category_grades)
|
|
9
9
|
"""
|
|
10
10
|
|
|
@@ -48,6 +48,43 @@ CATEGORY_WEIGHT: dict[Category, float] = {
|
|
|
48
48
|
|
|
49
49
|
CWE_TOP25_BONUS = 1.25
|
|
50
50
|
|
|
51
|
+
#: Which scoring model produced a number, for consumers that keep a trend.
|
|
52
|
+
#:
|
|
53
|
+
#: `maintainability-agent` stores this tool's `condition` in its scan history
|
|
54
|
+
#: and has to know when two readings are comparable. A delegated pillar can
|
|
55
|
+
#: change its scoring model **without changing its schema** — same shape, same
|
|
56
|
+
#: fields, a different number for the same repository — which is exactly what
|
|
57
|
+
#: D16 and D17 did. MA previously keyed on our release version, which is
|
|
58
|
+
#: correct but far too broad: it opened a new series on every release,
|
|
59
|
+
#: including ones that changed no scoring, and a signal that fires constantly
|
|
60
|
+
#: teaches people to ignore it.
|
|
61
|
+
#:
|
|
62
|
+
#: **Bump this when a repository's condition could differ for a reason that is
|
|
63
|
+
#: not the repository.** Concretely: the normalizer, the grade slope, any
|
|
64
|
+
#: weight table, the letter bands, the rank discount, `COUNT_LIKE_CATEGORIES`,
|
|
65
|
+
#: the scanner floor, or the built-in rule profile (D10).
|
|
66
|
+
#:
|
|
67
|
+
#: **Do not bump for** documentation, adapters, CLI flags, output formats,
|
|
68
|
+
#: performance, or a parser fix that does not change which findings are
|
|
69
|
+
#: produced.
|
|
70
|
+
#:
|
|
71
|
+
#: A new *rule* does bump it. That was the arguable case and it resolves
|
|
72
|
+
#: against intuition: a repository containing `yaml.unsafe_load` scores lower
|
|
73
|
+
#: the day that rule ships, with no change to the repository. Adding findings
|
|
74
|
+
#: *is* rescoring, because the score is a function of the finding set, and a
|
|
75
|
+
#: user must not read "we can see more now" as "your code got worse".
|
|
76
|
+
#:
|
|
77
|
+
#: **1 is reserved and is never emitted.** It denotes every release before
|
|
78
|
+
#: this field existed, and those releases do not share one model — the
|
|
79
|
+
#: corroboration merge, the rank discount and D16 all moved the numbers. A v1
|
|
80
|
+
#: document simply omits the field and MA keys those on the release version,
|
|
81
|
+
#: which fragments them correctly. Nothing may back-fill a 1.
|
|
82
|
+
#:
|
|
83
|
+
#: `tests/integration/test_scoring_drift.py` holds this honest: it digests the
|
|
84
|
+
#: weights, the bands and the output of the real `score()` over a fixed
|
|
85
|
+
#: matrix, so changing a constant *or* a formula without bumping this fails.
|
|
86
|
+
SCORING_MODEL = 2
|
|
87
|
+
|
|
51
88
|
|
|
52
89
|
# --- letter-grade boundaries (mirrors maintainability-agent) ---------------
|
|
53
90
|
|
|
@@ -173,7 +210,7 @@ def category_subtotal(findings: Iterable[Finding], category: Category) -> float:
|
|
|
173
210
|
|
|
174
211
|
A straight sum measures how many times a pattern matched, and that
|
|
175
212
|
tracks codebase size times how talkative the scanner is — not how much
|
|
176
|
-
risk is in the code.
|
|
213
|
+
risk is in the code. The normalizer was meant to cancel the size half.
|
|
177
214
|
Nothing cancelled the other half, and the corpus said so plainly: the
|
|
178
215
|
worst-first ordering read Python → JavaScript → Go/Ruby → Java, which is
|
|
179
216
|
the order of Bandit's verbosity, and Django, FastAPI, httpx and Flask all
|
|
@@ -216,15 +253,111 @@ def category_subtotal(findings: Iterable[Finding], category: Category) -> float:
|
|
|
216
253
|
)
|
|
217
254
|
|
|
218
255
|
|
|
219
|
-
|
|
220
|
-
|
|
256
|
+
#: Categories where a finding is a count rather than a rate.
|
|
257
|
+
#:
|
|
258
|
+
#: One committed credential is one committed credential regardless of how
|
|
259
|
+
#: much code surrounds it. Everything else here — injection sinks, weak
|
|
260
|
+
#: crypto calls, unsafe deserialization — genuinely does scale with how much
|
|
261
|
+
#: code there is, and comparing two repositories on those means comparing
|
|
262
|
+
#: rates. See `normalize` for the measurement that settled which is which.
|
|
263
|
+
COUNT_LIKE_CATEGORIES: frozenset[str] = frozenset({"secrets"})
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _category_name(category: Category | str) -> str:
|
|
267
|
+
return category.value if isinstance(category, Category) else str(category)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def normalize(subtotal: float, loc_scanned: int, category: Category | str | None = None) -> float:
|
|
271
|
+
"""Weighted findings per thousand lines of scanned code — a density.
|
|
272
|
+
|
|
273
|
+
This was `sqrt(LOC/1000)` and that under-corrected for size, so the
|
|
274
|
+
ranking followed how *big* a repository is rather than how much is wrong
|
|
275
|
+
with it. Measured across the examined corpus:
|
|
276
|
+
|
|
277
|
+
repo LOC weighted findings/kLOC sqrt-normalized
|
|
278
|
+
django 144,473 1.51 18.18 <- ranked worst
|
|
279
|
+
flask 7,841 3.35 9.38
|
|
280
|
+
fastapi 23,764 1.60 7.81
|
|
281
|
+
|
|
282
|
+
Flask carries **2.2x Django's finding density** and normalised at half
|
|
283
|
+
the value. Django sat fifth by density and first by penalty. Spearman
|
|
284
|
+
correlation of grade against size was -0.37 while correlation of density
|
|
285
|
+
against size was +0.12: the number was tracking the wrong variable.
|
|
286
|
+
|
|
287
|
+
Straight density fixes the ordering: re-measured over the same corpus
|
|
288
|
+
after the change, Spearman(LOC, grade) is +0.02, and the worst-ranked
|
|
289
|
+
repository is the densest one rather than the largest one. The slope
|
|
290
|
+
moves with it — see `category_grade` — because the two only make sense
|
|
291
|
+
together.
|
|
292
|
+
|
|
293
|
+
**`secrets` is not a density, and D17 is why.** Adding
|
|
294
|
+
vulnerable-by-design anchors to the corpus exposed the failure directly:
|
|
295
|
+
OWASP Juice Shop carries four hardcoded API keys and three private keys
|
|
296
|
+
and graded **B+**, because 115,340 lines of surrounding code divided
|
|
297
|
+
seven committed credentials down to nothing. Meanwhile Flask, with no
|
|
298
|
+
secrets at all, graded F. A committed private key is one committed
|
|
299
|
+
private key whether the repository is a thousand lines or a million; it
|
|
300
|
+
is a count, not a rate, and dividing it by size is how a training
|
|
301
|
+
application built to be insecure outscored a well-run library.
|
|
302
|
+
|
|
303
|
+
So `secrets` normalizes by `sqrt(LOC/1000)` instead. Not by nothing: a
|
|
304
|
+
larger codebase genuinely does carry more configuration surface, and an
|
|
305
|
+
absolute count made Django fail on two low-confidence hits. Measured over
|
|
306
|
+
the fourteen examined repositories, against whether a repository is
|
|
307
|
+
maintained or written to be vulnerable:
|
|
308
|
+
|
|
309
|
+
variant AUC separation Spearman(LOC, grade)
|
|
310
|
+
linear everywhere (D16) 0.80 -3.76 +0.14
|
|
311
|
+
sqrt everywhere (pre-D16) 0.91 -0.58 -0.32
|
|
312
|
+
linear; secrets absolute 0.90 +0.00 -0.24
|
|
313
|
+
linear; secrets sqrt 1.00 +0.65 -0.01 <-
|
|
314
|
+
|
|
315
|
+
AUC is the probability that a maintained repository outscores a
|
|
316
|
+
vulnerable-by-design one. Negative separation means the populations
|
|
317
|
+
overlap and *no* band table can tell them apart — which is what blocked
|
|
318
|
+
D5's band edges for as long as the corpus had no bad end in it.
|
|
319
|
+
"""
|
|
221
320
|
if loc_scanned <= 0:
|
|
222
321
|
return subtotal
|
|
223
|
-
|
|
322
|
+
per_kloc = max(loc_scanned, 1) / 1000
|
|
323
|
+
if category is not None and _category_name(category) in COUNT_LIKE_CATEGORIES:
|
|
324
|
+
return subtotal / math.sqrt(per_kloc)
|
|
325
|
+
return subtotal / per_kloc
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
#: Grade points lost per normalized weighted finding.
|
|
329
|
+
#:
|
|
330
|
+
#: The slope only rescales — it cannot reorder anything — so it is chosen
|
|
331
|
+
#: against two things the ordering does not fix: where the median of
|
|
332
|
+
#: well-maintained code lands, and how much of the corpus clamps at 0.0 and
|
|
333
|
+
#: loses its tail.
|
|
334
|
+
#:
|
|
335
|
+
#: 1.3 is the largest slope that keeps the maintained-corpus median inside the
|
|
336
|
+
#: B band [3.00, 3.50) *and* keeps the two populations from touching. Measured
|
|
337
|
+
#: across the range, holding the D17 normalizer fixed:
|
|
338
|
+
#:
|
|
339
|
+
#: slope median (maintained) AUC separation clamped at 0
|
|
340
|
+
#: 1.2 3.53 (B+) 1.00 +0.98 4
|
|
341
|
+
#: 1.3 3.41 (B) 1.00 +0.65 4 <- adopted
|
|
342
|
+
#: 1.4 3.28 (B) 1.00 +0.31 4
|
|
343
|
+
#: 1.5 3.16 (B) 0.95 +0.00 5
|
|
344
|
+
#:
|
|
345
|
+
#: At 1.5 a maintained repository joins the four vulnerable-by-design ones at
|
|
346
|
+
#: the clamp, the populations touch, and AUC falls. 1.3 has the widest margin
|
|
347
|
+
#: of the slopes that land the median in B.
|
|
348
|
+
#:
|
|
349
|
+
#: The cost, stated: a large repository with a handful of serious findings
|
|
350
|
+
#: still scores better than a small noisy one. A 135,841-line control carrying
|
|
351
|
+
#: SQL injection, `shell=True`, `pickle.loads`, MD5 and `eval` grades in the
|
|
352
|
+
#: A band. What protects that repository is the default `fail_on_new` gate,
|
|
353
|
+
#: which fails it outright, and the work order, which puts all seven findings
|
|
354
|
+
#: in §FIX. A grade is for comparing and for trend; it was never the thing
|
|
355
|
+
#: that catches a vulnerability. See D16 and D17.
|
|
356
|
+
GRADE_SLOPE = 1.3
|
|
224
357
|
|
|
225
358
|
|
|
226
359
|
def category_grade(normalized: float) -> float:
|
|
227
|
-
return max(0.0, min(5.0, 5.0 - (normalized *
|
|
360
|
+
return max(0.0, min(5.0, 5.0 - (normalized * GRADE_SLOPE)))
|
|
228
361
|
|
|
229
362
|
|
|
230
363
|
# --- overall score ---------------------------------------------------------
|
|
@@ -520,7 +653,7 @@ def score(
|
|
|
520
653
|
per_category[cat] = None
|
|
521
654
|
continue
|
|
522
655
|
subtotal = category_subtotal(findings, cat)
|
|
523
|
-
per_category[cat] = category_grade(normalize(subtotal, loc_scanned))
|
|
656
|
+
per_category[cat] = category_grade(normalize(subtotal, loc_scanned, cat))
|
|
524
657
|
|
|
525
658
|
for f in findings:
|
|
526
659
|
if not f.suppressed:
|
|
@@ -691,7 +824,26 @@ def _gate_fail_on_new(
|
|
|
691
824
|
]
|
|
692
825
|
if new_findings:
|
|
693
826
|
tripped.append("fail_on_new")
|
|
694
|
-
|
|
827
|
+
# What "new" means depends on whether there is anything to be new
|
|
828
|
+
# *against*. Saying "since baseline" when no baseline exists claims
|
|
829
|
+
# these findings appeared after one, which is the opposite of the
|
|
830
|
+
# truth on a first run.
|
|
831
|
+
baseline_state = gate_config.get("_baseline_state")
|
|
832
|
+
if baseline_state == "absent":
|
|
833
|
+
reasons.append(
|
|
834
|
+
f"{len(new_findings)} finding(s), and no baseline exists yet — on a first "
|
|
835
|
+
f"run everything is new because there is nothing to compare against. "
|
|
836
|
+
f"Work the order, then re-run with --bump-baseline to accept what is "
|
|
837
|
+
f"left and gate on regressions from there."
|
|
838
|
+
)
|
|
839
|
+
elif baseline_state == "unreadable":
|
|
840
|
+
reasons.append(
|
|
841
|
+
f"{len(new_findings)} finding(s) read as new because the baseline file "
|
|
842
|
+
f"could not be parsed. Fix or delete it — a broken baseline silently "
|
|
843
|
+
f"turns an established repository back into a first run."
|
|
844
|
+
)
|
|
845
|
+
else:
|
|
846
|
+
reasons.append(f"{len(new_findings)} new finding(s) since baseline")
|
|
695
847
|
|
|
696
848
|
|
|
697
849
|
def _gate_min_score(
|
|
@@ -95,8 +95,19 @@ class StandardsEntry:
|
|
|
95
95
|
_MAP: dict[tuple[str, str], StandardsEntry] = {
|
|
96
96
|
# ----- Bandit ----------------------------------------------------------
|
|
97
97
|
# Source: https://bandit.readthedocs.io/en/latest/plugins/index.html
|
|
98
|
+
# B102 read CWE-78 — *OS* command injection, the shell-injection weakness
|
|
99
|
+
# that B602/B603/B605/B607 cover. `exec()` does not invoke a shell; it
|
|
100
|
+
# compiles and runs Python. The mislabel travelled: into the OWASP
|
|
101
|
+
# mapping, into SARIF, into the work order, and into the Top-25 bonus,
|
|
102
|
+
# which CWE-78 carries and the true weakness does not directly.
|
|
103
|
+
#
|
|
104
|
+
# It also broke corroboration, which is how it was found. Bandit's B102
|
|
105
|
+
# and our own `sca.python.eval` fire on the same `exec(compile(...))`
|
|
106
|
+
# line in Flask's `config.py`; `_same_weakness` merges across scanners on
|
|
107
|
+
# a shared CWE, CWE-78 and CWE-95 are not shared, so one defect scored
|
|
108
|
+
# twice. Flask carried four such pairs and graded F partly on doubles.
|
|
98
109
|
("bandit", "B102"): StandardsEntry(
|
|
99
|
-
canonical_cwe="CWE-
|
|
110
|
+
canonical_cwe="CWE-95",
|
|
100
111
|
owasp_top10="A03",
|
|
101
112
|
asvs_section="V5.3.8",
|
|
102
113
|
nist_ssdf="PW.5.1",
|
|
@@ -106,6 +117,20 @@ _MAP: dict[tuple[str, str], StandardsEntry] = {
|
|
|
106
117
|
short_desc="Use of exec() — arbitrary code execution risk.",
|
|
107
118
|
fix_hint="Eliminate exec() entirely. If dynamic dispatch is required, use a typed registry / function map.",
|
|
108
119
|
),
|
|
120
|
+
# B307 (`eval`) had no curated entry, so `_make_finding` fell back to the
|
|
121
|
+
# CWE the scanner reports — and Bandit files `eval` under CWE-78 too. Same
|
|
122
|
+
# weakness as B102, same fix, and curating it is what stops the fallback.
|
|
123
|
+
("bandit", "B307"): StandardsEntry(
|
|
124
|
+
canonical_cwe="CWE-95",
|
|
125
|
+
owasp_top10="A03",
|
|
126
|
+
asvs_section="V5.2.4",
|
|
127
|
+
nist_ssdf="PW.5.1",
|
|
128
|
+
category=Category.CODE_VULNERABILITIES,
|
|
129
|
+
severity=Severity.HIGH,
|
|
130
|
+
confidence=Confidence.MEDIUM,
|
|
131
|
+
short_desc="Use of eval() — arbitrary code execution risk.",
|
|
132
|
+
fix_hint="Use ast.literal_eval for data. For dispatch, use a dict of callables rather than evaluating a name.",
|
|
133
|
+
),
|
|
109
134
|
("bandit", "B301"): StandardsEntry(
|
|
110
135
|
canonical_cwe="CWE-502",
|
|
111
136
|
owasp_top10="A08",
|
|
@@ -494,9 +519,41 @@ def lookup(scanner: str, rule_id: str) -> StandardsEntry | None:
|
|
|
494
519
|
return _MAP.get((scanner.lower(), "*"))
|
|
495
520
|
|
|
496
521
|
|
|
522
|
+
#: Child CWEs this project maps to, and the Top-25 entry each is a ChildOf.
|
|
523
|
+
#:
|
|
524
|
+
#: The Top-25 list names classes, and a scanner names the specific weakness
|
|
525
|
+
#: inside one. CWE-95 ("Eval Injection") is a documented ChildOf CWE-94
|
|
526
|
+
#: ("Improper Control of Generation of Code"), which is on the list — so a
|
|
527
|
+
#: confirmed eval injection *is* a Top-25 weakness, and a membership test
|
|
528
|
+
#: that only compares strings says it is not.
|
|
529
|
+
#:
|
|
530
|
+
#: This is deliberately a hand-checked handful rather than an imported CWE
|
|
531
|
+
#: hierarchy. Every entry is a relationship stated in the MITRE definition of
|
|
532
|
+
#: the child, and each is used by a rule this project actually maps. Adding a
|
|
533
|
+
#: parent here widens the 1.25x bonus, so it is a decision, not a lookup.
|
|
534
|
+
_TOP25_PARENT: dict[str, str] = {
|
|
535
|
+
# cwe.mitre.org/data/definitions/95.html — ChildOf 94
|
|
536
|
+
"CWE-95": "CWE-94",
|
|
537
|
+
# cwe.mitre.org/data/definitions/77.html is itself on the list; 78 is the
|
|
538
|
+
# OS-command child and is also listed, so neither needs an entry here.
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
|
|
497
542
|
def is_top25(canonical_cwe: str | None) -> bool:
|
|
498
|
-
"""Is this CWE on the MITRE Top 25 (2025) list
|
|
499
|
-
|
|
543
|
+
"""Is this CWE on the MITRE Top 25 (2025) list, directly or as a child?
|
|
544
|
+
|
|
545
|
+
Correcting Bandit's B102/B307 from CWE-78 to CWE-95 was right on the
|
|
546
|
+
weakness and would have quietly removed the Top-25 bonus from every
|
|
547
|
+
eval/exec finding in the corpus — CWE-78 is on the list and CWE-95 is
|
|
548
|
+
not. Losing the bonus for the *reason* "we now describe the weakness
|
|
549
|
+
accurately" is the wrong trade, and CWE-95 is a child of CWE-94, which
|
|
550
|
+
is listed. So membership follows the relationship.
|
|
551
|
+
"""
|
|
552
|
+
if canonical_cwe is None:
|
|
553
|
+
return False
|
|
554
|
+
if canonical_cwe in CWE_TOP25_2025:
|
|
555
|
+
return True
|
|
556
|
+
return _TOP25_PARENT.get(canonical_cwe, "") in CWE_TOP25_2025
|
|
500
557
|
|
|
501
558
|
|
|
502
559
|
def cwe_url(canonical_cwe: str) -> str:
|
|
@@ -29,7 +29,7 @@ import enum
|
|
|
29
29
|
import re
|
|
30
30
|
from collections.abc import Iterable
|
|
31
31
|
|
|
32
|
-
from secure_code_audit.findings import Confidence, Finding, Severity
|
|
32
|
+
from secure_code_audit.findings import Category, Confidence, Finding, Severity
|
|
33
33
|
|
|
34
34
|
|
|
35
35
|
class Tier(enum.Enum):
|
|
@@ -104,6 +104,123 @@ def _looks_like_a_credential(message: str) -> bool:
|
|
|
104
104
|
return len(value) >= 12 and any(c.isdigit() for c in value) and any(c.isupper() for c in value)
|
|
105
105
|
|
|
106
106
|
|
|
107
|
+
#: A credential *reference* — a name standing in for a value that is not here.
|
|
108
|
+
#:
|
|
109
|
+
#: Shell (`$TOKEN`, `${TOKEN}`), GitHub Actions (`${{ secrets.X }}`,
|
|
110
|
+
#: `${{ env.X }}`), Windows (`%TOKEN%`), and the ordinary code spellings.
|
|
111
|
+
_VARIABLE_REFERENCE = re.compile(
|
|
112
|
+
r"""
|
|
113
|
+
\$\{\{\s*(?:secrets|env|vars)\. # ${{ secrets.NAME }}
|
|
114
|
+
| \$\{[A-Za-z_][A-Za-z0-9_]* # ${NAME}
|
|
115
|
+
| \$[A-Za-z_][A-Za-z0-9_]* # $NAME
|
|
116
|
+
| %[A-Za-z_][A-Za-z0-9_]*% # %NAME%
|
|
117
|
+
| os\.environ | os\.getenv # Python
|
|
118
|
+
| process\.env\. # Node
|
|
119
|
+
| ENV\[ # Ruby
|
|
120
|
+
""",
|
|
121
|
+
re.VERBOSE,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
#: A URL, stripped before looking for a literal credential.
|
|
125
|
+
#:
|
|
126
|
+
#: This is not cosmetic. The first version of the literal test matched
|
|
127
|
+
#: `//sonarcloud` inside `https://sonarcloud.io/api` — twelve characters of
|
|
128
|
+
#: `[A-Za-z0-9+/_-]`, because `/` and `+` are base64 alphabet and a URL is
|
|
129
|
+
#: full of them. Every `curl` line has a URL on it, so the demotion never
|
|
130
|
+
#: fired on the exact case it was written for.
|
|
131
|
+
_URL = re.compile(r"\bhttps?://\S+", re.IGNORECASE)
|
|
132
|
+
|
|
133
|
+
#: A literal that could itself be the credential, sitting on the same line.
|
|
134
|
+
_LITERAL_RUN = re.compile(r"[A-Za-z0-9+/_\-]{12,}")
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _contains_a_literal_credential(text: str) -> bool:
|
|
138
|
+
"""Twelve-plus characters with a digit and an uppercase letter.
|
|
139
|
+
|
|
140
|
+
Deliberately the same dull test `_looks_like_a_credential` applies to
|
|
141
|
+
Bandit's quoted values — same shape, same reasons, and it keeps the two
|
|
142
|
+
heuristics from drifting into disagreement about what a secret looks
|
|
143
|
+
like.
|
|
144
|
+
|
|
145
|
+
**Known miss, stated:** an all-lowercase hex token such as
|
|
146
|
+
`a3f9c2b1d4e5` has no uppercase letter and is not caught. That costs a
|
|
147
|
+
demotion from FIX to REVIEW, and only on a line that *also* carries a
|
|
148
|
+
variable reference, which is an odd thing to write. The finding still
|
|
149
|
+
appears in the work order, still appears in the report, and still
|
|
150
|
+
escalates through `fail_on_category`. The error falls in the safe
|
|
151
|
+
direction.
|
|
152
|
+
"""
|
|
153
|
+
for run in _LITERAL_RUN.findall(_URL.sub(" ", text)):
|
|
154
|
+
if any(c.isdigit() for c in run) and any(c.isupper() for c in run):
|
|
155
|
+
return True
|
|
156
|
+
return False
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _line_of(finding: Finding) -> str | None:
|
|
160
|
+
"""Read back the source line a secrets finding points at.
|
|
161
|
+
|
|
162
|
+
Gitleaks is run with `--redact`, so the matched text never reaches the
|
|
163
|
+
report — the message reads `curl -sS -u REDACTED`, and the variable name
|
|
164
|
+
is precisely what was redacted. The only way to tell a reference from a
|
|
165
|
+
value is to look at the line, the way `verify._is_silenced` does.
|
|
166
|
+
|
|
167
|
+
Returns None whenever the line cannot be read with confidence: the file
|
|
168
|
+
is gone, the path is a directory, the line number is out of range, or the
|
|
169
|
+
finding came out of git history and the working tree has moved on. Every
|
|
170
|
+
one of those must leave the finding where it was.
|
|
171
|
+
"""
|
|
172
|
+
path = finding.file_path
|
|
173
|
+
if path is None or finding.line_start is None or finding.line_start < 1:
|
|
174
|
+
return None
|
|
175
|
+
try:
|
|
176
|
+
if not path.is_file():
|
|
177
|
+
return None
|
|
178
|
+
with path.open("r", encoding="utf-8", errors="replace") as handle:
|
|
179
|
+
for number, line in enumerate(handle, 1):
|
|
180
|
+
if number == finding.line_start:
|
|
181
|
+
return line
|
|
182
|
+
if number > finding.line_start:
|
|
183
|
+
break
|
|
184
|
+
except OSError:
|
|
185
|
+
return None
|
|
186
|
+
return None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _is_a_reference_not_a_value(finding: Finding) -> bool:
|
|
190
|
+
"""Is the flagged credential a variable name rather than a credential?
|
|
191
|
+
|
|
192
|
+
`gitleaks.curl-auth-user` fires CRITICAL on
|
|
193
|
+
`curl -sS -u "$SONAR_TOKEN:"` in a GitHub Actions workflow. No credential
|
|
194
|
+
is present; `$SONAR_TOKEN` is how you write *not* putting one there, and
|
|
195
|
+
the idiom is the recommended one. Reported independently on two
|
|
196
|
+
repositories on the same day, five occurrences on one of them.
|
|
197
|
+
|
|
198
|
+
This matters more than it used to. D17 made `secrets` count-like — it
|
|
199
|
+
normalizes by `sqrt(LOC/1000)` rather than by size — so a single false
|
|
200
|
+
critical now costs roughly two grade points on a 100k-line repository
|
|
201
|
+
where it previously cost two tenths. Raising the weight of a category
|
|
202
|
+
raises the cost of being wrong in it, and this is the corresponding
|
|
203
|
+
precision work.
|
|
204
|
+
|
|
205
|
+
Conservative in both directions. It demotes to REVIEW, never to ACCEPT:
|
|
206
|
+
the finding stays in the work order, stays in the report, and still
|
|
207
|
+
escalates through `fail_on_category: [secrets]` from any axis. And it
|
|
208
|
+
declines to act unless it can read the line and finds no literal token on
|
|
209
|
+
it, so `curl -u "$USER:hunter2Passw0rd"` is untouched.
|
|
210
|
+
"""
|
|
211
|
+
if finding.category is not Category.SECRETS:
|
|
212
|
+
return False
|
|
213
|
+
line = _line_of(finding)
|
|
214
|
+
if line is None:
|
|
215
|
+
return False
|
|
216
|
+
if not _VARIABLE_REFERENCE.search(line):
|
|
217
|
+
return False
|
|
218
|
+
# A reference and a literal on one line is still a leak. Strip the
|
|
219
|
+
# references first so their own names cannot satisfy the literal test.
|
|
220
|
+
without_references = _VARIABLE_REFERENCE.sub(" ", line)
|
|
221
|
+
return not _contains_a_literal_credential(without_references)
|
|
222
|
+
|
|
223
|
+
|
|
107
224
|
def tier_of(finding: Finding, axis: str = "primary") -> Tier:
|
|
108
225
|
"""Classify one finding.
|
|
109
226
|
|
|
@@ -118,6 +235,8 @@ def tier_of(finding: Finding, axis: str = "primary") -> Tier:
|
|
|
118
235
|
if _looks_like_a_credential(finding.message):
|
|
119
236
|
return Tier.FIX
|
|
120
237
|
return Tier.REVIEW
|
|
238
|
+
if _is_a_reference_not_a_value(finding):
|
|
239
|
+
return Tier.REVIEW
|
|
121
240
|
if finding.confidence is Confidence.LOW:
|
|
122
241
|
return Tier.REVIEW
|
|
123
242
|
return Tier.FIX
|
|
@@ -128,6 +247,15 @@ def reason_for(finding: Finding) -> str | None:
|
|
|
128
247
|
measured = LOW_PRECISION.get(finding.rule_id)
|
|
129
248
|
if measured:
|
|
130
249
|
return measured
|
|
250
|
+
if _is_a_reference_not_a_value(finding):
|
|
251
|
+
return (
|
|
252
|
+
"The line holds a variable reference, not a credential — a name "
|
|
253
|
+
"standing in for a value kept elsewhere, which is the recommended "
|
|
254
|
+
"way to write this. Checked by reading the line back, because "
|
|
255
|
+
"gitleaks runs with --redact and the redacted text is the "
|
|
256
|
+
"variable name itself. Confirm the value really is injected at "
|
|
257
|
+
"runtime, then suppress it."
|
|
258
|
+
)
|
|
131
259
|
if finding.confidence is Confidence.LOW:
|
|
132
260
|
return (
|
|
133
261
|
f"{finding.scanner} reported this at low confidence — it is "
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/SOURCES.txt
RENAMED
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/entry_points.txt
RENAMED
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/requires.txt
RENAMED
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/top_level.txt
RENAMED
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/data/semgrep-offline.yaml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanner_status.py
RENAMED
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/builtin_rules.py
RENAMED
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/floor.py
RENAMED
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/gosec_scanner.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/osv_scanner.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/trivy_scanner.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|