secure-code-agent 0.8.0__tar.gz → 0.10.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {secure_code_agent-0.8.0/src/secure_code_agent.egg-info → secure_code_agent-0.10.0}/PKG-INFO +1 -1
  2. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0/src/secure_code_agent.egg-info}/PKG-INFO +1 -1
  3. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/__init__.py +1 -1
  4. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/baseline.py +31 -0
  5. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/cli.py +32 -4
  6. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/config.py +87 -3
  7. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/git_tools.py +31 -4
  8. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/pillar.py +53 -2
  9. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scoring.py +161 -9
  10. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/standards.py +60 -3
  11. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/triage.py +129 -1
  12. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/LICENSE +0 -0
  13. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/README.md +0 -0
  14. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/pyproject.toml +0 -0
  15. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/setup.cfg +0 -0
  16. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/SOURCES.txt +0 -0
  17. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/dependency_links.txt +0 -0
  18. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/entry_points.txt +0 -0
  19. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/requires.txt +0 -0
  20. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_agent.egg-info/top_level.txt +0 -0
  21. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/data/semgrep-offline.yaml +0 -0
  22. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/findings.py +0 -0
  23. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/history.py +0 -0
  24. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/instructions.py +0 -0
  25. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/practice.py +0 -0
  26. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/remediation.py +0 -0
  27. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/renderers.py +0 -0
  28. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/ruleset.py +0 -0
  29. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/sarif.py +0 -0
  30. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanner_status.py +0 -0
  31. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/__init__.py +0 -0
  32. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/bandit_scanner.py +0 -0
  33. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/base.py +0 -0
  34. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/builtin_rules.py +0 -0
  35. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/checkov_scanner.py +0 -0
  36. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/floor.py +0 -0
  37. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/gitleaks_scanner.py +0 -0
  38. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/gosec_scanner.py +0 -0
  39. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/hadolint_scanner.py +0 -0
  40. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/njsscan_scanner.py +0 -0
  41. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/npm_audit_scanner.py +0 -0
  42. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/osv_scanner.py +0 -0
  43. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/pip_audit_scanner.py +0 -0
  44. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/rubocop_scanner.py +0 -0
  45. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/scorecard_scanner.py +0 -0
  46. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/semgrep_scanner.py +0 -0
  47. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/trivy_scanner.py +0 -0
  48. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/scanners/trufflehog_scanner.py +0 -0
  49. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/suppressions.py +0 -0
  50. {secure_code_agent-0.8.0 → secure_code_agent-0.10.0}/src/secure_code_audit/verify.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: secure-code-agent
3
- Version: 0.8.0
3
+ Version: 0.10.0
4
4
  Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
5
5
  Author: Marshall Guillory
6
6
  License: MIT
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: secure-code-agent
3
- Version: 0.8.0
3
+ Version: 0.10.0
4
4
  Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
5
5
  Author: Marshall Guillory
6
6
  License: MIT
@@ -12,4 +12,4 @@
12
12
  #:
13
13
  #: PyPI is immutable, so 0.4.0 stays wrong. 0.5.0 is the first build whose
14
14
  #: artifacts name their own producer correctly.
15
- __version__ = "0.8.0"
15
+ __version__ = "0.10.0"
@@ -10,6 +10,7 @@ operator has acknowledged at a point in time. On the next run:
10
10
  from __future__ import annotations
11
11
 
12
12
  import datetime
13
+ import enum
13
14
  import json
14
15
  import shutil
15
16
  import subprocess
@@ -32,6 +33,36 @@ class BaselineEntry:
32
33
  notes: str = ""
33
34
 
34
35
 
36
+ class State(enum.Enum):
37
+ """Why the baseline is the shape it is.
38
+
39
+ `load` returns an empty mapping for a baseline that is absent, one that
40
+ is unreadable, and one that is genuinely empty. Downstream those look
41
+ identical and every finding reads as new — which is correct for the
42
+ first case, a silent failure for the second, and worth saying out loud
43
+ in all three once `fail_on_new` gates by default.
44
+
45
+ On a first run the honest message is "there is nothing to compare
46
+ against yet", not "2 new findings since baseline", which claims a
47
+ baseline exists and that these appeared after it.
48
+ """
49
+
50
+ ABSENT = "absent"
51
+ UNREADABLE = "unreadable"
52
+ PRESENT = "present"
53
+
54
+
55
+ def state(path: Path) -> State:
56
+ """Distinguish a missing baseline from a broken one."""
57
+ if not path.exists():
58
+ return State.ABSENT
59
+ try:
60
+ raw = json.loads(path.read_text(encoding="utf-8"))
61
+ except (OSError, json.JSONDecodeError):
62
+ return State.UNREADABLE
63
+ return State.PRESENT if isinstance(raw, dict) else State.UNREADABLE
64
+
65
+
35
66
  def load(path: Path) -> dict[str, BaselineEntry]:
36
67
  if not path.exists():
37
68
  return {}
@@ -160,7 +160,11 @@ def _parser() -> argparse.ArgumentParser:
160
160
  "--target",
161
161
  action="append",
162
162
  default=[],
163
- help="Target for --init-agent-standards (codex, claude-code, cursor, copilot, windsurf, generic).",
163
+ help=(
164
+ "Agent name for --init-agent-standards (codex, claude-code, cursor, "
165
+ "copilot, windsurf, generic). NOT the repository to audit — pass that "
166
+ "as a positional path."
167
+ ),
164
168
  )
165
169
  p.add_argument(
166
170
  "--instructions-output-dir",
@@ -183,6 +187,22 @@ def main(argv: list[str] | None = None) -> int:
183
187
  if args.init_agent_standards:
184
188
  return _do_init_standards(args)
185
189
 
190
+ if args.target:
191
+ # `--target` names an *agent* for --init-agent-standards, and it reads
192
+ # exactly like the flag for "the repository to audit". It was accepted
193
+ # and silently discarded on an audit run, so `--target /some/repo`
194
+ # audited the current directory instead and reported a clean result
195
+ # for a repository nobody had looked at. Found by using the tool: a
196
+ # 136k-line control was audited from inside itself and appeared to
197
+ # honour the flag, which is the kind of coincidence that keeps a bug.
198
+ sys.stderr.write(
199
+ "ERROR: --target names an agent for --init-agent-standards, not a "
200
+ "repository to audit.\n"
201
+ f" To audit a repository, pass it as a path: "
202
+ f"secure-code-agent {args.target[0]}\n"
203
+ )
204
+ return 2
205
+
186
206
  try:
187
207
  return _do_preflight(args) if args.preflight else _do_audit(args)
188
208
  except ValueError as exc:
@@ -338,6 +358,7 @@ def _do_audit(args: argparse.Namespace) -> int:
338
358
  # ----- baseline -----
339
359
  baseline_path = _under_root(root, args.baseline or cfg.outputs["baseline_path"])
340
360
  baseline = baseline_mod.load(baseline_path)
361
+ baseline_state = baseline_mod.state(baseline_path)
341
362
  all_findings = baseline_mod.mark_new(all_findings, baseline)
342
363
 
343
364
  # ----- scoring -----
@@ -377,8 +398,9 @@ def _do_audit(args: argparse.Namespace) -> int:
377
398
  if cfg.loc_for_scoring:
378
399
  loc = int(cfg.loc_for_scoring.get("value", 0))
379
400
  test_loc = 0
401
+ docs_loc = 0
380
402
  else:
381
- loc, test_loc = loc_under(
403
+ loc, test_loc, docs_loc = loc_under(
382
404
  target,
383
405
  cfg.include_extensions,
384
406
  cfg.exclude_patterns,
@@ -386,6 +408,7 @@ def _do_audit(args: argparse.Namespace) -> int:
386
408
  # Same set the findings were filtered against. Numerator and
387
409
  # denominator have to describe the same repository.
388
410
  own_artifacts,
411
+ cfg.docs_patterns,
389
412
  )
390
413
  # Dependencies come off the code-condition score and onto their own axis.
391
414
  # A CVE in a pinned dependency is fixed with a version bump; an injection
@@ -405,7 +428,7 @@ def _do_audit(args: argparse.Namespace) -> int:
405
428
  score = score_findings(scored_findings, loc, measurable)
406
429
  axes = (
407
430
  summarize_axis("test tree", test_findings, test_loc),
408
- summarize_axis("documentation", docs_findings),
431
+ summarize_axis("documentation", docs_findings, docs_loc or None),
409
432
  summarize_axis("dependencies", dependency_findings),
410
433
  )
411
434
  # Naming an import on the command line asserts that it contributes coverage,
@@ -425,7 +448,12 @@ def _do_audit(args: argparse.Namespace) -> int:
425
448
  ]
426
449
  coverage = evaluate_coverage(executions, required)
427
450
  # Gates see the dependency advisories; the score does not.
428
- gate = evaluate_gates(gated, score, cfg.gates, coverage)
451
+ # The gate needs to know *why* the baseline is empty to describe a first
452
+ # run truthfully. Passed in the config dict rather than as a parameter so
453
+ # the gate signature stays the one every check shares.
454
+ gate = evaluate_gates(
455
+ gated, score, {**cfg.gates, "_baseline_state": baseline_state.value}, coverage
456
+ )
429
457
  verdict = build_verdict(score, cfg.gates, coverage)
430
458
 
431
459
  # ----- write outputs -----
@@ -8,7 +8,7 @@ from __future__ import annotations
8
8
 
9
9
  import json
10
10
  from dataclasses import dataclass, field
11
- from pathlib import Path
11
+ from pathlib import Path, PurePosixPath
12
12
  from typing import Any
13
13
 
14
14
  DEFAULT_CONFIG_PATH = Path("secure-code-agent.json")
@@ -26,6 +26,53 @@ DEFAULT_EXCLUDES: tuple[str, ...] = (
26
26
  ".mypy_cache/",
27
27
  "**/*.min.js",
28
28
  "**/*.lock",
29
+ # Lockfiles that are not named `.lock`. `**/*.lock` catches
30
+ # `poetry.lock`, `Gemfile.lock`, `Cargo.lock` and `yarn.lock` and misses
31
+ # every lockfile the JavaScript ecosystem actually ships:
32
+ # `package-lock.json` alone was 9,699 of axios's 17,532 non-code lines
33
+ # and 5,845 of lodash's 6,222. A generated dependency manifest is not
34
+ # source, and counting it inflates the denominator that decides the
35
+ # grade.
36
+ "**/package-lock.json",
37
+ "**/npm-shrinkwrap.json",
38
+ "**/pnpm-lock.yaml",
39
+ "**/bun.lockb",
40
+ # --- stored analysis output -------------------------------------------
41
+ #
42
+ # This tool's own output is never its input, *wherever* it is stored.
43
+ #
44
+ # `cli._own_artifacts` already removes the paths the current run is about
45
+ # to write, which is what stopped an audit scoring the report it had just
46
+ # produced. It cannot see a *copy* kept somewhere else, and two
47
+ # independent reports of that landed on the same day:
48
+ #
49
+ # - this repository scanned `calibration/.corpus` — fourteen cloned
50
+ # third-party projects, 556,808 LOC and 550 findings, all about code
51
+ # that is not ours;
52
+ # - `maintainability-agent` scanned `tools/validation/reports/` —
53
+ # 957,219 LOC of stored audit output *about other repositories*,
54
+ # 4,929 findings, which diluted five genuine criticals to an A-.
55
+ #
56
+ # A stored report is the worst possible input: it quotes findings
57
+ # verbatim, including the code snippets and the redacted secrets that
58
+ # produced them, so it manufactures findings about findings and inflates
59
+ # the denominator at the same time.
60
+ #
61
+ # Filename patterns rather than directory names, because the directory is
62
+ # whatever the operator chose and the filenames are ours. Derived from
63
+ # `DEFAULT_OUTPUTS` below so the two cannot drift — see
64
+ # `_own_output_globs`.
65
+ #
66
+ # Deliberately NOT here: `vendor/`, `third_party/` and their kin. Vendored
67
+ # code is deployed code, and excluding it by default would hide real
68
+ # vulnerabilities in exactly the place nobody is reading. Stored analysis
69
+ # output is not code at all; that is the whole difference.
70
+ ".secure-code/",
71
+ "**/.secure-code/",
72
+ # maintainability-agent's state and output directory. Same argument: its
73
+ # reports quote findings, and its history is append-only JSONL.
74
+ ".maintainability/",
75
+ "**/.maintainability/",
29
76
  )
30
77
 
31
78
  #: Conventional test-tree locations across the languages the floor reads.
@@ -96,6 +143,24 @@ DEFAULT_OUTPUTS: dict[str, str] = {
96
143
  "history_path": ".secure-code/history.jsonl",
97
144
  }
98
145
 
146
+
147
+ def _own_output_globs() -> tuple[str, ...]:
148
+ """Match this tool's default output filenames anywhere in a tree.
149
+
150
+ Kept as a function over `DEFAULT_OUTPUTS` rather than a hand-written list
151
+ so that adding an output cannot leave a file this tool writes readable by
152
+ the next run. `history_path` is already covered by the `.secure-code/`
153
+ directory entries; matching its basename anywhere would be wrong, since
154
+ `history.jsonl` is not a name this project owns.
155
+ """
156
+ names = {
157
+ PurePosixPath(path).name for key, path in DEFAULT_OUTPUTS.items() if key != "history_path"
158
+ }
159
+ return tuple(sorted(f"**/{name}" for name in names))
160
+
161
+
162
+ DEFAULT_EXCLUDES = DEFAULT_EXCLUDES + _own_output_globs()
163
+
99
164
  _SEVERITIES = {"critical", "high", "medium", "low", "informational"}
100
165
  _CATEGORIES = {
101
166
  "secrets",
@@ -169,7 +234,24 @@ class Config:
169
234
  scanners: dict[str, ScannerConfig] = field(default_factory=dict)
170
235
  severity_overrides: dict[str, str] = field(default_factory=dict)
171
236
  category_overrides: dict[str, str] = field(default_factory=dict)
172
- gates: dict[str, Any] = field(default_factory=dict)
237
+ #: Default policy: **ratchet on regressions**, not on absolute state.
238
+ #:
239
+ #: `fail_on_new` is the one gate measured to work. Severity-based
240
+ #: defaults cannot: `fail_on_severity: ["critical"]` caught none of four
241
+ #: known-vulnerable control repositories, and `["critical","high"]`
242
+ #: failed five of ten well-maintained ones while still missing SQL
243
+ #: injection and `pickle.loads`, which Bandit rates *medium*. See D15.
244
+ #:
245
+ #: The ratchet reads no severity at all, so it inherits none of that. A
246
+ #: false positive is baselined once and never asked about again, which
247
+ #: is the property a threshold cannot have. Measured across an adoption
248
+ #: lifecycle: first run fails (everything is new), `--bump-baseline`
249
+ #: accepts existing debt, an introduced SQL injection fails, reverting
250
+ #: passes.
251
+ #:
252
+ #: An empty `{}` was the previous default and provided no floor at all —
253
+ #: an absent gate cannot trip, so every audit "passed".
254
+ gates: dict[str, Any] = field(default_factory=lambda: {"fail_on_new": True})
173
255
  outputs: dict[str, str] = field(default_factory=lambda: dict(DEFAULT_OUTPUTS))
174
256
  suppressions_file: str = ".scignore.yaml"
175
257
  loc_for_scoring: dict[str, Any] | None = None
@@ -343,7 +425,9 @@ def _from_dict(raw: dict[str, Any]) -> Config:
343
425
  raise ValueError(f"invalid severity override: {', '.join(invalid)}")
344
426
  if invalid := sorted(set(cfg.category_overrides.values()) - _CATEGORIES):
345
427
  raise ValueError(f"invalid category override: {', '.join(invalid)}")
346
- cfg.gates = _validate_gates(raw.get("gates", {}))
428
+ # An operator who writes a `gates` block chooses their own policy
429
+ # entirely; the ratchet default applies only when they write none.
430
+ cfg.gates = _validate_gates(raw["gates"]) if "gates" in raw else dict(cfg.gates)
347
431
 
348
432
  outputs = raw.get("outputs", {})
349
433
  if not isinstance(outputs, dict):
@@ -58,9 +58,25 @@ def _matches(rel: str, name: str, pat: str) -> bool:
58
58
  `router/context_test.go` and never `context_test.go`. Gin keeps its tests
59
59
  beside the code they test, so three of its four "production" secrets were
60
60
  test fixtures at the repository root, and that alone held it at F.
61
+
62
+ **The `**/` strip has to happen on the directory branch too.** It did not,
63
+ and so `**/__pycache__/` matched *nothing*: the branch searched for a
64
+ literal `/**/__pycache__/` inside the path, and no real path contains
65
+ `/**/`. A bare `__pycache__/` already matches at any depth, so the two
66
+ spellings differed by everything — one worked and the one this
67
+ repository's own config used was inert. `.pyc` files were scanned as
68
+ source the whole time, and CI caught it only because a compiled test
69
+ fixture tripped a secrets rule.
70
+
71
+ An inert exclude pattern is the worst kind of configuration defect: it
72
+ reads as intent, it never errors, and the only symptom is findings the
73
+ operator believed they had excluded.
61
74
  """
62
75
  if pat.endswith("/"):
63
- return rel.startswith(pat) or f"/{pat}" in f"/{rel}/"
76
+ bare = pat[3:] if pat.startswith("**/") else pat
77
+ if not bare: # a lone `**/` would otherwise exclude the entire tree
78
+ return False
79
+ return rel.startswith(bare) or f"/{bare}" in f"/{rel}/"
64
80
  bare = pat[3:] if pat.startswith("**/") else pat
65
81
  return fnmatch.fnmatch(rel, pat) or fnmatch.fnmatch(rel, bare) or fnmatch.fnmatch(name, bare)
66
82
 
@@ -97,8 +113,9 @@ def loc_under(
97
113
  excludes: Iterable[str],
98
114
  test_patterns: Iterable[str] = (),
99
115
  skip: Iterable[Path] = (),
100
- ) -> tuple[int, int]:
101
- """Non-blank in-scope lines, split into (primary, test).
116
+ docs_patterns: Iterable[str] = (),
117
+ ) -> tuple[int, int, int]:
118
+ """Non-blank in-scope lines, split into (primary, test, docs).
102
119
 
103
120
  The split exists because the score's denominator has to move with its
104
121
  numerator. Scoring primary-tree findings over a LOC count that included the
@@ -106,6 +123,12 @@ def loc_under(
106
123
  tested — the same numerator/denominator mismatch that `exclude_patterns`
107
124
  already caused once, arriving by a different door.
108
125
 
126
+ Documentation is split for the same reason, and was not: its *findings*
127
+ move to their own axis and out of the score, while its *lines* stayed in
128
+ the primary denominator. FastAPI carries 7,160 lines of `docs/en/data/`
129
+ — translator and contributor lists — diluting the count its code is
130
+ graded against. Third occurrence of one mismatch.
131
+
109
132
  `skip` names the run's own artifacts — the report, the baseline, the
110
133
  suppressions file. Dropping their *findings* without dropping their
111
134
  *lines* is that same mismatch a third time: an audit that wrote a
@@ -114,9 +137,11 @@ def loc_under(
114
137
  unchanged repository returned 0.00 and 4.25.
115
138
  """
116
139
  test_patterns = tuple(test_patterns)
140
+ docs_patterns = tuple(docs_patterns)
117
141
  skip = {p.resolve() for p in skip}
118
142
  primary = 0
119
143
  test = 0
144
+ docs = 0
120
145
  # `rglob` on a file yields nothing, so a single-file audit reported zero
121
146
  # lines — and a zero denominator is not normalised at all, so the grade
122
147
  # became the raw subtotal. Auditing one file is supported; it should be
@@ -138,6 +163,8 @@ def loc_under(
138
163
  lines = sum(1 for line in text.splitlines() if line.strip())
139
164
  if test_patterns and is_test_path(path, root, test_patterns):
140
165
  test += lines
166
+ elif docs_patterns and is_test_path(path, root, docs_patterns):
167
+ docs += lines
141
168
  else:
142
169
  primary += lines
143
- return primary, test
170
+ return primary, test, docs
@@ -5,6 +5,41 @@ declares Security a `DELEGATED` pillar naming this tool, and reports it as
5
5
  `NotApplicable` so a reader never mistakes silence for safety. This module is
6
6
  the other half: the thing that makes that entry unnecessary.
7
7
 
8
+ **Changing this document's shape is a two-repository release, in order.**
9
+
10
+ MA refuses an unknown `schema` or `schema_version` outright and reports no
11
+ delegated pillar rather than a partial one — deliberately, because "the schema
12
+ string is the producer's promise about the shape, and guessing past it is how
13
+ a consumer starts reporting fields that mean something different." That is the
14
+ safe failure and a *silent* one.
15
+
16
+ So a schema bump ships in this sequence and no other:
17
+
18
+ 1. specify the new shape and send it to MA;
19
+ 2. **MA lands its reader first**, accepting the old version and the new;
20
+ 3. only then does this tool emit the new version.
21
+
22
+ Emitting first leaves the pillar unmeasured for the entire window between the
23
+ two releases, with nothing on either side reporting why. MA will not
24
+ pre-accept an unspecified shape, which is correct for the same reason this
25
+ rule exists. Agreed with the MA maintainer 2026-09-11; see D18, D19 and MA's
26
+ D155.
27
+
28
+ **v2 is live and was cut this way.** It is v1 plus one field, `scoring_model`
29
+ — every v1 key keeps its name, type and meaning. MA's reader accepts v1 and
30
+ v2 and keys its trend on `scoring_model` when present, falling back to the
31
+ producer version when absent, so its reader could land before this tool
32
+ emitted anything and no document was ever refused.
33
+
34
+ **`scoring_model` is load-bearing and is not ours alone.** MA keys trend
35
+ comparability on it, because a delegated pillar can change its scoring model
36
+ without changing its schema — which is exactly what D16 and D17 did. A wrong
37
+ value here silently splices two scoring models into one trend and presents it
38
+ as knowledge, so `tests/integration/test_scoring_drift.py` digests the weights,
39
+ the bands and the real `score()` output and fails when the model moves without
40
+ the integer. `producer.version` remains pinned to `__version__` for the same
41
+ reason, since it is still the key for any v1 document.
42
+
8
43
  **Everything structural here is MA's and is reused deliberately.** The scope
9
44
  vocabulary, the two-axis split, the posture matrix and its thresholds all come
10
45
  from `_pillars.py`. Two tools reporting "level 3" or "healthy" about the same
@@ -41,7 +76,7 @@ from typing import Any
41
76
  from secure_code_audit import __version__
42
77
  from secure_code_audit.practice import PracticeLevel
43
78
  from secure_code_audit.scanner_status import CoverageReport, CoverageStatus
44
- from secure_code_audit.scoring import AxisReport, ScoreReport, Verdict
79
+ from secure_code_audit.scoring import SCORING_MODEL, AxisReport, ScoreReport, Verdict
45
80
 
46
81
  #: MA's matrix thresholds, imported by value because the two tools must agree
47
82
  #: on where the cells fall. `_pillars.py` holds the originals.
@@ -151,8 +186,24 @@ def to_dict(pillar: SecurityPillar) -> dict[str, Any]:
151
186
  """The document MA reads. Both axes present, their mean absent."""
152
187
  return {
153
188
  "schema": "secure-code-agent/security-pillar",
154
- "schema_version": 1,
189
+ # v2 = v1 plus `scoring_model`. Nothing else moved: every v1 key keeps
190
+ # its name, its type and its meaning.
191
+ "schema_version": 2,
155
192
  "producer": {"tool": "secure-code-agent", "version": __version__},
193
+ # Which scoring model produced `condition`, for a consumer keeping a
194
+ # trend. Deliberately top-level rather than inside `producer`:
195
+ # `producer` says *who*, this says *what model*, and a consumer keying
196
+ # on it should not have to reach through an identity block.
197
+ #
198
+ # An integer, not a version string, because the only question is "same
199
+ # or different" — an integer cannot be padded, or compared as text, or
200
+ # read as ordering that means more than it does.
201
+ #
202
+ # MA keyed this on our release version before, which is correct and far
203
+ # too broad: a new series opened on every release, including ones that
204
+ # changed no scoring, and a signal that fires constantly teaches people
205
+ # to ignore it.
206
+ "scoring_model": SCORING_MODEL,
156
207
  "generated": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
157
208
  "pillar": "security",
158
209
  "scope": SCOPE,
@@ -3,8 +3,8 @@
3
3
  Implements the model documented in docs/scoring.md:
4
4
  · finding_score = severity × confidence × category × top25_bonus
5
5
  · category_subtotal = Σ finding_score per category
6
- · category_normalized = subtotal / sqrt(LOC / 1000)
7
- · category_grade = clamp(5.0 - (normalized × 0.5), 0, 5)
6
+ · category_normalized = subtotal / (LOC / 1000)
7
+ · category_grade = clamp(5.0 - (normalized × 1.5), 0, 5)
8
8
  · overall = min(category_grades)
9
9
  """
10
10
 
@@ -48,6 +48,43 @@ CATEGORY_WEIGHT: dict[Category, float] = {
48
48
 
49
49
  CWE_TOP25_BONUS = 1.25
50
50
 
51
+ #: Which scoring model produced a number, for consumers that keep a trend.
52
+ #:
53
+ #: `maintainability-agent` stores this tool's `condition` in its scan history
54
+ #: and has to know when two readings are comparable. A delegated pillar can
55
+ #: change its scoring model **without changing its schema** — same shape, same
56
+ #: fields, a different number for the same repository — which is exactly what
57
+ #: D16 and D17 did. MA previously keyed on our release version, which is
58
+ #: correct but far too broad: it opened a new series on every release,
59
+ #: including ones that changed no scoring, and a signal that fires constantly
60
+ #: teaches people to ignore it.
61
+ #:
62
+ #: **Bump this when a repository's condition could differ for a reason that is
63
+ #: not the repository.** Concretely: the normalizer, the grade slope, any
64
+ #: weight table, the letter bands, the rank discount, `COUNT_LIKE_CATEGORIES`,
65
+ #: the scanner floor, or the built-in rule profile (D10).
66
+ #:
67
+ #: **Do not bump for** documentation, adapters, CLI flags, output formats,
68
+ #: performance, or a parser fix that does not change which findings are
69
+ #: produced.
70
+ #:
71
+ #: A new *rule* does bump it. That was the arguable case and it resolves
72
+ #: against intuition: a repository containing `yaml.unsafe_load` scores lower
73
+ #: the day that rule ships, with no change to the repository. Adding findings
74
+ #: *is* rescoring, because the score is a function of the finding set, and a
75
+ #: user must not read "we can see more now" as "your code got worse".
76
+ #:
77
+ #: **1 is reserved and is never emitted.** It denotes every release before
78
+ #: this field existed, and those releases do not share one model — the
79
+ #: corroboration merge, the rank discount and D16 all moved the numbers. A v1
80
+ #: document simply omits the field and MA keys those on the release version,
81
+ #: which fragments them correctly. Nothing may back-fill a 1.
82
+ #:
83
+ #: `tests/integration/test_scoring_drift.py` holds this honest: it digests the
84
+ #: weights, the bands and the output of the real `score()` over a fixed
85
+ #: matrix, so changing a constant *or* a formula without bumping this fails.
86
+ SCORING_MODEL = 2
87
+
51
88
 
52
89
  # --- letter-grade boundaries (mirrors maintainability-agent) ---------------
53
90
 
@@ -173,7 +210,7 @@ def category_subtotal(findings: Iterable[Finding], category: Category) -> float:
173
210
 
174
211
  A straight sum measures how many times a pattern matched, and that
175
212
  tracks codebase size times how talkative the scanner is — not how much
176
- risk is in the code. `sqrt(LOC)` was meant to cancel the size half.
213
+ risk is in the code. The normalizer was meant to cancel the size half.
177
214
  Nothing cancelled the other half, and the corpus said so plainly: the
178
215
  worst-first ordering read Python → JavaScript → Go/Ruby → Java, which is
179
216
  the order of Bandit's verbosity, and Django, FastAPI, httpx and Flask all
@@ -216,15 +253,111 @@ def category_subtotal(findings: Iterable[Finding], category: Category) -> float:
216
253
  )
217
254
 
218
255
 
219
- def normalize(subtotal: float, loc_scanned: int) -> float:
220
- """sqrt(LOC/1000) dampener — see docs/scoring.md for the rationale."""
256
+ #: Categories where a finding is a count rather than a rate.
257
+ #:
258
+ #: One committed credential is one committed credential regardless of how
259
+ #: much code surrounds it. Everything else here — injection sinks, weak
260
+ #: crypto calls, unsafe deserialization — genuinely does scale with how much
261
+ #: code there is, and comparing two repositories on those means comparing
262
+ #: rates. See `normalize` for the measurement that settled which is which.
263
+ COUNT_LIKE_CATEGORIES: frozenset[str] = frozenset({"secrets"})
264
+
265
+
266
+ def _category_name(category: Category | str) -> str:
267
+ return category.value if isinstance(category, Category) else str(category)
268
+
269
+
270
+ def normalize(subtotal: float, loc_scanned: int, category: Category | str | None = None) -> float:
271
+ """Weighted findings per thousand lines of scanned code — a density.
272
+
273
+ This was `sqrt(LOC/1000)` and that under-corrected for size, so the
274
+ ranking followed how *big* a repository is rather than how much is wrong
275
+ with it. Measured across the examined corpus:
276
+
277
+ repo LOC weighted findings/kLOC sqrt-normalized
278
+ django 144,473 1.51 18.18 <- ranked worst
279
+ flask 7,841 3.35 9.38
280
+ fastapi 23,764 1.60 7.81
281
+
282
+ Flask carries **2.2x Django's finding density** and normalised at half
283
+ the value. Django sat fifth by density and first by penalty. Spearman
284
+ correlation of grade against size was -0.37 while correlation of density
285
+ against size was +0.12: the number was tracking the wrong variable.
286
+
287
+ Straight density fixes the ordering: re-measured over the same corpus
288
+ after the change, Spearman(LOC, grade) is +0.02, and the worst-ranked
289
+ repository is the densest one rather than the largest one. The slope
290
+ moves with it — see `category_grade` — because the two only make sense
291
+ together.
292
+
293
+ **`secrets` is not a density, and D17 is why.** Adding
294
+ vulnerable-by-design anchors to the corpus exposed the failure directly:
295
+ OWASP Juice Shop carries four hardcoded API keys and three private keys
296
+ and graded **B+**, because 115,340 lines of surrounding code divided
297
+ seven committed credentials down to nothing. Meanwhile Flask, with no
298
+ secrets at all, graded F. A committed private key is one committed
299
+ private key whether the repository is a thousand lines or a million; it
300
+ is a count, not a rate, and dividing it by size is how a training
301
+ application built to be insecure outscored a well-run library.
302
+
303
+ So `secrets` normalizes by `sqrt(LOC/1000)` instead. Not by nothing: a
304
+ larger codebase genuinely does carry more configuration surface, and an
305
+ absolute count made Django fail on two low-confidence hits. Measured over
306
+ the fourteen examined repositories, against whether a repository is
307
+ maintained or written to be vulnerable:
308
+
309
+ variant AUC separation Spearman(LOC, grade)
310
+ linear everywhere (D16) 0.80 -3.76 +0.14
311
+ sqrt everywhere (pre-D16) 0.91 -0.58 -0.32
312
+ linear; secrets absolute 0.90 +0.00 -0.24
313
+ linear; secrets sqrt 1.00 +0.65 -0.01 <-
314
+
315
+ AUC is the probability that a maintained repository outscores a
316
+ vulnerable-by-design one. Negative separation means the populations
317
+ overlap and *no* band table can tell them apart — which is what blocked
318
+ D5's band edges for as long as the corpus had no bad end in it.
319
+ """
221
320
  if loc_scanned <= 0:
222
321
  return subtotal
223
- return subtotal / math.sqrt(max(loc_scanned, 1) / 1000)
322
+ per_kloc = max(loc_scanned, 1) / 1000
323
+ if category is not None and _category_name(category) in COUNT_LIKE_CATEGORIES:
324
+ return subtotal / math.sqrt(per_kloc)
325
+ return subtotal / per_kloc
326
+
327
+
328
+ #: Grade points lost per normalized weighted finding.
329
+ #:
330
+ #: The slope only rescales — it cannot reorder anything — so it is chosen
331
+ #: against two things the ordering does not fix: where the median of
332
+ #: well-maintained code lands, and how much of the corpus clamps at 0.0 and
333
+ #: loses its tail.
334
+ #:
335
+ #: 1.3 is the largest slope that keeps the maintained-corpus median inside the
336
+ #: B band [3.00, 3.50) *and* keeps the two populations from touching. Measured
337
+ #: across the range, holding the D17 normalizer fixed:
338
+ #:
339
+ #: slope median (maintained) AUC separation clamped at 0
340
+ #: 1.2 3.53 (B+) 1.00 +0.98 4
341
+ #: 1.3 3.41 (B) 1.00 +0.65 4 <- adopted
342
+ #: 1.4 3.28 (B) 1.00 +0.31 4
343
+ #: 1.5 3.16 (B) 0.95 +0.00 5
344
+ #:
345
+ #: At 1.5 a maintained repository joins the four vulnerable-by-design ones at
346
+ #: the clamp, the populations touch, and AUC falls. 1.3 has the widest margin
347
+ #: of the slopes that land the median in B.
348
+ #:
349
+ #: The cost, stated: a large repository with a handful of serious findings
350
+ #: still scores better than a small noisy one. A 135,841-line control carrying
351
+ #: SQL injection, `shell=True`, `pickle.loads`, MD5 and `eval` grades in the
352
+ #: A band. What protects that repository is the default `fail_on_new` gate,
353
+ #: which fails it outright, and the work order, which puts all seven findings
354
+ #: in §FIX. A grade is for comparing and for trend; it was never the thing
355
+ #: that catches a vulnerability. See D16 and D17.
356
+ GRADE_SLOPE = 1.3
224
357
 
225
358
 
226
359
  def category_grade(normalized: float) -> float:
227
- return max(0.0, min(5.0, 5.0 - (normalized * 0.5)))
360
+ return max(0.0, min(5.0, 5.0 - (normalized * GRADE_SLOPE)))
228
361
 
229
362
 
230
363
  # --- overall score ---------------------------------------------------------
@@ -520,7 +653,7 @@ def score(
520
653
  per_category[cat] = None
521
654
  continue
522
655
  subtotal = category_subtotal(findings, cat)
523
- per_category[cat] = category_grade(normalize(subtotal, loc_scanned))
656
+ per_category[cat] = category_grade(normalize(subtotal, loc_scanned, cat))
524
657
 
525
658
  for f in findings:
526
659
  if not f.suppressed:
@@ -691,7 +824,26 @@ def _gate_fail_on_new(
691
824
  ]
692
825
  if new_findings:
693
826
  tripped.append("fail_on_new")
694
- reasons.append(f"{len(new_findings)} new finding(s) since baseline")
827
+ # What "new" means depends on whether there is anything to be new
828
+ # *against*. Saying "since baseline" when no baseline exists claims
829
+ # these findings appeared after one, which is the opposite of the
830
+ # truth on a first run.
831
+ baseline_state = gate_config.get("_baseline_state")
832
+ if baseline_state == "absent":
833
+ reasons.append(
834
+ f"{len(new_findings)} finding(s), and no baseline exists yet — on a first "
835
+ f"run everything is new because there is nothing to compare against. "
836
+ f"Work the order, then re-run with --bump-baseline to accept what is "
837
+ f"left and gate on regressions from there."
838
+ )
839
+ elif baseline_state == "unreadable":
840
+ reasons.append(
841
+ f"{len(new_findings)} finding(s) read as new because the baseline file "
842
+ f"could not be parsed. Fix or delete it — a broken baseline silently "
843
+ f"turns an established repository back into a first run."
844
+ )
845
+ else:
846
+ reasons.append(f"{len(new_findings)} new finding(s) since baseline")
695
847
 
696
848
 
697
849
  def _gate_min_score(
@@ -95,8 +95,19 @@ class StandardsEntry:
95
95
  _MAP: dict[tuple[str, str], StandardsEntry] = {
96
96
  # ----- Bandit ----------------------------------------------------------
97
97
  # Source: https://bandit.readthedocs.io/en/latest/plugins/index.html
98
+ # B102 read CWE-78 — *OS* command injection, the shell-injection weakness
99
+ # that B602/B603/B605/B607 cover. `exec()` does not invoke a shell; it
100
+ # compiles and runs Python. The mislabel travelled: into the OWASP
101
+ # mapping, into SARIF, into the work order, and into the Top-25 bonus,
102
+ # which CWE-78 carries and the true weakness does not directly.
103
+ #
104
+ # It also broke corroboration, which is how it was found. Bandit's B102
105
+ # and our own `sca.python.eval` fire on the same `exec(compile(...))`
106
+ # line in Flask's `config.py`; `_same_weakness` merges across scanners on
107
+ # a shared CWE, CWE-78 and CWE-95 are not shared, so one defect scored
108
+ # twice. Flask carried four such pairs and graded F partly on doubles.
98
109
  ("bandit", "B102"): StandardsEntry(
99
- canonical_cwe="CWE-78",
110
+ canonical_cwe="CWE-95",
100
111
  owasp_top10="A03",
101
112
  asvs_section="V5.3.8",
102
113
  nist_ssdf="PW.5.1",
@@ -106,6 +117,20 @@ _MAP: dict[tuple[str, str], StandardsEntry] = {
106
117
  short_desc="Use of exec() — arbitrary code execution risk.",
107
118
  fix_hint="Eliminate exec() entirely. If dynamic dispatch is required, use a typed registry / function map.",
108
119
  ),
120
+ # B307 (`eval`) had no curated entry, so `_make_finding` fell back to the
121
+ # CWE the scanner reports — and Bandit files `eval` under CWE-78 too. Same
122
+ # weakness as B102, same fix, and curating it is what stops the fallback.
123
+ ("bandit", "B307"): StandardsEntry(
124
+ canonical_cwe="CWE-95",
125
+ owasp_top10="A03",
126
+ asvs_section="V5.2.4",
127
+ nist_ssdf="PW.5.1",
128
+ category=Category.CODE_VULNERABILITIES,
129
+ severity=Severity.HIGH,
130
+ confidence=Confidence.MEDIUM,
131
+ short_desc="Use of eval() — arbitrary code execution risk.",
132
+ fix_hint="Use ast.literal_eval for data. For dispatch, use a dict of callables rather than evaluating a name.",
133
+ ),
109
134
  ("bandit", "B301"): StandardsEntry(
110
135
  canonical_cwe="CWE-502",
111
136
  owasp_top10="A08",
@@ -494,9 +519,41 @@ def lookup(scanner: str, rule_id: str) -> StandardsEntry | None:
494
519
  return _MAP.get((scanner.lower(), "*"))
495
520
 
496
521
 
522
+ #: Child CWEs this project maps to, and the Top-25 entry each is a ChildOf.
523
+ #:
524
+ #: The Top-25 list names classes, and a scanner names the specific weakness
525
+ #: inside one. CWE-95 ("Eval Injection") is a documented ChildOf CWE-94
526
+ #: ("Improper Control of Generation of Code"), which is on the list — so a
527
+ #: confirmed eval injection *is* a Top-25 weakness, and a membership test
528
+ #: that only compares strings says it is not.
529
+ #:
530
+ #: This is deliberately a hand-checked handful rather than an imported CWE
531
+ #: hierarchy. Every entry is a relationship stated in the MITRE definition of
532
+ #: the child, and each is used by a rule this project actually maps. Adding a
533
+ #: parent here widens the 1.25x bonus, so it is a decision, not a lookup.
534
+ _TOP25_PARENT: dict[str, str] = {
535
+ # cwe.mitre.org/data/definitions/95.html — ChildOf 94
536
+ "CWE-95": "CWE-94",
537
+ # cwe.mitre.org/data/definitions/77.html is itself on the list; 78 is the
538
+ # OS-command child and is also listed, so neither needs an entry here.
539
+ }
540
+
541
+
497
542
  def is_top25(canonical_cwe: str | None) -> bool:
498
- """Is this CWE on the MITRE Top 25 (2025) list? Used for scoring boost."""
499
- return canonical_cwe is not None and canonical_cwe in CWE_TOP25_2025
543
+ """Is this CWE on the MITRE Top 25 (2025) list, directly or as a child?
544
+
545
+ Correcting Bandit's B102/B307 from CWE-78 to CWE-95 was right on the
546
+ weakness and would have quietly removed the Top-25 bonus from every
547
+ eval/exec finding in the corpus — CWE-78 is on the list and CWE-95 is
548
+ not. Losing the bonus for the *reason* "we now describe the weakness
549
+ accurately" is the wrong trade, and CWE-95 is a child of CWE-94, which
550
+ is listed. So membership follows the relationship.
551
+ """
552
+ if canonical_cwe is None:
553
+ return False
554
+ if canonical_cwe in CWE_TOP25_2025:
555
+ return True
556
+ return _TOP25_PARENT.get(canonical_cwe, "") in CWE_TOP25_2025
500
557
 
501
558
 
502
559
  def cwe_url(canonical_cwe: str) -> str:
@@ -29,7 +29,7 @@ import enum
29
29
  import re
30
30
  from collections.abc import Iterable
31
31
 
32
- from secure_code_audit.findings import Confidence, Finding, Severity
32
+ from secure_code_audit.findings import Category, Confidence, Finding, Severity
33
33
 
34
34
 
35
35
  class Tier(enum.Enum):
@@ -104,6 +104,123 @@ def _looks_like_a_credential(message: str) -> bool:
104
104
  return len(value) >= 12 and any(c.isdigit() for c in value) and any(c.isupper() for c in value)
105
105
 
106
106
 
107
+ #: A credential *reference* — a name standing in for a value that is not here.
108
+ #:
109
+ #: Shell (`$TOKEN`, `${TOKEN}`), GitHub Actions (`${{ secrets.X }}`,
110
+ #: `${{ env.X }}`), Windows (`%TOKEN%`), and the ordinary code spellings.
111
+ _VARIABLE_REFERENCE = re.compile(
112
+ r"""
113
+ \$\{\{\s*(?:secrets|env|vars)\. # ${{ secrets.NAME }}
114
+ | \$\{[A-Za-z_][A-Za-z0-9_]* # ${NAME}
115
+ | \$[A-Za-z_][A-Za-z0-9_]* # $NAME
116
+ | %[A-Za-z_][A-Za-z0-9_]*% # %NAME%
117
+ | os\.environ | os\.getenv # Python
118
+ | process\.env\. # Node
119
+ | ENV\[ # Ruby
120
+ """,
121
+ re.VERBOSE,
122
+ )
123
+
124
+ #: A URL, stripped before looking for a literal credential.
125
+ #:
126
+ #: This is not cosmetic. The first version of the literal test matched
127
+ #: `//sonarcloud` inside `https://sonarcloud.io/api` — twelve characters of
128
+ #: `[A-Za-z0-9+/_-]`, because `/` and `+` are base64 alphabet and a URL is
129
+ #: full of them. Every `curl` line has a URL on it, so the demotion never
130
+ #: fired on the exact case it was written for.
131
+ _URL = re.compile(r"\bhttps?://\S+", re.IGNORECASE)
132
+
133
+ #: A literal that could itself be the credential, sitting on the same line.
134
+ _LITERAL_RUN = re.compile(r"[A-Za-z0-9+/_\-]{12,}")
135
+
136
+
137
+ def _contains_a_literal_credential(text: str) -> bool:
138
+ """Twelve-plus characters with a digit and an uppercase letter.
139
+
140
+ Deliberately the same dull test `_looks_like_a_credential` applies to
141
+ Bandit's quoted values — same shape, same reasons, and it keeps the two
142
+ heuristics from drifting into disagreement about what a secret looks
143
+ like.
144
+
145
+ **Known miss, stated:** an all-lowercase hex token such as
146
+ `a3f9c2b1d4e5` has no uppercase letter and is not caught. That costs a
147
+ demotion from FIX to REVIEW, and only on a line that *also* carries a
148
+ variable reference, which is an odd thing to write. The finding still
149
+ appears in the work order, still appears in the report, and still
150
+ escalates through `fail_on_category`. The error falls in the safe
151
+ direction.
152
+ """
153
+ for run in _LITERAL_RUN.findall(_URL.sub(" ", text)):
154
+ if any(c.isdigit() for c in run) and any(c.isupper() for c in run):
155
+ return True
156
+ return False
157
+
158
+
159
+ def _line_of(finding: Finding) -> str | None:
160
+ """Read back the source line a secrets finding points at.
161
+
162
+ Gitleaks is run with `--redact`, so the matched text never reaches the
163
+ report — the message reads `curl -sS -u REDACTED`, and the variable name
164
+ is precisely what was redacted. The only way to tell a reference from a
165
+ value is to look at the line, the way `verify._is_silenced` does.
166
+
167
+ Returns None whenever the line cannot be read with confidence: the file
168
+ is gone, the path is a directory, the line number is out of range, or the
169
+ finding came out of git history and the working tree has moved on. Every
170
+ one of those must leave the finding where it was.
171
+ """
172
+ path = finding.file_path
173
+ if path is None or finding.line_start is None or finding.line_start < 1:
174
+ return None
175
+ try:
176
+ if not path.is_file():
177
+ return None
178
+ with path.open("r", encoding="utf-8", errors="replace") as handle:
179
+ for number, line in enumerate(handle, 1):
180
+ if number == finding.line_start:
181
+ return line
182
+ if number > finding.line_start:
183
+ break
184
+ except OSError:
185
+ return None
186
+ return None
187
+
188
+
189
+ def _is_a_reference_not_a_value(finding: Finding) -> bool:
190
+ """Is the flagged credential a variable name rather than a credential?
191
+
192
+ `gitleaks.curl-auth-user` fires CRITICAL on
193
+ `curl -sS -u "$SONAR_TOKEN:"` in a GitHub Actions workflow. No credential
194
+ is present; `$SONAR_TOKEN` is how you write *not* putting one there, and
195
+ the idiom is the recommended one. Reported independently on two
196
+ repositories on the same day, five occurrences on one of them.
197
+
198
+ This matters more than it used to. D17 made `secrets` count-like — it
199
+ normalizes by `sqrt(LOC/1000)` rather than by size — so a single false
200
+ critical now costs roughly two grade points on a 100k-line repository
201
+ where it previously cost two tenths. Raising the weight of a category
202
+ raises the cost of being wrong in it, and this is the corresponding
203
+ precision work.
204
+
205
+ Conservative in both directions. It demotes to REVIEW, never to ACCEPT:
206
+ the finding stays in the work order, stays in the report, and still
207
+ escalates through `fail_on_category: [secrets]` from any axis. And it
208
+ declines to act unless it can read the line and finds no literal token on
209
+ it, so `curl -u "$USER:hunter2Passw0rd"` is untouched.
210
+ """
211
+ if finding.category is not Category.SECRETS:
212
+ return False
213
+ line = _line_of(finding)
214
+ if line is None:
215
+ return False
216
+ if not _VARIABLE_REFERENCE.search(line):
217
+ return False
218
+ # A reference and a literal on one line is still a leak. Strip the
219
+ # references first so their own names cannot satisfy the literal test.
220
+ without_references = _VARIABLE_REFERENCE.sub(" ", line)
221
+ return not _contains_a_literal_credential(without_references)
222
+
223
+
107
224
  def tier_of(finding: Finding, axis: str = "primary") -> Tier:
108
225
  """Classify one finding.
109
226
 
@@ -118,6 +235,8 @@ def tier_of(finding: Finding, axis: str = "primary") -> Tier:
118
235
  if _looks_like_a_credential(finding.message):
119
236
  return Tier.FIX
120
237
  return Tier.REVIEW
238
+ if _is_a_reference_not_a_value(finding):
239
+ return Tier.REVIEW
121
240
  if finding.confidence is Confidence.LOW:
122
241
  return Tier.REVIEW
123
242
  return Tier.FIX
@@ -128,6 +247,15 @@ def reason_for(finding: Finding) -> str | None:
128
247
  measured = LOW_PRECISION.get(finding.rule_id)
129
248
  if measured:
130
249
  return measured
250
+ if _is_a_reference_not_a_value(finding):
251
+ return (
252
+ "The line holds a variable reference, not a credential — a name "
253
+ "standing in for a value kept elsewhere, which is the recommended "
254
+ "way to write this. Checked by reading the line back, because "
255
+ "gitleaks runs with --redact and the redacted text is the "
256
+ "variable name itself. Confirm the value really is injected at "
257
+ "runtime, then suppress it."
258
+ )
131
259
  if finding.confidence is Confidence.LOW:
132
260
  return (
133
261
  f"{finding.scanner} reported this at low confidence — it is "