secure-code-agent 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {secure_code_agent-0.5.0/src/secure_code_agent.egg-info → secure_code_agent-0.6.0}/PKG-INFO +85 -3
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/README.md +84 -2
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0/src/secure_code_agent.egg-info}/PKG-INFO +85 -3
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/SOURCES.txt +3 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/__init__.py +1 -1
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/cli.py +343 -34
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/config.py +49 -1
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/data/semgrep-offline.yaml +27 -7
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/findings.py +111 -2
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/git_tools.py +41 -8
- secure_code_agent-0.6.0/src/secure_code_audit/history.py +160 -0
- secure_code_agent-0.6.0/src/secure_code_audit/remediation.py +303 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/renderers.py +69 -13
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/ruleset.py +1 -1
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/bandit_scanner.py +21 -1
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/base.py +74 -6
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/builtin_rules.py +7 -1
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/gitleaks_scanner.py +68 -16
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/gosec_scanner.py +5 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scoring.py +219 -49
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/standards.py +109 -0
- secure_code_agent-0.6.0/src/secure_code_audit/triage.py +155 -0
- secure_code_agent-0.6.0/src/secure_code_audit/verify.py +295 -0
- secure_code_agent-0.5.0/src/secure_code_audit/remediation.py +0 -168
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/LICENSE +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/pyproject.toml +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/setup.cfg +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/dependency_links.txt +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/entry_points.txt +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/requires.txt +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/top_level.txt +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/baseline.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/instructions.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/pillar.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/practice.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/sarif.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanner_status.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/__init__.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/checkov_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/floor.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/hadolint_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/njsscan_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/npm_audit_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/osv_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/pip_audit_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/rubocop_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/scorecard_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/semgrep_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/trivy_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/trufflehog_scanner.py +0 -0
- {secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_audit/suppressions.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: secure-code-agent
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
|
|
5
5
|
Author: Marshall Guillory
|
|
6
6
|
License: MIT
|
|
@@ -119,17 +119,98 @@ This is a constrained task, not a refactor.
|
|
|
119
119
|
10. Keep the patch small. If you find yourself rewriting a function
|
|
120
120
|
rather than patching it, stop and report the structural issue.
|
|
121
121
|
|
|
122
|
-
## §
|
|
122
|
+
## §FIX — patch these
|
|
123
123
|
...
|
|
124
124
|
```
|
|
125
125
|
|
|
126
|
-
Hand the
|
|
126
|
+
Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
|
|
127
|
+
|
|
128
|
+
**It is written on every run**, alongside the report — no flag required. The
|
|
129
|
+
report describes your code; the work order changes it. Set
|
|
130
|
+
`outputs.prompt_path` to `null` if you do not want one.
|
|
131
|
+
|
|
132
|
+
### Findings are tiered, because not all of them deserve equal attention
|
|
133
|
+
|
|
134
|
+
A flat list gives a `shell=True` command injection and a
|
|
135
|
+
`PASSWORD_FIELD = "password"` name-match the same billing, so an agent
|
|
136
|
+
working top to bottom spends its care on noise.
|
|
137
|
+
|
|
138
|
+
| tier | what it means | what the agent does |
|
|
139
|
+
| --- | --- | --- |
|
|
140
|
+
| **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
|
|
141
|
+
| **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
|
|
142
|
+
| **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
|
|
143
|
+
|
|
144
|
+
A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
|
|
145
|
+
useful hits out of 22 across the calibration corpus — empty defaults, field
|
|
146
|
+
names, `django-insecure-` — and still caught a planted hardcoded credential,
|
|
147
|
+
so the *value* is judged as well as the rule.
|
|
148
|
+
|
|
149
|
+
## Proving the work order actually helped
|
|
150
|
+
|
|
151
|
+
A work order nobody checks is a suggestion. After the agent has worked:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
secure-code-agent . --verify-against secure-code-report.json
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
This re-audits and reports what was **fixed**, what is **still open**, what
|
|
158
|
+
was **silenced rather than repaired**, and what this work **introduced**. It
|
|
159
|
+
exits nonzero unless the run passes, because a verification step that always
|
|
160
|
+
passes verifies nothing.
|
|
161
|
+
|
|
162
|
+
```
|
|
163
|
+
work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
|
|
164
|
+
SILENCED B602 app.py:4
|
|
165
|
+
SILENCED B324 app.py:7
|
|
166
|
+
! 2 finding(s) disappeared without the code being repaired — a suppression
|
|
167
|
+
entry now covers them, or the reported line gained an inline marker such
|
|
168
|
+
as `# nosec`.
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
That last case is the one worth having. **Lint disable** is in the
|
|
172
|
+
anti-pattern table above, hard constraint 6 forbids it, and forbidding is
|
|
173
|
+
not detecting — Bandit honours `# nosec` itself, so a silenced finding
|
|
174
|
+
simply stops arriving and reads as fixed. Verification reads the source back
|
|
175
|
+
and calls it what it is.
|
|
176
|
+
|
|
177
|
+
Findings in the test tree and documentation are **reported but not
|
|
178
|
+
required**: §ACCEPT tells the agent not to patch them, so demanding them
|
|
179
|
+
back would make any repository with fixtures impossible to verify.
|
|
180
|
+
|
|
181
|
+
## Trend, which is what a score is actually for
|
|
182
|
+
|
|
183
|
+
A single B− tells you little. A B− that was an A− three runs ago tells you
|
|
184
|
+
something happened. Every run appends one line to
|
|
185
|
+
`.secure-code/history.jsonl` and prints the movement:
|
|
186
|
+
|
|
187
|
+
```
|
|
188
|
+
trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
A run whose coverage was too thin to grade records `null` and is skipped
|
|
192
|
+
rather than plotted as a collapse to zero — the same rule the rest of the
|
|
193
|
+
tool follows: absence of evidence is not a bad grade.
|
|
127
194
|
|
|
128
195
|
## Standards anchored, not invented
|
|
129
196
|
|
|
130
197
|
Known rules map to fields from five public standards. Unmapped and scanner-control
|
|
131
198
|
findings retain null standards fields rather than receiving invented mappings.
|
|
132
199
|
|
|
200
|
+
A CWE comes from one of three places, in descending authority: an adapter's
|
|
201
|
+
explicit override, the curated map in `standards.py`, then whatever the
|
|
202
|
+
scanner itself declared. That last source was being discarded — Bandit
|
|
203
|
+
publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
|
|
204
|
+
were dropped on the floor, leaving **14% of real findings with any CWE at
|
|
205
|
+
all** against a corpus measurement. Reading them takes it to 100%, and the
|
|
206
|
+
OWASP category is derived from the CWE using OWASP's own published
|
|
207
|
+
category-to-CWE lists where the curated map has none.
|
|
208
|
+
|
|
209
|
+
Derivation stops where the standard does. `CWE-703` — Bandit's classification
|
|
210
|
+
for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
|
|
211
|
+
null. 62% of findings carry an OWASP category and the rest say nothing, which
|
|
212
|
+
is the honest answer.
|
|
213
|
+
|
|
133
214
|
| Source | What we use it for |
|
|
134
215
|
|----------------------------------------------|----------------------------------------------------------|
|
|
135
216
|
| [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
|
|
@@ -431,6 +512,7 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
|
|
|
431
512
|
- [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
|
|
432
513
|
- [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
|
|
433
514
|
- [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
|
|
515
|
+
- [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
|
|
434
516
|
- [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
|
|
435
517
|
- [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
|
|
436
518
|
|
|
@@ -73,17 +73,98 @@ This is a constrained task, not a refactor.
|
|
|
73
73
|
10. Keep the patch small. If you find yourself rewriting a function
|
|
74
74
|
rather than patching it, stop and report the structural issue.
|
|
75
75
|
|
|
76
|
-
## §
|
|
76
|
+
## §FIX — patch these
|
|
77
77
|
...
|
|
78
78
|
```
|
|
79
79
|
|
|
80
|
-
Hand the
|
|
80
|
+
Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
|
|
81
|
+
|
|
82
|
+
**It is written on every run**, alongside the report — no flag required. The
|
|
83
|
+
report describes your code; the work order changes it. Set
|
|
84
|
+
`outputs.prompt_path` to `null` if you do not want one.
|
|
85
|
+
|
|
86
|
+
### Findings are tiered, because not all of them deserve equal attention
|
|
87
|
+
|
|
88
|
+
A flat list gives a `shell=True` command injection and a
|
|
89
|
+
`PASSWORD_FIELD = "password"` name-match the same billing, so an agent
|
|
90
|
+
working top to bottom spends its care on noise.
|
|
91
|
+
|
|
92
|
+
| tier | what it means | what the agent does |
|
|
93
|
+
| --- | --- | --- |
|
|
94
|
+
| **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
|
|
95
|
+
| **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
|
|
96
|
+
| **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
|
|
97
|
+
|
|
98
|
+
A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
|
|
99
|
+
useful hits out of 22 across the calibration corpus — empty defaults, field
|
|
100
|
+
names, `django-insecure-` — and still caught a planted hardcoded credential,
|
|
101
|
+
so the *value* is judged as well as the rule.
|
|
102
|
+
|
|
103
|
+
## Proving the work order actually helped
|
|
104
|
+
|
|
105
|
+
A work order nobody checks is a suggestion. After the agent has worked:
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
secure-code-agent . --verify-against secure-code-report.json
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
This re-audits and reports what was **fixed**, what is **still open**, what
|
|
112
|
+
was **silenced rather than repaired**, and what this work **introduced**. It
|
|
113
|
+
exits nonzero unless the run passes, because a verification step that always
|
|
114
|
+
passes verifies nothing.
|
|
115
|
+
|
|
116
|
+
```
|
|
117
|
+
work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
|
|
118
|
+
SILENCED B602 app.py:4
|
|
119
|
+
SILENCED B324 app.py:7
|
|
120
|
+
! 2 finding(s) disappeared without the code being repaired — a suppression
|
|
121
|
+
entry now covers them, or the reported line gained an inline marker such
|
|
122
|
+
as `# nosec`.
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
That last case is the one worth having. **Lint disable** is in the
|
|
126
|
+
anti-pattern table above, hard constraint 6 forbids it, and forbidding is
|
|
127
|
+
not detecting — Bandit honours `# nosec` itself, so a silenced finding
|
|
128
|
+
simply stops arriving and reads as fixed. Verification reads the source back
|
|
129
|
+
and calls it what it is.
|
|
130
|
+
|
|
131
|
+
Findings in the test tree and documentation are **reported but not
|
|
132
|
+
required**: §ACCEPT tells the agent not to patch them, so demanding them
|
|
133
|
+
back would make any repository with fixtures impossible to verify.
|
|
134
|
+
|
|
135
|
+
## Trend, which is what a score is actually for
|
|
136
|
+
|
|
137
|
+
A single B− tells you little. A B− that was an A− three runs ago tells you
|
|
138
|
+
something happened. Every run appends one line to
|
|
139
|
+
`.secure-code/history.jsonl` and prints the movement:
|
|
140
|
+
|
|
141
|
+
```
|
|
142
|
+
trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
A run whose coverage was too thin to grade records `null` and is skipped
|
|
146
|
+
rather than plotted as a collapse to zero — the same rule the rest of the
|
|
147
|
+
tool follows: absence of evidence is not a bad grade.
|
|
81
148
|
|
|
82
149
|
## Standards anchored, not invented
|
|
83
150
|
|
|
84
151
|
Known rules map to fields from five public standards. Unmapped and scanner-control
|
|
85
152
|
findings retain null standards fields rather than receiving invented mappings.
|
|
86
153
|
|
|
154
|
+
A CWE comes from one of three places, in descending authority: an adapter's
|
|
155
|
+
explicit override, the curated map in `standards.py`, then whatever the
|
|
156
|
+
scanner itself declared. That last source was being discarded — Bandit
|
|
157
|
+
publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
|
|
158
|
+
were dropped on the floor, leaving **14% of real findings with any CWE at
|
|
159
|
+
all** against a corpus measurement. Reading them takes it to 100%, and the
|
|
160
|
+
OWASP category is derived from the CWE using OWASP's own published
|
|
161
|
+
category-to-CWE lists where the curated map has none.
|
|
162
|
+
|
|
163
|
+
Derivation stops where the standard does. `CWE-703` — Bandit's classification
|
|
164
|
+
for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
|
|
165
|
+
null. 62% of findings carry an OWASP category and the rest say nothing, which
|
|
166
|
+
is the honest answer.
|
|
167
|
+
|
|
87
168
|
| Source | What we use it for |
|
|
88
169
|
|----------------------------------------------|----------------------------------------------------------|
|
|
89
170
|
| [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
|
|
@@ -385,6 +466,7 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
|
|
|
385
466
|
- [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
|
|
386
467
|
- [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
|
|
387
468
|
- [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
|
|
469
|
+
- [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
|
|
388
470
|
- [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
|
|
389
471
|
- [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
|
|
390
472
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: secure-code-agent
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
|
|
5
5
|
Author: Marshall Guillory
|
|
6
6
|
License: MIT
|
|
@@ -119,17 +119,98 @@ This is a constrained task, not a refactor.
|
|
|
119
119
|
10. Keep the patch small. If you find yourself rewriting a function
|
|
120
120
|
rather than patching it, stop and report the structural issue.
|
|
121
121
|
|
|
122
|
-
## §
|
|
122
|
+
## §FIX — patch these
|
|
123
123
|
...
|
|
124
124
|
```
|
|
125
125
|
|
|
126
|
-
Hand the
|
|
126
|
+
Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
|
|
127
|
+
|
|
128
|
+
**It is written on every run**, alongside the report — no flag required. The
|
|
129
|
+
report describes your code; the work order changes it. Set
|
|
130
|
+
`outputs.prompt_path` to `null` if you do not want one.
|
|
131
|
+
|
|
132
|
+
### Findings are tiered, because not all of them deserve equal attention
|
|
133
|
+
|
|
134
|
+
A flat list gives a `shell=True` command injection and a
|
|
135
|
+
`PASSWORD_FIELD = "password"` name-match the same billing, so an agent
|
|
136
|
+
working top to bottom spends its care on noise.
|
|
137
|
+
|
|
138
|
+
| tier | what it means | what the agent does |
|
|
139
|
+
| --- | --- | --- |
|
|
140
|
+
| **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
|
|
141
|
+
| **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
|
|
142
|
+
| **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
|
|
143
|
+
|
|
144
|
+
A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
|
|
145
|
+
useful hits out of 22 across the calibration corpus — empty defaults, field
|
|
146
|
+
names, `django-insecure-` — and still caught a planted hardcoded credential,
|
|
147
|
+
so the *value* is judged as well as the rule.
|
|
148
|
+
|
|
149
|
+
## Proving the work order actually helped
|
|
150
|
+
|
|
151
|
+
A work order nobody checks is a suggestion. After the agent has worked:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
secure-code-agent . --verify-against secure-code-report.json
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
This re-audits and reports what was **fixed**, what is **still open**, what
|
|
158
|
+
was **silenced rather than repaired**, and what this work **introduced**. It
|
|
159
|
+
exits nonzero unless the run passes, because a verification step that always
|
|
160
|
+
passes verifies nothing.
|
|
161
|
+
|
|
162
|
+
```
|
|
163
|
+
work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
|
|
164
|
+
SILENCED B602 app.py:4
|
|
165
|
+
SILENCED B324 app.py:7
|
|
166
|
+
! 2 finding(s) disappeared without the code being repaired — a suppression
|
|
167
|
+
entry now covers them, or the reported line gained an inline marker such
|
|
168
|
+
as `# nosec`.
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
That last case is the one worth having. **Lint disable** is in the
|
|
172
|
+
anti-pattern table above, hard constraint 6 forbids it, and forbidding is
|
|
173
|
+
not detecting — Bandit honours `# nosec` itself, so a silenced finding
|
|
174
|
+
simply stops arriving and reads as fixed. Verification reads the source back
|
|
175
|
+
and calls it what it is.
|
|
176
|
+
|
|
177
|
+
Findings in the test tree and documentation are **reported but not
|
|
178
|
+
required**: §ACCEPT tells the agent not to patch them, so demanding them
|
|
179
|
+
back would make any repository with fixtures impossible to verify.
|
|
180
|
+
|
|
181
|
+
## Trend, which is what a score is actually for
|
|
182
|
+
|
|
183
|
+
A single B− tells you little. A B− that was an A− three runs ago tells you
|
|
184
|
+
something happened. Every run appends one line to
|
|
185
|
+
`.secure-code/history.jsonl` and prints the movement:
|
|
186
|
+
|
|
187
|
+
```
|
|
188
|
+
trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
A run whose coverage was too thin to grade records `null` and is skipped
|
|
192
|
+
rather than plotted as a collapse to zero — the same rule the rest of the
|
|
193
|
+
tool follows: absence of evidence is not a bad grade.
|
|
127
194
|
|
|
128
195
|
## Standards anchored, not invented
|
|
129
196
|
|
|
130
197
|
Known rules map to fields from five public standards. Unmapped and scanner-control
|
|
131
198
|
findings retain null standards fields rather than receiving invented mappings.
|
|
132
199
|
|
|
200
|
+
A CWE comes from one of three places, in descending authority: an adapter's
|
|
201
|
+
explicit override, the curated map in `standards.py`, then whatever the
|
|
202
|
+
scanner itself declared. That last source was being discarded — Bandit
|
|
203
|
+
publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
|
|
204
|
+
were dropped on the floor, leaving **14% of real findings with any CWE at
|
|
205
|
+
all** against a corpus measurement. Reading them takes it to 100%, and the
|
|
206
|
+
OWASP category is derived from the CWE using OWASP's own published
|
|
207
|
+
category-to-CWE lists where the curated map has none.
|
|
208
|
+
|
|
209
|
+
Derivation stops where the standard does. `CWE-703` — Bandit's classification
|
|
210
|
+
for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
|
|
211
|
+
null. 62% of findings carry an OWASP category and the rest say nothing, which
|
|
212
|
+
is the honest answer.
|
|
213
|
+
|
|
133
214
|
| Source | What we use it for |
|
|
134
215
|
|----------------------------------------------|----------------------------------------------------------|
|
|
135
216
|
| [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
|
|
@@ -431,6 +512,7 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
|
|
|
431
512
|
- [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
|
|
432
513
|
- [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
|
|
433
514
|
- [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
|
|
515
|
+
- [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
|
|
434
516
|
- [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
|
|
435
517
|
- [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
|
|
436
518
|
|
{secure_code_agent-0.5.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/SOURCES.txt
RENAMED
|
@@ -13,6 +13,7 @@ src/secure_code_audit/cli.py
|
|
|
13
13
|
src/secure_code_audit/config.py
|
|
14
14
|
src/secure_code_audit/findings.py
|
|
15
15
|
src/secure_code_audit/git_tools.py
|
|
16
|
+
src/secure_code_audit/history.py
|
|
16
17
|
src/secure_code_audit/instructions.py
|
|
17
18
|
src/secure_code_audit/pillar.py
|
|
18
19
|
src/secure_code_audit/practice.py
|
|
@@ -24,6 +25,8 @@ src/secure_code_audit/scanner_status.py
|
|
|
24
25
|
src/secure_code_audit/scoring.py
|
|
25
26
|
src/secure_code_audit/standards.py
|
|
26
27
|
src/secure_code_audit/suppressions.py
|
|
28
|
+
src/secure_code_audit/triage.py
|
|
29
|
+
src/secure_code_audit/verify.py
|
|
27
30
|
src/secure_code_audit/data/semgrep-offline.yaml
|
|
28
31
|
src/secure_code_audit/scanners/__init__.py
|
|
29
32
|
src/secure_code_audit/scanners/bandit_scanner.py
|