secure-code-agent 0.4.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {secure_code_agent-0.4.0/src/secure_code_agent.egg-info → secure_code_agent-0.6.0}/PKG-INFO +112 -3
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/README.md +111 -2
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/pyproject.toml +15 -1
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0/src/secure_code_agent.egg-info}/PKG-INFO +112 -3
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/SOURCES.txt +5 -0
- secure_code_agent-0.6.0/src/secure_code_audit/__init__.py +15 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/cli.py +416 -29
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/config.py +85 -1
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/data/semgrep-offline.yaml +27 -7
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/findings.py +111 -2
- secure_code_agent-0.6.0/src/secure_code_audit/git_tools.py +128 -0
- secure_code_agent-0.6.0/src/secure_code_audit/history.py +160 -0
- secure_code_agent-0.6.0/src/secure_code_audit/pillar.py +194 -0
- secure_code_agent-0.6.0/src/secure_code_audit/practice.py +300 -0
- secure_code_agent-0.6.0/src/secure_code_audit/remediation.py +303 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/renderers.py +159 -13
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/ruleset.py +1 -1
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/bandit_scanner.py +21 -1
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/base.py +74 -6
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/builtin_rules.py +7 -1
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/gitleaks_scanner.py +68 -16
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/gosec_scanner.py +5 -0
- secure_code_agent-0.6.0/src/secure_code_audit/scoring.py +712 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/standards.py +109 -0
- secure_code_agent-0.6.0/src/secure_code_audit/triage.py +155 -0
- secure_code_agent-0.6.0/src/secure_code_audit/verify.py +295 -0
- secure_code_agent-0.4.0/src/secure_code_audit/__init__.py +0 -3
- secure_code_agent-0.4.0/src/secure_code_audit/git_tools.py +0 -65
- secure_code_agent-0.4.0/src/secure_code_audit/remediation.py +0 -168
- secure_code_agent-0.4.0/src/secure_code_audit/scoring.py +0 -421
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/LICENSE +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/setup.cfg +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/dependency_links.txt +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/entry_points.txt +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/requires.txt +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/top_level.txt +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/baseline.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/instructions.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/sarif.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanner_status.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/__init__.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/checkov_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/floor.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/hadolint_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/njsscan_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/npm_audit_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/osv_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/pip_audit_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/rubocop_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/scorecard_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/semgrep_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/trivy_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/trufflehog_scanner.py +0 -0
- {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/suppressions.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: secure-code-agent
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
|
|
5
5
|
Author: Marshall Guillory
|
|
6
6
|
License: MIT
|
|
@@ -119,17 +119,98 @@ This is a constrained task, not a refactor.
|
|
|
119
119
|
10. Keep the patch small. If you find yourself rewriting a function
|
|
120
120
|
rather than patching it, stop and report the structural issue.
|
|
121
121
|
|
|
122
|
-
## §
|
|
122
|
+
## §FIX — patch these
|
|
123
123
|
...
|
|
124
124
|
```
|
|
125
125
|
|
|
126
|
-
Hand the
|
|
126
|
+
Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
|
|
127
|
+
|
|
128
|
+
**It is written on every run**, alongside the report — no flag required. The
|
|
129
|
+
report describes your code; the work order changes it. Set
|
|
130
|
+
`outputs.prompt_path` to `null` if you do not want one.
|
|
131
|
+
|
|
132
|
+
### Findings are tiered, because not all of them deserve equal attention
|
|
133
|
+
|
|
134
|
+
A flat list gives a `shell=True` command injection and a
|
|
135
|
+
`PASSWORD_FIELD = "password"` name-match the same billing, so an agent
|
|
136
|
+
working top to bottom spends its care on noise.
|
|
137
|
+
|
|
138
|
+
| tier | what it means | what the agent does |
|
|
139
|
+
| --- | --- | --- |
|
|
140
|
+
| **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
|
|
141
|
+
| **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
|
|
142
|
+
| **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
|
|
143
|
+
|
|
144
|
+
A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
|
|
145
|
+
useful hits out of 22 across the calibration corpus — empty defaults, field
|
|
146
|
+
names, `django-insecure-` — and still caught a planted hardcoded credential,
|
|
147
|
+
so the *value* is judged as well as the rule.
|
|
148
|
+
|
|
149
|
+
## Proving the work order actually helped
|
|
150
|
+
|
|
151
|
+
A work order nobody checks is a suggestion. After the agent has worked:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
secure-code-agent . --verify-against secure-code-report.json
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
This re-audits and reports what was **fixed**, what is **still open**, what
|
|
158
|
+
was **silenced rather than repaired**, and what this work **introduced**. It
|
|
159
|
+
exits nonzero unless the run passes, because a verification step that always
|
|
160
|
+
passes verifies nothing.
|
|
161
|
+
|
|
162
|
+
```
|
|
163
|
+
work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
|
|
164
|
+
SILENCED B602 app.py:4
|
|
165
|
+
SILENCED B324 app.py:7
|
|
166
|
+
! 2 finding(s) disappeared without the code being repaired — a suppression
|
|
167
|
+
entry now covers them, or the reported line gained an inline marker such
|
|
168
|
+
as `# nosec`.
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
That last case is the one worth having. **Lint disable** is in the
|
|
172
|
+
anti-pattern table above, hard constraint 6 forbids it, and forbidding is
|
|
173
|
+
not detecting — Bandit honours `# nosec` itself, so a silenced finding
|
|
174
|
+
simply stops arriving and reads as fixed. Verification reads the source back
|
|
175
|
+
and calls it what it is.
|
|
176
|
+
|
|
177
|
+
Findings in the test tree and documentation are **reported but not
|
|
178
|
+
required**: §ACCEPT tells the agent not to patch them, so demanding them
|
|
179
|
+
back would make any repository with fixtures impossible to verify.
|
|
180
|
+
|
|
181
|
+
## Trend, which is what a score is actually for
|
|
182
|
+
|
|
183
|
+
A single B− tells you little. A B− that was an A− three runs ago tells you
|
|
184
|
+
something happened. Every run appends one line to
|
|
185
|
+
`.secure-code/history.jsonl` and prints the movement:
|
|
186
|
+
|
|
187
|
+
```
|
|
188
|
+
trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
A run whose coverage was too thin to grade records `null` and is skipped
|
|
192
|
+
rather than plotted as a collapse to zero — the same rule the rest of the
|
|
193
|
+
tool follows: absence of evidence is not a bad grade.
|
|
127
194
|
|
|
128
195
|
## Standards anchored, not invented
|
|
129
196
|
|
|
130
197
|
Known rules map to fields from five public standards. Unmapped and scanner-control
|
|
131
198
|
findings retain null standards fields rather than receiving invented mappings.
|
|
132
199
|
|
|
200
|
+
A CWE comes from one of three places, in descending authority: an adapter's
|
|
201
|
+
explicit override, the curated map in `standards.py`, then whatever the
|
|
202
|
+
scanner itself declared. That last source was being discarded — Bandit
|
|
203
|
+
publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
|
|
204
|
+
were dropped on the floor, leaving **14% of real findings with any CWE at
|
|
205
|
+
all** against a corpus measurement. Reading them takes it to 100%, and the
|
|
206
|
+
OWASP category is derived from the CWE using OWASP's own published
|
|
207
|
+
category-to-CWE lists where the curated map has none.
|
|
208
|
+
|
|
209
|
+
Derivation stops where the standard does. `CWE-703` — Bandit's classification
|
|
210
|
+
for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
|
|
211
|
+
null. 62% of findings carry an OWASP category and the rest say nothing, which
|
|
212
|
+
is the honest answer.
|
|
213
|
+
|
|
133
214
|
| Source | What we use it for |
|
|
134
215
|
|----------------------------------------------|----------------------------------------------------------|
|
|
135
216
|
| [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
|
|
@@ -375,6 +456,31 @@ SARIF upload. Pin production usage to a full commit SHA; the version tag above
|
|
|
375
456
|
is shown for readability. See [`action.yml`](action.yml) and
|
|
376
457
|
[`examples/github-actions/`](examples/github-actions/) for full workflows.
|
|
377
458
|
|
|
459
|
+
## maintainability-agent integration
|
|
460
|
+
|
|
461
|
+
`maintainability-agent` declares Security a **delegated** pillar naming this
|
|
462
|
+
tool, and reports it as `NotApplicable` so silence is never read as safety.
|
|
463
|
+
This tool emits the artifact that completes the picture:
|
|
464
|
+
|
|
465
|
+
```bash
|
|
466
|
+
secure-code-agent . --security-pillar security-pillar.json
|
|
467
|
+
maintainability-agent . --security-pillar security-pillar.json
|
|
468
|
+
```
|
|
469
|
+
|
|
470
|
+
Standalone use is unaffected — without the flag nothing is written and every
|
|
471
|
+
other output is identical. MA never executes this tool, and this tool never
|
|
472
|
+
imports MA; two independently releasable packages exchanging one document.
|
|
473
|
+
|
|
474
|
+
The artifact carries **two values that are never averaged**: a practice level
|
|
475
|
+
read from configuration and CI (*is anything preventing the next
|
|
476
|
+
vulnerability?*) and a code condition read from the scanners (*what did they
|
|
477
|
+
find?*). `condition` is `null` whenever scanner coverage is incomplete — this
|
|
478
|
+
tool's score is a rate over findings, so removing scanners makes the raw number
|
|
479
|
+
go **up**, and an unscanned repository must not arrive at MA looking measured.
|
|
480
|
+
|
|
481
|
+
Full contract, invariants and the practice rubric:
|
|
482
|
+
[`docs/ma-integration.md`](docs/ma-integration.md).
|
|
483
|
+
|
|
378
484
|
## What this is NOT
|
|
379
485
|
|
|
380
486
|
- ❌ **Not a SAST engine.** We delegate to Semgrep / Bandit / CodeQL / etc. — we don't write yet another AST analyzer.
|
|
@@ -398,12 +504,15 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
|
|
|
398
504
|
|
|
399
505
|
- [`docs/product-intent.md`](docs/product-intent.md) — Why this exists, who it serves, what it refuses to become
|
|
400
506
|
- [`docs/decisions.md`](docs/decisions.md) — Decision register: rulings the code alone cannot answer
|
|
507
|
+
- [`docs/ma-integration.md`](docs/ma-integration.md) — The `security-pillar.json` contract maintainability-agent reads
|
|
508
|
+
- [`docs/calibration.md`](docs/calibration.md) — The calibration study, its corpus, and what it found
|
|
401
509
|
- [`docs/design.md`](docs/design.md) — Architecture + non-goals + scanner protocol
|
|
402
510
|
- [`docs/architecture.md`](docs/architecture.md) — Audit of the system as built + remediation sequence
|
|
403
511
|
- [`docs/release-blockers.md`](docs/release-blockers.md) — Open v0.3.0 release blockers (do not tag until closed)
|
|
404
512
|
- [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
|
|
405
513
|
- [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
|
|
406
514
|
- [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
|
|
515
|
+
- [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
|
|
407
516
|
- [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
|
|
408
517
|
- [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
|
|
409
518
|
|
|
@@ -73,17 +73,98 @@ This is a constrained task, not a refactor.
|
|
|
73
73
|
10. Keep the patch small. If you find yourself rewriting a function
|
|
74
74
|
rather than patching it, stop and report the structural issue.
|
|
75
75
|
|
|
76
|
-
## §
|
|
76
|
+
## §FIX — patch these
|
|
77
77
|
...
|
|
78
78
|
```
|
|
79
79
|
|
|
80
|
-
Hand the
|
|
80
|
+
Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
|
|
81
|
+
|
|
82
|
+
**It is written on every run**, alongside the report — no flag required. The
|
|
83
|
+
report describes your code; the work order changes it. Set
|
|
84
|
+
`outputs.prompt_path` to `null` if you do not want one.
|
|
85
|
+
|
|
86
|
+
### Findings are tiered, because not all of them deserve equal attention
|
|
87
|
+
|
|
88
|
+
A flat list gives a `shell=True` command injection and a
|
|
89
|
+
`PASSWORD_FIELD = "password"` name-match the same billing, so an agent
|
|
90
|
+
working top to bottom spends its care on noise.
|
|
91
|
+
|
|
92
|
+
| tier | what it means | what the agent does |
|
|
93
|
+
| --- | --- | --- |
|
|
94
|
+
| **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
|
|
95
|
+
| **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
|
|
96
|
+
| **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
|
|
97
|
+
|
|
98
|
+
A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
|
|
99
|
+
useful hits out of 22 across the calibration corpus — empty defaults, field
|
|
100
|
+
names, `django-insecure-` — and still caught a planted hardcoded credential,
|
|
101
|
+
so the *value* is judged as well as the rule.
|
|
102
|
+
|
|
103
|
+
## Proving the work order actually helped
|
|
104
|
+
|
|
105
|
+
A work order nobody checks is a suggestion. After the agent has worked:
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
secure-code-agent . --verify-against secure-code-report.json
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
This re-audits and reports what was **fixed**, what is **still open**, what
|
|
112
|
+
was **silenced rather than repaired**, and what this work **introduced**. It
|
|
113
|
+
exits nonzero unless the run passes, because a verification step that always
|
|
114
|
+
passes verifies nothing.
|
|
115
|
+
|
|
116
|
+
```
|
|
117
|
+
work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
|
|
118
|
+
SILENCED B602 app.py:4
|
|
119
|
+
SILENCED B324 app.py:7
|
|
120
|
+
! 2 finding(s) disappeared without the code being repaired — a suppression
|
|
121
|
+
entry now covers them, or the reported line gained an inline marker such
|
|
122
|
+
as `# nosec`.
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
That last case is the one worth having. **Lint disable** is in the
|
|
126
|
+
anti-pattern table above, hard constraint 6 forbids it, and forbidding is
|
|
127
|
+
not detecting — Bandit honours `# nosec` itself, so a silenced finding
|
|
128
|
+
simply stops arriving and reads as fixed. Verification reads the source back
|
|
129
|
+
and calls it what it is.
|
|
130
|
+
|
|
131
|
+
Findings in the test tree and documentation are **reported but not
|
|
132
|
+
required**: §ACCEPT tells the agent not to patch them, so demanding them
|
|
133
|
+
back would make any repository with fixtures impossible to verify.
|
|
134
|
+
|
|
135
|
+
## Trend, which is what a score is actually for
|
|
136
|
+
|
|
137
|
+
A single B− tells you little. A B− that was an A− three runs ago tells you
|
|
138
|
+
something happened. Every run appends one line to
|
|
139
|
+
`.secure-code/history.jsonl` and prints the movement:
|
|
140
|
+
|
|
141
|
+
```
|
|
142
|
+
trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
A run whose coverage was too thin to grade records `null` and is skipped
|
|
146
|
+
rather than plotted as a collapse to zero — the same rule the rest of the
|
|
147
|
+
tool follows: absence of evidence is not a bad grade.
|
|
81
148
|
|
|
82
149
|
## Standards anchored, not invented
|
|
83
150
|
|
|
84
151
|
Known rules map to fields from five public standards. Unmapped and scanner-control
|
|
85
152
|
findings retain null standards fields rather than receiving invented mappings.
|
|
86
153
|
|
|
154
|
+
A CWE comes from one of three places, in descending authority: an adapter's
|
|
155
|
+
explicit override, the curated map in `standards.py`, then whatever the
|
|
156
|
+
scanner itself declared. That last source was being discarded — Bandit
|
|
157
|
+
publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
|
|
158
|
+
were dropped on the floor, leaving **14% of real findings with any CWE at
|
|
159
|
+
all** against a corpus measurement. Reading them takes it to 100%, and the
|
|
160
|
+
OWASP category is derived from the CWE using OWASP's own published
|
|
161
|
+
category-to-CWE lists where the curated map has none.
|
|
162
|
+
|
|
163
|
+
Derivation stops where the standard does. `CWE-703` — Bandit's classification
|
|
164
|
+
for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
|
|
165
|
+
null. 62% of findings carry an OWASP category and the rest say nothing, which
|
|
166
|
+
is the honest answer.
|
|
167
|
+
|
|
87
168
|
| Source | What we use it for |
|
|
88
169
|
|----------------------------------------------|----------------------------------------------------------|
|
|
89
170
|
| [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
|
|
@@ -329,6 +410,31 @@ SARIF upload. Pin production usage to a full commit SHA; the version tag above
|
|
|
329
410
|
is shown for readability. See [`action.yml`](action.yml) and
|
|
330
411
|
[`examples/github-actions/`](examples/github-actions/) for full workflows.
|
|
331
412
|
|
|
413
|
+
## maintainability-agent integration
|
|
414
|
+
|
|
415
|
+
`maintainability-agent` declares Security a **delegated** pillar naming this
|
|
416
|
+
tool, and reports it as `NotApplicable` so silence is never read as safety.
|
|
417
|
+
This tool emits the artifact that completes the picture:
|
|
418
|
+
|
|
419
|
+
```bash
|
|
420
|
+
secure-code-agent . --security-pillar security-pillar.json
|
|
421
|
+
maintainability-agent . --security-pillar security-pillar.json
|
|
422
|
+
```
|
|
423
|
+
|
|
424
|
+
Standalone use is unaffected — without the flag nothing is written and every
|
|
425
|
+
other output is identical. MA never executes this tool, and this tool never
|
|
426
|
+
imports MA; two independently releasable packages exchanging one document.
|
|
427
|
+
|
|
428
|
+
The artifact carries **two values that are never averaged**: a practice level
|
|
429
|
+
read from configuration and CI (*is anything preventing the next
|
|
430
|
+
vulnerability?*) and a code condition read from the scanners (*what did they
|
|
431
|
+
find?*). `condition` is `null` whenever scanner coverage is incomplete — this
|
|
432
|
+
tool's score is a rate over findings, so removing scanners makes the raw number
|
|
433
|
+
go **up**, and an unscanned repository must not arrive at MA looking measured.
|
|
434
|
+
|
|
435
|
+
Full contract, invariants and the practice rubric:
|
|
436
|
+
[`docs/ma-integration.md`](docs/ma-integration.md).
|
|
437
|
+
|
|
332
438
|
## What this is NOT
|
|
333
439
|
|
|
334
440
|
- ❌ **Not a SAST engine.** We delegate to Semgrep / Bandit / CodeQL / etc. — we don't write yet another AST analyzer.
|
|
@@ -352,12 +458,15 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
|
|
|
352
458
|
|
|
353
459
|
- [`docs/product-intent.md`](docs/product-intent.md) — Why this exists, who it serves, what it refuses to become
|
|
354
460
|
- [`docs/decisions.md`](docs/decisions.md) — Decision register: rulings the code alone cannot answer
|
|
461
|
+
- [`docs/ma-integration.md`](docs/ma-integration.md) — The `security-pillar.json` contract maintainability-agent reads
|
|
462
|
+
- [`docs/calibration.md`](docs/calibration.md) — The calibration study, its corpus, and what it found
|
|
355
463
|
- [`docs/design.md`](docs/design.md) — Architecture + non-goals + scanner protocol
|
|
356
464
|
- [`docs/architecture.md`](docs/architecture.md) — Audit of the system as built + remediation sequence
|
|
357
465
|
- [`docs/release-blockers.md`](docs/release-blockers.md) — Open v0.3.0 release blockers (do not tag until closed)
|
|
358
466
|
- [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
|
|
359
467
|
- [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
|
|
360
468
|
- [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
|
|
469
|
+
- [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
|
|
361
470
|
- [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
|
|
362
471
|
- [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
|
|
363
472
|
|
|
@@ -4,7 +4,15 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "secure-code-agent"
|
|
7
|
-
|
|
7
|
+
# Derived from `secure_code_audit.__version__`, which is the only place the
|
|
8
|
+
# number is written. It used to be duplicated here, and the two drifted: this
|
|
9
|
+
# file said 0.4.0 while the package said 0.3.0, so the released v0.4.0 wheel
|
|
10
|
+
# stamped 0.3.0 into every SARIF document, JSON report, Markdown header,
|
|
11
|
+
# `--version` and the pillar artifact handed to maintainability-agent. The
|
|
12
|
+
# release workflow compared the tag against *this* line and never looked at the
|
|
13
|
+
# package, so nothing caught it. A test asserting the two agree would only
|
|
14
|
+
# police the duplication; removing it is the fix.
|
|
15
|
+
dynamic = ["version"]
|
|
8
16
|
description = "Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored."
|
|
9
17
|
readme = "README.md"
|
|
10
18
|
requires-python = ">=3.10"
|
|
@@ -86,6 +94,12 @@ Documentation = "https://github.com/marshallguillory86/secure-code-agent/tree/ma
|
|
|
86
94
|
Issues = "https://github.com/marshallguillory86/secure-code-agent/issues"
|
|
87
95
|
Changelog = "https://github.com/marshallguillory86/secure-code-agent/blob/main/CHANGELOG.md"
|
|
88
96
|
|
|
97
|
+
[tool.setuptools.dynamic]
|
|
98
|
+
# The single source of truth for the version. `src/secure_code_audit/__init__.py`
|
|
99
|
+
# holds the literal; this reads it. The two cannot drift because there is only
|
|
100
|
+
# one of them.
|
|
101
|
+
version = { attr = "secure_code_audit.__version__" }
|
|
102
|
+
|
|
89
103
|
[tool.setuptools.packages.find]
|
|
90
104
|
where = ["src"]
|
|
91
105
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: secure-code-agent
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
|
|
5
5
|
Author: Marshall Guillory
|
|
6
6
|
License: MIT
|
|
@@ -119,17 +119,98 @@ This is a constrained task, not a refactor.
|
|
|
119
119
|
10. Keep the patch small. If you find yourself rewriting a function
|
|
120
120
|
rather than patching it, stop and report the structural issue.
|
|
121
121
|
|
|
122
|
-
## §
|
|
122
|
+
## §FIX — patch these
|
|
123
123
|
...
|
|
124
124
|
```
|
|
125
125
|
|
|
126
|
-
Hand the
|
|
126
|
+
Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
|
|
127
|
+
|
|
128
|
+
**It is written on every run**, alongside the report — no flag required. The
|
|
129
|
+
report describes your code; the work order changes it. Set
|
|
130
|
+
`outputs.prompt_path` to `null` if you do not want one.
|
|
131
|
+
|
|
132
|
+
### Findings are tiered, because not all of them deserve equal attention
|
|
133
|
+
|
|
134
|
+
A flat list gives a `shell=True` command injection and a
|
|
135
|
+
`PASSWORD_FIELD = "password"` name-match the same billing, so an agent
|
|
136
|
+
working top to bottom spends its care on noise.
|
|
137
|
+
|
|
138
|
+
| tier | what it means | what the agent does |
|
|
139
|
+
| --- | --- | --- |
|
|
140
|
+
| **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
|
|
141
|
+
| **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
|
|
142
|
+
| **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
|
|
143
|
+
|
|
144
|
+
A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
|
|
145
|
+
useful hits out of 22 across the calibration corpus — empty defaults, field
|
|
146
|
+
names, `django-insecure-` — and still caught a planted hardcoded credential,
|
|
147
|
+
so the *value* is judged as well as the rule.
|
|
148
|
+
|
|
149
|
+
## Proving the work order actually helped
|
|
150
|
+
|
|
151
|
+
A work order nobody checks is a suggestion. After the agent has worked:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
secure-code-agent . --verify-against secure-code-report.json
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
This re-audits and reports what was **fixed**, what is **still open**, what
|
|
158
|
+
was **silenced rather than repaired**, and what this work **introduced**. It
|
|
159
|
+
exits nonzero unless the run passes, because a verification step that always
|
|
160
|
+
passes verifies nothing.
|
|
161
|
+
|
|
162
|
+
```
|
|
163
|
+
work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
|
|
164
|
+
SILENCED B602 app.py:4
|
|
165
|
+
SILENCED B324 app.py:7
|
|
166
|
+
! 2 finding(s) disappeared without the code being repaired — a suppression
|
|
167
|
+
entry now covers them, or the reported line gained an inline marker such
|
|
168
|
+
as `# nosec`.
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
That last case is the one worth having. **Lint disable** is in the
|
|
172
|
+
anti-pattern table above, hard constraint 6 forbids it, and forbidding is
|
|
173
|
+
not detecting — Bandit honours `# nosec` itself, so a silenced finding
|
|
174
|
+
simply stops arriving and reads as fixed. Verification reads the source back
|
|
175
|
+
and calls it what it is.
|
|
176
|
+
|
|
177
|
+
Findings in the test tree and documentation are **reported but not
|
|
178
|
+
required**: §ACCEPT tells the agent not to patch them, so demanding them
|
|
179
|
+
back would make any repository with fixtures impossible to verify.
|
|
180
|
+
|
|
181
|
+
## Trend, which is what a score is actually for
|
|
182
|
+
|
|
183
|
+
A single B− tells you little. A B− that was an A− three runs ago tells you
|
|
184
|
+
something happened. Every run appends one line to
|
|
185
|
+
`.secure-code/history.jsonl` and prints the movement:
|
|
186
|
+
|
|
187
|
+
```
|
|
188
|
+
trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
A run whose coverage was too thin to grade records `null` and is skipped
|
|
192
|
+
rather than plotted as a collapse to zero — the same rule the rest of the
|
|
193
|
+
tool follows: absence of evidence is not a bad grade.
|
|
127
194
|
|
|
128
195
|
## Standards anchored, not invented
|
|
129
196
|
|
|
130
197
|
Known rules map to fields from five public standards. Unmapped and scanner-control
|
|
131
198
|
findings retain null standards fields rather than receiving invented mappings.
|
|
132
199
|
|
|
200
|
+
A CWE comes from one of three places, in descending authority: an adapter's
|
|
201
|
+
explicit override, the curated map in `standards.py`, then whatever the
|
|
202
|
+
scanner itself declared. That last source was being discarded — Bandit
|
|
203
|
+
publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
|
|
204
|
+
were dropped on the floor, leaving **14% of real findings with any CWE at
|
|
205
|
+
all** against a corpus measurement. Reading them takes it to 100%, and the
|
|
206
|
+
OWASP category is derived from the CWE using OWASP's own published
|
|
207
|
+
category-to-CWE lists where the curated map has none.
|
|
208
|
+
|
|
209
|
+
Derivation stops where the standard does. `CWE-703` — Bandit's classification
|
|
210
|
+
for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
|
|
211
|
+
null. 62% of findings carry an OWASP category and the rest say nothing, which
|
|
212
|
+
is the honest answer.
|
|
213
|
+
|
|
133
214
|
| Source | What we use it for |
|
|
134
215
|
|----------------------------------------------|----------------------------------------------------------|
|
|
135
216
|
| [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
|
|
@@ -375,6 +456,31 @@ SARIF upload. Pin production usage to a full commit SHA; the version tag above
|
|
|
375
456
|
is shown for readability. See [`action.yml`](action.yml) and
|
|
376
457
|
[`examples/github-actions/`](examples/github-actions/) for full workflows.
|
|
377
458
|
|
|
459
|
+
## maintainability-agent integration
|
|
460
|
+
|
|
461
|
+
`maintainability-agent` declares Security a **delegated** pillar naming this
|
|
462
|
+
tool, and reports it as `NotApplicable` so silence is never read as safety.
|
|
463
|
+
This tool emits the artifact that completes the picture:
|
|
464
|
+
|
|
465
|
+
```bash
|
|
466
|
+
secure-code-agent . --security-pillar security-pillar.json
|
|
467
|
+
maintainability-agent . --security-pillar security-pillar.json
|
|
468
|
+
```
|
|
469
|
+
|
|
470
|
+
Standalone use is unaffected — without the flag nothing is written and every
|
|
471
|
+
other output is identical. MA never executes this tool, and this tool never
|
|
472
|
+
imports MA; two independently releasable packages exchanging one document.
|
|
473
|
+
|
|
474
|
+
The artifact carries **two values that are never averaged**: a practice level
|
|
475
|
+
read from configuration and CI (*is anything preventing the next
|
|
476
|
+
vulnerability?*) and a code condition read from the scanners (*what did they
|
|
477
|
+
find?*). `condition` is `null` whenever scanner coverage is incomplete — this
|
|
478
|
+
tool's score is a rate over findings, so removing scanners makes the raw number
|
|
479
|
+
go **up**, and an unscanned repository must not arrive at MA looking measured.
|
|
480
|
+
|
|
481
|
+
Full contract, invariants and the practice rubric:
|
|
482
|
+
[`docs/ma-integration.md`](docs/ma-integration.md).
|
|
483
|
+
|
|
378
484
|
## What this is NOT
|
|
379
485
|
|
|
380
486
|
- ❌ **Not a SAST engine.** We delegate to Semgrep / Bandit / CodeQL / etc. — we don't write yet another AST analyzer.
|
|
@@ -398,12 +504,15 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
|
|
|
398
504
|
|
|
399
505
|
- [`docs/product-intent.md`](docs/product-intent.md) — Why this exists, who it serves, what it refuses to become
|
|
400
506
|
- [`docs/decisions.md`](docs/decisions.md) — Decision register: rulings the code alone cannot answer
|
|
507
|
+
- [`docs/ma-integration.md`](docs/ma-integration.md) — The `security-pillar.json` contract maintainability-agent reads
|
|
508
|
+
- [`docs/calibration.md`](docs/calibration.md) — The calibration study, its corpus, and what it found
|
|
401
509
|
- [`docs/design.md`](docs/design.md) — Architecture + non-goals + scanner protocol
|
|
402
510
|
- [`docs/architecture.md`](docs/architecture.md) — Audit of the system as built + remediation sequence
|
|
403
511
|
- [`docs/release-blockers.md`](docs/release-blockers.md) — Open v0.3.0 release blockers (do not tag until closed)
|
|
404
512
|
- [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
|
|
405
513
|
- [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
|
|
406
514
|
- [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
|
|
515
|
+
- [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
|
|
407
516
|
- [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
|
|
408
517
|
- [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
|
|
409
518
|
|
{secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/SOURCES.txt
RENAMED
|
@@ -13,7 +13,10 @@ src/secure_code_audit/cli.py
|
|
|
13
13
|
src/secure_code_audit/config.py
|
|
14
14
|
src/secure_code_audit/findings.py
|
|
15
15
|
src/secure_code_audit/git_tools.py
|
|
16
|
+
src/secure_code_audit/history.py
|
|
16
17
|
src/secure_code_audit/instructions.py
|
|
18
|
+
src/secure_code_audit/pillar.py
|
|
19
|
+
src/secure_code_audit/practice.py
|
|
17
20
|
src/secure_code_audit/remediation.py
|
|
18
21
|
src/secure_code_audit/renderers.py
|
|
19
22
|
src/secure_code_audit/ruleset.py
|
|
@@ -22,6 +25,8 @@ src/secure_code_audit/scanner_status.py
|
|
|
22
25
|
src/secure_code_audit/scoring.py
|
|
23
26
|
src/secure_code_audit/standards.py
|
|
24
27
|
src/secure_code_audit/suppressions.py
|
|
28
|
+
src/secure_code_audit/triage.py
|
|
29
|
+
src/secure_code_audit/verify.py
|
|
25
30
|
src/secure_code_audit/data/semgrep-offline.yaml
|
|
26
31
|
src/secure_code_audit/scanners/__init__.py
|
|
27
32
|
src/secure_code_audit/scanners/bandit_scanner.py
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""secure-code-agent — deterministic security gate + bounded AI remediation prompts."""
|
|
2
|
+
|
|
3
|
+
#: The only place the version is written. `pyproject.toml` derives it via
|
|
4
|
+
#: `[tool.setuptools.dynamic]`, so the two cannot drift.
|
|
5
|
+
#:
|
|
6
|
+
#: They did drift once, and it shipped: pyproject said 0.4.0 while this said
|
|
7
|
+
#: 0.3.0, so the released v0.4.0 wheel stamped 0.3.0 into every SARIF document,
|
|
8
|
+
#: every JSON report, the Markdown header, `--version`, and the
|
|
9
|
+
#: `security-pillar.json` handed to maintainability-agent. The release workflow
|
|
10
|
+
#: compared the tag against pyproject's line and never looked here — it
|
|
11
|
+
#: verified the half that was right and shipped the half that was wrong.
|
|
12
|
+
#:
|
|
13
|
+
#: PyPI is immutable, so 0.4.0 stays wrong. 0.5.0 is the first build whose
|
|
14
|
+
#: artifacts name their own producer correctly.
|
|
15
|
+
__version__ = "0.6.0"
|