secure-code-agent 0.4.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. {secure_code_agent-0.4.0/src/secure_code_agent.egg-info → secure_code_agent-0.6.0}/PKG-INFO +112 -3
  2. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/README.md +111 -2
  3. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/pyproject.toml +15 -1
  4. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0/src/secure_code_agent.egg-info}/PKG-INFO +112 -3
  5. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/SOURCES.txt +5 -0
  6. secure_code_agent-0.6.0/src/secure_code_audit/__init__.py +15 -0
  7. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/cli.py +416 -29
  8. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/config.py +85 -1
  9. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/data/semgrep-offline.yaml +27 -7
  10. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/findings.py +111 -2
  11. secure_code_agent-0.6.0/src/secure_code_audit/git_tools.py +128 -0
  12. secure_code_agent-0.6.0/src/secure_code_audit/history.py +160 -0
  13. secure_code_agent-0.6.0/src/secure_code_audit/pillar.py +194 -0
  14. secure_code_agent-0.6.0/src/secure_code_audit/practice.py +300 -0
  15. secure_code_agent-0.6.0/src/secure_code_audit/remediation.py +303 -0
  16. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/renderers.py +159 -13
  17. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/ruleset.py +1 -1
  18. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/bandit_scanner.py +21 -1
  19. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/base.py +74 -6
  20. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/builtin_rules.py +7 -1
  21. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/gitleaks_scanner.py +68 -16
  22. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/gosec_scanner.py +5 -0
  23. secure_code_agent-0.6.0/src/secure_code_audit/scoring.py +712 -0
  24. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/standards.py +109 -0
  25. secure_code_agent-0.6.0/src/secure_code_audit/triage.py +155 -0
  26. secure_code_agent-0.6.0/src/secure_code_audit/verify.py +295 -0
  27. secure_code_agent-0.4.0/src/secure_code_audit/__init__.py +0 -3
  28. secure_code_agent-0.4.0/src/secure_code_audit/git_tools.py +0 -65
  29. secure_code_agent-0.4.0/src/secure_code_audit/remediation.py +0 -168
  30. secure_code_agent-0.4.0/src/secure_code_audit/scoring.py +0 -421
  31. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/LICENSE +0 -0
  32. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/setup.cfg +0 -0
  33. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/dependency_links.txt +0 -0
  34. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/entry_points.txt +0 -0
  35. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/requires.txt +0 -0
  36. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_agent.egg-info/top_level.txt +0 -0
  37. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/baseline.py +0 -0
  38. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/instructions.py +0 -0
  39. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/sarif.py +0 -0
  40. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanner_status.py +0 -0
  41. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/__init__.py +0 -0
  42. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/checkov_scanner.py +0 -0
  43. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/floor.py +0 -0
  44. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/hadolint_scanner.py +0 -0
  45. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/njsscan_scanner.py +0 -0
  46. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/npm_audit_scanner.py +0 -0
  47. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/osv_scanner.py +0 -0
  48. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/pip_audit_scanner.py +0 -0
  49. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/rubocop_scanner.py +0 -0
  50. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/scorecard_scanner.py +0 -0
  51. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/semgrep_scanner.py +0 -0
  52. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/trivy_scanner.py +0 -0
  53. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/scanners/trufflehog_scanner.py +0 -0
  54. {secure_code_agent-0.4.0 → secure_code_agent-0.6.0}/src/secure_code_audit/suppressions.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: secure-code-agent
3
- Version: 0.4.0
3
+ Version: 0.6.0
4
4
  Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
5
5
  Author: Marshall Guillory
6
6
  License: MIT
@@ -119,17 +119,98 @@ This is a constrained task, not a refactor.
119
119
  10. Keep the patch small. If you find yourself rewriting a function
120
120
  rather than patching it, stop and report the structural issue.
121
121
 
122
- ## §FINDINGS
122
+ ## §FIX — patch these
123
123
  ...
124
124
  ```
125
125
 
126
- Hand the prompt to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
126
+ Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
127
+
128
+ **It is written on every run**, alongside the report — no flag required. The
129
+ report describes your code; the work order changes it. Set
130
+ `outputs.prompt_path` to `null` if you do not want one.
131
+
132
+ ### Findings are tiered, because not all of them deserve equal attention
133
+
134
+ A flat list gives a `shell=True` command injection and a
135
+ `PASSWORD_FIELD = "password"` name-match the same billing, so an agent
136
+ working top to bottom spends its care on noise.
137
+
138
+ | tier | what it means | what the agent does |
139
+ | --- | --- | --- |
140
+ | **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
141
+ | **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
142
+ | **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
143
+
144
+ A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
145
+ useful hits out of 22 across the calibration corpus — empty defaults, field
146
+ names, `django-insecure-` — and still caught a planted hardcoded credential,
147
+ so the *value* is judged as well as the rule.
148
+
149
+ ## Proving the work order actually helped
150
+
151
+ A work order nobody checks is a suggestion. After the agent has worked:
152
+
153
+ ```bash
154
+ secure-code-agent . --verify-against secure-code-report.json
155
+ ```
156
+
157
+ This re-audits and reports what was **fixed**, what is **still open**, what
158
+ was **silenced rather than repaired**, and what this work **introduced**. It
159
+ exits nonzero unless the run passes, because a verification step that always
160
+ passes verifies nothing.
161
+
162
+ ```
163
+ work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
164
+ SILENCED B602 app.py:4
165
+ SILENCED B324 app.py:7
166
+ ! 2 finding(s) disappeared without the code being repaired — a suppression
167
+ entry now covers them, or the reported line gained an inline marker such
168
+ as `# nosec`.
169
+ ```
170
+
171
+ That last case is the one worth having. **Lint disable** is in the
172
+ anti-pattern table above, hard constraint 6 forbids it, and forbidding is
173
+ not detecting — Bandit honours `# nosec` itself, so a silenced finding
174
+ simply stops arriving and reads as fixed. Verification reads the source back
175
+ and calls it what it is.
176
+
177
+ Findings in the test tree and documentation are **reported but not
178
+ required**: §ACCEPT tells the agent not to patch them, so demanding them
179
+ back would make any repository with fixtures impossible to verify.
180
+
181
+ ## Trend, which is what a score is actually for
182
+
183
+ A single B− tells you little. A B− that was an A− three runs ago tells you
184
+ something happened. Every run appends one line to
185
+ `.secure-code/history.jsonl` and prints the movement:
186
+
187
+ ```
188
+ trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
189
+ ```
190
+
191
+ A run whose coverage was too thin to grade records `null` and is skipped
192
+ rather than plotted as a collapse to zero — the same rule the rest of the
193
+ tool follows: absence of evidence is not a bad grade.
127
194
 
128
195
  ## Standards anchored, not invented
129
196
 
130
197
  Known rules map to fields from five public standards. Unmapped and scanner-control
131
198
  findings retain null standards fields rather than receiving invented mappings.
132
199
 
200
+ A CWE comes from one of three places, in descending authority: an adapter's
201
+ explicit override, the curated map in `standards.py`, then whatever the
202
+ scanner itself declared. That last source was being discarded — Bandit
203
+ publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
204
+ were dropped on the floor, leaving **14% of real findings with any CWE at
205
+ all** against a corpus measurement. Reading them takes it to 100%, and the
206
+ OWASP category is derived from the CWE using OWASP's own published
207
+ category-to-CWE lists where the curated map has none.
208
+
209
+ Derivation stops where the standard does. `CWE-703` — Bandit's classification
210
+ for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
211
+ null. 62% of findings carry an OWASP category and the rest say nothing, which
212
+ is the honest answer.
213
+
133
214
  | Source | What we use it for |
134
215
  |----------------------------------------------|----------------------------------------------------------|
135
216
  | [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
@@ -375,6 +456,31 @@ SARIF upload. Pin production usage to a full commit SHA; the version tag above
375
456
  is shown for readability. See [`action.yml`](action.yml) and
376
457
  [`examples/github-actions/`](examples/github-actions/) for full workflows.
377
458
 
459
+ ## maintainability-agent integration
460
+
461
+ `maintainability-agent` declares Security a **delegated** pillar naming this
462
+ tool, and reports it as `NotApplicable` so silence is never read as safety.
463
+ This tool emits the artifact that completes the picture:
464
+
465
+ ```bash
466
+ secure-code-agent . --security-pillar security-pillar.json
467
+ maintainability-agent . --security-pillar security-pillar.json
468
+ ```
469
+
470
+ Standalone use is unaffected — without the flag nothing is written and every
471
+ other output is identical. MA never executes this tool, and this tool never
472
+ imports MA; two independently releasable packages exchanging one document.
473
+
474
+ The artifact carries **two values that are never averaged**: a practice level
475
+ read from configuration and CI (*is anything preventing the next
476
+ vulnerability?*) and a code condition read from the scanners (*what did they
477
+ find?*). `condition` is `null` whenever scanner coverage is incomplete — this
478
+ tool's score is a rate over findings, so removing scanners makes the raw number
479
+ go **up**, and an unscanned repository must not arrive at MA looking measured.
480
+
481
+ Full contract, invariants and the practice rubric:
482
+ [`docs/ma-integration.md`](docs/ma-integration.md).
483
+
378
484
  ## What this is NOT
379
485
 
380
486
  - ❌ **Not a SAST engine.** We delegate to Semgrep / Bandit / CodeQL / etc. — we don't write yet another AST analyzer.
@@ -398,12 +504,15 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
398
504
 
399
505
  - [`docs/product-intent.md`](docs/product-intent.md) — Why this exists, who it serves, what it refuses to become
400
506
  - [`docs/decisions.md`](docs/decisions.md) — Decision register: rulings the code alone cannot answer
507
+ - [`docs/ma-integration.md`](docs/ma-integration.md) — The `security-pillar.json` contract maintainability-agent reads
508
+ - [`docs/calibration.md`](docs/calibration.md) — The calibration study, its corpus, and what it found
401
509
  - [`docs/design.md`](docs/design.md) — Architecture + non-goals + scanner protocol
402
510
  - [`docs/architecture.md`](docs/architecture.md) — Audit of the system as built + remediation sequence
403
511
  - [`docs/release-blockers.md`](docs/release-blockers.md) — Open v0.3.0 release blockers (do not tag until closed)
404
512
  - [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
405
513
  - [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
406
514
  - [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
515
+ - [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
407
516
  - [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
408
517
  - [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
409
518
 
@@ -73,17 +73,98 @@ This is a constrained task, not a refactor.
73
73
  10. Keep the patch small. If you find yourself rewriting a function
74
74
  rather than patching it, stop and report the structural issue.
75
75
 
76
- ## §FINDINGS
76
+ ## §FIX — patch these
77
77
  ...
78
78
  ```
79
79
 
80
- Hand the prompt to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
80
+ Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
81
+
82
+ **It is written on every run**, alongside the report — no flag required. The
83
+ report describes your code; the work order changes it. Set
84
+ `outputs.prompt_path` to `null` if you do not want one.
85
+
86
+ ### Findings are tiered, because not all of them deserve equal attention
87
+
88
+ A flat list gives a `shell=True` command injection and a
89
+ `PASSWORD_FIELD = "password"` name-match the same billing, so an agent
90
+ working top to bottom spends its care on noise.
91
+
92
+ | tier | what it means | what the agent does |
93
+ | --- | --- | --- |
94
+ | **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
95
+ | **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
96
+ | **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
97
+
98
+ A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
99
+ useful hits out of 22 across the calibration corpus — empty defaults, field
100
+ names, `django-insecure-` — and still caught a planted hardcoded credential,
101
+ so the *value* is judged as well as the rule.
102
+
103
+ ## Proving the work order actually helped
104
+
105
+ A work order nobody checks is a suggestion. After the agent has worked:
106
+
107
+ ```bash
108
+ secure-code-agent . --verify-against secure-code-report.json
109
+ ```
110
+
111
+ This re-audits and reports what was **fixed**, what is **still open**, what
112
+ was **silenced rather than repaired**, and what this work **introduced**. It
113
+ exits nonzero unless the run passes, because a verification step that always
114
+ passes verifies nothing.
115
+
116
+ ```
117
+ work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
118
+ SILENCED B602 app.py:4
119
+ SILENCED B324 app.py:7
120
+ ! 2 finding(s) disappeared without the code being repaired — a suppression
121
+ entry now covers them, or the reported line gained an inline marker such
122
+ as `# nosec`.
123
+ ```
124
+
125
+ That last case is the one worth having. **Lint disable** is in the
126
+ anti-pattern table above, hard constraint 6 forbids it, and forbidding is
127
+ not detecting — Bandit honours `# nosec` itself, so a silenced finding
128
+ simply stops arriving and reads as fixed. Verification reads the source back
129
+ and calls it what it is.
130
+
131
+ Findings in the test tree and documentation are **reported but not
132
+ required**: §ACCEPT tells the agent not to patch them, so demanding them
133
+ back would make any repository with fixtures impossible to verify.
134
+
135
+ ## Trend, which is what a score is actually for
136
+
137
+ A single B− tells you little. A B− that was an A− three runs ago tells you
138
+ something happened. Every run appends one line to
139
+ `.secure-code/history.jsonl` and prints the movement:
140
+
141
+ ```
142
+ trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
143
+ ```
144
+
145
+ A run whose coverage was too thin to grade records `null` and is skipped
146
+ rather than plotted as a collapse to zero — the same rule the rest of the
147
+ tool follows: absence of evidence is not a bad grade.
81
148
 
82
149
  ## Standards anchored, not invented
83
150
 
84
151
  Known rules map to fields from five public standards. Unmapped and scanner-control
85
152
  findings retain null standards fields rather than receiving invented mappings.
86
153
 
154
+ A CWE comes from one of three places, in descending authority: an adapter's
155
+ explicit override, the curated map in `standards.py`, then whatever the
156
+ scanner itself declared. That last source was being discarded — Bandit
157
+ publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
158
+ were dropped on the floor, leaving **14% of real findings with any CWE at
159
+ all** against a corpus measurement. Reading them takes it to 100%, and the
160
+ OWASP category is derived from the CWE using OWASP's own published
161
+ category-to-CWE lists where the curated map has none.
162
+
163
+ Derivation stops where the standard does. `CWE-703` — Bandit's classification
164
+ for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
165
+ null. 62% of findings carry an OWASP category and the rest say nothing, which
166
+ is the honest answer.
167
+
87
168
  | Source | What we use it for |
88
169
  |----------------------------------------------|----------------------------------------------------------|
89
170
  | [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
@@ -329,6 +410,31 @@ SARIF upload. Pin production usage to a full commit SHA; the version tag above
329
410
  is shown for readability. See [`action.yml`](action.yml) and
330
411
  [`examples/github-actions/`](examples/github-actions/) for full workflows.
331
412
 
413
+ ## maintainability-agent integration
414
+
415
+ `maintainability-agent` declares Security a **delegated** pillar naming this
416
+ tool, and reports it as `NotApplicable` so silence is never read as safety.
417
+ This tool emits the artifact that completes the picture:
418
+
419
+ ```bash
420
+ secure-code-agent . --security-pillar security-pillar.json
421
+ maintainability-agent . --security-pillar security-pillar.json
422
+ ```
423
+
424
+ Standalone use is unaffected — without the flag nothing is written and every
425
+ other output is identical. MA never executes this tool, and this tool never
426
+ imports MA; two independently releasable packages exchanging one document.
427
+
428
+ The artifact carries **two values that are never averaged**: a practice level
429
+ read from configuration and CI (*is anything preventing the next
430
+ vulnerability?*) and a code condition read from the scanners (*what did they
431
+ find?*). `condition` is `null` whenever scanner coverage is incomplete — this
432
+ tool's score is a rate over findings, so removing scanners makes the raw number
433
+ go **up**, and an unscanned repository must not arrive at MA looking measured.
434
+
435
+ Full contract, invariants and the practice rubric:
436
+ [`docs/ma-integration.md`](docs/ma-integration.md).
437
+
332
438
  ## What this is NOT
333
439
 
334
440
  - ❌ **Not a SAST engine.** We delegate to Semgrep / Bandit / CodeQL / etc. — we don't write yet another AST analyzer.
@@ -352,12 +458,15 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
352
458
 
353
459
  - [`docs/product-intent.md`](docs/product-intent.md) — Why this exists, who it serves, what it refuses to become
354
460
  - [`docs/decisions.md`](docs/decisions.md) — Decision register: rulings the code alone cannot answer
461
+ - [`docs/ma-integration.md`](docs/ma-integration.md) — The `security-pillar.json` contract maintainability-agent reads
462
+ - [`docs/calibration.md`](docs/calibration.md) — The calibration study, its corpus, and what it found
355
463
  - [`docs/design.md`](docs/design.md) — Architecture + non-goals + scanner protocol
356
464
  - [`docs/architecture.md`](docs/architecture.md) — Audit of the system as built + remediation sequence
357
465
  - [`docs/release-blockers.md`](docs/release-blockers.md) — Open v0.3.0 release blockers (do not tag until closed)
358
466
  - [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
359
467
  - [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
360
468
  - [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
469
+ - [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
361
470
  - [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
362
471
  - [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
363
472
 
@@ -4,7 +4,15 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "secure-code-agent"
7
- version = "0.4.0"
7
+ # Derived from `secure_code_audit.__version__`, which is the only place the
8
+ # number is written. It used to be duplicated here, and the two drifted: this
9
+ # file said 0.4.0 while the package said 0.3.0, so the released v0.4.0 wheel
10
+ # stamped 0.3.0 into every SARIF document, JSON report, Markdown header,
11
+ # `--version` and the pillar artifact handed to maintainability-agent. The
12
+ # release workflow compared the tag against *this* line and never looked at the
13
+ # package, so nothing caught it. A test asserting the two agree would only
14
+ # police the duplication; removing it is the fix.
15
+ dynamic = ["version"]
8
16
  description = "Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored."
9
17
  readme = "README.md"
10
18
  requires-python = ">=3.10"
@@ -86,6 +94,12 @@ Documentation = "https://github.com/marshallguillory86/secure-code-agent/tree/ma
86
94
  Issues = "https://github.com/marshallguillory86/secure-code-agent/issues"
87
95
  Changelog = "https://github.com/marshallguillory86/secure-code-agent/blob/main/CHANGELOG.md"
88
96
 
97
+ [tool.setuptools.dynamic]
98
+ # The single source of truth for the version. `src/secure_code_audit/__init__.py`
99
+ # holds the literal; this reads it. The two cannot drift because there is only
100
+ # one of them.
101
+ version = { attr = "secure_code_audit.__version__" }
102
+
89
103
  [tool.setuptools.packages.find]
90
104
  where = ["src"]
91
105
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: secure-code-agent
3
- Version: 0.4.0
3
+ Version: 0.6.0
4
4
  Summary: Deterministic security gate + bounded AI remediation prompt generator. NIST SSDF / OWASP ASVS / CWE Top 25 anchored.
5
5
  Author: Marshall Guillory
6
6
  License: MIT
@@ -119,17 +119,98 @@ This is a constrained task, not a refactor.
119
119
  10. Keep the patch small. If you find yourself rewriting a function
120
120
  rather than patching it, stop and report the structural issue.
121
121
 
122
- ## §FINDINGS
122
+ ## §FIX — patch these
123
123
  ...
124
124
  ```
125
125
 
126
- Hand the prompt to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
126
+ Hand the work order to Claude Code, Codex, Cursor, Copilot, or any agent. The agent now has explicit boundaries. The full template + rationale lives in [`docs/remediation.md`](docs/remediation.md).
127
+
128
+ **It is written on every run**, alongside the report — no flag required. The
129
+ report describes your code; the work order changes it. Set
130
+ `outputs.prompt_path` to `null` if you do not want one.
131
+
132
+ ### Findings are tiered, because not all of them deserve equal attention
133
+
134
+ A flat list gives a `shell=True` command injection and a
135
+ `PASSWORD_FIELD = "password"` name-match the same billing, so an agent
136
+ working top to bottom spends its care on noise.
137
+
138
+ | tier | what it means | what the agent does |
139
+ | --- | --- | --- |
140
+ | **§FIX** | the scanner is confident and the rule has not been measured producing noise | patch it |
141
+ | **§REVIEW** | a rule measured producing non-defects, or a scanner reporting low confidence — the reason is stated per finding | confirm it is real first; a justified suppression is a *successful* outcome here |
142
+ | **§ACCEPT** | test tree and documentation | propose a suppression, do not patch |
143
+
144
+ A noisy rule is demoted, never dropped. Bandit's `B105` produced **zero**
145
+ useful hits out of 22 across the calibration corpus — empty defaults, field
146
+ names, `django-insecure-` — and still caught a planted hardcoded credential,
147
+ so the *value* is judged as well as the rule.
148
+
149
+ ## Proving the work order actually helped
150
+
151
+ A work order nobody checks is a suggestion. After the agent has worked:
152
+
153
+ ```bash
154
+ secure-code-agent . --verify-against secure-code-report.json
155
+ ```
156
+
157
+ This re-audits and reports what was **fixed**, what is **still open**, what
158
+ was **silenced rather than repaired**, and what this work **introduced**. It
159
+ exits nonzero unless the run passes, because a verification step that always
160
+ passes verifies nothing.
161
+
162
+ ```
163
+ work order verification · not proven: 0 fixed, 2 silenced rather than fixed, 1 still open
164
+ SILENCED B602 app.py:4
165
+ SILENCED B324 app.py:7
166
+ ! 2 finding(s) disappeared without the code being repaired — a suppression
167
+ entry now covers them, or the reported line gained an inline marker such
168
+ as `# nosec`.
169
+ ```
170
+
171
+ That last case is the one worth having. **Lint disable** is in the
172
+ anti-pattern table above, hard constraint 6 forbids it, and forbidding is
173
+ not detecting — Bandit honours `# nosec` itself, so a silenced finding
174
+ simply stops arriving and reads as fixed. Verification reads the source back
175
+ and calls it what it is.
176
+
177
+ Findings in the test tree and documentation are **reported but not
178
+ required**: §ACCEPT tells the agent not to patch them, so demanding them
179
+ back would make any repository with fixtures impossible to verify.
180
+
181
+ ## Trend, which is what a score is actually for
182
+
183
+ A single B− tells you little. A B− that was an A− three runs ago tells you
184
+ something happened. Every run appends one line to
185
+ `.secure-code/history.jsonl` and prints the movement:
186
+
187
+ ```
188
+ trend: 3.90 (B+) — up 3.86 from 0.04 (F), 3 scored runs
189
+ ```
190
+
191
+ A run whose coverage was too thin to grade records `null` and is skipped
192
+ rather than plotted as a collapse to zero — the same rule the rest of the
193
+ tool follows: absence of evidence is not a bad grade.
127
194
 
128
195
  ## Standards anchored, not invented
129
196
 
130
197
  Known rules map to fields from five public standards. Unmapped and scanner-control
131
198
  findings retain null standards fields rather than receiving invented mappings.
132
199
 
200
+ A CWE comes from one of three places, in descending authority: an adapter's
201
+ explicit override, the curated map in `standards.py`, then whatever the
202
+ scanner itself declared. That last source was being discarded — Bandit
203
+ publishes a CWE for all ~70 of its plugins and gosec for every rule, and both
204
+ were dropped on the floor, leaving **14% of real findings with any CWE at
205
+ all** against a corpus measurement. Reading them takes it to 100%, and the
206
+ OWASP category is derived from the CWE using OWASP's own published
207
+ category-to-CWE lists where the curated map has none.
208
+
209
+ Derivation stops where the standard does. `CWE-703` — Bandit's classification
210
+ for a bare `assert` — belongs to no OWASP Top 10 category, so that field stays
211
+ null. 62% of findings carry an OWASP category and the rest say nothing, which
212
+ is the honest answer.
213
+
133
214
  | Source | What we use it for |
134
215
  |----------------------------------------------|----------------------------------------------------------|
135
216
  | [NIST SSDF SP 800-218](https://csrc.nist.gov/pubs/sp/800/218/final) | Process practice id (e.g. `PW.5.1`) |
@@ -375,6 +456,31 @@ SARIF upload. Pin production usage to a full commit SHA; the version tag above
375
456
  is shown for readability. See [`action.yml`](action.yml) and
376
457
  [`examples/github-actions/`](examples/github-actions/) for full workflows.
377
458
 
459
+ ## maintainability-agent integration
460
+
461
+ `maintainability-agent` declares Security a **delegated** pillar naming this
462
+ tool, and reports it as `NotApplicable` so silence is never read as safety.
463
+ This tool emits the artifact that completes the picture:
464
+
465
+ ```bash
466
+ secure-code-agent . --security-pillar security-pillar.json
467
+ maintainability-agent . --security-pillar security-pillar.json
468
+ ```
469
+
470
+ Standalone use is unaffected — without the flag nothing is written and every
471
+ other output is identical. MA never executes this tool, and this tool never
472
+ imports MA; two independently releasable packages exchanging one document.
473
+
474
+ The artifact carries **two values that are never averaged**: a practice level
475
+ read from configuration and CI (*is anything preventing the next
476
+ vulnerability?*) and a code condition read from the scanners (*what did they
477
+ find?*). `condition` is `null` whenever scanner coverage is incomplete — this
478
+ tool's score is a rate over findings, so removing scanners makes the raw number
479
+ go **up**, and an unscanned repository must not arrive at MA looking measured.
480
+
481
+ Full contract, invariants and the practice rubric:
482
+ [`docs/ma-integration.md`](docs/ma-integration.md).
483
+
378
484
  ## What this is NOT
379
485
 
380
486
  - ❌ **Not a SAST engine.** We delegate to Semgrep / Bandit / CodeQL / etc. — we don't write yet another AST analyzer.
@@ -398,12 +504,15 @@ Full design philosophy in [`docs/design.md`](docs/design.md).
398
504
 
399
505
  - [`docs/product-intent.md`](docs/product-intent.md) — Why this exists, who it serves, what it refuses to become
400
506
  - [`docs/decisions.md`](docs/decisions.md) — Decision register: rulings the code alone cannot answer
507
+ - [`docs/ma-integration.md`](docs/ma-integration.md) — The `security-pillar.json` contract maintainability-agent reads
508
+ - [`docs/calibration.md`](docs/calibration.md) — The calibration study, its corpus, and what it found
401
509
  - [`docs/design.md`](docs/design.md) — Architecture + non-goals + scanner protocol
402
510
  - [`docs/architecture.md`](docs/architecture.md) — Audit of the system as built + remediation sequence
403
511
  - [`docs/release-blockers.md`](docs/release-blockers.md) — Open v0.3.0 release blockers (do not tag until closed)
404
512
  - [`docs/standards.md`](docs/standards.md) — NIST SSDF / OWASP / CWE / Scorecard / SARIF citations
405
513
  - [`docs/scoring.md`](docs/scoring.md) — Weighting model + worked examples
406
514
  - [`docs/scanners.md`](docs/scanners.md) — Per-scanner integrations + caveats
515
+ - [`docs/work-orders.md`](docs/work-orders.md) — **The first-class output**: audit → work order → fix → verify → trend
407
516
  - [`docs/remediation.md`](docs/remediation.md) — The prompt template + failure-mode rationale
408
517
  - [`docs/threat-model.md`](docs/threat-model.md) — What we defend against (and what we don't)
409
518
 
@@ -13,7 +13,10 @@ src/secure_code_audit/cli.py
13
13
  src/secure_code_audit/config.py
14
14
  src/secure_code_audit/findings.py
15
15
  src/secure_code_audit/git_tools.py
16
+ src/secure_code_audit/history.py
16
17
  src/secure_code_audit/instructions.py
18
+ src/secure_code_audit/pillar.py
19
+ src/secure_code_audit/practice.py
17
20
  src/secure_code_audit/remediation.py
18
21
  src/secure_code_audit/renderers.py
19
22
  src/secure_code_audit/ruleset.py
@@ -22,6 +25,8 @@ src/secure_code_audit/scanner_status.py
22
25
  src/secure_code_audit/scoring.py
23
26
  src/secure_code_audit/standards.py
24
27
  src/secure_code_audit/suppressions.py
28
+ src/secure_code_audit/triage.py
29
+ src/secure_code_audit/verify.py
25
30
  src/secure_code_audit/data/semgrep-offline.yaml
26
31
  src/secure_code_audit/scanners/__init__.py
27
32
  src/secure_code_audit/scanners/bandit_scanner.py
@@ -0,0 +1,15 @@
1
+ """secure-code-agent — deterministic security gate + bounded AI remediation prompts."""
2
+
3
+ #: The only place the version is written. `pyproject.toml` derives it via
4
+ #: `[tool.setuptools.dynamic]`, so the two cannot drift.
5
+ #:
6
+ #: They did drift once, and it shipped: pyproject said 0.4.0 while this said
7
+ #: 0.3.0, so the released v0.4.0 wheel stamped 0.3.0 into every SARIF document,
8
+ #: every JSON report, the Markdown header, `--version`, and the
9
+ #: `security-pillar.json` handed to maintainability-agent. The release workflow
10
+ #: compared the tag against pyproject's line and never looked here — it
11
+ #: verified the half that was right and shipped the half that was wrong.
12
+ #:
13
+ #: PyPI is immutable, so 0.4.0 stays wrong. 0.5.0 is the first build whose
14
+ #: artifacts name their own producer correctly.
15
+ __version__ = "0.6.0"