torusguard 2.1.0 → 2.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.torusguard/.manifest.json +47 -5
- package/.torusguard/core/__init__.py +146 -0
- package/.torusguard/core/agent_roles.py +104 -0
- package/.torusguard/core/ast_walker.py +283 -0
- package/.torusguard/core/authorization.py +218 -0
- package/.torusguard/core/browser_verifier.py +128 -0
- package/.torusguard/core/bundle.py +141 -0
- package/.torusguard/core/call_graph.py +184 -0
- package/.torusguard/core/clustering.py +275 -0
- package/.torusguard/core/confidence.py +120 -0
- package/.torusguard/core/cross_file_taint.py +101 -0
- package/.torusguard/core/exploit_checker.py +317 -0
- package/.torusguard/core/formatter.py +351 -0
- package/.torusguard/core/governance.py +210 -0
- package/.torusguard/core/identity.py +104 -0
- package/.torusguard/core/import_resolver.py +91 -0
- package/.torusguard/core/incremental.py +102 -0
- package/.torusguard/core/lifecycle.py +137 -0
- package/.torusguard/core/models.py +425 -0
- package/.torusguard/core/parallel.py +56 -0
- package/.torusguard/core/parser.py +202 -0
- package/.torusguard/core/rechecker.py +107 -0
- package/.torusguard/core/replay_trace.py +178 -0
- package/.torusguard/core/rules_registry.py +131 -0
- package/.torusguard/core/run_folder.py +60 -0
- package/.torusguard/core/run_manager.py +163 -0
- package/.torusguard/core/runtime_evidence.py +175 -0
- package/.torusguard/core/runtime_validator.py +246 -0
- package/.torusguard/core/safety_gate.py +139 -0
- package/.torusguard/core/sarif.py +189 -0
- package/.torusguard/core/stack_profiler.py +184 -0
- package/.torusguard/core/symbol_table.py +91 -0
- package/.torusguard/core/taint.py +133 -0
- package/.torusguard/core/taint_graph.py +235 -0
- package/.torusguard/core/taint_rules.py +268 -0
- package/.torusguard/core/v070_reporter.py +102 -0
- package/.torusguard/core/v070_workflow.py +339 -0
- package/.torusguard/core/v6_reporter.py +180 -0
- package/.torusguard/core/v6_workflow.py +221 -0
- package/.torusguard/core/watcher.py +58 -0
- package/.torusguard/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
- package/.torusguard/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
- package/.torusguard/scripts/__pycache__/audit_runner.cpython-314.pyc +0 -0
- package/.torusguard/scripts/__pycache__/finding_scorer.cpython-314.pyc +0 -0
- package/.torusguard/scripts/__pycache__/rules_sync.cpython-314.pyc +0 -0
- package/.torusguard/scripts/audit_runner.py +108 -10
- package/.torusguard/scripts/finding_scorer.py +43 -13
- package/.torusguard/scripts/skill_profiler.py +26 -0
- package/.torusguard/skills/torusguard/SKILL.md +6 -2
- package/.torusguard/skills/torusguard-audit/SKILL.md +109 -84
- package/.torusguard/workflows/audit.md +21 -17
- package/README.md +19 -11
- package/package.json +7 -2
- package/skills/torusguard/SKILL.md +6 -2
- package/skills/torusguard/__pycache__/bootstrap.cpython-314.pyc +0 -0
- package/skills/torusguard/bootstrap.py +3 -3
- package/skills/torusguard/payload/.manifest.json +48 -7
- package/skills/torusguard/payload/core/__init__.py +146 -0
- package/skills/torusguard/payload/core/agent_roles.py +104 -0
- package/skills/torusguard/payload/core/ast_walker.py +283 -0
- package/skills/torusguard/payload/core/authorization.py +218 -0
- package/skills/torusguard/payload/core/browser_verifier.py +128 -0
- package/skills/torusguard/payload/core/bundle.py +141 -0
- package/skills/torusguard/payload/core/call_graph.py +184 -0
- package/skills/torusguard/payload/core/clustering.py +275 -0
- package/skills/torusguard/payload/core/confidence.py +120 -0
- package/skills/torusguard/payload/core/cross_file_taint.py +101 -0
- package/skills/torusguard/payload/core/exploit_checker.py +317 -0
- package/skills/torusguard/payload/core/formatter.py +351 -0
- package/skills/torusguard/payload/core/governance.py +210 -0
- package/skills/torusguard/payload/core/identity.py +104 -0
- package/skills/torusguard/payload/core/import_resolver.py +91 -0
- package/skills/torusguard/payload/core/incremental.py +102 -0
- package/skills/torusguard/payload/core/lifecycle.py +137 -0
- package/skills/torusguard/payload/core/models.py +425 -0
- package/skills/torusguard/payload/core/parallel.py +56 -0
- package/skills/torusguard/payload/core/parser.py +202 -0
- package/skills/torusguard/payload/core/rechecker.py +107 -0
- package/skills/torusguard/payload/core/replay_trace.py +178 -0
- package/skills/torusguard/payload/core/rules_registry.py +131 -0
- package/skills/torusguard/payload/core/run_folder.py +60 -0
- package/skills/torusguard/payload/core/run_manager.py +163 -0
- package/skills/torusguard/payload/core/runtime_evidence.py +175 -0
- package/skills/torusguard/payload/core/runtime_validator.py +246 -0
- package/skills/torusguard/payload/core/safety_gate.py +139 -0
- package/skills/torusguard/payload/core/sarif.py +189 -0
- package/skills/torusguard/payload/core/stack_profiler.py +184 -0
- package/skills/torusguard/payload/core/symbol_table.py +91 -0
- package/skills/torusguard/payload/core/taint.py +133 -0
- package/skills/torusguard/payload/core/taint_graph.py +235 -0
- package/skills/torusguard/payload/core/taint_rules.py +268 -0
- package/skills/torusguard/payload/core/v070_reporter.py +102 -0
- package/skills/torusguard/payload/core/v070_workflow.py +339 -0
- package/skills/torusguard/payload/core/v6_reporter.py +180 -0
- package/skills/torusguard/payload/core/v6_workflow.py +221 -0
- package/skills/torusguard/payload/core/watcher.py +58 -0
- package/skills/torusguard/payload/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
- package/skills/torusguard/payload/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
- package/skills/torusguard/payload/rules/container/TG-CONT-001-root-user-execution.md +50 -50
- package/skills/torusguard/payload/rules/container/TG-CONT-002-docker-socket-mount.md +47 -47
- package/skills/torusguard/payload/rules/container/TG-CONT-003-privileged-container-mode.md +53 -53
- package/skills/torusguard/payload/rules/container/TG-CONT-004-build-arg-secret-exposure.md +43 -43
- package/skills/torusguard/payload/rules/git/TG-GIT-001-historical-secret-in-git-commit.md +44 -44
- package/skills/torusguard/payload/rules/git/TG-GIT-002-plaintext-credentials-in-git-config.md +41 -41
- package/skills/torusguard/payload/rules/git/TG-GIT-003-sensitive-tracked-file-gitignore-breach.md +40 -40
- package/skills/torusguard/payload/rules/rag/TG-RAG-001-untrusted-rag-context-injection.md +72 -72
- package/skills/torusguard/payload/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md +51 -51
- package/skills/torusguard/payload/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md +51 -51
- package/skills/torusguard/payload/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md +46 -46
- package/skills/torusguard/payload/rules/redos/TG-REDOS-002-unbounded-nested-quantifier.md +43 -43
- package/skills/torusguard/payload/scripts/audit_runner.py +108 -10
- package/skills/torusguard/payload/scripts/finding_scorer.py +43 -13
- package/skills/torusguard/payload/skills/torusguard/SKILL.md +6 -2
- package/skills/torusguard/payload/skills/torusguard/bootstrap.py +3 -3
- package/skills/torusguard/payload/skills/torusguard-ai-guard/SKILL.md +95 -95
- package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +109 -84
- package/skills/torusguard/payload/skills/torusguard-container/SKILL.md +94 -94
- package/skills/torusguard/payload/skills/torusguard-git-mine/SKILL.md +92 -92
- package/skills/torusguard/payload/skills/torusguard-ocr-scan/SKILL.md +94 -94
- package/skills/torusguard/payload/skills/torusguard-redos/SKILL.md +91 -91
- package/skills/torusguard/payload/workflows/ai-guard.md +31 -31
- package/skills/torusguard/payload/workflows/audit.md +21 -17
- package/skills/torusguard/payload/workflows/container.md +29 -29
- package/skills/torusguard/payload/workflows/git-mine.md +25 -25
- package/skills/torusguard/payload/workflows/ocr-scan.md +25 -25
- package/skills/torusguard/payload/workflows/redos.md +27 -27
- package/skills/torusguard/payload/workflows/torusguard-audit.md +35 -55
- package/skills/torusguard/references/csharp-security.md +41 -41
- package/skills/torusguard/references/go-security.md +41 -41
- package/skills/torusguard/references/java-security.md +40 -40
- package/skills/torusguard/references/polyglot-security-matrix.md +25 -25
- package/skills/torusguard/references/rust-security.md +40 -40
- package/skills/torusguard-audit/SKILL.md +107 -83
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TorusGuard Core Data Models (v0.5.4)
|
|
3
|
+
Defines canonical Finding, ProvenanceChain, ConfidenceScore, EvidencePackage, RetestRecord, RemediationPriority, and AuditReport objects.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from dataclasses import dataclass, field, asdict
|
|
7
|
+
from enum import Enum
|
|
8
|
+
from typing import List, Optional, Dict, Any
|
|
9
|
+
import datetime
|
|
10
|
+
import hashlib
|
|
11
|
+
import uuid
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class SeverityLevel(str, Enum):
|
|
16
|
+
CRITICAL = "Critical"
|
|
17
|
+
HIGH = "High"
|
|
18
|
+
MEDIUM = "Medium"
|
|
19
|
+
LOW = "Low"
|
|
20
|
+
INFORMATIONAL = "Informational"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class RemediationPriority(str, Enum):
|
|
24
|
+
IMMEDIATE = "Immediate (P0)" # Block deployment / immediate fix
|
|
25
|
+
NEAR_TERM = "Near-Term (P1)" # Fix in current sprint / patch cycle
|
|
26
|
+
BACKLOG = "Backlog (P2)" # Defense-in-depth hardening backlog
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ConfidenceBand(str, Enum):
|
|
30
|
+
CONFIRMED = "Confirmed" # 90 - 100
|
|
31
|
+
HIGH_CONFIDENCE = "High Confidence" # 70 - 89
|
|
32
|
+
MEDIUM_CONFIDENCE = "Medium Confidence" # 50 - 69
|
|
33
|
+
LOW_CONFIDENCE = "Low Confidence" # < 50
|
|
34
|
+
UNCONFIRMED = "Unconfirmed"
|
|
35
|
+
NEEDS_REVIEW = "Needs Review"
|
|
36
|
+
INFORMATIONAL = "Informational"
|
|
37
|
+
NOT_APPLICABLE = "Not Applicable"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class FindingStatus(str, Enum):
|
|
41
|
+
CONFIRMED = "Confirmed"
|
|
42
|
+
HIGH_CONFIDENCE = "High Confidence"
|
|
43
|
+
MEDIUM_CONFIDENCE = "Medium Confidence"
|
|
44
|
+
LOW_CONFIDENCE = "Low Confidence"
|
|
45
|
+
UNCONFIRMED = "Unconfirmed"
|
|
46
|
+
NEEDS_REVIEW = "Needs Review"
|
|
47
|
+
REMEDIATED = "Remediated"
|
|
48
|
+
VERIFIED_FIXED = "Verified Fixed"
|
|
49
|
+
SUPPRESSED = "Suppressed"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class LifecycleStage(str, Enum):
|
|
53
|
+
DETECT = "Detect"
|
|
54
|
+
CLASSIFY = "Classify"
|
|
55
|
+
VERIFY = "Verify"
|
|
56
|
+
REMEDIATE = "Remediate"
|
|
57
|
+
RECHECK = "Recheck"
|
|
58
|
+
ARCHIVE = "Archive"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class TaxonomyCategory(str, Enum):
|
|
62
|
+
AUTH = "authentication-authorization"
|
|
63
|
+
INPUT = "input-validation-encoding"
|
|
64
|
+
DATA = "data-access-orm"
|
|
65
|
+
FILES = "file-upload-handling"
|
|
66
|
+
SECRETS = "secrets-configuration"
|
|
67
|
+
SUPPLY_CHAIN = "dependency-supply-chain"
|
|
68
|
+
NETWORK = "network-ssrf-boundaries"
|
|
69
|
+
BUSINESS_LOGIC = "business-logic-rate-limiting"
|
|
70
|
+
CLIENT_PLATFORM = "client-cache-platform"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class EvidenceType(str, Enum):
|
|
74
|
+
SOURCE = "source"
|
|
75
|
+
RUNTIME = "runtime"
|
|
76
|
+
TEST = "test"
|
|
77
|
+
MANUAL_REVIEW = "manual_review"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class ProvenanceChain:
|
|
82
|
+
discovery_module: str
|
|
83
|
+
triggering_input: str
|
|
84
|
+
evidence_collected: List[str]
|
|
85
|
+
decision_path: List[str]
|
|
86
|
+
verification_step: str
|
|
87
|
+
agent_environment: str = "TorusGuard Engine v0.5.4"
|
|
88
|
+
timestamp: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
89
|
+
|
|
90
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
91
|
+
return asdict(self)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass
|
|
95
|
+
class ConfidenceFactors:
|
|
96
|
+
evidence_quality: int # Max: 35
|
|
97
|
+
reproduction_success: int # Max: 25
|
|
98
|
+
independent_confirmations: int # Max: 15
|
|
99
|
+
environmental_clarity: int # Max: 15
|
|
100
|
+
manual_review_status: int # Max: 10
|
|
101
|
+
|
|
102
|
+
def total_score(self) -> int:
|
|
103
|
+
return max(0, min(100, (
|
|
104
|
+
self.evidence_quality +
|
|
105
|
+
self.reproduction_success +
|
|
106
|
+
self.independent_confirmations +
|
|
107
|
+
self.environmental_clarity +
|
|
108
|
+
self.manual_review_status
|
|
109
|
+
)))
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@dataclass
|
|
113
|
+
class ConfidenceScore:
|
|
114
|
+
factors: ConfidenceFactors
|
|
115
|
+
score: int = field(init=False)
|
|
116
|
+
band: ConfidenceBand = field(init=False)
|
|
117
|
+
rationale: str = ""
|
|
118
|
+
|
|
119
|
+
def __post_init__(self):
|
|
120
|
+
self.score = self.factors.total_score()
|
|
121
|
+
if self.score >= 90:
|
|
122
|
+
self.band = ConfidenceBand.CONFIRMED
|
|
123
|
+
elif self.score >= 70:
|
|
124
|
+
self.band = ConfidenceBand.HIGH_CONFIDENCE
|
|
125
|
+
elif self.score >= 50:
|
|
126
|
+
self.band = ConfidenceBand.MEDIUM_CONFIDENCE
|
|
127
|
+
else:
|
|
128
|
+
self.band = ConfidenceBand.LOW_CONFIDENCE
|
|
129
|
+
|
|
130
|
+
@staticmethod
|
|
131
|
+
def calculate(
|
|
132
|
+
evidence_quality: int = 30,
|
|
133
|
+
reproduction_success: int = 25,
|
|
134
|
+
independent_confirmations: int = 15,
|
|
135
|
+
environmental_clarity: int = 15,
|
|
136
|
+
manual_review_status: int = 5,
|
|
137
|
+
rationale: str = ""
|
|
138
|
+
) -> 'ConfidenceScore':
|
|
139
|
+
factors = ConfidenceFactors(
|
|
140
|
+
evidence_quality=evidence_quality,
|
|
141
|
+
reproduction_success=reproduction_success,
|
|
142
|
+
independent_confirmations=independent_confirmations,
|
|
143
|
+
environmental_clarity=environmental_clarity,
|
|
144
|
+
manual_review_status=manual_review_status,
|
|
145
|
+
)
|
|
146
|
+
return ConfidenceScore(factors=factors, rationale=rationale)
|
|
147
|
+
|
|
148
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
149
|
+
return {
|
|
150
|
+
"score": self.score,
|
|
151
|
+
"band": self.band.value,
|
|
152
|
+
"factors": asdict(self.factors),
|
|
153
|
+
"rationale": self.rationale,
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def mask_sensitive_data(text: str) -> str:
|
|
158
|
+
"""Masks secrets, tokens, API keys, and passwords from report output."""
|
|
159
|
+
text = re.sub(r'sk_live_[0-9a-zA-Z_\-]{6,}', 'sk_live_***REDACTED***', text)
|
|
160
|
+
text = re.sub(r'ghp_[0-9a-zA-Z_\-]{6,}', 'ghp_***REDACTED***', text)
|
|
161
|
+
text = re.sub(r'(Bearer\s+)[A-Za-z0-9\-_=]+\.[A-Za-z0-9\-_=]+\.?[A-Za-z0-9\-_=]*', r'\1***REDACTED_JWT***', text)
|
|
162
|
+
|
|
163
|
+
def redact_kv(m):
|
|
164
|
+
val = m.group(3)
|
|
165
|
+
if "***REDACTED" in val:
|
|
166
|
+
return m.group(0)
|
|
167
|
+
return f"{m.group(1)}{m.group(2)}***REDACTED***{m.group(4)}"
|
|
168
|
+
|
|
169
|
+
text = re.sub(
|
|
170
|
+
r'(?i)(secret[_\-\w]*|password|api[_\-\w]*key|token|auth[_\-\w]*key)(\s*[:=]\s*[\'"])([^\'"]{4,})([\'"])',
|
|
171
|
+
redact_kv,
|
|
172
|
+
text
|
|
173
|
+
)
|
|
174
|
+
return text
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@dataclass
|
|
178
|
+
class Evidence:
|
|
179
|
+
type: EvidenceType
|
|
180
|
+
location: str
|
|
181
|
+
raw_snippet: str
|
|
182
|
+
rationale: str
|
|
183
|
+
confidence_level: ConfidenceBand
|
|
184
|
+
sha256_checksum: str = field(init=False)
|
|
185
|
+
context: Optional[str] = None
|
|
186
|
+
reproduction_notes: Optional[str] = None
|
|
187
|
+
reviewer_notes: Optional[str] = None
|
|
188
|
+
is_sufficient_for_confirmed: bool = False
|
|
189
|
+
collected_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
190
|
+
|
|
191
|
+
def __post_init__(self):
|
|
192
|
+
self.sha256_checksum = hashlib.sha256(self.raw_snippet.strip().encode("utf-8")).hexdigest()
|
|
193
|
+
|
|
194
|
+
def get_masked_snippet(self) -> str:
|
|
195
|
+
return mask_sensitive_data(self.raw_snippet)
|
|
196
|
+
|
|
197
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
198
|
+
return {
|
|
199
|
+
"type": self.type.value if isinstance(self.type, EvidenceType) else self.type,
|
|
200
|
+
"location": self.location,
|
|
201
|
+
"raw_snippet": self.get_masked_snippet(),
|
|
202
|
+
"sha256_checksum": self.sha256_checksum,
|
|
203
|
+
"collected_at": self.collected_at,
|
|
204
|
+
"context": self.context,
|
|
205
|
+
"rationale": self.rationale,
|
|
206
|
+
"confidence_level": self.confidence_level.value if isinstance(self.confidence_level, ConfidenceBand) else self.confidence_level,
|
|
207
|
+
"reproduction_notes": self.reproduction_notes,
|
|
208
|
+
"reviewer_notes": self.reviewer_notes,
|
|
209
|
+
"is_sufficient_for_confirmed": self.is_sufficient_for_confirmed,
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
@dataclass
|
|
214
|
+
class SeverityInfo:
|
|
215
|
+
level: SeverityLevel
|
|
216
|
+
rationale: str
|
|
217
|
+
rubric_justification: str
|
|
218
|
+
|
|
219
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
220
|
+
return {
|
|
221
|
+
"level": self.level.value if isinstance(self.level, SeverityLevel) else self.level,
|
|
222
|
+
"rationale": self.rationale,
|
|
223
|
+
"rubric_justification": self.rubric_justification,
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
@dataclass
|
|
228
|
+
class FrameworkPattern:
|
|
229
|
+
framework: str
|
|
230
|
+
unsafe_snippet: str
|
|
231
|
+
safe_snippet: str
|
|
232
|
+
least_invasive: bool = True
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
@dataclass
|
|
236
|
+
class Remediation:
|
|
237
|
+
problem_statement: str
|
|
238
|
+
risk_explanation: str
|
|
239
|
+
recommended_fix: str
|
|
240
|
+
framework_pattern: FrameworkPattern
|
|
241
|
+
verification_method: str
|
|
242
|
+
residual_risk_notes: str
|
|
243
|
+
|
|
244
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
245
|
+
return {
|
|
246
|
+
"problem_statement": self.problem_statement,
|
|
247
|
+
"risk_explanation": self.risk_explanation,
|
|
248
|
+
"recommended_fix": self.recommended_fix,
|
|
249
|
+
"framework_pattern": asdict(self.framework_pattern),
|
|
250
|
+
"verification_method": self.verification_method,
|
|
251
|
+
"residual_risk_notes": self.residual_risk_notes,
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
@dataclass
|
|
256
|
+
class AffectedComponent:
|
|
257
|
+
component_name: str
|
|
258
|
+
target_path: str
|
|
259
|
+
start_line: Optional[int] = None
|
|
260
|
+
end_line: Optional[int] = None
|
|
261
|
+
symbol: Optional[str] = None
|
|
262
|
+
|
|
263
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
264
|
+
return {k: v for k, v in asdict(self).items() if v is not None}
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
@dataclass
|
|
268
|
+
class ReproductionMethod:
|
|
269
|
+
step_by_step: List[str]
|
|
270
|
+
deterministic: bool = True
|
|
271
|
+
test_command: Optional[str] = None
|
|
272
|
+
expected_failure_response: Optional[str] = None
|
|
273
|
+
|
|
274
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
275
|
+
return {k: v for k, v in asdict(self).items() if v is not None}
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
@dataclass
|
|
279
|
+
class RetestRecord:
|
|
280
|
+
retest_performed: bool = False
|
|
281
|
+
closure_status: FindingStatus = FindingStatus.UNCONFIRMED
|
|
282
|
+
fix_applied: Optional[str] = None
|
|
283
|
+
retest_method: Optional[str] = None
|
|
284
|
+
retest_evidence_hash: Optional[str] = None
|
|
285
|
+
residual_risk: Optional[str] = None
|
|
286
|
+
verifier_notes: Optional[str] = None
|
|
287
|
+
retest_timestamp: Optional[str] = None
|
|
288
|
+
|
|
289
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
290
|
+
d = asdict(self)
|
|
291
|
+
d["closure_status"] = self.closure_status.value if isinstance(self.closure_status, FindingStatus) else self.closure_status
|
|
292
|
+
return {k: v for k, v in d.items() if v is not None}
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
@dataclass
|
|
296
|
+
class NotesRecord:
|
|
297
|
+
business_impact: str
|
|
298
|
+
technical_description: str
|
|
299
|
+
raw_facts_summary: str
|
|
300
|
+
ai_interpretation: str
|
|
301
|
+
|
|
302
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
303
|
+
return asdict(self)
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
@dataclass
|
|
307
|
+
class FindingTimestamps:
|
|
308
|
+
discovered_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
309
|
+
updated_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
310
|
+
verified_at: Optional[str] = None
|
|
311
|
+
remediated_at: Optional[str] = None
|
|
312
|
+
retested_at: Optional[str] = None
|
|
313
|
+
|
|
314
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
315
|
+
return {k: v for k, v in asdict(self).items() if v is not None}
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
@dataclass
|
|
319
|
+
class Finding:
|
|
320
|
+
rule_id: str
|
|
321
|
+
title: str
|
|
322
|
+
category: TaxonomyCategory
|
|
323
|
+
severity: SeverityInfo
|
|
324
|
+
confidence: ConfidenceScore
|
|
325
|
+
status: FindingStatus
|
|
326
|
+
affected_component: AffectedComponent
|
|
327
|
+
evidence: List[Evidence]
|
|
328
|
+
provenance: ProvenanceChain
|
|
329
|
+
reproduction_method: ReproductionMethod
|
|
330
|
+
remediation: Remediation
|
|
331
|
+
remediation_priority: RemediationPriority = RemediationPriority.IMMEDIATE
|
|
332
|
+
retest_result: RetestRecord = field(default_factory=RetestRecord)
|
|
333
|
+
timestamps: FindingTimestamps = field(default_factory=FindingTimestamps)
|
|
334
|
+
notes: NotesRecord = field(default_factory=lambda: NotesRecord(
|
|
335
|
+
business_impact="Exposure of sensitive application resources or unauthorized data modification.",
|
|
336
|
+
technical_description="Direct unmitigated pattern detected in application route or data layer.",
|
|
337
|
+
raw_facts_summary="Unmitigated pattern identified in source.",
|
|
338
|
+
ai_interpretation="High priority fix recommended.",
|
|
339
|
+
))
|
|
340
|
+
finding_id: str = field(default_factory=lambda: f"TG-FIND-{datetime.datetime.utcnow().year}-{uuid.uuid4().hex[:6]}")
|
|
341
|
+
lifecycle_stage: LifecycleStage = LifecycleStage.DETECT
|
|
342
|
+
asvs_control: Optional[str] = None
|
|
343
|
+
cwe: Optional[str] = None
|
|
344
|
+
nist_ssdf: Optional[str] = None
|
|
345
|
+
|
|
346
|
+
def __post_init__(self):
|
|
347
|
+
# Auto-derive remediation priority from severity level if default
|
|
348
|
+
if self.severity.level == SeverityLevel.CRITICAL:
|
|
349
|
+
self.remediation_priority = RemediationPriority.IMMEDIATE
|
|
350
|
+
elif self.severity.level == SeverityLevel.HIGH:
|
|
351
|
+
self.remediation_priority = RemediationPriority.NEAR_TERM
|
|
352
|
+
else:
|
|
353
|
+
self.remediation_priority = RemediationPriority.BACKLOG
|
|
354
|
+
|
|
355
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
356
|
+
return {
|
|
357
|
+
"finding_id": self.finding_id,
|
|
358
|
+
"rule_id": self.rule_id,
|
|
359
|
+
"title": self.title,
|
|
360
|
+
"category": self.category.value if isinstance(self.category, TaxonomyCategory) else self.category,
|
|
361
|
+
"severity": self.severity.to_dict(),
|
|
362
|
+
"confidence": self.confidence.to_dict(),
|
|
363
|
+
"status": self.status.value if isinstance(self.status, FindingStatus) else self.status,
|
|
364
|
+
"remediation_priority": self.remediation_priority.value if isinstance(self.remediation_priority, RemediationPriority) else self.remediation_priority,
|
|
365
|
+
"lifecycle_stage": self.lifecycle_stage.value if isinstance(self.lifecycle_stage, LifecycleStage) else self.lifecycle_stage,
|
|
366
|
+
"affected_component": self.affected_component.to_dict(),
|
|
367
|
+
"evidence": [e.to_dict() for e in self.evidence],
|
|
368
|
+
"provenance": self.provenance.to_dict(),
|
|
369
|
+
"reproduction_method": self.reproduction_method.to_dict(),
|
|
370
|
+
"remediation": self.remediation.to_dict(),
|
|
371
|
+
"retest_result": self.retest_result.to_dict(),
|
|
372
|
+
"requirement_reference": {
|
|
373
|
+
"asvs_v4": self.asvs_control,
|
|
374
|
+
"cwe": self.cwe,
|
|
375
|
+
"nist_ssdf": self.nist_ssdf,
|
|
376
|
+
},
|
|
377
|
+
"cwe": self.cwe,
|
|
378
|
+
"timestamps": self.timestamps.to_dict(),
|
|
379
|
+
"notes": self.notes.to_dict(),
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
@dataclass
|
|
384
|
+
class AuditReport:
|
|
385
|
+
project_name: str
|
|
386
|
+
detected_stack: Dict[str, Any]
|
|
387
|
+
findings: List[Finding]
|
|
388
|
+
summary_counts: Dict[str, Any] = field(default_factory=dict)
|
|
389
|
+
generated_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
390
|
+
torusguard_version: str = "v0.5.4"
|
|
391
|
+
report_owner: str = "TorusGuard Security Subsystem"
|
|
392
|
+
repository_ref: str = "workspace"
|
|
393
|
+
|
|
394
|
+
def calculate_summary(self) -> None:
|
|
395
|
+
total = len(self.findings)
|
|
396
|
+
avg_confidence = round(sum(f.confidence.score for f in self.findings) / total, 1) if total > 0 else 100.0
|
|
397
|
+
self.summary_counts = {
|
|
398
|
+
"total_findings": total,
|
|
399
|
+
"average_confidence_score": avg_confidence,
|
|
400
|
+
"critical": sum(1 for f in self.findings if f.severity.level == SeverityLevel.CRITICAL),
|
|
401
|
+
"high": sum(1 for f in self.findings if f.severity.level == SeverityLevel.HIGH),
|
|
402
|
+
"medium": sum(1 for f in self.findings if f.severity.level == SeverityLevel.MEDIUM),
|
|
403
|
+
"low": sum(1 for f in self.findings if f.severity.level == SeverityLevel.LOW),
|
|
404
|
+
"confirmed": sum(1 for f in self.findings if f.confidence.band == ConfidenceBand.CONFIRMED),
|
|
405
|
+
"high_confidence": sum(1 for f in self.findings if f.confidence.band == ConfidenceBand.HIGH_CONFIDENCE),
|
|
406
|
+
"needs_review": sum(1 for f in self.findings if f.confidence.band in (ConfidenceBand.NEEDS_REVIEW, ConfidenceBand.LOW_CONFIDENCE)),
|
|
407
|
+
"verified_fixed": sum(1 for f in self.findings if f.status == FindingStatus.VERIFIED_FIXED),
|
|
408
|
+
"remediated": sum(1 for f in self.findings if f.status in (FindingStatus.REMEDIATED, FindingStatus.VERIFIED_FIXED)),
|
|
409
|
+
"immediate_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.IMMEDIATE),
|
|
410
|
+
"near_term_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.NEAR_TERM),
|
|
411
|
+
"backlog_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.BACKLOG),
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
415
|
+
self.calculate_summary()
|
|
416
|
+
return {
|
|
417
|
+
"project_name": self.project_name,
|
|
418
|
+
"torusguard_version": self.torusguard_version,
|
|
419
|
+
"generated_at": self.generated_at,
|
|
420
|
+
"report_owner": self.report_owner,
|
|
421
|
+
"repository_ref": self.repository_ref,
|
|
422
|
+
"detected_stack": self.detected_stack,
|
|
423
|
+
"summary": self.summary_counts,
|
|
424
|
+
"findings": [f.to_dict() for f in self.findings],
|
|
425
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TorusGuard Parallel Audit Executor
|
|
3
|
+
Executes multi-threaded static security scanning across files with deterministic result collation.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import List, Dict, Any, Callable, Optional
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ParallelAuditExecutor:
|
|
13
|
+
"""Distributes file analysis across a thread pool with deterministic output ordering."""
|
|
14
|
+
|
|
15
|
+
def __init__(self, max_workers: Optional[int] = None):
|
|
16
|
+
self.max_workers = max_workers or min(os.cpu_count() or 4, 8)
|
|
17
|
+
|
|
18
|
+
def scan_files_parallel(
|
|
19
|
+
self,
|
|
20
|
+
files: List[Path],
|
|
21
|
+
scan_fn: Callable[[Path], List[Dict[str, Any]]]
|
|
22
|
+
) -> List[Dict[str, Any]]:
|
|
23
|
+
"""
|
|
24
|
+
Executes `scan_fn` across all files in parallel.
|
|
25
|
+
Returns deterministically ordered list of all findings.
|
|
26
|
+
"""
|
|
27
|
+
if not files:
|
|
28
|
+
return []
|
|
29
|
+
|
|
30
|
+
# If only a few files, avoid thread pool overhead
|
|
31
|
+
if len(files) <= 3:
|
|
32
|
+
all_findings = []
|
|
33
|
+
for f in files:
|
|
34
|
+
all_findings.extend(scan_fn(f))
|
|
35
|
+
return all_findings
|
|
36
|
+
|
|
37
|
+
all_findings: List[Dict[str, Any]] = []
|
|
38
|
+
file_results: Dict[str, List[Dict[str, Any]]] = {}
|
|
39
|
+
|
|
40
|
+
with ThreadPoolExecutor(max_workers=self.max_workers) as executor:
|
|
41
|
+
future_to_file = {executor.submit(scan_fn, f): str(f) for f in files}
|
|
42
|
+
for future in as_completed(future_to_file):
|
|
43
|
+
f_str = future_to_file[future]
|
|
44
|
+
try:
|
|
45
|
+
res = future.result()
|
|
46
|
+
file_results[f_str] = res
|
|
47
|
+
except Exception:
|
|
48
|
+
file_results[f_str] = []
|
|
49
|
+
|
|
50
|
+
# Deterministic sort by original file order
|
|
51
|
+
for f in files:
|
|
52
|
+
f_str = str(f)
|
|
53
|
+
if f_str in file_results:
|
|
54
|
+
all_findings.extend(file_results[f_str])
|
|
55
|
+
|
|
56
|
+
return all_findings
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TorusGuard Polyglot Parser
|
|
3
|
+
Unified Tree-sitter parser wrapper with language auto-detection and resilient
|
|
4
|
+
fallback parsing across Python, JavaScript, TypeScript, Go, Rust, Java, Ruby, PHP, and C#.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import Dict, List, Optional, Any
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
from core.ast_walker import FunctionCall, Assignment, ImportStatement, StringLiteral, TreeSitterWalker
|
|
13
|
+
from core.symbol_table import SymbolTable
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
LANGUAGE_EXTENSIONS: Dict[str, str] = {
|
|
17
|
+
".py": "python",
|
|
18
|
+
".js": "javascript",
|
|
19
|
+
".mjs": "javascript",
|
|
20
|
+
".cjs": "javascript",
|
|
21
|
+
".ts": "typescript",
|
|
22
|
+
".tsx": "tsx",
|
|
23
|
+
".go": "go",
|
|
24
|
+
".rs": "rust",
|
|
25
|
+
".java": "java",
|
|
26
|
+
".rb": "ruby",
|
|
27
|
+
".php": "php",
|
|
28
|
+
".cs": "csharp",
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class ParseResult:
|
|
34
|
+
file_path: str
|
|
35
|
+
language: str
|
|
36
|
+
function_calls: List[FunctionCall] = field(default_factory=list)
|
|
37
|
+
assignments: List[Assignment] = field(default_factory=list)
|
|
38
|
+
imports: List[ImportStatement] = field(default_factory=list)
|
|
39
|
+
string_literals: List[StringLiteral] = field(default_factory=list)
|
|
40
|
+
symbol_table: SymbolTable = field(default_factory=SymbolTable)
|
|
41
|
+
is_tree_sitter: bool = False
|
|
42
|
+
raw_content: str = ""
|
|
43
|
+
|
|
44
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
45
|
+
return {
|
|
46
|
+
"file_path": self.file_path,
|
|
47
|
+
"language": self.language,
|
|
48
|
+
"function_calls_count": len(self.function_calls),
|
|
49
|
+
"assignments_count": len(self.assignments),
|
|
50
|
+
"imports_count": len(self.imports),
|
|
51
|
+
"is_tree_sitter": self.is_tree_sitter,
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class PolyglotParser:
|
|
56
|
+
"""Parses source files into Tree-sitter ASTs or fallback token models with language auto-detection."""
|
|
57
|
+
|
|
58
|
+
def __init__(self):
|
|
59
|
+
self._parsers: Dict[str, Any] = {}
|
|
60
|
+
self._init_tree_sitter()
|
|
61
|
+
|
|
62
|
+
def _init_tree_sitter(self):
|
|
63
|
+
try:
|
|
64
|
+
import tree_sitter
|
|
65
|
+
# Python
|
|
66
|
+
try:
|
|
67
|
+
import tree_sitter_python
|
|
68
|
+
py_lang = tree_sitter.Language(tree_sitter_python.language())
|
|
69
|
+
self._parsers["python"] = tree_sitter.Parser(py_lang)
|
|
70
|
+
except Exception:
|
|
71
|
+
pass
|
|
72
|
+
|
|
73
|
+
# JavaScript
|
|
74
|
+
try:
|
|
75
|
+
import tree_sitter_javascript
|
|
76
|
+
js_lang = tree_sitter.Language(tree_sitter_javascript.language())
|
|
77
|
+
self._parsers["javascript"] = tree_sitter.Parser(js_lang)
|
|
78
|
+
except Exception:
|
|
79
|
+
pass
|
|
80
|
+
|
|
81
|
+
# TypeScript / TSX
|
|
82
|
+
try:
|
|
83
|
+
import tree_sitter_typescript
|
|
84
|
+
ts_lang = tree_sitter.Language(tree_sitter_typescript.language_typescript())
|
|
85
|
+
self._parsers["typescript"] = tree_sitter.Parser(ts_lang)
|
|
86
|
+
tsx_lang = tree_sitter.Language(tree_sitter_typescript.language_tsx())
|
|
87
|
+
self._parsers["tsx"] = tree_sitter.Parser(tsx_lang)
|
|
88
|
+
except Exception:
|
|
89
|
+
pass
|
|
90
|
+
|
|
91
|
+
# Go
|
|
92
|
+
try:
|
|
93
|
+
import tree_sitter_go
|
|
94
|
+
go_lang = tree_sitter.Language(tree_sitter_go.language())
|
|
95
|
+
self._parsers["go"] = tree_sitter.Parser(go_lang)
|
|
96
|
+
except Exception:
|
|
97
|
+
pass
|
|
98
|
+
|
|
99
|
+
except Exception:
|
|
100
|
+
pass
|
|
101
|
+
|
|
102
|
+
def detect_language(self, file_path: Path) -> str:
|
|
103
|
+
ext = file_path.suffix.lower()
|
|
104
|
+
return LANGUAGE_EXTENSIONS.get(ext, "unknown")
|
|
105
|
+
|
|
106
|
+
def parse_file(self, file_path: Path) -> ParseResult:
|
|
107
|
+
lang = self.detect_language(file_path)
|
|
108
|
+
try:
|
|
109
|
+
content = file_path.read_text(encoding="utf-8", errors="replace")
|
|
110
|
+
except Exception:
|
|
111
|
+
return ParseResult(file_path=str(file_path), language=lang)
|
|
112
|
+
|
|
113
|
+
source_bytes = content.encode("utf-8")
|
|
114
|
+
ts_parser = self._parsers.get(lang)
|
|
115
|
+
|
|
116
|
+
if ts_parser is not None:
|
|
117
|
+
try:
|
|
118
|
+
tree = ts_parser.parse(source_bytes)
|
|
119
|
+
calls = TreeSitterWalker.extract_function_calls(tree.root_node, source_bytes)
|
|
120
|
+
assigns = TreeSitterWalker.extract_assignments(tree.root_node, source_bytes)
|
|
121
|
+
imports = TreeSitterWalker.extract_imports(tree.root_node, source_bytes, lang)
|
|
122
|
+
strings = TreeSitterWalker.extract_string_literals(tree.root_node, source_bytes)
|
|
123
|
+
|
|
124
|
+
sym_table = SymbolTable(str(file_path))
|
|
125
|
+
for a in assigns:
|
|
126
|
+
sym_table.add_variable(name=a.target, line=a.line_number, expression=a.value_expression)
|
|
127
|
+
for imp in imports:
|
|
128
|
+
for alias, orig in imp.alias_map.items():
|
|
129
|
+
sym_table.add_import_alias(alias, orig)
|
|
130
|
+
|
|
131
|
+
return ParseResult(
|
|
132
|
+
file_path=str(file_path),
|
|
133
|
+
language=lang,
|
|
134
|
+
function_calls=calls,
|
|
135
|
+
assignments=assigns,
|
|
136
|
+
imports=imports,
|
|
137
|
+
string_literals=strings,
|
|
138
|
+
symbol_table=sym_table,
|
|
139
|
+
is_tree_sitter=True,
|
|
140
|
+
raw_content=content
|
|
141
|
+
)
|
|
142
|
+
except Exception:
|
|
143
|
+
pass
|
|
144
|
+
|
|
145
|
+
# Resilient Fallback Parser
|
|
146
|
+
return self._fallback_parse(str(file_path), content, lang)
|
|
147
|
+
|
|
148
|
+
def _fallback_parse(self, file_path: str, content: str, lang: str) -> ParseResult:
|
|
149
|
+
lines = content.splitlines()
|
|
150
|
+
calls: List[FunctionCall] = []
|
|
151
|
+
assigns: List[Assignment] = []
|
|
152
|
+
imports: List[ImportStatement] = []
|
|
153
|
+
strings: List[StringLiteral] = []
|
|
154
|
+
sym_table = SymbolTable(file_path)
|
|
155
|
+
|
|
156
|
+
for idx, line in enumerate(lines):
|
|
157
|
+
line_num = idx + 1
|
|
158
|
+
stripped = line.strip()
|
|
159
|
+
if not stripped or stripped.startswith(("#", "//", "/*", "*")):
|
|
160
|
+
continue
|
|
161
|
+
|
|
162
|
+
# Assignments
|
|
163
|
+
assign_match = re.match(r"^(?:const|let|var)?\s*([a-zA-Z_][a-zA-Z0-9_]*)\s*(?::=[=]?|=)\s*(.+)$", stripped)
|
|
164
|
+
if assign_match:
|
|
165
|
+
target = assign_match.group(1).strip()
|
|
166
|
+
val = assign_match.group(2).strip()
|
|
167
|
+
assigns.append(Assignment(target=target, value_expression=val, line_number=line_num, column=0))
|
|
168
|
+
sym_table.add_variable(target, line_num, expression=val)
|
|
169
|
+
|
|
170
|
+
# Function calls (e.g. foo(x), obj.method(y))
|
|
171
|
+
call_matches = re.finditer(r"([a-zA-Z_][a-zA-Z0-9_\.]*)\s*\((.*?)\)", stripped)
|
|
172
|
+
for cm in call_matches:
|
|
173
|
+
func_name = cm.group(1)
|
|
174
|
+
args_str = cm.group(2)
|
|
175
|
+
args = [a.strip() for a in args_str.split(",") if a.strip()]
|
|
176
|
+
calls.append(FunctionCall(
|
|
177
|
+
name=func_name,
|
|
178
|
+
full_call=cm.group(0),
|
|
179
|
+
arguments=args,
|
|
180
|
+
line_number=line_num,
|
|
181
|
+
column=cm.start()
|
|
182
|
+
))
|
|
183
|
+
|
|
184
|
+
# Imports
|
|
185
|
+
if "import " in stripped or "require(" in stripped or "from " in stripped:
|
|
186
|
+
imp_match = re.search(r"(?:from\s+([a-zA-Z0-9_\.]+)\s+import\s+([a-zA-Z0-9_,\s]+)|import\s+([a-zA-Z0-9_\.]+))", stripped)
|
|
187
|
+
if imp_match:
|
|
188
|
+
mod = imp_match.group(1) or imp_match.group(3) or ""
|
|
189
|
+
names = [n.strip() for n in (imp_match.group(2) or "").split(",") if n.strip()]
|
|
190
|
+
imports.append(ImportStatement(module=mod, imported_names=names, line_number=line_num))
|
|
191
|
+
|
|
192
|
+
return ParseResult(
|
|
193
|
+
file_path=file_path,
|
|
194
|
+
language=lang,
|
|
195
|
+
function_calls=calls,
|
|
196
|
+
assignments=assigns,
|
|
197
|
+
imports=imports,
|
|
198
|
+
string_literals=strings,
|
|
199
|
+
symbol_table=sym_table,
|
|
200
|
+
is_tree_sitter=False,
|
|
201
|
+
raw_content=content
|
|
202
|
+
)
|