torusguard 2.1.0 → 2.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/.torusguard/.manifest.json +47 -5
  2. package/.torusguard/core/__init__.py +146 -0
  3. package/.torusguard/core/agent_roles.py +104 -0
  4. package/.torusguard/core/ast_walker.py +283 -0
  5. package/.torusguard/core/authorization.py +218 -0
  6. package/.torusguard/core/browser_verifier.py +128 -0
  7. package/.torusguard/core/bundle.py +141 -0
  8. package/.torusguard/core/call_graph.py +184 -0
  9. package/.torusguard/core/clustering.py +275 -0
  10. package/.torusguard/core/confidence.py +120 -0
  11. package/.torusguard/core/cross_file_taint.py +101 -0
  12. package/.torusguard/core/exploit_checker.py +317 -0
  13. package/.torusguard/core/formatter.py +351 -0
  14. package/.torusguard/core/governance.py +210 -0
  15. package/.torusguard/core/identity.py +104 -0
  16. package/.torusguard/core/import_resolver.py +91 -0
  17. package/.torusguard/core/incremental.py +102 -0
  18. package/.torusguard/core/lifecycle.py +137 -0
  19. package/.torusguard/core/models.py +425 -0
  20. package/.torusguard/core/parallel.py +56 -0
  21. package/.torusguard/core/parser.py +202 -0
  22. package/.torusguard/core/rechecker.py +107 -0
  23. package/.torusguard/core/replay_trace.py +178 -0
  24. package/.torusguard/core/rules_registry.py +131 -0
  25. package/.torusguard/core/run_folder.py +60 -0
  26. package/.torusguard/core/run_manager.py +163 -0
  27. package/.torusguard/core/runtime_evidence.py +175 -0
  28. package/.torusguard/core/runtime_validator.py +246 -0
  29. package/.torusguard/core/safety_gate.py +139 -0
  30. package/.torusguard/core/sarif.py +189 -0
  31. package/.torusguard/core/stack_profiler.py +184 -0
  32. package/.torusguard/core/symbol_table.py +91 -0
  33. package/.torusguard/core/taint.py +133 -0
  34. package/.torusguard/core/taint_graph.py +235 -0
  35. package/.torusguard/core/taint_rules.py +268 -0
  36. package/.torusguard/core/v070_reporter.py +102 -0
  37. package/.torusguard/core/v070_workflow.py +339 -0
  38. package/.torusguard/core/v6_reporter.py +180 -0
  39. package/.torusguard/core/v6_workflow.py +221 -0
  40. package/.torusguard/core/watcher.py +58 -0
  41. package/.torusguard/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
  42. package/.torusguard/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
  43. package/.torusguard/scripts/__pycache__/audit_runner.cpython-314.pyc +0 -0
  44. package/.torusguard/scripts/__pycache__/finding_scorer.cpython-314.pyc +0 -0
  45. package/.torusguard/scripts/__pycache__/rules_sync.cpython-314.pyc +0 -0
  46. package/.torusguard/scripts/audit_runner.py +108 -10
  47. package/.torusguard/scripts/finding_scorer.py +43 -13
  48. package/.torusguard/scripts/skill_profiler.py +26 -0
  49. package/.torusguard/skills/torusguard/SKILL.md +6 -2
  50. package/.torusguard/skills/torusguard-audit/SKILL.md +109 -84
  51. package/.torusguard/workflows/audit.md +21 -17
  52. package/README.md +19 -11
  53. package/package.json +7 -2
  54. package/skills/torusguard/SKILL.md +6 -2
  55. package/skills/torusguard/__pycache__/bootstrap.cpython-314.pyc +0 -0
  56. package/skills/torusguard/bootstrap.py +3 -3
  57. package/skills/torusguard/payload/.manifest.json +48 -7
  58. package/skills/torusguard/payload/core/__init__.py +146 -0
  59. package/skills/torusguard/payload/core/agent_roles.py +104 -0
  60. package/skills/torusguard/payload/core/ast_walker.py +283 -0
  61. package/skills/torusguard/payload/core/authorization.py +218 -0
  62. package/skills/torusguard/payload/core/browser_verifier.py +128 -0
  63. package/skills/torusguard/payload/core/bundle.py +141 -0
  64. package/skills/torusguard/payload/core/call_graph.py +184 -0
  65. package/skills/torusguard/payload/core/clustering.py +275 -0
  66. package/skills/torusguard/payload/core/confidence.py +120 -0
  67. package/skills/torusguard/payload/core/cross_file_taint.py +101 -0
  68. package/skills/torusguard/payload/core/exploit_checker.py +317 -0
  69. package/skills/torusguard/payload/core/formatter.py +351 -0
  70. package/skills/torusguard/payload/core/governance.py +210 -0
  71. package/skills/torusguard/payload/core/identity.py +104 -0
  72. package/skills/torusguard/payload/core/import_resolver.py +91 -0
  73. package/skills/torusguard/payload/core/incremental.py +102 -0
  74. package/skills/torusguard/payload/core/lifecycle.py +137 -0
  75. package/skills/torusguard/payload/core/models.py +425 -0
  76. package/skills/torusguard/payload/core/parallel.py +56 -0
  77. package/skills/torusguard/payload/core/parser.py +202 -0
  78. package/skills/torusguard/payload/core/rechecker.py +107 -0
  79. package/skills/torusguard/payload/core/replay_trace.py +178 -0
  80. package/skills/torusguard/payload/core/rules_registry.py +131 -0
  81. package/skills/torusguard/payload/core/run_folder.py +60 -0
  82. package/skills/torusguard/payload/core/run_manager.py +163 -0
  83. package/skills/torusguard/payload/core/runtime_evidence.py +175 -0
  84. package/skills/torusguard/payload/core/runtime_validator.py +246 -0
  85. package/skills/torusguard/payload/core/safety_gate.py +139 -0
  86. package/skills/torusguard/payload/core/sarif.py +189 -0
  87. package/skills/torusguard/payload/core/stack_profiler.py +184 -0
  88. package/skills/torusguard/payload/core/symbol_table.py +91 -0
  89. package/skills/torusguard/payload/core/taint.py +133 -0
  90. package/skills/torusguard/payload/core/taint_graph.py +235 -0
  91. package/skills/torusguard/payload/core/taint_rules.py +268 -0
  92. package/skills/torusguard/payload/core/v070_reporter.py +102 -0
  93. package/skills/torusguard/payload/core/v070_workflow.py +339 -0
  94. package/skills/torusguard/payload/core/v6_reporter.py +180 -0
  95. package/skills/torusguard/payload/core/v6_workflow.py +221 -0
  96. package/skills/torusguard/payload/core/watcher.py +58 -0
  97. package/skills/torusguard/payload/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
  98. package/skills/torusguard/payload/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
  99. package/skills/torusguard/payload/rules/container/TG-CONT-001-root-user-execution.md +50 -50
  100. package/skills/torusguard/payload/rules/container/TG-CONT-002-docker-socket-mount.md +47 -47
  101. package/skills/torusguard/payload/rules/container/TG-CONT-003-privileged-container-mode.md +53 -53
  102. package/skills/torusguard/payload/rules/container/TG-CONT-004-build-arg-secret-exposure.md +43 -43
  103. package/skills/torusguard/payload/rules/git/TG-GIT-001-historical-secret-in-git-commit.md +44 -44
  104. package/skills/torusguard/payload/rules/git/TG-GIT-002-plaintext-credentials-in-git-config.md +41 -41
  105. package/skills/torusguard/payload/rules/git/TG-GIT-003-sensitive-tracked-file-gitignore-breach.md +40 -40
  106. package/skills/torusguard/payload/rules/rag/TG-RAG-001-untrusted-rag-context-injection.md +72 -72
  107. package/skills/torusguard/payload/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md +51 -51
  108. package/skills/torusguard/payload/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md +51 -51
  109. package/skills/torusguard/payload/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md +46 -46
  110. package/skills/torusguard/payload/rules/redos/TG-REDOS-002-unbounded-nested-quantifier.md +43 -43
  111. package/skills/torusguard/payload/scripts/audit_runner.py +108 -10
  112. package/skills/torusguard/payload/scripts/finding_scorer.py +43 -13
  113. package/skills/torusguard/payload/skills/torusguard/SKILL.md +6 -2
  114. package/skills/torusguard/payload/skills/torusguard/bootstrap.py +3 -3
  115. package/skills/torusguard/payload/skills/torusguard-ai-guard/SKILL.md +95 -95
  116. package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +109 -84
  117. package/skills/torusguard/payload/skills/torusguard-container/SKILL.md +94 -94
  118. package/skills/torusguard/payload/skills/torusguard-git-mine/SKILL.md +92 -92
  119. package/skills/torusguard/payload/skills/torusguard-ocr-scan/SKILL.md +94 -94
  120. package/skills/torusguard/payload/skills/torusguard-redos/SKILL.md +91 -91
  121. package/skills/torusguard/payload/workflows/ai-guard.md +31 -31
  122. package/skills/torusguard/payload/workflows/audit.md +21 -17
  123. package/skills/torusguard/payload/workflows/container.md +29 -29
  124. package/skills/torusguard/payload/workflows/git-mine.md +25 -25
  125. package/skills/torusguard/payload/workflows/ocr-scan.md +25 -25
  126. package/skills/torusguard/payload/workflows/redos.md +27 -27
  127. package/skills/torusguard/payload/workflows/torusguard-audit.md +35 -55
  128. package/skills/torusguard/references/csharp-security.md +41 -41
  129. package/skills/torusguard/references/go-security.md +41 -41
  130. package/skills/torusguard/references/java-security.md +40 -40
  131. package/skills/torusguard/references/polyglot-security-matrix.md +25 -25
  132. package/skills/torusguard/references/rust-security.md +40 -40
  133. package/skills/torusguard-audit/SKILL.md +107 -83
@@ -0,0 +1,425 @@
1
+ """
2
+ TorusGuard Core Data Models (v0.5.4)
3
+ Defines canonical Finding, ProvenanceChain, ConfidenceScore, EvidencePackage, RetestRecord, RemediationPriority, and AuditReport objects.
4
+ """
5
+
6
+ from dataclasses import dataclass, field, asdict
7
+ from enum import Enum
8
+ from typing import List, Optional, Dict, Any
9
+ import datetime
10
+ import hashlib
11
+ import uuid
12
+ import re
13
+
14
+
15
+ class SeverityLevel(str, Enum):
16
+ CRITICAL = "Critical"
17
+ HIGH = "High"
18
+ MEDIUM = "Medium"
19
+ LOW = "Low"
20
+ INFORMATIONAL = "Informational"
21
+
22
+
23
+ class RemediationPriority(str, Enum):
24
+ IMMEDIATE = "Immediate (P0)" # Block deployment / immediate fix
25
+ NEAR_TERM = "Near-Term (P1)" # Fix in current sprint / patch cycle
26
+ BACKLOG = "Backlog (P2)" # Defense-in-depth hardening backlog
27
+
28
+
29
+ class ConfidenceBand(str, Enum):
30
+ CONFIRMED = "Confirmed" # 90 - 100
31
+ HIGH_CONFIDENCE = "High Confidence" # 70 - 89
32
+ MEDIUM_CONFIDENCE = "Medium Confidence" # 50 - 69
33
+ LOW_CONFIDENCE = "Low Confidence" # < 50
34
+ UNCONFIRMED = "Unconfirmed"
35
+ NEEDS_REVIEW = "Needs Review"
36
+ INFORMATIONAL = "Informational"
37
+ NOT_APPLICABLE = "Not Applicable"
38
+
39
+
40
+ class FindingStatus(str, Enum):
41
+ CONFIRMED = "Confirmed"
42
+ HIGH_CONFIDENCE = "High Confidence"
43
+ MEDIUM_CONFIDENCE = "Medium Confidence"
44
+ LOW_CONFIDENCE = "Low Confidence"
45
+ UNCONFIRMED = "Unconfirmed"
46
+ NEEDS_REVIEW = "Needs Review"
47
+ REMEDIATED = "Remediated"
48
+ VERIFIED_FIXED = "Verified Fixed"
49
+ SUPPRESSED = "Suppressed"
50
+
51
+
52
+ class LifecycleStage(str, Enum):
53
+ DETECT = "Detect"
54
+ CLASSIFY = "Classify"
55
+ VERIFY = "Verify"
56
+ REMEDIATE = "Remediate"
57
+ RECHECK = "Recheck"
58
+ ARCHIVE = "Archive"
59
+
60
+
61
+ class TaxonomyCategory(str, Enum):
62
+ AUTH = "authentication-authorization"
63
+ INPUT = "input-validation-encoding"
64
+ DATA = "data-access-orm"
65
+ FILES = "file-upload-handling"
66
+ SECRETS = "secrets-configuration"
67
+ SUPPLY_CHAIN = "dependency-supply-chain"
68
+ NETWORK = "network-ssrf-boundaries"
69
+ BUSINESS_LOGIC = "business-logic-rate-limiting"
70
+ CLIENT_PLATFORM = "client-cache-platform"
71
+
72
+
73
+ class EvidenceType(str, Enum):
74
+ SOURCE = "source"
75
+ RUNTIME = "runtime"
76
+ TEST = "test"
77
+ MANUAL_REVIEW = "manual_review"
78
+
79
+
80
+ @dataclass
81
+ class ProvenanceChain:
82
+ discovery_module: str
83
+ triggering_input: str
84
+ evidence_collected: List[str]
85
+ decision_path: List[str]
86
+ verification_step: str
87
+ agent_environment: str = "TorusGuard Engine v0.5.4"
88
+ timestamp: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
89
+
90
+ def to_dict(self) -> Dict[str, Any]:
91
+ return asdict(self)
92
+
93
+
94
+ @dataclass
95
+ class ConfidenceFactors:
96
+ evidence_quality: int # Max: 35
97
+ reproduction_success: int # Max: 25
98
+ independent_confirmations: int # Max: 15
99
+ environmental_clarity: int # Max: 15
100
+ manual_review_status: int # Max: 10
101
+
102
+ def total_score(self) -> int:
103
+ return max(0, min(100, (
104
+ self.evidence_quality +
105
+ self.reproduction_success +
106
+ self.independent_confirmations +
107
+ self.environmental_clarity +
108
+ self.manual_review_status
109
+ )))
110
+
111
+
112
+ @dataclass
113
+ class ConfidenceScore:
114
+ factors: ConfidenceFactors
115
+ score: int = field(init=False)
116
+ band: ConfidenceBand = field(init=False)
117
+ rationale: str = ""
118
+
119
+ def __post_init__(self):
120
+ self.score = self.factors.total_score()
121
+ if self.score >= 90:
122
+ self.band = ConfidenceBand.CONFIRMED
123
+ elif self.score >= 70:
124
+ self.band = ConfidenceBand.HIGH_CONFIDENCE
125
+ elif self.score >= 50:
126
+ self.band = ConfidenceBand.MEDIUM_CONFIDENCE
127
+ else:
128
+ self.band = ConfidenceBand.LOW_CONFIDENCE
129
+
130
+ @staticmethod
131
+ def calculate(
132
+ evidence_quality: int = 30,
133
+ reproduction_success: int = 25,
134
+ independent_confirmations: int = 15,
135
+ environmental_clarity: int = 15,
136
+ manual_review_status: int = 5,
137
+ rationale: str = ""
138
+ ) -> 'ConfidenceScore':
139
+ factors = ConfidenceFactors(
140
+ evidence_quality=evidence_quality,
141
+ reproduction_success=reproduction_success,
142
+ independent_confirmations=independent_confirmations,
143
+ environmental_clarity=environmental_clarity,
144
+ manual_review_status=manual_review_status,
145
+ )
146
+ return ConfidenceScore(factors=factors, rationale=rationale)
147
+
148
+ def to_dict(self) -> Dict[str, Any]:
149
+ return {
150
+ "score": self.score,
151
+ "band": self.band.value,
152
+ "factors": asdict(self.factors),
153
+ "rationale": self.rationale,
154
+ }
155
+
156
+
157
+ def mask_sensitive_data(text: str) -> str:
158
+ """Masks secrets, tokens, API keys, and passwords from report output."""
159
+ text = re.sub(r'sk_live_[0-9a-zA-Z_\-]{6,}', 'sk_live_***REDACTED***', text)
160
+ text = re.sub(r'ghp_[0-9a-zA-Z_\-]{6,}', 'ghp_***REDACTED***', text)
161
+ text = re.sub(r'(Bearer\s+)[A-Za-z0-9\-_=]+\.[A-Za-z0-9\-_=]+\.?[A-Za-z0-9\-_=]*', r'\1***REDACTED_JWT***', text)
162
+
163
+ def redact_kv(m):
164
+ val = m.group(3)
165
+ if "***REDACTED" in val:
166
+ return m.group(0)
167
+ return f"{m.group(1)}{m.group(2)}***REDACTED***{m.group(4)}"
168
+
169
+ text = re.sub(
170
+ r'(?i)(secret[_\-\w]*|password|api[_\-\w]*key|token|auth[_\-\w]*key)(\s*[:=]\s*[\'"])([^\'"]{4,})([\'"])',
171
+ redact_kv,
172
+ text
173
+ )
174
+ return text
175
+
176
+
177
+ @dataclass
178
+ class Evidence:
179
+ type: EvidenceType
180
+ location: str
181
+ raw_snippet: str
182
+ rationale: str
183
+ confidence_level: ConfidenceBand
184
+ sha256_checksum: str = field(init=False)
185
+ context: Optional[str] = None
186
+ reproduction_notes: Optional[str] = None
187
+ reviewer_notes: Optional[str] = None
188
+ is_sufficient_for_confirmed: bool = False
189
+ collected_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
190
+
191
+ def __post_init__(self):
192
+ self.sha256_checksum = hashlib.sha256(self.raw_snippet.strip().encode("utf-8")).hexdigest()
193
+
194
+ def get_masked_snippet(self) -> str:
195
+ return mask_sensitive_data(self.raw_snippet)
196
+
197
+ def to_dict(self) -> Dict[str, Any]:
198
+ return {
199
+ "type": self.type.value if isinstance(self.type, EvidenceType) else self.type,
200
+ "location": self.location,
201
+ "raw_snippet": self.get_masked_snippet(),
202
+ "sha256_checksum": self.sha256_checksum,
203
+ "collected_at": self.collected_at,
204
+ "context": self.context,
205
+ "rationale": self.rationale,
206
+ "confidence_level": self.confidence_level.value if isinstance(self.confidence_level, ConfidenceBand) else self.confidence_level,
207
+ "reproduction_notes": self.reproduction_notes,
208
+ "reviewer_notes": self.reviewer_notes,
209
+ "is_sufficient_for_confirmed": self.is_sufficient_for_confirmed,
210
+ }
211
+
212
+
213
+ @dataclass
214
+ class SeverityInfo:
215
+ level: SeverityLevel
216
+ rationale: str
217
+ rubric_justification: str
218
+
219
+ def to_dict(self) -> Dict[str, Any]:
220
+ return {
221
+ "level": self.level.value if isinstance(self.level, SeverityLevel) else self.level,
222
+ "rationale": self.rationale,
223
+ "rubric_justification": self.rubric_justification,
224
+ }
225
+
226
+
227
+ @dataclass
228
+ class FrameworkPattern:
229
+ framework: str
230
+ unsafe_snippet: str
231
+ safe_snippet: str
232
+ least_invasive: bool = True
233
+
234
+
235
+ @dataclass
236
+ class Remediation:
237
+ problem_statement: str
238
+ risk_explanation: str
239
+ recommended_fix: str
240
+ framework_pattern: FrameworkPattern
241
+ verification_method: str
242
+ residual_risk_notes: str
243
+
244
+ def to_dict(self) -> Dict[str, Any]:
245
+ return {
246
+ "problem_statement": self.problem_statement,
247
+ "risk_explanation": self.risk_explanation,
248
+ "recommended_fix": self.recommended_fix,
249
+ "framework_pattern": asdict(self.framework_pattern),
250
+ "verification_method": self.verification_method,
251
+ "residual_risk_notes": self.residual_risk_notes,
252
+ }
253
+
254
+
255
+ @dataclass
256
+ class AffectedComponent:
257
+ component_name: str
258
+ target_path: str
259
+ start_line: Optional[int] = None
260
+ end_line: Optional[int] = None
261
+ symbol: Optional[str] = None
262
+
263
+ def to_dict(self) -> Dict[str, Any]:
264
+ return {k: v for k, v in asdict(self).items() if v is not None}
265
+
266
+
267
+ @dataclass
268
+ class ReproductionMethod:
269
+ step_by_step: List[str]
270
+ deterministic: bool = True
271
+ test_command: Optional[str] = None
272
+ expected_failure_response: Optional[str] = None
273
+
274
+ def to_dict(self) -> Dict[str, Any]:
275
+ return {k: v for k, v in asdict(self).items() if v is not None}
276
+
277
+
278
+ @dataclass
279
+ class RetestRecord:
280
+ retest_performed: bool = False
281
+ closure_status: FindingStatus = FindingStatus.UNCONFIRMED
282
+ fix_applied: Optional[str] = None
283
+ retest_method: Optional[str] = None
284
+ retest_evidence_hash: Optional[str] = None
285
+ residual_risk: Optional[str] = None
286
+ verifier_notes: Optional[str] = None
287
+ retest_timestamp: Optional[str] = None
288
+
289
+ def to_dict(self) -> Dict[str, Any]:
290
+ d = asdict(self)
291
+ d["closure_status"] = self.closure_status.value if isinstance(self.closure_status, FindingStatus) else self.closure_status
292
+ return {k: v for k, v in d.items() if v is not None}
293
+
294
+
295
+ @dataclass
296
+ class NotesRecord:
297
+ business_impact: str
298
+ technical_description: str
299
+ raw_facts_summary: str
300
+ ai_interpretation: str
301
+
302
+ def to_dict(self) -> Dict[str, Any]:
303
+ return asdict(self)
304
+
305
+
306
+ @dataclass
307
+ class FindingTimestamps:
308
+ discovered_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
309
+ updated_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
310
+ verified_at: Optional[str] = None
311
+ remediated_at: Optional[str] = None
312
+ retested_at: Optional[str] = None
313
+
314
+ def to_dict(self) -> Dict[str, Any]:
315
+ return {k: v for k, v in asdict(self).items() if v is not None}
316
+
317
+
318
+ @dataclass
319
+ class Finding:
320
+ rule_id: str
321
+ title: str
322
+ category: TaxonomyCategory
323
+ severity: SeverityInfo
324
+ confidence: ConfidenceScore
325
+ status: FindingStatus
326
+ affected_component: AffectedComponent
327
+ evidence: List[Evidence]
328
+ provenance: ProvenanceChain
329
+ reproduction_method: ReproductionMethod
330
+ remediation: Remediation
331
+ remediation_priority: RemediationPriority = RemediationPriority.IMMEDIATE
332
+ retest_result: RetestRecord = field(default_factory=RetestRecord)
333
+ timestamps: FindingTimestamps = field(default_factory=FindingTimestamps)
334
+ notes: NotesRecord = field(default_factory=lambda: NotesRecord(
335
+ business_impact="Exposure of sensitive application resources or unauthorized data modification.",
336
+ technical_description="Direct unmitigated pattern detected in application route or data layer.",
337
+ raw_facts_summary="Unmitigated pattern identified in source.",
338
+ ai_interpretation="High priority fix recommended.",
339
+ ))
340
+ finding_id: str = field(default_factory=lambda: f"TG-FIND-{datetime.datetime.utcnow().year}-{uuid.uuid4().hex[:6]}")
341
+ lifecycle_stage: LifecycleStage = LifecycleStage.DETECT
342
+ asvs_control: Optional[str] = None
343
+ cwe: Optional[str] = None
344
+ nist_ssdf: Optional[str] = None
345
+
346
+ def __post_init__(self):
347
+ # Auto-derive remediation priority from severity level if default
348
+ if self.severity.level == SeverityLevel.CRITICAL:
349
+ self.remediation_priority = RemediationPriority.IMMEDIATE
350
+ elif self.severity.level == SeverityLevel.HIGH:
351
+ self.remediation_priority = RemediationPriority.NEAR_TERM
352
+ else:
353
+ self.remediation_priority = RemediationPriority.BACKLOG
354
+
355
+ def to_dict(self) -> Dict[str, Any]:
356
+ return {
357
+ "finding_id": self.finding_id,
358
+ "rule_id": self.rule_id,
359
+ "title": self.title,
360
+ "category": self.category.value if isinstance(self.category, TaxonomyCategory) else self.category,
361
+ "severity": self.severity.to_dict(),
362
+ "confidence": self.confidence.to_dict(),
363
+ "status": self.status.value if isinstance(self.status, FindingStatus) else self.status,
364
+ "remediation_priority": self.remediation_priority.value if isinstance(self.remediation_priority, RemediationPriority) else self.remediation_priority,
365
+ "lifecycle_stage": self.lifecycle_stage.value if isinstance(self.lifecycle_stage, LifecycleStage) else self.lifecycle_stage,
366
+ "affected_component": self.affected_component.to_dict(),
367
+ "evidence": [e.to_dict() for e in self.evidence],
368
+ "provenance": self.provenance.to_dict(),
369
+ "reproduction_method": self.reproduction_method.to_dict(),
370
+ "remediation": self.remediation.to_dict(),
371
+ "retest_result": self.retest_result.to_dict(),
372
+ "requirement_reference": {
373
+ "asvs_v4": self.asvs_control,
374
+ "cwe": self.cwe,
375
+ "nist_ssdf": self.nist_ssdf,
376
+ },
377
+ "cwe": self.cwe,
378
+ "timestamps": self.timestamps.to_dict(),
379
+ "notes": self.notes.to_dict(),
380
+ }
381
+
382
+
383
+ @dataclass
384
+ class AuditReport:
385
+ project_name: str
386
+ detected_stack: Dict[str, Any]
387
+ findings: List[Finding]
388
+ summary_counts: Dict[str, Any] = field(default_factory=dict)
389
+ generated_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
390
+ torusguard_version: str = "v0.5.4"
391
+ report_owner: str = "TorusGuard Security Subsystem"
392
+ repository_ref: str = "workspace"
393
+
394
+ def calculate_summary(self) -> None:
395
+ total = len(self.findings)
396
+ avg_confidence = round(sum(f.confidence.score for f in self.findings) / total, 1) if total > 0 else 100.0
397
+ self.summary_counts = {
398
+ "total_findings": total,
399
+ "average_confidence_score": avg_confidence,
400
+ "critical": sum(1 for f in self.findings if f.severity.level == SeverityLevel.CRITICAL),
401
+ "high": sum(1 for f in self.findings if f.severity.level == SeverityLevel.HIGH),
402
+ "medium": sum(1 for f in self.findings if f.severity.level == SeverityLevel.MEDIUM),
403
+ "low": sum(1 for f in self.findings if f.severity.level == SeverityLevel.LOW),
404
+ "confirmed": sum(1 for f in self.findings if f.confidence.band == ConfidenceBand.CONFIRMED),
405
+ "high_confidence": sum(1 for f in self.findings if f.confidence.band == ConfidenceBand.HIGH_CONFIDENCE),
406
+ "needs_review": sum(1 for f in self.findings if f.confidence.band in (ConfidenceBand.NEEDS_REVIEW, ConfidenceBand.LOW_CONFIDENCE)),
407
+ "verified_fixed": sum(1 for f in self.findings if f.status == FindingStatus.VERIFIED_FIXED),
408
+ "remediated": sum(1 for f in self.findings if f.status in (FindingStatus.REMEDIATED, FindingStatus.VERIFIED_FIXED)),
409
+ "immediate_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.IMMEDIATE),
410
+ "near_term_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.NEAR_TERM),
411
+ "backlog_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.BACKLOG),
412
+ }
413
+
414
+ def to_dict(self) -> Dict[str, Any]:
415
+ self.calculate_summary()
416
+ return {
417
+ "project_name": self.project_name,
418
+ "torusguard_version": self.torusguard_version,
419
+ "generated_at": self.generated_at,
420
+ "report_owner": self.report_owner,
421
+ "repository_ref": self.repository_ref,
422
+ "detected_stack": self.detected_stack,
423
+ "summary": self.summary_counts,
424
+ "findings": [f.to_dict() for f in self.findings],
425
+ }
@@ -0,0 +1,56 @@
1
+ """
2
+ TorusGuard Parallel Audit Executor
3
+ Executes multi-threaded static security scanning across files with deterministic result collation.
4
+ """
5
+
6
+ import os
7
+ from concurrent.futures import ThreadPoolExecutor, as_completed
8
+ from pathlib import Path
9
+ from typing import List, Dict, Any, Callable, Optional
10
+
11
+
12
+ class ParallelAuditExecutor:
13
+ """Distributes file analysis across a thread pool with deterministic output ordering."""
14
+
15
+ def __init__(self, max_workers: Optional[int] = None):
16
+ self.max_workers = max_workers or min(os.cpu_count() or 4, 8)
17
+
18
+ def scan_files_parallel(
19
+ self,
20
+ files: List[Path],
21
+ scan_fn: Callable[[Path], List[Dict[str, Any]]]
22
+ ) -> List[Dict[str, Any]]:
23
+ """
24
+ Executes `scan_fn` across all files in parallel.
25
+ Returns deterministically ordered list of all findings.
26
+ """
27
+ if not files:
28
+ return []
29
+
30
+ # If only a few files, avoid thread pool overhead
31
+ if len(files) <= 3:
32
+ all_findings = []
33
+ for f in files:
34
+ all_findings.extend(scan_fn(f))
35
+ return all_findings
36
+
37
+ all_findings: List[Dict[str, Any]] = []
38
+ file_results: Dict[str, List[Dict[str, Any]]] = {}
39
+
40
+ with ThreadPoolExecutor(max_workers=self.max_workers) as executor:
41
+ future_to_file = {executor.submit(scan_fn, f): str(f) for f in files}
42
+ for future in as_completed(future_to_file):
43
+ f_str = future_to_file[future]
44
+ try:
45
+ res = future.result()
46
+ file_results[f_str] = res
47
+ except Exception:
48
+ file_results[f_str] = []
49
+
50
+ # Deterministic sort by original file order
51
+ for f in files:
52
+ f_str = str(f)
53
+ if f_str in file_results:
54
+ all_findings.extend(file_results[f_str])
55
+
56
+ return all_findings
@@ -0,0 +1,202 @@
1
+ """
2
+ TorusGuard Polyglot Parser
3
+ Unified Tree-sitter parser wrapper with language auto-detection and resilient
4
+ fallback parsing across Python, JavaScript, TypeScript, Go, Rust, Java, Ruby, PHP, and C#.
5
+ """
6
+
7
+ from pathlib import Path
8
+ from dataclasses import dataclass, field
9
+ from typing import Dict, List, Optional, Any
10
+ import re
11
+
12
+ from core.ast_walker import FunctionCall, Assignment, ImportStatement, StringLiteral, TreeSitterWalker
13
+ from core.symbol_table import SymbolTable
14
+
15
+
16
+ LANGUAGE_EXTENSIONS: Dict[str, str] = {
17
+ ".py": "python",
18
+ ".js": "javascript",
19
+ ".mjs": "javascript",
20
+ ".cjs": "javascript",
21
+ ".ts": "typescript",
22
+ ".tsx": "tsx",
23
+ ".go": "go",
24
+ ".rs": "rust",
25
+ ".java": "java",
26
+ ".rb": "ruby",
27
+ ".php": "php",
28
+ ".cs": "csharp",
29
+ }
30
+
31
+
32
+ @dataclass
33
+ class ParseResult:
34
+ file_path: str
35
+ language: str
36
+ function_calls: List[FunctionCall] = field(default_factory=list)
37
+ assignments: List[Assignment] = field(default_factory=list)
38
+ imports: List[ImportStatement] = field(default_factory=list)
39
+ string_literals: List[StringLiteral] = field(default_factory=list)
40
+ symbol_table: SymbolTable = field(default_factory=SymbolTable)
41
+ is_tree_sitter: bool = False
42
+ raw_content: str = ""
43
+
44
+ def to_dict(self) -> Dict[str, Any]:
45
+ return {
46
+ "file_path": self.file_path,
47
+ "language": self.language,
48
+ "function_calls_count": len(self.function_calls),
49
+ "assignments_count": len(self.assignments),
50
+ "imports_count": len(self.imports),
51
+ "is_tree_sitter": self.is_tree_sitter,
52
+ }
53
+
54
+
55
+ class PolyglotParser:
56
+ """Parses source files into Tree-sitter ASTs or fallback token models with language auto-detection."""
57
+
58
+ def __init__(self):
59
+ self._parsers: Dict[str, Any] = {}
60
+ self._init_tree_sitter()
61
+
62
+ def _init_tree_sitter(self):
63
+ try:
64
+ import tree_sitter
65
+ # Python
66
+ try:
67
+ import tree_sitter_python
68
+ py_lang = tree_sitter.Language(tree_sitter_python.language())
69
+ self._parsers["python"] = tree_sitter.Parser(py_lang)
70
+ except Exception:
71
+ pass
72
+
73
+ # JavaScript
74
+ try:
75
+ import tree_sitter_javascript
76
+ js_lang = tree_sitter.Language(tree_sitter_javascript.language())
77
+ self._parsers["javascript"] = tree_sitter.Parser(js_lang)
78
+ except Exception:
79
+ pass
80
+
81
+ # TypeScript / TSX
82
+ try:
83
+ import tree_sitter_typescript
84
+ ts_lang = tree_sitter.Language(tree_sitter_typescript.language_typescript())
85
+ self._parsers["typescript"] = tree_sitter.Parser(ts_lang)
86
+ tsx_lang = tree_sitter.Language(tree_sitter_typescript.language_tsx())
87
+ self._parsers["tsx"] = tree_sitter.Parser(tsx_lang)
88
+ except Exception:
89
+ pass
90
+
91
+ # Go
92
+ try:
93
+ import tree_sitter_go
94
+ go_lang = tree_sitter.Language(tree_sitter_go.language())
95
+ self._parsers["go"] = tree_sitter.Parser(go_lang)
96
+ except Exception:
97
+ pass
98
+
99
+ except Exception:
100
+ pass
101
+
102
+ def detect_language(self, file_path: Path) -> str:
103
+ ext = file_path.suffix.lower()
104
+ return LANGUAGE_EXTENSIONS.get(ext, "unknown")
105
+
106
+ def parse_file(self, file_path: Path) -> ParseResult:
107
+ lang = self.detect_language(file_path)
108
+ try:
109
+ content = file_path.read_text(encoding="utf-8", errors="replace")
110
+ except Exception:
111
+ return ParseResult(file_path=str(file_path), language=lang)
112
+
113
+ source_bytes = content.encode("utf-8")
114
+ ts_parser = self._parsers.get(lang)
115
+
116
+ if ts_parser is not None:
117
+ try:
118
+ tree = ts_parser.parse(source_bytes)
119
+ calls = TreeSitterWalker.extract_function_calls(tree.root_node, source_bytes)
120
+ assigns = TreeSitterWalker.extract_assignments(tree.root_node, source_bytes)
121
+ imports = TreeSitterWalker.extract_imports(tree.root_node, source_bytes, lang)
122
+ strings = TreeSitterWalker.extract_string_literals(tree.root_node, source_bytes)
123
+
124
+ sym_table = SymbolTable(str(file_path))
125
+ for a in assigns:
126
+ sym_table.add_variable(name=a.target, line=a.line_number, expression=a.value_expression)
127
+ for imp in imports:
128
+ for alias, orig in imp.alias_map.items():
129
+ sym_table.add_import_alias(alias, orig)
130
+
131
+ return ParseResult(
132
+ file_path=str(file_path),
133
+ language=lang,
134
+ function_calls=calls,
135
+ assignments=assigns,
136
+ imports=imports,
137
+ string_literals=strings,
138
+ symbol_table=sym_table,
139
+ is_tree_sitter=True,
140
+ raw_content=content
141
+ )
142
+ except Exception:
143
+ pass
144
+
145
+ # Resilient Fallback Parser
146
+ return self._fallback_parse(str(file_path), content, lang)
147
+
148
+ def _fallback_parse(self, file_path: str, content: str, lang: str) -> ParseResult:
149
+ lines = content.splitlines()
150
+ calls: List[FunctionCall] = []
151
+ assigns: List[Assignment] = []
152
+ imports: List[ImportStatement] = []
153
+ strings: List[StringLiteral] = []
154
+ sym_table = SymbolTable(file_path)
155
+
156
+ for idx, line in enumerate(lines):
157
+ line_num = idx + 1
158
+ stripped = line.strip()
159
+ if not stripped or stripped.startswith(("#", "//", "/*", "*")):
160
+ continue
161
+
162
+ # Assignments
163
+ assign_match = re.match(r"^(?:const|let|var)?\s*([a-zA-Z_][a-zA-Z0-9_]*)\s*(?::=[=]?|=)\s*(.+)$", stripped)
164
+ if assign_match:
165
+ target = assign_match.group(1).strip()
166
+ val = assign_match.group(2).strip()
167
+ assigns.append(Assignment(target=target, value_expression=val, line_number=line_num, column=0))
168
+ sym_table.add_variable(target, line_num, expression=val)
169
+
170
+ # Function calls (e.g. foo(x), obj.method(y))
171
+ call_matches = re.finditer(r"([a-zA-Z_][a-zA-Z0-9_\.]*)\s*\((.*?)\)", stripped)
172
+ for cm in call_matches:
173
+ func_name = cm.group(1)
174
+ args_str = cm.group(2)
175
+ args = [a.strip() for a in args_str.split(",") if a.strip()]
176
+ calls.append(FunctionCall(
177
+ name=func_name,
178
+ full_call=cm.group(0),
179
+ arguments=args,
180
+ line_number=line_num,
181
+ column=cm.start()
182
+ ))
183
+
184
+ # Imports
185
+ if "import " in stripped or "require(" in stripped or "from " in stripped:
186
+ imp_match = re.search(r"(?:from\s+([a-zA-Z0-9_\.]+)\s+import\s+([a-zA-Z0-9_,\s]+)|import\s+([a-zA-Z0-9_\.]+))", stripped)
187
+ if imp_match:
188
+ mod = imp_match.group(1) or imp_match.group(3) or ""
189
+ names = [n.strip() for n in (imp_match.group(2) or "").split(",") if n.strip()]
190
+ imports.append(ImportStatement(module=mod, imported_names=names, line_number=line_num))
191
+
192
+ return ParseResult(
193
+ file_path=file_path,
194
+ language=lang,
195
+ function_calls=calls,
196
+ assignments=assigns,
197
+ imports=imports,
198
+ string_literals=strings,
199
+ symbol_table=sym_table,
200
+ is_tree_sitter=False,
201
+ raw_content=content
202
+ )