torusguard 2.1.0 → 2.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/.torusguard/.manifest.json +55 -5
  2. package/.torusguard/core/__init__.py +146 -0
  3. package/.torusguard/core/agent_roles.py +104 -0
  4. package/.torusguard/core/ast_walker.py +283 -0
  5. package/.torusguard/core/authorization.py +218 -0
  6. package/.torusguard/core/browser_verifier.py +128 -0
  7. package/.torusguard/core/bundle.py +141 -0
  8. package/.torusguard/core/call_graph.py +184 -0
  9. package/.torusguard/core/clustering.py +275 -0
  10. package/.torusguard/core/confidence.py +120 -0
  11. package/.torusguard/core/cross_file_taint.py +101 -0
  12. package/.torusguard/core/exploit_checker.py +317 -0
  13. package/.torusguard/core/formatter.py +351 -0
  14. package/.torusguard/core/governance.py +210 -0
  15. package/.torusguard/core/identity.py +104 -0
  16. package/.torusguard/core/import_resolver.py +91 -0
  17. package/.torusguard/core/incremental.py +102 -0
  18. package/.torusguard/core/lifecycle.py +137 -0
  19. package/.torusguard/core/models.py +425 -0
  20. package/.torusguard/core/parallel.py +56 -0
  21. package/.torusguard/core/parser.py +202 -0
  22. package/.torusguard/core/rechecker.py +107 -0
  23. package/.torusguard/core/replay_trace.py +178 -0
  24. package/.torusguard/core/rules_registry.py +131 -0
  25. package/.torusguard/core/run_folder.py +60 -0
  26. package/.torusguard/core/run_manager.py +163 -0
  27. package/.torusguard/core/runtime_evidence.py +175 -0
  28. package/.torusguard/core/runtime_validator.py +246 -0
  29. package/.torusguard/core/safety_gate.py +139 -0
  30. package/.torusguard/core/sarif.py +189 -0
  31. package/.torusguard/core/stack_profiler.py +184 -0
  32. package/.torusguard/core/symbol_table.py +91 -0
  33. package/.torusguard/core/taint.py +133 -0
  34. package/.torusguard/core/taint_graph.py +235 -0
  35. package/.torusguard/core/taint_rules.py +268 -0
  36. package/.torusguard/core/v070_reporter.py +102 -0
  37. package/.torusguard/core/v070_workflow.py +339 -0
  38. package/.torusguard/core/v6_reporter.py +180 -0
  39. package/.torusguard/core/v6_workflow.py +221 -0
  40. package/.torusguard/core/watcher.py +58 -0
  41. package/.torusguard/custom_rules/README.md +24 -0
  42. package/.torusguard/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
  43. package/.torusguard/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
  44. package/.torusguard/scripts/__pycache__/audit_runner.cpython-314.pyc +0 -0
  45. package/.torusguard/scripts/__pycache__/finding_scorer.cpython-314.pyc +0 -0
  46. package/.torusguard/scripts/__pycache__/manifest_builder.cpython-314.pyc +0 -0
  47. package/.torusguard/scripts/__pycache__/rules_sync.cpython-314.pyc +0 -0
  48. package/.torusguard/scripts/audit_runner.py +108 -10
  49. package/.torusguard/scripts/deliberation_tournament.py +136 -0
  50. package/.torusguard/scripts/finding_scorer.py +43 -13
  51. package/.torusguard/scripts/reachability_analyzer.py +165 -0
  52. package/.torusguard/scripts/skill_profiler.py +26 -0
  53. package/.torusguard/scripts/stride_generator.py +198 -0
  54. package/.torusguard/skills/torusguard/SKILL.md +6 -2
  55. package/.torusguard/skills/torusguard-audit/SKILL.md +109 -84
  56. package/.torusguard/skills/torusguard-review/SKILL.md +27 -0
  57. package/.torusguard/skills/torusguard-threatmodel/SKILL.md +27 -0
  58. package/.torusguard/workflows/audit.md +21 -17
  59. package/.torusguard/workflows/review.md +11 -0
  60. package/.torusguard/workflows/threatmodel.md +9 -0
  61. package/README.md +149 -48
  62. package/package.json +10 -2
  63. package/skills/torusguard/SKILL.md +6 -2
  64. package/skills/torusguard/__pycache__/bootstrap.cpython-314.pyc +0 -0
  65. package/skills/torusguard/bootstrap.py +3 -3
  66. package/skills/torusguard/payload/.manifest.json +56 -7
  67. package/skills/torusguard/payload/core/__init__.py +146 -0
  68. package/skills/torusguard/payload/core/agent_roles.py +104 -0
  69. package/skills/torusguard/payload/core/ast_walker.py +283 -0
  70. package/skills/torusguard/payload/core/authorization.py +218 -0
  71. package/skills/torusguard/payload/core/browser_verifier.py +128 -0
  72. package/skills/torusguard/payload/core/bundle.py +141 -0
  73. package/skills/torusguard/payload/core/call_graph.py +184 -0
  74. package/skills/torusguard/payload/core/clustering.py +275 -0
  75. package/skills/torusguard/payload/core/confidence.py +120 -0
  76. package/skills/torusguard/payload/core/cross_file_taint.py +101 -0
  77. package/skills/torusguard/payload/core/exploit_checker.py +317 -0
  78. package/skills/torusguard/payload/core/formatter.py +351 -0
  79. package/skills/torusguard/payload/core/governance.py +210 -0
  80. package/skills/torusguard/payload/core/identity.py +104 -0
  81. package/skills/torusguard/payload/core/import_resolver.py +91 -0
  82. package/skills/torusguard/payload/core/incremental.py +102 -0
  83. package/skills/torusguard/payload/core/lifecycle.py +137 -0
  84. package/skills/torusguard/payload/core/models.py +425 -0
  85. package/skills/torusguard/payload/core/parallel.py +56 -0
  86. package/skills/torusguard/payload/core/parser.py +202 -0
  87. package/skills/torusguard/payload/core/rechecker.py +107 -0
  88. package/skills/torusguard/payload/core/replay_trace.py +178 -0
  89. package/skills/torusguard/payload/core/rules_registry.py +131 -0
  90. package/skills/torusguard/payload/core/run_folder.py +60 -0
  91. package/skills/torusguard/payload/core/run_manager.py +163 -0
  92. package/skills/torusguard/payload/core/runtime_evidence.py +175 -0
  93. package/skills/torusguard/payload/core/runtime_validator.py +246 -0
  94. package/skills/torusguard/payload/core/safety_gate.py +139 -0
  95. package/skills/torusguard/payload/core/sarif.py +189 -0
  96. package/skills/torusguard/payload/core/stack_profiler.py +184 -0
  97. package/skills/torusguard/payload/core/symbol_table.py +91 -0
  98. package/skills/torusguard/payload/core/taint.py +133 -0
  99. package/skills/torusguard/payload/core/taint_graph.py +235 -0
  100. package/skills/torusguard/payload/core/taint_rules.py +268 -0
  101. package/skills/torusguard/payload/core/v070_reporter.py +102 -0
  102. package/skills/torusguard/payload/core/v070_workflow.py +339 -0
  103. package/skills/torusguard/payload/core/v6_reporter.py +180 -0
  104. package/skills/torusguard/payload/core/v6_workflow.py +221 -0
  105. package/skills/torusguard/payload/core/watcher.py +58 -0
  106. package/skills/torusguard/payload/custom_rules/README.md +15 -0
  107. package/skills/torusguard/payload/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
  108. package/skills/torusguard/payload/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
  109. package/skills/torusguard/payload/rules/container/TG-CONT-001-root-user-execution.md +50 -50
  110. package/skills/torusguard/payload/rules/container/TG-CONT-002-docker-socket-mount.md +47 -47
  111. package/skills/torusguard/payload/rules/container/TG-CONT-003-privileged-container-mode.md +53 -53
  112. package/skills/torusguard/payload/rules/container/TG-CONT-004-build-arg-secret-exposure.md +43 -43
  113. package/skills/torusguard/payload/rules/git/TG-GIT-001-historical-secret-in-git-commit.md +44 -44
  114. package/skills/torusguard/payload/rules/git/TG-GIT-002-plaintext-credentials-in-git-config.md +41 -41
  115. package/skills/torusguard/payload/rules/git/TG-GIT-003-sensitive-tracked-file-gitignore-breach.md +40 -40
  116. package/skills/torusguard/payload/rules/rag/TG-RAG-001-untrusted-rag-context-injection.md +72 -72
  117. package/skills/torusguard/payload/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md +51 -51
  118. package/skills/torusguard/payload/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md +51 -51
  119. package/skills/torusguard/payload/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md +46 -46
  120. package/skills/torusguard/payload/rules/redos/TG-REDOS-002-unbounded-nested-quantifier.md +43 -43
  121. package/skills/torusguard/payload/scripts/audit_runner.py +108 -10
  122. package/skills/torusguard/payload/scripts/deliberation_tournament.py +127 -0
  123. package/skills/torusguard/payload/scripts/finding_scorer.py +43 -13
  124. package/skills/torusguard/payload/scripts/reachability_analyzer.py +164 -0
  125. package/skills/torusguard/payload/scripts/stride_generator.py +196 -0
  126. package/skills/torusguard/payload/skills/torusguard/SKILL.md +6 -2
  127. package/skills/torusguard/payload/skills/torusguard/bootstrap.py +3 -3
  128. package/skills/torusguard/payload/skills/torusguard-ai-guard/SKILL.md +95 -95
  129. package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +109 -84
  130. package/skills/torusguard/payload/skills/torusguard-container/SKILL.md +94 -94
  131. package/skills/torusguard/payload/skills/torusguard-git-mine/SKILL.md +92 -92
  132. package/skills/torusguard/payload/skills/torusguard-ocr-scan/SKILL.md +94 -94
  133. package/skills/torusguard/payload/skills/torusguard-redos/SKILL.md +91 -91
  134. package/skills/torusguard/payload/skills/torusguard-review/SKILL.md +27 -0
  135. package/skills/torusguard/payload/skills/torusguard-threatmodel/SKILL.md +27 -0
  136. package/skills/torusguard/payload/workflows/ai-guard.md +31 -31
  137. package/skills/torusguard/payload/workflows/audit.md +21 -17
  138. package/skills/torusguard/payload/workflows/container.md +29 -29
  139. package/skills/torusguard/payload/workflows/git-mine.md +25 -25
  140. package/skills/torusguard/payload/workflows/ocr-scan.md +25 -25
  141. package/skills/torusguard/payload/workflows/redos.md +27 -27
  142. package/skills/torusguard/payload/workflows/review.md +11 -0
  143. package/skills/torusguard/payload/workflows/threatmodel.md +9 -0
  144. package/skills/torusguard/payload/workflows/torusguard-audit.md +35 -55
  145. package/skills/torusguard/references/csharp-security.md +41 -41
  146. package/skills/torusguard/references/go-security.md +41 -41
  147. package/skills/torusguard/references/java-security.md +40 -40
  148. package/skills/torusguard/references/polyglot-security-matrix.md +25 -25
  149. package/skills/torusguard/references/rust-security.md +40 -40
  150. package/skills/torusguard-audit/SKILL.md +107 -83
  151. package/skills/torusguard-review/SKILL.md +27 -0
  152. package/skills/torusguard-threatmodel/SKILL.md +27 -0
@@ -0,0 +1,275 @@
1
+ """
2
+ TorusGuard v6 Root-Cause Clustering Engine
3
+ Groups individual static-analysis findings into cohesive root-cause clusters
4
+ so engineering teams can remediate systemic architectural issues rather than chasing isolated alerts.
5
+ """
6
+
7
+ from dataclasses import dataclass, field, asdict
8
+ from typing import List, Dict, Any, Optional
9
+ import hashlib
10
+
11
+
12
+ # Canonical Cluster Taxonomy Definitions
13
+ KNOWN_ROOT_CAUSES = {
14
+ "TG-DB-004": {
15
+ "cluster_id": "cluster-tenant-isolation",
16
+ "title": "Missing Multi-Tenant Query Scoping & Model Isolation",
17
+ "shared_remediation_path": "Implement tenant-aware BaseManager / default queryset filtering or tenancy middleware context.",
18
+ "shared_verification_plan": "Execute tenant cross-boundary query assertions and recheck query builders."
19
+ },
20
+ "TG-INPUT-006": {
21
+ "cluster_id": "cluster-path-traversal",
22
+ "title": "Unsafe File Upload Storage & Path Traversal Boundaries",
23
+ "shared_remediation_path": "Sanitize filenames using secure_filename() and enforce safe directory resolution with Path.resolve().",
24
+ "shared_verification_plan": "Run path traversal payload test suite and verify storage isolation."
25
+ },
26
+ "TG-INPUT-005": {
27
+ "cluster_id": "cluster-template-escaping",
28
+ "title": "Disabled Template Autoescaping & Unsafe HTML Rendering",
29
+ "shared_remediation_path": "Remove explicit mark_safe() / |safe filters and use autoescaped context variables.",
30
+ "shared_verification_plan": "Execute XSS payload injection test against rendered view outputs."
31
+ },
32
+ "TG-AUTH-008": {
33
+ "cluster_id": "cluster-header-trust",
34
+ "title": "Untrusted Client Header Trust & Role/Tenant Injection",
35
+ "shared_remediation_path": "Derive user identity and role scopes exclusively from cryptographically signed session tokens or trusted gateways.",
36
+ "shared_verification_plan": "Send spoofed client headers (X-User-Role, X-Tenant-ID) and assert rejection."
37
+ },
38
+ "TG-AUTH-007": {
39
+ "cluster_id": "cluster-idor-scoping",
40
+ "title": "Insecure Direct Object Reference (IDOR) on Primary Keys",
41
+ "shared_remediation_path": "Scope database queries with user_id or account ownership filters before returning model instances.",
42
+ "shared_verification_plan": "Run IDOR authorization matrix tests across test accounts."
43
+ },
44
+ "TG-RATE-001": {
45
+ "cluster_id": "cluster-rate-limiting",
46
+ "title": "Unbounded Resource Consumption & Missing Endpoint Throttling",
47
+ "shared_remediation_path": "Apply Redis/in-memory rate limiting middleware or DRF Throttling classes.",
48
+ "shared_verification_plan": "Execute burst traffic simulation and verify 429 Too Many Requests response."
49
+ },
50
+ "TG-SSRF-001": {
51
+ "cluster_id": "cluster-ssrf-network",
52
+ "title": "Unvalidated Outbound HTTP Requests & Network Boundary Leakage",
53
+ "shared_remediation_path": "Validate destination URLs against strict allowlists and block internal IP ranges (127.0.0.1, 169.254.169.254).",
54
+ "shared_verification_plan": "Attempt outbound requests to loopback and link-local metadata endpoints."
55
+ },
56
+ "TG-WEBHOOK-001": {
57
+ "cluster_id": "cluster-webhook-auth",
58
+ "title": "Unverified Inbound Webhook Signatures & Replay Vulnerability",
59
+ "shared_remediation_path": "Verify HMAC signatures using timing-safe comparisons and enforce timestamp freshness bounds.",
60
+ "shared_verification_plan": "Send unsigned and replay webhook payloads and confirm 401/403 rejection."
61
+ },
62
+ "TG-SEC-001": {
63
+ "cluster_id": "cluster-secrets",
64
+ "title": "Hardcoded Secrets & Sensitive Environment Configuration Exposure",
65
+ "shared_remediation_path": "Extract secrets into environment variables (.env / secrets manager) and exclude from version control.",
66
+ "shared_verification_plan": "Audit git history and scan source files for credential patterns."
67
+ },
68
+ "TG-AGENT-001": {
69
+ "cluster_id": "cluster-prompt-injection",
70
+ "title": "AI Agent User Prompt Concatenation & System Prompt Override",
71
+ "shared_remediation_path": "Isolate untrusted user input within explicit inert XML delimiters or structured user-role message arrays.",
72
+ "shared_verification_plan": "Execute prompt injection canary payloads and verify refusal to override system directives."
73
+ },
74
+ "TG-CSRF-001": {
75
+ "cluster_id": "cluster-csrf-missing",
76
+ "title": "Cross-Site Request Forgery (CSRF) & Missing SameSite Cookie Protection",
77
+ "shared_remediation_path": "Enforce SameSite=Lax/Strict on session cookies and validate cryptographic anti-CSRF tokens on state mutations.",
78
+ "shared_verification_plan": "Attempt cross-origin state-changing POST requests without CSRF token."
79
+ },
80
+ "TG-GQL-001": {
81
+ "cluster_id": "cluster-graphql-abuse",
82
+ "title": "Unbounded GraphQL Query Depth & Production Introspection Exposure",
83
+ "shared_remediation_path": "Configure GraphQL query depth limiting (max depth 6) and disable schema introspection in production.",
84
+ "shared_verification_plan": "Send deeply nested recursive query and introspection requests to production endpoint."
85
+ },
86
+ "TG-SUPPLY-001": {
87
+ "cluster_id": "cluster-supply-chain",
88
+ "title": "Supply Chain Vulnerability & Unpinned CI/CD Action Hashes",
89
+ "shared_remediation_path": "Pin GitHub Actions to full immutable commit SHAs and enforce lockfile integrity verification in CI.",
90
+ "shared_verification_plan": "Scan workflow files for mutable tag references and verify lockfile checksums."
91
+ },
92
+ "TG-BIZ-001": {
93
+ "cluster_id": "cluster-business-logic",
94
+ "title": "Business Logic Invariant Violation & Negative Quantity/Discount Bypass",
95
+ "shared_remediation_path": "Assert non-negative amount invariants, enforce transactional locks, and cap discounts server-side.",
96
+ "shared_verification_plan": "Submit negative amount and coupon boundary payloads in transactional API endpoints."
97
+ },
98
+ "TG-CACHE-001": {
99
+ "cluster_id": "cluster-cache-poisoning",
100
+ "title": "Web Cache Poisoning & Missing Cache-Control on Sensitive Responses",
101
+ "shared_remediation_path": "Sanitize unkeyed HTTP request headers and apply 'Cache-Control: no-store' on private or authenticated responses.",
102
+ "shared_verification_plan": "Send unkeyed headers (X-Forwarded-Host) and inspect cached responses."
103
+ },
104
+ "TG-WS-001": {
105
+ "cluster_id": "cluster-websocket-auth",
106
+ "title": "Missing WebSocket Handshake Authentication & Origin Verification",
107
+ "shared_remediation_path": "Authenticate client credentials during WebSocket upgrade and reject connections with unwhitelisted Origin headers.",
108
+ "shared_verification_plan": "Connect from untrusted origin and assert handshake rejection with 403 Forbidden."
109
+ },
110
+ "TG-EDGE-001": {
111
+ "cluster_id": "cluster-edge-abuse",
112
+ "title": "Serverless Subrequest Fan-Out Explosion & Missing Timeout Bounds",
113
+ "shared_remediation_path": "Enforce concurrency caps on edge subrequests and configure strict execution timeouts (<= 15s).",
114
+ "shared_verification_plan": "Simulate high subrequest fan-out and assert bounded concurrency."
115
+ },
116
+ "TG-CONT-001": {
117
+ "cluster_id": "cluster-container-hardening",
118
+ "title": "Container Security Misconfiguration & Root User Execution",
119
+ "shared_remediation_path": "Define non-root USER in Dockerfile, avoid mounting /var/run/docker.sock, and disallow privileged mode.",
120
+ "shared_verification_plan": "Inspect container image metadata and assert non-root UID."
121
+ },
122
+ "TG-GIT-001": {
123
+ "cluster_id": "cluster-git-secret-mining",
124
+ "title": "Committed Credentials & Leaked Tokens in Git Commit History",
125
+ "shared_remediation_path": "Rotate exposed keys immediately, purge secret commits using git-filter-repo, and install pre-commit secret hooks.",
126
+ "shared_verification_plan": "Mine git commit logs for high-entropy tokens and regex matches."
127
+ },
128
+ "TG-REDOS-001": {
129
+ "cluster_id": "cluster-regex-backtracking",
130
+ "title": "Regular Expression Denial of Service (ReDoS) via Nested Quantifiers",
131
+ "shared_remediation_path": "Refactor regex patterns to eliminate nested quantifiers (e.g. (a+)+) or enforce execution timeouts.",
132
+ "shared_verification_plan": "Execute polynomial/exponential adversarial string payloads against regex engine."
133
+ },
134
+ "TG-RAG-001": {
135
+ "cluster_id": "cluster-rag-tenant-leak",
136
+ "title": "RAG Pipeline Vector Database Multi-Tenant Isolation Leakage",
137
+ "shared_remediation_path": "Always include tenant_id or user_id in vector similarity filter metadata (e.g. filter={'tenant_id': tid}).",
138
+ "shared_verification_plan": "Query vector database with tenant A context requesting tenant B documents."
139
+ },
140
+ "TG-INPUT-007": {
141
+ "cluster_id": "cluster-open-redirect",
142
+ "title": "Unvalidated URL Redirection (Open Redirect)",
143
+ "shared_remediation_path": "Validate destination URLs against an allowlist of trusted domains and restrict to relative paths.",
144
+ "shared_verification_plan": "Test redirect endpoints with external and protocol-relative URLs."
145
+ },
146
+ "TG-INPUT-008": {
147
+ "cluster_id": "cluster-deserialization",
148
+ "title": "Insecure Object Deserialization via Untrusted Streams",
149
+ "shared_remediation_path": "Avoid pickle/yaml.load with untrusted input; use safe serialization formats (JSON) or SafeLoader.",
150
+ "shared_verification_plan": "Submit serialized gadget payload and confirm safe parser rejection."
151
+ }
152
+ }
153
+
154
+
155
+
156
+ @dataclass
157
+ class RootCauseCluster:
158
+ cluster_id: str
159
+ title: str
160
+ primary_rule: str
161
+ affected_files: List[str] = field(default_factory=list)
162
+ affected_locations: List[str] = field(default_factory=list)
163
+ finding_ids: List[str] = field(default_factory=list)
164
+ shared_remediation_path: str = ""
165
+ shared_verification_plan: str = ""
166
+ risk_severity: str = "High"
167
+ hotspot_module: str = ""
168
+ is_high_density: bool = False
169
+
170
+ def add_finding(self, finding_id: str, file_path: str, location_str: str, severity: str = "High"):
171
+ if finding_id not in self.finding_ids:
172
+ self.finding_ids.append(finding_id)
173
+ if file_path not in self.affected_files:
174
+ self.affected_files.append(file_path)
175
+ if location_str not in self.affected_locations:
176
+ self.affected_locations.append(location_str)
177
+
178
+ # Update hotspot module (top directory)
179
+ parts = file_path.replace("\\", "/").split("/")
180
+ if len(parts) > 1:
181
+ self.hotspot_module = "/".join(parts[:2])
182
+ else:
183
+ self.hotspot_module = parts[0]
184
+
185
+ # Density threshold (> 5 findings in one cluster is high density)
186
+ if len(self.finding_ids) >= 5:
187
+ self.is_high_density = True
188
+
189
+ # Escalate cluster severity if higher
190
+ severity_order = {"Critical": 4, "High": 3, "Medium": 2, "Low": 1, "Informational": 0}
191
+ if severity_order.get(severity, 0) > severity_order.get(self.risk_severity, 0):
192
+ self.risk_severity = severity
193
+
194
+ def to_dict(self) -> Dict[str, Any]:
195
+ return asdict(self)
196
+
197
+
198
+ # Generated / Vendor File Exclusion Patterns
199
+ IGNORED_PATTERNS = [
200
+ "migrations/", "node_modules/", "dist/", "build/", "vendor/",
201
+ ".venv/", "venv/", ".min.js", ".min.css", ".pb.go", "_pb2.py",
202
+ "bundle.js", ".map"
203
+ ]
204
+
205
+
206
+ def is_generated_file(file_path: str) -> bool:
207
+ norm = file_path.replace("\\", "/").lower()
208
+ return any(p in norm for p in IGNORED_PATTERNS)
209
+
210
+
211
+ class ClusteringEngine:
212
+ """
213
+ Analyzes findings and groups them into root-cause clusters with scale and density metrics.
214
+ """
215
+
216
+ @staticmethod
217
+ def cluster_findings(
218
+ findings: List[Dict[str, Any]],
219
+ filter_generated: bool = False
220
+ ) -> List[RootCauseCluster]:
221
+ clusters: Dict[str, RootCauseCluster] = {}
222
+
223
+ for f in findings:
224
+ target = f.get("target", {})
225
+ file_path = target.get("file_path", "unknown")
226
+
227
+ if filter_generated and is_generated_file(file_path):
228
+ continue
229
+
230
+ rule_id = f.get("rule_id", "TG-GENERIC")
231
+ finding_id = f.get("finding_id", "unknown")
232
+ start_line = target.get("line_start", 0)
233
+ end_line = target.get("line_end", 0)
234
+ loc_str = f"{file_path}:{start_line}-{end_line}"
235
+ severity = f.get("severity", "High")
236
+
237
+ # Determine cluster mapping
238
+ if rule_id in KNOWN_ROOT_CAUSES:
239
+ meta = KNOWN_ROOT_CAUSES[rule_id]
240
+ cid = meta["cluster_id"]
241
+ title = meta["title"]
242
+ rem_path = meta["shared_remediation_path"]
243
+ ver_plan = meta["shared_verification_plan"]
244
+ else:
245
+ cid = f"cluster-{rule_id.lower().replace('tg-', '')}"
246
+ title = f"Systemic {f.get('title', rule_id)} Issues"
247
+ rem_path = "Apply framework-native security controls as documented in rule reference."
248
+ ver_plan = "Re-audit all affected components with /torusguard recheck."
249
+
250
+ if cid not in clusters:
251
+ clusters[cid] = RootCauseCluster(
252
+ cluster_id=cid,
253
+ title=title,
254
+ primary_rule=rule_id,
255
+ shared_remediation_path=rem_path,
256
+ shared_verification_plan=ver_plan,
257
+ risk_severity=severity,
258
+ )
259
+
260
+ clusters[cid].add_finding(
261
+ finding_id=finding_id,
262
+ file_path=file_path,
263
+ location_str=loc_str,
264
+ severity=severity,
265
+ )
266
+
267
+ # Sort clusters by severity and finding count
268
+ severity_order = {"Critical": 4, "High": 3, "Medium": 2, "Low": 1, "Informational": 0}
269
+ sorted_clusters = sorted(
270
+ clusters.values(),
271
+ key=lambda c: (severity_order.get(c.risk_severity, 0), len(c.finding_ids)),
272
+ reverse=True
273
+ )
274
+
275
+ return sorted_clusters
@@ -0,0 +1,120 @@
1
+ """
2
+ TorusGuard Multi-Signal Evidence-Chain Confidence Calibration Engine
3
+ Calibrates finding confidence (0-100) using 7 weighted empirical signals:
4
+ rule severity, taint path confirmation, taint depth, sanitizer absence,
5
+ framework context match, evidence snippet quality, and test fixture suppression.
6
+ """
7
+
8
+ from dataclasses import dataclass, field, asdict
9
+ from typing import Dict, Any, Tuple, Optional
10
+
11
+
12
+ SEVERITY_BASE_SCORES = {
13
+ "Critical": 90,
14
+ "High": 75,
15
+ "Medium": 50,
16
+ "Low": 25,
17
+ "Informational": 10
18
+ }
19
+
20
+
21
+ @dataclass
22
+ class EvidenceSignals:
23
+ rule_severity: str = "High"
24
+ taint_path_confirmed: bool = False
25
+ taint_depth: Optional[int] = None
26
+ sanitizer_present: bool = False
27
+ framework_context_match: bool = True
28
+ has_multiline_evidence: bool = True
29
+ is_test_or_mock: bool = False
30
+ memory_boost: int = 0
31
+
32
+ def to_dict(self) -> Dict[str, Any]:
33
+ return asdict(self)
34
+
35
+
36
+ class ConfidenceCalibrator:
37
+ """Computes evidence-chain calibrated confidence scores."""
38
+
39
+ @staticmethod
40
+ def calculate_score(signals: EvidenceSignals) -> Tuple[int, str, Dict[str, Any]]:
41
+ # 1. rule_severity_base (weight 0.20)
42
+ sev_base = SEVERITY_BASE_SCORES.get(signals.rule_severity, 50)
43
+ w_sev = 0.20 * sev_base
44
+
45
+ # 2. taint_path_confirmed (weight 0.25)
46
+ taint_score = 100 if signals.taint_path_confirmed else 0
47
+ w_taint = 0.25 * taint_score
48
+
49
+ # 3. taint_depth (weight 0.10)
50
+ # direct=100, 1-hop=80, 2-hop=60, 3+=40; if no taint path, 0
51
+ if signals.taint_path_confirmed:
52
+ d = signals.taint_depth if signals.taint_depth is not None else 0
53
+ if d == 0:
54
+ depth_score = 100
55
+ elif d == 1:
56
+ depth_score = 80
57
+ elif d == 2:
58
+ depth_score = 60
59
+ else:
60
+ depth_score = 40
61
+ else:
62
+ depth_score = 0
63
+ w_depth = 0.10 * depth_score
64
+
65
+ # 4. sanitizer_absence (weight 0.15)
66
+ san_score = 0 if signals.sanitizer_present else 100
67
+ w_san = 0.15 * san_score
68
+
69
+ # 5. framework_context_match (weight 0.10)
70
+ ctx_score = 100 if signals.framework_context_match else 50
71
+ w_ctx = 0.10 * ctx_score
72
+
73
+ # 6. evidence_snippet_quality (weight 0.10)
74
+ snippet_score = 100 if signals.has_multiline_evidence else 50
75
+ w_snippet = 0.10 * snippet_score
76
+
77
+ # 7. test fixture penalty
78
+ test_penalty = -50 if signals.is_test_or_mock else 0
79
+
80
+ # Memory boost (-30 to +20)
81
+ mem_boost = signals.memory_boost
82
+
83
+ raw_total = (
84
+ w_sev +
85
+ w_taint +
86
+ w_depth +
87
+ w_san +
88
+ w_ctx +
89
+ w_snippet +
90
+ test_penalty +
91
+ mem_boost
92
+ )
93
+
94
+ final_score = int(round(min(max(raw_total, 0), 100)))
95
+
96
+ # Assign band
97
+ if final_score >= 90:
98
+ band = "Confirmed"
99
+ elif final_score >= 70:
100
+ band = "High Confidence"
101
+ elif final_score >= 50:
102
+ band = "Medium Confidence"
103
+ else:
104
+ band = "Needs Review"
105
+
106
+ breakdown = {
107
+ "rule_severity_score": sev_base,
108
+ "taint_path_confirmed": signals.taint_path_confirmed,
109
+ "taint_depth": signals.taint_depth,
110
+ "sanitizer_present": signals.sanitizer_present,
111
+ "framework_context_match": signals.framework_context_match,
112
+ "has_multiline_evidence": signals.has_multiline_evidence,
113
+ "is_test_or_mock": signals.is_test_or_mock,
114
+ "test_penalty": test_penalty,
115
+ "memory_boost": mem_boost,
116
+ "total_score": final_score,
117
+ "classification_band": band
118
+ }
119
+
120
+ return final_score, band, breakdown
@@ -0,0 +1,101 @@
1
+ """
2
+ TorusGuard Cross-File Interprocedural Taint Analyzer
3
+ Propagates taint across module and function boundaries using project call graphs.
4
+ Enforces strict 5-hop depth bound to prevent recursion and path explosion.
5
+ """
6
+
7
+ from pathlib import Path
8
+ from dataclasses import dataclass, field
9
+ from typing import List, Dict, Set, Optional, Tuple, Any
10
+
11
+ from core.taint import TaintNode, TaintPath
12
+ from core.taint_graph import TaintGraph
13
+ from core.parser import PolyglotParser, ParseResult
14
+ from core.call_graph import CallGraph, CallSite, FunctionDef
15
+
16
+
17
+ class CrossFileTaintAnalyzer:
18
+ """Interprocedural taint analysis across module boundaries."""
19
+
20
+ MAX_INTERPROCEDURAL_DEPTH = 5
21
+
22
+ def __init__(self, project_root: Path):
23
+ self.project_root = project_root.resolve()
24
+ self.parser = PolyglotParser()
25
+ self.call_graph = CallGraph(self.project_root)
26
+
27
+ def analyze_project(self, file_paths: List[Path]) -> List[TaintPath]:
28
+ """
29
+ Builds call graph and walks intra-file and cross-file dataflow chains.
30
+ """
31
+ all_paths: List[TaintPath] = []
32
+ file_graphs: Dict[str, TaintGraph] = {}
33
+ file_contents: Dict[str, str] = {}
34
+ file_languages: Dict[str, str] = {}
35
+
36
+ # 1. Build CallGraph across target files
37
+ self.call_graph.build_from_files(file_paths, self.parser)
38
+
39
+ # 2. Intra-file analysis on each file
40
+ for p in file_paths:
41
+ try:
42
+ rel_p = str(p.resolve().relative_to(self.project_root)).replace("\\", "/")
43
+ content = p.read_text(encoding="utf-8", errors="replace")
44
+ lang = self.parser.detect_language(p)
45
+ file_contents[rel_p] = content
46
+ file_languages[rel_p] = lang
47
+
48
+ tg = TaintGraph()
49
+ paths = tg.analyze_file_code(rel_p, content, lang)
50
+ file_graphs[rel_p] = tg
51
+ all_paths.extend(paths)
52
+ except Exception:
53
+ continue
54
+
55
+ # 3. Interprocedural Propagation:
56
+ # Check if function returns tainted value, propagate to call site LHS
57
+ # Check if call passes tainted argument, propagate to callee parameters
58
+ for (caller_key, call_sites) in self.call_graph.callees.items():
59
+ caller_file, caller_func = caller_key
60
+ tg_caller = file_graphs.get(caller_file)
61
+ if not tg_caller:
62
+ continue
63
+
64
+ for cs in call_sites:
65
+ if not cs.callee_file or cs.callee_file not in file_graphs:
66
+ continue
67
+
68
+ tg_callee = file_graphs[cs.callee_file]
69
+
70
+ # Check argument propagation (caller argument -> callee parameter)
71
+ for arg_expr in cs.arguments:
72
+ # Is this argument connected to a tainted source in caller?
73
+ for src_id, src_node in tg_caller.nodes.items():
74
+ if src_node.node_type == "source":
75
+ # Check if src reaches the argument expression
76
+ arg_node = TaintNode(
77
+ variable_name=arg_expr,
78
+ file_path=caller_file,
79
+ line_number=cs.caller_line,
80
+ node_type="transform",
81
+ expression=arg_expr
82
+ )
83
+ # Look for sinks in the callee file
84
+ for sink_id, sink_node in tg_callee.nodes.items():
85
+ if sink_node.node_type == "sink":
86
+ # Create interprocedural path
87
+ cross_path = [src_node, arg_node, sink_node]
88
+ is_sanitized, sanitizers = tg_callee.is_path_sanitized(
89
+ cross_path,
90
+ sink_node.metadata.get("category", ""),
91
+ file_languages.get(cs.callee_file, "python")
92
+ )
93
+ all_paths.append(TaintPath(
94
+ source=src_node,
95
+ sink=sink_node,
96
+ path=cross_path,
97
+ is_sanitized=is_sanitized,
98
+ sanitizers_found=sanitizers
99
+ ))
100
+
101
+ return all_paths