torusguard 2.0.0-alpha → 2.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.torusguard/.manifest.json +94 -30
- package/.torusguard/core/__init__.py +146 -0
- package/.torusguard/core/agent_roles.py +104 -0
- package/.torusguard/core/ast_walker.py +283 -0
- package/.torusguard/core/authorization.py +218 -0
- package/.torusguard/core/browser_verifier.py +128 -0
- package/.torusguard/core/bundle.py +141 -0
- package/.torusguard/core/call_graph.py +184 -0
- package/.torusguard/core/clustering.py +275 -0
- package/.torusguard/core/confidence.py +120 -0
- package/.torusguard/core/cross_file_taint.py +101 -0
- package/.torusguard/core/exploit_checker.py +317 -0
- package/.torusguard/core/formatter.py +351 -0
- package/.torusguard/core/governance.py +210 -0
- package/.torusguard/core/identity.py +104 -0
- package/.torusguard/core/import_resolver.py +91 -0
- package/.torusguard/core/incremental.py +102 -0
- package/.torusguard/core/lifecycle.py +137 -0
- package/.torusguard/core/models.py +425 -0
- package/.torusguard/core/parallel.py +56 -0
- package/.torusguard/core/parser.py +202 -0
- package/.torusguard/core/rechecker.py +107 -0
- package/.torusguard/core/replay_trace.py +178 -0
- package/.torusguard/core/rules_registry.py +131 -0
- package/.torusguard/core/run_folder.py +60 -0
- package/.torusguard/core/run_manager.py +163 -0
- package/.torusguard/core/runtime_evidence.py +175 -0
- package/.torusguard/core/runtime_validator.py +246 -0
- package/.torusguard/core/safety_gate.py +139 -0
- package/.torusguard/core/sarif.py +189 -0
- package/.torusguard/core/stack_profiler.py +184 -0
- package/.torusguard/core/symbol_table.py +91 -0
- package/.torusguard/core/taint.py +133 -0
- package/.torusguard/core/taint_graph.py +235 -0
- package/.torusguard/core/taint_rules.py +268 -0
- package/.torusguard/core/v070_reporter.py +102 -0
- package/.torusguard/core/v070_workflow.py +339 -0
- package/.torusguard/core/v6_reporter.py +180 -0
- package/.torusguard/core/v6_workflow.py +221 -0
- package/.torusguard/core/watcher.py +58 -0
- package/.torusguard/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
- package/.torusguard/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
- package/.torusguard/rules/container/TG-CONT-001-root-user-execution.md +50 -0
- package/.torusguard/rules/container/TG-CONT-002-docker-socket-mount.md +47 -0
- package/.torusguard/rules/container/TG-CONT-003-privileged-container-mode.md +53 -0
- package/.torusguard/rules/container/TG-CONT-004-build-arg-secret-exposure.md +43 -0
- package/.torusguard/rules/git/TG-GIT-001-historical-secret-in-git-commit.md +44 -0
- package/.torusguard/rules/git/TG-GIT-002-plaintext-credentials-in-git-config.md +41 -0
- package/.torusguard/rules/git/TG-GIT-003-sensitive-tracked-file-gitignore-breach.md +40 -0
- package/.torusguard/rules/rag/TG-RAG-001-untrusted-rag-context-injection.md +72 -0
- package/.torusguard/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md +51 -0
- package/.torusguard/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md +51 -0
- package/.torusguard/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md +46 -0
- package/.torusguard/rules/redos/TG-REDOS-002-unbounded-nested-quantifier.md +43 -0
- package/.torusguard/rules_catalog.json +96 -0
- package/.torusguard/scripts/__pycache__/audit_runner.cpython-314.pyc +0 -0
- package/.torusguard/scripts/__pycache__/finding_scorer.cpython-314.pyc +0 -0
- package/.torusguard/scripts/__pycache__/manifest_builder.cpython-314.pyc +0 -0
- package/.torusguard/scripts/__pycache__/rules_sync.cpython-314.pyc +0 -0
- package/.torusguard/scripts/audit_runner.py +108 -10
- package/.torusguard/scripts/finding_scorer.py +43 -13
- package/.torusguard/scripts/manifest_builder.py +1 -1
- package/.torusguard/scripts/skill_profiler.py +26 -0
- package/.torusguard/skills/torusguard/SKILL.md +74 -25
- package/.torusguard/skills/torusguard/bootstrap.py +57 -24
- package/.torusguard/skills/torusguard-ai-guard/SKILL.md +95 -0
- package/.torusguard/skills/torusguard-apply/SKILL.md +60 -34
- package/.torusguard/skills/torusguard-audit/SKILL.md +131 -57
- package/.torusguard/skills/torusguard-authorize/SKILL.md +48 -6
- package/.torusguard/skills/torusguard-container/SKILL.md +94 -0
- package/.torusguard/skills/torusguard-exploit-check/SKILL.md +50 -6
- package/.torusguard/skills/torusguard-full/SKILL.md +62 -19
- package/.torusguard/skills/torusguard-git-mine/SKILL.md +92 -0
- package/.torusguard/skills/torusguard-harden/SKILL.md +81 -50
- package/.torusguard/skills/torusguard-init/SKILL.md +61 -14
- package/.torusguard/skills/torusguard-ocr-scan/SKILL.md +94 -0
- package/.torusguard/skills/torusguard-recheck/SKILL.md +71 -18
- package/.torusguard/skills/torusguard-redos/SKILL.md +91 -0
- package/.torusguard/skills/torusguard-report/SKILL.md +50 -9
- package/.torusguard/skills/torusguard-status/SKILL.md +63 -10
- package/.torusguard/skills/torusguard-verify/SKILL.md +52 -10
- package/.torusguard/skills/torusguard-web-validate/SKILL.md +53 -8
- package/.torusguard/workflows/ai-guard.md +31 -0
- package/.torusguard/workflows/apply.md +32 -55
- package/.torusguard/workflows/audit.md +35 -49
- package/.torusguard/workflows/authorize.md +27 -50
- package/.torusguard/workflows/container.md +29 -0
- package/.torusguard/workflows/exploit-check.md +28 -50
- package/.torusguard/workflows/git-mine.md +25 -0
- package/.torusguard/workflows/harden.md +29 -48
- package/.torusguard/workflows/init.md +27 -50
- package/.torusguard/workflows/memory.md +18 -23
- package/.torusguard/workflows/ocr-scan.md +25 -0
- package/.torusguard/workflows/recheck.md +28 -46
- package/.torusguard/workflows/redos.md +27 -0
- package/.torusguard/workflows/report.md +33 -52
- package/.torusguard/workflows/status.md +31 -52
- package/.torusguard/workflows/verify.md +29 -49
- package/.torusguard/workflows/web-validate.md +22 -45
- package/README.md +96 -60
- package/package.json +7 -2
- package/skills/torusguard/SKILL.md +75 -24
- package/skills/torusguard/__pycache__/bootstrap.cpython-314.pyc +0 -0
- package/skills/torusguard/bootstrap.py +60 -71
- package/skills/torusguard/payload/.manifest.json +94 -31
- package/skills/torusguard/payload/core/__init__.py +146 -0
- package/skills/torusguard/payload/core/agent_roles.py +104 -0
- package/skills/torusguard/payload/core/ast_walker.py +283 -0
- package/skills/torusguard/payload/core/authorization.py +218 -0
- package/skills/torusguard/payload/core/browser_verifier.py +128 -0
- package/skills/torusguard/payload/core/bundle.py +141 -0
- package/skills/torusguard/payload/core/call_graph.py +184 -0
- package/skills/torusguard/payload/core/clustering.py +275 -0
- package/skills/torusguard/payload/core/confidence.py +120 -0
- package/skills/torusguard/payload/core/cross_file_taint.py +101 -0
- package/skills/torusguard/payload/core/exploit_checker.py +317 -0
- package/skills/torusguard/payload/core/formatter.py +351 -0
- package/skills/torusguard/payload/core/governance.py +210 -0
- package/skills/torusguard/payload/core/identity.py +104 -0
- package/skills/torusguard/payload/core/import_resolver.py +91 -0
- package/skills/torusguard/payload/core/incremental.py +102 -0
- package/skills/torusguard/payload/core/lifecycle.py +137 -0
- package/skills/torusguard/payload/core/models.py +425 -0
- package/skills/torusguard/payload/core/parallel.py +56 -0
- package/skills/torusguard/payload/core/parser.py +202 -0
- package/skills/torusguard/payload/core/rechecker.py +107 -0
- package/skills/torusguard/payload/core/replay_trace.py +178 -0
- package/skills/torusguard/payload/core/rules_registry.py +131 -0
- package/skills/torusguard/payload/core/run_folder.py +60 -0
- package/skills/torusguard/payload/core/run_manager.py +163 -0
- package/skills/torusguard/payload/core/runtime_evidence.py +175 -0
- package/skills/torusguard/payload/core/runtime_validator.py +246 -0
- package/skills/torusguard/payload/core/safety_gate.py +139 -0
- package/skills/torusguard/payload/core/sarif.py +189 -0
- package/skills/torusguard/payload/core/stack_profiler.py +184 -0
- package/skills/torusguard/payload/core/symbol_table.py +91 -0
- package/skills/torusguard/payload/core/taint.py +133 -0
- package/skills/torusguard/payload/core/taint_graph.py +235 -0
- package/skills/torusguard/payload/core/taint_rules.py +268 -0
- package/skills/torusguard/payload/core/v070_reporter.py +102 -0
- package/skills/torusguard/payload/core/v070_workflow.py +339 -0
- package/skills/torusguard/payload/core/v6_reporter.py +180 -0
- package/skills/torusguard/payload/core/v6_workflow.py +221 -0
- package/skills/torusguard/payload/core/watcher.py +58 -0
- package/skills/torusguard/payload/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
- package/skills/torusguard/payload/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
- package/skills/torusguard/payload/rules/container/TG-CONT-001-root-user-execution.md +50 -0
- package/skills/torusguard/payload/rules/container/TG-CONT-002-docker-socket-mount.md +47 -0
- package/skills/torusguard/payload/rules/container/TG-CONT-003-privileged-container-mode.md +53 -0
- package/skills/torusguard/payload/rules/container/TG-CONT-004-build-arg-secret-exposure.md +43 -0
- package/skills/torusguard/payload/rules/git/TG-GIT-001-historical-secret-in-git-commit.md +44 -0
- package/skills/torusguard/payload/rules/git/TG-GIT-002-plaintext-credentials-in-git-config.md +41 -0
- package/skills/torusguard/payload/rules/git/TG-GIT-003-sensitive-tracked-file-gitignore-breach.md +40 -0
- package/skills/torusguard/payload/rules/rag/TG-RAG-001-untrusted-rag-context-injection.md +72 -0
- package/skills/torusguard/payload/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md +51 -0
- package/skills/torusguard/payload/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md +51 -0
- package/skills/torusguard/payload/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md +46 -0
- package/skills/torusguard/payload/rules/redos/TG-REDOS-002-unbounded-nested-quantifier.md +43 -0
- package/skills/torusguard/payload/rules_catalog.json +338 -518
- package/skills/torusguard/payload/scripts/__pycache__/term_ui.cpython-314.pyc +0 -0
- package/skills/torusguard/payload/scripts/audit_runner.py +108 -10
- package/skills/torusguard/payload/scripts/finding_scorer.py +43 -13
- package/skills/torusguard/payload/scripts/manifest_builder.py +1 -1
- package/skills/torusguard/payload/skills/torusguard/SKILL.md +74 -25
- package/skills/torusguard/payload/skills/torusguard/bootstrap.py +57 -24
- package/skills/torusguard/payload/skills/torusguard/references/csharp-security.md +41 -41
- package/skills/torusguard/payload/skills/torusguard/references/go-security.md +41 -41
- package/skills/torusguard/payload/skills/torusguard/references/java-security.md +40 -40
- package/skills/torusguard/payload/skills/torusguard/references/polyglot-security-matrix.md +25 -25
- package/skills/torusguard/payload/skills/torusguard/references/rust-security.md +40 -40
- package/skills/torusguard/payload/skills/torusguard-ai-guard/SKILL.md +95 -0
- package/skills/torusguard/payload/skills/torusguard-apply/SKILL.md +60 -34
- package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +131 -57
- package/skills/torusguard/payload/skills/torusguard-authorize/SKILL.md +48 -6
- package/skills/torusguard/payload/skills/torusguard-container/SKILL.md +94 -0
- package/skills/torusguard/payload/skills/torusguard-exploit-check/SKILL.md +50 -6
- package/skills/torusguard/payload/skills/torusguard-full/SKILL.md +62 -19
- package/skills/torusguard/payload/skills/torusguard-git-mine/SKILL.md +92 -0
- package/skills/torusguard/payload/skills/torusguard-harden/SKILL.md +81 -50
- package/skills/torusguard/payload/skills/torusguard-init/SKILL.md +61 -14
- package/skills/torusguard/payload/skills/torusguard-ocr-scan/SKILL.md +94 -0
- package/skills/torusguard/payload/skills/torusguard-recheck/SKILL.md +71 -18
- package/skills/torusguard/payload/skills/torusguard-redos/SKILL.md +91 -0
- package/skills/torusguard/payload/skills/torusguard-report/SKILL.md +50 -9
- package/skills/torusguard/payload/skills/torusguard-status/SKILL.md +63 -10
- package/skills/torusguard/payload/skills/torusguard-verify/SKILL.md +52 -10
- package/skills/torusguard/payload/skills/torusguard-web-validate/SKILL.md +53 -8
- package/skills/torusguard/payload/workflows/ai-guard.md +31 -0
- package/skills/torusguard/payload/workflows/apply.md +31 -62
- package/skills/torusguard/payload/workflows/audit.md +35 -55
- package/skills/torusguard/payload/workflows/authorize.md +27 -50
- package/skills/torusguard/payload/workflows/container.md +29 -0
- package/skills/torusguard/payload/workflows/exploit-check.md +28 -50
- package/skills/torusguard/payload/workflows/git-mine.md +25 -0
- package/skills/torusguard/payload/workflows/harden.md +28 -52
- package/skills/torusguard/payload/workflows/init.md +27 -56
- package/skills/torusguard/payload/workflows/memory.md +18 -23
- package/skills/torusguard/payload/workflows/ocr-scan.md +25 -0
- package/skills/torusguard/payload/workflows/recheck.md +28 -46
- package/skills/torusguard/payload/workflows/redos.md +27 -0
- package/skills/torusguard/payload/workflows/report.md +39 -62
- package/skills/torusguard/payload/workflows/status.md +31 -55
- package/skills/torusguard/payload/workflows/torusguard-audit.md +35 -55
- package/skills/torusguard/payload/workflows/verify.md +29 -49
- package/skills/torusguard/payload/workflows/web-validate.md +22 -45
- package/skills/torusguard/references/csharp-security.md +41 -0
- package/skills/torusguard/references/go-security.md +41 -0
- package/skills/torusguard/references/java-security.md +40 -0
- package/skills/torusguard/references/polyglot-security-matrix.md +25 -0
- package/skills/torusguard/references/rust-security.md +40 -0
- package/skills/torusguard-ai-guard/SKILL.md +95 -0
- package/skills/torusguard-apply/SKILL.md +60 -34
- package/skills/torusguard-audit/SKILL.md +130 -57
- package/skills/torusguard-authorize/SKILL.md +48 -6
- package/skills/torusguard-container/SKILL.md +94 -0
- package/skills/torusguard-exploit-check/SKILL.md +50 -6
- package/skills/torusguard-full/SKILL.md +62 -19
- package/skills/torusguard-git-mine/SKILL.md +92 -0
- package/skills/torusguard-harden/SKILL.md +81 -50
- package/skills/torusguard-init/SKILL.md +61 -14
- package/skills/torusguard-ocr-scan/SKILL.md +94 -0
- package/skills/torusguard-recheck/SKILL.md +71 -18
- package/skills/torusguard-redos/SKILL.md +91 -0
- package/skills/torusguard-report/SKILL.md +50 -9
- package/skills/torusguard-status/SKILL.md +63 -10
- package/skills/torusguard-verify/SKILL.md +52 -10
- package/skills/torusguard-web-validate/SKILL.md +53 -8
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TorusGuard Core Data Models (v0.5.4)
|
|
3
|
+
Defines canonical Finding, ProvenanceChain, ConfidenceScore, EvidencePackage, RetestRecord, RemediationPriority, and AuditReport objects.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from dataclasses import dataclass, field, asdict
|
|
7
|
+
from enum import Enum
|
|
8
|
+
from typing import List, Optional, Dict, Any
|
|
9
|
+
import datetime
|
|
10
|
+
import hashlib
|
|
11
|
+
import uuid
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class SeverityLevel(str, Enum):
|
|
16
|
+
CRITICAL = "Critical"
|
|
17
|
+
HIGH = "High"
|
|
18
|
+
MEDIUM = "Medium"
|
|
19
|
+
LOW = "Low"
|
|
20
|
+
INFORMATIONAL = "Informational"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class RemediationPriority(str, Enum):
|
|
24
|
+
IMMEDIATE = "Immediate (P0)" # Block deployment / immediate fix
|
|
25
|
+
NEAR_TERM = "Near-Term (P1)" # Fix in current sprint / patch cycle
|
|
26
|
+
BACKLOG = "Backlog (P2)" # Defense-in-depth hardening backlog
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ConfidenceBand(str, Enum):
|
|
30
|
+
CONFIRMED = "Confirmed" # 90 - 100
|
|
31
|
+
HIGH_CONFIDENCE = "High Confidence" # 70 - 89
|
|
32
|
+
MEDIUM_CONFIDENCE = "Medium Confidence" # 50 - 69
|
|
33
|
+
LOW_CONFIDENCE = "Low Confidence" # < 50
|
|
34
|
+
UNCONFIRMED = "Unconfirmed"
|
|
35
|
+
NEEDS_REVIEW = "Needs Review"
|
|
36
|
+
INFORMATIONAL = "Informational"
|
|
37
|
+
NOT_APPLICABLE = "Not Applicable"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class FindingStatus(str, Enum):
|
|
41
|
+
CONFIRMED = "Confirmed"
|
|
42
|
+
HIGH_CONFIDENCE = "High Confidence"
|
|
43
|
+
MEDIUM_CONFIDENCE = "Medium Confidence"
|
|
44
|
+
LOW_CONFIDENCE = "Low Confidence"
|
|
45
|
+
UNCONFIRMED = "Unconfirmed"
|
|
46
|
+
NEEDS_REVIEW = "Needs Review"
|
|
47
|
+
REMEDIATED = "Remediated"
|
|
48
|
+
VERIFIED_FIXED = "Verified Fixed"
|
|
49
|
+
SUPPRESSED = "Suppressed"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class LifecycleStage(str, Enum):
|
|
53
|
+
DETECT = "Detect"
|
|
54
|
+
CLASSIFY = "Classify"
|
|
55
|
+
VERIFY = "Verify"
|
|
56
|
+
REMEDIATE = "Remediate"
|
|
57
|
+
RECHECK = "Recheck"
|
|
58
|
+
ARCHIVE = "Archive"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class TaxonomyCategory(str, Enum):
|
|
62
|
+
AUTH = "authentication-authorization"
|
|
63
|
+
INPUT = "input-validation-encoding"
|
|
64
|
+
DATA = "data-access-orm"
|
|
65
|
+
FILES = "file-upload-handling"
|
|
66
|
+
SECRETS = "secrets-configuration"
|
|
67
|
+
SUPPLY_CHAIN = "dependency-supply-chain"
|
|
68
|
+
NETWORK = "network-ssrf-boundaries"
|
|
69
|
+
BUSINESS_LOGIC = "business-logic-rate-limiting"
|
|
70
|
+
CLIENT_PLATFORM = "client-cache-platform"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class EvidenceType(str, Enum):
|
|
74
|
+
SOURCE = "source"
|
|
75
|
+
RUNTIME = "runtime"
|
|
76
|
+
TEST = "test"
|
|
77
|
+
MANUAL_REVIEW = "manual_review"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class ProvenanceChain:
|
|
82
|
+
discovery_module: str
|
|
83
|
+
triggering_input: str
|
|
84
|
+
evidence_collected: List[str]
|
|
85
|
+
decision_path: List[str]
|
|
86
|
+
verification_step: str
|
|
87
|
+
agent_environment: str = "TorusGuard Engine v0.5.4"
|
|
88
|
+
timestamp: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
89
|
+
|
|
90
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
91
|
+
return asdict(self)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass
|
|
95
|
+
class ConfidenceFactors:
|
|
96
|
+
evidence_quality: int # Max: 35
|
|
97
|
+
reproduction_success: int # Max: 25
|
|
98
|
+
independent_confirmations: int # Max: 15
|
|
99
|
+
environmental_clarity: int # Max: 15
|
|
100
|
+
manual_review_status: int # Max: 10
|
|
101
|
+
|
|
102
|
+
def total_score(self) -> int:
|
|
103
|
+
return max(0, min(100, (
|
|
104
|
+
self.evidence_quality +
|
|
105
|
+
self.reproduction_success +
|
|
106
|
+
self.independent_confirmations +
|
|
107
|
+
self.environmental_clarity +
|
|
108
|
+
self.manual_review_status
|
|
109
|
+
)))
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@dataclass
|
|
113
|
+
class ConfidenceScore:
|
|
114
|
+
factors: ConfidenceFactors
|
|
115
|
+
score: int = field(init=False)
|
|
116
|
+
band: ConfidenceBand = field(init=False)
|
|
117
|
+
rationale: str = ""
|
|
118
|
+
|
|
119
|
+
def __post_init__(self):
|
|
120
|
+
self.score = self.factors.total_score()
|
|
121
|
+
if self.score >= 90:
|
|
122
|
+
self.band = ConfidenceBand.CONFIRMED
|
|
123
|
+
elif self.score >= 70:
|
|
124
|
+
self.band = ConfidenceBand.HIGH_CONFIDENCE
|
|
125
|
+
elif self.score >= 50:
|
|
126
|
+
self.band = ConfidenceBand.MEDIUM_CONFIDENCE
|
|
127
|
+
else:
|
|
128
|
+
self.band = ConfidenceBand.LOW_CONFIDENCE
|
|
129
|
+
|
|
130
|
+
@staticmethod
|
|
131
|
+
def calculate(
|
|
132
|
+
evidence_quality: int = 30,
|
|
133
|
+
reproduction_success: int = 25,
|
|
134
|
+
independent_confirmations: int = 15,
|
|
135
|
+
environmental_clarity: int = 15,
|
|
136
|
+
manual_review_status: int = 5,
|
|
137
|
+
rationale: str = ""
|
|
138
|
+
) -> 'ConfidenceScore':
|
|
139
|
+
factors = ConfidenceFactors(
|
|
140
|
+
evidence_quality=evidence_quality,
|
|
141
|
+
reproduction_success=reproduction_success,
|
|
142
|
+
independent_confirmations=independent_confirmations,
|
|
143
|
+
environmental_clarity=environmental_clarity,
|
|
144
|
+
manual_review_status=manual_review_status,
|
|
145
|
+
)
|
|
146
|
+
return ConfidenceScore(factors=factors, rationale=rationale)
|
|
147
|
+
|
|
148
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
149
|
+
return {
|
|
150
|
+
"score": self.score,
|
|
151
|
+
"band": self.band.value,
|
|
152
|
+
"factors": asdict(self.factors),
|
|
153
|
+
"rationale": self.rationale,
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def mask_sensitive_data(text: str) -> str:
|
|
158
|
+
"""Masks secrets, tokens, API keys, and passwords from report output."""
|
|
159
|
+
text = re.sub(r'sk_live_[0-9a-zA-Z_\-]{6,}', 'sk_live_***REDACTED***', text)
|
|
160
|
+
text = re.sub(r'ghp_[0-9a-zA-Z_\-]{6,}', 'ghp_***REDACTED***', text)
|
|
161
|
+
text = re.sub(r'(Bearer\s+)[A-Za-z0-9\-_=]+\.[A-Za-z0-9\-_=]+\.?[A-Za-z0-9\-_=]*', r'\1***REDACTED_JWT***', text)
|
|
162
|
+
|
|
163
|
+
def redact_kv(m):
|
|
164
|
+
val = m.group(3)
|
|
165
|
+
if "***REDACTED" in val:
|
|
166
|
+
return m.group(0)
|
|
167
|
+
return f"{m.group(1)}{m.group(2)}***REDACTED***{m.group(4)}"
|
|
168
|
+
|
|
169
|
+
text = re.sub(
|
|
170
|
+
r'(?i)(secret[_\-\w]*|password|api[_\-\w]*key|token|auth[_\-\w]*key)(\s*[:=]\s*[\'"])([^\'"]{4,})([\'"])',
|
|
171
|
+
redact_kv,
|
|
172
|
+
text
|
|
173
|
+
)
|
|
174
|
+
return text
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@dataclass
|
|
178
|
+
class Evidence:
|
|
179
|
+
type: EvidenceType
|
|
180
|
+
location: str
|
|
181
|
+
raw_snippet: str
|
|
182
|
+
rationale: str
|
|
183
|
+
confidence_level: ConfidenceBand
|
|
184
|
+
sha256_checksum: str = field(init=False)
|
|
185
|
+
context: Optional[str] = None
|
|
186
|
+
reproduction_notes: Optional[str] = None
|
|
187
|
+
reviewer_notes: Optional[str] = None
|
|
188
|
+
is_sufficient_for_confirmed: bool = False
|
|
189
|
+
collected_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
190
|
+
|
|
191
|
+
def __post_init__(self):
|
|
192
|
+
self.sha256_checksum = hashlib.sha256(self.raw_snippet.strip().encode("utf-8")).hexdigest()
|
|
193
|
+
|
|
194
|
+
def get_masked_snippet(self) -> str:
|
|
195
|
+
return mask_sensitive_data(self.raw_snippet)
|
|
196
|
+
|
|
197
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
198
|
+
return {
|
|
199
|
+
"type": self.type.value if isinstance(self.type, EvidenceType) else self.type,
|
|
200
|
+
"location": self.location,
|
|
201
|
+
"raw_snippet": self.get_masked_snippet(),
|
|
202
|
+
"sha256_checksum": self.sha256_checksum,
|
|
203
|
+
"collected_at": self.collected_at,
|
|
204
|
+
"context": self.context,
|
|
205
|
+
"rationale": self.rationale,
|
|
206
|
+
"confidence_level": self.confidence_level.value if isinstance(self.confidence_level, ConfidenceBand) else self.confidence_level,
|
|
207
|
+
"reproduction_notes": self.reproduction_notes,
|
|
208
|
+
"reviewer_notes": self.reviewer_notes,
|
|
209
|
+
"is_sufficient_for_confirmed": self.is_sufficient_for_confirmed,
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
@dataclass
|
|
214
|
+
class SeverityInfo:
|
|
215
|
+
level: SeverityLevel
|
|
216
|
+
rationale: str
|
|
217
|
+
rubric_justification: str
|
|
218
|
+
|
|
219
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
220
|
+
return {
|
|
221
|
+
"level": self.level.value if isinstance(self.level, SeverityLevel) else self.level,
|
|
222
|
+
"rationale": self.rationale,
|
|
223
|
+
"rubric_justification": self.rubric_justification,
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
@dataclass
|
|
228
|
+
class FrameworkPattern:
|
|
229
|
+
framework: str
|
|
230
|
+
unsafe_snippet: str
|
|
231
|
+
safe_snippet: str
|
|
232
|
+
least_invasive: bool = True
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
@dataclass
|
|
236
|
+
class Remediation:
|
|
237
|
+
problem_statement: str
|
|
238
|
+
risk_explanation: str
|
|
239
|
+
recommended_fix: str
|
|
240
|
+
framework_pattern: FrameworkPattern
|
|
241
|
+
verification_method: str
|
|
242
|
+
residual_risk_notes: str
|
|
243
|
+
|
|
244
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
245
|
+
return {
|
|
246
|
+
"problem_statement": self.problem_statement,
|
|
247
|
+
"risk_explanation": self.risk_explanation,
|
|
248
|
+
"recommended_fix": self.recommended_fix,
|
|
249
|
+
"framework_pattern": asdict(self.framework_pattern),
|
|
250
|
+
"verification_method": self.verification_method,
|
|
251
|
+
"residual_risk_notes": self.residual_risk_notes,
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
@dataclass
|
|
256
|
+
class AffectedComponent:
|
|
257
|
+
component_name: str
|
|
258
|
+
target_path: str
|
|
259
|
+
start_line: Optional[int] = None
|
|
260
|
+
end_line: Optional[int] = None
|
|
261
|
+
symbol: Optional[str] = None
|
|
262
|
+
|
|
263
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
264
|
+
return {k: v for k, v in asdict(self).items() if v is not None}
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
@dataclass
|
|
268
|
+
class ReproductionMethod:
|
|
269
|
+
step_by_step: List[str]
|
|
270
|
+
deterministic: bool = True
|
|
271
|
+
test_command: Optional[str] = None
|
|
272
|
+
expected_failure_response: Optional[str] = None
|
|
273
|
+
|
|
274
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
275
|
+
return {k: v for k, v in asdict(self).items() if v is not None}
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
@dataclass
|
|
279
|
+
class RetestRecord:
|
|
280
|
+
retest_performed: bool = False
|
|
281
|
+
closure_status: FindingStatus = FindingStatus.UNCONFIRMED
|
|
282
|
+
fix_applied: Optional[str] = None
|
|
283
|
+
retest_method: Optional[str] = None
|
|
284
|
+
retest_evidence_hash: Optional[str] = None
|
|
285
|
+
residual_risk: Optional[str] = None
|
|
286
|
+
verifier_notes: Optional[str] = None
|
|
287
|
+
retest_timestamp: Optional[str] = None
|
|
288
|
+
|
|
289
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
290
|
+
d = asdict(self)
|
|
291
|
+
d["closure_status"] = self.closure_status.value if isinstance(self.closure_status, FindingStatus) else self.closure_status
|
|
292
|
+
return {k: v for k, v in d.items() if v is not None}
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
@dataclass
|
|
296
|
+
class NotesRecord:
|
|
297
|
+
business_impact: str
|
|
298
|
+
technical_description: str
|
|
299
|
+
raw_facts_summary: str
|
|
300
|
+
ai_interpretation: str
|
|
301
|
+
|
|
302
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
303
|
+
return asdict(self)
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
@dataclass
|
|
307
|
+
class FindingTimestamps:
|
|
308
|
+
discovered_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
309
|
+
updated_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
310
|
+
verified_at: Optional[str] = None
|
|
311
|
+
remediated_at: Optional[str] = None
|
|
312
|
+
retested_at: Optional[str] = None
|
|
313
|
+
|
|
314
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
315
|
+
return {k: v for k, v in asdict(self).items() if v is not None}
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
@dataclass
|
|
319
|
+
class Finding:
|
|
320
|
+
rule_id: str
|
|
321
|
+
title: str
|
|
322
|
+
category: TaxonomyCategory
|
|
323
|
+
severity: SeverityInfo
|
|
324
|
+
confidence: ConfidenceScore
|
|
325
|
+
status: FindingStatus
|
|
326
|
+
affected_component: AffectedComponent
|
|
327
|
+
evidence: List[Evidence]
|
|
328
|
+
provenance: ProvenanceChain
|
|
329
|
+
reproduction_method: ReproductionMethod
|
|
330
|
+
remediation: Remediation
|
|
331
|
+
remediation_priority: RemediationPriority = RemediationPriority.IMMEDIATE
|
|
332
|
+
retest_result: RetestRecord = field(default_factory=RetestRecord)
|
|
333
|
+
timestamps: FindingTimestamps = field(default_factory=FindingTimestamps)
|
|
334
|
+
notes: NotesRecord = field(default_factory=lambda: NotesRecord(
|
|
335
|
+
business_impact="Exposure of sensitive application resources or unauthorized data modification.",
|
|
336
|
+
technical_description="Direct unmitigated pattern detected in application route or data layer.",
|
|
337
|
+
raw_facts_summary="Unmitigated pattern identified in source.",
|
|
338
|
+
ai_interpretation="High priority fix recommended.",
|
|
339
|
+
))
|
|
340
|
+
finding_id: str = field(default_factory=lambda: f"TG-FIND-{datetime.datetime.utcnow().year}-{uuid.uuid4().hex[:6]}")
|
|
341
|
+
lifecycle_stage: LifecycleStage = LifecycleStage.DETECT
|
|
342
|
+
asvs_control: Optional[str] = None
|
|
343
|
+
cwe: Optional[str] = None
|
|
344
|
+
nist_ssdf: Optional[str] = None
|
|
345
|
+
|
|
346
|
+
def __post_init__(self):
|
|
347
|
+
# Auto-derive remediation priority from severity level if default
|
|
348
|
+
if self.severity.level == SeverityLevel.CRITICAL:
|
|
349
|
+
self.remediation_priority = RemediationPriority.IMMEDIATE
|
|
350
|
+
elif self.severity.level == SeverityLevel.HIGH:
|
|
351
|
+
self.remediation_priority = RemediationPriority.NEAR_TERM
|
|
352
|
+
else:
|
|
353
|
+
self.remediation_priority = RemediationPriority.BACKLOG
|
|
354
|
+
|
|
355
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
356
|
+
return {
|
|
357
|
+
"finding_id": self.finding_id,
|
|
358
|
+
"rule_id": self.rule_id,
|
|
359
|
+
"title": self.title,
|
|
360
|
+
"category": self.category.value if isinstance(self.category, TaxonomyCategory) else self.category,
|
|
361
|
+
"severity": self.severity.to_dict(),
|
|
362
|
+
"confidence": self.confidence.to_dict(),
|
|
363
|
+
"status": self.status.value if isinstance(self.status, FindingStatus) else self.status,
|
|
364
|
+
"remediation_priority": self.remediation_priority.value if isinstance(self.remediation_priority, RemediationPriority) else self.remediation_priority,
|
|
365
|
+
"lifecycle_stage": self.lifecycle_stage.value if isinstance(self.lifecycle_stage, LifecycleStage) else self.lifecycle_stage,
|
|
366
|
+
"affected_component": self.affected_component.to_dict(),
|
|
367
|
+
"evidence": [e.to_dict() for e in self.evidence],
|
|
368
|
+
"provenance": self.provenance.to_dict(),
|
|
369
|
+
"reproduction_method": self.reproduction_method.to_dict(),
|
|
370
|
+
"remediation": self.remediation.to_dict(),
|
|
371
|
+
"retest_result": self.retest_result.to_dict(),
|
|
372
|
+
"requirement_reference": {
|
|
373
|
+
"asvs_v4": self.asvs_control,
|
|
374
|
+
"cwe": self.cwe,
|
|
375
|
+
"nist_ssdf": self.nist_ssdf,
|
|
376
|
+
},
|
|
377
|
+
"cwe": self.cwe,
|
|
378
|
+
"timestamps": self.timestamps.to_dict(),
|
|
379
|
+
"notes": self.notes.to_dict(),
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
@dataclass
|
|
384
|
+
class AuditReport:
|
|
385
|
+
project_name: str
|
|
386
|
+
detected_stack: Dict[str, Any]
|
|
387
|
+
findings: List[Finding]
|
|
388
|
+
summary_counts: Dict[str, Any] = field(default_factory=dict)
|
|
389
|
+
generated_at: str = field(default_factory=lambda: datetime.datetime.utcnow().isoformat() + "Z")
|
|
390
|
+
torusguard_version: str = "v0.5.4"
|
|
391
|
+
report_owner: str = "TorusGuard Security Subsystem"
|
|
392
|
+
repository_ref: str = "workspace"
|
|
393
|
+
|
|
394
|
+
def calculate_summary(self) -> None:
|
|
395
|
+
total = len(self.findings)
|
|
396
|
+
avg_confidence = round(sum(f.confidence.score for f in self.findings) / total, 1) if total > 0 else 100.0
|
|
397
|
+
self.summary_counts = {
|
|
398
|
+
"total_findings": total,
|
|
399
|
+
"average_confidence_score": avg_confidence,
|
|
400
|
+
"critical": sum(1 for f in self.findings if f.severity.level == SeverityLevel.CRITICAL),
|
|
401
|
+
"high": sum(1 for f in self.findings if f.severity.level == SeverityLevel.HIGH),
|
|
402
|
+
"medium": sum(1 for f in self.findings if f.severity.level == SeverityLevel.MEDIUM),
|
|
403
|
+
"low": sum(1 for f in self.findings if f.severity.level == SeverityLevel.LOW),
|
|
404
|
+
"confirmed": sum(1 for f in self.findings if f.confidence.band == ConfidenceBand.CONFIRMED),
|
|
405
|
+
"high_confidence": sum(1 for f in self.findings if f.confidence.band == ConfidenceBand.HIGH_CONFIDENCE),
|
|
406
|
+
"needs_review": sum(1 for f in self.findings if f.confidence.band in (ConfidenceBand.NEEDS_REVIEW, ConfidenceBand.LOW_CONFIDENCE)),
|
|
407
|
+
"verified_fixed": sum(1 for f in self.findings if f.status == FindingStatus.VERIFIED_FIXED),
|
|
408
|
+
"remediated": sum(1 for f in self.findings if f.status in (FindingStatus.REMEDIATED, FindingStatus.VERIFIED_FIXED)),
|
|
409
|
+
"immediate_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.IMMEDIATE),
|
|
410
|
+
"near_term_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.NEAR_TERM),
|
|
411
|
+
"backlog_priority": sum(1 for f in self.findings if f.remediation_priority == RemediationPriority.BACKLOG),
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
415
|
+
self.calculate_summary()
|
|
416
|
+
return {
|
|
417
|
+
"project_name": self.project_name,
|
|
418
|
+
"torusguard_version": self.torusguard_version,
|
|
419
|
+
"generated_at": self.generated_at,
|
|
420
|
+
"report_owner": self.report_owner,
|
|
421
|
+
"repository_ref": self.repository_ref,
|
|
422
|
+
"detected_stack": self.detected_stack,
|
|
423
|
+
"summary": self.summary_counts,
|
|
424
|
+
"findings": [f.to_dict() for f in self.findings],
|
|
425
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TorusGuard Parallel Audit Executor
|
|
3
|
+
Executes multi-threaded static security scanning across files with deterministic result collation.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import List, Dict, Any, Callable, Optional
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ParallelAuditExecutor:
|
|
13
|
+
"""Distributes file analysis across a thread pool with deterministic output ordering."""
|
|
14
|
+
|
|
15
|
+
def __init__(self, max_workers: Optional[int] = None):
|
|
16
|
+
self.max_workers = max_workers or min(os.cpu_count() or 4, 8)
|
|
17
|
+
|
|
18
|
+
def scan_files_parallel(
|
|
19
|
+
self,
|
|
20
|
+
files: List[Path],
|
|
21
|
+
scan_fn: Callable[[Path], List[Dict[str, Any]]]
|
|
22
|
+
) -> List[Dict[str, Any]]:
|
|
23
|
+
"""
|
|
24
|
+
Executes `scan_fn` across all files in parallel.
|
|
25
|
+
Returns deterministically ordered list of all findings.
|
|
26
|
+
"""
|
|
27
|
+
if not files:
|
|
28
|
+
return []
|
|
29
|
+
|
|
30
|
+
# If only a few files, avoid thread pool overhead
|
|
31
|
+
if len(files) <= 3:
|
|
32
|
+
all_findings = []
|
|
33
|
+
for f in files:
|
|
34
|
+
all_findings.extend(scan_fn(f))
|
|
35
|
+
return all_findings
|
|
36
|
+
|
|
37
|
+
all_findings: List[Dict[str, Any]] = []
|
|
38
|
+
file_results: Dict[str, List[Dict[str, Any]]] = {}
|
|
39
|
+
|
|
40
|
+
with ThreadPoolExecutor(max_workers=self.max_workers) as executor:
|
|
41
|
+
future_to_file = {executor.submit(scan_fn, f): str(f) for f in files}
|
|
42
|
+
for future in as_completed(future_to_file):
|
|
43
|
+
f_str = future_to_file[future]
|
|
44
|
+
try:
|
|
45
|
+
res = future.result()
|
|
46
|
+
file_results[f_str] = res
|
|
47
|
+
except Exception:
|
|
48
|
+
file_results[f_str] = []
|
|
49
|
+
|
|
50
|
+
# Deterministic sort by original file order
|
|
51
|
+
for f in files:
|
|
52
|
+
f_str = str(f)
|
|
53
|
+
if f_str in file_results:
|
|
54
|
+
all_findings.extend(file_results[f_str])
|
|
55
|
+
|
|
56
|
+
return all_findings
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TorusGuard Polyglot Parser
|
|
3
|
+
Unified Tree-sitter parser wrapper with language auto-detection and resilient
|
|
4
|
+
fallback parsing across Python, JavaScript, TypeScript, Go, Rust, Java, Ruby, PHP, and C#.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import Dict, List, Optional, Any
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
from core.ast_walker import FunctionCall, Assignment, ImportStatement, StringLiteral, TreeSitterWalker
|
|
13
|
+
from core.symbol_table import SymbolTable
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
LANGUAGE_EXTENSIONS: Dict[str, str] = {
|
|
17
|
+
".py": "python",
|
|
18
|
+
".js": "javascript",
|
|
19
|
+
".mjs": "javascript",
|
|
20
|
+
".cjs": "javascript",
|
|
21
|
+
".ts": "typescript",
|
|
22
|
+
".tsx": "tsx",
|
|
23
|
+
".go": "go",
|
|
24
|
+
".rs": "rust",
|
|
25
|
+
".java": "java",
|
|
26
|
+
".rb": "ruby",
|
|
27
|
+
".php": "php",
|
|
28
|
+
".cs": "csharp",
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class ParseResult:
|
|
34
|
+
file_path: str
|
|
35
|
+
language: str
|
|
36
|
+
function_calls: List[FunctionCall] = field(default_factory=list)
|
|
37
|
+
assignments: List[Assignment] = field(default_factory=list)
|
|
38
|
+
imports: List[ImportStatement] = field(default_factory=list)
|
|
39
|
+
string_literals: List[StringLiteral] = field(default_factory=list)
|
|
40
|
+
symbol_table: SymbolTable = field(default_factory=SymbolTable)
|
|
41
|
+
is_tree_sitter: bool = False
|
|
42
|
+
raw_content: str = ""
|
|
43
|
+
|
|
44
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
45
|
+
return {
|
|
46
|
+
"file_path": self.file_path,
|
|
47
|
+
"language": self.language,
|
|
48
|
+
"function_calls_count": len(self.function_calls),
|
|
49
|
+
"assignments_count": len(self.assignments),
|
|
50
|
+
"imports_count": len(self.imports),
|
|
51
|
+
"is_tree_sitter": self.is_tree_sitter,
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class PolyglotParser:
|
|
56
|
+
"""Parses source files into Tree-sitter ASTs or fallback token models with language auto-detection."""
|
|
57
|
+
|
|
58
|
+
def __init__(self):
|
|
59
|
+
self._parsers: Dict[str, Any] = {}
|
|
60
|
+
self._init_tree_sitter()
|
|
61
|
+
|
|
62
|
+
def _init_tree_sitter(self):
|
|
63
|
+
try:
|
|
64
|
+
import tree_sitter
|
|
65
|
+
# Python
|
|
66
|
+
try:
|
|
67
|
+
import tree_sitter_python
|
|
68
|
+
py_lang = tree_sitter.Language(tree_sitter_python.language())
|
|
69
|
+
self._parsers["python"] = tree_sitter.Parser(py_lang)
|
|
70
|
+
except Exception:
|
|
71
|
+
pass
|
|
72
|
+
|
|
73
|
+
# JavaScript
|
|
74
|
+
try:
|
|
75
|
+
import tree_sitter_javascript
|
|
76
|
+
js_lang = tree_sitter.Language(tree_sitter_javascript.language())
|
|
77
|
+
self._parsers["javascript"] = tree_sitter.Parser(js_lang)
|
|
78
|
+
except Exception:
|
|
79
|
+
pass
|
|
80
|
+
|
|
81
|
+
# TypeScript / TSX
|
|
82
|
+
try:
|
|
83
|
+
import tree_sitter_typescript
|
|
84
|
+
ts_lang = tree_sitter.Language(tree_sitter_typescript.language_typescript())
|
|
85
|
+
self._parsers["typescript"] = tree_sitter.Parser(ts_lang)
|
|
86
|
+
tsx_lang = tree_sitter.Language(tree_sitter_typescript.language_tsx())
|
|
87
|
+
self._parsers["tsx"] = tree_sitter.Parser(tsx_lang)
|
|
88
|
+
except Exception:
|
|
89
|
+
pass
|
|
90
|
+
|
|
91
|
+
# Go
|
|
92
|
+
try:
|
|
93
|
+
import tree_sitter_go
|
|
94
|
+
go_lang = tree_sitter.Language(tree_sitter_go.language())
|
|
95
|
+
self._parsers["go"] = tree_sitter.Parser(go_lang)
|
|
96
|
+
except Exception:
|
|
97
|
+
pass
|
|
98
|
+
|
|
99
|
+
except Exception:
|
|
100
|
+
pass
|
|
101
|
+
|
|
102
|
+
def detect_language(self, file_path: Path) -> str:
|
|
103
|
+
ext = file_path.suffix.lower()
|
|
104
|
+
return LANGUAGE_EXTENSIONS.get(ext, "unknown")
|
|
105
|
+
|
|
106
|
+
def parse_file(self, file_path: Path) -> ParseResult:
|
|
107
|
+
lang = self.detect_language(file_path)
|
|
108
|
+
try:
|
|
109
|
+
content = file_path.read_text(encoding="utf-8", errors="replace")
|
|
110
|
+
except Exception:
|
|
111
|
+
return ParseResult(file_path=str(file_path), language=lang)
|
|
112
|
+
|
|
113
|
+
source_bytes = content.encode("utf-8")
|
|
114
|
+
ts_parser = self._parsers.get(lang)
|
|
115
|
+
|
|
116
|
+
if ts_parser is not None:
|
|
117
|
+
try:
|
|
118
|
+
tree = ts_parser.parse(source_bytes)
|
|
119
|
+
calls = TreeSitterWalker.extract_function_calls(tree.root_node, source_bytes)
|
|
120
|
+
assigns = TreeSitterWalker.extract_assignments(tree.root_node, source_bytes)
|
|
121
|
+
imports = TreeSitterWalker.extract_imports(tree.root_node, source_bytes, lang)
|
|
122
|
+
strings = TreeSitterWalker.extract_string_literals(tree.root_node, source_bytes)
|
|
123
|
+
|
|
124
|
+
sym_table = SymbolTable(str(file_path))
|
|
125
|
+
for a in assigns:
|
|
126
|
+
sym_table.add_variable(name=a.target, line=a.line_number, expression=a.value_expression)
|
|
127
|
+
for imp in imports:
|
|
128
|
+
for alias, orig in imp.alias_map.items():
|
|
129
|
+
sym_table.add_import_alias(alias, orig)
|
|
130
|
+
|
|
131
|
+
return ParseResult(
|
|
132
|
+
file_path=str(file_path),
|
|
133
|
+
language=lang,
|
|
134
|
+
function_calls=calls,
|
|
135
|
+
assignments=assigns,
|
|
136
|
+
imports=imports,
|
|
137
|
+
string_literals=strings,
|
|
138
|
+
symbol_table=sym_table,
|
|
139
|
+
is_tree_sitter=True,
|
|
140
|
+
raw_content=content
|
|
141
|
+
)
|
|
142
|
+
except Exception:
|
|
143
|
+
pass
|
|
144
|
+
|
|
145
|
+
# Resilient Fallback Parser
|
|
146
|
+
return self._fallback_parse(str(file_path), content, lang)
|
|
147
|
+
|
|
148
|
+
def _fallback_parse(self, file_path: str, content: str, lang: str) -> ParseResult:
|
|
149
|
+
lines = content.splitlines()
|
|
150
|
+
calls: List[FunctionCall] = []
|
|
151
|
+
assigns: List[Assignment] = []
|
|
152
|
+
imports: List[ImportStatement] = []
|
|
153
|
+
strings: List[StringLiteral] = []
|
|
154
|
+
sym_table = SymbolTable(file_path)
|
|
155
|
+
|
|
156
|
+
for idx, line in enumerate(lines):
|
|
157
|
+
line_num = idx + 1
|
|
158
|
+
stripped = line.strip()
|
|
159
|
+
if not stripped or stripped.startswith(("#", "//", "/*", "*")):
|
|
160
|
+
continue
|
|
161
|
+
|
|
162
|
+
# Assignments
|
|
163
|
+
assign_match = re.match(r"^(?:const|let|var)?\s*([a-zA-Z_][a-zA-Z0-9_]*)\s*(?::=[=]?|=)\s*(.+)$", stripped)
|
|
164
|
+
if assign_match:
|
|
165
|
+
target = assign_match.group(1).strip()
|
|
166
|
+
val = assign_match.group(2).strip()
|
|
167
|
+
assigns.append(Assignment(target=target, value_expression=val, line_number=line_num, column=0))
|
|
168
|
+
sym_table.add_variable(target, line_num, expression=val)
|
|
169
|
+
|
|
170
|
+
# Function calls (e.g. foo(x), obj.method(y))
|
|
171
|
+
call_matches = re.finditer(r"([a-zA-Z_][a-zA-Z0-9_\.]*)\s*\((.*?)\)", stripped)
|
|
172
|
+
for cm in call_matches:
|
|
173
|
+
func_name = cm.group(1)
|
|
174
|
+
args_str = cm.group(2)
|
|
175
|
+
args = [a.strip() for a in args_str.split(",") if a.strip()]
|
|
176
|
+
calls.append(FunctionCall(
|
|
177
|
+
name=func_name,
|
|
178
|
+
full_call=cm.group(0),
|
|
179
|
+
arguments=args,
|
|
180
|
+
line_number=line_num,
|
|
181
|
+
column=cm.start()
|
|
182
|
+
))
|
|
183
|
+
|
|
184
|
+
# Imports
|
|
185
|
+
if "import " in stripped or "require(" in stripped or "from " in stripped:
|
|
186
|
+
imp_match = re.search(r"(?:from\s+([a-zA-Z0-9_\.]+)\s+import\s+([a-zA-Z0-9_,\s]+)|import\s+([a-zA-Z0-9_\.]+))", stripped)
|
|
187
|
+
if imp_match:
|
|
188
|
+
mod = imp_match.group(1) or imp_match.group(3) or ""
|
|
189
|
+
names = [n.strip() for n in (imp_match.group(2) or "").split(",") if n.strip()]
|
|
190
|
+
imports.append(ImportStatement(module=mod, imported_names=names, line_number=line_num))
|
|
191
|
+
|
|
192
|
+
return ParseResult(
|
|
193
|
+
file_path=file_path,
|
|
194
|
+
language=lang,
|
|
195
|
+
function_calls=calls,
|
|
196
|
+
assignments=assigns,
|
|
197
|
+
imports=imports,
|
|
198
|
+
string_literals=strings,
|
|
199
|
+
symbol_table=sym_table,
|
|
200
|
+
is_tree_sitter=False,
|
|
201
|
+
raw_content=content
|
|
202
|
+
)
|