torusguard 2.1.0 → 2.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.torusguard/.manifest.json +47 -5
- package/.torusguard/core/__init__.py +146 -0
- package/.torusguard/core/agent_roles.py +104 -0
- package/.torusguard/core/ast_walker.py +283 -0
- package/.torusguard/core/authorization.py +218 -0
- package/.torusguard/core/browser_verifier.py +128 -0
- package/.torusguard/core/bundle.py +141 -0
- package/.torusguard/core/call_graph.py +184 -0
- package/.torusguard/core/clustering.py +275 -0
- package/.torusguard/core/confidence.py +120 -0
- package/.torusguard/core/cross_file_taint.py +101 -0
- package/.torusguard/core/exploit_checker.py +317 -0
- package/.torusguard/core/formatter.py +351 -0
- package/.torusguard/core/governance.py +210 -0
- package/.torusguard/core/identity.py +104 -0
- package/.torusguard/core/import_resolver.py +91 -0
- package/.torusguard/core/incremental.py +102 -0
- package/.torusguard/core/lifecycle.py +137 -0
- package/.torusguard/core/models.py +425 -0
- package/.torusguard/core/parallel.py +56 -0
- package/.torusguard/core/parser.py +202 -0
- package/.torusguard/core/rechecker.py +107 -0
- package/.torusguard/core/replay_trace.py +178 -0
- package/.torusguard/core/rules_registry.py +131 -0
- package/.torusguard/core/run_folder.py +60 -0
- package/.torusguard/core/run_manager.py +163 -0
- package/.torusguard/core/runtime_evidence.py +175 -0
- package/.torusguard/core/runtime_validator.py +246 -0
- package/.torusguard/core/safety_gate.py +139 -0
- package/.torusguard/core/sarif.py +189 -0
- package/.torusguard/core/stack_profiler.py +184 -0
- package/.torusguard/core/symbol_table.py +91 -0
- package/.torusguard/core/taint.py +133 -0
- package/.torusguard/core/taint_graph.py +235 -0
- package/.torusguard/core/taint_rules.py +268 -0
- package/.torusguard/core/v070_reporter.py +102 -0
- package/.torusguard/core/v070_workflow.py +339 -0
- package/.torusguard/core/v6_reporter.py +180 -0
- package/.torusguard/core/v6_workflow.py +221 -0
- package/.torusguard/core/watcher.py +58 -0
- package/.torusguard/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
- package/.torusguard/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
- package/.torusguard/scripts/__pycache__/audit_runner.cpython-314.pyc +0 -0
- package/.torusguard/scripts/__pycache__/finding_scorer.cpython-314.pyc +0 -0
- package/.torusguard/scripts/__pycache__/rules_sync.cpython-314.pyc +0 -0
- package/.torusguard/scripts/audit_runner.py +108 -10
- package/.torusguard/scripts/finding_scorer.py +43 -13
- package/.torusguard/scripts/skill_profiler.py +26 -0
- package/.torusguard/skills/torusguard/SKILL.md +6 -2
- package/.torusguard/skills/torusguard-audit/SKILL.md +109 -84
- package/.torusguard/workflows/audit.md +21 -17
- package/README.md +19 -11
- package/package.json +7 -2
- package/skills/torusguard/SKILL.md +6 -2
- package/skills/torusguard/__pycache__/bootstrap.cpython-314.pyc +0 -0
- package/skills/torusguard/bootstrap.py +3 -3
- package/skills/torusguard/payload/.manifest.json +48 -7
- package/skills/torusguard/payload/core/__init__.py +146 -0
- package/skills/torusguard/payload/core/agent_roles.py +104 -0
- package/skills/torusguard/payload/core/ast_walker.py +283 -0
- package/skills/torusguard/payload/core/authorization.py +218 -0
- package/skills/torusguard/payload/core/browser_verifier.py +128 -0
- package/skills/torusguard/payload/core/bundle.py +141 -0
- package/skills/torusguard/payload/core/call_graph.py +184 -0
- package/skills/torusguard/payload/core/clustering.py +275 -0
- package/skills/torusguard/payload/core/confidence.py +120 -0
- package/skills/torusguard/payload/core/cross_file_taint.py +101 -0
- package/skills/torusguard/payload/core/exploit_checker.py +317 -0
- package/skills/torusguard/payload/core/formatter.py +351 -0
- package/skills/torusguard/payload/core/governance.py +210 -0
- package/skills/torusguard/payload/core/identity.py +104 -0
- package/skills/torusguard/payload/core/import_resolver.py +91 -0
- package/skills/torusguard/payload/core/incremental.py +102 -0
- package/skills/torusguard/payload/core/lifecycle.py +137 -0
- package/skills/torusguard/payload/core/models.py +425 -0
- package/skills/torusguard/payload/core/parallel.py +56 -0
- package/skills/torusguard/payload/core/parser.py +202 -0
- package/skills/torusguard/payload/core/rechecker.py +107 -0
- package/skills/torusguard/payload/core/replay_trace.py +178 -0
- package/skills/torusguard/payload/core/rules_registry.py +131 -0
- package/skills/torusguard/payload/core/run_folder.py +60 -0
- package/skills/torusguard/payload/core/run_manager.py +163 -0
- package/skills/torusguard/payload/core/runtime_evidence.py +175 -0
- package/skills/torusguard/payload/core/runtime_validator.py +246 -0
- package/skills/torusguard/payload/core/safety_gate.py +139 -0
- package/skills/torusguard/payload/core/sarif.py +189 -0
- package/skills/torusguard/payload/core/stack_profiler.py +184 -0
- package/skills/torusguard/payload/core/symbol_table.py +91 -0
- package/skills/torusguard/payload/core/taint.py +133 -0
- package/skills/torusguard/payload/core/taint_graph.py +235 -0
- package/skills/torusguard/payload/core/taint_rules.py +268 -0
- package/skills/torusguard/payload/core/v070_reporter.py +102 -0
- package/skills/torusguard/payload/core/v070_workflow.py +339 -0
- package/skills/torusguard/payload/core/v6_reporter.py +180 -0
- package/skills/torusguard/payload/core/v6_workflow.py +221 -0
- package/skills/torusguard/payload/core/watcher.py +58 -0
- package/skills/torusguard/payload/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
- package/skills/torusguard/payload/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
- package/skills/torusguard/payload/rules/container/TG-CONT-001-root-user-execution.md +50 -50
- package/skills/torusguard/payload/rules/container/TG-CONT-002-docker-socket-mount.md +47 -47
- package/skills/torusguard/payload/rules/container/TG-CONT-003-privileged-container-mode.md +53 -53
- package/skills/torusguard/payload/rules/container/TG-CONT-004-build-arg-secret-exposure.md +43 -43
- package/skills/torusguard/payload/rules/git/TG-GIT-001-historical-secret-in-git-commit.md +44 -44
- package/skills/torusguard/payload/rules/git/TG-GIT-002-plaintext-credentials-in-git-config.md +41 -41
- package/skills/torusguard/payload/rules/git/TG-GIT-003-sensitive-tracked-file-gitignore-breach.md +40 -40
- package/skills/torusguard/payload/rules/rag/TG-RAG-001-untrusted-rag-context-injection.md +72 -72
- package/skills/torusguard/payload/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md +51 -51
- package/skills/torusguard/payload/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md +51 -51
- package/skills/torusguard/payload/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md +46 -46
- package/skills/torusguard/payload/rules/redos/TG-REDOS-002-unbounded-nested-quantifier.md +43 -43
- package/skills/torusguard/payload/scripts/audit_runner.py +108 -10
- package/skills/torusguard/payload/scripts/finding_scorer.py +43 -13
- package/skills/torusguard/payload/skills/torusguard/SKILL.md +6 -2
- package/skills/torusguard/payload/skills/torusguard/bootstrap.py +3 -3
- package/skills/torusguard/payload/skills/torusguard-ai-guard/SKILL.md +95 -95
- package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +109 -84
- package/skills/torusguard/payload/skills/torusguard-container/SKILL.md +94 -94
- package/skills/torusguard/payload/skills/torusguard-git-mine/SKILL.md +92 -92
- package/skills/torusguard/payload/skills/torusguard-ocr-scan/SKILL.md +94 -94
- package/skills/torusguard/payload/skills/torusguard-redos/SKILL.md +91 -91
- package/skills/torusguard/payload/workflows/ai-guard.md +31 -31
- package/skills/torusguard/payload/workflows/audit.md +21 -17
- package/skills/torusguard/payload/workflows/container.md +29 -29
- package/skills/torusguard/payload/workflows/git-mine.md +25 -25
- package/skills/torusguard/payload/workflows/ocr-scan.md +25 -25
- package/skills/torusguard/payload/workflows/redos.md +27 -27
- package/skills/torusguard/payload/workflows/torusguard-audit.md +35 -55
- package/skills/torusguard/references/csharp-security.md +41 -41
- package/skills/torusguard/references/go-security.md +41 -41
- package/skills/torusguard/references/java-security.md +40 -40
- package/skills/torusguard/references/polyglot-security-matrix.md +25 -25
- package/skills/torusguard/references/rust-security.md +40 -40
- package/skills/torusguard-audit/SKILL.md +107 -83
package/skills/torusguard/payload/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md
CHANGED
|
@@ -1,51 +1,51 @@
|
|
|
1
|
-
# TG-RAG-002: Autonomous LLM Tool Unsandboxed Call
|
|
2
|
-
|
|
3
|
-
## Severity
|
|
4
|
-
Critical. Executing system shell commands, raw SQL, or filesystem modifications based directly on model tool call outputs without schema validation or sandboxing allows remote code execution (RCE).
|
|
5
|
-
|
|
6
|
-
## Applies To
|
|
7
|
-
- LLM Function Calling, Agent Tool Calling, ReAct loops, Model Context Protocol (MCP) servers
|
|
8
|
-
- Python, Node.js, Go
|
|
9
|
-
|
|
10
|
-
## Why It Matters
|
|
11
|
-
When an AI agent calls external tools (e.g. `execute_code`, `query_database`, `run_bash`), the arguments originate from stochastic model generation. If the model was prompted or tricked via prompt injection to emit `rm -rf /` or `DROP TABLE users;`, executing those arguments without strict allowlists, parameterization, or sandboxing destroys data or compromises the server.
|
|
12
|
-
|
|
13
|
-
## What TorusGuard Looks For
|
|
14
|
-
1. Passing tool call arguments directly to `os.system`, `subprocess.run(..., shell=True)`, or `child_process.exec`.
|
|
15
|
-
2. Evaluating raw SQL emitted by LLM tool calls without parameterization.
|
|
16
|
-
3. Lack of human-in-the-loop confirmation on destructive tool invocations.
|
|
17
|
-
|
|
18
|
-
## Unsafe Example
|
|
19
|
-
```python
|
|
20
|
-
# UNSAFE: Unsandboxed execution of model tool call
|
|
21
|
-
def handle_tool_call(tool_call):
|
|
22
|
-
if tool_call.function.name == "run_command":
|
|
23
|
-
args = json.loads(tool_call.function.arguments)
|
|
24
|
-
# Directly executes arbitrary shell command generated by LLM!
|
|
25
|
-
return subprocess.check_output(args["cmd"], shell=True)
|
|
26
|
-
```
|
|
27
|
-
|
|
28
|
-
## Safe Example
|
|
29
|
-
```python
|
|
30
|
-
# SAFE: Strict schema validation, command allowlisting, and no shell=True
|
|
31
|
-
ALLOWED_COMMANDS = {"git status", "git diff", "npm test"}
|
|
32
|
-
|
|
33
|
-
def handle_tool_call(tool_call):
|
|
34
|
-
if tool_call.function.name == "run_command":
|
|
35
|
-
args = json.loads(tool_call.function.arguments)
|
|
36
|
-
cmd = args.get("cmd", "").strip()
|
|
37
|
-
|
|
38
|
-
if cmd not in ALLOWED_COMMANDS:
|
|
39
|
-
raise PermissionError(f"Command not permitted: {cmd}")
|
|
40
|
-
|
|
41
|
-
return subprocess.check_output(cmd.split(), shell=False)
|
|
42
|
-
```
|
|
43
|
-
|
|
44
|
-
## Remediation
|
|
45
|
-
1. Enforce strict Pydantic/Zod schemas on all tool arguments.
|
|
46
|
-
2. Ban `shell=True` when invoking sub-processes from AI tool calls.
|
|
47
|
-
3. Require explicit human confirmation (Human Gate) for state-altering, file-writing, or network operations.
|
|
48
|
-
|
|
49
|
-
## Related Rules
|
|
50
|
-
- `TG-AGENT-002`: Unsafe Tool Dispatch
|
|
51
|
-
- `TG-INPUT-003`: Unsafe Code Execution
|
|
1
|
+
# TG-RAG-002: Autonomous LLM Tool Unsandboxed Call
|
|
2
|
+
|
|
3
|
+
## Severity
|
|
4
|
+
Critical. Executing system shell commands, raw SQL, or filesystem modifications based directly on model tool call outputs without schema validation or sandboxing allows remote code execution (RCE).
|
|
5
|
+
|
|
6
|
+
## Applies To
|
|
7
|
+
- LLM Function Calling, Agent Tool Calling, ReAct loops, Model Context Protocol (MCP) servers
|
|
8
|
+
- Python, Node.js, Go
|
|
9
|
+
|
|
10
|
+
## Why It Matters
|
|
11
|
+
When an AI agent calls external tools (e.g. `execute_code`, `query_database`, `run_bash`), the arguments originate from stochastic model generation. If the model was prompted or tricked via prompt injection to emit `rm -rf /` or `DROP TABLE users;`, executing those arguments without strict allowlists, parameterization, or sandboxing destroys data or compromises the server.
|
|
12
|
+
|
|
13
|
+
## What TorusGuard Looks For
|
|
14
|
+
1. Passing tool call arguments directly to `os.system`, `subprocess.run(..., shell=True)`, or `child_process.exec`.
|
|
15
|
+
2. Evaluating raw SQL emitted by LLM tool calls without parameterization.
|
|
16
|
+
3. Lack of human-in-the-loop confirmation on destructive tool invocations.
|
|
17
|
+
|
|
18
|
+
## Unsafe Example
|
|
19
|
+
```python
|
|
20
|
+
# UNSAFE: Unsandboxed execution of model tool call
|
|
21
|
+
def handle_tool_call(tool_call):
|
|
22
|
+
if tool_call.function.name == "run_command":
|
|
23
|
+
args = json.loads(tool_call.function.arguments)
|
|
24
|
+
# Directly executes arbitrary shell command generated by LLM!
|
|
25
|
+
return subprocess.check_output(args["cmd"], shell=True)
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Safe Example
|
|
29
|
+
```python
|
|
30
|
+
# SAFE: Strict schema validation, command allowlisting, and no shell=True
|
|
31
|
+
ALLOWED_COMMANDS = {"git status", "git diff", "npm test"}
|
|
32
|
+
|
|
33
|
+
def handle_tool_call(tool_call):
|
|
34
|
+
if tool_call.function.name == "run_command":
|
|
35
|
+
args = json.loads(tool_call.function.arguments)
|
|
36
|
+
cmd = args.get("cmd", "").strip()
|
|
37
|
+
|
|
38
|
+
if cmd not in ALLOWED_COMMANDS:
|
|
39
|
+
raise PermissionError(f"Command not permitted: {cmd}")
|
|
40
|
+
|
|
41
|
+
return subprocess.check_output(cmd.split(), shell=False)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Remediation
|
|
45
|
+
1. Enforce strict Pydantic/Zod schemas on all tool arguments.
|
|
46
|
+
2. Ban `shell=True` when invoking sub-processes from AI tool calls.
|
|
47
|
+
3. Require explicit human confirmation (Human Gate) for state-altering, file-writing, or network operations.
|
|
48
|
+
|
|
49
|
+
## Related Rules
|
|
50
|
+
- `TG-AGENT-002`: Unsafe Tool Dispatch
|
|
51
|
+
- `TG-INPUT-003`: Unsafe Code Execution
|
package/skills/torusguard/payload/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md
CHANGED
|
@@ -1,51 +1,51 @@
|
|
|
1
|
-
# TG-RAG-003: Unpartitioned Vector Database Tenant Lookup
|
|
2
|
-
|
|
3
|
-
## Severity
|
|
4
|
-
High. Executing similarity searches across vector databases without multi-tenant metadata filters leaks private organization or user documents across tenant boundaries.
|
|
5
|
-
|
|
6
|
-
## Applies To
|
|
7
|
-
- Vector Databases: Pinecone, Qdrant, Chroma, Weaviate, Milvus, pgvector
|
|
8
|
-
- RAG applications with multi-tenant users or workspaces
|
|
9
|
-
|
|
10
|
-
## Why It Matters
|
|
11
|
-
Vector embeddings from different tenants exist in the same high-dimensional embedding space. If an embedding lookup only searches by cosine similarity without an explicit `filter={"tenant_id": user.tenant_id}` or namespace partition, queries from User A will return private embeddings, contracts, or records belonging to User B.
|
|
12
|
-
|
|
13
|
-
## What TorusGuard Looks For
|
|
14
|
-
1. Vector similarity searches lacking metadata filter arguments (e.g. `index.query(vector=..., top_k=5)` with no `filter`).
|
|
15
|
-
2. Missing tenant partitioning in vector retrieval endpoints.
|
|
16
|
-
|
|
17
|
-
## Unsafe Example
|
|
18
|
-
```python
|
|
19
|
-
# UNSAFE: Vector similarity search across all tenants
|
|
20
|
-
def search_knowledge_base(user: User, query_vector: list[float]):
|
|
21
|
-
results = pinecone_index.query(
|
|
22
|
-
vector=query_vector,
|
|
23
|
-
top_k=5,
|
|
24
|
-
include_metadata=True
|
|
25
|
-
# MISSING tenant filter!
|
|
26
|
-
)
|
|
27
|
-
return results
|
|
28
|
-
```
|
|
29
|
-
|
|
30
|
-
## Safe Example
|
|
31
|
-
```python
|
|
32
|
-
# SAFE: Mandatory tenant scoping in metadata filter
|
|
33
|
-
def search_knowledge_base(user: User, query_vector: list[float]):
|
|
34
|
-
results = pinecone_index.query(
|
|
35
|
-
vector=query_vector,
|
|
36
|
-
top_k=5,
|
|
37
|
-
include_metadata=True,
|
|
38
|
-
filter={
|
|
39
|
-
"tenant_id": {"$eq": user.tenant_id}
|
|
40
|
-
}
|
|
41
|
-
)
|
|
42
|
-
return results
|
|
43
|
-
```
|
|
44
|
-
|
|
45
|
-
## Remediation
|
|
46
|
-
1. Always scope vector similarity queries by tenant ID in the metadata filter.
|
|
47
|
-
2. In pgvector, enforce row-level security (RLS) or explicit `WHERE tenant_id = :tenant_id` clauses on embedding queries.
|
|
48
|
-
|
|
49
|
-
## Related Rules
|
|
50
|
-
- `TG-DB-001`: Missing Tenant Query Isolation
|
|
51
|
-
- `TG-RAG-001`: Untrusted RAG Context Injection
|
|
1
|
+
# TG-RAG-003: Unpartitioned Vector Database Tenant Lookup
|
|
2
|
+
|
|
3
|
+
## Severity
|
|
4
|
+
High. Executing similarity searches across vector databases without multi-tenant metadata filters leaks private organization or user documents across tenant boundaries.
|
|
5
|
+
|
|
6
|
+
## Applies To
|
|
7
|
+
- Vector Databases: Pinecone, Qdrant, Chroma, Weaviate, Milvus, pgvector
|
|
8
|
+
- RAG applications with multi-tenant users or workspaces
|
|
9
|
+
|
|
10
|
+
## Why It Matters
|
|
11
|
+
Vector embeddings from different tenants exist in the same high-dimensional embedding space. If an embedding lookup only searches by cosine similarity without an explicit `filter={"tenant_id": user.tenant_id}` or namespace partition, queries from User A will return private embeddings, contracts, or records belonging to User B.
|
|
12
|
+
|
|
13
|
+
## What TorusGuard Looks For
|
|
14
|
+
1. Vector similarity searches lacking metadata filter arguments (e.g. `index.query(vector=..., top_k=5)` with no `filter`).
|
|
15
|
+
2. Missing tenant partitioning in vector retrieval endpoints.
|
|
16
|
+
|
|
17
|
+
## Unsafe Example
|
|
18
|
+
```python
|
|
19
|
+
# UNSAFE: Vector similarity search across all tenants
|
|
20
|
+
def search_knowledge_base(user: User, query_vector: list[float]):
|
|
21
|
+
results = pinecone_index.query(
|
|
22
|
+
vector=query_vector,
|
|
23
|
+
top_k=5,
|
|
24
|
+
include_metadata=True
|
|
25
|
+
# MISSING tenant filter!
|
|
26
|
+
)
|
|
27
|
+
return results
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Safe Example
|
|
31
|
+
```python
|
|
32
|
+
# SAFE: Mandatory tenant scoping in metadata filter
|
|
33
|
+
def search_knowledge_base(user: User, query_vector: list[float]):
|
|
34
|
+
results = pinecone_index.query(
|
|
35
|
+
vector=query_vector,
|
|
36
|
+
top_k=5,
|
|
37
|
+
include_metadata=True,
|
|
38
|
+
filter={
|
|
39
|
+
"tenant_id": {"$eq": user.tenant_id}
|
|
40
|
+
}
|
|
41
|
+
)
|
|
42
|
+
return results
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Remediation
|
|
46
|
+
1. Always scope vector similarity queries by tenant ID in the metadata filter.
|
|
47
|
+
2. In pgvector, enforce row-level security (RLS) or explicit `WHERE tenant_id = :tenant_id` clauses on embedding queries.
|
|
48
|
+
|
|
49
|
+
## Related Rules
|
|
50
|
+
- `TG-DB-001`: Missing Tenant Query Isolation
|
|
51
|
+
- `TG-RAG-001`: Untrusted RAG Context Injection
|
package/skills/torusguard/payload/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md
CHANGED
|
@@ -1,46 +1,46 @@
|
|
|
1
|
-
# TG-REDOS-001: Catastrophic Exponential Backtracking in Regular Expression
|
|
2
|
-
|
|
3
|
-
## Severity
|
|
4
|
-
High. Regular expressions with catastrophic backtracking trigger exponential time complexity ($O(2^n)$) when evaluating non-matching input strings, freezing CPU cores and causing Denial of Service.
|
|
5
|
-
|
|
6
|
-
## Applies To
|
|
7
|
-
- JavaScript / TypeScript (`RegExp`, `pattern.test()`), Python (`re.match`, `re.search`), Go, Java, Ruby
|
|
8
|
-
- Input validation patterns, email validators, URL extractors
|
|
9
|
-
|
|
10
|
-
## Why It Matters
|
|
11
|
-
Traditional regex engines using NFA backtracking (e.g. JavaScript V8, Python `re`, Java `java.util.regex`, PCRE) explore all possible match paths on failure. When a pattern contains overlapping nested repetitions like `(a+)+$`, an input of 30 characters like `aaaaaaaaaaaaaaaaaaaaaaaaaaaaab` can require over 1 billion comparison operations, freezing the Node.js event loop or Python GIL.
|
|
12
|
-
|
|
13
|
-
## What TorusGuard Looks For
|
|
14
|
-
1. Nested repetitions: `([a-zA-Z0-9]+)+`, `(a+)+`, `(\d+)*`.
|
|
15
|
-
2. Overlapping alternations with outer quantifiers: `(a|aa)+`, `(x|x)*`.
|
|
16
|
-
3. Greedy repetition with overlapping prefix and suffix.
|
|
17
|
-
|
|
18
|
-
## Unsafe Example
|
|
19
|
-
```javascript
|
|
20
|
-
// UNSAFE: Catastrophic backtracking on non-matching strings
|
|
21
|
-
const EMAIL_REGEX = /^([a-zA-Z0-9_\.\-])+@(([a-zA-Z0-9\-])+\.)+([a-zA-Z0-9]{2,4})+$/;
|
|
22
|
-
|
|
23
|
-
// Freezes server:
|
|
24
|
-
EMAIL_REGEX.test("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa!");
|
|
25
|
-
```
|
|
26
|
-
|
|
27
|
-
## Safe Example
|
|
28
|
-
```javascript
|
|
29
|
-
// SAFE: Linear time validation using atomic checks, character class bounds, or validator libraries
|
|
30
|
-
const validator = require('validator');
|
|
31
|
-
if (!validator.isEmail(input)) {
|
|
32
|
-
throw new Error("Invalid email");
|
|
33
|
-
}
|
|
34
|
-
|
|
35
|
-
// Or constrained regex without nested quantifiers:
|
|
36
|
-
const SAFE_EMAIL = /^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$/;
|
|
37
|
-
```
|
|
38
|
-
|
|
39
|
-
## Remediation
|
|
40
|
-
1. Eliminate nested quantifiers (`(x+)+` -> `x+`).
|
|
41
|
-
2. Disallow overlapping tokens in alternations.
|
|
42
|
-
3. In Node.js, wrap untrusted input validation with `safe-regex` or strict input length bounds (e.g. `if (input.length > 256) return false;`).
|
|
43
|
-
|
|
44
|
-
## Related Rules
|
|
45
|
-
- `TG-REDOS-002`: Unbounded Nested Quantifier
|
|
46
|
-
- `TG-RATE-003`: Unbounded Resource Consumption
|
|
1
|
+
# TG-REDOS-001: Catastrophic Exponential Backtracking in Regular Expression
|
|
2
|
+
|
|
3
|
+
## Severity
|
|
4
|
+
High. Regular expressions with catastrophic backtracking trigger exponential time complexity ($O(2^n)$) when evaluating non-matching input strings, freezing CPU cores and causing Denial of Service.
|
|
5
|
+
|
|
6
|
+
## Applies To
|
|
7
|
+
- JavaScript / TypeScript (`RegExp`, `pattern.test()`), Python (`re.match`, `re.search`), Go, Java, Ruby
|
|
8
|
+
- Input validation patterns, email validators, URL extractors
|
|
9
|
+
|
|
10
|
+
## Why It Matters
|
|
11
|
+
Traditional regex engines using NFA backtracking (e.g. JavaScript V8, Python `re`, Java `java.util.regex`, PCRE) explore all possible match paths on failure. When a pattern contains overlapping nested repetitions like `(a+)+$`, an input of 30 characters like `aaaaaaaaaaaaaaaaaaaaaaaaaaaaab` can require over 1 billion comparison operations, freezing the Node.js event loop or Python GIL.
|
|
12
|
+
|
|
13
|
+
## What TorusGuard Looks For
|
|
14
|
+
1. Nested repetitions: `([a-zA-Z0-9]+)+`, `(a+)+`, `(\d+)*`.
|
|
15
|
+
2. Overlapping alternations with outer quantifiers: `(a|aa)+`, `(x|x)*`.
|
|
16
|
+
3. Greedy repetition with overlapping prefix and suffix.
|
|
17
|
+
|
|
18
|
+
## Unsafe Example
|
|
19
|
+
```javascript
|
|
20
|
+
// UNSAFE: Catastrophic backtracking on non-matching strings
|
|
21
|
+
const EMAIL_REGEX = /^([a-zA-Z0-9_\.\-])+@(([a-zA-Z0-9\-])+\.)+([a-zA-Z0-9]{2,4})+$/;
|
|
22
|
+
|
|
23
|
+
// Freezes server:
|
|
24
|
+
EMAIL_REGEX.test("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa!");
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Safe Example
|
|
28
|
+
```javascript
|
|
29
|
+
// SAFE: Linear time validation using atomic checks, character class bounds, or validator libraries
|
|
30
|
+
const validator = require('validator');
|
|
31
|
+
if (!validator.isEmail(input)) {
|
|
32
|
+
throw new Error("Invalid email");
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// Or constrained regex without nested quantifiers:
|
|
36
|
+
const SAFE_EMAIL = /^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$/;
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Remediation
|
|
40
|
+
1. Eliminate nested quantifiers (`(x+)+` -> `x+`).
|
|
41
|
+
2. Disallow overlapping tokens in alternations.
|
|
42
|
+
3. In Node.js, wrap untrusted input validation with `safe-regex` or strict input length bounds (e.g. `if (input.length > 256) return false;`).
|
|
43
|
+
|
|
44
|
+
## Related Rules
|
|
45
|
+
- `TG-REDOS-002`: Unbounded Nested Quantifier
|
|
46
|
+
- `TG-RATE-003`: Unbounded Resource Consumption
|
|
@@ -1,43 +1,43 @@
|
|
|
1
|
-
# TG-REDOS-002: Unbounded Nested Quantifier in Input Validation
|
|
2
|
-
|
|
3
|
-
## Severity
|
|
4
|
-
Medium. Unbounded repeated capture groups without boundary anchors cause polynomial ($O(n^2)$) or exponential degradation on large payloads.
|
|
5
|
-
|
|
6
|
-
## Applies To
|
|
7
|
-
- Input validation filters, route path matchers, sanitizer regexes
|
|
8
|
-
- Polyglot web backends and client-side form validators
|
|
9
|
-
|
|
10
|
-
## Why It Matters
|
|
11
|
-
When regexes use repeated capture groups like `(\w+\s*)+` without anchoring, trailing spaces or punctuation force the engine into deep recursive state branches. While not always pure exponential, large payloads (e.g. 50KB JSON strings) will peg CPU at 100% for minutes.
|
|
12
|
-
|
|
13
|
-
## What TorusGuard Looks For
|
|
14
|
-
1. Nested groups where both inner and outer components have greedy repetition (`+` or `*`).
|
|
15
|
-
2. Regexes evaluated on user-supplied strings without a preceding string length check.
|
|
16
|
-
|
|
17
|
-
## Unsafe Example
|
|
18
|
-
```python
|
|
19
|
-
# UNSAFE: Unbounded nested quantifier on user input
|
|
20
|
-
import re
|
|
21
|
-
|
|
22
|
-
TAG_REGEX = re.compile(r"^(<[a-z]+(\s+[a-z]+=[^>]+)*>)+$")
|
|
23
|
-
match = TAG_REGEX.match(user_payload)
|
|
24
|
-
```
|
|
25
|
-
|
|
26
|
-
## Safe Example
|
|
27
|
-
```python
|
|
28
|
-
# SAFE: Bounded input length check + non-nested linear pattern
|
|
29
|
-
import re
|
|
30
|
-
|
|
31
|
-
if len(user_payload) > 512:
|
|
32
|
-
return False
|
|
33
|
-
|
|
34
|
-
# Use a dedicated HTML parser (BeautifulSoup / html5lib) instead of regex
|
|
35
|
-
```
|
|
36
|
-
|
|
37
|
-
## Remediation
|
|
38
|
-
1. Bound user input length *before* regex execution.
|
|
39
|
-
2. Replace complex nested regexes with dedicated, parser-based validation libraries (e.g., standard parsers for HTML, URLs, and emails).
|
|
40
|
-
|
|
41
|
-
## Related Rules
|
|
42
|
-
- `TG-REDOS-001`: Catastrophic Exponential Backtracking
|
|
43
|
-
- `TG-INPUT-001`: Missing Server Validation
|
|
1
|
+
# TG-REDOS-002: Unbounded Nested Quantifier in Input Validation
|
|
2
|
+
|
|
3
|
+
## Severity
|
|
4
|
+
Medium. Unbounded repeated capture groups without boundary anchors cause polynomial ($O(n^2)$) or exponential degradation on large payloads.
|
|
5
|
+
|
|
6
|
+
## Applies To
|
|
7
|
+
- Input validation filters, route path matchers, sanitizer regexes
|
|
8
|
+
- Polyglot web backends and client-side form validators
|
|
9
|
+
|
|
10
|
+
## Why It Matters
|
|
11
|
+
When regexes use repeated capture groups like `(\w+\s*)+` without anchoring, trailing spaces or punctuation force the engine into deep recursive state branches. While not always pure exponential, large payloads (e.g. 50KB JSON strings) will peg CPU at 100% for minutes.
|
|
12
|
+
|
|
13
|
+
## What TorusGuard Looks For
|
|
14
|
+
1. Nested groups where both inner and outer components have greedy repetition (`+` or `*`).
|
|
15
|
+
2. Regexes evaluated on user-supplied strings without a preceding string length check.
|
|
16
|
+
|
|
17
|
+
## Unsafe Example
|
|
18
|
+
```python
|
|
19
|
+
# UNSAFE: Unbounded nested quantifier on user input
|
|
20
|
+
import re
|
|
21
|
+
|
|
22
|
+
TAG_REGEX = re.compile(r"^(<[a-z]+(\s+[a-z]+=[^>]+)*>)+$")
|
|
23
|
+
match = TAG_REGEX.match(user_payload)
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
## Safe Example
|
|
27
|
+
```python
|
|
28
|
+
# SAFE: Bounded input length check + non-nested linear pattern
|
|
29
|
+
import re
|
|
30
|
+
|
|
31
|
+
if len(user_payload) > 512:
|
|
32
|
+
return False
|
|
33
|
+
|
|
34
|
+
# Use a dedicated HTML parser (BeautifulSoup / html5lib) instead of regex
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Remediation
|
|
38
|
+
1. Bound user input length *before* regex execution.
|
|
39
|
+
2. Replace complex nested regexes with dedicated, parser-based validation libraries (e.g., standard parsers for HTML, URLs, and emails).
|
|
40
|
+
|
|
41
|
+
## Related Rules
|
|
42
|
+
- `TG-REDOS-001`: Catastrophic Exponential Backtracking
|
|
43
|
+
- `TG-INPUT-001`: Missing Server Validation
|
|
@@ -42,8 +42,16 @@ def get_ist_now() -> datetime.datetime:
|
|
|
42
42
|
|
|
43
43
|
# ─── UI Formatter Bridge ──────────────────────────────────────────────────────
|
|
44
44
|
scripts_dir = Path(__file__).resolve().parent
|
|
45
|
+
tg_root = Path(__file__).resolve().parent.parent
|
|
46
|
+
project_root = Path(__file__).resolve().parent.parent.parent
|
|
45
47
|
if str(scripts_dir) not in sys.path:
|
|
46
48
|
sys.path.insert(0, str(scripts_dir))
|
|
49
|
+
if str(tg_root) not in sys.path:
|
|
50
|
+
sys.path.insert(0, str(tg_root))
|
|
51
|
+
if str(project_root) not in sys.path:
|
|
52
|
+
sys.path.insert(0, str(project_root))
|
|
53
|
+
|
|
54
|
+
|
|
47
55
|
|
|
48
56
|
try:
|
|
49
57
|
import term_ui as tui
|
|
@@ -1240,9 +1248,10 @@ def scan_file(file_path: Path, target_root: Path) -> List[Dict[str, Any]]:
|
|
|
1240
1248
|
return findings
|
|
1241
1249
|
|
|
1242
1250
|
|
|
1243
|
-
def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path) -> Tuple[List[Dict[str, Any]], Dict[str, List[Dict[str, Any]]]]:
|
|
1251
|
+
def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path, taint_paths: Optional[List[Any]] = None) -> Tuple[List[Dict[str, Any]], Dict[str, List[Dict[str, Any]]]]:
|
|
1244
1252
|
"""
|
|
1245
|
-
Score each finding using finding_scorer.py
|
|
1253
|
+
Score each finding using finding_scorer.py, persistent memory patterns,
|
|
1254
|
+
and taint dataflow analysis.
|
|
1246
1255
|
Returns (scored_findings, clusters_map).
|
|
1247
1256
|
"""
|
|
1248
1257
|
try:
|
|
@@ -1253,6 +1262,15 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
|
|
|
1253
1262
|
scored = []
|
|
1254
1263
|
clusters: Dict[str, List[Dict[str, Any]]] = {}
|
|
1255
1264
|
|
|
1265
|
+
# Map taint paths by (file_path, line_number)
|
|
1266
|
+
taint_by_loc: Dict[Tuple[str, int], Any] = {}
|
|
1267
|
+
if taint_paths:
|
|
1268
|
+
for tp in taint_paths:
|
|
1269
|
+
sink_node = getattr(tp, "sink", None)
|
|
1270
|
+
if sink_node:
|
|
1271
|
+
norm_p = getattr(sink_node, "file_path", "").replace("\\", "/")
|
|
1272
|
+
taint_by_loc[(norm_p, getattr(sink_node, "line_number", 0))] = tp
|
|
1273
|
+
|
|
1256
1274
|
# Phase 1d: Pre-compute cross-file corroboration bonus
|
|
1257
1275
|
# If multiple rule families flag the same file, each finding gets +5 confidence
|
|
1258
1276
|
file_rule_families: Dict[str, set] = {}
|
|
@@ -1267,6 +1285,14 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
|
|
|
1267
1285
|
band = "High Confidence"
|
|
1268
1286
|
factors = {}
|
|
1269
1287
|
|
|
1288
|
+
# Check for correlated taint path
|
|
1289
|
+
matching_tp = taint_by_loc.get((f["file_path"], f["line_number"]))
|
|
1290
|
+
is_taint_confirmed = matching_tp is not None
|
|
1291
|
+
taint_depth = getattr(matching_tp, "depth", None) if matching_tp else None
|
|
1292
|
+
is_sanitized = getattr(matching_tp, "is_sanitized", False) if matching_tp else False
|
|
1293
|
+
if matching_tp and hasattr(matching_tp, "to_dict"):
|
|
1294
|
+
f["taint_path"] = matching_tp.to_dict()
|
|
1295
|
+
|
|
1270
1296
|
if finding_scorer:
|
|
1271
1297
|
try:
|
|
1272
1298
|
# Phase 1d: Dynamic evidence quality based on rule precision
|
|
@@ -1291,7 +1317,11 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
|
|
|
1291
1317
|
manual_review_status=mr,
|
|
1292
1318
|
rule_id=f["rule_id"],
|
|
1293
1319
|
file_path=f["file_path"],
|
|
1294
|
-
root_dir=target_root
|
|
1320
|
+
root_dir=target_root,
|
|
1321
|
+
taint_path_confirmed=is_taint_confirmed,
|
|
1322
|
+
taint_depth=taint_depth,
|
|
1323
|
+
sanitizer_present=is_sanitized,
|
|
1324
|
+
rule_severity=f.get("severity", "High")
|
|
1295
1325
|
)
|
|
1296
1326
|
score = s
|
|
1297
1327
|
band = b
|
|
@@ -1315,6 +1345,7 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
|
|
|
1315
1345
|
return scored, clusters
|
|
1316
1346
|
|
|
1317
1347
|
|
|
1348
|
+
|
|
1318
1349
|
def emit_run_artifacts(run_folder: Path, scored_findings: List[Dict[str, Any]], clusters: Dict[str, List[Dict[str, Any]]], target_root: Path) -> None:
|
|
1319
1350
|
"""Generate findings.json, findings.md, and summary.md into run folder."""
|
|
1320
1351
|
now_ist = get_ist_now()
|
|
@@ -1501,7 +1532,14 @@ def run_watch_mode(target_root: Path, severity_floor: str = "medium", json_outpu
|
|
|
1501
1532
|
print(f"\n {YELLOW}🛑 Watch mode stopped.{RESET}\n")
|
|
1502
1533
|
|
|
1503
1534
|
|
|
1504
|
-
def execute_audit(
|
|
1535
|
+
def execute_audit(
|
|
1536
|
+
target_root: Path,
|
|
1537
|
+
severity_floor: str = "medium",
|
|
1538
|
+
json_output: bool = False,
|
|
1539
|
+
include_tests: bool = False,
|
|
1540
|
+
incremental: bool = False,
|
|
1541
|
+
use_taint: bool = True
|
|
1542
|
+
) -> Dict[str, Any]:
|
|
1505
1543
|
"""Execute the full TorusGuard static security audit."""
|
|
1506
1544
|
start_time = time.perf_counter()
|
|
1507
1545
|
target_root = target_root.resolve()
|
|
@@ -1522,14 +1560,64 @@ def execute_audit(target_root: Path, severity_floor: str = "medium", json_output
|
|
|
1522
1560
|
except Exception:
|
|
1523
1561
|
pass
|
|
1524
1562
|
|
|
1525
|
-
# 2. Collect files
|
|
1563
|
+
# 2. Collect files to scan
|
|
1526
1564
|
files = find_files_to_scan(target_root, include_tests=include_tests)
|
|
1527
1565
|
all_findings = []
|
|
1528
|
-
for f in files:
|
|
1529
|
-
all_findings.extend(scan_file(f, target_root))
|
|
1530
1566
|
|
|
1531
|
-
|
|
1532
|
-
|
|
1567
|
+
inc_scanner = None
|
|
1568
|
+
files_to_scan = files
|
|
1569
|
+
unchanged_files = []
|
|
1570
|
+
|
|
1571
|
+
if incremental:
|
|
1572
|
+
try:
|
|
1573
|
+
from core.incremental import IncrementalScanner
|
|
1574
|
+
inc_scanner = IncrementalScanner(target_root)
|
|
1575
|
+
files_to_scan, unchanged_files = inc_scanner.get_changed_files(files)
|
|
1576
|
+
# Rehydrate findings for unchanged files
|
|
1577
|
+
for uf in unchanged_files:
|
|
1578
|
+
all_findings.extend(inc_scanner.get_cached_findings(uf))
|
|
1579
|
+
except Exception:
|
|
1580
|
+
files_to_scan = files
|
|
1581
|
+
unchanged_files = []
|
|
1582
|
+
|
|
1583
|
+
# Parallel or sequential scan on files_to_scan
|
|
1584
|
+
new_findings = []
|
|
1585
|
+
try:
|
|
1586
|
+
from core.parallel import ParallelAuditExecutor
|
|
1587
|
+
executor = ParallelAuditExecutor()
|
|
1588
|
+
new_findings = executor.scan_files_parallel(files_to_scan, lambda f: scan_file(f, target_root))
|
|
1589
|
+
except Exception:
|
|
1590
|
+
for f in files_to_scan:
|
|
1591
|
+
new_findings.extend(scan_file(f, target_root))
|
|
1592
|
+
|
|
1593
|
+
all_findings.extend(new_findings)
|
|
1594
|
+
|
|
1595
|
+
# Update cache if incremental scanner is active
|
|
1596
|
+
if inc_scanner:
|
|
1597
|
+
# Group new findings by file
|
|
1598
|
+
file_to_findings: Dict[Path, List[Dict[str, Any]]] = {f: [] for f in files_to_scan}
|
|
1599
|
+
for nf in new_findings:
|
|
1600
|
+
raw_fp = nf.get("file_path", "")
|
|
1601
|
+
target_f = target_root / raw_fp
|
|
1602
|
+
if target_f in file_to_findings:
|
|
1603
|
+
file_to_findings[target_f].append(nf)
|
|
1604
|
+
for target_f, f_list in file_to_findings.items():
|
|
1605
|
+
inc_scanner.update_file_cache(target_f, f_list)
|
|
1606
|
+
inc_scanner.save_cache()
|
|
1607
|
+
|
|
1608
|
+
# 2.5 Taint Dataflow Analysis (if enabled)
|
|
1609
|
+
taint_paths = []
|
|
1610
|
+
if use_taint:
|
|
1611
|
+
try:
|
|
1612
|
+
from core.cross_file_taint import CrossFileTaintAnalyzer
|
|
1613
|
+
analyzer = CrossFileTaintAnalyzer(target_root)
|
|
1614
|
+
# Analyze target files (capped to 200 files for high responsiveness)
|
|
1615
|
+
taint_paths = analyzer.analyze_project(files[:200])
|
|
1616
|
+
except Exception:
|
|
1617
|
+
taint_paths = []
|
|
1618
|
+
|
|
1619
|
+
# 3. Score & Cluster with Taint Evidence
|
|
1620
|
+
scored_findings, clusters = score_and_cluster_findings(all_findings, target_root, taint_paths=taint_paths)
|
|
1533
1621
|
|
|
1534
1622
|
# 4. Allocate run folder
|
|
1535
1623
|
runs_dir = target_root / ".torusguard" / "runs"
|
|
@@ -1595,6 +1683,8 @@ def main():
|
|
|
1595
1683
|
parser.add_argument("--scope", "-s", help="Alternative path to target project")
|
|
1596
1684
|
parser.add_argument("--severity", choices=["critical", "high", "medium", "low"], default="medium", help="Severity floor")
|
|
1597
1685
|
parser.add_argument("--watch", "-w", action="store_true", help="Continuous watch mode: re-scan on file save")
|
|
1686
|
+
parser.add_argument("--incremental", "-i", action="store_true", help="Incremental mode: only scan modified files")
|
|
1687
|
+
parser.add_argument("--no-taint", action="store_true", help="Disable taint-aware dataflow analysis")
|
|
1598
1688
|
parser.add_argument("--sarif", action="store_true", help="Automatically export findings to OASIS SARIF v2.1.0")
|
|
1599
1689
|
parser.add_argument("--sarif-out", help="Output file path for SARIF export")
|
|
1600
1690
|
parser.add_argument("--json", action="store_true", help="Output raw JSON")
|
|
@@ -1607,10 +1697,18 @@ def main():
|
|
|
1607
1697
|
run_watch_mode(target, severity_floor=args.severity, json_output=args.json, include_tests=args.include_tests, sarif=args.sarif, sarif_out=args.sarif_out)
|
|
1608
1698
|
sys.exit(0)
|
|
1609
1699
|
|
|
1610
|
-
res = execute_audit(
|
|
1700
|
+
res = execute_audit(
|
|
1701
|
+
target,
|
|
1702
|
+
severity_floor=args.severity,
|
|
1703
|
+
json_output=args.json,
|
|
1704
|
+
include_tests=args.include_tests,
|
|
1705
|
+
incremental=args.incremental,
|
|
1706
|
+
use_taint=(not args.no_taint)
|
|
1707
|
+
)
|
|
1611
1708
|
if args.sarif:
|
|
1612
1709
|
export_sarif(target, res.get("run_folder"), args.sarif_out)
|
|
1613
1710
|
|
|
1711
|
+
|
|
1614
1712
|
sys.exit(0 if res["critical_count"] == 0 else 1)
|
|
1615
1713
|
|
|
1616
1714
|
|