torusguard 2.1.0 → 2.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/.torusguard/.manifest.json +47 -5
  2. package/.torusguard/core/__init__.py +146 -0
  3. package/.torusguard/core/agent_roles.py +104 -0
  4. package/.torusguard/core/ast_walker.py +283 -0
  5. package/.torusguard/core/authorization.py +218 -0
  6. package/.torusguard/core/browser_verifier.py +128 -0
  7. package/.torusguard/core/bundle.py +141 -0
  8. package/.torusguard/core/call_graph.py +184 -0
  9. package/.torusguard/core/clustering.py +275 -0
  10. package/.torusguard/core/confidence.py +120 -0
  11. package/.torusguard/core/cross_file_taint.py +101 -0
  12. package/.torusguard/core/exploit_checker.py +317 -0
  13. package/.torusguard/core/formatter.py +351 -0
  14. package/.torusguard/core/governance.py +210 -0
  15. package/.torusguard/core/identity.py +104 -0
  16. package/.torusguard/core/import_resolver.py +91 -0
  17. package/.torusguard/core/incremental.py +102 -0
  18. package/.torusguard/core/lifecycle.py +137 -0
  19. package/.torusguard/core/models.py +425 -0
  20. package/.torusguard/core/parallel.py +56 -0
  21. package/.torusguard/core/parser.py +202 -0
  22. package/.torusguard/core/rechecker.py +107 -0
  23. package/.torusguard/core/replay_trace.py +178 -0
  24. package/.torusguard/core/rules_registry.py +131 -0
  25. package/.torusguard/core/run_folder.py +60 -0
  26. package/.torusguard/core/run_manager.py +163 -0
  27. package/.torusguard/core/runtime_evidence.py +175 -0
  28. package/.torusguard/core/runtime_validator.py +246 -0
  29. package/.torusguard/core/safety_gate.py +139 -0
  30. package/.torusguard/core/sarif.py +189 -0
  31. package/.torusguard/core/stack_profiler.py +184 -0
  32. package/.torusguard/core/symbol_table.py +91 -0
  33. package/.torusguard/core/taint.py +133 -0
  34. package/.torusguard/core/taint_graph.py +235 -0
  35. package/.torusguard/core/taint_rules.py +268 -0
  36. package/.torusguard/core/v070_reporter.py +102 -0
  37. package/.torusguard/core/v070_workflow.py +339 -0
  38. package/.torusguard/core/v6_reporter.py +180 -0
  39. package/.torusguard/core/v6_workflow.py +221 -0
  40. package/.torusguard/core/watcher.py +58 -0
  41. package/.torusguard/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
  42. package/.torusguard/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
  43. package/.torusguard/scripts/__pycache__/audit_runner.cpython-314.pyc +0 -0
  44. package/.torusguard/scripts/__pycache__/finding_scorer.cpython-314.pyc +0 -0
  45. package/.torusguard/scripts/__pycache__/rules_sync.cpython-314.pyc +0 -0
  46. package/.torusguard/scripts/audit_runner.py +108 -10
  47. package/.torusguard/scripts/finding_scorer.py +43 -13
  48. package/.torusguard/scripts/skill_profiler.py +26 -0
  49. package/.torusguard/skills/torusguard/SKILL.md +6 -2
  50. package/.torusguard/skills/torusguard-audit/SKILL.md +109 -84
  51. package/.torusguard/workflows/audit.md +21 -17
  52. package/README.md +19 -11
  53. package/package.json +7 -2
  54. package/skills/torusguard/SKILL.md +6 -2
  55. package/skills/torusguard/__pycache__/bootstrap.cpython-314.pyc +0 -0
  56. package/skills/torusguard/bootstrap.py +3 -3
  57. package/skills/torusguard/payload/.manifest.json +48 -7
  58. package/skills/torusguard/payload/core/__init__.py +146 -0
  59. package/skills/torusguard/payload/core/agent_roles.py +104 -0
  60. package/skills/torusguard/payload/core/ast_walker.py +283 -0
  61. package/skills/torusguard/payload/core/authorization.py +218 -0
  62. package/skills/torusguard/payload/core/browser_verifier.py +128 -0
  63. package/skills/torusguard/payload/core/bundle.py +141 -0
  64. package/skills/torusguard/payload/core/call_graph.py +184 -0
  65. package/skills/torusguard/payload/core/clustering.py +275 -0
  66. package/skills/torusguard/payload/core/confidence.py +120 -0
  67. package/skills/torusguard/payload/core/cross_file_taint.py +101 -0
  68. package/skills/torusguard/payload/core/exploit_checker.py +317 -0
  69. package/skills/torusguard/payload/core/formatter.py +351 -0
  70. package/skills/torusguard/payload/core/governance.py +210 -0
  71. package/skills/torusguard/payload/core/identity.py +104 -0
  72. package/skills/torusguard/payload/core/import_resolver.py +91 -0
  73. package/skills/torusguard/payload/core/incremental.py +102 -0
  74. package/skills/torusguard/payload/core/lifecycle.py +137 -0
  75. package/skills/torusguard/payload/core/models.py +425 -0
  76. package/skills/torusguard/payload/core/parallel.py +56 -0
  77. package/skills/torusguard/payload/core/parser.py +202 -0
  78. package/skills/torusguard/payload/core/rechecker.py +107 -0
  79. package/skills/torusguard/payload/core/replay_trace.py +178 -0
  80. package/skills/torusguard/payload/core/rules_registry.py +131 -0
  81. package/skills/torusguard/payload/core/run_folder.py +60 -0
  82. package/skills/torusguard/payload/core/run_manager.py +163 -0
  83. package/skills/torusguard/payload/core/runtime_evidence.py +175 -0
  84. package/skills/torusguard/payload/core/runtime_validator.py +246 -0
  85. package/skills/torusguard/payload/core/safety_gate.py +139 -0
  86. package/skills/torusguard/payload/core/sarif.py +189 -0
  87. package/skills/torusguard/payload/core/stack_profiler.py +184 -0
  88. package/skills/torusguard/payload/core/symbol_table.py +91 -0
  89. package/skills/torusguard/payload/core/taint.py +133 -0
  90. package/skills/torusguard/payload/core/taint_graph.py +235 -0
  91. package/skills/torusguard/payload/core/taint_rules.py +268 -0
  92. package/skills/torusguard/payload/core/v070_reporter.py +102 -0
  93. package/skills/torusguard/payload/core/v070_workflow.py +339 -0
  94. package/skills/torusguard/payload/core/v6_reporter.py +180 -0
  95. package/skills/torusguard/payload/core/v6_workflow.py +221 -0
  96. package/skills/torusguard/payload/core/watcher.py +58 -0
  97. package/skills/torusguard/payload/rules/TG-INPUT-007-unvalidated-redirect.md +53 -0
  98. package/skills/torusguard/payload/rules/TG-INPUT-008-insecure-deserialization.md +52 -0
  99. package/skills/torusguard/payload/rules/container/TG-CONT-001-root-user-execution.md +50 -50
  100. package/skills/torusguard/payload/rules/container/TG-CONT-002-docker-socket-mount.md +47 -47
  101. package/skills/torusguard/payload/rules/container/TG-CONT-003-privileged-container-mode.md +53 -53
  102. package/skills/torusguard/payload/rules/container/TG-CONT-004-build-arg-secret-exposure.md +43 -43
  103. package/skills/torusguard/payload/rules/git/TG-GIT-001-historical-secret-in-git-commit.md +44 -44
  104. package/skills/torusguard/payload/rules/git/TG-GIT-002-plaintext-credentials-in-git-config.md +41 -41
  105. package/skills/torusguard/payload/rules/git/TG-GIT-003-sensitive-tracked-file-gitignore-breach.md +40 -40
  106. package/skills/torusguard/payload/rules/rag/TG-RAG-001-untrusted-rag-context-injection.md +72 -72
  107. package/skills/torusguard/payload/rules/rag/TG-RAG-002-autonomous-llm-tool-unsandboxed-call.md +51 -51
  108. package/skills/torusguard/payload/rules/rag/TG-RAG-003-unpartitioned-vector-tenant-lookup.md +51 -51
  109. package/skills/torusguard/payload/rules/redos/TG-REDOS-001-catastrophic-exponential-backtracking.md +46 -46
  110. package/skills/torusguard/payload/rules/redos/TG-REDOS-002-unbounded-nested-quantifier.md +43 -43
  111. package/skills/torusguard/payload/scripts/audit_runner.py +108 -10
  112. package/skills/torusguard/payload/scripts/finding_scorer.py +43 -13
  113. package/skills/torusguard/payload/skills/torusguard/SKILL.md +6 -2
  114. package/skills/torusguard/payload/skills/torusguard/bootstrap.py +3 -3
  115. package/skills/torusguard/payload/skills/torusguard-ai-guard/SKILL.md +95 -95
  116. package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +109 -84
  117. package/skills/torusguard/payload/skills/torusguard-container/SKILL.md +94 -94
  118. package/skills/torusguard/payload/skills/torusguard-git-mine/SKILL.md +92 -92
  119. package/skills/torusguard/payload/skills/torusguard-ocr-scan/SKILL.md +94 -94
  120. package/skills/torusguard/payload/skills/torusguard-redos/SKILL.md +91 -91
  121. package/skills/torusguard/payload/workflows/ai-guard.md +31 -31
  122. package/skills/torusguard/payload/workflows/audit.md +21 -17
  123. package/skills/torusguard/payload/workflows/container.md +29 -29
  124. package/skills/torusguard/payload/workflows/git-mine.md +25 -25
  125. package/skills/torusguard/payload/workflows/ocr-scan.md +25 -25
  126. package/skills/torusguard/payload/workflows/redos.md +27 -27
  127. package/skills/torusguard/payload/workflows/torusguard-audit.md +35 -55
  128. package/skills/torusguard/references/csharp-security.md +41 -41
  129. package/skills/torusguard/references/go-security.md +41 -41
  130. package/skills/torusguard/references/java-security.md +40 -40
  131. package/skills/torusguard/references/polyglot-security-matrix.md +25 -25
  132. package/skills/torusguard/references/rust-security.md +40 -40
  133. package/skills/torusguard-audit/SKILL.md +107 -83
@@ -1,51 +1,51 @@
1
- # TG-RAG-002: Autonomous LLM Tool Unsandboxed Call
2
-
3
- ## Severity
4
- Critical. Executing system shell commands, raw SQL, or filesystem modifications based directly on model tool call outputs without schema validation or sandboxing allows remote code execution (RCE).
5
-
6
- ## Applies To
7
- - LLM Function Calling, Agent Tool Calling, ReAct loops, Model Context Protocol (MCP) servers
8
- - Python, Node.js, Go
9
-
10
- ## Why It Matters
11
- When an AI agent calls external tools (e.g. `execute_code`, `query_database`, `run_bash`), the arguments originate from stochastic model generation. If the model was prompted or tricked via prompt injection to emit `rm -rf /` or `DROP TABLE users;`, executing those arguments without strict allowlists, parameterization, or sandboxing destroys data or compromises the server.
12
-
13
- ## What TorusGuard Looks For
14
- 1. Passing tool call arguments directly to `os.system`, `subprocess.run(..., shell=True)`, or `child_process.exec`.
15
- 2. Evaluating raw SQL emitted by LLM tool calls without parameterization.
16
- 3. Lack of human-in-the-loop confirmation on destructive tool invocations.
17
-
18
- ## Unsafe Example
19
- ```python
20
- # UNSAFE: Unsandboxed execution of model tool call
21
- def handle_tool_call(tool_call):
22
- if tool_call.function.name == "run_command":
23
- args = json.loads(tool_call.function.arguments)
24
- # Directly executes arbitrary shell command generated by LLM!
25
- return subprocess.check_output(args["cmd"], shell=True)
26
- ```
27
-
28
- ## Safe Example
29
- ```python
30
- # SAFE: Strict schema validation, command allowlisting, and no shell=True
31
- ALLOWED_COMMANDS = {"git status", "git diff", "npm test"}
32
-
33
- def handle_tool_call(tool_call):
34
- if tool_call.function.name == "run_command":
35
- args = json.loads(tool_call.function.arguments)
36
- cmd = args.get("cmd", "").strip()
37
-
38
- if cmd not in ALLOWED_COMMANDS:
39
- raise PermissionError(f"Command not permitted: {cmd}")
40
-
41
- return subprocess.check_output(cmd.split(), shell=False)
42
- ```
43
-
44
- ## Remediation
45
- 1. Enforce strict Pydantic/Zod schemas on all tool arguments.
46
- 2. Ban `shell=True` when invoking sub-processes from AI tool calls.
47
- 3. Require explicit human confirmation (Human Gate) for state-altering, file-writing, or network operations.
48
-
49
- ## Related Rules
50
- - `TG-AGENT-002`: Unsafe Tool Dispatch
51
- - `TG-INPUT-003`: Unsafe Code Execution
1
+ # TG-RAG-002: Autonomous LLM Tool Unsandboxed Call
2
+
3
+ ## Severity
4
+ Critical. Executing system shell commands, raw SQL, or filesystem modifications based directly on model tool call outputs without schema validation or sandboxing allows remote code execution (RCE).
5
+
6
+ ## Applies To
7
+ - LLM Function Calling, Agent Tool Calling, ReAct loops, Model Context Protocol (MCP) servers
8
+ - Python, Node.js, Go
9
+
10
+ ## Why It Matters
11
+ When an AI agent calls external tools (e.g. `execute_code`, `query_database`, `run_bash`), the arguments originate from stochastic model generation. If the model was prompted or tricked via prompt injection to emit `rm -rf /` or `DROP TABLE users;`, executing those arguments without strict allowlists, parameterization, or sandboxing destroys data or compromises the server.
12
+
13
+ ## What TorusGuard Looks For
14
+ 1. Passing tool call arguments directly to `os.system`, `subprocess.run(..., shell=True)`, or `child_process.exec`.
15
+ 2. Evaluating raw SQL emitted by LLM tool calls without parameterization.
16
+ 3. Lack of human-in-the-loop confirmation on destructive tool invocations.
17
+
18
+ ## Unsafe Example
19
+ ```python
20
+ # UNSAFE: Unsandboxed execution of model tool call
21
+ def handle_tool_call(tool_call):
22
+ if tool_call.function.name == "run_command":
23
+ args = json.loads(tool_call.function.arguments)
24
+ # Directly executes arbitrary shell command generated by LLM!
25
+ return subprocess.check_output(args["cmd"], shell=True)
26
+ ```
27
+
28
+ ## Safe Example
29
+ ```python
30
+ # SAFE: Strict schema validation, command allowlisting, and no shell=True
31
+ ALLOWED_COMMANDS = {"git status", "git diff", "npm test"}
32
+
33
+ def handle_tool_call(tool_call):
34
+ if tool_call.function.name == "run_command":
35
+ args = json.loads(tool_call.function.arguments)
36
+ cmd = args.get("cmd", "").strip()
37
+
38
+ if cmd not in ALLOWED_COMMANDS:
39
+ raise PermissionError(f"Command not permitted: {cmd}")
40
+
41
+ return subprocess.check_output(cmd.split(), shell=False)
42
+ ```
43
+
44
+ ## Remediation
45
+ 1. Enforce strict Pydantic/Zod schemas on all tool arguments.
46
+ 2. Ban `shell=True` when invoking sub-processes from AI tool calls.
47
+ 3. Require explicit human confirmation (Human Gate) for state-altering, file-writing, or network operations.
48
+
49
+ ## Related Rules
50
+ - `TG-AGENT-002`: Unsafe Tool Dispatch
51
+ - `TG-INPUT-003`: Unsafe Code Execution
@@ -1,51 +1,51 @@
1
- # TG-RAG-003: Unpartitioned Vector Database Tenant Lookup
2
-
3
- ## Severity
4
- High. Executing similarity searches across vector databases without multi-tenant metadata filters leaks private organization or user documents across tenant boundaries.
5
-
6
- ## Applies To
7
- - Vector Databases: Pinecone, Qdrant, Chroma, Weaviate, Milvus, pgvector
8
- - RAG applications with multi-tenant users or workspaces
9
-
10
- ## Why It Matters
11
- Vector embeddings from different tenants exist in the same high-dimensional embedding space. If an embedding lookup only searches by cosine similarity without an explicit `filter={"tenant_id": user.tenant_id}` or namespace partition, queries from User A will return private embeddings, contracts, or records belonging to User B.
12
-
13
- ## What TorusGuard Looks For
14
- 1. Vector similarity searches lacking metadata filter arguments (e.g. `index.query(vector=..., top_k=5)` with no `filter`).
15
- 2. Missing tenant partitioning in vector retrieval endpoints.
16
-
17
- ## Unsafe Example
18
- ```python
19
- # UNSAFE: Vector similarity search across all tenants
20
- def search_knowledge_base(user: User, query_vector: list[float]):
21
- results = pinecone_index.query(
22
- vector=query_vector,
23
- top_k=5,
24
- include_metadata=True
25
- # MISSING tenant filter!
26
- )
27
- return results
28
- ```
29
-
30
- ## Safe Example
31
- ```python
32
- # SAFE: Mandatory tenant scoping in metadata filter
33
- def search_knowledge_base(user: User, query_vector: list[float]):
34
- results = pinecone_index.query(
35
- vector=query_vector,
36
- top_k=5,
37
- include_metadata=True,
38
- filter={
39
- "tenant_id": {"$eq": user.tenant_id}
40
- }
41
- )
42
- return results
43
- ```
44
-
45
- ## Remediation
46
- 1. Always scope vector similarity queries by tenant ID in the metadata filter.
47
- 2. In pgvector, enforce row-level security (RLS) or explicit `WHERE tenant_id = :tenant_id` clauses on embedding queries.
48
-
49
- ## Related Rules
50
- - `TG-DB-001`: Missing Tenant Query Isolation
51
- - `TG-RAG-001`: Untrusted RAG Context Injection
1
+ # TG-RAG-003: Unpartitioned Vector Database Tenant Lookup
2
+
3
+ ## Severity
4
+ High. Executing similarity searches across vector databases without multi-tenant metadata filters leaks private organization or user documents across tenant boundaries.
5
+
6
+ ## Applies To
7
+ - Vector Databases: Pinecone, Qdrant, Chroma, Weaviate, Milvus, pgvector
8
+ - RAG applications with multi-tenant users or workspaces
9
+
10
+ ## Why It Matters
11
+ Vector embeddings from different tenants exist in the same high-dimensional embedding space. If an embedding lookup only searches by cosine similarity without an explicit `filter={"tenant_id": user.tenant_id}` or namespace partition, queries from User A will return private embeddings, contracts, or records belonging to User B.
12
+
13
+ ## What TorusGuard Looks For
14
+ 1. Vector similarity searches lacking metadata filter arguments (e.g. `index.query(vector=..., top_k=5)` with no `filter`).
15
+ 2. Missing tenant partitioning in vector retrieval endpoints.
16
+
17
+ ## Unsafe Example
18
+ ```python
19
+ # UNSAFE: Vector similarity search across all tenants
20
+ def search_knowledge_base(user: User, query_vector: list[float]):
21
+ results = pinecone_index.query(
22
+ vector=query_vector,
23
+ top_k=5,
24
+ include_metadata=True
25
+ # MISSING tenant filter!
26
+ )
27
+ return results
28
+ ```
29
+
30
+ ## Safe Example
31
+ ```python
32
+ # SAFE: Mandatory tenant scoping in metadata filter
33
+ def search_knowledge_base(user: User, query_vector: list[float]):
34
+ results = pinecone_index.query(
35
+ vector=query_vector,
36
+ top_k=5,
37
+ include_metadata=True,
38
+ filter={
39
+ "tenant_id": {"$eq": user.tenant_id}
40
+ }
41
+ )
42
+ return results
43
+ ```
44
+
45
+ ## Remediation
46
+ 1. Always scope vector similarity queries by tenant ID in the metadata filter.
47
+ 2. In pgvector, enforce row-level security (RLS) or explicit `WHERE tenant_id = :tenant_id` clauses on embedding queries.
48
+
49
+ ## Related Rules
50
+ - `TG-DB-001`: Missing Tenant Query Isolation
51
+ - `TG-RAG-001`: Untrusted RAG Context Injection
@@ -1,46 +1,46 @@
1
- # TG-REDOS-001: Catastrophic Exponential Backtracking in Regular Expression
2
-
3
- ## Severity
4
- High. Regular expressions with catastrophic backtracking trigger exponential time complexity ($O(2^n)$) when evaluating non-matching input strings, freezing CPU cores and causing Denial of Service.
5
-
6
- ## Applies To
7
- - JavaScript / TypeScript (`RegExp`, `pattern.test()`), Python (`re.match`, `re.search`), Go, Java, Ruby
8
- - Input validation patterns, email validators, URL extractors
9
-
10
- ## Why It Matters
11
- Traditional regex engines using NFA backtracking (e.g. JavaScript V8, Python `re`, Java `java.util.regex`, PCRE) explore all possible match paths on failure. When a pattern contains overlapping nested repetitions like `(a+)+$`, an input of 30 characters like `aaaaaaaaaaaaaaaaaaaaaaaaaaaaab` can require over 1 billion comparison operations, freezing the Node.js event loop or Python GIL.
12
-
13
- ## What TorusGuard Looks For
14
- 1. Nested repetitions: `([a-zA-Z0-9]+)+`, `(a+)+`, `(\d+)*`.
15
- 2. Overlapping alternations with outer quantifiers: `(a|aa)+`, `(x|x)*`.
16
- 3. Greedy repetition with overlapping prefix and suffix.
17
-
18
- ## Unsafe Example
19
- ```javascript
20
- // UNSAFE: Catastrophic backtracking on non-matching strings
21
- const EMAIL_REGEX = /^([a-zA-Z0-9_\.\-])+@(([a-zA-Z0-9\-])+\.)+([a-zA-Z0-9]{2,4})+$/;
22
-
23
- // Freezes server:
24
- EMAIL_REGEX.test("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa!");
25
- ```
26
-
27
- ## Safe Example
28
- ```javascript
29
- // SAFE: Linear time validation using atomic checks, character class bounds, or validator libraries
30
- const validator = require('validator');
31
- if (!validator.isEmail(input)) {
32
- throw new Error("Invalid email");
33
- }
34
-
35
- // Or constrained regex without nested quantifiers:
36
- const SAFE_EMAIL = /^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$/;
37
- ```
38
-
39
- ## Remediation
40
- 1. Eliminate nested quantifiers (`(x+)+` -> `x+`).
41
- 2. Disallow overlapping tokens in alternations.
42
- 3. In Node.js, wrap untrusted input validation with `safe-regex` or strict input length bounds (e.g. `if (input.length > 256) return false;`).
43
-
44
- ## Related Rules
45
- - `TG-REDOS-002`: Unbounded Nested Quantifier
46
- - `TG-RATE-003`: Unbounded Resource Consumption
1
+ # TG-REDOS-001: Catastrophic Exponential Backtracking in Regular Expression
2
+
3
+ ## Severity
4
+ High. Regular expressions with catastrophic backtracking trigger exponential time complexity ($O(2^n)$) when evaluating non-matching input strings, freezing CPU cores and causing Denial of Service.
5
+
6
+ ## Applies To
7
+ - JavaScript / TypeScript (`RegExp`, `pattern.test()`), Python (`re.match`, `re.search`), Go, Java, Ruby
8
+ - Input validation patterns, email validators, URL extractors
9
+
10
+ ## Why It Matters
11
+ Traditional regex engines using NFA backtracking (e.g. JavaScript V8, Python `re`, Java `java.util.regex`, PCRE) explore all possible match paths on failure. When a pattern contains overlapping nested repetitions like `(a+)+$`, an input of 30 characters like `aaaaaaaaaaaaaaaaaaaaaaaaaaaaab` can require over 1 billion comparison operations, freezing the Node.js event loop or Python GIL.
12
+
13
+ ## What TorusGuard Looks For
14
+ 1. Nested repetitions: `([a-zA-Z0-9]+)+`, `(a+)+`, `(\d+)*`.
15
+ 2. Overlapping alternations with outer quantifiers: `(a|aa)+`, `(x|x)*`.
16
+ 3. Greedy repetition with overlapping prefix and suffix.
17
+
18
+ ## Unsafe Example
19
+ ```javascript
20
+ // UNSAFE: Catastrophic backtracking on non-matching strings
21
+ const EMAIL_REGEX = /^([a-zA-Z0-9_\.\-])+@(([a-zA-Z0-9\-])+\.)+([a-zA-Z0-9]{2,4})+$/;
22
+
23
+ // Freezes server:
24
+ EMAIL_REGEX.test("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa!");
25
+ ```
26
+
27
+ ## Safe Example
28
+ ```javascript
29
+ // SAFE: Linear time validation using atomic checks, character class bounds, or validator libraries
30
+ const validator = require('validator');
31
+ if (!validator.isEmail(input)) {
32
+ throw new Error("Invalid email");
33
+ }
34
+
35
+ // Or constrained regex without nested quantifiers:
36
+ const SAFE_EMAIL = /^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$/;
37
+ ```
38
+
39
+ ## Remediation
40
+ 1. Eliminate nested quantifiers (`(x+)+` -> `x+`).
41
+ 2. Disallow overlapping tokens in alternations.
42
+ 3. In Node.js, wrap untrusted input validation with `safe-regex` or strict input length bounds (e.g. `if (input.length > 256) return false;`).
43
+
44
+ ## Related Rules
45
+ - `TG-REDOS-002`: Unbounded Nested Quantifier
46
+ - `TG-RATE-003`: Unbounded Resource Consumption
@@ -1,43 +1,43 @@
1
- # TG-REDOS-002: Unbounded Nested Quantifier in Input Validation
2
-
3
- ## Severity
4
- Medium. Unbounded repeated capture groups without boundary anchors cause polynomial ($O(n^2)$) or exponential degradation on large payloads.
5
-
6
- ## Applies To
7
- - Input validation filters, route path matchers, sanitizer regexes
8
- - Polyglot web backends and client-side form validators
9
-
10
- ## Why It Matters
11
- When regexes use repeated capture groups like `(\w+\s*)+` without anchoring, trailing spaces or punctuation force the engine into deep recursive state branches. While not always pure exponential, large payloads (e.g. 50KB JSON strings) will peg CPU at 100% for minutes.
12
-
13
- ## What TorusGuard Looks For
14
- 1. Nested groups where both inner and outer components have greedy repetition (`+` or `*`).
15
- 2. Regexes evaluated on user-supplied strings without a preceding string length check.
16
-
17
- ## Unsafe Example
18
- ```python
19
- # UNSAFE: Unbounded nested quantifier on user input
20
- import re
21
-
22
- TAG_REGEX = re.compile(r"^(<[a-z]+(\s+[a-z]+=[^>]+)*>)+$")
23
- match = TAG_REGEX.match(user_payload)
24
- ```
25
-
26
- ## Safe Example
27
- ```python
28
- # SAFE: Bounded input length check + non-nested linear pattern
29
- import re
30
-
31
- if len(user_payload) > 512:
32
- return False
33
-
34
- # Use a dedicated HTML parser (BeautifulSoup / html5lib) instead of regex
35
- ```
36
-
37
- ## Remediation
38
- 1. Bound user input length *before* regex execution.
39
- 2. Replace complex nested regexes with dedicated, parser-based validation libraries (e.g., standard parsers for HTML, URLs, and emails).
40
-
41
- ## Related Rules
42
- - `TG-REDOS-001`: Catastrophic Exponential Backtracking
43
- - `TG-INPUT-001`: Missing Server Validation
1
+ # TG-REDOS-002: Unbounded Nested Quantifier in Input Validation
2
+
3
+ ## Severity
4
+ Medium. Unbounded repeated capture groups without boundary anchors cause polynomial ($O(n^2)$) or exponential degradation on large payloads.
5
+
6
+ ## Applies To
7
+ - Input validation filters, route path matchers, sanitizer regexes
8
+ - Polyglot web backends and client-side form validators
9
+
10
+ ## Why It Matters
11
+ When regexes use repeated capture groups like `(\w+\s*)+` without anchoring, trailing spaces or punctuation force the engine into deep recursive state branches. While not always pure exponential, large payloads (e.g. 50KB JSON strings) will peg CPU at 100% for minutes.
12
+
13
+ ## What TorusGuard Looks For
14
+ 1. Nested groups where both inner and outer components have greedy repetition (`+` or `*`).
15
+ 2. Regexes evaluated on user-supplied strings without a preceding string length check.
16
+
17
+ ## Unsafe Example
18
+ ```python
19
+ # UNSAFE: Unbounded nested quantifier on user input
20
+ import re
21
+
22
+ TAG_REGEX = re.compile(r"^(<[a-z]+(\s+[a-z]+=[^>]+)*>)+$")
23
+ match = TAG_REGEX.match(user_payload)
24
+ ```
25
+
26
+ ## Safe Example
27
+ ```python
28
+ # SAFE: Bounded input length check + non-nested linear pattern
29
+ import re
30
+
31
+ if len(user_payload) > 512:
32
+ return False
33
+
34
+ # Use a dedicated HTML parser (BeautifulSoup / html5lib) instead of regex
35
+ ```
36
+
37
+ ## Remediation
38
+ 1. Bound user input length *before* regex execution.
39
+ 2. Replace complex nested regexes with dedicated, parser-based validation libraries (e.g., standard parsers for HTML, URLs, and emails).
40
+
41
+ ## Related Rules
42
+ - `TG-REDOS-001`: Catastrophic Exponential Backtracking
43
+ - `TG-INPUT-001`: Missing Server Validation
@@ -42,8 +42,16 @@ def get_ist_now() -> datetime.datetime:
42
42
 
43
43
  # ─── UI Formatter Bridge ──────────────────────────────────────────────────────
44
44
  scripts_dir = Path(__file__).resolve().parent
45
+ tg_root = Path(__file__).resolve().parent.parent
46
+ project_root = Path(__file__).resolve().parent.parent.parent
45
47
  if str(scripts_dir) not in sys.path:
46
48
  sys.path.insert(0, str(scripts_dir))
49
+ if str(tg_root) not in sys.path:
50
+ sys.path.insert(0, str(tg_root))
51
+ if str(project_root) not in sys.path:
52
+ sys.path.insert(0, str(project_root))
53
+
54
+
47
55
 
48
56
  try:
49
57
  import term_ui as tui
@@ -1240,9 +1248,10 @@ def scan_file(file_path: Path, target_root: Path) -> List[Dict[str, Any]]:
1240
1248
  return findings
1241
1249
 
1242
1250
 
1243
- def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path) -> Tuple[List[Dict[str, Any]], Dict[str, List[Dict[str, Any]]]]:
1251
+ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path, taint_paths: Optional[List[Any]] = None) -> Tuple[List[Dict[str, Any]], Dict[str, List[Dict[str, Any]]]]:
1244
1252
  """
1245
- Score each finding using finding_scorer.py and persistent memory patterns.
1253
+ Score each finding using finding_scorer.py, persistent memory patterns,
1254
+ and taint dataflow analysis.
1246
1255
  Returns (scored_findings, clusters_map).
1247
1256
  """
1248
1257
  try:
@@ -1253,6 +1262,15 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
1253
1262
  scored = []
1254
1263
  clusters: Dict[str, List[Dict[str, Any]]] = {}
1255
1264
 
1265
+ # Map taint paths by (file_path, line_number)
1266
+ taint_by_loc: Dict[Tuple[str, int], Any] = {}
1267
+ if taint_paths:
1268
+ for tp in taint_paths:
1269
+ sink_node = getattr(tp, "sink", None)
1270
+ if sink_node:
1271
+ norm_p = getattr(sink_node, "file_path", "").replace("\\", "/")
1272
+ taint_by_loc[(norm_p, getattr(sink_node, "line_number", 0))] = tp
1273
+
1256
1274
  # Phase 1d: Pre-compute cross-file corroboration bonus
1257
1275
  # If multiple rule families flag the same file, each finding gets +5 confidence
1258
1276
  file_rule_families: Dict[str, set] = {}
@@ -1267,6 +1285,14 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
1267
1285
  band = "High Confidence"
1268
1286
  factors = {}
1269
1287
 
1288
+ # Check for correlated taint path
1289
+ matching_tp = taint_by_loc.get((f["file_path"], f["line_number"]))
1290
+ is_taint_confirmed = matching_tp is not None
1291
+ taint_depth = getattr(matching_tp, "depth", None) if matching_tp else None
1292
+ is_sanitized = getattr(matching_tp, "is_sanitized", False) if matching_tp else False
1293
+ if matching_tp and hasattr(matching_tp, "to_dict"):
1294
+ f["taint_path"] = matching_tp.to_dict()
1295
+
1270
1296
  if finding_scorer:
1271
1297
  try:
1272
1298
  # Phase 1d: Dynamic evidence quality based on rule precision
@@ -1291,7 +1317,11 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
1291
1317
  manual_review_status=mr,
1292
1318
  rule_id=f["rule_id"],
1293
1319
  file_path=f["file_path"],
1294
- root_dir=target_root
1320
+ root_dir=target_root,
1321
+ taint_path_confirmed=is_taint_confirmed,
1322
+ taint_depth=taint_depth,
1323
+ sanitizer_present=is_sanitized,
1324
+ rule_severity=f.get("severity", "High")
1295
1325
  )
1296
1326
  score = s
1297
1327
  band = b
@@ -1315,6 +1345,7 @@ def score_and_cluster_findings(findings: List[Dict[str, Any]], target_root: Path
1315
1345
  return scored, clusters
1316
1346
 
1317
1347
 
1348
+
1318
1349
  def emit_run_artifacts(run_folder: Path, scored_findings: List[Dict[str, Any]], clusters: Dict[str, List[Dict[str, Any]]], target_root: Path) -> None:
1319
1350
  """Generate findings.json, findings.md, and summary.md into run folder."""
1320
1351
  now_ist = get_ist_now()
@@ -1501,7 +1532,14 @@ def run_watch_mode(target_root: Path, severity_floor: str = "medium", json_outpu
1501
1532
  print(f"\n {YELLOW}🛑 Watch mode stopped.{RESET}\n")
1502
1533
 
1503
1534
 
1504
- def execute_audit(target_root: Path, severity_floor: str = "medium", json_output: bool = False, include_tests: bool = False) -> Dict[str, Any]:
1535
+ def execute_audit(
1536
+ target_root: Path,
1537
+ severity_floor: str = "medium",
1538
+ json_output: bool = False,
1539
+ include_tests: bool = False,
1540
+ incremental: bool = False,
1541
+ use_taint: bool = True
1542
+ ) -> Dict[str, Any]:
1505
1543
  """Execute the full TorusGuard static security audit."""
1506
1544
  start_time = time.perf_counter()
1507
1545
  target_root = target_root.resolve()
@@ -1522,14 +1560,64 @@ def execute_audit(target_root: Path, severity_floor: str = "medium", json_output
1522
1560
  except Exception:
1523
1561
  pass
1524
1562
 
1525
- # 2. Collect files & scan
1563
+ # 2. Collect files to scan
1526
1564
  files = find_files_to_scan(target_root, include_tests=include_tests)
1527
1565
  all_findings = []
1528
- for f in files:
1529
- all_findings.extend(scan_file(f, target_root))
1530
1566
 
1531
- # 3. Score & Cluster
1532
- scored_findings, clusters = score_and_cluster_findings(all_findings, target_root)
1567
+ inc_scanner = None
1568
+ files_to_scan = files
1569
+ unchanged_files = []
1570
+
1571
+ if incremental:
1572
+ try:
1573
+ from core.incremental import IncrementalScanner
1574
+ inc_scanner = IncrementalScanner(target_root)
1575
+ files_to_scan, unchanged_files = inc_scanner.get_changed_files(files)
1576
+ # Rehydrate findings for unchanged files
1577
+ for uf in unchanged_files:
1578
+ all_findings.extend(inc_scanner.get_cached_findings(uf))
1579
+ except Exception:
1580
+ files_to_scan = files
1581
+ unchanged_files = []
1582
+
1583
+ # Parallel or sequential scan on files_to_scan
1584
+ new_findings = []
1585
+ try:
1586
+ from core.parallel import ParallelAuditExecutor
1587
+ executor = ParallelAuditExecutor()
1588
+ new_findings = executor.scan_files_parallel(files_to_scan, lambda f: scan_file(f, target_root))
1589
+ except Exception:
1590
+ for f in files_to_scan:
1591
+ new_findings.extend(scan_file(f, target_root))
1592
+
1593
+ all_findings.extend(new_findings)
1594
+
1595
+ # Update cache if incremental scanner is active
1596
+ if inc_scanner:
1597
+ # Group new findings by file
1598
+ file_to_findings: Dict[Path, List[Dict[str, Any]]] = {f: [] for f in files_to_scan}
1599
+ for nf in new_findings:
1600
+ raw_fp = nf.get("file_path", "")
1601
+ target_f = target_root / raw_fp
1602
+ if target_f in file_to_findings:
1603
+ file_to_findings[target_f].append(nf)
1604
+ for target_f, f_list in file_to_findings.items():
1605
+ inc_scanner.update_file_cache(target_f, f_list)
1606
+ inc_scanner.save_cache()
1607
+
1608
+ # 2.5 Taint Dataflow Analysis (if enabled)
1609
+ taint_paths = []
1610
+ if use_taint:
1611
+ try:
1612
+ from core.cross_file_taint import CrossFileTaintAnalyzer
1613
+ analyzer = CrossFileTaintAnalyzer(target_root)
1614
+ # Analyze target files (capped to 200 files for high responsiveness)
1615
+ taint_paths = analyzer.analyze_project(files[:200])
1616
+ except Exception:
1617
+ taint_paths = []
1618
+
1619
+ # 3. Score & Cluster with Taint Evidence
1620
+ scored_findings, clusters = score_and_cluster_findings(all_findings, target_root, taint_paths=taint_paths)
1533
1621
 
1534
1622
  # 4. Allocate run folder
1535
1623
  runs_dir = target_root / ".torusguard" / "runs"
@@ -1595,6 +1683,8 @@ def main():
1595
1683
  parser.add_argument("--scope", "-s", help="Alternative path to target project")
1596
1684
  parser.add_argument("--severity", choices=["critical", "high", "medium", "low"], default="medium", help="Severity floor")
1597
1685
  parser.add_argument("--watch", "-w", action="store_true", help="Continuous watch mode: re-scan on file save")
1686
+ parser.add_argument("--incremental", "-i", action="store_true", help="Incremental mode: only scan modified files")
1687
+ parser.add_argument("--no-taint", action="store_true", help="Disable taint-aware dataflow analysis")
1598
1688
  parser.add_argument("--sarif", action="store_true", help="Automatically export findings to OASIS SARIF v2.1.0")
1599
1689
  parser.add_argument("--sarif-out", help="Output file path for SARIF export")
1600
1690
  parser.add_argument("--json", action="store_true", help="Output raw JSON")
@@ -1607,10 +1697,18 @@ def main():
1607
1697
  run_watch_mode(target, severity_floor=args.severity, json_output=args.json, include_tests=args.include_tests, sarif=args.sarif, sarif_out=args.sarif_out)
1608
1698
  sys.exit(0)
1609
1699
 
1610
- res = execute_audit(target, severity_floor=args.severity, json_output=args.json, include_tests=args.include_tests)
1700
+ res = execute_audit(
1701
+ target,
1702
+ severity_floor=args.severity,
1703
+ json_output=args.json,
1704
+ include_tests=args.include_tests,
1705
+ incremental=args.incremental,
1706
+ use_taint=(not args.no_taint)
1707
+ )
1611
1708
  if args.sarif:
1612
1709
  export_sarif(target, res.get("run_folder"), args.sarif_out)
1613
1710
 
1711
+
1614
1712
  sys.exit(0 if res["critical_count"] == 0 else 1)
1615
1713
 
1616
1714