rulevis 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- __init__.py +0 -0
- internal/analyzer.py +167 -0
- internal/generator.py +176 -0
- internal/visualizer.py +336 -0
- rulevis-1.0.0.dist-info/METADATA +168 -0
- rulevis-1.0.0.dist-info/RECORD +11 -0
- rulevis-1.0.0.dist-info/WHEEL +5 -0
- rulevis-1.0.0.dist-info/entry_points.txt +2 -0
- rulevis-1.0.0.dist-info/licenses/LICENSE +21 -0
- rulevis-1.0.0.dist-info/top_level.txt +3 -0
- rulevis.py +200 -0
__init__.py
ADDED
|
File without changes
|
internal/analyzer.py
ADDED
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import math
|
|
3
|
+
import pickle
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
from networkx import MultiDiGraph, descendants, ancestors, nodes_with_selfloops, simple_cycles
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Analyzer:
|
|
10
|
+
"""
|
|
11
|
+
Analyzes a graph from a pickle file and calculates key statistics.
|
|
12
|
+
"""
|
|
13
|
+
def __init__(self, graph_path: str) -> None:
|
|
14
|
+
"""
|
|
15
|
+
Initializes the Analyzer with the path to the graph file.
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
graph_path (str): The path to the .gpickle file.
|
|
19
|
+
|
|
20
|
+
Raises:
|
|
21
|
+
FileNotFoundError: If the graph file does not exist.
|
|
22
|
+
"""
|
|
23
|
+
if not os.path.isfile(graph_path):
|
|
24
|
+
raise FileNotFoundError(f"Graph file not found: {graph_path}")
|
|
25
|
+
self.graph_path = graph_path
|
|
26
|
+
self.G: MultiDiGraph = self._load_graph()
|
|
27
|
+
|
|
28
|
+
def _load_graph(self) -> MultiDiGraph:
|
|
29
|
+
"""Loads the graph from the pickle file."""
|
|
30
|
+
with open(self.graph_path, "rb") as f:
|
|
31
|
+
return pickle.load(f)
|
|
32
|
+
|
|
33
|
+
def calculate_statistics(self) -> dict:
|
|
34
|
+
"""
|
|
35
|
+
Calculates all the required statistics for the graph, ensuring no
|
|
36
|
+
duplicate cycles are reported.
|
|
37
|
+
"""
|
|
38
|
+
real_nodes = [n for n in self.G.nodes() if n != '0']
|
|
39
|
+
|
|
40
|
+
# --- Calculate Top 5 Lists ---
|
|
41
|
+
real_nodes = [n for n in self.G.nodes() if n != '0']
|
|
42
|
+
out_degrees = dict(self.G.out_degree(real_nodes))
|
|
43
|
+
in_degrees = dict(self.G.in_degree(real_nodes))
|
|
44
|
+
|
|
45
|
+
top_5_direct_descendants = sorted(out_degrees, key=out_degrees.get, reverse=True)[:5]
|
|
46
|
+
indirect_descendants_counts = {n: len(descendants(self.G, n)) for n in real_nodes}
|
|
47
|
+
top_5_indirect_descendants = sorted(indirect_descendants_counts,
|
|
48
|
+
key=indirect_descendants_counts.get, reverse=True)[:5]
|
|
49
|
+
top_5_direct_ancestors = sorted(in_degrees,
|
|
50
|
+
key=in_degrees.get, reverse=True)[:5]
|
|
51
|
+
indirect_ancestors_counts = {n: len(ancestors(self.G, n)) for n in real_nodes}
|
|
52
|
+
top_5_indirect_ancestors = sorted(indirect_ancestors_counts,
|
|
53
|
+
key=indirect_ancestors_counts.get, reverse=True)[:5]
|
|
54
|
+
|
|
55
|
+
isolated_rules = [
|
|
56
|
+
n for n in real_nodes
|
|
57
|
+
if self.G.out_degree(n) == 0 and list(self.G.predecessors(n)) == ['0']
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
# 1. Find all nodes with self-loops first. This is our definitive list.
|
|
61
|
+
logging.info("Detecting self-loops (e.g., A -> A)...")
|
|
62
|
+
self_loop_nodes = set(nodes_with_selfloops(self.G)) # Use a set for efficient lookup
|
|
63
|
+
logging.info(f"Found {len(self_loop_nodes)} nodes with self-loops: {self_loop_nodes}")
|
|
64
|
+
|
|
65
|
+
# 2. Find all other simple cycles.
|
|
66
|
+
logging.info("Detecting multi-node cycles...")
|
|
67
|
+
all_simple_cycles = list(simple_cycles(self.G))
|
|
68
|
+
|
|
69
|
+
# 3. Filter out any multi-node cycles that contain a node we've already
|
|
70
|
+
# identified as having a self-loop. This prevents double-reporting.
|
|
71
|
+
multi_node_cycles = [
|
|
72
|
+
cycle for cycle in all_simple_cycles
|
|
73
|
+
if not self_loop_nodes.intersection(cycle)
|
|
74
|
+
]
|
|
75
|
+
|
|
76
|
+
logging.info(f"Found {len(multi_node_cycles)} distinct multi-node cycles.")
|
|
77
|
+
if multi_node_cycles:
|
|
78
|
+
logging.info(f"Example multi-node cycle: {multi_node_cycles[0]}")
|
|
79
|
+
|
|
80
|
+
# 4. Format the multi-node cycles for display
|
|
81
|
+
for cycle in multi_node_cycles:
|
|
82
|
+
if cycle:
|
|
83
|
+
cycle.append(cycle[0])
|
|
84
|
+
|
|
85
|
+
stats = {
|
|
86
|
+
"top_direct_descendants": [{"id": n, "count": out_degrees[n]}
|
|
87
|
+
for n in top_5_direct_descendants],
|
|
88
|
+
"top_indirect_descendants": [{"id": n, "count": indirect_descendants_counts[n]}
|
|
89
|
+
for n in top_5_indirect_descendants],
|
|
90
|
+
"top_direct_ancestors": [{"id": n, "count": in_degrees[n]} for n in top_5_direct_ancestors],
|
|
91
|
+
"top_indirect_ancestors": [{"id": n, "count": indirect_ancestors_counts[n]}
|
|
92
|
+
for n in top_5_indirect_ancestors],
|
|
93
|
+
"isolated_rules": [{"id": n}
|
|
94
|
+
for n in isolated_rules],
|
|
95
|
+
"self_loops": [{"id": n}
|
|
96
|
+
for n in self_loop_nodes],
|
|
97
|
+
"cycles": multi_node_cycles
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
return stats
|
|
101
|
+
|
|
102
|
+
def calculate_heatmap_data(self, block_size: int = 10) -> dict:
|
|
103
|
+
"""
|
|
104
|
+
Generates data for a block-based heatmap, showing rule ID occupancy.
|
|
105
|
+
The heatmap range is dynamically calculated based on the highest rule ID found.
|
|
106
|
+
|
|
107
|
+
Args:
|
|
108
|
+
block_size (int): The size of each ID range block (e.g., 10).
|
|
109
|
+
|
|
110
|
+
Returns:
|
|
111
|
+
dict: A dictionary containing the list of blocks and metadata.
|
|
112
|
+
"""
|
|
113
|
+
all_rule_ids = {int(n) for n in self.G.nodes() if n.isdigit() and n != '0'}
|
|
114
|
+
|
|
115
|
+
if not all_rule_ids:
|
|
116
|
+
# Handle case where there are no integer rules
|
|
117
|
+
return {"metadata": {"block_size": block_size, "max_id": 0, "total_blocks": 0}, "blocks": []}
|
|
118
|
+
|
|
119
|
+
actual_max_id = max(all_rule_ids)
|
|
120
|
+
|
|
121
|
+
# Calculate the upper bound for the heatmap range.
|
|
122
|
+
# We use math.ceil to round up to the next full block.
|
|
123
|
+
# For example, if max ID is 101234 and block size is 1000, this becomes:
|
|
124
|
+
# ceil(101234 / 1000) * 1000 => ceil(101.234) * 1000 => 102 * 1000 => 102000
|
|
125
|
+
dynamic_max_range = math.ceil(actual_max_id / block_size) * block_size
|
|
126
|
+
|
|
127
|
+
# Ensure we have at least one block even if max_id is small
|
|
128
|
+
dynamic_max_range = max(dynamic_max_range, block_size)
|
|
129
|
+
|
|
130
|
+
blocks = []
|
|
131
|
+
# Use the dynamically calculated range for the loop
|
|
132
|
+
for i in range(0, dynamic_max_range, block_size):
|
|
133
|
+
start_range = i
|
|
134
|
+
end_range = i + block_size - 1
|
|
135
|
+
|
|
136
|
+
count = sum(1 for rule_id in all_rule_ids if start_range <= rule_id <= end_range)
|
|
137
|
+
|
|
138
|
+
blocks.append({
|
|
139
|
+
"id": f"{start_range}-{end_range}",
|
|
140
|
+
"count": count
|
|
141
|
+
})
|
|
142
|
+
|
|
143
|
+
return {
|
|
144
|
+
"metadata": {
|
|
145
|
+
"block_size": block_size,
|
|
146
|
+
"max_id": dynamic_max_range,
|
|
147
|
+
"total_blocks": len(blocks)
|
|
148
|
+
},
|
|
149
|
+
"blocks": blocks
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
def write_to_json(self, stats_output_path: str, heatmap_output_path: str) -> None:
|
|
153
|
+
"""Calculates all data and writes to respective files."""
|
|
154
|
+
|
|
155
|
+
logging.info("Calculating graph statistics...")
|
|
156
|
+
stats_data = self.calculate_statistics()
|
|
157
|
+
logging.info(f"Writing statistics to {stats_output_path}...")
|
|
158
|
+
with open(stats_output_path, "w") as f:
|
|
159
|
+
json.dump(stats_data, f, indent=4)
|
|
160
|
+
|
|
161
|
+
logging.info("Calculating heatmap data...")
|
|
162
|
+
heatmap_data = self.calculate_heatmap_data()
|
|
163
|
+
logging.info(f"Writing heatmap data to {heatmap_output_path}...")
|
|
164
|
+
with open(heatmap_output_path, "w") as f:
|
|
165
|
+
json.dump(heatmap_data, f, indent=2)
|
|
166
|
+
|
|
167
|
+
logging.info("All analysis complete.")
|
internal/generator.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
import pickle
|
|
4
|
+
import xml.etree.ElementTree as ET
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
from typing import Any, Final, Optional
|
|
7
|
+
|
|
8
|
+
import networkx as nx
|
|
9
|
+
|
|
10
|
+
ENCODING: Final[str] = "utf-8"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class GraphGenerator:
|
|
14
|
+
def __init__(self, paths: list[str], graph_file: str) -> None:
|
|
15
|
+
self.paths = paths
|
|
16
|
+
self.group_membership: dict[str, list[str]] = defaultdict(list)
|
|
17
|
+
self.G: nx.MultiDiGraph[Any] = nx.MultiDiGraph()
|
|
18
|
+
self.graph_file: str = graph_file
|
|
19
|
+
|
|
20
|
+
def get_all_xml_files(self) -> list[str]:
|
|
21
|
+
xml_files: list[str] = []
|
|
22
|
+
for path in self.paths:
|
|
23
|
+
for root, _, files in os.walk(path):
|
|
24
|
+
for file in files:
|
|
25
|
+
if file.lower().endswith('.xml'):
|
|
26
|
+
abs = os.path.abspath(os.path.join(root, file))
|
|
27
|
+
xml_files.append(abs)
|
|
28
|
+
|
|
29
|
+
print(f'Found {len(xml_files)} XML files in the given paths')
|
|
30
|
+
logging.info(f'Found {len(xml_files)} XML files in the given paths')
|
|
31
|
+
logging.info('Processing all files...')
|
|
32
|
+
return xml_files
|
|
33
|
+
|
|
34
|
+
def add_edge_with_type(self, source: str, target: str, relation_type: str) -> None:
|
|
35
|
+
if logging.getLogger().getEffectiveLevel() <= logging.DEBUG:
|
|
36
|
+
logging.debug(
|
|
37
|
+
f"Adding edge from {source} to {target} with type {relation_type}")
|
|
38
|
+
self.G.add_edge(source, target, relation_type=relation_type)
|
|
39
|
+
|
|
40
|
+
def add_relationship_edges(self, rule_id: str,
|
|
41
|
+
if_sid: Optional[str], if_matched_sid: Optional[str],
|
|
42
|
+
if_group: Optional[str], if_matched_group: Optional[str]) -> None:
|
|
43
|
+
if if_sid:
|
|
44
|
+
for sid in if_sid.split(','):
|
|
45
|
+
self.add_edge_with_type(sid.strip(), rule_id, 'if_sid')
|
|
46
|
+
|
|
47
|
+
if if_matched_sid:
|
|
48
|
+
for sid in if_matched_sid.split(','):
|
|
49
|
+
self.add_edge_with_type(sid.strip(), rule_id, 'if_matched_sid')
|
|
50
|
+
|
|
51
|
+
if if_group:
|
|
52
|
+
for group in if_group.split(','):
|
|
53
|
+
for parent_rule in self.group_membership.get(group.strip(), []):
|
|
54
|
+
self.add_edge_with_type(parent_rule, rule_id, 'if_group')
|
|
55
|
+
|
|
56
|
+
if if_matched_group:
|
|
57
|
+
for group in if_matched_group.split(','):
|
|
58
|
+
for parent_rule in self.group_membership.get(group.strip(), []):
|
|
59
|
+
self.add_edge_with_type(
|
|
60
|
+
parent_rule, rule_id, 'if_matched_group')
|
|
61
|
+
|
|
62
|
+
def parse_groups_and_rules(self, element: ET.Element, inherited_groups: list[str], xml_file: str) -> None:
|
|
63
|
+
if element.tag == 'rule':
|
|
64
|
+
rule_id = element.get('id', '0')
|
|
65
|
+
rule_level = element.get('level')
|
|
66
|
+
if_sid = element.findtext('if_sid', None)
|
|
67
|
+
if_matched_sid = element.findtext('if_matched_sid', None)
|
|
68
|
+
if_group = element.findtext('if_group', None)
|
|
69
|
+
if_matched_group = element.findtext('if_matched_group', None)
|
|
70
|
+
|
|
71
|
+
attributes = [(i.tag, i.text) for i in element]
|
|
72
|
+
rule_description = self.extract_rule_description(attributes)
|
|
73
|
+
all_groups = self.extract_rule_groups(inherited_groups, attributes)
|
|
74
|
+
|
|
75
|
+
if self.G.nodes.get(rule_id) is not None:
|
|
76
|
+
logging.debug(
|
|
77
|
+
f"Duplicate rule ID found: {rule_id}. Skipping this rule.")
|
|
78
|
+
|
|
79
|
+
else:
|
|
80
|
+
self.G.add_node(rule_id,
|
|
81
|
+
groups=all_groups,
|
|
82
|
+
description=rule_description,
|
|
83
|
+
level=rule_level,
|
|
84
|
+
file=os.path.basename(xml_file))
|
|
85
|
+
for group in all_groups:
|
|
86
|
+
self.group_membership[group].append(rule_id)
|
|
87
|
+
|
|
88
|
+
self.add_relationship_edges(
|
|
89
|
+
rule_id, if_sid, if_matched_sid, if_group, if_matched_group)
|
|
90
|
+
|
|
91
|
+
elif element.tag == 'group':
|
|
92
|
+
group_attribute = element.get('name', '')
|
|
93
|
+
internal_groups = [
|
|
94
|
+
gr for gr in group_attribute.split(',') if gr != '']
|
|
95
|
+
new_inherited_groups = inherited_groups + internal_groups
|
|
96
|
+
|
|
97
|
+
for child in element:
|
|
98
|
+
self.parse_groups_and_rules(child, new_inherited_groups, xml_file)
|
|
99
|
+
|
|
100
|
+
def extract_rule_groups(self, inherited_groups: list[str], children: list[tuple[str, Optional[str]]]) -> list[str]:
|
|
101
|
+
all_groups = list(inherited_groups)
|
|
102
|
+
for child in children:
|
|
103
|
+
if child[0] == 'group' and child[1]:
|
|
104
|
+
all_groups.extend([g for g in child[1].split(',') if g])
|
|
105
|
+
return all_groups
|
|
106
|
+
|
|
107
|
+
def extract_rule_description(self, attributes: list[tuple[str, Optional[str]]]) -> Optional[str]:
|
|
108
|
+
description: list[str] = []
|
|
109
|
+
for attr in attributes:
|
|
110
|
+
if attr[0] == 'description':
|
|
111
|
+
d = attr[1]
|
|
112
|
+
if d:
|
|
113
|
+
description.append(d)
|
|
114
|
+
if len(description) > 0:
|
|
115
|
+
return ' '.join(description)
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
def wrap_with_root(self, xml_content: str) -> str:
|
|
119
|
+
return f"<root>{xml_content}</root>"
|
|
120
|
+
|
|
121
|
+
def build_graph_from_xml(self) -> None:
|
|
122
|
+
xml_files = self.get_all_xml_files()
|
|
123
|
+
|
|
124
|
+
for xml_file in xml_files:
|
|
125
|
+
logging.info(f'Processing file: {xml_file}')
|
|
126
|
+
try:
|
|
127
|
+
with open(xml_file, 'r', encoding=ENCODING) as f:
|
|
128
|
+
xml_content = f.read()
|
|
129
|
+
except OSError as e:
|
|
130
|
+
logging.error(
|
|
131
|
+
f"Error reading file {xml_file}: {e}", exc_info=True)
|
|
132
|
+
continue
|
|
133
|
+
|
|
134
|
+
wrapped_content: str = self.wrap_with_root(xml_content)
|
|
135
|
+
|
|
136
|
+
try:
|
|
137
|
+
parsed_xml = ET.fromstring(wrapped_content)
|
|
138
|
+
root = parsed_xml
|
|
139
|
+
for child in root:
|
|
140
|
+
self.parse_groups_and_rules(child, [], xml_file)
|
|
141
|
+
except Exception as e:
|
|
142
|
+
logging.error(f"Error parsing {xml_file}: {e}", exc_info=True)
|
|
143
|
+
|
|
144
|
+
first_level_rules = [
|
|
145
|
+
node for node in self.G.nodes if self.G.in_degree(node) == 0]
|
|
146
|
+
|
|
147
|
+
# Add synthetic root and connect to top-level rules
|
|
148
|
+
synthetic_root = '0' # Root has ID of 0
|
|
149
|
+
self.G.add_node(
|
|
150
|
+
synthetic_root, description="Synthetic root node", groups=["__meta__"])
|
|
151
|
+
|
|
152
|
+
for node in first_level_rules:
|
|
153
|
+
self.add_edge_with_type(synthetic_root, node, "root")
|
|
154
|
+
|
|
155
|
+
# Pre-calculate and store all children for every node.
|
|
156
|
+
# This is crucial for the frontend to know if a node is fully expanded.
|
|
157
|
+
print("Pre-calculating child relationships...")
|
|
158
|
+
for node_id in list(self.G.nodes):
|
|
159
|
+
# G.successors(node_id) returns an iterator of all direct children
|
|
160
|
+
children_ids = list(self.G.successors(node_id))
|
|
161
|
+
# Store this list as a new attribute on the node itself.
|
|
162
|
+
self.G.nodes[node_id]['children_ids'] = children_ids
|
|
163
|
+
print("Child relationship calculation complete.")
|
|
164
|
+
|
|
165
|
+
print("Total nodes:", self.G.number_of_nodes())
|
|
166
|
+
print("First-level children (connected to root):",
|
|
167
|
+
len(list(self.G.successors("0"))))
|
|
168
|
+
|
|
169
|
+
def save_graph(self) -> None:
|
|
170
|
+
try:
|
|
171
|
+
output_path = self.graph_file
|
|
172
|
+
os.makedirs(os.path.dirname(output_path), exist_ok=True)
|
|
173
|
+
pickle.dump(self.G, open(output_path, 'wb'))
|
|
174
|
+
logging.info(f"Graph saved to {output_path}")
|
|
175
|
+
except Exception as e:
|
|
176
|
+
logging.error(f"Error saving graph: {e}", exc_info=True)
|
internal/visualizer.py
ADDED
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
import pickle
|
|
4
|
+
from typing import Any, Union
|
|
5
|
+
|
|
6
|
+
from flask import Flask, jsonify, render_template_string, request
|
|
7
|
+
from flask.wrappers import Response
|
|
8
|
+
from networkx import MultiDiGraph
|
|
9
|
+
|
|
10
|
+
PRECOMPUTED_BLOCK_SIZES: list[int] = [1, 10, 50, 100, 250, 500]
|
|
11
|
+
PRECOMPUTED_HEATMAPS: dict[int, Any] = {}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def parse_start(idstr: str) -> int:
|
|
15
|
+
if not idstr:
|
|
16
|
+
return 0
|
|
17
|
+
s = str(idstr).strip()
|
|
18
|
+
for sep in ("-", "–"):
|
|
19
|
+
if sep in s:
|
|
20
|
+
s = s.split(sep, 1)[0]
|
|
21
|
+
break
|
|
22
|
+
try:
|
|
23
|
+
return int(s)
|
|
24
|
+
except Exception:
|
|
25
|
+
return 0
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def precompute_heatmaps(base_blocks: list[dict[str, Any]]) -> None:
|
|
29
|
+
# Use both start and count!
|
|
30
|
+
starts_and_counts = [(parse_start(b.get("id")), int( # type: ignore
|
|
31
|
+
b.get("count", 0))) for b in base_blocks]
|
|
32
|
+
max_start = max((start for start, _ in starts_and_counts), default=0)
|
|
33
|
+
for bs in PRECOMPUTED_BLOCK_SIZES:
|
|
34
|
+
if bs == 1:
|
|
35
|
+
ids = sorted(str(start) for start, _ in starts_and_counts)
|
|
36
|
+
PRECOMPUTED_HEATMAPS[bs] = {
|
|
37
|
+
"block_size": 1,
|
|
38
|
+
"ids": ids
|
|
39
|
+
}
|
|
40
|
+
else:
|
|
41
|
+
acc: dict[str, Any] = {}
|
|
42
|
+
for start, count in starts_and_counts:
|
|
43
|
+
bucket = (start // bs) * bs
|
|
44
|
+
key = f"{bucket}-{bucket + bs - 1}"
|
|
45
|
+
acc[key] = acc.get(key, 0) + count # Use the original count!
|
|
46
|
+
blocks = [{"id": k, "count": v} for k, v in acc.items()]
|
|
47
|
+
blocks.sort(key=lambda block: int(block["id"].split("-")[0]))
|
|
48
|
+
PRECOMPUTED_HEATMAPS[bs] = {
|
|
49
|
+
"metadata": {"block_size": bs, "max_id": max_start, "total_blocks": len(blocks)},
|
|
50
|
+
"blocks": blocks
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def create_app(graph_path: str, stats_path: str, heatmap_path: str) -> Flask:
|
|
55
|
+
if not os.path.isfile(graph_path):
|
|
56
|
+
raise FileNotFoundError(f"Graph file not found: {graph_path}")
|
|
57
|
+
|
|
58
|
+
if not os.path.isfile(stats_path):
|
|
59
|
+
raise FileNotFoundError(f"Stats file not found: {stats_path}")
|
|
60
|
+
|
|
61
|
+
with open(graph_path, "rb") as bf:
|
|
62
|
+
G: MultiDiGraph = pickle.load(bf)
|
|
63
|
+
|
|
64
|
+
with open(stats_path, "r", encoding='utf-8') as f:
|
|
65
|
+
try:
|
|
66
|
+
# Validate that the stats file is valid JSON
|
|
67
|
+
STATS_DATA = json.load(f)
|
|
68
|
+
# Simple validation to ensure keys exist
|
|
69
|
+
required_keys = [
|
|
70
|
+
"cycles", "top_direct_descendants", "top_indirect_descendants",
|
|
71
|
+
"top_direct_ancestors", "top_indirect_ancestors", "isolated_rules"
|
|
72
|
+
]
|
|
73
|
+
if not all(key in STATS_DATA for key in required_keys):
|
|
74
|
+
raise ValueError("Stats file is missing required keys.")
|
|
75
|
+
except (json.JSONDecodeError, ValueError) as e:
|
|
76
|
+
raise RuntimeError(f"Stats file '{stats_path}' is corrupted or invalid: {e}")
|
|
77
|
+
|
|
78
|
+
if not os.path.isfile(heatmap_path):
|
|
79
|
+
raise FileNotFoundError(f"Heatmap file not found: {heatmap_path}")
|
|
80
|
+
|
|
81
|
+
with open(heatmap_path, "r") as f:
|
|
82
|
+
try:
|
|
83
|
+
HEATMAP_DATA = json.load(f)
|
|
84
|
+
except json.JSONDecodeError:
|
|
85
|
+
raise RuntimeError(f"Heatmap file '{heatmap_path}' is corrupted.")
|
|
86
|
+
|
|
87
|
+
# Precompute heatmaps for all sizes on app startup
|
|
88
|
+
base_blocks = HEATMAP_DATA.get("blocks", [])
|
|
89
|
+
precompute_heatmaps(base_blocks)
|
|
90
|
+
|
|
91
|
+
app = Flask(__name__)
|
|
92
|
+
|
|
93
|
+
# ---------- shared helpers ----------
|
|
94
|
+
def serialize_node(nid: str, displayed: set[str] | None = None) -> dict[str, Any]:
|
|
95
|
+
attrs = G.nodes[nid]
|
|
96
|
+
if displayed is None:
|
|
97
|
+
# caller doesn't care about expandable computation
|
|
98
|
+
return {"id": nid, **{k: v for k, v in attrs.items() if k != "expandable"}}
|
|
99
|
+
expandable = any(child not in displayed for child in G.successors(nid))
|
|
100
|
+
return {
|
|
101
|
+
"id": nid,
|
|
102
|
+
**{k: v for k, v in attrs.items() if k != "expandable"},
|
|
103
|
+
"expandable": expandable,
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
def relation_type(u: str, v: str) -> str:
|
|
107
|
+
data = G.get_edge_data(u, v)
|
|
108
|
+
if not data:
|
|
109
|
+
return "unknown"
|
|
110
|
+
return next(iter(data.values()), {}).get("relation_type", "unknown") # type: ignore
|
|
111
|
+
|
|
112
|
+
def make_edge(u: str, v: str) -> dict[str, Any]:
|
|
113
|
+
return {"source": u, "target": v, "relation_type": relation_type(u, v)}
|
|
114
|
+
|
|
115
|
+
def error(message: str, code: int = 400) -> tuple[Response, int]:
|
|
116
|
+
return jsonify({"error": message}), code
|
|
117
|
+
|
|
118
|
+
def _handle_root(displayed: set[str]) -> Union[Response, tuple[Response, int]]:
|
|
119
|
+
root = "0"
|
|
120
|
+
if root not in G:
|
|
121
|
+
return error("Root node not found", 404)
|
|
122
|
+
children = list(G.successors(root))
|
|
123
|
+
nodes = [serialize_node(root, displayed)] + [serialize_node(c, displayed) for c in children]
|
|
124
|
+
edges = [{"source": root, "target": c, "relation_type": "no_parent"} for c in children]
|
|
125
|
+
return jsonify({"nodes": nodes, "edges": edges})
|
|
126
|
+
|
|
127
|
+
def _handle_batch(ids_param: str, displayed: set[str]) -> Union[Response, tuple[Response, int]]:
|
|
128
|
+
node_ids = [nid for nid in ids_param.split(",") if nid]
|
|
129
|
+
if not node_ids:
|
|
130
|
+
return jsonify({"nodes": [], "edges": []})
|
|
131
|
+
nodes = [serialize_node(nid, displayed) for nid in node_ids if nid in G]
|
|
132
|
+
all_relevant = set(node_ids) | displayed
|
|
133
|
+
sub_edges = [
|
|
134
|
+
make_edge(u, v)
|
|
135
|
+
for u, v in G.subgraph(all_relevant).edges()
|
|
136
|
+
if u in node_ids or v in node_ids
|
|
137
|
+
]
|
|
138
|
+
return jsonify({"nodes": nodes, "edges": sub_edges})
|
|
139
|
+
|
|
140
|
+
def _handle_search(node_id: str, displayed: set[str]) -> Union[Response, tuple[Response, int]]:
|
|
141
|
+
if node_id not in G:
|
|
142
|
+
return error(f"Node '{node_id}' not found", 404)
|
|
143
|
+
|
|
144
|
+
node_obj = serialize_node(node_id, displayed)
|
|
145
|
+
edges = []
|
|
146
|
+
for p in G.predecessors(node_id):
|
|
147
|
+
if p in displayed:
|
|
148
|
+
edges.append(make_edge(p, node_id))
|
|
149
|
+
for c in G.successors(node_id):
|
|
150
|
+
if c in displayed:
|
|
151
|
+
edges.append(make_edge(node_id, c))
|
|
152
|
+
return jsonify({"nodes": [node_obj], "edges": edges})
|
|
153
|
+
|
|
154
|
+
def _handle_single_node(node_id: str, neighbor_mode: str,
|
|
155
|
+
include_details: bool, displayed: set[str]) -> Union[Response, tuple[Response, int]]:
|
|
156
|
+
|
|
157
|
+
if node_id not in G:
|
|
158
|
+
return error(f"Node '{node_id}' not found", 404)
|
|
159
|
+
|
|
160
|
+
if include_details:
|
|
161
|
+
node_data = G.nodes[node_id]
|
|
162
|
+
parents = [
|
|
163
|
+
{"id": p, "relation_type": make_edge(p, node_id)["relation_type"]} for p in G.predecessors(node_id)
|
|
164
|
+
]
|
|
165
|
+
children = [
|
|
166
|
+
{"id": c, "relation_type": make_edge(node_id, c)["relation_type"]} for c in G.successors(node_id)
|
|
167
|
+
]
|
|
168
|
+
return jsonify({
|
|
169
|
+
"id": node_id,
|
|
170
|
+
"description": node_data.get("description"),
|
|
171
|
+
"groups": node_data.get("groups", []),
|
|
172
|
+
"level": node_data.get("level"),
|
|
173
|
+
"file": node_data.get("file"),
|
|
174
|
+
"parents": parents,
|
|
175
|
+
"children": children
|
|
176
|
+
})
|
|
177
|
+
|
|
178
|
+
nodes = [serialize_node(node_id, displayed)]
|
|
179
|
+
edges = []
|
|
180
|
+
if neighbor_mode in {"parents", "both"}:
|
|
181
|
+
for p in G.predecessors(node_id):
|
|
182
|
+
nodes.append(serialize_node(p, displayed))
|
|
183
|
+
edges.append(make_edge(p, node_id))
|
|
184
|
+
if neighbor_mode in {"children", "both"}:
|
|
185
|
+
for c in G.successors(node_id):
|
|
186
|
+
nodes.append(serialize_node(c, displayed))
|
|
187
|
+
edges.append(make_edge(node_id, c))
|
|
188
|
+
return jsonify({"nodes": nodes, "edges": edges})
|
|
189
|
+
|
|
190
|
+
# ---------- UI ----------
|
|
191
|
+
@app.route("/")
|
|
192
|
+
def index() -> str:
|
|
193
|
+
return render_template_string("""
|
|
194
|
+
<!DOCTYPE html>
|
|
195
|
+
<html lang="en">
|
|
196
|
+
<head>
|
|
197
|
+
<meta charset="UTF-8">
|
|
198
|
+
<title>Rule Graph Explorer</title>
|
|
199
|
+
<meta name="viewport" content="width=device-width, initial-scale=1">
|
|
200
|
+
<script src="https://d3js.org/d3.v7.min.js"></script>
|
|
201
|
+
<link rel="stylesheet" href="static/styles.css">
|
|
202
|
+
</head>
|
|
203
|
+
<body>
|
|
204
|
+
<div class="navbar">
|
|
205
|
+
<div class="navbar-left">
|
|
206
|
+
<div class="navbar-title">Interactive Rule Graph Explorer</div>
|
|
207
|
+
</div>
|
|
208
|
+
<div class="navbar-links">
|
|
209
|
+
<button id="showHeatmapBtn">Show Heatmap</button>
|
|
210
|
+
<button id="showStatsBtn">Show Stats</button>
|
|
211
|
+
<button id="resetZoom">Reset Zoom</button>
|
|
212
|
+
<button id="resetGraph">Reset Graph</button>
|
|
213
|
+
<button id="toggleFocusBtn">Toggle focus off</button>
|
|
214
|
+
<input type="text" id="searchBox" placeholder="Search Rule ID">
|
|
215
|
+
<button id="searchBtn">Search</button>
|
|
216
|
+
</div>
|
|
217
|
+
</div>
|
|
218
|
+
|
|
219
|
+
<div class="content">
|
|
220
|
+
<canvas></canvas>
|
|
221
|
+
</div>
|
|
222
|
+
|
|
223
|
+
<div id="detailsPanel" class="side-panel">
|
|
224
|
+
<button id="detailsCloseBtn" class="panel-close-btn">×</button>
|
|
225
|
+
<div id="detailsContent" class="panel-content">
|
|
226
|
+
<p>Click on a node to see its details.</p>
|
|
227
|
+
</div>
|
|
228
|
+
</div>
|
|
229
|
+
|
|
230
|
+
<div id="statsPanel" class="side-panel">
|
|
231
|
+
<button id="statsCloseBtn" class="panel-close-btn">×</button>
|
|
232
|
+
<div id="statsContent" class="panel-content">
|
|
233
|
+
<p>Statistics loading...</p>
|
|
234
|
+
</div>
|
|
235
|
+
</div>
|
|
236
|
+
|
|
237
|
+
<div id="heatmapModal" class="modal-overlay">
|
|
238
|
+
<div id="heatmapContent">
|
|
239
|
+
</div>
|
|
240
|
+
</div>
|
|
241
|
+
|
|
242
|
+
<div class="footer">
|
|
243
|
+
<p>Visit official Wazuh documentation for
|
|
244
|
+
<a href="https://documentation.wazuh.com/current/user-manual/ruleset/ruleset-xml-syntax/rules.html" target="_blank">
|
|
245
|
+
Wazuh Rule Syntax</a>.
|
|
246
|
+
</p>
|
|
247
|
+
<p>© 2025 <a href="https://zaferbalkan.com" target="_blank">Zafer Balkan</a></p>
|
|
248
|
+
<p>The brand <a href="https://wazuh.com/" target="_blank">Wazuh</a> and related marks,
|
|
249
|
+
emblems and images are registered trademarks of their respective owners.
|
|
250
|
+
</p>
|
|
251
|
+
</div>
|
|
252
|
+
|
|
253
|
+
<script src="static/tutorial.js" defer></script>
|
|
254
|
+
<script src="static/graph.js" defer></script>
|
|
255
|
+
</body>
|
|
256
|
+
</html>
|
|
257
|
+
""")
|
|
258
|
+
|
|
259
|
+
# ---------- Consolidated APIs ----------
|
|
260
|
+
@app.route("/api/nodes", methods=["GET"])
|
|
261
|
+
def nodes() -> Union[Response, tuple[Response, int]]:
|
|
262
|
+
"""
|
|
263
|
+
Dispatches requests to the appropriate handler based on query params.
|
|
264
|
+
"""
|
|
265
|
+
mode = request.args.get("mode", "").strip().lower()
|
|
266
|
+
node_id = request.args.get("id", "").strip()
|
|
267
|
+
ids_param = request.args.get("ids", "").strip()
|
|
268
|
+
neighbor_mode = request.args.get("neighbors", "").strip().lower() or ("children" if node_id else "none")
|
|
269
|
+
include_details = request.args.get("include", "").strip().lower() == "details"
|
|
270
|
+
displayed: set[str] = set(filter(None, request.args.get("displayed", "").split(",")))
|
|
271
|
+
|
|
272
|
+
if mode == "root":
|
|
273
|
+
return _handle_root(displayed)
|
|
274
|
+
|
|
275
|
+
if ids_param:
|
|
276
|
+
return _handle_batch(ids_param, displayed)
|
|
277
|
+
|
|
278
|
+
if mode == "search":
|
|
279
|
+
if not node_id:
|
|
280
|
+
return error("mode=search requires an 'id' parameter", 400)
|
|
281
|
+
return _handle_search(node_id, displayed)
|
|
282
|
+
|
|
283
|
+
if node_id:
|
|
284
|
+
return _handle_single_node(node_id, neighbor_mode, include_details, displayed)
|
|
285
|
+
|
|
286
|
+
return error("Specify one of: mode=root | mode=search&id=... | id=... | ids=...", 400)
|
|
287
|
+
|
|
288
|
+
@app.route("/api/edges", methods=["POST"])
|
|
289
|
+
def edges() -> Response:
|
|
290
|
+
"""
|
|
291
|
+
Unifies: /api/connections
|
|
292
|
+
Body: { "ids": ["a","b","c"] }
|
|
293
|
+
Returns all edges among provided ids.
|
|
294
|
+
"""
|
|
295
|
+
payload = request.get_json(silent=True) or {}
|
|
296
|
+
node_ids = payload.get("ids", [])
|
|
297
|
+
if not node_ids:
|
|
298
|
+
return jsonify({"nodes": [], "edges": []})
|
|
299
|
+
edges_list = [make_edge(u, v) for u, v in G.subgraph(node_ids).edges()]
|
|
300
|
+
return jsonify({"nodes": [], "edges": edges_list})
|
|
301
|
+
|
|
302
|
+
@app.route("/api/stats", methods=["GET"])
|
|
303
|
+
def stats() -> Response:
|
|
304
|
+
"""Serves the pre-calculated graph statistics."""
|
|
305
|
+
return jsonify(STATS_DATA)
|
|
306
|
+
|
|
307
|
+
@app.route("/api/heatmap", methods=["GET"])
|
|
308
|
+
def heatmap() -> Union[Response, tuple[Response, int]]:
|
|
309
|
+
bs = request.args.get("block_size", type=int)
|
|
310
|
+
if not bs or bs < 1:
|
|
311
|
+
bs = 1
|
|
312
|
+
|
|
313
|
+
if bs in PRECOMPUTED_HEATMAPS:
|
|
314
|
+
return jsonify(PRECOMPUTED_HEATMAPS[bs])
|
|
315
|
+
|
|
316
|
+
# Fallback for non-precomputed block sizes
|
|
317
|
+
starts_and_counts = [(parse_start(b.get("id")), int(
|
|
318
|
+
b.get("count", 0))) for b in base_blocks]
|
|
319
|
+
if bs == 1:
|
|
320
|
+
ids = sorted(str(start) for start, _ in starts_and_counts)
|
|
321
|
+
return jsonify({"block_size": 1, "ids": ids})
|
|
322
|
+
acc: dict[str, Any] = {}
|
|
323
|
+
for start, count in starts_and_counts:
|
|
324
|
+
bucket = (start // bs) * bs
|
|
325
|
+
key = f"{bucket}-{bucket + bs - 1}"
|
|
326
|
+
acc[key] = acc.get(key, 0) + count # Use the original count!
|
|
327
|
+
blocks = [{"id": k, "count": v} for k, v in acc.items()]
|
|
328
|
+
blocks.sort(key=lambda block: int(block["id"].split("-")[0]))
|
|
329
|
+
meta = {
|
|
330
|
+
"block_size": bs,
|
|
331
|
+
"max_id": max((start for start, _ in starts_and_counts), default=0),
|
|
332
|
+
"total_blocks": len(blocks)
|
|
333
|
+
}
|
|
334
|
+
return jsonify({"metadata": meta, "blocks": blocks})
|
|
335
|
+
|
|
336
|
+
return app
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: rulevis
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Interactive Wazuh Rule Graph Explorer
|
|
5
|
+
Author-email: Zafer Balkan <zafer@zaferbalkan.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/zbalkan/rulevis
|
|
8
|
+
Project-URL: Repository, https://github.com/zbalkan/rulevis
|
|
9
|
+
Project-URL: Issues, https://github.com/zbalkan/rulevis/issues
|
|
10
|
+
Keywords: wazuh,detection,security,monitoring,logging,siem
|
|
11
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
12
|
+
Classifier: Intended Audience :: Information Technology
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
20
|
+
Requires-Python: >=3.9
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Requires-Dist: Flask>=3.1.2
|
|
24
|
+
Requires-Dist: networkx>=3.5
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# RuleVis: Interactive Wazuh Rule Graph Explorer
|
|
28
|
+
|
|
29
|
+
RuleVis is a powerful analysis tool that transforms your Wazuh ruleset into a dynamic, interactive force-directed graph. It helps you visualize the complex relationships between rules, identify critical dependencies, discover structural issues, and analyze the distribution of your rule IDs.
|
|
30
|
+
|
|
31
|
+
This tool is designed for security engineers, SOC analysts, and Wazuh administrators who need to understand, maintain, and develop complex custom rulesets.
|
|
32
|
+
|
|
33
|
+

|
|
34
|
+
|
|
35
|
+
## Features
|
|
36
|
+
|
|
37
|
+
* **Interactive Graph Visualization:** Renders your entire ruleset as a graph using D3.js and HTML Canvas for high performance.
|
|
38
|
+
* **Dependency Analysis:** Clearly shows parent-child relationships (`if_sid`, `if_group`, etc.) with directed edges.
|
|
39
|
+
* **Node Expansion:** Interactively expand nodes to reveal their parent or child dependencies on demand.
|
|
40
|
+
* **Detailed Rule Information:** Click on any rule to see its full description, groups, and a complete list of its parents and children.
|
|
41
|
+
* **Powerful Search:** Instantly find and focus on any rule by its ID.
|
|
42
|
+
* **Graph Statistics Panel:** Get at-a-glance insights into your ruleset with statistics like:
|
|
43
|
+
* Top 5 rules with the most direct children (foundational rules).
|
|
44
|
+
* Top 5 rules with the highest impact (most total descendants).
|
|
45
|
+
* Top 5 rules with the most complex dependencies.
|
|
46
|
+
* A list of isolated rules.
|
|
47
|
+
* Cycles in the rules
|
|
48
|
+
* **Rule ID Heatmap:** Visualize the entire rule ID space from 0 to 100,000+ to see which ID ranges are heavily used and which are available for custom rules.
|
|
49
|
+
* **Keyboard Shortcuts:** Pause the simulation (`Space`), close panels (`Esc`), and more for an efficient workflow.
|
|
50
|
+
* **Focus:** By default, when a node is highlighted, any node except for the selected node and its neighbords are dimmed. That allows better focus minimizing visual complexity.
|
|
51
|
+
|
|
52
|
+
## The Problem It Solves
|
|
53
|
+
|
|
54
|
+
Wazuh's rule engine builds a complex, tree-like structure in memory. While powerful, this structure is invisible to the user. It can be difficult to:
|
|
55
|
+
|
|
56
|
+
* Understand the full impact of changing a single rule.
|
|
57
|
+
* Find redundant rules or overly complex dependency chains.
|
|
58
|
+
* Identify structural issues like circular dependencies, which can impact performance.
|
|
59
|
+
* Know which ID ranges are safe to use for new custom rules.
|
|
60
|
+
|
|
61
|
+
RuleVis makes these invisible structures visible, turning abstract XML files into a tangible, explorable map.
|
|
62
|
+
|
|
63
|
+
## Installation
|
|
64
|
+
|
|
65
|
+
### Using pipx (recommended)
|
|
66
|
+
|
|
67
|
+
`pipx install rulevis`
|
|
68
|
+
|
|
69
|
+
### Using pipx (for testing)
|
|
70
|
+
|
|
71
|
+
`pipx install rulevis`
|
|
72
|
+
|
|
73
|
+
### Using source
|
|
74
|
+
|
|
75
|
+
1. **Clone the repository:**
|
|
76
|
+
|
|
77
|
+
```shell
|
|
78
|
+
git clone https://github.com/zbalkan/rulevis.git
|
|
79
|
+
cd rulevis
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
2. **Create and activate a Python virtual environment:**
|
|
83
|
+
|
|
84
|
+
```shell
|
|
85
|
+
python -m venv .venv
|
|
86
|
+
source .venv/bin/activate
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Or for Windows
|
|
90
|
+
|
|
91
|
+
```shell
|
|
92
|
+
python -m venv .venv
|
|
93
|
+
source .venv\Scripts\activate.ps1
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
3. **Install dependencies:**
|
|
97
|
+
|
|
98
|
+
```shell
|
|
99
|
+
pip install -r requirements.txt
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## Usage
|
|
103
|
+
|
|
104
|
+
The tool is run from the command line. You must provide the path to the directory (or directories) containing your Wazuh rule XML files.
|
|
105
|
+
|
|
106
|
+
```shell
|
|
107
|
+
python src/rulevis.py --path /var/ossec/ruleset/rules,/var/ossec/etc/rules
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
**Arguments:**
|
|
111
|
+
|
|
112
|
+
* `--path, -p`: **(Required)** A comma-separated list of paths to your Wazuh rule directories. This should include both the default rules and your custom rules.
|
|
113
|
+
* `-h, --help`: Show the help message.
|
|
114
|
+
|
|
115
|
+
Once executed, the script will:
|
|
116
|
+
|
|
117
|
+
1. Parse all `.xml` files in the specified paths.
|
|
118
|
+
2. Build a graph model of the rule relationships.
|
|
119
|
+
3. Pre-calculate statistics and heatmap data.
|
|
120
|
+
4. Start a local web server.
|
|
121
|
+
5. Automatically open the tool in your default web browser.
|
|
122
|
+
|
|
123
|
+
## Key Features in Action
|
|
124
|
+
|
|
125
|
+
### Graph Statistics
|
|
126
|
+
|
|
127
|
+
Quickly identify the most important and complex rules in your entire ruleset. Click on any rule in the list to instantly navigate to it in the main graph.
|
|
128
|
+
|
|
129
|
+

|
|
130
|
+
|
|
131
|
+
### Rule ID Heatmap
|
|
132
|
+
|
|
133
|
+
Get a bird's-eye view of your rule ID landscape. Dark gray blocks are unused and available for your custom rules, while brighter red blocks indicate heavily populated ranges. This is invaluable for planning and organizing a large custom ruleset.
|
|
134
|
+
|
|
135
|
+

|
|
136
|
+
|
|
137
|
+
## Technical Overview
|
|
138
|
+
|
|
139
|
+
The project is composed of three main Python modules and a JavaScript frontend:
|
|
140
|
+
|
|
141
|
+
1. **`generator.py`:** Parses the Wazuh XML rule files and uses the `networkx` library to build a `MultiDiGraph` object representing the rule relationships. It saves this graph to a temporary file.
|
|
142
|
+
2. **`analyzer.py`:** Loads the graph file and uses `networkx` to perform complex calculations (descendants, ancestors, etc.). It pre-calculates the data needed for the Statistics Panel and the Rule ID Heatmap and saves them to temporary JSON files.
|
|
143
|
+
3. **`visualizer.py`:** A Flask web application that serves the frontend and provides a clean API for the visualization to fetch graph, stats, and heatmap data.
|
|
144
|
+
4. **`graph.js`:** The core frontend logic. It uses **D3.js** for the force simulation and user interactions, and renders the main graph to an **HTML Canvas** for high performance. The interactive heatmap is rendered using **SVG** for its superior event handling and styling capabilities.
|
|
145
|
+
|
|
146
|
+
Here’s a **ready-to-paste README subsection** that explains `rulevis` logging clearly and professionally for your users. It assumes the per-user setup you’ve implemented.
|
|
147
|
+
|
|
148
|
+
---
|
|
149
|
+
|
|
150
|
+
## Logging
|
|
151
|
+
|
|
152
|
+
`rulevis` automatically writes diagnostic and operational logs to a user-specific location. Logs are plain text encoded in UTF-8 and include timestamps, module names, and severity levels. The application creates its log directory if it does not exist.
|
|
153
|
+
|
|
154
|
+
| Platform | Log file location | Example path |
|
|
155
|
+
| --------------- | ------------------------------------------------------------------------------------------- | -------------------------------------------------------- |
|
|
156
|
+
| Windows | `%LocalAppData%\rulevis\Logs\rulevis.log` | `C:\Users\<user>\AppData\Local\rulevis\Logs\rulevis.log` |
|
|
157
|
+
| macOS | `~/Library/Logs/rulevis/rulevis.log` | `/Users/<user>/Library/Logs/rulevis/rulevis.log` |
|
|
158
|
+
| Linux / BSD | `$XDG_STATE_HOME/rulevis/rulevis.log` or fallback `~/.local/share/rulevis/logs/rulevis.log` | `/home/<user>/.local/state/rulevis/rulevis.log` |
|
|
159
|
+
|
|
160
|
+
`rulevis` follows the [XDG Base Directory specification](https://specifications.freedesktop.org/basedir-spec/latest/) on Unix-like systems and Windows conventions under `%LocalAppData%`.
|
|
161
|
+
|
|
162
|
+
The log file records informational messages, warnings, and errors emitted during execution. You can safely delete it; a new one will be created automatically on the next run.
|
|
163
|
+
|
|
164
|
+
## Notes
|
|
165
|
+
|
|
166
|
+
While the documentation defines <if_level> as another condition creating a parent-child relationship, it has not been used in any built-in rules. And as a personal choicem I decided to omit that deliberately.
|
|
167
|
+
|
|
168
|
+
There is another `if`, called `<if_fts>`, that is used for *first time seen* events, not creating a parent-child relationship. Theefore it is not mentioned.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
rulevis.py,sha256=EaXrzPmJcXXPTN8BG7hozkL9aVlN8hoPGf47J_9dNwY,7147
|
|
3
|
+
internal/analyzer.py,sha256=Cuz6TrKJ3EfJcwHzF5NeGidUqEoMNQ53HPmYmVFo5pg,7161
|
|
4
|
+
internal/generator.py,sha256=sZu54uF9tdJq3VR8tjsuyZk1nN6SkLTjFXdH_8rbwSo,7659
|
|
5
|
+
internal/visualizer.py,sha256=Fy32UtUDddZJa18rjYA9XSMcpZ1aIK5B52FvZMH4hKY,13822
|
|
6
|
+
rulevis-1.0.0.dist-info/licenses/LICENSE,sha256=ztkkhuJWbd7G1tZrFebtUecU8LbQzI6Pm9wRc69eXu8,1088
|
|
7
|
+
rulevis-1.0.0.dist-info/METADATA,sha256=gMf172IcL6xMeYiDme7x2Ew40Gf4EXTFO4Kuau2tJfk,8733
|
|
8
|
+
rulevis-1.0.0.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
9
|
+
rulevis-1.0.0.dist-info/entry_points.txt,sha256=2_1Zn-V0eVSIqleW3izMY0mp4iidkD8Rb0-WrGLxSwM,41
|
|
10
|
+
rulevis-1.0.0.dist-info/top_level.txt,sha256=9ttBbMPxCJIx9wv_G5oYERYLOuOHIaq_f6-Afqk8kMM,26
|
|
11
|
+
rulevis-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Zafer Balkan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
rulevis.py
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import logging
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
import sys
|
|
8
|
+
import tempfile
|
|
9
|
+
import webbrowser
|
|
10
|
+
from threading import Timer
|
|
11
|
+
from typing import Final
|
|
12
|
+
|
|
13
|
+
from internal.analyzer import Analyzer
|
|
14
|
+
from internal.generator import GraphGenerator
|
|
15
|
+
from internal.visualizer import create_app
|
|
16
|
+
|
|
17
|
+
APP_NAME: Final[str] = 'rulevis'
|
|
18
|
+
DESCRIPTION: Final[str] = f"{APP_NAME} is a Wazuh rule visualization tool."
|
|
19
|
+
ENCODING: Final[str] = "utf-8"
|
|
20
|
+
|
|
21
|
+
# Precompiled regex to remove ANSI color/control sequences
|
|
22
|
+
ANSI_ESCAPE_RE: re.Pattern[str] = re.compile(
|
|
23
|
+
r"""
|
|
24
|
+
(?: # Non-capturing group for all patterns
|
|
25
|
+
\x1B\[ # ESC [ (CSI)
|
|
26
|
+
[0-?]*[ -/]*[@-~] # Parameter bytes + intermediate + final byte
|
|
27
|
+
| # OR
|
|
28
|
+
\x1B[@-Z\\-_] # 2-byte sequences
|
|
29
|
+
| # OR
|
|
30
|
+
\x1B\][^\x07]*(?:\x07|\x1B\\) # OSC sequences
|
|
31
|
+
| # OR literal representations (\x1b, <0x1b>)
|
|
32
|
+
(?:\\x1[bB]|\<0x1[bB]\>)(?:\[[0-?]*[ -/]*[@-~])?
|
|
33
|
+
)
|
|
34
|
+
""",
|
|
35
|
+
re.VERBOSE,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class CustomFileHandler(logging.FileHandler):
|
|
40
|
+
"""FileHandler that strips all escape sequences and representations."""
|
|
41
|
+
|
|
42
|
+
def emit(self, record) -> None:
|
|
43
|
+
# Escape ANSI Color Sequences
|
|
44
|
+
record.msg = ANSI_ESCAPE_RE.sub('', str(record.msg))
|
|
45
|
+
|
|
46
|
+
# Rename source
|
|
47
|
+
record.name = APP_NAME
|
|
48
|
+
|
|
49
|
+
super().emit(record)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Rulevis():
|
|
53
|
+
|
|
54
|
+
def __init__(self, paths: list[str]) -> None:
|
|
55
|
+
self.graph_path: str = tempfile.NamedTemporaryFile(delete=False).name
|
|
56
|
+
logging.info(f"Temporary graph file created at {self.graph_path}")
|
|
57
|
+
|
|
58
|
+
self.stats_path: str = tempfile.NamedTemporaryFile(delete=False).name
|
|
59
|
+
logging.info(f"Temporary stats file created at {self.stats_path}")
|
|
60
|
+
|
|
61
|
+
self.heatmap_path: str = tempfile.NamedTemporaryFile(delete=False).name
|
|
62
|
+
logging.info(f"Temporary heatmap file created at {self.heatmap_path}")
|
|
63
|
+
|
|
64
|
+
self.__validate_paths(paths)
|
|
65
|
+
self.__paths: list[str] = paths
|
|
66
|
+
|
|
67
|
+
def __del__(self) -> None:
|
|
68
|
+
try:
|
|
69
|
+
if hasattr(self, 'graph_path') and os.path.exists(self.graph_path):
|
|
70
|
+
os.remove(self.graph_path)
|
|
71
|
+
logging.info(
|
|
72
|
+
f"Temporary graph file {self.graph_path} deleted.")
|
|
73
|
+
except Exception as e:
|
|
74
|
+
logging.error(f"Error deleting temporary graph file: {e}")
|
|
75
|
+
|
|
76
|
+
try:
|
|
77
|
+
if hasattr(self, 'stats_path') and os.path.exists(self.stats_path):
|
|
78
|
+
os.remove(self.stats_path)
|
|
79
|
+
logging.info(
|
|
80
|
+
f"Temporary stats file {self.stats_path} deleted.")
|
|
81
|
+
except Exception as e:
|
|
82
|
+
logging.error(f"Error deleting temporary stats file: {e}")
|
|
83
|
+
|
|
84
|
+
try:
|
|
85
|
+
if hasattr(self, 'heatmap_path') and os.path.exists(self.heatmap_path):
|
|
86
|
+
os.remove(self.heatmap_path)
|
|
87
|
+
logging.info(
|
|
88
|
+
f"Temporary heatmap file {self.heatmap_path} deleted.")
|
|
89
|
+
except Exception as e:
|
|
90
|
+
logging.error(f"Error deleting temporary heatmap file: {e}")
|
|
91
|
+
|
|
92
|
+
def run(self) -> None:
|
|
93
|
+
self.__generate_graph()
|
|
94
|
+
self.__generate_stats()
|
|
95
|
+
self.__run_flask_app()
|
|
96
|
+
|
|
97
|
+
def __validate_paths(self, paths: list[str]) -> None:
|
|
98
|
+
for path in paths:
|
|
99
|
+
if not os.path:
|
|
100
|
+
logging.error(f"Invalid directory path: {path}", exc_info=True)
|
|
101
|
+
print(f"Error: Invalid directory path: {path}")
|
|
102
|
+
sys.exit(1)
|
|
103
|
+
|
|
104
|
+
def __generate_graph(self) -> None:
|
|
105
|
+
logging.info("Generating rule graph...")
|
|
106
|
+
generator = GraphGenerator(paths=self.__paths, graph_file=self.graph_path)
|
|
107
|
+
generator.build_graph_from_xml()
|
|
108
|
+
generator.save_graph()
|
|
109
|
+
logging.info("Graph generation complete.")
|
|
110
|
+
|
|
111
|
+
def __generate_stats(self) -> None:
|
|
112
|
+
logging.info("Generating rule stats...")
|
|
113
|
+
analyzer = Analyzer(self.graph_path)
|
|
114
|
+
analyzer.write_to_json(self.stats_path, self.heatmap_path)
|
|
115
|
+
logging.info("Stats generation complete.")
|
|
116
|
+
|
|
117
|
+
def __run_flask_app(self) -> None:
|
|
118
|
+
app = create_app(self.graph_path, self.stats_path, self.heatmap_path)
|
|
119
|
+
logging.info("Starting Flask app...")
|
|
120
|
+
Timer(1, self.__open_browser).start()
|
|
121
|
+
app.run(debug=True, use_reloader=False)
|
|
122
|
+
|
|
123
|
+
def __open_browser(self, ) -> None:
|
|
124
|
+
new_url = 'http://localhost:5000/'
|
|
125
|
+
|
|
126
|
+
if (webbrowser.get().name != 'gio'):
|
|
127
|
+
webbrowser.open_new(new_url)
|
|
128
|
+
|
|
129
|
+
print(f"Access the app over {new_url}")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def main() -> None:
|
|
133
|
+
parser = argparse.ArgumentParser(prog=APP_NAME, description=DESCRIPTION)
|
|
134
|
+
if len(sys.argv) == 1:
|
|
135
|
+
parser.print_help()
|
|
136
|
+
sys.exit(1)
|
|
137
|
+
|
|
138
|
+
parser.add_argument("--path", "-p", dest="path", required=True, type=str,
|
|
139
|
+
help="Path to the Wazuh rule directories. Comma-separated multiple paths are accepted.")
|
|
140
|
+
|
|
141
|
+
args: argparse.Namespace = parser.parse_args()
|
|
142
|
+
paths: list[str] = [p for p in str(args.path).split(',') if p != '']
|
|
143
|
+
logging.info(f"Paths: {paths}")
|
|
144
|
+
|
|
145
|
+
rulevis = Rulevis(paths)
|
|
146
|
+
rulevis.run()
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _get_log_path() -> str:
|
|
150
|
+
"""
|
|
151
|
+
Return a per-user log file path appropriate for Windows, Linux, and macOS.
|
|
152
|
+
Uses only os and sys modules.
|
|
153
|
+
"""
|
|
154
|
+
# Determine base OS type
|
|
155
|
+
if os.name == "nt": # Windows
|
|
156
|
+
base_dir = os.getenv(
|
|
157
|
+
"LOCALAPPDATA", os.path.expanduser("~\\AppData\\Local"))
|
|
158
|
+
log_dir = os.path.join(base_dir, APP_NAME, "Logs")
|
|
159
|
+
|
|
160
|
+
elif sys.platform == "darwin": # macOS
|
|
161
|
+
log_dir = os.path.expanduser(f"~/Library/Logs/{APP_NAME}")
|
|
162
|
+
|
|
163
|
+
else: # Linux / other Unix-like
|
|
164
|
+
xdg_state_home = os.getenv(
|
|
165
|
+
"XDG_STATE_HOME", os.path.expanduser("~/.local/state"))
|
|
166
|
+
log_dir = os.path.join(xdg_state_home, APP_NAME)
|
|
167
|
+
if not os.access(os.path.dirname(log_dir), os.W_OK):
|
|
168
|
+
log_dir = os.path.expanduser(f"~/.local/share/{APP_NAME}/logs")
|
|
169
|
+
|
|
170
|
+
os.makedirs(log_dir, exist_ok=True)
|
|
171
|
+
return os.path.abspath(os.path.join(log_dir, f"{APP_NAME}.log"))
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
if __name__ == "__main__":
|
|
175
|
+
try:
|
|
176
|
+
handler = CustomFileHandler(_get_log_path(), encoding=ENCODING)
|
|
177
|
+
|
|
178
|
+
logging.basicConfig(handlers=[handler],
|
|
179
|
+
format='%(asctime)s:%(name)s:%(levelname)s:%(message)s',
|
|
180
|
+
datefmt="%Y-%m-%dT%H:%M:%S%z",
|
|
181
|
+
level=logging.INFO)
|
|
182
|
+
|
|
183
|
+
excepthook = logging.error
|
|
184
|
+
logging.info('Starting')
|
|
185
|
+
main()
|
|
186
|
+
logging.info('Exiting.')
|
|
187
|
+
except KeyboardInterrupt:
|
|
188
|
+
print('Cancelled by user.')
|
|
189
|
+
logging.info('Cancelled by user.')
|
|
190
|
+
try:
|
|
191
|
+
sys.exit(0)
|
|
192
|
+
except SystemExit:
|
|
193
|
+
os._exit(0)
|
|
194
|
+
except Exception as ex:
|
|
195
|
+
print('ERROR: ' + str(ex))
|
|
196
|
+
logging.error(str(ex), exc_info=True)
|
|
197
|
+
try:
|
|
198
|
+
sys.exit(1)
|
|
199
|
+
except SystemExit:
|
|
200
|
+
os._exit(1)
|