parse_ioc 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
parse_ioc.py
ADDED
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
import ipaddress
|
|
2
|
+
import json
|
|
3
|
+
import re
|
|
4
|
+
import sys
|
|
5
|
+
import tomllib
|
|
6
|
+
from dataclasses import dataclass, field, asdict
|
|
7
|
+
from typing import Any, BinaryIO, List, TextIO
|
|
8
|
+
from urllib.parse import urlparse
|
|
9
|
+
# https://github.com/JoshData/python-email-validator
|
|
10
|
+
#from email_validator import validate_email
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
# TODO
|
|
14
|
+
# imphash, ssdeep, ja3+
|
|
15
|
+
# filenames from a filepath
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
# uv run -m tests.test -f tests/ioc_examples.txt
|
|
19
|
+
# uv run ioc_parse.py
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class ParseIOC:
|
|
24
|
+
"""Determines the type of an indicator of compromise (IOC), such as ipv4, domain, MD5, etc."""
|
|
25
|
+
ioc: str
|
|
26
|
+
ioc_type: str | None = None
|
|
27
|
+
# using field(default_factory=list) for mutable default lists
|
|
28
|
+
extra: list[dict[str, Any]] = field(default_factory=list)
|
|
29
|
+
|
|
30
|
+
def __post_init__(self) -> None:
|
|
31
|
+
"""Runs cleaning and processing automatically after __init__.
|
|
32
|
+
|
|
33
|
+
Runs .strip() on the IOC, checks for Punycode (IDNA), and then processes the IOC.
|
|
34
|
+
"""
|
|
35
|
+
self.ioc = self.ioc.strip()
|
|
36
|
+
self._check_punycode()
|
|
37
|
+
self._process()
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def to_dict(self) -> dict[str, Any]:
|
|
41
|
+
"""Returns a dictionary of the class attributes."""
|
|
42
|
+
out = asdict(self)
|
|
43
|
+
if not self.extra:
|
|
44
|
+
out.pop("extra")
|
|
45
|
+
return out
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def to_json(self) -> str:
|
|
49
|
+
"""Returns a JSON object of the class attributes."""
|
|
50
|
+
out = asdict(self)
|
|
51
|
+
if not self.extra:
|
|
52
|
+
out.pop("extra")
|
|
53
|
+
return json.dumps(out)
|
|
54
|
+
|
|
55
|
+
def _preclean(self) -> None:
|
|
56
|
+
"""Suppresses multiple forward and backward slashes. Does not affect _check_domain()."""
|
|
57
|
+
if "//" in self.ioc:
|
|
58
|
+
self.ioc = self.ioc.replace("//", "/")
|
|
59
|
+
self._preclean()
|
|
60
|
+
elif "\\\\" in self.ioc:
|
|
61
|
+
self.ioc = self.ioc.replace("\\\\", "\\")
|
|
62
|
+
self._preclean()
|
|
63
|
+
self.ioc = self.ioc.lower().replace("hxxp", "http").replace("[://]", "://").replace("**.", "").replace("*.", "")
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
def _check_punycode(self) -> str:
|
|
67
|
+
"""International punycode checks; searches each character individually and decodes the IOC if needed."""
|
|
68
|
+
is_it_punycode = False
|
|
69
|
+
for char in self.ioc:
|
|
70
|
+
#if not re.search("[A-Za-z0-9.-]", char):
|
|
71
|
+
if ord(char) > 127:
|
|
72
|
+
is_it_punycode = True
|
|
73
|
+
self.ioc = self.ioc.encode("idna").decode().strip()
|
|
74
|
+
break
|
|
75
|
+
|
|
76
|
+
def _check_file(self) -> bool:
|
|
77
|
+
"""Checks if the IOC is a file path for either Windows or Linux.
|
|
78
|
+
|
|
79
|
+
Handles absolute paths, relative paths, and paths with alternative data streams. AI helped write both large regex statements in this function.
|
|
80
|
+
"""
|
|
81
|
+
# remove quotes common in Windows
|
|
82
|
+
ioc = self.ioc.strip('"')
|
|
83
|
+
# Windows regex
|
|
84
|
+
# check for drive letter (C:\ etc), backslashes, alternative data streams, extensions at the end of the path
|
|
85
|
+
windows_path_pattern = re.compile(
|
|
86
|
+
r"^(?:[a-zA-Z]:(?:\\|/)|\\\\|/)?(?:[^<>:\"/\\|?*\n]+\\?|[^<>:\"/\\|?*\n]+/)*[^<>:\"/\\|?*\n]+(?:\.[a-zA-Z0-9]+)?(?::[^<>:\"/\\|?*\n]+)?$")
|
|
87
|
+
# Linux regex
|
|
88
|
+
# check for leading forward slashes (/home/user), forward slashes as path separators, file extension at the end
|
|
89
|
+
linux_path_pattern = re.compile(
|
|
90
|
+
r"^(?:/|~)?(?:(?:[^<>:\"/\\|?*\n]+/)*[^<>:\"/\\|?*\n]+)?(?:\.[a-zA-Z0-9]+)?$")
|
|
91
|
+
# Windows check
|
|
92
|
+
if (re.search(r"^[a-zA-Z]:\\", ioc) or "\\" in ioc or ":" in ioc) and windows_path_pattern.search(ioc):
|
|
93
|
+
# additional interesting extension paths
|
|
94
|
+
#if re.search(r"\.(exe|dll|txt|pdf|docx|zip|py|sh|bat|jpg|png|mshta)$", ioc, re.IGNORECASE):
|
|
95
|
+
self.ioc_type = "file_path_windows"
|
|
96
|
+
return True
|
|
97
|
+
# Linux check
|
|
98
|
+
elif "/" in ioc and linux_path_pattern.search(ioc):
|
|
99
|
+
#if re.search(r"\.(sh|py|txt|conf|log|bin|deb|rpm|tar\.gz)$", ioc, re.IGNORECASE) or ioc.startswith('/'):
|
|
100
|
+
self.ioc_type = "file_path_linux"
|
|
101
|
+
return True
|
|
102
|
+
# False if not a path
|
|
103
|
+
return False
|
|
104
|
+
|
|
105
|
+
def _check_sha512(self) -> bool:
|
|
106
|
+
"""Check for SHA-512 hash, only based on string length."""
|
|
107
|
+
if re.search("^[A-Za-z0-9]{128}$", self.ioc):
|
|
108
|
+
self.ioc_type = "sha512"
|
|
109
|
+
return True
|
|
110
|
+
|
|
111
|
+
def _check_sha256(self) -> bool:
|
|
112
|
+
"""Check for SHA-256 hash, only based on string length."""
|
|
113
|
+
if re.search("^[A-Za-z0-9]{64}$", self.ioc):
|
|
114
|
+
self.ioc_type = "sha256"
|
|
115
|
+
return True
|
|
116
|
+
|
|
117
|
+
def _check_md5(self) -> bool:
|
|
118
|
+
"""Check for MD5 hash, only based on string length."""
|
|
119
|
+
if re.search("^[A-Za-z0-9]{32}$", self.ioc):
|
|
120
|
+
self.ioc_type = "md5"
|
|
121
|
+
return True
|
|
122
|
+
|
|
123
|
+
def _check_email(self) -> bool:
|
|
124
|
+
"""Loosely check for email addresses based on regex."""
|
|
125
|
+
# future: use email_validator like this: validate_email(self.ioc, check_deliverability=False)
|
|
126
|
+
if re.search(r"^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$", self.ioc):
|
|
127
|
+
self.ioc_type = "email"
|
|
128
|
+
try:
|
|
129
|
+
ed = self.ioc.split("@")[1]
|
|
130
|
+
self.extra.append({"ioc":ed, "ioc_type":"domain"})
|
|
131
|
+
except:
|
|
132
|
+
return False
|
|
133
|
+
return True
|
|
134
|
+
|
|
135
|
+
def _check_ip(self) -> bool:
|
|
136
|
+
"""Checks for an IPv4 or IPv6 address or network."""
|
|
137
|
+
try:
|
|
138
|
+
addr = ipaddress.ip_network(self.ioc, strict=False)
|
|
139
|
+
# start with IPv4, assume it's a network unless /32
|
|
140
|
+
if isinstance(addr, ipaddress.IPv4Network):
|
|
141
|
+
self.ioc_type = "ipv4_network"
|
|
142
|
+
if addr.prefixlen == 32:
|
|
143
|
+
self.ioc_type = "ipv4"
|
|
144
|
+
return True
|
|
145
|
+
# next assume IPv6 network unless /128
|
|
146
|
+
elif isinstance(addr, ipaddress.IPv6Network):
|
|
147
|
+
self.ioc_type = "ipv6_network"
|
|
148
|
+
if addr.prefixlen == 128:
|
|
149
|
+
self.ioc_type = "ipv6"
|
|
150
|
+
return True
|
|
151
|
+
return True
|
|
152
|
+
except ValueError:
|
|
153
|
+
return False
|
|
154
|
+
|
|
155
|
+
def _check_domain(self) -> bool:
|
|
156
|
+
"""Checks if the IOC is a URL or domain."""
|
|
157
|
+
try:
|
|
158
|
+
# create local semi-cleaned version of self.ioc
|
|
159
|
+
# several .strip() because of edge cases with malformed strings needing safeguards
|
|
160
|
+
url = self.ioc.lower().lstrip("htxps:/")
|
|
161
|
+
# add schema to urlparse properly parses netloc (domain)
|
|
162
|
+
url = "https://" + url
|
|
163
|
+
parsed = urlparse(url)
|
|
164
|
+
if not parsed.hostname:
|
|
165
|
+
return False
|
|
166
|
+
if parsed.username:
|
|
167
|
+
creds = parsed.username
|
|
168
|
+
if parsed.password:
|
|
169
|
+
creds += f":{parsed.password}"
|
|
170
|
+
self.extra.append({"ioc":creds, "ioc_type":"credentials"})
|
|
171
|
+
if parsed.port:
|
|
172
|
+
# possible future feature, to optionally remove "common" high ports via argument
|
|
173
|
+
#if port > 1024 and port != [5353, 8000, 8080, 8443]:
|
|
174
|
+
self.extra.append({"ioc":parsed.port, "ioc_type":"port"})
|
|
175
|
+
# final check for if the domain is an IP
|
|
176
|
+
try:
|
|
177
|
+
# anything except an IP will trigger exception
|
|
178
|
+
ipaddress.ip_address(parsed.hostname)
|
|
179
|
+
# if no exception, update self.ioc and run _check_ip()
|
|
180
|
+
self.ioc = parsed.hostname
|
|
181
|
+
self._check_ip()
|
|
182
|
+
#print("A"*20, parsed.hostname)
|
|
183
|
+
return True
|
|
184
|
+
except ValueError:
|
|
185
|
+
self.ioc = parsed.hostname
|
|
186
|
+
self.ioc_type = "domain"
|
|
187
|
+
return True
|
|
188
|
+
except Exception as e:
|
|
189
|
+
#print("URL_EXCEPTION", str(e))
|
|
190
|
+
return False
|
|
191
|
+
|
|
192
|
+
def _process(self) -> None:
|
|
193
|
+
"""Main processing logic."""
|
|
194
|
+
self._preclean()
|
|
195
|
+
# if adding more check functions, add them here without ()
|
|
196
|
+
checks = [
|
|
197
|
+
self._check_sha512,
|
|
198
|
+
self._check_sha256,
|
|
199
|
+
self._check_md5,
|
|
200
|
+
self._check_email,
|
|
201
|
+
self._check_ip,
|
|
202
|
+
self._check_file,
|
|
203
|
+
self._check_domain
|
|
204
|
+
]
|
|
205
|
+
# check the checks
|
|
206
|
+
for check in checks:
|
|
207
|
+
# add () to "check", since it's a function name
|
|
208
|
+
# if a check function returns True, return
|
|
209
|
+
if check():
|
|
210
|
+
return
|
|
211
|
+
# fallback
|
|
212
|
+
if not self.ioc_type:
|
|
213
|
+
self.ioc_type = "unknown"
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def parse_multi(input_object: List[str] | TextIO, mode="combined") -> dict[str, Any] | List[Any]:
|
|
217
|
+
"""Parse list or file of indicators into a large combined structure or individual lines.
|
|
218
|
+
|
|
219
|
+
Args:
|
|
220
|
+
input_object: list of IOCs, or path to a text file containing IOCs
|
|
221
|
+
mode: "combined" produces a single returned object, "single" produces individual lines
|
|
222
|
+
"""
|
|
223
|
+
for_assembler = []
|
|
224
|
+
# read a list
|
|
225
|
+
if isinstance(input_object, list):
|
|
226
|
+
for item in input_object:
|
|
227
|
+
#item = repr(item)
|
|
228
|
+
if not item.startswith("#") and not item.startswith("="):
|
|
229
|
+
ioc = ParseIOC(item)
|
|
230
|
+
for_assembler.append(ioc.to_dict)
|
|
231
|
+
# read a file
|
|
232
|
+
else:
|
|
233
|
+
try:
|
|
234
|
+
with open(input_object) as input_file:
|
|
235
|
+
for line in input_file:
|
|
236
|
+
if not line.startswith("#") and not line.startswith("="):
|
|
237
|
+
ioc = ParseIOC(line)
|
|
238
|
+
for_assembler.append(ioc.to_dict)
|
|
239
|
+
except Exception as e:
|
|
240
|
+
print(str(e))
|
|
241
|
+
return for_assembler
|
|
242
|
+
# dict
|
|
243
|
+
if mode == "combined":
|
|
244
|
+
return _assembler(for_assembler)
|
|
245
|
+
# return list of dicts
|
|
246
|
+
elif mode == "single":
|
|
247
|
+
return for_assembler
|
|
248
|
+
#print(json.dumps(a, indent=4))
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _field_mapper(iocs: dict, map_file_name: str, chunk_size: int=0) -> dict:
|
|
252
|
+
"""Match field mapping to provided ioc_parse output.
|
|
253
|
+
|
|
254
|
+
This is the worker function that maps indicators to SIEM field names.
|
|
255
|
+
|
|
256
|
+
Args:
|
|
257
|
+
ioc: the post-processed IOCs in a dictionary
|
|
258
|
+
map_file_name: path to the TOML mapping file
|
|
259
|
+
chunk_size (unused): split IOCs into chunks of n-size
|
|
260
|
+
"""
|
|
261
|
+
try:
|
|
262
|
+
with open(map_file_name, "rb") as f:
|
|
263
|
+
map_data = tomllib.load(f)
|
|
264
|
+
except Exception as e:
|
|
265
|
+
return str(e)+": cannot open TOML field mapping file."
|
|
266
|
+
# inner chunking function
|
|
267
|
+
# chunking should be a user responsibility; may add helper function
|
|
268
|
+
def _inner_chunk(l, n) -> list:
|
|
269
|
+
"""Split a large list into smaller lists of n items, with one list of remainders."""
|
|
270
|
+
for i in range(0, len(l), n):
|
|
271
|
+
yield l[i:i+n]
|
|
272
|
+
out = {}
|
|
273
|
+
if map_data.get("field_map"):
|
|
274
|
+
# the type of field, ipv4, domain, etc
|
|
275
|
+
for field_type in map_data["field_map"]:
|
|
276
|
+
# if the key (ioc type) exists in the parsed IOCs
|
|
277
|
+
if iocs.get(field_type):
|
|
278
|
+
# each field we want in the final output, from map toml
|
|
279
|
+
for field_name in map_data["field_map"][field_type]:
|
|
280
|
+
# make a key if not already there
|
|
281
|
+
if not out.get(field_name):
|
|
282
|
+
out[field_name] = []
|
|
283
|
+
# extend the IOCs into the output mapping via shared key
|
|
284
|
+
out[field_name].extend(iocs[field_type])
|
|
285
|
+
else:
|
|
286
|
+
return "error: no key field_map in loaded TOML file"
|
|
287
|
+
# chunking should be a user responsibility; may add helper function
|
|
288
|
+
#if chunk_size > 0:
|
|
289
|
+
# for key in out:
|
|
290
|
+
# out[key]["values"] = list(_inner_chunk(out[key]["values"], chunk_size))
|
|
291
|
+
# out[key]["chunk_size"] = len(out[key]["values"])
|
|
292
|
+
#
|
|
293
|
+
#print("="*50)
|
|
294
|
+
#print(json.dumps(out, indent=4))
|
|
295
|
+
#mapper(map_data["field_map"], ioc_data, chunk_size=2)
|
|
296
|
+
return out
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def map_fields(input_object: List[str] | TextIO, map_file_name: str) -> dict:
|
|
300
|
+
"""Callable function to map IOCs to the fields in the TOML configuration.
|
|
301
|
+
|
|
302
|
+
This is the friendly entrypoint to the field-to-mapping functions.
|
|
303
|
+
|
|
304
|
+
Args:
|
|
305
|
+
input_object: list of IOCs, or path to a text file containing IOCs
|
|
306
|
+
map_file_name: path to the TOML mapping file
|
|
307
|
+
"""
|
|
308
|
+
parsed_iocs_combined = parse_multi(input_object, mode="combined")
|
|
309
|
+
return _field_mapper(parsed_iocs_combined, map_file_name)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _assembler(l: List[str]) -> dict:
|
|
313
|
+
"""Combines all IOCs, including nested "extra" entries, into one dictionary.
|
|
314
|
+
|
|
315
|
+
Use this dictionary with mappings to field names for automated SIEM or database queries.
|
|
316
|
+
|
|
317
|
+
Args:
|
|
318
|
+
l: list of IOCs to be assembled into the combined structure.
|
|
319
|
+
"""
|
|
320
|
+
out = {}
|
|
321
|
+
def inner(d):
|
|
322
|
+
if d["ioc_type"] not in out:
|
|
323
|
+
out[d["ioc_type"]] = []
|
|
324
|
+
out[d["ioc_type"]].append(d["ioc"])
|
|
325
|
+
else:
|
|
326
|
+
out[d["ioc_type"]].append(d["ioc"])
|
|
327
|
+
for d in l:
|
|
328
|
+
inner(d)
|
|
329
|
+
if d.get("extra"):
|
|
330
|
+
for dd in d["extra"]:
|
|
331
|
+
inner(dd)
|
|
332
|
+
# clean output
|
|
333
|
+
for k,v in out.items():
|
|
334
|
+
out[k] = list(set(v))
|
|
335
|
+
#print(json.dumps(out, indent=4))
|
|
336
|
+
return out
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
if __name__ == "__main__":
|
|
340
|
+
#
|
|
341
|
+
ioc_list = ["bob@email.local", "website.local", "https://anotherwebsite.local:9443", "https://username:password@securewebsite.local:8443", "192.168.1.1", "192.168.1.0/24", "https://192.168.20.20/bad.txt", "NHKやさしいことばニュース.com"]
|
|
342
|
+
#
|
|
343
|
+
# categorize a single IOC
|
|
344
|
+
print(" categorize a single indicator ".center(80, "="))
|
|
345
|
+
parsed_indicator = ParseIOC("https://192.168.20.20/bad.txt")
|
|
346
|
+
print(".to_dict:", type(parsed_indicator.to_dict), parsed_indicator.to_dict)
|
|
347
|
+
print(".to_json:", type(parsed_indicator.to_json), parsed_indicator.to_json)
|
|
348
|
+
#
|
|
349
|
+
# THIS IS THE FIRST PRIMARY OUTPUT
|
|
350
|
+
# parse a list or file IOCs into a large dict or json structure using parse_multi()
|
|
351
|
+
print(" parse a list or file of IOCs into a combined structure ".center(80, "="))
|
|
352
|
+
iocs_from_list = parse_multi(ioc_list, mode="combined")
|
|
353
|
+
iocs_from_file = parse_multi("ioc_examples.txt", mode="combined")
|
|
354
|
+
print(json.dumps(iocs_from_list, indent=4))
|
|
355
|
+
print(json.dumps(iocs_from_file, indent=4))
|
|
356
|
+
#
|
|
357
|
+
# alternatively parse a file line by line (mode=single) into dicts, instead of calling ParseIOC(indicator)
|
|
358
|
+
print(" yield dictionaries from a list or file, instead of a combined structure ".center(80, "="))
|
|
359
|
+
for item in parse_multi("ioc_examples.txt", mode="single"):
|
|
360
|
+
print(item)
|
|
361
|
+
#
|
|
362
|
+
# THIS IS THE SECOND PRIMARY OUTPUT
|
|
363
|
+
# use a field map to stage siem or database queries
|
|
364
|
+
print(" provide IOC (file or list) and TOML config (path) map_fields() ".center(80, "="))
|
|
365
|
+
#m = map_fields(ioc_list, "map_ecs.toml")
|
|
366
|
+
m = map_fields("ioc_examples.txt", "map_ecs.toml")
|
|
367
|
+
print(json.dumps(m, indent=4))
|