parse_ioc 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
parse_ioc.py ADDED
@@ -0,0 +1,367 @@
1
+ import ipaddress
2
+ import json
3
+ import re
4
+ import sys
5
+ import tomllib
6
+ from dataclasses import dataclass, field, asdict
7
+ from typing import Any, BinaryIO, List, TextIO
8
+ from urllib.parse import urlparse
9
+ # https://github.com/JoshData/python-email-validator
10
+ #from email_validator import validate_email
11
+
12
+
13
+ # TODO
14
+ # imphash, ssdeep, ja3+
15
+ # filenames from a filepath
16
+
17
+
18
+ # uv run -m tests.test -f tests/ioc_examples.txt
19
+ # uv run ioc_parse.py
20
+
21
+
22
+ @dataclass
23
+ class ParseIOC:
24
+ """Determines the type of an indicator of compromise (IOC), such as ipv4, domain, MD5, etc."""
25
+ ioc: str
26
+ ioc_type: str | None = None
27
+ # using field(default_factory=list) for mutable default lists
28
+ extra: list[dict[str, Any]] = field(default_factory=list)
29
+
30
+ def __post_init__(self) -> None:
31
+ """Runs cleaning and processing automatically after __init__.
32
+
33
+ Runs .strip() on the IOC, checks for Punycode (IDNA), and then processes the IOC.
34
+ """
35
+ self.ioc = self.ioc.strip()
36
+ self._check_punycode()
37
+ self._process()
38
+
39
+ @property
40
+ def to_dict(self) -> dict[str, Any]:
41
+ """Returns a dictionary of the class attributes."""
42
+ out = asdict(self)
43
+ if not self.extra:
44
+ out.pop("extra")
45
+ return out
46
+
47
+ @property
48
+ def to_json(self) -> str:
49
+ """Returns a JSON object of the class attributes."""
50
+ out = asdict(self)
51
+ if not self.extra:
52
+ out.pop("extra")
53
+ return json.dumps(out)
54
+
55
+ def _preclean(self) -> None:
56
+ """Suppresses multiple forward and backward slashes. Does not affect _check_domain()."""
57
+ if "//" in self.ioc:
58
+ self.ioc = self.ioc.replace("//", "/")
59
+ self._preclean()
60
+ elif "\\\\" in self.ioc:
61
+ self.ioc = self.ioc.replace("\\\\", "\\")
62
+ self._preclean()
63
+ self.ioc = self.ioc.lower().replace("hxxp", "http").replace("[://]", "://").replace("**.", "").replace("*.", "")
64
+ return
65
+
66
+ def _check_punycode(self) -> str:
67
+ """International punycode checks; searches each character individually and decodes the IOC if needed."""
68
+ is_it_punycode = False
69
+ for char in self.ioc:
70
+ #if not re.search("[A-Za-z0-9.-]", char):
71
+ if ord(char) > 127:
72
+ is_it_punycode = True
73
+ self.ioc = self.ioc.encode("idna").decode().strip()
74
+ break
75
+
76
+ def _check_file(self) -> bool:
77
+ """Checks if the IOC is a file path for either Windows or Linux.
78
+
79
+ Handles absolute paths, relative paths, and paths with alternative data streams. AI helped write both large regex statements in this function.
80
+ """
81
+ # remove quotes common in Windows
82
+ ioc = self.ioc.strip('"')
83
+ # Windows regex
84
+ # check for drive letter (C:\ etc), backslashes, alternative data streams, extensions at the end of the path
85
+ windows_path_pattern = re.compile(
86
+ r"^(?:[a-zA-Z]:(?:\\|/)|\\\\|/)?(?:[^<>:\"/\\|?*\n]+\\?|[^<>:\"/\\|?*\n]+/)*[^<>:\"/\\|?*\n]+(?:\.[a-zA-Z0-9]+)?(?::[^<>:\"/\\|?*\n]+)?$")
87
+ # Linux regex
88
+ # check for leading forward slashes (/home/user), forward slashes as path separators, file extension at the end
89
+ linux_path_pattern = re.compile(
90
+ r"^(?:/|~)?(?:(?:[^<>:\"/\\|?*\n]+/)*[^<>:\"/\\|?*\n]+)?(?:\.[a-zA-Z0-9]+)?$")
91
+ # Windows check
92
+ if (re.search(r"^[a-zA-Z]:\\", ioc) or "\\" in ioc or ":" in ioc) and windows_path_pattern.search(ioc):
93
+ # additional interesting extension paths
94
+ #if re.search(r"\.(exe|dll|txt|pdf|docx|zip|py|sh|bat|jpg|png|mshta)$", ioc, re.IGNORECASE):
95
+ self.ioc_type = "file_path_windows"
96
+ return True
97
+ # Linux check
98
+ elif "/" in ioc and linux_path_pattern.search(ioc):
99
+ #if re.search(r"\.(sh|py|txt|conf|log|bin|deb|rpm|tar\.gz)$", ioc, re.IGNORECASE) or ioc.startswith('/'):
100
+ self.ioc_type = "file_path_linux"
101
+ return True
102
+ # False if not a path
103
+ return False
104
+
105
+ def _check_sha512(self) -> bool:
106
+ """Check for SHA-512 hash, only based on string length."""
107
+ if re.search("^[A-Za-z0-9]{128}$", self.ioc):
108
+ self.ioc_type = "sha512"
109
+ return True
110
+
111
+ def _check_sha256(self) -> bool:
112
+ """Check for SHA-256 hash, only based on string length."""
113
+ if re.search("^[A-Za-z0-9]{64}$", self.ioc):
114
+ self.ioc_type = "sha256"
115
+ return True
116
+
117
+ def _check_md5(self) -> bool:
118
+ """Check for MD5 hash, only based on string length."""
119
+ if re.search("^[A-Za-z0-9]{32}$", self.ioc):
120
+ self.ioc_type = "md5"
121
+ return True
122
+
123
+ def _check_email(self) -> bool:
124
+ """Loosely check for email addresses based on regex."""
125
+ # future: use email_validator like this: validate_email(self.ioc, check_deliverability=False)
126
+ if re.search(r"^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$", self.ioc):
127
+ self.ioc_type = "email"
128
+ try:
129
+ ed = self.ioc.split("@")[1]
130
+ self.extra.append({"ioc":ed, "ioc_type":"domain"})
131
+ except:
132
+ return False
133
+ return True
134
+
135
+ def _check_ip(self) -> bool:
136
+ """Checks for an IPv4 or IPv6 address or network."""
137
+ try:
138
+ addr = ipaddress.ip_network(self.ioc, strict=False)
139
+ # start with IPv4, assume it's a network unless /32
140
+ if isinstance(addr, ipaddress.IPv4Network):
141
+ self.ioc_type = "ipv4_network"
142
+ if addr.prefixlen == 32:
143
+ self.ioc_type = "ipv4"
144
+ return True
145
+ # next assume IPv6 network unless /128
146
+ elif isinstance(addr, ipaddress.IPv6Network):
147
+ self.ioc_type = "ipv6_network"
148
+ if addr.prefixlen == 128:
149
+ self.ioc_type = "ipv6"
150
+ return True
151
+ return True
152
+ except ValueError:
153
+ return False
154
+
155
+ def _check_domain(self) -> bool:
156
+ """Checks if the IOC is a URL or domain."""
157
+ try:
158
+ # create local semi-cleaned version of self.ioc
159
+ # several .strip() because of edge cases with malformed strings needing safeguards
160
+ url = self.ioc.lower().lstrip("htxps:/")
161
+ # add schema to urlparse properly parses netloc (domain)
162
+ url = "https://" + url
163
+ parsed = urlparse(url)
164
+ if not parsed.hostname:
165
+ return False
166
+ if parsed.username:
167
+ creds = parsed.username
168
+ if parsed.password:
169
+ creds += f":{parsed.password}"
170
+ self.extra.append({"ioc":creds, "ioc_type":"credentials"})
171
+ if parsed.port:
172
+ # possible future feature, to optionally remove "common" high ports via argument
173
+ #if port > 1024 and port != [5353, 8000, 8080, 8443]:
174
+ self.extra.append({"ioc":parsed.port, "ioc_type":"port"})
175
+ # final check for if the domain is an IP
176
+ try:
177
+ # anything except an IP will trigger exception
178
+ ipaddress.ip_address(parsed.hostname)
179
+ # if no exception, update self.ioc and run _check_ip()
180
+ self.ioc = parsed.hostname
181
+ self._check_ip()
182
+ #print("A"*20, parsed.hostname)
183
+ return True
184
+ except ValueError:
185
+ self.ioc = parsed.hostname
186
+ self.ioc_type = "domain"
187
+ return True
188
+ except Exception as e:
189
+ #print("URL_EXCEPTION", str(e))
190
+ return False
191
+
192
+ def _process(self) -> None:
193
+ """Main processing logic."""
194
+ self._preclean()
195
+ # if adding more check functions, add them here without ()
196
+ checks = [
197
+ self._check_sha512,
198
+ self._check_sha256,
199
+ self._check_md5,
200
+ self._check_email,
201
+ self._check_ip,
202
+ self._check_file,
203
+ self._check_domain
204
+ ]
205
+ # check the checks
206
+ for check in checks:
207
+ # add () to "check", since it's a function name
208
+ # if a check function returns True, return
209
+ if check():
210
+ return
211
+ # fallback
212
+ if not self.ioc_type:
213
+ self.ioc_type = "unknown"
214
+
215
+
216
+ def parse_multi(input_object: List[str] | TextIO, mode="combined") -> dict[str, Any] | List[Any]:
217
+ """Parse list or file of indicators into a large combined structure or individual lines.
218
+
219
+ Args:
220
+ input_object: list of IOCs, or path to a text file containing IOCs
221
+ mode: "combined" produces a single returned object, "single" produces individual lines
222
+ """
223
+ for_assembler = []
224
+ # read a list
225
+ if isinstance(input_object, list):
226
+ for item in input_object:
227
+ #item = repr(item)
228
+ if not item.startswith("#") and not item.startswith("="):
229
+ ioc = ParseIOC(item)
230
+ for_assembler.append(ioc.to_dict)
231
+ # read a file
232
+ else:
233
+ try:
234
+ with open(input_object) as input_file:
235
+ for line in input_file:
236
+ if not line.startswith("#") and not line.startswith("="):
237
+ ioc = ParseIOC(line)
238
+ for_assembler.append(ioc.to_dict)
239
+ except Exception as e:
240
+ print(str(e))
241
+ return for_assembler
242
+ # dict
243
+ if mode == "combined":
244
+ return _assembler(for_assembler)
245
+ # return list of dicts
246
+ elif mode == "single":
247
+ return for_assembler
248
+ #print(json.dumps(a, indent=4))
249
+
250
+
251
+ def _field_mapper(iocs: dict, map_file_name: str, chunk_size: int=0) -> dict:
252
+ """Match field mapping to provided ioc_parse output.
253
+
254
+ This is the worker function that maps indicators to SIEM field names.
255
+
256
+ Args:
257
+ ioc: the post-processed IOCs in a dictionary
258
+ map_file_name: path to the TOML mapping file
259
+ chunk_size (unused): split IOCs into chunks of n-size
260
+ """
261
+ try:
262
+ with open(map_file_name, "rb") as f:
263
+ map_data = tomllib.load(f)
264
+ except Exception as e:
265
+ return str(e)+": cannot open TOML field mapping file."
266
+ # inner chunking function
267
+ # chunking should be a user responsibility; may add helper function
268
+ def _inner_chunk(l, n) -> list:
269
+ """Split a large list into smaller lists of n items, with one list of remainders."""
270
+ for i in range(0, len(l), n):
271
+ yield l[i:i+n]
272
+ out = {}
273
+ if map_data.get("field_map"):
274
+ # the type of field, ipv4, domain, etc
275
+ for field_type in map_data["field_map"]:
276
+ # if the key (ioc type) exists in the parsed IOCs
277
+ if iocs.get(field_type):
278
+ # each field we want in the final output, from map toml
279
+ for field_name in map_data["field_map"][field_type]:
280
+ # make a key if not already there
281
+ if not out.get(field_name):
282
+ out[field_name] = []
283
+ # extend the IOCs into the output mapping via shared key
284
+ out[field_name].extend(iocs[field_type])
285
+ else:
286
+ return "error: no key field_map in loaded TOML file"
287
+ # chunking should be a user responsibility; may add helper function
288
+ #if chunk_size > 0:
289
+ # for key in out:
290
+ # out[key]["values"] = list(_inner_chunk(out[key]["values"], chunk_size))
291
+ # out[key]["chunk_size"] = len(out[key]["values"])
292
+ #
293
+ #print("="*50)
294
+ #print(json.dumps(out, indent=4))
295
+ #mapper(map_data["field_map"], ioc_data, chunk_size=2)
296
+ return out
297
+
298
+
299
+ def map_fields(input_object: List[str] | TextIO, map_file_name: str) -> dict:
300
+ """Callable function to map IOCs to the fields in the TOML configuration.
301
+
302
+ This is the friendly entrypoint to the field-to-mapping functions.
303
+
304
+ Args:
305
+ input_object: list of IOCs, or path to a text file containing IOCs
306
+ map_file_name: path to the TOML mapping file
307
+ """
308
+ parsed_iocs_combined = parse_multi(input_object, mode="combined")
309
+ return _field_mapper(parsed_iocs_combined, map_file_name)
310
+
311
+
312
+ def _assembler(l: List[str]) -> dict:
313
+ """Combines all IOCs, including nested "extra" entries, into one dictionary.
314
+
315
+ Use this dictionary with mappings to field names for automated SIEM or database queries.
316
+
317
+ Args:
318
+ l: list of IOCs to be assembled into the combined structure.
319
+ """
320
+ out = {}
321
+ def inner(d):
322
+ if d["ioc_type"] not in out:
323
+ out[d["ioc_type"]] = []
324
+ out[d["ioc_type"]].append(d["ioc"])
325
+ else:
326
+ out[d["ioc_type"]].append(d["ioc"])
327
+ for d in l:
328
+ inner(d)
329
+ if d.get("extra"):
330
+ for dd in d["extra"]:
331
+ inner(dd)
332
+ # clean output
333
+ for k,v in out.items():
334
+ out[k] = list(set(v))
335
+ #print(json.dumps(out, indent=4))
336
+ return out
337
+
338
+
339
+ if __name__ == "__main__":
340
+ #
341
+ ioc_list = ["bob@email.local", "website.local", "https://anotherwebsite.local:9443", "https://username:password@securewebsite.local:8443", "192.168.1.1", "192.168.1.0/24", "https://192.168.20.20/bad.txt", "NHKやさしいことばニュース.com"]
342
+ #
343
+ # categorize a single IOC
344
+ print(" categorize a single indicator ".center(80, "="))
345
+ parsed_indicator = ParseIOC("https://192.168.20.20/bad.txt")
346
+ print(".to_dict:", type(parsed_indicator.to_dict), parsed_indicator.to_dict)
347
+ print(".to_json:", type(parsed_indicator.to_json), parsed_indicator.to_json)
348
+ #
349
+ # THIS IS THE FIRST PRIMARY OUTPUT
350
+ # parse a list or file IOCs into a large dict or json structure using parse_multi()
351
+ print(" parse a list or file of IOCs into a combined structure ".center(80, "="))
352
+ iocs_from_list = parse_multi(ioc_list, mode="combined")
353
+ iocs_from_file = parse_multi("ioc_examples.txt", mode="combined")
354
+ print(json.dumps(iocs_from_list, indent=4))
355
+ print(json.dumps(iocs_from_file, indent=4))
356
+ #
357
+ # alternatively parse a file line by line (mode=single) into dicts, instead of calling ParseIOC(indicator)
358
+ print(" yield dictionaries from a list or file, instead of a combined structure ".center(80, "="))
359
+ for item in parse_multi("ioc_examples.txt", mode="single"):
360
+ print(item)
361
+ #
362
+ # THIS IS THE SECOND PRIMARY OUTPUT
363
+ # use a field map to stage siem or database queries
364
+ print(" provide IOC (file or list) and TOML config (path) map_fields() ".center(80, "="))
365
+ #m = map_fields(ioc_list, "map_ecs.toml")
366
+ m = map_fields("ioc_examples.txt", "map_ecs.toml")
367
+ print(json.dumps(m, indent=4))