cryptsmith 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Jake Burre
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,67 @@
1
+ Metadata-Version: 2.4
2
+ Name: cryptsmith
3
+ Version: 0.1.0
4
+ Summary: Crypto attack primitives and analysis toolkit for CTF players and security researchers. Zero dependencies.
5
+ Author: Jake Burre
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/burrejak22/cryptsmith
8
+ Project-URL: Repository, https://github.com/burrejak22/cryptsmith
9
+ Keywords: crypto,ctf,cryptanalysis,rsa,pcap,security
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Information Technology
12
+ Classifier: Topic :: Security :: Cryptography
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Dynamic: license-file
19
+
20
+ # cryptsmith
21
+
22
+ Crypto attack primitives and analysis toolkit for CTF players and security researchers. Zero dependencies, stdlib only.
23
+
24
+ Built for challenges like NSA Codebreakers style pcaps: custom protocols, hidden enrollment keys, obfuscated traffic.
25
+
26
+ ## What it does
27
+
28
+ * **Classical ciphers** — caesar bruteforce with english scoring, vigenere, atbash, affine
29
+ * **XOR** — single byte bruteforce, repeating key break via hamming distance keysize guessing
30
+ * **RSA attacks** — Wiener's attack, Hastad broadcast, common modulus, small exponent
31
+ * **Analysis** — shannon entropy, index of coincidence, chi squared english scoring, frequency analysis
32
+ * **Encoding** — layered auto decode for base64, hex, url encoding, with the decode chain reported
33
+ * **Hash ID** — identify hash algorithms by digest length and charset
34
+ * **PCAP** — pure python pcap reader, IPv4/TCP parsing, TCP stream reassembly, payload carving, token hunting
35
+
36
+ ## Install
37
+
38
+ ```bash
39
+ pip install cryptsmith
40
+ ```
41
+
42
+ ## Quick use
43
+
44
+ ```python
45
+ from cryptsmith import xor, rsa, pcap, analysis
46
+
47
+ # break single byte xor
48
+ for key, plaintext, score in xor.single_byte_xor_break(data)[:3]:
49
+ print(key, plaintext)
50
+
51
+ # wiener attack on weak rsa
52
+ d = rsa.wiener_attack(n, e)
53
+
54
+ # pull tcp streams out of a pcap and hunt 12 char alpha tokens
55
+ streams = pcap.reassemble_streams(pcap.PcapReader("capture.pcap"))
56
+ for s in streams:
57
+ for token in pcap.hunt_alpha_tokens(s.payload, length=12):
58
+ print(s.label, token)
59
+ ```
60
+
61
+ ## Companion project
62
+
63
+ **flaghunter** builds on cryptsmith: point it at a challenge file or a pcap and it runs a full battery of attacks automatically and writes you a report.
64
+
65
+ ## License
66
+
67
+ MIT
@@ -0,0 +1,48 @@
1
+ # cryptsmith
2
+
3
+ Crypto attack primitives and analysis toolkit for CTF players and security researchers. Zero dependencies, stdlib only.
4
+
5
+ Built for challenges like NSA Codebreakers style pcaps: custom protocols, hidden enrollment keys, obfuscated traffic.
6
+
7
+ ## What it does
8
+
9
+ * **Classical ciphers** — caesar bruteforce with english scoring, vigenere, atbash, affine
10
+ * **XOR** — single byte bruteforce, repeating key break via hamming distance keysize guessing
11
+ * **RSA attacks** — Wiener's attack, Hastad broadcast, common modulus, small exponent
12
+ * **Analysis** — shannon entropy, index of coincidence, chi squared english scoring, frequency analysis
13
+ * **Encoding** — layered auto decode for base64, hex, url encoding, with the decode chain reported
14
+ * **Hash ID** — identify hash algorithms by digest length and charset
15
+ * **PCAP** — pure python pcap reader, IPv4/TCP parsing, TCP stream reassembly, payload carving, token hunting
16
+
17
+ ## Install
18
+
19
+ ```bash
20
+ pip install cryptsmith
21
+ ```
22
+
23
+ ## Quick use
24
+
25
+ ```python
26
+ from cryptsmith import xor, rsa, pcap, analysis
27
+
28
+ # break single byte xor
29
+ for key, plaintext, score in xor.single_byte_xor_break(data)[:3]:
30
+ print(key, plaintext)
31
+
32
+ # wiener attack on weak rsa
33
+ d = rsa.wiener_attack(n, e)
34
+
35
+ # pull tcp streams out of a pcap and hunt 12 char alpha tokens
36
+ streams = pcap.reassemble_streams(pcap.PcapReader("capture.pcap"))
37
+ for s in streams:
38
+ for token in pcap.hunt_alpha_tokens(s.payload, length=12):
39
+ print(s.label, token)
40
+ ```
41
+
42
+ ## Companion project
43
+
44
+ **flaghunter** builds on cryptsmith: point it at a challenge file or a pcap and it runs a full battery of attacks automatically and writes you a report.
45
+
46
+ ## License
47
+
48
+ MIT
@@ -0,0 +1,30 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "cryptsmith"
7
+ version = "0.1.0"
8
+ description = "Crypto attack primitives and analysis toolkit for CTF players and security researchers. Zero dependencies."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "Jake Burre" }]
13
+ keywords = ["crypto", "ctf", "cryptanalysis", "rsa", "pcap", "security"]
14
+ classifiers = [
15
+ "Development Status :: 3 - Alpha",
16
+ "Intended Audience :: Information Technology",
17
+ "Topic :: Security :: Cryptography",
18
+ "Programming Language :: Python :: 3",
19
+ "License :: OSI Approved :: MIT License",
20
+ ]
21
+
22
+ [project.urls]
23
+ Homepage = "https://github.com/burrejak22/cryptsmith"
24
+ Repository = "https://github.com/burrejak22/cryptsmith"
25
+
26
+ [tool.setuptools.packages.find]
27
+ where = ["src"]
28
+
29
+ [tool.pytest.ini_options]
30
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,6 @@
1
+ """cryptsmith: crypto attack primitives and analysis toolkit."""
2
+
3
+ from . import analysis, classical, encoding, hashid, pcap, rsa, xor
4
+
5
+ __version__ = "0.1.0"
6
+ __all__ = ["analysis", "classical", "encoding", "hashid", "pcap", "rsa", "xor"]
@@ -0,0 +1,77 @@
1
+ """Statistical analysis: entropy, frequency, english scoring."""
2
+ import math
3
+ from collections import Counter
4
+
5
+ ENGLISH_FREQ = {
6
+ "a": 0.08167, "b": 0.01492, "c": 0.02782, "d": 0.04253, "e": 0.12702,
7
+ "f": 0.02228, "g": 0.02015, "h": 0.06094, "i": 0.06966, "j": 0.00153,
8
+ "k": 0.00772, "l": 0.04025, "m": 0.02406, "n": 0.06749, "o": 0.07507,
9
+ "p": 0.01929, "q": 0.00095, "r": 0.05987, "s": 0.06327, "t": 0.09056,
10
+ "u": 0.02758, "v": 0.00978, "w": 0.02360, "x": 0.00150, "y": 0.01974,
11
+ "z": 0.00074,
12
+ }
13
+
14
+
15
+ def shannon_entropy(data: bytes) -> float:
16
+ """Shannon entropy in bits per byte. ~8.0 is random, ~4.7 is english text."""
17
+ if not data:
18
+ return 0.0
19
+ counts = Counter(data)
20
+ length = len(data)
21
+ return -sum((c / length) * math.log2(c / length) for c in counts.values())
22
+
23
+
24
+ def frequency_analysis(text: str) -> list[tuple[str, float]]:
25
+ """Letter frequencies sorted most common first."""
26
+ letters = [c.lower() for c in text if c.isalpha()]
27
+ if not letters:
28
+ return []
29
+ counts = Counter(letters)
30
+ total = len(letters)
31
+ return sorted(((ch, n / total) for ch, n in counts.items()), key=lambda x: -x[1])
32
+
33
+
34
+ def index_of_coincidence(text: str) -> float:
35
+ """IoC ~0.065 for english, ~0.038 for random. Useful to spot polyalphabetic ciphers."""
36
+ letters = [c.lower() for c in text if c.isalpha()]
37
+ n = len(letters)
38
+ if n < 2:
39
+ return 0.0
40
+ counts = Counter(letters)
41
+ return sum(c * (c - 1) for c in counts.values()) / (n * (n - 1))
42
+
43
+
44
+ def chi_squared(text: str) -> float:
45
+ """Chi squared distance from english letter frequencies. Lower is more english."""
46
+ letters = [c.lower() for c in text if c.isalpha()]
47
+ n = len(letters)
48
+ if n == 0:
49
+ return float("inf")
50
+ counts = Counter(letters)
51
+ score = 0.0
52
+ for ch, expected in ENGLISH_FREQ.items():
53
+ observed = counts.get(ch, 0)
54
+ exp = n * expected
55
+ score += ((observed - exp) ** 2) / exp if exp else 0
56
+ return score
57
+
58
+
59
+ def english_score(text: str) -> float:
60
+ """Higher is better. Combines chi squared with a bonus for common words and spaces."""
61
+ if not text:
62
+ return float("-inf")
63
+ try:
64
+ t = text if isinstance(text, str) else text.decode("latin1")
65
+ except Exception:
66
+ return float("-inf")
67
+ chi = chi_squared(t)
68
+ words = t.lower().split()
69
+ common = sum(1 for w in words if w.strip(".,!?;:\"'()") in {
70
+ "the", "and", "for", "with", "this", "that", "from", "have", "flag",
71
+ "key", "secret", "enroll", "hello", "server", "client",
72
+ })
73
+ space_bonus = t.count(" ") * 2
74
+ printable_ratio = sum(1 for c in t if 32 <= ord(c) < 127 or c in "\n\r\t") / max(len(t), 1)
75
+ if printable_ratio < 0.9:
76
+ return float("-inf")
77
+ return space_bonus + common * 25 - chi / 10
@@ -0,0 +1,91 @@
1
+ """Classical ciphers: caesar, vigenere, atbash, affine."""
2
+ from .analysis import english_score
3
+
4
+
5
+ def _shift_char(c: str, shift: int) -> str:
6
+ if "a" <= c <= "z":
7
+ return chr((ord(c) - 97 + shift) % 26 + 97)
8
+ if "A" <= c <= "Z":
9
+ return chr((ord(c) - 65 + shift) % 26 + 65)
10
+ return c
11
+
12
+
13
+ def caesar_decrypt(text: str, shift: int) -> str:
14
+ return "".join(_shift_char(c, -shift) for c in text)
15
+
16
+
17
+ def caesar_encrypt(text: str, shift: int) -> str:
18
+ return "".join(_shift_char(c, shift) for c in text)
19
+
20
+
21
+ def caesar_bruteforce(text: str, top: int = 3) -> list[tuple[int, str, float]]:
22
+ """Try all 26 shifts, return the top scoring (shift, plaintext, score)."""
23
+ scored = [(s, caesar_decrypt(text, s), english_score(caesar_decrypt(text, s))) for s in range(26)]
24
+ scored.sort(key=lambda x: -x[2])
25
+ return scored[:top]
26
+
27
+
28
+ def atbash(text: str) -> str:
29
+ def flip(c: str) -> str:
30
+ if "a" <= c <= "z":
31
+ return chr(122 - (ord(c) - 97))
32
+ if "A" <= c <= "Z":
33
+ return chr(90 - (ord(c) - 65))
34
+ return c
35
+ return "".join(flip(c) for c in text)
36
+
37
+
38
+ def vigenere_decrypt(text: str, key: str) -> str:
39
+ key = key.lower()
40
+ out = []
41
+ ki = 0
42
+ for c in text:
43
+ if c.isalpha():
44
+ shift = ord(key[ki % len(key)]) - 97
45
+ out.append(_shift_char(c, -shift))
46
+ ki += 1
47
+ else:
48
+ out.append(c)
49
+ return "".join(out)
50
+
51
+
52
+ def vigenere_encrypt(text: str, key: str) -> str:
53
+ key = key.lower()
54
+ out = []
55
+ ki = 0
56
+ for c in text:
57
+ if c.isalpha():
58
+ shift = ord(key[ki % len(key)]) - 97
59
+ out.append(_shift_char(c, shift))
60
+ ki += 1
61
+ else:
62
+ out.append(c)
63
+ return "".join(out)
64
+
65
+
66
+ def affine_decrypt(text: str, a: int, b: int) -> str | None:
67
+ from math import gcd
68
+ if gcd(a, 26) != 1:
69
+ return None
70
+ a_inv = pow(a, -1, 26)
71
+ def dec(c: str) -> str:
72
+ if "a" <= c <= "z":
73
+ return chr((a_inv * (ord(c) - 97 - b)) % 26 + 97)
74
+ if "A" <= c <= "Z":
75
+ return chr((a_inv * (ord(c) - 65 - b)) % 26 + 65)
76
+ return c
77
+ return "".join(dec(c) for c in text)
78
+
79
+
80
+ def affine_bruteforce(text: str, top: int = 5) -> list[tuple[int, int, str, float]]:
81
+ """Try all valid (a, b) pairs, return top scoring."""
82
+ from math import gcd
83
+ results = []
84
+ for a in range(1, 26):
85
+ if gcd(a, 26) != 1:
86
+ continue
87
+ for b in range(26):
88
+ pt = affine_decrypt(text, a, b)
89
+ results.append((a, b, pt, english_score(pt)))
90
+ results.sort(key=lambda x: -x[3])
91
+ return results[:top]
@@ -0,0 +1,77 @@
1
+ """Encoding detection and layered auto decode (base64, hex, url, ...)."""
2
+ import base64
3
+ import re
4
+ from urllib.parse import unquote
5
+
6
+ _B64_RE = re.compile(r"^[A-Za-z0-9+/=\s]+$")
7
+ _HEX_RE = re.compile(r"^(?:[0-9a-fA-F]{2})+$")
8
+
9
+
10
+ def looks_like_base64(s: str) -> bool:
11
+ s = s.strip()
12
+ return len(s) >= 8 and len(s) % 4 == 0 and bool(_B64_RE.match(s))
13
+
14
+
15
+ def looks_like_hex(s: str) -> bool:
16
+ s = s.strip()
17
+ return len(s) >= 8 and bool(_HEX_RE.match(s))
18
+
19
+
20
+ def looks_like_urlencoded(s: str) -> bool:
21
+ return "%" in s and bool(re.search(r"%[0-9a-fA-F]{2}", s))
22
+
23
+
24
+ def _try_b64(s: str) -> bytes | None:
25
+ try:
26
+ return base64.b64decode(s.strip(), validate=False)
27
+ except Exception:
28
+ return None
29
+
30
+
31
+ def _try_hex(s: str) -> bytes | None:
32
+ try:
33
+ return bytes.fromhex(s.strip())
34
+ except Exception:
35
+ return None
36
+
37
+
38
+ def smart_decode(data: bytes, max_depth: int = 4) -> list[tuple[str, bytes]]:
39
+ """Peel layered encodings. Returns list of (chain_label, decoded_bytes)."""
40
+ try:
41
+ text = data.decode("ascii").strip()
42
+ except Exception:
43
+ return []
44
+ results: list[tuple[str, bytes]] = []
45
+ stack = [(text, [])]
46
+ seen = set()
47
+ for _ in range(max_depth):
48
+ next_stack = []
49
+ for current, chain in stack:
50
+ if current in seen:
51
+ continue
52
+ seen.add(current)
53
+ candidates: list[tuple[str, bytes | None]] = []
54
+ if looks_like_base64(current):
55
+ candidates.append(("b64", _try_b64(current)))
56
+ if looks_like_hex(current):
57
+ candidates.append(("hex", _try_hex(current)))
58
+ if looks_like_urlencoded(current):
59
+ try:
60
+ candidates.append(("url", unquote(current).encode("latin1")))
61
+ except Exception:
62
+ pass
63
+ for label, decoded in candidates:
64
+ if decoded is None or decoded == current.encode():
65
+ continue
66
+ new_chain = chain + [label]
67
+ results.append(("+".join(new_chain), decoded))
68
+ try:
69
+ nxt = decoded.decode("ascii").strip()
70
+ if nxt != current:
71
+ next_stack.append((nxt, new_chain))
72
+ except Exception:
73
+ pass
74
+ stack = next_stack
75
+ if not stack:
76
+ break
77
+ return results
@@ -0,0 +1,22 @@
1
+ """Identify hash algorithms by digest length and charset."""
2
+ import re
3
+
4
+ _HASHES = [
5
+ (32, "md5", "md4"),
6
+ (40, "sha1", "ripemd160"),
7
+ (56, "sha224"),
8
+ (64, "sha256", "sha3_256", "blake2s"),
9
+ (96, "sha384", "sha3_384"),
10
+ (128, "sha512", "sha3_512", "blake2b", "whirlpool"),
11
+ ]
12
+
13
+
14
+ def identify(digest: str) -> list[str]:
15
+ """Return likely hash names for a hex digest string."""
16
+ d = digest.strip().lower()
17
+ if not re.fullmatch(r"[0-9a-f]+", d):
18
+ return []
19
+ for length, *names in _HASHES:
20
+ if len(d) == length:
21
+ return list(names)
22
+ return []
@@ -0,0 +1,331 @@
1
+ """Pure python pcap reading, TCP stream reassembly, and payload carving.
2
+
3
+ No dependencies. Handles the classic libpcap format (little and big endian,
4
+ microsecond and nanosecond), Ethernet + IPv4 + TCP.
5
+ """
6
+ import re
7
+ import struct
8
+ from dataclasses import dataclass, field
9
+
10
+
11
+ @dataclass
12
+ class Packet:
13
+ ts: float
14
+ data: bytes
15
+
16
+
17
+ class PcapReader:
18
+ """Iterate packets from a pcap file."""
19
+
20
+ def __init__(self, path: str):
21
+ self.path = path
22
+ with open(path, "rb") as f:
23
+ magic = f.read(4)
24
+ if magic == b"\xd4\xc3\xb2\xa1":
25
+ self._endian = "<"
26
+ elif magic == b"\xa1\xb2\xc3\xd4":
27
+ self._endian = ">"
28
+ elif magic in (b"\x4d\x3c\xb2\xa1", b"\xa1\xb2\x3c\x4d"):
29
+ self._endian = "<" if magic == b"\x4d\x3c\xb2\xa1" else ">"
30
+ self._nanosec = True
31
+ else:
32
+ raise ValueError("not a pcap file (bad magic)")
33
+ self._nanosec = getattr(self, "_nanosec", magic in (b"\x4d\x3c\xb2\xa1", b"\xa1\xb2\x3c\x4d"))
34
+
35
+ def packets(self):
36
+ e = self._endian
37
+ with open(self.path, "rb") as f:
38
+ f.read(24) # global header
39
+ while True:
40
+ hdr = f.read(16)
41
+ if len(hdr) < 16:
42
+ return
43
+ ts_sec, ts_frac, incl_len, _ = struct.unpack(e + "IIII", hdr)
44
+ data = f.read(incl_len)
45
+ if len(data) < incl_len:
46
+ return
47
+ ts = ts_sec + ts_frac / (1e9 if self._nanosec else 1e6)
48
+ yield Packet(ts=ts, data=data)
49
+
50
+
51
+ @dataclass
52
+ class TCPFlow:
53
+ src: str
54
+ sport: int
55
+ dst: str
56
+ dport: int
57
+ payload: bytes = b""
58
+
59
+ @property
60
+ def label(self) -> str:
61
+ return f"{self.src}:{self.sport} -> {self.dst}:{self.dport}"
62
+
63
+
64
+ @dataclass
65
+ class TCPStream:
66
+ """Both directions of one TCP conversation, reassembled."""
67
+ endpoint_a: tuple[str, int] = field(default=("", 0))
68
+ endpoint_b: tuple[str, int] = field(default=("", 0))
69
+ a_to_b: bytes = b""
70
+ b_to_a: bytes = b""
71
+ packet_count: int = 0
72
+
73
+ @property
74
+ def label(self) -> str:
75
+ a = f"{self.endpoint_a[0]}:{self.endpoint_a[1]}"
76
+ b = f"{self.endpoint_b[0]}:{self.endpoint_b[1]}"
77
+ return f"{a} <-> {b}"
78
+
79
+ @property
80
+ def payload(self) -> bytes:
81
+ return self.a_to_b + b"\n--- direction flip ---\n" + self.b_to_a
82
+
83
+
84
+ def _parse_ipv4(data: bytes):
85
+ if len(data) < 20:
86
+ return None
87
+ ihl = (data[0] & 0x0F) * 4
88
+ if len(data) < ihl or data[9] != 6: # protocol 6 = TCP
89
+ return None
90
+ src = ".".join(str(b) for b in data[12:16])
91
+ dst = ".".join(str(b) for b in data[16:20])
92
+ return src, dst, data[ihl:]
93
+
94
+
95
+ def _parse_tcp(data: bytes):
96
+ if len(data) < 20:
97
+ return None
98
+ sport, dport, seq = struct.unpack("!HHI", data[:8])
99
+ data_offset = (data[12] >> 4) * 4
100
+ if len(data) < data_offset:
101
+ return None
102
+ return sport, dport, seq, data[data_offset:]
103
+
104
+
105
+ def parse_packet(raw: bytes) -> TCPFlow | None:
106
+ """Parse one link layer frame into a TCPFlow. Returns None if not TCP/IPv4."""
107
+ if len(raw) < 14:
108
+ return None
109
+ ethertype = struct.unpack("!H", raw[12:14])[0]
110
+ if ethertype != 0x0800: # IPv4 only
111
+ return None
112
+ ip = _parse_ipv4(raw[14:])
113
+ if ip is None:
114
+ return None
115
+ src, dst, tcp_seg = ip
116
+ tcp = _parse_tcp(tcp_seg)
117
+ if tcp is None:
118
+ return None
119
+ sport, dport, seq, payload = tcp
120
+ flow = TCPFlow(src=src, sport=sport, dst=dst, dport=dport, payload=b"")
121
+ flow.seq = seq # type: ignore[attr-defined]
122
+ flow.payload = payload
123
+ return flow
124
+
125
+
126
+ def reassemble_streams(reader: PcapReader) -> list[TCPStream]:
127
+ """Reassemble TCP streams from a pcap. Returns streams with payloads in order."""
128
+ # key: frozenset of endpoints -> stream; per direction seq buckets
129
+ streams: dict[frozenset, TCPStream] = {}
130
+ buckets: dict[frozenset, dict[tuple, list[tuple[int, bytes]]]] = {}
131
+
132
+ for pkt in reader.packets():
133
+ flow = parse_packet(pkt.data)
134
+ if flow is None or not flow.payload:
135
+ continue
136
+ key = frozenset({(flow.src, flow.sport), (flow.dst, flow.dport)})
137
+ if key not in streams:
138
+ eps = sorted(key)
139
+ streams[key] = TCPStream(endpoint_a=eps[0], endpoint_b=eps[1])
140
+ buckets[key] = {}
141
+ direction = ((flow.src, flow.sport), (flow.dst, flow.dport))
142
+ buckets[key].setdefault(direction, []).append((flow.seq, flow.payload))
143
+ streams[key].packet_count += 1
144
+
145
+ for key, stream in streams.items():
146
+ for direction, segs in buckets[key].items():
147
+ segs.sort(key=lambda x: x[0])
148
+ merged = bytearray()
149
+ next_seq = None
150
+ for seq, payload in segs:
151
+ if next_seq is None:
152
+ merged += payload
153
+ next_seq = seq + len(payload)
154
+ elif seq >= next_seq:
155
+ merged += payload
156
+ next_seq = seq + len(payload)
157
+ elif seq + len(payload) > next_seq:
158
+ merged += payload[next_seq - seq:]
159
+ next_seq = seq + len(payload)
160
+ data = bytes(merged)
161
+ if direction[0] == stream.endpoint_a:
162
+ stream.a_to_b = data
163
+ else:
164
+ stream.b_to_a = data
165
+ return list(streams.values())
166
+
167
+
168
+ def carve_strings(data: bytes, min_len: int = 4) -> list[str]:
169
+ """Extract printable ascii strings like the `strings` utility."""
170
+ out = []
171
+ cur = []
172
+ for b in data:
173
+ if 32 <= b < 127:
174
+ cur.append(chr(b))
175
+ else:
176
+ if len(cur) >= min_len:
177
+ out.append("".join(cur))
178
+ cur = []
179
+ if len(cur) >= min_len:
180
+ out.append("".join(cur))
181
+ return out
182
+
183
+
184
+ def opaque_runs(data: bytes, min_len: int = 12) -> list[bytes]:
185
+ """Maximal runs of non printable bytes. Obfuscated blobs hide in these."""
186
+ runs = []
187
+ cur = bytearray()
188
+ for b in data:
189
+ if 32 <= b < 127:
190
+ if len(cur) >= min_len:
191
+ runs.append(bytes(cur))
192
+ cur = bytearray()
193
+ else:
194
+ cur.append(b)
195
+ if len(cur) >= min_len:
196
+ runs.append(bytes(cur))
197
+ return runs
198
+
199
+
200
+ def suspicious_spans(data: bytes, min_len: int = 16, max_printable: float = 0.7) -> list[bytes]:
201
+ """Spans that look obfuscated: mixed binary where xor'd text often hides.
202
+
203
+ Xor obfuscation with a byte key leaves some bytes printable, so strict
204
+ non printable runs miss it. These spans tolerate printable bytes up to
205
+ max_printable ratio.
206
+ """
207
+ flags = [32 <= b < 127 for b in data]
208
+ n = len(data)
209
+ spans = []
210
+ i = 0
211
+ while i < n:
212
+ if flags[i]:
213
+ i += 1
214
+ continue
215
+ j = i
216
+ while j < n:
217
+ seg = flags[i:j + 1]
218
+ if sum(seg) / len(seg) > max_printable and (j - i + 1) >= min_len:
219
+ break
220
+ j += 1
221
+ if j - i >= min_len:
222
+ spans.append(data[i:j])
223
+ i = j if j > i else i + 1
224
+ return spans
225
+
226
+
227
+ def hunt_alpha_tokens(data: bytes, length: int = 12) -> list[str]:
228
+ """Find alpha only tokens of exactly `length` chars, e.g. enrollment keys.
229
+
230
+ Looks at raw bytes and at printable strings, deduped, in order of appearance.
231
+ """
232
+ text = data.decode("latin1")
233
+ pattern = re.compile(rf"(?<![A-Za-z])[A-Za-z]{{{length}}}(?![A-Za-z])")
234
+ seen = set()
235
+ hits = []
236
+ for m in pattern.finditer(text):
237
+ tok = m.group(0)
238
+ if tok not in seen:
239
+ seen.add(tok)
240
+ hits.append(tok)
241
+ return hits
242
+
243
+
244
+ def hunt_alpha_runs(data: bytes, min_length: int = 12) -> list[str]:
245
+ """Find alpha only runs of at least min_length chars.
246
+
247
+ Looser than hunt_alpha_tokens: for decoded or deobfuscated text where
248
+ token boundaries are unreliable, e.g. a key glued to framing bytes.
249
+ """
250
+ text = data.decode("latin1")
251
+ seen = set()
252
+ hits = []
253
+ for m in re.finditer(r"[A-Za-z]+", text):
254
+ tok = m.group(0)
255
+ if len(tok) >= min_length and tok not in seen:
256
+ seen.add(tok)
257
+ hits.append(tok)
258
+ return hits
259
+
260
+
261
+ _ALPHA = set(b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz")
262
+
263
+ # BYTE_MASKS[b]: bitmask of xor keys k where (b ^ k) is an ascii letter.
264
+ # Precomputed once; key k corresponds to bit (1 << k).
265
+ _BYTE_MASKS: list[int] = []
266
+ for _b in range(256):
267
+ _m = 0
268
+ for _c in _ALPHA:
269
+ _m |= 1 << (_b ^ _c)
270
+ _BYTE_MASKS.append(_m)
271
+
272
+
273
+ def xor_token_scan(data: bytes, min_length: int = 12) -> list[tuple[int, int, str]]:
274
+ """Find (offset, key, token) where data[offset:] xored with key starts with
275
+ an alpha only run of at least min_length chars.
276
+
277
+ This catches xor obfuscated tokens buried in mixed binary traffic, where
278
+ whole blob english scoring fails because plaintext and obfuscated regions
279
+ mix. Bitmask sliding window makes it O(n): for each byte, the set of keys
280
+ yielding a letter; AND the sets across the window. Any nonzero result
281
+ means some key decodes the whole window to letters.
282
+ """
283
+ n = len(data)
284
+ if n < min_length:
285
+ return []
286
+ masks = [_BYTE_MASKS[b] for b in data]
287
+ hits: list[tuple[int, int, str]] = []
288
+ for start in range(n - min_length + 1):
289
+ window_and = masks[start]
290
+ for i in range(start + 1, start + min_length):
291
+ window_and &= masks[i]
292
+ if not window_and:
293
+ break
294
+ if not window_and:
295
+ continue
296
+ mm, k = window_and, 0
297
+ while mm:
298
+ if mm & 1:
299
+ end = start + min_length
300
+ while end < n and (data[end] ^ k) in _ALPHA:
301
+ end += 1
302
+ token = bytes(b ^ k for b in data[start:end]).decode("ascii")
303
+ # skip trivial keys: the token is already sitting in the raw
304
+ # bytes (e.g. key 0x00 identity or 0x20 case flip on plaintext)
305
+ raw = data[start:end].decode("latin1")
306
+ if raw.lower() != token.lower():
307
+ hits.append((start, k, token))
308
+ mm >>= 1
309
+ k += 1
310
+ # dedupe by (key, token), keep earliest offset
311
+ seen = set()
312
+ deduped = []
313
+ for offset, k, token in hits:
314
+ if (k, token) not in seen:
315
+ seen.add((k, token))
316
+ deduped.append((offset, k, token))
317
+ return deduped
318
+
319
+
320
+ def hunt_pattern(data: bytes, pattern: str) -> list[str]:
321
+ """Find all matches of a custom regex against the payload as latin1 text."""
322
+ rx = re.compile(pattern)
323
+ text = data.decode("latin1")
324
+ seen = set()
325
+ hits = []
326
+ for m in rx.finditer(text):
327
+ tok = m.group(0)
328
+ if tok not in seen:
329
+ seen.add(tok)
330
+ hits.append(tok)
331
+ return hits
@@ -0,0 +1,123 @@
1
+ """RSA attacks: Wiener, Hastad broadcast, common modulus, small exponent."""
2
+ import math
3
+
4
+
5
+ def egcd(a: int, b: int) -> tuple[int, int, int]:
6
+ if b == 0:
7
+ return (a, 1, 0)
8
+ g, x, y = egcd(b, a % b)
9
+ return (g, y, x - (a // b) * y)
10
+
11
+
12
+ def modinv(a: int, m: int) -> int:
13
+ g, x, _ = egcd(a, m)
14
+ if g != 1:
15
+ raise ValueError("no modular inverse")
16
+ return x % m
17
+
18
+
19
+ def int_nthroot(x: int, n: int) -> tuple[int, bool]:
20
+ """Integer nth root. Returns (root, exact)."""
21
+ if x < 0:
22
+ raise ValueError("negative")
23
+ if x == 0:
24
+ return (0, True)
25
+ bits = x.bit_length()
26
+ high = 1 << ((bits + n - 1) // n + 1)
27
+ low = 0
28
+ while low < high:
29
+ mid = (low + high) // 2
30
+ p = mid ** n
31
+ if p == x:
32
+ return (mid, True)
33
+ if p < x:
34
+ low = mid + 1
35
+ else:
36
+ high = mid
37
+ return (low - 1, (low - 1) ** n == x)
38
+
39
+
40
+ def _continued_fraction(num: int, den: int) -> list[int]:
41
+ cf = []
42
+ while den:
43
+ q = num // den
44
+ cf.append(q)
45
+ num, den = den, num - q * den
46
+ return cf
47
+
48
+
49
+ def _convergents(cf: list[int]):
50
+ convs = []
51
+ for i in range(len(cf)):
52
+ num, den = cf[i], 1
53
+ for j in range(i - 1, -1, -1):
54
+ num, den = den + num * cf[j], num
55
+ convs.append((den, num)) # (k, d) style: returns denominator, numerator
56
+ return convs
57
+
58
+
59
+ def wiener_attack(n: int, e: int) -> int | None:
60
+ """Recover d when d < n^0.25 / 3 via Wiener's continued fraction attack."""
61
+ cf = _continued_fraction(e, n)
62
+ for d, k in _convergents(cf):
63
+ if k == 0:
64
+ continue
65
+ if (e * d - 1) % k != 0:
66
+ continue
67
+ phi = (e * d - 1) // k
68
+ s = n - phi + 1
69
+ disc = s * s - 4 * n
70
+ if disc < 0:
71
+ continue
72
+ t = math.isqrt(disc)
73
+ if t * t != disc:
74
+ continue
75
+ if (s + t) % 2 == 0:
76
+ return d
77
+ return None
78
+
79
+
80
+ def hastad_broadcast(ns: list[int], e: int, cts: list[int]) -> int | None:
81
+ """Hastad broadcast attack: same m, small e, e+ ciphertexts under different moduli."""
82
+ if len(ns) < e or len(cts) < e:
83
+ return None
84
+ crt_val = _crt(cts[:e], ns[:e])
85
+ m, exact = int_nthroot(crt_val, e)
86
+ return m if exact else None
87
+
88
+
89
+ def _crt(remainders: list[int], moduli: list[int]) -> int:
90
+ m_prod = 1
91
+ for m in moduli:
92
+ m_prod *= m
93
+ result = 0
94
+ for r, m in zip(remainders, moduli):
95
+ p = m_prod // m
96
+ result += r * modinv(p, m) * p
97
+ return result % m_prod
98
+
99
+
100
+ def common_modulus(n: int, e1: int, e2: int, c1: int, c2: int) -> int | None:
101
+ """Recover m when the same m was encrypted under the same n with coprime exponents."""
102
+ g, a, b = egcd(e1, e2)
103
+ if g != 1:
104
+ return None
105
+ if a < 0:
106
+ c1 = modinv(c1, n)
107
+ a = -a
108
+ if b < 0:
109
+ c2 = modinv(c2, n)
110
+ b = -b
111
+ return (pow(c1, a, n) * pow(c2, b, n)) % n
112
+
113
+
114
+ def small_exponent_decrypt(n: int, e: int, c: int) -> int | None:
115
+ """Decrypt when m^e < n (no modular wraparound happened)."""
116
+ m, exact = int_nthroot(c, e)
117
+ return m if exact and pow(m, e, n) == c else None
118
+
119
+
120
+ def int_to_bytes(n: int) -> bytes:
121
+ if n == 0:
122
+ return b"\x00"
123
+ return n.to_bytes((n.bit_length() + 7) // 8, "big")
@@ -0,0 +1,96 @@
1
+ """XOR cryptanalysis: single byte bruteforce and repeating key break."""
2
+ from .analysis import english_score
3
+
4
+
5
+ def xor_bytes(data: bytes, key: bytes) -> bytes:
6
+ return bytes(b ^ key[i % len(key)] for i, b in enumerate(data))
7
+
8
+
9
+ def single_byte_xor_break(data: bytes, top: int = 5) -> list[tuple[int, bytes, float]]:
10
+ """Bruteforce all 256 single byte keys, return top scoring (key, plaintext, score)."""
11
+ results = []
12
+ for key in range(256):
13
+ pt = bytes(b ^ key for b in data)
14
+ results.append((key, pt, english_score(pt.decode("latin1"))))
15
+ results.sort(key=lambda x: -x[2])
16
+ return results[:top]
17
+
18
+
19
+ def hamming_distance(a: bytes, b: bytes) -> int:
20
+ return sum(bin(x ^ y).count("1") for x, y in zip(a, b))
21
+
22
+
23
+ def guess_keysize(data: bytes, max_keysize: int = 40, top: int = 3) -> list[tuple[int, float]]:
24
+ """Guess repeating xor keysize via normalized hamming distance. Lower is better."""
25
+ guesses = []
26
+ for ks in range(2, min(max_keysize + 1, len(data) // 8 + 1)):
27
+ nblocks = min(8, len(data) // ks)
28
+ blocks = [data[i * ks:(i + 1) * ks] for i in range(nblocks)]
29
+ dists = [hamming_distance(blocks[i], blocks[i + 1]) / ks for i in range(nblocks - 1)]
30
+ guesses.append((ks, sum(dists) / len(dists)))
31
+ guesses.sort(key=lambda x: x[1])
32
+ return guesses[:top]
33
+
34
+
35
+ def _refine_key(data: bytes, key: bytes) -> tuple[bytes, float]:
36
+ """Hill climb each key byte against the full plaintext english score."""
37
+ best = bytearray(key)
38
+ best_score = english_score(xor_bytes(data, bytes(best)).decode("latin1"))
39
+ improved = True
40
+ while improved:
41
+ improved = False
42
+ for i in range(len(best)):
43
+ for b in range(256):
44
+ if b == best[i]:
45
+ continue
46
+ cand = bytearray(best)
47
+ cand[i] = b
48
+ s = english_score(xor_bytes(data, bytes(cand)).decode("latin1"))
49
+ if s > best_score:
50
+ best, best_score = cand, s
51
+ improved = True
52
+ return bytes(best), best_score
53
+
54
+
55
+ def _partial_score(data: bytes, key: bytes, keysize: int) -> float:
56
+ """Score text decrypted with a partial key (first len(key) bytes of each keysize block)."""
57
+ m = len(key)
58
+ dec = bytearray()
59
+ for t in range(len(data)):
60
+ if t % keysize < m:
61
+ dec.append(data[t] ^ key[t % keysize])
62
+ if not dec:
63
+ return float("-inf")
64
+ return english_score(bytes(dec).decode("latin1"))
65
+
66
+
67
+ def _beam_search_key(data: bytes, keysize: int, width: int = 50, per_block: int = 8) -> bytes:
68
+ """Beam search over per-block key candidates, scored on progressively decrypted text."""
69
+ cands = [
70
+ [k for k, _, _ in single_byte_xor_break(data[i::keysize], top=per_block)]
71
+ for i in range(keysize)
72
+ ]
73
+ beam: list[bytes] = [b""]
74
+ for i in range(keysize):
75
+ expanded = []
76
+ for partial in beam:
77
+ for b in cands[i]:
78
+ full = partial + bytes([b])
79
+ expanded.append((full, _partial_score(data, full, keysize)))
80
+ expanded.sort(key=lambda x: -x[1])
81
+ beam = [k for k, _ in expanded[:width]]
82
+ return beam[0]
83
+
84
+
85
+ def repeating_key_xor_break(data: bytes, max_keysize: int = 40) -> list[tuple[bytes, bytes, float]]:
86
+ """Break repeating key xor. Returns (key, plaintext, score) for top keysize guesses."""
87
+ results = []
88
+ for keysize, _ in guess_keysize(data, max_keysize, top=6):
89
+ if keysize <= 12:
90
+ key = _beam_search_key(data, keysize)
91
+ else:
92
+ key = bytes(single_byte_xor_break(data[i::keysize], top=1)[0][0] for i in range(keysize))
93
+ key, score = _refine_key(data, key)
94
+ results.append((key, xor_bytes(data, key), score))
95
+ results.sort(key=lambda x: -x[2])
96
+ return results
@@ -0,0 +1,67 @@
1
+ Metadata-Version: 2.4
2
+ Name: cryptsmith
3
+ Version: 0.1.0
4
+ Summary: Crypto attack primitives and analysis toolkit for CTF players and security researchers. Zero dependencies.
5
+ Author: Jake Burre
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/burrejak22/cryptsmith
8
+ Project-URL: Repository, https://github.com/burrejak22/cryptsmith
9
+ Keywords: crypto,ctf,cryptanalysis,rsa,pcap,security
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Information Technology
12
+ Classifier: Topic :: Security :: Cryptography
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Dynamic: license-file
19
+
20
+ # cryptsmith
21
+
22
+ Crypto attack primitives and analysis toolkit for CTF players and security researchers. Zero dependencies, stdlib only.
23
+
24
+ Built for challenges like NSA Codebreakers style pcaps: custom protocols, hidden enrollment keys, obfuscated traffic.
25
+
26
+ ## What it does
27
+
28
+ * **Classical ciphers** — caesar bruteforce with english scoring, vigenere, atbash, affine
29
+ * **XOR** — single byte bruteforce, repeating key break via hamming distance keysize guessing
30
+ * **RSA attacks** — Wiener's attack, Hastad broadcast, common modulus, small exponent
31
+ * **Analysis** — shannon entropy, index of coincidence, chi squared english scoring, frequency analysis
32
+ * **Encoding** — layered auto decode for base64, hex, url encoding, with the decode chain reported
33
+ * **Hash ID** — identify hash algorithms by digest length and charset
34
+ * **PCAP** — pure python pcap reader, IPv4/TCP parsing, TCP stream reassembly, payload carving, token hunting
35
+
36
+ ## Install
37
+
38
+ ```bash
39
+ pip install cryptsmith
40
+ ```
41
+
42
+ ## Quick use
43
+
44
+ ```python
45
+ from cryptsmith import xor, rsa, pcap, analysis
46
+
47
+ # break single byte xor
48
+ for key, plaintext, score in xor.single_byte_xor_break(data)[:3]:
49
+ print(key, plaintext)
50
+
51
+ # wiener attack on weak rsa
52
+ d = rsa.wiener_attack(n, e)
53
+
54
+ # pull tcp streams out of a pcap and hunt 12 char alpha tokens
55
+ streams = pcap.reassemble_streams(pcap.PcapReader("capture.pcap"))
56
+ for s in streams:
57
+ for token in pcap.hunt_alpha_tokens(s.payload, length=12):
58
+ print(s.label, token)
59
+ ```
60
+
61
+ ## Companion project
62
+
63
+ **flaghunter** builds on cryptsmith: point it at a challenge file or a pcap and it runs a full battery of attacks automatically and writes you a report.
64
+
65
+ ## License
66
+
67
+ MIT
@@ -0,0 +1,16 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ src/cryptsmith/__init__.py
5
+ src/cryptsmith/analysis.py
6
+ src/cryptsmith/classical.py
7
+ src/cryptsmith/encoding.py
8
+ src/cryptsmith/hashid.py
9
+ src/cryptsmith/pcap.py
10
+ src/cryptsmith/rsa.py
11
+ src/cryptsmith/xor.py
12
+ src/cryptsmith.egg-info/PKG-INFO
13
+ src/cryptsmith.egg-info/SOURCES.txt
14
+ src/cryptsmith.egg-info/dependency_links.txt
15
+ src/cryptsmith.egg-info/top_level.txt
16
+ tests/test_cryptsmith.py
@@ -0,0 +1 @@
1
+ cryptsmith
@@ -0,0 +1,154 @@
1
+ import base64
2
+ import os
3
+ import struct
4
+ import tempfile
5
+
6
+ from cryptsmith import analysis, classical, encoding, hashid, pcap, rsa, xor
7
+
8
+
9
+ def test_caesar_roundtrip_and_bruteforce():
10
+ pt = "the quick brown fox jumps over the lazy dog"
11
+ ct = classical.caesar_encrypt(pt, 7)
12
+ assert classical.caesar_decrypt(ct, 7) == pt
13
+ top = classical.caesar_bruteforce(ct, top=1)[0]
14
+ assert top[0] == 7
15
+ assert "quick brown fox" in top[1]
16
+
17
+
18
+ def test_atbash():
19
+ assert classical.atbash("hello") == "svool"
20
+ assert classical.atbash(classical.atbash("Hello World")) == "Hello World"
21
+
22
+
23
+ def test_vigenere_roundtrip():
24
+ pt = "attack at dawn"
25
+ ct = classical.vigenere_encrypt(pt, "lemon")
26
+ assert classical.vigenere_decrypt(ct, "lemon") == pt
27
+
28
+
29
+ def test_affine_bruteforce():
30
+ pt = "this is a secret message hidden well"
31
+ a, b = 5, 8
32
+ enc = "".join(
33
+ chr((a * (ord(c) - 97) + b) % 26 + 97) if c.isalpha() else c for c in pt
34
+ )
35
+ best = classical.affine_bruteforce(enc, top=1)[0]
36
+ assert best[0] == a and best[1] == b
37
+ assert best[2] == pt
38
+
39
+
40
+ def test_single_byte_xor_break():
41
+ pt = b"the enrollment key is hidden in this packet capture data"
42
+ ct = xor.xor_bytes(pt, b"\x42")
43
+ key, recovered, score = xor.single_byte_xor_break(ct, top=1)[0]
44
+ assert key == 0x42
45
+ assert recovered == pt
46
+
47
+
48
+ def test_repeating_key_xor_break():
49
+ pt = (b"the quick brown fox jumps over the lazy dog and then some more text here "
50
+ b"so the sample is long enough for keysize detection to work well")
51
+ key = b"SECRET"
52
+ ct = xor.xor_bytes(pt, key)
53
+ guesses = xor.repeating_key_xor_break(ct)
54
+ assert guesses[0][0] == key
55
+ assert guesses[0][1] == pt
56
+
57
+
58
+ def test_wiener_attack():
59
+ # d must be small: d < n^0.25 / 3
60
+ p, q = 100003, 100019
61
+ n = p * q
62
+ phi = (p - 1) * (q - 1)
63
+ d = 53
64
+ e = pow(d, -1, phi)
65
+ assert rsa.wiener_attack(n, e) == d
66
+
67
+
68
+ def test_hastad_broadcast():
69
+ m = 500
70
+ e = 3
71
+ ns = [10007 * 10009, 10037 * 10039, 10061 * 10067]
72
+ cts = [pow(m, e, n) for n in ns]
73
+ assert rsa.hastad_broadcast(ns, e, cts) == m
74
+
75
+
76
+ def test_common_modulus():
77
+ p, q = 101, 103
78
+ n = p * q
79
+ m = 1234
80
+ e1, e2 = 7, 11
81
+ c1, c2 = pow(m, e1, n), pow(m, e2, n)
82
+ assert rsa.common_modulus(n, e1, e2, c1, c2) == m
83
+
84
+
85
+ def test_entropy_and_ioc():
86
+ assert analysis.shannon_entropy(b"a" * 100) < 1.0
87
+ assert analysis.shannon_entropy(os.urandom(1000)) > 7.0
88
+ english = ("the enrollment key was hidden inside the packet capture "
89
+ "and the analyst needed to find it before the server responded ") * 5
90
+ assert analysis.index_of_coincidence(english) > 0.06
91
+
92
+
93
+ def test_smart_decode_layered():
94
+ inner = base64.b64encode(b"hello world").decode()
95
+ outer = base64.b64encode(inner.encode()).decode()
96
+ chains = encoding.smart_decode(outer.encode())
97
+ labels = [c[0] for c in chains]
98
+ assert "b64" in labels
99
+ assert "b64+b64" in labels
100
+ assert any(c[1] == b"hello world" for c in chains)
101
+
102
+
103
+ def test_hashid():
104
+ assert "md5" in hashid.identify("5d41402abc4b2a76b9719d911017c592")
105
+ assert "sha256" in hashid.identify("a" * 64)
106
+ assert hashid.identify("xyz") == []
107
+
108
+
109
+ def _build_pcap(path):
110
+ def frame(src_ip, dst_ip, sport, dport, seq, payload):
111
+ eth = b"\x00" * 12 + struct.pack("!H", 0x0800)
112
+ ver_ihl, tos = 0x45, 0
113
+ total = 20 + 20 + len(payload)
114
+ ip = struct.pack("!BBHHHBBH4s4s", ver_ihl, tos, total, 0, 0, 64, 6, 0,
115
+ bytes(map(int, src_ip.split("."))), bytes(map(int, dst_ip.split("."))))
116
+ tcp = struct.pack("!HHIIHHHH", sport, dport, seq, 0, 0x5000, 0, 0, 0)
117
+ return eth + ip + tcp + payload
118
+
119
+ with open(path, "wb") as f:
120
+ f.write(struct.pack("<IHHIIII", 0xA1B2C3D4, 2, 4, 0, 0, 65535, 1))
121
+ pkts = [
122
+ frame("10.0.0.5", "10.0.0.9", 1337, 9000, 1, b"HELLO distsrv"),
123
+ frame("10.0.0.5", "10.0.0.9", 1337, 9000, 14, b"ENROLL QwErTyUiOpAs"),
124
+ frame("10.0.0.9", "10.0.0.5", 9000, 1337, 1, b"OK welcome"),
125
+ ]
126
+ for p in pkts:
127
+ f.write(struct.pack("<IIII", 0, 0, len(p), len(p)) + p)
128
+
129
+
130
+ def test_xor_token_scan_finds_obfuscated_key():
131
+ key = b"KxQwErTyUiOp"
132
+ blob = b"\x03\x10" + bytes(b ^ 0x42 for b in b"ENROLL:" + key) + b"\x05\x00GET /x"
133
+ hits = pcap.xor_token_scan(blob, min_length=12)
134
+ assert hits, "expected to find the xor hidden key"
135
+ assert any(k == 0x42 and key.decode() in t for _, k, t in hits)
136
+ # no trivial keys: plaintext words should not report under 0x00 or 0x20
137
+ plain_hits = pcap.xor_token_scan(b"xx communication yy", min_length=12)
138
+ assert all(k not in (0x00, 0x20) for _, k, _ in plain_hits)
139
+
140
+
141
+ def test_pcap_reassembly_and_token_hunt():
142
+ with tempfile.NamedTemporaryFile(suffix=".pcap", delete=False) as t:
143
+ path = t.name
144
+ try:
145
+ _build_pcap(path)
146
+ streams = pcap.reassemble_streams(pcap.PcapReader(path))
147
+ assert len(streams) == 1
148
+ s = streams[0]
149
+ assert b"HELLO distsrv" in s.a_to_b
150
+ assert b"OK welcome" in s.b_to_a
151
+ assert pcap.hunt_alpha_tokens(s.payload, length=12) == ["QwErTyUiOpAs"]
152
+ assert any("HELLO" in st for st in pcap.carve_strings(s.payload))
153
+ finally:
154
+ os.unlink(path)