iotsploit-fuzzer 0.0.9__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iotsploit_fuzzer/__init__.py +77 -0
- iotsploit_fuzzer/analysis/__init__.py +1 -0
- iotsploit_fuzzer/analysis/corpus.py +480 -0
- iotsploit_fuzzer/analysis/logger.py +54 -0
- iotsploit_fuzzer/analysis/outcome.py +273 -0
- iotsploit_fuzzer/core/__init__.py +67 -0
- iotsploit_fuzzer/core/bit_manipulator.py +405 -0
- iotsploit_fuzzer/core/config.py +41 -0
- iotsploit_fuzzer/core/fuzzing_engine.py +469 -0
- iotsploit_fuzzer/core/orchestrator.py +452 -0
- iotsploit_fuzzer/core/parser_campaign.py +380 -0
- iotsploit_fuzzer/core/strategies/__init__.py +43 -0
- iotsploit_fuzzer/core/strategies/bit_strategies.py +509 -0
- iotsploit_fuzzer/core/strategies/field_strategies.py +697 -0
- iotsploit_fuzzer/fuzz.py +276 -0
- iotsploit_fuzzer/generators/__init__.py +1 -0
- iotsploit_fuzzer/generators/base.py +14 -0
- iotsploit_fuzzer/generators/corpus_generator.py +219 -0
- iotsploit_fuzzer/generators/radamsa_generator.py +96 -0
- iotsploit_fuzzer/harnesses/__init__.py +1 -0
- iotsploit_fuzzer/harnesses/base.py +27 -0
- iotsploit_fuzzer/harnesses/can_harness.py +23 -0
- iotsploit_fuzzer/harnesses/parser_harness.py +323 -0
- iotsploit_fuzzer/harnesses/parser_targets.py +142 -0
- iotsploit_fuzzer/harnesses/parser_worker.py +180 -0
- iotsploit_fuzzer/harnesses/spi_harness.py +17 -0
- iotsploit_fuzzer/harnesses/uart_harness.py +23 -0
- iotsploit_fuzzer/interfaces/__init__.py +1 -0
- iotsploit_fuzzer/interfaces/base.py +17 -0
- iotsploit_fuzzer/interfaces/can_interface.py +48 -0
- iotsploit_fuzzer/interfaces/spi_interface.py +27 -0
- iotsploit_fuzzer/interfaces/uart_interface.py +26 -0
- iotsploit_fuzzer/monitoring/__init__.py +1 -0
- iotsploit_fuzzer/monitoring/boundary_monitor.py +187 -0
- iotsploit_fuzzer/monitoring/monitor.py +203 -0
- iotsploit_fuzzer/targets/__init__.py +1 -0
- iotsploit_fuzzer/targets/iotsploit.py +1004 -0
- iotsploit_fuzzer-0.0.9.dist-info/METADATA +320 -0
- iotsploit_fuzzer-0.0.9.dist-info/RECORD +40 -0
- iotsploit_fuzzer-0.0.9.dist-info/WHEEL +4 -0
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""
|
|
2
|
+
IoT Protocol Fuzzer
|
|
3
|
+
|
|
4
|
+
A modular fuzzing framework for IoT communication protocols including CAN, UART, and SPI.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
__version__ = "0.0.9"
|
|
8
|
+
__author__ = "IoT Security Research"
|
|
9
|
+
__description__ = "Modular fuzzing framework for IoT protocols"
|
|
10
|
+
|
|
11
|
+
# Make common imports available at package level
|
|
12
|
+
from .core.orchestrator import Orchestrator, CampaignConfig
|
|
13
|
+
from .core.bit_manipulator import BitManipulator, flip_bit, set_bit, get_bit, parse_target_bits
|
|
14
|
+
from .core.fuzzing_engine import (
|
|
15
|
+
FuzzingEngine, StrategyRegistry, FuzzingStrategy, FuzzingType,
|
|
16
|
+
FuzzTestCase, MutationResult, default_engine, default_registry
|
|
17
|
+
)
|
|
18
|
+
from .generators.radamsa_generator import RadamsaGenerator
|
|
19
|
+
from .analysis.corpus import CorpusStore
|
|
20
|
+
from .harnesses.can_harness import CANHarness
|
|
21
|
+
from .harnesses.parser_harness import ParserHarness
|
|
22
|
+
from .harnesses.parser_targets import ParseTarget, register
|
|
23
|
+
from .monitoring.boundary_monitor import BoundaryMonitor
|
|
24
|
+
from .harnesses.uart_harness import UARTHarness
|
|
25
|
+
from .harnesses.spi_harness import SPIHarness
|
|
26
|
+
|
|
27
|
+
# Import strategies
|
|
28
|
+
try:
|
|
29
|
+
from .core.strategies.bit_strategies import (
|
|
30
|
+
BitFlipStrategy, SequentialBitStrategy, RandomBitStrategy
|
|
31
|
+
)
|
|
32
|
+
from .core.strategies.field_strategies import (
|
|
33
|
+
FieldMutationStrategy, BoundaryValueStrategy, InjectionStrategy
|
|
34
|
+
)
|
|
35
|
+
except ImportError:
|
|
36
|
+
# Strategies may not be available in all environments
|
|
37
|
+
pass
|
|
38
|
+
|
|
39
|
+
__all__ = [
|
|
40
|
+
# Original components
|
|
41
|
+
"Orchestrator",
|
|
42
|
+
"CampaignConfig",
|
|
43
|
+
"BitManipulator",
|
|
44
|
+
"flip_bit",
|
|
45
|
+
"set_bit",
|
|
46
|
+
"get_bit",
|
|
47
|
+
"parse_target_bits",
|
|
48
|
+
"RadamsaGenerator",
|
|
49
|
+
"CANHarness",
|
|
50
|
+
"UARTHarness",
|
|
51
|
+
"SPIHarness",
|
|
52
|
+
|
|
53
|
+
# Inbound parser fuzzing: the same loop pointed at our own parsers
|
|
54
|
+
"ParserHarness",
|
|
55
|
+
"ParseTarget",
|
|
56
|
+
"register",
|
|
57
|
+
"CorpusStore",
|
|
58
|
+
"BoundaryMonitor",
|
|
59
|
+
|
|
60
|
+
# Fuzzing engine components
|
|
61
|
+
"FuzzingEngine",
|
|
62
|
+
"StrategyRegistry",
|
|
63
|
+
"FuzzingStrategy",
|
|
64
|
+
"FuzzingType",
|
|
65
|
+
"FuzzTestCase",
|
|
66
|
+
"MutationResult",
|
|
67
|
+
"default_engine",
|
|
68
|
+
"default_registry",
|
|
69
|
+
|
|
70
|
+
# Fuzzing strategies
|
|
71
|
+
"BitFlipStrategy",
|
|
72
|
+
"SequentialBitStrategy",
|
|
73
|
+
"RandomBitStrategy",
|
|
74
|
+
"FieldMutationStrategy",
|
|
75
|
+
"BoundaryValueStrategy",
|
|
76
|
+
"InjectionStrategy",
|
|
77
|
+
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Analysis and reporting
|
|
@@ -0,0 +1,480 @@
|
|
|
1
|
+
"""Content-addressed payloads and the boundary ledger that gives them meaning.
|
|
2
|
+
|
|
3
|
+
The fuzzer's existing ``artifacts/`` directory keys files on the iteration
|
|
4
|
+
index, so campaign N+1 overwrites campaign N wherever the indices overlap, and
|
|
5
|
+
``case_500.bin`` and ``crash_500.bin`` are usually from different runs. That is
|
|
6
|
+
a graveyard, not a corpus: nothing is attributable and nothing accumulates.
|
|
7
|
+
|
|
8
|
+
Here a payload's identity is its content, and the ledger records what each one
|
|
9
|
+
*did* -- its normalised outcome signature -- under the fingerprint of the
|
|
10
|
+
oracle that judged it. That is what makes a diff possible, and a diff is the
|
|
11
|
+
thing that keeps producing information after the crashes run dry.
|
|
12
|
+
|
|
13
|
+
Writes are atomic and locked per target. A nightly run that is killed halfway
|
|
14
|
+
leaves the previous ledger intact and readable.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
import json
|
|
21
|
+
import logging
|
|
22
|
+
import os
|
|
23
|
+
import zipfile
|
|
24
|
+
from contextlib import contextmanager
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from datetime import datetime, timezone
|
|
27
|
+
from io import BytesIO
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
from typing import Dict, Iterator, Optional, Tuple
|
|
30
|
+
|
|
31
|
+
from ..harnesses.parser_targets import ParseTarget
|
|
32
|
+
from .outcome import Outcome
|
|
33
|
+
|
|
34
|
+
logger = logging.getLogger("fuzzer.corpus")
|
|
35
|
+
|
|
36
|
+
try: # POSIX advisory locking; elsewhere the lock is best-effort.
|
|
37
|
+
import fcntl
|
|
38
|
+
except ImportError: # pragma: no cover - exercised only off POSIX
|
|
39
|
+
fcntl = None
|
|
40
|
+
|
|
41
|
+
#: Bumped when the on-disk shape changes. A ledger from an older version is
|
|
42
|
+
#: re-baselined rather than guessed at.
|
|
43
|
+
LEDGER_VERSION = 1
|
|
44
|
+
|
|
45
|
+
#: Payloads kept per distinct signature. Without a cap the corpus fills with
|
|
46
|
+
#: near-identical inputs that all prove the same point, and the gate replay
|
|
47
|
+
#: that has to run on every commit slows down for nothing.
|
|
48
|
+
MAX_EXEMPLARS = 3
|
|
49
|
+
|
|
50
|
+
#: Payloads kept per distinct accept/reject edge. Lower than the signature cap
|
|
51
|
+
#: because an edge is already a pair, and because crossing the line is the
|
|
52
|
+
#: common case for a mutator -- uncapped, this alone grows without bound.
|
|
53
|
+
MAX_EDGE_EXEMPLARS = 1
|
|
54
|
+
|
|
55
|
+
#: Absolute ceiling per signature, whatever the reason for keeping it. Edges
|
|
56
|
+
#: are capped per *pair*, so a signature reachable from many parents could
|
|
57
|
+
#: otherwise accumulate one exemplar per parent and escape both caps.
|
|
58
|
+
MAX_PER_SIGNATURE = 5
|
|
59
|
+
|
|
60
|
+
#: Distinct signatures one target may hold. The per-signature caps bound how
|
|
61
|
+
#: many payloads each behaviour keeps, but not how many behaviours there are,
|
|
62
|
+
#: and a result shape with six bucketed fields has a combinatorially large
|
|
63
|
+
#: space. Sized generously: reaching it means the shape is too fine-grained to
|
|
64
|
+
#: be a boundary, which is worth reporting rather than absorbing.
|
|
65
|
+
MAX_SIGNATURES = 200
|
|
66
|
+
|
|
67
|
+
#: Campaign manifests kept per target. Enough to compare a finding against the
|
|
68
|
+
#: runs around it; bounded so a nightly loop does not accumulate a manifest a
|
|
69
|
+
#: day for ever.
|
|
70
|
+
KEEP_MANIFESTS = 10
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class CorpusDamagedError(RuntimeError):
|
|
74
|
+
"""The ledger and the archive do not agree, so replay coverage is unknown.
|
|
75
|
+
|
|
76
|
+
Raised rather than worked around. A replay that silently skips the
|
|
77
|
+
payloads it cannot find still reports success, and a green gate that
|
|
78
|
+
checked fewer inputs than it claims to is worse than a red one.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def payload_id(payload: bytes) -> str:
|
|
83
|
+
"""A payload's identity: what it is, never where it appeared."""
|
|
84
|
+
return hashlib.sha256(payload).hexdigest()[:16]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass
|
|
88
|
+
class LedgerEntry:
|
|
89
|
+
"""One retained payload, and what it did the last time anyone looked."""
|
|
90
|
+
|
|
91
|
+
signature: str
|
|
92
|
+
kind: str
|
|
93
|
+
first_seen: str
|
|
94
|
+
#: The parent's signature, when this payload was retained because one
|
|
95
|
+
#: mutation moved it across the accept/reject line. Empty otherwise. The
|
|
96
|
+
#: pair is the point: one payload alone does not locate a boundary.
|
|
97
|
+
edge_of: str = ""
|
|
98
|
+
|
|
99
|
+
def to_dict(self) -> Dict[str, object]:
|
|
100
|
+
return {
|
|
101
|
+
"signature": self.signature,
|
|
102
|
+
"kind": self.kind,
|
|
103
|
+
"first_seen": self.first_seen,
|
|
104
|
+
"edge_of": self.edge_of,
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
@classmethod
|
|
108
|
+
def from_dict(cls, raw: Dict[str, object]) -> "LedgerEntry":
|
|
109
|
+
return cls(
|
|
110
|
+
signature=str(raw.get("signature", "")),
|
|
111
|
+
kind=str(raw.get("kind", "")),
|
|
112
|
+
first_seen=str(raw.get("first_seen", "")),
|
|
113
|
+
edge_of=str(raw.get("edge_of", "")),
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@contextmanager
|
|
118
|
+
def _locked(path: Path) -> Iterator[None]:
|
|
119
|
+
"""Hold the per-target lock for the duration of a write."""
|
|
120
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
121
|
+
handle = os.open(str(path), os.O_CREAT | os.O_RDWR, 0o644)
|
|
122
|
+
try:
|
|
123
|
+
if fcntl is not None:
|
|
124
|
+
fcntl.flock(handle, fcntl.LOCK_EX)
|
|
125
|
+
yield
|
|
126
|
+
finally:
|
|
127
|
+
if fcntl is not None:
|
|
128
|
+
try:
|
|
129
|
+
fcntl.flock(handle, fcntl.LOCK_UN)
|
|
130
|
+
except OSError: # pragma: no cover - unlock of a dead fd
|
|
131
|
+
pass
|
|
132
|
+
os.close(handle)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _write_atomic(path: Path, data: bytes) -> None:
|
|
136
|
+
"""Replace a file in one step, or not at all.
|
|
137
|
+
|
|
138
|
+
The fsync matters more than it looks: without it a machine that loses
|
|
139
|
+
power mid-campaign can leave a renamed but empty ledger, which reads as
|
|
140
|
+
"this target has never been fuzzed" and silently discards months of
|
|
141
|
+
boundary history.
|
|
142
|
+
"""
|
|
143
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
144
|
+
temporary = path.with_name(f"{path.name}.tmp-{os.getpid()}")
|
|
145
|
+
with open(temporary, "wb") as handle:
|
|
146
|
+
handle.write(data)
|
|
147
|
+
handle.flush()
|
|
148
|
+
os.fsync(handle.fileno())
|
|
149
|
+
os.replace(temporary, path)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class CorpusStore:
|
|
153
|
+
"""The retained payloads for one target, and the ledger over them."""
|
|
154
|
+
|
|
155
|
+
def __init__(self, root: Path | str, target: ParseTarget) -> None:
|
|
156
|
+
self.target = target
|
|
157
|
+
self.root = Path(root) / target.name
|
|
158
|
+
#: One archive per target rather than one file per payload. The
|
|
159
|
+
#: corpus is a thousand-odd inputs of a hundred-odd bytes each, and
|
|
160
|
+
#: as loose files it was 74% of the repository's tracked file count
|
|
161
|
+
#: for 1% of its bytes -- plus a 4 KB block apiece, which turned
|
|
162
|
+
#: 1.1 MB of payloads into 9.4 MB on disk.
|
|
163
|
+
self.archive_path = self.root / "payloads.zip"
|
|
164
|
+
#: Where payloads used to live, read once so an existing corpus
|
|
165
|
+
#: migrates itself on the next save.
|
|
166
|
+
self.legacy_dir = self.root / "payloads"
|
|
167
|
+
self.ledger_path = self.root / "ledger.json"
|
|
168
|
+
self.lock_path = self.root / ".lock"
|
|
169
|
+
self.entries: Dict[str, LedgerEntry] = {}
|
|
170
|
+
#: True when the ledger was written under a different oracle. Diffing
|
|
171
|
+
#: is refused until a re-baseline: a changed declared-exception set
|
|
172
|
+
#: moves the accept/reject line by definition, and reporting that as
|
|
173
|
+
#: thousands of boundary movements teaches everyone to ignore reports.
|
|
174
|
+
self.stale = False
|
|
175
|
+
self.load()
|
|
176
|
+
|
|
177
|
+
# -- reading -----------------------------------------------------------
|
|
178
|
+
|
|
179
|
+
def load(self) -> None:
|
|
180
|
+
self.entries = {}
|
|
181
|
+
#: Set when the ledger and the archive disagree. Not raised here -- a
|
|
182
|
+
#: caller may want to inspect a damaged corpus -- but anything that
|
|
183
|
+
#: replays or extends it must refuse. Assigned before the payloads
|
|
184
|
+
#: are read, because reading them is one of the things that can
|
|
185
|
+
#: discover the damage.
|
|
186
|
+
self.damaged = ""
|
|
187
|
+
self._payloads: Dict[str, bytes] = self._read_payloads()
|
|
188
|
+
self._counts: Dict[str, int] = {}
|
|
189
|
+
self.stale = False
|
|
190
|
+
#: Set when a new signature was turned away by MAX_SIGNATURES.
|
|
191
|
+
self.saturated = False
|
|
192
|
+
if not self.ledger_path.exists():
|
|
193
|
+
return
|
|
194
|
+
try:
|
|
195
|
+
raw = json.loads(self.ledger_path.read_text())
|
|
196
|
+
except (OSError, ValueError) as error:
|
|
197
|
+
logger.warning("ledger for %s is unreadable (%s)", self.target.name, error)
|
|
198
|
+
self.stale = True
|
|
199
|
+
return
|
|
200
|
+
if raw.get("ledger_version") != LEDGER_VERSION:
|
|
201
|
+
logger.warning("ledger for %s was written by an older version", self.target.name)
|
|
202
|
+
self.stale = True
|
|
203
|
+
if raw.get("fingerprint") != self.target.fingerprint:
|
|
204
|
+
logger.warning(
|
|
205
|
+
"ledger for %s was recorded against oracle %s, now %s",
|
|
206
|
+
self.target.name, raw.get("fingerprint"), self.target.fingerprint,
|
|
207
|
+
)
|
|
208
|
+
self.stale = True
|
|
209
|
+
self.entries = {
|
|
210
|
+
str(key): LedgerEntry.from_dict(value)
|
|
211
|
+
for key, value in (raw.get("entries") or {}).items()
|
|
212
|
+
if isinstance(value, dict)
|
|
213
|
+
}
|
|
214
|
+
for entry in self.entries.values():
|
|
215
|
+
self._counts[entry.signature] = self._counts.get(entry.signature, 0) + 1
|
|
216
|
+
self._check_archive_agrees()
|
|
217
|
+
|
|
218
|
+
def _check_archive_agrees(self) -> None:
|
|
219
|
+
"""The ledger and the archive are two files, written one after the
|
|
220
|
+
other under one lock -- which bounds the window but does not remove
|
|
221
|
+
it. A process killed between the two renames leaves a mixed pair.
|
|
222
|
+
|
|
223
|
+
A payload the ledger names and the archive does not have is the case
|
|
224
|
+
that must fail: ``payloads()`` would skip it, the replay would check
|
|
225
|
+
fewer inputs than the ledger claims, and it would still pass. The
|
|
226
|
+
other direction is harmless -- an unreferenced payload costs space
|
|
227
|
+
and nothing else -- so it is reported and ignored.
|
|
228
|
+
"""
|
|
229
|
+
if self.damaged:
|
|
230
|
+
return
|
|
231
|
+
named = set(self.entries)
|
|
232
|
+
held = set(self._payloads)
|
|
233
|
+
missing = named - held
|
|
234
|
+
if missing:
|
|
235
|
+
self.damaged = (
|
|
236
|
+
f"{len(missing)} payload(s) named by the ledger are not in the "
|
|
237
|
+
f"archive, e.g. {sorted(missing)[0]}"
|
|
238
|
+
)
|
|
239
|
+
return
|
|
240
|
+
orphaned = held - named
|
|
241
|
+
if orphaned:
|
|
242
|
+
logger.warning(
|
|
243
|
+
"%s: %d payload(s) in the archive are not in the ledger; ignoring",
|
|
244
|
+
self.target.name, len(orphaned),
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
def _read_payloads(self) -> Dict[str, bytes]:
|
|
248
|
+
"""Every retained payload, from the archive and any loose leftovers."""
|
|
249
|
+
found: Dict[str, bytes] = {}
|
|
250
|
+
if self.legacy_dir.is_dir():
|
|
251
|
+
for path in self.legacy_dir.glob("*.bin"):
|
|
252
|
+
found[path.stem] = path.read_bytes()
|
|
253
|
+
if self.archive_path.exists():
|
|
254
|
+
try:
|
|
255
|
+
with zipfile.ZipFile(self.archive_path) as archive:
|
|
256
|
+
for name in archive.namelist():
|
|
257
|
+
found[Path(name).stem] = archive.read(name)
|
|
258
|
+
except (OSError, zipfile.BadZipFile) as error:
|
|
259
|
+
# An unreadable archive is not an empty corpus. Treating it as
|
|
260
|
+
# one is how a replay reports success having checked nothing.
|
|
261
|
+
self.damaged = f"payloads.zip cannot be read: {error}"
|
|
262
|
+
return found
|
|
263
|
+
|
|
264
|
+
def payload(self, identity: str) -> Optional[bytes]:
|
|
265
|
+
return self._payloads.get(identity)
|
|
266
|
+
|
|
267
|
+
def payloads(self) -> Iterator[Tuple[str, bytes]]:
|
|
268
|
+
"""Every retained payload, for a replay or a new campaign's seeds."""
|
|
269
|
+
for identity in sorted(self.entries):
|
|
270
|
+
data = self.payload(identity)
|
|
271
|
+
if data is not None:
|
|
272
|
+
yield identity, data
|
|
273
|
+
|
|
274
|
+
def signature_counts(self) -> Dict[str, int]:
|
|
275
|
+
return dict(self._counts)
|
|
276
|
+
|
|
277
|
+
def known_signatures(self) -> set:
|
|
278
|
+
"""Every signature the ledger holds. Kept as a set, not rebuilt: this
|
|
279
|
+
is asked once per case, and rebuilding it made the campaign quadratic
|
|
280
|
+
in its own corpus."""
|
|
281
|
+
return set(self._counts)
|
|
282
|
+
|
|
283
|
+
# -- writing -----------------------------------------------------------
|
|
284
|
+
|
|
285
|
+
def admit(
|
|
286
|
+
self, payload: bytes, outcome: Outcome, campaign: str, *, edge_of: str = ""
|
|
287
|
+
) -> bool:
|
|
288
|
+
"""Retain a payload if this signature still has room, else discard it.
|
|
289
|
+
|
|
290
|
+
Returns whether it was kept -- a payload already in the ledger is not
|
|
291
|
+
kept again. What it used to do is left exactly as it is: that record
|
|
292
|
+
is what a boundary diff is made against, and only the monitor that
|
|
293
|
+
reported the movement may change it.
|
|
294
|
+
"""
|
|
295
|
+
identity = payload_id(payload)
|
|
296
|
+
entry = self.entries.get(identity)
|
|
297
|
+
if entry is not None:
|
|
298
|
+
if not entry.signature:
|
|
299
|
+
# Blanked by a re-baseline: this is the campaign re-deriving
|
|
300
|
+
# what the payload does under the new oracle. Without this the
|
|
301
|
+
# ledger never recovers -- every case stays "novel" forever
|
|
302
|
+
# because no signature is ever recorded again.
|
|
303
|
+
entry.signature = outcome.signature
|
|
304
|
+
entry.kind = outcome.kind
|
|
305
|
+
self._counts[outcome.signature] = self._counts.get(outcome.signature, 0) + 1
|
|
306
|
+
return False
|
|
307
|
+
if self._counts.get(outcome.signature, 0) >= MAX_PER_SIGNATURE:
|
|
308
|
+
return False
|
|
309
|
+
if outcome.signature not in self._counts and len(self._counts) >= MAX_SIGNATURES:
|
|
310
|
+
self.saturated = True
|
|
311
|
+
return False
|
|
312
|
+
if edge_of:
|
|
313
|
+
held = sum(
|
|
314
|
+
1 for entry in self.entries.values()
|
|
315
|
+
if entry.edge_of == edge_of and entry.signature == outcome.signature
|
|
316
|
+
)
|
|
317
|
+
cap = MAX_EDGE_EXEMPLARS
|
|
318
|
+
else:
|
|
319
|
+
held = self._counts.get(outcome.signature, 0)
|
|
320
|
+
cap = MAX_EXEMPLARS
|
|
321
|
+
if held >= cap and not self._displace(payload, outcome, edge_of):
|
|
322
|
+
return False
|
|
323
|
+
self._payloads[identity] = payload
|
|
324
|
+
self.entries[identity] = LedgerEntry(
|
|
325
|
+
signature=outcome.signature,
|
|
326
|
+
kind=outcome.kind,
|
|
327
|
+
first_seen=campaign,
|
|
328
|
+
edge_of=edge_of,
|
|
329
|
+
)
|
|
330
|
+
self._counts[outcome.signature] = self._counts.get(outcome.signature, 0) + 1
|
|
331
|
+
return True
|
|
332
|
+
|
|
333
|
+
def reclassify(self, identity: str, outcome: Outcome) -> None:
|
|
334
|
+
"""Record that a payload already held now does something else.
|
|
335
|
+
|
|
336
|
+
The one way an entry's signature may change after it is admitted, and
|
|
337
|
+
it lives here because the entry is not the only thing that has to
|
|
338
|
+
move: ``_counts`` backs ``known_signatures()``, the per-signature cap
|
|
339
|
+
and the saturation ceiling. The monitor used to assign to
|
|
340
|
+
``entry.signature`` directly, which left the ledger saying one thing
|
|
341
|
+
and the counts another -- a payload recorded as rejected while the
|
|
342
|
+
count that decides novelty still read accepted.
|
|
343
|
+
"""
|
|
344
|
+
entry = self.entries.get(identity)
|
|
345
|
+
if entry is None or entry.signature == outcome.signature:
|
|
346
|
+
return
|
|
347
|
+
if self._counts.get(entry.signature):
|
|
348
|
+
self._counts[entry.signature] -= 1
|
|
349
|
+
if not self._counts[entry.signature]:
|
|
350
|
+
del self._counts[entry.signature]
|
|
351
|
+
entry.signature = outcome.signature
|
|
352
|
+
entry.kind = outcome.kind
|
|
353
|
+
self._counts[outcome.signature] = self._counts.get(outcome.signature, 0) + 1
|
|
354
|
+
|
|
355
|
+
def _payload_size(self, identity: str) -> int:
|
|
356
|
+
return len(self._payloads.get(identity, b""))
|
|
357
|
+
|
|
358
|
+
def _displace(self, payload: bytes, outcome: Outcome, edge_of: str) -> bool:
|
|
359
|
+
"""Make room by dropping a larger payload that proves the same point.
|
|
360
|
+
|
|
361
|
+
Without this the corpus is bounded in entries but not in bytes, and
|
|
362
|
+
radamsa's repetition mutations exploit exactly that: a mutant grows,
|
|
363
|
+
is retained because its signature is new, becomes the next
|
|
364
|
+
generation's parent, and grows again. Measured from a 61-byte seed it
|
|
365
|
+
reached 567 KB in eight generations -- and radamsa's cost scales with
|
|
366
|
+
input size, so the loop makes itself slower as it runs.
|
|
367
|
+
|
|
368
|
+
Keeping the smallest example of each signature bounds the corpus in
|
|
369
|
+
bytes, keeps the gate replay fast, and hands triage the smallest
|
|
370
|
+
reproduction rather than whichever one happened to arrive first.
|
|
371
|
+
"""
|
|
372
|
+
candidates = [
|
|
373
|
+
identity for identity, entry in self.entries.items()
|
|
374
|
+
if entry.signature == outcome.signature
|
|
375
|
+
and (not edge_of or entry.edge_of == edge_of)
|
|
376
|
+
]
|
|
377
|
+
if not candidates:
|
|
378
|
+
return False
|
|
379
|
+
largest = max(candidates, key=self._payload_size)
|
|
380
|
+
if self._payload_size(largest) <= len(payload):
|
|
381
|
+
return False
|
|
382
|
+
self._drop(largest)
|
|
383
|
+
return True
|
|
384
|
+
|
|
385
|
+
def _drop(self, identity: str) -> None:
|
|
386
|
+
"""Remove one payload and the ledger's memory of it."""
|
|
387
|
+
entry = self.entries.pop(identity, None)
|
|
388
|
+
if entry is None:
|
|
389
|
+
return
|
|
390
|
+
if self._counts.get(entry.signature):
|
|
391
|
+
self._counts[entry.signature] -= 1
|
|
392
|
+
if not self._counts[entry.signature]:
|
|
393
|
+
del self._counts[entry.signature]
|
|
394
|
+
self._payloads.pop(identity, None)
|
|
395
|
+
|
|
396
|
+
def rebaseline(self) -> None:
|
|
397
|
+
"""Adopt the current oracle, discarding recorded signatures.
|
|
398
|
+
|
|
399
|
+
The payloads are kept -- they are the expensive part and are still
|
|
400
|
+
interesting inputs. Only the claim about what they *do* is dropped,
|
|
401
|
+
because it was made by a different oracle. The next campaign
|
|
402
|
+
re-derives it.
|
|
403
|
+
"""
|
|
404
|
+
for entry in self.entries.values():
|
|
405
|
+
entry.signature = ""
|
|
406
|
+
entry.kind = ""
|
|
407
|
+
self._counts = {}
|
|
408
|
+
self.stale = False
|
|
409
|
+
|
|
410
|
+
def _write_archive(self) -> None:
|
|
411
|
+
"""Rewrite the archive in one atomic step, holding only what the
|
|
412
|
+
ledger still names.
|
|
413
|
+
|
|
414
|
+
Written whole rather than appended to, because a payload displaced by
|
|
415
|
+
a smaller one has to leave.
|
|
416
|
+
|
|
417
|
+
This and the ledger are two renames under one lock. The lock keeps
|
|
418
|
+
two campaigns from interleaving; it does nothing about a kill between
|
|
419
|
+
the two writes, which can still leave a mixed pair. That is not
|
|
420
|
+
claimed to be transactional -- it is detected on load instead, by
|
|
421
|
+
:meth:`_check_archive_agrees`.
|
|
422
|
+
"""
|
|
423
|
+
buffer = BytesIO()
|
|
424
|
+
with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive:
|
|
425
|
+
for identity in sorted(self.entries):
|
|
426
|
+
payload = self._payloads.get(identity)
|
|
427
|
+
if payload is not None:
|
|
428
|
+
archive.writestr(f"{identity}.bin", payload)
|
|
429
|
+
_write_atomic(self.archive_path, buffer.getvalue())
|
|
430
|
+
if self.legacy_dir.is_dir():
|
|
431
|
+
for path in self.legacy_dir.glob("*.bin"):
|
|
432
|
+
path.unlink()
|
|
433
|
+
self.legacy_dir.rmdir()
|
|
434
|
+
|
|
435
|
+
def save(self, campaign: Optional[Dict[str, object]] = None) -> None:
|
|
436
|
+
"""Commit the ledger, and the manifest of the run that produced it."""
|
|
437
|
+
document = {
|
|
438
|
+
"ledger_version": LEDGER_VERSION,
|
|
439
|
+
"target": self.target.name,
|
|
440
|
+
"fingerprint": self.target.fingerprint,
|
|
441
|
+
"updated": datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
|
442
|
+
"entries": {k: v.to_dict() for k, v in sorted(self.entries.items())},
|
|
443
|
+
}
|
|
444
|
+
with _locked(self.lock_path):
|
|
445
|
+
if self._already_recorded(document):
|
|
446
|
+
# Nothing was learned, so nothing is written. A campaign that
|
|
447
|
+
# finds no new behaviour is the normal night, and rewriting
|
|
448
|
+
# the timestamp anyway put all 28 ledgers in every diff --
|
|
449
|
+
# which buries the one line that says a boundary moved, the
|
|
450
|
+
# thing tracking the corpus in git is for.
|
|
451
|
+
self._write_manifest(campaign)
|
|
452
|
+
return
|
|
453
|
+
self._write_archive()
|
|
454
|
+
_write_atomic(self.ledger_path, json.dumps(document, indent=2).encode())
|
|
455
|
+
self._write_manifest(campaign)
|
|
456
|
+
|
|
457
|
+
def _already_recorded(self, document: Dict[str, object]) -> bool:
|
|
458
|
+
"""Whether the ledger on disk already says this, timestamp aside."""
|
|
459
|
+
try:
|
|
460
|
+
existing = json.loads(self.ledger_path.read_text())
|
|
461
|
+
except (OSError, ValueError):
|
|
462
|
+
return False
|
|
463
|
+
comparable = dict(document)
|
|
464
|
+
comparable.pop("updated", None)
|
|
465
|
+
existing.pop("updated", None)
|
|
466
|
+
return existing == comparable
|
|
467
|
+
|
|
468
|
+
def _write_manifest(self, campaign: Optional[Dict[str, object]]) -> None:
|
|
469
|
+
"""The record of one run, which is written whether or not it learned
|
|
470
|
+
anything -- it is how a night that found nothing is distinguished
|
|
471
|
+
from a night that did not run."""
|
|
472
|
+
if not campaign:
|
|
473
|
+
return
|
|
474
|
+
manifests = self.root / "campaigns"
|
|
475
|
+
_write_atomic(
|
|
476
|
+
manifests / f"{campaign['campaign']}.json",
|
|
477
|
+
json.dumps(campaign, indent=2).encode(),
|
|
478
|
+
)
|
|
479
|
+
for stale in sorted(manifests.glob("*.json"))[:-KEEP_MANIFESTS]:
|
|
480
|
+
stale.unlink(missing_ok=True)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Callable, Optional
|
|
4
|
+
|
|
5
|
+
from ..harnesses.base import HarnessResult
|
|
6
|
+
from .corpus import payload_id
|
|
7
|
+
|
|
8
|
+
logger = logging.getLogger("fuzzer.logger")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class TestLogger:
|
|
12
|
+
"""Logs test cases to disk for later analysis, keyed on content.
|
|
13
|
+
|
|
14
|
+
Files were named ``case_{index}.bin``, which made a payload's identity its
|
|
15
|
+
position in a campaign: every run overwrote the previous run's cases
|
|
16
|
+
wherever the indices overlapped, ``case_500.bin`` and ``crash_500.bin``
|
|
17
|
+
came from different runs, and nothing on disk was attributable. Naming by
|
|
18
|
+
content hash instead makes a repeated payload one file and leaves earlier
|
|
19
|
+
campaigns intact.
|
|
20
|
+
|
|
21
|
+
``keep`` lets a caller that has its own notion of interesting -- the
|
|
22
|
+
parser loop, whose corpus is curated by signature -- stop routine cases
|
|
23
|
+
reaching disk at all, without giving up the crash record.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
def __init__(
|
|
27
|
+
self,
|
|
28
|
+
workdir: str = "artifacts",
|
|
29
|
+
keep: Optional[Callable[[bytes, HarnessResult], bool]] = None,
|
|
30
|
+
):
|
|
31
|
+
self.workdir = Path(workdir)
|
|
32
|
+
self.workdir.mkdir(parents=True, exist_ok=True)
|
|
33
|
+
self.keep = keep
|
|
34
|
+
self.total = 0
|
|
35
|
+
self.crashes = 0
|
|
36
|
+
|
|
37
|
+
def record(self, idx: int, payload: bytes, result: HarnessResult) -> None:
|
|
38
|
+
self.total += 1
|
|
39
|
+
if result.crashed:
|
|
40
|
+
self.crashes += 1
|
|
41
|
+
if self.keep is not None and not self.keep(payload, result):
|
|
42
|
+
return
|
|
43
|
+
identity = payload_id(payload)
|
|
44
|
+
self._write(f"case_{identity}.bin", payload)
|
|
45
|
+
if result.crashed:
|
|
46
|
+
self._write(f"crash_{identity}.bin", payload)
|
|
47
|
+
|
|
48
|
+
def _write(self, name: str, payload: bytes) -> None:
|
|
49
|
+
path = self.workdir / name
|
|
50
|
+
if not path.exists():
|
|
51
|
+
path.write_bytes(payload)
|
|
52
|
+
|
|
53
|
+
def summary(self):
|
|
54
|
+
logger.info("Cases: %d, crashes: %d", self.total, self.crashes)
|