iotsploit-fuzzer 0.0.9__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. iotsploit_fuzzer/__init__.py +77 -0
  2. iotsploit_fuzzer/analysis/__init__.py +1 -0
  3. iotsploit_fuzzer/analysis/corpus.py +480 -0
  4. iotsploit_fuzzer/analysis/logger.py +54 -0
  5. iotsploit_fuzzer/analysis/outcome.py +273 -0
  6. iotsploit_fuzzer/core/__init__.py +67 -0
  7. iotsploit_fuzzer/core/bit_manipulator.py +405 -0
  8. iotsploit_fuzzer/core/config.py +41 -0
  9. iotsploit_fuzzer/core/fuzzing_engine.py +469 -0
  10. iotsploit_fuzzer/core/orchestrator.py +452 -0
  11. iotsploit_fuzzer/core/parser_campaign.py +380 -0
  12. iotsploit_fuzzer/core/strategies/__init__.py +43 -0
  13. iotsploit_fuzzer/core/strategies/bit_strategies.py +509 -0
  14. iotsploit_fuzzer/core/strategies/field_strategies.py +697 -0
  15. iotsploit_fuzzer/fuzz.py +276 -0
  16. iotsploit_fuzzer/generators/__init__.py +1 -0
  17. iotsploit_fuzzer/generators/base.py +14 -0
  18. iotsploit_fuzzer/generators/corpus_generator.py +219 -0
  19. iotsploit_fuzzer/generators/radamsa_generator.py +96 -0
  20. iotsploit_fuzzer/harnesses/__init__.py +1 -0
  21. iotsploit_fuzzer/harnesses/base.py +27 -0
  22. iotsploit_fuzzer/harnesses/can_harness.py +23 -0
  23. iotsploit_fuzzer/harnesses/parser_harness.py +323 -0
  24. iotsploit_fuzzer/harnesses/parser_targets.py +142 -0
  25. iotsploit_fuzzer/harnesses/parser_worker.py +180 -0
  26. iotsploit_fuzzer/harnesses/spi_harness.py +17 -0
  27. iotsploit_fuzzer/harnesses/uart_harness.py +23 -0
  28. iotsploit_fuzzer/interfaces/__init__.py +1 -0
  29. iotsploit_fuzzer/interfaces/base.py +17 -0
  30. iotsploit_fuzzer/interfaces/can_interface.py +48 -0
  31. iotsploit_fuzzer/interfaces/spi_interface.py +27 -0
  32. iotsploit_fuzzer/interfaces/uart_interface.py +26 -0
  33. iotsploit_fuzzer/monitoring/__init__.py +1 -0
  34. iotsploit_fuzzer/monitoring/boundary_monitor.py +187 -0
  35. iotsploit_fuzzer/monitoring/monitor.py +203 -0
  36. iotsploit_fuzzer/targets/__init__.py +1 -0
  37. iotsploit_fuzzer/targets/iotsploit.py +1004 -0
  38. iotsploit_fuzzer-0.0.9.dist-info/METADATA +320 -0
  39. iotsploit_fuzzer-0.0.9.dist-info/RECORD +40 -0
  40. iotsploit_fuzzer-0.0.9.dist-info/WHEEL +4 -0
@@ -0,0 +1,77 @@
1
+ """
2
+ IoT Protocol Fuzzer
3
+
4
+ A modular fuzzing framework for IoT communication protocols including CAN, UART, and SPI.
5
+ """
6
+
7
+ __version__ = "0.0.9"
8
+ __author__ = "IoT Security Research"
9
+ __description__ = "Modular fuzzing framework for IoT protocols"
10
+
11
+ # Make common imports available at package level
12
+ from .core.orchestrator import Orchestrator, CampaignConfig
13
+ from .core.bit_manipulator import BitManipulator, flip_bit, set_bit, get_bit, parse_target_bits
14
+ from .core.fuzzing_engine import (
15
+ FuzzingEngine, StrategyRegistry, FuzzingStrategy, FuzzingType,
16
+ FuzzTestCase, MutationResult, default_engine, default_registry
17
+ )
18
+ from .generators.radamsa_generator import RadamsaGenerator
19
+ from .analysis.corpus import CorpusStore
20
+ from .harnesses.can_harness import CANHarness
21
+ from .harnesses.parser_harness import ParserHarness
22
+ from .harnesses.parser_targets import ParseTarget, register
23
+ from .monitoring.boundary_monitor import BoundaryMonitor
24
+ from .harnesses.uart_harness import UARTHarness
25
+ from .harnesses.spi_harness import SPIHarness
26
+
27
+ # Import strategies
28
+ try:
29
+ from .core.strategies.bit_strategies import (
30
+ BitFlipStrategy, SequentialBitStrategy, RandomBitStrategy
31
+ )
32
+ from .core.strategies.field_strategies import (
33
+ FieldMutationStrategy, BoundaryValueStrategy, InjectionStrategy
34
+ )
35
+ except ImportError:
36
+ # Strategies may not be available in all environments
37
+ pass
38
+
39
+ __all__ = [
40
+ # Original components
41
+ "Orchestrator",
42
+ "CampaignConfig",
43
+ "BitManipulator",
44
+ "flip_bit",
45
+ "set_bit",
46
+ "get_bit",
47
+ "parse_target_bits",
48
+ "RadamsaGenerator",
49
+ "CANHarness",
50
+ "UARTHarness",
51
+ "SPIHarness",
52
+
53
+ # Inbound parser fuzzing: the same loop pointed at our own parsers
54
+ "ParserHarness",
55
+ "ParseTarget",
56
+ "register",
57
+ "CorpusStore",
58
+ "BoundaryMonitor",
59
+
60
+ # Fuzzing engine components
61
+ "FuzzingEngine",
62
+ "StrategyRegistry",
63
+ "FuzzingStrategy",
64
+ "FuzzingType",
65
+ "FuzzTestCase",
66
+ "MutationResult",
67
+ "default_engine",
68
+ "default_registry",
69
+
70
+ # Fuzzing strategies
71
+ "BitFlipStrategy",
72
+ "SequentialBitStrategy",
73
+ "RandomBitStrategy",
74
+ "FieldMutationStrategy",
75
+ "BoundaryValueStrategy",
76
+ "InjectionStrategy",
77
+ ]
@@ -0,0 +1 @@
1
+ # Analysis and reporting
@@ -0,0 +1,480 @@
1
+ """Content-addressed payloads and the boundary ledger that gives them meaning.
2
+
3
+ The fuzzer's existing ``artifacts/`` directory keys files on the iteration
4
+ index, so campaign N+1 overwrites campaign N wherever the indices overlap, and
5
+ ``case_500.bin`` and ``crash_500.bin`` are usually from different runs. That is
6
+ a graveyard, not a corpus: nothing is attributable and nothing accumulates.
7
+
8
+ Here a payload's identity is its content, and the ledger records what each one
9
+ *did* -- its normalised outcome signature -- under the fingerprint of the
10
+ oracle that judged it. That is what makes a diff possible, and a diff is the
11
+ thing that keeps producing information after the crashes run dry.
12
+
13
+ Writes are atomic and locked per target. A nightly run that is killed halfway
14
+ leaves the previous ledger intact and readable.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import hashlib
20
+ import json
21
+ import logging
22
+ import os
23
+ import zipfile
24
+ from contextlib import contextmanager
25
+ from dataclasses import dataclass
26
+ from datetime import datetime, timezone
27
+ from io import BytesIO
28
+ from pathlib import Path
29
+ from typing import Dict, Iterator, Optional, Tuple
30
+
31
+ from ..harnesses.parser_targets import ParseTarget
32
+ from .outcome import Outcome
33
+
34
+ logger = logging.getLogger("fuzzer.corpus")
35
+
36
+ try: # POSIX advisory locking; elsewhere the lock is best-effort.
37
+ import fcntl
38
+ except ImportError: # pragma: no cover - exercised only off POSIX
39
+ fcntl = None
40
+
41
+ #: Bumped when the on-disk shape changes. A ledger from an older version is
42
+ #: re-baselined rather than guessed at.
43
+ LEDGER_VERSION = 1
44
+
45
+ #: Payloads kept per distinct signature. Without a cap the corpus fills with
46
+ #: near-identical inputs that all prove the same point, and the gate replay
47
+ #: that has to run on every commit slows down for nothing.
48
+ MAX_EXEMPLARS = 3
49
+
50
+ #: Payloads kept per distinct accept/reject edge. Lower than the signature cap
51
+ #: because an edge is already a pair, and because crossing the line is the
52
+ #: common case for a mutator -- uncapped, this alone grows without bound.
53
+ MAX_EDGE_EXEMPLARS = 1
54
+
55
+ #: Absolute ceiling per signature, whatever the reason for keeping it. Edges
56
+ #: are capped per *pair*, so a signature reachable from many parents could
57
+ #: otherwise accumulate one exemplar per parent and escape both caps.
58
+ MAX_PER_SIGNATURE = 5
59
+
60
+ #: Distinct signatures one target may hold. The per-signature caps bound how
61
+ #: many payloads each behaviour keeps, but not how many behaviours there are,
62
+ #: and a result shape with six bucketed fields has a combinatorially large
63
+ #: space. Sized generously: reaching it means the shape is too fine-grained to
64
+ #: be a boundary, which is worth reporting rather than absorbing.
65
+ MAX_SIGNATURES = 200
66
+
67
+ #: Campaign manifests kept per target. Enough to compare a finding against the
68
+ #: runs around it; bounded so a nightly loop does not accumulate a manifest a
69
+ #: day for ever.
70
+ KEEP_MANIFESTS = 10
71
+
72
+
73
+ class CorpusDamagedError(RuntimeError):
74
+ """The ledger and the archive do not agree, so replay coverage is unknown.
75
+
76
+ Raised rather than worked around. A replay that silently skips the
77
+ payloads it cannot find still reports success, and a green gate that
78
+ checked fewer inputs than it claims to is worse than a red one.
79
+ """
80
+
81
+
82
+ def payload_id(payload: bytes) -> str:
83
+ """A payload's identity: what it is, never where it appeared."""
84
+ return hashlib.sha256(payload).hexdigest()[:16]
85
+
86
+
87
+ @dataclass
88
+ class LedgerEntry:
89
+ """One retained payload, and what it did the last time anyone looked."""
90
+
91
+ signature: str
92
+ kind: str
93
+ first_seen: str
94
+ #: The parent's signature, when this payload was retained because one
95
+ #: mutation moved it across the accept/reject line. Empty otherwise. The
96
+ #: pair is the point: one payload alone does not locate a boundary.
97
+ edge_of: str = ""
98
+
99
+ def to_dict(self) -> Dict[str, object]:
100
+ return {
101
+ "signature": self.signature,
102
+ "kind": self.kind,
103
+ "first_seen": self.first_seen,
104
+ "edge_of": self.edge_of,
105
+ }
106
+
107
+ @classmethod
108
+ def from_dict(cls, raw: Dict[str, object]) -> "LedgerEntry":
109
+ return cls(
110
+ signature=str(raw.get("signature", "")),
111
+ kind=str(raw.get("kind", "")),
112
+ first_seen=str(raw.get("first_seen", "")),
113
+ edge_of=str(raw.get("edge_of", "")),
114
+ )
115
+
116
+
117
+ @contextmanager
118
+ def _locked(path: Path) -> Iterator[None]:
119
+ """Hold the per-target lock for the duration of a write."""
120
+ path.parent.mkdir(parents=True, exist_ok=True)
121
+ handle = os.open(str(path), os.O_CREAT | os.O_RDWR, 0o644)
122
+ try:
123
+ if fcntl is not None:
124
+ fcntl.flock(handle, fcntl.LOCK_EX)
125
+ yield
126
+ finally:
127
+ if fcntl is not None:
128
+ try:
129
+ fcntl.flock(handle, fcntl.LOCK_UN)
130
+ except OSError: # pragma: no cover - unlock of a dead fd
131
+ pass
132
+ os.close(handle)
133
+
134
+
135
+ def _write_atomic(path: Path, data: bytes) -> None:
136
+ """Replace a file in one step, or not at all.
137
+
138
+ The fsync matters more than it looks: without it a machine that loses
139
+ power mid-campaign can leave a renamed but empty ledger, which reads as
140
+ "this target has never been fuzzed" and silently discards months of
141
+ boundary history.
142
+ """
143
+ path.parent.mkdir(parents=True, exist_ok=True)
144
+ temporary = path.with_name(f"{path.name}.tmp-{os.getpid()}")
145
+ with open(temporary, "wb") as handle:
146
+ handle.write(data)
147
+ handle.flush()
148
+ os.fsync(handle.fileno())
149
+ os.replace(temporary, path)
150
+
151
+
152
+ class CorpusStore:
153
+ """The retained payloads for one target, and the ledger over them."""
154
+
155
+ def __init__(self, root: Path | str, target: ParseTarget) -> None:
156
+ self.target = target
157
+ self.root = Path(root) / target.name
158
+ #: One archive per target rather than one file per payload. The
159
+ #: corpus is a thousand-odd inputs of a hundred-odd bytes each, and
160
+ #: as loose files it was 74% of the repository's tracked file count
161
+ #: for 1% of its bytes -- plus a 4 KB block apiece, which turned
162
+ #: 1.1 MB of payloads into 9.4 MB on disk.
163
+ self.archive_path = self.root / "payloads.zip"
164
+ #: Where payloads used to live, read once so an existing corpus
165
+ #: migrates itself on the next save.
166
+ self.legacy_dir = self.root / "payloads"
167
+ self.ledger_path = self.root / "ledger.json"
168
+ self.lock_path = self.root / ".lock"
169
+ self.entries: Dict[str, LedgerEntry] = {}
170
+ #: True when the ledger was written under a different oracle. Diffing
171
+ #: is refused until a re-baseline: a changed declared-exception set
172
+ #: moves the accept/reject line by definition, and reporting that as
173
+ #: thousands of boundary movements teaches everyone to ignore reports.
174
+ self.stale = False
175
+ self.load()
176
+
177
+ # -- reading -----------------------------------------------------------
178
+
179
+ def load(self) -> None:
180
+ self.entries = {}
181
+ #: Set when the ledger and the archive disagree. Not raised here -- a
182
+ #: caller may want to inspect a damaged corpus -- but anything that
183
+ #: replays or extends it must refuse. Assigned before the payloads
184
+ #: are read, because reading them is one of the things that can
185
+ #: discover the damage.
186
+ self.damaged = ""
187
+ self._payloads: Dict[str, bytes] = self._read_payloads()
188
+ self._counts: Dict[str, int] = {}
189
+ self.stale = False
190
+ #: Set when a new signature was turned away by MAX_SIGNATURES.
191
+ self.saturated = False
192
+ if not self.ledger_path.exists():
193
+ return
194
+ try:
195
+ raw = json.loads(self.ledger_path.read_text())
196
+ except (OSError, ValueError) as error:
197
+ logger.warning("ledger for %s is unreadable (%s)", self.target.name, error)
198
+ self.stale = True
199
+ return
200
+ if raw.get("ledger_version") != LEDGER_VERSION:
201
+ logger.warning("ledger for %s was written by an older version", self.target.name)
202
+ self.stale = True
203
+ if raw.get("fingerprint") != self.target.fingerprint:
204
+ logger.warning(
205
+ "ledger for %s was recorded against oracle %s, now %s",
206
+ self.target.name, raw.get("fingerprint"), self.target.fingerprint,
207
+ )
208
+ self.stale = True
209
+ self.entries = {
210
+ str(key): LedgerEntry.from_dict(value)
211
+ for key, value in (raw.get("entries") or {}).items()
212
+ if isinstance(value, dict)
213
+ }
214
+ for entry in self.entries.values():
215
+ self._counts[entry.signature] = self._counts.get(entry.signature, 0) + 1
216
+ self._check_archive_agrees()
217
+
218
+ def _check_archive_agrees(self) -> None:
219
+ """The ledger and the archive are two files, written one after the
220
+ other under one lock -- which bounds the window but does not remove
221
+ it. A process killed between the two renames leaves a mixed pair.
222
+
223
+ A payload the ledger names and the archive does not have is the case
224
+ that must fail: ``payloads()`` would skip it, the replay would check
225
+ fewer inputs than the ledger claims, and it would still pass. The
226
+ other direction is harmless -- an unreferenced payload costs space
227
+ and nothing else -- so it is reported and ignored.
228
+ """
229
+ if self.damaged:
230
+ return
231
+ named = set(self.entries)
232
+ held = set(self._payloads)
233
+ missing = named - held
234
+ if missing:
235
+ self.damaged = (
236
+ f"{len(missing)} payload(s) named by the ledger are not in the "
237
+ f"archive, e.g. {sorted(missing)[0]}"
238
+ )
239
+ return
240
+ orphaned = held - named
241
+ if orphaned:
242
+ logger.warning(
243
+ "%s: %d payload(s) in the archive are not in the ledger; ignoring",
244
+ self.target.name, len(orphaned),
245
+ )
246
+
247
+ def _read_payloads(self) -> Dict[str, bytes]:
248
+ """Every retained payload, from the archive and any loose leftovers."""
249
+ found: Dict[str, bytes] = {}
250
+ if self.legacy_dir.is_dir():
251
+ for path in self.legacy_dir.glob("*.bin"):
252
+ found[path.stem] = path.read_bytes()
253
+ if self.archive_path.exists():
254
+ try:
255
+ with zipfile.ZipFile(self.archive_path) as archive:
256
+ for name in archive.namelist():
257
+ found[Path(name).stem] = archive.read(name)
258
+ except (OSError, zipfile.BadZipFile) as error:
259
+ # An unreadable archive is not an empty corpus. Treating it as
260
+ # one is how a replay reports success having checked nothing.
261
+ self.damaged = f"payloads.zip cannot be read: {error}"
262
+ return found
263
+
264
+ def payload(self, identity: str) -> Optional[bytes]:
265
+ return self._payloads.get(identity)
266
+
267
+ def payloads(self) -> Iterator[Tuple[str, bytes]]:
268
+ """Every retained payload, for a replay or a new campaign's seeds."""
269
+ for identity in sorted(self.entries):
270
+ data = self.payload(identity)
271
+ if data is not None:
272
+ yield identity, data
273
+
274
+ def signature_counts(self) -> Dict[str, int]:
275
+ return dict(self._counts)
276
+
277
+ def known_signatures(self) -> set:
278
+ """Every signature the ledger holds. Kept as a set, not rebuilt: this
279
+ is asked once per case, and rebuilding it made the campaign quadratic
280
+ in its own corpus."""
281
+ return set(self._counts)
282
+
283
+ # -- writing -----------------------------------------------------------
284
+
285
+ def admit(
286
+ self, payload: bytes, outcome: Outcome, campaign: str, *, edge_of: str = ""
287
+ ) -> bool:
288
+ """Retain a payload if this signature still has room, else discard it.
289
+
290
+ Returns whether it was kept -- a payload already in the ledger is not
291
+ kept again. What it used to do is left exactly as it is: that record
292
+ is what a boundary diff is made against, and only the monitor that
293
+ reported the movement may change it.
294
+ """
295
+ identity = payload_id(payload)
296
+ entry = self.entries.get(identity)
297
+ if entry is not None:
298
+ if not entry.signature:
299
+ # Blanked by a re-baseline: this is the campaign re-deriving
300
+ # what the payload does under the new oracle. Without this the
301
+ # ledger never recovers -- every case stays "novel" forever
302
+ # because no signature is ever recorded again.
303
+ entry.signature = outcome.signature
304
+ entry.kind = outcome.kind
305
+ self._counts[outcome.signature] = self._counts.get(outcome.signature, 0) + 1
306
+ return False
307
+ if self._counts.get(outcome.signature, 0) >= MAX_PER_SIGNATURE:
308
+ return False
309
+ if outcome.signature not in self._counts and len(self._counts) >= MAX_SIGNATURES:
310
+ self.saturated = True
311
+ return False
312
+ if edge_of:
313
+ held = sum(
314
+ 1 for entry in self.entries.values()
315
+ if entry.edge_of == edge_of and entry.signature == outcome.signature
316
+ )
317
+ cap = MAX_EDGE_EXEMPLARS
318
+ else:
319
+ held = self._counts.get(outcome.signature, 0)
320
+ cap = MAX_EXEMPLARS
321
+ if held >= cap and not self._displace(payload, outcome, edge_of):
322
+ return False
323
+ self._payloads[identity] = payload
324
+ self.entries[identity] = LedgerEntry(
325
+ signature=outcome.signature,
326
+ kind=outcome.kind,
327
+ first_seen=campaign,
328
+ edge_of=edge_of,
329
+ )
330
+ self._counts[outcome.signature] = self._counts.get(outcome.signature, 0) + 1
331
+ return True
332
+
333
+ def reclassify(self, identity: str, outcome: Outcome) -> None:
334
+ """Record that a payload already held now does something else.
335
+
336
+ The one way an entry's signature may change after it is admitted, and
337
+ it lives here because the entry is not the only thing that has to
338
+ move: ``_counts`` backs ``known_signatures()``, the per-signature cap
339
+ and the saturation ceiling. The monitor used to assign to
340
+ ``entry.signature`` directly, which left the ledger saying one thing
341
+ and the counts another -- a payload recorded as rejected while the
342
+ count that decides novelty still read accepted.
343
+ """
344
+ entry = self.entries.get(identity)
345
+ if entry is None or entry.signature == outcome.signature:
346
+ return
347
+ if self._counts.get(entry.signature):
348
+ self._counts[entry.signature] -= 1
349
+ if not self._counts[entry.signature]:
350
+ del self._counts[entry.signature]
351
+ entry.signature = outcome.signature
352
+ entry.kind = outcome.kind
353
+ self._counts[outcome.signature] = self._counts.get(outcome.signature, 0) + 1
354
+
355
+ def _payload_size(self, identity: str) -> int:
356
+ return len(self._payloads.get(identity, b""))
357
+
358
+ def _displace(self, payload: bytes, outcome: Outcome, edge_of: str) -> bool:
359
+ """Make room by dropping a larger payload that proves the same point.
360
+
361
+ Without this the corpus is bounded in entries but not in bytes, and
362
+ radamsa's repetition mutations exploit exactly that: a mutant grows,
363
+ is retained because its signature is new, becomes the next
364
+ generation's parent, and grows again. Measured from a 61-byte seed it
365
+ reached 567 KB in eight generations -- and radamsa's cost scales with
366
+ input size, so the loop makes itself slower as it runs.
367
+
368
+ Keeping the smallest example of each signature bounds the corpus in
369
+ bytes, keeps the gate replay fast, and hands triage the smallest
370
+ reproduction rather than whichever one happened to arrive first.
371
+ """
372
+ candidates = [
373
+ identity for identity, entry in self.entries.items()
374
+ if entry.signature == outcome.signature
375
+ and (not edge_of or entry.edge_of == edge_of)
376
+ ]
377
+ if not candidates:
378
+ return False
379
+ largest = max(candidates, key=self._payload_size)
380
+ if self._payload_size(largest) <= len(payload):
381
+ return False
382
+ self._drop(largest)
383
+ return True
384
+
385
+ def _drop(self, identity: str) -> None:
386
+ """Remove one payload and the ledger's memory of it."""
387
+ entry = self.entries.pop(identity, None)
388
+ if entry is None:
389
+ return
390
+ if self._counts.get(entry.signature):
391
+ self._counts[entry.signature] -= 1
392
+ if not self._counts[entry.signature]:
393
+ del self._counts[entry.signature]
394
+ self._payloads.pop(identity, None)
395
+
396
+ def rebaseline(self) -> None:
397
+ """Adopt the current oracle, discarding recorded signatures.
398
+
399
+ The payloads are kept -- they are the expensive part and are still
400
+ interesting inputs. Only the claim about what they *do* is dropped,
401
+ because it was made by a different oracle. The next campaign
402
+ re-derives it.
403
+ """
404
+ for entry in self.entries.values():
405
+ entry.signature = ""
406
+ entry.kind = ""
407
+ self._counts = {}
408
+ self.stale = False
409
+
410
+ def _write_archive(self) -> None:
411
+ """Rewrite the archive in one atomic step, holding only what the
412
+ ledger still names.
413
+
414
+ Written whole rather than appended to, because a payload displaced by
415
+ a smaller one has to leave.
416
+
417
+ This and the ledger are two renames under one lock. The lock keeps
418
+ two campaigns from interleaving; it does nothing about a kill between
419
+ the two writes, which can still leave a mixed pair. That is not
420
+ claimed to be transactional -- it is detected on load instead, by
421
+ :meth:`_check_archive_agrees`.
422
+ """
423
+ buffer = BytesIO()
424
+ with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive:
425
+ for identity in sorted(self.entries):
426
+ payload = self._payloads.get(identity)
427
+ if payload is not None:
428
+ archive.writestr(f"{identity}.bin", payload)
429
+ _write_atomic(self.archive_path, buffer.getvalue())
430
+ if self.legacy_dir.is_dir():
431
+ for path in self.legacy_dir.glob("*.bin"):
432
+ path.unlink()
433
+ self.legacy_dir.rmdir()
434
+
435
+ def save(self, campaign: Optional[Dict[str, object]] = None) -> None:
436
+ """Commit the ledger, and the manifest of the run that produced it."""
437
+ document = {
438
+ "ledger_version": LEDGER_VERSION,
439
+ "target": self.target.name,
440
+ "fingerprint": self.target.fingerprint,
441
+ "updated": datetime.now(timezone.utc).isoformat(timespec="seconds"),
442
+ "entries": {k: v.to_dict() for k, v in sorted(self.entries.items())},
443
+ }
444
+ with _locked(self.lock_path):
445
+ if self._already_recorded(document):
446
+ # Nothing was learned, so nothing is written. A campaign that
447
+ # finds no new behaviour is the normal night, and rewriting
448
+ # the timestamp anyway put all 28 ledgers in every diff --
449
+ # which buries the one line that says a boundary moved, the
450
+ # thing tracking the corpus in git is for.
451
+ self._write_manifest(campaign)
452
+ return
453
+ self._write_archive()
454
+ _write_atomic(self.ledger_path, json.dumps(document, indent=2).encode())
455
+ self._write_manifest(campaign)
456
+
457
+ def _already_recorded(self, document: Dict[str, object]) -> bool:
458
+ """Whether the ledger on disk already says this, timestamp aside."""
459
+ try:
460
+ existing = json.loads(self.ledger_path.read_text())
461
+ except (OSError, ValueError):
462
+ return False
463
+ comparable = dict(document)
464
+ comparable.pop("updated", None)
465
+ existing.pop("updated", None)
466
+ return existing == comparable
467
+
468
+ def _write_manifest(self, campaign: Optional[Dict[str, object]]) -> None:
469
+ """The record of one run, which is written whether or not it learned
470
+ anything -- it is how a night that found nothing is distinguished
471
+ from a night that did not run."""
472
+ if not campaign:
473
+ return
474
+ manifests = self.root / "campaigns"
475
+ _write_atomic(
476
+ manifests / f"{campaign['campaign']}.json",
477
+ json.dumps(campaign, indent=2).encode(),
478
+ )
479
+ for stale in sorted(manifests.glob("*.json"))[:-KEEP_MANIFESTS]:
480
+ stale.unlink(missing_ok=True)
@@ -0,0 +1,54 @@
1
+ import logging
2
+ from pathlib import Path
3
+ from typing import Callable, Optional
4
+
5
+ from ..harnesses.base import HarnessResult
6
+ from .corpus import payload_id
7
+
8
+ logger = logging.getLogger("fuzzer.logger")
9
+
10
+
11
+ class TestLogger:
12
+ """Logs test cases to disk for later analysis, keyed on content.
13
+
14
+ Files were named ``case_{index}.bin``, which made a payload's identity its
15
+ position in a campaign: every run overwrote the previous run's cases
16
+ wherever the indices overlapped, ``case_500.bin`` and ``crash_500.bin``
17
+ came from different runs, and nothing on disk was attributable. Naming by
18
+ content hash instead makes a repeated payload one file and leaves earlier
19
+ campaigns intact.
20
+
21
+ ``keep`` lets a caller that has its own notion of interesting -- the
22
+ parser loop, whose corpus is curated by signature -- stop routine cases
23
+ reaching disk at all, without giving up the crash record.
24
+ """
25
+
26
+ def __init__(
27
+ self,
28
+ workdir: str = "artifacts",
29
+ keep: Optional[Callable[[bytes, HarnessResult], bool]] = None,
30
+ ):
31
+ self.workdir = Path(workdir)
32
+ self.workdir.mkdir(parents=True, exist_ok=True)
33
+ self.keep = keep
34
+ self.total = 0
35
+ self.crashes = 0
36
+
37
+ def record(self, idx: int, payload: bytes, result: HarnessResult) -> None:
38
+ self.total += 1
39
+ if result.crashed:
40
+ self.crashes += 1
41
+ if self.keep is not None and not self.keep(payload, result):
42
+ return
43
+ identity = payload_id(payload)
44
+ self._write(f"case_{identity}.bin", payload)
45
+ if result.crashed:
46
+ self._write(f"crash_{identity}.bin", payload)
47
+
48
+ def _write(self, name: str, payload: bytes) -> None:
49
+ path = self.workdir / name
50
+ if not path.exists():
51
+ path.write_bytes(payload)
52
+
53
+ def summary(self):
54
+ logger.info("Cases: %d, crashes: %d", self.total, self.crashes)