iotsploit-protocols 0.0.9__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,753 @@
1
+ """Read a recorded CAN log back as frames, so a stored capture decodes like a live one.
2
+
3
+ What this buys is not convenience. A bus you can only watch live is a bus you
4
+ can only analyse while you are standing next to the vehicle, and a finding you
5
+ cannot replay is a finding nobody else can check. Feeding a file through the
6
+ same aggregator, the same codec, and the same target definitions as a live
7
+ capture is what makes a recorded window reviewable evidence.
8
+
9
+ The messages yielded here are deliberately *not* ``can.Message``. Nothing
10
+ downstream needs one: :class:`~iotsploit_exploits.canbus.live_capture.CaptureAggregator`
11
+ reads identity, payload, arrival time, and the frame-class flags by attribute
12
+ and nothing else. ASC stays a small native parser; binary BLF and the two other
13
+ widely used text formats delegate parsing to ``python-can`` without ever
14
+ constructing a CAN bus or touching a platform socket.
15
+
16
+ Three things in the Vector ASC format are worth knowing before changing this,
17
+ because each one silently produces plausible wrong numbers rather than an
18
+ error:
19
+
20
+ *The ``base`` directive governs identifiers only.* ``base hex`` does not make
21
+ DLC and data-length hexadecimal; those columns are decimal in every ASC this
22
+ has been checked against. Reading them as hex turns the CAN FD length codes
23
+ ``10 16``, ``12 24`` and ``13 32`` into 22, 36 and 50 bytes, which truncates
24
+ payloads on exactly the frames that carry the most signal.
25
+
26
+ *DLC is a length code, not a length.* Above eight, CAN FD's DLC indexes a
27
+ table (9 to 15 mean 12, 16, 20, 24, 32, 48, 64 bytes). Both columns are present
28
+ in the file and they must agree; a line where they do not is a line this reader
29
+ did not understand.
30
+
31
+ *There is more than one CAN FD column order in the wild.* Vector's own writer
32
+ puts direction before identifier; other tools emit the classic-CAN order with
33
+ identifier first. Both appear in real logs, so both are read here, anchored on
34
+ whichever column actually holds ``Rx``/``Tx`` rather than on a fixed offset.
35
+
36
+ A line this reader cannot parse is counted and skipped, never guessed at and
37
+ never fatal. A single corrupt line in the middle of fifty thousand must not end
38
+ a replay, and a count of what was skipped is what lets an operator tell a clean
39
+ read from a partial one.
40
+ """
41
+
42
+ from __future__ import annotations
43
+
44
+ import re
45
+ from dataclasses import dataclass, field
46
+ from datetime import datetime, timezone
47
+ from pathlib import Path
48
+ from typing import Any, Dict, Iterator, List, Optional, Sequence, Set, Tuple, Type
49
+
50
+ from iotsploit_protocols.errors import NotConfigured
51
+
52
+ #: CAN FD data-length codes above 8. ``linux/can.h`` calls this ``can_fd_dlc2len``.
53
+ DLC_TO_LENGTH: Dict[int, int] = {
54
+ **{code: code for code in range(9)},
55
+ 9: 12,
56
+ 10: 16,
57
+ 11: 20,
58
+ 12: 24,
59
+ 13: 32,
60
+ 14: 48,
61
+ 15: 64,
62
+ }
63
+
64
+ #: Every payload size a CAN or CAN FD frame can actually have.
65
+ VALID_LENGTHS = frozenset(DLC_TO_LENGTH.values())
66
+
67
+ #: Direction columns. ``TxRq`` is a transmit *request*, logged in addition to
68
+ #: the ``Tx`` that follows it -- counting both would double every frame this
69
+ #: host sent, so it is skipped rather than replayed.
70
+ _DIRECTIONS = frozenset({"Rx", "Tx"})
71
+ _TX_REQUEST = "TxRq"
72
+
73
+ #: An identifier column, optionally flagged extended with a trailing ``x``.
74
+ _ID_RE = re.compile(r"\A([0-9A-Fa-f]+)(x?)\Z")
75
+
76
+ #: Header forms this reader understands. Anything else in the header is ignored
77
+ #: rather than refused: ASC writers emit tool-specific directives freely, and a
78
+ #: replay must not fail because it met one it had not seen.
79
+ _DATE_FORMATS = (
80
+ "%a %b %d %I:%M:%S %p %Y",
81
+ "%a %b %d %I:%M:%S.%f %p %Y",
82
+ "%a %b %d %H:%M:%S %Y",
83
+ "%a %b %d %H:%M:%S.%f %Y",
84
+ )
85
+
86
+ MAX_STANDARD_FRAME_ID = 0x7FF
87
+ MAX_EXTENDED_FRAME_ID = 0x1FFFFFFF
88
+ LogChannel = int | str
89
+
90
+
91
+ class CanLogError(NotConfigured):
92
+ """The log cannot be read, or is not a format this understands.
93
+
94
+ A subclass of :class:`NotConfigured` because every case it covers is a
95
+ choice the operator can correct -- a path that is not there, a suffix
96
+ nothing here parses -- rather than a fault in the bus or the target.
97
+ """
98
+
99
+
100
+ @dataclass(frozen=True)
101
+ class ReplayMessage:
102
+ """One frame read back from a log.
103
+
104
+ The attribute names are ``python-can``'s because that is the shape the
105
+ aggregator reads. The class is a plain dataclass because that is all the
106
+ shape actually requires.
107
+ """
108
+
109
+ timestamp: float
110
+ arbitration_id: int
111
+ data: bytes
112
+ is_extended_id: bool = False
113
+ is_error_frame: bool = False
114
+ is_remote_frame: bool = False
115
+ is_fd: bool = False
116
+ channel: Optional[LogChannel] = None
117
+
118
+ @property
119
+ def dlc(self) -> int:
120
+ return len(self.data)
121
+
122
+
123
+ @dataclass
124
+ class LogReadStats:
125
+ """What a read actually saw, as opposed to what it was asked for.
126
+
127
+ Read after iterating. ``unparsable_lines`` is the number that matters: a
128
+ replay reporting 47,797 frames and 0 skipped lines is a clean read of the
129
+ whole file, and the same replay reporting 12,000 skipped lines is a partial
130
+ one wearing the same summary.
131
+ """
132
+
133
+ format: str = "asc"
134
+ frames: int = 0
135
+ error_frames: int = 0
136
+ unparsable_lines: int = 0
137
+ channels_present: Set[LogChannel] = field(default_factory=set)
138
+ first_timestamp: Optional[float] = None
139
+ last_timestamp: Optional[float] = None
140
+ #: Wall-clock start from the log's own ``date`` header, when it had one.
141
+ started_at: Optional[datetime] = None
142
+ truncated: bool = False
143
+
144
+ @property
145
+ def duration_s(self) -> float:
146
+ if self.first_timestamp is None or self.last_timestamp is None:
147
+ return 0.0
148
+ return round(self.last_timestamp - self.first_timestamp, 6)
149
+
150
+ def as_dict(self) -> Dict[str, Any]:
151
+ return {
152
+ "format": self.format,
153
+ "frames": self.frames,
154
+ "error_frames": self.error_frames,
155
+ "unparsable_lines": self.unparsable_lines,
156
+ "channels_present": sorted(self.channels_present, key=channel_sort_key),
157
+ "duration_s": self.duration_s,
158
+ "started_at": self.started_at.isoformat() if self.started_at else None,
159
+ "truncated": self.truncated,
160
+ }
161
+
162
+
163
+ def channel_sort_key(channel: LogChannel) -> Tuple[int, str]:
164
+ """Keep numeric channels ordered before named interfaces such as ``can0``."""
165
+ if isinstance(channel, int):
166
+ return 0, f"{channel:020d}"
167
+ return 1, channel
168
+
169
+
170
+ def normalize_log_channel(value: Any) -> Optional[LogChannel]:
171
+ """Normalize an API/CLI channel without losing candump interface names."""
172
+ if value is None:
173
+ return None
174
+ if isinstance(value, bool):
175
+ raise ValueError("log_channel must be a channel number or interface name")
176
+ if isinstance(value, int):
177
+ if value < 0:
178
+ raise ValueError("log_channel must not be negative")
179
+ return value
180
+ if isinstance(value, str):
181
+ channel = value.strip()
182
+ if not channel:
183
+ raise ValueError("log_channel must not be empty")
184
+ return int(channel) if channel.isdigit() else channel
185
+ raise ValueError("log_channel must be a channel number or interface name")
186
+
187
+
188
+ def select_log_channel(
189
+ channels: Set[LogChannel], requested: Optional[LogChannel]
190
+ ) -> LogChannel:
191
+ """Resolve the one bus a replay may decode against one target definition."""
192
+ if not channels:
193
+ raise CanLogError("the CAN log contains no readable channels")
194
+ if requested is not None:
195
+ if requested not in channels:
196
+ available = ", ".join(
197
+ str(channel) for channel in sorted(channels, key=channel_sort_key)
198
+ )
199
+ raise CanLogError(
200
+ f"channel {requested!r} is not present in the CAN log; "
201
+ f"available channels: {available}"
202
+ )
203
+ return requested
204
+ if len(channels) == 1:
205
+ return next(iter(channels))
206
+ available = ", ".join(
207
+ str(channel) for channel in sorted(channels, key=channel_sort_key)
208
+ )
209
+ raise CanLogError(
210
+ f"the CAN log contains multiple channels ({available}); choose log_channel"
211
+ )
212
+
213
+
214
+ def _parse_identifier(token: str, base: int) -> Optional[Tuple[int, bool]]:
215
+ match = _ID_RE.match(token)
216
+ if match is None:
217
+ return None
218
+ digits, extended_flag = match.group(1), match.group(2) == "x"
219
+ try:
220
+ frame_id = int(digits, base)
221
+ except ValueError:
222
+ return None
223
+ ceiling = MAX_EXTENDED_FRAME_ID if extended_flag else MAX_STANDARD_FRAME_ID
224
+ if frame_id > ceiling:
225
+ # An identifier too wide for the width it claims is a mis-read column,
226
+ # not a frame. Counting it would put an address on screen that no ECU
227
+ # can hold.
228
+ return None
229
+ return frame_id, extended_flag
230
+
231
+
232
+ def _parse_payload(tokens: Sequence[str], length: int) -> Optional[bytes]:
233
+ if len(tokens) < length:
234
+ return None
235
+ try:
236
+ return bytes(int(token, 16) for token in tokens[:length])
237
+ except ValueError:
238
+ return None
239
+
240
+
241
+ #: Longest decimal field this reader will convert. ``int()`` refuses a string
242
+ #: of more than 4300 digits and raises ValueError, and this reader's contract
243
+ #: is that a line it cannot parse is counted and skipped, never fatal. No
244
+ #: column in an ASC log -- a DLC, a channel, a length -- is more than a few
245
+ #: digits, so a longer one means the line is not what it claims to be.
246
+ MAX_DECIMAL_DIGITS = 18
247
+
248
+
249
+ def _is_decimal(token: str) -> bool:
250
+ return token.isdigit() and len(token) <= MAX_DECIMAL_DIGITS
251
+
252
+
253
+ def _consume_frame_body(
254
+ tokens: Sequence[str], *, fd: bool
255
+ ) -> Optional[Tuple[bool, bytes]]:
256
+ """Read ``[name] [brs esi] [d|r] <dlc> [length] <data...>`` into a payload.
257
+
258
+ The optional columns are what make one routine serve both the classic and
259
+ the CAN FD line shapes, and both CAN FD column orders. Everything optional
260
+ here is optional in some real writer's output; nothing is optional to make
261
+ a malformed line pass.
262
+
263
+ Returns the remote flag and the payload, or ``None`` when the columns do
264
+ not form a frame -- which is the signal to count the line as unparsable
265
+ rather than to publish a guess.
266
+ """
267
+ index = 0
268
+ # A symbolic message name, when the writer emitted one. Never a bare
269
+ # integer, so it cannot be confused with the numeric columns that follow.
270
+ if index < len(tokens) and not _is_decimal(tokens[index]) and tokens[index] not in {"d", "r"}:
271
+ index += 1
272
+
273
+ if fd:
274
+ # BRS and ESI. Present in every CAN FD line; their absence means the
275
+ # columns are not the ones this expects.
276
+ if index + 1 >= len(tokens) or not (
277
+ _is_decimal(tokens[index]) and _is_decimal(tokens[index + 1])
278
+ ):
279
+ return None
280
+ index += 2
281
+
282
+ is_remote = False
283
+ if index < len(tokens) and tokens[index] in {"d", "r"}:
284
+ is_remote = tokens[index] == "r"
285
+ index += 1
286
+
287
+ if index >= len(tokens) or not _is_decimal(tokens[index]):
288
+ return None
289
+ dlc = int(tokens[index])
290
+ index += 1
291
+
292
+ length = DLC_TO_LENGTH.get(dlc)
293
+ if length is None:
294
+ return None
295
+
296
+ # CAN FD states the byte count in its own column. When it is there the two
297
+ # must agree: a DLC and a length that disagree mean the columns have been
298
+ # read at the wrong offset, and the payload that follows is not the one
299
+ # this line describes.
300
+ if fd:
301
+ if index >= len(tokens) or not _is_decimal(tokens[index]):
302
+ return None
303
+ stated = int(tokens[index])
304
+ if stated != length:
305
+ return None
306
+ index += 1
307
+
308
+ if is_remote:
309
+ # A remote frame requests a payload rather than carrying one.
310
+ return True, b""
311
+
312
+ payload = _parse_payload(tokens[index:], length)
313
+ if payload is None:
314
+ return None
315
+ return False, payload
316
+
317
+
318
+ class AscLogReader:
319
+ """A Vector ASC log, read as a stream of frames.
320
+
321
+ Streaming rather than loading: an ASC of a busy bus runs to hundreds of
322
+ megabytes, and a replay that has to fit the whole log in memory before
323
+ showing the first row is a replay that fails on the logs worth replaying.
324
+
325
+ Read :attr:`stats` after iterating to find out what the read actually saw.
326
+ """
327
+
328
+ format = "asc"
329
+
330
+ def __init__(
331
+ self,
332
+ path: str | Path,
333
+ *,
334
+ channel: Optional[LogChannel] = None,
335
+ max_frames: Optional[int] = None,
336
+ ) -> None:
337
+ self.path = Path(path)
338
+ #: Which bus in the log to replay. A multi-channel ASC holds several
339
+ #: buses, and decoding all of them against one target bus is how a
340
+ #: replay produces values that look right and are not -- so a caller
341
+ #: that knows which channel it wants says so.
342
+ self.channel = channel
343
+ self.max_frames = max_frames
344
+ self.stats = LogReadStats(format=self.format)
345
+ self._base = 16
346
+ self._epoch = 0.0
347
+
348
+ def messages(self) -> Iterator[ReplayMessage]:
349
+ """Yield frames until the log or the frame budget runs out."""
350
+ try:
351
+ handle = self.path.open("r", encoding="utf-8", errors="replace")
352
+ except OSError as error:
353
+ raise CanLogError(f"cannot read CAN log {str(self.path)!r}: {error}") from error
354
+
355
+ with handle:
356
+ for line in handle:
357
+ tokens = line.split()
358
+ if not tokens:
359
+ continue
360
+ if self._read_directive(tokens):
361
+ continue
362
+
363
+ message = self._read_frame(tokens)
364
+ if message is None:
365
+ continue
366
+
367
+ observed_channel = message.channel if message.channel is not None else 0
368
+ self.stats.channels_present.add(observed_channel)
369
+ if self.channel is not None and observed_channel != self.channel:
370
+ continue
371
+
372
+ if message.is_error_frame:
373
+ self.stats.error_frames += 1
374
+ else:
375
+ self.stats.frames += 1
376
+ if self.stats.first_timestamp is None:
377
+ self.stats.first_timestamp = message.timestamp
378
+ self.stats.last_timestamp = message.timestamp
379
+
380
+ yield message
381
+
382
+ if self.max_frames is not None and self.stats.frames >= self.max_frames:
383
+ self.stats.truncated = True
384
+ return
385
+
386
+ # ── header ────────────────────────────────────────────────────────
387
+
388
+ def _read_directive(self, tokens: List[str]) -> bool:
389
+ """Consume a header line, returning whether it was one.
390
+
391
+ A directive is recognised by its keyword rather than by its position:
392
+ ASC writers put ``base`` and ``timestamps`` on one line or two, and in
393
+ either order.
394
+ """
395
+ head = tokens[0]
396
+ if head.startswith("//"):
397
+ return True
398
+ if head == "date":
399
+ self._read_date(tokens[1:])
400
+ return True
401
+ if head in {"base", "internal"} or (head in {"Begin", "End"} and len(tokens) > 1):
402
+ if "hex" in tokens:
403
+ self._base = 16
404
+ elif "dec" in tokens:
405
+ self._base = 10
406
+ return True
407
+ if head in {"Measurement", "previous"}:
408
+ return True
409
+ # A frame line always opens with a timestamp. Anything else that does
410
+ # not is a directive this reader has no use for.
411
+ try:
412
+ float(head)
413
+ except ValueError:
414
+ return True
415
+ return False
416
+
417
+ def _read_date(self, tokens: Sequence[str]) -> None:
418
+ """Anchor the log's relative timestamps to the wall clock it recorded.
419
+
420
+ Without this every ``first_seen_at`` in a replay renders as 1970, because
421
+ an ASC counts seconds from the start of its own measurement. The log
422
+ states when that was; using it is the difference between a timestamp an
423
+ operator can correlate with anything else and one they cannot.
424
+ """
425
+ text = " ".join(tokens)
426
+ for fmt in _DATE_FORMATS:
427
+ try:
428
+ parsed = datetime.strptime(text, fmt)
429
+ except ValueError:
430
+ continue
431
+ # An ASC date header is wall-clock time on the machine that
432
+ # recorded it, with no zone stated. Reading it as local time is the
433
+ # only available interpretation; stamping the zone on makes that
434
+ # assumption visible in the result instead of leaving a bare naive
435
+ # time that reads as UTC next to arrival times that are not.
436
+ self.stats.started_at = parsed.astimezone()
437
+ self._epoch = parsed.timestamp()
438
+ return
439
+
440
+ # ── frames ────────────────────────────────────────────────────────
441
+
442
+ def _read_frame(self, tokens: List[str]) -> Optional[ReplayMessage]:
443
+ try:
444
+ timestamp = float(tokens[0]) + self._epoch
445
+ except ValueError:
446
+ return None
447
+
448
+ if len(tokens) < 3:
449
+ self.stats.unparsable_lines += 1
450
+ return None
451
+
452
+ # A bus fault is not traffic, and the aggregator tallies it separately.
453
+ # Recognised before anything reads a column as an identifier, for the
454
+ # same reason the live path classifies before it reads identity.
455
+ if any(token.startswith("ErrorFrame") for token in tokens):
456
+ return ReplayMessage(
457
+ timestamp=timestamp,
458
+ arbitration_id=0,
459
+ data=b"",
460
+ is_error_frame=True,
461
+ channel=self._read_channel(tokens[1]),
462
+ )
463
+
464
+ if tokens[1] == "CANFD":
465
+ return self._read_fd_frame(timestamp, tokens[2:])
466
+ if tokens[1] == "CAN":
467
+ # Some writers label classic lines explicitly.
468
+ return self._read_classic_frame(timestamp, tokens[2:])
469
+ return self._read_classic_frame(timestamp, tokens[1:])
470
+
471
+ @staticmethod
472
+ def _read_channel(token: str) -> Optional[int]:
473
+ return int(token) if _is_decimal(token) else None
474
+
475
+ def _read_fd_frame(self, timestamp: float, tokens: List[str]) -> Optional[ReplayMessage]:
476
+ """Read a CAN FD line in either of the two column orders found in real logs.
477
+
478
+ Vector's writer emits ``channel direction identifier``; tools that grew
479
+ out of the classic-CAN line emit ``channel identifier direction``. The
480
+ direction column is what tells them apart, so it is located rather than
481
+ assumed.
482
+ """
483
+ if len(tokens) < 6:
484
+ self.stats.unparsable_lines += 1
485
+ return None
486
+ channel = self._read_channel(tokens[0])
487
+
488
+ if tokens[1] in _DIRECTIONS:
489
+ identifier, rest = tokens[2], tokens[3:]
490
+ elif len(tokens) > 2 and tokens[2] in _DIRECTIONS:
491
+ identifier, rest = tokens[1], tokens[3:]
492
+ elif _TX_REQUEST in tokens[1:3]:
493
+ # Logged alongside the Tx it precedes; replaying both would count
494
+ # every transmitted frame twice.
495
+ return None
496
+ else:
497
+ self.stats.unparsable_lines += 1
498
+ return None
499
+
500
+ parsed_id = _parse_identifier(identifier, self._base)
501
+ body = _consume_frame_body(rest, fd=True) if parsed_id else None
502
+ if parsed_id is None or body is None:
503
+ self.stats.unparsable_lines += 1
504
+ return None
505
+
506
+ frame_id, is_extended = parsed_id
507
+ is_remote, payload = body
508
+ return ReplayMessage(
509
+ timestamp=timestamp,
510
+ arbitration_id=frame_id,
511
+ data=payload,
512
+ is_extended_id=is_extended,
513
+ is_remote_frame=is_remote,
514
+ is_fd=True,
515
+ channel=channel,
516
+ )
517
+
518
+ def _read_classic_frame(
519
+ self, timestamp: float, tokens: List[str]
520
+ ) -> Optional[ReplayMessage]:
521
+ """Read ``<channel> <identifier> <direction> <d|r> <dlc> <data...>``."""
522
+ if len(tokens) < 4:
523
+ self.stats.unparsable_lines += 1
524
+ return None
525
+ channel = self._read_channel(tokens[0])
526
+
527
+ if tokens[2] in _DIRECTIONS:
528
+ identifier, rest = tokens[1], tokens[3:]
529
+ elif tokens[1] in _DIRECTIONS:
530
+ identifier, rest = tokens[2], tokens[3:]
531
+ elif _TX_REQUEST in tokens[1:3]:
532
+ return None
533
+ else:
534
+ self.stats.unparsable_lines += 1
535
+ return None
536
+
537
+ parsed_id = _parse_identifier(identifier, self._base)
538
+ body = _consume_frame_body(rest, fd=False) if parsed_id else None
539
+ if parsed_id is None or body is None:
540
+ self.stats.unparsable_lines += 1
541
+ return None
542
+
543
+ frame_id, is_extended = parsed_id
544
+ is_remote, payload = body
545
+ return ReplayMessage(
546
+ timestamp=timestamp,
547
+ arbitration_id=frame_id,
548
+ data=payload,
549
+ is_extended_id=is_extended,
550
+ is_remote_frame=is_remote,
551
+ channel=channel,
552
+ )
553
+
554
+
555
+ class PythonCanLogReader:
556
+ """Stream a format handled by python-can into the replay message shape."""
557
+
558
+ format = ""
559
+ reader_name = ""
560
+ one_based_channels = False
561
+
562
+ def __init__(
563
+ self,
564
+ path: str | Path,
565
+ *,
566
+ channel: Optional[LogChannel] = None,
567
+ max_frames: Optional[int] = None,
568
+ ) -> None:
569
+ self.path = Path(path)
570
+ self.channel = channel
571
+ self.max_frames = max_frames
572
+ self.stats = LogReadStats(format=self.format)
573
+
574
+ def messages(self) -> Iterator[ReplayMessage]:
575
+ try:
576
+ self._validate()
577
+ from can import io
578
+
579
+ reader_type: Type[Any] = getattr(io, self.reader_name)
580
+ with reader_type(self.path) as source:
581
+ for source_message in source:
582
+ message = self._convert(source_message)
583
+ observed_channel = message.channel if message.channel is not None else 0
584
+ self.stats.channels_present.add(observed_channel)
585
+ if self.channel is not None and observed_channel != self.channel:
586
+ continue
587
+
588
+ if message.is_error_frame:
589
+ self.stats.error_frames += 1
590
+ else:
591
+ self.stats.frames += 1
592
+ if self.stats.first_timestamp is None:
593
+ self.stats.first_timestamp = message.timestamp
594
+ self._set_started_at(message.timestamp)
595
+ self.stats.last_timestamp = message.timestamp
596
+
597
+ yield message
598
+
599
+ if self.max_frames is not None and self.stats.frames >= self.max_frames:
600
+ self.stats.truncated = True
601
+ return
602
+ except CanLogError:
603
+ raise
604
+ # Reader implementations also surface format-specific exceptions from
605
+ # struct/zlib. Keep those library details behind the public log error.
606
+ except Exception as error:
607
+ raise CanLogError(
608
+ f"cannot parse {self.format.upper()} CAN log {str(self.path)!r}: {error}"
609
+ ) from error
610
+
611
+ def _validate(self) -> None:
612
+ """Reject known unsupported variants before a reader can partially parse them."""
613
+
614
+ def _convert(self, message: Any) -> ReplayMessage:
615
+ channel = message.channel
616
+ if self.one_based_channels and isinstance(channel, int):
617
+ channel += 1
618
+ return ReplayMessage(
619
+ timestamp=float(message.timestamp),
620
+ arbitration_id=int(message.arbitration_id),
621
+ data=bytes(message.data),
622
+ is_extended_id=bool(message.is_extended_id),
623
+ is_error_frame=bool(message.is_error_frame),
624
+ is_remote_frame=bool(message.is_remote_frame),
625
+ is_fd=bool(message.is_fd),
626
+ channel=channel,
627
+ )
628
+
629
+ def _set_started_at(self, timestamp: float) -> None:
630
+ # Some BLF files contain only relative seconds and no wall-clock header.
631
+ # Presenting those as January 1970 would claim precision the file does
632
+ # not have; year 2000 is a conservative boundary for vehicle captures.
633
+ if timestamp < 946_684_800:
634
+ return
635
+ try:
636
+ self.stats.started_at = datetime.fromtimestamp(timestamp, timezone.utc)
637
+ except (OverflowError, OSError, ValueError):
638
+ self.stats.started_at = None
639
+
640
+
641
+ class BlfLogReader(PythonCanLogReader):
642
+ format = "blf"
643
+ reader_name = "BLFReader"
644
+ # python-can exposes BLF's one-based channel field as zero-based.
645
+ one_based_channels = True
646
+
647
+
648
+ class CandumpLogReader(PythonCanLogReader):
649
+ format = "candump"
650
+ reader_name = "CanutilsLogReader"
651
+
652
+
653
+ class TrcLogReader(PythonCanLogReader):
654
+ format = "trc"
655
+ reader_name = "TRCReader"
656
+
657
+ def _validate(self) -> None:
658
+ supported = {"1.0", "1.1", "1.3", "2.0", "2.1"}
659
+ with self.path.open("r", encoding="utf-8", errors="replace") as handle:
660
+ for line in handle:
661
+ if line.startswith(";$FILEVERSION="):
662
+ version = line.partition("=")[2].strip()
663
+ if version not in supported:
664
+ raise CanLogError(
665
+ f"unsupported PEAK TRC version {version!r}; "
666
+ "supported versions are 1.0, 1.1, 1.3, 2.0, and 2.1 "
667
+ "(TRC 3/CAN XL is not supported)"
668
+ )
669
+ return
670
+ if not line.startswith(";"):
671
+ # Headerless TRC is the legacy 1.0 form.
672
+ return
673
+
674
+
675
+ #: Suffixes this can replay, mapped to the reader that reads them.
676
+ READERS = {
677
+ ".asc": AscLogReader,
678
+ ".blf": BlfLogReader,
679
+ ".log": CandumpLogReader,
680
+ ".trc": TrcLogReader,
681
+ }
682
+
683
+
684
+ def open_log(
685
+ path: str | Path,
686
+ *,
687
+ channel: Optional[LogChannel] = None,
688
+ max_frames: Optional[int] = None,
689
+ ) -> AscLogReader | PythonCanLogReader:
690
+ """Pick a reader for a log by its suffix.
691
+
692
+ Refuses an unknown suffix by name rather than guessing from content. This
693
+ also gives the UI and CLI one explicit list of accepted formats.
694
+ """
695
+ resolved = Path(path)
696
+ reader = READERS.get(resolved.suffix.lower())
697
+ if reader is None:
698
+ supported = ", ".join(sorted(READERS))
699
+ raise CanLogError(
700
+ f"{resolved.name!r} is not a CAN log this can replay; "
701
+ f"supported formats: {supported}"
702
+ )
703
+ if not resolved.is_file():
704
+ raise CanLogError(
705
+ f"no CAN log at {str(resolved)!r}. The path is read on the host running "
706
+ "IoTSploit, which is not necessarily the host running the UI."
707
+ )
708
+ return reader(resolved, channel=channel, max_frames=max_frames)
709
+
710
+
711
+ def scan_log(
712
+ path: str | Path,
713
+ *,
714
+ channel: Optional[LogChannel] = None,
715
+ max_frames: Optional[int] = None,
716
+ ) -> LogReadStats:
717
+ """Read a log through once without keeping it, to learn what it covers.
718
+
719
+ A progress bar needs to know the length of the thing it is measuring before
720
+ the first frame is shown, and a log only states that by being read. So a
721
+ replay reads the file twice: once to learn its duration, frame count and
722
+ channels, and once to play it. The pass is cheap -- parsing is a few hundred
723
+ milliseconds for a 3 MB log -- and the alternative is a progress bar that
724
+ only learns its own scale at the moment it finishes, which is no progress
725
+ bar at all.
726
+ """
727
+ reader = open_log(path, channel=channel, max_frames=max_frames)
728
+ for _ in reader.messages():
729
+ pass
730
+ return reader.stats
731
+
732
+
733
+ def identities_from_log(
734
+ path: str | Path,
735
+ *,
736
+ channel: Optional[LogChannel] = None,
737
+ max_frames: Optional[int] = None,
738
+ ) -> Set[Tuple[int, bool]]:
739
+ """Distinct data-frame identities in a log, for scoring it against a target's buses.
740
+
741
+ The live counterpart of this is
742
+ :func:`~iotsploit_protocols.canbus.bus_match.observe_identities`, and it
743
+ exists for the same reason: picking the wrong bus does not fail, it decodes
744
+ every frame to a plausible wrong value. A log needs that check more than a
745
+ live capture does, not less -- whoever recorded it is often not whoever is
746
+ reading it back.
747
+ """
748
+ reader = open_log(path, channel=channel, max_frames=max_frames)
749
+ return {
750
+ (message.arbitration_id, message.is_extended_id)
751
+ for message in reader.messages()
752
+ if not message.is_error_frame and not message.is_remote_frame
753
+ }