iotsploit-protocols 0.0.9__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iotsploit_protocols/__init__.py +20 -0
- iotsploit_protocols/autosar/__init__.py +12 -0
- iotsploit_protocols/autosar/arxml.py +765 -0
- iotsploit_protocols/canbus/__init__.py +56 -0
- iotsploit_protocols/canbus/bus_match.py +121 -0
- iotsploit_protocols/canbus/catalog.py +425 -0
- iotsploit_protocols/canbus/codec.py +455 -0
- iotsploit_protocols/canbus/definitions.py +252 -0
- iotsploit_protocols/canbus/errorframes.py +154 -0
- iotsploit_protocols/canbus/errors.py +47 -0
- iotsploit_protocols/canbus/logfile.py +753 -0
- iotsploit_protocols/canbus/socketcan.py +385 -0
- iotsploit_protocols/doip/__init__.py +22 -0
- iotsploit_protocols/doip/client.py +267 -0
- iotsploit_protocols/doip/facet.py +51 -0
- iotsploit_protocols/doip/uds.py +262 -0
- iotsploit_protocols/errors.py +45 -0
- iotsploit_protocols/someip/__init__.py +19 -0
- iotsploit_protocols/someip/client.py +322 -0
- iotsploit_protocols/someip/codec.py +11 -0
- iotsploit_protocols/someip/facet.py +71 -0
- iotsploit_protocols/someip/sd.py +355 -0
- iotsploit_protocols-0.0.9.dist-info/METADATA +118 -0
- iotsploit_protocols-0.0.9.dist-info/RECORD +25 -0
- iotsploit_protocols-0.0.9.dist-info/WHEEL +4 -0
|
@@ -0,0 +1,753 @@
|
|
|
1
|
+
"""Read a recorded CAN log back as frames, so a stored capture decodes like a live one.
|
|
2
|
+
|
|
3
|
+
What this buys is not convenience. A bus you can only watch live is a bus you
|
|
4
|
+
can only analyse while you are standing next to the vehicle, and a finding you
|
|
5
|
+
cannot replay is a finding nobody else can check. Feeding a file through the
|
|
6
|
+
same aggregator, the same codec, and the same target definitions as a live
|
|
7
|
+
capture is what makes a recorded window reviewable evidence.
|
|
8
|
+
|
|
9
|
+
The messages yielded here are deliberately *not* ``can.Message``. Nothing
|
|
10
|
+
downstream needs one: :class:`~iotsploit_exploits.canbus.live_capture.CaptureAggregator`
|
|
11
|
+
reads identity, payload, arrival time, and the frame-class flags by attribute
|
|
12
|
+
and nothing else. ASC stays a small native parser; binary BLF and the two other
|
|
13
|
+
widely used text formats delegate parsing to ``python-can`` without ever
|
|
14
|
+
constructing a CAN bus or touching a platform socket.
|
|
15
|
+
|
|
16
|
+
Three things in the Vector ASC format are worth knowing before changing this,
|
|
17
|
+
because each one silently produces plausible wrong numbers rather than an
|
|
18
|
+
error:
|
|
19
|
+
|
|
20
|
+
*The ``base`` directive governs identifiers only.* ``base hex`` does not make
|
|
21
|
+
DLC and data-length hexadecimal; those columns are decimal in every ASC this
|
|
22
|
+
has been checked against. Reading them as hex turns the CAN FD length codes
|
|
23
|
+
``10 16``, ``12 24`` and ``13 32`` into 22, 36 and 50 bytes, which truncates
|
|
24
|
+
payloads on exactly the frames that carry the most signal.
|
|
25
|
+
|
|
26
|
+
*DLC is a length code, not a length.* Above eight, CAN FD's DLC indexes a
|
|
27
|
+
table (9 to 15 mean 12, 16, 20, 24, 32, 48, 64 bytes). Both columns are present
|
|
28
|
+
in the file and they must agree; a line where they do not is a line this reader
|
|
29
|
+
did not understand.
|
|
30
|
+
|
|
31
|
+
*There is more than one CAN FD column order in the wild.* Vector's own writer
|
|
32
|
+
puts direction before identifier; other tools emit the classic-CAN order with
|
|
33
|
+
identifier first. Both appear in real logs, so both are read here, anchored on
|
|
34
|
+
whichever column actually holds ``Rx``/``Tx`` rather than on a fixed offset.
|
|
35
|
+
|
|
36
|
+
A line this reader cannot parse is counted and skipped, never guessed at and
|
|
37
|
+
never fatal. A single corrupt line in the middle of fifty thousand must not end
|
|
38
|
+
a replay, and a count of what was skipped is what lets an operator tell a clean
|
|
39
|
+
read from a partial one.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
import re
|
|
45
|
+
from dataclasses import dataclass, field
|
|
46
|
+
from datetime import datetime, timezone
|
|
47
|
+
from pathlib import Path
|
|
48
|
+
from typing import Any, Dict, Iterator, List, Optional, Sequence, Set, Tuple, Type
|
|
49
|
+
|
|
50
|
+
from iotsploit_protocols.errors import NotConfigured
|
|
51
|
+
|
|
52
|
+
#: CAN FD data-length codes above 8. ``linux/can.h`` calls this ``can_fd_dlc2len``.
|
|
53
|
+
DLC_TO_LENGTH: Dict[int, int] = {
|
|
54
|
+
**{code: code for code in range(9)},
|
|
55
|
+
9: 12,
|
|
56
|
+
10: 16,
|
|
57
|
+
11: 20,
|
|
58
|
+
12: 24,
|
|
59
|
+
13: 32,
|
|
60
|
+
14: 48,
|
|
61
|
+
15: 64,
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
#: Every payload size a CAN or CAN FD frame can actually have.
|
|
65
|
+
VALID_LENGTHS = frozenset(DLC_TO_LENGTH.values())
|
|
66
|
+
|
|
67
|
+
#: Direction columns. ``TxRq`` is a transmit *request*, logged in addition to
|
|
68
|
+
#: the ``Tx`` that follows it -- counting both would double every frame this
|
|
69
|
+
#: host sent, so it is skipped rather than replayed.
|
|
70
|
+
_DIRECTIONS = frozenset({"Rx", "Tx"})
|
|
71
|
+
_TX_REQUEST = "TxRq"
|
|
72
|
+
|
|
73
|
+
#: An identifier column, optionally flagged extended with a trailing ``x``.
|
|
74
|
+
_ID_RE = re.compile(r"\A([0-9A-Fa-f]+)(x?)\Z")
|
|
75
|
+
|
|
76
|
+
#: Header forms this reader understands. Anything else in the header is ignored
|
|
77
|
+
#: rather than refused: ASC writers emit tool-specific directives freely, and a
|
|
78
|
+
#: replay must not fail because it met one it had not seen.
|
|
79
|
+
_DATE_FORMATS = (
|
|
80
|
+
"%a %b %d %I:%M:%S %p %Y",
|
|
81
|
+
"%a %b %d %I:%M:%S.%f %p %Y",
|
|
82
|
+
"%a %b %d %H:%M:%S %Y",
|
|
83
|
+
"%a %b %d %H:%M:%S.%f %Y",
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
MAX_STANDARD_FRAME_ID = 0x7FF
|
|
87
|
+
MAX_EXTENDED_FRAME_ID = 0x1FFFFFFF
|
|
88
|
+
LogChannel = int | str
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class CanLogError(NotConfigured):
|
|
92
|
+
"""The log cannot be read, or is not a format this understands.
|
|
93
|
+
|
|
94
|
+
A subclass of :class:`NotConfigured` because every case it covers is a
|
|
95
|
+
choice the operator can correct -- a path that is not there, a suffix
|
|
96
|
+
nothing here parses -- rather than a fault in the bus or the target.
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass(frozen=True)
|
|
101
|
+
class ReplayMessage:
|
|
102
|
+
"""One frame read back from a log.
|
|
103
|
+
|
|
104
|
+
The attribute names are ``python-can``'s because that is the shape the
|
|
105
|
+
aggregator reads. The class is a plain dataclass because that is all the
|
|
106
|
+
shape actually requires.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
timestamp: float
|
|
110
|
+
arbitration_id: int
|
|
111
|
+
data: bytes
|
|
112
|
+
is_extended_id: bool = False
|
|
113
|
+
is_error_frame: bool = False
|
|
114
|
+
is_remote_frame: bool = False
|
|
115
|
+
is_fd: bool = False
|
|
116
|
+
channel: Optional[LogChannel] = None
|
|
117
|
+
|
|
118
|
+
@property
|
|
119
|
+
def dlc(self) -> int:
|
|
120
|
+
return len(self.data)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@dataclass
|
|
124
|
+
class LogReadStats:
|
|
125
|
+
"""What a read actually saw, as opposed to what it was asked for.
|
|
126
|
+
|
|
127
|
+
Read after iterating. ``unparsable_lines`` is the number that matters: a
|
|
128
|
+
replay reporting 47,797 frames and 0 skipped lines is a clean read of the
|
|
129
|
+
whole file, and the same replay reporting 12,000 skipped lines is a partial
|
|
130
|
+
one wearing the same summary.
|
|
131
|
+
"""
|
|
132
|
+
|
|
133
|
+
format: str = "asc"
|
|
134
|
+
frames: int = 0
|
|
135
|
+
error_frames: int = 0
|
|
136
|
+
unparsable_lines: int = 0
|
|
137
|
+
channels_present: Set[LogChannel] = field(default_factory=set)
|
|
138
|
+
first_timestamp: Optional[float] = None
|
|
139
|
+
last_timestamp: Optional[float] = None
|
|
140
|
+
#: Wall-clock start from the log's own ``date`` header, when it had one.
|
|
141
|
+
started_at: Optional[datetime] = None
|
|
142
|
+
truncated: bool = False
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def duration_s(self) -> float:
|
|
146
|
+
if self.first_timestamp is None or self.last_timestamp is None:
|
|
147
|
+
return 0.0
|
|
148
|
+
return round(self.last_timestamp - self.first_timestamp, 6)
|
|
149
|
+
|
|
150
|
+
def as_dict(self) -> Dict[str, Any]:
|
|
151
|
+
return {
|
|
152
|
+
"format": self.format,
|
|
153
|
+
"frames": self.frames,
|
|
154
|
+
"error_frames": self.error_frames,
|
|
155
|
+
"unparsable_lines": self.unparsable_lines,
|
|
156
|
+
"channels_present": sorted(self.channels_present, key=channel_sort_key),
|
|
157
|
+
"duration_s": self.duration_s,
|
|
158
|
+
"started_at": self.started_at.isoformat() if self.started_at else None,
|
|
159
|
+
"truncated": self.truncated,
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def channel_sort_key(channel: LogChannel) -> Tuple[int, str]:
|
|
164
|
+
"""Keep numeric channels ordered before named interfaces such as ``can0``."""
|
|
165
|
+
if isinstance(channel, int):
|
|
166
|
+
return 0, f"{channel:020d}"
|
|
167
|
+
return 1, channel
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def normalize_log_channel(value: Any) -> Optional[LogChannel]:
|
|
171
|
+
"""Normalize an API/CLI channel without losing candump interface names."""
|
|
172
|
+
if value is None:
|
|
173
|
+
return None
|
|
174
|
+
if isinstance(value, bool):
|
|
175
|
+
raise ValueError("log_channel must be a channel number or interface name")
|
|
176
|
+
if isinstance(value, int):
|
|
177
|
+
if value < 0:
|
|
178
|
+
raise ValueError("log_channel must not be negative")
|
|
179
|
+
return value
|
|
180
|
+
if isinstance(value, str):
|
|
181
|
+
channel = value.strip()
|
|
182
|
+
if not channel:
|
|
183
|
+
raise ValueError("log_channel must not be empty")
|
|
184
|
+
return int(channel) if channel.isdigit() else channel
|
|
185
|
+
raise ValueError("log_channel must be a channel number or interface name")
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def select_log_channel(
|
|
189
|
+
channels: Set[LogChannel], requested: Optional[LogChannel]
|
|
190
|
+
) -> LogChannel:
|
|
191
|
+
"""Resolve the one bus a replay may decode against one target definition."""
|
|
192
|
+
if not channels:
|
|
193
|
+
raise CanLogError("the CAN log contains no readable channels")
|
|
194
|
+
if requested is not None:
|
|
195
|
+
if requested not in channels:
|
|
196
|
+
available = ", ".join(
|
|
197
|
+
str(channel) for channel in sorted(channels, key=channel_sort_key)
|
|
198
|
+
)
|
|
199
|
+
raise CanLogError(
|
|
200
|
+
f"channel {requested!r} is not present in the CAN log; "
|
|
201
|
+
f"available channels: {available}"
|
|
202
|
+
)
|
|
203
|
+
return requested
|
|
204
|
+
if len(channels) == 1:
|
|
205
|
+
return next(iter(channels))
|
|
206
|
+
available = ", ".join(
|
|
207
|
+
str(channel) for channel in sorted(channels, key=channel_sort_key)
|
|
208
|
+
)
|
|
209
|
+
raise CanLogError(
|
|
210
|
+
f"the CAN log contains multiple channels ({available}); choose log_channel"
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _parse_identifier(token: str, base: int) -> Optional[Tuple[int, bool]]:
|
|
215
|
+
match = _ID_RE.match(token)
|
|
216
|
+
if match is None:
|
|
217
|
+
return None
|
|
218
|
+
digits, extended_flag = match.group(1), match.group(2) == "x"
|
|
219
|
+
try:
|
|
220
|
+
frame_id = int(digits, base)
|
|
221
|
+
except ValueError:
|
|
222
|
+
return None
|
|
223
|
+
ceiling = MAX_EXTENDED_FRAME_ID if extended_flag else MAX_STANDARD_FRAME_ID
|
|
224
|
+
if frame_id > ceiling:
|
|
225
|
+
# An identifier too wide for the width it claims is a mis-read column,
|
|
226
|
+
# not a frame. Counting it would put an address on screen that no ECU
|
|
227
|
+
# can hold.
|
|
228
|
+
return None
|
|
229
|
+
return frame_id, extended_flag
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _parse_payload(tokens: Sequence[str], length: int) -> Optional[bytes]:
|
|
233
|
+
if len(tokens) < length:
|
|
234
|
+
return None
|
|
235
|
+
try:
|
|
236
|
+
return bytes(int(token, 16) for token in tokens[:length])
|
|
237
|
+
except ValueError:
|
|
238
|
+
return None
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
#: Longest decimal field this reader will convert. ``int()`` refuses a string
|
|
242
|
+
#: of more than 4300 digits and raises ValueError, and this reader's contract
|
|
243
|
+
#: is that a line it cannot parse is counted and skipped, never fatal. No
|
|
244
|
+
#: column in an ASC log -- a DLC, a channel, a length -- is more than a few
|
|
245
|
+
#: digits, so a longer one means the line is not what it claims to be.
|
|
246
|
+
MAX_DECIMAL_DIGITS = 18
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _is_decimal(token: str) -> bool:
|
|
250
|
+
return token.isdigit() and len(token) <= MAX_DECIMAL_DIGITS
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _consume_frame_body(
|
|
254
|
+
tokens: Sequence[str], *, fd: bool
|
|
255
|
+
) -> Optional[Tuple[bool, bytes]]:
|
|
256
|
+
"""Read ``[name] [brs esi] [d|r] <dlc> [length] <data...>`` into a payload.
|
|
257
|
+
|
|
258
|
+
The optional columns are what make one routine serve both the classic and
|
|
259
|
+
the CAN FD line shapes, and both CAN FD column orders. Everything optional
|
|
260
|
+
here is optional in some real writer's output; nothing is optional to make
|
|
261
|
+
a malformed line pass.
|
|
262
|
+
|
|
263
|
+
Returns the remote flag and the payload, or ``None`` when the columns do
|
|
264
|
+
not form a frame -- which is the signal to count the line as unparsable
|
|
265
|
+
rather than to publish a guess.
|
|
266
|
+
"""
|
|
267
|
+
index = 0
|
|
268
|
+
# A symbolic message name, when the writer emitted one. Never a bare
|
|
269
|
+
# integer, so it cannot be confused with the numeric columns that follow.
|
|
270
|
+
if index < len(tokens) and not _is_decimal(tokens[index]) and tokens[index] not in {"d", "r"}:
|
|
271
|
+
index += 1
|
|
272
|
+
|
|
273
|
+
if fd:
|
|
274
|
+
# BRS and ESI. Present in every CAN FD line; their absence means the
|
|
275
|
+
# columns are not the ones this expects.
|
|
276
|
+
if index + 1 >= len(tokens) or not (
|
|
277
|
+
_is_decimal(tokens[index]) and _is_decimal(tokens[index + 1])
|
|
278
|
+
):
|
|
279
|
+
return None
|
|
280
|
+
index += 2
|
|
281
|
+
|
|
282
|
+
is_remote = False
|
|
283
|
+
if index < len(tokens) and tokens[index] in {"d", "r"}:
|
|
284
|
+
is_remote = tokens[index] == "r"
|
|
285
|
+
index += 1
|
|
286
|
+
|
|
287
|
+
if index >= len(tokens) or not _is_decimal(tokens[index]):
|
|
288
|
+
return None
|
|
289
|
+
dlc = int(tokens[index])
|
|
290
|
+
index += 1
|
|
291
|
+
|
|
292
|
+
length = DLC_TO_LENGTH.get(dlc)
|
|
293
|
+
if length is None:
|
|
294
|
+
return None
|
|
295
|
+
|
|
296
|
+
# CAN FD states the byte count in its own column. When it is there the two
|
|
297
|
+
# must agree: a DLC and a length that disagree mean the columns have been
|
|
298
|
+
# read at the wrong offset, and the payload that follows is not the one
|
|
299
|
+
# this line describes.
|
|
300
|
+
if fd:
|
|
301
|
+
if index >= len(tokens) or not _is_decimal(tokens[index]):
|
|
302
|
+
return None
|
|
303
|
+
stated = int(tokens[index])
|
|
304
|
+
if stated != length:
|
|
305
|
+
return None
|
|
306
|
+
index += 1
|
|
307
|
+
|
|
308
|
+
if is_remote:
|
|
309
|
+
# A remote frame requests a payload rather than carrying one.
|
|
310
|
+
return True, b""
|
|
311
|
+
|
|
312
|
+
payload = _parse_payload(tokens[index:], length)
|
|
313
|
+
if payload is None:
|
|
314
|
+
return None
|
|
315
|
+
return False, payload
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
class AscLogReader:
|
|
319
|
+
"""A Vector ASC log, read as a stream of frames.
|
|
320
|
+
|
|
321
|
+
Streaming rather than loading: an ASC of a busy bus runs to hundreds of
|
|
322
|
+
megabytes, and a replay that has to fit the whole log in memory before
|
|
323
|
+
showing the first row is a replay that fails on the logs worth replaying.
|
|
324
|
+
|
|
325
|
+
Read :attr:`stats` after iterating to find out what the read actually saw.
|
|
326
|
+
"""
|
|
327
|
+
|
|
328
|
+
format = "asc"
|
|
329
|
+
|
|
330
|
+
def __init__(
|
|
331
|
+
self,
|
|
332
|
+
path: str | Path,
|
|
333
|
+
*,
|
|
334
|
+
channel: Optional[LogChannel] = None,
|
|
335
|
+
max_frames: Optional[int] = None,
|
|
336
|
+
) -> None:
|
|
337
|
+
self.path = Path(path)
|
|
338
|
+
#: Which bus in the log to replay. A multi-channel ASC holds several
|
|
339
|
+
#: buses, and decoding all of them against one target bus is how a
|
|
340
|
+
#: replay produces values that look right and are not -- so a caller
|
|
341
|
+
#: that knows which channel it wants says so.
|
|
342
|
+
self.channel = channel
|
|
343
|
+
self.max_frames = max_frames
|
|
344
|
+
self.stats = LogReadStats(format=self.format)
|
|
345
|
+
self._base = 16
|
|
346
|
+
self._epoch = 0.0
|
|
347
|
+
|
|
348
|
+
def messages(self) -> Iterator[ReplayMessage]:
|
|
349
|
+
"""Yield frames until the log or the frame budget runs out."""
|
|
350
|
+
try:
|
|
351
|
+
handle = self.path.open("r", encoding="utf-8", errors="replace")
|
|
352
|
+
except OSError as error:
|
|
353
|
+
raise CanLogError(f"cannot read CAN log {str(self.path)!r}: {error}") from error
|
|
354
|
+
|
|
355
|
+
with handle:
|
|
356
|
+
for line in handle:
|
|
357
|
+
tokens = line.split()
|
|
358
|
+
if not tokens:
|
|
359
|
+
continue
|
|
360
|
+
if self._read_directive(tokens):
|
|
361
|
+
continue
|
|
362
|
+
|
|
363
|
+
message = self._read_frame(tokens)
|
|
364
|
+
if message is None:
|
|
365
|
+
continue
|
|
366
|
+
|
|
367
|
+
observed_channel = message.channel if message.channel is not None else 0
|
|
368
|
+
self.stats.channels_present.add(observed_channel)
|
|
369
|
+
if self.channel is not None and observed_channel != self.channel:
|
|
370
|
+
continue
|
|
371
|
+
|
|
372
|
+
if message.is_error_frame:
|
|
373
|
+
self.stats.error_frames += 1
|
|
374
|
+
else:
|
|
375
|
+
self.stats.frames += 1
|
|
376
|
+
if self.stats.first_timestamp is None:
|
|
377
|
+
self.stats.first_timestamp = message.timestamp
|
|
378
|
+
self.stats.last_timestamp = message.timestamp
|
|
379
|
+
|
|
380
|
+
yield message
|
|
381
|
+
|
|
382
|
+
if self.max_frames is not None and self.stats.frames >= self.max_frames:
|
|
383
|
+
self.stats.truncated = True
|
|
384
|
+
return
|
|
385
|
+
|
|
386
|
+
# ── header ────────────────────────────────────────────────────────
|
|
387
|
+
|
|
388
|
+
def _read_directive(self, tokens: List[str]) -> bool:
|
|
389
|
+
"""Consume a header line, returning whether it was one.
|
|
390
|
+
|
|
391
|
+
A directive is recognised by its keyword rather than by its position:
|
|
392
|
+
ASC writers put ``base`` and ``timestamps`` on one line or two, and in
|
|
393
|
+
either order.
|
|
394
|
+
"""
|
|
395
|
+
head = tokens[0]
|
|
396
|
+
if head.startswith("//"):
|
|
397
|
+
return True
|
|
398
|
+
if head == "date":
|
|
399
|
+
self._read_date(tokens[1:])
|
|
400
|
+
return True
|
|
401
|
+
if head in {"base", "internal"} or (head in {"Begin", "End"} and len(tokens) > 1):
|
|
402
|
+
if "hex" in tokens:
|
|
403
|
+
self._base = 16
|
|
404
|
+
elif "dec" in tokens:
|
|
405
|
+
self._base = 10
|
|
406
|
+
return True
|
|
407
|
+
if head in {"Measurement", "previous"}:
|
|
408
|
+
return True
|
|
409
|
+
# A frame line always opens with a timestamp. Anything else that does
|
|
410
|
+
# not is a directive this reader has no use for.
|
|
411
|
+
try:
|
|
412
|
+
float(head)
|
|
413
|
+
except ValueError:
|
|
414
|
+
return True
|
|
415
|
+
return False
|
|
416
|
+
|
|
417
|
+
def _read_date(self, tokens: Sequence[str]) -> None:
|
|
418
|
+
"""Anchor the log's relative timestamps to the wall clock it recorded.
|
|
419
|
+
|
|
420
|
+
Without this every ``first_seen_at`` in a replay renders as 1970, because
|
|
421
|
+
an ASC counts seconds from the start of its own measurement. The log
|
|
422
|
+
states when that was; using it is the difference between a timestamp an
|
|
423
|
+
operator can correlate with anything else and one they cannot.
|
|
424
|
+
"""
|
|
425
|
+
text = " ".join(tokens)
|
|
426
|
+
for fmt in _DATE_FORMATS:
|
|
427
|
+
try:
|
|
428
|
+
parsed = datetime.strptime(text, fmt)
|
|
429
|
+
except ValueError:
|
|
430
|
+
continue
|
|
431
|
+
# An ASC date header is wall-clock time on the machine that
|
|
432
|
+
# recorded it, with no zone stated. Reading it as local time is the
|
|
433
|
+
# only available interpretation; stamping the zone on makes that
|
|
434
|
+
# assumption visible in the result instead of leaving a bare naive
|
|
435
|
+
# time that reads as UTC next to arrival times that are not.
|
|
436
|
+
self.stats.started_at = parsed.astimezone()
|
|
437
|
+
self._epoch = parsed.timestamp()
|
|
438
|
+
return
|
|
439
|
+
|
|
440
|
+
# ── frames ────────────────────────────────────────────────────────
|
|
441
|
+
|
|
442
|
+
def _read_frame(self, tokens: List[str]) -> Optional[ReplayMessage]:
|
|
443
|
+
try:
|
|
444
|
+
timestamp = float(tokens[0]) + self._epoch
|
|
445
|
+
except ValueError:
|
|
446
|
+
return None
|
|
447
|
+
|
|
448
|
+
if len(tokens) < 3:
|
|
449
|
+
self.stats.unparsable_lines += 1
|
|
450
|
+
return None
|
|
451
|
+
|
|
452
|
+
# A bus fault is not traffic, and the aggregator tallies it separately.
|
|
453
|
+
# Recognised before anything reads a column as an identifier, for the
|
|
454
|
+
# same reason the live path classifies before it reads identity.
|
|
455
|
+
if any(token.startswith("ErrorFrame") for token in tokens):
|
|
456
|
+
return ReplayMessage(
|
|
457
|
+
timestamp=timestamp,
|
|
458
|
+
arbitration_id=0,
|
|
459
|
+
data=b"",
|
|
460
|
+
is_error_frame=True,
|
|
461
|
+
channel=self._read_channel(tokens[1]),
|
|
462
|
+
)
|
|
463
|
+
|
|
464
|
+
if tokens[1] == "CANFD":
|
|
465
|
+
return self._read_fd_frame(timestamp, tokens[2:])
|
|
466
|
+
if tokens[1] == "CAN":
|
|
467
|
+
# Some writers label classic lines explicitly.
|
|
468
|
+
return self._read_classic_frame(timestamp, tokens[2:])
|
|
469
|
+
return self._read_classic_frame(timestamp, tokens[1:])
|
|
470
|
+
|
|
471
|
+
@staticmethod
|
|
472
|
+
def _read_channel(token: str) -> Optional[int]:
|
|
473
|
+
return int(token) if _is_decimal(token) else None
|
|
474
|
+
|
|
475
|
+
def _read_fd_frame(self, timestamp: float, tokens: List[str]) -> Optional[ReplayMessage]:
|
|
476
|
+
"""Read a CAN FD line in either of the two column orders found in real logs.
|
|
477
|
+
|
|
478
|
+
Vector's writer emits ``channel direction identifier``; tools that grew
|
|
479
|
+
out of the classic-CAN line emit ``channel identifier direction``. The
|
|
480
|
+
direction column is what tells them apart, so it is located rather than
|
|
481
|
+
assumed.
|
|
482
|
+
"""
|
|
483
|
+
if len(tokens) < 6:
|
|
484
|
+
self.stats.unparsable_lines += 1
|
|
485
|
+
return None
|
|
486
|
+
channel = self._read_channel(tokens[0])
|
|
487
|
+
|
|
488
|
+
if tokens[1] in _DIRECTIONS:
|
|
489
|
+
identifier, rest = tokens[2], tokens[3:]
|
|
490
|
+
elif len(tokens) > 2 and tokens[2] in _DIRECTIONS:
|
|
491
|
+
identifier, rest = tokens[1], tokens[3:]
|
|
492
|
+
elif _TX_REQUEST in tokens[1:3]:
|
|
493
|
+
# Logged alongside the Tx it precedes; replaying both would count
|
|
494
|
+
# every transmitted frame twice.
|
|
495
|
+
return None
|
|
496
|
+
else:
|
|
497
|
+
self.stats.unparsable_lines += 1
|
|
498
|
+
return None
|
|
499
|
+
|
|
500
|
+
parsed_id = _parse_identifier(identifier, self._base)
|
|
501
|
+
body = _consume_frame_body(rest, fd=True) if parsed_id else None
|
|
502
|
+
if parsed_id is None or body is None:
|
|
503
|
+
self.stats.unparsable_lines += 1
|
|
504
|
+
return None
|
|
505
|
+
|
|
506
|
+
frame_id, is_extended = parsed_id
|
|
507
|
+
is_remote, payload = body
|
|
508
|
+
return ReplayMessage(
|
|
509
|
+
timestamp=timestamp,
|
|
510
|
+
arbitration_id=frame_id,
|
|
511
|
+
data=payload,
|
|
512
|
+
is_extended_id=is_extended,
|
|
513
|
+
is_remote_frame=is_remote,
|
|
514
|
+
is_fd=True,
|
|
515
|
+
channel=channel,
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
def _read_classic_frame(
|
|
519
|
+
self, timestamp: float, tokens: List[str]
|
|
520
|
+
) -> Optional[ReplayMessage]:
|
|
521
|
+
"""Read ``<channel> <identifier> <direction> <d|r> <dlc> <data...>``."""
|
|
522
|
+
if len(tokens) < 4:
|
|
523
|
+
self.stats.unparsable_lines += 1
|
|
524
|
+
return None
|
|
525
|
+
channel = self._read_channel(tokens[0])
|
|
526
|
+
|
|
527
|
+
if tokens[2] in _DIRECTIONS:
|
|
528
|
+
identifier, rest = tokens[1], tokens[3:]
|
|
529
|
+
elif tokens[1] in _DIRECTIONS:
|
|
530
|
+
identifier, rest = tokens[2], tokens[3:]
|
|
531
|
+
elif _TX_REQUEST in tokens[1:3]:
|
|
532
|
+
return None
|
|
533
|
+
else:
|
|
534
|
+
self.stats.unparsable_lines += 1
|
|
535
|
+
return None
|
|
536
|
+
|
|
537
|
+
parsed_id = _parse_identifier(identifier, self._base)
|
|
538
|
+
body = _consume_frame_body(rest, fd=False) if parsed_id else None
|
|
539
|
+
if parsed_id is None or body is None:
|
|
540
|
+
self.stats.unparsable_lines += 1
|
|
541
|
+
return None
|
|
542
|
+
|
|
543
|
+
frame_id, is_extended = parsed_id
|
|
544
|
+
is_remote, payload = body
|
|
545
|
+
return ReplayMessage(
|
|
546
|
+
timestamp=timestamp,
|
|
547
|
+
arbitration_id=frame_id,
|
|
548
|
+
data=payload,
|
|
549
|
+
is_extended_id=is_extended,
|
|
550
|
+
is_remote_frame=is_remote,
|
|
551
|
+
channel=channel,
|
|
552
|
+
)
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
class PythonCanLogReader:
|
|
556
|
+
"""Stream a format handled by python-can into the replay message shape."""
|
|
557
|
+
|
|
558
|
+
format = ""
|
|
559
|
+
reader_name = ""
|
|
560
|
+
one_based_channels = False
|
|
561
|
+
|
|
562
|
+
def __init__(
|
|
563
|
+
self,
|
|
564
|
+
path: str | Path,
|
|
565
|
+
*,
|
|
566
|
+
channel: Optional[LogChannel] = None,
|
|
567
|
+
max_frames: Optional[int] = None,
|
|
568
|
+
) -> None:
|
|
569
|
+
self.path = Path(path)
|
|
570
|
+
self.channel = channel
|
|
571
|
+
self.max_frames = max_frames
|
|
572
|
+
self.stats = LogReadStats(format=self.format)
|
|
573
|
+
|
|
574
|
+
def messages(self) -> Iterator[ReplayMessage]:
|
|
575
|
+
try:
|
|
576
|
+
self._validate()
|
|
577
|
+
from can import io
|
|
578
|
+
|
|
579
|
+
reader_type: Type[Any] = getattr(io, self.reader_name)
|
|
580
|
+
with reader_type(self.path) as source:
|
|
581
|
+
for source_message in source:
|
|
582
|
+
message = self._convert(source_message)
|
|
583
|
+
observed_channel = message.channel if message.channel is not None else 0
|
|
584
|
+
self.stats.channels_present.add(observed_channel)
|
|
585
|
+
if self.channel is not None and observed_channel != self.channel:
|
|
586
|
+
continue
|
|
587
|
+
|
|
588
|
+
if message.is_error_frame:
|
|
589
|
+
self.stats.error_frames += 1
|
|
590
|
+
else:
|
|
591
|
+
self.stats.frames += 1
|
|
592
|
+
if self.stats.first_timestamp is None:
|
|
593
|
+
self.stats.first_timestamp = message.timestamp
|
|
594
|
+
self._set_started_at(message.timestamp)
|
|
595
|
+
self.stats.last_timestamp = message.timestamp
|
|
596
|
+
|
|
597
|
+
yield message
|
|
598
|
+
|
|
599
|
+
if self.max_frames is not None and self.stats.frames >= self.max_frames:
|
|
600
|
+
self.stats.truncated = True
|
|
601
|
+
return
|
|
602
|
+
except CanLogError:
|
|
603
|
+
raise
|
|
604
|
+
# Reader implementations also surface format-specific exceptions from
|
|
605
|
+
# struct/zlib. Keep those library details behind the public log error.
|
|
606
|
+
except Exception as error:
|
|
607
|
+
raise CanLogError(
|
|
608
|
+
f"cannot parse {self.format.upper()} CAN log {str(self.path)!r}: {error}"
|
|
609
|
+
) from error
|
|
610
|
+
|
|
611
|
+
def _validate(self) -> None:
|
|
612
|
+
"""Reject known unsupported variants before a reader can partially parse them."""
|
|
613
|
+
|
|
614
|
+
def _convert(self, message: Any) -> ReplayMessage:
|
|
615
|
+
channel = message.channel
|
|
616
|
+
if self.one_based_channels and isinstance(channel, int):
|
|
617
|
+
channel += 1
|
|
618
|
+
return ReplayMessage(
|
|
619
|
+
timestamp=float(message.timestamp),
|
|
620
|
+
arbitration_id=int(message.arbitration_id),
|
|
621
|
+
data=bytes(message.data),
|
|
622
|
+
is_extended_id=bool(message.is_extended_id),
|
|
623
|
+
is_error_frame=bool(message.is_error_frame),
|
|
624
|
+
is_remote_frame=bool(message.is_remote_frame),
|
|
625
|
+
is_fd=bool(message.is_fd),
|
|
626
|
+
channel=channel,
|
|
627
|
+
)
|
|
628
|
+
|
|
629
|
+
def _set_started_at(self, timestamp: float) -> None:
|
|
630
|
+
# Some BLF files contain only relative seconds and no wall-clock header.
|
|
631
|
+
# Presenting those as January 1970 would claim precision the file does
|
|
632
|
+
# not have; year 2000 is a conservative boundary for vehicle captures.
|
|
633
|
+
if timestamp < 946_684_800:
|
|
634
|
+
return
|
|
635
|
+
try:
|
|
636
|
+
self.stats.started_at = datetime.fromtimestamp(timestamp, timezone.utc)
|
|
637
|
+
except (OverflowError, OSError, ValueError):
|
|
638
|
+
self.stats.started_at = None
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
class BlfLogReader(PythonCanLogReader):
|
|
642
|
+
format = "blf"
|
|
643
|
+
reader_name = "BLFReader"
|
|
644
|
+
# python-can exposes BLF's one-based channel field as zero-based.
|
|
645
|
+
one_based_channels = True
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
class CandumpLogReader(PythonCanLogReader):
|
|
649
|
+
format = "candump"
|
|
650
|
+
reader_name = "CanutilsLogReader"
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
class TrcLogReader(PythonCanLogReader):
|
|
654
|
+
format = "trc"
|
|
655
|
+
reader_name = "TRCReader"
|
|
656
|
+
|
|
657
|
+
def _validate(self) -> None:
|
|
658
|
+
supported = {"1.0", "1.1", "1.3", "2.0", "2.1"}
|
|
659
|
+
with self.path.open("r", encoding="utf-8", errors="replace") as handle:
|
|
660
|
+
for line in handle:
|
|
661
|
+
if line.startswith(";$FILEVERSION="):
|
|
662
|
+
version = line.partition("=")[2].strip()
|
|
663
|
+
if version not in supported:
|
|
664
|
+
raise CanLogError(
|
|
665
|
+
f"unsupported PEAK TRC version {version!r}; "
|
|
666
|
+
"supported versions are 1.0, 1.1, 1.3, 2.0, and 2.1 "
|
|
667
|
+
"(TRC 3/CAN XL is not supported)"
|
|
668
|
+
)
|
|
669
|
+
return
|
|
670
|
+
if not line.startswith(";"):
|
|
671
|
+
# Headerless TRC is the legacy 1.0 form.
|
|
672
|
+
return
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
#: Suffixes this can replay, mapped to the reader that reads them.
|
|
676
|
+
READERS = {
|
|
677
|
+
".asc": AscLogReader,
|
|
678
|
+
".blf": BlfLogReader,
|
|
679
|
+
".log": CandumpLogReader,
|
|
680
|
+
".trc": TrcLogReader,
|
|
681
|
+
}
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
def open_log(
|
|
685
|
+
path: str | Path,
|
|
686
|
+
*,
|
|
687
|
+
channel: Optional[LogChannel] = None,
|
|
688
|
+
max_frames: Optional[int] = None,
|
|
689
|
+
) -> AscLogReader | PythonCanLogReader:
|
|
690
|
+
"""Pick a reader for a log by its suffix.
|
|
691
|
+
|
|
692
|
+
Refuses an unknown suffix by name rather than guessing from content. This
|
|
693
|
+
also gives the UI and CLI one explicit list of accepted formats.
|
|
694
|
+
"""
|
|
695
|
+
resolved = Path(path)
|
|
696
|
+
reader = READERS.get(resolved.suffix.lower())
|
|
697
|
+
if reader is None:
|
|
698
|
+
supported = ", ".join(sorted(READERS))
|
|
699
|
+
raise CanLogError(
|
|
700
|
+
f"{resolved.name!r} is not a CAN log this can replay; "
|
|
701
|
+
f"supported formats: {supported}"
|
|
702
|
+
)
|
|
703
|
+
if not resolved.is_file():
|
|
704
|
+
raise CanLogError(
|
|
705
|
+
f"no CAN log at {str(resolved)!r}. The path is read on the host running "
|
|
706
|
+
"IoTSploit, which is not necessarily the host running the UI."
|
|
707
|
+
)
|
|
708
|
+
return reader(resolved, channel=channel, max_frames=max_frames)
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
def scan_log(
|
|
712
|
+
path: str | Path,
|
|
713
|
+
*,
|
|
714
|
+
channel: Optional[LogChannel] = None,
|
|
715
|
+
max_frames: Optional[int] = None,
|
|
716
|
+
) -> LogReadStats:
|
|
717
|
+
"""Read a log through once without keeping it, to learn what it covers.
|
|
718
|
+
|
|
719
|
+
A progress bar needs to know the length of the thing it is measuring before
|
|
720
|
+
the first frame is shown, and a log only states that by being read. So a
|
|
721
|
+
replay reads the file twice: once to learn its duration, frame count and
|
|
722
|
+
channels, and once to play it. The pass is cheap -- parsing is a few hundred
|
|
723
|
+
milliseconds for a 3 MB log -- and the alternative is a progress bar that
|
|
724
|
+
only learns its own scale at the moment it finishes, which is no progress
|
|
725
|
+
bar at all.
|
|
726
|
+
"""
|
|
727
|
+
reader = open_log(path, channel=channel, max_frames=max_frames)
|
|
728
|
+
for _ in reader.messages():
|
|
729
|
+
pass
|
|
730
|
+
return reader.stats
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
def identities_from_log(
|
|
734
|
+
path: str | Path,
|
|
735
|
+
*,
|
|
736
|
+
channel: Optional[LogChannel] = None,
|
|
737
|
+
max_frames: Optional[int] = None,
|
|
738
|
+
) -> Set[Tuple[int, bool]]:
|
|
739
|
+
"""Distinct data-frame identities in a log, for scoring it against a target's buses.
|
|
740
|
+
|
|
741
|
+
The live counterpart of this is
|
|
742
|
+
:func:`~iotsploit_protocols.canbus.bus_match.observe_identities`, and it
|
|
743
|
+
exists for the same reason: picking the wrong bus does not fail, it decodes
|
|
744
|
+
every frame to a plausible wrong value. A log needs that check more than a
|
|
745
|
+
live capture does, not less -- whoever recorded it is often not whoever is
|
|
746
|
+
reading it back.
|
|
747
|
+
"""
|
|
748
|
+
reader = open_log(path, channel=channel, max_frames=max_frames)
|
|
749
|
+
return {
|
|
750
|
+
(message.arbitration_id, message.is_extended_id)
|
|
751
|
+
for message in reader.messages()
|
|
752
|
+
if not message.is_error_frame and not message.is_remote_frame
|
|
753
|
+
}
|