bdo-toolkit 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bdo_toolkit/__init__.py +87 -0
- bdo_toolkit/_async_sessions.py +651 -0
- bdo_toolkit/_capture_backend.py +194 -0
- bdo_toolkit/_capture_options.py +68 -0
- bdo_toolkit/_capture_runtime.py +626 -0
- bdo_toolkit/_deposit_origin.py +1599 -0
- bdo_toolkit/_engine.py +327 -0
- bdo_toolkit/_framing.py +904 -0
- bdo_toolkit/_profile_runtime.py +157 -0
- bdo_toolkit/_protocol.py +386 -0
- bdo_toolkit/_reassembly.py +654 -0
- bdo_toolkit/_specs.py +285 -0
- bdo_toolkit/_storage_destination_validation.py +167 -0
- bdo_toolkit/_storage_hydration.py +241 -0
- bdo_toolkit/_version.py +3 -0
- bdo_toolkit/calibration.py +3223 -0
- bdo_toolkit/capture.py +1713 -0
- bdo_toolkit/character_state.py +3506 -0
- bdo_toolkit/cli.py +948 -0
- bdo_toolkit/diagnostics.py +51 -0
- bdo_toolkit/events.py +214 -0
- bdo_toolkit/filters.py +105 -0
- bdo_toolkit/item_state.py +48 -0
- bdo_toolkit/origin_learning.py +779 -0
- bdo_toolkit/profiles.py +370 -0
- bdo_toolkit/py.typed +1 -0
- bdo_toolkit/remote_profiles.py +358 -0
- bdo_toolkit/solare/__init__.py +50 -0
- bdo_toolkit/solare/_constants.py +94 -0
- bdo_toolkit/solare/_detail_learning.py +1437 -0
- bdo_toolkit/solare/_details.py +796 -0
- bdo_toolkit/solare/_discovery.py +1212 -0
- bdo_toolkit/solare/_live_tracker.py +472 -0
- bdo_toolkit/solare/_replay_capture.py +182 -0
- bdo_toolkit/solare/_result.py +441 -0
- bdo_toolkit/solare/_scanner.py +203 -0
- bdo_toolkit/solare/_validation.py +11 -0
- bdo_toolkit/solare/async_session.py +444 -0
- bdo_toolkit/solare/models.py +806 -0
- bdo_toolkit/solare/replay.py +62 -0
- bdo_toolkit/solare/session.py +1051 -0
- bdo_toolkit/writers.py +30 -0
- bdo_toolkit-1.0.0.dist-info/METADATA +143 -0
- bdo_toolkit-1.0.0.dist-info/RECORD +48 -0
- bdo_toolkit-1.0.0.dist-info/WHEEL +5 -0
- bdo_toolkit-1.0.0.dist-info/entry_points.txt +2 -0
- bdo_toolkit-1.0.0.dist-info/licenses/LICENSE +21 -0
- bdo_toolkit-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,779 @@
|
|
|
1
|
+
"""Structural worker-companion discovery and explicit candidate persistence."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime as dt
|
|
6
|
+
import hashlib
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import shutil
|
|
10
|
+
import tempfile
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from threading import Lock
|
|
14
|
+
from typing import Any, Iterable, Optional
|
|
15
|
+
|
|
16
|
+
from ._protocol import FlowKey, MAX_TARGET_MESSAGE_LENGTH
|
|
17
|
+
from .profiles import ProfileError, load_opcode_profile
|
|
18
|
+
|
|
19
|
+
TOKEN_WIDTH = 8
|
|
20
|
+
# Five-byte diversity was just permissive enough to treat an initial-load
|
|
21
|
+
# timestamp as a transaction token. A worker token must now contain more than
|
|
22
|
+
# five distinct byte values.
|
|
23
|
+
MIN_TOKEN_UNIQUE_BYTES = 6
|
|
24
|
+
CANDIDATE_FORMAT_VERSION = 1
|
|
25
|
+
COMPANION_DETECTION = "shared-token-chain-v1"
|
|
26
|
+
DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES = 256
|
|
27
|
+
DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS = 10_000
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class OriginLearningLimitError(RuntimeError):
|
|
31
|
+
"""Raised before an origin learner would exceed its retention budget."""
|
|
32
|
+
|
|
33
|
+
def __init__(self, resource: str, limit: int) -> None:
|
|
34
|
+
self.resource = resource
|
|
35
|
+
self.limit = limit
|
|
36
|
+
super().__init__(
|
|
37
|
+
f"origin learning {resource} limit reached ({limit}); "
|
|
38
|
+
"save the current candidates and start a new learner or raise "
|
|
39
|
+
"the explicit limit"
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _opcode_text(opcode: int) -> str:
|
|
44
|
+
return f"0x{opcode:04X}"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _utc_text(timestamp: Optional[float] = None) -> str:
|
|
48
|
+
value = (
|
|
49
|
+
dt.datetime.fromtimestamp(timestamp, tz=dt.timezone.utc)
|
|
50
|
+
if timestamp is not None
|
|
51
|
+
else dt.datetime.now(tz=dt.timezone.utc)
|
|
52
|
+
)
|
|
53
|
+
return value.isoformat(timespec="seconds").replace("+00:00", "Z")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _parse_opcode(value: object, location: str) -> int:
|
|
57
|
+
if isinstance(value, bool):
|
|
58
|
+
raise ProfileError(f"{location} must be a uint16")
|
|
59
|
+
if isinstance(value, int):
|
|
60
|
+
opcode = value
|
|
61
|
+
elif isinstance(value, str):
|
|
62
|
+
try:
|
|
63
|
+
opcode = int(value, 16 if value.lower().startswith("0x") else 10)
|
|
64
|
+
except ValueError as exc:
|
|
65
|
+
raise ProfileError(f"{location} must be a uint16") from exc
|
|
66
|
+
else:
|
|
67
|
+
raise ProfileError(f"{location} must be a uint16")
|
|
68
|
+
if not 0 <= opcode <= 0xFFFF:
|
|
69
|
+
raise ProfileError(f"{location} must be a uint16")
|
|
70
|
+
return opcode
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _find_all(message: bytes, token: bytes) -> tuple[int, ...]:
|
|
74
|
+
positions: list[int] = []
|
|
75
|
+
search_at = 5 # Never correlate against length/flag/opcode header bytes.
|
|
76
|
+
while True:
|
|
77
|
+
offset = message.find(token, search_at)
|
|
78
|
+
if offset < 0:
|
|
79
|
+
break
|
|
80
|
+
positions.append(offset)
|
|
81
|
+
search_at = offset + 1
|
|
82
|
+
return tuple(positions)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _shared_token(
|
|
86
|
+
delta_message: bytes,
|
|
87
|
+
first_message: bytes,
|
|
88
|
+
second_message: bytes,
|
|
89
|
+
delta_prefix_end: int,
|
|
90
|
+
) -> Optional[tuple[bytes, tuple[int, int, int]]]:
|
|
91
|
+
"""Return the strongest informative token shared by three frames.
|
|
92
|
+
|
|
93
|
+
``delta_prefix_end`` is the decoded first-record boundary. Restricting
|
|
94
|
+
candidates to that prefix prevents item instances, storage instances, and
|
|
95
|
+
record timestamps from masquerading as operation tokens.
|
|
96
|
+
"""
|
|
97
|
+
candidates: list[tuple[int, int, bytes, tuple[int, int, int]]] = []
|
|
98
|
+
prefix_end = min(delta_prefix_end, len(delta_message))
|
|
99
|
+
for delta_offset in range(5, prefix_end - TOKEN_WIDTH + 1):
|
|
100
|
+
token = delta_message[delta_offset : delta_offset + TOKEN_WIDTH]
|
|
101
|
+
unique_bytes = len(set(token))
|
|
102
|
+
if unique_bytes < MIN_TOKEN_UNIQUE_BYTES:
|
|
103
|
+
continue
|
|
104
|
+
first_positions = _find_all(first_message, token)
|
|
105
|
+
if not first_positions:
|
|
106
|
+
continue
|
|
107
|
+
second_positions = _find_all(second_message, token)
|
|
108
|
+
if not second_positions:
|
|
109
|
+
continue
|
|
110
|
+
offsets = (delta_offset, first_positions[0], second_positions[0])
|
|
111
|
+
# Prefer higher-entropy tokens, then the earliest delta-prefix token.
|
|
112
|
+
candidates.append((unique_bytes, -delta_offset, token, offsets))
|
|
113
|
+
if not candidates:
|
|
114
|
+
return None
|
|
115
|
+
_, _, token, offsets = max(candidates)
|
|
116
|
+
return token, offsets
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@dataclass(frozen=True)
|
|
120
|
+
class CompanionObservation:
|
|
121
|
+
"""One structurally correlated storage-delta companion chain."""
|
|
122
|
+
|
|
123
|
+
timestamp: float
|
|
124
|
+
flow: FlowKey
|
|
125
|
+
stream_sequence: Optional[int]
|
|
126
|
+
delta_opcode: int
|
|
127
|
+
delta_length: int
|
|
128
|
+
companion_opcodes: tuple[int, int]
|
|
129
|
+
companion_lengths: tuple[int, int]
|
|
130
|
+
token_digest: str
|
|
131
|
+
token_offsets: tuple[int, int, int]
|
|
132
|
+
|
|
133
|
+
@property
|
|
134
|
+
def family_key(self) -> tuple[int, int, int, int, int]:
|
|
135
|
+
return (
|
|
136
|
+
self.delta_opcode,
|
|
137
|
+
self.companion_opcodes[0],
|
|
138
|
+
self.companion_opcodes[1],
|
|
139
|
+
self.companion_lengths[0],
|
|
140
|
+
self.companion_lengths[1],
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
@property
|
|
144
|
+
def observation_id(self) -> str:
|
|
145
|
+
identity = (
|
|
146
|
+
f"{self.flow}|{self.stream_sequence}|{self.timestamp:.9f}|"
|
|
147
|
+
f"{self.family_key}|{self.token_digest}"
|
|
148
|
+
).encode("utf-8")
|
|
149
|
+
return hashlib.blake2b(identity, digest_size=16).hexdigest()
|
|
150
|
+
|
|
151
|
+
def to_dict(self) -> dict[str, Any]:
|
|
152
|
+
return {
|
|
153
|
+
"detection": COMPANION_DETECTION,
|
|
154
|
+
"delta_opcode": _opcode_text(self.delta_opcode),
|
|
155
|
+
"delta_length": self.delta_length,
|
|
156
|
+
"companion_opcodes": [
|
|
157
|
+
_opcode_text(opcode) for opcode in self.companion_opcodes
|
|
158
|
+
],
|
|
159
|
+
"companion_lengths": list(self.companion_lengths),
|
|
160
|
+
"shared_token_digest": self.token_digest,
|
|
161
|
+
"token_offsets": {
|
|
162
|
+
"delta": self.token_offsets[0],
|
|
163
|
+
"first_companion": self.token_offsets[1],
|
|
164
|
+
"second_companion": self.token_offsets[2],
|
|
165
|
+
},
|
|
166
|
+
"observed_at": _utc_text(self.timestamp),
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def discover_companion_observation(
|
|
171
|
+
*,
|
|
172
|
+
delta_message: bytes,
|
|
173
|
+
first_message: bytes,
|
|
174
|
+
second_message: bytes,
|
|
175
|
+
timestamp: float,
|
|
176
|
+
flow: FlowKey,
|
|
177
|
+
stream_sequence: Optional[int],
|
|
178
|
+
delta_prefix_end: int,
|
|
179
|
+
) -> Optional[CompanionObservation]:
|
|
180
|
+
"""Recognize a three-frame worker chain without trusting opcode values."""
|
|
181
|
+
for label, message in (
|
|
182
|
+
("delta", delta_message),
|
|
183
|
+
("first companion", first_message),
|
|
184
|
+
("second companion", second_message),
|
|
185
|
+
):
|
|
186
|
+
if not 5 <= len(message) <= MAX_TARGET_MESSAGE_LENGTH:
|
|
187
|
+
return None
|
|
188
|
+
if int.from_bytes(message[0:2], "little") != len(message):
|
|
189
|
+
return None
|
|
190
|
+
if (
|
|
191
|
+
isinstance(delta_prefix_end, bool)
|
|
192
|
+
or not isinstance(delta_prefix_end, int)
|
|
193
|
+
or not 5 + TOKEN_WIDTH <= delta_prefix_end <= len(delta_message)
|
|
194
|
+
):
|
|
195
|
+
return None
|
|
196
|
+
delta_opcode = int.from_bytes(delta_message[3:5], "little")
|
|
197
|
+
first_opcode = int.from_bytes(first_message[3:5], "little")
|
|
198
|
+
second_opcode = int.from_bytes(second_message[3:5], "little")
|
|
199
|
+
# A second storage delta is a boundary, not a worker companion. The two
|
|
200
|
+
# observed worker companion roles are also distinct message families;
|
|
201
|
+
# rejecting a repeated ambient opcode removes a known false correlation.
|
|
202
|
+
if (
|
|
203
|
+
first_opcode == delta_opcode
|
|
204
|
+
or second_opcode == delta_opcode
|
|
205
|
+
or first_opcode == second_opcode
|
|
206
|
+
):
|
|
207
|
+
return None
|
|
208
|
+
shared = _shared_token(
|
|
209
|
+
delta_message,
|
|
210
|
+
first_message,
|
|
211
|
+
second_message,
|
|
212
|
+
delta_prefix_end,
|
|
213
|
+
)
|
|
214
|
+
if shared is None:
|
|
215
|
+
return None
|
|
216
|
+
token, offsets = shared
|
|
217
|
+
return CompanionObservation(
|
|
218
|
+
timestamp=timestamp,
|
|
219
|
+
flow=flow,
|
|
220
|
+
stream_sequence=stream_sequence,
|
|
221
|
+
delta_opcode=delta_opcode,
|
|
222
|
+
delta_length=len(delta_message),
|
|
223
|
+
companion_opcodes=(
|
|
224
|
+
first_opcode,
|
|
225
|
+
second_opcode,
|
|
226
|
+
),
|
|
227
|
+
companion_lengths=(len(first_message), len(second_message)),
|
|
228
|
+
token_digest=hashlib.blake2b(token, digest_size=8).hexdigest(),
|
|
229
|
+
token_offsets=offsets,
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
@dataclass(frozen=True)
|
|
234
|
+
class OriginCompanionCandidate:
|
|
235
|
+
"""Aggregated observations for one delta/companion opcode family."""
|
|
236
|
+
|
|
237
|
+
delta_opcode: int
|
|
238
|
+
companion_opcodes: tuple[int, int]
|
|
239
|
+
companion_lengths: tuple[int, int]
|
|
240
|
+
observations: int
|
|
241
|
+
distinct_tokens: int
|
|
242
|
+
delta_lengths: tuple[int, ...]
|
|
243
|
+
delta_token_offsets: tuple[int, ...]
|
|
244
|
+
first_token_offsets: tuple[int, ...]
|
|
245
|
+
second_token_offsets: tuple[int, ...]
|
|
246
|
+
first_seen: str
|
|
247
|
+
last_seen: str
|
|
248
|
+
observation_ids: tuple[str, ...] = field(repr=False)
|
|
249
|
+
token_digests: tuple[str, ...] = field(repr=False)
|
|
250
|
+
|
|
251
|
+
@property
|
|
252
|
+
def family_key(self) -> tuple[int, int, int, int, int]:
|
|
253
|
+
return (
|
|
254
|
+
self.delta_opcode,
|
|
255
|
+
self.companion_opcodes[0],
|
|
256
|
+
self.companion_opcodes[1],
|
|
257
|
+
self.companion_lengths[0],
|
|
258
|
+
self.companion_lengths[1],
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
def confirmed(self, min_observations: int) -> bool:
|
|
262
|
+
return self.observations >= min_observations
|
|
263
|
+
|
|
264
|
+
def to_dict(self, min_observations: int) -> dict[str, Any]:
|
|
265
|
+
return {
|
|
266
|
+
"status": (
|
|
267
|
+
"confirmed" if self.confirmed(min_observations) else "candidate"
|
|
268
|
+
),
|
|
269
|
+
"detection": COMPANION_DETECTION,
|
|
270
|
+
"delta_opcode": _opcode_text(self.delta_opcode),
|
|
271
|
+
"companion_opcodes": [
|
|
272
|
+
_opcode_text(opcode) for opcode in self.companion_opcodes
|
|
273
|
+
],
|
|
274
|
+
"companion_lengths": list(self.companion_lengths),
|
|
275
|
+
"observations": self.observations,
|
|
276
|
+
"distinct_tokens": self.distinct_tokens,
|
|
277
|
+
"delta_lengths": list(self.delta_lengths),
|
|
278
|
+
"token_offsets": {
|
|
279
|
+
"delta": list(self.delta_token_offsets),
|
|
280
|
+
"first_companion": list(self.first_token_offsets),
|
|
281
|
+
"second_companion": list(self.second_token_offsets),
|
|
282
|
+
},
|
|
283
|
+
"first_seen": self.first_seen,
|
|
284
|
+
"last_seen": self.last_seen,
|
|
285
|
+
# Opaque IDs prevent the same pcap from inflating the count on rerun.
|
|
286
|
+
"observation_ids": list(self.observation_ids),
|
|
287
|
+
"token_digests": list(self.token_digests),
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
@dataclass
|
|
292
|
+
class _CandidateState:
|
|
293
|
+
delta_opcode: int
|
|
294
|
+
companion_opcodes: tuple[int, int]
|
|
295
|
+
companion_lengths: tuple[int, int]
|
|
296
|
+
observation_ids: set[str] = field(default_factory=set)
|
|
297
|
+
token_digests: set[str] = field(default_factory=set)
|
|
298
|
+
delta_lengths: set[int] = field(default_factory=set)
|
|
299
|
+
delta_offsets: set[int] = field(default_factory=set)
|
|
300
|
+
first_offsets: set[int] = field(default_factory=set)
|
|
301
|
+
second_offsets: set[int] = field(default_factory=set)
|
|
302
|
+
first_timestamp: Optional[float] = None
|
|
303
|
+
last_timestamp: Optional[float] = None
|
|
304
|
+
first_seen_text: Optional[str] = None
|
|
305
|
+
last_seen_text: Optional[str] = None
|
|
306
|
+
|
|
307
|
+
def add(self, observation: CompanionObservation) -> bool:
|
|
308
|
+
observation_id = observation.observation_id
|
|
309
|
+
if observation_id in self.observation_ids:
|
|
310
|
+
return False
|
|
311
|
+
self.observation_ids.add(observation_id)
|
|
312
|
+
self.token_digests.add(observation.token_digest)
|
|
313
|
+
self.delta_lengths.add(observation.delta_length)
|
|
314
|
+
self.delta_offsets.add(observation.token_offsets[0])
|
|
315
|
+
self.first_offsets.add(observation.token_offsets[1])
|
|
316
|
+
self.second_offsets.add(observation.token_offsets[2])
|
|
317
|
+
if self.first_timestamp is None or observation.timestamp < self.first_timestamp:
|
|
318
|
+
self.first_timestamp = observation.timestamp
|
|
319
|
+
if self.last_timestamp is None or observation.timestamp > self.last_timestamp:
|
|
320
|
+
self.last_timestamp = observation.timestamp
|
|
321
|
+
return True
|
|
322
|
+
|
|
323
|
+
def snapshot(self) -> OriginCompanionCandidate:
|
|
324
|
+
first_values = [
|
|
325
|
+
value
|
|
326
|
+
for value in (
|
|
327
|
+
self.first_seen_text,
|
|
328
|
+
(
|
|
329
|
+
_utc_text(self.first_timestamp)
|
|
330
|
+
if self.first_timestamp is not None
|
|
331
|
+
else None
|
|
332
|
+
),
|
|
333
|
+
)
|
|
334
|
+
if value is not None
|
|
335
|
+
]
|
|
336
|
+
first_seen = min(first_values) if first_values else _utc_text()
|
|
337
|
+
last_values = [
|
|
338
|
+
value
|
|
339
|
+
for value in (
|
|
340
|
+
self.last_seen_text,
|
|
341
|
+
(
|
|
342
|
+
_utc_text(self.last_timestamp)
|
|
343
|
+
if self.last_timestamp is not None
|
|
344
|
+
else None
|
|
345
|
+
),
|
|
346
|
+
)
|
|
347
|
+
if value is not None
|
|
348
|
+
]
|
|
349
|
+
last_seen = max(last_values) if last_values else first_seen
|
|
350
|
+
return OriginCompanionCandidate(
|
|
351
|
+
delta_opcode=self.delta_opcode,
|
|
352
|
+
companion_opcodes=self.companion_opcodes,
|
|
353
|
+
companion_lengths=self.companion_lengths,
|
|
354
|
+
observations=len(self.observation_ids),
|
|
355
|
+
distinct_tokens=len(self.token_digests),
|
|
356
|
+
delta_lengths=tuple(sorted(self.delta_lengths)),
|
|
357
|
+
delta_token_offsets=tuple(sorted(self.delta_offsets)),
|
|
358
|
+
first_token_offsets=tuple(sorted(self.first_offsets)),
|
|
359
|
+
second_token_offsets=tuple(sorted(self.second_offsets)),
|
|
360
|
+
first_seen=first_seen,
|
|
361
|
+
last_seen=last_seen,
|
|
362
|
+
observation_ids=tuple(sorted(self.observation_ids)),
|
|
363
|
+
token_digests=tuple(sorted(self.token_digests)),
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
class OriginLearner:
|
|
368
|
+
"""Thread-safe in-memory aggregator for structurally discovered families."""
|
|
369
|
+
|
|
370
|
+
def __init__(
|
|
371
|
+
self,
|
|
372
|
+
min_observations: int = 2,
|
|
373
|
+
*,
|
|
374
|
+
max_candidates: int = DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES,
|
|
375
|
+
max_observations: int = DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS,
|
|
376
|
+
) -> None:
|
|
377
|
+
if (
|
|
378
|
+
isinstance(min_observations, bool)
|
|
379
|
+
or not isinstance(min_observations, int)
|
|
380
|
+
or min_observations <= 0
|
|
381
|
+
):
|
|
382
|
+
raise ValueError("min_observations must be a positive integer")
|
|
383
|
+
for name, value in (
|
|
384
|
+
("max_candidates", max_candidates),
|
|
385
|
+
("max_observations", max_observations),
|
|
386
|
+
):
|
|
387
|
+
if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
|
|
388
|
+
raise ValueError(f"{name} must be a positive integer")
|
|
389
|
+
self.min_observations = min_observations
|
|
390
|
+
self.max_candidates = max_candidates
|
|
391
|
+
self.max_observations = max_observations
|
|
392
|
+
self._states: dict[tuple[int, int, int, int, int], _CandidateState] = {}
|
|
393
|
+
self._observation_count = 0
|
|
394
|
+
self._lock = Lock()
|
|
395
|
+
|
|
396
|
+
@classmethod
|
|
397
|
+
def load(
|
|
398
|
+
cls,
|
|
399
|
+
path: str | Path,
|
|
400
|
+
*,
|
|
401
|
+
min_observations: Optional[int] = None,
|
|
402
|
+
max_candidates: Optional[int] = None,
|
|
403
|
+
max_observations: Optional[int] = None,
|
|
404
|
+
) -> "OriginLearner":
|
|
405
|
+
candidate_path = Path(path)
|
|
406
|
+
if not candidate_path.exists():
|
|
407
|
+
return cls(
|
|
408
|
+
2 if min_observations is None else min_observations,
|
|
409
|
+
max_candidates=(
|
|
410
|
+
max_candidates
|
|
411
|
+
if max_candidates is not None
|
|
412
|
+
else DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES
|
|
413
|
+
),
|
|
414
|
+
max_observations=(
|
|
415
|
+
max_observations
|
|
416
|
+
if max_observations is not None
|
|
417
|
+
else DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS
|
|
418
|
+
),
|
|
419
|
+
)
|
|
420
|
+
try:
|
|
421
|
+
data = json.loads(candidate_path.read_text(encoding="utf-8-sig"))
|
|
422
|
+
except (json.JSONDecodeError, UnicodeError) as exc:
|
|
423
|
+
raise ProfileError(
|
|
424
|
+
f"Could not parse origin candidates JSON {candidate_path}: {exc}"
|
|
425
|
+
) from exc
|
|
426
|
+
if not isinstance(data, dict):
|
|
427
|
+
raise ProfileError(f"Origin candidates {candidate_path} must be an object")
|
|
428
|
+
version = data.get("version")
|
|
429
|
+
if version != CANDIDATE_FORMAT_VERSION or isinstance(version, bool):
|
|
430
|
+
raise ProfileError(
|
|
431
|
+
f"version in {candidate_path} must be {CANDIDATE_FORMAT_VERSION}"
|
|
432
|
+
)
|
|
433
|
+
stored_min = data.get("min_observations", 2)
|
|
434
|
+
if (
|
|
435
|
+
isinstance(stored_min, bool)
|
|
436
|
+
or not isinstance(stored_min, int)
|
|
437
|
+
or stored_min <= 0
|
|
438
|
+
):
|
|
439
|
+
raise ProfileError(
|
|
440
|
+
f"min_observations in {candidate_path} must be a positive integer"
|
|
441
|
+
)
|
|
442
|
+
threshold = min_observations if min_observations is not None else stored_min
|
|
443
|
+
stored_max_candidates = data.get(
|
|
444
|
+
"max_candidates", DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES
|
|
445
|
+
)
|
|
446
|
+
stored_max_observations = data.get(
|
|
447
|
+
"max_observations", DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS
|
|
448
|
+
)
|
|
449
|
+
try:
|
|
450
|
+
learner = cls(
|
|
451
|
+
threshold,
|
|
452
|
+
max_candidates=(
|
|
453
|
+
max_candidates
|
|
454
|
+
if max_candidates is not None
|
|
455
|
+
else stored_max_candidates
|
|
456
|
+
),
|
|
457
|
+
max_observations=(
|
|
458
|
+
max_observations
|
|
459
|
+
if max_observations is not None
|
|
460
|
+
else stored_max_observations
|
|
461
|
+
),
|
|
462
|
+
)
|
|
463
|
+
except ValueError as exc:
|
|
464
|
+
if (
|
|
465
|
+
min_observations is not None
|
|
466
|
+
or max_candidates is not None
|
|
467
|
+
or max_observations is not None
|
|
468
|
+
):
|
|
469
|
+
raise
|
|
470
|
+
raise ProfileError(
|
|
471
|
+
f"invalid origin-learning limits in {candidate_path}: {exc}"
|
|
472
|
+
) from exc
|
|
473
|
+
entries = data.get("candidates", [])
|
|
474
|
+
if not isinstance(entries, list):
|
|
475
|
+
raise ProfileError(f"candidates in {candidate_path} must be a list")
|
|
476
|
+
for index, entry in enumerate(entries):
|
|
477
|
+
learner._load_entry(entry, f"candidates[{index}] in {candidate_path}")
|
|
478
|
+
return learner
|
|
479
|
+
|
|
480
|
+
def _load_entry(self, entry: object, location: str) -> None:
|
|
481
|
+
if not isinstance(entry, dict):
|
|
482
|
+
raise ProfileError(f"{location} must be an object")
|
|
483
|
+
if entry.get("detection") != COMPANION_DETECTION:
|
|
484
|
+
raise ProfileError(f"{location}.detection must be {COMPANION_DETECTION!r}")
|
|
485
|
+
delta_opcode = _parse_opcode(
|
|
486
|
+
entry.get("delta_opcode"), f"{location}.delta_opcode"
|
|
487
|
+
)
|
|
488
|
+
raw_opcodes = entry.get("companion_opcodes")
|
|
489
|
+
raw_lengths = entry.get("companion_lengths")
|
|
490
|
+
if not isinstance(raw_opcodes, list) or len(raw_opcodes) != 2:
|
|
491
|
+
raise ProfileError(f"{location}.companion_opcodes must contain two values")
|
|
492
|
+
if not isinstance(raw_lengths, list) or len(raw_lengths) != 2:
|
|
493
|
+
raise ProfileError(f"{location}.companion_lengths must contain two values")
|
|
494
|
+
opcodes = (
|
|
495
|
+
_parse_opcode(raw_opcodes[0], f"{location}.companion_opcodes"),
|
|
496
|
+
_parse_opcode(raw_opcodes[1], f"{location}.companion_opcodes"),
|
|
497
|
+
)
|
|
498
|
+
lengths: list[int] = []
|
|
499
|
+
for value in raw_lengths:
|
|
500
|
+
if (
|
|
501
|
+
isinstance(value, bool)
|
|
502
|
+
or not isinstance(value, int)
|
|
503
|
+
or not 5 <= value <= 0xFFFF
|
|
504
|
+
):
|
|
505
|
+
raise ProfileError(f"{location}.companion_lengths must be 5..65535")
|
|
506
|
+
lengths.append(value)
|
|
507
|
+
key = (delta_opcode, opcodes[0], opcodes[1], lengths[0], lengths[1])
|
|
508
|
+
state = _CandidateState(delta_opcode, opcodes, (lengths[0], lengths[1]))
|
|
509
|
+
observation_ids = entry.get("observation_ids", [])
|
|
510
|
+
token_digests = entry.get("token_digests", [])
|
|
511
|
+
delta_lengths = entry.get("delta_lengths", [])
|
|
512
|
+
if not isinstance(observation_ids, list) or any(
|
|
513
|
+
not isinstance(value, str) for value in observation_ids
|
|
514
|
+
):
|
|
515
|
+
raise ProfileError(f"{location}.observation_ids must be a string list")
|
|
516
|
+
if not isinstance(token_digests, list) or any(
|
|
517
|
+
not isinstance(value, str) for value in token_digests
|
|
518
|
+
):
|
|
519
|
+
raise ProfileError(f"{location}.token_digests must be a string list")
|
|
520
|
+
if not isinstance(delta_lengths, list) or any(
|
|
521
|
+
isinstance(value, bool)
|
|
522
|
+
or not isinstance(value, int)
|
|
523
|
+
or not 5 <= value <= 0xFFFF
|
|
524
|
+
for value in delta_lengths
|
|
525
|
+
):
|
|
526
|
+
raise ProfileError(f"{location}.delta_lengths must contain 5..65535")
|
|
527
|
+
unique_observation_ids = set(observation_ids)
|
|
528
|
+
for name, values in (
|
|
529
|
+
("token_digests", token_digests),
|
|
530
|
+
("delta_lengths", delta_lengths),
|
|
531
|
+
):
|
|
532
|
+
if len(set(values)) > len(unique_observation_ids):
|
|
533
|
+
raise ProfileError(
|
|
534
|
+
f"{location}.{name} cannot contain more distinct evidence "
|
|
535
|
+
"than observation_ids"
|
|
536
|
+
)
|
|
537
|
+
state.observation_ids.update(observation_ids)
|
|
538
|
+
state.token_digests.update(token_digests)
|
|
539
|
+
state.delta_lengths.update(delta_lengths)
|
|
540
|
+
offsets = entry.get("token_offsets", {})
|
|
541
|
+
if not isinstance(offsets, dict):
|
|
542
|
+
raise ProfileError(f"{location}.token_offsets must be an object")
|
|
543
|
+
for name, target in (
|
|
544
|
+
("delta", state.delta_offsets),
|
|
545
|
+
("first_companion", state.first_offsets),
|
|
546
|
+
("second_companion", state.second_offsets),
|
|
547
|
+
):
|
|
548
|
+
values = offsets.get(name, [])
|
|
549
|
+
if not isinstance(values, list) or any(
|
|
550
|
+
isinstance(value, bool) or not isinstance(value, int) or value < 0
|
|
551
|
+
for value in values
|
|
552
|
+
):
|
|
553
|
+
raise ProfileError(f"{location}.token_offsets.{name} is invalid")
|
|
554
|
+
if len(set(values)) > len(unique_observation_ids):
|
|
555
|
+
raise ProfileError(
|
|
556
|
+
f"{location}.token_offsets.{name} cannot contain more "
|
|
557
|
+
"distinct evidence than observation_ids"
|
|
558
|
+
)
|
|
559
|
+
target.update(values)
|
|
560
|
+
first_seen = entry.get("first_seen")
|
|
561
|
+
last_seen = entry.get("last_seen")
|
|
562
|
+
if first_seen is not None and not isinstance(first_seen, str):
|
|
563
|
+
raise ProfileError(f"{location}.first_seen must be a string")
|
|
564
|
+
if last_seen is not None and not isinstance(last_seen, str):
|
|
565
|
+
raise ProfileError(f"{location}.last_seen must be a string")
|
|
566
|
+
state.first_seen_text = first_seen
|
|
567
|
+
state.last_seen_text = last_seen
|
|
568
|
+
if key in self._states:
|
|
569
|
+
raise ProfileError(f"{location} duplicates an earlier candidate family")
|
|
570
|
+
if len(self._states) >= self.max_candidates:
|
|
571
|
+
raise OriginLearningLimitError("candidate-family", self.max_candidates)
|
|
572
|
+
retained_observations = len(state.observation_ids)
|
|
573
|
+
if self._observation_count + retained_observations > self.max_observations:
|
|
574
|
+
raise OriginLearningLimitError("observation", self.max_observations)
|
|
575
|
+
self._states[key] = state
|
|
576
|
+
self._observation_count += retained_observations
|
|
577
|
+
|
|
578
|
+
def observe(self, observation: CompanionObservation) -> OriginCompanionCandidate:
|
|
579
|
+
if not isinstance(observation, CompanionObservation):
|
|
580
|
+
raise TypeError("OriginLearner.observe expects CompanionObservation")
|
|
581
|
+
with self._lock:
|
|
582
|
+
state = self._states.get(observation.family_key)
|
|
583
|
+
if (
|
|
584
|
+
state is not None
|
|
585
|
+
and observation.observation_id in state.observation_ids
|
|
586
|
+
):
|
|
587
|
+
return state.snapshot()
|
|
588
|
+
if state is None:
|
|
589
|
+
if len(self._states) >= self.max_candidates:
|
|
590
|
+
raise OriginLearningLimitError(
|
|
591
|
+
"candidate-family", self.max_candidates
|
|
592
|
+
)
|
|
593
|
+
state = _CandidateState(
|
|
594
|
+
observation.delta_opcode,
|
|
595
|
+
observation.companion_opcodes,
|
|
596
|
+
observation.companion_lengths,
|
|
597
|
+
)
|
|
598
|
+
if self._observation_count >= self.max_observations:
|
|
599
|
+
raise OriginLearningLimitError("observation", self.max_observations)
|
|
600
|
+
added = state.add(observation)
|
|
601
|
+
if added:
|
|
602
|
+
self._observation_count += 1
|
|
603
|
+
self._states[observation.family_key] = state
|
|
604
|
+
return state.snapshot()
|
|
605
|
+
|
|
606
|
+
@property
|
|
607
|
+
def candidates(self) -> tuple[OriginCompanionCandidate, ...]:
|
|
608
|
+
with self._lock:
|
|
609
|
+
return tuple(self._states[key].snapshot() for key in sorted(self._states))
|
|
610
|
+
|
|
611
|
+
@property
|
|
612
|
+
def confirmed_candidates(self) -> tuple[OriginCompanionCandidate, ...]:
|
|
613
|
+
return tuple(
|
|
614
|
+
candidate
|
|
615
|
+
for candidate in self.candidates
|
|
616
|
+
if candidate.confirmed(self.min_observations)
|
|
617
|
+
)
|
|
618
|
+
|
|
619
|
+
def summary(self) -> str:
|
|
620
|
+
"""Return a human-readable multi-line candidate report."""
|
|
621
|
+
candidates = self.candidates
|
|
622
|
+
confirmed_count = sum(
|
|
623
|
+
candidate.confirmed(self.min_observations) for candidate in candidates
|
|
624
|
+
)
|
|
625
|
+
lines = [
|
|
626
|
+
f"observed {len(candidates)} origin companion family/families; "
|
|
627
|
+
f"{confirmed_count} confirmed for promotion "
|
|
628
|
+
f"(threshold: {self.min_observations} observation(s))"
|
|
629
|
+
]
|
|
630
|
+
if not candidates:
|
|
631
|
+
lines.append("no origin companion families observed")
|
|
632
|
+
return "\n".join(lines)
|
|
633
|
+
|
|
634
|
+
for candidate in candidates:
|
|
635
|
+
status = (
|
|
636
|
+
"confirmed"
|
|
637
|
+
if candidate.confirmed(self.min_observations)
|
|
638
|
+
else "candidate"
|
|
639
|
+
)
|
|
640
|
+
companions = " -> ".join(
|
|
641
|
+
f"0x{opcode:04X}" for opcode in candidate.companion_opcodes
|
|
642
|
+
)
|
|
643
|
+
lengths = ", ".join(str(length) for length in candidate.companion_lengths)
|
|
644
|
+
lines.append(
|
|
645
|
+
f"{status}: delta=0x{candidate.delta_opcode:04X} "
|
|
646
|
+
f"companions={companions} lengths=({lengths}) "
|
|
647
|
+
f"observations={candidate.observations} "
|
|
648
|
+
f"distinct_tokens={candidate.distinct_tokens}"
|
|
649
|
+
)
|
|
650
|
+
return "\n".join(lines)
|
|
651
|
+
|
|
652
|
+
def to_dict(self) -> dict[str, Any]:
|
|
653
|
+
return {
|
|
654
|
+
"version": CANDIDATE_FORMAT_VERSION,
|
|
655
|
+
"generated_at": _utc_text(),
|
|
656
|
+
"min_observations": self.min_observations,
|
|
657
|
+
"max_candidates": self.max_candidates,
|
|
658
|
+
"max_observations": self.max_observations,
|
|
659
|
+
"candidates": [
|
|
660
|
+
candidate.to_dict(self.min_observations)
|
|
661
|
+
for candidate in self.candidates
|
|
662
|
+
],
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
def save(self, path: str | Path) -> Path:
|
|
666
|
+
candidate_path = Path(path)
|
|
667
|
+
_atomic_write_text(
|
|
668
|
+
candidate_path,
|
|
669
|
+
json.dumps(self.to_dict(), indent=2, sort_keys=True) + "\n",
|
|
670
|
+
)
|
|
671
|
+
return candidate_path
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
@dataclass(frozen=True)
|
|
675
|
+
class OriginPromotion:
|
|
676
|
+
path: Path
|
|
677
|
+
added: tuple[OriginCompanionCandidate, ...]
|
|
678
|
+
backup_path: Optional[Path]
|
|
679
|
+
written: bool
|
|
680
|
+
|
|
681
|
+
|
|
682
|
+
def promote_origin_candidates(
|
|
683
|
+
candidates_path: str | Path,
|
|
684
|
+
profile_path: str | Path,
|
|
685
|
+
*,
|
|
686
|
+
min_observations: Optional[int] = None,
|
|
687
|
+
backup: bool = True,
|
|
688
|
+
) -> OriginPromotion:
|
|
689
|
+
"""Explicitly merge confirmed learned families into an opcode profile."""
|
|
690
|
+
learner = OriginLearner.load(
|
|
691
|
+
candidates_path,
|
|
692
|
+
min_observations=min_observations,
|
|
693
|
+
)
|
|
694
|
+
confirmed = learner.confirmed_candidates
|
|
695
|
+
destination = Path(profile_path)
|
|
696
|
+
load_opcode_profile(destination) # Strictly validate before preserving it.
|
|
697
|
+
data = json.loads(destination.read_text(encoding="utf-8-sig"))
|
|
698
|
+
families = data.setdefault("origin_companion_families", [])
|
|
699
|
+
if not isinstance(families, list):
|
|
700
|
+
raise ProfileError(f"origin_companion_families in {destination} must be a list")
|
|
701
|
+
|
|
702
|
+
existing: set[tuple[int, int, int, int, int]] = set()
|
|
703
|
+
for index, family in enumerate(families):
|
|
704
|
+
if not isinstance(family, dict):
|
|
705
|
+
raise ProfileError(
|
|
706
|
+
f"origin_companion_families[{index}] in {destination} must be an object"
|
|
707
|
+
)
|
|
708
|
+
raw_opcodes = family.get("companion_opcodes")
|
|
709
|
+
raw_lengths = family.get("companion_lengths")
|
|
710
|
+
if not isinstance(raw_opcodes, list) or len(raw_opcodes) != 2:
|
|
711
|
+
raise ProfileError("existing companion family has invalid opcodes")
|
|
712
|
+
if not isinstance(raw_lengths, list) or len(raw_lengths) != 2:
|
|
713
|
+
raise ProfileError("existing companion family has invalid lengths")
|
|
714
|
+
existing.add(
|
|
715
|
+
(
|
|
716
|
+
_parse_opcode(family.get("delta_opcode"), "delta_opcode"),
|
|
717
|
+
_parse_opcode(raw_opcodes[0], "companion opcode"),
|
|
718
|
+
_parse_opcode(raw_opcodes[1], "companion opcode"),
|
|
719
|
+
int(raw_lengths[0]),
|
|
720
|
+
int(raw_lengths[1]),
|
|
721
|
+
)
|
|
722
|
+
)
|
|
723
|
+
|
|
724
|
+
added = tuple(
|
|
725
|
+
candidate for candidate in confirmed if candidate.family_key not in existing
|
|
726
|
+
)
|
|
727
|
+
if not added:
|
|
728
|
+
return OriginPromotion(destination, (), None, False)
|
|
729
|
+
promoted_at = _utc_text()
|
|
730
|
+
for candidate in added:
|
|
731
|
+
families.append(
|
|
732
|
+
{
|
|
733
|
+
"detection": COMPANION_DETECTION,
|
|
734
|
+
"delta_opcode": _opcode_text(candidate.delta_opcode),
|
|
735
|
+
"companion_opcodes": [
|
|
736
|
+
_opcode_text(opcode) for opcode in candidate.companion_opcodes
|
|
737
|
+
],
|
|
738
|
+
"companion_lengths": list(candidate.companion_lengths),
|
|
739
|
+
"observations": candidate.observations,
|
|
740
|
+
"promoted_at": promoted_at,
|
|
741
|
+
}
|
|
742
|
+
)
|
|
743
|
+
data["updated_at"] = promoted_at
|
|
744
|
+
backup_path = None
|
|
745
|
+
if backup:
|
|
746
|
+
backup_path = _unique_backup_path(destination)
|
|
747
|
+
shutil.copy2(destination, backup_path)
|
|
748
|
+
_atomic_write_text(destination, json.dumps(data, indent=2, sort_keys=True) + "\n")
|
|
749
|
+
return OriginPromotion(destination, added, backup_path, True)
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
def _unique_backup_path(path: Path) -> Path:
|
|
753
|
+
backup_dir = path.parent / "opcodes_backups"
|
|
754
|
+
backup_dir.mkdir(parents=True, exist_ok=True)
|
|
755
|
+
stamp = dt.datetime.now(tz=dt.timezone.utc).strftime("%Y%m%d%H%M%S%f")
|
|
756
|
+
return backup_dir / f"{path.name}.bak.{stamp}"
|
|
757
|
+
|
|
758
|
+
|
|
759
|
+
def _atomic_write_text(path: Path, value: str) -> None:
|
|
760
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
761
|
+
temporary: Optional[Path] = None
|
|
762
|
+
try:
|
|
763
|
+
with tempfile.NamedTemporaryFile(
|
|
764
|
+
mode="w",
|
|
765
|
+
encoding="utf-8",
|
|
766
|
+
newline="\n",
|
|
767
|
+
dir=path.parent,
|
|
768
|
+
prefix=f".{path.name}.",
|
|
769
|
+
suffix=".tmp",
|
|
770
|
+
delete=False,
|
|
771
|
+
) as handle:
|
|
772
|
+
temporary = Path(handle.name)
|
|
773
|
+
handle.write(value)
|
|
774
|
+
handle.flush()
|
|
775
|
+
os.fsync(handle.fileno())
|
|
776
|
+
os.replace(temporary, path)
|
|
777
|
+
finally:
|
|
778
|
+
if temporary is not None and temporary.exists():
|
|
779
|
+
temporary.unlink()
|