bdo-toolkit 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. bdo_toolkit/__init__.py +87 -0
  2. bdo_toolkit/_async_sessions.py +651 -0
  3. bdo_toolkit/_capture_backend.py +194 -0
  4. bdo_toolkit/_capture_options.py +68 -0
  5. bdo_toolkit/_capture_runtime.py +626 -0
  6. bdo_toolkit/_deposit_origin.py +1599 -0
  7. bdo_toolkit/_engine.py +327 -0
  8. bdo_toolkit/_framing.py +904 -0
  9. bdo_toolkit/_profile_runtime.py +157 -0
  10. bdo_toolkit/_protocol.py +386 -0
  11. bdo_toolkit/_reassembly.py +654 -0
  12. bdo_toolkit/_specs.py +285 -0
  13. bdo_toolkit/_storage_destination_validation.py +167 -0
  14. bdo_toolkit/_storage_hydration.py +241 -0
  15. bdo_toolkit/_version.py +3 -0
  16. bdo_toolkit/calibration.py +3223 -0
  17. bdo_toolkit/capture.py +1713 -0
  18. bdo_toolkit/character_state.py +3506 -0
  19. bdo_toolkit/cli.py +948 -0
  20. bdo_toolkit/diagnostics.py +51 -0
  21. bdo_toolkit/events.py +214 -0
  22. bdo_toolkit/filters.py +105 -0
  23. bdo_toolkit/item_state.py +48 -0
  24. bdo_toolkit/origin_learning.py +779 -0
  25. bdo_toolkit/profiles.py +370 -0
  26. bdo_toolkit/py.typed +1 -0
  27. bdo_toolkit/remote_profiles.py +358 -0
  28. bdo_toolkit/solare/__init__.py +50 -0
  29. bdo_toolkit/solare/_constants.py +94 -0
  30. bdo_toolkit/solare/_detail_learning.py +1437 -0
  31. bdo_toolkit/solare/_details.py +796 -0
  32. bdo_toolkit/solare/_discovery.py +1212 -0
  33. bdo_toolkit/solare/_live_tracker.py +472 -0
  34. bdo_toolkit/solare/_replay_capture.py +182 -0
  35. bdo_toolkit/solare/_result.py +441 -0
  36. bdo_toolkit/solare/_scanner.py +203 -0
  37. bdo_toolkit/solare/_validation.py +11 -0
  38. bdo_toolkit/solare/async_session.py +444 -0
  39. bdo_toolkit/solare/models.py +806 -0
  40. bdo_toolkit/solare/replay.py +62 -0
  41. bdo_toolkit/solare/session.py +1051 -0
  42. bdo_toolkit/writers.py +30 -0
  43. bdo_toolkit-1.0.0.dist-info/METADATA +143 -0
  44. bdo_toolkit-1.0.0.dist-info/RECORD +48 -0
  45. bdo_toolkit-1.0.0.dist-info/WHEEL +5 -0
  46. bdo_toolkit-1.0.0.dist-info/entry_points.txt +2 -0
  47. bdo_toolkit-1.0.0.dist-info/licenses/LICENSE +21 -0
  48. bdo_toolkit-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,779 @@
1
+ """Structural worker-companion discovery and explicit candidate persistence."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime as dt
6
+ import hashlib
7
+ import json
8
+ import os
9
+ import shutil
10
+ import tempfile
11
+ from dataclasses import dataclass, field
12
+ from pathlib import Path
13
+ from threading import Lock
14
+ from typing import Any, Iterable, Optional
15
+
16
+ from ._protocol import FlowKey, MAX_TARGET_MESSAGE_LENGTH
17
+ from .profiles import ProfileError, load_opcode_profile
18
+
19
+ TOKEN_WIDTH = 8
20
+ # Five-byte diversity was just permissive enough to treat an initial-load
21
+ # timestamp as a transaction token. A worker token must now contain more than
22
+ # five distinct byte values.
23
+ MIN_TOKEN_UNIQUE_BYTES = 6
24
+ CANDIDATE_FORMAT_VERSION = 1
25
+ COMPANION_DETECTION = "shared-token-chain-v1"
26
+ DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES = 256
27
+ DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS = 10_000
28
+
29
+
30
+ class OriginLearningLimitError(RuntimeError):
31
+ """Raised before an origin learner would exceed its retention budget."""
32
+
33
+ def __init__(self, resource: str, limit: int) -> None:
34
+ self.resource = resource
35
+ self.limit = limit
36
+ super().__init__(
37
+ f"origin learning {resource} limit reached ({limit}); "
38
+ "save the current candidates and start a new learner or raise "
39
+ "the explicit limit"
40
+ )
41
+
42
+
43
+ def _opcode_text(opcode: int) -> str:
44
+ return f"0x{opcode:04X}"
45
+
46
+
47
+ def _utc_text(timestamp: Optional[float] = None) -> str:
48
+ value = (
49
+ dt.datetime.fromtimestamp(timestamp, tz=dt.timezone.utc)
50
+ if timestamp is not None
51
+ else dt.datetime.now(tz=dt.timezone.utc)
52
+ )
53
+ return value.isoformat(timespec="seconds").replace("+00:00", "Z")
54
+
55
+
56
+ def _parse_opcode(value: object, location: str) -> int:
57
+ if isinstance(value, bool):
58
+ raise ProfileError(f"{location} must be a uint16")
59
+ if isinstance(value, int):
60
+ opcode = value
61
+ elif isinstance(value, str):
62
+ try:
63
+ opcode = int(value, 16 if value.lower().startswith("0x") else 10)
64
+ except ValueError as exc:
65
+ raise ProfileError(f"{location} must be a uint16") from exc
66
+ else:
67
+ raise ProfileError(f"{location} must be a uint16")
68
+ if not 0 <= opcode <= 0xFFFF:
69
+ raise ProfileError(f"{location} must be a uint16")
70
+ return opcode
71
+
72
+
73
+ def _find_all(message: bytes, token: bytes) -> tuple[int, ...]:
74
+ positions: list[int] = []
75
+ search_at = 5 # Never correlate against length/flag/opcode header bytes.
76
+ while True:
77
+ offset = message.find(token, search_at)
78
+ if offset < 0:
79
+ break
80
+ positions.append(offset)
81
+ search_at = offset + 1
82
+ return tuple(positions)
83
+
84
+
85
+ def _shared_token(
86
+ delta_message: bytes,
87
+ first_message: bytes,
88
+ second_message: bytes,
89
+ delta_prefix_end: int,
90
+ ) -> Optional[tuple[bytes, tuple[int, int, int]]]:
91
+ """Return the strongest informative token shared by three frames.
92
+
93
+ ``delta_prefix_end`` is the decoded first-record boundary. Restricting
94
+ candidates to that prefix prevents item instances, storage instances, and
95
+ record timestamps from masquerading as operation tokens.
96
+ """
97
+ candidates: list[tuple[int, int, bytes, tuple[int, int, int]]] = []
98
+ prefix_end = min(delta_prefix_end, len(delta_message))
99
+ for delta_offset in range(5, prefix_end - TOKEN_WIDTH + 1):
100
+ token = delta_message[delta_offset : delta_offset + TOKEN_WIDTH]
101
+ unique_bytes = len(set(token))
102
+ if unique_bytes < MIN_TOKEN_UNIQUE_BYTES:
103
+ continue
104
+ first_positions = _find_all(first_message, token)
105
+ if not first_positions:
106
+ continue
107
+ second_positions = _find_all(second_message, token)
108
+ if not second_positions:
109
+ continue
110
+ offsets = (delta_offset, first_positions[0], second_positions[0])
111
+ # Prefer higher-entropy tokens, then the earliest delta-prefix token.
112
+ candidates.append((unique_bytes, -delta_offset, token, offsets))
113
+ if not candidates:
114
+ return None
115
+ _, _, token, offsets = max(candidates)
116
+ return token, offsets
117
+
118
+
119
+ @dataclass(frozen=True)
120
+ class CompanionObservation:
121
+ """One structurally correlated storage-delta companion chain."""
122
+
123
+ timestamp: float
124
+ flow: FlowKey
125
+ stream_sequence: Optional[int]
126
+ delta_opcode: int
127
+ delta_length: int
128
+ companion_opcodes: tuple[int, int]
129
+ companion_lengths: tuple[int, int]
130
+ token_digest: str
131
+ token_offsets: tuple[int, int, int]
132
+
133
+ @property
134
+ def family_key(self) -> tuple[int, int, int, int, int]:
135
+ return (
136
+ self.delta_opcode,
137
+ self.companion_opcodes[0],
138
+ self.companion_opcodes[1],
139
+ self.companion_lengths[0],
140
+ self.companion_lengths[1],
141
+ )
142
+
143
+ @property
144
+ def observation_id(self) -> str:
145
+ identity = (
146
+ f"{self.flow}|{self.stream_sequence}|{self.timestamp:.9f}|"
147
+ f"{self.family_key}|{self.token_digest}"
148
+ ).encode("utf-8")
149
+ return hashlib.blake2b(identity, digest_size=16).hexdigest()
150
+
151
+ def to_dict(self) -> dict[str, Any]:
152
+ return {
153
+ "detection": COMPANION_DETECTION,
154
+ "delta_opcode": _opcode_text(self.delta_opcode),
155
+ "delta_length": self.delta_length,
156
+ "companion_opcodes": [
157
+ _opcode_text(opcode) for opcode in self.companion_opcodes
158
+ ],
159
+ "companion_lengths": list(self.companion_lengths),
160
+ "shared_token_digest": self.token_digest,
161
+ "token_offsets": {
162
+ "delta": self.token_offsets[0],
163
+ "first_companion": self.token_offsets[1],
164
+ "second_companion": self.token_offsets[2],
165
+ },
166
+ "observed_at": _utc_text(self.timestamp),
167
+ }
168
+
169
+
170
+ def discover_companion_observation(
171
+ *,
172
+ delta_message: bytes,
173
+ first_message: bytes,
174
+ second_message: bytes,
175
+ timestamp: float,
176
+ flow: FlowKey,
177
+ stream_sequence: Optional[int],
178
+ delta_prefix_end: int,
179
+ ) -> Optional[CompanionObservation]:
180
+ """Recognize a three-frame worker chain without trusting opcode values."""
181
+ for label, message in (
182
+ ("delta", delta_message),
183
+ ("first companion", first_message),
184
+ ("second companion", second_message),
185
+ ):
186
+ if not 5 <= len(message) <= MAX_TARGET_MESSAGE_LENGTH:
187
+ return None
188
+ if int.from_bytes(message[0:2], "little") != len(message):
189
+ return None
190
+ if (
191
+ isinstance(delta_prefix_end, bool)
192
+ or not isinstance(delta_prefix_end, int)
193
+ or not 5 + TOKEN_WIDTH <= delta_prefix_end <= len(delta_message)
194
+ ):
195
+ return None
196
+ delta_opcode = int.from_bytes(delta_message[3:5], "little")
197
+ first_opcode = int.from_bytes(first_message[3:5], "little")
198
+ second_opcode = int.from_bytes(second_message[3:5], "little")
199
+ # A second storage delta is a boundary, not a worker companion. The two
200
+ # observed worker companion roles are also distinct message families;
201
+ # rejecting a repeated ambient opcode removes a known false correlation.
202
+ if (
203
+ first_opcode == delta_opcode
204
+ or second_opcode == delta_opcode
205
+ or first_opcode == second_opcode
206
+ ):
207
+ return None
208
+ shared = _shared_token(
209
+ delta_message,
210
+ first_message,
211
+ second_message,
212
+ delta_prefix_end,
213
+ )
214
+ if shared is None:
215
+ return None
216
+ token, offsets = shared
217
+ return CompanionObservation(
218
+ timestamp=timestamp,
219
+ flow=flow,
220
+ stream_sequence=stream_sequence,
221
+ delta_opcode=delta_opcode,
222
+ delta_length=len(delta_message),
223
+ companion_opcodes=(
224
+ first_opcode,
225
+ second_opcode,
226
+ ),
227
+ companion_lengths=(len(first_message), len(second_message)),
228
+ token_digest=hashlib.blake2b(token, digest_size=8).hexdigest(),
229
+ token_offsets=offsets,
230
+ )
231
+
232
+
233
+ @dataclass(frozen=True)
234
+ class OriginCompanionCandidate:
235
+ """Aggregated observations for one delta/companion opcode family."""
236
+
237
+ delta_opcode: int
238
+ companion_opcodes: tuple[int, int]
239
+ companion_lengths: tuple[int, int]
240
+ observations: int
241
+ distinct_tokens: int
242
+ delta_lengths: tuple[int, ...]
243
+ delta_token_offsets: tuple[int, ...]
244
+ first_token_offsets: tuple[int, ...]
245
+ second_token_offsets: tuple[int, ...]
246
+ first_seen: str
247
+ last_seen: str
248
+ observation_ids: tuple[str, ...] = field(repr=False)
249
+ token_digests: tuple[str, ...] = field(repr=False)
250
+
251
+ @property
252
+ def family_key(self) -> tuple[int, int, int, int, int]:
253
+ return (
254
+ self.delta_opcode,
255
+ self.companion_opcodes[0],
256
+ self.companion_opcodes[1],
257
+ self.companion_lengths[0],
258
+ self.companion_lengths[1],
259
+ )
260
+
261
+ def confirmed(self, min_observations: int) -> bool:
262
+ return self.observations >= min_observations
263
+
264
+ def to_dict(self, min_observations: int) -> dict[str, Any]:
265
+ return {
266
+ "status": (
267
+ "confirmed" if self.confirmed(min_observations) else "candidate"
268
+ ),
269
+ "detection": COMPANION_DETECTION,
270
+ "delta_opcode": _opcode_text(self.delta_opcode),
271
+ "companion_opcodes": [
272
+ _opcode_text(opcode) for opcode in self.companion_opcodes
273
+ ],
274
+ "companion_lengths": list(self.companion_lengths),
275
+ "observations": self.observations,
276
+ "distinct_tokens": self.distinct_tokens,
277
+ "delta_lengths": list(self.delta_lengths),
278
+ "token_offsets": {
279
+ "delta": list(self.delta_token_offsets),
280
+ "first_companion": list(self.first_token_offsets),
281
+ "second_companion": list(self.second_token_offsets),
282
+ },
283
+ "first_seen": self.first_seen,
284
+ "last_seen": self.last_seen,
285
+ # Opaque IDs prevent the same pcap from inflating the count on rerun.
286
+ "observation_ids": list(self.observation_ids),
287
+ "token_digests": list(self.token_digests),
288
+ }
289
+
290
+
291
+ @dataclass
292
+ class _CandidateState:
293
+ delta_opcode: int
294
+ companion_opcodes: tuple[int, int]
295
+ companion_lengths: tuple[int, int]
296
+ observation_ids: set[str] = field(default_factory=set)
297
+ token_digests: set[str] = field(default_factory=set)
298
+ delta_lengths: set[int] = field(default_factory=set)
299
+ delta_offsets: set[int] = field(default_factory=set)
300
+ first_offsets: set[int] = field(default_factory=set)
301
+ second_offsets: set[int] = field(default_factory=set)
302
+ first_timestamp: Optional[float] = None
303
+ last_timestamp: Optional[float] = None
304
+ first_seen_text: Optional[str] = None
305
+ last_seen_text: Optional[str] = None
306
+
307
+ def add(self, observation: CompanionObservation) -> bool:
308
+ observation_id = observation.observation_id
309
+ if observation_id in self.observation_ids:
310
+ return False
311
+ self.observation_ids.add(observation_id)
312
+ self.token_digests.add(observation.token_digest)
313
+ self.delta_lengths.add(observation.delta_length)
314
+ self.delta_offsets.add(observation.token_offsets[0])
315
+ self.first_offsets.add(observation.token_offsets[1])
316
+ self.second_offsets.add(observation.token_offsets[2])
317
+ if self.first_timestamp is None or observation.timestamp < self.first_timestamp:
318
+ self.first_timestamp = observation.timestamp
319
+ if self.last_timestamp is None or observation.timestamp > self.last_timestamp:
320
+ self.last_timestamp = observation.timestamp
321
+ return True
322
+
323
+ def snapshot(self) -> OriginCompanionCandidate:
324
+ first_values = [
325
+ value
326
+ for value in (
327
+ self.first_seen_text,
328
+ (
329
+ _utc_text(self.first_timestamp)
330
+ if self.first_timestamp is not None
331
+ else None
332
+ ),
333
+ )
334
+ if value is not None
335
+ ]
336
+ first_seen = min(first_values) if first_values else _utc_text()
337
+ last_values = [
338
+ value
339
+ for value in (
340
+ self.last_seen_text,
341
+ (
342
+ _utc_text(self.last_timestamp)
343
+ if self.last_timestamp is not None
344
+ else None
345
+ ),
346
+ )
347
+ if value is not None
348
+ ]
349
+ last_seen = max(last_values) if last_values else first_seen
350
+ return OriginCompanionCandidate(
351
+ delta_opcode=self.delta_opcode,
352
+ companion_opcodes=self.companion_opcodes,
353
+ companion_lengths=self.companion_lengths,
354
+ observations=len(self.observation_ids),
355
+ distinct_tokens=len(self.token_digests),
356
+ delta_lengths=tuple(sorted(self.delta_lengths)),
357
+ delta_token_offsets=tuple(sorted(self.delta_offsets)),
358
+ first_token_offsets=tuple(sorted(self.first_offsets)),
359
+ second_token_offsets=tuple(sorted(self.second_offsets)),
360
+ first_seen=first_seen,
361
+ last_seen=last_seen,
362
+ observation_ids=tuple(sorted(self.observation_ids)),
363
+ token_digests=tuple(sorted(self.token_digests)),
364
+ )
365
+
366
+
367
+ class OriginLearner:
368
+ """Thread-safe in-memory aggregator for structurally discovered families."""
369
+
370
+ def __init__(
371
+ self,
372
+ min_observations: int = 2,
373
+ *,
374
+ max_candidates: int = DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES,
375
+ max_observations: int = DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS,
376
+ ) -> None:
377
+ if (
378
+ isinstance(min_observations, bool)
379
+ or not isinstance(min_observations, int)
380
+ or min_observations <= 0
381
+ ):
382
+ raise ValueError("min_observations must be a positive integer")
383
+ for name, value in (
384
+ ("max_candidates", max_candidates),
385
+ ("max_observations", max_observations),
386
+ ):
387
+ if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
388
+ raise ValueError(f"{name} must be a positive integer")
389
+ self.min_observations = min_observations
390
+ self.max_candidates = max_candidates
391
+ self.max_observations = max_observations
392
+ self._states: dict[tuple[int, int, int, int, int], _CandidateState] = {}
393
+ self._observation_count = 0
394
+ self._lock = Lock()
395
+
396
+ @classmethod
397
+ def load(
398
+ cls,
399
+ path: str | Path,
400
+ *,
401
+ min_observations: Optional[int] = None,
402
+ max_candidates: Optional[int] = None,
403
+ max_observations: Optional[int] = None,
404
+ ) -> "OriginLearner":
405
+ candidate_path = Path(path)
406
+ if not candidate_path.exists():
407
+ return cls(
408
+ 2 if min_observations is None else min_observations,
409
+ max_candidates=(
410
+ max_candidates
411
+ if max_candidates is not None
412
+ else DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES
413
+ ),
414
+ max_observations=(
415
+ max_observations
416
+ if max_observations is not None
417
+ else DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS
418
+ ),
419
+ )
420
+ try:
421
+ data = json.loads(candidate_path.read_text(encoding="utf-8-sig"))
422
+ except (json.JSONDecodeError, UnicodeError) as exc:
423
+ raise ProfileError(
424
+ f"Could not parse origin candidates JSON {candidate_path}: {exc}"
425
+ ) from exc
426
+ if not isinstance(data, dict):
427
+ raise ProfileError(f"Origin candidates {candidate_path} must be an object")
428
+ version = data.get("version")
429
+ if version != CANDIDATE_FORMAT_VERSION or isinstance(version, bool):
430
+ raise ProfileError(
431
+ f"version in {candidate_path} must be {CANDIDATE_FORMAT_VERSION}"
432
+ )
433
+ stored_min = data.get("min_observations", 2)
434
+ if (
435
+ isinstance(stored_min, bool)
436
+ or not isinstance(stored_min, int)
437
+ or stored_min <= 0
438
+ ):
439
+ raise ProfileError(
440
+ f"min_observations in {candidate_path} must be a positive integer"
441
+ )
442
+ threshold = min_observations if min_observations is not None else stored_min
443
+ stored_max_candidates = data.get(
444
+ "max_candidates", DEFAULT_ORIGIN_LEARNING_MAX_CANDIDATES
445
+ )
446
+ stored_max_observations = data.get(
447
+ "max_observations", DEFAULT_ORIGIN_LEARNING_MAX_OBSERVATIONS
448
+ )
449
+ try:
450
+ learner = cls(
451
+ threshold,
452
+ max_candidates=(
453
+ max_candidates
454
+ if max_candidates is not None
455
+ else stored_max_candidates
456
+ ),
457
+ max_observations=(
458
+ max_observations
459
+ if max_observations is not None
460
+ else stored_max_observations
461
+ ),
462
+ )
463
+ except ValueError as exc:
464
+ if (
465
+ min_observations is not None
466
+ or max_candidates is not None
467
+ or max_observations is not None
468
+ ):
469
+ raise
470
+ raise ProfileError(
471
+ f"invalid origin-learning limits in {candidate_path}: {exc}"
472
+ ) from exc
473
+ entries = data.get("candidates", [])
474
+ if not isinstance(entries, list):
475
+ raise ProfileError(f"candidates in {candidate_path} must be a list")
476
+ for index, entry in enumerate(entries):
477
+ learner._load_entry(entry, f"candidates[{index}] in {candidate_path}")
478
+ return learner
479
+
480
+ def _load_entry(self, entry: object, location: str) -> None:
481
+ if not isinstance(entry, dict):
482
+ raise ProfileError(f"{location} must be an object")
483
+ if entry.get("detection") != COMPANION_DETECTION:
484
+ raise ProfileError(f"{location}.detection must be {COMPANION_DETECTION!r}")
485
+ delta_opcode = _parse_opcode(
486
+ entry.get("delta_opcode"), f"{location}.delta_opcode"
487
+ )
488
+ raw_opcodes = entry.get("companion_opcodes")
489
+ raw_lengths = entry.get("companion_lengths")
490
+ if not isinstance(raw_opcodes, list) or len(raw_opcodes) != 2:
491
+ raise ProfileError(f"{location}.companion_opcodes must contain two values")
492
+ if not isinstance(raw_lengths, list) or len(raw_lengths) != 2:
493
+ raise ProfileError(f"{location}.companion_lengths must contain two values")
494
+ opcodes = (
495
+ _parse_opcode(raw_opcodes[0], f"{location}.companion_opcodes"),
496
+ _parse_opcode(raw_opcodes[1], f"{location}.companion_opcodes"),
497
+ )
498
+ lengths: list[int] = []
499
+ for value in raw_lengths:
500
+ if (
501
+ isinstance(value, bool)
502
+ or not isinstance(value, int)
503
+ or not 5 <= value <= 0xFFFF
504
+ ):
505
+ raise ProfileError(f"{location}.companion_lengths must be 5..65535")
506
+ lengths.append(value)
507
+ key = (delta_opcode, opcodes[0], opcodes[1], lengths[0], lengths[1])
508
+ state = _CandidateState(delta_opcode, opcodes, (lengths[0], lengths[1]))
509
+ observation_ids = entry.get("observation_ids", [])
510
+ token_digests = entry.get("token_digests", [])
511
+ delta_lengths = entry.get("delta_lengths", [])
512
+ if not isinstance(observation_ids, list) or any(
513
+ not isinstance(value, str) for value in observation_ids
514
+ ):
515
+ raise ProfileError(f"{location}.observation_ids must be a string list")
516
+ if not isinstance(token_digests, list) or any(
517
+ not isinstance(value, str) for value in token_digests
518
+ ):
519
+ raise ProfileError(f"{location}.token_digests must be a string list")
520
+ if not isinstance(delta_lengths, list) or any(
521
+ isinstance(value, bool)
522
+ or not isinstance(value, int)
523
+ or not 5 <= value <= 0xFFFF
524
+ for value in delta_lengths
525
+ ):
526
+ raise ProfileError(f"{location}.delta_lengths must contain 5..65535")
527
+ unique_observation_ids = set(observation_ids)
528
+ for name, values in (
529
+ ("token_digests", token_digests),
530
+ ("delta_lengths", delta_lengths),
531
+ ):
532
+ if len(set(values)) > len(unique_observation_ids):
533
+ raise ProfileError(
534
+ f"{location}.{name} cannot contain more distinct evidence "
535
+ "than observation_ids"
536
+ )
537
+ state.observation_ids.update(observation_ids)
538
+ state.token_digests.update(token_digests)
539
+ state.delta_lengths.update(delta_lengths)
540
+ offsets = entry.get("token_offsets", {})
541
+ if not isinstance(offsets, dict):
542
+ raise ProfileError(f"{location}.token_offsets must be an object")
543
+ for name, target in (
544
+ ("delta", state.delta_offsets),
545
+ ("first_companion", state.first_offsets),
546
+ ("second_companion", state.second_offsets),
547
+ ):
548
+ values = offsets.get(name, [])
549
+ if not isinstance(values, list) or any(
550
+ isinstance(value, bool) or not isinstance(value, int) or value < 0
551
+ for value in values
552
+ ):
553
+ raise ProfileError(f"{location}.token_offsets.{name} is invalid")
554
+ if len(set(values)) > len(unique_observation_ids):
555
+ raise ProfileError(
556
+ f"{location}.token_offsets.{name} cannot contain more "
557
+ "distinct evidence than observation_ids"
558
+ )
559
+ target.update(values)
560
+ first_seen = entry.get("first_seen")
561
+ last_seen = entry.get("last_seen")
562
+ if first_seen is not None and not isinstance(first_seen, str):
563
+ raise ProfileError(f"{location}.first_seen must be a string")
564
+ if last_seen is not None and not isinstance(last_seen, str):
565
+ raise ProfileError(f"{location}.last_seen must be a string")
566
+ state.first_seen_text = first_seen
567
+ state.last_seen_text = last_seen
568
+ if key in self._states:
569
+ raise ProfileError(f"{location} duplicates an earlier candidate family")
570
+ if len(self._states) >= self.max_candidates:
571
+ raise OriginLearningLimitError("candidate-family", self.max_candidates)
572
+ retained_observations = len(state.observation_ids)
573
+ if self._observation_count + retained_observations > self.max_observations:
574
+ raise OriginLearningLimitError("observation", self.max_observations)
575
+ self._states[key] = state
576
+ self._observation_count += retained_observations
577
+
578
+ def observe(self, observation: CompanionObservation) -> OriginCompanionCandidate:
579
+ if not isinstance(observation, CompanionObservation):
580
+ raise TypeError("OriginLearner.observe expects CompanionObservation")
581
+ with self._lock:
582
+ state = self._states.get(observation.family_key)
583
+ if (
584
+ state is not None
585
+ and observation.observation_id in state.observation_ids
586
+ ):
587
+ return state.snapshot()
588
+ if state is None:
589
+ if len(self._states) >= self.max_candidates:
590
+ raise OriginLearningLimitError(
591
+ "candidate-family", self.max_candidates
592
+ )
593
+ state = _CandidateState(
594
+ observation.delta_opcode,
595
+ observation.companion_opcodes,
596
+ observation.companion_lengths,
597
+ )
598
+ if self._observation_count >= self.max_observations:
599
+ raise OriginLearningLimitError("observation", self.max_observations)
600
+ added = state.add(observation)
601
+ if added:
602
+ self._observation_count += 1
603
+ self._states[observation.family_key] = state
604
+ return state.snapshot()
605
+
606
+ @property
607
+ def candidates(self) -> tuple[OriginCompanionCandidate, ...]:
608
+ with self._lock:
609
+ return tuple(self._states[key].snapshot() for key in sorted(self._states))
610
+
611
+ @property
612
+ def confirmed_candidates(self) -> tuple[OriginCompanionCandidate, ...]:
613
+ return tuple(
614
+ candidate
615
+ for candidate in self.candidates
616
+ if candidate.confirmed(self.min_observations)
617
+ )
618
+
619
+ def summary(self) -> str:
620
+ """Return a human-readable multi-line candidate report."""
621
+ candidates = self.candidates
622
+ confirmed_count = sum(
623
+ candidate.confirmed(self.min_observations) for candidate in candidates
624
+ )
625
+ lines = [
626
+ f"observed {len(candidates)} origin companion family/families; "
627
+ f"{confirmed_count} confirmed for promotion "
628
+ f"(threshold: {self.min_observations} observation(s))"
629
+ ]
630
+ if not candidates:
631
+ lines.append("no origin companion families observed")
632
+ return "\n".join(lines)
633
+
634
+ for candidate in candidates:
635
+ status = (
636
+ "confirmed"
637
+ if candidate.confirmed(self.min_observations)
638
+ else "candidate"
639
+ )
640
+ companions = " -> ".join(
641
+ f"0x{opcode:04X}" for opcode in candidate.companion_opcodes
642
+ )
643
+ lengths = ", ".join(str(length) for length in candidate.companion_lengths)
644
+ lines.append(
645
+ f"{status}: delta=0x{candidate.delta_opcode:04X} "
646
+ f"companions={companions} lengths=({lengths}) "
647
+ f"observations={candidate.observations} "
648
+ f"distinct_tokens={candidate.distinct_tokens}"
649
+ )
650
+ return "\n".join(lines)
651
+
652
+ def to_dict(self) -> dict[str, Any]:
653
+ return {
654
+ "version": CANDIDATE_FORMAT_VERSION,
655
+ "generated_at": _utc_text(),
656
+ "min_observations": self.min_observations,
657
+ "max_candidates": self.max_candidates,
658
+ "max_observations": self.max_observations,
659
+ "candidates": [
660
+ candidate.to_dict(self.min_observations)
661
+ for candidate in self.candidates
662
+ ],
663
+ }
664
+
665
+ def save(self, path: str | Path) -> Path:
666
+ candidate_path = Path(path)
667
+ _atomic_write_text(
668
+ candidate_path,
669
+ json.dumps(self.to_dict(), indent=2, sort_keys=True) + "\n",
670
+ )
671
+ return candidate_path
672
+
673
+
674
+ @dataclass(frozen=True)
675
+ class OriginPromotion:
676
+ path: Path
677
+ added: tuple[OriginCompanionCandidate, ...]
678
+ backup_path: Optional[Path]
679
+ written: bool
680
+
681
+
682
+ def promote_origin_candidates(
683
+ candidates_path: str | Path,
684
+ profile_path: str | Path,
685
+ *,
686
+ min_observations: Optional[int] = None,
687
+ backup: bool = True,
688
+ ) -> OriginPromotion:
689
+ """Explicitly merge confirmed learned families into an opcode profile."""
690
+ learner = OriginLearner.load(
691
+ candidates_path,
692
+ min_observations=min_observations,
693
+ )
694
+ confirmed = learner.confirmed_candidates
695
+ destination = Path(profile_path)
696
+ load_opcode_profile(destination) # Strictly validate before preserving it.
697
+ data = json.loads(destination.read_text(encoding="utf-8-sig"))
698
+ families = data.setdefault("origin_companion_families", [])
699
+ if not isinstance(families, list):
700
+ raise ProfileError(f"origin_companion_families in {destination} must be a list")
701
+
702
+ existing: set[tuple[int, int, int, int, int]] = set()
703
+ for index, family in enumerate(families):
704
+ if not isinstance(family, dict):
705
+ raise ProfileError(
706
+ f"origin_companion_families[{index}] in {destination} must be an object"
707
+ )
708
+ raw_opcodes = family.get("companion_opcodes")
709
+ raw_lengths = family.get("companion_lengths")
710
+ if not isinstance(raw_opcodes, list) or len(raw_opcodes) != 2:
711
+ raise ProfileError("existing companion family has invalid opcodes")
712
+ if not isinstance(raw_lengths, list) or len(raw_lengths) != 2:
713
+ raise ProfileError("existing companion family has invalid lengths")
714
+ existing.add(
715
+ (
716
+ _parse_opcode(family.get("delta_opcode"), "delta_opcode"),
717
+ _parse_opcode(raw_opcodes[0], "companion opcode"),
718
+ _parse_opcode(raw_opcodes[1], "companion opcode"),
719
+ int(raw_lengths[0]),
720
+ int(raw_lengths[1]),
721
+ )
722
+ )
723
+
724
+ added = tuple(
725
+ candidate for candidate in confirmed if candidate.family_key not in existing
726
+ )
727
+ if not added:
728
+ return OriginPromotion(destination, (), None, False)
729
+ promoted_at = _utc_text()
730
+ for candidate in added:
731
+ families.append(
732
+ {
733
+ "detection": COMPANION_DETECTION,
734
+ "delta_opcode": _opcode_text(candidate.delta_opcode),
735
+ "companion_opcodes": [
736
+ _opcode_text(opcode) for opcode in candidate.companion_opcodes
737
+ ],
738
+ "companion_lengths": list(candidate.companion_lengths),
739
+ "observations": candidate.observations,
740
+ "promoted_at": promoted_at,
741
+ }
742
+ )
743
+ data["updated_at"] = promoted_at
744
+ backup_path = None
745
+ if backup:
746
+ backup_path = _unique_backup_path(destination)
747
+ shutil.copy2(destination, backup_path)
748
+ _atomic_write_text(destination, json.dumps(data, indent=2, sort_keys=True) + "\n")
749
+ return OriginPromotion(destination, added, backup_path, True)
750
+
751
+
752
+ def _unique_backup_path(path: Path) -> Path:
753
+ backup_dir = path.parent / "opcodes_backups"
754
+ backup_dir.mkdir(parents=True, exist_ok=True)
755
+ stamp = dt.datetime.now(tz=dt.timezone.utc).strftime("%Y%m%d%H%M%S%f")
756
+ return backup_dir / f"{path.name}.bak.{stamp}"
757
+
758
+
759
+ def _atomic_write_text(path: Path, value: str) -> None:
760
+ path.parent.mkdir(parents=True, exist_ok=True)
761
+ temporary: Optional[Path] = None
762
+ try:
763
+ with tempfile.NamedTemporaryFile(
764
+ mode="w",
765
+ encoding="utf-8",
766
+ newline="\n",
767
+ dir=path.parent,
768
+ prefix=f".{path.name}.",
769
+ suffix=".tmp",
770
+ delete=False,
771
+ ) as handle:
772
+ temporary = Path(handle.name)
773
+ handle.write(value)
774
+ handle.flush()
775
+ os.fsync(handle.fileno())
776
+ os.replace(temporary, path)
777
+ finally:
778
+ if temporary is not None and temporary.exists():
779
+ temporary.unlink()