python-gdb 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: python-gdb
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Classifier: Development Status :: 3 - Alpha
5
5
  Classifier: Intended Audience :: Science/Research
6
6
  Classifier: Topic :: Scientific/Engineering :: GIS
@@ -190,6 +190,7 @@ zensical serve
190
190
  boundary on reverse engineering.
191
191
  - [`docs/provenance/`](docs/provenance/index.md) — the original
192
192
  research log and write-up this implementation was derived from.
193
+ - [`CHANGELOG.md`](CHANGELOG.md) — what changed in each release.
193
194
 
194
195
  ## Testing
195
196
 
@@ -152,6 +152,7 @@ zensical serve
152
152
  boundary on reverse engineering.
153
153
  - [`docs/provenance/`](docs/provenance/index.md) — the original
154
154
  research log and write-up this implementation was derived from.
155
+ - [`CHANGELOG.md`](CHANGELOG.md) — what changed in each release.
155
156
 
156
157
  ## Testing
157
158
 
@@ -16,7 +16,7 @@ from __future__ import annotations
16
16
  import warnings
17
17
  from dataclasses import dataclass
18
18
  from pathlib import Path
19
- from typing import Dict, Iterator, List, Optional, Tuple, Union
19
+ from typing import Dict, Iterator, List, Optional, Sequence, Tuple, Union
20
20
 
21
21
  import numpy as np
22
22
 
@@ -60,6 +60,87 @@ LineRef = Union[str, Tuple[str, int], "LineRecord"]
60
60
  ChannelRef = Union[str, Tuple[str, int], "ChannelRecord"]
61
61
 
62
62
 
63
+ def _finite_values(values: np.ndarray) -> np.ndarray:
64
+ v = np.asarray(values, dtype=float).ravel()
65
+ return v[np.isfinite(v) & (np.abs(v) < 1e30)]
66
+
67
+
68
+ def _roughness(values: np.ndarray) -> Optional[float]:
69
+ """
70
+ Median row-to-row step divided by the 5-95% spread of the values.
71
+
72
+ Returns
73
+ -------
74
+ float or None
75
+ Small for data that varies smoothly along its rows, large for
76
+ the same values in a scrambled order. None if there are too few
77
+ finite values or they are constant.
78
+ """
79
+ v = _finite_values(values)
80
+ if len(v) < 50:
81
+ return None
82
+ spread = float(np.percentile(v, 95) - np.percentile(v, 5))
83
+ if spread == 0.0:
84
+ return None
85
+ return float(np.median(np.abs(np.diff(v))) / spread)
86
+
87
+
88
+ def _row_order_pick(first: np.ndarray, last: np.ndarray, factor: float = 3.0) -> Optional[int]:
89
+ """
90
+ Decide which of two copies of one channel is in acquisition order.
91
+
92
+ Parameters
93
+ ----------
94
+ first, last : numpy.ndarray
95
+ The earlier and later copy in the blob chain.
96
+ factor : float, optional
97
+ How many times rougher one copy must be than the other.
98
+
99
+ Returns
100
+ -------
101
+ int or None
102
+ 0 if the *first* copy is the smooth one and the last is a
103
+ markedly rougher reordering of it; 1 if the last is the smooth
104
+ one; None if the copies are not a pure reordering of each other
105
+ (different values or length) or neither is clearly smoother.
106
+
107
+ Notes
108
+ -----
109
+ A copy that is *itself* perfectly monotone is never judged: a sorted
110
+ ramp is smoother than any real signal, and a re-sort by a channel's
111
+ own value (a coordinate re-sorted by itself) leaves exactly that. It
112
+ could equally be a genuine ID or time channel, and nothing here can
113
+ tell the two apart, so such a pair is undecided.
114
+ """
115
+ if len(first) != len(last):
116
+ return None
117
+ a, b = np.asarray(first).ravel(), np.asarray(last).ravel()
118
+ try:
119
+ if not np.array_equal(np.sort(a), np.sort(b), equal_nan=True):
120
+ return None
121
+ except TypeError: # dtype without a NaN notion (integers)
122
+ if not np.array_equal(np.sort(a), np.sort(b)):
123
+ return None
124
+ if _is_monotone_reference(a) or _is_monotone_reference(b):
125
+ return None
126
+ ra, rb = _roughness(a), _roughness(b)
127
+ if ra is None or rb is None:
128
+ return None
129
+ if rb > 0 and rb >= factor * ra:
130
+ return 0
131
+ if ra > 0 and ra >= factor * rb:
132
+ return 1
133
+ return None
134
+
135
+
136
+ def _is_monotone_reference(values: np.ndarray) -> bool:
137
+ """A non-constant channel stored in non-decreasing order (an ID, date or time)."""
138
+ v = _finite_values(values)
139
+ if len(v) < 50 or v[0] == v[-1]:
140
+ return False
141
+ return bool(np.mean(np.diff(v) >= 0) >= 0.999)
142
+
143
+
63
144
  @dataclass
64
145
  class CompressionInfo:
65
146
  """
@@ -100,6 +181,20 @@ class GDB:
100
181
  ----------
101
182
  path : str
102
183
  Path to the `.gdb` file.
184
+ duplicate_blobs : {"last", "row_order"}, optional
185
+ What to do when a (line, channel) has more than one blob in the
186
+ blob chain (issue #2). `"last"` (the default) uses the last one
187
+ in chain order. `"row_order"` is an opt-in **heuristic**, not a
188
+ decoded field: for a duplicated numeric channel whose two
189
+ copies hold exactly the same values in a different row order
190
+ (a stale re-sorted copy), on a line that has an order-defining
191
+ channel (see Notes), it prefers the copy whose values vary
192
+ smoothly along the rows -- acquisition order, the order the
193
+ line's ID/time channels are stored in -- and keeps the last
194
+ copy in every other case. Either way one `GDBParseWarning`
195
+ names the affected pairs and, for `"row_order"`, which ones it
196
+ overrode. Choosing needs both copies decoded, so it is done
197
+ once, when the blob index is first built.
103
198
 
104
199
  Raises
105
200
  ------
@@ -108,7 +203,8 @@ class GDB:
108
203
  expected `.gdb` magic -- unlike the module-level functions in
109
204
  `gdb_reader`/`registry` (which warn and return empty results),
110
205
  since a `GDB` object that isn't backed by a real `.gdb` file
111
- can't usefully do anything at all.
206
+ can't usefully do anything at all. Also if `duplicate_blobs` is
207
+ not one of the values above.
112
208
 
113
209
  Examples
114
210
  --------
@@ -137,10 +233,32 @@ class GDB:
137
233
  the Rust-plan's M4 notes). Close it (`db.close()`, or use `GDB` as a
138
234
  context manager) when done with it, or just let it get
139
235
  garbage-collected -- `__del__` closes it too, as a safety net.
236
+
237
+ `duplicate_blobs="row_order"` only ever chooses between two copies
238
+ of one channel that are a pure reordering of each other, and only on
239
+ a line with an *order-defining channel*: a single-copy numeric
240
+ channel of the same length stored in monotone order (typically an
241
+ ID, date or time). A revised copy (different values) is never
242
+ second-guessed, and neither is a pair with a copy that is itself
243
+ perfectly monotone (a re-sort by a channel's own value leaves a
244
+ smooth ramp that looks like the best copy but is the stale one).
245
+ Among a qualifying pair, the copy at least 3x
246
+ rougher along the rows -- median row-to-row step over the 5-95%
247
+ spread of the values -- is treated as the stale one, since data in
248
+ acquisition order varies smoothly and a re-sort scrambles that. This
249
+ was validated against independent spreadsheet exports of the one
250
+ real file known to have such copies (it never contradicted them),
251
+ but it cannot decide a channel that is smooth in both orders, and no
252
+ on-disk marker has been found that would make it unnecessary.
140
253
  """
141
254
 
142
- def __init__(self, path: str):
255
+ def __init__(self, path: str, duplicate_blobs: str = "last"):
256
+ if duplicate_blobs not in ("last", "row_order"):
257
+ raise ValueError(
258
+ f"duplicate_blobs must be 'last' or 'row_order', got {duplicate_blobs!r}"
259
+ )
143
260
  self.path = path
261
+ self.duplicate_blobs = duplicate_blobs
144
262
  self._file = open(path, "rb")
145
263
  header = self._file.read(4096)
146
264
  if not check_magic(header):
@@ -435,16 +553,207 @@ class GDB:
435
553
  anything beyond an occasional one-off lookup -- this class
436
554
  always wants line/channel listings and random-access
437
555
  reads, so it always builds the index.
556
+
557
+ Warns
558
+ -----
559
+ GDBParseWarning
560
+ If a real line's channel has more than one blob in the
561
+ blob chain. By default the **last** one in chain order is
562
+ used, but that is not always the current copy (issue #2):
563
+ an older copy with the same values in a different row order
564
+ can sit either before or after the current one, and nothing
565
+ decoded so far says which is which. With
566
+ `duplicate_blobs="row_order"` the warning also says which
567
+ pairs were switched to an earlier copy. Duplicates in the
568
+ administrative slots past the last real line (the REG/IPJ
569
+ registry, whose stale copies are expected and handled by
570
+ `pygdb.registry`) are not reported.
438
571
  """
439
572
  if self._blob_index is None:
440
573
  chans_max = self.chans_max
441
- index: Dict[Tuple[int, int], BlobHeader] = {}
574
+ copies: Dict[Tuple[int, int], List[BlobHeader]] = {}
442
575
  for blob in iter_blobs(self.path):
443
- index[blob.line_channel(chans_max)] = blob
576
+ copies.setdefault(blob.line_channel(chans_max), []).append(blob)
577
+ index: Dict[Tuple[int, int], BlobHeader] = {k: v[-1] for k, v in copies.items()}
578
+ duplicated = [k for k, v in copies.items() if len(v) > 1]
444
579
  self._blob_index = index
445
580
  self._calibrate_line_indices()
581
+ if duplicated:
582
+ overridden: List[Tuple[int, int]] = []
583
+ if self.duplicate_blobs == "row_order":
584
+ overridden = self._prefer_row_order_copies(duplicated, copies, index)
585
+ self._warn_duplicate_blobs(duplicated, overridden)
446
586
  return self._blob_index
447
587
 
588
+ def _read_copy(self, blob: BlobHeader, channel: ChannelRecord) -> np.ndarray:
589
+ return read_blob_values(
590
+ self.path, blob, channel,
591
+ comp_level=self.comp_level or 0, page_size=self.page_size, file=self._file,
592
+ )
593
+
594
+ def _prefer_row_order_copies(
595
+ self,
596
+ duplicated: List[Tuple[int, int]],
597
+ copies: Dict[Tuple[int, int], List[BlobHeader]],
598
+ index: Dict[Tuple[int, int], BlobHeader],
599
+ ) -> List[Tuple[int, int]]:
600
+ """
601
+ Apply `duplicate_blobs="row_order"` to the duplicated pairs.
602
+
603
+ Parameters
604
+ ----------
605
+ duplicated : list of (int, int)
606
+ `(line_slot, channel_slot)` keys with more than one blob.
607
+ copies : dict of {(int, int) : list of BlobHeader}
608
+ Every blob for each key, in chain order.
609
+ index : dict of {(int, int) : BlobHeader}
610
+ The blob index being built; updated in place with the
611
+ earlier copy wherever one is preferred.
612
+
613
+ Returns
614
+ -------
615
+ list of (int, int)
616
+ The keys switched from the last copy to an earlier one.
617
+
618
+ Notes
619
+ -----
620
+ See the class docstring for exactly when a pair qualifies. A pair
621
+ that doesn't (a string or array channel, more than two copies,
622
+ different values, no order-defining channel on the line, or no
623
+ clear winner) simply keeps the last copy.
624
+ """
625
+ real_lines = {line.index for line in self.lines}
626
+ channels = {c.index: c for c in self.channels}
627
+ duplicated_keys = set(duplicated)
628
+ reference_cache: Dict[Tuple[int, int], bool] = {}
629
+ overridden: List[Tuple[int, int]] = []
630
+ for key in duplicated:
631
+ line_slot, channel_slot = key
632
+ channel = channels.get(channel_slot)
633
+ blobs = copies[key]
634
+ if (
635
+ line_slot not in real_lines or channel is None or len(blobs) != 2
636
+ or channel.is_string or channel.is_array
637
+ ):
638
+ continue
639
+ first, last = self._read_copy(blobs[0], channel), self._read_copy(blobs[1], channel)
640
+ if len(first) != len(last):
641
+ continue
642
+ ref_key = (line_slot, len(first))
643
+ if ref_key not in reference_cache:
644
+ reference_cache[ref_key] = self._line_has_order_reference(
645
+ line_slot, len(first), duplicated_keys, channels
646
+ )
647
+ if reference_cache[ref_key] and _row_order_pick(first, last) == 0:
648
+ index[key] = blobs[0]
649
+ overridden.append(key)
650
+ return overridden
651
+
652
+ def _line_has_order_reference(
653
+ self,
654
+ line_slot: int,
655
+ n_rows: int,
656
+ duplicated_keys: set,
657
+ channels: Dict[int, ChannelRecord],
658
+ ) -> bool:
659
+ """
660
+ Whether a line has a channel proving its rows are in some fixed order.
661
+
662
+ Parameters
663
+ ----------
664
+ line_slot : int
665
+ The line's slot.
666
+ n_rows : int
667
+ The row count of the duplicated copies being judged.
668
+ duplicated_keys : set of (int, int)
669
+ Keys with more than one blob -- excluded, since their own
670
+ order is what is in question.
671
+ channels : dict of {int : ChannelRecord}
672
+ Channel records by slot.
673
+
674
+ Returns
675
+ -------
676
+ bool
677
+ True if some single-copy, numeric, non-array channel of
678
+ `n_rows` values on this line is stored in monotone
679
+ (non-decreasing) order and isn't constant -- typically an
680
+ ID, date or time channel.
681
+ """
682
+ for (ls, cs), blob in sorted(self._ensure_blob_index().items()):
683
+ channel = channels.get(cs)
684
+ if (
685
+ ls != line_slot or (ls, cs) in duplicated_keys or channel is None
686
+ or channel.is_string or channel.is_array
687
+ ):
688
+ continue
689
+ values = self._read_copy(blob, channel)
690
+ if len(values) == n_rows and _is_monotone_reference(values):
691
+ return True
692
+ return False
693
+
694
+ def _warn_duplicate_blobs(
695
+ self,
696
+ duplicated: List[Tuple[int, int]],
697
+ overridden: Sequence[Tuple[int, int]] = (),
698
+ ) -> None:
699
+ """
700
+ Warn about (line, channel) pairs that have more than one blob.
701
+
702
+ Parameters
703
+ ----------
704
+ duplicated : list of (int, int)
705
+ `(line_slot, channel_slot)` keys seen more than once in the
706
+ blob chain.
707
+ overridden : sequence of (int, int), optional
708
+ The keys `duplicate_blobs="row_order"` switched to an
709
+ earlier copy.
710
+
711
+ Warns
712
+ -----
713
+ GDBParseWarning
714
+ Once, naming how many real (line, channel) pairs are
715
+ affected and a few examples (and, for `"row_order"`, which
716
+ ones were switched). Nothing is emitted if every duplicate
717
+ is in an administrative slot.
718
+ """
719
+ line_names = {line.index: line.name for line in self.lines}
720
+ channel_names = {c.index: c.name for c in self.channels}
721
+
722
+ def describe(keys, limit=3):
723
+ named = [
724
+ f"line {line_names[ls]!r} channel {channel_names.get(cs, f'#{cs}')!r}"
725
+ for ls, cs in keys if ls in line_names
726
+ ]
727
+ more = f" and {len(named) - limit} more" if len(named) > limit else ""
728
+ return ", ".join(named[:limit]) + more
729
+
730
+ n_real = sum(1 for ls, _ in duplicated if ls in line_names)
731
+ if not n_real:
732
+ return
733
+ head = (
734
+ f"{self.path}: {n_real} (line, channel) pair(s) have more than one "
735
+ f"blob in the blob chain ({describe(duplicated)})"
736
+ )
737
+ if self.duplicate_blobs == "row_order":
738
+ tail = (
739
+ f" -- duplicate_blobs='row_order': switched {len(overridden)} pair(s) to "
740
+ f"an earlier copy because the last copy was the same values in a "
741
+ f"markedly rougher row order ({describe(overridden)}); kept the last "
742
+ f"copy in chain order for the other {n_real - len(overridden)}. This "
743
+ f"is a heuristic, not a decoded field (see issue #2)"
744
+ if overridden else
745
+ f" -- duplicate_blobs='row_order': no pair needed switching, so the "
746
+ f"last copy in chain order was kept for all of them (see issue #2)"
747
+ )
748
+ else:
749
+ tail = (
750
+ " -- using the last one in chain order, which is not always the "
751
+ "current copy. If a channel's rows look scrambled against the line's "
752
+ "other channels, this is the likely cause; duplicate_blobs='row_order' "
753
+ "can pick the acquisition-order copy (see issue #2)"
754
+ )
755
+ warnings.warn(head + tail, GDBParseWarning, stacklevel=4)
756
+
448
757
  def _calibrate_line_indices(self) -> None:
449
758
  """
450
759
  Correct a possible small, fixed off-by-N in every line's index.
@@ -1682,11 +1682,16 @@ def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
1682
1682
  established ground truth exactly) and for both single- and
1683
1683
  multi-page DB_COMP_SPEED (a real 2-page `Northing_AMGz55` blob
1684
1684
  decodes to sane real coordinates with real `rDUMMY` sentinels).
1685
- See docs/provenance/notes.md section 6.6d: a multi-page blob is
1686
- simply one continuous compressed stream spanning the whole
1687
- `n_pages*page_size` span, not one independently-framed chunk per
1688
- page -- no special multi-page logic was actually needed once this
1689
- was verified, just reading the full span instead of one page.
1685
+ See docs/provenance/notes.md section 6.6d: a multi-page blob has no
1686
+ per-page re-framing -- just read the full `n_pages*page_size` span
1687
+ instead of one page. **A DB_COMP_SPEED blob is, however, a chain of
1688
+ chunks of at most 16368 decompressed bytes each, not one chunk**
1689
+ (docs/provenance/notes.md section 6.6e): only the first carries the
1690
+ 16-byte magic, later ones are a bare 12-byte sub-header plus
1691
+ payload, and the blob header's `+24` field is the total
1692
+ decompressed size across the chain. Decoding only the first chunk
1693
+ -- as this function once did -- silently truncated any channel
1694
+ longer than 2046 float64 values on a line.
1690
1695
 
1691
1696
  **A real third on-disk variant, auto-detected here rather than
1692
1697
  assumed away (docs/provenance/notes.md section 6.6b):** even
@@ -1769,11 +1774,11 @@ def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
1769
1774
  # to be a SINGLE continuous compressed stream spanning the whole
1770
1775
  # n_pages*page_size span -- NOT one independently-framed chunk per
1771
1776
  # page. There is no per-page re-framing to handle: reading the full
1772
- # span and decompressing it as one stream (zlib.decompressobj()
1773
- # naturally stops at the real end of stream and reports the rest as
1774
- # padding; the LZRW1 chunk header's own decompressed_length/
1775
- # chunk_length fields already span the full compressed length
1776
- # regardless of how many pages it spilled into) is sufficient.
1777
+ # span is sufficient for zlib (zlib.decompressobj() naturally stops
1778
+ # at the real end of stream and reports the rest as padding, and its
1779
+ # single stream always matches the blob header's `+24` total). LZRW1
1780
+ # is different: the span holds a chain of chunks, not one -- see the
1781
+ # DB_COMP_SPEED branch below.
1777
1782
  if page_size is None:
1778
1783
  with _file_handle(path, file) as f:
1779
1784
  header = f.read(128)
@@ -1838,9 +1843,19 @@ def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
1838
1843
  )
1839
1844
  return np.array([])
1840
1845
  elif subtype == _lzrw1.DB_COMP_SPEED:
1846
+ # A blob is a chain of chunks of at most 16368 decompressed bytes
1847
+ # each (docs/spec.md section 7.3); the blob header's `+24` field
1848
+ # is the total across all of them, and is what tells the decoder
1849
+ # where the chain ends (the bytes after the last chunk are page
1850
+ # padding, not a reliable terminator).
1851
+ with _file_handle(path, file) as f:
1852
+ f.seek(blob.offset + 24)
1853
+ total_field = f.read(4)
1854
+ total_decompressed = (
1855
+ struct.unpack("<i", total_field)[0] if len(total_field) == 4 else 0
1856
+ )
1841
1857
  try:
1842
- chunk = _lzrw1.parse_chunk_header(raw_span, 0)
1843
- decompressed = _lzrw1.decode_speed_chunk(raw_span, chunk)
1858
+ decompressed = _lzrw1.decode_speed_blob(raw_span, total_decompressed)
1844
1859
  except _lzrw1.LZRW1DecodeError as e:
1845
1860
  _warn(
1846
1861
  f"blob_index={blob.blob_index}: LZRW1 chunk decode failed ({e}) "
@@ -15,9 +15,10 @@ reference C wrapper:
15
15
  1. No 4-byte FLAG_BYTES prefix (the reference C code's own
16
16
  FLAG_COMPRESS/FLAG_COPY byte + 3 padding bytes) -- the control word
17
17
  starts immediately for a compressed chunk.
18
- 2. Each chunk (which may span several of the file's physical
19
- 1024-byte pages) is preceded by a 28-byte Geosoft-specific wrapper,
20
- not part of LZRW1 itself:
18
+ 2. The first chunk of a blob (which may span several of the file's
19
+ physical 1024-byte pages) is preceded by a 28-byte Geosoft-specific
20
+ wrapper, not part of LZRW1 itself (see point 4 for what follows
21
+ it):
21
22
  - 16 bytes: the magic sub-header shared with the `.grd` sibling
22
23
  format and with `.gdb`'s DB_COMP_SIZE (zlib) mode:
23
24
  `0f 0e ff fe 12 34 56 78 <subtype int32> <reserved int32>`
@@ -49,6 +50,21 @@ reference C wrapper:
49
50
  values). Every real marker value found across all 10 real
50
51
  Speed files was one of these two constants -- zero exceptions,
51
52
  zero unrecognized third values.
53
+ 4. **A blob is a chain of chunks, not one chunk** ([CONFIRMED] on
54
+ every Speed blob checked -- see docs/spec.md section 7.3/7.4). A
55
+ chunk decompresses to at most 16368 bytes (2046 float64 values);
56
+ a channel holding more data than that on one line is split across
57
+ several chunks stored back to back. Only the *first* chunk of a
58
+ blob carries the 16-byte magic; each later one is just its own
59
+ bare 12-byte `<decompressed_length> <chunk_length> <marker>`
60
+ sub-header immediately followed by its payload, starting
61
+ `chunk_length` bytes after the previous sub-header began. The
62
+ blob header (docs/spec.md section 7.4) records the total
63
+ decompressed size at `+24`, which is how a reader knows when to
64
+ stop -- the bytes after the last chunk are page padding, not
65
+ zeros, so they can't be relied on as a terminator. Every chunk
66
+ decoded independently (LZRW1 back-references never reach across
67
+ a chunk boundary). See `decode_speed_blob`.
52
68
 
53
69
  Validated exactly (not just "plausibly") against **all 10 real**
54
70
  DB_COMP_SPEED files now in this project's sample set (the original 4
@@ -72,6 +88,7 @@ from __future__ import annotations
72
88
  import struct
73
89
  import warnings
74
90
  from dataclasses import dataclass
91
+ from typing import List, Optional
75
92
 
76
93
  try:
77
94
  from . import _native as _native_ext
@@ -208,7 +225,8 @@ def _lzrw1_decompress_py(data: bytes, start: int, decompressed_length: int) -> b
208
225
 
209
226
  @dataclass
210
227
  class SpeedChunk:
211
- magic_offset: int # file offset of the 16-byte magic sub-header
228
+ magic_offset: Optional[int] # offset of the 16-byte magic sub-header; None for a
229
+ # continuation chunk, which has no magic of its own
212
230
  subtype: int # 1 = DB_COMP_SPEED, 2 = DB_COMP_SIZE
213
231
  decompressed_length: int
214
232
  chunk_length: int # includes the 12-byte length sub-header
@@ -361,11 +379,135 @@ def decode_speed_chunk(data: bytes, chunk: SpeedChunk):
361
379
  ) from e
362
380
 
363
381
 
382
+ def decode_speed_blob(data: bytes, total_decompressed_length: int = 0):
383
+ """
384
+ Decode every chunk of a DB_COMP_SPEED blob, in order.
385
+
386
+ Parameters
387
+ ----------
388
+ data : bytes or bytearray
389
+ The blob's compressed span, starting at the 16-byte magic of its
390
+ first chunk (i.e. everything after the blob header,
391
+ docs/spec.md section 7.4).
392
+ total_decompressed_length : int, optional
393
+ The blob's total decompressed size in bytes, from its header
394
+ (`+24`, docs/spec.md section 7.4). Decoding continues chunk by
395
+ chunk until this many bytes have been produced. If not positive
396
+ (a header that doesn't carry it, as in some hand-built
397
+ fixtures), only the first chunk is decoded.
398
+
399
+ Returns
400
+ -------
401
+ bytearray or bytes
402
+ The concatenated output of every chunk. A single-chunk blob
403
+ returns exactly what `decode_speed_chunk` does for it (no extra
404
+ copy); a multi-chunk one is a new, writable `bytearray`.
405
+
406
+ Raises
407
+ ------
408
+ LZRW1DecodeError
409
+ If any chunk fails to decode (see `decode_speed_chunk`), the
410
+ chain runs off the end of `data`, or the chunks don't add up
411
+ to exactly `total_decompressed_length`.
412
+
413
+ Notes
414
+ -----
415
+ A blob is a chain of chunks of at most 16368 decompressed bytes
416
+ each, not a single chunk (module docstring point 4): only the first
417
+ carries the 16-byte magic, and every later one is a bare 12-byte
418
+ sub-header plus payload starting `chunk_length` bytes after the
419
+ previous sub-header began. This reader used to decode only the first
420
+ chunk, silently truncating any channel longer than 2046 float64
421
+ values on a line to exactly that length.
422
+
423
+ Dispatches to the compiled `pygdb._native` extension when it's
424
+ available and `data` is `bytes` (same algorithm, ported to Rust --
425
+ see `rust/src/lib.rs`), falling back to the pure-Python
426
+ `_decode_speed_blob_py` below otherwise. The native version raises
427
+ `ValueError`/`IndexError` for the same conditions; both are turned
428
+ into `LZRW1DecodeError` here, so callers never see which backend
429
+ produced a failure.
430
+ """
431
+ if _native_ext is not None and isinstance(data, bytes):
432
+ try:
433
+ return _native_ext.decode_speed_blob(data, total_decompressed_length)
434
+ except (ValueError, IndexError) as e:
435
+ raise LZRW1DecodeError(str(e)) from e
436
+ return _decode_speed_blob_py(data, total_decompressed_length)
437
+
438
+
439
+ def _decode_speed_blob_py(data: bytes, total_decompressed_length: int = 0):
440
+ """
441
+ Pure-Python reference implementation of `decode_speed_blob`.
442
+
443
+ Parameters
444
+ ----------
445
+ data : bytes or bytearray
446
+ See `decode_speed_blob`.
447
+ total_decompressed_length : int, optional
448
+ See `decode_speed_blob`.
449
+
450
+ Returns
451
+ -------
452
+ bytearray or bytes
453
+ See `decode_speed_blob`.
454
+
455
+ Raises
456
+ ------
457
+ LZRW1DecodeError
458
+ See `decode_speed_blob`.
459
+ """
460
+ first = parse_chunk_header(data, 0)
461
+ out = decode_speed_chunk(data, first)
462
+ if total_decompressed_length <= len(out):
463
+ return out
464
+
465
+ parts: List[bytes] = [out]
466
+ produced = len(out)
467
+ header_start = first.payload_offset - 12 # where this chunk's sub-header began
468
+ chunk_length = first.chunk_length
469
+ while produced < total_decompressed_length:
470
+ if chunk_length < 12:
471
+ raise LZRW1DecodeError(
472
+ f"implausible chunk_length={chunk_length} -- corrupt chunk chain"
473
+ )
474
+ header_start += chunk_length
475
+ try:
476
+ decompressed_length, chunk_length, marker = struct.unpack_from(
477
+ "<iii", data, header_start
478
+ )
479
+ except struct.error as e:
480
+ raise LZRW1DecodeError(
481
+ f"chunk chain runs off the end of the data after {produced} of "
482
+ f"{total_decompressed_length} byte(s) -- truncated data"
483
+ ) from e
484
+ chunk = SpeedChunk(
485
+ magic_offset=None,
486
+ subtype=DB_COMP_SPEED,
487
+ decompressed_length=decompressed_length,
488
+ chunk_length=chunk_length,
489
+ marker=marker,
490
+ payload_offset=header_start + 12,
491
+ )
492
+ parts.append(decode_speed_chunk(data, chunk))
493
+ produced += decompressed_length
494
+
495
+ if produced != total_decompressed_length:
496
+ raise LZRW1DecodeError(
497
+ f"chunks decode to {produced} byte(s) but the blob header declares "
498
+ f"{total_decompressed_length}"
499
+ )
500
+ return bytearray().join(parts)
501
+
502
+
364
503
  def find_speed_chunks(data: bytes):
365
504
  """
366
505
  Yield every DB_COMP_SPEED (subtype==1) chunk found in `data`.
367
506
 
368
- Scans for the shared 16-byte magic byte-by-byte.
507
+ Scans for the shared 16-byte magic byte-by-byte. Since only the
508
+ *first* chunk of a blob carries that magic (see `decode_speed_blob`),
509
+ this finds one chunk per blob, not every chunk -- it's a scanning
510
+ helper for locating blobs, not a way to decode them.
369
511
 
370
512
  Parameters
371
513
  ----------
@@ -46,7 +46,7 @@ dependencies = [
46
46
 
47
47
  [[package]]
48
48
  name = "pygdb-native"
49
- version = "0.2.0"
49
+ version = "0.2.1"
50
50
  dependencies = [
51
51
  "flate2",
52
52
  "pyo3",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "pygdb-native"
3
- version = "0.2.0"
3
+ version = "0.2.1"
4
4
  edition = "2021"
5
5
  description = "Optional Rust-accelerated backend for python-gdb (pygdb._native)"
6
6
  license = "MIT"
@@ -148,6 +148,175 @@ fn lzrw1_decompress_impl(data: &[u8], start: usize, out: &mut [u8]) -> PyResult<
148
148
  Ok(())
149
149
  }
150
150
 
151
+ const DB_COMP_SPEED: i32 = 1;
152
+ const MARKER_COMPRESSED: i32 = -186263865; // 0xF4E5D6C7 -- payload is real LZRW1
153
+ const MARKER_STORED_RAW: i32 = -253635901; // 0xF0E1D2C3 -- payload is stored verbatim
154
+
155
+ /// A chunk's 12-byte `<decompressed_length> <chunk_length> <marker>`
156
+ /// sub-header, read from `data[header_start..]`.
157
+ fn read_chunk_subheader(data: &[u8], header_start: usize) -> PyResult<(i32, i32, i32)> {
158
+ let bytes = header_start
159
+ .checked_add(12)
160
+ .and_then(|end| data.get(header_start..end))
161
+ .ok_or_else(|| {
162
+ PyIndexError::new_err(
163
+ "decode_speed_blob: chunk chain runs off the end of the data -- truncated",
164
+ )
165
+ })?;
166
+ let field = |i: usize| i32::from_le_bytes([bytes[i], bytes[i + 1], bytes[i + 2], bytes[i + 3]]);
167
+ Ok((field(0), field(4), field(8)))
168
+ }
169
+
170
+ /// Decode one chunk's payload into `out` (exactly its `decompressed_length`
171
+ /// bytes) -- the Rust counterpart of `pygdb.lzrw1.decode_speed_chunk`,
172
+ /// with the same checks in the same order.
173
+ fn decode_speed_chunk_into(
174
+ data: &[u8],
175
+ header_start: usize,
176
+ chunk_length: i32,
177
+ marker: i32,
178
+ out: &mut [u8],
179
+ ) -> PyResult<()> {
180
+ let payload_offset = header_start + 12;
181
+ match marker {
182
+ MARKER_STORED_RAW => {
183
+ if chunk_length as i64 - 12 != out.len() as i64 {
184
+ return Err(PyValueError::new_err(format!(
185
+ "decode_speed_blob: stored-raw chunk should have chunk_length-12 == \
186
+ decompressed_length (got chunk_length-12={}, decompressed_length={})",
187
+ chunk_length as i64 - 12,
188
+ out.len(),
189
+ )));
190
+ }
191
+ let payload = data
192
+ .get(payload_offset..payload_offset + out.len())
193
+ .ok_or_else(|| {
194
+ PyIndexError::new_err(
195
+ "decode_speed_blob: truncated stored-raw payload -- file cut off mid-chunk?",
196
+ )
197
+ })?;
198
+ out.copy_from_slice(payload);
199
+ Ok(())
200
+ }
201
+ MARKER_COMPRESSED => lzrw1_decompress_impl(data, payload_offset, out),
202
+ other => Err(PyValueError::new_err(format!(
203
+ "decode_speed_blob: unrecognized marker value: {other}"
204
+ ))),
205
+ }
206
+ }
207
+
208
+ /// Walk a whole `DB_COMP_SPEED` blob's chain of chunks into `out`.
209
+ ///
210
+ /// `out.len()` is either the first chunk's `decompressed_length` (a
211
+ /// single-chunk blob, or no usable total) or the blob's declared total; the
212
+ /// loop stops as soon as `out` is full, and errors if a chunk would
213
+ /// overshoot it -- matching `pygdb.lzrw1._decode_speed_blob_py`.
214
+ fn decode_speed_chain_into(data: &[u8], out: &mut [u8]) -> PyResult<()> {
215
+ let mut written = 0usize;
216
+ let mut header_start = 16usize; // just past the first chunk's 16-byte magic
217
+ loop {
218
+ let (decompressed_length, chunk_length, marker) = read_chunk_subheader(data, header_start)?;
219
+ if !(0 < decompressed_length && decompressed_length < 200_000_000) {
220
+ return Err(PyValueError::new_err(format!(
221
+ "decode_speed_blob: implausible decompressed_length={decompressed_length} -- \
222
+ likely a misaligned or corrupt chunk header"
223
+ )));
224
+ }
225
+ let end = written + decompressed_length as usize;
226
+ if end > out.len() {
227
+ return Err(PyValueError::new_err(
228
+ "decode_speed_blob: chunks decode to more bytes than the blob header declares",
229
+ ));
230
+ }
231
+ decode_speed_chunk_into(
232
+ data,
233
+ header_start,
234
+ chunk_length,
235
+ marker,
236
+ &mut out[written..end],
237
+ )?;
238
+ written = end;
239
+ if written == out.len() {
240
+ return Ok(());
241
+ }
242
+ if chunk_length < 12 {
243
+ return Err(PyValueError::new_err(format!(
244
+ "decode_speed_blob: implausible chunk_length={chunk_length} -- corrupt chunk chain"
245
+ )));
246
+ }
247
+ header_start += chunk_length as usize;
248
+ }
249
+ }
250
+
251
+ /// Decode every chunk of a `DB_COMP_SPEED` blob into one writable Python
252
+ /// `bytearray`.
253
+ ///
254
+ /// Rust port of `pygdb.lzrw1.decode_speed_blob` -- see that module's
255
+ /// docstring (point 4) and docs/spec.md section 7.3/7.4 for the format:
256
+ /// a blob is a chain of chunks of at most 16368 decompressed bytes each,
257
+ /// only the first preceded by the 16-byte magic; `data` starts at that
258
+ /// magic, and `total_decompressed_length` is the blob header's `+24`
259
+ /// field (how the decoder knows the chain has ended -- the bytes after
260
+ /// the last chunk are page padding, not zeros). A total that isn't
261
+ /// larger than the first chunk (including 0 or negative, "no usable
262
+ /// total") decodes just the first chunk.
263
+ ///
264
+ /// Raises `ValueError` for a malformed chunk or chain (bad marker,
265
+ /// implausible lengths, a total the chain doesn't add up to) and
266
+ /// `IndexError` for truncated data -- `pygdb.lzrw1.decode_speed_blob`
267
+ /// turns both into `LZRW1DecodeError`, so callers never see which
268
+ /// backend produced the failure.
269
+ ///
270
+ /// The output buffer is sized once, up front, from the first chunk's
271
+ /// header and the declared total, then filled in place via
272
+ /// `PyByteArray::new_with` under `Python::detach` -- the same single-
273
+ /// allocation technique and the same GIL-release argument as
274
+ /// `lzrw1_decompress` (`data` is `&[u8]`, so an immutable `bytes`; the
275
+ /// target `bytearray` isn't Python-visible until this returns). A
276
+ /// declared total larger than any real chain could produce from `data`
277
+ /// (LZRW1 expands at most 8x, plus the 12-byte sub-headers) is rejected
278
+ /// before allocating, so a corrupt header can't request a huge buffer.
279
+ #[pyfunction]
280
+ fn decode_speed_blob<'py>(
281
+ py: Python<'py>,
282
+ data: &[u8],
283
+ total_decompressed_length: i64,
284
+ ) -> PyResult<Bound<'py, PyByteArray>> {
285
+ let subtype = data
286
+ .get(8..12)
287
+ .map(|b| i32::from_le_bytes([b[0], b[1], b[2], b[3]]))
288
+ .ok_or_else(|| PyIndexError::new_err("decode_speed_blob: truncated -- no chunk header"))?;
289
+ if subtype != DB_COMP_SPEED {
290
+ return Err(PyValueError::new_err(format!(
291
+ "decode_speed_blob: not a Speed chunk (subtype={subtype})"
292
+ )));
293
+ }
294
+ let (first_length, _, _) = read_chunk_subheader(data, 16)?;
295
+ if !(0 < first_length && first_length < 200_000_000) {
296
+ return Err(PyValueError::new_err(format!(
297
+ "decode_speed_blob: implausible decompressed_length={first_length} -- \
298
+ likely a misaligned or corrupt chunk header"
299
+ )));
300
+ }
301
+ let first_length = first_length as usize;
302
+ let out_len = if total_decompressed_length > first_length as i64 {
303
+ let total = total_decompressed_length as u64;
304
+ if total > (data.len() as u64).saturating_mul(9) {
305
+ return Err(PyValueError::new_err(format!(
306
+ "decode_speed_blob: blob header declares {total} decompressed byte(s), \
307
+ more than {} byte(s) of chunk data could possibly produce",
308
+ data.len(),
309
+ )));
310
+ }
311
+ total as usize
312
+ } else {
313
+ first_length
314
+ };
315
+ PyByteArray::new_with(py, out_len, |buf| {
316
+ py.detach(|| decode_speed_chain_into(data, buf))
317
+ })
318
+ }
319
+
151
320
  /// Decompress a `DB_COMP_SIZE` blob's raw zlib/DEFLATE stream into a
152
321
  /// writable Python `bytearray`.
153
322
  ///
@@ -483,6 +652,7 @@ fn decode_fixed_width_strings_ucs4<'py>(
483
652
  fn _native(m: &Bound<'_, PyModule>) -> PyResult<()> {
484
653
  m.add_function(wrap_pyfunction!(ping, m)?)?;
485
654
  m.add_function(wrap_pyfunction!(lzrw1_decompress, m)?)?;
655
+ m.add_function(wrap_pyfunction!(decode_speed_blob, m)?)?;
486
656
  m.add_function(wrap_pyfunction!(zlib_decompress, m)?)?;
487
657
  m.add_function(wrap_pyfunction!(decompress_grd_blocks, m)?)?;
488
658
  m.add_function(wrap_pyfunction!(decode_fixed_width_strings_ucs4, m)?)?;
File without changes
File without changes
File without changes
File without changes