python-gdb 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {python_gdb-0.2.0 → python_gdb-0.2.1}/PKG-INFO +2 -1
- {python_gdb-0.2.0 → python_gdb-0.2.1}/README.md +1 -0
- {python_gdb-0.2.0 → python_gdb-0.2.1}/pygdb/gdb.py +314 -5
- {python_gdb-0.2.0 → python_gdb-0.2.1}/pygdb/gdb_reader.py +27 -12
- {python_gdb-0.2.0 → python_gdb-0.2.1}/pygdb/lzrw1.py +147 -5
- {python_gdb-0.2.0 → python_gdb-0.2.1}/rust/Cargo.lock +1 -1
- {python_gdb-0.2.0 → python_gdb-0.2.1}/rust/Cargo.toml +1 -1
- {python_gdb-0.2.0 → python_gdb-0.2.1}/rust/src/lib.rs +170 -0
- {python_gdb-0.2.0 → python_gdb-0.2.1}/LICENSE +0 -0
- {python_gdb-0.2.0 → python_gdb-0.2.1}/pygdb/__init__.py +0 -0
- {python_gdb-0.2.0 → python_gdb-0.2.1}/pygdb/grd_reader.py +0 -0
- {python_gdb-0.2.0 → python_gdb-0.2.1}/pygdb/registry.py +0 -0
- {python_gdb-0.2.0 → python_gdb-0.2.1}/pyproject.toml +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: python-gdb
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Classifier: Development Status :: 3 - Alpha
|
|
5
5
|
Classifier: Intended Audience :: Science/Research
|
|
6
6
|
Classifier: Topic :: Scientific/Engineering :: GIS
|
|
@@ -190,6 +190,7 @@ zensical serve
|
|
|
190
190
|
boundary on reverse engineering.
|
|
191
191
|
- [`docs/provenance/`](docs/provenance/index.md) — the original
|
|
192
192
|
research log and write-up this implementation was derived from.
|
|
193
|
+
- [`CHANGELOG.md`](CHANGELOG.md) — what changed in each release.
|
|
193
194
|
|
|
194
195
|
## Testing
|
|
195
196
|
|
|
@@ -152,6 +152,7 @@ zensical serve
|
|
|
152
152
|
boundary on reverse engineering.
|
|
153
153
|
- [`docs/provenance/`](docs/provenance/index.md) — the original
|
|
154
154
|
research log and write-up this implementation was derived from.
|
|
155
|
+
- [`CHANGELOG.md`](CHANGELOG.md) — what changed in each release.
|
|
155
156
|
|
|
156
157
|
## Testing
|
|
157
158
|
|
|
@@ -16,7 +16,7 @@ from __future__ import annotations
|
|
|
16
16
|
import warnings
|
|
17
17
|
from dataclasses import dataclass
|
|
18
18
|
from pathlib import Path
|
|
19
|
-
from typing import Dict, Iterator, List, Optional, Tuple, Union
|
|
19
|
+
from typing import Dict, Iterator, List, Optional, Sequence, Tuple, Union
|
|
20
20
|
|
|
21
21
|
import numpy as np
|
|
22
22
|
|
|
@@ -60,6 +60,87 @@ LineRef = Union[str, Tuple[str, int], "LineRecord"]
|
|
|
60
60
|
ChannelRef = Union[str, Tuple[str, int], "ChannelRecord"]
|
|
61
61
|
|
|
62
62
|
|
|
63
|
+
def _finite_values(values: np.ndarray) -> np.ndarray:
|
|
64
|
+
v = np.asarray(values, dtype=float).ravel()
|
|
65
|
+
return v[np.isfinite(v) & (np.abs(v) < 1e30)]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _roughness(values: np.ndarray) -> Optional[float]:
|
|
69
|
+
"""
|
|
70
|
+
Median row-to-row step divided by the 5-95% spread of the values.
|
|
71
|
+
|
|
72
|
+
Returns
|
|
73
|
+
-------
|
|
74
|
+
float or None
|
|
75
|
+
Small for data that varies smoothly along its rows, large for
|
|
76
|
+
the same values in a scrambled order. None if there are too few
|
|
77
|
+
finite values or they are constant.
|
|
78
|
+
"""
|
|
79
|
+
v = _finite_values(values)
|
|
80
|
+
if len(v) < 50:
|
|
81
|
+
return None
|
|
82
|
+
spread = float(np.percentile(v, 95) - np.percentile(v, 5))
|
|
83
|
+
if spread == 0.0:
|
|
84
|
+
return None
|
|
85
|
+
return float(np.median(np.abs(np.diff(v))) / spread)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _row_order_pick(first: np.ndarray, last: np.ndarray, factor: float = 3.0) -> Optional[int]:
|
|
89
|
+
"""
|
|
90
|
+
Decide which of two copies of one channel is in acquisition order.
|
|
91
|
+
|
|
92
|
+
Parameters
|
|
93
|
+
----------
|
|
94
|
+
first, last : numpy.ndarray
|
|
95
|
+
The earlier and later copy in the blob chain.
|
|
96
|
+
factor : float, optional
|
|
97
|
+
How many times rougher one copy must be than the other.
|
|
98
|
+
|
|
99
|
+
Returns
|
|
100
|
+
-------
|
|
101
|
+
int or None
|
|
102
|
+
0 if the *first* copy is the smooth one and the last is a
|
|
103
|
+
markedly rougher reordering of it; 1 if the last is the smooth
|
|
104
|
+
one; None if the copies are not a pure reordering of each other
|
|
105
|
+
(different values or length) or neither is clearly smoother.
|
|
106
|
+
|
|
107
|
+
Notes
|
|
108
|
+
-----
|
|
109
|
+
A copy that is *itself* perfectly monotone is never judged: a sorted
|
|
110
|
+
ramp is smoother than any real signal, and a re-sort by a channel's
|
|
111
|
+
own value (a coordinate re-sorted by itself) leaves exactly that. It
|
|
112
|
+
could equally be a genuine ID or time channel, and nothing here can
|
|
113
|
+
tell the two apart, so such a pair is undecided.
|
|
114
|
+
"""
|
|
115
|
+
if len(first) != len(last):
|
|
116
|
+
return None
|
|
117
|
+
a, b = np.asarray(first).ravel(), np.asarray(last).ravel()
|
|
118
|
+
try:
|
|
119
|
+
if not np.array_equal(np.sort(a), np.sort(b), equal_nan=True):
|
|
120
|
+
return None
|
|
121
|
+
except TypeError: # dtype without a NaN notion (integers)
|
|
122
|
+
if not np.array_equal(np.sort(a), np.sort(b)):
|
|
123
|
+
return None
|
|
124
|
+
if _is_monotone_reference(a) or _is_monotone_reference(b):
|
|
125
|
+
return None
|
|
126
|
+
ra, rb = _roughness(a), _roughness(b)
|
|
127
|
+
if ra is None or rb is None:
|
|
128
|
+
return None
|
|
129
|
+
if rb > 0 and rb >= factor * ra:
|
|
130
|
+
return 0
|
|
131
|
+
if ra > 0 and ra >= factor * rb:
|
|
132
|
+
return 1
|
|
133
|
+
return None
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _is_monotone_reference(values: np.ndarray) -> bool:
|
|
137
|
+
"""A non-constant channel stored in non-decreasing order (an ID, date or time)."""
|
|
138
|
+
v = _finite_values(values)
|
|
139
|
+
if len(v) < 50 or v[0] == v[-1]:
|
|
140
|
+
return False
|
|
141
|
+
return bool(np.mean(np.diff(v) >= 0) >= 0.999)
|
|
142
|
+
|
|
143
|
+
|
|
63
144
|
@dataclass
|
|
64
145
|
class CompressionInfo:
|
|
65
146
|
"""
|
|
@@ -100,6 +181,20 @@ class GDB:
|
|
|
100
181
|
----------
|
|
101
182
|
path : str
|
|
102
183
|
Path to the `.gdb` file.
|
|
184
|
+
duplicate_blobs : {"last", "row_order"}, optional
|
|
185
|
+
What to do when a (line, channel) has more than one blob in the
|
|
186
|
+
blob chain (issue #2). `"last"` (the default) uses the last one
|
|
187
|
+
in chain order. `"row_order"` is an opt-in **heuristic**, not a
|
|
188
|
+
decoded field: for a duplicated numeric channel whose two
|
|
189
|
+
copies hold exactly the same values in a different row order
|
|
190
|
+
(a stale re-sorted copy), on a line that has an order-defining
|
|
191
|
+
channel (see Notes), it prefers the copy whose values vary
|
|
192
|
+
smoothly along the rows -- acquisition order, the order the
|
|
193
|
+
line's ID/time channels are stored in -- and keeps the last
|
|
194
|
+
copy in every other case. Either way one `GDBParseWarning`
|
|
195
|
+
names the affected pairs and, for `"row_order"`, which ones it
|
|
196
|
+
overrode. Choosing needs both copies decoded, so it is done
|
|
197
|
+
once, when the blob index is first built.
|
|
103
198
|
|
|
104
199
|
Raises
|
|
105
200
|
------
|
|
@@ -108,7 +203,8 @@ class GDB:
|
|
|
108
203
|
expected `.gdb` magic -- unlike the module-level functions in
|
|
109
204
|
`gdb_reader`/`registry` (which warn and return empty results),
|
|
110
205
|
since a `GDB` object that isn't backed by a real `.gdb` file
|
|
111
|
-
can't usefully do anything at all.
|
|
206
|
+
can't usefully do anything at all. Also if `duplicate_blobs` is
|
|
207
|
+
not one of the values above.
|
|
112
208
|
|
|
113
209
|
Examples
|
|
114
210
|
--------
|
|
@@ -137,10 +233,32 @@ class GDB:
|
|
|
137
233
|
the Rust-plan's M4 notes). Close it (`db.close()`, or use `GDB` as a
|
|
138
234
|
context manager) when done with it, or just let it get
|
|
139
235
|
garbage-collected -- `__del__` closes it too, as a safety net.
|
|
236
|
+
|
|
237
|
+
`duplicate_blobs="row_order"` only ever chooses between two copies
|
|
238
|
+
of one channel that are a pure reordering of each other, and only on
|
|
239
|
+
a line with an *order-defining channel*: a single-copy numeric
|
|
240
|
+
channel of the same length stored in monotone order (typically an
|
|
241
|
+
ID, date or time). A revised copy (different values) is never
|
|
242
|
+
second-guessed, and neither is a pair with a copy that is itself
|
|
243
|
+
perfectly monotone (a re-sort by a channel's own value leaves a
|
|
244
|
+
smooth ramp that looks like the best copy but is the stale one).
|
|
245
|
+
Among a qualifying pair, the copy at least 3x
|
|
246
|
+
rougher along the rows -- median row-to-row step over the 5-95%
|
|
247
|
+
spread of the values -- is treated as the stale one, since data in
|
|
248
|
+
acquisition order varies smoothly and a re-sort scrambles that. This
|
|
249
|
+
was validated against independent spreadsheet exports of the one
|
|
250
|
+
real file known to have such copies (it never contradicted them),
|
|
251
|
+
but it cannot decide a channel that is smooth in both orders, and no
|
|
252
|
+
on-disk marker has been found that would make it unnecessary.
|
|
140
253
|
"""
|
|
141
254
|
|
|
142
|
-
def __init__(self, path: str):
|
|
255
|
+
def __init__(self, path: str, duplicate_blobs: str = "last"):
|
|
256
|
+
if duplicate_blobs not in ("last", "row_order"):
|
|
257
|
+
raise ValueError(
|
|
258
|
+
f"duplicate_blobs must be 'last' or 'row_order', got {duplicate_blobs!r}"
|
|
259
|
+
)
|
|
143
260
|
self.path = path
|
|
261
|
+
self.duplicate_blobs = duplicate_blobs
|
|
144
262
|
self._file = open(path, "rb")
|
|
145
263
|
header = self._file.read(4096)
|
|
146
264
|
if not check_magic(header):
|
|
@@ -435,16 +553,207 @@ class GDB:
|
|
|
435
553
|
anything beyond an occasional one-off lookup -- this class
|
|
436
554
|
always wants line/channel listings and random-access
|
|
437
555
|
reads, so it always builds the index.
|
|
556
|
+
|
|
557
|
+
Warns
|
|
558
|
+
-----
|
|
559
|
+
GDBParseWarning
|
|
560
|
+
If a real line's channel has more than one blob in the
|
|
561
|
+
blob chain. By default the **last** one in chain order is
|
|
562
|
+
used, but that is not always the current copy (issue #2):
|
|
563
|
+
an older copy with the same values in a different row order
|
|
564
|
+
can sit either before or after the current one, and nothing
|
|
565
|
+
decoded so far says which is which. With
|
|
566
|
+
`duplicate_blobs="row_order"` the warning also says which
|
|
567
|
+
pairs were switched to an earlier copy. Duplicates in the
|
|
568
|
+
administrative slots past the last real line (the REG/IPJ
|
|
569
|
+
registry, whose stale copies are expected and handled by
|
|
570
|
+
`pygdb.registry`) are not reported.
|
|
438
571
|
"""
|
|
439
572
|
if self._blob_index is None:
|
|
440
573
|
chans_max = self.chans_max
|
|
441
|
-
|
|
574
|
+
copies: Dict[Tuple[int, int], List[BlobHeader]] = {}
|
|
442
575
|
for blob in iter_blobs(self.path):
|
|
443
|
-
|
|
576
|
+
copies.setdefault(blob.line_channel(chans_max), []).append(blob)
|
|
577
|
+
index: Dict[Tuple[int, int], BlobHeader] = {k: v[-1] for k, v in copies.items()}
|
|
578
|
+
duplicated = [k for k, v in copies.items() if len(v) > 1]
|
|
444
579
|
self._blob_index = index
|
|
445
580
|
self._calibrate_line_indices()
|
|
581
|
+
if duplicated:
|
|
582
|
+
overridden: List[Tuple[int, int]] = []
|
|
583
|
+
if self.duplicate_blobs == "row_order":
|
|
584
|
+
overridden = self._prefer_row_order_copies(duplicated, copies, index)
|
|
585
|
+
self._warn_duplicate_blobs(duplicated, overridden)
|
|
446
586
|
return self._blob_index
|
|
447
587
|
|
|
588
|
+
def _read_copy(self, blob: BlobHeader, channel: ChannelRecord) -> np.ndarray:
|
|
589
|
+
return read_blob_values(
|
|
590
|
+
self.path, blob, channel,
|
|
591
|
+
comp_level=self.comp_level or 0, page_size=self.page_size, file=self._file,
|
|
592
|
+
)
|
|
593
|
+
|
|
594
|
+
def _prefer_row_order_copies(
|
|
595
|
+
self,
|
|
596
|
+
duplicated: List[Tuple[int, int]],
|
|
597
|
+
copies: Dict[Tuple[int, int], List[BlobHeader]],
|
|
598
|
+
index: Dict[Tuple[int, int], BlobHeader],
|
|
599
|
+
) -> List[Tuple[int, int]]:
|
|
600
|
+
"""
|
|
601
|
+
Apply `duplicate_blobs="row_order"` to the duplicated pairs.
|
|
602
|
+
|
|
603
|
+
Parameters
|
|
604
|
+
----------
|
|
605
|
+
duplicated : list of (int, int)
|
|
606
|
+
`(line_slot, channel_slot)` keys with more than one blob.
|
|
607
|
+
copies : dict of {(int, int) : list of BlobHeader}
|
|
608
|
+
Every blob for each key, in chain order.
|
|
609
|
+
index : dict of {(int, int) : BlobHeader}
|
|
610
|
+
The blob index being built; updated in place with the
|
|
611
|
+
earlier copy wherever one is preferred.
|
|
612
|
+
|
|
613
|
+
Returns
|
|
614
|
+
-------
|
|
615
|
+
list of (int, int)
|
|
616
|
+
The keys switched from the last copy to an earlier one.
|
|
617
|
+
|
|
618
|
+
Notes
|
|
619
|
+
-----
|
|
620
|
+
See the class docstring for exactly when a pair qualifies. A pair
|
|
621
|
+
that doesn't (a string or array channel, more than two copies,
|
|
622
|
+
different values, no order-defining channel on the line, or no
|
|
623
|
+
clear winner) simply keeps the last copy.
|
|
624
|
+
"""
|
|
625
|
+
real_lines = {line.index for line in self.lines}
|
|
626
|
+
channels = {c.index: c for c in self.channels}
|
|
627
|
+
duplicated_keys = set(duplicated)
|
|
628
|
+
reference_cache: Dict[Tuple[int, int], bool] = {}
|
|
629
|
+
overridden: List[Tuple[int, int]] = []
|
|
630
|
+
for key in duplicated:
|
|
631
|
+
line_slot, channel_slot = key
|
|
632
|
+
channel = channels.get(channel_slot)
|
|
633
|
+
blobs = copies[key]
|
|
634
|
+
if (
|
|
635
|
+
line_slot not in real_lines or channel is None or len(blobs) != 2
|
|
636
|
+
or channel.is_string or channel.is_array
|
|
637
|
+
):
|
|
638
|
+
continue
|
|
639
|
+
first, last = self._read_copy(blobs[0], channel), self._read_copy(blobs[1], channel)
|
|
640
|
+
if len(first) != len(last):
|
|
641
|
+
continue
|
|
642
|
+
ref_key = (line_slot, len(first))
|
|
643
|
+
if ref_key not in reference_cache:
|
|
644
|
+
reference_cache[ref_key] = self._line_has_order_reference(
|
|
645
|
+
line_slot, len(first), duplicated_keys, channels
|
|
646
|
+
)
|
|
647
|
+
if reference_cache[ref_key] and _row_order_pick(first, last) == 0:
|
|
648
|
+
index[key] = blobs[0]
|
|
649
|
+
overridden.append(key)
|
|
650
|
+
return overridden
|
|
651
|
+
|
|
652
|
+
def _line_has_order_reference(
|
|
653
|
+
self,
|
|
654
|
+
line_slot: int,
|
|
655
|
+
n_rows: int,
|
|
656
|
+
duplicated_keys: set,
|
|
657
|
+
channels: Dict[int, ChannelRecord],
|
|
658
|
+
) -> bool:
|
|
659
|
+
"""
|
|
660
|
+
Whether a line has a channel proving its rows are in some fixed order.
|
|
661
|
+
|
|
662
|
+
Parameters
|
|
663
|
+
----------
|
|
664
|
+
line_slot : int
|
|
665
|
+
The line's slot.
|
|
666
|
+
n_rows : int
|
|
667
|
+
The row count of the duplicated copies being judged.
|
|
668
|
+
duplicated_keys : set of (int, int)
|
|
669
|
+
Keys with more than one blob -- excluded, since their own
|
|
670
|
+
order is what is in question.
|
|
671
|
+
channels : dict of {int : ChannelRecord}
|
|
672
|
+
Channel records by slot.
|
|
673
|
+
|
|
674
|
+
Returns
|
|
675
|
+
-------
|
|
676
|
+
bool
|
|
677
|
+
True if some single-copy, numeric, non-array channel of
|
|
678
|
+
`n_rows` values on this line is stored in monotone
|
|
679
|
+
(non-decreasing) order and isn't constant -- typically an
|
|
680
|
+
ID, date or time channel.
|
|
681
|
+
"""
|
|
682
|
+
for (ls, cs), blob in sorted(self._ensure_blob_index().items()):
|
|
683
|
+
channel = channels.get(cs)
|
|
684
|
+
if (
|
|
685
|
+
ls != line_slot or (ls, cs) in duplicated_keys or channel is None
|
|
686
|
+
or channel.is_string or channel.is_array
|
|
687
|
+
):
|
|
688
|
+
continue
|
|
689
|
+
values = self._read_copy(blob, channel)
|
|
690
|
+
if len(values) == n_rows and _is_monotone_reference(values):
|
|
691
|
+
return True
|
|
692
|
+
return False
|
|
693
|
+
|
|
694
|
+
def _warn_duplicate_blobs(
|
|
695
|
+
self,
|
|
696
|
+
duplicated: List[Tuple[int, int]],
|
|
697
|
+
overridden: Sequence[Tuple[int, int]] = (),
|
|
698
|
+
) -> None:
|
|
699
|
+
"""
|
|
700
|
+
Warn about (line, channel) pairs that have more than one blob.
|
|
701
|
+
|
|
702
|
+
Parameters
|
|
703
|
+
----------
|
|
704
|
+
duplicated : list of (int, int)
|
|
705
|
+
`(line_slot, channel_slot)` keys seen more than once in the
|
|
706
|
+
blob chain.
|
|
707
|
+
overridden : sequence of (int, int), optional
|
|
708
|
+
The keys `duplicate_blobs="row_order"` switched to an
|
|
709
|
+
earlier copy.
|
|
710
|
+
|
|
711
|
+
Warns
|
|
712
|
+
-----
|
|
713
|
+
GDBParseWarning
|
|
714
|
+
Once, naming how many real (line, channel) pairs are
|
|
715
|
+
affected and a few examples (and, for `"row_order"`, which
|
|
716
|
+
ones were switched). Nothing is emitted if every duplicate
|
|
717
|
+
is in an administrative slot.
|
|
718
|
+
"""
|
|
719
|
+
line_names = {line.index: line.name for line in self.lines}
|
|
720
|
+
channel_names = {c.index: c.name for c in self.channels}
|
|
721
|
+
|
|
722
|
+
def describe(keys, limit=3):
|
|
723
|
+
named = [
|
|
724
|
+
f"line {line_names[ls]!r} channel {channel_names.get(cs, f'#{cs}')!r}"
|
|
725
|
+
for ls, cs in keys if ls in line_names
|
|
726
|
+
]
|
|
727
|
+
more = f" and {len(named) - limit} more" if len(named) > limit else ""
|
|
728
|
+
return ", ".join(named[:limit]) + more
|
|
729
|
+
|
|
730
|
+
n_real = sum(1 for ls, _ in duplicated if ls in line_names)
|
|
731
|
+
if not n_real:
|
|
732
|
+
return
|
|
733
|
+
head = (
|
|
734
|
+
f"{self.path}: {n_real} (line, channel) pair(s) have more than one "
|
|
735
|
+
f"blob in the blob chain ({describe(duplicated)})"
|
|
736
|
+
)
|
|
737
|
+
if self.duplicate_blobs == "row_order":
|
|
738
|
+
tail = (
|
|
739
|
+
f" -- duplicate_blobs='row_order': switched {len(overridden)} pair(s) to "
|
|
740
|
+
f"an earlier copy because the last copy was the same values in a "
|
|
741
|
+
f"markedly rougher row order ({describe(overridden)}); kept the last "
|
|
742
|
+
f"copy in chain order for the other {n_real - len(overridden)}. This "
|
|
743
|
+
f"is a heuristic, not a decoded field (see issue #2)"
|
|
744
|
+
if overridden else
|
|
745
|
+
f" -- duplicate_blobs='row_order': no pair needed switching, so the "
|
|
746
|
+
f"last copy in chain order was kept for all of them (see issue #2)"
|
|
747
|
+
)
|
|
748
|
+
else:
|
|
749
|
+
tail = (
|
|
750
|
+
" -- using the last one in chain order, which is not always the "
|
|
751
|
+
"current copy. If a channel's rows look scrambled against the line's "
|
|
752
|
+
"other channels, this is the likely cause; duplicate_blobs='row_order' "
|
|
753
|
+
"can pick the acquisition-order copy (see issue #2)"
|
|
754
|
+
)
|
|
755
|
+
warnings.warn(head + tail, GDBParseWarning, stacklevel=4)
|
|
756
|
+
|
|
448
757
|
def _calibrate_line_indices(self) -> None:
|
|
449
758
|
"""
|
|
450
759
|
Correct a possible small, fixed off-by-N in every line's index.
|
|
@@ -1682,11 +1682,16 @@ def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
|
|
|
1682
1682
|
established ground truth exactly) and for both single- and
|
|
1683
1683
|
multi-page DB_COMP_SPEED (a real 2-page `Northing_AMGz55` blob
|
|
1684
1684
|
decodes to sane real coordinates with real `rDUMMY` sentinels).
|
|
1685
|
-
See docs/provenance/notes.md section 6.6d: a multi-page blob
|
|
1686
|
-
|
|
1687
|
-
|
|
1688
|
-
|
|
1689
|
-
|
|
1685
|
+
See docs/provenance/notes.md section 6.6d: a multi-page blob has no
|
|
1686
|
+
per-page re-framing -- just read the full `n_pages*page_size` span
|
|
1687
|
+
instead of one page. **A DB_COMP_SPEED blob is, however, a chain of
|
|
1688
|
+
chunks of at most 16368 decompressed bytes each, not one chunk**
|
|
1689
|
+
(docs/provenance/notes.md section 6.6e): only the first carries the
|
|
1690
|
+
16-byte magic, later ones are a bare 12-byte sub-header plus
|
|
1691
|
+
payload, and the blob header's `+24` field is the total
|
|
1692
|
+
decompressed size across the chain. Decoding only the first chunk
|
|
1693
|
+
-- as this function once did -- silently truncated any channel
|
|
1694
|
+
longer than 2046 float64 values on a line.
|
|
1690
1695
|
|
|
1691
1696
|
**A real third on-disk variant, auto-detected here rather than
|
|
1692
1697
|
assumed away (docs/provenance/notes.md section 6.6b):** even
|
|
@@ -1769,11 +1774,11 @@ def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
|
|
|
1769
1774
|
# to be a SINGLE continuous compressed stream spanning the whole
|
|
1770
1775
|
# n_pages*page_size span -- NOT one independently-framed chunk per
|
|
1771
1776
|
# page. There is no per-page re-framing to handle: reading the full
|
|
1772
|
-
# span
|
|
1773
|
-
#
|
|
1774
|
-
#
|
|
1775
|
-
#
|
|
1776
|
-
#
|
|
1777
|
+
# span is sufficient for zlib (zlib.decompressobj() naturally stops
|
|
1778
|
+
# at the real end of stream and reports the rest as padding, and its
|
|
1779
|
+
# single stream always matches the blob header's `+24` total). LZRW1
|
|
1780
|
+
# is different: the span holds a chain of chunks, not one -- see the
|
|
1781
|
+
# DB_COMP_SPEED branch below.
|
|
1777
1782
|
if page_size is None:
|
|
1778
1783
|
with _file_handle(path, file) as f:
|
|
1779
1784
|
header = f.read(128)
|
|
@@ -1838,9 +1843,19 @@ def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
|
|
|
1838
1843
|
)
|
|
1839
1844
|
return np.array([])
|
|
1840
1845
|
elif subtype == _lzrw1.DB_COMP_SPEED:
|
|
1846
|
+
# A blob is a chain of chunks of at most 16368 decompressed bytes
|
|
1847
|
+
# each (docs/spec.md section 7.3); the blob header's `+24` field
|
|
1848
|
+
# is the total across all of them, and is what tells the decoder
|
|
1849
|
+
# where the chain ends (the bytes after the last chunk are page
|
|
1850
|
+
# padding, not a reliable terminator).
|
|
1851
|
+
with _file_handle(path, file) as f:
|
|
1852
|
+
f.seek(blob.offset + 24)
|
|
1853
|
+
total_field = f.read(4)
|
|
1854
|
+
total_decompressed = (
|
|
1855
|
+
struct.unpack("<i", total_field)[0] if len(total_field) == 4 else 0
|
|
1856
|
+
)
|
|
1841
1857
|
try:
|
|
1842
|
-
|
|
1843
|
-
decompressed = _lzrw1.decode_speed_chunk(raw_span, chunk)
|
|
1858
|
+
decompressed = _lzrw1.decode_speed_blob(raw_span, total_decompressed)
|
|
1844
1859
|
except _lzrw1.LZRW1DecodeError as e:
|
|
1845
1860
|
_warn(
|
|
1846
1861
|
f"blob_index={blob.blob_index}: LZRW1 chunk decode failed ({e}) "
|
|
@@ -15,9 +15,10 @@ reference C wrapper:
|
|
|
15
15
|
1. No 4-byte FLAG_BYTES prefix (the reference C code's own
|
|
16
16
|
FLAG_COMPRESS/FLAG_COPY byte + 3 padding bytes) -- the control word
|
|
17
17
|
starts immediately for a compressed chunk.
|
|
18
|
-
2.
|
|
19
|
-
1024-byte pages) is preceded by a 28-byte Geosoft-specific
|
|
20
|
-
not part of LZRW1 itself
|
|
18
|
+
2. The first chunk of a blob (which may span several of the file's
|
|
19
|
+
physical 1024-byte pages) is preceded by a 28-byte Geosoft-specific
|
|
20
|
+
wrapper, not part of LZRW1 itself (see point 4 for what follows
|
|
21
|
+
it):
|
|
21
22
|
- 16 bytes: the magic sub-header shared with the `.grd` sibling
|
|
22
23
|
format and with `.gdb`'s DB_COMP_SIZE (zlib) mode:
|
|
23
24
|
`0f 0e ff fe 12 34 56 78 <subtype int32> <reserved int32>`
|
|
@@ -49,6 +50,21 @@ reference C wrapper:
|
|
|
49
50
|
values). Every real marker value found across all 10 real
|
|
50
51
|
Speed files was one of these two constants -- zero exceptions,
|
|
51
52
|
zero unrecognized third values.
|
|
53
|
+
4. **A blob is a chain of chunks, not one chunk** ([CONFIRMED] on
|
|
54
|
+
every Speed blob checked -- see docs/spec.md section 7.3/7.4). A
|
|
55
|
+
chunk decompresses to at most 16368 bytes (2046 float64 values);
|
|
56
|
+
a channel holding more data than that on one line is split across
|
|
57
|
+
several chunks stored back to back. Only the *first* chunk of a
|
|
58
|
+
blob carries the 16-byte magic; each later one is just its own
|
|
59
|
+
bare 12-byte `<decompressed_length> <chunk_length> <marker>`
|
|
60
|
+
sub-header immediately followed by its payload, starting
|
|
61
|
+
`chunk_length` bytes after the previous sub-header began. The
|
|
62
|
+
blob header (docs/spec.md section 7.4) records the total
|
|
63
|
+
decompressed size at `+24`, which is how a reader knows when to
|
|
64
|
+
stop -- the bytes after the last chunk are page padding, not
|
|
65
|
+
zeros, so they can't be relied on as a terminator. Every chunk
|
|
66
|
+
decoded independently (LZRW1 back-references never reach across
|
|
67
|
+
a chunk boundary). See `decode_speed_blob`.
|
|
52
68
|
|
|
53
69
|
Validated exactly (not just "plausibly") against **all 10 real**
|
|
54
70
|
DB_COMP_SPEED files now in this project's sample set (the original 4
|
|
@@ -72,6 +88,7 @@ from __future__ import annotations
|
|
|
72
88
|
import struct
|
|
73
89
|
import warnings
|
|
74
90
|
from dataclasses import dataclass
|
|
91
|
+
from typing import List, Optional
|
|
75
92
|
|
|
76
93
|
try:
|
|
77
94
|
from . import _native as _native_ext
|
|
@@ -208,7 +225,8 @@ def _lzrw1_decompress_py(data: bytes, start: int, decompressed_length: int) -> b
|
|
|
208
225
|
|
|
209
226
|
@dataclass
|
|
210
227
|
class SpeedChunk:
|
|
211
|
-
magic_offset: int
|
|
228
|
+
magic_offset: Optional[int] # offset of the 16-byte magic sub-header; None for a
|
|
229
|
+
# continuation chunk, which has no magic of its own
|
|
212
230
|
subtype: int # 1 = DB_COMP_SPEED, 2 = DB_COMP_SIZE
|
|
213
231
|
decompressed_length: int
|
|
214
232
|
chunk_length: int # includes the 12-byte length sub-header
|
|
@@ -361,11 +379,135 @@ def decode_speed_chunk(data: bytes, chunk: SpeedChunk):
|
|
|
361
379
|
) from e
|
|
362
380
|
|
|
363
381
|
|
|
382
|
+
def decode_speed_blob(data: bytes, total_decompressed_length: int = 0):
|
|
383
|
+
"""
|
|
384
|
+
Decode every chunk of a DB_COMP_SPEED blob, in order.
|
|
385
|
+
|
|
386
|
+
Parameters
|
|
387
|
+
----------
|
|
388
|
+
data : bytes or bytearray
|
|
389
|
+
The blob's compressed span, starting at the 16-byte magic of its
|
|
390
|
+
first chunk (i.e. everything after the blob header,
|
|
391
|
+
docs/spec.md section 7.4).
|
|
392
|
+
total_decompressed_length : int, optional
|
|
393
|
+
The blob's total decompressed size in bytes, from its header
|
|
394
|
+
(`+24`, docs/spec.md section 7.4). Decoding continues chunk by
|
|
395
|
+
chunk until this many bytes have been produced. If not positive
|
|
396
|
+
(a header that doesn't carry it, as in some hand-built
|
|
397
|
+
fixtures), only the first chunk is decoded.
|
|
398
|
+
|
|
399
|
+
Returns
|
|
400
|
+
-------
|
|
401
|
+
bytearray or bytes
|
|
402
|
+
The concatenated output of every chunk. A single-chunk blob
|
|
403
|
+
returns exactly what `decode_speed_chunk` does for it (no extra
|
|
404
|
+
copy); a multi-chunk one is a new, writable `bytearray`.
|
|
405
|
+
|
|
406
|
+
Raises
|
|
407
|
+
------
|
|
408
|
+
LZRW1DecodeError
|
|
409
|
+
If any chunk fails to decode (see `decode_speed_chunk`), the
|
|
410
|
+
chain runs off the end of `data`, or the chunks don't add up
|
|
411
|
+
to exactly `total_decompressed_length`.
|
|
412
|
+
|
|
413
|
+
Notes
|
|
414
|
+
-----
|
|
415
|
+
A blob is a chain of chunks of at most 16368 decompressed bytes
|
|
416
|
+
each, not a single chunk (module docstring point 4): only the first
|
|
417
|
+
carries the 16-byte magic, and every later one is a bare 12-byte
|
|
418
|
+
sub-header plus payload starting `chunk_length` bytes after the
|
|
419
|
+
previous sub-header began. This reader used to decode only the first
|
|
420
|
+
chunk, silently truncating any channel longer than 2046 float64
|
|
421
|
+
values on a line to exactly that length.
|
|
422
|
+
|
|
423
|
+
Dispatches to the compiled `pygdb._native` extension when it's
|
|
424
|
+
available and `data` is `bytes` (same algorithm, ported to Rust --
|
|
425
|
+
see `rust/src/lib.rs`), falling back to the pure-Python
|
|
426
|
+
`_decode_speed_blob_py` below otherwise. The native version raises
|
|
427
|
+
`ValueError`/`IndexError` for the same conditions; both are turned
|
|
428
|
+
into `LZRW1DecodeError` here, so callers never see which backend
|
|
429
|
+
produced a failure.
|
|
430
|
+
"""
|
|
431
|
+
if _native_ext is not None and isinstance(data, bytes):
|
|
432
|
+
try:
|
|
433
|
+
return _native_ext.decode_speed_blob(data, total_decompressed_length)
|
|
434
|
+
except (ValueError, IndexError) as e:
|
|
435
|
+
raise LZRW1DecodeError(str(e)) from e
|
|
436
|
+
return _decode_speed_blob_py(data, total_decompressed_length)
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def _decode_speed_blob_py(data: bytes, total_decompressed_length: int = 0):
|
|
440
|
+
"""
|
|
441
|
+
Pure-Python reference implementation of `decode_speed_blob`.
|
|
442
|
+
|
|
443
|
+
Parameters
|
|
444
|
+
----------
|
|
445
|
+
data : bytes or bytearray
|
|
446
|
+
See `decode_speed_blob`.
|
|
447
|
+
total_decompressed_length : int, optional
|
|
448
|
+
See `decode_speed_blob`.
|
|
449
|
+
|
|
450
|
+
Returns
|
|
451
|
+
-------
|
|
452
|
+
bytearray or bytes
|
|
453
|
+
See `decode_speed_blob`.
|
|
454
|
+
|
|
455
|
+
Raises
|
|
456
|
+
------
|
|
457
|
+
LZRW1DecodeError
|
|
458
|
+
See `decode_speed_blob`.
|
|
459
|
+
"""
|
|
460
|
+
first = parse_chunk_header(data, 0)
|
|
461
|
+
out = decode_speed_chunk(data, first)
|
|
462
|
+
if total_decompressed_length <= len(out):
|
|
463
|
+
return out
|
|
464
|
+
|
|
465
|
+
parts: List[bytes] = [out]
|
|
466
|
+
produced = len(out)
|
|
467
|
+
header_start = first.payload_offset - 12 # where this chunk's sub-header began
|
|
468
|
+
chunk_length = first.chunk_length
|
|
469
|
+
while produced < total_decompressed_length:
|
|
470
|
+
if chunk_length < 12:
|
|
471
|
+
raise LZRW1DecodeError(
|
|
472
|
+
f"implausible chunk_length={chunk_length} -- corrupt chunk chain"
|
|
473
|
+
)
|
|
474
|
+
header_start += chunk_length
|
|
475
|
+
try:
|
|
476
|
+
decompressed_length, chunk_length, marker = struct.unpack_from(
|
|
477
|
+
"<iii", data, header_start
|
|
478
|
+
)
|
|
479
|
+
except struct.error as e:
|
|
480
|
+
raise LZRW1DecodeError(
|
|
481
|
+
f"chunk chain runs off the end of the data after {produced} of "
|
|
482
|
+
f"{total_decompressed_length} byte(s) -- truncated data"
|
|
483
|
+
) from e
|
|
484
|
+
chunk = SpeedChunk(
|
|
485
|
+
magic_offset=None,
|
|
486
|
+
subtype=DB_COMP_SPEED,
|
|
487
|
+
decompressed_length=decompressed_length,
|
|
488
|
+
chunk_length=chunk_length,
|
|
489
|
+
marker=marker,
|
|
490
|
+
payload_offset=header_start + 12,
|
|
491
|
+
)
|
|
492
|
+
parts.append(decode_speed_chunk(data, chunk))
|
|
493
|
+
produced += decompressed_length
|
|
494
|
+
|
|
495
|
+
if produced != total_decompressed_length:
|
|
496
|
+
raise LZRW1DecodeError(
|
|
497
|
+
f"chunks decode to {produced} byte(s) but the blob header declares "
|
|
498
|
+
f"{total_decompressed_length}"
|
|
499
|
+
)
|
|
500
|
+
return bytearray().join(parts)
|
|
501
|
+
|
|
502
|
+
|
|
364
503
|
def find_speed_chunks(data: bytes):
|
|
365
504
|
"""
|
|
366
505
|
Yield every DB_COMP_SPEED (subtype==1) chunk found in `data`.
|
|
367
506
|
|
|
368
|
-
Scans for the shared 16-byte magic byte-by-byte.
|
|
507
|
+
Scans for the shared 16-byte magic byte-by-byte. Since only the
|
|
508
|
+
*first* chunk of a blob carries that magic (see `decode_speed_blob`),
|
|
509
|
+
this finds one chunk per blob, not every chunk -- it's a scanning
|
|
510
|
+
helper for locating blobs, not a way to decode them.
|
|
369
511
|
|
|
370
512
|
Parameters
|
|
371
513
|
----------
|
|
@@ -148,6 +148,175 @@ fn lzrw1_decompress_impl(data: &[u8], start: usize, out: &mut [u8]) -> PyResult<
|
|
|
148
148
|
Ok(())
|
|
149
149
|
}
|
|
150
150
|
|
|
151
|
+
const DB_COMP_SPEED: i32 = 1;
|
|
152
|
+
const MARKER_COMPRESSED: i32 = -186263865; // 0xF4E5D6C7 -- payload is real LZRW1
|
|
153
|
+
const MARKER_STORED_RAW: i32 = -253635901; // 0xF0E1D2C3 -- payload is stored verbatim
|
|
154
|
+
|
|
155
|
+
/// A chunk's 12-byte `<decompressed_length> <chunk_length> <marker>`
|
|
156
|
+
/// sub-header, read from `data[header_start..]`.
|
|
157
|
+
fn read_chunk_subheader(data: &[u8], header_start: usize) -> PyResult<(i32, i32, i32)> {
|
|
158
|
+
let bytes = header_start
|
|
159
|
+
.checked_add(12)
|
|
160
|
+
.and_then(|end| data.get(header_start..end))
|
|
161
|
+
.ok_or_else(|| {
|
|
162
|
+
PyIndexError::new_err(
|
|
163
|
+
"decode_speed_blob: chunk chain runs off the end of the data -- truncated",
|
|
164
|
+
)
|
|
165
|
+
})?;
|
|
166
|
+
let field = |i: usize| i32::from_le_bytes([bytes[i], bytes[i + 1], bytes[i + 2], bytes[i + 3]]);
|
|
167
|
+
Ok((field(0), field(4), field(8)))
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/// Decode one chunk's payload into `out` (exactly its `decompressed_length`
|
|
171
|
+
/// bytes) -- the Rust counterpart of `pygdb.lzrw1.decode_speed_chunk`,
|
|
172
|
+
/// with the same checks in the same order.
|
|
173
|
+
fn decode_speed_chunk_into(
|
|
174
|
+
data: &[u8],
|
|
175
|
+
header_start: usize,
|
|
176
|
+
chunk_length: i32,
|
|
177
|
+
marker: i32,
|
|
178
|
+
out: &mut [u8],
|
|
179
|
+
) -> PyResult<()> {
|
|
180
|
+
let payload_offset = header_start + 12;
|
|
181
|
+
match marker {
|
|
182
|
+
MARKER_STORED_RAW => {
|
|
183
|
+
if chunk_length as i64 - 12 != out.len() as i64 {
|
|
184
|
+
return Err(PyValueError::new_err(format!(
|
|
185
|
+
"decode_speed_blob: stored-raw chunk should have chunk_length-12 == \
|
|
186
|
+
decompressed_length (got chunk_length-12={}, decompressed_length={})",
|
|
187
|
+
chunk_length as i64 - 12,
|
|
188
|
+
out.len(),
|
|
189
|
+
)));
|
|
190
|
+
}
|
|
191
|
+
let payload = data
|
|
192
|
+
.get(payload_offset..payload_offset + out.len())
|
|
193
|
+
.ok_or_else(|| {
|
|
194
|
+
PyIndexError::new_err(
|
|
195
|
+
"decode_speed_blob: truncated stored-raw payload -- file cut off mid-chunk?",
|
|
196
|
+
)
|
|
197
|
+
})?;
|
|
198
|
+
out.copy_from_slice(payload);
|
|
199
|
+
Ok(())
|
|
200
|
+
}
|
|
201
|
+
MARKER_COMPRESSED => lzrw1_decompress_impl(data, payload_offset, out),
|
|
202
|
+
other => Err(PyValueError::new_err(format!(
|
|
203
|
+
"decode_speed_blob: unrecognized marker value: {other}"
|
|
204
|
+
))),
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/// Walk a whole `DB_COMP_SPEED` blob's chain of chunks into `out`.
|
|
209
|
+
///
|
|
210
|
+
/// `out.len()` is either the first chunk's `decompressed_length` (a
|
|
211
|
+
/// single-chunk blob, or no usable total) or the blob's declared total; the
|
|
212
|
+
/// loop stops as soon as `out` is full, and errors if a chunk would
|
|
213
|
+
/// overshoot it -- matching `pygdb.lzrw1._decode_speed_blob_py`.
|
|
214
|
+
fn decode_speed_chain_into(data: &[u8], out: &mut [u8]) -> PyResult<()> {
|
|
215
|
+
let mut written = 0usize;
|
|
216
|
+
let mut header_start = 16usize; // just past the first chunk's 16-byte magic
|
|
217
|
+
loop {
|
|
218
|
+
let (decompressed_length, chunk_length, marker) = read_chunk_subheader(data, header_start)?;
|
|
219
|
+
if !(0 < decompressed_length && decompressed_length < 200_000_000) {
|
|
220
|
+
return Err(PyValueError::new_err(format!(
|
|
221
|
+
"decode_speed_blob: implausible decompressed_length={decompressed_length} -- \
|
|
222
|
+
likely a misaligned or corrupt chunk header"
|
|
223
|
+
)));
|
|
224
|
+
}
|
|
225
|
+
let end = written + decompressed_length as usize;
|
|
226
|
+
if end > out.len() {
|
|
227
|
+
return Err(PyValueError::new_err(
|
|
228
|
+
"decode_speed_blob: chunks decode to more bytes than the blob header declares",
|
|
229
|
+
));
|
|
230
|
+
}
|
|
231
|
+
decode_speed_chunk_into(
|
|
232
|
+
data,
|
|
233
|
+
header_start,
|
|
234
|
+
chunk_length,
|
|
235
|
+
marker,
|
|
236
|
+
&mut out[written..end],
|
|
237
|
+
)?;
|
|
238
|
+
written = end;
|
|
239
|
+
if written == out.len() {
|
|
240
|
+
return Ok(());
|
|
241
|
+
}
|
|
242
|
+
if chunk_length < 12 {
|
|
243
|
+
return Err(PyValueError::new_err(format!(
|
|
244
|
+
"decode_speed_blob: implausible chunk_length={chunk_length} -- corrupt chunk chain"
|
|
245
|
+
)));
|
|
246
|
+
}
|
|
247
|
+
header_start += chunk_length as usize;
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
/// Decode every chunk of a `DB_COMP_SPEED` blob into one writable Python
|
|
252
|
+
/// `bytearray`.
|
|
253
|
+
///
|
|
254
|
+
/// Rust port of `pygdb.lzrw1.decode_speed_blob` -- see that module's
|
|
255
|
+
/// docstring (point 4) and docs/spec.md section 7.3/7.4 for the format:
|
|
256
|
+
/// a blob is a chain of chunks of at most 16368 decompressed bytes each,
|
|
257
|
+
/// only the first preceded by the 16-byte magic; `data` starts at that
|
|
258
|
+
/// magic, and `total_decompressed_length` is the blob header's `+24`
|
|
259
|
+
/// field (how the decoder knows the chain has ended -- the bytes after
|
|
260
|
+
/// the last chunk are page padding, not zeros). A total that isn't
|
|
261
|
+
/// larger than the first chunk (including 0 or negative, "no usable
|
|
262
|
+
/// total") decodes just the first chunk.
|
|
263
|
+
///
|
|
264
|
+
/// Raises `ValueError` for a malformed chunk or chain (bad marker,
|
|
265
|
+
/// implausible lengths, a total the chain doesn't add up to) and
|
|
266
|
+
/// `IndexError` for truncated data -- `pygdb.lzrw1.decode_speed_blob`
|
|
267
|
+
/// turns both into `LZRW1DecodeError`, so callers never see which
|
|
268
|
+
/// backend produced the failure.
|
|
269
|
+
///
|
|
270
|
+
/// The output buffer is sized once, up front, from the first chunk's
|
|
271
|
+
/// header and the declared total, then filled in place via
|
|
272
|
+
/// `PyByteArray::new_with` under `Python::detach` -- the same single-
|
|
273
|
+
/// allocation technique and the same GIL-release argument as
|
|
274
|
+
/// `lzrw1_decompress` (`data` is `&[u8]`, so an immutable `bytes`; the
|
|
275
|
+
/// target `bytearray` isn't Python-visible until this returns). A
|
|
276
|
+
/// declared total larger than any real chain could produce from `data`
|
|
277
|
+
/// (LZRW1 expands at most 8x, plus the 12-byte sub-headers) is rejected
|
|
278
|
+
/// before allocating, so a corrupt header can't request a huge buffer.
|
|
279
|
+
#[pyfunction]
|
|
280
|
+
fn decode_speed_blob<'py>(
|
|
281
|
+
py: Python<'py>,
|
|
282
|
+
data: &[u8],
|
|
283
|
+
total_decompressed_length: i64,
|
|
284
|
+
) -> PyResult<Bound<'py, PyByteArray>> {
|
|
285
|
+
let subtype = data
|
|
286
|
+
.get(8..12)
|
|
287
|
+
.map(|b| i32::from_le_bytes([b[0], b[1], b[2], b[3]]))
|
|
288
|
+
.ok_or_else(|| PyIndexError::new_err("decode_speed_blob: truncated -- no chunk header"))?;
|
|
289
|
+
if subtype != DB_COMP_SPEED {
|
|
290
|
+
return Err(PyValueError::new_err(format!(
|
|
291
|
+
"decode_speed_blob: not a Speed chunk (subtype={subtype})"
|
|
292
|
+
)));
|
|
293
|
+
}
|
|
294
|
+
let (first_length, _, _) = read_chunk_subheader(data, 16)?;
|
|
295
|
+
if !(0 < first_length && first_length < 200_000_000) {
|
|
296
|
+
return Err(PyValueError::new_err(format!(
|
|
297
|
+
"decode_speed_blob: implausible decompressed_length={first_length} -- \
|
|
298
|
+
likely a misaligned or corrupt chunk header"
|
|
299
|
+
)));
|
|
300
|
+
}
|
|
301
|
+
let first_length = first_length as usize;
|
|
302
|
+
let out_len = if total_decompressed_length > first_length as i64 {
|
|
303
|
+
let total = total_decompressed_length as u64;
|
|
304
|
+
if total > (data.len() as u64).saturating_mul(9) {
|
|
305
|
+
return Err(PyValueError::new_err(format!(
|
|
306
|
+
"decode_speed_blob: blob header declares {total} decompressed byte(s), \
|
|
307
|
+
more than {} byte(s) of chunk data could possibly produce",
|
|
308
|
+
data.len(),
|
|
309
|
+
)));
|
|
310
|
+
}
|
|
311
|
+
total as usize
|
|
312
|
+
} else {
|
|
313
|
+
first_length
|
|
314
|
+
};
|
|
315
|
+
PyByteArray::new_with(py, out_len, |buf| {
|
|
316
|
+
py.detach(|| decode_speed_chain_into(data, buf))
|
|
317
|
+
})
|
|
318
|
+
}
|
|
319
|
+
|
|
151
320
|
/// Decompress a `DB_COMP_SIZE` blob's raw zlib/DEFLATE stream into a
|
|
152
321
|
/// writable Python `bytearray`.
|
|
153
322
|
///
|
|
@@ -483,6 +652,7 @@ fn decode_fixed_width_strings_ucs4<'py>(
|
|
|
483
652
|
fn _native(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
|
484
653
|
m.add_function(wrap_pyfunction!(ping, m)?)?;
|
|
485
654
|
m.add_function(wrap_pyfunction!(lzrw1_decompress, m)?)?;
|
|
655
|
+
m.add_function(wrap_pyfunction!(decode_speed_blob, m)?)?;
|
|
486
656
|
m.add_function(wrap_pyfunction!(zlib_decompress, m)?)?;
|
|
487
657
|
m.add_function(wrap_pyfunction!(decompress_grd_blocks, m)?)?;
|
|
488
658
|
m.add_function(wrap_pyfunction!(decode_fixed_width_strings_ucs4, m)?)?;
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|