python-gdb 0.1.0__cp312-abi3-win32.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pygdb/__init__.py +60 -0
- pygdb/_native.pyd +0 -0
- pygdb/gdb.py +641 -0
- pygdb/gdb_reader.py +1368 -0
- pygdb/grd_reader.py +292 -0
- pygdb/lzrw1.py +323 -0
- pygdb/registry.py +104 -0
- python_gdb-0.1.0.dist-info/METADATA +183 -0
- python_gdb-0.1.0.dist-info/RECORD +12 -0
- python_gdb-0.1.0.dist-info/WHEEL +4 -0
- python_gdb-0.1.0.dist-info/licenses/LICENSE +21 -0
- python_gdb-0.1.0.dist-info/sboms/pygdb-native.cyclonedx.json +542 -0
pygdb/grd_reader.py
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Clean-room reader for Geosoft .grd grid files (version 2 header).
|
|
3
|
+
|
|
4
|
+
Derived entirely from public information:
|
|
5
|
+
- Header field layout: adapted from the independent, MIT-licensed
|
|
6
|
+
third-party reader at https://github.com/Loop3D/geosoft_grid
|
|
7
|
+
(grd2geotiff.py), itself citing
|
|
8
|
+
https://help.seequent.com/Oasis-montaj/9.9/en/Content/ss/glossary/grid_file_format__grd.htm
|
|
9
|
+
- Compressed-block layout (the offset table / size table / 16-byte
|
|
10
|
+
per-block sub-header / zlib streams): reverse-derived and verified in
|
|
11
|
+
this project by diffing a real compressed .grd against the byte-exact
|
|
12
|
+
real uncompressed version of the same grid (see
|
|
13
|
+
../samples/loop3d_grd_test/ and ../docs/provenance/notes.md section 4). This
|
|
14
|
+
implementation decompresses each block and checks the result matches
|
|
15
|
+
exactly what the vendor's own documentation and the third-party
|
|
16
|
+
reader implied it should -- confirmed byte-for-byte against real
|
|
17
|
+
files, not merely assumed.
|
|
18
|
+
- Compression algorithm: zlib, NOT LZRW1, despite whatever an on-disk
|
|
19
|
+
COMP_TYPE-style field might claim. This was independently discovered
|
|
20
|
+
by the Loop3D project (credited in their README to Evren
|
|
21
|
+
Pakyuz-Charrier) and independently reconfirmed here by successfully
|
|
22
|
+
zlib-decompressing real compressed blocks.
|
|
23
|
+
|
|
24
|
+
No Geosoft software of any kind was installed, imported, or executed to
|
|
25
|
+
produce this code.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import array
|
|
31
|
+
import struct
|
|
32
|
+
import warnings
|
|
33
|
+
import zlib
|
|
34
|
+
from dataclasses import dataclass, field
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class GRDParseWarning(RuntimeWarning):
|
|
38
|
+
"""
|
|
39
|
+
Warned when this reader hits a truncated file or a compressed block
|
|
40
|
+
that fails to decompress -- see `gdb_reader.GDBParseWarning` for the
|
|
41
|
+
same design rationale applied here: return whatever grid data was
|
|
42
|
+
successfully decoded (padded with the header's own dummy value for
|
|
43
|
+
the missing tail, so the array shape still matches `shape_e *
|
|
44
|
+
shape_v`) rather than raising and discarding a whole grid over one
|
|
45
|
+
bad/missing block.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _warn(msg: str) -> None:
|
|
50
|
+
warnings.warn(msg, GRDParseWarning, stacklevel=3)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
# Valid "ES" (bytes-per-element) values; +1024 marks a compressed grid.
|
|
54
|
+
_VALID_ES = (1, 2, 4, 8, 1024 + 1, 1024 + 2, 1024 + 4, 1024 + 8)
|
|
55
|
+
|
|
56
|
+
# Dummy (no-data) sentinel per element type. These match the GS_*DM
|
|
57
|
+
# constants read from Geosoft's own published geosoft/gxapi/__init__.py
|
|
58
|
+
# (see ../docs/provenance/notes.md section 2) -- an independent cross-check between two
|
|
59
|
+
# unrelated public sources (Geosoft's generated constants, and the
|
|
60
|
+
# Loop3D reader's own from-scratch table) that agree exactly.
|
|
61
|
+
_DUMMIES = {
|
|
62
|
+
"b": -127,
|
|
63
|
+
"B": 255,
|
|
64
|
+
"h": -32767,
|
|
65
|
+
"H": 65535,
|
|
66
|
+
"i": -2147483647,
|
|
67
|
+
"I": 4294967295,
|
|
68
|
+
"f": -1e32,
|
|
69
|
+
"d": -1e32,
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
HEADER_SIZE = 512
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class GrdHeader:
|
|
77
|
+
n_bytes_per_element: int
|
|
78
|
+
sign_flag: int
|
|
79
|
+
shape_e: int
|
|
80
|
+
shape_v: int
|
|
81
|
+
ordering: int
|
|
82
|
+
spacing_e: float
|
|
83
|
+
spacing_v: float
|
|
84
|
+
x_origin: float
|
|
85
|
+
y_origin: float
|
|
86
|
+
rotation: float
|
|
87
|
+
base_value: float
|
|
88
|
+
data_factor: float
|
|
89
|
+
raw: bytes = field(repr=False)
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def is_compressed(self) -> bool:
|
|
93
|
+
return self.n_bytes_per_element > 1024
|
|
94
|
+
|
|
95
|
+
@property
|
|
96
|
+
def element_size(self) -> int:
|
|
97
|
+
"""Bytes per element, with the +1024 compressed marker stripped."""
|
|
98
|
+
es = self.n_bytes_per_element
|
|
99
|
+
return es - 1024 if es > 1024 else es
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _array_typecode(element_size: int, sign_flag: int) -> str:
|
|
103
|
+
if element_size not in (1, 2, 4, 8):
|
|
104
|
+
raise NotImplementedError(f"unsupported element size {element_size}")
|
|
105
|
+
if element_size == 1:
|
|
106
|
+
return "B" if sign_flag == 0 else "b"
|
|
107
|
+
if element_size == 2:
|
|
108
|
+
return "H" if sign_flag == 0 else "h"
|
|
109
|
+
if element_size == 4:
|
|
110
|
+
if sign_flag == 0:
|
|
111
|
+
return "I"
|
|
112
|
+
if sign_flag == 1:
|
|
113
|
+
return "i"
|
|
114
|
+
if sign_flag == 2:
|
|
115
|
+
return "f"
|
|
116
|
+
raise NotImplementedError(f"unsupported sign_flag {sign_flag} for 4-byte element")
|
|
117
|
+
if element_size == 8:
|
|
118
|
+
return "d"
|
|
119
|
+
raise AssertionError("unreachable")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def parse_header(header_bytes: bytes) -> GrdHeader:
|
|
123
|
+
if len(header_bytes) < HEADER_SIZE:
|
|
124
|
+
raise ValueError(f".grd header must be {HEADER_SIZE} bytes, got {len(header_bytes)}")
|
|
125
|
+
|
|
126
|
+
es, sf, ne, nv, kx = struct.unpack_from("<5i", header_bytes, 0)
|
|
127
|
+
de, dv, x0, y0, rot = struct.unpack_from("<5d", header_bytes, 20)
|
|
128
|
+
zbase, zmult = struct.unpack_from("<2d", header_bytes, 60)
|
|
129
|
+
|
|
130
|
+
if es not in _VALID_ES:
|
|
131
|
+
raise NotImplementedError(f"unrecognized element-size field ES={es}")
|
|
132
|
+
|
|
133
|
+
return GrdHeader(
|
|
134
|
+
n_bytes_per_element=es,
|
|
135
|
+
sign_flag=sf,
|
|
136
|
+
shape_e=ne,
|
|
137
|
+
shape_v=nv,
|
|
138
|
+
ordering=kx,
|
|
139
|
+
spacing_e=de,
|
|
140
|
+
spacing_v=dv,
|
|
141
|
+
x_origin=x0,
|
|
142
|
+
y_origin=y0,
|
|
143
|
+
rotation=rot,
|
|
144
|
+
base_value=zbase,
|
|
145
|
+
data_factor=zmult,
|
|
146
|
+
raw=header_bytes[:HEADER_SIZE],
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _decompress_body(body: bytes) -> bytes:
|
|
151
|
+
"""
|
|
152
|
+
Decompress the post-header bytes of a compressed .grd file.
|
|
153
|
+
|
|
154
|
+
Layout (all confirmed by round-tripping a real compressed file to an
|
|
155
|
+
exact byte-for-byte match against its real uncompressed twin -- see
|
|
156
|
+
../docs/provenance/notes.md section 4):
|
|
157
|
+
|
|
158
|
+
offset 0..7 : 8-byte signature/comp-type field (not decoded)
|
|
159
|
+
offset 8 : n_blocks (int32)
|
|
160
|
+
offset 12 : vectors_per_block (int32)
|
|
161
|
+
offset 16 : n_blocks x int64 -- absolute file offset of each
|
|
162
|
+
block's slot (offsets are absolute against the
|
|
163
|
+
*whole file*, i.e. already include the 512-byte
|
|
164
|
+
header)
|
|
165
|
+
offset 16+8N : n_blocks x int32 -- length in bytes of each block's
|
|
166
|
+
slot, INCLUDING the 16-byte per-block sub-header
|
|
167
|
+
below (this differs slightly from how the
|
|
168
|
+
Loop3D reader frames the same arithmetic, but
|
|
169
|
+
lands on the identical byte range)
|
|
170
|
+
|
|
171
|
+
Each block's slot (at its absolute file offset, i.e.
|
|
172
|
+
`body[offset - HEADER_SIZE : ...]` here since `body` starts right
|
|
173
|
+
after the header):
|
|
174
|
+
16 bytes : a per-block sub-header. Confirmed present and its
|
|
175
|
+
length confirmed exactly (skipping exactly 16 bytes
|
|
176
|
+
always lands on a valid zlib stream, 0x78 0x01, in
|
|
177
|
+
every real block we tested). Internal meaning of the
|
|
178
|
+
16 bytes is NOT fully understood -- see docs/provenance/notes.md.
|
|
179
|
+
remainder : a standalone zlib stream for that block.
|
|
180
|
+
"""
|
|
181
|
+
try:
|
|
182
|
+
n_blocks, vectors_per_block = struct.unpack_from("<2i", body, 8)
|
|
183
|
+
offsets = struct.unpack_from(f"<{n_blocks}q", body, 16)
|
|
184
|
+
sizes = struct.unpack_from(f"<{n_blocks}i", body, 16 + n_blocks * 8)
|
|
185
|
+
except struct.error as e:
|
|
186
|
+
_warn(
|
|
187
|
+
f"compressed-grid block table is truncated/unreadable ({e}) -- "
|
|
188
|
+
f"file is likely cut off very early; returning no decoded data"
|
|
189
|
+
)
|
|
190
|
+
return b""
|
|
191
|
+
|
|
192
|
+
out = bytearray()
|
|
193
|
+
for i, (off, size) in enumerate(zip(offsets, sizes)):
|
|
194
|
+
rel = off - HEADER_SIZE # convert absolute file offset -> offset within `body`
|
|
195
|
+
chunk = body[rel + 16 : rel + size]
|
|
196
|
+
if len(chunk) < size - 16:
|
|
197
|
+
_warn(
|
|
198
|
+
f"block {i} of {n_blocks} is truncated (expected {size - 16} "
|
|
199
|
+
f"compressed byte(s), only {len(chunk)} available in the file) "
|
|
200
|
+
f"-- stopping here; returning the {i} block(s) already "
|
|
201
|
+
f"decompressed rather than the full grid"
|
|
202
|
+
)
|
|
203
|
+
break
|
|
204
|
+
try:
|
|
205
|
+
out += zlib.decompress(chunk)
|
|
206
|
+
except zlib.error as e:
|
|
207
|
+
_warn(
|
|
208
|
+
f"block {i} of {n_blocks} failed to decompress ({e}) -- "
|
|
209
|
+
f"corrupt or truncated data; stopping here; returning the "
|
|
210
|
+
f"{i} block(s) already decompressed rather than the full grid"
|
|
211
|
+
)
|
|
212
|
+
break
|
|
213
|
+
return bytes(out)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def read_grd(path: str):
|
|
217
|
+
"""
|
|
218
|
+
Read a .grd file and return (header, values) where `values` is an
|
|
219
|
+
`array.array` of the grid's raw (unscaled) element values in
|
|
220
|
+
on-disk order (row-major per the file's own `ordering`/KX flag).
|
|
221
|
+
numpy is a dependency of the `pygdb` package as a whole (see
|
|
222
|
+
`gdb_reader.py`'s VA/array-channel decoding), but this module's own
|
|
223
|
+
`.grd` reading doesn't need it -- a grid's shape is already fully
|
|
224
|
+
known from `shape_e`/`shape_v`, so reshaping is left to the caller
|
|
225
|
+
rather than done here.
|
|
226
|
+
|
|
227
|
+
A file too short to even hold the 512-byte header raises `ValueError`
|
|
228
|
+
(there's nothing to return at all in that case). Beyond that, this
|
|
229
|
+
degrades gracefully rather than hard-crashing: a truncated/corrupt
|
|
230
|
+
compressed block is skipped (see `_decompress_body`), and a decoded
|
|
231
|
+
element count that doesn't match `shape_e * shape_v` -- which for a
|
|
232
|
+
real, complete file never happens, so it's always a sign of trouble
|
|
233
|
+
-- is reported as a `GRDParseWarning` rather than a raised
|
|
234
|
+
`ValueError`; `values` is returned exactly as long as what was
|
|
235
|
+
actually decoded, so a caller can check `len(values)` against
|
|
236
|
+
`header.shape_e * header.shape_v` itself if it needs to know.
|
|
237
|
+
"""
|
|
238
|
+
with open(path, "rb") as f:
|
|
239
|
+
raw = f.read()
|
|
240
|
+
|
|
241
|
+
if len(raw) < HEADER_SIZE:
|
|
242
|
+
raise ValueError(
|
|
243
|
+
f"{path}: only {len(raw)} byte(s) available, need at least "
|
|
244
|
+
f"{HEADER_SIZE} for the header -- nothing to decode"
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
header = parse_header(raw[:HEADER_SIZE])
|
|
248
|
+
body = raw[HEADER_SIZE:]
|
|
249
|
+
|
|
250
|
+
if header.is_compressed:
|
|
251
|
+
body = _decompress_body(body)
|
|
252
|
+
|
|
253
|
+
typecode = _array_typecode(header.element_size, header.sign_flag)
|
|
254
|
+
values = array.array(typecode)
|
|
255
|
+
# array.frombytes requires a whole number of elements; a truncated
|
|
256
|
+
# tail (partial last element) is silently dropped rather than raising,
|
|
257
|
+
# consistent with "return everything that was actually decodable."
|
|
258
|
+
usable = len(body) - (len(body) % values.itemsize)
|
|
259
|
+
values.frombytes(body[:usable])
|
|
260
|
+
|
|
261
|
+
expected = header.shape_e * header.shape_v
|
|
262
|
+
if len(values) != expected:
|
|
263
|
+
_warn(
|
|
264
|
+
f"{path}: decoded {len(values)} element(s), expected "
|
|
265
|
+
f"shape_e*shape_v={expected} -- file is likely truncated or "
|
|
266
|
+
f"a compressed block failed to decompress (see any earlier "
|
|
267
|
+
f"GRDParseWarning for which); returning the {len(values)} "
|
|
268
|
+
f"element(s) actually decoded"
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
return header, values
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def dummy_value(header: GrdHeader):
|
|
275
|
+
typecode = _array_typecode(header.element_size, header.sign_flag)
|
|
276
|
+
return _DUMMIES[typecode]
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
if __name__ == "__main__":
|
|
280
|
+
import sys
|
|
281
|
+
|
|
282
|
+
if len(sys.argv) != 2:
|
|
283
|
+
print("usage: python -m pygdb.grd_reader <path-to.grd>")
|
|
284
|
+
raise SystemExit(1)
|
|
285
|
+
|
|
286
|
+
header, values = read_grd(sys.argv[1])
|
|
287
|
+
print(f"shape: {header.shape_e} x {header.shape_v} (ordering={header.ordering})")
|
|
288
|
+
print(f"compressed: {header.is_compressed} element_size: {header.element_size} bytes")
|
|
289
|
+
print(f"origin: ({header.x_origin}, {header.y_origin}) spacing: ({header.spacing_e}, {header.spacing_v})")
|
|
290
|
+
print(f"z scale: base={header.base_value} mult={header.data_factor}")
|
|
291
|
+
print(f"dummy value: {dummy_value(header)}")
|
|
292
|
+
print(f"decoded {len(values)} elements; first 5: {values[:5]}")
|
pygdb/lzrw1.py
ADDED
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Canonical LZRW1 decompressor (Ross Williams, 1991), ported from his own
|
|
3
|
+
public-domain reference C source
|
|
4
|
+
(http://www.ross.net/compression/download/original/old_lzrw1.c,
|
|
5
|
+
explicitly marked "This code is public domain" in its own header
|
|
6
|
+
comment), used here to decode Geosoft `.gdb` DB_COMP_SPEED page data.
|
|
7
|
+
|
|
8
|
+
[CONFIRMED] finding (see ../docs/provenance/notes.md section 6.5c/6.5d for the full
|
|
9
|
+
derivation): Geosoft's DB_COMP_SPEED mode wraps *exactly* Ross Williams'
|
|
10
|
+
canonical LZRW1 byte stream -- the same 2-byte control word + 1-byte
|
|
11
|
+
literal / 2-byte nibble-packed copy-item scheme as his original
|
|
12
|
+
`lzrw1_decompress()` -- with these Geosoft-specific differences from the
|
|
13
|
+
reference C wrapper:
|
|
14
|
+
|
|
15
|
+
1. No 4-byte FLAG_BYTES prefix (the reference C code's own
|
|
16
|
+
FLAG_COMPRESS/FLAG_COPY byte + 3 padding bytes) -- the control word
|
|
17
|
+
starts immediately for a compressed chunk.
|
|
18
|
+
2. Each chunk (which may span several of the file's physical
|
|
19
|
+
1024-byte pages) is preceded by a 28-byte Geosoft-specific wrapper,
|
|
20
|
+
not part of LZRW1 itself:
|
|
21
|
+
- 16 bytes: the magic sub-header shared with the `.grd` sibling
|
|
22
|
+
format and with `.gdb`'s DB_COMP_SIZE (zlib) mode:
|
|
23
|
+
`0f 0e ff fe 12 34 56 78 <subtype int32> <reserved int32>`
|
|
24
|
+
-- subtype is 1 for DB_COMP_SPEED (2 for DB_COMP_SIZE).
|
|
25
|
+
- 12 bytes: `<decompressed_length int32> <chunk_length int32>
|
|
26
|
+
<marker int32>`. `chunk_length` INCLUDES these 12 bytes (i.e.
|
|
27
|
+
chunk_length - 12 == the number of raw bytes that follow,
|
|
28
|
+
compressed or not -- see point 3).
|
|
29
|
+
3. **`marker` is itself a real flag, not just a validation sentinel**
|
|
30
|
+
(this was missed in an earlier pass that only tested against 4 of
|
|
31
|
+
the 10 real Speed files in this project's full sample set -- the
|
|
32
|
+
other 6, found and validated later, include real examples of the
|
|
33
|
+
second case below and are what exposed it):
|
|
34
|
+
- `marker == 0xF4E5D6C7` (-186263865 signed): this chunk's
|
|
35
|
+
payload is genuine LZRW1-compressed data (Ross Williams' own
|
|
36
|
+
reference implementation has an analogous `FLAG_COMPRESS`
|
|
37
|
+
case, using a different byte value/position; Geosoft's variant
|
|
38
|
+
repurposes this 4-byte marker field for the same purpose).
|
|
39
|
+
- `marker == 0xF0E1D2C3` (-253635901 signed): this chunk's
|
|
40
|
+
payload is **stored raw, uncompressed** -- i.e. Geosoft's
|
|
41
|
+
equivalent of Ross Williams' reference `FLAG_COPY` case (used
|
|
42
|
+
when LZRW1 compression didn't shrink the data, so the encoder
|
|
43
|
+
gave up and stored it verbatim instead). For every real
|
|
44
|
+
instance of this found, `chunk_length - 12 == decompressed_length`
|
|
45
|
+
exactly (no compression ratio at all, consistent with "stored
|
|
46
|
+
raw"), and reading `decompressed_length` bytes directly (no
|
|
47
|
+
decompression) produces plausible real survey data (checked as
|
|
48
|
+
float64: smoothly-varying, physically sane gravity/magnetic
|
|
49
|
+
values). Every real marker value found across all 10 real
|
|
50
|
+
Speed files was one of these two constants -- zero exceptions,
|
|
51
|
+
zero unrecognized third values.
|
|
52
|
+
|
|
53
|
+
Validated exactly (not just "plausibly") against **all 10 real**
|
|
54
|
+
DB_COMP_SPEED files now in this project's sample set (the original 4
|
|
55
|
+
GEOTEM/Questem EM/Mag files, plus 4 AGG/Mag files from the Melinda Downs
|
|
56
|
+
delivery, plus 2 Rad/Mag files from the Georgetown-AGSO delivery,
|
|
57
|
+
extracted specifically to check this): for every chunk in every file --
|
|
58
|
+
not a sample -- decoding exactly `decompressed_length` output bytes
|
|
59
|
+
(via LZRW1 decompression when marker indicates compressed, or a direct
|
|
60
|
+
copy when marker indicates raw/stored) consumes exactly
|
|
61
|
+
`chunk_length - 12` input bytes, with zero slack, and zero chunks with
|
|
62
|
+
an unrecognized marker value.
|
|
63
|
+
|
|
64
|
+
No Geosoft software of any kind was used to derive or produce this
|
|
65
|
+
module -- only Ross Williams' own public-domain reference source (read,
|
|
66
|
+
not executed against anything proprietary) and real, independently
|
|
67
|
+
obtained `.gdb` files.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
from __future__ import annotations
|
|
71
|
+
|
|
72
|
+
import struct
|
|
73
|
+
import warnings
|
|
74
|
+
from dataclasses import dataclass
|
|
75
|
+
|
|
76
|
+
try:
|
|
77
|
+
from . import _native as _native_ext
|
|
78
|
+
except ImportError:
|
|
79
|
+
_native_ext = None
|
|
80
|
+
|
|
81
|
+
CHUNK_MAGIC = bytes.fromhex("0f0efffe12345678")
|
|
82
|
+
DB_COMP_SPEED = 1
|
|
83
|
+
DB_COMP_SIZE = 2
|
|
84
|
+
|
|
85
|
+
# The two real marker values observed in the 12-byte chunk length
|
|
86
|
+
# sub-header, across all 10 real DB_COMP_SPEED files in this project,
|
|
87
|
+
# with zero exceptions and zero unrecognized third values.
|
|
88
|
+
MARKER_COMPRESSED = -186263865 # 0xF4E5D6C7 -- payload is real LZRW1
|
|
89
|
+
MARKER_STORED_RAW = -253635901 # 0xF0E1D2C3 -- payload is stored verbatim
|
|
90
|
+
|
|
91
|
+
# Backwards-compatible alias (earlier name, before the raw/stored case
|
|
92
|
+
# was found).
|
|
93
|
+
KNOWN_MARKER = MARKER_COMPRESSED
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def lzrw1_decompress(data: bytes, start: int, decompressed_length: int) -> bytes:
|
|
97
|
+
"""
|
|
98
|
+
Decompress exactly `decompressed_length` bytes of canonical LZRW1
|
|
99
|
+
data (Ross Williams' algorithm, no FLAG_BYTES prefix) starting at
|
|
100
|
+
`data[start:]`. Returns the decompressed bytes.
|
|
101
|
+
|
|
102
|
+
Dispatches to the compiled `pygdb._native` extension when it's
|
|
103
|
+
available (same algorithm, ported to Rust -- see `rust/src/lib.rs`;
|
|
104
|
+
~16x faster on real DB_COMP_SPEED data, since this per-byte loop is
|
|
105
|
+
this reader's one CPU-bound hot path), falling back to the pure-Python
|
|
106
|
+
`_lzrw1_decompress_py` below when it isn't. `_native` raises
|
|
107
|
+
`IndexError` under the same truncated/corrupt-input conditions as the
|
|
108
|
+
pure-Python version, so callers (`decode_speed_chunk`) don't need to
|
|
109
|
+
know which backend produced the error.
|
|
110
|
+
"""
|
|
111
|
+
if _native_ext is not None:
|
|
112
|
+
return bytes(_native_ext.lzrw1_decompress(data, start, decompressed_length))
|
|
113
|
+
return _lzrw1_decompress_py(data, start, decompressed_length)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _lzrw1_decompress_py(data: bytes, start: int, decompressed_length: int) -> bytes:
|
|
117
|
+
"""
|
|
118
|
+
Pure-Python reference implementation of `lzrw1_decompress` -- kept as
|
|
119
|
+
the always-available fallback when `pygdb._native` isn't built, and
|
|
120
|
+
as the documented, clean-room-derived source of truth for the
|
|
121
|
+
algorithm.
|
|
122
|
+
|
|
123
|
+
This is a direct, literal port of the core loop in Ross Williams'
|
|
124
|
+
own public-domain `lzrw1_decompress()` (see module docstring for the
|
|
125
|
+
source URL) -- same control-word/control-bit walk, same nibble
|
|
126
|
+
packing for copy-item offset/length. The only functional change from
|
|
127
|
+
his reference is that this operates on a `bytes` object with an
|
|
128
|
+
explicit output-length stop condition instead of a fixed-size output
|
|
129
|
+
buffer, and does not skip his 4-byte FLAG_BYTES prefix (Geosoft's
|
|
130
|
+
on-disk chunks don't have it -- the equivalent flag lives in the
|
|
131
|
+
12-byte length sub-header's `marker` field instead, see
|
|
132
|
+
`decode_speed_chunk`).
|
|
133
|
+
"""
|
|
134
|
+
p = start
|
|
135
|
+
out = bytearray()
|
|
136
|
+
while len(out) < decompressed_length:
|
|
137
|
+
control = data[p] | (data[p + 1] << 8)
|
|
138
|
+
p += 2
|
|
139
|
+
for _bit in range(16):
|
|
140
|
+
if len(out) >= decompressed_length:
|
|
141
|
+
break
|
|
142
|
+
if control & 1:
|
|
143
|
+
b0 = data[p]
|
|
144
|
+
b1 = data[p + 1]
|
|
145
|
+
p += 2
|
|
146
|
+
offset = ((b0 & 0xF0) << 4) + b1
|
|
147
|
+
length = (b0 & 0x0F) + 1
|
|
148
|
+
start_idx = len(out) - offset
|
|
149
|
+
for i in range(length):
|
|
150
|
+
if len(out) >= decompressed_length:
|
|
151
|
+
break
|
|
152
|
+
out.append(out[start_idx + i])
|
|
153
|
+
else:
|
|
154
|
+
out.append(data[p])
|
|
155
|
+
p += 1
|
|
156
|
+
control >>= 1
|
|
157
|
+
return bytes(out)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
@dataclass
|
|
161
|
+
class SpeedChunk:
|
|
162
|
+
magic_offset: int # file offset of the 16-byte magic sub-header
|
|
163
|
+
subtype: int # 1 = DB_COMP_SPEED, 2 = DB_COMP_SIZE
|
|
164
|
+
decompressed_length: int
|
|
165
|
+
chunk_length: int # includes the 12-byte length sub-header
|
|
166
|
+
marker: int
|
|
167
|
+
payload_offset: int # file offset where the payload starts
|
|
168
|
+
|
|
169
|
+
@property
|
|
170
|
+
def is_stored_raw(self) -> bool:
|
|
171
|
+
return self.marker == MARKER_STORED_RAW
|
|
172
|
+
|
|
173
|
+
@property
|
|
174
|
+
def is_compressed(self) -> bool:
|
|
175
|
+
return self.marker == MARKER_COMPRESSED
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
class LZRW1DecodeError(Exception):
|
|
179
|
+
"""
|
|
180
|
+
Raised by `decode_speed_chunk`/`parse_chunk_header` when a Speed-mode
|
|
181
|
+
chunk can't be decoded: truncated/corrupt data, or a subtype/marker
|
|
182
|
+
this module doesn't recognize. A single, deliberately narrow
|
|
183
|
+
exception type (rather than a bare `AssertionError`/`IndexError`/
|
|
184
|
+
`struct.error` grab-bag) so callers -- notably
|
|
185
|
+
`gdb_reader.read_blob_values` -- can catch exactly this and fail
|
|
186
|
+
gracefully (return whatever was already decoded elsewhere, emit a
|
|
187
|
+
clear warning) instead of crashing. See docs/provenance/notes.md's "reader
|
|
188
|
+
robustness" notes for the design rationale; this reader is not meant
|
|
189
|
+
to hard-crash on a truncated download or an unrecognized real-world
|
|
190
|
+
variant, per an explicit engineering request from the coordinator.
|
|
191
|
+
"""
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def parse_chunk_header(data: bytes, magic_offset: int) -> SpeedChunk:
|
|
195
|
+
try:
|
|
196
|
+
subtype, _reserved = struct.unpack_from("<ii", data, magic_offset + 8)
|
|
197
|
+
header_start = magic_offset + 16
|
|
198
|
+
decompressed_length, chunk_length, marker = struct.unpack_from(
|
|
199
|
+
"<iii", data, header_start
|
|
200
|
+
)
|
|
201
|
+
except struct.error as e:
|
|
202
|
+
raise LZRW1DecodeError(
|
|
203
|
+
f"not enough bytes at offset {magic_offset} to read a full chunk header "
|
|
204
|
+
f"(need 28 bytes from the magic; only {len(data) - magic_offset} available) "
|
|
205
|
+
f"-- truncated data"
|
|
206
|
+
) from e
|
|
207
|
+
return SpeedChunk(
|
|
208
|
+
magic_offset=magic_offset,
|
|
209
|
+
subtype=subtype,
|
|
210
|
+
decompressed_length=decompressed_length,
|
|
211
|
+
chunk_length=chunk_length,
|
|
212
|
+
marker=marker,
|
|
213
|
+
payload_offset=header_start + 12,
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def decode_speed_chunk(data: bytes, chunk: SpeedChunk) -> bytes:
|
|
218
|
+
"""
|
|
219
|
+
Decode one DB_COMP_SPEED chunk's data -- transparently handling both
|
|
220
|
+
the LZRW1-compressed case and the stored-raw case (see module
|
|
221
|
+
docstring point 3). Raises `LZRW1DecodeError` (never a bare
|
|
222
|
+
`AssertionError`/`IndexError`) if the chunk doesn't look like a
|
|
223
|
+
real, well-formed Speed chunk (wrong subtype, implausible lengths,
|
|
224
|
+
an unrecognized marker value, or truncated payload data) -- callers
|
|
225
|
+
should catch this one type and treat it as "this chunk couldn't be
|
|
226
|
+
decoded," not as a program bug.
|
|
227
|
+
"""
|
|
228
|
+
if chunk.subtype != DB_COMP_SPEED:
|
|
229
|
+
raise LZRW1DecodeError(f"not a Speed chunk (subtype={chunk.subtype})")
|
|
230
|
+
if not (0 < chunk.decompressed_length < 200_000_000):
|
|
231
|
+
raise LZRW1DecodeError(
|
|
232
|
+
f"implausible decompressed_length={chunk.decompressed_length} -- "
|
|
233
|
+
f"likely a misaligned or corrupt chunk header"
|
|
234
|
+
)
|
|
235
|
+
if chunk.is_stored_raw:
|
|
236
|
+
if chunk.chunk_length - 12 != chunk.decompressed_length:
|
|
237
|
+
raise LZRW1DecodeError(
|
|
238
|
+
"stored-raw chunk should have chunk_length-12 == decompressed_length "
|
|
239
|
+
f"(got chunk_length-12={chunk.chunk_length - 12}, "
|
|
240
|
+
f"decompressed_length={chunk.decompressed_length})"
|
|
241
|
+
)
|
|
242
|
+
payload = data[chunk.payload_offset: chunk.payload_offset + chunk.decompressed_length]
|
|
243
|
+
if len(payload) < chunk.decompressed_length:
|
|
244
|
+
raise LZRW1DecodeError(
|
|
245
|
+
f"truncated stored-raw payload: expected {chunk.decompressed_length} "
|
|
246
|
+
f"byte(s), only {len(payload)} available -- file cut off mid-chunk?"
|
|
247
|
+
)
|
|
248
|
+
return payload
|
|
249
|
+
if not chunk.is_compressed:
|
|
250
|
+
raise LZRW1DecodeError(f"unrecognized marker value: {chunk.marker}")
|
|
251
|
+
try:
|
|
252
|
+
return lzrw1_decompress(data, chunk.payload_offset, chunk.decompressed_length)
|
|
253
|
+
except IndexError as e:
|
|
254
|
+
raise LZRW1DecodeError(
|
|
255
|
+
f"ran out of input data while decompressing (need to produce "
|
|
256
|
+
f"{chunk.decompressed_length} bytes) -- truncated or corrupt payload"
|
|
257
|
+
) from e
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def find_speed_chunks(data: bytes):
|
|
261
|
+
"""
|
|
262
|
+
Yield SpeedChunk for every DB_COMP_SPEED (subtype==1) chunk found in
|
|
263
|
+
`data` by scanning for the shared 16-byte magic. A magic-byte match
|
|
264
|
+
too close to the end of `data` to hold a full 28-byte chunk header
|
|
265
|
+
(e.g. a truncated file, or a coincidental match inside real data
|
|
266
|
+
right before EOF -- both real, observed cases elsewhere in this
|
|
267
|
+
project) is skipped with a warning rather than raising -- this is a
|
|
268
|
+
scanning helper, not a strict decoder, so it degrades gracefully and
|
|
269
|
+
keeps looking rather than aborting the whole scan.
|
|
270
|
+
"""
|
|
271
|
+
start = 0
|
|
272
|
+
while True:
|
|
273
|
+
idx = data.find(CHUNK_MAGIC, start)
|
|
274
|
+
if idx == -1:
|
|
275
|
+
return
|
|
276
|
+
try:
|
|
277
|
+
chunk = parse_chunk_header(data, idx)
|
|
278
|
+
except LZRW1DecodeError as e:
|
|
279
|
+
warnings.warn(
|
|
280
|
+
f"find_speed_chunks: skipping a magic-byte match at offset {idx} "
|
|
281
|
+
f"that isn't a full chunk header ({e})",
|
|
282
|
+
RuntimeWarning,
|
|
283
|
+
stacklevel=2,
|
|
284
|
+
)
|
|
285
|
+
start = idx + 1
|
|
286
|
+
continue
|
|
287
|
+
if chunk.subtype == DB_COMP_SPEED:
|
|
288
|
+
yield chunk
|
|
289
|
+
start = idx + 1
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
if __name__ == "__main__":
|
|
293
|
+
import sys
|
|
294
|
+
|
|
295
|
+
path = sys.argv[1] if len(sys.argv) > 1 else "samples/GSQ_Data/extracted/holroy/em000293/DB_EM_293.gdb"
|
|
296
|
+
data = open(path, "rb").read()
|
|
297
|
+
n_ok = 0
|
|
298
|
+
n_compressed = 0
|
|
299
|
+
n_stored = 0
|
|
300
|
+
n_bad_marker = 0
|
|
301
|
+
n_total = 0
|
|
302
|
+
for chunk in find_speed_chunks(data):
|
|
303
|
+
n_total += 1
|
|
304
|
+
try:
|
|
305
|
+
out = decode_speed_chunk(data, chunk)
|
|
306
|
+
except LZRW1DecodeError as e:
|
|
307
|
+
n_bad_marker += 1
|
|
308
|
+
if n_bad_marker <= 3:
|
|
309
|
+
print(f"chunk @ {chunk.magic_offset}: FAILED ({e})")
|
|
310
|
+
continue
|
|
311
|
+
n_ok += 1
|
|
312
|
+
if chunk.is_stored_raw:
|
|
313
|
+
n_stored += 1
|
|
314
|
+
else:
|
|
315
|
+
n_compressed += 1
|
|
316
|
+
if n_total <= 5:
|
|
317
|
+
print(
|
|
318
|
+
f"chunk @ {chunk.magic_offset}: decompressed_length={chunk.decompressed_length} "
|
|
319
|
+
f"chunk_length={chunk.chunk_length} kind={'stored-raw' if chunk.is_stored_raw else 'compressed'} "
|
|
320
|
+
f"first_16_bytes={out[:16].hex()}"
|
|
321
|
+
)
|
|
322
|
+
print(f"\n{n_total} chunks total: {n_compressed} compressed, {n_stored} stored-raw, "
|
|
323
|
+
f"{n_bad_marker} unrecognized marker")
|