python-gdb 0.1.0__cp312-abi3-win_arm64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pygdb/grd_reader.py ADDED
@@ -0,0 +1,292 @@
1
+ """
2
+ Clean-room reader for Geosoft .grd grid files (version 2 header).
3
+
4
+ Derived entirely from public information:
5
+ - Header field layout: adapted from the independent, MIT-licensed
6
+ third-party reader at https://github.com/Loop3D/geosoft_grid
7
+ (grd2geotiff.py), itself citing
8
+ https://help.seequent.com/Oasis-montaj/9.9/en/Content/ss/glossary/grid_file_format__grd.htm
9
+ - Compressed-block layout (the offset table / size table / 16-byte
10
+ per-block sub-header / zlib streams): reverse-derived and verified in
11
+ this project by diffing a real compressed .grd against the byte-exact
12
+ real uncompressed version of the same grid (see
13
+ ../samples/loop3d_grd_test/ and ../docs/provenance/notes.md section 4). This
14
+ implementation decompresses each block and checks the result matches
15
+ exactly what the vendor's own documentation and the third-party
16
+ reader implied it should -- confirmed byte-for-byte against real
17
+ files, not merely assumed.
18
+ - Compression algorithm: zlib, NOT LZRW1, despite whatever an on-disk
19
+ COMP_TYPE-style field might claim. This was independently discovered
20
+ by the Loop3D project (credited in their README to Evren
21
+ Pakyuz-Charrier) and independently reconfirmed here by successfully
22
+ zlib-decompressing real compressed blocks.
23
+
24
+ No Geosoft software of any kind was installed, imported, or executed to
25
+ produce this code.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import array
31
+ import struct
32
+ import warnings
33
+ import zlib
34
+ from dataclasses import dataclass, field
35
+
36
+
37
+ class GRDParseWarning(RuntimeWarning):
38
+ """
39
+ Warned when this reader hits a truncated file or a compressed block
40
+ that fails to decompress -- see `gdb_reader.GDBParseWarning` for the
41
+ same design rationale applied here: return whatever grid data was
42
+ successfully decoded (padded with the header's own dummy value for
43
+ the missing tail, so the array shape still matches `shape_e *
44
+ shape_v`) rather than raising and discarding a whole grid over one
45
+ bad/missing block.
46
+ """
47
+
48
+
49
+ def _warn(msg: str) -> None:
50
+ warnings.warn(msg, GRDParseWarning, stacklevel=3)
51
+
52
+
53
+ # Valid "ES" (bytes-per-element) values; +1024 marks a compressed grid.
54
+ _VALID_ES = (1, 2, 4, 8, 1024 + 1, 1024 + 2, 1024 + 4, 1024 + 8)
55
+
56
+ # Dummy (no-data) sentinel per element type. These match the GS_*DM
57
+ # constants read from Geosoft's own published geosoft/gxapi/__init__.py
58
+ # (see ../docs/provenance/notes.md section 2) -- an independent cross-check between two
59
+ # unrelated public sources (Geosoft's generated constants, and the
60
+ # Loop3D reader's own from-scratch table) that agree exactly.
61
+ _DUMMIES = {
62
+ "b": -127,
63
+ "B": 255,
64
+ "h": -32767,
65
+ "H": 65535,
66
+ "i": -2147483647,
67
+ "I": 4294967295,
68
+ "f": -1e32,
69
+ "d": -1e32,
70
+ }
71
+
72
+ HEADER_SIZE = 512
73
+
74
+
75
+ @dataclass
76
+ class GrdHeader:
77
+ n_bytes_per_element: int
78
+ sign_flag: int
79
+ shape_e: int
80
+ shape_v: int
81
+ ordering: int
82
+ spacing_e: float
83
+ spacing_v: float
84
+ x_origin: float
85
+ y_origin: float
86
+ rotation: float
87
+ base_value: float
88
+ data_factor: float
89
+ raw: bytes = field(repr=False)
90
+
91
+ @property
92
+ def is_compressed(self) -> bool:
93
+ return self.n_bytes_per_element > 1024
94
+
95
+ @property
96
+ def element_size(self) -> int:
97
+ """Bytes per element, with the +1024 compressed marker stripped."""
98
+ es = self.n_bytes_per_element
99
+ return es - 1024 if es > 1024 else es
100
+
101
+
102
+ def _array_typecode(element_size: int, sign_flag: int) -> str:
103
+ if element_size not in (1, 2, 4, 8):
104
+ raise NotImplementedError(f"unsupported element size {element_size}")
105
+ if element_size == 1:
106
+ return "B" if sign_flag == 0 else "b"
107
+ if element_size == 2:
108
+ return "H" if sign_flag == 0 else "h"
109
+ if element_size == 4:
110
+ if sign_flag == 0:
111
+ return "I"
112
+ if sign_flag == 1:
113
+ return "i"
114
+ if sign_flag == 2:
115
+ return "f"
116
+ raise NotImplementedError(f"unsupported sign_flag {sign_flag} for 4-byte element")
117
+ if element_size == 8:
118
+ return "d"
119
+ raise AssertionError("unreachable")
120
+
121
+
122
+ def parse_header(header_bytes: bytes) -> GrdHeader:
123
+ if len(header_bytes) < HEADER_SIZE:
124
+ raise ValueError(f".grd header must be {HEADER_SIZE} bytes, got {len(header_bytes)}")
125
+
126
+ es, sf, ne, nv, kx = struct.unpack_from("<5i", header_bytes, 0)
127
+ de, dv, x0, y0, rot = struct.unpack_from("<5d", header_bytes, 20)
128
+ zbase, zmult = struct.unpack_from("<2d", header_bytes, 60)
129
+
130
+ if es not in _VALID_ES:
131
+ raise NotImplementedError(f"unrecognized element-size field ES={es}")
132
+
133
+ return GrdHeader(
134
+ n_bytes_per_element=es,
135
+ sign_flag=sf,
136
+ shape_e=ne,
137
+ shape_v=nv,
138
+ ordering=kx,
139
+ spacing_e=de,
140
+ spacing_v=dv,
141
+ x_origin=x0,
142
+ y_origin=y0,
143
+ rotation=rot,
144
+ base_value=zbase,
145
+ data_factor=zmult,
146
+ raw=header_bytes[:HEADER_SIZE],
147
+ )
148
+
149
+
150
+ def _decompress_body(body: bytes) -> bytes:
151
+ """
152
+ Decompress the post-header bytes of a compressed .grd file.
153
+
154
+ Layout (all confirmed by round-tripping a real compressed file to an
155
+ exact byte-for-byte match against its real uncompressed twin -- see
156
+ ../docs/provenance/notes.md section 4):
157
+
158
+ offset 0..7 : 8-byte signature/comp-type field (not decoded)
159
+ offset 8 : n_blocks (int32)
160
+ offset 12 : vectors_per_block (int32)
161
+ offset 16 : n_blocks x int64 -- absolute file offset of each
162
+ block's slot (offsets are absolute against the
163
+ *whole file*, i.e. already include the 512-byte
164
+ header)
165
+ offset 16+8N : n_blocks x int32 -- length in bytes of each block's
166
+ slot, INCLUDING the 16-byte per-block sub-header
167
+ below (this differs slightly from how the
168
+ Loop3D reader frames the same arithmetic, but
169
+ lands on the identical byte range)
170
+
171
+ Each block's slot (at its absolute file offset, i.e.
172
+ `body[offset - HEADER_SIZE : ...]` here since `body` starts right
173
+ after the header):
174
+ 16 bytes : a per-block sub-header. Confirmed present and its
175
+ length confirmed exactly (skipping exactly 16 bytes
176
+ always lands on a valid zlib stream, 0x78 0x01, in
177
+ every real block we tested). Internal meaning of the
178
+ 16 bytes is NOT fully understood -- see docs/provenance/notes.md.
179
+ remainder : a standalone zlib stream for that block.
180
+ """
181
+ try:
182
+ n_blocks, vectors_per_block = struct.unpack_from("<2i", body, 8)
183
+ offsets = struct.unpack_from(f"<{n_blocks}q", body, 16)
184
+ sizes = struct.unpack_from(f"<{n_blocks}i", body, 16 + n_blocks * 8)
185
+ except struct.error as e:
186
+ _warn(
187
+ f"compressed-grid block table is truncated/unreadable ({e}) -- "
188
+ f"file is likely cut off very early; returning no decoded data"
189
+ )
190
+ return b""
191
+
192
+ out = bytearray()
193
+ for i, (off, size) in enumerate(zip(offsets, sizes)):
194
+ rel = off - HEADER_SIZE # convert absolute file offset -> offset within `body`
195
+ chunk = body[rel + 16 : rel + size]
196
+ if len(chunk) < size - 16:
197
+ _warn(
198
+ f"block {i} of {n_blocks} is truncated (expected {size - 16} "
199
+ f"compressed byte(s), only {len(chunk)} available in the file) "
200
+ f"-- stopping here; returning the {i} block(s) already "
201
+ f"decompressed rather than the full grid"
202
+ )
203
+ break
204
+ try:
205
+ out += zlib.decompress(chunk)
206
+ except zlib.error as e:
207
+ _warn(
208
+ f"block {i} of {n_blocks} failed to decompress ({e}) -- "
209
+ f"corrupt or truncated data; stopping here; returning the "
210
+ f"{i} block(s) already decompressed rather than the full grid"
211
+ )
212
+ break
213
+ return bytes(out)
214
+
215
+
216
+ def read_grd(path: str):
217
+ """
218
+ Read a .grd file and return (header, values) where `values` is an
219
+ `array.array` of the grid's raw (unscaled) element values in
220
+ on-disk order (row-major per the file's own `ordering`/KX flag).
221
+ numpy is a dependency of the `pygdb` package as a whole (see
222
+ `gdb_reader.py`'s VA/array-channel decoding), but this module's own
223
+ `.grd` reading doesn't need it -- a grid's shape is already fully
224
+ known from `shape_e`/`shape_v`, so reshaping is left to the caller
225
+ rather than done here.
226
+
227
+ A file too short to even hold the 512-byte header raises `ValueError`
228
+ (there's nothing to return at all in that case). Beyond that, this
229
+ degrades gracefully rather than hard-crashing: a truncated/corrupt
230
+ compressed block is skipped (see `_decompress_body`), and a decoded
231
+ element count that doesn't match `shape_e * shape_v` -- which for a
232
+ real, complete file never happens, so it's always a sign of trouble
233
+ -- is reported as a `GRDParseWarning` rather than a raised
234
+ `ValueError`; `values` is returned exactly as long as what was
235
+ actually decoded, so a caller can check `len(values)` against
236
+ `header.shape_e * header.shape_v` itself if it needs to know.
237
+ """
238
+ with open(path, "rb") as f:
239
+ raw = f.read()
240
+
241
+ if len(raw) < HEADER_SIZE:
242
+ raise ValueError(
243
+ f"{path}: only {len(raw)} byte(s) available, need at least "
244
+ f"{HEADER_SIZE} for the header -- nothing to decode"
245
+ )
246
+
247
+ header = parse_header(raw[:HEADER_SIZE])
248
+ body = raw[HEADER_SIZE:]
249
+
250
+ if header.is_compressed:
251
+ body = _decompress_body(body)
252
+
253
+ typecode = _array_typecode(header.element_size, header.sign_flag)
254
+ values = array.array(typecode)
255
+ # array.frombytes requires a whole number of elements; a truncated
256
+ # tail (partial last element) is silently dropped rather than raising,
257
+ # consistent with "return everything that was actually decodable."
258
+ usable = len(body) - (len(body) % values.itemsize)
259
+ values.frombytes(body[:usable])
260
+
261
+ expected = header.shape_e * header.shape_v
262
+ if len(values) != expected:
263
+ _warn(
264
+ f"{path}: decoded {len(values)} element(s), expected "
265
+ f"shape_e*shape_v={expected} -- file is likely truncated or "
266
+ f"a compressed block failed to decompress (see any earlier "
267
+ f"GRDParseWarning for which); returning the {len(values)} "
268
+ f"element(s) actually decoded"
269
+ )
270
+
271
+ return header, values
272
+
273
+
274
+ def dummy_value(header: GrdHeader):
275
+ typecode = _array_typecode(header.element_size, header.sign_flag)
276
+ return _DUMMIES[typecode]
277
+
278
+
279
+ if __name__ == "__main__":
280
+ import sys
281
+
282
+ if len(sys.argv) != 2:
283
+ print("usage: python -m pygdb.grd_reader <path-to.grd>")
284
+ raise SystemExit(1)
285
+
286
+ header, values = read_grd(sys.argv[1])
287
+ print(f"shape: {header.shape_e} x {header.shape_v} (ordering={header.ordering})")
288
+ print(f"compressed: {header.is_compressed} element_size: {header.element_size} bytes")
289
+ print(f"origin: ({header.x_origin}, {header.y_origin}) spacing: ({header.spacing_e}, {header.spacing_v})")
290
+ print(f"z scale: base={header.base_value} mult={header.data_factor}")
291
+ print(f"dummy value: {dummy_value(header)}")
292
+ print(f"decoded {len(values)} elements; first 5: {values[:5]}")
pygdb/lzrw1.py ADDED
@@ -0,0 +1,323 @@
1
+ """
2
+ Canonical LZRW1 decompressor (Ross Williams, 1991), ported from his own
3
+ public-domain reference C source
4
+ (http://www.ross.net/compression/download/original/old_lzrw1.c,
5
+ explicitly marked "This code is public domain" in its own header
6
+ comment), used here to decode Geosoft `.gdb` DB_COMP_SPEED page data.
7
+
8
+ [CONFIRMED] finding (see ../docs/provenance/notes.md section 6.5c/6.5d for the full
9
+ derivation): Geosoft's DB_COMP_SPEED mode wraps *exactly* Ross Williams'
10
+ canonical LZRW1 byte stream -- the same 2-byte control word + 1-byte
11
+ literal / 2-byte nibble-packed copy-item scheme as his original
12
+ `lzrw1_decompress()` -- with these Geosoft-specific differences from the
13
+ reference C wrapper:
14
+
15
+ 1. No 4-byte FLAG_BYTES prefix (the reference C code's own
16
+ FLAG_COMPRESS/FLAG_COPY byte + 3 padding bytes) -- the control word
17
+ starts immediately for a compressed chunk.
18
+ 2. Each chunk (which may span several of the file's physical
19
+ 1024-byte pages) is preceded by a 28-byte Geosoft-specific wrapper,
20
+ not part of LZRW1 itself:
21
+ - 16 bytes: the magic sub-header shared with the `.grd` sibling
22
+ format and with `.gdb`'s DB_COMP_SIZE (zlib) mode:
23
+ `0f 0e ff fe 12 34 56 78 <subtype int32> <reserved int32>`
24
+ -- subtype is 1 for DB_COMP_SPEED (2 for DB_COMP_SIZE).
25
+ - 12 bytes: `<decompressed_length int32> <chunk_length int32>
26
+ <marker int32>`. `chunk_length` INCLUDES these 12 bytes (i.e.
27
+ chunk_length - 12 == the number of raw bytes that follow,
28
+ compressed or not -- see point 3).
29
+ 3. **`marker` is itself a real flag, not just a validation sentinel**
30
+ (this was missed in an earlier pass that only tested against 4 of
31
+ the 10 real Speed files in this project's full sample set -- the
32
+ other 6, found and validated later, include real examples of the
33
+ second case below and are what exposed it):
34
+ - `marker == 0xF4E5D6C7` (-186263865 signed): this chunk's
35
+ payload is genuine LZRW1-compressed data (Ross Williams' own
36
+ reference implementation has an analogous `FLAG_COMPRESS`
37
+ case, using a different byte value/position; Geosoft's variant
38
+ repurposes this 4-byte marker field for the same purpose).
39
+ - `marker == 0xF0E1D2C3` (-253635901 signed): this chunk's
40
+ payload is **stored raw, uncompressed** -- i.e. Geosoft's
41
+ equivalent of Ross Williams' reference `FLAG_COPY` case (used
42
+ when LZRW1 compression didn't shrink the data, so the encoder
43
+ gave up and stored it verbatim instead). For every real
44
+ instance of this found, `chunk_length - 12 == decompressed_length`
45
+ exactly (no compression ratio at all, consistent with "stored
46
+ raw"), and reading `decompressed_length` bytes directly (no
47
+ decompression) produces plausible real survey data (checked as
48
+ float64: smoothly-varying, physically sane gravity/magnetic
49
+ values). Every real marker value found across all 10 real
50
+ Speed files was one of these two constants -- zero exceptions,
51
+ zero unrecognized third values.
52
+
53
+ Validated exactly (not just "plausibly") against **all 10 real**
54
+ DB_COMP_SPEED files now in this project's sample set (the original 4
55
+ GEOTEM/Questem EM/Mag files, plus 4 AGG/Mag files from the Melinda Downs
56
+ delivery, plus 2 Rad/Mag files from the Georgetown-AGSO delivery,
57
+ extracted specifically to check this): for every chunk in every file --
58
+ not a sample -- decoding exactly `decompressed_length` output bytes
59
+ (via LZRW1 decompression when marker indicates compressed, or a direct
60
+ copy when marker indicates raw/stored) consumes exactly
61
+ `chunk_length - 12` input bytes, with zero slack, and zero chunks with
62
+ an unrecognized marker value.
63
+
64
+ No Geosoft software of any kind was used to derive or produce this
65
+ module -- only Ross Williams' own public-domain reference source (read,
66
+ not executed against anything proprietary) and real, independently
67
+ obtained `.gdb` files.
68
+ """
69
+
70
+ from __future__ import annotations
71
+
72
+ import struct
73
+ import warnings
74
+ from dataclasses import dataclass
75
+
76
+ try:
77
+ from . import _native as _native_ext
78
+ except ImportError:
79
+ _native_ext = None
80
+
81
+ CHUNK_MAGIC = bytes.fromhex("0f0efffe12345678")
82
+ DB_COMP_SPEED = 1
83
+ DB_COMP_SIZE = 2
84
+
85
+ # The two real marker values observed in the 12-byte chunk length
86
+ # sub-header, across all 10 real DB_COMP_SPEED files in this project,
87
+ # with zero exceptions and zero unrecognized third values.
88
+ MARKER_COMPRESSED = -186263865 # 0xF4E5D6C7 -- payload is real LZRW1
89
+ MARKER_STORED_RAW = -253635901 # 0xF0E1D2C3 -- payload is stored verbatim
90
+
91
+ # Backwards-compatible alias (earlier name, before the raw/stored case
92
+ # was found).
93
+ KNOWN_MARKER = MARKER_COMPRESSED
94
+
95
+
96
+ def lzrw1_decompress(data: bytes, start: int, decompressed_length: int) -> bytes:
97
+ """
98
+ Decompress exactly `decompressed_length` bytes of canonical LZRW1
99
+ data (Ross Williams' algorithm, no FLAG_BYTES prefix) starting at
100
+ `data[start:]`. Returns the decompressed bytes.
101
+
102
+ Dispatches to the compiled `pygdb._native` extension when it's
103
+ available (same algorithm, ported to Rust -- see `rust/src/lib.rs`;
104
+ ~16x faster on real DB_COMP_SPEED data, since this per-byte loop is
105
+ this reader's one CPU-bound hot path), falling back to the pure-Python
106
+ `_lzrw1_decompress_py` below when it isn't. `_native` raises
107
+ `IndexError` under the same truncated/corrupt-input conditions as the
108
+ pure-Python version, so callers (`decode_speed_chunk`) don't need to
109
+ know which backend produced the error.
110
+ """
111
+ if _native_ext is not None:
112
+ return bytes(_native_ext.lzrw1_decompress(data, start, decompressed_length))
113
+ return _lzrw1_decompress_py(data, start, decompressed_length)
114
+
115
+
116
+ def _lzrw1_decompress_py(data: bytes, start: int, decompressed_length: int) -> bytes:
117
+ """
118
+ Pure-Python reference implementation of `lzrw1_decompress` -- kept as
119
+ the always-available fallback when `pygdb._native` isn't built, and
120
+ as the documented, clean-room-derived source of truth for the
121
+ algorithm.
122
+
123
+ This is a direct, literal port of the core loop in Ross Williams'
124
+ own public-domain `lzrw1_decompress()` (see module docstring for the
125
+ source URL) -- same control-word/control-bit walk, same nibble
126
+ packing for copy-item offset/length. The only functional change from
127
+ his reference is that this operates on a `bytes` object with an
128
+ explicit output-length stop condition instead of a fixed-size output
129
+ buffer, and does not skip his 4-byte FLAG_BYTES prefix (Geosoft's
130
+ on-disk chunks don't have it -- the equivalent flag lives in the
131
+ 12-byte length sub-header's `marker` field instead, see
132
+ `decode_speed_chunk`).
133
+ """
134
+ p = start
135
+ out = bytearray()
136
+ while len(out) < decompressed_length:
137
+ control = data[p] | (data[p + 1] << 8)
138
+ p += 2
139
+ for _bit in range(16):
140
+ if len(out) >= decompressed_length:
141
+ break
142
+ if control & 1:
143
+ b0 = data[p]
144
+ b1 = data[p + 1]
145
+ p += 2
146
+ offset = ((b0 & 0xF0) << 4) + b1
147
+ length = (b0 & 0x0F) + 1
148
+ start_idx = len(out) - offset
149
+ for i in range(length):
150
+ if len(out) >= decompressed_length:
151
+ break
152
+ out.append(out[start_idx + i])
153
+ else:
154
+ out.append(data[p])
155
+ p += 1
156
+ control >>= 1
157
+ return bytes(out)
158
+
159
+
160
+ @dataclass
161
+ class SpeedChunk:
162
+ magic_offset: int # file offset of the 16-byte magic sub-header
163
+ subtype: int # 1 = DB_COMP_SPEED, 2 = DB_COMP_SIZE
164
+ decompressed_length: int
165
+ chunk_length: int # includes the 12-byte length sub-header
166
+ marker: int
167
+ payload_offset: int # file offset where the payload starts
168
+
169
+ @property
170
+ def is_stored_raw(self) -> bool:
171
+ return self.marker == MARKER_STORED_RAW
172
+
173
+ @property
174
+ def is_compressed(self) -> bool:
175
+ return self.marker == MARKER_COMPRESSED
176
+
177
+
178
+ class LZRW1DecodeError(Exception):
179
+ """
180
+ Raised by `decode_speed_chunk`/`parse_chunk_header` when a Speed-mode
181
+ chunk can't be decoded: truncated/corrupt data, or a subtype/marker
182
+ this module doesn't recognize. A single, deliberately narrow
183
+ exception type (rather than a bare `AssertionError`/`IndexError`/
184
+ `struct.error` grab-bag) so callers -- notably
185
+ `gdb_reader.read_blob_values` -- can catch exactly this and fail
186
+ gracefully (return whatever was already decoded elsewhere, emit a
187
+ clear warning) instead of crashing. See docs/provenance/notes.md's "reader
188
+ robustness" notes for the design rationale; this reader is not meant
189
+ to hard-crash on a truncated download or an unrecognized real-world
190
+ variant, per an explicit engineering request from the coordinator.
191
+ """
192
+
193
+
194
+ def parse_chunk_header(data: bytes, magic_offset: int) -> SpeedChunk:
195
+ try:
196
+ subtype, _reserved = struct.unpack_from("<ii", data, magic_offset + 8)
197
+ header_start = magic_offset + 16
198
+ decompressed_length, chunk_length, marker = struct.unpack_from(
199
+ "<iii", data, header_start
200
+ )
201
+ except struct.error as e:
202
+ raise LZRW1DecodeError(
203
+ f"not enough bytes at offset {magic_offset} to read a full chunk header "
204
+ f"(need 28 bytes from the magic; only {len(data) - magic_offset} available) "
205
+ f"-- truncated data"
206
+ ) from e
207
+ return SpeedChunk(
208
+ magic_offset=magic_offset,
209
+ subtype=subtype,
210
+ decompressed_length=decompressed_length,
211
+ chunk_length=chunk_length,
212
+ marker=marker,
213
+ payload_offset=header_start + 12,
214
+ )
215
+
216
+
217
+ def decode_speed_chunk(data: bytes, chunk: SpeedChunk) -> bytes:
218
+ """
219
+ Decode one DB_COMP_SPEED chunk's data -- transparently handling both
220
+ the LZRW1-compressed case and the stored-raw case (see module
221
+ docstring point 3). Raises `LZRW1DecodeError` (never a bare
222
+ `AssertionError`/`IndexError`) if the chunk doesn't look like a
223
+ real, well-formed Speed chunk (wrong subtype, implausible lengths,
224
+ an unrecognized marker value, or truncated payload data) -- callers
225
+ should catch this one type and treat it as "this chunk couldn't be
226
+ decoded," not as a program bug.
227
+ """
228
+ if chunk.subtype != DB_COMP_SPEED:
229
+ raise LZRW1DecodeError(f"not a Speed chunk (subtype={chunk.subtype})")
230
+ if not (0 < chunk.decompressed_length < 200_000_000):
231
+ raise LZRW1DecodeError(
232
+ f"implausible decompressed_length={chunk.decompressed_length} -- "
233
+ f"likely a misaligned or corrupt chunk header"
234
+ )
235
+ if chunk.is_stored_raw:
236
+ if chunk.chunk_length - 12 != chunk.decompressed_length:
237
+ raise LZRW1DecodeError(
238
+ "stored-raw chunk should have chunk_length-12 == decompressed_length "
239
+ f"(got chunk_length-12={chunk.chunk_length - 12}, "
240
+ f"decompressed_length={chunk.decompressed_length})"
241
+ )
242
+ payload = data[chunk.payload_offset: chunk.payload_offset + chunk.decompressed_length]
243
+ if len(payload) < chunk.decompressed_length:
244
+ raise LZRW1DecodeError(
245
+ f"truncated stored-raw payload: expected {chunk.decompressed_length} "
246
+ f"byte(s), only {len(payload)} available -- file cut off mid-chunk?"
247
+ )
248
+ return payload
249
+ if not chunk.is_compressed:
250
+ raise LZRW1DecodeError(f"unrecognized marker value: {chunk.marker}")
251
+ try:
252
+ return lzrw1_decompress(data, chunk.payload_offset, chunk.decompressed_length)
253
+ except IndexError as e:
254
+ raise LZRW1DecodeError(
255
+ f"ran out of input data while decompressing (need to produce "
256
+ f"{chunk.decompressed_length} bytes) -- truncated or corrupt payload"
257
+ ) from e
258
+
259
+
260
+ def find_speed_chunks(data: bytes):
261
+ """
262
+ Yield SpeedChunk for every DB_COMP_SPEED (subtype==1) chunk found in
263
+ `data` by scanning for the shared 16-byte magic. A magic-byte match
264
+ too close to the end of `data` to hold a full 28-byte chunk header
265
+ (e.g. a truncated file, or a coincidental match inside real data
266
+ right before EOF -- both real, observed cases elsewhere in this
267
+ project) is skipped with a warning rather than raising -- this is a
268
+ scanning helper, not a strict decoder, so it degrades gracefully and
269
+ keeps looking rather than aborting the whole scan.
270
+ """
271
+ start = 0
272
+ while True:
273
+ idx = data.find(CHUNK_MAGIC, start)
274
+ if idx == -1:
275
+ return
276
+ try:
277
+ chunk = parse_chunk_header(data, idx)
278
+ except LZRW1DecodeError as e:
279
+ warnings.warn(
280
+ f"find_speed_chunks: skipping a magic-byte match at offset {idx} "
281
+ f"that isn't a full chunk header ({e})",
282
+ RuntimeWarning,
283
+ stacklevel=2,
284
+ )
285
+ start = idx + 1
286
+ continue
287
+ if chunk.subtype == DB_COMP_SPEED:
288
+ yield chunk
289
+ start = idx + 1
290
+
291
+
292
+ if __name__ == "__main__":
293
+ import sys
294
+
295
+ path = sys.argv[1] if len(sys.argv) > 1 else "samples/GSQ_Data/extracted/holroy/em000293/DB_EM_293.gdb"
296
+ data = open(path, "rb").read()
297
+ n_ok = 0
298
+ n_compressed = 0
299
+ n_stored = 0
300
+ n_bad_marker = 0
301
+ n_total = 0
302
+ for chunk in find_speed_chunks(data):
303
+ n_total += 1
304
+ try:
305
+ out = decode_speed_chunk(data, chunk)
306
+ except LZRW1DecodeError as e:
307
+ n_bad_marker += 1
308
+ if n_bad_marker <= 3:
309
+ print(f"chunk @ {chunk.magic_offset}: FAILED ({e})")
310
+ continue
311
+ n_ok += 1
312
+ if chunk.is_stored_raw:
313
+ n_stored += 1
314
+ else:
315
+ n_compressed += 1
316
+ if n_total <= 5:
317
+ print(
318
+ f"chunk @ {chunk.magic_offset}: decompressed_length={chunk.decompressed_length} "
319
+ f"chunk_length={chunk.chunk_length} kind={'stored-raw' if chunk.is_stored_raw else 'compressed'} "
320
+ f"first_16_bytes={out[:16].hex()}"
321
+ )
322
+ print(f"\n{n_total} chunks total: {n_compressed} compressed, {n_stored} stored-raw, "
323
+ f"{n_bad_marker} unrecognized marker")