python-gdb 0.1.0__cp312-abi3-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pygdb/__init__.py +60 -0
- pygdb/_native.pyd +0 -0
- pygdb/gdb.py +641 -0
- pygdb/gdb_reader.py +1368 -0
- pygdb/grd_reader.py +292 -0
- pygdb/lzrw1.py +323 -0
- pygdb/registry.py +104 -0
- python_gdb-0.1.0.dist-info/METADATA +183 -0
- python_gdb-0.1.0.dist-info/RECORD +12 -0
- python_gdb-0.1.0.dist-info/WHEEL +4 -0
- python_gdb-0.1.0.dist-info/licenses/LICENSE +21 -0
- python_gdb-0.1.0.dist-info/sboms/pygdb-native.cyclonedx.json +572 -0
pygdb/gdb_reader.py
ADDED
|
@@ -0,0 +1,1368 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Clean-room reader for Geosoft .gdb database files.
|
|
3
|
+
|
|
4
|
+
Status: reads the file magic/header, walks the channel and line symbol
|
|
5
|
+
tables, and reads real channel data for every compression mode
|
|
6
|
+
(DB_COMP_NONE, DB_COMP_SPEED/LZRW1, DB_COMP_SIZE/zlib, single- or
|
|
7
|
+
multi-page blobs, and the "bare"/uncompressed-blob-inside-a-compressed-
|
|
8
|
+
file variant) via iter_blobs()/find_blob()/read_blob_values() -- see
|
|
9
|
+
docs/provenance/notes.md sections 6.6/6.6b/6.6d for the full derivation.
|
|
10
|
+
Still open: REG/IPJ registry content is only partially decoded (section
|
|
11
|
+
6.7/6.8), and a handful of header/record fields remain [UNKNOWN] -- see
|
|
12
|
+
docs/spec.md and docs/provenance/notes.md for the complete, current
|
|
13
|
+
picture.
|
|
14
|
+
|
|
15
|
+
Robustness: this reader is designed to degrade gracefully rather than
|
|
16
|
+
hard-crash on a blob/chunk/record it can't parse -- a truncated file
|
|
17
|
+
(cut-off download, or a blob chain that runs past EOF), an
|
|
18
|
+
administrative-blob variant it doesn't recognize, an unrecognized
|
|
19
|
+
channel type, or anything else that doesn't fit the confirmed
|
|
20
|
+
structure. Functions return whatever they successfully decoded up to
|
|
21
|
+
the point of trouble (an empty list/dict in the worst case) rather than
|
|
22
|
+
raising, and always pair that with a `GDBParseWarning` (see its
|
|
23
|
+
docstring) identifying what couldn't be decoded and why. This is a
|
|
24
|
+
deliberate engineering choice, not new format research -- see docs/provenance/notes.md
|
|
25
|
+
for the design rationale and docs/provenance/log.md for when/why it was added.
|
|
26
|
+
|
|
27
|
+
Confidence markers below mirror docs/provenance/notes.md: [CONFIRMED] =
|
|
28
|
+
verified against two independent real files with a falsifiable
|
|
29
|
+
structural test; [LIKELY] = passed one real test but not independently
|
|
30
|
+
cross-checked; [GUESS] = plausible pattern, not tested; values otherwise
|
|
31
|
+
unlabeled in comments are the [UNKNOWN] raw offsets, kept for whoever
|
|
32
|
+
continues this.
|
|
33
|
+
|
|
34
|
+
Every numeric constant used for interpretation (GS_* type codes,
|
|
35
|
+
DB_CHAN_FORMAT_*, DB_SYMB_NAME_SIZE, etc.) comes from reading Geosoft's
|
|
36
|
+
own published, BSD-licensed source at
|
|
37
|
+
https://github.com/GeosoftInc/gxpy/blob/master/geosoft/gxapi/__init__.py
|
|
38
|
+
-- publicly available vendor source, not obtained by running anything.
|
|
39
|
+
|
|
40
|
+
No Geosoft software of any kind was installed, imported, or executed to
|
|
41
|
+
produce this code.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
from __future__ import annotations
|
|
45
|
+
|
|
46
|
+
import contextlib
|
|
47
|
+
import re
|
|
48
|
+
import struct
|
|
49
|
+
import warnings
|
|
50
|
+
import zlib
|
|
51
|
+
from dataclasses import dataclass
|
|
52
|
+
|
|
53
|
+
from typing import BinaryIO, List, Optional, Tuple
|
|
54
|
+
|
|
55
|
+
import numpy as np
|
|
56
|
+
|
|
57
|
+
from . import lzrw1 as _lzrw1
|
|
58
|
+
|
|
59
|
+
try:
|
|
60
|
+
from . import _native as _native_ext
|
|
61
|
+
except ImportError:
|
|
62
|
+
_native_ext = None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class GDBParseWarning(RuntimeWarning):
|
|
66
|
+
"""
|
|
67
|
+
Warned (via `warnings.warn`) whenever this reader hits a blob,
|
|
68
|
+
chunk, or record it can't parse -- an unexpected byte sequence, a
|
|
69
|
+
file that ends prematurely (truncated download, or a blob chain
|
|
70
|
+
that runs past EOF), an administrative-blob variant it doesn't
|
|
71
|
+
recognize, or anything else that doesn't fit the confirmed
|
|
72
|
+
structure. This reader is designed to degrade gracefully rather
|
|
73
|
+
than hard-crash on this whole class of problem: functions return
|
|
74
|
+
whatever they successfully decoded up to the point of trouble
|
|
75
|
+
(a shorter-than-expected list, an empty list, or in the worst case
|
|
76
|
+
an empty result) instead of raising, and a `GDBParseWarning`
|
|
77
|
+
describing what couldn't be decoded and why is always issued
|
|
78
|
+
alongside, so a caller can tell a clean, complete result from a
|
|
79
|
+
partial one and go investigate. See docs/provenance/notes.md's "reader robustness"
|
|
80
|
+
notes for the design rationale (an explicit engineering request,
|
|
81
|
+
not new format research).
|
|
82
|
+
|
|
83
|
+
This does not apply to a handful of genuine precondition failures
|
|
84
|
+
that aren't "this file has an interesting anomaly" (e.g. calling
|
|
85
|
+
`read_blob_values` with a `channel`/`blob` pair that can't
|
|
86
|
+
possibly match) -- those still raise normally.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _warn(msg: str) -> None:
|
|
91
|
+
warnings.warn(msg, GDBParseWarning, stacklevel=3)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
MAGIC = b"!CBD"
|
|
95
|
+
# The 16-byte header opening was found byte-identical across both real
|
|
96
|
+
# sample files (Magnetic_Data.gdb and Radiometric_Data.gdb) -- see
|
|
97
|
+
# docs/provenance/notes.md 6.1. Treated here as a fixed format/version signature.
|
|
98
|
+
HEADER_SIGNATURE = bytes.fromhex("21434244000000000000021008010000".replace(" ", ""))[:16]
|
|
99
|
+
|
|
100
|
+
SYMBOL_RECORD_SIZE = 128 # [CONFIRMED] -- constant stride of every symbol
|
|
101
|
+
# table record (channel, line, and user records
|
|
102
|
+
# all observed at this stride)
|
|
103
|
+
|
|
104
|
+
# Vendor-published GX type codes (geosoft/gxapi/__init__.py). Positive
|
|
105
|
+
# values only here -- negative values in a channel record instead mean
|
|
106
|
+
# "string, N bytes wide" where N = -value (see decode_dtype below).
|
|
107
|
+
GS_TYPE_NAMES = {
|
|
108
|
+
0: "GS_BYTE",
|
|
109
|
+
1: "GS_USHORT",
|
|
110
|
+
2: "GS_SHORT",
|
|
111
|
+
3: "GS_LONG",
|
|
112
|
+
4: "GS_FLOAT",
|
|
113
|
+
5: "GS_DOUBLE",
|
|
114
|
+
6: "GS_UBYTE",
|
|
115
|
+
7: "GS_ULONG",
|
|
116
|
+
8: "GS_LONG64",
|
|
117
|
+
9: "GS_ULONG64",
|
|
118
|
+
10: "GS_FLOAT3D",
|
|
119
|
+
11: "GS_DOUBLE3D",
|
|
120
|
+
12: "GS_FLOAT2D",
|
|
121
|
+
13: "GS_DOUBLE2D",
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
# struct format codes for each GS_* type -- used for `_element_width`'s
|
|
125
|
+
# byte-width math (struct.calcsize), independent of whichever decode
|
|
126
|
+
# strategy actually reads the bytes.
|
|
127
|
+
GS_TYPE_STRUCT = {
|
|
128
|
+
0: "b", # GS_BYTE (signed, per GS_S1* constants)
|
|
129
|
+
1: "H", # GS_USHORT
|
|
130
|
+
2: "h", # GS_SHORT
|
|
131
|
+
3: "i", # GS_LONG
|
|
132
|
+
4: "f", # GS_FLOAT
|
|
133
|
+
5: "d", # GS_DOUBLE
|
|
134
|
+
6: "B", # GS_UBYTE
|
|
135
|
+
7: "I", # GS_ULONG
|
|
136
|
+
8: "q", # GS_LONG64
|
|
137
|
+
9: "Q", # GS_ULONG64
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
# Little-endian numpy dtype strings for each GS_* type (explicit `<`
|
|
141
|
+
# byte-order prefix, matching this format's confirmed little-endian
|
|
142
|
+
# layout everywhere else -- a platform-native dtype would silently
|
|
143
|
+
# misdecode on a big-endian host). Used by `_decode_numeric_or_string`
|
|
144
|
+
# for `np.frombuffer`; `GS_TYPE_STRUCT` above is kept separately since
|
|
145
|
+
# `_element_width` only needs a byte count, not a full dtype.
|
|
146
|
+
GS_TYPE_NUMPY_DTYPE = {
|
|
147
|
+
0: "<i1", # GS_BYTE (signed)
|
|
148
|
+
1: "<u2", # GS_USHORT
|
|
149
|
+
2: "<i2", # GS_SHORT
|
|
150
|
+
3: "<i4", # GS_LONG
|
|
151
|
+
4: "<f4", # GS_FLOAT
|
|
152
|
+
5: "<f8", # GS_DOUBLE
|
|
153
|
+
6: "<u1", # GS_UBYTE
|
|
154
|
+
7: "<u4", # GS_ULONG
|
|
155
|
+
8: "<i8", # GS_LONG64
|
|
156
|
+
9: "<u8", # GS_ULONG64
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
DB_CHAN_FORMAT_NAMES = {
|
|
160
|
+
0: "NORMAL",
|
|
161
|
+
1: "EXP",
|
|
162
|
+
2: "TIME",
|
|
163
|
+
3: "DATE",
|
|
164
|
+
4: "GEOGR",
|
|
165
|
+
5: "SIGDIG",
|
|
166
|
+
6: "HEX",
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
# Vendor-published DB_ARRAY_BASETYPE_* constants (geosoft/gxapi/__init__.py).
|
|
170
|
+
# [LIKELY] match for the int16 field at relative offset +86 -- see
|
|
171
|
+
# docs/provenance/notes.md "VA / array channels" section. Confirmed to hold value 1
|
|
172
|
+
# (TIME_WINDOWS) on real multi-gate TEM decay-curve array channels, but
|
|
173
|
+
# also seen as a constant non-zero value across *every* channel (including
|
|
174
|
+
# obviously-scalar ones) in three older real files, so treat this field's
|
|
175
|
+
# meaning with more caution than the array-width field below.
|
|
176
|
+
DB_ARRAY_BASETYPE_NAMES = {
|
|
177
|
+
0: "NONE",
|
|
178
|
+
1: "TIME_WINDOWS",
|
|
179
|
+
2: "TIMES",
|
|
180
|
+
3: "FREQUENCIES",
|
|
181
|
+
4: "ELEVATIONS",
|
|
182
|
+
5: "DEPTHS",
|
|
183
|
+
6: "VELOCITIES",
|
|
184
|
+
7: "DISCRETE_TIME_WINDOWS",
|
|
185
|
+
8: "ENERGIES",
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
@dataclass(eq=False)
|
|
190
|
+
class ChannelRecord:
|
|
191
|
+
# eq=False -- keep the default identity-based __eq__/__hash__ instead
|
|
192
|
+
# of dataclass's usual field-by-field one, so instances stay hashable
|
|
193
|
+
# (GDB.iter_line() yields these and documents `dict(...)` keyed by
|
|
194
|
+
# the record itself as safe -- see its docstring -- which needs
|
|
195
|
+
# __hash__ to actually work). Value equality between two separately-
|
|
196
|
+
# constructed-but-identical records is never used anywhere in this
|
|
197
|
+
# codebase; every real lookup returns the same cached instance from
|
|
198
|
+
# GDB.channels, so identity is all that's ever needed in practice.
|
|
199
|
+
index: int
|
|
200
|
+
offset: int
|
|
201
|
+
name: str
|
|
202
|
+
dtype_code: int # raw int16 value: positive=GS_* type, negative=-string_width
|
|
203
|
+
format_code: int
|
|
204
|
+
raw: bytes
|
|
205
|
+
array_width: int = 1 # [CONFIRMED] relative offset +118, int16. 1 = plain
|
|
206
|
+
# scalar channel (the overwhelming majority of real
|
|
207
|
+
# channels seen). >1 = a true VA/array channel
|
|
208
|
+
# storing that many elements per fiducial "cell" --
|
|
209
|
+
# e.g. 24 (time-decay gates) or 30 (depth layers) in
|
|
210
|
+
# the real AG106386 Georgetown conductivity file.
|
|
211
|
+
# Independently cross-checked against that same
|
|
212
|
+
# file's plain-text ASCII sibling (.dfn) format,
|
|
213
|
+
# which spells out "30F10.4" (Fortran-style: 30
|
|
214
|
+
# repetitions of a float field) for the exact same
|
|
215
|
+
# channel name -- see docs/provenance/notes.md.
|
|
216
|
+
array_basetype_code: int = 0 # [LIKELY] relative offset +86, int16.
|
|
217
|
+
name_is_clean: bool = True # False = name field was NUL-unterminated / had
|
|
218
|
+
# non-printable bytes -- see docs/provenance/notes.md re: older
|
|
219
|
+
# (pre-2020, e.g. 1990s GEOTEM) files sometimes
|
|
220
|
+
# leaving unused capacity slots un-zeroed rather
|
|
221
|
+
# than clean, unlike the 2020 USGS samples.
|
|
222
|
+
|
|
223
|
+
@property
|
|
224
|
+
def is_string(self) -> bool:
|
|
225
|
+
return self.dtype_code < 0
|
|
226
|
+
|
|
227
|
+
@property
|
|
228
|
+
def string_width(self) -> Optional[int]:
|
|
229
|
+
return -self.dtype_code if self.is_string else None
|
|
230
|
+
|
|
231
|
+
@property
|
|
232
|
+
def type_name(self) -> str:
|
|
233
|
+
if self.is_string:
|
|
234
|
+
return f"string[{self.string_width}]"
|
|
235
|
+
return GS_TYPE_NAMES.get(self.dtype_code, f"unknown({self.dtype_code})")
|
|
236
|
+
|
|
237
|
+
@property
|
|
238
|
+
def format_name(self) -> str:
|
|
239
|
+
return DB_CHAN_FORMAT_NAMES.get(self.format_code, f"unknown({self.format_code})")
|
|
240
|
+
|
|
241
|
+
@property
|
|
242
|
+
def is_array(self) -> bool:
|
|
243
|
+
"""True for a real VA/array channel (array_width > 1). [CONFIRMED]."""
|
|
244
|
+
return self.array_width > 1
|
|
245
|
+
|
|
246
|
+
@property
|
|
247
|
+
def array_basetype_name(self) -> str:
|
|
248
|
+
return DB_ARRAY_BASETYPE_NAMES.get(
|
|
249
|
+
self.array_basetype_code, f"unknown({self.array_basetype_code})"
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
@property
|
|
253
|
+
def looks_sane(self) -> bool:
|
|
254
|
+
"""
|
|
255
|
+
Heuristic sanity check distinguishing a real channel record from
|
|
256
|
+
leftover-garbage bytes that happen to decode a clean printable
|
|
257
|
+
name (observed for real in DB_Mag_833.gdb -- see docs/provenance/notes.md). Real
|
|
258
|
+
records seen so far always have dtype either a known GS_* code
|
|
259
|
+
(0-13) or a small negative string width, and a format code in the
|
|
260
|
+
known DB_CHAN_FORMAT_* range (0-6).
|
|
261
|
+
"""
|
|
262
|
+
dtype_ok = self.dtype_code in GS_TYPE_NAMES or -256 <= self.dtype_code < 0
|
|
263
|
+
format_ok = self.format_code in DB_CHAN_FORMAT_NAMES
|
|
264
|
+
return dtype_ok and format_ok
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _read_name(raw: bytes, offset: int, max_len: int = 64):
|
|
268
|
+
"""
|
|
269
|
+
Read a NUL-padded name field.
|
|
270
|
+
|
|
271
|
+
Returns (name, is_clean). is_clean is False when the field has no NUL
|
|
272
|
+
terminator within max_len, or contains non-printable bytes before the
|
|
273
|
+
terminator -- observed [CONFIRMED against a real 1992 file,
|
|
274
|
+
DB_Mag_293.gdb from GSQ's Holroy River survey] to happen for *unused*
|
|
275
|
+
channel-table capacity slots in at least one older (pre-2020) real
|
|
276
|
+
.gdb file: unlike the 2020 USGS samples (where unused capacity slots
|
|
277
|
+
are cleanly zeroed), this older file leaves unused slots holding
|
|
278
|
+
leftover/uninitialized bytes that happen to look like binary float
|
|
279
|
+
data, not padding. Treat is_clean=False slots as "unused capacity,
|
|
280
|
+
contents undefined" rather than as real channels.
|
|
281
|
+
"""
|
|
282
|
+
field = raw[offset : offset + max_len]
|
|
283
|
+
nul = field.find(b"\x00")
|
|
284
|
+
if nul == -1:
|
|
285
|
+
return field.decode("ascii", errors="replace"), False
|
|
286
|
+
text = field[:nul]
|
|
287
|
+
is_clean = all(32 <= b < 127 for b in text)
|
|
288
|
+
return text.decode("ascii", errors="replace"), is_clean
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def check_magic(data: bytes) -> bool:
|
|
292
|
+
"""
|
|
293
|
+
[CONFIRMED] the 4-byte "!CBD" prefix against 9/9 real files across two
|
|
294
|
+
independent sources (2020 USGS Mojave survey, 1990s-2020s GSQ
|
|
295
|
+
Queensland surveys from three different TEM systems/vendors).
|
|
296
|
+
|
|
297
|
+
The FULL 16-byte HEADER_SIGNATURE is only [LIKELY] -- it matched
|
|
298
|
+
exactly in 8/9 real files, but one real file
|
|
299
|
+
(DB_Mag_Elaine_1003.gdb, from GSQ's Mount Gordon delivery) has
|
|
300
|
+
`f0 f0 f0 f0` at bytes 8-11 instead of the usual `00 00 00 00`. That
|
|
301
|
+
file is otherwise structurally normal (chans_max/users_max/page_size
|
|
302
|
+
all decode sanely), so this looks like a real, if rare, variation in
|
|
303
|
+
that sub-block rather than a different format entirely -- flagged
|
|
304
|
+
[UNKNOWN] in docs/provenance/notes.md. Only the 4-byte magic is treated as a hard
|
|
305
|
+
requirement here; the rest of the signature is reported separately.
|
|
306
|
+
"""
|
|
307
|
+
return data[:4] == MAGIC
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def magic_signature_matches_common_case(data: bytes) -> bool:
|
|
311
|
+
"""True if bytes 0-15 exactly match the signature seen in most real files."""
|
|
312
|
+
return data[:16] == HEADER_SIGNATURE
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def header_fields(data: bytes) -> dict:
|
|
316
|
+
"""
|
|
317
|
+
Extract the header int32 fields whose approximate meaning we have
|
|
318
|
+
some confidence in. See docs/provenance/notes.md section 6.1 for the full table
|
|
319
|
+
including the still-unknown offsets, and for why each confidence
|
|
320
|
+
label was assigned.
|
|
321
|
+
|
|
322
|
+
Fails gracefully on a truncated/too-short header: any field that
|
|
323
|
+
can't be read (not enough bytes at its offset) is set to `None`
|
|
324
|
+
in the returned dict rather than raising, and a `GDBParseWarning`
|
|
325
|
+
is issued naming which field(s) were affected. Callers that need a
|
|
326
|
+
field should check for `None` before using it (every function in
|
|
327
|
+
this module that consumes `header_fields()` output does).
|
|
328
|
+
"""
|
|
329
|
+
result = {}
|
|
330
|
+
for name, offset in (("chans_max", 24), ("users_max", 40),
|
|
331
|
+
("page_size", 100), ("comp_level", 120)):
|
|
332
|
+
try:
|
|
333
|
+
result[name] = struct.unpack_from("<i", data, offset)[0]
|
|
334
|
+
except struct.error:
|
|
335
|
+
_warn(
|
|
336
|
+
f"header truncated: only {len(data)} byte(s) available, not enough "
|
|
337
|
+
f"to read '{name}' at offset {offset} -- returning None for it"
|
|
338
|
+
)
|
|
339
|
+
result[name] = None
|
|
340
|
+
# comp_level==1 (DB_COMP_SPEED) does NOT mean the payload is zlib --
|
|
341
|
+
# confirmed it is NOT (docs/provenance/notes.md section 6.5b), it's canonical LZRW1
|
|
342
|
+
# (section 6.5c). comp_level==2 (DB_COMP_SIZE) IS confirmed real zlib.
|
|
343
|
+
return result
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _parse_channel_record(data: bytes, rec_start: int, index: int) -> ChannelRecord:
|
|
347
|
+
raw = data[rec_start : rec_start + SYMBOL_RECORD_SIZE]
|
|
348
|
+
name, is_clean = _read_name(raw, 8)
|
|
349
|
+
dtype_code = struct.unpack_from("<h", raw, 84)[0]
|
|
350
|
+
array_basetype_code = struct.unpack_from("<h", raw, 86)[0]
|
|
351
|
+
format_code = struct.unpack_from("<h", raw, 92)[0]
|
|
352
|
+
array_width = struct.unpack_from("<h", raw, 118)[0]
|
|
353
|
+
return ChannelRecord(
|
|
354
|
+
index=index, offset=rec_start, name=name,
|
|
355
|
+
dtype_code=dtype_code, format_code=format_code, raw=raw,
|
|
356
|
+
array_width=array_width, array_basetype_code=array_basetype_code,
|
|
357
|
+
name_is_clean=is_clean,
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def find_channel_table(data: bytes, search_window=(0, None)) -> int:
|
|
362
|
+
"""
|
|
363
|
+
Locate the start of the channel symbol table.
|
|
364
|
+
|
|
365
|
+
Strategy [CONFIRMED against 2 real 2020 USGS files, RE-CONFIRMED --
|
|
366
|
+
with one revision -- against 5 more real 1990s-2020s GSQ files, see
|
|
367
|
+
docs/provenance/notes.md/docs/provenance/log.md "pressure test" round]: search for the default
|
|
368
|
+
super-user name (from GXDB.create()'s documented default
|
|
369
|
+
`super="SUPER"`). The channel table is found to occupy exactly
|
|
370
|
+
`chans_max` consecutive 128-byte records immediately before the user
|
|
371
|
+
table, i.e.
|
|
372
|
+
channel_table_start == offset_of(super_name) - 8 - chans_max*128
|
|
373
|
+
|
|
374
|
+
This isn't a generic file-format constant we can hardcode a single
|
|
375
|
+
offset for -- it depends on `chans_max`, which itself varies between
|
|
376
|
+
files -- so we compute it.
|
|
377
|
+
|
|
378
|
+
REVISION from the original derivation: the two 2020 USGS files both
|
|
379
|
+
had the default super-user name stored as literal uppercase ASCII
|
|
380
|
+
"SUPER". Five real 1990s-2020s GSQ files instead have it stored as
|
|
381
|
+
lowercase "super" -- confirmed to be the *same* structural pattern
|
|
382
|
+
(same 128-byte-per-record math, same position relative to the
|
|
383
|
+
channel table) once the case is corrected, not a different layout.
|
|
384
|
+
Search for both cases. (One of the GSQ files, DB_Mag_833.gdb, also
|
|
385
|
+
demonstrated that the literal string can coincidentally appear
|
|
386
|
+
elsewhere in a file, e.g. inside embedded metadata blobs, and that
|
|
387
|
+
the *word* "super"/"SUPER" appearing is not on its own sufficient --
|
|
388
|
+
a naive first-match there pointed at a bogus offset. Confirmed
|
|
389
|
+
correct instead via an independent generic 128-byte-periodicity scan
|
|
390
|
+
that landed on the identical answer once cross-checked.)
|
|
391
|
+
|
|
392
|
+
Every occurrence of "SUPER"/"super" is tried and the first one whose
|
|
393
|
+
implied table start decodes a *clean* (NUL-terminated, printable)
|
|
394
|
+
channel name is used.
|
|
395
|
+
"""
|
|
396
|
+
lo, hi = search_window
|
|
397
|
+
if hi is None:
|
|
398
|
+
hi = len(data)
|
|
399
|
+
chans_max = header_fields(data)["chans_max"]
|
|
400
|
+
|
|
401
|
+
start = lo
|
|
402
|
+
while True:
|
|
403
|
+
idx_upper = data.find(b"SUPER", start, hi)
|
|
404
|
+
idx_lower = data.find(b"super", start, hi)
|
|
405
|
+
candidates = [i for i in (idx_upper, idx_lower) if i != -1]
|
|
406
|
+
super_idx = min(candidates) if candidates else -1
|
|
407
|
+
if super_idx == -1:
|
|
408
|
+
raise ValueError(
|
|
409
|
+
"could not find a 'SUPER' user record that implies a valid "
|
|
410
|
+
"channel table in the search window; try widening `search_window`"
|
|
411
|
+
)
|
|
412
|
+
super_rec_start = super_idx - 8
|
|
413
|
+
table_start = super_rec_start - chans_max * SYMBOL_RECORD_SIZE
|
|
414
|
+
if table_start >= 0:
|
|
415
|
+
raw = data[table_start : table_start + SYMBOL_RECORD_SIZE]
|
|
416
|
+
if len(raw) == SYMBOL_RECORD_SIZE:
|
|
417
|
+
name, is_clean = _read_name(raw, 8)
|
|
418
|
+
if is_clean and name:
|
|
419
|
+
return table_start
|
|
420
|
+
start = super_idx + 1
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def read_channels(path: str) -> List[ChannelRecord]:
|
|
424
|
+
"""
|
|
425
|
+
Decode the channel symbol table. Fails gracefully: a file that
|
|
426
|
+
isn't a real `.gdb` (bad magic), has a truncated header, has no
|
|
427
|
+
locatable channel table, or has a channel table that's cut off
|
|
428
|
+
partway through all result in a `GDBParseWarning` plus whatever
|
|
429
|
+
channels *were* successfully decoded before the problem (an empty
|
|
430
|
+
list in the first three cases, since nothing was decodable yet; a
|
|
431
|
+
real, non-empty, shorter-than-`chans_max` list in the last case).
|
|
432
|
+
Never raises for these -- see `GDBParseWarning`'s docstring.
|
|
433
|
+
"""
|
|
434
|
+
with open(path, "rb") as f:
|
|
435
|
+
# Reading the whole file is wasteful for a 700MB+ real survey
|
|
436
|
+
# database, but the symbol table's exact byte extent isn't fully
|
|
437
|
+
# pinned down yet (docs/provenance/notes.md 6.1), so for correctness this reads
|
|
438
|
+
# generously. A production version should read a memory-mapped
|
|
439
|
+
# view instead -- left as a TODO once the header's table-size
|
|
440
|
+
# field (offset 104, currently [UNKNOWN]) is confirmed.
|
|
441
|
+
header = f.read(4096)
|
|
442
|
+
if not check_magic(header):
|
|
443
|
+
_warn(f"{path}: does not start with the expected '!CBD' magic -- "
|
|
444
|
+
f"not a recognized .gdb file, returning no channels")
|
|
445
|
+
return []
|
|
446
|
+
fields = header_fields(header)
|
|
447
|
+
if fields["chans_max"] is None:
|
|
448
|
+
_warn(f"{path}: header too short to read chans_max -- returning no channels")
|
|
449
|
+
return []
|
|
450
|
+
|
|
451
|
+
f.seek(0, 2)
|
|
452
|
+
size = f.tell()
|
|
453
|
+
f.seek(0)
|
|
454
|
+
# Only need enough of the file to reach the channel + user tables.
|
|
455
|
+
# Observed table offsets range from ~130KB to ~580KB across 9 real
|
|
456
|
+
# files so far, but read generously (all of a file up to 200MB,
|
|
457
|
+
# else the first 20MB) since the exact extent isn't pinned down.
|
|
458
|
+
data = f.read(size if size <= 200_000_000 else 20_000_000)
|
|
459
|
+
|
|
460
|
+
try:
|
|
461
|
+
table_start = find_channel_table(data)
|
|
462
|
+
except ValueError as e:
|
|
463
|
+
_warn(f"{path}: could not locate the channel symbol table ({e}) -- "
|
|
464
|
+
f"returning no channels")
|
|
465
|
+
return []
|
|
466
|
+
chans_max = fields["chans_max"]
|
|
467
|
+
|
|
468
|
+
channels = []
|
|
469
|
+
for i in range(chans_max):
|
|
470
|
+
rec_start = table_start + i * SYMBOL_RECORD_SIZE
|
|
471
|
+
if rec_start + SYMBOL_RECORD_SIZE > len(data):
|
|
472
|
+
_warn(
|
|
473
|
+
f"{path}: channel table truncated at record {i} of {chans_max} "
|
|
474
|
+
f"(need bytes up to {rec_start + SYMBOL_RECORD_SIZE}, only "
|
|
475
|
+
f"{len(data)} were read/available) -- returning the "
|
|
476
|
+
f"{len(channels)} channel(s) decoded so far"
|
|
477
|
+
)
|
|
478
|
+
break
|
|
479
|
+
rec = _parse_channel_record(data, rec_start, i)
|
|
480
|
+
if not rec.name:
|
|
481
|
+
continue # cleanly empty/unused slot (NUL name, zeroed record)
|
|
482
|
+
if not rec.name_is_clean:
|
|
483
|
+
# Unused capacity slot with leftover/uninitialized bytes rather
|
|
484
|
+
# than a clean NUL name -- observed in at least one real 1990s
|
|
485
|
+
# file (see _read_name docstring / docs/provenance/notes.md). Not a real channel.
|
|
486
|
+
continue
|
|
487
|
+
if not rec.looks_sane:
|
|
488
|
+
# A NUL-terminated printable "name" can still show up by pure
|
|
489
|
+
# coincidence inside leftover garbage bytes in an unused slot
|
|
490
|
+
# (observed in DB_Mag_833.gdb: "L2161" and "1", both leftover
|
|
491
|
+
# fragments of an embedded projection-name blob that happened
|
|
492
|
+
# to land in unused channel-table capacity). Real channel
|
|
493
|
+
# records always have a dtype matching a known GS_* code or a
|
|
494
|
+
# small negative string width, and a format code in the known
|
|
495
|
+
# DB_CHAN_FORMAT_* range -- garbage doesn't. See docs/provenance/notes.md.
|
|
496
|
+
continue
|
|
497
|
+
channels.append(rec)
|
|
498
|
+
return channels
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
DB_CATEGORY_LINE_NAMES = {
|
|
502
|
+
100: "NORMAL", # DB_CATEGORY_LINE_NORMAL / DB_CATEGORY_LINE_FLIGHT (same value)
|
|
503
|
+
200: "GROUP", # DB_CATEGORY_LINE_GROUP
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
_LINE_TABLE_EMPTY_CATEGORY = 65536 # [CONFIRMED] sentinel seen on unused line-table
|
|
507
|
+
# capacity slots -- docs/spec.md section 3.2
|
|
508
|
+
|
|
509
|
+
_NAME_LIKE_RE = re.compile(rb"[\x20-\x7e]{1,63}\x00")
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
@dataclass(eq=False)
|
|
513
|
+
class LineRecord:
|
|
514
|
+
"""
|
|
515
|
+
One 128-byte line-table record. [LIKELY]/[UNKNOWN] -- much less firmly
|
|
516
|
+
established than ChannelRecord: only the name (relative +32) and
|
|
517
|
+
category code (relative +108) fields are decoded, and locating the
|
|
518
|
+
table itself (find_line_table below) is a heuristic scan rather than
|
|
519
|
+
the structurally-proven SUPER-anchor technique used for the channel
|
|
520
|
+
table. See docs/spec.md section 3.2 and docs/provenance/notes.md section 6.3.
|
|
521
|
+
|
|
522
|
+
`eq=False` keeps the default identity-based `__eq__`/`__hash__`
|
|
523
|
+
instead of dataclass's usual field-by-field one -- needed both to
|
|
524
|
+
stay hashable (see `ChannelRecord`'s docstring for why) and because
|
|
525
|
+
`GDB._calibrate_line_indices` mutates `.index` in place on these
|
|
526
|
+
after construction; a value-based `__eq__`/`__hash__` pair would be
|
|
527
|
+
actively wrong for an object whose fields change post-construction.
|
|
528
|
+
"""
|
|
529
|
+
index: int # 0-based physical slot number -- this IS line_slot_index
|
|
530
|
+
# in the blob_index formula (BlobHeader.line_channel)
|
|
531
|
+
offset: int
|
|
532
|
+
name: str
|
|
533
|
+
category_code: Optional[int]
|
|
534
|
+
raw: bytes
|
|
535
|
+
name_is_clean: bool = True
|
|
536
|
+
|
|
537
|
+
@property
|
|
538
|
+
def category_name(self) -> str:
|
|
539
|
+
if self.category_code is None:
|
|
540
|
+
return "unknown"
|
|
541
|
+
return DB_CATEGORY_LINE_NAMES.get(self.category_code, f"unknown({self.category_code})")
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _parse_line_record(data: bytes, rec_start: int, index: int) -> LineRecord:
|
|
545
|
+
raw = data[rec_start : rec_start + SYMBOL_RECORD_SIZE]
|
|
546
|
+
name, is_clean = _read_name(raw, 32)
|
|
547
|
+
try:
|
|
548
|
+
category_code = struct.unpack_from("<i", raw, 108)[0]
|
|
549
|
+
except struct.error:
|
|
550
|
+
category_code = None
|
|
551
|
+
return LineRecord(
|
|
552
|
+
index=index, offset=rec_start, name=name,
|
|
553
|
+
category_code=category_code, raw=raw, name_is_clean=is_clean,
|
|
554
|
+
)
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
def find_line_table(data: bytes, search_window: Tuple[int, Optional[int]] = (128, None)) -> int:
|
|
558
|
+
"""
|
|
559
|
+
Locate the start of the line symbol table.
|
|
560
|
+
|
|
561
|
+
Unlike find_channel_table, there's no known default-name anchor (the
|
|
562
|
+
line table has nothing analogous to the channel table's "SUPER" user
|
|
563
|
+
record immediately after it) and no confirmed header field gives its
|
|
564
|
+
start offset directly -- reconciling one with the header's capacity
|
|
565
|
+
fields was tried and didn't cleanly round-trip (docs/provenance/log.md
|
|
566
|
+
Session 1 section 1.16, docs/provenance/notes.md section 6.3). This is
|
|
567
|
+
therefore a heuristic **[LIKELY]** scan, not the structurally-proven
|
|
568
|
+
technique used for the channel table: it looks for a run of 128-byte
|
|
569
|
+
records whose relative +32 field looks like a clean, NUL-terminated,
|
|
570
|
+
printable line name and whose relative +108 category field matches one
|
|
571
|
+
of the two confirmed real values (100=NORMAL/FLIGHT, 200=GROUP), then
|
|
572
|
+
returns the earliest such record in the run with the most hits at a
|
|
573
|
+
consistent 128-byte phase.
|
|
574
|
+
|
|
575
|
+
**Known limitation, found by real-file testing, not yet fixed here:**
|
|
576
|
+
if a table's true first slot(s) don't carry a category code in
|
|
577
|
+
{100, 200}, this returns a start that's one or more slots too late --
|
|
578
|
+
every subsequent LineRecord.index is then off by that same fixed
|
|
579
|
+
amount, which breaks blob_index lookups by line name. Observed for
|
|
580
|
+
real on a GSQ file (`rm001141`): physical slot 0 is a genuine, named
|
|
581
|
+
record (`"L0"`) with category `65636` (**[GUESS]**: `65536 + 100`,
|
|
582
|
+
plausibly "a NORMAL line that was since cleared," not confirmed),
|
|
583
|
+
which this function doesn't recognize, so it starts the table one
|
|
584
|
+
slot late. A generic backward-scan fix was tried and rejected: "keep
|
|
585
|
+
walking backward while the name field still looks clean" massively
|
|
586
|
+
over-extends on at least one real file (walked 30+ slots into what
|
|
587
|
+
turned out to be unrelated, legitimately-empty space before the real
|
|
588
|
+
table). `GDB` (in `gdb.py`) instead cross-validates and corrects this
|
|
589
|
+
against the actual blob chain, which is a strictly stronger signal
|
|
590
|
+
than anything available from the symbol-table bytes alone -- prefer
|
|
591
|
+
it over calling this function directly when correct line-indexed
|
|
592
|
+
data access matters, not just names.
|
|
593
|
+
|
|
594
|
+
`search_window` defaults to (128, end of `data`) -- callers should
|
|
595
|
+
generally narrow `hi` to `blob_region_start(data)` when known, since
|
|
596
|
+
every real file examined has its symbol tables (line, channel, user)
|
|
597
|
+
entirely before the blob region, and narrowing avoids false-positive
|
|
598
|
+
matches inside actual channel data.
|
|
599
|
+
"""
|
|
600
|
+
lo, hi = search_window
|
|
601
|
+
if hi is None:
|
|
602
|
+
hi = len(data)
|
|
603
|
+
|
|
604
|
+
phase_hits = {}
|
|
605
|
+
for m in _NAME_LIKE_RE.finditer(data, lo, hi):
|
|
606
|
+
rec_start = m.start() - 32
|
|
607
|
+
if rec_start < lo or rec_start + SYMBOL_RECORD_SIZE > hi:
|
|
608
|
+
continue
|
|
609
|
+
raw = data[rec_start : rec_start + SYMBOL_RECORD_SIZE]
|
|
610
|
+
try:
|
|
611
|
+
category_code = struct.unpack_from("<i", raw, 108)[0]
|
|
612
|
+
except struct.error:
|
|
613
|
+
continue
|
|
614
|
+
if category_code not in DB_CATEGORY_LINE_NAMES:
|
|
615
|
+
continue
|
|
616
|
+
phase_hits.setdefault(rec_start % SYMBOL_RECORD_SIZE, []).append(rec_start)
|
|
617
|
+
|
|
618
|
+
if not phase_hits:
|
|
619
|
+
raise ValueError(
|
|
620
|
+
"no line-record-shaped data (clean name + a known category "
|
|
621
|
+
"code) found in the search window"
|
|
622
|
+
)
|
|
623
|
+
best_phase = max(phase_hits, key=lambda p: len(phase_hits[p]))
|
|
624
|
+
return min(phase_hits[best_phase])
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def read_lines(path: str) -> List[LineRecord]:
|
|
628
|
+
"""
|
|
629
|
+
Decode the line symbol table. Heuristic (see find_line_table) --
|
|
630
|
+
less firmly established than read_channels. Fails gracefully in the
|
|
631
|
+
same style: a bad magic, truncated header, or unlocatable line table
|
|
632
|
+
all return `[]` with a `GDBParseWarning` rather than raising.
|
|
633
|
+
|
|
634
|
+
Since no confirmed header field gives the line table's slot capacity
|
|
635
|
+
(the way chans_max does for the channel table), this reads forward
|
|
636
|
+
from the located start until 8 consecutive records fail to look like
|
|
637
|
+
either a populated line record or clean unused capacity -- a
|
|
638
|
+
tolerance against one-off corruption/false-positive records, not a
|
|
639
|
+
precisely-known table boundary.
|
|
640
|
+
|
|
641
|
+
**`LineRecord.index` can be off by a small, fixed amount** on a file
|
|
642
|
+
where `find_line_table`'s heuristic starts one or more slots late --
|
|
643
|
+
see that function's docstring. This makes `.name` still correct but
|
|
644
|
+
`.index` (and therefore any blob_index lookup keyed on it) wrong.
|
|
645
|
+
`GDB` (in `gdb.py`) corrects this against the actual blob chain
|
|
646
|
+
before exposing lines by name; call it instead of this function
|
|
647
|
+
directly when you need working (line, channel) data access, not
|
|
648
|
+
just a list of names.
|
|
649
|
+
"""
|
|
650
|
+
with open(path, "rb") as f:
|
|
651
|
+
header = f.read(4096)
|
|
652
|
+
if not check_magic(header):
|
|
653
|
+
_warn(f"{path}: does not start with the expected '!CBD' magic -- "
|
|
654
|
+
f"not a recognized .gdb file, returning no lines")
|
|
655
|
+
return []
|
|
656
|
+
blob_start = blob_region_start(header)
|
|
657
|
+
f.seek(0, 2)
|
|
658
|
+
size = f.tell()
|
|
659
|
+
f.seek(0)
|
|
660
|
+
read_size = blob_start if (blob_start is not None and 0 < blob_start <= size) else min(size, 20_000_000)
|
|
661
|
+
data = f.read(read_size)
|
|
662
|
+
|
|
663
|
+
try:
|
|
664
|
+
table_start = find_line_table(data, search_window=(128, len(data)))
|
|
665
|
+
except ValueError as e:
|
|
666
|
+
_warn(f"{path}: could not locate the line symbol table ({e}) -- "
|
|
667
|
+
f"returning no lines")
|
|
668
|
+
return []
|
|
669
|
+
|
|
670
|
+
lines: List[LineRecord] = []
|
|
671
|
+
consecutive_bad = 0
|
|
672
|
+
i = 0
|
|
673
|
+
while True:
|
|
674
|
+
rec_start = table_start + i * SYMBOL_RECORD_SIZE
|
|
675
|
+
if rec_start + SYMBOL_RECORD_SIZE > len(data):
|
|
676
|
+
break
|
|
677
|
+
rec = _parse_line_record(data, rec_start, i)
|
|
678
|
+
i += 1
|
|
679
|
+
if rec.name and rec.name_is_clean and rec.category_code in DB_CATEGORY_LINE_NAMES:
|
|
680
|
+
lines.append(rec)
|
|
681
|
+
consecutive_bad = 0
|
|
682
|
+
elif not rec.name and rec.name_is_clean:
|
|
683
|
+
# Empty/unused capacity slot (matches the channel table's own
|
|
684
|
+
# zeroed-padding convention, or the 65536 empty-category
|
|
685
|
+
# sentinel confirmed in docs/spec.md section 3.2) -- keep
|
|
686
|
+
# scanning past it, it doesn't count as "bad".
|
|
687
|
+
consecutive_bad = 0
|
|
688
|
+
else:
|
|
689
|
+
consecutive_bad += 1
|
|
690
|
+
if consecutive_bad >= 8:
|
|
691
|
+
break
|
|
692
|
+
return lines
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
BLOB_MAGIC = b"\xcc\xcc\x00\xff"
|
|
696
|
+
BLOB_HEADER_SIZE = 48 # [CONFIRMED] -- see docs/provenance/notes.md section 6.6
|
|
697
|
+
|
|
698
|
+
# [CONFIRMED] (docs/provenance/notes.md section 6.6b): for COMPRESSED blobs specifically
|
|
699
|
+
# (DB_COMP_SPEED or DB_COMP_SIZE), the blob header is 56 bytes, not 48 --
|
|
700
|
+
# the extra 8 bytes hold a preview of the first chunk's decompressed
|
|
701
|
+
# length and total on-disk span (not fully decoded, see docs/provenance/notes.md section
|
|
702
|
+
# 6.5e). The already-known 16-byte page-primitive chunk magic
|
|
703
|
+
# (lzrw1.CHUNK_MAGIC) sits immediately after these 56 bytes, verified
|
|
704
|
+
# directly against real ground truth on AG106386 (DB_COMP_SIZE): the
|
|
705
|
+
# zlib payload for blob_index=0 (GA_project_number) is found at exactly
|
|
706
|
+
# blob.offset + COMPRESSED_BLOB_HEADER_SIZE + 16 and decompresses to the
|
|
707
|
+
# known real constant value 5027.
|
|
708
|
+
COMPRESSED_BLOB_HEADER_SIZE = 56
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
@dataclass
|
|
712
|
+
class BlobHeader:
|
|
713
|
+
"""
|
|
714
|
+
The per-channel-per-line data block header. [CONFIRMED] for fields
|
|
715
|
+
up to and including `blob_index` (verified byte-exact on 5 real
|
|
716
|
+
DB_COMP_NONE files via a whole-file, zero-error chain walk that
|
|
717
|
+
lands exactly on each file's true size -- see docs/provenance/notes.md section 6.6).
|
|
718
|
+
Fields from `timestamp` onward are only [LIKELY]/[UNKNOWN] and are
|
|
719
|
+
known NOT to decode sensibly at these byte offsets in at least one
|
|
720
|
+
real older (1991 GSQ) file -- kept here for the modern (2020 USGS)
|
|
721
|
+
case where they were verified, not assumed general.
|
|
722
|
+
"""
|
|
723
|
+
offset: int # absolute file offset of this header's first byte
|
|
724
|
+
n_pages: int # [CONFIRMED] -- this blob's total on-disk size, in
|
|
725
|
+
# pages (page_size from header_fields())
|
|
726
|
+
n_pages_dup: int # [LIKELY] -- always seen equal to n_pages
|
|
727
|
+
blob_index: int # [CONFIRMED] -- see line_slot/channel_slot below
|
|
728
|
+
timestamp: int # [LIKELY] modern files only, see docstring above
|
|
729
|
+
reserved_200: int # [UNKNOWN]
|
|
730
|
+
scale: float # [LIKELY] modern files only
|
|
731
|
+
row_count: int # [CONFIRMED] modern files only (verified against
|
|
732
|
+
# real ground-truth-matching decoded values)
|
|
733
|
+
gs_type_code: int # [CONFIRMED] modern files only (matches owning
|
|
734
|
+
# channel's own symbol-table dtype exactly)
|
|
735
|
+
|
|
736
|
+
def line_channel(self, chans_max: int):
|
|
737
|
+
"""
|
|
738
|
+
Decompose blob_index into (line_slot_index, channel_slot_index)
|
|
739
|
+
via the formula [CONFIRMED] in docs/provenance/notes.md section 6.6:
|
|
740
|
+
blob_index == line_slot_index * chans_max + channel_slot_index
|
|
741
|
+
Both are 0-based physical slot numbers in their respective
|
|
742
|
+
symbol tables (same indexing as ChannelRecord.index and the
|
|
743
|
+
line table walked ad hoc in docs/provenance/notes.md section 6.3).
|
|
744
|
+
"""
|
|
745
|
+
return divmod(self.blob_index, chans_max)
|
|
746
|
+
|
|
747
|
+
@property
|
|
748
|
+
def data_offset(self) -> int:
|
|
749
|
+
return self.offset + BLOB_HEADER_SIZE
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
def _parse_blob_header(raw: bytes, offset: int) -> Optional[BlobHeader]:
|
|
753
|
+
if len(raw) < BLOB_HEADER_SIZE or raw[:4] != BLOB_MAGIC:
|
|
754
|
+
return None
|
|
755
|
+
n_pages = struct.unpack_from("<i", raw, 4)[0]
|
|
756
|
+
n_pages_dup = struct.unpack_from("<i", raw, 8)[0]
|
|
757
|
+
blob_index = struct.unpack_from("<i", raw, 12)[0]
|
|
758
|
+
timestamp = struct.unpack_from("<i", raw, 16)[0]
|
|
759
|
+
reserved_200 = struct.unpack_from("<i", raw, 20)[0]
|
|
760
|
+
scale = struct.unpack_from("<d", raw, 32)[0]
|
|
761
|
+
row_count = struct.unpack_from("<i", raw, 40)[0]
|
|
762
|
+
gs_type_code = struct.unpack_from("<i", raw, 44)[0]
|
|
763
|
+
return BlobHeader(
|
|
764
|
+
offset=offset, n_pages=n_pages, n_pages_dup=n_pages_dup,
|
|
765
|
+
blob_index=blob_index, timestamp=timestamp,
|
|
766
|
+
reserved_200=reserved_200, scale=scale, row_count=row_count,
|
|
767
|
+
gs_type_code=gs_type_code,
|
|
768
|
+
)
|
|
769
|
+
|
|
770
|
+
|
|
771
|
+
def blob_region_start(data: bytes) -> Optional[int]:
|
|
772
|
+
"""
|
|
773
|
+
Absolute byte offset of the first real blob header.
|
|
774
|
+
|
|
775
|
+
[CONFIRMED] on 20+ real files (every compression mode, chans_max
|
|
776
|
+
20-500, ~1991-2020, all 3 agencies) -- see docs/provenance/notes.md section 6.6/
|
|
777
|
+
6.6b. Header offset 108 (int32) is a PAGE NUMBER; multiplying by
|
|
778
|
+
page_size (header offset 100) lands exactly on the CC CC 00 FF
|
|
779
|
+
magic every time. (Header offset 104, an earlier "live lead" for
|
|
780
|
+
this same purpose in this project's own notes, is a close-but-wrong
|
|
781
|
+
red herring -- it sits near, but not exactly on, the end of the
|
|
782
|
+
symbol tables, and isn't even page-aligned.)
|
|
783
|
+
|
|
784
|
+
Returns `None` (with a `GDBParseWarning`) if `data` is too short to
|
|
785
|
+
even read the two fields this needs (offset 108 + 4 bytes) -- a
|
|
786
|
+
severely truncated header.
|
|
787
|
+
"""
|
|
788
|
+
try:
|
|
789
|
+
page_size = struct.unpack_from("<i", data, 100)[0]
|
|
790
|
+
start_page = struct.unpack_from("<i", data, 108)[0]
|
|
791
|
+
except struct.error:
|
|
792
|
+
_warn(
|
|
793
|
+
f"header truncated: only {len(data)} byte(s) available, not enough "
|
|
794
|
+
f"to locate the blob region (need offset 108 + 4 bytes)"
|
|
795
|
+
)
|
|
796
|
+
return None
|
|
797
|
+
return start_page * page_size
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def iter_blobs(path: str, max_blobs: Optional[int] = None):
|
|
801
|
+
"""
|
|
802
|
+
Walk the self-describing blob chain from the start of the blob
|
|
803
|
+
region to end of file (or `max_blobs`, or the first framing
|
|
804
|
+
anomaly), yielding BlobHeader records in on-disk order.
|
|
805
|
+
|
|
806
|
+
[CONFIRMED] end-to-end (zero framing errors, landing exactly on the
|
|
807
|
+
true file size) on 20 real files spanning all 3 agencies this
|
|
808
|
+
project has files from and all three `DB_COMP_*` compression modes,
|
|
809
|
+
2MB to 1.93GB -- see docs/provenance/notes.md section 6.6b/6.6d/6.9.
|
|
810
|
+
|
|
811
|
+
As a generator, this already "returns partial results" in the most
|
|
812
|
+
natural way possible: whatever's been yielded before a problem is
|
|
813
|
+
hit stays with the caller (a `for blob in iter_blobs(path): ...`
|
|
814
|
+
loop simply ends, keeping everything already processed) -- nothing
|
|
815
|
+
is lost by stopping early. What this function adds on top of that
|
|
816
|
+
is a clear `GDBParseWarning` distinguishing *why* it stopped:
|
|
817
|
+
reaching the file's true end cleanly is silent (the expected,
|
|
818
|
+
common case), but a magic mismatch, a non-positive `n_pages`, a
|
|
819
|
+
file that ends mid-header, or landing short of true EOF by less
|
|
820
|
+
than one full header (i.e. real leftover bytes, not enough to be
|
|
821
|
+
read at all) are all real anomalies and each gets its own specific
|
|
822
|
+
warning identifying the offset and how many blobs were walked
|
|
823
|
+
first -- so a caller can tell "the chain looked completely normal
|
|
824
|
+
and just ended" from "something didn't fit the confirmed
|
|
825
|
+
structure" without having to guess from the return value alone.
|
|
826
|
+
Never raises for a bad/truncated file; only for a real precondition
|
|
827
|
+
problem (can't even open `path`, propagated normally from `open`).
|
|
828
|
+
"""
|
|
829
|
+
with open(path, "rb") as f:
|
|
830
|
+
header = f.read(128)
|
|
831
|
+
if not check_magic(header):
|
|
832
|
+
_warn(f"{path}: does not start with the expected '!CBD' magic -- "
|
|
833
|
+
f"no blobs to walk")
|
|
834
|
+
return
|
|
835
|
+
off = blob_region_start(header)
|
|
836
|
+
if off is None:
|
|
837
|
+
_warn(f"{path}: could not determine the blob region start -- "
|
|
838
|
+
f"no blobs to walk")
|
|
839
|
+
return
|
|
840
|
+
page_size = struct.unpack_from("<i", header, 100)[0]
|
|
841
|
+
f.seek(0, 2)
|
|
842
|
+
size = f.tell()
|
|
843
|
+
if off > size:
|
|
844
|
+
_warn(
|
|
845
|
+
f"{path}: computed blob region start ({off}) is past the end "
|
|
846
|
+
f"of the file ({size} byte(s)) -- file is likely severely "
|
|
847
|
+
f"truncated; no blobs to walk"
|
|
848
|
+
)
|
|
849
|
+
return
|
|
850
|
+
f.seek(off)
|
|
851
|
+
n = 0
|
|
852
|
+
while off + BLOB_HEADER_SIZE <= size:
|
|
853
|
+
if max_blobs is not None and n >= max_blobs:
|
|
854
|
+
return
|
|
855
|
+
raw = f.read(BLOB_HEADER_SIZE)
|
|
856
|
+
blob = _parse_blob_header(raw, off)
|
|
857
|
+
if blob is None:
|
|
858
|
+
if len(raw) < BLOB_HEADER_SIZE:
|
|
859
|
+
_warn(
|
|
860
|
+
f"{path}: blob chain ends mid-header at offset {off} "
|
|
861
|
+
f"(only {len(raw)} of {BLOB_HEADER_SIZE} expected "
|
|
862
|
+
f"byte(s) available) after {n} blob(s) successfully "
|
|
863
|
+
f"walked -- file is likely truncated; returning the "
|
|
864
|
+
f"{n} blob(s) already yielded"
|
|
865
|
+
)
|
|
866
|
+
else:
|
|
867
|
+
_warn(
|
|
868
|
+
f"{path}: blob magic mismatch at offset {off} "
|
|
869
|
+
f"(got {raw[:4].hex()}, expected {BLOB_MAGIC.hex()}) "
|
|
870
|
+
f"after {n} blob(s) successfully walked -- stopping "
|
|
871
|
+
f"the chain walk here and returning the {n} blob(s) "
|
|
872
|
+
f"already yielded; this may be a real structural "
|
|
873
|
+
f"anomaly or an administrative-blob variant not yet "
|
|
874
|
+
f"understood (docs/provenance/notes.md section 6.4/6.9)"
|
|
875
|
+
)
|
|
876
|
+
return
|
|
877
|
+
if blob.n_pages <= 0:
|
|
878
|
+
_warn(
|
|
879
|
+
f"{path}: blob at offset {off} (blob_index={blob.blob_index}) "
|
|
880
|
+
f"has a non-positive n_pages ({blob.n_pages}) after {n} "
|
|
881
|
+
f"blob(s) successfully walked -- cannot safely continue "
|
|
882
|
+
f"(don't know how far to skip to find the next header); "
|
|
883
|
+
f"returning the {n} blob(s) already yielded"
|
|
884
|
+
)
|
|
885
|
+
return
|
|
886
|
+
# NOTE: n_pages_dup (relative +8) is NOT always equal to n_pages
|
|
887
|
+
# (relative +4) -- confirmed on real Ontario GDS1251 files
|
|
888
|
+
# (MLGRAV.gdb/MLMAG.gdb), where a small number of "reserved/
|
|
889
|
+
# administrative" blobs (same class flagged [UNKNOWN] elsewhere
|
|
890
|
+
# in this section -- out-of-range line index, gs_type_code
|
|
891
|
+
# reading the same 4670802 constant) have n_pages_dup != n_pages.
|
|
892
|
+
# Directly verified: n_pages (not n_pages_dup) is the field that
|
|
893
|
+
# correctly lands on the next real blob header every time -- an
|
|
894
|
+
# earlier version of this function required the two to match and
|
|
895
|
+
# broke immediately on these files as a result. Trust n_pages
|
|
896
|
+
# alone; n_pages_dup is kept on BlobHeader for whoever wants to
|
|
897
|
+
# investigate what it actually means.
|
|
898
|
+
yield blob
|
|
899
|
+
skip = blob.n_pages * page_size - BLOB_HEADER_SIZE
|
|
900
|
+
f.seek(skip, 1)
|
|
901
|
+
off += blob.n_pages * page_size
|
|
902
|
+
n += 1
|
|
903
|
+
if n > 0 and off > size:
|
|
904
|
+
# The last blob successfully parsed claimed a page count that
|
|
905
|
+
# implies more data than the file actually contains -- off
|
|
906
|
+
# jumped past true EOF. A real, distinct anomaly: the file is
|
|
907
|
+
# cut off in the middle of what should have been that blob's
|
|
908
|
+
# data (or its padding).
|
|
909
|
+
_warn(
|
|
910
|
+
f"{path}: after {n} blob(s), the last one (offset "
|
|
911
|
+
f"{off - blob.n_pages * page_size}, blob_index={blob.blob_index}, "
|
|
912
|
+
f"n_pages={blob.n_pages}) claims data extending "
|
|
913
|
+
f"{off - size} byte(s) past the true end of file ({size} "
|
|
914
|
+
f"byte(s) total) -- file is truncated mid-blob; returning "
|
|
915
|
+
f"the {n} blob header(s) already yielded (note: that last "
|
|
916
|
+
f"blob's own data may itself be incomplete -- see "
|
|
917
|
+
f"read_blob_values()'s truncation handling)"
|
|
918
|
+
)
|
|
919
|
+
elif n > 0 and off != size:
|
|
920
|
+
# Loop condition failed (off + 48 > size) but we're not exactly
|
|
921
|
+
# at the true end either -- real leftover bytes, less than one
|
|
922
|
+
# full header's worth. Every real file checked in this project
|
|
923
|
+
# (docs/provenance/notes.md section 6.6b/6.9) ends with an EXACT match, so any
|
|
924
|
+
# slack here is new/unusual and worth flagging, not silently
|
|
925
|
+
# accepted.
|
|
926
|
+
_warn(
|
|
927
|
+
f"{path}: blob chain walk stopped {size - off} byte(s) short "
|
|
928
|
+
f"of the true end of file (at offset {off} of {size}) after "
|
|
929
|
+
f"{n} blob(s) -- less than one full header remains there, "
|
|
930
|
+
f"which doesn't match any real file checked in this project "
|
|
931
|
+
f"so far (they all end with an exact match); file may be "
|
|
932
|
+
f"truncated"
|
|
933
|
+
)
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
def find_blob(path: str, line_slot: int, channel_slot: int, chans_max: Optional[int] = None) -> Optional[BlobHeader]:
|
|
937
|
+
"""
|
|
938
|
+
Locate the blob for a specific (line, channel) pair by walking the
|
|
939
|
+
chain (see iter_blobs) and computing the target blob_index via the
|
|
940
|
+
formula [CONFIRMED] in docs/provenance/notes.md section 6.6. Returns `None` if the
|
|
941
|
+
chain ends (or breaks -- see `iter_blobs`'s `GDBParseWarning`s for
|
|
942
|
+
why) before the target is found, or if `chans_max` can't be
|
|
943
|
+
determined at all (bad magic / truncated header) -- never raises
|
|
944
|
+
for these, consistent with the rest of this module.
|
|
945
|
+
|
|
946
|
+
This does a linear walk from the start of the blob region every
|
|
947
|
+
call -- fine for occasional lookups or for building a full
|
|
948
|
+
line/channel -> offset index once (walk the whole chain yourself
|
|
949
|
+
with iter_blobs() and record every blob.offset keyed by
|
|
950
|
+
blob.line_channel(chans_max) if you need many lookups).
|
|
951
|
+
"""
|
|
952
|
+
if chans_max is None:
|
|
953
|
+
with open(path, "rb") as f:
|
|
954
|
+
header = f.read(128)
|
|
955
|
+
if not check_magic(header):
|
|
956
|
+
_warn(f"{path}: does not start with the expected '!CBD' magic -- "
|
|
957
|
+
f"cannot determine chans_max, blob not found")
|
|
958
|
+
return None
|
|
959
|
+
try:
|
|
960
|
+
chans_max = struct.unpack_from("<i", header, 24)[0]
|
|
961
|
+
except struct.error:
|
|
962
|
+
_warn(f"{path}: header too short to read chans_max -- blob not found")
|
|
963
|
+
return None
|
|
964
|
+
target = line_slot * chans_max + channel_slot
|
|
965
|
+
for blob in iter_blobs(path):
|
|
966
|
+
if blob.blob_index == target:
|
|
967
|
+
return blob
|
|
968
|
+
return None
|
|
969
|
+
|
|
970
|
+
|
|
971
|
+
def _element_width(channel: ChannelRecord) -> Optional[int]:
|
|
972
|
+
"""
|
|
973
|
+
Byte width of one element of `channel`'s data, or `None` if it's a
|
|
974
|
+
type this reader doesn't know how to decode -- e.g. one of the
|
|
975
|
+
multi-dimensional `GS_FLOAT3D`/`GS_DOUBLE3D`/`GS_FLOAT2D`/
|
|
976
|
+
`GS_DOUBLE2D` types, none of which have been seen in any real
|
|
977
|
+
sample yet (docs/provenance/notes.md section 4). Callers should check for `None`
|
|
978
|
+
and warn/return gracefully rather than assume a format exists.
|
|
979
|
+
"""
|
|
980
|
+
if channel.is_string:
|
|
981
|
+
return channel.string_width
|
|
982
|
+
fmt = GS_TYPE_STRUCT.get(channel.dtype_code)
|
|
983
|
+
return struct.calcsize(fmt) if fmt is not None else None
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
def _decode_numeric_or_string(raw: bytes, channel: ChannelRecord, row_count: Optional[int] = None):
|
|
987
|
+
"""
|
|
988
|
+
Interpret a raw byte buffer as `row_count` (or however many fit)
|
|
989
|
+
values of `channel`'s known type. Shared by the uncompressed and
|
|
990
|
+
compressed decode paths.
|
|
991
|
+
|
|
992
|
+
Always returns a numpy `ndarray`: 1-D `(n_rows,)` for an ordinary
|
|
993
|
+
scalar channel, or 2-D `(n_rows, channel.array_width)` for a VA/
|
|
994
|
+
array channel (docs/spec.md section 5 -- e.g. a 512-wide airborne
|
|
995
|
+
gamma-ray spectrum recorded per station; `array_width` is fixed per
|
|
996
|
+
channel, never seen to vary row-to-row, so this reshape is always a
|
|
997
|
+
clean rectangle). Numeric channels get the dtype matching their
|
|
998
|
+
`GS_*` type (`GS_TYPE_NUMPY_DTYPE`); string channels (including the
|
|
999
|
+
unconfirmed-but-handled case of a *string* array channel) get
|
|
1000
|
+
`dtype=object` holding plain Python `str`, since numpy has no
|
|
1001
|
+
variable-content fixed-dtype string type that round-trips this
|
|
1002
|
+
format's null-padded, variable-actual-length names cleanly.
|
|
1003
|
+
|
|
1004
|
+
String-typed channels dispatch to the compiled `pygdb._native`
|
|
1005
|
+
extension when it's available (same decode, ported to Rust -- see
|
|
1006
|
+
`rust/src/lib.rs`'s `decode_fixed_width_strings`; profiling found
|
|
1007
|
+
this the second real CPU-bound hot path in this reader besides
|
|
1008
|
+
LZRW1, unlike numeric decode which stays near memory-bandwidth speed
|
|
1009
|
+
via `np.frombuffer` either way), falling back to the pure-Python list
|
|
1010
|
+
comprehension below when it isn't.
|
|
1011
|
+
|
|
1012
|
+
Fails gracefully rather than raising: an unrecognized element type
|
|
1013
|
+
returns an empty array with a `GDBParseWarning`; a `raw` buffer
|
|
1014
|
+
shorter than needed for the requested `row_count` (the file was
|
|
1015
|
+
truncated mid-blob, a real scenario for a cut-off download) decodes
|
|
1016
|
+
as many *complete* elements as actually fit and warns about the
|
|
1017
|
+
shortfall, rather than raising a `struct.error` and discarding
|
|
1018
|
+
everything; for an array channel, a flat element count that isn't a
|
|
1019
|
+
whole multiple of `array_width` similarly warns and drops the
|
|
1020
|
+
trailing incomplete row rather than raising.
|
|
1021
|
+
"""
|
|
1022
|
+
width = _element_width(channel)
|
|
1023
|
+
if width is None:
|
|
1024
|
+
_warn(
|
|
1025
|
+
f"channel {channel.name!r} has dtype_code={channel.dtype_code}, a type "
|
|
1026
|
+
f"this reader doesn't know how to decode (likely a multi-dimensional "
|
|
1027
|
+
f"GS_FLOAT3D/GS_DOUBLE3D/etc type, never seen in a real sample) -- "
|
|
1028
|
+
f"returning no values for it"
|
|
1029
|
+
)
|
|
1030
|
+
return np.array([])
|
|
1031
|
+
n_available = len(raw) // width
|
|
1032
|
+
n = row_count if row_count is not None else n_available
|
|
1033
|
+
if n > n_available:
|
|
1034
|
+
_warn(
|
|
1035
|
+
f"channel {channel.name!r}: expected {n} row(s) but only enough raw "
|
|
1036
|
+
f"bytes for {n_available} complete element(s) (got {len(raw)} byte(s), "
|
|
1037
|
+
f"need {n * width}) -- data is truncated (file cut off mid-blob?); "
|
|
1038
|
+
f"returning the {n_available} row(s) that could be decoded"
|
|
1039
|
+
)
|
|
1040
|
+
n = n_available
|
|
1041
|
+
if channel.is_array:
|
|
1042
|
+
# `n` here is the FLAT element count (row_count == n_rows *
|
|
1043
|
+
# array_width for an array channel, confirmed docs/spec.md
|
|
1044
|
+
# section 5); truncate to the largest whole number of complete
|
|
1045
|
+
# rows before reshaping if it doesn't divide evenly.
|
|
1046
|
+
usable_rows, remainder = divmod(n, channel.array_width)
|
|
1047
|
+
if remainder:
|
|
1048
|
+
_warn(
|
|
1049
|
+
f"channel {channel.name!r}: {n} flat element(s) isn't a whole "
|
|
1050
|
+
f"multiple of array_width={channel.array_width} -- dropping the "
|
|
1051
|
+
f"trailing {remainder} incomplete element(s) rather than "
|
|
1052
|
+
f"returning a raggedly-shaped result"
|
|
1053
|
+
)
|
|
1054
|
+
n = usable_rows * channel.array_width
|
|
1055
|
+
if channel.is_string:
|
|
1056
|
+
if _native_ext is not None:
|
|
1057
|
+
values = _native_ext.decode_fixed_width_strings(raw, width, n)
|
|
1058
|
+
else:
|
|
1059
|
+
values = [
|
|
1060
|
+
raw[i * width : (i + 1) * width].split(b"\x00")[0].decode("ascii", errors="replace")
|
|
1061
|
+
for i in range(n)
|
|
1062
|
+
]
|
|
1063
|
+
arr = np.array(values, dtype=object)
|
|
1064
|
+
else:
|
|
1065
|
+
dtype = GS_TYPE_NUMPY_DTYPE[channel.dtype_code]
|
|
1066
|
+
arr = np.frombuffer(raw[: n * width], dtype=dtype).copy()
|
|
1067
|
+
if channel.is_array:
|
|
1068
|
+
arr = arr.reshape(-1, channel.array_width)
|
|
1069
|
+
return arr
|
|
1070
|
+
|
|
1071
|
+
|
|
1072
|
+
@contextlib.contextmanager
|
|
1073
|
+
def _file_handle(path: str, file: Optional[BinaryIO]):
|
|
1074
|
+
"""
|
|
1075
|
+
Yield `file` directly if given (an already-open handle a caller
|
|
1076
|
+
wants reused across many calls), otherwise open `path` fresh and
|
|
1077
|
+
close it on exit -- lets `read_blob_values` support both "just give
|
|
1078
|
+
me a path" (the default, used everywhere else in this module) and
|
|
1079
|
+
"reuse this open handle" (what `GDB` does, to avoid reopening the
|
|
1080
|
+
file on every single read -- benchmarked at ~1.7-1.9x slower per
|
|
1081
|
+
call otherwise, see the project's Rust-plan notes) with the same
|
|
1082
|
+
`with _file_handle(path, file) as f:` call sites either way.
|
|
1083
|
+
"""
|
|
1084
|
+
if file is not None:
|
|
1085
|
+
yield file
|
|
1086
|
+
else:
|
|
1087
|
+
with open(path, "rb") as f:
|
|
1088
|
+
yield f
|
|
1089
|
+
|
|
1090
|
+
|
|
1091
|
+
def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
|
|
1092
|
+
comp_level: int = 0, page_size: Optional[int] = None,
|
|
1093
|
+
file: Optional[BinaryIO] = None):
|
|
1094
|
+
"""
|
|
1095
|
+
Decode a found blob's real row data using the owning channel's
|
|
1096
|
+
already-known type (from the symbol table, docs/provenance/notes.md section 6.2).
|
|
1097
|
+
|
|
1098
|
+
[CONFIRMED] against real ground truth for GS_DOUBLE data and for
|
|
1099
|
+
fixed-width strings, for DB_COMP_NONE (docs/provenance/notes.md section 6.6: a real
|
|
1100
|
+
`fid` blob decoded this way reproduces the exact CSV ground-truth
|
|
1101
|
+
value, and a real `line`-channel blob decodes to the correct real
|
|
1102
|
+
line name repeated once per row).
|
|
1103
|
+
|
|
1104
|
+
Also handles compressed blobs (`comp_level` 1=DB_COMP_SPEED or
|
|
1105
|
+
2=DB_COMP_SIZE), **including multi-page ones** -- [CONFIRMED]
|
|
1106
|
+
against real ground truth for both single- and multi-page
|
|
1107
|
+
DB_COMP_SIZE (a real single-page blob_index=0 decodes to the known
|
|
1108
|
+
constant 5027; a real 36-page array-channel blob decodes to
|
|
1109
|
+
`LEI_Depth`'s exact known real depth profile, `0.0, 3.0, 6.3, 9.9,
|
|
1110
|
+
...`, repeated once per station -- both matching docs/provenance/notes.md section
|
|
1111
|
+
6.5/6.2b's independently-established ground truth exactly) and for
|
|
1112
|
+
both single- and multi-page DB_COMP_SPEED (a real 2-page
|
|
1113
|
+
`Northing_AMGz55` blob decodes to sane real coordinates with real
|
|
1114
|
+
`rDUMMY` sentinels). See docs/provenance/notes.md section 6.6d: a multi-page blob is
|
|
1115
|
+
simply one continuous compressed stream spanning the whole
|
|
1116
|
+
`n_pages*page_size` span, not one independently-framed chunk per
|
|
1117
|
+
page -- no special multi-page logic was actually needed once this
|
|
1118
|
+
was verified, just reading the full span instead of one page.
|
|
1119
|
+
|
|
1120
|
+
**A real third on-disk variant, auto-detected here rather than
|
|
1121
|
+
assumed away (docs/provenance/notes.md section 6.6b):** even inside a file that
|
|
1122
|
+
genuinely declares (and elsewhere uses) DB_COMP_SPEED, some
|
|
1123
|
+
individual blobs turn out to carry no chunk wrapper at all -- just
|
|
1124
|
+
the plain 48-byte DB_COMP_NONE-style header with real, directly
|
|
1125
|
+
readable data straight after it (confirmed on a real
|
|
1126
|
+
`Easting_AMGz55` blob in `DB_EM_293.gdb`: decoding it as if
|
|
1127
|
+
`comp_level==0` reproduces sane, real coordinate values with real
|
|
1128
|
+
`rDUMMY=-1.0E32` sentinels in the expected places). When
|
|
1129
|
+
`comp_level != 0`, this function checks for the 16-byte chunk magic
|
|
1130
|
+
at the 56-byte-header position first and only falls back to the
|
|
1131
|
+
genuinely-compressed path if it's actually there -- otherwise it
|
|
1132
|
+
decodes the blob exactly like a DB_COMP_NONE one.
|
|
1133
|
+
|
|
1134
|
+
**Fails gracefully, per an explicit engineering request:** a
|
|
1135
|
+
negative `row_count` (a reserved/administrative blob, docs/provenance/notes.md
|
|
1136
|
+
section 6.4/6.9, not real data), a channel type this reader can't
|
|
1137
|
+
decode, a truncated read (file cut off mid-blob), an unrecognized
|
|
1138
|
+
chunk subtype, or a chunk that fails to decompress (corrupt/
|
|
1139
|
+
truncated compressed data, or `lzrw1.LZRW1DecodeError`) all return
|
|
1140
|
+
an empty `ndarray` (`np.array([])`) with a `GDBParseWarning`
|
|
1141
|
+
describing what went wrong, instead of raising and losing the
|
|
1142
|
+
caller's place in a larger loop (e.g. a whole-file scan that's
|
|
1143
|
+
decoded hundreds of blobs already). The one exception where full
|
|
1144
|
+
graceful salvage wasn't attempted is a truncated/corrupt
|
|
1145
|
+
*compressed* stream: unlike the plain-data case, there's no simple
|
|
1146
|
+
way to hand back "the first K decoded values" from a partially-
|
|
1147
|
+
decompressed zlib/LZRW1 stream, so those cases warn and return an
|
|
1148
|
+
empty array rather than a partial decode -- documented here rather
|
|
1149
|
+
than silently implied to be as complete as the plain-data
|
|
1150
|
+
truncation handling.
|
|
1151
|
+
|
|
1152
|
+
Return shape/dtype: see `_decode_numeric_or_string`'s docstring --
|
|
1153
|
+
1-D `ndarray` for a scalar channel, 2-D `(n_rows, array_width)` for
|
|
1154
|
+
a VA/array channel, dtype matching the channel's `GS_*` type or
|
|
1155
|
+
`object` (holding `str`) for a string channel.
|
|
1156
|
+
|
|
1157
|
+
`file`: an already-open binary file handle for `path`, reused
|
|
1158
|
+
instead of opening `path` fresh -- pass this if you're calling this
|
|
1159
|
+
function many times for the same file (e.g. `GDB` does, internally).
|
|
1160
|
+
Reopening `path` on every call is real, measured overhead (~1.7-1.9x
|
|
1161
|
+
slower per call, benchmarked against this project's real sample
|
|
1162
|
+
corpus -- see the Rust-plan's M4 notes); `file=None` (the default)
|
|
1163
|
+
keeps this function's plain "just give me a path" behavior for every
|
|
1164
|
+
other caller.
|
|
1165
|
+
"""
|
|
1166
|
+
if comp_level == 0:
|
|
1167
|
+
if blob.row_count < 0:
|
|
1168
|
+
_warn(
|
|
1169
|
+
f"blob_index={blob.blob_index}: negative row_count "
|
|
1170
|
+
f"({blob.row_count}) -- this is one of the reserved/"
|
|
1171
|
+
f"administrative blobs flagged [UNKNOWN] in docs/provenance/notes.md "
|
|
1172
|
+
f"section 6.4/6.9, not a real data blob; returning no values"
|
|
1173
|
+
)
|
|
1174
|
+
return np.array([])
|
|
1175
|
+
width = _element_width(channel)
|
|
1176
|
+
if width is None:
|
|
1177
|
+
_warn(
|
|
1178
|
+
f"channel {channel.name!r} has dtype_code={channel.dtype_code}, a "
|
|
1179
|
+
f"type this reader doesn't know how to decode -- returning no values"
|
|
1180
|
+
)
|
|
1181
|
+
return np.array([])
|
|
1182
|
+
with _file_handle(path, file) as f:
|
|
1183
|
+
f.seek(blob.data_offset)
|
|
1184
|
+
raw = f.read(blob.row_count * width)
|
|
1185
|
+
return _decode_numeric_or_string(raw, channel, blob.row_count)
|
|
1186
|
+
|
|
1187
|
+
# comp_level != 0: could still be any of three real on-disk variants
|
|
1188
|
+
# (docs/provenance/notes.md section 6.6b) -- check which one this specific blob
|
|
1189
|
+
# actually is rather than assuming from the file-level comp_level.
|
|
1190
|
+
with _file_handle(path, file) as f:
|
|
1191
|
+
f.seek(blob.offset + COMPRESSED_BLOB_HEADER_SIZE)
|
|
1192
|
+
chunk_magic_probe = f.read(8)
|
|
1193
|
+
if len(chunk_magic_probe) < 8:
|
|
1194
|
+
_warn(
|
|
1195
|
+
f"blob_index={blob.blob_index}: file ends before the compressed-blob "
|
|
1196
|
+
f"header/chunk-magic region (offset {blob.offset + COMPRESSED_BLOB_HEADER_SIZE}) "
|
|
1197
|
+
f"could be fully read -- truncated mid-blob; returning no values"
|
|
1198
|
+
)
|
|
1199
|
+
return np.array([])
|
|
1200
|
+
if chunk_magic_probe != _lzrw1.CHUNK_MAGIC:
|
|
1201
|
+
# Variant 3: no chunk wrapper at all -- a "bare" blob, byte-for-byte
|
|
1202
|
+
# identical in layout to a DB_COMP_NONE one, just living inside an
|
|
1203
|
+
# otherwise-compressed file. Use the plain 48-byte-header fields,
|
|
1204
|
+
# which decoded sanely for real in this exact case.
|
|
1205
|
+
if blob.row_count < 0:
|
|
1206
|
+
_warn(
|
|
1207
|
+
f"blob_index={blob.blob_index}: negative row_count "
|
|
1208
|
+
f"({blob.row_count}) -- this is one of the reserved/"
|
|
1209
|
+
f"administrative blobs flagged [UNKNOWN] in docs/provenance/notes.md "
|
|
1210
|
+
f"section 6.4/6.9, not a real data blob; returning no values"
|
|
1211
|
+
)
|
|
1212
|
+
return np.array([])
|
|
1213
|
+
width = _element_width(channel)
|
|
1214
|
+
if width is None:
|
|
1215
|
+
_warn(
|
|
1216
|
+
f"channel {channel.name!r} has dtype_code={channel.dtype_code}, a "
|
|
1217
|
+
f"type this reader doesn't know how to decode -- returning no values"
|
|
1218
|
+
)
|
|
1219
|
+
return np.array([])
|
|
1220
|
+
with _file_handle(path, file) as f:
|
|
1221
|
+
f.seek(blob.data_offset)
|
|
1222
|
+
raw = f.read(blob.row_count * width)
|
|
1223
|
+
return _decode_numeric_or_string(raw, channel, blob.row_count)
|
|
1224
|
+
|
|
1225
|
+
# Compressed (DB_COMP_SPEED / DB_COMP_SIZE): 56-byte blob header,
|
|
1226
|
+
# then the shared 16-byte page-primitive chunk magic -- see
|
|
1227
|
+
# COMPRESSED_BLOB_HEADER_SIZE and docs/provenance/notes.md section 6.6b/6.6d.
|
|
1228
|
+
#
|
|
1229
|
+
# Multi-page blobs (blob.n_pages > 1) are [CONFIRMED] (section 6.6d)
|
|
1230
|
+
# to be a SINGLE continuous compressed stream spanning the whole
|
|
1231
|
+
# n_pages*page_size span -- NOT one independently-framed chunk per
|
|
1232
|
+
# page. There is no per-page re-framing to handle: reading the full
|
|
1233
|
+
# span and decompressing it as one stream (zlib.decompressobj()
|
|
1234
|
+
# naturally stops at the real end of stream and reports the rest as
|
|
1235
|
+
# padding; the LZRW1 chunk header's own decompressed_length/
|
|
1236
|
+
# chunk_length fields already span the full compressed length
|
|
1237
|
+
# regardless of how many pages it spilled into) is sufficient.
|
|
1238
|
+
if page_size is None:
|
|
1239
|
+
with _file_handle(path, file) as f:
|
|
1240
|
+
header = f.read(128)
|
|
1241
|
+
try:
|
|
1242
|
+
page_size = struct.unpack_from("<i", header, 100)[0]
|
|
1243
|
+
except struct.error:
|
|
1244
|
+
_warn(
|
|
1245
|
+
f"blob_index={blob.blob_index}: header too short to read "
|
|
1246
|
+
f"page_size -- cannot decode, returning no values"
|
|
1247
|
+
)
|
|
1248
|
+
return np.array([])
|
|
1249
|
+
with _file_handle(path, file) as f:
|
|
1250
|
+
f.seek(blob.offset + COMPRESSED_BLOB_HEADER_SIZE)
|
|
1251
|
+
expected_span = blob.n_pages * page_size - COMPRESSED_BLOB_HEADER_SIZE
|
|
1252
|
+
raw_span = f.read(expected_span)
|
|
1253
|
+
if len(raw_span) < expected_span:
|
|
1254
|
+
_warn(
|
|
1255
|
+
f"blob_index={blob.blob_index}: expected {expected_span} byte(s) of "
|
|
1256
|
+
f"compressed payload but the file only had {len(raw_span)} available "
|
|
1257
|
+
f"-- truncated mid-blob; attempting to decode what's there, but this "
|
|
1258
|
+
f"may fail or be incomplete"
|
|
1259
|
+
)
|
|
1260
|
+
if len(raw_span) < 16:
|
|
1261
|
+
_warn(
|
|
1262
|
+
f"blob_index={blob.blob_index}: not enough bytes to read even the "
|
|
1263
|
+
f"chunk sub-header ({len(raw_span)} available, need 16) -- cannot "
|
|
1264
|
+
f"decode, returning no values"
|
|
1265
|
+
)
|
|
1266
|
+
return np.array([])
|
|
1267
|
+
subtype = struct.unpack_from("<i", raw_span, 8)[0]
|
|
1268
|
+
if subtype == _lzrw1.DB_COMP_SIZE:
|
|
1269
|
+
try:
|
|
1270
|
+
d = zlib.decompressobj()
|
|
1271
|
+
decompressed = d.decompress(raw_span[16:])
|
|
1272
|
+
except zlib.error as e:
|
|
1273
|
+
_warn(
|
|
1274
|
+
f"blob_index={blob.blob_index}: zlib decompression failed ({e}) "
|
|
1275
|
+
f"-- likely truncated or corrupt compressed data; returning no values"
|
|
1276
|
+
)
|
|
1277
|
+
return np.array([])
|
|
1278
|
+
elif subtype == _lzrw1.DB_COMP_SPEED:
|
|
1279
|
+
try:
|
|
1280
|
+
chunk = _lzrw1.parse_chunk_header(raw_span, 0)
|
|
1281
|
+
decompressed = _lzrw1.decode_speed_chunk(raw_span, chunk)
|
|
1282
|
+
except _lzrw1.LZRW1DecodeError as e:
|
|
1283
|
+
_warn(
|
|
1284
|
+
f"blob_index={blob.blob_index}: LZRW1 chunk decode failed ({e}) "
|
|
1285
|
+
f"-- likely truncated or corrupt compressed data, or an "
|
|
1286
|
+
f"unrecognized chunk variant; returning no values"
|
|
1287
|
+
)
|
|
1288
|
+
return np.array([])
|
|
1289
|
+
else:
|
|
1290
|
+
_warn(
|
|
1291
|
+
f"blob_index={blob.blob_index}: unrecognized chunk subtype={subtype} "
|
|
1292
|
+
f"(expected {_lzrw1.DB_COMP_SIZE}=Size or {_lzrw1.DB_COMP_SPEED}=Speed) "
|
|
1293
|
+
f"-- returning no values"
|
|
1294
|
+
)
|
|
1295
|
+
return np.array([])
|
|
1296
|
+
return _decode_numeric_or_string(decompressed, channel)
|
|
1297
|
+
|
|
1298
|
+
|
|
1299
|
+
if __name__ == "__main__":
|
|
1300
|
+
import sys
|
|
1301
|
+
|
|
1302
|
+
if len(sys.argv) != 2:
|
|
1303
|
+
print("usage: python -m pygdb.gdb_reader <path-to.gdb>")
|
|
1304
|
+
raise SystemExit(1)
|
|
1305
|
+
|
|
1306
|
+
path = sys.argv[1]
|
|
1307
|
+
with open(path, "rb") as f:
|
|
1308
|
+
header = f.read(128)
|
|
1309
|
+
if not check_magic(header):
|
|
1310
|
+
print("WARNING: file does not start with the expected '!CBD' magic")
|
|
1311
|
+
elif not magic_signature_matches_common_case(header):
|
|
1312
|
+
print(
|
|
1313
|
+
"NOTE: '!CBD' magic OK, but bytes 8-15 differ from the common "
|
|
1314
|
+
f"case (got {header[4:16].hex()}) -- see docs/provenance/notes.md, seen once before"
|
|
1315
|
+
)
|
|
1316
|
+
fields = header_fields(header)
|
|
1317
|
+
print(f"header fields: {fields}")
|
|
1318
|
+
|
|
1319
|
+
channels = read_channels(path)
|
|
1320
|
+
print(f"\n{len(channels)} channel(s) found:\n")
|
|
1321
|
+
print(f"{'#':>3} {'name':30s} {'type':16s} {'format':8s} {'width':>5s} {'basetype':13s}")
|
|
1322
|
+
for c in channels:
|
|
1323
|
+
width = f"{c.array_width}*" if c.is_array else str(c.array_width)
|
|
1324
|
+
print(
|
|
1325
|
+
f"{c.index:3d} {c.name:30s} {c.type_name:16s} {c.format_name:8s} "
|
|
1326
|
+
f"{width:>5s} {c.array_basetype_name:13s}"
|
|
1327
|
+
)
|
|
1328
|
+
n_array = sum(1 for c in channels if c.is_array)
|
|
1329
|
+
if n_array:
|
|
1330
|
+
print(f"\n({n_array} of {len(channels)} channels are VA/array channels, marked with '*' in width)")
|
|
1331
|
+
|
|
1332
|
+
by_slot = {c.index: c for c in channels}
|
|
1333
|
+
comp_level = fields["comp_level"]
|
|
1334
|
+
print(
|
|
1335
|
+
f"\nWalking the blob chain (comp_level={comp_level}) looking for "
|
|
1336
|
+
"the first blob that decodes to real data (the first few are "
|
|
1337
|
+
"often reserved/administrative ones, see docs/provenance/notes.md section 6.6)..."
|
|
1338
|
+
)
|
|
1339
|
+
shown = 0
|
|
1340
|
+
for blob in iter_blobs(path, max_blobs=50):
|
|
1341
|
+
if blob.row_count is not None and blob.row_count <= 0 and comp_level == 0:
|
|
1342
|
+
continue # cheap skip for the common admin-blob case, comp_level==0 only
|
|
1343
|
+
line_slot, chan_slot = blob.line_channel(fields["chans_max"])
|
|
1344
|
+
chan = by_slot.get(chan_slot)
|
|
1345
|
+
if chan is None:
|
|
1346
|
+
continue
|
|
1347
|
+
# read_blob_values() never raises for a blob/channel it can't decode --
|
|
1348
|
+
# it warns (GDBParseWarning) and returns an empty array instead, so a
|
|
1349
|
+
# plain empty-result check is all that's needed here; no try/except
|
|
1350
|
+
# required.
|
|
1351
|
+
values = read_blob_values(path, blob, chan, comp_level=comp_level,
|
|
1352
|
+
page_size=fields["page_size"])
|
|
1353
|
+
if len(values) == 0:
|
|
1354
|
+
continue
|
|
1355
|
+
print(
|
|
1356
|
+
f" line_slot={line_slot} channel_slot={chan_slot} ({chan.name}), "
|
|
1357
|
+
f"n_pages={blob.n_pages}, offset={blob.offset}, first 3 values: {values[:3]}"
|
|
1358
|
+
)
|
|
1359
|
+
shown += 1
|
|
1360
|
+
if shown >= 3:
|
|
1361
|
+
break
|
|
1362
|
+
if shown == 0:
|
|
1363
|
+
print(" (no easily-decodable blob found in the first 50 of the chain)")
|
|
1364
|
+
print(
|
|
1365
|
+
"\nUse iter_blobs()/find_blob()/read_blob_values() to read real "
|
|
1366
|
+
"channel data for any (line, channel) pair, single- or multi-page, "
|
|
1367
|
+
"any compression mode -- see docs/provenance/notes.md section 6.6/6.6b/6.6d."
|
|
1368
|
+
)
|