python-gdb 0.1.0__cp312-abi3-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pygdb/gdb_reader.py ADDED
@@ -0,0 +1,1368 @@
1
+ """
2
+ Clean-room reader for Geosoft .gdb database files.
3
+
4
+ Status: reads the file magic/header, walks the channel and line symbol
5
+ tables, and reads real channel data for every compression mode
6
+ (DB_COMP_NONE, DB_COMP_SPEED/LZRW1, DB_COMP_SIZE/zlib, single- or
7
+ multi-page blobs, and the "bare"/uncompressed-blob-inside-a-compressed-
8
+ file variant) via iter_blobs()/find_blob()/read_blob_values() -- see
9
+ docs/provenance/notes.md sections 6.6/6.6b/6.6d for the full derivation.
10
+ Still open: REG/IPJ registry content is only partially decoded (section
11
+ 6.7/6.8), and a handful of header/record fields remain [UNKNOWN] -- see
12
+ docs/spec.md and docs/provenance/notes.md for the complete, current
13
+ picture.
14
+
15
+ Robustness: this reader is designed to degrade gracefully rather than
16
+ hard-crash on a blob/chunk/record it can't parse -- a truncated file
17
+ (cut-off download, or a blob chain that runs past EOF), an
18
+ administrative-blob variant it doesn't recognize, an unrecognized
19
+ channel type, or anything else that doesn't fit the confirmed
20
+ structure. Functions return whatever they successfully decoded up to
21
+ the point of trouble (an empty list/dict in the worst case) rather than
22
+ raising, and always pair that with a `GDBParseWarning` (see its
23
+ docstring) identifying what couldn't be decoded and why. This is a
24
+ deliberate engineering choice, not new format research -- see docs/provenance/notes.md
25
+ for the design rationale and docs/provenance/log.md for when/why it was added.
26
+
27
+ Confidence markers below mirror docs/provenance/notes.md: [CONFIRMED] =
28
+ verified against two independent real files with a falsifiable
29
+ structural test; [LIKELY] = passed one real test but not independently
30
+ cross-checked; [GUESS] = plausible pattern, not tested; values otherwise
31
+ unlabeled in comments are the [UNKNOWN] raw offsets, kept for whoever
32
+ continues this.
33
+
34
+ Every numeric constant used for interpretation (GS_* type codes,
35
+ DB_CHAN_FORMAT_*, DB_SYMB_NAME_SIZE, etc.) comes from reading Geosoft's
36
+ own published, BSD-licensed source at
37
+ https://github.com/GeosoftInc/gxpy/blob/master/geosoft/gxapi/__init__.py
38
+ -- publicly available vendor source, not obtained by running anything.
39
+
40
+ No Geosoft software of any kind was installed, imported, or executed to
41
+ produce this code.
42
+ """
43
+
44
+ from __future__ import annotations
45
+
46
+ import contextlib
47
+ import re
48
+ import struct
49
+ import warnings
50
+ import zlib
51
+ from dataclasses import dataclass
52
+
53
+ from typing import BinaryIO, List, Optional, Tuple
54
+
55
+ import numpy as np
56
+
57
+ from . import lzrw1 as _lzrw1
58
+
59
+ try:
60
+ from . import _native as _native_ext
61
+ except ImportError:
62
+ _native_ext = None
63
+
64
+
65
+ class GDBParseWarning(RuntimeWarning):
66
+ """
67
+ Warned (via `warnings.warn`) whenever this reader hits a blob,
68
+ chunk, or record it can't parse -- an unexpected byte sequence, a
69
+ file that ends prematurely (truncated download, or a blob chain
70
+ that runs past EOF), an administrative-blob variant it doesn't
71
+ recognize, or anything else that doesn't fit the confirmed
72
+ structure. This reader is designed to degrade gracefully rather
73
+ than hard-crash on this whole class of problem: functions return
74
+ whatever they successfully decoded up to the point of trouble
75
+ (a shorter-than-expected list, an empty list, or in the worst case
76
+ an empty result) instead of raising, and a `GDBParseWarning`
77
+ describing what couldn't be decoded and why is always issued
78
+ alongside, so a caller can tell a clean, complete result from a
79
+ partial one and go investigate. See docs/provenance/notes.md's "reader robustness"
80
+ notes for the design rationale (an explicit engineering request,
81
+ not new format research).
82
+
83
+ This does not apply to a handful of genuine precondition failures
84
+ that aren't "this file has an interesting anomaly" (e.g. calling
85
+ `read_blob_values` with a `channel`/`blob` pair that can't
86
+ possibly match) -- those still raise normally.
87
+ """
88
+
89
+
90
+ def _warn(msg: str) -> None:
91
+ warnings.warn(msg, GDBParseWarning, stacklevel=3)
92
+
93
+
94
+ MAGIC = b"!CBD"
95
+ # The 16-byte header opening was found byte-identical across both real
96
+ # sample files (Magnetic_Data.gdb and Radiometric_Data.gdb) -- see
97
+ # docs/provenance/notes.md 6.1. Treated here as a fixed format/version signature.
98
+ HEADER_SIGNATURE = bytes.fromhex("21434244000000000000021008010000".replace(" ", ""))[:16]
99
+
100
+ SYMBOL_RECORD_SIZE = 128 # [CONFIRMED] -- constant stride of every symbol
101
+ # table record (channel, line, and user records
102
+ # all observed at this stride)
103
+
104
+ # Vendor-published GX type codes (geosoft/gxapi/__init__.py). Positive
105
+ # values only here -- negative values in a channel record instead mean
106
+ # "string, N bytes wide" where N = -value (see decode_dtype below).
107
+ GS_TYPE_NAMES = {
108
+ 0: "GS_BYTE",
109
+ 1: "GS_USHORT",
110
+ 2: "GS_SHORT",
111
+ 3: "GS_LONG",
112
+ 4: "GS_FLOAT",
113
+ 5: "GS_DOUBLE",
114
+ 6: "GS_UBYTE",
115
+ 7: "GS_ULONG",
116
+ 8: "GS_LONG64",
117
+ 9: "GS_ULONG64",
118
+ 10: "GS_FLOAT3D",
119
+ 11: "GS_DOUBLE3D",
120
+ 12: "GS_FLOAT2D",
121
+ 13: "GS_DOUBLE2D",
122
+ }
123
+
124
+ # struct format codes for each GS_* type -- used for `_element_width`'s
125
+ # byte-width math (struct.calcsize), independent of whichever decode
126
+ # strategy actually reads the bytes.
127
+ GS_TYPE_STRUCT = {
128
+ 0: "b", # GS_BYTE (signed, per GS_S1* constants)
129
+ 1: "H", # GS_USHORT
130
+ 2: "h", # GS_SHORT
131
+ 3: "i", # GS_LONG
132
+ 4: "f", # GS_FLOAT
133
+ 5: "d", # GS_DOUBLE
134
+ 6: "B", # GS_UBYTE
135
+ 7: "I", # GS_ULONG
136
+ 8: "q", # GS_LONG64
137
+ 9: "Q", # GS_ULONG64
138
+ }
139
+
140
+ # Little-endian numpy dtype strings for each GS_* type (explicit `<`
141
+ # byte-order prefix, matching this format's confirmed little-endian
142
+ # layout everywhere else -- a platform-native dtype would silently
143
+ # misdecode on a big-endian host). Used by `_decode_numeric_or_string`
144
+ # for `np.frombuffer`; `GS_TYPE_STRUCT` above is kept separately since
145
+ # `_element_width` only needs a byte count, not a full dtype.
146
+ GS_TYPE_NUMPY_DTYPE = {
147
+ 0: "<i1", # GS_BYTE (signed)
148
+ 1: "<u2", # GS_USHORT
149
+ 2: "<i2", # GS_SHORT
150
+ 3: "<i4", # GS_LONG
151
+ 4: "<f4", # GS_FLOAT
152
+ 5: "<f8", # GS_DOUBLE
153
+ 6: "<u1", # GS_UBYTE
154
+ 7: "<u4", # GS_ULONG
155
+ 8: "<i8", # GS_LONG64
156
+ 9: "<u8", # GS_ULONG64
157
+ }
158
+
159
+ DB_CHAN_FORMAT_NAMES = {
160
+ 0: "NORMAL",
161
+ 1: "EXP",
162
+ 2: "TIME",
163
+ 3: "DATE",
164
+ 4: "GEOGR",
165
+ 5: "SIGDIG",
166
+ 6: "HEX",
167
+ }
168
+
169
+ # Vendor-published DB_ARRAY_BASETYPE_* constants (geosoft/gxapi/__init__.py).
170
+ # [LIKELY] match for the int16 field at relative offset +86 -- see
171
+ # docs/provenance/notes.md "VA / array channels" section. Confirmed to hold value 1
172
+ # (TIME_WINDOWS) on real multi-gate TEM decay-curve array channels, but
173
+ # also seen as a constant non-zero value across *every* channel (including
174
+ # obviously-scalar ones) in three older real files, so treat this field's
175
+ # meaning with more caution than the array-width field below.
176
+ DB_ARRAY_BASETYPE_NAMES = {
177
+ 0: "NONE",
178
+ 1: "TIME_WINDOWS",
179
+ 2: "TIMES",
180
+ 3: "FREQUENCIES",
181
+ 4: "ELEVATIONS",
182
+ 5: "DEPTHS",
183
+ 6: "VELOCITIES",
184
+ 7: "DISCRETE_TIME_WINDOWS",
185
+ 8: "ENERGIES",
186
+ }
187
+
188
+
189
+ @dataclass(eq=False)
190
+ class ChannelRecord:
191
+ # eq=False -- keep the default identity-based __eq__/__hash__ instead
192
+ # of dataclass's usual field-by-field one, so instances stay hashable
193
+ # (GDB.iter_line() yields these and documents `dict(...)` keyed by
194
+ # the record itself as safe -- see its docstring -- which needs
195
+ # __hash__ to actually work). Value equality between two separately-
196
+ # constructed-but-identical records is never used anywhere in this
197
+ # codebase; every real lookup returns the same cached instance from
198
+ # GDB.channels, so identity is all that's ever needed in practice.
199
+ index: int
200
+ offset: int
201
+ name: str
202
+ dtype_code: int # raw int16 value: positive=GS_* type, negative=-string_width
203
+ format_code: int
204
+ raw: bytes
205
+ array_width: int = 1 # [CONFIRMED] relative offset +118, int16. 1 = plain
206
+ # scalar channel (the overwhelming majority of real
207
+ # channels seen). >1 = a true VA/array channel
208
+ # storing that many elements per fiducial "cell" --
209
+ # e.g. 24 (time-decay gates) or 30 (depth layers) in
210
+ # the real AG106386 Georgetown conductivity file.
211
+ # Independently cross-checked against that same
212
+ # file's plain-text ASCII sibling (.dfn) format,
213
+ # which spells out "30F10.4" (Fortran-style: 30
214
+ # repetitions of a float field) for the exact same
215
+ # channel name -- see docs/provenance/notes.md.
216
+ array_basetype_code: int = 0 # [LIKELY] relative offset +86, int16.
217
+ name_is_clean: bool = True # False = name field was NUL-unterminated / had
218
+ # non-printable bytes -- see docs/provenance/notes.md re: older
219
+ # (pre-2020, e.g. 1990s GEOTEM) files sometimes
220
+ # leaving unused capacity slots un-zeroed rather
221
+ # than clean, unlike the 2020 USGS samples.
222
+
223
+ @property
224
+ def is_string(self) -> bool:
225
+ return self.dtype_code < 0
226
+
227
+ @property
228
+ def string_width(self) -> Optional[int]:
229
+ return -self.dtype_code if self.is_string else None
230
+
231
+ @property
232
+ def type_name(self) -> str:
233
+ if self.is_string:
234
+ return f"string[{self.string_width}]"
235
+ return GS_TYPE_NAMES.get(self.dtype_code, f"unknown({self.dtype_code})")
236
+
237
+ @property
238
+ def format_name(self) -> str:
239
+ return DB_CHAN_FORMAT_NAMES.get(self.format_code, f"unknown({self.format_code})")
240
+
241
+ @property
242
+ def is_array(self) -> bool:
243
+ """True for a real VA/array channel (array_width > 1). [CONFIRMED]."""
244
+ return self.array_width > 1
245
+
246
+ @property
247
+ def array_basetype_name(self) -> str:
248
+ return DB_ARRAY_BASETYPE_NAMES.get(
249
+ self.array_basetype_code, f"unknown({self.array_basetype_code})"
250
+ )
251
+
252
+ @property
253
+ def looks_sane(self) -> bool:
254
+ """
255
+ Heuristic sanity check distinguishing a real channel record from
256
+ leftover-garbage bytes that happen to decode a clean printable
257
+ name (observed for real in DB_Mag_833.gdb -- see docs/provenance/notes.md). Real
258
+ records seen so far always have dtype either a known GS_* code
259
+ (0-13) or a small negative string width, and a format code in the
260
+ known DB_CHAN_FORMAT_* range (0-6).
261
+ """
262
+ dtype_ok = self.dtype_code in GS_TYPE_NAMES or -256 <= self.dtype_code < 0
263
+ format_ok = self.format_code in DB_CHAN_FORMAT_NAMES
264
+ return dtype_ok and format_ok
265
+
266
+
267
+ def _read_name(raw: bytes, offset: int, max_len: int = 64):
268
+ """
269
+ Read a NUL-padded name field.
270
+
271
+ Returns (name, is_clean). is_clean is False when the field has no NUL
272
+ terminator within max_len, or contains non-printable bytes before the
273
+ terminator -- observed [CONFIRMED against a real 1992 file,
274
+ DB_Mag_293.gdb from GSQ's Holroy River survey] to happen for *unused*
275
+ channel-table capacity slots in at least one older (pre-2020) real
276
+ .gdb file: unlike the 2020 USGS samples (where unused capacity slots
277
+ are cleanly zeroed), this older file leaves unused slots holding
278
+ leftover/uninitialized bytes that happen to look like binary float
279
+ data, not padding. Treat is_clean=False slots as "unused capacity,
280
+ contents undefined" rather than as real channels.
281
+ """
282
+ field = raw[offset : offset + max_len]
283
+ nul = field.find(b"\x00")
284
+ if nul == -1:
285
+ return field.decode("ascii", errors="replace"), False
286
+ text = field[:nul]
287
+ is_clean = all(32 <= b < 127 for b in text)
288
+ return text.decode("ascii", errors="replace"), is_clean
289
+
290
+
291
+ def check_magic(data: bytes) -> bool:
292
+ """
293
+ [CONFIRMED] the 4-byte "!CBD" prefix against 9/9 real files across two
294
+ independent sources (2020 USGS Mojave survey, 1990s-2020s GSQ
295
+ Queensland surveys from three different TEM systems/vendors).
296
+
297
+ The FULL 16-byte HEADER_SIGNATURE is only [LIKELY] -- it matched
298
+ exactly in 8/9 real files, but one real file
299
+ (DB_Mag_Elaine_1003.gdb, from GSQ's Mount Gordon delivery) has
300
+ `f0 f0 f0 f0` at bytes 8-11 instead of the usual `00 00 00 00`. That
301
+ file is otherwise structurally normal (chans_max/users_max/page_size
302
+ all decode sanely), so this looks like a real, if rare, variation in
303
+ that sub-block rather than a different format entirely -- flagged
304
+ [UNKNOWN] in docs/provenance/notes.md. Only the 4-byte magic is treated as a hard
305
+ requirement here; the rest of the signature is reported separately.
306
+ """
307
+ return data[:4] == MAGIC
308
+
309
+
310
+ def magic_signature_matches_common_case(data: bytes) -> bool:
311
+ """True if bytes 0-15 exactly match the signature seen in most real files."""
312
+ return data[:16] == HEADER_SIGNATURE
313
+
314
+
315
+ def header_fields(data: bytes) -> dict:
316
+ """
317
+ Extract the header int32 fields whose approximate meaning we have
318
+ some confidence in. See docs/provenance/notes.md section 6.1 for the full table
319
+ including the still-unknown offsets, and for why each confidence
320
+ label was assigned.
321
+
322
+ Fails gracefully on a truncated/too-short header: any field that
323
+ can't be read (not enough bytes at its offset) is set to `None`
324
+ in the returned dict rather than raising, and a `GDBParseWarning`
325
+ is issued naming which field(s) were affected. Callers that need a
326
+ field should check for `None` before using it (every function in
327
+ this module that consumes `header_fields()` output does).
328
+ """
329
+ result = {}
330
+ for name, offset in (("chans_max", 24), ("users_max", 40),
331
+ ("page_size", 100), ("comp_level", 120)):
332
+ try:
333
+ result[name] = struct.unpack_from("<i", data, offset)[0]
334
+ except struct.error:
335
+ _warn(
336
+ f"header truncated: only {len(data)} byte(s) available, not enough "
337
+ f"to read '{name}' at offset {offset} -- returning None for it"
338
+ )
339
+ result[name] = None
340
+ # comp_level==1 (DB_COMP_SPEED) does NOT mean the payload is zlib --
341
+ # confirmed it is NOT (docs/provenance/notes.md section 6.5b), it's canonical LZRW1
342
+ # (section 6.5c). comp_level==2 (DB_COMP_SIZE) IS confirmed real zlib.
343
+ return result
344
+
345
+
346
+ def _parse_channel_record(data: bytes, rec_start: int, index: int) -> ChannelRecord:
347
+ raw = data[rec_start : rec_start + SYMBOL_RECORD_SIZE]
348
+ name, is_clean = _read_name(raw, 8)
349
+ dtype_code = struct.unpack_from("<h", raw, 84)[0]
350
+ array_basetype_code = struct.unpack_from("<h", raw, 86)[0]
351
+ format_code = struct.unpack_from("<h", raw, 92)[0]
352
+ array_width = struct.unpack_from("<h", raw, 118)[0]
353
+ return ChannelRecord(
354
+ index=index, offset=rec_start, name=name,
355
+ dtype_code=dtype_code, format_code=format_code, raw=raw,
356
+ array_width=array_width, array_basetype_code=array_basetype_code,
357
+ name_is_clean=is_clean,
358
+ )
359
+
360
+
361
+ def find_channel_table(data: bytes, search_window=(0, None)) -> int:
362
+ """
363
+ Locate the start of the channel symbol table.
364
+
365
+ Strategy [CONFIRMED against 2 real 2020 USGS files, RE-CONFIRMED --
366
+ with one revision -- against 5 more real 1990s-2020s GSQ files, see
367
+ docs/provenance/notes.md/docs/provenance/log.md "pressure test" round]: search for the default
368
+ super-user name (from GXDB.create()'s documented default
369
+ `super="SUPER"`). The channel table is found to occupy exactly
370
+ `chans_max` consecutive 128-byte records immediately before the user
371
+ table, i.e.
372
+ channel_table_start == offset_of(super_name) - 8 - chans_max*128
373
+
374
+ This isn't a generic file-format constant we can hardcode a single
375
+ offset for -- it depends on `chans_max`, which itself varies between
376
+ files -- so we compute it.
377
+
378
+ REVISION from the original derivation: the two 2020 USGS files both
379
+ had the default super-user name stored as literal uppercase ASCII
380
+ "SUPER". Five real 1990s-2020s GSQ files instead have it stored as
381
+ lowercase "super" -- confirmed to be the *same* structural pattern
382
+ (same 128-byte-per-record math, same position relative to the
383
+ channel table) once the case is corrected, not a different layout.
384
+ Search for both cases. (One of the GSQ files, DB_Mag_833.gdb, also
385
+ demonstrated that the literal string can coincidentally appear
386
+ elsewhere in a file, e.g. inside embedded metadata blobs, and that
387
+ the *word* "super"/"SUPER" appearing is not on its own sufficient --
388
+ a naive first-match there pointed at a bogus offset. Confirmed
389
+ correct instead via an independent generic 128-byte-periodicity scan
390
+ that landed on the identical answer once cross-checked.)
391
+
392
+ Every occurrence of "SUPER"/"super" is tried and the first one whose
393
+ implied table start decodes a *clean* (NUL-terminated, printable)
394
+ channel name is used.
395
+ """
396
+ lo, hi = search_window
397
+ if hi is None:
398
+ hi = len(data)
399
+ chans_max = header_fields(data)["chans_max"]
400
+
401
+ start = lo
402
+ while True:
403
+ idx_upper = data.find(b"SUPER", start, hi)
404
+ idx_lower = data.find(b"super", start, hi)
405
+ candidates = [i for i in (idx_upper, idx_lower) if i != -1]
406
+ super_idx = min(candidates) if candidates else -1
407
+ if super_idx == -1:
408
+ raise ValueError(
409
+ "could not find a 'SUPER' user record that implies a valid "
410
+ "channel table in the search window; try widening `search_window`"
411
+ )
412
+ super_rec_start = super_idx - 8
413
+ table_start = super_rec_start - chans_max * SYMBOL_RECORD_SIZE
414
+ if table_start >= 0:
415
+ raw = data[table_start : table_start + SYMBOL_RECORD_SIZE]
416
+ if len(raw) == SYMBOL_RECORD_SIZE:
417
+ name, is_clean = _read_name(raw, 8)
418
+ if is_clean and name:
419
+ return table_start
420
+ start = super_idx + 1
421
+
422
+
423
+ def read_channels(path: str) -> List[ChannelRecord]:
424
+ """
425
+ Decode the channel symbol table. Fails gracefully: a file that
426
+ isn't a real `.gdb` (bad magic), has a truncated header, has no
427
+ locatable channel table, or has a channel table that's cut off
428
+ partway through all result in a `GDBParseWarning` plus whatever
429
+ channels *were* successfully decoded before the problem (an empty
430
+ list in the first three cases, since nothing was decodable yet; a
431
+ real, non-empty, shorter-than-`chans_max` list in the last case).
432
+ Never raises for these -- see `GDBParseWarning`'s docstring.
433
+ """
434
+ with open(path, "rb") as f:
435
+ # Reading the whole file is wasteful for a 700MB+ real survey
436
+ # database, but the symbol table's exact byte extent isn't fully
437
+ # pinned down yet (docs/provenance/notes.md 6.1), so for correctness this reads
438
+ # generously. A production version should read a memory-mapped
439
+ # view instead -- left as a TODO once the header's table-size
440
+ # field (offset 104, currently [UNKNOWN]) is confirmed.
441
+ header = f.read(4096)
442
+ if not check_magic(header):
443
+ _warn(f"{path}: does not start with the expected '!CBD' magic -- "
444
+ f"not a recognized .gdb file, returning no channels")
445
+ return []
446
+ fields = header_fields(header)
447
+ if fields["chans_max"] is None:
448
+ _warn(f"{path}: header too short to read chans_max -- returning no channels")
449
+ return []
450
+
451
+ f.seek(0, 2)
452
+ size = f.tell()
453
+ f.seek(0)
454
+ # Only need enough of the file to reach the channel + user tables.
455
+ # Observed table offsets range from ~130KB to ~580KB across 9 real
456
+ # files so far, but read generously (all of a file up to 200MB,
457
+ # else the first 20MB) since the exact extent isn't pinned down.
458
+ data = f.read(size if size <= 200_000_000 else 20_000_000)
459
+
460
+ try:
461
+ table_start = find_channel_table(data)
462
+ except ValueError as e:
463
+ _warn(f"{path}: could not locate the channel symbol table ({e}) -- "
464
+ f"returning no channels")
465
+ return []
466
+ chans_max = fields["chans_max"]
467
+
468
+ channels = []
469
+ for i in range(chans_max):
470
+ rec_start = table_start + i * SYMBOL_RECORD_SIZE
471
+ if rec_start + SYMBOL_RECORD_SIZE > len(data):
472
+ _warn(
473
+ f"{path}: channel table truncated at record {i} of {chans_max} "
474
+ f"(need bytes up to {rec_start + SYMBOL_RECORD_SIZE}, only "
475
+ f"{len(data)} were read/available) -- returning the "
476
+ f"{len(channels)} channel(s) decoded so far"
477
+ )
478
+ break
479
+ rec = _parse_channel_record(data, rec_start, i)
480
+ if not rec.name:
481
+ continue # cleanly empty/unused slot (NUL name, zeroed record)
482
+ if not rec.name_is_clean:
483
+ # Unused capacity slot with leftover/uninitialized bytes rather
484
+ # than a clean NUL name -- observed in at least one real 1990s
485
+ # file (see _read_name docstring / docs/provenance/notes.md). Not a real channel.
486
+ continue
487
+ if not rec.looks_sane:
488
+ # A NUL-terminated printable "name" can still show up by pure
489
+ # coincidence inside leftover garbage bytes in an unused slot
490
+ # (observed in DB_Mag_833.gdb: "L2161" and "1", both leftover
491
+ # fragments of an embedded projection-name blob that happened
492
+ # to land in unused channel-table capacity). Real channel
493
+ # records always have a dtype matching a known GS_* code or a
494
+ # small negative string width, and a format code in the known
495
+ # DB_CHAN_FORMAT_* range -- garbage doesn't. See docs/provenance/notes.md.
496
+ continue
497
+ channels.append(rec)
498
+ return channels
499
+
500
+
501
+ DB_CATEGORY_LINE_NAMES = {
502
+ 100: "NORMAL", # DB_CATEGORY_LINE_NORMAL / DB_CATEGORY_LINE_FLIGHT (same value)
503
+ 200: "GROUP", # DB_CATEGORY_LINE_GROUP
504
+ }
505
+
506
+ _LINE_TABLE_EMPTY_CATEGORY = 65536 # [CONFIRMED] sentinel seen on unused line-table
507
+ # capacity slots -- docs/spec.md section 3.2
508
+
509
+ _NAME_LIKE_RE = re.compile(rb"[\x20-\x7e]{1,63}\x00")
510
+
511
+
512
+ @dataclass(eq=False)
513
+ class LineRecord:
514
+ """
515
+ One 128-byte line-table record. [LIKELY]/[UNKNOWN] -- much less firmly
516
+ established than ChannelRecord: only the name (relative +32) and
517
+ category code (relative +108) fields are decoded, and locating the
518
+ table itself (find_line_table below) is a heuristic scan rather than
519
+ the structurally-proven SUPER-anchor technique used for the channel
520
+ table. See docs/spec.md section 3.2 and docs/provenance/notes.md section 6.3.
521
+
522
+ `eq=False` keeps the default identity-based `__eq__`/`__hash__`
523
+ instead of dataclass's usual field-by-field one -- needed both to
524
+ stay hashable (see `ChannelRecord`'s docstring for why) and because
525
+ `GDB._calibrate_line_indices` mutates `.index` in place on these
526
+ after construction; a value-based `__eq__`/`__hash__` pair would be
527
+ actively wrong for an object whose fields change post-construction.
528
+ """
529
+ index: int # 0-based physical slot number -- this IS line_slot_index
530
+ # in the blob_index formula (BlobHeader.line_channel)
531
+ offset: int
532
+ name: str
533
+ category_code: Optional[int]
534
+ raw: bytes
535
+ name_is_clean: bool = True
536
+
537
+ @property
538
+ def category_name(self) -> str:
539
+ if self.category_code is None:
540
+ return "unknown"
541
+ return DB_CATEGORY_LINE_NAMES.get(self.category_code, f"unknown({self.category_code})")
542
+
543
+
544
+ def _parse_line_record(data: bytes, rec_start: int, index: int) -> LineRecord:
545
+ raw = data[rec_start : rec_start + SYMBOL_RECORD_SIZE]
546
+ name, is_clean = _read_name(raw, 32)
547
+ try:
548
+ category_code = struct.unpack_from("<i", raw, 108)[0]
549
+ except struct.error:
550
+ category_code = None
551
+ return LineRecord(
552
+ index=index, offset=rec_start, name=name,
553
+ category_code=category_code, raw=raw, name_is_clean=is_clean,
554
+ )
555
+
556
+
557
+ def find_line_table(data: bytes, search_window: Tuple[int, Optional[int]] = (128, None)) -> int:
558
+ """
559
+ Locate the start of the line symbol table.
560
+
561
+ Unlike find_channel_table, there's no known default-name anchor (the
562
+ line table has nothing analogous to the channel table's "SUPER" user
563
+ record immediately after it) and no confirmed header field gives its
564
+ start offset directly -- reconciling one with the header's capacity
565
+ fields was tried and didn't cleanly round-trip (docs/provenance/log.md
566
+ Session 1 section 1.16, docs/provenance/notes.md section 6.3). This is
567
+ therefore a heuristic **[LIKELY]** scan, not the structurally-proven
568
+ technique used for the channel table: it looks for a run of 128-byte
569
+ records whose relative +32 field looks like a clean, NUL-terminated,
570
+ printable line name and whose relative +108 category field matches one
571
+ of the two confirmed real values (100=NORMAL/FLIGHT, 200=GROUP), then
572
+ returns the earliest such record in the run with the most hits at a
573
+ consistent 128-byte phase.
574
+
575
+ **Known limitation, found by real-file testing, not yet fixed here:**
576
+ if a table's true first slot(s) don't carry a category code in
577
+ {100, 200}, this returns a start that's one or more slots too late --
578
+ every subsequent LineRecord.index is then off by that same fixed
579
+ amount, which breaks blob_index lookups by line name. Observed for
580
+ real on a GSQ file (`rm001141`): physical slot 0 is a genuine, named
581
+ record (`"L0"`) with category `65636` (**[GUESS]**: `65536 + 100`,
582
+ plausibly "a NORMAL line that was since cleared," not confirmed),
583
+ which this function doesn't recognize, so it starts the table one
584
+ slot late. A generic backward-scan fix was tried and rejected: "keep
585
+ walking backward while the name field still looks clean" massively
586
+ over-extends on at least one real file (walked 30+ slots into what
587
+ turned out to be unrelated, legitimately-empty space before the real
588
+ table). `GDB` (in `gdb.py`) instead cross-validates and corrects this
589
+ against the actual blob chain, which is a strictly stronger signal
590
+ than anything available from the symbol-table bytes alone -- prefer
591
+ it over calling this function directly when correct line-indexed
592
+ data access matters, not just names.
593
+
594
+ `search_window` defaults to (128, end of `data`) -- callers should
595
+ generally narrow `hi` to `blob_region_start(data)` when known, since
596
+ every real file examined has its symbol tables (line, channel, user)
597
+ entirely before the blob region, and narrowing avoids false-positive
598
+ matches inside actual channel data.
599
+ """
600
+ lo, hi = search_window
601
+ if hi is None:
602
+ hi = len(data)
603
+
604
+ phase_hits = {}
605
+ for m in _NAME_LIKE_RE.finditer(data, lo, hi):
606
+ rec_start = m.start() - 32
607
+ if rec_start < lo or rec_start + SYMBOL_RECORD_SIZE > hi:
608
+ continue
609
+ raw = data[rec_start : rec_start + SYMBOL_RECORD_SIZE]
610
+ try:
611
+ category_code = struct.unpack_from("<i", raw, 108)[0]
612
+ except struct.error:
613
+ continue
614
+ if category_code not in DB_CATEGORY_LINE_NAMES:
615
+ continue
616
+ phase_hits.setdefault(rec_start % SYMBOL_RECORD_SIZE, []).append(rec_start)
617
+
618
+ if not phase_hits:
619
+ raise ValueError(
620
+ "no line-record-shaped data (clean name + a known category "
621
+ "code) found in the search window"
622
+ )
623
+ best_phase = max(phase_hits, key=lambda p: len(phase_hits[p]))
624
+ return min(phase_hits[best_phase])
625
+
626
+
627
+ def read_lines(path: str) -> List[LineRecord]:
628
+ """
629
+ Decode the line symbol table. Heuristic (see find_line_table) --
630
+ less firmly established than read_channels. Fails gracefully in the
631
+ same style: a bad magic, truncated header, or unlocatable line table
632
+ all return `[]` with a `GDBParseWarning` rather than raising.
633
+
634
+ Since no confirmed header field gives the line table's slot capacity
635
+ (the way chans_max does for the channel table), this reads forward
636
+ from the located start until 8 consecutive records fail to look like
637
+ either a populated line record or clean unused capacity -- a
638
+ tolerance against one-off corruption/false-positive records, not a
639
+ precisely-known table boundary.
640
+
641
+ **`LineRecord.index` can be off by a small, fixed amount** on a file
642
+ where `find_line_table`'s heuristic starts one or more slots late --
643
+ see that function's docstring. This makes `.name` still correct but
644
+ `.index` (and therefore any blob_index lookup keyed on it) wrong.
645
+ `GDB` (in `gdb.py`) corrects this against the actual blob chain
646
+ before exposing lines by name; call it instead of this function
647
+ directly when you need working (line, channel) data access, not
648
+ just a list of names.
649
+ """
650
+ with open(path, "rb") as f:
651
+ header = f.read(4096)
652
+ if not check_magic(header):
653
+ _warn(f"{path}: does not start with the expected '!CBD' magic -- "
654
+ f"not a recognized .gdb file, returning no lines")
655
+ return []
656
+ blob_start = blob_region_start(header)
657
+ f.seek(0, 2)
658
+ size = f.tell()
659
+ f.seek(0)
660
+ read_size = blob_start if (blob_start is not None and 0 < blob_start <= size) else min(size, 20_000_000)
661
+ data = f.read(read_size)
662
+
663
+ try:
664
+ table_start = find_line_table(data, search_window=(128, len(data)))
665
+ except ValueError as e:
666
+ _warn(f"{path}: could not locate the line symbol table ({e}) -- "
667
+ f"returning no lines")
668
+ return []
669
+
670
+ lines: List[LineRecord] = []
671
+ consecutive_bad = 0
672
+ i = 0
673
+ while True:
674
+ rec_start = table_start + i * SYMBOL_RECORD_SIZE
675
+ if rec_start + SYMBOL_RECORD_SIZE > len(data):
676
+ break
677
+ rec = _parse_line_record(data, rec_start, i)
678
+ i += 1
679
+ if rec.name and rec.name_is_clean and rec.category_code in DB_CATEGORY_LINE_NAMES:
680
+ lines.append(rec)
681
+ consecutive_bad = 0
682
+ elif not rec.name and rec.name_is_clean:
683
+ # Empty/unused capacity slot (matches the channel table's own
684
+ # zeroed-padding convention, or the 65536 empty-category
685
+ # sentinel confirmed in docs/spec.md section 3.2) -- keep
686
+ # scanning past it, it doesn't count as "bad".
687
+ consecutive_bad = 0
688
+ else:
689
+ consecutive_bad += 1
690
+ if consecutive_bad >= 8:
691
+ break
692
+ return lines
693
+
694
+
695
+ BLOB_MAGIC = b"\xcc\xcc\x00\xff"
696
+ BLOB_HEADER_SIZE = 48 # [CONFIRMED] -- see docs/provenance/notes.md section 6.6
697
+
698
+ # [CONFIRMED] (docs/provenance/notes.md section 6.6b): for COMPRESSED blobs specifically
699
+ # (DB_COMP_SPEED or DB_COMP_SIZE), the blob header is 56 bytes, not 48 --
700
+ # the extra 8 bytes hold a preview of the first chunk's decompressed
701
+ # length and total on-disk span (not fully decoded, see docs/provenance/notes.md section
702
+ # 6.5e). The already-known 16-byte page-primitive chunk magic
703
+ # (lzrw1.CHUNK_MAGIC) sits immediately after these 56 bytes, verified
704
+ # directly against real ground truth on AG106386 (DB_COMP_SIZE): the
705
+ # zlib payload for blob_index=0 (GA_project_number) is found at exactly
706
+ # blob.offset + COMPRESSED_BLOB_HEADER_SIZE + 16 and decompresses to the
707
+ # known real constant value 5027.
708
+ COMPRESSED_BLOB_HEADER_SIZE = 56
709
+
710
+
711
+ @dataclass
712
+ class BlobHeader:
713
+ """
714
+ The per-channel-per-line data block header. [CONFIRMED] for fields
715
+ up to and including `blob_index` (verified byte-exact on 5 real
716
+ DB_COMP_NONE files via a whole-file, zero-error chain walk that
717
+ lands exactly on each file's true size -- see docs/provenance/notes.md section 6.6).
718
+ Fields from `timestamp` onward are only [LIKELY]/[UNKNOWN] and are
719
+ known NOT to decode sensibly at these byte offsets in at least one
720
+ real older (1991 GSQ) file -- kept here for the modern (2020 USGS)
721
+ case where they were verified, not assumed general.
722
+ """
723
+ offset: int # absolute file offset of this header's first byte
724
+ n_pages: int # [CONFIRMED] -- this blob's total on-disk size, in
725
+ # pages (page_size from header_fields())
726
+ n_pages_dup: int # [LIKELY] -- always seen equal to n_pages
727
+ blob_index: int # [CONFIRMED] -- see line_slot/channel_slot below
728
+ timestamp: int # [LIKELY] modern files only, see docstring above
729
+ reserved_200: int # [UNKNOWN]
730
+ scale: float # [LIKELY] modern files only
731
+ row_count: int # [CONFIRMED] modern files only (verified against
732
+ # real ground-truth-matching decoded values)
733
+ gs_type_code: int # [CONFIRMED] modern files only (matches owning
734
+ # channel's own symbol-table dtype exactly)
735
+
736
+ def line_channel(self, chans_max: int):
737
+ """
738
+ Decompose blob_index into (line_slot_index, channel_slot_index)
739
+ via the formula [CONFIRMED] in docs/provenance/notes.md section 6.6:
740
+ blob_index == line_slot_index * chans_max + channel_slot_index
741
+ Both are 0-based physical slot numbers in their respective
742
+ symbol tables (same indexing as ChannelRecord.index and the
743
+ line table walked ad hoc in docs/provenance/notes.md section 6.3).
744
+ """
745
+ return divmod(self.blob_index, chans_max)
746
+
747
+ @property
748
+ def data_offset(self) -> int:
749
+ return self.offset + BLOB_HEADER_SIZE
750
+
751
+
752
+ def _parse_blob_header(raw: bytes, offset: int) -> Optional[BlobHeader]:
753
+ if len(raw) < BLOB_HEADER_SIZE or raw[:4] != BLOB_MAGIC:
754
+ return None
755
+ n_pages = struct.unpack_from("<i", raw, 4)[0]
756
+ n_pages_dup = struct.unpack_from("<i", raw, 8)[0]
757
+ blob_index = struct.unpack_from("<i", raw, 12)[0]
758
+ timestamp = struct.unpack_from("<i", raw, 16)[0]
759
+ reserved_200 = struct.unpack_from("<i", raw, 20)[0]
760
+ scale = struct.unpack_from("<d", raw, 32)[0]
761
+ row_count = struct.unpack_from("<i", raw, 40)[0]
762
+ gs_type_code = struct.unpack_from("<i", raw, 44)[0]
763
+ return BlobHeader(
764
+ offset=offset, n_pages=n_pages, n_pages_dup=n_pages_dup,
765
+ blob_index=blob_index, timestamp=timestamp,
766
+ reserved_200=reserved_200, scale=scale, row_count=row_count,
767
+ gs_type_code=gs_type_code,
768
+ )
769
+
770
+
771
+ def blob_region_start(data: bytes) -> Optional[int]:
772
+ """
773
+ Absolute byte offset of the first real blob header.
774
+
775
+ [CONFIRMED] on 20+ real files (every compression mode, chans_max
776
+ 20-500, ~1991-2020, all 3 agencies) -- see docs/provenance/notes.md section 6.6/
777
+ 6.6b. Header offset 108 (int32) is a PAGE NUMBER; multiplying by
778
+ page_size (header offset 100) lands exactly on the CC CC 00 FF
779
+ magic every time. (Header offset 104, an earlier "live lead" for
780
+ this same purpose in this project's own notes, is a close-but-wrong
781
+ red herring -- it sits near, but not exactly on, the end of the
782
+ symbol tables, and isn't even page-aligned.)
783
+
784
+ Returns `None` (with a `GDBParseWarning`) if `data` is too short to
785
+ even read the two fields this needs (offset 108 + 4 bytes) -- a
786
+ severely truncated header.
787
+ """
788
+ try:
789
+ page_size = struct.unpack_from("<i", data, 100)[0]
790
+ start_page = struct.unpack_from("<i", data, 108)[0]
791
+ except struct.error:
792
+ _warn(
793
+ f"header truncated: only {len(data)} byte(s) available, not enough "
794
+ f"to locate the blob region (need offset 108 + 4 bytes)"
795
+ )
796
+ return None
797
+ return start_page * page_size
798
+
799
+
800
+ def iter_blobs(path: str, max_blobs: Optional[int] = None):
801
+ """
802
+ Walk the self-describing blob chain from the start of the blob
803
+ region to end of file (or `max_blobs`, or the first framing
804
+ anomaly), yielding BlobHeader records in on-disk order.
805
+
806
+ [CONFIRMED] end-to-end (zero framing errors, landing exactly on the
807
+ true file size) on 20 real files spanning all 3 agencies this
808
+ project has files from and all three `DB_COMP_*` compression modes,
809
+ 2MB to 1.93GB -- see docs/provenance/notes.md section 6.6b/6.6d/6.9.
810
+
811
+ As a generator, this already "returns partial results" in the most
812
+ natural way possible: whatever's been yielded before a problem is
813
+ hit stays with the caller (a `for blob in iter_blobs(path): ...`
814
+ loop simply ends, keeping everything already processed) -- nothing
815
+ is lost by stopping early. What this function adds on top of that
816
+ is a clear `GDBParseWarning` distinguishing *why* it stopped:
817
+ reaching the file's true end cleanly is silent (the expected,
818
+ common case), but a magic mismatch, a non-positive `n_pages`, a
819
+ file that ends mid-header, or landing short of true EOF by less
820
+ than one full header (i.e. real leftover bytes, not enough to be
821
+ read at all) are all real anomalies and each gets its own specific
822
+ warning identifying the offset and how many blobs were walked
823
+ first -- so a caller can tell "the chain looked completely normal
824
+ and just ended" from "something didn't fit the confirmed
825
+ structure" without having to guess from the return value alone.
826
+ Never raises for a bad/truncated file; only for a real precondition
827
+ problem (can't even open `path`, propagated normally from `open`).
828
+ """
829
+ with open(path, "rb") as f:
830
+ header = f.read(128)
831
+ if not check_magic(header):
832
+ _warn(f"{path}: does not start with the expected '!CBD' magic -- "
833
+ f"no blobs to walk")
834
+ return
835
+ off = blob_region_start(header)
836
+ if off is None:
837
+ _warn(f"{path}: could not determine the blob region start -- "
838
+ f"no blobs to walk")
839
+ return
840
+ page_size = struct.unpack_from("<i", header, 100)[0]
841
+ f.seek(0, 2)
842
+ size = f.tell()
843
+ if off > size:
844
+ _warn(
845
+ f"{path}: computed blob region start ({off}) is past the end "
846
+ f"of the file ({size} byte(s)) -- file is likely severely "
847
+ f"truncated; no blobs to walk"
848
+ )
849
+ return
850
+ f.seek(off)
851
+ n = 0
852
+ while off + BLOB_HEADER_SIZE <= size:
853
+ if max_blobs is not None and n >= max_blobs:
854
+ return
855
+ raw = f.read(BLOB_HEADER_SIZE)
856
+ blob = _parse_blob_header(raw, off)
857
+ if blob is None:
858
+ if len(raw) < BLOB_HEADER_SIZE:
859
+ _warn(
860
+ f"{path}: blob chain ends mid-header at offset {off} "
861
+ f"(only {len(raw)} of {BLOB_HEADER_SIZE} expected "
862
+ f"byte(s) available) after {n} blob(s) successfully "
863
+ f"walked -- file is likely truncated; returning the "
864
+ f"{n} blob(s) already yielded"
865
+ )
866
+ else:
867
+ _warn(
868
+ f"{path}: blob magic mismatch at offset {off} "
869
+ f"(got {raw[:4].hex()}, expected {BLOB_MAGIC.hex()}) "
870
+ f"after {n} blob(s) successfully walked -- stopping "
871
+ f"the chain walk here and returning the {n} blob(s) "
872
+ f"already yielded; this may be a real structural "
873
+ f"anomaly or an administrative-blob variant not yet "
874
+ f"understood (docs/provenance/notes.md section 6.4/6.9)"
875
+ )
876
+ return
877
+ if blob.n_pages <= 0:
878
+ _warn(
879
+ f"{path}: blob at offset {off} (blob_index={blob.blob_index}) "
880
+ f"has a non-positive n_pages ({blob.n_pages}) after {n} "
881
+ f"blob(s) successfully walked -- cannot safely continue "
882
+ f"(don't know how far to skip to find the next header); "
883
+ f"returning the {n} blob(s) already yielded"
884
+ )
885
+ return
886
+ # NOTE: n_pages_dup (relative +8) is NOT always equal to n_pages
887
+ # (relative +4) -- confirmed on real Ontario GDS1251 files
888
+ # (MLGRAV.gdb/MLMAG.gdb), where a small number of "reserved/
889
+ # administrative" blobs (same class flagged [UNKNOWN] elsewhere
890
+ # in this section -- out-of-range line index, gs_type_code
891
+ # reading the same 4670802 constant) have n_pages_dup != n_pages.
892
+ # Directly verified: n_pages (not n_pages_dup) is the field that
893
+ # correctly lands on the next real blob header every time -- an
894
+ # earlier version of this function required the two to match and
895
+ # broke immediately on these files as a result. Trust n_pages
896
+ # alone; n_pages_dup is kept on BlobHeader for whoever wants to
897
+ # investigate what it actually means.
898
+ yield blob
899
+ skip = blob.n_pages * page_size - BLOB_HEADER_SIZE
900
+ f.seek(skip, 1)
901
+ off += blob.n_pages * page_size
902
+ n += 1
903
+ if n > 0 and off > size:
904
+ # The last blob successfully parsed claimed a page count that
905
+ # implies more data than the file actually contains -- off
906
+ # jumped past true EOF. A real, distinct anomaly: the file is
907
+ # cut off in the middle of what should have been that blob's
908
+ # data (or its padding).
909
+ _warn(
910
+ f"{path}: after {n} blob(s), the last one (offset "
911
+ f"{off - blob.n_pages * page_size}, blob_index={blob.blob_index}, "
912
+ f"n_pages={blob.n_pages}) claims data extending "
913
+ f"{off - size} byte(s) past the true end of file ({size} "
914
+ f"byte(s) total) -- file is truncated mid-blob; returning "
915
+ f"the {n} blob header(s) already yielded (note: that last "
916
+ f"blob's own data may itself be incomplete -- see "
917
+ f"read_blob_values()'s truncation handling)"
918
+ )
919
+ elif n > 0 and off != size:
920
+ # Loop condition failed (off + 48 > size) but we're not exactly
921
+ # at the true end either -- real leftover bytes, less than one
922
+ # full header's worth. Every real file checked in this project
923
+ # (docs/provenance/notes.md section 6.6b/6.9) ends with an EXACT match, so any
924
+ # slack here is new/unusual and worth flagging, not silently
925
+ # accepted.
926
+ _warn(
927
+ f"{path}: blob chain walk stopped {size - off} byte(s) short "
928
+ f"of the true end of file (at offset {off} of {size}) after "
929
+ f"{n} blob(s) -- less than one full header remains there, "
930
+ f"which doesn't match any real file checked in this project "
931
+ f"so far (they all end with an exact match); file may be "
932
+ f"truncated"
933
+ )
934
+
935
+
936
+ def find_blob(path: str, line_slot: int, channel_slot: int, chans_max: Optional[int] = None) -> Optional[BlobHeader]:
937
+ """
938
+ Locate the blob for a specific (line, channel) pair by walking the
939
+ chain (see iter_blobs) and computing the target blob_index via the
940
+ formula [CONFIRMED] in docs/provenance/notes.md section 6.6. Returns `None` if the
941
+ chain ends (or breaks -- see `iter_blobs`'s `GDBParseWarning`s for
942
+ why) before the target is found, or if `chans_max` can't be
943
+ determined at all (bad magic / truncated header) -- never raises
944
+ for these, consistent with the rest of this module.
945
+
946
+ This does a linear walk from the start of the blob region every
947
+ call -- fine for occasional lookups or for building a full
948
+ line/channel -> offset index once (walk the whole chain yourself
949
+ with iter_blobs() and record every blob.offset keyed by
950
+ blob.line_channel(chans_max) if you need many lookups).
951
+ """
952
+ if chans_max is None:
953
+ with open(path, "rb") as f:
954
+ header = f.read(128)
955
+ if not check_magic(header):
956
+ _warn(f"{path}: does not start with the expected '!CBD' magic -- "
957
+ f"cannot determine chans_max, blob not found")
958
+ return None
959
+ try:
960
+ chans_max = struct.unpack_from("<i", header, 24)[0]
961
+ except struct.error:
962
+ _warn(f"{path}: header too short to read chans_max -- blob not found")
963
+ return None
964
+ target = line_slot * chans_max + channel_slot
965
+ for blob in iter_blobs(path):
966
+ if blob.blob_index == target:
967
+ return blob
968
+ return None
969
+
970
+
971
+ def _element_width(channel: ChannelRecord) -> Optional[int]:
972
+ """
973
+ Byte width of one element of `channel`'s data, or `None` if it's a
974
+ type this reader doesn't know how to decode -- e.g. one of the
975
+ multi-dimensional `GS_FLOAT3D`/`GS_DOUBLE3D`/`GS_FLOAT2D`/
976
+ `GS_DOUBLE2D` types, none of which have been seen in any real
977
+ sample yet (docs/provenance/notes.md section 4). Callers should check for `None`
978
+ and warn/return gracefully rather than assume a format exists.
979
+ """
980
+ if channel.is_string:
981
+ return channel.string_width
982
+ fmt = GS_TYPE_STRUCT.get(channel.dtype_code)
983
+ return struct.calcsize(fmt) if fmt is not None else None
984
+
985
+
986
+ def _decode_numeric_or_string(raw: bytes, channel: ChannelRecord, row_count: Optional[int] = None):
987
+ """
988
+ Interpret a raw byte buffer as `row_count` (or however many fit)
989
+ values of `channel`'s known type. Shared by the uncompressed and
990
+ compressed decode paths.
991
+
992
+ Always returns a numpy `ndarray`: 1-D `(n_rows,)` for an ordinary
993
+ scalar channel, or 2-D `(n_rows, channel.array_width)` for a VA/
994
+ array channel (docs/spec.md section 5 -- e.g. a 512-wide airborne
995
+ gamma-ray spectrum recorded per station; `array_width` is fixed per
996
+ channel, never seen to vary row-to-row, so this reshape is always a
997
+ clean rectangle). Numeric channels get the dtype matching their
998
+ `GS_*` type (`GS_TYPE_NUMPY_DTYPE`); string channels (including the
999
+ unconfirmed-but-handled case of a *string* array channel) get
1000
+ `dtype=object` holding plain Python `str`, since numpy has no
1001
+ variable-content fixed-dtype string type that round-trips this
1002
+ format's null-padded, variable-actual-length names cleanly.
1003
+
1004
+ String-typed channels dispatch to the compiled `pygdb._native`
1005
+ extension when it's available (same decode, ported to Rust -- see
1006
+ `rust/src/lib.rs`'s `decode_fixed_width_strings`; profiling found
1007
+ this the second real CPU-bound hot path in this reader besides
1008
+ LZRW1, unlike numeric decode which stays near memory-bandwidth speed
1009
+ via `np.frombuffer` either way), falling back to the pure-Python list
1010
+ comprehension below when it isn't.
1011
+
1012
+ Fails gracefully rather than raising: an unrecognized element type
1013
+ returns an empty array with a `GDBParseWarning`; a `raw` buffer
1014
+ shorter than needed for the requested `row_count` (the file was
1015
+ truncated mid-blob, a real scenario for a cut-off download) decodes
1016
+ as many *complete* elements as actually fit and warns about the
1017
+ shortfall, rather than raising a `struct.error` and discarding
1018
+ everything; for an array channel, a flat element count that isn't a
1019
+ whole multiple of `array_width` similarly warns and drops the
1020
+ trailing incomplete row rather than raising.
1021
+ """
1022
+ width = _element_width(channel)
1023
+ if width is None:
1024
+ _warn(
1025
+ f"channel {channel.name!r} has dtype_code={channel.dtype_code}, a type "
1026
+ f"this reader doesn't know how to decode (likely a multi-dimensional "
1027
+ f"GS_FLOAT3D/GS_DOUBLE3D/etc type, never seen in a real sample) -- "
1028
+ f"returning no values for it"
1029
+ )
1030
+ return np.array([])
1031
+ n_available = len(raw) // width
1032
+ n = row_count if row_count is not None else n_available
1033
+ if n > n_available:
1034
+ _warn(
1035
+ f"channel {channel.name!r}: expected {n} row(s) but only enough raw "
1036
+ f"bytes for {n_available} complete element(s) (got {len(raw)} byte(s), "
1037
+ f"need {n * width}) -- data is truncated (file cut off mid-blob?); "
1038
+ f"returning the {n_available} row(s) that could be decoded"
1039
+ )
1040
+ n = n_available
1041
+ if channel.is_array:
1042
+ # `n` here is the FLAT element count (row_count == n_rows *
1043
+ # array_width for an array channel, confirmed docs/spec.md
1044
+ # section 5); truncate to the largest whole number of complete
1045
+ # rows before reshaping if it doesn't divide evenly.
1046
+ usable_rows, remainder = divmod(n, channel.array_width)
1047
+ if remainder:
1048
+ _warn(
1049
+ f"channel {channel.name!r}: {n} flat element(s) isn't a whole "
1050
+ f"multiple of array_width={channel.array_width} -- dropping the "
1051
+ f"trailing {remainder} incomplete element(s) rather than "
1052
+ f"returning a raggedly-shaped result"
1053
+ )
1054
+ n = usable_rows * channel.array_width
1055
+ if channel.is_string:
1056
+ if _native_ext is not None:
1057
+ values = _native_ext.decode_fixed_width_strings(raw, width, n)
1058
+ else:
1059
+ values = [
1060
+ raw[i * width : (i + 1) * width].split(b"\x00")[0].decode("ascii", errors="replace")
1061
+ for i in range(n)
1062
+ ]
1063
+ arr = np.array(values, dtype=object)
1064
+ else:
1065
+ dtype = GS_TYPE_NUMPY_DTYPE[channel.dtype_code]
1066
+ arr = np.frombuffer(raw[: n * width], dtype=dtype).copy()
1067
+ if channel.is_array:
1068
+ arr = arr.reshape(-1, channel.array_width)
1069
+ return arr
1070
+
1071
+
1072
+ @contextlib.contextmanager
1073
+ def _file_handle(path: str, file: Optional[BinaryIO]):
1074
+ """
1075
+ Yield `file` directly if given (an already-open handle a caller
1076
+ wants reused across many calls), otherwise open `path` fresh and
1077
+ close it on exit -- lets `read_blob_values` support both "just give
1078
+ me a path" (the default, used everywhere else in this module) and
1079
+ "reuse this open handle" (what `GDB` does, to avoid reopening the
1080
+ file on every single read -- benchmarked at ~1.7-1.9x slower per
1081
+ call otherwise, see the project's Rust-plan notes) with the same
1082
+ `with _file_handle(path, file) as f:` call sites either way.
1083
+ """
1084
+ if file is not None:
1085
+ yield file
1086
+ else:
1087
+ with open(path, "rb") as f:
1088
+ yield f
1089
+
1090
+
1091
+ def read_blob_values(path: str, blob: BlobHeader, channel: ChannelRecord,
1092
+ comp_level: int = 0, page_size: Optional[int] = None,
1093
+ file: Optional[BinaryIO] = None):
1094
+ """
1095
+ Decode a found blob's real row data using the owning channel's
1096
+ already-known type (from the symbol table, docs/provenance/notes.md section 6.2).
1097
+
1098
+ [CONFIRMED] against real ground truth for GS_DOUBLE data and for
1099
+ fixed-width strings, for DB_COMP_NONE (docs/provenance/notes.md section 6.6: a real
1100
+ `fid` blob decoded this way reproduces the exact CSV ground-truth
1101
+ value, and a real `line`-channel blob decodes to the correct real
1102
+ line name repeated once per row).
1103
+
1104
+ Also handles compressed blobs (`comp_level` 1=DB_COMP_SPEED or
1105
+ 2=DB_COMP_SIZE), **including multi-page ones** -- [CONFIRMED]
1106
+ against real ground truth for both single- and multi-page
1107
+ DB_COMP_SIZE (a real single-page blob_index=0 decodes to the known
1108
+ constant 5027; a real 36-page array-channel blob decodes to
1109
+ `LEI_Depth`'s exact known real depth profile, `0.0, 3.0, 6.3, 9.9,
1110
+ ...`, repeated once per station -- both matching docs/provenance/notes.md section
1111
+ 6.5/6.2b's independently-established ground truth exactly) and for
1112
+ both single- and multi-page DB_COMP_SPEED (a real 2-page
1113
+ `Northing_AMGz55` blob decodes to sane real coordinates with real
1114
+ `rDUMMY` sentinels). See docs/provenance/notes.md section 6.6d: a multi-page blob is
1115
+ simply one continuous compressed stream spanning the whole
1116
+ `n_pages*page_size` span, not one independently-framed chunk per
1117
+ page -- no special multi-page logic was actually needed once this
1118
+ was verified, just reading the full span instead of one page.
1119
+
1120
+ **A real third on-disk variant, auto-detected here rather than
1121
+ assumed away (docs/provenance/notes.md section 6.6b):** even inside a file that
1122
+ genuinely declares (and elsewhere uses) DB_COMP_SPEED, some
1123
+ individual blobs turn out to carry no chunk wrapper at all -- just
1124
+ the plain 48-byte DB_COMP_NONE-style header with real, directly
1125
+ readable data straight after it (confirmed on a real
1126
+ `Easting_AMGz55` blob in `DB_EM_293.gdb`: decoding it as if
1127
+ `comp_level==0` reproduces sane, real coordinate values with real
1128
+ `rDUMMY=-1.0E32` sentinels in the expected places). When
1129
+ `comp_level != 0`, this function checks for the 16-byte chunk magic
1130
+ at the 56-byte-header position first and only falls back to the
1131
+ genuinely-compressed path if it's actually there -- otherwise it
1132
+ decodes the blob exactly like a DB_COMP_NONE one.
1133
+
1134
+ **Fails gracefully, per an explicit engineering request:** a
1135
+ negative `row_count` (a reserved/administrative blob, docs/provenance/notes.md
1136
+ section 6.4/6.9, not real data), a channel type this reader can't
1137
+ decode, a truncated read (file cut off mid-blob), an unrecognized
1138
+ chunk subtype, or a chunk that fails to decompress (corrupt/
1139
+ truncated compressed data, or `lzrw1.LZRW1DecodeError`) all return
1140
+ an empty `ndarray` (`np.array([])`) with a `GDBParseWarning`
1141
+ describing what went wrong, instead of raising and losing the
1142
+ caller's place in a larger loop (e.g. a whole-file scan that's
1143
+ decoded hundreds of blobs already). The one exception where full
1144
+ graceful salvage wasn't attempted is a truncated/corrupt
1145
+ *compressed* stream: unlike the plain-data case, there's no simple
1146
+ way to hand back "the first K decoded values" from a partially-
1147
+ decompressed zlib/LZRW1 stream, so those cases warn and return an
1148
+ empty array rather than a partial decode -- documented here rather
1149
+ than silently implied to be as complete as the plain-data
1150
+ truncation handling.
1151
+
1152
+ Return shape/dtype: see `_decode_numeric_or_string`'s docstring --
1153
+ 1-D `ndarray` for a scalar channel, 2-D `(n_rows, array_width)` for
1154
+ a VA/array channel, dtype matching the channel's `GS_*` type or
1155
+ `object` (holding `str`) for a string channel.
1156
+
1157
+ `file`: an already-open binary file handle for `path`, reused
1158
+ instead of opening `path` fresh -- pass this if you're calling this
1159
+ function many times for the same file (e.g. `GDB` does, internally).
1160
+ Reopening `path` on every call is real, measured overhead (~1.7-1.9x
1161
+ slower per call, benchmarked against this project's real sample
1162
+ corpus -- see the Rust-plan's M4 notes); `file=None` (the default)
1163
+ keeps this function's plain "just give me a path" behavior for every
1164
+ other caller.
1165
+ """
1166
+ if comp_level == 0:
1167
+ if blob.row_count < 0:
1168
+ _warn(
1169
+ f"blob_index={blob.blob_index}: negative row_count "
1170
+ f"({blob.row_count}) -- this is one of the reserved/"
1171
+ f"administrative blobs flagged [UNKNOWN] in docs/provenance/notes.md "
1172
+ f"section 6.4/6.9, not a real data blob; returning no values"
1173
+ )
1174
+ return np.array([])
1175
+ width = _element_width(channel)
1176
+ if width is None:
1177
+ _warn(
1178
+ f"channel {channel.name!r} has dtype_code={channel.dtype_code}, a "
1179
+ f"type this reader doesn't know how to decode -- returning no values"
1180
+ )
1181
+ return np.array([])
1182
+ with _file_handle(path, file) as f:
1183
+ f.seek(blob.data_offset)
1184
+ raw = f.read(blob.row_count * width)
1185
+ return _decode_numeric_or_string(raw, channel, blob.row_count)
1186
+
1187
+ # comp_level != 0: could still be any of three real on-disk variants
1188
+ # (docs/provenance/notes.md section 6.6b) -- check which one this specific blob
1189
+ # actually is rather than assuming from the file-level comp_level.
1190
+ with _file_handle(path, file) as f:
1191
+ f.seek(blob.offset + COMPRESSED_BLOB_HEADER_SIZE)
1192
+ chunk_magic_probe = f.read(8)
1193
+ if len(chunk_magic_probe) < 8:
1194
+ _warn(
1195
+ f"blob_index={blob.blob_index}: file ends before the compressed-blob "
1196
+ f"header/chunk-magic region (offset {blob.offset + COMPRESSED_BLOB_HEADER_SIZE}) "
1197
+ f"could be fully read -- truncated mid-blob; returning no values"
1198
+ )
1199
+ return np.array([])
1200
+ if chunk_magic_probe != _lzrw1.CHUNK_MAGIC:
1201
+ # Variant 3: no chunk wrapper at all -- a "bare" blob, byte-for-byte
1202
+ # identical in layout to a DB_COMP_NONE one, just living inside an
1203
+ # otherwise-compressed file. Use the plain 48-byte-header fields,
1204
+ # which decoded sanely for real in this exact case.
1205
+ if blob.row_count < 0:
1206
+ _warn(
1207
+ f"blob_index={blob.blob_index}: negative row_count "
1208
+ f"({blob.row_count}) -- this is one of the reserved/"
1209
+ f"administrative blobs flagged [UNKNOWN] in docs/provenance/notes.md "
1210
+ f"section 6.4/6.9, not a real data blob; returning no values"
1211
+ )
1212
+ return np.array([])
1213
+ width = _element_width(channel)
1214
+ if width is None:
1215
+ _warn(
1216
+ f"channel {channel.name!r} has dtype_code={channel.dtype_code}, a "
1217
+ f"type this reader doesn't know how to decode -- returning no values"
1218
+ )
1219
+ return np.array([])
1220
+ with _file_handle(path, file) as f:
1221
+ f.seek(blob.data_offset)
1222
+ raw = f.read(blob.row_count * width)
1223
+ return _decode_numeric_or_string(raw, channel, blob.row_count)
1224
+
1225
+ # Compressed (DB_COMP_SPEED / DB_COMP_SIZE): 56-byte blob header,
1226
+ # then the shared 16-byte page-primitive chunk magic -- see
1227
+ # COMPRESSED_BLOB_HEADER_SIZE and docs/provenance/notes.md section 6.6b/6.6d.
1228
+ #
1229
+ # Multi-page blobs (blob.n_pages > 1) are [CONFIRMED] (section 6.6d)
1230
+ # to be a SINGLE continuous compressed stream spanning the whole
1231
+ # n_pages*page_size span -- NOT one independently-framed chunk per
1232
+ # page. There is no per-page re-framing to handle: reading the full
1233
+ # span and decompressing it as one stream (zlib.decompressobj()
1234
+ # naturally stops at the real end of stream and reports the rest as
1235
+ # padding; the LZRW1 chunk header's own decompressed_length/
1236
+ # chunk_length fields already span the full compressed length
1237
+ # regardless of how many pages it spilled into) is sufficient.
1238
+ if page_size is None:
1239
+ with _file_handle(path, file) as f:
1240
+ header = f.read(128)
1241
+ try:
1242
+ page_size = struct.unpack_from("<i", header, 100)[0]
1243
+ except struct.error:
1244
+ _warn(
1245
+ f"blob_index={blob.blob_index}: header too short to read "
1246
+ f"page_size -- cannot decode, returning no values"
1247
+ )
1248
+ return np.array([])
1249
+ with _file_handle(path, file) as f:
1250
+ f.seek(blob.offset + COMPRESSED_BLOB_HEADER_SIZE)
1251
+ expected_span = blob.n_pages * page_size - COMPRESSED_BLOB_HEADER_SIZE
1252
+ raw_span = f.read(expected_span)
1253
+ if len(raw_span) < expected_span:
1254
+ _warn(
1255
+ f"blob_index={blob.blob_index}: expected {expected_span} byte(s) of "
1256
+ f"compressed payload but the file only had {len(raw_span)} available "
1257
+ f"-- truncated mid-blob; attempting to decode what's there, but this "
1258
+ f"may fail or be incomplete"
1259
+ )
1260
+ if len(raw_span) < 16:
1261
+ _warn(
1262
+ f"blob_index={blob.blob_index}: not enough bytes to read even the "
1263
+ f"chunk sub-header ({len(raw_span)} available, need 16) -- cannot "
1264
+ f"decode, returning no values"
1265
+ )
1266
+ return np.array([])
1267
+ subtype = struct.unpack_from("<i", raw_span, 8)[0]
1268
+ if subtype == _lzrw1.DB_COMP_SIZE:
1269
+ try:
1270
+ d = zlib.decompressobj()
1271
+ decompressed = d.decompress(raw_span[16:])
1272
+ except zlib.error as e:
1273
+ _warn(
1274
+ f"blob_index={blob.blob_index}: zlib decompression failed ({e}) "
1275
+ f"-- likely truncated or corrupt compressed data; returning no values"
1276
+ )
1277
+ return np.array([])
1278
+ elif subtype == _lzrw1.DB_COMP_SPEED:
1279
+ try:
1280
+ chunk = _lzrw1.parse_chunk_header(raw_span, 0)
1281
+ decompressed = _lzrw1.decode_speed_chunk(raw_span, chunk)
1282
+ except _lzrw1.LZRW1DecodeError as e:
1283
+ _warn(
1284
+ f"blob_index={blob.blob_index}: LZRW1 chunk decode failed ({e}) "
1285
+ f"-- likely truncated or corrupt compressed data, or an "
1286
+ f"unrecognized chunk variant; returning no values"
1287
+ )
1288
+ return np.array([])
1289
+ else:
1290
+ _warn(
1291
+ f"blob_index={blob.blob_index}: unrecognized chunk subtype={subtype} "
1292
+ f"(expected {_lzrw1.DB_COMP_SIZE}=Size or {_lzrw1.DB_COMP_SPEED}=Speed) "
1293
+ f"-- returning no values"
1294
+ )
1295
+ return np.array([])
1296
+ return _decode_numeric_or_string(decompressed, channel)
1297
+
1298
+
1299
+ if __name__ == "__main__":
1300
+ import sys
1301
+
1302
+ if len(sys.argv) != 2:
1303
+ print("usage: python -m pygdb.gdb_reader <path-to.gdb>")
1304
+ raise SystemExit(1)
1305
+
1306
+ path = sys.argv[1]
1307
+ with open(path, "rb") as f:
1308
+ header = f.read(128)
1309
+ if not check_magic(header):
1310
+ print("WARNING: file does not start with the expected '!CBD' magic")
1311
+ elif not magic_signature_matches_common_case(header):
1312
+ print(
1313
+ "NOTE: '!CBD' magic OK, but bytes 8-15 differ from the common "
1314
+ f"case (got {header[4:16].hex()}) -- see docs/provenance/notes.md, seen once before"
1315
+ )
1316
+ fields = header_fields(header)
1317
+ print(f"header fields: {fields}")
1318
+
1319
+ channels = read_channels(path)
1320
+ print(f"\n{len(channels)} channel(s) found:\n")
1321
+ print(f"{'#':>3} {'name':30s} {'type':16s} {'format':8s} {'width':>5s} {'basetype':13s}")
1322
+ for c in channels:
1323
+ width = f"{c.array_width}*" if c.is_array else str(c.array_width)
1324
+ print(
1325
+ f"{c.index:3d} {c.name:30s} {c.type_name:16s} {c.format_name:8s} "
1326
+ f"{width:>5s} {c.array_basetype_name:13s}"
1327
+ )
1328
+ n_array = sum(1 for c in channels if c.is_array)
1329
+ if n_array:
1330
+ print(f"\n({n_array} of {len(channels)} channels are VA/array channels, marked with '*' in width)")
1331
+
1332
+ by_slot = {c.index: c for c in channels}
1333
+ comp_level = fields["comp_level"]
1334
+ print(
1335
+ f"\nWalking the blob chain (comp_level={comp_level}) looking for "
1336
+ "the first blob that decodes to real data (the first few are "
1337
+ "often reserved/administrative ones, see docs/provenance/notes.md section 6.6)..."
1338
+ )
1339
+ shown = 0
1340
+ for blob in iter_blobs(path, max_blobs=50):
1341
+ if blob.row_count is not None and blob.row_count <= 0 and comp_level == 0:
1342
+ continue # cheap skip for the common admin-blob case, comp_level==0 only
1343
+ line_slot, chan_slot = blob.line_channel(fields["chans_max"])
1344
+ chan = by_slot.get(chan_slot)
1345
+ if chan is None:
1346
+ continue
1347
+ # read_blob_values() never raises for a blob/channel it can't decode --
1348
+ # it warns (GDBParseWarning) and returns an empty array instead, so a
1349
+ # plain empty-result check is all that's needed here; no try/except
1350
+ # required.
1351
+ values = read_blob_values(path, blob, chan, comp_level=comp_level,
1352
+ page_size=fields["page_size"])
1353
+ if len(values) == 0:
1354
+ continue
1355
+ print(
1356
+ f" line_slot={line_slot} channel_slot={chan_slot} ({chan.name}), "
1357
+ f"n_pages={blob.n_pages}, offset={blob.offset}, first 3 values: {values[:3]}"
1358
+ )
1359
+ shown += 1
1360
+ if shown >= 3:
1361
+ break
1362
+ if shown == 0:
1363
+ print(" (no easily-decodable blob found in the first 50 of the chain)")
1364
+ print(
1365
+ "\nUse iter_blobs()/find_blob()/read_blob_values() to read real "
1366
+ "channel data for any (line, channel) pair, single- or multi-page, "
1367
+ "any compression mode -- see docs/provenance/notes.md section 6.6/6.6b/6.6d."
1368
+ )