python-gdb 0.1.0__cp314-cp314t-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pygdb/__init__.py +60 -0
- pygdb/_native.cp314t-win_amd64.pyd +0 -0
- pygdb/gdb.py +641 -0
- pygdb/gdb_reader.py +1368 -0
- pygdb/grd_reader.py +292 -0
- pygdb/lzrw1.py +323 -0
- pygdb/registry.py +104 -0
- python_gdb-0.1.0.dist-info/METADATA +183 -0
- python_gdb-0.1.0.dist-info/RECORD +12 -0
- python_gdb-0.1.0.dist-info/WHEEL +4 -0
- python_gdb-0.1.0.dist-info/licenses/LICENSE +21 -0
- python_gdb-0.1.0.dist-info/sboms/pygdb-native.cyclonedx.json +572 -0
pygdb/gdb.py
ADDED
|
@@ -0,0 +1,641 @@
|
|
|
1
|
+
"""
|
|
2
|
+
A user-facing, high-level wrapper around a single `.gdb` file.
|
|
3
|
+
|
|
4
|
+
Everything here is built on top of the lower-level primitives in
|
|
5
|
+
`gdb_reader`/`registry` (header parsing, symbol tables, the blob chain,
|
|
6
|
+
value decoding, coordinate-system extraction) -- `GDB` just gives them a
|
|
7
|
+
single, name-based entry point: list the lines and channels, see which
|
|
8
|
+
channels actually have data on a given line (the format's sparse (line,
|
|
9
|
+
channel) grid, docs/spec.md section 1/6.1), read a specific (line,
|
|
10
|
+
channel) pair's values by name, and describe the file's compression mode
|
|
11
|
+
and coordinate reference system(s).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import warnings
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from typing import Dict, Iterator, List, Optional, Tuple, Union
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
|
|
22
|
+
from .gdb_reader import (
|
|
23
|
+
BlobHeader,
|
|
24
|
+
ChannelRecord,
|
|
25
|
+
GDBParseWarning,
|
|
26
|
+
LineRecord,
|
|
27
|
+
check_magic,
|
|
28
|
+
header_fields,
|
|
29
|
+
iter_blobs,
|
|
30
|
+
read_blob_values,
|
|
31
|
+
read_channels,
|
|
32
|
+
read_lines,
|
|
33
|
+
)
|
|
34
|
+
from .registry import find_coordinate_systems
|
|
35
|
+
|
|
36
|
+
# docs/spec.md section 7
|
|
37
|
+
_DB_COMP_NAMES = {
|
|
38
|
+
0: "DB_COMP_NONE",
|
|
39
|
+
1: "DB_COMP_SPEED",
|
|
40
|
+
2: "DB_COMP_SIZE",
|
|
41
|
+
}
|
|
42
|
+
_DB_COMP_CODECS = {
|
|
43
|
+
0: "none (raw values)",
|
|
44
|
+
1: "LZRW1 -- not zlib, despite comp_level implying otherwise (docs/spec.md section 7)",
|
|
45
|
+
2: "zlib/deflate",
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# A plain name is ambiguous whenever a file has more than one line or
|
|
50
|
+
# channel sharing it (real, if unusual -- see channel()/line()'s
|
|
51
|
+
# docstrings). `(name, occurrence)` -- occurrence is a 0-based index into
|
|
52
|
+
# every record sharing that name, in `.channels`/`.lines` order -- lets
|
|
53
|
+
# a caller pick a specific one explicitly instead of relying on context-
|
|
54
|
+
# based disambiguation (or hitting the ValueError it raises when even
|
|
55
|
+
# that's ambiguous). Accepted anywhere a plain name is.
|
|
56
|
+
LineRef = Union[str, Tuple[str, int], "LineRecord"]
|
|
57
|
+
ChannelRef = Union[str, Tuple[str, int], "ChannelRecord"]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class CompressionInfo:
|
|
62
|
+
"""
|
|
63
|
+
This file's *declared* compression mode (header offset 120,
|
|
64
|
+
docs/spec.md section 7). Note this describes what the file was
|
|
65
|
+
configured with, not a guarantee every blob actually used it --
|
|
66
|
+
some real files declare `DB_COMP_SPEED`/`DB_COMP_SIZE` but contain
|
|
67
|
+
zero compressed blobs (docs/spec.md section 7.6), and individual
|
|
68
|
+
"bare" blobs inside a genuinely-compressed file can skip compression
|
|
69
|
+
entirely (docs/spec.md section 7.4) -- `read_blob_values()` already
|
|
70
|
+
detects and handles both cases automatically per-blob.
|
|
71
|
+
"""
|
|
72
|
+
code: Optional[int]
|
|
73
|
+
name: str
|
|
74
|
+
codec: str
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class GDB:
|
|
78
|
+
"""
|
|
79
|
+
High-level, name-based view of a single `.gdb` file.
|
|
80
|
+
|
|
81
|
+
>>> db = GDB("survey.gdb")
|
|
82
|
+
>>> db.line_names[:3]
|
|
83
|
+
['L1000', 'L1001', 'L1010']
|
|
84
|
+
>>> db.channels_on_line("L1000")[:3]
|
|
85
|
+
['Fiducial', 'Easting', 'Northing']
|
|
86
|
+
>>> db.read("L1000", "Easting")[:3]
|
|
87
|
+
[612345.6, 612346.1, 612346.7]
|
|
88
|
+
>>> db.compression.name
|
|
89
|
+
'DB_COMP_NONE'
|
|
90
|
+
>>> db.coordinate_systems
|
|
91
|
+
['NAD83 / UTM zone 11N', 'WGS 84']
|
|
92
|
+
|
|
93
|
+
Channel and line tables are read once, lazily, on first access, and
|
|
94
|
+
cached; the (line, channel) -> blob index used by `read()` and
|
|
95
|
+
`channels_on_line()` is likewise built once (a full blob-chain walk)
|
|
96
|
+
on first use. Unlike the module-level `gdb_reader` functions this
|
|
97
|
+
class is built on (which each reopen `path` fresh, for statelessness),
|
|
98
|
+
`GDB` opens `path` once at construction and reuses that handle for
|
|
99
|
+
every `read()`/`iter_line()` call -- reopening per call was measured
|
|
100
|
+
at ~1.7-1.9x slower against this project's real sample corpus (see
|
|
101
|
+
the Rust-plan's M4 notes). Close it (`db.close()`, or use `GDB` as a
|
|
102
|
+
context manager) when done with it, or just let it get
|
|
103
|
+
garbage-collected -- `__del__` closes it too, as a safety net.
|
|
104
|
+
|
|
105
|
+
Raises `ValueError` at construction time if `path` doesn't start
|
|
106
|
+
with the expected `.gdb` magic -- unlike the module-level functions
|
|
107
|
+
in `gdb_reader`/`registry` (which warn and return empty results),
|
|
108
|
+
since a `GDB` object that isn't backed by a real `.gdb` file can't
|
|
109
|
+
usefully do anything at all.
|
|
110
|
+
"""
|
|
111
|
+
|
|
112
|
+
def __init__(self, path: str):
|
|
113
|
+
self.path = path
|
|
114
|
+
self._file = open(path, "rb")
|
|
115
|
+
header = self._file.read(4096)
|
|
116
|
+
if not check_magic(header):
|
|
117
|
+
self._file.close()
|
|
118
|
+
raise ValueError(
|
|
119
|
+
f"{path}: does not start with the expected '!CBD' magic -- "
|
|
120
|
+
f"not a recognized .gdb file"
|
|
121
|
+
)
|
|
122
|
+
self._fields = header_fields(header)
|
|
123
|
+
self._channels: Optional[List[ChannelRecord]] = None
|
|
124
|
+
self._lines: Optional[List[LineRecord]] = None
|
|
125
|
+
self._channels_by_name: Optional[Dict[str, List[ChannelRecord]]] = None
|
|
126
|
+
self._lines_by_name: Optional[Dict[str, List[LineRecord]]] = None
|
|
127
|
+
self._blob_index: Optional[Dict[Tuple[int, int], BlobHeader]] = None
|
|
128
|
+
self._coordinate_systems: Optional[List[str]] = None
|
|
129
|
+
|
|
130
|
+
def __repr__(self) -> str:
|
|
131
|
+
return f"GDB({self.path!r})"
|
|
132
|
+
|
|
133
|
+
def close(self) -> None:
|
|
134
|
+
"""Close the underlying file handle. Safe to call more than once."""
|
|
135
|
+
self._file.close()
|
|
136
|
+
|
|
137
|
+
def __enter__(self) -> "GDB":
|
|
138
|
+
return self
|
|
139
|
+
|
|
140
|
+
def __exit__(self, *exc_info) -> None:
|
|
141
|
+
self.close()
|
|
142
|
+
|
|
143
|
+
def __del__(self) -> None:
|
|
144
|
+
file = getattr(self, "_file", None)
|
|
145
|
+
if file is not None:
|
|
146
|
+
file.close()
|
|
147
|
+
|
|
148
|
+
# -- header-level info -------------------------------------------------
|
|
149
|
+
|
|
150
|
+
@property
|
|
151
|
+
def chans_max(self) -> Optional[int]:
|
|
152
|
+
return self._fields["chans_max"]
|
|
153
|
+
|
|
154
|
+
@property
|
|
155
|
+
def page_size(self) -> Optional[int]:
|
|
156
|
+
return self._fields["page_size"]
|
|
157
|
+
|
|
158
|
+
@property
|
|
159
|
+
def comp_level(self) -> Optional[int]:
|
|
160
|
+
return self._fields["comp_level"]
|
|
161
|
+
|
|
162
|
+
@property
|
|
163
|
+
def compression(self) -> CompressionInfo:
|
|
164
|
+
code = self.comp_level
|
|
165
|
+
return CompressionInfo(
|
|
166
|
+
code=code,
|
|
167
|
+
name=_DB_COMP_NAMES.get(code, f"unknown({code})"),
|
|
168
|
+
codec=_DB_COMP_CODECS.get(code, "unknown"),
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
@property
|
|
172
|
+
def coordinate_systems(self) -> List[str]:
|
|
173
|
+
"""
|
|
174
|
+
Best-effort list of coordinate-system/map-projection names found
|
|
175
|
+
in this file's REG/IPJ administrative-blob content (docs/spec.md
|
|
176
|
+
section 8-9). An empty list just means none were found -- not
|
|
177
|
+
every real file has this content, and even when it does, this is
|
|
178
|
+
a name-only extraction, not a full projection definition.
|
|
179
|
+
"""
|
|
180
|
+
if self._coordinate_systems is None:
|
|
181
|
+
max_real_line_slot = max((line.index for line in self.lines), default=-1)
|
|
182
|
+
self._coordinate_systems = find_coordinate_systems(
|
|
183
|
+
self.path, max_real_line_slot=max_real_line_slot
|
|
184
|
+
)
|
|
185
|
+
return self._coordinate_systems
|
|
186
|
+
|
|
187
|
+
# -- channels / lines ----------------------------------------------------
|
|
188
|
+
|
|
189
|
+
@property
|
|
190
|
+
def channels(self) -> List[ChannelRecord]:
|
|
191
|
+
if self._channels is None:
|
|
192
|
+
self._channels = read_channels(self.path)
|
|
193
|
+
self._channels_by_name = {}
|
|
194
|
+
for c in self._channels:
|
|
195
|
+
self._channels_by_name.setdefault(c.name, []).append(c)
|
|
196
|
+
return self._channels
|
|
197
|
+
|
|
198
|
+
@property
|
|
199
|
+
def channel_names(self) -> List[str]:
|
|
200
|
+
return [c.name for c in self.channels]
|
|
201
|
+
|
|
202
|
+
@property
|
|
203
|
+
def lines(self) -> List[LineRecord]:
|
|
204
|
+
if self._lines is None:
|
|
205
|
+
self._lines = read_lines(self.path)
|
|
206
|
+
self._lines_by_name = {}
|
|
207
|
+
for l in self._lines:
|
|
208
|
+
self._lines_by_name.setdefault(l.name, []).append(l)
|
|
209
|
+
return self._lines
|
|
210
|
+
|
|
211
|
+
@property
|
|
212
|
+
def line_names(self) -> List[str]:
|
|
213
|
+
return [l.name for l in self.lines]
|
|
214
|
+
|
|
215
|
+
def _nth_by_name(self, by_name: Dict[str, list], name: str, occurrence: int, kind: str):
|
|
216
|
+
"""
|
|
217
|
+
Shared lookup for the `(name, occurrence)` form `channel()`/
|
|
218
|
+
`line()` both accept: `occurrence` is a 0-based index into every
|
|
219
|
+
record sharing `name`, in `.channels`/`.lines` order -- an
|
|
220
|
+
explicit way to pick a specific one when a plain name is
|
|
221
|
+
ambiguous, rather than raising or guessing.
|
|
222
|
+
"""
|
|
223
|
+
matches = by_name.get(name)
|
|
224
|
+
if not matches:
|
|
225
|
+
raise KeyError(f"{self.path}: no {kind} named {name!r}")
|
|
226
|
+
try:
|
|
227
|
+
return matches[occurrence]
|
|
228
|
+
except IndexError:
|
|
229
|
+
raise IndexError(
|
|
230
|
+
f"{self.path}: only {len(matches)} {kind}(s) named {name!r} "
|
|
231
|
+
f"(requested occurrence {occurrence})"
|
|
232
|
+
) from None
|
|
233
|
+
|
|
234
|
+
def channel(self, name: ChannelRef) -> ChannelRecord:
|
|
235
|
+
"""
|
|
236
|
+
Look up a channel by name. Raises `KeyError` if no channel has
|
|
237
|
+
this name.
|
|
238
|
+
|
|
239
|
+
Raises `ValueError` if more than one channel shares this name --
|
|
240
|
+
a real, if unusual, on-disk possibility (confirmed for real on a
|
|
241
|
+
sample file with two channels each named `UTC`, `RADAR`, and
|
|
242
|
+
`RAWMAG`), for which there's no file-wide way to pick the
|
|
243
|
+
"right" one without a line to disambiguate against. `read()`
|
|
244
|
+
already disambiguates this automatically using line context
|
|
245
|
+
(see `_resolve_channel_on_line`).
|
|
246
|
+
|
|
247
|
+
Pass `(name, occurrence)` instead of a plain name (`occurrence`
|
|
248
|
+
a 0-based index into every channel sharing that name, in
|
|
249
|
+
`.channels` order) to pick a specific one explicitly rather than
|
|
250
|
+
relying on that, or hitting the `ValueError` above.
|
|
251
|
+
"""
|
|
252
|
+
if self._channels_by_name is None:
|
|
253
|
+
self.channels # populate the cache
|
|
254
|
+
if isinstance(name, tuple):
|
|
255
|
+
actual_name, occurrence = name
|
|
256
|
+
return self._nth_by_name(self._channels_by_name, actual_name, occurrence, "channel")
|
|
257
|
+
matches = self._channels_by_name.get(name)
|
|
258
|
+
if not matches:
|
|
259
|
+
raise KeyError(f"{self.path}: no channel named {name!r}")
|
|
260
|
+
if len(matches) > 1:
|
|
261
|
+
raise ValueError(
|
|
262
|
+
f"{self.path}: {len(matches)} channels are named {name!r} -- "
|
|
263
|
+
f"ambiguous without a line to disambiguate against; use "
|
|
264
|
+
f"read(line, name) (which resolves this using the line's "
|
|
265
|
+
f"own data), pass (name, occurrence) to pick a specific "
|
|
266
|
+
f"one explicitly, or pick a ChannelRecord from .channels "
|
|
267
|
+
f"yourself"
|
|
268
|
+
)
|
|
269
|
+
return matches[0]
|
|
270
|
+
|
|
271
|
+
def line(self, name: LineRef) -> LineRecord:
|
|
272
|
+
"""
|
|
273
|
+
Look up a line by name. Raises `KeyError` if no line has this
|
|
274
|
+
name.
|
|
275
|
+
|
|
276
|
+
Raises `ValueError` if more than one line shares this name --
|
|
277
|
+
the line table has the same on-disk shape as the channel table
|
|
278
|
+
(see `channel()`'s docstring), with nothing in the format
|
|
279
|
+
forbidding a duplicate name there either; not yet observed on a
|
|
280
|
+
real file, but handled the same way on principle rather than
|
|
281
|
+
left as a silent last-one-wins lookup.
|
|
282
|
+
|
|
283
|
+
Pass `(name, occurrence)` instead of a plain name (`occurrence`
|
|
284
|
+
a 0-based index into every line sharing that name, in `.lines`
|
|
285
|
+
order) to pick a specific one explicitly rather than hitting
|
|
286
|
+
that `ValueError`.
|
|
287
|
+
"""
|
|
288
|
+
if self._lines_by_name is None:
|
|
289
|
+
self.lines # populate the cache
|
|
290
|
+
if isinstance(name, tuple):
|
|
291
|
+
actual_name, occurrence = name
|
|
292
|
+
return self._nth_by_name(self._lines_by_name, actual_name, occurrence, "line")
|
|
293
|
+
matches = self._lines_by_name.get(name)
|
|
294
|
+
if not matches:
|
|
295
|
+
raise KeyError(f"{self.path}: no line named {name!r}")
|
|
296
|
+
if len(matches) > 1:
|
|
297
|
+
raise ValueError(
|
|
298
|
+
f"{self.path}: {len(matches)} lines are named {name!r} -- "
|
|
299
|
+
f"ambiguous; pass (name, occurrence) to pick a specific "
|
|
300
|
+
f"one explicitly, or pick a LineRecord from .lines yourself"
|
|
301
|
+
)
|
|
302
|
+
return matches[0]
|
|
303
|
+
|
|
304
|
+
def _resolve_line(self, line: LineRef) -> LineRecord:
|
|
305
|
+
return line if isinstance(line, LineRecord) else self.line(line)
|
|
306
|
+
|
|
307
|
+
# -- data access ---------------------------------------------------------
|
|
308
|
+
|
|
309
|
+
def _ensure_blob_index(self) -> Dict[Tuple[int, int], BlobHeader]:
|
|
310
|
+
"""
|
|
311
|
+
Build the full (line_slot, channel_slot) -> BlobHeader map with one
|
|
312
|
+
blob-chain walk, cached from then on. `iter_blobs`/`find_blob`
|
|
313
|
+
themselves recommend this for anything beyond an occasional
|
|
314
|
+
one-off lookup -- this class always wants line/channel listings
|
|
315
|
+
and random-access reads, so it always builds the index.
|
|
316
|
+
"""
|
|
317
|
+
if self._blob_index is None:
|
|
318
|
+
chans_max = self.chans_max
|
|
319
|
+
index: Dict[Tuple[int, int], BlobHeader] = {}
|
|
320
|
+
for blob in iter_blobs(self.path):
|
|
321
|
+
index[blob.line_channel(chans_max)] = blob
|
|
322
|
+
self._blob_index = index
|
|
323
|
+
self._calibrate_line_indices()
|
|
324
|
+
return self._blob_index
|
|
325
|
+
|
|
326
|
+
def _calibrate_line_indices(self) -> None:
|
|
327
|
+
"""
|
|
328
|
+
Correct a possible small, fixed off-by-N in every LineRecord.index
|
|
329
|
+
(see find_line_table's and read_lines's docstrings in
|
|
330
|
+
gdb_reader.py) by checking, for a handful of small integer
|
|
331
|
+
shifts, which one makes the most already-found lines actually
|
|
332
|
+
have at least one real data blob on disk for *some* channel --
|
|
333
|
+
then applying the winning shift to every LineRecord.index in
|
|
334
|
+
place. This is a strictly stronger signal than anything available
|
|
335
|
+
from the symbol-table bytes alone (it's checking against the
|
|
336
|
+
real, self-describing blob chain, not another heuristic guess),
|
|
337
|
+
confirmed to fix a real off-by-one found on a GSQ file
|
|
338
|
+
(`rm001141`) without disturbing any of the other real files this
|
|
339
|
+
package has been tested against (where the winning shift is 0,
|
|
340
|
+
i.e. a no-op).
|
|
341
|
+
|
|
342
|
+
Runs once, right after the blob index is first built -- cheap
|
|
343
|
+
relative to that index build itself (already O(number of real
|
|
344
|
+
lines) additional work, not another file scan).
|
|
345
|
+
"""
|
|
346
|
+
lines = self.lines
|
|
347
|
+
if not lines or not self._blob_index:
|
|
348
|
+
return
|
|
349
|
+
slots_with_data = {line_slot for line_slot, _channel_slot in self._blob_index}
|
|
350
|
+
best_offset, best_score = 0, -1
|
|
351
|
+
for offset in range(-4, 5):
|
|
352
|
+
score = sum(1 for l in lines if (l.index + offset) in slots_with_data)
|
|
353
|
+
if score > best_score:
|
|
354
|
+
best_score, best_offset = score, offset
|
|
355
|
+
if best_offset:
|
|
356
|
+
for l in lines:
|
|
357
|
+
l.index += best_offset
|
|
358
|
+
|
|
359
|
+
def _channels_with_data_on_line(self, line_rec: LineRecord) -> List[Tuple[ChannelRecord, BlobHeader]]:
|
|
360
|
+
"""
|
|
361
|
+
`(channel, blob)` for every channel that actually has a real data
|
|
362
|
+
blob recorded for `line_rec`, in `self.channels` order. Shared by
|
|
363
|
+
`channels_on_line` and `iter_line` so both agree on exactly which
|
|
364
|
+
channel matched -- looking a channel back up by name afterward
|
|
365
|
+
would be ambiguous for a file with duplicate channel names (real
|
|
366
|
+
channel records aren't guaranteed unique by name), so callers
|
|
367
|
+
that need the actual data should go through this, not re-resolve
|
|
368
|
+
`channels_on_line`'s returned names.
|
|
369
|
+
"""
|
|
370
|
+
index = self._ensure_blob_index()
|
|
371
|
+
return [
|
|
372
|
+
(c, blob) for c in self.channels
|
|
373
|
+
if (blob := index.get((line_rec.index, c.index))) is not None
|
|
374
|
+
and (blob.row_count is None or blob.row_count >= 0)
|
|
375
|
+
]
|
|
376
|
+
|
|
377
|
+
def _resolve_channel_on_line(
|
|
378
|
+
self, line_rec: LineRecord, name: str
|
|
379
|
+
) -> Tuple[ChannelRecord, Optional[BlobHeader]]:
|
|
380
|
+
"""
|
|
381
|
+
Resolve a channel name to `(ChannelRecord, BlobHeader-or-None)`
|
|
382
|
+
for a specific line, using the line's own data to disambiguate a
|
|
383
|
+
name shared by more than one channel (see `channel()`'s
|
|
384
|
+
docstring) -- picking whichever same-named channel actually has
|
|
385
|
+
data on this line, rather than an arbitrary one. Raises
|
|
386
|
+
`KeyError` if no channel has this name at all.
|
|
387
|
+
|
|
388
|
+
If more than one same-named channel has data on this same line,
|
|
389
|
+
that's genuinely ambiguous (not just "the file happens to reuse
|
|
390
|
+
this name") and raises `ValueError` -- every real duplicate-name
|
|
391
|
+
case found so far has only one of the duplicates actually
|
|
392
|
+
populated per line, so this hasn't been observed, but there's no
|
|
393
|
+
principled way to guess if it ever is.
|
|
394
|
+
"""
|
|
395
|
+
if self._channels_by_name is None:
|
|
396
|
+
self.channels # populate the cache
|
|
397
|
+
matches = self._channels_by_name.get(name)
|
|
398
|
+
if not matches:
|
|
399
|
+
raise KeyError(f"{self.path}: no channel named {name!r}")
|
|
400
|
+
if len(matches) == 1:
|
|
401
|
+
chan_rec = matches[0]
|
|
402
|
+
blob = self._ensure_blob_index().get((line_rec.index, chan_rec.index))
|
|
403
|
+
return chan_rec, blob
|
|
404
|
+
index = self._ensure_blob_index()
|
|
405
|
+
with_data = [
|
|
406
|
+
(c, blob) for c in matches
|
|
407
|
+
if (blob := index.get((line_rec.index, c.index))) is not None
|
|
408
|
+
and (blob.row_count is None or blob.row_count >= 0)
|
|
409
|
+
]
|
|
410
|
+
if len(with_data) > 1:
|
|
411
|
+
raise ValueError(
|
|
412
|
+
f"{self.path}: {len(with_data)} channels named {name!r} all "
|
|
413
|
+
f"have data on line {line_rec.name!r} -- genuinely "
|
|
414
|
+
f"ambiguous even with line context; pass (name, occurrence) "
|
|
415
|
+
f"to pick a specific one explicitly (occurrence is a 0-based "
|
|
416
|
+
f"index into every channel named {name!r}, in .channels "
|
|
417
|
+
f"order), or pick a ChannelRecord yourself"
|
|
418
|
+
)
|
|
419
|
+
if with_data:
|
|
420
|
+
return with_data[0]
|
|
421
|
+
# None of the same-named channels have data on this line -- report
|
|
422
|
+
# "no data" the same way an unambiguous miss would, using the
|
|
423
|
+
# first match's ChannelRecord just to name it in the warning.
|
|
424
|
+
return matches[0], None
|
|
425
|
+
|
|
426
|
+
def channels_on_line(self, line: LineRef) -> List[str]:
|
|
427
|
+
"""
|
|
428
|
+
Names of channels that actually have a real data blob recorded
|
|
429
|
+
for `line` -- the format stores a sparse (line, channel) grid
|
|
430
|
+
(docs/spec.md section 1), so most lines only populate a subset
|
|
431
|
+
of this file's full channel list. `line` may be a line name, a
|
|
432
|
+
`(name, occurrence)` pair (see `line()`), or a `LineRecord`.
|
|
433
|
+
|
|
434
|
+
If two channels share a name and both have data on this line,
|
|
435
|
+
that name appears twice here (a list, so nothing is silently
|
|
436
|
+
dropped) -- use `iter_line()` instead if you need the actual
|
|
437
|
+
`ChannelRecord` for each entry, not just its name.
|
|
438
|
+
"""
|
|
439
|
+
line_rec = self._resolve_line(line)
|
|
440
|
+
return [c.name for c, _blob in self._channels_with_data_on_line(line_rec)]
|
|
441
|
+
|
|
442
|
+
def read(self, line: LineRef, channel: ChannelRef) -> np.ndarray:
|
|
443
|
+
"""
|
|
444
|
+
Random access by name: decode and return every value recorded
|
|
445
|
+
for `channel` on `line`, as a numpy `ndarray` -- 1-D for an
|
|
446
|
+
ordinary scalar channel, 2-D `(n_rows, channel.array_width)`
|
|
447
|
+
for a VA/array channel (docs/spec.md section 5), dtype matching
|
|
448
|
+
the channel's `GS_*` type, or `object` (holding `str`) for a
|
|
449
|
+
string-typed channel. `line`/`channel` may be names,
|
|
450
|
+
`(name, occurrence)` pairs (see `line()`/`channel()`), or
|
|
451
|
+
`LineRecord`/`ChannelRecord` instances.
|
|
452
|
+
|
|
453
|
+
Raises `KeyError` if `line` or `channel` isn't a name this file
|
|
454
|
+
has. If `channel` is a plain name shared by more than one
|
|
455
|
+
channel (see `channel()`'s docstring), this resolves it using
|
|
456
|
+
`line`'s own data (`_resolve_channel_on_line`) rather than
|
|
457
|
+
picking an arbitrary one -- raising `ValueError` only if that's
|
|
458
|
+
*still* ambiguous (more than one same-named channel has data on
|
|
459
|
+
this exact line); pass `(name, occurrence)` or a specific
|
|
460
|
+
`ChannelRecord` to sidestep either lookup. Returns an empty
|
|
461
|
+
array (with a `GDBParseWarning`, per `read_blob_values`) if the
|
|
462
|
+
name is valid but this specific (line, channel) pair has no
|
|
463
|
+
data blob, or its data can't be decoded -- consistent with the
|
|
464
|
+
rest of this package's degrade-gracefully philosophy for
|
|
465
|
+
decode-time problems, as opposed to a plain lookup-by-name
|
|
466
|
+
mistake (which does raise).
|
|
467
|
+
"""
|
|
468
|
+
line_rec = self._resolve_line(line)
|
|
469
|
+
if isinstance(channel, ChannelRecord):
|
|
470
|
+
chan_rec = channel
|
|
471
|
+
blob = self._ensure_blob_index().get((line_rec.index, chan_rec.index))
|
|
472
|
+
elif isinstance(channel, tuple):
|
|
473
|
+
chan_rec = self.channel(channel)
|
|
474
|
+
blob = self._ensure_blob_index().get((line_rec.index, chan_rec.index))
|
|
475
|
+
else:
|
|
476
|
+
chan_rec, blob = self._resolve_channel_on_line(line_rec, channel)
|
|
477
|
+
if blob is None:
|
|
478
|
+
warnings.warn(
|
|
479
|
+
f"{self.path}: no data blob for line {line_rec.name!r}, "
|
|
480
|
+
f"channel {chan_rec.name!r} -- this (line, channel) pair "
|
|
481
|
+
f"was likely never recorded (the format's grid is sparse, "
|
|
482
|
+
f"docs/spec.md section 1)",
|
|
483
|
+
GDBParseWarning, stacklevel=2,
|
|
484
|
+
)
|
|
485
|
+
return np.array([])
|
|
486
|
+
return read_blob_values(
|
|
487
|
+
self.path, blob, chan_rec,
|
|
488
|
+
comp_level=self.comp_level or 0, page_size=self.page_size,
|
|
489
|
+
file=self._file,
|
|
490
|
+
)
|
|
491
|
+
|
|
492
|
+
def iter_line(self, line: LineRef) -> Iterator[Tuple[ChannelRecord, np.ndarray]]:
|
|
493
|
+
"""
|
|
494
|
+
Yield `(channel, values)` for every channel that actually has
|
|
495
|
+
data on `line`, via `_channels_with_data_on_line` directly
|
|
496
|
+
rather than one `read()` call per channel (saving the repeated
|
|
497
|
+
name lookups).
|
|
498
|
+
|
|
499
|
+
Yields the `ChannelRecord` itself, not just its name (`values`
|
|
500
|
+
is the same as `read(line, channel)` would give for that exact
|
|
501
|
+
channel) -- deliberately, so that if two channels share a name
|
|
502
|
+
and both have data on this line, both still come through as
|
|
503
|
+
distinct, fully-identified entries. `dict(db.iter_line(line))`
|
|
504
|
+
keyed by the records themselves preserves that; collapsing to
|
|
505
|
+
`channel.name` yourself reintroduces the same collision `read()`
|
|
506
|
+
raises on, so do that deliberately if you do it at all.
|
|
507
|
+
|
|
508
|
+
A `rayon`-based parallel batch decoder was tried for this and
|
|
509
|
+
removed: benchmarked against this project's real sample corpus,
|
|
510
|
+
it was consistently ~3x *slower* than plain sequential calls at
|
|
511
|
+
every scale tried, since this format's chunks are small enough
|
|
512
|
+
that the Rust decoder (`pygdb._native`) already finishes each one
|
|
513
|
+
in a fraction of a millisecond -- not enough work per chunk to
|
|
514
|
+
amortize rayon's per-task dispatch cost. See `rust/src/lib.rs`'s
|
|
515
|
+
module doc for the full note.
|
|
516
|
+
"""
|
|
517
|
+
line_rec = self._resolve_line(line)
|
|
518
|
+
for c, blob in self._channels_with_data_on_line(line_rec):
|
|
519
|
+
values = read_blob_values(
|
|
520
|
+
self.path, blob, c,
|
|
521
|
+
comp_level=self.comp_level or 0, page_size=self.page_size,
|
|
522
|
+
file=self._file,
|
|
523
|
+
)
|
|
524
|
+
yield c, values
|
|
525
|
+
|
|
526
|
+
def to_xarray(self, line: LineRef) -> "xr.Dataset":
|
|
527
|
+
"""
|
|
528
|
+
Build an `xarray.Dataset` for every channel that has data on
|
|
529
|
+
`line` -- one data variable per channel, sharing a common
|
|
530
|
+
`"station"` dimension. Needs the optional `xarray` dependency
|
|
531
|
+
(`pip install python-gdb[xarray]`), imported lazily here so
|
|
532
|
+
importing `pygdb` itself never requires it.
|
|
533
|
+
|
|
534
|
+
A VA/array channel (docs/spec.md section 5) gets its own
|
|
535
|
+
second dimension, `f"{name}_bin"` -- deliberately *not* shared
|
|
536
|
+
with any other array channel even when their `array_width`
|
|
537
|
+
happens to match (e.g. real `ISPD`/`ISPU` are both 512-wide in
|
|
538
|
+
a real USGS file): two channels having the same width is a
|
|
539
|
+
coincidence, not a guarantee they share a semantic axis. Align/
|
|
540
|
+
rename dimensions yourself afterward if you know two channels
|
|
541
|
+
genuinely do.
|
|
542
|
+
|
|
543
|
+
If two channels share a name and both have data on this line
|
|
544
|
+
(confirmed structurally possible -- see `channel()`'s
|
|
545
|
+
docstring -- though never yet observed with data on both), the
|
|
546
|
+
variable name for every occurrence after the first is
|
|
547
|
+
disambiguated as `f"{name}[{occurrence}]"`, `occurrence` being
|
|
548
|
+
the same 0-based index into every channel sharing that name (in
|
|
549
|
+
`.channels` order) that `channel()`'s `(name, occurrence)` form
|
|
550
|
+
uses -- so `ds["UTC[1]"]` and `db.channel(("UTC", 1))` refer to
|
|
551
|
+
the same channel. Raises a `GDBParseWarning` when this actually
|
|
552
|
+
triggers, since a caller not expecting a bracket-suffixed
|
|
553
|
+
variable name should be told why one showed up.
|
|
554
|
+
|
|
555
|
+
If channels on this line don't all decode to the same row
|
|
556
|
+
count (a truncated/corrupt file -- truncation only ever
|
|
557
|
+
shortens a channel, never lengthens it), the *shorter*
|
|
558
|
+
channel(s) keep their full (shorter) data rather than being cut
|
|
559
|
+
down further, or cutting the other channels down to match: a
|
|
560
|
+
short channel gets its own first dimension, `f"{name}_station"`,
|
|
561
|
+
instead of the shared `"station"` (the same "give it its own
|
|
562
|
+
dimension rather than lose data to fit one" principle as the
|
|
563
|
+
array-channel case above). Also raises a `GDBParseWarning`.
|
|
564
|
+
|
|
565
|
+
No channel is auto-promoted to a coordinate -- `"station"` is a
|
|
566
|
+
bare integer range index, and every channel (however
|
|
567
|
+
conventionally named) is a plain data variable; call
|
|
568
|
+
`ds.set_coords(...)` yourself if you want one -- no real file
|
|
569
|
+
names its channels consistently enough for this reader to
|
|
570
|
+
guess which one(s) you'd want without risking guessing wrong.
|
|
571
|
+
"""
|
|
572
|
+
try:
|
|
573
|
+
import xarray as xr
|
|
574
|
+
except ImportError as e:
|
|
575
|
+
raise ImportError(
|
|
576
|
+
"to_xarray() needs the optional 'xarray' dependency -- "
|
|
577
|
+
"install with `pip install python-gdb[xarray]`"
|
|
578
|
+
) from e
|
|
579
|
+
|
|
580
|
+
line_rec = self._resolve_line(line)
|
|
581
|
+
decoded = [
|
|
582
|
+
(c, read_blob_values(
|
|
583
|
+
self.path, blob, c,
|
|
584
|
+
comp_level=self.comp_level or 0, page_size=self.page_size,
|
|
585
|
+
file=self._file,
|
|
586
|
+
))
|
|
587
|
+
for c, blob in self._channels_with_data_on_line(line_rec)
|
|
588
|
+
]
|
|
589
|
+
if self._channels_by_name is None:
|
|
590
|
+
self.channels # populate the cache (for occurrence numbering)
|
|
591
|
+
|
|
592
|
+
station_length = max((len(values) for _c, values in decoded), default=0)
|
|
593
|
+
name_counts: Dict[str, int] = {}
|
|
594
|
+
for c, _values in decoded:
|
|
595
|
+
name_counts[c.name] = name_counts.get(c.name, 0) + 1
|
|
596
|
+
|
|
597
|
+
data_vars = {}
|
|
598
|
+
seen_so_far: Dict[str, int] = {}
|
|
599
|
+
for c, values in decoded:
|
|
600
|
+
seen_so_far[c.name] = seen_so_far.get(c.name, 0) + 1
|
|
601
|
+
if name_counts[c.name] > 1:
|
|
602
|
+
occurrence = self._channels_by_name[c.name].index(c)
|
|
603
|
+
var_name = c.name if seen_so_far[c.name] == 1 else f"{c.name}[{occurrence}]"
|
|
604
|
+
warnings.warn(
|
|
605
|
+
f"{self.path}: line {line_rec.name!r} has {name_counts[c.name]} "
|
|
606
|
+
f"channels named {c.name!r} with data -- using {var_name!r} "
|
|
607
|
+
f"for occurrence {occurrence} (pass (name, occurrence) to "
|
|
608
|
+
f"channel() for the same numbering)",
|
|
609
|
+
GDBParseWarning, stacklevel=2,
|
|
610
|
+
)
|
|
611
|
+
else:
|
|
612
|
+
var_name = c.name
|
|
613
|
+
|
|
614
|
+
if len(values) == station_length:
|
|
615
|
+
station_dim = "station"
|
|
616
|
+
else:
|
|
617
|
+
station_dim = f"{var_name}_station"
|
|
618
|
+
warnings.warn(
|
|
619
|
+
f"{self.path}: line {line_rec.name!r} channel {c.name!r} "
|
|
620
|
+
f"decoded {len(values)} row(s), expected {station_length} "
|
|
621
|
+
f"(the max across this line's channels) -- likely "
|
|
622
|
+
f"truncated; keeping its own {len(values)}-row dimension "
|
|
623
|
+
f"{station_dim!r} rather than cutting other channels down "
|
|
624
|
+
f"to match",
|
|
625
|
+
GDBParseWarning, stacklevel=2,
|
|
626
|
+
)
|
|
627
|
+
|
|
628
|
+
dims = (station_dim, f"{var_name}_bin") if c.is_array else (station_dim,)
|
|
629
|
+
attrs = {"type_name": c.type_name, "format_name": c.format_name}
|
|
630
|
+
if c.is_array:
|
|
631
|
+
attrs["array_basetype_name"] = c.array_basetype_name
|
|
632
|
+
data_vars[var_name] = (dims, values, attrs)
|
|
633
|
+
|
|
634
|
+
return xr.Dataset(
|
|
635
|
+
data_vars,
|
|
636
|
+
attrs={
|
|
637
|
+
"line_name": line_rec.name,
|
|
638
|
+
"line_category": line_rec.category_name,
|
|
639
|
+
"path": self.path,
|
|
640
|
+
},
|
|
641
|
+
)
|