python-gdb 0.1.0__cp314-cp314t-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pygdb/gdb.py ADDED
@@ -0,0 +1,641 @@
1
+ """
2
+ A user-facing, high-level wrapper around a single `.gdb` file.
3
+
4
+ Everything here is built on top of the lower-level primitives in
5
+ `gdb_reader`/`registry` (header parsing, symbol tables, the blob chain,
6
+ value decoding, coordinate-system extraction) -- `GDB` just gives them a
7
+ single, name-based entry point: list the lines and channels, see which
8
+ channels actually have data on a given line (the format's sparse (line,
9
+ channel) grid, docs/spec.md section 1/6.1), read a specific (line,
10
+ channel) pair's values by name, and describe the file's compression mode
11
+ and coordinate reference system(s).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import warnings
17
+ from dataclasses import dataclass
18
+ from typing import Dict, Iterator, List, Optional, Tuple, Union
19
+
20
+ import numpy as np
21
+
22
+ from .gdb_reader import (
23
+ BlobHeader,
24
+ ChannelRecord,
25
+ GDBParseWarning,
26
+ LineRecord,
27
+ check_magic,
28
+ header_fields,
29
+ iter_blobs,
30
+ read_blob_values,
31
+ read_channels,
32
+ read_lines,
33
+ )
34
+ from .registry import find_coordinate_systems
35
+
36
+ # docs/spec.md section 7
37
+ _DB_COMP_NAMES = {
38
+ 0: "DB_COMP_NONE",
39
+ 1: "DB_COMP_SPEED",
40
+ 2: "DB_COMP_SIZE",
41
+ }
42
+ _DB_COMP_CODECS = {
43
+ 0: "none (raw values)",
44
+ 1: "LZRW1 -- not zlib, despite comp_level implying otherwise (docs/spec.md section 7)",
45
+ 2: "zlib/deflate",
46
+ }
47
+
48
+
49
+ # A plain name is ambiguous whenever a file has more than one line or
50
+ # channel sharing it (real, if unusual -- see channel()/line()'s
51
+ # docstrings). `(name, occurrence)` -- occurrence is a 0-based index into
52
+ # every record sharing that name, in `.channels`/`.lines` order -- lets
53
+ # a caller pick a specific one explicitly instead of relying on context-
54
+ # based disambiguation (or hitting the ValueError it raises when even
55
+ # that's ambiguous). Accepted anywhere a plain name is.
56
+ LineRef = Union[str, Tuple[str, int], "LineRecord"]
57
+ ChannelRef = Union[str, Tuple[str, int], "ChannelRecord"]
58
+
59
+
60
+ @dataclass
61
+ class CompressionInfo:
62
+ """
63
+ This file's *declared* compression mode (header offset 120,
64
+ docs/spec.md section 7). Note this describes what the file was
65
+ configured with, not a guarantee every blob actually used it --
66
+ some real files declare `DB_COMP_SPEED`/`DB_COMP_SIZE` but contain
67
+ zero compressed blobs (docs/spec.md section 7.6), and individual
68
+ "bare" blobs inside a genuinely-compressed file can skip compression
69
+ entirely (docs/spec.md section 7.4) -- `read_blob_values()` already
70
+ detects and handles both cases automatically per-blob.
71
+ """
72
+ code: Optional[int]
73
+ name: str
74
+ codec: str
75
+
76
+
77
+ class GDB:
78
+ """
79
+ High-level, name-based view of a single `.gdb` file.
80
+
81
+ >>> db = GDB("survey.gdb")
82
+ >>> db.line_names[:3]
83
+ ['L1000', 'L1001', 'L1010']
84
+ >>> db.channels_on_line("L1000")[:3]
85
+ ['Fiducial', 'Easting', 'Northing']
86
+ >>> db.read("L1000", "Easting")[:3]
87
+ [612345.6, 612346.1, 612346.7]
88
+ >>> db.compression.name
89
+ 'DB_COMP_NONE'
90
+ >>> db.coordinate_systems
91
+ ['NAD83 / UTM zone 11N', 'WGS 84']
92
+
93
+ Channel and line tables are read once, lazily, on first access, and
94
+ cached; the (line, channel) -> blob index used by `read()` and
95
+ `channels_on_line()` is likewise built once (a full blob-chain walk)
96
+ on first use. Unlike the module-level `gdb_reader` functions this
97
+ class is built on (which each reopen `path` fresh, for statelessness),
98
+ `GDB` opens `path` once at construction and reuses that handle for
99
+ every `read()`/`iter_line()` call -- reopening per call was measured
100
+ at ~1.7-1.9x slower against this project's real sample corpus (see
101
+ the Rust-plan's M4 notes). Close it (`db.close()`, or use `GDB` as a
102
+ context manager) when done with it, or just let it get
103
+ garbage-collected -- `__del__` closes it too, as a safety net.
104
+
105
+ Raises `ValueError` at construction time if `path` doesn't start
106
+ with the expected `.gdb` magic -- unlike the module-level functions
107
+ in `gdb_reader`/`registry` (which warn and return empty results),
108
+ since a `GDB` object that isn't backed by a real `.gdb` file can't
109
+ usefully do anything at all.
110
+ """
111
+
112
+ def __init__(self, path: str):
113
+ self.path = path
114
+ self._file = open(path, "rb")
115
+ header = self._file.read(4096)
116
+ if not check_magic(header):
117
+ self._file.close()
118
+ raise ValueError(
119
+ f"{path}: does not start with the expected '!CBD' magic -- "
120
+ f"not a recognized .gdb file"
121
+ )
122
+ self._fields = header_fields(header)
123
+ self._channels: Optional[List[ChannelRecord]] = None
124
+ self._lines: Optional[List[LineRecord]] = None
125
+ self._channels_by_name: Optional[Dict[str, List[ChannelRecord]]] = None
126
+ self._lines_by_name: Optional[Dict[str, List[LineRecord]]] = None
127
+ self._blob_index: Optional[Dict[Tuple[int, int], BlobHeader]] = None
128
+ self._coordinate_systems: Optional[List[str]] = None
129
+
130
+ def __repr__(self) -> str:
131
+ return f"GDB({self.path!r})"
132
+
133
+ def close(self) -> None:
134
+ """Close the underlying file handle. Safe to call more than once."""
135
+ self._file.close()
136
+
137
+ def __enter__(self) -> "GDB":
138
+ return self
139
+
140
+ def __exit__(self, *exc_info) -> None:
141
+ self.close()
142
+
143
+ def __del__(self) -> None:
144
+ file = getattr(self, "_file", None)
145
+ if file is not None:
146
+ file.close()
147
+
148
+ # -- header-level info -------------------------------------------------
149
+
150
+ @property
151
+ def chans_max(self) -> Optional[int]:
152
+ return self._fields["chans_max"]
153
+
154
+ @property
155
+ def page_size(self) -> Optional[int]:
156
+ return self._fields["page_size"]
157
+
158
+ @property
159
+ def comp_level(self) -> Optional[int]:
160
+ return self._fields["comp_level"]
161
+
162
+ @property
163
+ def compression(self) -> CompressionInfo:
164
+ code = self.comp_level
165
+ return CompressionInfo(
166
+ code=code,
167
+ name=_DB_COMP_NAMES.get(code, f"unknown({code})"),
168
+ codec=_DB_COMP_CODECS.get(code, "unknown"),
169
+ )
170
+
171
+ @property
172
+ def coordinate_systems(self) -> List[str]:
173
+ """
174
+ Best-effort list of coordinate-system/map-projection names found
175
+ in this file's REG/IPJ administrative-blob content (docs/spec.md
176
+ section 8-9). An empty list just means none were found -- not
177
+ every real file has this content, and even when it does, this is
178
+ a name-only extraction, not a full projection definition.
179
+ """
180
+ if self._coordinate_systems is None:
181
+ max_real_line_slot = max((line.index for line in self.lines), default=-1)
182
+ self._coordinate_systems = find_coordinate_systems(
183
+ self.path, max_real_line_slot=max_real_line_slot
184
+ )
185
+ return self._coordinate_systems
186
+
187
+ # -- channels / lines ----------------------------------------------------
188
+
189
+ @property
190
+ def channels(self) -> List[ChannelRecord]:
191
+ if self._channels is None:
192
+ self._channels = read_channels(self.path)
193
+ self._channels_by_name = {}
194
+ for c in self._channels:
195
+ self._channels_by_name.setdefault(c.name, []).append(c)
196
+ return self._channels
197
+
198
+ @property
199
+ def channel_names(self) -> List[str]:
200
+ return [c.name for c in self.channels]
201
+
202
+ @property
203
+ def lines(self) -> List[LineRecord]:
204
+ if self._lines is None:
205
+ self._lines = read_lines(self.path)
206
+ self._lines_by_name = {}
207
+ for l in self._lines:
208
+ self._lines_by_name.setdefault(l.name, []).append(l)
209
+ return self._lines
210
+
211
+ @property
212
+ def line_names(self) -> List[str]:
213
+ return [l.name for l in self.lines]
214
+
215
+ def _nth_by_name(self, by_name: Dict[str, list], name: str, occurrence: int, kind: str):
216
+ """
217
+ Shared lookup for the `(name, occurrence)` form `channel()`/
218
+ `line()` both accept: `occurrence` is a 0-based index into every
219
+ record sharing `name`, in `.channels`/`.lines` order -- an
220
+ explicit way to pick a specific one when a plain name is
221
+ ambiguous, rather than raising or guessing.
222
+ """
223
+ matches = by_name.get(name)
224
+ if not matches:
225
+ raise KeyError(f"{self.path}: no {kind} named {name!r}")
226
+ try:
227
+ return matches[occurrence]
228
+ except IndexError:
229
+ raise IndexError(
230
+ f"{self.path}: only {len(matches)} {kind}(s) named {name!r} "
231
+ f"(requested occurrence {occurrence})"
232
+ ) from None
233
+
234
+ def channel(self, name: ChannelRef) -> ChannelRecord:
235
+ """
236
+ Look up a channel by name. Raises `KeyError` if no channel has
237
+ this name.
238
+
239
+ Raises `ValueError` if more than one channel shares this name --
240
+ a real, if unusual, on-disk possibility (confirmed for real on a
241
+ sample file with two channels each named `UTC`, `RADAR`, and
242
+ `RAWMAG`), for which there's no file-wide way to pick the
243
+ "right" one without a line to disambiguate against. `read()`
244
+ already disambiguates this automatically using line context
245
+ (see `_resolve_channel_on_line`).
246
+
247
+ Pass `(name, occurrence)` instead of a plain name (`occurrence`
248
+ a 0-based index into every channel sharing that name, in
249
+ `.channels` order) to pick a specific one explicitly rather than
250
+ relying on that, or hitting the `ValueError` above.
251
+ """
252
+ if self._channels_by_name is None:
253
+ self.channels # populate the cache
254
+ if isinstance(name, tuple):
255
+ actual_name, occurrence = name
256
+ return self._nth_by_name(self._channels_by_name, actual_name, occurrence, "channel")
257
+ matches = self._channels_by_name.get(name)
258
+ if not matches:
259
+ raise KeyError(f"{self.path}: no channel named {name!r}")
260
+ if len(matches) > 1:
261
+ raise ValueError(
262
+ f"{self.path}: {len(matches)} channels are named {name!r} -- "
263
+ f"ambiguous without a line to disambiguate against; use "
264
+ f"read(line, name) (which resolves this using the line's "
265
+ f"own data), pass (name, occurrence) to pick a specific "
266
+ f"one explicitly, or pick a ChannelRecord from .channels "
267
+ f"yourself"
268
+ )
269
+ return matches[0]
270
+
271
+ def line(self, name: LineRef) -> LineRecord:
272
+ """
273
+ Look up a line by name. Raises `KeyError` if no line has this
274
+ name.
275
+
276
+ Raises `ValueError` if more than one line shares this name --
277
+ the line table has the same on-disk shape as the channel table
278
+ (see `channel()`'s docstring), with nothing in the format
279
+ forbidding a duplicate name there either; not yet observed on a
280
+ real file, but handled the same way on principle rather than
281
+ left as a silent last-one-wins lookup.
282
+
283
+ Pass `(name, occurrence)` instead of a plain name (`occurrence`
284
+ a 0-based index into every line sharing that name, in `.lines`
285
+ order) to pick a specific one explicitly rather than hitting
286
+ that `ValueError`.
287
+ """
288
+ if self._lines_by_name is None:
289
+ self.lines # populate the cache
290
+ if isinstance(name, tuple):
291
+ actual_name, occurrence = name
292
+ return self._nth_by_name(self._lines_by_name, actual_name, occurrence, "line")
293
+ matches = self._lines_by_name.get(name)
294
+ if not matches:
295
+ raise KeyError(f"{self.path}: no line named {name!r}")
296
+ if len(matches) > 1:
297
+ raise ValueError(
298
+ f"{self.path}: {len(matches)} lines are named {name!r} -- "
299
+ f"ambiguous; pass (name, occurrence) to pick a specific "
300
+ f"one explicitly, or pick a LineRecord from .lines yourself"
301
+ )
302
+ return matches[0]
303
+
304
+ def _resolve_line(self, line: LineRef) -> LineRecord:
305
+ return line if isinstance(line, LineRecord) else self.line(line)
306
+
307
+ # -- data access ---------------------------------------------------------
308
+
309
+ def _ensure_blob_index(self) -> Dict[Tuple[int, int], BlobHeader]:
310
+ """
311
+ Build the full (line_slot, channel_slot) -> BlobHeader map with one
312
+ blob-chain walk, cached from then on. `iter_blobs`/`find_blob`
313
+ themselves recommend this for anything beyond an occasional
314
+ one-off lookup -- this class always wants line/channel listings
315
+ and random-access reads, so it always builds the index.
316
+ """
317
+ if self._blob_index is None:
318
+ chans_max = self.chans_max
319
+ index: Dict[Tuple[int, int], BlobHeader] = {}
320
+ for blob in iter_blobs(self.path):
321
+ index[blob.line_channel(chans_max)] = blob
322
+ self._blob_index = index
323
+ self._calibrate_line_indices()
324
+ return self._blob_index
325
+
326
+ def _calibrate_line_indices(self) -> None:
327
+ """
328
+ Correct a possible small, fixed off-by-N in every LineRecord.index
329
+ (see find_line_table's and read_lines's docstrings in
330
+ gdb_reader.py) by checking, for a handful of small integer
331
+ shifts, which one makes the most already-found lines actually
332
+ have at least one real data blob on disk for *some* channel --
333
+ then applying the winning shift to every LineRecord.index in
334
+ place. This is a strictly stronger signal than anything available
335
+ from the symbol-table bytes alone (it's checking against the
336
+ real, self-describing blob chain, not another heuristic guess),
337
+ confirmed to fix a real off-by-one found on a GSQ file
338
+ (`rm001141`) without disturbing any of the other real files this
339
+ package has been tested against (where the winning shift is 0,
340
+ i.e. a no-op).
341
+
342
+ Runs once, right after the blob index is first built -- cheap
343
+ relative to that index build itself (already O(number of real
344
+ lines) additional work, not another file scan).
345
+ """
346
+ lines = self.lines
347
+ if not lines or not self._blob_index:
348
+ return
349
+ slots_with_data = {line_slot for line_slot, _channel_slot in self._blob_index}
350
+ best_offset, best_score = 0, -1
351
+ for offset in range(-4, 5):
352
+ score = sum(1 for l in lines if (l.index + offset) in slots_with_data)
353
+ if score > best_score:
354
+ best_score, best_offset = score, offset
355
+ if best_offset:
356
+ for l in lines:
357
+ l.index += best_offset
358
+
359
+ def _channels_with_data_on_line(self, line_rec: LineRecord) -> List[Tuple[ChannelRecord, BlobHeader]]:
360
+ """
361
+ `(channel, blob)` for every channel that actually has a real data
362
+ blob recorded for `line_rec`, in `self.channels` order. Shared by
363
+ `channels_on_line` and `iter_line` so both agree on exactly which
364
+ channel matched -- looking a channel back up by name afterward
365
+ would be ambiguous for a file with duplicate channel names (real
366
+ channel records aren't guaranteed unique by name), so callers
367
+ that need the actual data should go through this, not re-resolve
368
+ `channels_on_line`'s returned names.
369
+ """
370
+ index = self._ensure_blob_index()
371
+ return [
372
+ (c, blob) for c in self.channels
373
+ if (blob := index.get((line_rec.index, c.index))) is not None
374
+ and (blob.row_count is None or blob.row_count >= 0)
375
+ ]
376
+
377
+ def _resolve_channel_on_line(
378
+ self, line_rec: LineRecord, name: str
379
+ ) -> Tuple[ChannelRecord, Optional[BlobHeader]]:
380
+ """
381
+ Resolve a channel name to `(ChannelRecord, BlobHeader-or-None)`
382
+ for a specific line, using the line's own data to disambiguate a
383
+ name shared by more than one channel (see `channel()`'s
384
+ docstring) -- picking whichever same-named channel actually has
385
+ data on this line, rather than an arbitrary one. Raises
386
+ `KeyError` if no channel has this name at all.
387
+
388
+ If more than one same-named channel has data on this same line,
389
+ that's genuinely ambiguous (not just "the file happens to reuse
390
+ this name") and raises `ValueError` -- every real duplicate-name
391
+ case found so far has only one of the duplicates actually
392
+ populated per line, so this hasn't been observed, but there's no
393
+ principled way to guess if it ever is.
394
+ """
395
+ if self._channels_by_name is None:
396
+ self.channels # populate the cache
397
+ matches = self._channels_by_name.get(name)
398
+ if not matches:
399
+ raise KeyError(f"{self.path}: no channel named {name!r}")
400
+ if len(matches) == 1:
401
+ chan_rec = matches[0]
402
+ blob = self._ensure_blob_index().get((line_rec.index, chan_rec.index))
403
+ return chan_rec, blob
404
+ index = self._ensure_blob_index()
405
+ with_data = [
406
+ (c, blob) for c in matches
407
+ if (blob := index.get((line_rec.index, c.index))) is not None
408
+ and (blob.row_count is None or blob.row_count >= 0)
409
+ ]
410
+ if len(with_data) > 1:
411
+ raise ValueError(
412
+ f"{self.path}: {len(with_data)} channels named {name!r} all "
413
+ f"have data on line {line_rec.name!r} -- genuinely "
414
+ f"ambiguous even with line context; pass (name, occurrence) "
415
+ f"to pick a specific one explicitly (occurrence is a 0-based "
416
+ f"index into every channel named {name!r}, in .channels "
417
+ f"order), or pick a ChannelRecord yourself"
418
+ )
419
+ if with_data:
420
+ return with_data[0]
421
+ # None of the same-named channels have data on this line -- report
422
+ # "no data" the same way an unambiguous miss would, using the
423
+ # first match's ChannelRecord just to name it in the warning.
424
+ return matches[0], None
425
+
426
+ def channels_on_line(self, line: LineRef) -> List[str]:
427
+ """
428
+ Names of channels that actually have a real data blob recorded
429
+ for `line` -- the format stores a sparse (line, channel) grid
430
+ (docs/spec.md section 1), so most lines only populate a subset
431
+ of this file's full channel list. `line` may be a line name, a
432
+ `(name, occurrence)` pair (see `line()`), or a `LineRecord`.
433
+
434
+ If two channels share a name and both have data on this line,
435
+ that name appears twice here (a list, so nothing is silently
436
+ dropped) -- use `iter_line()` instead if you need the actual
437
+ `ChannelRecord` for each entry, not just its name.
438
+ """
439
+ line_rec = self._resolve_line(line)
440
+ return [c.name for c, _blob in self._channels_with_data_on_line(line_rec)]
441
+
442
+ def read(self, line: LineRef, channel: ChannelRef) -> np.ndarray:
443
+ """
444
+ Random access by name: decode and return every value recorded
445
+ for `channel` on `line`, as a numpy `ndarray` -- 1-D for an
446
+ ordinary scalar channel, 2-D `(n_rows, channel.array_width)`
447
+ for a VA/array channel (docs/spec.md section 5), dtype matching
448
+ the channel's `GS_*` type, or `object` (holding `str`) for a
449
+ string-typed channel. `line`/`channel` may be names,
450
+ `(name, occurrence)` pairs (see `line()`/`channel()`), or
451
+ `LineRecord`/`ChannelRecord` instances.
452
+
453
+ Raises `KeyError` if `line` or `channel` isn't a name this file
454
+ has. If `channel` is a plain name shared by more than one
455
+ channel (see `channel()`'s docstring), this resolves it using
456
+ `line`'s own data (`_resolve_channel_on_line`) rather than
457
+ picking an arbitrary one -- raising `ValueError` only if that's
458
+ *still* ambiguous (more than one same-named channel has data on
459
+ this exact line); pass `(name, occurrence)` or a specific
460
+ `ChannelRecord` to sidestep either lookup. Returns an empty
461
+ array (with a `GDBParseWarning`, per `read_blob_values`) if the
462
+ name is valid but this specific (line, channel) pair has no
463
+ data blob, or its data can't be decoded -- consistent with the
464
+ rest of this package's degrade-gracefully philosophy for
465
+ decode-time problems, as opposed to a plain lookup-by-name
466
+ mistake (which does raise).
467
+ """
468
+ line_rec = self._resolve_line(line)
469
+ if isinstance(channel, ChannelRecord):
470
+ chan_rec = channel
471
+ blob = self._ensure_blob_index().get((line_rec.index, chan_rec.index))
472
+ elif isinstance(channel, tuple):
473
+ chan_rec = self.channel(channel)
474
+ blob = self._ensure_blob_index().get((line_rec.index, chan_rec.index))
475
+ else:
476
+ chan_rec, blob = self._resolve_channel_on_line(line_rec, channel)
477
+ if blob is None:
478
+ warnings.warn(
479
+ f"{self.path}: no data blob for line {line_rec.name!r}, "
480
+ f"channel {chan_rec.name!r} -- this (line, channel) pair "
481
+ f"was likely never recorded (the format's grid is sparse, "
482
+ f"docs/spec.md section 1)",
483
+ GDBParseWarning, stacklevel=2,
484
+ )
485
+ return np.array([])
486
+ return read_blob_values(
487
+ self.path, blob, chan_rec,
488
+ comp_level=self.comp_level or 0, page_size=self.page_size,
489
+ file=self._file,
490
+ )
491
+
492
+ def iter_line(self, line: LineRef) -> Iterator[Tuple[ChannelRecord, np.ndarray]]:
493
+ """
494
+ Yield `(channel, values)` for every channel that actually has
495
+ data on `line`, via `_channels_with_data_on_line` directly
496
+ rather than one `read()` call per channel (saving the repeated
497
+ name lookups).
498
+
499
+ Yields the `ChannelRecord` itself, not just its name (`values`
500
+ is the same as `read(line, channel)` would give for that exact
501
+ channel) -- deliberately, so that if two channels share a name
502
+ and both have data on this line, both still come through as
503
+ distinct, fully-identified entries. `dict(db.iter_line(line))`
504
+ keyed by the records themselves preserves that; collapsing to
505
+ `channel.name` yourself reintroduces the same collision `read()`
506
+ raises on, so do that deliberately if you do it at all.
507
+
508
+ A `rayon`-based parallel batch decoder was tried for this and
509
+ removed: benchmarked against this project's real sample corpus,
510
+ it was consistently ~3x *slower* than plain sequential calls at
511
+ every scale tried, since this format's chunks are small enough
512
+ that the Rust decoder (`pygdb._native`) already finishes each one
513
+ in a fraction of a millisecond -- not enough work per chunk to
514
+ amortize rayon's per-task dispatch cost. See `rust/src/lib.rs`'s
515
+ module doc for the full note.
516
+ """
517
+ line_rec = self._resolve_line(line)
518
+ for c, blob in self._channels_with_data_on_line(line_rec):
519
+ values = read_blob_values(
520
+ self.path, blob, c,
521
+ comp_level=self.comp_level or 0, page_size=self.page_size,
522
+ file=self._file,
523
+ )
524
+ yield c, values
525
+
526
+ def to_xarray(self, line: LineRef) -> "xr.Dataset":
527
+ """
528
+ Build an `xarray.Dataset` for every channel that has data on
529
+ `line` -- one data variable per channel, sharing a common
530
+ `"station"` dimension. Needs the optional `xarray` dependency
531
+ (`pip install python-gdb[xarray]`), imported lazily here so
532
+ importing `pygdb` itself never requires it.
533
+
534
+ A VA/array channel (docs/spec.md section 5) gets its own
535
+ second dimension, `f"{name}_bin"` -- deliberately *not* shared
536
+ with any other array channel even when their `array_width`
537
+ happens to match (e.g. real `ISPD`/`ISPU` are both 512-wide in
538
+ a real USGS file): two channels having the same width is a
539
+ coincidence, not a guarantee they share a semantic axis. Align/
540
+ rename dimensions yourself afterward if you know two channels
541
+ genuinely do.
542
+
543
+ If two channels share a name and both have data on this line
544
+ (confirmed structurally possible -- see `channel()`'s
545
+ docstring -- though never yet observed with data on both), the
546
+ variable name for every occurrence after the first is
547
+ disambiguated as `f"{name}[{occurrence}]"`, `occurrence` being
548
+ the same 0-based index into every channel sharing that name (in
549
+ `.channels` order) that `channel()`'s `(name, occurrence)` form
550
+ uses -- so `ds["UTC[1]"]` and `db.channel(("UTC", 1))` refer to
551
+ the same channel. Raises a `GDBParseWarning` when this actually
552
+ triggers, since a caller not expecting a bracket-suffixed
553
+ variable name should be told why one showed up.
554
+
555
+ If channels on this line don't all decode to the same row
556
+ count (a truncated/corrupt file -- truncation only ever
557
+ shortens a channel, never lengthens it), the *shorter*
558
+ channel(s) keep their full (shorter) data rather than being cut
559
+ down further, or cutting the other channels down to match: a
560
+ short channel gets its own first dimension, `f"{name}_station"`,
561
+ instead of the shared `"station"` (the same "give it its own
562
+ dimension rather than lose data to fit one" principle as the
563
+ array-channel case above). Also raises a `GDBParseWarning`.
564
+
565
+ No channel is auto-promoted to a coordinate -- `"station"` is a
566
+ bare integer range index, and every channel (however
567
+ conventionally named) is a plain data variable; call
568
+ `ds.set_coords(...)` yourself if you want one -- no real file
569
+ names its channels consistently enough for this reader to
570
+ guess which one(s) you'd want without risking guessing wrong.
571
+ """
572
+ try:
573
+ import xarray as xr
574
+ except ImportError as e:
575
+ raise ImportError(
576
+ "to_xarray() needs the optional 'xarray' dependency -- "
577
+ "install with `pip install python-gdb[xarray]`"
578
+ ) from e
579
+
580
+ line_rec = self._resolve_line(line)
581
+ decoded = [
582
+ (c, read_blob_values(
583
+ self.path, blob, c,
584
+ comp_level=self.comp_level or 0, page_size=self.page_size,
585
+ file=self._file,
586
+ ))
587
+ for c, blob in self._channels_with_data_on_line(line_rec)
588
+ ]
589
+ if self._channels_by_name is None:
590
+ self.channels # populate the cache (for occurrence numbering)
591
+
592
+ station_length = max((len(values) for _c, values in decoded), default=0)
593
+ name_counts: Dict[str, int] = {}
594
+ for c, _values in decoded:
595
+ name_counts[c.name] = name_counts.get(c.name, 0) + 1
596
+
597
+ data_vars = {}
598
+ seen_so_far: Dict[str, int] = {}
599
+ for c, values in decoded:
600
+ seen_so_far[c.name] = seen_so_far.get(c.name, 0) + 1
601
+ if name_counts[c.name] > 1:
602
+ occurrence = self._channels_by_name[c.name].index(c)
603
+ var_name = c.name if seen_so_far[c.name] == 1 else f"{c.name}[{occurrence}]"
604
+ warnings.warn(
605
+ f"{self.path}: line {line_rec.name!r} has {name_counts[c.name]} "
606
+ f"channels named {c.name!r} with data -- using {var_name!r} "
607
+ f"for occurrence {occurrence} (pass (name, occurrence) to "
608
+ f"channel() for the same numbering)",
609
+ GDBParseWarning, stacklevel=2,
610
+ )
611
+ else:
612
+ var_name = c.name
613
+
614
+ if len(values) == station_length:
615
+ station_dim = "station"
616
+ else:
617
+ station_dim = f"{var_name}_station"
618
+ warnings.warn(
619
+ f"{self.path}: line {line_rec.name!r} channel {c.name!r} "
620
+ f"decoded {len(values)} row(s), expected {station_length} "
621
+ f"(the max across this line's channels) -- likely "
622
+ f"truncated; keeping its own {len(values)}-row dimension "
623
+ f"{station_dim!r} rather than cutting other channels down "
624
+ f"to match",
625
+ GDBParseWarning, stacklevel=2,
626
+ )
627
+
628
+ dims = (station_dim, f"{var_name}_bin") if c.is_array else (station_dim,)
629
+ attrs = {"type_name": c.type_name, "format_name": c.format_name}
630
+ if c.is_array:
631
+ attrs["array_basetype_name"] = c.array_basetype_name
632
+ data_vars[var_name] = (dims, values, attrs)
633
+
634
+ return xr.Dataset(
635
+ data_vars,
636
+ attrs={
637
+ "line_name": line_rec.name,
638
+ "line_category": line_rec.category_name,
639
+ "path": self.path,
640
+ },
641
+ )