python-gdb 0.1.0__cp312-abi3-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pygdb/__init__.py +60 -0
- pygdb/_native.pyd +0 -0
- pygdb/gdb.py +641 -0
- pygdb/gdb_reader.py +1368 -0
- pygdb/grd_reader.py +292 -0
- pygdb/lzrw1.py +323 -0
- pygdb/registry.py +104 -0
- python_gdb-0.1.0.dist-info/METADATA +183 -0
- python_gdb-0.1.0.dist-info/RECORD +12 -0
- python_gdb-0.1.0.dist-info/WHEEL +4 -0
- python_gdb-0.1.0.dist-info/licenses/LICENSE +21 -0
- python_gdb-0.1.0.dist-info/sboms/pygdb-native.cyclonedx.json +572 -0
pygdb/registry.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Best-effort extraction of coordinate-system (map projection) names from a
|
|
3
|
+
`.gdb` file's REG/IPJ "reserved/administrative" blob region.
|
|
4
|
+
|
|
5
|
+
[CONFIRMED] present and real on every one of 3 independent agencies this
|
|
6
|
+
project has files from; [UNKNOWN] full binary framing beyond the specific
|
|
7
|
+
name marker decoded here. See docs/spec.md sections 8-9 and
|
|
8
|
+
docs/provenance/notes.md sections 6.7-6.8 for the full derivation --
|
|
9
|
+
including the important caveat that REG/IPJ content is **not universal**:
|
|
10
|
+
several real files (mostly ones apparently never interactively opened in
|
|
11
|
+
Oasis montaj) have none at all, which is a real, patterned absence, not a
|
|
12
|
+
bug in this scan.
|
|
13
|
+
|
|
14
|
+
Administrative blobs (holding REG/IPJ metadata rather than real survey
|
|
15
|
+
channel data) are addressed via the same blob_index formula as ordinary
|
|
16
|
+
data (gdb_reader.BlobHeader.line_channel), but with an out-of-range
|
|
17
|
+
line_slot -- one past every real line this file's own line table
|
|
18
|
+
(gdb_reader.read_lines) actually has. This project's original prototype
|
|
19
|
+
(docs/provenance/scripts/reg_ipj_full_scan.py) used a hardcoded line_slot
|
|
20
|
+
threshold (700) tuned to its own sample corpus; this module instead
|
|
21
|
+
derives the threshold from each file's real line count, so it isn't
|
|
22
|
+
tied to one corpus's line-numbering conventions.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import re
|
|
28
|
+
import warnings
|
|
29
|
+
from typing import List, Optional
|
|
30
|
+
|
|
31
|
+
from .gdb_reader import GDBParseWarning, check_magic, header_fields, iter_blobs, read_lines
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _warn(msg: str) -> None:
|
|
35
|
+
warnings.warn(msg, GDBParseWarning, stacklevel=3)
|
|
36
|
+
|
|
37
|
+
# The confirmed micro-pattern for how a working projected-CRS name is
|
|
38
|
+
# introduced inside an IPJ-tagged administrative blob (docs/spec.md
|
|
39
|
+
# section 8): the 4-byte tag " JPI" (a space plus the tail of the literal
|
|
40
|
+
# string "IPJ" read across an alignment boundary), an int32 count (always
|
|
41
|
+
# seen as 1), then a NUL-terminated name string.
|
|
42
|
+
_IPJ_NAME_RE = re.compile(rb" JPI\x01\x00\x00\x00([\x20-\x7e]+)\x00")
|
|
43
|
+
|
|
44
|
+
# How many bytes of each candidate administrative blob to read looking for
|
|
45
|
+
# the IPJ name marker. [GUESS] -- generous relative to the marker's own
|
|
46
|
+
# size, matches the original prototype script.
|
|
47
|
+
_PROBE_SIZE = 2000
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def find_coordinate_systems(path: str, max_real_line_slot: Optional[int] = None) -> List[str]:
|
|
51
|
+
"""
|
|
52
|
+
Scan `path` for coordinate-system (map projection) names.
|
|
53
|
+
|
|
54
|
+
Returns a de-duplicated, order-of-discovery list of name strings (e.g.
|
|
55
|
+
`"WGS 84 / UTM zone 54S"`), or an empty list if none were found --
|
|
56
|
+
which is expected and normal for a real file with no REG/IPJ content
|
|
57
|
+
at all (docs/spec.md section 9), not necessarily a sign of a problem.
|
|
58
|
+
|
|
59
|
+
`max_real_line_slot` is the highest physical line-table slot index
|
|
60
|
+
that corresponds to a real survey line -- blobs whose `line_slot`
|
|
61
|
+
(decoded via `BlobHeader.line_channel`) is beyond this are treated as
|
|
62
|
+
"administrative" and probed for IPJ content. If not given, it's
|
|
63
|
+
derived by calling `read_lines(path)` (an extra table scan) and using
|
|
64
|
+
the highest slot index found there; pass it explicitly if you already
|
|
65
|
+
have that file's `read_lines()` result to avoid repeating the scan.
|
|
66
|
+
|
|
67
|
+
Fails gracefully like the rest of this package: a bad magic or
|
|
68
|
+
truncated header returns `[]` with a `GDBParseWarning` rather than
|
|
69
|
+
raising.
|
|
70
|
+
"""
|
|
71
|
+
with open(path, "rb") as f:
|
|
72
|
+
header = f.read(128)
|
|
73
|
+
if not check_magic(header):
|
|
74
|
+
_warn(f"{path}: does not start with the expected '!CBD' magic -- "
|
|
75
|
+
f"no coordinate systems")
|
|
76
|
+
return []
|
|
77
|
+
fields = header_fields(header)
|
|
78
|
+
chans_max = fields["chans_max"]
|
|
79
|
+
if chans_max is None:
|
|
80
|
+
_warn(f"{path}: header too short to read chans_max -- "
|
|
81
|
+
f"no coordinate systems")
|
|
82
|
+
return []
|
|
83
|
+
|
|
84
|
+
if max_real_line_slot is None:
|
|
85
|
+
lines = read_lines(path)
|
|
86
|
+
max_real_line_slot = max((line.index for line in lines), default=-1)
|
|
87
|
+
|
|
88
|
+
names: List[str] = []
|
|
89
|
+
seen = set()
|
|
90
|
+
with open(path, "rb") as f:
|
|
91
|
+
for blob in iter_blobs(path):
|
|
92
|
+
line_slot, _channel_slot = blob.line_channel(chans_max)
|
|
93
|
+
if line_slot <= max_real_line_slot:
|
|
94
|
+
continue # a real survey line's data, not administrative metadata
|
|
95
|
+
f.seek(blob.offset)
|
|
96
|
+
chunk = f.read(_PROBE_SIZE)
|
|
97
|
+
m = _IPJ_NAME_RE.search(chunk)
|
|
98
|
+
if not m:
|
|
99
|
+
continue
|
|
100
|
+
name = m.group(1).decode("ascii", errors="replace")
|
|
101
|
+
if name not in seen:
|
|
102
|
+
seen.add(name)
|
|
103
|
+
names.append(name)
|
|
104
|
+
return names
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: python-gdb
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Classifier: Development Status :: 3 - Alpha
|
|
5
|
+
Classifier: Intended Audience :: Science/Research
|
|
6
|
+
Classifier: Topic :: Scientific/Engineering :: GIS
|
|
7
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Dist: numpy>=2.2
|
|
14
|
+
Requires-Dist: pytest>=7.0 ; extra == 'dev'
|
|
15
|
+
Requires-Dist: xarray ; extra == 'dev'
|
|
16
|
+
Requires-Dist: zensical ; extra == 'docs'
|
|
17
|
+
Requires-Dist: xarray ; extra == 'xarray'
|
|
18
|
+
Provides-Extra: dev
|
|
19
|
+
Provides-Extra: docs
|
|
20
|
+
Provides-Extra: xarray
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Summary: A clean-room Python reader for Geosoft .gdb database and .grd grid files
|
|
23
|
+
Author-email: Joseph Capriotti <josephrcapriotti@gmail.com>
|
|
24
|
+
License-Expression: MIT
|
|
25
|
+
Requires-Python: >=3.12
|
|
26
|
+
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
27
|
+
|
|
28
|
+
# pygdb
|
|
29
|
+
|
|
30
|
+
A clean-room Python reader for Geosoft's proprietary `.gdb` ("Geosoft
|
|
31
|
+
Database") binary format and the sibling `.grd` grid format.
|
|
32
|
+
|
|
33
|
+
> [!IMPORTANT]
|
|
34
|
+
> This project has **no affiliation with, and is not endorsed,
|
|
35
|
+
> sponsored, or certified by, Geosoft Inc., Seequent, Bentley Systems,
|
|
36
|
+
> or any of their successors.** "Geosoft" and "Oasis montaj" are
|
|
37
|
+
> trademarks of their respective owners. This is an independent,
|
|
38
|
+
> third-party implementation produced entirely by clean-room means —
|
|
39
|
+
> see [Provenance](docs/provenance/index.md) for the full research
|
|
40
|
+
> trail.
|
|
41
|
+
|
|
42
|
+
It was built using only publicly available information: vendor-published
|
|
43
|
+
open-source code and documentation, independent third-party format
|
|
44
|
+
readers, one openly-specified independent successor format (`.geoh5`,
|
|
45
|
+
read via the third-party `geoh5py` library), and byte-level analysis of
|
|
46
|
+
real, publicly downloaded `.gdb` files. No Geosoft software of any kind
|
|
47
|
+
(the `geosoft`/`gxapi`/`gxpy` compiled package, Oasis montaj, Geosoft
|
|
48
|
+
Desktop, or the free Geosoft Viewer) was installed, imported, or
|
|
49
|
+
executed at any point in producing it.
|
|
50
|
+
|
|
51
|
+
Reading only — writing or mutating `.gdb`/`.grd` files is out of scope.
|
|
52
|
+
|
|
53
|
+
## Installation
|
|
54
|
+
|
|
55
|
+
```sh
|
|
56
|
+
pip install python-gdb
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
The distribution is named `python-gdb` on PyPI (`pygdb` was already
|
|
60
|
+
registered there for an unrelated project), but the importable package
|
|
61
|
+
is `pygdb`. On a platform/Python version this project publishes a
|
|
62
|
+
prebuilt wheel for, that automatically includes the optional Rust
|
|
63
|
+
accelerator (see below) -- nothing extra to install or configure.
|
|
64
|
+
Elsewhere, `pip` falls back to building from source, which needs a Rust
|
|
65
|
+
toolchain. `numpy` is the one required dependency (used to return
|
|
66
|
+
correctly-shaped arrays -- see Quick start below); everything else is
|
|
67
|
+
optional.
|
|
68
|
+
|
|
69
|
+
## Quick start
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from pygdb import GDB
|
|
73
|
+
|
|
74
|
+
db = GDB("example.gdb")
|
|
75
|
+
|
|
76
|
+
db.compression # CompressionInfo(code=0, name='DB_COMP_NONE', ...)
|
|
77
|
+
db.coordinate_systems # ['NAD83 / UTM zone 11N', 'WGS 84'] (best-effort, may be [])
|
|
78
|
+
|
|
79
|
+
db.line_names[:5] # ['L1000', 'L1001', 'L1010', 'L1020', 'L1030']
|
|
80
|
+
db.channels_on_line("L1000") # channels that actually have data on this line
|
|
81
|
+
db.read("L1000", "Easting") # random access by (line name, channel name)
|
|
82
|
+
# -> ndarray, shape (n_rows,) for a scalar
|
|
83
|
+
# channel, (n_rows, array_width) for a
|
|
84
|
+
# VA/array channel (docs/spec.md section 5)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
See [the docs](docs/index.md) for the lower-level, slot-index-based
|
|
88
|
+
functions `GDB` is built on.
|
|
89
|
+
|
|
90
|
+
`db.to_xarray("L1000")` exports one line to an `xarray.Dataset`
|
|
91
|
+
(`pip install python-gdb[xarray]`, an optional dependency) — see
|
|
92
|
+
[the docs](docs/index.md#exporting-to-xarray) for how VA/array channels
|
|
93
|
+
and duplicate channel names come through.
|
|
94
|
+
|
|
95
|
+
## Optional Rust-accelerated backend
|
|
96
|
+
|
|
97
|
+
The pure-Python code in `pygdb/` is always the reference implementation
|
|
98
|
+
and always fully correct and usable on its own — nothing here depends
|
|
99
|
+
on Rust. This package is built with [`maturin`](https://www.maturin.rs/)
|
|
100
|
+
so that an optional Rust extension (`pygdb._native`, source under
|
|
101
|
+
[`rust/`](rust/)) rides along and is used automatically when present:
|
|
102
|
+
it accelerates the two real CPU-bound hot paths profiling found in this
|
|
103
|
+
reader — LZRW1 decompression and fixed-width string decoding — roughly
|
|
104
|
+
3-6x on real files, measured against this project's own sample corpus.
|
|
105
|
+
`pygdb/lzrw1.py`/`pygdb/gdb_reader.py` detect it at import time and fall
|
|
106
|
+
back to plain Python transparently if it isn't there.
|
|
107
|
+
|
|
108
|
+
Building it yourself (e.g. for local development, or a platform without
|
|
109
|
+
a published wheel) needs a Rust toolchain:
|
|
110
|
+
|
|
111
|
+
```sh
|
|
112
|
+
pip install -e ".[dev]" # compiles pygdb._native as part of the install
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
or, for a release-optimized build without an editable install:
|
|
116
|
+
|
|
117
|
+
```sh
|
|
118
|
+
pip install maturin
|
|
119
|
+
maturin build --release --manifest-path rust/Cargo.toml
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
See [`rust/src/lib.rs`](rust/src/lib.rs) for what's implemented (and,
|
|
123
|
+
just as importantly, what was tried and deliberately left out after
|
|
124
|
+
being benchmarked as not worth it — a parallel batch decoder and a
|
|
125
|
+
memory-mapped-file I/O path, both documented there with real numbers),
|
|
126
|
+
and [`.github/workflows/wheels.yml`](.github/workflows/wheels.yml) for
|
|
127
|
+
the released wheel matrix: an `abi3` wheel per platform covering every
|
|
128
|
+
non-free-threaded CPython ≥3.12, an `abi3.abi3t` wheel per platform
|
|
129
|
+
(PEP 803) covering both the GIL-enabled and free-threaded builds of
|
|
130
|
+
CPython ≥3.15 with a single wheel, and one version-specific wheel per
|
|
131
|
+
platform for 3.14's free-threaded build (`3.14t` has no stable-ABI
|
|
132
|
+
option — `abi3t` only exists from 3.15 onward).
|
|
133
|
+
|
|
134
|
+
## Documentation
|
|
135
|
+
|
|
136
|
+
Full documentation — the format specification, usage, contributing
|
|
137
|
+
guide, and research provenance — lives under [`docs/`](docs/index.md)
|
|
138
|
+
and is built with [Zensical](https://zensical.org/). To view it
|
|
139
|
+
locally:
|
|
140
|
+
|
|
141
|
+
```sh
|
|
142
|
+
pip install -e ".[docs]"
|
|
143
|
+
zensical serve
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
- [`docs/spec.md`](docs/spec.md) — the living reference specification
|
|
147
|
+
for the `.gdb`/`.grd` on-disk format, with a confidence rating
|
|
148
|
+
(confirmed / likely / guess / unknown) on every field. Updated as
|
|
149
|
+
more real example files are tested against it.
|
|
150
|
+
- [`docs/contributing.md`](docs/contributing.md) — how to report bugs
|
|
151
|
+
(reproducible example files welcome) and this project's hard
|
|
152
|
+
boundary on reverse engineering.
|
|
153
|
+
- [`docs/provenance/`](docs/provenance/index.md) — the original
|
|
154
|
+
research log and write-up this implementation was derived from.
|
|
155
|
+
|
|
156
|
+
## Testing
|
|
157
|
+
|
|
158
|
+
```sh
|
|
159
|
+
pip install -e ".[dev]"
|
|
160
|
+
pytest
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
Most of the suite is synthetic-fixture unit tests (`tests/test_*.py`,
|
|
164
|
+
minus `test_integration_samples.py`) that build minimal `.gdb`/`.grd`
|
|
165
|
+
byte layouts by hand — no sample data required, safe to run anywhere
|
|
166
|
+
including CI. `tests/test_integration_samples.py` additionally
|
|
167
|
+
cross-checks the reader against real files in a local, gitignored
|
|
168
|
+
`samples/` directory when present, and skips (not fails) when it's
|
|
169
|
+
absent.
|
|
170
|
+
|
|
171
|
+
## Sample data
|
|
172
|
+
|
|
173
|
+
Real `.gdb`/`.grd` sample files used during development are **not**
|
|
174
|
+
committed to this repository or otherwise redistributed — we don't
|
|
175
|
+
necessarily have redistribution rights for them. See
|
|
176
|
+
[`docs/provenance/notes.md`](docs/provenance/notes.md) for exact
|
|
177
|
+
provenance (source URL/DOI/portal, size) for every sample used, so they
|
|
178
|
+
can be re-downloaded independently.
|
|
179
|
+
|
|
180
|
+
## License
|
|
181
|
+
|
|
182
|
+
[MIT](LICENSE) — Copyright (c) 2026 Joseph Capriotti.
|
|
183
|
+
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
pygdb/__init__.py,sha256=0srdABOKNkLpKi0CJFa0vpGaLL8_kc17U2afcTVzRDM,1637
|
|
2
|
+
pygdb/_native.pyd,sha256=T3mim3jz-dDqoO9wpsJILUBf75sRQQ3YtDh6VLU6PNM,186368
|
|
3
|
+
pygdb/gdb.py,sha256=MDM5zcTNSdobFlSFEC1AqW0hTDKBmTC8_AUewML8xZM,29436
|
|
4
|
+
pygdb/gdb_reader.py,sha256=nwYpNdurB9oPmLh0kF_BGl0yqnyzjOc0sYNvOCCy4fs,65656
|
|
5
|
+
pygdb/grd_reader.py,sha256=pabp9vVW1v-leM9rfZowzV2DGn70AboSoQKV4Zeprzs,11655
|
|
6
|
+
pygdb/lzrw1.py,sha256=Hjne6ClmgXNyHGsN0Oel2c_yLLgE_uoAML9D3DUHVPk,14612
|
|
7
|
+
pygdb/registry.py,sha256=74cLbC9k9dpWYN0sOKBwUENymhu8LrtbuoIOdVrQhVs,4637
|
|
8
|
+
python_gdb-0.1.0.dist-info/METADATA,sha256=j9hTn_rLwIBsYVibUUm-vY1XJ5t2YjpBkO2X0eCSqVY,7635
|
|
9
|
+
python_gdb-0.1.0.dist-info/WHEEL,sha256=rjikdQI6uHLAgYXPVkr7bTnnX2G_U4MjYuRag5dSWZs,96
|
|
10
|
+
python_gdb-0.1.0.dist-info/licenses/LICENSE,sha256=PKGLyWd7bIQHdkI3zmIF0jpKUlD01tpRdmvNQ7hAlTI,1094
|
|
11
|
+
python_gdb-0.1.0.dist-info/sboms/pygdb-native.cyclonedx.json,sha256=DhLZJaU5Rimv09TbQKSWDEOURFmTplXLsxvAUbgzTIE,17316
|
|
12
|
+
python_gdb-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Joseph Capriotti
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|