DeepGPR 0.0.18__tar.gz → 0.0.20__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- deepgpr-0.0.20/MANIFEST.in +19 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/PKG-INFO +59 -3
- {deepgpr-0.0.18 → deepgpr-0.0.20}/README.md +58 -2
- {deepgpr-0.0.18 → deepgpr-0.0.20}/pyproject.toml +1 -1
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/__init__.py +23 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/compute2.py +515 -35
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr.cu +873 -77
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr.dll +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr.h +6 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr.so +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr_cpu.c +6 -4
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr_cpu.dll +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr_cpu.dylib +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/deepgpr_cpu.so +0 -0
- deepgpr-0.0.20/src/DeepGPR/requirements.txt +1 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR.egg-info/PKG-INFO +59 -3
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR.egg-info/SOURCES.txt +3 -4
- deepgpr-0.0.18/tests/test_discrete_adjoint.py +0 -530
- deepgpr-0.0.18/tests/test_numerics.py +0 -643
- deepgpr-0.0.18/tests/test_wavelet.py +0 -135
- {deepgpr-0.0.18 → deepgpr-0.0.20}/license +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/setup.cfg +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/common.py +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/lib/libomp.dylib +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/multiscale.py +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR/wavelet.py +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR.egg-info/dependency_links.txt +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR.egg-info/requires.txt +0 -0
- {deepgpr-0.0.18 → deepgpr-0.0.20}/src/DeepGPR.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
include README.md
|
|
2
|
+
include license
|
|
3
|
+
include pyproject.toml
|
|
4
|
+
|
|
5
|
+
recursive-include src/DeepGPR *.py *.txt *.cu *.h *.c *.dll *.so *.dylib
|
|
6
|
+
|
|
7
|
+
prune .agents
|
|
8
|
+
prune .codex
|
|
9
|
+
prune .github
|
|
10
|
+
prune Fig
|
|
11
|
+
prune docs
|
|
12
|
+
prune examples
|
|
13
|
+
prune scripts
|
|
14
|
+
prune tests
|
|
15
|
+
prune website
|
|
16
|
+
|
|
17
|
+
global-exclude .DS_Store
|
|
18
|
+
global-exclude *.py[cod]
|
|
19
|
+
global-exclude __pycache__
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: DeepGPR
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.20
|
|
4
4
|
Summary: PyTorch and CUDA for GPR FWI
|
|
5
5
|
Author-email: Lei Liu <liulei990222@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -23,6 +23,8 @@ Dynamic: license-file
|
|
|
23
23
|
|
|
24
24
|
# DeepGPR
|
|
25
25
|
|
|
26
|
+
**Official Website / Documentation:** [https://songc0a.github.io/DeepGPR/](https://songc0a.github.io/DeepGPR/)
|
|
27
|
+
|
|
26
28
|
DeepGPR provides a wave propagation module for PyTorch, designed for applications such as Ground Penetrating Radar (GPR) imaging and inversion. Its core concepts are derived from Deepwave. You can use it to perform both forward modeling and backpropagation—thereby enabling the simulation of wave propagation to generate synthetic data—as well as for Full Waveform Inversion (FWI). Furthermore, you can integrate this wave propagation functionality into a larger operational pipeline—incorporating various wavelets, loss functions, and other components—to achieve end-to-end forward and reverse propagation, powered by automatic differentiation and our high-performance operators.
|
|
27
29
|
|
|
28
30
|
|
|
@@ -241,7 +243,13 @@ This section defines the geometric observation system (coordinates) and the exci
|
|
|
241
243
|
| :--- | :--- | :--- | :--- |
|
|
242
244
|
| **`pmlthick`** | `int` / `list` / `Tensor`| Scalar or list of 6 | Thickness (in grid layers) of the PML (Perfectly Matched Layer) absorbing boundaries.<br>- Integer `p`: All six boundaries have thickness `p` (Z-boundaries are ignored in 2D).<br>- List `[x0, xm, y0, ym, z0, zm]`: Specific thicknesses for the 6 boundaries. |
|
|
243
245
|
| **`model_gradient_sampling_interval`**| `int` | Scalar | Wavefield sampling interval during forward propagation (Default: 1).<br>A larger integer reduces VRAM use for `E_saved` and `R_saved`, but uses an explicitly approximate model gradient. The last incomplete sampling block is weighted by its actual length. |
|
|
246
|
+
| **`save_wavefield_history`** | `bool` | Scalar | Independently controls allocation and native writes of the E/R histories used by adjoint model-gradient backward (Default: `True`). `False` still executes the complete FDTD, CPML, source-injection, and receiver-recording path, returns an empty `E_saved`, and raises a clear error if history-dependent backward is attempted. |
|
|
244
247
|
| **`wavefield_storage_dtype`** | `torch.dtype` / `str` | `float32`, `float16`, or `bfloat16` | Storage format for saved `E_saved` and `R_saved` model-gradient wavefields. FDTD propagation remains float32. `float16` and `bfloat16` halve saved-wavefield memory at the cost of gradient accuracy; `bfloat16` has the safer dynamic range. String aliases such as `"fp16"` and `"bf16"` are accepted. |
|
|
248
|
+
| **`wavefield_conversion_backend`** | `str` | `"auto"`, `"legacy"`, `"native_scalar"`, or `"native_vec2"` | CUDA FP16/BF16 history conversion. The audited default `"auto"` uses NVIDIA scalar intrinsics for CUDA FP16 and the legacy path otherwise. Explicit values retain the correctness/performance A/B paths; vec2 is not the default because it was slower on RTX 4090. |
|
|
249
|
+
| **`wavefield_compression`** | `str` | `"none"`, `"int8"`, or `"zfp"` | `"int8"` enables CUDA-native per-block symmetric INT8 histories with FP32 scales and inline decode in the fused material-gradient kernel. `"none"` is the unchanged default. `"zfp"` is reserved for an optional fused CUDA decoder and is rejected by the current dependency-free build instead of silently materializing a decoded global-memory history. |
|
|
250
|
+
| **`wavefield_compression_block_size`** | sequence / `None` | 2D or 3D spatial block | INT8 block shape; defaults to `(8, 8)` in 2D and `(4, 4, 4)` in 3D. The block volume must be a power of two no larger than 256. Partial boundary blocks are supported. |
|
|
251
|
+
| **`int8_reduction_backend`** | `str` | `"auto"`, `"current"`, `"cub_block"`, or `"warp_shuffle"` | Tile maximum reduction. The audited default `"auto"` selects NVIDIA CUB `BlockReduce` for 64-voxel tiles and preserves the prior shared-memory tree for other valid tile sizes. Explicit values expose the retained A/B implementations. |
|
|
252
|
+
| **`wavefield_compression_rate`** | scalar / `None` | Optional ZFP setting | Reserved for an optional ZFP backend. It is rejected unless that backend is selected and available. |
|
|
245
253
|
| **`use_async_offload`** | `bool` | Scalar | CUDA-only VRAM optimization flag (Default: `False`).<br>If `True`, `E_saved` and `R_saved` are asynchronously offloaded to page-locked host memory (`pin_memory` CPU RAM). This reduces GPU VRAM consumption at the cost of PCIe transfers. On CPU this option is ignored. |
|
|
246
254
|
|
|
247
255
|
### 4.1 FWI Gradient Mode
|
|
@@ -259,8 +267,55 @@ The backward solver applies the exact reverse-mode transpose of each executed op
|
|
|
259
267
|
|
|
260
268
|
CPML is treated as a fixed numerical boundary. Its boundary material averages are explicitly detached, and CPML cells are excluded from the returned material gradients. This separation must be retained when optimizing a model.
|
|
261
269
|
|
|
270
|
+
The material-gradient formulation in DeepGPR was informed in part by the differentiable FDTD implementation in [TIDE](https://github.com/Vcholerae1/tide-GPR), particularly its treatment of the discrete Maxwell electric-field update in gradient computation. We gratefully acknowledge the TIDE project and its authors for this work.
|
|
271
|
+
|
|
262
272
|
Use `model_gradient_sampling_interval=1` and `wavefield_storage_dtype=torch.float32` for a directional derivative check in the physical model region. Temporal subsampling and lower-precision storage deliberately approximate the gradient. Run [the gradient-check notebook](examples/4.GradientCheck.ipynb) after rebuilding the native ABI 6 libraries.
|
|
263
273
|
|
|
274
|
+
### 4.3 GPU-native block INT8 history
|
|
275
|
+
|
|
276
|
+
Only the sampled forward state used by the material adjoint is compressed. The
|
|
277
|
+
live Ex/Ey/Ez/Hx/Hy/Hz and CPML states remain float32. The executed discrete
|
|
278
|
+
update requires `E^n` and `R^n`, where `E^(n+1) = ca E^n + cb R^n`; consequently
|
|
279
|
+
`mode=2` stores compressed Ez/Rz and `mode=3` stores compressed Ex/Ey/Ez and all
|
|
280
|
+
three corresponding RHS components. Magnetic histories are not stored.
|
|
281
|
+
|
|
282
|
+
The packed tensor contains a contiguous signed-INT8 payload followed by a
|
|
283
|
+
four-byte-aligned contiguous FP32 scale array. Backward maps one CUDA block to
|
|
284
|
+
one compression tile, loads each E/R scale once into shared memory, decodes each
|
|
285
|
+
value in a register, and immediately accumulates the epsilon/conductivity
|
|
286
|
+
gradient. It does not allocate or write a reconstructed global-memory history.
|
|
287
|
+
`use_async_offload=True`, CPU execution, and a non-float32
|
|
288
|
+
`wavefield_storage_dtype` are explicitly incompatible with `"int8"`.
|
|
289
|
+
|
|
290
|
+
The current shared-memory maximum reduction remains available as
|
|
291
|
+
`int8_reduction_backend="current"`. The RTX 4090 audit selected CUB
|
|
292
|
+
`BlockReduce` for the default 64-voxel tiles; `"auto"` falls back to the current
|
|
293
|
+
tree for other supported power-of-two tile volumes.
|
|
294
|
+
|
|
295
|
+
`E_saved` is an opaque one-dimensional `torch.int8` packed tensor in this mode.
|
|
296
|
+
For diagnostics only, reconstruct it with
|
|
297
|
+
`DeepGPR.decompress_wavefield_history(E_saved, original_shape, block_size)`.
|
|
298
|
+
This helper materializes FP32 and is never called by autograd backward.
|
|
299
|
+
|
|
300
|
+
## CUDA Backend Build
|
|
301
|
+
|
|
302
|
+
Build the CUDA shared library from the repository root on Linux. `-lineinfo`
|
|
303
|
+
preserves source correlation for Nsight; `-Xptxas=-v` prints registers and spill
|
|
304
|
+
stores/loads for each kernel.
|
|
305
|
+
|
|
306
|
+
```bash
|
|
307
|
+
nvcc -std=c++14 -O3 -lineinfo -Xptxas=-v -arch=sm_89 --shared -Xcompiler -fPIC \
|
|
308
|
+
-o src/DeepGPR/lib/deepgpr.so src/DeepGPR/lib/deepgpr.cu
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
The command above is the audited RTX 4090 build path. Select the matching
|
|
312
|
+
architecture when building for a different GPU; do not add `--use_fast_math`
|
|
313
|
+
unless a separate numerical and gradient validation explicitly permits it.
|
|
314
|
+
|
|
315
|
+
The new INT8 path deliberately keeps native ABI 6 because the forward/backward
|
|
316
|
+
C signatures are unchanged. A capability symbol prevents an older ABI-6 CUDA
|
|
317
|
+
library from accepting the packed storage code and corrupting memory.
|
|
318
|
+
|
|
264
319
|
## CPU Backend Build
|
|
265
320
|
|
|
266
321
|
The CPU backend is a plain C shared library and is built with OpenMP by default. Build it into `src/DeepGPR/lib` before running with `device='cpu'`. You can control CPU thread count with `OMP_NUM_THREADS`.
|
|
@@ -309,11 +364,12 @@ return E_saved, (Ex, Ey, Ez), (Hx, Hy, Hz), (x0EPhi1...zmHPhi2), receiver_amplit
|
|
|
309
364
|
```
|
|
310
365
|
|
|
311
366
|
1. **`E_saved`**: The pre-update electric field history `E^n` saved for gradient calculation and diagnostics. An internal `R_saved` tensor stores the corresponding discrete right-hand side `R^n`.
|
|
367
|
+
* With `save_wavefield_history=False`, `E_saved` is a zero-length tensor and neither E nor R history storage/compression kernels are launched.
|
|
312
368
|
* **Shape when `mode=2`**: `(nt_saved, nstep, nx, ny, nz)`, storing Ez only.
|
|
313
369
|
* **Shape when `mode=3`**: `(3, nt_saved, nstep, nx, ny, nz)`, storing components in `[Ex, Ey, Ez]` order.
|
|
314
370
|
* `nt_saved` depends on `nt` and `model_gradient_sampling_interval`.
|
|
315
|
-
* Dtype is selected by `wavefield_storage_dtype
|
|
316
|
-
* Set `save_forward_wavefield_path="/path/to/output"` to save
|
|
371
|
+
* Dtype is selected by `wavefield_storage_dtype` when compression is disabled. With `wavefield_compression="int8"`, this is an opaque packed one-dimensional `torch.int8` tensor containing values and FP32 scales.
|
|
372
|
+
* Set `save_forward_wavefield_path="/path/to/output"` to save a CPU-loadable `.pt` file. Uncompressed modes save the tensor directly. INT8 mode saves a dictionary containing `wavefield`, `compression`, `block_size`, and `uncompressed_shape` so diagnostics can reconstruct it safely.
|
|
317
373
|
2. **`(Ex, Ey, Ez)`**: The 3D electric field state at the final time step.
|
|
318
374
|
3. **`(Hx, Hy, Hz)`**: The 3D magnetic field state at the final time step.
|
|
319
375
|
4. **`(PML_Tuple)`**: A tuple of 24 Tensors recording the final time step state of the PML auxiliary $\Phi$ variables.
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# DeepGPR
|
|
2
2
|
|
|
3
|
+
**Official Website / Documentation:** [https://songc0a.github.io/DeepGPR/](https://songc0a.github.io/DeepGPR/)
|
|
4
|
+
|
|
3
5
|
DeepGPR provides a wave propagation module for PyTorch, designed for applications such as Ground Penetrating Radar (GPR) imaging and inversion. Its core concepts are derived from Deepwave. You can use it to perform both forward modeling and backpropagation—thereby enabling the simulation of wave propagation to generate synthetic data—as well as for Full Waveform Inversion (FWI). Furthermore, you can integrate this wave propagation functionality into a larger operational pipeline—incorporating various wavelets, loss functions, and other components—to achieve end-to-end forward and reverse propagation, powered by automatic differentiation and our high-performance operators.
|
|
4
6
|
|
|
5
7
|
|
|
@@ -218,7 +220,13 @@ This section defines the geometric observation system (coordinates) and the exci
|
|
|
218
220
|
| :--- | :--- | :--- | :--- |
|
|
219
221
|
| **`pmlthick`** | `int` / `list` / `Tensor`| Scalar or list of 6 | Thickness (in grid layers) of the PML (Perfectly Matched Layer) absorbing boundaries.<br>- Integer `p`: All six boundaries have thickness `p` (Z-boundaries are ignored in 2D).<br>- List `[x0, xm, y0, ym, z0, zm]`: Specific thicknesses for the 6 boundaries. |
|
|
220
222
|
| **`model_gradient_sampling_interval`**| `int` | Scalar | Wavefield sampling interval during forward propagation (Default: 1).<br>A larger integer reduces VRAM use for `E_saved` and `R_saved`, but uses an explicitly approximate model gradient. The last incomplete sampling block is weighted by its actual length. |
|
|
223
|
+
| **`save_wavefield_history`** | `bool` | Scalar | Independently controls allocation and native writes of the E/R histories used by adjoint model-gradient backward (Default: `True`). `False` still executes the complete FDTD, CPML, source-injection, and receiver-recording path, returns an empty `E_saved`, and raises a clear error if history-dependent backward is attempted. |
|
|
221
224
|
| **`wavefield_storage_dtype`** | `torch.dtype` / `str` | `float32`, `float16`, or `bfloat16` | Storage format for saved `E_saved` and `R_saved` model-gradient wavefields. FDTD propagation remains float32. `float16` and `bfloat16` halve saved-wavefield memory at the cost of gradient accuracy; `bfloat16` has the safer dynamic range. String aliases such as `"fp16"` and `"bf16"` are accepted. |
|
|
225
|
+
| **`wavefield_conversion_backend`** | `str` | `"auto"`, `"legacy"`, `"native_scalar"`, or `"native_vec2"` | CUDA FP16/BF16 history conversion. The audited default `"auto"` uses NVIDIA scalar intrinsics for CUDA FP16 and the legacy path otherwise. Explicit values retain the correctness/performance A/B paths; vec2 is not the default because it was slower on RTX 4090. |
|
|
226
|
+
| **`wavefield_compression`** | `str` | `"none"`, `"int8"`, or `"zfp"` | `"int8"` enables CUDA-native per-block symmetric INT8 histories with FP32 scales and inline decode in the fused material-gradient kernel. `"none"` is the unchanged default. `"zfp"` is reserved for an optional fused CUDA decoder and is rejected by the current dependency-free build instead of silently materializing a decoded global-memory history. |
|
|
227
|
+
| **`wavefield_compression_block_size`** | sequence / `None` | 2D or 3D spatial block | INT8 block shape; defaults to `(8, 8)` in 2D and `(4, 4, 4)` in 3D. The block volume must be a power of two no larger than 256. Partial boundary blocks are supported. |
|
|
228
|
+
| **`int8_reduction_backend`** | `str` | `"auto"`, `"current"`, `"cub_block"`, or `"warp_shuffle"` | Tile maximum reduction. The audited default `"auto"` selects NVIDIA CUB `BlockReduce` for 64-voxel tiles and preserves the prior shared-memory tree for other valid tile sizes. Explicit values expose the retained A/B implementations. |
|
|
229
|
+
| **`wavefield_compression_rate`** | scalar / `None` | Optional ZFP setting | Reserved for an optional ZFP backend. It is rejected unless that backend is selected and available. |
|
|
222
230
|
| **`use_async_offload`** | `bool` | Scalar | CUDA-only VRAM optimization flag (Default: `False`).<br>If `True`, `E_saved` and `R_saved` are asynchronously offloaded to page-locked host memory (`pin_memory` CPU RAM). This reduces GPU VRAM consumption at the cost of PCIe transfers. On CPU this option is ignored. |
|
|
223
231
|
|
|
224
232
|
### 4.1 FWI Gradient Mode
|
|
@@ -236,8 +244,55 @@ The backward solver applies the exact reverse-mode transpose of each executed op
|
|
|
236
244
|
|
|
237
245
|
CPML is treated as a fixed numerical boundary. Its boundary material averages are explicitly detached, and CPML cells are excluded from the returned material gradients. This separation must be retained when optimizing a model.
|
|
238
246
|
|
|
247
|
+
The material-gradient formulation in DeepGPR was informed in part by the differentiable FDTD implementation in [TIDE](https://github.com/Vcholerae1/tide-GPR), particularly its treatment of the discrete Maxwell electric-field update in gradient computation. We gratefully acknowledge the TIDE project and its authors for this work.
|
|
248
|
+
|
|
239
249
|
Use `model_gradient_sampling_interval=1` and `wavefield_storage_dtype=torch.float32` for a directional derivative check in the physical model region. Temporal subsampling and lower-precision storage deliberately approximate the gradient. Run [the gradient-check notebook](examples/4.GradientCheck.ipynb) after rebuilding the native ABI 6 libraries.
|
|
240
250
|
|
|
251
|
+
### 4.3 GPU-native block INT8 history
|
|
252
|
+
|
|
253
|
+
Only the sampled forward state used by the material adjoint is compressed. The
|
|
254
|
+
live Ex/Ey/Ez/Hx/Hy/Hz and CPML states remain float32. The executed discrete
|
|
255
|
+
update requires `E^n` and `R^n`, where `E^(n+1) = ca E^n + cb R^n`; consequently
|
|
256
|
+
`mode=2` stores compressed Ez/Rz and `mode=3` stores compressed Ex/Ey/Ez and all
|
|
257
|
+
three corresponding RHS components. Magnetic histories are not stored.
|
|
258
|
+
|
|
259
|
+
The packed tensor contains a contiguous signed-INT8 payload followed by a
|
|
260
|
+
four-byte-aligned contiguous FP32 scale array. Backward maps one CUDA block to
|
|
261
|
+
one compression tile, loads each E/R scale once into shared memory, decodes each
|
|
262
|
+
value in a register, and immediately accumulates the epsilon/conductivity
|
|
263
|
+
gradient. It does not allocate or write a reconstructed global-memory history.
|
|
264
|
+
`use_async_offload=True`, CPU execution, and a non-float32
|
|
265
|
+
`wavefield_storage_dtype` are explicitly incompatible with `"int8"`.
|
|
266
|
+
|
|
267
|
+
The current shared-memory maximum reduction remains available as
|
|
268
|
+
`int8_reduction_backend="current"`. The RTX 4090 audit selected CUB
|
|
269
|
+
`BlockReduce` for the default 64-voxel tiles; `"auto"` falls back to the current
|
|
270
|
+
tree for other supported power-of-two tile volumes.
|
|
271
|
+
|
|
272
|
+
`E_saved` is an opaque one-dimensional `torch.int8` packed tensor in this mode.
|
|
273
|
+
For diagnostics only, reconstruct it with
|
|
274
|
+
`DeepGPR.decompress_wavefield_history(E_saved, original_shape, block_size)`.
|
|
275
|
+
This helper materializes FP32 and is never called by autograd backward.
|
|
276
|
+
|
|
277
|
+
## CUDA Backend Build
|
|
278
|
+
|
|
279
|
+
Build the CUDA shared library from the repository root on Linux. `-lineinfo`
|
|
280
|
+
preserves source correlation for Nsight; `-Xptxas=-v` prints registers and spill
|
|
281
|
+
stores/loads for each kernel.
|
|
282
|
+
|
|
283
|
+
```bash
|
|
284
|
+
nvcc -std=c++14 -O3 -lineinfo -Xptxas=-v -arch=sm_89 --shared -Xcompiler -fPIC \
|
|
285
|
+
-o src/DeepGPR/lib/deepgpr.so src/DeepGPR/lib/deepgpr.cu
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
The command above is the audited RTX 4090 build path. Select the matching
|
|
289
|
+
architecture when building for a different GPU; do not add `--use_fast_math`
|
|
290
|
+
unless a separate numerical and gradient validation explicitly permits it.
|
|
291
|
+
|
|
292
|
+
The new INT8 path deliberately keeps native ABI 6 because the forward/backward
|
|
293
|
+
C signatures are unchanged. A capability symbol prevents an older ABI-6 CUDA
|
|
294
|
+
library from accepting the packed storage code and corrupting memory.
|
|
295
|
+
|
|
241
296
|
## CPU Backend Build
|
|
242
297
|
|
|
243
298
|
The CPU backend is a plain C shared library and is built with OpenMP by default. Build it into `src/DeepGPR/lib` before running with `device='cpu'`. You can control CPU thread count with `OMP_NUM_THREADS`.
|
|
@@ -286,11 +341,12 @@ return E_saved, (Ex, Ey, Ez), (Hx, Hy, Hz), (x0EPhi1...zmHPhi2), receiver_amplit
|
|
|
286
341
|
```
|
|
287
342
|
|
|
288
343
|
1. **`E_saved`**: The pre-update electric field history `E^n` saved for gradient calculation and diagnostics. An internal `R_saved` tensor stores the corresponding discrete right-hand side `R^n`.
|
|
344
|
+
* With `save_wavefield_history=False`, `E_saved` is a zero-length tensor and neither E nor R history storage/compression kernels are launched.
|
|
289
345
|
* **Shape when `mode=2`**: `(nt_saved, nstep, nx, ny, nz)`, storing Ez only.
|
|
290
346
|
* **Shape when `mode=3`**: `(3, nt_saved, nstep, nx, ny, nz)`, storing components in `[Ex, Ey, Ez]` order.
|
|
291
347
|
* `nt_saved` depends on `nt` and `model_gradient_sampling_interval`.
|
|
292
|
-
* Dtype is selected by `wavefield_storage_dtype
|
|
293
|
-
* Set `save_forward_wavefield_path="/path/to/output"` to save
|
|
348
|
+
* Dtype is selected by `wavefield_storage_dtype` when compression is disabled. With `wavefield_compression="int8"`, this is an opaque packed one-dimensional `torch.int8` tensor containing values and FP32 scales.
|
|
349
|
+
* Set `save_forward_wavefield_path="/path/to/output"` to save a CPU-loadable `.pt` file. Uncompressed modes save the tensor directly. INT8 mode saves a dictionary containing `wavefield`, `compression`, `block_size`, and `uncompressed_shape` so diagnostics can reconstruct it safely.
|
|
294
350
|
2. **`(Ex, Ey, Ez)`**: The 3D electric field state at the final time step.
|
|
295
351
|
3. **`(Hx, Hy, Hz)`**: The 3D magnetic field state at the final time step.
|
|
296
352
|
4. **`(PML_Tuple)`**: A tuple of 24 Tensors recording the final time step state of the PML auxiliary $\Phi$ variables.
|
|
@@ -232,6 +232,29 @@ def _configure_deepgpr_library(lib: ctypes.CDLL) -> None:
|
|
|
232
232
|
lib.set_fdtd_order.argtypes = [ctypes.c_int]
|
|
233
233
|
lib.set_fdtd_order.restype = None
|
|
234
234
|
|
|
235
|
+
if hasattr(lib, "deepgpr_supports_int8_wavefield"):
|
|
236
|
+
lib.deepgpr_supports_int8_wavefield.argtypes = []
|
|
237
|
+
lib.deepgpr_supports_int8_wavefield.restype = ctypes.c_int
|
|
238
|
+
|
|
239
|
+
if hasattr(lib, "deepgpr_supports_conversion_backends"):
|
|
240
|
+
lib.deepgpr_supports_conversion_backends.argtypes = []
|
|
241
|
+
lib.deepgpr_supports_conversion_backends.restype = ctypes.c_int
|
|
242
|
+
|
|
243
|
+
if hasattr(lib, "deepgpr_supports_int8_reduction_backends"):
|
|
244
|
+
lib.deepgpr_supports_int8_reduction_backends.argtypes = []
|
|
245
|
+
lib.deepgpr_supports_int8_reduction_backends.restype = ctypes.c_int
|
|
246
|
+
|
|
247
|
+
if hasattr(lib, "deepgpr_test_wavefield_conversion"):
|
|
248
|
+
lib.deepgpr_test_wavefield_conversion.argtypes = [
|
|
249
|
+
_FLOAT_P,
|
|
250
|
+
_VOID_P,
|
|
251
|
+
_FLOAT_P,
|
|
252
|
+
ctypes.c_longlong,
|
|
253
|
+
ctypes.c_int,
|
|
254
|
+
ctypes.c_int,
|
|
255
|
+
]
|
|
256
|
+
lib.deepgpr_test_wavefield_conversion.restype = None
|
|
257
|
+
|
|
235
258
|
lib._deepgpr_argtypes_configured = True
|
|
236
259
|
|
|
237
260
|
|