DeepGPR 0.0.16__tar.gz → 0.0.18__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deepgpr-0.0.16 → deepgpr-0.0.18}/PKG-INFO +1 -1
- {deepgpr-0.0.16 → deepgpr-0.0.18}/pyproject.toml +1 -1
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/common.py +150 -22
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/compute2.py +6 -2
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr.cu +246 -73
- deepgpr-0.0.18/src/DeepGPR/lib/deepgpr.dll +0 -0
- deepgpr-0.0.18/src/DeepGPR/lib/deepgpr.so +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.dll +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/PKG-INFO +1 -1
- {deepgpr-0.0.16 → deepgpr-0.0.18}/tests/test_discrete_adjoint.py +155 -8
- {deepgpr-0.0.16 → deepgpr-0.0.18}/tests/test_numerics.py +99 -0
- deepgpr-0.0.16/src/DeepGPR/lib/deepgpr.dll +0 -0
- deepgpr-0.0.16/src/DeepGPR/lib/deepgpr.so +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/README.md +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/license +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/setup.cfg +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/__init__.py +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr.h +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.c +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.dylib +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.so +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/libomp.dylib +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/multiscale.py +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/wavelet.py +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/SOURCES.txt +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/dependency_links.txt +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/requires.txt +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/top_level.txt +0 -0
- {deepgpr-0.0.16 → deepgpr-0.0.18}/tests/test_wavelet.py +0 -0
|
@@ -80,6 +80,7 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
80
80
|
pmlthick: PML thickness as an int, list, or tensor.
|
|
81
81
|
fdtd_order: Spatial finite-difference order used for the CFL check.
|
|
82
82
|
"""
|
|
83
|
+
device = torch.device("cpu" if device is None else device)
|
|
83
84
|
dtype=torch.float32
|
|
84
85
|
spacing = _normalize_grid_spacing(dx)
|
|
85
86
|
try:
|
|
@@ -100,6 +101,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
100
101
|
elif len(se.shape) != 3:
|
|
101
102
|
raise ValueError('The shape of sigma should be 2-d or 3-d.')
|
|
102
103
|
|
|
104
|
+
if mr is not None and not torch.is_tensor(mr):
|
|
105
|
+
raise TypeError("mr must be a PyTorch tensor or None.")
|
|
103
106
|
if mr is not None:
|
|
104
107
|
if len(mr.shape) == 2:
|
|
105
108
|
mr = mr.reshape(*mr.shape, 1)
|
|
@@ -108,6 +111,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
108
111
|
|
|
109
112
|
if er.shape != se.shape:
|
|
110
113
|
raise ValueError('The shape of epsilon and sigma should be the same.')
|
|
114
|
+
if any(size < 1 for size in er.shape):
|
|
115
|
+
raise ValueError("The material model dimensions must all be non-empty.")
|
|
111
116
|
nx, ny, nz = er.shape
|
|
112
117
|
mode = 2 if nz == 1 else 3
|
|
113
118
|
er=er.to(device=device, dtype=dtype)
|
|
@@ -119,17 +124,36 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
119
124
|
else:
|
|
120
125
|
raise ValueError('The shape of mr should be the same as epsilon and sigma.')
|
|
121
126
|
|
|
122
|
-
if source_location
|
|
123
|
-
raise
|
|
127
|
+
if not torch.is_tensor(source_location) or not torch.is_tensor(receiver_location):
|
|
128
|
+
raise TypeError("source_location and receiver_location must be PyTorch tensors.")
|
|
124
129
|
if source_location.ndim != 3 or receiver_location.ndim != 3:
|
|
125
130
|
raise ValueError("source_location and receiver_location must have shape (nstep, count, 3).")
|
|
126
131
|
if source_location.shape[2] != 3 or receiver_location.shape[2] != 3:
|
|
127
132
|
raise ValueError("The last dimension of source_location and receiver_location must be 3.")
|
|
133
|
+
if source_location.shape[0] != receiver_location.shape[0]:
|
|
134
|
+
raise ValueError('The first dimension (nstep) of source_location and receiver_location should be the same.')
|
|
128
135
|
nstep=source_location.shape[0]
|
|
129
136
|
nsr=source_location.shape[1]
|
|
130
137
|
nrx=receiver_location.shape[1]
|
|
131
|
-
|
|
132
|
-
|
|
138
|
+
if nstep < 1:
|
|
139
|
+
raise ValueError("At least one shot is required.")
|
|
140
|
+
if nsr < 1:
|
|
141
|
+
raise ValueError("At least one source per shot is required.")
|
|
142
|
+
if nrx < 1:
|
|
143
|
+
raise ValueError("At least one receiver per shot is required.")
|
|
144
|
+
if device.type == "cuda" and nstep > 65535:
|
|
145
|
+
raise ValueError(
|
|
146
|
+
"CUDA supports at most 65535 shots in one DeepGPR compute call. "
|
|
147
|
+
"Split a larger acquisition batch into smaller calls."
|
|
148
|
+
)
|
|
149
|
+
for name, locations in (
|
|
150
|
+
("source_location", source_location),
|
|
151
|
+
("receiver_location", receiver_location),
|
|
152
|
+
):
|
|
153
|
+
if locations.dtype == torch.bool or locations.is_complex():
|
|
154
|
+
raise TypeError(f"{name} must contain real integer-valued coordinates.")
|
|
155
|
+
source_location=source_location.to(device=device).contiguous()
|
|
156
|
+
receiver_location=receiver_location.to(device=device).contiguous()
|
|
133
157
|
|
|
134
158
|
if not torch.is_tensor(source_amplitudes):
|
|
135
159
|
raise TypeError("source_amplitudes must be a PyTorch tensor.")
|
|
@@ -165,6 +189,16 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
165
189
|
)
|
|
166
190
|
|
|
167
191
|
shape_tensor = torch.tensor((nx, ny, nz), dtype=torch.int32, device=device)
|
|
192
|
+
source_integral = (
|
|
193
|
+
(torch.isfinite(source_location) & (source_location == source_location.trunc())).all()
|
|
194
|
+
if source_location.is_floating_point()
|
|
195
|
+
else torch.ones((), dtype=torch.bool, device=device)
|
|
196
|
+
)
|
|
197
|
+
receiver_integral = (
|
|
198
|
+
(torch.isfinite(receiver_location) & (receiver_location == receiver_location.trunc())).all()
|
|
199
|
+
if receiver_location.is_floating_point()
|
|
200
|
+
else torch.ones((), dtype=torch.bool, device=device)
|
|
201
|
+
)
|
|
168
202
|
source_valid = ((source_location >= 0) & (source_location < shape_tensor)).all()
|
|
169
203
|
receiver_valid = ((receiver_location >= 0) & (receiver_location < shape_tensor)).all()
|
|
170
204
|
source_in_pml = _locations_in_pml(source_location, (nx, ny, nz), pml_values)
|
|
@@ -181,6 +215,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
181
215
|
(er.detach() * mr.detach()).amin(),
|
|
182
216
|
source_valid.to(dtype),
|
|
183
217
|
receiver_valid.to(dtype),
|
|
218
|
+
source_integral.to(dtype),
|
|
219
|
+
receiver_integral.to(dtype),
|
|
184
220
|
source_in_pml.sum().to(dtype),
|
|
185
221
|
receiver_in_pml.sum().to(dtype),
|
|
186
222
|
)
|
|
@@ -188,7 +224,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
188
224
|
er_finite, se_finite, mr_finite, source_finite = (bool(value) for value in stats[:4])
|
|
189
225
|
er_min, se_min, mr_min, min_er_mr = stats[4:8]
|
|
190
226
|
source_valid, receiver_valid = (bool(value) for value in stats[8:10])
|
|
191
|
-
|
|
227
|
+
source_integral, receiver_integral = (bool(value) for value in stats[10:12])
|
|
228
|
+
source_pml_count, receiver_pml_count = (int(value) for value in stats[12:14])
|
|
192
229
|
|
|
193
230
|
for name, finite in (
|
|
194
231
|
("er", er_finite), ("se", se_finite), ("mr", mr_finite),
|
|
@@ -202,6 +239,10 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
|
|
|
202
239
|
raise ValueError('The values of sigma is incorrect.(should be non-negative)')
|
|
203
240
|
if mr_min <= 0:
|
|
204
241
|
raise ValueError('The values of mr are incorrect (must be positive).')
|
|
242
|
+
if not source_integral:
|
|
243
|
+
raise ValueError("source_location must contain finite integer-valued coordinates.")
|
|
244
|
+
if not receiver_integral:
|
|
245
|
+
raise ValueError("receiver_location must contain finite integer-valued coordinates.")
|
|
205
246
|
if not source_valid:
|
|
206
247
|
raise ValueError(
|
|
207
248
|
"Error: Source coordinates out of range! "
|
|
@@ -292,24 +333,46 @@ def pmlthick_revert(p, er):
|
|
|
292
333
|
p: PML thickness as an int, list, or tensor.
|
|
293
334
|
er: Relative permittivity tensor used to detect 2D or 3D mode.
|
|
294
335
|
"""
|
|
295
|
-
if isinstance(p,
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
return torch.tensor(p + [0, 0], dtype=torch.int32)
|
|
336
|
+
if isinstance(p, bool):
|
|
337
|
+
raise TypeError("PML thickness must contain integer values, not bool.")
|
|
338
|
+
if isinstance(p, int):
|
|
339
|
+
values = [p, p, p, p, 0, 0] if er.shape[2] == 1 else [p] * 6
|
|
340
|
+
elif isinstance(p, (list, tuple)):
|
|
341
|
+
if len(p) == 4:
|
|
342
|
+
values = [*p, 0, 0]
|
|
343
|
+
elif len(p) == 6:
|
|
344
|
+
values = list(p)
|
|
305
345
|
else:
|
|
306
|
-
raise ValueError(f"Unsupported
|
|
307
|
-
|
|
346
|
+
raise ValueError(f"Unsupported PML length: {len(p)}. Must be 4 or 6.")
|
|
308
347
|
elif isinstance(p, torch.Tensor):
|
|
309
|
-
|
|
310
|
-
|
|
348
|
+
if p.ndim != 1:
|
|
349
|
+
raise ValueError("PML thickness tensor must be one-dimensional.")
|
|
350
|
+
if p.numel() == 4:
|
|
351
|
+
values = [*p.detach().cpu().tolist(), 0, 0]
|
|
352
|
+
elif p.numel() == 6:
|
|
353
|
+
values = p.detach().cpu().tolist()
|
|
354
|
+
else:
|
|
355
|
+
raise ValueError(
|
|
356
|
+
f"Unsupported PML length: {p.numel()}. Must be 4 or 6."
|
|
357
|
+
)
|
|
311
358
|
else:
|
|
312
|
-
raise TypeError(f"Unsupported type: {type(p)}")
|
|
359
|
+
raise TypeError(f"Unsupported PML thickness type: {type(p)}")
|
|
360
|
+
|
|
361
|
+
normalized = []
|
|
362
|
+
for value in values:
|
|
363
|
+
if isinstance(value, bool):
|
|
364
|
+
raise TypeError("PML thickness must contain integer values, not bool.")
|
|
365
|
+
try:
|
|
366
|
+
numeric = float(value)
|
|
367
|
+
except (TypeError, ValueError, OverflowError) as exc:
|
|
368
|
+
raise TypeError("PML thickness must contain finite integer values.") from exc
|
|
369
|
+
if not math.isfinite(numeric) or not numeric.is_integer():
|
|
370
|
+
raise ValueError("PML thickness must contain finite integer values.")
|
|
371
|
+
integer = int(numeric)
|
|
372
|
+
if integer < 0 or integer > torch.iinfo(torch.int32).max:
|
|
373
|
+
raise ValueError("PML thickness values must fit in non-negative int32.")
|
|
374
|
+
normalized.append(integer)
|
|
375
|
+
return torch.tensor(normalized, dtype=torch.int32)
|
|
313
376
|
|
|
314
377
|
|
|
315
378
|
class TVRegularization(nn.Module):
|
|
@@ -734,8 +797,29 @@ def build_pml_phi(x0,xm,y0,ym,z0,zm,nstep,PML,device):
|
|
|
734
797
|
z0EPhi1, z0EPhi2, z0HPhi1, z0HPhi2,
|
|
735
798
|
zmEPhi1, zmEPhi2, zmHPhi1, zmHPhi2) = [torch.empty(0) for _ in range(24)]
|
|
736
799
|
|
|
737
|
-
|
|
738
|
-
|
|
800
|
+
descriptors = (x0, xm, y0, ym, z0, zm)
|
|
801
|
+
if PML is None:
|
|
802
|
+
PML = (None,) * 24
|
|
803
|
+
elif not isinstance(PML, (list, tuple)) or len(PML) != 24:
|
|
804
|
+
raise ValueError("PML must contain exactly 24 CPML auxiliary tensors.")
|
|
805
|
+
|
|
806
|
+
for face, descriptor in enumerate(descriptors):
|
|
807
|
+
group = PML[4 * face:4 * face + 4]
|
|
808
|
+
supplied = tuple(value is not None for value in group)
|
|
809
|
+
if any(supplied) and not all(supplied):
|
|
810
|
+
raise ValueError(
|
|
811
|
+
f"CPML face {face} must provide all four auxiliary tensors or none."
|
|
812
|
+
)
|
|
813
|
+
if all(supplied) and not all(torch.is_tensor(value) for value in group):
|
|
814
|
+
raise TypeError(f"CPML face {face} state must contain PyTorch tensors.")
|
|
815
|
+
if (
|
|
816
|
+
all(supplied)
|
|
817
|
+
and descriptor.numel() == 0
|
|
818
|
+
and any(value.numel() != 0 for value in group)
|
|
819
|
+
):
|
|
820
|
+
raise ValueError(
|
|
821
|
+
f"CPML face {face} state must be empty for a zero-thickness boundary."
|
|
822
|
+
)
|
|
739
823
|
|
|
740
824
|
if x0.numel()!=0 and PML[0]==None:
|
|
741
825
|
x0EPhi1=torch.zeros((nstep, int(x0[2]-x0[1]+1), int(x0[4]-x0[3]), int(x0[6]-x0[5]+1)),dtype=torch.float, device=device)
|
|
@@ -811,6 +895,50 @@ def build_pml_phi(x0,xm,y0,ym,z0,zm,nstep,PML,device):
|
|
|
811
895
|
z0EPhi1,z0EPhi2,z0HPhi1,z0HPhi2,
|
|
812
896
|
zmEPhi1,zmEPhi2,zmHPhi1,zmHPhi2,
|
|
813
897
|
)
|
|
898
|
+
|
|
899
|
+
def expected_shapes(axis, descriptor):
|
|
900
|
+
if descriptor.numel() == 0:
|
|
901
|
+
return (None,) * 4
|
|
902
|
+
a = int(descriptor[2] - descriptor[1])
|
|
903
|
+
b = int(descriptor[4] - descriptor[3])
|
|
904
|
+
c = int(descriptor[6] - descriptor[5])
|
|
905
|
+
if axis == 0:
|
|
906
|
+
return (
|
|
907
|
+
(nstep, a + 1, b, c + 1),
|
|
908
|
+
(nstep, a + 1, b + 1, c),
|
|
909
|
+
(nstep, a, b + 1, c),
|
|
910
|
+
(nstep, a, b, c + 1),
|
|
911
|
+
)
|
|
912
|
+
if axis == 1:
|
|
913
|
+
return (
|
|
914
|
+
(nstep, a, b + 1, c + 1),
|
|
915
|
+
(nstep, a + 1, b + 1, c),
|
|
916
|
+
(nstep, a + 1, b, c),
|
|
917
|
+
(nstep, a, b, c + 1),
|
|
918
|
+
)
|
|
919
|
+
return (
|
|
920
|
+
(nstep, a, b + 1, c + 1),
|
|
921
|
+
(nstep, a + 1, b, c + 1),
|
|
922
|
+
(nstep, a + 1, b, c),
|
|
923
|
+
(nstep, a, b + 1, c),
|
|
924
|
+
)
|
|
925
|
+
|
|
926
|
+
for face, descriptor in enumerate(descriptors):
|
|
927
|
+
group = tensors[4 * face:4 * face + 4]
|
|
928
|
+
for component, (tensor, expected) in enumerate(
|
|
929
|
+
zip(group, expected_shapes(face // 2, descriptor))
|
|
930
|
+
):
|
|
931
|
+
if expected is None:
|
|
932
|
+
if tensor.numel() != 0:
|
|
933
|
+
raise ValueError(
|
|
934
|
+
f"CPML face {face} component {component} must be empty."
|
|
935
|
+
)
|
|
936
|
+
elif tuple(tensor.shape) != expected:
|
|
937
|
+
raise ValueError(
|
|
938
|
+
f"CPML face {face} component {component} has shape "
|
|
939
|
+
f"{tuple(tensor.shape)}; expected {expected}."
|
|
940
|
+
)
|
|
941
|
+
|
|
814
942
|
return tuple(
|
|
815
943
|
tensor.to(device=device, dtype=torch.float32).contiguous()
|
|
816
944
|
for tensor in tensors
|
|
@@ -700,8 +700,12 @@ class DeepGPR(torch.autograd.Function):
|
|
|
700
700
|
"""
|
|
701
701
|
|
|
702
702
|
source_amplitudes = source_amplitudes.contiguous()
|
|
703
|
-
source_location=source_location.to(
|
|
704
|
-
|
|
703
|
+
source_location=source_location.to(
|
|
704
|
+
device=device, dtype=torch.int32
|
|
705
|
+
).contiguous()
|
|
706
|
+
receiver_location=receiver_location.to(
|
|
707
|
+
device=device, dtype=torch.int32
|
|
708
|
+
).contiguous()
|
|
705
709
|
c_lib = get_deepgpr_lib(device)
|
|
706
710
|
set_library_fdtd_order(c_lib, fdtd_order)
|
|
707
711
|
ctx.save_for_backward(mu_r, source_location, receiver_location,
|
|
@@ -175,6 +175,146 @@ __device__ __forceinline__ float load_wavefield_value_device(
|
|
|
175
175
|
} \
|
|
176
176
|
} while (0)
|
|
177
177
|
|
|
178
|
+
static int clamp_cpml_index(int value, int extent)
|
|
179
|
+
{
|
|
180
|
+
if (value < 0) return 0;
|
|
181
|
+
if (value > extent) return extent;
|
|
182
|
+
return value;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/*
|
|
186
|
+
* Partition the CPML shell into six disjoint boxes. The x boxes own edges and
|
|
187
|
+
* corners, the y boxes exclude the x boxes, and the z boxes exclude both.
|
|
188
|
+
* Return the largest box so one grid row can be assigned to each face.
|
|
189
|
+
*/
|
|
190
|
+
static long long cpml_max_region_cells(
|
|
191
|
+
int NX, int NY, int NZ,
|
|
192
|
+
int pml0, int pml1, int pml2, int pml3, int pml4, int pml5,
|
|
193
|
+
int electric)
|
|
194
|
+
{
|
|
195
|
+
int x_low_end = pml0 > 0
|
|
196
|
+
? clamp_cpml_index(pml0 + (electric ? 1 : 0), NX) : 0;
|
|
197
|
+
int y_low_end = pml2 > 0
|
|
198
|
+
? clamp_cpml_index(pml2 + (electric ? 1 : 0), NY) : 0;
|
|
199
|
+
int z_low_end = pml4 > 0
|
|
200
|
+
? clamp_cpml_index(pml4 + (electric ? 1 : 0), NZ) : 0;
|
|
201
|
+
int x_high_begin = pml1 > 0
|
|
202
|
+
? clamp_cpml_index(NX - 1 - pml1, NX) : NX;
|
|
203
|
+
int y_high_begin = pml3 > 0
|
|
204
|
+
? clamp_cpml_index(NY - 1 - pml3, NY) : NY;
|
|
205
|
+
int z_high_begin = pml5 > 0
|
|
206
|
+
? clamp_cpml_index(NZ - 1 - pml5, NZ) : NZ;
|
|
207
|
+
|
|
208
|
+
int x_mid_end = x_high_begin > x_low_end ? x_high_begin : x_low_end;
|
|
209
|
+
int y_mid_end = y_high_begin > y_low_end ? y_high_begin : y_low_end;
|
|
210
|
+
int z_mid_end = z_high_begin > z_low_end ? z_high_begin : z_low_end;
|
|
211
|
+
long long x_mid = x_mid_end - x_low_end;
|
|
212
|
+
long long y_mid = y_mid_end - y_low_end;
|
|
213
|
+
|
|
214
|
+
long long regions[6] = {
|
|
215
|
+
(long long)x_low_end * NY * NZ,
|
|
216
|
+
pml1 > 0 ? (long long)(NX - x_mid_end) * NY * NZ : 0,
|
|
217
|
+
x_mid * (long long)y_low_end * NZ,
|
|
218
|
+
pml3 > 0 ? x_mid * (long long)(NY - y_mid_end) * NZ : 0,
|
|
219
|
+
x_mid * y_mid * (long long)z_low_end,
|
|
220
|
+
pml5 > 0 ? x_mid * y_mid * (long long)(NZ - z_mid_end) : 0,
|
|
221
|
+
};
|
|
222
|
+
long long maximum = 0;
|
|
223
|
+
for (int region = 0; region < 6; ++region) {
|
|
224
|
+
if (regions[region] > maximum) maximum = regions[region];
|
|
225
|
+
}
|
|
226
|
+
return maximum;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
template<int ELECTRIC>
|
|
230
|
+
__device__ inline bool map_cpml_region_work(
|
|
231
|
+
int region, long long local,
|
|
232
|
+
int NX, int NY, int NZ,
|
|
233
|
+
int pml0, int pml1, int pml2, int pml3, int pml4, int pml5,
|
|
234
|
+
long long* i, long long* j, long long* k)
|
|
235
|
+
{
|
|
236
|
+
int x_low_end = pml0 > 0 ? pml0 + ELECTRIC : 0;
|
|
237
|
+
int y_low_end = pml2 > 0 ? pml2 + ELECTRIC : 0;
|
|
238
|
+
int z_low_end = pml4 > 0 ? pml4 + ELECTRIC : 0;
|
|
239
|
+
if (x_low_end < 0) x_low_end = 0;
|
|
240
|
+
if (y_low_end < 0) y_low_end = 0;
|
|
241
|
+
if (z_low_end < 0) z_low_end = 0;
|
|
242
|
+
if (x_low_end > NX) x_low_end = NX;
|
|
243
|
+
if (y_low_end > NY) y_low_end = NY;
|
|
244
|
+
if (z_low_end > NZ) z_low_end = NZ;
|
|
245
|
+
|
|
246
|
+
int x_high_begin = pml1 > 0 ? NX - 1 - pml1 : NX;
|
|
247
|
+
int y_high_begin = pml3 > 0 ? NY - 1 - pml3 : NY;
|
|
248
|
+
int z_high_begin = pml5 > 0 ? NZ - 1 - pml5 : NZ;
|
|
249
|
+
if (x_high_begin < 0) x_high_begin = 0;
|
|
250
|
+
if (y_high_begin < 0) y_high_begin = 0;
|
|
251
|
+
if (z_high_begin < 0) z_high_begin = 0;
|
|
252
|
+
|
|
253
|
+
int x_mid_end = x_high_begin > x_low_end ? x_high_begin : x_low_end;
|
|
254
|
+
int y_mid_end = y_high_begin > y_low_end ? y_high_begin : y_low_end;
|
|
255
|
+
int z_mid_end = z_high_begin > z_low_end ? z_high_begin : z_low_end;
|
|
256
|
+
|
|
257
|
+
int i0 = 0, ni = 0;
|
|
258
|
+
int j0 = 0, nj = 0;
|
|
259
|
+
int k0 = 0, nk = 0;
|
|
260
|
+
switch (region) {
|
|
261
|
+
case 0:
|
|
262
|
+
ni = x_low_end;
|
|
263
|
+
nj = NY;
|
|
264
|
+
nk = NZ;
|
|
265
|
+
break;
|
|
266
|
+
case 1:
|
|
267
|
+
if (pml1 <= 0) return false;
|
|
268
|
+
i0 = x_mid_end;
|
|
269
|
+
ni = NX - x_mid_end;
|
|
270
|
+
nj = NY;
|
|
271
|
+
nk = NZ;
|
|
272
|
+
break;
|
|
273
|
+
case 2:
|
|
274
|
+
i0 = x_low_end;
|
|
275
|
+
ni = x_mid_end - x_low_end;
|
|
276
|
+
nj = y_low_end;
|
|
277
|
+
nk = NZ;
|
|
278
|
+
break;
|
|
279
|
+
case 3:
|
|
280
|
+
if (pml3 <= 0) return false;
|
|
281
|
+
i0 = x_low_end;
|
|
282
|
+
ni = x_mid_end - x_low_end;
|
|
283
|
+
j0 = y_mid_end;
|
|
284
|
+
nj = NY - y_mid_end;
|
|
285
|
+
nk = NZ;
|
|
286
|
+
break;
|
|
287
|
+
case 4:
|
|
288
|
+
i0 = x_low_end;
|
|
289
|
+
ni = x_mid_end - x_low_end;
|
|
290
|
+
j0 = y_low_end;
|
|
291
|
+
nj = y_mid_end - y_low_end;
|
|
292
|
+
nk = z_low_end;
|
|
293
|
+
break;
|
|
294
|
+
case 5:
|
|
295
|
+
if (pml5 <= 0) return false;
|
|
296
|
+
i0 = x_low_end;
|
|
297
|
+
ni = x_mid_end - x_low_end;
|
|
298
|
+
j0 = y_low_end;
|
|
299
|
+
nj = y_mid_end - y_low_end;
|
|
300
|
+
k0 = z_mid_end;
|
|
301
|
+
nk = NZ - z_mid_end;
|
|
302
|
+
break;
|
|
303
|
+
default:
|
|
304
|
+
return false;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
if (ni <= 0 || nj <= 0 || nk <= 0) return false;
|
|
308
|
+
long long jk = (long long)nj * nk;
|
|
309
|
+
long long region_size = (long long)ni * jk;
|
|
310
|
+
if (local >= region_size) return false;
|
|
311
|
+
*i = i0 + local / jk;
|
|
312
|
+
long long remainder = local % jk;
|
|
313
|
+
*j = j0 + remainder / nk;
|
|
314
|
+
*k = k0 + remainder % nk;
|
|
315
|
+
return true;
|
|
316
|
+
}
|
|
317
|
+
|
|
178
318
|
static int g_fdtd_order = 2;
|
|
179
319
|
|
|
180
320
|
DEEPGPR_API int deepgpr_abi_version(void)
|
|
@@ -439,7 +579,7 @@ __global__ void build_update_coeffs_gpu(const float* __restrict__ eps_r_pad, con
|
|
|
439
579
|
float* __restrict__ ch_hist, float* __restrict__ ch_curl, float* __restrict__ ch_rhs,
|
|
440
580
|
int NX_FIELDS, int NY_FIELDS, int NZ_FIELDS, float dt, float dx)
|
|
441
581
|
{
|
|
442
|
-
long long idx = blockIdx.x * blockDim.x + threadIdx.x;
|
|
582
|
+
long long idx = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
443
583
|
long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
|
|
444
584
|
if (idx >= (long long)NX_FIELDS * ny_nz) return;
|
|
445
585
|
|
|
@@ -489,7 +629,7 @@ __global__ void sample_receivers_gpu(
|
|
|
489
629
|
int NX, int NY, int NZ, int N_ITER, int receiver_component)
|
|
490
630
|
{
|
|
491
631
|
long long field_stride = (long long)NX * NY * NZ;
|
|
492
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
632
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
493
633
|
long long total = (long long)step * NRX;
|
|
494
634
|
if (work >= total) return;
|
|
495
635
|
|
|
@@ -530,7 +670,7 @@ __global__ void inject_sources_gpu(
|
|
|
530
670
|
int NX, int NY, int NZ, int nsrc, int polarisation, int nt)
|
|
531
671
|
{
|
|
532
672
|
long long field_stride = (long long)NX * NY * NZ;
|
|
533
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
673
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
534
674
|
long long total = (long long)step * nsrc;
|
|
535
675
|
if (work >= total) return;
|
|
536
676
|
|
|
@@ -547,9 +687,10 @@ __global__ void inject_sources_gpu(
|
|
|
547
687
|
long long id3 = i * NY * NZ + j * NZ + k;
|
|
548
688
|
long long id4 = (long long)s * field_stride + id3;
|
|
549
689
|
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
else if (polarisation ==
|
|
690
|
+
float value = -ce_rhs[id3] * scale;
|
|
691
|
+
if (polarisation == 0) atomicAdd(&Ex[id4], value);
|
|
692
|
+
else if (polarisation == 1) atomicAdd(&Ey[id4], value);
|
|
693
|
+
else if (polarisation == 2) atomicAdd(&Ez[id4], value);
|
|
553
694
|
}
|
|
554
695
|
|
|
555
696
|
|
|
@@ -577,9 +718,9 @@ __global__ void inject_sources_and_sample_gpu(
|
|
|
577
718
|
float scale = srcwaveforms[src * nt + iteration] * dipole_length / (dx * dy * dz);
|
|
578
719
|
float value = ce_rhs[material_idx] * scale;
|
|
579
720
|
|
|
580
|
-
if (source_component == 0) Ex[field_idx]
|
|
581
|
-
else if (source_component == 1) Ey[field_idx]
|
|
582
|
-
else Ez[field_idx]
|
|
721
|
+
if (source_component == 0) atomicAdd(&Ex[field_idx], -value);
|
|
722
|
+
else if (source_component == 1) atomicAdd(&Ey[field_idx], -value);
|
|
723
|
+
else atomicAdd(&Ez[field_idx], -value);
|
|
583
724
|
}
|
|
584
725
|
|
|
585
726
|
__syncthreads();
|
|
@@ -608,7 +749,7 @@ __global__ void update_e_gpu(
|
|
|
608
749
|
{
|
|
609
750
|
long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
|
|
610
751
|
long long field_stride = (long long)NX_FIELDS * ny_nz;
|
|
611
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
752
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
612
753
|
if (work >= (long long)step * field_stride) return;
|
|
613
754
|
|
|
614
755
|
long long idx = work % field_stride;
|
|
@@ -667,16 +808,16 @@ __global__ void cpml_e_gpu(
|
|
|
667
808
|
{
|
|
668
809
|
long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
|
|
669
810
|
long long field_stride = (long long)NX_FIELDS * ny_nz;
|
|
670
|
-
long long
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
long long
|
|
679
|
-
long long
|
|
811
|
+
long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
812
|
+
int region = (int)blockIdx.y;
|
|
813
|
+
int s = (int)blockIdx.z;
|
|
814
|
+
long long i, j, k;
|
|
815
|
+
if (s >= step ||
|
|
816
|
+
!map_cpml_region_work<1>(region, region_work,
|
|
817
|
+
NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
818
|
+
pml0, pml1, pml2, pml3, pml4, pml5, &i, &j, &k)) return;
|
|
819
|
+
long long idx = i * ny_nz + j * NZ_FIELDS + k;
|
|
820
|
+
long long work = (long long)s * field_stride + idx;
|
|
680
821
|
|
|
681
822
|
bool in_x0 = (pml0 > 0 && i > 0 && i <= pml0 && j < NY_FIELDS && k < NZ_FIELDS);
|
|
682
823
|
bool in_xm = (pml1 > 0 && i >= NX_FIELDS - 1 - pml1 && i < NX_FIELDS - 1 && j < NY_FIELDS && k < NZ_FIELDS);
|
|
@@ -818,7 +959,7 @@ __global__ void update_h_gpu(
|
|
|
818
959
|
{
|
|
819
960
|
long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
|
|
820
961
|
long long field_stride = (long long)NX_FIELDS * ny_nz;
|
|
821
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
962
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
822
963
|
if (work >= (long long)step * field_stride) return;
|
|
823
964
|
|
|
824
965
|
long long idx = work % field_stride;
|
|
@@ -862,7 +1003,7 @@ __global__ void adjoint_e_gpu(
|
|
|
862
1003
|
{
|
|
863
1004
|
long long ny_nz = (long long)NY * NZ;
|
|
864
1005
|
long long field_stride = (long long)NX * ny_nz;
|
|
865
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
1006
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
866
1007
|
if (work >= (long long)step * field_stride) return;
|
|
867
1008
|
long long idx = work % field_stride;
|
|
868
1009
|
long long i = idx / ny_nz;
|
|
@@ -904,7 +1045,7 @@ __global__ void adjoint_h_gpu(
|
|
|
904
1045
|
{
|
|
905
1046
|
long long ny_nz = (long long)NY * NZ;
|
|
906
1047
|
long long field_stride = (long long)NX * ny_nz;
|
|
907
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
1048
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
908
1049
|
if (work >= (long long)step * field_stride) return;
|
|
909
1050
|
long long idx = work % field_stride;
|
|
910
1051
|
long long i = idx / ny_nz;
|
|
@@ -961,16 +1102,16 @@ __global__ void cpml_h_gpu(
|
|
|
961
1102
|
{
|
|
962
1103
|
long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
|
|
963
1104
|
long long field_stride = (long long)NX_FIELDS * ny_nz;
|
|
964
|
-
long long
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
long long
|
|
973
|
-
long long
|
|
1105
|
+
long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1106
|
+
int region = (int)blockIdx.y;
|
|
1107
|
+
int s = (int)blockIdx.z;
|
|
1108
|
+
long long i, j, k;
|
|
1109
|
+
if (s >= step ||
|
|
1110
|
+
!map_cpml_region_work<0>(region, region_work,
|
|
1111
|
+
NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1112
|
+
pml0, pml1, pml2, pml3, pml4, pml5, &i, &j, &k)) return;
|
|
1113
|
+
long long idx = i * ny_nz + j * NZ_FIELDS + k;
|
|
1114
|
+
long long work = (long long)s * field_stride + idx;
|
|
974
1115
|
|
|
975
1116
|
bool in_x0 = (pml0 > 0 && i < pml0 && j < NY_FIELDS && k < NZ_FIELDS);
|
|
976
1117
|
bool in_xm = (pml1 > 0 && i >= NX_FIELDS - 1 - pml1 && i < NX_FIELDS - 1 && j < NY_FIELDS && k < NZ_FIELDS);
|
|
@@ -1115,14 +1256,16 @@ __global__ void adjoint_cpml_e_gpu(
|
|
|
1115
1256
|
{
|
|
1116
1257
|
long long ny_nz = (long long)NY * NZ;
|
|
1117
1258
|
long long field_stride = (long long)NX * ny_nz;
|
|
1118
|
-
long long
|
|
1119
|
-
|
|
1120
|
-
int s = (int)
|
|
1121
|
-
long long
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1259
|
+
long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1260
|
+
int region = (int)blockIdx.y;
|
|
1261
|
+
int s = (int)blockIdx.z;
|
|
1262
|
+
long long i, j, k;
|
|
1263
|
+
if (s >= step ||
|
|
1264
|
+
!map_cpml_region_work<1>(region, region_work,
|
|
1265
|
+
NX, NY, NZ, pml0, pml1, pml2, pml3, pml4, pml5,
|
|
1266
|
+
&i, &j, &k)) return;
|
|
1267
|
+
long long idx = i * ny_nz + j * NZ + k;
|
|
1268
|
+
long long work = (long long)s * field_stride + idx;
|
|
1126
1269
|
float upd = update[idx];
|
|
1127
1270
|
|
|
1128
1271
|
#define APPLY_E_PML_GPU(R, P, p, q, stride, coord, n, spacing, field, source, sign) \
|
|
@@ -1215,14 +1358,16 @@ __global__ void adjoint_cpml_h_gpu(
|
|
|
1215
1358
|
{
|
|
1216
1359
|
long long ny_nz = (long long)NY * NZ;
|
|
1217
1360
|
long long field_stride = (long long)NX * ny_nz;
|
|
1218
|
-
long long
|
|
1219
|
-
|
|
1220
|
-
int s = (int)
|
|
1221
|
-
long long
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1361
|
+
long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1362
|
+
int region = (int)blockIdx.y;
|
|
1363
|
+
int s = (int)blockIdx.z;
|
|
1364
|
+
long long i, j, k;
|
|
1365
|
+
if (s >= step ||
|
|
1366
|
+
!map_cpml_region_work<0>(region, region_work,
|
|
1367
|
+
NX, NY, NZ, pml0, pml1, pml2, pml3, pml4, pml5,
|
|
1368
|
+
&i, &j, &k)) return;
|
|
1369
|
+
long long idx = i * ny_nz + j * NZ + k;
|
|
1370
|
+
long long work = (long long)s * field_stride + idx;
|
|
1226
1371
|
float upd = update[idx];
|
|
1227
1372
|
|
|
1228
1373
|
#define APPLY_H_PML_GPU(R, P, p, q, stride, coord, n, spacing, field, source, sign) \
|
|
@@ -1324,7 +1469,7 @@ __global__ void adjoint_receivers_gpu(
|
|
|
1324
1469
|
int NX, int NY, int NZ, int nsr, int polarisation, int iterations
|
|
1325
1470
|
){
|
|
1326
1471
|
long long field_stride = (long long)NX * NY * NZ;
|
|
1327
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
1472
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1328
1473
|
long long total = (long long)step * nsr;
|
|
1329
1474
|
if (work >= total) return;
|
|
1330
1475
|
|
|
@@ -1339,9 +1484,9 @@ __global__ void adjoint_receivers_gpu(
|
|
|
1339
1484
|
float waveform_value = srcwaveforms[index];
|
|
1340
1485
|
long long id4 = (long long)s * field_stride + i * NY * NZ + j * NZ + k;
|
|
1341
1486
|
|
|
1342
|
-
if (polarisation == 0) lambda_ex[id4]
|
|
1343
|
-
else if (polarisation == 1) lambda_ey[id4]
|
|
1344
|
-
else if (polarisation == 2) lambda_ez[id4]
|
|
1487
|
+
if (polarisation == 0) atomicAdd(&lambda_ex[id4], waveform_value);
|
|
1488
|
+
else if (polarisation == 1) atomicAdd(&lambda_ey[id4], waveform_value);
|
|
1489
|
+
else if (polarisation == 2) atomicAdd(&lambda_ez[id4], waveform_value);
|
|
1345
1490
|
}
|
|
1346
1491
|
|
|
1347
1492
|
|
|
@@ -1355,7 +1500,7 @@ __global__ void adjoint_source_injection_gpu(
|
|
|
1355
1500
|
float* __restrict__ grad_source)
|
|
1356
1501
|
{
|
|
1357
1502
|
long long field_stride = (long long)NX * NY * NZ;
|
|
1358
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
1503
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1359
1504
|
long long total = (long long)step * nsrc;
|
|
1360
1505
|
if (work >= total) return;
|
|
1361
1506
|
|
|
@@ -1392,7 +1537,7 @@ __global__ void save_e_snapshot_gpu(
|
|
|
1392
1537
|
{
|
|
1393
1538
|
long long nx1 = NX - 1, ny1 = NY - 1, nz1 = NZ - 1;
|
|
1394
1539
|
long long total = nx1 * ny1 * nz1;
|
|
1395
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
1540
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1396
1541
|
if (work >= (long long)step * total) return;
|
|
1397
1542
|
|
|
1398
1543
|
int s = (int)(work / total);
|
|
@@ -1425,7 +1570,7 @@ __global__ void save_rhs_snapshot_gpu(
|
|
|
1425
1570
|
{
|
|
1426
1571
|
long long nx1 = NX - 1, ny1 = NY - 1, nz1 = NZ - 1;
|
|
1427
1572
|
long long total = nx1 * ny1 * nz1;
|
|
1428
|
-
long long work = blockIdx.x * blockDim.x + threadIdx.x;
|
|
1573
|
+
long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1429
1574
|
if (work >= (long long)step * total) return;
|
|
1430
1575
|
|
|
1431
1576
|
int s = (int)(work / total);
|
|
@@ -1491,7 +1636,7 @@ __global__ void accumulate_material_gradients_gpu(
|
|
|
1491
1636
|
long long snap_stride = (long long)step * total_cells;
|
|
1492
1637
|
long long component_stride = (long long)nt_saved * snap_stride;
|
|
1493
1638
|
int components = (fwi_mode == 3) ? 3 : 1;
|
|
1494
|
-
long long idx = blockIdx.x * blockDim.x + threadIdx.x;
|
|
1639
|
+
long long idx = (long long)blockIdx.x * blockDim.x + threadIdx.x;
|
|
1495
1640
|
|
|
1496
1641
|
if (idx >= total_cells) return;
|
|
1497
1642
|
|
|
@@ -1666,6 +1811,20 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
|
|
|
1666
1811
|
long long total_fields = (long long)NX_FIELDS * NY_FIELDS * NZ_FIELDS;
|
|
1667
1812
|
dim3 grid_material(CEIL_DIV(total_fields, blockSize));
|
|
1668
1813
|
dim3 grid_fields(CEIL_DIV((long long)step * total_fields, blockSize));
|
|
1814
|
+
long long cpml_e_region_max = cpml_max_region_cells(
|
|
1815
|
+
NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1816
|
+
pml0, pml1, pml2, pml3, pml4, pml5, 1);
|
|
1817
|
+
long long cpml_h_region_max = cpml_max_region_cells(
|
|
1818
|
+
NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1819
|
+
pml0, pml1, pml2, pml3, pml4, pml5, 0);
|
|
1820
|
+
int has_cpml_e = has_cpml && cpml_e_region_max > 0;
|
|
1821
|
+
int has_cpml_h = has_cpml && cpml_h_region_max > 0;
|
|
1822
|
+
dim3 grid_cpml_e(
|
|
1823
|
+
has_cpml_e ? CEIL_DIV(cpml_e_region_max, blockSize) : 1,
|
|
1824
|
+
6, step);
|
|
1825
|
+
dim3 grid_cpml_h(
|
|
1826
|
+
has_cpml_h ? CEIL_DIV(cpml_h_region_max, blockSize) : 1,
|
|
1827
|
+
6, step);
|
|
1669
1828
|
|
|
1670
1829
|
build_update_coeffs_gpu<<<grid_material, blockSize, 0, stream_comp>>>(eps_r_pad, sigma_pad, mu_r_pad, ce_hist, ce_curl, ce_rhs, ch_hist, ch_curl, ch_rhs, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dt, dx);
|
|
1671
1830
|
CUDA_CHECK_LAST();
|
|
@@ -1717,8 +1876,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
|
|
|
1717
1876
|
if (fdtd_order == 8) {
|
|
1718
1877
|
update_h_gpu<8><<<grid_fields, blockSize, 0, stream_comp>>>(ch_hist, ch_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
|
|
1719
1878
|
CUDA_CHECK_LAST();
|
|
1720
|
-
if (
|
|
1721
|
-
cpml_h_gpu<8><<<
|
|
1879
|
+
if (has_cpml_h) {
|
|
1880
|
+
cpml_h_gpu<8><<<grid_cpml_h, blockSize, 0, stream_comp>>>(
|
|
1722
1881
|
Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1723
1882
|
pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
|
|
1724
1883
|
x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2, z0HPhi1, z0HPhi2, zmHPhi1, zmHPhi2);
|
|
@@ -1726,8 +1885,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
|
|
|
1726
1885
|
}
|
|
1727
1886
|
update_e_gpu<8><<<grid_fields, blockSize, 0, stream_comp>>>(ce_hist, ce_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
|
|
1728
1887
|
CUDA_CHECK_LAST();
|
|
1729
|
-
if (
|
|
1730
|
-
cpml_e_gpu<8><<<
|
|
1888
|
+
if (has_cpml_e) {
|
|
1889
|
+
cpml_e_gpu<8><<<grid_cpml_e, blockSize, 0, stream_comp>>>(
|
|
1731
1890
|
Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1732
1891
|
pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
|
|
1733
1892
|
x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2, z0EPhi1, z0EPhi2, zmEPhi1, zmEPhi2);
|
|
@@ -1736,8 +1895,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
|
|
|
1736
1895
|
} else if (fdtd_order == 4) {
|
|
1737
1896
|
update_h_gpu<4><<<grid_fields, blockSize, 0, stream_comp>>>(ch_hist, ch_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
|
|
1738
1897
|
CUDA_CHECK_LAST();
|
|
1739
|
-
if (
|
|
1740
|
-
cpml_h_gpu<4><<<
|
|
1898
|
+
if (has_cpml_h) {
|
|
1899
|
+
cpml_h_gpu<4><<<grid_cpml_h, blockSize, 0, stream_comp>>>(
|
|
1741
1900
|
Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1742
1901
|
pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
|
|
1743
1902
|
x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2, z0HPhi1, z0HPhi2, zmHPhi1, zmHPhi2);
|
|
@@ -1745,8 +1904,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
|
|
|
1745
1904
|
}
|
|
1746
1905
|
update_e_gpu<4><<<grid_fields, blockSize, 0, stream_comp>>>(ce_hist, ce_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
|
|
1747
1906
|
CUDA_CHECK_LAST();
|
|
1748
|
-
if (
|
|
1749
|
-
cpml_e_gpu<4><<<
|
|
1907
|
+
if (has_cpml_e) {
|
|
1908
|
+
cpml_e_gpu<4><<<grid_cpml_e, blockSize, 0, stream_comp>>>(
|
|
1750
1909
|
Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1751
1910
|
pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
|
|
1752
1911
|
x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2, z0EPhi1, z0EPhi2, zmEPhi1, zmEPhi2);
|
|
@@ -1755,8 +1914,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
|
|
|
1755
1914
|
} else {
|
|
1756
1915
|
update_h_gpu<2><<<grid_fields, blockSize, 0, stream_comp>>>(ch_hist, ch_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
|
|
1757
1916
|
CUDA_CHECK_LAST();
|
|
1758
|
-
if (
|
|
1759
|
-
cpml_h_gpu<2><<<
|
|
1917
|
+
if (has_cpml_h) {
|
|
1918
|
+
cpml_h_gpu<2><<<grid_cpml_h, blockSize, 0, stream_comp>>>(
|
|
1760
1919
|
Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1761
1920
|
pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
|
|
1762
1921
|
x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2, z0HPhi1, z0HPhi2, zmHPhi1, zmHPhi2);
|
|
@@ -1764,8 +1923,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
|
|
|
1764
1923
|
}
|
|
1765
1924
|
update_e_gpu<2><<<grid_fields, blockSize, 0, stream_comp>>>(ce_hist, ce_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
|
|
1766
1925
|
CUDA_CHECK_LAST();
|
|
1767
|
-
if (
|
|
1768
|
-
cpml_e_gpu<2><<<
|
|
1926
|
+
if (has_cpml_e) {
|
|
1927
|
+
cpml_e_gpu<2><<<grid_cpml_e, blockSize, 0, stream_comp>>>(
|
|
1769
1928
|
Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
1770
1929
|
pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
|
|
1771
1930
|
x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2, z0EPhi1, z0EPhi2, zmEPhi1, zmEPhi2);
|
|
@@ -1964,6 +2123,20 @@ DEEPGPR_API void backward(const float* __restrict__ eps_r_pad, const float* __re
|
|
|
1964
2123
|
long long total_fields = (long long)NX_FIELDS * NY_FIELDS * NZ_FIELDS;
|
|
1965
2124
|
dim3 grid_material(CEIL_DIV(total_fields, blockSize));
|
|
1966
2125
|
dim3 grid_fields(CEIL_DIV((long long)step * total_fields, blockSize));
|
|
2126
|
+
long long cpml_e_region_max = cpml_max_region_cells(
|
|
2127
|
+
NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
2128
|
+
pml0, pml1, pml2, pml3, pml4, pml5, 1);
|
|
2129
|
+
long long cpml_h_region_max = cpml_max_region_cells(
|
|
2130
|
+
NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
2131
|
+
pml0, pml1, pml2, pml3, pml4, pml5, 0);
|
|
2132
|
+
int has_cpml_e = has_cpml && cpml_e_region_max > 0;
|
|
2133
|
+
int has_cpml_h = has_cpml && cpml_h_region_max > 0;
|
|
2134
|
+
dim3 grid_cpml_e(
|
|
2135
|
+
has_cpml_e ? CEIL_DIV(cpml_e_region_max, blockSize) : 1,
|
|
2136
|
+
6, step);
|
|
2137
|
+
dim3 grid_cpml_h(
|
|
2138
|
+
has_cpml_h ? CEIL_DIV(cpml_h_region_max, blockSize) : 1,
|
|
2139
|
+
6, step);
|
|
1967
2140
|
|
|
1968
2141
|
build_update_coeffs_gpu<<<grid_material, blockSize, 0, stream_comp>>>(eps_r_pad, sigma_pad, mu_r_pad, ce_hist, ce_curl, ce_rhs, ch_hist, ch_curl, ch_rhs, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dt, dx);
|
|
1969
2142
|
CUDA_CHECK_LAST();
|
|
@@ -2032,8 +2205,8 @@ DEEPGPR_API void backward(const float* __restrict__ eps_r_pad, const float* __re
|
|
|
2032
2205
|
}
|
|
2033
2206
|
|
|
2034
2207
|
/* E CPML^T -> E update^T -> H CPML^T -> H update^T. */
|
|
2035
|
-
if (
|
|
2036
|
-
LAUNCH_ORDER_KERNEL(adjoint_cpml_e_gpu,
|
|
2208
|
+
if (has_cpml_e) {
|
|
2209
|
+
LAUNCH_ORDER_KERNEL(adjoint_cpml_e_gpu, grid_cpml_e, blockSize, stream_comp, fdtd_order,
|
|
2037
2210
|
lambda_ex, lambda_ey, lambda_ez, lambda_hx, lambda_hy, lambda_hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
2038
2211
|
pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
|
|
2039
2212
|
x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2,
|
|
@@ -2044,8 +2217,8 @@ DEEPGPR_API void backward(const float* __restrict__ eps_r_pad, const float* __re
|
|
|
2044
2217
|
ce_hist, ce_curl, lambda_ex, lambda_ey, lambda_ez, lambda_hx, lambda_hy, lambda_hz,
|
|
2045
2218
|
step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
|
|
2046
2219
|
CUDA_CHECK_LAST();
|
|
2047
|
-
if (
|
|
2048
|
-
LAUNCH_ORDER_KERNEL(adjoint_cpml_h_gpu,
|
|
2220
|
+
if (has_cpml_h) {
|
|
2221
|
+
LAUNCH_ORDER_KERNEL(adjoint_cpml_h_gpu, grid_cpml_h, blockSize, stream_comp, fdtd_order,
|
|
2049
2222
|
lambda_ex, lambda_ey, lambda_ez, lambda_hx, lambda_hy, lambda_hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
|
|
2050
2223
|
pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
|
|
2051
2224
|
x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2,
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -67,23 +67,24 @@ def _locations(shape):
|
|
|
67
67
|
return torch.tensor([[center]], dtype=torch.int32)
|
|
68
68
|
|
|
69
69
|
|
|
70
|
-
def _state_problem(order, pml):
|
|
70
|
+
def _state_problem(order, pml, device="cpu"):
|
|
71
71
|
torch.manual_seed(1100 + order + pml)
|
|
72
|
+
device = torch.device(device)
|
|
72
73
|
shape = (9, 10, 8)
|
|
73
74
|
nx, ny, nz = shape
|
|
74
|
-
x = torch.linspace(0.0, 1.0, nx)[:, None, None]
|
|
75
|
-
y = torch.linspace(0.0, 1.0, ny)[None, :, None]
|
|
76
|
-
z = torch.linspace(0.0, 1.0, nz)[None, None, :]
|
|
75
|
+
x = torch.linspace(0.0, 1.0, nx, device=device)[:, None, None]
|
|
76
|
+
y = torch.linspace(0.0, 1.0, ny, device=device)[None, :, None]
|
|
77
|
+
z = torch.linspace(0.0, 1.0, nz, device=device)[None, None, :]
|
|
77
78
|
eps_r = 3.5 + 0.7 * x + 0.2 * y + 0.1 * z
|
|
78
79
|
sigma = 1.0e-4 + 1.0e-4 * (x + y + z) / 3.0
|
|
79
80
|
mu_r = 1.0 + 0.15 * x + 0.0 * y + 0.05 * z
|
|
80
|
-
location = _locations(shape)
|
|
81
|
-
source = torch.zeros((1, 1, 1), dtype=torch.float32)
|
|
81
|
+
location = _locations(shape).to(device)
|
|
82
|
+
source = torch.zeros((1, 1, 1), dtype=torch.float32, device=device)
|
|
82
83
|
spacing = (0.020, 0.017, 0.013)
|
|
83
84
|
dt = 1.0e-11
|
|
84
85
|
|
|
85
86
|
initial_e, initial_h, initial_cpml = DeepGPR.checkpoint_initial_field(
|
|
86
|
-
device=
|
|
87
|
+
device=device,
|
|
87
88
|
dx=spacing,
|
|
88
89
|
dt=dt,
|
|
89
90
|
source_amplitudes=source,
|
|
@@ -102,7 +103,7 @@ def _state_problem(order, pml):
|
|
|
102
103
|
]
|
|
103
104
|
state = [tensor.clone().requires_grad_(True) for tensor in base_state]
|
|
104
105
|
result = DeepGPR.compute(
|
|
105
|
-
device=
|
|
106
|
+
device=device,
|
|
106
107
|
dx=spacing,
|
|
107
108
|
dt=dt,
|
|
108
109
|
source_amplitudes=source,
|
|
@@ -269,6 +270,15 @@ class DiscreteAdjointTests(unittest.TestCase):
|
|
|
269
270
|
def test_native_cpml_single_step_dot_product_all_faces(self):
|
|
270
271
|
self.assertLess(_state_problem(order=4, pml=2), 3.0e-5)
|
|
271
272
|
|
|
273
|
+
@unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
|
|
274
|
+
def test_cuda_cpml_single_step_dot_products_orders_2_4_8(self):
|
|
275
|
+
for order in (2, 4, 8):
|
|
276
|
+
with self.subTest(order=order):
|
|
277
|
+
self.assertLess(
|
|
278
|
+
_state_problem(order=order, pml=2, device="cuda"),
|
|
279
|
+
2.0e-4,
|
|
280
|
+
)
|
|
281
|
+
|
|
272
282
|
def test_source_waveform_gradient_matches_finite_difference(self):
|
|
273
283
|
nx, ny, nt = 12, 14, 60
|
|
274
284
|
eps_r = torch.full((nx, ny), 4.0)
|
|
@@ -378,6 +388,143 @@ class DiscreteAdjointTests(unittest.TestCase):
|
|
|
378
388
|
for reference, candidate in (*zip(cpu, cuda), *zip(cuda, cuda_offload)):
|
|
379
389
|
torch.testing.assert_close(reference, candidate, rtol=5.0e-4, atol=1.0e-6)
|
|
380
390
|
|
|
391
|
+
@unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
|
|
392
|
+
def test_cuda_coincident_sources_and_receivers_accumulate(self):
|
|
393
|
+
device = torch.device("cuda")
|
|
394
|
+
nt = 48
|
|
395
|
+
eps_r = torch.full((12, 14), 4.0, device=device)
|
|
396
|
+
sigma = torch.zeros_like(eps_r)
|
|
397
|
+
first = torch.linspace(-0.5, 0.8, nt, device=device)
|
|
398
|
+
second = torch.linspace(0.2, -0.4, nt, device=device)
|
|
399
|
+
source_pair = torch.stack((first, second), dim=0).unsqueeze(-1)
|
|
400
|
+
source_sum = (first + second).reshape(1, nt, 1)
|
|
401
|
+
one_source = torch.tensor([[[5, 5, 0]]], dtype=torch.int32, device=device)
|
|
402
|
+
two_sources = one_source.repeat(1, 2, 1)
|
|
403
|
+
one_receiver = torch.tensor([[[5, 8, 0]]], dtype=torch.int32, device=device)
|
|
404
|
+
|
|
405
|
+
common = dict(
|
|
406
|
+
device=device, dx=0.02, dt=3.0e-11,
|
|
407
|
+
receiver_location=one_receiver,
|
|
408
|
+
er=eps_r, se=sigma, pmlthick=0,
|
|
409
|
+
fdtd_order=2, mode=2,
|
|
410
|
+
)
|
|
411
|
+
pair_data = DeepGPR.compute(
|
|
412
|
+
source_amplitudes=source_pair,
|
|
413
|
+
source_location=two_sources,
|
|
414
|
+
**common,
|
|
415
|
+
)[-1]
|
|
416
|
+
sum_data = DeepGPR.compute(
|
|
417
|
+
source_amplitudes=source_sum,
|
|
418
|
+
source_location=one_source,
|
|
419
|
+
**common,
|
|
420
|
+
)[-1]
|
|
421
|
+
torch.cuda.synchronize()
|
|
422
|
+
torch.testing.assert_close(pair_data, sum_data, rtol=2.0e-5, atol=1.0e-6)
|
|
423
|
+
|
|
424
|
+
def receiver_gradient(receiver_location):
|
|
425
|
+
model = eps_r.detach().clone().requires_grad_(True)
|
|
426
|
+
data = DeepGPR.compute(
|
|
427
|
+
device=device, dx=0.02, dt=3.0e-11,
|
|
428
|
+
source_amplitudes=source_sum,
|
|
429
|
+
source_location=one_source,
|
|
430
|
+
receiver_location=receiver_location,
|
|
431
|
+
er=model, se=sigma, pmlthick=0,
|
|
432
|
+
fdtd_order=2, mode=2,
|
|
433
|
+
model_gradient_sampling_interval=1,
|
|
434
|
+
wavefield_storage_dtype=torch.float32,
|
|
435
|
+
)[-1]
|
|
436
|
+
data.sum().backward()
|
|
437
|
+
return model.grad
|
|
438
|
+
|
|
439
|
+
single_gradient = receiver_gradient(one_receiver)
|
|
440
|
+
duplicate_gradient = receiver_gradient(one_receiver.repeat(1, 2, 1))
|
|
441
|
+
torch.cuda.synchronize()
|
|
442
|
+
torch.testing.assert_close(
|
|
443
|
+
duplicate_gradient, 2.0 * single_gradient, rtol=2.0e-4, atol=1.0e-6
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
@unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
|
|
447
|
+
def test_cuda_large_3d_all_face_cpml_async_bounds(self):
|
|
448
|
+
device = torch.device("cuda")
|
|
449
|
+
nx, ny, nz, nt = 120, 120, 100, 5
|
|
450
|
+
eps_r = torch.full(
|
|
451
|
+
(nx, ny, nz), 4.0, dtype=torch.float32, device=device,
|
|
452
|
+
requires_grad=True,
|
|
453
|
+
)
|
|
454
|
+
sigma = torch.full_like(eps_r, 1.0e-4)
|
|
455
|
+
source = torch.zeros((1, nt, 1), dtype=torch.float32, device=device)
|
|
456
|
+
source[0, 0, 0] = 1.0
|
|
457
|
+
source_location = torch.tensor(
|
|
458
|
+
[[[60, 60, 50]]], dtype=torch.int32
|
|
459
|
+
)
|
|
460
|
+
receiver_location = torch.tensor(
|
|
461
|
+
[[[60, 62, 50]]], dtype=torch.int32
|
|
462
|
+
)
|
|
463
|
+
|
|
464
|
+
result = DeepGPR.compute(
|
|
465
|
+
device=device,
|
|
466
|
+
dx=(0.020, 0.017, 0.014),
|
|
467
|
+
dt=1.0e-11,
|
|
468
|
+
source_amplitudes=source,
|
|
469
|
+
source_location=source_location,
|
|
470
|
+
receiver_location=receiver_location,
|
|
471
|
+
er=eps_r,
|
|
472
|
+
se=sigma,
|
|
473
|
+
pmlthick=10,
|
|
474
|
+
fdtd_order=2,
|
|
475
|
+
mode=3,
|
|
476
|
+
model_gradient_sampling_interval=3,
|
|
477
|
+
wavefield_storage_dtype=torch.float16,
|
|
478
|
+
use_async_offload=True,
|
|
479
|
+
)
|
|
480
|
+
final_fields = (*result[1], *result[2])
|
|
481
|
+
loss = result[-1].square().sum()
|
|
482
|
+
loss = loss + sum(field.square().mean() for field in final_fields)
|
|
483
|
+
loss.backward()
|
|
484
|
+
torch.cuda.synchronize()
|
|
485
|
+
|
|
486
|
+
self.assertTrue(torch.isfinite(result[-1]).all().item())
|
|
487
|
+
self.assertIsNotNone(eps_r.grad)
|
|
488
|
+
self.assertTrue(torch.isfinite(eps_r.grad).all().item())
|
|
489
|
+
|
|
490
|
+
@unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
|
|
491
|
+
def test_cuda_asymmetric_cpml_and_incomplete_sampling_block(self):
|
|
492
|
+
device = torch.device("cuda")
|
|
493
|
+
shape = (31, 29, 27)
|
|
494
|
+
nt = 11
|
|
495
|
+
eps_r = torch.full(shape, 4.0, device=device, requires_grad=True)
|
|
496
|
+
sigma = torch.full_like(eps_r, 2.0e-4, requires_grad=True)
|
|
497
|
+
source = torch.linspace(-0.3, 0.7, nt, device=device).reshape(1, nt, 1)
|
|
498
|
+
source_location = torch.tensor(
|
|
499
|
+
[[[15, 14, 13]]], dtype=torch.int32, device=device
|
|
500
|
+
)
|
|
501
|
+
receiver_location = torch.tensor(
|
|
502
|
+
[[[16, 15, 14]]], dtype=torch.int32, device=device
|
|
503
|
+
)
|
|
504
|
+
|
|
505
|
+
receiver = DeepGPR.compute(
|
|
506
|
+
device=device,
|
|
507
|
+
dx=(0.020, 0.016, 0.013),
|
|
508
|
+
dt=1.0e-11,
|
|
509
|
+
source_amplitudes=source,
|
|
510
|
+
source_location=source_location,
|
|
511
|
+
receiver_location=receiver_location,
|
|
512
|
+
er=eps_r,
|
|
513
|
+
se=sigma,
|
|
514
|
+
pmlthick=[3, 5, 4, 2, 6, 3],
|
|
515
|
+
fdtd_order=8,
|
|
516
|
+
mode=3,
|
|
517
|
+
model_gradient_sampling_interval=4,
|
|
518
|
+
wavefield_storage_dtype=torch.bfloat16,
|
|
519
|
+
use_async_offload=True,
|
|
520
|
+
)[-1]
|
|
521
|
+
receiver.square().sum().backward()
|
|
522
|
+
torch.cuda.synchronize()
|
|
523
|
+
|
|
524
|
+
self.assertTrue(torch.isfinite(receiver).all().item())
|
|
525
|
+
self.assertTrue(torch.isfinite(eps_r.grad).all().item())
|
|
526
|
+
self.assertTrue(torch.isfinite(sigma.grad).all().item())
|
|
527
|
+
|
|
381
528
|
|
|
382
529
|
if __name__ == "__main__":
|
|
383
530
|
unittest.main(verbosity=2)
|
|
@@ -113,6 +113,70 @@ class NumericsValidationTests(unittest.TestCase):
|
|
|
113
113
|
location, location, 0.02, 3.0e-11, 2, 2,
|
|
114
114
|
)
|
|
115
115
|
|
|
116
|
+
def test_initialization_rejects_empty_native_launch_dimensions(self):
|
|
117
|
+
source = torch.zeros((1, 4, 1))
|
|
118
|
+
location = torch.tensor([[[1, 1, 0]]], dtype=torch.int32)
|
|
119
|
+
|
|
120
|
+
with self.assertRaisesRegex(ValueError, "model dimensions"):
|
|
121
|
+
initialization(
|
|
122
|
+
"cpu", torch.empty((0, 4)), torch.empty((0, 4)), None,
|
|
123
|
+
source, location, location, 0.02, 3.0e-11, 0, 2,
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
empty_cases = (
|
|
127
|
+
(
|
|
128
|
+
torch.empty((0, 1, 3), dtype=torch.int32),
|
|
129
|
+
torch.empty((0, 1, 3), dtype=torch.int32),
|
|
130
|
+
"shot",
|
|
131
|
+
),
|
|
132
|
+
(
|
|
133
|
+
torch.empty((1, 0, 3), dtype=torch.int32),
|
|
134
|
+
location,
|
|
135
|
+
"source",
|
|
136
|
+
),
|
|
137
|
+
(
|
|
138
|
+
location,
|
|
139
|
+
torch.empty((1, 0, 3), dtype=torch.int32),
|
|
140
|
+
"receiver",
|
|
141
|
+
),
|
|
142
|
+
)
|
|
143
|
+
er = torch.full((4, 4), 4.0)
|
|
144
|
+
se = torch.zeros_like(er)
|
|
145
|
+
for source_location, receiver_location, message in empty_cases:
|
|
146
|
+
with self.subTest(message=message), self.assertRaisesRegex(ValueError, message):
|
|
147
|
+
initialization(
|
|
148
|
+
"cpu", er, se, None, source,
|
|
149
|
+
source_location, receiver_location, 0.02, 3.0e-11, 0, 2,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
def test_initialization_validates_coordinates_before_int32_conversion(self):
|
|
153
|
+
er = torch.full((4, 4), 4.0)
|
|
154
|
+
se = torch.zeros_like(er)
|
|
155
|
+
source = torch.zeros((1, 4, 1))
|
|
156
|
+
location = torch.tensor([[[1, 1, 0]]], dtype=torch.int64)
|
|
157
|
+
|
|
158
|
+
fractional = location.to(torch.float64)
|
|
159
|
+
fractional[0, 0, 1] = 1.5
|
|
160
|
+
with self.assertRaisesRegex(ValueError, "integer-valued"):
|
|
161
|
+
initialization(
|
|
162
|
+
"cpu", er, se, None, source,
|
|
163
|
+
fractional, location, 0.02, 3.0e-11, 0, 2,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
wrapped = location.clone()
|
|
167
|
+
wrapped[0, 0, 0] = 2**32 + 1
|
|
168
|
+
with self.assertRaisesRegex(ValueError, "out of range"):
|
|
169
|
+
initialization(
|
|
170
|
+
"cpu", er, se, None, source,
|
|
171
|
+
wrapped, location, 0.02, 3.0e-11, 0, 2,
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
with self.assertRaisesRegex(ValueError, "finite integer"):
|
|
175
|
+
initialization(
|
|
176
|
+
"cpu", er, se, None, source,
|
|
177
|
+
location, location, 0.02, 3.0e-11, [0, 0, 0.5, 0], 2,
|
|
178
|
+
)
|
|
179
|
+
|
|
116
180
|
def test_wavefield_storage_aliases(self):
|
|
117
181
|
self.assertIs(_normalize_wavefield_storage_dtype("fp32"), torch.float32)
|
|
118
182
|
self.assertIs(_normalize_wavefield_storage_dtype("fp16"), torch.float16)
|
|
@@ -228,6 +292,41 @@ class NumericsValidationTests(unittest.TestCase):
|
|
|
228
292
|
_pml_phi_elements(nx, ny, nz, nstep, pml),
|
|
229
293
|
)
|
|
230
294
|
|
|
295
|
+
def test_cpml_checkpoint_shapes_are_validated_before_native_call(self):
|
|
296
|
+
nx, ny, nt = 8, 10, 4
|
|
297
|
+
er = torch.full((nx, ny), 4.0)
|
|
298
|
+
se = torch.zeros_like(er)
|
|
299
|
+
source = torch.zeros((1, nt, 1))
|
|
300
|
+
location = torch.tensor([[[4, 5, 0]]], dtype=torch.int32)
|
|
301
|
+
_, _, pml_state = DeepGPR.checkpoint_initial_field(
|
|
302
|
+
device="cpu",
|
|
303
|
+
dx=0.02,
|
|
304
|
+
dt=3.0e-11,
|
|
305
|
+
source_amplitudes=source,
|
|
306
|
+
source_location=location,
|
|
307
|
+
receiver_location=location,
|
|
308
|
+
er=er,
|
|
309
|
+
se=se,
|
|
310
|
+
pmlthick=2,
|
|
311
|
+
)
|
|
312
|
+
bad_state = list(pml_state)
|
|
313
|
+
bad_state[0] = bad_state[0][..., :-1]
|
|
314
|
+
|
|
315
|
+
with self.assertRaisesRegex(ValueError, "CPML face 0 component 0"):
|
|
316
|
+
DeepGPR.compute(
|
|
317
|
+
device="cpu",
|
|
318
|
+
dx=0.02,
|
|
319
|
+
dt=3.0e-11,
|
|
320
|
+
source_amplitudes=source,
|
|
321
|
+
source_location=location,
|
|
322
|
+
receiver_location=location,
|
|
323
|
+
er=er,
|
|
324
|
+
se=se,
|
|
325
|
+
pmlthick=2,
|
|
326
|
+
PML=tuple(bad_state),
|
|
327
|
+
mode=2,
|
|
328
|
+
)
|
|
329
|
+
|
|
231
330
|
def test_anisotropic_cpml_coefficients_use_axis_spacing(self):
|
|
232
331
|
nx, ny, nz = 10, 12, 14
|
|
233
332
|
er = torch.full((nx, ny, nz), 4.0)
|
|
Binary file
|
|
Binary file
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|