DeepGPR 0.0.16__tar.gz → 0.0.18__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. {deepgpr-0.0.16 → deepgpr-0.0.18}/PKG-INFO +1 -1
  2. {deepgpr-0.0.16 → deepgpr-0.0.18}/pyproject.toml +1 -1
  3. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/common.py +150 -22
  4. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/compute2.py +6 -2
  5. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr.cu +246 -73
  6. deepgpr-0.0.18/src/DeepGPR/lib/deepgpr.dll +0 -0
  7. deepgpr-0.0.18/src/DeepGPR/lib/deepgpr.so +0 -0
  8. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.dll +0 -0
  9. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/PKG-INFO +1 -1
  10. {deepgpr-0.0.16 → deepgpr-0.0.18}/tests/test_discrete_adjoint.py +155 -8
  11. {deepgpr-0.0.16 → deepgpr-0.0.18}/tests/test_numerics.py +99 -0
  12. deepgpr-0.0.16/src/DeepGPR/lib/deepgpr.dll +0 -0
  13. deepgpr-0.0.16/src/DeepGPR/lib/deepgpr.so +0 -0
  14. {deepgpr-0.0.16 → deepgpr-0.0.18}/README.md +0 -0
  15. {deepgpr-0.0.16 → deepgpr-0.0.18}/license +0 -0
  16. {deepgpr-0.0.16 → deepgpr-0.0.18}/setup.cfg +0 -0
  17. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/__init__.py +0 -0
  18. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr.h +0 -0
  19. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.c +0 -0
  20. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.dylib +0 -0
  21. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/deepgpr_cpu.so +0 -0
  22. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/lib/libomp.dylib +0 -0
  23. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/multiscale.py +0 -0
  24. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR/wavelet.py +0 -0
  25. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/SOURCES.txt +0 -0
  26. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/dependency_links.txt +0 -0
  27. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/requires.txt +0 -0
  28. {deepgpr-0.0.16 → deepgpr-0.0.18}/src/DeepGPR.egg-info/top_level.txt +0 -0
  29. {deepgpr-0.0.16 → deepgpr-0.0.18}/tests/test_wavelet.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: DeepGPR
3
- Version: 0.0.16
3
+ Version: 0.0.18
4
4
  Summary: PyTorch and CUDA for GPR FWI
5
5
  Author-email: Lei Liu <liulei990222@gmail.com>
6
6
  License-Expression: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "DeepGPR"
7
- version = "0.0.16"
7
+ version = "0.0.18"
8
8
  authors = [
9
9
  { name = "Lei Liu", email = "liulei990222@gmail.com" }
10
10
  ]
@@ -80,6 +80,7 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
80
80
  pmlthick: PML thickness as an int, list, or tensor.
81
81
  fdtd_order: Spatial finite-difference order used for the CFL check.
82
82
  """
83
+ device = torch.device("cpu" if device is None else device)
83
84
  dtype=torch.float32
84
85
  spacing = _normalize_grid_spacing(dx)
85
86
  try:
@@ -100,6 +101,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
100
101
  elif len(se.shape) != 3:
101
102
  raise ValueError('The shape of sigma should be 2-d or 3-d.')
102
103
 
104
+ if mr is not None and not torch.is_tensor(mr):
105
+ raise TypeError("mr must be a PyTorch tensor or None.")
103
106
  if mr is not None:
104
107
  if len(mr.shape) == 2:
105
108
  mr = mr.reshape(*mr.shape, 1)
@@ -108,6 +111,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
108
111
 
109
112
  if er.shape != se.shape:
110
113
  raise ValueError('The shape of epsilon and sigma should be the same.')
114
+ if any(size < 1 for size in er.shape):
115
+ raise ValueError("The material model dimensions must all be non-empty.")
111
116
  nx, ny, nz = er.shape
112
117
  mode = 2 if nz == 1 else 3
113
118
  er=er.to(device=device, dtype=dtype)
@@ -119,17 +124,36 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
119
124
  else:
120
125
  raise ValueError('The shape of mr should be the same as epsilon and sigma.')
121
126
 
122
- if source_location.shape[0] != receiver_location.shape[0]:
123
- raise ValueError('The first dimension (nstep) of source_location and receiver_location should be the same.')
127
+ if not torch.is_tensor(source_location) or not torch.is_tensor(receiver_location):
128
+ raise TypeError("source_location and receiver_location must be PyTorch tensors.")
124
129
  if source_location.ndim != 3 or receiver_location.ndim != 3:
125
130
  raise ValueError("source_location and receiver_location must have shape (nstep, count, 3).")
126
131
  if source_location.shape[2] != 3 or receiver_location.shape[2] != 3:
127
132
  raise ValueError("The last dimension of source_location and receiver_location must be 3.")
133
+ if source_location.shape[0] != receiver_location.shape[0]:
134
+ raise ValueError('The first dimension (nstep) of source_location and receiver_location should be the same.')
128
135
  nstep=source_location.shape[0]
129
136
  nsr=source_location.shape[1]
130
137
  nrx=receiver_location.shape[1]
131
- source_location=source_location.to(device=device, dtype=torch.int32).contiguous()
132
- receiver_location=receiver_location.to(device=device, dtype=torch.int32).contiguous()
138
+ if nstep < 1:
139
+ raise ValueError("At least one shot is required.")
140
+ if nsr < 1:
141
+ raise ValueError("At least one source per shot is required.")
142
+ if nrx < 1:
143
+ raise ValueError("At least one receiver per shot is required.")
144
+ if device.type == "cuda" and nstep > 65535:
145
+ raise ValueError(
146
+ "CUDA supports at most 65535 shots in one DeepGPR compute call. "
147
+ "Split a larger acquisition batch into smaller calls."
148
+ )
149
+ for name, locations in (
150
+ ("source_location", source_location),
151
+ ("receiver_location", receiver_location),
152
+ ):
153
+ if locations.dtype == torch.bool or locations.is_complex():
154
+ raise TypeError(f"{name} must contain real integer-valued coordinates.")
155
+ source_location=source_location.to(device=device).contiguous()
156
+ receiver_location=receiver_location.to(device=device).contiguous()
133
157
 
134
158
  if not torch.is_tensor(source_amplitudes):
135
159
  raise TypeError("source_amplitudes must be a PyTorch tensor.")
@@ -165,6 +189,16 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
165
189
  )
166
190
 
167
191
  shape_tensor = torch.tensor((nx, ny, nz), dtype=torch.int32, device=device)
192
+ source_integral = (
193
+ (torch.isfinite(source_location) & (source_location == source_location.trunc())).all()
194
+ if source_location.is_floating_point()
195
+ else torch.ones((), dtype=torch.bool, device=device)
196
+ )
197
+ receiver_integral = (
198
+ (torch.isfinite(receiver_location) & (receiver_location == receiver_location.trunc())).all()
199
+ if receiver_location.is_floating_point()
200
+ else torch.ones((), dtype=torch.bool, device=device)
201
+ )
168
202
  source_valid = ((source_location >= 0) & (source_location < shape_tensor)).all()
169
203
  receiver_valid = ((receiver_location >= 0) & (receiver_location < shape_tensor)).all()
170
204
  source_in_pml = _locations_in_pml(source_location, (nx, ny, nz), pml_values)
@@ -181,6 +215,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
181
215
  (er.detach() * mr.detach()).amin(),
182
216
  source_valid.to(dtype),
183
217
  receiver_valid.to(dtype),
218
+ source_integral.to(dtype),
219
+ receiver_integral.to(dtype),
184
220
  source_in_pml.sum().to(dtype),
185
221
  receiver_in_pml.sum().to(dtype),
186
222
  )
@@ -188,7 +224,8 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
188
224
  er_finite, se_finite, mr_finite, source_finite = (bool(value) for value in stats[:4])
189
225
  er_min, se_min, mr_min, min_er_mr = stats[4:8]
190
226
  source_valid, receiver_valid = (bool(value) for value in stats[8:10])
191
- source_pml_count, receiver_pml_count = (int(value) for value in stats[10:12])
227
+ source_integral, receiver_integral = (bool(value) for value in stats[10:12])
228
+ source_pml_count, receiver_pml_count = (int(value) for value in stats[12:14])
192
229
 
193
230
  for name, finite in (
194
231
  ("er", er_finite), ("se", se_finite), ("mr", mr_finite),
@@ -202,6 +239,10 @@ def initialization(device, er,se,mr,source_amplitudes,source_location,receiver_l
202
239
  raise ValueError('The values of sigma is incorrect.(should be non-negative)')
203
240
  if mr_min <= 0:
204
241
  raise ValueError('The values of mr are incorrect (must be positive).')
242
+ if not source_integral:
243
+ raise ValueError("source_location must contain finite integer-valued coordinates.")
244
+ if not receiver_integral:
245
+ raise ValueError("receiver_location must contain finite integer-valued coordinates.")
205
246
  if not source_valid:
206
247
  raise ValueError(
207
248
  "Error: Source coordinates out of range! "
@@ -292,24 +333,46 @@ def pmlthick_revert(p, er):
292
333
  p: PML thickness as an int, list, or tensor.
293
334
  er: Relative permittivity tensor used to detect 2D or 3D mode.
294
335
  """
295
- if isinstance(p, int):
296
- if er.shape[2] == 1:
297
- return torch.tensor([p, p, p, p, 0, 0], dtype=torch.int32)
298
- return torch.tensor([p]*6, dtype=torch.int32)
299
-
300
- elif isinstance(p, list):
301
- if len(p) == 6:
302
- return torch.tensor(p, dtype=torch.int32)
303
- elif len(p) == 4:
304
- return torch.tensor(p + [0, 0], dtype=torch.int32)
336
+ if isinstance(p, bool):
337
+ raise TypeError("PML thickness must contain integer values, not bool.")
338
+ if isinstance(p, int):
339
+ values = [p, p, p, p, 0, 0] if er.shape[2] == 1 else [p] * 6
340
+ elif isinstance(p, (list, tuple)):
341
+ if len(p) == 4:
342
+ values = [*p, 0, 0]
343
+ elif len(p) == 6:
344
+ values = list(p)
305
345
  else:
306
- raise ValueError(f"Unsupported list length: {len(p)}. Must be 4 or 6.")
307
-
346
+ raise ValueError(f"Unsupported PML length: {len(p)}. Must be 4 or 6.")
308
347
  elif isinstance(p, torch.Tensor):
309
- return p.detach().to(device="cpu", dtype=torch.int32)
310
-
348
+ if p.ndim != 1:
349
+ raise ValueError("PML thickness tensor must be one-dimensional.")
350
+ if p.numel() == 4:
351
+ values = [*p.detach().cpu().tolist(), 0, 0]
352
+ elif p.numel() == 6:
353
+ values = p.detach().cpu().tolist()
354
+ else:
355
+ raise ValueError(
356
+ f"Unsupported PML length: {p.numel()}. Must be 4 or 6."
357
+ )
311
358
  else:
312
- raise TypeError(f"Unsupported type: {type(p)}")
359
+ raise TypeError(f"Unsupported PML thickness type: {type(p)}")
360
+
361
+ normalized = []
362
+ for value in values:
363
+ if isinstance(value, bool):
364
+ raise TypeError("PML thickness must contain integer values, not bool.")
365
+ try:
366
+ numeric = float(value)
367
+ except (TypeError, ValueError, OverflowError) as exc:
368
+ raise TypeError("PML thickness must contain finite integer values.") from exc
369
+ if not math.isfinite(numeric) or not numeric.is_integer():
370
+ raise ValueError("PML thickness must contain finite integer values.")
371
+ integer = int(numeric)
372
+ if integer < 0 or integer > torch.iinfo(torch.int32).max:
373
+ raise ValueError("PML thickness values must fit in non-negative int32.")
374
+ normalized.append(integer)
375
+ return torch.tensor(normalized, dtype=torch.int32)
313
376
 
314
377
 
315
378
  class TVRegularization(nn.Module):
@@ -734,8 +797,29 @@ def build_pml_phi(x0,xm,y0,ym,z0,zm,nstep,PML,device):
734
797
  z0EPhi1, z0EPhi2, z0HPhi1, z0HPhi2,
735
798
  zmEPhi1, zmEPhi2, zmHPhi1, zmHPhi2) = [torch.empty(0) for _ in range(24)]
736
799
 
737
- if PML==None:
738
- PML=(None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None,None)
800
+ descriptors = (x0, xm, y0, ym, z0, zm)
801
+ if PML is None:
802
+ PML = (None,) * 24
803
+ elif not isinstance(PML, (list, tuple)) or len(PML) != 24:
804
+ raise ValueError("PML must contain exactly 24 CPML auxiliary tensors.")
805
+
806
+ for face, descriptor in enumerate(descriptors):
807
+ group = PML[4 * face:4 * face + 4]
808
+ supplied = tuple(value is not None for value in group)
809
+ if any(supplied) and not all(supplied):
810
+ raise ValueError(
811
+ f"CPML face {face} must provide all four auxiliary tensors or none."
812
+ )
813
+ if all(supplied) and not all(torch.is_tensor(value) for value in group):
814
+ raise TypeError(f"CPML face {face} state must contain PyTorch tensors.")
815
+ if (
816
+ all(supplied)
817
+ and descriptor.numel() == 0
818
+ and any(value.numel() != 0 for value in group)
819
+ ):
820
+ raise ValueError(
821
+ f"CPML face {face} state must be empty for a zero-thickness boundary."
822
+ )
739
823
 
740
824
  if x0.numel()!=0 and PML[0]==None:
741
825
  x0EPhi1=torch.zeros((nstep, int(x0[2]-x0[1]+1), int(x0[4]-x0[3]), int(x0[6]-x0[5]+1)),dtype=torch.float, device=device)
@@ -811,6 +895,50 @@ def build_pml_phi(x0,xm,y0,ym,z0,zm,nstep,PML,device):
811
895
  z0EPhi1,z0EPhi2,z0HPhi1,z0HPhi2,
812
896
  zmEPhi1,zmEPhi2,zmHPhi1,zmHPhi2,
813
897
  )
898
+
899
+ def expected_shapes(axis, descriptor):
900
+ if descriptor.numel() == 0:
901
+ return (None,) * 4
902
+ a = int(descriptor[2] - descriptor[1])
903
+ b = int(descriptor[4] - descriptor[3])
904
+ c = int(descriptor[6] - descriptor[5])
905
+ if axis == 0:
906
+ return (
907
+ (nstep, a + 1, b, c + 1),
908
+ (nstep, a + 1, b + 1, c),
909
+ (nstep, a, b + 1, c),
910
+ (nstep, a, b, c + 1),
911
+ )
912
+ if axis == 1:
913
+ return (
914
+ (nstep, a, b + 1, c + 1),
915
+ (nstep, a + 1, b + 1, c),
916
+ (nstep, a + 1, b, c),
917
+ (nstep, a, b, c + 1),
918
+ )
919
+ return (
920
+ (nstep, a, b + 1, c + 1),
921
+ (nstep, a + 1, b, c + 1),
922
+ (nstep, a + 1, b, c),
923
+ (nstep, a, b + 1, c),
924
+ )
925
+
926
+ for face, descriptor in enumerate(descriptors):
927
+ group = tensors[4 * face:4 * face + 4]
928
+ for component, (tensor, expected) in enumerate(
929
+ zip(group, expected_shapes(face // 2, descriptor))
930
+ ):
931
+ if expected is None:
932
+ if tensor.numel() != 0:
933
+ raise ValueError(
934
+ f"CPML face {face} component {component} must be empty."
935
+ )
936
+ elif tuple(tensor.shape) != expected:
937
+ raise ValueError(
938
+ f"CPML face {face} component {component} has shape "
939
+ f"{tuple(tensor.shape)}; expected {expected}."
940
+ )
941
+
814
942
  return tuple(
815
943
  tensor.to(device=device, dtype=torch.float32).contiguous()
816
944
  for tensor in tensors
@@ -700,8 +700,12 @@ class DeepGPR(torch.autograd.Function):
700
700
  """
701
701
 
702
702
  source_amplitudes = source_amplitudes.contiguous()
703
- source_location=source_location.to(torch.int32).contiguous()
704
- receiver_location=receiver_location.to(torch.int32).contiguous()
703
+ source_location=source_location.to(
704
+ device=device, dtype=torch.int32
705
+ ).contiguous()
706
+ receiver_location=receiver_location.to(
707
+ device=device, dtype=torch.int32
708
+ ).contiguous()
705
709
  c_lib = get_deepgpr_lib(device)
706
710
  set_library_fdtd_order(c_lib, fdtd_order)
707
711
  ctx.save_for_backward(mu_r, source_location, receiver_location,
@@ -175,6 +175,146 @@ __device__ __forceinline__ float load_wavefield_value_device(
175
175
  } \
176
176
  } while (0)
177
177
 
178
+ static int clamp_cpml_index(int value, int extent)
179
+ {
180
+ if (value < 0) return 0;
181
+ if (value > extent) return extent;
182
+ return value;
183
+ }
184
+
185
+ /*
186
+ * Partition the CPML shell into six disjoint boxes. The x boxes own edges and
187
+ * corners, the y boxes exclude the x boxes, and the z boxes exclude both.
188
+ * Return the largest box so one grid row can be assigned to each face.
189
+ */
190
+ static long long cpml_max_region_cells(
191
+ int NX, int NY, int NZ,
192
+ int pml0, int pml1, int pml2, int pml3, int pml4, int pml5,
193
+ int electric)
194
+ {
195
+ int x_low_end = pml0 > 0
196
+ ? clamp_cpml_index(pml0 + (electric ? 1 : 0), NX) : 0;
197
+ int y_low_end = pml2 > 0
198
+ ? clamp_cpml_index(pml2 + (electric ? 1 : 0), NY) : 0;
199
+ int z_low_end = pml4 > 0
200
+ ? clamp_cpml_index(pml4 + (electric ? 1 : 0), NZ) : 0;
201
+ int x_high_begin = pml1 > 0
202
+ ? clamp_cpml_index(NX - 1 - pml1, NX) : NX;
203
+ int y_high_begin = pml3 > 0
204
+ ? clamp_cpml_index(NY - 1 - pml3, NY) : NY;
205
+ int z_high_begin = pml5 > 0
206
+ ? clamp_cpml_index(NZ - 1 - pml5, NZ) : NZ;
207
+
208
+ int x_mid_end = x_high_begin > x_low_end ? x_high_begin : x_low_end;
209
+ int y_mid_end = y_high_begin > y_low_end ? y_high_begin : y_low_end;
210
+ int z_mid_end = z_high_begin > z_low_end ? z_high_begin : z_low_end;
211
+ long long x_mid = x_mid_end - x_low_end;
212
+ long long y_mid = y_mid_end - y_low_end;
213
+
214
+ long long regions[6] = {
215
+ (long long)x_low_end * NY * NZ,
216
+ pml1 > 0 ? (long long)(NX - x_mid_end) * NY * NZ : 0,
217
+ x_mid * (long long)y_low_end * NZ,
218
+ pml3 > 0 ? x_mid * (long long)(NY - y_mid_end) * NZ : 0,
219
+ x_mid * y_mid * (long long)z_low_end,
220
+ pml5 > 0 ? x_mid * y_mid * (long long)(NZ - z_mid_end) : 0,
221
+ };
222
+ long long maximum = 0;
223
+ for (int region = 0; region < 6; ++region) {
224
+ if (regions[region] > maximum) maximum = regions[region];
225
+ }
226
+ return maximum;
227
+ }
228
+
229
+ template<int ELECTRIC>
230
+ __device__ inline bool map_cpml_region_work(
231
+ int region, long long local,
232
+ int NX, int NY, int NZ,
233
+ int pml0, int pml1, int pml2, int pml3, int pml4, int pml5,
234
+ long long* i, long long* j, long long* k)
235
+ {
236
+ int x_low_end = pml0 > 0 ? pml0 + ELECTRIC : 0;
237
+ int y_low_end = pml2 > 0 ? pml2 + ELECTRIC : 0;
238
+ int z_low_end = pml4 > 0 ? pml4 + ELECTRIC : 0;
239
+ if (x_low_end < 0) x_low_end = 0;
240
+ if (y_low_end < 0) y_low_end = 0;
241
+ if (z_low_end < 0) z_low_end = 0;
242
+ if (x_low_end > NX) x_low_end = NX;
243
+ if (y_low_end > NY) y_low_end = NY;
244
+ if (z_low_end > NZ) z_low_end = NZ;
245
+
246
+ int x_high_begin = pml1 > 0 ? NX - 1 - pml1 : NX;
247
+ int y_high_begin = pml3 > 0 ? NY - 1 - pml3 : NY;
248
+ int z_high_begin = pml5 > 0 ? NZ - 1 - pml5 : NZ;
249
+ if (x_high_begin < 0) x_high_begin = 0;
250
+ if (y_high_begin < 0) y_high_begin = 0;
251
+ if (z_high_begin < 0) z_high_begin = 0;
252
+
253
+ int x_mid_end = x_high_begin > x_low_end ? x_high_begin : x_low_end;
254
+ int y_mid_end = y_high_begin > y_low_end ? y_high_begin : y_low_end;
255
+ int z_mid_end = z_high_begin > z_low_end ? z_high_begin : z_low_end;
256
+
257
+ int i0 = 0, ni = 0;
258
+ int j0 = 0, nj = 0;
259
+ int k0 = 0, nk = 0;
260
+ switch (region) {
261
+ case 0:
262
+ ni = x_low_end;
263
+ nj = NY;
264
+ nk = NZ;
265
+ break;
266
+ case 1:
267
+ if (pml1 <= 0) return false;
268
+ i0 = x_mid_end;
269
+ ni = NX - x_mid_end;
270
+ nj = NY;
271
+ nk = NZ;
272
+ break;
273
+ case 2:
274
+ i0 = x_low_end;
275
+ ni = x_mid_end - x_low_end;
276
+ nj = y_low_end;
277
+ nk = NZ;
278
+ break;
279
+ case 3:
280
+ if (pml3 <= 0) return false;
281
+ i0 = x_low_end;
282
+ ni = x_mid_end - x_low_end;
283
+ j0 = y_mid_end;
284
+ nj = NY - y_mid_end;
285
+ nk = NZ;
286
+ break;
287
+ case 4:
288
+ i0 = x_low_end;
289
+ ni = x_mid_end - x_low_end;
290
+ j0 = y_low_end;
291
+ nj = y_mid_end - y_low_end;
292
+ nk = z_low_end;
293
+ break;
294
+ case 5:
295
+ if (pml5 <= 0) return false;
296
+ i0 = x_low_end;
297
+ ni = x_mid_end - x_low_end;
298
+ j0 = y_low_end;
299
+ nj = y_mid_end - y_low_end;
300
+ k0 = z_mid_end;
301
+ nk = NZ - z_mid_end;
302
+ break;
303
+ default:
304
+ return false;
305
+ }
306
+
307
+ if (ni <= 0 || nj <= 0 || nk <= 0) return false;
308
+ long long jk = (long long)nj * nk;
309
+ long long region_size = (long long)ni * jk;
310
+ if (local >= region_size) return false;
311
+ *i = i0 + local / jk;
312
+ long long remainder = local % jk;
313
+ *j = j0 + remainder / nk;
314
+ *k = k0 + remainder % nk;
315
+ return true;
316
+ }
317
+
178
318
  static int g_fdtd_order = 2;
179
319
 
180
320
  DEEPGPR_API int deepgpr_abi_version(void)
@@ -439,7 +579,7 @@ __global__ void build_update_coeffs_gpu(const float* __restrict__ eps_r_pad, con
439
579
  float* __restrict__ ch_hist, float* __restrict__ ch_curl, float* __restrict__ ch_rhs,
440
580
  int NX_FIELDS, int NY_FIELDS, int NZ_FIELDS, float dt, float dx)
441
581
  {
442
- long long idx = blockIdx.x * blockDim.x + threadIdx.x;
582
+ long long idx = (long long)blockIdx.x * blockDim.x + threadIdx.x;
443
583
  long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
444
584
  if (idx >= (long long)NX_FIELDS * ny_nz) return;
445
585
 
@@ -489,7 +629,7 @@ __global__ void sample_receivers_gpu(
489
629
  int NX, int NY, int NZ, int N_ITER, int receiver_component)
490
630
  {
491
631
  long long field_stride = (long long)NX * NY * NZ;
492
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
632
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
493
633
  long long total = (long long)step * NRX;
494
634
  if (work >= total) return;
495
635
 
@@ -530,7 +670,7 @@ __global__ void inject_sources_gpu(
530
670
  int NX, int NY, int NZ, int nsrc, int polarisation, int nt)
531
671
  {
532
672
  long long field_stride = (long long)NX * NY * NZ;
533
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
673
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
534
674
  long long total = (long long)step * nsrc;
535
675
  if (work >= total) return;
536
676
 
@@ -547,9 +687,10 @@ __global__ void inject_sources_gpu(
547
687
  long long id3 = i * NY * NZ + j * NZ + k;
548
688
  long long id4 = (long long)s * field_stride + id3;
549
689
 
550
- if (polarisation == 0) Ex[id4] -= ce_rhs[id3] * scale;
551
- else if (polarisation == 1) Ey[id4] -= ce_rhs[id3] * scale;
552
- else if (polarisation == 2) Ez[id4] -= ce_rhs[id3] * scale;
690
+ float value = -ce_rhs[id3] * scale;
691
+ if (polarisation == 0) atomicAdd(&Ex[id4], value);
692
+ else if (polarisation == 1) atomicAdd(&Ey[id4], value);
693
+ else if (polarisation == 2) atomicAdd(&Ez[id4], value);
553
694
  }
554
695
 
555
696
 
@@ -577,9 +718,9 @@ __global__ void inject_sources_and_sample_gpu(
577
718
  float scale = srcwaveforms[src * nt + iteration] * dipole_length / (dx * dy * dz);
578
719
  float value = ce_rhs[material_idx] * scale;
579
720
 
580
- if (source_component == 0) Ex[field_idx] -= value;
581
- else if (source_component == 1) Ey[field_idx] -= value;
582
- else Ez[field_idx] -= value;
721
+ if (source_component == 0) atomicAdd(&Ex[field_idx], -value);
722
+ else if (source_component == 1) atomicAdd(&Ey[field_idx], -value);
723
+ else atomicAdd(&Ez[field_idx], -value);
583
724
  }
584
725
 
585
726
  __syncthreads();
@@ -608,7 +749,7 @@ __global__ void update_e_gpu(
608
749
  {
609
750
  long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
610
751
  long long field_stride = (long long)NX_FIELDS * ny_nz;
611
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
752
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
612
753
  if (work >= (long long)step * field_stride) return;
613
754
 
614
755
  long long idx = work % field_stride;
@@ -667,16 +808,16 @@ __global__ void cpml_e_gpu(
667
808
  {
668
809
  long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
669
810
  long long field_stride = (long long)NX_FIELDS * ny_nz;
670
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
671
- if (work >= (long long)step * field_stride) return;
672
-
673
- int s = (int)(work / field_stride);
674
- long long idx = work % field_stride;
675
-
676
- long long i = idx / ny_nz;
677
- long long rem = idx % ny_nz;
678
- long long j = rem / NZ_FIELDS;
679
- long long k = rem % NZ_FIELDS;
811
+ long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
812
+ int region = (int)blockIdx.y;
813
+ int s = (int)blockIdx.z;
814
+ long long i, j, k;
815
+ if (s >= step ||
816
+ !map_cpml_region_work<1>(region, region_work,
817
+ NX_FIELDS, NY_FIELDS, NZ_FIELDS,
818
+ pml0, pml1, pml2, pml3, pml4, pml5, &i, &j, &k)) return;
819
+ long long idx = i * ny_nz + j * NZ_FIELDS + k;
820
+ long long work = (long long)s * field_stride + idx;
680
821
 
681
822
  bool in_x0 = (pml0 > 0 && i > 0 && i <= pml0 && j < NY_FIELDS && k < NZ_FIELDS);
682
823
  bool in_xm = (pml1 > 0 && i >= NX_FIELDS - 1 - pml1 && i < NX_FIELDS - 1 && j < NY_FIELDS && k < NZ_FIELDS);
@@ -818,7 +959,7 @@ __global__ void update_h_gpu(
818
959
  {
819
960
  long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
820
961
  long long field_stride = (long long)NX_FIELDS * ny_nz;
821
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
962
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
822
963
  if (work >= (long long)step * field_stride) return;
823
964
 
824
965
  long long idx = work % field_stride;
@@ -862,7 +1003,7 @@ __global__ void adjoint_e_gpu(
862
1003
  {
863
1004
  long long ny_nz = (long long)NY * NZ;
864
1005
  long long field_stride = (long long)NX * ny_nz;
865
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1006
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
866
1007
  if (work >= (long long)step * field_stride) return;
867
1008
  long long idx = work % field_stride;
868
1009
  long long i = idx / ny_nz;
@@ -904,7 +1045,7 @@ __global__ void adjoint_h_gpu(
904
1045
  {
905
1046
  long long ny_nz = (long long)NY * NZ;
906
1047
  long long field_stride = (long long)NX * ny_nz;
907
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1048
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
908
1049
  if (work >= (long long)step * field_stride) return;
909
1050
  long long idx = work % field_stride;
910
1051
  long long i = idx / ny_nz;
@@ -961,16 +1102,16 @@ __global__ void cpml_h_gpu(
961
1102
  {
962
1103
  long long ny_nz = (long long)NY_FIELDS * NZ_FIELDS;
963
1104
  long long field_stride = (long long)NX_FIELDS * ny_nz;
964
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
965
- if (work >= (long long)step * field_stride) return;
966
-
967
- int s = (int)(work / field_stride);
968
- long long idx = work % field_stride;
969
-
970
- long long i = idx / ny_nz;
971
- long long rem = idx % ny_nz;
972
- long long j = rem / NZ_FIELDS;
973
- long long k = rem % NZ_FIELDS;
1105
+ long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1106
+ int region = (int)blockIdx.y;
1107
+ int s = (int)blockIdx.z;
1108
+ long long i, j, k;
1109
+ if (s >= step ||
1110
+ !map_cpml_region_work<0>(region, region_work,
1111
+ NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1112
+ pml0, pml1, pml2, pml3, pml4, pml5, &i, &j, &k)) return;
1113
+ long long idx = i * ny_nz + j * NZ_FIELDS + k;
1114
+ long long work = (long long)s * field_stride + idx;
974
1115
 
975
1116
  bool in_x0 = (pml0 > 0 && i < pml0 && j < NY_FIELDS && k < NZ_FIELDS);
976
1117
  bool in_xm = (pml1 > 0 && i >= NX_FIELDS - 1 - pml1 && i < NX_FIELDS - 1 && j < NY_FIELDS && k < NZ_FIELDS);
@@ -1115,14 +1256,16 @@ __global__ void adjoint_cpml_e_gpu(
1115
1256
  {
1116
1257
  long long ny_nz = (long long)NY * NZ;
1117
1258
  long long field_stride = (long long)NX * ny_nz;
1118
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1119
- if (work >= (long long)step * field_stride) return;
1120
- int s = (int)(work / field_stride);
1121
- long long idx = work % field_stride;
1122
- long long i = idx / ny_nz;
1123
- long long rem = idx % ny_nz;
1124
- long long j = rem / NZ;
1125
- long long k = rem % NZ;
1259
+ long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1260
+ int region = (int)blockIdx.y;
1261
+ int s = (int)blockIdx.z;
1262
+ long long i, j, k;
1263
+ if (s >= step ||
1264
+ !map_cpml_region_work<1>(region, region_work,
1265
+ NX, NY, NZ, pml0, pml1, pml2, pml3, pml4, pml5,
1266
+ &i, &j, &k)) return;
1267
+ long long idx = i * ny_nz + j * NZ + k;
1268
+ long long work = (long long)s * field_stride + idx;
1126
1269
  float upd = update[idx];
1127
1270
 
1128
1271
  #define APPLY_E_PML_GPU(R, P, p, q, stride, coord, n, spacing, field, source, sign) \
@@ -1215,14 +1358,16 @@ __global__ void adjoint_cpml_h_gpu(
1215
1358
  {
1216
1359
  long long ny_nz = (long long)NY * NZ;
1217
1360
  long long field_stride = (long long)NX * ny_nz;
1218
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1219
- if (work >= (long long)step * field_stride) return;
1220
- int s = (int)(work / field_stride);
1221
- long long idx = work % field_stride;
1222
- long long i = idx / ny_nz;
1223
- long long rem = idx % ny_nz;
1224
- long long j = rem / NZ;
1225
- long long k = rem % NZ;
1361
+ long long region_work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1362
+ int region = (int)blockIdx.y;
1363
+ int s = (int)blockIdx.z;
1364
+ long long i, j, k;
1365
+ if (s >= step ||
1366
+ !map_cpml_region_work<0>(region, region_work,
1367
+ NX, NY, NZ, pml0, pml1, pml2, pml3, pml4, pml5,
1368
+ &i, &j, &k)) return;
1369
+ long long idx = i * ny_nz + j * NZ + k;
1370
+ long long work = (long long)s * field_stride + idx;
1226
1371
  float upd = update[idx];
1227
1372
 
1228
1373
  #define APPLY_H_PML_GPU(R, P, p, q, stride, coord, n, spacing, field, source, sign) \
@@ -1324,7 +1469,7 @@ __global__ void adjoint_receivers_gpu(
1324
1469
  int NX, int NY, int NZ, int nsr, int polarisation, int iterations
1325
1470
  ){
1326
1471
  long long field_stride = (long long)NX * NY * NZ;
1327
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1472
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1328
1473
  long long total = (long long)step * nsr;
1329
1474
  if (work >= total) return;
1330
1475
 
@@ -1339,9 +1484,9 @@ __global__ void adjoint_receivers_gpu(
1339
1484
  float waveform_value = srcwaveforms[index];
1340
1485
  long long id4 = (long long)s * field_stride + i * NY * NZ + j * NZ + k;
1341
1486
 
1342
- if (polarisation == 0) lambda_ex[id4] += waveform_value;
1343
- else if (polarisation == 1) lambda_ey[id4] += waveform_value;
1344
- else if (polarisation == 2) lambda_ez[id4] += waveform_value;
1487
+ if (polarisation == 0) atomicAdd(&lambda_ex[id4], waveform_value);
1488
+ else if (polarisation == 1) atomicAdd(&lambda_ey[id4], waveform_value);
1489
+ else if (polarisation == 2) atomicAdd(&lambda_ez[id4], waveform_value);
1345
1490
  }
1346
1491
 
1347
1492
 
@@ -1355,7 +1500,7 @@ __global__ void adjoint_source_injection_gpu(
1355
1500
  float* __restrict__ grad_source)
1356
1501
  {
1357
1502
  long long field_stride = (long long)NX * NY * NZ;
1358
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1503
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1359
1504
  long long total = (long long)step * nsrc;
1360
1505
  if (work >= total) return;
1361
1506
 
@@ -1392,7 +1537,7 @@ __global__ void save_e_snapshot_gpu(
1392
1537
  {
1393
1538
  long long nx1 = NX - 1, ny1 = NY - 1, nz1 = NZ - 1;
1394
1539
  long long total = nx1 * ny1 * nz1;
1395
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1540
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1396
1541
  if (work >= (long long)step * total) return;
1397
1542
 
1398
1543
  int s = (int)(work / total);
@@ -1425,7 +1570,7 @@ __global__ void save_rhs_snapshot_gpu(
1425
1570
  {
1426
1571
  long long nx1 = NX - 1, ny1 = NY - 1, nz1 = NZ - 1;
1427
1572
  long long total = nx1 * ny1 * nz1;
1428
- long long work = blockIdx.x * blockDim.x + threadIdx.x;
1573
+ long long work = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1429
1574
  if (work >= (long long)step * total) return;
1430
1575
 
1431
1576
  int s = (int)(work / total);
@@ -1491,7 +1636,7 @@ __global__ void accumulate_material_gradients_gpu(
1491
1636
  long long snap_stride = (long long)step * total_cells;
1492
1637
  long long component_stride = (long long)nt_saved * snap_stride;
1493
1638
  int components = (fwi_mode == 3) ? 3 : 1;
1494
- long long idx = blockIdx.x * blockDim.x + threadIdx.x;
1639
+ long long idx = (long long)blockIdx.x * blockDim.x + threadIdx.x;
1495
1640
 
1496
1641
  if (idx >= total_cells) return;
1497
1642
 
@@ -1666,6 +1811,20 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
1666
1811
  long long total_fields = (long long)NX_FIELDS * NY_FIELDS * NZ_FIELDS;
1667
1812
  dim3 grid_material(CEIL_DIV(total_fields, blockSize));
1668
1813
  dim3 grid_fields(CEIL_DIV((long long)step * total_fields, blockSize));
1814
+ long long cpml_e_region_max = cpml_max_region_cells(
1815
+ NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1816
+ pml0, pml1, pml2, pml3, pml4, pml5, 1);
1817
+ long long cpml_h_region_max = cpml_max_region_cells(
1818
+ NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1819
+ pml0, pml1, pml2, pml3, pml4, pml5, 0);
1820
+ int has_cpml_e = has_cpml && cpml_e_region_max > 0;
1821
+ int has_cpml_h = has_cpml && cpml_h_region_max > 0;
1822
+ dim3 grid_cpml_e(
1823
+ has_cpml_e ? CEIL_DIV(cpml_e_region_max, blockSize) : 1,
1824
+ 6, step);
1825
+ dim3 grid_cpml_h(
1826
+ has_cpml_h ? CEIL_DIV(cpml_h_region_max, blockSize) : 1,
1827
+ 6, step);
1669
1828
 
1670
1829
  build_update_coeffs_gpu<<<grid_material, blockSize, 0, stream_comp>>>(eps_r_pad, sigma_pad, mu_r_pad, ce_hist, ce_curl, ce_rhs, ch_hist, ch_curl, ch_rhs, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dt, dx);
1671
1830
  CUDA_CHECK_LAST();
@@ -1717,8 +1876,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
1717
1876
  if (fdtd_order == 8) {
1718
1877
  update_h_gpu<8><<<grid_fields, blockSize, 0, stream_comp>>>(ch_hist, ch_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
1719
1878
  CUDA_CHECK_LAST();
1720
- if (has_cpml) {
1721
- cpml_h_gpu<8><<<grid_fields, blockSize, 0, stream_comp>>>(
1879
+ if (has_cpml_h) {
1880
+ cpml_h_gpu<8><<<grid_cpml_h, blockSize, 0, stream_comp>>>(
1722
1881
  Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1723
1882
  pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
1724
1883
  x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2, z0HPhi1, z0HPhi2, zmHPhi1, zmHPhi2);
@@ -1726,8 +1885,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
1726
1885
  }
1727
1886
  update_e_gpu<8><<<grid_fields, blockSize, 0, stream_comp>>>(ce_hist, ce_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
1728
1887
  CUDA_CHECK_LAST();
1729
- if (has_cpml) {
1730
- cpml_e_gpu<8><<<grid_fields, blockSize, 0, stream_comp>>>(
1888
+ if (has_cpml_e) {
1889
+ cpml_e_gpu<8><<<grid_cpml_e, blockSize, 0, stream_comp>>>(
1731
1890
  Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1732
1891
  pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
1733
1892
  x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2, z0EPhi1, z0EPhi2, zmEPhi1, zmEPhi2);
@@ -1736,8 +1895,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
1736
1895
  } else if (fdtd_order == 4) {
1737
1896
  update_h_gpu<4><<<grid_fields, blockSize, 0, stream_comp>>>(ch_hist, ch_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
1738
1897
  CUDA_CHECK_LAST();
1739
- if (has_cpml) {
1740
- cpml_h_gpu<4><<<grid_fields, blockSize, 0, stream_comp>>>(
1898
+ if (has_cpml_h) {
1899
+ cpml_h_gpu<4><<<grid_cpml_h, blockSize, 0, stream_comp>>>(
1741
1900
  Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1742
1901
  pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
1743
1902
  x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2, z0HPhi1, z0HPhi2, zmHPhi1, zmHPhi2);
@@ -1745,8 +1904,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
1745
1904
  }
1746
1905
  update_e_gpu<4><<<grid_fields, blockSize, 0, stream_comp>>>(ce_hist, ce_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
1747
1906
  CUDA_CHECK_LAST();
1748
- if (has_cpml) {
1749
- cpml_e_gpu<4><<<grid_fields, blockSize, 0, stream_comp>>>(
1907
+ if (has_cpml_e) {
1908
+ cpml_e_gpu<4><<<grid_cpml_e, blockSize, 0, stream_comp>>>(
1750
1909
  Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1751
1910
  pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
1752
1911
  x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2, z0EPhi1, z0EPhi2, zmEPhi1, zmEPhi2);
@@ -1755,8 +1914,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
1755
1914
  } else {
1756
1915
  update_h_gpu<2><<<grid_fields, blockSize, 0, stream_comp>>>(ch_hist, ch_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
1757
1916
  CUDA_CHECK_LAST();
1758
- if (has_cpml) {
1759
- cpml_h_gpu<2><<<grid_fields, blockSize, 0, stream_comp>>>(
1917
+ if (has_cpml_h) {
1918
+ cpml_h_gpu<2><<<grid_cpml_h, blockSize, 0, stream_comp>>>(
1760
1919
  Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1761
1920
  pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
1762
1921
  x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2, z0HPhi1, z0HPhi2, zmHPhi1, zmHPhi2);
@@ -1764,8 +1923,8 @@ DEEPGPR_API void forward(const float* __restrict__ eps_r_pad, const float* __res
1764
1923
  }
1765
1924
  update_e_gpu<2><<<grid_fields, blockSize, 0, stream_comp>>>(ce_hist, ce_curl, Ex, Ey, Ez, Hx, Hy, Hz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
1766
1925
  CUDA_CHECK_LAST();
1767
- if (has_cpml) {
1768
- cpml_e_gpu<2><<<grid_fields, blockSize, 0, stream_comp>>>(
1926
+ if (has_cpml_e) {
1927
+ cpml_e_gpu<2><<<grid_cpml_e, blockSize, 0, stream_comp>>>(
1769
1928
  Ex, Ey, Ez, Hx, Hy, Hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
1770
1929
  pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
1771
1930
  x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2, z0EPhi1, z0EPhi2, zmEPhi1, zmEPhi2);
@@ -1964,6 +2123,20 @@ DEEPGPR_API void backward(const float* __restrict__ eps_r_pad, const float* __re
1964
2123
  long long total_fields = (long long)NX_FIELDS * NY_FIELDS * NZ_FIELDS;
1965
2124
  dim3 grid_material(CEIL_DIV(total_fields, blockSize));
1966
2125
  dim3 grid_fields(CEIL_DIV((long long)step * total_fields, blockSize));
2126
+ long long cpml_e_region_max = cpml_max_region_cells(
2127
+ NX_FIELDS, NY_FIELDS, NZ_FIELDS,
2128
+ pml0, pml1, pml2, pml3, pml4, pml5, 1);
2129
+ long long cpml_h_region_max = cpml_max_region_cells(
2130
+ NX_FIELDS, NY_FIELDS, NZ_FIELDS,
2131
+ pml0, pml1, pml2, pml3, pml4, pml5, 0);
2132
+ int has_cpml_e = has_cpml && cpml_e_region_max > 0;
2133
+ int has_cpml_h = has_cpml && cpml_h_region_max > 0;
2134
+ dim3 grid_cpml_e(
2135
+ has_cpml_e ? CEIL_DIV(cpml_e_region_max, blockSize) : 1,
2136
+ 6, step);
2137
+ dim3 grid_cpml_h(
2138
+ has_cpml_h ? CEIL_DIV(cpml_h_region_max, blockSize) : 1,
2139
+ 6, step);
1967
2140
 
1968
2141
  build_update_coeffs_gpu<<<grid_material, blockSize, 0, stream_comp>>>(eps_r_pad, sigma_pad, mu_r_pad, ce_hist, ce_curl, ce_rhs, ch_hist, ch_curl, ch_rhs, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dt, dx);
1969
2142
  CUDA_CHECK_LAST();
@@ -2032,8 +2205,8 @@ DEEPGPR_API void backward(const float* __restrict__ eps_r_pad, const float* __re
2032
2205
  }
2033
2206
 
2034
2207
  /* E CPML^T -> E update^T -> H CPML^T -> H update^T. */
2035
- if (has_cpml) {
2036
- LAUNCH_ORDER_KERNEL(adjoint_cpml_e_gpu, grid_fields, blockSize, stream_comp, fdtd_order,
2208
+ if (has_cpml_e) {
2209
+ LAUNCH_ORDER_KERNEL(adjoint_cpml_e_gpu, grid_cpml_e, blockSize, stream_comp, fdtd_order,
2037
2210
  lambda_ex, lambda_ey, lambda_ez, lambda_hx, lambda_hy, lambda_hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
2038
2211
  pml0, pml1, pml2, pml3, pml4, pml5, x0ER, xmER, y0ER, ymER, z0ER, zmER, ce_rhs,
2039
2212
  x0EPhi1, x0EPhi2, xmEPhi1, xmEPhi2, y0EPhi1, y0EPhi2, ymEPhi1, ymEPhi2,
@@ -2044,8 +2217,8 @@ DEEPGPR_API void backward(const float* __restrict__ eps_r_pad, const float* __re
2044
2217
  ce_hist, ce_curl, lambda_ex, lambda_ey, lambda_ez, lambda_hx, lambda_hy, lambda_hz,
2045
2218
  step, NX_FIELDS, NY_FIELDS, NZ_FIELDS, dx, dy, dz);
2046
2219
  CUDA_CHECK_LAST();
2047
- if (has_cpml) {
2048
- LAUNCH_ORDER_KERNEL(adjoint_cpml_h_gpu, grid_fields, blockSize, stream_comp, fdtd_order,
2220
+ if (has_cpml_h) {
2221
+ LAUNCH_ORDER_KERNEL(adjoint_cpml_h_gpu, grid_cpml_h, blockSize, stream_comp, fdtd_order,
2049
2222
  lambda_ex, lambda_ey, lambda_ez, lambda_hx, lambda_hy, lambda_hz, dx, dy, dz, step, NX_FIELDS, NY_FIELDS, NZ_FIELDS,
2050
2223
  pml0, pml1, pml2, pml3, pml4, pml5, x0HR, xmHR, y0HR, ymHR, z0HR, zmHR, ch_rhs,
2051
2224
  x0HPhi1, x0HPhi2, xmHPhi1, xmHPhi2, y0HPhi1, y0HPhi2, ymHPhi1, ymHPhi2,
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: DeepGPR
3
- Version: 0.0.16
3
+ Version: 0.0.18
4
4
  Summary: PyTorch and CUDA for GPR FWI
5
5
  Author-email: Lei Liu <liulei990222@gmail.com>
6
6
  License-Expression: MIT
@@ -67,23 +67,24 @@ def _locations(shape):
67
67
  return torch.tensor([[center]], dtype=torch.int32)
68
68
 
69
69
 
70
- def _state_problem(order, pml):
70
+ def _state_problem(order, pml, device="cpu"):
71
71
  torch.manual_seed(1100 + order + pml)
72
+ device = torch.device(device)
72
73
  shape = (9, 10, 8)
73
74
  nx, ny, nz = shape
74
- x = torch.linspace(0.0, 1.0, nx)[:, None, None]
75
- y = torch.linspace(0.0, 1.0, ny)[None, :, None]
76
- z = torch.linspace(0.0, 1.0, nz)[None, None, :]
75
+ x = torch.linspace(0.0, 1.0, nx, device=device)[:, None, None]
76
+ y = torch.linspace(0.0, 1.0, ny, device=device)[None, :, None]
77
+ z = torch.linspace(0.0, 1.0, nz, device=device)[None, None, :]
77
78
  eps_r = 3.5 + 0.7 * x + 0.2 * y + 0.1 * z
78
79
  sigma = 1.0e-4 + 1.0e-4 * (x + y + z) / 3.0
79
80
  mu_r = 1.0 + 0.15 * x + 0.0 * y + 0.05 * z
80
- location = _locations(shape)
81
- source = torch.zeros((1, 1, 1), dtype=torch.float32)
81
+ location = _locations(shape).to(device)
82
+ source = torch.zeros((1, 1, 1), dtype=torch.float32, device=device)
82
83
  spacing = (0.020, 0.017, 0.013)
83
84
  dt = 1.0e-11
84
85
 
85
86
  initial_e, initial_h, initial_cpml = DeepGPR.checkpoint_initial_field(
86
- device="cpu",
87
+ device=device,
87
88
  dx=spacing,
88
89
  dt=dt,
89
90
  source_amplitudes=source,
@@ -102,7 +103,7 @@ def _state_problem(order, pml):
102
103
  ]
103
104
  state = [tensor.clone().requires_grad_(True) for tensor in base_state]
104
105
  result = DeepGPR.compute(
105
- device="cpu",
106
+ device=device,
106
107
  dx=spacing,
107
108
  dt=dt,
108
109
  source_amplitudes=source,
@@ -269,6 +270,15 @@ class DiscreteAdjointTests(unittest.TestCase):
269
270
  def test_native_cpml_single_step_dot_product_all_faces(self):
270
271
  self.assertLess(_state_problem(order=4, pml=2), 3.0e-5)
271
272
 
273
+ @unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
274
+ def test_cuda_cpml_single_step_dot_products_orders_2_4_8(self):
275
+ for order in (2, 4, 8):
276
+ with self.subTest(order=order):
277
+ self.assertLess(
278
+ _state_problem(order=order, pml=2, device="cuda"),
279
+ 2.0e-4,
280
+ )
281
+
272
282
  def test_source_waveform_gradient_matches_finite_difference(self):
273
283
  nx, ny, nt = 12, 14, 60
274
284
  eps_r = torch.full((nx, ny), 4.0)
@@ -378,6 +388,143 @@ class DiscreteAdjointTests(unittest.TestCase):
378
388
  for reference, candidate in (*zip(cpu, cuda), *zip(cuda, cuda_offload)):
379
389
  torch.testing.assert_close(reference, candidate, rtol=5.0e-4, atol=1.0e-6)
380
390
 
391
+ @unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
392
+ def test_cuda_coincident_sources_and_receivers_accumulate(self):
393
+ device = torch.device("cuda")
394
+ nt = 48
395
+ eps_r = torch.full((12, 14), 4.0, device=device)
396
+ sigma = torch.zeros_like(eps_r)
397
+ first = torch.linspace(-0.5, 0.8, nt, device=device)
398
+ second = torch.linspace(0.2, -0.4, nt, device=device)
399
+ source_pair = torch.stack((first, second), dim=0).unsqueeze(-1)
400
+ source_sum = (first + second).reshape(1, nt, 1)
401
+ one_source = torch.tensor([[[5, 5, 0]]], dtype=torch.int32, device=device)
402
+ two_sources = one_source.repeat(1, 2, 1)
403
+ one_receiver = torch.tensor([[[5, 8, 0]]], dtype=torch.int32, device=device)
404
+
405
+ common = dict(
406
+ device=device, dx=0.02, dt=3.0e-11,
407
+ receiver_location=one_receiver,
408
+ er=eps_r, se=sigma, pmlthick=0,
409
+ fdtd_order=2, mode=2,
410
+ )
411
+ pair_data = DeepGPR.compute(
412
+ source_amplitudes=source_pair,
413
+ source_location=two_sources,
414
+ **common,
415
+ )[-1]
416
+ sum_data = DeepGPR.compute(
417
+ source_amplitudes=source_sum,
418
+ source_location=one_source,
419
+ **common,
420
+ )[-1]
421
+ torch.cuda.synchronize()
422
+ torch.testing.assert_close(pair_data, sum_data, rtol=2.0e-5, atol=1.0e-6)
423
+
424
+ def receiver_gradient(receiver_location):
425
+ model = eps_r.detach().clone().requires_grad_(True)
426
+ data = DeepGPR.compute(
427
+ device=device, dx=0.02, dt=3.0e-11,
428
+ source_amplitudes=source_sum,
429
+ source_location=one_source,
430
+ receiver_location=receiver_location,
431
+ er=model, se=sigma, pmlthick=0,
432
+ fdtd_order=2, mode=2,
433
+ model_gradient_sampling_interval=1,
434
+ wavefield_storage_dtype=torch.float32,
435
+ )[-1]
436
+ data.sum().backward()
437
+ return model.grad
438
+
439
+ single_gradient = receiver_gradient(one_receiver)
440
+ duplicate_gradient = receiver_gradient(one_receiver.repeat(1, 2, 1))
441
+ torch.cuda.synchronize()
442
+ torch.testing.assert_close(
443
+ duplicate_gradient, 2.0 * single_gradient, rtol=2.0e-4, atol=1.0e-6
444
+ )
445
+
446
+ @unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
447
+ def test_cuda_large_3d_all_face_cpml_async_bounds(self):
448
+ device = torch.device("cuda")
449
+ nx, ny, nz, nt = 120, 120, 100, 5
450
+ eps_r = torch.full(
451
+ (nx, ny, nz), 4.0, dtype=torch.float32, device=device,
452
+ requires_grad=True,
453
+ )
454
+ sigma = torch.full_like(eps_r, 1.0e-4)
455
+ source = torch.zeros((1, nt, 1), dtype=torch.float32, device=device)
456
+ source[0, 0, 0] = 1.0
457
+ source_location = torch.tensor(
458
+ [[[60, 60, 50]]], dtype=torch.int32
459
+ )
460
+ receiver_location = torch.tensor(
461
+ [[[60, 62, 50]]], dtype=torch.int32
462
+ )
463
+
464
+ result = DeepGPR.compute(
465
+ device=device,
466
+ dx=(0.020, 0.017, 0.014),
467
+ dt=1.0e-11,
468
+ source_amplitudes=source,
469
+ source_location=source_location,
470
+ receiver_location=receiver_location,
471
+ er=eps_r,
472
+ se=sigma,
473
+ pmlthick=10,
474
+ fdtd_order=2,
475
+ mode=3,
476
+ model_gradient_sampling_interval=3,
477
+ wavefield_storage_dtype=torch.float16,
478
+ use_async_offload=True,
479
+ )
480
+ final_fields = (*result[1], *result[2])
481
+ loss = result[-1].square().sum()
482
+ loss = loss + sum(field.square().mean() for field in final_fields)
483
+ loss.backward()
484
+ torch.cuda.synchronize()
485
+
486
+ self.assertTrue(torch.isfinite(result[-1]).all().item())
487
+ self.assertIsNotNone(eps_r.grad)
488
+ self.assertTrue(torch.isfinite(eps_r.grad).all().item())
489
+
490
+ @unittest.skipUnless(torch.cuda.is_available(), "CUDA is not available")
491
+ def test_cuda_asymmetric_cpml_and_incomplete_sampling_block(self):
492
+ device = torch.device("cuda")
493
+ shape = (31, 29, 27)
494
+ nt = 11
495
+ eps_r = torch.full(shape, 4.0, device=device, requires_grad=True)
496
+ sigma = torch.full_like(eps_r, 2.0e-4, requires_grad=True)
497
+ source = torch.linspace(-0.3, 0.7, nt, device=device).reshape(1, nt, 1)
498
+ source_location = torch.tensor(
499
+ [[[15, 14, 13]]], dtype=torch.int32, device=device
500
+ )
501
+ receiver_location = torch.tensor(
502
+ [[[16, 15, 14]]], dtype=torch.int32, device=device
503
+ )
504
+
505
+ receiver = DeepGPR.compute(
506
+ device=device,
507
+ dx=(0.020, 0.016, 0.013),
508
+ dt=1.0e-11,
509
+ source_amplitudes=source,
510
+ source_location=source_location,
511
+ receiver_location=receiver_location,
512
+ er=eps_r,
513
+ se=sigma,
514
+ pmlthick=[3, 5, 4, 2, 6, 3],
515
+ fdtd_order=8,
516
+ mode=3,
517
+ model_gradient_sampling_interval=4,
518
+ wavefield_storage_dtype=torch.bfloat16,
519
+ use_async_offload=True,
520
+ )[-1]
521
+ receiver.square().sum().backward()
522
+ torch.cuda.synchronize()
523
+
524
+ self.assertTrue(torch.isfinite(receiver).all().item())
525
+ self.assertTrue(torch.isfinite(eps_r.grad).all().item())
526
+ self.assertTrue(torch.isfinite(sigma.grad).all().item())
527
+
381
528
 
382
529
  if __name__ == "__main__":
383
530
  unittest.main(verbosity=2)
@@ -113,6 +113,70 @@ class NumericsValidationTests(unittest.TestCase):
113
113
  location, location, 0.02, 3.0e-11, 2, 2,
114
114
  )
115
115
 
116
+ def test_initialization_rejects_empty_native_launch_dimensions(self):
117
+ source = torch.zeros((1, 4, 1))
118
+ location = torch.tensor([[[1, 1, 0]]], dtype=torch.int32)
119
+
120
+ with self.assertRaisesRegex(ValueError, "model dimensions"):
121
+ initialization(
122
+ "cpu", torch.empty((0, 4)), torch.empty((0, 4)), None,
123
+ source, location, location, 0.02, 3.0e-11, 0, 2,
124
+ )
125
+
126
+ empty_cases = (
127
+ (
128
+ torch.empty((0, 1, 3), dtype=torch.int32),
129
+ torch.empty((0, 1, 3), dtype=torch.int32),
130
+ "shot",
131
+ ),
132
+ (
133
+ torch.empty((1, 0, 3), dtype=torch.int32),
134
+ location,
135
+ "source",
136
+ ),
137
+ (
138
+ location,
139
+ torch.empty((1, 0, 3), dtype=torch.int32),
140
+ "receiver",
141
+ ),
142
+ )
143
+ er = torch.full((4, 4), 4.0)
144
+ se = torch.zeros_like(er)
145
+ for source_location, receiver_location, message in empty_cases:
146
+ with self.subTest(message=message), self.assertRaisesRegex(ValueError, message):
147
+ initialization(
148
+ "cpu", er, se, None, source,
149
+ source_location, receiver_location, 0.02, 3.0e-11, 0, 2,
150
+ )
151
+
152
+ def test_initialization_validates_coordinates_before_int32_conversion(self):
153
+ er = torch.full((4, 4), 4.0)
154
+ se = torch.zeros_like(er)
155
+ source = torch.zeros((1, 4, 1))
156
+ location = torch.tensor([[[1, 1, 0]]], dtype=torch.int64)
157
+
158
+ fractional = location.to(torch.float64)
159
+ fractional[0, 0, 1] = 1.5
160
+ with self.assertRaisesRegex(ValueError, "integer-valued"):
161
+ initialization(
162
+ "cpu", er, se, None, source,
163
+ fractional, location, 0.02, 3.0e-11, 0, 2,
164
+ )
165
+
166
+ wrapped = location.clone()
167
+ wrapped[0, 0, 0] = 2**32 + 1
168
+ with self.assertRaisesRegex(ValueError, "out of range"):
169
+ initialization(
170
+ "cpu", er, se, None, source,
171
+ wrapped, location, 0.02, 3.0e-11, 0, 2,
172
+ )
173
+
174
+ with self.assertRaisesRegex(ValueError, "finite integer"):
175
+ initialization(
176
+ "cpu", er, se, None, source,
177
+ location, location, 0.02, 3.0e-11, [0, 0, 0.5, 0], 2,
178
+ )
179
+
116
180
  def test_wavefield_storage_aliases(self):
117
181
  self.assertIs(_normalize_wavefield_storage_dtype("fp32"), torch.float32)
118
182
  self.assertIs(_normalize_wavefield_storage_dtype("fp16"), torch.float16)
@@ -228,6 +292,41 @@ class NumericsValidationTests(unittest.TestCase):
228
292
  _pml_phi_elements(nx, ny, nz, nstep, pml),
229
293
  )
230
294
 
295
+ def test_cpml_checkpoint_shapes_are_validated_before_native_call(self):
296
+ nx, ny, nt = 8, 10, 4
297
+ er = torch.full((nx, ny), 4.0)
298
+ se = torch.zeros_like(er)
299
+ source = torch.zeros((1, nt, 1))
300
+ location = torch.tensor([[[4, 5, 0]]], dtype=torch.int32)
301
+ _, _, pml_state = DeepGPR.checkpoint_initial_field(
302
+ device="cpu",
303
+ dx=0.02,
304
+ dt=3.0e-11,
305
+ source_amplitudes=source,
306
+ source_location=location,
307
+ receiver_location=location,
308
+ er=er,
309
+ se=se,
310
+ pmlthick=2,
311
+ )
312
+ bad_state = list(pml_state)
313
+ bad_state[0] = bad_state[0][..., :-1]
314
+
315
+ with self.assertRaisesRegex(ValueError, "CPML face 0 component 0"):
316
+ DeepGPR.compute(
317
+ device="cpu",
318
+ dx=0.02,
319
+ dt=3.0e-11,
320
+ source_amplitudes=source,
321
+ source_location=location,
322
+ receiver_location=location,
323
+ er=er,
324
+ se=se,
325
+ pmlthick=2,
326
+ PML=tuple(bad_state),
327
+ mode=2,
328
+ )
329
+
231
330
  def test_anisotropic_cpml_coefficients_use_axis_spacing(self):
232
331
  nx, ny, nz = 10, 12, 14
233
332
  er = torch.full((nx, ny, nz), 4.0)
Binary file
Binary file
File without changes
File without changes
File without changes
File without changes