cuwave 0.1.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {cuwave-0.1.0 → cuwave-0.3.0}/PKG-INFO +3 -23
  2. {cuwave-0.1.0 → cuwave-0.3.0}/README.md +2 -22
  3. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/anisotropic.py +4 -11
  4. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/elastic.py +5 -6
  5. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/common.cuh +16 -0
  6. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/scalar.cu +107 -7
  7. cuwave-0.3.0/cuwave/kernels/scalar_sensitivity.cu +302 -0
  8. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/maxwell.py +5 -6
  9. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/optimization.py +9 -7
  10. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/scalar.py +53 -16
  11. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/sensitivity.py +146 -92
  12. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/wave.py +285 -104
  13. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/PKG-INFO +3 -23
  14. {cuwave-0.1.0 → cuwave-0.3.0}/pyproject.toml +1 -1
  15. cuwave-0.1.0/cuwave/kernels/scalar_sensitivity.cu +0 -140
  16. {cuwave-0.1.0 → cuwave-0.3.0}/LICENSE +0 -0
  17. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/__init__.py +0 -0
  18. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/boundary.py +0 -0
  19. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/evals.py +0 -0
  20. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/geometry.py +0 -0
  21. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/__init__.py +0 -0
  22. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/anisotropic.cu +0 -0
  23. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/anisotropic_sensitivity.cu +0 -0
  24. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/elastic.cu +0 -0
  25. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/elastic_sensitivity.cu +0 -0
  26. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/maxwell.cu +0 -0
  27. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/maxwell_sensitivity.cu +0 -0
  28. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/nn.py +0 -0
  29. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/postprocessing.py +0 -0
  30. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/regularization.py +0 -0
  31. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/signals.py +0 -0
  32. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/stencils.py +0 -0
  33. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/utils.py +0 -0
  34. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/SOURCES.txt +0 -0
  35. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/dependency_links.txt +0 -0
  36. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/requires.txt +0 -0
  37. {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/top_level.txt +0 -0
  38. {cuwave-0.1.0 → cuwave-0.3.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cuwave
3
- Version: 0.1.0
3
+ Version: 0.3.0
4
4
  Summary: GPU finite-difference wave solver with differentiable adjoints
5
5
  Author-email: Leon Herrmann <herrmann.leon@pm.me>
6
6
  License-Expression: MIT
@@ -93,31 +93,11 @@ Additional benefits of **CuWave** are
93
93
 
94
94
  ## Install
95
95
 
96
- Dependencies are kept **lightweight**. Only CuPy is required beyond standard Python library.
97
-
98
96
  ```bash
99
- pip install cupy-cuda12x # or cupy-cuda11x, to match your CUDA
100
- pip install cuwave # or `pip install -e .` from a checkout
97
+ pip install cuwave
101
98
  ```
102
99
 
103
- CuPy must be installed separately because the wheel depends on your CUDA toolkit;
104
- all remaining dependencies are declared in `pyproject.toml`.
105
-
106
- PyTorch is optional for the regularization via neural optimization; see [pytorch](https://pytorch.org/get-started/locally/) for the installation. Otherwise it is not needed.
107
-
108
- > [!NOTE]
109
- > Match PyTorch's CUDA version to CuPy's, or the two runtimes clash at the first kernel launch.
110
- > With `cupy-cuda12x`:
111
- > ```bash
112
- > pip install torch --index-url https://download.pytorch.org/whl/cu128
113
- > ```
114
-
115
- The tests under `tests/` are `unittest` classes, but `pytest` is the recommended runner:
116
-
117
- ```bash
118
- pip install pytest
119
- python -m pytest tests/ -q # ~20 s on a GPU, ~3 s without: CUDA and PyTorch tests skip when unavailable
120
- ```
100
+ Requires an NVIDIA GPU and [CuPy](https://docs.cupy.dev/en/stable/install.html) matching your CUDA toolkit (e.g. `pip install cupy-cuda12x`), which is not pulled in automatically. For the full installation, including the optional PyTorch and running the tests, see [docs/install.md](https://github.com/cmpmech/cuwave/blob/main/docs/install.md).
121
101
 
122
102
  ## References
123
103
 
@@ -67,31 +67,11 @@ Additional benefits of **CuWave** are
67
67
 
68
68
  ## Install
69
69
 
70
- Dependencies are kept **lightweight**. Only CuPy is required beyond standard Python library.
71
-
72
70
  ```bash
73
- pip install cupy-cuda12x # or cupy-cuda11x, to match your CUDA
74
- pip install cuwave # or `pip install -e .` from a checkout
71
+ pip install cuwave
75
72
  ```
76
73
 
77
- CuPy must be installed separately because the wheel depends on your CUDA toolkit;
78
- all remaining dependencies are declared in `pyproject.toml`.
79
-
80
- PyTorch is optional for the regularization via neural optimization; see [pytorch](https://pytorch.org/get-started/locally/) for the installation. Otherwise it is not needed.
81
-
82
- > [!NOTE]
83
- > Match PyTorch's CUDA version to CuPy's, or the two runtimes clash at the first kernel launch.
84
- > With `cupy-cuda12x`:
85
- > ```bash
86
- > pip install torch --index-url https://download.pytorch.org/whl/cu128
87
- > ```
88
-
89
- The tests under `tests/` are `unittest` classes, but `pytest` is the recommended runner:
90
-
91
- ```bash
92
- pip install pytest
93
- python -m pytest tests/ -q # ~20 s on a GPU, ~3 s without: CUDA and PyTorch tests skip when unavailable
94
- ```
74
+ Requires an NVIDIA GPU and [CuPy](https://docs.cupy.dev/en/stable/install.html) matching your CUDA toolkit (e.g. `pip install cupy-cuda12x`), which is not pulled in automatically. For the full installation, including the optional PyTorch and running the tests, see [docs/install.md](https://github.com/cmpmech/cuwave/blob/main/docs/install.md).
95
75
 
96
76
  ## References
97
77
 
@@ -25,7 +25,7 @@ import numpy.typing as npt
25
25
 
26
26
  from .boundary import Clamped, Traction, faces_with
27
27
  from .elastic import voigt
28
- from .wave import PAIRS, Simulation, apply_cell_weights, grid_block
28
+ from .wave import PAIRS, Simulation, apply_cell_weights, axis_geometry, grid_block
29
29
 
30
30
  KERNEL_PATH = Path(__file__).parent / "kernels" / "anisotropic.cu"
31
31
  SENSITIVITY_PATH = Path(__file__).parent / "kernels" / "anisotropic_sensitivity.cu"
@@ -253,13 +253,6 @@ class AnisotropicElasticWave(Simulation):
253
253
  """Source scaling, unscaled since rho0 is already folded into the lumped inertia."""
254
254
  return 1.0
255
255
 
256
- def axis_geometry(self) -> list:
257
- """Extents and previous-axis strides, the step kernel's tail without the factors."""
258
- geom = [self.Nx[0]]
259
- for d in range(1, self.ndim):
260
- geom += [self.Nx[d], self.strides[d - 1]]
261
- return geom
262
-
263
256
  def gradient_fields(self, mat: dict) -> dict[str, cpt.NDArray]:
264
257
  """Nodal accumulators plus the cell one the stiffness density lands in first."""
265
258
  grads = {
@@ -285,7 +278,7 @@ class AnisotropicElasticWave(Simulation):
285
278
  grads["material"],
286
279
  grads["design"],
287
280
  self.dtype(1.0 / 2.0**self.ndim),
288
- *self.axis_geometry(),
281
+ *axis_geometry(self),
289
282
  ],
290
283
  )
291
284
  return {"mass": grads["mass"], "stiff": grads["stiff"]}
@@ -302,7 +295,7 @@ class AnisotropicElasticWave(Simulation):
302
295
  mat["stencil"],
303
296
  mass_factor,
304
297
  np.int32(self.comp_stride),
305
- *self.axis_geometry(),
298
+ *axis_geometry(self),
306
299
  ]
307
300
 
308
301
  def gradient_step(u0, u1, u2, l1):
@@ -323,7 +316,7 @@ class AnisotropicElasticWave(Simulation):
323
316
  self.dtype(sign * self.density * volume / (2.0 * self.dt) ** 2),
324
317
  self.dtype(-sign),
325
318
  np.int32(self.comp_stride),
326
- *self.axis_geometry(),
319
+ *axis_geometry(self),
327
320
  ]
328
321
 
329
322
  def frechet_step(u0, u1, u2):
@@ -26,13 +26,12 @@ from .wave import (
26
26
  Simulation,
27
27
  apply_cell_weights,
28
28
  axis_geometry,
29
- component_weights,
30
29
  grid_block,
31
30
  pair_average,
32
31
  pair_average_adjoint,
33
- pair_weights,
34
32
  point_average,
35
33
  point_average_adjoint,
34
+ wall_weights,
36
35
  )
37
36
 
38
37
  KERNEL_PATH = Path(__file__).parent / "kernels" / "elastic.cu"
@@ -160,7 +159,7 @@ class ElasticWave(Simulation):
160
159
  for c in range(self.ncomp):
161
160
  mass = (
162
161
  self.density
163
- * component_weights(self, c)
162
+ * wall_weights(self, (c,))
164
163
  * point_average(self, gamma, c)
165
164
  )
166
165
  minv[c] = 1.0 / cp.maximum(mass, cp.finfo(self.dtype).tiny)
@@ -178,7 +177,7 @@ class ElasticWave(Simulation):
178
177
  if self.ndim > 1:
179
178
  gshear = cp.zeros((self.npairs, *self.Nx_padded), dtype=self.dtype)
180
179
  for p, axes in enumerate(PAIRS[self.ndim][self.ndim :]):
181
- gshear[p] = pair_weights(self, axes) * pair_average(self, gamma, axes)
180
+ gshear[p] = wall_weights(self, axes) * pair_average(self, gamma, axes)
182
181
  mat["gshear"] = cp.ascontiguousarray(gshear)
183
182
  if self.damping is not None:
184
183
  mat["damping"] = self.damping
@@ -269,12 +268,12 @@ class ElasticWave(Simulation):
269
268
  gamma = grads["design"]
270
269
  g_mass = cp.zeros(self.Nx_padded, dtype=self.dtype)
271
270
  for c in range(self.ncomp):
272
- density = grads["mass"][c] * component_weights(self, c) * self.density
271
+ density = grads["mass"][c] * wall_weights(self, (c,)) * self.density
273
272
  g_mass += point_average_adjoint(self, density, c)
274
273
  # the normal density sits on the nodes, so its chain rule is the weight alone
275
274
  g_stiff = apply_cell_weights(self, grads["normal"].copy())
276
275
  for p, axes in enumerate(PAIRS[self.ndim][self.ndim :]):
277
- density = grads["shear"][p] * pair_weights(self, axes)
276
+ density = grads["shear"][p] * wall_weights(self, axes)
278
277
  g_stiff += pair_average_adjoint(self, density, gamma, axes)
279
278
  return {"mass": g_mass, "stiff": g_stiff}
280
279
 
@@ -69,6 +69,22 @@ excitation_kernel(real_t *__restrict__ u, const real_t *__restrict__ source,
69
69
  }
70
70
  }
71
71
 
72
+ // ------------------------------------------------------------------------------------
73
+ __global__ void adjoint_excitation_kernel(
74
+ real_t *__restrict__ l2, const real_t *__restrict__ signal,
75
+ const int offset, const int *__restrict__ lin_index, const int num_sensors,
76
+ const real_t *__restrict__ weight, real_t *__restrict__ g_mass,
77
+ const real_t *__restrict__ u1, const real_t mf) {
78
+ const int idx = blockIdx.x * blockDim.x + threadIdx.x;
79
+ if (idx < num_sensors) {
80
+ const int n = lin_index[idx];
81
+ const real_t load = weight[idx] * signal[offset + idx];
82
+ atomicAdd(&l2[n], load);
83
+ // the injected load is part of the second time difference of lambda
84
+ atomicAdd(&g_mass[n], -mf * u1[n] * load);
85
+ }
86
+ }
87
+
72
88
  // ------------------------------------------------------------------------------------
73
89
  __global__ void get_signal_kernel(const real_t *__restrict__ u,
74
90
  real_t *__restrict__ um, const int offset,
@@ -2,6 +2,7 @@
2
2
  // Compile-time configuration this file responds to:
3
3
  // NDIM = 1 | 2 | 3
4
4
  // USE_DAMPING
5
+ // USE_DOMAIN
5
6
 
6
7
  // spatial finite difference stencil for Laplacian
7
8
  __device__ __forceinline__ real_t flux_divergence_axis(
@@ -25,6 +26,81 @@ __device__ __forceinline__ real_t flux_divergence_axis(
25
26
  return factor * (Dp * gp - Dm * gm); // outer grad (incl. inner grad)
26
27
  }
27
28
 
29
+ // ----------------------------------- domain helpers
30
+ #ifdef USE_DOMAIN
31
+ #define DIRICHLET 7 // the radius code of a cell open onto a node held at zero
32
+
33
+ // the same with each cell at its own radius, zero for a cell leaving the domain
34
+ __device__ __forceinline__ real_t flux_divergence_cells(
35
+ const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
36
+ const int idx, const int s, const real_t uc, const real_t sc,
37
+ const real_t factor, const int rp, const int rm) {
38
+ real_t Dp = 0, Dm = 0; // a closed cell carries no flux
39
+ if (rp == DIRICHLET)
40
+ Dp = -uc * sc; // onto a zero halfway across, at the node's own stiffness
41
+ else if (rp) {
42
+ const real_t sp = stiff[idx + s];
43
+ Dp = OP_W(rp, 1) * (u1[idx + s] - uc);
44
+ #pragma unroll
45
+ for (int k = 2; k <= STENCIL_RADIUS; ++k)
46
+ if (k <= rp)
47
+ Dp += OP_W(rp, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]);
48
+ Dp *= sc * sp / (sc + sp); // harmonic mean
49
+ }
50
+ if (rm == DIRICHLET)
51
+ Dm = uc * sc;
52
+ else if (rm) {
53
+ const real_t sm = stiff[idx - s];
54
+ Dm = OP_W(rm, 1) * (uc - u1[idx - s]);
55
+ #pragma unroll
56
+ for (int k = 2; k <= STENCIL_RADIUS; ++k)
57
+ if (k <= rm)
58
+ Dm += OP_W(rm, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]);
59
+ Dm *= sc * sm / (sc + sm); // harmonic mean
60
+ }
61
+ return factor * (Dp - Dm);
62
+ }
63
+ #endif
64
+
65
+ #ifdef USE_DOMAIN
66
+ #define BLOCK_X (tile & 1023) // one block per domain tile, 10 bits an axis
67
+ #define BLOCK_Y ((tile >> 10) & 1023)
68
+ #define BLOCK_Z ((tile >> 20) & 1023)
69
+ #else
70
+ #define BLOCK_X blockIdx.x
71
+ #define BLOCK_Y blockIdx.y
72
+ #define BLOCK_Z blockIdx.z
73
+ #endif
74
+
75
+ #ifdef USE_DOMAIN
76
+ #define CELLS(s, f, d) \
77
+ flux_divergence_cells(u1, stiff, idx, s, uc, sc, f, (code >> 6 * (d)) & 7, \
78
+ (code >> (6 * (d) + 3)) & 7)
79
+
80
+ // out of line, so the rare wall node costs the deep ones no registers
81
+ __device__ __noinline__ real_t
82
+ wall_laplacian(const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
83
+ const int idx, const real_t uc, const real_t sc, const int code,
84
+ const real_t f0
85
+ #if NDIM >= 2
86
+ ,
87
+ const real_t f1, const int s0
88
+ #endif
89
+ #if NDIM >= 3
90
+ ,
91
+ const real_t f2, const int s1
92
+ #endif
93
+ ) {
94
+ #if NDIM == 1
95
+ return CELLS(1, f0, 0);
96
+ #elif NDIM == 2
97
+ return CELLS(s0, f0, 0) + CELLS(1, f1, 1);
98
+ #elif NDIM == 3
99
+ return CELLS(s0, f0, 0) + CELLS(s1, f1, 1) + CELLS(1, f2, 2);
100
+ #endif
101
+ }
102
+ #endif
103
+
28
104
  // ----------------------------- boundary condition helper
29
105
  #if NDIM == 1
30
106
  #define BC_PARAMS const int N0
@@ -80,6 +156,9 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
80
156
  const real_t *__restrict__ minv, const int derive_inertia,
81
157
  #ifdef USE_DAMPING
82
158
  const real_t *__restrict__ damping, const real_t dt,
159
+ #endif
160
+ #ifdef USE_DOMAIN
161
+ const int *__restrict__ cells, const int *__restrict__ tiles,
83
162
  #endif
84
163
  const real_t f0, const int N0
85
164
  #if NDIM >= 2
@@ -91,23 +170,26 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
91
170
  const real_t f2, const int N2, const int s1
92
171
  #endif
93
172
  ) {
173
+ #ifdef USE_DOMAIN
174
+ const int tile = tiles[blockIdx.x];
175
+ #endif
94
176
  #if NDIM == 1
95
- const int a0 = blockIdx.x * blockDim.x + threadIdx.x;
177
+ const int a0 = BLOCK_X * blockDim.x + threadIdx.x;
96
178
  if (!(a0 > 0 && a0 < N0 - 1))
97
179
  return;
98
180
  const int idx = a0;
99
181
  const int r0 = CLOSURE(a0, N0);
100
182
  #elif NDIM == 2
101
- const int a1 = blockIdx.x * blockDim.x + threadIdx.x;
102
- const int a0 = blockIdx.y * blockDim.y + threadIdx.y;
183
+ const int a1 = BLOCK_X * blockDim.x + threadIdx.x;
184
+ const int a0 = BLOCK_Y * blockDim.y + threadIdx.y;
103
185
  if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1))
104
186
  return;
105
187
  const int idx = a0 * s0 + a1;
106
188
  const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1);
107
189
  #elif NDIM == 3
108
- const int a2 = blockIdx.x * blockDim.x + threadIdx.x;
109
- const int a1 = blockIdx.y * blockDim.y + threadIdx.y;
110
- const int a0 = blockIdx.z * blockDim.z + threadIdx.z;
190
+ const int a2 = BLOCK_X * blockDim.x + threadIdx.x;
191
+ const int a1 = BLOCK_Y * blockDim.y + threadIdx.y;
192
+ const int a0 = BLOCK_Z * blockDim.z + threadIdx.z;
111
193
  if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1 && a2 > 0 &&
112
194
  a2 < N2 - 1))
113
195
  return;
@@ -117,7 +199,6 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
117
199
 
118
200
  const real_t uc = u1[idx]; // load once
119
201
  const real_t sc = stiff[idx]; // load once
120
-
121
202
  #if NDIM == 1
122
203
  real_t laplacian = flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f0, r0);
123
204
  #elif NDIM == 2
@@ -128,6 +209,25 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
128
209
  flux_divergence_axis(u1, stiff, idx, s1, uc, sc, f1, r1) +
129
210
  flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f2, r2);
130
211
  #endif
212
+ #ifdef USE_DOMAIN
213
+ // bit 30 marks a tile touching the wall, so the branch is uniform per block
214
+ if (tile >> 30) {
215
+ const int code = cells[idx]; // 0 outside, bit 30 deep, else the cell radii
216
+ if (!code)
217
+ return; // outside the domain nothing is stepped
218
+ #if NDIM == 1
219
+ if (!(code >> 30))
220
+ laplacian = wall_laplacian(u1, stiff, idx, uc, sc, code, f0);
221
+ #elif NDIM == 2
222
+ if (!(code >> 30))
223
+ laplacian = wall_laplacian(u1, stiff, idx, uc, sc, code, f0, f1, s0);
224
+ #elif NDIM == 3
225
+ if (!(code >> 30))
226
+ laplacian =
227
+ wall_laplacian(u1, stiff, idx, uc, sc, code, f0, f1, s0, f2, s1);
228
+ #endif
229
+ }
230
+ #endif
131
231
 
132
232
  const real_t mi = derive_inertia ? 1.f / sc : minv[idx];
133
233
  #ifdef USE_DAMPING
@@ -0,0 +1,302 @@
1
+ // Prepended by wave.compile_kernels: stencils.preamble, then common.cuh.
2
+ // Compile-time configuration this file responds to:
3
+ // NDIM = 1 | 2 | 3
4
+ // USE_DAMPING
5
+
6
+ // ------------------------------- interior guard macros
7
+ #if NDIM == 1
8
+ #define INTERIOR_OR_RETURN \
9
+ const int a0 = blockIdx.x * blockDim.x + threadIdx.x; \
10
+ if (!(a0 > 0 && a0 < N0 - 1)) \
11
+ return; \
12
+ const int idx = a0
13
+ #define AXIS_RADII const int r0 = CLOSURE(a0, N0)
14
+ #define AXIS_OFFSETS const int o0 = 1
15
+ #elif NDIM == 2
16
+ #define INTERIOR_OR_RETURN \
17
+ const int a1 = blockIdx.x * blockDim.x + threadIdx.x; \
18
+ const int a0 = blockIdx.y * blockDim.y + threadIdx.y; \
19
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1)) \
20
+ return; \
21
+ const int idx = a0 * s0 + a1
22
+ #define AXIS_RADII const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1)
23
+ #define AXIS_OFFSETS const int o0 = s0, o1 = 1
24
+ #elif NDIM == 3
25
+ #define INTERIOR_OR_RETURN \
26
+ const int a2 = blockIdx.x * blockDim.x + threadIdx.x; \
27
+ const int a1 = blockIdx.y * blockDim.y + threadIdx.y; \
28
+ const int a0 = blockIdx.z * blockDim.z + threadIdx.z; \
29
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1 && a2 > 0 && \
30
+ a2 < N2 - 1)) \
31
+ return; \
32
+ const int idx = a0 * s0 + a1 * s1 + a2
33
+ #define AXIS_RADII \
34
+ const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1), r2 = CLOSURE(a2, N2)
35
+ #define AXIS_OFFSETS const int o0 = s0, o1 = s1, o2 = 1
36
+ #endif
37
+
38
+ // ---------------------------------- gradient helpers
39
+ __device__ __forceinline__ real_t stiffness_gradient_axis(
40
+ const real_t *__restrict__ u1, const real_t *__restrict__ l1,
41
+ const real_t *__restrict__ stiff, const int idx, const int s,
42
+ const real_t uc, const real_t lc, const real_t sc, const real_t factor,
43
+ const int r) {
44
+ const real_t sp = stiff[idx + s]; // plus of sc
45
+ const real_t sm = stiff[idx - s]; // minus of sc
46
+ const real_t dgp = sp * sp / ((sc + sp) * (sc + sp)); // d(harmonic mean)/dsc
47
+ const real_t dgm = sm * sm / ((sc + sm) * (sc + sm)); // d(harmonic mean)/dsc
48
+ real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // inner grad
49
+ real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // inner grad
50
+ #pragma unroll
51
+ for (int k = 2; k <= STENCIL_RADIUS; ++k)
52
+ if (k <= r) {
53
+ Dp += OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]);
54
+ Dm += OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]);
55
+ }
56
+ return factor * (dgp * Dp * (l1[idx + s] - lc) +
57
+ dgm * Dm * (lc - l1[idx - s])); // both cells of the node
58
+ }
59
+
60
+ // the adjoint step's flux divergence and the stiffness gradient of one axis,
61
+ // sharing the cell stiffnesses
62
+ __device__ __forceinline__ void adjoint_gradient_axis(
63
+ const real_t *__restrict__ u1, const real_t *__restrict__ l1,
64
+ const real_t *__restrict__ stiff, const int idx, const int s,
65
+ const real_t uc, const real_t lc, const real_t sc, const real_t factor,
66
+ const int r, real_t &div_l, real_t &g) {
67
+ const real_t sp = stiff[idx + s]; // plus of sc
68
+ const real_t sm = stiff[idx - s]; // minus of sc
69
+ const real_t gp = sc * sp / (sc + sp); // harmonic mean
70
+ const real_t gm = sc * sm / (sc + sm); // harmonic mean
71
+ const real_t dgp = sp * sp / ((sc + sp) * (sc + sp)); // d(harmonic mean)/dsc
72
+ const real_t dgm = sm * sm / ((sc + sm) * (sc + sm)); // d(harmonic mean)/dsc
73
+ const real_t lp = l1[idx + s];
74
+ const real_t lm = l1[idx - s];
75
+ real_t Lp = OP_W(r, 1) * (lp - lc); // inner grad of lambda
76
+ real_t Lm = OP_W(r, 1) * (lc - lm); // inner grad of lambda
77
+ real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // inner grad of u
78
+ real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // inner grad of u
79
+ #pragma unroll
80
+ for (int k = 2; k <= STENCIL_RADIUS; ++k)
81
+ if (k <= r) {
82
+ Lp += OP_W(r, k) * (l1[idx + k * s] - l1[idx - (k - 1) * s]);
83
+ Lm += OP_W(r, k) * (l1[idx + (k - 1) * s] - l1[idx - k * s]);
84
+ Dp += OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]);
85
+ Dm += OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]);
86
+ }
87
+ div_l += factor * (Lp * gp - Lm * gm); // the step's flux divergence
88
+ g += factor * (dgp * Dp * (lp - lc) + dgm * Dm * (lc - lm));
89
+ }
90
+
91
+ // byte-identical to scalar.cu: each file is its own compilation unit
92
+ __device__ __forceinline__ real_t flux_divergence_axis(
93
+ const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
94
+ const int idx, const int s, const real_t uc, const real_t sc,
95
+ const real_t factor, const int r) {
96
+ const real_t sp = stiff[idx + s]; // plus of sc
97
+ const real_t sm = stiff[idx - s]; // minus of sc
98
+ const real_t gp = sc * sp / (sc + sp); // harmonic mean
99
+ const real_t gm = sc * sm / (sc + sm); // harmonic mean
100
+ real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // initialization: inner grad
101
+ real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // initialization: inner grad
102
+ #pragma unroll
103
+ for (int k = 2; k <= STENCIL_RADIUS; ++k)
104
+ if (k <= r) {
105
+ Dp +=
106
+ OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]); // inner grad
107
+ Dm +=
108
+ OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]); // inner grad
109
+ }
110
+ return factor * (Dp * gp - Dm * gm); // outer grad (incl. inner grad)
111
+ }
112
+
113
+ // -------------------------------------- kernels
114
+ extern "C" {
115
+
116
+ // ------------------------------------------------------------------------------------
117
+ __global__ void
118
+ gradient_kernel(real_t *__restrict__ g_mass, real_t *__restrict__ g_stiff,
119
+ const real_t *__restrict__ u0, const real_t *__restrict__ u1,
120
+ const real_t *__restrict__ u2, const real_t *__restrict__ l1,
121
+ const real_t *__restrict__ stiff, const real_t inv_dt2,
122
+ const real_t F0, const int N0
123
+ #if NDIM >= 2
124
+ ,
125
+ const real_t F1, const int N1, const int s0
126
+ #endif
127
+ #if NDIM >= 3
128
+ ,
129
+ const real_t F2, const int N2, const int s1
130
+ #endif
131
+ ) {
132
+ INTERIOR_OR_RETURN;
133
+ AXIS_RADII;
134
+
135
+ const real_t uc = u1[idx];
136
+ const real_t lc = l1[idx];
137
+
138
+ // dJ/dmass: no neighbour and no material load
139
+ g_mass[idx] -= inv_dt2 * lc * (u2[idx] - 2.f * uc + u0[idx]);
140
+
141
+ // dJ/dstiff: scalar.cu's harmonic cell mean differentiated in place
142
+ const real_t sc = stiff[idx];
143
+ #if NDIM == 1
144
+ const real_t g =
145
+ stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F0, r0);
146
+ #elif NDIM == 2
147
+ const real_t g =
148
+ stiffness_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, F0, r0) +
149
+ stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F1, r1);
150
+ #elif NDIM == 3
151
+ const real_t g =
152
+ stiffness_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, F0, r0) +
153
+ stiffness_gradient_axis(u1, l1, stiff, idx, s1, uc, lc, sc, F1, r1) +
154
+ stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F2, r2);
155
+ #endif
156
+ g_stiff[idx] -= g;
157
+ }
158
+
159
+ // ------------------------------------------------------------------------------------
160
+ __global__ void frechet_kernel(real_t *__restrict__ acc_mass,
161
+ real_t *__restrict__ acc_stiff,
162
+ const real_t *__restrict__ u0,
163
+ const real_t *__restrict__ u1,
164
+ const real_t *__restrict__ u2, const real_t ft,
165
+ const real_t f0, const int N0
166
+ #if NDIM >= 2
167
+ ,
168
+ const real_t f1, const int N1, const int s0
169
+ #endif
170
+ #if NDIM >= 3
171
+ ,
172
+ const real_t f2, const int N2, const int s1
173
+ #endif
174
+ ) {
175
+ INTERIOR_OR_RETURN;
176
+ AXIS_OFFSETS;
177
+
178
+ const real_t dudt = u2[idx] - u0[idx];
179
+ acc_mass[idx] += ft * dudt * dudt;
180
+
181
+ const real_t g0 = u1[idx + o0] - u1[idx - o0];
182
+ real_t sum = f0 * g0 * g0;
183
+ #if NDIM >= 2
184
+ const real_t g1 = u1[idx + o1] - u1[idx - o1];
185
+ sum += f1 * g1 * g1;
186
+ #endif
187
+ #if NDIM >= 3
188
+ const real_t g2 = u1[idx + o2] - u1[idx - o2];
189
+ sum += f2 * g2 * g2;
190
+ #endif
191
+ acc_stiff[idx] += sum;
192
+ }
193
+
194
+ // ------------------------------------------------------------------------------------
195
+ __global__ void adjoint_gradient_kernel(
196
+ real_t *__restrict__ l0, const real_t *__restrict__ l1,
197
+ real_t *__restrict__ g_mass, real_t *__restrict__ g_stiff,
198
+ const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
199
+ const real_t *__restrict__ minv, const int derive_inertia,
200
+ #ifdef USE_DAMPING
201
+ const real_t *__restrict__ damping, const real_t dt,
202
+ #endif
203
+ const real_t mf, const real_t f0, const int N0
204
+ #if NDIM >= 2
205
+ ,
206
+ const real_t f1, const int N1, const int s0
207
+ #endif
208
+ #if NDIM >= 3
209
+ ,
210
+ const real_t f2, const int N2, const int s1
211
+ #endif
212
+ ) {
213
+ INTERIOR_OR_RETURN;
214
+ AXIS_RADII;
215
+
216
+ const real_t uc = u1[idx];
217
+ const real_t lc = l1[idx];
218
+ const real_t sc = stiff[idx];
219
+ real_t div_l = 0, g = 0;
220
+ #if NDIM == 1
221
+ adjoint_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, f0, r0, div_l, g);
222
+ #elif NDIM == 2
223
+ adjoint_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, f0, r0, div_l, g);
224
+ adjoint_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, f1, r1, div_l, g);
225
+ #elif NDIM == 3
226
+ adjoint_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, f0, r0, div_l, g);
227
+ adjoint_gradient_axis(u1, l1, stiff, idx, s1, uc, lc, sc, f1, r1, div_l, g);
228
+ adjoint_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, f2, r2, div_l, g);
229
+ #endif
230
+
231
+ const real_t mi = derive_inertia ? 1.f / sc : minv[idx];
232
+ const real_t lo = l0[idx];
233
+ #ifdef USE_DAMPING
234
+ const real_t beta = 0.5f * mi * damping[idx] * dt;
235
+ const real_t ln = (2.f * lc - lo * (1.f - beta) + mi * div_l) / (1.f + beta);
236
+ #else
237
+ const real_t ln = -lo + 2.f * lc + mi * div_l;
238
+ #endif
239
+ l0[idx] = ln;
240
+ // dJ/dmass by parts in time: u against the second difference of lambda
241
+ g_mass[idx] -= mf * uc * (ln - 2.f * lc + lo);
242
+ g_stiff[idx] -= mf * g;
243
+ }
244
+
245
+ // ------------------------------------------------------------------------------------
246
+ __global__ void superposed_kernel(
247
+ const real_t *__restrict__ u0, const real_t *__restrict__ u1,
248
+ real_t *__restrict__ u2, real_t *__restrict__ acc_mass,
249
+ real_t *__restrict__ acc_stiff, const real_t *__restrict__ stiff,
250
+ const real_t *__restrict__ minv, const int derive_inertia, const real_t ft,
251
+ const real_t fs, const real_t f0, const int N0
252
+ #if NDIM >= 2
253
+ ,
254
+ const real_t f1, const int N1, const int s0
255
+ #endif
256
+ #if NDIM >= 3
257
+ ,
258
+ const real_t f2, const int N2, const int s1
259
+ #endif
260
+ ) {
261
+ INTERIOR_OR_RETURN;
262
+ AXIS_RADII;
263
+ AXIS_OFFSETS;
264
+
265
+ const real_t uc = u1[idx];
266
+ const real_t sc = stiff[idx];
267
+ #if NDIM == 1
268
+ const real_t laplacian =
269
+ flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f0, r0);
270
+ #elif NDIM == 2
271
+ const real_t laplacian =
272
+ flux_divergence_axis(u1, stiff, idx, s0, uc, sc, f0, r0) +
273
+ flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f1, r1);
274
+ #elif NDIM == 3
275
+ const real_t laplacian =
276
+ flux_divergence_axis(u1, stiff, idx, s0, uc, sc, f0, r0) +
277
+ flux_divergence_axis(u1, stiff, idx, s1, uc, sc, f1, r1) +
278
+ flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f2, r2);
279
+ #endif
280
+
281
+ // the mass density of the triplet before, whose oldest slot u2 still holds
282
+ const real_t dudt = uc - u2[idx];
283
+ acc_mass[idx] += ft * dudt * dudt;
284
+
285
+ // the stiffness density of this triplet, its middle slot being u1
286
+ const real_t g0 = u1[idx + o0] - u1[idx - o0];
287
+ real_t sum = f0 * g0 * g0;
288
+ #if NDIM >= 2
289
+ const real_t g1 = u1[idx + o1] - u1[idx - o1];
290
+ sum += f1 * g1 * g1;
291
+ #endif
292
+ #if NDIM >= 3
293
+ const real_t g2 = u1[idx + o2] - u1[idx - o2];
294
+ sum += f2 * g2 * g2;
295
+ #endif
296
+ acc_stiff[idx] += fs * sum;
297
+
298
+ const real_t mi = derive_inertia ? 1.f / sc : minv[idx];
299
+ u2[idx] = -u0[idx] + 2.f * uc + mi * laplacian;
300
+ }
301
+
302
+ } // extern "C"