cuwave 0.1.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cuwave-0.1.0 → cuwave-0.3.0}/PKG-INFO +3 -23
- {cuwave-0.1.0 → cuwave-0.3.0}/README.md +2 -22
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/anisotropic.py +4 -11
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/elastic.py +5 -6
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/common.cuh +16 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/scalar.cu +107 -7
- cuwave-0.3.0/cuwave/kernels/scalar_sensitivity.cu +302 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/maxwell.py +5 -6
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/optimization.py +9 -7
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/scalar.py +53 -16
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/sensitivity.py +146 -92
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/wave.py +285 -104
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/PKG-INFO +3 -23
- {cuwave-0.1.0 → cuwave-0.3.0}/pyproject.toml +1 -1
- cuwave-0.1.0/cuwave/kernels/scalar_sensitivity.cu +0 -140
- {cuwave-0.1.0 → cuwave-0.3.0}/LICENSE +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/__init__.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/boundary.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/evals.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/geometry.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/__init__.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/anisotropic.cu +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/anisotropic_sensitivity.cu +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/elastic.cu +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/elastic_sensitivity.cu +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/maxwell.cu +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/kernels/maxwell_sensitivity.cu +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/nn.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/postprocessing.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/regularization.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/signals.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/stencils.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave/utils.py +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/SOURCES.txt +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/dependency_links.txt +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/requires.txt +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/cuwave.egg-info/top_level.txt +0 -0
- {cuwave-0.1.0 → cuwave-0.3.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cuwave
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: GPU finite-difference wave solver with differentiable adjoints
|
|
5
5
|
Author-email: Leon Herrmann <herrmann.leon@pm.me>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -93,31 +93,11 @@ Additional benefits of **CuWave** are
|
|
|
93
93
|
|
|
94
94
|
## Install
|
|
95
95
|
|
|
96
|
-
Dependencies are kept **lightweight**. Only CuPy is required beyond standard Python library.
|
|
97
|
-
|
|
98
96
|
```bash
|
|
99
|
-
pip install
|
|
100
|
-
pip install cuwave # or `pip install -e .` from a checkout
|
|
97
|
+
pip install cuwave
|
|
101
98
|
```
|
|
102
99
|
|
|
103
|
-
CuPy
|
|
104
|
-
all remaining dependencies are declared in `pyproject.toml`.
|
|
105
|
-
|
|
106
|
-
PyTorch is optional for the regularization via neural optimization; see [pytorch](https://pytorch.org/get-started/locally/) for the installation. Otherwise it is not needed.
|
|
107
|
-
|
|
108
|
-
> [!NOTE]
|
|
109
|
-
> Match PyTorch's CUDA version to CuPy's, or the two runtimes clash at the first kernel launch.
|
|
110
|
-
> With `cupy-cuda12x`:
|
|
111
|
-
> ```bash
|
|
112
|
-
> pip install torch --index-url https://download.pytorch.org/whl/cu128
|
|
113
|
-
> ```
|
|
114
|
-
|
|
115
|
-
The tests under `tests/` are `unittest` classes, but `pytest` is the recommended runner:
|
|
116
|
-
|
|
117
|
-
```bash
|
|
118
|
-
pip install pytest
|
|
119
|
-
python -m pytest tests/ -q # ~20 s on a GPU, ~3 s without: CUDA and PyTorch tests skip when unavailable
|
|
120
|
-
```
|
|
100
|
+
Requires an NVIDIA GPU and [CuPy](https://docs.cupy.dev/en/stable/install.html) matching your CUDA toolkit (e.g. `pip install cupy-cuda12x`), which is not pulled in automatically. For the full installation, including the optional PyTorch and running the tests, see [docs/install.md](https://github.com/cmpmech/cuwave/blob/main/docs/install.md).
|
|
121
101
|
|
|
122
102
|
## References
|
|
123
103
|
|
|
@@ -67,31 +67,11 @@ Additional benefits of **CuWave** are
|
|
|
67
67
|
|
|
68
68
|
## Install
|
|
69
69
|
|
|
70
|
-
Dependencies are kept **lightweight**. Only CuPy is required beyond standard Python library.
|
|
71
|
-
|
|
72
70
|
```bash
|
|
73
|
-
pip install
|
|
74
|
-
pip install cuwave # or `pip install -e .` from a checkout
|
|
71
|
+
pip install cuwave
|
|
75
72
|
```
|
|
76
73
|
|
|
77
|
-
CuPy
|
|
78
|
-
all remaining dependencies are declared in `pyproject.toml`.
|
|
79
|
-
|
|
80
|
-
PyTorch is optional for the regularization via neural optimization; see [pytorch](https://pytorch.org/get-started/locally/) for the installation. Otherwise it is not needed.
|
|
81
|
-
|
|
82
|
-
> [!NOTE]
|
|
83
|
-
> Match PyTorch's CUDA version to CuPy's, or the two runtimes clash at the first kernel launch.
|
|
84
|
-
> With `cupy-cuda12x`:
|
|
85
|
-
> ```bash
|
|
86
|
-
> pip install torch --index-url https://download.pytorch.org/whl/cu128
|
|
87
|
-
> ```
|
|
88
|
-
|
|
89
|
-
The tests under `tests/` are `unittest` classes, but `pytest` is the recommended runner:
|
|
90
|
-
|
|
91
|
-
```bash
|
|
92
|
-
pip install pytest
|
|
93
|
-
python -m pytest tests/ -q # ~20 s on a GPU, ~3 s without: CUDA and PyTorch tests skip when unavailable
|
|
94
|
-
```
|
|
74
|
+
Requires an NVIDIA GPU and [CuPy](https://docs.cupy.dev/en/stable/install.html) matching your CUDA toolkit (e.g. `pip install cupy-cuda12x`), which is not pulled in automatically. For the full installation, including the optional PyTorch and running the tests, see [docs/install.md](https://github.com/cmpmech/cuwave/blob/main/docs/install.md).
|
|
95
75
|
|
|
96
76
|
## References
|
|
97
77
|
|
|
@@ -25,7 +25,7 @@ import numpy.typing as npt
|
|
|
25
25
|
|
|
26
26
|
from .boundary import Clamped, Traction, faces_with
|
|
27
27
|
from .elastic import voigt
|
|
28
|
-
from .wave import PAIRS, Simulation, apply_cell_weights, grid_block
|
|
28
|
+
from .wave import PAIRS, Simulation, apply_cell_weights, axis_geometry, grid_block
|
|
29
29
|
|
|
30
30
|
KERNEL_PATH = Path(__file__).parent / "kernels" / "anisotropic.cu"
|
|
31
31
|
SENSITIVITY_PATH = Path(__file__).parent / "kernels" / "anisotropic_sensitivity.cu"
|
|
@@ -253,13 +253,6 @@ class AnisotropicElasticWave(Simulation):
|
|
|
253
253
|
"""Source scaling, unscaled since rho0 is already folded into the lumped inertia."""
|
|
254
254
|
return 1.0
|
|
255
255
|
|
|
256
|
-
def axis_geometry(self) -> list:
|
|
257
|
-
"""Extents and previous-axis strides, the step kernel's tail without the factors."""
|
|
258
|
-
geom = [self.Nx[0]]
|
|
259
|
-
for d in range(1, self.ndim):
|
|
260
|
-
geom += [self.Nx[d], self.strides[d - 1]]
|
|
261
|
-
return geom
|
|
262
|
-
|
|
263
256
|
def gradient_fields(self, mat: dict) -> dict[str, cpt.NDArray]:
|
|
264
257
|
"""Nodal accumulators plus the cell one the stiffness density lands in first."""
|
|
265
258
|
grads = {
|
|
@@ -285,7 +278,7 @@ class AnisotropicElasticWave(Simulation):
|
|
|
285
278
|
grads["material"],
|
|
286
279
|
grads["design"],
|
|
287
280
|
self.dtype(1.0 / 2.0**self.ndim),
|
|
288
|
-
*
|
|
281
|
+
*axis_geometry(self),
|
|
289
282
|
],
|
|
290
283
|
)
|
|
291
284
|
return {"mass": grads["mass"], "stiff": grads["stiff"]}
|
|
@@ -302,7 +295,7 @@ class AnisotropicElasticWave(Simulation):
|
|
|
302
295
|
mat["stencil"],
|
|
303
296
|
mass_factor,
|
|
304
297
|
np.int32(self.comp_stride),
|
|
305
|
-
*
|
|
298
|
+
*axis_geometry(self),
|
|
306
299
|
]
|
|
307
300
|
|
|
308
301
|
def gradient_step(u0, u1, u2, l1):
|
|
@@ -323,7 +316,7 @@ class AnisotropicElasticWave(Simulation):
|
|
|
323
316
|
self.dtype(sign * self.density * volume / (2.0 * self.dt) ** 2),
|
|
324
317
|
self.dtype(-sign),
|
|
325
318
|
np.int32(self.comp_stride),
|
|
326
|
-
*
|
|
319
|
+
*axis_geometry(self),
|
|
327
320
|
]
|
|
328
321
|
|
|
329
322
|
def frechet_step(u0, u1, u2):
|
|
@@ -26,13 +26,12 @@ from .wave import (
|
|
|
26
26
|
Simulation,
|
|
27
27
|
apply_cell_weights,
|
|
28
28
|
axis_geometry,
|
|
29
|
-
component_weights,
|
|
30
29
|
grid_block,
|
|
31
30
|
pair_average,
|
|
32
31
|
pair_average_adjoint,
|
|
33
|
-
pair_weights,
|
|
34
32
|
point_average,
|
|
35
33
|
point_average_adjoint,
|
|
34
|
+
wall_weights,
|
|
36
35
|
)
|
|
37
36
|
|
|
38
37
|
KERNEL_PATH = Path(__file__).parent / "kernels" / "elastic.cu"
|
|
@@ -160,7 +159,7 @@ class ElasticWave(Simulation):
|
|
|
160
159
|
for c in range(self.ncomp):
|
|
161
160
|
mass = (
|
|
162
161
|
self.density
|
|
163
|
-
*
|
|
162
|
+
* wall_weights(self, (c,))
|
|
164
163
|
* point_average(self, gamma, c)
|
|
165
164
|
)
|
|
166
165
|
minv[c] = 1.0 / cp.maximum(mass, cp.finfo(self.dtype).tiny)
|
|
@@ -178,7 +177,7 @@ class ElasticWave(Simulation):
|
|
|
178
177
|
if self.ndim > 1:
|
|
179
178
|
gshear = cp.zeros((self.npairs, *self.Nx_padded), dtype=self.dtype)
|
|
180
179
|
for p, axes in enumerate(PAIRS[self.ndim][self.ndim :]):
|
|
181
|
-
gshear[p] =
|
|
180
|
+
gshear[p] = wall_weights(self, axes) * pair_average(self, gamma, axes)
|
|
182
181
|
mat["gshear"] = cp.ascontiguousarray(gshear)
|
|
183
182
|
if self.damping is not None:
|
|
184
183
|
mat["damping"] = self.damping
|
|
@@ -269,12 +268,12 @@ class ElasticWave(Simulation):
|
|
|
269
268
|
gamma = grads["design"]
|
|
270
269
|
g_mass = cp.zeros(self.Nx_padded, dtype=self.dtype)
|
|
271
270
|
for c in range(self.ncomp):
|
|
272
|
-
density = grads["mass"][c] *
|
|
271
|
+
density = grads["mass"][c] * wall_weights(self, (c,)) * self.density
|
|
273
272
|
g_mass += point_average_adjoint(self, density, c)
|
|
274
273
|
# the normal density sits on the nodes, so its chain rule is the weight alone
|
|
275
274
|
g_stiff = apply_cell_weights(self, grads["normal"].copy())
|
|
276
275
|
for p, axes in enumerate(PAIRS[self.ndim][self.ndim :]):
|
|
277
|
-
density = grads["shear"][p] *
|
|
276
|
+
density = grads["shear"][p] * wall_weights(self, axes)
|
|
278
277
|
g_stiff += pair_average_adjoint(self, density, gamma, axes)
|
|
279
278
|
return {"mass": g_mass, "stiff": g_stiff}
|
|
280
279
|
|
|
@@ -69,6 +69,22 @@ excitation_kernel(real_t *__restrict__ u, const real_t *__restrict__ source,
|
|
|
69
69
|
}
|
|
70
70
|
}
|
|
71
71
|
|
|
72
|
+
// ------------------------------------------------------------------------------------
|
|
73
|
+
__global__ void adjoint_excitation_kernel(
|
|
74
|
+
real_t *__restrict__ l2, const real_t *__restrict__ signal,
|
|
75
|
+
const int offset, const int *__restrict__ lin_index, const int num_sensors,
|
|
76
|
+
const real_t *__restrict__ weight, real_t *__restrict__ g_mass,
|
|
77
|
+
const real_t *__restrict__ u1, const real_t mf) {
|
|
78
|
+
const int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
|
79
|
+
if (idx < num_sensors) {
|
|
80
|
+
const int n = lin_index[idx];
|
|
81
|
+
const real_t load = weight[idx] * signal[offset + idx];
|
|
82
|
+
atomicAdd(&l2[n], load);
|
|
83
|
+
// the injected load is part of the second time difference of lambda
|
|
84
|
+
atomicAdd(&g_mass[n], -mf * u1[n] * load);
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
72
88
|
// ------------------------------------------------------------------------------------
|
|
73
89
|
__global__ void get_signal_kernel(const real_t *__restrict__ u,
|
|
74
90
|
real_t *__restrict__ um, const int offset,
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
// Compile-time configuration this file responds to:
|
|
3
3
|
// NDIM = 1 | 2 | 3
|
|
4
4
|
// USE_DAMPING
|
|
5
|
+
// USE_DOMAIN
|
|
5
6
|
|
|
6
7
|
// spatial finite difference stencil for Laplacian
|
|
7
8
|
__device__ __forceinline__ real_t flux_divergence_axis(
|
|
@@ -25,6 +26,81 @@ __device__ __forceinline__ real_t flux_divergence_axis(
|
|
|
25
26
|
return factor * (Dp * gp - Dm * gm); // outer grad (incl. inner grad)
|
|
26
27
|
}
|
|
27
28
|
|
|
29
|
+
// ----------------------------------- domain helpers
|
|
30
|
+
#ifdef USE_DOMAIN
|
|
31
|
+
#define DIRICHLET 7 // the radius code of a cell open onto a node held at zero
|
|
32
|
+
|
|
33
|
+
// the same with each cell at its own radius, zero for a cell leaving the domain
|
|
34
|
+
__device__ __forceinline__ real_t flux_divergence_cells(
|
|
35
|
+
const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
|
|
36
|
+
const int idx, const int s, const real_t uc, const real_t sc,
|
|
37
|
+
const real_t factor, const int rp, const int rm) {
|
|
38
|
+
real_t Dp = 0, Dm = 0; // a closed cell carries no flux
|
|
39
|
+
if (rp == DIRICHLET)
|
|
40
|
+
Dp = -uc * sc; // onto a zero halfway across, at the node's own stiffness
|
|
41
|
+
else if (rp) {
|
|
42
|
+
const real_t sp = stiff[idx + s];
|
|
43
|
+
Dp = OP_W(rp, 1) * (u1[idx + s] - uc);
|
|
44
|
+
#pragma unroll
|
|
45
|
+
for (int k = 2; k <= STENCIL_RADIUS; ++k)
|
|
46
|
+
if (k <= rp)
|
|
47
|
+
Dp += OP_W(rp, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]);
|
|
48
|
+
Dp *= sc * sp / (sc + sp); // harmonic mean
|
|
49
|
+
}
|
|
50
|
+
if (rm == DIRICHLET)
|
|
51
|
+
Dm = uc * sc;
|
|
52
|
+
else if (rm) {
|
|
53
|
+
const real_t sm = stiff[idx - s];
|
|
54
|
+
Dm = OP_W(rm, 1) * (uc - u1[idx - s]);
|
|
55
|
+
#pragma unroll
|
|
56
|
+
for (int k = 2; k <= STENCIL_RADIUS; ++k)
|
|
57
|
+
if (k <= rm)
|
|
58
|
+
Dm += OP_W(rm, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]);
|
|
59
|
+
Dm *= sc * sm / (sc + sm); // harmonic mean
|
|
60
|
+
}
|
|
61
|
+
return factor * (Dp - Dm);
|
|
62
|
+
}
|
|
63
|
+
#endif
|
|
64
|
+
|
|
65
|
+
#ifdef USE_DOMAIN
|
|
66
|
+
#define BLOCK_X (tile & 1023) // one block per domain tile, 10 bits an axis
|
|
67
|
+
#define BLOCK_Y ((tile >> 10) & 1023)
|
|
68
|
+
#define BLOCK_Z ((tile >> 20) & 1023)
|
|
69
|
+
#else
|
|
70
|
+
#define BLOCK_X blockIdx.x
|
|
71
|
+
#define BLOCK_Y blockIdx.y
|
|
72
|
+
#define BLOCK_Z blockIdx.z
|
|
73
|
+
#endif
|
|
74
|
+
|
|
75
|
+
#ifdef USE_DOMAIN
|
|
76
|
+
#define CELLS(s, f, d) \
|
|
77
|
+
flux_divergence_cells(u1, stiff, idx, s, uc, sc, f, (code >> 6 * (d)) & 7, \
|
|
78
|
+
(code >> (6 * (d) + 3)) & 7)
|
|
79
|
+
|
|
80
|
+
// out of line, so the rare wall node costs the deep ones no registers
|
|
81
|
+
__device__ __noinline__ real_t
|
|
82
|
+
wall_laplacian(const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
|
|
83
|
+
const int idx, const real_t uc, const real_t sc, const int code,
|
|
84
|
+
const real_t f0
|
|
85
|
+
#if NDIM >= 2
|
|
86
|
+
,
|
|
87
|
+
const real_t f1, const int s0
|
|
88
|
+
#endif
|
|
89
|
+
#if NDIM >= 3
|
|
90
|
+
,
|
|
91
|
+
const real_t f2, const int s1
|
|
92
|
+
#endif
|
|
93
|
+
) {
|
|
94
|
+
#if NDIM == 1
|
|
95
|
+
return CELLS(1, f0, 0);
|
|
96
|
+
#elif NDIM == 2
|
|
97
|
+
return CELLS(s0, f0, 0) + CELLS(1, f1, 1);
|
|
98
|
+
#elif NDIM == 3
|
|
99
|
+
return CELLS(s0, f0, 0) + CELLS(s1, f1, 1) + CELLS(1, f2, 2);
|
|
100
|
+
#endif
|
|
101
|
+
}
|
|
102
|
+
#endif
|
|
103
|
+
|
|
28
104
|
// ----------------------------- boundary condition helper
|
|
29
105
|
#if NDIM == 1
|
|
30
106
|
#define BC_PARAMS const int N0
|
|
@@ -80,6 +156,9 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
|
|
|
80
156
|
const real_t *__restrict__ minv, const int derive_inertia,
|
|
81
157
|
#ifdef USE_DAMPING
|
|
82
158
|
const real_t *__restrict__ damping, const real_t dt,
|
|
159
|
+
#endif
|
|
160
|
+
#ifdef USE_DOMAIN
|
|
161
|
+
const int *__restrict__ cells, const int *__restrict__ tiles,
|
|
83
162
|
#endif
|
|
84
163
|
const real_t f0, const int N0
|
|
85
164
|
#if NDIM >= 2
|
|
@@ -91,23 +170,26 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
|
|
|
91
170
|
const real_t f2, const int N2, const int s1
|
|
92
171
|
#endif
|
|
93
172
|
) {
|
|
173
|
+
#ifdef USE_DOMAIN
|
|
174
|
+
const int tile = tiles[blockIdx.x];
|
|
175
|
+
#endif
|
|
94
176
|
#if NDIM == 1
|
|
95
|
-
const int a0 =
|
|
177
|
+
const int a0 = BLOCK_X * blockDim.x + threadIdx.x;
|
|
96
178
|
if (!(a0 > 0 && a0 < N0 - 1))
|
|
97
179
|
return;
|
|
98
180
|
const int idx = a0;
|
|
99
181
|
const int r0 = CLOSURE(a0, N0);
|
|
100
182
|
#elif NDIM == 2
|
|
101
|
-
const int a1 =
|
|
102
|
-
const int a0 =
|
|
183
|
+
const int a1 = BLOCK_X * blockDim.x + threadIdx.x;
|
|
184
|
+
const int a0 = BLOCK_Y * blockDim.y + threadIdx.y;
|
|
103
185
|
if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1))
|
|
104
186
|
return;
|
|
105
187
|
const int idx = a0 * s0 + a1;
|
|
106
188
|
const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1);
|
|
107
189
|
#elif NDIM == 3
|
|
108
|
-
const int a2 =
|
|
109
|
-
const int a1 =
|
|
110
|
-
const int a0 =
|
|
190
|
+
const int a2 = BLOCK_X * blockDim.x + threadIdx.x;
|
|
191
|
+
const int a1 = BLOCK_Y * blockDim.y + threadIdx.y;
|
|
192
|
+
const int a0 = BLOCK_Z * blockDim.z + threadIdx.z;
|
|
111
193
|
if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1 && a2 > 0 &&
|
|
112
194
|
a2 < N2 - 1))
|
|
113
195
|
return;
|
|
@@ -117,7 +199,6 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
|
|
|
117
199
|
|
|
118
200
|
const real_t uc = u1[idx]; // load once
|
|
119
201
|
const real_t sc = stiff[idx]; // load once
|
|
120
|
-
|
|
121
202
|
#if NDIM == 1
|
|
122
203
|
real_t laplacian = flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f0, r0);
|
|
123
204
|
#elif NDIM == 2
|
|
@@ -128,6 +209,25 @@ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
|
|
|
128
209
|
flux_divergence_axis(u1, stiff, idx, s1, uc, sc, f1, r1) +
|
|
129
210
|
flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f2, r2);
|
|
130
211
|
#endif
|
|
212
|
+
#ifdef USE_DOMAIN
|
|
213
|
+
// bit 30 marks a tile touching the wall, so the branch is uniform per block
|
|
214
|
+
if (tile >> 30) {
|
|
215
|
+
const int code = cells[idx]; // 0 outside, bit 30 deep, else the cell radii
|
|
216
|
+
if (!code)
|
|
217
|
+
return; // outside the domain nothing is stepped
|
|
218
|
+
#if NDIM == 1
|
|
219
|
+
if (!(code >> 30))
|
|
220
|
+
laplacian = wall_laplacian(u1, stiff, idx, uc, sc, code, f0);
|
|
221
|
+
#elif NDIM == 2
|
|
222
|
+
if (!(code >> 30))
|
|
223
|
+
laplacian = wall_laplacian(u1, stiff, idx, uc, sc, code, f0, f1, s0);
|
|
224
|
+
#elif NDIM == 3
|
|
225
|
+
if (!(code >> 30))
|
|
226
|
+
laplacian =
|
|
227
|
+
wall_laplacian(u1, stiff, idx, uc, sc, code, f0, f1, s0, f2, s1);
|
|
228
|
+
#endif
|
|
229
|
+
}
|
|
230
|
+
#endif
|
|
131
231
|
|
|
132
232
|
const real_t mi = derive_inertia ? 1.f / sc : minv[idx];
|
|
133
233
|
#ifdef USE_DAMPING
|
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
// Prepended by wave.compile_kernels: stencils.preamble, then common.cuh.
|
|
2
|
+
// Compile-time configuration this file responds to:
|
|
3
|
+
// NDIM = 1 | 2 | 3
|
|
4
|
+
// USE_DAMPING
|
|
5
|
+
|
|
6
|
+
// ------------------------------- interior guard macros
|
|
7
|
+
#if NDIM == 1
|
|
8
|
+
#define INTERIOR_OR_RETURN \
|
|
9
|
+
const int a0 = blockIdx.x * blockDim.x + threadIdx.x; \
|
|
10
|
+
if (!(a0 > 0 && a0 < N0 - 1)) \
|
|
11
|
+
return; \
|
|
12
|
+
const int idx = a0
|
|
13
|
+
#define AXIS_RADII const int r0 = CLOSURE(a0, N0)
|
|
14
|
+
#define AXIS_OFFSETS const int o0 = 1
|
|
15
|
+
#elif NDIM == 2
|
|
16
|
+
#define INTERIOR_OR_RETURN \
|
|
17
|
+
const int a1 = blockIdx.x * blockDim.x + threadIdx.x; \
|
|
18
|
+
const int a0 = blockIdx.y * blockDim.y + threadIdx.y; \
|
|
19
|
+
if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1)) \
|
|
20
|
+
return; \
|
|
21
|
+
const int idx = a0 * s0 + a1
|
|
22
|
+
#define AXIS_RADII const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1)
|
|
23
|
+
#define AXIS_OFFSETS const int o0 = s0, o1 = 1
|
|
24
|
+
#elif NDIM == 3
|
|
25
|
+
#define INTERIOR_OR_RETURN \
|
|
26
|
+
const int a2 = blockIdx.x * blockDim.x + threadIdx.x; \
|
|
27
|
+
const int a1 = blockIdx.y * blockDim.y + threadIdx.y; \
|
|
28
|
+
const int a0 = blockIdx.z * blockDim.z + threadIdx.z; \
|
|
29
|
+
if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1 && a2 > 0 && \
|
|
30
|
+
a2 < N2 - 1)) \
|
|
31
|
+
return; \
|
|
32
|
+
const int idx = a0 * s0 + a1 * s1 + a2
|
|
33
|
+
#define AXIS_RADII \
|
|
34
|
+
const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1), r2 = CLOSURE(a2, N2)
|
|
35
|
+
#define AXIS_OFFSETS const int o0 = s0, o1 = s1, o2 = 1
|
|
36
|
+
#endif
|
|
37
|
+
|
|
38
|
+
// ---------------------------------- gradient helpers
|
|
39
|
+
__device__ __forceinline__ real_t stiffness_gradient_axis(
|
|
40
|
+
const real_t *__restrict__ u1, const real_t *__restrict__ l1,
|
|
41
|
+
const real_t *__restrict__ stiff, const int idx, const int s,
|
|
42
|
+
const real_t uc, const real_t lc, const real_t sc, const real_t factor,
|
|
43
|
+
const int r) {
|
|
44
|
+
const real_t sp = stiff[idx + s]; // plus of sc
|
|
45
|
+
const real_t sm = stiff[idx - s]; // minus of sc
|
|
46
|
+
const real_t dgp = sp * sp / ((sc + sp) * (sc + sp)); // d(harmonic mean)/dsc
|
|
47
|
+
const real_t dgm = sm * sm / ((sc + sm) * (sc + sm)); // d(harmonic mean)/dsc
|
|
48
|
+
real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // inner grad
|
|
49
|
+
real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // inner grad
|
|
50
|
+
#pragma unroll
|
|
51
|
+
for (int k = 2; k <= STENCIL_RADIUS; ++k)
|
|
52
|
+
if (k <= r) {
|
|
53
|
+
Dp += OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]);
|
|
54
|
+
Dm += OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]);
|
|
55
|
+
}
|
|
56
|
+
return factor * (dgp * Dp * (l1[idx + s] - lc) +
|
|
57
|
+
dgm * Dm * (lc - l1[idx - s])); // both cells of the node
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// the adjoint step's flux divergence and the stiffness gradient of one axis,
|
|
61
|
+
// sharing the cell stiffnesses
|
|
62
|
+
__device__ __forceinline__ void adjoint_gradient_axis(
|
|
63
|
+
const real_t *__restrict__ u1, const real_t *__restrict__ l1,
|
|
64
|
+
const real_t *__restrict__ stiff, const int idx, const int s,
|
|
65
|
+
const real_t uc, const real_t lc, const real_t sc, const real_t factor,
|
|
66
|
+
const int r, real_t &div_l, real_t &g) {
|
|
67
|
+
const real_t sp = stiff[idx + s]; // plus of sc
|
|
68
|
+
const real_t sm = stiff[idx - s]; // minus of sc
|
|
69
|
+
const real_t gp = sc * sp / (sc + sp); // harmonic mean
|
|
70
|
+
const real_t gm = sc * sm / (sc + sm); // harmonic mean
|
|
71
|
+
const real_t dgp = sp * sp / ((sc + sp) * (sc + sp)); // d(harmonic mean)/dsc
|
|
72
|
+
const real_t dgm = sm * sm / ((sc + sm) * (sc + sm)); // d(harmonic mean)/dsc
|
|
73
|
+
const real_t lp = l1[idx + s];
|
|
74
|
+
const real_t lm = l1[idx - s];
|
|
75
|
+
real_t Lp = OP_W(r, 1) * (lp - lc); // inner grad of lambda
|
|
76
|
+
real_t Lm = OP_W(r, 1) * (lc - lm); // inner grad of lambda
|
|
77
|
+
real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // inner grad of u
|
|
78
|
+
real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // inner grad of u
|
|
79
|
+
#pragma unroll
|
|
80
|
+
for (int k = 2; k <= STENCIL_RADIUS; ++k)
|
|
81
|
+
if (k <= r) {
|
|
82
|
+
Lp += OP_W(r, k) * (l1[idx + k * s] - l1[idx - (k - 1) * s]);
|
|
83
|
+
Lm += OP_W(r, k) * (l1[idx + (k - 1) * s] - l1[idx - k * s]);
|
|
84
|
+
Dp += OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]);
|
|
85
|
+
Dm += OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]);
|
|
86
|
+
}
|
|
87
|
+
div_l += factor * (Lp * gp - Lm * gm); // the step's flux divergence
|
|
88
|
+
g += factor * (dgp * Dp * (lp - lc) + dgm * Dm * (lc - lm));
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// byte-identical to scalar.cu: each file is its own compilation unit
|
|
92
|
+
__device__ __forceinline__ real_t flux_divergence_axis(
|
|
93
|
+
const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
|
|
94
|
+
const int idx, const int s, const real_t uc, const real_t sc,
|
|
95
|
+
const real_t factor, const int r) {
|
|
96
|
+
const real_t sp = stiff[idx + s]; // plus of sc
|
|
97
|
+
const real_t sm = stiff[idx - s]; // minus of sc
|
|
98
|
+
const real_t gp = sc * sp / (sc + sp); // harmonic mean
|
|
99
|
+
const real_t gm = sc * sm / (sc + sm); // harmonic mean
|
|
100
|
+
real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // initialization: inner grad
|
|
101
|
+
real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // initialization: inner grad
|
|
102
|
+
#pragma unroll
|
|
103
|
+
for (int k = 2; k <= STENCIL_RADIUS; ++k)
|
|
104
|
+
if (k <= r) {
|
|
105
|
+
Dp +=
|
|
106
|
+
OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]); // inner grad
|
|
107
|
+
Dm +=
|
|
108
|
+
OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]); // inner grad
|
|
109
|
+
}
|
|
110
|
+
return factor * (Dp * gp - Dm * gm); // outer grad (incl. inner grad)
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// -------------------------------------- kernels
|
|
114
|
+
extern "C" {
|
|
115
|
+
|
|
116
|
+
// ------------------------------------------------------------------------------------
|
|
117
|
+
__global__ void
|
|
118
|
+
gradient_kernel(real_t *__restrict__ g_mass, real_t *__restrict__ g_stiff,
|
|
119
|
+
const real_t *__restrict__ u0, const real_t *__restrict__ u1,
|
|
120
|
+
const real_t *__restrict__ u2, const real_t *__restrict__ l1,
|
|
121
|
+
const real_t *__restrict__ stiff, const real_t inv_dt2,
|
|
122
|
+
const real_t F0, const int N0
|
|
123
|
+
#if NDIM >= 2
|
|
124
|
+
,
|
|
125
|
+
const real_t F1, const int N1, const int s0
|
|
126
|
+
#endif
|
|
127
|
+
#if NDIM >= 3
|
|
128
|
+
,
|
|
129
|
+
const real_t F2, const int N2, const int s1
|
|
130
|
+
#endif
|
|
131
|
+
) {
|
|
132
|
+
INTERIOR_OR_RETURN;
|
|
133
|
+
AXIS_RADII;
|
|
134
|
+
|
|
135
|
+
const real_t uc = u1[idx];
|
|
136
|
+
const real_t lc = l1[idx];
|
|
137
|
+
|
|
138
|
+
// dJ/dmass: no neighbour and no material load
|
|
139
|
+
g_mass[idx] -= inv_dt2 * lc * (u2[idx] - 2.f * uc + u0[idx]);
|
|
140
|
+
|
|
141
|
+
// dJ/dstiff: scalar.cu's harmonic cell mean differentiated in place
|
|
142
|
+
const real_t sc = stiff[idx];
|
|
143
|
+
#if NDIM == 1
|
|
144
|
+
const real_t g =
|
|
145
|
+
stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F0, r0);
|
|
146
|
+
#elif NDIM == 2
|
|
147
|
+
const real_t g =
|
|
148
|
+
stiffness_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, F0, r0) +
|
|
149
|
+
stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F1, r1);
|
|
150
|
+
#elif NDIM == 3
|
|
151
|
+
const real_t g =
|
|
152
|
+
stiffness_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, F0, r0) +
|
|
153
|
+
stiffness_gradient_axis(u1, l1, stiff, idx, s1, uc, lc, sc, F1, r1) +
|
|
154
|
+
stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F2, r2);
|
|
155
|
+
#endif
|
|
156
|
+
g_stiff[idx] -= g;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// ------------------------------------------------------------------------------------
|
|
160
|
+
__global__ void frechet_kernel(real_t *__restrict__ acc_mass,
|
|
161
|
+
real_t *__restrict__ acc_stiff,
|
|
162
|
+
const real_t *__restrict__ u0,
|
|
163
|
+
const real_t *__restrict__ u1,
|
|
164
|
+
const real_t *__restrict__ u2, const real_t ft,
|
|
165
|
+
const real_t f0, const int N0
|
|
166
|
+
#if NDIM >= 2
|
|
167
|
+
,
|
|
168
|
+
const real_t f1, const int N1, const int s0
|
|
169
|
+
#endif
|
|
170
|
+
#if NDIM >= 3
|
|
171
|
+
,
|
|
172
|
+
const real_t f2, const int N2, const int s1
|
|
173
|
+
#endif
|
|
174
|
+
) {
|
|
175
|
+
INTERIOR_OR_RETURN;
|
|
176
|
+
AXIS_OFFSETS;
|
|
177
|
+
|
|
178
|
+
const real_t dudt = u2[idx] - u0[idx];
|
|
179
|
+
acc_mass[idx] += ft * dudt * dudt;
|
|
180
|
+
|
|
181
|
+
const real_t g0 = u1[idx + o0] - u1[idx - o0];
|
|
182
|
+
real_t sum = f0 * g0 * g0;
|
|
183
|
+
#if NDIM >= 2
|
|
184
|
+
const real_t g1 = u1[idx + o1] - u1[idx - o1];
|
|
185
|
+
sum += f1 * g1 * g1;
|
|
186
|
+
#endif
|
|
187
|
+
#if NDIM >= 3
|
|
188
|
+
const real_t g2 = u1[idx + o2] - u1[idx - o2];
|
|
189
|
+
sum += f2 * g2 * g2;
|
|
190
|
+
#endif
|
|
191
|
+
acc_stiff[idx] += sum;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
// ------------------------------------------------------------------------------------
|
|
195
|
+
__global__ void adjoint_gradient_kernel(
|
|
196
|
+
real_t *__restrict__ l0, const real_t *__restrict__ l1,
|
|
197
|
+
real_t *__restrict__ g_mass, real_t *__restrict__ g_stiff,
|
|
198
|
+
const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
|
|
199
|
+
const real_t *__restrict__ minv, const int derive_inertia,
|
|
200
|
+
#ifdef USE_DAMPING
|
|
201
|
+
const real_t *__restrict__ damping, const real_t dt,
|
|
202
|
+
#endif
|
|
203
|
+
const real_t mf, const real_t f0, const int N0
|
|
204
|
+
#if NDIM >= 2
|
|
205
|
+
,
|
|
206
|
+
const real_t f1, const int N1, const int s0
|
|
207
|
+
#endif
|
|
208
|
+
#if NDIM >= 3
|
|
209
|
+
,
|
|
210
|
+
const real_t f2, const int N2, const int s1
|
|
211
|
+
#endif
|
|
212
|
+
) {
|
|
213
|
+
INTERIOR_OR_RETURN;
|
|
214
|
+
AXIS_RADII;
|
|
215
|
+
|
|
216
|
+
const real_t uc = u1[idx];
|
|
217
|
+
const real_t lc = l1[idx];
|
|
218
|
+
const real_t sc = stiff[idx];
|
|
219
|
+
real_t div_l = 0, g = 0;
|
|
220
|
+
#if NDIM == 1
|
|
221
|
+
adjoint_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, f0, r0, div_l, g);
|
|
222
|
+
#elif NDIM == 2
|
|
223
|
+
adjoint_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, f0, r0, div_l, g);
|
|
224
|
+
adjoint_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, f1, r1, div_l, g);
|
|
225
|
+
#elif NDIM == 3
|
|
226
|
+
adjoint_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, f0, r0, div_l, g);
|
|
227
|
+
adjoint_gradient_axis(u1, l1, stiff, idx, s1, uc, lc, sc, f1, r1, div_l, g);
|
|
228
|
+
adjoint_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, f2, r2, div_l, g);
|
|
229
|
+
#endif
|
|
230
|
+
|
|
231
|
+
const real_t mi = derive_inertia ? 1.f / sc : minv[idx];
|
|
232
|
+
const real_t lo = l0[idx];
|
|
233
|
+
#ifdef USE_DAMPING
|
|
234
|
+
const real_t beta = 0.5f * mi * damping[idx] * dt;
|
|
235
|
+
const real_t ln = (2.f * lc - lo * (1.f - beta) + mi * div_l) / (1.f + beta);
|
|
236
|
+
#else
|
|
237
|
+
const real_t ln = -lo + 2.f * lc + mi * div_l;
|
|
238
|
+
#endif
|
|
239
|
+
l0[idx] = ln;
|
|
240
|
+
// dJ/dmass by parts in time: u against the second difference of lambda
|
|
241
|
+
g_mass[idx] -= mf * uc * (ln - 2.f * lc + lo);
|
|
242
|
+
g_stiff[idx] -= mf * g;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
// ------------------------------------------------------------------------------------
|
|
246
|
+
__global__ void superposed_kernel(
|
|
247
|
+
const real_t *__restrict__ u0, const real_t *__restrict__ u1,
|
|
248
|
+
real_t *__restrict__ u2, real_t *__restrict__ acc_mass,
|
|
249
|
+
real_t *__restrict__ acc_stiff, const real_t *__restrict__ stiff,
|
|
250
|
+
const real_t *__restrict__ minv, const int derive_inertia, const real_t ft,
|
|
251
|
+
const real_t fs, const real_t f0, const int N0
|
|
252
|
+
#if NDIM >= 2
|
|
253
|
+
,
|
|
254
|
+
const real_t f1, const int N1, const int s0
|
|
255
|
+
#endif
|
|
256
|
+
#if NDIM >= 3
|
|
257
|
+
,
|
|
258
|
+
const real_t f2, const int N2, const int s1
|
|
259
|
+
#endif
|
|
260
|
+
) {
|
|
261
|
+
INTERIOR_OR_RETURN;
|
|
262
|
+
AXIS_RADII;
|
|
263
|
+
AXIS_OFFSETS;
|
|
264
|
+
|
|
265
|
+
const real_t uc = u1[idx];
|
|
266
|
+
const real_t sc = stiff[idx];
|
|
267
|
+
#if NDIM == 1
|
|
268
|
+
const real_t laplacian =
|
|
269
|
+
flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f0, r0);
|
|
270
|
+
#elif NDIM == 2
|
|
271
|
+
const real_t laplacian =
|
|
272
|
+
flux_divergence_axis(u1, stiff, idx, s0, uc, sc, f0, r0) +
|
|
273
|
+
flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f1, r1);
|
|
274
|
+
#elif NDIM == 3
|
|
275
|
+
const real_t laplacian =
|
|
276
|
+
flux_divergence_axis(u1, stiff, idx, s0, uc, sc, f0, r0) +
|
|
277
|
+
flux_divergence_axis(u1, stiff, idx, s1, uc, sc, f1, r1) +
|
|
278
|
+
flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f2, r2);
|
|
279
|
+
#endif
|
|
280
|
+
|
|
281
|
+
// the mass density of the triplet before, whose oldest slot u2 still holds
|
|
282
|
+
const real_t dudt = uc - u2[idx];
|
|
283
|
+
acc_mass[idx] += ft * dudt * dudt;
|
|
284
|
+
|
|
285
|
+
// the stiffness density of this triplet, its middle slot being u1
|
|
286
|
+
const real_t g0 = u1[idx + o0] - u1[idx - o0];
|
|
287
|
+
real_t sum = f0 * g0 * g0;
|
|
288
|
+
#if NDIM >= 2
|
|
289
|
+
const real_t g1 = u1[idx + o1] - u1[idx - o1];
|
|
290
|
+
sum += f1 * g1 * g1;
|
|
291
|
+
#endif
|
|
292
|
+
#if NDIM >= 3
|
|
293
|
+
const real_t g2 = u1[idx + o2] - u1[idx - o2];
|
|
294
|
+
sum += f2 * g2 * g2;
|
|
295
|
+
#endif
|
|
296
|
+
acc_stiff[idx] += fs * sum;
|
|
297
|
+
|
|
298
|
+
const real_t mi = derive_inertia ? 1.f / sc : minv[idx];
|
|
299
|
+
u2[idx] = -u0[idx] + 2.f * uc + mi * laplacian;
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
} // extern "C"
|