cuwave 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,139 @@
1
+ // Prepended by wave.compile_kernels: stencils.preamble, then common.cuh.
2
+ // Compile-time configuration this file responds to:
3
+ // NDIM = 2 | 3
4
+ // USE_MAGNETIC
5
+
6
+ #if NDIM == 1
7
+ #error "a single in-plane component has no curl; use maxwell.ElectricWave"
8
+ #endif
9
+
10
+ #if NDIM == 2
11
+ #define AXIS_PARAMS \
12
+ const real_t f0, const int N0, const real_t f1, const int N1, const int s0
13
+ #define INTERIOR_OR_RETURN \
14
+ const int a1 = blockIdx.x * blockDim.x + threadIdx.x; \
15
+ const int a0 = blockIdx.y * blockDim.y + threadIdx.y; \
16
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1)) \
17
+ return; \
18
+ const int idx = a0 * s0 + a1
19
+ #define AXIS_GEOM \
20
+ const int A[2] = {a0, a1}, S[2] = {s0, 1}, NN[2] = {N0, N1}
21
+ #define AXIS_FACTORS const real_t F[2] = {f0, f1}
22
+ #elif NDIM == 3
23
+ #define AXIS_PARAMS \
24
+ const real_t f0, const int N0, const real_t f1, const int N1, const int s0, \
25
+ const real_t f2, const int N2, const int s1
26
+ #define INTERIOR_OR_RETURN \
27
+ const int a2 = blockIdx.x * blockDim.x + threadIdx.x; \
28
+ const int a1 = blockIdx.y * blockDim.y + threadIdx.y; \
29
+ const int a0 = blockIdx.z * blockDim.z + threadIdx.z; \
30
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1 && a2 > 0 && \
31
+ a2 < N2 - 1)) \
32
+ return; \
33
+ const int idx = a0 * s0 + a1 * s1 + a2
34
+ #define AXIS_GEOM \
35
+ const int A[3] = {a0, a1, a2}, S[3] = {s0, s1, 1}, NN[3] = {N0, N1, N2}
36
+ #define AXIS_FACTORS const real_t F[3] = {f0, f1, f2}
37
+ #endif
38
+
39
+ #ifdef USE_MAGNETIC
40
+ // ------------------------------------ curl helpers
41
+ // byte-identical to maxwell.cu: each file is its own compilation unit
42
+ __device__ __forceinline__ real_t
43
+ curl_component(const real_t *__restrict__ u, const int idx, const int cs,
44
+ const int k, const int l, const int *A, const int *S,
45
+ const int *NN, const real_t *F) {
46
+ const int r1 = rad_half(A[l], NN[l]); // d u_k / d x_l
47
+ const int r2 = rad_half(A[k], NN[k]); // d u_l / d x_k
48
+ real_t b = (real_t)0;
49
+ #pragma unroll
50
+ for (int j = 1; j <= STENCIL_RADIUS; ++j) {
51
+ if (j <= r1)
52
+ b -= F[l] * SG_W(r1, j) *
53
+ (u[k * cs + idx + j * S[l]] - u[k * cs + idx - (j - 1) * S[l]]);
54
+ if (j <= r2)
55
+ b += F[k] * SG_W(r2, j) *
56
+ (u[l * cs + idx + j * S[k]] - u[l * cs + idx - (j - 1) * S[k]]);
57
+ }
58
+ return b;
59
+ }
60
+ #endif
61
+
62
+ // -------------------------------------- kernels
63
+ extern "C" {
64
+
65
+ // ------------------------------------------------------------------------------------
66
+ __global__ void gradient_kernel(real_t *__restrict__ g_mass,
67
+ #ifdef USE_MAGNETIC
68
+ real_t *__restrict__ g_nu,
69
+ #endif
70
+ const real_t *__restrict__ u0,
71
+ const real_t *__restrict__ u1,
72
+ const real_t *__restrict__ u2,
73
+ const real_t *__restrict__ l1, const real_t mf,
74
+ const int cs, AXIS_PARAMS) {
75
+ INTERIOR_OR_RETURN;
76
+ AXIS_GEOM;
77
+
78
+ // dJ/dpermittivity on the component points: no neighbour and no material load
79
+ #pragma unroll
80
+ for (int c = 0; c < NDIM; ++c) {
81
+ if (A[c] > NN[c] - 3)
82
+ continue;
83
+ const int n = c * cs + idx;
84
+ g_mass[n] -= mf * l1[n] * (u2[n] - 2.f * u1[n] + u0[n]);
85
+ }
86
+
87
+ #ifdef USE_MAGNETIC
88
+ AXIS_FACTORS;
89
+ // dJ/d(1/mu) on the pair points: forward curl against adjoint curl
90
+ #pragma unroll
91
+ for (int k = 0; k < NDIM - 1; ++k)
92
+ #pragma unroll
93
+ for (int l = k + 1; l < NDIM; ++l) {
94
+ if (A[k] > NN[k] - 3 || A[l] > NN[l] - 3)
95
+ continue;
96
+ const int p = PAIR_ROW(k, l) - NDIM;
97
+ g_nu[p * cs + idx] -= curl_component(l1, idx, cs, k, l, A, S, NN, F) *
98
+ curl_component(u1, idx, cs, k, l, A, S, NN, F);
99
+ }
100
+ #endif
101
+ }
102
+
103
+ // ------------------------------------------------------------------------------------
104
+ __global__ void frechet_kernel(real_t *__restrict__ acc_mass,
105
+ #ifdef USE_MAGNETIC
106
+ real_t *__restrict__ acc_nu,
107
+ #endif
108
+ const real_t *__restrict__ u0,
109
+ const real_t *__restrict__ u1,
110
+ const real_t *__restrict__ u2, const real_t ft,
111
+ const real_t fs, const int cs, AXIS_PARAMS) {
112
+ INTERIOR_OR_RETURN;
113
+ AXIS_GEOM;
114
+
115
+ #pragma unroll
116
+ for (int c = 0; c < NDIM; ++c) {
117
+ if (A[c] > NN[c] - 3)
118
+ continue;
119
+ const int n = c * cs + idx;
120
+ const real_t dudt = u2[n] - u0[n];
121
+ acc_mass[n] += ft * dudt * dudt;
122
+ }
123
+
124
+ #ifdef USE_MAGNETIC
125
+ AXIS_FACTORS;
126
+ #pragma unroll
127
+ for (int k = 0; k < NDIM - 1; ++k)
128
+ #pragma unroll
129
+ for (int l = k + 1; l < NDIM; ++l) {
130
+ if (A[k] > NN[k] - 3 || A[l] > NN[l] - 3)
131
+ continue;
132
+ const int p = PAIR_ROW(k, l) - NDIM;
133
+ const real_t b = curl_component(u1, idx, cs, k, l, A, S, NN, F);
134
+ acc_nu[p * cs + idx] += fs * b * b;
135
+ }
136
+ #endif
137
+ }
138
+
139
+ } // extern "C"
@@ -0,0 +1,164 @@
1
+ // Prepended by wave.compile_kernels: stencils.preamble, then common.cuh.
2
+ // Compile-time configuration this file responds to:
3
+ // NDIM = 1 | 2 | 3
4
+ // USE_DAMPING
5
+
6
+ // spatial finite difference stencil for Laplacian
7
+ __device__ __forceinline__ real_t flux_divergence_axis(
8
+ const real_t *__restrict__ u1, const real_t *__restrict__ stiff,
9
+ const int idx, const int s, const real_t uc, const real_t sc,
10
+ const real_t factor, const int r) {
11
+ const real_t sp = stiff[idx + s]; // plus of sc
12
+ const real_t sm = stiff[idx - s]; // minus of sc
13
+ const real_t gp = sc * sp / (sc + sp); // harmonic mean
14
+ const real_t gm = sc * sm / (sc + sm); // harmonic mean
15
+ real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // initialization: inner grad
16
+ real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // initialization: inner grad
17
+ #pragma unroll
18
+ for (int k = 2; k <= STENCIL_RADIUS; ++k)
19
+ if (k <= r) {
20
+ Dp +=
21
+ OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]); // inner grad
22
+ Dm +=
23
+ OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]); // inner grad
24
+ }
25
+ return factor * (Dp * gp - Dm * gm); // outer grad (incl. inner grad)
26
+ }
27
+
28
+ // ----------------------------- boundary condition helper
29
+ #if NDIM == 1
30
+ #define BC_PARAMS const int N0
31
+ #define BC_GEOM const int n[1] = {N0}, s[1] = {1}
32
+ #elif NDIM == 2
33
+ #define BC_PARAMS const int N0, const int N1, const int s0
34
+ #define BC_GEOM const int n[2] = {N0, N1}, s[2] = {s0, 1}
35
+ #elif NDIM == 3
36
+ #define BC_PARAMS \
37
+ const int N0, const int N1, const int s0, const int N2, const int s1
38
+ #define BC_GEOM const int n[3] = {N0, N1, N2}, s[3] = {s0, s1, 1}
39
+ #endif
40
+
41
+ __device__ __forceinline__ bool bc_ghost(int t, const int faces, const int *n,
42
+ const int *s, int &ghost,
43
+ int &normal) {
44
+ #pragma unroll
45
+ for (int d = 0; d < NDIM; ++d) {
46
+ int face = 1; // nodes on one ghost face of axis d
47
+ #pragma unroll
48
+ for (int k = 0; k < NDIM; ++k)
49
+ if (k != d)
50
+ face *= n[k] - 2;
51
+
52
+ const int lo = (faces >> (2 * d)) & 1, hi = (faces >> (2 * d + 1)) & 1;
53
+ if (t < (lo + hi) * face) {
54
+ const int high =
55
+ (lo && t < face) ? 0 : 1; // low ghost (0) or high (n[d] - 1)
56
+ int r = t - (high ? lo * face : 0);
57
+ int off = 0; // position within the face, axis d excluded
58
+ #pragma unroll
59
+ for (int k = NDIM - 1; k >= 0; --k)
60
+ if (k != d) {
61
+ off += (r % (n[k] - 2) + 1) * s[k];
62
+ r /= n[k] - 2;
63
+ }
64
+ ghost = off + (high ? n[d] - 1 : 0) * s[d];
65
+ normal = high ? -s[d] : s[d];
66
+ return true;
67
+ }
68
+ t -= (lo + hi) * face;
69
+ }
70
+ return false;
71
+ }
72
+
73
+ // -------------------------------------- kernels
74
+ extern "C" {
75
+
76
+ // ------------------------------------------------------------------------------------
77
+ __global__ void
78
+ fd_kernel(const real_t *__restrict__ u0, const real_t *__restrict__ u1,
79
+ real_t *__restrict__ u2, const real_t *__restrict__ stiff,
80
+ const real_t *__restrict__ minv, const int derive_inertia,
81
+ #ifdef USE_DAMPING
82
+ const real_t *__restrict__ damping, const real_t dt,
83
+ #endif
84
+ const real_t f0, const int N0
85
+ #if NDIM >= 2
86
+ ,
87
+ const real_t f1, const int N1, const int s0
88
+ #endif
89
+ #if NDIM >= 3
90
+ ,
91
+ const real_t f2, const int N2, const int s1
92
+ #endif
93
+ ) {
94
+ #if NDIM == 1
95
+ const int a0 = blockIdx.x * blockDim.x + threadIdx.x;
96
+ if (!(a0 > 0 && a0 < N0 - 1))
97
+ return;
98
+ const int idx = a0;
99
+ const int r0 = CLOSURE(a0, N0);
100
+ #elif NDIM == 2
101
+ const int a1 = blockIdx.x * blockDim.x + threadIdx.x;
102
+ const int a0 = blockIdx.y * blockDim.y + threadIdx.y;
103
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1))
104
+ return;
105
+ const int idx = a0 * s0 + a1;
106
+ const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1);
107
+ #elif NDIM == 3
108
+ const int a2 = blockIdx.x * blockDim.x + threadIdx.x;
109
+ const int a1 = blockIdx.y * blockDim.y + threadIdx.y;
110
+ const int a0 = blockIdx.z * blockDim.z + threadIdx.z;
111
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1 && a2 > 0 &&
112
+ a2 < N2 - 1))
113
+ return;
114
+ const int idx = a0 * s0 + a1 * s1 + a2;
115
+ const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1), r2 = CLOSURE(a2, N2);
116
+ #endif
117
+
118
+ const real_t uc = u1[idx]; // load once
119
+ const real_t sc = stiff[idx]; // load once
120
+
121
+ #if NDIM == 1
122
+ real_t laplacian = flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f0, r0);
123
+ #elif NDIM == 2
124
+ real_t laplacian = flux_divergence_axis(u1, stiff, idx, s0, uc, sc, f0, r0) +
125
+ flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f1, r1);
126
+ #elif NDIM == 3
127
+ real_t laplacian = flux_divergence_axis(u1, stiff, idx, s0, uc, sc, f0, r0) +
128
+ flux_divergence_axis(u1, stiff, idx, s1, uc, sc, f1, r1) +
129
+ flux_divergence_axis(u1, stiff, idx, 1, uc, sc, f2, r2);
130
+ #endif
131
+
132
+ const real_t mi = derive_inertia ? 1.f / sc : minv[idx];
133
+ #ifdef USE_DAMPING
134
+ const real_t beta = 0.5f * mi * damping[idx] * dt;
135
+ u2[idx] = (2.f * uc - u0[idx] * (1.f - beta) + mi * laplacian) / (1.f + beta);
136
+ #else
137
+ u2[idx] = -u0[idx] + 2.f * uc + mi * laplacian;
138
+ #endif
139
+ }
140
+
141
+ // ------------------------------------------------------------------------------------
142
+ __global__ void homogeneous_neumann_kernel(real_t *__restrict__ u,
143
+ const int faces, BC_PARAMS) {
144
+ BC_GEOM;
145
+ int ghost, normal;
146
+ if (!bc_ghost(blockIdx.x * blockDim.x + threadIdx.x, faces, n, s, ghost,
147
+ normal))
148
+ return;
149
+ u[ghost] = u[ghost + 2 * normal];
150
+ }
151
+
152
+ // ------------------------------------------------------------------------------------
153
+ __global__ void homogeneous_dirichlet_kernel(real_t *__restrict__ u,
154
+ const int faces, BC_PARAMS) {
155
+ BC_GEOM;
156
+ int ghost, normal;
157
+ if (!bc_ghost(blockIdx.x * blockDim.x + threadIdx.x, faces, n, s, ghost,
158
+ normal))
159
+ return;
160
+ u[ghost] = -u[ghost + 2 * normal];
161
+ }
162
+
163
+
164
+ } // extern "C"
@@ -0,0 +1,140 @@
1
+ // Prepended by wave.compile_kernels: stencils.preamble, then common.cuh.
2
+ // Compile-time configuration this file responds to:
3
+ // NDIM = 1 | 2 | 3
4
+
5
+ // ------------------------------- interior guard macros
6
+ #if NDIM == 1
7
+ #define INTERIOR_OR_RETURN \
8
+ const int a0 = blockIdx.x * blockDim.x + threadIdx.x; \
9
+ if (!(a0 > 0 && a0 < N0 - 1)) \
10
+ return; \
11
+ const int idx = a0
12
+ #define AXIS_RADII const int r0 = CLOSURE(a0, N0)
13
+ #define AXIS_OFFSETS const int o0 = 1
14
+ #elif NDIM == 2
15
+ #define INTERIOR_OR_RETURN \
16
+ const int a1 = blockIdx.x * blockDim.x + threadIdx.x; \
17
+ const int a0 = blockIdx.y * blockDim.y + threadIdx.y; \
18
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1)) \
19
+ return; \
20
+ const int idx = a0 * s0 + a1
21
+ #define AXIS_RADII const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1)
22
+ #define AXIS_OFFSETS const int o0 = s0, o1 = 1
23
+ #elif NDIM == 3
24
+ #define INTERIOR_OR_RETURN \
25
+ const int a2 = blockIdx.x * blockDim.x + threadIdx.x; \
26
+ const int a1 = blockIdx.y * blockDim.y + threadIdx.y; \
27
+ const int a0 = blockIdx.z * blockDim.z + threadIdx.z; \
28
+ if (!(a0 > 0 && a0 < N0 - 1 && a1 > 0 && a1 < N1 - 1 && a2 > 0 && \
29
+ a2 < N2 - 1)) \
30
+ return; \
31
+ const int idx = a0 * s0 + a1 * s1 + a2
32
+ #define AXIS_RADII \
33
+ const int r0 = CLOSURE(a0, N0), r1 = CLOSURE(a1, N1), r2 = CLOSURE(a2, N2)
34
+ #define AXIS_OFFSETS const int o0 = s0, o1 = s1, o2 = 1
35
+ #endif
36
+
37
+ // ---------------------------------- gradient helpers
38
+ __device__ __forceinline__ real_t stiffness_gradient_axis(
39
+ const real_t *__restrict__ u1, const real_t *__restrict__ l1,
40
+ const real_t *__restrict__ stiff, const int idx, const int s,
41
+ const real_t uc, const real_t lc, const real_t sc, const real_t factor,
42
+ const int r) {
43
+ const real_t sp = stiff[idx + s]; // plus of sc
44
+ const real_t sm = stiff[idx - s]; // minus of sc
45
+ const real_t dgp = sp * sp / ((sc + sp) * (sc + sp)); // d(harmonic mean)/dsc
46
+ const real_t dgm = sm * sm / ((sc + sm) * (sc + sm)); // d(harmonic mean)/dsc
47
+ real_t Dp = OP_W(r, 1) * (u1[idx + s] - uc); // inner grad
48
+ real_t Dm = OP_W(r, 1) * (uc - u1[idx - s]); // inner grad
49
+ #pragma unroll
50
+ for (int k = 2; k <= STENCIL_RADIUS; ++k)
51
+ if (k <= r) {
52
+ Dp += OP_W(r, k) * (u1[idx + k * s] - u1[idx - (k - 1) * s]);
53
+ Dm += OP_W(r, k) * (u1[idx + (k - 1) * s] - u1[idx - k * s]);
54
+ }
55
+ return factor * (dgp * Dp * (l1[idx + s] - lc) +
56
+ dgm * Dm * (lc - l1[idx - s])); // both cells of the node
57
+ }
58
+
59
+ // -------------------------------------- kernels
60
+ extern "C" {
61
+
62
+ // ------------------------------------------------------------------------------------
63
+ __global__ void
64
+ gradient_kernel(real_t *__restrict__ g_mass, real_t *__restrict__ g_stiff,
65
+ const real_t *__restrict__ u0, const real_t *__restrict__ u1,
66
+ const real_t *__restrict__ u2, const real_t *__restrict__ l1,
67
+ const real_t *__restrict__ stiff, const real_t inv_dt2,
68
+ const real_t F0, const int N0
69
+ #if NDIM >= 2
70
+ ,
71
+ const real_t F1, const int N1, const int s0
72
+ #endif
73
+ #if NDIM >= 3
74
+ ,
75
+ const real_t F2, const int N2, const int s1
76
+ #endif
77
+ ) {
78
+ INTERIOR_OR_RETURN;
79
+ AXIS_RADII;
80
+
81
+ const real_t uc = u1[idx];
82
+ const real_t lc = l1[idx];
83
+
84
+ // dJ/dmass: no neighbour and no material load
85
+ g_mass[idx] -= inv_dt2 * lc * (u2[idx] - 2.f * uc + u0[idx]);
86
+
87
+ // dJ/dstiff: scalar.cu's harmonic cell mean differentiated in place
88
+ const real_t sc = stiff[idx];
89
+ #if NDIM == 1
90
+ const real_t g =
91
+ stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F0, r0);
92
+ #elif NDIM == 2
93
+ const real_t g =
94
+ stiffness_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, F0, r0) +
95
+ stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F1, r1);
96
+ #elif NDIM == 3
97
+ const real_t g =
98
+ stiffness_gradient_axis(u1, l1, stiff, idx, s0, uc, lc, sc, F0, r0) +
99
+ stiffness_gradient_axis(u1, l1, stiff, idx, s1, uc, lc, sc, F1, r1) +
100
+ stiffness_gradient_axis(u1, l1, stiff, idx, 1, uc, lc, sc, F2, r2);
101
+ #endif
102
+ g_stiff[idx] -= g;
103
+ }
104
+
105
+ // ------------------------------------------------------------------------------------
106
+ __global__ void frechet_kernel(real_t *__restrict__ acc_mass,
107
+ real_t *__restrict__ acc_stiff,
108
+ const real_t *__restrict__ u0,
109
+ const real_t *__restrict__ u1,
110
+ const real_t *__restrict__ u2, const real_t ft,
111
+ const real_t f0, const int N0
112
+ #if NDIM >= 2
113
+ ,
114
+ const real_t f1, const int N1, const int s0
115
+ #endif
116
+ #if NDIM >= 3
117
+ ,
118
+ const real_t f2, const int N2, const int s1
119
+ #endif
120
+ ) {
121
+ INTERIOR_OR_RETURN;
122
+ AXIS_OFFSETS;
123
+
124
+ const real_t dudt = u2[idx] - u0[idx];
125
+ acc_mass[idx] += ft * dudt * dudt;
126
+
127
+ const real_t g0 = u1[idx + o0] - u1[idx - o0];
128
+ real_t sum = f0 * g0 * g0;
129
+ #if NDIM >= 2
130
+ const real_t g1 = u1[idx + o1] - u1[idx - o1];
131
+ sum += f1 * g1 * g1;
132
+ #endif
133
+ #if NDIM >= 3
134
+ const real_t g2 = u1[idx + o2] - u1[idx - o2];
135
+ sum += f2 * g2 * g2;
136
+ #endif
137
+ acc_stiff[idx] += sum;
138
+ }
139
+
140
+ } // extern "C"