cuwave 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cuwave/sensitivity.py ADDED
@@ -0,0 +1,535 @@
1
+ """Adjoint sensitivities in three variants, same arguments and same return.
2
+
3
+ `sensitivity` stores the forward field (N + 2 grids) and is the exact transpose of
4
+ the discretisation. `reconstruction_sensitivity` rebuilds that field by a reverse
5
+ march instead, storing only the strip a damping layer makes irreversible, and stays
6
+ exact wherever the strip shields it. `superposition_sensitivity` reconstructs it in
7
+ three slots with no strip at all, trading exactness and a `scale` the caller has to
8
+ set for a footprint independent of N. All three share the cell weights and the
9
+ adjoint excitation below; docs/sensitivity.md carries the derivations.
10
+
11
+ `source_sensitivity` shares those arguments but differentiates with respect to
12
+ the source signal rather than the material, which needs no forward field at all.
13
+ """
14
+
15
+ import warnings
16
+ from collections.abc import Callable
17
+
18
+ import cupy as cp
19
+ import cupy.typing as cpt
20
+ import cupyx.scipy.ndimage as ndi
21
+
22
+ from .boundary import define_boundary
23
+ from .wave import (
24
+ Simulation,
25
+ Source,
26
+ compile_kernels,
27
+ define_excitation,
28
+ define_get_signal,
29
+ define_set_signal,
30
+ define_step_method,
31
+ flatten_indices,
32
+ grid_rows,
33
+ )
34
+
35
+ ADJOINT_DELAY = 1 # lines the adjoint up with the reconstructed forward triplet
36
+
37
+ # cancellation past which B(w, w) - B(u, u) has eaten too much of the mantissa to trust
38
+ CANCELLATION_LIMIT = {"float32": 1e5, "float64": 1e12}
39
+
40
+
41
+ # -------------------------------------- helpers --------------------------------------
42
+ def windowed_misfit(observed: cpt.NDArray, window: cpt.NDArray) -> Callable:
43
+ """Objective factory: `l2_misfit` restricted to a time window.
44
+
45
+ Args:
46
+ observed: the (N, num_sensors) measured record to fit.
47
+ window: (N, 1) or (N, num_sensors) weights, zero outside the window kept.
48
+
49
+ Returns:
50
+ the objective `sensitivity` takes, its derivative carrying the window twice
51
+ so that it stays the exact derivative of the windowed cost.
52
+ """
53
+
54
+ def objective(traces):
55
+ residual = window * (traces - observed)
56
+ return 0.5 * float(cp.sum(residual**2)), window * residual
57
+
58
+ return objective
59
+
60
+
61
+ def _accumulated(accs: dict) -> float:
62
+ """Norm over every accumulator, so the diagnostic names no material field."""
63
+ return float(sum(float(cp.linalg.norm(f)) ** 2 for f in accs.values()) ** 0.5)
64
+
65
+
66
+ def l2_misfit(observed: cpt.NDArray) -> Callable:
67
+ """Objective factory: J = 1/2 sum (traces - observed)^2, and its derivative."""
68
+
69
+ def objective(traces):
70
+ residual = traces - observed
71
+ return 0.5 * float(cp.sum(residual**2)), residual
72
+
73
+ return objective
74
+
75
+
76
+ def require_interior(
77
+ sim: Simulation, position: cpt.NDArray[cp.int32], what: str
78
+ ) -> None:
79
+ """Raise if any node of `position` sits on a ghost node, naming it `what`."""
80
+ # a ghost node carries no equation, so a sensor there corrupts the whole gradient
81
+ rows = grid_rows(sim, position)
82
+ lo = int(cp.min(rows))
83
+ if lo < 1:
84
+ raise ValueError(f"{what} on a ghost node: index {lo} < 1")
85
+ for d in range(sim.ndim):
86
+ hi = int(cp.max(rows[d]))
87
+ if hi > sim.Nx[d] - 2:
88
+ raise ValueError(
89
+ f"{what} on a ghost node: axis {d} index {hi} exceeds "
90
+ f"Nx[{d}] - 2 = {sim.Nx[d] - 2}"
91
+ )
92
+
93
+
94
+ def require_lossless(sim: Simulation) -> None:
95
+ """Raise if `sim.damping` is set: reconstructing by time reversal needs losslessness."""
96
+ if sim.damping is not None:
97
+ raise NotImplementedError(
98
+ "superposition_sensitivity rebuilds the forward field by running it "
99
+ "backwards, which only a lossless operator allows; use sensitivity"
100
+ )
101
+
102
+
103
+ def reconstruction_nodes(
104
+ sim: Simulation,
105
+ ) -> tuple[cpt.NDArray[cp.int32], cpt.NDArray[cp.bool_]]:
106
+ """Nodes a reverse march has to replay, and where its gradient stays exact.
107
+
108
+ Which nodes those are is decided by `sim.damping`, so the caller states neither.
109
+
110
+ Returns:
111
+ (strip, valid): the (ndim, num) grid indices to record and replay each step,
112
+ the damped ones a lossless node reads across the interface, and the mask of
113
+ every lossless interior node, which is where the rebuilt triplet is exact.
114
+ """
115
+ # the reverse step of a node reaches this far, so a strip that thin feeds it
116
+ radius = sim.reach
117
+ interior = cp.zeros(sim.Nx_padded, dtype=cp.bool_)
118
+ interior[tuple(slice(1, n - 1) for n in sim.Nx)] = True
119
+ if sim.damping is None:
120
+ return cp.zeros((sim.node_rows, 0), dtype=cp.int32), interior
121
+ lossless = interior & ~(sim.damping > 0)
122
+ reach = ndi.binary_dilation(lossless, iterations=radius, brute_force=True)
123
+ strip = cp.stack(cp.nonzero(interior & ~lossless & reach)).astype(cp.int32)
124
+ return with_components(sim, strip), lossless
125
+
126
+
127
+ def with_components(
128
+ sim: Simulation, nodes: cpt.NDArray[cp.int32]
129
+ ) -> cpt.NDArray[cp.int32]:
130
+ """Repeat spatial `nodes` once per field component, the component row prepended."""
131
+ if sim.ncomp == 1:
132
+ return nodes
133
+ tiled = cp.tile(nodes, (1, sim.ncomp))
134
+ row = cp.repeat(cp.arange(sim.ncomp, dtype=cp.int32), nodes.shape[1])
135
+ return cp.ascontiguousarray(cp.concatenate((row[None, :], tiled), axis=0))
136
+
137
+
138
+ def require_reconstructable(sim: Simulation, valid: cpt.NDArray[cp.bool_]) -> None:
139
+ """Raise if `sim.damping` leaves no lossless interior for a reverse march to rebuild."""
140
+ if not bool(cp.any(valid)):
141
+ raise ValueError(
142
+ "damping covers every interior node, so there is nothing to reconstruct; "
143
+ "damp only the faces with boundary.sponge, or use sensitivity"
144
+ )
145
+
146
+
147
+ def adjoint_signal(
148
+ sim: Simulation,
149
+ dphi: cpt.NDArray,
150
+ sensors: cpt.NDArray[cp.int32],
151
+ scale: float = 1.0,
152
+ delay: int = 0,
153
+ ) -> cpt.NDArray:
154
+ """Adjoint excitation: `dphi` reversed, over W and source_factor, times `scale`.
155
+
156
+ Args:
157
+ dphi: (N, num_sensors) derivative of the cost with respect to the traces.
158
+ sensors: the nodes it is injected on, one column each.
159
+ scale: the superposition k; 1 for the exact adjoint.
160
+ delay: entries dropped from the front and zero-padded at the back, which
161
+ starts the adjoint recursion that many steps earlier in its own sequence.
162
+ """
163
+ # define_excitation supplies the dt^2 source_factor minv the recursion wants
164
+ signal = cp.asarray(dphi, dtype=sim.dtype)[::-1] / sim.adjoint_weights(sensors)
165
+ if delay:
166
+ signal = cp.concatenate((signal[delay:], cp.zeros_like(signal[:delay])))
167
+ return cp.ascontiguousarray(sim.dtype(scale) * signal, dtype=sim.dtype)
168
+
169
+
170
+ # ---------------------------------- adjoint solvers ----------------------------------
171
+ def sensitivity(
172
+ sim: Simulation,
173
+ source: Source,
174
+ indicator: cpt.NDArray,
175
+ sensors: cpt.NDArray[cp.int32],
176
+ objective: Callable,
177
+ ) -> tuple[float, dict[str, cpt.NDArray], cpt.NDArray, dict]:
178
+ """Cost and its gradients d(cost)/d(mass, stiff) over the padded grid.
179
+
180
+ Args:
181
+ sim: the simulation the forward and adjoint passes both step.
182
+ source: the shot to differentiate, its position interior nodes only.
183
+ indicator: the design field the materials are built from.
184
+ sensors: (ndim, num_sensors) interior grid indices.
185
+ objective: takes the (N, num_sensors) record, returns (cost, dcost/dtraces).
186
+ The derivative drives the adjoint field, so any differentiable cost
187
+ works; reparametrize by chain rule at the call site with
188
+ `sim.parametrization_jacobian()`.
189
+
190
+ A `sim.damping` field is stepped by the same kernel in both passes, since marching
191
+ the adjoint backwards is what transposes the damped recursion.
192
+
193
+ Returns:
194
+ (cost, {"mass": ..., "stiff": ...}, traces, info), the gradients fields over
195
+ the padded grid, `traces` the (N, num_sensors) record the cost was read from,
196
+ and `info` empty; this variant has nothing to report.
197
+ """
198
+ require_interior(sim, sensors, "sensor")
199
+ require_interior(sim, source.position, "source")
200
+
201
+ mat = sim.build_materials(indicator)
202
+ kernels = compile_kernels(sim)
203
+ sens_kernels = compile_kernels(sim, sim.sensitivity_path)
204
+
205
+ fd_step = define_step_method(sim, kernels, mat)
206
+ bc_step = define_boundary(sim, kernels)
207
+ excitation_step = define_excitation(sim, source.position, kernels, mat)
208
+ grads = sim.gradient_fields(mat)
209
+ gradient_step = sim.define_gradient(sens_kernels, mat, grads)
210
+
211
+ # ------------------------------------ forward pass -----------------------------------
212
+ # stepped straight into the history, so the leading zeros are u^-2 / u^-1
213
+ V = cp.zeros((sim.N + 2, *sim.field_shape), dtype=sim.dtype)
214
+ # the slot views made once: V[t] is a host slice costing more than its own kernel
215
+ slot = [V[t] for t in range(sim.N + 2)]
216
+
217
+ for t in range(sim.N):
218
+ u = fd_step(slot[t], slot[t + 1], slot[t + 2])
219
+ u = excitation_step(u, source.signal, t)
220
+ u = bc_step(u)
221
+
222
+ # gathered off the history rather than probed per step, saving one launch a step
223
+ um = V[2:].reshape(sim.N, -1)[:, flatten_indices(sim, sensors)]
224
+
225
+ cost, dphi = objective(um)
226
+
227
+ # --------------------------------- adjoint excitation --------------------------------
228
+ signal = adjoint_signal(sim, dphi, sensors)
229
+ adjoint_excitation = define_excitation(sim, sensors, kernels, mat)
230
+
231
+ # ----------------------------------- backward pass -----------------------------------
232
+ P = cp.zeros((2, *sim.field_shape), dtype=sim.dtype)
233
+ p0, p1 = P[0], P[1]
234
+ for m in range(sim.N):
235
+ n = sim.N - 1 - m
236
+ p0 = fd_step(p0, p1, p0)
237
+ p0 = adjoint_excitation(p0, signal, m)
238
+ p0 = bc_step(p0)
239
+ p1, p0 = p0, p1 # p1 now holds lambda^n
240
+ # inertia pairs lambda^n with the whole triplet, stiffness with its middle slot
241
+ gradient_step(slot[n], slot[n + 1], slot[n + 2], p1)
242
+
243
+ return cost, sim.finalize_gradients(grads, sens_kernels), um, {}
244
+
245
+
246
+ def reconstruction_sensitivity(
247
+ sim: Simulation,
248
+ source: Source,
249
+ indicator: cpt.NDArray,
250
+ sensors: cpt.NDArray[cp.int32],
251
+ objective: Callable,
252
+ ) -> tuple[float, dict[str, cpt.NDArray], cpt.NDArray, dict]:
253
+ """Cost and its gradients as `sensitivity`, storing a boundary strip and not the field.
254
+
255
+ The lossless recursion is symmetric in its two outer slots, so the same step kernel
256
+ run with them swapped marches the forward field backwards. Damping breaks that, so
257
+ the nodes a damped one reaches are recorded each step and replayed on the way back:
258
+ the reverse march then never reads an irreversible node, and the gradient stays the
259
+ exact transpose wherever the strip shields it. This is the variant to reach for
260
+ once the history no longer fits and the domain is open, since
261
+ `superposition_sensitivity` refuses a damping field outright.
262
+
263
+ Args:
264
+ sim: the simulation both passes step, damped or lossless. Damping covering
265
+ every interior node is rejected: nothing is left to rebuild from.
266
+ source: the shot to differentiate, its position interior nodes only.
267
+ indicator: the design field the materials are built from.
268
+ sensors: (ndim, num_sensors) interior grid indices.
269
+ objective: takes the (N, num_sensors) record, returns (cost, dcost/dtraces).
270
+
271
+ Returns:
272
+ (cost, {"mass": ..., "stiff": ...}, traces, info) as `sensitivity`, the
273
+ gradients zeroed outside the region the strip shields, and `info` carrying the
274
+ `strip` node count and the `drift` the reverse march accumulated.
275
+ """
276
+ require_interior(sim, sensors, "sensor")
277
+ require_interior(sim, source.position, "source")
278
+ strip, valid = reconstruction_nodes(sim)
279
+ require_reconstructable(sim, valid)
280
+ num_strip = strip.shape[1]
281
+
282
+ mat = sim.build_materials(indicator)
283
+ kernels = compile_kernels(sim)
284
+ sens_kernels = compile_kernels(sim, sim.sensitivity_path)
285
+
286
+ fd_step = define_step_method(sim, kernels, mat)
287
+ bc_step = define_boundary(sim, kernels)
288
+ excitation_step = define_excitation(sim, source.position, kernels, mat)
289
+ get_signal = define_get_signal(sim, sensors, kernels)
290
+ grads = sim.gradient_fields(mat)
291
+ gradient_step = sim.define_gradient(sens_kernels, mat, grads)
292
+ if num_strip:
293
+ record_strip = define_get_signal(sim, strip, kernels)
294
+ replay_strip = define_set_signal(sim, strip, kernels)
295
+
296
+ U = cp.zeros((3, *sim.field_shape), dtype=sim.dtype)
297
+ u0, u1, u2 = U[0], U[1], U[2]
298
+ um = cp.zeros((sim.N, sensors.shape[1]), dtype=sim.dtype)
299
+ # two leading zero rows, so row t + 2 is u^t and the initial states need no branch
300
+ strip_store = cp.zeros((sim.N + 2, num_strip), dtype=sim.dtype)
301
+
302
+ # ------------------------------------ forward pass -----------------------------------
303
+ for t in range(sim.N):
304
+ u2 = fd_step(u0, u1, u2)
305
+ u2 = excitation_step(u2, source.signal, t)
306
+ u2 = bc_step(u2)
307
+ get_signal(u2, um, t)
308
+ if num_strip:
309
+ record_strip(u2, strip_store, t + 2)
310
+ u0, u1, u2 = u1, u2, u0
311
+
312
+ cost, dphi = objective(um)
313
+
314
+ # --------------------------------- adjoint excitation --------------------------------
315
+ signal = adjoint_signal(sim, dphi, sensors)
316
+ adjoint_excitation = define_excitation(sim, sensors, kernels, mat)
317
+
318
+ # ----------------------------------- backward pass -----------------------------------
319
+ P = cp.zeros((2, *sim.field_shape), dtype=sim.dtype)
320
+ p0, p1 = P[0], P[1]
321
+ # the forward rotation left the last three states live, which is the whole seed
322
+ a, b, c = u2, u0, u1
323
+ seed = float(cp.linalg.norm(c * valid))
324
+
325
+ for m in range(sim.N):
326
+ n = sim.N - 1 - m
327
+ p0 = fd_step(p0, p1, p0)
328
+ p0 = adjoint_excitation(p0, signal, m)
329
+ p0 = bc_step(p0)
330
+ p1, p0 = p0, p1 # p1 now holds lambda^n
331
+ # inertia pairs lambda^n with the whole triplet, stiffness with its middle slot
332
+ gradient_step(a, b, c, p1)
333
+ if n < 1:
334
+ break
335
+ # read backwards, so the source rides two steps ahead of the state it rebuilds
336
+ c = fd_step(b, a, c)
337
+ c = excitation_step(c, source.signal, n - 1)
338
+ if num_strip:
339
+ replay_strip(c, strip_store, n - 1)
340
+ c = bc_step(c)
341
+ a, b, c = c, a, b
342
+
343
+ grads = sim.finalize_gradients(grads, sens_kernels)
344
+ # assigned, not multiplied, so a NaN outside cannot survive as 0 * NaN
345
+ outside = ~valid
346
+ for field in grads.values():
347
+ field[outside] = 0.0
348
+ # the march ends on the initial state, which is zero, so what is left is round-off
349
+ drift = float(cp.linalg.norm(a * valid)) / seed if seed > 0.0 else float("inf")
350
+ info = {"strip": num_strip, "drift": drift}
351
+ return cost, grads, um, info
352
+
353
+
354
+ def superposition_sensitivity(
355
+ sim: Simulation,
356
+ source: Source,
357
+ indicator: cpt.NDArray,
358
+ sensors: cpt.NDArray[cp.int32],
359
+ objective: Callable,
360
+ scale: float = 1.0,
361
+ ) -> tuple[float, dict[str, cpt.NDArray], cpt.NDArray, dict]:
362
+ """Cost and its gradients as `sensitivity`, in three field slots instead of N + 2.
363
+
364
+ The gradient densities are bilinear in the forward and adjoint fields, so the
365
+ cross term can be read off the diagonal of B(u + k lambda) alone, and the closed
366
+ lossless domain is time-reversible, so u itself never has to be stored. The prices
367
+ are a k^2 bias traded against round-off, and a symmetric Frechet form that is
368
+ consistent with `sensitivity` rather than equal to it. docs/sensitivity.md has
369
+ both, with the measurements.
370
+
371
+ Args:
372
+ sim: the simulation both passes step, which must be lossless: a `damping`
373
+ field is rejected, since the time reversal needs one. `sensitivity`
374
+ takes one.
375
+ source: the shot to differentiate, its position interior nodes only.
376
+ indicator: the design field the materials are built from.
377
+ sensors: (ndim, num_sensors) interior grid indices.
378
+ objective: takes the (N, num_sensors) record, returns (cost, dcost/dtraces).
379
+ scale: the superposition k, trading the k**2 bias against round-off. Set it
380
+ from `info["cancellation"]`, aiming near 1e4 in float32 or 1e6 in float64.
381
+
382
+ Returns:
383
+ (cost, {"mass": ..., "stiff": ...}, traces, info) as `sensitivity`, with
384
+ `info` carrying `scale` and the `cancellation` the subtraction cost.
385
+ """
386
+ require_lossless(sim)
387
+ require_interior(sim, sensors, "sensor")
388
+ require_interior(sim, source.position, "source")
389
+
390
+ mat = sim.build_materials(indicator)
391
+ kernels = compile_kernels(sim)
392
+ sens_kernels = compile_kernels(sim, sim.sensitivity_path)
393
+
394
+ fd_step = define_step_method(sim, kernels, mat)
395
+ bc_step = define_boundary(sim, kernels)
396
+ excitation_step = define_excitation(sim, source.position, kernels, mat)
397
+ get_signal = define_get_signal(sim, sensors, kernels)
398
+ accs = sim.gradient_fields(mat)
399
+ subtract_step = sim.define_frechet(sens_kernels, accs, -1.0)
400
+ add_step = sim.define_frechet(sens_kernels, accs, 1.0)
401
+
402
+ U = cp.zeros((3, *sim.field_shape), dtype=sim.dtype)
403
+ u0, u1, u2 = U[0], U[1], U[2]
404
+ um = cp.zeros((sim.N, sensors.shape[1]), dtype=sim.dtype)
405
+
406
+ # ------------------------------------ forward pass -----------------------------------
407
+ # records the traces and subtracts the forward diagonal B(u, u)
408
+ for t in range(sim.N):
409
+ u2 = fd_step(u0, u1, u2)
410
+ u2 = excitation_step(u2, source.signal, t)
411
+ u2 = bc_step(u2)
412
+ get_signal(u2, um, t)
413
+ subtract_step(u0, u1, u2)
414
+ u0, u1, u2 = u1, u2, u0
415
+
416
+ cost, dphi = objective(um)
417
+
418
+ # --------------------------------- adjoint excitation --------------------------------
419
+ # the forward diagonal before the backward pass cancels it, for `cancellation`
420
+ before = _accumulated(accs)
421
+ # concatenated into one launch, sound because excitation_kernel uses atomicAdd
422
+ backward_position = cp.concatenate((sensors, source.position), axis=1)
423
+ backward_signal = cp.ascontiguousarray(
424
+ cp.concatenate(
425
+ (
426
+ adjoint_signal(sim, dphi, sensors, scale, ADJOINT_DELAY),
427
+ source.signal[: sim.N][::-1],
428
+ ),
429
+ axis=1,
430
+ )
431
+ )
432
+ backward_excitation = define_excitation(sim, backward_position, kernels, mat)
433
+
434
+ # ----------------------------------- backward pass -----------------------------------
435
+ # u0 / u1 hold u^(N-1) / u^(N-2), so the one array carries u + k lambda
436
+ u0, u1 = u1, u0
437
+ for t in range(sim.N):
438
+ u2 = fd_step(u0, u1, u2)
439
+ u2 = backward_excitation(u2, backward_signal, t)
440
+ u2 = bc_step(u2)
441
+ add_step(u0, u1, u2)
442
+ u0, u1, u2 = u1, u2, u0
443
+
444
+ after = _accumulated(accs)
445
+ cancellation = before / after if after > 0.0 else float("inf")
446
+ limit = CANCELLATION_LIMIT[sim.precision]
447
+ if cancellation > limit:
448
+ warnings.warn(
449
+ f"superposition scale={scale:g} leaves a cancellation of "
450
+ f"{cancellation:.1e} in {sim.precision}, past the usable {limit:.0e}: the "
451
+ f"gradient is largely round-off. Raise scale by about "
452
+ f"{cancellation / limit:.0e}.",
453
+ RuntimeWarning,
454
+ stacklevel=2,
455
+ )
456
+
457
+ # B(u, lambda) = [B(w, w) - B(u, u)] / 2k, scaled in place to spare a field
458
+ norm = sim.dtype(1.0 / (2.0 * scale))
459
+ accs = sim.finalize_gradients(accs, sens_kernels)
460
+ for field in accs.values():
461
+ field *= norm
462
+ info = {"scale": scale, "cancellation": cancellation}
463
+ return cost, accs, um, info
464
+
465
+
466
+ def source_sensitivity(
467
+ sim: Simulation,
468
+ source: Source,
469
+ indicator: cpt.NDArray,
470
+ sensors: cpt.NDArray[cp.int32],
471
+ objective: Callable,
472
+ ) -> tuple[float, cpt.NDArray, cpt.NDArray, dict]:
473
+ """Cost and its gradient d(cost)/d(source.signal), the adjoint field at the source.
474
+
475
+ Args:
476
+ sim: the simulation the forward and adjoint passes both step.
477
+ source: the shot to differentiate, its position interior nodes only.
478
+ indicator: the design field the materials are built from, held fixed here.
479
+ sensors: (ndim, num_sensors) interior grid indices.
480
+ objective: takes the (N, num_sensors) record, returns (cost, dcost/dtraces).
481
+
482
+ The cost is linear in the signal, so the gradient pairs no forward field against
483
+ the adjoint one and this variant stores neither: four grids, whatever N is.
484
+
485
+ Returns:
486
+ (cost, gradient, traces, info), the gradient the (N, num_sources) derivative
487
+ with respect to `source.signal`, `traces` the (N, num_sensors) record the cost
488
+ was read from, and `info` empty; this variant has nothing to report.
489
+ """
490
+ require_interior(sim, sensors, "sensor")
491
+ require_interior(sim, source.position, "source")
492
+
493
+ mat = sim.build_materials(indicator)
494
+ kernels = compile_kernels(sim)
495
+
496
+ fd_step = define_step_method(sim, kernels, mat)
497
+ bc_step = define_boundary(sim, kernels)
498
+ excitation_step = define_excitation(sim, source.position, kernels, mat)
499
+ get_signal = define_get_signal(sim, sensors, kernels)
500
+ probe = define_get_signal(sim, source.position, kernels)
501
+
502
+ # ------------------------------------ forward pass -----------------------------------
503
+ U = cp.zeros((2, *sim.field_shape), dtype=sim.dtype)
504
+ u0, u1 = U[0], U[1]
505
+ um = cp.zeros((sim.N, sensors.shape[1]), dtype=sim.dtype)
506
+
507
+ for t in range(sim.N):
508
+ u0 = fd_step(u0, u1, u0)
509
+ u0 = excitation_step(u0, source.signal, t)
510
+ u0 = bc_step(u0)
511
+ u1, u0 = u0, u1
512
+ get_signal(u1, um, t)
513
+
514
+ cost, dphi = objective(um)
515
+
516
+ # --------------------------------- adjoint excitation --------------------------------
517
+ signal = adjoint_signal(sim, dphi, sensors)
518
+ adjoint_excitation = define_excitation(sim, sensors, kernels, mat)
519
+
520
+ # ----------------------------------- backward pass -----------------------------------
521
+ P = cp.zeros((2, *sim.field_shape), dtype=sim.dtype)
522
+ p0, p1 = P[0], P[1]
523
+ lam = cp.zeros((sim.N, source.position.shape[1]), dtype=sim.dtype)
524
+
525
+ for m in range(sim.N):
526
+ n = sim.N - 1 - m
527
+ p0 = fd_step(p0, p1, p0)
528
+ p0 = adjoint_excitation(p0, signal, m)
529
+ p0 = bc_step(p0)
530
+ p1, p0 = p0, p1 # p1 now holds lambda^n
531
+ probe(p1, lam, n) # row n not row m, so the record runs forward in time
532
+
533
+ # the transpose of adjoint_signal: over the same weights, and not reversed
534
+ gradient = lam * sim.adjoint_weights(source.position)
535
+ return cost, gradient, um, {}
cuwave/signals.py ADDED
@@ -0,0 +1,71 @@
1
+ import numpy as np
2
+ import numpy.typing as npt
3
+
4
+
5
+ def sineburst(
6
+ t: npt.NDArray[np.float64], amplitude: float, frequency: float, cycles: int
7
+ ) -> npt.NDArray[np.float64]:
8
+ """Hann-windowed sine burst of `cycles` periods at `frequency`, zero outside its support."""
9
+ mask = (t > 0) & (t <= cycles / frequency)
10
+ return (
11
+ amplitude
12
+ * mask
13
+ * np.sin(2 * np.pi * frequency * t)
14
+ * np.sin(np.pi * frequency * t / cycles) ** 2
15
+ )
16
+
17
+
18
+ def ricker(
19
+ t: npt.NDArray[np.float64],
20
+ amplitude: float,
21
+ frequency: float,
22
+ delay: float | None = None,
23
+ ) -> npt.NDArray[np.float64]:
24
+ """Ricker wavelet at `frequency`, centered at `delay` (default one period)."""
25
+ if delay is None:
26
+ delay = 1.0 / frequency
27
+ arg = (np.pi * frequency * (t - delay)) ** 2
28
+ return amplitude * (1.0 - 2.0 * arg) * np.exp(-arg)
29
+
30
+
31
+ def gabor(
32
+ t: npt.NDArray[np.float64],
33
+ amplitude: float,
34
+ frequency: float,
35
+ cycles: float,
36
+ delay: float | None = None,
37
+ ) -> npt.NDArray[np.float64]:
38
+ """Gaussian-modulated sine at `frequency`, its 1/e envelope `cycles` periods wide.
39
+
40
+ Odd about `delay`, three envelope widths in by default, so the mean is zero.
41
+ """
42
+ tau = 0.5 * cycles / frequency
43
+ if delay is None:
44
+ delay = 3.0 * tau
45
+ lag = t - delay
46
+ return amplitude * np.exp(-((lag / tau) ** 2)) * np.sin(2 * np.pi * frequency * lag)
47
+
48
+
49
+ def chirp(
50
+ t: npt.NDArray[np.float64],
51
+ amplitude: float,
52
+ band: tuple[float, float],
53
+ duration: float,
54
+ taper: float = 0.1,
55
+ ) -> npt.NDArray[np.float64]:
56
+ """Linear sweep across `band` over `duration`, zero outside it.
57
+
58
+ `taper` is the share of `duration` spent ramping, split between the two ends. A
59
+ Tukey window and not `sineburst`'s Hann, whose rise would attenuate the band edges
60
+ the sweep is there to cover flatly.
61
+ """
62
+ mask = (t > 0) & (t <= duration)
63
+ edge = 0.5 * taper * duration
64
+ window = np.sin(0.5 * np.pi * np.clip(np.minimum(t, duration - t) / edge, 0.0, 1.0))
65
+ rate = (band[1] - band[0]) / duration
66
+ return (
67
+ amplitude
68
+ * mask
69
+ * np.sin(2 * np.pi * t * (band[0] + 0.5 * rate * t))
70
+ * window**2
71
+ )
cuwave/stencils.py ADDED
@@ -0,0 +1,48 @@
1
+ import numpy as np
2
+ import numpy.typing as npt
3
+
4
+
5
+ def weights(R: int) -> npt.NDArray[np.float64]:
6
+ """Central second-derivative weights of order 2R, `w[-R..R]`, via a Vandermonde solve."""
7
+ x = np.arange(-R, R + 1)
8
+ A = x[None, :] ** np.arange(2 * R + 1)[:, None]
9
+ rhs = np.zeros(2 * R + 1)
10
+ rhs[2] = 2.0
11
+ return np.linalg.solve(A, rhs)
12
+
13
+
14
+ def cell_coefficients(R: int) -> npt.NDArray[np.float64]:
15
+ """Cumulative tail sums of `weights(R)`: the coefficients of the cell flux at radius `R`."""
16
+ w = weights(R)
17
+ return np.array([w[R + 1 + k :].sum() for k in range(R)])
18
+
19
+
20
+ def staggered_weights(R: int) -> npt.NDArray[np.float64]:
21
+ """First-derivative weights of order 2R on the half offsets (2k - 1) / 2, `w[1..R]`."""
22
+ # antisymmetric taps, so only the odd moments constrain the R coefficients
23
+ x = np.arange(1, R + 1) - 0.5
24
+ A = x[None, :] ** (2 * np.arange(R)[:, None] + 1)
25
+ rhs = np.zeros(R)
26
+ rhs[0] = 0.5
27
+ return np.linalg.solve(A, rhs)
28
+
29
+
30
+ def preamble(space_order: int) -> str:
31
+ """CUDA preamble: `STENCIL_RADIUS` plus the coefficient tables, one row per radius."""
32
+ R = space_order // 2
33
+ flux = np.zeros((R, R))
34
+ stag = np.zeros((R, R))
35
+ for r in range(1, R + 1):
36
+ flux[r - 1, :r] = cell_coefficients(r)
37
+ stag[r - 1, :r] = staggered_weights(r)
38
+
39
+ def rows(table):
40
+ return ", ".join(
41
+ "{" + ", ".join(repr(float(v)) for v in row) + "}" for row in table
42
+ )
43
+
44
+ return (
45
+ f"#define STENCIL_RADIUS {R}\n"
46
+ f"#define OP_COEFFS {{ {rows(flux)} }}\n"
47
+ f"#define STAG_COEFFS {{ {rows(stag)} }}\n"
48
+ )