specmod 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. specmod/__init__.py +17 -0
  2. specmod/_vendor/__init__.py +21 -0
  3. specmod/_vendor/qiinv.py +243 -0
  4. specmod/acquire.py +358 -0
  5. specmod/api.py +480 -0
  6. specmod/cli.py +139 -0
  7. specmod/config/__init__.py +44 -0
  8. specmod/config/layers.py +168 -0
  9. specmod/config/provenance.py +77 -0
  10. specmod/config/sections.py +385 -0
  11. specmod/config/serialize.py +58 -0
  12. specmod/core/__init__.py +41 -0
  13. specmod/core/bandwidth.py +187 -0
  14. specmod/core/collection.py +549 -0
  15. specmod/core/noise.py +478 -0
  16. specmod/core/scalogram.py +234 -0
  17. specmod/core/spectrum.py +326 -0
  18. specmod/core/units.py +116 -0
  19. specmod/datasets.py +316 -0
  20. specmod/distance.py +190 -0
  21. specmod/exceptions.py +58 -0
  22. specmod/fitting/__init__.py +58 -0
  23. specmod/fitting/base.py +50 -0
  24. specmod/fitting/event.py +284 -0
  25. specmod/fitting/guess.py +170 -0
  26. specmod/fitting/spectrum.py +330 -0
  27. specmod/io.py +241 -0
  28. specmod/magnitude.py +312 -0
  29. specmod/picks/__init__.py +182 -0
  30. specmod/picks/base.py +250 -0
  31. specmod/picks/delimited.py +224 -0
  32. specmod/picks/events.py +157 -0
  33. specmod/picks/resolution.py +149 -0
  34. specmod/picks/snuffler.py +92 -0
  35. specmod/pipeline.py +280 -0
  36. specmod/plotting.py +203 -0
  37. specmod/preprocess.py +554 -0
  38. specmod/smoothing/__init__.py +50 -0
  39. specmod/smoothing/base.py +56 -0
  40. specmod/smoothing/konno_ohmachi.py +83 -0
  41. specmod/smoothing/log_bins.py +171 -0
  42. specmod/sources/__init__.py +65 -0
  43. specmod/sources/attenuation.py +110 -0
  44. specmod/sources/composite.py +135 -0
  45. specmod/sources/motion.py +40 -0
  46. specmod/sources/source.py +147 -0
  47. specmod/spreading.py +209 -0
  48. specmod/staged.py +523 -0
  49. specmod/tables.py +110 -0
  50. specmod/transforms/__init__.py +50 -0
  51. specmod/transforms/base.py +242 -0
  52. specmod/transforms/cwt.py +219 -0
  53. specmod/transforms/fft.py +157 -0
  54. specmod/transforms/multitaper.py +357 -0
  55. specmod/transforms/prieto.py +272 -0
  56. specmod/transforms/quadratic.py +221 -0
  57. specmod/utils.py +305 -0
  58. specmod-0.2.0.dist-info/METADATA +294 -0
  59. specmod-0.2.0.dist-info/RECORD +62 -0
  60. specmod-0.2.0.dist-info/WHEEL +4 -0
  61. specmod-0.2.0.dist-info/entry_points.txt +2 -0
  62. specmod-0.2.0.dist-info/licenses/LICENSE +21 -0
specmod/core/noise.py ADDED
@@ -0,0 +1,478 @@
1
+ """Models for the noise level a signal is judged against.
2
+
3
+ A recorded noise window understates the noise underneath a strong signal: it
4
+ is a sample of the same process, but taken where the signal is not. Comparing
5
+ a signal against it unmodified selects a band wider than the data supports,
6
+ particularly at the low end where ``Omega`` is read.
7
+
8
+ How to correct for that is a modelling choice, not a fact, so this module
9
+ holds a **set** of methods rather than one. They share a signature — given the
10
+ frequencies, the noise and the signal, return a multiplicative factor to apply
11
+ to the noise — which is what lets the band search stay indifferent to which
12
+ was used. :data:`NOISE_MODELS` maps names to implementations and
13
+ :func:`get_noise_model` resolves one, mirroring how
14
+ :mod:`specmod.transforms` handles estimators.
15
+
16
+ The factor is returned rather than the corrected array because it has to be
17
+ applied to two things: the binned noise and the unbinned noise, the latter by
18
+ interpolation onto the finer axis. Returning the factor keeps those two from
19
+ drifting apart.
20
+
21
+ Currently implemented
22
+ ---------------------
23
+ ``boost``
24
+ The default, described below. Lifts the low and high tails independently
25
+ until each touches the signal.
26
+ ``rotate``
27
+ The legacy ``ROT_METHOD = 1``, described below. Rotates the log-log
28
+ spectrum about the origin instead of scaling it.
29
+ ``none``
30
+ The recorded noise, uncorrected. The control the other two are read
31
+ against.
32
+
33
+ Anything added here should say what it assumes about the noise, because that
34
+ assumption is the whole content of the method — and it propagates into every
35
+ bandwidth, and so into every ``Omega``.
36
+
37
+
38
+ The boost method
39
+ ----------------
40
+
41
+ Noise rotation exists because a raw noise spectrum understates the noise at the
42
+ edges of the band. The recorded noise window is a sample of a process, and at
43
+ frequencies where the signal is strong the *same* process is present underneath
44
+ it — so a band chosen against the unmodified noise level runs wider than the
45
+ data supports, particularly at the low end where ``Omega`` is read.
46
+
47
+ ``ROT_METHOD = 2`` in the legacy module, the shipped default, and the one the
48
+ Magna study uses. It raises the low and high tails independently until each
49
+ touches the signal, then keeps the larger of the two at every frequency.
50
+
51
+ **It assumes** the noise beneath the signal follows the same spectral shape as
52
+ the recorded noise, scaled by a power of a frequency ramp — so it can be
53
+ corrected by a smooth monotone lift anchored at the point where noise first
54
+ meets signal. That is an assumption about the noise process, and a different
55
+ one would give a different band.
56
+
57
+
58
+ The rotate method
59
+ -----------------
60
+
61
+ ``ROT_METHOD = 1``, described in the legacy source as "actual rotation, quite
62
+ aggressive". In log-log space it rotates the noise spectrum about the origin,
63
+
64
+ .. code-block:: text
65
+
66
+ Y'(theta) = X sin(theta) + Y cos(theta) + Y[0] * theta
67
+
68
+ with ``X = log10(f)`` and ``Y = log10(noise)``, the trailing term holding the
69
+ low-frequency end in place so the rotation pivots there rather than about the
70
+ axis origin. As with ``boost``, one angle is found for the low half and one for
71
+ the high, and the larger of the two results is kept at every frequency.
72
+
73
+ **It assumes** the discrepancy between recorded and underlying noise is a *tilt*
74
+ — the recorded window has the right level somewhere in the middle and the wrong
75
+ slope — rather than the level offset ``boost`` assumes. That is a genuinely
76
+ different claim about the noise process, which is why both are kept: comparing
77
+ the two bands is the only way to see how much of a result is the method.
78
+
79
+ A single rotation is not monotone — tilting the spectrum raises one end and
80
+ lowers the other — so it is not obvious that the spliced result can only raise
81
+ the noise, the way ``boost`` provably can. There is no proof of it here, but it
82
+ was not observed to fail: over 4000 randomised spectra spanning three decades
83
+ of frequency, spectral slopes from ``f**-3`` to ``f**1``, and noise between
84
+ 0.1% and 99% of signal, the smallest factor returned was ``1 - 2e-15``. The
85
+ registry-wide test asserts the property; if some real spectrum ever violates
86
+ it, that test is where it will show up, and the honest fix is to relax the
87
+ claim rather than to clamp the method.
88
+
89
+ Two deliberate departures from the legacy implementation, neither of which can
90
+ move a published number — ``ROT_METHOD = 1`` was commented out on ``master``,
91
+ so it has never produced one:
92
+
93
+ * **The angle is solved, not stepped.** The legacy loop advanced ``theta`` by a
94
+ fixed ``inc`` and stopped at the first trial past the touching point, which
95
+ made the result a step function of its input — the same defect that made
96
+ ``boost`` machine-dependent. Here the touching angle is bracketed and then
97
+ bisected to a tolerance, so it is continuous in the input.
98
+ * **The split between halves is taken from the signal.** The legacy version
99
+ used the *noise* centroid here and the *signal* centroid in the boost path.
100
+ The signal is the defensible one, for the reason given at
101
+ :func:`centroid_frequency`, and using it in both makes the two bands
102
+ comparable — which is the point of having a registry rather than a flag.
103
+ """
104
+
105
+ from __future__ import annotations
106
+
107
+ from dataclasses import dataclass
108
+ from typing import Protocol, runtime_checkable
109
+
110
+ import numpy as np
111
+ from numpy.typing import NDArray
112
+
113
+ __all__ = [
114
+ "NOISE_MODELS",
115
+ "BoostNoise",
116
+ "NoiseModel",
117
+ "RotateNoise",
118
+ "boost_noise",
119
+ "centroid_frequency",
120
+ "get_noise_model",
121
+ "rotate_log_spectrum",
122
+ "rotate_noise",
123
+ ]
124
+
125
+
126
+ def centroid_frequency(freq: NDArray[np.float64], amp: NDArray[np.float64]) -> float:
127
+ """Amplitude-weighted mean frequency.
128
+
129
+ Used to split the spectrum into a low and a high half. It is taken from the
130
+ **signal**, not the noise — the split should follow where the energy
131
+ actually is, and the noise has no reason to share that shape.
132
+ """
133
+ return float(np.sum(freq * amp) / np.sum(amp))
134
+
135
+
136
+ def boost_noise(
137
+ freq: NDArray[np.float64],
138
+ noise_amp: NDArray[np.float64],
139
+ signal_amp: NDArray[np.float64],
140
+ *,
141
+ inc: float = 0.05,
142
+ space: tuple[float, float] = (0.001, 1.001),
143
+ max_iter: int = 1000,
144
+ ) -> NDArray[np.float64]:
145
+ """Multiplicative factor lifting ``noise_amp`` toward ``signal_amp``.
146
+
147
+ Returns the *factor*, not the boosted array, because it has to be applied
148
+ to two things — the binned noise and the unbinned noise, the latter by
149
+ interpolation onto the finer axis. Returning the factor keeps the two
150
+ applications from drifting apart.
151
+
152
+ Each frequency is mapped onto ``space``, which runs from near zero at the
153
+ low end to just above one at the high end. Dividing by ``sample ** n`` with
154
+ ``sample < 1`` therefore *raises* the low frequencies fastest, and reversing
155
+ the mapping does the same for the high ones. The exponent grows by ``inc``
156
+ until any point in the half being lifted reaches the signal.
157
+ """
158
+ if noise_amp.shape != signal_amp.shape or noise_amp.shape != freq.shape:
159
+ raise ValueError(
160
+ f"freq {freq.shape}, noise {noise_amp.shape} and signal "
161
+ f"{signal_amp.shape} must all match"
162
+ )
163
+
164
+ centroid = centroid_frequency(freq, signal_amp)
165
+ low = freq <= centroid
166
+ high = ~low
167
+
168
+ sample = np.interp(freq, [freq.min(), freq.max()], space)
169
+
170
+ lifted_low = _lift(noise_amp, signal_amp, sample, low, inc=inc, max_iter=max_iter)
171
+ lifted_high = _lift(
172
+ noise_amp, signal_amp, sample[::-1], high, inc=inc, max_iter=max_iter
173
+ )
174
+
175
+ return np.asarray(np.maximum(lifted_low, lifted_high) / noise_amp)
176
+
177
+
178
+ def _lift(
179
+ noise_amp: NDArray[np.float64],
180
+ signal_amp: NDArray[np.float64],
181
+ sample: NDArray[np.float64],
182
+ where: NDArray[np.bool_],
183
+ *,
184
+ inc: float,
185
+ max_iter: int,
186
+ ) -> NDArray[np.float64]:
187
+ """Raise ``noise_amp`` until any point in ``where`` reaches the signal.
188
+
189
+ Solved rather than searched. The legacy code stepped the exponent by
190
+ ``inc`` up to a thousand times and stopped at the first trial that touched
191
+ the signal; the exponent it lands on has a closed form. For a bin to touch,
192
+
193
+ .. code-block:: text
194
+
195
+ noise * sample ** -n >= signal
196
+ n >= ln(signal / noise) / -ln(sample)
197
+
198
+ so the first exponent at which *any* bin in ``where`` touches is the
199
+ minimum of that expression over the bins the lift actually raises — those
200
+ with ``sample < 1``, since dividing by a larger ``sample`` lowers them.
201
+
202
+ **The exponent is used exactly, not rounded up to a multiple of ``inc``.**
203
+ That is the change that makes this reproducible. The old loop stepped and
204
+ stopped at the first multiple past the touching point, so the result was a
205
+ step function of its input: two machines differing in the last bit could
206
+ land on either side of a step and move the noise by ``1.41x``. Solving for
207
+ the touching point directly makes the exponent a continuous function of the
208
+ input, so a last-bit difference in gives a last-bit difference out.
209
+
210
+ It also removes a bias. Rounding was always *upward*, so the old code
211
+ consistently overshot and overstated the lifted noise — a median ``1.18x``
212
+ and up to ``1.41x`` across 39 lifts on the 28 PNR windows. That made the
213
+ signal-to-noise pessimistic at exactly the band edges the ratio is read
214
+ from. The exact touching point is what the algorithm was always trying to
215
+ compute; the stepping was a crude search for it.
216
+
217
+ ``inc`` is therefore no longer used, and is accepted only so that callers
218
+ and stored configurations do not have to change.
219
+ """
220
+ del max_iter, inc # the search they parameterised is gone
221
+
222
+ noise, signal, scale = noise_amp[where], signal_amp[where], sample[where]
223
+ if noise.size == 0:
224
+ return noise_amp
225
+
226
+ log_scale = np.log(scale)
227
+ rises = log_scale < 0.0
228
+ if not np.any(rises):
229
+ # Nothing in this half can be raised: dividing by a `sample` above one
230
+ # lowers it. The loop would have exhausted `max_iter` and returned the
231
+ # record unchanged.
232
+ return noise_amp
233
+
234
+ # A bin that *already* touches needs a negative exponent to get there, so
235
+ # clamping the minimum at zero is what "no lift needed" means. Written this
236
+ # way rather than as an `if np.any(noise >= signal)` guard on purpose: the
237
+ # guard is a branch, and a branch is a step. As a bin crosses from below
238
+ # the signal to above it, `needed` passes smoothly through zero and the
239
+ # clamp keeps the result continuous. The guard instead jumped, which is why
240
+ # four CWT stations still disagreed across machines after the other three
241
+ # discontinuities were fixed.
242
+ needed = np.log(signal[rises] / noise[rises]) / -log_scale[rises]
243
+ exponent = max(0.0, float(np.min(needed)))
244
+ return np.asarray(noise_amp / sample**exponent)
245
+
246
+
247
+ #: Angles beyond this are not a correction to a noise level. At a quarter turn
248
+ #: the frequency and amplitude axes have swapped, and past it the rotated
249
+ #: spectrum runs backwards in frequency. The legacy loop would have kept going
250
+ #: to 250 radians — forty full turns — before giving up and returning zero;
251
+ #: anything needing more than this has already stopped meaning anything.
252
+ MAX_ROTATION = np.pi / 2
253
+
254
+ #: Coarse steps used to bracket the touching angle before bisection. Only the
255
+ #: bracket depends on this; the answer inside it does not, which is what keeps
256
+ #: the result continuous as the root crosses a step boundary.
257
+ _BRACKET_STEPS = 128
258
+
259
+
260
+ def rotate_log_spectrum(
261
+ log_freq: NDArray[np.float64], log_amp: NDArray[np.float64], theta: float
262
+ ) -> NDArray[np.float64]:
263
+ """Rotate a log-log spectrum through ``theta`` radians.
264
+
265
+ The trailing ``log_amp[0] * theta`` term is what makes this a rotation
266
+ *about the low-frequency end* rather than about the axis origin: without it
267
+ the whole curve translates as well as tilts, and the correction stops being
268
+ anchored to the one part of the noise record that is least contaminated.
269
+ """
270
+ return np.asarray(
271
+ log_freq * np.sin(theta) + log_amp * np.cos(theta) + log_amp[0] * theta
272
+ )
273
+
274
+
275
+ def _touching_angle(
276
+ log_freq: NDArray[np.float64],
277
+ log_noise: NDArray[np.float64],
278
+ log_signal: NDArray[np.float64],
279
+ where: NDArray[np.bool_],
280
+ *,
281
+ backwards: bool,
282
+ tol: float = 1e-12,
283
+ ) -> float:
284
+ """Smallest rotation at which any point in ``where`` reaches the signal.
285
+
286
+ Bracketed then bisected rather than stepped. The margin
287
+
288
+ .. code-block:: text
289
+
290
+ g(t) = max over `where` of [ rotate(t) - log_signal ]
291
+
292
+ is continuous in ``t`` and negative at ``t = 0`` unless the two already
293
+ touch, so the first ``t`` with ``g(t) >= 0`` is a root of a continuous
294
+ function and moves continuously with the data. Stepping to a multiple of a
295
+ fixed increment does not, which is what made the legacy version — and
296
+ ``boost`` before it was solved — differ between machines on last-bit
297
+ input differences.
298
+
299
+ Returns ``0.0`` when the halves already touch, and when no rotation within
300
+ :data:`MAX_ROTATION` brings them together. Both are the legacy fallbacks;
301
+ the difference is that the second no longer takes five thousand iterations
302
+ to discover.
303
+ """
304
+ if not np.any(where):
305
+ return 0.0
306
+
307
+ sign = -1.0 if backwards else 1.0
308
+ target = log_signal[where]
309
+
310
+ def margin(t: float) -> float:
311
+ rotated = rotate_log_spectrum(log_freq, log_noise, sign * t)[where]
312
+ return float(np.max(rotated - target))
313
+
314
+ if margin(0.0) >= 0.0:
315
+ return 0.0
316
+
317
+ step = MAX_ROTATION / _BRACKET_STEPS
318
+ lo = 0.0
319
+ for i in range(1, _BRACKET_STEPS + 1):
320
+ hi = i * step
321
+ if margin(hi) >= 0.0:
322
+ break
323
+ lo = hi
324
+ else:
325
+ return 0.0
326
+
327
+ while hi - lo > tol:
328
+ mid = 0.5 * (lo + hi)
329
+ if margin(mid) >= 0.0:
330
+ hi = mid
331
+ else:
332
+ lo = mid
333
+ return sign * hi
334
+
335
+
336
+ def rotate_noise(
337
+ freq: NDArray[np.float64],
338
+ noise_amp: NDArray[np.float64],
339
+ signal_amp: NDArray[np.float64],
340
+ ) -> NDArray[np.float64]:
341
+ """Multiplicative factor from rotating the noise toward the signal.
342
+
343
+ See the module docstring for what this assumes and how it differs from
344
+ :func:`boost_noise`. Returns the factor, for the same reason
345
+ :func:`boost_noise` does.
346
+ """
347
+ if noise_amp.shape != signal_amp.shape or noise_amp.shape != freq.shape:
348
+ raise ValueError(
349
+ f"freq {freq.shape}, noise {noise_amp.shape} and signal "
350
+ f"{signal_amp.shape} must all match"
351
+ )
352
+
353
+ log_freq = np.log10(freq)
354
+ log_noise = np.log10(noise_amp)
355
+ log_signal = np.log10(signal_amp)
356
+
357
+ centroid = centroid_frequency(freq, signal_amp)
358
+ low = freq <= centroid
359
+
360
+ back = _touching_angle(log_freq, log_noise, log_signal, low, backwards=True)
361
+ forward = _touching_angle(log_freq, log_noise, log_signal, ~low, backwards=False)
362
+
363
+ lifted = np.maximum(
364
+ 10.0 ** rotate_log_spectrum(log_freq, log_noise, back),
365
+ 10.0 ** rotate_log_spectrum(log_freq, log_noise, forward),
366
+ )
367
+ return np.asarray(lifted / noise_amp)
368
+
369
+
370
+ @runtime_checkable
371
+ class NoiseModel(Protocol):
372
+ """Anything that produces a multiplicative correction to a noise spectrum.
373
+
374
+ Implementations are frozen dataclasses carrying their own parameters, so a
375
+ configured model can be stored, compared and recorded in provenance.
376
+ """
377
+
378
+ @property
379
+ def name(self) -> str:
380
+ """Short identifier, recorded alongside the result."""
381
+ ...
382
+
383
+ def factor(
384
+ self,
385
+ freq: NDArray[np.float64],
386
+ noise_amp: NDArray[np.float64],
387
+ signal_amp: NDArray[np.float64],
388
+ ) -> NDArray[np.float64]:
389
+ """Multiply the noise by this to get the level to judge against."""
390
+ ...
391
+
392
+
393
+ @dataclass(frozen=True)
394
+ class BoostNoise:
395
+ """The default: lift each tail until it touches the signal.
396
+
397
+ See the module docstring for the assumption this encodes and
398
+ :func:`boost_noise` for the derivation of the exponent.
399
+ """
400
+
401
+ space: tuple[float, float] = (0.001, 1.001)
402
+
403
+ @property
404
+ def name(self) -> str:
405
+ return "boost"
406
+
407
+ def factor(
408
+ self,
409
+ freq: NDArray[np.float64],
410
+ noise_amp: NDArray[np.float64],
411
+ signal_amp: NDArray[np.float64],
412
+ ) -> NDArray[np.float64]:
413
+ return boost_noise(freq, noise_amp, signal_amp, space=self.space)
414
+
415
+
416
+ @dataclass(frozen=True)
417
+ class RotateNoise:
418
+ """Tilt the noise toward the signal rather than scaling it.
419
+
420
+ The legacy ``ROT_METHOD = 1``. See the module docstring for the assumption
421
+ this encodes, how it differs from :class:`BoostNoise`, and the two
422
+ departures from the legacy implementation.
423
+ """
424
+
425
+ @property
426
+ def name(self) -> str:
427
+ return "rotate"
428
+
429
+ def factor(
430
+ self,
431
+ freq: NDArray[np.float64],
432
+ noise_amp: NDArray[np.float64],
433
+ signal_amp: NDArray[np.float64],
434
+ ) -> NDArray[np.float64]:
435
+ return rotate_noise(freq, noise_amp, signal_amp)
436
+
437
+
438
+ @dataclass(frozen=True)
439
+ class NoNoiseModel:
440
+ """Use the recorded noise as measured.
441
+
442
+ Not a placeholder — it is the honest choice when the noise window is
443
+ genuinely representative, and it is what a run needs in order to show what
444
+ the correction is doing. Every other model here should be compared against
445
+ it before being trusted.
446
+ """
447
+
448
+ @property
449
+ def name(self) -> str:
450
+ return "none"
451
+
452
+ def factor(
453
+ self,
454
+ freq: NDArray[np.float64],
455
+ noise_amp: NDArray[np.float64],
456
+ signal_amp: NDArray[np.float64],
457
+ ) -> NDArray[np.float64]:
458
+ del freq, signal_amp
459
+ return np.ones_like(noise_amp)
460
+
461
+
462
+ #: Registered noise models, by the name configuration refers to them by.
463
+ NOISE_MODELS: dict[str, type[BoostNoise] | type[RotateNoise] | type[NoNoiseModel]] = {
464
+ "boost": BoostNoise,
465
+ "rotate": RotateNoise,
466
+ "none": NoNoiseModel,
467
+ }
468
+
469
+
470
+ def get_noise_model(name: str) -> NoiseModel:
471
+ """Resolve a registered model by name, with its defaults."""
472
+ try:
473
+ cls = NOISE_MODELS[name]
474
+ except KeyError:
475
+ raise ValueError(
476
+ f"Unknown noise model {name!r}. Available: {sorted(NOISE_MODELS)}."
477
+ ) from None
478
+ return cls()