das-mccc 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,107 @@
1
+ Metadata-Version: 2.4
2
+ Name: das-mccc
3
+ Version: 0.2.0
4
+ Summary: Iterative network MCCC refinement of DAS arrival curves with absolute anchoring and Ricker polarity QC
5
+ Author-email: Jaewoo Kim <jk103@rice.edu>
6
+ License-Expression: MIT
7
+ Project-URL: Repository, https://github.com/Jaewoo-Kim-Rice/das-mccc
8
+ Project-URL: Algorithm, https://github.com/Jaewoo-Kim-Rice/das-mccc/blob/main/docs/algorithm.md
9
+ Keywords: DAS,distributed acoustic sensing,microseismic,cross-correlation,phase picking
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.10
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Topic :: Scientific/Engineering :: Physics
17
+ Requires-Python: >=3.10
18
+ Description-Content-Type: text/markdown
19
+ License-File: LICENSE
20
+ Requires-Dist: numpy
21
+ Requires-Dist: scipy
22
+ Provides-Extra: numba
23
+ Requires-Dist: numba; extra == "numba"
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest; extra == "dev"
26
+ Requires-Dist: ruff; extra == "dev"
27
+ Dynamic: license-file
28
+
29
+ # das-mccc
30
+
31
+ Refine a DAS arrival curve with an iterative network multi-channel cross-correlation
32
+ (MCCC), anchor its absolute level on the aligned stack, and return per-channel polarity
33
+ and quality measures. Arrays in, arrays out: no site, file or path concepts.
34
+
35
+ The core is the refiner of [das-focmec](https://github.com/Jaewoo-Kim-Rice/das-focmec)
36
+ (`das_focmec.processing.mccc_core.ultra_mccc_iterative`), extracted with its history so it
37
+ can be used by any picker. Started from a rough curve (a VLM trace, a bracket, a
38
+ theoretical moveout) it recovers the shape of the arrival to the precision of a human
39
+ curve: on 24 CAPE 2025 reads the shape MAD against human picks went from 2.2 to 1.95 ms
40
+ (P) and 4.5 to 4.15 ms (S), and a curve started on the wrong lobe went from 5.6 to 2.65 ms.
41
+
42
+ ## Install
43
+
44
+ ```
45
+ pip install -e . # numpy, scipy
46
+ pip install -e '.[numba]' # fast pairwise correlation (strongly recommended)
47
+ pip install -e '.[dev]' # pytest, ruff
48
+ ```
49
+
50
+ Without numba the pairwise correlation runs in pure numpy: identical results, one to two
51
+ orders of magnitude slower, and a warning is logged at import.
52
+
53
+ ## Use
54
+
55
+ ```python
56
+ import numpy as np
57
+ from dasmccc import refine_curve, refine_phases, DIRECT, SECONDARY
58
+
59
+ # waveform: (n_channels, n_samples) float, filtered as you like
60
+ # curve : (n_channels,) arrival in samples, NaN where the phase is not picked
61
+ res = refine_curve(waveform, curve, DIRECT)
62
+
63
+ res.curve # refined arrival (samples), NaN where the input was NaN
64
+ res.curve_relative # same shape, at the initial curve's level (no anchor)
65
+ res.anchor_offset # samples added by the anchor rule (NaN if the rule refused)
66
+ res.polarity # -1 / 0 / +1 per channel
67
+ res.snr, res.coherence, res.kept
68
+ res.aligned, res.stack # the aligned window and its stack, for plots
69
+
70
+ # several phases of one gather, strongest first; refined phases are masked for the next and a
71
+ # narrow tapered pre-mask (+-40 samples, 10 taper) keeps them out of the correlation
72
+ out = refine_phases(waveform, {"S": s_curve, "P": p_curve, "SP": sp_curve})
73
+ # several curves of one tag: any keys, plus a key -> tag map
74
+ out = refine_phases(waveform, {"S": s_curve, "R1": r1, "R2": r2},
75
+ tags={"S": "S", "R1": "REFL", "R2": "REFL"})
76
+ ```
77
+
78
+ Settings live in `RefineConfig` (everything in samples and channels). `DIRECT` is the
79
+ das-focmec configuration for direct waves (window 200, corr_len 200, smoothness 50, four
80
+ passes, pre-mask 100); `SECONDARY` narrows it for conversions and reflections (window
81
+ 120, corr_len 100, three passes, pre-mask 50). Both were calibrated at 1 kHz and 2 m channel
82
+ spacing; `docs/algorithm.md` gives the conversion to other rates and spacings, what each
83
+ knob does, and the anchoring and masking rules.
84
+
85
+ ## For das-focmec
86
+
87
+ `dasmccc.legacy` exposes `ultra_mccc_iterative` and `diff_corr_ric` with the das-focmec
88
+ signatures and identical results, so `das_focmec.processing.workflows` only changes its
89
+ import line.
90
+
91
+ ## What it does not do
92
+
93
+ * It does not re-pick. The initial curve decides which arrival and roughly which lobe is
94
+ refined; MCCC measures relative delays within `pair_slope` samples per channel of it.
95
+ * The anchor moves the whole curve by one offset measured on the stack (secondary phases
96
+ inherit their parent's). The first-lobe rule sits about 5 ms before the human pick for P
97
+ and within a few ms of it for S on the CAPE 2025 fibres, with a per-fibre constant;
98
+ calibrate it per site against a few human picks when onsets are needed.
99
+ * Sub-sample precision: the alignment is integer; tau is a float but the returned curve
100
+ inherits the integer initial alignment plus the smoothed tau.
101
+
102
+ ## Development
103
+
104
+ ```
105
+ PYTHONPATH=src pytest -q
106
+ ruff check src tests && ruff format --check src tests
107
+ ```
@@ -0,0 +1,13 @@
1
+ das_mccc-0.2.0.dist-info/licenses/LICENSE,sha256=sryuJWrcza8-CCtUmy0VR6_D7vK90fPOXfhE4VxpiHw,1072
2
+ dasmccc/__init__.py,sha256=txpbzWpT2g2Qmx97HhcP3TQWlcS5InMDptJWNGYnrTQ,1051
3
+ dasmccc/anchor.py,sha256=VrbKqi4cHaFbaH47QW9lfLCFrFEuWz0lVTAAeWSTVxA,3317
4
+ dasmccc/core.py,sha256=S6yYlh27x14CkHJlCyAqgK4WxGGol9EjItwKHto3JJ0,8331
5
+ dasmccc/legacy.py,sha256=U-5uQSe6_tApURflt9DGCT19xE3JXAnrHTu3o5brZc8,3642
6
+ dasmccc/ops.py,sha256=66efQzMPLbz8EB1jAi5_O5Tp-Tdc3WVh0AXenGDvX5o,2731
7
+ dasmccc/pipeline.py,sha256=2JruQ7CpusuMMwlRSuNt7GzbA4VxOoTA7DQvCksxf7Q,27159
8
+ dasmccc/polarity.py,sha256=jcsZ82BVjnx5H-EIeOdN3oNoR_2zn_qzQA_4z1y1lsY,5730
9
+ dasmccc/signal.py,sha256=KjZ70-Jo29bRVOadI2LN2TP4vL3yr1l5zxbkKfq1hac,5338
10
+ das_mccc-0.2.0.dist-info/METADATA,sha256=CJdgKKNdntuXMnzibzyETPloaTzjWUU6knw5fhtUBUQ,4890
11
+ das_mccc-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
12
+ das_mccc-0.2.0.dist-info/top_level.txt,sha256=abRPwUZpu0ZjfM0cQVQfjwOg5D72N4Ay-Z9DqASjBKo,8
13
+ das_mccc-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025-2026 Jaewoo Kim
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ dasmccc
dasmccc/__init__.py ADDED
@@ -0,0 +1,45 @@
1
+ """dasmccc: iterative network MCCC refinement of DAS arrival curves.
2
+
3
+ from dasmccc import refine_curve, DIRECT
4
+ res = refine_curve(waveform, curve, DIRECT) # arrays in, RefineResult out
5
+ """
6
+
7
+ __version__ = "0.2.0"
8
+
9
+ from .anchor import first_lobe, stack_peak
10
+ from .core import iterate_align, mccc, pairwise_lags, partner_pairs, solve_tau
11
+ from .pipeline import (
12
+ DIRECT,
13
+ SECONDARY,
14
+ NothingToRefine,
15
+ RefineConfig,
16
+ RefineResult,
17
+ refine_curve,
18
+ refine_phases,
19
+ )
20
+ from .polarity import PolarityConfig, PolarityResult, RickerWindows, ricker_polarity, ricker_windows
21
+ from .signal import NUMBA_AVAILABLE
22
+
23
+ __all__ = [
24
+ "DIRECT",
25
+ "__version__",
26
+ "NUMBA_AVAILABLE",
27
+ "NothingToRefine",
28
+ "SECONDARY",
29
+ "PolarityConfig",
30
+ "PolarityResult",
31
+ "RefineConfig",
32
+ "RefineResult",
33
+ "RickerWindows",
34
+ "first_lobe",
35
+ "iterate_align",
36
+ "mccc",
37
+ "pairwise_lags",
38
+ "partner_pairs",
39
+ "refine_curve",
40
+ "refine_phases",
41
+ "ricker_polarity",
42
+ "ricker_windows",
43
+ "solve_tau",
44
+ "stack_peak",
45
+ ]
dasmccc/anchor.py ADDED
@@ -0,0 +1,79 @@
1
+ """Absolute anchoring of a relatively aligned gather.
2
+
3
+ Network MCCC fixes only relative delays; the level of the refined curve is whatever the
4
+ initial curve's level was (the lobe the initial picker traced). The rules here measure one
5
+ offset on the aligned stack so that the curve is moved to a reproducible feature of the
6
+ wavelet. All offsets are in samples relative to ``centre`` (the alignment sample).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import logging
12
+
13
+ import numpy as np
14
+
15
+ log = logging.getLogger("dasmccc")
16
+
17
+
18
+ def stack_peak(stack: np.ndarray, centre: int) -> float:
19
+ """Offset of the |stack| maximum. Reproducible on strong reads, but the peak lobe is
20
+ not the same lobe on every read (spread of tens of ms against human onsets)."""
21
+ return float(int(np.argmax(np.abs(stack))) - centre)
22
+
23
+
24
+ def first_lobe(
25
+ stack: np.ndarray,
26
+ centre: int,
27
+ min_frac: float = 0.4,
28
+ guard: int | None = 40,
29
+ contiguous: bool = True,
30
+ window: tuple[int, int] | None = (-30, 10),
31
+ ) -> float:
32
+ """Offset of the centre of the first lobe of the stack.
33
+
34
+ Candidates are the local maxima of |stack|. ``window`` = (lo, hi) restricts the search to
35
+ ``centre + lo .. centre + hi`` (samples): the prior that the initial picker traced a lobe
36
+ of the arrival, so the onset lies at most one wavelet before it and hardly after it. The
37
+ reference amplitude is the |stack| peak inside the window. With ``contiguous`` the rule
38
+ walks back from that peak lobe by lobe while each lobe keeps at least ``min_frac`` of the
39
+ peak and returns the earliest lobe of that run (a precursor separated by a weaker lobe is
40
+ not the onset); without it the earliest candidate above ``min_frac`` anywhere before the
41
+ peak is taken (the original rule). The peak itself is returned when no earlier lobe
42
+ qualifies.
43
+
44
+ Tuned on 616 CAPE 2025 reads with human picks (das-phase-agent research record,
45
+ `docs/11_anchor_tuning.md`): window (-30, 10), contiguous, min_frac 0.4 on the
46
+ channel-normalised stack removed every refusal and halved the gross anchor errors
47
+ against the original rule (window None, contiguous False, min_frac 0.3, plain stack).
48
+
49
+ ``guard`` bounds |offset|: a larger offset is judged unreliable, a warning is logged and
50
+ NaN is returned so the caller keeps the relative level.
51
+ """
52
+ a = np.abs(np.asarray(stack, float))
53
+ n = len(a)
54
+ if window is None:
55
+ lo, hi = 0, n
56
+ else:
57
+ lo, hi = max(0, centre + int(window[0])), min(n, centre + int(window[1]) + 1)
58
+ if hi - lo < 3:
59
+ raise ValueError(f"anchor window {window} leaves no samples around centre {centre}")
60
+ pk = lo + int(np.argmax(a[lo:hi]))
61
+ ext = [i for i in range(max(1, lo), pk) if a[i] >= a[i - 1] and a[i] >= a[i + 1]]
62
+ first = pk
63
+ if contiguous:
64
+ for i in reversed(ext):
65
+ if a[i] >= min_frac * a[pk]:
66
+ first = i
67
+ else:
68
+ break
69
+ else:
70
+ ok = [i for i in ext if a[i] >= min_frac * a[pk]]
71
+ if ok:
72
+ first = ok[0]
73
+ offset = float(first - centre)
74
+ if guard is not None and abs(offset) > guard:
75
+ log.warning(
76
+ "first_lobe anchor %+.0f samples exceeds guard %d; anchor not applied", offset, guard
77
+ )
78
+ return float("nan")
79
+ return offset
dasmccc/core.py ADDED
@@ -0,0 +1,210 @@
1
+ """Network multi-channel cross-correlation (MCCC) on an aligned DAS gather.
2
+
3
+ One MCCC pass measures, for every channel, the lag that maximises the absolute
4
+ correlation with about fifty partner channels drawn from a normal distribution
5
+ within +-corr_len channels, then solves the sparse least-squares problem
6
+
7
+ min_tau || lamb * (tau_i - tau_j - lag_ij) ||^2 + || smoothness * (tau_{c+1} - tau_c) ||^2
8
+
9
+ and smooths tau with a moving average. ``iterate_align`` repeats the pass, applying tau
10
+ to the gather and median-filtering it along the fibre between passes.
11
+
12
+ Sign convention: a positive tau[c] means channel c currently arrives *later* than its
13
+ partners by tau samples (the aligned trace has to be advanced by tau).
14
+
15
+ Every quantity is in samples or channels; the per-pair lag bound
16
+ ``max(pair_slope * |i - j|, pair_min_shift)`` is a bound on the *residual* moveout
17
+ slope relative to the initial curve, not on the absolute moveout.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import logging
23
+
24
+ import numpy as np
25
+ from scipy import sparse
26
+ from scipy.sparse.linalg import lsqr
27
+
28
+ from .ops import moving_avg, spatial_median, tau_shift
29
+ from .signal import NUMBA_AVAILABLE, jit, limited_cc, normal_distribution, prange
30
+
31
+ log = logging.getLogger("dasmccc")
32
+
33
+
34
+ def partner_pairs(
35
+ n_channels: int, corr_len: int, n_partners: int = 50, partner_std: float = 20.0
36
+ ) -> tuple[np.ndarray, np.ndarray]:
37
+ """Channel pairs (i, j) with j > i to correlate.
38
+
39
+ For each channel i, ``n_partners`` candidates are drawn as quantiles of a normal
40
+ distribution (std ``partner_std`` channels) truncated to
41
+ [i - corr_len, min(n_channels - 1, i + corr_len)]; candidates outside the fibre or
42
+ with j <= i are dropped, so each channel keeps roughly n_partners / 2 partners ahead
43
+ of it (the pairs behind it come from the earlier channels).
44
+ """
45
+ pairs_i, pairs_j = [], []
46
+ for i in range(n_channels):
47
+ cand = normal_distribution(
48
+ i - corr_len, min(n_channels - 1, i + corr_len), n_partners, partner_std
49
+ )
50
+ j = cand[(cand >= 0) & (cand < n_channels) & (cand > i)]
51
+ if j.size:
52
+ pairs_i.append(np.full(j.size, i, dtype=np.int64))
53
+ pairs_j.append(j.astype(np.int64))
54
+ if not pairs_i:
55
+ return np.zeros(0, np.int64), np.zeros(0, np.int64)
56
+ return np.concatenate(pairs_i), np.concatenate(pairs_j)
57
+
58
+
59
+ @jit(nopython=True, parallel=True, cache=True)
60
+ def _pair_lags_numba(data, pairs_i, pairs_j, pair_max_shift, lags): # pragma: no cover - compiled
61
+ n_samples = data.shape[1]
62
+ for k in prange(len(pairs_i)):
63
+ tr_i = data[pairs_i[k], :]
64
+ tr_j = data[pairs_j[k], :]
65
+ ms = pair_max_shift[k]
66
+ best_shift = 0
67
+ best_abs = 0.0
68
+ for s in range(2 * ms + 1):
69
+ shift = s - ms
70
+ total = 0.0
71
+ if shift < 0:
72
+ for idx in range(n_samples + shift):
73
+ total += tr_i[idx] * tr_j[idx - shift]
74
+ elif shift > 0:
75
+ for idx in range(n_samples - shift):
76
+ total += tr_i[idx + shift] * tr_j[idx]
77
+ else:
78
+ for idx in range(n_samples):
79
+ total += tr_i[idx] * tr_j[idx]
80
+ if abs(total) > best_abs:
81
+ best_abs = abs(total)
82
+ best_shift = shift
83
+ lags[k] = best_shift
84
+
85
+
86
+ def pairwise_lags(
87
+ data: np.ndarray,
88
+ pairs_i: np.ndarray,
89
+ pairs_j: np.ndarray,
90
+ pair_slope: float = 0.2,
91
+ pair_min_shift: int = 3,
92
+ use_numba: bool | None = None,
93
+ ) -> np.ndarray:
94
+ """Lag (samples) maximising |cross-correlation| of each pair, searched within
95
+ +-max(pair_slope * (j - i), pair_min_shift) samples. Sign-agnostic, so a polarity
96
+ flip between channels does not break the alignment.
97
+ """
98
+ if use_numba is None:
99
+ use_numba = NUMBA_AVAILABLE
100
+ if use_numba and not NUMBA_AVAILABLE:
101
+ raise RuntimeError("use_numba=True requested but numba is not installed")
102
+ pair_max_shift = np.maximum((pairs_j - pairs_i) * pair_slope, pair_min_shift).astype(np.int64)
103
+ lags = np.zeros(len(pairs_i), dtype=np.float64)
104
+ if len(pairs_i) == 0:
105
+ return lags
106
+ data_c = np.ascontiguousarray(data, dtype=np.float64)
107
+ if use_numba:
108
+ _pair_lags_numba(data_c, pairs_i, pairs_j, pair_max_shift, lags)
109
+ else:
110
+ for k in range(len(pairs_i)):
111
+ ms = int(pair_max_shift[k])
112
+ corr = limited_cc(data_c[pairs_i[k]], data_c[pairs_j[k]], ms, use_numba=False)
113
+ lags[k] = np.argmax(np.abs(corr)) - ms
114
+ return lags
115
+
116
+
117
+ def solve_tau(
118
+ n_channels: int,
119
+ pairs_i: np.ndarray,
120
+ pairs_j: np.ndarray,
121
+ lags: np.ndarray,
122
+ lamb: float = 1.0,
123
+ smoothness: float = 0.0,
124
+ reference_dt: np.ndarray | None = None,
125
+ tau_avg: int = 100,
126
+ ) -> np.ndarray:
127
+ """Least-squares tau (n_channels,) from pairwise lags, with an optional first-difference
128
+ smoothness term and a final moving average of ``tau_avg`` channels.
129
+
130
+ ``reference_dt`` (n_channels - 1,) makes the smoothness term target that adjacent-channel
131
+ difference instead of zero (follow a theoretical moveout while correlating).
132
+ """
133
+ n_pairs = len(pairs_i)
134
+ rows = np.repeat(np.arange(n_pairs), 2)
135
+ cols = np.column_stack([pairs_i, pairs_j]).ravel()
136
+ vals = np.tile([1.0, -1.0], n_pairs)
137
+ diff = sparse.csr_matrix((lamb * vals, (rows, cols)), shape=(n_pairs, n_channels))
138
+ b = lamb * np.asarray(lags, float)
139
+ if smoothness > 0:
140
+ m = n_channels - 1
141
+ d_rows = np.repeat(np.arange(m), 2)
142
+ d_cols = np.column_stack([np.arange(m), np.arange(1, n_channels)]).ravel()
143
+ d_vals = np.tile([-1.0, 1.0], m)
144
+ d_mat = sparse.csr_matrix((smoothness * d_vals, (d_rows, d_cols)), shape=(m, n_channels))
145
+ if reference_dt is None:
146
+ b_smooth = np.zeros(m)
147
+ else:
148
+ reference_dt = np.asarray(reference_dt, float)
149
+ if reference_dt.shape != (m,):
150
+ raise ValueError(f"reference_dt must have shape ({m},), got {reference_dt.shape}")
151
+ b_smooth = reference_dt
152
+ diff = sparse.vstack([diff, d_mat]).tocsr()
153
+ b = np.concatenate([b, smoothness * b_smooth])
154
+ tau = lsqr(diff, b, atol=1e-10, btol=1e-10)[0]
155
+ if tau_avg > n_channels:
156
+ log.warning(
157
+ "tau_avg %d longer than the %d refined channels; averaging over all of them",
158
+ tau_avg,
159
+ n_channels,
160
+ )
161
+ tau_avg = n_channels
162
+ return moving_avg(tau, tau_avg)
163
+
164
+
165
+ def mccc(
166
+ data: np.ndarray,
167
+ corr_len: int,
168
+ n_partners: int = 50,
169
+ partner_std: float = 20.0,
170
+ pair_slope: float = 0.2,
171
+ pair_min_shift: int = 3,
172
+ lamb: float = 1.0,
173
+ smoothness: float = 0.0,
174
+ reference_dt: np.ndarray | None = None,
175
+ tau_avg: int = 100,
176
+ use_numba: bool | None = None,
177
+ ) -> np.ndarray:
178
+ """One MCCC pass on an aligned gather (n_channels, n_samples); returns tau (n_channels,)."""
179
+ pairs_i, pairs_j = partner_pairs(data.shape[0], corr_len, n_partners, partner_std)
180
+ lags = pairwise_lags(data, pairs_i, pairs_j, pair_slope, pair_min_shift, use_numba)
181
+ return solve_tau(data.shape[0], pairs_i, pairs_j, lags, lamb, smoothness, reference_dt, tau_avg)
182
+
183
+
184
+ def iterate_align(
185
+ aligned: np.ndarray,
186
+ corr_len: int,
187
+ n_iter: int = 4,
188
+ medfilt_channels: int = 25,
189
+ medfilt_iters: tuple[int, ...] = (1, 2, 3),
190
+ use_numba: bool | None = None,
191
+ **mccc_kwargs,
192
+ ) -> tuple[np.ndarray, np.ndarray, list[np.ndarray]]:
193
+ """Iterated MCCC on a pre-aligned, windowed gather.
194
+
195
+ Pass i (1-based) correlates within ``corr_len // i`` channels, applies tau, and median
196
+ filters the gather along the fibre when i is in ``medfilt_iters``. Returns
197
+ (aligned, total_tau, taus); ``total_tau`` (n_channels,) is the summed tau, so the
198
+ refined arrival of channel c is ``round(initial_pick[c]) + total_tau[c]`` in the
199
+ original sample axis (see ``pipeline.refine_curve``).
200
+ """
201
+ taus = []
202
+ total = np.zeros(aligned.shape[0])
203
+ for i in range(1, n_iter + 1):
204
+ tau = mccc(aligned, corr_len // i, use_numba=use_numba, **mccc_kwargs)
205
+ taus.append(tau)
206
+ total += tau
207
+ aligned = tau_shift(aligned, tau)
208
+ if i in medfilt_iters:
209
+ aligned = spatial_median(aligned, medfilt_channels)
210
+ return aligned, total, taus
dasmccc/legacy.py ADDED
@@ -0,0 +1,91 @@
1
+ """das-focmec compatible entry points.
2
+
3
+ ``das_focmec.processing.workflows.mccc_pipeline`` was written against
4
+ ``ultra_mccc_iterative`` and ``diff_corr_ric`` from its own ``mccc_core``; these wrappers
5
+ keep those signatures and return values exactly (intermediate arrays, legacy (n, 2) pick
6
+ and tau layouts) on top of the dasmccc building blocks, so that the focal-mechanism
7
+ pipeline reproduces bit for bit while new code uses ``dasmccc.refine_curve``.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import logging
13
+ import warnings
14
+
15
+ import numpy as np
16
+
17
+ from .core import mccc
18
+ from .ops import shift_arr, spatial_median, tau_shift
19
+ from .polarity import ricker_windows
20
+
21
+ log = logging.getLogger("dasmccc")
22
+
23
+
24
+ def ultra_mccc_iterative(
25
+ das_arr: np.ndarray,
26
+ pick: np.ndarray,
27
+ corr_len: int,
28
+ max_shift: int | None,
29
+ lamb: float,
30
+ n_iterations: int = 3,
31
+ shrinked_window_length: int = 300,
32
+ medfilt_iterations=(1, 2, 3),
33
+ smoothness: float = 0.0,
34
+ reference_dt: np.ndarray | None = None,
35
+ pre_mccc_mask_half_width: int | None = None,
36
+ ):
37
+ """Legacy iterative MCCC.
38
+
39
+ pick : (n_channels, 2) array [channel index, pick sample] as in das-focmec.
40
+ max_shift : accepted for compatibility and ignored, as it always was: the per-pair lag
41
+ bound is max(0.2 * channel gap, 3) samples.
42
+
43
+ Returns ``(intermediates, total_shift, [first_shifts, base_time], tau_list)`` where
44
+ ``intermediates = [das_arr, shifted, after pass 1, ...]``, the refined pick is
45
+ ``base_time - total_shift`` and each tau is an (n_channels, 2) [index, tau] array.
46
+ """
47
+ if max_shift is not None:
48
+ warnings.warn(
49
+ "ultra_mccc_iterative: max_shift has no effect (per-pair bound is "
50
+ "max(0.2 * gap, 3) samples); pass None to silence",
51
+ DeprecationWarning,
52
+ stacklevel=2,
53
+ )
54
+ half = shrinked_window_length // 2
55
+ base_time = das_arr.shape[1] // 2
56
+ pick = np.asarray(pick, float)
57
+ if pick.shape != (das_arr.shape[0], 2):
58
+ raise ValueError(f"pick must be (n_channels, 2), got {pick.shape}")
59
+ log.info("legacy ultra_mccc_iterative: initial shift")
60
+ shifted, first_shifts = shift_arr(das_arr, pick[:, 1], base_time)
61
+ shifted = shifted[:, base_time - half : base_time + half]
62
+ if pre_mccc_mask_half_width is not None:
63
+ centre = shifted.shape[1] // 2
64
+ keep = np.zeros_like(shifted)
65
+ sl = slice(centre - pre_mccc_mask_half_width, centre + pre_mccc_mask_half_width)
66
+ keep[:, sl] = shifted[:, sl]
67
+ shifted = keep
68
+ log.info(" pre-MCCC mask +-%d samples", pre_mccc_mask_half_width)
69
+ current = shifted
70
+ intermediates = [das_arr, shifted]
71
+ tau_list = []
72
+ total_shift = first_shifts.astype(np.float64).copy()
73
+ for i in range(1, n_iterations + 1):
74
+ log.info(" MCCC pass %d", i)
75
+ tau = mccc(
76
+ current, corr_len // i, lamb=lamb, smoothness=smoothness, reference_dt=reference_dt
77
+ )
78
+ tau_list.append(np.column_stack([np.arange(tau.size), tau]))
79
+ current = tau_shift(current, tau)
80
+ if i in medfilt_iterations:
81
+ current = spatial_median(current, 25)
82
+ intermediates.append(current)
83
+ total_shift -= tau
84
+ return intermediates, total_shift, [first_shifts, base_time], tau_list
85
+
86
+
87
+ def diff_corr_ric(shifted_arr, max_shift, snr_thresh=5, mmad_thresh=4.0, ricker_freq=50):
88
+ """Legacy tuple form of :func:`dasmccc.polarity.ricker_windows`:
89
+ ``(dts, polarities, amps, snrs, wins)``."""
90
+ r = ricker_windows(shifted_arr, max_shift, snr_thresh, mmad_thresh, ricker_freq)
91
+ return r.dt, r.polarity, r.amplitude, r.snr, r.window
dasmccc/ops.py ADDED
@@ -0,0 +1,61 @@
1
+ """Array operations on a (n_channels, n_samples) gather: integer time shifts,
2
+ tau smoothing and the spatial (channel-axis) median filter."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import numpy as np
7
+ from scipy.signal import medfilt
8
+
9
+ PAD_VALUE = 1e-14 # samples rolled in from outside the trace (tiny, not exactly zero)
10
+
11
+
12
+ def _shift_trace(trace: np.ndarray, shift: int) -> np.ndarray:
13
+ """Delay ``trace`` by ``shift`` samples (negative = advance), padding with PAD_VALUE."""
14
+ if shift > 0:
15
+ return np.concatenate([np.full(shift, PAD_VALUE), trace[:-shift]])
16
+ if shift < 0:
17
+ return np.concatenate([trace[-shift:], np.full(-shift, PAD_VALUE)])
18
+ return trace.copy()
19
+
20
+
21
+ def shift_arr(data: np.ndarray, picks: np.ndarray, base_time: int) -> tuple[np.ndarray, np.ndarray]:
22
+ """Shift every channel so that its pick (sample index, rounded) lands on ``base_time``.
23
+
24
+ Returns (shifted, shifts) with ``shifts[c] = round(base_time - picks[c])`` in samples.
25
+ """
26
+ picks = np.asarray(picks, float)
27
+ if picks.shape != (data.shape[0],):
28
+ raise ValueError(f"picks must have shape ({data.shape[0]},), got {picks.shape}")
29
+ if not np.all(np.isfinite(picks)):
30
+ raise ValueError("picks must be finite for every channel")
31
+ shifts = np.round(base_time - picks).astype(int)
32
+ shifted = np.stack([_shift_trace(data[c], int(shifts[c])) for c in range(data.shape[0])])
33
+ return shifted, shifts
34
+
35
+
36
+ def tau_shift(data: np.ndarray, tau: np.ndarray) -> np.ndarray:
37
+ """Advance every channel by ``round(tau[c])`` samples (tau > 0 moves the trace earlier)."""
38
+ tau = np.asarray(tau, float)
39
+ if tau.shape != (data.shape[0],):
40
+ raise ValueError(f"tau must have shape ({data.shape[0]},), got {tau.shape}")
41
+ return np.stack([_shift_trace(data[c], int(round(-tau[c]))) for c in range(data.shape[0])])
42
+
43
+
44
+ def moving_avg(x: np.ndarray, win: int) -> np.ndarray:
45
+ """Centred moving average of length ``win`` with reflected edges; same length as ``x``."""
46
+ if win < 2:
47
+ return np.asarray(x, float).copy()
48
+ if win > len(x):
49
+ raise ValueError(f"moving_avg window {win} longer than the array ({len(x)})")
50
+ pad_left = win // 2
51
+ pad_right = win - 1 - pad_left
52
+ padded = np.pad(np.asarray(x, float), (pad_left, pad_right), mode="reflect")
53
+ return np.convolve(padded, np.ones(win) / win, mode="valid")
54
+
55
+
56
+ def spatial_median(data: np.ndarray, n_channels: int) -> np.ndarray:
57
+ """Median filter along the channel axis with an odd kernel of ``n_channels``
58
+ (zero-padded at the fibre ends, as scipy.signal.medfilt does)."""
59
+ if n_channels % 2 == 0:
60
+ raise ValueError("spatial median kernel must be odd")
61
+ return medfilt(data, kernel_size=(n_channels, 1))