das-mccc 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- das_mccc-0.2.0.dist-info/METADATA +107 -0
- das_mccc-0.2.0.dist-info/RECORD +13 -0
- das_mccc-0.2.0.dist-info/WHEEL +5 -0
- das_mccc-0.2.0.dist-info/licenses/LICENSE +21 -0
- das_mccc-0.2.0.dist-info/top_level.txt +1 -0
- dasmccc/__init__.py +45 -0
- dasmccc/anchor.py +79 -0
- dasmccc/core.py +210 -0
- dasmccc/legacy.py +91 -0
- dasmccc/ops.py +61 -0
- dasmccc/pipeline.py +597 -0
- dasmccc/polarity.py +150 -0
- dasmccc/signal.py +147 -0
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: das-mccc
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Iterative network MCCC refinement of DAS arrival curves with absolute anchoring and Ricker polarity QC
|
|
5
|
+
Author-email: Jaewoo Kim <jk103@rice.edu>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/Jaewoo-Kim-Rice/das-mccc
|
|
8
|
+
Project-URL: Algorithm, https://github.com/Jaewoo-Kim-Rice/das-mccc/blob/main/docs/algorithm.md
|
|
9
|
+
Keywords: DAS,distributed acoustic sensing,microseismic,cross-correlation,phase picking
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Physics
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Requires-Dist: numpy
|
|
21
|
+
Requires-Dist: scipy
|
|
22
|
+
Provides-Extra: numba
|
|
23
|
+
Requires-Dist: numba; extra == "numba"
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: pytest; extra == "dev"
|
|
26
|
+
Requires-Dist: ruff; extra == "dev"
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# das-mccc
|
|
30
|
+
|
|
31
|
+
Refine a DAS arrival curve with an iterative network multi-channel cross-correlation
|
|
32
|
+
(MCCC), anchor its absolute level on the aligned stack, and return per-channel polarity
|
|
33
|
+
and quality measures. Arrays in, arrays out: no site, file or path concepts.
|
|
34
|
+
|
|
35
|
+
The core is the refiner of [das-focmec](https://github.com/Jaewoo-Kim-Rice/das-focmec)
|
|
36
|
+
(`das_focmec.processing.mccc_core.ultra_mccc_iterative`), extracted with its history so it
|
|
37
|
+
can be used by any picker. Started from a rough curve (a VLM trace, a bracket, a
|
|
38
|
+
theoretical moveout) it recovers the shape of the arrival to the precision of a human
|
|
39
|
+
curve: on 24 CAPE 2025 reads the shape MAD against human picks went from 2.2 to 1.95 ms
|
|
40
|
+
(P) and 4.5 to 4.15 ms (S), and a curve started on the wrong lobe went from 5.6 to 2.65 ms.
|
|
41
|
+
|
|
42
|
+
## Install
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
pip install -e . # numpy, scipy
|
|
46
|
+
pip install -e '.[numba]' # fast pairwise correlation (strongly recommended)
|
|
47
|
+
pip install -e '.[dev]' # pytest, ruff
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Without numba the pairwise correlation runs in pure numpy: identical results, one to two
|
|
51
|
+
orders of magnitude slower, and a warning is logged at import.
|
|
52
|
+
|
|
53
|
+
## Use
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
import numpy as np
|
|
57
|
+
from dasmccc import refine_curve, refine_phases, DIRECT, SECONDARY
|
|
58
|
+
|
|
59
|
+
# waveform: (n_channels, n_samples) float, filtered as you like
|
|
60
|
+
# curve : (n_channels,) arrival in samples, NaN where the phase is not picked
|
|
61
|
+
res = refine_curve(waveform, curve, DIRECT)
|
|
62
|
+
|
|
63
|
+
res.curve # refined arrival (samples), NaN where the input was NaN
|
|
64
|
+
res.curve_relative # same shape, at the initial curve's level (no anchor)
|
|
65
|
+
res.anchor_offset # samples added by the anchor rule (NaN if the rule refused)
|
|
66
|
+
res.polarity # -1 / 0 / +1 per channel
|
|
67
|
+
res.snr, res.coherence, res.kept
|
|
68
|
+
res.aligned, res.stack # the aligned window and its stack, for plots
|
|
69
|
+
|
|
70
|
+
# several phases of one gather, strongest first; refined phases are masked for the next and a
|
|
71
|
+
# narrow tapered pre-mask (+-40 samples, 10 taper) keeps them out of the correlation
|
|
72
|
+
out = refine_phases(waveform, {"S": s_curve, "P": p_curve, "SP": sp_curve})
|
|
73
|
+
# several curves of one tag: any keys, plus a key -> tag map
|
|
74
|
+
out = refine_phases(waveform, {"S": s_curve, "R1": r1, "R2": r2},
|
|
75
|
+
tags={"S": "S", "R1": "REFL", "R2": "REFL"})
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Settings live in `RefineConfig` (everything in samples and channels). `DIRECT` is the
|
|
79
|
+
das-focmec configuration for direct waves (window 200, corr_len 200, smoothness 50, four
|
|
80
|
+
passes, pre-mask 100); `SECONDARY` narrows it for conversions and reflections (window
|
|
81
|
+
120, corr_len 100, three passes, pre-mask 50). Both were calibrated at 1 kHz and 2 m channel
|
|
82
|
+
spacing; `docs/algorithm.md` gives the conversion to other rates and spacings, what each
|
|
83
|
+
knob does, and the anchoring and masking rules.
|
|
84
|
+
|
|
85
|
+
## For das-focmec
|
|
86
|
+
|
|
87
|
+
`dasmccc.legacy` exposes `ultra_mccc_iterative` and `diff_corr_ric` with the das-focmec
|
|
88
|
+
signatures and identical results, so `das_focmec.processing.workflows` only changes its
|
|
89
|
+
import line.
|
|
90
|
+
|
|
91
|
+
## What it does not do
|
|
92
|
+
|
|
93
|
+
* It does not re-pick. The initial curve decides which arrival and roughly which lobe is
|
|
94
|
+
refined; MCCC measures relative delays within `pair_slope` samples per channel of it.
|
|
95
|
+
* The anchor moves the whole curve by one offset measured on the stack (secondary phases
|
|
96
|
+
inherit their parent's). The first-lobe rule sits about 5 ms before the human pick for P
|
|
97
|
+
and within a few ms of it for S on the CAPE 2025 fibres, with a per-fibre constant;
|
|
98
|
+
calibrate it per site against a few human picks when onsets are needed.
|
|
99
|
+
* Sub-sample precision: the alignment is integer; tau is a float but the returned curve
|
|
100
|
+
inherits the integer initial alignment plus the smoothed tau.
|
|
101
|
+
|
|
102
|
+
## Development
|
|
103
|
+
|
|
104
|
+
```
|
|
105
|
+
PYTHONPATH=src pytest -q
|
|
106
|
+
ruff check src tests && ruff format --check src tests
|
|
107
|
+
```
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
das_mccc-0.2.0.dist-info/licenses/LICENSE,sha256=sryuJWrcza8-CCtUmy0VR6_D7vK90fPOXfhE4VxpiHw,1072
|
|
2
|
+
dasmccc/__init__.py,sha256=txpbzWpT2g2Qmx97HhcP3TQWlcS5InMDptJWNGYnrTQ,1051
|
|
3
|
+
dasmccc/anchor.py,sha256=VrbKqi4cHaFbaH47QW9lfLCFrFEuWz0lVTAAeWSTVxA,3317
|
|
4
|
+
dasmccc/core.py,sha256=S6yYlh27x14CkHJlCyAqgK4WxGGol9EjItwKHto3JJ0,8331
|
|
5
|
+
dasmccc/legacy.py,sha256=U-5uQSe6_tApURflt9DGCT19xE3JXAnrHTu3o5brZc8,3642
|
|
6
|
+
dasmccc/ops.py,sha256=66efQzMPLbz8EB1jAi5_O5Tp-Tdc3WVh0AXenGDvX5o,2731
|
|
7
|
+
dasmccc/pipeline.py,sha256=2JruQ7CpusuMMwlRSuNt7GzbA4VxOoTA7DQvCksxf7Q,27159
|
|
8
|
+
dasmccc/polarity.py,sha256=jcsZ82BVjnx5H-EIeOdN3oNoR_2zn_qzQA_4z1y1lsY,5730
|
|
9
|
+
dasmccc/signal.py,sha256=KjZ70-Jo29bRVOadI2LN2TP4vL3yr1l5zxbkKfq1hac,5338
|
|
10
|
+
das_mccc-0.2.0.dist-info/METADATA,sha256=CJdgKKNdntuXMnzibzyETPloaTzjWUU6knw5fhtUBUQ,4890
|
|
11
|
+
das_mccc-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
12
|
+
das_mccc-0.2.0.dist-info/top_level.txt,sha256=abRPwUZpu0ZjfM0cQVQfjwOg5D72N4Ay-Z9DqASjBKo,8
|
|
13
|
+
das_mccc-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025-2026 Jaewoo Kim
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
dasmccc
|
dasmccc/__init__.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""dasmccc: iterative network MCCC refinement of DAS arrival curves.
|
|
2
|
+
|
|
3
|
+
from dasmccc import refine_curve, DIRECT
|
|
4
|
+
res = refine_curve(waveform, curve, DIRECT) # arrays in, RefineResult out
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
__version__ = "0.2.0"
|
|
8
|
+
|
|
9
|
+
from .anchor import first_lobe, stack_peak
|
|
10
|
+
from .core import iterate_align, mccc, pairwise_lags, partner_pairs, solve_tau
|
|
11
|
+
from .pipeline import (
|
|
12
|
+
DIRECT,
|
|
13
|
+
SECONDARY,
|
|
14
|
+
NothingToRefine,
|
|
15
|
+
RefineConfig,
|
|
16
|
+
RefineResult,
|
|
17
|
+
refine_curve,
|
|
18
|
+
refine_phases,
|
|
19
|
+
)
|
|
20
|
+
from .polarity import PolarityConfig, PolarityResult, RickerWindows, ricker_polarity, ricker_windows
|
|
21
|
+
from .signal import NUMBA_AVAILABLE
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"DIRECT",
|
|
25
|
+
"__version__",
|
|
26
|
+
"NUMBA_AVAILABLE",
|
|
27
|
+
"NothingToRefine",
|
|
28
|
+
"SECONDARY",
|
|
29
|
+
"PolarityConfig",
|
|
30
|
+
"PolarityResult",
|
|
31
|
+
"RefineConfig",
|
|
32
|
+
"RefineResult",
|
|
33
|
+
"RickerWindows",
|
|
34
|
+
"first_lobe",
|
|
35
|
+
"iterate_align",
|
|
36
|
+
"mccc",
|
|
37
|
+
"pairwise_lags",
|
|
38
|
+
"partner_pairs",
|
|
39
|
+
"refine_curve",
|
|
40
|
+
"refine_phases",
|
|
41
|
+
"ricker_polarity",
|
|
42
|
+
"ricker_windows",
|
|
43
|
+
"solve_tau",
|
|
44
|
+
"stack_peak",
|
|
45
|
+
]
|
dasmccc/anchor.py
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Absolute anchoring of a relatively aligned gather.
|
|
2
|
+
|
|
3
|
+
Network MCCC fixes only relative delays; the level of the refined curve is whatever the
|
|
4
|
+
initial curve's level was (the lobe the initial picker traced). The rules here measure one
|
|
5
|
+
offset on the aligned stack so that the curve is moved to a reproducible feature of the
|
|
6
|
+
wavelet. All offsets are in samples relative to ``centre`` (the alignment sample).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
log = logging.getLogger("dasmccc")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def stack_peak(stack: np.ndarray, centre: int) -> float:
|
|
19
|
+
"""Offset of the |stack| maximum. Reproducible on strong reads, but the peak lobe is
|
|
20
|
+
not the same lobe on every read (spread of tens of ms against human onsets)."""
|
|
21
|
+
return float(int(np.argmax(np.abs(stack))) - centre)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def first_lobe(
|
|
25
|
+
stack: np.ndarray,
|
|
26
|
+
centre: int,
|
|
27
|
+
min_frac: float = 0.4,
|
|
28
|
+
guard: int | None = 40,
|
|
29
|
+
contiguous: bool = True,
|
|
30
|
+
window: tuple[int, int] | None = (-30, 10),
|
|
31
|
+
) -> float:
|
|
32
|
+
"""Offset of the centre of the first lobe of the stack.
|
|
33
|
+
|
|
34
|
+
Candidates are the local maxima of |stack|. ``window`` = (lo, hi) restricts the search to
|
|
35
|
+
``centre + lo .. centre + hi`` (samples): the prior that the initial picker traced a lobe
|
|
36
|
+
of the arrival, so the onset lies at most one wavelet before it and hardly after it. The
|
|
37
|
+
reference amplitude is the |stack| peak inside the window. With ``contiguous`` the rule
|
|
38
|
+
walks back from that peak lobe by lobe while each lobe keeps at least ``min_frac`` of the
|
|
39
|
+
peak and returns the earliest lobe of that run (a precursor separated by a weaker lobe is
|
|
40
|
+
not the onset); without it the earliest candidate above ``min_frac`` anywhere before the
|
|
41
|
+
peak is taken (the original rule). The peak itself is returned when no earlier lobe
|
|
42
|
+
qualifies.
|
|
43
|
+
|
|
44
|
+
Tuned on 616 CAPE 2025 reads with human picks (das-phase-agent research record,
|
|
45
|
+
`docs/11_anchor_tuning.md`): window (-30, 10), contiguous, min_frac 0.4 on the
|
|
46
|
+
channel-normalised stack removed every refusal and halved the gross anchor errors
|
|
47
|
+
against the original rule (window None, contiguous False, min_frac 0.3, plain stack).
|
|
48
|
+
|
|
49
|
+
``guard`` bounds |offset|: a larger offset is judged unreliable, a warning is logged and
|
|
50
|
+
NaN is returned so the caller keeps the relative level.
|
|
51
|
+
"""
|
|
52
|
+
a = np.abs(np.asarray(stack, float))
|
|
53
|
+
n = len(a)
|
|
54
|
+
if window is None:
|
|
55
|
+
lo, hi = 0, n
|
|
56
|
+
else:
|
|
57
|
+
lo, hi = max(0, centre + int(window[0])), min(n, centre + int(window[1]) + 1)
|
|
58
|
+
if hi - lo < 3:
|
|
59
|
+
raise ValueError(f"anchor window {window} leaves no samples around centre {centre}")
|
|
60
|
+
pk = lo + int(np.argmax(a[lo:hi]))
|
|
61
|
+
ext = [i for i in range(max(1, lo), pk) if a[i] >= a[i - 1] and a[i] >= a[i + 1]]
|
|
62
|
+
first = pk
|
|
63
|
+
if contiguous:
|
|
64
|
+
for i in reversed(ext):
|
|
65
|
+
if a[i] >= min_frac * a[pk]:
|
|
66
|
+
first = i
|
|
67
|
+
else:
|
|
68
|
+
break
|
|
69
|
+
else:
|
|
70
|
+
ok = [i for i in ext if a[i] >= min_frac * a[pk]]
|
|
71
|
+
if ok:
|
|
72
|
+
first = ok[0]
|
|
73
|
+
offset = float(first - centre)
|
|
74
|
+
if guard is not None and abs(offset) > guard:
|
|
75
|
+
log.warning(
|
|
76
|
+
"first_lobe anchor %+.0f samples exceeds guard %d; anchor not applied", offset, guard
|
|
77
|
+
)
|
|
78
|
+
return float("nan")
|
|
79
|
+
return offset
|
dasmccc/core.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Network multi-channel cross-correlation (MCCC) on an aligned DAS gather.
|
|
2
|
+
|
|
3
|
+
One MCCC pass measures, for every channel, the lag that maximises the absolute
|
|
4
|
+
correlation with about fifty partner channels drawn from a normal distribution
|
|
5
|
+
within +-corr_len channels, then solves the sparse least-squares problem
|
|
6
|
+
|
|
7
|
+
min_tau || lamb * (tau_i - tau_j - lag_ij) ||^2 + || smoothness * (tau_{c+1} - tau_c) ||^2
|
|
8
|
+
|
|
9
|
+
and smooths tau with a moving average. ``iterate_align`` repeats the pass, applying tau
|
|
10
|
+
to the gather and median-filtering it along the fibre between passes.
|
|
11
|
+
|
|
12
|
+
Sign convention: a positive tau[c] means channel c currently arrives *later* than its
|
|
13
|
+
partners by tau samples (the aligned trace has to be advanced by tau).
|
|
14
|
+
|
|
15
|
+
Every quantity is in samples or channels; the per-pair lag bound
|
|
16
|
+
``max(pair_slope * |i - j|, pair_min_shift)`` is a bound on the *residual* moveout
|
|
17
|
+
slope relative to the initial curve, not on the absolute moveout.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import logging
|
|
23
|
+
|
|
24
|
+
import numpy as np
|
|
25
|
+
from scipy import sparse
|
|
26
|
+
from scipy.sparse.linalg import lsqr
|
|
27
|
+
|
|
28
|
+
from .ops import moving_avg, spatial_median, tau_shift
|
|
29
|
+
from .signal import NUMBA_AVAILABLE, jit, limited_cc, normal_distribution, prange
|
|
30
|
+
|
|
31
|
+
log = logging.getLogger("dasmccc")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def partner_pairs(
|
|
35
|
+
n_channels: int, corr_len: int, n_partners: int = 50, partner_std: float = 20.0
|
|
36
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
37
|
+
"""Channel pairs (i, j) with j > i to correlate.
|
|
38
|
+
|
|
39
|
+
For each channel i, ``n_partners`` candidates are drawn as quantiles of a normal
|
|
40
|
+
distribution (std ``partner_std`` channels) truncated to
|
|
41
|
+
[i - corr_len, min(n_channels - 1, i + corr_len)]; candidates outside the fibre or
|
|
42
|
+
with j <= i are dropped, so each channel keeps roughly n_partners / 2 partners ahead
|
|
43
|
+
of it (the pairs behind it come from the earlier channels).
|
|
44
|
+
"""
|
|
45
|
+
pairs_i, pairs_j = [], []
|
|
46
|
+
for i in range(n_channels):
|
|
47
|
+
cand = normal_distribution(
|
|
48
|
+
i - corr_len, min(n_channels - 1, i + corr_len), n_partners, partner_std
|
|
49
|
+
)
|
|
50
|
+
j = cand[(cand >= 0) & (cand < n_channels) & (cand > i)]
|
|
51
|
+
if j.size:
|
|
52
|
+
pairs_i.append(np.full(j.size, i, dtype=np.int64))
|
|
53
|
+
pairs_j.append(j.astype(np.int64))
|
|
54
|
+
if not pairs_i:
|
|
55
|
+
return np.zeros(0, np.int64), np.zeros(0, np.int64)
|
|
56
|
+
return np.concatenate(pairs_i), np.concatenate(pairs_j)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@jit(nopython=True, parallel=True, cache=True)
|
|
60
|
+
def _pair_lags_numba(data, pairs_i, pairs_j, pair_max_shift, lags): # pragma: no cover - compiled
|
|
61
|
+
n_samples = data.shape[1]
|
|
62
|
+
for k in prange(len(pairs_i)):
|
|
63
|
+
tr_i = data[pairs_i[k], :]
|
|
64
|
+
tr_j = data[pairs_j[k], :]
|
|
65
|
+
ms = pair_max_shift[k]
|
|
66
|
+
best_shift = 0
|
|
67
|
+
best_abs = 0.0
|
|
68
|
+
for s in range(2 * ms + 1):
|
|
69
|
+
shift = s - ms
|
|
70
|
+
total = 0.0
|
|
71
|
+
if shift < 0:
|
|
72
|
+
for idx in range(n_samples + shift):
|
|
73
|
+
total += tr_i[idx] * tr_j[idx - shift]
|
|
74
|
+
elif shift > 0:
|
|
75
|
+
for idx in range(n_samples - shift):
|
|
76
|
+
total += tr_i[idx + shift] * tr_j[idx]
|
|
77
|
+
else:
|
|
78
|
+
for idx in range(n_samples):
|
|
79
|
+
total += tr_i[idx] * tr_j[idx]
|
|
80
|
+
if abs(total) > best_abs:
|
|
81
|
+
best_abs = abs(total)
|
|
82
|
+
best_shift = shift
|
|
83
|
+
lags[k] = best_shift
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def pairwise_lags(
|
|
87
|
+
data: np.ndarray,
|
|
88
|
+
pairs_i: np.ndarray,
|
|
89
|
+
pairs_j: np.ndarray,
|
|
90
|
+
pair_slope: float = 0.2,
|
|
91
|
+
pair_min_shift: int = 3,
|
|
92
|
+
use_numba: bool | None = None,
|
|
93
|
+
) -> np.ndarray:
|
|
94
|
+
"""Lag (samples) maximising |cross-correlation| of each pair, searched within
|
|
95
|
+
+-max(pair_slope * (j - i), pair_min_shift) samples. Sign-agnostic, so a polarity
|
|
96
|
+
flip between channels does not break the alignment.
|
|
97
|
+
"""
|
|
98
|
+
if use_numba is None:
|
|
99
|
+
use_numba = NUMBA_AVAILABLE
|
|
100
|
+
if use_numba and not NUMBA_AVAILABLE:
|
|
101
|
+
raise RuntimeError("use_numba=True requested but numba is not installed")
|
|
102
|
+
pair_max_shift = np.maximum((pairs_j - pairs_i) * pair_slope, pair_min_shift).astype(np.int64)
|
|
103
|
+
lags = np.zeros(len(pairs_i), dtype=np.float64)
|
|
104
|
+
if len(pairs_i) == 0:
|
|
105
|
+
return lags
|
|
106
|
+
data_c = np.ascontiguousarray(data, dtype=np.float64)
|
|
107
|
+
if use_numba:
|
|
108
|
+
_pair_lags_numba(data_c, pairs_i, pairs_j, pair_max_shift, lags)
|
|
109
|
+
else:
|
|
110
|
+
for k in range(len(pairs_i)):
|
|
111
|
+
ms = int(pair_max_shift[k])
|
|
112
|
+
corr = limited_cc(data_c[pairs_i[k]], data_c[pairs_j[k]], ms, use_numba=False)
|
|
113
|
+
lags[k] = np.argmax(np.abs(corr)) - ms
|
|
114
|
+
return lags
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def solve_tau(
|
|
118
|
+
n_channels: int,
|
|
119
|
+
pairs_i: np.ndarray,
|
|
120
|
+
pairs_j: np.ndarray,
|
|
121
|
+
lags: np.ndarray,
|
|
122
|
+
lamb: float = 1.0,
|
|
123
|
+
smoothness: float = 0.0,
|
|
124
|
+
reference_dt: np.ndarray | None = None,
|
|
125
|
+
tau_avg: int = 100,
|
|
126
|
+
) -> np.ndarray:
|
|
127
|
+
"""Least-squares tau (n_channels,) from pairwise lags, with an optional first-difference
|
|
128
|
+
smoothness term and a final moving average of ``tau_avg`` channels.
|
|
129
|
+
|
|
130
|
+
``reference_dt`` (n_channels - 1,) makes the smoothness term target that adjacent-channel
|
|
131
|
+
difference instead of zero (follow a theoretical moveout while correlating).
|
|
132
|
+
"""
|
|
133
|
+
n_pairs = len(pairs_i)
|
|
134
|
+
rows = np.repeat(np.arange(n_pairs), 2)
|
|
135
|
+
cols = np.column_stack([pairs_i, pairs_j]).ravel()
|
|
136
|
+
vals = np.tile([1.0, -1.0], n_pairs)
|
|
137
|
+
diff = sparse.csr_matrix((lamb * vals, (rows, cols)), shape=(n_pairs, n_channels))
|
|
138
|
+
b = lamb * np.asarray(lags, float)
|
|
139
|
+
if smoothness > 0:
|
|
140
|
+
m = n_channels - 1
|
|
141
|
+
d_rows = np.repeat(np.arange(m), 2)
|
|
142
|
+
d_cols = np.column_stack([np.arange(m), np.arange(1, n_channels)]).ravel()
|
|
143
|
+
d_vals = np.tile([-1.0, 1.0], m)
|
|
144
|
+
d_mat = sparse.csr_matrix((smoothness * d_vals, (d_rows, d_cols)), shape=(m, n_channels))
|
|
145
|
+
if reference_dt is None:
|
|
146
|
+
b_smooth = np.zeros(m)
|
|
147
|
+
else:
|
|
148
|
+
reference_dt = np.asarray(reference_dt, float)
|
|
149
|
+
if reference_dt.shape != (m,):
|
|
150
|
+
raise ValueError(f"reference_dt must have shape ({m},), got {reference_dt.shape}")
|
|
151
|
+
b_smooth = reference_dt
|
|
152
|
+
diff = sparse.vstack([diff, d_mat]).tocsr()
|
|
153
|
+
b = np.concatenate([b, smoothness * b_smooth])
|
|
154
|
+
tau = lsqr(diff, b, atol=1e-10, btol=1e-10)[0]
|
|
155
|
+
if tau_avg > n_channels:
|
|
156
|
+
log.warning(
|
|
157
|
+
"tau_avg %d longer than the %d refined channels; averaging over all of them",
|
|
158
|
+
tau_avg,
|
|
159
|
+
n_channels,
|
|
160
|
+
)
|
|
161
|
+
tau_avg = n_channels
|
|
162
|
+
return moving_avg(tau, tau_avg)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def mccc(
|
|
166
|
+
data: np.ndarray,
|
|
167
|
+
corr_len: int,
|
|
168
|
+
n_partners: int = 50,
|
|
169
|
+
partner_std: float = 20.0,
|
|
170
|
+
pair_slope: float = 0.2,
|
|
171
|
+
pair_min_shift: int = 3,
|
|
172
|
+
lamb: float = 1.0,
|
|
173
|
+
smoothness: float = 0.0,
|
|
174
|
+
reference_dt: np.ndarray | None = None,
|
|
175
|
+
tau_avg: int = 100,
|
|
176
|
+
use_numba: bool | None = None,
|
|
177
|
+
) -> np.ndarray:
|
|
178
|
+
"""One MCCC pass on an aligned gather (n_channels, n_samples); returns tau (n_channels,)."""
|
|
179
|
+
pairs_i, pairs_j = partner_pairs(data.shape[0], corr_len, n_partners, partner_std)
|
|
180
|
+
lags = pairwise_lags(data, pairs_i, pairs_j, pair_slope, pair_min_shift, use_numba)
|
|
181
|
+
return solve_tau(data.shape[0], pairs_i, pairs_j, lags, lamb, smoothness, reference_dt, tau_avg)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def iterate_align(
|
|
185
|
+
aligned: np.ndarray,
|
|
186
|
+
corr_len: int,
|
|
187
|
+
n_iter: int = 4,
|
|
188
|
+
medfilt_channels: int = 25,
|
|
189
|
+
medfilt_iters: tuple[int, ...] = (1, 2, 3),
|
|
190
|
+
use_numba: bool | None = None,
|
|
191
|
+
**mccc_kwargs,
|
|
192
|
+
) -> tuple[np.ndarray, np.ndarray, list[np.ndarray]]:
|
|
193
|
+
"""Iterated MCCC on a pre-aligned, windowed gather.
|
|
194
|
+
|
|
195
|
+
Pass i (1-based) correlates within ``corr_len // i`` channels, applies tau, and median
|
|
196
|
+
filters the gather along the fibre when i is in ``medfilt_iters``. Returns
|
|
197
|
+
(aligned, total_tau, taus); ``total_tau`` (n_channels,) is the summed tau, so the
|
|
198
|
+
refined arrival of channel c is ``round(initial_pick[c]) + total_tau[c]`` in the
|
|
199
|
+
original sample axis (see ``pipeline.refine_curve``).
|
|
200
|
+
"""
|
|
201
|
+
taus = []
|
|
202
|
+
total = np.zeros(aligned.shape[0])
|
|
203
|
+
for i in range(1, n_iter + 1):
|
|
204
|
+
tau = mccc(aligned, corr_len // i, use_numba=use_numba, **mccc_kwargs)
|
|
205
|
+
taus.append(tau)
|
|
206
|
+
total += tau
|
|
207
|
+
aligned = tau_shift(aligned, tau)
|
|
208
|
+
if i in medfilt_iters:
|
|
209
|
+
aligned = spatial_median(aligned, medfilt_channels)
|
|
210
|
+
return aligned, total, taus
|
dasmccc/legacy.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""das-focmec compatible entry points.
|
|
2
|
+
|
|
3
|
+
``das_focmec.processing.workflows.mccc_pipeline`` was written against
|
|
4
|
+
``ultra_mccc_iterative`` and ``diff_corr_ric`` from its own ``mccc_core``; these wrappers
|
|
5
|
+
keep those signatures and return values exactly (intermediate arrays, legacy (n, 2) pick
|
|
6
|
+
and tau layouts) on top of the dasmccc building blocks, so that the focal-mechanism
|
|
7
|
+
pipeline reproduces bit for bit while new code uses ``dasmccc.refine_curve``.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import logging
|
|
13
|
+
import warnings
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
|
|
17
|
+
from .core import mccc
|
|
18
|
+
from .ops import shift_arr, spatial_median, tau_shift
|
|
19
|
+
from .polarity import ricker_windows
|
|
20
|
+
|
|
21
|
+
log = logging.getLogger("dasmccc")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def ultra_mccc_iterative(
|
|
25
|
+
das_arr: np.ndarray,
|
|
26
|
+
pick: np.ndarray,
|
|
27
|
+
corr_len: int,
|
|
28
|
+
max_shift: int | None,
|
|
29
|
+
lamb: float,
|
|
30
|
+
n_iterations: int = 3,
|
|
31
|
+
shrinked_window_length: int = 300,
|
|
32
|
+
medfilt_iterations=(1, 2, 3),
|
|
33
|
+
smoothness: float = 0.0,
|
|
34
|
+
reference_dt: np.ndarray | None = None,
|
|
35
|
+
pre_mccc_mask_half_width: int | None = None,
|
|
36
|
+
):
|
|
37
|
+
"""Legacy iterative MCCC.
|
|
38
|
+
|
|
39
|
+
pick : (n_channels, 2) array [channel index, pick sample] as in das-focmec.
|
|
40
|
+
max_shift : accepted for compatibility and ignored, as it always was: the per-pair lag
|
|
41
|
+
bound is max(0.2 * channel gap, 3) samples.
|
|
42
|
+
|
|
43
|
+
Returns ``(intermediates, total_shift, [first_shifts, base_time], tau_list)`` where
|
|
44
|
+
``intermediates = [das_arr, shifted, after pass 1, ...]``, the refined pick is
|
|
45
|
+
``base_time - total_shift`` and each tau is an (n_channels, 2) [index, tau] array.
|
|
46
|
+
"""
|
|
47
|
+
if max_shift is not None:
|
|
48
|
+
warnings.warn(
|
|
49
|
+
"ultra_mccc_iterative: max_shift has no effect (per-pair bound is "
|
|
50
|
+
"max(0.2 * gap, 3) samples); pass None to silence",
|
|
51
|
+
DeprecationWarning,
|
|
52
|
+
stacklevel=2,
|
|
53
|
+
)
|
|
54
|
+
half = shrinked_window_length // 2
|
|
55
|
+
base_time = das_arr.shape[1] // 2
|
|
56
|
+
pick = np.asarray(pick, float)
|
|
57
|
+
if pick.shape != (das_arr.shape[0], 2):
|
|
58
|
+
raise ValueError(f"pick must be (n_channels, 2), got {pick.shape}")
|
|
59
|
+
log.info("legacy ultra_mccc_iterative: initial shift")
|
|
60
|
+
shifted, first_shifts = shift_arr(das_arr, pick[:, 1], base_time)
|
|
61
|
+
shifted = shifted[:, base_time - half : base_time + half]
|
|
62
|
+
if pre_mccc_mask_half_width is not None:
|
|
63
|
+
centre = shifted.shape[1] // 2
|
|
64
|
+
keep = np.zeros_like(shifted)
|
|
65
|
+
sl = slice(centre - pre_mccc_mask_half_width, centre + pre_mccc_mask_half_width)
|
|
66
|
+
keep[:, sl] = shifted[:, sl]
|
|
67
|
+
shifted = keep
|
|
68
|
+
log.info(" pre-MCCC mask +-%d samples", pre_mccc_mask_half_width)
|
|
69
|
+
current = shifted
|
|
70
|
+
intermediates = [das_arr, shifted]
|
|
71
|
+
tau_list = []
|
|
72
|
+
total_shift = first_shifts.astype(np.float64).copy()
|
|
73
|
+
for i in range(1, n_iterations + 1):
|
|
74
|
+
log.info(" MCCC pass %d", i)
|
|
75
|
+
tau = mccc(
|
|
76
|
+
current, corr_len // i, lamb=lamb, smoothness=smoothness, reference_dt=reference_dt
|
|
77
|
+
)
|
|
78
|
+
tau_list.append(np.column_stack([np.arange(tau.size), tau]))
|
|
79
|
+
current = tau_shift(current, tau)
|
|
80
|
+
if i in medfilt_iterations:
|
|
81
|
+
current = spatial_median(current, 25)
|
|
82
|
+
intermediates.append(current)
|
|
83
|
+
total_shift -= tau
|
|
84
|
+
return intermediates, total_shift, [first_shifts, base_time], tau_list
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def diff_corr_ric(shifted_arr, max_shift, snr_thresh=5, mmad_thresh=4.0, ricker_freq=50):
|
|
88
|
+
"""Legacy tuple form of :func:`dasmccc.polarity.ricker_windows`:
|
|
89
|
+
``(dts, polarities, amps, snrs, wins)``."""
|
|
90
|
+
r = ricker_windows(shifted_arr, max_shift, snr_thresh, mmad_thresh, ricker_freq)
|
|
91
|
+
return r.dt, r.polarity, r.amplitude, r.snr, r.window
|
dasmccc/ops.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Array operations on a (n_channels, n_samples) gather: integer time shifts,
|
|
2
|
+
tau smoothing and the spatial (channel-axis) median filter."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
from scipy.signal import medfilt
|
|
8
|
+
|
|
9
|
+
PAD_VALUE = 1e-14 # samples rolled in from outside the trace (tiny, not exactly zero)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _shift_trace(trace: np.ndarray, shift: int) -> np.ndarray:
|
|
13
|
+
"""Delay ``trace`` by ``shift`` samples (negative = advance), padding with PAD_VALUE."""
|
|
14
|
+
if shift > 0:
|
|
15
|
+
return np.concatenate([np.full(shift, PAD_VALUE), trace[:-shift]])
|
|
16
|
+
if shift < 0:
|
|
17
|
+
return np.concatenate([trace[-shift:], np.full(-shift, PAD_VALUE)])
|
|
18
|
+
return trace.copy()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def shift_arr(data: np.ndarray, picks: np.ndarray, base_time: int) -> tuple[np.ndarray, np.ndarray]:
|
|
22
|
+
"""Shift every channel so that its pick (sample index, rounded) lands on ``base_time``.
|
|
23
|
+
|
|
24
|
+
Returns (shifted, shifts) with ``shifts[c] = round(base_time - picks[c])`` in samples.
|
|
25
|
+
"""
|
|
26
|
+
picks = np.asarray(picks, float)
|
|
27
|
+
if picks.shape != (data.shape[0],):
|
|
28
|
+
raise ValueError(f"picks must have shape ({data.shape[0]},), got {picks.shape}")
|
|
29
|
+
if not np.all(np.isfinite(picks)):
|
|
30
|
+
raise ValueError("picks must be finite for every channel")
|
|
31
|
+
shifts = np.round(base_time - picks).astype(int)
|
|
32
|
+
shifted = np.stack([_shift_trace(data[c], int(shifts[c])) for c in range(data.shape[0])])
|
|
33
|
+
return shifted, shifts
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def tau_shift(data: np.ndarray, tau: np.ndarray) -> np.ndarray:
|
|
37
|
+
"""Advance every channel by ``round(tau[c])`` samples (tau > 0 moves the trace earlier)."""
|
|
38
|
+
tau = np.asarray(tau, float)
|
|
39
|
+
if tau.shape != (data.shape[0],):
|
|
40
|
+
raise ValueError(f"tau must have shape ({data.shape[0]},), got {tau.shape}")
|
|
41
|
+
return np.stack([_shift_trace(data[c], int(round(-tau[c]))) for c in range(data.shape[0])])
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def moving_avg(x: np.ndarray, win: int) -> np.ndarray:
|
|
45
|
+
"""Centred moving average of length ``win`` with reflected edges; same length as ``x``."""
|
|
46
|
+
if win < 2:
|
|
47
|
+
return np.asarray(x, float).copy()
|
|
48
|
+
if win > len(x):
|
|
49
|
+
raise ValueError(f"moving_avg window {win} longer than the array ({len(x)})")
|
|
50
|
+
pad_left = win // 2
|
|
51
|
+
pad_right = win - 1 - pad_left
|
|
52
|
+
padded = np.pad(np.asarray(x, float), (pad_left, pad_right), mode="reflect")
|
|
53
|
+
return np.convolve(padded, np.ones(win) / win, mode="valid")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def spatial_median(data: np.ndarray, n_channels: int) -> np.ndarray:
|
|
57
|
+
"""Median filter along the channel axis with an odd kernel of ``n_channels``
|
|
58
|
+
(zero-padded at the fibre ends, as scipy.signal.medfilt does)."""
|
|
59
|
+
if n_channels % 2 == 0:
|
|
60
|
+
raise ValueError("spatial median kernel must be odd")
|
|
61
|
+
return medfilt(data, kernel_size=(n_channels, 1))
|