nltools 0.6.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nltools/__init__.py +55 -0
- nltools/algorithms/__init__.py +90 -0
- nltools/algorithms/alignment/__init__.py +21 -0
- nltools/algorithms/alignment/procrustes.py +565 -0
- nltools/algorithms/alignment/srm.py +758 -0
- nltools/algorithms/backends.py +1059 -0
- nltools/algorithms/corrections.py +177 -0
- nltools/algorithms/decoding.py +327 -0
- nltools/algorithms/inference/__init__.py +50 -0
- nltools/algorithms/inference/bootstrap.py +1386 -0
- nltools/algorithms/inference/correlation.py +373 -0
- nltools/algorithms/inference/intersubject.py +422 -0
- nltools/algorithms/inference/isc.py +1554 -0
- nltools/algorithms/inference/matrix.py +602 -0
- nltools/algorithms/inference/one_sample.py +288 -0
- nltools/algorithms/inference/random.py +122 -0
- nltools/algorithms/inference/timeseries.py +347 -0
- nltools/algorithms/inference/two_sample.py +212 -0
- nltools/algorithms/inference/utils.py +58 -0
- nltools/algorithms/inference/validation.py +282 -0
- nltools/algorithms/neighborhoods.py +207 -0
- nltools/algorithms/outliers.py +308 -0
- nltools/algorithms/regression.py +83 -0
- nltools/algorithms/signal.py +303 -0
- nltools/algorithms/similarity.py +234 -0
- nltools/algorithms/validation.py +151 -0
- nltools/cross_validation.py +72 -0
- nltools/data/__init__.py +30 -0
- nltools/data/adjacency/__init__.py +875 -0
- nltools/data/adjacency/io.py +111 -0
- nltools/data/adjacency/modeling.py +569 -0
- nltools/data/adjacency/plotting.py +174 -0
- nltools/data/adjacency/state.py +349 -0
- nltools/data/adjacency/stats.py +596 -0
- nltools/data/adjacency/utils.py +79 -0
- nltools/data/atlases/__init__.py +23 -0
- nltools/data/atlases/labeling.py +158 -0
- nltools/data/atlases/loading.py +76 -0
- nltools/data/atlases/registry.py +96 -0
- nltools/data/atlases/reporting.py +456 -0
- nltools/data/braindata/__init__.py +2170 -0
- nltools/data/braindata/analysis.py +1381 -0
- nltools/data/braindata/bootstrap.py +398 -0
- nltools/data/braindata/io.py +896 -0
- nltools/data/braindata/modeling.py +594 -0
- nltools/data/braindata/plotting.py +501 -0
- nltools/data/braindata/prediction.py +1250 -0
- nltools/data/braindata/utils.py +348 -0
- nltools/data/braindata/validation.py +197 -0
- nltools/data/braindata/viewer.js +266 -0
- nltools/data/braindata/viewer.py +770 -0
- nltools/data/combine.py +27 -0
- nltools/data/designmatrix/__init__.py +1032 -0
- nltools/data/designmatrix/append.py +518 -0
- nltools/data/designmatrix/diagnostics.py +248 -0
- nltools/data/designmatrix/io.py +356 -0
- nltools/data/designmatrix/plotting.py +291 -0
- nltools/data/designmatrix/regressors.py +463 -0
- nltools/data/designmatrix/transforms.py +200 -0
- nltools/data/designmatrix/utils.py +350 -0
- nltools/data/ownership.py +129 -0
- nltools/data/results.py +291 -0
- nltools/data/roc/__init__.py +398 -0
- nltools/data/simulator/__init__.py +927 -0
- nltools/data/simulator/haxby.py +124 -0
- nltools/data/validation.py +83 -0
- nltools/datasets.py +218 -0
- nltools/io/__init__.py +10 -0
- nltools/io/events.py +67 -0
- nltools/io/h5.py +246 -0
- nltools/mask.py +403 -0
- nltools/models/__init__.py +11 -0
- nltools/models/glm.py +543 -0
- nltools/models/results.py +49 -0
- nltools/models/ridge.py +1303 -0
- nltools/models/validation.py +26 -0
- nltools/plotting/__init__.py +32 -0
- nltools/plotting/adjacency.py +421 -0
- nltools/plotting/brain.py +669 -0
- nltools/plotting/decomposition.py +111 -0
- nltools/plotting/prediction.py +110 -0
- nltools/resources/covariates_example.csv +161 -0
- nltools/resources/onsets_example.csv +40 -0
- nltools/templates/__init__.py +51 -0
- nltools/templates/config.py +144 -0
- nltools/templates/fetch.py +260 -0
- nltools/templates/matching.py +183 -0
- nltools/templates/paths.py +106 -0
- nltools/templates/registry.py +25 -0
- nltools/utils.py +230 -0
- nltools/version.py +13 -0
- nltools-0.6.0.dev0.dist-info/METADATA +95 -0
- nltools-0.6.0.dev0.dist-info/RECORD +95 -0
- nltools-0.6.0.dev0.dist-info/WHEEL +4 -0
- nltools-0.6.0.dev0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,347 @@
|
|
|
1
|
+
"""Permutation tests for autocorrelated time series.
|
|
2
|
+
|
|
3
|
+
Shuffling the samples of a time series destroys its autocorrelation and inflates
|
|
4
|
+
false positives. These surrogate-data methods build the null while preserving
|
|
5
|
+
temporal structure:
|
|
6
|
+
|
|
7
|
+
- `circle_shift`: rotate the series (preserves autocorrelation exactly)
|
|
8
|
+
- `phase_randomize`: randomize Fourier phases (preserves the power spectrum)
|
|
9
|
+
- `_timeseries_correlation_permutation_test`: correlation test using either method
|
|
10
|
+
|
|
11
|
+
Permutations run on joblib workers; `n_jobs` sets how many, and a given
|
|
12
|
+
`random_state` gives the same result at any worker count.
|
|
13
|
+
|
|
14
|
+
References:
|
|
15
|
+
Theiler, J., Galdrikian, B., Longtin, A., Eubank, S., & Farmer, J. D. (1991).
|
|
16
|
+
Testing for nonlinearity in time series: the method of surrogate data
|
|
17
|
+
(No. LA-UR-91-3343; CONF-9108181-1). Los Alamos National Lab., NM (United States).
|
|
18
|
+
|
|
19
|
+
Lancaster, G., Iatsenko, D., Pidde, A., Ticcinelli, V., & Stefanovska, A. (2018).
|
|
20
|
+
Surrogate data for hypothesis testing of physical systems. Physics Reports, 748, 1-60.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import numpy as np
|
|
24
|
+
from typing import Literal
|
|
25
|
+
from sklearn.utils import check_random_state
|
|
26
|
+
|
|
27
|
+
from .utils import _maybe_tqdm
|
|
28
|
+
from ..validation import _compute_pvalue, _validate_tail_parameter
|
|
29
|
+
from .correlation import _select_corr_func
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def circle_shift(
|
|
33
|
+
data: np.ndarray,
|
|
34
|
+
shift_amount: int | np.ndarray | None = None,
|
|
35
|
+
random_state: int | np.random.RandomState | None = None,
|
|
36
|
+
) -> np.ndarray:
|
|
37
|
+
"""Circular shift for time-series data.
|
|
38
|
+
|
|
39
|
+
Performs a circular shift that preserves autocorrelation structure.
|
|
40
|
+
Useful for permutation tests on autocorrelated time series (e.g., fMRI).
|
|
41
|
+
For 1D data, shifts by a single amount. For 2D data, shifts each
|
|
42
|
+
feature (column) independently.
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
data (np.ndarray): Time series, shape (n_samples,) or (n_samples, n_features).
|
|
46
|
+
shift_amount (int | np.ndarray | None): Shift amount: an int for 1D data,
|
|
47
|
+
or an array of length n_features (one shift per column) for 2D data.
|
|
48
|
+
None draws random shift(s). Defaults to None.
|
|
49
|
+
random_state (int | np.random.RandomState | None): Random seed used when
|
|
50
|
+
`shift_amount` is None.
|
|
51
|
+
|
|
52
|
+
Returns:
|
|
53
|
+
np.ndarray: Circularly shifted data with the same shape as the input.
|
|
54
|
+
|
|
55
|
+
Examples:
|
|
56
|
+
```python
|
|
57
|
+
x = np.array([1, 2, 3, 4, 5])
|
|
58
|
+
circle_shift(x, shift_amount=2) # → array([4, 5, 1, 2, 3])
|
|
59
|
+
|
|
60
|
+
X = np.array([[1, 10], [2, 20], [3, 30], [4, 40]])
|
|
61
|
+
circle_shift(X, shift_amount=np.array([1, 2]))
|
|
62
|
+
# → array([[ 4, 30],
|
|
63
|
+
# [ 1, 40],
|
|
64
|
+
# [ 2, 10],
|
|
65
|
+
# [ 3, 20]])
|
|
66
|
+
```
|
|
67
|
+
"""
|
|
68
|
+
data = np.asarray(data)
|
|
69
|
+
rng = check_random_state(random_state)
|
|
70
|
+
|
|
71
|
+
# 1D case
|
|
72
|
+
if data.ndim == 1:
|
|
73
|
+
if shift_amount is None:
|
|
74
|
+
shift_amount = rng.randint(1, len(data))
|
|
75
|
+
shift_amount = int(shift_amount)
|
|
76
|
+
return np.concatenate([data[-shift_amount:], data[:-shift_amount]])
|
|
77
|
+
|
|
78
|
+
# 2D case
|
|
79
|
+
if data.ndim == 2:
|
|
80
|
+
n_samples, n_features = data.shape
|
|
81
|
+
if shift_amount is None:
|
|
82
|
+
# Each feature gets an independent random shift (with replacement)
|
|
83
|
+
# This allows n_features > n_samples (e.g., more voxels than timepoints)
|
|
84
|
+
shift_amount = rng.randint(1, n_samples, size=n_features)
|
|
85
|
+
shift_amount = np.asarray(shift_amount, dtype=int)
|
|
86
|
+
|
|
87
|
+
if shift_amount.shape != (n_features,):
|
|
88
|
+
raise ValueError(
|
|
89
|
+
f"shift_amount must have length n_features={n_features}, "
|
|
90
|
+
f"got shape {shift_amount.shape}"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# Shift each feature independently
|
|
94
|
+
shifted = np.empty_like(data)
|
|
95
|
+
for i, shift in enumerate(shift_amount):
|
|
96
|
+
shifted[:, i] = np.concatenate([data[-shift:, i], data[:-shift, i]])
|
|
97
|
+
return shifted
|
|
98
|
+
|
|
99
|
+
raise ValueError(f"data must be 1D or 2D, got shape {data.shape}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def phase_randomize(
|
|
103
|
+
data: np.ndarray,
|
|
104
|
+
*,
|
|
105
|
+
random_state: int | np.random.RandomState | None = None,
|
|
106
|
+
) -> np.ndarray:
|
|
107
|
+
"""FFT-based phase randomization for time-series data.
|
|
108
|
+
|
|
109
|
+
Preserves the power spectrum (and therefore the autocorrelation) exactly, up
|
|
110
|
+
to numerical precision, while destroying nonlinear temporal structure. Used to
|
|
111
|
+
test whether data was generated by a linear Gaussian process or contains
|
|
112
|
+
nonlinear dynamics.
|
|
113
|
+
|
|
114
|
+
The signal is transformed with an FFT, each positive frequency is multiplied
|
|
115
|
+
by `exp(iφ)` with φ drawn uniformly from [0, 2π], the matching negative
|
|
116
|
+
frequency by the conjugate `exp(-iφ)` so the inverse FFT is real, and the
|
|
117
|
+
result is transformed back.
|
|
118
|
+
|
|
119
|
+
Args:
|
|
120
|
+
data (np.ndarray): Time series, shape (n_samples,) or (n_samples, n_features).
|
|
121
|
+
random_state (int | np.random.RandomState | None): Random seed for
|
|
122
|
+
reproducibility.
|
|
123
|
+
|
|
124
|
+
Returns:
|
|
125
|
+
np.ndarray: Phase-randomized data with the same shape as the input.
|
|
126
|
+
|
|
127
|
+
Examples:
|
|
128
|
+
```python
|
|
129
|
+
x = np.sin(np.linspace(0, 10 * np.pi, 100))
|
|
130
|
+
x_rand = phase_randomize(x, random_state=42)
|
|
131
|
+
# Power spectrum is preserved
|
|
132
|
+
np.allclose(np.abs(np.fft.rfft(x)) ** 2, np.abs(np.fft.rfft(x_rand)) ** 2) # → True
|
|
133
|
+
```
|
|
134
|
+
"""
|
|
135
|
+
data = np.asarray(data)
|
|
136
|
+
rng = check_random_state(random_state)
|
|
137
|
+
|
|
138
|
+
# Compute FFT
|
|
139
|
+
fft_data = np.fft.fft(data, axis=0)
|
|
140
|
+
n_samples = data.shape[0]
|
|
141
|
+
|
|
142
|
+
# Determine positive and negative frequency indices
|
|
143
|
+
if n_samples % 2 == 0:
|
|
144
|
+
pos_freq = np.arange(1, n_samples // 2)
|
|
145
|
+
neg_freq = np.arange(n_samples - 1, n_samples // 2, -1)
|
|
146
|
+
else:
|
|
147
|
+
pos_freq = np.arange(1, (n_samples - 1) // 2 + 1)
|
|
148
|
+
neg_freq = np.arange(n_samples - 1, (n_samples - 1) // 2, -1)
|
|
149
|
+
|
|
150
|
+
# Generate random phases and apply to FFT
|
|
151
|
+
if data.ndim == 1:
|
|
152
|
+
phase_shifts = rng.uniform(0, 2 * np.pi, size=len(pos_freq))
|
|
153
|
+
fft_data[pos_freq] *= np.exp(1j * phase_shifts)
|
|
154
|
+
fft_data[neg_freq] *= np.exp(-1j * phase_shifts)
|
|
155
|
+
else:
|
|
156
|
+
n_features = data.shape[1]
|
|
157
|
+
phase_shifts = rng.uniform(0, 2 * np.pi, size=(len(pos_freq), n_features))
|
|
158
|
+
fft_data[pos_freq, :] *= np.exp(1j * phase_shifts)
|
|
159
|
+
fft_data[neg_freq, :] *= np.exp(-1j * phase_shifts)
|
|
160
|
+
|
|
161
|
+
# Inverse FFT and return real part
|
|
162
|
+
return np.real(np.fft.ifft(fft_data, axis=0))
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _timeseries_correlation_cpu_parallel(
|
|
166
|
+
data1: np.ndarray,
|
|
167
|
+
data2: np.ndarray,
|
|
168
|
+
*,
|
|
169
|
+
n_permute: int,
|
|
170
|
+
method: str,
|
|
171
|
+
metric: str,
|
|
172
|
+
tail: int | str,
|
|
173
|
+
return_null: bool,
|
|
174
|
+
n_jobs: int,
|
|
175
|
+
random_state: int | np.random.RandomState | None,
|
|
176
|
+
progress_bar: bool = False,
|
|
177
|
+
) -> dict:
|
|
178
|
+
"""Surrogate-data correlation test parallelized across CPU cores with joblib.
|
|
179
|
+
|
|
180
|
+
Pre-generates one seed per permutation, so the surrogate a permutation sees
|
|
181
|
+
depends only on its index and never on which worker ran it. Only `data1` is
|
|
182
|
+
randomized: randomizing both series reduces power and tests a different
|
|
183
|
+
hypothesis than H0: correlation = 0.
|
|
184
|
+
|
|
185
|
+
Args:
|
|
186
|
+
data1 (np.ndarray): First time series, shape `(n_samples,)`; the series
|
|
187
|
+
the surrogates are built from.
|
|
188
|
+
data2 (np.ndarray): Second time series, shape `(n_samples,)`; held fixed.
|
|
189
|
+
n_permute (int): Number of permutations.
|
|
190
|
+
method (str): `'circle_shift'` or `'phase_randomize'`.
|
|
191
|
+
metric (str): `'pearson'`, `'spearman'`, or `'kendall'`.
|
|
192
|
+
tail (int | str): `2` or `'two'` for two-tailed; `1` or `'one'` for
|
|
193
|
+
one-tailed.
|
|
194
|
+
return_null (bool): Whether to return the null distribution.
|
|
195
|
+
n_jobs (int): Number of joblib workers (-1 = all cores).
|
|
196
|
+
random_state (int | np.random.RandomState | None): Random seed for
|
|
197
|
+
reproducibility.
|
|
198
|
+
progress_bar (bool): Whether to display a tqdm progress bar.
|
|
199
|
+
|
|
200
|
+
Returns:
|
|
201
|
+
dict: Same format as `_timeseries_correlation_permutation_test`.
|
|
202
|
+
"""
|
|
203
|
+
from joblib import Parallel, delayed
|
|
204
|
+
|
|
205
|
+
corr_func = _select_corr_func(metric)
|
|
206
|
+
|
|
207
|
+
obs_corr = corr_func(data1, data2)
|
|
208
|
+
if isinstance(obs_corr, np.ndarray):
|
|
209
|
+
obs_corr = obs_corr[0]
|
|
210
|
+
obs_corr = np.asarray(obs_corr)
|
|
211
|
+
|
|
212
|
+
rng = check_random_state(random_state)
|
|
213
|
+
MAX_INT = 2**31 - 1
|
|
214
|
+
seeds = rng.randint(MAX_INT, size=n_permute)
|
|
215
|
+
|
|
216
|
+
surrogate = circle_shift if method == "circle_shift" else phase_randomize
|
|
217
|
+
|
|
218
|
+
def _compute_one_perm(seed):
|
|
219
|
+
"""Correlate one surrogate of `data1` against the fixed `data2`."""
|
|
220
|
+
corr = corr_func(surrogate(data1, random_state=seed), data2)
|
|
221
|
+
return corr[0] if isinstance(corr, np.ndarray) else corr
|
|
222
|
+
|
|
223
|
+
null_dist = Parallel(n_jobs=n_jobs)(
|
|
224
|
+
delayed(_compute_one_perm)(seeds[i])
|
|
225
|
+
for i in _maybe_tqdm(
|
|
226
|
+
range(n_permute),
|
|
227
|
+
progress_bar=progress_bar,
|
|
228
|
+
desc=f"{method} perms",
|
|
229
|
+
unit="perm",
|
|
230
|
+
)
|
|
231
|
+
)
|
|
232
|
+
null_dist = np.array(null_dist)
|
|
233
|
+
|
|
234
|
+
p_value = _compute_pvalue(obs_corr, null_dist, tail=tail)
|
|
235
|
+
|
|
236
|
+
results = {
|
|
237
|
+
"correlation": float(obs_corr),
|
|
238
|
+
"p": p_value.item() if hasattr(p_value, "item") else float(p_value),
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
if return_null:
|
|
242
|
+
results["null_dist"] = null_dist
|
|
243
|
+
|
|
244
|
+
return results
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _timeseries_correlation_permutation_test(
|
|
248
|
+
data1: np.ndarray,
|
|
249
|
+
data2: np.ndarray,
|
|
250
|
+
*,
|
|
251
|
+
method: Literal["circle_shift", "phase_randomize"] = "circle_shift",
|
|
252
|
+
n_permute: int = 5000,
|
|
253
|
+
metric: Literal["pearson", "spearman", "kendall"] = "pearson",
|
|
254
|
+
tail: int | str = 2,
|
|
255
|
+
n_jobs: int = -1,
|
|
256
|
+
return_null: bool = False,
|
|
257
|
+
random_state: int | np.random.RandomState | None = None,
|
|
258
|
+
progress_bar: bool = False,
|
|
259
|
+
) -> dict:
|
|
260
|
+
"""Permutation test for the correlation between two autocorrelated time series.
|
|
261
|
+
|
|
262
|
+
Standard permutation tests shuffle samples independently, which destroys
|
|
263
|
+
autocorrelation and inflates Type I error on time series. This test instead
|
|
264
|
+
builds the null from surrogates of `data1` that preserve temporal structure
|
|
265
|
+
— circular shifts (`'circle_shift'`) or Fourier phase randomization
|
|
266
|
+
(`'phase_randomize'`) — while `data2` stays fixed. For independent
|
|
267
|
+
observations use `correlation_permutation_test`.
|
|
268
|
+
|
|
269
|
+
Args:
|
|
270
|
+
data1 (np.ndarray): First time series, shape (n_samples,) or (n_samples, 1).
|
|
271
|
+
data2 (np.ndarray): Second time series, shape (n_samples,) or (n_samples, 1).
|
|
272
|
+
method (str): `'circle_shift'` rotates the series (preserves
|
|
273
|
+
autocorrelation; fast, and suitable for most fMRI time series);
|
|
274
|
+
`'phase_randomize'` randomizes Fourier phases (preserves the power
|
|
275
|
+
spectrum exactly; tests for nonlinear structure). Defaults to
|
|
276
|
+
'circle_shift'.
|
|
277
|
+
n_permute (int): Number of permutations. Defaults to 5000.
|
|
278
|
+
metric (str): Correlation type, one of 'pearson', 'spearman', or
|
|
279
|
+
'kendall'. Defaults to 'pearson'.
|
|
280
|
+
tail (int | str): `2` or `'two'` for a two-tailed test; `1` or `'one'` for
|
|
281
|
+
a one-tailed test in the positive direction (negate one series for the
|
|
282
|
+
other direction). Defaults to 2.
|
|
283
|
+
n_jobs (int): Number of joblib workers, -1 = all cores. Defaults to -1.
|
|
284
|
+
Results are identical at every worker count.
|
|
285
|
+
return_null (bool): Also return the null distribution. Defaults to False.
|
|
286
|
+
random_state (int | np.random.RandomState | None): Random seed for
|
|
287
|
+
reproducibility.
|
|
288
|
+
progress_bar (bool): Show a progress bar over permutations. Defaults to
|
|
289
|
+
False.
|
|
290
|
+
|
|
291
|
+
Returns:
|
|
292
|
+
dict: Keys 'correlation' (float, observed correlation), 'p' (float), and
|
|
293
|
+
'null_dist' (np.ndarray of shape (n_permute,)) when
|
|
294
|
+
`return_null=True`.
|
|
295
|
+
|
|
296
|
+
Examples:
|
|
297
|
+
```python
|
|
298
|
+
import numpy as np
|
|
299
|
+
from nltools.algorithms import _timeseries_correlation_permutation_test
|
|
300
|
+
|
|
301
|
+
rng = np.random.default_rng(0)
|
|
302
|
+
x = np.sin(np.linspace(0, 10 * np.pi, 100)) # strongly autocorrelated
|
|
303
|
+
y = x + rng.standard_normal(100) * 0.5
|
|
304
|
+
result = _timeseries_correlation_permutation_test(
|
|
305
|
+
x, y, method="circle_shift", n_permute=1000, random_state=42
|
|
306
|
+
)
|
|
307
|
+
result["correlation"] # → 0.853
|
|
308
|
+
result["p"] # → 0.078 — the autocorrelation-aware null is far wider
|
|
309
|
+
# than a sample-shuffling null would be
|
|
310
|
+
```
|
|
311
|
+
"""
|
|
312
|
+
# Validate tail up front (like one_sample/two_sample/matrix) so an invalid
|
|
313
|
+
# value fails immediately rather than after every permutation has run.
|
|
314
|
+
_validate_tail_parameter(tail)
|
|
315
|
+
|
|
316
|
+
# Validate inputs
|
|
317
|
+
data1 = np.asarray(data1).squeeze()
|
|
318
|
+
data2 = np.asarray(data2).squeeze()
|
|
319
|
+
|
|
320
|
+
if data1.ndim != 1 or data2.ndim != 1:
|
|
321
|
+
raise ValueError("data1 and data2 must be 1D arrays")
|
|
322
|
+
|
|
323
|
+
if len(data1) != len(data2):
|
|
324
|
+
raise ValueError("data1 and data2 must have the same length")
|
|
325
|
+
|
|
326
|
+
if method not in ["circle_shift", "phase_randomize"]:
|
|
327
|
+
raise ValueError(
|
|
328
|
+
f"method must be 'circle_shift' or 'phase_randomize', got '{method}'"
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
if metric not in ["pearson", "spearman", "kendall"]:
|
|
332
|
+
raise ValueError(
|
|
333
|
+
f"metric must be 'pearson', 'spearman', or 'kendall', got '{metric}'"
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
return _timeseries_correlation_cpu_parallel(
|
|
337
|
+
data1,
|
|
338
|
+
data2,
|
|
339
|
+
n_permute=n_permute,
|
|
340
|
+
method=method,
|
|
341
|
+
metric=metric,
|
|
342
|
+
tail=tail,
|
|
343
|
+
return_null=return_null,
|
|
344
|
+
n_jobs=n_jobs,
|
|
345
|
+
random_state=random_state,
|
|
346
|
+
progress_bar=progress_bar,
|
|
347
|
+
)
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Two-sample permutation test (group-label shuffling).
|
|
2
|
+
|
|
3
|
+
Tests whether two independent groups differ in mean by randomly reassigning
|
|
4
|
+
observations to groups — the permutation analogue of an independent-samples
|
|
5
|
+
t-test. Permutations run on joblib workers; `n_jobs` sets how many, and a
|
|
6
|
+
given `random_state` gives the same result at any worker count.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
|
|
11
|
+
from .utils import _maybe_tqdm
|
|
12
|
+
from .validation import _validate_array_shape_range
|
|
13
|
+
from ..validation import _compute_pvalue, _validate_tail_parameter
|
|
14
|
+
from .random import _generate_seeds
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _two_sample_permutation_cpu_parallel(
|
|
18
|
+
data1: np.ndarray,
|
|
19
|
+
data2: np.ndarray,
|
|
20
|
+
*,
|
|
21
|
+
n_permute: int,
|
|
22
|
+
tail: int,
|
|
23
|
+
return_null: bool,
|
|
24
|
+
n_jobs: int,
|
|
25
|
+
random_state: int | None,
|
|
26
|
+
single_feature: bool = False,
|
|
27
|
+
progress_bar: bool = False,
|
|
28
|
+
) -> dict:
|
|
29
|
+
"""Two-sample permutation test parallelized across CPU cores with joblib.
|
|
30
|
+
|
|
31
|
+
Each worker shuffles the group labels for one permutation (from its own
|
|
32
|
+
pre-drawn seed) and computes the mean difference, so results are
|
|
33
|
+
reproducible regardless of worker count.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
data1 (np.ndarray): Group 1 data, shape `(n_samples1, n_features)`.
|
|
37
|
+
data2 (np.ndarray): Group 2 data, shape `(n_samples2, n_features)`.
|
|
38
|
+
n_permute (int): Number of permutations.
|
|
39
|
+
tail (int | str): `2` or `'two'` for two-tailed; `1` or `'one'` for
|
|
40
|
+
one-tailed.
|
|
41
|
+
return_null (bool): Whether to return the null distribution.
|
|
42
|
+
n_jobs (int): Number of parallel jobs (-1 = all cores).
|
|
43
|
+
random_state (int | None): Random seed for reproducibility.
|
|
44
|
+
single_feature (bool): Whether the caller passed 1D data (results are
|
|
45
|
+
returned as scalars).
|
|
46
|
+
progress_bar (bool): Whether to display a tqdm progress bar.
|
|
47
|
+
|
|
48
|
+
Returns:
|
|
49
|
+
dict: Same format as `two_sample_permutation_test`.
|
|
50
|
+
"""
|
|
51
|
+
from joblib import Parallel, delayed
|
|
52
|
+
|
|
53
|
+
# Setup random state and generate seeds for workers
|
|
54
|
+
seeds = _generate_seeds(n_permute, random_state=random_state)
|
|
55
|
+
|
|
56
|
+
# Get dimensions (data already reshaped by caller)
|
|
57
|
+
n1, n_features = data1.shape
|
|
58
|
+
n2 = data2.shape[0]
|
|
59
|
+
n_total = n1 + n2
|
|
60
|
+
|
|
61
|
+
# Compute observed mean difference
|
|
62
|
+
obs_diff = np.nanmean(data1, axis=0) - np.nanmean(data2, axis=0)
|
|
63
|
+
|
|
64
|
+
# Concatenate data for permutation
|
|
65
|
+
combined = np.vstack([data1, data2]) # (n_total, n_features)
|
|
66
|
+
|
|
67
|
+
# Define worker function (each processes ONE permutation)
|
|
68
|
+
def _compute_one_perm(seed):
|
|
69
|
+
"""Compute mean difference for one group permutation."""
|
|
70
|
+
perm_rng = np.random.RandomState(seed)
|
|
71
|
+
# Randomly shuffle indices
|
|
72
|
+
indices = perm_rng.permutation(n_total)
|
|
73
|
+
# Split into two groups
|
|
74
|
+
group1_indices = indices[:n1]
|
|
75
|
+
group2_indices = indices[n1:]
|
|
76
|
+
# Compute mean difference
|
|
77
|
+
mean1 = np.nanmean(combined[group1_indices], axis=0)
|
|
78
|
+
mean2 = np.nanmean(combined[group2_indices], axis=0)
|
|
79
|
+
return mean1 - mean2
|
|
80
|
+
|
|
81
|
+
# Execute in parallel with progress bar
|
|
82
|
+
null_dist = Parallel(n_jobs=n_jobs)(
|
|
83
|
+
delayed(_compute_one_perm)(seeds[i])
|
|
84
|
+
for i in _maybe_tqdm(
|
|
85
|
+
range(n_permute),
|
|
86
|
+
progress_bar=progress_bar,
|
|
87
|
+
desc="CPU parallel perms",
|
|
88
|
+
unit="perm",
|
|
89
|
+
)
|
|
90
|
+
)
|
|
91
|
+
null_dist = np.array(null_dist) # Shape: (n_permute, n_features)
|
|
92
|
+
|
|
93
|
+
# Compute p-values
|
|
94
|
+
p_values = _compute_pvalue(obs_diff, null_dist, tail=tail)
|
|
95
|
+
|
|
96
|
+
# Return to original shape
|
|
97
|
+
if single_feature:
|
|
98
|
+
obs_diff = obs_diff.item() if hasattr(obs_diff, "item") else float(obs_diff[0])
|
|
99
|
+
p_values = p_values.item() if hasattr(p_values, "item") else float(p_values[0])
|
|
100
|
+
|
|
101
|
+
# Build result
|
|
102
|
+
result = {
|
|
103
|
+
"mean_diff": obs_diff,
|
|
104
|
+
"p": p_values,
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
if return_null:
|
|
108
|
+
if single_feature:
|
|
109
|
+
null_dist = null_dist.squeeze()
|
|
110
|
+
result["null_dist"] = null_dist
|
|
111
|
+
|
|
112
|
+
return result
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def two_sample_permutation_test(
|
|
116
|
+
data1: np.ndarray,
|
|
117
|
+
data2: np.ndarray,
|
|
118
|
+
*,
|
|
119
|
+
n_permute: int = 5000,
|
|
120
|
+
tail: int | str = 2,
|
|
121
|
+
return_null: bool = False,
|
|
122
|
+
n_jobs: int = -1,
|
|
123
|
+
random_state: int | None = None,
|
|
124
|
+
progress_bar: bool = False,
|
|
125
|
+
) -> dict:
|
|
126
|
+
"""Two-sample permutation test using group-label shuffling.
|
|
127
|
+
|
|
128
|
+
Tests whether two independent groups have different means by randomly
|
|
129
|
+
reassigning observations to groups — the permutation analogue of an
|
|
130
|
+
independent-samples t-test. Group sizes may differ. Multi-feature
|
|
131
|
+
(voxel-wise) data tests each column independently against the same
|
|
132
|
+
permutations.
|
|
133
|
+
|
|
134
|
+
Assumes exchangeability under the null (group assignment is arbitrary):
|
|
135
|
+
independent samples from similarly shaped distributions. NaN observations
|
|
136
|
+
are dropped from the observed and every permuted mean (`np.nanmean`),
|
|
137
|
+
feature by feature.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
data1 (np.ndarray): Group 1 data, shape `(n_samples1,)` for a single
|
|
141
|
+
feature or `(n_samples1, n_features)` for voxel-wise data. May
|
|
142
|
+
contain NaN observations.
|
|
143
|
+
data2 (np.ndarray): Group 2 data, shape `(n_samples2,)` or
|
|
144
|
+
`(n_samples2, n_features)`; must have the same number of features
|
|
145
|
+
as `data1`. May contain NaN observations.
|
|
146
|
+
n_permute (int): Number of permutations. Defaults to 5000.
|
|
147
|
+
tail (int | str): `2` or `'two'` (default) for a two-tailed test
|
|
148
|
+
(mean1 != mean2); `1` or `'one'` for a one-tailed test of
|
|
149
|
+
mean1 > mean2 (swap the groups for the other direction — the fixed
|
|
150
|
+
direction keeps multiple-comparison correction valid).
|
|
151
|
+
return_null (bool): If True, include the full null distribution in the
|
|
152
|
+
result. Defaults to False.
|
|
153
|
+
n_jobs (int): Number of joblib workers. Defaults to -1 (all cores).
|
|
154
|
+
Results are identical at every worker count.
|
|
155
|
+
random_state (int | None): Random seed for reproducibility.
|
|
156
|
+
progress_bar (bool): Whether to display a progress bar. Defaults to False.
|
|
157
|
+
|
|
158
|
+
Returns:
|
|
159
|
+
dict: Keys `'mean_diff'` (float or np.ndarray, observed
|
|
160
|
+
`mean(data1) - mean(data2)`), `'p'` (float or np.ndarray,
|
|
161
|
+
p-value(s)), and — when `return_null=True` — `'null_dist'`
|
|
162
|
+
(np.ndarray, shape `(n_permute,)` or `(n_permute, n_features)`).
|
|
163
|
+
|
|
164
|
+
Examples:
|
|
165
|
+
```python
|
|
166
|
+
# Single feature
|
|
167
|
+
data1 = np.random.randn(20) # Group 1: 20 subjects
|
|
168
|
+
data2 = np.random.randn(25) # Group 2: 25 subjects
|
|
169
|
+
result = two_sample_permutation_test(data1, data2, n_permute=5000)
|
|
170
|
+
result["p"] # → 0.45
|
|
171
|
+
|
|
172
|
+
# Voxel-wise test
|
|
173
|
+
data1 = np.random.randn(20, 10000) # 20 subjects, 10K voxels
|
|
174
|
+
data2 = np.random.randn(25, 10000) # 25 subjects, 10K voxels
|
|
175
|
+
result = two_sample_permutation_test(data1, data2, n_permute=5000)
|
|
176
|
+
result["mean_diff"].shape # → (10000,)
|
|
177
|
+
result["p"].shape # → (10000,)
|
|
178
|
+
```
|
|
179
|
+
"""
|
|
180
|
+
# Input validation
|
|
181
|
+
data1 = np.asarray(data1, dtype=np.float64)
|
|
182
|
+
data2 = np.asarray(data2, dtype=np.float64)
|
|
183
|
+
|
|
184
|
+
_validate_array_shape_range(data1, 1, 2, name="data1")
|
|
185
|
+
_validate_array_shape_range(data2, 1, 2, name="data2")
|
|
186
|
+
_validate_tail_parameter(tail)
|
|
187
|
+
|
|
188
|
+
# Handle shape
|
|
189
|
+
single_feature = data1.ndim == 1 and data2.ndim == 1
|
|
190
|
+
if data1.ndim == 1:
|
|
191
|
+
data1 = data1[:, np.newaxis]
|
|
192
|
+
if data2.ndim == 1:
|
|
193
|
+
data2 = data2[:, np.newaxis]
|
|
194
|
+
|
|
195
|
+
# Check feature dimensions match
|
|
196
|
+
if data1.shape[1] != data2.shape[1]:
|
|
197
|
+
raise ValueError(
|
|
198
|
+
f"data1 and data2 must have same number of features, "
|
|
199
|
+
f"got {data1.shape[1]} and {data2.shape[1]}"
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
return _two_sample_permutation_cpu_parallel(
|
|
203
|
+
data1,
|
|
204
|
+
data2,
|
|
205
|
+
n_permute=n_permute,
|
|
206
|
+
tail=tail,
|
|
207
|
+
return_null=return_null,
|
|
208
|
+
n_jobs=n_jobs,
|
|
209
|
+
random_state=random_state,
|
|
210
|
+
single_feature=single_feature,
|
|
211
|
+
progress_bar=progress_bar,
|
|
212
|
+
)
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Shared helpers for the permutation tests: sign flips, z-from-p, the stability epsilon and progress bars."""
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
from ...utils import _NullProgressBar, _make_progress_bar, _maybe_tqdm # noqa: F401
|
|
5
|
+
from .random import _generate_sign_flips as _generate_sign_flips # noqa: F401
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
# ============================================================================
|
|
9
|
+
# Numerical Stability Constants
|
|
10
|
+
# ============================================================================
|
|
11
|
+
|
|
12
|
+
# Small constant added to denominators to prevent division by zero
|
|
13
|
+
# Value: 1e-10 is standard in scientific computing for float64 precision
|
|
14
|
+
# - Well above machine epsilon (2.22e-16 for float64)
|
|
15
|
+
# - Small enough not to affect correlation values
|
|
16
|
+
# - Matches established practice in neuroimaging libraries
|
|
17
|
+
EPSILON = 1e-10
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _signed_z_from_p(t_like_arr, p_arr, tail_internal: str = "two") -> np.ndarray:
|
|
21
|
+
"""Compute a signed z-score map from a p-value map.
|
|
22
|
+
|
|
23
|
+
The z-from-p conversion used by `BrainData.ttest` and `Adjacency.ttest`.
|
|
24
|
+
The clipping policy below must live in exactly one place.
|
|
25
|
+
|
|
26
|
+
Two-tailed p: ``|z| = norm.isf(p/2)`` so that p=0.05 → |z|≈1.96, matching
|
|
27
|
+
nilearn's ``output_type='z_score'`` convention, with the sign copied from
|
|
28
|
+
the accompanying statistic. One-tailed (upper) p: ``z = norm.isf(p)`` —
|
|
29
|
+
a one-sided p already encodes direction, so no sign copy is needed.
|
|
30
|
+
|
|
31
|
+
p is clipped to the open interval (0, 1) so z stays finite in BOTH
|
|
32
|
+
directions: the lower bound is the smallest positive normal float64
|
|
33
|
+
(z ≈ +37.7), the upper bound the largest float64 below 1 (z ≈ -8.3).
|
|
34
|
+
The bounds look asymmetric because float64 resolves p near 0 far more
|
|
35
|
+
finely than near 1; both sit beyond any meaningful statistical
|
|
36
|
+
resolution. Without the upper clip, ``p == 1.0`` — reachable on the
|
|
37
|
+
one-tailed path, e.g. ``t.sf(-45, 29) == 1.0`` — maps to ``-inf`` and
|
|
38
|
+
poisons downstream percentiles and plotting.
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
t_like_arr (np.ndarray): Statistic supplying the sign (t map or similar).
|
|
42
|
+
p_arr (np.ndarray): P-value map (two-tailed, or one-tailed upper).
|
|
43
|
+
tail_internal (str): `'two'` (default) or `'upper'`.
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
np.ndarray: Signed z map, finite everywhere.
|
|
47
|
+
"""
|
|
48
|
+
from scipy.stats import norm
|
|
49
|
+
|
|
50
|
+
p_clipped = np.clip(
|
|
51
|
+
np.asarray(p_arr, dtype=np.float64),
|
|
52
|
+
np.finfo(np.float64).tiny,
|
|
53
|
+
np.nextafter(1.0, 0.0),
|
|
54
|
+
)
|
|
55
|
+
if tail_internal == "upper":
|
|
56
|
+
return norm.isf(p_clipped)
|
|
57
|
+
z_abs = norm.isf(p_clipped / 2.0)
|
|
58
|
+
return np.sign(np.asarray(t_like_arr)) * z_abs
|