nltools 0.6.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. nltools/__init__.py +55 -0
  2. nltools/algorithms/__init__.py +90 -0
  3. nltools/algorithms/alignment/__init__.py +21 -0
  4. nltools/algorithms/alignment/procrustes.py +565 -0
  5. nltools/algorithms/alignment/srm.py +758 -0
  6. nltools/algorithms/backends.py +1059 -0
  7. nltools/algorithms/corrections.py +177 -0
  8. nltools/algorithms/decoding.py +327 -0
  9. nltools/algorithms/inference/__init__.py +50 -0
  10. nltools/algorithms/inference/bootstrap.py +1386 -0
  11. nltools/algorithms/inference/correlation.py +373 -0
  12. nltools/algorithms/inference/intersubject.py +422 -0
  13. nltools/algorithms/inference/isc.py +1554 -0
  14. nltools/algorithms/inference/matrix.py +602 -0
  15. nltools/algorithms/inference/one_sample.py +288 -0
  16. nltools/algorithms/inference/random.py +122 -0
  17. nltools/algorithms/inference/timeseries.py +347 -0
  18. nltools/algorithms/inference/two_sample.py +212 -0
  19. nltools/algorithms/inference/utils.py +58 -0
  20. nltools/algorithms/inference/validation.py +282 -0
  21. nltools/algorithms/neighborhoods.py +207 -0
  22. nltools/algorithms/outliers.py +308 -0
  23. nltools/algorithms/regression.py +83 -0
  24. nltools/algorithms/signal.py +303 -0
  25. nltools/algorithms/similarity.py +234 -0
  26. nltools/algorithms/validation.py +151 -0
  27. nltools/cross_validation.py +72 -0
  28. nltools/data/__init__.py +30 -0
  29. nltools/data/adjacency/__init__.py +875 -0
  30. nltools/data/adjacency/io.py +111 -0
  31. nltools/data/adjacency/modeling.py +569 -0
  32. nltools/data/adjacency/plotting.py +174 -0
  33. nltools/data/adjacency/state.py +349 -0
  34. nltools/data/adjacency/stats.py +596 -0
  35. nltools/data/adjacency/utils.py +79 -0
  36. nltools/data/atlases/__init__.py +23 -0
  37. nltools/data/atlases/labeling.py +158 -0
  38. nltools/data/atlases/loading.py +76 -0
  39. nltools/data/atlases/registry.py +96 -0
  40. nltools/data/atlases/reporting.py +456 -0
  41. nltools/data/braindata/__init__.py +2170 -0
  42. nltools/data/braindata/analysis.py +1381 -0
  43. nltools/data/braindata/bootstrap.py +398 -0
  44. nltools/data/braindata/io.py +896 -0
  45. nltools/data/braindata/modeling.py +594 -0
  46. nltools/data/braindata/plotting.py +501 -0
  47. nltools/data/braindata/prediction.py +1250 -0
  48. nltools/data/braindata/utils.py +348 -0
  49. nltools/data/braindata/validation.py +197 -0
  50. nltools/data/braindata/viewer.js +266 -0
  51. nltools/data/braindata/viewer.py +770 -0
  52. nltools/data/combine.py +27 -0
  53. nltools/data/designmatrix/__init__.py +1032 -0
  54. nltools/data/designmatrix/append.py +518 -0
  55. nltools/data/designmatrix/diagnostics.py +248 -0
  56. nltools/data/designmatrix/io.py +356 -0
  57. nltools/data/designmatrix/plotting.py +291 -0
  58. nltools/data/designmatrix/regressors.py +463 -0
  59. nltools/data/designmatrix/transforms.py +200 -0
  60. nltools/data/designmatrix/utils.py +350 -0
  61. nltools/data/ownership.py +129 -0
  62. nltools/data/results.py +291 -0
  63. nltools/data/roc/__init__.py +398 -0
  64. nltools/data/simulator/__init__.py +927 -0
  65. nltools/data/simulator/haxby.py +124 -0
  66. nltools/data/validation.py +83 -0
  67. nltools/datasets.py +218 -0
  68. nltools/io/__init__.py +10 -0
  69. nltools/io/events.py +67 -0
  70. nltools/io/h5.py +246 -0
  71. nltools/mask.py +403 -0
  72. nltools/models/__init__.py +11 -0
  73. nltools/models/glm.py +543 -0
  74. nltools/models/results.py +49 -0
  75. nltools/models/ridge.py +1303 -0
  76. nltools/models/validation.py +26 -0
  77. nltools/plotting/__init__.py +32 -0
  78. nltools/plotting/adjacency.py +421 -0
  79. nltools/plotting/brain.py +669 -0
  80. nltools/plotting/decomposition.py +111 -0
  81. nltools/plotting/prediction.py +110 -0
  82. nltools/resources/covariates_example.csv +161 -0
  83. nltools/resources/onsets_example.csv +40 -0
  84. nltools/templates/__init__.py +51 -0
  85. nltools/templates/config.py +144 -0
  86. nltools/templates/fetch.py +260 -0
  87. nltools/templates/matching.py +183 -0
  88. nltools/templates/paths.py +106 -0
  89. nltools/templates/registry.py +25 -0
  90. nltools/utils.py +230 -0
  91. nltools/version.py +13 -0
  92. nltools-0.6.0.dev0.dist-info/METADATA +95 -0
  93. nltools-0.6.0.dev0.dist-info/RECORD +95 -0
  94. nltools-0.6.0.dev0.dist-info/WHEEL +4 -0
  95. nltools-0.6.0.dev0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,347 @@
1
+ """Permutation tests for autocorrelated time series.
2
+
3
+ Shuffling the samples of a time series destroys its autocorrelation and inflates
4
+ false positives. These surrogate-data methods build the null while preserving
5
+ temporal structure:
6
+
7
+ - `circle_shift`: rotate the series (preserves autocorrelation exactly)
8
+ - `phase_randomize`: randomize Fourier phases (preserves the power spectrum)
9
+ - `_timeseries_correlation_permutation_test`: correlation test using either method
10
+
11
+ Permutations run on joblib workers; `n_jobs` sets how many, and a given
12
+ `random_state` gives the same result at any worker count.
13
+
14
+ References:
15
+ Theiler, J., Galdrikian, B., Longtin, A., Eubank, S., & Farmer, J. D. (1991).
16
+ Testing for nonlinearity in time series: the method of surrogate data
17
+ (No. LA-UR-91-3343; CONF-9108181-1). Los Alamos National Lab., NM (United States).
18
+
19
+ Lancaster, G., Iatsenko, D., Pidde, A., Ticcinelli, V., & Stefanovska, A. (2018).
20
+ Surrogate data for hypothesis testing of physical systems. Physics Reports, 748, 1-60.
21
+ """
22
+
23
+ import numpy as np
24
+ from typing import Literal
25
+ from sklearn.utils import check_random_state
26
+
27
+ from .utils import _maybe_tqdm
28
+ from ..validation import _compute_pvalue, _validate_tail_parameter
29
+ from .correlation import _select_corr_func
30
+
31
+
32
+ def circle_shift(
33
+ data: np.ndarray,
34
+ shift_amount: int | np.ndarray | None = None,
35
+ random_state: int | np.random.RandomState | None = None,
36
+ ) -> np.ndarray:
37
+ """Circular shift for time-series data.
38
+
39
+ Performs a circular shift that preserves autocorrelation structure.
40
+ Useful for permutation tests on autocorrelated time series (e.g., fMRI).
41
+ For 1D data, shifts by a single amount. For 2D data, shifts each
42
+ feature (column) independently.
43
+
44
+ Args:
45
+ data (np.ndarray): Time series, shape (n_samples,) or (n_samples, n_features).
46
+ shift_amount (int | np.ndarray | None): Shift amount: an int for 1D data,
47
+ or an array of length n_features (one shift per column) for 2D data.
48
+ None draws random shift(s). Defaults to None.
49
+ random_state (int | np.random.RandomState | None): Random seed used when
50
+ `shift_amount` is None.
51
+
52
+ Returns:
53
+ np.ndarray: Circularly shifted data with the same shape as the input.
54
+
55
+ Examples:
56
+ ```python
57
+ x = np.array([1, 2, 3, 4, 5])
58
+ circle_shift(x, shift_amount=2) # → array([4, 5, 1, 2, 3])
59
+
60
+ X = np.array([[1, 10], [2, 20], [3, 30], [4, 40]])
61
+ circle_shift(X, shift_amount=np.array([1, 2]))
62
+ # → array([[ 4, 30],
63
+ # [ 1, 40],
64
+ # [ 2, 10],
65
+ # [ 3, 20]])
66
+ ```
67
+ """
68
+ data = np.asarray(data)
69
+ rng = check_random_state(random_state)
70
+
71
+ # 1D case
72
+ if data.ndim == 1:
73
+ if shift_amount is None:
74
+ shift_amount = rng.randint(1, len(data))
75
+ shift_amount = int(shift_amount)
76
+ return np.concatenate([data[-shift_amount:], data[:-shift_amount]])
77
+
78
+ # 2D case
79
+ if data.ndim == 2:
80
+ n_samples, n_features = data.shape
81
+ if shift_amount is None:
82
+ # Each feature gets an independent random shift (with replacement)
83
+ # This allows n_features > n_samples (e.g., more voxels than timepoints)
84
+ shift_amount = rng.randint(1, n_samples, size=n_features)
85
+ shift_amount = np.asarray(shift_amount, dtype=int)
86
+
87
+ if shift_amount.shape != (n_features,):
88
+ raise ValueError(
89
+ f"shift_amount must have length n_features={n_features}, "
90
+ f"got shape {shift_amount.shape}"
91
+ )
92
+
93
+ # Shift each feature independently
94
+ shifted = np.empty_like(data)
95
+ for i, shift in enumerate(shift_amount):
96
+ shifted[:, i] = np.concatenate([data[-shift:, i], data[:-shift, i]])
97
+ return shifted
98
+
99
+ raise ValueError(f"data must be 1D or 2D, got shape {data.shape}")
100
+
101
+
102
+ def phase_randomize(
103
+ data: np.ndarray,
104
+ *,
105
+ random_state: int | np.random.RandomState | None = None,
106
+ ) -> np.ndarray:
107
+ """FFT-based phase randomization for time-series data.
108
+
109
+ Preserves the power spectrum (and therefore the autocorrelation) exactly, up
110
+ to numerical precision, while destroying nonlinear temporal structure. Used to
111
+ test whether data was generated by a linear Gaussian process or contains
112
+ nonlinear dynamics.
113
+
114
+ The signal is transformed with an FFT, each positive frequency is multiplied
115
+ by `exp(iφ)` with φ drawn uniformly from [0, 2π], the matching negative
116
+ frequency by the conjugate `exp(-iφ)` so the inverse FFT is real, and the
117
+ result is transformed back.
118
+
119
+ Args:
120
+ data (np.ndarray): Time series, shape (n_samples,) or (n_samples, n_features).
121
+ random_state (int | np.random.RandomState | None): Random seed for
122
+ reproducibility.
123
+
124
+ Returns:
125
+ np.ndarray: Phase-randomized data with the same shape as the input.
126
+
127
+ Examples:
128
+ ```python
129
+ x = np.sin(np.linspace(0, 10 * np.pi, 100))
130
+ x_rand = phase_randomize(x, random_state=42)
131
+ # Power spectrum is preserved
132
+ np.allclose(np.abs(np.fft.rfft(x)) ** 2, np.abs(np.fft.rfft(x_rand)) ** 2) # → True
133
+ ```
134
+ """
135
+ data = np.asarray(data)
136
+ rng = check_random_state(random_state)
137
+
138
+ # Compute FFT
139
+ fft_data = np.fft.fft(data, axis=0)
140
+ n_samples = data.shape[0]
141
+
142
+ # Determine positive and negative frequency indices
143
+ if n_samples % 2 == 0:
144
+ pos_freq = np.arange(1, n_samples // 2)
145
+ neg_freq = np.arange(n_samples - 1, n_samples // 2, -1)
146
+ else:
147
+ pos_freq = np.arange(1, (n_samples - 1) // 2 + 1)
148
+ neg_freq = np.arange(n_samples - 1, (n_samples - 1) // 2, -1)
149
+
150
+ # Generate random phases and apply to FFT
151
+ if data.ndim == 1:
152
+ phase_shifts = rng.uniform(0, 2 * np.pi, size=len(pos_freq))
153
+ fft_data[pos_freq] *= np.exp(1j * phase_shifts)
154
+ fft_data[neg_freq] *= np.exp(-1j * phase_shifts)
155
+ else:
156
+ n_features = data.shape[1]
157
+ phase_shifts = rng.uniform(0, 2 * np.pi, size=(len(pos_freq), n_features))
158
+ fft_data[pos_freq, :] *= np.exp(1j * phase_shifts)
159
+ fft_data[neg_freq, :] *= np.exp(-1j * phase_shifts)
160
+
161
+ # Inverse FFT and return real part
162
+ return np.real(np.fft.ifft(fft_data, axis=0))
163
+
164
+
165
+ def _timeseries_correlation_cpu_parallel(
166
+ data1: np.ndarray,
167
+ data2: np.ndarray,
168
+ *,
169
+ n_permute: int,
170
+ method: str,
171
+ metric: str,
172
+ tail: int | str,
173
+ return_null: bool,
174
+ n_jobs: int,
175
+ random_state: int | np.random.RandomState | None,
176
+ progress_bar: bool = False,
177
+ ) -> dict:
178
+ """Surrogate-data correlation test parallelized across CPU cores with joblib.
179
+
180
+ Pre-generates one seed per permutation, so the surrogate a permutation sees
181
+ depends only on its index and never on which worker ran it. Only `data1` is
182
+ randomized: randomizing both series reduces power and tests a different
183
+ hypothesis than H0: correlation = 0.
184
+
185
+ Args:
186
+ data1 (np.ndarray): First time series, shape `(n_samples,)`; the series
187
+ the surrogates are built from.
188
+ data2 (np.ndarray): Second time series, shape `(n_samples,)`; held fixed.
189
+ n_permute (int): Number of permutations.
190
+ method (str): `'circle_shift'` or `'phase_randomize'`.
191
+ metric (str): `'pearson'`, `'spearman'`, or `'kendall'`.
192
+ tail (int | str): `2` or `'two'` for two-tailed; `1` or `'one'` for
193
+ one-tailed.
194
+ return_null (bool): Whether to return the null distribution.
195
+ n_jobs (int): Number of joblib workers (-1 = all cores).
196
+ random_state (int | np.random.RandomState | None): Random seed for
197
+ reproducibility.
198
+ progress_bar (bool): Whether to display a tqdm progress bar.
199
+
200
+ Returns:
201
+ dict: Same format as `_timeseries_correlation_permutation_test`.
202
+ """
203
+ from joblib import Parallel, delayed
204
+
205
+ corr_func = _select_corr_func(metric)
206
+
207
+ obs_corr = corr_func(data1, data2)
208
+ if isinstance(obs_corr, np.ndarray):
209
+ obs_corr = obs_corr[0]
210
+ obs_corr = np.asarray(obs_corr)
211
+
212
+ rng = check_random_state(random_state)
213
+ MAX_INT = 2**31 - 1
214
+ seeds = rng.randint(MAX_INT, size=n_permute)
215
+
216
+ surrogate = circle_shift if method == "circle_shift" else phase_randomize
217
+
218
+ def _compute_one_perm(seed):
219
+ """Correlate one surrogate of `data1` against the fixed `data2`."""
220
+ corr = corr_func(surrogate(data1, random_state=seed), data2)
221
+ return corr[0] if isinstance(corr, np.ndarray) else corr
222
+
223
+ null_dist = Parallel(n_jobs=n_jobs)(
224
+ delayed(_compute_one_perm)(seeds[i])
225
+ for i in _maybe_tqdm(
226
+ range(n_permute),
227
+ progress_bar=progress_bar,
228
+ desc=f"{method} perms",
229
+ unit="perm",
230
+ )
231
+ )
232
+ null_dist = np.array(null_dist)
233
+
234
+ p_value = _compute_pvalue(obs_corr, null_dist, tail=tail)
235
+
236
+ results = {
237
+ "correlation": float(obs_corr),
238
+ "p": p_value.item() if hasattr(p_value, "item") else float(p_value),
239
+ }
240
+
241
+ if return_null:
242
+ results["null_dist"] = null_dist
243
+
244
+ return results
245
+
246
+
247
+ def _timeseries_correlation_permutation_test(
248
+ data1: np.ndarray,
249
+ data2: np.ndarray,
250
+ *,
251
+ method: Literal["circle_shift", "phase_randomize"] = "circle_shift",
252
+ n_permute: int = 5000,
253
+ metric: Literal["pearson", "spearman", "kendall"] = "pearson",
254
+ tail: int | str = 2,
255
+ n_jobs: int = -1,
256
+ return_null: bool = False,
257
+ random_state: int | np.random.RandomState | None = None,
258
+ progress_bar: bool = False,
259
+ ) -> dict:
260
+ """Permutation test for the correlation between two autocorrelated time series.
261
+
262
+ Standard permutation tests shuffle samples independently, which destroys
263
+ autocorrelation and inflates Type I error on time series. This test instead
264
+ builds the null from surrogates of `data1` that preserve temporal structure
265
+ — circular shifts (`'circle_shift'`) or Fourier phase randomization
266
+ (`'phase_randomize'`) — while `data2` stays fixed. For independent
267
+ observations use `correlation_permutation_test`.
268
+
269
+ Args:
270
+ data1 (np.ndarray): First time series, shape (n_samples,) or (n_samples, 1).
271
+ data2 (np.ndarray): Second time series, shape (n_samples,) or (n_samples, 1).
272
+ method (str): `'circle_shift'` rotates the series (preserves
273
+ autocorrelation; fast, and suitable for most fMRI time series);
274
+ `'phase_randomize'` randomizes Fourier phases (preserves the power
275
+ spectrum exactly; tests for nonlinear structure). Defaults to
276
+ 'circle_shift'.
277
+ n_permute (int): Number of permutations. Defaults to 5000.
278
+ metric (str): Correlation type, one of 'pearson', 'spearman', or
279
+ 'kendall'. Defaults to 'pearson'.
280
+ tail (int | str): `2` or `'two'` for a two-tailed test; `1` or `'one'` for
281
+ a one-tailed test in the positive direction (negate one series for the
282
+ other direction). Defaults to 2.
283
+ n_jobs (int): Number of joblib workers, -1 = all cores. Defaults to -1.
284
+ Results are identical at every worker count.
285
+ return_null (bool): Also return the null distribution. Defaults to False.
286
+ random_state (int | np.random.RandomState | None): Random seed for
287
+ reproducibility.
288
+ progress_bar (bool): Show a progress bar over permutations. Defaults to
289
+ False.
290
+
291
+ Returns:
292
+ dict: Keys 'correlation' (float, observed correlation), 'p' (float), and
293
+ 'null_dist' (np.ndarray of shape (n_permute,)) when
294
+ `return_null=True`.
295
+
296
+ Examples:
297
+ ```python
298
+ import numpy as np
299
+ from nltools.algorithms import _timeseries_correlation_permutation_test
300
+
301
+ rng = np.random.default_rng(0)
302
+ x = np.sin(np.linspace(0, 10 * np.pi, 100)) # strongly autocorrelated
303
+ y = x + rng.standard_normal(100) * 0.5
304
+ result = _timeseries_correlation_permutation_test(
305
+ x, y, method="circle_shift", n_permute=1000, random_state=42
306
+ )
307
+ result["correlation"] # → 0.853
308
+ result["p"] # → 0.078 — the autocorrelation-aware null is far wider
309
+ # than a sample-shuffling null would be
310
+ ```
311
+ """
312
+ # Validate tail up front (like one_sample/two_sample/matrix) so an invalid
313
+ # value fails immediately rather than after every permutation has run.
314
+ _validate_tail_parameter(tail)
315
+
316
+ # Validate inputs
317
+ data1 = np.asarray(data1).squeeze()
318
+ data2 = np.asarray(data2).squeeze()
319
+
320
+ if data1.ndim != 1 or data2.ndim != 1:
321
+ raise ValueError("data1 and data2 must be 1D arrays")
322
+
323
+ if len(data1) != len(data2):
324
+ raise ValueError("data1 and data2 must have the same length")
325
+
326
+ if method not in ["circle_shift", "phase_randomize"]:
327
+ raise ValueError(
328
+ f"method must be 'circle_shift' or 'phase_randomize', got '{method}'"
329
+ )
330
+
331
+ if metric not in ["pearson", "spearman", "kendall"]:
332
+ raise ValueError(
333
+ f"metric must be 'pearson', 'spearman', or 'kendall', got '{metric}'"
334
+ )
335
+
336
+ return _timeseries_correlation_cpu_parallel(
337
+ data1,
338
+ data2,
339
+ n_permute=n_permute,
340
+ method=method,
341
+ metric=metric,
342
+ tail=tail,
343
+ return_null=return_null,
344
+ n_jobs=n_jobs,
345
+ random_state=random_state,
346
+ progress_bar=progress_bar,
347
+ )
@@ -0,0 +1,212 @@
1
+ """Two-sample permutation test (group-label shuffling).
2
+
3
+ Tests whether two independent groups differ in mean by randomly reassigning
4
+ observations to groups — the permutation analogue of an independent-samples
5
+ t-test. Permutations run on joblib workers; `n_jobs` sets how many, and a
6
+ given `random_state` gives the same result at any worker count.
7
+ """
8
+
9
+ import numpy as np
10
+
11
+ from .utils import _maybe_tqdm
12
+ from .validation import _validate_array_shape_range
13
+ from ..validation import _compute_pvalue, _validate_tail_parameter
14
+ from .random import _generate_seeds
15
+
16
+
17
+ def _two_sample_permutation_cpu_parallel(
18
+ data1: np.ndarray,
19
+ data2: np.ndarray,
20
+ *,
21
+ n_permute: int,
22
+ tail: int,
23
+ return_null: bool,
24
+ n_jobs: int,
25
+ random_state: int | None,
26
+ single_feature: bool = False,
27
+ progress_bar: bool = False,
28
+ ) -> dict:
29
+ """Two-sample permutation test parallelized across CPU cores with joblib.
30
+
31
+ Each worker shuffles the group labels for one permutation (from its own
32
+ pre-drawn seed) and computes the mean difference, so results are
33
+ reproducible regardless of worker count.
34
+
35
+ Args:
36
+ data1 (np.ndarray): Group 1 data, shape `(n_samples1, n_features)`.
37
+ data2 (np.ndarray): Group 2 data, shape `(n_samples2, n_features)`.
38
+ n_permute (int): Number of permutations.
39
+ tail (int | str): `2` or `'two'` for two-tailed; `1` or `'one'` for
40
+ one-tailed.
41
+ return_null (bool): Whether to return the null distribution.
42
+ n_jobs (int): Number of parallel jobs (-1 = all cores).
43
+ random_state (int | None): Random seed for reproducibility.
44
+ single_feature (bool): Whether the caller passed 1D data (results are
45
+ returned as scalars).
46
+ progress_bar (bool): Whether to display a tqdm progress bar.
47
+
48
+ Returns:
49
+ dict: Same format as `two_sample_permutation_test`.
50
+ """
51
+ from joblib import Parallel, delayed
52
+
53
+ # Setup random state and generate seeds for workers
54
+ seeds = _generate_seeds(n_permute, random_state=random_state)
55
+
56
+ # Get dimensions (data already reshaped by caller)
57
+ n1, n_features = data1.shape
58
+ n2 = data2.shape[0]
59
+ n_total = n1 + n2
60
+
61
+ # Compute observed mean difference
62
+ obs_diff = np.nanmean(data1, axis=0) - np.nanmean(data2, axis=0)
63
+
64
+ # Concatenate data for permutation
65
+ combined = np.vstack([data1, data2]) # (n_total, n_features)
66
+
67
+ # Define worker function (each processes ONE permutation)
68
+ def _compute_one_perm(seed):
69
+ """Compute mean difference for one group permutation."""
70
+ perm_rng = np.random.RandomState(seed)
71
+ # Randomly shuffle indices
72
+ indices = perm_rng.permutation(n_total)
73
+ # Split into two groups
74
+ group1_indices = indices[:n1]
75
+ group2_indices = indices[n1:]
76
+ # Compute mean difference
77
+ mean1 = np.nanmean(combined[group1_indices], axis=0)
78
+ mean2 = np.nanmean(combined[group2_indices], axis=0)
79
+ return mean1 - mean2
80
+
81
+ # Execute in parallel with progress bar
82
+ null_dist = Parallel(n_jobs=n_jobs)(
83
+ delayed(_compute_one_perm)(seeds[i])
84
+ for i in _maybe_tqdm(
85
+ range(n_permute),
86
+ progress_bar=progress_bar,
87
+ desc="CPU parallel perms",
88
+ unit="perm",
89
+ )
90
+ )
91
+ null_dist = np.array(null_dist) # Shape: (n_permute, n_features)
92
+
93
+ # Compute p-values
94
+ p_values = _compute_pvalue(obs_diff, null_dist, tail=tail)
95
+
96
+ # Return to original shape
97
+ if single_feature:
98
+ obs_diff = obs_diff.item() if hasattr(obs_diff, "item") else float(obs_diff[0])
99
+ p_values = p_values.item() if hasattr(p_values, "item") else float(p_values[0])
100
+
101
+ # Build result
102
+ result = {
103
+ "mean_diff": obs_diff,
104
+ "p": p_values,
105
+ }
106
+
107
+ if return_null:
108
+ if single_feature:
109
+ null_dist = null_dist.squeeze()
110
+ result["null_dist"] = null_dist
111
+
112
+ return result
113
+
114
+
115
+ def two_sample_permutation_test(
116
+ data1: np.ndarray,
117
+ data2: np.ndarray,
118
+ *,
119
+ n_permute: int = 5000,
120
+ tail: int | str = 2,
121
+ return_null: bool = False,
122
+ n_jobs: int = -1,
123
+ random_state: int | None = None,
124
+ progress_bar: bool = False,
125
+ ) -> dict:
126
+ """Two-sample permutation test using group-label shuffling.
127
+
128
+ Tests whether two independent groups have different means by randomly
129
+ reassigning observations to groups — the permutation analogue of an
130
+ independent-samples t-test. Group sizes may differ. Multi-feature
131
+ (voxel-wise) data tests each column independently against the same
132
+ permutations.
133
+
134
+ Assumes exchangeability under the null (group assignment is arbitrary):
135
+ independent samples from similarly shaped distributions. NaN observations
136
+ are dropped from the observed and every permuted mean (`np.nanmean`),
137
+ feature by feature.
138
+
139
+ Args:
140
+ data1 (np.ndarray): Group 1 data, shape `(n_samples1,)` for a single
141
+ feature or `(n_samples1, n_features)` for voxel-wise data. May
142
+ contain NaN observations.
143
+ data2 (np.ndarray): Group 2 data, shape `(n_samples2,)` or
144
+ `(n_samples2, n_features)`; must have the same number of features
145
+ as `data1`. May contain NaN observations.
146
+ n_permute (int): Number of permutations. Defaults to 5000.
147
+ tail (int | str): `2` or `'two'` (default) for a two-tailed test
148
+ (mean1 != mean2); `1` or `'one'` for a one-tailed test of
149
+ mean1 > mean2 (swap the groups for the other direction — the fixed
150
+ direction keeps multiple-comparison correction valid).
151
+ return_null (bool): If True, include the full null distribution in the
152
+ result. Defaults to False.
153
+ n_jobs (int): Number of joblib workers. Defaults to -1 (all cores).
154
+ Results are identical at every worker count.
155
+ random_state (int | None): Random seed for reproducibility.
156
+ progress_bar (bool): Whether to display a progress bar. Defaults to False.
157
+
158
+ Returns:
159
+ dict: Keys `'mean_diff'` (float or np.ndarray, observed
160
+ `mean(data1) - mean(data2)`), `'p'` (float or np.ndarray,
161
+ p-value(s)), and — when `return_null=True` — `'null_dist'`
162
+ (np.ndarray, shape `(n_permute,)` or `(n_permute, n_features)`).
163
+
164
+ Examples:
165
+ ```python
166
+ # Single feature
167
+ data1 = np.random.randn(20) # Group 1: 20 subjects
168
+ data2 = np.random.randn(25) # Group 2: 25 subjects
169
+ result = two_sample_permutation_test(data1, data2, n_permute=5000)
170
+ result["p"] # → 0.45
171
+
172
+ # Voxel-wise test
173
+ data1 = np.random.randn(20, 10000) # 20 subjects, 10K voxels
174
+ data2 = np.random.randn(25, 10000) # 25 subjects, 10K voxels
175
+ result = two_sample_permutation_test(data1, data2, n_permute=5000)
176
+ result["mean_diff"].shape # → (10000,)
177
+ result["p"].shape # → (10000,)
178
+ ```
179
+ """
180
+ # Input validation
181
+ data1 = np.asarray(data1, dtype=np.float64)
182
+ data2 = np.asarray(data2, dtype=np.float64)
183
+
184
+ _validate_array_shape_range(data1, 1, 2, name="data1")
185
+ _validate_array_shape_range(data2, 1, 2, name="data2")
186
+ _validate_tail_parameter(tail)
187
+
188
+ # Handle shape
189
+ single_feature = data1.ndim == 1 and data2.ndim == 1
190
+ if data1.ndim == 1:
191
+ data1 = data1[:, np.newaxis]
192
+ if data2.ndim == 1:
193
+ data2 = data2[:, np.newaxis]
194
+
195
+ # Check feature dimensions match
196
+ if data1.shape[1] != data2.shape[1]:
197
+ raise ValueError(
198
+ f"data1 and data2 must have same number of features, "
199
+ f"got {data1.shape[1]} and {data2.shape[1]}"
200
+ )
201
+
202
+ return _two_sample_permutation_cpu_parallel(
203
+ data1,
204
+ data2,
205
+ n_permute=n_permute,
206
+ tail=tail,
207
+ return_null=return_null,
208
+ n_jobs=n_jobs,
209
+ random_state=random_state,
210
+ single_feature=single_feature,
211
+ progress_bar=progress_bar,
212
+ )
@@ -0,0 +1,58 @@
1
+ """Shared helpers for the permutation tests: sign flips, z-from-p, the stability epsilon and progress bars."""
2
+
3
+ import numpy as np
4
+ from ...utils import _NullProgressBar, _make_progress_bar, _maybe_tqdm # noqa: F401
5
+ from .random import _generate_sign_flips as _generate_sign_flips # noqa: F401
6
+
7
+
8
+ # ============================================================================
9
+ # Numerical Stability Constants
10
+ # ============================================================================
11
+
12
+ # Small constant added to denominators to prevent division by zero
13
+ # Value: 1e-10 is standard in scientific computing for float64 precision
14
+ # - Well above machine epsilon (2.22e-16 for float64)
15
+ # - Small enough not to affect correlation values
16
+ # - Matches established practice in neuroimaging libraries
17
+ EPSILON = 1e-10
18
+
19
+
20
+ def _signed_z_from_p(t_like_arr, p_arr, tail_internal: str = "two") -> np.ndarray:
21
+ """Compute a signed z-score map from a p-value map.
22
+
23
+ The z-from-p conversion used by `BrainData.ttest` and `Adjacency.ttest`.
24
+ The clipping policy below must live in exactly one place.
25
+
26
+ Two-tailed p: ``|z| = norm.isf(p/2)`` so that p=0.05 → |z|≈1.96, matching
27
+ nilearn's ``output_type='z_score'`` convention, with the sign copied from
28
+ the accompanying statistic. One-tailed (upper) p: ``z = norm.isf(p)`` —
29
+ a one-sided p already encodes direction, so no sign copy is needed.
30
+
31
+ p is clipped to the open interval (0, 1) so z stays finite in BOTH
32
+ directions: the lower bound is the smallest positive normal float64
33
+ (z ≈ +37.7), the upper bound the largest float64 below 1 (z ≈ -8.3).
34
+ The bounds look asymmetric because float64 resolves p near 0 far more
35
+ finely than near 1; both sit beyond any meaningful statistical
36
+ resolution. Without the upper clip, ``p == 1.0`` — reachable on the
37
+ one-tailed path, e.g. ``t.sf(-45, 29) == 1.0`` — maps to ``-inf`` and
38
+ poisons downstream percentiles and plotting.
39
+
40
+ Args:
41
+ t_like_arr (np.ndarray): Statistic supplying the sign (t map or similar).
42
+ p_arr (np.ndarray): P-value map (two-tailed, or one-tailed upper).
43
+ tail_internal (str): `'two'` (default) or `'upper'`.
44
+
45
+ Returns:
46
+ np.ndarray: Signed z map, finite everywhere.
47
+ """
48
+ from scipy.stats import norm
49
+
50
+ p_clipped = np.clip(
51
+ np.asarray(p_arr, dtype=np.float64),
52
+ np.finfo(np.float64).tiny,
53
+ np.nextafter(1.0, 0.0),
54
+ )
55
+ if tail_internal == "upper":
56
+ return norm.isf(p_clipped)
57
+ z_abs = norm.isf(p_clipped / 2.0)
58
+ return np.sign(np.asarray(t_like_arr)) * z_abs