fdnkit 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fdnkit/__init__.py +94 -0
- fdnkit/classify.py +310 -0
- fdnkit/cli.py +175 -0
- fdnkit/dfa.py +88 -0
- fdnkit/features.py +212 -0
- fdnkit/fodn.py +346 -0
- fdnkit/io.py +163 -0
- fdnkit/mfdfa.py +233 -0
- fdnkit/preprocessing.py +125 -0
- fdnkit/synthetic.py +173 -0
- fdnkit/viz.py +146 -0
- fdnkit-1.0.0.dist-info/METADATA +192 -0
- fdnkit-1.0.0.dist-info/RECORD +16 -0
- fdnkit-1.0.0.dist-info/WHEEL +4 -0
- fdnkit-1.0.0.dist-info/entry_points.txt +2 -0
- fdnkit-1.0.0.dist-info/licenses/LICENSE +21 -0
fdnkit/io.py
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""Input/output: load recordings, load trial labels, read/write feature tables.
|
|
2
|
+
|
|
3
|
+
Heavy IO dependencies are optional and imported lazily so the core analysis
|
|
4
|
+
stack (numpy/scipy/pandas/scikit-learn) stays lightweight:
|
|
5
|
+
|
|
6
|
+
* EDF reading uses **MNE-Python** (``pip install fdnkit[io]``).
|
|
7
|
+
* HDF5 reading uses **h5py**.
|
|
8
|
+
|
|
9
|
+
Both raise a clear, actionable error if the backend is missing.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
import pandas as pd
|
|
19
|
+
|
|
20
|
+
__all__ = ["Recording", "load_edf", "load_h5", "load_labels_excel",
|
|
21
|
+
"save_features", "load_features"]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class Recording:
|
|
26
|
+
"""A loaded multi-channel recording.
|
|
27
|
+
|
|
28
|
+
Attributes
|
|
29
|
+
----------
|
|
30
|
+
signals : numpy.ndarray
|
|
31
|
+
Shape ``(n_channels, n_samples)``.
|
|
32
|
+
fs : float
|
|
33
|
+
Sampling frequency in Hz.
|
|
34
|
+
channel_names : list of str
|
|
35
|
+
times : numpy.ndarray | None
|
|
36
|
+
Optional per-sample time vector (seconds).
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
signals: np.ndarray
|
|
40
|
+
fs: float
|
|
41
|
+
channel_names: list
|
|
42
|
+
times: np.ndarray | None = None
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def n_channels(self) -> int:
|
|
46
|
+
return self.signals.shape[0]
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def n_samples(self) -> int:
|
|
50
|
+
return self.signals.shape[1]
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def duration(self) -> float:
|
|
54
|
+
return self.n_samples / self.fs
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def load_edf(path, *, preload: bool = True) -> Recording:
|
|
58
|
+
"""Load an EDF/EDF+ recording via MNE-Python.
|
|
59
|
+
|
|
60
|
+
Parameters
|
|
61
|
+
----------
|
|
62
|
+
path : str | pathlib.Path
|
|
63
|
+
preload : bool
|
|
64
|
+
Read sample data into memory immediately.
|
|
65
|
+
|
|
66
|
+
Returns
|
|
67
|
+
-------
|
|
68
|
+
Recording
|
|
69
|
+
"""
|
|
70
|
+
try:
|
|
71
|
+
import mne
|
|
72
|
+
except ImportError as exc: # pragma: no cover - exercised only without mne
|
|
73
|
+
raise ImportError(
|
|
74
|
+
"Reading EDF requires MNE-Python. Install with `pip install fdnkit[io]` "
|
|
75
|
+
"or `pip install mne`."
|
|
76
|
+
) from exc
|
|
77
|
+
|
|
78
|
+
raw = mne.io.read_raw_edf(str(path), preload=preload, verbose="ERROR")
|
|
79
|
+
signals = raw.get_data()
|
|
80
|
+
fs = float(raw.info["sfreq"])
|
|
81
|
+
names = list(raw.ch_names)
|
|
82
|
+
times = raw.times.copy()
|
|
83
|
+
return Recording(signals=signals, fs=fs, channel_names=names, times=times)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def load_h5(path, *, signals_key="data/Signals", time_key="data/Time",
|
|
87
|
+
names_key="metadata/channel_names", fs: float = 1000.0) -> Recording:
|
|
88
|
+
"""Load a recording from an HDF5 file (h5py).
|
|
89
|
+
|
|
90
|
+
Defaults match the reference layout: signals at ``data/Signals`` with shape
|
|
91
|
+
``(n_channels, n_samples)``, an optional time vector at ``data/Time``, and
|
|
92
|
+
optional channel names at ``metadata/channel_names``.
|
|
93
|
+
"""
|
|
94
|
+
try:
|
|
95
|
+
import h5py
|
|
96
|
+
except ImportError as exc: # pragma: no cover
|
|
97
|
+
raise ImportError(
|
|
98
|
+
"Reading HDF5 requires h5py. Install with `pip install fdnkit[io]` "
|
|
99
|
+
"or `pip install h5py`."
|
|
100
|
+
) from exc
|
|
101
|
+
|
|
102
|
+
with h5py.File(str(path), "r") as f:
|
|
103
|
+
signals = np.asarray(f[signals_key][:])
|
|
104
|
+
times = np.asarray(f[time_key][:]) if time_key in f else None
|
|
105
|
+
if names_key in f:
|
|
106
|
+
raw_names = f[names_key][:]
|
|
107
|
+
names = [n.decode("utf-8") if isinstance(n, bytes) else str(n) for n in raw_names]
|
|
108
|
+
else:
|
|
109
|
+
names = [f"CH{i + 1}" for i in range(signals.shape[0])]
|
|
110
|
+
|
|
111
|
+
if times is not None and times.size > 1:
|
|
112
|
+
dt = float(np.median(np.diff(times)))
|
|
113
|
+
if dt > 0:
|
|
114
|
+
fs = 1.0 / dt
|
|
115
|
+
return Recording(signals=signals, fs=fs, channel_names=names, times=times)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def load_labels_excel(path, *, id_col="Patient_Session_Trial", score_col="Math_Score",
|
|
119
|
+
success_code="M1", failure_code="M0", header: int = 1) -> dict:
|
|
120
|
+
"""Load a ``{trial_id: 0/1}`` label mapping from an Excel scoresheet.
|
|
121
|
+
|
|
122
|
+
Ports ``utils/label_strategies.ExcelMathScoreLabeler``: rows whose score
|
|
123
|
+
equals ``success_code`` map to 1, ``failure_code`` to 0; anything else
|
|
124
|
+
(blank, "MC", ...) is skipped.
|
|
125
|
+
|
|
126
|
+
Parameters
|
|
127
|
+
----------
|
|
128
|
+
path : str | pathlib.Path
|
|
129
|
+
id_col, score_col : str
|
|
130
|
+
Column names for the trial identifier and the score.
|
|
131
|
+
success_code, failure_code : str
|
|
132
|
+
Score strings mapped to 1 and 0 respectively (compared case-insensitively).
|
|
133
|
+
header : int
|
|
134
|
+
Row index (0-based) of the header. Defaults to 1 (second row).
|
|
135
|
+
|
|
136
|
+
Returns
|
|
137
|
+
-------
|
|
138
|
+
dict[str, int]
|
|
139
|
+
"""
|
|
140
|
+
df = pd.read_excel(path, header=header)
|
|
141
|
+
df = df[[id_col, score_col]].dropna(subset=[id_col])
|
|
142
|
+
mapping: dict = {}
|
|
143
|
+
for _, row in df.iterrows():
|
|
144
|
+
tid = str(row[id_col]).strip()
|
|
145
|
+
score = str(row[score_col]).strip().upper()
|
|
146
|
+
if score == success_code.upper():
|
|
147
|
+
mapping[tid] = 1
|
|
148
|
+
elif score == failure_code.upper():
|
|
149
|
+
mapping[tid] = 0
|
|
150
|
+
return mapping
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def save_features(df: pd.DataFrame, path) -> Path:
|
|
154
|
+
"""Write a feature DataFrame to CSV (index omitted). Returns the path."""
|
|
155
|
+
path = Path(path)
|
|
156
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
157
|
+
df.to_csv(path, index=False)
|
|
158
|
+
return path
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def load_features(path) -> pd.DataFrame:
|
|
162
|
+
"""Read a feature table CSV back into a DataFrame."""
|
|
163
|
+
return pd.read_csv(path)
|
fdnkit/mfdfa.py
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Multifractal detrended fluctuation analysis (MFDFA).
|
|
2
|
+
|
|
3
|
+
Ports the validated reference implementation, cross-checked against its MATLAB
|
|
4
|
+
counterpart, into a clean, array-first API.
|
|
5
|
+
|
|
6
|
+
The generalized Hurst exponent ``h(q)`` describes how the ``q``-th order
|
|
7
|
+
fluctuation of a signal scales with window size. A signal is *monofractal* when
|
|
8
|
+
``h(q)`` is (nearly) constant in ``q`` and *multifractal* when it varies; the
|
|
9
|
+
width ``delta_h = max h(q) - min h(q)`` quantifies multifractality.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
|
|
18
|
+
__all__ = ["MFDFAResult", "mfdfa", "generalized_hurst", "delta_hq", "multifractal_spectrum"]
|
|
19
|
+
|
|
20
|
+
# Default analysis grids (match the validated reference / MATLAB settings).
|
|
21
|
+
DEFAULT_SCALES = np.array([4, 6, 8, 12, 16, 24, 32, 48, 64, 96, 128, 192, 256])
|
|
22
|
+
DEFAULT_Q = np.array([-5, -3, -2, -1, 0, 1, 2, 3, 5], dtype=float)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class MFDFAResult:
|
|
27
|
+
"""Container for the output of :func:`mfdfa`.
|
|
28
|
+
|
|
29
|
+
Attributes
|
|
30
|
+
----------
|
|
31
|
+
hurst : float
|
|
32
|
+
Monofractal Hurst exponent ``H`` (slope of ``log2 F`` vs ``log2 scale``,
|
|
33
|
+
equivalent to ``h(q=2)``).
|
|
34
|
+
hq : numpy.ndarray
|
|
35
|
+
Generalized Hurst exponents, one per entry of :attr:`q`.
|
|
36
|
+
q : numpy.ndarray
|
|
37
|
+
The moment orders used.
|
|
38
|
+
scales : numpy.ndarray
|
|
39
|
+
The window sizes used.
|
|
40
|
+
fluct : numpy.ndarray
|
|
41
|
+
Standard (``q=2``) fluctuation function ``F`` per scale.
|
|
42
|
+
fluct_q : numpy.ndarray
|
|
43
|
+
Fluctuation function per ``(q, scale)`` -- shape ``(len(q), len(scales))``.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
hurst: float
|
|
47
|
+
hq: np.ndarray
|
|
48
|
+
q: np.ndarray
|
|
49
|
+
scales: np.ndarray
|
|
50
|
+
fluct: np.ndarray
|
|
51
|
+
fluct_q: np.ndarray
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def delta_h(self) -> float:
|
|
55
|
+
"""Multifractal width ``max h(q) - min h(q)`` (0 for a monofractal)."""
|
|
56
|
+
finite = self.hq[np.isfinite(self.hq)]
|
|
57
|
+
if finite.size == 0:
|
|
58
|
+
return float("nan")
|
|
59
|
+
return float(finite.max() - finite.min())
|
|
60
|
+
|
|
61
|
+
def hq_at(self, q_value: float) -> float:
|
|
62
|
+
"""Return ``h(q)`` at the grid point nearest ``q_value``."""
|
|
63
|
+
idx = int(np.argmin(np.abs(self.q - q_value)))
|
|
64
|
+
return float(self.hq[idx])
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _fluctuations(signal, scales, order, rel_floor: float = 1e-3):
|
|
68
|
+
"""Local detrended RMS fluctuations for each scale (q=2 base quantities).
|
|
69
|
+
|
|
70
|
+
Returns a list whose ``i``-th entry is the array of per-segment RMS values at
|
|
71
|
+
``scales[i]`` (empty if the scale exceeds the signal length).
|
|
72
|
+
|
|
73
|
+
``rel_floor`` sets a *scale-relative* lower bound on each RMS value. Quantized
|
|
74
|
+
or degenerately-detrended segments (common in real, integer-stored recordings,
|
|
75
|
+
especially at small scales) can drive a segment's RMS to ~0; raised to a
|
|
76
|
+
negative moment ``q`` this dominates ``F_q(s)`` and fabricates a huge apparent
|
|
77
|
+
multifractal width. Flooring at ``rel_floor * median(RMS)`` for the scale --
|
|
78
|
+
rather than at machine epsilon -- tames this without perturbing well-behaved
|
|
79
|
+
(e.g. synthetic) signals, whose fluctuations never approach the floor.
|
|
80
|
+
"""
|
|
81
|
+
eps = np.finfo(float).eps
|
|
82
|
+
x = np.asarray(signal, dtype=float)
|
|
83
|
+
n = x.size
|
|
84
|
+
y = np.cumsum(x - x.mean())
|
|
85
|
+
|
|
86
|
+
rms_per_scale = []
|
|
87
|
+
for s in scales:
|
|
88
|
+
s = int(s)
|
|
89
|
+
segs = n // s
|
|
90
|
+
if segs == 0: # scale larger than the signal
|
|
91
|
+
rms_per_scale.append(np.empty(0))
|
|
92
|
+
continue
|
|
93
|
+
t = np.arange(s)
|
|
94
|
+
rms = np.empty(segs)
|
|
95
|
+
for v in range(segs):
|
|
96
|
+
seg = y[v * s : (v + 1) * s]
|
|
97
|
+
coef = np.polyfit(t, seg, order)
|
|
98
|
+
fit = np.polyval(coef, t)
|
|
99
|
+
rms[v] = np.sqrt(np.mean((seg - fit) ** 2))
|
|
100
|
+
positive = rms[rms > 0]
|
|
101
|
+
floor = rel_floor * np.median(positive) if positive.size else eps
|
|
102
|
+
floor = max(floor, eps)
|
|
103
|
+
np.maximum(rms, floor, out=rms)
|
|
104
|
+
rms_per_scale.append(rms)
|
|
105
|
+
return rms_per_scale
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _loglog_slope(scales, values):
|
|
109
|
+
"""Slope of log2(values) vs log2(scales) over finite, positive points."""
|
|
110
|
+
lg_s = np.log2(np.asarray(scales, dtype=float))
|
|
111
|
+
lg_v = np.log2(np.asarray(values, dtype=float))
|
|
112
|
+
good = np.isfinite(lg_v) & np.isfinite(lg_s)
|
|
113
|
+
if good.sum() < 2:
|
|
114
|
+
return float("nan")
|
|
115
|
+
# Ordinary least-squares slope.
|
|
116
|
+
xs, ys = lg_s[good], lg_v[good]
|
|
117
|
+
xm, ym = xs.mean(), ys.mean()
|
|
118
|
+
denom = np.sum((xs - xm) ** 2)
|
|
119
|
+
if denom == 0:
|
|
120
|
+
return float("nan")
|
|
121
|
+
return float(np.sum((xs - xm) * (ys - ym)) / denom)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def mfdfa(signal, scales=None, q=None, order: int = 1, rel_floor: float = 1e-3) -> MFDFAResult:
|
|
125
|
+
"""Run multifractal detrended fluctuation analysis on a 1-D signal.
|
|
126
|
+
|
|
127
|
+
Parameters
|
|
128
|
+
----------
|
|
129
|
+
signal : array-like
|
|
130
|
+
1-D time series.
|
|
131
|
+
scales : array-like, optional
|
|
132
|
+
Window sizes (in samples). Defaults to
|
|
133
|
+
``[4, 6, 8, 12, 16, 24, 32, 48, 64, 96, 128, 192, 256]``. Every scale
|
|
134
|
+
should be smaller than ``len(signal)``.
|
|
135
|
+
q : array-like, optional
|
|
136
|
+
Moment orders. Defaults to ``[-5, -3, -2, -1, 0, 1, 2, 3, 5]``.
|
|
137
|
+
order : int
|
|
138
|
+
Order of the polynomial used to detrend each segment (1 = linear).
|
|
139
|
+
rel_floor : float
|
|
140
|
+
Scale-relative floor on per-segment fluctuations, as a fraction of the
|
|
141
|
+
median fluctuation at each scale. Guards the negative-``q`` moments
|
|
142
|
+
against degenerate near-zero segments in real (e.g. integer-quantized)
|
|
143
|
+
recordings; set to 0 to disable. Does not affect well-behaved signals.
|
|
144
|
+
|
|
145
|
+
Returns
|
|
146
|
+
-------
|
|
147
|
+
MFDFAResult
|
|
148
|
+
|
|
149
|
+
Notes
|
|
150
|
+
-----
|
|
151
|
+
Follows Kantelhardt et al. (2002) with a forward (non-overlapping) segment
|
|
152
|
+
partition, matching the validated reference implementation. The ``q = 0``
|
|
153
|
+
moment uses the logarithmic-average limit
|
|
154
|
+
``F_0(s) = exp(0.5 * mean(log RMS^2))``.
|
|
155
|
+
"""
|
|
156
|
+
eps = np.finfo(float).eps
|
|
157
|
+
scales = DEFAULT_SCALES if scales is None else np.asarray(scales)
|
|
158
|
+
scales = np.asarray(scales, dtype=int)
|
|
159
|
+
q = DEFAULT_Q if q is None else np.asarray(q, dtype=float)
|
|
160
|
+
|
|
161
|
+
x = np.asarray(signal, dtype=float).ravel()
|
|
162
|
+
if x.size < int(scales.min()) * 2:
|
|
163
|
+
raise ValueError(
|
|
164
|
+
f"signal length {x.size} too short for smallest scale {int(scales.min())}"
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
rms_per_scale = _fluctuations(x, scales, order, rel_floor=rel_floor)
|
|
168
|
+
|
|
169
|
+
fluct = np.full(len(scales), np.nan)
|
|
170
|
+
fluct_q = np.full((len(q), len(scales)), np.nan)
|
|
171
|
+
for i, rms in enumerate(rms_per_scale):
|
|
172
|
+
if rms.size == 0:
|
|
173
|
+
continue
|
|
174
|
+
fluct[i] = np.sqrt(np.mean(rms**2))
|
|
175
|
+
for j, qq in enumerate(q):
|
|
176
|
+
if qq == 0:
|
|
177
|
+
fluct_q[j, i] = np.exp(0.5 * np.mean(np.log(rms**2)))
|
|
178
|
+
else:
|
|
179
|
+
fluct_q[j, i] = np.mean(rms**qq) ** (1.0 / qq)
|
|
180
|
+
|
|
181
|
+
hurst = _loglog_slope(scales, fluct + eps)
|
|
182
|
+
hq = np.array([_loglog_slope(scales, fluct_q[j] + eps) for j in range(len(q))])
|
|
183
|
+
|
|
184
|
+
return MFDFAResult(
|
|
185
|
+
hurst=hurst, hq=hq, q=np.asarray(q, dtype=float),
|
|
186
|
+
scales=np.asarray(scales), fluct=fluct, fluct_q=fluct_q,
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def generalized_hurst(signal, scales=None, q=None, order: int = 1):
|
|
191
|
+
"""Convenience wrapper returning ``(q, h(q))`` arrays only."""
|
|
192
|
+
res = mfdfa(signal, scales=scales, q=q, order=order)
|
|
193
|
+
return res.q, res.hq
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def delta_hq(signal, scales=None, q=None, order: int = 1) -> float:
|
|
197
|
+
"""Multifractal width ``max h(q) - min h(q)`` for a signal."""
|
|
198
|
+
return mfdfa(signal, scales=scales, q=q, order=order).delta_h
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def multifractal_spectrum(result: MFDFAResult):
|
|
202
|
+
"""Legendre-transform the generalized Hurst exponents to ``(alpha, f(alpha))``.
|
|
203
|
+
|
|
204
|
+
Uses the standard MFDFA relations
|
|
205
|
+
|
|
206
|
+
``tau(q) = q * h(q) - 1``,
|
|
207
|
+
``alpha = d tau / d q``,
|
|
208
|
+
``f(alpha) = q * alpha - tau(q)``.
|
|
209
|
+
|
|
210
|
+
Parameters
|
|
211
|
+
----------
|
|
212
|
+
result : MFDFAResult
|
|
213
|
+
Output of :func:`mfdfa` (needs at least 3 finite ``h(q)`` points).
|
|
214
|
+
|
|
215
|
+
Returns
|
|
216
|
+
-------
|
|
217
|
+
alpha : numpy.ndarray
|
|
218
|
+
Holder exponents (singularity strengths).
|
|
219
|
+
f_alpha : numpy.ndarray
|
|
220
|
+
Singularity spectrum values.
|
|
221
|
+
"""
|
|
222
|
+
q = result.q
|
|
223
|
+
hq = result.hq
|
|
224
|
+
good = np.isfinite(hq)
|
|
225
|
+
q, hq = q[good], hq[good]
|
|
226
|
+
if q.size < 3:
|
|
227
|
+
raise ValueError("need at least 3 finite h(q) points for a spectrum")
|
|
228
|
+
order = np.argsort(q)
|
|
229
|
+
q, hq = q[order], hq[order]
|
|
230
|
+
tau = q * hq - 1.0
|
|
231
|
+
alpha = np.gradient(tau, q)
|
|
232
|
+
f_alpha = q * alpha - tau
|
|
233
|
+
return alpha, f_alpha
|
fdnkit/preprocessing.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Preprocessing helpers: normalization, artifact-channel flagging, windowing.
|
|
2
|
+
|
|
3
|
+
Kept deliberately small -- FDNkit depends on MNE-Python for montages, filtering,
|
|
4
|
+
and ICA rather than reimplementing them. These utilities cover only what the
|
|
5
|
+
fractal/FODN pipeline needs: z-scoring, dropping obviously-bad channels, and
|
|
6
|
+
cutting a recording into fixed-length analysis windows.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
|
|
13
|
+
__all__ = ["zscore", "flag_bad_channels", "segment", "sliding_windows"]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def zscore(signals, axis: int = -1, eps: float = 1e-12) -> np.ndarray:
|
|
17
|
+
"""Z-score signals along ``axis`` (per-channel by default).
|
|
18
|
+
|
|
19
|
+
Parameters
|
|
20
|
+
----------
|
|
21
|
+
signals : array-like
|
|
22
|
+
1-D or 2-D ``(n_channels, n_samples)`` array.
|
|
23
|
+
axis : int
|
|
24
|
+
Axis along which to standardize (default ``-1`` = time).
|
|
25
|
+
eps : float
|
|
26
|
+
Floor for the standard deviation to avoid divide-by-zero on flat channels.
|
|
27
|
+
"""
|
|
28
|
+
x = np.asarray(signals, dtype=float)
|
|
29
|
+
mean = np.mean(x, axis=axis, keepdims=True)
|
|
30
|
+
std = np.std(x, axis=axis, keepdims=True)
|
|
31
|
+
return (x - mean) / np.maximum(std, eps)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def flag_bad_channels(
|
|
35
|
+
signals,
|
|
36
|
+
channel_names=None,
|
|
37
|
+
*,
|
|
38
|
+
flat_std: float = 1e-8,
|
|
39
|
+
amplitude_z: float = 5.0,
|
|
40
|
+
name_prefixes=("EKG", "ECG", "X1 DC", "DC", "TRIG", "STI"),
|
|
41
|
+
) -> list[int]:
|
|
42
|
+
"""Return indices of channels that look like artifacts.
|
|
43
|
+
|
|
44
|
+
A channel is flagged if it is (a) effectively flat, (b) has an
|
|
45
|
+
root-mean-square amplitude more than ``amplitude_z`` robust-SDs from the
|
|
46
|
+
median across channels, or (c) its name starts with a known non-neural
|
|
47
|
+
prefix (EKG, trigger, DC, ...).
|
|
48
|
+
|
|
49
|
+
Parameters
|
|
50
|
+
----------
|
|
51
|
+
signals : array-like, shape (n_channels, n_samples)
|
|
52
|
+
channel_names : sequence of str, optional
|
|
53
|
+
Used only for prefix-based flagging.
|
|
54
|
+
flat_std : float
|
|
55
|
+
Channels with standard deviation below this are flat.
|
|
56
|
+
amplitude_z : float
|
|
57
|
+
Robust z-threshold on per-channel RMS.
|
|
58
|
+
name_prefixes : tuple of str
|
|
59
|
+
Case-insensitive channel-name prefixes to treat as non-neural.
|
|
60
|
+
"""
|
|
61
|
+
x = np.asarray(signals, dtype=float)
|
|
62
|
+
if x.ndim != 2:
|
|
63
|
+
raise ValueError("signals must be 2-D (n_channels, n_samples)")
|
|
64
|
+
n_ch = x.shape[0]
|
|
65
|
+
bad = set()
|
|
66
|
+
|
|
67
|
+
std = x.std(axis=1)
|
|
68
|
+
bad.update(np.where(std < flat_std)[0].tolist())
|
|
69
|
+
|
|
70
|
+
rms = np.sqrt(np.mean(x**2, axis=1))
|
|
71
|
+
med = np.median(rms)
|
|
72
|
+
mad = np.median(np.abs(rms - med)) + 1e-12
|
|
73
|
+
robust_z = 0.6745 * (rms - med) / mad
|
|
74
|
+
bad.update(np.where(np.abs(robust_z) > amplitude_z)[0].tolist())
|
|
75
|
+
|
|
76
|
+
if channel_names is not None:
|
|
77
|
+
prefixes = tuple(p.upper() for p in name_prefixes)
|
|
78
|
+
for i, name in enumerate(channel_names):
|
|
79
|
+
if i < n_ch and str(name).upper().startswith(prefixes):
|
|
80
|
+
bad.add(i)
|
|
81
|
+
|
|
82
|
+
return sorted(bad)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def segment(signals, window: int, *, step: int = None, min_size: int = None):
|
|
86
|
+
"""Cut a signal into non-overlapping (or strided) windows along time.
|
|
87
|
+
|
|
88
|
+
Parameters
|
|
89
|
+
----------
|
|
90
|
+
signals : array-like
|
|
91
|
+
1-D ``(n_samples,)`` or 2-D ``(n_channels, n_samples)`` array.
|
|
92
|
+
window : int
|
|
93
|
+
Window length in samples.
|
|
94
|
+
step : int, optional
|
|
95
|
+
Hop size in samples. Defaults to ``window`` (non-overlapping).
|
|
96
|
+
min_size : int, optional
|
|
97
|
+
Discard a trailing window shorter than this. Defaults to ``window``
|
|
98
|
+
(i.e. drop any partial final window).
|
|
99
|
+
|
|
100
|
+
Yields
|
|
101
|
+
------
|
|
102
|
+
(start, stop, chunk) : tuple[int, int, numpy.ndarray]
|
|
103
|
+
Sample bounds and the windowed data (``chunk`` keeps the input ndim).
|
|
104
|
+
"""
|
|
105
|
+
x = np.asarray(signals)
|
|
106
|
+
if window <= 0:
|
|
107
|
+
raise ValueError("window must be positive")
|
|
108
|
+
step = window if step is None else step
|
|
109
|
+
if step <= 0:
|
|
110
|
+
raise ValueError("step must be positive")
|
|
111
|
+
min_size = window if min_size is None else min_size
|
|
112
|
+
n = x.shape[-1]
|
|
113
|
+
start = 0
|
|
114
|
+
while start < n:
|
|
115
|
+
stop = min(start + window, n)
|
|
116
|
+
if stop - start < min_size:
|
|
117
|
+
break
|
|
118
|
+
chunk = x[..., start:stop]
|
|
119
|
+
yield start, stop, chunk
|
|
120
|
+
start += step
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def sliding_windows(signals, window: int, *, step: int = None, min_size: int = None):
|
|
124
|
+
"""List form of :func:`segment` -- returns ``[(start, stop, chunk), ...]``."""
|
|
125
|
+
return list(segment(signals, window, step=step, min_size=min_size))
|
fdnkit/synthetic.py
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""Synthetic signal generators for testing, examples, and quickstarts.
|
|
2
|
+
|
|
3
|
+
These let FDNkit's tutorials and test suite run with **no data download** and give
|
|
4
|
+
analyses ground-truth answers to check against:
|
|
5
|
+
|
|
6
|
+
* :func:`fgn` / :func:`fbm` -- fractional Gaussian noise / motion with a known
|
|
7
|
+
Hurst exponent (exact Davies-Harte circulant-embedding synthesis). Detrended
|
|
8
|
+
fluctuation analysis of ``fgn(H)`` should recover ``H``.
|
|
9
|
+
* :func:`binomial_cascade` -- a multiplicative cascade whose multifractal
|
|
10
|
+
spectrum is known in closed form; used to check that MFDFA reports a genuinely
|
|
11
|
+
broad ``h(q)``.
|
|
12
|
+
* :func:`synthetic_ieeg` -- a small multi-channel, sparsely-coupled recording
|
|
13
|
+
that stands in for an intracranial-EEG trial in examples and FODN tests.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import numpy as np
|
|
19
|
+
|
|
20
|
+
__all__ = ["fgn", "fbm", "binomial_cascade", "synthetic_ieeg"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _as_rng(seed):
|
|
24
|
+
if isinstance(seed, np.random.Generator):
|
|
25
|
+
return seed
|
|
26
|
+
return np.random.default_rng(seed)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def fgn(n: int, hurst: float = 0.7, *, seed=None) -> np.ndarray:
|
|
30
|
+
"""Exact fractional Gaussian noise via Davies-Harte circulant embedding.
|
|
31
|
+
|
|
32
|
+
Parameters
|
|
33
|
+
----------
|
|
34
|
+
n : int
|
|
35
|
+
Number of samples.
|
|
36
|
+
hurst : float
|
|
37
|
+
Target Hurst exponent in (0, 1). ``0.5`` gives ordinary white noise.
|
|
38
|
+
seed : int | numpy.random.Generator | None
|
|
39
|
+
Seed or generator for reproducibility.
|
|
40
|
+
|
|
41
|
+
Returns
|
|
42
|
+
-------
|
|
43
|
+
numpy.ndarray
|
|
44
|
+
Length-``n`` unit-variance fractional Gaussian noise. Its DFA exponent is
|
|
45
|
+
(asymptotically) ``hurst``.
|
|
46
|
+
|
|
47
|
+
Notes
|
|
48
|
+
-----
|
|
49
|
+
Davies, R. B. & Harte, D. S. (1987). *Tests for Hurst effect.* Biometrika 74.
|
|
50
|
+
The circulant embedding is exact when all embedded eigenvalues are
|
|
51
|
+
non-negative; for the fGn autocovariance and ``0 < H < 1`` this holds.
|
|
52
|
+
"""
|
|
53
|
+
if n < 2:
|
|
54
|
+
raise ValueError("n must be >= 2")
|
|
55
|
+
if not 0.0 < hurst < 1.0:
|
|
56
|
+
raise ValueError("hurst must be in the open interval (0, 1)")
|
|
57
|
+
rng = _as_rng(seed)
|
|
58
|
+
H = float(hurst)
|
|
59
|
+
|
|
60
|
+
# fGn autocovariance gamma(k) for unit-variance increments.
|
|
61
|
+
k = np.arange(n)
|
|
62
|
+
gamma = 0.5 * (
|
|
63
|
+
np.abs(k - 1) ** (2 * H) - 2 * np.abs(k) ** (2 * H) + np.abs(k + 1) ** (2 * H)
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
# First row of the size-M = 2(n-1) circulant embedding.
|
|
67
|
+
row = np.concatenate([gamma, gamma[-2:0:-1]])
|
|
68
|
+
m = row.size # 2*(n-1)
|
|
69
|
+
eig = np.fft.fft(row).real
|
|
70
|
+
# Clip tiny negative eigenvalues from floating error; warn only if large.
|
|
71
|
+
if np.any(eig < -1e-8 * np.abs(eig).max()):
|
|
72
|
+
eig = np.clip(eig, 0.0, None)
|
|
73
|
+
else:
|
|
74
|
+
eig = np.clip(eig, 0.0, None)
|
|
75
|
+
|
|
76
|
+
# Build spectral coefficients W with the exact Davies-Harte weighting.
|
|
77
|
+
w = np.zeros(m, dtype=complex)
|
|
78
|
+
half = m // 2
|
|
79
|
+
v1 = rng.standard_normal(m)
|
|
80
|
+
v2 = rng.standard_normal(m)
|
|
81
|
+
w[0] = np.sqrt(eig[0]) * v1[0]
|
|
82
|
+
w[half] = np.sqrt(eig[half]) * v1[half]
|
|
83
|
+
idx = np.arange(1, half)
|
|
84
|
+
w[idx] = np.sqrt(eig[idx] / 2.0) * (v1[idx] + 1j * v2[idx])
|
|
85
|
+
w[m - idx] = np.conj(w[idx])
|
|
86
|
+
|
|
87
|
+
y = np.fft.fft(w) / np.sqrt(m)
|
|
88
|
+
return y.real[:n]
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def fbm(n: int, hurst: float = 0.7, *, seed=None) -> np.ndarray:
|
|
92
|
+
"""Fractional Brownian motion: the cumulative sum of :func:`fgn`.
|
|
93
|
+
|
|
94
|
+
Returns a length-``n`` path with Hurst exponent ``hurst`` (DFA exponent
|
|
95
|
+
``hurst + 1``).
|
|
96
|
+
"""
|
|
97
|
+
return np.cumsum(fgn(n, hurst, seed=seed))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def binomial_cascade(n_levels: int = 12, p: float = 0.3, *, seed=None) -> np.ndarray:
|
|
101
|
+
"""Deterministic-weight binomial multiplicative cascade (a multifractal).
|
|
102
|
+
|
|
103
|
+
Builds a measure on ``2**n_levels`` points by repeatedly splitting mass with
|
|
104
|
+
multipliers ``p`` and ``1 - p``. The result is strongly multifractal, so
|
|
105
|
+
MFDFA should return a wide ``h(q)`` (large ``delta_h``). With ``p = 0.5`` the
|
|
106
|
+
cascade is uniform (monofractal).
|
|
107
|
+
|
|
108
|
+
Parameters
|
|
109
|
+
----------
|
|
110
|
+
n_levels : int
|
|
111
|
+
Number of cascade levels; output length is ``2 ** n_levels``.
|
|
112
|
+
p : float
|
|
113
|
+
Multiplier in (0, 1). Distance of ``p`` from 0.5 sets multifractal width.
|
|
114
|
+
seed : int | Generator | None
|
|
115
|
+
Randomises which child receives ``p`` at each split (order only).
|
|
116
|
+
"""
|
|
117
|
+
if not 0.0 < p < 1.0:
|
|
118
|
+
raise ValueError("p must be in (0, 1)")
|
|
119
|
+
rng = _as_rng(seed)
|
|
120
|
+
measure = np.array([1.0])
|
|
121
|
+
for _ in range(n_levels):
|
|
122
|
+
left = rng.random(measure.size) < 0.5
|
|
123
|
+
m_left = np.where(left, p, 1 - p)
|
|
124
|
+
m_right = 1.0 - m_left
|
|
125
|
+
nxt = np.empty(measure.size * 2)
|
|
126
|
+
nxt[0::2] = measure * m_left
|
|
127
|
+
nxt[1::2] = measure * m_right
|
|
128
|
+
measure = nxt
|
|
129
|
+
# Return as increments of the cumulative measure (a fluctuating series).
|
|
130
|
+
return measure * measure.size
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def synthetic_ieeg(
|
|
134
|
+
n_channels: int = 8,
|
|
135
|
+
n_samples: int = 5000,
|
|
136
|
+
fs: float = 1000.0,
|
|
137
|
+
*,
|
|
138
|
+
hurst: float = 0.7,
|
|
139
|
+
coupling_density: float = 0.25,
|
|
140
|
+
coupling_strength: float = 0.35,
|
|
141
|
+
noise: float = 0.1,
|
|
142
|
+
seed=None,
|
|
143
|
+
) -> tuple[np.ndarray, list[str]]:
|
|
144
|
+
"""A small, sparsely-coupled multi-channel recording resembling iEEG.
|
|
145
|
+
|
|
146
|
+
Each channel starts as long-range-correlated fractional Gaussian noise, then
|
|
147
|
+
a sparse random coupling matrix mixes a fraction of each channel's past into
|
|
148
|
+
its neighbours. This yields data with (a) non-trivial Hurst exponents and
|
|
149
|
+
(b) a recoverable directed coupling structure -- suitable for exercising the
|
|
150
|
+
DFA, MFDFA, and FODN paths in examples and tests.
|
|
151
|
+
|
|
152
|
+
Returns
|
|
153
|
+
-------
|
|
154
|
+
signals : numpy.ndarray
|
|
155
|
+
Array of shape ``(n_channels, n_samples)``.
|
|
156
|
+
channel_names : list of str
|
|
157
|
+
Names like ``["CH1", "CH2", ...]``.
|
|
158
|
+
"""
|
|
159
|
+
rng = _as_rng(seed)
|
|
160
|
+
base = np.vstack([fgn(n_samples, hurst, seed=rng) for _ in range(n_channels)])
|
|
161
|
+
|
|
162
|
+
# Sparse random coupling matrix A (off-diagonal drives).
|
|
163
|
+
A = np.zeros((n_channels, n_channels))
|
|
164
|
+
mask = rng.random((n_channels, n_channels)) < coupling_density
|
|
165
|
+
np.fill_diagonal(mask, False)
|
|
166
|
+
A[mask] = coupling_strength * rng.standard_normal(mask.sum())
|
|
167
|
+
|
|
168
|
+
signals = base.copy()
|
|
169
|
+
for t in range(1, n_samples):
|
|
170
|
+
signals[:, t] += A @ signals[:, t - 1] + noise * rng.standard_normal(n_channels)
|
|
171
|
+
|
|
172
|
+
names = [f"CH{i + 1}" for i in range(n_channels)]
|
|
173
|
+
return signals, names
|