tools 1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tools/__init__.py +14 -0
- tools/array.py +39 -0
- tools/data.py +90 -0
- tools/exp.py +165 -0
- tools/hydra/__init__.py +6 -0
- tools/modules.py +117 -0
- tools/numpy/__init__.py +8 -0
- tools/numpy/_f.py +60 -0
- tools/numpy/_utils.py +117 -0
- tools/os.py +67 -0
- tools/pandas/__init__.py +45 -0
- tools/plot/__init__.py +5 -0
- tools/plot/sklearn.py +28 -0
- tools/plot/utils.py +70 -0
- tools/random.py +35 -0
- tools/sklearn/__init__.py +2 -0
- tools/sklearn/metrics.py +82 -0
- tools/sklearn/model_selection.py +229 -0
- tools/sklearn/preprocessing.py +125 -0
- tools/stats/__init__.py +43 -0
- tools/tools.py +568 -0
- tools/torch/__init__.py +15 -0
- tools/torch/_pandas.py +12 -0
- tools/torch/data.py +81 -0
- tools/torch/estimator.py +65 -0
- tools/torch/federated_learning.py +376 -0
- tools/torch/layers.py +13 -0
- tools/torch/model.py +397 -0
- tools/torch/optim/__init__.py +2 -0
- tools/torch/optim/lr_scheduler.py +195 -0
- tools/torch/plot.py +13 -0
- tools/torch/utils.py +121 -0
- tools-1.0.dist-info/LICENSE.txt +21 -0
- tools-1.0.dist-info/METADATA +37 -0
- tools-1.0.dist-info/RECORD +37 -0
- tools-1.0.dist-info/WHEEL +5 -0
- tools-1.0.dist-info/top_level.txt +1 -0
tools/pandas/__init__.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import numbers
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
def isclose(a, b, rtol=1e-09, atol=0.0, equal_nan=False):
|
|
7
|
+
"""
|
|
8
|
+
Wraparound np.isclose for pandas.DataFrame
|
|
9
|
+
|
|
10
|
+
Parameters
|
|
11
|
+
----------
|
|
12
|
+
a, b : pandas.DataFrame
|
|
13
|
+
Input arrays to compare.
|
|
14
|
+
rtol : float
|
|
15
|
+
The relative tolerance parameter (see Notes).
|
|
16
|
+
atol : float
|
|
17
|
+
The absolute tolerance parameter (see Notes).
|
|
18
|
+
equal_nan : bool
|
|
19
|
+
Whether to compare NaN's as equal. If True, NaN's in `a` will be
|
|
20
|
+
considered equal to NaN's in `b` in the output array.
|
|
21
|
+
|
|
22
|
+
Returns
|
|
23
|
+
-------
|
|
24
|
+
C : pandas.DataFrame
|
|
25
|
+
Returns a boolean array of where `a` and `b` are equal within the
|
|
26
|
+
given tolerance. If both `a` and `b` are scalars, returns a single
|
|
27
|
+
boolean value.
|
|
28
|
+
"""
|
|
29
|
+
assert a.shape == b.shape, "Shape mismatch"
|
|
30
|
+
assert (a.columns == b.columns).all(), "Column mismatch"
|
|
31
|
+
assert (a.index == b.index).all(), "Index mismatch"
|
|
32
|
+
|
|
33
|
+
close_list = []
|
|
34
|
+
for col in a.columns:
|
|
35
|
+
if issubclass(a[col].dtype.type, numbers.Number):
|
|
36
|
+
close = np.isclose(a[col], b[col], rtol=rtol, atol=atol, equal_nan=equal_nan)
|
|
37
|
+
else:
|
|
38
|
+
close = (a[col] == b[col]).to_numpy()
|
|
39
|
+
close_list.append(close)
|
|
40
|
+
close = np.stack(close_list, axis=1)
|
|
41
|
+
|
|
42
|
+
return pd.DataFrame(close, columns=a.columns, index=a.index)
|
|
43
|
+
|
|
44
|
+
if __name__ == '__main__':
|
|
45
|
+
pass
|
tools/plot/__init__.py
ADDED
tools/plot/sklearn.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import sklearn.metrics as metrics
|
|
2
|
+
import seaborn as sns
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import matplotlib.pyplot as plt
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
'confusion_matrix',
|
|
8
|
+
'plot_confusion_matrix',
|
|
9
|
+
]
|
|
10
|
+
|
|
11
|
+
def plot_confusion_matrix(y_true, y_pred, ax=None, **kwargs):
|
|
12
|
+
'''Just a wrapper to generate the confusion matrix and the plot'''
|
|
13
|
+
c_matrix = metrics.confusion_matrix(y_true, y_pred)
|
|
14
|
+
return confusion_matrix(c_matrix, ax=ax)
|
|
15
|
+
|
|
16
|
+
def confusion_matrix(c_matrix, ax=None, **kwargs):
|
|
17
|
+
'''Plot confusion matrix heatmap'''
|
|
18
|
+
df_cm = pd.DataFrame(c_matrix)
|
|
19
|
+
if ax is None:
|
|
20
|
+
ax = sns.heatmap(df_cm, annot=True, **kwargs)
|
|
21
|
+
else:
|
|
22
|
+
ax = sns.heatmap(df_cm, annot=True, ax=ax, **kwargs)
|
|
23
|
+
ax.set_ylabel('y_true')
|
|
24
|
+
ax.set_xlabel('y_pred')
|
|
25
|
+
return ax
|
|
26
|
+
|
|
27
|
+
if __name__ == '__main__':
|
|
28
|
+
pass
|
tools/plot/utils.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# import logging
|
|
2
|
+
import os
|
|
3
|
+
import warnings
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import matplotlib.colors as mcolors
|
|
7
|
+
from .. import numpy as tnp
|
|
8
|
+
# log = logging.getLogger(__name__)
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
'get_xtick_seconds',
|
|
12
|
+
'multisave',
|
|
13
|
+
]
|
|
14
|
+
|
|
15
|
+
# %%
|
|
16
|
+
def get_xtick_seconds(ax, sr):
|
|
17
|
+
'''
|
|
18
|
+
:param ax: axis instance
|
|
19
|
+
:param sec: second which maximum length indicates. it is assumed the xtick starts at 0
|
|
20
|
+
'''
|
|
21
|
+
# xticks = np.linspace(0,sec,len(ax_real.get_xticklabels())-2)
|
|
22
|
+
# xticks = np.round(xticks, 3)
|
|
23
|
+
# d = xticks[1]-xticks[0]
|
|
24
|
+
# xticks = np.concatenate([[xticks[0] - d],xticks,[xticks[-1]+d]])
|
|
25
|
+
# return xticks
|
|
26
|
+
return ax.get_xticks() / sr
|
|
27
|
+
|
|
28
|
+
def multisave(fig, path, dpi=300):
|
|
29
|
+
'''
|
|
30
|
+
save given figure in .png, .eps, .svg format
|
|
31
|
+
:param fig: pyplot figure instance
|
|
32
|
+
:param path: path of the figure without extension (.extension)
|
|
33
|
+
'''
|
|
34
|
+
if '.' in os.path.basename(path):
|
|
35
|
+
warnings.warn("Do not specify '.' in name. assuming it's part of the filename...")
|
|
36
|
+
|
|
37
|
+
# log.warning('Do not specify extension in multisave(). Removing extension...')
|
|
38
|
+
# path = os.path.splitext(path)[0]
|
|
39
|
+
|
|
40
|
+
fig.savefig(path+'.png', dpi=dpi)
|
|
41
|
+
fig.savefig(path+'.tiff', dpi=dpi)
|
|
42
|
+
fig.savefig(path+'.eps')
|
|
43
|
+
fig.savefig(path+'.svg')
|
|
44
|
+
|
|
45
|
+
def plot_values(valuetracker_list, ax=None):
|
|
46
|
+
n_line = len(valuetracker_list)
|
|
47
|
+
if ax is None:
|
|
48
|
+
fig, ax = plt.subplots()
|
|
49
|
+
color_list = it.cycle(mcolors.TABLEAU_COLORS)
|
|
50
|
+
for valuetracker, color in zip(valuetracker_list, color_list):
|
|
51
|
+
y_smooth = tnp.moving_mean(valuetracker.y, 9)
|
|
52
|
+
ax.plot(valuetracker.x, valuetracker.y, color=color, alpha=0.4)
|
|
53
|
+
ax.plot(valuetracker.x, y_smooth, color=color)
|
|
54
|
+
return ax
|
|
55
|
+
|
|
56
|
+
def plot_trainval(valuetracker_list, ax=None):
|
|
57
|
+
n_line = len(valuetracker_list)
|
|
58
|
+
if ax is None:
|
|
59
|
+
fig, ax = plt.subplots()
|
|
60
|
+
color_list = ['tab:blue', 'tab:orange']
|
|
61
|
+
line_list = ['-', 'x-']
|
|
62
|
+
labels = ['train', 'val']
|
|
63
|
+
for valuetracker, color, line, label in zip(valuetracker_list, color_list, line_list, labels):
|
|
64
|
+
y_smooth = tnp.moving_mean(valuetracker.y, 9)
|
|
65
|
+
ax.plot(valuetracker.x, valuetracker.y, color=color, alpha=0.4, label=label)
|
|
66
|
+
ax.plot(valuetracker.x, y_smooth, line, color=color, label=label+'_smooth')
|
|
67
|
+
return ax
|
|
68
|
+
|
|
69
|
+
if __name__ == '__main__':
|
|
70
|
+
pass
|
tools/random.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import random
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import torch
|
|
5
|
+
|
|
6
|
+
# %%
|
|
7
|
+
def choice(a, size=None, replace=True):
|
|
8
|
+
'''
|
|
9
|
+
Randomly choose elements from given iterable.
|
|
10
|
+
|
|
11
|
+
Parameters
|
|
12
|
+
----------
|
|
13
|
+
a: int or list
|
|
14
|
+
choice of
|
|
15
|
+
|
|
16
|
+
'''
|
|
17
|
+
m = len(a)
|
|
18
|
+
# Single sample
|
|
19
|
+
if size==None:
|
|
20
|
+
random_i = np.random.randint(m)
|
|
21
|
+
return a[random_i]
|
|
22
|
+
else:
|
|
23
|
+
num_sample = np.prod(size)
|
|
24
|
+
|
|
25
|
+
# Multi-Sample
|
|
26
|
+
a=np.asarray(a)
|
|
27
|
+
if replace==True:
|
|
28
|
+
random_i = np.random.randint(m, size = num_sample)
|
|
29
|
+
return a[random_i].reshape(size)
|
|
30
|
+
else:
|
|
31
|
+
assert m >= num_sample, 'entries of array cannot exceed number of samples'
|
|
32
|
+
random_i = np.arange(m)
|
|
33
|
+
np.random.shuffle(random_i)
|
|
34
|
+
random_i = random_i[:num_sample]
|
|
35
|
+
return a[random_i].reshape(size)
|
tools/sklearn/metrics.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
from collections.abc import Iterable
|
|
2
|
+
import warnings
|
|
3
|
+
|
|
4
|
+
import numpy as np
|
|
5
|
+
from sklearn.metrics import r2_score as r2_score_sklearn
|
|
6
|
+
|
|
7
|
+
def squared_error(y_true, y_pred, axis=None):
|
|
8
|
+
'''MSE without mean
|
|
9
|
+
if axis is provided, it will return the mean along the axis
|
|
10
|
+
'''
|
|
11
|
+
score = (y_true-y_pred)**2
|
|
12
|
+
if axis is not None:
|
|
13
|
+
score = np.mean(score, axis=axis)
|
|
14
|
+
return score
|
|
15
|
+
|
|
16
|
+
def absolute_error(y_true, y_pred, axis=None):
|
|
17
|
+
'''MAE without mean'''
|
|
18
|
+
score = np.abs(y_true-y_pred)
|
|
19
|
+
if axis is not None:
|
|
20
|
+
score = np.mean(score, axis=axis)
|
|
21
|
+
return score
|
|
22
|
+
|
|
23
|
+
def r2_score(y_true, y_pred, axis=None, multioutput='raw_values'):
|
|
24
|
+
"""
|
|
25
|
+
R^2 score for multidimensional predictions.
|
|
26
|
+
collapses all axes except the specified axis.
|
|
27
|
+
|
|
28
|
+
Parameters
|
|
29
|
+
----------
|
|
30
|
+
y_true : np.ndarray
|
|
31
|
+
y_pred : np.ndarray
|
|
32
|
+
axis: int or iterable of int, default=None
|
|
33
|
+
Axis to collapse.
|
|
34
|
+
It must be specified if y_true and y_pred dimensions are > 2
|
|
35
|
+
multioutput : Reference to `sklearn.metrics.r2_score`
|
|
36
|
+
https://scikit-learn.org/stable/modules/generated/sklearn.metrics.r2_score.html
|
|
37
|
+
|
|
38
|
+
Returns
|
|
39
|
+
-------
|
|
40
|
+
z : np.ndarray
|
|
41
|
+
if axis is specified, returns an array of shape (y_true.shape[axis],)
|
|
42
|
+
"""
|
|
43
|
+
assert y_true.shape == y_pred.shape, f"y_true and y_pred must have the same shape, received: {y_true.shape} and {y_pred.shape}"
|
|
44
|
+
if axis is not None:
|
|
45
|
+
if not isinstance(axis, Iterable):
|
|
46
|
+
axis = [axis]
|
|
47
|
+
else:
|
|
48
|
+
axis = list(axis)
|
|
49
|
+
if len(axis) > y_true.ndim:
|
|
50
|
+
raise ValueError("Axis is greater than the number of dimensions of y_true and y_pred")
|
|
51
|
+
|
|
52
|
+
# shape_final = np.array(y_true.shape)[axis]
|
|
53
|
+
# shape_collapse = list(set(range(y_true.ndim)).difference(axis))
|
|
54
|
+
dim_collapse = axis
|
|
55
|
+
dim_final = list(set(range(y_true.ndim)).difference(axis))
|
|
56
|
+
shape_final = np.array(y_true.shape)[dim_final]
|
|
57
|
+
|
|
58
|
+
y_true = np.transpose(y_true, (*dim_collapse, *dim_final)).reshape(-1, np.prod(shape_final)) # Move axis to the end, and flatten the rest
|
|
59
|
+
y_pred = np.transpose(y_pred, (*dim_collapse, *dim_final)).reshape(-1, np.prod(shape_final))
|
|
60
|
+
|
|
61
|
+
score = r2_score_sklearn(y_true, y_pred, multioutput=multioutput)
|
|
62
|
+
|
|
63
|
+
if type(score) == float and np.isnan(score).item():
|
|
64
|
+
warnings.warn("R2 score is a single NaN, shape matching with NaN")
|
|
65
|
+
score = np.full(shape_final, np.nan)
|
|
66
|
+
return score
|
|
67
|
+
|
|
68
|
+
if multioutput == 'raw_values':
|
|
69
|
+
return score.reshape(*shape_final)
|
|
70
|
+
elif multioutput == 'uniform_average':
|
|
71
|
+
return score
|
|
72
|
+
elif multioutput == 'variance_weighted':
|
|
73
|
+
# return np.mean(score, weights=np.var(y_true, axis=0))
|
|
74
|
+
raise NotImplementedError("variance_weighted is not implemented yet")
|
|
75
|
+
else:
|
|
76
|
+
raise ValueError("multioutput must be one of ['raw_values', 'uniform_average', 'variance_weighted']")
|
|
77
|
+
|
|
78
|
+
else:
|
|
79
|
+
assert (y_true.ndim <= 2) and (y_pred.ndim <= 2), "If axis is None, y_true and y_pred must be smaller than 2D"
|
|
80
|
+
return r2_score_sklearn(y_true, y_pred, multioutput=multioutput)
|
|
81
|
+
|
|
82
|
+
# %%
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
# %%
|
|
2
|
+
from sklearn.model_selection import GridSearchCV
|
|
3
|
+
from sklearn.model_selection import StratifiedShuffleSplit, ShuffleSplit, train_test_split, StratifiedKFold, KFold
|
|
4
|
+
import numpy as np
|
|
5
|
+
import pandas as pd
|
|
6
|
+
import torch
|
|
7
|
+
|
|
8
|
+
from .. import torch as ttorch
|
|
9
|
+
|
|
10
|
+
# %%
|
|
11
|
+
'''
|
|
12
|
+
Train/(Validation)/Test Split index functions
|
|
13
|
+
'''
|
|
14
|
+
def stratified_train_test_split_i(y, test_size=0.15, random_state=None):
|
|
15
|
+
''':return: indices of train_i, test_i'''
|
|
16
|
+
x = np.empty(len(y))
|
|
17
|
+
|
|
18
|
+
sss = StratifiedShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
|
|
19
|
+
train_i, test_i = next(sss.split(x, y))
|
|
20
|
+
return train_i, test_i
|
|
21
|
+
|
|
22
|
+
def stratified_train_val_test_split_i(y, val_size=0.15, test_size=0.15, random_state=None):
|
|
23
|
+
x = np.empty(len(y))
|
|
24
|
+
|
|
25
|
+
sss = StratifiedShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
|
|
26
|
+
train_val_i, test_i = next(sss.split(x, y))
|
|
27
|
+
|
|
28
|
+
train_val_y = y[train_val_i]
|
|
29
|
+
x = np.empty(len(train_val_i))
|
|
30
|
+
sss = StratifiedShuffleSplit(n_splits=1, test_size=val_size/(1-test_size), random_state=random_state)
|
|
31
|
+
train_i_, val_i_ = next(sss.split(x, train_val_y))
|
|
32
|
+
train_i = train_val_i[train_i_]
|
|
33
|
+
val_i = train_val_i[val_i_]
|
|
34
|
+
|
|
35
|
+
return train_i, val_i, test_i
|
|
36
|
+
|
|
37
|
+
def train_test_split_i(y, test_size=0.15, random_state=None):
|
|
38
|
+
x = np.empty(len(y))
|
|
39
|
+
|
|
40
|
+
ss = ShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
|
|
41
|
+
train_i, test_i = next(ss.split(x, y))
|
|
42
|
+
|
|
43
|
+
return train_i, test_i
|
|
44
|
+
|
|
45
|
+
def train_val_test_split_i(y, val_size=0.15, test_size=0.15, random_state=None):
|
|
46
|
+
x = np.empty(len(y))
|
|
47
|
+
|
|
48
|
+
ss = ShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
|
|
49
|
+
train_val_i, test_i = next(ss.split(x, y))
|
|
50
|
+
|
|
51
|
+
train_val_y = y[train_val_i]
|
|
52
|
+
x = np.empty(len(train_val_i))
|
|
53
|
+
ss = ShuffleSplit(n_splits=1, test_size=val_size/(1-test_size), random_state=random_state)
|
|
54
|
+
train_i_, val_i_ = next(ss.split(x, train_val_y))
|
|
55
|
+
train_i = train_val_i[train_i_]
|
|
56
|
+
val_i = train_val_i[val_i_]
|
|
57
|
+
|
|
58
|
+
return train_i, val_i, test_i
|
|
59
|
+
|
|
60
|
+
def stratified_kfold_split_i(y, n_splits, split_i, shuffle=True, random_state=None):
|
|
61
|
+
'''
|
|
62
|
+
return train, test indices of "split_i"-th split of "n_splits"-fold split
|
|
63
|
+
'''
|
|
64
|
+
skf = StratifiedKFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
|
|
65
|
+
x = np.zeros(len(y))
|
|
66
|
+
skf_ = skf.split(x, y)
|
|
67
|
+
|
|
68
|
+
for i in range(split_i):
|
|
69
|
+
next(skf_)
|
|
70
|
+
train_i, test_i = next(skf_)
|
|
71
|
+
|
|
72
|
+
return train_i, test_i
|
|
73
|
+
|
|
74
|
+
def stratified_nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
|
|
75
|
+
'''
|
|
76
|
+
return train, val, test indices of "split_i"-th split of "n_splits"-fold split
|
|
77
|
+
'''
|
|
78
|
+
# Outer split
|
|
79
|
+
skf = StratifiedKFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
|
|
80
|
+
x = np.zeros(len(y))
|
|
81
|
+
skf_ = skf.split(x, y)
|
|
82
|
+
|
|
83
|
+
for i in range(split_i):
|
|
84
|
+
next(skf_)
|
|
85
|
+
train_val_i, test_i = next(skf_)
|
|
86
|
+
|
|
87
|
+
# Inner split
|
|
88
|
+
skf = StratifiedKFold(n_splits=m_splits, shuffle=shuffle, random_state=random_state)
|
|
89
|
+
x = np.zeros(len(y[train_val_i]))
|
|
90
|
+
skf_ = skf.split(x, y[train_val_i])
|
|
91
|
+
|
|
92
|
+
for j in range(split_j):
|
|
93
|
+
next(skf_)
|
|
94
|
+
train_i, val_i = next(skf_)
|
|
95
|
+
train_i, val_i = train_val_i[train_i], train_val_i[val_i]
|
|
96
|
+
|
|
97
|
+
return train_i, val_i, test_i
|
|
98
|
+
|
|
99
|
+
def kfold_split_i(y, n_splits, split_i, shuffle=True, random_state=None):
|
|
100
|
+
'''
|
|
101
|
+
return train, test indices of "split_i"-th split of "n_splits"-fold split
|
|
102
|
+
'''
|
|
103
|
+
kf = KFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
|
|
104
|
+
kf_ = kf.split(y)
|
|
105
|
+
|
|
106
|
+
for i in range(split_i):
|
|
107
|
+
next(kf_)
|
|
108
|
+
train_i, test_i = next(kf_)
|
|
109
|
+
|
|
110
|
+
return train_i, test_i
|
|
111
|
+
|
|
112
|
+
def nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
|
|
113
|
+
'''
|
|
114
|
+
return train, val, test indices of "split_i"-th split of "n_splits"-fold split
|
|
115
|
+
'''
|
|
116
|
+
# Outer split
|
|
117
|
+
kf = KFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
|
|
118
|
+
kf_ = kf.split(y)
|
|
119
|
+
|
|
120
|
+
for i in range(split_i):
|
|
121
|
+
next(kf_)
|
|
122
|
+
train_val_i, test_i = next(kf_)
|
|
123
|
+
|
|
124
|
+
# Inner split
|
|
125
|
+
kf = KFold(n_splits=m_splits, shuffle=shuffle, random_state=random_state)
|
|
126
|
+
kf_ = kf.split(y[train_val_i])
|
|
127
|
+
|
|
128
|
+
for j in range(split_j):
|
|
129
|
+
next(kf_)
|
|
130
|
+
train_i, val_i = next(kf_)
|
|
131
|
+
train_i, val_i = train_val_i[train_i], train_val_i[val_i]
|
|
132
|
+
|
|
133
|
+
return train_i, val_i, test_i
|
|
134
|
+
|
|
135
|
+
# %%
|
|
136
|
+
'''
|
|
137
|
+
Train/(Validation)/Test split dataset functions
|
|
138
|
+
'''
|
|
139
|
+
def nested_kfold_split_data(data, y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
|
|
140
|
+
'''
|
|
141
|
+
y doesn't matter since it's not stratified
|
|
142
|
+
'''
|
|
143
|
+
train_i, val_i, test_i = nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle, random_state)
|
|
144
|
+
train_data, val_data, test_data = wrap_data(data, train_i), wrap_data(data, val_i), wrap_data(data, test_i)
|
|
145
|
+
return train_data, val_data, test_data
|
|
146
|
+
|
|
147
|
+
def stratified_nested_kfold_split_data(data, y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
|
|
148
|
+
"""
|
|
149
|
+
Split data into train, validation, and test set
|
|
150
|
+
|
|
151
|
+
Parameters
|
|
152
|
+
----------
|
|
153
|
+
data : pd.DataFrame or np.ndarray or torch.Tensor or torch.dat.Dataset object or iterable object (i.e. list)
|
|
154
|
+
Data to split
|
|
155
|
+
|
|
156
|
+
y: pd.Series or np.ndarray or torch.Tensor or iterable object (i.e. list)
|
|
157
|
+
Target variable to be used in stratified split
|
|
158
|
+
|
|
159
|
+
Returns
|
|
160
|
+
-------
|
|
161
|
+
train_data : train data of the same variable format
|
|
162
|
+
val_data : validation data of the same variable format
|
|
163
|
+
test_data : test data of the same variable format
|
|
164
|
+
"""
|
|
165
|
+
train_i, val_i, test_i = stratified_nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle, random_state)
|
|
166
|
+
train_data, val_data, test_data = wrap_data(data, train_i), wrap_data(data, val_i), wrap_data(data, test_i)
|
|
167
|
+
return train_data, val_data, test_data
|
|
168
|
+
|
|
169
|
+
def wrap_data(data, i):
|
|
170
|
+
if isinstance(data, np.ndarray):
|
|
171
|
+
return data[i]
|
|
172
|
+
elif isinstance(data, pd.DataFrame):
|
|
173
|
+
return data.iloc[i]
|
|
174
|
+
elif isinstance(data, torch.Tensor):
|
|
175
|
+
return data[i]
|
|
176
|
+
elif isinstance(data, torch.utils.data.Dataset):
|
|
177
|
+
return ttorch.data.ProxyDataset(dataset=data, idxs=i)
|
|
178
|
+
else:
|
|
179
|
+
return data[i]
|
|
180
|
+
|
|
181
|
+
# %%
|
|
182
|
+
# def train_val_test_split(x, val_size=0.1, test_size=0.1, random_state=None):
|
|
183
|
+
# if type(x)==int:
|
|
184
|
+
# x = np.zeros(x)
|
|
185
|
+
# ss = ShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
|
|
186
|
+
# train_val_i, test_i = next(ss.split(x))
|
|
187
|
+
# train_i, val_i = train_test_split(train_val_i, test_size=val_size/(1-test_size), random_state=random_state)
|
|
188
|
+
#
|
|
189
|
+
# # assert len(set(np.concatenate((train_i, val_i, test_i)))) == len(x)
|
|
190
|
+
# return train_i, val_i, test_i
|
|
191
|
+
|
|
192
|
+
class MultiGridSearchCV():
|
|
193
|
+
'''Perform Grid Search over multiple models
|
|
194
|
+
|
|
195
|
+
Parameters
|
|
196
|
+
----------
|
|
197
|
+
estimators: dict
|
|
198
|
+
Name of model. This name has to correspond to model name of param_grid
|
|
199
|
+
param_grids: dict
|
|
200
|
+
Parameter Grid to Search. It must be dict nested in dict(Double dict).
|
|
201
|
+
**kwargs: other parameters from GridSearchCV
|
|
202
|
+
Look GridSearchCV
|
|
203
|
+
|
|
204
|
+
examples of **kwargs
|
|
205
|
+
--------------------
|
|
206
|
+
cv: number of k-fold CV
|
|
207
|
+
scoring: scoring method. either can be prebuilt sklearn string or a callable function
|
|
208
|
+
|
|
209
|
+
'''
|
|
210
|
+
def __init__(self, estimators, param_grids, **kwargs):
|
|
211
|
+
self.estimators = estimators
|
|
212
|
+
self.param_grids = param_grids
|
|
213
|
+
self.kwargs = kwargs
|
|
214
|
+
self.grid_searches = {estimator_name: GridSearchCV(estimator=estimator, param_grid=self.param_grids[estimator_name], **self.kwargs) for estimator_name, estimator in self.estimators.items()}
|
|
215
|
+
|
|
216
|
+
def fit(self, x, y):
|
|
217
|
+
for grid_search in self.grid_searches.values():
|
|
218
|
+
grid_search.fit(x,y)
|
|
219
|
+
self.cv_results = {estimator_name: grid_search.cv_results_ for estimator_name, grid_search in self.grid_searches.items()}
|
|
220
|
+
best_estimator_name_, best_grid_search_ = max(self.grid_searches.items(), key = lambda x: x[1].best_score_) # x[0]: key, x[1]: value
|
|
221
|
+
self.best_score_ = best_grid_search_.best_score_
|
|
222
|
+
self.best_estimator_ = best_grid_search_.best_estimator_
|
|
223
|
+
self.best_params_ = best_grid_search_.best_params_
|
|
224
|
+
self.best_estimator_name_ = best_estimator_name_
|
|
225
|
+
|
|
226
|
+
def predict(self, x):
|
|
227
|
+
assert hasattr(self, 'best_estimator_'), 'There is no best_estimator_. Need to call "fit" first.'
|
|
228
|
+
assert hasattr(self.best_estimator_,'predict'), 'Best estimator does not support "predict" method.'
|
|
229
|
+
return self.best_estimator_.predict(x)
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
from sklearn.preprocessing import StandardScaler, MinMaxScaler
|
|
2
|
+
from sklearn.decomposition import PCA
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
def standardize(x, scaler=None):
|
|
6
|
+
"""
|
|
7
|
+
Standardize (z-score/gaussian normalization) x without having to make a StandardScaler() object.
|
|
8
|
+
|
|
9
|
+
Parameters
|
|
10
|
+
----------
|
|
11
|
+
x : ndarray of shape (n_observation, n_channel)
|
|
12
|
+
data to be standardized.
|
|
13
|
+
scaler: sklearn.preprocessing.StandardScaler(), default=None
|
|
14
|
+
scaler to use (that has already fit())
|
|
15
|
+
|
|
16
|
+
Returns
|
|
17
|
+
-------
|
|
18
|
+
x_standardized : ndarray of shape (n_observation, n_channel)
|
|
19
|
+
Normalized data
|
|
20
|
+
scaler : sklearn.preprocessing.StandardScaler()
|
|
21
|
+
Scaler used for normalization
|
|
22
|
+
"""
|
|
23
|
+
if scaler is None:
|
|
24
|
+
scaler = StandardScaler()
|
|
25
|
+
x_standardized = scaler.fit_transform(x)
|
|
26
|
+
else:
|
|
27
|
+
x_standardized = scaler.transform(x)
|
|
28
|
+
|
|
29
|
+
return x_standardized, scaler
|
|
30
|
+
|
|
31
|
+
def project_pca(x, var_threshold=None, n_pc=None, solver=None, random_state=0):
|
|
32
|
+
"""
|
|
33
|
+
project x onto PC space without having to make a PCA() object.
|
|
34
|
+
|
|
35
|
+
Parameters
|
|
36
|
+
----------
|
|
37
|
+
x : ndarray of shape (n_observation, n_channel)
|
|
38
|
+
data to be projected.
|
|
39
|
+
solver: sklearn.decomposition.PCA(), default=None
|
|
40
|
+
solver to use (that has already fit())
|
|
41
|
+
|
|
42
|
+
Returns
|
|
43
|
+
-------
|
|
44
|
+
x_projected : ndarray of shape (n_observation, n_channel)
|
|
45
|
+
Normalized data
|
|
46
|
+
solver : sklearn.decomposition.PCA()
|
|
47
|
+
Solver used for PCA projection
|
|
48
|
+
"""
|
|
49
|
+
# If svd_solver is not full, result is random
|
|
50
|
+
if solver is None:
|
|
51
|
+
solver = PCA(svd_solver='full', random_state=random_state)
|
|
52
|
+
x_pc = solver.fit_transform(x)
|
|
53
|
+
else:
|
|
54
|
+
x_pc = solver.transform(x)
|
|
55
|
+
|
|
56
|
+
if var_threshold is not None:
|
|
57
|
+
n_pc = np.argmax(np.cumsum(solver.explained_variance_ratio_) > var_threshold) + 1
|
|
58
|
+
elif n_pc is not None:
|
|
59
|
+
pass
|
|
60
|
+
else:
|
|
61
|
+
n_pc = solver.n_components_
|
|
62
|
+
x_pc = x_pc[:,:n_pc]
|
|
63
|
+
|
|
64
|
+
return x_pc, solver
|
|
65
|
+
|
|
66
|
+
# def minmaxscale(x, scaler=None, feature_range=(0, 1), *, copy=True, clip=False):
|
|
67
|
+
def minmaxscale(x, scaler=None, **kwargs):
|
|
68
|
+
"""
|
|
69
|
+
Min-max scale x without having to make a MinMaxScaler() object.
|
|
70
|
+
|
|
71
|
+
Parameters
|
|
72
|
+
----------
|
|
73
|
+
x : ndarray of shape (n_observation, n_channel)
|
|
74
|
+
data to be standardized.
|
|
75
|
+
scaler: sklearn.preprocessing.MinMaxScaler(), default=None
|
|
76
|
+
scaler to use (that has already fit())
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
Following arguments are applied if scaler is None:
|
|
80
|
+
|
|
81
|
+
feature_range : tuple (min, max), default=(0, 1)
|
|
82
|
+
Desired range of transformed data.
|
|
83
|
+
The default is (0, 1).
|
|
84
|
+
copy : bool, default=True
|
|
85
|
+
Set to False to perform inplace normalization and avoid a copy (if the input is already a numpy array).
|
|
86
|
+
clip : bool, default=False
|
|
87
|
+
Set to True to clip transformed values of held-out data.
|
|
88
|
+
|
|
89
|
+
Returns
|
|
90
|
+
-------
|
|
91
|
+
x_scaled : ndarray of shape (n_observation, n_channel)
|
|
92
|
+
Normalized data
|
|
93
|
+
scaler : sklearn.preprocessing.MinMaxScaler()
|
|
94
|
+
Scaler used for normalization
|
|
95
|
+
"""
|
|
96
|
+
if scaler is None:
|
|
97
|
+
# scaler = MinMaxScaler(feature_range=feature_range, copy=copy, clip=clip)
|
|
98
|
+
scaler = MinMaxScaler(**kwargs)
|
|
99
|
+
x_scaled = scaler.fit_transform(x)
|
|
100
|
+
else:
|
|
101
|
+
x_scaled = scaler.transform(x)
|
|
102
|
+
|
|
103
|
+
return x_scaled, scaler
|
|
104
|
+
|
|
105
|
+
if __name__=='__main__':
|
|
106
|
+
import tools as T
|
|
107
|
+
|
|
108
|
+
x = np.random.rand(2,3,5)
|
|
109
|
+
print(x.shape)
|
|
110
|
+
squeezer = T.numpy.Squeezer()
|
|
111
|
+
x = squeezer.squeeze(x)
|
|
112
|
+
print(x.shape)
|
|
113
|
+
x_s, scaler = T.sklearn.preprocessing.standardize(x)
|
|
114
|
+
x_s = squeezer.unsqueeze(x_s)
|
|
115
|
+
print(x_s.shape)
|
|
116
|
+
|
|
117
|
+
# %%
|
|
118
|
+
x = np.random.rand(2,3,5)
|
|
119
|
+
print(x.shape)
|
|
120
|
+
squeezer = T.numpy.Squeezer()
|
|
121
|
+
x = squeezer.squeeze(x)
|
|
122
|
+
print(x.shape)
|
|
123
|
+
x_pc, solver = T.sklearn.preprocessing.project_pca(x, n_pc=3)
|
|
124
|
+
x_pc = squeezer.unsqueeze(x_pc, strict=False)
|
|
125
|
+
print(x_pc.shape)
|
tools/stats/__init__.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import scipy.stats as st
|
|
3
|
+
|
|
4
|
+
def ci(std, n, p=0.95):
|
|
5
|
+
z = st.norm.ppf((1+p)/2) # Two-tailed confidence interval
|
|
6
|
+
half_range = z*std/np.sqrt(n)
|
|
7
|
+
return half_range
|
|
8
|
+
|
|
9
|
+
def interval(distrib, confidence, **kwargs):
|
|
10
|
+
if 'std' in kwargs and 'n' in kwargs:
|
|
11
|
+
scale = kwargs['std']/np.sqrt(kwargs['n'])
|
|
12
|
+
assert 'scale' not in kwargs, 'scale is given but std and n are also given'
|
|
13
|
+
kwargs['scale'] = scale
|
|
14
|
+
del kwargs['std'], kwargs['n']
|
|
15
|
+
|
|
16
|
+
return distrib.interval(confidence, **kwargs)
|
|
17
|
+
|
|
18
|
+
def ci_stats(mean, std, n, p=0.95):
|
|
19
|
+
'''
|
|
20
|
+
calculate confidence interval based on given statistics
|
|
21
|
+
:param mean: 1d array, sample mean
|
|
22
|
+
:param std: 1d array, sample mean
|
|
23
|
+
:param n: 1d array, sample size
|
|
24
|
+
:param p: 1d array, confidence level
|
|
25
|
+
'''
|
|
26
|
+
# z = st.norm.ppf(p)
|
|
27
|
+
# half_range = z*std/n
|
|
28
|
+
half_range = ci(std=std, n=n)
|
|
29
|
+
|
|
30
|
+
confidence_interval = {
|
|
31
|
+
'low': mean-half_range,
|
|
32
|
+
'high': mean+half_range
|
|
33
|
+
}
|
|
34
|
+
return confidence_interval
|
|
35
|
+
|
|
36
|
+
def ci_samples(x, p=0.95):
|
|
37
|
+
'''
|
|
38
|
+
calculate confidence interval based on samples
|
|
39
|
+
'''
|
|
40
|
+
mean = np.mean(x)
|
|
41
|
+
std = np.std(x)
|
|
42
|
+
n = len(x)
|
|
43
|
+
return ci_stats(mean, std, n, p=p)
|