tools 1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,45 @@
1
+ import numbers
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+
6
+ def isclose(a, b, rtol=1e-09, atol=0.0, equal_nan=False):
7
+ """
8
+ Wraparound np.isclose for pandas.DataFrame
9
+
10
+ Parameters
11
+ ----------
12
+ a, b : pandas.DataFrame
13
+ Input arrays to compare.
14
+ rtol : float
15
+ The relative tolerance parameter (see Notes).
16
+ atol : float
17
+ The absolute tolerance parameter (see Notes).
18
+ equal_nan : bool
19
+ Whether to compare NaN's as equal. If True, NaN's in `a` will be
20
+ considered equal to NaN's in `b` in the output array.
21
+
22
+ Returns
23
+ -------
24
+ C : pandas.DataFrame
25
+ Returns a boolean array of where `a` and `b` are equal within the
26
+ given tolerance. If both `a` and `b` are scalars, returns a single
27
+ boolean value.
28
+ """
29
+ assert a.shape == b.shape, "Shape mismatch"
30
+ assert (a.columns == b.columns).all(), "Column mismatch"
31
+ assert (a.index == b.index).all(), "Index mismatch"
32
+
33
+ close_list = []
34
+ for col in a.columns:
35
+ if issubclass(a[col].dtype.type, numbers.Number):
36
+ close = np.isclose(a[col], b[col], rtol=rtol, atol=atol, equal_nan=equal_nan)
37
+ else:
38
+ close = (a[col] == b[col]).to_numpy()
39
+ close_list.append(close)
40
+ close = np.stack(close_list, axis=1)
41
+
42
+ return pd.DataFrame(close, columns=a.columns, index=a.index)
43
+
44
+ if __name__ == '__main__':
45
+ pass
tools/plot/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ from .sklearn import *
2
+ from .utils import *
3
+
4
+ if __name__ == '__main__':
5
+ pass
tools/plot/sklearn.py ADDED
@@ -0,0 +1,28 @@
1
+ import sklearn.metrics as metrics
2
+ import seaborn as sns
3
+ import pandas as pd
4
+ import matplotlib.pyplot as plt
5
+
6
+ __all__ = [
7
+ 'confusion_matrix',
8
+ 'plot_confusion_matrix',
9
+ ]
10
+
11
+ def plot_confusion_matrix(y_true, y_pred, ax=None, **kwargs):
12
+ '''Just a wrapper to generate the confusion matrix and the plot'''
13
+ c_matrix = metrics.confusion_matrix(y_true, y_pred)
14
+ return confusion_matrix(c_matrix, ax=ax)
15
+
16
+ def confusion_matrix(c_matrix, ax=None, **kwargs):
17
+ '''Plot confusion matrix heatmap'''
18
+ df_cm = pd.DataFrame(c_matrix)
19
+ if ax is None:
20
+ ax = sns.heatmap(df_cm, annot=True, **kwargs)
21
+ else:
22
+ ax = sns.heatmap(df_cm, annot=True, ax=ax, **kwargs)
23
+ ax.set_ylabel('y_true')
24
+ ax.set_xlabel('y_pred')
25
+ return ax
26
+
27
+ if __name__ == '__main__':
28
+ pass
tools/plot/utils.py ADDED
@@ -0,0 +1,70 @@
1
+ # import logging
2
+ import os
3
+ import warnings
4
+
5
+ import numpy as np
6
+ import matplotlib.colors as mcolors
7
+ from .. import numpy as tnp
8
+ # log = logging.getLogger(__name__)
9
+
10
+ __all__ = [
11
+ 'get_xtick_seconds',
12
+ 'multisave',
13
+ ]
14
+
15
+ # %%
16
+ def get_xtick_seconds(ax, sr):
17
+ '''
18
+ :param ax: axis instance
19
+ :param sec: second which maximum length indicates. it is assumed the xtick starts at 0
20
+ '''
21
+ # xticks = np.linspace(0,sec,len(ax_real.get_xticklabels())-2)
22
+ # xticks = np.round(xticks, 3)
23
+ # d = xticks[1]-xticks[0]
24
+ # xticks = np.concatenate([[xticks[0] - d],xticks,[xticks[-1]+d]])
25
+ # return xticks
26
+ return ax.get_xticks() / sr
27
+
28
+ def multisave(fig, path, dpi=300):
29
+ '''
30
+ save given figure in .png, .eps, .svg format
31
+ :param fig: pyplot figure instance
32
+ :param path: path of the figure without extension (.extension)
33
+ '''
34
+ if '.' in os.path.basename(path):
35
+ warnings.warn("Do not specify '.' in name. assuming it's part of the filename...")
36
+
37
+ # log.warning('Do not specify extension in multisave(). Removing extension...')
38
+ # path = os.path.splitext(path)[0]
39
+
40
+ fig.savefig(path+'.png', dpi=dpi)
41
+ fig.savefig(path+'.tiff', dpi=dpi)
42
+ fig.savefig(path+'.eps')
43
+ fig.savefig(path+'.svg')
44
+
45
+ def plot_values(valuetracker_list, ax=None):
46
+ n_line = len(valuetracker_list)
47
+ if ax is None:
48
+ fig, ax = plt.subplots()
49
+ color_list = it.cycle(mcolors.TABLEAU_COLORS)
50
+ for valuetracker, color in zip(valuetracker_list, color_list):
51
+ y_smooth = tnp.moving_mean(valuetracker.y, 9)
52
+ ax.plot(valuetracker.x, valuetracker.y, color=color, alpha=0.4)
53
+ ax.plot(valuetracker.x, y_smooth, color=color)
54
+ return ax
55
+
56
+ def plot_trainval(valuetracker_list, ax=None):
57
+ n_line = len(valuetracker_list)
58
+ if ax is None:
59
+ fig, ax = plt.subplots()
60
+ color_list = ['tab:blue', 'tab:orange']
61
+ line_list = ['-', 'x-']
62
+ labels = ['train', 'val']
63
+ for valuetracker, color, line, label in zip(valuetracker_list, color_list, line_list, labels):
64
+ y_smooth = tnp.moving_mean(valuetracker.y, 9)
65
+ ax.plot(valuetracker.x, valuetracker.y, color=color, alpha=0.4, label=label)
66
+ ax.plot(valuetracker.x, y_smooth, line, color=color, label=label+'_smooth')
67
+ return ax
68
+
69
+ if __name__ == '__main__':
70
+ pass
tools/random.py ADDED
@@ -0,0 +1,35 @@
1
+ import random
2
+
3
+ import numpy as np
4
+ import torch
5
+
6
+ # %%
7
+ def choice(a, size=None, replace=True):
8
+ '''
9
+ Randomly choose elements from given iterable.
10
+
11
+ Parameters
12
+ ----------
13
+ a: int or list
14
+ choice of
15
+
16
+ '''
17
+ m = len(a)
18
+ # Single sample
19
+ if size==None:
20
+ random_i = np.random.randint(m)
21
+ return a[random_i]
22
+ else:
23
+ num_sample = np.prod(size)
24
+
25
+ # Multi-Sample
26
+ a=np.asarray(a)
27
+ if replace==True:
28
+ random_i = np.random.randint(m, size = num_sample)
29
+ return a[random_i].reshape(size)
30
+ else:
31
+ assert m >= num_sample, 'entries of array cannot exceed number of samples'
32
+ random_i = np.arange(m)
33
+ np.random.shuffle(random_i)
34
+ random_i = random_i[:num_sample]
35
+ return a[random_i].reshape(size)
@@ -0,0 +1,2 @@
1
+
2
+ from . import model_selection, preprocessing
@@ -0,0 +1,82 @@
1
+ from collections.abc import Iterable
2
+ import warnings
3
+
4
+ import numpy as np
5
+ from sklearn.metrics import r2_score as r2_score_sklearn
6
+
7
+ def squared_error(y_true, y_pred, axis=None):
8
+ '''MSE without mean
9
+ if axis is provided, it will return the mean along the axis
10
+ '''
11
+ score = (y_true-y_pred)**2
12
+ if axis is not None:
13
+ score = np.mean(score, axis=axis)
14
+ return score
15
+
16
+ def absolute_error(y_true, y_pred, axis=None):
17
+ '''MAE without mean'''
18
+ score = np.abs(y_true-y_pred)
19
+ if axis is not None:
20
+ score = np.mean(score, axis=axis)
21
+ return score
22
+
23
+ def r2_score(y_true, y_pred, axis=None, multioutput='raw_values'):
24
+ """
25
+ R^2 score for multidimensional predictions.
26
+ collapses all axes except the specified axis.
27
+
28
+ Parameters
29
+ ----------
30
+ y_true : np.ndarray
31
+ y_pred : np.ndarray
32
+ axis: int or iterable of int, default=None
33
+ Axis to collapse.
34
+ It must be specified if y_true and y_pred dimensions are > 2
35
+ multioutput : Reference to `sklearn.metrics.r2_score`
36
+ https://scikit-learn.org/stable/modules/generated/sklearn.metrics.r2_score.html
37
+
38
+ Returns
39
+ -------
40
+ z : np.ndarray
41
+ if axis is specified, returns an array of shape (y_true.shape[axis],)
42
+ """
43
+ assert y_true.shape == y_pred.shape, f"y_true and y_pred must have the same shape, received: {y_true.shape} and {y_pred.shape}"
44
+ if axis is not None:
45
+ if not isinstance(axis, Iterable):
46
+ axis = [axis]
47
+ else:
48
+ axis = list(axis)
49
+ if len(axis) > y_true.ndim:
50
+ raise ValueError("Axis is greater than the number of dimensions of y_true and y_pred")
51
+
52
+ # shape_final = np.array(y_true.shape)[axis]
53
+ # shape_collapse = list(set(range(y_true.ndim)).difference(axis))
54
+ dim_collapse = axis
55
+ dim_final = list(set(range(y_true.ndim)).difference(axis))
56
+ shape_final = np.array(y_true.shape)[dim_final]
57
+
58
+ y_true = np.transpose(y_true, (*dim_collapse, *dim_final)).reshape(-1, np.prod(shape_final)) # Move axis to the end, and flatten the rest
59
+ y_pred = np.transpose(y_pred, (*dim_collapse, *dim_final)).reshape(-1, np.prod(shape_final))
60
+
61
+ score = r2_score_sklearn(y_true, y_pred, multioutput=multioutput)
62
+
63
+ if type(score) == float and np.isnan(score).item():
64
+ warnings.warn("R2 score is a single NaN, shape matching with NaN")
65
+ score = np.full(shape_final, np.nan)
66
+ return score
67
+
68
+ if multioutput == 'raw_values':
69
+ return score.reshape(*shape_final)
70
+ elif multioutput == 'uniform_average':
71
+ return score
72
+ elif multioutput == 'variance_weighted':
73
+ # return np.mean(score, weights=np.var(y_true, axis=0))
74
+ raise NotImplementedError("variance_weighted is not implemented yet")
75
+ else:
76
+ raise ValueError("multioutput must be one of ['raw_values', 'uniform_average', 'variance_weighted']")
77
+
78
+ else:
79
+ assert (y_true.ndim <= 2) and (y_pred.ndim <= 2), "If axis is None, y_true and y_pred must be smaller than 2D"
80
+ return r2_score_sklearn(y_true, y_pred, multioutput=multioutput)
81
+
82
+ # %%
@@ -0,0 +1,229 @@
1
+ # %%
2
+ from sklearn.model_selection import GridSearchCV
3
+ from sklearn.model_selection import StratifiedShuffleSplit, ShuffleSplit, train_test_split, StratifiedKFold, KFold
4
+ import numpy as np
5
+ import pandas as pd
6
+ import torch
7
+
8
+ from .. import torch as ttorch
9
+
10
+ # %%
11
+ '''
12
+ Train/(Validation)/Test Split index functions
13
+ '''
14
+ def stratified_train_test_split_i(y, test_size=0.15, random_state=None):
15
+ ''':return: indices of train_i, test_i'''
16
+ x = np.empty(len(y))
17
+
18
+ sss = StratifiedShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
19
+ train_i, test_i = next(sss.split(x, y))
20
+ return train_i, test_i
21
+
22
+ def stratified_train_val_test_split_i(y, val_size=0.15, test_size=0.15, random_state=None):
23
+ x = np.empty(len(y))
24
+
25
+ sss = StratifiedShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
26
+ train_val_i, test_i = next(sss.split(x, y))
27
+
28
+ train_val_y = y[train_val_i]
29
+ x = np.empty(len(train_val_i))
30
+ sss = StratifiedShuffleSplit(n_splits=1, test_size=val_size/(1-test_size), random_state=random_state)
31
+ train_i_, val_i_ = next(sss.split(x, train_val_y))
32
+ train_i = train_val_i[train_i_]
33
+ val_i = train_val_i[val_i_]
34
+
35
+ return train_i, val_i, test_i
36
+
37
+ def train_test_split_i(y, test_size=0.15, random_state=None):
38
+ x = np.empty(len(y))
39
+
40
+ ss = ShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
41
+ train_i, test_i = next(ss.split(x, y))
42
+
43
+ return train_i, test_i
44
+
45
+ def train_val_test_split_i(y, val_size=0.15, test_size=0.15, random_state=None):
46
+ x = np.empty(len(y))
47
+
48
+ ss = ShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
49
+ train_val_i, test_i = next(ss.split(x, y))
50
+
51
+ train_val_y = y[train_val_i]
52
+ x = np.empty(len(train_val_i))
53
+ ss = ShuffleSplit(n_splits=1, test_size=val_size/(1-test_size), random_state=random_state)
54
+ train_i_, val_i_ = next(ss.split(x, train_val_y))
55
+ train_i = train_val_i[train_i_]
56
+ val_i = train_val_i[val_i_]
57
+
58
+ return train_i, val_i, test_i
59
+
60
+ def stratified_kfold_split_i(y, n_splits, split_i, shuffle=True, random_state=None):
61
+ '''
62
+ return train, test indices of "split_i"-th split of "n_splits"-fold split
63
+ '''
64
+ skf = StratifiedKFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
65
+ x = np.zeros(len(y))
66
+ skf_ = skf.split(x, y)
67
+
68
+ for i in range(split_i):
69
+ next(skf_)
70
+ train_i, test_i = next(skf_)
71
+
72
+ return train_i, test_i
73
+
74
+ def stratified_nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
75
+ '''
76
+ return train, val, test indices of "split_i"-th split of "n_splits"-fold split
77
+ '''
78
+ # Outer split
79
+ skf = StratifiedKFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
80
+ x = np.zeros(len(y))
81
+ skf_ = skf.split(x, y)
82
+
83
+ for i in range(split_i):
84
+ next(skf_)
85
+ train_val_i, test_i = next(skf_)
86
+
87
+ # Inner split
88
+ skf = StratifiedKFold(n_splits=m_splits, shuffle=shuffle, random_state=random_state)
89
+ x = np.zeros(len(y[train_val_i]))
90
+ skf_ = skf.split(x, y[train_val_i])
91
+
92
+ for j in range(split_j):
93
+ next(skf_)
94
+ train_i, val_i = next(skf_)
95
+ train_i, val_i = train_val_i[train_i], train_val_i[val_i]
96
+
97
+ return train_i, val_i, test_i
98
+
99
+ def kfold_split_i(y, n_splits, split_i, shuffle=True, random_state=None):
100
+ '''
101
+ return train, test indices of "split_i"-th split of "n_splits"-fold split
102
+ '''
103
+ kf = KFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
104
+ kf_ = kf.split(y)
105
+
106
+ for i in range(split_i):
107
+ next(kf_)
108
+ train_i, test_i = next(kf_)
109
+
110
+ return train_i, test_i
111
+
112
+ def nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
113
+ '''
114
+ return train, val, test indices of "split_i"-th split of "n_splits"-fold split
115
+ '''
116
+ # Outer split
117
+ kf = KFold(n_splits=n_splits, shuffle=shuffle, random_state=random_state)
118
+ kf_ = kf.split(y)
119
+
120
+ for i in range(split_i):
121
+ next(kf_)
122
+ train_val_i, test_i = next(kf_)
123
+
124
+ # Inner split
125
+ kf = KFold(n_splits=m_splits, shuffle=shuffle, random_state=random_state)
126
+ kf_ = kf.split(y[train_val_i])
127
+
128
+ for j in range(split_j):
129
+ next(kf_)
130
+ train_i, val_i = next(kf_)
131
+ train_i, val_i = train_val_i[train_i], train_val_i[val_i]
132
+
133
+ return train_i, val_i, test_i
134
+
135
+ # %%
136
+ '''
137
+ Train/(Validation)/Test split dataset functions
138
+ '''
139
+ def nested_kfold_split_data(data, y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
140
+ '''
141
+ y doesn't matter since it's not stratified
142
+ '''
143
+ train_i, val_i, test_i = nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle, random_state)
144
+ train_data, val_data, test_data = wrap_data(data, train_i), wrap_data(data, val_i), wrap_data(data, test_i)
145
+ return train_data, val_data, test_data
146
+
147
+ def stratified_nested_kfold_split_data(data, y, n_splits, m_splits, split_i, split_j, shuffle=True, random_state=None):
148
+ """
149
+ Split data into train, validation, and test set
150
+
151
+ Parameters
152
+ ----------
153
+ data : pd.DataFrame or np.ndarray or torch.Tensor or torch.dat.Dataset object or iterable object (i.e. list)
154
+ Data to split
155
+
156
+ y: pd.Series or np.ndarray or torch.Tensor or iterable object (i.e. list)
157
+ Target variable to be used in stratified split
158
+
159
+ Returns
160
+ -------
161
+ train_data : train data of the same variable format
162
+ val_data : validation data of the same variable format
163
+ test_data : test data of the same variable format
164
+ """
165
+ train_i, val_i, test_i = stratified_nested_kfold_split_i(y, n_splits, m_splits, split_i, split_j, shuffle, random_state)
166
+ train_data, val_data, test_data = wrap_data(data, train_i), wrap_data(data, val_i), wrap_data(data, test_i)
167
+ return train_data, val_data, test_data
168
+
169
+ def wrap_data(data, i):
170
+ if isinstance(data, np.ndarray):
171
+ return data[i]
172
+ elif isinstance(data, pd.DataFrame):
173
+ return data.iloc[i]
174
+ elif isinstance(data, torch.Tensor):
175
+ return data[i]
176
+ elif isinstance(data, torch.utils.data.Dataset):
177
+ return ttorch.data.ProxyDataset(dataset=data, idxs=i)
178
+ else:
179
+ return data[i]
180
+
181
+ # %%
182
+ # def train_val_test_split(x, val_size=0.1, test_size=0.1, random_state=None):
183
+ # if type(x)==int:
184
+ # x = np.zeros(x)
185
+ # ss = ShuffleSplit(n_splits=1, test_size=test_size, random_state=random_state)
186
+ # train_val_i, test_i = next(ss.split(x))
187
+ # train_i, val_i = train_test_split(train_val_i, test_size=val_size/(1-test_size), random_state=random_state)
188
+ #
189
+ # # assert len(set(np.concatenate((train_i, val_i, test_i)))) == len(x)
190
+ # return train_i, val_i, test_i
191
+
192
+ class MultiGridSearchCV():
193
+ '''Perform Grid Search over multiple models
194
+
195
+ Parameters
196
+ ----------
197
+ estimators: dict
198
+ Name of model. This name has to correspond to model name of param_grid
199
+ param_grids: dict
200
+ Parameter Grid to Search. It must be dict nested in dict(Double dict).
201
+ **kwargs: other parameters from GridSearchCV
202
+ Look GridSearchCV
203
+
204
+ examples of **kwargs
205
+ --------------------
206
+ cv: number of k-fold CV
207
+ scoring: scoring method. either can be prebuilt sklearn string or a callable function
208
+
209
+ '''
210
+ def __init__(self, estimators, param_grids, **kwargs):
211
+ self.estimators = estimators
212
+ self.param_grids = param_grids
213
+ self.kwargs = kwargs
214
+ self.grid_searches = {estimator_name: GridSearchCV(estimator=estimator, param_grid=self.param_grids[estimator_name], **self.kwargs) for estimator_name, estimator in self.estimators.items()}
215
+
216
+ def fit(self, x, y):
217
+ for grid_search in self.grid_searches.values():
218
+ grid_search.fit(x,y)
219
+ self.cv_results = {estimator_name: grid_search.cv_results_ for estimator_name, grid_search in self.grid_searches.items()}
220
+ best_estimator_name_, best_grid_search_ = max(self.grid_searches.items(), key = lambda x: x[1].best_score_) # x[0]: key, x[1]: value
221
+ self.best_score_ = best_grid_search_.best_score_
222
+ self.best_estimator_ = best_grid_search_.best_estimator_
223
+ self.best_params_ = best_grid_search_.best_params_
224
+ self.best_estimator_name_ = best_estimator_name_
225
+
226
+ def predict(self, x):
227
+ assert hasattr(self, 'best_estimator_'), 'There is no best_estimator_. Need to call "fit" first.'
228
+ assert hasattr(self.best_estimator_,'predict'), 'Best estimator does not support "predict" method.'
229
+ return self.best_estimator_.predict(x)
@@ -0,0 +1,125 @@
1
+ from sklearn.preprocessing import StandardScaler, MinMaxScaler
2
+ from sklearn.decomposition import PCA
3
+ import numpy as np
4
+
5
+ def standardize(x, scaler=None):
6
+ """
7
+ Standardize (z-score/gaussian normalization) x without having to make a StandardScaler() object.
8
+
9
+ Parameters
10
+ ----------
11
+ x : ndarray of shape (n_observation, n_channel)
12
+ data to be standardized.
13
+ scaler: sklearn.preprocessing.StandardScaler(), default=None
14
+ scaler to use (that has already fit())
15
+
16
+ Returns
17
+ -------
18
+ x_standardized : ndarray of shape (n_observation, n_channel)
19
+ Normalized data
20
+ scaler : sklearn.preprocessing.StandardScaler()
21
+ Scaler used for normalization
22
+ """
23
+ if scaler is None:
24
+ scaler = StandardScaler()
25
+ x_standardized = scaler.fit_transform(x)
26
+ else:
27
+ x_standardized = scaler.transform(x)
28
+
29
+ return x_standardized, scaler
30
+
31
+ def project_pca(x, var_threshold=None, n_pc=None, solver=None, random_state=0):
32
+ """
33
+ project x onto PC space without having to make a PCA() object.
34
+
35
+ Parameters
36
+ ----------
37
+ x : ndarray of shape (n_observation, n_channel)
38
+ data to be projected.
39
+ solver: sklearn.decomposition.PCA(), default=None
40
+ solver to use (that has already fit())
41
+
42
+ Returns
43
+ -------
44
+ x_projected : ndarray of shape (n_observation, n_channel)
45
+ Normalized data
46
+ solver : sklearn.decomposition.PCA()
47
+ Solver used for PCA projection
48
+ """
49
+ # If svd_solver is not full, result is random
50
+ if solver is None:
51
+ solver = PCA(svd_solver='full', random_state=random_state)
52
+ x_pc = solver.fit_transform(x)
53
+ else:
54
+ x_pc = solver.transform(x)
55
+
56
+ if var_threshold is not None:
57
+ n_pc = np.argmax(np.cumsum(solver.explained_variance_ratio_) > var_threshold) + 1
58
+ elif n_pc is not None:
59
+ pass
60
+ else:
61
+ n_pc = solver.n_components_
62
+ x_pc = x_pc[:,:n_pc]
63
+
64
+ return x_pc, solver
65
+
66
+ # def minmaxscale(x, scaler=None, feature_range=(0, 1), *, copy=True, clip=False):
67
+ def minmaxscale(x, scaler=None, **kwargs):
68
+ """
69
+ Min-max scale x without having to make a MinMaxScaler() object.
70
+
71
+ Parameters
72
+ ----------
73
+ x : ndarray of shape (n_observation, n_channel)
74
+ data to be standardized.
75
+ scaler: sklearn.preprocessing.MinMaxScaler(), default=None
76
+ scaler to use (that has already fit())
77
+
78
+
79
+ Following arguments are applied if scaler is None:
80
+
81
+ feature_range : tuple (min, max), default=(0, 1)
82
+ Desired range of transformed data.
83
+ The default is (0, 1).
84
+ copy : bool, default=True
85
+ Set to False to perform inplace normalization and avoid a copy (if the input is already a numpy array).
86
+ clip : bool, default=False
87
+ Set to True to clip transformed values of held-out data.
88
+
89
+ Returns
90
+ -------
91
+ x_scaled : ndarray of shape (n_observation, n_channel)
92
+ Normalized data
93
+ scaler : sklearn.preprocessing.MinMaxScaler()
94
+ Scaler used for normalization
95
+ """
96
+ if scaler is None:
97
+ # scaler = MinMaxScaler(feature_range=feature_range, copy=copy, clip=clip)
98
+ scaler = MinMaxScaler(**kwargs)
99
+ x_scaled = scaler.fit_transform(x)
100
+ else:
101
+ x_scaled = scaler.transform(x)
102
+
103
+ return x_scaled, scaler
104
+
105
+ if __name__=='__main__':
106
+ import tools as T
107
+
108
+ x = np.random.rand(2,3,5)
109
+ print(x.shape)
110
+ squeezer = T.numpy.Squeezer()
111
+ x = squeezer.squeeze(x)
112
+ print(x.shape)
113
+ x_s, scaler = T.sklearn.preprocessing.standardize(x)
114
+ x_s = squeezer.unsqueeze(x_s)
115
+ print(x_s.shape)
116
+
117
+ # %%
118
+ x = np.random.rand(2,3,5)
119
+ print(x.shape)
120
+ squeezer = T.numpy.Squeezer()
121
+ x = squeezer.squeeze(x)
122
+ print(x.shape)
123
+ x_pc, solver = T.sklearn.preprocessing.project_pca(x, n_pc=3)
124
+ x_pc = squeezer.unsqueeze(x_pc, strict=False)
125
+ print(x_pc.shape)
@@ -0,0 +1,43 @@
1
+ import numpy as np
2
+ import scipy.stats as st
3
+
4
+ def ci(std, n, p=0.95):
5
+ z = st.norm.ppf((1+p)/2) # Two-tailed confidence interval
6
+ half_range = z*std/np.sqrt(n)
7
+ return half_range
8
+
9
+ def interval(distrib, confidence, **kwargs):
10
+ if 'std' in kwargs and 'n' in kwargs:
11
+ scale = kwargs['std']/np.sqrt(kwargs['n'])
12
+ assert 'scale' not in kwargs, 'scale is given but std and n are also given'
13
+ kwargs['scale'] = scale
14
+ del kwargs['std'], kwargs['n']
15
+
16
+ return distrib.interval(confidence, **kwargs)
17
+
18
+ def ci_stats(mean, std, n, p=0.95):
19
+ '''
20
+ calculate confidence interval based on given statistics
21
+ :param mean: 1d array, sample mean
22
+ :param std: 1d array, sample mean
23
+ :param n: 1d array, sample size
24
+ :param p: 1d array, confidence level
25
+ '''
26
+ # z = st.norm.ppf(p)
27
+ # half_range = z*std/n
28
+ half_range = ci(std=std, n=n)
29
+
30
+ confidence_interval = {
31
+ 'low': mean-half_range,
32
+ 'high': mean+half_range
33
+ }
34
+ return confidence_interval
35
+
36
+ def ci_samples(x, p=0.95):
37
+ '''
38
+ calculate confidence interval based on samples
39
+ '''
40
+ mean = np.mean(x)
41
+ std = np.std(x)
42
+ n = len(x)
43
+ return ci_stats(mean, std, n, p=p)