pyrolite 0.0.14__zip

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__init__.py +10 -0
  2. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/__init__.cpython-36.pyc +0 -0
  3. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/_version.cpython-36.pyc +0 -0
  4. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/alteration.cpython-36.pyc +0 -0
  5. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/classification.cpython-36.pyc +0 -0
  6. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/compositions.cpython-36.pyc +0 -0
  7. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/geochem.cpython-36.pyc +0 -0
  8. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/melts.cpython-36.pyc +0 -0
  9. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/norm.cpython-36.pyc +0 -0
  10. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/normalisation.cpython-36.pyc +0 -0
  11. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/plot.cpython-36.pyc +0 -0
  12. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/_version.py +21 -0
  13. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/alteration.py +66 -0
  14. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/classification.py +222 -0
  15. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__init__.py +9 -0
  16. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/__init__.cpython-36.pyc +0 -0
  17. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/aggregate.cpython-36.pyc +0 -0
  18. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/codata.cpython-36.pyc +0 -0
  19. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/impute.cpython-36.pyc +0 -0
  20. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/renorm.cpython-36.pyc +0 -0
  21. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/aggregate.py +391 -0
  22. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/codata.py +266 -0
  23. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/impute.py +82 -0
  24. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/renorm.py +40 -0
  25. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/compositions.py +524 -0
  26. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_CFB_Dataset_List.csv +42 -0
  27. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_Convergent_Dataset_List.csv +42 -0
  28. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OBFB_Dataset_List.csv +5 -0
  29. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OIB_Dataset_List.csv +49 -0
  30. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OceanicPlateau_Dataset_List.csv +18 -0
  31. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/contents.json +1 -0
  32. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-35.pyc +0 -0
  33. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-36.pyc +0 -0
  34. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/env.py +1063 -0
  35. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ba.modelfield +0 -0
  36. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Bs.modelfield +0 -0
  37. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.F.modelfield +0 -0
  38. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O1.modelfield +0 -0
  39. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O2.modelfield +0 -0
  40. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O3.modelfield +0 -0
  41. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Pc.modelfield +0 -0
  42. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ph.modelfield +0 -0
  43. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.R.modelfield +0 -0
  44. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S1.modelfield +0 -0
  45. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S2.modelfield +0 -0
  46. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S3.modelfield +0 -0
  47. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T1.modelfield +0 -0
  48. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T2.modelfield +0 -0
  49. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U1.modelfield +0 -0
  50. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U2.modelfield +0 -0
  51. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U3.modelfield +0 -0
  52. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.modelfields +0 -0
  53. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.nan.modelfield +0 -0
  54. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.none.modelfield +0 -0
  55. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS.clsf.gz +0 -0
  56. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/aphanitic.clsf.gz +0 -0
  57. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/gabbroic.clsf.gz +0 -0
  58. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/peralkalinity.clsf.gz +0 -0
  59. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/phaneritic.clsf.gz +0 -0
  60. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/ultramafic.clsf.gz +0 -0
  61. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/CH_PalmeONeill2014.csv +95 -0
  62. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DDMM_WorkmanHart2005.csv +105 -0
  63. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DMM_WorkmanHart2005.csv +105 -0
  64. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DM_SaltersStrake2004.csv +95 -0
  65. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/EDMM_WorkmanHart2005.csv +105 -0
  66. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/PM_PalmeONeill2014.csv +95 -0
  67. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/timescale/geotimescale_spans.csv +180 -0
  68. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/geochem.py +821 -0
  69. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/melts.py +92 -0
  70. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__init__.py +10 -0
  71. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/__init__.cpython-36.pyc +0 -0
  72. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/db.cpython-36.pyc +0 -0
  73. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/ions.cpython-36.pyc +0 -0
  74. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/mineral.cpython-36.pyc +0 -0
  75. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/sites.cpython-36.pyc +0 -0
  76. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/db.py +88 -0
  77. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/ions.py +78 -0
  78. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/mineral.py +587 -0
  79. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/sites.py +134 -0
  80. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/norm.py +224 -0
  81. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/normalisation.py +204 -0
  82. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/plot.py +514 -0
  83. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__init__.py +13 -0
  84. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/__init__.cpython-36.pyc +0 -0
  85. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/database.cpython-36.pyc +0 -0
  86. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/env.cpython-36.pyc +0 -0
  87. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/general.cpython-36.pyc +0 -0
  88. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/georoc.cpython-36.pyc +0 -0
  89. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/math.cpython-36.pyc +0 -0
  90. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/melts.cpython-36.pyc +0 -0
  91. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multip.cpython-36.pyc +0 -0
  92. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multiprocessing.cpython-36.pyc +0 -0
  93. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/pd.cpython-36.pyc +0 -0
  94. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/plot.cpython-36.pyc +0 -0
  95. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/skl.cpython-36.pyc +0 -0
  96. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/spatial.cpython-36.pyc +0 -0
  97. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/text.cpython-36.pyc +0 -0
  98. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/time.cpython-36.pyc +0 -0
  99. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/wfs.cpython-36.pyc +0 -0
  100. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/database.py +88 -0
  101. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/env.py +81 -0
  102. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/general.py +266 -0
  103. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/georoc.py +444 -0
  104. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/math.py +371 -0
  105. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/melts.py +397 -0
  106. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multip.py +29 -0
  107. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multiprocessing.py +29 -0
  108. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/pd.py +214 -0
  109. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/plot.py +345 -0
  110. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/skl.py +847 -0
  111. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/spatial.py +91 -0
  112. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/text.py +207 -0
  113. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/time.py +224 -0
  114. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/wfs.py +10 -0
  115. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/PKG-INFO +61 -0
  116. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/SOURCES.txt +83 -0
  117. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/dependency_links.txt +1 -0
  118. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/requires.txt +47 -0
  119. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/top_level.txt +1 -0
@@ -0,0 +1,391 @@
1
+ import numpy as np
2
+ import pandas as pd
3
+ import warnings
4
+ from .codata import alr, inv_alr
5
+ from .impute import *
6
+
7
+ import logging
8
+
9
+ logging.getLogger(__name__).addHandler(logging.NullHandler())
10
+ logger = logging.getLogger(__name__)
11
+
12
+
13
+ def get_full_column(X: np.ndarray):
14
+ """
15
+ Returns the index of the first array column which contains only finite
16
+ numbers (i.e. no missing data, nan, inf).
17
+
18
+ Parameters
19
+ ---------------
20
+ X: np.ndarray
21
+ Array for which to find the first full column within.
22
+ """
23
+ if len(X.shape) == 1:
24
+ X = X.reshape((1, *X.shape))
25
+ inds = np.arange(X.shape[1])
26
+ wherenonnan = np.isfinite(X).all(axis=0)
27
+ ind = inds[wherenonnan][0]
28
+ return ind
29
+
30
+
31
+ def weights_from_array(X: np.ndarray):
32
+ """
33
+ Returns a set of equal weights with size equal to that of the first axis of
34
+ an array.
35
+
36
+ Parameters
37
+ ---------------
38
+ X: np.ndarray
39
+ Array of compositions to produce weights for.
40
+ """
41
+ wts = np.ones((X.shape[0]))
42
+ return wts / np.sum(wts)
43
+
44
+
45
+ def nan_weighted_mean(X: np.ndarray, weights=None):
46
+ """
47
+ Returns a weighted mean of compositions, where weights are renormalised
48
+ to account for missing data.
49
+
50
+ Parameters
51
+ ---------------
52
+ X: np.ndarray
53
+ Array of compositions to take a weighted mean of.
54
+ weights: np.ndarray
55
+ Array of weights.
56
+ """
57
+ if weights is None:
58
+ weights = weights_from_array(X)
59
+ weights = np.array(weights) / np.nansum(weights)
60
+
61
+ mask = (np.isnan(X) + np.isinf(X)) > 0
62
+ if not mask.any():
63
+ return np.average(X, weights=weights, axis=0)
64
+ else:
65
+ return np.ma.average(np.ma.array(X, mask=mask), weights=weights, axis=0)
66
+
67
+
68
+ def compositional_mean(df, weights=[], **kwargs):
69
+ """
70
+ Implements an aggregation using a compositional weighted mean.
71
+
72
+ Parameters
73
+ ---------------
74
+ df: pd.DataFrame
75
+ Dataframe of compositions to aggregate.
76
+ weights: np.ndarray
77
+ Array of weights.
78
+ """
79
+ non_nan_cols = df.dropna(axis=1, how="all").columns
80
+ assert not df.loc[:, non_nan_cols].isna().values.any()
81
+ mean = df.iloc[0, :].copy()
82
+ if not weights:
83
+ weights = np.ones(len(df.index.values))
84
+ weights = np.array(weights) / np.nansum(weights)
85
+
86
+ logmean = alr(df.loc[:, non_nan_cols].values).T @ weights[:, np.newaxis]
87
+ # this renormalises by default
88
+ mean.loc[non_nan_cols] = inv_alr(logmean.T.squeeze())
89
+ return mean
90
+
91
+
92
+ def nan_weighted_compositional_mean(
93
+ X: np.ndarray, weights=None, ind=None, renorm=True, **kwargs
94
+ ):
95
+ """
96
+ Implements an aggregation using a weighted mean, but accounts
97
+ for nans. Requires at least one non-nan column for alr mean.
98
+
99
+ When used for internal standardisation, there should be only a single
100
+ common element - this would be used by default as the divisor here. When
101
+ used for multiple-standardisation, the [specified] or first common
102
+ element will be used.
103
+
104
+ Parameters
105
+ ---------------
106
+ X: np.ndarray
107
+ Array of compositions to aggregate.
108
+ weights: np.ndarray
109
+ Array of weights.
110
+ ind: int
111
+ Index of the column to use as the alr divisor.
112
+ renorm: bool, True
113
+ Whether to renormalise the output compositional mean to unity.
114
+
115
+ Returns
116
+ -------
117
+ np.ndarray
118
+ An array with the mean composition.
119
+ """
120
+ if X.ndim == 1: # if it's a single row
121
+ return X
122
+ else:
123
+ if weights is None:
124
+ weights = weights_from_array(X)
125
+ else:
126
+ weights = np.array(weights) / np.sum(weights, axis=-1)
127
+
128
+ if ind is None: # take the first column which has no nans
129
+ ind = get_full_column(X)
130
+
131
+ if X.ndim < 3 and X.shape[0] == 1:
132
+ div = X[:, ind].squeeze() # check this
133
+ else:
134
+ div = X[:, ind].squeeze()[:, np.newaxis]
135
+
136
+ logvals = np.log(np.divide(X, div))
137
+ mean = np.nan * np.ones(X.shape[1:])
138
+
139
+ ixs = np.arange(logvals.shape[1])
140
+ if X.ndim == 2:
141
+ indexes = ixs
142
+ elif X.ndim == 3:
143
+ iys = np.arange(logvals.shape[2])
144
+ indexes = np.ixs_(ixs, iys)
145
+
146
+ mean[indexes] = nan_weighted_mean(logvals[:, indexes], weights=weights)
147
+
148
+ mean = np.exp(mean.squeeze())
149
+ if renorm:
150
+ mean /= np.nansum(mean)
151
+ return mean
152
+
153
+
154
+ def cross_ratios(df: pd.DataFrame):
155
+ """
156
+ Takes ratios of values across a dataframe, such that columns are
157
+ denominators and the row indexes the numerators, to create a square array.
158
+ Returns one array per record.
159
+
160
+ Parameters
161
+ ---------------
162
+ df: pd.DataFrame
163
+ Dataframe of compositions to create ratios of.
164
+
165
+ Returns
166
+ -------
167
+ np.ndarray
168
+ A 3D array of ratios.
169
+ """
170
+ ratios = np.ones((len(df.index), len(df.columns), len(df.columns)))
171
+ for idx in range(df.index.size):
172
+ row_vals = df.iloc[idx, :].values
173
+ r1 = row_vals.T[:, np.newaxis] @ np.ones_like(row_vals)[np.newaxis, :]
174
+ ratios[idx] = r1 / r1.T
175
+ return ratios
176
+
177
+
178
+ def np_cross_ratios(X: np.ndarray, debug=False):
179
+ """
180
+ Takes ratios of values across an array, such that columns are
181
+ denominators and the row indexes the numerators, to create a square array.
182
+ Returns one array per record.
183
+
184
+ Parameters
185
+ ---------------
186
+ X: np.ndarray
187
+ Array of compositions to create ratios of.
188
+
189
+ Returns
190
+ -------
191
+ np.ndarray
192
+ A 3D array of ratios.
193
+ """
194
+ X = X.copy()
195
+ with warnings.catch_warnings():
196
+ # can get invalid values which raise RuntimeWarnings
197
+ # consider changing to np.errstate
198
+ warnings.simplefilter("ignore", category=RuntimeWarning)
199
+ X[X <= 0] = np.nan
200
+ if X.ndim == 1:
201
+ index_length = 1
202
+ X = X.reshape((1, *X.shape))
203
+ else:
204
+ index_length = X.shape[0]
205
+ dims = X.shape[-1]
206
+ ratios = np.ones((index_length, dims, dims))
207
+ for idx in range(index_length):
208
+ row_vals = X[idx, :]
209
+ r1 = row_vals.T[:, np.newaxis] @ np.ones_like(row_vals)[np.newaxis, :]
210
+ ratios[idx] = r1.T / r1
211
+
212
+ if debug:
213
+ try:
214
+ diags = ratios[:, np.arange(dims), np.arange(dims)]
215
+ # check all diags are 1.
216
+ assert np.allclose(diags, 1.0)
217
+ except:
218
+ # check all diags are 1. or nan
219
+ assert np.allclose(diags[~np.isnan(diags)], 1.0)
220
+
221
+ return ratios
222
+
223
+
224
+ def standardise_aggregate(
225
+ df: pd.DataFrame, int_std=None, fixed_record_idx=0, renorm=True, **kwargs
226
+ ):
227
+ """
228
+ Performs internal standardisation and aggregates dissimilar geochemical
229
+ records. Note: this changes the closure parameter, and is generally
230
+ intended to integrate major and trace element records.
231
+
232
+ Parameters
233
+ ---------------
234
+ df: pd.DataFrame
235
+ Dataframe of compositions to aggregate of.
236
+ int_std: str
237
+ Name of the internal standard column.
238
+ fixed_record_idx: int
239
+ Numeric index of a specific record's for which to retain the internal
240
+ standard value (e.g for standardising trace element data).
241
+ renorm: bool, True
242
+ Whether to renormalise to unity.
243
+
244
+ Returns
245
+ -------
246
+ pd.Series
247
+ A series representing the internally standardised record.
248
+ """
249
+ if df.index.size == 1: # catch single records
250
+ return df
251
+ else:
252
+ if int_std is None:
253
+ # Get the 'internal standard column'
254
+ potential_int_stds = df.count()[df.count() == df.count().max()].index.values
255
+ assert len(potential_int_stds) > 0
256
+ # Use an internal standard
257
+ int_std = potential_int_stds[0]
258
+ if len(potential_int_stds) > 1:
259
+ logging.info("Multiple int. stds possible. Using " + str(int_std))
260
+
261
+ non_nan_cols = df.dropna(axis=1, how="all").columns
262
+ assert len(non_nan_cols)
263
+ mean = nan_weighted_compositional_mean(
264
+ df.values, ind=df.columns.get_loc(int_std), renorm=False
265
+ )
266
+ ser = pd.Series(mean, index=df.columns)
267
+ multiplier = (
268
+ df.iloc[fixed_record_idx, df.columns.get_loc(int_std)] / ser[int_std]
269
+ )
270
+ ser *= multiplier
271
+ if renorm:
272
+ ser /= np.nansum(ser.values)
273
+ return ser
274
+
275
+
276
+ def complex_standardise_aggregate(
277
+ df, int_std=None, fixed_record_idx=0, renorm=True # fallback parameters
278
+ ):
279
+ """
280
+ Aggregation function which attempts to incorporate all ratio information
281
+ but creating a ratio matrix for each record which is then imputed for
282
+ missing data where possible. Falls back to internal standardistion where
283
+ int_std is specified.
284
+
285
+ Parameters
286
+ ---------------
287
+ df: pd.DataFrame
288
+ Dataframe of compositions to aggregate of.
289
+ int_std: str
290
+ Name of the internal standard column.
291
+ fixed_record_idx: int
292
+ Numeric index of a specific record's for which to retain the internal
293
+ standard value (e.g for standardising trace element data).
294
+ renorm: bool, True
295
+ Whether to renormalise to unity.
296
+
297
+ Returns
298
+ -------
299
+ pd.Series
300
+ A series representing the internally standardised record.
301
+ """
302
+ if int_std is None:
303
+ # create a n x d x d matrix for aggregating ratios
304
+ non_nan_cols = df.dropna(axis=1, how="all").columns
305
+ ratios = cross_ratios(df.loc[:, non_nan_cols])
306
+ # Average across record matricies
307
+ with warnings.catch_warnings():
308
+ # can get empty arrays which raise RuntimeWarnings
309
+ # consider changing to np.errstate
310
+ warnings.simplefilter("ignore", category=RuntimeWarning)
311
+ mean_ratios = pd.DataFrame(
312
+ np.exp(np.nanmean(np.log(ratios), axis=0)),
313
+ columns=non_nan_cols,
314
+ index=non_nan_cols,
315
+ )
316
+ # Filling in the null values in a ratio matrix
317
+ imputed_ratios = impute_ratios(mean_ratios)
318
+ # We simply pick the first non-nan column.
319
+ IS = non_nan_cols[0]
320
+ mean = np.exp(
321
+ np.mean(np.log(imputed_ratios / imputed_ratios.loc[IS, :]), axis=1)
322
+ )
323
+ # This needs to be renormalised to make logical sense
324
+ mean /= np.nansum(mean.values)
325
+
326
+ out = np.ones((1, len(df.columns))) * np.nan
327
+ out[:, [list(df.columns).index(c) for c in non_nan_cols]] = mean
328
+ return pd.Series(out.squeeze(), index=df.columns)
329
+ else:
330
+ # fallback to internal standardisation
331
+ return standardise_aggregate(
332
+ df, int_std=int_std, fixed_record_idx=fixed_record_idx, renorm=renorm
333
+ )
334
+
335
+
336
+ def np_complex_standardise_aggregate(
337
+ df, int_std=None, renorm=True, fixed_record_idx=0 # fallback parameters
338
+ ):
339
+ """
340
+ Aggregation function which attempts to incorporate all ratio information
341
+ but creating a ratio matrix for each record which is then imputed for
342
+ missing data where possible. Falls back to internal standardistion where
343
+ int_std is specified. Uses numpy more extensively.
344
+
345
+ Parameters
346
+ ---------------
347
+ df: pd.DataFrame
348
+ Dataframe of compositions to aggregate of.
349
+ int_std: str
350
+ Name of the internal standard column.
351
+ fixed_record_idx: int
352
+ Numeric index of a specific record's for which to retain the internal
353
+ standard value (e.g for standardising trace element data).
354
+ renorm: bool, True
355
+ Whether to renormalise to unity.
356
+
357
+ Returns
358
+ -------
359
+ pd.Series
360
+ A series representing the internally standardised record.
361
+ """
362
+
363
+ if int_std is None:
364
+ # create a n x d x d matrix for aggregating ratios
365
+ non_nan_cols = df.dropna(axis=1, how="all").columns
366
+ assert len(non_nan_cols) > 0
367
+ ratios = np_cross_ratios(df.loc[:, non_nan_cols].values)
368
+ # Take the mean across the cross-ratio matricies
369
+ with warnings.catch_warnings():
370
+ # can get empty arrays which raise RuntimeWarnings
371
+ # consider changing to np.errstate
372
+ warnings.simplefilter("ignore", category=RuntimeWarning)
373
+ mean_logratios = np.nanmean(np.log(ratios), axis=0)
374
+ # Filling in the null values in a ratio matrix
375
+ imputed_log_ratios = np_impute_ratios(mean_logratios)
376
+ # We simply pick the first non-nan column.
377
+ # IS = 0
378
+ IS = np.argmax(np.count_nonzero(~np.isnan(imputed_log_ratios), axis=0))
379
+ # Convert to a composition by subtracting a row and taking negative
380
+ div_log_ratios = -(imputed_log_ratios - imputed_log_ratios[IS, :])
381
+ comp_abund = np.exp(np.nanmean(div_log_ratios, axis=1))
382
+ comp_abund /= np.nansum(comp_abund)
383
+ out = np.ones((1, len(df.columns))) * np.nan
384
+ inds = np.array([list(df.columns).index(c) for c in non_nan_cols])
385
+ out[:, inds] = comp_abund
386
+ return pd.Series(out.squeeze(), index=df.columns)
387
+ else:
388
+ # fallback to internal standardisation
389
+ return standardise_aggregate(
390
+ df, int_std=int_std, fixed_record_idx=fixed_record_idx, renorm=renorm
391
+ )
@@ -0,0 +1,266 @@
1
+ import numpy as np
2
+ import pandas as pd
3
+ import pandas_flavor as pf
4
+ import scipy.stats as scpstats
5
+ import scipy.special as scpspec
6
+ #from .renorm import renormalise, close
7
+ from ..util.math import orthagonal_basis
8
+ import logging
9
+
10
+ logging.getLogger(__name__).addHandler(logging.NullHandler())
11
+ logger = logging.getLogger(__name__)
12
+
13
+
14
+ def close(X: np.ndarray):
15
+ if X.ndim == 2:
16
+ return np.divide(X, np.sum(X, axis=1)[:, np.newaxis])
17
+ else:
18
+ return np.divide(X, np.sum(X, axis=0))
19
+
20
+ @pf.register_series_method
21
+ @pf.register_dataframe_method
22
+ def renormalise(df: pd.DataFrame, components: list = [], scale=100.0):
23
+ """
24
+ Renormalises compositional data to ensure closure.
25
+
26
+ Parameters
27
+ ------------
28
+ df: pd.DataFrame
29
+ Dataframe to renomalise.
30
+ components: list
31
+ Option subcompositon to renormalise to 100. Useful for the use case
32
+ where compostional data and non-compositional data are stored in the
33
+ same dataframe.
34
+ scale: float, 100.
35
+ Closure parameter. Typically either 100 or 1.
36
+ """
37
+ dfc = df.copy(deep=True)
38
+ if components:
39
+ cmpnts = [c for c in components if c in dfc.columns]
40
+ dfc.loc[:, cmpnts] = scale * dfc.loc[:, cmpnts].divide(
41
+ dfc.loc[:, cmpnts].sum(axis=1).replace(0, np.nan), axis=0
42
+ )
43
+ return dfc
44
+ else:
45
+ dfc = dfc.divide(dfc.sum(axis=1).replace(0, 100.0), axis=0) * scale
46
+ return dfc
47
+
48
+
49
+ def additive_log_ratio(X: np.ndarray, ind: int = -1, null_col=False):
50
+ """
51
+ Inverse Additive Log Ratio transformation.
52
+
53
+ Parameters
54
+ ---------------
55
+ X: np.ndarray
56
+ Array on which to perform the inverse transformation.
57
+ ind: int
58
+ Index of column used as denominator.
59
+ """
60
+
61
+ Y = X.copy()
62
+ assert Y.ndim in [1, 2]
63
+ dimensions = Y.shape[Y.ndim - 1]
64
+ if ind < 0:
65
+ ind += dimensions
66
+
67
+ if Y.ndim == 2:
68
+ Y = np.divide(Y, Y[:, ind][:, np.newaxis])
69
+ if not null_col:
70
+ Y = Y[:, [i for i in range(dimensions) if not i == ind]]
71
+ else:
72
+ Y = np.divide(X, X[ind])
73
+ if not null_col:
74
+ Y = Y[[i for i in range(dimensions) if not i == ind]]
75
+
76
+ return np.log(Y)
77
+
78
+
79
+ def inverse_additive_log_ratio(Y: np.ndarray, ind=-1, null_col=False):
80
+ """
81
+ Inverse Centred Log Ratio transformation.
82
+
83
+ Parameters
84
+ ---------------
85
+ X: np.ndarray
86
+ Array on which to perform the inverse transformation.
87
+ ind: int
88
+ Index of column used as denominator.
89
+ """
90
+ assert Y.ndim in [1, 2]
91
+
92
+ X = Y.copy()
93
+ dimensions = X.shape[X.ndim - 1]
94
+ if not null_col:
95
+ idx = np.arange(0, dimensions + 1)
96
+
97
+ if ind != -1:
98
+ idx = np.array(list(idx[idx < ind]) + [-1] + list(idx[idx >= ind + 1] - 1))
99
+
100
+ # Add a zero-column and reorder columns
101
+ if Y.ndim == 2:
102
+ X = np.concatenate((X, np.zeros((X.shape[0], 1))), axis=1)
103
+ X = X[:, idx]
104
+ else:
105
+ X = np.append(X, np.array([0]))
106
+ X = X[idx]
107
+
108
+ # Inverse log and closure operations
109
+ X = np.exp(X)
110
+ X = close(X)
111
+ return X
112
+
113
+
114
+ def alr(*args, **kwargs):
115
+ """
116
+ Short form of Additive Log Ratio transformation.
117
+
118
+ Parameters
119
+ ---------------
120
+ Y: np.ndarray
121
+ Array on which to perform the inverse transformation.
122
+ ind: int
123
+ Index of column used as denominator.
124
+ """
125
+ return additive_log_ratio(*args, **kwargs)
126
+
127
+
128
+ def inv_alr(*args, **kwargs):
129
+ """
130
+ Short form of Inverse Additive Log Ratio transformation.
131
+
132
+ Parameters
133
+ ---------------
134
+ Y: np.ndarray
135
+ Array on which to perform the inverse transformation.
136
+ ind: int
137
+ Index of column used as denominator.
138
+ """
139
+ return inverse_additive_log_ratio(*args, **kwargs)
140
+
141
+
142
+ def clr(X: np.ndarray):
143
+ """
144
+ Centred Log Ratio transformation.
145
+
146
+ Parameters
147
+ ---------------
148
+ X: np.ndarray
149
+ Array on which to perform the transformation.
150
+ """
151
+ X = np.divide(X, np.sum(X, axis=1)[:, np.newaxis]) # Closure operation
152
+ Y = np.log(X) # Log operation
153
+ Y -= 1 / X.shape[1] * np.nansum(Y, axis=1)[:, np.newaxis]
154
+ return Y
155
+
156
+
157
+ def inv_clr(Y: np.ndarray):
158
+ """
159
+ Inverse Centred Log Ratio transformation.
160
+
161
+ Parameters
162
+ ---------------
163
+ Y: np.ndarray
164
+ Array on which to perform the inverse transformation.
165
+ """
166
+ # Inverse of log operation
167
+ X = np.exp(Y)
168
+ # Closure operation
169
+ X = np.divide(X, np.nansum(X, axis=1)[:, np.newaxis])
170
+ return X
171
+
172
+
173
+ def ilr(X: np.ndarray):
174
+ """
175
+ Isotmetric Log Ratio transformation.
176
+
177
+ Parameters
178
+ ---------------
179
+ X: np.ndarray
180
+ Array on which to perform the transformation.
181
+ """
182
+ d = X.shape[1]
183
+ Y = clr(X)
184
+ psi = orthagonal_basis(X) # Get a basis
185
+ psi = orthagonal_basis(clr(X)) # trying to get right algorithm
186
+ assert np.allclose(psi @ psi.T, np.eye(d - 1))
187
+ return Y @ psi.T
188
+
189
+
190
+ def inv_ilr(Y: np.ndarray, X: np.ndarray = None):
191
+ """
192
+ Inverse Isometric Log Ratio transformation.
193
+
194
+ Parameters
195
+ ---------------
196
+ Y: np.ndarray
197
+ Array on which to perform the inverse transformation.
198
+ """
199
+ psi = orthagonal_basis(X)
200
+ C = Y @ psi
201
+ X = inv_clr(C) # Inverse log operation
202
+ return X
203
+
204
+
205
+ def boxcox(
206
+ X: np.ndarray,
207
+ lmbda=None,
208
+ lmbda_search_space=(-1, 5),
209
+ search_steps=100,
210
+ return_lmbda=False,
211
+ ):
212
+ """
213
+ Box-Cox transformation.
214
+
215
+ Parameters
216
+ ---------------
217
+ Y: np.ndarray
218
+ Array on which to perform the transformation.
219
+ lmbda: {None, np.float}
220
+ Lambda value used to forward-transform values. If none, it will be calculated
221
+ using the mean
222
+ """
223
+ if isinstance(X, pd.DataFrame) or isinstance(X, pd.Series):
224
+ _X = X.values
225
+ else:
226
+ _X = X.copy()
227
+
228
+ if lmbda is None:
229
+ l_search = np.linspace(*lmbda_search_space, search_steps)
230
+ llf = np.apply_along_axis(scpstats.boxcox_llf, 0, np.array([l_search]), _X.T)
231
+ if llf.shape[0] == 1:
232
+ mean_llf = llf[0]
233
+ else:
234
+ mean_llf = np.nansum(llf, axis=0)
235
+
236
+ lmbda = l_search[mean_llf == np.nanmax(mean_llf)]
237
+ if _X.ndim < 2:
238
+ out = scpstats.boxcox(_X, lmbda)
239
+ elif _X.shape[0] == 1:
240
+ out = scpstats.boxcox(np.squeeze(_X), lmbda)
241
+ else:
242
+ out = np.apply_along_axis(scpstats.boxcox, 0, _X, lmbda)
243
+
244
+ if isinstance(_X, pd.DataFrame) or isinstance(_X, pd.Series):
245
+ _out = X.copy()
246
+ _out.loc[:, :] = out
247
+ out = _out
248
+
249
+ if return_lmbda:
250
+ return out, lmbda
251
+ else:
252
+ return out
253
+
254
+
255
+ def inv_boxcox(Y: np.ndarray, lmbda):
256
+ """
257
+ Inverse Box-Cox transformation.
258
+
259
+ Parameters
260
+ ---------------
261
+ Y: np.ndarray
262
+ Array on which to perform the transformation.
263
+ lmbda: np.float
264
+ Lambda value used to forward-transform values.
265
+ """
266
+ return scpspec.inv_boxcox(Y, lmbda)