pyrolite 0.0.14__zip
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__init__.py +10 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/_version.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/alteration.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/classification.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/compositions.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/geochem.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/melts.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/norm.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/normalisation.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/plot.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/_version.py +21 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/alteration.py +66 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/classification.py +222 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__init__.py +9 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/aggregate.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/codata.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/impute.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/renorm.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/aggregate.py +391 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/codata.py +266 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/impute.py +82 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/renorm.py +40 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/compositions.py +524 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_CFB_Dataset_List.csv +42 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_Convergent_Dataset_List.csv +42 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OBFB_Dataset_List.csv +5 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OIB_Dataset_List.csv +49 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OceanicPlateau_Dataset_List.csv +18 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/contents.json +1 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-35.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/env.py +1063 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ba.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Bs.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.F.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O3.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Pc.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ph.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.R.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S3.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U3.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.modelfields +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.nan.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.none.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/aphanitic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/gabbroic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/peralkalinity.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/phaneritic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/ultramafic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/CH_PalmeONeill2014.csv +95 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DDMM_WorkmanHart2005.csv +105 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DMM_WorkmanHart2005.csv +105 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DM_SaltersStrake2004.csv +95 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/EDMM_WorkmanHart2005.csv +105 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/PM_PalmeONeill2014.csv +95 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/timescale/geotimescale_spans.csv +180 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/geochem.py +821 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/melts.py +92 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__init__.py +10 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/db.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/ions.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/mineral.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/sites.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/db.py +88 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/ions.py +78 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/mineral.py +587 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/sites.py +134 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/norm.py +224 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/normalisation.py +204 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/plot.py +514 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__init__.py +13 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/database.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/env.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/general.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/georoc.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/math.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/melts.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multip.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multiprocessing.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/pd.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/plot.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/skl.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/spatial.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/text.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/time.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/wfs.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/database.py +88 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/env.py +81 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/general.py +266 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/georoc.py +444 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/math.py +371 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/melts.py +397 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multip.py +29 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multiprocessing.py +29 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/pd.py +214 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/plot.py +345 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/skl.py +847 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/spatial.py +91 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/text.py +207 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/time.py +224 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/wfs.py +10 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/PKG-INFO +61 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/SOURCES.txt +83 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/dependency_links.txt +1 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/requires.txt +47 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import pandas as pd
|
|
3
|
+
import warnings
|
|
4
|
+
from .codata import alr, inv_alr
|
|
5
|
+
from .impute import *
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
|
|
9
|
+
logging.getLogger(__name__).addHandler(logging.NullHandler())
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def get_full_column(X: np.ndarray):
|
|
14
|
+
"""
|
|
15
|
+
Returns the index of the first array column which contains only finite
|
|
16
|
+
numbers (i.e. no missing data, nan, inf).
|
|
17
|
+
|
|
18
|
+
Parameters
|
|
19
|
+
---------------
|
|
20
|
+
X: np.ndarray
|
|
21
|
+
Array for which to find the first full column within.
|
|
22
|
+
"""
|
|
23
|
+
if len(X.shape) == 1:
|
|
24
|
+
X = X.reshape((1, *X.shape))
|
|
25
|
+
inds = np.arange(X.shape[1])
|
|
26
|
+
wherenonnan = np.isfinite(X).all(axis=0)
|
|
27
|
+
ind = inds[wherenonnan][0]
|
|
28
|
+
return ind
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def weights_from_array(X: np.ndarray):
|
|
32
|
+
"""
|
|
33
|
+
Returns a set of equal weights with size equal to that of the first axis of
|
|
34
|
+
an array.
|
|
35
|
+
|
|
36
|
+
Parameters
|
|
37
|
+
---------------
|
|
38
|
+
X: np.ndarray
|
|
39
|
+
Array of compositions to produce weights for.
|
|
40
|
+
"""
|
|
41
|
+
wts = np.ones((X.shape[0]))
|
|
42
|
+
return wts / np.sum(wts)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def nan_weighted_mean(X: np.ndarray, weights=None):
|
|
46
|
+
"""
|
|
47
|
+
Returns a weighted mean of compositions, where weights are renormalised
|
|
48
|
+
to account for missing data.
|
|
49
|
+
|
|
50
|
+
Parameters
|
|
51
|
+
---------------
|
|
52
|
+
X: np.ndarray
|
|
53
|
+
Array of compositions to take a weighted mean of.
|
|
54
|
+
weights: np.ndarray
|
|
55
|
+
Array of weights.
|
|
56
|
+
"""
|
|
57
|
+
if weights is None:
|
|
58
|
+
weights = weights_from_array(X)
|
|
59
|
+
weights = np.array(weights) / np.nansum(weights)
|
|
60
|
+
|
|
61
|
+
mask = (np.isnan(X) + np.isinf(X)) > 0
|
|
62
|
+
if not mask.any():
|
|
63
|
+
return np.average(X, weights=weights, axis=0)
|
|
64
|
+
else:
|
|
65
|
+
return np.ma.average(np.ma.array(X, mask=mask), weights=weights, axis=0)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def compositional_mean(df, weights=[], **kwargs):
|
|
69
|
+
"""
|
|
70
|
+
Implements an aggregation using a compositional weighted mean.
|
|
71
|
+
|
|
72
|
+
Parameters
|
|
73
|
+
---------------
|
|
74
|
+
df: pd.DataFrame
|
|
75
|
+
Dataframe of compositions to aggregate.
|
|
76
|
+
weights: np.ndarray
|
|
77
|
+
Array of weights.
|
|
78
|
+
"""
|
|
79
|
+
non_nan_cols = df.dropna(axis=1, how="all").columns
|
|
80
|
+
assert not df.loc[:, non_nan_cols].isna().values.any()
|
|
81
|
+
mean = df.iloc[0, :].copy()
|
|
82
|
+
if not weights:
|
|
83
|
+
weights = np.ones(len(df.index.values))
|
|
84
|
+
weights = np.array(weights) / np.nansum(weights)
|
|
85
|
+
|
|
86
|
+
logmean = alr(df.loc[:, non_nan_cols].values).T @ weights[:, np.newaxis]
|
|
87
|
+
# this renormalises by default
|
|
88
|
+
mean.loc[non_nan_cols] = inv_alr(logmean.T.squeeze())
|
|
89
|
+
return mean
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def nan_weighted_compositional_mean(
|
|
93
|
+
X: np.ndarray, weights=None, ind=None, renorm=True, **kwargs
|
|
94
|
+
):
|
|
95
|
+
"""
|
|
96
|
+
Implements an aggregation using a weighted mean, but accounts
|
|
97
|
+
for nans. Requires at least one non-nan column for alr mean.
|
|
98
|
+
|
|
99
|
+
When used for internal standardisation, there should be only a single
|
|
100
|
+
common element - this would be used by default as the divisor here. When
|
|
101
|
+
used for multiple-standardisation, the [specified] or first common
|
|
102
|
+
element will be used.
|
|
103
|
+
|
|
104
|
+
Parameters
|
|
105
|
+
---------------
|
|
106
|
+
X: np.ndarray
|
|
107
|
+
Array of compositions to aggregate.
|
|
108
|
+
weights: np.ndarray
|
|
109
|
+
Array of weights.
|
|
110
|
+
ind: int
|
|
111
|
+
Index of the column to use as the alr divisor.
|
|
112
|
+
renorm: bool, True
|
|
113
|
+
Whether to renormalise the output compositional mean to unity.
|
|
114
|
+
|
|
115
|
+
Returns
|
|
116
|
+
-------
|
|
117
|
+
np.ndarray
|
|
118
|
+
An array with the mean composition.
|
|
119
|
+
"""
|
|
120
|
+
if X.ndim == 1: # if it's a single row
|
|
121
|
+
return X
|
|
122
|
+
else:
|
|
123
|
+
if weights is None:
|
|
124
|
+
weights = weights_from_array(X)
|
|
125
|
+
else:
|
|
126
|
+
weights = np.array(weights) / np.sum(weights, axis=-1)
|
|
127
|
+
|
|
128
|
+
if ind is None: # take the first column which has no nans
|
|
129
|
+
ind = get_full_column(X)
|
|
130
|
+
|
|
131
|
+
if X.ndim < 3 and X.shape[0] == 1:
|
|
132
|
+
div = X[:, ind].squeeze() # check this
|
|
133
|
+
else:
|
|
134
|
+
div = X[:, ind].squeeze()[:, np.newaxis]
|
|
135
|
+
|
|
136
|
+
logvals = np.log(np.divide(X, div))
|
|
137
|
+
mean = np.nan * np.ones(X.shape[1:])
|
|
138
|
+
|
|
139
|
+
ixs = np.arange(logvals.shape[1])
|
|
140
|
+
if X.ndim == 2:
|
|
141
|
+
indexes = ixs
|
|
142
|
+
elif X.ndim == 3:
|
|
143
|
+
iys = np.arange(logvals.shape[2])
|
|
144
|
+
indexes = np.ixs_(ixs, iys)
|
|
145
|
+
|
|
146
|
+
mean[indexes] = nan_weighted_mean(logvals[:, indexes], weights=weights)
|
|
147
|
+
|
|
148
|
+
mean = np.exp(mean.squeeze())
|
|
149
|
+
if renorm:
|
|
150
|
+
mean /= np.nansum(mean)
|
|
151
|
+
return mean
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def cross_ratios(df: pd.DataFrame):
|
|
155
|
+
"""
|
|
156
|
+
Takes ratios of values across a dataframe, such that columns are
|
|
157
|
+
denominators and the row indexes the numerators, to create a square array.
|
|
158
|
+
Returns one array per record.
|
|
159
|
+
|
|
160
|
+
Parameters
|
|
161
|
+
---------------
|
|
162
|
+
df: pd.DataFrame
|
|
163
|
+
Dataframe of compositions to create ratios of.
|
|
164
|
+
|
|
165
|
+
Returns
|
|
166
|
+
-------
|
|
167
|
+
np.ndarray
|
|
168
|
+
A 3D array of ratios.
|
|
169
|
+
"""
|
|
170
|
+
ratios = np.ones((len(df.index), len(df.columns), len(df.columns)))
|
|
171
|
+
for idx in range(df.index.size):
|
|
172
|
+
row_vals = df.iloc[idx, :].values
|
|
173
|
+
r1 = row_vals.T[:, np.newaxis] @ np.ones_like(row_vals)[np.newaxis, :]
|
|
174
|
+
ratios[idx] = r1 / r1.T
|
|
175
|
+
return ratios
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def np_cross_ratios(X: np.ndarray, debug=False):
|
|
179
|
+
"""
|
|
180
|
+
Takes ratios of values across an array, such that columns are
|
|
181
|
+
denominators and the row indexes the numerators, to create a square array.
|
|
182
|
+
Returns one array per record.
|
|
183
|
+
|
|
184
|
+
Parameters
|
|
185
|
+
---------------
|
|
186
|
+
X: np.ndarray
|
|
187
|
+
Array of compositions to create ratios of.
|
|
188
|
+
|
|
189
|
+
Returns
|
|
190
|
+
-------
|
|
191
|
+
np.ndarray
|
|
192
|
+
A 3D array of ratios.
|
|
193
|
+
"""
|
|
194
|
+
X = X.copy()
|
|
195
|
+
with warnings.catch_warnings():
|
|
196
|
+
# can get invalid values which raise RuntimeWarnings
|
|
197
|
+
# consider changing to np.errstate
|
|
198
|
+
warnings.simplefilter("ignore", category=RuntimeWarning)
|
|
199
|
+
X[X <= 0] = np.nan
|
|
200
|
+
if X.ndim == 1:
|
|
201
|
+
index_length = 1
|
|
202
|
+
X = X.reshape((1, *X.shape))
|
|
203
|
+
else:
|
|
204
|
+
index_length = X.shape[0]
|
|
205
|
+
dims = X.shape[-1]
|
|
206
|
+
ratios = np.ones((index_length, dims, dims))
|
|
207
|
+
for idx in range(index_length):
|
|
208
|
+
row_vals = X[idx, :]
|
|
209
|
+
r1 = row_vals.T[:, np.newaxis] @ np.ones_like(row_vals)[np.newaxis, :]
|
|
210
|
+
ratios[idx] = r1.T / r1
|
|
211
|
+
|
|
212
|
+
if debug:
|
|
213
|
+
try:
|
|
214
|
+
diags = ratios[:, np.arange(dims), np.arange(dims)]
|
|
215
|
+
# check all diags are 1.
|
|
216
|
+
assert np.allclose(diags, 1.0)
|
|
217
|
+
except:
|
|
218
|
+
# check all diags are 1. or nan
|
|
219
|
+
assert np.allclose(diags[~np.isnan(diags)], 1.0)
|
|
220
|
+
|
|
221
|
+
return ratios
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def standardise_aggregate(
|
|
225
|
+
df: pd.DataFrame, int_std=None, fixed_record_idx=0, renorm=True, **kwargs
|
|
226
|
+
):
|
|
227
|
+
"""
|
|
228
|
+
Performs internal standardisation and aggregates dissimilar geochemical
|
|
229
|
+
records. Note: this changes the closure parameter, and is generally
|
|
230
|
+
intended to integrate major and trace element records.
|
|
231
|
+
|
|
232
|
+
Parameters
|
|
233
|
+
---------------
|
|
234
|
+
df: pd.DataFrame
|
|
235
|
+
Dataframe of compositions to aggregate of.
|
|
236
|
+
int_std: str
|
|
237
|
+
Name of the internal standard column.
|
|
238
|
+
fixed_record_idx: int
|
|
239
|
+
Numeric index of a specific record's for which to retain the internal
|
|
240
|
+
standard value (e.g for standardising trace element data).
|
|
241
|
+
renorm: bool, True
|
|
242
|
+
Whether to renormalise to unity.
|
|
243
|
+
|
|
244
|
+
Returns
|
|
245
|
+
-------
|
|
246
|
+
pd.Series
|
|
247
|
+
A series representing the internally standardised record.
|
|
248
|
+
"""
|
|
249
|
+
if df.index.size == 1: # catch single records
|
|
250
|
+
return df
|
|
251
|
+
else:
|
|
252
|
+
if int_std is None:
|
|
253
|
+
# Get the 'internal standard column'
|
|
254
|
+
potential_int_stds = df.count()[df.count() == df.count().max()].index.values
|
|
255
|
+
assert len(potential_int_stds) > 0
|
|
256
|
+
# Use an internal standard
|
|
257
|
+
int_std = potential_int_stds[0]
|
|
258
|
+
if len(potential_int_stds) > 1:
|
|
259
|
+
logging.info("Multiple int. stds possible. Using " + str(int_std))
|
|
260
|
+
|
|
261
|
+
non_nan_cols = df.dropna(axis=1, how="all").columns
|
|
262
|
+
assert len(non_nan_cols)
|
|
263
|
+
mean = nan_weighted_compositional_mean(
|
|
264
|
+
df.values, ind=df.columns.get_loc(int_std), renorm=False
|
|
265
|
+
)
|
|
266
|
+
ser = pd.Series(mean, index=df.columns)
|
|
267
|
+
multiplier = (
|
|
268
|
+
df.iloc[fixed_record_idx, df.columns.get_loc(int_std)] / ser[int_std]
|
|
269
|
+
)
|
|
270
|
+
ser *= multiplier
|
|
271
|
+
if renorm:
|
|
272
|
+
ser /= np.nansum(ser.values)
|
|
273
|
+
return ser
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def complex_standardise_aggregate(
|
|
277
|
+
df, int_std=None, fixed_record_idx=0, renorm=True # fallback parameters
|
|
278
|
+
):
|
|
279
|
+
"""
|
|
280
|
+
Aggregation function which attempts to incorporate all ratio information
|
|
281
|
+
but creating a ratio matrix for each record which is then imputed for
|
|
282
|
+
missing data where possible. Falls back to internal standardistion where
|
|
283
|
+
int_std is specified.
|
|
284
|
+
|
|
285
|
+
Parameters
|
|
286
|
+
---------------
|
|
287
|
+
df: pd.DataFrame
|
|
288
|
+
Dataframe of compositions to aggregate of.
|
|
289
|
+
int_std: str
|
|
290
|
+
Name of the internal standard column.
|
|
291
|
+
fixed_record_idx: int
|
|
292
|
+
Numeric index of a specific record's for which to retain the internal
|
|
293
|
+
standard value (e.g for standardising trace element data).
|
|
294
|
+
renorm: bool, True
|
|
295
|
+
Whether to renormalise to unity.
|
|
296
|
+
|
|
297
|
+
Returns
|
|
298
|
+
-------
|
|
299
|
+
pd.Series
|
|
300
|
+
A series representing the internally standardised record.
|
|
301
|
+
"""
|
|
302
|
+
if int_std is None:
|
|
303
|
+
# create a n x d x d matrix for aggregating ratios
|
|
304
|
+
non_nan_cols = df.dropna(axis=1, how="all").columns
|
|
305
|
+
ratios = cross_ratios(df.loc[:, non_nan_cols])
|
|
306
|
+
# Average across record matricies
|
|
307
|
+
with warnings.catch_warnings():
|
|
308
|
+
# can get empty arrays which raise RuntimeWarnings
|
|
309
|
+
# consider changing to np.errstate
|
|
310
|
+
warnings.simplefilter("ignore", category=RuntimeWarning)
|
|
311
|
+
mean_ratios = pd.DataFrame(
|
|
312
|
+
np.exp(np.nanmean(np.log(ratios), axis=0)),
|
|
313
|
+
columns=non_nan_cols,
|
|
314
|
+
index=non_nan_cols,
|
|
315
|
+
)
|
|
316
|
+
# Filling in the null values in a ratio matrix
|
|
317
|
+
imputed_ratios = impute_ratios(mean_ratios)
|
|
318
|
+
# We simply pick the first non-nan column.
|
|
319
|
+
IS = non_nan_cols[0]
|
|
320
|
+
mean = np.exp(
|
|
321
|
+
np.mean(np.log(imputed_ratios / imputed_ratios.loc[IS, :]), axis=1)
|
|
322
|
+
)
|
|
323
|
+
# This needs to be renormalised to make logical sense
|
|
324
|
+
mean /= np.nansum(mean.values)
|
|
325
|
+
|
|
326
|
+
out = np.ones((1, len(df.columns))) * np.nan
|
|
327
|
+
out[:, [list(df.columns).index(c) for c in non_nan_cols]] = mean
|
|
328
|
+
return pd.Series(out.squeeze(), index=df.columns)
|
|
329
|
+
else:
|
|
330
|
+
# fallback to internal standardisation
|
|
331
|
+
return standardise_aggregate(
|
|
332
|
+
df, int_std=int_std, fixed_record_idx=fixed_record_idx, renorm=renorm
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def np_complex_standardise_aggregate(
|
|
337
|
+
df, int_std=None, renorm=True, fixed_record_idx=0 # fallback parameters
|
|
338
|
+
):
|
|
339
|
+
"""
|
|
340
|
+
Aggregation function which attempts to incorporate all ratio information
|
|
341
|
+
but creating a ratio matrix for each record which is then imputed for
|
|
342
|
+
missing data where possible. Falls back to internal standardistion where
|
|
343
|
+
int_std is specified. Uses numpy more extensively.
|
|
344
|
+
|
|
345
|
+
Parameters
|
|
346
|
+
---------------
|
|
347
|
+
df: pd.DataFrame
|
|
348
|
+
Dataframe of compositions to aggregate of.
|
|
349
|
+
int_std: str
|
|
350
|
+
Name of the internal standard column.
|
|
351
|
+
fixed_record_idx: int
|
|
352
|
+
Numeric index of a specific record's for which to retain the internal
|
|
353
|
+
standard value (e.g for standardising trace element data).
|
|
354
|
+
renorm: bool, True
|
|
355
|
+
Whether to renormalise to unity.
|
|
356
|
+
|
|
357
|
+
Returns
|
|
358
|
+
-------
|
|
359
|
+
pd.Series
|
|
360
|
+
A series representing the internally standardised record.
|
|
361
|
+
"""
|
|
362
|
+
|
|
363
|
+
if int_std is None:
|
|
364
|
+
# create a n x d x d matrix for aggregating ratios
|
|
365
|
+
non_nan_cols = df.dropna(axis=1, how="all").columns
|
|
366
|
+
assert len(non_nan_cols) > 0
|
|
367
|
+
ratios = np_cross_ratios(df.loc[:, non_nan_cols].values)
|
|
368
|
+
# Take the mean across the cross-ratio matricies
|
|
369
|
+
with warnings.catch_warnings():
|
|
370
|
+
# can get empty arrays which raise RuntimeWarnings
|
|
371
|
+
# consider changing to np.errstate
|
|
372
|
+
warnings.simplefilter("ignore", category=RuntimeWarning)
|
|
373
|
+
mean_logratios = np.nanmean(np.log(ratios), axis=0)
|
|
374
|
+
# Filling in the null values in a ratio matrix
|
|
375
|
+
imputed_log_ratios = np_impute_ratios(mean_logratios)
|
|
376
|
+
# We simply pick the first non-nan column.
|
|
377
|
+
# IS = 0
|
|
378
|
+
IS = np.argmax(np.count_nonzero(~np.isnan(imputed_log_ratios), axis=0))
|
|
379
|
+
# Convert to a composition by subtracting a row and taking negative
|
|
380
|
+
div_log_ratios = -(imputed_log_ratios - imputed_log_ratios[IS, :])
|
|
381
|
+
comp_abund = np.exp(np.nanmean(div_log_ratios, axis=1))
|
|
382
|
+
comp_abund /= np.nansum(comp_abund)
|
|
383
|
+
out = np.ones((1, len(df.columns))) * np.nan
|
|
384
|
+
inds = np.array([list(df.columns).index(c) for c in non_nan_cols])
|
|
385
|
+
out[:, inds] = comp_abund
|
|
386
|
+
return pd.Series(out.squeeze(), index=df.columns)
|
|
387
|
+
else:
|
|
388
|
+
# fallback to internal standardisation
|
|
389
|
+
return standardise_aggregate(
|
|
390
|
+
df, int_std=int_std, fixed_record_idx=fixed_record_idx, renorm=renorm
|
|
391
|
+
)
|
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import pandas as pd
|
|
3
|
+
import pandas_flavor as pf
|
|
4
|
+
import scipy.stats as scpstats
|
|
5
|
+
import scipy.special as scpspec
|
|
6
|
+
#from .renorm import renormalise, close
|
|
7
|
+
from ..util.math import orthagonal_basis
|
|
8
|
+
import logging
|
|
9
|
+
|
|
10
|
+
logging.getLogger(__name__).addHandler(logging.NullHandler())
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def close(X: np.ndarray):
|
|
15
|
+
if X.ndim == 2:
|
|
16
|
+
return np.divide(X, np.sum(X, axis=1)[:, np.newaxis])
|
|
17
|
+
else:
|
|
18
|
+
return np.divide(X, np.sum(X, axis=0))
|
|
19
|
+
|
|
20
|
+
@pf.register_series_method
|
|
21
|
+
@pf.register_dataframe_method
|
|
22
|
+
def renormalise(df: pd.DataFrame, components: list = [], scale=100.0):
|
|
23
|
+
"""
|
|
24
|
+
Renormalises compositional data to ensure closure.
|
|
25
|
+
|
|
26
|
+
Parameters
|
|
27
|
+
------------
|
|
28
|
+
df: pd.DataFrame
|
|
29
|
+
Dataframe to renomalise.
|
|
30
|
+
components: list
|
|
31
|
+
Option subcompositon to renormalise to 100. Useful for the use case
|
|
32
|
+
where compostional data and non-compositional data are stored in the
|
|
33
|
+
same dataframe.
|
|
34
|
+
scale: float, 100.
|
|
35
|
+
Closure parameter. Typically either 100 or 1.
|
|
36
|
+
"""
|
|
37
|
+
dfc = df.copy(deep=True)
|
|
38
|
+
if components:
|
|
39
|
+
cmpnts = [c for c in components if c in dfc.columns]
|
|
40
|
+
dfc.loc[:, cmpnts] = scale * dfc.loc[:, cmpnts].divide(
|
|
41
|
+
dfc.loc[:, cmpnts].sum(axis=1).replace(0, np.nan), axis=0
|
|
42
|
+
)
|
|
43
|
+
return dfc
|
|
44
|
+
else:
|
|
45
|
+
dfc = dfc.divide(dfc.sum(axis=1).replace(0, 100.0), axis=0) * scale
|
|
46
|
+
return dfc
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def additive_log_ratio(X: np.ndarray, ind: int = -1, null_col=False):
|
|
50
|
+
"""
|
|
51
|
+
Inverse Additive Log Ratio transformation.
|
|
52
|
+
|
|
53
|
+
Parameters
|
|
54
|
+
---------------
|
|
55
|
+
X: np.ndarray
|
|
56
|
+
Array on which to perform the inverse transformation.
|
|
57
|
+
ind: int
|
|
58
|
+
Index of column used as denominator.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
Y = X.copy()
|
|
62
|
+
assert Y.ndim in [1, 2]
|
|
63
|
+
dimensions = Y.shape[Y.ndim - 1]
|
|
64
|
+
if ind < 0:
|
|
65
|
+
ind += dimensions
|
|
66
|
+
|
|
67
|
+
if Y.ndim == 2:
|
|
68
|
+
Y = np.divide(Y, Y[:, ind][:, np.newaxis])
|
|
69
|
+
if not null_col:
|
|
70
|
+
Y = Y[:, [i for i in range(dimensions) if not i == ind]]
|
|
71
|
+
else:
|
|
72
|
+
Y = np.divide(X, X[ind])
|
|
73
|
+
if not null_col:
|
|
74
|
+
Y = Y[[i for i in range(dimensions) if not i == ind]]
|
|
75
|
+
|
|
76
|
+
return np.log(Y)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def inverse_additive_log_ratio(Y: np.ndarray, ind=-1, null_col=False):
|
|
80
|
+
"""
|
|
81
|
+
Inverse Centred Log Ratio transformation.
|
|
82
|
+
|
|
83
|
+
Parameters
|
|
84
|
+
---------------
|
|
85
|
+
X: np.ndarray
|
|
86
|
+
Array on which to perform the inverse transformation.
|
|
87
|
+
ind: int
|
|
88
|
+
Index of column used as denominator.
|
|
89
|
+
"""
|
|
90
|
+
assert Y.ndim in [1, 2]
|
|
91
|
+
|
|
92
|
+
X = Y.copy()
|
|
93
|
+
dimensions = X.shape[X.ndim - 1]
|
|
94
|
+
if not null_col:
|
|
95
|
+
idx = np.arange(0, dimensions + 1)
|
|
96
|
+
|
|
97
|
+
if ind != -1:
|
|
98
|
+
idx = np.array(list(idx[idx < ind]) + [-1] + list(idx[idx >= ind + 1] - 1))
|
|
99
|
+
|
|
100
|
+
# Add a zero-column and reorder columns
|
|
101
|
+
if Y.ndim == 2:
|
|
102
|
+
X = np.concatenate((X, np.zeros((X.shape[0], 1))), axis=1)
|
|
103
|
+
X = X[:, idx]
|
|
104
|
+
else:
|
|
105
|
+
X = np.append(X, np.array([0]))
|
|
106
|
+
X = X[idx]
|
|
107
|
+
|
|
108
|
+
# Inverse log and closure operations
|
|
109
|
+
X = np.exp(X)
|
|
110
|
+
X = close(X)
|
|
111
|
+
return X
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def alr(*args, **kwargs):
|
|
115
|
+
"""
|
|
116
|
+
Short form of Additive Log Ratio transformation.
|
|
117
|
+
|
|
118
|
+
Parameters
|
|
119
|
+
---------------
|
|
120
|
+
Y: np.ndarray
|
|
121
|
+
Array on which to perform the inverse transformation.
|
|
122
|
+
ind: int
|
|
123
|
+
Index of column used as denominator.
|
|
124
|
+
"""
|
|
125
|
+
return additive_log_ratio(*args, **kwargs)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def inv_alr(*args, **kwargs):
|
|
129
|
+
"""
|
|
130
|
+
Short form of Inverse Additive Log Ratio transformation.
|
|
131
|
+
|
|
132
|
+
Parameters
|
|
133
|
+
---------------
|
|
134
|
+
Y: np.ndarray
|
|
135
|
+
Array on which to perform the inverse transformation.
|
|
136
|
+
ind: int
|
|
137
|
+
Index of column used as denominator.
|
|
138
|
+
"""
|
|
139
|
+
return inverse_additive_log_ratio(*args, **kwargs)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def clr(X: np.ndarray):
|
|
143
|
+
"""
|
|
144
|
+
Centred Log Ratio transformation.
|
|
145
|
+
|
|
146
|
+
Parameters
|
|
147
|
+
---------------
|
|
148
|
+
X: np.ndarray
|
|
149
|
+
Array on which to perform the transformation.
|
|
150
|
+
"""
|
|
151
|
+
X = np.divide(X, np.sum(X, axis=1)[:, np.newaxis]) # Closure operation
|
|
152
|
+
Y = np.log(X) # Log operation
|
|
153
|
+
Y -= 1 / X.shape[1] * np.nansum(Y, axis=1)[:, np.newaxis]
|
|
154
|
+
return Y
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def inv_clr(Y: np.ndarray):
|
|
158
|
+
"""
|
|
159
|
+
Inverse Centred Log Ratio transformation.
|
|
160
|
+
|
|
161
|
+
Parameters
|
|
162
|
+
---------------
|
|
163
|
+
Y: np.ndarray
|
|
164
|
+
Array on which to perform the inverse transformation.
|
|
165
|
+
"""
|
|
166
|
+
# Inverse of log operation
|
|
167
|
+
X = np.exp(Y)
|
|
168
|
+
# Closure operation
|
|
169
|
+
X = np.divide(X, np.nansum(X, axis=1)[:, np.newaxis])
|
|
170
|
+
return X
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def ilr(X: np.ndarray):
|
|
174
|
+
"""
|
|
175
|
+
Isotmetric Log Ratio transformation.
|
|
176
|
+
|
|
177
|
+
Parameters
|
|
178
|
+
---------------
|
|
179
|
+
X: np.ndarray
|
|
180
|
+
Array on which to perform the transformation.
|
|
181
|
+
"""
|
|
182
|
+
d = X.shape[1]
|
|
183
|
+
Y = clr(X)
|
|
184
|
+
psi = orthagonal_basis(X) # Get a basis
|
|
185
|
+
psi = orthagonal_basis(clr(X)) # trying to get right algorithm
|
|
186
|
+
assert np.allclose(psi @ psi.T, np.eye(d - 1))
|
|
187
|
+
return Y @ psi.T
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def inv_ilr(Y: np.ndarray, X: np.ndarray = None):
|
|
191
|
+
"""
|
|
192
|
+
Inverse Isometric Log Ratio transformation.
|
|
193
|
+
|
|
194
|
+
Parameters
|
|
195
|
+
---------------
|
|
196
|
+
Y: np.ndarray
|
|
197
|
+
Array on which to perform the inverse transformation.
|
|
198
|
+
"""
|
|
199
|
+
psi = orthagonal_basis(X)
|
|
200
|
+
C = Y @ psi
|
|
201
|
+
X = inv_clr(C) # Inverse log operation
|
|
202
|
+
return X
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def boxcox(
|
|
206
|
+
X: np.ndarray,
|
|
207
|
+
lmbda=None,
|
|
208
|
+
lmbda_search_space=(-1, 5),
|
|
209
|
+
search_steps=100,
|
|
210
|
+
return_lmbda=False,
|
|
211
|
+
):
|
|
212
|
+
"""
|
|
213
|
+
Box-Cox transformation.
|
|
214
|
+
|
|
215
|
+
Parameters
|
|
216
|
+
---------------
|
|
217
|
+
Y: np.ndarray
|
|
218
|
+
Array on which to perform the transformation.
|
|
219
|
+
lmbda: {None, np.float}
|
|
220
|
+
Lambda value used to forward-transform values. If none, it will be calculated
|
|
221
|
+
using the mean
|
|
222
|
+
"""
|
|
223
|
+
if isinstance(X, pd.DataFrame) or isinstance(X, pd.Series):
|
|
224
|
+
_X = X.values
|
|
225
|
+
else:
|
|
226
|
+
_X = X.copy()
|
|
227
|
+
|
|
228
|
+
if lmbda is None:
|
|
229
|
+
l_search = np.linspace(*lmbda_search_space, search_steps)
|
|
230
|
+
llf = np.apply_along_axis(scpstats.boxcox_llf, 0, np.array([l_search]), _X.T)
|
|
231
|
+
if llf.shape[0] == 1:
|
|
232
|
+
mean_llf = llf[0]
|
|
233
|
+
else:
|
|
234
|
+
mean_llf = np.nansum(llf, axis=0)
|
|
235
|
+
|
|
236
|
+
lmbda = l_search[mean_llf == np.nanmax(mean_llf)]
|
|
237
|
+
if _X.ndim < 2:
|
|
238
|
+
out = scpstats.boxcox(_X, lmbda)
|
|
239
|
+
elif _X.shape[0] == 1:
|
|
240
|
+
out = scpstats.boxcox(np.squeeze(_X), lmbda)
|
|
241
|
+
else:
|
|
242
|
+
out = np.apply_along_axis(scpstats.boxcox, 0, _X, lmbda)
|
|
243
|
+
|
|
244
|
+
if isinstance(_X, pd.DataFrame) or isinstance(_X, pd.Series):
|
|
245
|
+
_out = X.copy()
|
|
246
|
+
_out.loc[:, :] = out
|
|
247
|
+
out = _out
|
|
248
|
+
|
|
249
|
+
if return_lmbda:
|
|
250
|
+
return out, lmbda
|
|
251
|
+
else:
|
|
252
|
+
return out
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def inv_boxcox(Y: np.ndarray, lmbda):
|
|
256
|
+
"""
|
|
257
|
+
Inverse Box-Cox transformation.
|
|
258
|
+
|
|
259
|
+
Parameters
|
|
260
|
+
---------------
|
|
261
|
+
Y: np.ndarray
|
|
262
|
+
Array on which to perform the transformation.
|
|
263
|
+
lmbda: np.float
|
|
264
|
+
Lambda value used to forward-transform values.
|
|
265
|
+
"""
|
|
266
|
+
return scpspec.inv_boxcox(Y, lmbda)
|