pyrolite 0.0.14__zip

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__init__.py +10 -0
  2. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/__init__.cpython-36.pyc +0 -0
  3. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/_version.cpython-36.pyc +0 -0
  4. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/alteration.cpython-36.pyc +0 -0
  5. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/classification.cpython-36.pyc +0 -0
  6. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/compositions.cpython-36.pyc +0 -0
  7. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/geochem.cpython-36.pyc +0 -0
  8. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/melts.cpython-36.pyc +0 -0
  9. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/norm.cpython-36.pyc +0 -0
  10. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/normalisation.cpython-36.pyc +0 -0
  11. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/plot.cpython-36.pyc +0 -0
  12. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/_version.py +21 -0
  13. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/alteration.py +66 -0
  14. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/classification.py +222 -0
  15. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__init__.py +9 -0
  16. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/__init__.cpython-36.pyc +0 -0
  17. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/aggregate.cpython-36.pyc +0 -0
  18. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/codata.cpython-36.pyc +0 -0
  19. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/impute.cpython-36.pyc +0 -0
  20. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/renorm.cpython-36.pyc +0 -0
  21. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/aggregate.py +391 -0
  22. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/codata.py +266 -0
  23. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/impute.py +82 -0
  24. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/renorm.py +40 -0
  25. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/compositions.py +524 -0
  26. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_CFB_Dataset_List.csv +42 -0
  27. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_Convergent_Dataset_List.csv +42 -0
  28. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OBFB_Dataset_List.csv +5 -0
  29. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OIB_Dataset_List.csv +49 -0
  30. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OceanicPlateau_Dataset_List.csv +18 -0
  31. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/contents.json +1 -0
  32. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-35.pyc +0 -0
  33. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-36.pyc +0 -0
  34. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/env.py +1063 -0
  35. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ba.modelfield +0 -0
  36. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Bs.modelfield +0 -0
  37. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.F.modelfield +0 -0
  38. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O1.modelfield +0 -0
  39. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O2.modelfield +0 -0
  40. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O3.modelfield +0 -0
  41. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Pc.modelfield +0 -0
  42. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ph.modelfield +0 -0
  43. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.R.modelfield +0 -0
  44. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S1.modelfield +0 -0
  45. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S2.modelfield +0 -0
  46. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S3.modelfield +0 -0
  47. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T1.modelfield +0 -0
  48. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T2.modelfield +0 -0
  49. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U1.modelfield +0 -0
  50. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U2.modelfield +0 -0
  51. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U3.modelfield +0 -0
  52. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.modelfields +0 -0
  53. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.nan.modelfield +0 -0
  54. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.none.modelfield +0 -0
  55. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS.clsf.gz +0 -0
  56. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/aphanitic.clsf.gz +0 -0
  57. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/gabbroic.clsf.gz +0 -0
  58. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/peralkalinity.clsf.gz +0 -0
  59. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/phaneritic.clsf.gz +0 -0
  60. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/ultramafic.clsf.gz +0 -0
  61. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/CH_PalmeONeill2014.csv +95 -0
  62. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DDMM_WorkmanHart2005.csv +105 -0
  63. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DMM_WorkmanHart2005.csv +105 -0
  64. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DM_SaltersStrake2004.csv +95 -0
  65. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/EDMM_WorkmanHart2005.csv +105 -0
  66. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/PM_PalmeONeill2014.csv +95 -0
  67. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/timescale/geotimescale_spans.csv +180 -0
  68. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/geochem.py +821 -0
  69. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/melts.py +92 -0
  70. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__init__.py +10 -0
  71. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/__init__.cpython-36.pyc +0 -0
  72. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/db.cpython-36.pyc +0 -0
  73. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/ions.cpython-36.pyc +0 -0
  74. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/mineral.cpython-36.pyc +0 -0
  75. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/sites.cpython-36.pyc +0 -0
  76. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/db.py +88 -0
  77. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/ions.py +78 -0
  78. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/mineral.py +587 -0
  79. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/sites.py +134 -0
  80. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/norm.py +224 -0
  81. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/normalisation.py +204 -0
  82. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/plot.py +514 -0
  83. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__init__.py +13 -0
  84. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/__init__.cpython-36.pyc +0 -0
  85. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/database.cpython-36.pyc +0 -0
  86. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/env.cpython-36.pyc +0 -0
  87. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/general.cpython-36.pyc +0 -0
  88. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/georoc.cpython-36.pyc +0 -0
  89. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/math.cpython-36.pyc +0 -0
  90. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/melts.cpython-36.pyc +0 -0
  91. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multip.cpython-36.pyc +0 -0
  92. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multiprocessing.cpython-36.pyc +0 -0
  93. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/pd.cpython-36.pyc +0 -0
  94. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/plot.cpython-36.pyc +0 -0
  95. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/skl.cpython-36.pyc +0 -0
  96. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/spatial.cpython-36.pyc +0 -0
  97. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/text.cpython-36.pyc +0 -0
  98. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/time.cpython-36.pyc +0 -0
  99. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/wfs.cpython-36.pyc +0 -0
  100. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/database.py +88 -0
  101. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/env.py +81 -0
  102. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/general.py +266 -0
  103. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/georoc.py +444 -0
  104. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/math.py +371 -0
  105. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/melts.py +397 -0
  106. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multip.py +29 -0
  107. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multiprocessing.py +29 -0
  108. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/pd.py +214 -0
  109. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/plot.py +345 -0
  110. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/skl.py +847 -0
  111. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/spatial.py +91 -0
  112. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/text.py +207 -0
  113. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/time.py +224 -0
  114. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/wfs.py +10 -0
  115. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/PKG-INFO +61 -0
  116. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/SOURCES.txt +83 -0
  117. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/dependency_links.txt +1 -0
  118. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/requires.txt +47 -0
  119. ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/top_level.txt +1 -0
@@ -0,0 +1,82 @@
1
+ import numpy as np
2
+ import pandas as pd
3
+ import warnings
4
+ import logging
5
+
6
+ logging.getLogger(__name__).addHandler(logging.NullHandler())
7
+ logger = logging.getLogger(__name__)
8
+
9
+
10
+ def impute_ratios(ratios: pd.DataFrame):
11
+ """
12
+ Imputation function utilizing pandas which is used to fill out the
13
+ aggregated ratio matrix via chained ratio multiplication akin to
14
+ internal standardisation (e.g. Ti / MgO = Ti/SiO2 * SiO2 / MgO).
15
+
16
+ Parameters
17
+ ---------------
18
+ ratios: pd.DataFrame
19
+ Dataframe of ratios to impute.
20
+
21
+ Returns
22
+ -------
23
+ pd.DataFrame
24
+ A DataFrame of imputed ratios.
25
+ """
26
+ with warnings.catch_warnings():
27
+ # can get empty arrays which raise RuntimeWarnings
28
+ # consider changing to np.errstate
29
+ warnings.simplefilter("ignore", category=RuntimeWarning)
30
+ for IS in ratios.columns:
31
+ ser = ratios.loc[:, IS]
32
+ if ser.isnull().any():
33
+ non_null_idxs = ser.loc[~ser.isnull()].index.values
34
+ null_idxs = ser.loc[ser.isnull()].index.values
35
+ for null in null_idxs: # e.g. Ti / MgO = Ti/SiO2 * SiO2 / MgO
36
+ # e.g. SiO2/MgO ratios
37
+ inverse_ratios = ratios.loc[null, non_null_idxs]
38
+ # e.g. Ti/SiO2 ratios
39
+ non_null_ISratios = ratios.loc[non_null_idxs, IS]
40
+ predicted_ratios = inverse_ratios * non_null_ISratios
41
+ ratios.loc[null, IS] = np.exp(np.nanmean(np.log(predicted_ratios)))
42
+ return ratios
43
+
44
+
45
+ def np_impute_ratios(ratios: np.ndarray):
46
+ """
47
+ Imputation function utilizing numpy which is used to fill out the
48
+ aggregated ratio matrix via chained ratio multiplication akin to
49
+ internal standardisation (e.g. Ti / MgO = Ti/SiO2 * SiO2 / MgO).
50
+
51
+ Parameters
52
+ ---------------
53
+ ratios: np.ndarray
54
+ Array of ratios to impute.
55
+
56
+ Returns
57
+ -------
58
+ np.ndarray
59
+ Array of imputed ratios.
60
+ """
61
+ finite = np.isfinite(ratios)
62
+ not_finite = ~finite
63
+ if not_finite.any():
64
+ where_not_finite = np.argwhere(not_finite)
65
+ _ixs, _iys = where_not_finite.T
66
+ ixs = _ixs[~(_ixs == _iys)]
67
+ iys = _iys[~(_ixs == _iys)]
68
+ where_not_finite = np.stack((ixs, iys)).T
69
+ excludes = np.empty((ixs.size, ratios.shape[0] - 2)).astype(int)
70
+ indicies = np.arange(ratios.shape[0]).astype(int)
71
+ for enm_ix in np.arange(ixs.size):
72
+ excludes[enm_ix] = np.setdiff1d(indicies, where_not_finite[enm_ix])
73
+
74
+ for enm_ix in np.arange(ixs.size):
75
+ ex = excludes[enm_ix]
76
+ ix, iy = where_not_finite[enm_ix].T
77
+ with warnings.catch_warnings():
78
+ # can get empty arrays which raise RuntimeWarnings
79
+ # consider changing to np.errstate
80
+ warnings.simplefilter("ignore", category=RuntimeWarning)
81
+ ratios[ix, iy] = np.nanmean(ratios[ix, ex] + ratios[ex, iy])
82
+ return ratios
@@ -0,0 +1,40 @@
1
+ import numpy as np
2
+ import pandas as pd
3
+ import logging
4
+
5
+ logging.getLogger(__name__).addHandler(logging.NullHandler())
6
+ logger = logging.getLogger(__name__)
7
+
8
+
9
+ def close(X: np.ndarray):
10
+ if X.ndim == 2:
11
+ return np.divide(X, np.sum(X, axis=1)[:, np.newaxis])
12
+ else:
13
+ return np.divide(X, np.sum(X, axis=0))
14
+
15
+
16
+ def renormalise(df: pd.DataFrame, components: list = [], scale=100.0):
17
+ """
18
+ Renormalises compositional data to ensure closure.
19
+
20
+ Parameters
21
+ ------------
22
+ df: pd.DataFrame
23
+ Dataframe to renomalise.
24
+ components: list
25
+ Option subcompositon to renormalise to 100. Useful for the use case
26
+ where compostional data and non-compositional data are stored in the
27
+ same dataframe.
28
+ scale: float, 100.
29
+ Closure parameter. Typically either 100 or 1.
30
+ """
31
+ dfc = df.copy(deep=True)
32
+ if components:
33
+ cmpnts = [c for c in components if c in dfc.columns]
34
+ dfc.loc[:, cmpnts] = scale * dfc.loc[:, cmpnts].divide(
35
+ dfc.loc[:, cmpnts].sum(axis=1).replace(0, np.nan), axis=0
36
+ )
37
+ return dfc
38
+ else:
39
+ dfc = dfc.divide(dfc.sum(axis=1).replace(0, 100.0), axis=0) * scale
40
+ return dfc
@@ -0,0 +1,524 @@
1
+ from types import MethodType
2
+ import numpy as np
3
+ import pandas as pd
4
+ import scipy
5
+ from sklearn.base import TransformerMixin
6
+ import logging
7
+
8
+ logging.getLogger(__name__).addHandler(logging.NullHandler())
9
+ logger = logging.getLogger(__name__)
10
+
11
+
12
+ def close(X: np.ndarray):
13
+ if X.ndim == 2:
14
+ return np.divide(X, np.sum(X, axis=1)[:, np.newaxis])
15
+ else:
16
+ return np.divide(X, np.sum(X, axis=0))
17
+
18
+
19
+ def get_nonnan_column(arr:np.ndarray):
20
+ """Returns the first column without nans in it."""
21
+ if len(arr.shape)==1:
22
+ arr = arr.reshape((1, *arr.shape))
23
+ inds = np.arange(arr.shape[1])
24
+ wherenonnan = ~np.isnan(arr).any(axis=0)
25
+ ind = inds[wherenonnan][0]
26
+ return ind
27
+
28
+
29
+ def weights_from_array(arr:np.ndarray):
30
+ """
31
+ Returns a set of equal weights for components
32
+ along the first axis of an array.
33
+ """
34
+ wts = np.ones((arr.shape[0]))
35
+ wts = wts/np.sum(wts)
36
+ wts = wts
37
+ return wts
38
+
39
+
40
+ def nan_weighted_mean(arr:np.ndarray, weights=None,):
41
+ if weights is None:
42
+ weights = weights_from_array(arr)
43
+ weights = np.array(weights)/np.nansum(weights)
44
+
45
+ mask = (np.isnan(arr) + np.isinf(arr)) > 0
46
+ if not mask.any():
47
+ return np.average(arr,
48
+ weights=weights,
49
+ axis=0)
50
+ else:
51
+ return np.ma.average(np.ma.array(arr, mask=mask),
52
+ weights=weights,
53
+ axis=0)
54
+
55
+
56
+
57
+ def compositional_mean(df, weights=[], **kwargs):
58
+ """
59
+ Implements an aggregation using a weighted mean.
60
+ """
61
+ non_nan_cols = df.dropna(axis=1, how='all').columns
62
+ assert not df.loc[:, non_nan_cols].isna().values.any()
63
+ mean = df.iloc[0, :].copy()
64
+ if not weights:
65
+ weights = np.ones(len(df.index.values))
66
+ weights = np.array(weights)/np.nansum(weights)
67
+
68
+ logmean = alr(df.loc[:, non_nan_cols].values).T @ weights[:, np.newaxis]
69
+ mean.loc[non_nan_cols] = inv_alr(logmean.T.squeeze()) # this renormalises by default
70
+ return mean
71
+
72
+
73
+ def nan_weighted_compositional_mean(arr: np.ndarray,
74
+ weights=None,
75
+ ind=None,
76
+ renorm=True,
77
+ **kwargs):
78
+ """
79
+ Implements an aggregation using a weighted mean, but accounts
80
+ for nans. Requires at least one non-nan column for alr mean.
81
+
82
+ When used for internal standardisation, there should be only a single
83
+ common element - this would be used by default as the divisor here.
84
+
85
+ When used for multiple-standardisation, the [specified] or first common
86
+ element will be used.
87
+
88
+ Input array has analyses along the first axis.
89
+ """
90
+ if arr.ndim == 1: #if it's a single row
91
+ return arr
92
+ else:
93
+ if weights is None:
94
+ weights = weights_from_array(arr)
95
+ else:
96
+ weights = np.array(weights)/np.sum(weights, axis=-1)
97
+
98
+ if ind is None: # take the first column which has no nans
99
+ ind = get_nonnan_column(arr)
100
+
101
+ if arr.ndim < 3 and arr.shape[0] == 1:
102
+ div = arr[:, ind].squeeze() # check this
103
+ else:
104
+ div = arr[:, ind].squeeze()[:, np.newaxis]
105
+
106
+ logvals = np.log(np.divide(arr, div))
107
+ mean = np.nan * np.ones(arr.shape[1:])
108
+
109
+ ixs = np.arange(logvals.shape[1])
110
+ if arr.ndim == 2:
111
+ indexes = ixs
112
+ elif arr.ndim == 3:
113
+ iys = np.arange(logvals.shape[2])
114
+ indexes = np.ixs_(ixs, iys)
115
+
116
+ mean[indexes] = nan_weighted_mean(logvals[:, indexes],
117
+ weights=weights)
118
+
119
+ mean = np.exp(mean.squeeze())
120
+ if renorm: mean /= np.nansum(mean)
121
+ return mean
122
+
123
+
124
+ def cross_ratios(df: pd.DataFrame):
125
+ """
126
+ Takes ratios of values across a a dataframe,
127
+ such that columns are denominators and the row indexes the numerators,
128
+ to create a square array. Returns one array per record.
129
+ """
130
+ ratios = np.ones((len(df.index), len(df.columns), len(df.columns)))
131
+ for idx in range(df.index.size):
132
+ row_vals = df.iloc[idx, :].values
133
+ r1 = row_vals.T[:, np.newaxis] @ np.ones_like(row_vals)[np.newaxis, :]
134
+ ratios[idx] = r1 / r1.T
135
+ return ratios
136
+
137
+
138
+ def np_cross_ratios(arr: np.ndarray, debug=False):
139
+ """
140
+ Takes ratios of values across an array to create a square array,
141
+ such that columns are numerators and the row indexes the denominators.
142
+ Returns an array of arrays (one per record).
143
+ """
144
+ arr[arr <= 0] = np.nan
145
+ if arr.ndim == 1:
146
+ index_length = 1
147
+ arr = arr.reshape((1, *arr.shape))
148
+ else:
149
+ index_length = arr.shape[0]
150
+ dims = arr.shape[-1]
151
+ ratios = np.ones((index_length, dims, dims))
152
+ for idx in range(index_length):
153
+ row_vals = arr[idx, :]
154
+ r1 = row_vals.T[:, np.newaxis] @ np.ones_like(row_vals)[np.newaxis, :]
155
+ ratios[idx] = r1.T / r1
156
+
157
+ if debug:
158
+ try:
159
+ diags = ratios[:, np.arange(dims), np.arange(dims)]
160
+ # check all diags are 1.
161
+ assert np.allclose(diags, 1.)
162
+ except:
163
+ # check all diags are 1. or nan
164
+ assert np.allclose(diags[~np.isnan(diags)], 1.)
165
+
166
+ return ratios
167
+
168
+
169
+ def impute_ratios(ratios: pd.DataFrame):
170
+ """
171
+ Pandas version of ratio matrix imputation.
172
+ """
173
+ for IS in ratios.columns:
174
+ ser = ratios.loc[:, IS]
175
+ if ser.isnull().any():
176
+ non_null_idxs = ser.loc[~ser.isnull()].index.values
177
+ null_idxs = ser.loc[ser.isnull()].index.values
178
+ for null in null_idxs:
179
+ # e.g. Ti / MgO = Ti/SiO2 * SiO2 / MgO
180
+ inverse_ratios = ratios.loc[null, non_null_idxs] # e.g. SiO2/MgO ratios
181
+ non_null_ISratios = ratios.loc[non_null_idxs, IS] # e.g. Ti/SiO2 ratios
182
+ predicted_ratios = inverse_ratios * non_null_ISratios
183
+ ratios.loc[null, IS] = np.exp(np.nanmean(np.log(predicted_ratios)))
184
+ return ratios
185
+
186
+
187
+ def np_impute_ratios(ratios: np.ndarray):
188
+ """
189
+ Numpy version of ratio matrix imputation.
190
+ """
191
+ finite = np.isfinite(ratios)
192
+ not_finite = ~finite
193
+ if not_finite.any():
194
+ where_not_finite = np.argwhere(not_finite)
195
+ print(where_not_finite)
196
+ _ixs, _iys = where_not_finite.T
197
+ ixs = _ixs[~(_ixs == _iys)]
198
+ iys = _iys[~(_ixs == _iys)]
199
+ where_not_finite = np.stack((ixs, iys)).T
200
+ excludes = np.empty((ixs.size, ratios.shape[0]-2)).astype(int)
201
+ indicies = np.arange(ratios.shape[0]).astype(int)
202
+ for enm_ix in np.arange(ixs.size):
203
+ excludes[enm_ix] = np.setdiff1d(indicies, where_not_finite[enm_ix])
204
+
205
+ for enm_ix in np.arange(ixs.size):
206
+ ex = excludes[enm_ix]
207
+ ix, iy = where_not_finite[enm_ix].T
208
+ ratios[ix, iy] = np.nanmean(ratios[ix, ex] + ratios[ex, iy])
209
+ return ratios
210
+
211
+
212
+ def standardise_aggregate(df: pd.DataFrame,
213
+ int_std=None,
214
+ fixed_record_idx=0,
215
+ renorm=True,
216
+ **kwargs):
217
+ """
218
+ Performs internal standardisation and aggregates dissimilar geochemical records.
219
+ Note: this changes the closure parameter, and is generally intended to integrate
220
+ major and trace element records.
221
+ """
222
+ if df.index.size == 1: # catch single records
223
+ return df
224
+ else:
225
+ if int_std is None:
226
+ # Get the 'internal standard column'
227
+ potential_int_stds = df.count()[df.count()==df.count().max()].index.values
228
+ assert len(potential_int_stds) > 0
229
+ # Use an internal standard
230
+ int_std = potential_int_stds[0]
231
+ if len(potential_int_stds) > 1:
232
+ logging.info('Multiple int. stds possible. Using '+str(int_std))
233
+
234
+ non_nan_cols = df.dropna(axis=1, how='all').columns
235
+ assert len(non_nan_cols)
236
+ mean = nan_weighted_compositional_mean(df.values,
237
+ ind=df.columns.get_loc(int_std),
238
+ renorm=False)
239
+ ser = pd.Series(mean, index=df.columns)
240
+ multiplier = df.iloc[fixed_record_idx, df.columns.get_loc(int_std)] /\
241
+ ser[int_std]
242
+ ser *= multiplier
243
+ if renorm: ser /= np.nansum(ser.values)
244
+ return ser
245
+
246
+
247
+ def complex_standardise_aggregate(df,
248
+ int_std=None, # fallback parameters
249
+ renorm=True,
250
+ fixed_record_idx=0):
251
+
252
+ if int_std is None:
253
+ # create a n x d x d matrix for aggregating ratios
254
+ non_nan_cols = df.dropna(axis=1, how='all').columns
255
+ ratios = cross_ratios(df.loc[:, non_nan_cols])
256
+ # Average across record matricies
257
+ mean_ratios = pd.DataFrame(np.exp(np.nanmean(np.log(ratios), axis=0)),
258
+ columns=non_nan_cols,
259
+ index=non_nan_cols)
260
+ # Filling in the null values in a ratio matrix
261
+ imputed_ratios = impute_ratios(mean_ratios)
262
+ # We simply pick the first non-nan column.
263
+ IS = non_nan_cols[0]
264
+ mean = np.exp(np.mean(np.log(imputed_ratios/imputed_ratios.loc[IS, :]),
265
+ axis=1)
266
+ )
267
+ # This needs to be renormalised to make logical sense
268
+ mean /= np.nansum(mean.values)
269
+
270
+ out = np.ones((1, len(df.columns))) * np.nan
271
+ out[:, [list(df.columns).index(c) for c in non_nan_cols]] = mean
272
+ return pd.Series(out.squeeze(), index=df.columns)
273
+ else:
274
+ # fallback to internal standardisation
275
+ return standardise_aggregate(df,
276
+ int_std=int_std,
277
+ fixed_record_idx=fixed_record_idx,
278
+ renorm=renorm)
279
+
280
+
281
+ def np_complex_standardise_aggregate(df,
282
+ int_std=None, # fallback parameters
283
+ renorm=True,
284
+ fixed_record_idx=0):
285
+ """
286
+ Numpy version of complex internal standardisation.
287
+ """
288
+
289
+ if int_std is None:
290
+ # create a n x d x d matrix for aggregating ratios
291
+ non_nan_cols = df.dropna(axis=1, how='all').columns
292
+ assert len(non_nan_cols) > 0
293
+ ratios = np_cross_ratios(df.loc[:, non_nan_cols].values)
294
+ # Take the mean across the cross-ratio matricies
295
+ mean_logratios = np.nanmean(np.log(ratios), axis=0)
296
+ # Filling in the null values in a ratio matrix
297
+ imputed_log_ratios = np_impute_ratios(mean_logratios)
298
+ # We simply pick the first non-nan column.
299
+ #IS = 0
300
+ IS = np.argmax(np.count_nonzero(~np.isnan(imputed_log_ratios), axis=0))
301
+ # Convert to a composition by subtracting a row and taking negative
302
+ div_log_ratios = -(imputed_log_ratios - imputed_log_ratios[IS, :])
303
+ comp_abund = np.exp(np.nanmean(div_log_ratios, axis=1))
304
+ comp_abund /= np.nansum(comp_abund)
305
+ out = np.ones((1, len(df.columns))) * np.nan
306
+ inds = np.array([list(df.columns).index(c) for c in non_nan_cols])
307
+ out[:, inds] = comp_abund
308
+ return pd.Series(out.squeeze(), index=df.columns)
309
+ else:
310
+ # fallback to internal standardisation
311
+ return standardise_aggregate(df,
312
+ int_std=int_std,
313
+ fixed_record_idx=fixed_record_idx,
314
+ renorm=renorm)
315
+
316
+
317
+ def nancov(X, method='replace'):
318
+ """
319
+ Generates a covariance matrix excluding nan-components.
320
+ Done on a column-column/pairwise basis.
321
+ The result Y may not be a positive definite matrix.
322
+ """
323
+ if method=='rowexclude':
324
+ Xnanfree = X[np.all(np.isfinite(X), axis=1), :].T
325
+ #assert Xnanfree.shape[1] > Xnanfree.shape[0]
326
+ #(1/m)X^T*X
327
+ return np.cov(Xnanfree)
328
+ else:
329
+ X = np.array(X, ndmin=2, dtype=float)
330
+ X -= np.nanmean(X, axis=0)#[:, np.newaxis]
331
+ cov = np.empty((X.shape[1], X.shape[1]))
332
+ cols = range(X.shape[1])
333
+ for n in cols:
334
+ for m in [i for i in cols if i>=n] :
335
+ fn = np.isfinite(X[:, n])
336
+ fm = np.isfinite(X[:, m])
337
+ if method=='replace':
338
+ X[~fn, n] = 0
339
+ X[~fm, m] = 0
340
+ fact = fn.shape[0] - 1
341
+ c= np.dot(X[:, n], X[:, m])/fact
342
+ else:
343
+ f = fn & fm
344
+ fact = f.shape[0] - 1
345
+ c = np.dot(X[f, n], X[f, m])/fact
346
+ cov[n, m] = c
347
+ cov[m, n] = c
348
+ return cov
349
+
350
+
351
+ def renormalise(df: pd.DataFrame, components:list=[], scale=100.):
352
+ """
353
+ Renormalises compositional data to ensure closure.
354
+ A subset of components can be used for flexibility.
355
+ For data which sums to 0, 100 is returned - e.g. for TE-only datasets
356
+ """
357
+ dfc = df.copy()
358
+ if components:
359
+ cmpnts = [c for c in components if c in dfc.columns]
360
+ dfc.loc[:, cmpnts] = scale * dfc.loc[:, cmpnts].divide(
361
+ dfc.loc[:, cmpnts].sum(axis=1).replace(0, np.nan),
362
+ axis=0)
363
+ return dfc
364
+ else:
365
+ dfc = dfc.divide(dfc.sum(axis=1).replace(0, 100), axis=0) * scale
366
+ return dfc
367
+
368
+
369
+ def additive_log_ratio(X: np.ndarray, ind: int=-1):
370
+ """Additive log ratio transform. """
371
+
372
+ Y = X.copy()
373
+ assert Y.ndim in [1, 2]
374
+ dimensions = Y.shape[Y.ndim-1]
375
+ if ind < 0: ind += dimensions
376
+
377
+ if Y.ndim == 2:
378
+ Y = np.divide(Y, Y[:, ind][:, np.newaxis])
379
+ Y = np.log(Y[:, [i for i in range(dimensions) if not i==ind]])
380
+ else:
381
+ Y = np.divide(X, X[ind])
382
+ Y = np.log(Y[[i for i in range(dimensions) if not i==ind]])
383
+
384
+ return Y
385
+
386
+ def inverse_additive_log_ratio(Y: np.ndarray, ind=-1):
387
+ """
388
+ Inverse additive log ratio transform.
389
+ """
390
+ assert Y.ndim in [1, 2]
391
+
392
+ X = Y.copy()
393
+ dimensions = X.shape[X.ndim-1]
394
+ idx = np.arange(0, dimensions+1)
395
+
396
+ if ind != -1:
397
+ idx = np.array(list(idx[idx < ind]) +
398
+ [-1] +
399
+ list(idx[idx >= ind+1]-1))
400
+
401
+ # Add a zero-column and reorder columns
402
+ if Y.ndim == 2:
403
+ X = np.concatenate((X, np.zeros((X.shape[0], 1))), axis=1)
404
+ X = X[:, idx]
405
+ else:
406
+ X = np.append(X, np.array([0]))
407
+ X = X[idx]
408
+
409
+ # Inverse log and closure operations
410
+ X = np.exp(X)
411
+ X = close(X)
412
+ return X
413
+
414
+
415
+ def alr(*args, **kwargs):
416
+ return additive_log_ratio(*args, **kwargs)
417
+
418
+
419
+ def inv_alr(*args, **kwargs):
420
+ return inverse_additive_log_ratio(*args, **kwargs)
421
+
422
+
423
+ def clr(X: np.ndarray):
424
+ X = np.divide(X, np.sum(X, axis=1)[:, np.newaxis]) # Closure operation
425
+ Y = np.log(X) # Log operation
426
+ Y -= 1/X.shape[1] * np.nansum(Y, axis=1)[:, np.newaxis]
427
+ return Y
428
+
429
+
430
+ def inv_clr(Y: np.ndarray):
431
+ X = np.exp(Y) # Inverse of log operation
432
+ X = np.divide(X, np.nansum(X, axis=1)[:, np.newaxis]) #Closure operation
433
+ return X
434
+
435
+
436
+ def orthagonal_basis(X: np.ndarray):
437
+ D = X.shape[1]
438
+ H = scipy.linalg.helmert(D, full=False) # D-1, D Helmert matrix, exact representation of ψ as in Egozogue's book
439
+ return H[::-1]
440
+
441
+
442
+ def ilr(X: np.ndarray):
443
+ d = X.shape[1]
444
+ Y = clr(X)
445
+ psi = orthagonal_basis(X) # Get a basis
446
+ psi = orthagonal_basis(clr(X)) # trying to get right algorithm
447
+ assert np.allclose(psi @ psi.T, np.eye(d-1))
448
+ return Y @ psi.T
449
+
450
+
451
+ def inv_ilr(Y: np.ndarray, X: np.ndarray=None):
452
+ psi = orthagonal_basis(X)
453
+ C = Y @ psi
454
+ X = inv_clr(C) # Inverse log operation
455
+ return X
456
+
457
+
458
+ class LinearTransform(TransformerMixin):
459
+ def __init__(self, **kwargs):
460
+ self.kpairs = kwargs
461
+ self.label = 'Crude'
462
+
463
+ def transform(self, X, *args):
464
+ X = np.array(X)
465
+ return X
466
+
467
+ def inverse_transform(self, Y, *args):
468
+ Y = np.array(Y)
469
+ return Y
470
+
471
+ def fit(self, X, *args):
472
+ return self
473
+
474
+
475
+ class ALRTransform(TransformerMixin):
476
+ def __init__(self, **kwargs):
477
+ self.kpairs = kwargs
478
+ self.label = 'ALR'
479
+
480
+ def transform(self, X, *args, **kwargs):
481
+ X = np.array(X)
482
+ return alr(X, *args, **kwargs)
483
+
484
+ def inverse_transform(self, Y, *args, **kwargs):
485
+ Y = np.array(Y)
486
+ return inv_alr(Y, *args, **kwargs)
487
+
488
+ def fit(self, X, *args, **kwargs):
489
+ return self
490
+
491
+
492
+ class CLRTransform(TransformerMixin):
493
+ def __init__(self, **kwargs):
494
+ self.kpairs = kwargs
495
+ self.label = 'CLR'
496
+
497
+ def transform(self, X, *args, **kwargs):
498
+ X = np.array(X)
499
+ return clr(X, *args, **kwargs)
500
+
501
+ def inverse_transform(self, Y, *args, **kwargs):
502
+ Y = np.array(Y)
503
+ return inv_clr(Y, *args, **kwargs)
504
+
505
+ def fit(self, X, *args, **kwargs):
506
+ return self
507
+
508
+
509
+ class ILRTransform(TransformerMixin):
510
+ def __init__(self, **kwargs):
511
+ self.kpairs = kwargs
512
+ self.label = 'ILR'
513
+
514
+ def transform(self, X, *args, **kwargs):
515
+ X = np.array(X)
516
+ self.X = X
517
+ return ilr(X, *args, **kwargs)
518
+
519
+ def inverse_transform(self, Y, *args, **kwargs):
520
+ Y = np.array(Y)
521
+ return inv_ilr(Y, X=self.X, *args, **kwargs)
522
+
523
+ def fit(self, X, *args, **kwargs):
524
+ return self
@@ -0,0 +1,42 @@
1
+ AUSTRALIA.csv
2
+ AVANAVERA LARGE IGNEOUS PROVINCE.csv
3
+ CENTRAL ATLANTIC MAGMATIC PROVINCE - CAMP.csv
4
+ CHILCOTIN PLATEAU BASALTS.csv
5
+ COMEI LARGE IGNEOUS PROVINCE.csv
6
+ DECCAN.csv
7
+ EMEISHAN.csv
8
+ ETENDEKA PROVINCE.csv
9
+ ETHIOPIAN PLATEAU.csv
10
+ FRANKLIN LARGE IGNEOUS PROVINCE.csv
11
+ GUNBARREL IGNEOUS EVENT - MAMMOTH-WESTERN CHANNEL LARGE IGNEOUS PROVINCE.csv
12
+ HIGH ARCTIC LARGE IGNEOUS PROVINCE.csv
13
+ KAROO AND FERRAR PROVINCES.csv
14
+ KUONAMKA LARGE IGNEOUS PROVINCE.csv
15
+ KUZNETSK BASIN OR KUZBASS TRAPS.csv
16
+ LAKE VICTORIA LARGE IGNEOUS PROVINCE.csv
17
+ MACKENZIE LARGE IGNEOUS PROVINCE.csv
18
+ MADAGASCAR FLOOD BASALT.csv
19
+ MAIMECHA-KOTUI PROVINCE.csv
20
+ MALANI MAGMATIC PROVINCE;INDIA.csv
21
+ MARATHON LARGE IGNEOUS PROVINCE.csv
22
+ MARNDA MOORN LARGE IGNEOUS PROVINCE.csv
23
+ MATACHEWAN LARGE IGNEOUS PROVINCE.csv
24
+ MIDCONTINENT RIFT SYSTEM - KEWEENAWAN.csv
25
+ NANDALING - YANSHAN BELT.csv
26
+ NILUFER UNIT - YENISEHIR ASSOCIATION - PONTIDES.csv
27
+ NORTH ATLANTIC IGNEOUS PROVINCE OR NAIP.csv
28
+ NORTH GREENLAND PROTEROZOIC.csv
29
+ PANJAL-SOUTH QIANGTANG LARGE IGNEOUS PROVINCE.csv
30
+ PARANA.csv
31
+ QIANGTANG FLOOD BASALT PROVINCE.csv
32
+ RAJAHMUNDRY TRAPS.csv
33
+ RAJMAHAL-BENGAL-SYLHET.csv
34
+ RAMPUR-GARHWAL-MANDI-DARLA PROVINCE.csv
35
+ SIBERIAN TRAPS.csv
36
+ SOUTH TETHYAN SUTURE ZONE - PAKISTAN.csv
37
+ TARIM LARGE IGNEOUS PROVINCE.csv
38
+ TIBESTI VOLCANIC PROVINCE.csv
39
+ UMKONDO LARGE IGNEOUS PROVINCE.csv
40
+ WRANGELLIA.csv
41
+ YELLOWSTONE-SNAKE RIVER PLAIN VOLCANIC PROVINCE.csv
42
+ YEMEN PLATEAU.csv