pyrolite 0.0.14__zip
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__init__.py +10 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/_version.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/alteration.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/classification.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/compositions.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/geochem.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/melts.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/norm.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/normalisation.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/__pycache__/plot.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/_version.py +21 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/alteration.py +66 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/classification.py +222 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__init__.py +9 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/aggregate.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/codata.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/impute.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/__pycache__/renorm.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/aggregate.py +391 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/codata.py +266 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/impute.py +82 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/comp/renorm.py +40 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/compositions.py +524 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_CFB_Dataset_List.csv +42 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_Convergent_Dataset_List.csv +42 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OBFB_Dataset_List.csv +5 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OIB_Dataset_List.csv +49 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/GEOROC_OceanicPlateau_Dataset_List.csv +18 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/georoc/contents.json +1 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-35.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/__pycache__/env.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/melts/env.py +1063 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ba.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Bs.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.F.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.O3.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Pc.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.Ph.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.R.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.S3.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.T2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U1.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U2.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.U3.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.modelfields +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.nan.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS/TAS.none.modelfield +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/TAS.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/aphanitic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/gabbroic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/peralkalinity.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/phaneritic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/models/ultramafic.clsf.gz +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/CH_PalmeONeill2014.csv +95 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DDMM_WorkmanHart2005.csv +105 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DMM_WorkmanHart2005.csv +105 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/DM_SaltersStrake2004.csv +95 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/EDMM_WorkmanHart2005.csv +105 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/refcomp/PM_PalmeONeill2014.csv +95 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/data/timescale/geotimescale_spans.csv +180 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/geochem.py +821 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/melts.py +92 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__init__.py +10 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/db.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/ions.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/mineral.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/__pycache__/sites.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/db.py +88 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/ions.py +78 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/mineral.py +587 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/mineral/sites.py +134 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/norm.py +224 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/normalisation.py +204 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/plot.py +514 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__init__.py +13 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/__init__.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/database.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/env.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/general.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/georoc.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/math.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/melts.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multip.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/multiprocessing.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/pd.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/plot.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/skl.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/spatial.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/text.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/time.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/__pycache__/wfs.cpython-36.pyc +0 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/database.py +88 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/env.py +81 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/general.py +266 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/georoc.py +444 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/math.py +371 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/melts.py +397 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multip.py +29 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/multiprocessing.py +29 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/pd.py +214 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/plot.py +345 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/skl.py +847 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/spatial.py +91 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/text.py +207 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/time.py +224 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite/util/wfs.py +10 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/PKG-INFO +61 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/SOURCES.txt +83 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/dependency_links.txt +1 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/requires.txt +47 -0
- ProgramData/Anaconda3_64/Lib/site-packages/pyrolite-0.0.14-py3.6.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,444 @@
|
|
|
1
|
+
import urllib
|
|
2
|
+
from bs4 import BeautifulSoup
|
|
3
|
+
import requests
|
|
4
|
+
from http.client import HTTPResponse
|
|
5
|
+
import json
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
import pandas as pd
|
|
8
|
+
import numpy as np
|
|
9
|
+
from functools import partial
|
|
10
|
+
import re
|
|
11
|
+
import logging
|
|
12
|
+
|
|
13
|
+
from .pd import *
|
|
14
|
+
from .text import titlecase, parse_entry, split_records
|
|
15
|
+
from .general import (
|
|
16
|
+
temp_path,
|
|
17
|
+
urlify,
|
|
18
|
+
pyrolite_datafolder,
|
|
19
|
+
pathify,
|
|
20
|
+
iscollection,
|
|
21
|
+
internet_connection,
|
|
22
|
+
)
|
|
23
|
+
from ..geochem import tochem, check_multiple_cation_inclusion, aggregate_cation
|
|
24
|
+
from ..norm import scale_multiplier
|
|
25
|
+
|
|
26
|
+
logging.getLogger(__name__).addHandler(logging.NullHandler())
|
|
27
|
+
logger = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
# -----------------------------
|
|
30
|
+
# GEOROC INFO
|
|
31
|
+
# -----------------------------
|
|
32
|
+
__value_rx__ = r"(\s)*?(?P<value>[\.,\s\w]+\b)((\s)*?\[)?(?P<key>\w*)(\])?(\s)*?"
|
|
33
|
+
__cit_rx__ = r"(\s)*?(\[)?(?P<key>\w*)(\])?(\s)*?(?P<value>[\.\w]+)(\s)*?"
|
|
34
|
+
__full_cit_rx__ = r"(\s)*?\[(?P<key>\w*)\](\s)*(?P<value>.+)$"
|
|
35
|
+
__doi_rx__ = r"(.)*(doi(\s)*?:*)(\s)*(?P<value>\S*)"
|
|
36
|
+
|
|
37
|
+
_contents_file = pyrolite_datafolder(subfolder="georoc") / "contents.json"
|
|
38
|
+
|
|
39
|
+
if _contents_file.exists():
|
|
40
|
+
with open(str(_contents_file)) as fh:
|
|
41
|
+
__CONTENTS__ = json.loads(fh.read())
|
|
42
|
+
else:
|
|
43
|
+
__CONTENTS__ = {}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def subsitute_commas(entry):
|
|
47
|
+
if iscollection(entry):
|
|
48
|
+
return [x.replace(",", ";") for x in entry]
|
|
49
|
+
else:
|
|
50
|
+
return entry.replace(",", ";")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def parse_values(entry, sub=subsitute_commas, **kwargs):
|
|
54
|
+
"""
|
|
55
|
+
Wrapper for parse_entry for GEOROC formatted values.
|
|
56
|
+
|
|
57
|
+
Parameters
|
|
58
|
+
-------------
|
|
59
|
+
entry: pd.Series | str
|
|
60
|
+
String series formated as sequences of 'VALUE [NUMERIC_CITATION]'
|
|
61
|
+
separated by '/'. Else a string entry itself.
|
|
62
|
+
sub: function
|
|
63
|
+
Secondary subsitution function, here used for subsitution
|
|
64
|
+
(e.g. of commas).
|
|
65
|
+
"""
|
|
66
|
+
f = partial(parse_entry, regex=__value_rx__, delimiter="/", **kwargs)
|
|
67
|
+
if isinstance(entry, pd.Series):
|
|
68
|
+
return entry.apply(f).apply(sub)
|
|
69
|
+
else:
|
|
70
|
+
return sub(f(entry))
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def parse_citations(entry, **kwargs):
|
|
74
|
+
"""
|
|
75
|
+
Wrapper for parse_entry for GEOROC formatted citations.
|
|
76
|
+
|
|
77
|
+
Parameters
|
|
78
|
+
-------------
|
|
79
|
+
ser: pd.Series
|
|
80
|
+
String series formated as sequences of '[NUMERIC_CITATION] Citation'.
|
|
81
|
+
"""
|
|
82
|
+
f = partial(
|
|
83
|
+
parse_entry, regex=__full_cit_rx__, values_only=False, delimiter=None, **kwargs
|
|
84
|
+
)
|
|
85
|
+
if isinstance(entry, pd.Series):
|
|
86
|
+
return entry.apply(f)
|
|
87
|
+
else:
|
|
88
|
+
return f(entry)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def parse_DOI(entry, link=True, **kwargs):
|
|
92
|
+
"""
|
|
93
|
+
Wrapper for parse_entry for GEOROC formatted dois.
|
|
94
|
+
|
|
95
|
+
Parameters
|
|
96
|
+
-------------
|
|
97
|
+
ser: pd.Series
|
|
98
|
+
String series formated as sequences of 'Citation doi: DOI'.
|
|
99
|
+
"""
|
|
100
|
+
f = partial(
|
|
101
|
+
parse_entry,
|
|
102
|
+
regex=__doi_rx__,
|
|
103
|
+
values_only=True,
|
|
104
|
+
delimiter=None,
|
|
105
|
+
first_only=True,
|
|
106
|
+
replace_nan="",
|
|
107
|
+
**kwargs
|
|
108
|
+
)
|
|
109
|
+
if isinstance(entry, pd.Series):
|
|
110
|
+
return entry.apply(lambda x: r"{}{}".format(["", "dx.doi.org/"][link], f(x)))
|
|
111
|
+
else:
|
|
112
|
+
return r"{}{}".format(["", "dx.doi.org/"][link], f(entry))
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def get_georoc_links(
|
|
116
|
+
page="http://georoc.mpch-mainz.gwdg.de/georoc/CompFiles.aspx",
|
|
117
|
+
exclude=["Minerals", "Rocks", "Inclusions", "Georoc"],
|
|
118
|
+
):
|
|
119
|
+
"""
|
|
120
|
+
Parameters
|
|
121
|
+
------------
|
|
122
|
+
page: {str, HTTPResponse}
|
|
123
|
+
String URL or http.client.HTTPResponse to scrape for links.
|
|
124
|
+
exclude: list
|
|
125
|
+
List of collections not to get links for.
|
|
126
|
+
"""
|
|
127
|
+
if isinstance(page, str):
|
|
128
|
+
page = urllib.request.urlopen(page)
|
|
129
|
+
|
|
130
|
+
soup = BeautifulSoup(page, "html.parser")
|
|
131
|
+
links = [
|
|
132
|
+
link.get("href")
|
|
133
|
+
for link in list(soup.find_all("a"))
|
|
134
|
+
if not link.get("href") is None
|
|
135
|
+
]
|
|
136
|
+
pathlinks = [Path(i) for i in links if "_comp" in i]
|
|
137
|
+
groups = set([l.parent.name for l in pathlinks])
|
|
138
|
+
contents = {}
|
|
139
|
+
for g in groups:
|
|
140
|
+
name = titlecase(g.replace("_comp", "").replace("_", " "))
|
|
141
|
+
if name not in exclude:
|
|
142
|
+
abbrv = "".join([s for s in g if s == s.upper() and not s in ["_", "-"]])
|
|
143
|
+
# File names which include url_suffix:
|
|
144
|
+
grp = ["".join([g, "/", i.name]) for i in pathlinks if i.parent.name == g]
|
|
145
|
+
contents[name] = {"files": grp, "abbrv": abbrv}
|
|
146
|
+
|
|
147
|
+
return contents
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def update_georoc_filelist(
|
|
151
|
+
filepath=pyrolite_datafolder(subfolder="georoc") / "contents.json"
|
|
152
|
+
):
|
|
153
|
+
"""
|
|
154
|
+
Update a local copy listing the compilations available from GEOROC.
|
|
155
|
+
"""
|
|
156
|
+
try:
|
|
157
|
+
assert internet_connection(target="georoc.mpch-mainz.gwdg.de")
|
|
158
|
+
contents = get_georoc_links()
|
|
159
|
+
with open(str(filepath), "w") as fh:
|
|
160
|
+
fh.write(json.dumps(contents))
|
|
161
|
+
except AssertionError:
|
|
162
|
+
msg = "Unable to make onnection to GEOROC to update compilation lists."
|
|
163
|
+
logger.warning(msg)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def bulk_GEOROC_download(
|
|
167
|
+
output_folder=Path("~/Downloads/GEOROC"),
|
|
168
|
+
reservoirs=None,
|
|
169
|
+
redownload: bool = False,
|
|
170
|
+
write_hdf: bool = False,
|
|
171
|
+
write_pickle: bool = False,
|
|
172
|
+
):
|
|
173
|
+
"""
|
|
174
|
+
Download utility for GEOROC data. Facilitates incremental and resumed
|
|
175
|
+
downloadsself. Output data will be organised into folders by reservoir, and
|
|
176
|
+
stored as both i) individual CSVs and ii) a picked pd.DataFrame.
|
|
177
|
+
|
|
178
|
+
Notes
|
|
179
|
+
-----
|
|
180
|
+
Chemical abundance data are output as Wt% by default.
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
Parameters
|
|
184
|
+
----------
|
|
185
|
+
output_folder: {pathlib.Path('~/Downloads/GEOROC'), :obj:`str`}
|
|
186
|
+
Path to folder to store output data.
|
|
187
|
+
reservoirs: {None, :obj:`list`}
|
|
188
|
+
List of names (e.g. 'ConvergentMargins') or abbrevaitions (e.g. 'CM') for
|
|
189
|
+
GEOROC compilations to download.
|
|
190
|
+
redownload: {False, True}
|
|
191
|
+
Whether to redownload prevoiusly downloaded compilations.
|
|
192
|
+
write_hdf: {True, False}
|
|
193
|
+
Whether to create HDF5 files for each compilation.
|
|
194
|
+
write_pickle: {False, True}
|
|
195
|
+
Whether to create pickle files for each compilation.
|
|
196
|
+
"""
|
|
197
|
+
|
|
198
|
+
output_folder = output_folder or temp_path()
|
|
199
|
+
output_folder = Path(output_folder)
|
|
200
|
+
output_folder = output_folder.expanduser()
|
|
201
|
+
|
|
202
|
+
update_georoc_filelist()
|
|
203
|
+
|
|
204
|
+
reservoirs = reservoirs or __CONTENTS__.keys()
|
|
205
|
+
abbrvs = {__CONTENTS__[k]["abbrv"]: k for k in __CONTENTS__}
|
|
206
|
+
logger.info("Downloading only undownloaded files.")
|
|
207
|
+
if not redownload:
|
|
208
|
+
logger.info("Bulk download for {} beginning.".format(", ".join(reservoirs)))
|
|
209
|
+
|
|
210
|
+
completed = []
|
|
211
|
+
for res in reservoirs:
|
|
212
|
+
if res in __CONTENTS__.keys():
|
|
213
|
+
resname = res
|
|
214
|
+
resabbrv = v["abbrv"]
|
|
215
|
+
elif res in abbrvs:
|
|
216
|
+
resname = abbrvs[res]
|
|
217
|
+
resabbrv = res
|
|
218
|
+
else:
|
|
219
|
+
msg = "Unknown reservoir requested: {}".format(res)
|
|
220
|
+
logger.warn(msg)
|
|
221
|
+
|
|
222
|
+
if resname:
|
|
223
|
+
v = __CONTENTS__[resname]
|
|
224
|
+
|
|
225
|
+
resdir = output_folder / res
|
|
226
|
+
if not resdir.exists():
|
|
227
|
+
resdir.mkdir(parents=True)
|
|
228
|
+
|
|
229
|
+
out_aggfile = resdir / ("_" + res)
|
|
230
|
+
|
|
231
|
+
# Compilation List of Targets
|
|
232
|
+
filenames = v["files"]
|
|
233
|
+
|
|
234
|
+
# URL target
|
|
235
|
+
host = r"http://georoc.mpch-mainz.gwdg.de"
|
|
236
|
+
base_url = host + "/georoc/Csv_Downloads"
|
|
237
|
+
|
|
238
|
+
# Files yet to download, continuing from last 'save'
|
|
239
|
+
dwnld_fns = filenames
|
|
240
|
+
if not redownload:
|
|
241
|
+
# Just get the ones we don't have,
|
|
242
|
+
dwnld_stems = [(resdir / urlify(f)).stem for f in dwnld_fns]
|
|
243
|
+
current_files = [f.stem for f in resdir.iterdir() if f.is_file()]
|
|
244
|
+
dwnld_fns = [
|
|
245
|
+
f for f, s in zip(dwnld_fns, dwnld_stems) if not s in current_files
|
|
246
|
+
]
|
|
247
|
+
|
|
248
|
+
dataseturls = [
|
|
249
|
+
(urlify(d), base_url + r"/" + urlify(d)) for d in dwnld_fns if d.strip()
|
|
250
|
+
]
|
|
251
|
+
|
|
252
|
+
for name, url in dataseturls:
|
|
253
|
+
if "/" in name:
|
|
254
|
+
name = name.split("/")[-1]
|
|
255
|
+
outfile = (resdir / name).with_suffix("")
|
|
256
|
+
msg = "Downloading {} {} dataset to {}.".format(res, name, outfile)
|
|
257
|
+
logger.info(msg)
|
|
258
|
+
try:
|
|
259
|
+
df = download_GEOROC_compilation(url)
|
|
260
|
+
df.to_csv(outfile.with_suffix(".csv"))
|
|
261
|
+
except requests.exceptions.HTTPError as e:
|
|
262
|
+
pass
|
|
263
|
+
|
|
264
|
+
if write_hdf or write_pickle:
|
|
265
|
+
aggdf = df_from_csvs(resdir.glob("*.csv"), ignore_index=True)
|
|
266
|
+
msg = "Aggregated {} datasets ({} records).".format(
|
|
267
|
+
res, aggdf.index.size
|
|
268
|
+
)
|
|
269
|
+
logger.info(msg)
|
|
270
|
+
|
|
271
|
+
# Save the compilation
|
|
272
|
+
if write_pickle:
|
|
273
|
+
sparse_pickle_df(aggdf, out_aggfile)
|
|
274
|
+
|
|
275
|
+
if write_hdf:
|
|
276
|
+
min_itemsize = {
|
|
277
|
+
c: 100 for c in aggdf.columns[aggdf.dtypes == "object"]
|
|
278
|
+
}
|
|
279
|
+
min_itemsize.update({"Citations": 1200})
|
|
280
|
+
aggdf.to_hdf(
|
|
281
|
+
out_aggfile.with_suffix(".h5"),
|
|
282
|
+
out_aggfile.stem,
|
|
283
|
+
min_itemsize=min_itemsize,
|
|
284
|
+
mode="w",
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
logger.info("Download and aggregation for {} finished.".format(res))
|
|
288
|
+
completed.append(res)
|
|
289
|
+
logger.info("Bulk download for {} completed.".format(", ".join(completed)))
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def download_GEOROC_compilation(url: str):
|
|
293
|
+
"""
|
|
294
|
+
Downloads a specific GEOROC compilation and returns a cleaned and formatted
|
|
295
|
+
pd.DataFrame.
|
|
296
|
+
|
|
297
|
+
Parameters
|
|
298
|
+
----------
|
|
299
|
+
url: str
|
|
300
|
+
URL of specific compilation to download as a csv.
|
|
301
|
+
|
|
302
|
+
Returns
|
|
303
|
+
-------
|
|
304
|
+
pd.DataFrame
|
|
305
|
+
Dataframe representation of the GEOROC data.
|
|
306
|
+
"""
|
|
307
|
+
with requests.Session() as s:
|
|
308
|
+
response = s.get(url)
|
|
309
|
+
if response.status_code == requests.codes.ok:
|
|
310
|
+
logger.debug("Response recieved from {}.".format(url))
|
|
311
|
+
return format_GEOROC_response(response.content.decode("latin-1"))
|
|
312
|
+
else:
|
|
313
|
+
msg = "Failed download - bad status code at {}".format(url)
|
|
314
|
+
logger.warning(msg)
|
|
315
|
+
response.raise_for_status()
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def format_GEOROC_response(content: str, start_chem="SiO2", end_chem="Nd143Nd144"):
|
|
319
|
+
"""
|
|
320
|
+
Formats decoded content from GEOROC as a pd.DataFrame
|
|
321
|
+
|
|
322
|
+
Parameters
|
|
323
|
+
---------
|
|
324
|
+
content
|
|
325
|
+
Decoded string from GEOROC response.
|
|
326
|
+
|
|
327
|
+
Returns
|
|
328
|
+
-------
|
|
329
|
+
pd.DataFrame
|
|
330
|
+
"""
|
|
331
|
+
# GEOROC Specific Data Working
|
|
332
|
+
data, ref = re.split("\s?References:\s+", content)
|
|
333
|
+
datalines = [re.split(r'"\s?,\s?"', line) for line in re.split(r",\r", data)]
|
|
334
|
+
cols = [i.replace('"', "").replace(",", "") for i in datalines[0]]
|
|
335
|
+
cols = [titlecase(h, abbrv=["ID"]) for h in cols]
|
|
336
|
+
start = 1
|
|
337
|
+
finish = len(datalines)
|
|
338
|
+
if datalines[-1][0].strip().startswith("Abbreviations"):
|
|
339
|
+
finish -= 1
|
|
340
|
+
df = pd.DataFrame(datalines[start:finish], columns=cols)
|
|
341
|
+
cols = list(df.columns)
|
|
342
|
+
df = df.applymap(lambda x: str(x).replace('"', ""))
|
|
343
|
+
|
|
344
|
+
# Location names are extended with newlines
|
|
345
|
+
df.Location = df.Location.apply(lambda x: str(x).replace("\r\n", " / "))
|
|
346
|
+
|
|
347
|
+
df.Citations = df.Citations.apply(lambda x: re.findall(r"[\d]+", x))
|
|
348
|
+
# df = df.drop(index=df.index[~df.Citations.apply(lambda x: len(x))])
|
|
349
|
+
# Drop Empty Rows
|
|
350
|
+
df = df.dropna(how="all", axis=0)
|
|
351
|
+
df = df.set_index("UniqueID", drop=True)
|
|
352
|
+
df = df.apply(parse_values, axis=1)
|
|
353
|
+
|
|
354
|
+
# Translate headers and data units
|
|
355
|
+
cols = tochem([c.replace("(wt%)", "").replace("(ppm)", "") for c in df.columns])
|
|
356
|
+
start = cols.index("SiO2")
|
|
357
|
+
end = cols.index("143Nd144Nd")
|
|
358
|
+
where_ppm = [
|
|
359
|
+
(("ppm" in t) and (ix >= start and ix <= end))
|
|
360
|
+
for ix, t in enumerate(df.columns)
|
|
361
|
+
]
|
|
362
|
+
|
|
363
|
+
# Rename columns
|
|
364
|
+
df.columns = cols
|
|
365
|
+
headercols = list(df.columns[:start])
|
|
366
|
+
chemcols = list(df.columns[start:end])
|
|
367
|
+
trailingcols = list(df.columns[end:]) # trailing are generally isotope ratios
|
|
368
|
+
# Numeric data
|
|
369
|
+
|
|
370
|
+
numheaders = [
|
|
371
|
+
"ElevationMin",
|
|
372
|
+
"ElevationMax",
|
|
373
|
+
"LatitudeMin",
|
|
374
|
+
"LatitudeMax",
|
|
375
|
+
"LongitudeMin",
|
|
376
|
+
"LongitudeMax",
|
|
377
|
+
"Min.Age(yrs.)",
|
|
378
|
+
"Max.Age(yrs.)",
|
|
379
|
+
]
|
|
380
|
+
|
|
381
|
+
numeric_cols = numheaders + chemcols + trailingcols
|
|
382
|
+
# can include duplicates at this stage.
|
|
383
|
+
numeric_cols = [i for i in df.columns if i in numeric_cols]
|
|
384
|
+
numeric_ixs = [ix for ix, i in enumerate(df.columns) if i in numeric_cols]
|
|
385
|
+
df[numeric_cols] = df.iloc[:, numeric_ixs].apply(
|
|
386
|
+
pd.to_numeric, errors="coerce", axis=1
|
|
387
|
+
)
|
|
388
|
+
# remove <0.
|
|
389
|
+
chem_ixs = [ix for ix, i in enumerate(df.columns) if i in chemcols]
|
|
390
|
+
df.iloc[:, chem_ixs] = df.iloc[:, chem_ixs].mask(
|
|
391
|
+
df.iloc[:, chem_ixs] <= 0.0, other=np.nan
|
|
392
|
+
)
|
|
393
|
+
|
|
394
|
+
# units conversion -- convert to Wt%
|
|
395
|
+
df.iloc[:, where_ppm] *= scale_multiplier("ppm", "Wt%")
|
|
396
|
+
|
|
397
|
+
# deal with duplicate columns
|
|
398
|
+
collist = list(df.columns)
|
|
399
|
+
dup_chemcols = df.columns[
|
|
400
|
+
df.columns.duplicated() & [i in chemcols for i in collist]
|
|
401
|
+
]
|
|
402
|
+
for chem in dup_chemcols:
|
|
403
|
+
# replace the first (non-duplicated) column with the sum
|
|
404
|
+
ix = collist.index(chem)
|
|
405
|
+
df.iloc[:, ix] = df.loc[:, chem].apply(np.nansum, axis=1)
|
|
406
|
+
|
|
407
|
+
df = df.iloc[:, ~df.columns.duplicated()]
|
|
408
|
+
|
|
409
|
+
# Process the reference data.
|
|
410
|
+
reflines = split_records(ref)
|
|
411
|
+
reflines = [line.replace('"', "") for line in reflines]
|
|
412
|
+
reflines = [line.replace("\r\n", "") for line in reflines]
|
|
413
|
+
reflines = [parse_citations(i) for i in reflines if i]
|
|
414
|
+
refdf = pd.DataFrame.from_records(reflines).set_index("key", drop=True)
|
|
415
|
+
# Replace the reference indexes with references.
|
|
416
|
+
df.Citations = df.Citations.apply(
|
|
417
|
+
lambda lst: "; ".join([refdf.loc[x, "value"] for x in lst])
|
|
418
|
+
)
|
|
419
|
+
df["doi"] = df.Citations.apply(parse_DOI)
|
|
420
|
+
return df
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def load_georoc_frame(path):
|
|
424
|
+
"""
|
|
425
|
+
Munge GEOROC Data from pickle
|
|
426
|
+
Data should be converted to numeric and units already.
|
|
427
|
+
"""
|
|
428
|
+
df = load_sparse_pickle_df(path)
|
|
429
|
+
return df
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def georoc_munge(df):
|
|
433
|
+
"""
|
|
434
|
+
Collection of munging and feature adding functions for GEROROC data.
|
|
435
|
+
|
|
436
|
+
Todo: GEOL + AGE = AGE
|
|
437
|
+
"""
|
|
438
|
+
mulitiple_cations = check_multiple_cation_inclusion(df)
|
|
439
|
+
df = aggregate_cation(df, "Ti", form="element")
|
|
440
|
+
df.loc[:, "GeolAge"] = df.loc[:, "Geol."].replace("None", "") + df.Age
|
|
441
|
+
|
|
442
|
+
df.loc[:, "Lat"] = (df.LatitudeMax + df.LatitudeMin) / 2.0
|
|
443
|
+
df.loc[:, "Long"] = (df.LongitudeMax + df.LongitudeMin) / 2.0
|
|
444
|
+
return df
|