hinode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hinode/__init__.py +29 -0
- hinode/py.typed +0 -0
- hinode/xrt/__init__.py +17 -0
- hinode/xrt/_data.py +558 -0
- hinode/xrt/_data_test.py +638 -0
- hinode/xrt/_filtergrams.py +576 -0
- hinode/xrt/_filtergrams_test.py +518 -0
- hinode/xrt/_xrt.py +63 -0
- hinode/xrt/_xrt_test.py +14 -0
- hinode-0.1.0.dist-info/METADATA +77 -0
- hinode-0.1.0.dist-info/RECORD +14 -0
- hinode-0.1.0.dist-info/WHEEL +5 -0
- hinode-0.1.0.dist-info/licenses/LICENSE +28 -0
- hinode-0.1.0.dist-info/top_level.txt +1 -0
hinode/__init__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Download and analyze observations from the Hinode satellite.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import pathlib
|
|
6
|
+
import joblib
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"directory_default",
|
|
10
|
+
"memory",
|
|
11
|
+
"xrt",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
directory_default = pathlib.Path.home() / ".hinode/cache"
|
|
15
|
+
"""The default directory for downloaded files."""
|
|
16
|
+
|
|
17
|
+
memory = joblib.Memory(location=directory_default / "joblib", verbose=0)
|
|
18
|
+
"""
|
|
19
|
+
A representation of the cache which stores intermediate results,
|
|
20
|
+
such as the headers of the files in the archives.
|
|
21
|
+
|
|
22
|
+
The cache has a directory of its own, apart from the downloaded files,
|
|
23
|
+
so that :meth:`joblib.Memory.clear` does not delete them.
|
|
24
|
+
Assign another :class:`joblib.Memory` to this attribute to move the cache,
|
|
25
|
+
or one whose location is :obj:`None` to stop caching.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
# Imported after the attributes above, which the subpackages use
|
|
29
|
+
from . import xrt # noqa: E402
|
hinode/py.typed
ADDED
|
File without changes
|
hinode/xrt/__init__.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Download and analyze images from the X-Ray Telescope (XRT).
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from ._data import (
|
|
6
|
+
urls,
|
|
7
|
+
download,
|
|
8
|
+
)
|
|
9
|
+
from ._filtergrams import Filtergram
|
|
10
|
+
from ._xrt import open
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"urls",
|
|
14
|
+
"download",
|
|
15
|
+
"Filtergram",
|
|
16
|
+
"open",
|
|
17
|
+
]
|
hinode/xrt/_data.py
ADDED
|
@@ -0,0 +1,558 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import re
|
|
3
|
+
import time
|
|
4
|
+
import uuid
|
|
5
|
+
import typing
|
|
6
|
+
import pathlib
|
|
7
|
+
import datetime
|
|
8
|
+
import warnings
|
|
9
|
+
import functools
|
|
10
|
+
import concurrent.futures
|
|
11
|
+
import requests
|
|
12
|
+
import joblib
|
|
13
|
+
import numpy as np
|
|
14
|
+
import astropy.units as u
|
|
15
|
+
import astropy.time
|
|
16
|
+
import astropy.io.fits
|
|
17
|
+
import named_arrays as na
|
|
18
|
+
import hinode
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"urls",
|
|
22
|
+
"download",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
_url_level_1 = "https://umbra.nascom.nasa.gov/hinode/xrt/level1/"
|
|
26
|
+
"""
|
|
27
|
+
The archive of Level 1 XRT images at the Solar Data Analysis Center,
|
|
28
|
+
with one directory for each hour.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
_pattern_file = re.compile(r'href="(L1_XRT(\d{8})_(\d{6}\.\d)\.fits)"')
|
|
32
|
+
"""
|
|
33
|
+
The name of a Level 1 file in a directory listing of the archive,
|
|
34
|
+
which holds the date and the start time of the exposure,
|
|
35
|
+
as in ``L1_XRT20190930_180837.5.fits``.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
_size_block = 2880
|
|
39
|
+
"""The number of bytes in a block of a FITS file."""
|
|
40
|
+
|
|
41
|
+
_size_card = 80
|
|
42
|
+
"""The number of bytes in a card of a FITS header."""
|
|
43
|
+
|
|
44
|
+
_num_workers = 8
|
|
45
|
+
"""
|
|
46
|
+
The number of requests made to the archive at once,
|
|
47
|
+
since most of the time of each one is spent waiting for the server.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
_delay_retry = 1
|
|
51
|
+
"""
|
|
52
|
+
The number of seconds to wait before trying a request again,
|
|
53
|
+
which is doubled before each further try.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
_status_retry = (408, 429)
|
|
57
|
+
"""
|
|
58
|
+
The statuses below 500 with which a server refuses a request that may
|
|
59
|
+
succeed if it is tried again.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
_errors_retry = (
|
|
63
|
+
requests.exceptions.ConnectionError,
|
|
64
|
+
requests.exceptions.Timeout,
|
|
65
|
+
requests.exceptions.ChunkedEncodingError,
|
|
66
|
+
)
|
|
67
|
+
"""
|
|
68
|
+
The errors of a request which may not happen if it is tried again.
|
|
69
|
+
|
|
70
|
+
The archive sometimes ends a response before the length it promised,
|
|
71
|
+
which :mod:`urllib3` reports as a
|
|
72
|
+
:class:`~requests.exceptions.ChunkedEncodingError`.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
_errors_header = (
|
|
76
|
+
FileNotFoundError,
|
|
77
|
+
requests.exceptions.HTTPError,
|
|
78
|
+
ValueError,
|
|
79
|
+
)
|
|
80
|
+
"""
|
|
81
|
+
The errors of reading the header of a file in the archive
|
|
82
|
+
which are the fault of that file alone,
|
|
83
|
+
such as the file having been removed since the directory was listed,
|
|
84
|
+
or not being a FITS file.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
_delay_replace = 0.1
|
|
88
|
+
"""
|
|
89
|
+
The number of seconds to wait before trying to move a downloaded file
|
|
90
|
+
into place again,
|
|
91
|
+
which is doubled before each further try.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
_signature_fits = b"SIMPLE ="
|
|
95
|
+
"""The first bytes of every FITS file."""
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _get(
|
|
99
|
+
url: str,
|
|
100
|
+
num_retry: int = 5,
|
|
101
|
+
headers: None | dict[str, str] = None,
|
|
102
|
+
) -> requests.Response:
|
|
103
|
+
"""
|
|
104
|
+
Get a URL, trying again if the connection fails, the server is busy or
|
|
105
|
+
has an error, or the response is cut short.
|
|
106
|
+
|
|
107
|
+
Each try waits twice as long as the one before it,
|
|
108
|
+
starting from :data:`_delay_retry`.
|
|
109
|
+
|
|
110
|
+
Parameters
|
|
111
|
+
----------
|
|
112
|
+
url
|
|
113
|
+
The URL to get.
|
|
114
|
+
num_retry
|
|
115
|
+
The number of times to try to connect to the server.
|
|
116
|
+
headers
|
|
117
|
+
Additional HTTP headers to send with the request.
|
|
118
|
+
|
|
119
|
+
Raises
|
|
120
|
+
------
|
|
121
|
+
FileNotFoundError
|
|
122
|
+
If the server does not have the URL.
|
|
123
|
+
requests.exceptions.HTTPError
|
|
124
|
+
If the server refuses the request for a reason which trying again
|
|
125
|
+
cannot change, such as a 403 status.
|
|
126
|
+
requests.exceptions.RequestException
|
|
127
|
+
If the request cannot be made at all, such as for a malformed URL.
|
|
128
|
+
ConnectionError
|
|
129
|
+
If no attempt succeeded.
|
|
130
|
+
"""
|
|
131
|
+
error = None
|
|
132
|
+
for i in range(num_retry):
|
|
133
|
+
if i > 0:
|
|
134
|
+
time.sleep(_delay_retry * 2 ** (i - 1))
|
|
135
|
+
try:
|
|
136
|
+
response = requests.get(url, headers=headers, timeout=60)
|
|
137
|
+
response.raise_for_status()
|
|
138
|
+
except requests.exceptions.HTTPError as e:
|
|
139
|
+
status = None if e.response is None else e.response.status_code
|
|
140
|
+
if status == 404:
|
|
141
|
+
raise FileNotFoundError(url) from e
|
|
142
|
+
if status is not None and status < 500 and status not in _status_retry:
|
|
143
|
+
raise
|
|
144
|
+
error = e
|
|
145
|
+
except _errors_retry as e:
|
|
146
|
+
error = e
|
|
147
|
+
else:
|
|
148
|
+
return response
|
|
149
|
+
|
|
150
|
+
raise ConnectionError(f"Could not get {url} in {num_retry} tries.") from error
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _hours(
|
|
154
|
+
start: str | astropy.time.Time,
|
|
155
|
+
stop: str | astropy.time.Time,
|
|
156
|
+
) -> list[datetime.datetime]:
|
|
157
|
+
"""
|
|
158
|
+
The hours of the directories of the archive which hold the files that
|
|
159
|
+
began during a time range.
|
|
160
|
+
|
|
161
|
+
The directories are named in UTC,
|
|
162
|
+
and the hours are counted as :class:`datetime.datetime` objects,
|
|
163
|
+
which, unlike UTC times, have no leap seconds,
|
|
164
|
+
so an hour with a leap second at its end is not counted twice.
|
|
165
|
+
|
|
166
|
+
Parameters
|
|
167
|
+
----------
|
|
168
|
+
start
|
|
169
|
+
The start of the time range.
|
|
170
|
+
stop
|
|
171
|
+
The end of the time range, which is not included,
|
|
172
|
+
so a range which stops on the hour does not include that hour.
|
|
173
|
+
"""
|
|
174
|
+
format_hour = "%Y-%m-%dT%H"
|
|
175
|
+
|
|
176
|
+
def hour_utc(time: astropy.time.Time) -> datetime.datetime:
|
|
177
|
+
text = str(time.strftime(format_hour))
|
|
178
|
+
return datetime.datetime.strptime(text, format_hour)
|
|
179
|
+
|
|
180
|
+
start = astropy.time.Time(start, scale="utc")
|
|
181
|
+
stop = astropy.time.Time(stop, scale="utc")
|
|
182
|
+
|
|
183
|
+
hour = hour_utc(start)
|
|
184
|
+
last = hour_utc(stop)
|
|
185
|
+
if astropy.time.Time(last.isoformat(), scale="utc") == stop:
|
|
186
|
+
last -= datetime.timedelta(hours=1)
|
|
187
|
+
|
|
188
|
+
result = []
|
|
189
|
+
while hour <= last:
|
|
190
|
+
result.append(hour)
|
|
191
|
+
hour += datetime.timedelta(hours=1)
|
|
192
|
+
|
|
193
|
+
return result
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _files(
|
|
197
|
+
hour: datetime.datetime,
|
|
198
|
+
num_retry: int = 5,
|
|
199
|
+
) -> list[tuple[str, str]]:
|
|
200
|
+
"""
|
|
201
|
+
The URL and the start time, from the name, of every Level 1 file in the
|
|
202
|
+
directory of the archive which holds a given hour.
|
|
203
|
+
|
|
204
|
+
The start time is in UTC, in the ISOT format.
|
|
205
|
+
|
|
206
|
+
Parameters
|
|
207
|
+
----------
|
|
208
|
+
hour
|
|
209
|
+
The hour, in UTC.
|
|
210
|
+
num_retry
|
|
211
|
+
The number of times to try to connect to the server.
|
|
212
|
+
"""
|
|
213
|
+
url = f"{_url_level_1}{hour.strftime('%Y/%m/%d/H%H00')}/"
|
|
214
|
+
|
|
215
|
+
try:
|
|
216
|
+
response = _get(url, num_retry)
|
|
217
|
+
except FileNotFoundError:
|
|
218
|
+
return []
|
|
219
|
+
|
|
220
|
+
result = []
|
|
221
|
+
for name, date, clock in dict.fromkeys(_pattern_file.findall(response.text)):
|
|
222
|
+
isot = f"{date[:4]}-{date[4:6]}-{date[6:]}T{clock[:2]}:{clock[2:4]}:{clock[4:]}"
|
|
223
|
+
result.append((url + name, isot))
|
|
224
|
+
|
|
225
|
+
return result
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _header_string(
|
|
229
|
+
url: str,
|
|
230
|
+
num_retry: int = 5,
|
|
231
|
+
) -> str:
|
|
232
|
+
"""
|
|
233
|
+
The text of the primary header of a FITS file on a server,
|
|
234
|
+
read from the start of the file without downloading its data.
|
|
235
|
+
|
|
236
|
+
Parameters
|
|
237
|
+
----------
|
|
238
|
+
url
|
|
239
|
+
The URL of the FITS file.
|
|
240
|
+
num_retry
|
|
241
|
+
The number of times to try to connect to the server.
|
|
242
|
+
"""
|
|
243
|
+
num_blocks = 8
|
|
244
|
+
|
|
245
|
+
while True:
|
|
246
|
+
size = num_blocks * _size_block
|
|
247
|
+
response = _get(
|
|
248
|
+
url=url,
|
|
249
|
+
num_retry=num_retry,
|
|
250
|
+
headers={"Range": f"bytes=0-{size - 1}"},
|
|
251
|
+
)
|
|
252
|
+
content = response.content
|
|
253
|
+
|
|
254
|
+
for start in range(0, len(content) - _size_card + 1, _size_card):
|
|
255
|
+
card = content[start : start + _size_card]
|
|
256
|
+
if card.rstrip() == b"END":
|
|
257
|
+
return content[: start + _size_card].decode("ascii")
|
|
258
|
+
|
|
259
|
+
if len(content) < size:
|
|
260
|
+
raise ValueError(f"The header of {url} has no END card.")
|
|
261
|
+
|
|
262
|
+
num_blocks *= 2
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
@functools.cache
|
|
266
|
+
def _header_string_cached(
|
|
267
|
+
memory: joblib.Memory,
|
|
268
|
+
) -> typing.Callable[[str, int], str]:
|
|
269
|
+
"""
|
|
270
|
+
:func:`_header_string`, cached in the given cache,
|
|
271
|
+
since the archive does not change.
|
|
272
|
+
|
|
273
|
+
The cached function is made once for each cache,
|
|
274
|
+
so that the cache records the code of the function it caches only once,
|
|
275
|
+
rather than from every thread at once.
|
|
276
|
+
|
|
277
|
+
Parameters
|
|
278
|
+
----------
|
|
279
|
+
memory
|
|
280
|
+
The cache to store the headers in.
|
|
281
|
+
"""
|
|
282
|
+
result = memory.cache(_header_string, ignore=["num_retry"])
|
|
283
|
+
return typing.cast(typing.Callable[[str, int], str], result)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _header(
|
|
287
|
+
url: str,
|
|
288
|
+
num_retry: int = 5,
|
|
289
|
+
) -> astropy.io.fits.Header:
|
|
290
|
+
"""
|
|
291
|
+
The primary header of a FITS file on a server,
|
|
292
|
+
cached in :data:`hinode.memory`, since the archive does not change.
|
|
293
|
+
|
|
294
|
+
Parameters
|
|
295
|
+
----------
|
|
296
|
+
url
|
|
297
|
+
The URL of the FITS file.
|
|
298
|
+
num_retry
|
|
299
|
+
The number of times to try to connect to the server.
|
|
300
|
+
"""
|
|
301
|
+
text = _header_string_cached(hinode.memory)(url, num_retry)
|
|
302
|
+
return astropy.io.fits.Header.fromstring(text)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _filter(header: astropy.io.fits.Header) -> str:
|
|
306
|
+
"""
|
|
307
|
+
The filters an image was taken through,
|
|
308
|
+
named as in :func:`urls`.
|
|
309
|
+
|
|
310
|
+
Parameters
|
|
311
|
+
----------
|
|
312
|
+
header
|
|
313
|
+
The primary header of an XRT file.
|
|
314
|
+
"""
|
|
315
|
+
names = [str(header[key]).strip() for key in ["EC_FW1_", "EC_FW2_"]]
|
|
316
|
+
names = [name for name in names if name != "Open"]
|
|
317
|
+
if not names:
|
|
318
|
+
return "Open"
|
|
319
|
+
return "/".join(names)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def urls(
|
|
323
|
+
time_start: str | astropy.time.Time,
|
|
324
|
+
time_stop: str | astropy.time.Time,
|
|
325
|
+
filter: str = "Al_poly",
|
|
326
|
+
axis_time: str = "time",
|
|
327
|
+
num_retry: int = 5,
|
|
328
|
+
) -> na.ScalarArray:
|
|
329
|
+
"""
|
|
330
|
+
Find the URLs of the Level 1 XRT images in the archive of the Solar Data
|
|
331
|
+
Analysis Center which began during a given time range and were taken
|
|
332
|
+
through a given filter.
|
|
333
|
+
|
|
334
|
+
The archive does not say which filter each image was taken through,
|
|
335
|
+
so the start of each file is downloaded to read its header,
|
|
336
|
+
and the headers are cached in :data:`hinode.memory`.
|
|
337
|
+
Only the images whose type, ``EC_IMTY_``, is ``normal`` are found,
|
|
338
|
+
which leaves out the dark frames.
|
|
339
|
+
A file whose header cannot be read,
|
|
340
|
+
such as one removed from the archive after its directory was listed,
|
|
341
|
+
is left out with a warning.
|
|
342
|
+
|
|
343
|
+
Parameters
|
|
344
|
+
----------
|
|
345
|
+
time_start
|
|
346
|
+
The earliest start time of the images.
|
|
347
|
+
time_stop
|
|
348
|
+
The time before which an image must begin.
|
|
349
|
+
filter
|
|
350
|
+
The filters in the two filter wheels of XRT,
|
|
351
|
+
named as in the ``EC_FW1_`` and ``EC_FW2_`` header keywords,
|
|
352
|
+
with the open positions left out and the filters joined by a slash.
|
|
353
|
+
For example ``"Al_poly"``, ``"Ti_poly"``, ``"Al_poly/Ti_poly"``,
|
|
354
|
+
``"Gband"``, or ``"Open"`` if both wheels are open.
|
|
355
|
+
axis_time
|
|
356
|
+
The logical axis corresponding to changes in time.
|
|
357
|
+
num_retry
|
|
358
|
+
The number of times to try to connect to the server.
|
|
359
|
+
|
|
360
|
+
Examples
|
|
361
|
+
--------
|
|
362
|
+
|
|
363
|
+
Find the Al_poly images captured while the EUV Snapshot Imaging
|
|
364
|
+
Spectrograph (ESIS) was observing the Sun on 2019 September 30.
|
|
365
|
+
|
|
366
|
+
.. jupyter-execute::
|
|
367
|
+
|
|
368
|
+
import hinode
|
|
369
|
+
|
|
370
|
+
hinode.xrt.urls(
|
|
371
|
+
time_start="2019-09-30T18:06:11",
|
|
372
|
+
time_stop="2019-09-30T18:11:01",
|
|
373
|
+
filter="Al_poly",
|
|
374
|
+
)
|
|
375
|
+
"""
|
|
376
|
+
start = astropy.time.Time(time_start)
|
|
377
|
+
stop = astropy.time.Time(time_stop)
|
|
378
|
+
|
|
379
|
+
def header(url: str) -> astropy.io.fits.Header | Exception:
|
|
380
|
+
try:
|
|
381
|
+
return _header(url, num_retry)
|
|
382
|
+
except _errors_header as e:
|
|
383
|
+
return e
|
|
384
|
+
|
|
385
|
+
result: list[str] = []
|
|
386
|
+
|
|
387
|
+
with concurrent.futures.ThreadPoolExecutor(_num_workers) as executor:
|
|
388
|
+
|
|
389
|
+
listings = executor.map(
|
|
390
|
+
lambda hour: _files(hour, num_retry),
|
|
391
|
+
_hours(time_start, time_stop),
|
|
392
|
+
)
|
|
393
|
+
files = dict(file for listing in listings for file in listing)
|
|
394
|
+
|
|
395
|
+
# The name holds the start time cut to a tenth of a second,
|
|
396
|
+
# so it can be a little earlier than the start time in the header.
|
|
397
|
+
candidates = []
|
|
398
|
+
if files:
|
|
399
|
+
time_name = astropy.time.Time(list(files.values()), scale="utc")
|
|
400
|
+
margin = astropy.time.TimeDelta(1 * u.s)
|
|
401
|
+
where = (start - margin <= time_name) & (time_name < stop)
|
|
402
|
+
candidates = [url for url, w in zip(files, where) if w]
|
|
403
|
+
|
|
404
|
+
# The first header is read before the others, so that the cache
|
|
405
|
+
# records the code of the function it caches from one thread,
|
|
406
|
+
# rather than from every thread at once.
|
|
407
|
+
headers = [header(url) for url in candidates[:1]]
|
|
408
|
+
headers += executor.map(header, candidates[1:])
|
|
409
|
+
|
|
410
|
+
selected = []
|
|
411
|
+
for url, h in zip(candidates, headers):
|
|
412
|
+
if isinstance(h, Exception):
|
|
413
|
+
warnings.warn(
|
|
414
|
+
f"Left out {url}, since its header could not be read: {h!r}",
|
|
415
|
+
stacklevel=2,
|
|
416
|
+
)
|
|
417
|
+
elif _filter(h) == filter and h.get("EC_IMTY_") == "normal":
|
|
418
|
+
selected.append((url, h["DATE_OBS"]))
|
|
419
|
+
|
|
420
|
+
if selected:
|
|
421
|
+
urls_selected, dates = zip(*selected)
|
|
422
|
+
time_header = astropy.time.Time(list(dates), scale="utc")
|
|
423
|
+
where = (start <= time_header) & (time_header < stop)
|
|
424
|
+
result = [urls_selected[i] for i in time_header.argsort() if where[i]]
|
|
425
|
+
|
|
426
|
+
return na.ScalarArray(np.array(result, dtype=str), axes=axis_time)
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def _replace(
|
|
430
|
+
source: pathlib.Path,
|
|
431
|
+
destination: pathlib.Path,
|
|
432
|
+
num_retry: int = 5,
|
|
433
|
+
) -> None:
|
|
434
|
+
"""
|
|
435
|
+
Move a file onto another, trying again if the move is not permitted.
|
|
436
|
+
|
|
437
|
+
Windows does not move a file while another process has it open,
|
|
438
|
+
as a virus scanner may have a new file for a moment.
|
|
439
|
+
Each try waits twice as long as the one before it,
|
|
440
|
+
starting from :data:`_delay_replace`.
|
|
441
|
+
|
|
442
|
+
Parameters
|
|
443
|
+
----------
|
|
444
|
+
source
|
|
445
|
+
The file to move.
|
|
446
|
+
destination
|
|
447
|
+
Where to move the file, replacing any file there.
|
|
448
|
+
num_retry
|
|
449
|
+
The number of times to try to move the file.
|
|
450
|
+
"""
|
|
451
|
+
for i in range(num_retry - 1):
|
|
452
|
+
try:
|
|
453
|
+
os.replace(source, destination)
|
|
454
|
+
return
|
|
455
|
+
except PermissionError:
|
|
456
|
+
time.sleep(_delay_replace * 2**i)
|
|
457
|
+
|
|
458
|
+
os.replace(source, destination)
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def download(
|
|
462
|
+
urls: na.AbstractScalarArray,
|
|
463
|
+
directory: None | pathlib.Path = None,
|
|
464
|
+
overwrite: bool = False,
|
|
465
|
+
num_retry: int = 5,
|
|
466
|
+
) -> na.ScalarArray:
|
|
467
|
+
"""
|
|
468
|
+
Download the given URLs to a directory,
|
|
469
|
+
unless they have been downloaded already.
|
|
470
|
+
|
|
471
|
+
The files are placed under `directory` with the same paths as on the
|
|
472
|
+
server, and each one is checked against the length the server promised,
|
|
473
|
+
since the archive sometimes ends a download early,
|
|
474
|
+
and checked to be a FITS file.
|
|
475
|
+
|
|
476
|
+
Parameters
|
|
477
|
+
----------
|
|
478
|
+
urls
|
|
479
|
+
The URLs to download.
|
|
480
|
+
directory
|
|
481
|
+
The directory to place the downloaded files in.
|
|
482
|
+
If :obj:`None` (the default), :data:`hinode.directory_default` is used.
|
|
483
|
+
overwrite
|
|
484
|
+
Boolean flag controlling whether to download files which are already
|
|
485
|
+
in `directory`.
|
|
486
|
+
num_retry
|
|
487
|
+
The number of times to try to connect to the server.
|
|
488
|
+
|
|
489
|
+
Examples
|
|
490
|
+
--------
|
|
491
|
+
|
|
492
|
+
Download one of the Al_poly images captured while ESIS was observing
|
|
493
|
+
the Sun.
|
|
494
|
+
|
|
495
|
+
.. jupyter-execute::
|
|
496
|
+
|
|
497
|
+
import hinode
|
|
498
|
+
|
|
499
|
+
urls = hinode.xrt.urls(
|
|
500
|
+
time_start="2019-09-30T18:08:30",
|
|
501
|
+
time_stop="2019-09-30T18:08:40",
|
|
502
|
+
)
|
|
503
|
+
|
|
504
|
+
hinode.xrt.download(urls)
|
|
505
|
+
"""
|
|
506
|
+
if directory is None:
|
|
507
|
+
directory = hinode.directory_default
|
|
508
|
+
|
|
509
|
+
ndarray = np.asarray(urls.ndarray)
|
|
510
|
+
|
|
511
|
+
def get(url: str) -> str:
|
|
512
|
+
relative = "/".join(url.split("/")[3:])
|
|
513
|
+
if not relative or relative.endswith("/"):
|
|
514
|
+
raise ValueError(f"{url} is not the URL of a file.")
|
|
515
|
+
path = directory / relative
|
|
516
|
+
|
|
517
|
+
if overwrite or not path.is_file():
|
|
518
|
+
content = _get(url, num_retry).content
|
|
519
|
+
|
|
520
|
+
if not content.startswith(_signature_fits):
|
|
521
|
+
raise ValueError(
|
|
522
|
+
f"{url} is not a FITS file, it starts with {content[:40]!r}."
|
|
523
|
+
)
|
|
524
|
+
|
|
525
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
526
|
+
|
|
527
|
+
# Written next to its final place, under a name no other download
|
|
528
|
+
# uses, and then moved, so an interrupted download never looks
|
|
529
|
+
# finished.
|
|
530
|
+
# It is made with :func:`open` rather than :mod:`tempfile`,
|
|
531
|
+
# so that it has the permissions the umask gives a new file.
|
|
532
|
+
part = path.with_name(f"{path.name}.{uuid.uuid4().hex}.part")
|
|
533
|
+
try:
|
|
534
|
+
with open(part, "xb") as file:
|
|
535
|
+
file.write(content)
|
|
536
|
+
try:
|
|
537
|
+
_replace(part, path)
|
|
538
|
+
except PermissionError:
|
|
539
|
+
# Another process may have downloaded the file first and
|
|
540
|
+
# still have it open, in which case its copy is kept,
|
|
541
|
+
# unless a new copy was asked for.
|
|
542
|
+
if overwrite or not path.is_file():
|
|
543
|
+
raise
|
|
544
|
+
finally:
|
|
545
|
+
part.unlink(missing_ok=True)
|
|
546
|
+
|
|
547
|
+
return str(path)
|
|
548
|
+
|
|
549
|
+
# Each URL is downloaded once, even if it is given more than once.
|
|
550
|
+
unique = list(dict.fromkeys(str(url) for url in ndarray.flat))
|
|
551
|
+
|
|
552
|
+
with concurrent.futures.ThreadPoolExecutor(_num_workers) as executor:
|
|
553
|
+
paths = dict(zip(unique, executor.map(get, unique)))
|
|
554
|
+
|
|
555
|
+
paths = [paths[str(url)] for url in ndarray.flat]
|
|
556
|
+
paths = np.array(paths, dtype=str).reshape(ndarray.shape)
|
|
557
|
+
|
|
558
|
+
return na.ScalarArray(paths, axes=urls.axes)
|