hinode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hinode/__init__.py ADDED
@@ -0,0 +1,29 @@
1
+ """
2
+ Download and analyze observations from the Hinode satellite.
3
+ """
4
+
5
+ import pathlib
6
+ import joblib
7
+
8
+ __all__ = [
9
+ "directory_default",
10
+ "memory",
11
+ "xrt",
12
+ ]
13
+
14
+ directory_default = pathlib.Path.home() / ".hinode/cache"
15
+ """The default directory for downloaded files."""
16
+
17
+ memory = joblib.Memory(location=directory_default / "joblib", verbose=0)
18
+ """
19
+ A representation of the cache which stores intermediate results,
20
+ such as the headers of the files in the archives.
21
+
22
+ The cache has a directory of its own, apart from the downloaded files,
23
+ so that :meth:`joblib.Memory.clear` does not delete them.
24
+ Assign another :class:`joblib.Memory` to this attribute to move the cache,
25
+ or one whose location is :obj:`None` to stop caching.
26
+ """
27
+
28
+ # Imported after the attributes above, which the subpackages use
29
+ from . import xrt # noqa: E402
hinode/py.typed ADDED
File without changes
hinode/xrt/__init__.py ADDED
@@ -0,0 +1,17 @@
1
+ """
2
+ Download and analyze images from the X-Ray Telescope (XRT).
3
+ """
4
+
5
+ from ._data import (
6
+ urls,
7
+ download,
8
+ )
9
+ from ._filtergrams import Filtergram
10
+ from ._xrt import open
11
+
12
+ __all__ = [
13
+ "urls",
14
+ "download",
15
+ "Filtergram",
16
+ "open",
17
+ ]
hinode/xrt/_data.py ADDED
@@ -0,0 +1,558 @@
1
+ import os
2
+ import re
3
+ import time
4
+ import uuid
5
+ import typing
6
+ import pathlib
7
+ import datetime
8
+ import warnings
9
+ import functools
10
+ import concurrent.futures
11
+ import requests
12
+ import joblib
13
+ import numpy as np
14
+ import astropy.units as u
15
+ import astropy.time
16
+ import astropy.io.fits
17
+ import named_arrays as na
18
+ import hinode
19
+
20
+ __all__ = [
21
+ "urls",
22
+ "download",
23
+ ]
24
+
25
+ _url_level_1 = "https://umbra.nascom.nasa.gov/hinode/xrt/level1/"
26
+ """
27
+ The archive of Level 1 XRT images at the Solar Data Analysis Center,
28
+ with one directory for each hour.
29
+ """
30
+
31
+ _pattern_file = re.compile(r'href="(L1_XRT(\d{8})_(\d{6}\.\d)\.fits)"')
32
+ """
33
+ The name of a Level 1 file in a directory listing of the archive,
34
+ which holds the date and the start time of the exposure,
35
+ as in ``L1_XRT20190930_180837.5.fits``.
36
+ """
37
+
38
+ _size_block = 2880
39
+ """The number of bytes in a block of a FITS file."""
40
+
41
+ _size_card = 80
42
+ """The number of bytes in a card of a FITS header."""
43
+
44
+ _num_workers = 8
45
+ """
46
+ The number of requests made to the archive at once,
47
+ since most of the time of each one is spent waiting for the server.
48
+ """
49
+
50
+ _delay_retry = 1
51
+ """
52
+ The number of seconds to wait before trying a request again,
53
+ which is doubled before each further try.
54
+ """
55
+
56
+ _status_retry = (408, 429)
57
+ """
58
+ The statuses below 500 with which a server refuses a request that may
59
+ succeed if it is tried again.
60
+ """
61
+
62
+ _errors_retry = (
63
+ requests.exceptions.ConnectionError,
64
+ requests.exceptions.Timeout,
65
+ requests.exceptions.ChunkedEncodingError,
66
+ )
67
+ """
68
+ The errors of a request which may not happen if it is tried again.
69
+
70
+ The archive sometimes ends a response before the length it promised,
71
+ which :mod:`urllib3` reports as a
72
+ :class:`~requests.exceptions.ChunkedEncodingError`.
73
+ """
74
+
75
+ _errors_header = (
76
+ FileNotFoundError,
77
+ requests.exceptions.HTTPError,
78
+ ValueError,
79
+ )
80
+ """
81
+ The errors of reading the header of a file in the archive
82
+ which are the fault of that file alone,
83
+ such as the file having been removed since the directory was listed,
84
+ or not being a FITS file.
85
+ """
86
+
87
+ _delay_replace = 0.1
88
+ """
89
+ The number of seconds to wait before trying to move a downloaded file
90
+ into place again,
91
+ which is doubled before each further try.
92
+ """
93
+
94
+ _signature_fits = b"SIMPLE ="
95
+ """The first bytes of every FITS file."""
96
+
97
+
98
+ def _get(
99
+ url: str,
100
+ num_retry: int = 5,
101
+ headers: None | dict[str, str] = None,
102
+ ) -> requests.Response:
103
+ """
104
+ Get a URL, trying again if the connection fails, the server is busy or
105
+ has an error, or the response is cut short.
106
+
107
+ Each try waits twice as long as the one before it,
108
+ starting from :data:`_delay_retry`.
109
+
110
+ Parameters
111
+ ----------
112
+ url
113
+ The URL to get.
114
+ num_retry
115
+ The number of times to try to connect to the server.
116
+ headers
117
+ Additional HTTP headers to send with the request.
118
+
119
+ Raises
120
+ ------
121
+ FileNotFoundError
122
+ If the server does not have the URL.
123
+ requests.exceptions.HTTPError
124
+ If the server refuses the request for a reason which trying again
125
+ cannot change, such as a 403 status.
126
+ requests.exceptions.RequestException
127
+ If the request cannot be made at all, such as for a malformed URL.
128
+ ConnectionError
129
+ If no attempt succeeded.
130
+ """
131
+ error = None
132
+ for i in range(num_retry):
133
+ if i > 0:
134
+ time.sleep(_delay_retry * 2 ** (i - 1))
135
+ try:
136
+ response = requests.get(url, headers=headers, timeout=60)
137
+ response.raise_for_status()
138
+ except requests.exceptions.HTTPError as e:
139
+ status = None if e.response is None else e.response.status_code
140
+ if status == 404:
141
+ raise FileNotFoundError(url) from e
142
+ if status is not None and status < 500 and status not in _status_retry:
143
+ raise
144
+ error = e
145
+ except _errors_retry as e:
146
+ error = e
147
+ else:
148
+ return response
149
+
150
+ raise ConnectionError(f"Could not get {url} in {num_retry} tries.") from error
151
+
152
+
153
+ def _hours(
154
+ start: str | astropy.time.Time,
155
+ stop: str | astropy.time.Time,
156
+ ) -> list[datetime.datetime]:
157
+ """
158
+ The hours of the directories of the archive which hold the files that
159
+ began during a time range.
160
+
161
+ The directories are named in UTC,
162
+ and the hours are counted as :class:`datetime.datetime` objects,
163
+ which, unlike UTC times, have no leap seconds,
164
+ so an hour with a leap second at its end is not counted twice.
165
+
166
+ Parameters
167
+ ----------
168
+ start
169
+ The start of the time range.
170
+ stop
171
+ The end of the time range, which is not included,
172
+ so a range which stops on the hour does not include that hour.
173
+ """
174
+ format_hour = "%Y-%m-%dT%H"
175
+
176
+ def hour_utc(time: astropy.time.Time) -> datetime.datetime:
177
+ text = str(time.strftime(format_hour))
178
+ return datetime.datetime.strptime(text, format_hour)
179
+
180
+ start = astropy.time.Time(start, scale="utc")
181
+ stop = astropy.time.Time(stop, scale="utc")
182
+
183
+ hour = hour_utc(start)
184
+ last = hour_utc(stop)
185
+ if astropy.time.Time(last.isoformat(), scale="utc") == stop:
186
+ last -= datetime.timedelta(hours=1)
187
+
188
+ result = []
189
+ while hour <= last:
190
+ result.append(hour)
191
+ hour += datetime.timedelta(hours=1)
192
+
193
+ return result
194
+
195
+
196
+ def _files(
197
+ hour: datetime.datetime,
198
+ num_retry: int = 5,
199
+ ) -> list[tuple[str, str]]:
200
+ """
201
+ The URL and the start time, from the name, of every Level 1 file in the
202
+ directory of the archive which holds a given hour.
203
+
204
+ The start time is in UTC, in the ISOT format.
205
+
206
+ Parameters
207
+ ----------
208
+ hour
209
+ The hour, in UTC.
210
+ num_retry
211
+ The number of times to try to connect to the server.
212
+ """
213
+ url = f"{_url_level_1}{hour.strftime('%Y/%m/%d/H%H00')}/"
214
+
215
+ try:
216
+ response = _get(url, num_retry)
217
+ except FileNotFoundError:
218
+ return []
219
+
220
+ result = []
221
+ for name, date, clock in dict.fromkeys(_pattern_file.findall(response.text)):
222
+ isot = f"{date[:4]}-{date[4:6]}-{date[6:]}T{clock[:2]}:{clock[2:4]}:{clock[4:]}"
223
+ result.append((url + name, isot))
224
+
225
+ return result
226
+
227
+
228
+ def _header_string(
229
+ url: str,
230
+ num_retry: int = 5,
231
+ ) -> str:
232
+ """
233
+ The text of the primary header of a FITS file on a server,
234
+ read from the start of the file without downloading its data.
235
+
236
+ Parameters
237
+ ----------
238
+ url
239
+ The URL of the FITS file.
240
+ num_retry
241
+ The number of times to try to connect to the server.
242
+ """
243
+ num_blocks = 8
244
+
245
+ while True:
246
+ size = num_blocks * _size_block
247
+ response = _get(
248
+ url=url,
249
+ num_retry=num_retry,
250
+ headers={"Range": f"bytes=0-{size - 1}"},
251
+ )
252
+ content = response.content
253
+
254
+ for start in range(0, len(content) - _size_card + 1, _size_card):
255
+ card = content[start : start + _size_card]
256
+ if card.rstrip() == b"END":
257
+ return content[: start + _size_card].decode("ascii")
258
+
259
+ if len(content) < size:
260
+ raise ValueError(f"The header of {url} has no END card.")
261
+
262
+ num_blocks *= 2
263
+
264
+
265
+ @functools.cache
266
+ def _header_string_cached(
267
+ memory: joblib.Memory,
268
+ ) -> typing.Callable[[str, int], str]:
269
+ """
270
+ :func:`_header_string`, cached in the given cache,
271
+ since the archive does not change.
272
+
273
+ The cached function is made once for each cache,
274
+ so that the cache records the code of the function it caches only once,
275
+ rather than from every thread at once.
276
+
277
+ Parameters
278
+ ----------
279
+ memory
280
+ The cache to store the headers in.
281
+ """
282
+ result = memory.cache(_header_string, ignore=["num_retry"])
283
+ return typing.cast(typing.Callable[[str, int], str], result)
284
+
285
+
286
+ def _header(
287
+ url: str,
288
+ num_retry: int = 5,
289
+ ) -> astropy.io.fits.Header:
290
+ """
291
+ The primary header of a FITS file on a server,
292
+ cached in :data:`hinode.memory`, since the archive does not change.
293
+
294
+ Parameters
295
+ ----------
296
+ url
297
+ The URL of the FITS file.
298
+ num_retry
299
+ The number of times to try to connect to the server.
300
+ """
301
+ text = _header_string_cached(hinode.memory)(url, num_retry)
302
+ return astropy.io.fits.Header.fromstring(text)
303
+
304
+
305
+ def _filter(header: astropy.io.fits.Header) -> str:
306
+ """
307
+ The filters an image was taken through,
308
+ named as in :func:`urls`.
309
+
310
+ Parameters
311
+ ----------
312
+ header
313
+ The primary header of an XRT file.
314
+ """
315
+ names = [str(header[key]).strip() for key in ["EC_FW1_", "EC_FW2_"]]
316
+ names = [name for name in names if name != "Open"]
317
+ if not names:
318
+ return "Open"
319
+ return "/".join(names)
320
+
321
+
322
+ def urls(
323
+ time_start: str | astropy.time.Time,
324
+ time_stop: str | astropy.time.Time,
325
+ filter: str = "Al_poly",
326
+ axis_time: str = "time",
327
+ num_retry: int = 5,
328
+ ) -> na.ScalarArray:
329
+ """
330
+ Find the URLs of the Level 1 XRT images in the archive of the Solar Data
331
+ Analysis Center which began during a given time range and were taken
332
+ through a given filter.
333
+
334
+ The archive does not say which filter each image was taken through,
335
+ so the start of each file is downloaded to read its header,
336
+ and the headers are cached in :data:`hinode.memory`.
337
+ Only the images whose type, ``EC_IMTY_``, is ``normal`` are found,
338
+ which leaves out the dark frames.
339
+ A file whose header cannot be read,
340
+ such as one removed from the archive after its directory was listed,
341
+ is left out with a warning.
342
+
343
+ Parameters
344
+ ----------
345
+ time_start
346
+ The earliest start time of the images.
347
+ time_stop
348
+ The time before which an image must begin.
349
+ filter
350
+ The filters in the two filter wheels of XRT,
351
+ named as in the ``EC_FW1_`` and ``EC_FW2_`` header keywords,
352
+ with the open positions left out and the filters joined by a slash.
353
+ For example ``"Al_poly"``, ``"Ti_poly"``, ``"Al_poly/Ti_poly"``,
354
+ ``"Gband"``, or ``"Open"`` if both wheels are open.
355
+ axis_time
356
+ The logical axis corresponding to changes in time.
357
+ num_retry
358
+ The number of times to try to connect to the server.
359
+
360
+ Examples
361
+ --------
362
+
363
+ Find the Al_poly images captured while the EUV Snapshot Imaging
364
+ Spectrograph (ESIS) was observing the Sun on 2019 September 30.
365
+
366
+ .. jupyter-execute::
367
+
368
+ import hinode
369
+
370
+ hinode.xrt.urls(
371
+ time_start="2019-09-30T18:06:11",
372
+ time_stop="2019-09-30T18:11:01",
373
+ filter="Al_poly",
374
+ )
375
+ """
376
+ start = astropy.time.Time(time_start)
377
+ stop = astropy.time.Time(time_stop)
378
+
379
+ def header(url: str) -> astropy.io.fits.Header | Exception:
380
+ try:
381
+ return _header(url, num_retry)
382
+ except _errors_header as e:
383
+ return e
384
+
385
+ result: list[str] = []
386
+
387
+ with concurrent.futures.ThreadPoolExecutor(_num_workers) as executor:
388
+
389
+ listings = executor.map(
390
+ lambda hour: _files(hour, num_retry),
391
+ _hours(time_start, time_stop),
392
+ )
393
+ files = dict(file for listing in listings for file in listing)
394
+
395
+ # The name holds the start time cut to a tenth of a second,
396
+ # so it can be a little earlier than the start time in the header.
397
+ candidates = []
398
+ if files:
399
+ time_name = astropy.time.Time(list(files.values()), scale="utc")
400
+ margin = astropy.time.TimeDelta(1 * u.s)
401
+ where = (start - margin <= time_name) & (time_name < stop)
402
+ candidates = [url for url, w in zip(files, where) if w]
403
+
404
+ # The first header is read before the others, so that the cache
405
+ # records the code of the function it caches from one thread,
406
+ # rather than from every thread at once.
407
+ headers = [header(url) for url in candidates[:1]]
408
+ headers += executor.map(header, candidates[1:])
409
+
410
+ selected = []
411
+ for url, h in zip(candidates, headers):
412
+ if isinstance(h, Exception):
413
+ warnings.warn(
414
+ f"Left out {url}, since its header could not be read: {h!r}",
415
+ stacklevel=2,
416
+ )
417
+ elif _filter(h) == filter and h.get("EC_IMTY_") == "normal":
418
+ selected.append((url, h["DATE_OBS"]))
419
+
420
+ if selected:
421
+ urls_selected, dates = zip(*selected)
422
+ time_header = astropy.time.Time(list(dates), scale="utc")
423
+ where = (start <= time_header) & (time_header < stop)
424
+ result = [urls_selected[i] for i in time_header.argsort() if where[i]]
425
+
426
+ return na.ScalarArray(np.array(result, dtype=str), axes=axis_time)
427
+
428
+
429
+ def _replace(
430
+ source: pathlib.Path,
431
+ destination: pathlib.Path,
432
+ num_retry: int = 5,
433
+ ) -> None:
434
+ """
435
+ Move a file onto another, trying again if the move is not permitted.
436
+
437
+ Windows does not move a file while another process has it open,
438
+ as a virus scanner may have a new file for a moment.
439
+ Each try waits twice as long as the one before it,
440
+ starting from :data:`_delay_replace`.
441
+
442
+ Parameters
443
+ ----------
444
+ source
445
+ The file to move.
446
+ destination
447
+ Where to move the file, replacing any file there.
448
+ num_retry
449
+ The number of times to try to move the file.
450
+ """
451
+ for i in range(num_retry - 1):
452
+ try:
453
+ os.replace(source, destination)
454
+ return
455
+ except PermissionError:
456
+ time.sleep(_delay_replace * 2**i)
457
+
458
+ os.replace(source, destination)
459
+
460
+
461
+ def download(
462
+ urls: na.AbstractScalarArray,
463
+ directory: None | pathlib.Path = None,
464
+ overwrite: bool = False,
465
+ num_retry: int = 5,
466
+ ) -> na.ScalarArray:
467
+ """
468
+ Download the given URLs to a directory,
469
+ unless they have been downloaded already.
470
+
471
+ The files are placed under `directory` with the same paths as on the
472
+ server, and each one is checked against the length the server promised,
473
+ since the archive sometimes ends a download early,
474
+ and checked to be a FITS file.
475
+
476
+ Parameters
477
+ ----------
478
+ urls
479
+ The URLs to download.
480
+ directory
481
+ The directory to place the downloaded files in.
482
+ If :obj:`None` (the default), :data:`hinode.directory_default` is used.
483
+ overwrite
484
+ Boolean flag controlling whether to download files which are already
485
+ in `directory`.
486
+ num_retry
487
+ The number of times to try to connect to the server.
488
+
489
+ Examples
490
+ --------
491
+
492
+ Download one of the Al_poly images captured while ESIS was observing
493
+ the Sun.
494
+
495
+ .. jupyter-execute::
496
+
497
+ import hinode
498
+
499
+ urls = hinode.xrt.urls(
500
+ time_start="2019-09-30T18:08:30",
501
+ time_stop="2019-09-30T18:08:40",
502
+ )
503
+
504
+ hinode.xrt.download(urls)
505
+ """
506
+ if directory is None:
507
+ directory = hinode.directory_default
508
+
509
+ ndarray = np.asarray(urls.ndarray)
510
+
511
+ def get(url: str) -> str:
512
+ relative = "/".join(url.split("/")[3:])
513
+ if not relative or relative.endswith("/"):
514
+ raise ValueError(f"{url} is not the URL of a file.")
515
+ path = directory / relative
516
+
517
+ if overwrite or not path.is_file():
518
+ content = _get(url, num_retry).content
519
+
520
+ if not content.startswith(_signature_fits):
521
+ raise ValueError(
522
+ f"{url} is not a FITS file, it starts with {content[:40]!r}."
523
+ )
524
+
525
+ path.parent.mkdir(parents=True, exist_ok=True)
526
+
527
+ # Written next to its final place, under a name no other download
528
+ # uses, and then moved, so an interrupted download never looks
529
+ # finished.
530
+ # It is made with :func:`open` rather than :mod:`tempfile`,
531
+ # so that it has the permissions the umask gives a new file.
532
+ part = path.with_name(f"{path.name}.{uuid.uuid4().hex}.part")
533
+ try:
534
+ with open(part, "xb") as file:
535
+ file.write(content)
536
+ try:
537
+ _replace(part, path)
538
+ except PermissionError:
539
+ # Another process may have downloaded the file first and
540
+ # still have it open, in which case its copy is kept,
541
+ # unless a new copy was asked for.
542
+ if overwrite or not path.is_file():
543
+ raise
544
+ finally:
545
+ part.unlink(missing_ok=True)
546
+
547
+ return str(path)
548
+
549
+ # Each URL is downloaded once, even if it is given more than once.
550
+ unique = list(dict.fromkeys(str(url) for url in ndarray.flat))
551
+
552
+ with concurrent.futures.ThreadPoolExecutor(_num_workers) as executor:
553
+ paths = dict(zip(unique, executor.map(get, unique)))
554
+
555
+ paths = [paths[str(url)] for url in ndarray.flat]
556
+ paths = np.array(paths, dtype=str).reshape(ndarray.shape)
557
+
558
+ return na.ScalarArray(paths, axes=urls.axes)