dwdopen 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dwdopen/__init__.py +61 -0
- dwdopen/_fileserver/__init__.py +4 -0
- dwdopen/_fileserver/download.py +265 -0
- dwdopen/_fileserver/http.py +60 -0
- dwdopen/_fileserver/listing.py +108 -0
- dwdopen/_fileserver/naming.py +70 -0
- dwdopen/_fileserver/paths.py +48 -0
- dwdopen/_fileserver/traversal.py +404 -0
- dwdopen/client.py +164 -0
- dwdopen/exceptions.py +147 -0
- dwdopen/nwp/__init__.py +22 -0
- dwdopen/nwp/catalogue.py +125 -0
- dwdopen/nwp/durations.py +165 -0
- dwdopen/nwp/model.py +244 -0
- dwdopen/nwp/query.py +395 -0
- dwdopen/nwp/request.py +492 -0
- dwdopen/nwp/run.py +68 -0
- dwdopen/nwp/selectors.py +176 -0
- dwdopen/py.typed +0 -0
- dwdopen-0.2.1.dist-info/METADATA +226 -0
- dwdopen-0.2.1.dist-info/RECORD +23 -0
- dwdopen-0.2.1.dist-info/WHEEL +4 -0
- dwdopen-0.2.1.dist-info/licenses/LICENSE +21 -0
dwdopen/__init__.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Python access to DWD Open Data."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
6
|
+
|
|
7
|
+
from dwdopen.client import DWD, NWP
|
|
8
|
+
from dwdopen.exceptions import (
|
|
9
|
+
AmbiguousSelectionError,
|
|
10
|
+
CatalogueError,
|
|
11
|
+
CatalogueUnavailableError,
|
|
12
|
+
DownloadError,
|
|
13
|
+
DWDOpenError,
|
|
14
|
+
IncompleteRunError,
|
|
15
|
+
InvalidSelectorError,
|
|
16
|
+
MissingAssetError,
|
|
17
|
+
NoMatchingRunError,
|
|
18
|
+
ResolutionError,
|
|
19
|
+
RunExpiredError,
|
|
20
|
+
SelectionError,
|
|
21
|
+
UnknownModelError,
|
|
22
|
+
UnknownParameterError,
|
|
23
|
+
)
|
|
24
|
+
from dwdopen.nwp.durations import hours, minutes
|
|
25
|
+
from dwdopen.nwp.model import Model, ParameterInfo
|
|
26
|
+
from dwdopen.nwp.request import DownloadResult
|
|
27
|
+
from dwdopen.nwp.run import Run
|
|
28
|
+
from dwdopen.nwp.selectors import Between, Every, LevelType
|
|
29
|
+
|
|
30
|
+
__all__ = [
|
|
31
|
+
"DWD",
|
|
32
|
+
"NWP",
|
|
33
|
+
"AmbiguousSelectionError",
|
|
34
|
+
"Between",
|
|
35
|
+
"CatalogueError",
|
|
36
|
+
"CatalogueUnavailableError",
|
|
37
|
+
"DWDOpenError",
|
|
38
|
+
"DownloadError",
|
|
39
|
+
"DownloadResult",
|
|
40
|
+
"Every",
|
|
41
|
+
"IncompleteRunError",
|
|
42
|
+
"InvalidSelectorError",
|
|
43
|
+
"LevelType",
|
|
44
|
+
"MissingAssetError",
|
|
45
|
+
"Model",
|
|
46
|
+
"NoMatchingRunError",
|
|
47
|
+
"ParameterInfo",
|
|
48
|
+
"ResolutionError",
|
|
49
|
+
"Run",
|
|
50
|
+
"RunExpiredError",
|
|
51
|
+
"SelectionError",
|
|
52
|
+
"UnknownModelError",
|
|
53
|
+
"UnknownParameterError",
|
|
54
|
+
"hours",
|
|
55
|
+
"minutes",
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
try:
|
|
59
|
+
__version__ = version("dwdopen")
|
|
60
|
+
except PackageNotFoundError: # running from a source tree, not installed
|
|
61
|
+
__version__ = "0.0.0.dev0"
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
"""Fetching assets over HTTP, in parallel, onto disk."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
import secrets
|
|
8
|
+
import shutil
|
|
9
|
+
import time
|
|
10
|
+
from collections.abc import Sequence
|
|
11
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from dwdopen._fileserver.http import HttpClient
|
|
15
|
+
from dwdopen._fileserver.naming import already_complete, local_name, temp_name
|
|
16
|
+
from dwdopen.exceptions import DownloadError
|
|
17
|
+
from dwdopen.nwp.request import Asset, CombineMode, Fetched, human_size
|
|
18
|
+
from dwdopen.nwp.selectors import LevelType
|
|
19
|
+
|
|
20
|
+
__all__ = ["HttpDownloader"]
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger("dwdopen")
|
|
23
|
+
|
|
24
|
+
LARGE_DOWNLOAD = 2 * 1024**3
|
|
25
|
+
"""Log when a user plans to download large amounts of data already in the
|
|
26
|
+
planning stage.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
RETRY_STATUS = {404, 408, 425, 429, 500, 502, 503, 504}
|
|
30
|
+
"""HTTP error statuses worth a retry. 404 is handled separately."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class HttpDownloader:
|
|
34
|
+
"""Fetches assets concurrently and writes them atomically.
|
|
35
|
+
Takes an HttpClient so transport configuration stays in one place.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
def __init__(
|
|
39
|
+
self,
|
|
40
|
+
http: HttpClient,
|
|
41
|
+
*,
|
|
42
|
+
max_workers: int = 16,
|
|
43
|
+
max_attempts: int = 4,
|
|
44
|
+
backoff: float = 0.5
|
|
45
|
+
) -> None:
|
|
46
|
+
"""
|
|
47
|
+
max_workers
|
|
48
|
+
Concurrent downloads. Bounded by the client's connection pool.
|
|
49
|
+
max_attempts
|
|
50
|
+
Attempts per asset for a failure.
|
|
51
|
+
backoff
|
|
52
|
+
Seconds before the second attempt; doubles each further attempt.
|
|
53
|
+
"""
|
|
54
|
+
self._http = http
|
|
55
|
+
self._max_workers = max_workers
|
|
56
|
+
self._max_attempts = max_attempts
|
|
57
|
+
self._backoff = backoff
|
|
58
|
+
|
|
59
|
+
def fetch(
|
|
60
|
+
self,
|
|
61
|
+
assets: Sequence[Asset],
|
|
62
|
+
destination: Path,
|
|
63
|
+
*,
|
|
64
|
+
combine: CombineMode = "all",
|
|
65
|
+
temp_dir: Path | None = None,
|
|
66
|
+
) -> Fetched:
|
|
67
|
+
"""Fetches the given assets to the destination path. Calls either fetch
|
|
68
|
+
separately or combined depending on the combine flag.
|
|
69
|
+
"""
|
|
70
|
+
destination = Path(destination)
|
|
71
|
+
if not assets:
|
|
72
|
+
logger.info("nothing to download")
|
|
73
|
+
return Fetched(files=(), bytes_downloaded=0)
|
|
74
|
+
|
|
75
|
+
self._announce(assets, destination, combine)
|
|
76
|
+
if combine == "none":
|
|
77
|
+
fetched = self._fetch_separately(assets, destination, temp_dir)
|
|
78
|
+
else:
|
|
79
|
+
fetched = self._fetch_combined(assets, destination, temp_dir)
|
|
80
|
+
self._report(fetched, combine)
|
|
81
|
+
return fetched
|
|
82
|
+
|
|
83
|
+
def _fetch_separately(
|
|
84
|
+
self, assets: Sequence[Asset], destination: Path, temp_dir: Path | None
|
|
85
|
+
) -> Fetched:
|
|
86
|
+
"""One file per asset, named after its keys, inside one directory."""
|
|
87
|
+
destination.mkdir(parents=True, exist_ok=True)
|
|
88
|
+
staging = destination if temp_dir is None else temp_dir
|
|
89
|
+
|
|
90
|
+
target_paths = [destination / local_name(asset.keys) for asset in assets]
|
|
91
|
+
sizes = self._fetch_all(assets, target_paths, staging)
|
|
92
|
+
# A zero size asset means the file was already there, thus it was skipped.
|
|
93
|
+
return Fetched(
|
|
94
|
+
files=tuple(target_paths),
|
|
95
|
+
bytes_downloaded=sum(sizes),
|
|
96
|
+
skipped=sum(1 for size in sizes if size == 0),
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
def _fetch_combined(
|
|
100
|
+
self, assets: Sequence[Asset], destination: Path, temp_dir: Path | None
|
|
101
|
+
) -> Fetched:
|
|
102
|
+
"""Every message concatenated into one GRIB2 file.
|
|
103
|
+
GRIB2 messages are self-delimiting, each carrying its own length and
|
|
104
|
+
terminator, so joining the bytes of several files is a valid file.
|
|
105
|
+
The parts are downloaded into temp_dir if given, before concatenating.
|
|
106
|
+
"""
|
|
107
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
108
|
+
parent = destination.parent if temp_dir is None else temp_dir
|
|
109
|
+
staging_dir = self._private_directory(parent)
|
|
110
|
+
|
|
111
|
+
try:
|
|
112
|
+
parts = [staging_dir / f"{index:08d}.part" for index in range(len(assets))]
|
|
113
|
+
sizes = self._fetch_all(assets, parts, staging_dir)
|
|
114
|
+
|
|
115
|
+
# Merge into the staging directory first, never straight into
|
|
116
|
+
# the destination. Writing there directly would leave a partial
|
|
117
|
+
# file under the final name if the process died mid-merge.
|
|
118
|
+
merged = staging_dir / "merged"
|
|
119
|
+
with merged.open("wb") as out:
|
|
120
|
+
for part in parts:
|
|
121
|
+
with part.open("rb") as chunk:
|
|
122
|
+
shutil.copyfileobj(chunk, out)
|
|
123
|
+
|
|
124
|
+
# os.replace is atomic within a filesystem: the destination holds
|
|
125
|
+
# either the old file or the whole new one, never half.
|
|
126
|
+
os.replace(merged, destination)
|
|
127
|
+
finally:
|
|
128
|
+
# Remove temp files.
|
|
129
|
+
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
130
|
+
|
|
131
|
+
return Fetched(files=(destination,), bytes_downloaded=sum(sizes))
|
|
132
|
+
|
|
133
|
+
def _fetch_all(
|
|
134
|
+
self, assets: Sequence[Asset], targets: Sequence[Path], staging: Path
|
|
135
|
+
) -> list[int]:
|
|
136
|
+
"""Fetch every asset into its target, in parallel."""
|
|
137
|
+
if len(assets) != len(targets):
|
|
138
|
+
raise ValueError(
|
|
139
|
+
f"{len(assets)} assets but {len(targets)} targets"
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
# Num workers: As specified, or number of assets to get.
|
|
143
|
+
workers = max(1, min(self._max_workers, len(assets)))
|
|
144
|
+
|
|
145
|
+
def fetch(asset: Asset, target: Path) -> int:
|
|
146
|
+
return self._fetch_one(asset, target, staging)
|
|
147
|
+
|
|
148
|
+
with ThreadPoolExecutor(max_workers=workers) as pool:
|
|
149
|
+
# Returns list of download sizes for the assets.
|
|
150
|
+
return list(pool.map(fetch, assets, targets))
|
|
151
|
+
|
|
152
|
+
def _fetch_one(self, asset: Asset, target: Path, staging: Path) -> int:
|
|
153
|
+
"""Fetch one asset, write it to the specified target.
|
|
154
|
+
Returns the bytes transferred, so zero means the file was already there
|
|
155
|
+
and nothing had to be fetched.
|
|
156
|
+
"""
|
|
157
|
+
if already_complete(target, asset.size):
|
|
158
|
+
logger.debug("skipping %s, already downloaded", target)
|
|
159
|
+
return 0
|
|
160
|
+
|
|
161
|
+
# Write into a partial/staging directory instead of the target path.
|
|
162
|
+
# This makes sure two processes downloading the same asset to not have
|
|
163
|
+
# a race condition on the file access and write. After downloading,
|
|
164
|
+
# os.replace() makes an atomic move into the target file.
|
|
165
|
+
payload = self._get_with_retries(asset)
|
|
166
|
+
partial = staging / temp_name(target.name)
|
|
167
|
+
partial.write_bytes(payload)
|
|
168
|
+
os.replace(partial, target)
|
|
169
|
+
logger.debug("wrote %s (%d bytes)", target, len(payload))
|
|
170
|
+
return len(payload)
|
|
171
|
+
|
|
172
|
+
def _get_with_retries(self, asset: Asset) -> bytes:
|
|
173
|
+
"""Fetch one path, retrying failures with exponential backoff."""
|
|
174
|
+
attempt = 1
|
|
175
|
+
while True:
|
|
176
|
+
try:
|
|
177
|
+
return self._http.get_asset(asset.path)
|
|
178
|
+
except DownloadError as exc:
|
|
179
|
+
# Get limit of retries for the error code.
|
|
180
|
+
limit = 1
|
|
181
|
+
if exc.status is None or exc.status in RETRY_STATUS:
|
|
182
|
+
# Set limit based on error code, if retrying makes actual sense.
|
|
183
|
+
limit = self._max_attempts
|
|
184
|
+
if attempt >= limit:
|
|
185
|
+
raise # Abort after N retries
|
|
186
|
+
delay = self._backoff * 2 ** (attempt - 1)
|
|
187
|
+
logger.warning(
|
|
188
|
+
"attempt %d/%d for %s failed (%s), retrying in %.1fs",
|
|
189
|
+
attempt, limit, asset.path, exc, delay,
|
|
190
|
+
)
|
|
191
|
+
time.sleep(delay)
|
|
192
|
+
attempt += 1
|
|
193
|
+
|
|
194
|
+
# Private helpers
|
|
195
|
+
|
|
196
|
+
def _private_directory(self, parent: Path) -> Path:
|
|
197
|
+
"""A staging directory private to this process."""
|
|
198
|
+
parent.mkdir(parents=True, exist_ok=True)
|
|
199
|
+
staging = parent / f".dwdopen-{temp_name('merge')}-{secrets.token_hex(4)}"
|
|
200
|
+
staging.mkdir()
|
|
201
|
+
return staging
|
|
202
|
+
|
|
203
|
+
def _announce(
|
|
204
|
+
self, assets: Sequence[Asset], destination: Path, combine: CombineMode
|
|
205
|
+
) -> None:
|
|
206
|
+
"""Announcing the download before it starts (e.g. inform user about size)."""
|
|
207
|
+
asset_sizes = [asset.size for asset in assets if asset.size is not None]
|
|
208
|
+
total = sum(asset_sizes) if len(asset_sizes) == len(assets) else None
|
|
209
|
+
size = "unknown size" if total is None else human_size(total)
|
|
210
|
+
|
|
211
|
+
message = "downloading %d assets (%s) to %s, combine=%s"
|
|
212
|
+
args = (len(assets), size, destination, combine)
|
|
213
|
+
|
|
214
|
+
if total is not None and total >= LARGE_DOWNLOAD:
|
|
215
|
+
# Inform user about a large request.
|
|
216
|
+
logger.info(message + " - this is a large request.", *args)
|
|
217
|
+
else:
|
|
218
|
+
logger.info(message, *args)
|
|
219
|
+
|
|
220
|
+
# Also warn the user if there are multiple level types in the request and
|
|
221
|
+
# combine="all" is set: Multiple level types in a single GRIB file is valid
|
|
222
|
+
# but makes many problems with software reading GRIB files.
|
|
223
|
+
if combine == "all":
|
|
224
|
+
self._warn_about_mixed_level_types(assets)
|
|
225
|
+
|
|
226
|
+
@staticmethod
|
|
227
|
+
def _report(fetched: Fetched, combine: CombineMode) -> None:
|
|
228
|
+
"""Report the finished download.
|
|
229
|
+
"""
|
|
230
|
+
size = human_size(fetched.bytes_downloaded)
|
|
231
|
+
if fetched.skipped:
|
|
232
|
+
size += f", {fetched.skipped} already downloaded"
|
|
233
|
+
if combine == "none":
|
|
234
|
+
logger.info(
|
|
235
|
+
"saved %d files (%s) to %s",
|
|
236
|
+
len(fetched.files), size, fetched.files[0].parent,
|
|
237
|
+
)
|
|
238
|
+
else:
|
|
239
|
+
logger.info("saved %s (%s)", fetched.files[0], size)
|
|
240
|
+
|
|
241
|
+
@staticmethod
|
|
242
|
+
def _warn_about_mixed_level_types(assets: Sequence[Asset]) -> None:
|
|
243
|
+
"""Warn when one file will hold several vertical coordinate types.
|
|
244
|
+
Concatenating them is legal GRIB2 and stays allowed, but readers do
|
|
245
|
+
struggle with it. CDO in particular wants one vertical axis per file.
|
|
246
|
+
"""
|
|
247
|
+
kinds = {
|
|
248
|
+
asset.level_type.code
|
|
249
|
+
for asset in assets
|
|
250
|
+
if asset.level_type is not None
|
|
251
|
+
}
|
|
252
|
+
# A 2-D field alongside a 3-D one is also problematic.
|
|
253
|
+
if any(asset.level_type is None for asset in assets):
|
|
254
|
+
kinds.add(-1)
|
|
255
|
+
if len(kinds) > 1:
|
|
256
|
+
named = ", ".join(
|
|
257
|
+
"surface" if code == -1 else str(LevelType.of(code))
|
|
258
|
+
for code in sorted(kinds)
|
|
259
|
+
)
|
|
260
|
+
logger.warning(
|
|
261
|
+
"combining several level types into one file (%s). This is valid "
|
|
262
|
+
"GRIB2, but some readers want one vertical axis per file. "
|
|
263
|
+
"cdo splitzaxis, or combine=\"none\", separates them",
|
|
264
|
+
named,
|
|
265
|
+
)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""HTTP access to the Open Data file server."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import httpx
|
|
6
|
+
|
|
7
|
+
from dwdopen.exceptions import CatalogueUnavailableError, DownloadError
|
|
8
|
+
|
|
9
|
+
__all__ = ["HttpClient"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class HttpClient:
|
|
13
|
+
"""Fetches paths below a base URL and maps HTTP failures onto dwdopen errors.
|
|
14
|
+
Contains separate methods for fetching a listing and fetching an actual file,
|
|
15
|
+
to separate potential errors.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
base_url: str,
|
|
21
|
+
*,
|
|
22
|
+
timeout: float = 30.0,
|
|
23
|
+
max_connections: int = 16,
|
|
24
|
+
) -> None:
|
|
25
|
+
self._client = httpx.Client(
|
|
26
|
+
base_url=base_url.rstrip("/"),
|
|
27
|
+
timeout=timeout,
|
|
28
|
+
follow_redirects=True,
|
|
29
|
+
headers={"User-Agent": "dwdopen"},
|
|
30
|
+
limits=httpx.Limits(max_connections=max_connections),
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
def get_listing(self, path: str) -> str:
|
|
34
|
+
"""Fetch a directory listing as text."""
|
|
35
|
+
try:
|
|
36
|
+
response = self._client.get(path)
|
|
37
|
+
response.raise_for_status()
|
|
38
|
+
except httpx.HTTPError as exc:
|
|
39
|
+
raise CatalogueUnavailableError(f"could not read {path}: {exc}") from exc
|
|
40
|
+
return response.text
|
|
41
|
+
|
|
42
|
+
def get_asset(self, path: str) -> bytes:
|
|
43
|
+
"""Fetch one file.
|
|
44
|
+
The status is put in the error so the caller can decide how to recover
|
|
45
|
+
from the error based on the failed state.
|
|
46
|
+
"""
|
|
47
|
+
try:
|
|
48
|
+
response = self._client.get(path)
|
|
49
|
+
response.raise_for_status()
|
|
50
|
+
except httpx.HTTPStatusError as exc:
|
|
51
|
+
raise DownloadError(
|
|
52
|
+
f"could not download {path}: {exc}",
|
|
53
|
+
status=exc.response.status_code,
|
|
54
|
+
) from exc
|
|
55
|
+
except httpx.HTTPError as exc:
|
|
56
|
+
raise DownloadError(f"could not download {path}: {exc}") from exc
|
|
57
|
+
return response.content
|
|
58
|
+
|
|
59
|
+
def close(self) -> None:
|
|
60
|
+
self._client.close()
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Parsing of nginx directory listings.
|
|
2
|
+
|
|
3
|
+
This is the only module in the package that looks at the raw HTML
|
|
4
|
+
returned by the OpenData DWD server.
|
|
5
|
+
To the abstract layers, only ListingEntries are exposed.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from datetime import UTC, datetime
|
|
13
|
+
from urllib.parse import unquote
|
|
14
|
+
|
|
15
|
+
__all__ = ["ListingEntry", "parse_listing"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class ListingEntry:
|
|
20
|
+
"""One row of a directory listing.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
name: str
|
|
24
|
+
"""The URL-decoded path."""
|
|
25
|
+
|
|
26
|
+
is_dir: bool
|
|
27
|
+
|
|
28
|
+
modified: datetime | None = None
|
|
29
|
+
"""In UTC. Used to see if files have settled: We only accept a list entry
|
|
30
|
+
if it already exists on the server for a minute or so. Otherwise, the writing
|
|
31
|
+
might not have finished on the server, or, as DWD has multiple servers that
|
|
32
|
+
are not exactly in sync, we might later want to request that file from a
|
|
33
|
+
server that still does not have the file yet.
|
|
34
|
+
|
|
35
|
+
Might be not available if entry is a directory."""
|
|
36
|
+
|
|
37
|
+
size: int | None = None
|
|
38
|
+
"""Bytes. None for directories."""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# opendata.dwd.de serves nginx autoindex: one entry per line inside a single
|
|
42
|
+
# <pre>:
|
|
43
|
+
#
|
|
44
|
+
# <a href="2026-09-15T06%3A00/">2026-09-15T06:00/</a> 15-Sep-2026 08:42:25 -
|
|
45
|
+
# <a href="PT000H00M.grib2">PT000H00M.grib2</a> 16-Sep-2026 02:40:29 877797
|
|
46
|
+
#
|
|
47
|
+
# The href name should be parsed, as the displayed name might be truncated.
|
|
48
|
+
# Date and size are optional in this pattern. Thanks to Claude for the regex :)
|
|
49
|
+
_ENTRY = re.compile(
|
|
50
|
+
r'<a\s+href="(?P<href>[^"]*)"[^>]*>.*?</a>'
|
|
51
|
+
r"(?:\s+(?P<date>\d{2}-[A-Za-z]{3}-\d{4}\s+\d{2}:\d{2}:\d{2})"
|
|
52
|
+
r"\s+(?P<size>-|\d+))?",
|
|
53
|
+
re.IGNORECASE,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Define months, can't parse them from strptime as that one is locale based.
|
|
57
|
+
_MONTHS = {
|
|
58
|
+
"jan": 1, "feb": 2, "mar": 3, "apr": 4, "may": 5, "jun": 6,
|
|
59
|
+
"jul": 7, "aug": 8, "sep": 9, "oct": 10, "nov": 11, "dec": 12,
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def parse_listing(html: str) -> list[ListingEntry]:
|
|
64
|
+
"""Parse one nginx autoindex page.
|
|
65
|
+
|
|
66
|
+
The parent dir entry is skipped; every other entry is a child of the listed
|
|
67
|
+
directory.
|
|
68
|
+
An unreadable page yields an empty list. The caller should know which
|
|
69
|
+
directories have children. An empty page might be as valid, if there are
|
|
70
|
+
no files yet for the chosen run.
|
|
71
|
+
"""
|
|
72
|
+
entries: list[ListingEntry] = []
|
|
73
|
+
# Iterate through regex occurrences; each match is a file or folder.
|
|
74
|
+
for match in _ENTRY.finditer(html):
|
|
75
|
+
href = match.group("href")
|
|
76
|
+
if href == "../":
|
|
77
|
+
continue
|
|
78
|
+
|
|
79
|
+
size = match.group("size")
|
|
80
|
+
entries.append(
|
|
81
|
+
ListingEntry(
|
|
82
|
+
name=unquote(href.rstrip("/")),
|
|
83
|
+
is_dir=href.endswith("/"),
|
|
84
|
+
modified=_parse_date(match.group("date")),
|
|
85
|
+
size=None if size in (None, "-") else int(size), # None for directories
|
|
86
|
+
)
|
|
87
|
+
)
|
|
88
|
+
return entries
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _parse_date(value: str | None) -> datetime | None:
|
|
92
|
+
"""Parse a listing date, given for example by "16-Sep-2026 02:40:29" as UTC.
|
|
93
|
+
|
|
94
|
+
We assume the timestamps are always UTC and not locale dependent.
|
|
95
|
+
Returns None if parsing fails.
|
|
96
|
+
"""
|
|
97
|
+
if value is None:
|
|
98
|
+
return None
|
|
99
|
+
day, month, rest = value.split("-", 2)
|
|
100
|
+
year, clock = rest.split(maxsplit=1)
|
|
101
|
+
hour, minute, second = clock.split(":")
|
|
102
|
+
try:
|
|
103
|
+
return datetime(
|
|
104
|
+
int(year), _MONTHS[month.lower()], int(day),
|
|
105
|
+
int(hour), int(minute), int(second), tzinfo=UTC,
|
|
106
|
+
)
|
|
107
|
+
except (KeyError, ValueError):
|
|
108
|
+
return None
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Turning a server path into a local file name.
|
|
2
|
+
The name is derived from the asset's key/token pairs, so it is deterministic.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import secrets
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from dwdopen._fileserver.paths import Segment
|
|
12
|
+
|
|
13
|
+
__all__ = ["already_complete", "local_name", "temp_name"]
|
|
14
|
+
|
|
15
|
+
_SAFE = set(
|
|
16
|
+
"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789._-"
|
|
17
|
+
)
|
|
18
|
+
"""Allowed characters for file names. Characters that both Linux and Windows
|
|
19
|
+
accept and also removed some weird chars that might be awkward in shells.
|
|
20
|
+
(In practice, this is mainly about the ':' character on Windows)
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def local_name(keys: tuple[Segment, ...]) -> str:
|
|
25
|
+
"""Build the file name for one asset from its path keys.
|
|
26
|
+
|
|
27
|
+
m/icon-eu, p/T_2M, r/2026-09-18T09:00, s/PT006H00M.grib2 becomes
|
|
28
|
+
``m-icon-eu_p-T_2M_r-2026-09-18T09%3A00_s-PT006H00M.grib2``.
|
|
29
|
+
"""
|
|
30
|
+
return "_".join(f"{key}-{_escape(token)}" for key, token in keys)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def temp_name(final: str) -> str:
|
|
34
|
+
"""Build a partial-download name that no other writer will pick.
|
|
35
|
+
The temporary name carries the process id and a random token, so
|
|
36
|
+
multiple download processes won't interfere.
|
|
37
|
+
"""
|
|
38
|
+
return f"{final}.{os.getpid()}.{secrets.token_hex(4)}.part"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _escape(token: str) -> str:
|
|
42
|
+
"""Escape characters not in _SAFE list."""
|
|
43
|
+
return "".join(c if c in _SAFE else f"%{ord(c):02X}" for c in token)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
GRIB_MAGIC = b"GRIB"
|
|
47
|
+
GRIB_TERMINATOR = b"7777"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def already_complete(target: Path, expected_size: int | None) -> bool:
|
|
51
|
+
"""Whether a file on disk can be trusted as a finished download.
|
|
52
|
+
Two checks: The size has to match what the catalogue listed,
|
|
53
|
+
and the file has to look like GRIB2: every message opens with "GRIB" and
|
|
54
|
+
closes with "7777".
|
|
55
|
+
"""
|
|
56
|
+
if expected_size is None:
|
|
57
|
+
return False
|
|
58
|
+
try:
|
|
59
|
+
if target.stat().st_size != expected_size:
|
|
60
|
+
return False
|
|
61
|
+
if expected_size < len(GRIB_MAGIC) + len(GRIB_TERMINATOR):
|
|
62
|
+
return False
|
|
63
|
+
with target.open("rb") as handle:
|
|
64
|
+
if handle.read(4) != GRIB_MAGIC:
|
|
65
|
+
return False
|
|
66
|
+
handle.seek(-4, os.SEEK_END)
|
|
67
|
+
return handle.read(4) == GRIB_TERMINATOR
|
|
68
|
+
except OSError:
|
|
69
|
+
# Missing or unreadable.
|
|
70
|
+
return False
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""Building URL paths for accessing the server from ordered key/token pairs."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = ["Segment", "build_path"]
|
|
6
|
+
|
|
7
|
+
Segment = tuple[str, str]
|
|
8
|
+
"""One path key and its token value, e.g. ("lvt1", "100")."""
|
|
9
|
+
|
|
10
|
+
V1_ROOT = "weather/nwp/v1"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def build_path(
|
|
14
|
+
*segments: Segment,
|
|
15
|
+
key: str | None = None,
|
|
16
|
+
directory: bool = True,
|
|
17
|
+
) -> str:
|
|
18
|
+
"""Join key/value pairs into a path below the v1 root.
|
|
19
|
+
Each key value pair describes one file attribute, for example the model,
|
|
20
|
+
level type, ensemble member and so on.
|
|
21
|
+
Adding a key without a token value returns a directory listing all possible
|
|
22
|
+
values for that key: build_path(("m", "icon-eu"), ("p", "T"),
|
|
23
|
+
key="lvt1") gives the directory whose entries are the available level type codes.
|
|
24
|
+
|
|
25
|
+
The builder does not know or check the order of the keys. DWD's order is
|
|
26
|
+
m, p, [wvl1], [lvt1, lv1], r, [e], s, where [] keys are optional, for example
|
|
27
|
+
wvl1 is the wavelength for ICON ART, lvt1/lv1 is the level type and level value
|
|
28
|
+
for 3-D grids (absent for 2-D), and e is the ensemble member for ensemble forecasts.
|
|
29
|
+
|
|
30
|
+
DWD's newsletter of 2 September 2026 documents only m, p, lvt1, lv1, r, e, s
|
|
31
|
+
and states that lvt1/lv1 are omitted for single-level parameters. wvl1 is
|
|
32
|
+
not in that list at all: it appears under ICON-ART parameters on the server
|
|
33
|
+
but is documented not in that document.
|
|
34
|
+
Therefore, we allow unknown keys to be passed through rather than validating against
|
|
35
|
+
a fixed set of keys (m, r, e, s...). This also allows us in future to
|
|
36
|
+
support further
|
|
37
|
+
keys directly without having to change any code here.
|
|
38
|
+
|
|
39
|
+
Directories get a trailing slash because nginx answers 301 without one.
|
|
40
|
+
"""
|
|
41
|
+
parts = [V1_ROOT]
|
|
42
|
+
for segment_key, token in segments:
|
|
43
|
+
parts.append(segment_key)
|
|
44
|
+
parts.append(token)
|
|
45
|
+
if key is not None:
|
|
46
|
+
parts.append(key)
|
|
47
|
+
path = "/".join(parts)
|
|
48
|
+
return f"{path}/" if directory else path
|