boldcurator 3.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- boldcurator/__init__.py +18 -0
- boldcurator/assets/icon.icns +0 -0
- boldcurator/assets/icon.ico +0 -0
- boldcurator/build/__init__.py +0 -0
- boldcurator/build/fetch_snapshot.py +491 -0
- boldcurator/build/snapshot_builder.py +823 -0
- boldcurator/build/verify.py +242 -0
- boldcurator/cli.py +642 -0
- boldcurator/config/__init__.py +0 -0
- boldcurator/config/constants.py +566 -0
- boldcurator/core/__init__.py +0 -0
- boldcurator/core/bags.py +191 -0
- boldcurator/core/bins.py +157 -0
- boldcurator/core/frames.py +41 -0
- boldcurator/core/grouping.py +272 -0
- boldcurator/core/phylogeny.py +391 -0
- boldcurator/core/pipeline.py +334 -0
- boldcurator/core/ranking.py +65 -0
- boldcurator/core/refalign.py +226 -0
- boldcurator/core/scoring.py +199 -0
- boldcurator/core/selection.py +74 -0
- boldcurator/core/species.py +204 -0
- boldcurator/core/summaries.py +149 -0
- boldcurator/core/table.py +310 -0
- boldcurator/data/__init__.py +0 -0
- boldcurator/data/queries.py +559 -0
- boldcurator/data/schema.py +250 -0
- boldcurator/data/snapshot.py +161 -0
- boldcurator/desktop.py +483 -0
- boldcurator/io/__init__.py +0 -0
- boldcurator/io/annotations.py +229 -0
- boldcurator/io/exports.py +409 -0
- boldcurator/io/session.py +254 -0
- boldcurator/launcher.py +63 -0
- boldcurator/shortcuts.py +327 -0
- boldcurator/ui/__init__.py +27 -0
- boldcurator/ui/app.py +2647 -0
- boldcurator/ui/format.py +133 -0
- boldcurator/ui/setup.py +291 -0
- boldcurator/ui/state.py +493 -0
- boldcurator/ui/static/phylo/phylo-init.js +409 -0
- boldcurator/ui/static/phylo/phylo.css +31 -0
- boldcurator-3.5.0.dist-info/METADATA +618 -0
- boldcurator-3.5.0.dist-info/RECORD +46 -0
- boldcurator-3.5.0.dist-info/WHEEL +4 -0
- boldcurator-3.5.0.dist-info/entry_points.txt +8 -0
boldcurator/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""BOLDcuratoR -- offline curation of BOLD specimen records.
|
|
2
|
+
|
|
3
|
+
``__version__`` is read from the installed package metadata (what
|
|
4
|
+
``pyproject.toml``'s own ``[project] version`` becomes once the package is
|
|
5
|
+
installed, editable or not), so it can never drift from the one place that
|
|
6
|
+
actually defines it. Falls back to a fixed placeholder only when the
|
|
7
|
+
metadata genuinely isn't there yet -- a source checkout nobody has run
|
|
8
|
+
``pip install -e .`` in, which is not a state that should crash on import.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from importlib.metadata import PackageNotFoundError, version as _pkg_version
|
|
14
|
+
|
|
15
|
+
try:
|
|
16
|
+
__version__ = _pkg_version("boldcurator")
|
|
17
|
+
except PackageNotFoundError:
|
|
18
|
+
__version__ = "0.0.0+unknown"
|
|
Binary file
|
|
Binary file
|
|
File without changes
|
|
@@ -0,0 +1,491 @@
|
|
|
1
|
+
"""Get a pre-built snapshot onto disk -- plan 5.1.
|
|
2
|
+
|
|
3
|
+
Three ways in, all converging on the same download-and-verify core:
|
|
4
|
+
|
|
5
|
+
* ``--url`` + ``--sha256`` -- a plain link, checked against a known digest.
|
|
6
|
+
* ``--manifest`` -- a small JSON file (plan 5.2's shape: ``url``, ``sha256``,
|
|
7
|
+
``snapshot_id``, and optionally ``row_count``/``schema_version``), fetched
|
|
8
|
+
first so this tool never has to be told the digest by hand. A concept DOI's
|
|
9
|
+
manifest always describes the *latest* release, which is what makes
|
|
10
|
+
re-running this a safe "check for updates" -- the file only re-downloads
|
|
11
|
+
when its content actually changed.
|
|
12
|
+
* ``--record`` -- a Zenodo record or concept id, resolved through the public
|
|
13
|
+
REST API (``developers.zenodo.org``) to find the snapshot file and its
|
|
14
|
+
checksum without a manifest at all. A concept id always resolves to the
|
|
15
|
+
newest version, which is the whole point of publishing under one.
|
|
16
|
+
|
|
17
|
+
Nothing here is BOLD-specific or Zenodo-specific below the resolution step --
|
|
18
|
+
any host that can serve a file over HTTP(S) and publish a sha256 works with
|
|
19
|
+
``--url``. Only ``--record`` talks to Zenodo's API.
|
|
20
|
+
|
|
21
|
+
Uses the standard library's ``urllib`` rather than ``requests``: this project
|
|
22
|
+
has stayed dependency-light throughout, and a one-shot streamed download with
|
|
23
|
+
a progress readout does not need more than that. The one addition is
|
|
24
|
+
``truststore``, for *which certificates to trust* -- see ``ssl_context``.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import argparse
|
|
30
|
+
import gzip
|
|
31
|
+
import hashlib
|
|
32
|
+
import json
|
|
33
|
+
import re
|
|
34
|
+
import ssl
|
|
35
|
+
import sys
|
|
36
|
+
import urllib.request
|
|
37
|
+
from dataclasses import dataclass
|
|
38
|
+
from datetime import date
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
from urllib.error import HTTPError, URLError
|
|
41
|
+
|
|
42
|
+
import truststore
|
|
43
|
+
|
|
44
|
+
ZENODO_API = "https://zenodo.org/api/records/{record_id}"
|
|
45
|
+
|
|
46
|
+
#: Matches the numeric id out of a full DOI (``10.5281/zenodo.22849516``), a
|
|
47
|
+
#: ``doi.org``/``zenodo.org`` URL, or the id on its own -- so a curator (or
|
|
48
|
+
#: ``DEFAULT_SNAPSHOT_ZENODO_DOI``) can hand this module whichever form is at
|
|
49
|
+
#: hand. Anchored on ``zenodo.<digits>`` specifically so ``zenodo.org`` itself
|
|
50
|
+
#: (no digits after the dot) never matches.
|
|
51
|
+
_ZENODO_ID_IN_DOI = re.compile(r"zenodo\.(\d+)\b")
|
|
52
|
+
|
|
53
|
+
#: The date this project's own published snapshots carry in their filename
|
|
54
|
+
#: (``bold_snapshot_2026-09-11.duckdb.gz``) -- the same value
|
|
55
|
+
#: ``snapshot_builder.build`` stamps into the file itself as ``snapshot_id``
|
|
56
|
+
#: (``snapshot_id or date.today().isoformat()``). Used to recover a
|
|
57
|
+
#: *comparable* ``Source.snapshot_id`` out of a Zenodo file listing -- see
|
|
58
|
+
#: ``resolve_zenodo_record``.
|
|
59
|
+
_SNAPSHOT_DATE_IN_FILENAME = re.compile(r"(\d{4}-\d{2}-\d{2})")
|
|
60
|
+
|
|
61
|
+
#: Chunk size for the streamed download and the running sha256/md5. A few
|
|
62
|
+
#: hundred KB balances syscall overhead against progress-readout granularity
|
|
63
|
+
#: for files in the hundreds-of-MB to low-GB range this exists to move.
|
|
64
|
+
CHUNK_SIZE = 1024 * 1024
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class FetchError(RuntimeError):
|
|
68
|
+
pass
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass
|
|
72
|
+
class Source:
|
|
73
|
+
"""Where to download from, and how to know the download is right."""
|
|
74
|
+
|
|
75
|
+
url: str
|
|
76
|
+
#: ``"sha256:<hex>"`` or ``"md5:<hex>"`` -- Zenodo publishes md5 by
|
|
77
|
+
#: default, this project's own ``manifest.json`` (plan 5.2) publishes
|
|
78
|
+
#: sha256. ``None`` means the download is not checked, which is only ever
|
|
79
|
+
#: allowed for a bare ``--url`` and prints a loud warning either way.
|
|
80
|
+
checksum: str | None = None
|
|
81
|
+
snapshot_id: str = ""
|
|
82
|
+
row_count: int | None = None
|
|
83
|
+
schema_version: str = ""
|
|
84
|
+
#: The published filename, when known (Zenodo's own ``key``) -- round 4,
|
|
85
|
+
#: item 2's date-named, gzipped published snapshots
|
|
86
|
+
#: (``bold_snapshot_2026-09-11.duckdb.gz``) are told apart from a plain
|
|
87
|
+
#: ``.duckdb`` by this, not by ``url`` alone, which is not guaranteed to
|
|
88
|
+
#: end in the real filename for every host. Falls back to ``url`` itself
|
|
89
|
+
#: when a source (a bare ``--url``, most manifests) doesn't carry one.
|
|
90
|
+
filename: str = ""
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def ssl_context() -> ssl.SSLContext:
|
|
94
|
+
"""Verify HTTPS against the operating system's own trusted certificates.
|
|
95
|
+
|
|
96
|
+
Plain ``urlopen`` verifies against OpenSSL's CA file, found at a path
|
|
97
|
+
compiled into whichever Python built the app. In the frozen macOS build
|
|
98
|
+
that path belongs to the CI runner and doesn't exist on a curator's Mac,
|
|
99
|
+
so every Zenodo request failed with ``CERTIFICATE_VERIFY_FAILED: unable
|
|
100
|
+
to get local issuer certificate`` (a real report, Intel Mac, V3.3).
|
|
101
|
+
``truststore`` asks the OS instead: the macOS Keychain, the Windows
|
|
102
|
+
certificate store, the usual distro CA bundles on Linux. That also
|
|
103
|
+
picks up an institution's own root certificate if its network inspects
|
|
104
|
+
HTTPS, which a bundled list like ``certifi`` would reject.
|
|
105
|
+
"""
|
|
106
|
+
return truststore.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _urlopen(url: str):
|
|
110
|
+
return urllib.request.urlopen(url, timeout=30, context=ssl_context())
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _get_json(url: str) -> dict:
|
|
114
|
+
try:
|
|
115
|
+
with _urlopen(url) as resp:
|
|
116
|
+
return json.loads(resp.read().decode("utf-8"))
|
|
117
|
+
except (HTTPError, URLError) as exc:
|
|
118
|
+
raise FetchError(f"Could not reach {url}: {exc}") from exc
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def resolve_manifest(location: str) -> Source:
|
|
122
|
+
"""A manifest is JSON, local or remote -- read either the same way.
|
|
123
|
+
|
|
124
|
+
Plan 5.2's shape: ``url`` (absolute), ``sha256``, ``snapshot_id``, and
|
|
125
|
+
optionally ``row_count``/``schema_version``. A relative ``url`` is
|
|
126
|
+
refused rather than guessed at -- the manifest is meant to be the one
|
|
127
|
+
place that has to get this right.
|
|
128
|
+
"""
|
|
129
|
+
if location.startswith(("http://", "https://")):
|
|
130
|
+
data = _get_json(location)
|
|
131
|
+
else:
|
|
132
|
+
path = Path(location)
|
|
133
|
+
if not path.exists():
|
|
134
|
+
raise FetchError(f"No manifest at {path}")
|
|
135
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
136
|
+
|
|
137
|
+
url = data.get("url", "")
|
|
138
|
+
if not url or "://" not in url:
|
|
139
|
+
raise FetchError(
|
|
140
|
+
f"manifest.json must give an absolute 'url' field, got {url!r}")
|
|
141
|
+
sha256 = data.get("sha256", "")
|
|
142
|
+
return Source(
|
|
143
|
+
url=url,
|
|
144
|
+
checksum=f"sha256:{sha256}" if sha256 else None,
|
|
145
|
+
snapshot_id=data.get("snapshot_id", ""),
|
|
146
|
+
row_count=data.get("row_count"),
|
|
147
|
+
schema_version=data.get("schema_version", ""),
|
|
148
|
+
filename=data.get("filename", ""),
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _clean_zenodo_id(record_id: str) -> str:
|
|
153
|
+
"""Accept a bare id, a full DOI, or a doi.org/zenodo.org URL alike.
|
|
154
|
+
|
|
155
|
+
``DEFAULT_SNAPSHOT_ZENODO_DOI`` is given as a full DOI (round 4, item 2)
|
|
156
|
+
so it reads the same as the citation on the Zenodo page itself, rather
|
|
157
|
+
than requiring a curator (or this project's own code) to know Zenodo's
|
|
158
|
+
internal numeric id separately.
|
|
159
|
+
"""
|
|
160
|
+
match = _ZENODO_ID_IN_DOI.search(record_id)
|
|
161
|
+
if match:
|
|
162
|
+
return match.group(1)
|
|
163
|
+
return record_id.strip().rstrip("/").rsplit("/", 1)[-1]
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def resolve_zenodo_record(record_id: str, *, filename: str | None = None) -> Source:
|
|
167
|
+
"""Resolve a Zenodo record (or concept) id to one file's URL and checksum.
|
|
168
|
+
|
|
169
|
+
A **concept** id (the one that does not change between versions) always
|
|
170
|
+
redirects to the record's latest version, which is what makes this the
|
|
171
|
+
right id to hand out for "always get the newest snapshot" -- a specific
|
|
172
|
+
version id pins to that version forever, which is a deliberate choice
|
|
173
|
+
too, just a different one.
|
|
174
|
+
|
|
175
|
+
Republished snapshots are date-named and gzipped (round 4, item 2:
|
|
176
|
+
``bold_snapshot_2026-09-11.duckdb.gz``) -- matched here the same way a
|
|
177
|
+
plain ``.duckdb`` already was, so a record holding either (or, someday,
|
|
178
|
+
both across versions) resolves the same way with nothing curator-facing
|
|
179
|
+
to change.
|
|
180
|
+
"""
|
|
181
|
+
record_id = _clean_zenodo_id(record_id)
|
|
182
|
+
data = _get_json(ZENODO_API.format(record_id=record_id))
|
|
183
|
+
files = data.get("files", [])
|
|
184
|
+
if not files:
|
|
185
|
+
raise FetchError(f"Zenodo record {record_id} lists no files")
|
|
186
|
+
|
|
187
|
+
if filename:
|
|
188
|
+
matches = [f for f in files if f.get("key") == filename]
|
|
189
|
+
if not matches:
|
|
190
|
+
available = ", ".join(f.get("key", "?") for f in files)
|
|
191
|
+
raise FetchError(
|
|
192
|
+
f"No file named {filename!r} in record {record_id}. "
|
|
193
|
+
f"Available: {available}")
|
|
194
|
+
elif len(files) == 1:
|
|
195
|
+
matches = files
|
|
196
|
+
else:
|
|
197
|
+
matches = [f for f in files
|
|
198
|
+
if str(f.get("key", "")).endswith((".duckdb", ".duckdb.gz"))]
|
|
199
|
+
if len(matches) != 1:
|
|
200
|
+
available = ", ".join(f.get("key", "?") for f in files)
|
|
201
|
+
raise FetchError(
|
|
202
|
+
f"Record {record_id} has {len(files)} files; pass --filename "
|
|
203
|
+
f"to pick one. Available: {available}")
|
|
204
|
+
|
|
205
|
+
entry = matches[0]
|
|
206
|
+
checksum = entry.get("checksum", "") # Zenodo's own form: "md5:<hex>"
|
|
207
|
+
metadata = data.get("metadata", {})
|
|
208
|
+
filename = str(entry.get("key", ""))
|
|
209
|
+
|
|
210
|
+
# ``snapshot_id`` needs to be *comparable* to what's already on disk --
|
|
211
|
+
# ``_local_snapshot_id`` reads the date ``snapshot_builder`` stamped into
|
|
212
|
+
# the file at build time (e.g. "2026-09-11"). Zenodo's own record id
|
|
213
|
+
# (a new one is minted for every version) is never that date, so using
|
|
214
|
+
# it here made ``fetch()``'s "already have this one, skip" check (and
|
|
215
|
+
# this module's own ``check_for_update``) silently never match for a
|
|
216
|
+
# Zenodo-record source -- only a manifest, which supplies its own
|
|
217
|
+
# ``snapshot_id`` field directly, ever actually hit it. Recovered from
|
|
218
|
+
# the published filename instead, which carries the same date by
|
|
219
|
+
# convention (``bold_snapshot_2026-09-11.duckdb.gz``); the record id is
|
|
220
|
+
# kept as a fallback for a file named some other way, so this never
|
|
221
|
+
# raises, just stops being comparable.
|
|
222
|
+
date_match = _SNAPSHOT_DATE_IN_FILENAME.search(filename)
|
|
223
|
+
snapshot_id = date_match.group(1) if date_match else str(data.get("id", record_id))
|
|
224
|
+
|
|
225
|
+
return Source(
|
|
226
|
+
url=entry["links"]["self"],
|
|
227
|
+
checksum=checksum or None,
|
|
228
|
+
snapshot_id=snapshot_id,
|
|
229
|
+
schema_version=str(metadata.get("version", "")),
|
|
230
|
+
filename=filename,
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _local_snapshot_id(path: Path) -> str | None:
|
|
235
|
+
"""The snapshot id already on disk, or ``None`` if there isn't one yet.
|
|
236
|
+
|
|
237
|
+
Failing to open it (partial download, not a DuckDB file, wrong format)
|
|
238
|
+
is treated the same as "nothing here" -- the download proceeds and
|
|
239
|
+
overwrites it, which is the right outcome for a corrupt leftover.
|
|
240
|
+
"""
|
|
241
|
+
if not path.exists():
|
|
242
|
+
return None
|
|
243
|
+
try:
|
|
244
|
+
from ..data.snapshot import SnapshotStore
|
|
245
|
+
|
|
246
|
+
with SnapshotStore(path) as store:
|
|
247
|
+
return store.info().snapshot_id
|
|
248
|
+
except Exception:
|
|
249
|
+
return None
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _parse_snapshot_date(snapshot_id: str | None) -> date | None:
|
|
253
|
+
"""``snapshot_id`` as a real date, or ``None`` when it isn't one.
|
|
254
|
+
|
|
255
|
+
Not every ``snapshot_id`` is a date -- ``resolve_zenodo_record`` falls
|
|
256
|
+
back to Zenodo's own record id when a file's name carries no
|
|
257
|
+
``YYYY-MM-DD`` (its own docstring explains why). Comparing two
|
|
258
|
+
non-dates, or a date against a non-date, has no meaningful direction,
|
|
259
|
+
so callers should treat ``None`` here as "not comparable", not "equal"
|
|
260
|
+
or "different" in either direction.
|
|
261
|
+
"""
|
|
262
|
+
if not snapshot_id:
|
|
263
|
+
return None
|
|
264
|
+
try:
|
|
265
|
+
return date.fromisoformat(snapshot_id)
|
|
266
|
+
except ValueError:
|
|
267
|
+
return None
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
@dataclass
|
|
271
|
+
class UpdateCheck:
|
|
272
|
+
"""The result of asking Zenodo what's latest, without downloading it."""
|
|
273
|
+
|
|
274
|
+
up_to_date: bool
|
|
275
|
+
local_snapshot_id: str | None
|
|
276
|
+
remote_snapshot_id: str
|
|
277
|
+
remote_filename: str
|
|
278
|
+
|
|
279
|
+
@property
|
|
280
|
+
def comparison(self) -> str:
|
|
281
|
+
"""One of ``"up_to_date"``, ``"remote_newer"``, ``"remote_older"``,
|
|
282
|
+
or ``"different"`` (not equal, but not comparable as dates either --
|
|
283
|
+
e.g. one side is a bare Zenodo record id, not a date-named file).
|
|
284
|
+
|
|
285
|
+
``up_to_date`` (equality) is decided once, in :func:`check_for_update`
|
|
286
|
+
itself -- this only has to work out *which direction* the difference
|
|
287
|
+
goes, for a curator-facing message that shouldn't claim "newer" when
|
|
288
|
+
the local snapshot is actually the more recent one (round found
|
|
289
|
+
during Phylogeny-tab field testing: a local build dated after the
|
|
290
|
+
latest Zenodo publish was reported as having a "newer" one
|
|
291
|
+
available, going backwards in time).
|
|
292
|
+
"""
|
|
293
|
+
if self.up_to_date:
|
|
294
|
+
return "up_to_date"
|
|
295
|
+
local_date = _parse_snapshot_date(self.local_snapshot_id)
|
|
296
|
+
remote_date = _parse_snapshot_date(self.remote_snapshot_id)
|
|
297
|
+
if local_date is None or remote_date is None:
|
|
298
|
+
return "different"
|
|
299
|
+
return "remote_newer" if remote_date > local_date else "remote_older"
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def check_for_update(record_id: str, local_path: Path) -> UpdateCheck:
|
|
303
|
+
"""Ask Zenodo what the latest snapshot is, and compare it to ``local_path``.
|
|
304
|
+
|
|
305
|
+
One small API call (``resolve_zenodo_record``), never a download -- for a
|
|
306
|
+
"is a newer snapshot available?" check the app can run any time, not only
|
|
307
|
+
when a curator is already committing to a multi-GB transfer.
|
|
308
|
+
|
|
309
|
+
``record_id`` should be a **concept** id/DOI (``DEFAULT_SNAPSHOT_ZENODO_DOI``)
|
|
310
|
+
so this always compares against the newest published version, not one
|
|
311
|
+
pinned release. Raises :class:`FetchError` on a network failure, the same
|
|
312
|
+
as every other Zenodo-talking function here -- callers already have to
|
|
313
|
+
handle that for the download path, so there is nothing new to catch.
|
|
314
|
+
|
|
315
|
+
``local_path`` not existing, or not being a readable snapshot, reads as
|
|
316
|
+
"no local version to compare" (``local_snapshot_id=None``,
|
|
317
|
+
``up_to_date=False``) rather than an error -- a curator with no snapshot
|
|
318
|
+
yet still wants to know a snapshot is available, not a crash.
|
|
319
|
+
"""
|
|
320
|
+
source = resolve_zenodo_record(record_id)
|
|
321
|
+
local_id = _local_snapshot_id(local_path)
|
|
322
|
+
return UpdateCheck(
|
|
323
|
+
up_to_date=local_id is not None and local_id == source.snapshot_id,
|
|
324
|
+
local_snapshot_id=local_id,
|
|
325
|
+
remote_snapshot_id=source.snapshot_id,
|
|
326
|
+
remote_filename=source.filename,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _verify(path: Path, checksum: str) -> None:
|
|
331
|
+
algo, _, expected = checksum.partition(":")
|
|
332
|
+
if algo not in ("sha256", "md5"):
|
|
333
|
+
raise FetchError(f"Unsupported checksum kind {algo!r}")
|
|
334
|
+
digest = hashlib.new(algo)
|
|
335
|
+
with open(path, "rb") as fh:
|
|
336
|
+
while chunk := fh.read(CHUNK_SIZE):
|
|
337
|
+
digest.update(chunk)
|
|
338
|
+
actual = digest.hexdigest()
|
|
339
|
+
if actual.lower() != expected.lower():
|
|
340
|
+
raise FetchError(
|
|
341
|
+
f"{algo} mismatch: expected {expected}, got {actual}. The download "
|
|
342
|
+
"is corrupt or the source file changed underneath it -- deleted, "
|
|
343
|
+
"not kept, since a silently wrong snapshot is worse than none.")
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _decompress_gzip(src: Path, dst: Path, *, progress=print) -> None:
|
|
347
|
+
"""Gunzip ``src`` into ``dst``, chunked -- a snapshot can be gigabytes,
|
|
348
|
+
and this is what keeps decompression from doubling that in memory."""
|
|
349
|
+
written = 0
|
|
350
|
+
with gzip.open(src, "rb") as fh_in, open(dst, "wb") as fh_out:
|
|
351
|
+
while chunk := fh_in.read(CHUNK_SIZE):
|
|
352
|
+
fh_out.write(chunk)
|
|
353
|
+
written += len(chunk)
|
|
354
|
+
progress(f"\rDecompressing... {written / 1e6:.0f} MB", end="")
|
|
355
|
+
progress("")
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def download(source: Source, out: Path, *, progress=print) -> Path:
|
|
359
|
+
"""Stream ``source.url`` to a temp file beside ``out``, verify, rename.
|
|
360
|
+
|
|
361
|
+
The temp file (not ``out`` itself) is what a failed or interrupted
|
|
362
|
+
download leaves behind, so ``out`` is never observed half-written.
|
|
363
|
+
|
|
364
|
+
Round 4, item 2: a published snapshot may be gzipped
|
|
365
|
+
(``bold_snapshot_2026-09-11.duckdb.gz``) -- detected from
|
|
366
|
+
``source.filename`` (Zenodo's own name for the file) or, failing that,
|
|
367
|
+
``source.url`` itself. The checksum Zenodo (or a manifest) publishes is
|
|
368
|
+
for the file as uploaded, so it is verified against the *compressed*
|
|
369
|
+
download, before decompressing into ``out``.
|
|
370
|
+
"""
|
|
371
|
+
is_gzipped = (source.filename or source.url).split("?")[0].endswith(".gz")
|
|
372
|
+
tmp = out.with_suffix(out.suffix + (".gz.part" if is_gzipped else ".part"))
|
|
373
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
374
|
+
|
|
375
|
+
try:
|
|
376
|
+
with _urlopen(source.url) as resp:
|
|
377
|
+
total = int(resp.headers.get("Content-Length") or 0)
|
|
378
|
+
written = 0
|
|
379
|
+
with open(tmp, "wb") as fh:
|
|
380
|
+
while chunk := resp.read(CHUNK_SIZE):
|
|
381
|
+
fh.write(chunk)
|
|
382
|
+
written += len(chunk)
|
|
383
|
+
if total:
|
|
384
|
+
progress(f"\r{written / total:.0%} "
|
|
385
|
+
f"({written / 1e6:.0f} / {total / 1e6:.0f} MB)",
|
|
386
|
+
end="")
|
|
387
|
+
else:
|
|
388
|
+
progress(f"\r{written / 1e6:.0f} MB", end="")
|
|
389
|
+
progress("")
|
|
390
|
+
except (HTTPError, URLError) as exc:
|
|
391
|
+
tmp.unlink(missing_ok=True)
|
|
392
|
+
raise FetchError(f"Download failed: {exc}") from exc
|
|
393
|
+
|
|
394
|
+
if source.checksum:
|
|
395
|
+
progress("Verifying checksum...")
|
|
396
|
+
try:
|
|
397
|
+
_verify(tmp, source.checksum)
|
|
398
|
+
except FetchError:
|
|
399
|
+
tmp.unlink(missing_ok=True)
|
|
400
|
+
raise
|
|
401
|
+
else:
|
|
402
|
+
progress("No checksum given -- integrity of this download is NOT verified.")
|
|
403
|
+
|
|
404
|
+
if is_gzipped:
|
|
405
|
+
decompressed = out.with_suffix(out.suffix + ".part")
|
|
406
|
+
try:
|
|
407
|
+
_decompress_gzip(tmp, decompressed, progress=progress)
|
|
408
|
+
except (OSError, gzip.BadGzipFile) as exc:
|
|
409
|
+
decompressed.unlink(missing_ok=True)
|
|
410
|
+
tmp.unlink(missing_ok=True)
|
|
411
|
+
raise FetchError(f"Could not decompress the download: {exc}") from exc
|
|
412
|
+
tmp.unlink()
|
|
413
|
+
decompressed.replace(out)
|
|
414
|
+
else:
|
|
415
|
+
tmp.replace(out)
|
|
416
|
+
return out
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def fetch(args: argparse.Namespace) -> int:
|
|
420
|
+
if args.manifest:
|
|
421
|
+
source = resolve_manifest(args.manifest)
|
|
422
|
+
elif args.record:
|
|
423
|
+
source = resolve_zenodo_record(args.record, filename=args.filename)
|
|
424
|
+
else:
|
|
425
|
+
source = Source(url=args.url, checksum=(f"sha256:{args.sha256}"
|
|
426
|
+
if args.sha256 else None))
|
|
427
|
+
|
|
428
|
+
if not args.force and source.snapshot_id:
|
|
429
|
+
current = _local_snapshot_id(args.out)
|
|
430
|
+
if current == source.snapshot_id:
|
|
431
|
+
print(f"{args.out} is already snapshot {source.snapshot_id} -- "
|
|
432
|
+
"nothing to do. Pass --force to re-download anyway.")
|
|
433
|
+
return 0
|
|
434
|
+
|
|
435
|
+
print(f"Fetching {source.url}")
|
|
436
|
+
download(source, args.out)
|
|
437
|
+
print(f"Wrote {args.out}")
|
|
438
|
+
if source.snapshot_id:
|
|
439
|
+
print(f"snapshot_id: {source.snapshot_id}")
|
|
440
|
+
if source.row_count:
|
|
441
|
+
print(f"row_count: {source.row_count:,}")
|
|
442
|
+
print(
|
|
443
|
+
"This data is CC BY-SA 4.0 (Barcode of Life Data System, "
|
|
444
|
+
"boldsystems.org) -- attribute BOLD Systems and share any "
|
|
445
|
+
"redistributed or adapted dataset under the same licence."
|
|
446
|
+
)
|
|
447
|
+
return 0
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def add_fetch_args(p: argparse.ArgumentParser) -> None:
|
|
451
|
+
"""Shared with ``cli.py``'s ``fetch-snapshot`` subcommand, so the two
|
|
452
|
+
argument sets cannot drift apart."""
|
|
453
|
+
p.add_argument("--out", required=True, type=Path,
|
|
454
|
+
help="where to write the snapshot .duckdb file")
|
|
455
|
+
source = p.add_mutually_exclusive_group(required=True)
|
|
456
|
+
source.add_argument("--url", help="a direct URL to the snapshot file")
|
|
457
|
+
source.add_argument("--manifest",
|
|
458
|
+
help="URL or local path to a manifest.json (plan 5.2) "
|
|
459
|
+
"naming the file, its sha256 and its snapshot_id")
|
|
460
|
+
source.add_argument("--record",
|
|
461
|
+
help="a Zenodo record or concept id, resolved via the "
|
|
462
|
+
"REST API -- a concept id always resolves to the "
|
|
463
|
+
"latest version")
|
|
464
|
+
p.add_argument("--filename",
|
|
465
|
+
help="which file to fetch when --record's record has more "
|
|
466
|
+
"than one (default: the only file, or the one .duckdb)")
|
|
467
|
+
p.add_argument("--sha256",
|
|
468
|
+
help="expected checksum for --url; --manifest and --record "
|
|
469
|
+
"supply their own")
|
|
470
|
+
p.add_argument("--force", action="store_true",
|
|
471
|
+
help="download even if --out already reports this "
|
|
472
|
+
"snapshot_id")
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
476
|
+
p = argparse.ArgumentParser(description=__doc__)
|
|
477
|
+
add_fetch_args(p)
|
|
478
|
+
return p
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def main(argv: list[str] | None = None) -> int:
|
|
482
|
+
args = build_parser().parse_args(argv)
|
|
483
|
+
try:
|
|
484
|
+
return fetch(args)
|
|
485
|
+
except FetchError as exc:
|
|
486
|
+
print(f"Error: {exc}", file=sys.stderr)
|
|
487
|
+
return 1
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
if __name__ == "__main__":
|
|
491
|
+
raise SystemExit(main())
|