boldcurator 3.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. boldcurator/__init__.py +18 -0
  2. boldcurator/assets/icon.icns +0 -0
  3. boldcurator/assets/icon.ico +0 -0
  4. boldcurator/build/__init__.py +0 -0
  5. boldcurator/build/fetch_snapshot.py +491 -0
  6. boldcurator/build/snapshot_builder.py +823 -0
  7. boldcurator/build/verify.py +242 -0
  8. boldcurator/cli.py +642 -0
  9. boldcurator/config/__init__.py +0 -0
  10. boldcurator/config/constants.py +566 -0
  11. boldcurator/core/__init__.py +0 -0
  12. boldcurator/core/bags.py +191 -0
  13. boldcurator/core/bins.py +157 -0
  14. boldcurator/core/frames.py +41 -0
  15. boldcurator/core/grouping.py +272 -0
  16. boldcurator/core/phylogeny.py +391 -0
  17. boldcurator/core/pipeline.py +334 -0
  18. boldcurator/core/ranking.py +65 -0
  19. boldcurator/core/refalign.py +226 -0
  20. boldcurator/core/scoring.py +199 -0
  21. boldcurator/core/selection.py +74 -0
  22. boldcurator/core/species.py +204 -0
  23. boldcurator/core/summaries.py +149 -0
  24. boldcurator/core/table.py +310 -0
  25. boldcurator/data/__init__.py +0 -0
  26. boldcurator/data/queries.py +559 -0
  27. boldcurator/data/schema.py +250 -0
  28. boldcurator/data/snapshot.py +161 -0
  29. boldcurator/desktop.py +483 -0
  30. boldcurator/io/__init__.py +0 -0
  31. boldcurator/io/annotations.py +229 -0
  32. boldcurator/io/exports.py +409 -0
  33. boldcurator/io/session.py +254 -0
  34. boldcurator/launcher.py +63 -0
  35. boldcurator/shortcuts.py +327 -0
  36. boldcurator/ui/__init__.py +27 -0
  37. boldcurator/ui/app.py +2647 -0
  38. boldcurator/ui/format.py +133 -0
  39. boldcurator/ui/setup.py +291 -0
  40. boldcurator/ui/state.py +493 -0
  41. boldcurator/ui/static/phylo/phylo-init.js +409 -0
  42. boldcurator/ui/static/phylo/phylo.css +31 -0
  43. boldcurator-3.5.0.dist-info/METADATA +618 -0
  44. boldcurator-3.5.0.dist-info/RECORD +46 -0
  45. boldcurator-3.5.0.dist-info/WHEEL +4 -0
  46. boldcurator-3.5.0.dist-info/entry_points.txt +8 -0
@@ -0,0 +1,18 @@
1
+ """BOLDcuratoR -- offline curation of BOLD specimen records.
2
+
3
+ ``__version__`` is read from the installed package metadata (what
4
+ ``pyproject.toml``'s own ``[project] version`` becomes once the package is
5
+ installed, editable or not), so it can never drift from the one place that
6
+ actually defines it. Falls back to a fixed placeholder only when the
7
+ metadata genuinely isn't there yet -- a source checkout nobody has run
8
+ ``pip install -e .`` in, which is not a state that should crash on import.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from importlib.metadata import PackageNotFoundError, version as _pkg_version
14
+
15
+ try:
16
+ __version__ = _pkg_version("boldcurator")
17
+ except PackageNotFoundError:
18
+ __version__ = "0.0.0+unknown"
Binary file
Binary file
File without changes
@@ -0,0 +1,491 @@
1
+ """Get a pre-built snapshot onto disk -- plan 5.1.
2
+
3
+ Three ways in, all converging on the same download-and-verify core:
4
+
5
+ * ``--url`` + ``--sha256`` -- a plain link, checked against a known digest.
6
+ * ``--manifest`` -- a small JSON file (plan 5.2's shape: ``url``, ``sha256``,
7
+ ``snapshot_id``, and optionally ``row_count``/``schema_version``), fetched
8
+ first so this tool never has to be told the digest by hand. A concept DOI's
9
+ manifest always describes the *latest* release, which is what makes
10
+ re-running this a safe "check for updates" -- the file only re-downloads
11
+ when its content actually changed.
12
+ * ``--record`` -- a Zenodo record or concept id, resolved through the public
13
+ REST API (``developers.zenodo.org``) to find the snapshot file and its
14
+ checksum without a manifest at all. A concept id always resolves to the
15
+ newest version, which is the whole point of publishing under one.
16
+
17
+ Nothing here is BOLD-specific or Zenodo-specific below the resolution step --
18
+ any host that can serve a file over HTTP(S) and publish a sha256 works with
19
+ ``--url``. Only ``--record`` talks to Zenodo's API.
20
+
21
+ Uses the standard library's ``urllib`` rather than ``requests``: this project
22
+ has stayed dependency-light throughout, and a one-shot streamed download with
23
+ a progress readout does not need more than that. The one addition is
24
+ ``truststore``, for *which certificates to trust* -- see ``ssl_context``.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import argparse
30
+ import gzip
31
+ import hashlib
32
+ import json
33
+ import re
34
+ import ssl
35
+ import sys
36
+ import urllib.request
37
+ from dataclasses import dataclass
38
+ from datetime import date
39
+ from pathlib import Path
40
+ from urllib.error import HTTPError, URLError
41
+
42
+ import truststore
43
+
44
+ ZENODO_API = "https://zenodo.org/api/records/{record_id}"
45
+
46
+ #: Matches the numeric id out of a full DOI (``10.5281/zenodo.22849516``), a
47
+ #: ``doi.org``/``zenodo.org`` URL, or the id on its own -- so a curator (or
48
+ #: ``DEFAULT_SNAPSHOT_ZENODO_DOI``) can hand this module whichever form is at
49
+ #: hand. Anchored on ``zenodo.<digits>`` specifically so ``zenodo.org`` itself
50
+ #: (no digits after the dot) never matches.
51
+ _ZENODO_ID_IN_DOI = re.compile(r"zenodo\.(\d+)\b")
52
+
53
+ #: The date this project's own published snapshots carry in their filename
54
+ #: (``bold_snapshot_2026-09-11.duckdb.gz``) -- the same value
55
+ #: ``snapshot_builder.build`` stamps into the file itself as ``snapshot_id``
56
+ #: (``snapshot_id or date.today().isoformat()``). Used to recover a
57
+ #: *comparable* ``Source.snapshot_id`` out of a Zenodo file listing -- see
58
+ #: ``resolve_zenodo_record``.
59
+ _SNAPSHOT_DATE_IN_FILENAME = re.compile(r"(\d{4}-\d{2}-\d{2})")
60
+
61
+ #: Chunk size for the streamed download and the running sha256/md5. A few
62
+ #: hundred KB balances syscall overhead against progress-readout granularity
63
+ #: for files in the hundreds-of-MB to low-GB range this exists to move.
64
+ CHUNK_SIZE = 1024 * 1024
65
+
66
+
67
+ class FetchError(RuntimeError):
68
+ pass
69
+
70
+
71
+ @dataclass
72
+ class Source:
73
+ """Where to download from, and how to know the download is right."""
74
+
75
+ url: str
76
+ #: ``"sha256:<hex>"`` or ``"md5:<hex>"`` -- Zenodo publishes md5 by
77
+ #: default, this project's own ``manifest.json`` (plan 5.2) publishes
78
+ #: sha256. ``None`` means the download is not checked, which is only ever
79
+ #: allowed for a bare ``--url`` and prints a loud warning either way.
80
+ checksum: str | None = None
81
+ snapshot_id: str = ""
82
+ row_count: int | None = None
83
+ schema_version: str = ""
84
+ #: The published filename, when known (Zenodo's own ``key``) -- round 4,
85
+ #: item 2's date-named, gzipped published snapshots
86
+ #: (``bold_snapshot_2026-09-11.duckdb.gz``) are told apart from a plain
87
+ #: ``.duckdb`` by this, not by ``url`` alone, which is not guaranteed to
88
+ #: end in the real filename for every host. Falls back to ``url`` itself
89
+ #: when a source (a bare ``--url``, most manifests) doesn't carry one.
90
+ filename: str = ""
91
+
92
+
93
+ def ssl_context() -> ssl.SSLContext:
94
+ """Verify HTTPS against the operating system's own trusted certificates.
95
+
96
+ Plain ``urlopen`` verifies against OpenSSL's CA file, found at a path
97
+ compiled into whichever Python built the app. In the frozen macOS build
98
+ that path belongs to the CI runner and doesn't exist on a curator's Mac,
99
+ so every Zenodo request failed with ``CERTIFICATE_VERIFY_FAILED: unable
100
+ to get local issuer certificate`` (a real report, Intel Mac, V3.3).
101
+ ``truststore`` asks the OS instead: the macOS Keychain, the Windows
102
+ certificate store, the usual distro CA bundles on Linux. That also
103
+ picks up an institution's own root certificate if its network inspects
104
+ HTTPS, which a bundled list like ``certifi`` would reject.
105
+ """
106
+ return truststore.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
107
+
108
+
109
+ def _urlopen(url: str):
110
+ return urllib.request.urlopen(url, timeout=30, context=ssl_context())
111
+
112
+
113
+ def _get_json(url: str) -> dict:
114
+ try:
115
+ with _urlopen(url) as resp:
116
+ return json.loads(resp.read().decode("utf-8"))
117
+ except (HTTPError, URLError) as exc:
118
+ raise FetchError(f"Could not reach {url}: {exc}") from exc
119
+
120
+
121
+ def resolve_manifest(location: str) -> Source:
122
+ """A manifest is JSON, local or remote -- read either the same way.
123
+
124
+ Plan 5.2's shape: ``url`` (absolute), ``sha256``, ``snapshot_id``, and
125
+ optionally ``row_count``/``schema_version``. A relative ``url`` is
126
+ refused rather than guessed at -- the manifest is meant to be the one
127
+ place that has to get this right.
128
+ """
129
+ if location.startswith(("http://", "https://")):
130
+ data = _get_json(location)
131
+ else:
132
+ path = Path(location)
133
+ if not path.exists():
134
+ raise FetchError(f"No manifest at {path}")
135
+ data = json.loads(path.read_text(encoding="utf-8"))
136
+
137
+ url = data.get("url", "")
138
+ if not url or "://" not in url:
139
+ raise FetchError(
140
+ f"manifest.json must give an absolute 'url' field, got {url!r}")
141
+ sha256 = data.get("sha256", "")
142
+ return Source(
143
+ url=url,
144
+ checksum=f"sha256:{sha256}" if sha256 else None,
145
+ snapshot_id=data.get("snapshot_id", ""),
146
+ row_count=data.get("row_count"),
147
+ schema_version=data.get("schema_version", ""),
148
+ filename=data.get("filename", ""),
149
+ )
150
+
151
+
152
+ def _clean_zenodo_id(record_id: str) -> str:
153
+ """Accept a bare id, a full DOI, or a doi.org/zenodo.org URL alike.
154
+
155
+ ``DEFAULT_SNAPSHOT_ZENODO_DOI`` is given as a full DOI (round 4, item 2)
156
+ so it reads the same as the citation on the Zenodo page itself, rather
157
+ than requiring a curator (or this project's own code) to know Zenodo's
158
+ internal numeric id separately.
159
+ """
160
+ match = _ZENODO_ID_IN_DOI.search(record_id)
161
+ if match:
162
+ return match.group(1)
163
+ return record_id.strip().rstrip("/").rsplit("/", 1)[-1]
164
+
165
+
166
+ def resolve_zenodo_record(record_id: str, *, filename: str | None = None) -> Source:
167
+ """Resolve a Zenodo record (or concept) id to one file's URL and checksum.
168
+
169
+ A **concept** id (the one that does not change between versions) always
170
+ redirects to the record's latest version, which is what makes this the
171
+ right id to hand out for "always get the newest snapshot" -- a specific
172
+ version id pins to that version forever, which is a deliberate choice
173
+ too, just a different one.
174
+
175
+ Republished snapshots are date-named and gzipped (round 4, item 2:
176
+ ``bold_snapshot_2026-09-11.duckdb.gz``) -- matched here the same way a
177
+ plain ``.duckdb`` already was, so a record holding either (or, someday,
178
+ both across versions) resolves the same way with nothing curator-facing
179
+ to change.
180
+ """
181
+ record_id = _clean_zenodo_id(record_id)
182
+ data = _get_json(ZENODO_API.format(record_id=record_id))
183
+ files = data.get("files", [])
184
+ if not files:
185
+ raise FetchError(f"Zenodo record {record_id} lists no files")
186
+
187
+ if filename:
188
+ matches = [f for f in files if f.get("key") == filename]
189
+ if not matches:
190
+ available = ", ".join(f.get("key", "?") for f in files)
191
+ raise FetchError(
192
+ f"No file named {filename!r} in record {record_id}. "
193
+ f"Available: {available}")
194
+ elif len(files) == 1:
195
+ matches = files
196
+ else:
197
+ matches = [f for f in files
198
+ if str(f.get("key", "")).endswith((".duckdb", ".duckdb.gz"))]
199
+ if len(matches) != 1:
200
+ available = ", ".join(f.get("key", "?") for f in files)
201
+ raise FetchError(
202
+ f"Record {record_id} has {len(files)} files; pass --filename "
203
+ f"to pick one. Available: {available}")
204
+
205
+ entry = matches[0]
206
+ checksum = entry.get("checksum", "") # Zenodo's own form: "md5:<hex>"
207
+ metadata = data.get("metadata", {})
208
+ filename = str(entry.get("key", ""))
209
+
210
+ # ``snapshot_id`` needs to be *comparable* to what's already on disk --
211
+ # ``_local_snapshot_id`` reads the date ``snapshot_builder`` stamped into
212
+ # the file at build time (e.g. "2026-09-11"). Zenodo's own record id
213
+ # (a new one is minted for every version) is never that date, so using
214
+ # it here made ``fetch()``'s "already have this one, skip" check (and
215
+ # this module's own ``check_for_update``) silently never match for a
216
+ # Zenodo-record source -- only a manifest, which supplies its own
217
+ # ``snapshot_id`` field directly, ever actually hit it. Recovered from
218
+ # the published filename instead, which carries the same date by
219
+ # convention (``bold_snapshot_2026-09-11.duckdb.gz``); the record id is
220
+ # kept as a fallback for a file named some other way, so this never
221
+ # raises, just stops being comparable.
222
+ date_match = _SNAPSHOT_DATE_IN_FILENAME.search(filename)
223
+ snapshot_id = date_match.group(1) if date_match else str(data.get("id", record_id))
224
+
225
+ return Source(
226
+ url=entry["links"]["self"],
227
+ checksum=checksum or None,
228
+ snapshot_id=snapshot_id,
229
+ schema_version=str(metadata.get("version", "")),
230
+ filename=filename,
231
+ )
232
+
233
+
234
+ def _local_snapshot_id(path: Path) -> str | None:
235
+ """The snapshot id already on disk, or ``None`` if there isn't one yet.
236
+
237
+ Failing to open it (partial download, not a DuckDB file, wrong format)
238
+ is treated the same as "nothing here" -- the download proceeds and
239
+ overwrites it, which is the right outcome for a corrupt leftover.
240
+ """
241
+ if not path.exists():
242
+ return None
243
+ try:
244
+ from ..data.snapshot import SnapshotStore
245
+
246
+ with SnapshotStore(path) as store:
247
+ return store.info().snapshot_id
248
+ except Exception:
249
+ return None
250
+
251
+
252
+ def _parse_snapshot_date(snapshot_id: str | None) -> date | None:
253
+ """``snapshot_id`` as a real date, or ``None`` when it isn't one.
254
+
255
+ Not every ``snapshot_id`` is a date -- ``resolve_zenodo_record`` falls
256
+ back to Zenodo's own record id when a file's name carries no
257
+ ``YYYY-MM-DD`` (its own docstring explains why). Comparing two
258
+ non-dates, or a date against a non-date, has no meaningful direction,
259
+ so callers should treat ``None`` here as "not comparable", not "equal"
260
+ or "different" in either direction.
261
+ """
262
+ if not snapshot_id:
263
+ return None
264
+ try:
265
+ return date.fromisoformat(snapshot_id)
266
+ except ValueError:
267
+ return None
268
+
269
+
270
+ @dataclass
271
+ class UpdateCheck:
272
+ """The result of asking Zenodo what's latest, without downloading it."""
273
+
274
+ up_to_date: bool
275
+ local_snapshot_id: str | None
276
+ remote_snapshot_id: str
277
+ remote_filename: str
278
+
279
+ @property
280
+ def comparison(self) -> str:
281
+ """One of ``"up_to_date"``, ``"remote_newer"``, ``"remote_older"``,
282
+ or ``"different"`` (not equal, but not comparable as dates either --
283
+ e.g. one side is a bare Zenodo record id, not a date-named file).
284
+
285
+ ``up_to_date`` (equality) is decided once, in :func:`check_for_update`
286
+ itself -- this only has to work out *which direction* the difference
287
+ goes, for a curator-facing message that shouldn't claim "newer" when
288
+ the local snapshot is actually the more recent one (round found
289
+ during Phylogeny-tab field testing: a local build dated after the
290
+ latest Zenodo publish was reported as having a "newer" one
291
+ available, going backwards in time).
292
+ """
293
+ if self.up_to_date:
294
+ return "up_to_date"
295
+ local_date = _parse_snapshot_date(self.local_snapshot_id)
296
+ remote_date = _parse_snapshot_date(self.remote_snapshot_id)
297
+ if local_date is None or remote_date is None:
298
+ return "different"
299
+ return "remote_newer" if remote_date > local_date else "remote_older"
300
+
301
+
302
+ def check_for_update(record_id: str, local_path: Path) -> UpdateCheck:
303
+ """Ask Zenodo what the latest snapshot is, and compare it to ``local_path``.
304
+
305
+ One small API call (``resolve_zenodo_record``), never a download -- for a
306
+ "is a newer snapshot available?" check the app can run any time, not only
307
+ when a curator is already committing to a multi-GB transfer.
308
+
309
+ ``record_id`` should be a **concept** id/DOI (``DEFAULT_SNAPSHOT_ZENODO_DOI``)
310
+ so this always compares against the newest published version, not one
311
+ pinned release. Raises :class:`FetchError` on a network failure, the same
312
+ as every other Zenodo-talking function here -- callers already have to
313
+ handle that for the download path, so there is nothing new to catch.
314
+
315
+ ``local_path`` not existing, or not being a readable snapshot, reads as
316
+ "no local version to compare" (``local_snapshot_id=None``,
317
+ ``up_to_date=False``) rather than an error -- a curator with no snapshot
318
+ yet still wants to know a snapshot is available, not a crash.
319
+ """
320
+ source = resolve_zenodo_record(record_id)
321
+ local_id = _local_snapshot_id(local_path)
322
+ return UpdateCheck(
323
+ up_to_date=local_id is not None and local_id == source.snapshot_id,
324
+ local_snapshot_id=local_id,
325
+ remote_snapshot_id=source.snapshot_id,
326
+ remote_filename=source.filename,
327
+ )
328
+
329
+
330
+ def _verify(path: Path, checksum: str) -> None:
331
+ algo, _, expected = checksum.partition(":")
332
+ if algo not in ("sha256", "md5"):
333
+ raise FetchError(f"Unsupported checksum kind {algo!r}")
334
+ digest = hashlib.new(algo)
335
+ with open(path, "rb") as fh:
336
+ while chunk := fh.read(CHUNK_SIZE):
337
+ digest.update(chunk)
338
+ actual = digest.hexdigest()
339
+ if actual.lower() != expected.lower():
340
+ raise FetchError(
341
+ f"{algo} mismatch: expected {expected}, got {actual}. The download "
342
+ "is corrupt or the source file changed underneath it -- deleted, "
343
+ "not kept, since a silently wrong snapshot is worse than none.")
344
+
345
+
346
+ def _decompress_gzip(src: Path, dst: Path, *, progress=print) -> None:
347
+ """Gunzip ``src`` into ``dst``, chunked -- a snapshot can be gigabytes,
348
+ and this is what keeps decompression from doubling that in memory."""
349
+ written = 0
350
+ with gzip.open(src, "rb") as fh_in, open(dst, "wb") as fh_out:
351
+ while chunk := fh_in.read(CHUNK_SIZE):
352
+ fh_out.write(chunk)
353
+ written += len(chunk)
354
+ progress(f"\rDecompressing... {written / 1e6:.0f} MB", end="")
355
+ progress("")
356
+
357
+
358
+ def download(source: Source, out: Path, *, progress=print) -> Path:
359
+ """Stream ``source.url`` to a temp file beside ``out``, verify, rename.
360
+
361
+ The temp file (not ``out`` itself) is what a failed or interrupted
362
+ download leaves behind, so ``out`` is never observed half-written.
363
+
364
+ Round 4, item 2: a published snapshot may be gzipped
365
+ (``bold_snapshot_2026-09-11.duckdb.gz``) -- detected from
366
+ ``source.filename`` (Zenodo's own name for the file) or, failing that,
367
+ ``source.url`` itself. The checksum Zenodo (or a manifest) publishes is
368
+ for the file as uploaded, so it is verified against the *compressed*
369
+ download, before decompressing into ``out``.
370
+ """
371
+ is_gzipped = (source.filename or source.url).split("?")[0].endswith(".gz")
372
+ tmp = out.with_suffix(out.suffix + (".gz.part" if is_gzipped else ".part"))
373
+ out.parent.mkdir(parents=True, exist_ok=True)
374
+
375
+ try:
376
+ with _urlopen(source.url) as resp:
377
+ total = int(resp.headers.get("Content-Length") or 0)
378
+ written = 0
379
+ with open(tmp, "wb") as fh:
380
+ while chunk := resp.read(CHUNK_SIZE):
381
+ fh.write(chunk)
382
+ written += len(chunk)
383
+ if total:
384
+ progress(f"\r{written / total:.0%} "
385
+ f"({written / 1e6:.0f} / {total / 1e6:.0f} MB)",
386
+ end="")
387
+ else:
388
+ progress(f"\r{written / 1e6:.0f} MB", end="")
389
+ progress("")
390
+ except (HTTPError, URLError) as exc:
391
+ tmp.unlink(missing_ok=True)
392
+ raise FetchError(f"Download failed: {exc}") from exc
393
+
394
+ if source.checksum:
395
+ progress("Verifying checksum...")
396
+ try:
397
+ _verify(tmp, source.checksum)
398
+ except FetchError:
399
+ tmp.unlink(missing_ok=True)
400
+ raise
401
+ else:
402
+ progress("No checksum given -- integrity of this download is NOT verified.")
403
+
404
+ if is_gzipped:
405
+ decompressed = out.with_suffix(out.suffix + ".part")
406
+ try:
407
+ _decompress_gzip(tmp, decompressed, progress=progress)
408
+ except (OSError, gzip.BadGzipFile) as exc:
409
+ decompressed.unlink(missing_ok=True)
410
+ tmp.unlink(missing_ok=True)
411
+ raise FetchError(f"Could not decompress the download: {exc}") from exc
412
+ tmp.unlink()
413
+ decompressed.replace(out)
414
+ else:
415
+ tmp.replace(out)
416
+ return out
417
+
418
+
419
+ def fetch(args: argparse.Namespace) -> int:
420
+ if args.manifest:
421
+ source = resolve_manifest(args.manifest)
422
+ elif args.record:
423
+ source = resolve_zenodo_record(args.record, filename=args.filename)
424
+ else:
425
+ source = Source(url=args.url, checksum=(f"sha256:{args.sha256}"
426
+ if args.sha256 else None))
427
+
428
+ if not args.force and source.snapshot_id:
429
+ current = _local_snapshot_id(args.out)
430
+ if current == source.snapshot_id:
431
+ print(f"{args.out} is already snapshot {source.snapshot_id} -- "
432
+ "nothing to do. Pass --force to re-download anyway.")
433
+ return 0
434
+
435
+ print(f"Fetching {source.url}")
436
+ download(source, args.out)
437
+ print(f"Wrote {args.out}")
438
+ if source.snapshot_id:
439
+ print(f"snapshot_id: {source.snapshot_id}")
440
+ if source.row_count:
441
+ print(f"row_count: {source.row_count:,}")
442
+ print(
443
+ "This data is CC BY-SA 4.0 (Barcode of Life Data System, "
444
+ "boldsystems.org) -- attribute BOLD Systems and share any "
445
+ "redistributed or adapted dataset under the same licence."
446
+ )
447
+ return 0
448
+
449
+
450
+ def add_fetch_args(p: argparse.ArgumentParser) -> None:
451
+ """Shared with ``cli.py``'s ``fetch-snapshot`` subcommand, so the two
452
+ argument sets cannot drift apart."""
453
+ p.add_argument("--out", required=True, type=Path,
454
+ help="where to write the snapshot .duckdb file")
455
+ source = p.add_mutually_exclusive_group(required=True)
456
+ source.add_argument("--url", help="a direct URL to the snapshot file")
457
+ source.add_argument("--manifest",
458
+ help="URL or local path to a manifest.json (plan 5.2) "
459
+ "naming the file, its sha256 and its snapshot_id")
460
+ source.add_argument("--record",
461
+ help="a Zenodo record or concept id, resolved via the "
462
+ "REST API -- a concept id always resolves to the "
463
+ "latest version")
464
+ p.add_argument("--filename",
465
+ help="which file to fetch when --record's record has more "
466
+ "than one (default: the only file, or the one .duckdb)")
467
+ p.add_argument("--sha256",
468
+ help="expected checksum for --url; --manifest and --record "
469
+ "supply their own")
470
+ p.add_argument("--force", action="store_true",
471
+ help="download even if --out already reports this "
472
+ "snapshot_id")
473
+
474
+
475
+ def build_parser() -> argparse.ArgumentParser:
476
+ p = argparse.ArgumentParser(description=__doc__)
477
+ add_fetch_args(p)
478
+ return p
479
+
480
+
481
+ def main(argv: list[str] | None = None) -> int:
482
+ args = build_parser().parse_args(argv)
483
+ try:
484
+ return fetch(args)
485
+ except FetchError as exc:
486
+ print(f"Error: {exc}", file=sys.stderr)
487
+ return 1
488
+
489
+
490
+ if __name__ == "__main__":
491
+ raise SystemExit(main())