parseo 0.4.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. parseo/__init__.py +53 -0
  2. parseo/_epsg_lookup.py +145 -0
  3. parseo/_field_mappings.py +219 -0
  4. parseo/_json.py +14 -0
  5. parseo/_tile_systems.py +51 -0
  6. parseo/assembler.py +225 -0
  7. parseo/cli.py +383 -0
  8. parseo/parser.py +784 -0
  9. parseo/schema_registry.py +246 -0
  10. parseo/schemas/copernicus/clms/clc/clc_filename_v1_0_0.json +48 -0
  11. parseo/schemas/copernicus/clms/clc/clc_filename_v1_1_0.json +65 -0
  12. parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_0.json +90 -0
  13. parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_1.json +80 -0
  14. parseo/schemas/copernicus/clms/egms/egms_gnss_model_filename_v0_0_0.json +46 -0
  15. parseo/schemas/copernicus/clms/egms/egms_l2a_product_filename_v0_0_0.json +71 -0
  16. parseo/schemas/copernicus/clms/egms/egms_l3_velocity_grid_filename_v0_0_0.json +64 -0
  17. parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_0.json +115 -0
  18. parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_1.json +103 -0
  19. parseo/schemas/copernicus/clms/hr-vpp/st_filename_v0_0_0.json +82 -0
  20. parseo/schemas/copernicus/clms/hr-vpp/vi_filename_v0_0_0.json +89 -0
  21. parseo/schemas/copernicus/clms/hr-vpp/vpp_filename_v0_0_0.json +109 -0
  22. parseo/schemas/copernicus/clms/hr-wsi/cc_filename_v0_0_0.json +84 -0
  23. parseo/schemas/copernicus/clms/hr-wsi/fsc_filename_v0_0_0.json +88 -0
  24. parseo/schemas/copernicus/clms/hr-wsi/gfsc_filename_v0_0_0.json +83 -0
  25. parseo/schemas/copernicus/clms/hr-wsi/icd_filename_v0_0_0.json +96 -0
  26. parseo/schemas/copernicus/clms/hr-wsi/sp_filename_v0_0_0.json +177 -0
  27. parseo/schemas/copernicus/clms/hr-wsi/sws_filename_v0_0_0.json +83 -0
  28. parseo/schemas/copernicus/clms/hr-wsi/wcd_filename_v0_0_0.json +96 -0
  29. parseo/schemas/copernicus/clms/hr-wsi/wds_filename_v0_0_0.json +83 -0
  30. parseo/schemas/copernicus/clms/hr-wsi/wic_comb_filename_v0_0_0.json +82 -0
  31. parseo/schemas/copernicus/clms/hr-wsi/wic_filename_v0_0_0.json +90 -0
  32. parseo/schemas/copernicus/clms/hrl/fty_filename_v0_0_0.json +48 -0
  33. parseo/schemas/copernicus/clms/hrl/gra_filename_v0_0_0.json +48 -0
  34. parseo/schemas/copernicus/clms/hrl/ibu_filename_v0_0_0.json +57 -0
  35. parseo/schemas/copernicus/clms/hrl/imp_filename_v0_0_0.json +48 -0
  36. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_0.json +73 -0
  37. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_1.json +89 -0
  38. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_2.json +89 -0
  39. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_0.json +48 -0
  40. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_1.json +49 -0
  41. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_2.json +93 -0
  42. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_3.json +99 -0
  43. parseo/schemas/copernicus/clms/hrl/tcd_filename_v0_0_0.json +48 -0
  44. parseo/schemas/copernicus/clms/hrl/vlcc_filename_v0_0_0.json +105 -0
  45. parseo/schemas/copernicus/clms/hrl/waw_filename_v0_0_0.json +48 -0
  46. parseo/schemas/copernicus/clms/n2k/n2k_filename_v1_0_0.json +43 -0
  47. parseo/schemas/copernicus/clms/pa/pa_filename_v1_0_0.json +99 -0
  48. parseo/schemas/copernicus/clms/riparian-zones/rpz_filename_v0_0_0.json +50 -0
  49. parseo/schemas/copernicus/clms/urban-atlas/ua_dhm_filename_v0_0_0.json +53 -0
  50. parseo/schemas/copernicus/clms/urban-atlas/ua_filename_v0_0_0.json +105 -0
  51. parseo/schemas/copernicus/clms/urban-atlas/ua_stl_filename_v0_0_0.json +89 -0
  52. parseo/schemas/copernicus/sentinel/s1_filename_v1_0_0.json +78 -0
  53. parseo/schemas/copernicus/sentinel/s2_filename_v1_0_0.json +62 -0
  54. parseo/schemas/copernicus/sentinel/s3_filename_v1_0_0.json +22 -0
  55. parseo/schemas/copernicus/sentinel/s4_filename_v1_0_0.json +21 -0
  56. parseo/schemas/copernicus/sentinel/s5p_filename_v1_0_0.json +21 -0
  57. parseo/schemas/copernicus/sentinel/s6_filename_v1_0_0.json +23 -0
  58. parseo/schemas/eumetsat/metop_filename_v1_0_0.json +68 -0
  59. parseo/schemas/eumetsat/mtg_filename_v1_0_0.json +65 -0
  60. parseo/schemas/nasa/modis_filename_v1_0_0.json +54 -0
  61. parseo/schemas/usgs/landsat/landsat_filename_v1_0_0.json +50 -0
  62. parseo/stac_http.py +267 -0
  63. parseo/stac_scraper.py +115 -0
  64. parseo/template.py +78 -0
  65. parseo-0.4.4.dist-info/METADATA +402 -0
  66. parseo-0.4.4.dist-info/RECORD +70 -0
  67. parseo-0.4.4.dist-info/WHEEL +5 -0
  68. parseo-0.4.4.dist-info/entry_points.txt +2 -0
  69. parseo-0.4.4.dist-info/licenses/LICENSE.txt +287 -0
  70. parseo-0.4.4.dist-info/top_level.txt +1 -0
@@ -0,0 +1,68 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "schema_id": "eumetsat:metop",
4
+ "schema_version": "1.0.0",
5
+ "status": "current",
6
+ "description": "EUMETSAT Metop product filename (extension optional).",
7
+ "fields": {
8
+ "platform": {
9
+ "type": "string",
10
+ "enum": [
11
+ "M01",
12
+ "M02",
13
+ "M03"
14
+ ],
15
+ "description": "Spacecraft unit"
16
+ },
17
+ "instrument": {
18
+ "type": "string",
19
+ "enum": [
20
+ "AVHR",
21
+ "IASI",
22
+ "ASCAT"
23
+ ],
24
+ "description": "Instrument short name"
25
+ },
26
+ "processing_level": {
27
+ "type": "string",
28
+ "enum": [
29
+ "L1B",
30
+ "L2"
31
+ ],
32
+ "description": "Processing level"
33
+ },
34
+ "start_datetime": {
35
+ "type": "string",
36
+ "pattern": "^\\d{8}T\\d{6}Z$",
37
+ "description": "Acquisition start time (UTC, YYYYMMDDTHHMMSSZ)"
38
+ },
39
+ "end_datetime": {
40
+ "type": "string",
41
+ "pattern": "^\\d{8}T\\d{6}Z$",
42
+ "description": "Acquisition end time (UTC, YYYYMMDDTHHMMSSZ)"
43
+ },
44
+ "orbit_number": {
45
+ "type": "string",
46
+ "pattern": "^O\\d{6}$",
47
+ "description": "Orbit number"
48
+ },
49
+ "collection": {
50
+ "type": "string",
51
+ "pattern": "^C\\d{2}$",
52
+ "description": "Collection identifier"
53
+ },
54
+ "version": {
55
+ "type": "string",
56
+ "pattern": "^\\d{2}$",
57
+ "description": "Product version"
58
+ },
59
+ "extension": {
60
+ "type": "string",
61
+ "enum": [
62
+ "nc"
63
+ ],
64
+ "description": "File extension without leading dot"
65
+ }
66
+ },
67
+ "template": "{platform}_{instrument}_{processing_level}_{start_datetime}_{end_datetime}_{orbit_number}_{collection}_V{version}[.{extension}]"
68
+ }
@@ -0,0 +1,65 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "schema_id": "eumetsat:mtg",
4
+ "schema_version": "1.0.0",
5
+ "status": "current",
6
+ "description": "EUMETSAT MTG product filename (extension optional).",
7
+ "fields": {
8
+ "platform": {
9
+ "type": "string",
10
+ "enum": [
11
+ "MTG-I1",
12
+ "MTG-I2",
13
+ "MTG-S1",
14
+ "MTG-S2"
15
+ ],
16
+ "description": "Spacecraft unit"
17
+ },
18
+ "instrument": {
19
+ "type": "string",
20
+ "enum": [
21
+ "FCI",
22
+ "LI",
23
+ "IRS",
24
+ "UVN"
25
+ ],
26
+ "description": "Instrument short name"
27
+ },
28
+ "product_type": {
29
+ "type": "string",
30
+ "pattern": "^[A-Z0-9_]{2,10}$",
31
+ "description": "Product short name"
32
+ },
33
+ "processing_level": {
34
+ "type": "string",
35
+ "enum": [
36
+ "L1",
37
+ "L2"
38
+ ],
39
+ "description": "Processing level"
40
+ },
41
+ "start_datetime": {
42
+ "type": "string",
43
+ "pattern": "^\\d{8}T\\d{6}Z$",
44
+ "description": "Acquisition start time (UTC, YYYYMMDDTHHMMSSZ)"
45
+ },
46
+ "end_datetime": {
47
+ "type": "string",
48
+ "pattern": "^\\d{8}T\\d{6}Z$",
49
+ "description": "Acquisition end time (UTC, YYYYMMDDTHHMMSSZ)"
50
+ },
51
+ "version": {
52
+ "type": "string",
53
+ "pattern": "^\\d{2}$",
54
+ "description": "Product version"
55
+ },
56
+ "extension": {
57
+ "type": "string",
58
+ "enum": [
59
+ "nc"
60
+ ],
61
+ "description": "File extension without leading dot"
62
+ }
63
+ },
64
+ "template": "{platform}_{instrument}_{product_type}_{processing_level}_{start_datetime}_{end_datetime}_V{version}[.{extension}]"
65
+ }
@@ -0,0 +1,54 @@
1
+ {
2
+ "schema_id": "nasa:modis",
3
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
4
+ "schema_version": "1.0.0",
5
+ "status": "current",
6
+ "stac_version": "1.1.0",
7
+
8
+ "description": "MODIS product filename (extension optional).",
9
+ "fields": {
10
+ "platform": {
11
+ "type": "string",
12
+ "enum": ["MOD","MYD","MCD"],
13
+ "description": "Spacecraft platform (MOD=Terra, MYD=Aqua, MCD=Combined)",
14
+ "stac_map": {
15
+ "MOD": {
16
+ "platform": "Terra",
17
+ "instruments": ["MODIS"]
18
+ },
19
+ "MYD": {
20
+ "platform": "Aqua",
21
+ "instruments": ["MODIS"]
22
+ },
23
+ "MCD": {
24
+ "platform": "Combined",
25
+ "instruments": ["MODIS"]
26
+ }
27
+ }
28
+ },
29
+ "product": {
30
+ "type": "string",
31
+ "pattern": "^\\d{2}$",
32
+ "description": "Two-digit product code (e.g., '09' for Surface Reflectance)"
33
+ },
34
+ "variant": {
35
+ "type": "string",
36
+ "pattern": "^[A-Za-z0-9]+$",
37
+ "description": "Product variant or sub-type (e.g., GA, A1)"
38
+ },
39
+ "acq_date": {"type": "string", "pattern": "^\\d{7}$", "description": "Acquisition date (YYYYDDD)"},
40
+ "tile_id": {"type": "string", "pattern": "^h\\d{2}v\\d{2}$", "description": "MODIS sinusoidal tile"},
41
+ "collection": {"type": "string", "pattern": "^\\d{3}$", "description": "Collection number"},
42
+ "proc_date": {"type": "string", "pattern": "^\\d{13}$", "description": "Production date (YYYYDDDHHMMSS)"},
43
+ "extension": {"type": "string", "pattern": "^[A-Za-z0-9]+$", "description": "File extension."}
44
+ },
45
+ "template": "{platform}{product}{variant}.A{acq_date}.{tile_id}.{collection}.{proc_date}[.{extension}]",
46
+ "examples": [
47
+ "MOD09GA.A2021123.h18v04.006.2021132234506.hdf",
48
+ "MYD11A1.A2021001.h21v09.061.2021015190034"
49
+ ],
50
+ "counter_examples": [
51
+ "MOD09GA.X2021123.h18v04.006.2021132234506.hdf",
52
+ "MOD09GA.A2021123.H18V04.006.2021132234506.hdf"
53
+ ]
54
+ }
@@ -0,0 +1,50 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "schema_id": "usgs:landsat:landsat",
4
+ "schema_version": "1.0.0",
5
+ "status": "current",
6
+ "stac_version": "1.1.0",
7
+
8
+ "description": "Landsat Collection 2 scene identifier (extension optional).",
9
+ "fields": {
10
+ "platform": {
11
+ "type": "string",
12
+ "pattern": "^(LT04|LT05|LE07|LC08|LC09)$",
13
+ "description": "Platform and sensor code",
14
+ "stac_map": {
15
+ "LT04": {
16
+ "platform": "landsat-4",
17
+ "instruments": ["TM"]
18
+ },
19
+ "LT05": {
20
+ "platform": "landsat-5",
21
+ "instruments": ["TM"]
22
+ },
23
+ "LE07": {
24
+ "platform": "landsat-7",
25
+ "instruments": ["ETM+"]
26
+ },
27
+ "LC08": {
28
+ "platform": "landsat-8",
29
+ "instruments": ["OLI_TIRS"]
30
+ },
31
+ "LC09": {
32
+ "platform": "landsat-9",
33
+ "instruments": ["OLI_TIRS"]
34
+ }
35
+ }
36
+ },
37
+ "processing_level": {"type": "string", "enum": ["L1TP", "L1GT", "L1GS", "L2SP", "L2SR"], "description": "Processing level"},
38
+ "wrs_path": {"type": "string", "pattern": "^\\d{3}$", "description": "WRS-2 path"},
39
+ "wrs_row": {"type": "string", "pattern": "^\\d{3}$", "description": "WRS-2 row"},
40
+ "acq_date": {"type": "string", "pattern": "^\\d{8}$", "description": "Acquisition date (YYYYMMDD)"},
41
+ "proc_date": {"type": "string", "pattern": "^\\d{8}$", "description": "Processing date (YYYYMMDD)"},
42
+ "collection_number": {"type": "string", "pattern": "^\\d{2}$", "description": "Collection number"},
43
+ "tier": {"type": "string", "enum": ["T1", "T2", "RT"], "description": "Tier classification"},
44
+ "extension": {"type": "string", "pattern": "^[A-Za-z0-9]+$", "description": "File extension without leading dot"}
45
+ },
46
+ "template": "{platform}_{processing_level}_{wrs_path}{wrs_row}_{acq_date}_{proc_date}_{collection_number}_{tier}[.{extension}]",
47
+ "examples": [
48
+ "LC08_L1TP_190026_20200101_20200114_02_T1.tar"
49
+ ]
50
+ }
parseo/stac_http.py ADDED
@@ -0,0 +1,267 @@
1
+ """Helpers for querying STAC APIs via the standard library.
2
+
3
+ The routines in this module intentionally rely only on :mod:`urllib` from the
4
+ Python standard library to avoid pulling in heavier dependencies. For a
5
+ ``pystac-client`` based alternative see :mod:`parseo.stac_scraper`. All helper
6
+ functions require explicitly passing the ``base_url`` of the STAC service.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from collections.abc import Iterable
11
+ from functools import lru_cache
12
+ import itertools
13
+ import json
14
+ from pathlib import Path
15
+ import re
16
+ from string import Template
17
+ from typing import Union
18
+ import urllib.error
19
+ from urllib.parse import urljoin
20
+ from urllib.parse import urlparse
21
+ import urllib.request
22
+
23
+
24
+ class StacClientError(IOError):
25
+ """Raised when a STAC API request fails."""
26
+
27
+
28
+ def _norm_collection_id(collection_id: str, *, base_url: str) -> str:
29
+ """Resolve ``collection_id`` to the official ID from the STAC API."""
30
+
31
+ norm = re.sub(r"[^A-Za-z0-9]", "", collection_id).upper()
32
+ for cid in _list_collections_cached(base_url):
33
+ if re.sub(r"[^A-Za-z0-9]", "", cid).upper() == norm:
34
+ return cid
35
+ return collection_id
36
+
37
+
38
+ def _norm_base(base_url: str) -> str:
39
+ """Return ``base_url`` with exactly one trailing slash."""
40
+ return base_url.rstrip("/") + "/"
41
+
42
+
43
+ def _read_json(url: str) -> dict:
44
+ try:
45
+ with urllib.request.urlopen(url) as resp: # type: ignore[call-arg]
46
+ return json.load(resp)
47
+ except urllib.error.URLError as err:
48
+ raise StacClientError(f"Could not connect to {url}: {err.reason}") from err
49
+
50
+
51
+ def list_collections_http(base_url: str, *, deep: bool = False) -> list[str]:
52
+ """Return available collection IDs from the STAC API using ``urllib``.
53
+
54
+ This lightweight helper performs raw HTTP requests without requiring
55
+ third-party libraries. If ``deep`` is ``True`` the function follows
56
+ ``rel='child'`` links and gathers collection IDs from nested catalogs as
57
+ well. For a variant powered by ``pystac-client`` see
58
+ :func:`parseo.stac_scraper.list_collections_client`.
59
+ """
60
+ base = _norm_base(base_url)
61
+
62
+ # First, fetch the standard ``/collections`` endpoint which should expose
63
+ # top-level collections for STAC APIs.
64
+ url = urljoin(base, "collections")
65
+ try:
66
+ data = _read_json(url)
67
+ except urllib.error.HTTPError as err:
68
+ raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
69
+
70
+ collections = {c["id"] for c in data.get("collections", [])}
71
+
72
+ if not deep:
73
+ return sorted(collections)
74
+
75
+ # Breadth-first traversal of child links starting from the catalog root.
76
+ to_visit = [base]
77
+ visited: set[str] = set()
78
+
79
+ while to_visit:
80
+ cur = to_visit.pop()
81
+ if cur in visited:
82
+ continue
83
+ visited.add(cur)
84
+ try:
85
+ data = _read_json(cur)
86
+ except urllib.error.HTTPError as err:
87
+ raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
88
+
89
+ # Collect IDs if this document represents a collection or includes
90
+ # embedded collections.
91
+ if data.get("type") == "Collection":
92
+ cid = data.get("id")
93
+ if cid:
94
+ collections.add(cid)
95
+ for coll in data.get("collections", []):
96
+ cid = coll.get("id")
97
+ if cid:
98
+ collections.add(cid)
99
+
100
+ # Queue any child links for further traversal.
101
+ for link in data.get("links", []):
102
+ if link.get("rel") == "child":
103
+ href = link.get("href")
104
+ if href:
105
+ base_cur = cur if cur.endswith("/") else cur + "/"
106
+ to_visit.append(urljoin(base_cur, href))
107
+
108
+ return sorted(collections)
109
+
110
+
111
+ @lru_cache(maxsize=32)
112
+ def _list_collections_cached(base_url: str) -> tuple[str, ...]:
113
+ """Cached helper returning collection IDs for ``base_url``."""
114
+ # Use a deep listing to include collections nested in child catalogs.
115
+ # This ensures that `_norm_collection_id` can resolve aliases for
116
+ # collections not present at the top-level of the STAC service.
117
+ return tuple(list_collections_http(base_url, deep=True))
118
+
119
+
120
+ def iter_asset_filenames(
121
+ collection_id: str,
122
+ *,
123
+ base_url: str,
124
+ limit: int = 100,
125
+ asset_role: Union[str, None] = None,
126
+ ) -> Iterable[str]:
127
+ """Yield asset filenames from items of a collection.
128
+
129
+ Pagination links (``rel="next"``) are followed until all pages are
130
+ exhausted or ``limit`` filenames have been yielded. When ``asset_role`` is
131
+ provided, only assets declaring that role are considered. Resulting
132
+ filenames are sanitized: directory components are stripped and any
133
+ characters outside ``[A-Za-z0-9._-]`` are replaced with ``_``.
134
+ """
135
+ base = _norm_base(base_url)
136
+ collection_id = _norm_collection_id(collection_id, base_url=base)
137
+ url = urljoin(base, f"collections/{collection_id}/items?limit={limit}")
138
+ remaining = limit
139
+ first_request = True
140
+ while url and remaining > 0:
141
+ try:
142
+ data = _read_json(url)
143
+ except urllib.error.HTTPError as err:
144
+ if first_request and err.code == 404:
145
+ raise StacClientError(
146
+ f"Collection '{collection_id}' not found at {base}. "
147
+ "Use `parseo stac-sample <collection> --stac-url <url>` with a valid collection ID."
148
+ ) from err
149
+ raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
150
+ first_request = False
151
+ for feat in data.get("features", []):
152
+ props = feat.get("properties", {})
153
+ assets = feat.get("assets", {})
154
+ for asset in assets.values():
155
+ if asset_role and asset_role not in (asset.get("roles") or []):
156
+ continue
157
+ title = asset.get("title")
158
+ href = asset.get("href")
159
+ filename = None
160
+ # Many assets use a generic title like "Product" which does not
161
+ # convey the actual filename. In such cases prefer extracting
162
+ # the name from the href. Only fall back to the title if it is
163
+ # present and not the generic "Product".
164
+ if title and title.strip().lower() != "product":
165
+ filename = title
166
+ elif href:
167
+ if "$" in href:
168
+ href_sub = Template(href).safe_substitute(props)
169
+ if re.search(r"\$(?!value\b)\w+", href_sub):
170
+ continue
171
+ href = href_sub
172
+ m = re.search(r"Products\('([^']+)'\)", href)
173
+ if m:
174
+ filename = m.group(1)
175
+ else:
176
+ parsed_url = urlparse(href)
177
+ path = Path(parsed_url.path)
178
+ filename = path.name
179
+ if not path.suffix:
180
+ filename += ".dat"
181
+ else:
182
+ continue
183
+ if filename.startswith("$"):
184
+ continue
185
+ filename = Path(filename).name
186
+ filename = re.sub(r"[^A-Za-z0-9._-]", "_", filename)
187
+ yield filename
188
+ remaining -= 1
189
+ if remaining == 0:
190
+ return
191
+ url = None
192
+ for link in data.get("links", []):
193
+ if link.get("rel") == "next":
194
+ url = link.get("href")
195
+ break
196
+
197
+ def iter_collection_tree(
198
+ collection_id: str,
199
+ *,
200
+ base_url: str,
201
+ limit: int = 100,
202
+ asset_role: Union[str, None] = None,
203
+ ) -> Iterable[tuple[str, str]]:
204
+ """Yield ``(collection_id, filename)`` pairs for all leaf collections.
205
+
206
+ The function follows ``links`` with ``rel="child"`` starting from
207
+ ``collection_id`` until it reaches leaf collections. For each leaf it
208
+ yields filenames from :func:`iter_asset_filenames`. The ``asset_role``
209
+ parameter is forwarded to :func:`iter_asset_filenames`.
210
+ """
211
+ base = _norm_base(base_url)
212
+ collection_id = _norm_collection_id(collection_id, base_url=base)
213
+ url = urljoin(base, f"collections/{collection_id}")
214
+ try:
215
+ data = _read_json(url)
216
+ except urllib.error.HTTPError as err:
217
+ if err.code == 404:
218
+ raise StacClientError(
219
+ f"Collection '{collection_id}' not found at {base}. "
220
+ "Use `parseo stac-sample <collection> --stac-url <url>` with a valid collection ID."
221
+ ) from err
222
+ raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
223
+
224
+ children = [
225
+ link.get("href", "").rstrip("/").split("/")[-1]
226
+ for link in data.get("links", [])
227
+ if link.get("rel") == "child"
228
+ ]
229
+
230
+ if children:
231
+ for child in children:
232
+ yield from iter_collection_tree(
233
+ child, base_url=base, limit=limit, asset_role=asset_role
234
+ )
235
+ else:
236
+ for fn in itertools.islice(
237
+ iter_asset_filenames(
238
+ collection_id, base_url=base, limit=limit, asset_role=asset_role
239
+ ),
240
+ limit,
241
+ ):
242
+ yield collection_id, fn
243
+
244
+
245
+ def sample_collection_filenames(
246
+ collection_id: str,
247
+ samples: int = 5,
248
+ *,
249
+ base_url: str,
250
+ asset_role: Union[str, None] = None,
251
+ ) -> dict[str, list[str]]:
252
+ """Return ``samples`` filenames for each leaf collection.
253
+
254
+ ``collection_id`` may be the official STAC ID or any case/format variant
255
+ resolvable via :func:`list_collections_http`. When ``collection_id`` has
256
+ child collections, a sample is collected from each leaf. Only assets whose
257
+ ``roles`` include ``asset_role`` are returned when the parameter is
258
+ supplied.
259
+ """
260
+ out: dict[str, list[str]] = {}
261
+ for cid, fn in iter_collection_tree(
262
+ collection_id, base_url=base_url, limit=samples, asset_role=asset_role
263
+ ):
264
+ lst = out.setdefault(cid, [])
265
+ if len(lst) < samples:
266
+ lst.append(fn)
267
+ return out
parseo/stac_scraper.py ADDED
@@ -0,0 +1,115 @@
1
+ """STAC helpers backed by ``pystac-client``.
2
+
3
+ This module mirrors the utilities in :mod:`parseo.stac_http` but relies on
4
+ ``pystac-client`` for STAC catalog traversal. Use these helpers when the extra
5
+ features of ``pystac-client`` are required. For a lightweight alternative that
6
+ only depends on the Python standard library see :mod:`parseo.stac_http`.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from pathlib import Path
11
+ from typing import Union
12
+ from urllib.parse import urlparse
13
+
14
+ def list_collections_client(base_url: str, *, deep: bool = False) -> list[str]:
15
+ """Return collection IDs from a STAC API using ``pystac-client``.
16
+
17
+ Parameters mirror :func:`parseo.stac_http.list_collections_http` but
18
+ this variant requires the optional ``pystac-client`` dependency. It is
19
+ suitable when more advanced STAC handling is needed, at the cost of pulling
20
+ in the external library.
21
+
22
+ Raises
23
+ ------
24
+ ImportError
25
+ If ``pystac-client`` is not installed.
26
+ """
27
+ try:
28
+ from pystac_client import Client
29
+ except Exception as exc: # pragma: no cover - exercised when dependency missing
30
+ raise ImportError(
31
+ "pystac-client is required for list_collections_client"
32
+ ) from exc
33
+
34
+ client = Client.open(base_url)
35
+ collections = {c.id for c in client.get_collections()}
36
+ if not deep:
37
+ return sorted(collections)
38
+
39
+ # Breadth-first traversal of child catalogs.
40
+ to_visit = list(client.get_children())
41
+ visited: set[str] = set()
42
+ while to_visit:
43
+ child = to_visit.pop()
44
+ href = getattr(child, "href", None) or getattr(child, "target", None)
45
+ if not href or href in visited:
46
+ continue
47
+ visited.add(href)
48
+ sub_client = Client.open(href)
49
+ collections.update(c.id for c in sub_client.get_collections())
50
+ to_visit.extend(sub_client.get_children())
51
+
52
+ return sorted(collections)
53
+
54
+
55
+ def search_stac_and_download(
56
+ *,
57
+ stac_url: str,
58
+ collections: list[str],
59
+ bbox: Union[list[float], tuple[float, float, float, float]],
60
+ datetime: str,
61
+ dest_dir: Union[str, Path],
62
+ ) -> Path:
63
+ """Download the first asset matching a STAC search.
64
+
65
+ The search is performed via :mod:`pystac-client` and the asset is retrieved
66
+ with :mod:`requests`. ``dest_dir`` is created if needed and the path to the
67
+ downloaded file is returned.
68
+
69
+ Raises
70
+ ------
71
+ ImportError
72
+ If ``pystac-client`` or ``requests`` is not installed.
73
+ FileNotFoundError
74
+ If the STAC search yields no downloadable assets or all downloads
75
+ fail.
76
+ """
77
+
78
+ try:
79
+ from pystac_client import Client
80
+ except Exception as exc: # pragma: no cover - exercised when dependency missing
81
+ raise ImportError(
82
+ "pystac-client is required for search_stac_and_download"
83
+ ) from exc
84
+
85
+ try:
86
+ import requests
87
+ except Exception as exc: # pragma: no cover - exercised when dependency missing
88
+ raise ImportError(
89
+ "requests is required for search_stac_and_download"
90
+ ) from exc
91
+
92
+ client = Client.open(stac_url)
93
+ search = client.search(collections=collections, bbox=bbox, datetime=datetime)
94
+ for item in search.items():
95
+ for asset in item.assets.values():
96
+ href = getattr(asset, "href", None)
97
+ if not href:
98
+ continue
99
+ name = getattr(asset, "title", None)
100
+ if not name:
101
+ name = Path(urlparse(href).path).name
102
+ dest_dir_path = Path(dest_dir)
103
+ dest_dir_path.mkdir(parents=True, exist_ok=True)
104
+ dest_path = dest_dir_path / name
105
+ try:
106
+ with requests.get(href, stream=True) as resp:
107
+ resp.raise_for_status()
108
+ with open(dest_path, "wb") as fh:
109
+ for chunk in resp.iter_content(chunk_size=8192):
110
+ if chunk:
111
+ fh.write(chunk)
112
+ return dest_path
113
+ except requests.HTTPError:
114
+ continue
115
+ raise FileNotFoundError("No matching assets found")