parseo 0.4.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parseo/__init__.py +53 -0
- parseo/_epsg_lookup.py +145 -0
- parseo/_field_mappings.py +219 -0
- parseo/_json.py +14 -0
- parseo/_tile_systems.py +51 -0
- parseo/assembler.py +225 -0
- parseo/cli.py +383 -0
- parseo/parser.py +784 -0
- parseo/schema_registry.py +246 -0
- parseo/schemas/copernicus/clms/clc/clc_filename_v1_0_0.json +48 -0
- parseo/schemas/copernicus/clms/clc/clc_filename_v1_1_0.json +65 -0
- parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_0.json +90 -0
- parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_1.json +80 -0
- parseo/schemas/copernicus/clms/egms/egms_gnss_model_filename_v0_0_0.json +46 -0
- parseo/schemas/copernicus/clms/egms/egms_l2a_product_filename_v0_0_0.json +71 -0
- parseo/schemas/copernicus/clms/egms/egms_l3_velocity_grid_filename_v0_0_0.json +64 -0
- parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_0.json +115 -0
- parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_1.json +103 -0
- parseo/schemas/copernicus/clms/hr-vpp/st_filename_v0_0_0.json +82 -0
- parseo/schemas/copernicus/clms/hr-vpp/vi_filename_v0_0_0.json +89 -0
- parseo/schemas/copernicus/clms/hr-vpp/vpp_filename_v0_0_0.json +109 -0
- parseo/schemas/copernicus/clms/hr-wsi/cc_filename_v0_0_0.json +84 -0
- parseo/schemas/copernicus/clms/hr-wsi/fsc_filename_v0_0_0.json +88 -0
- parseo/schemas/copernicus/clms/hr-wsi/gfsc_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/icd_filename_v0_0_0.json +96 -0
- parseo/schemas/copernicus/clms/hr-wsi/sp_filename_v0_0_0.json +177 -0
- parseo/schemas/copernicus/clms/hr-wsi/sws_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/wcd_filename_v0_0_0.json +96 -0
- parseo/schemas/copernicus/clms/hr-wsi/wds_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/wic_comb_filename_v0_0_0.json +82 -0
- parseo/schemas/copernicus/clms/hr-wsi/wic_filename_v0_0_0.json +90 -0
- parseo/schemas/copernicus/clms/hrl/fty_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/gra_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/ibu_filename_v0_0_0.json +57 -0
- parseo/schemas/copernicus/clms/hrl/imp_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_0.json +73 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_1.json +89 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_2.json +89 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_1.json +49 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_2.json +93 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_3.json +99 -0
- parseo/schemas/copernicus/clms/hrl/tcd_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/vlcc_filename_v0_0_0.json +105 -0
- parseo/schemas/copernicus/clms/hrl/waw_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/n2k/n2k_filename_v1_0_0.json +43 -0
- parseo/schemas/copernicus/clms/pa/pa_filename_v1_0_0.json +99 -0
- parseo/schemas/copernicus/clms/riparian-zones/rpz_filename_v0_0_0.json +50 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_dhm_filename_v0_0_0.json +53 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_filename_v0_0_0.json +105 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_stl_filename_v0_0_0.json +89 -0
- parseo/schemas/copernicus/sentinel/s1_filename_v1_0_0.json +78 -0
- parseo/schemas/copernicus/sentinel/s2_filename_v1_0_0.json +62 -0
- parseo/schemas/copernicus/sentinel/s3_filename_v1_0_0.json +22 -0
- parseo/schemas/copernicus/sentinel/s4_filename_v1_0_0.json +21 -0
- parseo/schemas/copernicus/sentinel/s5p_filename_v1_0_0.json +21 -0
- parseo/schemas/copernicus/sentinel/s6_filename_v1_0_0.json +23 -0
- parseo/schemas/eumetsat/metop_filename_v1_0_0.json +68 -0
- parseo/schemas/eumetsat/mtg_filename_v1_0_0.json +65 -0
- parseo/schemas/nasa/modis_filename_v1_0_0.json +54 -0
- parseo/schemas/usgs/landsat/landsat_filename_v1_0_0.json +50 -0
- parseo/stac_http.py +267 -0
- parseo/stac_scraper.py +115 -0
- parseo/template.py +78 -0
- parseo-0.4.4.dist-info/METADATA +402 -0
- parseo-0.4.4.dist-info/RECORD +70 -0
- parseo-0.4.4.dist-info/WHEEL +5 -0
- parseo-0.4.4.dist-info/entry_points.txt +2 -0
- parseo-0.4.4.dist-info/licenses/LICENSE.txt +287 -0
- parseo-0.4.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"schema_id": "eumetsat:metop",
|
|
4
|
+
"schema_version": "1.0.0",
|
|
5
|
+
"status": "current",
|
|
6
|
+
"description": "EUMETSAT Metop product filename (extension optional).",
|
|
7
|
+
"fields": {
|
|
8
|
+
"platform": {
|
|
9
|
+
"type": "string",
|
|
10
|
+
"enum": [
|
|
11
|
+
"M01",
|
|
12
|
+
"M02",
|
|
13
|
+
"M03"
|
|
14
|
+
],
|
|
15
|
+
"description": "Spacecraft unit"
|
|
16
|
+
},
|
|
17
|
+
"instrument": {
|
|
18
|
+
"type": "string",
|
|
19
|
+
"enum": [
|
|
20
|
+
"AVHR",
|
|
21
|
+
"IASI",
|
|
22
|
+
"ASCAT"
|
|
23
|
+
],
|
|
24
|
+
"description": "Instrument short name"
|
|
25
|
+
},
|
|
26
|
+
"processing_level": {
|
|
27
|
+
"type": "string",
|
|
28
|
+
"enum": [
|
|
29
|
+
"L1B",
|
|
30
|
+
"L2"
|
|
31
|
+
],
|
|
32
|
+
"description": "Processing level"
|
|
33
|
+
},
|
|
34
|
+
"start_datetime": {
|
|
35
|
+
"type": "string",
|
|
36
|
+
"pattern": "^\\d{8}T\\d{6}Z$",
|
|
37
|
+
"description": "Acquisition start time (UTC, YYYYMMDDTHHMMSSZ)"
|
|
38
|
+
},
|
|
39
|
+
"end_datetime": {
|
|
40
|
+
"type": "string",
|
|
41
|
+
"pattern": "^\\d{8}T\\d{6}Z$",
|
|
42
|
+
"description": "Acquisition end time (UTC, YYYYMMDDTHHMMSSZ)"
|
|
43
|
+
},
|
|
44
|
+
"orbit_number": {
|
|
45
|
+
"type": "string",
|
|
46
|
+
"pattern": "^O\\d{6}$",
|
|
47
|
+
"description": "Orbit number"
|
|
48
|
+
},
|
|
49
|
+
"collection": {
|
|
50
|
+
"type": "string",
|
|
51
|
+
"pattern": "^C\\d{2}$",
|
|
52
|
+
"description": "Collection identifier"
|
|
53
|
+
},
|
|
54
|
+
"version": {
|
|
55
|
+
"type": "string",
|
|
56
|
+
"pattern": "^\\d{2}$",
|
|
57
|
+
"description": "Product version"
|
|
58
|
+
},
|
|
59
|
+
"extension": {
|
|
60
|
+
"type": "string",
|
|
61
|
+
"enum": [
|
|
62
|
+
"nc"
|
|
63
|
+
],
|
|
64
|
+
"description": "File extension without leading dot"
|
|
65
|
+
}
|
|
66
|
+
},
|
|
67
|
+
"template": "{platform}_{instrument}_{processing_level}_{start_datetime}_{end_datetime}_{orbit_number}_{collection}_V{version}[.{extension}]"
|
|
68
|
+
}
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"schema_id": "eumetsat:mtg",
|
|
4
|
+
"schema_version": "1.0.0",
|
|
5
|
+
"status": "current",
|
|
6
|
+
"description": "EUMETSAT MTG product filename (extension optional).",
|
|
7
|
+
"fields": {
|
|
8
|
+
"platform": {
|
|
9
|
+
"type": "string",
|
|
10
|
+
"enum": [
|
|
11
|
+
"MTG-I1",
|
|
12
|
+
"MTG-I2",
|
|
13
|
+
"MTG-S1",
|
|
14
|
+
"MTG-S2"
|
|
15
|
+
],
|
|
16
|
+
"description": "Spacecraft unit"
|
|
17
|
+
},
|
|
18
|
+
"instrument": {
|
|
19
|
+
"type": "string",
|
|
20
|
+
"enum": [
|
|
21
|
+
"FCI",
|
|
22
|
+
"LI",
|
|
23
|
+
"IRS",
|
|
24
|
+
"UVN"
|
|
25
|
+
],
|
|
26
|
+
"description": "Instrument short name"
|
|
27
|
+
},
|
|
28
|
+
"product_type": {
|
|
29
|
+
"type": "string",
|
|
30
|
+
"pattern": "^[A-Z0-9_]{2,10}$",
|
|
31
|
+
"description": "Product short name"
|
|
32
|
+
},
|
|
33
|
+
"processing_level": {
|
|
34
|
+
"type": "string",
|
|
35
|
+
"enum": [
|
|
36
|
+
"L1",
|
|
37
|
+
"L2"
|
|
38
|
+
],
|
|
39
|
+
"description": "Processing level"
|
|
40
|
+
},
|
|
41
|
+
"start_datetime": {
|
|
42
|
+
"type": "string",
|
|
43
|
+
"pattern": "^\\d{8}T\\d{6}Z$",
|
|
44
|
+
"description": "Acquisition start time (UTC, YYYYMMDDTHHMMSSZ)"
|
|
45
|
+
},
|
|
46
|
+
"end_datetime": {
|
|
47
|
+
"type": "string",
|
|
48
|
+
"pattern": "^\\d{8}T\\d{6}Z$",
|
|
49
|
+
"description": "Acquisition end time (UTC, YYYYMMDDTHHMMSSZ)"
|
|
50
|
+
},
|
|
51
|
+
"version": {
|
|
52
|
+
"type": "string",
|
|
53
|
+
"pattern": "^\\d{2}$",
|
|
54
|
+
"description": "Product version"
|
|
55
|
+
},
|
|
56
|
+
"extension": {
|
|
57
|
+
"type": "string",
|
|
58
|
+
"enum": [
|
|
59
|
+
"nc"
|
|
60
|
+
],
|
|
61
|
+
"description": "File extension without leading dot"
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
"template": "{platform}_{instrument}_{product_type}_{processing_level}_{start_datetime}_{end_datetime}_V{version}[.{extension}]"
|
|
65
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_id": "nasa:modis",
|
|
3
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
4
|
+
"schema_version": "1.0.0",
|
|
5
|
+
"status": "current",
|
|
6
|
+
"stac_version": "1.1.0",
|
|
7
|
+
|
|
8
|
+
"description": "MODIS product filename (extension optional).",
|
|
9
|
+
"fields": {
|
|
10
|
+
"platform": {
|
|
11
|
+
"type": "string",
|
|
12
|
+
"enum": ["MOD","MYD","MCD"],
|
|
13
|
+
"description": "Spacecraft platform (MOD=Terra, MYD=Aqua, MCD=Combined)",
|
|
14
|
+
"stac_map": {
|
|
15
|
+
"MOD": {
|
|
16
|
+
"platform": "Terra",
|
|
17
|
+
"instruments": ["MODIS"]
|
|
18
|
+
},
|
|
19
|
+
"MYD": {
|
|
20
|
+
"platform": "Aqua",
|
|
21
|
+
"instruments": ["MODIS"]
|
|
22
|
+
},
|
|
23
|
+
"MCD": {
|
|
24
|
+
"platform": "Combined",
|
|
25
|
+
"instruments": ["MODIS"]
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"product": {
|
|
30
|
+
"type": "string",
|
|
31
|
+
"pattern": "^\\d{2}$",
|
|
32
|
+
"description": "Two-digit product code (e.g., '09' for Surface Reflectance)"
|
|
33
|
+
},
|
|
34
|
+
"variant": {
|
|
35
|
+
"type": "string",
|
|
36
|
+
"pattern": "^[A-Za-z0-9]+$",
|
|
37
|
+
"description": "Product variant or sub-type (e.g., GA, A1)"
|
|
38
|
+
},
|
|
39
|
+
"acq_date": {"type": "string", "pattern": "^\\d{7}$", "description": "Acquisition date (YYYYDDD)"},
|
|
40
|
+
"tile_id": {"type": "string", "pattern": "^h\\d{2}v\\d{2}$", "description": "MODIS sinusoidal tile"},
|
|
41
|
+
"collection": {"type": "string", "pattern": "^\\d{3}$", "description": "Collection number"},
|
|
42
|
+
"proc_date": {"type": "string", "pattern": "^\\d{13}$", "description": "Production date (YYYYDDDHHMMSS)"},
|
|
43
|
+
"extension": {"type": "string", "pattern": "^[A-Za-z0-9]+$", "description": "File extension."}
|
|
44
|
+
},
|
|
45
|
+
"template": "{platform}{product}{variant}.A{acq_date}.{tile_id}.{collection}.{proc_date}[.{extension}]",
|
|
46
|
+
"examples": [
|
|
47
|
+
"MOD09GA.A2021123.h18v04.006.2021132234506.hdf",
|
|
48
|
+
"MYD11A1.A2021001.h21v09.061.2021015190034"
|
|
49
|
+
],
|
|
50
|
+
"counter_examples": [
|
|
51
|
+
"MOD09GA.X2021123.h18v04.006.2021132234506.hdf",
|
|
52
|
+
"MOD09GA.A2021123.H18V04.006.2021132234506.hdf"
|
|
53
|
+
]
|
|
54
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"schema_id": "usgs:landsat:landsat",
|
|
4
|
+
"schema_version": "1.0.0",
|
|
5
|
+
"status": "current",
|
|
6
|
+
"stac_version": "1.1.0",
|
|
7
|
+
|
|
8
|
+
"description": "Landsat Collection 2 scene identifier (extension optional).",
|
|
9
|
+
"fields": {
|
|
10
|
+
"platform": {
|
|
11
|
+
"type": "string",
|
|
12
|
+
"pattern": "^(LT04|LT05|LE07|LC08|LC09)$",
|
|
13
|
+
"description": "Platform and sensor code",
|
|
14
|
+
"stac_map": {
|
|
15
|
+
"LT04": {
|
|
16
|
+
"platform": "landsat-4",
|
|
17
|
+
"instruments": ["TM"]
|
|
18
|
+
},
|
|
19
|
+
"LT05": {
|
|
20
|
+
"platform": "landsat-5",
|
|
21
|
+
"instruments": ["TM"]
|
|
22
|
+
},
|
|
23
|
+
"LE07": {
|
|
24
|
+
"platform": "landsat-7",
|
|
25
|
+
"instruments": ["ETM+"]
|
|
26
|
+
},
|
|
27
|
+
"LC08": {
|
|
28
|
+
"platform": "landsat-8",
|
|
29
|
+
"instruments": ["OLI_TIRS"]
|
|
30
|
+
},
|
|
31
|
+
"LC09": {
|
|
32
|
+
"platform": "landsat-9",
|
|
33
|
+
"instruments": ["OLI_TIRS"]
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
},
|
|
37
|
+
"processing_level": {"type": "string", "enum": ["L1TP", "L1GT", "L1GS", "L2SP", "L2SR"], "description": "Processing level"},
|
|
38
|
+
"wrs_path": {"type": "string", "pattern": "^\\d{3}$", "description": "WRS-2 path"},
|
|
39
|
+
"wrs_row": {"type": "string", "pattern": "^\\d{3}$", "description": "WRS-2 row"},
|
|
40
|
+
"acq_date": {"type": "string", "pattern": "^\\d{8}$", "description": "Acquisition date (YYYYMMDD)"},
|
|
41
|
+
"proc_date": {"type": "string", "pattern": "^\\d{8}$", "description": "Processing date (YYYYMMDD)"},
|
|
42
|
+
"collection_number": {"type": "string", "pattern": "^\\d{2}$", "description": "Collection number"},
|
|
43
|
+
"tier": {"type": "string", "enum": ["T1", "T2", "RT"], "description": "Tier classification"},
|
|
44
|
+
"extension": {"type": "string", "pattern": "^[A-Za-z0-9]+$", "description": "File extension without leading dot"}
|
|
45
|
+
},
|
|
46
|
+
"template": "{platform}_{processing_level}_{wrs_path}{wrs_row}_{acq_date}_{proc_date}_{collection_number}_{tier}[.{extension}]",
|
|
47
|
+
"examples": [
|
|
48
|
+
"LC08_L1TP_190026_20200101_20200114_02_T1.tar"
|
|
49
|
+
]
|
|
50
|
+
}
|
parseo/stac_http.py
ADDED
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
"""Helpers for querying STAC APIs via the standard library.
|
|
2
|
+
|
|
3
|
+
The routines in this module intentionally rely only on :mod:`urllib` from the
|
|
4
|
+
Python standard library to avoid pulling in heavier dependencies. For a
|
|
5
|
+
``pystac-client`` based alternative see :mod:`parseo.stac_scraper`. All helper
|
|
6
|
+
functions require explicitly passing the ``base_url`` of the STAC service.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections.abc import Iterable
|
|
11
|
+
from functools import lru_cache
|
|
12
|
+
import itertools
|
|
13
|
+
import json
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
import re
|
|
16
|
+
from string import Template
|
|
17
|
+
from typing import Union
|
|
18
|
+
import urllib.error
|
|
19
|
+
from urllib.parse import urljoin
|
|
20
|
+
from urllib.parse import urlparse
|
|
21
|
+
import urllib.request
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class StacClientError(IOError):
|
|
25
|
+
"""Raised when a STAC API request fails."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _norm_collection_id(collection_id: str, *, base_url: str) -> str:
|
|
29
|
+
"""Resolve ``collection_id`` to the official ID from the STAC API."""
|
|
30
|
+
|
|
31
|
+
norm = re.sub(r"[^A-Za-z0-9]", "", collection_id).upper()
|
|
32
|
+
for cid in _list_collections_cached(base_url):
|
|
33
|
+
if re.sub(r"[^A-Za-z0-9]", "", cid).upper() == norm:
|
|
34
|
+
return cid
|
|
35
|
+
return collection_id
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _norm_base(base_url: str) -> str:
|
|
39
|
+
"""Return ``base_url`` with exactly one trailing slash."""
|
|
40
|
+
return base_url.rstrip("/") + "/"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _read_json(url: str) -> dict:
|
|
44
|
+
try:
|
|
45
|
+
with urllib.request.urlopen(url) as resp: # type: ignore[call-arg]
|
|
46
|
+
return json.load(resp)
|
|
47
|
+
except urllib.error.URLError as err:
|
|
48
|
+
raise StacClientError(f"Could not connect to {url}: {err.reason}") from err
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def list_collections_http(base_url: str, *, deep: bool = False) -> list[str]:
|
|
52
|
+
"""Return available collection IDs from the STAC API using ``urllib``.
|
|
53
|
+
|
|
54
|
+
This lightweight helper performs raw HTTP requests without requiring
|
|
55
|
+
third-party libraries. If ``deep`` is ``True`` the function follows
|
|
56
|
+
``rel='child'`` links and gathers collection IDs from nested catalogs as
|
|
57
|
+
well. For a variant powered by ``pystac-client`` see
|
|
58
|
+
:func:`parseo.stac_scraper.list_collections_client`.
|
|
59
|
+
"""
|
|
60
|
+
base = _norm_base(base_url)
|
|
61
|
+
|
|
62
|
+
# First, fetch the standard ``/collections`` endpoint which should expose
|
|
63
|
+
# top-level collections for STAC APIs.
|
|
64
|
+
url = urljoin(base, "collections")
|
|
65
|
+
try:
|
|
66
|
+
data = _read_json(url)
|
|
67
|
+
except urllib.error.HTTPError as err:
|
|
68
|
+
raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
|
|
69
|
+
|
|
70
|
+
collections = {c["id"] for c in data.get("collections", [])}
|
|
71
|
+
|
|
72
|
+
if not deep:
|
|
73
|
+
return sorted(collections)
|
|
74
|
+
|
|
75
|
+
# Breadth-first traversal of child links starting from the catalog root.
|
|
76
|
+
to_visit = [base]
|
|
77
|
+
visited: set[str] = set()
|
|
78
|
+
|
|
79
|
+
while to_visit:
|
|
80
|
+
cur = to_visit.pop()
|
|
81
|
+
if cur in visited:
|
|
82
|
+
continue
|
|
83
|
+
visited.add(cur)
|
|
84
|
+
try:
|
|
85
|
+
data = _read_json(cur)
|
|
86
|
+
except urllib.error.HTTPError as err:
|
|
87
|
+
raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
|
|
88
|
+
|
|
89
|
+
# Collect IDs if this document represents a collection or includes
|
|
90
|
+
# embedded collections.
|
|
91
|
+
if data.get("type") == "Collection":
|
|
92
|
+
cid = data.get("id")
|
|
93
|
+
if cid:
|
|
94
|
+
collections.add(cid)
|
|
95
|
+
for coll in data.get("collections", []):
|
|
96
|
+
cid = coll.get("id")
|
|
97
|
+
if cid:
|
|
98
|
+
collections.add(cid)
|
|
99
|
+
|
|
100
|
+
# Queue any child links for further traversal.
|
|
101
|
+
for link in data.get("links", []):
|
|
102
|
+
if link.get("rel") == "child":
|
|
103
|
+
href = link.get("href")
|
|
104
|
+
if href:
|
|
105
|
+
base_cur = cur if cur.endswith("/") else cur + "/"
|
|
106
|
+
to_visit.append(urljoin(base_cur, href))
|
|
107
|
+
|
|
108
|
+
return sorted(collections)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@lru_cache(maxsize=32)
|
|
112
|
+
def _list_collections_cached(base_url: str) -> tuple[str, ...]:
|
|
113
|
+
"""Cached helper returning collection IDs for ``base_url``."""
|
|
114
|
+
# Use a deep listing to include collections nested in child catalogs.
|
|
115
|
+
# This ensures that `_norm_collection_id` can resolve aliases for
|
|
116
|
+
# collections not present at the top-level of the STAC service.
|
|
117
|
+
return tuple(list_collections_http(base_url, deep=True))
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def iter_asset_filenames(
|
|
121
|
+
collection_id: str,
|
|
122
|
+
*,
|
|
123
|
+
base_url: str,
|
|
124
|
+
limit: int = 100,
|
|
125
|
+
asset_role: Union[str, None] = None,
|
|
126
|
+
) -> Iterable[str]:
|
|
127
|
+
"""Yield asset filenames from items of a collection.
|
|
128
|
+
|
|
129
|
+
Pagination links (``rel="next"``) are followed until all pages are
|
|
130
|
+
exhausted or ``limit`` filenames have been yielded. When ``asset_role`` is
|
|
131
|
+
provided, only assets declaring that role are considered. Resulting
|
|
132
|
+
filenames are sanitized: directory components are stripped and any
|
|
133
|
+
characters outside ``[A-Za-z0-9._-]`` are replaced with ``_``.
|
|
134
|
+
"""
|
|
135
|
+
base = _norm_base(base_url)
|
|
136
|
+
collection_id = _norm_collection_id(collection_id, base_url=base)
|
|
137
|
+
url = urljoin(base, f"collections/{collection_id}/items?limit={limit}")
|
|
138
|
+
remaining = limit
|
|
139
|
+
first_request = True
|
|
140
|
+
while url and remaining > 0:
|
|
141
|
+
try:
|
|
142
|
+
data = _read_json(url)
|
|
143
|
+
except urllib.error.HTTPError as err:
|
|
144
|
+
if first_request and err.code == 404:
|
|
145
|
+
raise StacClientError(
|
|
146
|
+
f"Collection '{collection_id}' not found at {base}. "
|
|
147
|
+
"Use `parseo stac-sample <collection> --stac-url <url>` with a valid collection ID."
|
|
148
|
+
) from err
|
|
149
|
+
raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
|
|
150
|
+
first_request = False
|
|
151
|
+
for feat in data.get("features", []):
|
|
152
|
+
props = feat.get("properties", {})
|
|
153
|
+
assets = feat.get("assets", {})
|
|
154
|
+
for asset in assets.values():
|
|
155
|
+
if asset_role and asset_role not in (asset.get("roles") or []):
|
|
156
|
+
continue
|
|
157
|
+
title = asset.get("title")
|
|
158
|
+
href = asset.get("href")
|
|
159
|
+
filename = None
|
|
160
|
+
# Many assets use a generic title like "Product" which does not
|
|
161
|
+
# convey the actual filename. In such cases prefer extracting
|
|
162
|
+
# the name from the href. Only fall back to the title if it is
|
|
163
|
+
# present and not the generic "Product".
|
|
164
|
+
if title and title.strip().lower() != "product":
|
|
165
|
+
filename = title
|
|
166
|
+
elif href:
|
|
167
|
+
if "$" in href:
|
|
168
|
+
href_sub = Template(href).safe_substitute(props)
|
|
169
|
+
if re.search(r"\$(?!value\b)\w+", href_sub):
|
|
170
|
+
continue
|
|
171
|
+
href = href_sub
|
|
172
|
+
m = re.search(r"Products\('([^']+)'\)", href)
|
|
173
|
+
if m:
|
|
174
|
+
filename = m.group(1)
|
|
175
|
+
else:
|
|
176
|
+
parsed_url = urlparse(href)
|
|
177
|
+
path = Path(parsed_url.path)
|
|
178
|
+
filename = path.name
|
|
179
|
+
if not path.suffix:
|
|
180
|
+
filename += ".dat"
|
|
181
|
+
else:
|
|
182
|
+
continue
|
|
183
|
+
if filename.startswith("$"):
|
|
184
|
+
continue
|
|
185
|
+
filename = Path(filename).name
|
|
186
|
+
filename = re.sub(r"[^A-Za-z0-9._-]", "_", filename)
|
|
187
|
+
yield filename
|
|
188
|
+
remaining -= 1
|
|
189
|
+
if remaining == 0:
|
|
190
|
+
return
|
|
191
|
+
url = None
|
|
192
|
+
for link in data.get("links", []):
|
|
193
|
+
if link.get("rel") == "next":
|
|
194
|
+
url = link.get("href")
|
|
195
|
+
break
|
|
196
|
+
|
|
197
|
+
def iter_collection_tree(
|
|
198
|
+
collection_id: str,
|
|
199
|
+
*,
|
|
200
|
+
base_url: str,
|
|
201
|
+
limit: int = 100,
|
|
202
|
+
asset_role: Union[str, None] = None,
|
|
203
|
+
) -> Iterable[tuple[str, str]]:
|
|
204
|
+
"""Yield ``(collection_id, filename)`` pairs for all leaf collections.
|
|
205
|
+
|
|
206
|
+
The function follows ``links`` with ``rel="child"`` starting from
|
|
207
|
+
``collection_id`` until it reaches leaf collections. For each leaf it
|
|
208
|
+
yields filenames from :func:`iter_asset_filenames`. The ``asset_role``
|
|
209
|
+
parameter is forwarded to :func:`iter_asset_filenames`.
|
|
210
|
+
"""
|
|
211
|
+
base = _norm_base(base_url)
|
|
212
|
+
collection_id = _norm_collection_id(collection_id, base_url=base)
|
|
213
|
+
url = urljoin(base, f"collections/{collection_id}")
|
|
214
|
+
try:
|
|
215
|
+
data = _read_json(url)
|
|
216
|
+
except urllib.error.HTTPError as err:
|
|
217
|
+
if err.code == 404:
|
|
218
|
+
raise StacClientError(
|
|
219
|
+
f"Collection '{collection_id}' not found at {base}. "
|
|
220
|
+
"Use `parseo stac-sample <collection> --stac-url <url>` with a valid collection ID."
|
|
221
|
+
) from err
|
|
222
|
+
raise StacClientError(f"HTTP error {err.code} for {err.geturl()}") from err
|
|
223
|
+
|
|
224
|
+
children = [
|
|
225
|
+
link.get("href", "").rstrip("/").split("/")[-1]
|
|
226
|
+
for link in data.get("links", [])
|
|
227
|
+
if link.get("rel") == "child"
|
|
228
|
+
]
|
|
229
|
+
|
|
230
|
+
if children:
|
|
231
|
+
for child in children:
|
|
232
|
+
yield from iter_collection_tree(
|
|
233
|
+
child, base_url=base, limit=limit, asset_role=asset_role
|
|
234
|
+
)
|
|
235
|
+
else:
|
|
236
|
+
for fn in itertools.islice(
|
|
237
|
+
iter_asset_filenames(
|
|
238
|
+
collection_id, base_url=base, limit=limit, asset_role=asset_role
|
|
239
|
+
),
|
|
240
|
+
limit,
|
|
241
|
+
):
|
|
242
|
+
yield collection_id, fn
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def sample_collection_filenames(
|
|
246
|
+
collection_id: str,
|
|
247
|
+
samples: int = 5,
|
|
248
|
+
*,
|
|
249
|
+
base_url: str,
|
|
250
|
+
asset_role: Union[str, None] = None,
|
|
251
|
+
) -> dict[str, list[str]]:
|
|
252
|
+
"""Return ``samples`` filenames for each leaf collection.
|
|
253
|
+
|
|
254
|
+
``collection_id`` may be the official STAC ID or any case/format variant
|
|
255
|
+
resolvable via :func:`list_collections_http`. When ``collection_id`` has
|
|
256
|
+
child collections, a sample is collected from each leaf. Only assets whose
|
|
257
|
+
``roles`` include ``asset_role`` are returned when the parameter is
|
|
258
|
+
supplied.
|
|
259
|
+
"""
|
|
260
|
+
out: dict[str, list[str]] = {}
|
|
261
|
+
for cid, fn in iter_collection_tree(
|
|
262
|
+
collection_id, base_url=base_url, limit=samples, asset_role=asset_role
|
|
263
|
+
):
|
|
264
|
+
lst = out.setdefault(cid, [])
|
|
265
|
+
if len(lst) < samples:
|
|
266
|
+
lst.append(fn)
|
|
267
|
+
return out
|
parseo/stac_scraper.py
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""STAC helpers backed by ``pystac-client``.
|
|
2
|
+
|
|
3
|
+
This module mirrors the utilities in :mod:`parseo.stac_http` but relies on
|
|
4
|
+
``pystac-client`` for STAC catalog traversal. Use these helpers when the extra
|
|
5
|
+
features of ``pystac-client`` are required. For a lightweight alternative that
|
|
6
|
+
only depends on the Python standard library see :mod:`parseo.stac_http`.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Union
|
|
12
|
+
from urllib.parse import urlparse
|
|
13
|
+
|
|
14
|
+
def list_collections_client(base_url: str, *, deep: bool = False) -> list[str]:
|
|
15
|
+
"""Return collection IDs from a STAC API using ``pystac-client``.
|
|
16
|
+
|
|
17
|
+
Parameters mirror :func:`parseo.stac_http.list_collections_http` but
|
|
18
|
+
this variant requires the optional ``pystac-client`` dependency. It is
|
|
19
|
+
suitable when more advanced STAC handling is needed, at the cost of pulling
|
|
20
|
+
in the external library.
|
|
21
|
+
|
|
22
|
+
Raises
|
|
23
|
+
------
|
|
24
|
+
ImportError
|
|
25
|
+
If ``pystac-client`` is not installed.
|
|
26
|
+
"""
|
|
27
|
+
try:
|
|
28
|
+
from pystac_client import Client
|
|
29
|
+
except Exception as exc: # pragma: no cover - exercised when dependency missing
|
|
30
|
+
raise ImportError(
|
|
31
|
+
"pystac-client is required for list_collections_client"
|
|
32
|
+
) from exc
|
|
33
|
+
|
|
34
|
+
client = Client.open(base_url)
|
|
35
|
+
collections = {c.id for c in client.get_collections()}
|
|
36
|
+
if not deep:
|
|
37
|
+
return sorted(collections)
|
|
38
|
+
|
|
39
|
+
# Breadth-first traversal of child catalogs.
|
|
40
|
+
to_visit = list(client.get_children())
|
|
41
|
+
visited: set[str] = set()
|
|
42
|
+
while to_visit:
|
|
43
|
+
child = to_visit.pop()
|
|
44
|
+
href = getattr(child, "href", None) or getattr(child, "target", None)
|
|
45
|
+
if not href or href in visited:
|
|
46
|
+
continue
|
|
47
|
+
visited.add(href)
|
|
48
|
+
sub_client = Client.open(href)
|
|
49
|
+
collections.update(c.id for c in sub_client.get_collections())
|
|
50
|
+
to_visit.extend(sub_client.get_children())
|
|
51
|
+
|
|
52
|
+
return sorted(collections)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def search_stac_and_download(
|
|
56
|
+
*,
|
|
57
|
+
stac_url: str,
|
|
58
|
+
collections: list[str],
|
|
59
|
+
bbox: Union[list[float], tuple[float, float, float, float]],
|
|
60
|
+
datetime: str,
|
|
61
|
+
dest_dir: Union[str, Path],
|
|
62
|
+
) -> Path:
|
|
63
|
+
"""Download the first asset matching a STAC search.
|
|
64
|
+
|
|
65
|
+
The search is performed via :mod:`pystac-client` and the asset is retrieved
|
|
66
|
+
with :mod:`requests`. ``dest_dir`` is created if needed and the path to the
|
|
67
|
+
downloaded file is returned.
|
|
68
|
+
|
|
69
|
+
Raises
|
|
70
|
+
------
|
|
71
|
+
ImportError
|
|
72
|
+
If ``pystac-client`` or ``requests`` is not installed.
|
|
73
|
+
FileNotFoundError
|
|
74
|
+
If the STAC search yields no downloadable assets or all downloads
|
|
75
|
+
fail.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
try:
|
|
79
|
+
from pystac_client import Client
|
|
80
|
+
except Exception as exc: # pragma: no cover - exercised when dependency missing
|
|
81
|
+
raise ImportError(
|
|
82
|
+
"pystac-client is required for search_stac_and_download"
|
|
83
|
+
) from exc
|
|
84
|
+
|
|
85
|
+
try:
|
|
86
|
+
import requests
|
|
87
|
+
except Exception as exc: # pragma: no cover - exercised when dependency missing
|
|
88
|
+
raise ImportError(
|
|
89
|
+
"requests is required for search_stac_and_download"
|
|
90
|
+
) from exc
|
|
91
|
+
|
|
92
|
+
client = Client.open(stac_url)
|
|
93
|
+
search = client.search(collections=collections, bbox=bbox, datetime=datetime)
|
|
94
|
+
for item in search.items():
|
|
95
|
+
for asset in item.assets.values():
|
|
96
|
+
href = getattr(asset, "href", None)
|
|
97
|
+
if not href:
|
|
98
|
+
continue
|
|
99
|
+
name = getattr(asset, "title", None)
|
|
100
|
+
if not name:
|
|
101
|
+
name = Path(urlparse(href).path).name
|
|
102
|
+
dest_dir_path = Path(dest_dir)
|
|
103
|
+
dest_dir_path.mkdir(parents=True, exist_ok=True)
|
|
104
|
+
dest_path = dest_dir_path / name
|
|
105
|
+
try:
|
|
106
|
+
with requests.get(href, stream=True) as resp:
|
|
107
|
+
resp.raise_for_status()
|
|
108
|
+
with open(dest_path, "wb") as fh:
|
|
109
|
+
for chunk in resp.iter_content(chunk_size=8192):
|
|
110
|
+
if chunk:
|
|
111
|
+
fh.write(chunk)
|
|
112
|
+
return dest_path
|
|
113
|
+
except requests.HTTPError:
|
|
114
|
+
continue
|
|
115
|
+
raise FileNotFoundError("No matching assets found")
|