parseo 0.4.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parseo/__init__.py +53 -0
- parseo/_epsg_lookup.py +145 -0
- parseo/_field_mappings.py +219 -0
- parseo/_json.py +14 -0
- parseo/_tile_systems.py +51 -0
- parseo/assembler.py +225 -0
- parseo/cli.py +383 -0
- parseo/parser.py +784 -0
- parseo/schema_registry.py +246 -0
- parseo/schemas/copernicus/clms/clc/clc_filename_v1_0_0.json +48 -0
- parseo/schemas/copernicus/clms/clc/clc_filename_v1_1_0.json +65 -0
- parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_0.json +90 -0
- parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_1.json +80 -0
- parseo/schemas/copernicus/clms/egms/egms_gnss_model_filename_v0_0_0.json +46 -0
- parseo/schemas/copernicus/clms/egms/egms_l2a_product_filename_v0_0_0.json +71 -0
- parseo/schemas/copernicus/clms/egms/egms_l3_velocity_grid_filename_v0_0_0.json +64 -0
- parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_0.json +115 -0
- parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_1.json +103 -0
- parseo/schemas/copernicus/clms/hr-vpp/st_filename_v0_0_0.json +82 -0
- parseo/schemas/copernicus/clms/hr-vpp/vi_filename_v0_0_0.json +89 -0
- parseo/schemas/copernicus/clms/hr-vpp/vpp_filename_v0_0_0.json +109 -0
- parseo/schemas/copernicus/clms/hr-wsi/cc_filename_v0_0_0.json +84 -0
- parseo/schemas/copernicus/clms/hr-wsi/fsc_filename_v0_0_0.json +88 -0
- parseo/schemas/copernicus/clms/hr-wsi/gfsc_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/icd_filename_v0_0_0.json +96 -0
- parseo/schemas/copernicus/clms/hr-wsi/sp_filename_v0_0_0.json +177 -0
- parseo/schemas/copernicus/clms/hr-wsi/sws_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/wcd_filename_v0_0_0.json +96 -0
- parseo/schemas/copernicus/clms/hr-wsi/wds_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/wic_comb_filename_v0_0_0.json +82 -0
- parseo/schemas/copernicus/clms/hr-wsi/wic_filename_v0_0_0.json +90 -0
- parseo/schemas/copernicus/clms/hrl/fty_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/gra_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/ibu_filename_v0_0_0.json +57 -0
- parseo/schemas/copernicus/clms/hrl/imp_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_0.json +73 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_1.json +89 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_2.json +89 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_1.json +49 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_2.json +93 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_3.json +99 -0
- parseo/schemas/copernicus/clms/hrl/tcd_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/vlcc_filename_v0_0_0.json +105 -0
- parseo/schemas/copernicus/clms/hrl/waw_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/n2k/n2k_filename_v1_0_0.json +43 -0
- parseo/schemas/copernicus/clms/pa/pa_filename_v1_0_0.json +99 -0
- parseo/schemas/copernicus/clms/riparian-zones/rpz_filename_v0_0_0.json +50 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_dhm_filename_v0_0_0.json +53 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_filename_v0_0_0.json +105 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_stl_filename_v0_0_0.json +89 -0
- parseo/schemas/copernicus/sentinel/s1_filename_v1_0_0.json +78 -0
- parseo/schemas/copernicus/sentinel/s2_filename_v1_0_0.json +62 -0
- parseo/schemas/copernicus/sentinel/s3_filename_v1_0_0.json +22 -0
- parseo/schemas/copernicus/sentinel/s4_filename_v1_0_0.json +21 -0
- parseo/schemas/copernicus/sentinel/s5p_filename_v1_0_0.json +21 -0
- parseo/schemas/copernicus/sentinel/s6_filename_v1_0_0.json +23 -0
- parseo/schemas/eumetsat/metop_filename_v1_0_0.json +68 -0
- parseo/schemas/eumetsat/mtg_filename_v1_0_0.json +65 -0
- parseo/schemas/nasa/modis_filename_v1_0_0.json +54 -0
- parseo/schemas/usgs/landsat/landsat_filename_v1_0_0.json +50 -0
- parseo/stac_http.py +267 -0
- parseo/stac_scraper.py +115 -0
- parseo/template.py +78 -0
- parseo-0.4.4.dist-info/METADATA +402 -0
- parseo-0.4.4.dist-info/RECORD +70 -0
- parseo-0.4.4.dist-info/WHEEL +5 -0
- parseo-0.4.4.dist-info/entry_points.txt +2 -0
- parseo-0.4.4.dist-info/licenses/LICENSE.txt +287 -0
- parseo-0.4.4.dist-info/top_level.txt +1 -0
parseo/__init__.py
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Top level package for parseo."""
|
|
2
|
+
|
|
3
|
+
from importlib import metadata
|
|
4
|
+
from .assembler import assemble
|
|
5
|
+
from .assembler import assemble_auto
|
|
6
|
+
from .assembler import clear_schema_cache
|
|
7
|
+
from .parser import parse
|
|
8
|
+
from .parser import parse_auto
|
|
9
|
+
from .parser import validate_schema
|
|
10
|
+
from .schema_registry import get_schema_path
|
|
11
|
+
from .schema_registry import list_schema_families
|
|
12
|
+
from .schema_registry import list_schema_versions
|
|
13
|
+
|
|
14
|
+
try: # pragma: no cover - import failure handled for graceful degradation
|
|
15
|
+
from . import parser # noqa: F401 # import for side effect and re-export
|
|
16
|
+
except Exception: # ImportError and others
|
|
17
|
+
parser = None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
try: # pragma: no cover - gracefully handle missing distribution
|
|
21
|
+
__version__ = metadata.version("parseo")
|
|
22
|
+
except metadata.PackageNotFoundError: # pragma: no cover - defensive
|
|
23
|
+
__version__ = "unknown"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def info() -> dict[str, str]:
|
|
27
|
+
"""Return information about the installed :mod:`parseo` package.
|
|
28
|
+
|
|
29
|
+
Returns
|
|
30
|
+
-------
|
|
31
|
+
dict[str, str]
|
|
32
|
+
A dictionary containing the installed version under the ``"version"`` key.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
return {"version": __version__}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
__all__ = [
|
|
39
|
+
"parse",
|
|
40
|
+
"parse_auto",
|
|
41
|
+
"assemble",
|
|
42
|
+
"assemble_auto",
|
|
43
|
+
"clear_schema_cache",
|
|
44
|
+
"validate_schema",
|
|
45
|
+
"list_schema_families",
|
|
46
|
+
"list_schema_versions",
|
|
47
|
+
"get_schema_path",
|
|
48
|
+
"info",
|
|
49
|
+
"__version__",
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
if parser is not None:
|
|
53
|
+
__all__.append("parser")
|
parseo/_epsg_lookup.py
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Lookup helpers for deriving EPSG codes from tile identifiers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import math
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
_MGRS_LATITUDE_BANDS = {
|
|
11
|
+
"C": "south",
|
|
12
|
+
"D": "south",
|
|
13
|
+
"E": "south",
|
|
14
|
+
"F": "south",
|
|
15
|
+
"G": "south",
|
|
16
|
+
"H": "south",
|
|
17
|
+
"J": "south",
|
|
18
|
+
"K": "south",
|
|
19
|
+
"L": "south",
|
|
20
|
+
"M": "south",
|
|
21
|
+
"N": "north",
|
|
22
|
+
"P": "north",
|
|
23
|
+
"Q": "north",
|
|
24
|
+
"R": "north",
|
|
25
|
+
"S": "north",
|
|
26
|
+
"T": "north",
|
|
27
|
+
"U": "north",
|
|
28
|
+
"V": "north",
|
|
29
|
+
"W": "north",
|
|
30
|
+
"X": "north",
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def mgrs_tile_to_epsg(tile: str) -> Optional[str]:
|
|
35
|
+
"""Return the EPSG code associated with a Sentinel-2 MGRS tile.
|
|
36
|
+
|
|
37
|
+
Parameters
|
|
38
|
+
----------
|
|
39
|
+
tile:
|
|
40
|
+
The tile identifier (e.g. ``"T32TNS"``).
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
if not isinstance(tile, str) or len(tile) < 4:
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
tile = tile.strip().upper()
|
|
47
|
+
if not tile.startswith("T"):
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
zone_part = tile[1:3]
|
|
51
|
+
try:
|
|
52
|
+
zone = int(zone_part)
|
|
53
|
+
except ValueError:
|
|
54
|
+
return None
|
|
55
|
+
if not 1 <= zone <= 60:
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
band = tile[3]
|
|
59
|
+
hemisphere = _MGRS_LATITUDE_BANDS.get(band)
|
|
60
|
+
if hemisphere is None:
|
|
61
|
+
return None
|
|
62
|
+
|
|
63
|
+
if hemisphere == "north":
|
|
64
|
+
epsg = 32600 + zone
|
|
65
|
+
else:
|
|
66
|
+
epsg = 32700 + zone
|
|
67
|
+
|
|
68
|
+
return f"{epsg:05d}"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass(frozen=True)
|
|
72
|
+
class _WRSOrbitConstants:
|
|
73
|
+
"""Constants derived from the WRS-2 orbital configuration."""
|
|
74
|
+
|
|
75
|
+
orbital_period_days: float = 16.0
|
|
76
|
+
paths: int = 233
|
|
77
|
+
rows: int = 248
|
|
78
|
+
inclination_deg: float = 98.2
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
_WRS_CONSTANTS = _WRSOrbitConstants()
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _path_to_longitude(path: int) -> float:
|
|
85
|
+
"""Approximate the longitude of the descending node for *path* (degrees)."""
|
|
86
|
+
|
|
87
|
+
fraction = (path - 1) / _WRS_CONSTANTS.paths
|
|
88
|
+
longitude = (fraction * 360.0) % 360.0
|
|
89
|
+
if longitude > 180.0:
|
|
90
|
+
longitude -= 360.0
|
|
91
|
+
return longitude
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _row_to_latitude(row: int) -> float:
|
|
95
|
+
"""Approximate the latitude of the scene centre for *row* (degrees)."""
|
|
96
|
+
|
|
97
|
+
# Rows increase from north to south. We approximate the relationship using
|
|
98
|
+
# a linear fit anchored at the documented WRS limits (81°N and 81°S).
|
|
99
|
+
total_rows = _WRS_CONSTANTS.rows
|
|
100
|
+
span = 162.0 # 81°N to 81°S
|
|
101
|
+
step = span / (total_rows - 1)
|
|
102
|
+
return 81.0 - (row - 1) * step
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def landsat_path_row_to_epsg(path: str, row: str) -> Optional[str]:
|
|
106
|
+
"""Return the EPSG code inferred from a Landsat WRS path/row pair."""
|
|
107
|
+
|
|
108
|
+
try:
|
|
109
|
+
path_num = int(path)
|
|
110
|
+
row_num = int(row)
|
|
111
|
+
except (TypeError, ValueError):
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
if not 1 <= path_num <= _WRS_CONSTANTS.paths:
|
|
115
|
+
return None
|
|
116
|
+
if not 1 <= row_num <= _WRS_CONSTANTS.rows:
|
|
117
|
+
return None
|
|
118
|
+
|
|
119
|
+
longitude = _path_to_longitude(path_num)
|
|
120
|
+
latitude = _row_to_latitude(row_num)
|
|
121
|
+
|
|
122
|
+
zone = int(math.floor((longitude + 180.0) / 6.0)) + 1
|
|
123
|
+
zone = max(1, min(zone, 60))
|
|
124
|
+
|
|
125
|
+
# Special handling for Norway and Svalbard following the UTM specification.
|
|
126
|
+
if 56.0 <= latitude < 64.0 and 3.0 <= longitude < 12.0:
|
|
127
|
+
zone = 32
|
|
128
|
+
if 72.0 <= latitude <= 84.0:
|
|
129
|
+
if 0.0 <= longitude < 9.0:
|
|
130
|
+
zone = 31
|
|
131
|
+
elif 9.0 <= longitude < 21.0:
|
|
132
|
+
zone = 33
|
|
133
|
+
elif 21.0 <= longitude < 33.0:
|
|
134
|
+
zone = 35
|
|
135
|
+
elif 33.0 <= longitude < 42.0:
|
|
136
|
+
zone = 37
|
|
137
|
+
|
|
138
|
+
hemisphere = "north" if latitude >= 0.0 else "south"
|
|
139
|
+
if hemisphere == "north":
|
|
140
|
+
epsg = 32600 + zone
|
|
141
|
+
else:
|
|
142
|
+
epsg = 32700 + zone
|
|
143
|
+
|
|
144
|
+
return f"{epsg:05d}"
|
|
145
|
+
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Utilities for translating between schema tokens and STAC values."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any
|
|
7
|
+
from typing import Dict
|
|
8
|
+
from typing import Mapping
|
|
9
|
+
from typing import Optional
|
|
10
|
+
|
|
11
|
+
from ._epsg_lookup import landsat_path_row_to_epsg
|
|
12
|
+
from ._epsg_lookup import mgrs_tile_to_epsg
|
|
13
|
+
from ._tile_systems import TileSystem
|
|
14
|
+
from ._tile_systems import detect_tile_system
|
|
15
|
+
from ._tile_systems import normalize_tile
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class FieldMapping:
|
|
20
|
+
"""Represents a mapping between a filename token and STAC fields."""
|
|
21
|
+
|
|
22
|
+
preserve_as: Optional[str]
|
|
23
|
+
token_map: Dict[str, Dict[str, Any]]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _normalize_mapping_values(values: Mapping[str, Any]) -> Dict[str, Dict[str, Any]]:
|
|
27
|
+
normalized: Dict[str, Dict[str, Any]] = {}
|
|
28
|
+
for token, targets in values.items():
|
|
29
|
+
if not isinstance(targets, Mapping):
|
|
30
|
+
continue
|
|
31
|
+
normalized[str(token)] = {str(k): v for k, v in targets.items()}
|
|
32
|
+
return normalized
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def get_schema_field_mappings(schema: Mapping[str, Any]) -> Dict[str, FieldMapping]:
|
|
36
|
+
"""Extract field mappings declared in *schema*."""
|
|
37
|
+
|
|
38
|
+
fields = schema.get("fields", {})
|
|
39
|
+
if not isinstance(fields, Mapping):
|
|
40
|
+
return {}
|
|
41
|
+
|
|
42
|
+
mappings: Dict[str, FieldMapping] = {}
|
|
43
|
+
for field_name, spec in fields.items():
|
|
44
|
+
if not isinstance(spec, Mapping):
|
|
45
|
+
continue
|
|
46
|
+
raw_map = spec.get("stac_map")
|
|
47
|
+
if not isinstance(raw_map, Mapping):
|
|
48
|
+
continue
|
|
49
|
+
|
|
50
|
+
has_explicit_values = "values" in raw_map
|
|
51
|
+
preserve_raw = raw_map.get("preserve_original_as")
|
|
52
|
+
preserve_as: Optional[str]
|
|
53
|
+
|
|
54
|
+
if isinstance(preserve_raw, str):
|
|
55
|
+
preserve_as = preserve_raw.strip() or None
|
|
56
|
+
elif isinstance(preserve_raw, bool):
|
|
57
|
+
preserve_as = f"{field_name}_code" if preserve_raw else None
|
|
58
|
+
elif preserve_raw is not None:
|
|
59
|
+
preserve_as = str(preserve_raw).strip() or None
|
|
60
|
+
else:
|
|
61
|
+
preserve_as = None if has_explicit_values else f"{field_name}_code"
|
|
62
|
+
|
|
63
|
+
values: Any
|
|
64
|
+
if "values" in raw_map:
|
|
65
|
+
values = raw_map.get("values")
|
|
66
|
+
else:
|
|
67
|
+
values = raw_map
|
|
68
|
+
if not isinstance(values, Mapping):
|
|
69
|
+
continue
|
|
70
|
+
|
|
71
|
+
normalized = _normalize_mapping_values(values)
|
|
72
|
+
if not normalized:
|
|
73
|
+
continue
|
|
74
|
+
|
|
75
|
+
mappings[str(field_name)] = FieldMapping(preserve_as=preserve_as, token_map=normalized)
|
|
76
|
+
|
|
77
|
+
return mappings
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _augment_with_tile_variants(fields: Dict[str, Any]) -> Dict[str, Any]:
|
|
81
|
+
enriched = dict(fields)
|
|
82
|
+
|
|
83
|
+
mgrs_value = enriched.get("mgrs_tile")
|
|
84
|
+
if isinstance(mgrs_value, str) and mgrs_value and "tile_id" not in enriched:
|
|
85
|
+
enriched["tile_id"] = mgrs_value
|
|
86
|
+
|
|
87
|
+
for candidate in ("tile", "tile_id"):
|
|
88
|
+
value = enriched.get(candidate)
|
|
89
|
+
if not isinstance(value, str):
|
|
90
|
+
continue
|
|
91
|
+
|
|
92
|
+
system = detect_tile_system(value)
|
|
93
|
+
if system is None:
|
|
94
|
+
continue
|
|
95
|
+
|
|
96
|
+
normalized = normalize_tile(value)
|
|
97
|
+
enriched[candidate] = normalized
|
|
98
|
+
|
|
99
|
+
if system is TileSystem.MGRS and "tile_id" not in enriched:
|
|
100
|
+
enriched["tile_id"] = normalized
|
|
101
|
+
if system is TileSystem.EEA:
|
|
102
|
+
if "tile_id" not in enriched:
|
|
103
|
+
enriched["tile_id"] = normalized
|
|
104
|
+
if "epsg_code" not in enriched:
|
|
105
|
+
enriched["epsg_code"] = "03035"
|
|
106
|
+
|
|
107
|
+
enriched.pop("mgrs_tile", None)
|
|
108
|
+
return enriched
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _backfill_tile_tokens(
|
|
112
|
+
fields: Dict[str, Any], schema: Mapping[str, Any]
|
|
113
|
+
) -> Dict[str, Any]:
|
|
114
|
+
translated = dict(fields)
|
|
115
|
+
specs = schema.get("fields", {})
|
|
116
|
+
|
|
117
|
+
if "tile" in specs and "tile" not in translated:
|
|
118
|
+
for key in ("tile_id", "mgrs_tile"):
|
|
119
|
+
value = translated.get(key)
|
|
120
|
+
if isinstance(value, str) and value:
|
|
121
|
+
translated["tile"] = normalize_tile(value)
|
|
122
|
+
break
|
|
123
|
+
|
|
124
|
+
if "tile_id" in specs and "tile_id" not in translated:
|
|
125
|
+
for key in ("tile", "mgrs_tile"):
|
|
126
|
+
value = translated.get(key)
|
|
127
|
+
if isinstance(value, str) and value:
|
|
128
|
+
translated["tile_id"] = normalize_tile(value)
|
|
129
|
+
break
|
|
130
|
+
|
|
131
|
+
translated.pop("mgrs_tile", None)
|
|
132
|
+
return translated
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def apply_schema_mappings(
|
|
136
|
+
extracted: Dict[str, Any], schema: Mapping[str, Any]
|
|
137
|
+
) -> Dict[str, Any]:
|
|
138
|
+
"""Augment *extracted* fields with STAC values defined in *schema*."""
|
|
139
|
+
|
|
140
|
+
mappings = get_schema_field_mappings(schema)
|
|
141
|
+
|
|
142
|
+
enriched = _augment_with_tile_variants(extracted)
|
|
143
|
+
for field_name, mapping in mappings.items():
|
|
144
|
+
token = extracted.get(field_name)
|
|
145
|
+
if token is None:
|
|
146
|
+
continue
|
|
147
|
+
|
|
148
|
+
if mapping.preserve_as:
|
|
149
|
+
enriched[mapping.preserve_as] = token
|
|
150
|
+
|
|
151
|
+
targets = mapping.token_map.get(str(token))
|
|
152
|
+
if not isinstance(targets, Mapping):
|
|
153
|
+
continue
|
|
154
|
+
|
|
155
|
+
for target_field, target_value in targets.items():
|
|
156
|
+
enriched[target_field] = target_value
|
|
157
|
+
|
|
158
|
+
schema_id = str(schema.get("schema_id", "")).lower()
|
|
159
|
+
|
|
160
|
+
if "epsg_code" not in enriched:
|
|
161
|
+
tile = enriched.get("tile_id")
|
|
162
|
+
if (
|
|
163
|
+
tile
|
|
164
|
+
and "sentinel:s2" in schema_id
|
|
165
|
+
and detect_tile_system(str(tile)) is TileSystem.MGRS
|
|
166
|
+
):
|
|
167
|
+
epsg = mgrs_tile_to_epsg(str(tile))
|
|
168
|
+
if epsg:
|
|
169
|
+
enriched["epsg_code"] = epsg
|
|
170
|
+
|
|
171
|
+
if "epsg_code" not in enriched:
|
|
172
|
+
path = enriched.get("wrs_path")
|
|
173
|
+
row = enriched.get("wrs_row")
|
|
174
|
+
if path and row and schema_id.startswith("usgs:landsat"):
|
|
175
|
+
epsg = landsat_path_row_to_epsg(str(path), str(row))
|
|
176
|
+
if epsg:
|
|
177
|
+
enriched["epsg_code"] = epsg
|
|
178
|
+
|
|
179
|
+
return enriched
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def translate_fields_to_tokens(
|
|
183
|
+
fields: Dict[str, Any], schema: Mapping[str, Any]
|
|
184
|
+
) -> Dict[str, Any]:
|
|
185
|
+
"""Translate STAC field values in *fields* back to schema tokens."""
|
|
186
|
+
|
|
187
|
+
mappings = get_schema_field_mappings(schema)
|
|
188
|
+
if not mappings:
|
|
189
|
+
return fields
|
|
190
|
+
|
|
191
|
+
translated = _augment_with_tile_variants(fields)
|
|
192
|
+
for field_name, mapping in mappings.items():
|
|
193
|
+
token: Any = None
|
|
194
|
+
|
|
195
|
+
preserved = None
|
|
196
|
+
if mapping.preserve_as:
|
|
197
|
+
preserved = fields.get(mapping.preserve_as)
|
|
198
|
+
if preserved not in (None, ""):
|
|
199
|
+
token = str(preserved)
|
|
200
|
+
|
|
201
|
+
if token is None:
|
|
202
|
+
current_value = fields.get(field_name)
|
|
203
|
+
if isinstance(current_value, str) and current_value in mapping.token_map:
|
|
204
|
+
token = current_value
|
|
205
|
+
|
|
206
|
+
if token is None:
|
|
207
|
+
for candidate, targets in mapping.token_map.items():
|
|
208
|
+
if all(fields.get(k) == v for k, v in targets.items()):
|
|
209
|
+
token = candidate
|
|
210
|
+
break
|
|
211
|
+
|
|
212
|
+
if token is None:
|
|
213
|
+
continue
|
|
214
|
+
|
|
215
|
+
translated[field_name] = token
|
|
216
|
+
|
|
217
|
+
translated = _backfill_tile_tokens(translated, schema)
|
|
218
|
+
return translated
|
|
219
|
+
|
parseo/_json.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any
|
|
6
|
+
from typing import Dict
|
|
7
|
+
from typing import Union
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def load_json(path: Union[str, Path]) -> Dict[str, Any]:
|
|
11
|
+
"""Load a JSON file, handling optional UTF-8 BOM."""
|
|
12
|
+
p = Path(path)
|
|
13
|
+
text = p.read_text(encoding="utf-8-sig")
|
|
14
|
+
return json.loads(text)
|
parseo/_tile_systems.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Helpers for recognizing spatial tile identifier systems."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from enum import Enum
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class TileSystem(str, Enum):
|
|
11
|
+
"""Known spatial tiling systems."""
|
|
12
|
+
|
|
13
|
+
MGRS = "mgrs"
|
|
14
|
+
EEA = "eea"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
_MGRS_PATTERN = re.compile(r"^T\d{2}[C-HJ-NP-X][A-Z]{2}$")
|
|
18
|
+
_EEA_PATTERN = re.compile(r"^[EW]\d{2,3}[NS]\d{2,3}$")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def detect_tile_system(tile: str) -> Optional[TileSystem]:
|
|
22
|
+
"""Return the :class:`TileSystem` matching *tile*, if any."""
|
|
23
|
+
|
|
24
|
+
if not isinstance(tile, str):
|
|
25
|
+
return None
|
|
26
|
+
|
|
27
|
+
candidate = tile.strip().upper()
|
|
28
|
+
if not candidate:
|
|
29
|
+
return None
|
|
30
|
+
|
|
31
|
+
if _MGRS_PATTERN.fullmatch(candidate):
|
|
32
|
+
return TileSystem.MGRS
|
|
33
|
+
|
|
34
|
+
if _EEA_PATTERN.fullmatch(candidate):
|
|
35
|
+
return TileSystem.EEA
|
|
36
|
+
|
|
37
|
+
return None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def normalize_tile(tile: str) -> str:
|
|
41
|
+
"""Return a normalised representation of *tile*.
|
|
42
|
+
|
|
43
|
+
MGRS and EEA tiles are uppercased. Tiles from unrecognised systems
|
|
44
|
+
(e.g., MODIS sinusoidal ``h18v04``) are returned as-is.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
t = tile.strip()
|
|
48
|
+
if detect_tile_system(t):
|
|
49
|
+
return t.upper()
|
|
50
|
+
return t
|
|
51
|
+
|
parseo/assembler.py
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# src/parseo/assembler.py
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from functools import lru_cache
|
|
5
|
+
from importlib.resources import as_file
|
|
6
|
+
from importlib.resources import files
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
import re
|
|
9
|
+
from typing import Any
|
|
10
|
+
from typing import Dict
|
|
11
|
+
from typing import Union
|
|
12
|
+
|
|
13
|
+
from ._field_mappings import translate_fields_to_tokens
|
|
14
|
+
from .parser import _get_fields_order
|
|
15
|
+
from .parser import _pattern_from_schema
|
|
16
|
+
from .schema_registry import _load_json_from_path
|
|
17
|
+
from .schema_registry import get_schema_path
|
|
18
|
+
from .schema_registry import SCHEMAS_ROOT
|
|
19
|
+
from .template import _field_regex
|
|
20
|
+
from .template import compile_template
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def clear_schema_cache() -> None:
|
|
24
|
+
"""Clear all cached schema data."""
|
|
25
|
+
_load_json_from_path.cache_clear()
|
|
26
|
+
from .parser import _SCHEMA_PATTERN_CACHE, _SCHEMA_ORDER_CACHE
|
|
27
|
+
|
|
28
|
+
_SCHEMA_PATTERN_CACHE.clear()
|
|
29
|
+
_SCHEMA_ORDER_CACHE.clear()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _assemble_from_template(template: str, fields: Dict[str, Any]) -> str:
|
|
33
|
+
"""Render *template* using *fields*.
|
|
34
|
+
|
|
35
|
+
Optional segments denoted by ``[ ... ]`` are dropped if any enclosed field
|
|
36
|
+
is missing. ``{field}`` placeholders are replaced by values from *fields*.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def render(segment: str) -> str:
|
|
40
|
+
result = ""
|
|
41
|
+
i = 0
|
|
42
|
+
while i < len(segment):
|
|
43
|
+
ch = segment[i]
|
|
44
|
+
if ch == "{":
|
|
45
|
+
j = segment.index("}", i)
|
|
46
|
+
name = segment[i + 1 : j]
|
|
47
|
+
if name not in fields:
|
|
48
|
+
raise KeyError(name)
|
|
49
|
+
result += str(fields[name])
|
|
50
|
+
i = j + 1
|
|
51
|
+
elif ch == "[":
|
|
52
|
+
depth = 1
|
|
53
|
+
j = i + 1
|
|
54
|
+
while j < len(segment) and depth:
|
|
55
|
+
if segment[j] == "[":
|
|
56
|
+
depth += 1
|
|
57
|
+
elif segment[j] == "]":
|
|
58
|
+
depth -= 1
|
|
59
|
+
j += 1
|
|
60
|
+
inner = segment[i + 1 : j - 1]
|
|
61
|
+
try:
|
|
62
|
+
result += render(inner)
|
|
63
|
+
except KeyError:
|
|
64
|
+
pass
|
|
65
|
+
i = j
|
|
66
|
+
else:
|
|
67
|
+
result += ch
|
|
68
|
+
i += 1
|
|
69
|
+
return result
|
|
70
|
+
|
|
71
|
+
return render(template)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _assemble_schema(schema_path: Union[str, Path], fields: Dict[str, Any]) -> str:
|
|
75
|
+
"""Assemble a filename using a JSON schema.
|
|
76
|
+
|
|
77
|
+
Schemas must define a ``template`` string following parseo's mini-template
|
|
78
|
+
syntax. The template is rendered using the provided *fields* and optional
|
|
79
|
+
segments are dropped if any enclosed field is missing.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
sch = _load_json_from_path(schema_path)
|
|
83
|
+
prepared_fields = translate_fields_to_tokens(fields, sch)
|
|
84
|
+
|
|
85
|
+
template = sch.get("template")
|
|
86
|
+
if not isinstance(template, str):
|
|
87
|
+
raise ValueError(f"Schema {schema_path} missing 'template' string.")
|
|
88
|
+
|
|
89
|
+
epsg_value = prepared_fields.get("epsg_code")
|
|
90
|
+
if (
|
|
91
|
+
isinstance(epsg_value, str)
|
|
92
|
+
and "EPSG{epsg_code}" in template
|
|
93
|
+
and epsg_value
|
|
94
|
+
):
|
|
95
|
+
prepared_fields["epsg_code"] = epsg_value.lstrip("0") or "0"
|
|
96
|
+
|
|
97
|
+
# Validate provided fields against schema definitions
|
|
98
|
+
specs = sch.get("fields", {})
|
|
99
|
+
for name, value in prepared_fields.items():
|
|
100
|
+
spec = specs.get(name)
|
|
101
|
+
if not spec:
|
|
102
|
+
continue
|
|
103
|
+
if "enum" in spec and str(value) not in spec["enum"]:
|
|
104
|
+
raise ValueError(
|
|
105
|
+
f"Field '{name}' must be one of {spec['enum']}, got {value!r}."
|
|
106
|
+
)
|
|
107
|
+
if "pattern" in spec:
|
|
108
|
+
pattern = _field_regex({"pattern": spec["pattern"]})
|
|
109
|
+
regex = re.compile(f"^{pattern}$")
|
|
110
|
+
if not regex.match(str(value)):
|
|
111
|
+
raise ValueError(
|
|
112
|
+
f"Field '{name}' with value {value!r} does not match pattern {spec['pattern']}."
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
try:
|
|
116
|
+
return _assemble_from_template(template, prepared_fields)
|
|
117
|
+
except KeyError as exc:
|
|
118
|
+
name = exc.args[0]
|
|
119
|
+
raise ValueError(
|
|
120
|
+
f"Missing field '{name}' for schema {schema_path}"
|
|
121
|
+
) from exc
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def assemble(
|
|
125
|
+
fields: Dict[str, Any],
|
|
126
|
+
family: Union[str, None] = None,
|
|
127
|
+
version: Union[str, None] = None,
|
|
128
|
+
schema_path: Union[str, Path, None] = None,
|
|
129
|
+
) -> str:
|
|
130
|
+
"""Assemble a filename from *fields*.
|
|
131
|
+
|
|
132
|
+
Provide either a *schema_path* or a schema *family* (with optional
|
|
133
|
+
*version*) to select the schema. If neither is given the schema is
|
|
134
|
+
auto-selected based on the supplied *fields*.
|
|
135
|
+
"""
|
|
136
|
+
|
|
137
|
+
if schema_path is not None:
|
|
138
|
+
return _assemble_schema(schema_path, fields)
|
|
139
|
+
if family is not None:
|
|
140
|
+
resolved = get_schema_path(family, version=version)
|
|
141
|
+
return _assemble_schema(resolved, fields)
|
|
142
|
+
return assemble_auto(fields)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _iter_schema_paths() -> list[Path]:
|
|
146
|
+
"""Return all packaged schema JSON paths."""
|
|
147
|
+
base = files(__package__).joinpath(SCHEMAS_ROOT)
|
|
148
|
+
with as_file(base) as bp:
|
|
149
|
+
root = Path(bp)
|
|
150
|
+
return list(root.rglob("*.json"))
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _select_schema_by_first_compulsory(fields: Dict[str, Any]) -> Path:
|
|
154
|
+
"""Select the most appropriate schema based on provided fields.
|
|
155
|
+
|
|
156
|
+
Eligibility requires the user to provide the first compulsory field as
|
|
157
|
+
derived from the schema's template. Among eligible schemas the one with the
|
|
158
|
+
largest overlap of provided keys is chosen. A longer field order acts as a
|
|
159
|
+
tie breaker.
|
|
160
|
+
"""
|
|
161
|
+
best: Union[tuple[int, int, str], None] = None
|
|
162
|
+
best_path: Union[Path, None] = None
|
|
163
|
+
seen_first_keys: set[str] = set()
|
|
164
|
+
|
|
165
|
+
for p in _iter_schema_paths():
|
|
166
|
+
try:
|
|
167
|
+
sch = _load_json_from_path(p)
|
|
168
|
+
except (OSError, ValueError):
|
|
169
|
+
continue
|
|
170
|
+
|
|
171
|
+
template = sch.get("template")
|
|
172
|
+
if not isinstance(template, str):
|
|
173
|
+
continue
|
|
174
|
+
_pattern_from_schema(sch)
|
|
175
|
+
order = _get_fields_order(sch)
|
|
176
|
+
if not order:
|
|
177
|
+
continue
|
|
178
|
+
|
|
179
|
+
specs = sch.get("fields", {})
|
|
180
|
+
first_key = None
|
|
181
|
+
for name in order:
|
|
182
|
+
spec = specs.get(name, {})
|
|
183
|
+
enums = spec.get("enum")
|
|
184
|
+
field_val = fields.get(name)
|
|
185
|
+
if enums is not None:
|
|
186
|
+
enums = [str(e) for e in enums]
|
|
187
|
+
if len(enums) == 1 and field_val == enums[0]:
|
|
188
|
+
continue
|
|
189
|
+
if field_val is not None and field_val not in enums:
|
|
190
|
+
first_key = None
|
|
191
|
+
break
|
|
192
|
+
first_key = name
|
|
193
|
+
break
|
|
194
|
+
else:
|
|
195
|
+
first_key = name
|
|
196
|
+
break
|
|
197
|
+
if not first_key:
|
|
198
|
+
continue
|
|
199
|
+
|
|
200
|
+
seen_first_keys.add(first_key)
|
|
201
|
+
|
|
202
|
+
if first_key not in fields:
|
|
203
|
+
continue
|
|
204
|
+
|
|
205
|
+
overlap = sum(1 for k in fields.keys() if k in order)
|
|
206
|
+
key = (overlap, len(order), str(p))
|
|
207
|
+
if best is None or key > best:
|
|
208
|
+
best = key
|
|
209
|
+
best_path = p
|
|
210
|
+
|
|
211
|
+
if not best_path:
|
|
212
|
+
sample = ", ".join(sorted(seen_first_keys)) or "<no schemas found>"
|
|
213
|
+
raise ValueError(
|
|
214
|
+
"Could not select a schema. "
|
|
215
|
+
"Include the schema's FIRST compulsory field among your inputs.\n"
|
|
216
|
+
f"Examples of first fields from packaged schemas: {sample}"
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
return best_path
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def assemble_auto(fields: Dict[str, Any]) -> str:
|
|
223
|
+
"""Assemble a filename by auto-selecting the appropriate schema."""
|
|
224
|
+
schema_path = _select_schema_by_first_compulsory(fields)
|
|
225
|
+
return _assemble_schema(str(schema_path), fields)
|