parseo 0.4.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. parseo/__init__.py +53 -0
  2. parseo/_epsg_lookup.py +145 -0
  3. parseo/_field_mappings.py +219 -0
  4. parseo/_json.py +14 -0
  5. parseo/_tile_systems.py +51 -0
  6. parseo/assembler.py +225 -0
  7. parseo/cli.py +383 -0
  8. parseo/parser.py +784 -0
  9. parseo/schema_registry.py +246 -0
  10. parseo/schemas/copernicus/clms/clc/clc_filename_v1_0_0.json +48 -0
  11. parseo/schemas/copernicus/clms/clc/clc_filename_v1_1_0.json +65 -0
  12. parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_0.json +90 -0
  13. parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_1.json +80 -0
  14. parseo/schemas/copernicus/clms/egms/egms_gnss_model_filename_v0_0_0.json +46 -0
  15. parseo/schemas/copernicus/clms/egms/egms_l2a_product_filename_v0_0_0.json +71 -0
  16. parseo/schemas/copernicus/clms/egms/egms_l3_velocity_grid_filename_v0_0_0.json +64 -0
  17. parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_0.json +115 -0
  18. parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_1.json +103 -0
  19. parseo/schemas/copernicus/clms/hr-vpp/st_filename_v0_0_0.json +82 -0
  20. parseo/schemas/copernicus/clms/hr-vpp/vi_filename_v0_0_0.json +89 -0
  21. parseo/schemas/copernicus/clms/hr-vpp/vpp_filename_v0_0_0.json +109 -0
  22. parseo/schemas/copernicus/clms/hr-wsi/cc_filename_v0_0_0.json +84 -0
  23. parseo/schemas/copernicus/clms/hr-wsi/fsc_filename_v0_0_0.json +88 -0
  24. parseo/schemas/copernicus/clms/hr-wsi/gfsc_filename_v0_0_0.json +83 -0
  25. parseo/schemas/copernicus/clms/hr-wsi/icd_filename_v0_0_0.json +96 -0
  26. parseo/schemas/copernicus/clms/hr-wsi/sp_filename_v0_0_0.json +177 -0
  27. parseo/schemas/copernicus/clms/hr-wsi/sws_filename_v0_0_0.json +83 -0
  28. parseo/schemas/copernicus/clms/hr-wsi/wcd_filename_v0_0_0.json +96 -0
  29. parseo/schemas/copernicus/clms/hr-wsi/wds_filename_v0_0_0.json +83 -0
  30. parseo/schemas/copernicus/clms/hr-wsi/wic_comb_filename_v0_0_0.json +82 -0
  31. parseo/schemas/copernicus/clms/hr-wsi/wic_filename_v0_0_0.json +90 -0
  32. parseo/schemas/copernicus/clms/hrl/fty_filename_v0_0_0.json +48 -0
  33. parseo/schemas/copernicus/clms/hrl/gra_filename_v0_0_0.json +48 -0
  34. parseo/schemas/copernicus/clms/hrl/ibu_filename_v0_0_0.json +57 -0
  35. parseo/schemas/copernicus/clms/hrl/imp_filename_v0_0_0.json +48 -0
  36. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_0.json +73 -0
  37. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_1.json +89 -0
  38. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_2.json +89 -0
  39. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_0.json +48 -0
  40. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_1.json +49 -0
  41. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_2.json +93 -0
  42. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_3.json +99 -0
  43. parseo/schemas/copernicus/clms/hrl/tcd_filename_v0_0_0.json +48 -0
  44. parseo/schemas/copernicus/clms/hrl/vlcc_filename_v0_0_0.json +105 -0
  45. parseo/schemas/copernicus/clms/hrl/waw_filename_v0_0_0.json +48 -0
  46. parseo/schemas/copernicus/clms/n2k/n2k_filename_v1_0_0.json +43 -0
  47. parseo/schemas/copernicus/clms/pa/pa_filename_v1_0_0.json +99 -0
  48. parseo/schemas/copernicus/clms/riparian-zones/rpz_filename_v0_0_0.json +50 -0
  49. parseo/schemas/copernicus/clms/urban-atlas/ua_dhm_filename_v0_0_0.json +53 -0
  50. parseo/schemas/copernicus/clms/urban-atlas/ua_filename_v0_0_0.json +105 -0
  51. parseo/schemas/copernicus/clms/urban-atlas/ua_stl_filename_v0_0_0.json +89 -0
  52. parseo/schemas/copernicus/sentinel/s1_filename_v1_0_0.json +78 -0
  53. parseo/schemas/copernicus/sentinel/s2_filename_v1_0_0.json +62 -0
  54. parseo/schemas/copernicus/sentinel/s3_filename_v1_0_0.json +22 -0
  55. parseo/schemas/copernicus/sentinel/s4_filename_v1_0_0.json +21 -0
  56. parseo/schemas/copernicus/sentinel/s5p_filename_v1_0_0.json +21 -0
  57. parseo/schemas/copernicus/sentinel/s6_filename_v1_0_0.json +23 -0
  58. parseo/schemas/eumetsat/metop_filename_v1_0_0.json +68 -0
  59. parseo/schemas/eumetsat/mtg_filename_v1_0_0.json +65 -0
  60. parseo/schemas/nasa/modis_filename_v1_0_0.json +54 -0
  61. parseo/schemas/usgs/landsat/landsat_filename_v1_0_0.json +50 -0
  62. parseo/stac_http.py +267 -0
  63. parseo/stac_scraper.py +115 -0
  64. parseo/template.py +78 -0
  65. parseo-0.4.4.dist-info/METADATA +402 -0
  66. parseo-0.4.4.dist-info/RECORD +70 -0
  67. parseo-0.4.4.dist-info/WHEEL +5 -0
  68. parseo-0.4.4.dist-info/entry_points.txt +2 -0
  69. parseo-0.4.4.dist-info/licenses/LICENSE.txt +287 -0
  70. parseo-0.4.4.dist-info/top_level.txt +1 -0
parseo/__init__.py ADDED
@@ -0,0 +1,53 @@
1
+ """Top level package for parseo."""
2
+
3
+ from importlib import metadata
4
+ from .assembler import assemble
5
+ from .assembler import assemble_auto
6
+ from .assembler import clear_schema_cache
7
+ from .parser import parse
8
+ from .parser import parse_auto
9
+ from .parser import validate_schema
10
+ from .schema_registry import get_schema_path
11
+ from .schema_registry import list_schema_families
12
+ from .schema_registry import list_schema_versions
13
+
14
+ try: # pragma: no cover - import failure handled for graceful degradation
15
+ from . import parser # noqa: F401 # import for side effect and re-export
16
+ except Exception: # ImportError and others
17
+ parser = None
18
+
19
+
20
+ try: # pragma: no cover - gracefully handle missing distribution
21
+ __version__ = metadata.version("parseo")
22
+ except metadata.PackageNotFoundError: # pragma: no cover - defensive
23
+ __version__ = "unknown"
24
+
25
+
26
+ def info() -> dict[str, str]:
27
+ """Return information about the installed :mod:`parseo` package.
28
+
29
+ Returns
30
+ -------
31
+ dict[str, str]
32
+ A dictionary containing the installed version under the ``"version"`` key.
33
+ """
34
+
35
+ return {"version": __version__}
36
+
37
+
38
+ __all__ = [
39
+ "parse",
40
+ "parse_auto",
41
+ "assemble",
42
+ "assemble_auto",
43
+ "clear_schema_cache",
44
+ "validate_schema",
45
+ "list_schema_families",
46
+ "list_schema_versions",
47
+ "get_schema_path",
48
+ "info",
49
+ "__version__",
50
+ ]
51
+
52
+ if parser is not None:
53
+ __all__.append("parser")
parseo/_epsg_lookup.py ADDED
@@ -0,0 +1,145 @@
1
+ """Lookup helpers for deriving EPSG codes from tile identifiers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import math
7
+ from typing import Optional
8
+
9
+
10
+ _MGRS_LATITUDE_BANDS = {
11
+ "C": "south",
12
+ "D": "south",
13
+ "E": "south",
14
+ "F": "south",
15
+ "G": "south",
16
+ "H": "south",
17
+ "J": "south",
18
+ "K": "south",
19
+ "L": "south",
20
+ "M": "south",
21
+ "N": "north",
22
+ "P": "north",
23
+ "Q": "north",
24
+ "R": "north",
25
+ "S": "north",
26
+ "T": "north",
27
+ "U": "north",
28
+ "V": "north",
29
+ "W": "north",
30
+ "X": "north",
31
+ }
32
+
33
+
34
+ def mgrs_tile_to_epsg(tile: str) -> Optional[str]:
35
+ """Return the EPSG code associated with a Sentinel-2 MGRS tile.
36
+
37
+ Parameters
38
+ ----------
39
+ tile:
40
+ The tile identifier (e.g. ``"T32TNS"``).
41
+ """
42
+
43
+ if not isinstance(tile, str) or len(tile) < 4:
44
+ return None
45
+
46
+ tile = tile.strip().upper()
47
+ if not tile.startswith("T"):
48
+ return None
49
+
50
+ zone_part = tile[1:3]
51
+ try:
52
+ zone = int(zone_part)
53
+ except ValueError:
54
+ return None
55
+ if not 1 <= zone <= 60:
56
+ return None
57
+
58
+ band = tile[3]
59
+ hemisphere = _MGRS_LATITUDE_BANDS.get(band)
60
+ if hemisphere is None:
61
+ return None
62
+
63
+ if hemisphere == "north":
64
+ epsg = 32600 + zone
65
+ else:
66
+ epsg = 32700 + zone
67
+
68
+ return f"{epsg:05d}"
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class _WRSOrbitConstants:
73
+ """Constants derived from the WRS-2 orbital configuration."""
74
+
75
+ orbital_period_days: float = 16.0
76
+ paths: int = 233
77
+ rows: int = 248
78
+ inclination_deg: float = 98.2
79
+
80
+
81
+ _WRS_CONSTANTS = _WRSOrbitConstants()
82
+
83
+
84
+ def _path_to_longitude(path: int) -> float:
85
+ """Approximate the longitude of the descending node for *path* (degrees)."""
86
+
87
+ fraction = (path - 1) / _WRS_CONSTANTS.paths
88
+ longitude = (fraction * 360.0) % 360.0
89
+ if longitude > 180.0:
90
+ longitude -= 360.0
91
+ return longitude
92
+
93
+
94
+ def _row_to_latitude(row: int) -> float:
95
+ """Approximate the latitude of the scene centre for *row* (degrees)."""
96
+
97
+ # Rows increase from north to south. We approximate the relationship using
98
+ # a linear fit anchored at the documented WRS limits (81°N and 81°S).
99
+ total_rows = _WRS_CONSTANTS.rows
100
+ span = 162.0 # 81°N to 81°S
101
+ step = span / (total_rows - 1)
102
+ return 81.0 - (row - 1) * step
103
+
104
+
105
+ def landsat_path_row_to_epsg(path: str, row: str) -> Optional[str]:
106
+ """Return the EPSG code inferred from a Landsat WRS path/row pair."""
107
+
108
+ try:
109
+ path_num = int(path)
110
+ row_num = int(row)
111
+ except (TypeError, ValueError):
112
+ return None
113
+
114
+ if not 1 <= path_num <= _WRS_CONSTANTS.paths:
115
+ return None
116
+ if not 1 <= row_num <= _WRS_CONSTANTS.rows:
117
+ return None
118
+
119
+ longitude = _path_to_longitude(path_num)
120
+ latitude = _row_to_latitude(row_num)
121
+
122
+ zone = int(math.floor((longitude + 180.0) / 6.0)) + 1
123
+ zone = max(1, min(zone, 60))
124
+
125
+ # Special handling for Norway and Svalbard following the UTM specification.
126
+ if 56.0 <= latitude < 64.0 and 3.0 <= longitude < 12.0:
127
+ zone = 32
128
+ if 72.0 <= latitude <= 84.0:
129
+ if 0.0 <= longitude < 9.0:
130
+ zone = 31
131
+ elif 9.0 <= longitude < 21.0:
132
+ zone = 33
133
+ elif 21.0 <= longitude < 33.0:
134
+ zone = 35
135
+ elif 33.0 <= longitude < 42.0:
136
+ zone = 37
137
+
138
+ hemisphere = "north" if latitude >= 0.0 else "south"
139
+ if hemisphere == "north":
140
+ epsg = 32600 + zone
141
+ else:
142
+ epsg = 32700 + zone
143
+
144
+ return f"{epsg:05d}"
145
+
@@ -0,0 +1,219 @@
1
+ """Utilities for translating between schema tokens and STAC values."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from typing import Any
7
+ from typing import Dict
8
+ from typing import Mapping
9
+ from typing import Optional
10
+
11
+ from ._epsg_lookup import landsat_path_row_to_epsg
12
+ from ._epsg_lookup import mgrs_tile_to_epsg
13
+ from ._tile_systems import TileSystem
14
+ from ._tile_systems import detect_tile_system
15
+ from ._tile_systems import normalize_tile
16
+
17
+
18
+ @dataclass(frozen=True)
19
+ class FieldMapping:
20
+ """Represents a mapping between a filename token and STAC fields."""
21
+
22
+ preserve_as: Optional[str]
23
+ token_map: Dict[str, Dict[str, Any]]
24
+
25
+
26
+ def _normalize_mapping_values(values: Mapping[str, Any]) -> Dict[str, Dict[str, Any]]:
27
+ normalized: Dict[str, Dict[str, Any]] = {}
28
+ for token, targets in values.items():
29
+ if not isinstance(targets, Mapping):
30
+ continue
31
+ normalized[str(token)] = {str(k): v for k, v in targets.items()}
32
+ return normalized
33
+
34
+
35
+ def get_schema_field_mappings(schema: Mapping[str, Any]) -> Dict[str, FieldMapping]:
36
+ """Extract field mappings declared in *schema*."""
37
+
38
+ fields = schema.get("fields", {})
39
+ if not isinstance(fields, Mapping):
40
+ return {}
41
+
42
+ mappings: Dict[str, FieldMapping] = {}
43
+ for field_name, spec in fields.items():
44
+ if not isinstance(spec, Mapping):
45
+ continue
46
+ raw_map = spec.get("stac_map")
47
+ if not isinstance(raw_map, Mapping):
48
+ continue
49
+
50
+ has_explicit_values = "values" in raw_map
51
+ preserve_raw = raw_map.get("preserve_original_as")
52
+ preserve_as: Optional[str]
53
+
54
+ if isinstance(preserve_raw, str):
55
+ preserve_as = preserve_raw.strip() or None
56
+ elif isinstance(preserve_raw, bool):
57
+ preserve_as = f"{field_name}_code" if preserve_raw else None
58
+ elif preserve_raw is not None:
59
+ preserve_as = str(preserve_raw).strip() or None
60
+ else:
61
+ preserve_as = None if has_explicit_values else f"{field_name}_code"
62
+
63
+ values: Any
64
+ if "values" in raw_map:
65
+ values = raw_map.get("values")
66
+ else:
67
+ values = raw_map
68
+ if not isinstance(values, Mapping):
69
+ continue
70
+
71
+ normalized = _normalize_mapping_values(values)
72
+ if not normalized:
73
+ continue
74
+
75
+ mappings[str(field_name)] = FieldMapping(preserve_as=preserve_as, token_map=normalized)
76
+
77
+ return mappings
78
+
79
+
80
+ def _augment_with_tile_variants(fields: Dict[str, Any]) -> Dict[str, Any]:
81
+ enriched = dict(fields)
82
+
83
+ mgrs_value = enriched.get("mgrs_tile")
84
+ if isinstance(mgrs_value, str) and mgrs_value and "tile_id" not in enriched:
85
+ enriched["tile_id"] = mgrs_value
86
+
87
+ for candidate in ("tile", "tile_id"):
88
+ value = enriched.get(candidate)
89
+ if not isinstance(value, str):
90
+ continue
91
+
92
+ system = detect_tile_system(value)
93
+ if system is None:
94
+ continue
95
+
96
+ normalized = normalize_tile(value)
97
+ enriched[candidate] = normalized
98
+
99
+ if system is TileSystem.MGRS and "tile_id" not in enriched:
100
+ enriched["tile_id"] = normalized
101
+ if system is TileSystem.EEA:
102
+ if "tile_id" not in enriched:
103
+ enriched["tile_id"] = normalized
104
+ if "epsg_code" not in enriched:
105
+ enriched["epsg_code"] = "03035"
106
+
107
+ enriched.pop("mgrs_tile", None)
108
+ return enriched
109
+
110
+
111
+ def _backfill_tile_tokens(
112
+ fields: Dict[str, Any], schema: Mapping[str, Any]
113
+ ) -> Dict[str, Any]:
114
+ translated = dict(fields)
115
+ specs = schema.get("fields", {})
116
+
117
+ if "tile" in specs and "tile" not in translated:
118
+ for key in ("tile_id", "mgrs_tile"):
119
+ value = translated.get(key)
120
+ if isinstance(value, str) and value:
121
+ translated["tile"] = normalize_tile(value)
122
+ break
123
+
124
+ if "tile_id" in specs and "tile_id" not in translated:
125
+ for key in ("tile", "mgrs_tile"):
126
+ value = translated.get(key)
127
+ if isinstance(value, str) and value:
128
+ translated["tile_id"] = normalize_tile(value)
129
+ break
130
+
131
+ translated.pop("mgrs_tile", None)
132
+ return translated
133
+
134
+
135
+ def apply_schema_mappings(
136
+ extracted: Dict[str, Any], schema: Mapping[str, Any]
137
+ ) -> Dict[str, Any]:
138
+ """Augment *extracted* fields with STAC values defined in *schema*."""
139
+
140
+ mappings = get_schema_field_mappings(schema)
141
+
142
+ enriched = _augment_with_tile_variants(extracted)
143
+ for field_name, mapping in mappings.items():
144
+ token = extracted.get(field_name)
145
+ if token is None:
146
+ continue
147
+
148
+ if mapping.preserve_as:
149
+ enriched[mapping.preserve_as] = token
150
+
151
+ targets = mapping.token_map.get(str(token))
152
+ if not isinstance(targets, Mapping):
153
+ continue
154
+
155
+ for target_field, target_value in targets.items():
156
+ enriched[target_field] = target_value
157
+
158
+ schema_id = str(schema.get("schema_id", "")).lower()
159
+
160
+ if "epsg_code" not in enriched:
161
+ tile = enriched.get("tile_id")
162
+ if (
163
+ tile
164
+ and "sentinel:s2" in schema_id
165
+ and detect_tile_system(str(tile)) is TileSystem.MGRS
166
+ ):
167
+ epsg = mgrs_tile_to_epsg(str(tile))
168
+ if epsg:
169
+ enriched["epsg_code"] = epsg
170
+
171
+ if "epsg_code" not in enriched:
172
+ path = enriched.get("wrs_path")
173
+ row = enriched.get("wrs_row")
174
+ if path and row and schema_id.startswith("usgs:landsat"):
175
+ epsg = landsat_path_row_to_epsg(str(path), str(row))
176
+ if epsg:
177
+ enriched["epsg_code"] = epsg
178
+
179
+ return enriched
180
+
181
+
182
+ def translate_fields_to_tokens(
183
+ fields: Dict[str, Any], schema: Mapping[str, Any]
184
+ ) -> Dict[str, Any]:
185
+ """Translate STAC field values in *fields* back to schema tokens."""
186
+
187
+ mappings = get_schema_field_mappings(schema)
188
+ if not mappings:
189
+ return fields
190
+
191
+ translated = _augment_with_tile_variants(fields)
192
+ for field_name, mapping in mappings.items():
193
+ token: Any = None
194
+
195
+ preserved = None
196
+ if mapping.preserve_as:
197
+ preserved = fields.get(mapping.preserve_as)
198
+ if preserved not in (None, ""):
199
+ token = str(preserved)
200
+
201
+ if token is None:
202
+ current_value = fields.get(field_name)
203
+ if isinstance(current_value, str) and current_value in mapping.token_map:
204
+ token = current_value
205
+
206
+ if token is None:
207
+ for candidate, targets in mapping.token_map.items():
208
+ if all(fields.get(k) == v for k, v in targets.items()):
209
+ token = candidate
210
+ break
211
+
212
+ if token is None:
213
+ continue
214
+
215
+ translated[field_name] = token
216
+
217
+ translated = _backfill_tile_tokens(translated, schema)
218
+ return translated
219
+
parseo/_json.py ADDED
@@ -0,0 +1,14 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Any
6
+ from typing import Dict
7
+ from typing import Union
8
+
9
+
10
+ def load_json(path: Union[str, Path]) -> Dict[str, Any]:
11
+ """Load a JSON file, handling optional UTF-8 BOM."""
12
+ p = Path(path)
13
+ text = p.read_text(encoding="utf-8-sig")
14
+ return json.loads(text)
@@ -0,0 +1,51 @@
1
+ """Helpers for recognizing spatial tile identifier systems."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from enum import Enum
7
+ from typing import Optional
8
+
9
+
10
+ class TileSystem(str, Enum):
11
+ """Known spatial tiling systems."""
12
+
13
+ MGRS = "mgrs"
14
+ EEA = "eea"
15
+
16
+
17
+ _MGRS_PATTERN = re.compile(r"^T\d{2}[C-HJ-NP-X][A-Z]{2}$")
18
+ _EEA_PATTERN = re.compile(r"^[EW]\d{2,3}[NS]\d{2,3}$")
19
+
20
+
21
+ def detect_tile_system(tile: str) -> Optional[TileSystem]:
22
+ """Return the :class:`TileSystem` matching *tile*, if any."""
23
+
24
+ if not isinstance(tile, str):
25
+ return None
26
+
27
+ candidate = tile.strip().upper()
28
+ if not candidate:
29
+ return None
30
+
31
+ if _MGRS_PATTERN.fullmatch(candidate):
32
+ return TileSystem.MGRS
33
+
34
+ if _EEA_PATTERN.fullmatch(candidate):
35
+ return TileSystem.EEA
36
+
37
+ return None
38
+
39
+
40
+ def normalize_tile(tile: str) -> str:
41
+ """Return a normalised representation of *tile*.
42
+
43
+ MGRS and EEA tiles are uppercased. Tiles from unrecognised systems
44
+ (e.g., MODIS sinusoidal ``h18v04``) are returned as-is.
45
+ """
46
+
47
+ t = tile.strip()
48
+ if detect_tile_system(t):
49
+ return t.upper()
50
+ return t
51
+
parseo/assembler.py ADDED
@@ -0,0 +1,225 @@
1
+ # src/parseo/assembler.py
2
+ from __future__ import annotations
3
+
4
+ from functools import lru_cache
5
+ from importlib.resources import as_file
6
+ from importlib.resources import files
7
+ from pathlib import Path
8
+ import re
9
+ from typing import Any
10
+ from typing import Dict
11
+ from typing import Union
12
+
13
+ from ._field_mappings import translate_fields_to_tokens
14
+ from .parser import _get_fields_order
15
+ from .parser import _pattern_from_schema
16
+ from .schema_registry import _load_json_from_path
17
+ from .schema_registry import get_schema_path
18
+ from .schema_registry import SCHEMAS_ROOT
19
+ from .template import _field_regex
20
+ from .template import compile_template
21
+
22
+
23
+ def clear_schema_cache() -> None:
24
+ """Clear all cached schema data."""
25
+ _load_json_from_path.cache_clear()
26
+ from .parser import _SCHEMA_PATTERN_CACHE, _SCHEMA_ORDER_CACHE
27
+
28
+ _SCHEMA_PATTERN_CACHE.clear()
29
+ _SCHEMA_ORDER_CACHE.clear()
30
+
31
+
32
+ def _assemble_from_template(template: str, fields: Dict[str, Any]) -> str:
33
+ """Render *template* using *fields*.
34
+
35
+ Optional segments denoted by ``[ ... ]`` are dropped if any enclosed field
36
+ is missing. ``{field}`` placeholders are replaced by values from *fields*.
37
+ """
38
+
39
+ def render(segment: str) -> str:
40
+ result = ""
41
+ i = 0
42
+ while i < len(segment):
43
+ ch = segment[i]
44
+ if ch == "{":
45
+ j = segment.index("}", i)
46
+ name = segment[i + 1 : j]
47
+ if name not in fields:
48
+ raise KeyError(name)
49
+ result += str(fields[name])
50
+ i = j + 1
51
+ elif ch == "[":
52
+ depth = 1
53
+ j = i + 1
54
+ while j < len(segment) and depth:
55
+ if segment[j] == "[":
56
+ depth += 1
57
+ elif segment[j] == "]":
58
+ depth -= 1
59
+ j += 1
60
+ inner = segment[i + 1 : j - 1]
61
+ try:
62
+ result += render(inner)
63
+ except KeyError:
64
+ pass
65
+ i = j
66
+ else:
67
+ result += ch
68
+ i += 1
69
+ return result
70
+
71
+ return render(template)
72
+
73
+
74
+ def _assemble_schema(schema_path: Union[str, Path], fields: Dict[str, Any]) -> str:
75
+ """Assemble a filename using a JSON schema.
76
+
77
+ Schemas must define a ``template`` string following parseo's mini-template
78
+ syntax. The template is rendered using the provided *fields* and optional
79
+ segments are dropped if any enclosed field is missing.
80
+ """
81
+
82
+ sch = _load_json_from_path(schema_path)
83
+ prepared_fields = translate_fields_to_tokens(fields, sch)
84
+
85
+ template = sch.get("template")
86
+ if not isinstance(template, str):
87
+ raise ValueError(f"Schema {schema_path} missing 'template' string.")
88
+
89
+ epsg_value = prepared_fields.get("epsg_code")
90
+ if (
91
+ isinstance(epsg_value, str)
92
+ and "EPSG{epsg_code}" in template
93
+ and epsg_value
94
+ ):
95
+ prepared_fields["epsg_code"] = epsg_value.lstrip("0") or "0"
96
+
97
+ # Validate provided fields against schema definitions
98
+ specs = sch.get("fields", {})
99
+ for name, value in prepared_fields.items():
100
+ spec = specs.get(name)
101
+ if not spec:
102
+ continue
103
+ if "enum" in spec and str(value) not in spec["enum"]:
104
+ raise ValueError(
105
+ f"Field '{name}' must be one of {spec['enum']}, got {value!r}."
106
+ )
107
+ if "pattern" in spec:
108
+ pattern = _field_regex({"pattern": spec["pattern"]})
109
+ regex = re.compile(f"^{pattern}$")
110
+ if not regex.match(str(value)):
111
+ raise ValueError(
112
+ f"Field '{name}' with value {value!r} does not match pattern {spec['pattern']}."
113
+ )
114
+
115
+ try:
116
+ return _assemble_from_template(template, prepared_fields)
117
+ except KeyError as exc:
118
+ name = exc.args[0]
119
+ raise ValueError(
120
+ f"Missing field '{name}' for schema {schema_path}"
121
+ ) from exc
122
+
123
+
124
+ def assemble(
125
+ fields: Dict[str, Any],
126
+ family: Union[str, None] = None,
127
+ version: Union[str, None] = None,
128
+ schema_path: Union[str, Path, None] = None,
129
+ ) -> str:
130
+ """Assemble a filename from *fields*.
131
+
132
+ Provide either a *schema_path* or a schema *family* (with optional
133
+ *version*) to select the schema. If neither is given the schema is
134
+ auto-selected based on the supplied *fields*.
135
+ """
136
+
137
+ if schema_path is not None:
138
+ return _assemble_schema(schema_path, fields)
139
+ if family is not None:
140
+ resolved = get_schema_path(family, version=version)
141
+ return _assemble_schema(resolved, fields)
142
+ return assemble_auto(fields)
143
+
144
+
145
+ def _iter_schema_paths() -> list[Path]:
146
+ """Return all packaged schema JSON paths."""
147
+ base = files(__package__).joinpath(SCHEMAS_ROOT)
148
+ with as_file(base) as bp:
149
+ root = Path(bp)
150
+ return list(root.rglob("*.json"))
151
+
152
+
153
+ def _select_schema_by_first_compulsory(fields: Dict[str, Any]) -> Path:
154
+ """Select the most appropriate schema based on provided fields.
155
+
156
+ Eligibility requires the user to provide the first compulsory field as
157
+ derived from the schema's template. Among eligible schemas the one with the
158
+ largest overlap of provided keys is chosen. A longer field order acts as a
159
+ tie breaker.
160
+ """
161
+ best: Union[tuple[int, int, str], None] = None
162
+ best_path: Union[Path, None] = None
163
+ seen_first_keys: set[str] = set()
164
+
165
+ for p in _iter_schema_paths():
166
+ try:
167
+ sch = _load_json_from_path(p)
168
+ except (OSError, ValueError):
169
+ continue
170
+
171
+ template = sch.get("template")
172
+ if not isinstance(template, str):
173
+ continue
174
+ _pattern_from_schema(sch)
175
+ order = _get_fields_order(sch)
176
+ if not order:
177
+ continue
178
+
179
+ specs = sch.get("fields", {})
180
+ first_key = None
181
+ for name in order:
182
+ spec = specs.get(name, {})
183
+ enums = spec.get("enum")
184
+ field_val = fields.get(name)
185
+ if enums is not None:
186
+ enums = [str(e) for e in enums]
187
+ if len(enums) == 1 and field_val == enums[0]:
188
+ continue
189
+ if field_val is not None and field_val not in enums:
190
+ first_key = None
191
+ break
192
+ first_key = name
193
+ break
194
+ else:
195
+ first_key = name
196
+ break
197
+ if not first_key:
198
+ continue
199
+
200
+ seen_first_keys.add(first_key)
201
+
202
+ if first_key not in fields:
203
+ continue
204
+
205
+ overlap = sum(1 for k in fields.keys() if k in order)
206
+ key = (overlap, len(order), str(p))
207
+ if best is None or key > best:
208
+ best = key
209
+ best_path = p
210
+
211
+ if not best_path:
212
+ sample = ", ".join(sorted(seen_first_keys)) or "<no schemas found>"
213
+ raise ValueError(
214
+ "Could not select a schema. "
215
+ "Include the schema's FIRST compulsory field among your inputs.\n"
216
+ f"Examples of first fields from packaged schemas: {sample}"
217
+ )
218
+
219
+ return best_path
220
+
221
+
222
+ def assemble_auto(fields: Dict[str, Any]) -> str:
223
+ """Assemble a filename by auto-selecting the appropriate schema."""
224
+ schema_path = _select_schema_by_first_compulsory(fields)
225
+ return _assemble_schema(str(schema_path), fields)