parseo 0.4.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parseo/__init__.py +53 -0
- parseo/_epsg_lookup.py +145 -0
- parseo/_field_mappings.py +219 -0
- parseo/_json.py +14 -0
- parseo/_tile_systems.py +51 -0
- parseo/assembler.py +225 -0
- parseo/cli.py +383 -0
- parseo/parser.py +784 -0
- parseo/schema_registry.py +246 -0
- parseo/schemas/copernicus/clms/clc/clc_filename_v1_0_0.json +48 -0
- parseo/schemas/copernicus/clms/clc/clc_filename_v1_1_0.json +65 -0
- parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_0.json +90 -0
- parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_1.json +80 -0
- parseo/schemas/copernicus/clms/egms/egms_gnss_model_filename_v0_0_0.json +46 -0
- parseo/schemas/copernicus/clms/egms/egms_l2a_product_filename_v0_0_0.json +71 -0
- parseo/schemas/copernicus/clms/egms/egms_l3_velocity_grid_filename_v0_0_0.json +64 -0
- parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_0.json +115 -0
- parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_1.json +103 -0
- parseo/schemas/copernicus/clms/hr-vpp/st_filename_v0_0_0.json +82 -0
- parseo/schemas/copernicus/clms/hr-vpp/vi_filename_v0_0_0.json +89 -0
- parseo/schemas/copernicus/clms/hr-vpp/vpp_filename_v0_0_0.json +109 -0
- parseo/schemas/copernicus/clms/hr-wsi/cc_filename_v0_0_0.json +84 -0
- parseo/schemas/copernicus/clms/hr-wsi/fsc_filename_v0_0_0.json +88 -0
- parseo/schemas/copernicus/clms/hr-wsi/gfsc_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/icd_filename_v0_0_0.json +96 -0
- parseo/schemas/copernicus/clms/hr-wsi/sp_filename_v0_0_0.json +177 -0
- parseo/schemas/copernicus/clms/hr-wsi/sws_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/wcd_filename_v0_0_0.json +96 -0
- parseo/schemas/copernicus/clms/hr-wsi/wds_filename_v0_0_0.json +83 -0
- parseo/schemas/copernicus/clms/hr-wsi/wic_comb_filename_v0_0_0.json +82 -0
- parseo/schemas/copernicus/clms/hr-wsi/wic_filename_v0_0_0.json +90 -0
- parseo/schemas/copernicus/clms/hrl/fty_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/gra_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/ibu_filename_v0_0_0.json +57 -0
- parseo/schemas/copernicus/clms/hrl/imp_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_0.json +73 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_1.json +89 -0
- parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_2.json +89 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_1.json +49 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_2.json +93 -0
- parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_3.json +99 -0
- parseo/schemas/copernicus/clms/hrl/tcd_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/hrl/vlcc_filename_v0_0_0.json +105 -0
- parseo/schemas/copernicus/clms/hrl/waw_filename_v0_0_0.json +48 -0
- parseo/schemas/copernicus/clms/n2k/n2k_filename_v1_0_0.json +43 -0
- parseo/schemas/copernicus/clms/pa/pa_filename_v1_0_0.json +99 -0
- parseo/schemas/copernicus/clms/riparian-zones/rpz_filename_v0_0_0.json +50 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_dhm_filename_v0_0_0.json +53 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_filename_v0_0_0.json +105 -0
- parseo/schemas/copernicus/clms/urban-atlas/ua_stl_filename_v0_0_0.json +89 -0
- parseo/schemas/copernicus/sentinel/s1_filename_v1_0_0.json +78 -0
- parseo/schemas/copernicus/sentinel/s2_filename_v1_0_0.json +62 -0
- parseo/schemas/copernicus/sentinel/s3_filename_v1_0_0.json +22 -0
- parseo/schemas/copernicus/sentinel/s4_filename_v1_0_0.json +21 -0
- parseo/schemas/copernicus/sentinel/s5p_filename_v1_0_0.json +21 -0
- parseo/schemas/copernicus/sentinel/s6_filename_v1_0_0.json +23 -0
- parseo/schemas/eumetsat/metop_filename_v1_0_0.json +68 -0
- parseo/schemas/eumetsat/mtg_filename_v1_0_0.json +65 -0
- parseo/schemas/nasa/modis_filename_v1_0_0.json +54 -0
- parseo/schemas/usgs/landsat/landsat_filename_v1_0_0.json +50 -0
- parseo/stac_http.py +267 -0
- parseo/stac_scraper.py +115 -0
- parseo/template.py +78 -0
- parseo-0.4.4.dist-info/METADATA +402 -0
- parseo-0.4.4.dist-info/RECORD +70 -0
- parseo-0.4.4.dist-info/WHEEL +5 -0
- parseo-0.4.4.dist-info/entry_points.txt +2 -0
- parseo-0.4.4.dist-info/licenses/LICENSE.txt +287 -0
- parseo-0.4.4.dist-info/top_level.txt +1 -0
parseo/parser.py
ADDED
|
@@ -0,0 +1,784 @@
|
|
|
1
|
+
# src/parseo/parser.py
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from functools import lru_cache
|
|
6
|
+
from importlib.resources import as_file
|
|
7
|
+
from importlib.resources import files
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
import re
|
|
10
|
+
from typing import Any
|
|
11
|
+
from typing import Dict
|
|
12
|
+
from typing import Iterable
|
|
13
|
+
from typing import Iterator
|
|
14
|
+
from typing import Optional
|
|
15
|
+
from typing import Union
|
|
16
|
+
|
|
17
|
+
from ._field_mappings import apply_schema_mappings
|
|
18
|
+
from .schema_registry import _discover_family_info
|
|
19
|
+
from .schema_registry import _get_schema_paths
|
|
20
|
+
from .schema_registry import _load_json_from_path
|
|
21
|
+
from .schema_registry import get_schema_path
|
|
22
|
+
from .schema_registry import list_schema_families
|
|
23
|
+
from .schema_registry import SCHEMAS_ROOT
|
|
24
|
+
from .schema_registry import to_display_family
|
|
25
|
+
from .template import _field_regex
|
|
26
|
+
from .template import compile_template
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class ParseResult:
|
|
32
|
+
"""Result of a parsing attempt."""
|
|
33
|
+
valid: bool
|
|
34
|
+
fields: Dict[str, str]
|
|
35
|
+
version: Optional[str] = None
|
|
36
|
+
status: Optional[str] = None
|
|
37
|
+
match_family: Optional[str] = None # e.g., "S1", "S2", "LANDSAT"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class ParseError(Exception):
|
|
42
|
+
"""Raised when a filename nearly matches a schema but fails on a field."""
|
|
43
|
+
|
|
44
|
+
field: str
|
|
45
|
+
expected: str
|
|
46
|
+
value: str
|
|
47
|
+
schema_id: Optional[str] = None
|
|
48
|
+
match_family: Optional[str] = None
|
|
49
|
+
|
|
50
|
+
def __post_init__(self) -> None:
|
|
51
|
+
super().__init__(str(self))
|
|
52
|
+
|
|
53
|
+
def __str__(self) -> str: # pragma: no cover - trivial
|
|
54
|
+
base = (
|
|
55
|
+
f"Invalid value '{self.value}' for field '{self.field}': expected {self.expected}"
|
|
56
|
+
)
|
|
57
|
+
extras = []
|
|
58
|
+
if self.match_family:
|
|
59
|
+
extras.append(f"schema family '{self.match_family}'")
|
|
60
|
+
if self.schema_id:
|
|
61
|
+
extras.append(f"schema '{self.schema_id}'")
|
|
62
|
+
if extras:
|
|
63
|
+
base = f"{base} (nearest match: {', '.join(extras)})"
|
|
64
|
+
return base
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# ---------------------------
|
|
68
|
+
# Core helpers
|
|
69
|
+
# ---------------------------
|
|
70
|
+
|
|
71
|
+
# Per-schema caches — avoid mutating lru_cache'd schema dicts
|
|
72
|
+
_SCHEMA_PATTERN_CACHE: dict[int, str] = {}
|
|
73
|
+
_SCHEMA_ORDER_CACHE: dict[int, list[str]] = {}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@lru_cache(maxsize=512)
|
|
77
|
+
def _compile_pattern(pattern: str, ignore_case: bool = False) -> re.Pattern:
|
|
78
|
+
flags = re.IGNORECASE if ignore_case else 0
|
|
79
|
+
return re.compile(pattern, flags)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _pattern_from_schema(schema: Dict) -> Optional[str]:
|
|
83
|
+
"""Return a compiled regex pattern derived from a schema's template.
|
|
84
|
+
|
|
85
|
+
The compiled pattern and field order are cached in module-level dicts
|
|
86
|
+
keyed by ``id(schema)`` to avoid mutating the lru_cache'd schema object,
|
|
87
|
+
which would be thread-unsafe.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
sid = id(schema)
|
|
91
|
+
cached = _SCHEMA_PATTERN_CACHE.get(sid)
|
|
92
|
+
if cached is not None:
|
|
93
|
+
return cached
|
|
94
|
+
|
|
95
|
+
template = schema.get("template")
|
|
96
|
+
if isinstance(template, str):
|
|
97
|
+
fields = schema.get("fields", {})
|
|
98
|
+
pattern, order = compile_template(template, fields)
|
|
99
|
+
_SCHEMA_PATTERN_CACHE[sid] = pattern
|
|
100
|
+
if order:
|
|
101
|
+
_SCHEMA_ORDER_CACHE.setdefault(sid, order)
|
|
102
|
+
return pattern
|
|
103
|
+
|
|
104
|
+
return None
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _get_fields_order(schema: Dict) -> list[str]:
|
|
108
|
+
"""Return the field order for *schema*, from cache or the schema itself."""
|
|
109
|
+
return _SCHEMA_ORDER_CACHE.get(id(schema)) or schema.get("fields_order", [])
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _match_filename(name: str, schema: Dict, ignore_case: bool = False) -> Optional[re.Match]:
|
|
113
|
+
patt = _pattern_from_schema(schema)
|
|
114
|
+
if not isinstance(patt, str) or not patt:
|
|
115
|
+
return None
|
|
116
|
+
rx = _compile_pattern(patt, ignore_case=ignore_case)
|
|
117
|
+
return rx.match(name)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _family_from_path(path: Path, info: Dict[str, Any]) -> Optional[str]:
|
|
121
|
+
for fam_name, meta in info.items():
|
|
122
|
+
if getattr(meta, "schema_path", None) == path:
|
|
123
|
+
return fam_name
|
|
124
|
+
versions = getattr(meta, "versions", {})
|
|
125
|
+
for ver_path, _status in getattr(versions, "values", lambda: [])():
|
|
126
|
+
if ver_path == path:
|
|
127
|
+
return fam_name
|
|
128
|
+
return None
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _guess_product_family(name: str, info: Dict[str, Any]) -> Optional[str]:
|
|
132
|
+
"""Return a schema family hint derived from *name*."""
|
|
133
|
+
|
|
134
|
+
upper_name = name.upper()
|
|
135
|
+
for fam, meta in info.items():
|
|
136
|
+
tokens = getattr(meta, "tokens", ())
|
|
137
|
+
if any(upper_name.startswith(tok) for tok in tokens):
|
|
138
|
+
return fam
|
|
139
|
+
return None
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _normalize_epsg_fields(fields: Dict[str, Any]) -> Dict[str, Any]:
|
|
143
|
+
"""Normalize EPSG-related fields to consistently use 5-digit codes."""
|
|
144
|
+
|
|
145
|
+
normalized = {k: v for k, v in fields.items() if v is not None}
|
|
146
|
+
for key, value in list(normalized.items()):
|
|
147
|
+
if not isinstance(key, str):
|
|
148
|
+
continue
|
|
149
|
+
key_lower = key.lower()
|
|
150
|
+
if "epsg" not in key_lower and key_lower not in {"tile", "tile_id"}:
|
|
151
|
+
continue
|
|
152
|
+
if isinstance(value, str) and value.isdigit() and len(value) in {4, 5}:
|
|
153
|
+
normalized[key] = value.zfill(5)
|
|
154
|
+
return normalized
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _extract_fields(name: str, schema: Dict, ignore_case: bool = False) -> Dict[str, str]:
|
|
158
|
+
"""
|
|
159
|
+
Extract named groups as fields from 'name' using the schema's regex.
|
|
160
|
+
If the regex doesn't match, return an empty dict.
|
|
161
|
+
"""
|
|
162
|
+
m = _match_filename(name, schema, ignore_case=ignore_case)
|
|
163
|
+
if not m:
|
|
164
|
+
return {}
|
|
165
|
+
extracted = m.groupdict()
|
|
166
|
+
enriched = apply_schema_mappings(extracted, schema)
|
|
167
|
+
return _normalize_epsg_fields(enriched)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _try_validate(name: str, schema: Dict, ignore_case: bool = False) -> bool:
|
|
171
|
+
return _match_filename(name, schema, ignore_case=ignore_case) is not None
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _attempt_parse(
|
|
175
|
+
name: str,
|
|
176
|
+
info: Dict[str, Any],
|
|
177
|
+
candidates: Iterable[Path],
|
|
178
|
+
product_hint: Optional[str],
|
|
179
|
+
ignore_case: bool = False,
|
|
180
|
+
) -> tuple[Optional[ParseResult], Optional[ParseError], Optional[Exception]]:
|
|
181
|
+
"""Attempt to parse *name* once and return result with diagnostics."""
|
|
182
|
+
|
|
183
|
+
near_miss: Optional[ParseError] = None
|
|
184
|
+
first_error: Optional[Exception] = None
|
|
185
|
+
|
|
186
|
+
hinted_meta = info.get(product_hint) if product_hint else None
|
|
187
|
+
hinted = hinted_meta.schema_path if hinted_meta else None
|
|
188
|
+
if hinted and hinted.exists():
|
|
189
|
+
try:
|
|
190
|
+
schema = _load_json_from_path(hinted)
|
|
191
|
+
canonical_family = product_hint or _family_from_path(hinted, info)
|
|
192
|
+
if _try_validate(name, schema, ignore_case=ignore_case):
|
|
193
|
+
display_family = to_display_family(canonical_family)
|
|
194
|
+
return (
|
|
195
|
+
ParseResult(
|
|
196
|
+
valid=True,
|
|
197
|
+
fields=_extract_fields(name, schema, ignore_case=ignore_case),
|
|
198
|
+
version=hinted_meta.version if hinted_meta else None,
|
|
199
|
+
status=hinted_meta.status if hinted_meta else None,
|
|
200
|
+
match_family=display_family,
|
|
201
|
+
),
|
|
202
|
+
None,
|
|
203
|
+
None,
|
|
204
|
+
)
|
|
205
|
+
mismatch = _explain_match_failure(name, schema, ignore_case=ignore_case)
|
|
206
|
+
if mismatch:
|
|
207
|
+
field, expected, value = mismatch
|
|
208
|
+
display_family = to_display_family(canonical_family)
|
|
209
|
+
near_miss = ParseError(
|
|
210
|
+
field,
|
|
211
|
+
expected,
|
|
212
|
+
value,
|
|
213
|
+
schema_id=schema.get("schema_id"),
|
|
214
|
+
match_family=display_family,
|
|
215
|
+
)
|
|
216
|
+
except ParseError as err:
|
|
217
|
+
near_miss = err
|
|
218
|
+
except Exception:
|
|
219
|
+
# Schema loading/parsing can raise OSError (I/O), ValueError
|
|
220
|
+
# (JSON), or re.error (bad pattern). Skip and try next schema.
|
|
221
|
+
pass
|
|
222
|
+
|
|
223
|
+
for p in candidates:
|
|
224
|
+
try:
|
|
225
|
+
schema = _load_json_from_path(p)
|
|
226
|
+
except Exception as exc:
|
|
227
|
+
if first_error is None:
|
|
228
|
+
first_error = exc
|
|
229
|
+
continue
|
|
230
|
+
if _try_validate(name, schema, ignore_case=ignore_case):
|
|
231
|
+
matched_family = None
|
|
232
|
+
version = None
|
|
233
|
+
status = None
|
|
234
|
+
for fam_name, meta in info.items():
|
|
235
|
+
if meta.schema_path == p:
|
|
236
|
+
matched_family = fam_name
|
|
237
|
+
version = meta.version
|
|
238
|
+
status = meta.status
|
|
239
|
+
break
|
|
240
|
+
for ver, (path, st) in meta.versions.items():
|
|
241
|
+
if path == p:
|
|
242
|
+
matched_family = fam_name
|
|
243
|
+
version = ver
|
|
244
|
+
status = st
|
|
245
|
+
break
|
|
246
|
+
if matched_family:
|
|
247
|
+
break
|
|
248
|
+
display_family = to_display_family(matched_family or product_hint)
|
|
249
|
+
return (
|
|
250
|
+
ParseResult(
|
|
251
|
+
valid=True,
|
|
252
|
+
fields=_extract_fields(name, schema, ignore_case=ignore_case),
|
|
253
|
+
version=version,
|
|
254
|
+
status=status,
|
|
255
|
+
match_family=display_family,
|
|
256
|
+
),
|
|
257
|
+
None,
|
|
258
|
+
None,
|
|
259
|
+
)
|
|
260
|
+
if near_miss is None:
|
|
261
|
+
mismatch = _explain_match_failure(name, schema, ignore_case=ignore_case)
|
|
262
|
+
if mismatch:
|
|
263
|
+
field, expected, value = mismatch
|
|
264
|
+
canonical_family = _family_from_path(p, info)
|
|
265
|
+
display_family = to_display_family(canonical_family or product_hint)
|
|
266
|
+
near_miss = ParseError(
|
|
267
|
+
field,
|
|
268
|
+
expected,
|
|
269
|
+
value,
|
|
270
|
+
schema_id=schema.get("schema_id"),
|
|
271
|
+
match_family=display_family,
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
return None, near_miss, first_error
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _named_group_spans(pattern: str) -> Dict[str, tuple[int, int]]:
|
|
278
|
+
spans: Dict[str, tuple[int, int]] = {}
|
|
279
|
+
stack: list[tuple[Optional[str], int]] = []
|
|
280
|
+
i = 0
|
|
281
|
+
in_class = False
|
|
282
|
+
while i < len(pattern):
|
|
283
|
+
ch = pattern[i]
|
|
284
|
+
if ch == "\\":
|
|
285
|
+
i += 2
|
|
286
|
+
continue
|
|
287
|
+
if in_class:
|
|
288
|
+
if ch == "]":
|
|
289
|
+
in_class = False
|
|
290
|
+
i += 1
|
|
291
|
+
continue
|
|
292
|
+
if ch == "[":
|
|
293
|
+
in_class = True
|
|
294
|
+
i += 1
|
|
295
|
+
continue
|
|
296
|
+
if ch == "(":
|
|
297
|
+
if pattern.startswith("(?P<", i):
|
|
298
|
+
j = i + 4
|
|
299
|
+
k = pattern.index(">", j)
|
|
300
|
+
name = pattern[j:k]
|
|
301
|
+
stack.append((name, i))
|
|
302
|
+
i = k + 1
|
|
303
|
+
else:
|
|
304
|
+
stack.append((None, i))
|
|
305
|
+
i += 1
|
|
306
|
+
continue
|
|
307
|
+
if ch == ")":
|
|
308
|
+
name, start = stack.pop()
|
|
309
|
+
if name:
|
|
310
|
+
spans[name] = (start, i + 1)
|
|
311
|
+
i += 1
|
|
312
|
+
continue
|
|
313
|
+
i += 1
|
|
314
|
+
return spans
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def _balanced_prefix(pattern: str, index: int) -> str:
|
|
318
|
+
"""Return the longest balanced prefix of *pattern* up to *index*."""
|
|
319
|
+
|
|
320
|
+
index = min(index, len(pattern))
|
|
321
|
+
in_class = False
|
|
322
|
+
depth = 0
|
|
323
|
+
last_balanced = 0
|
|
324
|
+
i = 0
|
|
325
|
+
while i < index:
|
|
326
|
+
ch = pattern[i]
|
|
327
|
+
if ch == "\\":
|
|
328
|
+
i += 2
|
|
329
|
+
if depth == 0:
|
|
330
|
+
last_balanced = min(i, index)
|
|
331
|
+
continue
|
|
332
|
+
if in_class:
|
|
333
|
+
if ch == "]":
|
|
334
|
+
in_class = False
|
|
335
|
+
i += 1
|
|
336
|
+
if depth == 0:
|
|
337
|
+
last_balanced = min(i, index)
|
|
338
|
+
continue
|
|
339
|
+
if ch == "[":
|
|
340
|
+
in_class = True
|
|
341
|
+
i += 1
|
|
342
|
+
continue
|
|
343
|
+
if ch == "(":
|
|
344
|
+
depth += 1
|
|
345
|
+
i += 1
|
|
346
|
+
continue
|
|
347
|
+
if ch == ")":
|
|
348
|
+
if depth > 0:
|
|
349
|
+
depth -= 1
|
|
350
|
+
i += 1
|
|
351
|
+
if depth == 0:
|
|
352
|
+
last_balanced = min(i, index)
|
|
353
|
+
continue
|
|
354
|
+
i += 1
|
|
355
|
+
if depth == 0:
|
|
356
|
+
last_balanced = min(i, index)
|
|
357
|
+
|
|
358
|
+
return pattern[:last_balanced]
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _balanced_slice(pattern: str, end: int) -> str:
|
|
362
|
+
"""Return a balanced slice of *pattern* that includes *end*."""
|
|
363
|
+
|
|
364
|
+
end = min(end, len(pattern))
|
|
365
|
+
# Track parenthesis depth up to ``end``.
|
|
366
|
+
stack: list[str] = []
|
|
367
|
+
in_class = False
|
|
368
|
+
i = 0
|
|
369
|
+
while i < end:
|
|
370
|
+
ch = pattern[i]
|
|
371
|
+
if ch == "\\":
|
|
372
|
+
i += 2
|
|
373
|
+
continue
|
|
374
|
+
if in_class:
|
|
375
|
+
if ch == "]":
|
|
376
|
+
in_class = False
|
|
377
|
+
i += 1
|
|
378
|
+
continue
|
|
379
|
+
if ch == "[":
|
|
380
|
+
in_class = True
|
|
381
|
+
i += 1
|
|
382
|
+
continue
|
|
383
|
+
if ch == "(":
|
|
384
|
+
stack.append("(")
|
|
385
|
+
i += 1
|
|
386
|
+
continue
|
|
387
|
+
if ch == ")":
|
|
388
|
+
if stack:
|
|
389
|
+
stack.pop()
|
|
390
|
+
i += 1
|
|
391
|
+
continue
|
|
392
|
+
i += 1
|
|
393
|
+
|
|
394
|
+
j = end
|
|
395
|
+
in_class_after = in_class
|
|
396
|
+
stack_after = list(stack)
|
|
397
|
+
while stack_after and j < len(pattern):
|
|
398
|
+
ch = pattern[j]
|
|
399
|
+
if ch == "\\":
|
|
400
|
+
j += 2
|
|
401
|
+
continue
|
|
402
|
+
if in_class_after:
|
|
403
|
+
if ch == "]":
|
|
404
|
+
in_class_after = False
|
|
405
|
+
j += 1
|
|
406
|
+
continue
|
|
407
|
+
if ch == "[":
|
|
408
|
+
in_class_after = True
|
|
409
|
+
j += 1
|
|
410
|
+
continue
|
|
411
|
+
if ch == "(":
|
|
412
|
+
stack_after.append("(")
|
|
413
|
+
j += 1
|
|
414
|
+
continue
|
|
415
|
+
if ch == ")":
|
|
416
|
+
stack_after.pop()
|
|
417
|
+
j += 1
|
|
418
|
+
continue
|
|
419
|
+
j += 1
|
|
420
|
+
|
|
421
|
+
# Include trailing quantifiers that modify the just-closed group.
|
|
422
|
+
while j < len(pattern):
|
|
423
|
+
ch = pattern[j]
|
|
424
|
+
if ch in "?*+":
|
|
425
|
+
j += 1
|
|
426
|
+
continue
|
|
427
|
+
if ch == "{":
|
|
428
|
+
depth = 1
|
|
429
|
+
k = j + 1
|
|
430
|
+
while k < len(pattern) and depth:
|
|
431
|
+
nxt = pattern[k]
|
|
432
|
+
if nxt == "\\":
|
|
433
|
+
k += 2
|
|
434
|
+
continue
|
|
435
|
+
if nxt == "{":
|
|
436
|
+
depth += 1
|
|
437
|
+
k += 1
|
|
438
|
+
continue
|
|
439
|
+
if nxt == "}":
|
|
440
|
+
depth -= 1
|
|
441
|
+
k += 1
|
|
442
|
+
continue
|
|
443
|
+
k += 1
|
|
444
|
+
j = k
|
|
445
|
+
if j < len(pattern) and pattern[j] == "?":
|
|
446
|
+
j += 1
|
|
447
|
+
continue
|
|
448
|
+
break
|
|
449
|
+
|
|
450
|
+
return pattern[:j]
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _explain_match_failure(name: str, schema: Dict, ignore_case: bool = False) -> Optional[tuple[str, str, str]]:
|
|
454
|
+
pattern = _pattern_from_schema(schema)
|
|
455
|
+
fields = schema.get("fields", {})
|
|
456
|
+
order = _get_fields_order(schema)
|
|
457
|
+
if not pattern or not order:
|
|
458
|
+
return None
|
|
459
|
+
spans = _named_group_spans(pattern)
|
|
460
|
+
for i, field in enumerate(order):
|
|
461
|
+
next_end = spans[order[i + 1]][1] if i + 1 < len(order) else len(pattern)
|
|
462
|
+
prefix_pat = _balanced_slice(pattern, next_end) + ".*$"
|
|
463
|
+
if not _compile_pattern(prefix_pat, ignore_case=ignore_case).match(name):
|
|
464
|
+
before_pat = _balanced_prefix(pattern, spans[field][0])
|
|
465
|
+
m_before = _compile_pattern(before_pat, ignore_case=ignore_case).match(name)
|
|
466
|
+
start_pos = len(m_before.group(0)) if m_before else 0
|
|
467
|
+
target_field = field
|
|
468
|
+
spec = fields.get(target_field, {})
|
|
469
|
+
field_rx = re.compile(_field_regex(spec))
|
|
470
|
+
m_field = field_rx.match(name[start_pos:])
|
|
471
|
+
if m_field and i + 1 < len(order):
|
|
472
|
+
next_field = order[i + 1]
|
|
473
|
+
next_spec = fields.get(next_field, {})
|
|
474
|
+
before_next = _balanced_prefix(pattern, spans[next_field][0])
|
|
475
|
+
m_before_next = _compile_pattern(before_next, ignore_case=ignore_case).match(name)
|
|
476
|
+
if m_before_next:
|
|
477
|
+
start_pos = len(m_before_next.group(0))
|
|
478
|
+
else:
|
|
479
|
+
current_end = _balanced_slice(pattern, spans[field][1])
|
|
480
|
+
m_current_end = _compile_pattern(current_end, ignore_case=ignore_case).match(name)
|
|
481
|
+
start_pos = len(m_current_end.group(0)) if m_current_end else start_pos
|
|
482
|
+
target_field = next_field
|
|
483
|
+
spec = next_spec
|
|
484
|
+
field_rx = re.compile(_field_regex(spec))
|
|
485
|
+
m_field = field_rx.match(name[start_pos:])
|
|
486
|
+
if m_field:
|
|
487
|
+
# Both the current and subsequent field values satisfy
|
|
488
|
+
# their specifications; defer to later iterations.
|
|
489
|
+
continue
|
|
490
|
+
if m_field:
|
|
491
|
+
# No informative mismatch could be identified.
|
|
492
|
+
continue
|
|
493
|
+
end_pos = len(name)
|
|
494
|
+
for sep in ["_", ".", "-"]:
|
|
495
|
+
idx = name.find(sep, start_pos)
|
|
496
|
+
if idx != -1:
|
|
497
|
+
boundary = idx if idx > start_pos else idx + 1
|
|
498
|
+
end_pos = min(end_pos, boundary)
|
|
499
|
+
value = name[start_pos:end_pos]
|
|
500
|
+
if "enum" in spec:
|
|
501
|
+
expected = f"one of {spec['enum']}"
|
|
502
|
+
elif "pattern" in spec:
|
|
503
|
+
expected = f"pattern {spec['pattern']}"
|
|
504
|
+
else:
|
|
505
|
+
expected = "a different value"
|
|
506
|
+
return target_field, expected, value
|
|
507
|
+
return None
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
# ---------------------------
|
|
511
|
+
# Public API
|
|
512
|
+
# ---------------------------
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def list_schemas(pkg: str = __package__) -> list[str]:
|
|
516
|
+
"""Return a list of available mission family names."""
|
|
517
|
+
return list_schema_families(pkg)
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def describe_schema(
|
|
521
|
+
family: str, version: Optional[str] = None, pkg: str = __package__
|
|
522
|
+
) -> dict[str, Any]:
|
|
523
|
+
"""Return schema metadata and field descriptions for ``family``.
|
|
524
|
+
|
|
525
|
+
Parameters
|
|
526
|
+
----------
|
|
527
|
+
family:
|
|
528
|
+
Mission family name (case-insensitive).
|
|
529
|
+
version:
|
|
530
|
+
Optional semantic version. When omitted, the schema marked as
|
|
531
|
+
``"current"`` for the family is used.
|
|
532
|
+
pkg:
|
|
533
|
+
Package that hosts the schemas. Defaults to the installed
|
|
534
|
+
:mod:`parseo` package.
|
|
535
|
+
"""
|
|
536
|
+
|
|
537
|
+
try:
|
|
538
|
+
schema_path = get_schema_path(family, version=version, pkg=pkg)
|
|
539
|
+
except KeyError:
|
|
540
|
+
raise
|
|
541
|
+
|
|
542
|
+
schema = _load_json_from_path(schema_path)
|
|
543
|
+
fields: Dict[str, Dict[str, Any]] = {}
|
|
544
|
+
for name, spec in schema.get("fields", {}).items():
|
|
545
|
+
if isinstance(spec, dict):
|
|
546
|
+
fields[name] = {
|
|
547
|
+
k: spec[k]
|
|
548
|
+
for k in ("type", "enum", "pattern", "description")
|
|
549
|
+
if k in spec
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
out: Dict[str, Any] = {
|
|
553
|
+
"schema_id": schema.get("schema_id"),
|
|
554
|
+
"schema_version": schema.get("schema_version"),
|
|
555
|
+
"status": schema.get("status"),
|
|
556
|
+
"description": schema.get("description"),
|
|
557
|
+
"fields": fields,
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
template = schema.get("template")
|
|
561
|
+
if isinstance(template, str):
|
|
562
|
+
out["template"] = template
|
|
563
|
+
|
|
564
|
+
examples = schema.get("examples")
|
|
565
|
+
if isinstance(examples, list):
|
|
566
|
+
out["examples"] = [e for e in examples if isinstance(e, str)]
|
|
567
|
+
|
|
568
|
+
return out
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def parse(
|
|
572
|
+
name: str,
|
|
573
|
+
schema_path: Union[str, Path, None] = None,
|
|
574
|
+
*,
|
|
575
|
+
family: Optional[str] = None,
|
|
576
|
+
version: Optional[str] = None,
|
|
577
|
+
pkg: str = __package__,
|
|
578
|
+
ignore_case: bool = False,
|
|
579
|
+
) -> ParseResult:
|
|
580
|
+
"""Parse ``name`` using a specific schema.
|
|
581
|
+
|
|
582
|
+
Parameters
|
|
583
|
+
----------
|
|
584
|
+
name:
|
|
585
|
+
Filename to parse.
|
|
586
|
+
schema_path:
|
|
587
|
+
Path to the schema JSON file. Optional when ``family`` is provided.
|
|
588
|
+
family:
|
|
589
|
+
Schema family identifier (e.g. ``"S2"``). Ignored when
|
|
590
|
+
``schema_path`` is given.
|
|
591
|
+
version:
|
|
592
|
+
Semantic version string to resolve together with ``family``.
|
|
593
|
+
pkg:
|
|
594
|
+
Package that hosts the schemas. Defaults to the installed
|
|
595
|
+
:mod:`parseo` package.
|
|
596
|
+
ignore_case:
|
|
597
|
+
When ``True``, perform case-insensitive matching. Defaults to ``False``.
|
|
598
|
+
"""
|
|
599
|
+
|
|
600
|
+
if schema_path is None:
|
|
601
|
+
if not family:
|
|
602
|
+
raise ValueError("Provide either 'schema_path' or 'family'.")
|
|
603
|
+
schema_path = get_schema_path(family, version=version, pkg=pkg)
|
|
604
|
+
|
|
605
|
+
resolved_path = Path(schema_path)
|
|
606
|
+
schema = _load_json_from_path(resolved_path)
|
|
607
|
+
|
|
608
|
+
if not _try_validate(name, schema, ignore_case=ignore_case):
|
|
609
|
+
schema_id = schema.get("schema_id") if isinstance(schema.get("schema_id"), str) else None
|
|
610
|
+
family_hint = None
|
|
611
|
+
if schema_id:
|
|
612
|
+
family_hint = schema_id.split(":")[-1]
|
|
613
|
+
elif family:
|
|
614
|
+
family_hint = family
|
|
615
|
+
display_family = to_display_family(family_hint) if family_hint else None
|
|
616
|
+
mismatch = _explain_match_failure(name, schema, ignore_case=ignore_case)
|
|
617
|
+
if mismatch:
|
|
618
|
+
field, expected, value = mismatch
|
|
619
|
+
raise ParseError(
|
|
620
|
+
field=field,
|
|
621
|
+
expected=expected,
|
|
622
|
+
value=value,
|
|
623
|
+
schema_id=schema_id,
|
|
624
|
+
match_family=display_family,
|
|
625
|
+
)
|
|
626
|
+
raise ParseError(
|
|
627
|
+
field="filename",
|
|
628
|
+
expected=f"pattern defined by schema {schema_id or resolved_path}",
|
|
629
|
+
value=name,
|
|
630
|
+
schema_id=schema_id,
|
|
631
|
+
match_family=display_family,
|
|
632
|
+
)
|
|
633
|
+
|
|
634
|
+
fields = _extract_fields(name, schema, ignore_case=ignore_case)
|
|
635
|
+
version_info = schema.get("schema_version")
|
|
636
|
+
status_info = schema.get("status")
|
|
637
|
+
schema_id = schema.get("schema_id") if isinstance(schema.get("schema_id"), str) else None
|
|
638
|
+
family_hint = None
|
|
639
|
+
if schema_id:
|
|
640
|
+
family_hint = schema_id.split(":")[-1]
|
|
641
|
+
elif family:
|
|
642
|
+
family_hint = family
|
|
643
|
+
display_family = to_display_family(family_hint) if family_hint else None
|
|
644
|
+
|
|
645
|
+
return ParseResult(
|
|
646
|
+
valid=True,
|
|
647
|
+
fields=fields,
|
|
648
|
+
version=version_info if isinstance(version_info, str) else None,
|
|
649
|
+
status=status_info if isinstance(status_info, str) else None,
|
|
650
|
+
match_family=display_family,
|
|
651
|
+
)
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def parse_auto(name: str, ignore_case: bool = False) -> ParseResult:
|
|
655
|
+
"""
|
|
656
|
+
Try to parse `name` by matching it against any schema under schemas/**.json.
|
|
657
|
+
A quick 'family' hint is derived from the filename prefix by dynamically
|
|
658
|
+
inspecting available schema files. Returns a ParseResult on success;
|
|
659
|
+
raises RuntimeError if nothing matches.
|
|
660
|
+
|
|
661
|
+
Parameters
|
|
662
|
+
----------
|
|
663
|
+
name:
|
|
664
|
+
Filename to parse.
|
|
665
|
+
ignore_case:
|
|
666
|
+
When ``True``, perform case-insensitive matching. Defaults to ``False``.
|
|
667
|
+
"""
|
|
668
|
+
pkg = __package__ # e.g., "parseo"
|
|
669
|
+
info = _discover_family_info(pkg)
|
|
670
|
+
candidates = _get_schema_paths(pkg)
|
|
671
|
+
if not candidates:
|
|
672
|
+
# No schema files packaged at all
|
|
673
|
+
raise FileNotFoundError(f"No schemas packaged under {pkg}/{SCHEMAS_ROOT}.")
|
|
674
|
+
|
|
675
|
+
near_miss: Optional[ParseError] = None
|
|
676
|
+
first_error: Optional[Exception] = None
|
|
677
|
+
|
|
678
|
+
product_hint = _guess_product_family(name, info)
|
|
679
|
+
result, attempt_near_miss, attempt_first_error = _attempt_parse(
|
|
680
|
+
name, info, candidates, product_hint, ignore_case=ignore_case
|
|
681
|
+
)
|
|
682
|
+
if result is not None:
|
|
683
|
+
return result
|
|
684
|
+
if near_miss is None and attempt_near_miss is not None:
|
|
685
|
+
near_miss = attempt_near_miss
|
|
686
|
+
if first_error is None and attempt_first_error is not None:
|
|
687
|
+
first_error = attempt_first_error
|
|
688
|
+
|
|
689
|
+
# Nothing matched — provide a helpful error listing what we saw
|
|
690
|
+
with as_file(files(pkg).joinpath(SCHEMAS_ROOT)) as rp:
|
|
691
|
+
base = Path(rp)
|
|
692
|
+
seen = [str(q.relative_to(base)) for q in base.rglob("*filename_v*.json")] if base.exists() else []
|
|
693
|
+
msg = (
|
|
694
|
+
"No schema matched the provided name. "
|
|
695
|
+
f"Looked recursively under {pkg}/{SCHEMAS_ROOT}/ and found "
|
|
696
|
+
f"{len(seen)} file(s): {seen[:8]}{'…' if len(seen) > 8 else ''}"
|
|
697
|
+
)
|
|
698
|
+
if near_miss is not None:
|
|
699
|
+
raise near_miss
|
|
700
|
+
if first_error is not None:
|
|
701
|
+
msg += f". First error while reading schemas: {first_error}"
|
|
702
|
+
raise RuntimeError(msg) from first_error
|
|
703
|
+
raise RuntimeError(msg)
|
|
704
|
+
|
|
705
|
+
|
|
706
|
+
def validate_schema(
|
|
707
|
+
paths: Union[str, Path, Iterable[Union[str, Path]], None] = None,
|
|
708
|
+
pkg: str = __package__,
|
|
709
|
+
verbose: bool = False,
|
|
710
|
+
) -> None:
|
|
711
|
+
"""Validate example filenames declared in schema files.
|
|
712
|
+
|
|
713
|
+
Parameters
|
|
714
|
+
----------
|
|
715
|
+
paths: Union[str, Path, Iterable[Union[str, Path]]], optional
|
|
716
|
+
Specific schema JSON file(s) to validate. Accepts either a single path
|
|
717
|
+
or an iterable of paths. When omitted, all bundled schemas for *pkg*
|
|
718
|
+
are checked.
|
|
719
|
+
pkg: str, optional
|
|
720
|
+
Package name from which to discover schemas when *paths* is ``None``.
|
|
721
|
+
verbose: bool, optional
|
|
722
|
+
When ``True``, print each schema path and example as they are validated,
|
|
723
|
+
along with a summary count of successful validations.
|
|
724
|
+
|
|
725
|
+
Raises
|
|
726
|
+
------
|
|
727
|
+
ValueError
|
|
728
|
+
If an example cannot be parsed or fails to reassemble to the original
|
|
729
|
+
string.
|
|
730
|
+
"""
|
|
731
|
+
|
|
732
|
+
_get_schema_paths.cache_clear()
|
|
733
|
+
if paths is None:
|
|
734
|
+
schema_paths = _get_schema_paths(pkg)
|
|
735
|
+
elif isinstance(paths, (str, Path)):
|
|
736
|
+
schema_paths = [Path(paths)]
|
|
737
|
+
else:
|
|
738
|
+
schema_paths = [Path(p) for p in paths]
|
|
739
|
+
|
|
740
|
+
from .assembler import assemble # local import to avoid cycle
|
|
741
|
+
|
|
742
|
+
validated = 0
|
|
743
|
+
for schema_path in schema_paths:
|
|
744
|
+
if verbose:
|
|
745
|
+
print(schema_path)
|
|
746
|
+
schema = _load_json_from_path(schema_path)
|
|
747
|
+
examples = schema.get("examples")
|
|
748
|
+
if not isinstance(examples, list):
|
|
749
|
+
continue
|
|
750
|
+
for example in examples:
|
|
751
|
+
if not isinstance(example, str):
|
|
752
|
+
continue
|
|
753
|
+
res = parse_auto(example)
|
|
754
|
+
if not res.valid:
|
|
755
|
+
raise ValueError(f"Parsing failed for {example}")
|
|
756
|
+
fields = {
|
|
757
|
+
k: v
|
|
758
|
+
for k, v in _extract_fields(example, schema).items()
|
|
759
|
+
if v is not None
|
|
760
|
+
}
|
|
761
|
+
if not fields:
|
|
762
|
+
raise ValueError(
|
|
763
|
+
"Example parsed globally but not by its declared schema: "
|
|
764
|
+
f"{schema_path} -> {example}"
|
|
765
|
+
)
|
|
766
|
+
assembled = assemble(fields, schema_path=schema_path)
|
|
767
|
+
if assembled != example:
|
|
768
|
+
raise ValueError(f"Round trip failed for {example}")
|
|
769
|
+
validated += 1
|
|
770
|
+
if verbose:
|
|
771
|
+
print(f" {example}")
|
|
772
|
+
counter_examples = schema.get("counter_examples")
|
|
773
|
+
if isinstance(counter_examples, list):
|
|
774
|
+
for bad in counter_examples:
|
|
775
|
+
if not isinstance(bad, str):
|
|
776
|
+
continue
|
|
777
|
+
if _match_filename(bad, schema):
|
|
778
|
+
raise ValueError(
|
|
779
|
+
f"Counter-example was unexpectedly matched by its "
|
|
780
|
+
f"schema: {schema_path} -> {bad}"
|
|
781
|
+
)
|
|
782
|
+
if verbose:
|
|
783
|
+
print(f"Validated {validated} examples")
|
|
784
|
+
|