parseo 0.4.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. parseo/__init__.py +53 -0
  2. parseo/_epsg_lookup.py +145 -0
  3. parseo/_field_mappings.py +219 -0
  4. parseo/_json.py +14 -0
  5. parseo/_tile_systems.py +51 -0
  6. parseo/assembler.py +225 -0
  7. parseo/cli.py +383 -0
  8. parseo/parser.py +784 -0
  9. parseo/schema_registry.py +246 -0
  10. parseo/schemas/copernicus/clms/clc/clc_filename_v1_0_0.json +48 -0
  11. parseo/schemas/copernicus/clms/clc/clc_filename_v1_1_0.json +65 -0
  12. parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_0.json +90 -0
  13. parseo/schemas/copernicus/clms/clcplus/ras/clcplus_filename_v0_0_1.json +80 -0
  14. parseo/schemas/copernicus/clms/egms/egms_gnss_model_filename_v0_0_0.json +46 -0
  15. parseo/schemas/copernicus/clms/egms/egms_l2a_product_filename_v0_0_0.json +71 -0
  16. parseo/schemas/copernicus/clms/egms/egms_l3_velocity_grid_filename_v0_0_0.json +64 -0
  17. parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_0.json +115 -0
  18. parseo/schemas/copernicus/clms/euhydro/euhydro_filename_v0_0_1.json +103 -0
  19. parseo/schemas/copernicus/clms/hr-vpp/st_filename_v0_0_0.json +82 -0
  20. parseo/schemas/copernicus/clms/hr-vpp/vi_filename_v0_0_0.json +89 -0
  21. parseo/schemas/copernicus/clms/hr-vpp/vpp_filename_v0_0_0.json +109 -0
  22. parseo/schemas/copernicus/clms/hr-wsi/cc_filename_v0_0_0.json +84 -0
  23. parseo/schemas/copernicus/clms/hr-wsi/fsc_filename_v0_0_0.json +88 -0
  24. parseo/schemas/copernicus/clms/hr-wsi/gfsc_filename_v0_0_0.json +83 -0
  25. parseo/schemas/copernicus/clms/hr-wsi/icd_filename_v0_0_0.json +96 -0
  26. parseo/schemas/copernicus/clms/hr-wsi/sp_filename_v0_0_0.json +177 -0
  27. parseo/schemas/copernicus/clms/hr-wsi/sws_filename_v0_0_0.json +83 -0
  28. parseo/schemas/copernicus/clms/hr-wsi/wcd_filename_v0_0_0.json +96 -0
  29. parseo/schemas/copernicus/clms/hr-wsi/wds_filename_v0_0_0.json +83 -0
  30. parseo/schemas/copernicus/clms/hr-wsi/wic_comb_filename_v0_0_0.json +82 -0
  31. parseo/schemas/copernicus/clms/hr-wsi/wic_filename_v0_0_0.json +90 -0
  32. parseo/schemas/copernicus/clms/hrl/fty_filename_v0_0_0.json +48 -0
  33. parseo/schemas/copernicus/clms/hrl/gra_filename_v0_0_0.json +48 -0
  34. parseo/schemas/copernicus/clms/hrl/ibu_filename_v0_0_0.json +57 -0
  35. parseo/schemas/copernicus/clms/hrl/imp_filename_v0_0_0.json +48 -0
  36. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_0.json +73 -0
  37. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_1.json +89 -0
  38. parseo/schemas/copernicus/clms/hrl/nvlcc_filename_v0_0_2.json +89 -0
  39. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_0.json +48 -0
  40. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_1.json +49 -0
  41. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_2.json +93 -0
  42. parseo/schemas/copernicus/clms/hrl/swf_filename_v0_0_3.json +99 -0
  43. parseo/schemas/copernicus/clms/hrl/tcd_filename_v0_0_0.json +48 -0
  44. parseo/schemas/copernicus/clms/hrl/vlcc_filename_v0_0_0.json +105 -0
  45. parseo/schemas/copernicus/clms/hrl/waw_filename_v0_0_0.json +48 -0
  46. parseo/schemas/copernicus/clms/n2k/n2k_filename_v1_0_0.json +43 -0
  47. parseo/schemas/copernicus/clms/pa/pa_filename_v1_0_0.json +99 -0
  48. parseo/schemas/copernicus/clms/riparian-zones/rpz_filename_v0_0_0.json +50 -0
  49. parseo/schemas/copernicus/clms/urban-atlas/ua_dhm_filename_v0_0_0.json +53 -0
  50. parseo/schemas/copernicus/clms/urban-atlas/ua_filename_v0_0_0.json +105 -0
  51. parseo/schemas/copernicus/clms/urban-atlas/ua_stl_filename_v0_0_0.json +89 -0
  52. parseo/schemas/copernicus/sentinel/s1_filename_v1_0_0.json +78 -0
  53. parseo/schemas/copernicus/sentinel/s2_filename_v1_0_0.json +62 -0
  54. parseo/schemas/copernicus/sentinel/s3_filename_v1_0_0.json +22 -0
  55. parseo/schemas/copernicus/sentinel/s4_filename_v1_0_0.json +21 -0
  56. parseo/schemas/copernicus/sentinel/s5p_filename_v1_0_0.json +21 -0
  57. parseo/schemas/copernicus/sentinel/s6_filename_v1_0_0.json +23 -0
  58. parseo/schemas/eumetsat/metop_filename_v1_0_0.json +68 -0
  59. parseo/schemas/eumetsat/mtg_filename_v1_0_0.json +65 -0
  60. parseo/schemas/nasa/modis_filename_v1_0_0.json +54 -0
  61. parseo/schemas/usgs/landsat/landsat_filename_v1_0_0.json +50 -0
  62. parseo/stac_http.py +267 -0
  63. parseo/stac_scraper.py +115 -0
  64. parseo/template.py +78 -0
  65. parseo-0.4.4.dist-info/METADATA +402 -0
  66. parseo-0.4.4.dist-info/RECORD +70 -0
  67. parseo-0.4.4.dist-info/WHEEL +5 -0
  68. parseo-0.4.4.dist-info/entry_points.txt +2 -0
  69. parseo-0.4.4.dist-info/licenses/LICENSE.txt +287 -0
  70. parseo-0.4.4.dist-info/top_level.txt +1 -0
parseo/parser.py ADDED
@@ -0,0 +1,784 @@
1
+ # src/parseo/parser.py
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import dataclass
5
+ from functools import lru_cache
6
+ from importlib.resources import as_file
7
+ from importlib.resources import files
8
+ from pathlib import Path
9
+ import re
10
+ from typing import Any
11
+ from typing import Dict
12
+ from typing import Iterable
13
+ from typing import Iterator
14
+ from typing import Optional
15
+ from typing import Union
16
+
17
+ from ._field_mappings import apply_schema_mappings
18
+ from .schema_registry import _discover_family_info
19
+ from .schema_registry import _get_schema_paths
20
+ from .schema_registry import _load_json_from_path
21
+ from .schema_registry import get_schema_path
22
+ from .schema_registry import list_schema_families
23
+ from .schema_registry import SCHEMAS_ROOT
24
+ from .schema_registry import to_display_family
25
+ from .template import _field_regex
26
+ from .template import compile_template
27
+
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class ParseResult:
32
+ """Result of a parsing attempt."""
33
+ valid: bool
34
+ fields: Dict[str, str]
35
+ version: Optional[str] = None
36
+ status: Optional[str] = None
37
+ match_family: Optional[str] = None # e.g., "S1", "S2", "LANDSAT"
38
+
39
+
40
+ @dataclass
41
+ class ParseError(Exception):
42
+ """Raised when a filename nearly matches a schema but fails on a field."""
43
+
44
+ field: str
45
+ expected: str
46
+ value: str
47
+ schema_id: Optional[str] = None
48
+ match_family: Optional[str] = None
49
+
50
+ def __post_init__(self) -> None:
51
+ super().__init__(str(self))
52
+
53
+ def __str__(self) -> str: # pragma: no cover - trivial
54
+ base = (
55
+ f"Invalid value '{self.value}' for field '{self.field}': expected {self.expected}"
56
+ )
57
+ extras = []
58
+ if self.match_family:
59
+ extras.append(f"schema family '{self.match_family}'")
60
+ if self.schema_id:
61
+ extras.append(f"schema '{self.schema_id}'")
62
+ if extras:
63
+ base = f"{base} (nearest match: {', '.join(extras)})"
64
+ return base
65
+
66
+
67
+ # ---------------------------
68
+ # Core helpers
69
+ # ---------------------------
70
+
71
+ # Per-schema caches — avoid mutating lru_cache'd schema dicts
72
+ _SCHEMA_PATTERN_CACHE: dict[int, str] = {}
73
+ _SCHEMA_ORDER_CACHE: dict[int, list[str]] = {}
74
+
75
+
76
+ @lru_cache(maxsize=512)
77
+ def _compile_pattern(pattern: str, ignore_case: bool = False) -> re.Pattern:
78
+ flags = re.IGNORECASE if ignore_case else 0
79
+ return re.compile(pattern, flags)
80
+
81
+
82
+ def _pattern_from_schema(schema: Dict) -> Optional[str]:
83
+ """Return a compiled regex pattern derived from a schema's template.
84
+
85
+ The compiled pattern and field order are cached in module-level dicts
86
+ keyed by ``id(schema)`` to avoid mutating the lru_cache'd schema object,
87
+ which would be thread-unsafe.
88
+ """
89
+
90
+ sid = id(schema)
91
+ cached = _SCHEMA_PATTERN_CACHE.get(sid)
92
+ if cached is not None:
93
+ return cached
94
+
95
+ template = schema.get("template")
96
+ if isinstance(template, str):
97
+ fields = schema.get("fields", {})
98
+ pattern, order = compile_template(template, fields)
99
+ _SCHEMA_PATTERN_CACHE[sid] = pattern
100
+ if order:
101
+ _SCHEMA_ORDER_CACHE.setdefault(sid, order)
102
+ return pattern
103
+
104
+ return None
105
+
106
+
107
+ def _get_fields_order(schema: Dict) -> list[str]:
108
+ """Return the field order for *schema*, from cache or the schema itself."""
109
+ return _SCHEMA_ORDER_CACHE.get(id(schema)) or schema.get("fields_order", [])
110
+
111
+
112
+ def _match_filename(name: str, schema: Dict, ignore_case: bool = False) -> Optional[re.Match]:
113
+ patt = _pattern_from_schema(schema)
114
+ if not isinstance(patt, str) or not patt:
115
+ return None
116
+ rx = _compile_pattern(patt, ignore_case=ignore_case)
117
+ return rx.match(name)
118
+
119
+
120
+ def _family_from_path(path: Path, info: Dict[str, Any]) -> Optional[str]:
121
+ for fam_name, meta in info.items():
122
+ if getattr(meta, "schema_path", None) == path:
123
+ return fam_name
124
+ versions = getattr(meta, "versions", {})
125
+ for ver_path, _status in getattr(versions, "values", lambda: [])():
126
+ if ver_path == path:
127
+ return fam_name
128
+ return None
129
+
130
+
131
+ def _guess_product_family(name: str, info: Dict[str, Any]) -> Optional[str]:
132
+ """Return a schema family hint derived from *name*."""
133
+
134
+ upper_name = name.upper()
135
+ for fam, meta in info.items():
136
+ tokens = getattr(meta, "tokens", ())
137
+ if any(upper_name.startswith(tok) for tok in tokens):
138
+ return fam
139
+ return None
140
+
141
+
142
+ def _normalize_epsg_fields(fields: Dict[str, Any]) -> Dict[str, Any]:
143
+ """Normalize EPSG-related fields to consistently use 5-digit codes."""
144
+
145
+ normalized = {k: v for k, v in fields.items() if v is not None}
146
+ for key, value in list(normalized.items()):
147
+ if not isinstance(key, str):
148
+ continue
149
+ key_lower = key.lower()
150
+ if "epsg" not in key_lower and key_lower not in {"tile", "tile_id"}:
151
+ continue
152
+ if isinstance(value, str) and value.isdigit() and len(value) in {4, 5}:
153
+ normalized[key] = value.zfill(5)
154
+ return normalized
155
+
156
+
157
+ def _extract_fields(name: str, schema: Dict, ignore_case: bool = False) -> Dict[str, str]:
158
+ """
159
+ Extract named groups as fields from 'name' using the schema's regex.
160
+ If the regex doesn't match, return an empty dict.
161
+ """
162
+ m = _match_filename(name, schema, ignore_case=ignore_case)
163
+ if not m:
164
+ return {}
165
+ extracted = m.groupdict()
166
+ enriched = apply_schema_mappings(extracted, schema)
167
+ return _normalize_epsg_fields(enriched)
168
+
169
+
170
+ def _try_validate(name: str, schema: Dict, ignore_case: bool = False) -> bool:
171
+ return _match_filename(name, schema, ignore_case=ignore_case) is not None
172
+
173
+
174
+ def _attempt_parse(
175
+ name: str,
176
+ info: Dict[str, Any],
177
+ candidates: Iterable[Path],
178
+ product_hint: Optional[str],
179
+ ignore_case: bool = False,
180
+ ) -> tuple[Optional[ParseResult], Optional[ParseError], Optional[Exception]]:
181
+ """Attempt to parse *name* once and return result with diagnostics."""
182
+
183
+ near_miss: Optional[ParseError] = None
184
+ first_error: Optional[Exception] = None
185
+
186
+ hinted_meta = info.get(product_hint) if product_hint else None
187
+ hinted = hinted_meta.schema_path if hinted_meta else None
188
+ if hinted and hinted.exists():
189
+ try:
190
+ schema = _load_json_from_path(hinted)
191
+ canonical_family = product_hint or _family_from_path(hinted, info)
192
+ if _try_validate(name, schema, ignore_case=ignore_case):
193
+ display_family = to_display_family(canonical_family)
194
+ return (
195
+ ParseResult(
196
+ valid=True,
197
+ fields=_extract_fields(name, schema, ignore_case=ignore_case),
198
+ version=hinted_meta.version if hinted_meta else None,
199
+ status=hinted_meta.status if hinted_meta else None,
200
+ match_family=display_family,
201
+ ),
202
+ None,
203
+ None,
204
+ )
205
+ mismatch = _explain_match_failure(name, schema, ignore_case=ignore_case)
206
+ if mismatch:
207
+ field, expected, value = mismatch
208
+ display_family = to_display_family(canonical_family)
209
+ near_miss = ParseError(
210
+ field,
211
+ expected,
212
+ value,
213
+ schema_id=schema.get("schema_id"),
214
+ match_family=display_family,
215
+ )
216
+ except ParseError as err:
217
+ near_miss = err
218
+ except Exception:
219
+ # Schema loading/parsing can raise OSError (I/O), ValueError
220
+ # (JSON), or re.error (bad pattern). Skip and try next schema.
221
+ pass
222
+
223
+ for p in candidates:
224
+ try:
225
+ schema = _load_json_from_path(p)
226
+ except Exception as exc:
227
+ if first_error is None:
228
+ first_error = exc
229
+ continue
230
+ if _try_validate(name, schema, ignore_case=ignore_case):
231
+ matched_family = None
232
+ version = None
233
+ status = None
234
+ for fam_name, meta in info.items():
235
+ if meta.schema_path == p:
236
+ matched_family = fam_name
237
+ version = meta.version
238
+ status = meta.status
239
+ break
240
+ for ver, (path, st) in meta.versions.items():
241
+ if path == p:
242
+ matched_family = fam_name
243
+ version = ver
244
+ status = st
245
+ break
246
+ if matched_family:
247
+ break
248
+ display_family = to_display_family(matched_family or product_hint)
249
+ return (
250
+ ParseResult(
251
+ valid=True,
252
+ fields=_extract_fields(name, schema, ignore_case=ignore_case),
253
+ version=version,
254
+ status=status,
255
+ match_family=display_family,
256
+ ),
257
+ None,
258
+ None,
259
+ )
260
+ if near_miss is None:
261
+ mismatch = _explain_match_failure(name, schema, ignore_case=ignore_case)
262
+ if mismatch:
263
+ field, expected, value = mismatch
264
+ canonical_family = _family_from_path(p, info)
265
+ display_family = to_display_family(canonical_family or product_hint)
266
+ near_miss = ParseError(
267
+ field,
268
+ expected,
269
+ value,
270
+ schema_id=schema.get("schema_id"),
271
+ match_family=display_family,
272
+ )
273
+
274
+ return None, near_miss, first_error
275
+
276
+
277
+ def _named_group_spans(pattern: str) -> Dict[str, tuple[int, int]]:
278
+ spans: Dict[str, tuple[int, int]] = {}
279
+ stack: list[tuple[Optional[str], int]] = []
280
+ i = 0
281
+ in_class = False
282
+ while i < len(pattern):
283
+ ch = pattern[i]
284
+ if ch == "\\":
285
+ i += 2
286
+ continue
287
+ if in_class:
288
+ if ch == "]":
289
+ in_class = False
290
+ i += 1
291
+ continue
292
+ if ch == "[":
293
+ in_class = True
294
+ i += 1
295
+ continue
296
+ if ch == "(":
297
+ if pattern.startswith("(?P<", i):
298
+ j = i + 4
299
+ k = pattern.index(">", j)
300
+ name = pattern[j:k]
301
+ stack.append((name, i))
302
+ i = k + 1
303
+ else:
304
+ stack.append((None, i))
305
+ i += 1
306
+ continue
307
+ if ch == ")":
308
+ name, start = stack.pop()
309
+ if name:
310
+ spans[name] = (start, i + 1)
311
+ i += 1
312
+ continue
313
+ i += 1
314
+ return spans
315
+
316
+
317
+ def _balanced_prefix(pattern: str, index: int) -> str:
318
+ """Return the longest balanced prefix of *pattern* up to *index*."""
319
+
320
+ index = min(index, len(pattern))
321
+ in_class = False
322
+ depth = 0
323
+ last_balanced = 0
324
+ i = 0
325
+ while i < index:
326
+ ch = pattern[i]
327
+ if ch == "\\":
328
+ i += 2
329
+ if depth == 0:
330
+ last_balanced = min(i, index)
331
+ continue
332
+ if in_class:
333
+ if ch == "]":
334
+ in_class = False
335
+ i += 1
336
+ if depth == 0:
337
+ last_balanced = min(i, index)
338
+ continue
339
+ if ch == "[":
340
+ in_class = True
341
+ i += 1
342
+ continue
343
+ if ch == "(":
344
+ depth += 1
345
+ i += 1
346
+ continue
347
+ if ch == ")":
348
+ if depth > 0:
349
+ depth -= 1
350
+ i += 1
351
+ if depth == 0:
352
+ last_balanced = min(i, index)
353
+ continue
354
+ i += 1
355
+ if depth == 0:
356
+ last_balanced = min(i, index)
357
+
358
+ return pattern[:last_balanced]
359
+
360
+
361
+ def _balanced_slice(pattern: str, end: int) -> str:
362
+ """Return a balanced slice of *pattern* that includes *end*."""
363
+
364
+ end = min(end, len(pattern))
365
+ # Track parenthesis depth up to ``end``.
366
+ stack: list[str] = []
367
+ in_class = False
368
+ i = 0
369
+ while i < end:
370
+ ch = pattern[i]
371
+ if ch == "\\":
372
+ i += 2
373
+ continue
374
+ if in_class:
375
+ if ch == "]":
376
+ in_class = False
377
+ i += 1
378
+ continue
379
+ if ch == "[":
380
+ in_class = True
381
+ i += 1
382
+ continue
383
+ if ch == "(":
384
+ stack.append("(")
385
+ i += 1
386
+ continue
387
+ if ch == ")":
388
+ if stack:
389
+ stack.pop()
390
+ i += 1
391
+ continue
392
+ i += 1
393
+
394
+ j = end
395
+ in_class_after = in_class
396
+ stack_after = list(stack)
397
+ while stack_after and j < len(pattern):
398
+ ch = pattern[j]
399
+ if ch == "\\":
400
+ j += 2
401
+ continue
402
+ if in_class_after:
403
+ if ch == "]":
404
+ in_class_after = False
405
+ j += 1
406
+ continue
407
+ if ch == "[":
408
+ in_class_after = True
409
+ j += 1
410
+ continue
411
+ if ch == "(":
412
+ stack_after.append("(")
413
+ j += 1
414
+ continue
415
+ if ch == ")":
416
+ stack_after.pop()
417
+ j += 1
418
+ continue
419
+ j += 1
420
+
421
+ # Include trailing quantifiers that modify the just-closed group.
422
+ while j < len(pattern):
423
+ ch = pattern[j]
424
+ if ch in "?*+":
425
+ j += 1
426
+ continue
427
+ if ch == "{":
428
+ depth = 1
429
+ k = j + 1
430
+ while k < len(pattern) and depth:
431
+ nxt = pattern[k]
432
+ if nxt == "\\":
433
+ k += 2
434
+ continue
435
+ if nxt == "{":
436
+ depth += 1
437
+ k += 1
438
+ continue
439
+ if nxt == "}":
440
+ depth -= 1
441
+ k += 1
442
+ continue
443
+ k += 1
444
+ j = k
445
+ if j < len(pattern) and pattern[j] == "?":
446
+ j += 1
447
+ continue
448
+ break
449
+
450
+ return pattern[:j]
451
+
452
+
453
+ def _explain_match_failure(name: str, schema: Dict, ignore_case: bool = False) -> Optional[tuple[str, str, str]]:
454
+ pattern = _pattern_from_schema(schema)
455
+ fields = schema.get("fields", {})
456
+ order = _get_fields_order(schema)
457
+ if not pattern or not order:
458
+ return None
459
+ spans = _named_group_spans(pattern)
460
+ for i, field in enumerate(order):
461
+ next_end = spans[order[i + 1]][1] if i + 1 < len(order) else len(pattern)
462
+ prefix_pat = _balanced_slice(pattern, next_end) + ".*$"
463
+ if not _compile_pattern(prefix_pat, ignore_case=ignore_case).match(name):
464
+ before_pat = _balanced_prefix(pattern, spans[field][0])
465
+ m_before = _compile_pattern(before_pat, ignore_case=ignore_case).match(name)
466
+ start_pos = len(m_before.group(0)) if m_before else 0
467
+ target_field = field
468
+ spec = fields.get(target_field, {})
469
+ field_rx = re.compile(_field_regex(spec))
470
+ m_field = field_rx.match(name[start_pos:])
471
+ if m_field and i + 1 < len(order):
472
+ next_field = order[i + 1]
473
+ next_spec = fields.get(next_field, {})
474
+ before_next = _balanced_prefix(pattern, spans[next_field][0])
475
+ m_before_next = _compile_pattern(before_next, ignore_case=ignore_case).match(name)
476
+ if m_before_next:
477
+ start_pos = len(m_before_next.group(0))
478
+ else:
479
+ current_end = _balanced_slice(pattern, spans[field][1])
480
+ m_current_end = _compile_pattern(current_end, ignore_case=ignore_case).match(name)
481
+ start_pos = len(m_current_end.group(0)) if m_current_end else start_pos
482
+ target_field = next_field
483
+ spec = next_spec
484
+ field_rx = re.compile(_field_regex(spec))
485
+ m_field = field_rx.match(name[start_pos:])
486
+ if m_field:
487
+ # Both the current and subsequent field values satisfy
488
+ # their specifications; defer to later iterations.
489
+ continue
490
+ if m_field:
491
+ # No informative mismatch could be identified.
492
+ continue
493
+ end_pos = len(name)
494
+ for sep in ["_", ".", "-"]:
495
+ idx = name.find(sep, start_pos)
496
+ if idx != -1:
497
+ boundary = idx if idx > start_pos else idx + 1
498
+ end_pos = min(end_pos, boundary)
499
+ value = name[start_pos:end_pos]
500
+ if "enum" in spec:
501
+ expected = f"one of {spec['enum']}"
502
+ elif "pattern" in spec:
503
+ expected = f"pattern {spec['pattern']}"
504
+ else:
505
+ expected = "a different value"
506
+ return target_field, expected, value
507
+ return None
508
+
509
+
510
+ # ---------------------------
511
+ # Public API
512
+ # ---------------------------
513
+
514
+
515
+ def list_schemas(pkg: str = __package__) -> list[str]:
516
+ """Return a list of available mission family names."""
517
+ return list_schema_families(pkg)
518
+
519
+
520
+ def describe_schema(
521
+ family: str, version: Optional[str] = None, pkg: str = __package__
522
+ ) -> dict[str, Any]:
523
+ """Return schema metadata and field descriptions for ``family``.
524
+
525
+ Parameters
526
+ ----------
527
+ family:
528
+ Mission family name (case-insensitive).
529
+ version:
530
+ Optional semantic version. When omitted, the schema marked as
531
+ ``"current"`` for the family is used.
532
+ pkg:
533
+ Package that hosts the schemas. Defaults to the installed
534
+ :mod:`parseo` package.
535
+ """
536
+
537
+ try:
538
+ schema_path = get_schema_path(family, version=version, pkg=pkg)
539
+ except KeyError:
540
+ raise
541
+
542
+ schema = _load_json_from_path(schema_path)
543
+ fields: Dict[str, Dict[str, Any]] = {}
544
+ for name, spec in schema.get("fields", {}).items():
545
+ if isinstance(spec, dict):
546
+ fields[name] = {
547
+ k: spec[k]
548
+ for k in ("type", "enum", "pattern", "description")
549
+ if k in spec
550
+ }
551
+
552
+ out: Dict[str, Any] = {
553
+ "schema_id": schema.get("schema_id"),
554
+ "schema_version": schema.get("schema_version"),
555
+ "status": schema.get("status"),
556
+ "description": schema.get("description"),
557
+ "fields": fields,
558
+ }
559
+
560
+ template = schema.get("template")
561
+ if isinstance(template, str):
562
+ out["template"] = template
563
+
564
+ examples = schema.get("examples")
565
+ if isinstance(examples, list):
566
+ out["examples"] = [e for e in examples if isinstance(e, str)]
567
+
568
+ return out
569
+
570
+
571
+ def parse(
572
+ name: str,
573
+ schema_path: Union[str, Path, None] = None,
574
+ *,
575
+ family: Optional[str] = None,
576
+ version: Optional[str] = None,
577
+ pkg: str = __package__,
578
+ ignore_case: bool = False,
579
+ ) -> ParseResult:
580
+ """Parse ``name`` using a specific schema.
581
+
582
+ Parameters
583
+ ----------
584
+ name:
585
+ Filename to parse.
586
+ schema_path:
587
+ Path to the schema JSON file. Optional when ``family`` is provided.
588
+ family:
589
+ Schema family identifier (e.g. ``"S2"``). Ignored when
590
+ ``schema_path`` is given.
591
+ version:
592
+ Semantic version string to resolve together with ``family``.
593
+ pkg:
594
+ Package that hosts the schemas. Defaults to the installed
595
+ :mod:`parseo` package.
596
+ ignore_case:
597
+ When ``True``, perform case-insensitive matching. Defaults to ``False``.
598
+ """
599
+
600
+ if schema_path is None:
601
+ if not family:
602
+ raise ValueError("Provide either 'schema_path' or 'family'.")
603
+ schema_path = get_schema_path(family, version=version, pkg=pkg)
604
+
605
+ resolved_path = Path(schema_path)
606
+ schema = _load_json_from_path(resolved_path)
607
+
608
+ if not _try_validate(name, schema, ignore_case=ignore_case):
609
+ schema_id = schema.get("schema_id") if isinstance(schema.get("schema_id"), str) else None
610
+ family_hint = None
611
+ if schema_id:
612
+ family_hint = schema_id.split(":")[-1]
613
+ elif family:
614
+ family_hint = family
615
+ display_family = to_display_family(family_hint) if family_hint else None
616
+ mismatch = _explain_match_failure(name, schema, ignore_case=ignore_case)
617
+ if mismatch:
618
+ field, expected, value = mismatch
619
+ raise ParseError(
620
+ field=field,
621
+ expected=expected,
622
+ value=value,
623
+ schema_id=schema_id,
624
+ match_family=display_family,
625
+ )
626
+ raise ParseError(
627
+ field="filename",
628
+ expected=f"pattern defined by schema {schema_id or resolved_path}",
629
+ value=name,
630
+ schema_id=schema_id,
631
+ match_family=display_family,
632
+ )
633
+
634
+ fields = _extract_fields(name, schema, ignore_case=ignore_case)
635
+ version_info = schema.get("schema_version")
636
+ status_info = schema.get("status")
637
+ schema_id = schema.get("schema_id") if isinstance(schema.get("schema_id"), str) else None
638
+ family_hint = None
639
+ if schema_id:
640
+ family_hint = schema_id.split(":")[-1]
641
+ elif family:
642
+ family_hint = family
643
+ display_family = to_display_family(family_hint) if family_hint else None
644
+
645
+ return ParseResult(
646
+ valid=True,
647
+ fields=fields,
648
+ version=version_info if isinstance(version_info, str) else None,
649
+ status=status_info if isinstance(status_info, str) else None,
650
+ match_family=display_family,
651
+ )
652
+
653
+
654
+ def parse_auto(name: str, ignore_case: bool = False) -> ParseResult:
655
+ """
656
+ Try to parse `name` by matching it against any schema under schemas/**.json.
657
+ A quick 'family' hint is derived from the filename prefix by dynamically
658
+ inspecting available schema files. Returns a ParseResult on success;
659
+ raises RuntimeError if nothing matches.
660
+
661
+ Parameters
662
+ ----------
663
+ name:
664
+ Filename to parse.
665
+ ignore_case:
666
+ When ``True``, perform case-insensitive matching. Defaults to ``False``.
667
+ """
668
+ pkg = __package__ # e.g., "parseo"
669
+ info = _discover_family_info(pkg)
670
+ candidates = _get_schema_paths(pkg)
671
+ if not candidates:
672
+ # No schema files packaged at all
673
+ raise FileNotFoundError(f"No schemas packaged under {pkg}/{SCHEMAS_ROOT}.")
674
+
675
+ near_miss: Optional[ParseError] = None
676
+ first_error: Optional[Exception] = None
677
+
678
+ product_hint = _guess_product_family(name, info)
679
+ result, attempt_near_miss, attempt_first_error = _attempt_parse(
680
+ name, info, candidates, product_hint, ignore_case=ignore_case
681
+ )
682
+ if result is not None:
683
+ return result
684
+ if near_miss is None and attempt_near_miss is not None:
685
+ near_miss = attempt_near_miss
686
+ if first_error is None and attempt_first_error is not None:
687
+ first_error = attempt_first_error
688
+
689
+ # Nothing matched — provide a helpful error listing what we saw
690
+ with as_file(files(pkg).joinpath(SCHEMAS_ROOT)) as rp:
691
+ base = Path(rp)
692
+ seen = [str(q.relative_to(base)) for q in base.rglob("*filename_v*.json")] if base.exists() else []
693
+ msg = (
694
+ "No schema matched the provided name. "
695
+ f"Looked recursively under {pkg}/{SCHEMAS_ROOT}/ and found "
696
+ f"{len(seen)} file(s): {seen[:8]}{'…' if len(seen) > 8 else ''}"
697
+ )
698
+ if near_miss is not None:
699
+ raise near_miss
700
+ if first_error is not None:
701
+ msg += f". First error while reading schemas: {first_error}"
702
+ raise RuntimeError(msg) from first_error
703
+ raise RuntimeError(msg)
704
+
705
+
706
+ def validate_schema(
707
+ paths: Union[str, Path, Iterable[Union[str, Path]], None] = None,
708
+ pkg: str = __package__,
709
+ verbose: bool = False,
710
+ ) -> None:
711
+ """Validate example filenames declared in schema files.
712
+
713
+ Parameters
714
+ ----------
715
+ paths: Union[str, Path, Iterable[Union[str, Path]]], optional
716
+ Specific schema JSON file(s) to validate. Accepts either a single path
717
+ or an iterable of paths. When omitted, all bundled schemas for *pkg*
718
+ are checked.
719
+ pkg: str, optional
720
+ Package name from which to discover schemas when *paths* is ``None``.
721
+ verbose: bool, optional
722
+ When ``True``, print each schema path and example as they are validated,
723
+ along with a summary count of successful validations.
724
+
725
+ Raises
726
+ ------
727
+ ValueError
728
+ If an example cannot be parsed or fails to reassemble to the original
729
+ string.
730
+ """
731
+
732
+ _get_schema_paths.cache_clear()
733
+ if paths is None:
734
+ schema_paths = _get_schema_paths(pkg)
735
+ elif isinstance(paths, (str, Path)):
736
+ schema_paths = [Path(paths)]
737
+ else:
738
+ schema_paths = [Path(p) for p in paths]
739
+
740
+ from .assembler import assemble # local import to avoid cycle
741
+
742
+ validated = 0
743
+ for schema_path in schema_paths:
744
+ if verbose:
745
+ print(schema_path)
746
+ schema = _load_json_from_path(schema_path)
747
+ examples = schema.get("examples")
748
+ if not isinstance(examples, list):
749
+ continue
750
+ for example in examples:
751
+ if not isinstance(example, str):
752
+ continue
753
+ res = parse_auto(example)
754
+ if not res.valid:
755
+ raise ValueError(f"Parsing failed for {example}")
756
+ fields = {
757
+ k: v
758
+ for k, v in _extract_fields(example, schema).items()
759
+ if v is not None
760
+ }
761
+ if not fields:
762
+ raise ValueError(
763
+ "Example parsed globally but not by its declared schema: "
764
+ f"{schema_path} -> {example}"
765
+ )
766
+ assembled = assemble(fields, schema_path=schema_path)
767
+ if assembled != example:
768
+ raise ValueError(f"Round trip failed for {example}")
769
+ validated += 1
770
+ if verbose:
771
+ print(f" {example}")
772
+ counter_examples = schema.get("counter_examples")
773
+ if isinstance(counter_examples, list):
774
+ for bad in counter_examples:
775
+ if not isinstance(bad, str):
776
+ continue
777
+ if _match_filename(bad, schema):
778
+ raise ValueError(
779
+ f"Counter-example was unexpectedly matched by its "
780
+ f"schema: {schema_path} -> {bad}"
781
+ )
782
+ if verbose:
783
+ print(f"Validated {validated} examples")
784
+