eosframes 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
eosframes/naming.py ADDED
@@ -0,0 +1,610 @@
1
+ """Naming convention parsing and validation for Ersilia output files.
2
+
3
+ This module is the single source of truth for the eosframes naming
4
+ convention. Every read- and write-side gate in the library funnels through
5
+ the helpers here, so the rules below are enforced uniformly across CSV,
6
+ HDF5, chunk directories, sidecar files, transformer JSONs, and stack
7
+ outputs.
8
+
9
+ Canonical patterns
10
+ ------------------
11
+ * Data file: ``[prefix_]<model_id>_<version>.<ext>``
12
+ * Chunks directory: ``[prefix_]<model_id>_<version>_chunks``
13
+ * Sidecar CSV: ``[prefix_]<model_id>_<version>_<kind>.csv``
14
+ where ``kind`` is ``info``, ``columns``, or ``summary``
15
+ * Transformer: ``[prefix_]<model_id>_<version>_transformer.json``
16
+ * Stack, Mode A: ``[prefix_]eosmix.csv`` (column names carry provenance)
17
+ * Stack, Mode B: ``[prefix_]<m1>_<v1>_..._<mN>_<vN>.csv`` (N >= 2)
18
+
19
+ with
20
+
21
+ * ``model_id`` matching ``eos\\d[A-Za-z0-9]{3}``,
22
+ * ``version`` matching ``v\\d+``,
23
+ * ``ext`` in ``{"csv", "h5"}``,
24
+ * ``prefix`` an optional alphanumeric token, possibly with internal
25
+ underscores.
26
+
27
+ Two helpers anchor the API: :func:`parse_name` (strict, returns the full
28
+ tuple of components) and :func:`get_model_id_from_path` (lenient, scans the
29
+ basename for any model-ID-shaped substring). Strict gates use the former;
30
+ read paths and informational helpers use the latter.
31
+ """
32
+
33
+ import os
34
+ import re
35
+ from typing import Dict, List, Optional, Tuple
36
+
37
+ # Matches the eos<digit><3 alphanumeric> pattern anywhere in a string
38
+ _MODEL_ID_RE = re.compile(r"(?<![A-Za-z0-9])eos\d[A-Za-z0-9]{3}(?![A-Za-z0-9])")
39
+
40
+ # Matches <model_id>_<version> at the end of a stem (allows leading prefix tokens)
41
+ _STEM_RE = re.compile(r"(?:^|_)(eos\d[A-Za-z0-9]{3})_(v\d+)$")
42
+
43
+ # Matches <model_id>_<version>_info / _columns / _summary / _transformer stems
44
+ _INFO_STEM_RE = re.compile(r"(?:^|_)(eos\d[A-Za-z0-9]{3})_(v\d+)_info$")
45
+ _COLUMNS_STEM_RE = re.compile(r"(?:^|_)(eos\d[A-Za-z0-9]{3})_(v\d+)_columns$")
46
+ _SUMMARY_STEM_RE = re.compile(r"(?:^|_)(eos\d[A-Za-z0-9]{3})_(v\d+)_summary$")
47
+ _TRANSFORMER_STEM_RE = re.compile(r"(?:^|_)(eos\d[A-Za-z0-9]{3})_(v\d+)_transformer$")
48
+
49
+ # Stack outputs come in two flavours:
50
+ # Mode A (eosmix): [prefix]_eosmix.csv — feature cols get _<model_id>_<version>
51
+ # Mode B (explicit): [prefix]_<m1>_<v1>_..._<mN>_<vN>.csv (N>=2) — bare cols
52
+ # We match them with dedicated regexes rather than folding into parse_name,
53
+ # since Mode B overlaps syntactically with a regular data file that has a
54
+ # long prefix (see tests).
55
+ _EOSMIX_STEM_RE = re.compile(r"(?:^|_)eosmix$")
56
+ # One (model_id, version) pair — matched repeatedly to walk a stack_explicit stem.
57
+ _MODEL_VER_PAIR_RE = re.compile(r"(eos\d[A-Za-z0-9]{3})_(v\d+)")
58
+
59
+ VALID_EXTENSIONS = {"csv", "h5"}
60
+
61
+
62
+ def is_model_id_valid(model_id: str) -> bool:
63
+ """Return True if *model_id* matches the Ersilia pattern ``eos<digit><3 alphanumeric>``.
64
+
65
+ Parameters
66
+ ----------
67
+ model_id : str
68
+ Candidate model identifier.
69
+
70
+ Returns
71
+ -------
72
+ bool
73
+ """
74
+ return bool(re.fullmatch(r"eos\d[A-Za-z0-9]{3}", model_id))
75
+
76
+
77
+ def get_model_id_from_path(path: str) -> Optional[str]:
78
+ """Extract a model ID from a file or directory path.
79
+
80
+ Unlike :func:`parse_name`, no version suffix is required — the function
81
+ finds the first Ersilia model identifier anywhere in the basename.
82
+
83
+ Parameters
84
+ ----------
85
+ path : str
86
+ File path, directory path, or bare filename.
87
+
88
+ Returns
89
+ -------
90
+ str or None
91
+ The model identifier if found, otherwise ``None``.
92
+ """
93
+ basename = os.path.basename(path.rstrip("/\\"))
94
+ m = _MODEL_ID_RE.search(basename)
95
+ return m.group() if m else None
96
+
97
+
98
+ def parse_name(filename: str) -> Optional[Dict]:
99
+ """Parse a filename or directory name and return structured components.
100
+
101
+ Recognizes:
102
+
103
+ * ``<model_id>_<version>.csv`` → ``name_type="csv"``
104
+ * ``<model_id>_<version>.h5`` → ``name_type="h5"``
105
+ * ``<model_id>_<version>_chunks`` → ``name_type="chunks_dir"``
106
+ * ``<model_id>_<version>_info.csv`` → ``name_type="info"``
107
+ * ``<model_id>_<version>_columns.csv`` → ``name_type="columns"``
108
+ * ``<model_id>_<version>_summary.csv`` → ``name_type="summary"``
109
+ * ``<prefix>_<model_id>_<version>.csv`` → prefix allowed before model_id
110
+
111
+ Parameters
112
+ ----------
113
+ filename : str
114
+ Basename or full path; only the basename is used for matching.
115
+
116
+ Returns
117
+ -------
118
+ dict or None
119
+ ``{"model_id", "version", "extension", "name_type"}`` on success,
120
+ ``None`` if the filename does not match the convention.
121
+ """
122
+ basename = os.path.basename(filename.rstrip("/\\"))
123
+
124
+ # Chunks directory
125
+ if basename.endswith("_chunks"):
126
+ stem = basename[: -len("_chunks")]
127
+ m = _STEM_RE.search(stem)
128
+ if m and is_model_id_valid(m.group(1)):
129
+ return {
130
+ "model_id": m.group(1),
131
+ "version": m.group(2),
132
+ "extension": None,
133
+ "name_type": "chunks_dir",
134
+ }
135
+ return None
136
+
137
+ # File with extension
138
+ if "." not in basename:
139
+ return None
140
+ stem, ext = basename.rsplit(".", 1)
141
+ if ext not in VALID_EXTENSIONS:
142
+ return None
143
+
144
+ # Sidecar CSV files (_info.csv, _columns.csv, _summary.csv) are checked
145
+ # before the generic data-file pattern so the trailing token is not
146
+ # swallowed as part of a data-file prefix.
147
+ if ext == "csv":
148
+ for regex, name_type in (
149
+ (_INFO_STEM_RE, "info"),
150
+ (_COLUMNS_STEM_RE, "columns"),
151
+ (_SUMMARY_STEM_RE, "summary"),
152
+ ):
153
+ m = regex.search(stem)
154
+ if m and is_model_id_valid(m.group(1)):
155
+ return {
156
+ "model_id": m.group(1),
157
+ "version": m.group(2),
158
+ "extension": ext,
159
+ "name_type": name_type,
160
+ }
161
+
162
+ m = _STEM_RE.search(stem)
163
+ if m and is_model_id_valid(m.group(1)):
164
+ return {
165
+ "model_id": m.group(1),
166
+ "version": m.group(2),
167
+ "extension": ext,
168
+ "name_type": ext,
169
+ }
170
+ return None
171
+
172
+
173
+ def make_output_name(model_id: str, version: str, ext: str) -> str:
174
+ """Build a canonical output filename.
175
+
176
+ Parameters
177
+ ----------
178
+ model_id : str
179
+ Model identifier matching ``eos\\d[A-Za-z0-9]{3}``.
180
+ version : str
181
+ Version string matching ``v\\d+``.
182
+ ext : str
183
+ ``"csv"`` or ``"h5"`` (without leading dot)
184
+
185
+ Returns
186
+ -------
187
+ str
188
+ ``"<model_id>_<version>.<ext>"``.
189
+
190
+ Raises
191
+ ------
192
+ ValueError
193
+ If any argument is invalid.
194
+ """
195
+ if not is_model_id_valid(model_id):
196
+ raise ValueError(f"Invalid model_id: {model_id!r}")
197
+ if not re.match(r"^v\d+$", version):
198
+ raise ValueError(f"Invalid version: {version!r}. Expected format: v1, v2, ...")
199
+ if ext not in VALID_EXTENSIONS:
200
+ raise ValueError(
201
+ f"Unsupported extension: {ext!r}. Must be one of {VALID_EXTENSIONS}"
202
+ )
203
+ return f"{model_id}_{version}.{ext}"
204
+
205
+
206
+ def make_chunks_dir_name(model_id: str, version: str) -> str:
207
+ """Build a canonical chunks directory name.
208
+
209
+ Returns
210
+ -------
211
+ str
212
+ ``"<model_id>_<version>_chunks"``.
213
+
214
+ Raises
215
+ ------
216
+ ValueError
217
+ If any argument is invalid.
218
+ """
219
+ if not is_model_id_valid(model_id):
220
+ raise ValueError(f"Invalid model_id: {model_id!r}")
221
+ if not re.match(r"^v\d+$", version):
222
+ raise ValueError(f"Invalid version: {version!r}. Expected format: v1, v2, ...")
223
+ return f"{model_id}_{version}_chunks"
224
+
225
+
226
+ def get_version_from_path(path: str) -> Optional[str]:
227
+ """Extract the version token (matching ``v\\d+``) from a filename or path.
228
+
229
+ Parameters
230
+ ----------
231
+ path : str
232
+
233
+ Returns
234
+ -------
235
+ str or None
236
+ """
237
+ parsed = parse_name(path)
238
+ return parsed["version"] if parsed is not None else None
239
+
240
+
241
+ def is_valid_name(path: str) -> bool:
242
+ """Return ``True`` if *path* is a valid Ersilia data file or directory.
243
+
244
+ Accepts CSV files, H5 files, and chunks directories. Sidecar files
245
+ (``_info.csv``, ``_columns.csv``, ``_summary.csv``) are **not**
246
+ considered valid data names and are rejected here — use the dedicated
247
+ ``is_valid_*_name`` helpers for those.
248
+
249
+ Parameters
250
+ ----------
251
+ path : str
252
+ File or directory path; only the basename is considered.
253
+
254
+ Returns
255
+ -------
256
+ bool
257
+ """
258
+ parsed = parse_name(path)
259
+ return parsed is not None and parsed["name_type"] in {"csv", "h5", "chunks_dir"}
260
+
261
+
262
+ def is_valid_info_name(path: str) -> bool:
263
+ """Return ``True`` if *path* follows the info-sidecar convention.
264
+
265
+ Matches ``[prefix_]<model_id>_<version>_info.csv``.
266
+
267
+ Parameters
268
+ ----------
269
+ path : str
270
+
271
+ Returns
272
+ -------
273
+ bool
274
+ """
275
+ parsed = parse_name(path)
276
+ return parsed is not None and parsed["name_type"] == "info"
277
+
278
+
279
+ def is_valid_columns_name(path: str) -> bool:
280
+ """Return ``True`` if *path* follows the columns-sidecar convention.
281
+
282
+ Matches ``[prefix_]<model_id>_<version>_columns.csv``.
283
+
284
+ Parameters
285
+ ----------
286
+ path : str
287
+
288
+ Returns
289
+ -------
290
+ bool
291
+ """
292
+ parsed = parse_name(path)
293
+ return parsed is not None and parsed["name_type"] == "columns"
294
+
295
+
296
+ def is_valid_summary_name(path: str) -> bool:
297
+ """Return ``True`` if *path* follows the summary-sidecar convention.
298
+
299
+ Matches ``[prefix_]<model_id>_<version>_summary.csv``.
300
+
301
+ Parameters
302
+ ----------
303
+ path : str
304
+
305
+ Returns
306
+ -------
307
+ bool
308
+ """
309
+ parsed = parse_name(path)
310
+ return parsed is not None and parsed["name_type"] == "summary"
311
+
312
+
313
+ def _make_sidecar_name(
314
+ model_id: str, version: str, kind: str, prefix: Optional[str] = None
315
+ ) -> str:
316
+ """Build a canonical sidecar filename (``_info.csv``, ``_columns.csv``, ``_summary.csv``)."""
317
+ if not is_model_id_valid(model_id):
318
+ raise ValueError(f"Invalid model_id: {model_id!r}")
319
+ if not re.match(r"^v\d+$", version):
320
+ raise ValueError(f"Invalid version: {version!r}. Expected format: v1, v2, ...")
321
+ _validate_prefix(prefix)
322
+ stem = f"{model_id}_{version}_{kind}"
323
+ return f"{prefix}_{stem}.csv" if prefix else f"{stem}.csv"
324
+
325
+
326
+ def make_info_name(model_id: str, version: str, prefix: Optional[str] = None) -> str:
327
+ """Build a canonical info-sidecar filename.
328
+
329
+ Returns
330
+ -------
331
+ str
332
+ ``"[<prefix>_]<model_id>_<version>_info.csv"``.
333
+
334
+ Raises
335
+ ------
336
+ ValueError
337
+ If any argument is invalid.
338
+ """
339
+ return _make_sidecar_name(model_id, version, "info", prefix)
340
+
341
+
342
+ def make_columns_name(model_id: str, version: str, prefix: Optional[str] = None) -> str:
343
+ """Build a canonical columns-sidecar filename.
344
+
345
+ Returns
346
+ -------
347
+ str
348
+ ``"[<prefix>_]<model_id>_<version>_columns.csv"``.
349
+
350
+ Raises
351
+ ------
352
+ ValueError
353
+ If any argument is invalid.
354
+ """
355
+ return _make_sidecar_name(model_id, version, "columns", prefix)
356
+
357
+
358
+ def make_summary_name(model_id: str, version: str, prefix: Optional[str] = None) -> str:
359
+ """Build a canonical summary-sidecar filename.
360
+
361
+ Returns
362
+ -------
363
+ str
364
+ ``"[<prefix>_]<model_id>_<version>_summary.csv"``.
365
+
366
+ Raises
367
+ ------
368
+ ValueError
369
+ If any argument is invalid.
370
+ """
371
+ return _make_sidecar_name(model_id, version, "summary", prefix)
372
+
373
+
374
+ def parse_transformer_name(path: str) -> Optional[Dict]:
375
+ """Parse a transformer/scaler filename and return its components.
376
+
377
+ Transformer filenames follow ``[prefix_]<model_id>_<version>_transformer.json``.
378
+
379
+ Parameters
380
+ ----------
381
+ path : str
382
+ Basename or full path.
383
+
384
+ Returns
385
+ -------
386
+ dict or None
387
+ ``{"model_id", "version"}`` on success, ``None`` if the path does
388
+ not match the convention.
389
+ """
390
+ basename = os.path.basename(path.rstrip("/\\"))
391
+ if not basename.endswith(".json"):
392
+ return None
393
+ stem = basename[: -len(".json")]
394
+ m = _TRANSFORMER_STEM_RE.search(stem)
395
+ if m and is_model_id_valid(m.group(1)):
396
+ return {"model_id": m.group(1), "version": m.group(2)}
397
+ return None
398
+
399
+
400
+ def is_valid_transformer_name(path: str) -> bool:
401
+ """Return ``True`` if *path* follows the transformer/scaler naming convention.
402
+
403
+ Matches ``[prefix_]<model_id>_<version>_transformer.json``.
404
+
405
+ Parameters
406
+ ----------
407
+ path : str
408
+
409
+ Returns
410
+ -------
411
+ bool
412
+ """
413
+ return parse_transformer_name(path) is not None
414
+
415
+
416
+ def make_transformer_name(
417
+ model_id: str, version: str, prefix: Optional[str] = None
418
+ ) -> str:
419
+ """Build a canonical transformer/scaler filename.
420
+
421
+ Returns
422
+ -------
423
+ str
424
+ ``"[<prefix>_]<model_id>_<version>_transformer.json"``.
425
+
426
+ Raises
427
+ ------
428
+ ValueError
429
+ If any argument is invalid.
430
+ """
431
+ if not is_model_id_valid(model_id):
432
+ raise ValueError(f"Invalid model_id: {model_id!r}")
433
+ if not re.match(r"^v\d+$", version):
434
+ raise ValueError(f"Invalid version: {version!r}. Expected format: v1, v2, ...")
435
+ _validate_prefix(prefix)
436
+ stem = f"{model_id}_{version}_transformer"
437
+ return f"{prefix}_{stem}.json" if prefix else f"{stem}.json"
438
+
439
+
440
+ # ---------------------------------------------------------------------------
441
+ # Horizontal-stack outputs (two modes, mutually exclusive)
442
+ # ---------------------------------------------------------------------------
443
+
444
+
445
+ def _validate_prefix(prefix: Optional[str]) -> None:
446
+ if prefix is None:
447
+ return
448
+ if not re.fullmatch(r"[A-Za-z0-9]+(?:_[A-Za-z0-9]+)*", prefix):
449
+ raise ValueError(
450
+ f"Invalid prefix: {prefix!r}. Must be alphanumeric tokens joined by underscores."
451
+ )
452
+
453
+
454
+ def parse_stack_mix_name(path: str) -> Optional[Dict]:
455
+ """Parse a Mode A stack filename and return its prefix.
456
+
457
+ Mode A filenames look like ``[prefix]_eosmix.csv``. The mixture itself
458
+ has no version and no model id — column names carry the provenance.
459
+
460
+ Returns
461
+ -------
462
+ dict or None
463
+ ``{"prefix": <str or None>}`` on success, ``None`` if *path* does
464
+ not follow Mode A. Prefix is ``None`` for the bare ``eosmix.csv``.
465
+ """
466
+ basename = os.path.basename(path.rstrip("/\\"))
467
+ if "." not in basename:
468
+ return None
469
+ stem, ext = basename.rsplit(".", 1)
470
+ if ext != "csv":
471
+ return None
472
+ m = _EOSMIX_STEM_RE.search(stem)
473
+ if not m:
474
+ return None
475
+ # m.start() is 0 when the stem IS "eosmix"; otherwise it's the index of
476
+ # the leading "_" before "eosmix".
477
+ prefix = stem[: m.start()] if m.start() > 0 else ""
478
+ return {"prefix": prefix or None}
479
+
480
+
481
+ def is_valid_stack_mix_name(path: str) -> bool:
482
+ """Return ``True`` if *path* follows the Mode A stack convention.
483
+
484
+ Matches ``[prefix_]eosmix.csv``.
485
+
486
+ Parameters
487
+ ----------
488
+ path : str
489
+
490
+ Returns
491
+ -------
492
+ bool
493
+ """
494
+ return parse_stack_mix_name(path) is not None
495
+
496
+
497
+ def make_stack_mix_name(prefix: Optional[str] = None) -> str:
498
+ """Build a canonical Mode A (``eosmix``) stack filename.
499
+
500
+ Returns
501
+ -------
502
+ str
503
+ ``"eosmix.csv"`` or ``"<prefix>_eosmix.csv"``.
504
+ """
505
+ _validate_prefix(prefix)
506
+ return f"{prefix}_eosmix.csv" if prefix else "eosmix.csv"
507
+
508
+
509
+ def parse_stack_explicit_name(path: str) -> Optional[Dict]:
510
+ """Parse a Mode B stack filename: prefix + ordered model list.
511
+
512
+ A Mode B filename looks like ``[prefix]_<m1>_<v1>_<m2>_<v2>...<mN>_<vN>.csv``
513
+ with N >= 2. The trailing sequence of ``<model_id>_<version>`` pairs must
514
+ cover the stem; anything preceding the first pair (and its trailing ``_``)
515
+ is the prefix.
516
+
517
+ Returns
518
+ -------
519
+ dict or None
520
+ ``{"prefix": <str or None>, "models": [(model_id, version), ...]}``
521
+ on success; the model list has at least 2 entries. Returns ``None``
522
+ if *path* does not follow Mode B (including the single-model case,
523
+ which is a regular data file).
524
+ """
525
+ basename = os.path.basename(path.rstrip("/\\"))
526
+ if "." not in basename:
527
+ return None
528
+ stem, ext = basename.rsplit(".", 1)
529
+ if ext != "csv":
530
+ return None
531
+
532
+ # Walk the stem right-to-left collecting trailing model_id_version pairs.
533
+ pairs: List[Tuple[str, str]] = []
534
+ remaining = stem
535
+ prefix = ""
536
+ while True:
537
+ m = re.search(r"(?:^|_)(eos\d[A-Za-z0-9]{3})_(v\d+)$", remaining)
538
+ if not m:
539
+ # Whatever is left is the prefix (may be "").
540
+ prefix = remaining
541
+ break
542
+ pairs.append((m.group(1), m.group(2)))
543
+ end = m.start()
544
+ if end == 0:
545
+ prefix = ""
546
+ break
547
+ remaining = remaining[:end]
548
+
549
+ if len(pairs) < 2:
550
+ return None
551
+
552
+ pairs.reverse()
553
+ return {"prefix": prefix or None, "models": pairs}
554
+
555
+
556
+ def is_valid_stack_explicit_name(path: str) -> bool:
557
+ """Return ``True`` if *path* follows the Mode B stack convention.
558
+
559
+ Matches ``[prefix_]<m1>_<v1>_..._<mN>_<vN>.csv`` with N >= 2.
560
+
561
+ Parameters
562
+ ----------
563
+ path : str
564
+
565
+ Returns
566
+ -------
567
+ bool
568
+ """
569
+ return parse_stack_explicit_name(path) is not None
570
+
571
+
572
+ def make_stack_explicit_name(
573
+ model_versions: List[Tuple[str, str]], prefix: Optional[str] = None
574
+ ) -> str:
575
+ """Build a canonical Mode B stack filename from an ordered list of models.
576
+
577
+ Parameters
578
+ ----------
579
+ model_versions : list of (model_id, version)
580
+ At least two pairs, in the order the models were stacked.
581
+ prefix : str, optional
582
+ Optional prefix (alphanumeric tokens joined by underscores).
583
+
584
+ Returns
585
+ -------
586
+ str
587
+ ``"[<prefix>_]<m1>_<v1>_..._<mN>_<vN>.csv"``.
588
+
589
+ Raises
590
+ ------
591
+ ValueError
592
+ If fewer than two pairs are given, any ``model_id`` / ``version`` is
593
+ invalid, or the prefix is malformed.
594
+ """
595
+ if len(model_versions) < 2:
596
+ raise ValueError(
597
+ "Mode B stack filenames require at least 2 (model_id, version) pairs."
598
+ )
599
+ tokens = []
600
+ for model_id, version in model_versions:
601
+ if not is_model_id_valid(model_id):
602
+ raise ValueError(f"Invalid model_id: {model_id!r}")
603
+ if not re.match(r"^v\d+$", version):
604
+ raise ValueError(
605
+ f"Invalid version: {version!r}. Expected format: v1, v2, ..."
606
+ )
607
+ tokens.append(f"{model_id}_{version}")
608
+ _validate_prefix(prefix)
609
+ stem = "_".join(tokens)
610
+ return f"{prefix}_{stem}.csv" if prefix else f"{stem}.csv"