eosframes 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eosframes/__init__.py +121 -0
- eosframes/cli.py +764 -0
- eosframes/exceptions.py +20 -0
- eosframes/hub.py +152 -0
- eosframes/logger.py +127 -0
- eosframes/naming.py +610 -0
- eosframes/ops.py +711 -0
- eosframes/read.py +213 -0
- eosframes/scale.py +1713 -0
- eosframes/stack.py +204 -0
- eosframes/utils.py +23 -0
- eosframes/write.py +201 -0
- eosframes-1.1.0.dist-info/METADATA +112 -0
- eosframes-1.1.0.dist-info/RECORD +17 -0
- eosframes-1.1.0.dist-info/WHEEL +4 -0
- eosframes-1.1.0.dist-info/entry_points.txt +3 -0
- eosframes-1.1.0.dist-info/licenses/LICENSE +21 -0
eosframes/cli.py
ADDED
|
@@ -0,0 +1,764 @@
|
|
|
1
|
+
"""Click-based command-line interface for ``eosframes``.
|
|
2
|
+
|
|
3
|
+
Each command is a thin wrapper around a public function in
|
|
4
|
+
:mod:`eosframes.ops`, :mod:`eosframes.scale`, or :mod:`eosframes.hub`.
|
|
5
|
+
Business logic lives in those modules; this file only handles argument
|
|
6
|
+
parsing, output formatting, and translating
|
|
7
|
+
:class:`~eosframes.EosframesError` into a user-friendly
|
|
8
|
+
:class:`click.ClickException` (which Click renders with its standard
|
|
9
|
+
``Error:`` prefix and a non-zero exit code).
|
|
10
|
+
|
|
11
|
+
Run ``eosframes --help`` to list every command, or
|
|
12
|
+
``eosframes <command> --help`` for the per-command help text.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
from typing import Callable, Optional
|
|
18
|
+
|
|
19
|
+
import click
|
|
20
|
+
import pandas as pd
|
|
21
|
+
from rich import box
|
|
22
|
+
from rich.console import Console
|
|
23
|
+
from rich.table import Table
|
|
24
|
+
|
|
25
|
+
from . import hub, ops
|
|
26
|
+
from . import scale as _scale
|
|
27
|
+
from .exceptions import EosframesError
|
|
28
|
+
from .logger import get_logger
|
|
29
|
+
from .naming import (
|
|
30
|
+
is_valid_columns_name,
|
|
31
|
+
is_valid_info_name,
|
|
32
|
+
is_valid_summary_name,
|
|
33
|
+
make_columns_name,
|
|
34
|
+
make_info_name,
|
|
35
|
+
make_summary_name,
|
|
36
|
+
parse_name,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _err(e: EosframesError) -> click.ClickException:
|
|
41
|
+
"""Wrap an :class:`EosframesError` for Click's exit-code handling."""
|
|
42
|
+
return click.ClickException(str(e))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _parse_input_or_fail(input_path: str) -> dict:
|
|
46
|
+
"""Parse a data file/dir path or raise a ClickException with a helpful message."""
|
|
47
|
+
parsed = parse_name(input_path)
|
|
48
|
+
basename = os.path.basename(input_path.rstrip("/\\"))
|
|
49
|
+
if parsed is None:
|
|
50
|
+
raise click.ClickException(
|
|
51
|
+
f"'{basename}' does not follow the Ersilia naming convention.\n\n"
|
|
52
|
+
"Expected one of (prefix optional):\n"
|
|
53
|
+
" [<prefix>_]<model_id>_<version>.csv\n"
|
|
54
|
+
" [<prefix>_]<model_id>_<version>.h5\n"
|
|
55
|
+
" [<prefix>_]<model_id>_<version>_chunks/\n\n"
|
|
56
|
+
"model_id = 'eos' + 1 digit + 3 alphanumeric characters.\n"
|
|
57
|
+
"version = 'v' + integer.\n"
|
|
58
|
+
)
|
|
59
|
+
if parsed["name_type"] not in {"csv", "h5", "chunks_dir"}:
|
|
60
|
+
raise click.ClickException(
|
|
61
|
+
f"'{basename}' is a '{parsed['name_type']}' sidecar file, not a data file.\n"
|
|
62
|
+
"Pass the original .csv / .h5 / _chunks/ file instead."
|
|
63
|
+
)
|
|
64
|
+
return parsed
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _resolve_sidecar_output(
|
|
68
|
+
output: Optional[str], parsed_input: dict, kind: str
|
|
69
|
+
) -> Optional[str]:
|
|
70
|
+
"""Validate an info/columns/summary sidecar output path.
|
|
71
|
+
|
|
72
|
+
``kind`` is one of ``"info"``, ``"columns"``, ``"summary"``. Returns
|
|
73
|
+
``output`` on success, ``None`` if ``output`` is ``None``, and raises
|
|
74
|
+
``ClickException`` with a suggested filename on any violation.
|
|
75
|
+
"""
|
|
76
|
+
if output is None:
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
model_id = parsed_input["model_id"]
|
|
80
|
+
version = parsed_input["version"]
|
|
81
|
+
make_name = {
|
|
82
|
+
"info": make_info_name,
|
|
83
|
+
"columns": make_columns_name,
|
|
84
|
+
"summary": make_summary_name,
|
|
85
|
+
}[kind]
|
|
86
|
+
validator = {
|
|
87
|
+
"info": is_valid_info_name,
|
|
88
|
+
"columns": is_valid_columns_name,
|
|
89
|
+
"summary": is_valid_summary_name,
|
|
90
|
+
}[kind]
|
|
91
|
+
suggested = make_name(model_id, version)
|
|
92
|
+
suggested_with_prefix = make_name(model_id, version, prefix="example")
|
|
93
|
+
|
|
94
|
+
out_basename = os.path.basename(output)
|
|
95
|
+
if not validator(output):
|
|
96
|
+
raise click.ClickException(
|
|
97
|
+
f"'{out_basename}' is not a valid {kind}-sidecar filename.\n\n"
|
|
98
|
+
f"The filename must end with the literal suffix '_{kind}.csv' — the "
|
|
99
|
+
f"'_{kind}' token is required, not optional.\n\n"
|
|
100
|
+
f"Expected: [<prefix>_]<model_id>_<version>_{kind}.csv\n"
|
|
101
|
+
" where:\n"
|
|
102
|
+
" <prefix> optional alphanumeric token (underscores allowed)\n"
|
|
103
|
+
f" model_id must match --input ('{model_id}')\n"
|
|
104
|
+
f" version must match --input ('{version}')\n"
|
|
105
|
+
f" _{kind} literal token (required)\n\n"
|
|
106
|
+
f"Try: {suggested}\n"
|
|
107
|
+
f"Or with a prefix: {suggested_with_prefix}"
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
out_parsed = parse_name(output)
|
|
111
|
+
if out_parsed["model_id"] != model_id:
|
|
112
|
+
raise click.ClickException(
|
|
113
|
+
f"Model ID mismatch: input file is for '{model_id}' but '{out_basename}' "
|
|
114
|
+
f"encodes '{out_parsed['model_id']}'.\n"
|
|
115
|
+
f"Try: {suggested}"
|
|
116
|
+
)
|
|
117
|
+
if out_parsed["version"] != version:
|
|
118
|
+
raise click.ClickException(
|
|
119
|
+
f"Version mismatch: input file is '{version}' but '{out_basename}' "
|
|
120
|
+
f"encodes '{out_parsed['version']}'.\n"
|
|
121
|
+
f"Try: {suggested}"
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
if os.path.exists(output):
|
|
125
|
+
raise click.ClickException(
|
|
126
|
+
f"Output file '{output}' already exists. Remove it first."
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
return output
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _flatten_metadata_value(v) -> str:
|
|
133
|
+
"""Stringify a value pulled from ``hub.fetch_metadata`` for tabular display."""
|
|
134
|
+
if isinstance(v, list):
|
|
135
|
+
return " | ".join(str(item) for item in v)
|
|
136
|
+
if isinstance(v, dict):
|
|
137
|
+
return json.dumps(v)
|
|
138
|
+
if v is None:
|
|
139
|
+
return ""
|
|
140
|
+
return str(v)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _render_sidecar(
|
|
144
|
+
*,
|
|
145
|
+
input_file: str,
|
|
146
|
+
rows: pd.DataFrame,
|
|
147
|
+
resolved_out: Optional[str],
|
|
148
|
+
csv_label: str,
|
|
149
|
+
highlight_first_col: bool = True,
|
|
150
|
+
) -> None:
|
|
151
|
+
"""Render a hub-fetched sidecar as a Rich table and (optionally) write CSV.
|
|
152
|
+
|
|
153
|
+
Shared assembly + presentation for the ``info`` and ``columns``
|
|
154
|
+
commands. The two callers differ only in *how* they fetch their
|
|
155
|
+
rows; everything from the rule banner downward is identical, so it
|
|
156
|
+
lives here.
|
|
157
|
+
"""
|
|
158
|
+
console = Console()
|
|
159
|
+
console.print()
|
|
160
|
+
console.rule(f"[bold cyan]{os.path.basename(input_file)}[/bold cyan]")
|
|
161
|
+
table = Table(
|
|
162
|
+
box=box.SIMPLE_HEAD,
|
|
163
|
+
show_header=True,
|
|
164
|
+
header_style="bold magenta",
|
|
165
|
+
show_edge=False,
|
|
166
|
+
)
|
|
167
|
+
for i, col in enumerate(rows.columns):
|
|
168
|
+
is_first = i == 0 and highlight_first_col
|
|
169
|
+
table.add_column(
|
|
170
|
+
str(col),
|
|
171
|
+
style="cyan" if is_first else None,
|
|
172
|
+
overflow="fold",
|
|
173
|
+
no_wrap=is_first,
|
|
174
|
+
)
|
|
175
|
+
for _, row in rows.iterrows():
|
|
176
|
+
table.add_row(
|
|
177
|
+
*(str(v) if pd.notna(v) and v != "" else "[dim]—[/dim]" for v in row)
|
|
178
|
+
)
|
|
179
|
+
console.print(table)
|
|
180
|
+
|
|
181
|
+
if resolved_out is not None:
|
|
182
|
+
rows.to_csv(resolved_out, index=False)
|
|
183
|
+
get_logger().info(
|
|
184
|
+
"%s written to %s (%d row(s))", csv_label, resolved_out, len(rows)
|
|
185
|
+
)
|
|
186
|
+
click.echo(resolved_out)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _fetch_or_clickfail(
|
|
190
|
+
fetch: Callable, *args, **kwargs
|
|
191
|
+
):
|
|
192
|
+
"""Call a hub fetcher and convert ``EosframesError`` to ``ClickException``."""
|
|
193
|
+
try:
|
|
194
|
+
return fetch(*args, **kwargs)
|
|
195
|
+
except EosframesError as e:
|
|
196
|
+
raise _err(e) from e
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
@click.group()
|
|
200
|
+
def main():
|
|
201
|
+
"""eosframes — Ersilia output data utilities.
|
|
202
|
+
|
|
203
|
+
Manipulate inputs and outputs from the Ersilia Model Hub.
|
|
204
|
+
|
|
205
|
+
Output filenames follow a strict naming convention:
|
|
206
|
+
|
|
207
|
+
\b
|
|
208
|
+
[prefix_]<model_id>_<version>.<ext> data files (.csv, .h5)
|
|
209
|
+
[prefix_]<model_id>_<version>_chunks/ folder of chunk CSVs
|
|
210
|
+
[prefix_]<model_id>_<version>_<kind>.csv sidecars (info/columns/summary)
|
|
211
|
+
[prefix_]<model_id>_<version>_transformer.json saved scaler
|
|
212
|
+
|
|
213
|
+
where model_id matches eos<digit><3 alphanumeric>
|
|
214
|
+
and version matches v<integer>.
|
|
215
|
+
|
|
216
|
+
Read paths are lenient — only a recognizable model ID is required
|
|
217
|
+
on inputs. ``eosframes split`` is the single full exception.
|
|
218
|
+
|
|
219
|
+
Set EOSFRAMES_LOG_LEVEL=DEBUG to see verbose logs.
|
|
220
|
+
"""
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
@main.command(short_help="Split a CSV into numbered chunk files.")
|
|
224
|
+
@click.argument("input_csv", type=click.Path(exists=True, dir_okay=False))
|
|
225
|
+
@click.option(
|
|
226
|
+
"--output",
|
|
227
|
+
"-o",
|
|
228
|
+
"output_folder",
|
|
229
|
+
required=True,
|
|
230
|
+
type=click.Path(),
|
|
231
|
+
help="Destination folder (must not exist; will be created).",
|
|
232
|
+
)
|
|
233
|
+
@click.option(
|
|
234
|
+
"--chunksize",
|
|
235
|
+
default=10000,
|
|
236
|
+
show_default=True,
|
|
237
|
+
help="Number of rows per chunk file.",
|
|
238
|
+
)
|
|
239
|
+
def split(input_csv: str, output_folder: str, chunksize: int) -> None:
|
|
240
|
+
"""Split the input CSV into numbered chunk files inside the output folder.
|
|
241
|
+
|
|
242
|
+
Works with any CSV file — raw input data, Ersilia model outputs, or
|
|
243
|
+
any other tabular file. The column header is preserved in every chunk.
|
|
244
|
+
No model ID is required in the input filename.
|
|
245
|
+
|
|
246
|
+
Chunk files are named chunk_<N>.csv with zero-padding sized to fit the
|
|
247
|
+
total chunk count (e.g. chunk_0.csv for 1–9, chunk_00.csv for 10–99).
|
|
248
|
+
"""
|
|
249
|
+
try:
|
|
250
|
+
ops.split_csv(input_csv, output_folder, chunksize)
|
|
251
|
+
except EosframesError as e:
|
|
252
|
+
raise _err(e) from e
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
@main.command(short_help="Convert between CSV, H5, and chunk folders.")
|
|
256
|
+
@click.argument("input_path", type=click.Path(exists=True))
|
|
257
|
+
@click.option(
|
|
258
|
+
"--output",
|
|
259
|
+
"-o",
|
|
260
|
+
"output_path",
|
|
261
|
+
required=True,
|
|
262
|
+
type=click.Path(),
|
|
263
|
+
help="Output file. Must follow the naming convention: <model_id>_<version>.<csv|h5>.",
|
|
264
|
+
)
|
|
265
|
+
def convert(input_path: str, output_path: str) -> None:
|
|
266
|
+
"""Convert between CSV, H5, and chunk folders, inferring format from file extensions.
|
|
267
|
+
|
|
268
|
+
The input can be a CSV file, an H5 file, or a folder of chunk CSV files.
|
|
269
|
+
The output must follow the naming convention <model_id>_<version>.<csv|h5>.
|
|
270
|
+
|
|
271
|
+
Supported conversions:
|
|
272
|
+
|
|
273
|
+
\b
|
|
274
|
+
folder/ → CSV (assemble chunks into CSV)
|
|
275
|
+
folder/ → H5 (assemble chunks into H5)
|
|
276
|
+
.csv → H5 (CSV to H5)
|
|
277
|
+
.h5 → CSV (H5 to CSV)
|
|
278
|
+
"""
|
|
279
|
+
try:
|
|
280
|
+
ops.convert_file(input_path, output_path)
|
|
281
|
+
except EosframesError as e:
|
|
282
|
+
raise _err(e) from e
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
@main.command(short_help="Horizontally stack outputs from multiple models.")
|
|
286
|
+
@click.argument(
|
|
287
|
+
"inputs", nargs=-1, required=True, type=click.Path(exists=True, dir_okay=False)
|
|
288
|
+
)
|
|
289
|
+
@click.option(
|
|
290
|
+
"--output",
|
|
291
|
+
"-o",
|
|
292
|
+
required=True,
|
|
293
|
+
type=click.Path(),
|
|
294
|
+
help=(
|
|
295
|
+
"Output CSV. Must follow one of the two stack conventions: "
|
|
296
|
+
"[prefix]_eosmix.csv (columns suffixed with _<model_id>_<version>) "
|
|
297
|
+
"or [prefix]_<m1>_<v1>_..._<mN>_<vN>.csv in input order (columns bare)."
|
|
298
|
+
),
|
|
299
|
+
)
|
|
300
|
+
def stack(inputs: tuple, output: str) -> None:
|
|
301
|
+
"""Horizontally stack outputs from multiple Ersilia models into one CSV.
|
|
302
|
+
|
|
303
|
+
Each INPUT must be a CSV or H5 file following the naming convention. All
|
|
304
|
+
files must contain the same inputs in the same order — this is validated
|
|
305
|
+
before stacking.
|
|
306
|
+
|
|
307
|
+
The 'key' and 'input' columns are kept only once in the output.
|
|
308
|
+
|
|
309
|
+
The naming of the output file picks the column convention (pick one):
|
|
310
|
+
|
|
311
|
+
\b
|
|
312
|
+
Mode A: [<prefix>_]eosmix.csv
|
|
313
|
+
→ each feature column becomes <column>_<model_id>_<version>
|
|
314
|
+
Mode B: [<prefix>_]<m1>_<v1>_..._<mN>_<vN>.csv
|
|
315
|
+
→ feature columns stay bare; each model appears in the
|
|
316
|
+
filename in the same order as the INPUTS.
|
|
317
|
+
"""
|
|
318
|
+
try:
|
|
319
|
+
ops.stack_files(list(inputs), output)
|
|
320
|
+
except EosframesError as e:
|
|
321
|
+
raise _err(e) from e
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
@main.command(short_help="Split a stacked CSV back into per-model files.")
|
|
325
|
+
@click.argument("input_file", type=click.Path(exists=True, dir_okay=False))
|
|
326
|
+
@click.option(
|
|
327
|
+
"--output",
|
|
328
|
+
"-o",
|
|
329
|
+
"output_folder",
|
|
330
|
+
required=True,
|
|
331
|
+
type=click.Path(),
|
|
332
|
+
help="Destination folder (must not exist; will be created).",
|
|
333
|
+
)
|
|
334
|
+
def unstack(input_file: str, output_folder: str) -> None:
|
|
335
|
+
"""Split a stacked CSV back into per-model files in a fresh folder.
|
|
336
|
+
|
|
337
|
+
The mode is resolved from the input filename:
|
|
338
|
+
|
|
339
|
+
\b
|
|
340
|
+
Mode A ([<prefix>_]eosmix.csv):
|
|
341
|
+
Column names carry the provenance. Columns are grouped by their
|
|
342
|
+
_<model_id>_<version> suffix; the suffix is stripped when writing
|
|
343
|
+
each per-model file.
|
|
344
|
+
Mode B ([<prefix>_]<m1>_<v1>_..._<mN>_<vN>.csv):
|
|
345
|
+
Column names are bare. Each model's run_columns.csv is fetched
|
|
346
|
+
from GitHub and columns are distributed by name.
|
|
347
|
+
|
|
348
|
+
Each per-model file is written as [<prefix>_]<model_id>_<version>.csv
|
|
349
|
+
with the prefix inherited from the stacked filename.
|
|
350
|
+
"""
|
|
351
|
+
try:
|
|
352
|
+
ops.unstack_file(input_file, output_folder)
|
|
353
|
+
except EosframesError as e:
|
|
354
|
+
raise _err(e) from e
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
@main.command(short_help="Vertically concatenate files from the same model.")
|
|
358
|
+
@click.argument(
|
|
359
|
+
"inputs", nargs=-1, required=True, type=click.Path(exists=True, dir_okay=False)
|
|
360
|
+
)
|
|
361
|
+
@click.option(
|
|
362
|
+
"--output",
|
|
363
|
+
"-o",
|
|
364
|
+
required=True,
|
|
365
|
+
type=click.Path(),
|
|
366
|
+
help="Output file path (must follow the naming convention <model_id>_<version>.<csv|h5>).",
|
|
367
|
+
)
|
|
368
|
+
def append(inputs: tuple, output: str) -> None:
|
|
369
|
+
"""Vertically concatenate files from the same Ersilia model.
|
|
370
|
+
|
|
371
|
+
All INPUT files must belong to the same model (same model ID). Rows are
|
|
372
|
+
appended in the order the files are given. All files must have identical
|
|
373
|
+
columns.
|
|
374
|
+
|
|
375
|
+
The output must follow the naming convention and its model ID must match
|
|
376
|
+
the inputs. Format is inferred from the output extension (.csv or .h5).
|
|
377
|
+
"""
|
|
378
|
+
try:
|
|
379
|
+
ops.append_files(list(inputs), output)
|
|
380
|
+
except EosframesError as e:
|
|
381
|
+
raise _err(e) from e
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
@main.command(short_help="Remove duplicate rows by key.")
|
|
385
|
+
@click.argument("input_file", type=click.Path(exists=True, dir_okay=False))
|
|
386
|
+
@click.option(
|
|
387
|
+
"--output",
|
|
388
|
+
"-o",
|
|
389
|
+
"output_file",
|
|
390
|
+
required=True,
|
|
391
|
+
type=click.Path(),
|
|
392
|
+
help="Output file (must follow naming convention; same model ID as input).",
|
|
393
|
+
)
|
|
394
|
+
def dedupe(input_file: str, output_file: str) -> None:
|
|
395
|
+
"""Remove duplicate rows by key, keeping the first occurrence.
|
|
396
|
+
|
|
397
|
+
Reads the input, drops any row whose 'key' value has already appeared,
|
|
398
|
+
and writes the result to the output file. Both files must share the
|
|
399
|
+
same model ID. Output format is inferred from the extension (.csv or .h5).
|
|
400
|
+
"""
|
|
401
|
+
try:
|
|
402
|
+
ops.dedupe_file(input_file, output_file)
|
|
403
|
+
except EosframesError as e:
|
|
404
|
+
raise _err(e) from e
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
@main.command(short_help="Summarize an Ersilia output file.")
|
|
408
|
+
@click.argument("input_file", type=click.Path(exists=True, dir_okay=False))
|
|
409
|
+
@click.option(
|
|
410
|
+
"--output",
|
|
411
|
+
"-o",
|
|
412
|
+
default=None,
|
|
413
|
+
type=click.Path(),
|
|
414
|
+
help=(
|
|
415
|
+
"Optional sidecar CSV to write with per-feature stats (column, dtype, "
|
|
416
|
+
"missing, min, mean, max). Must follow "
|
|
417
|
+
"[prefix]_<model_id>_<version>_summary.csv with matching model_id / version. "
|
|
418
|
+
"Omit to print only."
|
|
419
|
+
),
|
|
420
|
+
)
|
|
421
|
+
def summary(input_file: str, output: str) -> None:
|
|
422
|
+
"""Summarize an Ersilia output file.
|
|
423
|
+
|
|
424
|
+
Displays file metadata, row/column counts, duplicate/missing-data flags,
|
|
425
|
+
and per-feature statistics (dtype, missing, min, mean, max).
|
|
426
|
+
|
|
427
|
+
With --output, the per-feature stats are also written as a sidecar CSV
|
|
428
|
+
(one row per feature column).
|
|
429
|
+
"""
|
|
430
|
+
console = Console()
|
|
431
|
+
df = ops._read_file(input_file)
|
|
432
|
+
|
|
433
|
+
# Resolve filename-based metadata (model_id, version) — required for the
|
|
434
|
+
# -o naming-convention check and surfaced in the header block.
|
|
435
|
+
parsed_in = parse_name(input_file)
|
|
436
|
+
resolved_out = (
|
|
437
|
+
_resolve_sidecar_output(output, parsed_in, "summary") if parsed_in else None
|
|
438
|
+
)
|
|
439
|
+
if output is not None and parsed_in is None:
|
|
440
|
+
# -o given but input filename doesn't follow the convention — we
|
|
441
|
+
# can't validate/build a matching sidecar name. Reject explicitly.
|
|
442
|
+
_parse_input_or_fail(input_file) # raises with the verbose error
|
|
443
|
+
|
|
444
|
+
model_id = getattr(df, "model_id", "unknown")
|
|
445
|
+
version = parsed_in["version"] if parsed_in else "unknown"
|
|
446
|
+
fmt = os.path.splitext(input_file)[1].lstrip(".")
|
|
447
|
+
file_size_kb = os.path.getsize(input_file) / 1024
|
|
448
|
+
|
|
449
|
+
meta_cols = [c for c in ("key", "input") if c in df.columns]
|
|
450
|
+
feature_cols = [c for c in df.columns if c not in {"key", "input"}]
|
|
451
|
+
uniq_col = (
|
|
452
|
+
"key" if "key" in df.columns else ("input" if "input" in df.columns else None)
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
# Compute per-feature stats once — reused by pretty-print and CSV output.
|
|
456
|
+
def _fmt(v: float) -> str:
|
|
457
|
+
return f"{v:.0f}" if v == int(v) else f"{v:.4g}"
|
|
458
|
+
|
|
459
|
+
stats_rows = ops._compute_summary_stats(df)
|
|
460
|
+
|
|
461
|
+
def _label_row(label: str, value: str) -> None:
|
|
462
|
+
# Pad label to 14 chars so values line up (longest label = "Unique inputs:" = 14).
|
|
463
|
+
console.print(f" [bold]{label + ':':<15}[/bold]{value}")
|
|
464
|
+
|
|
465
|
+
console.print()
|
|
466
|
+
console.rule(f"[bold cyan]{os.path.basename(input_file)}[/bold cyan]")
|
|
467
|
+
_label_row("Model ID", model_id)
|
|
468
|
+
_label_row("Version", version)
|
|
469
|
+
_label_row("Format", fmt.upper())
|
|
470
|
+
_label_row("Size", f"{file_size_kb:.1f} KB")
|
|
471
|
+
_label_row(
|
|
472
|
+
"Columns",
|
|
473
|
+
f"{len(df.columns)} ({len(meta_cols)} meta + {len(feature_cols)} features)",
|
|
474
|
+
)
|
|
475
|
+
_label_row("Rows", f"{len(df):,}")
|
|
476
|
+
if uniq_col is not None:
|
|
477
|
+
total = len(df)
|
|
478
|
+
unique = int(df[uniq_col].nunique())
|
|
479
|
+
_label_row(
|
|
480
|
+
"Unique keys" if uniq_col == "key" else "Unique inputs",
|
|
481
|
+
f"{unique:,}",
|
|
482
|
+
)
|
|
483
|
+
if unique == total:
|
|
484
|
+
dup_str = "[green]no[/green]"
|
|
485
|
+
else:
|
|
486
|
+
dup_str = f"[yellow]yes ({total - unique:,})[/yellow]"
|
|
487
|
+
_label_row("Duplicates", dup_str)
|
|
488
|
+
|
|
489
|
+
total_missing = sum(r["missing"] for r in stats_rows)
|
|
490
|
+
if total_missing == 0:
|
|
491
|
+
miss_str = "[green]no[/green]"
|
|
492
|
+
else:
|
|
493
|
+
n_cols_with_missing = sum(1 for r in stats_rows if r["missing"] > 0)
|
|
494
|
+
miss_str = (
|
|
495
|
+
f"[yellow]yes ({total_missing:,} cells in "
|
|
496
|
+
f"{n_cols_with_missing} column{'s' if n_cols_with_missing != 1 else ''})[/yellow]"
|
|
497
|
+
)
|
|
498
|
+
_label_row("Missing data", miss_str)
|
|
499
|
+
|
|
500
|
+
if not feature_cols:
|
|
501
|
+
console.print("\n [dim]No feature columns found.[/dim]")
|
|
502
|
+
return
|
|
503
|
+
|
|
504
|
+
console.print()
|
|
505
|
+
table = Table(
|
|
506
|
+
box=box.SIMPLE_HEAD,
|
|
507
|
+
show_header=True,
|
|
508
|
+
header_style="bold magenta",
|
|
509
|
+
show_edge=False,
|
|
510
|
+
)
|
|
511
|
+
table.add_column("column", style="cyan", no_wrap=True)
|
|
512
|
+
table.add_column("dtype", justify="center")
|
|
513
|
+
table.add_column("missing", justify="right")
|
|
514
|
+
table.add_column("min", justify="right")
|
|
515
|
+
table.add_column("mean", justify="right")
|
|
516
|
+
table.add_column("max", justify="right")
|
|
517
|
+
|
|
518
|
+
for row in stats_rows:
|
|
519
|
+
series = df[row["column"]]
|
|
520
|
+
missing_str = (
|
|
521
|
+
str(row["missing"])
|
|
522
|
+
if row["missing"] == 0
|
|
523
|
+
else f"[yellow]{row['missing']}[/yellow]"
|
|
524
|
+
)
|
|
525
|
+
if pd.api.types.is_numeric_dtype(series):
|
|
526
|
+
if row["min"] is None:
|
|
527
|
+
min_s = mean_s = max_s = "[dim]—[/dim]"
|
|
528
|
+
else:
|
|
529
|
+
min_s = _fmt(row["min"])
|
|
530
|
+
mean_s = _fmt(row["mean"])
|
|
531
|
+
max_s = _fmt(row["max"])
|
|
532
|
+
else:
|
|
533
|
+
n_unique = series.nunique()
|
|
534
|
+
min_s = "[dim]—[/dim]"
|
|
535
|
+
mean_s = f"[dim]{n_unique} unique[/dim]"
|
|
536
|
+
max_s = "[dim]—[/dim]"
|
|
537
|
+
table.add_row(row["column"], row["dtype"], missing_str, min_s, mean_s, max_s)
|
|
538
|
+
|
|
539
|
+
console.print(table)
|
|
540
|
+
|
|
541
|
+
if resolved_out is not None:
|
|
542
|
+
pd.DataFrame(stats_rows).to_csv(resolved_out, index=False)
|
|
543
|
+
get_logger().info(
|
|
544
|
+
"Summary written to %s (%d feature(s))", resolved_out, len(stats_rows)
|
|
545
|
+
)
|
|
546
|
+
click.echo(resolved_out)
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
@main.command(short_help="Show metadata for a model.")
|
|
550
|
+
@click.argument("input_file", type=click.Path(exists=True))
|
|
551
|
+
@click.option(
|
|
552
|
+
"--output",
|
|
553
|
+
"-o",
|
|
554
|
+
default=None,
|
|
555
|
+
type=click.Path(),
|
|
556
|
+
help=(
|
|
557
|
+
"Optional sidecar CSV to write. Must follow "
|
|
558
|
+
"[prefix]_<model_id>_<version>_info.csv, with model_id and version "
|
|
559
|
+
"matching the input file. Omit to print only."
|
|
560
|
+
),
|
|
561
|
+
)
|
|
562
|
+
def info(input_file: str, output: str) -> None:
|
|
563
|
+
"""Show metadata for the model identified by the input file.
|
|
564
|
+
|
|
565
|
+
The input must follow the Ersilia naming convention; the model ID and
|
|
566
|
+
version are resolved from its name.
|
|
567
|
+
|
|
568
|
+
With --output, a sidecar CSV is also written. The output must follow
|
|
569
|
+
[<prefix>_]<model_id>_<version>_info.csv and its model_id / version
|
|
570
|
+
must match the input.
|
|
571
|
+
"""
|
|
572
|
+
parsed_in = _parse_input_or_fail(input_file)
|
|
573
|
+
resolved_out = _resolve_sidecar_output(output, parsed_in, "info")
|
|
574
|
+
metadata = dict(_fetch_or_clickfail(hub.fetch_metadata, parsed_in["model_id"]))
|
|
575
|
+
identifier = metadata.pop("Identifier", None)
|
|
576
|
+
fields = ["Identifier", "Version", *metadata.keys()]
|
|
577
|
+
values = [
|
|
578
|
+
_flatten_metadata_value(identifier),
|
|
579
|
+
parsed_in["version"],
|
|
580
|
+
*(_flatten_metadata_value(v) for v in metadata.values()),
|
|
581
|
+
]
|
|
582
|
+
rows = pd.DataFrame({"field": fields, "value": values})
|
|
583
|
+
_render_sidecar(
|
|
584
|
+
input_file=input_file,
|
|
585
|
+
rows=rows,
|
|
586
|
+
resolved_out=resolved_out,
|
|
587
|
+
csv_label="Metadata",
|
|
588
|
+
)
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
@main.command(short_help="Show feature column definitions for a model version.")
|
|
592
|
+
@click.argument("input_file", type=click.Path(exists=True))
|
|
593
|
+
@click.option(
|
|
594
|
+
"--output",
|
|
595
|
+
"-o",
|
|
596
|
+
default=None,
|
|
597
|
+
type=click.Path(),
|
|
598
|
+
help=(
|
|
599
|
+
"Optional sidecar CSV to write. Must follow "
|
|
600
|
+
"[prefix]_<model_id>_<version>_columns.csv, with model_id and version "
|
|
601
|
+
"matching the input file. Omit to print only."
|
|
602
|
+
),
|
|
603
|
+
)
|
|
604
|
+
def columns(input_file: str, output: str) -> None:
|
|
605
|
+
"""Show the feature column definitions for the model version of the input.
|
|
606
|
+
|
|
607
|
+
The input must follow the Ersilia naming convention; the model ID and
|
|
608
|
+
version are resolved from its name.
|
|
609
|
+
|
|
610
|
+
With --output, a sidecar CSV is also written. The output must follow
|
|
611
|
+
[<prefix>_]<model_id>_<version>_columns.csv and its model_id / version
|
|
612
|
+
must match the input.
|
|
613
|
+
"""
|
|
614
|
+
parsed_in = _parse_input_or_fail(input_file)
|
|
615
|
+
resolved_out = _resolve_sidecar_output(output, parsed_in, "columns")
|
|
616
|
+
rows = _fetch_or_clickfail(
|
|
617
|
+
hub.fetch_columns, parsed_in["model_id"], parsed_in["version"]
|
|
618
|
+
)
|
|
619
|
+
_render_sidecar(
|
|
620
|
+
input_file=input_file,
|
|
621
|
+
rows=rows,
|
|
622
|
+
resolved_out=resolved_out,
|
|
623
|
+
csv_label="Columns",
|
|
624
|
+
)
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
@main.command(short_help="Fit a scaler and save parameters to a JSON file.")
|
|
628
|
+
@click.argument("input_file", type=click.Path(exists=True, dir_okay=False))
|
|
629
|
+
@click.option(
|
|
630
|
+
"--scaler",
|
|
631
|
+
"-s",
|
|
632
|
+
required=True,
|
|
633
|
+
type=click.Path(),
|
|
634
|
+
help="Path where the scaler JSON file will be saved.",
|
|
635
|
+
)
|
|
636
|
+
@click.option(
|
|
637
|
+
"--output",
|
|
638
|
+
"-o",
|
|
639
|
+
default=None,
|
|
640
|
+
type=click.Path(),
|
|
641
|
+
help="If provided, also write the scaled data here (fit-transform).",
|
|
642
|
+
)
|
|
643
|
+
@click.option(
|
|
644
|
+
"--quantize",
|
|
645
|
+
"quantize",
|
|
646
|
+
is_flag=True,
|
|
647
|
+
default=False,
|
|
648
|
+
help=(
|
|
649
|
+
"Only meaningful with -o: quantize the inline-transform output to "
|
|
650
|
+
"int8 in [-127, 127] (sentinel -128 for missing). Has no effect on "
|
|
651
|
+
"the saved scaler JSON, which is dtype-agnostic."
|
|
652
|
+
),
|
|
653
|
+
)
|
|
654
|
+
@click.option(
|
|
655
|
+
"--impute",
|
|
656
|
+
"impute",
|
|
657
|
+
is_flag=True,
|
|
658
|
+
default=False,
|
|
659
|
+
help=(
|
|
660
|
+
"Only meaningful with -o: replace input NaN with each column's "
|
|
661
|
+
"fit-time median (rounded to int for integer columns) before the "
|
|
662
|
+
"inline transform. Has no effect on the saved scaler JSON — the "
|
|
663
|
+
"impute value is always recorded; the flag only opts in at "
|
|
664
|
+
"transform time."
|
|
665
|
+
),
|
|
666
|
+
)
|
|
667
|
+
def fit(input_file: str, scaler: str, output: str, quantize: bool, impute: bool) -> None:
|
|
668
|
+
"""Fit a type-aware robust scaler on INPUT_FILE and save parameters to SCALER.
|
|
669
|
+
|
|
670
|
+
Each numeric feature column is auto-classified (constant, binary, count,
|
|
671
|
+
or continuous with right/left/centered sub-cases) and a type-specific
|
|
672
|
+
transform is fitted. Every numeric column is fitted — all-NaN columns
|
|
673
|
+
fall back to `type: constant, value: 0`. The key and input columns are
|
|
674
|
+
ignored.
|
|
675
|
+
|
|
676
|
+
When -o is given the scaled output is also written immediately
|
|
677
|
+
(fit-transform). Neither --quantize nor --impute changes the saved
|
|
678
|
+
scaler JSON — both only affect the inline output. Both choices are
|
|
679
|
+
available at `eosframes transform` time on the saved scaler.
|
|
680
|
+
"""
|
|
681
|
+
if quantize and output is None:
|
|
682
|
+
click.echo(
|
|
683
|
+
"Warning: --quantize has no effect without -o; the saved scaler "
|
|
684
|
+
"JSON is dtype-agnostic. Pass --quantize to `eosframes transform` "
|
|
685
|
+
"to quantize at transform time.",
|
|
686
|
+
err=True,
|
|
687
|
+
)
|
|
688
|
+
if impute and output is None:
|
|
689
|
+
click.echo(
|
|
690
|
+
"Warning: --impute has no effect without -o; the impute value "
|
|
691
|
+
"is always recorded in the scaler JSON. Pass --impute to "
|
|
692
|
+
"`eosframes transform` to opt in at transform time.",
|
|
693
|
+
err=True,
|
|
694
|
+
)
|
|
695
|
+
output_dtype = "int8" if quantize else "float32"
|
|
696
|
+
try:
|
|
697
|
+
_scale.fit_file(
|
|
698
|
+
input_file,
|
|
699
|
+
scaler,
|
|
700
|
+
output_path=output,
|
|
701
|
+
output_dtype=output_dtype,
|
|
702
|
+
impute=impute,
|
|
703
|
+
)
|
|
704
|
+
except EosframesError as e:
|
|
705
|
+
raise _err(e) from e
|
|
706
|
+
click.echo(output if output is not None else scaler)
|
|
707
|
+
|
|
708
|
+
|
|
709
|
+
@main.command(short_help="Apply a saved scaler to scale numeric feature columns.")
|
|
710
|
+
@click.argument("input_file", type=click.Path(exists=True, dir_okay=False))
|
|
711
|
+
@click.option(
|
|
712
|
+
"--scaler",
|
|
713
|
+
"-s",
|
|
714
|
+
required=True,
|
|
715
|
+
type=click.Path(exists=True),
|
|
716
|
+
help="Scaler JSON file produced by `eosframes fit`.",
|
|
717
|
+
)
|
|
718
|
+
@click.option(
|
|
719
|
+
"--output",
|
|
720
|
+
"-o",
|
|
721
|
+
required=True,
|
|
722
|
+
type=click.Path(),
|
|
723
|
+
help="Output file path.",
|
|
724
|
+
)
|
|
725
|
+
@click.option(
|
|
726
|
+
"--quantize",
|
|
727
|
+
"quantize",
|
|
728
|
+
is_flag=True,
|
|
729
|
+
default=False,
|
|
730
|
+
help=(
|
|
731
|
+
"Quantize output to int8 in [-127, 127] (sentinel -128 for missing). "
|
|
732
|
+
"Default is float32."
|
|
733
|
+
),
|
|
734
|
+
)
|
|
735
|
+
@click.option(
|
|
736
|
+
"--impute",
|
|
737
|
+
"impute",
|
|
738
|
+
is_flag=True,
|
|
739
|
+
default=False,
|
|
740
|
+
help=(
|
|
741
|
+
"Replace input NaN with each column's fit-time median (rounded to "
|
|
742
|
+
"int for integer columns) before applying the transform. The "
|
|
743
|
+
"output column will have no NaN entries (and, under --quantize, "
|
|
744
|
+
"no -128 sentinels)."
|
|
745
|
+
),
|
|
746
|
+
)
|
|
747
|
+
def transform(input_file: str, scaler: str, output: str, quantize: bool, impute: bool) -> None:
|
|
748
|
+
"""Apply a saved scaler to INPUT_FILE and write scaled data to OUTPUT.
|
|
749
|
+
|
|
750
|
+
Loads the scaler parameters from SCALER and applies them to the numeric
|
|
751
|
+
feature columns of INPUT_FILE. The key and input columns pass through
|
|
752
|
+
unchanged. Pass --quantize to write int8 output; default is float32.
|
|
753
|
+
Pass --impute to substitute each column's fit-time median for any input
|
|
754
|
+
NaN before the transform — the output column will then have no NaN
|
|
755
|
+
entries.
|
|
756
|
+
"""
|
|
757
|
+
output_dtype = "int8" if quantize else "float32"
|
|
758
|
+
try:
|
|
759
|
+
out = _scale.transform_file(
|
|
760
|
+
input_file, scaler, output, output_dtype=output_dtype, impute=impute
|
|
761
|
+
)
|
|
762
|
+
except EosframesError as e:
|
|
763
|
+
raise _err(e) from e
|
|
764
|
+
click.echo(out)
|