eosframes 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eosframes/__init__.py +121 -0
- eosframes/cli.py +764 -0
- eosframes/exceptions.py +20 -0
- eosframes/hub.py +152 -0
- eosframes/logger.py +127 -0
- eosframes/naming.py +610 -0
- eosframes/ops.py +711 -0
- eosframes/read.py +213 -0
- eosframes/scale.py +1713 -0
- eosframes/stack.py +204 -0
- eosframes/utils.py +23 -0
- eosframes/write.py +201 -0
- eosframes-1.1.0.dist-info/METADATA +112 -0
- eosframes-1.1.0.dist-info/RECORD +17 -0
- eosframes-1.1.0.dist-info/WHEEL +4 -0
- eosframes-1.1.0.dist-info/entry_points.txt +3 -0
- eosframes-1.1.0.dist-info/licenses/LICENSE +21 -0
eosframes/stack.py
ADDED
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""Stack Ersilia DataFrames horizontally or vertically.
|
|
2
|
+
|
|
3
|
+
These are the in-memory counterparts of :func:`eosframes.stack_files`
|
|
4
|
+
(horizontal, multi-model) and :func:`eosframes.append_files` (vertical,
|
|
5
|
+
single model). Use them when you already have DataFrames in hand —
|
|
6
|
+
typically from :func:`eosframes.read_csv` / :func:`eosframes.read_h5`,
|
|
7
|
+
which attach the ``model_id`` and ``version`` attributes that these
|
|
8
|
+
functions require.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from typing import List
|
|
12
|
+
|
|
13
|
+
import pandas as pd
|
|
14
|
+
|
|
15
|
+
from .exceptions import EosframesError
|
|
16
|
+
from .logger import get_logger
|
|
17
|
+
from .naming import is_model_id_valid
|
|
18
|
+
|
|
19
|
+
_STACK_MODES = ("eosmix", "explicit")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def hstack(df_list: List[pd.DataFrame], mode: str) -> pd.DataFrame:
|
|
23
|
+
"""Stack Ersilia DataFrames horizontally (one model per frame).
|
|
24
|
+
|
|
25
|
+
All frames must share the same ``input`` column in the same order.
|
|
26
|
+
The resulting DataFrame has the shared ``key`` / ``input`` columns
|
|
27
|
+
once, followed by feature columns from each input in input order.
|
|
28
|
+
|
|
29
|
+
Two naming strategies are available, matched to the two stack
|
|
30
|
+
filename modes (see :func:`eosframes.naming.is_valid_stack_mix_name`
|
|
31
|
+
and :func:`eosframes.naming.is_valid_stack_explicit_name`):
|
|
32
|
+
|
|
33
|
+
* ``mode="eosmix"`` — feature column names are suffixed with
|
|
34
|
+
``_<model_id>_<version>`` so provenance lives in the columns.
|
|
35
|
+
* ``mode="explicit"`` — feature column names are kept as-is, so
|
|
36
|
+
provenance must live in the destination filename (which will list
|
|
37
|
+
every model's ``model_id`` and ``version``).
|
|
38
|
+
|
|
39
|
+
Parameters
|
|
40
|
+
----------
|
|
41
|
+
df_list : list of pandas.DataFrame
|
|
42
|
+
DataFrames to stack. Must be non-empty, and each frame must
|
|
43
|
+
have ``model_id`` and ``version`` attributes (both are set
|
|
44
|
+
automatically by :func:`eosframes.read_csv` /
|
|
45
|
+
:func:`eosframes.read_h5` when the filename encodes them).
|
|
46
|
+
mode : {"eosmix", "explicit"}
|
|
47
|
+
Column-naming strategy.
|
|
48
|
+
|
|
49
|
+
Returns
|
|
50
|
+
-------
|
|
51
|
+
pandas.DataFrame
|
|
52
|
+
Without ``model_id`` / ``version`` attributes — a horizontal
|
|
53
|
+
stack is multi-model by definition.
|
|
54
|
+
|
|
55
|
+
Raises
|
|
56
|
+
------
|
|
57
|
+
EosframesError
|
|
58
|
+
If *df_list* is empty, *mode* is unknown, any DataFrame is
|
|
59
|
+
missing ``model_id`` or ``version``, the same
|
|
60
|
+
``(model_id, version)`` pair appears twice, or row inputs
|
|
61
|
+
don't match across frames.
|
|
62
|
+
"""
|
|
63
|
+
logger = get_logger()
|
|
64
|
+
if not df_list:
|
|
65
|
+
raise EosframesError("hstack received an empty df_list.")
|
|
66
|
+
if mode not in _STACK_MODES:
|
|
67
|
+
raise EosframesError(
|
|
68
|
+
f"Unknown stack mode: {mode!r}. Expected one of {_STACK_MODES}."
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
pairs = []
|
|
72
|
+
for i, df in enumerate(df_list):
|
|
73
|
+
model_id = getattr(df, "model_id", None)
|
|
74
|
+
version = getattr(df, "version", None)
|
|
75
|
+
if model_id is None:
|
|
76
|
+
raise EosframesError(
|
|
77
|
+
f"DataFrame #{i} does not have a 'model_id' attribute."
|
|
78
|
+
)
|
|
79
|
+
if not is_model_id_valid(model_id):
|
|
80
|
+
raise EosframesError(f"Invalid model_id: {model_id!r}")
|
|
81
|
+
if version is None:
|
|
82
|
+
raise EosframesError(
|
|
83
|
+
f"DataFrame #{i} (model_id={model_id}) does not have a 'version' "
|
|
84
|
+
"attribute. Read the file with a name that encodes the version "
|
|
85
|
+
"(<model_id>_<version>.csv) so df.version is set."
|
|
86
|
+
)
|
|
87
|
+
pairs.append((model_id, version))
|
|
88
|
+
|
|
89
|
+
# Same (model_id, version) must not appear twice — columns would collide
|
|
90
|
+
# in eosmix mode, and the explicit-mode filename would be ambiguous.
|
|
91
|
+
seen: set = set()
|
|
92
|
+
for p in pairs:
|
|
93
|
+
if p in seen:
|
|
94
|
+
raise EosframesError(
|
|
95
|
+
f"Duplicate (model_id, version) in stack inputs: {p}. "
|
|
96
|
+
"Each model/version combination may appear at most once."
|
|
97
|
+
)
|
|
98
|
+
seen.add(p)
|
|
99
|
+
|
|
100
|
+
reference_inputs = df_list[0]["input"].tolist()
|
|
101
|
+
for i, df in enumerate(df_list[1:], start=2):
|
|
102
|
+
if df["input"].tolist() != reference_inputs:
|
|
103
|
+
raise EosframesError(
|
|
104
|
+
f"Input mismatch: DataFrame #{i} has different inputs or row order "
|
|
105
|
+
"than DataFrame #1."
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
key_list = None
|
|
109
|
+
for df in df_list:
|
|
110
|
+
if "key" in df.columns:
|
|
111
|
+
key_list = df["key"].tolist()
|
|
112
|
+
break
|
|
113
|
+
|
|
114
|
+
meta = (
|
|
115
|
+
{"input": reference_inputs}
|
|
116
|
+
if key_list is None
|
|
117
|
+
else {"key": key_list, "input": reference_inputs}
|
|
118
|
+
)
|
|
119
|
+
result = pd.DataFrame(meta)
|
|
120
|
+
|
|
121
|
+
feature_count = 0
|
|
122
|
+
for (model_id, version), df in zip(pairs, df_list):
|
|
123
|
+
feature_cols = [c for c in df.columns if c not in {"key", "input"}]
|
|
124
|
+
block = df[feature_cols].reset_index(drop=True)
|
|
125
|
+
if mode == "eosmix":
|
|
126
|
+
block = block.rename(
|
|
127
|
+
columns={c: f"{c}_{model_id}_{version}" for c in feature_cols}
|
|
128
|
+
)
|
|
129
|
+
# mode == "explicit": leave column names as-is.
|
|
130
|
+
result = pd.concat([result, block], axis=1)
|
|
131
|
+
feature_count += len(feature_cols)
|
|
132
|
+
|
|
133
|
+
logger.info(
|
|
134
|
+
"hstack: %d frames × %d rows → %d feature columns (mode=%s)",
|
|
135
|
+
len(df_list),
|
|
136
|
+
len(reference_inputs),
|
|
137
|
+
feature_count,
|
|
138
|
+
mode,
|
|
139
|
+
)
|
|
140
|
+
return result
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def vstack(df_list: List[pd.DataFrame]) -> pd.DataFrame:
|
|
144
|
+
"""Stack Ersilia DataFrames vertically (same model, multiple batches).
|
|
145
|
+
|
|
146
|
+
All frames must share the same columns and the same ``model_id``.
|
|
147
|
+
The resulting DataFrame inherits the shared ``model_id`` so it can
|
|
148
|
+
be written directly with :func:`eosframes.write_csv` /
|
|
149
|
+
:func:`eosframes.write_h5`.
|
|
150
|
+
|
|
151
|
+
Parameters
|
|
152
|
+
----------
|
|
153
|
+
df_list : list of pandas.DataFrame
|
|
154
|
+
DataFrames to stack. Must be non-empty; each frame must have a
|
|
155
|
+
``model_id`` attribute.
|
|
156
|
+
|
|
157
|
+
Returns
|
|
158
|
+
-------
|
|
159
|
+
pandas.DataFrame
|
|
160
|
+
With ``model_id`` set to the shared model ID. Note that
|
|
161
|
+
``version`` is intentionally **not** propagated — vertical
|
|
162
|
+
concatenation across versions is allowed (it's the same model)
|
|
163
|
+
but the result no longer corresponds to a single version.
|
|
164
|
+
|
|
165
|
+
Raises
|
|
166
|
+
------
|
|
167
|
+
EosframesError
|
|
168
|
+
If *df_list* is empty, columns differ across frames, any
|
|
169
|
+
``model_id`` attribute is missing, or the model IDs do not all
|
|
170
|
+
match.
|
|
171
|
+
"""
|
|
172
|
+
logger = get_logger()
|
|
173
|
+
if not df_list:
|
|
174
|
+
raise EosframesError("vstack received an empty df_list.")
|
|
175
|
+
model_ids = [getattr(df, "model_id", None) for df in df_list]
|
|
176
|
+
for i, model_id in enumerate(model_ids):
|
|
177
|
+
if model_id is None:
|
|
178
|
+
raise EosframesError(
|
|
179
|
+
f"DataFrame #{i} does not have a 'model_id' attribute."
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
unique_ids = set(model_ids)
|
|
183
|
+
if len(unique_ids) > 1:
|
|
184
|
+
raise EosframesError(
|
|
185
|
+
f"Cannot vstack DataFrames with different model IDs: {sorted(unique_ids)}"
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
reference_cols = df_list[0].columns.tolist()
|
|
189
|
+
for i, df in enumerate(df_list[1:], start=2):
|
|
190
|
+
if df.columns.tolist() != reference_cols:
|
|
191
|
+
raise EosframesError(
|
|
192
|
+
f"Column mismatch: DataFrame #{i} has columns {df.columns.tolist()} "
|
|
193
|
+
f"but expected {reference_cols}."
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
result = pd.concat(df_list, axis=0).reset_index(drop=True)
|
|
197
|
+
result.model_id = model_ids[0]
|
|
198
|
+
logger.info(
|
|
199
|
+
"vstack: %d frames → %d rows (model_id=%s)",
|
|
200
|
+
len(df_list),
|
|
201
|
+
len(result),
|
|
202
|
+
model_ids[0],
|
|
203
|
+
)
|
|
204
|
+
return result
|
eosframes/utils.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""General-purpose utilities."""
|
|
2
|
+
|
|
3
|
+
from typing import Iterator
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def chunker(df: pd.DataFrame, chunksize: int = 10000) -> Iterator[pd.DataFrame]:
|
|
9
|
+
"""Yield successive non-overlapping chunks of *df*.
|
|
10
|
+
|
|
11
|
+
Parameters
|
|
12
|
+
----------
|
|
13
|
+
df : pd.DataFrame
|
|
14
|
+
The DataFrame to split.
|
|
15
|
+
chunksize : int
|
|
16
|
+
Number of rows per chunk (default 10 000).
|
|
17
|
+
|
|
18
|
+
Yields
|
|
19
|
+
------
|
|
20
|
+
pd.DataFrame
|
|
21
|
+
"""
|
|
22
|
+
for start in range(0, len(df), chunksize):
|
|
23
|
+
yield df.iloc[start : start + chunksize]
|
eosframes/write.py
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Writers for Ersilia output files (CSV, H5, chunked CSVs).
|
|
2
|
+
|
|
3
|
+
Every write path in this module enforces three invariants:
|
|
4
|
+
|
|
5
|
+
1. **No silent overwrite.** Writers refuse to clobber an existing file or
|
|
6
|
+
directory and raise :class:`~eosframes.EosframesError` with guidance
|
|
7
|
+
to delete the existing target first.
|
|
8
|
+
2. **Model-ID match.** Writers extract the model ID from the destination
|
|
9
|
+
path with :func:`eosframes.naming.get_model_id_from_path` and compare
|
|
10
|
+
it against ``df.model_id``; mismatches raise.
|
|
11
|
+
3. **Required attributes.** Every DataFrame passed in must have a
|
|
12
|
+
``model_id`` attribute. Set it via ``df.model_id = "..."`` after
|
|
13
|
+
transformations that drop loose attributes (``concat`` /
|
|
14
|
+
``drop_duplicates``).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import os
|
|
18
|
+
from typing import Union
|
|
19
|
+
|
|
20
|
+
import h5py
|
|
21
|
+
import numpy as np
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
from .exceptions import EosframesError
|
|
25
|
+
from .logger import get_logger
|
|
26
|
+
from .naming import get_model_id_from_path
|
|
27
|
+
from .utils import chunker
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def write_csv(df: pd.DataFrame, csv_path: str) -> None:
|
|
31
|
+
"""Save a DataFrame as a CSV file in Ersilia format.
|
|
32
|
+
|
|
33
|
+
The CSV is written with no row index. The model ID encoded in
|
|
34
|
+
*csv_path* must match ``df.model_id`` (extracted leniently from the
|
|
35
|
+
path basename — the strict canonical naming pattern is enforced by
|
|
36
|
+
the higher-level operations in :mod:`eosframes.ops`, not here).
|
|
37
|
+
|
|
38
|
+
Parameters
|
|
39
|
+
----------
|
|
40
|
+
df : pandas.DataFrame
|
|
41
|
+
DataFrame to save. Must have a ``model_id`` attribute set.
|
|
42
|
+
csv_path : str
|
|
43
|
+
Destination path. Must end in ``.csv`` and contain a valid
|
|
44
|
+
Ersilia model identifier somewhere in its basename.
|
|
45
|
+
|
|
46
|
+
Raises
|
|
47
|
+
------
|
|
48
|
+
EosframesError
|
|
49
|
+
If the file already exists, the extension is wrong, the model
|
|
50
|
+
ID cannot be resolved from the path, ``df.model_id`` is missing,
|
|
51
|
+
or the path-encoded and DataFrame model IDs disagree.
|
|
52
|
+
"""
|
|
53
|
+
logger = get_logger()
|
|
54
|
+
if os.path.exists(csv_path):
|
|
55
|
+
raise EosframesError(
|
|
56
|
+
f"File '{csv_path}' already exists. Remove it before saving."
|
|
57
|
+
)
|
|
58
|
+
if not csv_path.endswith(".csv"):
|
|
59
|
+
raise EosframesError(f"Output path must end in '.csv', got: '{csv_path}'")
|
|
60
|
+
path_model_id = get_model_id_from_path(csv_path)
|
|
61
|
+
if path_model_id is None:
|
|
62
|
+
raise EosframesError(
|
|
63
|
+
f"Could not extract a model ID from '{csv_path}'. "
|
|
64
|
+
"The filename must contain an Ersilia model identifier."
|
|
65
|
+
)
|
|
66
|
+
df_model_id = getattr(df, "model_id", None)
|
|
67
|
+
if df_model_id is None:
|
|
68
|
+
raise EosframesError("DataFrame does not have a 'model_id' attribute.")
|
|
69
|
+
if path_model_id != df_model_id:
|
|
70
|
+
raise EosframesError(
|
|
71
|
+
f"Model ID mismatch: filename encodes '{path_model_id}' "
|
|
72
|
+
f"but DataFrame has model_id='{df_model_id}'."
|
|
73
|
+
)
|
|
74
|
+
logger.info("Writing CSV: %s (%d rows)", csv_path, len(df))
|
|
75
|
+
df.reset_index(drop=True).to_csv(csv_path, index=False)
|
|
76
|
+
logger.info("Done: %s", csv_path)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def write_h5(df: pd.DataFrame, h5_path: str, dtype: Union[np.dtype, str]) -> None:
|
|
80
|
+
"""Save a DataFrame as an HDF5 file in Ersilia format.
|
|
81
|
+
|
|
82
|
+
Writes four datasets:
|
|
83
|
+
|
|
84
|
+
* ``key`` — UTF-8 strings, ``(N,)`` (only if ``df`` has a ``key`` column)
|
|
85
|
+
* ``input`` — UTF-8 strings, ``(N,)``
|
|
86
|
+
* ``features`` — UTF-8 strings, ``(F,)`` (the feature column names)
|
|
87
|
+
* ``values`` — numeric, ``(N, F)``, with the dtype provided by *dtype*
|
|
88
|
+
|
|
89
|
+
Parameters
|
|
90
|
+
----------
|
|
91
|
+
df : pandas.DataFrame
|
|
92
|
+
DataFrame to save. Must have a ``model_id`` attribute and an
|
|
93
|
+
``input`` column.
|
|
94
|
+
h5_path : str
|
|
95
|
+
Destination path. Must contain a valid Ersilia model identifier
|
|
96
|
+
somewhere in its basename.
|
|
97
|
+
dtype : numpy.dtype or str
|
|
98
|
+
NumPy dtype for the ``values`` dataset (e.g. ``numpy.float32``,
|
|
99
|
+
``numpy.int8``). Non-float dtypes will quantize the values.
|
|
100
|
+
|
|
101
|
+
Raises
|
|
102
|
+
------
|
|
103
|
+
EosframesError
|
|
104
|
+
If the file already exists, the model ID cannot be resolved,
|
|
105
|
+
``df.model_id`` is missing, or the path-encoded and DataFrame
|
|
106
|
+
model IDs disagree.
|
|
107
|
+
"""
|
|
108
|
+
logger = get_logger()
|
|
109
|
+
if os.path.exists(h5_path):
|
|
110
|
+
raise EosframesError(
|
|
111
|
+
f"File '{h5_path}' already exists. Remove it before saving."
|
|
112
|
+
)
|
|
113
|
+
path_model_id = get_model_id_from_path(h5_path)
|
|
114
|
+
if path_model_id is None:
|
|
115
|
+
raise EosframesError(
|
|
116
|
+
f"Could not extract a model ID from '{h5_path}'. "
|
|
117
|
+
"The filename must contain an Ersilia model identifier."
|
|
118
|
+
)
|
|
119
|
+
df_model_id = getattr(df, "model_id", None)
|
|
120
|
+
if df_model_id is None:
|
|
121
|
+
raise EosframesError("DataFrame does not have a 'model_id' attribute.")
|
|
122
|
+
if path_model_id != df_model_id:
|
|
123
|
+
raise EosframesError(
|
|
124
|
+
f"Model ID mismatch: filename encodes '{path_model_id}' "
|
|
125
|
+
f"but DataFrame has model_id='{df_model_id}'."
|
|
126
|
+
)
|
|
127
|
+
df = df.reset_index(drop=True)
|
|
128
|
+
feature_cols = [c for c in df.columns if c not in {"key", "input"}]
|
|
129
|
+
logger.info("Writing H5: %s (%d rows)", h5_path, len(df))
|
|
130
|
+
with h5py.File(h5_path, "w") as f:
|
|
131
|
+
dt_str = h5py.string_dtype(encoding="utf-8")
|
|
132
|
+
if "key" in df.columns:
|
|
133
|
+
f.create_dataset("key", data=df["key"].astype(str).tolist(), dtype=dt_str)
|
|
134
|
+
f.create_dataset("input", data=df["input"].astype(str).tolist(), dtype=dt_str)
|
|
135
|
+
f.create_dataset("features", data=feature_cols, dtype=dt_str)
|
|
136
|
+
f.create_dataset("values", data=df[feature_cols].values, dtype=dtype)
|
|
137
|
+
logger.info("Done: %s", h5_path)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def write_chunked_csvs(df: pd.DataFrame, dir_path: str, chunksize: int) -> None:
|
|
141
|
+
"""Split a DataFrame into chunk CSV files inside *dir_path*.
|
|
142
|
+
|
|
143
|
+
Creates *dir_path* (which must not already exist) and writes
|
|
144
|
+
``chunk_<N>.csv`` files into it, where ``<N>`` is a zero-padded
|
|
145
|
+
integer wide enough to accommodate the largest chunk index. Every
|
|
146
|
+
chunk has the same column header.
|
|
147
|
+
|
|
148
|
+
Parameters
|
|
149
|
+
----------
|
|
150
|
+
df : pandas.DataFrame
|
|
151
|
+
DataFrame to chunk. Must have a ``model_id`` attribute.
|
|
152
|
+
dir_path : str
|
|
153
|
+
Directory to create. Must contain a valid Ersilia model
|
|
154
|
+
identifier in its basename and must not already exist.
|
|
155
|
+
chunksize : int
|
|
156
|
+
Rows per chunk. Hard-capped at 100 000.
|
|
157
|
+
|
|
158
|
+
Raises
|
|
159
|
+
------
|
|
160
|
+
EosframesError
|
|
161
|
+
If the directory already exists, the model ID cannot be
|
|
162
|
+
resolved, ``df.model_id`` is missing, the path-encoded and
|
|
163
|
+
DataFrame model IDs disagree, or *chunksize* exceeds the limit.
|
|
164
|
+
"""
|
|
165
|
+
logger = get_logger()
|
|
166
|
+
if chunksize > 100_000:
|
|
167
|
+
raise EosframesError(f"chunksize {chunksize} exceeds the limit of 100 000.")
|
|
168
|
+
path_model_id = get_model_id_from_path(dir_path)
|
|
169
|
+
if path_model_id is None:
|
|
170
|
+
raise EosframesError(
|
|
171
|
+
f"Could not extract a model ID from '{dir_path}'. "
|
|
172
|
+
"The directory name must contain an Ersilia model identifier."
|
|
173
|
+
)
|
|
174
|
+
df_model_id = getattr(df, "model_id", None)
|
|
175
|
+
if df_model_id is None:
|
|
176
|
+
raise EosframesError("DataFrame does not have a 'model_id' attribute.")
|
|
177
|
+
if path_model_id != df_model_id:
|
|
178
|
+
raise EosframesError(
|
|
179
|
+
f"Model ID mismatch: directory encodes '{path_model_id}' "
|
|
180
|
+
f"but DataFrame has model_id='{df_model_id}'."
|
|
181
|
+
)
|
|
182
|
+
dir_path = os.path.abspath(dir_path)
|
|
183
|
+
if os.path.exists(dir_path):
|
|
184
|
+
raise EosframesError(
|
|
185
|
+
f"Directory '{dir_path}' already exists. Remove it before saving."
|
|
186
|
+
)
|
|
187
|
+
os.mkdir(dir_path)
|
|
188
|
+
num_chunks = (len(df) + chunksize - 1) // chunksize
|
|
189
|
+
zfill = len(str(max(num_chunks - 1, 0)))
|
|
190
|
+
logger.info(
|
|
191
|
+
"Writing %d rows to %s in %d chunks of up to %d rows each",
|
|
192
|
+
len(df),
|
|
193
|
+
dir_path,
|
|
194
|
+
num_chunks,
|
|
195
|
+
chunksize,
|
|
196
|
+
)
|
|
197
|
+
for i, chunk in enumerate(chunker(df.reset_index(drop=True), chunksize)):
|
|
198
|
+
chunk.to_csv(
|
|
199
|
+
os.path.join(dir_path, f"chunk_{str(i).zfill(zfill)}.csv"), index=False
|
|
200
|
+
)
|
|
201
|
+
logger.info("Done: %s (%d chunk files)", dir_path, num_chunks)
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: eosframes
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: Ersilia utilities for working with tabular output data
|
|
5
|
+
License: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Keywords: ersilia,cheminformatics,data,machine-learning
|
|
8
|
+
Author: Ersilia Open Source Initiative
|
|
9
|
+
Author-email: hello@ersilia.io
|
|
10
|
+
Requires-Python: >=3.8
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Chemistry
|
|
24
|
+
Requires-Dist: click (>=8.0)
|
|
25
|
+
Requires-Dist: h5py (>=3.10.0)
|
|
26
|
+
Requires-Dist: numpy (>=1.24.0)
|
|
27
|
+
Requires-Dist: pandas (>=2.0.0)
|
|
28
|
+
Requires-Dist: requests (>=2.31)
|
|
29
|
+
Requires-Dist: rich (>=10.0)
|
|
30
|
+
Project-URL: Homepage, https://github.com/ersilia-os/eosframes
|
|
31
|
+
Project-URL: Repository, https://github.com/ersilia-os/eosframes
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+

|
|
35
|
+
|
|
36
|
+
# Manipulating Ersilia's dataframes
|
|
37
|
+
|
|
38
|
+
`eosframes` is a library for manipulating inputs and outputs from the [Ersilia Model Hub](https://github.com/ersilia-os/ersilia). It splits, assembles, converts, scales, and summarises tabular model output files.
|
|
39
|
+
|
|
40
|
+
## Installation
|
|
41
|
+
|
|
42
|
+
Python ≥ 3.8 is required.
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install eosframes
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Or from source:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
git clone https://github.com/ersilia-os/eosframes.git
|
|
52
|
+
cd eosframes
|
|
53
|
+
pip install -e .
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Quick start
|
|
57
|
+
|
|
58
|
+
Every file the library reads or writes encodes a model ID and version in its filename, e.g. `eos4e40_v1.csv` (model `eos4e40`, version `v1`).
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
# Slice a big input CSV into chunks for parallel model runs
|
|
62
|
+
eosframes split compounds.csv -o chunks/ --chunksize 10000
|
|
63
|
+
|
|
64
|
+
# Stitch the per-batch outputs back into one file
|
|
65
|
+
eosframes append eos4e40_v1_000.csv eos4e40_v1_001.csv -o eos4e40_v1.csv
|
|
66
|
+
|
|
67
|
+
# Combine outputs from multiple models, side by side
|
|
68
|
+
eosframes stack eos4e40_v1.csv eos7m30_v1.csv -o project_eosmix.csv
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Everything the CLI does is also importable:
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from eosframes import read_csv, hstack, fit, transform
|
|
75
|
+
|
|
76
|
+
df = read_csv("eos4e40_v1.csv")
|
|
77
|
+
params = fit(df)
|
|
78
|
+
scaled = transform(df, params)
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Run `eosframes --help` (or `eosframes <command> --help`) for inline help.
|
|
82
|
+
|
|
83
|
+
## Commands
|
|
84
|
+
|
|
85
|
+
| Command | Purpose |
|
|
86
|
+
|-------------|---------------------------------------------------------------|
|
|
87
|
+
| `split` | Slice any CSV into chunk files for parallel model runs. |
|
|
88
|
+
| `convert` | CSV ↔ H5, or assemble a chunks folder. |
|
|
89
|
+
| `append` | Vertically concatenate batches from the same model. |
|
|
90
|
+
| `dedupe` | Drop duplicate rows by `key`. |
|
|
91
|
+
| `stack` | Horizontally combine outputs from different models. |
|
|
92
|
+
| `unstack` | Split a stacked file back into per-model files. |
|
|
93
|
+
| `summary` | Per-feature stats from a local file. |
|
|
94
|
+
| `info` | Model metadata fetched from GitHub. |
|
|
95
|
+
| `columns` | Feature definitions fetched from GitHub. |
|
|
96
|
+
| `fit` | Fit a type-aware robust scaler and save its parameters. |
|
|
97
|
+
| `transform` | Apply a saved scaler to a file. |
|
|
98
|
+
|
|
99
|
+
See [`docs/cli.md`](docs/cli.md) for every flag, example, and refusal condition.
|
|
100
|
+
|
|
101
|
+
## Documentation
|
|
102
|
+
|
|
103
|
+
- [`docs/cli.md`](docs/cli.md) — every CLI command, all flags, examples, and error patterns.
|
|
104
|
+
- [`docs/nomenclature.md`](docs/nomenclature.md) — every recognised filename / directory pattern, the strict/lenient contract, and the two stack modes.
|
|
105
|
+
- [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, and quantization / imputation.
|
|
106
|
+
|
|
107
|
+
## About the Ersilia Open Source Initiative
|
|
108
|
+
|
|
109
|
+
The [Ersilia Open Source Initiative](https://ersilia.io) is a tech-nonprofit fueling sustainable research in the Global South. Ersilia's main asset is the [Ersilia Model Hub](https://github.com/ersilia-os/ersilia), an open-source repository of AI/ML models for drug discovery.
|
|
110
|
+
|
|
111
|
+

|
|
112
|
+
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
eosframes/__init__.py,sha256=i6XqWcdR_VGqnwS4lqHKYXPTCHEoVgUcy2IqJ0aP_zg,3478
|
|
2
|
+
eosframes/cli.py,sha256=lnNxecjLBw-XdtaRpPPXX8m3hkzYWWQSEdri2T5vhMs,26538
|
|
3
|
+
eosframes/exceptions.py,sha256=sncH1ZUO_jJFspO-r_iqij0vGeUEHAhqhvJQcSWlX2w,758
|
|
4
|
+
eosframes/hub.py,sha256=jlu02v-s42Vohq539THMNAIRvI6liwp1F7UnLt6lSVM,5065
|
|
5
|
+
eosframes/logger.py,sha256=9Vs7KrLE1NutbLrhIvZfbso8RJ8L8QdBgctUiVNKms4,4432
|
|
6
|
+
eosframes/naming.py,sha256=9qxdmL8eloZyLuRcq-MF3SLpHGp-Y_ch2z6lkvZBFec,18519
|
|
7
|
+
eosframes/ops.py,sha256=XUl2xCXJmWpg9p-sLI3zZ0mKgWn13E-Ve0v9nBEFVe0,25944
|
|
8
|
+
eosframes/read.py,sha256=QIxPiU-a93rTNSGmORAVeURcr7tSHPTM20uWsIIrMs0,7502
|
|
9
|
+
eosframes/scale.py,sha256=jMoR2MNygTmd16Lh55dxRIu8E188jMnmv-6wyLdEoMs,71821
|
|
10
|
+
eosframes/stack.py,sha256=OuHMofTR8OWCVgALqnAe8NTbhPKcM8F8IjVGLUagOu0,7355
|
|
11
|
+
eosframes/utils.py,sha256=XYJ7bzvqiuvhUaqnOq19euM7Ay4TRGteu6sznmXShbg,522
|
|
12
|
+
eosframes/write.py,sha256=AAzlXyUM7X37l-vUfPYSNSbG5Jw6rMPlE7BOsQKhWwI,7995
|
|
13
|
+
eosframes-1.1.0.dist-info/METADATA,sha256=qq0OA3P8tQWY0VWO6jOVj5fOchd4_dB08WoVZmBAWjI,4582
|
|
14
|
+
eosframes-1.1.0.dist-info/WHEEL,sha256=EGEvSphFYqXKs23-kQBeyNoJP1nrT8ZJKQoi5p5DYL8,88
|
|
15
|
+
eosframes-1.1.0.dist-info/entry_points.txt,sha256=ePFIPifLmlju1MtwrrUE8PLeUHf3GrK86kNUHe_mb9E,48
|
|
16
|
+
eosframes-1.1.0.dist-info/licenses/LICENSE,sha256=2WVpPv5BJoPm-wsGXHxnR6uvrgPYGxWkbhl0kX3AqKc,1087
|
|
17
|
+
eosframes-1.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Ersilia Open Source Initiative
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|