eosframes 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eosframes/__init__.py +121 -0
- eosframes/cli.py +764 -0
- eosframes/exceptions.py +20 -0
- eosframes/hub.py +152 -0
- eosframes/logger.py +127 -0
- eosframes/naming.py +610 -0
- eosframes/ops.py +711 -0
- eosframes/read.py +213 -0
- eosframes/scale.py +1713 -0
- eosframes/stack.py +204 -0
- eosframes/utils.py +23 -0
- eosframes/write.py +201 -0
- eosframes-1.1.0.dist-info/METADATA +112 -0
- eosframes-1.1.0.dist-info/RECORD +17 -0
- eosframes-1.1.0.dist-info/WHEEL +4 -0
- eosframes-1.1.0.dist-info/entry_points.txt +3 -0
- eosframes-1.1.0.dist-info/licenses/LICENSE +21 -0
eosframes/read.py
ADDED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Readers for Ersilia output files (CSV, H5, chunked CSVs).
|
|
2
|
+
|
|
3
|
+
Every reader in this module:
|
|
4
|
+
|
|
5
|
+
* Extracts the model ID from the filename via
|
|
6
|
+
:func:`eosframes.naming.get_model_id_from_path` (lenient — the
|
|
7
|
+
``_v<N>`` token is not strictly required on inputs).
|
|
8
|
+
* Attaches the model ID as ``df.model_id`` and the version as
|
|
9
|
+
``df.version`` (``None`` when no version token is present).
|
|
10
|
+
* Validates the file structure: ``key`` / ``input`` columns for CSV,
|
|
11
|
+
``values`` / ``features`` / ``input`` datasets for HDF5.
|
|
12
|
+
|
|
13
|
+
The attached attributes are the foundation of the library's downstream
|
|
14
|
+
contract — :func:`eosframes.write.write_csv`, :func:`eosframes.hstack`,
|
|
15
|
+
and other operations rely on them. See ``CLAUDE.md`` for the full
|
|
16
|
+
``model_id`` attribute contract.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import os
|
|
20
|
+
|
|
21
|
+
import h5py
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
from .exceptions import EosframesError
|
|
25
|
+
from .logger import get_logger
|
|
26
|
+
from .naming import get_model_id_from_path, get_version_from_path
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def read_csv(file_path: str) -> pd.DataFrame:
|
|
30
|
+
"""Read an Ersilia-format CSV file into a DataFrame.
|
|
31
|
+
|
|
32
|
+
The file is expected to have at least ``key`` and ``input`` columns
|
|
33
|
+
followed by one or more feature columns. The model ID and version
|
|
34
|
+
are extracted from the filename and attached as ``df.model_id`` and
|
|
35
|
+
``df.version`` (the latter is ``None`` if the filename has no
|
|
36
|
+
``_v<N>`` token).
|
|
37
|
+
|
|
38
|
+
Parameters
|
|
39
|
+
----------
|
|
40
|
+
file_path : str
|
|
41
|
+
Path to the CSV file. Must contain a recognisable model ID in
|
|
42
|
+
its basename.
|
|
43
|
+
|
|
44
|
+
Returns
|
|
45
|
+
-------
|
|
46
|
+
pandas.DataFrame
|
|
47
|
+
With ``df.model_id`` and ``df.version`` attached as loose
|
|
48
|
+
attributes.
|
|
49
|
+
|
|
50
|
+
Raises
|
|
51
|
+
------
|
|
52
|
+
EosframesError
|
|
53
|
+
If the file does not exist, has no recognisable model ID in its
|
|
54
|
+
name, or is missing the required ``key`` / ``input`` columns.
|
|
55
|
+
"""
|
|
56
|
+
logger = get_logger()
|
|
57
|
+
if not os.path.exists(file_path):
|
|
58
|
+
raise EosframesError(f"File not found: '{file_path}'")
|
|
59
|
+
model_id = get_model_id_from_path(file_path)
|
|
60
|
+
if model_id is None:
|
|
61
|
+
raise EosframesError(
|
|
62
|
+
f"Could not extract a model ID from filename '{file_path}'. "
|
|
63
|
+
"The filename must contain an Ersilia model identifier matching the pattern eos<digit><3 alnum>."
|
|
64
|
+
)
|
|
65
|
+
logger.info("Reading CSV: %s", file_path)
|
|
66
|
+
df = pd.read_csv(file_path)
|
|
67
|
+
for col in ("key", "input"):
|
|
68
|
+
if col not in df.columns:
|
|
69
|
+
raise EosframesError(
|
|
70
|
+
f"'{file_path}' is missing the required '{col}' column."
|
|
71
|
+
)
|
|
72
|
+
df.model_id = model_id
|
|
73
|
+
df.version = get_version_from_path(file_path)
|
|
74
|
+
logger.info("Loaded %d rows (model_id=%s)", len(df), model_id)
|
|
75
|
+
return df
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def read_h5(h5_path: str) -> pd.DataFrame:
|
|
79
|
+
"""Read an Ersilia-format HDF5 file into a DataFrame.
|
|
80
|
+
|
|
81
|
+
Expected datasets:
|
|
82
|
+
|
|
83
|
+
* ``values`` — ``(N, F)`` float values
|
|
84
|
+
* ``features`` — ``(F,)`` UTF-8 feature column names
|
|
85
|
+
* ``input`` — ``(N,)`` UTF-8 input strings (e.g. SMILES)
|
|
86
|
+
* ``key`` — ``(N,)`` UTF-8 keys (optional)
|
|
87
|
+
|
|
88
|
+
The model ID and version are extracted from the filename and
|
|
89
|
+
attached as ``df.model_id`` and ``df.version``.
|
|
90
|
+
|
|
91
|
+
Parameters
|
|
92
|
+
----------
|
|
93
|
+
h5_path : str
|
|
94
|
+
Path to the HDF5 file. Must contain a recognisable model ID in
|
|
95
|
+
its basename.
|
|
96
|
+
|
|
97
|
+
Returns
|
|
98
|
+
-------
|
|
99
|
+
pandas.DataFrame
|
|
100
|
+
With ``df.model_id`` and ``df.version`` attached as loose
|
|
101
|
+
attributes.
|
|
102
|
+
|
|
103
|
+
Raises
|
|
104
|
+
------
|
|
105
|
+
EosframesError
|
|
106
|
+
If the file does not exist, has no recognisable model ID, or is
|
|
107
|
+
missing the required ``values`` / ``features`` / ``input``
|
|
108
|
+
datasets.
|
|
109
|
+
"""
|
|
110
|
+
logger = get_logger()
|
|
111
|
+
if not os.path.exists(h5_path):
|
|
112
|
+
raise EosframesError(f"File not found: '{h5_path}'")
|
|
113
|
+
model_id = get_model_id_from_path(h5_path)
|
|
114
|
+
if model_id is None:
|
|
115
|
+
raise EosframesError(
|
|
116
|
+
f"Could not extract a model ID from filename '{h5_path}'. "
|
|
117
|
+
"The filename must contain an Ersilia model identifier matching the pattern eos<digit><3 alnum>."
|
|
118
|
+
)
|
|
119
|
+
logger.info("Reading H5: %s", h5_path)
|
|
120
|
+
with h5py.File(h5_path, "r") as f:
|
|
121
|
+
if "values" not in f:
|
|
122
|
+
raise EosframesError(
|
|
123
|
+
f"'{h5_path}' is missing the required 'values' dataset."
|
|
124
|
+
)
|
|
125
|
+
values = f["values"][:]
|
|
126
|
+
columns = [x.decode("utf-8") for x in f["features"][:]]
|
|
127
|
+
keys = [x.decode("utf-8") for x in f["key"][:]] if "key" in f else None
|
|
128
|
+
inputs = [x.decode("utf-8") for x in f["input"][:]]
|
|
129
|
+
|
|
130
|
+
meta = {"input": inputs} if keys is None else {"key": keys, "input": inputs}
|
|
131
|
+
df = pd.concat([pd.DataFrame(meta), pd.DataFrame(values, columns=columns)], axis=1)
|
|
132
|
+
df.model_id = model_id
|
|
133
|
+
df.version = get_version_from_path(h5_path)
|
|
134
|
+
logger.info("Loaded %d rows (model_id=%s)", len(df), model_id)
|
|
135
|
+
return df
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def read_chunked_csvs(dir_path: str) -> pd.DataFrame:
|
|
139
|
+
"""Read a folder of chunk CSVs produced by :func:`~eosframes.split_csv`.
|
|
140
|
+
|
|
141
|
+
Files must be named ``<prefix>_<N>.csv`` (typically ``chunk_<N>.csv``)
|
|
142
|
+
with a zero-padded numeric index ``N``. All files in the directory
|
|
143
|
+
must share the same prefix and the same column layout. The model ID
|
|
144
|
+
and version are extracted from the directory name and attached as
|
|
145
|
+
``df.model_id`` / ``df.version``.
|
|
146
|
+
|
|
147
|
+
Parameters
|
|
148
|
+
----------
|
|
149
|
+
dir_path : str
|
|
150
|
+
Path to the directory containing the chunk CSV files.
|
|
151
|
+
|
|
152
|
+
Returns
|
|
153
|
+
-------
|
|
154
|
+
pandas.DataFrame
|
|
155
|
+
All chunks concatenated in ascending index order, with row
|
|
156
|
+
indices reset.
|
|
157
|
+
|
|
158
|
+
Raises
|
|
159
|
+
------
|
|
160
|
+
EosframesError
|
|
161
|
+
If the directory does not exist, has no recognisable model ID,
|
|
162
|
+
contains unexpected non-CSV files, or contains chunks with
|
|
163
|
+
mismatched prefixes.
|
|
164
|
+
"""
|
|
165
|
+
logger = get_logger()
|
|
166
|
+
if not os.path.exists(dir_path):
|
|
167
|
+
raise EosframesError(f"Directory not found: '{dir_path}'")
|
|
168
|
+
model_id = get_model_id_from_path(dir_path)
|
|
169
|
+
if model_id is None:
|
|
170
|
+
raise EosframesError(
|
|
171
|
+
f"Could not extract a model ID from directory name '{dir_path}'."
|
|
172
|
+
)
|
|
173
|
+
logger.info("Reading chunked CSVs from: %s", dir_path)
|
|
174
|
+
|
|
175
|
+
filenames = os.listdir(dir_path)
|
|
176
|
+
if not filenames:
|
|
177
|
+
raise EosframesError(
|
|
178
|
+
f"Directory '{dir_path}' is empty. Expected one or more "
|
|
179
|
+
"chunk_<N>.csv files."
|
|
180
|
+
)
|
|
181
|
+
batch_ids = []
|
|
182
|
+
zfill = 0
|
|
183
|
+
prefixes = []
|
|
184
|
+
for fn in filenames:
|
|
185
|
+
if not fn.endswith(".csv") or not fn.startswith("chunk"):
|
|
186
|
+
raise EosframesError(
|
|
187
|
+
f"Unexpected file '{fn}' in '{dir_path}'. "
|
|
188
|
+
"Chunk folders should contain only files named chunk_<N>.csv."
|
|
189
|
+
)
|
|
190
|
+
parts = fn.split("_")
|
|
191
|
+
batch_id_str = parts[-1].split(".")[0]
|
|
192
|
+
zfill = len(batch_id_str)
|
|
193
|
+
batch_ids.append(int(batch_id_str))
|
|
194
|
+
prefixes.append("_".join(parts[:-1]))
|
|
195
|
+
|
|
196
|
+
if len(set(prefixes)) > 1:
|
|
197
|
+
raise EosframesError(
|
|
198
|
+
f"Multiple file prefixes found in '{dir_path}': {set(prefixes)}. "
|
|
199
|
+
"All chunk files must share the same prefix."
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
prefix = prefixes[0]
|
|
203
|
+
frames = [
|
|
204
|
+
pd.read_csv(os.path.join(dir_path, f"{prefix}_{str(i).zfill(zfill)}.csv"))
|
|
205
|
+
for i in sorted(batch_ids)
|
|
206
|
+
]
|
|
207
|
+
df = pd.concat(frames, axis=0).reset_index(drop=True)
|
|
208
|
+
df.model_id = model_id
|
|
209
|
+
df.version = get_version_from_path(dir_path)
|
|
210
|
+
logger.info(
|
|
211
|
+
"Loaded %d rows from %d chunks (model_id=%s)", len(df), len(frames), model_id
|
|
212
|
+
)
|
|
213
|
+
return df
|