eosframes 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
eosframes/read.py ADDED
@@ -0,0 +1,213 @@
1
+ """Readers for Ersilia output files (CSV, H5, chunked CSVs).
2
+
3
+ Every reader in this module:
4
+
5
+ * Extracts the model ID from the filename via
6
+ :func:`eosframes.naming.get_model_id_from_path` (lenient — the
7
+ ``_v<N>`` token is not strictly required on inputs).
8
+ * Attaches the model ID as ``df.model_id`` and the version as
9
+ ``df.version`` (``None`` when no version token is present).
10
+ * Validates the file structure: ``key`` / ``input`` columns for CSV,
11
+ ``values`` / ``features`` / ``input`` datasets for HDF5.
12
+
13
+ The attached attributes are the foundation of the library's downstream
14
+ contract — :func:`eosframes.write.write_csv`, :func:`eosframes.hstack`,
15
+ and other operations rely on them. See ``CLAUDE.md`` for the full
16
+ ``model_id`` attribute contract.
17
+ """
18
+
19
+ import os
20
+
21
+ import h5py
22
+ import pandas as pd
23
+
24
+ from .exceptions import EosframesError
25
+ from .logger import get_logger
26
+ from .naming import get_model_id_from_path, get_version_from_path
27
+
28
+
29
+ def read_csv(file_path: str) -> pd.DataFrame:
30
+ """Read an Ersilia-format CSV file into a DataFrame.
31
+
32
+ The file is expected to have at least ``key`` and ``input`` columns
33
+ followed by one or more feature columns. The model ID and version
34
+ are extracted from the filename and attached as ``df.model_id`` and
35
+ ``df.version`` (the latter is ``None`` if the filename has no
36
+ ``_v<N>`` token).
37
+
38
+ Parameters
39
+ ----------
40
+ file_path : str
41
+ Path to the CSV file. Must contain a recognisable model ID in
42
+ its basename.
43
+
44
+ Returns
45
+ -------
46
+ pandas.DataFrame
47
+ With ``df.model_id`` and ``df.version`` attached as loose
48
+ attributes.
49
+
50
+ Raises
51
+ ------
52
+ EosframesError
53
+ If the file does not exist, has no recognisable model ID in its
54
+ name, or is missing the required ``key`` / ``input`` columns.
55
+ """
56
+ logger = get_logger()
57
+ if not os.path.exists(file_path):
58
+ raise EosframesError(f"File not found: '{file_path}'")
59
+ model_id = get_model_id_from_path(file_path)
60
+ if model_id is None:
61
+ raise EosframesError(
62
+ f"Could not extract a model ID from filename '{file_path}'. "
63
+ "The filename must contain an Ersilia model identifier matching the pattern eos<digit><3 alnum>."
64
+ )
65
+ logger.info("Reading CSV: %s", file_path)
66
+ df = pd.read_csv(file_path)
67
+ for col in ("key", "input"):
68
+ if col not in df.columns:
69
+ raise EosframesError(
70
+ f"'{file_path}' is missing the required '{col}' column."
71
+ )
72
+ df.model_id = model_id
73
+ df.version = get_version_from_path(file_path)
74
+ logger.info("Loaded %d rows (model_id=%s)", len(df), model_id)
75
+ return df
76
+
77
+
78
+ def read_h5(h5_path: str) -> pd.DataFrame:
79
+ """Read an Ersilia-format HDF5 file into a DataFrame.
80
+
81
+ Expected datasets:
82
+
83
+ * ``values`` — ``(N, F)`` float values
84
+ * ``features`` — ``(F,)`` UTF-8 feature column names
85
+ * ``input`` — ``(N,)`` UTF-8 input strings (e.g. SMILES)
86
+ * ``key`` — ``(N,)`` UTF-8 keys (optional)
87
+
88
+ The model ID and version are extracted from the filename and
89
+ attached as ``df.model_id`` and ``df.version``.
90
+
91
+ Parameters
92
+ ----------
93
+ h5_path : str
94
+ Path to the HDF5 file. Must contain a recognisable model ID in
95
+ its basename.
96
+
97
+ Returns
98
+ -------
99
+ pandas.DataFrame
100
+ With ``df.model_id`` and ``df.version`` attached as loose
101
+ attributes.
102
+
103
+ Raises
104
+ ------
105
+ EosframesError
106
+ If the file does not exist, has no recognisable model ID, or is
107
+ missing the required ``values`` / ``features`` / ``input``
108
+ datasets.
109
+ """
110
+ logger = get_logger()
111
+ if not os.path.exists(h5_path):
112
+ raise EosframesError(f"File not found: '{h5_path}'")
113
+ model_id = get_model_id_from_path(h5_path)
114
+ if model_id is None:
115
+ raise EosframesError(
116
+ f"Could not extract a model ID from filename '{h5_path}'. "
117
+ "The filename must contain an Ersilia model identifier matching the pattern eos<digit><3 alnum>."
118
+ )
119
+ logger.info("Reading H5: %s", h5_path)
120
+ with h5py.File(h5_path, "r") as f:
121
+ if "values" not in f:
122
+ raise EosframesError(
123
+ f"'{h5_path}' is missing the required 'values' dataset."
124
+ )
125
+ values = f["values"][:]
126
+ columns = [x.decode("utf-8") for x in f["features"][:]]
127
+ keys = [x.decode("utf-8") for x in f["key"][:]] if "key" in f else None
128
+ inputs = [x.decode("utf-8") for x in f["input"][:]]
129
+
130
+ meta = {"input": inputs} if keys is None else {"key": keys, "input": inputs}
131
+ df = pd.concat([pd.DataFrame(meta), pd.DataFrame(values, columns=columns)], axis=1)
132
+ df.model_id = model_id
133
+ df.version = get_version_from_path(h5_path)
134
+ logger.info("Loaded %d rows (model_id=%s)", len(df), model_id)
135
+ return df
136
+
137
+
138
+ def read_chunked_csvs(dir_path: str) -> pd.DataFrame:
139
+ """Read a folder of chunk CSVs produced by :func:`~eosframes.split_csv`.
140
+
141
+ Files must be named ``<prefix>_<N>.csv`` (typically ``chunk_<N>.csv``)
142
+ with a zero-padded numeric index ``N``. All files in the directory
143
+ must share the same prefix and the same column layout. The model ID
144
+ and version are extracted from the directory name and attached as
145
+ ``df.model_id`` / ``df.version``.
146
+
147
+ Parameters
148
+ ----------
149
+ dir_path : str
150
+ Path to the directory containing the chunk CSV files.
151
+
152
+ Returns
153
+ -------
154
+ pandas.DataFrame
155
+ All chunks concatenated in ascending index order, with row
156
+ indices reset.
157
+
158
+ Raises
159
+ ------
160
+ EosframesError
161
+ If the directory does not exist, has no recognisable model ID,
162
+ contains unexpected non-CSV files, or contains chunks with
163
+ mismatched prefixes.
164
+ """
165
+ logger = get_logger()
166
+ if not os.path.exists(dir_path):
167
+ raise EosframesError(f"Directory not found: '{dir_path}'")
168
+ model_id = get_model_id_from_path(dir_path)
169
+ if model_id is None:
170
+ raise EosframesError(
171
+ f"Could not extract a model ID from directory name '{dir_path}'."
172
+ )
173
+ logger.info("Reading chunked CSVs from: %s", dir_path)
174
+
175
+ filenames = os.listdir(dir_path)
176
+ if not filenames:
177
+ raise EosframesError(
178
+ f"Directory '{dir_path}' is empty. Expected one or more "
179
+ "chunk_<N>.csv files."
180
+ )
181
+ batch_ids = []
182
+ zfill = 0
183
+ prefixes = []
184
+ for fn in filenames:
185
+ if not fn.endswith(".csv") or not fn.startswith("chunk"):
186
+ raise EosframesError(
187
+ f"Unexpected file '{fn}' in '{dir_path}'. "
188
+ "Chunk folders should contain only files named chunk_<N>.csv."
189
+ )
190
+ parts = fn.split("_")
191
+ batch_id_str = parts[-1].split(".")[0]
192
+ zfill = len(batch_id_str)
193
+ batch_ids.append(int(batch_id_str))
194
+ prefixes.append("_".join(parts[:-1]))
195
+
196
+ if len(set(prefixes)) > 1:
197
+ raise EosframesError(
198
+ f"Multiple file prefixes found in '{dir_path}': {set(prefixes)}. "
199
+ "All chunk files must share the same prefix."
200
+ )
201
+
202
+ prefix = prefixes[0]
203
+ frames = [
204
+ pd.read_csv(os.path.join(dir_path, f"{prefix}_{str(i).zfill(zfill)}.csv"))
205
+ for i in sorted(batch_ids)
206
+ ]
207
+ df = pd.concat(frames, axis=0).reset_index(drop=True)
208
+ df.model_id = model_id
209
+ df.version = get_version_from_path(dir_path)
210
+ logger.info(
211
+ "Loaded %d rows from %d chunks (model_id=%s)", len(df), len(frames), model_id
212
+ )
213
+ return df