eosframes 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Ersilia Open Source Initiative
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,112 @@
1
+ Metadata-Version: 2.4
2
+ Name: eosframes
3
+ Version: 1.1.0
4
+ Summary: Ersilia utilities for working with tabular output data
5
+ License: MIT
6
+ License-File: LICENSE
7
+ Keywords: ersilia,cheminformatics,data,machine-learning
8
+ Author: Ersilia Open Source Initiative
9
+ Author-email: hello@ersilia.io
10
+ Requires-Python: >=3.8
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.8
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Programming Language :: Python :: 3.14
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Classifier: Topic :: Scientific/Engineering :: Chemistry
24
+ Requires-Dist: click (>=8.0)
25
+ Requires-Dist: h5py (>=3.10.0)
26
+ Requires-Dist: numpy (>=1.24.0)
27
+ Requires-Dist: pandas (>=2.0.0)
28
+ Requires-Dist: requests (>=2.31)
29
+ Requires-Dist: rich (>=10.0)
30
+ Project-URL: Homepage, https://github.com/ersilia-os/eosframes
31
+ Project-URL: Repository, https://github.com/ersilia-os/eosframes
32
+ Description-Content-Type: text/markdown
33
+
34
+ ![Work in Progress](https://img.shields.io/badge/status-work%20in%20progress-orange)
35
+
36
+ # Manipulating Ersilia's dataframes
37
+
38
+ `eosframes` is a library for manipulating inputs and outputs from the [Ersilia Model Hub](https://github.com/ersilia-os/ersilia). It splits, assembles, converts, scales, and summarises tabular model output files.
39
+
40
+ ## Installation
41
+
42
+ Python ≥ 3.8 is required.
43
+
44
+ ```bash
45
+ pip install eosframes
46
+ ```
47
+
48
+ Or from source:
49
+
50
+ ```bash
51
+ git clone https://github.com/ersilia-os/eosframes.git
52
+ cd eosframes
53
+ pip install -e .
54
+ ```
55
+
56
+ ## Quick start
57
+
58
+ Every file the library reads or writes encodes a model ID and version in its filename, e.g. `eos4e40_v1.csv` (model `eos4e40`, version `v1`).
59
+
60
+ ```bash
61
+ # Slice a big input CSV into chunks for parallel model runs
62
+ eosframes split compounds.csv -o chunks/ --chunksize 10000
63
+
64
+ # Stitch the per-batch outputs back into one file
65
+ eosframes append eos4e40_v1_000.csv eos4e40_v1_001.csv -o eos4e40_v1.csv
66
+
67
+ # Combine outputs from multiple models, side by side
68
+ eosframes stack eos4e40_v1.csv eos7m30_v1.csv -o project_eosmix.csv
69
+ ```
70
+
71
+ Everything the CLI does is also importable:
72
+
73
+ ```python
74
+ from eosframes import read_csv, hstack, fit, transform
75
+
76
+ df = read_csv("eos4e40_v1.csv")
77
+ params = fit(df)
78
+ scaled = transform(df, params)
79
+ ```
80
+
81
+ Run `eosframes --help` (or `eosframes <command> --help`) for inline help.
82
+
83
+ ## Commands
84
+
85
+ | Command | Purpose |
86
+ |-------------|---------------------------------------------------------------|
87
+ | `split` | Slice any CSV into chunk files for parallel model runs. |
88
+ | `convert` | CSV ↔ H5, or assemble a chunks folder. |
89
+ | `append` | Vertically concatenate batches from the same model. |
90
+ | `dedupe` | Drop duplicate rows by `key`. |
91
+ | `stack` | Horizontally combine outputs from different models. |
92
+ | `unstack` | Split a stacked file back into per-model files. |
93
+ | `summary` | Per-feature stats from a local file. |
94
+ | `info` | Model metadata fetched from GitHub. |
95
+ | `columns` | Feature definitions fetched from GitHub. |
96
+ | `fit` | Fit a type-aware robust scaler and save its parameters. |
97
+ | `transform` | Apply a saved scaler to a file. |
98
+
99
+ See [`docs/cli.md`](docs/cli.md) for every flag, example, and refusal condition.
100
+
101
+ ## Documentation
102
+
103
+ - [`docs/cli.md`](docs/cli.md) — every CLI command, all flags, examples, and error patterns.
104
+ - [`docs/nomenclature.md`](docs/nomenclature.md) — every recognised filename / directory pattern, the strict/lenient contract, and the two stack modes.
105
+ - [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, and quantization / imputation.
106
+
107
+ ## About the Ersilia Open Source Initiative
108
+
109
+ The [Ersilia Open Source Initiative](https://ersilia.io) is a tech-nonprofit fueling sustainable research in the Global South. Ersilia's main asset is the [Ersilia Model Hub](https://github.com/ersilia-os/ersilia), an open-source repository of AI/ML models for drug discovery.
110
+
111
+ ![Ersilia Logo](assets/Ersilia_Brand.png)
112
+
@@ -0,0 +1,78 @@
1
+ ![Work in Progress](https://img.shields.io/badge/status-work%20in%20progress-orange)
2
+
3
+ # Manipulating Ersilia's dataframes
4
+
5
+ `eosframes` is a library for manipulating inputs and outputs from the [Ersilia Model Hub](https://github.com/ersilia-os/ersilia). It splits, assembles, converts, scales, and summarises tabular model output files.
6
+
7
+ ## Installation
8
+
9
+ Python ≥ 3.8 is required.
10
+
11
+ ```bash
12
+ pip install eosframes
13
+ ```
14
+
15
+ Or from source:
16
+
17
+ ```bash
18
+ git clone https://github.com/ersilia-os/eosframes.git
19
+ cd eosframes
20
+ pip install -e .
21
+ ```
22
+
23
+ ## Quick start
24
+
25
+ Every file the library reads or writes encodes a model ID and version in its filename, e.g. `eos4e40_v1.csv` (model `eos4e40`, version `v1`).
26
+
27
+ ```bash
28
+ # Slice a big input CSV into chunks for parallel model runs
29
+ eosframes split compounds.csv -o chunks/ --chunksize 10000
30
+
31
+ # Stitch the per-batch outputs back into one file
32
+ eosframes append eos4e40_v1_000.csv eos4e40_v1_001.csv -o eos4e40_v1.csv
33
+
34
+ # Combine outputs from multiple models, side by side
35
+ eosframes stack eos4e40_v1.csv eos7m30_v1.csv -o project_eosmix.csv
36
+ ```
37
+
38
+ Everything the CLI does is also importable:
39
+
40
+ ```python
41
+ from eosframes import read_csv, hstack, fit, transform
42
+
43
+ df = read_csv("eos4e40_v1.csv")
44
+ params = fit(df)
45
+ scaled = transform(df, params)
46
+ ```
47
+
48
+ Run `eosframes --help` (or `eosframes <command> --help`) for inline help.
49
+
50
+ ## Commands
51
+
52
+ | Command | Purpose |
53
+ |-------------|---------------------------------------------------------------|
54
+ | `split` | Slice any CSV into chunk files for parallel model runs. |
55
+ | `convert` | CSV ↔ H5, or assemble a chunks folder. |
56
+ | `append` | Vertically concatenate batches from the same model. |
57
+ | `dedupe` | Drop duplicate rows by `key`. |
58
+ | `stack` | Horizontally combine outputs from different models. |
59
+ | `unstack` | Split a stacked file back into per-model files. |
60
+ | `summary` | Per-feature stats from a local file. |
61
+ | `info` | Model metadata fetched from GitHub. |
62
+ | `columns` | Feature definitions fetched from GitHub. |
63
+ | `fit` | Fit a type-aware robust scaler and save its parameters. |
64
+ | `transform` | Apply a saved scaler to a file. |
65
+
66
+ See [`docs/cli.md`](docs/cli.md) for every flag, example, and refusal condition.
67
+
68
+ ## Documentation
69
+
70
+ - [`docs/cli.md`](docs/cli.md) — every CLI command, all flags, examples, and error patterns.
71
+ - [`docs/nomenclature.md`](docs/nomenclature.md) — every recognised filename / directory pattern, the strict/lenient contract, and the two stack modes.
72
+ - [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, and quantization / imputation.
73
+
74
+ ## About the Ersilia Open Source Initiative
75
+
76
+ The [Ersilia Open Source Initiative](https://ersilia.io) is a tech-nonprofit fueling sustainable research in the Global South. Ersilia's main asset is the [Ersilia Model Hub](https://github.com/ersilia-os/ersilia), an open-source repository of AI/ML models for drug discovery.
77
+
78
+ ![Ersilia Logo](assets/Ersilia_Brand.png)
@@ -0,0 +1,88 @@
1
+ [build-system]
2
+ requires = ["poetry-core>=1.0.0", "poetry-dynamic-versioning>=1.0.0,<2.0.0"]
3
+ build-backend = "poetry_dynamic_versioning.backend"
4
+
5
+ [tool.poetry-dynamic-versioning]
6
+ enable = false
7
+ vcs = "git"
8
+ style = "pep440"
9
+
10
+ [tool.poetry]
11
+ name = "eosframes"
12
+ version = "1.1.0"
13
+ description = "Ersilia utilities for working with tabular output data"
14
+ authors = ["Ersilia Open Source Initiative <hello@ersilia.io>"]
15
+ license = "MIT"
16
+ readme = "README.md"
17
+ homepage = "https://github.com/ersilia-os/eosframes"
18
+ repository = "https://github.com/ersilia-os/eosframes"
19
+ keywords = ["ersilia", "cheminformatics", "data", "machine-learning"]
20
+ classifiers = [
21
+ "Development Status :: 3 - Alpha",
22
+ "Intended Audience :: Science/Research",
23
+ "License :: OSI Approved :: MIT License",
24
+ "Programming Language :: Python :: 3",
25
+ "Programming Language :: Python :: 3.8",
26
+ "Programming Language :: Python :: 3.9",
27
+ "Programming Language :: Python :: 3.10",
28
+ "Programming Language :: Python :: 3.11",
29
+ "Programming Language :: Python :: 3.12",
30
+ "Topic :: Scientific/Engineering :: Chemistry",
31
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
32
+ ]
33
+ packages = [{include = "eosframes", from = "src"}]
34
+
35
+ [tool.poetry.dependencies]
36
+ python = ">=3.8"
37
+ pandas = ">=2.0.0"
38
+ numpy = ">=1.24.0"
39
+ h5py = ">=3.10.0"
40
+ requests = ">=2.31"
41
+ click = ">=8.0"
42
+ rich = ">=10.0"
43
+
44
+ [tool.poetry.group.dev.dependencies]
45
+ ruff = ">=0.4,<1"
46
+ pytest = ">=7.0"
47
+
48
+ [tool.poetry.scripts]
49
+ eosframes = "eosframes.cli:main"
50
+
51
+ [tool.ruff]
52
+ line-length = 88
53
+ target-version = "py38"
54
+
55
+ [tool.ruff.lint]
56
+ select = ["E", "F", "W", "I", "UP", "B", "SIM", "D"]
57
+ ignore = [
58
+ "E501", # line too long — handled by formatter
59
+ "B008", # do not perform function calls in default arguments (Click uses this)
60
+ "UP007", # use X | Y for type annotations (not available in 3.8)
61
+ "UP006", # use type instead of Type (not available in 3.8)
62
+ "D203", # conflicts with D211 (no blank line before class docstring)
63
+ "D213", # conflicts with D212 (summary on first line)
64
+ "D301", # backslashes — Click uses \b in help text, raw-strings hurt readability
65
+ "D401", # imperative mood — too opinionated for descriptive docstrings
66
+ "D403", # first-word capitalisation — the brand "eosframes" stays lowercase
67
+ "D415", # first line ends with punctuation — D400 already covers periods
68
+ ]
69
+
70
+ [tool.ruff.lint.pydocstyle]
71
+ convention = "numpy"
72
+
73
+ [tool.ruff.lint.per-file-ignores]
74
+ # Tests document themselves via descriptive names; don't require docstrings.
75
+ "tests/*" = ["D"]
76
+ # Subpackage __init__.py docstrings live in the modules they expose.
77
+ "src/eosframes/**/__init__.py" = ["D104"]
78
+
79
+ [tool.ruff.lint.isort]
80
+ known-first-party = ["eosframes"]
81
+
82
+ [tool.pytest.ini_options]
83
+ testpaths = ["tests"]
84
+ # Disable pytest's fd-level capture so the eosframes logger's handler
85
+ # (pinned to sys.__stderr__ in tests/conftest.py) reaches the real terminal
86
+ # during test runs. Without -s, pytest captures fd 2 even before the
87
+ # Python-level sys.__stderr__ stream sees it.
88
+ addopts = "-s"
@@ -0,0 +1,121 @@
1
+ """``eosframes`` — utilities for Ersilia model output dataframes.
2
+
3
+ Split, assemble, convert, stack, scale, and summarise tabular outputs from
4
+ the `Ersilia Model Hub <https://github.com/ersilia-os/ersilia>`_. The CLI
5
+ entry point is ``eosframes`` (registered as a ``poetry`` script); the same
6
+ operations are available as Python functions imported from this package.
7
+
8
+ The architecture is layered: ``cli`` → ``ops`` / ``scale`` →
9
+ ``read`` / ``write`` / ``stack`` / ``naming`` / ``hub``. Every
10
+ DataFrame that flows through the library carries two loose attributes —
11
+ ``df.model_id`` (always) and ``df.version`` (when the filename encodes
12
+ it) — and write paths cross-validate them against the destination
13
+ filename. See ``CLAUDE.md`` for the full attribute and naming contracts.
14
+ """
15
+
16
+ from importlib.metadata import PackageNotFoundError
17
+ from importlib.metadata import version as _pkg_version
18
+
19
+ try:
20
+ __version__ = _pkg_version("eosframes")
21
+ except PackageNotFoundError:
22
+ # Not installed (e.g. running from a source checkout without
23
+ # `pip install -e .`). Keep the package importable but signal the
24
+ # situation — scaler files will fail the version-match check.
25
+ __version__ = "0.0.0+unknown"
26
+
27
+ from .exceptions import EosframesError
28
+ from .hub import fetch_columns, fetch_metadata
29
+ from .logger import get_logger, set_verbosity
30
+ from .naming import (
31
+ get_model_id_from_path,
32
+ get_version_from_path,
33
+ is_model_id_valid,
34
+ is_valid_columns_name,
35
+ is_valid_info_name,
36
+ is_valid_name,
37
+ is_valid_stack_explicit_name,
38
+ is_valid_stack_mix_name,
39
+ is_valid_summary_name,
40
+ is_valid_transformer_name,
41
+ make_chunks_dir_name,
42
+ make_columns_name,
43
+ make_info_name,
44
+ make_output_name,
45
+ make_stack_explicit_name,
46
+ make_stack_mix_name,
47
+ make_summary_name,
48
+ make_transformer_name,
49
+ parse_name,
50
+ parse_stack_explicit_name,
51
+ parse_stack_mix_name,
52
+ parse_transformer_name,
53
+ )
54
+ from .ops import (
55
+ append_files,
56
+ convert_file,
57
+ dedupe_file,
58
+ split_csv,
59
+ stack_files,
60
+ unstack_file,
61
+ )
62
+ from .read import read_chunked_csvs, read_csv, read_h5
63
+ from .scale import fit, fit_file, transform, transform_file
64
+ from .stack import hstack, vstack
65
+ from .write import write_chunked_csvs, write_csv, write_h5
66
+
67
+ __all__ = [
68
+ "__version__",
69
+ # Low-level I/O
70
+ "read_csv",
71
+ "read_h5",
72
+ "read_chunked_csvs",
73
+ "write_csv",
74
+ "write_h5",
75
+ "write_chunked_csvs",
76
+ # DataFrame operations
77
+ "hstack",
78
+ "vstack",
79
+ # File-level operations
80
+ "split_csv",
81
+ "convert_file",
82
+ "stack_files",
83
+ "unstack_file",
84
+ "append_files",
85
+ "dedupe_file",
86
+ # GitHub / hub
87
+ "fetch_metadata",
88
+ "fetch_columns",
89
+ # Scaling
90
+ "fit",
91
+ "transform",
92
+ "fit_file",
93
+ "transform_file",
94
+ # Naming utilities
95
+ "parse_name",
96
+ "make_output_name",
97
+ "make_chunks_dir_name",
98
+ "make_info_name",
99
+ "make_columns_name",
100
+ "make_summary_name",
101
+ "make_transformer_name",
102
+ "make_stack_mix_name",
103
+ "make_stack_explicit_name",
104
+ "parse_stack_mix_name",
105
+ "parse_stack_explicit_name",
106
+ "parse_transformer_name",
107
+ "get_version_from_path",
108
+ "get_model_id_from_path",
109
+ "is_valid_name",
110
+ "is_valid_info_name",
111
+ "is_valid_columns_name",
112
+ "is_valid_summary_name",
113
+ "is_valid_stack_mix_name",
114
+ "is_valid_stack_explicit_name",
115
+ "is_valid_transformer_name",
116
+ "is_model_id_valid",
117
+ # Misc
118
+ "get_logger",
119
+ "set_verbosity",
120
+ "EosframesError",
121
+ ]