chemsplit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- chemsplit-0.1.0/LICENSE +21 -0
- chemsplit-0.1.0/PKG-INFO +186 -0
- chemsplit-0.1.0/README.md +141 -0
- chemsplit-0.1.0/pyproject.toml +117 -0
- chemsplit-0.1.0/setup.cfg +4 -0
- chemsplit-0.1.0/src/chemsplit/__init__.py +250 -0
- chemsplit-0.1.0/src/chemsplit/__main__.py +8 -0
- chemsplit-0.1.0/src/chemsplit/_devtools.py +420 -0
- chemsplit-0.1.0/src/chemsplit/_fp_similarity.py +154 -0
- chemsplit-0.1.0/src/chemsplit/_optimize.py +781 -0
- chemsplit-0.1.0/src/chemsplit/_pair_assign.py +143 -0
- chemsplit-0.1.0/src/chemsplit/_unionfind.py +94 -0
- chemsplit-0.1.0/src/chemsplit/audit.py +577 -0
- chemsplit-0.1.0/src/chemsplit/base.py +977 -0
- chemsplit-0.1.0/src/chemsplit/cli.py +135 -0
- chemsplit-0.1.0/src/chemsplit/clustering.py +288 -0
- chemsplit-0.1.0/src/chemsplit/datasets.py +656 -0
- chemsplit-0.1.0/src/chemsplit/determinism.py +160 -0
- chemsplit-0.1.0/src/chemsplit/exceptions.py +252 -0
- chemsplit-0.1.0/src/chemsplit/featurizers/__init__.py +101 -0
- chemsplit-0.1.0/src/chemsplit/featurizers/descriptors.py +78 -0
- chemsplit-0.1.0/src/chemsplit/featurizers/fingerprints.py +205 -0
- chemsplit-0.1.0/src/chemsplit/featurizers/precomputed.py +28 -0
- chemsplit-0.1.0/src/chemsplit/metrics.py +284 -0
- chemsplit-0.1.0/src/chemsplit/preprocess.py +418 -0
- chemsplit-0.1.0/src/chemsplit/registry.py +306 -0
- chemsplit-0.1.0/src/chemsplit/scaffolds.py +331 -0
- chemsplit-0.1.0/src/chemsplit/splitters/__init__.py +4 -0
- chemsplit-0.1.0/src/chemsplit/splitters/baseline.py +816 -0
- chemsplit-0.1.0/src/chemsplit/splitters/biomolecular.py +579 -0
- chemsplit-0.1.0/src/chemsplit/splitters/embedding.py +645 -0
- chemsplit-0.1.0/src/chemsplit/splitters/lineage.py +967 -0
- chemsplit-0.1.0/src/chemsplit/splitters/property_.py +983 -0
- chemsplit-0.1.0/src/chemsplit/splitters/protocol.py +772 -0
- chemsplit-0.1.0/src/chemsplit/splitters/scaffold.py +1220 -0
- chemsplit-0.1.0/src/chemsplit/splitters/similarity.py +1613 -0
- chemsplit-0.1.0/src/chemsplit/splitters/task.py +1692 -0
- chemsplit-0.1.0/src/chemsplit/types.py +41 -0
- chemsplit-0.1.0/src/chemsplit.egg-info/PKG-INFO +186 -0
- chemsplit-0.1.0/src/chemsplit.egg-info/SOURCES.txt +64 -0
- chemsplit-0.1.0/src/chemsplit.egg-info/dependency_links.txt +1 -0
- chemsplit-0.1.0/src/chemsplit.egg-info/entry_points.txt +2 -0
- chemsplit-0.1.0/src/chemsplit.egg-info/requires.txt +32 -0
- chemsplit-0.1.0/src/chemsplit.egg-info/top_level.txt +1 -0
- chemsplit-0.1.0/tests/test_api_contract.py +405 -0
- chemsplit-0.1.0/tests/test_audit.py +185 -0
- chemsplit-0.1.0/tests/test_base.py +710 -0
- chemsplit-0.1.0/tests/test_cli.py +124 -0
- chemsplit-0.1.0/tests/test_clustering.py +226 -0
- chemsplit-0.1.0/tests/test_datasets.py +172 -0
- chemsplit-0.1.0/tests/test_determinism.py +87 -0
- chemsplit-0.1.0/tests/test_determinism_lint.py +35 -0
- chemsplit-0.1.0/tests/test_devtools.py +157 -0
- chemsplit-0.1.0/tests/test_exceptions.py +60 -0
- chemsplit-0.1.0/tests/test_featurizers.py +213 -0
- chemsplit-0.1.0/tests/test_fp_similarity.py +115 -0
- chemsplit-0.1.0/tests/test_golden.py +86 -0
- chemsplit-0.1.0/tests/test_metrics.py +254 -0
- chemsplit-0.1.0/tests/test_optimize.py +189 -0
- chemsplit-0.1.0/tests/test_package_import.py +71 -0
- chemsplit-0.1.0/tests/test_pair_assign.py +111 -0
- chemsplit-0.1.0/tests/test_preprocess.py +311 -0
- chemsplit-0.1.0/tests/test_property_based.py +439 -0
- chemsplit-0.1.0/tests/test_registry.py +100 -0
- chemsplit-0.1.0/tests/test_scaffolds.py +159 -0
- chemsplit-0.1.0/tests/test_unionfind.py +25 -0
chemsplit-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Olivier J. M. Béquignon
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
chemsplit-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: chemsplit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A self-contained, scikit-learn-compatible library of dataset-splitting strategies for cheminformatics machine learning.
|
|
5
|
+
Author-email: "Olivier J. M. Béquignon" <olivier.bequignon.maintainer@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/OlivierBeq/chemsplit
|
|
8
|
+
Keywords: cheminformatics,machine-learning,dataset-splitting,scikit-learn,rdkit
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Chemistry
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Python: <3.14,>=3.11
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: numpy<3,>=1.24
|
|
22
|
+
Requires-Dist: scipy>=1.10
|
|
23
|
+
Requires-Dist: scikit-learn>=1.3
|
|
24
|
+
Requires-Dist: rdkit>=2023.09.1
|
|
25
|
+
Requires-Dist: pandas>=2.0
|
|
26
|
+
Requires-Dist: scaffound>=0.0.1
|
|
27
|
+
Provides-Extra: umap
|
|
28
|
+
Requires-Dist: umap-learn>=0.5.4; extra == "umap"
|
|
29
|
+
Provides-Extra: hdbscan
|
|
30
|
+
Requires-Dist: scikit-learn>=1.3; extra == "hdbscan"
|
|
31
|
+
Provides-Extra: ga
|
|
32
|
+
Requires-Dist: deap>=1.4; extra == "ga"
|
|
33
|
+
Provides-Extra: bio
|
|
34
|
+
Requires-Dist: biopython>=1.81; extra == "bio"
|
|
35
|
+
Requires-Dist: parasail>=1.3; (sys_platform != "darwin" or platform_machine != "arm64") and extra == "bio"
|
|
36
|
+
Provides-Extra: mmpa
|
|
37
|
+
Provides-Extra: all
|
|
38
|
+
Requires-Dist: chemsplit[bio,ga,hdbscan,mmpa,umap]; extra == "all"
|
|
39
|
+
Provides-Extra: dev
|
|
40
|
+
Requires-Dist: pytest>=7.4; extra == "dev"
|
|
41
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
42
|
+
Requires-Dist: hypothesis>=6.90; extra == "dev"
|
|
43
|
+
Requires-Dist: ruff>=0.16; extra == "dev"
|
|
44
|
+
Dynamic: license-file
|
|
45
|
+
|
|
46
|
+
<div align="center">
|
|
47
|
+
|
|
48
|
+
# ✂️ chemsplit
|
|
49
|
+
|
|
50
|
+
[](https://pypi.org/project/chemsplit/)
|
|
51
|
+
[](https://pypi.org/project/chemsplit/)
|
|
52
|
+
[](https://opensource.org/licenses/MIT)
|
|
53
|
+
[](https://github.com/OlivierBeq/chemsplit/actions/workflows/ci.yml)
|
|
54
|
+
[](https://github.com/astral-sh/ruff)
|
|
55
|
+
|
|
56
|
+
</div>
|
|
57
|
+
|
|
58
|
+
A self-contained, scikit-learn-compatible Python library of dataset-splitting strategies for cheminformatics machine learning. `chemsplit` implements **52 splitting strategies across nine families** — baseline, scaffold, similarity, embedding, property, lineage, task, biomolecular, and protocol splitters — behind one coherent, deterministic API, plus a leakage-audit module and a set of reference/synthetic datasets to try them on.
|
|
59
|
+
|
|
60
|
+
## ✨ Features
|
|
61
|
+
|
|
62
|
+
- 🧩 **9 families, 52 strategies** — from a plain random split to scaffold-tree pruning, Butina/spectral clustering, UMAP-space holdouts, temporal and provenance cuts, protein-family and binding-site holdouts, drug-target cold-start benchmarks, and full CV/nested-CV protocol wrappers.
|
|
63
|
+
- 🎯 **Deterministic by construction** — every splitter accepts a `random_state` and produces bit-identical output regardless of record order or `n_jobs`, checked continuously by a golden-file regression suite and Hypothesis property tests.
|
|
64
|
+
- 🛡️ **Contract-checked results** — every `SplitResult` is validated against five structural invariants (index coverage, disjointness, group-label consistency, JSON round-tripping of `params`, id format) before it ever reaches your code.
|
|
65
|
+
- 🔍 **Built-in leakage auditing** — `chemsplit.audit` reports nearest-neighbour similarity, adversarial-validation AUC, exact/scaffold/ring-system overlap, and property/label shift between train and test.
|
|
66
|
+
- 🧪 **Chemistry-native featurization** — ECFP/FCFP/MACCS/Avalon/atom-pair/topological-torsion fingerprints and physicochemical descriptors, behind a pluggable `Featurizer` protocol for your own.
|
|
67
|
+
- 💻 **CLI included** — run, audit, and list any registered splitter without writing a line of Python.
|
|
68
|
+
- 📚 **One example notebook per family** — runnable, narrated walkthroughs of every splitter class under [`notebooks/`](notebooks/).
|
|
69
|
+
|
|
70
|
+
## 📦 Installation
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
pip install chemsplit
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Optional extras enable additional splitters and featurizers:
|
|
77
|
+
|
|
78
|
+
| Extra | Enables |
|
|
79
|
+
|---|---|
|
|
80
|
+
| `chemsplit[umap]` | `UMAPClusterSplitter` (UMAP embedding + clustering) |
|
|
81
|
+
| `chemsplit[hdbscan]` | `DensityClusterSplitter` (HDBSCAN density clustering) |
|
|
82
|
+
| `chemsplit[ga]` | `SIMPDSplitter` (genetic-algorithm pseudo-time optimization, via `deap`) |
|
|
83
|
+
| `chemsplit[bio]` | Protein-sequence splitters with accelerated alignment (`biopython`, `parasail`) |
|
|
84
|
+
| `chemsplit[mmpa]` | `MatchedMolecularSeriesSplitter` matched-series extraction |
|
|
85
|
+
| `chemsplit[all]` | Everything above |
|
|
86
|
+
|
|
87
|
+
> **Note:** every extra has a dependency-free fallback where one makes sense (e.g. a Hamming-distance fallback for sequence identity without `bio`) — an extra buys you a better implementation, not a hard requirement.
|
|
88
|
+
|
|
89
|
+
## 🛠️ Requirements
|
|
90
|
+
|
|
91
|
+
- Python 3.11 – 3.13
|
|
92
|
+
- [RDKit](https://www.rdkit.org/) (installed automatically as a core dependency — no separate conda step needed)
|
|
93
|
+
|
|
94
|
+
## 💡 Usage
|
|
95
|
+
|
|
96
|
+
### Quickstart
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
from chemsplit import datasets, get_splitter
|
|
100
|
+
|
|
101
|
+
fx = datasets.make_scaffold_families(n_scaffolds=10, per_scaffold=15)
|
|
102
|
+
|
|
103
|
+
splitter = get_splitter("scaffold_tree", train_size=0.8, test_size=0.2)
|
|
104
|
+
result = splitter.split_result(fx.smiles)[0]
|
|
105
|
+
|
|
106
|
+
train_smiles = [fx.smiles[i] for i in result.train]
|
|
107
|
+
test_smiles = [fx.smiles[i] for i in result.test]
|
|
108
|
+
print(f"{len(train_smiles)} train / {len(test_smiles)} test records")
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Every splitter is resolved the same way, by its `splitter_id` (see `chemsplit.list_splitters()` for the full table) or by importing the class directly:
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
from chemsplit import ButinaSplitter
|
|
115
|
+
|
|
116
|
+
splitter = ButinaSplitter(cutoff=0.4, train_size=0.7, test_size=0.3, random_state=0)
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
<details>
|
|
120
|
+
<summary><strong>🔍 Auditing a split for leakage</strong></summary>
|
|
121
|
+
|
|
122
|
+
```python
|
|
123
|
+
from chemsplit import audit_split
|
|
124
|
+
|
|
125
|
+
report = audit_split(result, fx.smiles)
|
|
126
|
+
print(report.summary())
|
|
127
|
+
```
|
|
128
|
+
```text
|
|
129
|
+
chemsplit LeakageReport
|
|
130
|
+
------------------------
|
|
131
|
+
n_train=120 n_valid=0 n_test=30 n_discard=0
|
|
132
|
+
max cross-partition similarity: 0.4375
|
|
133
|
+
median NN similarity (test->train): 0.6667
|
|
134
|
+
exact duplicates across partitions: 0
|
|
135
|
+
shared scaffolds: 0 shared ring systems: 0
|
|
136
|
+
adversarial AUC: 1.0000 (95% CI 1.0000-1.0000)
|
|
137
|
+
flags: MEDIAN_NN_ABOVE_0.6, HIGH_ADVERSARIAL_AUC, ...
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
`LeakageReport` is purely descriptive — it never fails your pipeline, it tells you what to look at.
|
|
141
|
+
</details>
|
|
142
|
+
|
|
143
|
+
<details>
|
|
144
|
+
<summary><strong>💻 The CLI</strong></summary>
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
# List every registered splitter, optionally filtered by family
|
|
148
|
+
chemsplit list --family scaffold
|
|
149
|
+
|
|
150
|
+
# Run a splitter over a CSV and write the split to disk
|
|
151
|
+
chemsplit split --splitter scaffold_tree --input data.csv --smiles-col smiles \
|
|
152
|
+
--train-size 0.8 --test-size 0.2 --seed 0 --out split.json
|
|
153
|
+
|
|
154
|
+
# Audit that split for leakage
|
|
155
|
+
chemsplit audit --split split.json --input data.csv --smiles-col smiles --out report.json
|
|
156
|
+
|
|
157
|
+
chemsplit --help
|
|
158
|
+
```
|
|
159
|
+
</details>
|
|
160
|
+
|
|
161
|
+
## 🧭 Design principles
|
|
162
|
+
|
|
163
|
+
1. **Determinism.** Same inputs + same `random_state` ⇒ byte-identical outputs, on any platform, any CPU count, any `n_jobs`.
|
|
164
|
+
2. **Explicitness.** No silent fallbacks. If a requested configuration is infeasible, raise, never approximate.
|
|
165
|
+
3. **Honesty.** Every splitter's docstring discloses its pitfalls with the same prominence as its advantages.
|
|
166
|
+
4. **Composability.** Every group-forming splitter exposes its group labels, so any grouping can be fed to any protocol wrapper.
|
|
167
|
+
5. **scikit-learn compatibility.** `.split()` is drop-in usable in `cross_val_score`, `GridSearchCV(cv=...)`, and `cross_validate`.
|
|
168
|
+
|
|
169
|
+
## 📚 Learn more
|
|
170
|
+
|
|
171
|
+
One narrated, runnable notebook per family, under [`notebooks/`](notebooks/):
|
|
172
|
+
|
|
173
|
+
- [`baseline.ipynb`](notebooks/baseline.ipynb) — uninformed and gold-standard controls
|
|
174
|
+
- [`scaffold.ipynb`](notebooks/scaffold.ipynb) — chemotype-aware generalization tests
|
|
175
|
+
- [`similarity.ipynb`](notebooks/similarity.ipynb) — fingerprint-clustering holdouts
|
|
176
|
+
- [`embedding.ipynb`](notebooks/embedding.ipynb) — latent-space holdouts
|
|
177
|
+
- [`property.ipynb`](notebooks/property.ipynb) — label-shift and distributional stress tests
|
|
178
|
+
- [`lineage.ipynb`](notebooks/lineage.ipynb) — date and provenance-aware holdouts
|
|
179
|
+
- [`biomolecular.ipynb`](notebooks/biomolecular.ipynb) — protein-axis holdouts
|
|
180
|
+
- [`task.ipynb`](notebooks/task.ipynb) — drug-target interaction benchmarks
|
|
181
|
+
- [`protocol.ipynb`](notebooks/protocol.ipynb) — cross-validation and evaluation harnesses
|
|
182
|
+
|
|
183
|
+
## 📄 License
|
|
184
|
+
|
|
185
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
186
|
+
</content>
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+
# ✂️ chemsplit
|
|
4
|
+
|
|
5
|
+
[](https://pypi.org/project/chemsplit/)
|
|
6
|
+
[](https://pypi.org/project/chemsplit/)
|
|
7
|
+
[](https://opensource.org/licenses/MIT)
|
|
8
|
+
[](https://github.com/OlivierBeq/chemsplit/actions/workflows/ci.yml)
|
|
9
|
+
[](https://github.com/astral-sh/ruff)
|
|
10
|
+
|
|
11
|
+
</div>
|
|
12
|
+
|
|
13
|
+
A self-contained, scikit-learn-compatible Python library of dataset-splitting strategies for cheminformatics machine learning. `chemsplit` implements **52 splitting strategies across nine families** — baseline, scaffold, similarity, embedding, property, lineage, task, biomolecular, and protocol splitters — behind one coherent, deterministic API, plus a leakage-audit module and a set of reference/synthetic datasets to try them on.
|
|
14
|
+
|
|
15
|
+
## ✨ Features
|
|
16
|
+
|
|
17
|
+
- 🧩 **9 families, 52 strategies** — from a plain random split to scaffold-tree pruning, Butina/spectral clustering, UMAP-space holdouts, temporal and provenance cuts, protein-family and binding-site holdouts, drug-target cold-start benchmarks, and full CV/nested-CV protocol wrappers.
|
|
18
|
+
- 🎯 **Deterministic by construction** — every splitter accepts a `random_state` and produces bit-identical output regardless of record order or `n_jobs`, checked continuously by a golden-file regression suite and Hypothesis property tests.
|
|
19
|
+
- 🛡️ **Contract-checked results** — every `SplitResult` is validated against five structural invariants (index coverage, disjointness, group-label consistency, JSON round-tripping of `params`, id format) before it ever reaches your code.
|
|
20
|
+
- 🔍 **Built-in leakage auditing** — `chemsplit.audit` reports nearest-neighbour similarity, adversarial-validation AUC, exact/scaffold/ring-system overlap, and property/label shift between train and test.
|
|
21
|
+
- 🧪 **Chemistry-native featurization** — ECFP/FCFP/MACCS/Avalon/atom-pair/topological-torsion fingerprints and physicochemical descriptors, behind a pluggable `Featurizer` protocol for your own.
|
|
22
|
+
- 💻 **CLI included** — run, audit, and list any registered splitter without writing a line of Python.
|
|
23
|
+
- 📚 **One example notebook per family** — runnable, narrated walkthroughs of every splitter class under [`notebooks/`](notebooks/).
|
|
24
|
+
|
|
25
|
+
## 📦 Installation
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install chemsplit
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Optional extras enable additional splitters and featurizers:
|
|
32
|
+
|
|
33
|
+
| Extra | Enables |
|
|
34
|
+
|---|---|
|
|
35
|
+
| `chemsplit[umap]` | `UMAPClusterSplitter` (UMAP embedding + clustering) |
|
|
36
|
+
| `chemsplit[hdbscan]` | `DensityClusterSplitter` (HDBSCAN density clustering) |
|
|
37
|
+
| `chemsplit[ga]` | `SIMPDSplitter` (genetic-algorithm pseudo-time optimization, via `deap`) |
|
|
38
|
+
| `chemsplit[bio]` | Protein-sequence splitters with accelerated alignment (`biopython`, `parasail`) |
|
|
39
|
+
| `chemsplit[mmpa]` | `MatchedMolecularSeriesSplitter` matched-series extraction |
|
|
40
|
+
| `chemsplit[all]` | Everything above |
|
|
41
|
+
|
|
42
|
+
> **Note:** every extra has a dependency-free fallback where one makes sense (e.g. a Hamming-distance fallback for sequence identity without `bio`) — an extra buys you a better implementation, not a hard requirement.
|
|
43
|
+
|
|
44
|
+
## 🛠️ Requirements
|
|
45
|
+
|
|
46
|
+
- Python 3.11 – 3.13
|
|
47
|
+
- [RDKit](https://www.rdkit.org/) (installed automatically as a core dependency — no separate conda step needed)
|
|
48
|
+
|
|
49
|
+
## 💡 Usage
|
|
50
|
+
|
|
51
|
+
### Quickstart
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
from chemsplit import datasets, get_splitter
|
|
55
|
+
|
|
56
|
+
fx = datasets.make_scaffold_families(n_scaffolds=10, per_scaffold=15)
|
|
57
|
+
|
|
58
|
+
splitter = get_splitter("scaffold_tree", train_size=0.8, test_size=0.2)
|
|
59
|
+
result = splitter.split_result(fx.smiles)[0]
|
|
60
|
+
|
|
61
|
+
train_smiles = [fx.smiles[i] for i in result.train]
|
|
62
|
+
test_smiles = [fx.smiles[i] for i in result.test]
|
|
63
|
+
print(f"{len(train_smiles)} train / {len(test_smiles)} test records")
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Every splitter is resolved the same way, by its `splitter_id` (see `chemsplit.list_splitters()` for the full table) or by importing the class directly:
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from chemsplit import ButinaSplitter
|
|
70
|
+
|
|
71
|
+
splitter = ButinaSplitter(cutoff=0.4, train_size=0.7, test_size=0.3, random_state=0)
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
<details>
|
|
75
|
+
<summary><strong>🔍 Auditing a split for leakage</strong></summary>
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from chemsplit import audit_split
|
|
79
|
+
|
|
80
|
+
report = audit_split(result, fx.smiles)
|
|
81
|
+
print(report.summary())
|
|
82
|
+
```
|
|
83
|
+
```text
|
|
84
|
+
chemsplit LeakageReport
|
|
85
|
+
------------------------
|
|
86
|
+
n_train=120 n_valid=0 n_test=30 n_discard=0
|
|
87
|
+
max cross-partition similarity: 0.4375
|
|
88
|
+
median NN similarity (test->train): 0.6667
|
|
89
|
+
exact duplicates across partitions: 0
|
|
90
|
+
shared scaffolds: 0 shared ring systems: 0
|
|
91
|
+
adversarial AUC: 1.0000 (95% CI 1.0000-1.0000)
|
|
92
|
+
flags: MEDIAN_NN_ABOVE_0.6, HIGH_ADVERSARIAL_AUC, ...
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
`LeakageReport` is purely descriptive — it never fails your pipeline, it tells you what to look at.
|
|
96
|
+
</details>
|
|
97
|
+
|
|
98
|
+
<details>
|
|
99
|
+
<summary><strong>💻 The CLI</strong></summary>
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
# List every registered splitter, optionally filtered by family
|
|
103
|
+
chemsplit list --family scaffold
|
|
104
|
+
|
|
105
|
+
# Run a splitter over a CSV and write the split to disk
|
|
106
|
+
chemsplit split --splitter scaffold_tree --input data.csv --smiles-col smiles \
|
|
107
|
+
--train-size 0.8 --test-size 0.2 --seed 0 --out split.json
|
|
108
|
+
|
|
109
|
+
# Audit that split for leakage
|
|
110
|
+
chemsplit audit --split split.json --input data.csv --smiles-col smiles --out report.json
|
|
111
|
+
|
|
112
|
+
chemsplit --help
|
|
113
|
+
```
|
|
114
|
+
</details>
|
|
115
|
+
|
|
116
|
+
## 🧭 Design principles
|
|
117
|
+
|
|
118
|
+
1. **Determinism.** Same inputs + same `random_state` ⇒ byte-identical outputs, on any platform, any CPU count, any `n_jobs`.
|
|
119
|
+
2. **Explicitness.** No silent fallbacks. If a requested configuration is infeasible, raise, never approximate.
|
|
120
|
+
3. **Honesty.** Every splitter's docstring discloses its pitfalls with the same prominence as its advantages.
|
|
121
|
+
4. **Composability.** Every group-forming splitter exposes its group labels, so any grouping can be fed to any protocol wrapper.
|
|
122
|
+
5. **scikit-learn compatibility.** `.split()` is drop-in usable in `cross_val_score`, `GridSearchCV(cv=...)`, and `cross_validate`.
|
|
123
|
+
|
|
124
|
+
## 📚 Learn more
|
|
125
|
+
|
|
126
|
+
One narrated, runnable notebook per family, under [`notebooks/`](notebooks/):
|
|
127
|
+
|
|
128
|
+
- [`baseline.ipynb`](notebooks/baseline.ipynb) — uninformed and gold-standard controls
|
|
129
|
+
- [`scaffold.ipynb`](notebooks/scaffold.ipynb) — chemotype-aware generalization tests
|
|
130
|
+
- [`similarity.ipynb`](notebooks/similarity.ipynb) — fingerprint-clustering holdouts
|
|
131
|
+
- [`embedding.ipynb`](notebooks/embedding.ipynb) — latent-space holdouts
|
|
132
|
+
- [`property.ipynb`](notebooks/property.ipynb) — label-shift and distributional stress tests
|
|
133
|
+
- [`lineage.ipynb`](notebooks/lineage.ipynb) — date and provenance-aware holdouts
|
|
134
|
+
- [`biomolecular.ipynb`](notebooks/biomolecular.ipynb) — protein-axis holdouts
|
|
135
|
+
- [`task.ipynb`](notebooks/task.ipynb) — drug-target interaction benchmarks
|
|
136
|
+
- [`protocol.ipynb`](notebooks/protocol.ipynb) — cross-validation and evaluation harnesses
|
|
137
|
+
|
|
138
|
+
## 📄 License
|
|
139
|
+
|
|
140
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
141
|
+
</content>
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "chemsplit"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "A self-contained, scikit-learn-compatible library of dataset-splitting strategies for cheminformatics machine learning."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11,<3.14"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Olivier J. M. Béquignon", email = "olivier.bequignon.maintainer@gmail.com" },
|
|
15
|
+
]
|
|
16
|
+
keywords = [
|
|
17
|
+
"cheminformatics",
|
|
18
|
+
"machine-learning",
|
|
19
|
+
"dataset-splitting",
|
|
20
|
+
"scikit-learn",
|
|
21
|
+
"rdkit",
|
|
22
|
+
]
|
|
23
|
+
classifiers = [
|
|
24
|
+
"Development Status :: 3 - Alpha",
|
|
25
|
+
"Intended Audience :: Science/Research",
|
|
26
|
+
"Programming Language :: Python :: 3",
|
|
27
|
+
"Programming Language :: Python :: 3.11",
|
|
28
|
+
"Programming Language :: Python :: 3.12",
|
|
29
|
+
"Programming Language :: Python :: 3.13",
|
|
30
|
+
"Topic :: Scientific/Engineering :: Chemistry",
|
|
31
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
32
|
+
"Typing :: Typed",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
dependencies = [
|
|
36
|
+
"numpy>=1.24,<3",
|
|
37
|
+
"scipy>=1.10",
|
|
38
|
+
"scikit-learn>=1.3",
|
|
39
|
+
"rdkit>=2023.09.1",
|
|
40
|
+
"pandas>=2.0",
|
|
41
|
+
"scaffound>=0.0.1",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
[project.optional-dependencies]
|
|
45
|
+
umap = ["umap-learn>=0.5.4"]
|
|
46
|
+
hdbscan = ["scikit-learn>=1.3"]
|
|
47
|
+
ga = ["deap>=1.4"]
|
|
48
|
+
bio = ["biopython>=1.81", "parasail>=1.3; sys_platform != 'darwin' or platform_machine != 'arm64'"]
|
|
49
|
+
mmpa = []
|
|
50
|
+
all = [
|
|
51
|
+
"chemsplit[umap,hdbscan,ga,bio,mmpa]",
|
|
52
|
+
]
|
|
53
|
+
dev = [
|
|
54
|
+
"pytest>=7.4",
|
|
55
|
+
"pytest-cov",
|
|
56
|
+
"hypothesis>=6.90",
|
|
57
|
+
"ruff>=0.16",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
[project.urls]
|
|
61
|
+
Homepage = "https://github.com/OlivierBeq/chemsplit"
|
|
62
|
+
|
|
63
|
+
[project.scripts]
|
|
64
|
+
chemsplit = "chemsplit.cli:main"
|
|
65
|
+
|
|
66
|
+
[tool.setuptools.dynamic]
|
|
67
|
+
version = { attr = "chemsplit.__version__" }
|
|
68
|
+
|
|
69
|
+
[tool.setuptools.packages.find]
|
|
70
|
+
where = ["src"]
|
|
71
|
+
include = ["chemsplit*"]
|
|
72
|
+
|
|
73
|
+
[tool.pytest.ini_options]
|
|
74
|
+
minversion = "7.4"
|
|
75
|
+
markers = [
|
|
76
|
+
"core: fast, no-network tests (< 60s total)",
|
|
77
|
+
"slow: performance / large-n tests",
|
|
78
|
+
"network: tests that download real datasets",
|
|
79
|
+
"extra_umap: requires the umap extra",
|
|
80
|
+
"extra_ga: requires the ga extra",
|
|
81
|
+
"extra_bio: requires the bio extra",
|
|
82
|
+
"golden: byte-exact / tolerance golden-file regression tests",
|
|
83
|
+
]
|
|
84
|
+
testpaths = ["tests"]
|
|
85
|
+
|
|
86
|
+
[tool.coverage.run]
|
|
87
|
+
source = ["chemsplit"]
|
|
88
|
+
omit = ["chemsplit/cli.py"]
|
|
89
|
+
|
|
90
|
+
[tool.coverage.report]
|
|
91
|
+
fail_under = 90
|
|
92
|
+
|
|
93
|
+
[tool.ruff]
|
|
94
|
+
line-length = 100
|
|
95
|
+
target-version = "py311"
|
|
96
|
+
|
|
97
|
+
[tool.ruff.lint]
|
|
98
|
+
select = ["E", "F", "I", "UP", "B", "ANN"]
|
|
99
|
+
ignore = [
|
|
100
|
+
"ANN401", # `Any` is legitimate for genuinely dynamic values (e.g. **kwargs passthrough)
|
|
101
|
+
"E501", # line length unenforced: chemistry-domain identifiers and docstring prose read
|
|
102
|
+
# better unwrapped; no autoformatter is configured to rewrap consistently
|
|
103
|
+
"B023", # argmax_tiebreak/argmin_tiebreak key-functions capture the enclosing loop
|
|
104
|
+
# variable but are always invoked synchronously within the same iteration, so
|
|
105
|
+
# the late-binding closure hazard this rule warns about cannot occur here
|
|
106
|
+
]
|
|
107
|
+
|
|
108
|
+
[tool.ruff.lint.per-file-ignores]
|
|
109
|
+
"tests/*" = [
|
|
110
|
+
"ANN",
|
|
111
|
+
"B017", # `pytest.raises(Exception)` is a deliberate, accepted "this call must fail" pattern
|
|
112
|
+
]
|
|
113
|
+
"tools/*" = ["ANN"]
|
|
114
|
+
"notebooks/*" = [
|
|
115
|
+
"ANN",
|
|
116
|
+
"E402", # imports interleaved with narrative markdown/setup cells is normal notebook structure
|
|
117
|
+
]
|