chemsplit 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. chemsplit-0.1.0/LICENSE +21 -0
  2. chemsplit-0.1.0/PKG-INFO +186 -0
  3. chemsplit-0.1.0/README.md +141 -0
  4. chemsplit-0.1.0/pyproject.toml +117 -0
  5. chemsplit-0.1.0/setup.cfg +4 -0
  6. chemsplit-0.1.0/src/chemsplit/__init__.py +250 -0
  7. chemsplit-0.1.0/src/chemsplit/__main__.py +8 -0
  8. chemsplit-0.1.0/src/chemsplit/_devtools.py +420 -0
  9. chemsplit-0.1.0/src/chemsplit/_fp_similarity.py +154 -0
  10. chemsplit-0.1.0/src/chemsplit/_optimize.py +781 -0
  11. chemsplit-0.1.0/src/chemsplit/_pair_assign.py +143 -0
  12. chemsplit-0.1.0/src/chemsplit/_unionfind.py +94 -0
  13. chemsplit-0.1.0/src/chemsplit/audit.py +577 -0
  14. chemsplit-0.1.0/src/chemsplit/base.py +977 -0
  15. chemsplit-0.1.0/src/chemsplit/cli.py +135 -0
  16. chemsplit-0.1.0/src/chemsplit/clustering.py +288 -0
  17. chemsplit-0.1.0/src/chemsplit/datasets.py +656 -0
  18. chemsplit-0.1.0/src/chemsplit/determinism.py +160 -0
  19. chemsplit-0.1.0/src/chemsplit/exceptions.py +252 -0
  20. chemsplit-0.1.0/src/chemsplit/featurizers/__init__.py +101 -0
  21. chemsplit-0.1.0/src/chemsplit/featurizers/descriptors.py +78 -0
  22. chemsplit-0.1.0/src/chemsplit/featurizers/fingerprints.py +205 -0
  23. chemsplit-0.1.0/src/chemsplit/featurizers/precomputed.py +28 -0
  24. chemsplit-0.1.0/src/chemsplit/metrics.py +284 -0
  25. chemsplit-0.1.0/src/chemsplit/preprocess.py +418 -0
  26. chemsplit-0.1.0/src/chemsplit/registry.py +306 -0
  27. chemsplit-0.1.0/src/chemsplit/scaffolds.py +331 -0
  28. chemsplit-0.1.0/src/chemsplit/splitters/__init__.py +4 -0
  29. chemsplit-0.1.0/src/chemsplit/splitters/baseline.py +816 -0
  30. chemsplit-0.1.0/src/chemsplit/splitters/biomolecular.py +579 -0
  31. chemsplit-0.1.0/src/chemsplit/splitters/embedding.py +645 -0
  32. chemsplit-0.1.0/src/chemsplit/splitters/lineage.py +967 -0
  33. chemsplit-0.1.0/src/chemsplit/splitters/property_.py +983 -0
  34. chemsplit-0.1.0/src/chemsplit/splitters/protocol.py +772 -0
  35. chemsplit-0.1.0/src/chemsplit/splitters/scaffold.py +1220 -0
  36. chemsplit-0.1.0/src/chemsplit/splitters/similarity.py +1613 -0
  37. chemsplit-0.1.0/src/chemsplit/splitters/task.py +1692 -0
  38. chemsplit-0.1.0/src/chemsplit/types.py +41 -0
  39. chemsplit-0.1.0/src/chemsplit.egg-info/PKG-INFO +186 -0
  40. chemsplit-0.1.0/src/chemsplit.egg-info/SOURCES.txt +64 -0
  41. chemsplit-0.1.0/src/chemsplit.egg-info/dependency_links.txt +1 -0
  42. chemsplit-0.1.0/src/chemsplit.egg-info/entry_points.txt +2 -0
  43. chemsplit-0.1.0/src/chemsplit.egg-info/requires.txt +32 -0
  44. chemsplit-0.1.0/src/chemsplit.egg-info/top_level.txt +1 -0
  45. chemsplit-0.1.0/tests/test_api_contract.py +405 -0
  46. chemsplit-0.1.0/tests/test_audit.py +185 -0
  47. chemsplit-0.1.0/tests/test_base.py +710 -0
  48. chemsplit-0.1.0/tests/test_cli.py +124 -0
  49. chemsplit-0.1.0/tests/test_clustering.py +226 -0
  50. chemsplit-0.1.0/tests/test_datasets.py +172 -0
  51. chemsplit-0.1.0/tests/test_determinism.py +87 -0
  52. chemsplit-0.1.0/tests/test_determinism_lint.py +35 -0
  53. chemsplit-0.1.0/tests/test_devtools.py +157 -0
  54. chemsplit-0.1.0/tests/test_exceptions.py +60 -0
  55. chemsplit-0.1.0/tests/test_featurizers.py +213 -0
  56. chemsplit-0.1.0/tests/test_fp_similarity.py +115 -0
  57. chemsplit-0.1.0/tests/test_golden.py +86 -0
  58. chemsplit-0.1.0/tests/test_metrics.py +254 -0
  59. chemsplit-0.1.0/tests/test_optimize.py +189 -0
  60. chemsplit-0.1.0/tests/test_package_import.py +71 -0
  61. chemsplit-0.1.0/tests/test_pair_assign.py +111 -0
  62. chemsplit-0.1.0/tests/test_preprocess.py +311 -0
  63. chemsplit-0.1.0/tests/test_property_based.py +439 -0
  64. chemsplit-0.1.0/tests/test_registry.py +100 -0
  65. chemsplit-0.1.0/tests/test_scaffolds.py +159 -0
  66. chemsplit-0.1.0/tests/test_unionfind.py +25 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Olivier J. M. Béquignon
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,186 @@
1
+ Metadata-Version: 2.4
2
+ Name: chemsplit
3
+ Version: 0.1.0
4
+ Summary: A self-contained, scikit-learn-compatible library of dataset-splitting strategies for cheminformatics machine learning.
5
+ Author-email: "Olivier J. M. Béquignon" <olivier.bequignon.maintainer@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/OlivierBeq/chemsplit
8
+ Keywords: cheminformatics,machine-learning,dataset-splitting,scikit-learn,rdkit
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Topic :: Scientific/Engineering :: Chemistry
16
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
17
+ Classifier: Typing :: Typed
18
+ Requires-Python: <3.14,>=3.11
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: numpy<3,>=1.24
22
+ Requires-Dist: scipy>=1.10
23
+ Requires-Dist: scikit-learn>=1.3
24
+ Requires-Dist: rdkit>=2023.09.1
25
+ Requires-Dist: pandas>=2.0
26
+ Requires-Dist: scaffound>=0.0.1
27
+ Provides-Extra: umap
28
+ Requires-Dist: umap-learn>=0.5.4; extra == "umap"
29
+ Provides-Extra: hdbscan
30
+ Requires-Dist: scikit-learn>=1.3; extra == "hdbscan"
31
+ Provides-Extra: ga
32
+ Requires-Dist: deap>=1.4; extra == "ga"
33
+ Provides-Extra: bio
34
+ Requires-Dist: biopython>=1.81; extra == "bio"
35
+ Requires-Dist: parasail>=1.3; (sys_platform != "darwin" or platform_machine != "arm64") and extra == "bio"
36
+ Provides-Extra: mmpa
37
+ Provides-Extra: all
38
+ Requires-Dist: chemsplit[bio,ga,hdbscan,mmpa,umap]; extra == "all"
39
+ Provides-Extra: dev
40
+ Requires-Dist: pytest>=7.4; extra == "dev"
41
+ Requires-Dist: pytest-cov; extra == "dev"
42
+ Requires-Dist: hypothesis>=6.90; extra == "dev"
43
+ Requires-Dist: ruff>=0.16; extra == "dev"
44
+ Dynamic: license-file
45
+
46
+ <div align="center">
47
+
48
+ # ✂️ chemsplit
49
+
50
+ [![PyPI version](https://img.shields.io/pypi/v/chemsplit.svg)](https://pypi.org/project/chemsplit/)
51
+ [![Supported Python versions](https://img.shields.io/pypi/pyversions/chemsplit.svg)](https://pypi.org/project/chemsplit/)
52
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
53
+ [![Tests](https://github.com/OlivierBeq/chemsplit/actions/workflows/ci.yml/badge.svg)](https://github.com/OlivierBeq/chemsplit/actions/workflows/ci.yml)
54
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
55
+
56
+ </div>
57
+
58
+ A self-contained, scikit-learn-compatible Python library of dataset-splitting strategies for cheminformatics machine learning. `chemsplit` implements **52 splitting strategies across nine families** — baseline, scaffold, similarity, embedding, property, lineage, task, biomolecular, and protocol splitters — behind one coherent, deterministic API, plus a leakage-audit module and a set of reference/synthetic datasets to try them on.
59
+
60
+ ## ✨ Features
61
+
62
+ - 🧩 **9 families, 52 strategies** — from a plain random split to scaffold-tree pruning, Butina/spectral clustering, UMAP-space holdouts, temporal and provenance cuts, protein-family and binding-site holdouts, drug-target cold-start benchmarks, and full CV/nested-CV protocol wrappers.
63
+ - 🎯 **Deterministic by construction** — every splitter accepts a `random_state` and produces bit-identical output regardless of record order or `n_jobs`, checked continuously by a golden-file regression suite and Hypothesis property tests.
64
+ - 🛡️ **Contract-checked results** — every `SplitResult` is validated against five structural invariants (index coverage, disjointness, group-label consistency, JSON round-tripping of `params`, id format) before it ever reaches your code.
65
+ - 🔍 **Built-in leakage auditing** — `chemsplit.audit` reports nearest-neighbour similarity, adversarial-validation AUC, exact/scaffold/ring-system overlap, and property/label shift between train and test.
66
+ - 🧪 **Chemistry-native featurization** — ECFP/FCFP/MACCS/Avalon/atom-pair/topological-torsion fingerprints and physicochemical descriptors, behind a pluggable `Featurizer` protocol for your own.
67
+ - 💻 **CLI included** — run, audit, and list any registered splitter without writing a line of Python.
68
+ - 📚 **One example notebook per family** — runnable, narrated walkthroughs of every splitter class under [`notebooks/`](notebooks/).
69
+
70
+ ## 📦 Installation
71
+
72
+ ```bash
73
+ pip install chemsplit
74
+ ```
75
+
76
+ Optional extras enable additional splitters and featurizers:
77
+
78
+ | Extra | Enables |
79
+ |---|---|
80
+ | `chemsplit[umap]` | `UMAPClusterSplitter` (UMAP embedding + clustering) |
81
+ | `chemsplit[hdbscan]` | `DensityClusterSplitter` (HDBSCAN density clustering) |
82
+ | `chemsplit[ga]` | `SIMPDSplitter` (genetic-algorithm pseudo-time optimization, via `deap`) |
83
+ | `chemsplit[bio]` | Protein-sequence splitters with accelerated alignment (`biopython`, `parasail`) |
84
+ | `chemsplit[mmpa]` | `MatchedMolecularSeriesSplitter` matched-series extraction |
85
+ | `chemsplit[all]` | Everything above |
86
+
87
+ > **Note:** every extra has a dependency-free fallback where one makes sense (e.g. a Hamming-distance fallback for sequence identity without `bio`) — an extra buys you a better implementation, not a hard requirement.
88
+
89
+ ## 🛠️ Requirements
90
+
91
+ - Python 3.11 – 3.13
92
+ - [RDKit](https://www.rdkit.org/) (installed automatically as a core dependency — no separate conda step needed)
93
+
94
+ ## 💡 Usage
95
+
96
+ ### Quickstart
97
+
98
+ ```python
99
+ from chemsplit import datasets, get_splitter
100
+
101
+ fx = datasets.make_scaffold_families(n_scaffolds=10, per_scaffold=15)
102
+
103
+ splitter = get_splitter("scaffold_tree", train_size=0.8, test_size=0.2)
104
+ result = splitter.split_result(fx.smiles)[0]
105
+
106
+ train_smiles = [fx.smiles[i] for i in result.train]
107
+ test_smiles = [fx.smiles[i] for i in result.test]
108
+ print(f"{len(train_smiles)} train / {len(test_smiles)} test records")
109
+ ```
110
+
111
+ Every splitter is resolved the same way, by its `splitter_id` (see `chemsplit.list_splitters()` for the full table) or by importing the class directly:
112
+
113
+ ```python
114
+ from chemsplit import ButinaSplitter
115
+
116
+ splitter = ButinaSplitter(cutoff=0.4, train_size=0.7, test_size=0.3, random_state=0)
117
+ ```
118
+
119
+ <details>
120
+ <summary><strong>🔍 Auditing a split for leakage</strong></summary>
121
+
122
+ ```python
123
+ from chemsplit import audit_split
124
+
125
+ report = audit_split(result, fx.smiles)
126
+ print(report.summary())
127
+ ```
128
+ ```text
129
+ chemsplit LeakageReport
130
+ ------------------------
131
+ n_train=120 n_valid=0 n_test=30 n_discard=0
132
+ max cross-partition similarity: 0.4375
133
+ median NN similarity (test->train): 0.6667
134
+ exact duplicates across partitions: 0
135
+ shared scaffolds: 0 shared ring systems: 0
136
+ adversarial AUC: 1.0000 (95% CI 1.0000-1.0000)
137
+ flags: MEDIAN_NN_ABOVE_0.6, HIGH_ADVERSARIAL_AUC, ...
138
+ ```
139
+
140
+ `LeakageReport` is purely descriptive — it never fails your pipeline, it tells you what to look at.
141
+ </details>
142
+
143
+ <details>
144
+ <summary><strong>💻 The CLI</strong></summary>
145
+
146
+ ```bash
147
+ # List every registered splitter, optionally filtered by family
148
+ chemsplit list --family scaffold
149
+
150
+ # Run a splitter over a CSV and write the split to disk
151
+ chemsplit split --splitter scaffold_tree --input data.csv --smiles-col smiles \
152
+ --train-size 0.8 --test-size 0.2 --seed 0 --out split.json
153
+
154
+ # Audit that split for leakage
155
+ chemsplit audit --split split.json --input data.csv --smiles-col smiles --out report.json
156
+
157
+ chemsplit --help
158
+ ```
159
+ </details>
160
+
161
+ ## 🧭 Design principles
162
+
163
+ 1. **Determinism.** Same inputs + same `random_state` ⇒ byte-identical outputs, on any platform, any CPU count, any `n_jobs`.
164
+ 2. **Explicitness.** No silent fallbacks. If a requested configuration is infeasible, raise, never approximate.
165
+ 3. **Honesty.** Every splitter's docstring discloses its pitfalls with the same prominence as its advantages.
166
+ 4. **Composability.** Every group-forming splitter exposes its group labels, so any grouping can be fed to any protocol wrapper.
167
+ 5. **scikit-learn compatibility.** `.split()` is drop-in usable in `cross_val_score`, `GridSearchCV(cv=...)`, and `cross_validate`.
168
+
169
+ ## 📚 Learn more
170
+
171
+ One narrated, runnable notebook per family, under [`notebooks/`](notebooks/):
172
+
173
+ - [`baseline.ipynb`](notebooks/baseline.ipynb) — uninformed and gold-standard controls
174
+ - [`scaffold.ipynb`](notebooks/scaffold.ipynb) — chemotype-aware generalization tests
175
+ - [`similarity.ipynb`](notebooks/similarity.ipynb) — fingerprint-clustering holdouts
176
+ - [`embedding.ipynb`](notebooks/embedding.ipynb) — latent-space holdouts
177
+ - [`property.ipynb`](notebooks/property.ipynb) — label-shift and distributional stress tests
178
+ - [`lineage.ipynb`](notebooks/lineage.ipynb) — date and provenance-aware holdouts
179
+ - [`biomolecular.ipynb`](notebooks/biomolecular.ipynb) — protein-axis holdouts
180
+ - [`task.ipynb`](notebooks/task.ipynb) — drug-target interaction benchmarks
181
+ - [`protocol.ipynb`](notebooks/protocol.ipynb) — cross-validation and evaluation harnesses
182
+
183
+ ## 📄 License
184
+
185
+ This project is licensed under the [MIT License](LICENSE).
186
+ </content>
@@ -0,0 +1,141 @@
1
+ <div align="center">
2
+
3
+ # ✂️ chemsplit
4
+
5
+ [![PyPI version](https://img.shields.io/pypi/v/chemsplit.svg)](https://pypi.org/project/chemsplit/)
6
+ [![Supported Python versions](https://img.shields.io/pypi/pyversions/chemsplit.svg)](https://pypi.org/project/chemsplit/)
7
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
8
+ [![Tests](https://github.com/OlivierBeq/chemsplit/actions/workflows/ci.yml/badge.svg)](https://github.com/OlivierBeq/chemsplit/actions/workflows/ci.yml)
9
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
10
+
11
+ </div>
12
+
13
+ A self-contained, scikit-learn-compatible Python library of dataset-splitting strategies for cheminformatics machine learning. `chemsplit` implements **52 splitting strategies across nine families** — baseline, scaffold, similarity, embedding, property, lineage, task, biomolecular, and protocol splitters — behind one coherent, deterministic API, plus a leakage-audit module and a set of reference/synthetic datasets to try them on.
14
+
15
+ ## ✨ Features
16
+
17
+ - 🧩 **9 families, 52 strategies** — from a plain random split to scaffold-tree pruning, Butina/spectral clustering, UMAP-space holdouts, temporal and provenance cuts, protein-family and binding-site holdouts, drug-target cold-start benchmarks, and full CV/nested-CV protocol wrappers.
18
+ - 🎯 **Deterministic by construction** — every splitter accepts a `random_state` and produces bit-identical output regardless of record order or `n_jobs`, checked continuously by a golden-file regression suite and Hypothesis property tests.
19
+ - 🛡️ **Contract-checked results** — every `SplitResult` is validated against five structural invariants (index coverage, disjointness, group-label consistency, JSON round-tripping of `params`, id format) before it ever reaches your code.
20
+ - 🔍 **Built-in leakage auditing** — `chemsplit.audit` reports nearest-neighbour similarity, adversarial-validation AUC, exact/scaffold/ring-system overlap, and property/label shift between train and test.
21
+ - 🧪 **Chemistry-native featurization** — ECFP/FCFP/MACCS/Avalon/atom-pair/topological-torsion fingerprints and physicochemical descriptors, behind a pluggable `Featurizer` protocol for your own.
22
+ - 💻 **CLI included** — run, audit, and list any registered splitter without writing a line of Python.
23
+ - 📚 **One example notebook per family** — runnable, narrated walkthroughs of every splitter class under [`notebooks/`](notebooks/).
24
+
25
+ ## 📦 Installation
26
+
27
+ ```bash
28
+ pip install chemsplit
29
+ ```
30
+
31
+ Optional extras enable additional splitters and featurizers:
32
+
33
+ | Extra | Enables |
34
+ |---|---|
35
+ | `chemsplit[umap]` | `UMAPClusterSplitter` (UMAP embedding + clustering) |
36
+ | `chemsplit[hdbscan]` | `DensityClusterSplitter` (HDBSCAN density clustering) |
37
+ | `chemsplit[ga]` | `SIMPDSplitter` (genetic-algorithm pseudo-time optimization, via `deap`) |
38
+ | `chemsplit[bio]` | Protein-sequence splitters with accelerated alignment (`biopython`, `parasail`) |
39
+ | `chemsplit[mmpa]` | `MatchedMolecularSeriesSplitter` matched-series extraction |
40
+ | `chemsplit[all]` | Everything above |
41
+
42
+ > **Note:** every extra has a dependency-free fallback where one makes sense (e.g. a Hamming-distance fallback for sequence identity without `bio`) — an extra buys you a better implementation, not a hard requirement.
43
+
44
+ ## 🛠️ Requirements
45
+
46
+ - Python 3.11 – 3.13
47
+ - [RDKit](https://www.rdkit.org/) (installed automatically as a core dependency — no separate conda step needed)
48
+
49
+ ## 💡 Usage
50
+
51
+ ### Quickstart
52
+
53
+ ```python
54
+ from chemsplit import datasets, get_splitter
55
+
56
+ fx = datasets.make_scaffold_families(n_scaffolds=10, per_scaffold=15)
57
+
58
+ splitter = get_splitter("scaffold_tree", train_size=0.8, test_size=0.2)
59
+ result = splitter.split_result(fx.smiles)[0]
60
+
61
+ train_smiles = [fx.smiles[i] for i in result.train]
62
+ test_smiles = [fx.smiles[i] for i in result.test]
63
+ print(f"{len(train_smiles)} train / {len(test_smiles)} test records")
64
+ ```
65
+
66
+ Every splitter is resolved the same way, by its `splitter_id` (see `chemsplit.list_splitters()` for the full table) or by importing the class directly:
67
+
68
+ ```python
69
+ from chemsplit import ButinaSplitter
70
+
71
+ splitter = ButinaSplitter(cutoff=0.4, train_size=0.7, test_size=0.3, random_state=0)
72
+ ```
73
+
74
+ <details>
75
+ <summary><strong>🔍 Auditing a split for leakage</strong></summary>
76
+
77
+ ```python
78
+ from chemsplit import audit_split
79
+
80
+ report = audit_split(result, fx.smiles)
81
+ print(report.summary())
82
+ ```
83
+ ```text
84
+ chemsplit LeakageReport
85
+ ------------------------
86
+ n_train=120 n_valid=0 n_test=30 n_discard=0
87
+ max cross-partition similarity: 0.4375
88
+ median NN similarity (test->train): 0.6667
89
+ exact duplicates across partitions: 0
90
+ shared scaffolds: 0 shared ring systems: 0
91
+ adversarial AUC: 1.0000 (95% CI 1.0000-1.0000)
92
+ flags: MEDIAN_NN_ABOVE_0.6, HIGH_ADVERSARIAL_AUC, ...
93
+ ```
94
+
95
+ `LeakageReport` is purely descriptive — it never fails your pipeline, it tells you what to look at.
96
+ </details>
97
+
98
+ <details>
99
+ <summary><strong>💻 The CLI</strong></summary>
100
+
101
+ ```bash
102
+ # List every registered splitter, optionally filtered by family
103
+ chemsplit list --family scaffold
104
+
105
+ # Run a splitter over a CSV and write the split to disk
106
+ chemsplit split --splitter scaffold_tree --input data.csv --smiles-col smiles \
107
+ --train-size 0.8 --test-size 0.2 --seed 0 --out split.json
108
+
109
+ # Audit that split for leakage
110
+ chemsplit audit --split split.json --input data.csv --smiles-col smiles --out report.json
111
+
112
+ chemsplit --help
113
+ ```
114
+ </details>
115
+
116
+ ## 🧭 Design principles
117
+
118
+ 1. **Determinism.** Same inputs + same `random_state` ⇒ byte-identical outputs, on any platform, any CPU count, any `n_jobs`.
119
+ 2. **Explicitness.** No silent fallbacks. If a requested configuration is infeasible, raise, never approximate.
120
+ 3. **Honesty.** Every splitter's docstring discloses its pitfalls with the same prominence as its advantages.
121
+ 4. **Composability.** Every group-forming splitter exposes its group labels, so any grouping can be fed to any protocol wrapper.
122
+ 5. **scikit-learn compatibility.** `.split()` is drop-in usable in `cross_val_score`, `GridSearchCV(cv=...)`, and `cross_validate`.
123
+
124
+ ## 📚 Learn more
125
+
126
+ One narrated, runnable notebook per family, under [`notebooks/`](notebooks/):
127
+
128
+ - [`baseline.ipynb`](notebooks/baseline.ipynb) — uninformed and gold-standard controls
129
+ - [`scaffold.ipynb`](notebooks/scaffold.ipynb) — chemotype-aware generalization tests
130
+ - [`similarity.ipynb`](notebooks/similarity.ipynb) — fingerprint-clustering holdouts
131
+ - [`embedding.ipynb`](notebooks/embedding.ipynb) — latent-space holdouts
132
+ - [`property.ipynb`](notebooks/property.ipynb) — label-shift and distributional stress tests
133
+ - [`lineage.ipynb`](notebooks/lineage.ipynb) — date and provenance-aware holdouts
134
+ - [`biomolecular.ipynb`](notebooks/biomolecular.ipynb) — protein-axis holdouts
135
+ - [`task.ipynb`](notebooks/task.ipynb) — drug-target interaction benchmarks
136
+ - [`protocol.ipynb`](notebooks/protocol.ipynb) — cross-validation and evaluation harnesses
137
+
138
+ ## 📄 License
139
+
140
+ This project is licensed under the [MIT License](LICENSE).
141
+ </content>
@@ -0,0 +1,117 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "chemsplit"
7
+ dynamic = ["version"]
8
+ description = "A self-contained, scikit-learn-compatible library of dataset-splitting strategies for cheminformatics machine learning."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11,<3.14"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [
14
+ { name = "Olivier J. M. Béquignon", email = "olivier.bequignon.maintainer@gmail.com" },
15
+ ]
16
+ keywords = [
17
+ "cheminformatics",
18
+ "machine-learning",
19
+ "dataset-splitting",
20
+ "scikit-learn",
21
+ "rdkit",
22
+ ]
23
+ classifiers = [
24
+ "Development Status :: 3 - Alpha",
25
+ "Intended Audience :: Science/Research",
26
+ "Programming Language :: Python :: 3",
27
+ "Programming Language :: Python :: 3.11",
28
+ "Programming Language :: Python :: 3.12",
29
+ "Programming Language :: Python :: 3.13",
30
+ "Topic :: Scientific/Engineering :: Chemistry",
31
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
32
+ "Typing :: Typed",
33
+ ]
34
+
35
+ dependencies = [
36
+ "numpy>=1.24,<3",
37
+ "scipy>=1.10",
38
+ "scikit-learn>=1.3",
39
+ "rdkit>=2023.09.1",
40
+ "pandas>=2.0",
41
+ "scaffound>=0.0.1",
42
+ ]
43
+
44
+ [project.optional-dependencies]
45
+ umap = ["umap-learn>=0.5.4"]
46
+ hdbscan = ["scikit-learn>=1.3"]
47
+ ga = ["deap>=1.4"]
48
+ bio = ["biopython>=1.81", "parasail>=1.3; sys_platform != 'darwin' or platform_machine != 'arm64'"]
49
+ mmpa = []
50
+ all = [
51
+ "chemsplit[umap,hdbscan,ga,bio,mmpa]",
52
+ ]
53
+ dev = [
54
+ "pytest>=7.4",
55
+ "pytest-cov",
56
+ "hypothesis>=6.90",
57
+ "ruff>=0.16",
58
+ ]
59
+
60
+ [project.urls]
61
+ Homepage = "https://github.com/OlivierBeq/chemsplit"
62
+
63
+ [project.scripts]
64
+ chemsplit = "chemsplit.cli:main"
65
+
66
+ [tool.setuptools.dynamic]
67
+ version = { attr = "chemsplit.__version__" }
68
+
69
+ [tool.setuptools.packages.find]
70
+ where = ["src"]
71
+ include = ["chemsplit*"]
72
+
73
+ [tool.pytest.ini_options]
74
+ minversion = "7.4"
75
+ markers = [
76
+ "core: fast, no-network tests (< 60s total)",
77
+ "slow: performance / large-n tests",
78
+ "network: tests that download real datasets",
79
+ "extra_umap: requires the umap extra",
80
+ "extra_ga: requires the ga extra",
81
+ "extra_bio: requires the bio extra",
82
+ "golden: byte-exact / tolerance golden-file regression tests",
83
+ ]
84
+ testpaths = ["tests"]
85
+
86
+ [tool.coverage.run]
87
+ source = ["chemsplit"]
88
+ omit = ["chemsplit/cli.py"]
89
+
90
+ [tool.coverage.report]
91
+ fail_under = 90
92
+
93
+ [tool.ruff]
94
+ line-length = 100
95
+ target-version = "py311"
96
+
97
+ [tool.ruff.lint]
98
+ select = ["E", "F", "I", "UP", "B", "ANN"]
99
+ ignore = [
100
+ "ANN401", # `Any` is legitimate for genuinely dynamic values (e.g. **kwargs passthrough)
101
+ "E501", # line length unenforced: chemistry-domain identifiers and docstring prose read
102
+ # better unwrapped; no autoformatter is configured to rewrap consistently
103
+ "B023", # argmax_tiebreak/argmin_tiebreak key-functions capture the enclosing loop
104
+ # variable but are always invoked synchronously within the same iteration, so
105
+ # the late-binding closure hazard this rule warns about cannot occur here
106
+ ]
107
+
108
+ [tool.ruff.lint.per-file-ignores]
109
+ "tests/*" = [
110
+ "ANN",
111
+ "B017", # `pytest.raises(Exception)` is a deliberate, accepted "this call must fail" pattern
112
+ ]
113
+ "tools/*" = ["ANN"]
114
+ "notebooks/*" = [
115
+ "ANN",
116
+ "E402", # imports interleaved with narrative markdown/setup cells is normal notebook structure
117
+ ]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+