molito 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- molito-0.1.0/LICENSE +21 -0
- molito-0.1.0/PKG-INFO +258 -0
- molito-0.1.0/README.md +209 -0
- molito-0.1.0/molito/__init__.py +33 -0
- molito-0.1.0/molito/arrays.py +72 -0
- molito-0.1.0/molito/convert.py +211 -0
- molito-0.1.0/molito/core/__init__.py +14 -0
- molito-0.1.0/molito/core/_checks.py +142 -0
- molito-0.1.0/molito/core/atoms.py +643 -0
- molito-0.1.0/molito/core/bonds.py +580 -0
- molito-0.1.0/molito/core/confs.py +415 -0
- molito-0.1.0/molito/core/format.py +79 -0
- molito-0.1.0/molito/core/lazydata.py +82 -0
- molito-0.1.0/molito/core/meta.py +357 -0
- molito-0.1.0/molito/core/pharmacophore.py +223 -0
- molito-0.1.0/molito/core/presets.py +96 -0
- molito-0.1.0/molito/core/pt.py +35 -0
- molito-0.1.0/molito/core/vocab.py +279 -0
- molito-0.1.0/molito/defs/pharmacophore.fdef +204 -0
- molito-0.1.0/molito/geometry/__init__.py +10 -0
- molito-0.1.0/molito/geometry/align.py +211 -0
- molito-0.1.0/molito/geometry/common.py +233 -0
- molito-0.1.0/molito/geometry/mmff.py +126 -0
- molito-0.1.0/molito/geometry/xtb.py +180 -0
- molito-0.1.0/molito/mol/__init__.py +10 -0
- molito-0.1.0/molito/mol/complex.py +514 -0
- molito-0.1.0/molito/mol/graph.py +1075 -0
- molito-0.1.0/molito/mol/interactions.py +463 -0
- molito-0.1.0/molito/mol/protein.py +648 -0
- molito-0.1.0/molito/py.typed +0 -0
- molito-0.1.0/molito.egg-info/PKG-INFO +258 -0
- molito-0.1.0/molito.egg-info/SOURCES.txt +35 -0
- molito-0.1.0/molito.egg-info/dependency_links.txt +1 -0
- molito-0.1.0/molito.egg-info/requires.txt +25 -0
- molito-0.1.0/molito.egg-info/top_level.txt +1 -0
- molito-0.1.0/pyproject.toml +109 -0
- molito-0.1.0/setup.cfg +4 -0
molito-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ross Irwin
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
molito-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: molito
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Molecular representation and processing toolkit for storing and processing small molecules into a training-ready format
|
|
5
|
+
Author-email: Ross Irwin <rssrwn@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/rssrwn/molito
|
|
8
|
+
Project-URL: Documentation, https://rssrwn.github.io/molito
|
|
9
|
+
Project-URL: Repository, https://github.com/rssrwn/molito
|
|
10
|
+
Project-URL: Issues, https://github.com/rssrwn/molito/issues
|
|
11
|
+
Keywords: cheminformatics,molecules,machine-learning,rdkit,hdf5,conformers,drug-discovery
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Chemistry
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.11
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: numpy>=2.0
|
|
27
|
+
Requires-Dist: scipy>=1.12
|
|
28
|
+
Requires-Dist: rdkit>=2024.3
|
|
29
|
+
Requires-Dist: h5py>=3.10
|
|
30
|
+
Requires-Dist: biotite>=1.0
|
|
31
|
+
Requires-Dist: more-itertools>=10.0
|
|
32
|
+
Provides-Extra: interactions
|
|
33
|
+
Requires-Dist: prolif>=2.0; extra == "interactions"
|
|
34
|
+
Requires-Dist: MDAnalysis; extra == "interactions"
|
|
35
|
+
Provides-Extra: dev
|
|
36
|
+
Requires-Dist: matplotlib; extra == "dev"
|
|
37
|
+
Requires-Dist: jupyter; extra == "dev"
|
|
38
|
+
Requires-Dist: ipykernel; extra == "dev"
|
|
39
|
+
Requires-Dist: py3Dmol; extra == "dev"
|
|
40
|
+
Requires-Dist: ruff==0.16.0; extra == "dev"
|
|
41
|
+
Requires-Dist: mypy==2.3.0; extra == "dev"
|
|
42
|
+
Requires-Dist: build; extra == "dev"
|
|
43
|
+
Requires-Dist: twine; extra == "dev"
|
|
44
|
+
Provides-Extra: docs
|
|
45
|
+
Requires-Dist: mkdocs-material>=9.5; extra == "docs"
|
|
46
|
+
Requires-Dist: mkdocstrings[python]>=0.24; extra == "docs"
|
|
47
|
+
Requires-Dist: ruff==0.16.0; extra == "docs"
|
|
48
|
+
Dynamic: license-file
|
|
49
|
+
|
|
50
|
+
# *Molito* - Small Molecule Data Processing Utility
|
|
51
|
+
|
|
52
|
+
A Python toolkit for processing and storing small molecule and protein data into a training-ready format.
|
|
53
|
+
|
|
54
|
+
Molito provides compact, serialisable representations for molecular graphs with atoms, bonds, and 3D conformer ensembles. It's designed for machine learning workflows where you need to go from RDKit molecules to structured arrays efficiently, with full support for stereochemistry (E/Z bond directions, tetrahedral chirality) preserved through round-trips. It also supports protein structures and protein-ligand complexes for binding data preprocessing, and integrates directly with RDKit, biotite and numpy.
|
|
55
|
+
|
|
56
|
+
## Features
|
|
57
|
+
|
|
58
|
+
- **Compact storage** — atomic numbers (uint8), charges (int8), chirality (int8), bonds (int16)
|
|
59
|
+
- **Stereochemistry** — E/Z bond directions and chirality preserved through canonicalisation and atom reordering
|
|
60
|
+
- **HDF5 persistence** — sharded save/load, with array data read on demand rather than up front
|
|
61
|
+
- **Conformer ensembles** — multiple 3D conformers per molecule with optional Boltzmann weights
|
|
62
|
+
- **Protein support** — residue annotations, biotite integration, prolif interaction detection
|
|
63
|
+
- **Configurable vocabularies** — toggle chirality and E/Z directions without rewriting your dataset
|
|
64
|
+
- **Lightweight** — core package needs only numpy, rdkit, scipy, h5py, biotite
|
|
65
|
+
|
|
66
|
+
## Installation
|
|
67
|
+
|
|
68
|
+
> Not yet on PyPI. For now, install editable from a local clone (see [Development](#development)).
|
|
69
|
+
> Once published the install will be `pip install molito`.
|
|
70
|
+
|
|
71
|
+
Requires Python >= 3.11.
|
|
72
|
+
|
|
73
|
+
Optional extras:
|
|
74
|
+
|
|
75
|
+
| extra | adds | enables |
|
|
76
|
+
|---|---|---|
|
|
77
|
+
| `interactions` | prolif, MDAnalysis | `Protein.to_prolif()`, `InteractionSet.from_system()` |
|
|
78
|
+
| `dev` | matplotlib, jupyter, ruff, mypy | development tooling |
|
|
79
|
+
| `docs` | mkdocs-material, mkdocstrings | building the documentation site |
|
|
80
|
+
|
|
81
|
+
`optimise_mol_xtb()` additionally requires `xtb-python`, which only installs reliably from
|
|
82
|
+
conda-forge:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
mamba install -c conda-forge xtb-python
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Quick Start
|
|
89
|
+
|
|
90
|
+
### Reading molecules
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
from molito.mol import GraphBatch, GraphMol
|
|
94
|
+
|
|
95
|
+
mol = GraphMol.from_smiles("C/C=C/C")
|
|
96
|
+
|
|
97
|
+
# A whole SDF at once. Tags on each record become mol.meta, and 3D coordinates
|
|
98
|
+
# are kept. Records holding only 2D depiction coordinates load without conformers.
|
|
99
|
+
batch = GraphBatch.from_sdf("ligands.sdf", remove_hs=True)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
RDKit molecules go in directly, which is also where the atom-ordering choice lives:
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
from rdkit import Chem
|
|
106
|
+
|
|
107
|
+
mol = GraphMol.from_rdkit(Chem.MolFromSmiles("C/C=C/C"))
|
|
108
|
+
|
|
109
|
+
# Opt in to canonical atom ordering when ingesting the same compounds from
|
|
110
|
+
# different formats, so the stored representation matches. Off by default.
|
|
111
|
+
mol = GraphMol.from_rdkit(rdkit_mol, canonicalise=True)
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
### Accessing properties
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
mol.atomics # uint8 array of atomic numbers
|
|
118
|
+
mol.charges # int8 array of formal charges
|
|
119
|
+
mol.tokens # ['C_0', 'C_0', 'C_0', 'C_0'] — includes chirality if present
|
|
120
|
+
mol.charged_symbols # ['C_0', 'C_0', 'C_0', 'C_0'] — without chirality
|
|
121
|
+
mol.bond_indices # [n_bonds, 2] array
|
|
122
|
+
mol.bond_types # bond encoding indices
|
|
123
|
+
mol.coords # [n_confs, n_atoms, 3] float32, or None
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
### Writing molecules back out
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
mol.to_smiles() # stereochemistry preserved
|
|
130
|
+
rdkit_mol = mol.to_rdkit() # None if the graph cannot be sanitised
|
|
131
|
+
batch.to_sdf("out.sdf") # meta entries are written as SDF tags
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
### Saving and loading datasets
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
batch = GraphBatch([mol1, mol2, mol3])
|
|
138
|
+
batch.save("my_dataset/", shard_size=1000, columnar_meta=True)
|
|
139
|
+
|
|
140
|
+
loaded = GraphBatch.load("my_dataset/")
|
|
141
|
+
mol = loaded[0]
|
|
142
|
+
mol.atomics # array data is read from HDF5 here, not at load time
|
|
143
|
+
loaded.close_hdf5() # loaded mols stop working after this - see below
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
For datasets beyond a million or so molecules, `materialise=False` skips building the
|
|
147
|
+
Python objects until each molecule is asked for, and `meta_column` scans a metadata key
|
|
148
|
+
without building any at all:
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
import numpy as np
|
|
152
|
+
|
|
153
|
+
loaded = GraphBatch.load("my_dataset/", materialise=False)
|
|
154
|
+
|
|
155
|
+
ids = loaded.meta_column("mol_id") # one HDF5 column read
|
|
156
|
+
train = loaded.subset(np.where(ids != "")[0]) # builds only the selection
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Molecules from a loaded batch read their arrays from the open file, so they stop working
|
|
160
|
+
once you call `close_hdf5()`. Call `mol.read()` to detach the ones you want to keep.
|
|
161
|
+
|
|
162
|
+
### Vocabularies for model training
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
from molito.core import VocabConfig
|
|
166
|
+
|
|
167
|
+
# Defaults: chirality and E/Z directions enabled
|
|
168
|
+
n_atom_types = len(VocabConfig.atoms) # includes PAD, MASK, CW/CCW variants
|
|
169
|
+
n_bond_types = len(VocabConfig.bonds)
|
|
170
|
+
|
|
171
|
+
# Convert tokens to model indices (with chirality fallback)
|
|
172
|
+
indices = VocabConfig.atoms.resolve_tokens(mol.tokens)
|
|
173
|
+
|
|
174
|
+
# Disable features you don't need. This changes the model index space only —
|
|
175
|
+
# the stored dataset is untouched.
|
|
176
|
+
VocabConfig.set_chirality(False)
|
|
177
|
+
VocabConfig.set_directions(False)
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
### Conformer generation and geometry
|
|
181
|
+
|
|
182
|
+
These take RDKit molecules that already have hydrogens and a conformer; they return `None`
|
|
183
|
+
rather than raising when the underlying force field cannot run.
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
from rdkit import Chem
|
|
187
|
+
from rdkit.Chem import AllChem
|
|
188
|
+
from molito.geometry import calc_energy_mmff, optimise_mol_mmff, sample_ensemble
|
|
189
|
+
|
|
190
|
+
rdkit_mol = Chem.AddHs(Chem.MolFromSmiles("CCO"))
|
|
191
|
+
AllChem.EmbedMolecule(rdkit_mol)
|
|
192
|
+
|
|
193
|
+
energy = calc_energy_mmff(rdkit_mol)
|
|
194
|
+
opt_mol = optimise_mol_mmff(rdkit_mol, max_iters=500)
|
|
195
|
+
|
|
196
|
+
# A Boltzmann-weighted ensemble: deduplicated, strain-filtered conformers plus weights
|
|
197
|
+
final_mol, weights, e_min = sample_ensemble(rdkit_mol, max_confs=128)
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
### Proteins and complexes
|
|
201
|
+
|
|
202
|
+
```python
|
|
203
|
+
import biotite.structure.io as strucio
|
|
204
|
+
from molito.mol import BindingComplex, Protein
|
|
205
|
+
|
|
206
|
+
atom_array = strucio.load_structure("protein.pdb")
|
|
207
|
+
protein = Protein.from_biotite(atom_array)
|
|
208
|
+
|
|
209
|
+
system = BindingComplex(protein, ligand_mol)
|
|
210
|
+
system.atomics # ligand atoms first, then protein
|
|
211
|
+
system.coords # [n_complex_atoms, 3]
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
## Documentation
|
|
215
|
+
|
|
216
|
+
- [Tutorial](docs/tutorial.md) — an SDF through to a padded training batch
|
|
217
|
+
- [Concepts](docs/concepts.md) — bond encodings vs vocabulary indices, atom ordering, deferred loading, metadata
|
|
218
|
+
- [Stereochemistry](docs/stereochemistry.md) — how stereo survives reordering, and what would break it
|
|
219
|
+
- [Getting Started](docs/getting-started.md) — installation and the core classes
|
|
220
|
+
|
|
221
|
+
Build the site locally with `mkdocs serve` after installing the `docs` extra.
|
|
222
|
+
|
|
223
|
+
## Package Structure
|
|
224
|
+
|
|
225
|
+
```
|
|
226
|
+
molito/
|
|
227
|
+
core/ # AtomSet, BondSet, ConfSet, vocab, metadata, on-disk format
|
|
228
|
+
mol/ # GraphMol, Protein, BindingComplex, InteractionSet
|
|
229
|
+
geometry/ # conformer sampling, alignment, MMFF, xTB
|
|
230
|
+
convert.py # RDKit <-> molito conversion
|
|
231
|
+
arrays.py # numpy array ops (padding, one-hot, adjacency)
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
## Development
|
|
235
|
+
|
|
236
|
+
Clone and install editable:
|
|
237
|
+
|
|
238
|
+
```bash
|
|
239
|
+
git clone https://github.com/rssrwn/molito
|
|
240
|
+
cd molito
|
|
241
|
+
pip install -e ".[interactions,dev]"
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
Run the tests:
|
|
245
|
+
|
|
246
|
+
```bash
|
|
247
|
+
python -m unittest discover tests/ -v
|
|
248
|
+
```
|
|
249
|
+
|
|
250
|
+
CI enforces lint, formatting and type checking, so run these before opening a PR:
|
|
251
|
+
|
|
252
|
+
```bash
|
|
253
|
+
ruff format . && ruff check . && mypy
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
## License
|
|
257
|
+
|
|
258
|
+
MIT — see [LICENSE](LICENSE).
|
molito-0.1.0/README.md
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
# *Molito* - Small Molecule Data Processing Utility
|
|
2
|
+
|
|
3
|
+
A Python toolkit for processing and storing small molecule and protein data into a training-ready format.
|
|
4
|
+
|
|
5
|
+
Molito provides compact, serialisable representations for molecular graphs with atoms, bonds, and 3D conformer ensembles. It's designed for machine learning workflows where you need to go from RDKit molecules to structured arrays efficiently, with full support for stereochemistry (E/Z bond directions, tetrahedral chirality) preserved through round-trips. It also supports protein structures and protein-ligand complexes for binding data preprocessing, and integrates directly with RDKit, biotite and numpy.
|
|
6
|
+
|
|
7
|
+
## Features
|
|
8
|
+
|
|
9
|
+
- **Compact storage** — atomic numbers (uint8), charges (int8), chirality (int8), bonds (int16)
|
|
10
|
+
- **Stereochemistry** — E/Z bond directions and chirality preserved through canonicalisation and atom reordering
|
|
11
|
+
- **HDF5 persistence** — sharded save/load, with array data read on demand rather than up front
|
|
12
|
+
- **Conformer ensembles** — multiple 3D conformers per molecule with optional Boltzmann weights
|
|
13
|
+
- **Protein support** — residue annotations, biotite integration, prolif interaction detection
|
|
14
|
+
- **Configurable vocabularies** — toggle chirality and E/Z directions without rewriting your dataset
|
|
15
|
+
- **Lightweight** — core package needs only numpy, rdkit, scipy, h5py, biotite
|
|
16
|
+
|
|
17
|
+
## Installation
|
|
18
|
+
|
|
19
|
+
> Not yet on PyPI. For now, install editable from a local clone (see [Development](#development)).
|
|
20
|
+
> Once published the install will be `pip install molito`.
|
|
21
|
+
|
|
22
|
+
Requires Python >= 3.11.
|
|
23
|
+
|
|
24
|
+
Optional extras:
|
|
25
|
+
|
|
26
|
+
| extra | adds | enables |
|
|
27
|
+
|---|---|---|
|
|
28
|
+
| `interactions` | prolif, MDAnalysis | `Protein.to_prolif()`, `InteractionSet.from_system()` |
|
|
29
|
+
| `dev` | matplotlib, jupyter, ruff, mypy | development tooling |
|
|
30
|
+
| `docs` | mkdocs-material, mkdocstrings | building the documentation site |
|
|
31
|
+
|
|
32
|
+
`optimise_mol_xtb()` additionally requires `xtb-python`, which only installs reliably from
|
|
33
|
+
conda-forge:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
mamba install -c conda-forge xtb-python
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Quick Start
|
|
40
|
+
|
|
41
|
+
### Reading molecules
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from molito.mol import GraphBatch, GraphMol
|
|
45
|
+
|
|
46
|
+
mol = GraphMol.from_smiles("C/C=C/C")
|
|
47
|
+
|
|
48
|
+
# A whole SDF at once. Tags on each record become mol.meta, and 3D coordinates
|
|
49
|
+
# are kept. Records holding only 2D depiction coordinates load without conformers.
|
|
50
|
+
batch = GraphBatch.from_sdf("ligands.sdf", remove_hs=True)
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
RDKit molecules go in directly, which is also where the atom-ordering choice lives:
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from rdkit import Chem
|
|
57
|
+
|
|
58
|
+
mol = GraphMol.from_rdkit(Chem.MolFromSmiles("C/C=C/C"))
|
|
59
|
+
|
|
60
|
+
# Opt in to canonical atom ordering when ingesting the same compounds from
|
|
61
|
+
# different formats, so the stored representation matches. Off by default.
|
|
62
|
+
mol = GraphMol.from_rdkit(rdkit_mol, canonicalise=True)
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
### Accessing properties
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
mol.atomics # uint8 array of atomic numbers
|
|
69
|
+
mol.charges # int8 array of formal charges
|
|
70
|
+
mol.tokens # ['C_0', 'C_0', 'C_0', 'C_0'] — includes chirality if present
|
|
71
|
+
mol.charged_symbols # ['C_0', 'C_0', 'C_0', 'C_0'] — without chirality
|
|
72
|
+
mol.bond_indices # [n_bonds, 2] array
|
|
73
|
+
mol.bond_types # bond encoding indices
|
|
74
|
+
mol.coords # [n_confs, n_atoms, 3] float32, or None
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
### Writing molecules back out
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
mol.to_smiles() # stereochemistry preserved
|
|
81
|
+
rdkit_mol = mol.to_rdkit() # None if the graph cannot be sanitised
|
|
82
|
+
batch.to_sdf("out.sdf") # meta entries are written as SDF tags
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
### Saving and loading datasets
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
batch = GraphBatch([mol1, mol2, mol3])
|
|
89
|
+
batch.save("my_dataset/", shard_size=1000, columnar_meta=True)
|
|
90
|
+
|
|
91
|
+
loaded = GraphBatch.load("my_dataset/")
|
|
92
|
+
mol = loaded[0]
|
|
93
|
+
mol.atomics # array data is read from HDF5 here, not at load time
|
|
94
|
+
loaded.close_hdf5() # loaded mols stop working after this - see below
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
For datasets beyond a million or so molecules, `materialise=False` skips building the
|
|
98
|
+
Python objects until each molecule is asked for, and `meta_column` scans a metadata key
|
|
99
|
+
without building any at all:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
import numpy as np
|
|
103
|
+
|
|
104
|
+
loaded = GraphBatch.load("my_dataset/", materialise=False)
|
|
105
|
+
|
|
106
|
+
ids = loaded.meta_column("mol_id") # one HDF5 column read
|
|
107
|
+
train = loaded.subset(np.where(ids != "")[0]) # builds only the selection
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Molecules from a loaded batch read their arrays from the open file, so they stop working
|
|
111
|
+
once you call `close_hdf5()`. Call `mol.read()` to detach the ones you want to keep.
|
|
112
|
+
|
|
113
|
+
### Vocabularies for model training
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
from molito.core import VocabConfig
|
|
117
|
+
|
|
118
|
+
# Defaults: chirality and E/Z directions enabled
|
|
119
|
+
n_atom_types = len(VocabConfig.atoms) # includes PAD, MASK, CW/CCW variants
|
|
120
|
+
n_bond_types = len(VocabConfig.bonds)
|
|
121
|
+
|
|
122
|
+
# Convert tokens to model indices (with chirality fallback)
|
|
123
|
+
indices = VocabConfig.atoms.resolve_tokens(mol.tokens)
|
|
124
|
+
|
|
125
|
+
# Disable features you don't need. This changes the model index space only —
|
|
126
|
+
# the stored dataset is untouched.
|
|
127
|
+
VocabConfig.set_chirality(False)
|
|
128
|
+
VocabConfig.set_directions(False)
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
### Conformer generation and geometry
|
|
132
|
+
|
|
133
|
+
These take RDKit molecules that already have hydrogens and a conformer; they return `None`
|
|
134
|
+
rather than raising when the underlying force field cannot run.
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
from rdkit import Chem
|
|
138
|
+
from rdkit.Chem import AllChem
|
|
139
|
+
from molito.geometry import calc_energy_mmff, optimise_mol_mmff, sample_ensemble
|
|
140
|
+
|
|
141
|
+
rdkit_mol = Chem.AddHs(Chem.MolFromSmiles("CCO"))
|
|
142
|
+
AllChem.EmbedMolecule(rdkit_mol)
|
|
143
|
+
|
|
144
|
+
energy = calc_energy_mmff(rdkit_mol)
|
|
145
|
+
opt_mol = optimise_mol_mmff(rdkit_mol, max_iters=500)
|
|
146
|
+
|
|
147
|
+
# A Boltzmann-weighted ensemble: deduplicated, strain-filtered conformers plus weights
|
|
148
|
+
final_mol, weights, e_min = sample_ensemble(rdkit_mol, max_confs=128)
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
### Proteins and complexes
|
|
152
|
+
|
|
153
|
+
```python
|
|
154
|
+
import biotite.structure.io as strucio
|
|
155
|
+
from molito.mol import BindingComplex, Protein
|
|
156
|
+
|
|
157
|
+
atom_array = strucio.load_structure("protein.pdb")
|
|
158
|
+
protein = Protein.from_biotite(atom_array)
|
|
159
|
+
|
|
160
|
+
system = BindingComplex(protein, ligand_mol)
|
|
161
|
+
system.atomics # ligand atoms first, then protein
|
|
162
|
+
system.coords # [n_complex_atoms, 3]
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
## Documentation
|
|
166
|
+
|
|
167
|
+
- [Tutorial](docs/tutorial.md) — an SDF through to a padded training batch
|
|
168
|
+
- [Concepts](docs/concepts.md) — bond encodings vs vocabulary indices, atom ordering, deferred loading, metadata
|
|
169
|
+
- [Stereochemistry](docs/stereochemistry.md) — how stereo survives reordering, and what would break it
|
|
170
|
+
- [Getting Started](docs/getting-started.md) — installation and the core classes
|
|
171
|
+
|
|
172
|
+
Build the site locally with `mkdocs serve` after installing the `docs` extra.
|
|
173
|
+
|
|
174
|
+
## Package Structure
|
|
175
|
+
|
|
176
|
+
```
|
|
177
|
+
molito/
|
|
178
|
+
core/ # AtomSet, BondSet, ConfSet, vocab, metadata, on-disk format
|
|
179
|
+
mol/ # GraphMol, Protein, BindingComplex, InteractionSet
|
|
180
|
+
geometry/ # conformer sampling, alignment, MMFF, xTB
|
|
181
|
+
convert.py # RDKit <-> molito conversion
|
|
182
|
+
arrays.py # numpy array ops (padding, one-hot, adjacency)
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
## Development
|
|
186
|
+
|
|
187
|
+
Clone and install editable:
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
git clone https://github.com/rssrwn/molito
|
|
191
|
+
cd molito
|
|
192
|
+
pip install -e ".[interactions,dev]"
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Run the tests:
|
|
196
|
+
|
|
197
|
+
```bash
|
|
198
|
+
python -m unittest discover tests/ -v
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
CI enforces lint, formatting and type checking, so run these before opening a PR:
|
|
202
|
+
|
|
203
|
+
```bash
|
|
204
|
+
ruff format . && ruff check . && mypy
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
## License
|
|
208
|
+
|
|
209
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""molito: Molecular representation and processing toolkit.
|
|
2
|
+
|
|
3
|
+
Provides data structures for representing molecules as graphs with atoms, bonds, and 3D conformers,
|
|
4
|
+
with efficient HDF5 serialization. Supports chirality, E/Z stereo, protein structures, and
|
|
5
|
+
protein-ligand complexes.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
9
|
+
|
|
10
|
+
from molito.core import PT, AtomSet, BondEncoding, BondSet, ConfSet, VocabConfig
|
|
11
|
+
from molito.mol import BindingComplex, ComplexBatch, GraphBatch, GraphMol, Protein, ProteinBatch
|
|
12
|
+
|
|
13
|
+
try:
|
|
14
|
+
__version__ = version("molito")
|
|
15
|
+
except PackageNotFoundError:
|
|
16
|
+
# Package isn't installed (e.g. running from a source checkout without `pip install -e .`)
|
|
17
|
+
__version__ = "0.0.0+unknown"
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"PT",
|
|
21
|
+
"AtomSet",
|
|
22
|
+
"BindingComplex",
|
|
23
|
+
"BondEncoding",
|
|
24
|
+
"BondSet",
|
|
25
|
+
"ComplexBatch",
|
|
26
|
+
"ConfSet",
|
|
27
|
+
"GraphBatch",
|
|
28
|
+
"GraphMol",
|
|
29
|
+
"Protein",
|
|
30
|
+
"ProteinBatch",
|
|
31
|
+
"VocabConfig",
|
|
32
|
+
"__version__",
|
|
33
|
+
]
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
TArr = np.ndarray
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def pad_arrays(arrays: list[TArr]) -> TArr:
|
|
9
|
+
"""Pad a list of arrays with zeros along the first dimension.
|
|
10
|
+
|
|
11
|
+
All dimensions other than the first must have the same shape across arrays.
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
arrays: List of numpy arrays to pad.
|
|
15
|
+
|
|
16
|
+
Returns:
|
|
17
|
+
np.ndarray: Batched, padded array, shape [B, L, *] where L is length of longest array.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
if len(arrays) == 0:
|
|
21
|
+
return np.array([])
|
|
22
|
+
|
|
23
|
+
max_len = max(arr.shape[0] for arr in arrays)
|
|
24
|
+
batch_shape = (len(arrays), max_len, *arrays[0].shape[1:])
|
|
25
|
+
padded = np.zeros(batch_shape, dtype=arrays[0].dtype)
|
|
26
|
+
|
|
27
|
+
for i, arr in enumerate(arrays):
|
|
28
|
+
padded[i, : arr.shape[0]] = arr
|
|
29
|
+
|
|
30
|
+
return padded
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def one_hot_encode(indices: TArr, vocab_size: int) -> TArr:
|
|
34
|
+
"""Create one-hot encodings from indices.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
indices: Indices into one-hot vectors, shape [*].
|
|
38
|
+
vocab_size: Length of returned vectors.
|
|
39
|
+
|
|
40
|
+
Returns:
|
|
41
|
+
np.ndarray: One-hot encoded vectors, shape [*, vocab_size].
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
one_hots = np.zeros((*indices.shape, vocab_size), dtype=np.long)
|
|
45
|
+
np.put_along_axis(one_hots, np.expand_dims(indices, -1), 1.0, axis=-1)
|
|
46
|
+
return one_hots
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def adj_from_edges(edge_indices: TArr, edge_types: TArr, n_nodes: int, symmetric: bool = False) -> TArr:
|
|
50
|
+
"""Create adjacency matrix from edge indices and types.
|
|
51
|
+
|
|
52
|
+
Args:
|
|
53
|
+
edge_indices: Edge list, shape [n_edges, 2]. Pairs of (from_idx, to_idx).
|
|
54
|
+
edge_types: Edge types, shape [n_edges].
|
|
55
|
+
n_nodes: Number of nodes in the adjacency matrix.
|
|
56
|
+
symmetric: If True, fill both (i,j) and (j,i) for each edge.
|
|
57
|
+
|
|
58
|
+
Returns:
|
|
59
|
+
np.ndarray: Adjacency matrix, shape [n_nodes, n_nodes].
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
adj = np.zeros((n_nodes, n_nodes), dtype=edge_types.dtype)
|
|
63
|
+
|
|
64
|
+
if len(edge_indices) == 0:
|
|
65
|
+
return adj
|
|
66
|
+
|
|
67
|
+
adj[edge_indices[:, 0], edge_indices[:, 1]] = edge_types
|
|
68
|
+
|
|
69
|
+
if symmetric:
|
|
70
|
+
adj[edge_indices[:, 1], edge_indices[:, 0]] = edge_types
|
|
71
|
+
|
|
72
|
+
return adj
|