molito 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. molito-0.1.0/LICENSE +21 -0
  2. molito-0.1.0/PKG-INFO +258 -0
  3. molito-0.1.0/README.md +209 -0
  4. molito-0.1.0/molito/__init__.py +33 -0
  5. molito-0.1.0/molito/arrays.py +72 -0
  6. molito-0.1.0/molito/convert.py +211 -0
  7. molito-0.1.0/molito/core/__init__.py +14 -0
  8. molito-0.1.0/molito/core/_checks.py +142 -0
  9. molito-0.1.0/molito/core/atoms.py +643 -0
  10. molito-0.1.0/molito/core/bonds.py +580 -0
  11. molito-0.1.0/molito/core/confs.py +415 -0
  12. molito-0.1.0/molito/core/format.py +79 -0
  13. molito-0.1.0/molito/core/lazydata.py +82 -0
  14. molito-0.1.0/molito/core/meta.py +357 -0
  15. molito-0.1.0/molito/core/pharmacophore.py +223 -0
  16. molito-0.1.0/molito/core/presets.py +96 -0
  17. molito-0.1.0/molito/core/pt.py +35 -0
  18. molito-0.1.0/molito/core/vocab.py +279 -0
  19. molito-0.1.0/molito/defs/pharmacophore.fdef +204 -0
  20. molito-0.1.0/molito/geometry/__init__.py +10 -0
  21. molito-0.1.0/molito/geometry/align.py +211 -0
  22. molito-0.1.0/molito/geometry/common.py +233 -0
  23. molito-0.1.0/molito/geometry/mmff.py +126 -0
  24. molito-0.1.0/molito/geometry/xtb.py +180 -0
  25. molito-0.1.0/molito/mol/__init__.py +10 -0
  26. molito-0.1.0/molito/mol/complex.py +514 -0
  27. molito-0.1.0/molito/mol/graph.py +1075 -0
  28. molito-0.1.0/molito/mol/interactions.py +463 -0
  29. molito-0.1.0/molito/mol/protein.py +648 -0
  30. molito-0.1.0/molito/py.typed +0 -0
  31. molito-0.1.0/molito.egg-info/PKG-INFO +258 -0
  32. molito-0.1.0/molito.egg-info/SOURCES.txt +35 -0
  33. molito-0.1.0/molito.egg-info/dependency_links.txt +1 -0
  34. molito-0.1.0/molito.egg-info/requires.txt +25 -0
  35. molito-0.1.0/molito.egg-info/top_level.txt +1 -0
  36. molito-0.1.0/pyproject.toml +109 -0
  37. molito-0.1.0/setup.cfg +4 -0
molito-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ross Irwin
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
molito-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,258 @@
1
+ Metadata-Version: 2.4
2
+ Name: molito
3
+ Version: 0.1.0
4
+ Summary: Molecular representation and processing toolkit for storing and processing small molecules into a training-ready format
5
+ Author-email: Ross Irwin <rssrwn@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/rssrwn/molito
8
+ Project-URL: Documentation, https://rssrwn.github.io/molito
9
+ Project-URL: Repository, https://github.com/rssrwn/molito
10
+ Project-URL: Issues, https://github.com/rssrwn/molito/issues
11
+ Keywords: cheminformatics,molecules,machine-learning,rdkit,hdf5,conformers,drug-discovery
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Chemistry
20
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.11
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: numpy>=2.0
27
+ Requires-Dist: scipy>=1.12
28
+ Requires-Dist: rdkit>=2024.3
29
+ Requires-Dist: h5py>=3.10
30
+ Requires-Dist: biotite>=1.0
31
+ Requires-Dist: more-itertools>=10.0
32
+ Provides-Extra: interactions
33
+ Requires-Dist: prolif>=2.0; extra == "interactions"
34
+ Requires-Dist: MDAnalysis; extra == "interactions"
35
+ Provides-Extra: dev
36
+ Requires-Dist: matplotlib; extra == "dev"
37
+ Requires-Dist: jupyter; extra == "dev"
38
+ Requires-Dist: ipykernel; extra == "dev"
39
+ Requires-Dist: py3Dmol; extra == "dev"
40
+ Requires-Dist: ruff==0.16.0; extra == "dev"
41
+ Requires-Dist: mypy==2.3.0; extra == "dev"
42
+ Requires-Dist: build; extra == "dev"
43
+ Requires-Dist: twine; extra == "dev"
44
+ Provides-Extra: docs
45
+ Requires-Dist: mkdocs-material>=9.5; extra == "docs"
46
+ Requires-Dist: mkdocstrings[python]>=0.24; extra == "docs"
47
+ Requires-Dist: ruff==0.16.0; extra == "docs"
48
+ Dynamic: license-file
49
+
50
+ # *Molito* - Small Molecule Data Processing Utility
51
+
52
+ A Python toolkit for processing and storing small molecule and protein data into a training-ready format.
53
+
54
+ Molito provides compact, serialisable representations for molecular graphs with atoms, bonds, and 3D conformer ensembles. It's designed for machine learning workflows where you need to go from RDKit molecules to structured arrays efficiently, with full support for stereochemistry (E/Z bond directions, tetrahedral chirality) preserved through round-trips. It also supports protein structures and protein-ligand complexes for binding data preprocessing, and integrates directly with RDKit, biotite and numpy.
55
+
56
+ ## Features
57
+
58
+ - **Compact storage** — atomic numbers (uint8), charges (int8), chirality (int8), bonds (int16)
59
+ - **Stereochemistry** — E/Z bond directions and chirality preserved through canonicalisation and atom reordering
60
+ - **HDF5 persistence** — sharded save/load, with array data read on demand rather than up front
61
+ - **Conformer ensembles** — multiple 3D conformers per molecule with optional Boltzmann weights
62
+ - **Protein support** — residue annotations, biotite integration, prolif interaction detection
63
+ - **Configurable vocabularies** — toggle chirality and E/Z directions without rewriting your dataset
64
+ - **Lightweight** — core package needs only numpy, rdkit, scipy, h5py, biotite
65
+
66
+ ## Installation
67
+
68
+ > Not yet on PyPI. For now, install editable from a local clone (see [Development](#development)).
69
+ > Once published the install will be `pip install molito`.
70
+
71
+ Requires Python >= 3.11.
72
+
73
+ Optional extras:
74
+
75
+ | extra | adds | enables |
76
+ |---|---|---|
77
+ | `interactions` | prolif, MDAnalysis | `Protein.to_prolif()`, `InteractionSet.from_system()` |
78
+ | `dev` | matplotlib, jupyter, ruff, mypy | development tooling |
79
+ | `docs` | mkdocs-material, mkdocstrings | building the documentation site |
80
+
81
+ `optimise_mol_xtb()` additionally requires `xtb-python`, which only installs reliably from
82
+ conda-forge:
83
+
84
+ ```bash
85
+ mamba install -c conda-forge xtb-python
86
+ ```
87
+
88
+ ## Quick Start
89
+
90
+ ### Reading molecules
91
+
92
+ ```python
93
+ from molito.mol import GraphBatch, GraphMol
94
+
95
+ mol = GraphMol.from_smiles("C/C=C/C")
96
+
97
+ # A whole SDF at once. Tags on each record become mol.meta, and 3D coordinates
98
+ # are kept. Records holding only 2D depiction coordinates load without conformers.
99
+ batch = GraphBatch.from_sdf("ligands.sdf", remove_hs=True)
100
+ ```
101
+
102
+ RDKit molecules go in directly, which is also where the atom-ordering choice lives:
103
+
104
+ ```python
105
+ from rdkit import Chem
106
+
107
+ mol = GraphMol.from_rdkit(Chem.MolFromSmiles("C/C=C/C"))
108
+
109
+ # Opt in to canonical atom ordering when ingesting the same compounds from
110
+ # different formats, so the stored representation matches. Off by default.
111
+ mol = GraphMol.from_rdkit(rdkit_mol, canonicalise=True)
112
+ ```
113
+
114
+ ### Accessing properties
115
+
116
+ ```python
117
+ mol.atomics # uint8 array of atomic numbers
118
+ mol.charges # int8 array of formal charges
119
+ mol.tokens # ['C_0', 'C_0', 'C_0', 'C_0'] — includes chirality if present
120
+ mol.charged_symbols # ['C_0', 'C_0', 'C_0', 'C_0'] — without chirality
121
+ mol.bond_indices # [n_bonds, 2] array
122
+ mol.bond_types # bond encoding indices
123
+ mol.coords # [n_confs, n_atoms, 3] float32, or None
124
+ ```
125
+
126
+ ### Writing molecules back out
127
+
128
+ ```python
129
+ mol.to_smiles() # stereochemistry preserved
130
+ rdkit_mol = mol.to_rdkit() # None if the graph cannot be sanitised
131
+ batch.to_sdf("out.sdf") # meta entries are written as SDF tags
132
+ ```
133
+
134
+ ### Saving and loading datasets
135
+
136
+ ```python
137
+ batch = GraphBatch([mol1, mol2, mol3])
138
+ batch.save("my_dataset/", shard_size=1000, columnar_meta=True)
139
+
140
+ loaded = GraphBatch.load("my_dataset/")
141
+ mol = loaded[0]
142
+ mol.atomics # array data is read from HDF5 here, not at load time
143
+ loaded.close_hdf5() # loaded mols stop working after this - see below
144
+ ```
145
+
146
+ For datasets beyond a million or so molecules, `materialise=False` skips building the
147
+ Python objects until each molecule is asked for, and `meta_column` scans a metadata key
148
+ without building any at all:
149
+
150
+ ```python
151
+ import numpy as np
152
+
153
+ loaded = GraphBatch.load("my_dataset/", materialise=False)
154
+
155
+ ids = loaded.meta_column("mol_id") # one HDF5 column read
156
+ train = loaded.subset(np.where(ids != "")[0]) # builds only the selection
157
+ ```
158
+
159
+ Molecules from a loaded batch read their arrays from the open file, so they stop working
160
+ once you call `close_hdf5()`. Call `mol.read()` to detach the ones you want to keep.
161
+
162
+ ### Vocabularies for model training
163
+
164
+ ```python
165
+ from molito.core import VocabConfig
166
+
167
+ # Defaults: chirality and E/Z directions enabled
168
+ n_atom_types = len(VocabConfig.atoms) # includes PAD, MASK, CW/CCW variants
169
+ n_bond_types = len(VocabConfig.bonds)
170
+
171
+ # Convert tokens to model indices (with chirality fallback)
172
+ indices = VocabConfig.atoms.resolve_tokens(mol.tokens)
173
+
174
+ # Disable features you don't need. This changes the model index space only —
175
+ # the stored dataset is untouched.
176
+ VocabConfig.set_chirality(False)
177
+ VocabConfig.set_directions(False)
178
+ ```
179
+
180
+ ### Conformer generation and geometry
181
+
182
+ These take RDKit molecules that already have hydrogens and a conformer; they return `None`
183
+ rather than raising when the underlying force field cannot run.
184
+
185
+ ```python
186
+ from rdkit import Chem
187
+ from rdkit.Chem import AllChem
188
+ from molito.geometry import calc_energy_mmff, optimise_mol_mmff, sample_ensemble
189
+
190
+ rdkit_mol = Chem.AddHs(Chem.MolFromSmiles("CCO"))
191
+ AllChem.EmbedMolecule(rdkit_mol)
192
+
193
+ energy = calc_energy_mmff(rdkit_mol)
194
+ opt_mol = optimise_mol_mmff(rdkit_mol, max_iters=500)
195
+
196
+ # A Boltzmann-weighted ensemble: deduplicated, strain-filtered conformers plus weights
197
+ final_mol, weights, e_min = sample_ensemble(rdkit_mol, max_confs=128)
198
+ ```
199
+
200
+ ### Proteins and complexes
201
+
202
+ ```python
203
+ import biotite.structure.io as strucio
204
+ from molito.mol import BindingComplex, Protein
205
+
206
+ atom_array = strucio.load_structure("protein.pdb")
207
+ protein = Protein.from_biotite(atom_array)
208
+
209
+ system = BindingComplex(protein, ligand_mol)
210
+ system.atomics # ligand atoms first, then protein
211
+ system.coords # [n_complex_atoms, 3]
212
+ ```
213
+
214
+ ## Documentation
215
+
216
+ - [Tutorial](docs/tutorial.md) — an SDF through to a padded training batch
217
+ - [Concepts](docs/concepts.md) — bond encodings vs vocabulary indices, atom ordering, deferred loading, metadata
218
+ - [Stereochemistry](docs/stereochemistry.md) — how stereo survives reordering, and what would break it
219
+ - [Getting Started](docs/getting-started.md) — installation and the core classes
220
+
221
+ Build the site locally with `mkdocs serve` after installing the `docs` extra.
222
+
223
+ ## Package Structure
224
+
225
+ ```
226
+ molito/
227
+ core/ # AtomSet, BondSet, ConfSet, vocab, metadata, on-disk format
228
+ mol/ # GraphMol, Protein, BindingComplex, InteractionSet
229
+ geometry/ # conformer sampling, alignment, MMFF, xTB
230
+ convert.py # RDKit <-> molito conversion
231
+ arrays.py # numpy array ops (padding, one-hot, adjacency)
232
+ ```
233
+
234
+ ## Development
235
+
236
+ Clone and install editable:
237
+
238
+ ```bash
239
+ git clone https://github.com/rssrwn/molito
240
+ cd molito
241
+ pip install -e ".[interactions,dev]"
242
+ ```
243
+
244
+ Run the tests:
245
+
246
+ ```bash
247
+ python -m unittest discover tests/ -v
248
+ ```
249
+
250
+ CI enforces lint, formatting and type checking, so run these before opening a PR:
251
+
252
+ ```bash
253
+ ruff format . && ruff check . && mypy
254
+ ```
255
+
256
+ ## License
257
+
258
+ MIT — see [LICENSE](LICENSE).
molito-0.1.0/README.md ADDED
@@ -0,0 +1,209 @@
1
+ # *Molito* - Small Molecule Data Processing Utility
2
+
3
+ A Python toolkit for processing and storing small molecule and protein data into a training-ready format.
4
+
5
+ Molito provides compact, serialisable representations for molecular graphs with atoms, bonds, and 3D conformer ensembles. It's designed for machine learning workflows where you need to go from RDKit molecules to structured arrays efficiently, with full support for stereochemistry (E/Z bond directions, tetrahedral chirality) preserved through round-trips. It also supports protein structures and protein-ligand complexes for binding data preprocessing, and integrates directly with RDKit, biotite and numpy.
6
+
7
+ ## Features
8
+
9
+ - **Compact storage** — atomic numbers (uint8), charges (int8), chirality (int8), bonds (int16)
10
+ - **Stereochemistry** — E/Z bond directions and chirality preserved through canonicalisation and atom reordering
11
+ - **HDF5 persistence** — sharded save/load, with array data read on demand rather than up front
12
+ - **Conformer ensembles** — multiple 3D conformers per molecule with optional Boltzmann weights
13
+ - **Protein support** — residue annotations, biotite integration, prolif interaction detection
14
+ - **Configurable vocabularies** — toggle chirality and E/Z directions without rewriting your dataset
15
+ - **Lightweight** — core package needs only numpy, rdkit, scipy, h5py, biotite
16
+
17
+ ## Installation
18
+
19
+ > Not yet on PyPI. For now, install editable from a local clone (see [Development](#development)).
20
+ > Once published the install will be `pip install molito`.
21
+
22
+ Requires Python >= 3.11.
23
+
24
+ Optional extras:
25
+
26
+ | extra | adds | enables |
27
+ |---|---|---|
28
+ | `interactions` | prolif, MDAnalysis | `Protein.to_prolif()`, `InteractionSet.from_system()` |
29
+ | `dev` | matplotlib, jupyter, ruff, mypy | development tooling |
30
+ | `docs` | mkdocs-material, mkdocstrings | building the documentation site |
31
+
32
+ `optimise_mol_xtb()` additionally requires `xtb-python`, which only installs reliably from
33
+ conda-forge:
34
+
35
+ ```bash
36
+ mamba install -c conda-forge xtb-python
37
+ ```
38
+
39
+ ## Quick Start
40
+
41
+ ### Reading molecules
42
+
43
+ ```python
44
+ from molito.mol import GraphBatch, GraphMol
45
+
46
+ mol = GraphMol.from_smiles("C/C=C/C")
47
+
48
+ # A whole SDF at once. Tags on each record become mol.meta, and 3D coordinates
49
+ # are kept. Records holding only 2D depiction coordinates load without conformers.
50
+ batch = GraphBatch.from_sdf("ligands.sdf", remove_hs=True)
51
+ ```
52
+
53
+ RDKit molecules go in directly, which is also where the atom-ordering choice lives:
54
+
55
+ ```python
56
+ from rdkit import Chem
57
+
58
+ mol = GraphMol.from_rdkit(Chem.MolFromSmiles("C/C=C/C"))
59
+
60
+ # Opt in to canonical atom ordering when ingesting the same compounds from
61
+ # different formats, so the stored representation matches. Off by default.
62
+ mol = GraphMol.from_rdkit(rdkit_mol, canonicalise=True)
63
+ ```
64
+
65
+ ### Accessing properties
66
+
67
+ ```python
68
+ mol.atomics # uint8 array of atomic numbers
69
+ mol.charges # int8 array of formal charges
70
+ mol.tokens # ['C_0', 'C_0', 'C_0', 'C_0'] — includes chirality if present
71
+ mol.charged_symbols # ['C_0', 'C_0', 'C_0', 'C_0'] — without chirality
72
+ mol.bond_indices # [n_bonds, 2] array
73
+ mol.bond_types # bond encoding indices
74
+ mol.coords # [n_confs, n_atoms, 3] float32, or None
75
+ ```
76
+
77
+ ### Writing molecules back out
78
+
79
+ ```python
80
+ mol.to_smiles() # stereochemistry preserved
81
+ rdkit_mol = mol.to_rdkit() # None if the graph cannot be sanitised
82
+ batch.to_sdf("out.sdf") # meta entries are written as SDF tags
83
+ ```
84
+
85
+ ### Saving and loading datasets
86
+
87
+ ```python
88
+ batch = GraphBatch([mol1, mol2, mol3])
89
+ batch.save("my_dataset/", shard_size=1000, columnar_meta=True)
90
+
91
+ loaded = GraphBatch.load("my_dataset/")
92
+ mol = loaded[0]
93
+ mol.atomics # array data is read from HDF5 here, not at load time
94
+ loaded.close_hdf5() # loaded mols stop working after this - see below
95
+ ```
96
+
97
+ For datasets beyond a million or so molecules, `materialise=False` skips building the
98
+ Python objects until each molecule is asked for, and `meta_column` scans a metadata key
99
+ without building any at all:
100
+
101
+ ```python
102
+ import numpy as np
103
+
104
+ loaded = GraphBatch.load("my_dataset/", materialise=False)
105
+
106
+ ids = loaded.meta_column("mol_id") # one HDF5 column read
107
+ train = loaded.subset(np.where(ids != "")[0]) # builds only the selection
108
+ ```
109
+
110
+ Molecules from a loaded batch read their arrays from the open file, so they stop working
111
+ once you call `close_hdf5()`. Call `mol.read()` to detach the ones you want to keep.
112
+
113
+ ### Vocabularies for model training
114
+
115
+ ```python
116
+ from molito.core import VocabConfig
117
+
118
+ # Defaults: chirality and E/Z directions enabled
119
+ n_atom_types = len(VocabConfig.atoms) # includes PAD, MASK, CW/CCW variants
120
+ n_bond_types = len(VocabConfig.bonds)
121
+
122
+ # Convert tokens to model indices (with chirality fallback)
123
+ indices = VocabConfig.atoms.resolve_tokens(mol.tokens)
124
+
125
+ # Disable features you don't need. This changes the model index space only —
126
+ # the stored dataset is untouched.
127
+ VocabConfig.set_chirality(False)
128
+ VocabConfig.set_directions(False)
129
+ ```
130
+
131
+ ### Conformer generation and geometry
132
+
133
+ These take RDKit molecules that already have hydrogens and a conformer; they return `None`
134
+ rather than raising when the underlying force field cannot run.
135
+
136
+ ```python
137
+ from rdkit import Chem
138
+ from rdkit.Chem import AllChem
139
+ from molito.geometry import calc_energy_mmff, optimise_mol_mmff, sample_ensemble
140
+
141
+ rdkit_mol = Chem.AddHs(Chem.MolFromSmiles("CCO"))
142
+ AllChem.EmbedMolecule(rdkit_mol)
143
+
144
+ energy = calc_energy_mmff(rdkit_mol)
145
+ opt_mol = optimise_mol_mmff(rdkit_mol, max_iters=500)
146
+
147
+ # A Boltzmann-weighted ensemble: deduplicated, strain-filtered conformers plus weights
148
+ final_mol, weights, e_min = sample_ensemble(rdkit_mol, max_confs=128)
149
+ ```
150
+
151
+ ### Proteins and complexes
152
+
153
+ ```python
154
+ import biotite.structure.io as strucio
155
+ from molito.mol import BindingComplex, Protein
156
+
157
+ atom_array = strucio.load_structure("protein.pdb")
158
+ protein = Protein.from_biotite(atom_array)
159
+
160
+ system = BindingComplex(protein, ligand_mol)
161
+ system.atomics # ligand atoms first, then protein
162
+ system.coords # [n_complex_atoms, 3]
163
+ ```
164
+
165
+ ## Documentation
166
+
167
+ - [Tutorial](docs/tutorial.md) — an SDF through to a padded training batch
168
+ - [Concepts](docs/concepts.md) — bond encodings vs vocabulary indices, atom ordering, deferred loading, metadata
169
+ - [Stereochemistry](docs/stereochemistry.md) — how stereo survives reordering, and what would break it
170
+ - [Getting Started](docs/getting-started.md) — installation and the core classes
171
+
172
+ Build the site locally with `mkdocs serve` after installing the `docs` extra.
173
+
174
+ ## Package Structure
175
+
176
+ ```
177
+ molito/
178
+ core/ # AtomSet, BondSet, ConfSet, vocab, metadata, on-disk format
179
+ mol/ # GraphMol, Protein, BindingComplex, InteractionSet
180
+ geometry/ # conformer sampling, alignment, MMFF, xTB
181
+ convert.py # RDKit <-> molito conversion
182
+ arrays.py # numpy array ops (padding, one-hot, adjacency)
183
+ ```
184
+
185
+ ## Development
186
+
187
+ Clone and install editable:
188
+
189
+ ```bash
190
+ git clone https://github.com/rssrwn/molito
191
+ cd molito
192
+ pip install -e ".[interactions,dev]"
193
+ ```
194
+
195
+ Run the tests:
196
+
197
+ ```bash
198
+ python -m unittest discover tests/ -v
199
+ ```
200
+
201
+ CI enforces lint, formatting and type checking, so run these before opening a PR:
202
+
203
+ ```bash
204
+ ruff format . && ruff check . && mypy
205
+ ```
206
+
207
+ ## License
208
+
209
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,33 @@
1
+ """molito: Molecular representation and processing toolkit.
2
+
3
+ Provides data structures for representing molecules as graphs with atoms, bonds, and 3D conformers,
4
+ with efficient HDF5 serialization. Supports chirality, E/Z stereo, protein structures, and
5
+ protein-ligand complexes.
6
+ """
7
+
8
+ from importlib.metadata import PackageNotFoundError, version
9
+
10
+ from molito.core import PT, AtomSet, BondEncoding, BondSet, ConfSet, VocabConfig
11
+ from molito.mol import BindingComplex, ComplexBatch, GraphBatch, GraphMol, Protein, ProteinBatch
12
+
13
+ try:
14
+ __version__ = version("molito")
15
+ except PackageNotFoundError:
16
+ # Package isn't installed (e.g. running from a source checkout without `pip install -e .`)
17
+ __version__ = "0.0.0+unknown"
18
+
19
+ __all__ = [
20
+ "PT",
21
+ "AtomSet",
22
+ "BindingComplex",
23
+ "BondEncoding",
24
+ "BondSet",
25
+ "ComplexBatch",
26
+ "ConfSet",
27
+ "GraphBatch",
28
+ "GraphMol",
29
+ "Protein",
30
+ "ProteinBatch",
31
+ "VocabConfig",
32
+ "__version__",
33
+ ]
@@ -0,0 +1,72 @@
1
+ from __future__ import annotations
2
+
3
+ import numpy as np
4
+
5
+ TArr = np.ndarray
6
+
7
+
8
+ def pad_arrays(arrays: list[TArr]) -> TArr:
9
+ """Pad a list of arrays with zeros along the first dimension.
10
+
11
+ All dimensions other than the first must have the same shape across arrays.
12
+
13
+ Args:
14
+ arrays: List of numpy arrays to pad.
15
+
16
+ Returns:
17
+ np.ndarray: Batched, padded array, shape [B, L, *] where L is length of longest array.
18
+ """
19
+
20
+ if len(arrays) == 0:
21
+ return np.array([])
22
+
23
+ max_len = max(arr.shape[0] for arr in arrays)
24
+ batch_shape = (len(arrays), max_len, *arrays[0].shape[1:])
25
+ padded = np.zeros(batch_shape, dtype=arrays[0].dtype)
26
+
27
+ for i, arr in enumerate(arrays):
28
+ padded[i, : arr.shape[0]] = arr
29
+
30
+ return padded
31
+
32
+
33
+ def one_hot_encode(indices: TArr, vocab_size: int) -> TArr:
34
+ """Create one-hot encodings from indices.
35
+
36
+ Args:
37
+ indices: Indices into one-hot vectors, shape [*].
38
+ vocab_size: Length of returned vectors.
39
+
40
+ Returns:
41
+ np.ndarray: One-hot encoded vectors, shape [*, vocab_size].
42
+ """
43
+
44
+ one_hots = np.zeros((*indices.shape, vocab_size), dtype=np.long)
45
+ np.put_along_axis(one_hots, np.expand_dims(indices, -1), 1.0, axis=-1)
46
+ return one_hots
47
+
48
+
49
+ def adj_from_edges(edge_indices: TArr, edge_types: TArr, n_nodes: int, symmetric: bool = False) -> TArr:
50
+ """Create adjacency matrix from edge indices and types.
51
+
52
+ Args:
53
+ edge_indices: Edge list, shape [n_edges, 2]. Pairs of (from_idx, to_idx).
54
+ edge_types: Edge types, shape [n_edges].
55
+ n_nodes: Number of nodes in the adjacency matrix.
56
+ symmetric: If True, fill both (i,j) and (j,i) for each edge.
57
+
58
+ Returns:
59
+ np.ndarray: Adjacency matrix, shape [n_nodes, n_nodes].
60
+ """
61
+
62
+ adj = np.zeros((n_nodes, n_nodes), dtype=edge_types.dtype)
63
+
64
+ if len(edge_indices) == 0:
65
+ return adj
66
+
67
+ adj[edge_indices[:, 0], edge_indices[:, 1]] = edge_types
68
+
69
+ if symmetric:
70
+ adj[edge_indices[:, 1], edge_indices[:, 0]] = edge_types
71
+
72
+ return adj