mlchem-ul 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. mlchem_ul-1.1.0/LICENSE +27 -0
  2. mlchem_ul-1.1.0/MANIFEST.in +3 -0
  3. mlchem_ul-1.1.0/PKG-INFO +31 -0
  4. mlchem_ul-1.1.0/README.md +454 -0
  5. mlchem_ul-1.1.0/__init__.py +37 -0
  6. mlchem_ul-1.1.0/chem/__init__.py +0 -0
  7. mlchem_ul-1.1.0/chem/calculator/__init__.py +32 -0
  8. mlchem_ul-1.1.0/chem/calculator/descriptors.py +1210 -0
  9. mlchem_ul-1.1.0/chem/calculator/tools.py +409 -0
  10. mlchem_ul-1.1.0/chem/manipulation.py +10416 -0
  11. mlchem_ul-1.1.0/chem/visualise/__init__.py +32 -0
  12. mlchem_ul-1.1.0/chem/visualise/drawing.py +1305 -0
  13. mlchem_ul-1.1.0/chem/visualise/simmaps.py +459 -0
  14. mlchem_ul-1.1.0/chem/visualise/space.py +662 -0
  15. mlchem_ul-1.1.0/helper.py +1518 -0
  16. mlchem_ul-1.1.0/importables.py +1859 -0
  17. mlchem_ul-1.1.0/metrics.py +690 -0
  18. mlchem_ul-1.1.0/ml/__init__.py +32 -0
  19. mlchem_ul-1.1.0/ml/feature_selection/__init__.py +32 -0
  20. mlchem_ul-1.1.0/ml/feature_selection/filters.py +177 -0
  21. mlchem_ul-1.1.0/ml/feature_selection/wrappers.py +1334 -0
  22. mlchem_ul-1.1.0/ml/modelling/__init__.py +32 -0
  23. mlchem_ul-1.1.0/ml/modelling/model_evaluation.py +688 -0
  24. mlchem_ul-1.1.0/ml/modelling/model_interpretation.py +1051 -0
  25. mlchem_ul-1.1.0/ml/preprocessing/__init__.py +32 -0
  26. mlchem_ul-1.1.0/ml/preprocessing/dimensional_reduction.py +593 -0
  27. mlchem_ul-1.1.0/ml/preprocessing/feature_transformation.py +70 -0
  28. mlchem_ul-1.1.0/ml/preprocessing/scaling.py +419 -0
  29. mlchem_ul-1.1.0/ml/preprocessing/undersampling.py +183 -0
  30. mlchem_ul-1.1.0/mlchem_ul.egg-info/PKG-INFO +31 -0
  31. mlchem_ul-1.1.0/mlchem_ul.egg-info/SOURCES.txt +63 -0
  32. mlchem_ul-1.1.0/mlchem_ul.egg-info/dependency_links.txt +1 -0
  33. mlchem_ul-1.1.0/mlchem_ul.egg-info/requires.txt +23 -0
  34. mlchem_ul-1.1.0/mlchem_ul.egg-info/top_level.txt +1 -0
  35. mlchem_ul-1.1.0/requirements.txt +23 -0
  36. mlchem_ul-1.1.0/setup.cfg +4 -0
  37. mlchem_ul-1.1.0/setup.py +63 -0
  38. mlchem_ul-1.1.0/tests/test_helper.py +408 -0
  39. mlchem_ul-1.1.0/tests/test_importables.py +205 -0
  40. mlchem_ul-1.1.0/tests/test_metrics.py +228 -0
@@ -0,0 +1,27 @@
1
+ mlchem - cheminformatics library
2
+ Copyright © 2025 as Unilever Global IP Limited
3
+
4
+ Redistribution and use in source and binary forms, with or without modification,
5
+ are permitted under the terms of the BSD-3 License, provided that the following conditions are met:
6
+
7
+ 1. Redistributions of source code must retain the above copyright
8
+ notice, this list of conditions and the following disclaimer.
9
+ 2. Redistributions in binary form must reproduce the above copyright
10
+ notice, this list of conditions and the following disclaimer in
11
+ the documentation and/or other materials provided with the distribution.
12
+
13
+ 3. Neither the name of the copyright holder nor the names of its
14
+ contributors may be used to endorse or promote products derived
15
+ from this software without specific prior written permission.
16
+
17
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS “AS IS”
18
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO,
19
+ THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
20
+ PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS
21
+ BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
22
+ CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE
23
+ GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
24
+ HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
25
+ STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING
26
+ IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
27
+ POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,3 @@
1
+ include requirements.txt
2
+ include README.md
3
+ include LICENSE
@@ -0,0 +1,31 @@
1
+ Metadata-Version: 2.4
2
+ Name: mlchem-ul
3
+ Version: 1.1.0
4
+ Requires-Python: >=3.12
5
+ License-File: LICENSE
6
+ Requires-Dist: bokeh==3.9.1
7
+ Requires-Dist: chembl_structure_pipeline==1.2.4
8
+ Requires-Dist: ipython==9.14.1
9
+ Requires-Dist: joblib==1.5.3
10
+ Requires-Dist: matplotlib==3.11.0
11
+ Requires-Dist: mordredcommunity[full]==2.0.7
12
+ Requires-Dist: myst-parser==5.1.0
13
+ Requires-Dist: numpy==2.4.6
14
+ Requires-Dist: openpyxl==3.1.5
15
+ Requires-Dist: pandas==3.0.3
16
+ Requires-Dist: Pillow==12.2.0
17
+ Requires-Dist: py3Dmol==2.5.5
18
+ Requires-Dist: pytest==9.1.1
19
+ Requires-Dist: rdkit==2026.3.3
20
+ Requires-Dist: scikit_learn==1.9.0
21
+ Requires-Dist: scipy==1.18.0
22
+ Requires-Dist: seaborn==0.13.2
23
+ Requires-Dist: selfies==2.2.0
24
+ Requires-Dist: shap==0.52.0
25
+ Requires-Dist: sphinx==9.1.0
26
+ Requires-Dist: tqdm==4.67.1
27
+ Requires-Dist: umap-learn==0.5.12
28
+ Requires-Dist: xgboost==3.3.0
29
+ Dynamic: license-file
30
+ Dynamic: requires-dist
31
+ Dynamic: requires-python
@@ -0,0 +1,454 @@
1
+ # mlchem
2
+
3
+ [![Static Badge](https://img.shields.io/badge/python_version-3.12,3.13,3.14-limegreen)](https://www.python.org/)
4
+ [![Static Badge](https://img.shields.io/badge/powered_by-RDKit-0626FA?labelColor=black)](https://www.rdkit.org/)
5
+ [![Line Coverage](assets/coverage.svg)](https://github.com/seacunilever/mlchem/blob/master/assets/coverage.svg)
6
+ [![Branch Coverage](assets/coverage-branch.svg)](https://github.com/seacunilever/mlchem/blob/master/assets/coverage-branch.svg)
7
+
8
+ **mlchem** is a Python cheminformatics library designed for the scientific community. It provides a comprehensive set of tools for data handling, molecule manipulation, drawing, machine learning, and plotting.
9
+ The library has been tested for python 3.12, 3.13 and 3.14 (experimental).
10
+
11
+ ## Documentation
12
+
13
+ Available at [seacunilever.github.io/mlchem](https://seacunilever.github.io/mlchem/).
14
+
15
+ ## Features
16
+
17
+ - **Data Handling**: Efficiently manage and process chemical data, including loading, cleaning, and transforming datasets.
18
+ - **Molecule Manipulation**: Tools for manipulating molecular structures, such as adding or removing atoms, modifying bonds, and generating molecular conformations.
19
+ - **Pattern Recognition**: An extensive list of functions to search for specific structural patterns.
20
+ - **Molecule Drawing**: Visualise molecules with customisable drawing options, creating high-quality images for presentations and publications.
21
+ - **Machine Learning**: Implement machine learning models for cheminformatics, including training, evaluating, and deploying models to predict chemical properties and activities.
22
+ - **Feature Analysis and Interpretation**: Interpret model features and provide insightful plots.
23
+
24
+ ## Architecture
25
+
26
+ ![image](assets/figure1.png)
27
+
28
+
29
+ ## Modules
30
+
31
+ ### chem.visualise/
32
+
33
+ - **space.py**: Computes and visualises datasets in a lower-dimensional space.
34
+ - **simmaps.py**: Generates "rdkit-like" similarity maps based on atomic importance weights.
35
+ - **drawing.py**: Handles the drawing of molecular structures with many customisable options.
36
+
37
+ ### chem.calculator/
38
+
39
+ - **tools.py**: Provides numerous tools for chemical calculations.
40
+ - **descriptors.py**: Calculates various descriptors for molecules, including RDKit and Mordred descriptors, atomic descriptors, chemotypes, fingerprints, and some quantum chemistry properties.
41
+
42
+ ### chem.manipulation.py
43
+
44
+ The `mlchem.chem.manipulation` module offers a variety of tools for creating, converting, manipulating molecular structures, generate new molecules and recognise molecular patterns.
45
+
46
+ ### ml.feature_selection/
47
+
48
+ - **filters.py**: Provides functionalities for filtering features.
49
+ - **wrappers.py**: Offers simplified interfaces for feature selection.
50
+
51
+ ### ml.modelling/
52
+
53
+ - **model_interpretation.py**: Provides tools for interpreting machine learning models.
54
+ - **model_evaluation.py**: Contains tools for evaluating machine learning models.
55
+
56
+ ### ml.preprocessing/
57
+
58
+ - **dimensional_reduction.py**: Provides functionalities for compressing dataframes using various dimensionality reduction techniques.
59
+ - **feature_transformation.py**: Expands features to polynomial features.
60
+ - **scaling.py**: Provides functionalities for scaling dataframes using different scaling techniques.
61
+ - **undersampling.py**: Contains techniques for handling imbalanced datasets.
62
+
63
+ ## Installation
64
+
65
+ To install **mlchem**, open your command prompt and use the following command:
66
+
67
+ ```bash
68
+ pip install git+https://github.com/seacunilever/mlchem.git
69
+ ```
70
+
71
+ When a release is published to PyPI, install with:
72
+
73
+ ```bash
74
+ pip install mlchem-ul
75
+ ```
76
+
77
+ Then import in Python as:
78
+
79
+ ```python
80
+ import mlchem
81
+ ```
82
+
83
+ Development installation, to modify the code or contribute with some changes:
84
+
85
+ ```bash
86
+ # Clone the repository
87
+ git clone https://github.com/seacunilever/mlchem
88
+ cd mlchem
89
+
90
+ # (Optional: create a virtual environment)
91
+ python -m venv _venv
92
+
93
+ # Activate on macOS/Linux:
94
+ source _venv/bin/activate
95
+
96
+ # Activate on Windows (PowerShell):
97
+ .\_venv\Scripts\Activate.ps1
98
+
99
+ # Activate on Windows (cmd.exe):
100
+ _venv\Scripts\activate.bat
101
+
102
+ # Make an editable install of mlchem from the source tree
103
+ pip install -e .
104
+
105
+ # and install requirements
106
+ pip install -r requirements.txt
107
+ ```
108
+
109
+ ## Logging
110
+
111
+ **mlchem** emits diagnostic logs during pipeline execution (feature selection, model evaluation, data preprocessing) to help users track progress and debug issues. Logs are emitted at the `INFO` level by default and appear on the console.
112
+
113
+ ### Basic Usage
114
+
115
+ Logs appear automatically when running mlchem functions:
116
+
117
+ ```python
118
+ from mlchem.ml.feature_selection.wrappers import SequentialForwardSelection
119
+ from mlchem.metrics import get_geometric_S
120
+
121
+ sfs = SequentialForwardSelection(estimator=..., metric=get_geometric_S, ...)
122
+ sfs.fit(X_train, y_train, X_test, y_test)
123
+ # Logs appear on console: e.g., "10:35:55 - mlchem.ml.feature_selection.wrappers - INFO - SFS start: ..."
124
+ ```
125
+
126
+ ### Controlling Log Level
127
+
128
+ Change the logging level to filter output (e.g., show only warnings, suppress info messages):
129
+
130
+ ```python
131
+ sfs = SequentialForwardSelection(..., log_level='WARNING')
132
+ sfs.fit(...) # Only WARNING+ logs appear
133
+ ```
134
+
135
+ Valid log levels: `DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`
136
+
137
+ ### Post-Pipeline Inspection and File Logging
138
+
139
+ Capture logs to memory or file for later inspection without modifying function calls:
140
+
141
+ ```python
142
+ from mlchem.helper import start_logging
143
+
144
+ # Capture to console + memory
145
+ logs = start_logging(log_level='INFO', to_console=True)
146
+ sfs.fit(X_train, y_train, X_test, y_test)
147
+ print(logs.get_logs()) # View captured logs
148
+
149
+ # Capture to file + console
150
+ logs = start_logging(log_level='INFO', to_file='pipeline.log')
151
+ sfs.fit(...)
152
+ # Logs written to pipeline.log + displayed on console
153
+
154
+ # Capture to file only (silent)
155
+ logs = start_logging(to_console=False, to_file='pipeline.log')
156
+ sfs.fit(...) # Silent; logs only in file
157
+ ```
158
+
159
+ Use this for non-interactive scripts and production environments.
160
+
161
+ ## Compatibility checks (Python 3.12, 3.13, 3.14)
162
+
163
+ This repository now includes a local matrix runner and CI workflow scaffold to
164
+ keep cross-version support visible on every push.
165
+
166
+ For most local development, running `python -m pytest -vv tests` from repo root is enough. The commands in
167
+ this section are mainly for maintainers, release checks, or contributors who
168
+ want local parity with CI across multiple Python versions.
169
+
170
+ Prerequisite for all commands below: activate your project virtual environment first.
171
+
172
+ Warning policy note:
173
+
174
+ - A narrowly scoped `pytest` warning filter is used for an upstream SHAP
175
+ `PendingDeprecationWarning` (`shap.plots.colors._colors`) tied to matplotlib
176
+ colormap API changes.
177
+ - Keep this filter temporary and remove it once SHAP resolves the upstream issue.
178
+ - To periodically audit all warnings explicitly, run:
179
+
180
+ ```bash
181
+ python -m pytest -vv tests -W default
182
+ ```
183
+
184
+ Coverage baseline (canonical: Python 3.12):
185
+
186
+ ```bash
187
+ python -m pytest -vv tests --cov=mlchem --cov-config=.coveragerc --cov-branch --cov-report=term --cov-report=xml:coverage.xml
188
+ python - <<'PY'
189
+ import xml.etree.ElementTree as ET
190
+ root = ET.parse('coverage.xml').getroot()
191
+ line_pct = round(float(root.get('line-rate', 0.0)) * 100)
192
+ branch_pct = round(float(root.get('branch-rate', 0.0)) * 100)
193
+ print(f'line={line_pct}, branch={branch_pct}')
194
+ PY
195
+ python -m anybadge --label "line cov" --value <LINE_PERCENT> --file assets/coverage.svg --overwrite 50=red 60=orange 70=yellow 80=yellowgreen 90=green
196
+ python -m anybadge --label "branch cov" --value <BRANCH_PERCENT> --file assets/coverage-branch.svg --overwrite 50=red 60=orange 70=yellow 80=yellowgreen 90=green
197
+ ```
198
+
199
+ Note: both badges (`assets/coverage.svg` for line coverage and `assets/coverage-branch.svg` for branch coverage) are refreshed automatically by GitHub CI on push (py312 job), so local regeneration is optional and mainly useful for previewing changes before pushing.
200
+
201
+ Current policy:
202
+
203
+ - Python 3.12 and 3.13 are required to pass.
204
+ - Python 3.14 is currently experimental (reported, not blocking).
205
+
206
+ ### Local matrix (default envs under ~/Envs)
207
+
208
+ The matrix helper expects existing Python environments (for example py312, py313, py314). If you do not use this layout, use the tox entrypoint below instead.
209
+
210
+ Run all environments in fast mode (default: no dependency reinstall):
211
+
212
+ From repository root:
213
+
214
+ ```bash
215
+ python scripts/run_local_matrix.py
216
+ ```
217
+
218
+ From `scripts/` directory:
219
+
220
+ ```bash
221
+ python run_local_matrix.py
222
+ ```
223
+
224
+ On Windows, if output appears buffered/silent, use unbuffered mode:
225
+
226
+ ```bash
227
+ py -u scripts/run_local_matrix.py
228
+ ```
229
+
230
+ By default, the runner streams live progress (active env/step and pytest output).
231
+
232
+ Run all environments with full dependency reinstall + tests:
233
+
234
+ ```bash
235
+ python scripts/run_local_matrix.py --full-install -- -vv tests
236
+ ```
237
+
238
+ Strict mode (make Python 3.14 failures blocking):
239
+
240
+ ```bash
241
+ python scripts/run_local_matrix.py --strict-314 -- -vv tests
242
+ ```
243
+
244
+ Quiet mode (disable live streaming and print only summary):
245
+
246
+ ```bash
247
+ python scripts/run_local_matrix.py --no-live-output -- -vv tests
248
+ ```
249
+
250
+ ### tox entrypoint
251
+
252
+ You can also run the same idea via tox:
253
+
254
+ ```bash
255
+ python -m pip install tox
256
+ ```
257
+
258
+ ```bash
259
+ python -m tox -e py312,py313,py314
260
+ ```
261
+
262
+ The `py314` tox environment is marked non-blocking during early adoption.
263
+
264
+ ## Usage
265
+
266
+ Here's some basic examples of how to use **mlchem**:
267
+
268
+ ### calculate rdkit descriptors for two molecules
269
+ ```python
270
+ from mlchem.chem.manipulation import create_molecule
271
+ from mlchem.chem.calculator import descriptors
272
+ mol1 = create_molecule('c1ccccc1CCCO')
273
+ mol2 = create_molecule('CCCCCN')
274
+ desc_df = descriptors.get_rdkitDesc([mol1, mol2],include_3D=True)
275
+ ```
276
+
277
+ ### calculate chemotypes faster on larger datasets
278
+ ```python
279
+ from mlchem.chem.calculator import descriptors
280
+
281
+ smiles_list = ['CCO', 'CCN', 'COCC', 'c1ccccc1O']
282
+
283
+ # n_jobs=1 keeps serial execution (default)
284
+ # n_jobs>1 enables multi-threaded molecule processing
285
+ # n_jobs=-1 uses all available CPU cores
286
+ chemotypes = descriptors.get_chemotypes(smiles_list, n_jobs=4)
287
+ ```
288
+
289
+ Performance note: chemotype execution now reuses per-molecule rule results
290
+ and avoids repeated molecule preparation. This is especially important for
291
+ large rule dictionaries and medium-to-large training sets.
292
+
293
+ ### control ML verbosity in notebooks and development runs
294
+ ```python
295
+ import logging
296
+ from sklearn.linear_model import LogisticRegression
297
+ from mlchem.ml.feature_selection.wrappers import (
298
+ SequentialForwardSelection,
299
+ CombinatorialSelection,
300
+ )
301
+
302
+ # Enable library logs in your notebook session
303
+ logging.basicConfig(level=logging.INFO)
304
+
305
+ sfs = SequentialForwardSelection(
306
+ estimator=LogisticRegression(),
307
+ estimator_string='lr',
308
+ metric=lambda y_true, y_pred: (y_true == y_pred).mean(),
309
+ log_level='INFO',
310
+ )
311
+
312
+ # Runtime toggle (use DEBUG for very verbose traces)
313
+ sfs.set_log_level('DEBUG')
314
+ sfs.set_log_level('WARNING')
315
+
316
+ cs = CombinatorialSelection(
317
+ estimator=LogisticRegression(),
318
+ metric=lambda y_true, y_pred: (y_true == y_pred).mean(),
319
+ log_level='INFO',
320
+ )
321
+ ```
322
+
323
+ ### optional diagnostics for undersampling and y-scrambling
324
+ ```python
325
+ from mlchem.ml.preprocessing.undersampling import undersample
326
+ from mlchem.ml.modelling.model_evaluation import y_scrambling
327
+
328
+ train_balanced, test_updated = undersample(
329
+ train_set=train_df,
330
+ test_set=test_df,
331
+ class_column='class',
332
+ desired_proportion_majority=0.6,
333
+ log_level='INFO',
334
+ )
335
+
336
+ y_scrambling(
337
+ estimator=model,
338
+ train_set=X_train,
339
+ y_train=y_train,
340
+ test_set=X_test,
341
+ y_test=y_test,
342
+ metric_function=metric_fn,
343
+ n_iter=50,
344
+ plot=False,
345
+ log_level='INFO',
346
+ )
347
+ ```
348
+
349
+ ### calculate fingerprints
350
+ ```python
351
+ from mlchem.chem.calculator import descriptors
352
+
353
+ smiles_list = ['CCO', 'CCN', 'CCC']
354
+
355
+ # Morgan bit-vectors (2048 bits by default)
356
+ fp_df = descriptors.get_fingerprint_df(smiles_list, fp_type='m', nBits=2048)
357
+
358
+ # Include bit info for interpretability on a single molecule
359
+ fp, bit_info = descriptors.get_fingerprint('CCO', fp_type='m', include_bit_info=True)
360
+ ```
361
+
362
+ ### pattern recognition
363
+ ![image](assets/figure2.png)
364
+
365
+ ### de novo molecule generation and cleaning
366
+ ![image](assets/figure3.png)
367
+
368
+ ### show pre-defined colour palette
369
+ ![image](assets/figure4.png)
370
+
371
+
372
+ More examples in the [examples](https://github.com/seacunilever/mlchem/tree/master/examples) folder.
373
+
374
+ ## Building the documentation
375
+
376
+ The documentation is built with [Sphinx](https://www.sphinx-doc.org/) using the `autodoc` and [`myst-parser`](https://myst-parser.readthedocs.io/) extensions. Source files live under `docs/source/`, build output lands in `docs/build/html/`, and a small post-build script (`docs/_publish.py`) mirrors that build into `docs/` so GitHub Pages always serves the latest version.
377
+
378
+ > Single-source content: `docs/source/welcome.md` `{include}`s this README, so you only edit `README.md` — never duplicate content into the welcome page.
379
+
380
+ ### What is tracked, what is not
381
+
382
+ Only the inputs and the published output are tracked in git:
383
+
384
+ - **Tracked (do edit / commit)**
385
+ - `docs/source/` — Sphinx inputs (`conf.py`, `*.rst`, `welcome.md`, `_static/custom.css`).
386
+ - `docs/Makefile`, `docs/make.bat`, `docs/_publish.py` — build entry points.
387
+ - `docs/.nojekyll` — tells GitHub Pages to keep `_static/` and `_sources/`.
388
+ - `docs/*.html`, `docs/_static/`, `docs/_sources/`, `docs/_images/`, `docs/objects.inv`, `docs/searchindex.js` — the **published mirror** that GitHub Pages serves; updated automatically by `make html` via `_publish.py`.
389
+ - **Not tracked (regenerated on every build, ignored via `.gitignore`)**
390
+ - `docs/build/` — Sphinx scratch output, including `build/html/.doctrees/` and `build/html/.buildinfo` incremental-build caches.
391
+ - `docs/*warnings*.{log,txt}` — ad-hoc diagnostic logs.
392
+
393
+ ### Prerequisites
394
+
395
+ The documentation toolchain is part of `requirements.txt`. If you only want the doc deps:
396
+
397
+ ```bash
398
+ pip install sphinx myst-parser
399
+ ```
400
+
401
+ ### Build & publish
402
+
403
+ Run the commands from the `docs/` directory (NOT from `docs/source/` — `source` is the value of `SOURCEDIR` inside the `Makefile` / `make.bat`, not the working directory):
404
+
405
+ ```bash
406
+ # from the repository root
407
+ cd docs
408
+
409
+ # wipe previous build artefacts (clears docs/build/)
410
+ make clean
411
+
412
+ # build HTML and automatically mirror docs/build/html/ -> docs/
413
+ make html
414
+ ```
415
+
416
+ `make html` runs Sphinx and then invokes `_publish.py`, which:
417
+
418
+ 1. removes every stale published asset at the root of `docs/` (everything except `source/`, `build/`, `Makefile`, `make.bat`, `_publish.py`, `.nojekyll`, `.gitignore`);
419
+ 2. copies the freshly built site from `docs/build/html/` into `docs/`;
420
+ 3. ensures the `.nojekyll` marker is present so GitHub Pages keeps `_static/` and `_sources/`.
421
+
422
+ Then commit the regenerated files at the root of `docs/` — that is what gets published. `docs/build/` stays local.
423
+
424
+ If you ever build with a raw `sphinx-build` invocation, run the mirror step manually:
425
+
426
+ ```bash
427
+ make publish
428
+ ```
429
+
430
+ On Windows the same targets are dispatched through `make.bat`, so the commands work in both `cmd` and PowerShell as long as `sphinx-build` and `python` are on the `PATH`.
431
+
432
+ > Common pitfalls: running `make html` from `docs/source/` (no `Makefile` there → "no rule" / "missing Makefile" error), or typing `make build` (the Sphinx target is `html`; `build` is the *output directory*, not a target).
433
+
434
+ ## Contributing
435
+
436
+ We welcome contributions to **mlchem**. Users are free to propose new functionalities, flag new bugs, fix old bugs and issue pull requests. Please consult the [contribution guide](https://github.com/seacunilever/mlchem/blob/master/CONTRIBUTING.md) on how to properly propose and submit changes.
437
+
438
+ ## Third-Party Dependencies
439
+
440
+ This project uses the [SELFIES](https://github.com/aspuru-guzik-group/selfies) Python package for molecular string representations.
441
+ SELFIES is licensed under the [Apache License 2.0](https://www.apache.org/). In accordance with its license, the relevant license is included in this repository.
442
+
443
+ This project uses and adapts code from the [RDKit](https://www.rdkit.org) cheminformatics toolkit, which is licensed under the [BSD 3-Clause License](https://interoperable-europe.ec.europa.eu/licence/bsd-3-clause-clear-license).
444
+
445
+ ## License
446
+
447
+ This project is licensed under the BSD-3 License.
448
+
449
+ Note: This project includes components licensed under the Apache License 2.0 (e.g., the SELFIES package), as well as source code taken and adapted from RDKit library.
450
+
451
+
452
+ ## Acknowledgements
453
+
454
+ Special thanks to the Safety, Environmental & Regulatory Science (SERS) Department at Unilever.
@@ -0,0 +1,37 @@
1
+ # mlchem - cheminformatics library
2
+ # Copyright © 2025 as Unilever Global IP Limited
3
+
4
+ # Redistribution and use in source and binary forms, with or without modification,
5
+ # are permitted under the terms of the BSD-3 License, provided that the following conditions are met:
6
+
7
+ # 1. Redistributions of source code must retain the above copyright
8
+ # notice, this list of conditions and the following disclaimer.
9
+ #
10
+ # 2. Redistributions in binary form must reproduce the above copyright
11
+ # notice, this list of conditions and the following disclaimer in
12
+ # the documentation and/or other materials provided with the distribution.
13
+ #
14
+ # 3. Neither the name of the copyright holder nor the names of its
15
+ # contributors may be used to endorse or promote products derived
16
+ # from this software without specific prior written permission.
17
+
18
+ # THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS “AS IS”
19
+ # AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO,
20
+ # THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
21
+ # PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS
22
+ # BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
23
+ # CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE
24
+ # GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
25
+ # HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
26
+ # STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING
27
+ # IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
28
+ # POSSIBILITY OF SUCH DAMAGE.
29
+
30
+ # You should have received a copy of the BSD-3 License along with mlchem.
31
+ # If not, see https://interoperable-europe.ec.europa.eu/licence/bsd-3-clause-new-or-revised-license .
32
+ # It is the responsibility of mlchem users to familiarise themselves with all dependencies and their associated licenses.
33
+
34
+ from rdkit import RDLogger
35
+ RDLogger.DisableLog('rdApp.*') # suppress unsolicited RDKit warnings
36
+
37
+ __version__ = "1.0.0"
File without changes
@@ -0,0 +1,32 @@
1
+ # mlchem - cheminformatics library
2
+ # Copyright © 2025 as Unilever Global IP Limited
3
+
4
+ # Redistribution and use in source and binary forms, with or without modification,
5
+ # are permitted under the terms of the BSD-3 License, provided that the following conditions are met:
6
+
7
+ # 1. Redistributions of source code must retain the above copyright
8
+ # notice, this list of conditions and the following disclaimer.
9
+ #
10
+ # 2. Redistributions in binary form must reproduce the above copyright
11
+ # notice, this list of conditions and the following disclaimer in
12
+ # the documentation and/or other materials provided with the distribution.
13
+ #
14
+ # 3. Neither the name of the copyright holder nor the names of its
15
+ # contributors may be used to endorse or promote products derived
16
+ # from this software without specific prior written permission.
17
+
18
+ # THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS “AS IS”
19
+ # AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO,
20
+ # THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
21
+ # PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS
22
+ # BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
23
+ # CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE
24
+ # GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
25
+ # HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
26
+ # STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING
27
+ # IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
28
+ # POSSIBILITY OF SUCH DAMAGE.
29
+
30
+ # You should have received a copy of the BSD-3 License along with mlchem.
31
+ # If not, see https://interoperable-europe.ec.europa.eu/licence/bsd-3-clause-new-or-revised-license .
32
+ # It is the responsibility of mlchem users to familiarise themselves with all dependencies and their associated licenses.