papyrus-scripts 1.0.2__tar.gz → 1.0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/PKG-INFO +10 -3
  2. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/README.md +9 -2
  3. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/__init__.py +1 -1
  4. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/modelling.py +49 -19
  5. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/preprocess.py +2 -1
  6. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts.egg-info/PKG-INFO +10 -3
  7. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/LICENSE +0 -0
  8. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/setup.cfg +0 -0
  9. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/setup.py +0 -0
  10. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/__main__.py +0 -0
  11. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/cli.py +0 -0
  12. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/download.py +0 -0
  13. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/fingerprint.py +0 -0
  14. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/matchRCSB.py +0 -0
  15. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/neuralnet.py +0 -0
  16. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/reader.py +0 -0
  17. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/subsim_search.py +0 -0
  18. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/utils/IO.py +0 -0
  19. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/utils/UniprotMatch.py +0 -0
  20. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/utils/__init__.py +0 -0
  21. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/utils/links.json +0 -0
  22. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts/utils/mol_reader.py +0 -0
  23. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts.egg-info/SOURCES.txt +0 -0
  24. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts.egg-info/dependency_links.txt +0 -0
  25. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts.egg-info/entry_points.txt +0 -0
  26. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts.egg-info/requires.txt +0 -0
  27. {papyrus_scripts-1.0.2 → papyrus_scripts-1.0.3}/src/papyrus_scripts.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: papyrus_scripts
3
- Version: 1.0.2
3
+ Version: 1.0.3
4
4
  Summary: A collection of scripts to handle the Papyrus bioactivity dataset
5
5
  Home-page: https://github.com/OlivierBeq/Papyrus-scripts
6
6
  Author: Olivier J. M. Béquignon - Brandon J. Bongers - Willem Jespers
@@ -28,6 +28,13 @@ Collection of scripts to interact with the Papyrus bioactivity dataset.
28
28
 
29
29
  <br/>
30
30
 
31
+ **Associated Article:** <a href="https://doi.org/10.1186/s13321-022-00672-x">10.1186/s13321-022-00672-x</a>
32
+ ```
33
+ Béquignon OJM, Bongers BJ, Jespers W, IJzerman AP, van de Water B, van Westen GJP.
34
+ Papyrus - A large scale curated dataset aimed at bioactivity predictions.
35
+ J Cheminform 15, 3 (2023). https://doi.org/10.1186/s13321-022-00672-x
36
+ ```
37
+
31
38
  **Associated Preprint:** <a href="https://doi.org/10.33774/chemrxiv-2021-1rxhk">10.33774/chemrxiv-2021-1rxhk</a>
32
39
  ```
33
40
  Béquignon OJM, Bongers BJ, Jespers W, IJzerman AP, van de Water B, van Westen GJP.
@@ -163,9 +170,9 @@ The scripts used to extract subsets, generate models and obtain visualizations c
163
170
 
164
171
  - [x] Substructure and similarity molecular searches
165
172
  - [x] ability to use DNN models
166
- - [ ] adapt models to QSPRpred
167
173
  - [x] ability to repeat model training over multiple seeds
168
- - [ ] y-scrambling
174
+ - [x] y-scrambling
175
+ - [ ] adapt models to QSPRpred
169
176
 
170
177
  ## Examples to come
171
178
 
@@ -6,6 +6,13 @@ Collection of scripts to interact with the Papyrus bioactivity dataset.
6
6
 
7
7
  <br/>
8
8
 
9
+ **Associated Article:** <a href="https://doi.org/10.1186/s13321-022-00672-x">10.1186/s13321-022-00672-x</a>
10
+ ```
11
+ Béquignon OJM, Bongers BJ, Jespers W, IJzerman AP, van de Water B, van Westen GJP.
12
+ Papyrus - A large scale curated dataset aimed at bioactivity predictions.
13
+ J Cheminform 15, 3 (2023). https://doi.org/10.1186/s13321-022-00672-x
14
+ ```
15
+
9
16
  **Associated Preprint:** <a href="https://doi.org/10.33774/chemrxiv-2021-1rxhk">10.33774/chemrxiv-2021-1rxhk</a>
10
17
  ```
11
18
  Béquignon OJM, Bongers BJ, Jespers W, IJzerman AP, van de Water B, van Westen GJP.
@@ -141,9 +148,9 @@ The scripts used to extract subsets, generate models and obtain visualizations c
141
148
 
142
149
  - [x] Substructure and similarity molecular searches
143
150
  - [x] ability to use DNN models
144
- - [ ] adapt models to QSPRpred
145
151
  - [x] ability to repeat model training over multiple seeds
146
- - [ ] y-scrambling
152
+ - [x] y-scrambling
153
+ - [ ] adapt models to QSPRpred
147
154
 
148
155
  ## Examples to come
149
156
 
@@ -16,4 +16,4 @@ from .modelling import qsar, pcm
16
16
  from .utils.mol_reader import MolSupplier
17
17
  from .utils import IO, UniprotMatch
18
18
 
19
- __version__ = '1.0.2'
19
+ __version__ = '1.0.3'
@@ -256,7 +256,7 @@ def train_test_proportional_group_split(data: pd.DataFrame,
256
256
  test_size: float = 0.30,
257
257
  verbose: bool = False
258
258
  ) -> Tuple[pd.DataFrame, pd.DataFrame, List[int], List[int]]:
259
- """Split the data into training and test sets according to the groups that respect most test_size
259
+ """Split the data into training and test sets according to the groups that respect most test_size (based on MSE)
260
260
 
261
261
  :param data: the data to be split up into training and test sets
262
262
  :param groups: groups to split the data according to
@@ -323,19 +323,24 @@ def qsar(data: pd.DataFrame,
323
323
  :param model: machine learning model to be used for QSAR modelling
324
324
  :param folds: number of cross-validation folds to be performed
325
325
  :param stratify: whether to stratify folds for cross validation, ignored if model is RegressorMixin
326
- :param split_by: how should folds be determined {'random', 'Year', 'cluster', 'custom'}
326
+ :param split_by: how should folds be determined {'random', 'Year', 'cluster', 'custom-cluster' 'custom'}
327
327
  If 'random', exactly test_set_size is extracted for test set.
328
328
  If 'Year', the size of the test and training set are not looked at
329
- If 'cluster' or 'custom', the groups giving proportion closest to test_set_size will be used to defined the test set
329
+ If 'cluster', 'custom-cluster', the groups giving proportion closest to test_set_size will be used to
330
+ define the test set. 'cluster' uses `cluster_method` to define groups while 'custom-cluster' uses user provided
331
+ groups and creates the best suited proportional split among them.
332
+ If 'custom', the groups to be used untouched, specifying either 'training' or 'test' for each entry (other labels
333
+ are disregarded).
330
334
  :param split_year: Year from which on the test set is extracted (ignored if split_by is not 'Year')
331
335
  :param test_set_size: proportion of the dataset to be used as test set
332
336
  :param cluster_method: clustering method to use to extract test set and cross-validation folds
333
337
  (ignored if split_by is not 'cluster')
334
338
  :param custom_groups: custom groups to use to extract test set and cross-validation fold
335
- (ignored if split_by is not 'custom').
336
- Groups must be a pandas DataFrame with only two Series. The first Series is either InChIKey or connectivity
339
+ (ignored if split_by is not 'custom-cluster' or 'custom').
340
+ Groups must be a pandas DataFrame with only two Series.The first Series is either InChIKey or connectivity
337
341
  (depending on whether stereochemistry data are being use or not). The second Series must be the group assignment
338
- of each compound.
342
+ of each compound specifying either 'training' or 'test' for each entry (other labels are disregarded) when
343
+ `split_by` is 'custom' or cluster membership when `split_by` is 'custom-cluster'.
339
344
  :param scale: should the features be scaled using the custom scaling_method
340
345
  :param scale_method: scaling method to be applied to features (ignored if scale is False)
341
346
  :param yscramble: should the endpoint be shuffled to compare performance to the unshuffled endpoint
@@ -347,8 +352,9 @@ def qsar(data: pd.DataFrame,
347
352
  the data splitter for cross-validation, and for each accession in the data:
348
353
  the fitted models on each cross-validation fold and the model fitted on the complete training set.
349
354
  """
350
- if split_by.lower() not in ['year', 'random', 'cluster', 'custom']:
351
- raise ValueError("split not supported, must be one of {'Year', 'random', 'cluster', 'custom'}")
355
+ if split_by.lower() not in ['year', 'random', 'cluster', 'custom-cluster', 'custom']:
356
+ raise ValueError("split not supported, must be one of {'Year', 'random', 'cluster',"
357
+ "'custom-cluster', 'custom'}")
352
358
  if not isinstance(model, (RegressorMixin, ClassifierMixin)):
353
359
  raise ValueError('model type can only be a Scikit-Learn compliant regressor or classifier')
354
360
  warnings.filterwarnings("ignore", category=RuntimeWarning)
@@ -375,11 +381,12 @@ def qsar(data: pd.DataFrame,
375
381
  # Change endpoint
376
382
  endpoint = 'Activity_class'
377
383
  del preserved, active, inactive
378
- # Get and merge molecular descriptors
384
+ # Get and merge molecular descriptors
379
385
  descs = read_molecular_descriptors(descriptors, 'connectivity' not in data.columns,
380
386
  version, descriptor_chunksize, descriptor_path)
381
387
  descs = filter_molecular_descriptors(descs, merge_on, data[merge_on].unique())
382
388
  data = data.merge(descs, on=merge_on)
389
+ merge_on_values = data[[merge_on]]
383
390
  data = data.drop(columns=[merge_on])
384
391
  del descs
385
392
  # Table of results
@@ -391,6 +398,7 @@ def qsar(data: pd.DataFrame,
391
398
  # Build QSAR model for targets reaching criteria
392
399
  for i_target in range(n_targets - 1, -1, -1):
393
400
  tmp_data = data[data['target_id'] == targets[i_target]]
401
+ tmp_merge_on_values = merge_on_values[merge_on_values.index.isin(tmp_data.index)]
394
402
  if verbose:
395
403
  pbar.set_description(f'Building QSAR for target: {targets[i_target]} #datapoints {tmp_data.shape[0]}',
396
404
  refresh=True)
@@ -520,12 +528,18 @@ def qsar(data: pd.DataFrame,
520
528
  training_set, test_set, training_groups, _ = train_test_proportional_group_split(tmp_data, groups,
521
529
  test_set_size,
522
530
  verbose=verbose)
523
- elif split_by.lower() == 'custom':
531
+ elif split_by.lower() == 'custom-cluster':
524
532
  # Merge from custom split DataFrame
525
- groups = tmp_data[[merge_on]].merge(custom_groups, on=merge_on).iloc[:, 1].tolist()
533
+ groups = tmp_merge_on_values.merge(custom_groups, on=merge_on).iloc[:, 1].tolist()
526
534
  training_set, test_set, training_groups, _ = train_test_proportional_group_split(tmp_data, groups,
527
535
  test_set_size,
528
536
  verbose=verbose)
537
+ elif split_by.lower() == 'custom':
538
+ # Merge from custom split DataFrame
539
+ groups = tmp_merge_on_values.merge(custom_groups, on=merge_on)
540
+ training_set = tmp_data[tmp_merge_on_values.squeeze().isin(groups[groups.iloc[:, 1] == 'training'][merge_on])]
541
+ test_set = tmp_data[tmp_merge_on_values.squeeze().isin(groups[groups.iloc[:, 1] == 'test'][merge_on])]
542
+ training_groups = None
529
543
  # Drop columns not used for training
530
544
  training_set = training_set.drop(columns=['Year', 'target_id'])
531
545
  test_set = test_set.drop(columns=['Year', 'target_id'])
@@ -665,19 +679,24 @@ def pcm(data: pd.DataFrame,
665
679
  :param model: machine learning model to be used for PCM modelling
666
680
  :param folds: number of cross-validation folds to be performed
667
681
  :param stratify: whether to stratify folds for cross validation, ignored if model is RegressorMixin
668
- :param split_by: how should folds be determined {'random', 'Year', 'cluster', 'custom'}
682
+ :param split_by: how should folds be determined {'random', 'Year', 'cluster', 'custom-cluster' 'custom'}
669
683
  If 'random', exactly test_set_size is extracted for test set.
670
684
  If 'Year', the size of the test and training set are not looked at
671
- If 'cluster' or 'custom', the groups giving proportion closest to test_set_size will be used to defined the test set
685
+ If 'cluster', 'custom-cluster', the groups giving proportion closest to test_set_size will be used to
686
+ define the test set. 'cluster' uses `cluster_method` to define groups while 'custom-cluster' uses user provided
687
+ groups and creates the best suited proportional split among them.
688
+ If 'custom', the groups to be used untouched, specifying either 'training' or 'test' for each entry (other labels
689
+ are disregarded).
672
690
  :param split_year: Year from which on the test set is extracted (ignored if split_by is not 'Year')
673
691
  :param test_set_size: proportion of the dataset to be used as test set
674
692
  :param cluster_method: clustering method to use to extract test set and cross-validation folds
675
693
  (ignored if split_by is not 'cluster')
676
694
  :param custom_groups: custom groups to use to extract test set and cross-validation fold
677
- (ignored if split_by is not 'custom').
695
+ (ignored if split_by is not 'custom-cluster' or 'custom').
678
696
  Groups must be a pandas DataFrame with only two Series.The first Series is either InChIKey or connectivity
679
697
  (depending on whether stereochemistry data are being use or not). The second Series must be the group assignment
680
- of each compound.
698
+ of each compound specifying either 'training' or 'test' for each entry (other labels are disregarded) when
699
+ `split_by` is 'custom' or cluster membership when `split_by` is 'custom-cluster'.
681
700
  :param scale: should the features be scaled using the custom scaling_method
682
701
  :param scale_method: scaling method to be applied to features (ignored if scale is False)
683
702
  :param yscramble: should the endpoint be shuffled to compare performance to the unshuffled endpoint
@@ -689,8 +708,9 @@ def pcm(data: pd.DataFrame,
689
708
  the data splitter for cross-validation, fitted models on each cross-validation fold,
690
709
  the model fitted on the complete training set.
691
710
  """
692
- if split_by.lower() not in ['year', 'random', 'cluster', 'custom']:
693
- raise ValueError("split not supported, must be one of {'Year', 'random', 'cluster', 'custom'}")
711
+ if split_by.lower() not in ['year', 'random', 'cluster', 'custom-cluster', 'custom']:
712
+ raise ValueError("split not supported, must be one of {'Year', 'random', 'cluster', "
713
+ "'custom-cluster', 'custom'}")
694
714
  if not isinstance(model, (RegressorMixin, ClassifierMixin)):
695
715
  raise ValueError('model type can only be a Scikit-Learn compliant regressor or classifier')
696
716
  warnings.filterwarnings("ignore", category=RuntimeWarning)
@@ -722,6 +742,7 @@ def pcm(data: pd.DataFrame,
722
742
  version, mol_descriptor_chunksize, mol_descriptor_path)
723
743
  mol_descs = filter_molecular_descriptors(mol_descs, merge_on, data[merge_on].unique())
724
744
  data = data.merge(mol_descs, on=merge_on)
745
+ merge_on_values = data[[merge_on]]
725
746
  data = data.drop(columns=[merge_on])
726
747
  # Get and merge protein descriptors
727
748
  prot_descs = read_protein_descriptors(prot_descriptors, version, prot_descriptor_chunksize,
@@ -760,12 +781,21 @@ def pcm(data: pd.DataFrame,
760
781
  training_set, test_set, training_groups, _ = train_test_proportional_group_split(data, groups,
761
782
  test_set_size,
762
783
  verbose=verbose)
763
- elif split_by.lower() == 'custom':
784
+ elif split_by.lower() == 'custom-cluster':
764
785
  # Merge from custom split DataFrame
765
- groups = data[[merge_on]].merge(custom_groups, on=merge_on).iloc[:, 1].tolist()
786
+ groups = merge_on_values.merge(custom_groups, on=merge_on).iloc[:, 1].tolist()
766
787
  training_set, test_set, training_groups, _ = train_test_proportional_group_split(data, groups,
767
788
  test_set_size,
768
789
  verbose=verbose)
790
+ elif split_by.lower() == 'custom':
791
+ # groups = custom_groups.iloc[:, 1]
792
+ # training_set = data[merge_on_values.squeeze().isin(custom_groups[groups == 'training'][merge_on])]
793
+ # test_set = data[merge_on_values.squeeze().isin(custom_groups[groups == 'test'][merge_on])]
794
+ # Merge from custom split DataFrame
795
+ groups = merge_on_values.merge(custom_groups, on=merge_on)
796
+ training_set = data[merge_on_values.squeeze().isin(groups[groups.iloc[:, 1] == 'training'][merge_on])]
797
+ test_set = data[merge_on_values.squeeze().isin(groups[groups.iloc[:, 1] == 'test'][merge_on])]
798
+ training_groups = None
769
799
  # Drop columns not used for training
770
800
  training_set = training_set.drop(columns=['Year'])
771
801
  test_set = test_set.drop(columns=['Year'])
@@ -159,7 +159,8 @@ def keep_source(data: Union[pd.DataFrame, PandasTextFileReader, Iterator], sourc
159
159
  return data
160
160
  # Source not defined
161
161
  elif set(source).difference(sources):
162
- raise ValueError(f'Source not supported, must be one of {sources}')
162
+ # Supplied source not in data sources
163
+ return data[data.source == 'SOURCE UNAVAILABLE'] # Ensures an empty dataframe with colnames is returned
163
164
  # Sources are defined
164
165
  else:
165
166
  # Columns with optional multiple values
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: papyrus-scripts
3
- Version: 1.0.2
3
+ Version: 1.0.3
4
4
  Summary: A collection of scripts to handle the Papyrus bioactivity dataset
5
5
  Home-page: https://github.com/OlivierBeq/Papyrus-scripts
6
6
  Author: Olivier J. M. Béquignon - Brandon J. Bongers - Willem Jespers
@@ -28,6 +28,13 @@ Collection of scripts to interact with the Papyrus bioactivity dataset.
28
28
 
29
29
  <br/>
30
30
 
31
+ **Associated Article:** <a href="https://doi.org/10.1186/s13321-022-00672-x">10.1186/s13321-022-00672-x</a>
32
+ ```
33
+ Béquignon OJM, Bongers BJ, Jespers W, IJzerman AP, van de Water B, van Westen GJP.
34
+ Papyrus - A large scale curated dataset aimed at bioactivity predictions.
35
+ J Cheminform 15, 3 (2023). https://doi.org/10.1186/s13321-022-00672-x
36
+ ```
37
+
31
38
  **Associated Preprint:** <a href="https://doi.org/10.33774/chemrxiv-2021-1rxhk">10.33774/chemrxiv-2021-1rxhk</a>
32
39
  ```
33
40
  Béquignon OJM, Bongers BJ, Jespers W, IJzerman AP, van de Water B, van Westen GJP.
@@ -163,9 +170,9 @@ The scripts used to extract subsets, generate models and obtain visualizations c
163
170
 
164
171
  - [x] Substructure and similarity molecular searches
165
172
  - [x] ability to use DNN models
166
- - [ ] adapt models to QSPRpred
167
173
  - [x] ability to repeat model training over multiple seeds
168
- - [ ] y-scrambling
174
+ - [x] y-scrambling
175
+ - [ ] adapt models to QSPRpred
169
176
 
170
177
  ## Examples to come
171
178
 
File without changes