dcatoolkit 0.1.1__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dcatoolkit-0.1.1/src/dcatoolkit.egg-info → dcatoolkit-0.1.2}/PKG-INFO +2 -1
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/pyproject.toml +2 -1
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/src/dcatoolkit/__init__.py +1 -1
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/src/dcatoolkit/representation.py +169 -34
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2/src/dcatoolkit.egg-info}/PKG-INFO +2 -1
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/src/dcatoolkit.egg-info/requires.txt +1 -0
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/LICENSE +0 -0
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/README.md +0 -0
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/setup.cfg +0 -0
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/src/dcatoolkit.egg-info/SOURCES.txt +0 -0
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/src/dcatoolkit.egg-info/dependency_links.txt +0 -0
- {dcatoolkit-0.1.1 → dcatoolkit-0.1.2}/src/dcatoolkit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
@@ -48,6 +48,7 @@ Provides-Extra: tests
|
|
|
48
48
|
Requires-Dist: pytest; extra == "tests"
|
|
49
49
|
Provides-Extra: docs
|
|
50
50
|
Requires-Dist: sphinx; extra == "docs"
|
|
51
|
+
Requires-Dist: pdoc; extra == "docs"
|
|
51
52
|
Requires-Dist: numpydoc; extra == "docs"
|
|
52
53
|
Provides-Extra: lint
|
|
53
54
|
Requires-Dist: ruffle; extra == "lint"
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dcatoolkit"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.2"
|
|
8
8
|
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
9
|
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
10
|
|
|
@@ -46,6 +46,7 @@ tests = [
|
|
|
46
46
|
]
|
|
47
47
|
docs = [
|
|
48
48
|
"sphinx",
|
|
49
|
+
"pdoc",
|
|
49
50
|
"numpydoc"
|
|
50
51
|
]
|
|
51
52
|
lint = [
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import numpy as np
|
|
2
2
|
import pandas as pd
|
|
3
|
+
from collections.abc import Iterable
|
|
3
4
|
from scipy.spatial.distance import cdist
|
|
4
5
|
|
|
5
6
|
import biotite.structure.io.pdbx as pdbx
|
|
@@ -462,6 +463,42 @@ class DirectInformationData:
|
|
|
462
463
|
"""
|
|
463
464
|
return DI_data[abs(DI_data['residue1'] - DI_data['residue2']) > 4]
|
|
464
465
|
|
|
466
|
+
@staticmethod
|
|
467
|
+
def find_DI_with_residues(critical_residues_1 : Iterable[int], critical_residues_2 : Iterable[int], max_rank: Optional[int]=None, *mapped_resi_arrs: Iterable[npt.NDArray]) -> list[tuple[list, int]]:
|
|
468
|
+
"""
|
|
469
|
+
Function that takes an n number of ranked, mapped DI pairs and checks to see if they're in a list of potential residue indices.
|
|
470
|
+
|
|
471
|
+
Parameters
|
|
472
|
+
----------
|
|
473
|
+
critical_residues_1 : collections.abc.Iterable of int
|
|
474
|
+
Specific residue indices that a DI pair will be compared to. If the first residue of the DI pair is not one of these indices, it will not be appended to results.
|
|
475
|
+
crtical_residues_2 : collections.abc.Iterable of int
|
|
476
|
+
Specific residue indices that a DI pair will be compared to. If the second residue of the DI pair is not one of these indices, it will not be appended to results.
|
|
477
|
+
threshold : int, optional
|
|
478
|
+
Maximum "rank" of the DI pair considered.
|
|
479
|
+
*mapped_resi_arrs : tuple of numpy.ndarray
|
|
480
|
+
Tuple of ranked, mapped pairs that are compared to critical residue indices and appended to results if in those indices and within threshold.
|
|
481
|
+
|
|
482
|
+
Returns
|
|
483
|
+
-------
|
|
484
|
+
results : list of tuple of list of int, int
|
|
485
|
+
Results which consist of tuples where the first element is a list of residue1, residue2, and DI score, whereas the second element is the rank.
|
|
486
|
+
"""
|
|
487
|
+
results = []
|
|
488
|
+
for mapped_resi_arr in mapped_resi_arrs:
|
|
489
|
+
# count_rank represents the rank of the DI pair being evaluated, iterating over every new row considered.
|
|
490
|
+
count_rank = 0
|
|
491
|
+
for row in mapped_resi_arr:
|
|
492
|
+
row_as_list = list(row)
|
|
493
|
+
count_rank += 1
|
|
494
|
+
if max_rank:
|
|
495
|
+
if row_as_list[0] in critical_residues_1 and row_as_list[1] in critical_residues_2 and count_rank <= max_rank:
|
|
496
|
+
results.append((row_as_list, count_rank))
|
|
497
|
+
else:
|
|
498
|
+
if row_as_list[0] in critical_residues_1 and row_as_list[1] in critical_residues_2:
|
|
499
|
+
results.append((row_as_list, count_rank))
|
|
500
|
+
return results
|
|
501
|
+
|
|
465
502
|
@staticmethod
|
|
466
503
|
def get_dist_commands(model1: str | int, model2: str | int, chain1: str, chain2: str, pairs: npt.NDArray, ca_only: bool=True, auth_res_ids=False) -> list[str]:
|
|
467
504
|
"""
|
|
@@ -543,29 +580,78 @@ class StructureInformation:
|
|
|
543
580
|
Structure obtained from an RCSB entry with a provided pdbx/mmcif file with a specified model number.
|
|
544
581
|
pdbx_file : biotite.io.pdbx.CIFFile
|
|
545
582
|
mmCIF file that contains generic information and atomic information of the protein structure categorized into mmCIF blocks.
|
|
583
|
+
model_num : int
|
|
584
|
+
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
546
585
|
|
|
547
586
|
Attributes
|
|
548
587
|
----------
|
|
549
|
-
|
|
550
|
-
|
|
588
|
+
self.atom_data : numpy.ndarray, optional
|
|
589
|
+
Entries in the format 'ATOM', residue index, chain ID, auth residue index, auth chain ID, model number
|
|
590
|
+
self.het_atom_data : numpy.ndarray, optional
|
|
591
|
+
Array of entries in the format 'HETATM', residue index, chain ID, auth residue index, auth chain ID, model number
|
|
592
|
+
self.unique_chains : numpy.ndarray, optional
|
|
593
|
+
Array of unique asym_id entries which corresponds to unique chain IDs.
|
|
594
|
+
self.chain_auth_dict : dict of str, str, optional
|
|
595
|
+
Uses chain id as a key and provides auth chain id as a value.
|
|
596
|
+
self.res_auth_dict : dict of str, tuple of int, int or optional
|
|
597
|
+
Uses chain id as a key and an array of residue index and auth residue index as a value.
|
|
551
598
|
"""
|
|
552
|
-
def __init__(self, structure, pdbx_file: pdbx.CIFFile):
|
|
599
|
+
def __init__(self, structure, pdbx_file: pdbx.CIFFile, model_num: int):
|
|
553
600
|
self.structure = structure
|
|
554
601
|
self.pdbx_file = pdbx_file
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
602
|
+
self.model_num = model_num
|
|
603
|
+
self.generate_auth_info()
|
|
604
|
+
|
|
605
|
+
def generate_auth_info(self) -> None:
|
|
606
|
+
"""
|
|
607
|
+
Ran as part of constructor function. Generates information needed to access auth information including auth_seq_id and auth_asym_id, which correspond to alternative chain ids and alternative residue indices.
|
|
608
|
+
|
|
609
|
+
Note
|
|
610
|
+
----
|
|
611
|
+
See attributes for details.
|
|
612
|
+
|
|
613
|
+
Returns
|
|
614
|
+
-------
|
|
615
|
+
None
|
|
616
|
+
"""
|
|
617
|
+
if len(self.pdbx_file.keys()) > 0:
|
|
618
|
+
self.first_block = list(self.pdbx_file)[0]
|
|
619
|
+
self.atom_site_category = self.pdbx_file[self.first_block].get('atom_site')
|
|
620
|
+
self.chain_auth_dict = {}
|
|
621
|
+
self.res_auth_dict = {}
|
|
622
|
+
if self.atom_site_category:
|
|
623
|
+
group_pdbs = []
|
|
624
|
+
seq_ids = []
|
|
625
|
+
asym_ids = []
|
|
626
|
+
auth_seq_ids = []
|
|
627
|
+
auth_asym_ids = []
|
|
628
|
+
model_nums = []
|
|
629
|
+
for col_name, col in self.atom_site_category.items():
|
|
630
|
+
if col_name == 'group_PDB':
|
|
631
|
+
group_pdbs = col.as_array()
|
|
632
|
+
elif col_name == 'label_seq_id':
|
|
633
|
+
seq_ids = col.as_array()
|
|
634
|
+
elif col_name == 'label_asym_id':
|
|
635
|
+
asym_ids = col.as_array()
|
|
636
|
+
elif col_name == 'auth_seq_id':
|
|
637
|
+
auth_seq_ids = col.as_array()
|
|
638
|
+
elif col_name == 'auth_asym_id':
|
|
639
|
+
auth_asym_ids = col.as_array()
|
|
640
|
+
elif col_name == 'pdbx_PDB_model_num':
|
|
641
|
+
model_nums = col.as_array()
|
|
642
|
+
|
|
643
|
+
atom_site_data = np.unique(np.column_stack((group_pdbs, seq_ids, asym_ids, auth_seq_ids, auth_asym_ids, model_nums)), axis=0)
|
|
644
|
+
atom_site_data = atom_site_data[atom_site_data[:,5] == str(self.model_num)]
|
|
645
|
+
self.atom_data = atom_site_data[atom_site_data[:,0] == "ATOM"]
|
|
646
|
+
self.het_atom_data = atom_site_data[atom_site_data[:,0] == "HETATM"]
|
|
647
|
+
self.unique_chains = np.unique(self.atom_data[:,2])
|
|
648
|
+
for unique_chain in self.unique_chains:
|
|
649
|
+
unique_entry = self.atom_data[self.atom_data[:,2] == unique_chain][0]
|
|
650
|
+
self.chain_auth_dict[unique_entry[2]] = unique_entry[4]
|
|
651
|
+
self.res_auth_dict[unique_entry[2]] = unique_entry[[1,3]].astype('int')
|
|
652
|
+
else:
|
|
653
|
+
self.atom_site_category = None
|
|
654
|
+
|
|
569
655
|
@staticmethod
|
|
570
656
|
def fetch_pdb(pdb_id: str, model_num: int=1, struc_format: str="mmcif") -> 'StructureInformation':
|
|
571
657
|
"""
|
|
@@ -594,7 +680,7 @@ class StructureInformation:
|
|
|
594
680
|
if fetched_data is None:
|
|
595
681
|
raise TypeError("RCSB fetch failed. Try fetch again.")
|
|
596
682
|
pdbx_file = pdbx.CIFFile.read(fetched_data)
|
|
597
|
-
return StructureInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file)
|
|
683
|
+
return StructureInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
|
|
598
684
|
|
|
599
685
|
@staticmethod
|
|
600
686
|
def read_pdb_mmCIF(pdb_filepath: str, model_num: int=1) -> 'StructureInformation':
|
|
@@ -614,7 +700,7 @@ class StructureInformation:
|
|
|
614
700
|
StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
|
|
615
701
|
"""
|
|
616
702
|
pdbx_file = pdbx.CIFFile.read(pdb_filepath)
|
|
617
|
-
return StructureInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file)
|
|
703
|
+
return StructureInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
|
|
618
704
|
|
|
619
705
|
def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
|
|
620
706
|
"""
|
|
@@ -669,6 +755,31 @@ class StructureInformation:
|
|
|
669
755
|
dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
|
|
670
756
|
return (chain1_structure, chain2_structure, dist_matrix)
|
|
671
757
|
|
|
758
|
+
def get_shift_values(self, chain1: str, chain2: str) -> tuple[int, int]:
|
|
759
|
+
"""
|
|
760
|
+
Get shift values needed for production of auth residue ids.
|
|
761
|
+
|
|
762
|
+
Parameters
|
|
763
|
+
----------
|
|
764
|
+
chain1 : str
|
|
765
|
+
Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
|
|
766
|
+
chain2 : str
|
|
767
|
+
Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
|
|
768
|
+
|
|
769
|
+
Returns
|
|
770
|
+
-------
|
|
771
|
+
(shift1, shift2) : tuple of int, int
|
|
772
|
+
Tuple containing both shift values, the difference between the auth_res_id and res_id.
|
|
773
|
+
"""
|
|
774
|
+
shift1 = 0
|
|
775
|
+
shift2 = 0
|
|
776
|
+
if self.atom_site_category:
|
|
777
|
+
shift1 = abs(self.res_auth_dict[chain1][0] - self.res_auth_dict[chain1][1])
|
|
778
|
+
shift2 = abs(self.res_auth_dict[chain2][0] - self.res_auth_dict[chain2][1])
|
|
779
|
+
return shift1, shift2
|
|
780
|
+
else:
|
|
781
|
+
return shift1, shift2
|
|
782
|
+
|
|
672
783
|
def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
|
|
673
784
|
"""
|
|
674
785
|
Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
|
|
@@ -687,15 +798,7 @@ class StructureInformation:
|
|
|
687
798
|
min_dist_pairs_atoms_arr : numpy.ndarray
|
|
688
799
|
Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,str,str]}
|
|
689
800
|
"""
|
|
690
|
-
shift1 =
|
|
691
|
-
shift2 = 0
|
|
692
|
-
for row in list(self.struct_ref_seq):
|
|
693
|
-
ref_seq_chain, ref_seq_beg, auth_ref_seq_beg = row
|
|
694
|
-
if ref_seq_chain == chain1:
|
|
695
|
-
shift1 = int(auth_ref_seq_beg) - int(ref_seq_beg)
|
|
696
|
-
if ref_seq_chain == chain2:
|
|
697
|
-
shift2 = int(auth_ref_seq_beg) - int(ref_seq_beg)
|
|
698
|
-
|
|
801
|
+
shift1, shift2 = self.get_shift_values(chain1, chain2)
|
|
699
802
|
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True)
|
|
700
803
|
min_dist_pairs_atoms = []
|
|
701
804
|
for row in pairs:
|
|
@@ -711,11 +814,11 @@ class StructureInformation:
|
|
|
711
814
|
# Generate the auth ids of the residues in the pairs ndarray
|
|
712
815
|
auth_res_id1 = row['residue1'] + shift1
|
|
713
816
|
auth_res_id2 = row['residue2'] + shift2
|
|
714
|
-
min_dist_pairs_atoms.append((row['residue1'], row['residue2'],auth_res_id1, auth_res_id2, chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
|
|
817
|
+
min_dist_pairs_atoms.append((row['residue1'], row['residue2'], auth_res_id1, auth_res_id2, chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
|
|
715
818
|
min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
|
|
716
819
|
return min_dist_pairs_atoms_arr
|
|
717
820
|
|
|
718
|
-
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str) -> set[tuple[int, int]]:
|
|
821
|
+
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_contacts: bool=False) -> set[tuple[int, int]]:
|
|
719
822
|
"""
|
|
720
823
|
Get contacts from the structure attribute where the distance between two residues is less than the threshold.
|
|
721
824
|
|
|
@@ -729,15 +832,47 @@ class StructureInformation:
|
|
|
729
832
|
Chain id corresponding to the first column of residues in the structure.
|
|
730
833
|
chain2 : str
|
|
731
834
|
Chain id corresponding to the second column of residues in the structure.
|
|
835
|
+
auth_contacts : bool
|
|
836
|
+
True if supplying alt_ids for residues indices, False if cif residue indexing is needed.
|
|
732
837
|
|
|
733
838
|
Returns
|
|
734
839
|
-------
|
|
735
840
|
contacts_set : set of tuple of ints
|
|
736
841
|
Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
|
|
737
842
|
"""
|
|
843
|
+
shift1, shift2 = self.get_shift_values(chain1, chain2)
|
|
738
844
|
chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2)
|
|
739
|
-
|
|
845
|
+
thresh_ind = np.argwhere(dist_matrix <= threshold)
|
|
740
846
|
contacts_set = set()
|
|
741
|
-
for
|
|
742
|
-
|
|
743
|
-
|
|
847
|
+
for indices in thresh_ind:
|
|
848
|
+
chain1_atom = chain1_structure[indices[0]]
|
|
849
|
+
chain2_atom = chain2_structure[indices[1]]
|
|
850
|
+
res1 = chain1_atom.res_id
|
|
851
|
+
res2 = chain2_atom.res_id
|
|
852
|
+
if not(chain1==chain2 and res1 >= res2):
|
|
853
|
+
if auth_contacts:
|
|
854
|
+
contacts_set.add((res1 + shift1, res2 + shift2))
|
|
855
|
+
else:
|
|
856
|
+
contacts_set.add((res1, res2))
|
|
857
|
+
return contacts_set
|
|
858
|
+
|
|
859
|
+
@staticmethod
|
|
860
|
+
def write_contacts_set(filepath : str, contacts_set : set[tuple[int, int]]) -> None:
|
|
861
|
+
"""
|
|
862
|
+
Write the contacts generated from get_contacts or general set of tuples of pairs.
|
|
863
|
+
|
|
864
|
+
Parameters
|
|
865
|
+
----------
|
|
866
|
+
filepath : str
|
|
867
|
+
Path of file to output contacts_set to.
|
|
868
|
+
contacts_set : set of tuple of int, int
|
|
869
|
+
Set of tuples of pairs that represent contacts.
|
|
870
|
+
|
|
871
|
+
Returns
|
|
872
|
+
-------
|
|
873
|
+
None
|
|
874
|
+
"""
|
|
875
|
+
contacts_list = list(sorted(contacts_set))
|
|
876
|
+
with open(filepath, 'w') as fs:
|
|
877
|
+
for pair in contacts_list:
|
|
878
|
+
fs.write(str(pair[0]) + "\t" + str(pair[1]) + "\n")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
@@ -48,6 +48,7 @@ Provides-Extra: tests
|
|
|
48
48
|
Requires-Dist: pytest; extra == "tests"
|
|
49
49
|
Provides-Extra: docs
|
|
50
50
|
Requires-Dist: sphinx; extra == "docs"
|
|
51
|
+
Requires-Dist: pdoc; extra == "docs"
|
|
51
52
|
Requires-Dist: numpydoc; extra == "docs"
|
|
52
53
|
Provides-Extra: lint
|
|
53
54
|
Requires-Dist: ruffle; extra == "lint"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|