dcatoolkit 0.1.8__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dcatoolkit-0.1.8/src/dcatoolkit.egg-info → dcatoolkit-0.2.0}/PKG-INFO +2 -2
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/pyproject.toml +2 -2
- dcatoolkit-0.2.0/src/dcatoolkit/__init__.py +6 -0
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/src/dcatoolkit/analytics.py +3 -2
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/src/dcatoolkit/representation.py +343 -79
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0/src/dcatoolkit.egg-info}/PKG-INFO +2 -2
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/src/dcatoolkit.egg-info/SOURCES.txt +2 -1
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/src/dcatoolkit.egg-info/requires.txt +1 -1
- dcatoolkit-0.2.0/tests/test_contacts.py +95 -0
- dcatoolkit-0.1.8/src/dcatoolkit/__init__.py +0 -4
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/LICENSE +0 -0
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/README.md +0 -0
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/setup.cfg +0 -0
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/src/dcatoolkit.egg-info/dependency_links.txt +0 -0
- {dcatoolkit-0.1.8 → dcatoolkit-0.2.0}/src/dcatoolkit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
@@ -38,7 +38,7 @@ Classifier: Programming Language :: Python :: 3.12
|
|
|
38
38
|
Requires-Python: >=3.10
|
|
39
39
|
Description-Content-Type: text/markdown
|
|
40
40
|
License-File: LICENSE
|
|
41
|
-
Requires-Dist: biotite
|
|
41
|
+
Requires-Dist: biotite>=1.0.1
|
|
42
42
|
Requires-Dist: matplotlib>=3.8.0
|
|
43
43
|
Requires-Dist: numpy>=1.26.0
|
|
44
44
|
Requires-Dist: pandas>=2.1.0
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dcatoolkit"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
9
|
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
10
|
|
|
@@ -21,7 +21,7 @@ maintainers = [
|
|
|
21
21
|
]
|
|
22
22
|
|
|
23
23
|
dependencies = [
|
|
24
|
-
"biotite",
|
|
24
|
+
"biotite>=1.0.1",
|
|
25
25
|
"matplotlib>=3.8.0",
|
|
26
26
|
"numpy>=1.26.0",
|
|
27
27
|
"pandas>=2.1.0",
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
|
|
2
|
+
__version__ = "0.1.9"
|
|
3
|
+
from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment, MMCIFInformation, PDBInformation
|
|
4
|
+
from .analytics import MSATools
|
|
5
|
+
|
|
6
|
+
__all__ = ['Pairs', 'DirectInformationData', 'StructureInformation', 'ResidueAlignment', 'MMCIFInformation', 'PDBInformation', 'MSATools']
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import re
|
|
2
2
|
from collections import Counter
|
|
3
3
|
from typing import Optional, Union
|
|
4
|
-
import string
|
|
4
|
+
import string
|
|
5
|
+
import io
|
|
5
6
|
|
|
6
7
|
class MSATools:
|
|
7
8
|
"""
|
|
@@ -107,7 +108,7 @@ class MSATools:
|
|
|
107
108
|
List of entries that are valid in that their sequences' number of maximum continuous gaps is within the threshold supplied as `max_gaps`.
|
|
108
109
|
"""
|
|
109
110
|
table = str.maketrans('', '', string.ascii_lowercase+".")
|
|
110
|
-
if max_gaps
|
|
111
|
+
if max_gaps is None:
|
|
111
112
|
kept_entries = []
|
|
112
113
|
for header, sequence in self.MSA:
|
|
113
114
|
sequence = sequence.translate(table)
|
|
@@ -4,10 +4,11 @@ from scipy.spatial.distance import cdist
|
|
|
4
4
|
|
|
5
5
|
import biotite.structure as struc
|
|
6
6
|
import biotite.structure.io.pdbx as pdbx
|
|
7
|
+
import biotite.structure.io.pdb as pdb
|
|
7
8
|
import biotite.database.rcsb as rcsb
|
|
8
9
|
|
|
9
10
|
from collections.abc import Iterable
|
|
10
|
-
from typing import Optional, Union
|
|
11
|
+
from typing import Optional, Union, Literal, overload
|
|
11
12
|
import numpy.typing as npt
|
|
12
13
|
|
|
13
14
|
|
|
@@ -619,9 +620,124 @@ class StructureInformation:
|
|
|
619
620
|
"""
|
|
620
621
|
Information regarding a protein structure, obtained from a protein structure file.
|
|
621
622
|
|
|
623
|
+
Uses fetch_pdb() to pull protein structure information from RCSB. Uses read_x_file() to supply a filepath to pull protein structure information from a file.
|
|
624
|
+
"""
|
|
625
|
+
@overload
|
|
626
|
+
@staticmethod
|
|
627
|
+
def fetch_pdb(pdb_id: str, struc_format: Literal["mmcif"]="mmcif", model_num: int=1) -> 'MMCIFInformation':
|
|
628
|
+
...
|
|
629
|
+
|
|
630
|
+
@overload
|
|
631
|
+
@staticmethod
|
|
632
|
+
def fetch_pdb(pdb_id: str, struc_format: Literal["pdb"], model_num: int=1) -> 'PDBInformation':
|
|
633
|
+
...
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
@staticmethod
|
|
637
|
+
def fetch_pdb(pdb_id: str, struc_format: Literal["mmcif", "pdb"]="mmcif", model_num: int=1) -> Union['MMCIFInformation', 'PDBInformation']:
|
|
638
|
+
"""
|
|
639
|
+
Fetches PDB as mmCIF file from RCSB and compiles the information into a StructureInformation instance.
|
|
640
|
+
|
|
641
|
+
Parameters
|
|
642
|
+
----------
|
|
643
|
+
pdb_id : str
|
|
644
|
+
PDB ID to be fetched from the RCSB database.
|
|
645
|
+
struc_format : str
|
|
646
|
+
The format of the file to pull from the RCSB database.
|
|
647
|
+
model_num : int
|
|
648
|
+
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
649
|
+
|
|
650
|
+
Returns
|
|
651
|
+
-------
|
|
652
|
+
StructureInformation
|
|
653
|
+
StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB.
|
|
654
|
+
|
|
655
|
+
Raises
|
|
656
|
+
------
|
|
657
|
+
TypeError
|
|
658
|
+
Fetched data was not found and returned None instead.
|
|
659
|
+
ValueError
|
|
660
|
+
Structure format may be invalid (not PDBx/mmCIF or PDB).
|
|
661
|
+
"""
|
|
662
|
+
fetched_data = rcsb.fetch(pdb_id, struc_format)
|
|
663
|
+
if fetched_data is None:
|
|
664
|
+
raise TypeError("RCSB fetch failed. Try fetch again.")
|
|
665
|
+
elif struc_format == "mmcif":
|
|
666
|
+
pdbx_file = pdbx.CIFFile.read(fetched_data)
|
|
667
|
+
return MMCIFInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
|
|
668
|
+
elif struc_format == "pdb":
|
|
669
|
+
pdb_file = pdb.PDBFile.read(fetched_data)
|
|
670
|
+
return PDBInformation(pdb.get_structure(pdb_file=pdb_file, model=model_num), pdb_file=pdb_file, model_num=model_num)
|
|
671
|
+
else:
|
|
672
|
+
raise ValueError(f"struc_format {struc_format} is not valid or currently supported by DCA Toolkit")
|
|
673
|
+
@staticmethod
|
|
674
|
+
def read_mmCIF_file(pdbx_filepath: str, model_num: int=1) -> 'MMCIFInformation':
|
|
675
|
+
"""
|
|
676
|
+
Reads PDB mmCIF file from filepath and compiles the information into a CIFInformation instance.
|
|
677
|
+
|
|
678
|
+
Parameters
|
|
679
|
+
----------
|
|
680
|
+
pdbx_filepath : str
|
|
681
|
+
Filepath of the PDB mmCIF file to be read.
|
|
682
|
+
model_num : int
|
|
683
|
+
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
684
|
+
|
|
685
|
+
Returns
|
|
686
|
+
-------
|
|
687
|
+
CIFInformation
|
|
688
|
+
CIFInformation generated from pdbx.get_structure() function using the PDBx file read from the pdbx_filepath.
|
|
689
|
+
"""
|
|
690
|
+
pdbx_file = pdbx.CIFFile.read(pdbx_filepath)
|
|
691
|
+
return MMCIFInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
|
|
692
|
+
|
|
693
|
+
@staticmethod
|
|
694
|
+
def read_pdb_file(pdb_filepath: str, model_num: int=1) -> 'PDBInformation':
|
|
695
|
+
"""
|
|
696
|
+
Reads PDB file from filepath and compiles the information into a PDBInformation instance.
|
|
697
|
+
|
|
698
|
+
Parameters
|
|
699
|
+
----------
|
|
700
|
+
pdb_filepath : str
|
|
701
|
+
Filepath of the PDB mmCIF file to be read.
|
|
702
|
+
model_num : int
|
|
703
|
+
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
704
|
+
|
|
705
|
+
Returns
|
|
706
|
+
-------
|
|
707
|
+
PDBInformation
|
|
708
|
+
PDBInformation generated from pdb.get_structure() function using the PDB file read from the pdb_filepath.
|
|
709
|
+
"""
|
|
710
|
+
pdb_file = pdb.PDBFile.read(pdb_filepath)
|
|
711
|
+
return PDBInformation(pdb.get_structure(pdb_file, model=model_num), pdb_file, model_num)
|
|
712
|
+
|
|
713
|
+
@staticmethod
|
|
714
|
+
def write_contacts_set(filepath : str, contacts_set : set[tuple[int, int]]) -> None:
|
|
715
|
+
"""
|
|
716
|
+
Write the contacts generated from get_contacts or general set of tuples of pairs.
|
|
717
|
+
|
|
718
|
+
Parameters
|
|
719
|
+
----------
|
|
720
|
+
filepath : str
|
|
721
|
+
Path of file to output contacts_set to.
|
|
722
|
+
contacts_set : set of tuple of int, int
|
|
723
|
+
Set of tuples of pairs that represent contacts.
|
|
724
|
+
|
|
725
|
+
Returns
|
|
726
|
+
-------
|
|
727
|
+
None
|
|
728
|
+
"""
|
|
729
|
+
contacts_list = list(sorted(contacts_set))
|
|
730
|
+
with open(filepath, 'w') as fs:
|
|
731
|
+
for pair in contacts_list:
|
|
732
|
+
fs.write(str(pair[0]) + "\t" + str(pair[1]) + "\n")
|
|
733
|
+
|
|
734
|
+
class MMCIFInformation(StructureInformation):
|
|
735
|
+
"""
|
|
736
|
+
Information regarding a protein structure, obtained from a protein structure file.
|
|
737
|
+
|
|
622
738
|
Parameters
|
|
623
739
|
----------
|
|
624
|
-
structure : biotite.structure
|
|
740
|
+
structure : biotite.structure.AtomArray
|
|
625
741
|
Structure obtained from an RCSB entry with a provided pdbx/mmcif file with a specified model number.
|
|
626
742
|
pdbx_file : biotite.io.pdbx.CIFFile
|
|
627
743
|
mmCIF file that contains generic information and atomic information of the protein structure categorized into mmCIF blocks.
|
|
@@ -707,56 +823,6 @@ class StructureInformation:
|
|
|
707
823
|
self.res_auth_dict[unique_entry[2]] = unique_entry[[1,3]].astype('int')
|
|
708
824
|
else:
|
|
709
825
|
self.atom_site_category = None
|
|
710
|
-
|
|
711
|
-
@staticmethod
|
|
712
|
-
def fetch_pdb(pdb_id: str, model_num: int=1, struc_format: str="mmcif") -> 'StructureInformation':
|
|
713
|
-
"""
|
|
714
|
-
Fetches PDB as mmCIF file from RCSB and compiles the information into a StructureInformation instance.
|
|
715
|
-
|
|
716
|
-
Parameters
|
|
717
|
-
----------
|
|
718
|
-
pdb_id : str
|
|
719
|
-
PDB ID to be fetched from the RCSB database.
|
|
720
|
-
model_num : int
|
|
721
|
-
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
722
|
-
struc_format : str
|
|
723
|
-
The format of the file to pull from the RCSB database.
|
|
724
|
-
|
|
725
|
-
Returns
|
|
726
|
-
-------
|
|
727
|
-
StructureInformation
|
|
728
|
-
StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
|
|
729
|
-
|
|
730
|
-
Raises
|
|
731
|
-
------
|
|
732
|
-
TypeError
|
|
733
|
-
Fetched data was not found and returned None instead.
|
|
734
|
-
"""
|
|
735
|
-
fetched_data = rcsb.fetch(pdb_id, struc_format)
|
|
736
|
-
if fetched_data is None:
|
|
737
|
-
raise TypeError("RCSB fetch failed. Try fetch again.")
|
|
738
|
-
pdbx_file = pdbx.CIFFile.read(fetched_data)
|
|
739
|
-
return StructureInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
|
|
740
|
-
|
|
741
|
-
@staticmethod
|
|
742
|
-
def read_pdb_mmCIF(pdb_filepath: str, model_num: int=1) -> 'StructureInformation':
|
|
743
|
-
"""
|
|
744
|
-
Reads PDB mmCIF file from filepath and compiles the information into a StructureInformation instance.
|
|
745
|
-
|
|
746
|
-
Parameters
|
|
747
|
-
----------
|
|
748
|
-
pdb_filepath : str
|
|
749
|
-
Filepath of the PDB mmCIF file to be read.
|
|
750
|
-
model_num : int
|
|
751
|
-
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
752
|
-
|
|
753
|
-
Returns
|
|
754
|
-
-------
|
|
755
|
-
StructureInformation
|
|
756
|
-
StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
|
|
757
|
-
"""
|
|
758
|
-
pdbx_file = pdbx.CIFFile.read(pdb_filepath)
|
|
759
|
-
return StructureInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
|
|
760
826
|
|
|
761
827
|
def get_full_sequence(self, chain_id: str, auth_chain_id_supplied: bool=False) -> str:
|
|
762
828
|
"""
|
|
@@ -801,7 +867,7 @@ class StructureInformation:
|
|
|
801
867
|
else:
|
|
802
868
|
return self.non_missing_sequences[chain_id]
|
|
803
869
|
|
|
804
|
-
def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
|
|
870
|
+
def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True, auth_chain_id_supplied: bool=False) -> tuple:
|
|
805
871
|
"""
|
|
806
872
|
Subsets structure attribute to select for chain specific portions of the structure.
|
|
807
873
|
|
|
@@ -815,12 +881,18 @@ class StructureInformation:
|
|
|
815
881
|
Chain id corresponding to the second column of residues in the structure.
|
|
816
882
|
remove_hetero : bool, default=True
|
|
817
883
|
If true, the structure will also be subsetted for atom entries where the hetero annotation is False, thus removing heteroatoms.
|
|
884
|
+
auth_chain_id_supplied : bool
|
|
885
|
+
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
818
886
|
|
|
819
887
|
Returns
|
|
820
888
|
-------
|
|
821
889
|
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray
|
|
822
890
|
Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
|
|
823
891
|
"""
|
|
892
|
+
if auth_chain_id_supplied:
|
|
893
|
+
chain1 = self.auth_chain_dict[chain1]
|
|
894
|
+
chain2 = self.auth_chain_dict[chain2]
|
|
895
|
+
|
|
824
896
|
selected_structure = self.structure
|
|
825
897
|
if remove_hetero:
|
|
826
898
|
# Remove hetero atoms via hetero column of structure ndarray
|
|
@@ -832,7 +904,7 @@ class StructureInformation:
|
|
|
832
904
|
chain2_structure = selected_structure[selected_structure.chain_id == chain2]
|
|
833
905
|
return (chain1_structure, chain2_structure)
|
|
834
906
|
|
|
835
|
-
def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str):
|
|
907
|
+
def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str, auth_chain_id_supplied: bool=False):
|
|
836
908
|
"""
|
|
837
909
|
Generates distance matrix between two chains in the structure attribute.
|
|
838
910
|
|
|
@@ -844,32 +916,40 @@ class StructureInformation:
|
|
|
844
916
|
Chain id corresponding to the first column of residues in the structure.
|
|
845
917
|
chain2 : str
|
|
846
918
|
Chain id corresponding to the first column of residues in the structure.
|
|
919
|
+
auth_chain_id_supplied : bool
|
|
920
|
+
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
847
921
|
|
|
848
922
|
Returns
|
|
849
923
|
-------
|
|
850
924
|
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
|
|
851
925
|
Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
|
|
852
926
|
"""
|
|
853
|
-
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True)
|
|
927
|
+
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
854
928
|
dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
|
|
855
929
|
return (chain1_structure, chain2_structure, dist_matrix)
|
|
856
930
|
|
|
857
|
-
def get_shift_values(self, chain1: str, chain2: str) -> tuple[int, int]:
|
|
931
|
+
def get_shift_values(self, chain1: str, chain2: str, auth_chain_id_supplied: bool=False) -> tuple[int, int]:
|
|
858
932
|
"""
|
|
859
933
|
Get shift values needed for production of auth residue ids.
|
|
860
934
|
|
|
861
935
|
Parameters
|
|
862
936
|
----------
|
|
863
937
|
chain1 : str
|
|
864
|
-
Name of the chain id present
|
|
938
|
+
Name of the chain id present referring to the second column of residues.
|
|
865
939
|
chain2 : str
|
|
866
|
-
Name of the chain id present
|
|
940
|
+
Name of the chain id present referring to the second column of residues.
|
|
941
|
+
auth_chain_id_supplied : bool
|
|
942
|
+
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
867
943
|
|
|
868
944
|
Returns
|
|
869
945
|
-------
|
|
870
946
|
(shift1, shift2) : tuple of int, int
|
|
871
947
|
Tuple containing both shift values, the difference between the auth_res_id and res_id.
|
|
872
948
|
"""
|
|
949
|
+
if auth_chain_id_supplied:
|
|
950
|
+
chain1 = self.auth_chain_dict[chain1]
|
|
951
|
+
chain2 = self.auth_chain_dict[chain2]
|
|
952
|
+
|
|
873
953
|
shift1 = 0
|
|
874
954
|
shift2 = 0
|
|
875
955
|
if self.atom_site_category:
|
|
@@ -879,7 +959,7 @@ class StructureInformation:
|
|
|
879
959
|
else:
|
|
880
960
|
return shift1, shift2
|
|
881
961
|
|
|
882
|
-
def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
|
|
962
|
+
def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str, auth_chain_id_supplied: bool=False) -> npt.NDArray:
|
|
883
963
|
"""
|
|
884
964
|
Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
|
|
885
965
|
|
|
@@ -891,14 +971,16 @@ class StructureInformation:
|
|
|
891
971
|
Chain id corresponding to the first column of residues in the structure.
|
|
892
972
|
chain2 : str
|
|
893
973
|
Chain id corresponding to the second column of residues in the structure.
|
|
894
|
-
|
|
974
|
+
auth_chain_id_supplied : bool
|
|
975
|
+
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
976
|
+
|
|
895
977
|
Returns
|
|
896
978
|
-------
|
|
897
979
|
min_dist_pairs_atoms_arr : numpy.ndarray
|
|
898
|
-
Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,
|
|
980
|
+
Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']}
|
|
899
981
|
"""
|
|
900
|
-
shift1, shift2 = self.get_shift_values(chain1, chain2)
|
|
901
|
-
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True)
|
|
982
|
+
shift1, shift2 = self.get_shift_values(chain1, chain2, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
983
|
+
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
902
984
|
min_dist_pairs_atoms = []
|
|
903
985
|
for row in pairs:
|
|
904
986
|
# Obtain structure information for chains 1 and 2
|
|
@@ -917,7 +999,7 @@ class StructureInformation:
|
|
|
917
999
|
min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
|
|
918
1000
|
return min_dist_pairs_atoms_arr
|
|
919
1001
|
|
|
920
|
-
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_contacts: bool=False) -> set[tuple[int, int]]:
|
|
1002
|
+
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_contacts: bool=False, auth_chain_id_supplied: bool=False) -> set[tuple[int, int]]:
|
|
921
1003
|
"""
|
|
922
1004
|
Get contacts from the structure attribute where the distance between two residues is less than the threshold.
|
|
923
1005
|
|
|
@@ -932,15 +1014,17 @@ class StructureInformation:
|
|
|
932
1014
|
chain2 : str
|
|
933
1015
|
Chain id corresponding to the second column of residues in the structure.
|
|
934
1016
|
auth_contacts : bool
|
|
935
|
-
True if
|
|
1017
|
+
True if you want alt_ids for residues indices, False if cif residue indexing is needed.
|
|
1018
|
+
auth_chain_id_supplied : bool
|
|
1019
|
+
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
936
1020
|
|
|
937
1021
|
Returns
|
|
938
1022
|
-------
|
|
939
1023
|
contacts_set : set of tuple of ints
|
|
940
1024
|
Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
|
|
941
1025
|
"""
|
|
942
|
-
|
|
943
|
-
chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2)
|
|
1026
|
+
|
|
1027
|
+
chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
944
1028
|
thresh_ind = np.argwhere(dist_matrix <= threshold)
|
|
945
1029
|
contacts_set = set()
|
|
946
1030
|
for indices in thresh_ind:
|
|
@@ -950,28 +1034,208 @@ class StructureInformation:
|
|
|
950
1034
|
res2 = chain2_atom.res_id
|
|
951
1035
|
if not(chain1==chain2 and res1 >= res2):
|
|
952
1036
|
if auth_contacts:
|
|
1037
|
+
shift1, shift2 = self.get_shift_values(chain1, chain2, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
953
1038
|
contacts_set.add((res1 + shift1, res2 + shift2))
|
|
954
1039
|
else:
|
|
955
1040
|
contacts_set.add((res1, res2))
|
|
956
1041
|
return contacts_set
|
|
1042
|
+
|
|
1043
|
+
class PDBInformation(StructureInformation):
|
|
1044
|
+
"""
|
|
1045
|
+
Information regarding a protein structure, obtained from a protein structure file.
|
|
1046
|
+
|
|
1047
|
+
Parameters
|
|
1048
|
+
----------
|
|
1049
|
+
structure : biotite.structure.AtomArray
|
|
1050
|
+
Structure obtained from an RCSB entry with a provided pdbx/mmcif file with a specified model number.
|
|
1051
|
+
pdb_file : biotite.io.pdb.PDBFile
|
|
1052
|
+
mmCIF file that contains generic information and atomic information of the protein structure categorized into mmCIF blocks.
|
|
1053
|
+
model_num : int
|
|
1054
|
+
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
1055
|
+
|
|
1056
|
+
Attributes
|
|
1057
|
+
----------
|
|
1058
|
+
self.non_missing_sequences : dict of str, biotite.sequence.ProteinSequence
|
|
1059
|
+
The protein sequences, without missing residues, compiled in the structure of the StructureInformation instance stored in a dictionary where chain_id is the key and the sequence string is the value.
|
|
957
1060
|
|
|
958
|
-
|
|
959
|
-
def
|
|
1061
|
+
"""
|
|
1062
|
+
def __init__(self, structure, pdb_file: pdb.PDBFile, model_num: int):
|
|
1063
|
+
self.structure = structure
|
|
1064
|
+
self.pdb_file = pdb_file
|
|
1065
|
+
self.model_num = model_num
|
|
1066
|
+
non_hetero_structure = self.structure[self.structure.hetero == False]
|
|
1067
|
+
self.non_missing_sequences = {str(chain): str(sequence) for (chain, sequence) in list(zip(struc.get_chains(non_hetero_structure), struc.to_sequence(non_hetero_structure)[0]))}
|
|
1068
|
+
self.unique_chains = struc.get_chains(non_hetero_structure)
|
|
1069
|
+
|
|
1070
|
+
def get_non_missing_sequence(self, chain_id: str) -> str:
|
|
960
1071
|
"""
|
|
961
|
-
|
|
1072
|
+
Get sequence, including only non-missing residues, from the specified chain.
|
|
962
1073
|
|
|
963
1074
|
Parameters
|
|
964
1075
|
----------
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
1076
|
+
chain_id : str
|
|
1077
|
+
Chain id supplied. The full sequence, including only non-missing residues, of this chain will be returned.
|
|
1078
|
+
|
|
1079
|
+
Returns
|
|
1080
|
+
-------
|
|
1081
|
+
str
|
|
1082
|
+
The full sequence, including only non-missing residues, of the chain specified.
|
|
1083
|
+
"""
|
|
1084
|
+
return self.non_missing_sequences[chain_id]
|
|
1085
|
+
|
|
1086
|
+
def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
|
|
1087
|
+
"""
|
|
1088
|
+
Subsets structure attribute to select for chain specific portions of the structure.
|
|
1089
|
+
|
|
1090
|
+
Parameters
|
|
1091
|
+
----------
|
|
1092
|
+
ca_only : bool
|
|
1093
|
+
If true, the structure will also be subsetted for atom entries where the atom_name annotation is "CA" (referring to alpha-carbons)
|
|
1094
|
+
chain1 : str
|
|
1095
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
1096
|
+
chain2 : str
|
|
1097
|
+
Chain id corresponding to the second column of residues in the structure.
|
|
1098
|
+
remove_hetero : bool, default=True
|
|
1099
|
+
If true, the structure will also be subsetted for atom entries where the hetero annotation is False, thus removing heteroatoms.
|
|
1100
|
+
|
|
1101
|
+
Returns
|
|
1102
|
+
-------
|
|
1103
|
+
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray
|
|
1104
|
+
Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
|
|
1105
|
+
"""
|
|
1106
|
+
|
|
1107
|
+
selected_structure = self.structure
|
|
1108
|
+
if remove_hetero:
|
|
1109
|
+
# Remove hetero atoms via hetero column of structure ndarray
|
|
1110
|
+
selected_structure = self.structure[self.structure.hetero == False]
|
|
1111
|
+
if ca_only:
|
|
1112
|
+
# Consider selection of alpha-carbon atoms only
|
|
1113
|
+
selected_structure = selected_structure[selected_structure.atom_name == "CA"]
|
|
1114
|
+
chain1_structure = selected_structure[selected_structure.chain_id == chain1]
|
|
1115
|
+
chain2_structure = selected_structure[selected_structure.chain_id == chain2]
|
|
1116
|
+
return (chain1_structure, chain2_structure)
|
|
1117
|
+
|
|
1118
|
+
def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str):
|
|
1119
|
+
"""
|
|
1120
|
+
Generates distance matrix between two chains in the structure attribute.
|
|
1121
|
+
|
|
1122
|
+
Parameters
|
|
1123
|
+
----------
|
|
1124
|
+
ca_only : bool
|
|
1125
|
+
If True, only atoms that have the name "CA" are selected in the chains the distance matrix is calculated between.
|
|
1126
|
+
chain1 : str
|
|
1127
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
1128
|
+
chain2 : str
|
|
1129
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
1130
|
+
|
|
1131
|
+
Returns
|
|
1132
|
+
-------
|
|
1133
|
+
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
|
|
1134
|
+
Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
|
|
1135
|
+
"""
|
|
1136
|
+
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True)
|
|
1137
|
+
dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
|
|
1138
|
+
return (chain1_structure, chain2_structure, dist_matrix)
|
|
1139
|
+
|
|
1140
|
+
def get_shift_values(self, chain1: str, chain2: str) -> tuple[int, int]:
|
|
1141
|
+
"""
|
|
1142
|
+
Get shift values needed for production of auth residue ids.
|
|
969
1143
|
|
|
1144
|
+
Parameters
|
|
1145
|
+
----------
|
|
1146
|
+
chain1 : str
|
|
1147
|
+
Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
|
|
1148
|
+
chain2 : str
|
|
1149
|
+
Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
|
|
1150
|
+
|
|
970
1151
|
Returns
|
|
971
1152
|
-------
|
|
972
|
-
|
|
1153
|
+
(shift1, shift2) : tuple of int, int
|
|
1154
|
+
Tuple containing both shift values, the difference between the auth_res_id and res_id.
|
|
973
1155
|
"""
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
1156
|
+
non_hetero_structure = self.structure[self.structure.hetero == False]
|
|
1157
|
+
shift1 = 0
|
|
1158
|
+
shift2 = 0
|
|
1159
|
+
if chain1 in self.unique_chains and chain2 in self.unique_chains:
|
|
1160
|
+
shift1 = non_hetero_structure[non_hetero_structure.chain_id == chain1][0].res_id - 1
|
|
1161
|
+
shift2 = non_hetero_structure[non_hetero_structure.chain_id == chain2][0].res_id - 1
|
|
1162
|
+
return shift1, shift2
|
|
1163
|
+
else:
|
|
1164
|
+
return shift1, shift2
|
|
1165
|
+
|
|
1166
|
+
def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
|
|
1167
|
+
"""
|
|
1168
|
+
Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
|
|
1169
|
+
|
|
1170
|
+
Parameters
|
|
1171
|
+
----------
|
|
1172
|
+
pairs : numpy.ndarray
|
|
1173
|
+
Pairs structured ndarray with "residue1" and "residue2" columns.
|
|
1174
|
+
chain1 : str
|
|
1175
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
1176
|
+
chain2 : str
|
|
1177
|
+
Chain id corresponding to the second column of residues in the structure.
|
|
1178
|
+
|
|
1179
|
+
Returns
|
|
1180
|
+
-------
|
|
1181
|
+
min_dist_pairs_atoms_arr : numpy.ndarray
|
|
1182
|
+
Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']}
|
|
1183
|
+
"""
|
|
1184
|
+
shift1, shift2 = self.get_shift_values(chain1, chain2)
|
|
1185
|
+
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True)
|
|
1186
|
+
min_dist_pairs_atoms = []
|
|
1187
|
+
for row in pairs:
|
|
1188
|
+
# Obtain structure information for chains 1 and 2
|
|
1189
|
+
chain1_res1_structure = chain1_structure[chain1_structure.res_id == row['residue1']]
|
|
1190
|
+
chain2_res2_structure = chain2_structure[chain2_structure.res_id == row['residue2']]
|
|
1191
|
+
|
|
1192
|
+
# Calculate a distance matrix and find the indices of the minimal value in the matrix
|
|
1193
|
+
dist_matrix = cdist(chain1_res1_structure.coord, chain2_res2_structure.coord)
|
|
1194
|
+
|
|
1195
|
+
ind = np.unravel_index(np.argmin(dist_matrix), dist_matrix.shape)
|
|
1196
|
+
# Use the indices to access the atom in the atom array and get the correct atom name.
|
|
1197
|
+
# Generate the auth ids of the residues in the pairs ndarray
|
|
1198
|
+
orig_res_id1 = row['residue1'] - shift1
|
|
1199
|
+
orig_res_id2 = row['residue2'] - shift2
|
|
1200
|
+
min_dist_pairs_atoms.append((orig_res_id1, orig_res_id2, row['residue1'], row['residue2'], chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
|
|
1201
|
+
min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
|
|
1202
|
+
return min_dist_pairs_atoms_arr
|
|
1203
|
+
|
|
1204
|
+
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_contacts: bool=False) -> set[tuple[int, int]]:
|
|
1205
|
+
"""
|
|
1206
|
+
Get contacts from the structure attribute where the distance between two residues is less than the threshold.
|
|
1207
|
+
|
|
1208
|
+
Parameters
|
|
1209
|
+
----------
|
|
1210
|
+
ca_only : bool
|
|
1211
|
+
If true, only consider alpha-carbon to alpha-carbon distances.
|
|
1212
|
+
threshold : float
|
|
1213
|
+
Maximum distance to consider between two atoms.
|
|
1214
|
+
chain1 : str
|
|
1215
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
1216
|
+
chain2 : str
|
|
1217
|
+
Chain id corresponding to the second column of residues in the structure.
|
|
1218
|
+
auth_contacts : bool
|
|
1219
|
+
True if you want alt_ids for residues indices, False if cif residue indexing is needed.
|
|
1220
|
+
|
|
1221
|
+
Returns
|
|
1222
|
+
-------
|
|
1223
|
+
contacts_set : set of tuple of ints
|
|
1224
|
+
Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
|
|
1225
|
+
"""
|
|
1226
|
+
|
|
1227
|
+
chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2)
|
|
1228
|
+
thresh_ind = np.argwhere(dist_matrix <= threshold)
|
|
1229
|
+
contacts_set = set()
|
|
1230
|
+
for indices in thresh_ind:
|
|
1231
|
+
chain1_atom = chain1_structure[indices[0]]
|
|
1232
|
+
chain2_atom = chain2_structure[indices[1]]
|
|
1233
|
+
res1 = chain1_atom.res_id
|
|
1234
|
+
res2 = chain2_atom.res_id
|
|
1235
|
+
if not(chain1==chain2 and res1 >= res2):
|
|
1236
|
+
if not auth_contacts:
|
|
1237
|
+
shift1, shift2 = self.get_shift_values(chain1, chain2)
|
|
1238
|
+
contacts_set.add((res1 - shift1, res2 - shift2))
|
|
1239
|
+
else:
|
|
1240
|
+
contacts_set.add((res1, res2))
|
|
1241
|
+
return contacts_set
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
@@ -38,7 +38,7 @@ Classifier: Programming Language :: Python :: 3.12
|
|
|
38
38
|
Requires-Python: >=3.10
|
|
39
39
|
Description-Content-Type: text/markdown
|
|
40
40
|
License-File: LICENSE
|
|
41
|
-
Requires-Dist: biotite
|
|
41
|
+
Requires-Dist: biotite>=1.0.1
|
|
42
42
|
Requires-Dist: matplotlib>=3.8.0
|
|
43
43
|
Requires-Dist: numpy>=1.26.0
|
|
44
44
|
Requires-Dist: pandas>=2.1.0
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
from context import MMCIFInformation, PDBInformation
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
import biotite.structure.io.pdbx as pdbx
|
|
4
|
+
import biotite.database.rcsb as rcsb
|
|
5
|
+
|
|
6
|
+
pdb_id_chain_map = {}
|
|
7
|
+
with open("tests/pdb_info/pdb_ids.txt", "r") as fs:
|
|
8
|
+
pdb_id_data = fs.read().splitlines()
|
|
9
|
+
for line in pdb_id_data:
|
|
10
|
+
pdb_id, chain1, chain2 = line.split()
|
|
11
|
+
if len(chain1.split("/")) > 1:
|
|
12
|
+
chain1, auth_chain1 = chain1.split("/")
|
|
13
|
+
else:
|
|
14
|
+
auth_chain1 = chain1
|
|
15
|
+
if len(chain2.split("/")) > 1:
|
|
16
|
+
chain2, auth_chain2 = chain2.split("/")
|
|
17
|
+
else:
|
|
18
|
+
auth_chain2 = chain2
|
|
19
|
+
pdb_id_chain_map[pdb_id] = (chain1, auth_chain1, chain2, auth_chain2)
|
|
20
|
+
|
|
21
|
+
def read_contacts(input_filepath: str) -> set[tuple[int, int]]:
|
|
22
|
+
with open(input_filepath, 'r') as fs:
|
|
23
|
+
data = fs.read().splitlines()
|
|
24
|
+
results = set()
|
|
25
|
+
for pair in data:
|
|
26
|
+
res1, res2 = pair.split()
|
|
27
|
+
res1 = int(res1)
|
|
28
|
+
res2 = int(res2)
|
|
29
|
+
if res1 < res2:
|
|
30
|
+
results.add((int(res1), int(res2)))
|
|
31
|
+
return results
|
|
32
|
+
#return set([(int(x.split()[0]), int(x.split()[1])) for x in data])
|
|
33
|
+
|
|
34
|
+
def drop_inord_res(contacts: set[tuple[int, int]]) -> set[tuple[int, int]]:
|
|
35
|
+
results = set()
|
|
36
|
+
for contact in contacts:
|
|
37
|
+
if contact[0] < contact[1]:
|
|
38
|
+
results.add(contact)
|
|
39
|
+
return results
|
|
40
|
+
|
|
41
|
+
def compare(*args) -> None:
|
|
42
|
+
for arg1 in args:
|
|
43
|
+
for arg2 in args:
|
|
44
|
+
assert arg1 == arg2
|
|
45
|
+
|
|
46
|
+
def check_contacts(test_CA: bool, threshold: float):
|
|
47
|
+
cif_files = list(Path("tests/pdb_info").glob("*.cif"))
|
|
48
|
+
if test_CA:
|
|
49
|
+
print("Testing CA")
|
|
50
|
+
else:
|
|
51
|
+
print("Testing AA")
|
|
52
|
+
|
|
53
|
+
for cif_file in cif_files:
|
|
54
|
+
pdb_id = cif_file.stem.upper()
|
|
55
|
+
if test_CA:
|
|
56
|
+
corresponding_file = f"tests/pdb_info/monomer_calpha_{pdb_id}_10.0"
|
|
57
|
+
else:
|
|
58
|
+
corresponding_file = f"tests/pdb_info/monomer_allatom_{pdb_id}_8.0"
|
|
59
|
+
|
|
60
|
+
cif_file_contacts = read_contacts(corresponding_file)
|
|
61
|
+
chain1, auth_chain1, chain2, auth_chain2 = pdb_id_chain_map[pdb_id]
|
|
62
|
+
fetch_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id).get_contacts(test_CA, threshold, chain1, chain2, auth_contacts=True)}
|
|
63
|
+
read_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, chain1, chain2, auth_contacts=True)}
|
|
64
|
+
fetch_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id).get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True, auth_chain_id_supplied=True)}
|
|
65
|
+
read_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True, auth_chain_id_supplied=True)}
|
|
66
|
+
fetch_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.fetch_pdb(pdb_id, struc_format="pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True)}
|
|
67
|
+
read_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.read_pdb_file(f"tests/pdb_info/{pdb_id.lower()}.pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True)}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
fetch_cif_contacts = drop_inord_res(fetch_cif_contacts)
|
|
71
|
+
fetch_authchain_cif_contacts = drop_inord_res(fetch_authchain_cif_contacts)
|
|
72
|
+
read_cif_contacts = drop_inord_res(read_cif_contacts)
|
|
73
|
+
read_authchain_cif_contacts = drop_inord_res(read_authchain_cif_contacts)
|
|
74
|
+
fetch_pdb_contacts = drop_inord_res(fetch_pdb_contacts)
|
|
75
|
+
read_pdb_contacts = drop_inord_res(read_pdb_contacts)
|
|
76
|
+
|
|
77
|
+
if pdbx.get_model_count(pdbx.CIFFile.read(rcsb.fetch(pdb_id, format="mmcif"))) <= 1:
|
|
78
|
+
compare(fetch_cif_contacts, fetch_pdb_contacts, read_cif_contacts, read_pdb_contacts)
|
|
79
|
+
print(f"{pdb_id} has no difference between cif and pdb reading.")
|
|
80
|
+
|
|
81
|
+
compare(fetch_cif_contacts, fetch_authchain_cif_contacts, read_cif_contacts, read_authchain_cif_contacts)
|
|
82
|
+
print(f"{pdb_id} has no issue reading with auth chains and asym chains.")
|
|
83
|
+
|
|
84
|
+
compare(fetch_cif_contacts, cif_file_contacts)
|
|
85
|
+
print(f"{pdb_id} has no difference between cif and interface_contacts_X.py reading.")
|
|
86
|
+
else:
|
|
87
|
+
compare(fetch_cif_contacts, fetch_pdb_contacts, read_cif_contacts, read_pdb_contacts)
|
|
88
|
+
print(f"{pdb_id} has no difference between cif and pdb reading.")
|
|
89
|
+
|
|
90
|
+
compare(fetch_cif_contacts, fetch_authchain_cif_contacts, read_cif_contacts, read_authchain_cif_contacts)
|
|
91
|
+
print(f"{pdb_id} has no issue reading with auth chains and asym chains.")
|
|
92
|
+
|
|
93
|
+
def test_contacts():
|
|
94
|
+
check_contacts(test_CA=True, threshold=10)
|
|
95
|
+
check_contacts(test_CA=False, threshold=8)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|