dcatoolkit 0.1.8__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: dcatoolkit
3
- Version: 0.1.8
3
+ Version: 0.2.0
4
4
  Summary: Collection of useful modules and representations for managing DCA output data.
5
5
  Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
6
6
  Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
@@ -38,7 +38,7 @@ Classifier: Programming Language :: Python :: 3.12
38
38
  Requires-Python: >=3.10
39
39
  Description-Content-Type: text/markdown
40
40
  License-File: LICENSE
41
- Requires-Dist: biotite
41
+ Requires-Dist: biotite>=1.0.1
42
42
  Requires-Dist: matplotlib>=3.8.0
43
43
  Requires-Dist: numpy>=1.26.0
44
44
  Requires-Dist: pandas>=2.1.0
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dcatoolkit"
7
- version = "0.1.8"
7
+ version = "0.2.0"
8
8
  description = "Collection of useful modules and representations for managing DCA output data."
9
9
  keywords = ["dca", "toolkit", "DI", "coevolution"]
10
10
 
@@ -21,7 +21,7 @@ maintainers = [
21
21
  ]
22
22
 
23
23
  dependencies = [
24
- "biotite",
24
+ "biotite>=1.0.1",
25
25
  "matplotlib>=3.8.0",
26
26
  "numpy>=1.26.0",
27
27
  "pandas>=2.1.0",
@@ -0,0 +1,6 @@
1
+
2
+ __version__ = "0.1.9"
3
+ from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment, MMCIFInformation, PDBInformation
4
+ from .analytics import MSATools
5
+
6
+ __all__ = ['Pairs', 'DirectInformationData', 'StructureInformation', 'ResidueAlignment', 'MMCIFInformation', 'PDBInformation', 'MSATools']
@@ -1,7 +1,8 @@
1
1
  import re
2
2
  from collections import Counter
3
3
  from typing import Optional, Union
4
- import string, io
4
+ import string
5
+ import io
5
6
 
6
7
  class MSATools:
7
8
  """
@@ -107,7 +108,7 @@ class MSATools:
107
108
  List of entries that are valid in that their sequences' number of maximum continuous gaps is within the threshold supplied as `max_gaps`.
108
109
  """
109
110
  table = str.maketrans('', '', string.ascii_lowercase+".")
110
- if max_gaps == None:
111
+ if max_gaps is None:
111
112
  kept_entries = []
112
113
  for header, sequence in self.MSA:
113
114
  sequence = sequence.translate(table)
@@ -4,10 +4,11 @@ from scipy.spatial.distance import cdist
4
4
 
5
5
  import biotite.structure as struc
6
6
  import biotite.structure.io.pdbx as pdbx
7
+ import biotite.structure.io.pdb as pdb
7
8
  import biotite.database.rcsb as rcsb
8
9
 
9
10
  from collections.abc import Iterable
10
- from typing import Optional, Union
11
+ from typing import Optional, Union, Literal, overload
11
12
  import numpy.typing as npt
12
13
 
13
14
 
@@ -619,9 +620,124 @@ class StructureInformation:
619
620
  """
620
621
  Information regarding a protein structure, obtained from a protein structure file.
621
622
 
623
+ Uses fetch_pdb() to pull protein structure information from RCSB. Uses read_x_file() to supply a filepath to pull protein structure information from a file.
624
+ """
625
+ @overload
626
+ @staticmethod
627
+ def fetch_pdb(pdb_id: str, struc_format: Literal["mmcif"]="mmcif", model_num: int=1) -> 'MMCIFInformation':
628
+ ...
629
+
630
+ @overload
631
+ @staticmethod
632
+ def fetch_pdb(pdb_id: str, struc_format: Literal["pdb"], model_num: int=1) -> 'PDBInformation':
633
+ ...
634
+
635
+
636
+ @staticmethod
637
+ def fetch_pdb(pdb_id: str, struc_format: Literal["mmcif", "pdb"]="mmcif", model_num: int=1) -> Union['MMCIFInformation', 'PDBInformation']:
638
+ """
639
+ Fetches PDB as mmCIF file from RCSB and compiles the information into a StructureInformation instance.
640
+
641
+ Parameters
642
+ ----------
643
+ pdb_id : str
644
+ PDB ID to be fetched from the RCSB database.
645
+ struc_format : str
646
+ The format of the file to pull from the RCSB database.
647
+ model_num : int
648
+ The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
649
+
650
+ Returns
651
+ -------
652
+ StructureInformation
653
+ StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB.
654
+
655
+ Raises
656
+ ------
657
+ TypeError
658
+ Fetched data was not found and returned None instead.
659
+ ValueError
660
+ Structure format may be invalid (not PDBx/mmCIF or PDB).
661
+ """
662
+ fetched_data = rcsb.fetch(pdb_id, struc_format)
663
+ if fetched_data is None:
664
+ raise TypeError("RCSB fetch failed. Try fetch again.")
665
+ elif struc_format == "mmcif":
666
+ pdbx_file = pdbx.CIFFile.read(fetched_data)
667
+ return MMCIFInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
668
+ elif struc_format == "pdb":
669
+ pdb_file = pdb.PDBFile.read(fetched_data)
670
+ return PDBInformation(pdb.get_structure(pdb_file=pdb_file, model=model_num), pdb_file=pdb_file, model_num=model_num)
671
+ else:
672
+ raise ValueError(f"struc_format {struc_format} is not valid or currently supported by DCA Toolkit")
673
+ @staticmethod
674
+ def read_mmCIF_file(pdbx_filepath: str, model_num: int=1) -> 'MMCIFInformation':
675
+ """
676
+ Reads PDB mmCIF file from filepath and compiles the information into a CIFInformation instance.
677
+
678
+ Parameters
679
+ ----------
680
+ pdbx_filepath : str
681
+ Filepath of the PDB mmCIF file to be read.
682
+ model_num : int
683
+ The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
684
+
685
+ Returns
686
+ -------
687
+ CIFInformation
688
+ CIFInformation generated from pdbx.get_structure() function using the PDBx file read from the pdbx_filepath.
689
+ """
690
+ pdbx_file = pdbx.CIFFile.read(pdbx_filepath)
691
+ return MMCIFInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
692
+
693
+ @staticmethod
694
+ def read_pdb_file(pdb_filepath: str, model_num: int=1) -> 'PDBInformation':
695
+ """
696
+ Reads PDB file from filepath and compiles the information into a PDBInformation instance.
697
+
698
+ Parameters
699
+ ----------
700
+ pdb_filepath : str
701
+ Filepath of the PDB mmCIF file to be read.
702
+ model_num : int
703
+ The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
704
+
705
+ Returns
706
+ -------
707
+ PDBInformation
708
+ PDBInformation generated from pdb.get_structure() function using the PDB file read from the pdb_filepath.
709
+ """
710
+ pdb_file = pdb.PDBFile.read(pdb_filepath)
711
+ return PDBInformation(pdb.get_structure(pdb_file, model=model_num), pdb_file, model_num)
712
+
713
+ @staticmethod
714
+ def write_contacts_set(filepath : str, contacts_set : set[tuple[int, int]]) -> None:
715
+ """
716
+ Write the contacts generated from get_contacts or general set of tuples of pairs.
717
+
718
+ Parameters
719
+ ----------
720
+ filepath : str
721
+ Path of file to output contacts_set to.
722
+ contacts_set : set of tuple of int, int
723
+ Set of tuples of pairs that represent contacts.
724
+
725
+ Returns
726
+ -------
727
+ None
728
+ """
729
+ contacts_list = list(sorted(contacts_set))
730
+ with open(filepath, 'w') as fs:
731
+ for pair in contacts_list:
732
+ fs.write(str(pair[0]) + "\t" + str(pair[1]) + "\n")
733
+
734
+ class MMCIFInformation(StructureInformation):
735
+ """
736
+ Information regarding a protein structure, obtained from a protein structure file.
737
+
622
738
  Parameters
623
739
  ----------
624
- structure : biotite.structure
740
+ structure : biotite.structure.AtomArray
625
741
  Structure obtained from an RCSB entry with a provided pdbx/mmcif file with a specified model number.
626
742
  pdbx_file : biotite.io.pdbx.CIFFile
627
743
  mmCIF file that contains generic information and atomic information of the protein structure categorized into mmCIF blocks.
@@ -707,56 +823,6 @@ class StructureInformation:
707
823
  self.res_auth_dict[unique_entry[2]] = unique_entry[[1,3]].astype('int')
708
824
  else:
709
825
  self.atom_site_category = None
710
-
711
- @staticmethod
712
- def fetch_pdb(pdb_id: str, model_num: int=1, struc_format: str="mmcif") -> 'StructureInformation':
713
- """
714
- Fetches PDB as mmCIF file from RCSB and compiles the information into a StructureInformation instance.
715
-
716
- Parameters
717
- ----------
718
- pdb_id : str
719
- PDB ID to be fetched from the RCSB database.
720
- model_num : int
721
- The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
722
- struc_format : str
723
- The format of the file to pull from the RCSB database.
724
-
725
- Returns
726
- -------
727
- StructureInformation
728
- StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
729
-
730
- Raises
731
- ------
732
- TypeError
733
- Fetched data was not found and returned None instead.
734
- """
735
- fetched_data = rcsb.fetch(pdb_id, struc_format)
736
- if fetched_data is None:
737
- raise TypeError("RCSB fetch failed. Try fetch again.")
738
- pdbx_file = pdbx.CIFFile.read(fetched_data)
739
- return StructureInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
740
-
741
- @staticmethod
742
- def read_pdb_mmCIF(pdb_filepath: str, model_num: int=1) -> 'StructureInformation':
743
- """
744
- Reads PDB mmCIF file from filepath and compiles the information into a StructureInformation instance.
745
-
746
- Parameters
747
- ----------
748
- pdb_filepath : str
749
- Filepath of the PDB mmCIF file to be read.
750
- model_num : int
751
- The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
752
-
753
- Returns
754
- -------
755
- StructureInformation
756
- StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
757
- """
758
- pdbx_file = pdbx.CIFFile.read(pdb_filepath)
759
- return StructureInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file, model_num)
760
826
 
761
827
  def get_full_sequence(self, chain_id: str, auth_chain_id_supplied: bool=False) -> str:
762
828
  """
@@ -801,7 +867,7 @@ class StructureInformation:
801
867
  else:
802
868
  return self.non_missing_sequences[chain_id]
803
869
 
804
- def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
870
+ def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True, auth_chain_id_supplied: bool=False) -> tuple:
805
871
  """
806
872
  Subsets structure attribute to select for chain specific portions of the structure.
807
873
 
@@ -815,12 +881,18 @@ class StructureInformation:
815
881
  Chain id corresponding to the second column of residues in the structure.
816
882
  remove_hetero : bool, default=True
817
883
  If true, the structure will also be subsetted for atom entries where the hetero annotation is False, thus removing heteroatoms.
884
+ auth_chain_id_supplied : bool
885
+ If True, the chain_id supplied is the auth chain id found on the RCSB website.
818
886
 
819
887
  Returns
820
888
  -------
821
889
  tuple of biotite.structure.AtomArray, biotite.structure.AtomArray
822
890
  Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
823
891
  """
892
+ if auth_chain_id_supplied:
893
+ chain1 = self.auth_chain_dict[chain1]
894
+ chain2 = self.auth_chain_dict[chain2]
895
+
824
896
  selected_structure = self.structure
825
897
  if remove_hetero:
826
898
  # Remove hetero atoms via hetero column of structure ndarray
@@ -832,7 +904,7 @@ class StructureInformation:
832
904
  chain2_structure = selected_structure[selected_structure.chain_id == chain2]
833
905
  return (chain1_structure, chain2_structure)
834
906
 
835
- def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str):
907
+ def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str, auth_chain_id_supplied: bool=False):
836
908
  """
837
909
  Generates distance matrix between two chains in the structure attribute.
838
910
 
@@ -844,32 +916,40 @@ class StructureInformation:
844
916
  Chain id corresponding to the first column of residues in the structure.
845
917
  chain2 : str
846
918
  Chain id corresponding to the first column of residues in the structure.
919
+ auth_chain_id_supplied : bool
920
+ If True, the chain_id supplied is the auth chain id found on the RCSB website.
847
921
 
848
922
  Returns
849
923
  -------
850
924
  tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
851
925
  Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
852
926
  """
853
- chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True)
927
+ chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
854
928
  dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
855
929
  return (chain1_structure, chain2_structure, dist_matrix)
856
930
 
857
- def get_shift_values(self, chain1: str, chain2: str) -> tuple[int, int]:
931
+ def get_shift_values(self, chain1: str, chain2: str, auth_chain_id_supplied: bool=False) -> tuple[int, int]:
858
932
  """
859
933
  Get shift values needed for production of auth residue ids.
860
934
 
861
935
  Parameters
862
936
  ----------
863
937
  chain1 : str
864
- Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
938
+ Name of the chain id present referring to the second column of residues.
865
939
  chain2 : str
866
- Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
940
+ Name of the chain id present referring to the second column of residues.
941
+ auth_chain_id_supplied : bool
942
+ If True, the chain_id supplied is the auth chain id found on the RCSB website.
867
943
 
868
944
  Returns
869
945
  -------
870
946
  (shift1, shift2) : tuple of int, int
871
947
  Tuple containing both shift values, the difference between the auth_res_id and res_id.
872
948
  """
949
+ if auth_chain_id_supplied:
950
+ chain1 = self.auth_chain_dict[chain1]
951
+ chain2 = self.auth_chain_dict[chain2]
952
+
873
953
  shift1 = 0
874
954
  shift2 = 0
875
955
  if self.atom_site_category:
@@ -879,7 +959,7 @@ class StructureInformation:
879
959
  else:
880
960
  return shift1, shift2
881
961
 
882
- def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
962
+ def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str, auth_chain_id_supplied: bool=False) -> npt.NDArray:
883
963
  """
884
964
  Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
885
965
 
@@ -891,14 +971,16 @@ class StructureInformation:
891
971
  Chain id corresponding to the first column of residues in the structure.
892
972
  chain2 : str
893
973
  Chain id corresponding to the second column of residues in the structure.
894
-
974
+ auth_chain_id_supplied : bool
975
+ If True, the chain_id supplied is the auth chain id found on the RCSB website.
976
+
895
977
  Returns
896
978
  -------
897
979
  min_dist_pairs_atoms_arr : numpy.ndarray
898
- Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,str,str]}
980
+ Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']}
899
981
  """
900
- shift1, shift2 = self.get_shift_values(chain1, chain2)
901
- chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True)
982
+ shift1, shift2 = self.get_shift_values(chain1, chain2, auth_chain_id_supplied=auth_chain_id_supplied)
983
+ chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
902
984
  min_dist_pairs_atoms = []
903
985
  for row in pairs:
904
986
  # Obtain structure information for chains 1 and 2
@@ -917,7 +999,7 @@ class StructureInformation:
917
999
  min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
918
1000
  return min_dist_pairs_atoms_arr
919
1001
 
920
- def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_contacts: bool=False) -> set[tuple[int, int]]:
1002
+ def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_contacts: bool=False, auth_chain_id_supplied: bool=False) -> set[tuple[int, int]]:
921
1003
  """
922
1004
  Get contacts from the structure attribute where the distance between two residues is less than the threshold.
923
1005
 
@@ -932,15 +1014,17 @@ class StructureInformation:
932
1014
  chain2 : str
933
1015
  Chain id corresponding to the second column of residues in the structure.
934
1016
  auth_contacts : bool
935
- True if supplying alt_ids for residues indices, False if cif residue indexing is needed.
1017
+ True if you want alt_ids for residues indices, False if cif residue indexing is needed.
1018
+ auth_chain_id_supplied : bool
1019
+ If True, the chain_id supplied is the auth chain id found on the RCSB website.
936
1020
 
937
1021
  Returns
938
1022
  -------
939
1023
  contacts_set : set of tuple of ints
940
1024
  Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
941
1025
  """
942
- shift1, shift2 = self.get_shift_values(chain1, chain2)
943
- chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2)
1026
+
1027
+ chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2, auth_chain_id_supplied=auth_chain_id_supplied)
944
1028
  thresh_ind = np.argwhere(dist_matrix <= threshold)
945
1029
  contacts_set = set()
946
1030
  for indices in thresh_ind:
@@ -950,28 +1034,208 @@ class StructureInformation:
950
1034
  res2 = chain2_atom.res_id
951
1035
  if not(chain1==chain2 and res1 >= res2):
952
1036
  if auth_contacts:
1037
+ shift1, shift2 = self.get_shift_values(chain1, chain2, auth_chain_id_supplied=auth_chain_id_supplied)
953
1038
  contacts_set.add((res1 + shift1, res2 + shift2))
954
1039
  else:
955
1040
  contacts_set.add((res1, res2))
956
1041
  return contacts_set
1042
+
1043
+ class PDBInformation(StructureInformation):
1044
+ """
1045
+ Information regarding a protein structure, obtained from a protein structure file.
1046
+
1047
+ Parameters
1048
+ ----------
1049
+ structure : biotite.structure.AtomArray
1050
+ Structure obtained from an RCSB entry with a provided pdbx/mmcif file with a specified model number.
1051
+ pdb_file : biotite.io.pdb.PDBFile
1052
+ mmCIF file that contains generic information and atomic information of the protein structure categorized into mmCIF blocks.
1053
+ model_num : int
1054
+ The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
1055
+
1056
+ Attributes
1057
+ ----------
1058
+ self.non_missing_sequences : dict of str, biotite.sequence.ProteinSequence
1059
+ The protein sequences, without missing residues, compiled in the structure of the StructureInformation instance stored in a dictionary where chain_id is the key and the sequence string is the value.
957
1060
 
958
- @staticmethod
959
- def write_contacts_set(filepath : str, contacts_set : set[tuple[int, int]]) -> None:
1061
+ """
1062
+ def __init__(self, structure, pdb_file: pdb.PDBFile, model_num: int):
1063
+ self.structure = structure
1064
+ self.pdb_file = pdb_file
1065
+ self.model_num = model_num
1066
+ non_hetero_structure = self.structure[self.structure.hetero == False]
1067
+ self.non_missing_sequences = {str(chain): str(sequence) for (chain, sequence) in list(zip(struc.get_chains(non_hetero_structure), struc.to_sequence(non_hetero_structure)[0]))}
1068
+ self.unique_chains = struc.get_chains(non_hetero_structure)
1069
+
1070
+ def get_non_missing_sequence(self, chain_id: str) -> str:
960
1071
  """
961
- Write the contacts generated from get_contacts or general set of tuples of pairs.
1072
+ Get sequence, including only non-missing residues, from the specified chain.
962
1073
 
963
1074
  Parameters
964
1075
  ----------
965
- filepath : str
966
- Path of file to output contacts_set to.
967
- contacts_set : set of tuple of int, int
968
- Set of tuples of pairs that represent contacts.
1076
+ chain_id : str
1077
+ Chain id supplied. The full sequence, including only non-missing residues, of this chain will be returned.
1078
+
1079
+ Returns
1080
+ -------
1081
+ str
1082
+ The full sequence, including only non-missing residues, of the chain specified.
1083
+ """
1084
+ return self.non_missing_sequences[chain_id]
1085
+
1086
+ def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
1087
+ """
1088
+ Subsets structure attribute to select for chain specific portions of the structure.
1089
+
1090
+ Parameters
1091
+ ----------
1092
+ ca_only : bool
1093
+ If true, the structure will also be subsetted for atom entries where the atom_name annotation is "CA" (referring to alpha-carbons)
1094
+ chain1 : str
1095
+ Chain id corresponding to the first column of residues in the structure.
1096
+ chain2 : str
1097
+ Chain id corresponding to the second column of residues in the structure.
1098
+ remove_hetero : bool, default=True
1099
+ If true, the structure will also be subsetted for atom entries where the hetero annotation is False, thus removing heteroatoms.
1100
+
1101
+ Returns
1102
+ -------
1103
+ tuple of biotite.structure.AtomArray, biotite.structure.AtomArray
1104
+ Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
1105
+ """
1106
+
1107
+ selected_structure = self.structure
1108
+ if remove_hetero:
1109
+ # Remove hetero atoms via hetero column of structure ndarray
1110
+ selected_structure = self.structure[self.structure.hetero == False]
1111
+ if ca_only:
1112
+ # Consider selection of alpha-carbon atoms only
1113
+ selected_structure = selected_structure[selected_structure.atom_name == "CA"]
1114
+ chain1_structure = selected_structure[selected_structure.chain_id == chain1]
1115
+ chain2_structure = selected_structure[selected_structure.chain_id == chain2]
1116
+ return (chain1_structure, chain2_structure)
1117
+
1118
+ def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str):
1119
+ """
1120
+ Generates distance matrix between two chains in the structure attribute.
1121
+
1122
+ Parameters
1123
+ ----------
1124
+ ca_only : bool
1125
+ If True, only atoms that have the name "CA" are selected in the chains the distance matrix is calculated between.
1126
+ chain1 : str
1127
+ Chain id corresponding to the first column of residues in the structure.
1128
+ chain2 : str
1129
+ Chain id corresponding to the first column of residues in the structure.
1130
+
1131
+ Returns
1132
+ -------
1133
+ tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
1134
+ Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
1135
+ """
1136
+ chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True)
1137
+ dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
1138
+ return (chain1_structure, chain2_structure, dist_matrix)
1139
+
1140
+ def get_shift_values(self, chain1: str, chain2: str) -> tuple[int, int]:
1141
+ """
1142
+ Get shift values needed for production of auth residue ids.
969
1143
 
1144
+ Parameters
1145
+ ----------
1146
+ chain1 : str
1147
+ Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
1148
+ chain2 : str
1149
+ Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
1150
+
970
1151
  Returns
971
1152
  -------
972
- None
1153
+ (shift1, shift2) : tuple of int, int
1154
+ Tuple containing both shift values, the difference between the auth_res_id and res_id.
973
1155
  """
974
- contacts_list = list(sorted(contacts_set))
975
- with open(filepath, 'w') as fs:
976
- for pair in contacts_list:
977
- fs.write(str(pair[0]) + "\t" + str(pair[1]) + "\n")
1156
+ non_hetero_structure = self.structure[self.structure.hetero == False]
1157
+ shift1 = 0
1158
+ shift2 = 0
1159
+ if chain1 in self.unique_chains and chain2 in self.unique_chains:
1160
+ shift1 = non_hetero_structure[non_hetero_structure.chain_id == chain1][0].res_id - 1
1161
+ shift2 = non_hetero_structure[non_hetero_structure.chain_id == chain2][0].res_id - 1
1162
+ return shift1, shift2
1163
+ else:
1164
+ return shift1, shift2
1165
+
1166
+ def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
1167
+ """
1168
+ Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
1169
+
1170
+ Parameters
1171
+ ----------
1172
+ pairs : numpy.ndarray
1173
+ Pairs structured ndarray with "residue1" and "residue2" columns.
1174
+ chain1 : str
1175
+ Chain id corresponding to the first column of residues in the structure.
1176
+ chain2 : str
1177
+ Chain id corresponding to the second column of residues in the structure.
1178
+
1179
+ Returns
1180
+ -------
1181
+ min_dist_pairs_atoms_arr : numpy.ndarray
1182
+ Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']}
1183
+ """
1184
+ shift1, shift2 = self.get_shift_values(chain1, chain2)
1185
+ chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True)
1186
+ min_dist_pairs_atoms = []
1187
+ for row in pairs:
1188
+ # Obtain structure information for chains 1 and 2
1189
+ chain1_res1_structure = chain1_structure[chain1_structure.res_id == row['residue1']]
1190
+ chain2_res2_structure = chain2_structure[chain2_structure.res_id == row['residue2']]
1191
+
1192
+ # Calculate a distance matrix and find the indices of the minimal value in the matrix
1193
+ dist_matrix = cdist(chain1_res1_structure.coord, chain2_res2_structure.coord)
1194
+
1195
+ ind = np.unravel_index(np.argmin(dist_matrix), dist_matrix.shape)
1196
+ # Use the indices to access the atom in the atom array and get the correct atom name.
1197
+ # Generate the auth ids of the residues in the pairs ndarray
1198
+ orig_res_id1 = row['residue1'] - shift1
1199
+ orig_res_id2 = row['residue2'] - shift2
1200
+ min_dist_pairs_atoms.append((orig_res_id1, orig_res_id2, row['residue1'], row['residue2'], chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
1201
+ min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
1202
+ return min_dist_pairs_atoms_arr
1203
+
1204
+ def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_contacts: bool=False) -> set[tuple[int, int]]:
1205
+ """
1206
+ Get contacts from the structure attribute where the distance between two residues is less than the threshold.
1207
+
1208
+ Parameters
1209
+ ----------
1210
+ ca_only : bool
1211
+ If true, only consider alpha-carbon to alpha-carbon distances.
1212
+ threshold : float
1213
+ Maximum distance to consider between two atoms.
1214
+ chain1 : str
1215
+ Chain id corresponding to the first column of residues in the structure.
1216
+ chain2 : str
1217
+ Chain id corresponding to the second column of residues in the structure.
1218
+ auth_contacts : bool
1219
+ True if you want alt_ids for residues indices, False if cif residue indexing is needed.
1220
+
1221
+ Returns
1222
+ -------
1223
+ contacts_set : set of tuple of ints
1224
+ Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
1225
+ """
1226
+
1227
+ chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2)
1228
+ thresh_ind = np.argwhere(dist_matrix <= threshold)
1229
+ contacts_set = set()
1230
+ for indices in thresh_ind:
1231
+ chain1_atom = chain1_structure[indices[0]]
1232
+ chain2_atom = chain2_structure[indices[1]]
1233
+ res1 = chain1_atom.res_id
1234
+ res2 = chain2_atom.res_id
1235
+ if not(chain1==chain2 and res1 >= res2):
1236
+ if not auth_contacts:
1237
+ shift1, shift2 = self.get_shift_values(chain1, chain2)
1238
+ contacts_set.add((res1 - shift1, res2 - shift2))
1239
+ else:
1240
+ contacts_set.add((res1, res2))
1241
+ return contacts_set
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: dcatoolkit
3
- Version: 0.1.8
3
+ Version: 0.2.0
4
4
  Summary: Collection of useful modules and representations for managing DCA output data.
5
5
  Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
6
6
  Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
@@ -38,7 +38,7 @@ Classifier: Programming Language :: Python :: 3.12
38
38
  Requires-Python: >=3.10
39
39
  Description-Content-Type: text/markdown
40
40
  License-File: LICENSE
41
- Requires-Dist: biotite
41
+ Requires-Dist: biotite>=1.0.1
42
42
  Requires-Dist: matplotlib>=3.8.0
43
43
  Requires-Dist: numpy>=1.26.0
44
44
  Requires-Dist: pandas>=2.1.0
@@ -8,4 +8,5 @@ src/dcatoolkit.egg-info/PKG-INFO
8
8
  src/dcatoolkit.egg-info/SOURCES.txt
9
9
  src/dcatoolkit.egg-info/dependency_links.txt
10
10
  src/dcatoolkit.egg-info/requires.txt
11
- src/dcatoolkit.egg-info/top_level.txt
11
+ src/dcatoolkit.egg-info/top_level.txt
12
+ tests/test_contacts.py
@@ -1,4 +1,4 @@
1
- biotite
1
+ biotite>=1.0.1
2
2
  matplotlib>=3.8.0
3
3
  numpy>=1.26.0
4
4
  pandas>=2.1.0
@@ -0,0 +1,95 @@
1
+ from context import MMCIFInformation, PDBInformation
2
+ from pathlib import Path
3
+ import biotite.structure.io.pdbx as pdbx
4
+ import biotite.database.rcsb as rcsb
5
+
6
+ pdb_id_chain_map = {}
7
+ with open("tests/pdb_info/pdb_ids.txt", "r") as fs:
8
+ pdb_id_data = fs.read().splitlines()
9
+ for line in pdb_id_data:
10
+ pdb_id, chain1, chain2 = line.split()
11
+ if len(chain1.split("/")) > 1:
12
+ chain1, auth_chain1 = chain1.split("/")
13
+ else:
14
+ auth_chain1 = chain1
15
+ if len(chain2.split("/")) > 1:
16
+ chain2, auth_chain2 = chain2.split("/")
17
+ else:
18
+ auth_chain2 = chain2
19
+ pdb_id_chain_map[pdb_id] = (chain1, auth_chain1, chain2, auth_chain2)
20
+
21
+ def read_contacts(input_filepath: str) -> set[tuple[int, int]]:
22
+ with open(input_filepath, 'r') as fs:
23
+ data = fs.read().splitlines()
24
+ results = set()
25
+ for pair in data:
26
+ res1, res2 = pair.split()
27
+ res1 = int(res1)
28
+ res2 = int(res2)
29
+ if res1 < res2:
30
+ results.add((int(res1), int(res2)))
31
+ return results
32
+ #return set([(int(x.split()[0]), int(x.split()[1])) for x in data])
33
+
34
+ def drop_inord_res(contacts: set[tuple[int, int]]) -> set[tuple[int, int]]:
35
+ results = set()
36
+ for contact in contacts:
37
+ if contact[0] < contact[1]:
38
+ results.add(contact)
39
+ return results
40
+
41
+ def compare(*args) -> None:
42
+ for arg1 in args:
43
+ for arg2 in args:
44
+ assert arg1 == arg2
45
+
46
+ def check_contacts(test_CA: bool, threshold: float):
47
+ cif_files = list(Path("tests/pdb_info").glob("*.cif"))
48
+ if test_CA:
49
+ print("Testing CA")
50
+ else:
51
+ print("Testing AA")
52
+
53
+ for cif_file in cif_files:
54
+ pdb_id = cif_file.stem.upper()
55
+ if test_CA:
56
+ corresponding_file = f"tests/pdb_info/monomer_calpha_{pdb_id}_10.0"
57
+ else:
58
+ corresponding_file = f"tests/pdb_info/monomer_allatom_{pdb_id}_8.0"
59
+
60
+ cif_file_contacts = read_contacts(corresponding_file)
61
+ chain1, auth_chain1, chain2, auth_chain2 = pdb_id_chain_map[pdb_id]
62
+ fetch_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id).get_contacts(test_CA, threshold, chain1, chain2, auth_contacts=True)}
63
+ read_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, chain1, chain2, auth_contacts=True)}
64
+ fetch_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id).get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True, auth_chain_id_supplied=True)}
65
+ read_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True, auth_chain_id_supplied=True)}
66
+ fetch_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.fetch_pdb(pdb_id, struc_format="pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True)}
67
+ read_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.read_pdb_file(f"tests/pdb_info/{pdb_id.lower()}.pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_contacts=True)}
68
+
69
+
70
+ fetch_cif_contacts = drop_inord_res(fetch_cif_contacts)
71
+ fetch_authchain_cif_contacts = drop_inord_res(fetch_authchain_cif_contacts)
72
+ read_cif_contacts = drop_inord_res(read_cif_contacts)
73
+ read_authchain_cif_contacts = drop_inord_res(read_authchain_cif_contacts)
74
+ fetch_pdb_contacts = drop_inord_res(fetch_pdb_contacts)
75
+ read_pdb_contacts = drop_inord_res(read_pdb_contacts)
76
+
77
+ if pdbx.get_model_count(pdbx.CIFFile.read(rcsb.fetch(pdb_id, format="mmcif"))) <= 1:
78
+ compare(fetch_cif_contacts, fetch_pdb_contacts, read_cif_contacts, read_pdb_contacts)
79
+ print(f"{pdb_id} has no difference between cif and pdb reading.")
80
+
81
+ compare(fetch_cif_contacts, fetch_authchain_cif_contacts, read_cif_contacts, read_authchain_cif_contacts)
82
+ print(f"{pdb_id} has no issue reading with auth chains and asym chains.")
83
+
84
+ compare(fetch_cif_contacts, cif_file_contacts)
85
+ print(f"{pdb_id} has no difference between cif and interface_contacts_X.py reading.")
86
+ else:
87
+ compare(fetch_cif_contacts, fetch_pdb_contacts, read_cif_contacts, read_pdb_contacts)
88
+ print(f"{pdb_id} has no difference between cif and pdb reading.")
89
+
90
+ compare(fetch_cif_contacts, fetch_authchain_cif_contacts, read_cif_contacts, read_authchain_cif_contacts)
91
+ print(f"{pdb_id} has no issue reading with auth chains and asym chains.")
92
+
93
+ def test_contacts():
94
+ check_contacts(test_CA=True, threshold=10)
95
+ check_contacts(test_CA=False, threshold=8)
@@ -1,4 +0,0 @@
1
-
2
- __version__ = "0.1.8"
3
- from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment
4
- from .analytics import MSATools
File without changes
File without changes
File without changes