pidibble 1.4.2__tar.gz → 1.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. {pidibble-1.4.2 → pidibble-1.5.0}/PKG-INFO +1 -1
  2. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/usage.rst +8 -2
  3. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/pdbparse.py +96 -80
  4. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/resources/pdb_format.yaml +1 -0
  5. {pidibble-1.4.2 → pidibble-1.5.0}/pyproject.toml +1 -1
  6. {pidibble-1.4.2 → pidibble-1.5.0}/.github/workflows/release.yaml +0 -0
  7. {pidibble-1.4.2 → pidibble-1.5.0}/.gitignore +0 -0
  8. {pidibble-1.4.2 → pidibble-1.5.0}/.readthedocs.yaml +0 -0
  9. {pidibble-1.4.2 → pidibble-1.5.0}/LICENSE +0 -0
  10. {pidibble-1.4.2 → pidibble-1.5.0}/MANIFEST.in +0 -0
  11. {pidibble-1.4.2 → pidibble-1.5.0}/README.md +0 -0
  12. {pidibble-1.4.2 → pidibble-1.5.0}/docs/Makefile +0 -0
  13. {pidibble-1.4.2 → pidibble-1.5.0}/docs/make.bat +0 -0
  14. {pidibble-1.4.2 → pidibble-1.5.0}/docs/requirements.txt +0 -0
  15. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/_static/css/custom.css +0 -0
  16. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/API.rst +0 -0
  17. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.baseparsers.rst +0 -0
  18. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.baserecord.rst +0 -0
  19. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.hex.rst +0 -0
  20. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.mmcif_parse.rst +0 -0
  21. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.pdbparse.rst +0 -0
  22. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.pdbrecord.rst +0 -0
  23. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.resources.rst +0 -0
  24. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.rst +0 -0
  25. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/conf.py +0 -0
  26. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/index.rst +0 -0
  27. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/installation.rst +0 -0
  28. {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/notes.md +0 -0
  29. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/__init__.py +0 -0
  30. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/baseparsers.py +0 -0
  31. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/baserecord.py +0 -0
  32. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/hex.py +0 -0
  33. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/mmcif_parse.py +0 -0
  34. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/pdbrecord.py +0 -0
  35. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/resources/__init__.py +0 -0
  36. {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/resources/mmcif_format.yaml +0 -0
  37. {pidibble-1.4.2 → pidibble-1.5.0}/tests/__init__.py +0 -0
  38. {pidibble-1.4.2 → pidibble-1.5.0}/tests/conftest.py +0 -0
  39. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_hex/my_system.pdb +0 -0
  40. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_hex.py +0 -0
  41. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4tvp.cif +0 -0
  42. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4tvp.pdb +0 -0
  43. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4zmj-newresnames.pdb +0 -0
  44. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4zmj.cif +0 -0
  45. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4zmj.pdb +0 -0
  46. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/6m0j.pdb +0 -0
  47. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/8fae.cif +0 -0
  48. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/G.pdb +0 -0
  49. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/GG.pdb +0 -0
  50. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/test.pdb +0 -0
  51. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/test_pdb_format.yaml +0 -0
  52. {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pidibble
3
- Version: 1.4.2
3
+ Version: 1.5.0
4
4
  Summary: A complete Protein Data Bank (PDB) file parser
5
5
  Project-URL: Source, https://github.com/cameronabrams/pidibble
6
6
  Project-URL: Documentation, https://pidibble.readthedocs.io/en/latest/
@@ -7,16 +7,22 @@ Usage Example
7
7
  Let's parse the PDB entry '4ZMJ', which is a trimeric ectodomain construct of the HIV-1 envelope glycoprotein:
8
8
 
9
9
  >>> from pidibble.pdbparse import PDBParser
10
- >>> p=PDBParser(PDBcode='4zmj').parse()
10
+ >>> p = PDBParser(source_db='rcsb', source_id='4zmj').parse()
11
11
 
12
12
  The ``PDBParser()`` call creates a new ``PDBParser`` object, and the member function ``parse()`` executes (optionally) downloading the PDB file of the code entered with the ``PDBcode`` keyword argument to ``PDBParser()``, followed by parsing into a member dictionary ``parsed``. (The file is downloaded from the RSCB only if it is not found in the current working directory.)
13
13
 
14
14
  Alternatively, a ``PDBParser()`` invocation can fetch from the AlphaFold model database by providing the accession code with the ``alphafold`` keyword:
15
15
 
16
- >>> p=PDBParser(alphafold='O46077').parse()
16
+ >>> p = PDBParser(source_db='alphafold', sourced_id='O46077').parse()
17
17
 
18
18
  (Note that this fetches a model for the odor receptor OR2a from *D. melanogaster*. For the rest of this example, we'll work with the HIV-1 Env trimer 4zmj above.)
19
19
 
20
+ Finally, one can also retrieve entries from OPM:
21
+
22
+ >>> p = PDBParser(source_db='opm', source_id='7f1r').parse()
23
+
24
+ (Note that this fetches the sweet receptor with dummy atoms that denote locations of lipid headgroups.)
25
+
20
26
  >>> type(p.parsed)
21
27
  <class 'dict'>
22
28
 
@@ -7,24 +7,27 @@
7
7
  .. moduleauthor: Cameron F. Abrams, <cfa22@drexel.edu>
8
8
 
9
9
  """
10
- import urllib.request
11
- import os
10
+ import importlib.metadata
11
+ import json
12
12
  import logging
13
+ import os
14
+ import urllib.request
13
15
  import yaml
16
+ from typing import Callable
17
+
14
18
  import numpy as np
15
- from pathlib import Path
19
+
16
20
  from mmcif.io.IoAdapterCore import IoAdapterCore
21
+ from pathlib import Path
22
+
17
23
  from . import resources
18
24
  from .baseparsers import ListParsers, ListParser, str2int_sig, safe_float
19
25
  from .baserecord import BaseRecordParser
20
26
  from .pdbrecord import PDBRecord, PDBRecordDict, PDBRecordList
21
27
  from .mmcif_parse import MMCIF_Parser
22
- from pidibble import resources
23
- logger=logging.getLogger(__name__)
24
- import importlib.metadata
25
- import json
26
28
  from .hex import str2atomSerial, hex_reset
27
29
 
30
+ logger = logging.getLogger(__name__)
28
31
  __version__ = importlib.metadata.version("pidibble")
29
32
 
30
33
  class PDBParser:
@@ -50,46 +53,42 @@ class PDBParser:
50
53
  A dictionary containing the parsed mmCIF data. Empty if no input file is provided.
51
54
  """
52
55
  def __init__(self,
53
- input_format='PDB',
54
- overwrite=False,
55
- alphafold='',
56
- filepath: Path = None,
57
- mappers={'HxInteger':str2atomSerial,'Integer':str2int_sig,'String':str,'Float':safe_float},
58
- pdbcode_synonyms=['PDBcode','pdb_code','pdbCode','pdbcode','pdbID','pdb_id','pdbid','PDBID','PDBId','PDBid'],
59
- comment_chars=['#'],
60
- pdb_format_file='pdb_format.yaml',
61
- mmcif_format_file='mmcif_format.yaml',
56
+ input_format: str = 'PDB',
57
+ overwrite: bool = False,
58
+ source_db: str = None,
59
+ source_id: str = None,
60
+ filepath: str | Path = None,
61
+ mappers: dict[str, Callable] = {'HxInteger':str2atomSerial, 'Integer':str2int_sig, 'String':str, 'Float':safe_float},
62
+ comment_chars: list[str] = ['#'],
63
+ pdb_format_file: str ='pdb_format.yaml',
64
+ mmcif_format_file: str = 'mmcif_format.yaml',
62
65
  **kwargs):
63
66
  logger.debug(f'Pidibble v. {__version__}')
64
- self.input_format=input_format
65
- self.overwrite=overwrite
66
- self.alphafold=alphafold
67
- self.filepath=filepath
68
- self.mappers=mappers
67
+ self.input_format = input_format
68
+ self.overwrite = overwrite
69
+ self.source_db = source_db
70
+ self.source_id = source_id
71
+ self.filepath = Path(filepath) if filepath else None
72
+ self.mappers = mappers
69
73
  self.mappers.update(ListParsers)
70
- self.pdb_code=''
71
- for k in pdbcode_synonyms:
72
- if k in kwargs:
73
- self.pdb_code=kwargs[k]
74
- break
75
- self.comment_chars=comment_chars
76
- self.pdb_lines=[]
77
- self.cif_data={}
74
+ self.comment_chars = comment_chars
75
+ self.pdb_lines = []
76
+ self.cif_data = {}
78
77
 
79
- self.parsed=PDBRecordDict()
80
- self.pdb_format_file=pdb_format_file
78
+ self.parsed = PDBRecordDict()
79
+ self.pdb_format_file = pdb_format_file
81
80
  if not os.path.isfile(self.pdb_format_file):
82
81
  # if pdb_format_file is not a file in the CWD, assume it is a relative path to the resources directory
83
82
  # this is useful for testing
84
- self.pdb_format_file=os.path.join(os.path.dirname(resources.__file__),pdb_format_file)
85
- self.mmcif_format_file=mmcif_format_file
83
+ self.pdb_format_file = os.path.join(os.path.dirname(resources.__file__), pdb_format_file)
84
+ self.mmcif_format_file = mmcif_format_file
86
85
  if not os.path.isfile(self.mmcif_format_file):
87
86
  # if mmcif_format_file is not a file in the CWD, assume it is a relative path to the resources directory
88
87
  # this is useful for testing
89
- self.mmcif_format_file=os.path.join(os.path.dirname(resources.__file__),mmcif_format_file)
88
+ self.mmcif_format_file = os.path.join(os.path.dirname(resources.__file__), mmcif_format_file)
90
89
  if os.path.exists(self.pdb_format_file):
91
- with open(self.pdb_format_file,'r') as f:
92
- self.pdb_format_dict=yaml.safe_load(f)
90
+ with open(self.pdb_format_file, 'r') as f:
91
+ self.pdb_format_dict = yaml.safe_load(f)
93
92
  logger.debug(f'Pidibble uses the installed config file {self.pdb_format_file}')
94
93
  else:
95
94
  raise FileNotFoundError(f'{self.pdb_format_file} not found, either locally ({os.getcwd()}) or in resources ({os.path.dirname(resources.__file__)})')
@@ -101,14 +100,14 @@ class PDBParser:
101
100
  raise FileNotFoundError(f'{self.mmcif_format_file} not found, either locally ({os.getcwd()}) or in resources ({os.path.dirname(resources.__file__)})')
102
101
 
103
102
  # update mappers with delimiters and custom formats
104
- delimiter_dict=self.pdb_format_dict.get('delimiters',{})
103
+ delimiter_dict = self.pdb_format_dict.get('delimiters', {})
105
104
  for map,d in delimiter_dict.items():
106
105
  if not map in self.mappers:
107
- self.mappers[map]=ListParser(d).parse
108
- cformat_dict=self.pdb_format_dict.get('custom_formats',{})
106
+ self.mappers[map] = ListParser(d).parse
107
+ cformat_dict = self.pdb_format_dict.get('custom_formats', {})
109
108
  for cname,cformat in cformat_dict.items():
110
109
  if not cname in self.mappers:
111
- self.mappers[cname]=BaseRecordParser(cformat,self.mappers).parse
110
+ self.mappers[cname] = BaseRecordParser(cformat, self.mappers).parse
112
111
 
113
112
  def fetch(self):
114
113
  """
@@ -121,50 +120,67 @@ class PDBParser:
121
120
  bool
122
121
  True if the file was successfully fetched, False otherwise.
123
122
  """
124
- assert self.pdb_code!='' or self.alphafold!='' or self.filepath is not None, "PDB code or AlphaFold ID must be provided, or a path to a PDB file must be specified."
125
- if self.pdb_code!='':
126
- if self.input_format=='PDB':
127
- self.filepath=f'{self.pdb_code}.pdb'
128
- elif self.input_format=='mmCIF':
129
- self.filepath=f'{self.pdb_code}.cif'
130
- else:
131
- logger.warning(f'Input format {self.input_format} not recognized; using PDB')
132
- self.filepath=f'{self.pdb_code}.pdb'
133
- BASE_URL=self.pdb_format_dict['BASE_URL']
134
- target_url=os.path.join(BASE_URL,self.filepath)
135
- if not os.path.exists(self.filepath) or self.overwrite:
123
+ assert self.source_db is not None or self.filepath is not None, f'source_db or filepath must be specified for fetch()'
124
+ if self.source_db is not None and self.source_id is None:
125
+ raise ValueError(f'You must specify a source ID code for source_db {self.source_db}')
126
+ if self.source_db is not None and self.source_db not in ['rcsb', 'alphafold', 'opm']:
127
+ raise ValueError(f'Source db {self.source_db} is not recognized.')
128
+
129
+ if self.filepath is not None:
130
+ if not self.filepath.exists():
131
+ raise FileNotFoundError(f'{self.filepath.name} not found.')
132
+ return True
133
+
134
+ match self.source_db:
135
+ case 'rcsb':
136
+ if self.input_format == 'PDB':
137
+ self.filepath = Path(f'{self.source_id}.pdb')
138
+ elif self.input_format == 'mmCIF':
139
+ self.filepath = Path(f'{self.source_id}.cif')
140
+ else:
141
+ logger.warning(f'Input format {self.input_format} not recognized; using PDB')
142
+ self.filepath = Path(f'{self.source_id}.pdb')
143
+ BASE_URL = self.pdb_format_dict['BASE_URL']
144
+ target_url = os.path.join(BASE_URL, self.filepath.name)
145
+ if not self.filepath.exists() or self.overwrite:
146
+ try:
147
+ urllib.request.urlretrieve(target_url, self.filepath.name)
148
+ except:
149
+ logger.warning(f'Could not fetch {self.filepath.name} from {self.source_db}')
150
+ return False
151
+ return True
152
+ case 'alphafold':
153
+ self.filepath = Path(f'{self.source_id}.pdb')
154
+ BASE_URL = self.pdb_format_dict['ALPHAFOLD_API_URL']
155
+ target_url = os.path.join(BASE_URL, self.source_id)
136
156
  try:
137
- urllib.request.urlretrieve(target_url,self.filepath)
157
+ urllib.request.urlretrieve(target_url + r'?key=' + self.pdb_format_dict['ALPHAFOLD_API_KEY'], f'{self.source_id}.json')
138
158
  except:
139
- logger.warning(f'Could not fetch {self.filepath}')
159
+ logger.warning(f'Could not fetch metadata for entry with accession code {self.source_id} from AlphaFold')
140
160
  return False
141
- return True
142
- elif self.alphafold!='':
143
- self.filepath=f'{self.alphafold}.pdb'
144
- BASE_URL=self.pdb_format_dict['ALPHAFOLD_API_URL']
145
- target_url=os.path.join(BASE_URL,self.alphafold)
146
- logger.debug(f'target url {target_url}')
147
- try:
148
- urllib.request.urlretrieve(target_url+r'?key='+self.pdb_format_dict['ALPHAFOLD_API_KEY'],f'{self.alphafold}.json')
149
- except:
150
- logger.warning(f'Could not fetch metadata for entry with accession code {self.alphafold} from AlphaFold')
151
- return False
152
- with open(f'{self.alphafold}.json') as f:
153
- result=json.load(f)
154
- try:
155
- urllib.request.urlretrieve(result[0]['pdbUrl'],self.filepath)
156
- except:
157
- logger.warning(f'Could not retrieve {result[0]["pdbUrl"]}')
158
- return False
159
- return True
160
- elif self.filepath is not None:
161
- if not os.path.exists(self.filepath):
162
- logger.warning(f'File {self.filepath} does not exist.')
161
+ with open(f'{self.source_id}.json') as f:
162
+ result = json.load(f)
163
+ try:
164
+ urllib.request.urlretrieve(result[0]['pdbUrl'], self.filepath.name)
165
+ except:
166
+ logger.warning(f'Could not retrieve {result[0]["pdbUrl"]}')
167
+ return False
168
+ return True
169
+ case 'opm':
170
+ self.filepath = Path(f'{self.source_id}.pdb')
171
+ BASE_URL = self.pdb_format_dict['OPM_URL']
172
+ target_url = os.path.join(BASE_URL, self.filepath.name)
173
+ if not self.filepath.exists() or self.overwrite:
174
+ try:
175
+ urllib.request.urlretrieve(target_url, self.filepath.name)
176
+ except:
177
+ logger.warning(f'Could not fetch {self.filepath.name} from {self.source_db}')
178
+ return False
179
+ return True
180
+ case '_':
181
+ logger.debug(f'Source db {self.source_db} is not recognized.')
163
182
  return False
164
- return True
165
- else:
166
- pass # assert statement at top of method body should suppress this branch
167
-
183
+
168
184
  def read_PDB(self):
169
185
  """
170
186
  Read the PDB file and store its lines in :attr:`PDBParser.pdb_lines`.
@@ -13,6 +13,7 @@
13
13
  BASE_URL: https://files.rcsb.org/download
14
14
  ALPHAFOLD_API_URL: https://alphafold.ebi.ac.uk/api/prediction
15
15
  ALPHAFOLD_API_KEY: AIzaSyCeurAJz7ZGjPQUtEaerUkBZ3TaBkXrY94
16
+ OPM_URL: https://biomembhub.org/shared/opm-assets/pdb
16
17
  record_types:
17
18
  1: one-time-single-line
18
19
  2: one-time-multiple-line
@@ -3,7 +3,7 @@ requires = ["hatchling"]
3
3
  build-backend = "hatchling.build"
4
4
  [project]
5
5
  name = "pidibble"
6
- version = "1.4.2"
6
+ version = "1.5.0"
7
7
  authors = [
8
8
  { name="Cameron F Abrams", email="cfa22@drexel.edu" },
9
9
  ]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes