pidibble 1.4.2__tar.gz → 1.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pidibble-1.4.2 → pidibble-1.5.0}/PKG-INFO +1 -1
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/usage.rst +8 -2
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/pdbparse.py +96 -80
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/resources/pdb_format.yaml +1 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pyproject.toml +1 -1
- {pidibble-1.4.2 → pidibble-1.5.0}/.github/workflows/release.yaml +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/.gitignore +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/.readthedocs.yaml +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/LICENSE +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/MANIFEST.in +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/README.md +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/Makefile +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/make.bat +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/requirements.txt +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/_static/css/custom.css +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/API.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.baseparsers.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.baserecord.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.hex.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.mmcif_parse.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.pdbparse.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.pdbrecord.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.resources.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/api/pidibble.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/conf.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/index.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/installation.rst +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/docs/source/notes.md +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/__init__.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/baseparsers.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/baserecord.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/hex.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/mmcif_parse.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/pdbrecord.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/resources/__init__.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/pidibble/resources/mmcif_format.yaml +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/__init__.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/conftest.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_hex/my_system.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_hex.py +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4tvp.cif +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4tvp.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4zmj-newresnames.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4zmj.cif +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/4zmj.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/6m0j.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/8fae.cif +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/G.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/GG.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/test.pdb +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb/test_pdb_format.yaml +0 -0
- {pidibble-1.4.2 → pidibble-1.5.0}/tests/unit/test_rcsb.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pidibble
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.5.0
|
|
4
4
|
Summary: A complete Protein Data Bank (PDB) file parser
|
|
5
5
|
Project-URL: Source, https://github.com/cameronabrams/pidibble
|
|
6
6
|
Project-URL: Documentation, https://pidibble.readthedocs.io/en/latest/
|
|
@@ -7,16 +7,22 @@ Usage Example
|
|
|
7
7
|
Let's parse the PDB entry '4ZMJ', which is a trimeric ectodomain construct of the HIV-1 envelope glycoprotein:
|
|
8
8
|
|
|
9
9
|
>>> from pidibble.pdbparse import PDBParser
|
|
10
|
-
>>> p=PDBParser(
|
|
10
|
+
>>> p = PDBParser(source_db='rcsb', source_id='4zmj').parse()
|
|
11
11
|
|
|
12
12
|
The ``PDBParser()`` call creates a new ``PDBParser`` object, and the member function ``parse()`` executes (optionally) downloading the PDB file of the code entered with the ``PDBcode`` keyword argument to ``PDBParser()``, followed by parsing into a member dictionary ``parsed``. (The file is downloaded from the RSCB only if it is not found in the current working directory.)
|
|
13
13
|
|
|
14
14
|
Alternatively, a ``PDBParser()`` invocation can fetch from the AlphaFold model database by providing the accession code with the ``alphafold`` keyword:
|
|
15
15
|
|
|
16
|
-
>>> p=PDBParser(alphafold='O46077').parse()
|
|
16
|
+
>>> p = PDBParser(source_db='alphafold', sourced_id='O46077').parse()
|
|
17
17
|
|
|
18
18
|
(Note that this fetches a model for the odor receptor OR2a from *D. melanogaster*. For the rest of this example, we'll work with the HIV-1 Env trimer 4zmj above.)
|
|
19
19
|
|
|
20
|
+
Finally, one can also retrieve entries from OPM:
|
|
21
|
+
|
|
22
|
+
>>> p = PDBParser(source_db='opm', source_id='7f1r').parse()
|
|
23
|
+
|
|
24
|
+
(Note that this fetches the sweet receptor with dummy atoms that denote locations of lipid headgroups.)
|
|
25
|
+
|
|
20
26
|
>>> type(p.parsed)
|
|
21
27
|
<class 'dict'>
|
|
22
28
|
|
|
@@ -7,24 +7,27 @@
|
|
|
7
7
|
.. moduleauthor: Cameron F. Abrams, <cfa22@drexel.edu>
|
|
8
8
|
|
|
9
9
|
"""
|
|
10
|
-
import
|
|
11
|
-
import
|
|
10
|
+
import importlib.metadata
|
|
11
|
+
import json
|
|
12
12
|
import logging
|
|
13
|
+
import os
|
|
14
|
+
import urllib.request
|
|
13
15
|
import yaml
|
|
16
|
+
from typing import Callable
|
|
17
|
+
|
|
14
18
|
import numpy as np
|
|
15
|
-
|
|
19
|
+
|
|
16
20
|
from mmcif.io.IoAdapterCore import IoAdapterCore
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
17
23
|
from . import resources
|
|
18
24
|
from .baseparsers import ListParsers, ListParser, str2int_sig, safe_float
|
|
19
25
|
from .baserecord import BaseRecordParser
|
|
20
26
|
from .pdbrecord import PDBRecord, PDBRecordDict, PDBRecordList
|
|
21
27
|
from .mmcif_parse import MMCIF_Parser
|
|
22
|
-
from pidibble import resources
|
|
23
|
-
logger=logging.getLogger(__name__)
|
|
24
|
-
import importlib.metadata
|
|
25
|
-
import json
|
|
26
28
|
from .hex import str2atomSerial, hex_reset
|
|
27
29
|
|
|
30
|
+
logger = logging.getLogger(__name__)
|
|
28
31
|
__version__ = importlib.metadata.version("pidibble")
|
|
29
32
|
|
|
30
33
|
class PDBParser:
|
|
@@ -50,46 +53,42 @@ class PDBParser:
|
|
|
50
53
|
A dictionary containing the parsed mmCIF data. Empty if no input file is provided.
|
|
51
54
|
"""
|
|
52
55
|
def __init__(self,
|
|
53
|
-
input_format='PDB',
|
|
54
|
-
overwrite=False,
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
comment_chars=['#'],
|
|
60
|
-
pdb_format_file='pdb_format.yaml',
|
|
61
|
-
mmcif_format_file='mmcif_format.yaml',
|
|
56
|
+
input_format: str = 'PDB',
|
|
57
|
+
overwrite: bool = False,
|
|
58
|
+
source_db: str = None,
|
|
59
|
+
source_id: str = None,
|
|
60
|
+
filepath: str | Path = None,
|
|
61
|
+
mappers: dict[str, Callable] = {'HxInteger':str2atomSerial, 'Integer':str2int_sig, 'String':str, 'Float':safe_float},
|
|
62
|
+
comment_chars: list[str] = ['#'],
|
|
63
|
+
pdb_format_file: str ='pdb_format.yaml',
|
|
64
|
+
mmcif_format_file: str = 'mmcif_format.yaml',
|
|
62
65
|
**kwargs):
|
|
63
66
|
logger.debug(f'Pidibble v. {__version__}')
|
|
64
|
-
self.input_format=input_format
|
|
65
|
-
self.overwrite=overwrite
|
|
66
|
-
self.
|
|
67
|
-
self.
|
|
68
|
-
self.
|
|
67
|
+
self.input_format = input_format
|
|
68
|
+
self.overwrite = overwrite
|
|
69
|
+
self.source_db = source_db
|
|
70
|
+
self.source_id = source_id
|
|
71
|
+
self.filepath = Path(filepath) if filepath else None
|
|
72
|
+
self.mappers = mappers
|
|
69
73
|
self.mappers.update(ListParsers)
|
|
70
|
-
self.
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
self.pdb_code=kwargs[k]
|
|
74
|
-
break
|
|
75
|
-
self.comment_chars=comment_chars
|
|
76
|
-
self.pdb_lines=[]
|
|
77
|
-
self.cif_data={}
|
|
74
|
+
self.comment_chars = comment_chars
|
|
75
|
+
self.pdb_lines = []
|
|
76
|
+
self.cif_data = {}
|
|
78
77
|
|
|
79
|
-
self.parsed=PDBRecordDict()
|
|
80
|
-
self.pdb_format_file=pdb_format_file
|
|
78
|
+
self.parsed = PDBRecordDict()
|
|
79
|
+
self.pdb_format_file = pdb_format_file
|
|
81
80
|
if not os.path.isfile(self.pdb_format_file):
|
|
82
81
|
# if pdb_format_file is not a file in the CWD, assume it is a relative path to the resources directory
|
|
83
82
|
# this is useful for testing
|
|
84
|
-
self.pdb_format_file=os.path.join(os.path.dirname(resources.__file__),pdb_format_file)
|
|
85
|
-
self.mmcif_format_file=mmcif_format_file
|
|
83
|
+
self.pdb_format_file = os.path.join(os.path.dirname(resources.__file__), pdb_format_file)
|
|
84
|
+
self.mmcif_format_file = mmcif_format_file
|
|
86
85
|
if not os.path.isfile(self.mmcif_format_file):
|
|
87
86
|
# if mmcif_format_file is not a file in the CWD, assume it is a relative path to the resources directory
|
|
88
87
|
# this is useful for testing
|
|
89
|
-
self.mmcif_format_file=os.path.join(os.path.dirname(resources.__file__),mmcif_format_file)
|
|
88
|
+
self.mmcif_format_file = os.path.join(os.path.dirname(resources.__file__), mmcif_format_file)
|
|
90
89
|
if os.path.exists(self.pdb_format_file):
|
|
91
|
-
with open(self.pdb_format_file,'r') as f:
|
|
92
|
-
self.pdb_format_dict=yaml.safe_load(f)
|
|
90
|
+
with open(self.pdb_format_file, 'r') as f:
|
|
91
|
+
self.pdb_format_dict = yaml.safe_load(f)
|
|
93
92
|
logger.debug(f'Pidibble uses the installed config file {self.pdb_format_file}')
|
|
94
93
|
else:
|
|
95
94
|
raise FileNotFoundError(f'{self.pdb_format_file} not found, either locally ({os.getcwd()}) or in resources ({os.path.dirname(resources.__file__)})')
|
|
@@ -101,14 +100,14 @@ class PDBParser:
|
|
|
101
100
|
raise FileNotFoundError(f'{self.mmcif_format_file} not found, either locally ({os.getcwd()}) or in resources ({os.path.dirname(resources.__file__)})')
|
|
102
101
|
|
|
103
102
|
# update mappers with delimiters and custom formats
|
|
104
|
-
delimiter_dict=self.pdb_format_dict.get('delimiters',{})
|
|
103
|
+
delimiter_dict = self.pdb_format_dict.get('delimiters', {})
|
|
105
104
|
for map,d in delimiter_dict.items():
|
|
106
105
|
if not map in self.mappers:
|
|
107
|
-
self.mappers[map]=ListParser(d).parse
|
|
108
|
-
cformat_dict=self.pdb_format_dict.get('custom_formats',{})
|
|
106
|
+
self.mappers[map] = ListParser(d).parse
|
|
107
|
+
cformat_dict = self.pdb_format_dict.get('custom_formats', {})
|
|
109
108
|
for cname,cformat in cformat_dict.items():
|
|
110
109
|
if not cname in self.mappers:
|
|
111
|
-
self.mappers[cname]=BaseRecordParser(cformat,self.mappers).parse
|
|
110
|
+
self.mappers[cname] = BaseRecordParser(cformat, self.mappers).parse
|
|
112
111
|
|
|
113
112
|
def fetch(self):
|
|
114
113
|
"""
|
|
@@ -121,50 +120,67 @@ class PDBParser:
|
|
|
121
120
|
bool
|
|
122
121
|
True if the file was successfully fetched, False otherwise.
|
|
123
122
|
"""
|
|
124
|
-
assert self.
|
|
125
|
-
if self.
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
123
|
+
assert self.source_db is not None or self.filepath is not None, f'source_db or filepath must be specified for fetch()'
|
|
124
|
+
if self.source_db is not None and self.source_id is None:
|
|
125
|
+
raise ValueError(f'You must specify a source ID code for source_db {self.source_db}')
|
|
126
|
+
if self.source_db is not None and self.source_db not in ['rcsb', 'alphafold', 'opm']:
|
|
127
|
+
raise ValueError(f'Source db {self.source_db} is not recognized.')
|
|
128
|
+
|
|
129
|
+
if self.filepath is not None:
|
|
130
|
+
if not self.filepath.exists():
|
|
131
|
+
raise FileNotFoundError(f'{self.filepath.name} not found.')
|
|
132
|
+
return True
|
|
133
|
+
|
|
134
|
+
match self.source_db:
|
|
135
|
+
case 'rcsb':
|
|
136
|
+
if self.input_format == 'PDB':
|
|
137
|
+
self.filepath = Path(f'{self.source_id}.pdb')
|
|
138
|
+
elif self.input_format == 'mmCIF':
|
|
139
|
+
self.filepath = Path(f'{self.source_id}.cif')
|
|
140
|
+
else:
|
|
141
|
+
logger.warning(f'Input format {self.input_format} not recognized; using PDB')
|
|
142
|
+
self.filepath = Path(f'{self.source_id}.pdb')
|
|
143
|
+
BASE_URL = self.pdb_format_dict['BASE_URL']
|
|
144
|
+
target_url = os.path.join(BASE_URL, self.filepath.name)
|
|
145
|
+
if not self.filepath.exists() or self.overwrite:
|
|
146
|
+
try:
|
|
147
|
+
urllib.request.urlretrieve(target_url, self.filepath.name)
|
|
148
|
+
except:
|
|
149
|
+
logger.warning(f'Could not fetch {self.filepath.name} from {self.source_db}')
|
|
150
|
+
return False
|
|
151
|
+
return True
|
|
152
|
+
case 'alphafold':
|
|
153
|
+
self.filepath = Path(f'{self.source_id}.pdb')
|
|
154
|
+
BASE_URL = self.pdb_format_dict['ALPHAFOLD_API_URL']
|
|
155
|
+
target_url = os.path.join(BASE_URL, self.source_id)
|
|
136
156
|
try:
|
|
137
|
-
urllib.request.urlretrieve(target_url,self.
|
|
157
|
+
urllib.request.urlretrieve(target_url + r'?key=' + self.pdb_format_dict['ALPHAFOLD_API_KEY'], f'{self.source_id}.json')
|
|
138
158
|
except:
|
|
139
|
-
logger.warning(f'Could not fetch {self.
|
|
159
|
+
logger.warning(f'Could not fetch metadata for entry with accession code {self.source_id} from AlphaFold')
|
|
140
160
|
return False
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
logger.warning(f'File {self.filepath} does not exist.')
|
|
161
|
+
with open(f'{self.source_id}.json') as f:
|
|
162
|
+
result = json.load(f)
|
|
163
|
+
try:
|
|
164
|
+
urllib.request.urlretrieve(result[0]['pdbUrl'], self.filepath.name)
|
|
165
|
+
except:
|
|
166
|
+
logger.warning(f'Could not retrieve {result[0]["pdbUrl"]}')
|
|
167
|
+
return False
|
|
168
|
+
return True
|
|
169
|
+
case 'opm':
|
|
170
|
+
self.filepath = Path(f'{self.source_id}.pdb')
|
|
171
|
+
BASE_URL = self.pdb_format_dict['OPM_URL']
|
|
172
|
+
target_url = os.path.join(BASE_URL, self.filepath.name)
|
|
173
|
+
if not self.filepath.exists() or self.overwrite:
|
|
174
|
+
try:
|
|
175
|
+
urllib.request.urlretrieve(target_url, self.filepath.name)
|
|
176
|
+
except:
|
|
177
|
+
logger.warning(f'Could not fetch {self.filepath.name} from {self.source_db}')
|
|
178
|
+
return False
|
|
179
|
+
return True
|
|
180
|
+
case '_':
|
|
181
|
+
logger.debug(f'Source db {self.source_db} is not recognized.')
|
|
163
182
|
return False
|
|
164
|
-
|
|
165
|
-
else:
|
|
166
|
-
pass # assert statement at top of method body should suppress this branch
|
|
167
|
-
|
|
183
|
+
|
|
168
184
|
def read_PDB(self):
|
|
169
185
|
"""
|
|
170
186
|
Read the PDB file and store its lines in :attr:`PDBParser.pdb_lines`.
|
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
BASE_URL: https://files.rcsb.org/download
|
|
14
14
|
ALPHAFOLD_API_URL: https://alphafold.ebi.ac.uk/api/prediction
|
|
15
15
|
ALPHAFOLD_API_KEY: AIzaSyCeurAJz7ZGjPQUtEaerUkBZ3TaBkXrY94
|
|
16
|
+
OPM_URL: https://biomembhub.org/shared/opm-assets/pdb
|
|
16
17
|
record_types:
|
|
17
18
|
1: one-time-single-line
|
|
18
19
|
2: one-time-multiple-line
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|