pidibble 1.7.1__tar.gz → 1.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pidibble-1.7.2/.github/workflows/tests.yaml +45 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/CHANGELOG.md +19 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/PKG-INFO +32 -8
- pidibble-1.7.2/README.md +101 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/conf.py +10 -1
- pidibble-1.7.2/docs/source/guide/advanced.rst +94 -0
- pidibble-1.7.2/docs/source/guide/assemblies.rst +88 -0
- pidibble-1.7.2/docs/source/guide/data_model.rst +110 -0
- pidibble-1.7.2/docs/source/guide/index.rst +20 -0
- pidibble-1.7.2/docs/source/guide/large_structures.rst +41 -0
- pidibble-1.7.2/docs/source/guide/loading.rst +116 -0
- pidibble-1.7.2/docs/source/guide/mmcif.rst +103 -0
- pidibble-1.7.2/docs/source/guide/nonconformance.rst +61 -0
- pidibble-1.7.2/docs/source/guide/record_reference.rst +213 -0
- pidibble-1.7.2/docs/source/guide/records.rst +115 -0
- pidibble-1.7.2/docs/source/index.rst +61 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/installation.rst +1 -1
- pidibble-1.7.2/docs/source/quickstart.rst +86 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pyproject.toml +1 -1
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb.py +23 -0
- pidibble-1.7.1/README.md +0 -77
- pidibble-1.7.1/docs/source/changelog.md +0 -5
- pidibble-1.7.1/docs/source/index.rst +0 -33
- pidibble-1.7.1/docs/source/notes.md +0 -34
- pidibble-1.7.1/docs/source/usage.rst +0 -174
- {pidibble-1.7.1 → pidibble-1.7.2}/.envrc +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/.github/workflows/release.yaml +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/.gitignore +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/.readthedocs.yaml +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/LICENSE +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/Makefile +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/make.bat +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/mmcif_coverage.md +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/requirements.txt +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/_static/css/custom.css +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/API.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.baseparsers.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.baserecord.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.hex.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.mmcif_parse.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.pdbparse.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.pdbrecord.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.resources.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/changelog.rst +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/__init__.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/baseparsers.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/baserecord.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/hex.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/mmcif_parse.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/pdbparse.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/pdbrecord.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/resources/mmcif_format.yaml +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/resources/pdb_format.yaml +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/scripts/release.sh +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/__init__.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/conftest.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_hex/my_system.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_hex.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_nonconformance.py +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/1ca2.cif +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/1ca2.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4tvp.cif +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4tvp.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4zmj-newresnames.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4zmj.cif +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4zmj.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/6m0j.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/8fae.cif +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/G.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/GG.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/test.pdb +0 -0
- {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/test_pdb_format.yaml +0 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
name: Tests
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
workflow_dispatch:
|
|
8
|
+
|
|
9
|
+
jobs:
|
|
10
|
+
pytest:
|
|
11
|
+
name: pytest (py${{ matrix.python-version }})
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
strategy:
|
|
14
|
+
fail-fast: false
|
|
15
|
+
matrix:
|
|
16
|
+
python-version: ["3.10", "3.11", "3.12"]
|
|
17
|
+
steps:
|
|
18
|
+
- uses: actions/checkout@v4
|
|
19
|
+
- name: Set up Python ${{ matrix.python-version }}
|
|
20
|
+
uses: actions/setup-python@v5
|
|
21
|
+
with:
|
|
22
|
+
python-version: ${{ matrix.python-version }}
|
|
23
|
+
- name: Install package with test extras
|
|
24
|
+
run: |
|
|
25
|
+
python -m pip install --upgrade pip
|
|
26
|
+
pip install -e .[test]
|
|
27
|
+
- name: Run unit tests
|
|
28
|
+
run: pytest tests/unit -q
|
|
29
|
+
|
|
30
|
+
doctest:
|
|
31
|
+
name: doctest (docs examples)
|
|
32
|
+
runs-on: ubuntu-latest
|
|
33
|
+
steps:
|
|
34
|
+
- uses: actions/checkout@v4
|
|
35
|
+
- name: Set up Python
|
|
36
|
+
uses: actions/setup-python@v5
|
|
37
|
+
with:
|
|
38
|
+
python-version: "3.11"
|
|
39
|
+
- name: Install docs requirements and local package
|
|
40
|
+
run: |
|
|
41
|
+
python -m pip install --upgrade pip
|
|
42
|
+
pip install -r docs/requirements.txt
|
|
43
|
+
pip install -e . # test the working tree, not the PyPI build
|
|
44
|
+
- name: Run documentation doctests
|
|
45
|
+
run: python -m sphinx -b doctest docs/source docs/_build/doctest
|
|
@@ -5,6 +5,25 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
|
5
5
|
|
|
6
6
|
## [Unreleased]
|
|
7
7
|
|
|
8
|
+
## [1.7.2] - 2026-07-21
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
- Continuous-integration workflow (`tests.yaml`) running the unit tests on
|
|
12
|
+
Python 3.10–3.12 and the documentation doctests on every push and pull
|
|
13
|
+
request. Previously tests ran only on tagged releases.
|
|
14
|
+
- README now documents PDBx/mmCIF parsing (with a worked example) and carries
|
|
15
|
+
version, Python-versions, license, tests, docs, and downloads badges.
|
|
16
|
+
|
|
17
|
+
### Changed
|
|
18
|
+
- Documentation overhauled from a single quickstart page into a multi-page User
|
|
19
|
+
Guide — loading structures, the parsed data model, working with records,
|
|
20
|
+
biological assemblies and symmetry, PDBx/mmCIF, large structures, nonconformant
|
|
21
|
+
files, and advanced/customization — plus a supported-record-types coverage
|
|
22
|
+
table. Numerous example corrections; every doctest is now validated
|
|
23
|
+
(`sphinx -b doctest`), with `NORMALIZE_WHITESPACE` enabled so pretty-printed
|
|
24
|
+
output validates on content rather than incidental spacing. The docs landing
|
|
25
|
+
page is retitled from "Welcome to Pidibble's documentation!" to "pidibble".
|
|
26
|
+
|
|
8
27
|
## [1.7.1] - 2026-07-15
|
|
9
28
|
|
|
10
29
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pidibble
|
|
3
|
-
Version: 1.7.
|
|
3
|
+
Version: 1.7.2
|
|
4
4
|
Summary: A complete Protein Data Bank (PDB) file parser
|
|
5
5
|
Project-URL: Source, https://github.com/cameronabrams/pidibble
|
|
6
6
|
Project-URL: Documentation, https://pidibble.readthedocs.io/en/latest/
|
|
@@ -19,15 +19,20 @@ Requires-Dist: pytest; extra == 'test'
|
|
|
19
19
|
Description-Content-Type: text/markdown
|
|
20
20
|
|
|
21
21
|
# Pidibble
|
|
22
|
-
> a complete PDB
|
|
22
|
+
> a complete parser for PDB and PDBx/mmCIF files
|
|
23
23
|
|
|
24
|
+
[](https://pypi.org/project/pidibble/)
|
|
25
|
+
[](https://pypi.org/project/pidibble/)
|
|
26
|
+
[](https://github.com/cameronabrams/pidibble/blob/main/LICENSE)
|
|
27
|
+
[](https://github.com/cameronabrams/pidibble/actions/workflows/tests.yaml)
|
|
28
|
+
[](https://pidibble.readthedocs.io/en/latest/)
|
|
24
29
|
[](https://pepy.tech/projects/pidibble)
|
|
25
30
|
|
|
26
|
-
Pidibble is a Python package for parsing
|
|
31
|
+
Pidibble is a Python package for parsing Protein Data Bank structures in both the legacy [PDB format](https://www.wwpdb.org/documentation/file-format-content/format33/v3.3.html) (v.3.3 Atomic Coordinate Entry Format, ca. 2011) and the modern [PDBx/mmCIF format](https://mmcif.wwpdb.org/) that is now standard at the RCSB.
|
|
27
32
|
|
|
28
33
|
Unlike parsers like that found in packages like [BioPython](https://biopython.org/wiki/PDBParser), `pidibble` provides meaningfully parsed objects from *all* standard PDB record types, not just ATOMs and CONECTs.
|
|
29
34
|
|
|
30
|
-
Once installed, the user has access to the `PDBParser` class in the `pidibble.
|
|
35
|
+
Once installed, the user has access to the `PDBParser` class in the `pidibble.pdbparse` module. Full documentation is at [pidibble.readthedocs.io](https://pidibble.readthedocs.io/en/latest/).
|
|
31
36
|
|
|
32
37
|
# Example interactive usage
|
|
33
38
|
|
|
@@ -40,9 +45,11 @@ VIRAL PROTEIN
|
|
|
40
45
|
04-MAY-15
|
|
41
46
|
>>> print (p.parsed['HEADER'].idCode)
|
|
42
47
|
4ZMJ
|
|
43
|
-
>>> keys=
|
|
44
|
-
>>>
|
|
45
|
-
|
|
48
|
+
>>> keys=sorted(p.parsed.keys())
|
|
49
|
+
>>> len(keys)
|
|
50
|
+
60
|
|
51
|
+
>>> keys[:8]
|
|
52
|
+
['ANISOU', 'ATOM', 'AUTHOR', 'CISPEP', 'COMPND', 'CONECT', 'CRYST1', 'DBREF']
|
|
46
53
|
>>> header=p.parsed['HEADER']
|
|
47
54
|
>>> print(header.pstr())
|
|
48
55
|
HEADER
|
|
@@ -72,7 +79,7 @@ ATOM
|
|
|
72
79
|
|
|
73
80
|
```
|
|
74
81
|
>>> from pidibble.pdbparse import PDBParser
|
|
75
|
-
>>> p=PDBParser(alphafold='O46077').parse()
|
|
82
|
+
>>> p=PDBParser(source_db='alphafold', source_id='O46077').parse()
|
|
76
83
|
>>> p.parsed['TITLE'].title
|
|
77
84
|
'ALPHAFOLD MONOMER V2.0 PREDICTION FOR ODORANT RECEPTOR 2A (O46077)'
|
|
78
85
|
>>> print(p.parsed['ATOM'][0].pstr())
|
|
@@ -92,6 +99,23 @@ ATOM
|
|
|
92
99
|
|
|
93
100
|
```
|
|
94
101
|
|
|
102
|
+
# Example reading a PDBx/mmCIF file
|
|
103
|
+
|
|
104
|
+
Many recent RCSB entries have no legacy PDB file at all. Pass `input_format='mmCIF'` and pidibble parses the mmCIF/PDBx file into the *same* record objects (using author numbering), so downstream code stays the same:
|
|
105
|
+
|
|
106
|
+
```
|
|
107
|
+
>>> from pidibble.pdbparse import PDBParser
|
|
108
|
+
>>> p=PDBParser(source_db='rcsb', source_id='4tvp', input_format='mmCIF').parse()
|
|
109
|
+
>>> p.parsed['HEADER'].idCode
|
|
110
|
+
'4TVP'
|
|
111
|
+
>>> p.parsed['HEADER'].classification
|
|
112
|
+
'VIRAL PROTEIN/IMMUNE SYSTEM'
|
|
113
|
+
>>> len(p.parsed['ATOM'])
|
|
114
|
+
11344
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
See the [mmCIF guide](https://pidibble.readthedocs.io/en/latest/guide/mmcif.html) for the full coverage and the small, deliberate differences from the PDB parse.
|
|
118
|
+
|
|
95
119
|
## Changelog
|
|
96
120
|
|
|
97
121
|
See [CHANGELOG.md](CHANGELOG.md) for the full release history.
|
pidibble-1.7.2/README.md
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# Pidibble
|
|
2
|
+
> a complete parser for PDB and PDBx/mmCIF files
|
|
3
|
+
|
|
4
|
+
[](https://pypi.org/project/pidibble/)
|
|
5
|
+
[](https://pypi.org/project/pidibble/)
|
|
6
|
+
[](https://github.com/cameronabrams/pidibble/blob/main/LICENSE)
|
|
7
|
+
[](https://github.com/cameronabrams/pidibble/actions/workflows/tests.yaml)
|
|
8
|
+
[](https://pidibble.readthedocs.io/en/latest/)
|
|
9
|
+
[](https://pepy.tech/projects/pidibble)
|
|
10
|
+
|
|
11
|
+
Pidibble is a Python package for parsing Protein Data Bank structures in both the legacy [PDB format](https://www.wwpdb.org/documentation/file-format-content/format33/v3.3.html) (v.3.3 Atomic Coordinate Entry Format, ca. 2011) and the modern [PDBx/mmCIF format](https://mmcif.wwpdb.org/) that is now standard at the RCSB.
|
|
12
|
+
|
|
13
|
+
Unlike parsers like that found in packages like [BioPython](https://biopython.org/wiki/PDBParser), `pidibble` provides meaningfully parsed objects from *all* standard PDB record types, not just ATOMs and CONECTs.
|
|
14
|
+
|
|
15
|
+
Once installed, the user has access to the `PDBParser` class in the `pidibble.pdbparse` module. Full documentation is at [pidibble.readthedocs.io](https://pidibble.readthedocs.io/en/latest/).
|
|
16
|
+
|
|
17
|
+
# Example interactive usage
|
|
18
|
+
|
|
19
|
+
```
|
|
20
|
+
>>> from pidibble.pdbparse import PDBParser
|
|
21
|
+
>>> p=PDBParser(PDBcode='4zmj').parse()
|
|
22
|
+
>>> print (p.parsed['HEADER'].classification)
|
|
23
|
+
VIRAL PROTEIN
|
|
24
|
+
>>> print (p.parsed['HEADER'].depDate)
|
|
25
|
+
04-MAY-15
|
|
26
|
+
>>> print (p.parsed['HEADER'].idCode)
|
|
27
|
+
4ZMJ
|
|
28
|
+
>>> keys=sorted(p.parsed.keys())
|
|
29
|
+
>>> len(keys)
|
|
30
|
+
60
|
|
31
|
+
>>> keys[:8]
|
|
32
|
+
['ANISOU', 'ATOM', 'AUTHOR', 'CISPEP', 'COMPND', 'CONECT', 'CRYST1', 'DBREF']
|
|
33
|
+
>>> header=p.parsed['HEADER']
|
|
34
|
+
>>> print(header.pstr())
|
|
35
|
+
HEADER
|
|
36
|
+
classification: VIRAL PROTEIN
|
|
37
|
+
depDate: 04-MAY-15
|
|
38
|
+
idCode: 4ZMJ
|
|
39
|
+
>>> atoms=p.parsed['ATOM']
|
|
40
|
+
>>> len(atoms)
|
|
41
|
+
4518
|
|
42
|
+
>>> print(atoms[0].pstr())
|
|
43
|
+
ATOM
|
|
44
|
+
serial: 1
|
|
45
|
+
name: N
|
|
46
|
+
altLoc:
|
|
47
|
+
residue: resName: LEU; chainID: G; seqNum: 34; iCode:
|
|
48
|
+
x: -0.092
|
|
49
|
+
y: 99.33
|
|
50
|
+
z: 57.967
|
|
51
|
+
occupancy: 1.0
|
|
52
|
+
tempFactor: 137.71
|
|
53
|
+
element: N
|
|
54
|
+
charge:
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
# Example downloading from AlphaFold
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
>>> from pidibble.pdbparse import PDBParser
|
|
62
|
+
>>> p=PDBParser(source_db='alphafold', source_id='O46077').parse()
|
|
63
|
+
>>> p.parsed['TITLE'].title
|
|
64
|
+
'ALPHAFOLD MONOMER V2.0 PREDICTION FOR ODORANT RECEPTOR 2A (O46077)'
|
|
65
|
+
>>> print(p.parsed['ATOM'][0].pstr())
|
|
66
|
+
ATOM
|
|
67
|
+
serial: 1
|
|
68
|
+
name: N
|
|
69
|
+
altLoc:
|
|
70
|
+
residue: resName: MET; chainID: A; seqNum: 1; iCode:
|
|
71
|
+
x: -0.553
|
|
72
|
+
y: 26.513
|
|
73
|
+
z: 23.174
|
|
74
|
+
occupancy: 1.0
|
|
75
|
+
tempFactor: 39.74
|
|
76
|
+
element: N
|
|
77
|
+
charge:
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
# Example reading a PDBx/mmCIF file
|
|
83
|
+
|
|
84
|
+
Many recent RCSB entries have no legacy PDB file at all. Pass `input_format='mmCIF'` and pidibble parses the mmCIF/PDBx file into the *same* record objects (using author numbering), so downstream code stays the same:
|
|
85
|
+
|
|
86
|
+
```
|
|
87
|
+
>>> from pidibble.pdbparse import PDBParser
|
|
88
|
+
>>> p=PDBParser(source_db='rcsb', source_id='4tvp', input_format='mmCIF').parse()
|
|
89
|
+
>>> p.parsed['HEADER'].idCode
|
|
90
|
+
'4TVP'
|
|
91
|
+
>>> p.parsed['HEADER'].classification
|
|
92
|
+
'VIRAL PROTEIN/IMMUNE SYSTEM'
|
|
93
|
+
>>> len(p.parsed['ATOM'])
|
|
94
|
+
11344
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
See the [mmCIF guide](https://pidibble.readthedocs.io/en/latest/guide/mmcif.html) for the full coverage and the small, deliberate differences from the PDB parse.
|
|
98
|
+
|
|
99
|
+
## Changelog
|
|
100
|
+
|
|
101
|
+
See [CHANGELOG.md](CHANGELOG.md) for the full release history.
|
|
@@ -2,12 +2,21 @@
|
|
|
2
2
|
|
|
3
3
|
# -- Project information
|
|
4
4
|
|
|
5
|
+
import doctest
|
|
5
6
|
import importlib.metadata
|
|
6
7
|
|
|
8
|
+
# Pretty-printed record output (``pstr()``) carries trailing spaces and a final
|
|
9
|
+
# blank line, and numpy array reprs pad columns in version-dependent ways.
|
|
10
|
+
# Normalizing whitespace lets the doctests validate content without being held
|
|
11
|
+
# hostage to invisible spacing.
|
|
12
|
+
doctest_default_flags = (doctest.ELLIPSIS
|
|
13
|
+
| doctest.NORMALIZE_WHITESPACE
|
|
14
|
+
| doctest.IGNORE_EXCEPTION_DETAIL)
|
|
15
|
+
|
|
7
16
|
project = 'pidibble'
|
|
8
17
|
release = importlib.metadata.version(project)
|
|
9
18
|
version = '.'.join(release.split('.')[:2]) # major.minor
|
|
10
|
-
copyright = '2023-
|
|
19
|
+
copyright = '2023-2026, Cameron F. Abrams'
|
|
11
20
|
author = 'cfa22@drexel.edu'
|
|
12
21
|
|
|
13
22
|
# -- General configuration
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
Advanced usage and customization
|
|
2
|
+
================================
|
|
3
|
+
|
|
4
|
+
pidibble is driven by a declarative description of the PDB and mmCIF formats, and
|
|
5
|
+
several constructor arguments let you extend or override it.
|
|
6
|
+
|
|
7
|
+
Custom field mappers
|
|
8
|
+
--------------------
|
|
9
|
+
|
|
10
|
+
Every field in the format specification names a *type*, and each type maps to a
|
|
11
|
+
callable that turns the raw column text into a Python value. The built-in types
|
|
12
|
+
are ``String``, ``Integer``, ``Float`` and ``HxInteger`` (the hexadecimal-aware
|
|
13
|
+
atom-serial type), plus a family of list parsers.
|
|
14
|
+
|
|
15
|
+
You can add or override these with the ``mappers`` argument — a dict from type
|
|
16
|
+
name to a one-argument callable. This is the hook for a custom type used by a
|
|
17
|
+
custom format, or for changing how an existing type is coerced:
|
|
18
|
+
|
|
19
|
+
.. code-block:: python
|
|
20
|
+
|
|
21
|
+
def tenths(text):
|
|
22
|
+
return round(float(text), 1)
|
|
23
|
+
|
|
24
|
+
p = PDBParser(source_db='rcsb', source_id='4zmj',
|
|
25
|
+
mappers={'Float': tenths}).parse()
|
|
26
|
+
|
|
27
|
+
Your mappers are merged over the defaults, so you only supply what you want to
|
|
28
|
+
change.
|
|
29
|
+
|
|
30
|
+
Comment characters
|
|
31
|
+
------------------
|
|
32
|
+
|
|
33
|
+
By default, lines beginning with ``#`` are treated as comments and skipped. Pass
|
|
34
|
+
``comment_chars`` to change that set — for example, to also ignore ``!`` lines:
|
|
35
|
+
|
|
36
|
+
.. code-block:: python
|
|
37
|
+
|
|
38
|
+
p = PDBParser(filepath='annotated.pdb', comment_chars=['#', '!']).parse()
|
|
39
|
+
|
|
40
|
+
Overriding the format specification
|
|
41
|
+
-----------------------------------
|
|
42
|
+
|
|
43
|
+
The PDB and mmCIF record definitions are shipped as YAML resources
|
|
44
|
+
(``pdb_format.yaml`` and ``mmcif_format.yaml``). To parse a nonstandard dialect
|
|
45
|
+
or add a record type, point the parser at your own files:
|
|
46
|
+
|
|
47
|
+
.. code-block:: python
|
|
48
|
+
|
|
49
|
+
p = PDBParser(filepath='custom.pdb',
|
|
50
|
+
pdb_format_file='my_pdb_format.yaml').parse()
|
|
51
|
+
|
|
52
|
+
c = PDBParser(filepath='custom.cif', input_format='mmCIF',
|
|
53
|
+
mmcif_format_file='my_mmcif_format.yaml').parse()
|
|
54
|
+
|
|
55
|
+
The format-specification schema
|
|
56
|
+
-------------------------------
|
|
57
|
+
|
|
58
|
+
Each record type in ``pdb_format.yaml`` is described by a small set of keys:
|
|
59
|
+
|
|
60
|
+
``type``
|
|
61
|
+
The record's cardinality/shape: one-time-one-line, one-time-multiple-lines,
|
|
62
|
+
multiple-times-one-line, multiple-times-multiple-lines, grouping, or other.
|
|
63
|
+
|
|
64
|
+
``fields``
|
|
65
|
+
A mapping of field name to a ``[type, [start, end]]`` pair giving the field's
|
|
66
|
+
data type and its (1-based, inclusive) column range.
|
|
67
|
+
|
|
68
|
+
``continues``
|
|
69
|
+
The fields that a continuation record appends to or extends.
|
|
70
|
+
|
|
71
|
+
``token_formats``
|
|
72
|
+
How a field's text is broken into named tokens (with optional ``determinants``
|
|
73
|
+
that group token/value pairs) — this is what produces the token groups of
|
|
74
|
+
``COMPND`` and ``SOURCE``.
|
|
75
|
+
|
|
76
|
+
``concatenate``
|
|
77
|
+
New fields built by list-concatenating other fields.
|
|
78
|
+
|
|
79
|
+
``allowed``
|
|
80
|
+
Per-field lists of allowed values, for validation.
|
|
81
|
+
|
|
82
|
+
``determinants``
|
|
83
|
+
The fields whose values decide whether a line starts a new record instance or
|
|
84
|
+
continues the current one.
|
|
85
|
+
|
|
86
|
+
``subrecords``
|
|
87
|
+
How to parse variant sub-formats, selected by the value of a ``branchon``
|
|
88
|
+
field.
|
|
89
|
+
|
|
90
|
+
``tables``
|
|
91
|
+
How embedded tables (such as the ``BIOMT`` matrices) are parsed out.
|
|
92
|
+
|
|
93
|
+
Reusable sub-formats (for example the ``Residue`` variants and the ``Biomt`` row)
|
|
94
|
+
live under ``custom_formats`` and are referenced by name from field definitions.
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
Biological assemblies and symmetry
|
|
2
|
+
==================================
|
|
3
|
+
|
|
4
|
+
The biologically relevant molecule is often larger than the contents of the
|
|
5
|
+
asymmetric unit — it is generated by applying a set of rotation/translation
|
|
6
|
+
operators to some subset of chains. The PDB records this under ``REMARK 350``,
|
|
7
|
+
and pidibble parses each operator into its own indexed sub-record.
|
|
8
|
+
|
|
9
|
+
>>> from pidibble.pdbparse import PDBParser
|
|
10
|
+
>>> p = PDBParser(source_db='rcsb', source_id='4zmj').parse()
|
|
11
|
+
|
|
12
|
+
Assembly transforms
|
|
13
|
+
-------------------
|
|
14
|
+
|
|
15
|
+
Each transform for biological assembly 1 is a record keyed
|
|
16
|
+
``REMARK.350.BIOMOLECULE1.TRANSFORM<n>``:
|
|
17
|
+
|
|
18
|
+
>>> t1 = p.parsed['REMARK.350.BIOMOLECULE1.TRANSFORM1']
|
|
19
|
+
|
|
20
|
+
The ``header`` attribute of the *first* transform of an assembly lists the chains
|
|
21
|
+
the operators apply to:
|
|
22
|
+
|
|
23
|
+
>>> t1.header
|
|
24
|
+
['G', 'B', 'A', 'C', 'D']
|
|
25
|
+
|
|
26
|
+
Turning a transform into matrices
|
|
27
|
+
---------------------------------
|
|
28
|
+
|
|
29
|
+
The helper :func:`~pidibble.pdbparse.get_symm_ops` converts a transform record
|
|
30
|
+
into a 3×3 rotation matrix ``M`` and a translation vector ``T`` as
|
|
31
|
+
:class:`numpy.ndarray`\ s:
|
|
32
|
+
|
|
33
|
+
>>> from pidibble.pdbparse import get_symm_ops
|
|
34
|
+
>>> M, T = get_symm_ops(t1)
|
|
35
|
+
>>> M
|
|
36
|
+
array([[1., 0., 0.],
|
|
37
|
+
[0., 1., 0.],
|
|
38
|
+
[0., 0., 1.]])
|
|
39
|
+
>>> T
|
|
40
|
+
array([0., 0., 0.])
|
|
41
|
+
|
|
42
|
+
The first transform is the identity (the asymmetric unit itself); the others
|
|
43
|
+
build out the assembly. For 4ZMJ — a trimer — transforms 2 and 3 are the ±120°
|
|
44
|
+
rotations of a three-fold axis:
|
|
45
|
+
|
|
46
|
+
>>> M, T = get_symm_ops(p.parsed['REMARK.350.BIOMOLECULE1.TRANSFORM2'])
|
|
47
|
+
>>> M
|
|
48
|
+
array([[-0.5 , -0.866025, 0. ],
|
|
49
|
+
[ 0.866025, -0.5 , 0. ],
|
|
50
|
+
[ 0. , 0. , 1. ]])
|
|
51
|
+
>>> T
|
|
52
|
+
array([107.18 , 185.64121, 0. ])
|
|
53
|
+
|
|
54
|
+
Applying an operator to coordinates is then just ``M @ xyz + T``:
|
|
55
|
+
|
|
56
|
+
.. code-block:: python
|
|
57
|
+
|
|
58
|
+
import numpy as np
|
|
59
|
+
|
|
60
|
+
assembly = []
|
|
61
|
+
for key in sorted(k for k in p.parsed if k.startswith('REMARK.350.BIOMOLECULE1.TRANSFORM')):
|
|
62
|
+
M, T = get_symm_ops(p.parsed[key])
|
|
63
|
+
for atom in p.parsed['ATOM']:
|
|
64
|
+
if atom.residue.chainID in p.parsed['REMARK.350.BIOMOLECULE1.TRANSFORM1'].header:
|
|
65
|
+
xyz = np.array([atom.x, atom.y, atom.z])
|
|
66
|
+
assembly.append(M @ xyz + T)
|
|
67
|
+
|
|
68
|
+
Crystallographic symmetry
|
|
69
|
+
-------------------------
|
|
70
|
+
|
|
71
|
+
Crystallographic symmetry operators (the space-group operations under
|
|
72
|
+
``REMARK 290``) are parsed the same way, into
|
|
73
|
+
``REMARK.290.CRYSTSYMMTRANS.<n>`` sub-records that also work with
|
|
74
|
+
:func:`~pidibble.pdbparse.get_symm_ops`:
|
|
75
|
+
|
|
76
|
+
>>> ops = sorted(k for k in p.parsed if k.startswith('REMARK.290.CRYSTSYMMTRANS'))
|
|
77
|
+
>>> len(ops)
|
|
78
|
+
6
|
|
79
|
+
>>> M, T = get_symm_ops(p.parsed[ops[0]]) # the first operator is the identity
|
|
80
|
+
|
|
81
|
+
The unit cell and space group themselves are in ``CRYST1`` (with fields ``a``,
|
|
82
|
+
``b``, ``c``, ``alpha``, ``beta``, ``gamma``, ``sGroup`` and ``z``):
|
|
83
|
+
|
|
84
|
+
>>> c = p.parsed['CRYST1']
|
|
85
|
+
>>> c.sGroup
|
|
86
|
+
'P 63'
|
|
87
|
+
>>> c.a, c.b, c.c
|
|
88
|
+
(107.18, 107.18, 103.06)
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
The parsed data model
|
|
2
|
+
=====================
|
|
3
|
+
|
|
4
|
+
After :meth:`~pidibble.pdbparse.PDBParser.parse`, everything pidibble extracted
|
|
5
|
+
lives in one attribute:
|
|
6
|
+
|
|
7
|
+
>>> from pidibble.pdbparse import PDBParser
|
|
8
|
+
>>> p = PDBParser(source_db='rcsb', source_id='4zmj').parse()
|
|
9
|
+
>>> p.parsed # doctest: +ELLIPSIS
|
|
10
|
+
{...}
|
|
11
|
+
|
|
12
|
+
``parsed`` is a :class:`~pidibble.pdbrecord.PDBRecordDict` — a dictionary keyed
|
|
13
|
+
by record type.
|
|
14
|
+
|
|
15
|
+
Single records versus record lists
|
|
16
|
+
----------------------------------
|
|
17
|
+
|
|
18
|
+
Each value is one of two things:
|
|
19
|
+
|
|
20
|
+
* a single :class:`~pidibble.pdbrecord.PDBRecord`, for records that occur once
|
|
21
|
+
per structure (``HEADER``, ``TITLE``, ``CRYST1``, …); or
|
|
22
|
+
* a :class:`~pidibble.pdbrecord.PDBRecordList` of ``PDBRecord`` instances, for
|
|
23
|
+
*multiple-entry* records that recur (``ATOM``, ``SEQRES``, ``LINK``, …).
|
|
24
|
+
|
|
25
|
+
``PDBRecordList`` subclasses :class:`collections.UserList`, so test for it with
|
|
26
|
+
``isinstance`` — a plain ``type(v) == list`` check never matches, because a
|
|
27
|
+
``UserList`` is not a ``list``:
|
|
28
|
+
|
|
29
|
+
>>> from pidibble.pdbrecord import PDBRecordList
|
|
30
|
+
>>> isinstance(p.parsed['ATOM'], PDBRecordList) # a multiple-entry record
|
|
31
|
+
True
|
|
32
|
+
>>> isinstance(p.parsed['HEADER'], PDBRecordList) # a single-instance record
|
|
33
|
+
False
|
|
34
|
+
>>> [k for k, v in p.parsed.items() if type(v) == list] # the naive check never matches
|
|
35
|
+
[]
|
|
36
|
+
|
|
37
|
+
For everyday use, a ``PDBRecordList`` behaves like a list — index it, slice it,
|
|
38
|
+
iterate it, take its ``len``:
|
|
39
|
+
|
|
40
|
+
>>> atoms = p.parsed['ATOM']
|
|
41
|
+
>>> len(atoms)
|
|
42
|
+
4518
|
|
43
|
+
>>> first, last = atoms[0], atoms[-1]
|
|
44
|
+
|
|
45
|
+
Which records land in which bucket is documented per record type in
|
|
46
|
+
:doc:`record_reference`.
|
|
47
|
+
|
|
48
|
+
Records and their fields
|
|
49
|
+
------------------------
|
|
50
|
+
|
|
51
|
+
A :class:`~pidibble.pdbrecord.PDBRecord` stores each parsed field as an instance
|
|
52
|
+
attribute of the appropriate Python type — strings, ``int``, ``float``, or
|
|
53
|
+
nested record objects. The quickest way to see what a record contains is
|
|
54
|
+
:meth:`~pidibble.baserecord.BaseRecord.pstr` ("pretty string"):
|
|
55
|
+
|
|
56
|
+
>>> print(p.parsed['HEADER'].pstr())
|
|
57
|
+
HEADER
|
|
58
|
+
classification: VIRAL PROTEIN
|
|
59
|
+
depDate: 04-MAY-15
|
|
60
|
+
idCode: 4ZMJ
|
|
61
|
+
|
|
62
|
+
The keys shown by ``pstr()`` are exactly the attribute names:
|
|
63
|
+
|
|
64
|
+
>>> h = p.parsed['HEADER']
|
|
65
|
+
>>> h.classification, h.depDate, h.idCode
|
|
66
|
+
('VIRAL PROTEIN', '04-MAY-15', '4ZMJ')
|
|
67
|
+
|
|
68
|
+
``pstr()`` takes an ``excludes`` list (bookkeeping fields ``key``, ``format`` and
|
|
69
|
+
``continuation`` are hidden by default) and a ``pad`` width for the labels, so
|
|
70
|
+
you can widen or narrow the display or reveal the hidden internals.
|
|
71
|
+
|
|
72
|
+
Residues and other structured sub-fields
|
|
73
|
+
-----------------------------------------
|
|
74
|
+
|
|
75
|
+
Some fields are not scalars but small structured objects. The most common is a
|
|
76
|
+
**residue**, which groups a residue name, chain id, sequence number and
|
|
77
|
+
insertion code:
|
|
78
|
+
|
|
79
|
+
>>> a = p.parsed['ATOM'][0]
|
|
80
|
+
>>> print(a.residue) # doctest: +NORMALIZE_WHITESPACE
|
|
81
|
+
resName: LEU; chainID: G; seqNum: 34; iCode:
|
|
82
|
+
>>> a.residue.resName, a.residue.chainID, a.residue.seqNum
|
|
83
|
+
('LEU', 'G', 34)
|
|
84
|
+
|
|
85
|
+
Records that describe a *relationship between* residues expose more than one:
|
|
86
|
+
|
|
87
|
+
>>> b = p.parsed['SSBOND'][0]
|
|
88
|
+
>>> b.residue1.resName, b.residue1.chainID, b.residue1.seqNum
|
|
89
|
+
('CYS', 'G', 54)
|
|
90
|
+
>>> b.residue2.resName, b.residue2.chainID, b.residue2.seqNum
|
|
91
|
+
('CYS', 'G', 74)
|
|
92
|
+
|
|
93
|
+
Working with the model as a whole
|
|
94
|
+
---------------------------------
|
|
95
|
+
|
|
96
|
+
Because ``parsed`` is an ordinary mapping, the usual idioms apply — membership
|
|
97
|
+
tests, iteration, comprehensions:
|
|
98
|
+
|
|
99
|
+
>>> 'ATOM' in p.parsed and 'HETATM' in p.parsed
|
|
100
|
+
True
|
|
101
|
+
>>> n_hetatm = len(p.parsed['HETATM'])
|
|
102
|
+
>>> chains = {a.residue.chainID for a in p.parsed['ATOM']}
|
|
103
|
+
>>> sorted(chains)
|
|
104
|
+
['B', 'G']
|
|
105
|
+
|
|
106
|
+
A record type that was not present in the file is simply absent from ``parsed``,
|
|
107
|
+
so guard with ``in`` (or ``dict.get``) before reaching for an optional record:
|
|
108
|
+
|
|
109
|
+
>>> 'SPLIT' in p.parsed
|
|
110
|
+
False
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
.. _user-guide:
|
|
2
|
+
|
|
3
|
+
User Guide
|
|
4
|
+
==========
|
|
5
|
+
|
|
6
|
+
The guide works through pidibble one topic at a time. If you just want a taste,
|
|
7
|
+
start with the :doc:`../quickstart`.
|
|
8
|
+
|
|
9
|
+
.. toctree::
|
|
10
|
+
:maxdepth: 2
|
|
11
|
+
|
|
12
|
+
loading
|
|
13
|
+
data_model
|
|
14
|
+
records
|
|
15
|
+
assemblies
|
|
16
|
+
mmcif
|
|
17
|
+
large_structures
|
|
18
|
+
nonconformance
|
|
19
|
+
advanced
|
|
20
|
+
record_reference
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
Large structures and hexadecimal serials
|
|
2
|
+
=========================================
|
|
3
|
+
|
|
4
|
+
The legacy PDB format allots five columns to the atom serial number, so it can
|
|
5
|
+
only count up to ``99999``. Structures with more atoms than that continue the
|
|
6
|
+
count in **hexadecimal** (``100000`` is written ``186A0``), a widely used
|
|
7
|
+
convention that a naive integer parse would get wrong.
|
|
8
|
+
|
|
9
|
+
pidibble handles this automatically — there is nothing you need to configure.
|
|
10
|
+
Each :class:`~pidibble.pdbparse.PDBParser` owns an
|
|
11
|
+
:class:`~pidibble.hex.AtomSerialParser` that watches the atom serial column and
|
|
12
|
+
switches to hexadecimal parsing at the right moment, so ``serial`` (and the
|
|
13
|
+
serial references in ``CONECT``, ``TER`` and ``ANISOU``) stay correct past
|
|
14
|
+
``99999``:
|
|
15
|
+
|
|
16
|
+
.. code-block:: python
|
|
17
|
+
|
|
18
|
+
p = PDBParser(source_db='rcsb', source_id='<large-entry>').parse()
|
|
19
|
+
p.parsed['ATOM'][100000].serial # a correct integer, not a mis-parsed hex string
|
|
20
|
+
|
|
21
|
+
How the switch is detected
|
|
22
|
+
--------------------------
|
|
23
|
+
|
|
24
|
+
The transition to hexadecimal is recognized two ways, so it works whether or not
|
|
25
|
+
the early serial numbers happen to contain the digits ``a``–``f``:
|
|
26
|
+
|
|
27
|
+
* **By content** — as soon as a serial contains a hexadecimal letter, subsequent
|
|
28
|
+
serials are read as hexadecimal.
|
|
29
|
+
* **By magnitude** — if the running value exceeds ``99999``, the parser trips into
|
|
30
|
+
hexadecimal mode even for an all-numeric field.
|
|
31
|
+
|
|
32
|
+
The switch is *latching*: once tripped it stays in hexadecimal mode for the rest
|
|
33
|
+
of the parse. Because the state lives on the parser instance (not in a global),
|
|
34
|
+
parsing several structures in the same program — or in several threads — never
|
|
35
|
+
lets one file's numbering leak into another's.
|
|
36
|
+
|
|
37
|
+
.. note::
|
|
38
|
+
|
|
39
|
+
Hexadecimal detection is applied **only** to atom-serial fields. Ordinary
|
|
40
|
+
integer fields (residue sequence numbers, counts, and the like) are always
|
|
41
|
+
read as decimal, including legitimately negative values.
|