pidibble 1.7.1__tar.gz → 1.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. pidibble-1.7.2/.github/workflows/tests.yaml +45 -0
  2. {pidibble-1.7.1 → pidibble-1.7.2}/CHANGELOG.md +19 -0
  3. {pidibble-1.7.1 → pidibble-1.7.2}/PKG-INFO +32 -8
  4. pidibble-1.7.2/README.md +101 -0
  5. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/conf.py +10 -1
  6. pidibble-1.7.2/docs/source/guide/advanced.rst +94 -0
  7. pidibble-1.7.2/docs/source/guide/assemblies.rst +88 -0
  8. pidibble-1.7.2/docs/source/guide/data_model.rst +110 -0
  9. pidibble-1.7.2/docs/source/guide/index.rst +20 -0
  10. pidibble-1.7.2/docs/source/guide/large_structures.rst +41 -0
  11. pidibble-1.7.2/docs/source/guide/loading.rst +116 -0
  12. pidibble-1.7.2/docs/source/guide/mmcif.rst +103 -0
  13. pidibble-1.7.2/docs/source/guide/nonconformance.rst +61 -0
  14. pidibble-1.7.2/docs/source/guide/record_reference.rst +213 -0
  15. pidibble-1.7.2/docs/source/guide/records.rst +115 -0
  16. pidibble-1.7.2/docs/source/index.rst +61 -0
  17. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/installation.rst +1 -1
  18. pidibble-1.7.2/docs/source/quickstart.rst +86 -0
  19. {pidibble-1.7.1 → pidibble-1.7.2}/pyproject.toml +1 -1
  20. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb.py +23 -0
  21. pidibble-1.7.1/README.md +0 -77
  22. pidibble-1.7.1/docs/source/changelog.md +0 -5
  23. pidibble-1.7.1/docs/source/index.rst +0 -33
  24. pidibble-1.7.1/docs/source/notes.md +0 -34
  25. pidibble-1.7.1/docs/source/usage.rst +0 -174
  26. {pidibble-1.7.1 → pidibble-1.7.2}/.envrc +0 -0
  27. {pidibble-1.7.1 → pidibble-1.7.2}/.github/workflows/release.yaml +0 -0
  28. {pidibble-1.7.1 → pidibble-1.7.2}/.gitignore +0 -0
  29. {pidibble-1.7.1 → pidibble-1.7.2}/.readthedocs.yaml +0 -0
  30. {pidibble-1.7.1 → pidibble-1.7.2}/LICENSE +0 -0
  31. {pidibble-1.7.1 → pidibble-1.7.2}/docs/Makefile +0 -0
  32. {pidibble-1.7.1 → pidibble-1.7.2}/docs/make.bat +0 -0
  33. {pidibble-1.7.1 → pidibble-1.7.2}/docs/mmcif_coverage.md +0 -0
  34. {pidibble-1.7.1 → pidibble-1.7.2}/docs/requirements.txt +0 -0
  35. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/_static/css/custom.css +0 -0
  36. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/API.rst +0 -0
  37. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.baseparsers.rst +0 -0
  38. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.baserecord.rst +0 -0
  39. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.hex.rst +0 -0
  40. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.mmcif_parse.rst +0 -0
  41. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.pdbparse.rst +0 -0
  42. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.pdbrecord.rst +0 -0
  43. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.resources.rst +0 -0
  44. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/api/pidibble.rst +0 -0
  45. {pidibble-1.7.1 → pidibble-1.7.2}/docs/source/changelog.rst +0 -0
  46. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/__init__.py +0 -0
  47. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/baseparsers.py +0 -0
  48. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/baserecord.py +0 -0
  49. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/hex.py +0 -0
  50. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/mmcif_parse.py +0 -0
  51. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/pdbparse.py +0 -0
  52. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/pdbrecord.py +0 -0
  53. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/resources/mmcif_format.yaml +0 -0
  54. {pidibble-1.7.1 → pidibble-1.7.2}/pidibble/resources/pdb_format.yaml +0 -0
  55. {pidibble-1.7.1 → pidibble-1.7.2}/scripts/release.sh +0 -0
  56. {pidibble-1.7.1 → pidibble-1.7.2}/tests/__init__.py +0 -0
  57. {pidibble-1.7.1 → pidibble-1.7.2}/tests/conftest.py +0 -0
  58. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_hex/my_system.pdb +0 -0
  59. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_hex.py +0 -0
  60. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_nonconformance.py +0 -0
  61. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/1ca2.cif +0 -0
  62. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/1ca2.pdb +0 -0
  63. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4tvp.cif +0 -0
  64. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4tvp.pdb +0 -0
  65. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4zmj-newresnames.pdb +0 -0
  66. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4zmj.cif +0 -0
  67. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/4zmj.pdb +0 -0
  68. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/6m0j.pdb +0 -0
  69. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/8fae.cif +0 -0
  70. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/G.pdb +0 -0
  71. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/GG.pdb +0 -0
  72. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/test.pdb +0 -0
  73. {pidibble-1.7.1 → pidibble-1.7.2}/tests/unit/test_rcsb/test_pdb_format.yaml +0 -0
@@ -0,0 +1,45 @@
1
+ name: Tests
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+ workflow_dispatch:
8
+
9
+ jobs:
10
+ pytest:
11
+ name: pytest (py${{ matrix.python-version }})
12
+ runs-on: ubuntu-latest
13
+ strategy:
14
+ fail-fast: false
15
+ matrix:
16
+ python-version: ["3.10", "3.11", "3.12"]
17
+ steps:
18
+ - uses: actions/checkout@v4
19
+ - name: Set up Python ${{ matrix.python-version }}
20
+ uses: actions/setup-python@v5
21
+ with:
22
+ python-version: ${{ matrix.python-version }}
23
+ - name: Install package with test extras
24
+ run: |
25
+ python -m pip install --upgrade pip
26
+ pip install -e .[test]
27
+ - name: Run unit tests
28
+ run: pytest tests/unit -q
29
+
30
+ doctest:
31
+ name: doctest (docs examples)
32
+ runs-on: ubuntu-latest
33
+ steps:
34
+ - uses: actions/checkout@v4
35
+ - name: Set up Python
36
+ uses: actions/setup-python@v5
37
+ with:
38
+ python-version: "3.11"
39
+ - name: Install docs requirements and local package
40
+ run: |
41
+ python -m pip install --upgrade pip
42
+ pip install -r docs/requirements.txt
43
+ pip install -e . # test the working tree, not the PyPI build
44
+ - name: Run documentation doctests
45
+ run: python -m sphinx -b doctest docs/source docs/_build/doctest
@@ -5,6 +5,25 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
5
5
 
6
6
  ## [Unreleased]
7
7
 
8
+ ## [1.7.2] - 2026-07-21
9
+
10
+ ### Added
11
+ - Continuous-integration workflow (`tests.yaml`) running the unit tests on
12
+ Python 3.10–3.12 and the documentation doctests on every push and pull
13
+ request. Previously tests ran only on tagged releases.
14
+ - README now documents PDBx/mmCIF parsing (with a worked example) and carries
15
+ version, Python-versions, license, tests, docs, and downloads badges.
16
+
17
+ ### Changed
18
+ - Documentation overhauled from a single quickstart page into a multi-page User
19
+ Guide — loading structures, the parsed data model, working with records,
20
+ biological assemblies and symmetry, PDBx/mmCIF, large structures, nonconformant
21
+ files, and advanced/customization — plus a supported-record-types coverage
22
+ table. Numerous example corrections; every doctest is now validated
23
+ (`sphinx -b doctest`), with `NORMALIZE_WHITESPACE` enabled so pretty-printed
24
+ output validates on content rather than incidental spacing. The docs landing
25
+ page is retitled from "Welcome to Pidibble's documentation!" to "pidibble".
26
+
8
27
  ## [1.7.1] - 2026-07-15
9
28
 
10
29
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pidibble
3
- Version: 1.7.1
3
+ Version: 1.7.2
4
4
  Summary: A complete Protein Data Bank (PDB) file parser
5
5
  Project-URL: Source, https://github.com/cameronabrams/pidibble
6
6
  Project-URL: Documentation, https://pidibble.readthedocs.io/en/latest/
@@ -19,15 +19,20 @@ Requires-Dist: pytest; extra == 'test'
19
19
  Description-Content-Type: text/markdown
20
20
 
21
21
  # Pidibble
22
- > a complete PDB-file parser
22
+ > a complete parser for PDB and PDBx/mmCIF files
23
23
 
24
+ [![PyPI version](https://img.shields.io/pypi/v/pidibble.svg)](https://pypi.org/project/pidibble/)
25
+ [![Python versions](https://img.shields.io/pypi/pyversions/pidibble.svg)](https://pypi.org/project/pidibble/)
26
+ [![License](https://img.shields.io/pypi/l/pidibble.svg)](https://github.com/cameronabrams/pidibble/blob/main/LICENSE)
27
+ [![Tests](https://github.com/cameronabrams/pidibble/actions/workflows/tests.yaml/badge.svg)](https://github.com/cameronabrams/pidibble/actions/workflows/tests.yaml)
28
+ [![Documentation Status](https://readthedocs.org/projects/pidibble/badge/?version=latest)](https://pidibble.readthedocs.io/en/latest/)
24
29
  [![PyPI Downloads](https://static.pepy.tech/badge/pidibble)](https://pepy.tech/projects/pidibble)
25
30
 
26
- Pidibble is a Python package for parsing standard Protein Data Bank (PDB) files. It conforms to the [most recent standard](https://www.wwpdb.org/documentation/file-format-content/format33/v3.3.html) (v.3.3 Atomic Coordinate Entry Format, ca. 2011).
31
+ Pidibble is a Python package for parsing Protein Data Bank structures in both the legacy [PDB format](https://www.wwpdb.org/documentation/file-format-content/format33/v3.3.html) (v.3.3 Atomic Coordinate Entry Format, ca. 2011) and the modern [PDBx/mmCIF format](https://mmcif.wwpdb.org/) that is now standard at the RCSB.
27
32
 
28
33
  Unlike parsers like that found in packages like [BioPython](https://biopython.org/wiki/PDBParser), `pidibble` provides meaningfully parsed objects from *all* standard PDB record types, not just ATOMs and CONECTs.
29
34
 
30
- Once installed, the user has access to the `PDBParser` class in the `pidibble.pdbparser` module.
35
+ Once installed, the user has access to the `PDBParser` class in the `pidibble.pdbparse` module. Full documentation is at [pidibble.readthedocs.io](https://pidibble.readthedocs.io/en/latest/).
31
36
 
32
37
  # Example interactive usage
33
38
 
@@ -40,9 +45,11 @@ VIRAL PROTEIN
40
45
  04-MAY-15
41
46
  >>> print (p.parsed['HEADER'].idCode)
42
47
  4ZMJ
43
- >>> keys=list(sorted(list(p.parsed.keys())))
44
- >>> print(keys)
45
- ['ANISOU', 'ATOM', 'AUTHOR', 'CISPEP', 'COMPND', 'CONECT', 'CRYST1', 'DBREF', 'END', 'EXPDTA', 'FORMUL', 'HEADER', 'HELIX', 'HET', 'HETATM', 'HETNAM', 'JRNL.AUTH', 'JRNL.DOI', 'JRNL.PMID', 'JRNL.REF', 'JRNL.REFN', 'JRNL.TITL', 'KEYWDS', 'LINK', 'MASTER', 'ORIGX1', 'ORIGX2', 'ORIGX3', 'REMARK.100', 'REMARK.2', 'REMARK.200', 'REMARK.280', 'REMARK.290', 'REMARK.290.CRYSTSYMMTRANS', 'REMARK.3', 'REMARK.300', 'REMARK.350', 'REMARK.350.BIOMOLECULE.1', 'REMARK.4', 'REMARK.465', 'REMARK.500', 'REVDAT', 'SCALE1', 'SCALE2', 'SCALE3', 'SEQADV', 'SEQRES', 'SHEET', 'SOURCE', 'SSBOND', 'TER', 'TITLE']
48
+ >>> keys=sorted(p.parsed.keys())
49
+ >>> len(keys)
50
+ 60
51
+ >>> keys[:8]
52
+ ['ANISOU', 'ATOM', 'AUTHOR', 'CISPEP', 'COMPND', 'CONECT', 'CRYST1', 'DBREF']
46
53
  >>> header=p.parsed['HEADER']
47
54
  >>> print(header.pstr())
48
55
  HEADER
@@ -72,7 +79,7 @@ ATOM
72
79
 
73
80
  ```
74
81
  >>> from pidibble.pdbparse import PDBParser
75
- >>> p=PDBParser(alphafold='O46077').parse()
82
+ >>> p=PDBParser(source_db='alphafold', source_id='O46077').parse()
76
83
  >>> p.parsed['TITLE'].title
77
84
  'ALPHAFOLD MONOMER V2.0 PREDICTION FOR ODORANT RECEPTOR 2A (O46077)'
78
85
  >>> print(p.parsed['ATOM'][0].pstr())
@@ -92,6 +99,23 @@ ATOM
92
99
 
93
100
  ```
94
101
 
102
+ # Example reading a PDBx/mmCIF file
103
+
104
+ Many recent RCSB entries have no legacy PDB file at all. Pass `input_format='mmCIF'` and pidibble parses the mmCIF/PDBx file into the *same* record objects (using author numbering), so downstream code stays the same:
105
+
106
+ ```
107
+ >>> from pidibble.pdbparse import PDBParser
108
+ >>> p=PDBParser(source_db='rcsb', source_id='4tvp', input_format='mmCIF').parse()
109
+ >>> p.parsed['HEADER'].idCode
110
+ '4TVP'
111
+ >>> p.parsed['HEADER'].classification
112
+ 'VIRAL PROTEIN/IMMUNE SYSTEM'
113
+ >>> len(p.parsed['ATOM'])
114
+ 11344
115
+ ```
116
+
117
+ See the [mmCIF guide](https://pidibble.readthedocs.io/en/latest/guide/mmcif.html) for the full coverage and the small, deliberate differences from the PDB parse.
118
+
95
119
  ## Changelog
96
120
 
97
121
  See [CHANGELOG.md](CHANGELOG.md) for the full release history.
@@ -0,0 +1,101 @@
1
+ # Pidibble
2
+ > a complete parser for PDB and PDBx/mmCIF files
3
+
4
+ [![PyPI version](https://img.shields.io/pypi/v/pidibble.svg)](https://pypi.org/project/pidibble/)
5
+ [![Python versions](https://img.shields.io/pypi/pyversions/pidibble.svg)](https://pypi.org/project/pidibble/)
6
+ [![License](https://img.shields.io/pypi/l/pidibble.svg)](https://github.com/cameronabrams/pidibble/blob/main/LICENSE)
7
+ [![Tests](https://github.com/cameronabrams/pidibble/actions/workflows/tests.yaml/badge.svg)](https://github.com/cameronabrams/pidibble/actions/workflows/tests.yaml)
8
+ [![Documentation Status](https://readthedocs.org/projects/pidibble/badge/?version=latest)](https://pidibble.readthedocs.io/en/latest/)
9
+ [![PyPI Downloads](https://static.pepy.tech/badge/pidibble)](https://pepy.tech/projects/pidibble)
10
+
11
+ Pidibble is a Python package for parsing Protein Data Bank structures in both the legacy [PDB format](https://www.wwpdb.org/documentation/file-format-content/format33/v3.3.html) (v.3.3 Atomic Coordinate Entry Format, ca. 2011) and the modern [PDBx/mmCIF format](https://mmcif.wwpdb.org/) that is now standard at the RCSB.
12
+
13
+ Unlike parsers like that found in packages like [BioPython](https://biopython.org/wiki/PDBParser), `pidibble` provides meaningfully parsed objects from *all* standard PDB record types, not just ATOMs and CONECTs.
14
+
15
+ Once installed, the user has access to the `PDBParser` class in the `pidibble.pdbparse` module. Full documentation is at [pidibble.readthedocs.io](https://pidibble.readthedocs.io/en/latest/).
16
+
17
+ # Example interactive usage
18
+
19
+ ```
20
+ >>> from pidibble.pdbparse import PDBParser
21
+ >>> p=PDBParser(PDBcode='4zmj').parse()
22
+ >>> print (p.parsed['HEADER'].classification)
23
+ VIRAL PROTEIN
24
+ >>> print (p.parsed['HEADER'].depDate)
25
+ 04-MAY-15
26
+ >>> print (p.parsed['HEADER'].idCode)
27
+ 4ZMJ
28
+ >>> keys=sorted(p.parsed.keys())
29
+ >>> len(keys)
30
+ 60
31
+ >>> keys[:8]
32
+ ['ANISOU', 'ATOM', 'AUTHOR', 'CISPEP', 'COMPND', 'CONECT', 'CRYST1', 'DBREF']
33
+ >>> header=p.parsed['HEADER']
34
+ >>> print(header.pstr())
35
+ HEADER
36
+ classification: VIRAL PROTEIN
37
+ depDate: 04-MAY-15
38
+ idCode: 4ZMJ
39
+ >>> atoms=p.parsed['ATOM']
40
+ >>> len(atoms)
41
+ 4518
42
+ >>> print(atoms[0].pstr())
43
+ ATOM
44
+ serial: 1
45
+ name: N
46
+ altLoc:
47
+ residue: resName: LEU; chainID: G; seqNum: 34; iCode:
48
+ x: -0.092
49
+ y: 99.33
50
+ z: 57.967
51
+ occupancy: 1.0
52
+ tempFactor: 137.71
53
+ element: N
54
+ charge:
55
+
56
+ ```
57
+
58
+ # Example downloading from AlphaFold
59
+
60
+ ```
61
+ >>> from pidibble.pdbparse import PDBParser
62
+ >>> p=PDBParser(source_db='alphafold', source_id='O46077').parse()
63
+ >>> p.parsed['TITLE'].title
64
+ 'ALPHAFOLD MONOMER V2.0 PREDICTION FOR ODORANT RECEPTOR 2A (O46077)'
65
+ >>> print(p.parsed['ATOM'][0].pstr())
66
+ ATOM
67
+ serial: 1
68
+ name: N
69
+ altLoc:
70
+ residue: resName: MET; chainID: A; seqNum: 1; iCode:
71
+ x: -0.553
72
+ y: 26.513
73
+ z: 23.174
74
+ occupancy: 1.0
75
+ tempFactor: 39.74
76
+ element: N
77
+ charge:
78
+
79
+
80
+ ```
81
+
82
+ # Example reading a PDBx/mmCIF file
83
+
84
+ Many recent RCSB entries have no legacy PDB file at all. Pass `input_format='mmCIF'` and pidibble parses the mmCIF/PDBx file into the *same* record objects (using author numbering), so downstream code stays the same:
85
+
86
+ ```
87
+ >>> from pidibble.pdbparse import PDBParser
88
+ >>> p=PDBParser(source_db='rcsb', source_id='4tvp', input_format='mmCIF').parse()
89
+ >>> p.parsed['HEADER'].idCode
90
+ '4TVP'
91
+ >>> p.parsed['HEADER'].classification
92
+ 'VIRAL PROTEIN/IMMUNE SYSTEM'
93
+ >>> len(p.parsed['ATOM'])
94
+ 11344
95
+ ```
96
+
97
+ See the [mmCIF guide](https://pidibble.readthedocs.io/en/latest/guide/mmcif.html) for the full coverage and the small, deliberate differences from the PDB parse.
98
+
99
+ ## Changelog
100
+
101
+ See [CHANGELOG.md](CHANGELOG.md) for the full release history.
@@ -2,12 +2,21 @@
2
2
 
3
3
  # -- Project information
4
4
 
5
+ import doctest
5
6
  import importlib.metadata
6
7
 
8
+ # Pretty-printed record output (``pstr()``) carries trailing spaces and a final
9
+ # blank line, and numpy array reprs pad columns in version-dependent ways.
10
+ # Normalizing whitespace lets the doctests validate content without being held
11
+ # hostage to invisible spacing.
12
+ doctest_default_flags = (doctest.ELLIPSIS
13
+ | doctest.NORMALIZE_WHITESPACE
14
+ | doctest.IGNORE_EXCEPTION_DETAIL)
15
+
7
16
  project = 'pidibble'
8
17
  release = importlib.metadata.version(project)
9
18
  version = '.'.join(release.split('.')[:2]) # major.minor
10
- copyright = '2023-2025, Cameron F. Abrams'
19
+ copyright = '2023-2026, Cameron F. Abrams'
11
20
  author = 'cfa22@drexel.edu'
12
21
 
13
22
  # -- General configuration
@@ -0,0 +1,94 @@
1
+ Advanced usage and customization
2
+ ================================
3
+
4
+ pidibble is driven by a declarative description of the PDB and mmCIF formats, and
5
+ several constructor arguments let you extend or override it.
6
+
7
+ Custom field mappers
8
+ --------------------
9
+
10
+ Every field in the format specification names a *type*, and each type maps to a
11
+ callable that turns the raw column text into a Python value. The built-in types
12
+ are ``String``, ``Integer``, ``Float`` and ``HxInteger`` (the hexadecimal-aware
13
+ atom-serial type), plus a family of list parsers.
14
+
15
+ You can add or override these with the ``mappers`` argument — a dict from type
16
+ name to a one-argument callable. This is the hook for a custom type used by a
17
+ custom format, or for changing how an existing type is coerced:
18
+
19
+ .. code-block:: python
20
+
21
+ def tenths(text):
22
+ return round(float(text), 1)
23
+
24
+ p = PDBParser(source_db='rcsb', source_id='4zmj',
25
+ mappers={'Float': tenths}).parse()
26
+
27
+ Your mappers are merged over the defaults, so you only supply what you want to
28
+ change.
29
+
30
+ Comment characters
31
+ ------------------
32
+
33
+ By default, lines beginning with ``#`` are treated as comments and skipped. Pass
34
+ ``comment_chars`` to change that set — for example, to also ignore ``!`` lines:
35
+
36
+ .. code-block:: python
37
+
38
+ p = PDBParser(filepath='annotated.pdb', comment_chars=['#', '!']).parse()
39
+
40
+ Overriding the format specification
41
+ -----------------------------------
42
+
43
+ The PDB and mmCIF record definitions are shipped as YAML resources
44
+ (``pdb_format.yaml`` and ``mmcif_format.yaml``). To parse a nonstandard dialect
45
+ or add a record type, point the parser at your own files:
46
+
47
+ .. code-block:: python
48
+
49
+ p = PDBParser(filepath='custom.pdb',
50
+ pdb_format_file='my_pdb_format.yaml').parse()
51
+
52
+ c = PDBParser(filepath='custom.cif', input_format='mmCIF',
53
+ mmcif_format_file='my_mmcif_format.yaml').parse()
54
+
55
+ The format-specification schema
56
+ -------------------------------
57
+
58
+ Each record type in ``pdb_format.yaml`` is described by a small set of keys:
59
+
60
+ ``type``
61
+ The record's cardinality/shape: one-time-one-line, one-time-multiple-lines,
62
+ multiple-times-one-line, multiple-times-multiple-lines, grouping, or other.
63
+
64
+ ``fields``
65
+ A mapping of field name to a ``[type, [start, end]]`` pair giving the field's
66
+ data type and its (1-based, inclusive) column range.
67
+
68
+ ``continues``
69
+ The fields that a continuation record appends to or extends.
70
+
71
+ ``token_formats``
72
+ How a field's text is broken into named tokens (with optional ``determinants``
73
+ that group token/value pairs) — this is what produces the token groups of
74
+ ``COMPND`` and ``SOURCE``.
75
+
76
+ ``concatenate``
77
+ New fields built by list-concatenating other fields.
78
+
79
+ ``allowed``
80
+ Per-field lists of allowed values, for validation.
81
+
82
+ ``determinants``
83
+ The fields whose values decide whether a line starts a new record instance or
84
+ continues the current one.
85
+
86
+ ``subrecords``
87
+ How to parse variant sub-formats, selected by the value of a ``branchon``
88
+ field.
89
+
90
+ ``tables``
91
+ How embedded tables (such as the ``BIOMT`` matrices) are parsed out.
92
+
93
+ Reusable sub-formats (for example the ``Residue`` variants and the ``Biomt`` row)
94
+ live under ``custom_formats`` and are referenced by name from field definitions.
@@ -0,0 +1,88 @@
1
+ Biological assemblies and symmetry
2
+ ==================================
3
+
4
+ The biologically relevant molecule is often larger than the contents of the
5
+ asymmetric unit — it is generated by applying a set of rotation/translation
6
+ operators to some subset of chains. The PDB records this under ``REMARK 350``,
7
+ and pidibble parses each operator into its own indexed sub-record.
8
+
9
+ >>> from pidibble.pdbparse import PDBParser
10
+ >>> p = PDBParser(source_db='rcsb', source_id='4zmj').parse()
11
+
12
+ Assembly transforms
13
+ -------------------
14
+
15
+ Each transform for biological assembly 1 is a record keyed
16
+ ``REMARK.350.BIOMOLECULE1.TRANSFORM<n>``:
17
+
18
+ >>> t1 = p.parsed['REMARK.350.BIOMOLECULE1.TRANSFORM1']
19
+
20
+ The ``header`` attribute of the *first* transform of an assembly lists the chains
21
+ the operators apply to:
22
+
23
+ >>> t1.header
24
+ ['G', 'B', 'A', 'C', 'D']
25
+
26
+ Turning a transform into matrices
27
+ ---------------------------------
28
+
29
+ The helper :func:`~pidibble.pdbparse.get_symm_ops` converts a transform record
30
+ into a 3×3 rotation matrix ``M`` and a translation vector ``T`` as
31
+ :class:`numpy.ndarray`\ s:
32
+
33
+ >>> from pidibble.pdbparse import get_symm_ops
34
+ >>> M, T = get_symm_ops(t1)
35
+ >>> M
36
+ array([[1., 0., 0.],
37
+ [0., 1., 0.],
38
+ [0., 0., 1.]])
39
+ >>> T
40
+ array([0., 0., 0.])
41
+
42
+ The first transform is the identity (the asymmetric unit itself); the others
43
+ build out the assembly. For 4ZMJ — a trimer — transforms 2 and 3 are the ±120°
44
+ rotations of a three-fold axis:
45
+
46
+ >>> M, T = get_symm_ops(p.parsed['REMARK.350.BIOMOLECULE1.TRANSFORM2'])
47
+ >>> M
48
+ array([[-0.5 , -0.866025, 0. ],
49
+ [ 0.866025, -0.5 , 0. ],
50
+ [ 0. , 0. , 1. ]])
51
+ >>> T
52
+ array([107.18 , 185.64121, 0. ])
53
+
54
+ Applying an operator to coordinates is then just ``M @ xyz + T``:
55
+
56
+ .. code-block:: python
57
+
58
+ import numpy as np
59
+
60
+ assembly = []
61
+ for key in sorted(k for k in p.parsed if k.startswith('REMARK.350.BIOMOLECULE1.TRANSFORM')):
62
+ M, T = get_symm_ops(p.parsed[key])
63
+ for atom in p.parsed['ATOM']:
64
+ if atom.residue.chainID in p.parsed['REMARK.350.BIOMOLECULE1.TRANSFORM1'].header:
65
+ xyz = np.array([atom.x, atom.y, atom.z])
66
+ assembly.append(M @ xyz + T)
67
+
68
+ Crystallographic symmetry
69
+ -------------------------
70
+
71
+ Crystallographic symmetry operators (the space-group operations under
72
+ ``REMARK 290``) are parsed the same way, into
73
+ ``REMARK.290.CRYSTSYMMTRANS.<n>`` sub-records that also work with
74
+ :func:`~pidibble.pdbparse.get_symm_ops`:
75
+
76
+ >>> ops = sorted(k for k in p.parsed if k.startswith('REMARK.290.CRYSTSYMMTRANS'))
77
+ >>> len(ops)
78
+ 6
79
+ >>> M, T = get_symm_ops(p.parsed[ops[0]]) # the first operator is the identity
80
+
81
+ The unit cell and space group themselves are in ``CRYST1`` (with fields ``a``,
82
+ ``b``, ``c``, ``alpha``, ``beta``, ``gamma``, ``sGroup`` and ``z``):
83
+
84
+ >>> c = p.parsed['CRYST1']
85
+ >>> c.sGroup
86
+ 'P 63'
87
+ >>> c.a, c.b, c.c
88
+ (107.18, 107.18, 103.06)
@@ -0,0 +1,110 @@
1
+ The parsed data model
2
+ =====================
3
+
4
+ After :meth:`~pidibble.pdbparse.PDBParser.parse`, everything pidibble extracted
5
+ lives in one attribute:
6
+
7
+ >>> from pidibble.pdbparse import PDBParser
8
+ >>> p = PDBParser(source_db='rcsb', source_id='4zmj').parse()
9
+ >>> p.parsed # doctest: +ELLIPSIS
10
+ {...}
11
+
12
+ ``parsed`` is a :class:`~pidibble.pdbrecord.PDBRecordDict` — a dictionary keyed
13
+ by record type.
14
+
15
+ Single records versus record lists
16
+ ----------------------------------
17
+
18
+ Each value is one of two things:
19
+
20
+ * a single :class:`~pidibble.pdbrecord.PDBRecord`, for records that occur once
21
+ per structure (``HEADER``, ``TITLE``, ``CRYST1``, …); or
22
+ * a :class:`~pidibble.pdbrecord.PDBRecordList` of ``PDBRecord`` instances, for
23
+ *multiple-entry* records that recur (``ATOM``, ``SEQRES``, ``LINK``, …).
24
+
25
+ ``PDBRecordList`` subclasses :class:`collections.UserList`, so test for it with
26
+ ``isinstance`` — a plain ``type(v) == list`` check never matches, because a
27
+ ``UserList`` is not a ``list``:
28
+
29
+ >>> from pidibble.pdbrecord import PDBRecordList
30
+ >>> isinstance(p.parsed['ATOM'], PDBRecordList) # a multiple-entry record
31
+ True
32
+ >>> isinstance(p.parsed['HEADER'], PDBRecordList) # a single-instance record
33
+ False
34
+ >>> [k for k, v in p.parsed.items() if type(v) == list] # the naive check never matches
35
+ []
36
+
37
+ For everyday use, a ``PDBRecordList`` behaves like a list — index it, slice it,
38
+ iterate it, take its ``len``:
39
+
40
+ >>> atoms = p.parsed['ATOM']
41
+ >>> len(atoms)
42
+ 4518
43
+ >>> first, last = atoms[0], atoms[-1]
44
+
45
+ Which records land in which bucket is documented per record type in
46
+ :doc:`record_reference`.
47
+
48
+ Records and their fields
49
+ ------------------------
50
+
51
+ A :class:`~pidibble.pdbrecord.PDBRecord` stores each parsed field as an instance
52
+ attribute of the appropriate Python type — strings, ``int``, ``float``, or
53
+ nested record objects. The quickest way to see what a record contains is
54
+ :meth:`~pidibble.baserecord.BaseRecord.pstr` ("pretty string"):
55
+
56
+ >>> print(p.parsed['HEADER'].pstr())
57
+ HEADER
58
+ classification: VIRAL PROTEIN
59
+ depDate: 04-MAY-15
60
+ idCode: 4ZMJ
61
+
62
+ The keys shown by ``pstr()`` are exactly the attribute names:
63
+
64
+ >>> h = p.parsed['HEADER']
65
+ >>> h.classification, h.depDate, h.idCode
66
+ ('VIRAL PROTEIN', '04-MAY-15', '4ZMJ')
67
+
68
+ ``pstr()`` takes an ``excludes`` list (bookkeeping fields ``key``, ``format`` and
69
+ ``continuation`` are hidden by default) and a ``pad`` width for the labels, so
70
+ you can widen or narrow the display or reveal the hidden internals.
71
+
72
+ Residues and other structured sub-fields
73
+ -----------------------------------------
74
+
75
+ Some fields are not scalars but small structured objects. The most common is a
76
+ **residue**, which groups a residue name, chain id, sequence number and
77
+ insertion code:
78
+
79
+ >>> a = p.parsed['ATOM'][0]
80
+ >>> print(a.residue) # doctest: +NORMALIZE_WHITESPACE
81
+ resName: LEU; chainID: G; seqNum: 34; iCode:
82
+ >>> a.residue.resName, a.residue.chainID, a.residue.seqNum
83
+ ('LEU', 'G', 34)
84
+
85
+ Records that describe a *relationship between* residues expose more than one:
86
+
87
+ >>> b = p.parsed['SSBOND'][0]
88
+ >>> b.residue1.resName, b.residue1.chainID, b.residue1.seqNum
89
+ ('CYS', 'G', 54)
90
+ >>> b.residue2.resName, b.residue2.chainID, b.residue2.seqNum
91
+ ('CYS', 'G', 74)
92
+
93
+ Working with the model as a whole
94
+ ---------------------------------
95
+
96
+ Because ``parsed`` is an ordinary mapping, the usual idioms apply — membership
97
+ tests, iteration, comprehensions:
98
+
99
+ >>> 'ATOM' in p.parsed and 'HETATM' in p.parsed
100
+ True
101
+ >>> n_hetatm = len(p.parsed['HETATM'])
102
+ >>> chains = {a.residue.chainID for a in p.parsed['ATOM']}
103
+ >>> sorted(chains)
104
+ ['B', 'G']
105
+
106
+ A record type that was not present in the file is simply absent from ``parsed``,
107
+ so guard with ``in`` (or ``dict.get``) before reaching for an optional record:
108
+
109
+ >>> 'SPLIT' in p.parsed
110
+ False
@@ -0,0 +1,20 @@
1
+ .. _user-guide:
2
+
3
+ User Guide
4
+ ==========
5
+
6
+ The guide works through pidibble one topic at a time. If you just want a taste,
7
+ start with the :doc:`../quickstart`.
8
+
9
+ .. toctree::
10
+ :maxdepth: 2
11
+
12
+ loading
13
+ data_model
14
+ records
15
+ assemblies
16
+ mmcif
17
+ large_structures
18
+ nonconformance
19
+ advanced
20
+ record_reference
@@ -0,0 +1,41 @@
1
+ Large structures and hexadecimal serials
2
+ =========================================
3
+
4
+ The legacy PDB format allots five columns to the atom serial number, so it can
5
+ only count up to ``99999``. Structures with more atoms than that continue the
6
+ count in **hexadecimal** (``100000`` is written ``186A0``), a widely used
7
+ convention that a naive integer parse would get wrong.
8
+
9
+ pidibble handles this automatically — there is nothing you need to configure.
10
+ Each :class:`~pidibble.pdbparse.PDBParser` owns an
11
+ :class:`~pidibble.hex.AtomSerialParser` that watches the atom serial column and
12
+ switches to hexadecimal parsing at the right moment, so ``serial`` (and the
13
+ serial references in ``CONECT``, ``TER`` and ``ANISOU``) stay correct past
14
+ ``99999``:
15
+
16
+ .. code-block:: python
17
+
18
+ p = PDBParser(source_db='rcsb', source_id='<large-entry>').parse()
19
+ p.parsed['ATOM'][100000].serial # a correct integer, not a mis-parsed hex string
20
+
21
+ How the switch is detected
22
+ --------------------------
23
+
24
+ The transition to hexadecimal is recognized two ways, so it works whether or not
25
+ the early serial numbers happen to contain the digits ``a``–``f``:
26
+
27
+ * **By content** — as soon as a serial contains a hexadecimal letter, subsequent
28
+ serials are read as hexadecimal.
29
+ * **By magnitude** — if the running value exceeds ``99999``, the parser trips into
30
+ hexadecimal mode even for an all-numeric field.
31
+
32
+ The switch is *latching*: once tripped it stays in hexadecimal mode for the rest
33
+ of the parse. Because the state lives on the parser instance (not in a global),
34
+ parsing several structures in the same program — or in several threads — never
35
+ lets one file's numbering leak into another's.
36
+
37
+ .. note::
38
+
39
+ Hexadecimal detection is applied **only** to atom-serial fields. Ordinary
40
+ integer fields (residue sequence numbers, counts, and the like) are always
41
+ read as decimal, including legitimately negative values.