pidibble 1.7.2__tar.gz → 1.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. {pidibble-1.7.2 → pidibble-1.8.0}/.github/workflows/release.yaml +1 -1
  2. {pidibble-1.7.2 → pidibble-1.8.0}/.github/workflows/tests.yaml +4 -4
  3. {pidibble-1.7.2 → pidibble-1.8.0}/CHANGELOG.md +50 -0
  4. {pidibble-1.7.2 → pidibble-1.8.0}/PKG-INFO +1 -1
  5. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/baseparsers.py +3 -1
  6. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/hex.py +40 -0
  7. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/pdbparse.py +81 -4
  8. pidibble-1.8.0/pidibble/pdbwrite.py +531 -0
  9. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/resources/pdb_format.yaml +74 -21
  10. {pidibble-1.7.2 → pidibble-1.8.0}/pyproject.toml +1 -1
  11. pidibble-1.8.0/tests/unit/test_pdbwrite/.gitignore +4 -0
  12. pidibble-1.8.0/tests/unit/test_pdbwrite/4zmj.pdb +10656 -0
  13. pidibble-1.8.0/tests/unit/test_pdbwrite/charmm_glycan.pdb +10 -0
  14. pidibble-1.8.0/tests/unit/test_pdbwrite.py +368 -0
  15. {pidibble-1.7.2 → pidibble-1.8.0}/.envrc +0 -0
  16. {pidibble-1.7.2 → pidibble-1.8.0}/.gitignore +0 -0
  17. {pidibble-1.7.2 → pidibble-1.8.0}/.readthedocs.yaml +0 -0
  18. {pidibble-1.7.2 → pidibble-1.8.0}/LICENSE +0 -0
  19. {pidibble-1.7.2 → pidibble-1.8.0}/README.md +0 -0
  20. {pidibble-1.7.2 → pidibble-1.8.0}/docs/Makefile +0 -0
  21. {pidibble-1.7.2 → pidibble-1.8.0}/docs/make.bat +0 -0
  22. {pidibble-1.7.2 → pidibble-1.8.0}/docs/mmcif_coverage.md +0 -0
  23. {pidibble-1.7.2 → pidibble-1.8.0}/docs/requirements.txt +0 -0
  24. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/_static/css/custom.css +0 -0
  25. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/API.rst +0 -0
  26. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.baseparsers.rst +0 -0
  27. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.baserecord.rst +0 -0
  28. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.hex.rst +0 -0
  29. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.mmcif_parse.rst +0 -0
  30. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.pdbparse.rst +0 -0
  31. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.pdbrecord.rst +0 -0
  32. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.resources.rst +0 -0
  33. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/api/pidibble.rst +0 -0
  34. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/changelog.rst +0 -0
  35. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/conf.py +0 -0
  36. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/advanced.rst +0 -0
  37. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/assemblies.rst +0 -0
  38. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/data_model.rst +0 -0
  39. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/index.rst +0 -0
  40. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/large_structures.rst +0 -0
  41. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/loading.rst +0 -0
  42. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/mmcif.rst +0 -0
  43. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/nonconformance.rst +0 -0
  44. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/record_reference.rst +0 -0
  45. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/guide/records.rst +0 -0
  46. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/index.rst +0 -0
  47. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/installation.rst +0 -0
  48. {pidibble-1.7.2 → pidibble-1.8.0}/docs/source/quickstart.rst +0 -0
  49. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/__init__.py +0 -0
  50. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/baserecord.py +0 -0
  51. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/mmcif_parse.py +0 -0
  52. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/pdbrecord.py +0 -0
  53. {pidibble-1.7.2 → pidibble-1.8.0}/pidibble/resources/mmcif_format.yaml +0 -0
  54. {pidibble-1.7.2 → pidibble-1.8.0}/scripts/release.sh +0 -0
  55. {pidibble-1.7.2 → pidibble-1.8.0}/tests/__init__.py +0 -0
  56. {pidibble-1.7.2 → pidibble-1.8.0}/tests/conftest.py +0 -0
  57. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_hex/my_system.pdb +0 -0
  58. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_hex.py +0 -0
  59. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_nonconformance.py +0 -0
  60. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/1ca2.cif +0 -0
  61. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/1ca2.pdb +0 -0
  62. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/4tvp.cif +0 -0
  63. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/4tvp.pdb +0 -0
  64. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/4zmj-newresnames.pdb +0 -0
  65. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/4zmj.cif +0 -0
  66. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/4zmj.pdb +0 -0
  67. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/6m0j.pdb +0 -0
  68. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/8fae.cif +0 -0
  69. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/G.pdb +0 -0
  70. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/GG.pdb +0 -0
  71. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/test.pdb +0 -0
  72. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb/test_pdb_format.yaml +0 -0
  73. {pidibble-1.7.2 → pidibble-1.8.0}/tests/unit/test_rcsb.py +0 -0
@@ -21,7 +21,7 @@ jobs:
21
21
  id-token: write
22
22
  steps:
23
23
  - name: Download dist artifacts
24
- uses: actions/download-artifact@v4
24
+ uses: actions/download-artifact@v7
25
25
  with:
26
26
  name: dist
27
27
  path: dist/
@@ -15,9 +15,9 @@ jobs:
15
15
  matrix:
16
16
  python-version: ["3.10", "3.11", "3.12"]
17
17
  steps:
18
- - uses: actions/checkout@v4
18
+ - uses: actions/checkout@v5
19
19
  - name: Set up Python ${{ matrix.python-version }}
20
- uses: actions/setup-python@v5
20
+ uses: actions/setup-python@v6
21
21
  with:
22
22
  python-version: ${{ matrix.python-version }}
23
23
  - name: Install package with test extras
@@ -31,9 +31,9 @@ jobs:
31
31
  name: doctest (docs examples)
32
32
  runs-on: ubuntu-latest
33
33
  steps:
34
- - uses: actions/checkout@v4
34
+ - uses: actions/checkout@v5
35
35
  - name: Set up Python
36
- uses: actions/setup-python@v5
36
+ uses: actions/setup-python@v6
37
37
  with:
38
38
  python-version: "3.11"
39
39
  - name: Install docs requirements and local package
@@ -5,6 +5,56 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
5
5
 
6
6
  ## [Unreleased]
7
7
 
8
+ ## [1.8.0] - 2026-07-22
9
+
10
+ ### Added
11
+ - PDB *writing*: parsed structures can be serialized back to conformant
12
+ fixed-column PDB. `PDBParser.write_PDB()` assembles a document in canonical
13
+ section order, reconstructs the coordinate section (`ATOM` with interleaved
14
+ `ANISOU` and chain-terminating `TER` cards, then `HETATM`), and regenerates
15
+ the `MASTER`/`END` bookkeeping records from the emitted content. The
16
+ record-level engine (`pidibble.pdbwrite.PDBWriter`) is the inverse of the
17
+ parser, driven by the same field specs plus optional per-field writer hints
18
+ (`{prec, just}`) carried as a third element in the YAML field definitions.
19
+ - Coverage spans all four writable record families: single-line records
20
+ (types 1/3, plus `TER`), continuation records (type 2 — `TITLE`, `COMPND`,
21
+ `SOURCE`, `KEYWDS`, `AUTHOR`, …), and determinant-group records (type 4 —
22
+ `SEQRES`, `HETNAM`, `HETSYN`, `FORMUL`, `SITE`, and the multi-line `REVDAT`),
23
+ re-wrapped/chunked across numbered continuation lines. `REMARK` and `JRNL`
24
+ (type 6) are re-emitted verbatim from the source lines.
25
+ - A full `parse -> write -> re-parse` round-trip preserves every parsed record
26
+ type and all field values on 4ZMJ (60 keys) and 4TVP (64 keys, incl. `SITE`);
27
+ the regenerated `MASTER` matches the original entry's byte-for-byte, and the
28
+ coordinate/`SEQRES`/`HETNAM`/`FORMUL`/`KEYWDS`/`TITLE` records re-serialize
29
+ byte-exactly.
30
+ - Hexadecimal serial numbers for structures with more than 99999 atoms (e.g.
31
+ large solvated systems): `HexSerialEncoder` is the exact inverse of the
32
+ parser's `AtomSerialParser`, switching to hex once a serial passes 99999 and
33
+ staying hex thereafter — including small `CONECT` back-references, which the
34
+ parser reads as hex once tripped. Serials round-trip up to the 5-column
35
+ hybrid-hex ceiling (`0xFFFFF` = 1 048 575 atoms).
36
+ - CHARMM read/write **dialect** (`PDBParser(dialect='charmm')`,
37
+ `write_PDB(dialect='charmm')`), for coordinate PDBs that must stay
38
+ column-congruent with a CHARMM/psfgen PSF. It widens `resName` to 6 columns
39
+ for CHARMM/glycan names (e.g. `BGLCNA`, `ANE5AC`) and writes the authoritative
40
+ `segID` column (73-76) that psfgen `coordpdb` depends on, while **pinning
41
+ x/y/z at columns 31-54** regardless of resName width — designing out the
42
+ wide-resName coordinate-column drift that scrambles fixed-column readers.
43
+ Parser and writer share one column model (`charmm_formats`/`ResidueCharmm` in
44
+ the YAML) so they are exact inverses; the default `'standard'` dialect remains
45
+ strict wwPDB. A `parse -> write -> parse` round-trip on a glycan structure
46
+ with a populated segID column is the identity for every coordinate field.
47
+
48
+ ### Changed
49
+ - Field specs may now carry an optional third element with writer formatting
50
+ hints; parsing ignores it (the byte-range unpacking is now width-tolerant), so
51
+ the change is fully backward-compatible.
52
+
53
+ ### Not yet supported
54
+ - Re-serialization of `REMARK`/`JRNL` from the parsed model (they are passed
55
+ through from the source instead, so they are omitted when the input was
56
+ mmCIF), and multi-model coordinate sections.
57
+
8
58
  ## [1.7.2] - 2026-07-21
9
59
 
10
60
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pidibble
3
- Version: 1.7.2
3
+ Version: 1.8.0
4
4
  Summary: A complete Protein Data Bank (PDB) file parser
5
5
  Project-URL: Source, https://github.com/cameronabrams/pidibble
6
6
  Project-URL: Documentation, https://pidibble.readthedocs.io/en/latest/
@@ -200,7 +200,9 @@ class StringParser:
200
200
  input_dict = {}
201
201
  record += ' ' * (80 - len(record)) # pad
202
202
  for k, v in self.fields.items():
203
- typestring, byte_range = v
203
+ # a field spec is [typestring, byte_range] with an optional third
204
+ # element carrying writer hints (prec/just); parsing ignores it
205
+ typestring, byte_range = v[0], v[1]
204
206
  typ = self.typemap[typestring]
205
207
  assert byte_range[1] <= len(record), f'{record} {byte_range}'
206
208
  # using columns beginning with "1" not "0"
@@ -44,6 +44,46 @@ class AtomSerialParser:
44
44
  self._hex_tripped = False
45
45
 
46
46
 
47
+ class HexSerialEncoder:
48
+ """
49
+ Inverse of :class:`AtomSerialParser`: render an atom serial number for a
50
+ fixed-width field, switching from decimal to hexadecimal once any serial
51
+ exceeds 99999 and staying hexadecimal thereafter.
52
+
53
+ The switch is stateful and permanent, mirroring the parser's ``_hex_tripped``
54
+ flag exactly. This matters because the parser, once tripped, reads *every*
55
+ subsequent serial field as hex — including small back-references in
56
+ ``CONECT`` — so those must be encoded as hex too (e.g. serial ``10`` becomes
57
+ ``"A"``) for the file to round-trip. One encoder instance must therefore be
58
+ threaded through a whole document's serial fields in emission order.
59
+ """
60
+ def __init__(self):
61
+ self._hex_tripped = False
62
+
63
+ def __call__(self, serial: int) -> str:
64
+ """
65
+ Encode one serial as a digit string (the caller pads it to width).
66
+
67
+ Parameters
68
+ ----------
69
+ serial : int
70
+ The atom serial number.
71
+
72
+ Returns
73
+ -------
74
+ str
75
+ Decimal digits before the decimal-to-hex trip, uppercase hex after.
76
+ """
77
+ iv = int(serial)
78
+ if iv > 99999:
79
+ self._hex_tripped = True
80
+ return format(iv, 'X') if self._hex_tripped else str(iv)
81
+
82
+ def reset(self):
83
+ """Reset to decimal mode for a fresh document."""
84
+ self._hex_tripped = False
85
+
86
+
47
87
  # Module-level instance kept for backward compatibility
48
88
  _default_parser = AtomSerialParser()
49
89
 
@@ -60,6 +60,7 @@ class PDBParser:
60
60
  filepath: str | Path = None,
61
61
  mappers: dict[str, Callable] = None,
62
62
  comment_chars: list[str] = ['#'],
63
+ dialect: str = 'standard',
63
64
  pdb_format_file: str = 'pdb_format.yaml',
64
65
  mmcif_format_file: str = 'mmcif_format.yaml',
65
66
  **kwargs):
@@ -83,6 +84,7 @@ class PDBParser:
83
84
  self.mappers.update(mappers)
84
85
  self.mappers.update(ListParsers)
85
86
  self.comment_chars = comment_chars
87
+ self.dialect = dialect
86
88
  self.pdb_lines = []
87
89
  self.cif_data = {}
88
90
 
@@ -90,6 +92,9 @@ class PDBParser:
90
92
  self.nonconformances = NonconformanceRegistry()
91
93
  self.pdb_format_dict = self._load_format(pdb_format_file)
92
94
  self.mmcif_format_dict = self._load_format(mmcif_format_file)
95
+ # the active record-format table for this instance's dialect; both
96
+ # parsing and writing read this so they stay exact inverses
97
+ self.record_formats = self._apply_dialect(self.pdb_format_dict, dialect)
93
98
 
94
99
  # update mappers with delimiters and custom formats
95
100
  delimiter_dict = self.pdb_format_dict.get('delimiters', {})
@@ -101,6 +106,36 @@ class PDBParser:
101
106
  if not cname in self.mappers:
102
107
  self.mappers[cname] = BaseRecordParser(cformat, self.mappers).parse
103
108
 
109
+ @staticmethod
110
+ def _apply_dialect(pdb_format_dict, dialect):
111
+ """
112
+ Build the record-format table for a dialect.
113
+
114
+ Returns a copy of the base ``record_formats`` with the dialect's
115
+ coordinate-record overrides merged in. ``'standard'`` is strict wwPDB;
116
+ ``'charmm'`` swaps ``ATOM``/``HETATM``/``TER`` for the wide-resName,
117
+ segID-bearing CHARMM layouts (``charmm_formats`` in the YAML), keeping
118
+ x/y/z pinned at columns 31-54.
119
+
120
+ Parameters
121
+ ----------
122
+ pdb_format_dict : dict
123
+ The loaded PDB format file.
124
+ dialect : str
125
+ ``'standard'`` (default) or ``'charmm'``.
126
+
127
+ Returns
128
+ -------
129
+ dict
130
+ The active ``{record_key: format}`` mapping for the dialect.
131
+ """
132
+ record_formats = dict(pdb_format_dict['record_formats'])
133
+ if dialect == 'charmm':
134
+ record_formats.update(pdb_format_dict.get('charmm_formats', {}))
135
+ elif dialect != 'standard':
136
+ raise ValueError(f"unknown dialect {dialect!r}; expected 'standard' or 'charmm'")
137
+ return record_formats
138
+
104
139
  @staticmethod
105
140
  def _load_format(filename: str) -> dict:
106
141
  """Load a YAML format file, checking CWD first then the package resources."""
@@ -252,7 +287,7 @@ class PDBParser:
252
287
  This method uses the :class:`.mmcif_parse.MMCIF_Parser` to parse the mmCIF data and store the parsed records
253
288
  in :attr:`PDBParser.parsed`.
254
289
  """
255
- mmcif_parser = MMCIF_Parser(self.mmcif_format_dict, self.pdb_format_dict['record_formats'], self.cif_data)
290
+ mmcif_parser = MMCIF_Parser(self.mmcif_format_dict, self.record_formats, self.cif_data)
256
291
  self.parsed = mmcif_parser.parse()
257
292
 
258
293
  def parse_PDB(self):
@@ -264,7 +299,7 @@ class PDBParser:
264
299
  """
265
300
  self._atom_serial_parser.reset()
266
301
  self.nonconformances = NonconformanceRegistry()
267
- record_formats = self.pdb_format_dict['record_formats']
302
+ record_formats = self.record_formats
268
303
  key = ''
269
304
  record_format = {}
270
305
  group_open_record = None
@@ -368,12 +403,12 @@ class PDBParser:
368
403
  if isinstance(p, PDBRecord):
369
404
  rf = p.format
370
405
  if 'embedded_records' in rf:
371
- new_parsed_records.update(p.parse_embedded(self.pdb_format_dict['record_formats'], self.mappers))
406
+ new_parsed_records.update(p.parse_embedded(self.record_formats, self.mappers))
372
407
  elif isinstance(p, PDBRecordList):
373
408
  for q in p:
374
409
  rf = q.format
375
410
  if 'embedded_records' in rf:
376
- new_parsed_records.update(q.parse_embedded(self.pdb_format_dict['record_formats'], self.mappers))
411
+ new_parsed_records.update(q.parse_embedded(self.record_formats, self.mappers))
377
412
  self.parsed.update(new_parsed_records)
378
413
 
379
414
  def parse_tokens(self):
@@ -430,6 +465,48 @@ class PDBParser:
430
465
  logger.warning(f'No data.')
431
466
  return self
432
467
 
468
+ def write_PDB(self, filename: str = None, anisou: bool = True, include_master: bool = True,
469
+ dialect: str = None):
470
+ """
471
+ Write the parsed structure back out as a conformant PDB file.
472
+
473
+ This assembles every writable record in canonical section order —
474
+ single-line records (types 1/3), continuation and determinant-group
475
+ records (types 2/4, e.g. ``TITLE``, ``COMPND``, ``SEQRES``, ``SITE``),
476
+ and ``TER`` — reconstructs the coordinate section, passes ``REMARK`` and
477
+ ``JRNL`` through verbatim from the source, and regenerates the
478
+ ``MASTER``/``END`` bookkeeping records. When the parse came from mmCIF
479
+ (no PDB source lines), ``REMARK``/``JRNL`` are omitted and reported to
480
+ the logger.
481
+
482
+ Parameters
483
+ ----------
484
+ filename : str, optional
485
+ Destination path. If omitted, no file is written.
486
+ anisou : bool, optional
487
+ Interleave ``ANISOU`` records after their atoms (default True).
488
+ include_master : bool, optional
489
+ Regenerate a ``MASTER`` record from the emitted counts (default True).
490
+ dialect : str, optional
491
+ Column dialect to write, ``'standard'`` or ``'charmm'``. Defaults to
492
+ the dialect this parser was constructed with. The CHARMM dialect
493
+ widens ``resName`` to 6 columns and writes the authoritative segID
494
+ column (73-76) while pinning x/y/z at columns 31-54.
495
+
496
+ Returns
497
+ -------
498
+ list of str
499
+ The assembled document as a list of record lines.
500
+ """
501
+ from .pdbwrite import assemble_pdb
502
+ record_formats = self._apply_dialect(self.pdb_format_dict, dialect) if dialect else None
503
+ lines = assemble_pdb(self, anisou=anisou, include_master=include_master,
504
+ record_formats=record_formats)
505
+ if filename:
506
+ with open(filename, 'w') as f:
507
+ f.write('\n'.join(lines) + '\n')
508
+ return lines
509
+
433
510
  def get_symm_ops(rec: PDBRecord):
434
511
  """
435
512
  Extract the symmetry operations from a PDB record.