pidibble 1.7.2__tar.gz → 1.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. {pidibble-1.7.2 → pidibble-1.9.0}/.github/workflows/release.yaml +1 -1
  2. {pidibble-1.7.2 → pidibble-1.9.0}/.github/workflows/tests.yaml +4 -4
  3. {pidibble-1.7.2 → pidibble-1.9.0}/CHANGELOG.md +90 -0
  4. {pidibble-1.7.2 → pidibble-1.9.0}/PKG-INFO +1 -1
  5. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/data_model.rst +35 -0
  6. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/mmcif.rst +20 -0
  7. pidibble-1.9.0/docs/testing-notes.md +147 -0
  8. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/baseparsers.py +128 -7
  9. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/hex.py +40 -0
  10. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/mmcif_parse.py +61 -1
  11. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/pdbparse.py +82 -4
  12. pidibble-1.9.0/pidibble/pdbwrite.py +531 -0
  13. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/resources/pdb_format.yaml +74 -21
  14. {pidibble-1.7.2 → pidibble-1.9.0}/pyproject.toml +1 -1
  15. pidibble-1.9.0/tests/unit/test_baseparsers.py +131 -0
  16. pidibble-1.9.0/tests/unit/test_hex.py +36 -0
  17. pidibble-1.9.0/tests/unit/test_pdbwrite/.gitignore +6 -0
  18. pidibble-1.9.0/tests/unit/test_pdbwrite/4zmj.pdb +10656 -0
  19. pidibble-1.9.0/tests/unit/test_pdbwrite/charmm_glycan.pdb +10 -0
  20. pidibble-1.9.0/tests/unit/test_pdbwrite.py +418 -0
  21. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb.py +29 -0
  22. pidibble-1.7.2/tests/unit/test_hex.py +0 -22
  23. {pidibble-1.7.2 → pidibble-1.9.0}/.envrc +0 -0
  24. {pidibble-1.7.2 → pidibble-1.9.0}/.gitignore +0 -0
  25. {pidibble-1.7.2 → pidibble-1.9.0}/.readthedocs.yaml +0 -0
  26. {pidibble-1.7.2 → pidibble-1.9.0}/LICENSE +0 -0
  27. {pidibble-1.7.2 → pidibble-1.9.0}/README.md +0 -0
  28. {pidibble-1.7.2 → pidibble-1.9.0}/docs/Makefile +0 -0
  29. {pidibble-1.7.2 → pidibble-1.9.0}/docs/make.bat +0 -0
  30. {pidibble-1.7.2 → pidibble-1.9.0}/docs/mmcif_coverage.md +0 -0
  31. {pidibble-1.7.2 → pidibble-1.9.0}/docs/requirements.txt +0 -0
  32. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/_static/css/custom.css +0 -0
  33. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/API.rst +0 -0
  34. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.baseparsers.rst +0 -0
  35. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.baserecord.rst +0 -0
  36. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.hex.rst +0 -0
  37. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.mmcif_parse.rst +0 -0
  38. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.pdbparse.rst +0 -0
  39. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.pdbrecord.rst +0 -0
  40. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.resources.rst +0 -0
  41. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/api/pidibble.rst +0 -0
  42. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/changelog.rst +0 -0
  43. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/conf.py +0 -0
  44. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/advanced.rst +0 -0
  45. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/assemblies.rst +0 -0
  46. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/index.rst +0 -0
  47. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/large_structures.rst +0 -0
  48. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/loading.rst +0 -0
  49. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/nonconformance.rst +0 -0
  50. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/record_reference.rst +0 -0
  51. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/guide/records.rst +0 -0
  52. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/index.rst +0 -0
  53. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/installation.rst +0 -0
  54. {pidibble-1.7.2 → pidibble-1.9.0}/docs/source/quickstart.rst +0 -0
  55. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/__init__.py +0 -0
  56. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/baserecord.py +0 -0
  57. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/pdbrecord.py +0 -0
  58. {pidibble-1.7.2 → pidibble-1.9.0}/pidibble/resources/mmcif_format.yaml +0 -0
  59. {pidibble-1.7.2 → pidibble-1.9.0}/scripts/release.sh +0 -0
  60. {pidibble-1.7.2 → pidibble-1.9.0}/tests/__init__.py +0 -0
  61. {pidibble-1.7.2 → pidibble-1.9.0}/tests/conftest.py +0 -0
  62. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_hex/my_system.pdb +0 -0
  63. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_nonconformance.py +0 -0
  64. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/1ca2.cif +0 -0
  65. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/1ca2.pdb +0 -0
  66. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/4tvp.cif +0 -0
  67. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/4tvp.pdb +0 -0
  68. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/4zmj-newresnames.pdb +0 -0
  69. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/4zmj.cif +0 -0
  70. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/4zmj.pdb +0 -0
  71. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/6m0j.pdb +0 -0
  72. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/8fae.cif +0 -0
  73. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/G.pdb +0 -0
  74. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/GG.pdb +0 -0
  75. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/test.pdb +0 -0
  76. {pidibble-1.7.2 → pidibble-1.9.0}/tests/unit/test_rcsb/test_pdb_format.yaml +0 -0
@@ -21,7 +21,7 @@ jobs:
21
21
  id-token: write
22
22
  steps:
23
23
  - name: Download dist artifacts
24
- uses: actions/download-artifact@v4
24
+ uses: actions/download-artifact@v7
25
25
  with:
26
26
  name: dist
27
27
  path: dist/
@@ -15,9 +15,9 @@ jobs:
15
15
  matrix:
16
16
  python-version: ["3.10", "3.11", "3.12"]
17
17
  steps:
18
- - uses: actions/checkout@v4
18
+ - uses: actions/checkout@v5
19
19
  - name: Set up Python ${{ matrix.python-version }}
20
- uses: actions/setup-python@v5
20
+ uses: actions/setup-python@v6
21
21
  with:
22
22
  python-version: ${{ matrix.python-version }}
23
23
  - name: Install package with test extras
@@ -31,9 +31,9 @@ jobs:
31
31
  name: doctest (docs examples)
32
32
  runs-on: ubuntu-latest
33
33
  steps:
34
- - uses: actions/checkout@v4
34
+ - uses: actions/checkout@v5
35
35
  - name: Set up Python
36
- uses: actions/setup-python@v5
36
+ uses: actions/setup-python@v6
37
37
  with:
38
38
  python-version: "3.11"
39
39
  - name: Install docs requirements and local package
@@ -5,6 +5,96 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
5
5
 
6
6
  ## [Unreleased]
7
7
 
8
+ ## [1.9.0] - 2026-07-31
9
+
10
+ ### Changed
11
+ - Fields declared `Float`, `Integer`, or `HxInteger` that the source record
12
+ leaves blank (or whose text cannot be coerced) now parse to
13
+ `baseparsers.EmptyField` rather than a bare `''`. It is a `str` subclass equal
14
+ to `''`, falsy, written back out as blanks, and accepted anywhere `''` was
15
+ before — including `record.empty()`, JSON, and YAML — so existing
16
+ `if field == '':` / `!= ''` checks are unaffected. What it refuses is
17
+ impersonating a number: `float()`, `int()`, and the arithmetic operators raise
18
+ `TypeError` naming the field and why it is empty. Previously a blank `Float`
19
+ field silently evaluated `rec.length * 2` to `''` by string repetition, and
20
+ `float(rec.length)` raised a `ValueError` identifying neither record nor
21
+ field. Fields declared `String` are unchanged: there `''` is a legitimate
22
+ value, not a missing one.
23
+ - The mmCIF path applies the same rule. mmCIF carries no per-attribute type, so
24
+ a missing value (`.`, `?`, or an omitted attribute) previously came back as a
25
+ bare `''` whatever the field was; the declared type is now taken from the PDB
26
+ record spec, at both record and sub-record level (`residue.seqNum` and
27
+ friends, typed from `custom_formats`). A record read from `.cif` and the same
28
+ record read from `.pdb` now behave identically. `MMCIF_Parser` takes a new
29
+ optional `custom_formats` argument to do it.
30
+ - The writer tests' 4ZMJ structure is now a committed fixture rather than a
31
+ run-time RCSB download, so the unit suite runs offline on a clean checkout and
32
+ the byte-for-byte round-trip assertions are pinned to a known release.
33
+
34
+ ### Fixed
35
+ - `baseparsers.safe_float` now recognizes a padded or capitalized NaN sentinel
36
+ (`' nan'`, `'NaN'`) and not only the bare lowercase string. Float fields are
37
+ right-justified in fixed-column PDB, so the guard previously almost never
38
+ fired and `float('nan')` let a NaN propagate silently into coordinates,
39
+ occupancies, and B-factors.
40
+
41
+ ### Added
42
+ - Regression tests for two previously untested paths: the `safe_float` NaN
43
+ sentinel (`tests/unit/test_baseparsers.py`) and the strict `> 99999` decimal-
44
+ to-hex trip threshold in `hex.AtomSerialParser` (`tests/unit/test_hex.py`).
45
+ - An "Empty fields" section in the data-model guide documenting the convention
46
+ above.
47
+
48
+ ## [1.8.0] - 2026-07-22
49
+
50
+ ### Added
51
+ - PDB *writing*: parsed structures can be serialized back to conformant
52
+ fixed-column PDB. `PDBParser.write_PDB()` assembles a document in canonical
53
+ section order, reconstructs the coordinate section (`ATOM` with interleaved
54
+ `ANISOU` and chain-terminating `TER` cards, then `HETATM`), and regenerates
55
+ the `MASTER`/`END` bookkeeping records from the emitted content. The
56
+ record-level engine (`pidibble.pdbwrite.PDBWriter`) is the inverse of the
57
+ parser, driven by the same field specs plus optional per-field writer hints
58
+ (`{prec, just}`) carried as a third element in the YAML field definitions.
59
+ - Coverage spans all four writable record families: single-line records
60
+ (types 1/3, plus `TER`), continuation records (type 2 — `TITLE`, `COMPND`,
61
+ `SOURCE`, `KEYWDS`, `AUTHOR`, …), and determinant-group records (type 4 —
62
+ `SEQRES`, `HETNAM`, `HETSYN`, `FORMUL`, `SITE`, and the multi-line `REVDAT`),
63
+ re-wrapped/chunked across numbered continuation lines. `REMARK` and `JRNL`
64
+ (type 6) are re-emitted verbatim from the source lines.
65
+ - A full `parse -> write -> re-parse` round-trip preserves every parsed record
66
+ type and all field values on 4ZMJ (60 keys) and 4TVP (64 keys, incl. `SITE`);
67
+ the regenerated `MASTER` matches the original entry's byte-for-byte, and the
68
+ coordinate/`SEQRES`/`HETNAM`/`FORMUL`/`KEYWDS`/`TITLE` records re-serialize
69
+ byte-exactly.
70
+ - Hexadecimal serial numbers for structures with more than 99999 atoms (e.g.
71
+ large solvated systems): `HexSerialEncoder` is the exact inverse of the
72
+ parser's `AtomSerialParser`, switching to hex once a serial passes 99999 and
73
+ staying hex thereafter — including small `CONECT` back-references, which the
74
+ parser reads as hex once tripped. Serials round-trip up to the 5-column
75
+ hybrid-hex ceiling (`0xFFFFF` = 1 048 575 atoms).
76
+ - CHARMM read/write **dialect** (`PDBParser(dialect='charmm')`,
77
+ `write_PDB(dialect='charmm')`), for coordinate PDBs that must stay
78
+ column-congruent with a CHARMM/psfgen PSF. It widens `resName` to 6 columns
79
+ for CHARMM/glycan names (e.g. `BGLCNA`, `ANE5AC`) and writes the authoritative
80
+ `segID` column (73-76) that psfgen `coordpdb` depends on, while **pinning
81
+ x/y/z at columns 31-54** regardless of resName width — designing out the
82
+ wide-resName coordinate-column drift that scrambles fixed-column readers.
83
+ Parser and writer share one column model (`charmm_formats`/`ResidueCharmm` in
84
+ the YAML) so they are exact inverses; the default `'standard'` dialect remains
85
+ strict wwPDB. A `parse -> write -> parse` round-trip on a glycan structure
86
+ with a populated segID column is the identity for every coordinate field.
87
+
88
+ ### Changed
89
+ - Field specs may now carry an optional third element with writer formatting
90
+ hints; parsing ignores it (the byte-range unpacking is now width-tolerant), so
91
+ the change is fully backward-compatible.
92
+
93
+ ### Not yet supported
94
+ - Re-serialization of `REMARK`/`JRNL` from the parsed model (they are passed
95
+ through from the source instead, so they are omitted when the input was
96
+ mmCIF), and multi-model coordinate sections.
97
+
8
98
  ## [1.7.2] - 2026-07-21
9
99
 
10
100
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pidibble
3
- Version: 1.7.2
3
+ Version: 1.9.0
4
4
  Summary: A complete Protein Data Bank (PDB) file parser
5
5
  Project-URL: Source, https://github.com/cameronabrams/pidibble
6
6
  Project-URL: Documentation, https://pidibble.readthedocs.io/en/latest/
@@ -69,6 +69,41 @@ The keys shown by ``pstr()`` are exactly the attribute names:
69
69
  ``continuation`` are hidden by default) and a ``pad`` width for the labels, so
70
70
  you can widen or narrow the display or reveal the hidden internals.
71
71
 
72
+ .. _empty-fields:
73
+
74
+ Empty fields
75
+ ------------
76
+
77
+ A PDB record may simply stop before its optional trailing fields, and many
78
+ records leave interior columns blank. Any such field reads as empty:
79
+
80
+ >>> c = p.parsed['CONECT'][0]
81
+ >>> c.partner2 == '', bool(c.partner2)
82
+ (True, False)
83
+
84
+ For a field declared ``String`` that is a plain ``''``. For a field declared
85
+ ``Float``, ``Integer`` or ``HxInteger``, pidibble hands back an
86
+ :class:`~pidibble.baseparsers.EmptyField` instead — still equal to ``''``, still
87
+ falsy, still written back out as blanks, but it refuses to stand in for the
88
+ number its type promises:
89
+
90
+ >>> c.partner2 * 2 # doctest: +IGNORE_EXCEPTION_DETAIL
91
+ Traceback (most recent call last):
92
+ TypeError: HxInteger field 'partner2' is empty: absent from the source record.
93
+ It has no value to multiply; test `if rec.partner2 != '':` first.
94
+
95
+ Without that guard a plain ``''`` would quietly evaluate ``c.partner2 * 2`` to
96
+ ``''`` by string repetition, and ``float(c.partner2)`` would raise a
97
+ ``ValueError`` naming neither the record nor the field. Test for emptiness
98
+ before doing arithmetic:
99
+
100
+ >>> partners = [c.partner1, c.partner2, c.partner3, c.partner4]
101
+ >>> [q for q in partners if q != '']
102
+ [332]
103
+
104
+ The same applies to records parsed from mmCIF, where the missing-value markers
105
+ ``.`` and ``?`` land in the same place — see :doc:`mmcif`.
106
+
72
107
  Residues and other structured sub-fields
73
108
  -----------------------------------------
74
109
 
@@ -75,6 +75,26 @@ A few purely representational fields differ where the formats themselves differ.
75
75
  (``2014-06-27``) rather than the PDB ``DD-MON-YY`` form (``27-JUN-14``), and
76
76
  ``TITLE``/``KEYWDS`` text is upper-cased to match the PDB convention.
77
77
 
78
+ Missing values
79
+ --------------
80
+
81
+ mmCIF writes an inapplicable value as ``.`` and an unknown one as ``?``; an
82
+ attribute the file omits is missing too. All three parse to an empty field, and
83
+ because mmCIF carries no per-attribute type, the field's declared type is taken
84
+ from the PDB record spec — so a missing value in a field declared ``Float``,
85
+ ``Integer`` or ``HxInteger`` yields the same guarded
86
+ :class:`~pidibble.baseparsers.EmptyField` as it would from a ``.pdb`` file (see
87
+ :ref:`empty-fields`). It equals ``''`` but refuses to act as a number:
88
+
89
+ .. code-block:: python
90
+
91
+ het = cif.parsed['HETATM'][0]
92
+ het.residue.seqNum == '' # True -- label_seq_id is '.' for a non-polymer
93
+ het.residue.seqNum + 1 # TypeError: Integer field 'seqNum' is empty:
94
+ # absent from the mmCIF ('.', '?', or attribute
95
+ # not present) ...
96
+ het.residue_auth.seqNum # 1 -- the auth_* copy carries the number
97
+
78
98
  Comparing the two parses
79
99
  ------------------------
80
100
 
@@ -0,0 +1,147 @@
1
+ # Testing notes
2
+
3
+ **Status:** note as of 2026-07-31 (pidibble 1.8.0). Developer reference, not
4
+ end-user documentation. Everything below has since been acted on; see
5
+ "Resolution" under each item.
6
+
7
+ ## A failing large-list assertion makes the suite appear to hang
8
+
9
+ ### Symptom
10
+
11
+ If a regression breaks serial-number round-tripping, running the unit suite
12
+ appears to hang indefinitely rather than report a failure:
13
+
14
+ ```
15
+ tests/unit/test_pdbwrite.py::test_hex_serial_encoder_mirrors_parser PASSED [ 61%]
16
+ tests/unit/test_pdbwrite.py::test_big_serial_document_roundtrip
17
+ <no further output; one core pinned at 100%, runs past 15 minutes>
18
+ ```
19
+
20
+ The process is CPU-bound, not blocked on I/O or the network, which makes it look
21
+ like an infinite loop in the parser.
22
+
23
+ ### Cause
24
+
25
+ It is not a pidibble bug. The cost is entirely inside pytest's assertion-diff
26
+ rendering.
27
+
28
+ `test_big_serial_document_roundtrip` ends with whole-list comparisons:
29
+
30
+ ```python
31
+ assert [r.serial for r in parser.parsed['ATOM']] == [r.serial for r in q.parsed['ATOM']]
32
+ ```
33
+
34
+ For 4ZMJ that is **4518 elements per side**. While the lists are equal the
35
+ assertion is cheap. Once a regression makes them differ, pytest builds a
36
+ human-readable diff through `difflib`, and `difflib._fancy_replace` recurses via
37
+ `_fancy_helper` computing pairwise similarity ratios across the mismatched
38
+ block. On lists this large, made of similar-looking values (`100001`, `100002`,
39
+ …), that descent is effectively unbounded.
40
+
41
+ Observed stack at the point of the "hang":
42
+
43
+ ```
44
+ difflib.py:642 in quick_ratio
45
+ difflib.py:938 in _fancy_replace
46
+ difflib.py:997 in _fancy_helper
47
+ difflib.py:985 in _fancy_replace <- recursing
48
+ ...
49
+ _pytest/assertion/_compare_sequence.py:33 in _compare_eq_iterable
50
+ tests/unit/test_pdbwrite.py:262 in test_big_serial_document_roundtrip
51
+ ```
52
+
53
+ ### Confirmation
54
+
55
+ Disabling assertion introspection removes the cost completely, while the test
56
+ still fails as it should:
57
+
58
+ ```console
59
+ $ pytest tests/unit/test_pdbwrite.py::test_big_serial_document_roundtrip -q
60
+ <no completion; killed after 120s>
61
+
62
+ $ pytest tests/unit/test_pdbwrite.py::test_big_serial_document_roundtrip -q --assert=plain
63
+ 1 failed in 0.72s
64
+ ```
65
+
66
+ That is the diagnostic to reach for first: if `--assert=plain` returns
67
+ instantly, the time is going into diff rendering, not into pidibble.
68
+
69
+ Note that `-x` also hides the problem, because earlier tests in the file fail
70
+ first and stop the run before this test is reached. A regression can therefore
71
+ look like an ordinary failure under `-x` and like a hang without it.
72
+
73
+ ### Resolution (applied)
74
+
75
+ `test_pdbwrite.py` now routes all three whole-list serial comparisons
76
+ (`ATOM`, `TER`, `CONECT`) through a `_assert_serials_equal(expected, got, label)`
77
+ helper that checks the length, then collects mismatches itself and reports only
78
+ a bounded slice — so `difflib` is never handed a large similar-element diff:
79
+
80
+ ```python
81
+ assert len(got) == len(expected), f'{label}: got {len(got)} records, expected {len(expected)}'
82
+ mismatch = [(i, e, g) for i, (e, g) in enumerate(zip(expected, got)) if e != g]
83
+ assert not mismatch, f'{label}: {len(mismatch)} mismatches; first 10: {mismatch[:10]}'
84
+ ```
85
+
86
+ The test still fails for exactly the same regressions. Re-checked against the
87
+ column mutation below: it now fails in **0.55s** instead of hanging, reporting
88
+
89
+ ```
90
+ AssertionError: ATOM: 4518 mismatches; first 10: [(0, 100001, 34465), (1, 100002, 34466), ...]
91
+ ```
92
+
93
+ which points straight at the first bad serial.
94
+
95
+ ### How this was found
96
+
97
+ While using the suite as a measurement target, one-line mutations were applied
98
+ to source files to generate realistic red runs. The mutation that exposed this
99
+ was in the fixed-column field slice in `baseparsers.StringParser.parse`:
100
+
101
+ ```python
102
+ # original (columns are 1-based in the PDB spec)
103
+ fieldstring = record[byte_range[0] - 1:byte_range[1]]
104
+ # mutated: drops the 1-based correction, shifting every field by one column
105
+ fieldstring = record[byte_range[0]:byte_range[1]]
106
+ ```
107
+
108
+ Any regression that garbles parsed serials will reproduce it; that particular
109
+ mutation is just a convenient trigger.
110
+
111
+ Two unrelated observations from the same exercise, with what came of them:
112
+
113
+ - Two plausible mutations **survived** the suite — `safe_float`'s `'nan'` guard
114
+ in `baseparsers.py` could be deleted, and the `> 99999` hex-trip threshold in
115
+ `hex.AtomSerialParser.__call__` could be changed to `>= 99999`, with all 103
116
+ tests still passing. Both were boundary/sentinel paths with no test.
117
+
118
+ **Resolution (applied).** Both are now covered, and writing the tests turned
119
+ up a real defect behind the first one. `safe_float` compared the *raw* field
120
+ string to `'nan'`, but fixed-column float fields are right-justified, so a
121
+ NaN written by an upstream tool arrives as `' nan'` — the guard almost
122
+ never fired, and `float('nan')` then let a NaN propagate silently into
123
+ coordinates and B-factors. The comparison now strips and case-folds first
124
+ (`x.strip().lower() == 'nan'`). Covered by
125
+ `tests/unit/test_baseparsers.py` (direct, plus through `StringParser.parse`
126
+ on a record with `nan` in a coordinate column) and by
127
+ `test_hex_trip_threshold_is_strictly_above_99999` in `tests/unit/test_hex.py`,
128
+ which pins that parsing `99999` must *not* trip hex mode — under `>=`, a
129
+ following `CONECT` back-reference of `'10'` would silently read as 16. Each
130
+ mutation was re-applied to confirm the new tests fail on it.
131
+
132
+ - The network-dependence claim recorded here originally was **backwards**, and
133
+ is corrected as follows. `test_rcsb.py` (72 tests) and `test_hex.py` run
134
+ fully offline: every structure they name — `1ca2`, `4tvp`, `4zmj`,
135
+ `4zmj-newresnames`, `6m0j`, `8fae`, `test`, `my_system` — is a committed
136
+ fixture under `tests/unit/test_rcsb/` or `tests/unit/test_hex/`, and
137
+ `PDBParser.fetch()` only downloads when the file is absent. The one module
138
+ that *did* hit RCSB was `test_pdbwrite.py`, whose `4zmj.pdb` was gitignored,
139
+ so the byte-exactness round-trip ran against the live entry.
140
+
141
+ **Resolution (applied).** That fixture is now committed under
142
+ `tests/unit/test_pdbwrite/` and removed from the directory's `.gitignore`.
143
+ The whole unit suite therefore runs offline on a clean checkout, and the
144
+ byte-exact assertions are pinned to a known release rather than to whatever
145
+ RCSB currently serves — a re-release of 4ZMJ can no longer break them for
146
+ reasons unrelated to the writer. To pick up a revised entry deliberately,
147
+ delete the file and re-run the suite; `fetch()` will download it again.
@@ -7,6 +7,9 @@
7
7
 
8
8
  """
9
9
  import logging
10
+
11
+ import yaml
12
+
10
13
  logger = logging.getLogger(__name__)
11
14
 
12
15
  class ListParser:
@@ -153,6 +156,110 @@ class NonconformanceRegistry:
153
156
  log.info(f' {name}: {e["count"]} value(s) {kind}{eg}')
154
157
 
155
158
 
159
+ class EmptyField(str):
160
+ """
161
+ The value of a non-String field that the source record gave no usable value
162
+ for -- either the columns were blank (a truncated or minimal record) or the
163
+ text there could not be coerced to the declared type.
164
+
165
+ It *is* the empty string: ``field == ''`` is True, ``bool(field)`` is False,
166
+ ``f'{field}'`` is empty, and the writer renders it as blanks. Every existing
167
+ emptiness test keeps working unchanged.
168
+
169
+ What it refuses to do is impersonate a number. A plain ``''`` in a field
170
+ declared ``Float`` bites silently -- ``rec.length * 2`` evaluates to ``''``
171
+ by string repetition rather than raising -- so the numeric operators are
172
+ overridden here to fail loudly, naming the field and why it is empty::
173
+
174
+ rec.length * 2
175
+ TypeError: Float field 'length' is empty: absent from the source
176
+ record. It has no value to multiply; test `if rec.length != '':` first.
177
+
178
+ Attributes
179
+ ----------
180
+ field : str
181
+ Name of the field this stood in for.
182
+ typestring : str
183
+ The field's declared type, e.g. ``'Float'``.
184
+ reason : str
185
+ Why it is empty: ``'absent from the source record'`` or
186
+ ``"not coercible to Float (found 'O-')"``.
187
+ """
188
+
189
+ def __new__(cls, field='', typestring='', reason='absent from the source record'):
190
+ self = super().__new__(cls, '')
191
+ self.field = field
192
+ self.typestring = typestring
193
+ self.reason = reason
194
+ return self
195
+
196
+ def _complain(self, op):
197
+ name = f'{self.typestring} field {self.field!r}' if self.typestring else f'field {self.field!r}'
198
+ return TypeError(f'{name} is empty: {self.reason}. It has no value to {op}; '
199
+ f"test `if rec.{self.field or 'field'} != '':` first.")
200
+
201
+ # numeric coercions -- ValueError/AttributeError from these would be cryptic
202
+ def __float__(self):
203
+ raise self._complain('convert to float')
204
+
205
+ def __int__(self):
206
+ raise self._complain('convert to int')
207
+
208
+ def __index__(self):
209
+ raise self._complain('use as an index')
210
+
211
+ # str inherits these and they succeed silently on '', which is the trap:
212
+ # '' * 2 == '' and '' % x == ''. Concatenation with another string is
213
+ # still allowed -- it is meaningful, and the result is a plain str.
214
+ def __mul__(self, other):
215
+ raise self._complain('multiply')
216
+
217
+ __rmul__ = __mul__
218
+
219
+ def __mod__(self, other):
220
+ raise self._complain('use as a format string')
221
+
222
+ def __add__(self, other):
223
+ if isinstance(other, str):
224
+ return str(self) + other
225
+ raise self._complain('add')
226
+
227
+ # arithmetic str does not define at all: those already raise, but the
228
+ # default message names only the type, so give the same guided one
229
+ def __sub__(self, other):
230
+ raise self._complain('subtract from')
231
+
232
+ def __rsub__(self, other):
233
+ raise self._complain('subtract')
234
+
235
+ def __truediv__(self, other):
236
+ raise self._complain('divide')
237
+
238
+ def __rtruediv__(self, other):
239
+ raise self._complain('divide by')
240
+
241
+ def __round__(self, ndigits=None):
242
+ raise self._complain('round')
243
+
244
+ def __abs__(self):
245
+ raise self._complain('take the absolute value of')
246
+
247
+ def __neg__(self):
248
+ raise self._complain('negate')
249
+
250
+ def __repr__(self):
251
+ return f'<empty {self.typestring or "field"} {self.field!r}: {self.reason}>'
252
+
253
+
254
+ # PyYAML dispatches on exact type, so an unregistered str subclass raises
255
+ # RepresenterError; a caller dumping parsed records must keep seeing ''.
256
+ for _dumper in ('Dumper', 'SafeDumper', 'CDumper', 'CSafeDumper'):
257
+ if hasattr(yaml, _dumper): # the C dumpers need libyaml
258
+ yaml.add_representer(EmptyField, lambda dumper, data: dumper.represent_str(''),
259
+ Dumper=getattr(yaml, _dumper))
260
+ del _dumper
261
+
262
+
156
263
  class StringParser:
157
264
  """
158
265
  A parser for fixed-width strings, with a customizable field map.
@@ -200,21 +307,29 @@ class StringParser:
200
307
  input_dict = {}
201
308
  record += ' ' * (80 - len(record)) # pad
202
309
  for k, v in self.fields.items():
203
- typestring, byte_range = v
310
+ # a field spec is [typestring, byte_range] with an optional third
311
+ # element carrying writer hints (prec/just); parsing ignores it
312
+ typestring, byte_range = v[0], v[1]
204
313
  typ = self.typemap[typestring]
205
314
  assert byte_range[1] <= len(record), f'{record} {byte_range}'
206
315
  # using columns beginning with "1" not "0"
207
316
  fieldstring = record[byte_range[0] - 1:byte_range[1]]
208
317
  fieldstring = fieldstring.rstrip()
318
+ # a field with no usable value becomes EmptyField -- equal to '' for
319
+ # every emptiness test, but loud if used as the number its declared
320
+ # type promises. String fields keep a plain '': there the empty
321
+ # string is a legitimate value, not a missing one.
209
322
  try:
210
- # if len(fieldstring)>0 and not typ==str:
211
- # fieldstring=''
212
- input_dict[k] = '' if fieldstring == '' else typ(fieldstring)
323
+ if fieldstring == '':
324
+ input_dict[k] = '' if typ == str else EmptyField(k, typestring)
325
+ else:
326
+ input_dict[k] = typ(fieldstring)
213
327
  except (ValueError, TypeError):
214
328
  self.nonconformances.append({'field': k, 'kind': f'not coercible to {typestring}',
215
329
  'byte_range': byte_range, 'value': fieldstring})
216
330
  self.report_field_error(record, k)
217
- input_dict[k] = ''
331
+ input_dict[k] = EmptyField(k, typestring,
332
+ f'not coercible to {typestring} (found {fieldstring!r})')
218
333
  if typ == str:
219
334
  input_dict[k] = input_dict[k].strip()
220
335
  if fieldstring in self.allowed:
@@ -255,9 +370,15 @@ class StringParser:
255
370
 
256
371
  def safe_float(x):
257
372
  """
258
- Convert a string to a float, returning 0.0 if the string is 'nan'.
373
+ Convert a string to a float, returning 0.0 if the string denotes NaN.
374
+
375
+ Fixed-column float fields are right-justified, so a NaN written by an
376
+ upstream tool arrives padded (``' nan'``); the comparison is made on the
377
+ stripped, case-folded value so those are caught too. ``float()`` accepts
378
+ 'nan' happily, so without this the sentinel would propagate silently into
379
+ every downstream calculation.
259
380
  """
260
- if x == 'nan':
381
+ if isinstance(x, str) and x.strip().lower() == 'nan':
261
382
  return 0.0
262
383
  return float(x)
263
384
 
@@ -44,6 +44,46 @@ class AtomSerialParser:
44
44
  self._hex_tripped = False
45
45
 
46
46
 
47
+ class HexSerialEncoder:
48
+ """
49
+ Inverse of :class:`AtomSerialParser`: render an atom serial number for a
50
+ fixed-width field, switching from decimal to hexadecimal once any serial
51
+ exceeds 99999 and staying hexadecimal thereafter.
52
+
53
+ The switch is stateful and permanent, mirroring the parser's ``_hex_tripped``
54
+ flag exactly. This matters because the parser, once tripped, reads *every*
55
+ subsequent serial field as hex — including small back-references in
56
+ ``CONECT`` — so those must be encoded as hex too (e.g. serial ``10`` becomes
57
+ ``"A"``) for the file to round-trip. One encoder instance must therefore be
58
+ threaded through a whole document's serial fields in emission order.
59
+ """
60
+ def __init__(self):
61
+ self._hex_tripped = False
62
+
63
+ def __call__(self, serial: int) -> str:
64
+ """
65
+ Encode one serial as a digit string (the caller pads it to width).
66
+
67
+ Parameters
68
+ ----------
69
+ serial : int
70
+ The atom serial number.
71
+
72
+ Returns
73
+ -------
74
+ str
75
+ Decimal digits before the decimal-to-hex trip, uppercase hex after.
76
+ """
77
+ iv = int(serial)
78
+ if iv > 99999:
79
+ self._hex_tripped = True
80
+ return format(iv, 'X') if self._hex_tripped else str(iv)
81
+
82
+ def reset(self):
83
+ """Reset to decimal mode for a fresh document."""
84
+ self._hex_tripped = False
85
+
86
+
47
87
  # Module-level instance kept for backward compatibility
48
88
  _default_parser = AtomSerialParser()
49
89
 
@@ -10,9 +10,19 @@
10
10
 
11
11
  from .pdbrecord import PDBRecord, PDBRecordDict, PDBRecordList
12
12
  from .baserecord import BaseRecord
13
+ from .baseparsers import EmptyField
13
14
  import logging
14
15
  logger = logging.getLogger(__name__)
15
16
 
17
+ # mmCIF spells a missing value '.' (inapplicable) or '?' (unknown); an attribute
18
+ # the file simply omits is missing too. rectify() flattens all three to ''.
19
+ MMCIF_EMPTY_REASON = "absent from the mmCIF ('.', '?', or attribute not present)"
20
+
21
+ def _is_bare_empty(value):
22
+ """True for a plain empty string -- not for one already guarded, and not for
23
+ the empty list that ``as_list`` fields legitimately carry."""
24
+ return type(value) is str and value == ''
25
+
16
26
  def split_ri(ri):
17
27
  """
18
28
  Split a residue identifier into its sequence number and insertion code.
@@ -76,10 +86,15 @@ class MMCIF_Parser:
76
86
  A dictionary defining the PDB formats to be parsed.
77
87
  cif_data : object
78
88
  An object containing the CIF data to be parsed.
89
+ custom_formats : dict, optional
90
+ The ``custom_formats`` block of the PDB format file, used to type the
91
+ fields of nested sub-records such as residues. Without it, empty
92
+ sub-record fields stay plain ``''``.
79
93
  """
80
- def __init__(self, mmcif_formats, pdb_formats, cif_data):
94
+ def __init__(self, mmcif_formats, pdb_formats, cif_data, custom_formats=None):
81
95
  self.formats = mmcif_formats
82
96
  self.pdb_formats = pdb_formats
97
+ self.custom_formats = custom_formats or {}
83
98
  self.global_maps = {}
84
99
  self.global_ids = {}
85
100
  self.cif_data = cif_data
@@ -389,6 +404,50 @@ class MMCIF_Parser:
389
404
  idict[k] = v.upper()
390
405
  return idicts
391
406
 
407
+ def _guard_empty_fields(self, rectype, idict):
408
+ """
409
+ Replace bare ``''`` values with :class:`~pidibble.baseparsers.EmptyField`
410
+ wherever the corresponding PDB record field is declared numeric.
411
+
412
+ mmCIF carries no per-attribute type, so :func:`rectify` infers one from
413
+ the text and returns ``''`` for a missing value whatever the field is.
414
+ The PDB format spec *does* declare the type, and the same record read
415
+ from a ``.pdb`` file gets a guarded empty there (see
416
+ :meth:`~pidibble.baseparsers.StringParser.parse`) -- so apply the same
417
+ rule here, and a caller can treat both sources identically.
418
+
419
+ Parameters
420
+ ----------
421
+ rectype : str
422
+ The PDB record type these fields belong to, e.g. ``'SSBOND'``.
423
+ idict : dict
424
+ The assembled ``{field: value}`` dict, modified in place.
425
+
426
+ Returns
427
+ -------
428
+ dict
429
+ The same dict, for convenience.
430
+ """
431
+ fields = (self.pdb_formats.get(rectype) or {}).get('fields', {})
432
+ for k, v in idict.items():
433
+ spec = fields.get(k)
434
+ if spec is None: # bookkeeping key, or field the PDB spec has no slot for
435
+ continue
436
+ if isinstance(v, (PDBRecord, BaseRecord)):
437
+ self._guard_subrecord(spec[0], v)
438
+ elif _is_bare_empty(v) and spec[0] != 'String':
439
+ idict[k] = EmptyField(k, spec[0], MMCIF_EMPTY_REASON)
440
+ return idict
441
+
442
+ def _guard_subrecord(self, typestring, record):
443
+ """Apply :meth:`_guard_empty_fields`' rule to a nested sub-record whose
444
+ layout is named by a custom format (``Residue11``, ``Biomtrow``, …)."""
445
+ subfields = self.custom_formats.get(typestring, {})
446
+ for k, v in vars(record).items():
447
+ spec = subfields.get(k)
448
+ if spec is not None and _is_bare_empty(v) and spec[0] != 'String':
449
+ setattr(record, k, EmptyField(k, spec[0], MMCIF_EMPTY_REASON))
450
+
392
451
  def parse(self):
393
452
  """
394
453
  Parse the mmCIF data and generate a dictionary of :class:`pdbrecord.PDBRecord` instances.
@@ -403,6 +462,7 @@ class MMCIF_Parser:
403
462
  for rectype, mapspec in self.formats.items():
404
463
  idicts = self.gen_dict(mapspec)
405
464
  for idict in idicts:
465
+ self._guard_empty_fields(rectype, idict)
406
466
  this_key = idict.get('tmp_label', '')
407
467
  reckey = rectype if not this_key else f'{rectype}.{this_key}'
408
468
  if reckey in recdict: