molforge 0.2.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (223) hide show
  1. {molforge-0.2.0 → molforge-0.4.0}/CHANGELOG.md +763 -159
  2. {molforge-0.2.0 → molforge-0.4.0}/PKG-INFO +12 -4
  3. {molforge-0.2.0 → molforge-0.4.0}/README.md +6 -2
  4. {molforge-0.2.0 → molforge-0.4.0}/pyproject.toml +7 -2
  5. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/__init__.py +1 -1
  6. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/__init__.py +4 -0
  7. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/metadata_keys.py +27 -0
  8. molforge-0.4.0/src/molforge/core/provenance.py +350 -0
  9. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/io/__init__.py +35 -5
  10. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/io/dispatch.py +16 -6
  11. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/io/mmcif.py +103 -15
  12. molforge-0.4.0/src/molforge/io/mol2.py +361 -0
  13. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/io/pdb_alphafold.py +10 -0
  14. molforge-0.4.0/src/molforge/io/pdbqt.py +265 -0
  15. molforge-0.4.0/src/molforge/io/pqr.py +232 -0
  16. molforge-0.4.0/src/molforge/io/sdf.py +309 -0
  17. molforge-0.4.0/src/molforge/io/trajectory.py +399 -0
  18. molforge-0.4.0/src/molforge/prep/__init__.py +62 -0
  19. molforge-0.4.0/src/molforge/prep/_deps.py +58 -0
  20. molforge-0.4.0/src/molforge/prep/_provenance.py +65 -0
  21. molforge-0.4.0/src/molforge/prep/clean.py +157 -0
  22. molforge-0.4.0/src/molforge/prep/fix.py +256 -0
  23. molforge-0.4.0/src/molforge/prep/pipeline.py +120 -0
  24. molforge-0.4.0/src/molforge/prep/protonate.py +123 -0
  25. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/docking/__init__.py +1 -1
  26. molforge-0.4.0/src/molforge/wrappers/docking/diffdock.py +407 -0
  27. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/docking/vina.py +60 -9
  28. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/folding/__init__.py +0 -3
  29. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/folding/_base.py +2 -2
  30. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/folding/alphafold.py +18 -0
  31. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/folding/boltz.py +13 -0
  32. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/folding/esmfold.py +18 -0
  33. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/folding/rosettafold.py +12 -0
  34. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/generative/proteinmpnn.py +40 -6
  35. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/generative/rfdiffusion.py +24 -0
  36. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/md/__init__.py +3 -2
  37. molforge-0.4.0/src/molforge/wrappers/md/gromacs.py +716 -0
  38. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/md/openmm.py +105 -9
  39. molforge-0.4.0/tests/fixtures/pdb/ala_tripeptide_heavy.pdb +26 -0
  40. molforge-0.4.0/tests/unit/core/test_provenance.py +339 -0
  41. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/io/test_dispatch.py +30 -5
  42. molforge-0.4.0/tests/unit/io/test_mmcif.py +767 -0
  43. molforge-0.4.0/tests/unit/io/test_mol2.py +353 -0
  44. molforge-0.4.0/tests/unit/io/test_pdbqt.py +244 -0
  45. molforge-0.4.0/tests/unit/io/test_pqr.py +205 -0
  46. molforge-0.4.0/tests/unit/io/test_sdf.py +256 -0
  47. molforge-0.4.0/tests/unit/io/test_trajectory.py +251 -0
  48. molforge-0.4.0/tests/unit/prep/test_clean.py +124 -0
  49. molforge-0.4.0/tests/unit/prep/test_fix.py +161 -0
  50. molforge-0.4.0/tests/unit/prep/test_pipeline.py +123 -0
  51. molforge-0.4.0/tests/unit/prep/test_protonate.py +121 -0
  52. molforge-0.4.0/tests/unit/wrappers/__init__.py +0 -0
  53. molforge-0.4.0/tests/unit/wrappers/test_diffdock.py +308 -0
  54. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_docking_base.py +0 -30
  55. molforge-0.4.0/tests/unit/wrappers/test_gromacs.py +588 -0
  56. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_md_base.py +0 -45
  57. molforge-0.4.0/tests/unit/wrappers/test_openmm.py +235 -0
  58. molforge-0.4.0/tests/unit/wrappers/test_provenance_adoption.py +692 -0
  59. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_rfdiffusion.py +146 -0
  60. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_rosettafold.py +0 -26
  61. molforge-0.2.0/src/molforge/io/mol2.py +0 -33
  62. molforge-0.2.0/src/molforge/io/pdbqt.py +0 -33
  63. molforge-0.2.0/src/molforge/io/pqr.py +0 -33
  64. molforge-0.2.0/src/molforge/io/sdf.py +0 -33
  65. molforge-0.2.0/src/molforge/wrappers/docking/diffdock.py +0 -54
  66. molforge-0.2.0/src/molforge/wrappers/folding/rosetta.py +0 -55
  67. molforge-0.2.0/src/molforge/wrappers/md/gromacs.py +0 -78
  68. molforge-0.2.0/tests/unit/io/test_mmcif.py +0 -171
  69. molforge-0.2.0/tests/unit/wrappers/test_openmm.py +0 -157
  70. {molforge-0.2.0 → molforge-0.4.0}/.gitignore +0 -0
  71. {molforge-0.2.0 → molforge-0.4.0}/LICENSE +0 -0
  72. {molforge-0.2.0 → molforge-0.4.0}/data/README.md +0 -0
  73. {molforge-0.2.0 → molforge-0.4.0}/notebooks/README.md +0 -0
  74. {molforge-0.2.0 → molforge-0.4.0}/plugins/README.md +0 -0
  75. {molforge-0.2.0 → molforge-0.4.0}/plugins/example_plugin/README.md +0 -0
  76. {molforge-0.2.0 → molforge-0.4.0}/plugins/example_plugin/pyproject.toml +0 -0
  77. {molforge-0.2.0 → molforge-0.4.0}/requirements/README.md +0 -0
  78. {molforge-0.2.0 → molforge-0.4.0}/scripts/README.md +0 -0
  79. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/atom.py +0 -0
  80. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/atom_array.py +0 -0
  81. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/chain.py +0 -0
  82. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/constants.py +0 -0
  83. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/protein.py +0 -0
  84. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/core/residue.py +0 -0
  85. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/docking/__init__.py +0 -0
  86. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ensembles/__init__.py +0 -0
  87. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ensembles/clustering.py +0 -0
  88. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ensembles/consensus.py +0 -0
  89. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ensembles/density.py +0 -0
  90. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ensembles/geometry.py +0 -0
  91. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ensembles/weighting.py +0 -0
  92. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/generative.py +0 -0
  93. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/io/fasta.py +0 -0
  94. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/io/pdb.py +0 -0
  95. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/md/__init__.py +0 -0
  96. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/metrics/__init__.py +0 -0
  97. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/metrics/dockq.py +0 -0
  98. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/metrics/gdt.py +0 -0
  99. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/metrics/lddt.py +0 -0
  100. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/metrics/tm.py +0 -0
  101. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ml/__init__.py +0 -0
  102. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ml/embeddings.py +0 -0
  103. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ml/graph.py +0 -0
  104. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ml/sequence_features.py +0 -0
  105. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/ml/structure_features.py +0 -0
  106. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/plugins/__init__.py +0 -0
  107. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/plugins/registry.py +0 -0
  108. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/py.typed +0 -0
  109. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/sequence/__init__.py +0 -0
  110. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/sequence/alignment.py +0 -0
  111. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/sequence/composition.py +0 -0
  112. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/sequence/matrices.py +0 -0
  113. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/sequence/mutations.py +0 -0
  114. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/__init__.py +0 -0
  115. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/contacts.py +0 -0
  116. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/dihedrals.py +0 -0
  117. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/dssp.py +0 -0
  118. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/geometry.py +0 -0
  119. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/rmsd.py +0 -0
  120. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/sasa.py +0 -0
  121. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/structure/superposition.py +0 -0
  122. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/validation/__init__.py +0 -0
  123. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/validation/criteria.py +0 -0
  124. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/validation/orchestration.py +0 -0
  125. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/validation/verdict.py +0 -0
  126. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/__init__.py +0 -0
  127. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/docking/_base.py +0 -0
  128. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/docking/prep.py +0 -0
  129. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/generative/__init__.py +0 -0
  130. {molforge-0.2.0 → molforge-0.4.0}/src/molforge/wrappers/md/_base.py +0 -0
  131. {molforge-0.2.0 → molforge-0.4.0}/tests/__init__.py +0 -0
  132. {molforge-0.2.0 → molforge-0.4.0}/tests/benchmarks/__init__.py +0 -0
  133. {molforge-0.2.0 → molforge-0.4.0}/tests/benchmarks/conftest.py +0 -0
  134. {molforge-0.2.0 → molforge-0.4.0}/tests/benchmarks/test_perf.py +0 -0
  135. {molforge-0.2.0 → molforge-0.4.0}/tests/conftest.py +0 -0
  136. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/cif/dipeptide.cif +0 -0
  137. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/fasta/.gitkeep +0 -0
  138. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/fasta/multiline_with_digits.fasta +0 -0
  139. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/fasta/simple.fasta +0 -0
  140. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/.gitkeep +0 -0
  141. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/alphafold_mock.pdb +0 -0
  142. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/dipeptide.pdb +0 -0
  143. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/helix.pdb +0 -0
  144. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/mini_beta_sheet.pdb +0 -0
  145. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/mini_complex_bad.pdb +0 -0
  146. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/mini_complex_good.pdb +0 -0
  147. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/mini_complex_native.pdb +0 -0
  148. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/mini_ensemble.pdb +0 -0
  149. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/mini_mixed.pdb +0 -0
  150. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/mini_with_ligand.pdb +0 -0
  151. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/multi_model.pdb +0 -0
  152. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/real_small_protein.pdb +0 -0
  153. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/real_with_altloc_sidechains.pdb +0 -0
  154. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/real_with_ligand_realistic.pdb +0 -0
  155. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/tripeptide.pdb +0 -0
  156. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/with_altloc.pdb +0 -0
  157. {molforge-0.2.0 → molforge-0.4.0}/tests/fixtures/pdb/with_insertion_code.pdb +0 -0
  158. {molforge-0.2.0 → molforge-0.4.0}/tests/integration/__init__.py +0 -0
  159. {molforge-0.2.0 → molforge-0.4.0}/tests/integration/test_fixtures.py +0 -0
  160. {molforge-0.2.0 → molforge-0.4.0}/tests/integration/test_real_fixtures.py +0 -0
  161. {molforge-0.2.0 → molforge-0.4.0}/tests/integration/test_smoke.py +0 -0
  162. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/__init__.py +0 -0
  163. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/core/__init__.py +0 -0
  164. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/core/test_atom_array.py +0 -0
  165. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/core/test_constants.py +0 -0
  166. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/core/test_core_smoke.py +0 -0
  167. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/core/test_hierarchy.py +0 -0
  168. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/core/test_metadata_keys.py +0 -0
  169. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/docking/__init__.py +0 -0
  170. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/docking/test_docking_smoke.py +0 -0
  171. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/__init__.py +0 -0
  172. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/conftest.py +0 -0
  173. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/test_clustering.py +0 -0
  174. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/test_consensus.py +0 -0
  175. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/test_density.py +0 -0
  176. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/test_geometry.py +0 -0
  177. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/test_smoke.py +0 -0
  178. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ensembles/test_weighting.py +0 -0
  179. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/io/__init__.py +0 -0
  180. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/io/test_alphafold.py +0 -0
  181. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/io/test_fasta.py +0 -0
  182. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/io/test_io_smoke.py +0 -0
  183. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/io/test_pdb.py +0 -0
  184. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/md/__init__.py +0 -0
  185. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/metrics/__init__.py +0 -0
  186. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/metrics/test_dockq.py +0 -0
  187. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/metrics/test_gdt.py +0 -0
  188. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/metrics/test_lddt.py +0 -0
  189. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/metrics/test_tm.py +0 -0
  190. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ml/__init__.py +0 -0
  191. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ml/test_embeddings.py +0 -0
  192. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ml/test_graph.py +0 -0
  193. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ml/test_sequence_features.py +0 -0
  194. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/ml/test_structure_features.py +0 -0
  195. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/plugins/__init__.py +0 -0
  196. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/plugins/test_plugins_smoke.py +0 -0
  197. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/plugins/test_registry.py +0 -0
  198. {molforge-0.2.0/tests/unit/sequence → molforge-0.4.0/tests/unit/prep}/__init__.py +0 -0
  199. {molforge-0.2.0/tests/unit/structure → molforge-0.4.0/tests/unit/sequence}/__init__.py +0 -0
  200. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/sequence/test_alignment.py +0 -0
  201. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/sequence/test_composition.py +0 -0
  202. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/sequence/test_matrices.py +0 -0
  203. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/sequence/test_mutations.py +0 -0
  204. {molforge-0.2.0/tests/unit/wrappers → molforge-0.4.0/tests/unit/structure}/__init__.py +0 -0
  205. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_contacts.py +0 -0
  206. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_dihedrals.py +0 -0
  207. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_dssp.py +0 -0
  208. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_geometry.py +0 -0
  209. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_rmsd.py +0 -0
  210. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_sasa.py +0 -0
  211. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_structure_smoke.py +0 -0
  212. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/structure/test_superposition.py +0 -0
  213. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/test_typing.py +0 -0
  214. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/validation/test_criteria.py +0 -0
  215. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/validation/test_orchestration.py +0 -0
  216. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/validation/test_verdict.py +0 -0
  217. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_alphafold.py +0 -0
  218. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_boltz.py +0 -0
  219. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_esmfold.py +0 -0
  220. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_folding_base.py +0 -0
  221. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_prep.py +0 -0
  222. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_proteinmpnn.py +0 -0
  223. {molforge-0.2.0 → molforge-0.4.0}/tests/unit/wrappers/test_vina.py +0 -0
@@ -7,6 +7,614 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.4.0] 2026-06-25
11
+
12
+ ### Added
13
+
14
+ - **Provenance adoption pass 2 — MD wrappers and prep functions.**
15
+ Completes the wrapper-adoption work. Where
16
+ pass 1 demonstrated cross-wrapper chaining (ESMFold → Vina),
17
+ pass 2 exercises chaining **within** a single wrapper's
18
+ multi-step pipeline and across the composable prep functions —
19
+ the harder case and the one that proves the design composes.
20
+
21
+ Adopted (2 wrappers + 5 functions):
22
+
23
+ - **`molforge.wrappers.md.OpenMM`** — `prepare` / `minimize` /
24
+ `run` each attach a `Provenance` to the output's metadata. Each
25
+ step's parent is the previous step's `Provenance`, so a full
26
+ pipeline leaves the final `Trajectory` with a 3-deep chain
27
+ that reads as `["OpenMM.prepare", "OpenMM.minimize", "OpenMM.run"]`
28
+ oldest-first. When the input `Protein` has its own `Provenance`
29
+ (e.g. it came from ESMFold), the chain extends back through it
30
+ — a sequence-to-trajectory workflow ends with a 4-deep chain
31
+ that traces all the way to the sequence.
32
+
33
+ - **`molforge.wrappers.md.GROMACS`** — same three-step pattern
34
+ as OpenMM, with the engine strings `"GROMACS.prepare"`,
35
+ `"GROMACS.minimize"`, `"GROMACS.run"`. The minimize step
36
+ appends to `simulation.metadata` (since minimize returns the
37
+ same Simulation in-place, not a new one), preserving the
38
+ chain across the mutation.
39
+
40
+ - **`molforge.prep.{remove_heterogens, fix_missing_atoms,
41
+ add_caps, add_hydrogens, prepare_for_md}`** — each prep
42
+ function chains a `Provenance` step onto the output's
43
+ metadata. `prepare_for_md` (which composes the other four in
44
+ sequence) leaves the result with a 4-deep chain naturally:
45
+ `["molforge.prep.remove_heterogens",
46
+ "molforge.prep.fix_missing_atoms",
47
+ "molforge.prep.add_caps",
48
+ "molforge.prep.add_hydrogens"]`. No special handling needed
49
+ in the composite — the inner functions chain themselves.
50
+
51
+ Two helpers, both private:
52
+
53
+ - **`molforge.wrappers.md.{openmm,gromacs}._parent_provenance(meta)`**
54
+ — extracts a `Provenance | None` from a free-form metadata
55
+ dict, narrowing the type. Per-wrapper rather than shared so
56
+ each MD wrapper stays self-contained.
57
+
58
+ - **`molforge.prep._provenance.chain_prep_provenance(output, *,
59
+ engine, parameters, input_protein)`** — the DRY-up for the
60
+ five prep functions. One place to evolve the prep-side
61
+ provenance shape later (e.g. when we want to also stamp the
62
+ PDBFixer / OpenMM versions used).
63
+
64
+ Backwards compatibility: every existing ad-hoc metadata key
65
+ (`metadata["engine"]`, `metadata["run_dir"]`, `metadata["emtol"]`,
66
+ etc.) is preserved unchanged. The new `metadata[PROVENANCE]` is
67
+ additive.
68
+
69
+ Tests (`tests/unit/wrappers/test_provenance_adoption.py`, 8 new
70
+ in addition to the 11 from pass 1):
71
+
72
+ - `TestOpenMMProvenanceChain` (2 tests) drives a full prepare →
73
+ minimize → run pipeline against a real OpenMM install
74
+ (skipped if openmm missing). The headline scenario
75
+ `["ESMFold", "OpenMM.prepare", "OpenMM.run"]` is exercised
76
+ end-to-end.
77
+
78
+ - `TestGROMACSProvenanceWiring` (2 tests) — GROMACS needs the
79
+ `gmx` binary which CI usually lacks. The tests inspect the
80
+ module source to assert the three step engine strings appear
81
+ and that `_parent_provenance(...)` is used — a regression net
82
+ that catches future code that bypasses the helper without
83
+ needing the real engine to run.
84
+
85
+ - `TestPrepProvenanceChain` (4 tests) — exercises each prep
86
+ function individually, then a two-function chain, then the
87
+ full `prepare_for_md` 4-step chain, then the "ESMFold + 4
88
+ prep steps" 5-deep chain (the most realistic scenario).
89
+ Skipped if openmm + pdbfixer aren't both installed.
90
+
91
+ With pass 2 complete, the headline scenario from the roadmap
92
+ ("20 designs from ProteinMPNN, docked with Vina, refined with
93
+ OpenMM") is now fully traceable: every output object's
94
+ `metadata[PROVENANCE].chain()` reads as the producer pipeline,
95
+ with sufficient detail in each step's `parameters` to
96
+ reconstruct the exact call. The remaining provenance work is
97
+ optional polish: persistence to a sidecar format, hash-keyed
98
+ caching (the next roadmap item), and richer engine-version
99
+ introspection.
100
+
101
+ - **Provenance adoption across folding, docking, and generative
102
+ wrappers (pass 1).** Builds on the `Provenance` surface added in
103
+ `9fafbba`. Every wrapper in scope now attaches a `Provenance` to
104
+ the output's `metadata[PROVENANCE]` alongside its existing ad-hoc
105
+ keys, so the "20 designs from ProteinMPNN, docked with Vina"
106
+ scenario is now traceable end-to-end.
107
+
108
+ Wrappers adopted:
109
+ - **Folding**: ESMFold, AlphaFold, Boltz, RoseTTAFold. Each
110
+ records its `__init__` config (model name, device, msa mode,
111
+ recycles, etc.) in `parameters` and the input sequence in
112
+ `inputs`. Folding has no upstream wrapper, so `parent` is
113
+ `None`.
114
+ - **Docking**: Vina, DiffDock. Each `DockingResult.metadata`
115
+ gets a `Provenance` covering the run (box, exhaustiveness,
116
+ seed, etc. in `parameters`; receptor + ligand refs in
117
+ `inputs`); when the receptor was a `Protein` with its own
118
+ `Provenance`, that becomes the `parent` so a Vina pose
119
+ docked against an ESMFold prediction chains back to the
120
+ sequence. Per-pose `Pose.metadata` keeps existing per-pose
121
+ keys (`confidence`, `source_file`) — poses aren't
122
+ independently produced, so they share the result-level
123
+ provenance rather than each carrying their own.
124
+ - **Generative**: RFdiffusion, ProteinMPNN. Each returned
125
+ design (a `Protein` for RFdiffusion, a `DesignedSequence`
126
+ for ProteinMPNN) gets its own `Provenance` — all designs
127
+ from one call share the same Provenance object (frozen +
128
+ immutable, so by-reference sharing is safe). `design_index`
129
+ stays as a separate metadata key, not part of `parameters`,
130
+ since it identifies *which* design rather than the engine
131
+ config.
132
+ - **`molforge.io.load_alphafold`**: the loader helper. The
133
+ engine name reflects that this is the loader, not the
134
+ AlphaFold run itself (`engine="load_alphafold"`); the file
135
+ path goes in `inputs["path"]`.
136
+
137
+ Existing ad-hoc metadata keys are preserved across every change
138
+ — `metadata["engine"]`, `metadata["source_sequence"]`,
139
+ `metadata["source_args"]`, etc. continue to work for 1.x
140
+ backwards compatibility. The new `metadata[PROVENANCE]` is
141
+ *additive*, not a replacement, until 2.x.
142
+
143
+ Two wrapper-side surface changes worth noting:
144
+ - `Vina._parse_poses_pdbqt` and `DiffDock._parse_outputs`
145
+ gained optional `provenance_parameters` /
146
+ `provenance_inputs` / `provenance_parent` kwargs (and
147
+ DiffDock additionally `receptor_ref` / `ligand_ref`). The
148
+ kwargs are optional so legacy tests calling the parsers
149
+ directly without those refs still work — they just don't
150
+ get a Provenance attached. The main `dock()` entry points
151
+ always pass them.
152
+ - The `Vina` module gained a `_provenance_ref` helper for
153
+ converting a `Protein | str | PathLike` to a JSON-native
154
+ string identifier.
155
+
156
+ Tests (`tests/unit/wrappers/test_provenance_adoption.py`, 11
157
+ new): one assertion per adopted wrapper plus a parent-chaining
158
+ integration test that exercises the headline scenario
159
+ (`ESMFold -> Vina` chain). Each adoption test holds the wrapper
160
+ to a uniform contract: `engine` matches the documented name,
161
+ `parameters` contains every promised key, `inputs` contains the
162
+ expected input identifier(s). The chaining test confirms
163
+ `result.metadata[PROVENANCE].chain()` reads as
164
+ `["ESMFold", "Vina"]` oldest-first.
165
+
166
+ - **First-class provenance tracking: `molforge.core.Provenance`.** A
167
+ raw PDB and a folded AlphaFold prediction look identical at the
168
+ AtomArray level; the only difference is *how the structure was
169
+ produced*. Pre-existing engine wrappers already wrote some of this
170
+ information into `metadata` ad-hoc — ESMFold sets
171
+ `metadata["engine"] = "ESMFold"`, RFdiffusion sets
172
+ `metadata["source_args"]` — but the keys are scattered, the shapes
173
+ disagree across engines, and there's no concept of a *parent*
174
+ output, so a chain of operations (fold -> dock -> MD) is not
175
+ traceable. This commit canonicalises the shape.
176
+
177
+ The new `Provenance` dataclass (frozen, JSON-round-trippable) has:
178
+ - **`engine`** — producer name; an engine ("ESMFold", "Vina")
179
+ or a molforge function path ("molforge.prep.prepare_for_md").
180
+ - **`engine_version`** — engine's own version string.
181
+ - **`molforge_version`** — auto-filled from `molforge.__version__`.
182
+ - **`timestamp`** — ISO-8601 UTC, auto-filled.
183
+ - **`parameters`** — engine-specific arguments (must be
184
+ JSON-native; validated eagerly at construction so a wrapper
185
+ can't smuggle in a `Path` or NumPy array that crashes much
186
+ later at serialisation).
187
+ - **`inputs`** — identifiers of the input data (e.g.
188
+ `{"sequence": "MKTVRQ..."}`, `{"receptor": "/path/to.pdb"}`).
189
+ - **`parent`** — the provenance of the input this step
190
+ *consumed*. Recursively a `Provenance` or `None`. This is what
191
+ makes the system compositional: walking the parent chain
192
+ reconstructs the whole history.
193
+
194
+ Construction goes through `Provenance.from_engine(engine=...,
195
+ parameters=..., inputs=..., parent=...)` which auto-fills the two
196
+ version fields and the timestamp so wrappers don't have to think
197
+ about them. The dataclass is frozen — mutating an attached
198
+ provenance would corrupt the audit trail; use `.replace(**changes)`
199
+ to derive an amended copy.
200
+
201
+ Traversal helpers: `walk()` yields self then ancestors newest-first;
202
+ `chain()` returns the same list oldest-first (suitable for printing
203
+ as a left-to-right pipeline); `depth` is the step count.
204
+
205
+ Serialisation: `to_dict()` / `from_dict()` give a stable plain-dict
206
+ shape with parents nested recursively; `to_json()` / `from_json()`
207
+ are JSON convenience wrappers. The on-disk shape is part of the
208
+ stability commitment.
209
+
210
+ The new `metadata_keys.PROVENANCE = "provenance"` constant is the
211
+ documented key; `ProteinMetadata` TypedDict declares it. The
212
+ intended use is `protein.metadata[mk.PROVENANCE] = prov`.
213
+ **Wrappers are NOT updated in this commit** — that's deliberately
214
+ separate work. The existing ad-hoc `metadata["engine"]` keys
215
+ continue to work for the 1.x series; engines opt into the
216
+ `Provenance` system gradually.
217
+
218
+ *NOT persisted through PDB / mmCIF writers* — those preserve only
219
+ the six documented IO header keys (per the mmCIF audit in
220
+ `c3a012e`). Provenance is an in-memory concept; users wanting
221
+ persistence serialise via `to_json` to a sidecar file. This is a
222
+ documented limitation, not a bug. A future "molforge bundle"
223
+ format could carry provenance alongside structure data; out of
224
+ scope here.
225
+
226
+ 33 new tests in `tests/unit/core/test_provenance.py` covering
227
+ construction (minimal, autofill, defensive copies, parent),
228
+ strict JSON-input validation, immutability (FrozenInstanceError +
229
+ `.replace`), traversal (walk / chain / depth), serialisation
230
+ (dict shape, JSON round-trip, missing-engine error, forward-
231
+ compatible deserialisation of older shapes), equality, and the
232
+ metadata-key integration. Module coverage 96.8%; the residual
233
+ two lines are a defensive `_molforge_version` exception fallback.
234
+
235
+ - **mmCIF writer round-trip audit and fidelity fixes.** A systematic
236
+ audit of `write_cif_string` against every PDB fixture in the repo
237
+ surfaced five concrete fidelity bugs in the pre-audit writer; all
238
+ five are now fixed.
239
+ 1. **`model_id == 0` was clobbered to 1.** The old writer used
240
+ `int(model_id) or 1`, which turned every single-model PDB's
241
+ `model_id=0` (the `read_pdb` convention for files without
242
+ MODEL records) into 1 on write. The reader's matching
243
+ `or 1` default reinforced the change — every PDB → CIF
244
+ round-trip silently lost the convention. Writer now emits
245
+ the value verbatim; reader's default flipped from 1 to 0.
246
+ *Affected every PDB fixture (19/19) before the fix.*
247
+ 2. **Partial / non-integer charges were truncated to int.** The
248
+ old writer emitted `f"{int(charge):d}"`, turning typical
249
+ PDBQT / PQR partial charges like `-0.297` into `0` and
250
+ `-1.5` into `-1`. Writer now emits `f"{charge:.4f}"` (4
251
+ decimal places preserves enough precision for typical
252
+ force-field partial charges); zero still emits the `?`
253
+ sentinel so "no charge information" round-trips cleanly.
254
+ 3. **`metadata['classification']` and `metadata['deposition_date']`
255
+ not emitted at all.** Both PDB HEADER fields were captured
256
+ by `read_pdb` and dropped by `write_cif`. Writer now emits
257
+ `_struct_keywords.text` for classification and
258
+ `_pdbx_database_status.recvd_initial_deposition_date` for
259
+ the deposition date; reader picks them up symmetrically.
260
+ 4. **`_entry.id` and `data_<id>` block name could disagree.**
261
+ The old writer used `protein.name` for the block name but
262
+ `metadata[pdb_id]` for `_entry.id`. When those differed, the
263
+ reader's `_entry.id` won and silently rewrote both fields on
264
+ round-trip (e.g. dipeptide.pdb's `name='dipeptide'` became
265
+ `'TEST'` because the HEADER's PDB id was `TEST`). The fix
266
+ uses one chosen identifier — preferring `metadata[pdb_id]`,
267
+ falling back to `protein.name`, then `"molforge"` — for both.
268
+ Identifiers with embedded whitespace (which `read_pdb`
269
+ tolerates from malformed HEADER lines) are now quoted in
270
+ `_entry.id` even though the block name has to substitute
271
+ underscores, so `pdb_id` whitespace survives round-trip.
272
+ When no `pdb_id` exists at all, the writer emits the
273
+ `_entry.id .` mmCIF sentinel and the reader knows to leave
274
+ `metadata[pdb_id]` absent rather than manufacturing one from
275
+ the block name.
276
+ 5. **`serial == 0` was clobbered to `i+1`.** Latent twin of #1;
277
+ no fixture triggered it but the bug was there. Same fix
278
+ pattern: only synthesize a default when `serial <= 0`.
279
+ 38 new tests in `tests/unit/io/test_mmcif.py` codify each fix as
280
+ a regression guard. The `TestFixtureSweep` parametrized test
281
+ iterates every PDB fixture in the repo and asserts that
282
+ coordinates, residue/chain/atom-name fields, residue_id, model_id,
283
+ serial, insertion_code, altloc, record_type, entity_type, and the
284
+ six tracked metadata keys all survive a PDB → CIF → in-memory
285
+ round-trip. Three classes — `TestEntryIdAndBlockNameConsistency`,
286
+ `TestAltlocRoundTrip`, and `TestFixtureSweep` — explicitly
287
+ document the two structural mmCIF limitations: (a) `Protein.name`
288
+ is recovered from `metadata[pdb_id]` on round-trip because mmCIF
289
+ carries only one identifier slot, and (b) altloc round-trip
290
+ requires the caller to pass `altloc="all"` (the default strategy
291
+ collapses to highest-occupancy and drops the label). Module
292
+ coverage at 82.9%; the residual misses are defensive error
293
+ branches in the tokenizer and reader.
294
+ - **Trajectory I/O: `read_trajectory`, `iter_trajectory`,
295
+ `write_trajectory`.** `Trajectory` was previously in-memory only —
296
+ any real MD trajectory bigger than RAM had no way into molforge.
297
+ This commit adds binary-trajectory I/O wrapping mdtraj, exposed
298
+ from `molforge.io`. Supported formats are everything mdtraj
299
+ handles: `.xtc` (GROMACS lossy, the common case), `.trr` (GROMACS
300
+ lossless), `.dcd` (CHARMM / NAMD / OpenMM), `.nc` / `.netcdf`
301
+ (AMBER), `.h5` / `.h5md` (HDF5-based), and multi-MODEL PDB. Three
302
+ functions: `read_trajectory(path, topology=..., stride=1,
303
+ atom_indices=None)` loads a whole file into a
304
+ `molforge.md.Trajectory` (use when it fits in memory);
305
+ `iter_trajectory(path, topology=..., chunk_size=100, stride=1,
306
+ atom_indices=None)` yields chunks of frames as `Trajectory` objects
307
+ with bounded memory; `write_trajectory(trajectory, path)` writes
308
+ out, format inferred from the extension. Coordinates are converted
309
+ nm ↔ Å automatically (molforge convention is Å throughout; mdtraj's
310
+ is nm), times in picoseconds either side. The `topology` argument
311
+ is required for binary formats that don't embed it (XTC, TRR, DCD,
312
+ NetCDF); it accepts either a `molforge.core.Protein` (reused
313
+ directly — no PDB round-trip when the caller passes a Protein in)
314
+ or a path to a PDB. PDB-format trajectories can pass
315
+ `topology=None`. The `atom_indices` parameter slices both the
316
+ coordinate array AND the returned topology, so callers analyzing
317
+ a subset (e.g. backbone atoms only) get a self-consistent
318
+ Trajectory rather than coords-and-topology mismatched. Lazy mdtraj
319
+ import: importing `molforge.io` does not import mdtraj; only the
320
+ trajectory functions do, and they raise
321
+ `MDEngineNotInstalledError` with install instructions when mdtraj
322
+ is absent. NOT wired into `load()` / `save()`: trajectories return
323
+ a different type than the dispatcher's `Protein` / `list[Protein]`
324
+ contract, and the topology argument is something the dispatcher
325
+ has no way to supply. Kept as dedicated entry points. 21 new tests
326
+ in `tests/unit/io/test_trajectory.py` covering reading from PDB,
327
+ the Å unit conversion, Protein-topology reuse identity, path-string
328
+ topology, no-topology PDB, atom-indices subset (including the
329
+ metadata-preserving Protein-slice path), stride, time-array
330
+ carry-through, the "source: mdtraj" metadata marker, the streaming
331
+ chunk count (3+3+3+1 from 10 frames), chunk validation, the
332
+ stride-plus-chunking compose case, the DCD / XTC / PDB write paths
333
+ (XTC round-trip with the documented 0.001-nm precision), and the
334
+ missing-mdtraj error path; mdtraj-using tests are
335
+ `pytest.importorskip` gated. The new module is at high coverage
336
+ (the few uncovered branches are unreachable error-handling
337
+ fallbacks). `pdbfixer.*` was added to mypy's missing-stubs override
338
+ list when `molforge.prep` shipped; no new mypy override is needed
339
+ here since mdtraj is already there.
340
+ - **New subpackage: `molforge.prep` for MD-system preparation.** A
341
+ raw PDB from AlphaFold, RoseTTAFold, the RCSB, or a docking engine
342
+ almost always needs the same clean-up before MD: drop
343
+ crystallographic clutter (waters, buffer salts, sometimes the
344
+ ligand), rebuild missing heavy atoms, cap free termini with ACE /
345
+ NME, and add explicit hydrogens at the right pH. The new
346
+ `molforge.prep` subpackage exposes one composable function per
347
+ step plus a convenience pipeline:
348
+ - `remove_heterogens(protein)` — pure-Python residue-name filter.
349
+ By default drops waters, ions, ligands, and everything outside
350
+ the 20 canonical amino acids + standard nucleotides. `keep_water`
351
+ / `keep_ions` / `keep_ligands` toggles plus an explicit `keep`
352
+ allow-list for cofactors. Recognises multiple water aliases
353
+ (HOH, WAT, H2O, SOL, TIP*) and the common monatomic ions.
354
+ - `fix_missing_atoms(protein)` — wraps PDBFixer's rotamer-library
355
+ rebuild for incomplete side chains. `fix_missing_residues=False`
356
+ by default (de-novo loop modelling is risky);
357
+ `replace_nonstandard=True` by default (MSE → MET, etc.).
358
+ - `add_caps(protein)` — wraps PDBFixer to add ACE / NME caps at
359
+ free termini of every protein chain. Multi-chain aware;
360
+ non-protein chains (ligands, DNA) skipped. Either cap can be
361
+ disabled with an empty string.
362
+ - `add_hydrogens(protein, pH=7.4)` — wraps OpenMM
363
+ `Modeller.addHydrogens` for pH-aware protonation. Idempotent on
364
+ already-protonated input. The `force_field` kwarg accepts both
365
+ registered aliases (`"amber14"`, `"charmm36"`) and bare XML
366
+ filenames.
367
+ - `prepare_for_md(protein)` — convenience entry point that chains
368
+ the four steps with sensible defaults for an
369
+ AlphaFold-PDB-to-OpenMM workflow. Each step's options are
370
+ forwarded; individual steps can be turned off
371
+ (`add_caps_to_termini=False`, `add_explicit_hydrogens=False`).
372
+
373
+ ## [v0.3.0] 2026-06-22
374
+
375
+ ### Added
376
+ - **`molforge.io.read_pqr` / `write_pqr` are implemented.** PQR
377
+ (PDB2PQR / APBS) was a committed import path but a
378
+ `NotImplementedError` stub; it now parses and writes PQR files,
379
+ completing the format-I/O backlog (SDF, MOL2, PDBQT, PQR all real).
380
+ PQR is a PDB-like format that appends per-atom partial charge and
381
+ atomic radius as whitespace-separated trailing fields. Unlike PDB
382
+ or PDBQT, PQR is **not** strictly fixed-column past the
383
+ coordinates — different generators (PDB2PQR, AMBER, CHARMM, APBS)
384
+ emit different widths. The reader handles all of them by parsing
385
+ columns 1-54 as fixed (the atom record through coordinates,
386
+ PDB-compatible) and whitespace-splitting the remainder for charge
387
+ and radius. Charges land on `AtomArray.charge`; radii land on
388
+ `protein.metadata["radii"]` as a per-atom list (no native
389
+ `radius` field on `AtomArray` — electrostatics is a small enough
390
+ slice of the surface that this didn't warrant a core-type change).
391
+ The writer is the symmetric operation: it calls `write_pdb_string`,
392
+ truncates each atom line at column 54, then appends
393
+ `charge radius`. When no radii are recorded in metadata, a default
394
+ 1.5 Å is used (a reasonable middle-of-the-road heavy-atom radius
395
+ that lets a charge-only Protein still be written as PQR). 19 new
396
+ tests in `tests/unit/io/test_pqr.py` covering the charge/radius
397
+ extractor (clean tokens, trailing garbage, defaults), reading from
398
+ string and from disk, the dispatcher routes, two variant-width
399
+ tails seen in real-world PDB2PQR / AMBER output, the full
400
+ round-trip (coordinates, charges, radii), and the default-radius
401
+ writer path; `pqr.py` is at 92.0% coverage.
402
+ `test_dispatch.py`: the two stub-format tests (load and save) now
403
+ monkeypatch a synthetic planned format rather than depending on
404
+ any real format being unimplemented — the dispatcher's
405
+ planned-readers fallback machinery still exists for future use, so
406
+ the tests are still meaningful. The `_PLANNED_READERS` dict in
407
+ `dispatch.py` is now empty.
408
+ - **`molforge.io.read_pdbqt` / `write_pdbqt` are implemented.** PDBQT
409
+ (AutoDock / Vina) was a committed import path but a
410
+ `NotImplementedError` stub; it now parses and writes PDBQT files.
411
+ PDBQT is a thin extension of PDB — columns 1-66 are PDB-compatible,
412
+ columns 71-76 hold the per-atom partial charge, and columns 78-79
413
+ hold the AutoDock atom type (`C`, `OA`, `HD`, `NA`, ...). The
414
+ reader reuses `molforge.io.read_pdb_string` for the heavy lifting
415
+ (atom-array construction, altloc handling, entity classification,
416
+ multi-MODEL parsing) and post-processes each atom line to pick up
417
+ the extra columns: charges are written to `AtomArray.charge`,
418
+ AutoDock types land on `protein.metadata["autodock_types"]` as a
419
+ per-atom list. `ROOT` / `BRANCH` / `TORSDOF` rotatable-bond markers
420
+ are read-tolerated (recognised and skipped — `AtomArray` doesn't
421
+ carry bond topology). The writer is the symmetric operation: it
422
+ calls `write_pdb_string`, then rewrites each `ATOM` / `HETATM` line
423
+ to append the charge and AutoDock-type columns; when no AutoDock
424
+ type is recorded in metadata, the element is used as a documented
425
+ best-effort fallback. Round-tripping preserves coordinates,
426
+ charges, and types. The Vina wrapper's pose parser is refactored to
427
+ go through `read_pdbqt_string` rather than its previous "truncate
428
+ every atom line to 66 columns and feed to the PDB reader" hack — so
429
+ per-atom charges now propagate to `Pose.ligand` instead of being
430
+ silently discarded. 23 new tests in `tests/unit/io/test_pdbqt.py`
431
+ covering the column extractors (charge and AutoDock type, including
432
+ the whitespace-split fallback), reading from string and from disk,
433
+ the dispatcher routes, the full round-trip (coordinates, charges,
434
+ types), the element-fallback writer path, multi-MODEL handling
435
+ (Vina pose output), and `ROOT` / `BRANCH` / `TORSDOF` tolerance;
436
+ `pdbqt.py` is at 93.8% coverage. The `test_dispatch.py` stub-format
437
+ tests are updated to use `.pqr` (the only remaining stub format).
438
+ - **`molforge.io.read_mol2` / `write_mol2` are implemented.** MOL2
439
+ (Tripos) was a committed import path but a `NotImplementedError`
440
+ stub; it now parses and writes Tripos MOL2 files. Like the SDF
441
+ reader, `read_mol2` is multi-molecule by default and returns
442
+ `list[Protein]` (the format supports multi-molecule files via
443
+ repeated `@<TRIPOS>MOLECULE` markers, common in docking output
444
+ libraries). The reader populates coordinates, elements (extracted
445
+ from the prefix of the Tripos atom type — `C.ar` → `C`, `N.am` →
446
+ `N`, two-letter `Cl`/`Br` preserved), atom names, per-atom partial
447
+ charges from the atom line's last column, and substructure info
448
+ (residue id / name from the MOL2 `subst_id` / `subst_name`
449
+ columns). Short atom lines (optional trailing columns omitted) and
450
+ non-conforming writers that emit `***` for the subst_id or a
451
+ non-numeric charge are tolerated with silent fallbacks rather
452
+ than crashing the whole molecule. Bond orders, ring information,
453
+ stereochemistry, and the `@<TRIPOS>SUBSTRUCTURE` /
454
+ `@<TRIPOS>CRYSIN` / `@<TRIPOS>UNITY` sections are intentionally
455
+ dropped — those need a chemistry toolkit. The writer emits a
456
+ minimal, spec-conformant MOL2 with an empty `@<TRIPOS>BOND` section
457
+ (some downstream tools error without the tag). The MOLECULE header
458
+ declares the atom count; a mismatch between that and the ATOM
459
+ section raises a clear error. `read_mol2` is wired into
460
+ `molforge.io.load`. 34 new tests in `tests/unit/io/test_mol2.py`
461
+ covering single/multi-molecule reading, Tripos atom-type element
462
+ extraction, two-letter elements, partial charges, optional-column
463
+ fallbacks, blank-line tolerance, every error path, dispatcher
464
+ integration, and the full round-trip; mol2.py is at 94.4% coverage
465
+ (the residual misses are unreachable defensive branches).
466
+ - **`molforge.io.read_sdf` / `write_sdf` are implemented.** SDF was a
467
+ committed import path but a `NotImplementedError` stub; it now
468
+ parses and writes V2000 SDF / MOL files. The reader is multi-
469
+ molecule by default, returning a `list[Protein]` (a single-molecule
470
+ `.mol` file still returns a one-element list, keeping the return
471
+ type uniform across callers). The atom block, title line, and the
472
+ ``> <Name>`` / value property block all round-trip; properties land
473
+ on `Protein.metadata["properties"]`. The implementation uses no
474
+ chemistry toolkit — the V2000 atom block has a fixed positional
475
+ layout that's enough for everything molforge does downstream
476
+ (coordinates, pose ranking, distance calculations). Bond orders,
477
+ aromaticity, and stereochemistry are intentionally dropped; users
478
+ who need them should call RDKit directly. V3000 files are detected
479
+ and raise a clear error pointing at conversion paths. `read_sdf` is
480
+ wired into `molforge.io.load`, so `load("foo.sdf")` works. The
481
+ DiffDock wrapper, which previously parsed SDF inline, now goes
482
+ through `molforge.io.sdf.read_sdf_string` — the inline
483
+ `_ligand_from_sdf` helper is removed (-1 duplicated parser).
484
+ `api-stability.md` is updated; MOL2, PDBQT, and PQR remain
485
+ tentative. 26 new tests in `tests/unit/io/test_sdf.py` covering
486
+ single/multi-molecule reading, the title and property block,
487
+ round-trip writing, dispatcher integration, and every error path;
488
+ the eight DiffDock tests that exercised the inline parser are
489
+ removed (now subsumed by the SDF tests).
490
+ - **`GROMACS` is now a real MD engine.** `GROMACS`
491
+ (`molforge.wrappers.md`) was a coherent stub whose `prepare` /
492
+ `minimize` / `run` all raised `NotImplementedError`; it is now
493
+ fully implemented. [GROMACS](https://www.gromacs.org/) is a
494
+ command-line program (`gmx`), not a Python library, so the wrapper
495
+ drives it as a subprocess. One `prepare` / `minimize` / `run` cycle
496
+ maps onto the standard GROMACS workflow: `prepare` runs
497
+ `pdb2gmx` → `editconf` → (optionally) `solvate`; `minimize` writes
498
+ a steepest-descent `.mdp`, then `grompp` → `mdrun`; `run` writes a
499
+ production `.mdp`, runs `grompp` → `mdrun`, then reads the frames
500
+ back with `trjconv` and per-frame energies with `gmx energy`. All
501
+ state for a simulation lives in one run directory whose path is
502
+ carried on `Simulation.engine_handle` (and mirrored in
503
+ `metadata["run_dir"]`), so `minimize` and `run` continue from
504
+ whatever `prepare` produced. Trajectory frames are read back by
505
+ asking GROMACS itself (`gmx trjconv`) to convert its binary `.xtc`
506
+ to a multi-model PDB, which molforge's own PDB reader then parses —
507
+ the wrapper deliberately takes no dependency on a third-party
508
+ binary-trajectory library. Three small fixed-layout parsers
509
+ (`.gro` coordinates, multi-model PDB, `.xvg` columns) handle the
510
+ GROMACS outputs directly. `gmx` is resolved lazily via
511
+ `shutil.which`, so construction never touches the filesystem; a
512
+ clear `MDEngineNotInstalledError` (pointing at OpenMM) is raised
513
+ when it is absent. A constructor flag covers the water model, box
514
+ margin/type, and a `verbose` pass-through. 36 tests (a new
515
+ `test_gromacs.py`), covering construction and validation, `gmx`
516
+ resolution, the three parsers and their error paths, and the full
517
+ `prepare` / `minimize` / `run` pipeline driven by a mocked
518
+ `subprocess.run` that writes the files each `gmx` step would
519
+ produce; the wrapper module is at 94% coverage (the residual
520
+ misses are defensive `except` branches for corrupt output GROMACS
521
+ would never actually emit). The 6 obsolete `TestGROMACSStub` tests
522
+ are removed.
523
+ - **`DiffDock` is now a real docking engine.** `DiffDock`
524
+ (`molforge.wrappers.docking`) was a coherent stub whose `dock()`
525
+ raised `NotImplementedError`; it is now fully implemented.
526
+ [DiffDock](https://github.com/gcorso/DiffDock) is a
527
+ diffusion-generative model for *blind* protein-ligand docking — it
528
+ needs no search box, sampling poses over the whole receptor and
529
+ ranking them with a learned confidence model. Like the
530
+ `RoseTTAFold` wrapper, DiffDock ships as a research repository
531
+ rather than a pip package, so the wrapper drives it as a
532
+ subprocess: it locates the cloned repo (`$DIFFDOCK_HOME` or an
533
+ explicit `repo_dir`), materializes the receptor to PDB, accepts the
534
+ ligand as a SMILES string or a path to an SDF/MOL2 file, runs
535
+ `python -m inference`, and parses the ranked
536
+ `rank{N}_confidence{C}.sdf` output into a `DockingResult`. DiffDock
537
+ reports a *confidence* (higher = better), the opposite of Vina's
538
+ affinity convention; the wrapper stores the raw value in
539
+ `Pose.metadata["confidence"]` and sets `Pose.score` to its
540
+ negation, so `score` ascending is best-first for every engine. SDF
541
+ poses are parsed by reading the V2000 atom block directly (molforge's
542
+ RDKit-backed SDF reader is still a stub, and the atom block —
543
+ 3D coordinates plus element symbols — needs no chemistry toolkit).
544
+ A constructor flag covers `samples_per_complex`, `inference_steps`,
545
+ and `batch_size`. 30 tests (a new `test_diffdock.py`), covering
546
+ construction and validation, install resolution, SDF and
547
+ confidence-from-filename parsing, and the `_run_cli` subprocess seam
548
+ via a mocked `subprocess.run`; the wrapper module is at 100%
549
+ coverage. The 4 obsolete `TestDiffDockStub` tests are removed.
550
+ - **OpenMM wrapper test coverage raised from 24% to 95%.** The
551
+ OpenMM tests previously gated every real path behind
552
+ `skipif(openmm installed)` — so `prepare` / `minimize` / `run`
553
+ were exercised by *nothing*: when openmm was absent they couldn't
554
+ run, and when present the negative-path tests skipped. The file is
555
+ restructured into a dependency-free half (construction, the
556
+ force-field registry, the missing-dependency errors) and a new
557
+ `TestRealOpenMM` class that runs `prepare` / `minimize` / `run`
558
+ end to end against a real OpenMM install — system building,
559
+ hydrogen addition, the minimizer, the integration loop, trajectory
560
+ assembly, and argument validation. A new chemically complete
561
+ heavy-atom fixture, `tests/fixtures/pdb/ala_tripeptide_heavy.pdb`
562
+ (ALA-ALA-ALA with all standard heavy atoms plus the C-terminal
563
+ OXT), gives the force field something it can template. The
564
+ real-engine tests are deliberately *not* marked `slow` — the
565
+ tripeptide is tiny and a 20-step run is sub-second — so they run
566
+ in the normal suite wherever openmm is installed, and skip cleanly
567
+ (9 skips) where it isn't. A new CI job, `md-openmm`, installs the
568
+ `[md]` extra and runs the MD wrapper tests on every push so
569
+ `TestRealOpenMM` is actually exercised.
570
+ - **RFdiffusion wrapper test coverage raised from 84% to 99%.** The
571
+ `_run_cli` subprocess seam of `wrappers.generative.rfdiffusion` —
572
+ previously untested — now has direct tests via a mocked
573
+ `subprocess.run` (no RFdiffusion or torch needed): command and
574
+ Hydra-arg assembly, the `design_*.pdb` output-parsing path, the
575
+ no-output `RuntimeError`, `CalledProcessError` → `RuntimeError`
576
+ translation, the public `generate()` entry point, and
577
+ `contigs` / `symmetry` pass-through. 5 new tests (16 → 21 in the
578
+ file). Mirrors the ProteinMPNN coverage work from the previous
579
+ cycle.
580
+
581
+ ### Removed
582
+ - **BREAKING `molforge.wrappers.folding.Rosetta` removed.** The `Rosetta`
583
+ name was a placeholder from the `0.0.x` series whose meaning was
584
+ ambiguous — it could read as PyRosetta (the classical
585
+ sequence-design library) or RoseTTAFold (the deep-learning
586
+ model). The real wrapper now lives at `RoseTTAFold`. `Rosetta`
587
+ had been kept this cycle as a `DeprecationWarning`-emitting alias,
588
+ but since it never appeared in a tagged release, carrying it —
589
+ and the day-one deprecation it implies — into the 1.0 stable
590
+ surface added nothing. It is removed outright: import
591
+ `RoseTTAFold` instead. A PyRosetta wrapper, if ever added, would
592
+ be a separate class (`PyRosetta`) in its own module, since
593
+ PyRosetta's surface is far wider than the `FoldingEngine`
594
+ contract. The 3 alias tests are removed (folding-engine count
595
+ unchanged: ESMFold, AlphaFold, Boltz, RoseTTAFold).
596
+
597
+ ### Fixed
598
+ +- **`OpenMM.prepare()` now adds missing hydrogens, so heavy-atom
599
+ structures are usable.** A force field needs explicit hydrogens,
600
+ but `prepare()` called `ForceField.createSystem()` directly on
601
+ whatever atoms the input had. Heavy-atom structures — the normal
602
+ output of every folding and docking engine, and what most PDB
603
+ files on disk contain — therefore failed with a cryptic
604
+ OpenMM "no template found for residue" error, making the wrapper
605
+ effectively unusable on exactly the structures molforge produces.
606
+ `prepare()` now runs `Modeller.addHydrogens()` before building the
607
+ system; the step is idempotent, so an already-protonated structure
608
+ is unaffected. Because adding hydrogens changes the atom count, the
609
+ molforge `Protein` attached to the returned `Simulation` is rebuilt
610
+ from the protonated structure, so its topology and the coordinate
611
+ array agree (previously a heavy-atom topology could be paired with
612
+ a protonated coordinate array). A new `add_hydrogens` constructor
613
+ flag (default `True`) lets callers who have pre-protonated their
614
+ structure opt out.
615
+
616
+ ## [v0.2.0] 2026-05-26
617
+
10
618
  ### Added
11
619
  - **ProteinMPNN wrapper test coverage raised from 69% to 96%.** The
12
620
  two previously-untested seams of `wrappers.generative.proteinmpnn`
@@ -72,148 +680,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
72
680
  still permitted but carry no cross-version stability guarantee. 15
73
681
  new tests, including consistency checks that the parsers only emit
74
682
  documented keys and that the TypedDict matches `DOCUMENTED_KEYS`.
75
-
76
- ### Fixed
77
- - **`load_alphafold` now emits the uniform confidence metadata keys.**
78
- `molforge.io.load_alphafold` previously wrote only AlphaFold-specific
79
- keys (`plddt`, `plddt_per_residue`, `mean_plddt`, `source`), while
80
- the AlphaFold *wrapper* wrote the cross-engine-uniform keys
81
- (`confidence_per_atom`, `confidence_per_residue`, `mean_confidence`,
82
- `engine`). Downstream code reading confidence uniformly across
83
- engines silently missed AlphaFold structures loaded from disk.
84
- `load_alphafold` now populates both sets (uniform keys preferred,
85
- legacy keys retained for backward compatibility); the two carry
86
- identical values. Surfaced by the API audit.
87
- - **`GROMACS` and `DiffDock` are now coherent stubs.** Both are
88
- exported (committed import paths) but unimplemented. Previously
89
- they were *incoherent*: `GROMACS` didn't implement its `MDEngine`
90
- abstract methods at all, so `GROMACS()` failed with a cryptic
91
- "Can't instantiate abstract class" `TypeError` rather than a
92
- meaningful message; both engines' methods raised a bare
93
- `NotImplementedError` with no text. They are now coherent stubs —
94
- instantiable, satisfying their respective engine ABCs
95
- (`MDEngine` / `DockingEngine`), with every method raising
96
- `NotImplementedError` carrying a clear message that points at the
97
- working alternative (`OpenMM` / `Vina`) and the tracking issue.
98
- 10 new tests. Surfaced by the API audit.
99
- - - **Lint drift from a Ruff version bump cleared; CI lint job green
100
- again.** `.pre-commit-config.yaml` pinned `ruff-pre-commit` at
101
- `v0.5.0`, but the `[dev]` extra installs `ruff>=0.5` unpinned, so
102
- CI resolved a much newer Ruff (0.15.x) whose added rules flagged
103
- 33 pre-existing issues — meaning the CI `lint` job was effectively
104
- red. All 33 are now resolved: a genuine dead variable in
105
- `ensembles.clustering` removed, an unused `shutil` import dropped,
106
- five `pytest.raises(match=...)` patterns with unescaped regex
107
- metacharacters made explicit (raw strings / escaped dots), a
108
- `zip()` given an explicit `strict=`, four nested `with` statements
109
- collapsed, a `getattr()` call with a string literal in
110
- `ensembles.weighting` replaced by a `cast`-backed direct attribute
111
- access (dropping a now-misplaced `# noqa`), and a Ruff-version
112
- formatting refresh applied across 24 files (cosmetic line-joining
113
- only). Two intentional-notation cases
114
- are configured rather than rewritten: `allowed-confusables`
115
- permits `×`, `σ`, and `–` in docstrings (matrix dimensions, the
116
- standard deviation, prose dashes), and `RUF022` is per-file-ignored
117
- for the two modules whose `__all__` is deliberately grouped by
118
- category with section comments. The `ruff` and `mypy` pre-commit
119
- pins are bumped to the versions CI resolves, so the two stay in
120
- lock-step and this drift cannot silently recur. No source-behaviour
121
- or test-count change (918 pass + 11 skipped, unchanged).
122
-
123
- ### Changed
124
- - **BREAKING: `cross_validate` now defaults to `on_error="raise"`.**
125
- Previously `cross_validate` defaulted to `on_error="record"` —
126
- exceptions raised by individual validators were silently caught,
127
- recorded in verdict metadata, and the verdict marked
128
- `passed=False`. The problem: a validator that throws on every
129
- design (a misconfigured engine, a missing dependency, a bad
130
- input) produced a full list of `passed=False` verdicts that
131
- *looked* like a real result, hiding the bug. The new default
132
- fails loud. Code that genuinely wants a batch to survive
133
- individual validator failures must now pass `on_error="record"`
134
- explicitly. Flagged and resolved by the API audit.
135
- - **The entire `molforge` package is now `mypy --strict` clean.**
136
- With `wrappers` and `plugins` brought up to strict, all 77 source
137
- modules across every subpackage pass `mypy --strict` with zero
138
- errors. The CI `typecheck` job is correspondingly simplified: the
139
- previous two-step arrangement (a strict gate on the clean
140
- subpackages plus a non-blocking informational full-tree run)
141
- collapses to a single `mypy src` gate that fails the build on any
142
- type error. The `tests/unit/test_typing.py` regression test is
143
- likewise simplified to one whole-package check. 31 errors fixed in
144
- this final tranche: 20 stale `# type: ignore` comments (made
145
- redundant when the optional heavy dependencies were added to the
146
- mypy `ignore_missing_imports` override), four deliberate engine-
147
- method `# type: ignore[override]` annotations (the concrete
148
- engine wrappers refine the permissive `**kwargs` signatures of
149
- their `DockingEngine` / `MDEngine` / `GenerativeEngine` abstract
150
- bases — an intentional, documented refinement that mypy's strict
151
- Liskov check cannot model), `cast`s for the opaque
152
- `Simulation.engine_handle` inside the OpenMM wrapper and for the
153
- unstubbed-dependency return values, and `Vina.dock`'s receptor
154
- narrowing switched from `hasattr` to `isinstance` (a more correct
155
- check that mypy can also narrow on).
156
- - **`molforge.ml` is now `mypy --strict` clean.** The ML subpackage
157
- (sequence/structure featurization, protein-language-model
158
- embeddings) joins the strict gate — eight strict-clean
159
- subpackages in total, 51 source files. Six errors fixed: the four
160
- numpy-widening `no-any-return`s in `embeddings.py` (resolved with
161
- `cast`s), and two real type bugs in `structure_features.py` —
162
- `pair_distances` and `pair_distance_features` declared
163
- `atom_choice: str` but pass it to `distance_map`, which requires
164
- the `Literal["ca","cb","heavy","all"]` the docstrings already
165
- specify, and a coordinate feature array silently upcast to
166
- float64 by a division. The `torch` and `transformers` (and
167
- `colabfold`, `meeko`, `vina`) optional heavy dependencies, which
168
- ship no type stubs, are added to the mypy `ignore_missing_imports`
169
- override alongside the existing `Bio` / `biotite` / `mdtraj` /
170
- `openmm` / `rdkit` entries. CI strict gate and the
171
- `tests/unit/test_typing.py` regression test updated; only
172
- `plugins` and `wrappers` remain outside the gate.
173
- - **Six more subpackages are now `mypy --strict` clean.**
174
- `molforge.io`, `molforge.sequence`, `molforge.structure`,
175
- `molforge.metrics`, `molforge.ensembles`, and
176
- `molforge.validation` now pass `mypy --strict` with zero errors,
177
- joining `molforge.core` — seven strict-clean subpackages in total,
178
- 46 source files. The 12 errors fixed were mostly numpy operations
179
- mypy widens to `Any` (resolved with explicit `cast`s that document
180
- the known array dtype) and two stale `type: ignore` comments; two
181
- were genuine annotation bugs — `_place_hydrogens` in `dssp.py` was
182
- declared to return a single array but actually returns a
183
- `(coords, mask)` tuple, and `_score` in `alignment.py` was
184
- declared `NDArray[np.int_]` but builds an `int32` array (`np.int_`
185
- is `int64` on 64-bit platforms). The CI strict gate now covers all
186
- seven subpackages; the regression test
187
- (`tests/unit/test_typing.py`, moved up from `tests/unit/core/` and
188
- parametrized) checks each one in-suite. The remaining subpackages
189
- (`ml`, `plugins`, `wrappers`) are still tracked by the
190
- non-blocking informational `mypy src` CI step.
191
- - **`molforge.core` is now `mypy --strict` clean, and CI enforces
192
- it.** The `core` subpackage — the data model the rest of the
193
- library is built on — now passes `mypy --strict` with zero
194
- errors (fixed: two missing `NDArray` type arguments in
195
- `AtomArray`, an `Any`-return in `Atom.coord`, and an untyped
196
- `Chain.__iter__` that was suppressed with a `type: ignore`). The
197
- CI `typecheck` job now runs `mypy --strict src/molforge/core/`
198
- as a hard gate, with a separate non-blocking full-tree `mypy src`
199
- step that keeps the remaining (out-of-`core`) type errors visible
200
- while they're worked through. A new `slow`-marked regression test
201
- (`tests/unit/core/test_typing.py`) runs the strict check in-suite
202
- so a `core` type regression is caught locally too.
203
-
204
- ### Documented
205
- - **`Simulation.engine_handle` contract clarified.** The attribute
206
- type (`object | None`) is correct — it really is an opaque,
207
- engine-specific handle — but the contract was under-specified.
208
- The docstring now states explicitly that `engine_handle` is
209
- engine-private (callers must not inspect it or set it), is **not
210
- serialized** (it typically wraps unpicklable C-extension state;
211
- persistence layers must drop it and let the engine wrapper
212
- rebuild it on resume), and carries **no semver guarantee**. For
213
- inspectable per-simulation data, `Simulation.metadata` is the
214
- supported field. No code change. Flagged by the API audit.
215
-
216
- ### Added
217
683
  - **RoseTTAFold All-Atom folding wrapper.** New file
218
684
  `src/molforge/wrappers/folding/rosettafold.py` implements a real
219
685
  wrapper around the Baker lab's RoseTTAFold-All-Atom (Krishna et
@@ -241,25 +707,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
241
707
  prediction matching the rest of the folding wrappers; protein-
242
708
  ligand and covalent-modification co-folding (RFAA's headline
243
709
  capability) need a separate `predict_complex()` surface and
244
- remain planned. 47 new tests (45 passing + 2 correctly skipped:
710
+ remain planned.
711
+ - 47 new tests (45 passing + 2 correctly skipped:
245
712
  one for the torch tensor conversion when torch isn't installed,
246
713
  one @slow end-to-end requiring `$RFAA_HOME`). Total test count:
247
714
  830 → 875 passed + 11 skipped.
248
-
249
- ### Deprecated
250
- - **`molforge.wrappers.folding.Rosetta` is now a deprecated alias
251
- for `RoseTTAFold`.** The original `rosetta.py` placeholder was
252
- ambiguous about whether it referred to PyRosetta (the Baker lab's
253
- classical sequence-design library) or RoseTTAFold (the deep-
254
- learning model). The new real wrapper lives at
255
- `RoseTTAFold` for clarity. `Rosetta` is retained as a thin
256
- subclass that emits `DeprecationWarning` on construction so
257
- existing imports / isinstance checks keep working through the
258
- next minor release. A PyRosetta wrapper, if added, would live in
259
- a separate module (`pyrosetta.py`) since PyRosetta's surface is
260
- much wider than the `FoldingEngine` contract.
261
-
262
- ### Added
263
715
  - **Boltz / Boltz-2 folding wrapper.** Real implementation replacing
264
716
  the `boltz.py` stub. Drives the `boltz predict` CLI via subprocess
265
717
  against a temporary directory and parses the resulting mmCIF +
@@ -275,7 +727,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
275
727
  metadata follows the uniform folding-engine convention
276
728
  (`confidence_per_residue`, `confidence_per_atom`, `mean_confidence`)
277
729
  and additionally surfaces Boltz-specific `ptm`, `iptm`, and
278
- `confidence_score` from the JSON sidecar. 47 new tests (46 passing
730
+ `confidence_score` from the JSON sidecar.
731
+ - 47 new tests (46 passing
279
732
  + 1 correctly skipped @slow end-to-end), structured as a series of
280
733
  testable seams: construction, sequence validation, YAML input
281
734
  construction, command-line assembly, environment setup, output
@@ -477,6 +930,51 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
477
930
  the empty-entry-points fallthrough.
478
931
 
479
932
  ### Fixed
933
+ - **`load_alphafold` now emits the uniform confidence metadata keys.**
934
+ `molforge.io.load_alphafold` previously wrote only AlphaFold-specific
935
+ keys (`plddt`, `plddt_per_residue`, `mean_plddt`, `source`), while
936
+ the AlphaFold *wrapper* wrote the cross-engine-uniform keys
937
+ (`confidence_per_atom`, `confidence_per_residue`, `mean_confidence`,
938
+ `engine`). Downstream code reading confidence uniformly across
939
+ engines silently missed AlphaFold structures loaded from disk.
940
+ `load_alphafold` now populates both sets (uniform keys preferred,
941
+ legacy keys retained for backward compatibility); the two carry
942
+ identical values. Surfaced by the API audit.
943
+ - **`GROMACS` and `DiffDock` are now coherent stubs.** Both are
944
+ exported (committed import paths) but unimplemented. Previously
945
+ they were *incoherent*: `GROMACS` didn't implement its `MDEngine`
946
+ abstract methods at all, so `GROMACS()` failed with a cryptic
947
+ "Can't instantiate abstract class" `TypeError` rather than a
948
+ meaningful message; both engines' methods raised a bare
949
+ `NotImplementedError` with no text. They are now coherent stubs —
950
+ instantiable, satisfying their respective engine ABCs
951
+ (`MDEngine` / `DockingEngine`), with every method raising
952
+ `NotImplementedError` carrying a clear message that points at the
953
+ working alternative (`OpenMM` / `Vina`) and the tracking issue.
954
+ 10 new tests. Surfaced by the API audit.
955
+ - - **Lint drift from a Ruff version bump cleared; CI lint job green
956
+ again.** `.pre-commit-config.yaml` pinned `ruff-pre-commit` at
957
+ `v0.5.0`, but the `[dev]` extra installs `ruff>=0.5` unpinned, so
958
+ CI resolved a much newer Ruff (0.15.x) whose added rules flagged
959
+ 33 pre-existing issues — meaning the CI `lint` job was effectively
960
+ red. All 33 are now resolved: a genuine dead variable in
961
+ `ensembles.clustering` removed, an unused `shutil` import dropped,
962
+ five `pytest.raises(match=...)` patterns with unescaped regex
963
+ metacharacters made explicit (raw strings / escaped dots), a
964
+ `zip()` given an explicit `strict=`, four nested `with` statements
965
+ collapsed, a `getattr()` call with a string literal in
966
+ `ensembles.weighting` replaced by a `cast`-backed direct attribute
967
+ access (dropping a now-misplaced `# noqa`), and a Ruff-version
968
+ formatting refresh applied across 24 files (cosmetic line-joining
969
+ only). Two intentional-notation cases
970
+ are configured rather than rewritten: `allowed-confusables`
971
+ permits `×`, `σ`, and `–` in docstrings (matrix dimensions, the
972
+ standard deviation, prose dashes), and `RUF022` is per-file-ignored
973
+ for the two modules whose `__all__` is deliberately grouped by
974
+ category with section comments. The `ruff` and `mypy` pre-commit
975
+ pins are bumped to the versions CI resolves, so the two stay in
976
+ lock-step and this drift cannot silently recur. No source-behaviour
977
+ or test-count change (918 pass + 11 skipped, unchanged).
480
978
  - **Docs notebooks no longer use symlinks.** The walkthrough and
481
979
  example notebooks were previously symlinked from `docs/` into the
482
980
  canonical `notebooks/` directory. Symlinks broke two things: (1)
@@ -499,6 +997,112 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
499
997
  the actual public attributes. Discovered while writing ensemble
500
998
  test fixtures.
501
999
 
1000
+ ### Changed
1001
+ - **BREAKING: `cross_validate` now defaults to `on_error="raise"`.**
1002
+ Previously `cross_validate` defaulted to `on_error="record"` —
1003
+ exceptions raised by individual validators were silently caught,
1004
+ recorded in verdict metadata, and the verdict marked
1005
+ `passed=False`. The problem: a validator that throws on every
1006
+ design (a misconfigured engine, a missing dependency, a bad
1007
+ input) produced a full list of `passed=False` verdicts that
1008
+ *looked* like a real result, hiding the bug. The new default
1009
+ fails loud. Code that genuinely wants a batch to survive
1010
+ individual validator failures must now pass `on_error="record"`
1011
+ explicitly. Flagged and resolved by the API audit.
1012
+ - **The entire `molforge` package is now `mypy --strict` clean.**
1013
+ With `wrappers` and `plugins` brought up to strict, all 77 source
1014
+ modules across every subpackage pass `mypy --strict` with zero
1015
+ errors. The CI `typecheck` job is correspondingly simplified: the
1016
+ previous two-step arrangement (a strict gate on the clean
1017
+ subpackages plus a non-blocking informational full-tree run)
1018
+ collapses to a single `mypy src` gate that fails the build on any
1019
+ type error. The `tests/unit/test_typing.py` regression test is
1020
+ likewise simplified to one whole-package check. 31 errors fixed in
1021
+ this final tranche: 20 stale `# type: ignore` comments (made
1022
+ redundant when the optional heavy dependencies were added to the
1023
+ mypy `ignore_missing_imports` override), four deliberate engine-
1024
+ method `# type: ignore[override]` annotations (the concrete
1025
+ engine wrappers refine the permissive `**kwargs` signatures of
1026
+ their `DockingEngine` / `MDEngine` / `GenerativeEngine` abstract
1027
+ bases — an intentional, documented refinement that mypy's strict
1028
+ Liskov check cannot model), `cast`s for the opaque
1029
+ `Simulation.engine_handle` inside the OpenMM wrapper and for the
1030
+ unstubbed-dependency return values, and `Vina.dock`'s receptor
1031
+ narrowing switched from `hasattr` to `isinstance` (a more correct
1032
+ check that mypy can also narrow on).
1033
+ - **`molforge.ml` is now `mypy --strict` clean.** The ML subpackage
1034
+ (sequence/structure featurization, protein-language-model
1035
+ embeddings) joins the strict gate — eight strict-clean
1036
+ subpackages in total, 51 source files. Six errors fixed: the four
1037
+ numpy-widening `no-any-return`s in `embeddings.py` (resolved with
1038
+ `cast`s), and two real type bugs in `structure_features.py` —
1039
+ `pair_distances` and `pair_distance_features` declared
1040
+ `atom_choice: str` but pass it to `distance_map`, which requires
1041
+ the `Literal["ca","cb","heavy","all"]` the docstrings already
1042
+ specify, and a coordinate feature array silently upcast to
1043
+ float64 by a division. The `torch` and `transformers` (and
1044
+ `colabfold`, `meeko`, `vina`) optional heavy dependencies, which
1045
+ ship no type stubs, are added to the mypy `ignore_missing_imports`
1046
+ override alongside the existing `Bio` / `biotite` / `mdtraj` /
1047
+ `openmm` / `rdkit` entries. CI strict gate and the
1048
+ `tests/unit/test_typing.py` regression test updated; only
1049
+ `plugins` and `wrappers` remain outside the gate.
1050
+ - **Six more subpackages are now `mypy --strict` clean.**
1051
+ `molforge.io`, `molforge.sequence`, `molforge.structure`,
1052
+ `molforge.metrics`, `molforge.ensembles`, and
1053
+ `molforge.validation` now pass `mypy --strict` with zero errors,
1054
+ joining `molforge.core` — seven strict-clean subpackages in total,
1055
+ 46 source files. The 12 errors fixed were mostly numpy operations
1056
+ mypy widens to `Any` (resolved with explicit `cast`s that document
1057
+ the known array dtype) and two stale `type: ignore` comments; two
1058
+ were genuine annotation bugs — `_place_hydrogens` in `dssp.py` was
1059
+ declared to return a single array but actually returns a
1060
+ `(coords, mask)` tuple, and `_score` in `alignment.py` was
1061
+ declared `NDArray[np.int_]` but builds an `int32` array (`np.int_`
1062
+ is `int64` on 64-bit platforms). The CI strict gate now covers all
1063
+ seven subpackages; the regression test
1064
+ (`tests/unit/test_typing.py`, moved up from `tests/unit/core/` and
1065
+ parametrized) checks each one in-suite. The remaining subpackages
1066
+ (`ml`, `plugins`, `wrappers`) are still tracked by the
1067
+ non-blocking informational `mypy src` CI step.
1068
+ - **`molforge.core` is now `mypy --strict` clean, and CI enforces
1069
+ it.** The `core` subpackage — the data model the rest of the
1070
+ library is built on — now passes `mypy --strict` with zero
1071
+ errors (fixed: two missing `NDArray` type arguments in
1072
+ `AtomArray`, an `Any`-return in `Atom.coord`, and an untyped
1073
+ `Chain.__iter__` that was suppressed with a `type: ignore`). The
1074
+ CI `typecheck` job now runs `mypy --strict src/molforge/core/`
1075
+ as a hard gate, with a separate non-blocking full-tree `mypy src`
1076
+ step that keeps the remaining (out-of-`core`) type errors visible
1077
+ while they're worked through. A new `slow`-marked regression test
1078
+ (`tests/unit/core/test_typing.py`) runs the strict check in-suite
1079
+ so a `core` type regression is caught locally too.
1080
+
1081
+ ### Documented
1082
+ - **`Simulation.engine_handle` contract clarified.** The attribute
1083
+ type (`object | None`) is correct — it really is an opaque,
1084
+ engine-specific handle — but the contract was under-specified.
1085
+ The docstring now states explicitly that `engine_handle` is
1086
+ engine-private (callers must not inspect it or set it), is **not
1087
+ serialized** (it typically wraps unpicklable C-extension state;
1088
+ persistence layers must drop it and let the engine wrapper
1089
+ rebuild it on resume), and carries **no semver guarantee**. For
1090
+ inspectable per-simulation data, `Simulation.metadata` is the
1091
+ supported field. No code change. Flagged by the API audit.
1092
+
1093
+ ### Deprecated
1094
+ - **`molforge.wrappers.folding.Rosetta` is now a deprecated alias
1095
+ for `RoseTTAFold`.** The original `rosetta.py` placeholder was
1096
+ ambiguous about whether it referred to PyRosetta (the Baker lab's
1097
+ classical sequence-design library) or RoseTTAFold (the deep-
1098
+ learning model). The new real wrapper lives at
1099
+ `RoseTTAFold` for clarity. `Rosetta` is retained as a thin
1100
+ subclass that emits `DeprecationWarning` on construction so
1101
+ existing imports / isinstance checks keep working through the
1102
+ next minor release. A PyRosetta wrapper, if added, would live in
1103
+ a separate module (`pyrosetta.py`) since PyRosetta's surface is
1104
+ much wider than the `FoldingEngine` contract.
1105
+
502
1106
  ### Removed
503
1107
  - **`tests/unit/core/test_core_types.py`.** A pre-existing fossil
504
1108
  from before the view-based data-model refactor: it imported from