biorazer 0.9.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- biorazer-0.9.3/LICENSE +21 -0
- biorazer-0.9.3/PKG-INFO +149 -0
- biorazer-0.9.3/README.md +111 -0
- biorazer-0.9.3/biorazer/__init__.py +17 -0
- biorazer-0.9.3/biorazer/access/cli.py +139 -0
- biorazer-0.9.3/biorazer/access/database/AFDB/files.py +105 -0
- biorazer-0.9.3/biorazer/access/database/AFDB/info.py +47 -0
- biorazer-0.9.3/biorazer/access/database/AFDB/query.py +21 -0
- biorazer-0.9.3/biorazer/access/database/RCSB/__init__.py +0 -0
- biorazer-0.9.3/biorazer/access/database/RCSB/files.py +43 -0
- biorazer-0.9.3/biorazer/access/database/RCSB/info.py +337 -0
- biorazer-0.9.3/biorazer/access/database/RCSB/query.py +27 -0
- biorazer-0.9.3/biorazer/access/database/ensembl/utils.py +82 -0
- biorazer-0.9.3/biorazer/access/database/uniprot/files.py +41 -0
- biorazer-0.9.3/biorazer/access/database/uniprot/load.py +27 -0
- biorazer-0.9.3/biorazer/access/database/uniprot/property.py +21 -0
- biorazer-0.9.3/biorazer/access/database/uniprot/response.py +52 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/api.py +120 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/cli.py +92 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/http.py +41 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/io.py +103 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/paired.py +119 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/pipeline.py +313 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/template.py +454 -0
- biorazer-0.9.3/biorazer/access/server/colabfold_msa/unpaired.py +123 -0
- biorazer-0.9.3/biorazer/cli.py +34 -0
- biorazer-0.9.3/biorazer/database/alphabet/__init__.py +55 -0
- biorazer-0.9.3/biorazer/database/alphabet/aa_list.py +146 -0
- biorazer-0.9.3/biorazer/database/alphabet/aa_types.py +17 -0
- biorazer-0.9.3/biorazer/database/alphabet/protein.py +28 -0
- biorazer-0.9.3/biorazer/database/codon/__init__.py +23 -0
- biorazer-0.9.3/biorazer/database/codon/tables.py +118 -0
- biorazer-0.9.3/biorazer/database/codon/usage.py +44 -0
- biorazer-0.9.3/biorazer/database/molecule/__init__.py +132 -0
- biorazer-0.9.3/biorazer/database/molecule/atom/__init__.py +10 -0
- biorazer-0.9.3/biorazer/database/molecule/atom/charge.py +8 -0
- biorazer-0.9.3/biorazer/database/molecule/atom/radius.py +36 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/__init__.py +19 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/angle/__init__.py +13 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/angle/generic.py +86 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/angle/protein.py +347 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/dihedral/__init__.py +15 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/dihedral/generic.py +6 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/dihedral/protein/__init__.py +57 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/dihedral/protein/by_residue.py +386 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/dihedral/protein/by_ss.py +249 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/length/__init__.py +12 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/length/protein/__init__.py +34 -0
- biorazer-0.9.3/biorazer/database/molecule/bond/length/protein/by_residue.py +238 -0
- biorazer-0.9.3/biorazer/database/molecule/icoor/__init__.py +9 -0
- biorazer-0.9.3/biorazer/database/molecule/icoor/protein/__init__.py +9 -0
- biorazer-0.9.3/biorazer/database/molecule/icoor/protein/template.py +425 -0
- biorazer-0.9.3/biorazer/database/molecule/icoor/protein/topology.py +290 -0
- biorazer-0.9.3/biorazer/database/molecule/rotamer/__init__.py +72 -0
- biorazer-0.9.3/biorazer/database/molecule/rotamer/pymol/__init__.py +26 -0
- biorazer-0.9.3/biorazer/database/molecule/rotamer/pymol/pymol.py +265 -0
- biorazer-0.9.3/biorazer/database/molecule/rotamer/rosetta/__init__.py +32 -0
- biorazer-0.9.3/biorazer/database/molecule/rotamer/rosetta/rosetta.py +287 -0
- biorazer-0.9.3/biorazer/database/molecule/rotamer/rotamer.py +118 -0
- biorazer-0.9.3/biorazer/design/__init__.py +1 -0
- biorazer-0.9.3/biorazer/design/basic/__init__.py +9 -0
- biorazer-0.9.3/biorazer/design/basic/_shared.py +12 -0
- biorazer-0.9.3/biorazer/design/basic/entry.py +241 -0
- biorazer-0.9.3/biorazer/design/basic/library.py +414 -0
- biorazer-0.9.3/biorazer/design/basic/single_test.py +189 -0
- biorazer-0.9.3/biorazer/design/sequence.py +57 -0
- biorazer-0.9.3/biorazer/display.py +45 -0
- biorazer-0.9.3/biorazer/io.py +84 -0
- biorazer-0.9.3/biorazer/logger.py +14 -0
- biorazer-0.9.3/biorazer/sequence/__init__.py +5 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/__init__.py +2 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/calculation/__init__.py +2 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/calculation/array.py +20 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/calculation/scaler.py +21 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/plot/__init__.py +5 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/plot/cli.py +22 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/plot/msa.py +156 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/plot/msa_coverage.py +211 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/plot/seqlogo.py +1027 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/report/__init__.py +1 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/report/conservativity.py +116 -0
- biorazer-0.9.3/biorazer/sequence/analysis/alignment/util.py +106 -0
- biorazer-0.9.3/biorazer/sequence/analysis/single/property.py +69 -0
- biorazer-0.9.3/biorazer/sequence/archive/nucleotide/__init__.py +0 -0
- biorazer-0.9.3/biorazer/sequence/archive/nucleotide/scripts/alignment.py +302 -0
- biorazer-0.9.3/biorazer/sequence/archive/nucleotide/utils/__init__.py +1 -0
- biorazer-0.9.3/biorazer/sequence/archive/nucleotide/utils/nt_utils.py +18 -0
- biorazer-0.9.3/biorazer/sequence/archive/protein/__init__.py +1 -0
- biorazer-0.9.3/biorazer/sequence/archive/protein/scripts/MSA_visualizer.py +117 -0
- biorazer-0.9.3/biorazer/sequence/archive/protein/scripts/alignment_archvie.py +91 -0
- biorazer-0.9.3/biorazer/sequence/archive/protein/scripts/msa_analyzer.py +50 -0
- biorazer-0.9.3/biorazer/sequence/archive/protein/scripts/sequence_comparer.py +213 -0
- biorazer-0.9.3/biorazer/sequence/bridge/central_dogma.py +39 -0
- biorazer-0.9.3/biorazer/sequence/bridge/translation/__init__.py +0 -0
- biorazer-0.9.3/biorazer/sequence/bridge/translation/reverse.py +53 -0
- biorazer-0.9.3/biorazer/sequence/io/__init__.py +13 -0
- biorazer-0.9.3/biorazer/sequence/io/nucleotide.py +0 -0
- biorazer-0.9.3/biorazer/sequence/io/protein.py +87 -0
- biorazer-0.9.3/biorazer/sequence/io/string.py +229 -0
- biorazer-0.9.3/biorazer/sequence/manipulation/annotation.py +0 -0
- biorazer-0.9.3/biorazer/sequence/manipulation/modification.py +107 -0
- biorazer-0.9.3/biorazer/sequence/manipulation/util.py +0 -0
- biorazer-0.9.3/biorazer/structure/__init__.py +3 -0
- biorazer-0.9.3/biorazer/structure/analysis/dynamic/md_traj/README.md +0 -0
- biorazer-0.9.3/biorazer/structure/analysis/dynamic/md_traj/__init__.py +1 -0
- biorazer-0.9.3/biorazer/structure/analysis/dynamic/md_traj/scripts/__init__.py +0 -0
- biorazer-0.9.3/biorazer/structure/analysis/dynamic/md_traj/scripts/view_trajectory.py +29 -0
- biorazer-0.9.3/biorazer/structure/analysis/dynamic/md_traj/scripts/xpm_plot.py +175 -0
- biorazer-0.9.3/biorazer/structure/analysis/dynamic/md_traj/scripts/xvg_plot.py +207 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/__init__.py +2 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/calculation/__init__.py +10 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/calculation/array.py +82 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/calculation/scaler.py +68 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/check.py +91 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/report/__init__.py +19 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/report/cli.py +16 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/report/contact.py +609 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/report/hbond.py +122 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/report/patch.py +468 -0
- biorazer-0.9.3/biorazer/structure/analysis/static/report/util.py +9 -0
- biorazer-0.9.3/biorazer/structure/bridge/__init__.py +30 -0
- biorazer-0.9.3/biorazer/structure/bridge/atom_array.py +327 -0
- biorazer-0.9.3/biorazer/structure/bridge/icchain.py +86 -0
- biorazer-0.9.3/biorazer/structure/bridge/sequence.py +67 -0
- biorazer-0.9.3/biorazer/structure/io/__init__.py +1 -0
- biorazer-0.9.3/biorazer/structure/io/organic.py +259 -0
- biorazer-0.9.3/biorazer/structure/io/protein/__init__.py +243 -0
- biorazer-0.9.3/biorazer/structure/io/protein/_icchain.py +27 -0
- biorazer-0.9.3/biorazer/structure/io/protein/_io.py +29 -0
- biorazer-0.9.3/biorazer/structure/io/protein/_pdb_records.py +174 -0
- biorazer-0.9.3/biorazer/structure/io/protein/_pose.py +76 -0
- biorazer-0.9.3/biorazer/structure/manipulation/atom_array/__init__.py +54 -0
- biorazer-0.9.3/biorazer/structure/manipulation/atom_array/annotation.py +138 -0
- biorazer-0.9.3/biorazer/structure/manipulation/atom_array/modification.py +369 -0
- biorazer-0.9.3/biorazer/structure/manipulation/atom_array/mutation.py +638 -0
- biorazer-0.9.3/biorazer/structure/manipulation/atom_array/util.py +154 -0
- biorazer-0.9.3/biorazer/structure/manipulation/internal_coord/__init__.py +18 -0
- biorazer-0.9.3/biorazer/structure/manipulation/internal_coord/modification.py +195 -0
- biorazer-0.9.3/biorazer/structure/objects/__init__.py +66 -0
- biorazer-0.9.3/biorazer/structure/objects/bp_icchain.py +42 -0
- biorazer-0.9.3/biorazer/structure/objects/bt_atom_array.py +21 -0
- biorazer-0.9.3/biorazer/structure/objects/bt_bond.py +20 -0
- biorazer-0.9.3/biorazer/structure/objects/internal_coords.py +592 -0
- biorazer-0.9.3/biorazer/structure/objects/pr_pose.py +42 -0
- biorazer-0.9.3/biorazer/structure/objects/rd_mol.py +18 -0
- biorazer-0.9.3/biorazer/structure/selection/__init__.py +44 -0
- biorazer-0.9.3/biorazer/structure/selection/index/__init__.py +0 -0
- biorazer-0.9.3/biorazer/structure/selection/index/annotation.py +264 -0
- biorazer-0.9.3/biorazer/structure/selection/index/spatial.py +0 -0
- biorazer-0.9.3/biorazer/structure/selection/mask/__init__.py +0 -0
- biorazer-0.9.3/biorazer/structure/selection/mask/annotation.py +269 -0
- biorazer-0.9.3/biorazer/structure/selection/mask/complex.py +122 -0
- biorazer-0.9.3/biorazer/structure/selection/mask/report.py +80 -0
- biorazer-0.9.3/biorazer/structure/selection/mask/spatial.py +57 -0
- biorazer-0.9.3/biorazer/structure/util/geometry/sphere/__init__.py +1 -0
- biorazer-0.9.3/biorazer/structure/util/geometry/sphere/fibonacci_sphere_grid.pyx +46 -0
- biorazer-0.9.3/biorazer/structure/util/geometry/surface.py +57 -0
- biorazer-0.9.3/biorazer/structure/util/geometry/triangle.py +33 -0
- biorazer-0.9.3/biorazer/structure/util/report.py +38 -0
- biorazer-0.9.3/biorazer/structure/util/selection.py +35 -0
- biorazer-0.9.3/pyproject.toml +76 -0
- biorazer-0.9.3/scripts/build.py +53 -0
biorazer-0.9.3/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Fanlin Wang
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
biorazer-0.9.3/PKG-INFO
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: biorazer
|
|
3
|
+
Version: 0.9.3
|
|
4
|
+
Summary: A platform for analyzing various biological information
|
|
5
|
+
License: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Author: Fanlin Wang
|
|
8
|
+
Author-email: flmaximwang@icloud.com
|
|
9
|
+
Requires-Python: >=3.11,<4.0
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
20
|
+
Provides-Extra: pyrosetta
|
|
21
|
+
Requires-Dist: biopython (>=1.80,<2.0)
|
|
22
|
+
Requires-Dist: biotite (>=1.6.0,<2.0.0)
|
|
23
|
+
Requires-Dist: hydride (>=1.2.3,<2.0.0)
|
|
24
|
+
Requires-Dist: matplotlib (>=3.7.2,<4.0.0)
|
|
25
|
+
Requires-Dist: numpy (>=2,<3)
|
|
26
|
+
Requires-Dist: pandas (>=2.2,<3.0)
|
|
27
|
+
Requires-Dist: python-codon-tables (>=0.1.18,<0.2.0)
|
|
28
|
+
Requires-Dist: rcsb-api (>=1.4.2,<2.0.0)
|
|
29
|
+
Requires-Dist: rdkit (>=2024.3.5)
|
|
30
|
+
Requires-Dist: scipy (>=1.16.1,<2.0.0)
|
|
31
|
+
Requires-Dist: tabulate (>=0.9.0,<0.10.0)
|
|
32
|
+
Requires-Dist: umap-learn (>=0.5.9.post2,<0.6.0)
|
|
33
|
+
Project-URL: Homepage, https://github.com/flmaximwang/BioRazer
|
|
34
|
+
Project-URL: Issues, https://github.com/flmaximwang/BioRazer/issues
|
|
35
|
+
Project-URL: Repository, https://github.com/flmaximwang/BioRazer
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# BioRazer
|
|
39
|
+
|
|
40
|
+
A Python package for analyzing various biological information, built from practical lab experience. Install with pip and you're ready for common bioinformatics analysis work.
|
|
41
|
+
|
|
42
|
+
## Features
|
|
43
|
+
|
|
44
|
+
- **Protein sequences** — Translation, reverse translation, codon/protein dictionaries
|
|
45
|
+
- **Multiple Sequence Alignment (MSA)** — Generate MSA via ColabFold MMseqs2 API, visualize coverage, analyze amino acid frequencies
|
|
46
|
+
- **Structure analysis** — Static analysis (contacts, hydrogen bonds, surface selection), dynamic trajectory analysis (MD trajectory view, XVG/XPM plots)
|
|
47
|
+
- **Database access** — Query AFDB, RCSB PDB, UniProt, Ensembl
|
|
48
|
+
- **Rotamer libraries** — Read external side-chain rotamer databases: PyMOL's bundled Dunbrack pickles (`sc_bb_ind` / `sc_bb_dep`) and Rosetta's Dunbrack 2002 / Shapovalov 2010 text libraries
|
|
49
|
+
- **Protein design** — Sequence design, library generation, single test entries
|
|
50
|
+
|
|
51
|
+
## Installation
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
pip install biorazer
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
### Dependencies
|
|
58
|
+
|
|
59
|
+
- Python >= 3.11
|
|
60
|
+
- biotite, numpy, scipy, matplotlib, hydride, umap-learn, rcsb-api
|
|
61
|
+
- tabulate (for formatted output)
|
|
62
|
+
|
|
63
|
+
### Development
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
# Install with dev and test dependencies
|
|
67
|
+
poetry install --with dev,test
|
|
68
|
+
|
|
69
|
+
# Or using pip with test dependencies
|
|
70
|
+
pip install biorazer
|
|
71
|
+
pip install pytest pytest-cov
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Usage
|
|
75
|
+
|
|
76
|
+
### ColabFold MSA via MMseqs2 API
|
|
77
|
+
|
|
78
|
+
Generate protein MSA by calling the ColabFold public API — zero additional dependencies, pure stdlib:
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
from biorazer.sequence.protein.analysis.alignment.query import run_search
|
|
82
|
+
|
|
83
|
+
# Single-chain MSA (unpaired, default)
|
|
84
|
+
files, _ = run_search(
|
|
85
|
+
["MTSENLYFQGAMG..."],
|
|
86
|
+
out_dir="msa_out/",
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
# Multi-chain paired MSA (for AF3 multimers)
|
|
90
|
+
files, _ = run_search(
|
|
91
|
+
["CHAIN1_SEQUENCE", "CHAIN2_SEQUENCE"],
|
|
92
|
+
out_dir="msa_out/",
|
|
93
|
+
pair_mode="paired",
|
|
94
|
+
pair_strategy="greedy", # or "complete"
|
|
95
|
+
)
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Output: A3M files (`uniref.a3m`, `bfd.mgnify30.*.a3m`, `pair.a3m`) ready for downstream folding pipelines. Supports template search (`--templates`) and custom MMseqs2 server URLs.
|
|
99
|
+
|
|
100
|
+
### MSA Visualization
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from biorazer.sequence.protein.analysis.alignment import plot_msa
|
|
104
|
+
|
|
105
|
+
fig, ax = plot_msa(
|
|
106
|
+
sequences=["MTSENLYFQG", "MTSENLXFQG"],
|
|
107
|
+
labels=["Wild-type", "Mutant"],
|
|
108
|
+
)
|
|
109
|
+
fig.savefig("msa_plot.png")
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Testing
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
pytest tests/ -v
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Tests cover: FASTA parsing, sequence validation, A3M merging, and module constants. All tests are pure (no network required).
|
|
119
|
+
|
|
120
|
+
## Project Structure
|
|
121
|
+
|
|
122
|
+
```
|
|
123
|
+
biorazer/
|
|
124
|
+
├── access/ # External database APIs (AFDB, RCSB, UniProt, Ensembl)
|
|
125
|
+
├── database/ # Reference data & external-library readers
|
|
126
|
+
│ └── molecule/ # per-molecule data
|
|
127
|
+
│ ├── atom/ # vdW radii, charges
|
|
128
|
+
│ ├── bond/ # bond length / angle / dihedral
|
|
129
|
+
│ ├── icoor/ # internal-coordinate topology & templates
|
|
130
|
+
│ └── rotamer/ # external rotamer readers, split by source
|
|
131
|
+
│ ├── rosetta/ # Dunbrack 2002 / Shapovalov 2010 text libraries
|
|
132
|
+
│ └── pymol/ # PyMOL's bundled Dunbrack pickles
|
|
133
|
+
├── design/ # Protein design tools
|
|
134
|
+
├── sequence/ # Sequence analysis
|
|
135
|
+
│ ├── nucleotide/
|
|
136
|
+
│ ├── protein/
|
|
137
|
+
│ ├── analysis/alignment/ # MSA analysis & plotting
|
|
138
|
+
│ │ └── scripts/ # MSA visualizer, analyzer
|
|
139
|
+
│ └── translation/
|
|
140
|
+
├── structure/ # Structure analysis & I/O
|
|
141
|
+
└── util/ # Utility modules (dictionaries)
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
## License
|
|
145
|
+
|
|
146
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
147
|
+
|
|
148
|
+
This project is intended for academic and research use.
|
|
149
|
+
|
biorazer-0.9.3/README.md
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
# BioRazer
|
|
2
|
+
|
|
3
|
+
A Python package for analyzing various biological information, built from practical lab experience. Install with pip and you're ready for common bioinformatics analysis work.
|
|
4
|
+
|
|
5
|
+
## Features
|
|
6
|
+
|
|
7
|
+
- **Protein sequences** — Translation, reverse translation, codon/protein dictionaries
|
|
8
|
+
- **Multiple Sequence Alignment (MSA)** — Generate MSA via ColabFold MMseqs2 API, visualize coverage, analyze amino acid frequencies
|
|
9
|
+
- **Structure analysis** — Static analysis (contacts, hydrogen bonds, surface selection), dynamic trajectory analysis (MD trajectory view, XVG/XPM plots)
|
|
10
|
+
- **Database access** — Query AFDB, RCSB PDB, UniProt, Ensembl
|
|
11
|
+
- **Rotamer libraries** — Read external side-chain rotamer databases: PyMOL's bundled Dunbrack pickles (`sc_bb_ind` / `sc_bb_dep`) and Rosetta's Dunbrack 2002 / Shapovalov 2010 text libraries
|
|
12
|
+
- **Protein design** — Sequence design, library generation, single test entries
|
|
13
|
+
|
|
14
|
+
## Installation
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
pip install biorazer
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
### Dependencies
|
|
21
|
+
|
|
22
|
+
- Python >= 3.11
|
|
23
|
+
- biotite, numpy, scipy, matplotlib, hydride, umap-learn, rcsb-api
|
|
24
|
+
- tabulate (for formatted output)
|
|
25
|
+
|
|
26
|
+
### Development
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
# Install with dev and test dependencies
|
|
30
|
+
poetry install --with dev,test
|
|
31
|
+
|
|
32
|
+
# Or using pip with test dependencies
|
|
33
|
+
pip install biorazer
|
|
34
|
+
pip install pytest pytest-cov
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Usage
|
|
38
|
+
|
|
39
|
+
### ColabFold MSA via MMseqs2 API
|
|
40
|
+
|
|
41
|
+
Generate protein MSA by calling the ColabFold public API — zero additional dependencies, pure stdlib:
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from biorazer.sequence.protein.analysis.alignment.query import run_search
|
|
45
|
+
|
|
46
|
+
# Single-chain MSA (unpaired, default)
|
|
47
|
+
files, _ = run_search(
|
|
48
|
+
["MTSENLYFQGAMG..."],
|
|
49
|
+
out_dir="msa_out/",
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
# Multi-chain paired MSA (for AF3 multimers)
|
|
53
|
+
files, _ = run_search(
|
|
54
|
+
["CHAIN1_SEQUENCE", "CHAIN2_SEQUENCE"],
|
|
55
|
+
out_dir="msa_out/",
|
|
56
|
+
pair_mode="paired",
|
|
57
|
+
pair_strategy="greedy", # or "complete"
|
|
58
|
+
)
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Output: A3M files (`uniref.a3m`, `bfd.mgnify30.*.a3m`, `pair.a3m`) ready for downstream folding pipelines. Supports template search (`--templates`) and custom MMseqs2 server URLs.
|
|
62
|
+
|
|
63
|
+
### MSA Visualization
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
from biorazer.sequence.protein.analysis.alignment import plot_msa
|
|
67
|
+
|
|
68
|
+
fig, ax = plot_msa(
|
|
69
|
+
sequences=["MTSENLYFQG", "MTSENLXFQG"],
|
|
70
|
+
labels=["Wild-type", "Mutant"],
|
|
71
|
+
)
|
|
72
|
+
fig.savefig("msa_plot.png")
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## Testing
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pytest tests/ -v
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Tests cover: FASTA parsing, sequence validation, A3M merging, and module constants. All tests are pure (no network required).
|
|
82
|
+
|
|
83
|
+
## Project Structure
|
|
84
|
+
|
|
85
|
+
```
|
|
86
|
+
biorazer/
|
|
87
|
+
├── access/ # External database APIs (AFDB, RCSB, UniProt, Ensembl)
|
|
88
|
+
├── database/ # Reference data & external-library readers
|
|
89
|
+
│ └── molecule/ # per-molecule data
|
|
90
|
+
│ ├── atom/ # vdW radii, charges
|
|
91
|
+
│ ├── bond/ # bond length / angle / dihedral
|
|
92
|
+
│ ├── icoor/ # internal-coordinate topology & templates
|
|
93
|
+
│ └── rotamer/ # external rotamer readers, split by source
|
|
94
|
+
│ ├── rosetta/ # Dunbrack 2002 / Shapovalov 2010 text libraries
|
|
95
|
+
│ └── pymol/ # PyMOL's bundled Dunbrack pickles
|
|
96
|
+
├── design/ # Protein design tools
|
|
97
|
+
├── sequence/ # Sequence analysis
|
|
98
|
+
│ ├── nucleotide/
|
|
99
|
+
│ ├── protein/
|
|
100
|
+
│ ├── analysis/alignment/ # MSA analysis & plotting
|
|
101
|
+
│ │ └── scripts/ # MSA visualizer, analyzer
|
|
102
|
+
│ └── translation/
|
|
103
|
+
├── structure/ # Structure analysis & I/O
|
|
104
|
+
└── util/ # Utility modules (dictionaries)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## License
|
|
108
|
+
|
|
109
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
110
|
+
|
|
111
|
+
This project is intended for academic and research use.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""BioRazer: a platform for analyzing various biological information.
|
|
2
|
+
|
|
3
|
+
这是一个**常规包** (有本文件), 而不是原来的 PEP 420 命名空间包。差别是实测过的:
|
|
4
|
+
命名空间包会把 ``sys.path`` 上**每一个**同名目录都并进 ``__path__``, 于是第二份
|
|
5
|
+
checkout (worktree、副本、陈旧的 editable 路径) 可以**悄悄只供应一部分模块**;
|
|
6
|
+
而且扫描时一个"带 ``__init__.py``"的同名目录会**整个顶掉**先出现的命名空间
|
|
7
|
+
portion —— 发现器遇到没有 ``__init__.py`` 的目录只记成 portion 继续往后找, 直到
|
|
8
|
+
碰上第一个常规包才停。有了本文件, 这个名字只属于一个目录, ``biorazer.__file__``
|
|
9
|
+
也会指出来是哪一份。
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
13
|
+
|
|
14
|
+
try: # 版本只有一个来源: pyproject.toml (经安装后的 metadata 暴露)
|
|
15
|
+
__version__ = version("biorazer")
|
|
16
|
+
except PackageNotFoundError: # 直接在源码树上 import, 没装过
|
|
17
|
+
__version__ = "0.0.0.dev0"
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""biorazer access fetch 子命令: 从 RCSB / UniProt / AlphaFold DB 下载条目文件。
|
|
2
|
+
|
|
3
|
+
用法: biorazer fetch <id> --source RCSB --fmt pdb -o 3e2c.pdb
|
|
4
|
+
-o 缺省下载到当前目录; 传已存在目录或以 / 结尾视为目录, 否则视为目标文件名。
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import argparse
|
|
8
|
+
import shutil
|
|
9
|
+
import sys
|
|
10
|
+
import tempfile
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import requests
|
|
14
|
+
|
|
15
|
+
from biorazer.access.database.AFDB.files import fetch as afdb_fetch
|
|
16
|
+
from biorazer.access.database.RCSB.files import fetch as rcsb_fetch
|
|
17
|
+
from biorazer.access.database.uniprot.files import fetch as uniprot_fetch
|
|
18
|
+
from biorazer.logger import initialize_logger
|
|
19
|
+
|
|
20
|
+
# 数据来源 -> (fetch 函数, 缺省格式)
|
|
21
|
+
SOURCES = {
|
|
22
|
+
"RCSB": (rcsb_fetch, "pdb"),
|
|
23
|
+
"UNIPROT": (uniprot_fetch, "fasta"),
|
|
24
|
+
"AFDB": (afdb_fetch, "cif"),
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _normalize_source(source: str) -> str:
|
|
29
|
+
"""来源参数大小写不敏感。"""
|
|
30
|
+
return source.upper()
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _resolve_target(identifier: str, fmt: str, output: str | None) -> Path:
|
|
34
|
+
"""-o 语义: 省略 -> ./{id}.{fmt};
|
|
35
|
+
目录(已存在或以 / 结尾) -> 下载到目录内 {id}.{fmt};
|
|
36
|
+
否则视为完整文件路径。"""
|
|
37
|
+
if output is None:
|
|
38
|
+
return Path(f"{identifier}.{fmt}")
|
|
39
|
+
raw = str(output)
|
|
40
|
+
if raw.endswith(("/", "\\")):
|
|
41
|
+
# 以 / 结尾但目录尚不存在: 先创建, 使 _download_to 中 target.is_dir() 判定正确,
|
|
42
|
+
# 否则会被误当成目标文件名
|
|
43
|
+
path = Path(output)
|
|
44
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
return path / f"{identifier}.{fmt}"
|
|
46
|
+
if Path(output).is_dir():
|
|
47
|
+
return Path(output) / f"{identifier}.{fmt}"
|
|
48
|
+
return Path(output)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _download_to(source_fetch, identifier, fmt, target: Path, overwrite: bool, logger) -> Path:
|
|
52
|
+
"""经源 fetch 下载到临时目录, 再移动到目标路径。
|
|
53
|
+
|
|
54
|
+
各源 fetch 均写入 {id}.{fmt} 风格文件名, 临时目录中转使 -o 自定义文件名
|
|
55
|
+
无需改动 access 层接口。
|
|
56
|
+
|
|
57
|
+
target 为最终目标文件路径 (由 _resolve_target 解析), 源 fetch 产物改名为
|
|
58
|
+
target.name 后移动到位。
|
|
59
|
+
"""
|
|
60
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
61
|
+
with tempfile.TemporaryDirectory(prefix="biorazer_fetch_") as tmp:
|
|
62
|
+
# source_fetch writes into the temp dir; its generated name is discarded,
|
|
63
|
+
# we rename to the resolved target name (which may differ, e.g. -o out.pdb)
|
|
64
|
+
produced_path = source_fetch(identifier, fmt=fmt, download_dir=tmp, overwrite=True, logger=logger)
|
|
65
|
+
produced = list(Path(tmp).iterdir())
|
|
66
|
+
if len(produced) != 1:
|
|
67
|
+
raise RuntimeError(
|
|
68
|
+
f"fetch 在临时目录产生了 {len(produced)} 个文件, 预期 1 个: {produced}"
|
|
69
|
+
)
|
|
70
|
+
produced_path = produced[0]
|
|
71
|
+
shutil.move(str(produced_path), target)
|
|
72
|
+
if str(target.parent) == ".":
|
|
73
|
+
logger.info(f"{produced_path.name} moved to current directory")
|
|
74
|
+
else:
|
|
75
|
+
logger.info(f"{produced_path.name} moved to {target.parent}")
|
|
76
|
+
return target
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _run_fetch(args) -> None:
|
|
80
|
+
"""执行 fetch 子命令。"""
|
|
81
|
+
fetch, default_fmt = SOURCES[args.source]
|
|
82
|
+
fmt = args.fmt or default_fmt
|
|
83
|
+
target = _resolve_target(args.id, fmt, args.output)
|
|
84
|
+
logger = initialize_logger(__name__)
|
|
85
|
+
try:
|
|
86
|
+
if target.exists() and not args.overwrite:
|
|
87
|
+
logger.warning(f"{target} already exists, skipping")
|
|
88
|
+
else:
|
|
89
|
+
_download_to(fetch, args.id, fmt, target, args.overwrite, logger)
|
|
90
|
+
except ValueError as e:
|
|
91
|
+
print(f"错误: {e}", file=sys.stderr)
|
|
92
|
+
sys.exit(1)
|
|
93
|
+
except requests.exceptions.RequestException as e:
|
|
94
|
+
print(f"网络错误: {e}", file=sys.stderr)
|
|
95
|
+
sys.exit(1)
|
|
96
|
+
print(target)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _add_fetch_parser(sub) -> argparse.ArgumentParser:
|
|
100
|
+
"""在 argparse subparsers 上注册 fetch 子命令。"""
|
|
101
|
+
p = sub.add_parser(
|
|
102
|
+
"fetch",
|
|
103
|
+
help="从 RCSB / UniProt / AlphaFold DB 下载条目文件",
|
|
104
|
+
description="从 access 支持的各数据库下载条目文件到本地",
|
|
105
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
106
|
+
)
|
|
107
|
+
p.add_argument(
|
|
108
|
+
"id",
|
|
109
|
+
help="数据库条目 ID (如 PDB 代码 3e2c、UniProt 登录号 P0DP23)",
|
|
110
|
+
)
|
|
111
|
+
p.add_argument(
|
|
112
|
+
"-s", "--source",
|
|
113
|
+
required=True,
|
|
114
|
+
type=_normalize_source,
|
|
115
|
+
choices=["RCSB", "UNIPROT", "AFDB"],
|
|
116
|
+
help="数据来源 (缺省格式: RCSB=pdb, UNIPROT=fasta, AFDB=cif)",
|
|
117
|
+
)
|
|
118
|
+
p.add_argument(
|
|
119
|
+
"--fmt",
|
|
120
|
+
default=None,
|
|
121
|
+
help="下载格式, 缺省按来源取默认 (RCSB=pdb / UNIPROT=fasta / AFDB=cif)",
|
|
122
|
+
)
|
|
123
|
+
p.add_argument(
|
|
124
|
+
"-o", "--output",
|
|
125
|
+
default=None,
|
|
126
|
+
help="输出路径: 省略 -> 当前目录 ./{id}.{fmt}; 目录(已存在或以 / 结尾) -> 下载到目录内; 否则视为目标文件名",
|
|
127
|
+
)
|
|
128
|
+
p.add_argument(
|
|
129
|
+
"--overwrite",
|
|
130
|
+
action="store_true",
|
|
131
|
+
help="已存在同名文件时强制重新下载",
|
|
132
|
+
)
|
|
133
|
+
p.set_defaults(func=_run_fetch)
|
|
134
|
+
return p
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def register_subcommand(sub) -> None:
|
|
138
|
+
"""在 argparse subparsers 上注册 access fetch 子命令。"""
|
|
139
|
+
_add_fetch_parser(sub)
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""AlphaFold DB 预测结构文件下载。"""
|
|
2
|
+
import requests
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from ....logger import initialize_logger
|
|
6
|
+
from .info import AFDBEntry
|
|
7
|
+
from .query import uniprot_to_entries
|
|
8
|
+
|
|
9
|
+
# AlphaFold DB 提供的文件类型 (与 AFDBEntry.file_types 一致)
|
|
10
|
+
SUPPORTED_FMT = ["bcif", "cif", "pdb", "pdbImage", "plddtDoc", "paeDoc"]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def fetch(
|
|
14
|
+
uniprot_id: str,
|
|
15
|
+
fmt: str = "cif",
|
|
16
|
+
download_dir: str | Path = ".",
|
|
17
|
+
overwrite=False,
|
|
18
|
+
logger=None,
|
|
19
|
+
):
|
|
20
|
+
"""下载 AlphaFold DB 中该 UniProt 登录号对应的预测结构文件。
|
|
21
|
+
|
|
22
|
+
Parameters
|
|
23
|
+
----------
|
|
24
|
+
uniprot_id : str
|
|
25
|
+
UniProt 登录号, 例如 "P0DP23"。也支持完整 AlphaFold ID 如 "AF-P0DP23-F1-model-v4"。
|
|
26
|
+
fmt : str
|
|
27
|
+
文件类型, 见 SUPPORTED_FMT。
|
|
28
|
+
download_dir : str | Path
|
|
29
|
+
下载目录。
|
|
30
|
+
overwrite : bool
|
|
31
|
+
已存在同名文件时是否覆盖。
|
|
32
|
+
logger : logging.Logger
|
|
33
|
+
日志器, 缺省自动初始化。
|
|
34
|
+
|
|
35
|
+
Returns
|
|
36
|
+
-------
|
|
37
|
+
Path
|
|
38
|
+
下载得到的文件路径。
|
|
39
|
+
"""
|
|
40
|
+
if not logger:
|
|
41
|
+
logger = initialize_logger(__name__)
|
|
42
|
+
if fmt not in SUPPORTED_FMT:
|
|
43
|
+
raise ValueError(
|
|
44
|
+
f"Unsupported format: {fmt} (supported: {', '.join(SUPPORTED_FMT)})"
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
# Parse requested version from full AFDB ID
|
|
48
|
+
requested_version = None
|
|
49
|
+
if uniprot_id.startswith("AF-") and "-model-v" in uniprot_id:
|
|
50
|
+
# Extract version: AF-P0AFB5-F1-model-v4 -> v4 -> 4
|
|
51
|
+
version_part = uniprot_id.split("-model-v")[-1]
|
|
52
|
+
if version_part.isdigit():
|
|
53
|
+
requested_version = int(version_part)
|
|
54
|
+
|
|
55
|
+
entries = uniprot_to_entries(uniprot_id)
|
|
56
|
+
if not isinstance(entries, list) or not entries or not hasattr(entries[0], 'data'):
|
|
57
|
+
raise ValueError(f"No AFDB entry found for {uniprot_id}")
|
|
58
|
+
|
|
59
|
+
# Select entry based on requested version
|
|
60
|
+
entry = entries[0] # Default to first entry (latest version if multiple fragments)
|
|
61
|
+
if requested_version is not None:
|
|
62
|
+
# Check if requested version is available
|
|
63
|
+
all_versions = entry.data.get("allVersions", [])
|
|
64
|
+
if requested_version not in all_versions:
|
|
65
|
+
raise ValueError(
|
|
66
|
+
f"Requested version v{requested_version} not available for {uniprot_id}. "
|
|
67
|
+
f"Available versions: {all_versions}"
|
|
68
|
+
)
|
|
69
|
+
# Check if the file exists - AlphaFold DB only keeps latest version available for download
|
|
70
|
+
url = f"https://alphafold.ebi.ac.uk/files/{entry.id}-model_v{requested_version}.{fmt}"
|
|
71
|
+
r = requests.head(url)
|
|
72
|
+
if r.status_code != 200:
|
|
73
|
+
raise ValueError(
|
|
74
|
+
f"Version v{requested_version} is listed in API but no download available. "
|
|
75
|
+
f"AlphaFold DB only keeps the latest version (v{entry.data['latestVersion']}) "
|
|
76
|
+
f"publicly available."
|
|
77
|
+
)
|
|
78
|
+
else:
|
|
79
|
+
entry.requested_version = None
|
|
80
|
+
|
|
81
|
+
# The entry already has correct URLs from API (for latest version or requested version)
|
|
82
|
+
# The filename from URL already includes version, so we don't need to construct it
|
|
83
|
+
url_key = f"{fmt}Url"
|
|
84
|
+
if requested_version is None:
|
|
85
|
+
# Get filename from the API URL which already has version
|
|
86
|
+
url = entry.data.get(url_key, "")
|
|
87
|
+
if url:
|
|
88
|
+
file_name = url.split("/")[-1]
|
|
89
|
+
else:
|
|
90
|
+
file_name = f"{entry.id}.{fmt}"
|
|
91
|
+
else:
|
|
92
|
+
# We'll construct URL in download(), filename will be correct there
|
|
93
|
+
# entryId is AF-P0AFB5-F1, filename will be AF-P0AFB5-F1-model_vX.fmt
|
|
94
|
+
file_name = f"{entry.id}-model_v{requested_version}.{fmt}"
|
|
95
|
+
|
|
96
|
+
file_path = Path(download_dir) / file_name
|
|
97
|
+
if file_path.exists() and not overwrite:
|
|
98
|
+
logger.warning(f"{file_path} already exists, skipping")
|
|
99
|
+
return file_path
|
|
100
|
+
Path(download_dir).mkdir(parents=True, exist_ok=True)
|
|
101
|
+
# 只传 download 必需参数; requested_version 已在 files.py 中用于 URL/文件名解析,
|
|
102
|
+
# 真实 AFDBEntry.download 内部会自行构造 URL (其第 3 参有默认值)
|
|
103
|
+
entry.download(fmt, folder_dir=str(download_dir))
|
|
104
|
+
logger.info(f"{file_name} downloaded to {download_dir}")
|
|
105
|
+
return file_path
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
import requests
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass
|
|
7
|
+
class AFDBEntry:
|
|
8
|
+
|
|
9
|
+
data: dict = None
|
|
10
|
+
|
|
11
|
+
@property
|
|
12
|
+
def id(self):
|
|
13
|
+
return self.data["entryId"]
|
|
14
|
+
|
|
15
|
+
@property
|
|
16
|
+
def file_types(self):
|
|
17
|
+
return ["bcif", "cif", "pdb", "pdbImage", "plddtDoc", "paeDoc"]
|
|
18
|
+
|
|
19
|
+
@property
|
|
20
|
+
def file_Urls(self):
|
|
21
|
+
result = {}
|
|
22
|
+
for file_type in self.file_types:
|
|
23
|
+
key = f"{file_type}Url"
|
|
24
|
+
result[key] = self.data.get(key, None)
|
|
25
|
+
return result
|
|
26
|
+
|
|
27
|
+
def download(self, file_type: str, folder_dir=".", requested_version=None):
|
|
28
|
+
# If requested specific version, construct URL manually
|
|
29
|
+
if requested_version is not None:
|
|
30
|
+
# AlphaFold filename convention: AF-{uniprot}-F1-model_v{version}.{ext}
|
|
31
|
+
# entryId is already AF-{uniprot}-F1
|
|
32
|
+
# Note: it's model_v{version} with underscore, not hyphen
|
|
33
|
+
url = f"https://alphafold.ebi.ac.uk/files/{self.id}-model_v{requested_version}.{file_type}"
|
|
34
|
+
else:
|
|
35
|
+
url = self.file_Urls[f"{file_type}Url"]
|
|
36
|
+
if url is None:
|
|
37
|
+
raise ValueError(f"File type {file_type} not found for entry {self.id}")
|
|
38
|
+
|
|
39
|
+
r = requests.get(url)
|
|
40
|
+
if r.status_code != 200:
|
|
41
|
+
raise ValueError(f"Failed to download {file_type} from {url} (status code: {r.status_code})")
|
|
42
|
+
|
|
43
|
+
# Get filename from URL always (includes version)
|
|
44
|
+
filename = url.split("/")[-1]
|
|
45
|
+
|
|
46
|
+
with open(Path(folder_dir) / filename, "wb") as f:
|
|
47
|
+
f.write(r.content)
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import requests
|
|
2
|
+
from .info import AFDBEntry
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def uniprot_to_entries(uniprot_id: str):
|
|
6
|
+
# If input is full AFDB ID like AF-P0AFB5-F1-model-v4, extract UniProt ID
|
|
7
|
+
if uniprot_id.startswith("AF-"):
|
|
8
|
+
# Format: AF-{uniprot}-F1-model-v{...}
|
|
9
|
+
parts = uniprot_id.split("-")
|
|
10
|
+
if len(parts) >= 2:
|
|
11
|
+
uniprot_id = parts[1]
|
|
12
|
+
r = requests.get(f"https://alphafold.ebi.ac.uk/api/prediction/{uniprot_id}")
|
|
13
|
+
entries = r.json()
|
|
14
|
+
# Handle API errors - API returns dict with "error" instead of list
|
|
15
|
+
if isinstance(entries, dict) and "error" in entries:
|
|
16
|
+
raise ValueError(f"AlphaFold DB API error: {entries['error']}")
|
|
17
|
+
if not isinstance(entries, list):
|
|
18
|
+
raise ValueError(f"AlphaFold DB API returned unexpected format: {type(entries)}")
|
|
19
|
+
for i in range(len(entries)):
|
|
20
|
+
entries[i] = AFDBEntry(data=entries[i])
|
|
21
|
+
return entries
|
|
File without changes
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
import logging, requests, re
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from xml.etree import ElementTree as ET
|
|
5
|
+
from html import unescape
|
|
6
|
+
from ....logger import initialize_logger
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class PDBStructure:
|
|
11
|
+
pdb_code: str = None
|
|
12
|
+
|
|
13
|
+
def download(
|
|
14
|
+
self,
|
|
15
|
+
fmt="pdb",
|
|
16
|
+
folder_dir=".",
|
|
17
|
+
overwrite=False,
|
|
18
|
+
logger: logging.Logger = None,
|
|
19
|
+
):
|
|
20
|
+
if not logger:
|
|
21
|
+
logger = initialize_logger(__name__)
|
|
22
|
+
logger.info(f"Downloading {self.pdb_code}.{fmt} to {folder_dir}")
|
|
23
|
+
if Path(f"{folder_dir}/{self.pdb_code}.{fmt}").exists() and not overwrite:
|
|
24
|
+
logger.warning(
|
|
25
|
+
f"{self.pdb_code}.{fmt} already exists in {folder_dir}, skipping"
|
|
26
|
+
)
|
|
27
|
+
else:
|
|
28
|
+
r = requests.get(f"https://files.rcsb.org/download/{self.pdb_code}.{fmt}")
|
|
29
|
+
if r.status_code != 200:
|
|
30
|
+
logger.warning(f"\n{r.text}")
|
|
31
|
+
if not Path(folder_dir).exists():
|
|
32
|
+
Path(folder_dir).mkdir(parents=True)
|
|
33
|
+
with open(Path(f"{folder_dir}") / f"{self.pdb_code}.{fmt}", "wb") as f:
|
|
34
|
+
f.write(r.content)
|
|
35
|
+
if logger:
|
|
36
|
+
logger.info(f"{self.pdb_code}.{fmt} downloaded to {folder_dir}")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def fetch(pdb_code: str, fmt="pdb", download_dir=".", overwrite=False, logger=None):
|
|
40
|
+
structure = PDBStructure(pdb_code)
|
|
41
|
+
structure.download(
|
|
42
|
+
fmt=fmt, folder_dir=download_dir, overwrite=overwrite, logger=logger
|
|
43
|
+
)
|