protspace 4.0.2__tar.gz → 4.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {protspace-4.0.2 → protspace-4.2.0}/CHANGELOG.md +61 -0
- {protspace-4.0.2 → protspace-4.2.0}/CLAUDE.md +2 -2
- {protspace-4.0.2 → protspace-4.2.0}/PKG-INFO +4 -5
- {protspace-4.0.2 → protspace-4.2.0}/docs/annotations.md +1 -1
- {protspace-4.0.2 → protspace-4.2.0}/notebooks/ProtSpace_Preparation.ipynb +1 -1
- {protspace-4.0.2 → protspace-4.2.0}/pyproject.toml +4 -6
- protspace-4.2.0/src/protspace/__init__.py +1 -0
- protspace-4.2.0/src/protspace/data/annotations/retrievers/http_utils.py +39 -0
- protspace-4.2.0/src/protspace/data/annotations/retrievers/taxonomy_retriever.py +124 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/uniprot_retriever.py +23 -30
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/parsers/uniprot_parser.py +8 -4
- protspace-4.2.0/tests/test_taxonomy_annotation_retriever.py +446 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_uniprot_annotation_retriever.py +50 -82
- {protspace-4.0.2 → protspace-4.2.0}/uv.lock +5 -32
- protspace-4.0.2/src/protspace/__init__.py +0 -1
- protspace-4.0.2/src/protspace/data/annotations/retrievers/taxonomy_retriever.py +0 -226
- protspace-4.0.2/tests/test_taxonomy_annotation_retriever.py +0 -225
- {protspace-4.0.2 → protspace-4.2.0}/.dockerignore +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/.env.example +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/.githooks/pre-commit +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/.github/workflows/ci.yml +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/.github/workflows/publish.yml +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/.github/workflows/release.yml +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/.gitignore +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/.python-version +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/Dockerfile +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/LICENSE +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/README.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/3FTx.csv +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/parquetbundle/3FTx_accession.csv +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/parquetbundle/3ftx_with_db_styled.parquetbundle +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/parquetbundle/styles.json +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/Pla2g2.csv +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/Pla2g2.fasta +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/Pla2g2_pdb.zip +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/protspace_files/Pla2g2.json +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/protspace_files/Pla2g2_customized.json +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/style.json +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/toxins/protspace/projections_data.parquet +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/toxins/protspace/projections_metadata.parquet +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/toxins/protspace/selected_annotations.parquet +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/toxins/toxins.json +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/data/toxins/toxins.parquetbundle +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/docs/cli.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/docs/protspace_example.png +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/docs/styling.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/notebooks/ClickThrough_GenerateEmbeddings.ipynb +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/count_h5_rows.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/count_parquetbundle_entries.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/download_clan_fastas.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/generate_examples/datasets.toml +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/generate_examples/generate.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/pfam_clans/README.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/pfam_clans/create_pfam_clan_query.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/scripts/pfam_clans/merge_h5_batches.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/app.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/annotated_image.png +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/custom.css +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_faq.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_how_it_works.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_json.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_overview.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/rostlab_logo.png +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/annotate.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/app.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/bundle.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/common_options.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/embed.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/prepare.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/project.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/serve.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/style.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/core/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/core/config.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/core/constants.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/configuration.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/manager.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/merging.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/base_retriever.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/interpro_retriever.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/scores.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/interpro_transforms.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/transformer.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/uniprot_transforms.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/embedding/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/embedding/biocentral.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/bundle.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/fasta.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/formatters.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/settings_converter.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/writers.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/embedding_set.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/fasta.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/h5.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/query.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/similarity.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/parsers/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/processors/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/processors/base_processor.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/processors/pipeline.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/main.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/callbacks.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/layout.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/styles.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/add_annotation_style.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/arrow_reader.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/constants.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/reducers.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/visualization/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/visualization/molstar.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/visualization/plotting.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/src/protspace/wsgi.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/README.md +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/__init__.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_annotation_manager.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_base_data_processor.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_biocentral_embedder.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_bundle_settings.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_config.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_config_validation.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_fasta.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_formatters.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_h5_parse_identifier.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_interpro_annotation_retriever.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_output_combinations.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_pipeline_utils.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_reducers.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_settings_converter.py +0 -0
- {protspace-4.0.2 → protspace-4.2.0}/tests/test_transformer.py +0 -0
|
@@ -1,6 +1,67 @@
|
|
|
1
1
|
# CHANGELOG
|
|
2
2
|
|
|
3
3
|
|
|
4
|
+
## v4.2.0 (2026-03-28)
|
|
5
|
+
|
|
6
|
+
### Features
|
|
7
|
+
|
|
8
|
+
* feat: replace unipressed with direct UniProt REST API calls
|
|
9
|
+
|
|
10
|
+
Replace the unipressed library (community UniProt API wrapper) with
|
|
11
|
+
direct HTTP calls to rest.uniprot.org. Adds _fetch_many_accessions()
|
|
12
|
+
and _search_sec_acc() helpers using the same Link-header pagination
|
|
13
|
+
pattern as the taxonomy retriever.
|
|
14
|
+
|
|
15
|
+
Simplifies the sec_acc search fallback from 8 lines of page-parsing
|
|
16
|
+
to a single function call.
|
|
17
|
+
|
|
18
|
+
Closes #32
|
|
19
|
+
|
|
20
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`5605a93`](https://github.com/tsenoner/protspace/commit/5605a935e680132ee7d22adb30a45b3ff26df02b))
|
|
21
|
+
|
|
22
|
+
### Refactoring
|
|
23
|
+
|
|
24
|
+
* refactor: extract shared paginated_get() utility for REST API calls
|
|
25
|
+
|
|
26
|
+
Consolidate the duplicated Link-header pagination loop (4 instances
|
|
27
|
+
across uniprot_retriever, taxonomy_retriever, and uniprot_parser) into
|
|
28
|
+
a single paginated_get() function in http_utils.py. Each caller is
|
|
29
|
+
now a 1-3 line function.
|
|
30
|
+
|
|
31
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`cd528ed`](https://github.com/tsenoner/protspace/commit/cd528ed68ba7cf9bf3a33183345b483c00b4f9d1))
|
|
32
|
+
|
|
33
|
+
### Unknown
|
|
34
|
+
|
|
35
|
+
* Merge pull request #39 from tsenoner/feat/replace-unipressed-with-direct-api
|
|
36
|
+
|
|
37
|
+
Replace unipressed with direct UniProt REST API calls ([`5d365c7`](https://github.com/tsenoner/protspace/commit/5d365c70dfb66fe1754fb610c7df3acb3cfae480))
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
## v4.1.0 (2026-03-28)
|
|
41
|
+
|
|
42
|
+
### Features
|
|
43
|
+
|
|
44
|
+
* feat: replace taxopy with UniProt Taxonomy API for taxonomy lookups
|
|
45
|
+
|
|
46
|
+
Replace the taxopy-based taxonomy retriever (which required downloading
|
|
47
|
+
the full NCBI taxonomy database ~50 MB on first use) with the UniProt
|
|
48
|
+
Taxonomy API (/taxonomy/search). This eliminates the slow first-run
|
|
49
|
+
download, weekly cache refresh, and ~120 lines of cache management code.
|
|
50
|
+
|
|
51
|
+
Also fix typer[all] → typer (the [all] extra was removed) and add
|
|
52
|
+
requests as an explicit core dependency.
|
|
53
|
+
|
|
54
|
+
Closes #36
|
|
55
|
+
|
|
56
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`754fa9b`](https://github.com/tsenoner/protspace/commit/754fa9b45cf5a913a97823b2b45012c49221ca7a))
|
|
57
|
+
|
|
58
|
+
### Unknown
|
|
59
|
+
|
|
60
|
+
* Merge pull request #37 from tsenoner/feat/replace-taxopy-with-uniprot-api
|
|
61
|
+
|
|
62
|
+
Replace taxopy with UniProt Taxonomy API ([`034c0ec`](https://github.com/tsenoner/protspace/commit/034c0ece32238144fe68819ce2b6e5e43d8b57a5))
|
|
63
|
+
|
|
64
|
+
|
|
4
65
|
## v4.0.2 (2026-03-28)
|
|
5
66
|
|
|
6
67
|
### Fixes
|
|
@@ -205,7 +205,7 @@ uv run pytest tests/ --cov=src/protspace # With coverage
|
|
|
205
205
|
| `test_biocentral_embedder.py` | 23 | Biocentral API client, embedding flow |
|
|
206
206
|
| `test_fasta.py` | 17 | FASTA parsing, edge cases, CSV annotation loading |
|
|
207
207
|
| `test_config_validation.py` | 12 | DimensionReductionConfig parameter validation |
|
|
208
|
-
| `test_taxonomy_annotation_retriever.py` |
|
|
208
|
+
| `test_taxonomy_annotation_retriever.py` | 15 | Taxonomy via UniProt Taxonomy API (mocked + integration) |
|
|
209
209
|
| `test_h5_parse_identifier.py` | 9 | HDF5 key parsing, identifier extraction |
|
|
210
210
|
| `test_base_data_processor.py` | 8 | BaseProcessor: reduction, output creation, save |
|
|
211
211
|
| `test_formatters.py` | 5 | ProteinAnnotations → DataFrame formatting |
|
|
@@ -225,7 +225,7 @@ Located in `notebooks/`:
|
|
|
225
225
|
|
|
226
226
|
## Dependencies
|
|
227
227
|
|
|
228
|
-
**Core:** h5py, scikit-learn, umap-learn, pacmap (includes annoy), numpy, pandas, pyarrow, tqdm,
|
|
228
|
+
**Core:** h5py, scikit-learn, umap-learn, pacmap (includes annoy), numpy, pandas, pyarrow, tqdm, requests, pymmseqs, biocentral-api, typer, rich
|
|
229
229
|
|
|
230
230
|
**Frontend (optional):** dash, plotly, dash-bootstrap-components, dash-molstar, gunicorn
|
|
231
231
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: protspace
|
|
3
|
-
Version: 4.0
|
|
3
|
+
Version: 4.2.0
|
|
4
4
|
Summary: A visualisation tool for protein embeddings from pLMs
|
|
5
5
|
Author-email: Tobias Senoner <tobias.senoner@tum.de>
|
|
6
6
|
License-Expression: GPL-3.0
|
|
@@ -13,13 +13,12 @@ Requires-Dist: pacmap>=0.8.0
|
|
|
13
13
|
Requires-Dist: pandas>=2.0.0
|
|
14
14
|
Requires-Dist: pyarrow>=20.0.0
|
|
15
15
|
Requires-Dist: pymmseqs>=1.0.4
|
|
16
|
-
Requires-Dist:
|
|
16
|
+
Requires-Dist: requests>=2.32.4
|
|
17
|
+
Requires-Dist: rich>=14.3.3
|
|
17
18
|
Requires-Dist: scikit-learn>=1.6.1
|
|
18
|
-
Requires-Dist: taxopy>=0.14.0
|
|
19
19
|
Requires-Dist: tqdm>=4.67.1
|
|
20
|
-
Requires-Dist: typer
|
|
20
|
+
Requires-Dist: typer>=0.24.1
|
|
21
21
|
Requires-Dist: umap-learn>=0.5.10
|
|
22
|
-
Requires-Dist: unipressed>=1.4.0
|
|
23
22
|
Provides-Extra: frontend
|
|
24
23
|
Requires-Dist: dash-bootstrap-components>=1.6.0; extra == 'frontend'
|
|
25
24
|
Requires-Dist: dash-daq>=0.5.0; extra == 'frontend'
|
|
@@ -46,7 +46,7 @@ With `--keep-tmp`, only API-fetched annotations are cached; the CSV is always re
|
|
|
46
46
|
|
|
47
47
|
## UniProt Annotations
|
|
48
48
|
|
|
49
|
-
14 annotations retrieved from the [UniProt REST API](https://rest.uniprot.org/)
|
|
49
|
+
14 annotations retrieved from the [UniProt REST API](https://rest.uniprot.org/) (batch size: 100):
|
|
50
50
|
|
|
51
51
|
| Name | Description | Example |
|
|
52
52
|
| ------------------------- | ------------------------------------ | -------------------------------------------------------------- |
|
|
@@ -45,7 +45,7 @@
|
|
|
45
45
|
"cellView": "form"
|
|
46
46
|
},
|
|
47
47
|
"outputs": [],
|
|
48
|
-
"source": "# @title 3. Generate & Download {display-mode: \"form\"}\n# @markdown Configure methods and annotations, then click Generate.\n# @markdown Processing time depends on dataset size and selected methods.\n\nfrom ipywidgets import (\n HTML,\n Button,\n Checkbox,\n Dropdown,\n FileUpload,\n FloatSlider,\n GridBox,\n HBox,\n IntSlider,\n Output,\n ToggleButton,\n VBox,\n)\n\nMETHODS = [\"PCA\", \"UMAP\", \"t-SNE\", \"MDS\", \"PaCMAP\", \"LocalMAP\"]\nANNOTATIONS = {\n \"UniProt\": [\"annotation_score\", \"cc_subcellular_location\", \"ec\", \"fragment\", \"go_bp\", \"go_cc\", \"go_mf\", \"keyword\", \"length\", \"protein_existence\", \"protein_families\", \"reviewed\", \"xref_pdb\"],\n \"InterPro\": [\"cath\", \"cdd\", \"panther\", \"pfam\", \"prints\", \"prosite\", \"signal_peptide\", \"smart\", \"superfamily\"],\n \"Taxonomy\": [\"root\", \"domain\", \"kingdom\", \"phylum\", \"class\", \"order\", \"family\", \"genus\", \"species\"],\n}\nANNOTATION_DEFAULTS = {\"ec\", \"keyword\", \"length\", \"protein_families\", \"reviewed\"}\n\n# --- Methods ---\nmethod_toggles = {}\nfor m in METHODS:\n sel = m in {\"PCA\", \"UMAP\"}\n method_toggles[m] = ToggleButton(value=sel, description=m, button_style=\"info\" if sel else \"\", layout={\"width\": \"auto\", \"height\": \"32px\"})\n\n# --- DR Parameters ---\n_s = {\"description_width\": \"110px\"}\n_l = {\"width\": \"100%\"}\n_g = {\"border\": \"1px solid #666\", \"padding\": \"6px 10px\", \"margin\": \"2px\", \"flex\": \"1 1 220px\", \"overflow\": \"hidden\"}\n\nparam_groups = []\nw_nn = IntSlider(value=25, min=2, max=500, description=\"n_neighbors:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP / PaCMAP / LocalMAP</b>\"), w_nn], layout=_g), {\"UMAP\", \"PaCMAP\", \"LocalMAP\"}))\nw_md = FloatSlider(value=0.4, min=0.0, max=0.99, step=0.01, description=\"min_dist:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP</b>\"), w_md], layout=_g), {\"UMAP\"}))\nw_mn = FloatSlider(value=0.5, min=0.0, max=1.0, step=0.1, description=\"mn_ratio:\", style=_s, layout=_l)\nw_fp = FloatSlider(value=2.0, min=0.0, max=5.0, step=0.1, description=\"fp_ratio:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>PaCMAP / LocalMAP</b>\"), w_mn, w_fp], layout=_g), {\"PaCMAP\", \"LocalMAP\"}))\nw_pp = FloatSlider(value=30.0, min=5.0, max=500.0, step=1.0, description=\"perplexity:\", style=_s, layout=_l)\nw_lr = FloatSlider(value=200.0, min=1.0, max=1000.0, step=10.0, description=\"learning_rate:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>t-SNE</b>\"), w_pp, w_lr], layout=_g), {\"t-SNE\"}))\n\nno_params = HTML(\"<i>PCA/MDS have no adjustable parameters.</i>\")\npw = {\"n_neighbors\": w_nn, \"min_dist\": w_md, \"perplexity\": w_pp, \"learning_rate\": w_lr, \"mn_ratio\": w_mn, \"fp_ratio\": w_fp}\nparam_grid = HBox([g for g, _ in param_groups], layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\", \"width\": \"100%\"})\n\n\ndef _update_params(_c=None):\n sel = {m for m, t in method_toggles.items() if t.value}\n any_vis = False\n for gw, req in param_groups:\n vis = bool(req & sel)\n gw.layout.display = \"\" if vis else \"none\"\n if vis:\n any_vis = True\n no_params.layout.display = \"\" if not any_vis else \"none\"\n\n\nfor tb in method_toggles.values():\n def _toggle(c):\n c[\"owner\"].button_style = \"info\" if c[\"new\"] else \"\"\n _update_params()\n tb.observe(_toggle, names=\"value\")\n_update_params()\n\n# --- Annotations ---\nann_preset = Dropdown(\n options=[\"Default (recommended)\", \"All annotations\", \"Custom...\"],\n value=\"Default (recommended)\",\n description=\"Annotations:\",\n style={\"description_width\": \"initial\"},\n)\nann_cbs = {}\nann_cols = []\nfor cat, names in ANNOTATIONS.items():\n cbs = {n: Checkbox(value=n in ANNOTATION_DEFAULTS, description=n, indent=False, layout={\"width\": \"auto\"}) for n in names}\n ann_cbs[cat] = cbs\n grid = GridBox(list(cbs.values()), layout={\"grid_template_columns\": \"repeat(2, auto)\", \"grid_gap\": \"0 6px\"})\n ann_cols.append(VBox([HTML(f\"<b>{cat}</b>\"), grid], layout={\"flex\": \"1 1 200px\", \"border\": \"1px solid #666\", \"padding\": \"6px\", \"margin\": \"2px\"}))\ncustom_ann = HBox(ann_cols, layout={\"flex_flow\": \"row wrap\"})\ncustom_ann.layout.display = \"none\"\nann_preset.observe(lambda c: setattr(custom_ann.layout, \"display\", \"\" if c[\"new\"] == \"Custom...\" else \"none\"), names=\"value\")\n\n# --- Custom CSV upload ---\ncsv_upload = FileUpload(accept=\".csv,.tsv\", multiple=False, description=\"Choose CSV/TSV\")\n\n\ndef _get_ann():\n p = ann_preset.value\n if p == \"Default (recommended)\":\n return [\"default\"]\n if p == \"All annotations\":\n return [\"all\"]\n sel = [n for cbs in ann_cbs.values() for n, cb in cbs.items() if cb.value]\n return sel if sel else [\"default\"]\n\n\ndef _save_csv():\n \"\"\"Save uploaded CSV to disk and return the path, or None.\"\"\"\n if not csv_upload.value:\n return None\n fname = list(csv_upload.value.keys())[0]\n content = csv_upload.value[fname][\"content\"]\n Path(fname).write_bytes(content)\n return fname\n\n\n# --- Generate ---\ngen_btn = Button(description=\"Generate\", button_style=\"primary\", icon=\"play\", layout={\"width\": \"180px\", \"height\": \"40px\"})\ngen_out = Output()\n\n\ndef _on_gen(_b):\n with gen_out:\n clear_output()\n if not input_args:\n print(\"Select input data in cell above first.\")\n return\n method_specs = [m.lower().replace(\"-\", \"\") + \"2\" for m, t in method_toggles.items() if t.value]\n if not method_specs:\n print(\"Select at least one method.\")\n return\n\n ann = _get_ann() or []\n csv_path = _save_csv()\n if csv_path:\n ann.append(csv_path)\n if not ann:\n ann = None\n\n inp = input_args\n\n out_dir = Path(\"output\")\n out_dir.mkdir(exist_ok=True)\n cache_dir = out_dir / \"tmp\"\n cache_dir.mkdir(exist_ok=True)\n output_path = out_dir / \"data.parquetbundle\"\n\n step_html = HTML(value=\"<b>Step 1/4: Loading embeddings...</b>\")\n display(step_html)\n\n try:\n t0 = _time.time()\n embedding_sets = []\n\n if inp[\"type\"] == \"query\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n step_html.value = \"<b>Step 1/6: Downloading FASTA...</b>\"\n headers, fasta_path = query_uniprot(inp[\"query\"])\n if not headers:\n print(f\"No sequences found for query: {inp['query']}\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 2/6: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(fasta_path, emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = fasta_path\n embedding_sets.append(emb_set)\n elif inp[\"type\"] == \"fasta\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 1/5: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(Path(inp[\"path\"]), emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = Path(inp[\"path\"])\n embedding_sets.append(emb_set)\n else:\n h5_path = Path(inp[\"path\"])\n name_override = inp.get(\"name\")\n emb_set = load_h5([h5_path], name_override=name_override)\n embedding_sets.append(emb_set)\n\n n_proteins = len(embedding_sets[0].headers)\n\n # Build pipeline with caching enabled\n reducer_params = ReducerParams(\n n_neighbors=pw[\"n_neighbors\"].value,\n min_dist=pw[\"min_dist\"].value,\n perplexity=pw[\"perplexity\"].value,\n learning_rate=pw[\"learning_rate\"].value,\n mn_ratio=pw[\"mn_ratio\"].value,\n fp_ratio=pw[\"fp_ratio\"].value,\n )\n config = PipelineConfig(\n methods=method_specs,\n output_path=output_path,\n bundled=True,\n keep_tmp=True,\n intermediate_dir=cache_dir,\n annotations=ann,\n reducer_params=reducer_params,\n )\n pipeline = ReductionPipeline(config)\n\n # Step 2: Annotations (cached after first run)\n step_html.value = \"<b>Step 2/4: Fetching annotations...</b>\"\n metadata = pipeline._fetch_annotations(embedding_sets[0].headers)\n\n # Step 3: Dimensionality reduction\n step_html.value = \"<b>Step 3/4: Reducing dimensions...</b>\"\n\n # Build full metadata\n all_headers = embedding_sets[0].headers\n full_metadata = _pd.DataFrame({\"identifier\": all_headers})\n if len(metadata.columns) > 1:\n metadata = metadata.astype(str)\n id_col = metadata.columns[0]\n if id_col != \"identifier\":\n metadata = metadata.rename(columns={id_col: \"identifier\"})\n full_metadata = full_metadata.merge(\n metadata.drop_duplicates(\"identifier\"),\n on=\"identifier\",\n how=\"left\",\n )\n metadata = full_metadata\n\n all_reductions = pipeline._run_reductions(embedding_sets)\n\n # Step 4: Bundle & download\n step_html.value = \"<b>Step 4/4: Bundling & downloading...</b>\"\n output = pipeline.base.create_output(metadata, all_reductions, all_headers)\n pipeline.base.save_output(output, output_path, bundled=True)\n\n kb = output_path.stat().st_size / 1024\n total_time = _time.time() - t0\n step_html.value = f\"<b>Done! ({kb:.0f} KB, {total_time:.0f}s)</b>\"\n print(f\"Processed {n_proteins} proteins with {len(method_specs)} method(s)\")\n print(f\"\\nUpload at: https://protspace.app/explore\")\n files.download(str(output_path))\n\n except Exception as e:\n step_html.value = \"<b>Failed</b>\"\n print(f\"\\nError: {e}\")\n print(\"\\nTroubleshooting:\")\n print(\"- Check that protein IDs are UniProt accessions (e.g., P12345)\")\n print(\"- Try setting Annotations to 'None' to skip annotation fetching\")\n\n\ngen_btn.on_click(_on_gen)\n\ndisplay(VBox([\n HTML(\"<h4>Methods</h4>\"),\n HBox(list(method_toggles.values()), layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\"}),\n no_params,\n param_grid,\n HTML(\"<h4>Annotations</h4>\"),\n HBox([ann_preset, csv_upload], layout={\"gap\": \"10px\", \"align_items\": \"center\"}),\n HTML(\"<p><i>Optional: upload a CSV/TSV with custom annotations (first column = protein IDs). \"\n \"Taxonomy requires a one-time database download (~1 min).</i></p>\"),\n custom_ann,\n gen_btn,\n HTML(\"<p><i>Processing time depends on dataset size and selected methods.</i></p>\"),\n gen_out,\n]))"
|
|
48
|
+
"source": "# @title 3. Generate & Download {display-mode: \"form\"}\n# @markdown Configure methods and annotations, then click Generate.\n# @markdown Processing time depends on dataset size and selected methods.\n\nfrom ipywidgets import (\n HTML,\n Button,\n Checkbox,\n Dropdown,\n FileUpload,\n FloatSlider,\n GridBox,\n HBox,\n IntSlider,\n Output,\n ToggleButton,\n VBox,\n)\n\nMETHODS = [\"PCA\", \"UMAP\", \"t-SNE\", \"MDS\", \"PaCMAP\", \"LocalMAP\"]\nANNOTATIONS = {\n \"UniProt\": [\"annotation_score\", \"cc_subcellular_location\", \"ec\", \"fragment\", \"go_bp\", \"go_cc\", \"go_mf\", \"keyword\", \"length\", \"protein_existence\", \"protein_families\", \"reviewed\", \"xref_pdb\"],\n \"InterPro\": [\"cath\", \"cdd\", \"panther\", \"pfam\", \"prints\", \"prosite\", \"signal_peptide\", \"smart\", \"superfamily\"],\n \"Taxonomy\": [\"root\", \"domain\", \"kingdom\", \"phylum\", \"class\", \"order\", \"family\", \"genus\", \"species\"],\n}\nANNOTATION_DEFAULTS = {\"ec\", \"keyword\", \"length\", \"protein_families\", \"reviewed\"}\n\n# --- Methods ---\nmethod_toggles = {}\nfor m in METHODS:\n sel = m in {\"PCA\", \"UMAP\"}\n method_toggles[m] = ToggleButton(value=sel, description=m, button_style=\"info\" if sel else \"\", layout={\"width\": \"auto\", \"height\": \"32px\"})\n\n# --- DR Parameters ---\n_s = {\"description_width\": \"110px\"}\n_l = {\"width\": \"100%\"}\n_g = {\"border\": \"1px solid #666\", \"padding\": \"6px 10px\", \"margin\": \"2px\", \"flex\": \"1 1 220px\", \"overflow\": \"hidden\"}\n\nparam_groups = []\nw_nn = IntSlider(value=25, min=2, max=500, description=\"n_neighbors:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP / PaCMAP / LocalMAP</b>\"), w_nn], layout=_g), {\"UMAP\", \"PaCMAP\", \"LocalMAP\"}))\nw_md = FloatSlider(value=0.4, min=0.0, max=0.99, step=0.01, description=\"min_dist:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP</b>\"), w_md], layout=_g), {\"UMAP\"}))\nw_mn = FloatSlider(value=0.5, min=0.0, max=1.0, step=0.1, description=\"mn_ratio:\", style=_s, layout=_l)\nw_fp = FloatSlider(value=2.0, min=0.0, max=5.0, step=0.1, description=\"fp_ratio:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>PaCMAP / LocalMAP</b>\"), w_mn, w_fp], layout=_g), {\"PaCMAP\", \"LocalMAP\"}))\nw_pp = FloatSlider(value=30.0, min=5.0, max=500.0, step=1.0, description=\"perplexity:\", style=_s, layout=_l)\nw_lr = FloatSlider(value=200.0, min=1.0, max=1000.0, step=10.0, description=\"learning_rate:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>t-SNE</b>\"), w_pp, w_lr], layout=_g), {\"t-SNE\"}))\n\nno_params = HTML(\"<i>PCA/MDS have no adjustable parameters.</i>\")\npw = {\"n_neighbors\": w_nn, \"min_dist\": w_md, \"perplexity\": w_pp, \"learning_rate\": w_lr, \"mn_ratio\": w_mn, \"fp_ratio\": w_fp}\nparam_grid = HBox([g for g, _ in param_groups], layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\", \"width\": \"100%\"})\n\n\ndef _update_params(_c=None):\n sel = {m for m, t in method_toggles.items() if t.value}\n any_vis = False\n for gw, req in param_groups:\n vis = bool(req & sel)\n gw.layout.display = \"\" if vis else \"none\"\n if vis:\n any_vis = True\n no_params.layout.display = \"\" if not any_vis else \"none\"\n\n\nfor tb in method_toggles.values():\n def _toggle(c):\n c[\"owner\"].button_style = \"info\" if c[\"new\"] else \"\"\n _update_params()\n tb.observe(_toggle, names=\"value\")\n_update_params()\n\n# --- Annotations ---\nann_preset = Dropdown(\n options=[\"Default (recommended)\", \"All annotations\", \"Custom...\"],\n value=\"Default (recommended)\",\n description=\"Annotations:\",\n style={\"description_width\": \"initial\"},\n)\nann_cbs = {}\nann_cols = []\nfor cat, names in ANNOTATIONS.items():\n cbs = {n: Checkbox(value=n in ANNOTATION_DEFAULTS, description=n, indent=False, layout={\"width\": \"auto\"}) for n in names}\n ann_cbs[cat] = cbs\n grid = GridBox(list(cbs.values()), layout={\"grid_template_columns\": \"repeat(2, auto)\", \"grid_gap\": \"0 6px\"})\n ann_cols.append(VBox([HTML(f\"<b>{cat}</b>\"), grid], layout={\"flex\": \"1 1 200px\", \"border\": \"1px solid #666\", \"padding\": \"6px\", \"margin\": \"2px\"}))\ncustom_ann = HBox(ann_cols, layout={\"flex_flow\": \"row wrap\"})\ncustom_ann.layout.display = \"none\"\nann_preset.observe(lambda c: setattr(custom_ann.layout, \"display\", \"\" if c[\"new\"] == \"Custom...\" else \"none\"), names=\"value\")\n\n# --- Custom CSV upload ---\ncsv_upload = FileUpload(accept=\".csv,.tsv\", multiple=False, description=\"Choose CSV/TSV\")\n\n\ndef _get_ann():\n p = ann_preset.value\n if p == \"Default (recommended)\":\n return [\"default\"]\n if p == \"All annotations\":\n return [\"all\"]\n sel = [n for cbs in ann_cbs.values() for n, cb in cbs.items() if cb.value]\n return sel if sel else [\"default\"]\n\n\ndef _save_csv():\n \"\"\"Save uploaded CSV to disk and return the path, or None.\"\"\"\n if not csv_upload.value:\n return None\n fname = list(csv_upload.value.keys())[0]\n content = csv_upload.value[fname][\"content\"]\n Path(fname).write_bytes(content)\n return fname\n\n\n# --- Generate ---\ngen_btn = Button(description=\"Generate\", button_style=\"primary\", icon=\"play\", layout={\"width\": \"180px\", \"height\": \"40px\"})\ngen_out = Output()\n\n\ndef _on_gen(_b):\n with gen_out:\n clear_output()\n if not input_args:\n print(\"Select input data in cell above first.\")\n return\n method_specs = [m.lower().replace(\"-\", \"\") + \"2\" for m, t in method_toggles.items() if t.value]\n if not method_specs:\n print(\"Select at least one method.\")\n return\n\n ann = _get_ann() or []\n csv_path = _save_csv()\n if csv_path:\n ann.append(csv_path)\n if not ann:\n ann = None\n\n inp = input_args\n\n out_dir = Path(\"output\")\n out_dir.mkdir(exist_ok=True)\n cache_dir = out_dir / \"tmp\"\n cache_dir.mkdir(exist_ok=True)\n output_path = out_dir / \"data.parquetbundle\"\n\n step_html = HTML(value=\"<b>Step 1/4: Loading embeddings...</b>\")\n display(step_html)\n\n try:\n t0 = _time.time()\n embedding_sets = []\n\n if inp[\"type\"] == \"query\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n step_html.value = \"<b>Step 1/6: Downloading FASTA...</b>\"\n headers, fasta_path = query_uniprot(inp[\"query\"])\n if not headers:\n print(f\"No sequences found for query: {inp['query']}\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 2/6: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(fasta_path, emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = fasta_path\n embedding_sets.append(emb_set)\n elif inp[\"type\"] == \"fasta\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 1/5: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(Path(inp[\"path\"]), emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = Path(inp[\"path\"])\n embedding_sets.append(emb_set)\n else:\n h5_path = Path(inp[\"path\"])\n name_override = inp.get(\"name\")\n emb_set = load_h5([h5_path], name_override=name_override)\n embedding_sets.append(emb_set)\n\n n_proteins = len(embedding_sets[0].headers)\n\n # Build pipeline with caching enabled\n reducer_params = ReducerParams(\n n_neighbors=pw[\"n_neighbors\"].value,\n min_dist=pw[\"min_dist\"].value,\n perplexity=pw[\"perplexity\"].value,\n learning_rate=pw[\"learning_rate\"].value,\n mn_ratio=pw[\"mn_ratio\"].value,\n fp_ratio=pw[\"fp_ratio\"].value,\n )\n config = PipelineConfig(\n methods=method_specs,\n output_path=output_path,\n bundled=True,\n keep_tmp=True,\n intermediate_dir=cache_dir,\n annotations=ann,\n reducer_params=reducer_params,\n )\n pipeline = ReductionPipeline(config)\n\n # Step 2: Annotations (cached after first run)\n step_html.value = \"<b>Step 2/4: Fetching annotations...</b>\"\n metadata = pipeline._fetch_annotations(embedding_sets[0].headers)\n\n # Step 3: Dimensionality reduction\n step_html.value = \"<b>Step 3/4: Reducing dimensions...</b>\"\n\n # Build full metadata\n all_headers = embedding_sets[0].headers\n full_metadata = _pd.DataFrame({\"identifier\": all_headers})\n if len(metadata.columns) > 1:\n metadata = metadata.astype(str)\n id_col = metadata.columns[0]\n if id_col != \"identifier\":\n metadata = metadata.rename(columns={id_col: \"identifier\"})\n full_metadata = full_metadata.merge(\n metadata.drop_duplicates(\"identifier\"),\n on=\"identifier\",\n how=\"left\",\n )\n metadata = full_metadata\n\n all_reductions = pipeline._run_reductions(embedding_sets)\n\n # Step 4: Bundle & download\n step_html.value = \"<b>Step 4/4: Bundling & downloading...</b>\"\n output = pipeline.base.create_output(metadata, all_reductions, all_headers)\n pipeline.base.save_output(output, output_path, bundled=True)\n\n kb = output_path.stat().st_size / 1024\n total_time = _time.time() - t0\n step_html.value = f\"<b>Done! ({kb:.0f} KB, {total_time:.0f}s)</b>\"\n print(f\"Processed {n_proteins} proteins with {len(method_specs)} method(s)\")\n print(f\"\\nUpload at: https://protspace.app/explore\")\n files.download(str(output_path))\n\n except Exception as e:\n step_html.value = \"<b>Failed</b>\"\n print(f\"\\nError: {e}\")\n print(\"\\nTroubleshooting:\")\n print(\"- Check that protein IDs are UniProt accessions (e.g., P12345)\")\n print(\"- Try setting Annotations to 'None' to skip annotation fetching\")\n\n\ngen_btn.on_click(_on_gen)\n\ndisplay(VBox([\n HTML(\"<h4>Methods</h4>\"),\n HBox(list(method_toggles.values()), layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\"}),\n no_params,\n param_grid,\n HTML(\"<h4>Annotations</h4>\"),\n HBox([ann_preset, csv_upload], layout={\"gap\": \"10px\", \"align_items\": \"center\"}),\n HTML(\"<p><i>Optional: upload a CSV/TSV with custom annotations (first column = protein IDs).</i></p>\"),\n custom_ann,\n gen_btn,\n HTML(\"<p><i>Processing time depends on dataset size and selected methods.</i></p>\"),\n gen_out,\n]))"
|
|
49
49
|
},
|
|
50
50
|
{
|
|
51
51
|
"cell_type": "markdown",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "protspace"
|
|
3
|
-
version = "4.0
|
|
3
|
+
version = "4.2.0"
|
|
4
4
|
description = "A visualisation tool for protein embeddings from pLMs"
|
|
5
5
|
authors = [{ name = "Tobias Senoner", email = "tobias.senoner@tum.de" }]
|
|
6
6
|
readme = "README.md"
|
|
@@ -15,13 +15,12 @@ dependencies = [
|
|
|
15
15
|
"numpy>=1.23.0",
|
|
16
16
|
"pandas>=2.0.0",
|
|
17
17
|
"tqdm>=4.67.1",
|
|
18
|
-
"taxopy>=0.14.0",
|
|
19
18
|
"pymmseqs>=1.0.4",
|
|
20
19
|
"pyarrow>=20.0.0",
|
|
21
|
-
"unipressed>=1.4.0",
|
|
22
20
|
"biocentral-api>=1.1.2",
|
|
23
|
-
"
|
|
24
|
-
"
|
|
21
|
+
"requests>=2.32.4",
|
|
22
|
+
"typer>=0.24.1",
|
|
23
|
+
"rich>=14.3.3",
|
|
25
24
|
]
|
|
26
25
|
|
|
27
26
|
[project.optional-dependencies]
|
|
@@ -60,7 +59,6 @@ dev = [
|
|
|
60
59
|
"seaborn>=0.13.2",
|
|
61
60
|
"pytest>=8.3.3",
|
|
62
61
|
"polars>=1.26.0",
|
|
63
|
-
"taxopy>=0.14.0",
|
|
64
62
|
"jupytext>=1.17.1",
|
|
65
63
|
"pytest-cov>=7.0.0",
|
|
66
64
|
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "4.2.0"
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Shared HTTP utilities for UniProt-style REST API calls."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
|
|
5
|
+
import requests
|
|
6
|
+
|
|
7
|
+
logger = logging.getLogger(__name__)
|
|
8
|
+
|
|
9
|
+
API_TIMEOUT = 30
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def paginated_get(
|
|
13
|
+
url: str,
|
|
14
|
+
params: dict | None = None,
|
|
15
|
+
timeout: int = API_TIMEOUT,
|
|
16
|
+
result_key: str = "results",
|
|
17
|
+
) -> list[dict]:
|
|
18
|
+
"""Fetch all pages from a UniProt-style REST API endpoint.
|
|
19
|
+
|
|
20
|
+
Follows Link headers with rel="next" for automatic pagination.
|
|
21
|
+
Returns the concatenated contents of the ``result_key`` array
|
|
22
|
+
across all pages.
|
|
23
|
+
"""
|
|
24
|
+
results = []
|
|
25
|
+
|
|
26
|
+
while url:
|
|
27
|
+
resp = requests.get(url, params=params, timeout=timeout)
|
|
28
|
+
resp.raise_for_status()
|
|
29
|
+
data = resp.json()
|
|
30
|
+
results.extend(data.get(result_key, []))
|
|
31
|
+
|
|
32
|
+
# Follow Link header for next page
|
|
33
|
+
link = resp.headers.get("Link", "")
|
|
34
|
+
url = None
|
|
35
|
+
params = None # next-page URL already contains all params
|
|
36
|
+
if 'rel="next"' in link:
|
|
37
|
+
url = link.split(";")[0].strip(" <>")
|
|
38
|
+
|
|
39
|
+
return results
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
from tqdm import tqdm
|
|
5
|
+
|
|
6
|
+
from protspace.data.annotations.retrievers.base_retriever import BaseAnnotationRetriever
|
|
7
|
+
from protspace.data.annotations.retrievers.http_utils import paginated_get
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
TAXONOMY_API_URL = "https://rest.uniprot.org/taxonomy/search"
|
|
12
|
+
_BATCH_SIZE = 100 # Max taxon IDs per request (URL length safety)
|
|
13
|
+
|
|
14
|
+
# Taxonomy annotations
|
|
15
|
+
TAXONOMY_ANNOTATIONS = [
|
|
16
|
+
"root",
|
|
17
|
+
"domain",
|
|
18
|
+
"kingdom",
|
|
19
|
+
"phylum",
|
|
20
|
+
"class",
|
|
21
|
+
"order",
|
|
22
|
+
"family",
|
|
23
|
+
"genus",
|
|
24
|
+
"species",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class TaxonomyRetriever(BaseAnnotationRetriever):
|
|
29
|
+
"""Retrieves taxonomy lineage data from the UniProt Taxonomy API."""
|
|
30
|
+
|
|
31
|
+
def __init__(self, taxon_ids: list[int], annotations: list = None):
|
|
32
|
+
# Don't call super().__init__() as we use taxon_ids instead of headers
|
|
33
|
+
self.taxon_ids = self._validate_taxon_ids(taxon_ids)
|
|
34
|
+
self.annotations = annotations
|
|
35
|
+
|
|
36
|
+
def fetch_annotations(self) -> dict[int, dict[str, Any]]:
|
|
37
|
+
result = {}
|
|
38
|
+
|
|
39
|
+
with tqdm(
|
|
40
|
+
total=len(self.taxon_ids),
|
|
41
|
+
desc="Fetching taxonomy annotations",
|
|
42
|
+
unit="taxon",
|
|
43
|
+
) as pbar:
|
|
44
|
+
taxonomies_info = self._get_taxonomy_info(self.taxon_ids)
|
|
45
|
+
|
|
46
|
+
for taxon_id in self.taxon_ids:
|
|
47
|
+
if taxon_id in taxonomies_info:
|
|
48
|
+
result[taxon_id] = {"annotations": taxonomies_info[taxon_id]}
|
|
49
|
+
else:
|
|
50
|
+
result[taxon_id] = {
|
|
51
|
+
"annotations": dict.fromkeys(self.annotations, "")
|
|
52
|
+
}
|
|
53
|
+
pbar.update(1)
|
|
54
|
+
|
|
55
|
+
return result
|
|
56
|
+
|
|
57
|
+
def _validate_taxon_ids(self, taxon_ids: list[int]) -> list[int]:
|
|
58
|
+
for taxon_id in taxon_ids:
|
|
59
|
+
if not isinstance(taxon_id, int):
|
|
60
|
+
raise ValueError(f"Taxon ID {taxon_id} is not an integer")
|
|
61
|
+
return taxon_ids
|
|
62
|
+
|
|
63
|
+
def _get_taxonomy_info(self, taxon_ids: list[int]) -> dict[int, dict[str, str]]:
|
|
64
|
+
result = {}
|
|
65
|
+
|
|
66
|
+
# Fetch in batches to stay within URL length limits
|
|
67
|
+
for i in range(0, len(taxon_ids), _BATCH_SIZE):
|
|
68
|
+
batch = taxon_ids[i : i + _BATCH_SIZE]
|
|
69
|
+
batch_results = self._fetch_batch(batch)
|
|
70
|
+
result.update(batch_results)
|
|
71
|
+
|
|
72
|
+
return result
|
|
73
|
+
|
|
74
|
+
def _fetch_batch(self, taxon_ids: list[int]) -> dict[int, dict[str, str]]:
|
|
75
|
+
"""Fetch taxonomy info for a batch of taxon IDs from UniProt Taxonomy API."""
|
|
76
|
+
result = {}
|
|
77
|
+
query = " OR ".join(f"id:{tid}" for tid in taxon_ids)
|
|
78
|
+
|
|
79
|
+
try:
|
|
80
|
+
entries = paginated_get(
|
|
81
|
+
TAXONOMY_API_URL,
|
|
82
|
+
params={"query": query, "format": "json", "size": "500"},
|
|
83
|
+
)
|
|
84
|
+
for entry in entries:
|
|
85
|
+
taxon_id = entry.get("taxonId")
|
|
86
|
+
if taxon_id is not None:
|
|
87
|
+
result[taxon_id] = self._extract_taxonomy(entry)
|
|
88
|
+
|
|
89
|
+
except Exception as e:
|
|
90
|
+
logger.error(f"Failed to fetch taxonomy batch: {e}")
|
|
91
|
+
for tid in taxon_ids:
|
|
92
|
+
if tid not in result:
|
|
93
|
+
result[tid] = dict.fromkeys(self.annotations, "")
|
|
94
|
+
|
|
95
|
+
return result
|
|
96
|
+
|
|
97
|
+
def _extract_taxonomy(self, entry: dict) -> dict[str, str]:
|
|
98
|
+
"""Extract taxonomy ranks from a UniProt Taxonomy API result."""
|
|
99
|
+
lineage = entry.get("lineage", [])
|
|
100
|
+
rank_map = {item["rank"]: item["scientificName"] for item in lineage}
|
|
101
|
+
|
|
102
|
+
# The lineage only contains ancestors. If the taxon itself is a
|
|
103
|
+
# species (e.g., Human 9606), include it from the entry's own rank.
|
|
104
|
+
own_rank = entry.get("rank", "")
|
|
105
|
+
if own_rank and own_rank not in rank_map:
|
|
106
|
+
rank_map[own_rank] = entry.get("scientificName", "")
|
|
107
|
+
|
|
108
|
+
full_taxonomy_info = {
|
|
109
|
+
"root": rank_map.get("no rank", ""),
|
|
110
|
+
"domain": rank_map.get("domain", "") or rank_map.get("realm", ""),
|
|
111
|
+
"kingdom": rank_map.get("kingdom", ""),
|
|
112
|
+
"phylum": rank_map.get("phylum", ""),
|
|
113
|
+
"class": rank_map.get("class", ""),
|
|
114
|
+
"order": rank_map.get("order", ""),
|
|
115
|
+
"family": rank_map.get("family", ""),
|
|
116
|
+
"genus": rank_map.get("genus", ""),
|
|
117
|
+
"species": rank_map.get("species", ""),
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
# Filter based on requested taxonomy annotations
|
|
121
|
+
return {
|
|
122
|
+
annotation: full_taxonomy_info.get(annotation, "")
|
|
123
|
+
for annotation in self.annotations
|
|
124
|
+
}
|
{protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/uniprot_retriever.py
RENAMED
|
@@ -4,15 +4,14 @@ UniProt annotation retriever.
|
|
|
4
4
|
This module fetches protein annotations from the UniProt API.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
import json
|
|
8
7
|
import logging
|
|
9
8
|
from collections import namedtuple
|
|
10
9
|
|
|
11
10
|
import requests
|
|
12
11
|
from tqdm import tqdm
|
|
13
|
-
from unipressed import UniprotkbClient
|
|
14
12
|
|
|
15
13
|
from protspace.data.annotations.retrievers.base_retriever import BaseAnnotationRetriever
|
|
14
|
+
from protspace.data.annotations.retrievers.http_utils import API_TIMEOUT, paginated_get
|
|
16
15
|
from protspace.data.parsers.uniprot_parser import UniProtEntry
|
|
17
16
|
|
|
18
17
|
logger = logging.getLogger(__name__)
|
|
@@ -42,15 +41,8 @@ UNIPROT_ANNOTATIONS = [
|
|
|
42
41
|
ProteinAnnotations = namedtuple("ProteinAnnotations", ["identifier", "annotations"])
|
|
43
42
|
|
|
44
43
|
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
def _fetch_one_with_timeout(accession: str, timeout: int = _API_TIMEOUT) -> dict:
|
|
49
|
-
"""Fetch a single UniProt entry with a timeout.
|
|
50
|
-
|
|
51
|
-
The unipressed library's fetch_one() uses requests.get without a timeout,
|
|
52
|
-
which can hang indefinitely. This wrapper adds timeout protection.
|
|
53
|
-
"""
|
|
44
|
+
def _fetch_one_with_timeout(accession: str, timeout: int = API_TIMEOUT) -> dict:
|
|
45
|
+
"""Fetch a single UniProt entry by accession with timeout protection."""
|
|
54
46
|
url = f"https://rest.uniprot.org/uniprotkb/{accession}.json"
|
|
55
47
|
resp = requests.get(url, timeout=timeout)
|
|
56
48
|
resp.raise_for_status()
|
|
@@ -58,15 +50,9 @@ def _fetch_one_with_timeout(accession: str, timeout: int = _API_TIMEOUT) -> dict
|
|
|
58
50
|
|
|
59
51
|
|
|
60
52
|
def _fetch_uniparc_sequence(
|
|
61
|
-
uniparc_id: str, timeout: int =
|
|
53
|
+
uniparc_id: str, timeout: int = API_TIMEOUT
|
|
62
54
|
) -> tuple[str, int]:
|
|
63
|
-
"""Fetch sequence and length from UniParc.
|
|
64
|
-
|
|
65
|
-
Deleted UniProt entries still have their sequence archived in UniParc.
|
|
66
|
-
|
|
67
|
-
Returns:
|
|
68
|
-
Tuple of (sequence_string, length) or ("", 0) on failure.
|
|
69
|
-
"""
|
|
55
|
+
"""Fetch sequence and length from UniParc for deleted entries."""
|
|
70
56
|
url = f"https://rest.uniprot.org/uniparc/{uniparc_id}.json"
|
|
71
57
|
try:
|
|
72
58
|
resp = requests.get(url, timeout=timeout)
|
|
@@ -81,6 +67,22 @@ def _fetch_uniparc_sequence(
|
|
|
81
67
|
return "", 0
|
|
82
68
|
|
|
83
69
|
|
|
70
|
+
def _fetch_many_accessions(accessions: list[str]) -> list[dict]:
|
|
71
|
+
"""Fetch multiple UniProt entries by accession."""
|
|
72
|
+
return paginated_get(
|
|
73
|
+
"https://rest.uniprot.org/uniprotkb/accessions",
|
|
74
|
+
params={"accessions": ",".join(accessions)},
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _search_sec_acc(accession: str) -> list[dict]:
|
|
79
|
+
"""Search UniProt by secondary accession (fallback for inactive entries)."""
|
|
80
|
+
return paginated_get(
|
|
81
|
+
"https://rest.uniprot.org/uniprotkb/search",
|
|
82
|
+
params={"query": f"sec_acc:{accession}", "format": "json", "size": "500"},
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
84
86
|
class UniProtRetriever(BaseAnnotationRetriever):
|
|
85
87
|
"""Retrieves annotations from UniProt API."""
|
|
86
88
|
|
|
@@ -207,15 +209,7 @@ class UniProtRetriever(BaseAnnotationRetriever):
|
|
|
207
209
|
except Exception:
|
|
208
210
|
# fetch_one failed — fall back to sec_acc: search
|
|
209
211
|
try:
|
|
210
|
-
records =
|
|
211
|
-
for page in UniprotkbClient.search(
|
|
212
|
-
query=f"sec_acc:{accession}", format="json"
|
|
213
|
-
).each_page():
|
|
214
|
-
content = page.read() if hasattr(page, "read") else page
|
|
215
|
-
parsed = (
|
|
216
|
-
json.loads(content) if isinstance(content, str) else content
|
|
217
|
-
)
|
|
218
|
-
records.extend(parsed.get("results", []))
|
|
212
|
+
records = _search_sec_acc(accession)
|
|
219
213
|
if records:
|
|
220
214
|
entry = UniProtEntry(records[0])
|
|
221
215
|
annotations_dict = self._extract_annotations(entry)
|
|
@@ -277,8 +271,7 @@ class UniProtRetriever(BaseAnnotationRetriever):
|
|
|
277
271
|
batch = self.headers[i : i + batch_size]
|
|
278
272
|
|
|
279
273
|
try:
|
|
280
|
-
|
|
281
|
-
records = UniprotkbClient.fetch_many(batch)
|
|
274
|
+
records = _fetch_many_accessions(batch)
|
|
282
275
|
|
|
283
276
|
# Parse each record and track returned identifiers
|
|
284
277
|
returned_ids = set()
|
|
@@ -50,7 +50,8 @@ xref_pdb - PDB cross-references (list)
|
|
|
50
50
|
from typing import Any
|
|
51
51
|
|
|
52
52
|
import pandas as pd
|
|
53
|
-
|
|
53
|
+
|
|
54
|
+
from protspace.data.annotations.retrievers.http_utils import paginated_get
|
|
54
55
|
|
|
55
56
|
# ECO evidence code mapping (ECO ID → short human-readable code)
|
|
56
57
|
ECO_TO_SHORT: dict[str, str] = {
|
|
@@ -586,9 +587,12 @@ def fetch_uniprot_data(
|
|
|
586
587
|
f"Invalid properties: {invalid}. Available: {AVAILABLE_PROPERTIES}"
|
|
587
588
|
)
|
|
588
589
|
|
|
589
|
-
# Fetch records
|
|
590
|
-
|
|
591
|
-
|
|
590
|
+
# Fetch records via UniProt REST API
|
|
591
|
+
results = paginated_get(
|
|
592
|
+
"https://rest.uniprot.org/uniprotkb/accessions",
|
|
593
|
+
params={"accessions": ",".join(accessions)},
|
|
594
|
+
)
|
|
595
|
+
entries = [UniProtEntry(record) for record in results]
|
|
592
596
|
|
|
593
597
|
# Extract properties
|
|
594
598
|
data = []
|