protspace 4.0.2__tar.gz → 4.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. {protspace-4.0.2 → protspace-4.2.0}/CHANGELOG.md +61 -0
  2. {protspace-4.0.2 → protspace-4.2.0}/CLAUDE.md +2 -2
  3. {protspace-4.0.2 → protspace-4.2.0}/PKG-INFO +4 -5
  4. {protspace-4.0.2 → protspace-4.2.0}/docs/annotations.md +1 -1
  5. {protspace-4.0.2 → protspace-4.2.0}/notebooks/ProtSpace_Preparation.ipynb +1 -1
  6. {protspace-4.0.2 → protspace-4.2.0}/pyproject.toml +4 -6
  7. protspace-4.2.0/src/protspace/__init__.py +1 -0
  8. protspace-4.2.0/src/protspace/data/annotations/retrievers/http_utils.py +39 -0
  9. protspace-4.2.0/src/protspace/data/annotations/retrievers/taxonomy_retriever.py +124 -0
  10. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/uniprot_retriever.py +23 -30
  11. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/parsers/uniprot_parser.py +8 -4
  12. protspace-4.2.0/tests/test_taxonomy_annotation_retriever.py +446 -0
  13. {protspace-4.0.2 → protspace-4.2.0}/tests/test_uniprot_annotation_retriever.py +50 -82
  14. {protspace-4.0.2 → protspace-4.2.0}/uv.lock +5 -32
  15. protspace-4.0.2/src/protspace/__init__.py +0 -1
  16. protspace-4.0.2/src/protspace/data/annotations/retrievers/taxonomy_retriever.py +0 -226
  17. protspace-4.0.2/tests/test_taxonomy_annotation_retriever.py +0 -225
  18. {protspace-4.0.2 → protspace-4.2.0}/.dockerignore +0 -0
  19. {protspace-4.0.2 → protspace-4.2.0}/.env.example +0 -0
  20. {protspace-4.0.2 → protspace-4.2.0}/.githooks/pre-commit +0 -0
  21. {protspace-4.0.2 → protspace-4.2.0}/.github/workflows/ci.yml +0 -0
  22. {protspace-4.0.2 → protspace-4.2.0}/.github/workflows/publish.yml +0 -0
  23. {protspace-4.0.2 → protspace-4.2.0}/.github/workflows/release.yml +0 -0
  24. {protspace-4.0.2 → protspace-4.2.0}/.gitignore +0 -0
  25. {protspace-4.0.2 → protspace-4.2.0}/.python-version +0 -0
  26. {protspace-4.0.2 → protspace-4.2.0}/Dockerfile +0 -0
  27. {protspace-4.0.2 → protspace-4.2.0}/LICENSE +0 -0
  28. {protspace-4.0.2 → protspace-4.2.0}/README.md +0 -0
  29. {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/3FTx.csv +0 -0
  30. {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/parquetbundle/3FTx_accession.csv +0 -0
  31. {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/parquetbundle/3ftx_with_db_styled.parquetbundle +0 -0
  32. {protspace-4.0.2 → protspace-4.2.0}/data/3FTx/parquetbundle/styles.json +0 -0
  33. {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/Pla2g2.csv +0 -0
  34. {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/Pla2g2.fasta +0 -0
  35. {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/Pla2g2_pdb.zip +0 -0
  36. {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/protspace_files/Pla2g2.json +0 -0
  37. {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/protspace_files/Pla2g2_customized.json +0 -0
  38. {protspace-4.0.2 → protspace-4.2.0}/data/Pla2g2/style.json +0 -0
  39. {protspace-4.0.2 → protspace-4.2.0}/data/toxins/protspace/projections_data.parquet +0 -0
  40. {protspace-4.0.2 → protspace-4.2.0}/data/toxins/protspace/projections_metadata.parquet +0 -0
  41. {protspace-4.0.2 → protspace-4.2.0}/data/toxins/protspace/selected_annotations.parquet +0 -0
  42. {protspace-4.0.2 → protspace-4.2.0}/data/toxins/toxins.json +0 -0
  43. {protspace-4.0.2 → protspace-4.2.0}/data/toxins/toxins.parquetbundle +0 -0
  44. {protspace-4.0.2 → protspace-4.2.0}/docs/cli.md +0 -0
  45. {protspace-4.0.2 → protspace-4.2.0}/docs/protspace_example.png +0 -0
  46. {protspace-4.0.2 → protspace-4.2.0}/docs/styling.md +0 -0
  47. {protspace-4.0.2 → protspace-4.2.0}/notebooks/ClickThrough_GenerateEmbeddings.ipynb +0 -0
  48. {protspace-4.0.2 → protspace-4.2.0}/scripts/count_h5_rows.py +0 -0
  49. {protspace-4.0.2 → protspace-4.2.0}/scripts/count_parquetbundle_entries.py +0 -0
  50. {protspace-4.0.2 → protspace-4.2.0}/scripts/download_clan_fastas.py +0 -0
  51. {protspace-4.0.2 → protspace-4.2.0}/scripts/generate_examples/datasets.toml +0 -0
  52. {protspace-4.0.2 → protspace-4.2.0}/scripts/generate_examples/generate.py +0 -0
  53. {protspace-4.0.2 → protspace-4.2.0}/scripts/pfam_clans/README.md +0 -0
  54. {protspace-4.0.2 → protspace-4.2.0}/scripts/pfam_clans/create_pfam_clan_query.py +0 -0
  55. {protspace-4.0.2 → protspace-4.2.0}/scripts/pfam_clans/merge_h5_batches.py +0 -0
  56. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/app.py +0 -0
  57. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/__init__.py +0 -0
  58. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/annotated_image.png +0 -0
  59. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/custom.css +0 -0
  60. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/__init__.py +0 -0
  61. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_faq.md +0 -0
  62. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_how_it_works.md +0 -0
  63. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_json.md +0 -0
  64. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/help_content/help_overview.md +0 -0
  65. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/assets/rostlab_logo.png +0 -0
  66. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/__init__.py +0 -0
  67. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/annotate.py +0 -0
  68. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/app.py +0 -0
  69. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/bundle.py +0 -0
  70. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/common_options.py +0 -0
  71. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/embed.py +0 -0
  72. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/prepare.py +0 -0
  73. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/project.py +0 -0
  74. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/serve.py +0 -0
  75. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/cli/style.py +0 -0
  76. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/core/__init__.py +0 -0
  77. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/core/config.py +0 -0
  78. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/core/constants.py +0 -0
  79. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/__init__.py +0 -0
  80. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/__init__.py +0 -0
  81. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/configuration.py +0 -0
  82. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/manager.py +0 -0
  83. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/merging.py +0 -0
  84. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/__init__.py +0 -0
  85. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/base_retriever.py +0 -0
  86. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/retrievers/interpro_retriever.py +0 -0
  87. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/scores.py +0 -0
  88. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/__init__.py +0 -0
  89. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/interpro_transforms.py +0 -0
  90. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/transformer.py +0 -0
  91. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/annotations/transformers/uniprot_transforms.py +0 -0
  92. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/embedding/__init__.py +0 -0
  93. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/embedding/biocentral.py +0 -0
  94. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/__init__.py +0 -0
  95. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/bundle.py +0 -0
  96. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/fasta.py +0 -0
  97. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/formatters.py +0 -0
  98. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/settings_converter.py +0 -0
  99. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/io/writers.py +0 -0
  100. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/__init__.py +0 -0
  101. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/embedding_set.py +0 -0
  102. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/fasta.py +0 -0
  103. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/h5.py +0 -0
  104. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/query.py +0 -0
  105. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/loaders/similarity.py +0 -0
  106. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/parsers/__init__.py +0 -0
  107. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/processors/__init__.py +0 -0
  108. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/processors/base_processor.py +0 -0
  109. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/data/processors/pipeline.py +0 -0
  110. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/main.py +0 -0
  111. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/__init__.py +0 -0
  112. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/callbacks.py +0 -0
  113. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/layout.py +0 -0
  114. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/ui/styles.py +0 -0
  115. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/__init__.py +0 -0
  116. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/add_annotation_style.py +0 -0
  117. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/arrow_reader.py +0 -0
  118. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/constants.py +0 -0
  119. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/utils/reducers.py +0 -0
  120. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/visualization/__init__.py +0 -0
  121. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/visualization/molstar.py +0 -0
  122. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/visualization/plotting.py +0 -0
  123. {protspace-4.0.2 → protspace-4.2.0}/src/protspace/wsgi.py +0 -0
  124. {protspace-4.0.2 → protspace-4.2.0}/tests/README.md +0 -0
  125. {protspace-4.0.2 → protspace-4.2.0}/tests/__init__.py +0 -0
  126. {protspace-4.0.2 → protspace-4.2.0}/tests/test_annotation_manager.py +0 -0
  127. {protspace-4.0.2 → protspace-4.2.0}/tests/test_base_data_processor.py +0 -0
  128. {protspace-4.0.2 → protspace-4.2.0}/tests/test_biocentral_embedder.py +0 -0
  129. {protspace-4.0.2 → protspace-4.2.0}/tests/test_bundle_settings.py +0 -0
  130. {protspace-4.0.2 → protspace-4.2.0}/tests/test_config.py +0 -0
  131. {protspace-4.0.2 → protspace-4.2.0}/tests/test_config_validation.py +0 -0
  132. {protspace-4.0.2 → protspace-4.2.0}/tests/test_fasta.py +0 -0
  133. {protspace-4.0.2 → protspace-4.2.0}/tests/test_formatters.py +0 -0
  134. {protspace-4.0.2 → protspace-4.2.0}/tests/test_h5_parse_identifier.py +0 -0
  135. {protspace-4.0.2 → protspace-4.2.0}/tests/test_interpro_annotation_retriever.py +0 -0
  136. {protspace-4.0.2 → protspace-4.2.0}/tests/test_output_combinations.py +0 -0
  137. {protspace-4.0.2 → protspace-4.2.0}/tests/test_pipeline_utils.py +0 -0
  138. {protspace-4.0.2 → protspace-4.2.0}/tests/test_reducers.py +0 -0
  139. {protspace-4.0.2 → protspace-4.2.0}/tests/test_settings_converter.py +0 -0
  140. {protspace-4.0.2 → protspace-4.2.0}/tests/test_transformer.py +0 -0
@@ -1,6 +1,67 @@
1
1
  # CHANGELOG
2
2
 
3
3
 
4
+ ## v4.2.0 (2026-03-28)
5
+
6
+ ### Features
7
+
8
+ * feat: replace unipressed with direct UniProt REST API calls
9
+
10
+ Replace the unipressed library (community UniProt API wrapper) with
11
+ direct HTTP calls to rest.uniprot.org. Adds _fetch_many_accessions()
12
+ and _search_sec_acc() helpers using the same Link-header pagination
13
+ pattern as the taxonomy retriever.
14
+
15
+ Simplifies the sec_acc search fallback from 8 lines of page-parsing
16
+ to a single function call.
17
+
18
+ Closes #32
19
+
20
+ Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`5605a93`](https://github.com/tsenoner/protspace/commit/5605a935e680132ee7d22adb30a45b3ff26df02b))
21
+
22
+ ### Refactoring
23
+
24
+ * refactor: extract shared paginated_get() utility for REST API calls
25
+
26
+ Consolidate the duplicated Link-header pagination loop (4 instances
27
+ across uniprot_retriever, taxonomy_retriever, and uniprot_parser) into
28
+ a single paginated_get() function in http_utils.py. Each caller is
29
+ now a 1-3 line function.
30
+
31
+ Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`cd528ed`](https://github.com/tsenoner/protspace/commit/cd528ed68ba7cf9bf3a33183345b483c00b4f9d1))
32
+
33
+ ### Unknown
34
+
35
+ * Merge pull request #39 from tsenoner/feat/replace-unipressed-with-direct-api
36
+
37
+ Replace unipressed with direct UniProt REST API calls ([`5d365c7`](https://github.com/tsenoner/protspace/commit/5d365c70dfb66fe1754fb610c7df3acb3cfae480))
38
+
39
+
40
+ ## v4.1.0 (2026-03-28)
41
+
42
+ ### Features
43
+
44
+ * feat: replace taxopy with UniProt Taxonomy API for taxonomy lookups
45
+
46
+ Replace the taxopy-based taxonomy retriever (which required downloading
47
+ the full NCBI taxonomy database ~50 MB on first use) with the UniProt
48
+ Taxonomy API (/taxonomy/search). This eliminates the slow first-run
49
+ download, weekly cache refresh, and ~120 lines of cache management code.
50
+
51
+ Also fix typer[all] → typer (the [all] extra was removed) and add
52
+ requests as an explicit core dependency.
53
+
54
+ Closes #36
55
+
56
+ Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`754fa9b`](https://github.com/tsenoner/protspace/commit/754fa9b45cf5a913a97823b2b45012c49221ca7a))
57
+
58
+ ### Unknown
59
+
60
+ * Merge pull request #37 from tsenoner/feat/replace-taxopy-with-uniprot-api
61
+
62
+ Replace taxopy with UniProt Taxonomy API ([`034c0ec`](https://github.com/tsenoner/protspace/commit/034c0ece32238144fe68819ce2b6e5e43d8b57a5))
63
+
64
+
4
65
  ## v4.0.2 (2026-03-28)
5
66
 
6
67
  ### Fixes
@@ -205,7 +205,7 @@ uv run pytest tests/ --cov=src/protspace # With coverage
205
205
  | `test_biocentral_embedder.py` | 23 | Biocentral API client, embedding flow |
206
206
  | `test_fasta.py` | 17 | FASTA parsing, edge cases, CSV annotation loading |
207
207
  | `test_config_validation.py` | 12 | DimensionReductionConfig parameter validation |
208
- | `test_taxonomy_annotation_retriever.py` | 12 | Taxonomy database lookups |
208
+ | `test_taxonomy_annotation_retriever.py` | 15 | Taxonomy via UniProt Taxonomy API (mocked + integration) |
209
209
  | `test_h5_parse_identifier.py` | 9 | HDF5 key parsing, identifier extraction |
210
210
  | `test_base_data_processor.py` | 8 | BaseProcessor: reduction, output creation, save |
211
211
  | `test_formatters.py` | 5 | ProteinAnnotations → DataFrame formatting |
@@ -225,7 +225,7 @@ Located in `notebooks/`:
225
225
 
226
226
  ## Dependencies
227
227
 
228
- **Core:** h5py, scikit-learn, umap-learn, pacmap (includes annoy), numpy, pandas, pyarrow, tqdm, taxopy, pymmseqs, unipressed, biocentral-api, typer, rich
228
+ **Core:** h5py, scikit-learn, umap-learn, pacmap (includes annoy), numpy, pandas, pyarrow, tqdm, requests, pymmseqs, biocentral-api, typer, rich
229
229
 
230
230
  **Frontend (optional):** dash, plotly, dash-bootstrap-components, dash-molstar, gunicorn
231
231
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: protspace
3
- Version: 4.0.2
3
+ Version: 4.2.0
4
4
  Summary: A visualisation tool for protein embeddings from pLMs
5
5
  Author-email: Tobias Senoner <tobias.senoner@tum.de>
6
6
  License-Expression: GPL-3.0
@@ -13,13 +13,12 @@ Requires-Dist: pacmap>=0.8.0
13
13
  Requires-Dist: pandas>=2.0.0
14
14
  Requires-Dist: pyarrow>=20.0.0
15
15
  Requires-Dist: pymmseqs>=1.0.4
16
- Requires-Dist: rich>=13.0.0
16
+ Requires-Dist: requests>=2.32.4
17
+ Requires-Dist: rich>=14.3.3
17
18
  Requires-Dist: scikit-learn>=1.6.1
18
- Requires-Dist: taxopy>=0.14.0
19
19
  Requires-Dist: tqdm>=4.67.1
20
- Requires-Dist: typer[all]>=0.15.0
20
+ Requires-Dist: typer>=0.24.1
21
21
  Requires-Dist: umap-learn>=0.5.10
22
- Requires-Dist: unipressed>=1.4.0
23
22
  Provides-Extra: frontend
24
23
  Requires-Dist: dash-bootstrap-components>=1.6.0; extra == 'frontend'
25
24
  Requires-Dist: dash-daq>=0.5.0; extra == 'frontend'
@@ -46,7 +46,7 @@ With `--keep-tmp`, only API-fetched annotations are cached; the CSV is always re
46
46
 
47
47
  ## UniProt Annotations
48
48
 
49
- 14 annotations retrieved from the [UniProt REST API](https://rest.uniprot.org/) via `unipressed` (batch size: 100):
49
+ 14 annotations retrieved from the [UniProt REST API](https://rest.uniprot.org/) (batch size: 100):
50
50
 
51
51
  | Name | Description | Example |
52
52
  | ------------------------- | ------------------------------------ | -------------------------------------------------------------- |
@@ -45,7 +45,7 @@
45
45
  "cellView": "form"
46
46
  },
47
47
  "outputs": [],
48
- "source": "# @title 3. Generate & Download {display-mode: \"form\"}\n# @markdown Configure methods and annotations, then click Generate.\n# @markdown Processing time depends on dataset size and selected methods.\n\nfrom ipywidgets import (\n HTML,\n Button,\n Checkbox,\n Dropdown,\n FileUpload,\n FloatSlider,\n GridBox,\n HBox,\n IntSlider,\n Output,\n ToggleButton,\n VBox,\n)\n\nMETHODS = [\"PCA\", \"UMAP\", \"t-SNE\", \"MDS\", \"PaCMAP\", \"LocalMAP\"]\nANNOTATIONS = {\n \"UniProt\": [\"annotation_score\", \"cc_subcellular_location\", \"ec\", \"fragment\", \"go_bp\", \"go_cc\", \"go_mf\", \"keyword\", \"length\", \"protein_existence\", \"protein_families\", \"reviewed\", \"xref_pdb\"],\n \"InterPro\": [\"cath\", \"cdd\", \"panther\", \"pfam\", \"prints\", \"prosite\", \"signal_peptide\", \"smart\", \"superfamily\"],\n \"Taxonomy\": [\"root\", \"domain\", \"kingdom\", \"phylum\", \"class\", \"order\", \"family\", \"genus\", \"species\"],\n}\nANNOTATION_DEFAULTS = {\"ec\", \"keyword\", \"length\", \"protein_families\", \"reviewed\"}\n\n# --- Methods ---\nmethod_toggles = {}\nfor m in METHODS:\n sel = m in {\"PCA\", \"UMAP\"}\n method_toggles[m] = ToggleButton(value=sel, description=m, button_style=\"info\" if sel else \"\", layout={\"width\": \"auto\", \"height\": \"32px\"})\n\n# --- DR Parameters ---\n_s = {\"description_width\": \"110px\"}\n_l = {\"width\": \"100%\"}\n_g = {\"border\": \"1px solid #666\", \"padding\": \"6px 10px\", \"margin\": \"2px\", \"flex\": \"1 1 220px\", \"overflow\": \"hidden\"}\n\nparam_groups = []\nw_nn = IntSlider(value=25, min=2, max=500, description=\"n_neighbors:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP / PaCMAP / LocalMAP</b>\"), w_nn], layout=_g), {\"UMAP\", \"PaCMAP\", \"LocalMAP\"}))\nw_md = FloatSlider(value=0.4, min=0.0, max=0.99, step=0.01, description=\"min_dist:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP</b>\"), w_md], layout=_g), {\"UMAP\"}))\nw_mn = FloatSlider(value=0.5, min=0.0, max=1.0, step=0.1, description=\"mn_ratio:\", style=_s, layout=_l)\nw_fp = FloatSlider(value=2.0, min=0.0, max=5.0, step=0.1, description=\"fp_ratio:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>PaCMAP / LocalMAP</b>\"), w_mn, w_fp], layout=_g), {\"PaCMAP\", \"LocalMAP\"}))\nw_pp = FloatSlider(value=30.0, min=5.0, max=500.0, step=1.0, description=\"perplexity:\", style=_s, layout=_l)\nw_lr = FloatSlider(value=200.0, min=1.0, max=1000.0, step=10.0, description=\"learning_rate:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>t-SNE</b>\"), w_pp, w_lr], layout=_g), {\"t-SNE\"}))\n\nno_params = HTML(\"<i>PCA/MDS have no adjustable parameters.</i>\")\npw = {\"n_neighbors\": w_nn, \"min_dist\": w_md, \"perplexity\": w_pp, \"learning_rate\": w_lr, \"mn_ratio\": w_mn, \"fp_ratio\": w_fp}\nparam_grid = HBox([g for g, _ in param_groups], layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\", \"width\": \"100%\"})\n\n\ndef _update_params(_c=None):\n sel = {m for m, t in method_toggles.items() if t.value}\n any_vis = False\n for gw, req in param_groups:\n vis = bool(req & sel)\n gw.layout.display = \"\" if vis else \"none\"\n if vis:\n any_vis = True\n no_params.layout.display = \"\" if not any_vis else \"none\"\n\n\nfor tb in method_toggles.values():\n def _toggle(c):\n c[\"owner\"].button_style = \"info\" if c[\"new\"] else \"\"\n _update_params()\n tb.observe(_toggle, names=\"value\")\n_update_params()\n\n# --- Annotations ---\nann_preset = Dropdown(\n options=[\"Default (recommended)\", \"All annotations\", \"Custom...\"],\n value=\"Default (recommended)\",\n description=\"Annotations:\",\n style={\"description_width\": \"initial\"},\n)\nann_cbs = {}\nann_cols = []\nfor cat, names in ANNOTATIONS.items():\n cbs = {n: Checkbox(value=n in ANNOTATION_DEFAULTS, description=n, indent=False, layout={\"width\": \"auto\"}) for n in names}\n ann_cbs[cat] = cbs\n grid = GridBox(list(cbs.values()), layout={\"grid_template_columns\": \"repeat(2, auto)\", \"grid_gap\": \"0 6px\"})\n ann_cols.append(VBox([HTML(f\"<b>{cat}</b>\"), grid], layout={\"flex\": \"1 1 200px\", \"border\": \"1px solid #666\", \"padding\": \"6px\", \"margin\": \"2px\"}))\ncustom_ann = HBox(ann_cols, layout={\"flex_flow\": \"row wrap\"})\ncustom_ann.layout.display = \"none\"\nann_preset.observe(lambda c: setattr(custom_ann.layout, \"display\", \"\" if c[\"new\"] == \"Custom...\" else \"none\"), names=\"value\")\n\n# --- Custom CSV upload ---\ncsv_upload = FileUpload(accept=\".csv,.tsv\", multiple=False, description=\"Choose CSV/TSV\")\n\n\ndef _get_ann():\n p = ann_preset.value\n if p == \"Default (recommended)\":\n return [\"default\"]\n if p == \"All annotations\":\n return [\"all\"]\n sel = [n for cbs in ann_cbs.values() for n, cb in cbs.items() if cb.value]\n return sel if sel else [\"default\"]\n\n\ndef _save_csv():\n \"\"\"Save uploaded CSV to disk and return the path, or None.\"\"\"\n if not csv_upload.value:\n return None\n fname = list(csv_upload.value.keys())[0]\n content = csv_upload.value[fname][\"content\"]\n Path(fname).write_bytes(content)\n return fname\n\n\n# --- Generate ---\ngen_btn = Button(description=\"Generate\", button_style=\"primary\", icon=\"play\", layout={\"width\": \"180px\", \"height\": \"40px\"})\ngen_out = Output()\n\n\ndef _on_gen(_b):\n with gen_out:\n clear_output()\n if not input_args:\n print(\"Select input data in cell above first.\")\n return\n method_specs = [m.lower().replace(\"-\", \"\") + \"2\" for m, t in method_toggles.items() if t.value]\n if not method_specs:\n print(\"Select at least one method.\")\n return\n\n ann = _get_ann() or []\n csv_path = _save_csv()\n if csv_path:\n ann.append(csv_path)\n if not ann:\n ann = None\n\n inp = input_args\n\n out_dir = Path(\"output\")\n out_dir.mkdir(exist_ok=True)\n cache_dir = out_dir / \"tmp\"\n cache_dir.mkdir(exist_ok=True)\n output_path = out_dir / \"data.parquetbundle\"\n\n step_html = HTML(value=\"<b>Step 1/4: Loading embeddings...</b>\")\n display(step_html)\n\n try:\n t0 = _time.time()\n embedding_sets = []\n\n if inp[\"type\"] == \"query\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n step_html.value = \"<b>Step 1/6: Downloading FASTA...</b>\"\n headers, fasta_path = query_uniprot(inp[\"query\"])\n if not headers:\n print(f\"No sequences found for query: {inp['query']}\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 2/6: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(fasta_path, emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = fasta_path\n embedding_sets.append(emb_set)\n elif inp[\"type\"] == \"fasta\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 1/5: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(Path(inp[\"path\"]), emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = Path(inp[\"path\"])\n embedding_sets.append(emb_set)\n else:\n h5_path = Path(inp[\"path\"])\n name_override = inp.get(\"name\")\n emb_set = load_h5([h5_path], name_override=name_override)\n embedding_sets.append(emb_set)\n\n n_proteins = len(embedding_sets[0].headers)\n\n # Build pipeline with caching enabled\n reducer_params = ReducerParams(\n n_neighbors=pw[\"n_neighbors\"].value,\n min_dist=pw[\"min_dist\"].value,\n perplexity=pw[\"perplexity\"].value,\n learning_rate=pw[\"learning_rate\"].value,\n mn_ratio=pw[\"mn_ratio\"].value,\n fp_ratio=pw[\"fp_ratio\"].value,\n )\n config = PipelineConfig(\n methods=method_specs,\n output_path=output_path,\n bundled=True,\n keep_tmp=True,\n intermediate_dir=cache_dir,\n annotations=ann,\n reducer_params=reducer_params,\n )\n pipeline = ReductionPipeline(config)\n\n # Step 2: Annotations (cached after first run)\n step_html.value = \"<b>Step 2/4: Fetching annotations...</b>\"\n metadata = pipeline._fetch_annotations(embedding_sets[0].headers)\n\n # Step 3: Dimensionality reduction\n step_html.value = \"<b>Step 3/4: Reducing dimensions...</b>\"\n\n # Build full metadata\n all_headers = embedding_sets[0].headers\n full_metadata = _pd.DataFrame({\"identifier\": all_headers})\n if len(metadata.columns) > 1:\n metadata = metadata.astype(str)\n id_col = metadata.columns[0]\n if id_col != \"identifier\":\n metadata = metadata.rename(columns={id_col: \"identifier\"})\n full_metadata = full_metadata.merge(\n metadata.drop_duplicates(\"identifier\"),\n on=\"identifier\",\n how=\"left\",\n )\n metadata = full_metadata\n\n all_reductions = pipeline._run_reductions(embedding_sets)\n\n # Step 4: Bundle & download\n step_html.value = \"<b>Step 4/4: Bundling & downloading...</b>\"\n output = pipeline.base.create_output(metadata, all_reductions, all_headers)\n pipeline.base.save_output(output, output_path, bundled=True)\n\n kb = output_path.stat().st_size / 1024\n total_time = _time.time() - t0\n step_html.value = f\"<b>Done! ({kb:.0f} KB, {total_time:.0f}s)</b>\"\n print(f\"Processed {n_proteins} proteins with {len(method_specs)} method(s)\")\n print(f\"\\nUpload at: https://protspace.app/explore\")\n files.download(str(output_path))\n\n except Exception as e:\n step_html.value = \"<b>Failed</b>\"\n print(f\"\\nError: {e}\")\n print(\"\\nTroubleshooting:\")\n print(\"- Check that protein IDs are UniProt accessions (e.g., P12345)\")\n print(\"- Try setting Annotations to 'None' to skip annotation fetching\")\n\n\ngen_btn.on_click(_on_gen)\n\ndisplay(VBox([\n HTML(\"<h4>Methods</h4>\"),\n HBox(list(method_toggles.values()), layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\"}),\n no_params,\n param_grid,\n HTML(\"<h4>Annotations</h4>\"),\n HBox([ann_preset, csv_upload], layout={\"gap\": \"10px\", \"align_items\": \"center\"}),\n HTML(\"<p><i>Optional: upload a CSV/TSV with custom annotations (first column = protein IDs). \"\n \"Taxonomy requires a one-time database download (~1 min).</i></p>\"),\n custom_ann,\n gen_btn,\n HTML(\"<p><i>Processing time depends on dataset size and selected methods.</i></p>\"),\n gen_out,\n]))"
48
+ "source": "# @title 3. Generate & Download {display-mode: \"form\"}\n# @markdown Configure methods and annotations, then click Generate.\n# @markdown Processing time depends on dataset size and selected methods.\n\nfrom ipywidgets import (\n HTML,\n Button,\n Checkbox,\n Dropdown,\n FileUpload,\n FloatSlider,\n GridBox,\n HBox,\n IntSlider,\n Output,\n ToggleButton,\n VBox,\n)\n\nMETHODS = [\"PCA\", \"UMAP\", \"t-SNE\", \"MDS\", \"PaCMAP\", \"LocalMAP\"]\nANNOTATIONS = {\n \"UniProt\": [\"annotation_score\", \"cc_subcellular_location\", \"ec\", \"fragment\", \"go_bp\", \"go_cc\", \"go_mf\", \"keyword\", \"length\", \"protein_existence\", \"protein_families\", \"reviewed\", \"xref_pdb\"],\n \"InterPro\": [\"cath\", \"cdd\", \"panther\", \"pfam\", \"prints\", \"prosite\", \"signal_peptide\", \"smart\", \"superfamily\"],\n \"Taxonomy\": [\"root\", \"domain\", \"kingdom\", \"phylum\", \"class\", \"order\", \"family\", \"genus\", \"species\"],\n}\nANNOTATION_DEFAULTS = {\"ec\", \"keyword\", \"length\", \"protein_families\", \"reviewed\"}\n\n# --- Methods ---\nmethod_toggles = {}\nfor m in METHODS:\n sel = m in {\"PCA\", \"UMAP\"}\n method_toggles[m] = ToggleButton(value=sel, description=m, button_style=\"info\" if sel else \"\", layout={\"width\": \"auto\", \"height\": \"32px\"})\n\n# --- DR Parameters ---\n_s = {\"description_width\": \"110px\"}\n_l = {\"width\": \"100%\"}\n_g = {\"border\": \"1px solid #666\", \"padding\": \"6px 10px\", \"margin\": \"2px\", \"flex\": \"1 1 220px\", \"overflow\": \"hidden\"}\n\nparam_groups = []\nw_nn = IntSlider(value=25, min=2, max=500, description=\"n_neighbors:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP / PaCMAP / LocalMAP</b>\"), w_nn], layout=_g), {\"UMAP\", \"PaCMAP\", \"LocalMAP\"}))\nw_md = FloatSlider(value=0.4, min=0.0, max=0.99, step=0.01, description=\"min_dist:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>UMAP</b>\"), w_md], layout=_g), {\"UMAP\"}))\nw_mn = FloatSlider(value=0.5, min=0.0, max=1.0, step=0.1, description=\"mn_ratio:\", style=_s, layout=_l)\nw_fp = FloatSlider(value=2.0, min=0.0, max=5.0, step=0.1, description=\"fp_ratio:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>PaCMAP / LocalMAP</b>\"), w_mn, w_fp], layout=_g), {\"PaCMAP\", \"LocalMAP\"}))\nw_pp = FloatSlider(value=30.0, min=5.0, max=500.0, step=1.0, description=\"perplexity:\", style=_s, layout=_l)\nw_lr = FloatSlider(value=200.0, min=1.0, max=1000.0, step=10.0, description=\"learning_rate:\", style=_s, layout=_l)\nparam_groups.append((VBox([HTML(\"<b>t-SNE</b>\"), w_pp, w_lr], layout=_g), {\"t-SNE\"}))\n\nno_params = HTML(\"<i>PCA/MDS have no adjustable parameters.</i>\")\npw = {\"n_neighbors\": w_nn, \"min_dist\": w_md, \"perplexity\": w_pp, \"learning_rate\": w_lr, \"mn_ratio\": w_mn, \"fp_ratio\": w_fp}\nparam_grid = HBox([g for g, _ in param_groups], layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\", \"width\": \"100%\"})\n\n\ndef _update_params(_c=None):\n sel = {m for m, t in method_toggles.items() if t.value}\n any_vis = False\n for gw, req in param_groups:\n vis = bool(req & sel)\n gw.layout.display = \"\" if vis else \"none\"\n if vis:\n any_vis = True\n no_params.layout.display = \"\" if not any_vis else \"none\"\n\n\nfor tb in method_toggles.values():\n def _toggle(c):\n c[\"owner\"].button_style = \"info\" if c[\"new\"] else \"\"\n _update_params()\n tb.observe(_toggle, names=\"value\")\n_update_params()\n\n# --- Annotations ---\nann_preset = Dropdown(\n options=[\"Default (recommended)\", \"All annotations\", \"Custom...\"],\n value=\"Default (recommended)\",\n description=\"Annotations:\",\n style={\"description_width\": \"initial\"},\n)\nann_cbs = {}\nann_cols = []\nfor cat, names in ANNOTATIONS.items():\n cbs = {n: Checkbox(value=n in ANNOTATION_DEFAULTS, description=n, indent=False, layout={\"width\": \"auto\"}) for n in names}\n ann_cbs[cat] = cbs\n grid = GridBox(list(cbs.values()), layout={\"grid_template_columns\": \"repeat(2, auto)\", \"grid_gap\": \"0 6px\"})\n ann_cols.append(VBox([HTML(f\"<b>{cat}</b>\"), grid], layout={\"flex\": \"1 1 200px\", \"border\": \"1px solid #666\", \"padding\": \"6px\", \"margin\": \"2px\"}))\ncustom_ann = HBox(ann_cols, layout={\"flex_flow\": \"row wrap\"})\ncustom_ann.layout.display = \"none\"\nann_preset.observe(lambda c: setattr(custom_ann.layout, \"display\", \"\" if c[\"new\"] == \"Custom...\" else \"none\"), names=\"value\")\n\n# --- Custom CSV upload ---\ncsv_upload = FileUpload(accept=\".csv,.tsv\", multiple=False, description=\"Choose CSV/TSV\")\n\n\ndef _get_ann():\n p = ann_preset.value\n if p == \"Default (recommended)\":\n return [\"default\"]\n if p == \"All annotations\":\n return [\"all\"]\n sel = [n for cbs in ann_cbs.values() for n, cb in cbs.items() if cb.value]\n return sel if sel else [\"default\"]\n\n\ndef _save_csv():\n \"\"\"Save uploaded CSV to disk and return the path, or None.\"\"\"\n if not csv_upload.value:\n return None\n fname = list(csv_upload.value.keys())[0]\n content = csv_upload.value[fname][\"content\"]\n Path(fname).write_bytes(content)\n return fname\n\n\n# --- Generate ---\ngen_btn = Button(description=\"Generate\", button_style=\"primary\", icon=\"play\", layout={\"width\": \"180px\", \"height\": \"40px\"})\ngen_out = Output()\n\n\ndef _on_gen(_b):\n with gen_out:\n clear_output()\n if not input_args:\n print(\"Select input data in cell above first.\")\n return\n method_specs = [m.lower().replace(\"-\", \"\") + \"2\" for m, t in method_toggles.items() if t.value]\n if not method_specs:\n print(\"Select at least one method.\")\n return\n\n ann = _get_ann() or []\n csv_path = _save_csv()\n if csv_path:\n ann.append(csv_path)\n if not ann:\n ann = None\n\n inp = input_args\n\n out_dir = Path(\"output\")\n out_dir.mkdir(exist_ok=True)\n cache_dir = out_dir / \"tmp\"\n cache_dir.mkdir(exist_ok=True)\n output_path = out_dir / \"data.parquetbundle\"\n\n step_html = HTML(value=\"<b>Step 1/4: Loading embeddings...</b>\")\n display(step_html)\n\n try:\n t0 = _time.time()\n embedding_sets = []\n\n if inp[\"type\"] == \"query\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n step_html.value = \"<b>Step 1/6: Downloading FASTA...</b>\"\n headers, fasta_path = query_uniprot(inp[\"query\"])\n if not headers:\n print(f\"No sequences found for query: {inp['query']}\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 2/6: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(fasta_path, emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = fasta_path\n embedding_sets.append(emb_set)\n elif inp[\"type\"] == \"fasta\":\n embs = _get_selected_embedders()\n if not embs:\n print(\"Select at least one embedder.\")\n return\n from protspace.data.embedding.biocentral import EmbedConfig\n for emb_name in embs:\n step_html.value = f\"<b>Step 1/5: Computing {emb_name} embeddings...</b>\"\n emb_set = embed_fasta(Path(inp[\"path\"]), emb_name, embed_config=EmbedConfig())\n emb_set.fasta_path = Path(inp[\"path\"])\n embedding_sets.append(emb_set)\n else:\n h5_path = Path(inp[\"path\"])\n name_override = inp.get(\"name\")\n emb_set = load_h5([h5_path], name_override=name_override)\n embedding_sets.append(emb_set)\n\n n_proteins = len(embedding_sets[0].headers)\n\n # Build pipeline with caching enabled\n reducer_params = ReducerParams(\n n_neighbors=pw[\"n_neighbors\"].value,\n min_dist=pw[\"min_dist\"].value,\n perplexity=pw[\"perplexity\"].value,\n learning_rate=pw[\"learning_rate\"].value,\n mn_ratio=pw[\"mn_ratio\"].value,\n fp_ratio=pw[\"fp_ratio\"].value,\n )\n config = PipelineConfig(\n methods=method_specs,\n output_path=output_path,\n bundled=True,\n keep_tmp=True,\n intermediate_dir=cache_dir,\n annotations=ann,\n reducer_params=reducer_params,\n )\n pipeline = ReductionPipeline(config)\n\n # Step 2: Annotations (cached after first run)\n step_html.value = \"<b>Step 2/4: Fetching annotations...</b>\"\n metadata = pipeline._fetch_annotations(embedding_sets[0].headers)\n\n # Step 3: Dimensionality reduction\n step_html.value = \"<b>Step 3/4: Reducing dimensions...</b>\"\n\n # Build full metadata\n all_headers = embedding_sets[0].headers\n full_metadata = _pd.DataFrame({\"identifier\": all_headers})\n if len(metadata.columns) > 1:\n metadata = metadata.astype(str)\n id_col = metadata.columns[0]\n if id_col != \"identifier\":\n metadata = metadata.rename(columns={id_col: \"identifier\"})\n full_metadata = full_metadata.merge(\n metadata.drop_duplicates(\"identifier\"),\n on=\"identifier\",\n how=\"left\",\n )\n metadata = full_metadata\n\n all_reductions = pipeline._run_reductions(embedding_sets)\n\n # Step 4: Bundle & download\n step_html.value = \"<b>Step 4/4: Bundling & downloading...</b>\"\n output = pipeline.base.create_output(metadata, all_reductions, all_headers)\n pipeline.base.save_output(output, output_path, bundled=True)\n\n kb = output_path.stat().st_size / 1024\n total_time = _time.time() - t0\n step_html.value = f\"<b>Done! ({kb:.0f} KB, {total_time:.0f}s)</b>\"\n print(f\"Processed {n_proteins} proteins with {len(method_specs)} method(s)\")\n print(f\"\\nUpload at: https://protspace.app/explore\")\n files.download(str(output_path))\n\n except Exception as e:\n step_html.value = \"<b>Failed</b>\"\n print(f\"\\nError: {e}\")\n print(\"\\nTroubleshooting:\")\n print(\"- Check that protein IDs are UniProt accessions (e.g., P12345)\")\n print(\"- Try setting Annotations to 'None' to skip annotation fetching\")\n\n\ngen_btn.on_click(_on_gen)\n\ndisplay(VBox([\n HTML(\"<h4>Methods</h4>\"),\n HBox(list(method_toggles.values()), layout={\"flex_flow\": \"row wrap\", \"gap\": \"6px\"}),\n no_params,\n param_grid,\n HTML(\"<h4>Annotations</h4>\"),\n HBox([ann_preset, csv_upload], layout={\"gap\": \"10px\", \"align_items\": \"center\"}),\n HTML(\"<p><i>Optional: upload a CSV/TSV with custom annotations (first column = protein IDs).</i></p>\"),\n custom_ann,\n gen_btn,\n HTML(\"<p><i>Processing time depends on dataset size and selected methods.</i></p>\"),\n gen_out,\n]))"
49
49
  },
50
50
  {
51
51
  "cell_type": "markdown",
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "protspace"
3
- version = "4.0.2"
3
+ version = "4.2.0"
4
4
  description = "A visualisation tool for protein embeddings from pLMs"
5
5
  authors = [{ name = "Tobias Senoner", email = "tobias.senoner@tum.de" }]
6
6
  readme = "README.md"
@@ -15,13 +15,12 @@ dependencies = [
15
15
  "numpy>=1.23.0",
16
16
  "pandas>=2.0.0",
17
17
  "tqdm>=4.67.1",
18
- "taxopy>=0.14.0",
19
18
  "pymmseqs>=1.0.4",
20
19
  "pyarrow>=20.0.0",
21
- "unipressed>=1.4.0",
22
20
  "biocentral-api>=1.1.2",
23
- "typer[all]>=0.15.0",
24
- "rich>=13.0.0",
21
+ "requests>=2.32.4",
22
+ "typer>=0.24.1",
23
+ "rich>=14.3.3",
25
24
  ]
26
25
 
27
26
  [project.optional-dependencies]
@@ -60,7 +59,6 @@ dev = [
60
59
  "seaborn>=0.13.2",
61
60
  "pytest>=8.3.3",
62
61
  "polars>=1.26.0",
63
- "taxopy>=0.14.0",
64
62
  "jupytext>=1.17.1",
65
63
  "pytest-cov>=7.0.0",
66
64
  ]
@@ -0,0 +1 @@
1
+ __version__ = "4.2.0"
@@ -0,0 +1,39 @@
1
+ """Shared HTTP utilities for UniProt-style REST API calls."""
2
+
3
+ import logging
4
+
5
+ import requests
6
+
7
+ logger = logging.getLogger(__name__)
8
+
9
+ API_TIMEOUT = 30
10
+
11
+
12
+ def paginated_get(
13
+ url: str,
14
+ params: dict | None = None,
15
+ timeout: int = API_TIMEOUT,
16
+ result_key: str = "results",
17
+ ) -> list[dict]:
18
+ """Fetch all pages from a UniProt-style REST API endpoint.
19
+
20
+ Follows Link headers with rel="next" for automatic pagination.
21
+ Returns the concatenated contents of the ``result_key`` array
22
+ across all pages.
23
+ """
24
+ results = []
25
+
26
+ while url:
27
+ resp = requests.get(url, params=params, timeout=timeout)
28
+ resp.raise_for_status()
29
+ data = resp.json()
30
+ results.extend(data.get(result_key, []))
31
+
32
+ # Follow Link header for next page
33
+ link = resp.headers.get("Link", "")
34
+ url = None
35
+ params = None # next-page URL already contains all params
36
+ if 'rel="next"' in link:
37
+ url = link.split(";")[0].strip(" <>")
38
+
39
+ return results
@@ -0,0 +1,124 @@
1
+ import logging
2
+ from typing import Any
3
+
4
+ from tqdm import tqdm
5
+
6
+ from protspace.data.annotations.retrievers.base_retriever import BaseAnnotationRetriever
7
+ from protspace.data.annotations.retrievers.http_utils import paginated_get
8
+
9
+ logger = logging.getLogger(__name__)
10
+
11
+ TAXONOMY_API_URL = "https://rest.uniprot.org/taxonomy/search"
12
+ _BATCH_SIZE = 100 # Max taxon IDs per request (URL length safety)
13
+
14
+ # Taxonomy annotations
15
+ TAXONOMY_ANNOTATIONS = [
16
+ "root",
17
+ "domain",
18
+ "kingdom",
19
+ "phylum",
20
+ "class",
21
+ "order",
22
+ "family",
23
+ "genus",
24
+ "species",
25
+ ]
26
+
27
+
28
+ class TaxonomyRetriever(BaseAnnotationRetriever):
29
+ """Retrieves taxonomy lineage data from the UniProt Taxonomy API."""
30
+
31
+ def __init__(self, taxon_ids: list[int], annotations: list = None):
32
+ # Don't call super().__init__() as we use taxon_ids instead of headers
33
+ self.taxon_ids = self._validate_taxon_ids(taxon_ids)
34
+ self.annotations = annotations
35
+
36
+ def fetch_annotations(self) -> dict[int, dict[str, Any]]:
37
+ result = {}
38
+
39
+ with tqdm(
40
+ total=len(self.taxon_ids),
41
+ desc="Fetching taxonomy annotations",
42
+ unit="taxon",
43
+ ) as pbar:
44
+ taxonomies_info = self._get_taxonomy_info(self.taxon_ids)
45
+
46
+ for taxon_id in self.taxon_ids:
47
+ if taxon_id in taxonomies_info:
48
+ result[taxon_id] = {"annotations": taxonomies_info[taxon_id]}
49
+ else:
50
+ result[taxon_id] = {
51
+ "annotations": dict.fromkeys(self.annotations, "")
52
+ }
53
+ pbar.update(1)
54
+
55
+ return result
56
+
57
+ def _validate_taxon_ids(self, taxon_ids: list[int]) -> list[int]:
58
+ for taxon_id in taxon_ids:
59
+ if not isinstance(taxon_id, int):
60
+ raise ValueError(f"Taxon ID {taxon_id} is not an integer")
61
+ return taxon_ids
62
+
63
+ def _get_taxonomy_info(self, taxon_ids: list[int]) -> dict[int, dict[str, str]]:
64
+ result = {}
65
+
66
+ # Fetch in batches to stay within URL length limits
67
+ for i in range(0, len(taxon_ids), _BATCH_SIZE):
68
+ batch = taxon_ids[i : i + _BATCH_SIZE]
69
+ batch_results = self._fetch_batch(batch)
70
+ result.update(batch_results)
71
+
72
+ return result
73
+
74
+ def _fetch_batch(self, taxon_ids: list[int]) -> dict[int, dict[str, str]]:
75
+ """Fetch taxonomy info for a batch of taxon IDs from UniProt Taxonomy API."""
76
+ result = {}
77
+ query = " OR ".join(f"id:{tid}" for tid in taxon_ids)
78
+
79
+ try:
80
+ entries = paginated_get(
81
+ TAXONOMY_API_URL,
82
+ params={"query": query, "format": "json", "size": "500"},
83
+ )
84
+ for entry in entries:
85
+ taxon_id = entry.get("taxonId")
86
+ if taxon_id is not None:
87
+ result[taxon_id] = self._extract_taxonomy(entry)
88
+
89
+ except Exception as e:
90
+ logger.error(f"Failed to fetch taxonomy batch: {e}")
91
+ for tid in taxon_ids:
92
+ if tid not in result:
93
+ result[tid] = dict.fromkeys(self.annotations, "")
94
+
95
+ return result
96
+
97
+ def _extract_taxonomy(self, entry: dict) -> dict[str, str]:
98
+ """Extract taxonomy ranks from a UniProt Taxonomy API result."""
99
+ lineage = entry.get("lineage", [])
100
+ rank_map = {item["rank"]: item["scientificName"] for item in lineage}
101
+
102
+ # The lineage only contains ancestors. If the taxon itself is a
103
+ # species (e.g., Human 9606), include it from the entry's own rank.
104
+ own_rank = entry.get("rank", "")
105
+ if own_rank and own_rank not in rank_map:
106
+ rank_map[own_rank] = entry.get("scientificName", "")
107
+
108
+ full_taxonomy_info = {
109
+ "root": rank_map.get("no rank", ""),
110
+ "domain": rank_map.get("domain", "") or rank_map.get("realm", ""),
111
+ "kingdom": rank_map.get("kingdom", ""),
112
+ "phylum": rank_map.get("phylum", ""),
113
+ "class": rank_map.get("class", ""),
114
+ "order": rank_map.get("order", ""),
115
+ "family": rank_map.get("family", ""),
116
+ "genus": rank_map.get("genus", ""),
117
+ "species": rank_map.get("species", ""),
118
+ }
119
+
120
+ # Filter based on requested taxonomy annotations
121
+ return {
122
+ annotation: full_taxonomy_info.get(annotation, "")
123
+ for annotation in self.annotations
124
+ }
@@ -4,15 +4,14 @@ UniProt annotation retriever.
4
4
  This module fetches protein annotations from the UniProt API.
5
5
  """
6
6
 
7
- import json
8
7
  import logging
9
8
  from collections import namedtuple
10
9
 
11
10
  import requests
12
11
  from tqdm import tqdm
13
- from unipressed import UniprotkbClient
14
12
 
15
13
  from protspace.data.annotations.retrievers.base_retriever import BaseAnnotationRetriever
14
+ from protspace.data.annotations.retrievers.http_utils import API_TIMEOUT, paginated_get
16
15
  from protspace.data.parsers.uniprot_parser import UniProtEntry
17
16
 
18
17
  logger = logging.getLogger(__name__)
@@ -42,15 +41,8 @@ UNIPROT_ANNOTATIONS = [
42
41
  ProteinAnnotations = namedtuple("ProteinAnnotations", ["identifier", "annotations"])
43
42
 
44
43
 
45
- _API_TIMEOUT = 30 # seconds per HTTP request
46
-
47
-
48
- def _fetch_one_with_timeout(accession: str, timeout: int = _API_TIMEOUT) -> dict:
49
- """Fetch a single UniProt entry with a timeout.
50
-
51
- The unipressed library's fetch_one() uses requests.get without a timeout,
52
- which can hang indefinitely. This wrapper adds timeout protection.
53
- """
44
+ def _fetch_one_with_timeout(accession: str, timeout: int = API_TIMEOUT) -> dict:
45
+ """Fetch a single UniProt entry by accession with timeout protection."""
54
46
  url = f"https://rest.uniprot.org/uniprotkb/{accession}.json"
55
47
  resp = requests.get(url, timeout=timeout)
56
48
  resp.raise_for_status()
@@ -58,15 +50,9 @@ def _fetch_one_with_timeout(accession: str, timeout: int = _API_TIMEOUT) -> dict
58
50
 
59
51
 
60
52
  def _fetch_uniparc_sequence(
61
- uniparc_id: str, timeout: int = _API_TIMEOUT
53
+ uniparc_id: str, timeout: int = API_TIMEOUT
62
54
  ) -> tuple[str, int]:
63
- """Fetch sequence and length from UniParc.
64
-
65
- Deleted UniProt entries still have their sequence archived in UniParc.
66
-
67
- Returns:
68
- Tuple of (sequence_string, length) or ("", 0) on failure.
69
- """
55
+ """Fetch sequence and length from UniParc for deleted entries."""
70
56
  url = f"https://rest.uniprot.org/uniparc/{uniparc_id}.json"
71
57
  try:
72
58
  resp = requests.get(url, timeout=timeout)
@@ -81,6 +67,22 @@ def _fetch_uniparc_sequence(
81
67
  return "", 0
82
68
 
83
69
 
70
+ def _fetch_many_accessions(accessions: list[str]) -> list[dict]:
71
+ """Fetch multiple UniProt entries by accession."""
72
+ return paginated_get(
73
+ "https://rest.uniprot.org/uniprotkb/accessions",
74
+ params={"accessions": ",".join(accessions)},
75
+ )
76
+
77
+
78
+ def _search_sec_acc(accession: str) -> list[dict]:
79
+ """Search UniProt by secondary accession (fallback for inactive entries)."""
80
+ return paginated_get(
81
+ "https://rest.uniprot.org/uniprotkb/search",
82
+ params={"query": f"sec_acc:{accession}", "format": "json", "size": "500"},
83
+ )
84
+
85
+
84
86
  class UniProtRetriever(BaseAnnotationRetriever):
85
87
  """Retrieves annotations from UniProt API."""
86
88
 
@@ -207,15 +209,7 @@ class UniProtRetriever(BaseAnnotationRetriever):
207
209
  except Exception:
208
210
  # fetch_one failed — fall back to sec_acc: search
209
211
  try:
210
- records = []
211
- for page in UniprotkbClient.search(
212
- query=f"sec_acc:{accession}", format="json"
213
- ).each_page():
214
- content = page.read() if hasattr(page, "read") else page
215
- parsed = (
216
- json.loads(content) if isinstance(content, str) else content
217
- )
218
- records.extend(parsed.get("results", []))
212
+ records = _search_sec_acc(accession)
219
213
  if records:
220
214
  entry = UniProtEntry(records[0])
221
215
  annotations_dict = self._extract_annotations(entry)
@@ -277,8 +271,7 @@ class UniProtRetriever(BaseAnnotationRetriever):
277
271
  batch = self.headers[i : i + batch_size]
278
272
 
279
273
  try:
280
- # Fetch records using unipressed
281
- records = UniprotkbClient.fetch_many(batch)
274
+ records = _fetch_many_accessions(batch)
282
275
 
283
276
  # Parse each record and track returned identifiers
284
277
  returned_ids = set()
@@ -50,7 +50,8 @@ xref_pdb - PDB cross-references (list)
50
50
  from typing import Any
51
51
 
52
52
  import pandas as pd
53
- from unipressed import UniprotkbClient
53
+
54
+ from protspace.data.annotations.retrievers.http_utils import paginated_get
54
55
 
55
56
  # ECO evidence code mapping (ECO ID → short human-readable code)
56
57
  ECO_TO_SHORT: dict[str, str] = {
@@ -586,9 +587,12 @@ def fetch_uniprot_data(
586
587
  f"Invalid properties: {invalid}. Available: {AVAILABLE_PROPERTIES}"
587
588
  )
588
589
 
589
- # Fetch records
590
- records = UniprotkbClient.fetch_many(accessions)
591
- entries = [UniProtEntry(record) for record in records]
590
+ # Fetch records via UniProt REST API
591
+ results = paginated_get(
592
+ "https://rest.uniprot.org/uniprotkb/accessions",
593
+ params={"accessions": ",".join(accessions)},
594
+ )
595
+ entries = [UniProtEntry(record) for record in results]
592
596
 
593
597
  # Extract properties
594
598
  data = []