protspace 4.2.0__tar.gz → 4.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {protspace-4.2.0 → protspace-4.3.0}/CHANGELOG.md +224 -0
- {protspace-4.2.0 → protspace-4.3.0}/CLAUDE.md +6 -3
- {protspace-4.2.0 → protspace-4.3.0}/PKG-INFO +1 -1
- {protspace-4.2.0 → protspace-4.3.0}/docs/annotations.md +69 -19
- {protspace-4.2.0 → protspace-4.3.0}/docs/cli.md +14 -6
- protspace-4.3.0/notebooks/ProtSpace_Preparation.ipynb +665 -0
- {protspace-4.2.0 → protspace-4.3.0}/pyproject.toml +1 -1
- protspace-4.3.0/src/protspace/__init__.py +1 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/prepare.py +165 -33
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/configuration.py +68 -42
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/manager.py +89 -13
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/merging.py +30 -25
- protspace-4.3.0/src/protspace/data/annotations/retrievers/biocentral_retriever.py +229 -0
- protspace-4.3.0/src/protspace/data/annotations/retrievers/cath_names.py +107 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/retrievers/interpro_retriever.py +25 -5
- protspace-4.3.0/src/protspace/data/annotations/retrievers/ted_retriever.py +103 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/retrievers/uniprot_retriever.py +49 -25
- protspace-4.3.0/src/protspace/data/annotations/transformers/interpro_transforms.py +141 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/transformers/transformer.py +9 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/embedding/biocentral.py +17 -6
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/loaders/similarity.py +41 -2
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/processors/pipeline.py +149 -14
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_annotation_manager.py +25 -11
- protspace-4.3.0/tests/test_biocentral_retriever.py +124 -0
- protspace-4.3.0/tests/test_cath_names.py +62 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_interpro_annotation_retriever.py +47 -43
- protspace-4.3.0/tests/test_pfam_clan.py +78 -0
- protspace-4.3.0/tests/test_ted_retriever.py +170 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_uniprot_annotation_retriever.py +3 -52
- {protspace-4.2.0 → protspace-4.3.0}/uv.lock +1 -1
- protspace-4.2.0/notebooks/ProtSpace_Preparation.ipynb +0 -83
- protspace-4.2.0/src/protspace/__init__.py +0 -1
- protspace-4.2.0/src/protspace/data/annotations/transformers/interpro_transforms.py +0 -65
- {protspace-4.2.0 → protspace-4.3.0}/.dockerignore +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/.env.example +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/.githooks/pre-commit +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/.github/workflows/ci.yml +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/.github/workflows/publish.yml +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/.github/workflows/release.yml +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/.gitignore +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/.python-version +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/Dockerfile +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/LICENSE +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/README.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/3FTx/3FTx.csv +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/3FTx/parquetbundle/3FTx_accession.csv +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/3FTx/parquetbundle/3ftx_with_db_styled.parquetbundle +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/3FTx/parquetbundle/styles.json +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/Pla2g2/Pla2g2.csv +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/Pla2g2/Pla2g2.fasta +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/Pla2g2/Pla2g2_pdb.zip +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/Pla2g2/protspace_files/Pla2g2.json +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/Pla2g2/protspace_files/Pla2g2_customized.json +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/Pla2g2/style.json +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/toxins/protspace/projections_data.parquet +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/toxins/protspace/projections_metadata.parquet +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/toxins/protspace/selected_annotations.parquet +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/toxins/toxins.json +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/data/toxins/toxins.parquetbundle +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/docs/protspace_example.png +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/docs/styling.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/notebooks/ClickThrough_GenerateEmbeddings.ipynb +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/count_h5_rows.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/count_parquetbundle_entries.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/download_clan_fastas.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/generate_examples/datasets.toml +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/generate_examples/generate.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/pfam_clans/README.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/pfam_clans/create_pfam_clan_query.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/scripts/pfam_clans/merge_h5_batches.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/app.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/annotated_image.png +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/custom.css +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/help_content/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/help_content/help_faq.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/help_content/help_how_it_works.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/help_content/help_json.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/help_content/help_overview.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/assets/rostlab_logo.png +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/annotate.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/app.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/bundle.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/common_options.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/embed.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/project.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/serve.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/cli/style.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/core/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/core/config.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/core/constants.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/retrievers/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/retrievers/base_retriever.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/retrievers/http_utils.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/retrievers/taxonomy_retriever.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/scores.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/transformers/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/annotations/transformers/uniprot_transforms.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/embedding/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/io/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/io/bundle.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/io/fasta.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/io/formatters.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/io/settings_converter.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/io/writers.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/loaders/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/loaders/embedding_set.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/loaders/fasta.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/loaders/h5.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/loaders/query.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/parsers/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/parsers/uniprot_parser.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/processors/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/data/processors/base_processor.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/main.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/ui/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/ui/callbacks.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/ui/layout.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/ui/styles.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/utils/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/utils/add_annotation_style.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/utils/arrow_reader.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/utils/constants.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/utils/reducers.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/visualization/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/visualization/molstar.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/visualization/plotting.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/src/protspace/wsgi.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/README.md +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/__init__.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_base_data_processor.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_biocentral_embedder.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_bundle_settings.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_config.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_config_validation.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_fasta.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_formatters.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_h5_parse_identifier.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_output_combinations.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_pipeline_utils.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_reducers.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_settings_converter.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_taxonomy_annotation_retriever.py +0 -0
- {protspace-4.2.0 → protspace-4.3.0}/tests/test_transformer.py +0 -0
|
@@ -1,6 +1,230 @@
|
|
|
1
1
|
# CHANGELOG
|
|
2
2
|
|
|
3
3
|
|
|
4
|
+
## v4.3.0 (2026-04-01)
|
|
5
|
+
|
|
6
|
+
### Continuous Integration
|
|
7
|
+
|
|
8
|
+
* ci: re-trigger release after corrupted merge event for PR #41
|
|
9
|
+
|
|
10
|
+
The merge commit (3725f87) landed on main but GitHub never processed
|
|
11
|
+
the push event due to a network issue, so CI, release, and issue
|
|
12
|
+
auto-close were all skipped. ([`6fae84d`](https://github.com/tsenoner/protspace/commit/6fae84dbbd65697e502936312208ea29a7173e27))
|
|
13
|
+
|
|
14
|
+
### Documentation
|
|
15
|
+
|
|
16
|
+
* docs: cache FASTA and embeddings in Colab notebook
|
|
17
|
+
|
|
18
|
+
Pass embedding_cache to embed_fasta() and cache UniProt query FASTA
|
|
19
|
+
to output/tmp/ so re-runs skip expensive API calls. Also clean up
|
|
20
|
+
unused imports flagged by ruff.
|
|
21
|
+
|
|
22
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`7e17236`](https://github.com/tsenoner/protspace/commit/7e17236912bc73227fe8ef6f05dbcf6b7d2cea4e))
|
|
23
|
+
|
|
24
|
+
* docs: update CLI caching section, test table, and UniProt ID handling
|
|
25
|
+
|
|
26
|
+
- Expand docs/cli.md caching section to document all 5 cached items
|
|
27
|
+
(FASTA, embeddings, annotations, similarity, DR projections)
|
|
28
|
+
- Add 3 new test files to CLAUDE.md test table (pfam_clan, ted, biocentral)
|
|
29
|
+
- Update UniProt ID handling docs: identifiers must be bare accessions
|
|
30
|
+
- Update test counts (uniprot_retriever: 29→24 after _manage_headers removal)
|
|
31
|
+
|
|
32
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`d4a4a14`](https://github.com/tsenoner/protspace/commit/d4a4a1481d74f22128c00bc5ece60d714b4c08a7))
|
|
33
|
+
|
|
34
|
+
* docs: update annotations.md with all five data sources and groups
|
|
35
|
+
|
|
36
|
+
Fix header (three → five sources), add TED and Biocentral rows to
|
|
37
|
+
summary table, update InterPro count (9 → 10 for pfam_clan), add
|
|
38
|
+
ted and biocentral to group presets table, add CLI example.
|
|
39
|
+
|
|
40
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`8fe8f82`](https://github.com/tsenoner/protspace/commit/8fe8f82ede0223cdbfa54f04ca1a8948dc8f15b3))
|
|
41
|
+
|
|
42
|
+
* docs: update Colab notebook with new annotation sources
|
|
43
|
+
|
|
44
|
+
Add pfam_clan, TED Domains, and Biocentral prediction annotations
|
|
45
|
+
to the ANNOTATIONS dict in the preparation notebook.
|
|
46
|
+
|
|
47
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`d667d93`](https://github.com/tsenoner/protspace/commit/d667d937499e78899d3b8a49d39274f315764ddf))
|
|
48
|
+
|
|
49
|
+
### Features
|
|
50
|
+
|
|
51
|
+
* feat: replace --force-refetch with granular --refetch <stages>
|
|
52
|
+
|
|
53
|
+
Replace the all-or-nothing --force-refetch boolean with --refetch
|
|
54
|
+
accepting comma-separated stage names for selective cache invalidation:
|
|
55
|
+
query, embed, similarity, projections, uniprot, taxonomy, interpro,
|
|
56
|
+
ted, biocentral. Shorthands: all, annotations.
|
|
57
|
+
|
|
58
|
+
Also fixes a bug where --force-refetch skipped TED and Biocentral
|
|
59
|
+
annotations, and suppresses the biocentral API length warning.
|
|
60
|
+
|
|
61
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`7dadeff`](https://github.com/tsenoner/protspace/commit/7dadeffec3e0b998befd2b2b3de3af901967ab00))
|
|
62
|
+
|
|
63
|
+
* feat: use official CATH names file for all-level code resolution
|
|
64
|
+
|
|
65
|
+
Replace InterPro XML-based CATH name lookup with the authoritative
|
|
66
|
+
cath-names-v4_4_0.txt file (393 KB vs 90 MB). This provides names at
|
|
67
|
+
all 4 CATH hierarchy levels (Class, Architecture, Topology, Superfamily),
|
|
68
|
+
fixing resolution of partial codes like 2.60.40 from the AlphaFold API.
|
|
69
|
+
|
|
70
|
+
- New shared module: cath_names.py (download, cache 30 days, parse)
|
|
71
|
+
- TED retriever: direct lookup at any level, no G3DSA: prefix needed
|
|
72
|
+
- InterPro retriever: CATH names from CATH file, SSF/PANTHER still from
|
|
73
|
+
InterPro XML
|
|
74
|
+
- Unnamed superfamilies inherit parent topology name as fallback
|
|
75
|
+
|
|
76
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`9b906db`](https://github.com/tsenoner/protspace/commit/9b906db1ad42b177140a8a19ec3af1d7e404deea))
|
|
77
|
+
|
|
78
|
+
* feat: cache FASTA downloads, MMseqs2 similarity, and DR projections
|
|
79
|
+
|
|
80
|
+
When keep_tmp is active (default), all intermediate results are now
|
|
81
|
+
cached under {output}/tmp/ and reused on subsequent runs:
|
|
82
|
+
|
|
83
|
+
- FASTA: skip re-download if tmp/sequences.fasta exists
|
|
84
|
+
- Similarity: save/load similarity_matrix.npy + similarity_headers.npy
|
|
85
|
+
- DR projections: save/load .npz files keyed by (embedding, method,
|
|
86
|
+
dims, params_hash) so different parameters produce separate caches
|
|
87
|
+
|
|
88
|
+
All caches are bypassed with --force-refetch (help text updated to
|
|
89
|
+
reflect its broader scope). Cache hits log a WARNING for visibility.
|
|
90
|
+
|
|
91
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`038e502`](https://github.com/tsenoner/protspace/commit/038e5025dfa1be368da0c9b8f9fc96c5585334cd))
|
|
92
|
+
|
|
93
|
+
* feat: add Biocentral prediction annotations (subcellular location, membrane, signal peptide, transmembrane)
|
|
94
|
+
|
|
95
|
+
Fetch per-protein predictions from the Biocentral API:
|
|
96
|
+
- predicted_subcellular_location (LightAttention, 10 classes)
|
|
97
|
+
- predicted_membrane (LightAttention, Membrane/Soluble)
|
|
98
|
+
- predicted_signal_peptide (TMbed-derived, True/False)
|
|
99
|
+
- predicted_transmembrane (TMbed-derived, none/alpha-helical/beta-barrel)
|
|
100
|
+
|
|
101
|
+
Closes #40
|
|
102
|
+
|
|
103
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`8a4609a`](https://github.com/tsenoner/protspace/commit/8a4609a72f2d3129643eef3737e83fd430d3179b))
|
|
104
|
+
|
|
105
|
+
* feat: add TED domain annotations via AlphaFold Database API
|
|
106
|
+
|
|
107
|
+
Query alphafold.ebi.ac.uk/api/domains/{acc} per protein to get TED
|
|
108
|
+
(The Encyclopedia of Domains) structural domain annotations. Resolves
|
|
109
|
+
CATH superfamily codes to names using the existing InterPro CATH-Gene3D
|
|
110
|
+
name map.
|
|
111
|
+
|
|
112
|
+
Output format: "2.60.40.720 (Immunoglobulin-like)|95.1;3.40.50.300|88.3"
|
|
113
|
+
|
|
114
|
+
Closes #22
|
|
115
|
+
|
|
116
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`91fc68d`](https://github.com/tsenoner/protspace/commit/91fc68d087d9756422ffab304ea0376d3dda956f))
|
|
117
|
+
|
|
118
|
+
* feat: add pfam_clan annotation — maps Pfam families to CLANS
|
|
119
|
+
|
|
120
|
+
Downloads Pfam-A.clans.tsv from EBI FTP (cached 30 days), maps Pfam
|
|
121
|
+
accessions from InterPro annotations to clan IDs with names.
|
|
122
|
+
Output format: "CL0023 (P-loop_NTPase);CL0192 (HAD)"
|
|
123
|
+
|
|
124
|
+
Closes #38
|
|
125
|
+
|
|
126
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`44a1774`](https://github.com/tsenoner/protspace/commit/44a17748712be37e8b1f2341c76146e18361681e))
|
|
127
|
+
|
|
128
|
+
### Fixes
|
|
129
|
+
|
|
130
|
+
* fix: use CATH latest-release URL instead of hardcoded v4_4_0
|
|
131
|
+
|
|
132
|
+
The latest-release/ path is a stable alias that always points to the
|
|
133
|
+
current CATH release, so we automatically pick up new versions.
|
|
134
|
+
|
|
135
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`bbb2dbf`](https://github.com/tsenoner/protspace/commit/bbb2dbf4c0d8a887ff469823a1b755f54a9fd3ab))
|
|
136
|
+
|
|
137
|
+
* fix: consolidate repetitive cache-hit messages into compact summaries
|
|
138
|
+
|
|
139
|
+
Group per-item cache warnings (projections, embeddings) into single
|
|
140
|
+
summary lines, remove repeated --force-refetch hints in favor of one
|
|
141
|
+
at the end, and demote verbose per-item logs to INFO level.
|
|
142
|
+
|
|
143
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`9245e8d`](https://github.com/tsenoner/protspace/commit/9245e8dcbafc694d0128d114ebb015fdb52a711f))
|
|
144
|
+
|
|
145
|
+
* fix: warn when using cached annotations and when cache is all empty
|
|
146
|
+
|
|
147
|
+
Change cache-hit message from INFO (only visible with -v) to WARNING
|
|
148
|
+
so users always know when cached data is being used. Also detect and
|
|
149
|
+
warn about all-empty cached annotations with actionable advice
|
|
150
|
+
(--force-refetch or -f).
|
|
151
|
+
|
|
152
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`64e3cbd`](https://github.com/tsenoner/protspace/commit/64e3cbd990d2a8011eda8200670fdc0f4771ee28))
|
|
153
|
+
|
|
154
|
+
* fix: simplify UniProt ID handling and document annotation input requirements
|
|
155
|
+
|
|
156
|
+
Remove _manage_headers() — identifiers must be valid UniProt accessions
|
|
157
|
+
directly. Non-matching IDs are skipped with a clear warning that
|
|
158
|
+
distinguishes accession-dependent (UniProt, Taxonomy, TED) from
|
|
159
|
+
sequence-dependent (InterPro, Biocentral) annotations.
|
|
160
|
+
|
|
161
|
+
Also:
|
|
162
|
+
- Fix _add_required_annotations() to include 'sequence' for Biocentral
|
|
163
|
+
- Simplify _build_sequence_map() (no reverse mapping needed)
|
|
164
|
+
- Document input requirements in docs/annotations.md
|
|
165
|
+
|
|
166
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`65ab8ac`](https://github.com/tsenoner/protspace/commit/65ab8ac6436f011e3d566b2deaa2c54fd9876e00))
|
|
167
|
+
|
|
168
|
+
* fix: improve annotation validation error with suggestions and group list
|
|
169
|
+
|
|
170
|
+
Show fuzzy-matched suggestions (via difflib), list available groups,
|
|
171
|
+
and link to online annotation reference. Example output:
|
|
172
|
+
|
|
173
|
+
Unknown annotation 'biocentra'. Did you mean: biocentral?
|
|
174
|
+
Groups: all, biocentral, default, interpro, taxonomy, ted, uniprot
|
|
175
|
+
See https://github.com/tsenoner/protspace/blob/main/docs/annotations.md
|
|
176
|
+
|
|
177
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`27fb824`](https://github.com/tsenoner/protspace/commit/27fb82484314bac9ce40f1ad4081ceed21397d2c))
|
|
178
|
+
|
|
179
|
+
* fix: attach -f FASTA path to H5 embedding sets for sequence reuse
|
|
180
|
+
|
|
181
|
+
When user provides H5 embeddings with -f fasta.fasta, store the FASTA
|
|
182
|
+
path on the EmbeddingSet so sequences are available for Biocentral
|
|
183
|
+
predictions and InterPro without needing UniProt accessions.
|
|
184
|
+
|
|
185
|
+
Also improve the warning message when no sequences are available.
|
|
186
|
+
|
|
187
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`1c0bafd`](https://github.com/tsenoner/protspace/commit/1c0bafdc83cabaf110d664e42ce987fd30c067cf))
|
|
188
|
+
|
|
189
|
+
* fix: use cache dir for MMseqs2 temp files instead of system temp
|
|
190
|
+
|
|
191
|
+
Pass cache_dir from CLI to compute_similarity() so MMseqs2 temp files
|
|
192
|
+
are stored alongside other cached data. Falls back to system temp
|
|
193
|
+
when no cache dir is available. Only cleans up temp files when using
|
|
194
|
+
system temp (cache dir is preserved for reuse).
|
|
195
|
+
|
|
196
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`dd9b273`](https://github.com/tsenoner/protspace/commit/dd9b27346ee6bc75452bcc49ee12f3047381550d))
|
|
197
|
+
|
|
198
|
+
* fix: pass FASTA sequences through pipeline and deduplicate for Biocentral
|
|
199
|
+
|
|
200
|
+
- Extract sequences from EmbeddingSet.fasta_path in the pipeline and
|
|
201
|
+
pass them to ProteinAnnotationManager, avoiding redundant UniProt
|
|
202
|
+
sequence re-fetches for FASTA/Query input modes
|
|
203
|
+
- Merge local sequences (priority) with UniProt sequences (fallback)
|
|
204
|
+
in both _fetch_interpro() and _fetch_biocentral()
|
|
205
|
+
- Deduplicate sequences before sending to Biocentral API (rejects
|
|
206
|
+
duplicate sequences) and map predictions back to all headers sharing
|
|
207
|
+
the same sequence
|
|
208
|
+
|
|
209
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`b778a7e`](https://github.com/tsenoner/protspace/commit/b778a7ecd1eacccdfcb8a57244f262b31b149b0f))
|
|
210
|
+
|
|
211
|
+
* fix: add tests for new annotations and update CLI help text
|
|
212
|
+
|
|
213
|
+
- Add 7 unit tests for Pfam CLAN transformer (mapping, dedup, edge cases)
|
|
214
|
+
- Add 7 unit tests for TED retriever (mocked AlphaFold API, CATH names)
|
|
215
|
+
- Add 14 unit tests for Biocentral retriever (TMbed parsing, per-sequence)
|
|
216
|
+
- Update CLI help text to include ted and biocentral groups
|
|
217
|
+
- Update annotations.md overview with all five sources and group presets
|
|
218
|
+
|
|
219
|
+
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> ([`6506c83`](https://github.com/tsenoner/protspace/commit/6506c83caae225d8b2dc7a2384ff491b0ed1357f))
|
|
220
|
+
|
|
221
|
+
### Unknown
|
|
222
|
+
|
|
223
|
+
* Merge pull request #41 from tsenoner/feat/extend-annotations
|
|
224
|
+
|
|
225
|
+
feat: extend annotations, improve caching, and fix sequence handling ([`3725f87`](https://github.com/tsenoner/protspace/commit/3725f87a00845dd0fdcfbb7c05c83e161fd6e715))
|
|
226
|
+
|
|
227
|
+
|
|
4
228
|
## v4.2.0 (2026-03-28)
|
|
5
229
|
|
|
6
230
|
### Features
|
|
@@ -153,7 +153,7 @@ Six methods supported, all in `src/protspace/utils/reducers.py`:
|
|
|
153
153
|
|
|
154
154
|
- **Float16 upcast:** HDF5 embeddings (often float16 from pLMs) are upcast to float32 in `data/loaders/h5.py:load_h5()` to prevent matrix overflow. A safety-net upcast also exists in `base_processor.py`.
|
|
155
155
|
- **HDF5 loading:** `load_h5()` in `data/loaders/h5.py` handles both flat and grouped HDF5 layouts, validates embedding dimensions are consistent, and rejects per-residue embeddings with a clear error message.
|
|
156
|
-
- **
|
|
156
|
+
- **UniProt ID validation:** `uniprot_retriever.py` pre-filters identifiers with a UniProt accession regex — non-matching IDs (e.g., `NCBI|...`, `sp|P12345|NAME`) are skipped with a summary warning. Identifiers must be bare accessions (e.g., `P12345`, `A0A2P1BSS8`). Inactive entries are resolved via `fetch_one()` (returns merged target or inactive reason + UniParc ID). Deleted entries recover their sequence from UniParc.
|
|
157
157
|
- **EC name resolution:** `uniprot_transforms.py` appends enzyme names to EC numbers using the ExPASy ENZYME database (`enzyme.dat` for fully specified ECs, `enzclass.txt` for partial ECs like `3.4.-.-`). Both files are downloaded and cached together in `~/.cache/protspace/enzyme/` with a 7-day TTL.
|
|
158
158
|
- **Warning suppression:** `base_processor.py` suppresses harmless sklearn RuntimeWarnings (randomized SVD overflow) and umap/pacmap UserWarnings during `fit_transform`.
|
|
159
159
|
- **Annoy fallback:** `reducers.py` includes a lazy annoy health check. On platforms where annoy is broken (e.g., macOS ARM64 segfaults), it monkey-patches pacmap to use sklearn `NearestNeighbors` instead. The check only runs when PaCMAP/LocalMAP are first used.
|
|
@@ -200,14 +200,17 @@ uv run pytest tests/ --cov=src/protspace # With coverage
|
|
|
200
200
|
| `test_reducers.py` | 51 | All 6 DR methods: shapes, finite output, float16, config validation |
|
|
201
201
|
| `test_interpro_annotation_retriever.py` | 46 | InterPro API mocking, parsing |
|
|
202
202
|
| `test_settings_converter.py` | 31 | Settings table ↔ visualization state conversion |
|
|
203
|
-
| `test_uniprot_annotation_retriever.py` |
|
|
203
|
+
| `test_uniprot_annotation_retriever.py` | 24 | UniProt API mocking, inactive entry resolution |
|
|
204
204
|
| `test_pipeline_utils.py` | 28 | ReductionPipeline, EmbeddingSet, method parsing |
|
|
205
205
|
| `test_biocentral_embedder.py` | 23 | Biocentral API client, embedding flow |
|
|
206
206
|
| `test_fasta.py` | 17 | FASTA parsing, edge cases, CSV annotation loading |
|
|
207
|
-
| `
|
|
207
|
+
| `test_biocentral_retriever.py` | 14 | Biocentral prediction retriever (TMbed parsing, per-sequence) |
|
|
208
208
|
| `test_taxonomy_annotation_retriever.py` | 15 | Taxonomy via UniProt Taxonomy API (mocked + integration) |
|
|
209
|
+
| `test_config_validation.py` | 12 | DimensionReductionConfig parameter validation |
|
|
209
210
|
| `test_h5_parse_identifier.py` | 9 | HDF5 key parsing, identifier extraction |
|
|
210
211
|
| `test_base_data_processor.py` | 8 | BaseProcessor: reduction, output creation, save |
|
|
212
|
+
| `test_ted_retriever.py` | 7 | TED domain retriever (mocked AlphaFold API, CATH names) |
|
|
213
|
+
| `test_pfam_clan.py` | 7 | Pfam CLAN transformer (mapping, dedup, edge cases) |
|
|
211
214
|
| `test_formatters.py` | 5 | ProteinAnnotations → DataFrame formatting |
|
|
212
215
|
| `test_output_combinations.py` | 4 | Output format flag combinations |
|
|
213
216
|
| `test_bundle_settings.py` | 4 | Parquetbundle settings read/write |
|
|
@@ -1,26 +1,41 @@
|
|
|
1
1
|
# Annotation Reference
|
|
2
2
|
|
|
3
|
-
ProtSpace retrieves annotations from
|
|
3
|
+
ProtSpace retrieves annotations from five data sources: **UniProt**, **InterPro**, **Taxonomy**, **TED Domains**, and **Biocentral Predictions**. Select annotations with `-a` in `protspace prepare`.
|
|
4
4
|
|
|
5
5
|
## Available Annotations
|
|
6
6
|
|
|
7
|
-
| Source
|
|
8
|
-
|
|
|
9
|
-
| **UniProt** (14)
|
|
10
|
-
| **InterPro** (
|
|
11
|
-
| **Taxonomy** (9)
|
|
7
|
+
| Source | Annotations |
|
|
8
|
+
| -------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
9
|
+
| **UniProt** (14) | `annotation_score`, `cc_subcellular_location`, `ec`, `fragment`, `gene_name`, `go_bp`, `go_cc`, `go_mf`, `keyword`, `length`, `protein_existence`, `protein_families`, `reviewed`, `xref_pdb` |
|
|
10
|
+
| **InterPro** (10) | `cath`, `cdd`, `panther`, `pfam`, `pfam_clan`, `prints`, `prosite`, `signal_peptide`, `smart`, `superfamily` |
|
|
11
|
+
| **Taxonomy** (9) | `root`, `domain`, `kingdom`, `phylum`, `class`, `order`, `family`, `genus`, `species` |
|
|
12
|
+
| **TED** (1) | `ted_domains` |
|
|
13
|
+
| **Biocentral** (4) | `predicted_subcellular_location`, `predicted_membrane`, `predicted_signal_peptide`, `predicted_transmembrane` |
|
|
12
14
|
|
|
13
15
|
_Always included_: `gene_name`, `protein_name`, `uniprot_kb_id` (fetched regardless of selection).
|
|
14
16
|
|
|
17
|
+
### Input Requirements
|
|
18
|
+
|
|
19
|
+
Annotation sources have different requirements for protein identifiers:
|
|
20
|
+
|
|
21
|
+
| Requirement | Sources | Works with `-f` FASTA? |
|
|
22
|
+
| ----------- | ------- | ---------------------- |
|
|
23
|
+
| **UniProt accession** | UniProt, Taxonomy, TED | No — accession needed |
|
|
24
|
+
| **Protein sequence** | InterPro, Biocentral, Pfam CLANS | Yes — provide `-f` |
|
|
25
|
+
|
|
26
|
+
If your H5 keys are not valid UniProt accessions (e.g., `NCBI|...`, custom IDs), accession-dependent annotations will be empty. Sequence-dependent annotations can still work if you provide the original FASTA file with `-f`.
|
|
27
|
+
|
|
15
28
|
## Group Presets
|
|
16
29
|
|
|
17
|
-
| Group
|
|
18
|
-
|
|
|
19
|
-
| `default`
|
|
20
|
-
| `all`
|
|
21
|
-
| `uniprot`
|
|
22
|
-
| `interpro`
|
|
23
|
-
| `taxonomy`
|
|
30
|
+
| Group | Contents |
|
|
31
|
+
| ------------ | --------------------------------------------------------------------------- |
|
|
32
|
+
| `default` | `ec`, `keyword`, `length`, `protein_families`, `reviewed` |
|
|
33
|
+
| `all` | All annotations from all sources |
|
|
34
|
+
| `uniprot` | All UniProt annotations |
|
|
35
|
+
| `interpro` | All InterPro annotations (incl. `pfam_clan`) |
|
|
36
|
+
| `taxonomy` | All taxonomy annotations |
|
|
37
|
+
| `ted` | All TED domain annotations |
|
|
38
|
+
| `biocentral` | All Biocentral prediction annotations |
|
|
24
39
|
|
|
25
40
|
Groups are mixable with individual names. If `-a` is omitted, `default` is used.
|
|
26
41
|
|
|
@@ -28,6 +43,7 @@ Groups are mixable with individual names. If `-a` is omitted, `default` is used.
|
|
|
28
43
|
protspace prepare -i data.h5:prot_t5 # default group
|
|
29
44
|
protspace prepare -i data.h5:prot_t5 -a all # everything
|
|
30
45
|
protspace prepare -i data.h5:prot_t5 -a default,interpro,kingdom
|
|
46
|
+
protspace prepare -i data.h5:prot_t5 -a ted,biocentral # all predictions
|
|
31
47
|
protspace prepare -i data.h5:prot_t5 -a pfam,cath,reviewed
|
|
32
48
|
protspace prepare -q "..." -e prot_t5 -a interpro,kingdom
|
|
33
49
|
```
|
|
@@ -120,16 +136,50 @@ Codes are derived from [ECO (Evidence & Conclusion Ontology)](https://www.eviden
|
|
|
120
136
|
|
|
121
137
|
Three databases (`cath`, `superfamily`, `panther`) resolve human-readable entry names via InterPro FTP XML (cached 7 days). `cath` has `G3DSA:` prefix removed; `signal_peptide` is converted to `True`/`False`.
|
|
122
138
|
|
|
139
|
+
### Derived Annotation
|
|
140
|
+
|
|
141
|
+
| Name | Source | Description |
|
|
142
|
+
| ----------- | ------ | --------------------------------------------------------- |
|
|
143
|
+
| `pfam_clan` | Pfam | Maps Pfam families to CLANS (higher-level groupings) |
|
|
144
|
+
|
|
145
|
+
**Output format**: `CL0023 (P-loop_NTPase);CL0192 (HAD)` — semicolon-separated unique clan IDs with names. Requires `pfam` (fetched automatically). Clan mapping from [Pfam FTP](https://ftp.ebi.ac.uk/pub/databases/Pfam/current_release/Pfam-A.clans.tsv.gz) (cached 30 days).
|
|
146
|
+
|
|
147
|
+
## TED Domain Annotations
|
|
148
|
+
|
|
149
|
+
Structure-based domain annotations from [TED (The Encyclopedia of Domains)](https://ted.cathdb.info/) via the [AlphaFold Database API](https://alphafold.ebi.ac.uk/):
|
|
150
|
+
|
|
151
|
+
| Name | Description |
|
|
152
|
+
| ------------- | ---------------------------------------------------------- |
|
|
153
|
+
| `ted_domains` | Structural domains with CATH classification and confidence |
|
|
154
|
+
|
|
155
|
+
**Output format**: `2.60.40.720 (Immunoglobulin-like)|95.1;3.40.50.300|88.3` — semicolon-separated domains. Each domain has a CATH superfamily code, name (when available, resolved from InterPro CATH-Gene3D cache), and pLDDT confidence score. Unclassified domains show as `unclassified|{plddt}`.
|
|
156
|
+
|
|
157
|
+
**Data source**: Per-protein lookup via `alphafold.ebi.ac.uk/api/domains/{accession}`. Domains are predicted from AlphaFold structures using a consensus of Chainsaw, Merizo, and UniDoc methods.
|
|
158
|
+
|
|
123
159
|
## Taxonomy Annotations
|
|
124
160
|
|
|
125
|
-
9 taxonomic ranks resolved via
|
|
161
|
+
9 taxonomic ranks resolved via the [UniProt Taxonomy API](https://rest.uniprot.org/taxonomy/search): `root`, `domain`, `kingdom`, `phylum`, `class`, `order`, `family`, `genus`, `species`. `root` is the cellular/acellular classification; `domain` is the top-level biological domain (e.g. Bacteria, Archaea, Eukaryota). Requires `organism_id` from UniProt (fetched automatically).
|
|
162
|
+
|
|
163
|
+
## Biocentral Prediction Annotations
|
|
164
|
+
|
|
165
|
+
Per-protein predictions from the [Biocentral API](https://biocentral.rostlab.org/) using pre-trained models. Requires protein sequences (fetched automatically from UniProt).
|
|
166
|
+
|
|
167
|
+
| Name | Model | Description |
|
|
168
|
+
| -------------------------------- | ------------------------------------ | ------------------------------------------ |
|
|
169
|
+
| `predicted_subcellular_location` | LightAttention | 10-class subcellular localization |
|
|
170
|
+
| `predicted_membrane` | LightAttention | Membrane / Soluble |
|
|
171
|
+
| `predicted_signal_peptide` | TMbed | True / False (derived from topology) |
|
|
172
|
+
| `predicted_transmembrane` | TMbed | none / alpha-helical / beta-barrel |
|
|
173
|
+
|
|
174
|
+
**Data source**: Batch predictions via Biocentral API (`api.predict()`). TMbed provides per-residue topology labels (`H`=TM helix, `B`=TM beta strand, `S`=signal peptide); signal peptide and transmembrane type are summarized from these labels.
|
|
126
175
|
|
|
127
176
|
## Caching
|
|
128
177
|
|
|
129
|
-
| Cache | Location
|
|
130
|
-
| -------------- |
|
|
131
|
-
|
|
|
132
|
-
| InterPro names | `~/.cache/protspace/interpro/`
|
|
133
|
-
| EC names | `~/.cache/protspace/enzyme/`
|
|
178
|
+
| Cache | Location | Max Age | Purpose |
|
|
179
|
+
| -------------- | --------------------------------- | -------- | ------------------------------------------------- |
|
|
180
|
+
| CATH names | `~/.cache/protspace/cath/` | 30 days | CATH hierarchy names (all levels) for TED and cath |
|
|
181
|
+
| InterPro names | `~/.cache/protspace/interpro/` | 7 days | Domain entry names for superfamily, panther |
|
|
182
|
+
| EC names | `~/.cache/protspace/enzyme/` | 7 days | Enzyme descriptions from ExPASy |
|
|
183
|
+
| Pfam clans | `~/.cache/protspace/pfam_clans/` | 30 days | Pfam family → clan mapping |
|
|
134
184
|
|
|
135
185
|
The `default` group only requires the UniProt REST API (+ ExPASy for EC names). For `--keep-tmp` annotation caching, see [CLI Reference](cli.md#annotation-caching---keep-tmp).
|
|
@@ -83,7 +83,7 @@ protspace prepare -i external.h5:prot_t5 -m pca2 -o output
|
|
|
83
83
|
| ---- | ----------- | ------- |
|
|
84
84
|
| `-a, --annotations` | Annotation sources: groups, individual names, or a CSV/TSV file path. See [Annotation Reference](annotations.md). | `default` |
|
|
85
85
|
| `--scores / --no-scores` | Include annotation confidence scores. | on |
|
|
86
|
-
| `--
|
|
86
|
+
| `--refetch STAGES` | Recompute specific stages (comma-separated): query, embed, similarity, projections, uniprot, taxonomy, interpro, ted, biocentral. Shorthands: `all`, `annotations`. | off |
|
|
87
87
|
|
|
88
88
|
#### Output
|
|
89
89
|
|
|
@@ -170,12 +170,20 @@ protspace prepare -i embeddings/prot_t5.h5 -m pca2 -o output
|
|
|
170
170
|
|
|
171
171
|
Check if an HDF5 file has the attribute: `python -c "import h5py; print(dict(h5py.File('file.h5','r').attrs))"`
|
|
172
172
|
|
|
173
|
-
##
|
|
173
|
+
## Intermediate Caching (`--keep-tmp`)
|
|
174
174
|
|
|
175
|
-
With `--keep-tmp
|
|
175
|
+
With `--keep-tmp` (default), all intermediate results are cached in `{output}/tmp/` and reused on subsequent runs:
|
|
176
176
|
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
177
|
+
| Cached item | File | Reuse behavior |
|
|
178
|
+
| ----------- | ---- | -------------- |
|
|
179
|
+
| FASTA sequences | `sequences.fasta` | Skip UniProt query download |
|
|
180
|
+
| Embeddings | `{embedder}.h5` | Skip already-embedded proteins |
|
|
181
|
+
| Annotations | `all_annotations.parquet` | Fetch only missing annotation sources |
|
|
182
|
+
| Similarity matrix | `similarity_matrix.npy` | Skip MMseqs2 recomputation |
|
|
183
|
+
| DR projections | `proj_{name}_{method}_{hash}.npz` | Skip dimensionality reduction |
|
|
184
|
+
|
|
185
|
+
- Annotation cache always includes scores regardless of `--no-scores`
|
|
186
|
+
- DR projection caches are keyed by embedding name, method, dimensions, and all parameters — changing any parameter creates a new cache entry
|
|
187
|
+
- Use `--refetch all` to bypass all caches, or `--refetch <stages>` selectively (e.g., `--refetch ted,biocentral`)
|
|
180
188
|
|
|
181
189
|
See also: [Annotation Reference](annotations.md) | [Annotation Styling](styling.md)
|